axstack 0.20.22 → 0.20.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/bin/axstack.js +1 -0
- package/docs/installation.md +4 -4
- package/docs/workflows.md +4 -4
- package/package.json +1 -1
- package/profiles/presets/claude-only.json +20 -2
- package/profiles/presets/codex-only.json +20 -2
- package/profiles/presets/mixed.json +20 -2
- package/skills/axstack/references/design-lens.md +1 -1
- package/skills/axstack/references/orca-runtime.md +1 -1
- package/skills/axstack/references/routing.md +21 -23
- package/skills/axstack-align/SKILL.md +19 -16
- package/src/installer.js +10 -0
- package/src/roles.js +4 -2
package/README.md
CHANGED
|
@@ -99,7 +99,7 @@ upgrades, conflicts, and uninstalling.
|
|
|
99
99
|
inline or from an explicitly scoped backlog without touching active, manual,
|
|
100
100
|
uncertain, user-owned, dirty, unpushed, or useful unmerged work.
|
|
101
101
|
|
|
102
|
-
Choose an explicit role preset:
|
|
102
|
+
Choose an explicit role preset (26 stable role IDs in each):
|
|
103
103
|
[mixed](profiles/presets/mixed.json),
|
|
104
104
|
[codex-only](profiles/presets/codex-only.json), or
|
|
105
105
|
[claude-only](profiles/presets/claude-only.json).
|
package/bin/axstack.js
CHANGED
|
@@ -335,6 +335,7 @@ async function main() {
|
|
|
335
335
|
} else {
|
|
336
336
|
if (summary.added.length) console.log(`added: ${summary.added.join(', ')}`);
|
|
337
337
|
if (summary.updated.length) console.log(`updated: ${summary.updated.join(', ')}`);
|
|
338
|
+
if (summary.addedRoleIds?.length) console.log(`added role IDs: ${summary.addedRoleIds.join(', ')}`);
|
|
338
339
|
if (summary.removed.length) console.log(`removed: ${summary.removed.join(', ')}`);
|
|
339
340
|
if (summary.unchanged.length) console.log(`unchanged: ${summary.unchanged.join(', ')}`);
|
|
340
341
|
}
|
package/docs/installation.md
CHANGED
|
@@ -71,7 +71,7 @@ profiles/presets/codex-only.json
|
|
|
71
71
|
profiles/presets/claude-only.json
|
|
72
72
|
```
|
|
73
73
|
|
|
74
|
-
Each has exactly `{ "version": 1, "roles": [...] }` with the same
|
|
74
|
+
Each has exactly `{ "version": 1, "roles": [...] }` with the same 26 stable
|
|
75
75
|
role IDs. Installation writes `<skills-dir>/axstack/roles.json` as
|
|
76
76
|
`{ "version": 1, "preset": "<selected preset>", "roles": [...] }` and records
|
|
77
77
|
its ownership hash like every other installed skill asset. There is no second
|
|
@@ -140,10 +140,10 @@ The complete bundle is validated before writes:
|
|
|
140
140
|
the filename's selected identity supplied by the caller, and the same role-ID
|
|
141
141
|
set as its peers;
|
|
142
142
|
- every role has valid preserved fields, while the mixed checker,
|
|
143
|
-
`axstack-research-web-google`,
|
|
143
|
+
`axstack-research-web-google`, `axstack-research-x`, and both arena candidate launch-by-agent-id
|
|
144
144
|
routes explicitly permit `model: null`;
|
|
145
145
|
in each single-provider preset, the unavailable adviser and its matching arena
|
|
146
|
-
judge seat explicitly permit `model: null`, as do both cross-provider research routes;
|
|
146
|
+
judge seat explicitly permit `model: null`, as do both cross-provider research routes and both arena candidate seats;
|
|
147
147
|
- obsolete runtime configuration flags fail before mutation with migration
|
|
148
148
|
guidance.
|
|
149
149
|
|
|
@@ -171,7 +171,7 @@ to rewrite them.
|
|
|
171
171
|
## Role behavior after installation
|
|
172
172
|
|
|
173
173
|
The runtime reads `roles.json` from the installed shared root `skills/axstack/`.
|
|
174
|
-
A new run records the selected preset plus all
|
|
174
|
+
A new run records the selected preset plus all 26 role rows. An active run keeps
|
|
175
175
|
that snapshot after a later preset install unless the user explicitly changes
|
|
176
176
|
it and accepts the resulting evidence invalidation.
|
|
177
177
|
|
package/docs/workflows.md
CHANGED
|
@@ -48,7 +48,7 @@ only affected work.
|
|
|
48
48
|
|
|
49
49
|
Installation requires one explicit canonical preset. The three bundle files
|
|
50
50
|
under `profiles/presets/` each contain exactly
|
|
51
|
-
`{ "version": 1, "roles": [...] }` and the same
|
|
51
|
+
`{ "version": 1, "roles": [...] }` and the same 26 stable IDs.
|
|
52
52
|
|
|
53
53
|
The current chat drives on whatever model runs it; no preset carries a driver
|
|
54
54
|
role.
|
|
@@ -122,9 +122,9 @@ session and evidence remain valid.
|
|
|
122
122
|
- `axstack-align` maps facts and dependencies, asks prioritized questions, and
|
|
123
123
|
consults Astra and Fable independently with the same bounded evidence and
|
|
124
124
|
question. It synthesizes disagreements and reuses unchanged receipts. For a
|
|
125
|
-
hard-to-reverse design choice it runs one arena round instead: Astra
|
|
126
|
-
Fable each author a candidate, `axstack-arena-judge-astra` and
|
|
127
|
-
`axstack-arena-judge-fable` score
|
|
125
|
+
hard-to-reverse design choice it runs one arena round instead: Astra,
|
|
126
|
+
Fable, Grok, and Antigravity each author a candidate, `axstack-arena-judge-astra` and
|
|
127
|
+
`axstack-arena-judge-fable` score every candidate against the driver's rubric, and the
|
|
128
128
|
driver picks a base, grafts the losers' strong ideas, and presents the
|
|
129
129
|
synthesis as the recommendation; the note lands as `Decisions` rows in the
|
|
130
130
|
run record.
|
package/package.json
CHANGED
|
@@ -206,7 +206,7 @@
|
|
|
206
206
|
"model": null,
|
|
207
207
|
"modeId": "bypassPermissions",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Required Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the claude-only preset; the arena-grade decision holds without substitution."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
212
|
"id": "axstack-arena-judge-fable",
|
|
@@ -215,7 +215,25 @@
|
|
|
215
215
|
"model": "claude-fable-5-1",
|
|
216
216
|
"modeId": "bypassPermissions",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
218
|
+
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-candidate-grok",
|
|
222
|
+
"name": "Axstack arena candidate Grok (unavailable)",
|
|
223
|
+
"provider": "claude",
|
|
224
|
+
"model": null,
|
|
225
|
+
"modeId": "bypassPermissions",
|
|
226
|
+
"thinkingOptionId": "high",
|
|
227
|
+
"notes": "Intentional single-provider absence in the claude-only preset; Grok candidate unavailable. Arena holds without substitution."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
231
|
+
"name": "Axstack arena candidate Antigravity (unavailable)",
|
|
232
|
+
"provider": "claude",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "bypassPermissions",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Intentional single-provider absence in the claude-only preset; Antigravity candidate unavailable. Arena holds without substitution."
|
|
219
237
|
}
|
|
220
238
|
]
|
|
221
239
|
}
|
|
@@ -206,7 +206,7 @@
|
|
|
206
206
|
"model": "gpt-6-astra",
|
|
207
207
|
"modeId": "full-access",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
212
|
"id": "axstack-arena-judge-fable",
|
|
@@ -215,7 +215,25 @@
|
|
|
215
215
|
"model": null,
|
|
216
216
|
"modeId": "full-access",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "Required Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
218
|
+
"notes": "Required Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged. Intentionally absent in the codex-only preset; the arena-grade decision holds without substitution."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-candidate-grok",
|
|
222
|
+
"name": "Axstack arena candidate Grok (unavailable)",
|
|
223
|
+
"provider": "codex",
|
|
224
|
+
"model": null,
|
|
225
|
+
"modeId": "full-access",
|
|
226
|
+
"thinkingOptionId": "high",
|
|
227
|
+
"notes": "Intentional single-provider absence in the codex-only preset; Grok candidate unavailable. Arena holds without substitution."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
231
|
+
"name": "Axstack arena candidate Antigravity (unavailable)",
|
|
232
|
+
"provider": "codex",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "full-access",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Intentional single-provider absence in the codex-only preset; Antigravity candidate unavailable. Arena holds without substitution."
|
|
219
237
|
}
|
|
220
238
|
]
|
|
221
239
|
}
|
|
@@ -206,7 +206,7 @@
|
|
|
206
206
|
"model": "gpt-6-astra",
|
|
207
207
|
"modeId": "full-access",
|
|
208
208
|
"thinkingOptionId": "xhigh",
|
|
209
|
-
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
209
|
+
"notes": "Arena judge (Astra seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
210
210
|
},
|
|
211
211
|
{
|
|
212
212
|
"id": "axstack-arena-judge-fable",
|
|
@@ -215,7 +215,25 @@
|
|
|
215
215
|
"model": "claude-fable-5-1",
|
|
216
216
|
"modeId": "bypassPermissions",
|
|
217
217
|
"thinkingOptionId": "xhigh",
|
|
218
|
-
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and
|
|
218
|
+
"notes": "Arena judge (Fable seat). Read-only cross-judge for an arena-grade Align decision: receives the rubric and every candidate by label, scores each criterion, and recommends a base with rationale. Never authors a candidate, never cross-reads the other judge, never mutates. Disagreement between judges is surfaced by the driver, never averaged."
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"id": "axstack-arena-candidate-grok",
|
|
222
|
+
"name": "Axstack arena candidate Grok",
|
|
223
|
+
"provider": "grok",
|
|
224
|
+
"model": null,
|
|
225
|
+
"modeId": "full-access",
|
|
226
|
+
"thinkingOptionId": "high",
|
|
227
|
+
"notes": "Independent Grok Rung 2 design candidate. Launch by agent id grok; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
|
|
228
|
+
},
|
|
229
|
+
{
|
|
230
|
+
"id": "axstack-arena-candidate-antigravity",
|
|
231
|
+
"name": "Axstack arena candidate Antigravity",
|
|
232
|
+
"provider": "antigravity",
|
|
233
|
+
"model": null,
|
|
234
|
+
"modeId": "full-access",
|
|
235
|
+
"thinkingOptionId": "high",
|
|
236
|
+
"notes": "Independent Antigravity Rung 2 design candidate. Launch by agent id antigravity; model selected by the TUI default. Receives the same brief without cross-reading; returns design, rationale, and rejected alternatives."
|
|
219
237
|
}
|
|
220
238
|
]
|
|
221
239
|
}
|
|
@@ -15,7 +15,7 @@ sketch through the scope identity.
|
|
|
15
15
|
not reclassify work: Rung 1 can stay small. An unsettled material design
|
|
16
16
|
question still makes routing reassess size.
|
|
17
17
|
- **Rung 2 — arena.** A Rung 1 design that also meets the existing ADR test:
|
|
18
|
-
a meaningful, hard-to-reverse, non-obvious trade-off. Use Align's
|
|
18
|
+
a meaningful, hard-to-reverse, non-obvious trade-off. Use Align's all-family
|
|
19
19
|
arena for that question.
|
|
20
20
|
|
|
21
21
|
There is no numeric threshold, file-count gate, or class-count gate.
|
|
@@ -38,7 +38,7 @@ Read `roles.json` from the installed shared root `skills/axstack/`. The installe
|
|
|
38
38
|
shape is `{ "version": 1, "preset": "<name>", "roles": [...] }`. Bundled
|
|
39
39
|
profiles are setup inputs shaped as
|
|
40
40
|
`{ "version": 1, "roles": [...] }`. A new run records the selected preset and
|
|
41
|
-
all
|
|
41
|
+
all 26 role rows once. An active run keeps the exact snapshot until the user
|
|
42
42
|
explicitly changes it.
|
|
43
43
|
|
|
44
44
|
Select the requested role by stable ID. A missing or null model holds only that role;
|
|
@@ -5,14 +5,14 @@ Driver entry sweep follows [Workspace hygiene](workspace-hygiene.md).
|
|
|
5
5
|
|
|
6
6
|
## Role routing
|
|
7
7
|
|
|
8
|
-
Presets: `mixed`, `codex-only`, `claude-only`. For
|
|
8
|
+
Presets: `mixed`, `codex-only`, `claude-only`. For new runs, use
|
|
9
9
|
`profiles.preset` from `.axstack-manifest.json` at the actually loaded
|
|
10
|
-
skills root, or an explicit user selection recorded in the run record.
|
|
11
|
-
|
|
10
|
+
skills root, or an explicit user selection recorded in the run record. Use one
|
|
11
|
+
unambiguous preset; missing or contradictory sources are
|
|
12
12
|
a setup gap: hold. Never infer from live profiles or `list_profiles`, harness,
|
|
13
13
|
tools, credentials, quota, subscription, or default to `mixed`.
|
|
14
14
|
|
|
15
|
-
At run start, snapshot all
|
|
15
|
+
At run start, snapshot all 26 role IDs with provider/model/mode/effort; absent
|
|
16
16
|
or unconfigured roles are recorded explicitly; invent no provider default.
|
|
17
17
|
Such a role holds only that role's work. A role installed or changed later must not
|
|
18
18
|
silently enter the snapshot; adding it needs an explicit user decision. Live profiles
|
|
@@ -25,8 +25,6 @@ revalidation. Unavailable models, unsupported efforts, missing roles, and
|
|
|
25
25
|
incompatible overrides hold only affected work; no automatic fallback, quota
|
|
26
26
|
routing, subscription inference, or silent provider/model/effort substitution.
|
|
27
27
|
|
|
28
|
-
Role IDs:
|
|
29
|
-
|
|
30
28
|
- Chat drives (no role ID); `axstack-owner` owns one PR and
|
|
31
29
|
`axstack-author` its sole writer.
|
|
32
30
|
- `axstack-reviewer-primary` and `axstack-reviewer-secondary` are the ordered
|
|
@@ -38,10 +36,11 @@ Role IDs:
|
|
|
38
36
|
| `mixed` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-primary` (`codex/gpt-6-sol` medium) |
|
|
39
37
|
| `codex-only` | Codex / Sol (`codex/gpt-6-sol`) | `axstack-reviewer-secondary` (`codex/gpt-6-luna` xhigh) |
|
|
40
38
|
| `claude-only` | Claude / Opus (`claude/claude-opus-5-5`) | `axstack-reviewer-secondary` (`claude/claude-sonnet-5` xhigh) |
|
|
41
|
-
- `axstack-advisor-astra
|
|
42
|
-
|
|
43
|
-
`axstack-arena-
|
|
44
|
-
`axstack-
|
|
39
|
+
- `axstack-advisor-astra`/`axstack-advisor-fable` advise independently and
|
|
40
|
+
author arena candidates; `axstack-arena-candidate-grok`/
|
|
41
|
+
`axstack-arena-candidate-antigravity` add families.
|
|
42
|
+
`axstack-arena-judge-astra`/`axstack-arena-judge-fable` judge them.
|
|
43
|
+
`axstack-auditor` audits; `axstack-checker` reports discrepancies.
|
|
45
44
|
- `axstack-explainer`/`axstack-explainer-review`: explain/review.
|
|
46
45
|
`axstack-monitor`: standalone watch never sends; chat-run watch: bounded
|
|
47
46
|
internal reports to its Run and original driver.
|
|
@@ -51,17 +50,16 @@ Provenance is matched on provider/model ID; effort never maps. Missing table-row
|
|
|
51
50
|
provenance is unsupported and `INCOMPLETE`; report it and ask the user. Never
|
|
52
51
|
infer from slot, driver, owner, or provider. Author and owner never review.
|
|
53
52
|
|
|
54
|
-
|
|
55
|
-
step (3) for user routing
|
|
53
|
+
`axstack-implement` requires `mixed`; single-provider presets hold at
|
|
54
|
+
step (3) for user routing: no substitution or same-provider review.
|
|
56
55
|
|
|
57
56
|
## Direct routes (no spec ceremony)
|
|
58
57
|
|
|
59
58
|
- One bounded research question -> `axstack-research`: verify primary sources
|
|
60
|
-
and code
|
|
61
|
-
questions.
|
|
59
|
+
and code; return a cited note with limitations. Fan out only distinct questions.
|
|
62
60
|
- Understanding a system, change, or implementation gap -> `axstack-explain`:
|
|
63
|
-
current/intended behavior, evidence dimensions,
|
|
64
|
-
|
|
61
|
+
current/intended behavior, evidence dimensions, bounded gaps from project docs and
|
|
62
|
+
rendered behavior. "What could this break" follows
|
|
65
63
|
[Blast radius](blast-radius.md). Publication needs separate authority.
|
|
66
64
|
- A bug, failing test, regression, or wrong behavior, red loop wanted ->
|
|
67
65
|
`axstack-debug`: diagnose, escalate via adviser-directed investigators, hand
|
|
@@ -69,11 +67,11 @@ step (3) for user routing, with no substitution or same-provider review.
|
|
|
69
67
|
- Codebase-quality or refactor discovery -> `axstack-improve`: inspect bounded
|
|
70
68
|
scope, rank evidenced candidates, report only; no spec, tickets, or source
|
|
71
69
|
edits.
|
|
72
|
-
- Accepted worker/Task/Run completion or bounded backlog request ->
|
|
73
|
-
`axstack-cleanup` inline
|
|
70
|
+
- Accepted worker/Task/Run completion or bounded backlog request -> driver invokes
|
|
71
|
+
`axstack-cleanup` inline; never dispatch it.
|
|
74
72
|
- Preparation completion, watch expiry, resume, or reconciliation -> the
|
|
75
|
-
[lifecycle](lifecycle.md#native-handoff-and-resume): reconcile
|
|
76
|
-
|
|
73
|
+
[lifecycle](lifecycle.md#native-handoff-and-resume): reconcile run record,
|
|
74
|
+
keep owner, launch no native handoff.
|
|
77
75
|
- Explicit user-requested ownership transfer -> the same lifecycle section.
|
|
78
76
|
Load the [Orca runtime boundary](orca-runtime.md), follow the runtime-owned
|
|
79
77
|
handoff guide, and require explicit recipient acceptance before ownership
|
|
@@ -81,10 +79,10 @@ step (3) for user routing, with no substitution or same-provider review.
|
|
|
81
79
|
- Colleague PR review -> `axstack-review`, peer mode.
|
|
82
80
|
- Codebase review -> `axstack-review` codebase mode, report only.
|
|
83
81
|
- A status question about an own open PR or stack ("check now", "what's left",
|
|
84
|
-
"are we done", or "is it approved") -> `axstack-watch`
|
|
82
|
+
"are we done", or "is it approved") -> `axstack-watch` observation-only
|
|
85
83
|
mode. Explicit "address", "patch", or "fix" grants authorized maintenance.
|
|
86
|
-
- Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs
|
|
87
|
-
|
|
84
|
+
- Chat-run PR watch -> `axstack-watch`: original driver; verified run PRs and
|
|
85
|
+
explicit adoptions only.
|
|
88
86
|
- Other own PR work -> `axstack-review` authored mode or `axstack-watch`
|
|
89
87
|
adoption.
|
|
90
88
|
|
|
@@ -81,29 +81,30 @@ hold Align; safe fact work may continue without substitution.
|
|
|
81
81
|
|
|
82
82
|
## Arena for hard-to-reverse design choices
|
|
83
83
|
|
|
84
|
-
Critique of one draft anchors every reader to that draft's shape.
|
|
85
|
-
|
|
84
|
+
Critique of one draft anchors every reader to that draft's shape. Rung 2 designs
|
|
85
|
+
alone enter the arena: they meet the same test as for an ADR (a meaningful,
|
|
86
86
|
hard-to-reverse, non-obvious trade-off: architecture, module boundaries, data
|
|
87
|
-
model, migration strategy)
|
|
87
|
+
model, migration strategy). Replace the critique round for that question with
|
|
88
88
|
one arena round. Small or routine questions never enter the arena.
|
|
89
89
|
|
|
90
90
|
1. **Frame.** The driver writes the brief (the artifact, its constraints, the
|
|
91
91
|
settled decisions it must respect) and three to six gradeable rubric
|
|
92
92
|
criteria. Candidates receive only the brief; the rubric is for judging.
|
|
93
|
-
2. **Fan out.**
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
3. **Cross-judge.** After
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
93
|
+
2. **Fan out.** Produce one candidate per configured family independently from the same brief,
|
|
94
|
+
without cross-reading: `axstack-advisor-astra`, `axstack-advisor-fable`,
|
|
95
|
+
`axstack-arena-candidate-grok`, and `axstack-arena-candidate-antigravity`.
|
|
96
|
+
Each gives a design, rationale, and rejected alternatives. The driver authors no candidate.
|
|
97
|
+
3. **Cross-judge.** After every candidate completes, give the judges anonymized,
|
|
98
|
+
relabeled candidates against the rubric.
|
|
99
|
+
`axstack-arena-judge-astra` and `axstack-arena-judge-fable` each independently
|
|
100
|
+
score every candidate against the driver-tailored rubric per criterion and
|
|
101
|
+
recommend a base with a reason. Judges never author, never cross-read each other.
|
|
101
102
|
4. **Pick.** The driver reads every candidate end to end and scores per
|
|
102
103
|
criterion, not on holistic feel, then compares with both judges. Agreement
|
|
103
104
|
confirms the base. Disagreement between judges or with the driver means one
|
|
104
|
-
reading is biased or the rubric was ambiguous: re-read
|
|
105
|
+
reading is biased or the rubric was ambiguous: re-read the rationales and
|
|
105
106
|
decide with a stated reason; never average verdicts or fabricate consensus.
|
|
106
|
-
5. **Graft.** Walk the losing
|
|
107
|
+
5. **Graft.** Walk the losing candidates once more for the one or two ideas
|
|
107
108
|
worth porting and fold them into the base by hand so the result stays
|
|
108
109
|
coherent under one mental model. Convergence on the same shape is a strong
|
|
109
110
|
agreement signal: adopt the consensus shape, no graft. Wide divergence
|
|
@@ -117,9 +118,11 @@ Record the synthesis note (base, grafts and their source candidate, rejections,
|
|
|
117
118
|
dropouts, both judge verdicts) as `Decisions` rows in the
|
|
118
119
|
[run record](../axstack/references/run-record.md). Load
|
|
119
120
|
[Orca runtime](../axstack/references/orca-runtime.md) immediately before the
|
|
120
|
-
first candidate or judge dispatch. If
|
|
121
|
-
unavailable, hold that question without
|
|
122
|
-
and
|
|
121
|
+
first candidate or judge dispatch. If any configured candidate or judge seat is
|
|
122
|
+
unavailable at launch or returns a failed receipt, hold that question without
|
|
123
|
+
substitution, record the gap, and ask: the user decides whether to proceed without it.
|
|
124
|
+
For an uncertain dispatch, reconcile natively; it is never treated as absent.
|
|
125
|
+
Unaffected fact work and questions continue.
|
|
123
126
|
|
|
124
127
|
## Bound the interview
|
|
125
128
|
|
package/src/installer.js
CHANGED
|
@@ -626,6 +626,16 @@ export async function installBundle({
|
|
|
626
626
|
if (rel in ownedFiles) installedHashes[rel] = ownedFiles[rel];
|
|
627
627
|
}
|
|
628
628
|
}
|
|
629
|
+
const previousRoles = desired.find(({ rel }) => rel === 'axstack/roles.json')?.current;
|
|
630
|
+
if (previousRoles && summary.updated.some((rel) => rel.startsWith('axstack/roles.json'))) {
|
|
631
|
+
try {
|
|
632
|
+
const old = JSON.parse(previousRoles.toString());
|
|
633
|
+
if (Array.isArray(old.roles)) {
|
|
634
|
+
const oldIds = new Set(old.roles.map((role) => role?.id));
|
|
635
|
+
summary.addedRoleIds = bundle.bundleRoles.filter((role) => !oldIds.has(role.id)).map((role) => role.id);
|
|
636
|
+
}
|
|
637
|
+
} catch { /* No reliable role-ID diff for malformed prior bytes. */ }
|
|
638
|
+
}
|
|
629
639
|
|
|
630
640
|
// Stale manifest entries (owned files the bundle no longer ships):
|
|
631
641
|
// the phase-1 plan validated every stale destination read-only
|
package/src/roles.js
CHANGED
|
@@ -66,8 +66,10 @@ export function assessRoleReadiness(roles, preset) {
|
|
|
66
66
|
(preset === 'mixed' && role.id === 'axstack-checker' && role.provider === 'antigravity') ||
|
|
67
67
|
(preset === 'mixed' && role.id === 'axstack-research-web-google' && role.provider === 'antigravity') ||
|
|
68
68
|
(preset === 'mixed' && role.id === 'axstack-research-x' && role.provider === 'grok') ||
|
|
69
|
-
(preset === '
|
|
70
|
-
(preset === '
|
|
69
|
+
(preset === 'mixed' && role.id === 'axstack-arena-candidate-grok' && role.provider === 'grok') ||
|
|
70
|
+
(preset === 'mixed' && role.id === 'axstack-arena-candidate-antigravity' && role.provider === 'antigravity') ||
|
|
71
|
+
(preset === 'codex-only' && ['axstack-advisor-fable', 'axstack-arena-judge-fable', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id)) ||
|
|
72
|
+
(preset === 'claude-only' && ['axstack-advisor-astra', 'axstack-arena-judge-astra', 'axstack-research-web-google', 'axstack-research-x', 'axstack-arena-candidate-grok', 'axstack-arena-candidate-antigravity'].includes(role.id))
|
|
71
73
|
);
|
|
72
74
|
for (const role of roles) {
|
|
73
75
|
if (!bounds.has(role.provider)) {
|