model-orchestrator 0.1.34 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/AGENTS.md +31 -21
  2. package/CHANGELOG.md +51 -1
  3. package/README.md +127 -110
  4. package/bin/README.md +57 -6
  5. package/bin/aunx.js +7 -0
  6. package/bin/cli-run.mjs +21 -15
  7. package/bin/cli.js +376 -257
  8. package/docs/README.md +15 -18
  9. package/docs/catalog.md +228 -38
  10. package/docs/companions.md +28 -10
  11. package/docs/guarantees.md +21 -12
  12. package/docs/how-it-routes.md +49 -42
  13. package/docs/install.md +135 -33
  14. package/docs/part-1-beginner.md +37 -45
  15. package/docs/part-2-intermediate.md +34 -52
  16. package/docs/part-3-advanced.md +36 -26
  17. package/docs/security-review-history.md +38 -0
  18. package/llms.txt +24 -25
  19. package/package.json +16 -8
  20. package/proof/README.md +100 -0
  21. package/proof/gate-demo.cast +9 -0
  22. package/proof/gate-demo.gif +0 -0
  23. package/proof/results.json +198 -0
  24. package/proof/scripts/check-gate.js +26 -0
  25. package/proof/scripts/install-time.js +16 -0
  26. package/proof/scripts/lib.js +73 -0
  27. package/proof/scripts/measure.js +15 -0
  28. package/proof/scripts/missing-results.js +30 -0
  29. package/proof/scripts/record-gate.js +38 -0
  30. package/proof/scripts/render.js +18 -0
  31. package/proof/scripts/runner-overhead.js +21 -0
  32. package/src/README.md +9 -3
  33. package/src/activation-ownership.js +19 -0
  34. package/src/apply-companions.js +104 -0
  35. package/src/apply-snippets.js +60 -28
  36. package/src/aunx.js +262 -0
  37. package/src/catalog.js +253 -117
  38. package/src/install.js +478 -209
  39. package/src/plugin.js +13 -4
  40. package/src/postinstall.js +57 -0
  41. package/src/roles.js +184 -0
  42. package/src/uninstall.js +125 -8
  43. package/templates/README.md +19 -2
  44. package/templates/advanced/README.md +2 -2
  45. package/templates/advanced/vm/PRIVACY_GATES.md +17 -19
  46. package/templates/advanced/vm/README.md +25 -20
  47. package/templates/advanced/vm/box-CLAUDE.md +19 -18
  48. package/templates/advanced/vm/jobs/README.md +3 -1
  49. package/templates/advanced/vm/jobs/weekly-audit.service +3 -0
  50. package/templates/advanced/vm/jobs/weekly-audit.sh +2 -2
  51. package/templates/advanced/vm/setup-vm.sh +49 -2
  52. package/templates/agents/README.md +2 -2
  53. package/templates/agents/agy/README.md +20 -3
  54. package/templates/agents/agy/builder.md +11 -7
  55. package/templates/agents/agy/bulk-worker.md +9 -7
  56. package/templates/agents/agy/code-reviewer.md +13 -7
  57. package/templates/agents/agy/deep-planner.md +10 -7
  58. package/templates/agents/agy/done-verifier.md +13 -22
  59. package/templates/agents/agy/finding-verifier.md +14 -22
  60. package/templates/agents/agy/live-researcher.md +10 -7
  61. package/templates/agents/agy/reader.md +10 -12
  62. package/templates/agents/claude-code/README.md +18 -14
  63. package/templates/agents/claude-code/builder.md +10 -15
  64. package/templates/agents/claude-code/bulk-worker.md +8 -10
  65. package/templates/agents/claude-code/code-reviewer.md +11 -17
  66. package/templates/agents/claude-code/deep-planner.md +9 -11
  67. package/templates/agents/claude-code/done-verifier.md +12 -33
  68. package/templates/agents/claude-code/finding-verifier.md +13 -39
  69. package/templates/agents/claude-code/live-researcher.md +9 -11
  70. package/templates/agents/claude-code/reader.md +9 -18
  71. package/templates/agents/snippets/chat.md +9 -10
  72. package/templates/agents/snippets/claude-code.md +17 -18
  73. package/templates/agents/snippets/generic.md +9 -11
  74. package/templates/agents/snippets/route-gate.mjs +2 -2
  75. package/templates/agents/snippets/route-metrics.mjs +1 -1
  76. package/templates/agents/snippets/subagent-context.mjs +4 -4
  77. package/templates/beginner/ORCHESTRATOR.md +31 -36
  78. package/templates/beginner/README.md +1 -1
  79. package/templates/common/ACCEPTANCE_CHECKS.json +12 -0
  80. package/templates/common/CONTEXT.md +37 -0
  81. package/templates/common/DECISIONS.md +11 -0
  82. package/templates/common/README.md +24 -11
  83. package/templates/common/TASK_BRIEF.md +84 -0
  84. package/templates/common/protocols/README.md +14 -11
  85. package/templates/common/protocols/acceptance-checks.md +14 -0
  86. package/templates/common/protocols/build-protocol.md +91 -106
  87. package/templates/common/protocols/context-file.md +10 -0
  88. package/templates/common/protocols/decision-log.md +9 -0
  89. package/templates/common/protocols/deep-research.md +20 -34
  90. package/templates/common/protocols/docs-then-prove.md +13 -18
  91. package/templates/common/protocols/gap-analysis.md +15 -21
  92. package/templates/common/protocols/memory-and-record.md +21 -20
  93. package/templates/common/protocols/numbers-and-logic.md +20 -26
  94. package/templates/common/protocols/propagate.md +18 -27
  95. package/templates/intermediate/CLI-RUN.md +83 -113
  96. package/templates/intermediate/DELEGATION_MATRIX.md +9 -3
  97. package/templates/intermediate/README.md +3 -3
  98. package/templates/intermediate/RESEARCH_TRIAGE.md +23 -15
  99. package/templates/intermediate/ROUTING.md +54 -51
  100. package/templates/intermediate/TIERS.md +37 -76
  101. package/templates/tools/README.md +1 -1
  102. package/templates/tools/obsidian-tc/OBSIDIAN-TC.md +1 -1
  103. package/docs/audit-brief.md +0 -148
  104. package/scripts/README.md +0 -7
  105. package/scripts/gen-catalog.js +0 -81
  106. package/scripts/gen-plugin.js +0 -16
  107. package/scripts/record-demo.sh +0 -45
  108. package/templates/common/TASK_BUNDLE.md +0 -56
@@ -0,0 +1,198 @@
1
+ {
2
+ "schemaVersion": 1,
3
+ "environment": {
4
+ "node": "v22.22.3",
5
+ "platform": "darwin",
6
+ "arch": "arm64"
7
+ },
8
+ "entries": [
9
+ {
10
+ "id": "install-dry-ms",
11
+ "label": "Dry install wall time",
12
+ "kind": "reproducible",
13
+ "value": 42.220875,
14
+ "unit": "ms median",
15
+ "measuredAt": "2026-09-27",
16
+ "method": "Spawn a fresh Node installer process per sample; level 2, Claude Code + Codex, no companions, --dry. Includes Node startup and planning; writes no install files. Isolated home and PATH, no real vendors.",
17
+ "sampleSize": 7,
18
+ "script": "proof/scripts/install-time.js",
19
+ "expiresAt": "2026-10-11",
20
+ "samples": [
21
+ 51.577042,
22
+ 48.655542,
23
+ 46.568834,
24
+ 42.220875,
25
+ 41.573667,
26
+ 39.881333,
27
+ 40.983417
28
+ ]
29
+ },
30
+ {
31
+ "id": "runner-overhead-ms",
32
+ "label": "Lane runner overhead",
33
+ "kind": "reproducible",
34
+ "value": 63.742250999999996,
35
+ "unit": "ms median difference",
36
+ "measuredAt": "2026-09-27",
37
+ "method": "Paired fresh processes: direct Node stub versus cli-run hermes with the same stub. Alternates pair order. Includes wrapper startup, validation and local log writes; excludes vendor/network/model time.",
38
+ "sampleSize": 7,
39
+ "script": "proof/scripts/runner-overhead.js",
40
+ "expiresAt": "2026-10-11",
41
+ "pairs": [
42
+ {
43
+ "directMs": 26.925375,
44
+ "wrappedMs": 409.562,
45
+ "overheadMs": 382.63662500000004
46
+ },
47
+ {
48
+ "directMs": 28.554875,
49
+ "wrappedMs": 95.06725,
50
+ "overheadMs": 66.512375
51
+ },
52
+ {
53
+ "directMs": 28.981666,
54
+ "wrappedMs": 94.610209,
55
+ "overheadMs": 65.628543
56
+ },
57
+ {
58
+ "directMs": 26.295583,
59
+ "wrappedMs": 86.161958,
60
+ "overheadMs": 59.866375
61
+ },
62
+ {
63
+ "directMs": 28.71575,
64
+ "wrappedMs": 82.585583,
65
+ "overheadMs": 53.869833
66
+ },
67
+ {
68
+ "directMs": 26.668791,
69
+ "wrappedMs": 90.411042,
70
+ "overheadMs": 63.742250999999996
71
+ },
72
+ {
73
+ "directMs": 34.081333,
74
+ "wrappedMs": 82.568041,
75
+ "overheadMs": 48.48670799999999
76
+ }
77
+ ]
78
+ },
79
+ {
80
+ "id": "missing-results-caught",
81
+ "label": "Empty results flagged",
82
+ "kind": "reproducible",
83
+ "value": 10,
84
+ "unit": "fixtures rejected",
85
+ "measuredAt": "2026-09-27",
86
+ "method": "Run cli-run against an exit-0 stub for every supported lane, once with empty stdout and once with an empty native final result. Count exit 10/11 only. A successful Hermes response is the positive control. Synthetic fixtures measure these shapes only.",
87
+ "sampleSize": 10,
88
+ "script": "proof/scripts/missing-results.js",
89
+ "expiresAt": "2026-10-11",
90
+ "outcomes": [
91
+ {
92
+ "lane": "grok",
93
+ "fixture": "empty-stdout",
94
+ "exitCode": 11
95
+ },
96
+ {
97
+ "lane": "grok",
98
+ "fixture": "empty-final-result",
99
+ "exitCode": 10
100
+ },
101
+ {
102
+ "lane": "codex",
103
+ "fixture": "empty-stdout",
104
+ "exitCode": 11
105
+ },
106
+ {
107
+ "lane": "codex",
108
+ "fixture": "empty-final-result",
109
+ "exitCode": 10
110
+ },
111
+ {
112
+ "lane": "agy",
113
+ "fixture": "empty-stdout",
114
+ "exitCode": 11
115
+ },
116
+ {
117
+ "lane": "agy",
118
+ "fixture": "empty-final-result",
119
+ "exitCode": 10
120
+ },
121
+ {
122
+ "lane": "hermes",
123
+ "fixture": "empty-stdout",
124
+ "exitCode": 11
125
+ },
126
+ {
127
+ "lane": "hermes",
128
+ "fixture": "empty-final-result",
129
+ "exitCode": 11
130
+ },
131
+ {
132
+ "lane": "qwen",
133
+ "fixture": "empty-stdout",
134
+ "exitCode": 11
135
+ },
136
+ {
137
+ "lane": "qwen",
138
+ "fixture": "empty-final-result",
139
+ "exitCode": 10
140
+ }
141
+ ]
142
+ },
143
+ {
144
+ "id": "acceptance-failures-blocked",
145
+ "label": "Acceptance failures blocked",
146
+ "kind": "reproducible",
147
+ "value": 4,
148
+ "unit": "fixtures rejected",
149
+ "measuredAt": "2026-09-27",
150
+ "method": "Run aunx checks run against nonzero, manual, missing-program and timeout fixtures. Each must exit 1; a passing command must exit 0. This is a local command gate, activated by the user in their release sequence.",
151
+ "sampleSize": 4,
152
+ "script": "proof/scripts/check-gate.js",
153
+ "expiresAt": "2026-10-11",
154
+ "outcomes": [
155
+ {
156
+ "fixture": "nonzero",
157
+ "exitCode": 1
158
+ },
159
+ {
160
+ "fixture": "manual",
161
+ "exitCode": 1
162
+ },
163
+ {
164
+ "fixture": "missing",
165
+ "exitCode": 1
166
+ },
167
+ {
168
+ "fixture": "timeout",
169
+ "exitCode": 1
170
+ }
171
+ ]
172
+ },
173
+ {
174
+ "id": "author-main-browser-tokens",
175
+ "label": "Main conversation browser tokens",
176
+ "kind": "author-setup",
177
+ "value": 408147,
178
+ "unit": "tokens per browser step (median)",
179
+ "measuredAt": "2026-09-26",
180
+ "method": "Measured on the author's Claude Code sessions: every browser tool call in the transcripts, counting tokens re-read by the main conversation per step. Median: 408,147 tokens per browser step across 145 sessions and 9,775 browser steps. Sample size counts browser steps.",
181
+ "sampleSize": 9775,
182
+ "script": "author setup, re-measured locally",
183
+ "expiresAt": "2026-10-27"
184
+ },
185
+ {
186
+ "id": "author-subagent-browser-tokens",
187
+ "label": "Small browser subagent tokens",
188
+ "kind": "author-setup",
189
+ "value": 17197,
190
+ "unit": "tokens per step (highest of 5 runs)",
191
+ "measuredAt": "2026-09-27",
192
+ "method": "At least 23x fewer tokens per step in this sample: the main conversation median of 408,147 divided by the highest subagent run of 17,197 is 23.73x. Measured on the author's Claude Code sessions, with the same kind of work handed to a small browser subagent. Five runs, tokens divided by steps or tool calls per run: 17,197 (206,369 tokens / 12 steps, 2026-09-26); 4,077 (93,773 / 23), 4,201 (105,034 / 25), 4,585 (91,709 / 20), and 4,489 (94,263 / 21), all four on 2026-09-27. Range: 4,077 to 17,197; median of 5 runs: 4,489. Typical context, using the median of 5 runs: about 91x fewer tokens per step. These are measurements from the author's own sessions, not a controlled comparison or a guarantee for other setups. The four 2026-09-27 runs shared one browser pane, so some steps were spent recovering a drifting tab, which raises the step count and lowers per-step tokens. The first run counted steps; the later runs counted tool calls. Sample size counts runs.",
193
+ "sampleSize": 5,
194
+ "script": "author setup, re-measured locally",
195
+ "expiresAt": "2026-10-27"
196
+ }
197
+ ]
198
+ }
@@ -0,0 +1,26 @@
1
+ import { writeFileSync } from 'node:fs';
2
+ import { join } from 'node:path';
3
+ import { clean, entry, isMain, ROOT, run, scratch } from './lib.js';
4
+
5
+ export function measureGate() {
6
+ const dir = scratch();
7
+ try {
8
+ const file = join(dir, 'checks.json');
9
+ const fixtures = [
10
+ { id: 'nonzero', command: ['node', '-e', 'process.exit(7)'] },
11
+ { id: 'manual', manual: true },
12
+ { id: 'missing', command: ['model-orchestrator-missing-fixture-program'] },
13
+ { id: 'timeout', timeoutMs: 50, command: ['node', '-e', 'setInterval(() => {}, 1000)'] }
14
+ ];
15
+ const outcomes = fixtures.map(check => {
16
+ writeFileSync(file, JSON.stringify({ version: 1, checks: [check] }));
17
+ return { fixture: check.id, exitCode: run([join(ROOT, 'bin', 'aunx.js'), 'checks', 'run', file]).status };
18
+ });
19
+ writeFileSync(file, JSON.stringify({ version: 1, checks: [{ id: 'positive', command: ['node', '-e', 'process.exit(0)'] }] }));
20
+ if (run([join(ROOT, 'bin', 'aunx.js'), 'checks', 'run', file]).status !== 0) throw new Error('Positive acceptance check control failed');
21
+ const caught = outcomes.filter(o => o.exitCode === 1).length;
22
+ if (caught !== outcomes.length) throw new Error('A failing acceptance fixture escaped the gate');
23
+ return entry('acceptance-failures-blocked', 'Acceptance failures blocked', caught, 'fixtures rejected', outcomes.length, 'Run aunx checks run against nonzero, manual, missing-program and timeout fixtures. Each must exit 1; a passing command must exit 0. This is a local command gate, activated by the user in their release sequence.', 'check-gate.js', { outcomes });
24
+ } finally { clean(dir); }
25
+ }
26
+ if (isMain(import.meta.url)) console.log(JSON.stringify(measureGate(), null, 2));
@@ -0,0 +1,16 @@
1
+ import { join } from 'node:path';
2
+ import { clean, entry, fixtureEnv, isMain, median, ROOT, run, scratch, timed } from './lib.js';
3
+
4
+ export function measureInstall() {
5
+ const dir = scratch();
6
+ try {
7
+ const samples = [];
8
+ for (let i = 0; i < 7; i++) {
9
+ const { result, ms } = timed(() => run([join(ROOT, 'bin', 'cli.js'), '--yes', '--level', '2', '--ais', 'claude-code,codex', '--primary', 'claude-code', '--dir', join(dir, 'rules'), '--project', dir, '--dry'], { env: fixtureEnv(dir) }));
10
+ if (result.status !== 0) throw new Error(`Dry install failed: ${result.stderr}`);
11
+ samples.push(ms);
12
+ }
13
+ return entry('install-dry-ms', 'Dry install wall time', median(samples), 'ms median', samples.length, 'Spawn a fresh Node installer process per sample; level 2, Claude Code + Codex, no companions, --dry. Includes Node startup and planning; writes no install files. Isolated home and PATH, no real vendors.', 'install-time.js', { samples });
14
+ } finally { clean(dir); }
15
+ }
16
+ if (isMain(import.meta.url)) console.log(JSON.stringify(measureInstall(), null, 2));
@@ -0,0 +1,73 @@
1
+ import { spawnSync } from 'node:child_process';
2
+ import { mkdtempSync, mkdirSync, readFileSync, rmSync, writeFileSync } from 'node:fs';
3
+ import { dirname, join, resolve } from 'node:path';
4
+ import { fileURLToPath } from 'node:url';
5
+ import { tmpdir } from 'node:os';
6
+
7
+ export const ROOT = dirname(dirname(dirname(fileURLToPath(import.meta.url))));
8
+ export const RESULTS = join(ROOT, 'proof', 'results.json');
9
+ export const isMain = url => Boolean(process.argv[1] && resolve(process.argv[1]) === fileURLToPath(url));
10
+ export function scratch() { return mkdtempSync(join(tmpdir(), 'model-orchestrator-proof-')); }
11
+ export function clean(path) { rmSync(path, { recursive: true, force: true }); }
12
+ export function run(args, options = {}) {
13
+ const result = spawnSync(process.execPath, args, { encoding: 'utf8', timeout: 30000, cwd: ROOT, ...options });
14
+ if (result.error) throw result.error;
15
+ return result;
16
+ }
17
+ export function median(samples) {
18
+ const sorted = [...samples].sort((a, b) => a - b);
19
+ const center = Math.floor(sorted.length / 2);
20
+ return sorted.length % 2 ? sorted[center] : (sorted[center - 1] + sorted[center]) / 2;
21
+ }
22
+ export function timed(fn) {
23
+ const start = process.hrtime.bigint();
24
+ const result = fn();
25
+ return { result, ms: Number(process.hrtime.bigint() - start) / 1e6 };
26
+ }
27
+ export function fixtureEnv(dir) {
28
+ const env = { ...process.env };
29
+ for (const key of Object.keys(env)) if (['PATH', 'HOME', 'USERPROFILE'].includes(key.toUpperCase())) delete env[key];
30
+ return { ...env, PATH: join(dir, 'bin'), HOME: dir, USERPROFILE: dir };
31
+ }
32
+ export function stubLane(dir, name, source) {
33
+ mkdirSync(join(dir, 'bin'), { recursive: true });
34
+ const script = join(dir, 'bin', `${name}.cjs`);
35
+ writeFileSync(script, source);
36
+ if (process.platform === 'win32') {
37
+ writeFileSync(join(dir, 'bin', `${name}.cmd`), '@ECHO off\r\nSET "_prog=node"\r\n"%_prog%" "%dp0%'+name+'.cjs" %*\r\n');
38
+ } else {
39
+ writeFileSync(join(dir, 'bin', name), '#!' + process.execPath + '\n' + source, { mode: 0o755 });
40
+ }
41
+ return script;
42
+ }
43
+ export function entry(id, label, value, unit, sampleSize, method, script, extra = {}) {
44
+ const now = new Date();
45
+ const expires = new Date(now);
46
+ expires.setUTCDate(expires.getUTCDate() + 14);
47
+ return { id, label, kind: 'reproducible', value, unit, measuredAt: now.toISOString().slice(0, 10), method, sampleSize, script: `proof/scripts/${script}`, expiresAt: expires.toISOString().slice(0, 10), ...extra };
48
+ }
49
+ export function readResults() { return JSON.parse(readFileSync(RESULTS, 'utf8')); }
50
+ export function validateResults(data, now = new Date()) {
51
+ const errors = [];
52
+ if (!data || data.schemaVersion !== 1 || !Array.isArray(data.entries) || !data.entries.length) return ['Expected schemaVersion 1 with measured entries'];
53
+ const today = now.toISOString().slice(0, 10);
54
+ const ids = new Set();
55
+ const validDate = value => typeof value === 'string' && /^\d{4}-\d{2}-\d{2}$/.test(value) && Number.isFinite(Date.parse(value)) && new Date(value).toISOString().slice(0, 10) === value;
56
+ for (const item of data.entries) {
57
+ if (!item || typeof item.id !== 'string' || ids.has(item.id)) { errors.push('Entry id must be unique'); continue; }
58
+ ids.add(item.id);
59
+ if (typeof item.value !== 'number' || !Number.isFinite(item.value)) errors.push(`${item.id}: value must be finite`);
60
+ if (!Number.isInteger(item.sampleSize) || item.sampleSize < 1) errors.push(`${item.id}: sample size must be positive`);
61
+ if (!['reproducible', 'author-setup'].includes(item.kind)) errors.push(`${item.id}: unknown measurement kind`);
62
+ for (const key of ['label', 'unit', 'method', 'script']) if (typeof item[key] !== 'string' || !item[key].trim()) errors.push(`${item.id}: ${key} is required`);
63
+ if (!validDate(item.measuredAt) || !validDate(item.expiresAt)) errors.push(`${item.id}: valid ISO dates required`);
64
+ else {
65
+ if (item.measuredAt > today) errors.push(`${item.id}: measurement is in the future`);
66
+ if (item.expiresAt < item.measuredAt) errors.push(`${item.id}: expiry precedes measurement`);
67
+ if (item.expiresAt < today) errors.push(`${item.id}: expired ${item.expiresAt}`);
68
+ }
69
+ const localAuthorSource = item.kind === 'author-setup' && item.script === 'author setup, re-measured locally';
70
+ if (!localAuthorSource && (typeof item.script !== 'string' || !/^proof\/scripts\/[A-Za-z0-9_-]+\.js$/.test(item.script))) errors.push(`${item.id}: script must name a proof script`);
71
+ }
72
+ return errors;
73
+ }
@@ -0,0 +1,15 @@
1
+ import { writeFileSync } from 'node:fs';
2
+ import { measureInstall } from './install-time.js';
3
+ import { measureRunner } from './runner-overhead.js';
4
+ import { measureMissingResults } from './missing-results.js';
5
+ import { measureGate } from './check-gate.js';
6
+ import { readResults, RESULTS } from './lib.js';
7
+ import { renderPage } from './render.js';
8
+
9
+ // Run serially: overlapping timing samples would measure this harness competing with itself.
10
+ const entries = [measureInstall(), measureRunner(), measureMissingResults(), measureGate()];
11
+ let author = [];
12
+ try { author = readResults().entries.filter(e => e.kind === 'author-setup'); } catch (error) { if (error.code !== 'ENOENT') throw error; }
13
+ writeFileSync(RESULTS, JSON.stringify({ schemaVersion: 1, environment: { node: process.version, platform: process.platform, arch: process.arch }, entries: [...entries, ...author] }, null, 2) + '\n');
14
+ renderPage();
15
+ for (const item of entries) console.log(`${item.id}: ${item.value} ${item.unit}; n=${item.sampleSize}; measured ${item.measuredAt}; expires ${item.expiresAt}`);
@@ -0,0 +1,30 @@
1
+ import { join } from 'node:path';
2
+ import { clean, entry, fixtureEnv, isMain, ROOT, run, scratch, stubLane } from './lib.js';
3
+
4
+ export function measureMissingResults() {
5
+ const dir = scratch();
6
+ try {
7
+ const shapes = {
8
+ grok: JSON.stringify({ stopReason: 'end_turn', text: '' }),
9
+ codex: JSON.stringify({ type: 'turn.completed' }),
10
+ agy: JSON.stringify({ event: 'result', result: { status: 'SUCCESS', response: '' } }),
11
+ hermes: ' ',
12
+ qwen: JSON.stringify([{ type: 'result', subtype: 'success', result: '' }])
13
+ };
14
+ const outcomes = [];
15
+ for (const [lane, shape] of Object.entries(shapes)) {
16
+ for (const [fixture, text] of [['empty-stdout', ''], ['empty-final-result', shape]]) {
17
+ stubLane(dir, lane, `process.stdout.write(${JSON.stringify(text)});`);
18
+ const result = run([join(ROOT, 'bin', 'cli-run.mjs'), lane, 'fixture', '--quiet'], { env: fixtureEnv(dir) });
19
+ outcomes.push({ lane, fixture, exitCode: result.status });
20
+ }
21
+ }
22
+ // A success control makes a wrapper that rejects everything fail this measurement.
23
+ stubLane(dir, 'hermes', 'console.log("fixture result");');
24
+ if (run([join(ROOT, 'bin', 'cli-run.mjs'), 'hermes', 'control', '--quiet'], { env: fixtureEnv(dir) }).status !== 0) throw new Error('Positive runner control failed');
25
+ const caught = outcomes.filter(o => o.exitCode === 10 || o.exitCode === 11).length;
26
+ if (caught !== outcomes.length) throw new Error('An empty-result fixture escaped the expected failure class');
27
+ return entry('missing-results-caught', 'Empty results flagged', caught, 'fixtures rejected', outcomes.length, 'Run cli-run against an exit-0 stub for every supported lane, once with empty stdout and once with an empty native final result. Count exit 10/11 only. A successful Hermes response is the positive control. Synthetic fixtures measure these shapes only.', 'missing-results.js', { outcomes });
28
+ } finally { clean(dir); }
29
+ }
30
+ if (isMain(import.meta.url)) console.log(JSON.stringify(measureMissingResults(), null, 2));
@@ -0,0 +1,38 @@
1
+ // Capture real fixture output, then render a terminal GIF with an already installed agg.
2
+ // The recorder installs no tools and leaves the repository's promo trailer untouched.
3
+ import { spawnSync } from 'node:child_process';
4
+ import { writeFileSync } from 'node:fs';
5
+ import { join } from 'node:path';
6
+ import { clean, ROOT, run, scratch } from './lib.js';
7
+
8
+ const dir = scratch();
9
+ try {
10
+ const file = join(dir, 'checks.json');
11
+ writeFileSync(file, JSON.stringify({ version: 1, checks: [{ id: 'output-exists', command: ['node', '-e', 'process.exit(require("node:fs").existsSync("output.txt") ? 0 : 1)'] }] }));
12
+ const before = run([join(ROOT, 'bin', 'aunx.js'), 'checks', 'run', file], { cwd: dir });
13
+ if (before.status !== 1) throw new Error('Expected the missing-output check to fail');
14
+ const create = run(['-e', 'require("node:fs").writeFileSync("output.txt", "verified output")'], { cwd: dir });
15
+ if (create.status !== 0) throw new Error('Could not create the fixture output');
16
+ const after = run([join(ROOT, 'bin', 'aunx.js'), 'checks', 'run', file], { cwd: dir });
17
+ if (after.status !== 0) throw new Error('Expected the existing-output check to pass');
18
+ const terminal = s => s.replace(/\r?\n/g, '\r\n');
19
+ const events = [
20
+ { version: 2, width: 86, height: 15, title: 'Acceptance checks: red, then green', env: { TERM: 'xterm-256color' } },
21
+ [0, 'o', '\u001b[1;32m$\u001b[0m aunx checks run checks.json\r\n'],
22
+ [0.6, 'o', '\u001b[31m' + terminal(before.stdout) + '\u001b[0m'],
23
+ [1.2, 'o', '$ echo $?\r\n' + before.status + '\r\n'],
24
+ [3, 'o', '\r\n$ node -e \'require("node:fs").writeFileSync("output.txt","verified output")\'\r\n'],
25
+ [4.5, 'o', '\r\n\u001b[1;32m$\u001b[0m aunx checks run checks.json\r\n'],
26
+ [5, 'o', '\u001b[32m' + terminal(after.stdout) + '\u001b[0m'],
27
+ [5.5, 'o', '$ echo $?\r\n' + after.status + '\r\n'],
28
+ [7, 'o', '\r\nThe exit code gates the next command in your release sequence.\r\n']
29
+ ];
30
+ const cast = join(ROOT, 'proof', 'gate-demo.cast');
31
+ const gif = join(ROOT, 'proof', 'gate-demo.gif');
32
+ writeFileSync(cast, events.map(e => JSON.stringify(e)).join('\n') + '\n');
33
+ const rendered = spawnSync('agg', ['--theme', 'monokai', '--font-size', '18', '--last-frame-duration', '3', cast, gif], { encoding: 'utf8', timeout: 60000 });
34
+ if (rendered.error || rendered.status !== 0) {
35
+ throw new Error(`The cast is saved. To render the GIF, install agg yourself from https://github.com/asciinema/agg and run this script again. ${rendered.error?.code || rendered.stderr}`);
36
+ }
37
+ console.log('Recorded proof/gate-demo.cast and proof/gate-demo.gif from real failing and passing commands.');
38
+ } finally { clean(dir); }
@@ -0,0 +1,18 @@
1
+ import { writeFileSync } from 'node:fs';
2
+ import { join, resolve } from 'node:path';
3
+ import { fileURLToPath } from 'node:url';
4
+ import { readResults, ROOT, validateResults } from './lib.js';
5
+
6
+ export function proofMarkdown(data) {
7
+ const source = (e, label) => e.kind === 'author-setup' && e.script === 'author setup, re-measured locally' ? e.script : `[${label}](../${e.script})`;
8
+ const rows = data.entries.map(e => `| ${e.label} | ${e.value.toFixed(Number.isInteger(e.value) ? 0 : 2)} ${e.unit} | ${e.sampleSize} | ${e.measuredAt} | ${e.expiresAt} | ${source(e, 'script')} |`);
9
+ const methods = data.entries.map(e => `### ${e.label}\n\n${e.method}\n\nKind: ${e.kind === 'author-setup' ? "measured on the author's setup" : 'reproducible local measurement'}. Sample size: ${e.sampleSize}. Measured: ${e.measuredAt}. Expires: ${e.expiresAt}.\n\nSource: ${source(e, e.script)}.`).join('\n\n');
10
+ return `# Reproduce the measurements\n\nGenerated from [results.json](results.json). Each figure has a method, sample size, measurement date and expiry. Run the scripts on your own machine to compare.\n\nEnvironment: Node ${data.environment.node}, ${data.environment.platform} ${data.environment.arch}. Timing varies with startup caches and other work on the machine. Synthetic cases show what those fixtures exercise.\n\n| Measurement | Result | Sample size | Measured | Expires | Reproduce |\n|---|---|---|---|---|---|\n${rows.join('\n')}\n\n## Run the proof scripts\n\n\`\`\`sh\nnode proof/scripts/measure.js\nnode proof/scripts/render.js\nnpm test\n\`\`\`\n\nThe measurement command refreshes reproducible entries and preserves separately sourced author-setup entries. The test suite rejects expired, future-dated or incomplete entries and checks this page against the data. The weekly [refresh workflow](../.github/workflows/proof.yml) reruns the scripts and commits their data and generated page.\n\n## Try the acceptance gate\n\n\`\`\`sh\naunx checks ACCEPTANCE_CHECKS.json\naunx checks run ACCEPTANCE_CHECKS.json\n\`\`\`\n\nThe scaffold starts red. Replace the sample with commands that prove your requirements, then put \`aunx checks run ACCEPTANCE_CHECKS.json && <your-release-command>\` in your own release sequence. Commands are local code you review before running. Manual evidence stays UNVERIFIED and blocks the gate.\n\n![Acceptance gate rejects a missing output, then passes after the output exists](gate-demo.gif)\n\nThe [recording script](scripts/record-gate.js) captures real command output into an asciicast, then renders it with an already installed agg. Companion tools are installed by their users.\n\n## Measurement methods\n\n${methods}\n\n## Operation and verification\n\n- **What and why:** executable measurements keep public figures traceable to current output.\n- **Trigger:** weekly schedule, workflow dispatch, or \`node proof/scripts/measure.js\`.\n- **Invocation chain:** workflow -> measurement functions -> isolated Node fixtures -> results.json -> this page -> npm test.\n- **Dependencies:** Node and the repository. The optional GIF recorder uses agg from the asciinema project.\n- **Reads:** package scripts, the installer, runner and acceptance-check runner. Fixture tests use an isolated home and PATH.\n- **Writes:** results.json, this generated page, temporary fixture directories and local fixture logs. The recorder writes gate-demo.cast and gate-demo.gif.\n- **Closed loop:** the workflow fails when measurement or tests fail. GitHub Actions records the failure; repository notification settings decide who receives it. No separate alert service is configured.\n- **Failure modes:** runner behavior changes, an expired catalog snapshot, missing runtime, unavailable write permission, or timing noise. Review the failed job, rerun locally, and send a reproducible issue to the repository maintainers.\n- **Run and verify:** run the commands above, inspect sample arrays and fixture exit codes in results.json, and require npm test to pass. A future-time unit test proves expiry can fail.\n- **Source of truth:** results.json and the scripts it names. Author-setup evidence is added separately by its owner.\n`;
11
+ }
12
+ export function renderPage() {
13
+ const data = readResults();
14
+ const errors = validateResults(data);
15
+ if (errors.length) throw new Error(errors.join('\n'));
16
+ writeFileSync(join(ROOT, 'proof', 'README.md'), proofMarkdown(data));
17
+ }
18
+ if (process.argv[1] && resolve(process.argv[1]) === fileURLToPath(import.meta.url)) renderPage();
@@ -0,0 +1,21 @@
1
+ import { join } from 'node:path';
2
+ import { clean, entry, fixtureEnv, isMain, median, ROOT, run, scratch, stubLane, timed } from './lib.js';
3
+
4
+ export function measureRunner() {
5
+ const dir = scratch();
6
+ try {
7
+ const stub = stubLane(dir, 'hermes', 'console.log("fixture result");');
8
+ const env = fixtureEnv(dir);
9
+ const pairs = [];
10
+ for (let i = 0; i < 7; i++) {
11
+ const direct = () => timed(() => run([stub], { env }));
12
+ const wrapped = () => timed(() => run([join(ROOT, 'bin', 'cli-run.mjs'), 'hermes', 'fixture', '--quiet'], { env }));
13
+ let before, after;
14
+ if (i % 2) { after = wrapped(); before = direct(); } else { before = direct(); after = wrapped(); }
15
+ if (before.result.status !== 0 || after.result.status !== 0) throw new Error('Stub runner measurement failed');
16
+ pairs.push({ directMs: before.ms, wrappedMs: after.ms, overheadMs: after.ms - before.ms });
17
+ }
18
+ return entry('runner-overhead-ms', 'Lane runner overhead', median(pairs.map(p => p.overheadMs)), 'ms median difference', pairs.length, 'Paired fresh processes: direct Node stub versus cli-run hermes with the same stub. Alternates pair order. Includes wrapper startup, validation and local log writes; excludes vendor/network/model time.', 'runner-overhead.js', { pairs });
19
+ } finally { clean(dir); }
20
+ }
21
+ if (isMain(import.meta.url)) console.log(JSON.stringify(measureRunner(), null, 2));
package/src/README.md CHANGED
@@ -3,9 +3,15 @@
3
3
  | File | Job |
4
4
  |---|---|
5
5
  | `catalog.js` | the single list of levels and AIs. Add an AI here and the prompts, docs tables, delegation matrix, gateway config and installer all pick it up. Nothing else lists AIs. |
6
+ | `roles.js` | pure role assignment from selected catalog capability facts, billing and selection order. Renders the stack table and manifest roles, and infers the main agent from its supported surfaces. Unknown facts remain unverified; review requires a known different model family and private work requires local execution. |
7
+ | `aunx.js` | command dispatch for briefs, context, checks, routing and runner calls. Route suggestions read manifest roles through a capped regular-file JSON reader; symlinks and malformed files are ignored. Route lookup executes no project code. A project's runner requires explicit `--dir`. |
6
8
  | `detect.js` | PATH lookup for a binary, plus the few places vendor installers drop binaries without touching PATH. No shell-outs. |
7
- | `install.js` | pure planner: turns (level, selection, primary) into a list of files to write, rendering templates and computing every generated table. `writeFiles` is the only thing that touches disk. `activationSteps()` and `snippetFor()` live here so the terminal summary and the generated README render the same list. `subagentsLoadRules(primary)` gates every delegate-by-default render var (builder-by-default wording, the route-gate table, the inline-threshold note) on the one verified premise: a claude-code subagent loads CLAUDE.md. |
8
- | `apply-snippets.js` | validates the two Claude Code activation targets before any writes, builds a replaceable marked block and merges hook entries. `writeFiles` applies these opt-in entries with backups and rollback; they stay outside the uninstall manifest. |
9
- | `plugin.js` | the plan for the Claude Code plugin bundle in `plugin/`: `planPluginFiles()` renders the claude-code agents and the two read-only hooks from the same templates `install.js` uses, with plugin render vars (the installer's default rules paths, the setup hint for a project with no rules) instead of per-install ones. Pure; `scripts/gen-plugin.js` writes it and `test/plugin.test.js` checks the committed copy. |
9
+ | `install.js` | pure planner: turns (level, selection, primary) into a list of files to write, rendering templates and computed role assignments. `writeFiles` is the only thing that touches disk. `activationSteps()` and `snippetFor()` live here so the terminal summary and the generated README render the same list. `subagentsLoadRules(primary)` selects loading guidance from catalog facts. File ownership hashes preserve edits during runtime upgrades and document updates, migrate the old brief name, and keep uninstall ownership of existing companion files. |
10
+ | `apply-snippets.js` | validates the main agent's catalog-supported rules target and Claude Code hook settings before writes, builds a replaceable marked block and merges hook entries. `writeFiles` applies the entries with backups and rollback, and records ownership for uninstall. Interactive activation defaults on; headless `--yes` requires `--apply-snippets`, and `--no-apply` keeps activation manual. |
11
+ | `activation-ownership.js` | validates recorded project paths and rules, hook and MCP ownership before upgrades or uninstall. The shared `globalConfigProblem()` check in `install.js` refuses home-level agent configuration targets during installation and uninstall. |
12
+ | `apply-companions.js` | plans selected companion registration for the main agent's catalog-supported project MCP config, preserves existing server entries and emits remaining setup steps. Claude Code's `.mcp.json` is the supported automatic target; other hosts keep manual registration guidance. |
13
+ | `postinstall.js` | runs the local presence health check and checks sign-in only through reliable catalog-listed status commands. Live canaries and vendor login flows remain explicit user actions. |
14
+ | `uninstall.js` | validates manifest ownership and target paths before removing unedited managed files and unchanged applied rules, hooks or MCP entries; preserves user changes and backups. |
15
+ | `plugin.js` | the plan for the Claude Code plugin bundle in `plugin/`: `planPluginFiles()` renders the claude-code agents and the two read-only hooks from the same templates `install.js` uses, with plugin render vars (the installer's default rules paths, the setup hint for a project with no rules) instead of per-install ones. Pure; `scripts/gen-plugin.js` writes it and the marketplace listing and `test/plugin.test.js` checks the committed copy. |
10
16
  | `prompt.js` | line-buffered questions for the interactive path; piped answers are queued, EOF mid-prompt aborts instead of confirming a write. |
11
17
  | `render.js` | `{{KEY}}` substitution. Throws on an unknown key, so a template typo fails the test suite instead of shipping a literal placeholder. |
@@ -0,0 +1,19 @@
1
+ const object = (value) => value !== null && typeof value === 'object' && !Array.isArray(value);
2
+ const digest = (value) => typeof value === 'string' && /^[a-f0-9]{64}$/.test(value);
3
+
4
+ // Shared by upgrades and uninstall: persisted ownership is untrusted input.
5
+ export function validateActivationOwnership(key, ownership) {
6
+ const relative = typeof key === 'string' && key.startsWith('[project] ') ? key.slice('[project] '.length) : '';
7
+ if (!relative || /[\\:\x00-\x1f\x7f]/.test(relative) || relative.split('/').some((part) => !part || part === '.' || part === '..')
8
+ || !object(ownership) || typeof ownership.created !== 'boolean') return `invalid activation ownership: ${key}`;
9
+ if (ownership.kind === 'rules') {
10
+ if (!digest(ownership.blockHash) || !['', '\n', '\n\n'].includes(ownership.addedPrefix) || !['', '\n'].includes(ownership.addedSuffix)) return `invalid rules ownership: ${key}`;
11
+ } else if (ownership.kind === 'hooks') {
12
+ if (!Array.isArray(ownership.hooks) || typeof ownership.hadHooks !== 'boolean' || !Array.isArray(ownership.originalEvents)
13
+ || ownership.originalEvents.some((event) => typeof event !== 'string')
14
+ || ownership.hooks.some((record) => !object(record) || typeof record.event !== 'string' || !object(record.group) || !object(record.hook))) return `invalid hook ownership: ${key}`;
15
+ } else if (ownership.kind === 'mcp') {
16
+ if (ownership.key !== 'mcpServers' || !object(ownership.servers) || typeof ownership.hadKey !== 'boolean') return `invalid MCP ownership: ${key}`;
17
+ } else return `unknown activation ownership: ${key}`;
18
+ return null;
19
+ }
@@ -0,0 +1,104 @@
1
+ import { existsSync, readFileSync, statSync } from 'node:fs';
2
+ import { join, resolve } from 'node:path';
3
+ import { isDeepStrictEqual } from 'node:util';
4
+ import { preflight, ACTIVATION_JSON_BYTE_CAP } from './install.js';
5
+
6
+ const object = (value) => value !== null && typeof value === 'object' && !Array.isArray(value);
7
+ const refuse = (message) => Object.assign(new Error(message), { code: 'PREFLIGHT' });
8
+ const configPlaceholder = '/ABSOLUTE/PATH/TO/obsidian-tc.config.json';
9
+ const needsConfiguration = (server) => server?.env?.OBSIDIAN_TC_CONFIG === configPlaceholder;
10
+
11
+ // Only cataloged project-local targets are eligible. A snippet for a vendor's
12
+ // global config is useful setup guidance, never permission to edit that file.
13
+ export function planCompanionApplication({ primary, tools = [], project, files, platform = process.platform }) {
14
+ const host = primary?.projectMcp;
15
+ const selected = tools.filter((tool) => tool.mcpSnippets?.[primary?.id]);
16
+ if (!host || !selected.length) return [];
17
+ const problems = preflight([{ rel: host.file }], project);
18
+ if (problems.length) throw refuse(problems.join('; '));
19
+ const path = join(project, host.file);
20
+ // R1: refuse an oversized pre-existing MCP config before it is read, the
21
+ // same way invalid JSON is refused today.
22
+ if (existsSync(path) && statSync(path).size > ACTIVATION_JSON_BYTE_CAP) {
23
+ throw refuse(`${path}: larger than the ${ACTIVATION_JSON_BYTE_CAP} byte (10 MB) cap on a pre-existing settings/MCP JSON file; nothing written`);
24
+ }
25
+ const original = existsSync(path) ? readFileSync(path) : null;
26
+ let settings = {};
27
+ if (original) {
28
+ try { settings = JSON.parse(original.toString('utf8')); }
29
+ catch { throw refuse(`${path}: invalid JSON; nothing written`); }
30
+ }
31
+ if (!object(settings) || (settings[host.key] !== undefined && !object(settings[host.key]))) {
32
+ throw refuse(`${path}: expected a JSON object with an optional ${host.key} object`);
33
+ }
34
+ const servers = { ...settings[host.key] };
35
+ const added = {};
36
+ const companionRegistrations = [];
37
+ for (const tool of selected) {
38
+ const rel = tool.mcpSnippets[primary.id];
39
+ const snippet = files.find((file) => (file.root || 'dir') === 'dir' && file.rel.split('\\').join('/') === rel);
40
+ if (!snippet) throw refuse(`${tool.id}: missing registration snippet ${rel}`);
41
+ let incoming;
42
+ try { incoming = JSON.parse(snippet.content.toString())[host.key]?.[tool.id]; }
43
+ catch { throw refuse(`${tool.id}: invalid registration snippet ${rel}`); }
44
+ if (!object(incoming)) throw refuse(`${tool.id}: missing server object in ${rel}`);
45
+ // The installed JSON templates use npx for this stdio server. Windows
46
+ // requires the vendor-documented cmd wrapper (OBSIDIAN-TC.md Windows).
47
+ if (platform === 'win32' && incoming.command === 'npx') {
48
+ incoming = { ...incoming, command: 'cmd', args: ['/d', '/c', 'npx', ...(incoming.args || [])] };
49
+ }
50
+ let status;
51
+ if (Object.hasOwn(servers, tool.id)) {
52
+ status = isDeepStrictEqual(servers[tool.id], incoming) ? 'present' : 'conflict';
53
+ } else {
54
+ status = 'added';
55
+ servers[tool.id] = incoming;
56
+ added[tool.id] = incoming;
57
+ }
58
+ companionRegistrations.push({ id: tool.id, status, needsConfiguration: needsConfiguration(servers[tool.id]) });
59
+ }
60
+ const content = Object.keys(added).length
61
+ ? JSON.stringify({ ...settings, [host.key]: servers }, null, 2) + '\n'
62
+ : original;
63
+ return [{
64
+ rel: host.file, root: 'project', mode: 0o644, original, content,
65
+ applySnippet: true,
66
+ activation: { kind: 'mcp', key: host.key, hadKey: Object.hasOwn(settings, host.key), servers: added },
67
+ companionRegistrations
68
+ }];
69
+ }
70
+
71
+ // The generated README can use conditional guidance before a merge is planned;
72
+ // the terminal supplies registrations to describe the observed merge result.
73
+ export function companionRegistrationSteps({ primary, tools = [], project, dir, applySnippets = false, registrations } = {}) {
74
+ const root = resolve(project || process.cwd());
75
+ const docs = resolve(dir || 'ai-orchestrator');
76
+ const host = primary?.projectMcp;
77
+ const states = registrations?.flatMap((entry) => entry.companionRegistrations || []);
78
+ const steps = [];
79
+ for (const tool of tools) {
80
+ const rel = tool.mcpSnippets?.[primary?.id];
81
+ const snippet = rel ? join(docs, rel) : join(docs, 'mcp');
82
+ const state = states?.find((item) => item.id === tool.id);
83
+ const registered = applySnippets && host && rel;
84
+ if (!registered) {
85
+ if (!rel) {
86
+ steps.push(`${tool.id}: follow ${join(docs, tool.id.toUpperCase() + '.md')} to set it up with ${primary?.chatName || primary?.name || 'your agent'}`);
87
+ continue;
88
+ }
89
+ const target = host ? join(root, host.file) : 'your agent\'s MCP config (user-managed)';
90
+ steps.push(`${tool.id}: merge ${snippet} into ${target}; follow ${join(docs, tool.id.toUpperCase() + '.md')} for prerequisites`);
91
+ continue;
92
+ }
93
+ if (state?.status === 'conflict') {
94
+ steps.push(`${tool.id}: review the existing ${tool.id} entry in ${join(root, host.file)} against ${snippet} (existing entry kept)`);
95
+ continue;
96
+ }
97
+ if (tool.id === 'codecalc') {
98
+ steps.push(`codecalc: if uv or Python 3.10+ is missing, install it using ${join(docs, 'CODECALC.md')}`);
99
+ } else if (tool.id === 'obsidian-tc' && (!state || state.needsConfiguration)) {
100
+ steps.push(`obsidian-tc: set OBSIDIAN_TC_CONFIG in ${join(root, host.file)} to your config file path; follow ${join(docs, 'OBSIDIAN-TC.md')} for vault setup`);
101
+ }
102
+ }
103
+ return steps;
104
+ }