model-orchestrator 0.1.35 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/AGENTS.md +31 -21
  2. package/CHANGELOG.md +43 -1
  3. package/README.md +127 -110
  4. package/bin/README.md +57 -6
  5. package/bin/aunx.js +7 -0
  6. package/bin/cli-run.mjs +21 -15
  7. package/bin/cli.js +376 -257
  8. package/docs/README.md +15 -18
  9. package/docs/catalog.md +228 -38
  10. package/docs/companions.md +28 -10
  11. package/docs/guarantees.md +21 -12
  12. package/docs/how-it-routes.md +49 -42
  13. package/docs/install.md +135 -33
  14. package/docs/part-1-beginner.md +37 -45
  15. package/docs/part-2-intermediate.md +34 -52
  16. package/docs/part-3-advanced.md +36 -26
  17. package/docs/security-review-history.md +38 -0
  18. package/llms.txt +24 -25
  19. package/package.json +15 -8
  20. package/proof/README.md +100 -0
  21. package/proof/gate-demo.cast +9 -0
  22. package/proof/gate-demo.gif +0 -0
  23. package/proof/results.json +198 -0
  24. package/proof/scripts/check-gate.js +26 -0
  25. package/proof/scripts/install-time.js +16 -0
  26. package/proof/scripts/lib.js +73 -0
  27. package/proof/scripts/measure.js +15 -0
  28. package/proof/scripts/missing-results.js +30 -0
  29. package/proof/scripts/record-gate.js +38 -0
  30. package/proof/scripts/render.js +18 -0
  31. package/proof/scripts/runner-overhead.js +21 -0
  32. package/src/README.md +9 -3
  33. package/src/activation-ownership.js +19 -0
  34. package/src/apply-companions.js +104 -0
  35. package/src/apply-snippets.js +60 -28
  36. package/src/aunx.js +262 -0
  37. package/src/catalog.js +253 -117
  38. package/src/install.js +478 -209
  39. package/src/plugin.js +13 -4
  40. package/src/postinstall.js +57 -0
  41. package/src/roles.js +184 -0
  42. package/src/uninstall.js +125 -8
  43. package/templates/README.md +19 -2
  44. package/templates/advanced/README.md +2 -2
  45. package/templates/advanced/vm/PRIVACY_GATES.md +17 -19
  46. package/templates/advanced/vm/README.md +25 -20
  47. package/templates/advanced/vm/box-CLAUDE.md +19 -18
  48. package/templates/advanced/vm/jobs/README.md +3 -1
  49. package/templates/advanced/vm/jobs/weekly-audit.service +3 -0
  50. package/templates/advanced/vm/jobs/weekly-audit.sh +2 -2
  51. package/templates/advanced/vm/setup-vm.sh +49 -2
  52. package/templates/agents/README.md +2 -2
  53. package/templates/agents/agy/README.md +20 -3
  54. package/templates/agents/agy/builder.md +11 -7
  55. package/templates/agents/agy/bulk-worker.md +9 -7
  56. package/templates/agents/agy/code-reviewer.md +13 -7
  57. package/templates/agents/agy/deep-planner.md +10 -7
  58. package/templates/agents/agy/done-verifier.md +13 -22
  59. package/templates/agents/agy/finding-verifier.md +14 -22
  60. package/templates/agents/agy/live-researcher.md +10 -7
  61. package/templates/agents/agy/reader.md +10 -12
  62. package/templates/agents/claude-code/README.md +18 -14
  63. package/templates/agents/claude-code/builder.md +10 -15
  64. package/templates/agents/claude-code/bulk-worker.md +8 -10
  65. package/templates/agents/claude-code/code-reviewer.md +11 -17
  66. package/templates/agents/claude-code/deep-planner.md +9 -11
  67. package/templates/agents/claude-code/done-verifier.md +12 -33
  68. package/templates/agents/claude-code/finding-verifier.md +13 -39
  69. package/templates/agents/claude-code/live-researcher.md +9 -11
  70. package/templates/agents/claude-code/reader.md +9 -18
  71. package/templates/agents/snippets/chat.md +9 -10
  72. package/templates/agents/snippets/claude-code.md +17 -18
  73. package/templates/agents/snippets/generic.md +9 -11
  74. package/templates/agents/snippets/route-gate.mjs +2 -2
  75. package/templates/agents/snippets/route-metrics.mjs +1 -1
  76. package/templates/agents/snippets/subagent-context.mjs +4 -4
  77. package/templates/beginner/ORCHESTRATOR.md +31 -36
  78. package/templates/beginner/README.md +1 -1
  79. package/templates/common/ACCEPTANCE_CHECKS.json +12 -0
  80. package/templates/common/CONTEXT.md +37 -0
  81. package/templates/common/DECISIONS.md +11 -0
  82. package/templates/common/README.md +24 -11
  83. package/templates/common/TASK_BRIEF.md +84 -0
  84. package/templates/common/protocols/README.md +14 -11
  85. package/templates/common/protocols/acceptance-checks.md +14 -0
  86. package/templates/common/protocols/build-protocol.md +91 -106
  87. package/templates/common/protocols/context-file.md +10 -0
  88. package/templates/common/protocols/decision-log.md +9 -0
  89. package/templates/common/protocols/deep-research.md +20 -34
  90. package/templates/common/protocols/docs-then-prove.md +13 -18
  91. package/templates/common/protocols/gap-analysis.md +15 -21
  92. package/templates/common/protocols/memory-and-record.md +21 -20
  93. package/templates/common/protocols/numbers-and-logic.md +20 -26
  94. package/templates/common/protocols/propagate.md +18 -27
  95. package/templates/intermediate/CLI-RUN.md +83 -113
  96. package/templates/intermediate/DELEGATION_MATRIX.md +9 -3
  97. package/templates/intermediate/README.md +3 -3
  98. package/templates/intermediate/RESEARCH_TRIAGE.md +23 -15
  99. package/templates/intermediate/ROUTING.md +54 -51
  100. package/templates/intermediate/TIERS.md +37 -76
  101. package/templates/tools/README.md +1 -1
  102. package/templates/tools/obsidian-tc/OBSIDIAN-TC.md +1 -1
  103. package/docs/audit-brief.md +0 -148
  104. package/scripts/README.md +0 -7
  105. package/scripts/gen-catalog.js +0 -81
  106. package/scripts/gen-plugin.js +0 -16
  107. package/scripts/record-demo.sh +0 -45
  108. package/templates/common/TASK_BUNDLE.md +0 -56
@@ -1,79 +1,61 @@
1
- # Part 2 · Intermediate: many AIs, called through their CLIs
1
+ # Part 2: delegate across your AI CLIs
2
2
 
3
- Everything in Part 1, plus lanes. One agent stays the orchestrator; every other AI becomes a lane it calls from the terminal.
3
+ Keep one main agent coordinating the work. Each other AI is a **lane**: a CLI or model the main agent can call with a scoped task brief.
4
4
 
5
- ## Plans and automatic effort
6
-
7
- The installer can record plans with `--plans codex=pro-20x,agy=ultra-5x`. Plan headroom changes volume allocation only, never capability or the independent-review rule. `--effort-auto` is explicit consent to write `auto` for eligible high or max headroom CLI lanes. Auto resolves to medium or high from prompt size, and a codex audit is always high. It is a heuristic, not a measurement: name xhigh explicitly for security-critical or irreversible work.
8
-
9
- ## 1. Two kinds of lane
10
-
11
- **Lane A, subscription CLIs.** Claude Code, Codex, Antigravity, Grok, Hermes. Already paid for, $0 per call, used for interactive and agentic work. **Lane B, metered APIs.** Per token, used for programmatic bulk where a subscription CLI cannot serve. **Local.** A privacy lane, never a cost lane.
12
-
13
- Rule: never spend a frontier token on a task a cheap tier finishes correctly. Escalate on signal, not by default. And an external lane must earn the hop with a real strength; when in doubt, stay in-house.
14
-
15
- ## 2. One job per lane
5
+ ## Subscription lanes, pay-per-token lanes and local models
16
6
 
17
- | Lane | Wins at |
18
- |---|---|
19
- | the orchestrator (Claude Code, or whichever you chose) | routes, maps, builds, verifies, records; drives the others as CLIs |
20
- | Codex | second coder and second-opinion reviewer: a different model family reading your diff |
21
- | Antigravity `agy` | deep research sweeps; concurrent fan-out (its subagent call takes an array) |
22
- | Grok CLI | X and live web reads at $0 (the same search on the API bills per call) |
23
- | Hermes | the free tier: rough drafts, first-pass summaries, divergent reads, cron jobs |
24
- | Qwen Code + a cheap metered model | structured bulk output; never anything that cites a line, number or source |
25
- | Ollama | anything that must not leave the machine |
7
+ - **Subscription lanes:** use the tools and quota included in your vendor plan. Check that plan's current limits before assigning volume.
8
+ - **Pay-per-token lanes:** use metered APIs for programmatic work, with an explicit budget and model choice.
9
+ - **Local models:** keep private input on your machine when the task requires that boundary.
26
10
 
27
- One driver, no second AI in the mix: the orchestrator invokes the CLIs; it never hands control to another agent.
11
+ The generated **Your stack: who does what** table assigns roles using capability facts, billing and selection order. The delegation matrix states each assignment and its reason. An independent reviewer needs a known different model family from the main agent; an unknown family cannot establish independence. Check current model names, permissions and tool reach before assigning a section. [Assignment rules](how-it-routes.md#your-stack-who-does-what).
28
12
 
29
- ## 3. Exit 0 is a lie on every lane
13
+ ## Lane runner (`aunx cli-run`)
30
14
 
31
- Every agent CLI can report success and deliver nothing. `bin/cli-run.mjs` builds the right invocation per lane, reads that lane's **native** terminal event, and exits `10` when a run produced no deliverable, `12` on timeout, `13` when the lane is missing, and `14` to `18` when the lane's own error says why (auth, quota, rejected, refused, cut short), so a missing API key is never blamed on the model. Byte count is not a check either; a run can emit hundreds of kilobytes and no conclusion. One lane's own success flags lie outright (an upstream 400 reported as success), so its judge reads the two honest signals instead.
15
+ Replace `<lane>` with a supported CLI named in your installed stack table.
32
16
 
33
- Every call goes through it. "This lane is flaky" becomes a query over its log instead of an argument. `node bin/cli-run.mjs --doctor` is the first thing to run after install: enabled lanes, binaries on PATH, the route each lane is pinned to, and with `--run` a one-word canary per lane.
17
+ ```bash
18
+ aunx cli-run --doctor
19
+ # Direct form from the installed rules folder:
20
+ node bin/cli-run.mjs --doctor
21
+ aunx cli-run '<lane>' --brief TASK_BRIEF.md --effort high
22
+ # Direct form: node bin/cli-run.mjs '<lane>' --brief TASK_BRIEF.md --effort high
23
+ ```
34
24
 
35
- There is a second thing a lane can be quietly wrong about. Left unpinned, it runs on **its own config file**, which the runner cannot see: a CLI set up months ago at a low reasoning effort keeps auditing at that effort while your routing docs describe a second-opinion pass, and no error is ever raised. `--model` and `--effort` pin it per call, `defaults` in `bin/lanes.json` pins it per lane, and every run records the value requested and where it came from (`flag`, `lanes.json`, `lane_default`). The log claims no actual: grok reports a model id in its output, the other four lanes report none, so the field would be populated for one lane and empty for four, and it would be a provider-supplied string the durable log never holds.
25
+ `cli-run` reads each vendor's terminal result and returns nonzero when the response is missing, interrupted, timed out or rejected by an output contract. It distinguishes authentication, quota and unavailable-tool failures so the next action can address the cause. Use `--expect-file` or `--expect-json` when your task needs a specific output shape.
36
26
 
37
- ## 4. Every delegation carries a task bundle, on both surfaces
27
+ `--model` and `--effort` set a request for one call. `defaults` in `bin/lanes.json` supplies per-tool defaults. The local log records requested values and their source; the vendor's own output is the place to confirm actual model execution.
38
28
 
39
- Subagents and CLI lanes are close to the same problem: something that may hold none of your rules, and broad tool access. A Claude Code subagent is the one documented exception, loading the project's CLAUDE.md hierarchy at start, so it keeps the standing rules but not this task's scope; a CLI lane and a fresh chat window get no such credit. The brief (purpose, task class, scope, capabilities, denied actions, conventions, report contract, exit parameters) goes in the prompt or in the file passed to `--brief` either way. If you can, gate it mechanically: a pre-dispatch hook that refuses a brief missing purpose, denied actions or a report contract. On claude-code, a `SubagentStart` hook can inject the essentials (where the rules and the brief format live) automatically; `.claude/hooks/subagent-context.mjs` is the generated example. A third hook, `.claude/hooks/route-metrics.mjs`, turns that same delegation into a measurement instead of an assumption: it logs every turn, dispatch, subagent start/stop and the lane named in the reply's hidden route marker, and `--summary` reports route-marker coverage, dispatches with no matching start, and duration per agent type.
40
-
41
- ## 5. Research: three engines, one triager
42
-
43
- Fan the same plan to three model families (web sweep, second-opinion read, live data), each as one `cli-run` call. The orchestrator opens the primary sources itself, marks every claim, and writes the only durable record. Expect one engine to return confident unsourced numerics; downgrade it. Weight the engines that report their own gaps. Count dispositions, not briefs.
44
-
45
- ## 5a. A finding is a claim, not a fact
46
-
47
- An audit that returns six findings has returned six claims. Hand them to `finding-verifier` before any of them causes a repair: it reads the cited line, states what would trigger the problem, then hunts for the guard, caller or test that makes it impossible, and answers CONFIRMED, NOT_REPRODUCED or INCONCLUSIVE. Only CONFIRMED earns a change. Use a different family from the one that produced the finding, and let INCONCLUSIVE stand: rounding it up to be safe buys unnecessary repairs, rounding it down to be tidy hides real ones.
29
+ ## Plans and automatic effort
48
30
 
49
- ## 6. Gap analysis gets a second family
31
+ `--plans codex=pro-20x,agy=ultra-5x` records plan headroom for volume allocation. Capability and independent review still follow the task and available tools. `--effort-auto` opts eligible high-headroom tools into the runner's prompt-size heuristic. Explicitly choose higher effort for security or irreversible work when the heuristic is insufficient. [Runner reference](../bin/README.md).
50
32
 
51
- The second pass is now a different model reading the same artifact, in read-only audit mode. Disagreement between families is the cheapest signal that something is soft.
33
+ ## Task brief (`aunx brief`)
52
34
 
53
- ## 7. The build protocol, bound to lanes
35
+ Each worker reads the shared context file and a brief naming the whole build, its own section, permissions, non-goals, interfaces to preserve and acceptance commands. On a split build, identify who merges and require conflicts to be named before the audit of the final combined result.
54
36
 
55
- Stage 1 Map: the orchestrator sweeps; CLI lanes critique the map at $0. Stage 2: deep tier, one named weak spot and one gap in the request. Stage 4: scanners on the added lines, refuses by default. Stage 5: security-shaped diff → the second coder in read-only audit mode; architecture-shaped → deep tier reviewing build against plan; never both. Two deep checkpoints per build; CLI lanes are uncapped.
37
+ When a requested write is refused by the worker's sandbox, hand that exact file change to an authorized writer and continue independent work. For a background call, arrange a heartbeat; the protocol defines when unchanged output requires investigation.
56
38
 
57
- ## 8. Privacy gate
39
+ ## Review and verify
58
40
 
59
- Name the lanes that never see private notes, client data or personal records. An unnamed bar is not enforced.
41
+ Use one audit step. One reviewer checks build against scope; a companion reviewer checks scope against the user's request at the same time. Prefer different model families from the author. If an independent reviewer is unavailable, record that limitation and arrange the required review before release.
60
42
 
61
- ## 9. Re-derive every figure a cheap lane returns
43
+ The verification role tries to reproduce each claim and probe the named definition of done. Where your main agent has an installed agent set, `finding-verifier` and `done-verifier` supply these prompts. Repairs follow confirmed findings and each fix gets a regression that is demonstrated to fail before the fix.
62
44
 
63
- Measured on the cheapest metered lane: conclusions right, 0 of 11 line citations correct, fabricated arithmetic attached to true observations. That survives a skim. So a number from a lane is a lead until a tool computes it: [codecalc](https://github.com/The-40-Thieves/codecalc) on the orchestrator's side, registered for Codex, Antigravity and Qwen Code with the snippets in `CODECALC.md`.
45
+ ## Research and shared notes
64
46
 
65
- ## 10. One writer, and a store the lanes can all read
47
+ Give independent research questions to tools that can answer them, then inspect primary sources before accepting the claims. Compute consequential figures independently. Use one writer for the final shared record.
66
48
 
67
- With several lanes proposing, the store is where they meet. obsidian-tc (optional) gives every CLI the same `semantic_search`, `get_backlinks` and compare-and-swap `write_note`, with folder ACLs so a research lane can read what it needs and write nothing. The orchestrator stays the one writer.
49
+ Optional companions can help: codecalc for execution and calculations, obsidian-tc for searchable notes, Context7 for current library docs. Without them, use the runtime, notes and official documentation already available to your agent.
68
50
 
69
- ## 11. Docs, then prove, across lanes
51
+ ## Measure your own routing
70
52
 
71
- Every lane's recall of a library's API is a lead, the same as its arithmetic (see item 9 above). Context7 (optional) gives every CLI the same current, version-aware docs lookup, registered for Claude Code, Cursor, Codex and Qwen Code with the snippets in `CONTEXT7.md`. It pairs with codecalc: a lane's claim about what a library does, cited from memory or from a doc, is confirmed by a run before code ships on it.
53
+ `aunx route-metrics --summary` reads your local Claude Code routing log. It reports where work went, route-marker coverage and subagent durations. Your own measurements are the basis for changing assignments and checking whether the rules are being followed.
72
54
 
73
55
  ## What the installer gives you at this level
74
56
 
75
- Everything from Part 1, plus `ROUTING.md` · `TIERS.md` · `DELEGATION_MATRIX.md` (generated from your selection) · `RESEARCH_TRIAGE.md` · `CLI-RUN.md` · `bin/cli-run.mjs` · `bin/lanes.json`.
57
+ Everything from [Part 1](part-1-beginner.md), plus `ROUTING.md`, `TIERS.md`, `DELEGATION_MATRIX.md`, `RESEARCH_TRIAGE.md`, `CLI-RUN.md`, `bin/cli-run.mjs` and `bin/lanes.json`.
76
58
 
77
- ## When you have outgrown it
59
+ ## Run scheduled work
78
60
 
79
- You want the audit to run on a Monday without you, a gateway so nothing but one process holds a key, and a machine that is always on. That is [Part 3](part-3-advanced.md).
61
+ When the setup needs an always-on host or scheduled review, move to [Part 3](part-3-advanced.md).
@@ -1,50 +1,60 @@
1
- # Part 3 · Advanced: everything above, plus a virtual machine
1
+ # Part 3 · Advanced: templates for your always-on Linux machine
2
2
 
3
- A small always-on Linux box owns the schedule. Your laptop stays the interactive driver.
3
+ Level 3 gives configuration and setup scripts for a Linux machine you provision. The box owns the schedule; your laptop stays the interactive driver.
4
4
 
5
- ## 1. Only the gateway holds keys
5
+ ## 1. Keep provider keys in the gateway
6
6
 
7
- One OpenAI-compatible gateway (LiteLLM) fronts every metered provider. Nothing else on the box holds a credential: not the orchestrator, not a job, not a container. Rotating a key is a change in one place. The gateway binds to loopback or a private mesh, never to the public interface.
7
+ One OpenAI-compatible gateway (LiteLLM) fronts your selected pay-per-token providers. Inject provider keys into the gateway and have jobs use its authenticated endpoint. Keep the gateway's access key in your secrets manager and supply it to authorized jobs. The supplied Compose ports bind to loopback; keep any private-network access within your authorized boundary.
8
8
 
9
- Subscription CLIs keep their own sign-in state and stay off the gateway; they are already $0.
9
+ Subscription CLIs keep their own sign-in state and use the access and quota included in your plan.
10
10
 
11
- ## 1b. A subscription is not an API key
11
+ ## 1b. Configure subscription access and API access separately
12
12
 
13
- The installer asks which metered API keys you hold separately from which CLIs you use. A Claude Code plan gives you `claude`; it does not give you an Anthropic API key, and the gateway only serves what a key unlocks. Gateway lanes and the variable names in `vm/ENVIRONMENT.md` are rendered from the keys, never from the CLI list.
13
+ Choose metered API providers with `--apis` or the level 3 edit menu, separately from the CLIs you use. A Claude Code plan gives you `claude`; it does not give you an Anthropic API key. API-backed gateway lanes and their variable names in `vm/ENVIRONMENT.md` come from the selected API providers. Selecting Ollama adds the local alias separately. Level 3 always needs an explicit level choice.
14
14
 
15
- ## 2. Bind to aliases, pin the cheapest lane
15
+ ## 2. Bind to aliases and configure job policies
16
16
 
17
- Each gateway lane is an alias (`bulk-cheap`, `standard`, `deep`, `long-context`, `live-fast`, `local-small`), so a vendor rename is a one-line repoint. Each job is pinned to the cheapest lane that does its work, with a token cap, and climbs the ladder (local → free → cheap → standard → deep) only on failure, low confidence, or an explicit "expensive to get wrong".
17
+ Selected providers and the optional local runtime get gateway aliases such as `bulk-cheap`, `standard`, `deep` and `local-small`. Update an alias's configured model when changing providers or models.
18
+
19
+ For jobs you add, configure an eligible lane, token limits, spending limits and any escalation policy. Define the checks that justify a retry or a stronger model and the authorization each step needs.
20
+
21
+ The supplied weekly audit uses the assigned review lane, falling back to the assigned bulk lane when review is unavailable. The installed stack table explains which capabilities and billing facts chose it. Each run calls that fixed lane through `cli-run` with a timeout. Its behavior is a single report request to that selected lane; token caps and escalation require your own job configuration. A bulk-lane fallback does not establish independent review.
18
22
 
19
23
  ## 3. Dispatch on the box
20
24
 
21
- 1. Deterministic pre-triage at zero tokens: a keyword table routes the obvious cases.
22
- 2. Judgment dispatch for the rest, by the orchestrator, with a logged reason.
23
- 3. Free lane first.
24
- 4. **Unattended means no human-gated escalation.** An unresolved irreversible call is surfaced and stopped, never executed on a model's confidence.
25
- 5. One writer. Every other engine proposes.
25
+ Use `vm/box-CLAUDE.md` as the starting policy for your main agent:
26
+
27
+ 1. When the task is obvious, use the routing table or `aunx route` suggestion, then verify tools and scope.
28
+ 2. When judgment is needed, apply `ROUTING.md` and `DELEGATION_MATRIX.md` and record the choice with a reason.
29
+ 3. When a cheaper eligible lane can satisfy the acceptance checks, select it. When checks fail, diagnose and choose an authorized fallback.
30
+ 4. When an unattended action exceeds the existing authorization, preserve the result and request approval through your configured channel.
31
+ 5. When recording shared state, keep one writer and have other workers propose updates.
26
32
 
27
33
  ## 4. The weekly gap analysis becomes a job
28
34
 
29
- A timer enumerates live state (gateway lanes, timers, CLI versions), diffs it against the delegation matrix, and lets the free lane draft the report. It catches the dead lane and the silently renamed model. Its "watched by" line starts as `nothing`, and that line is the one that tells you what to build next.
35
+ The timer runs a script that collects gateway alias IDs, the user timer listing and version strings for CLIs found on its `PATH`. It passes those observations, `DELEGATION_MATRIX.md` and the gap-analysis protocol to the fixed CLI lane. The lane drafts a report comparing the supplied observations with the intended configuration and names what it could not assess. Failed probes are marked `UNVERIFIED`.
36
+
37
+ Successful output replaces the dated report; a failed run preserves the previous report and keeps partial output separately. Nothing watches the weekly job until you configure a notifier and record it in `vm/jobs/README.md`.
30
38
 
31
39
  ## 5. What runs where
32
40
 
33
41
  | Surface | Role |
34
42
  |---|---|
35
43
  | the orchestrator CLI | interactive driver over SSH; dispatch brain for jobs |
36
- | `cli-run` lanes | the other agent CLIs, headless, signed in by device code |
37
- | the gateway | every metered provider behind one endpoint |
44
+ | `cli-run` lanes | supported agent CLIs, headless, using their vendor sign-ins |
45
+ | the gateway | selected pay-per-token providers behind one endpoint |
38
46
  | a local runtime | the privacy lane |
39
47
  | user-level systemd timers | the schedule |
40
48
 
41
- Headless Linux gotchas the setup script handles: install a keyring or the CLIs re-prompt for auth on every launch; run device-code sign-ins inside `tmux`; invoke CLIs by absolute path from non-login shells.
49
+ Review `vm/setup-vm.sh` before running it: that separately invoked deployment script installs system dependencies, including a keyring, and selected npm vendor CLIs. Complete vendor sign-ins inside `tmux` and inject the names in `vm/ENVIRONMENT.md`. Then, from `vm/`, run `bash setup-vm.sh --start-services`. It starts Compose; when Ollama is selected, it pulls the configured model inside that service and verifies a completion through `local-small`.
50
+
51
+ Before enabling the timer, check `command -v node` and the selected CLI on the box. The service supplies `Environment="PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"`; add their actual absolute directories there when needed. The script resolves `node` and vendor CLIs through that `PATH`. systemd does not expand `$PATH`, `$HOME` or `~` in this setting. Follow `vm/jobs/README.md` to configure the unit and verify a manual run.
42
52
 
43
- ## 6. Rules that do not bend on a box
53
+ ## 6. Set spending and privacy boundaries
44
54
 
45
- - No payment card on any compute lane without a human saying so. Free credit only.
46
- - Local first; cloud only when local genuinely cannot.
47
- - Private notes, client data and personal records never go to a third-party bulk lane. Name the barred lanes.
55
+ - Configure spending and token limits before scheduling a compute job. Include a check that stops for approval when a job would exceed its approved budget.
56
+ - Choose local or hosted execution from capacity, privacy and the approved budget.
57
+ - Fill in `vm/PRIVACY_GATES.md`: classify the data and name the allowed and barred lanes. Confidential data needs explicitly approved processors or a local runtime; personal data needs explicitly authorized processing with the required privacy boundary. Keep protected data out of unapproved bulk lanes.
48
58
  - Nothing binds to `0.0.0.0`.
49
59
  - A secret is never printed, never in argv, never in a file in the repo.
50
60
 
@@ -54,15 +64,15 @@ What and why · trigger · invocation chain · dependencies · reads · writes
54
64
 
55
65
  ## 8. codecalc on the box
56
66
 
57
- Runs as a stdio MCP server next to the orchestrator CLI: offline, no key, nothing to bind. The weekly audit's figures (lane counts, version deltas, spend) are computed there, not estimated by the free lane that drafts the report. See [codecalc](https://github.com/The-40-Thieves/codecalc).
67
+ If selected and separately installed, codecalc runs as a stdio MCP server next to the orchestrator CLI: offline, no key, nothing to bind. For jobs that need arithmetic, configure calculation through codecalc or an available calculator or runtime and supply the inputs to verify. The weekly audit collects the observations listed above and asks its selected lane to compare them with the delegation matrix. See [codecalc](https://github.com/The-40-Thieves/codecalc).
58
68
 
59
69
  ## 9. obsidian-tc on the box
60
70
 
61
- Stdio next to the orchestrator, or the upstream Docker service against a bind-mounted vault. Embeddings on the box's Ollama, so nothing leaves the machine. HTTP transport stays off unless every caller is on the private mesh and auth is on.
71
+ If selected, use stdio next to the orchestrator, or the upstream Docker service against a bind-mounted vault. Without it, use a searchable notes folder with one writer. Embeddings on the box's Ollama, so nothing leaves the machine. HTTP transport stays off unless every caller is on the private mesh and auth is on.
62
72
 
63
- ## 10. context7 on the box
73
+ ## 10. Context7 on the box
64
74
 
65
- Unlike the other two companions, it is never fully local: the hosted endpoint is a network call over HTTPS from the box, or a local `npx` server over stdio still needs no cloud account to run anonymously. Either way, only the library name and the query text leave the box, never source code. Scheduled jobs that write code against a vendored dependency pull its current docs through Context7 first, then prove the shape with codecalc before it ships.
75
+ If selected, Context7 retrieves documentation over the network, including when its MCP server runs locally. Keep private source and secrets out of query text. Without it, read current official docs or upstream source. Then test the API shape with your local runtime before release.
66
76
 
67
77
  ## What the installer gives you at this level
68
78
 
@@ -0,0 +1,38 @@
1
+ # Security review history
2
+
3
+ This page summarizes completed reviews recorded in the [changelog](../CHANGELOG.md) and their regression checks. It records the releases reviewed, what failed and what changed. It carries no claim that a past review verifies a later release.
4
+
5
+ ## Review rounds and fixes
6
+
7
+ | Release | Review recorded | What the review found | What changed and where it is checked |
8
+ |---|---|---|---|
9
+ | 0.1.0 | Initial review and follow-up round, with a second model-family review | Paths could escape the write roots, partial writes could remain, malformed arguments could proceed, and empty results could appear successful | Containment preflight, exclusive writes with rollback, strict flag parsing and vendor-specific result checks; `test/install.test.js`, `test/cli.test.js`, `test/judges.test.js` |
10
+ | 0.1.1 | Follow-up on issues #1 through #10 | Child processes could survive timeouts; a failed scheduled run could replace a good report; logs could contain provider text | Process-group cleanup, temporary report plus rename, bounded probes and fixed log codes; `test/cli.test.js`, `test/install.test.js` |
11
+ | 0.1.2 | Follow-up on issues #12 through #15 | Interrupt cleanup, split UTF-8 output and stale existing files could produce misleading results | Interrupt handlers, streaming decoders and a pre-run snapshot for `--expect-file`; `test/cli.test.js` |
12
+ | 0.1.3 | Independent repository review | Version pins, runtime-upgrade reporting and setup documentation disagreed | Shared catalog rendering, explicit upgrade reports and catalog checks; `test/catalog.test.js`, `test/install.test.js` |
13
+ | 0.1.15 | Pre-release review of routing hooks | Open stdin and non-regular rules files could hang hooks; some review agents were described more restrictively than their tools enforced | Bounded asynchronous stdin, regular-file checks and bounded reads; tool-grant wording corrected; `test/hooks.test.js`, `test/install.test.js` |
14
+ | 0.1.16 | Permission-claim follow-up | Agents with Bash were described as read-only without qualifying the command permission | Every such claim now distinguishes prompt instructions from actual tool restrictions; `test/install.test.js` |
15
+ | 0.1.18 | Windows execution checks and CI follow-up | npm shims needed safe direct execution; watchdog ordering could hide a timeout | Resolve supported shims to Node, refuse unresolved batch targets for runner calls, preserve timeout markers; `test/judges.test.js`, `test/cli.test.js`, `test/install.test.js` |
16
+ | 0.1.20 | Pre-release plugin review | Fallback hook text could exceed the output cap; the safety test missed asynchronous writes and subprocesses | Bound emitted strings and expand the guard with failing examples; `test/hooks.test.js`, `test/plugin.test.js` |
17
+ | 0.1.23 | Pre-release review of failure classification | Redaction order, escaped strings, unbounded scans, transcript containment and clipped-error classification could fail | Redact before clipping, bound scans, resolve containment with path semantics, classify full trusted error fields; `test/classify.test.js` |
18
+ | 0.1.31 | Uninstall validation recorded with release | Managed removal needed full manifest validation and preservation of user edits | Validate paths, hashes and target roots before removing anything; retain edited and foreign files; `test/uninstall.test.js` |
19
+
20
+ The current suite prints its own counts when you run `npm test`. [Platform support](../README.md) documents the tests that require POSIX behavior and the Windows limitations.
21
+
22
+ ## Incident: a successful exit with no new file
23
+
24
+ A worker could finish successfully while leaving an old output file untouched. A filesystem snapshot taken before the call made that observable: `--expect-file` now requires a new or changed file. The regression checks the old-file case as well as a real update. Source: [0.1.2 release record](../CHANGELOG.md#012---2026-09-05), `test/cli.test.js`.
25
+
26
+ ## Incident: a hook waiting forever for input
27
+
28
+ A routing hook waited for stdin to close, and a non-regular rules path could block a file read. The hook now bounds its stdin wait, verifies that a target is a regular file and reads into a bounded buffer. The tests leave stdin open and place a FIFO at the rules path on supported platforms. Source: [0.1.15 release record](../CHANGELOG.md#0115---2026-09-10), `test/hooks.test.js`.
29
+
30
+ ## Incident: clipping before redaction
31
+
32
+ A shortened error string could lose the delimiter needed to recognize a sensitive value. Redacting the complete bounded text before shortening it fixed the ordering problem. The regression covers escaped and unterminated values without publishing real credentials. Source: [0.1.23 release record](../CHANGELOG.md#0123---2026-09-15), `test/classify.test.js`.
33
+
34
+ ## Verify your own installation
35
+
36
+ Run `npm test` for fixtures and stubs. Run `aunx cli-run --doctor --run` through your own sign-ins to check current vendor behavior (direct form from the installed rules folder: `node bin/cli-run.mjs --doctor --run`). Fixture checks establish the behavior they exercise; a live run adds evidence about your installed vendors, credentials and quota.
37
+
38
+ Review current guarantees in [guarantees.md](guarantees.md). Report a vulnerability through the private channel in [SECURITY.md](../SECURITY.md).
package/llms.txt CHANGED
@@ -1,40 +1,39 @@
1
1
  # model-orchestrator
2
2
 
3
- > Model orchestrator for AI coding agents and LLMs (Claude Code, Codex, Antigravity (Google), Grok, Qwen, Ollama). One installer writes routing rules, subagent definitions and a CLI lane runner for the AI tools you already have. The rules tell your agent which model, subagent or CLI to use for each task, so small work goes to cheap tiers and fewer tokens go to frontier models. It sits above the request layer: your agent reads the rules and picks the lane, so the decision stays readable, versioned and editable. Request-level routers and gateways compose underneath it.
3
+ > Model router for AI coding agents: installs routing rules, 8 subagents, hooks and a CLI runner so your AI picks model and effort per task and saves tokens. A lane is an AI tool or model that can receive work; a tier describes model strength and cost. It sits above the request layer: your agent reads the rules and picks the lane. Request-level proxies and gateways can carry API calls underneath it.
4
4
 
5
- Install and run: `npx model-orchestrator` (interactive), or headless: `npx model-orchestrator --yes --level 2 --ais claude-code,codex --project . --dir ./ai-orchestrator`. Preview without writing: add `--dry-run`. List every supported AI: `npx model-orchestrator --list`. Node 18 or newer, zero runtime dependencies, MIT licence.
5
+ Install: `npx model-orchestrator`. Interactive installation applies the main agent's catalog-supported project rules and settings with backups under one confirmation. Use `--no-apply` or the edit screen to keep activation manual. Headless: `npx model-orchestrator --yes --level 2 --ais claude-code,codex --project . --dir ./ai-orchestrator`. Headless activation requires `--apply-snippets`; add `--dry-run` to preview. Every non-dry install runs a local CLI presence health check; live canaries require `--doctor --run` explicitly. "What's left for you" lists only remaining actions, including one paste step for a chat app.
6
6
 
7
- Levels: 1 beginner (one agent or chat app), 2 intermediate (several agent CLIs, each called through `cli-run`), 3 advanced (adds a virtual machine with a gateway and a scheduled audit job). On Claude Code the install also delegates execution to subagents by default and ships three hooks: `route-gate` (UserPromptSubmit) and `subagent-context` (SubagentStart) inject the routing table every turn, and `route-metrics` logs route markers and subagent dispatches.
7
+ Companions default to none, including with `--yes`. With activation enabled, supported project MCP configuration is merged with backups; global configuration stays manual. The installer prints third-party setup commands and never runs installs or login flows. Codex's reliable status check removes an already-completed sign-in from the list; other CLIs receive conditional sign-in instructions. Uninstall removes unchanged recorded activation blocks and added hook or MCP entries while preserving surrounding content and backups.
8
8
 
9
- Claude Code plugin: `/plugin marketplace add aunysillyme/model-orchestrator`, then `/plugin install model-orchestrator@model-orchestrator`. It installs the two read-only hooks (`route-gate`, `subagent-context`) and eight subagents; the routing rules still come from `npx model-orchestrator`, and the routing log (`route-metrics.mjs`) is npm-only.
9
+ Command: `npm install -g model-orchestrator` makes `aunx` available. `aunx` alone runs the installer; `aunx cli-run` runs an AI CLI; `aunx route-metrics --summary` shows local routing activity; `aunx brief`, `aunx context` and `aunx checks` scaffold task context and verification. `aunx checks run ACCEPTANCE_CHECKS.json` executes trusted check commands and exits 1 on any failure. `aunx route "rename this file"` prints an explained suggestion without launching a worker.
10
10
 
11
- Pick a model proxy (LiteLLM, Portkey, OpenRouter, claude-code-router) for per-request model routing underneath an agent; pick model-orchestrator for installed routing rules, subagents, hooks and a lane runner for coding agents and subscription CLIs. They compose.
11
+ Pick a model proxy (LiteLLM, Portkey, OpenRouter, claude-code-router) for per-request model routing underneath an agent; pick model-orchestrator for task delegation using installed rules, subagents, hooks and a CLI runner. They compose.
12
12
 
13
13
  ## Docs
14
14
 
15
- - [README](https://github.com/aunysillyme/model-orchestrator/blob/main/README.md): what it is, a real dry-run plan, the levels, the AIs, the principles
16
- - [Installing](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/install.md): every flag, the two folders a run writes to, headless examples, the full file list
17
- - [How it routes](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/how-it-routes.md): role, complexity and stakes; the three verifier agents; pinning model and effort per lane
18
- - [Guarantees](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/guarantees.md): what is enforced by code, what is delegated to a vendor flag, what is only an instruction
19
- - [Companion tools](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/companions.md): codecalc, obsidian-tc and Context7, what each closes and what it needs first
20
- - [Part 1: beginner](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/part-1-beginner.md): one agent, tiers, the task bundle every delegation carries
21
- - [Part 2: intermediate](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/part-2-intermediate.md): several agent CLIs, the lane runner, pinning model and effort per lane
22
- - [Part 3: advanced](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/part-3-advanced.md): the VM, the gateway, scheduled jobs, privacy gates
23
- - [AI catalog](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/catalog.md): every supported AI, how to sign in, how each is detected
15
+ - [Start here](https://github.com/aunysillyme/model-orchestrator/blob/main/README.md): model routing, activation, commands, search-phrased FAQ and related projects
16
+ - [Install and upgrade](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/install.md): flags, walkthrough, 0.1.x upgrades, project commands and uninstall
17
+ - [How routing works](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/how-it-routes.md): role, complexity, stakes, model and effort, independent verification
18
+ - [Guarantees](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/guarantees.md): code-enforced properties, vendor permissions and agent instructions
19
+ - [Optional companions](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/companions.md): other authors' tools, upstream setup and support, fallback tools
20
+ - [Beginner](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/part-1-beginner.md): model tiers, a task brief, a context file and acceptance checks inside one agent
21
+ - [Intermediate](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/part-2-intermediate.md): subscription lanes, pay-per-token lanes, research and shared work
22
+ - [Advanced](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/part-3-advanced.md): Linux host, gateway templates, scheduled work and privacy boundaries
23
+ - [Catalog](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/catalog.md): supported AIs, install commands, sign-in and detection
24
24
 
25
25
  ## Reference
26
26
 
27
- - [Claude Code plugin](https://github.com/aunysillyme/model-orchestrator/blob/main/plugin/README.md): installing the hooks and subagents with `/plugin install`, what the plugin reads, and what it leaves to the installer
28
- - [CLI runner](https://github.com/aunysillyme/model-orchestrator/blob/main/bin/README.md): `cli-run` lanes, exit codes, the route logged per run
29
- - [Templates](https://github.com/aunysillyme/model-orchestrator/blob/main/templates/README.md): the routing, tiers, task bundle and protocol files the installer renders
30
- - [Changelog](https://github.com/aunysillyme/model-orchestrator/blob/main/CHANGELOG.md): every release and the issue behind each fix
31
- - [Agent instructions](https://github.com/aunysillyme/model-orchestrator/blob/main/AGENTS.md): running the installer from an agent, and contributing
32
-
33
- ## Plans and automatic effort
34
-
35
- - [Plans and automatic effort](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/install.md#plans-and-automatic-effort): `--plans AI=plan` states a subscription plan so the guidance allocates volume by headroom; opt-in `--effort-auto` sizes effort per call on high or max headroom CLI lanes, medium or high, with xhigh named explicitly for security-critical or irreversible work
27
+ - [Commands](https://github.com/aunysillyme/model-orchestrator/blob/main/bin/README.md): aunx subcommands, CLI runner exit codes and output contracts
28
+ - [Templates](https://github.com/aunysillyme/model-orchestrator/blob/main/templates/README.md): routing rules, task brief, context, acceptance checks and protocols
29
+ - [Claude Code plugin](https://github.com/aunysillyme/model-orchestrator/blob/main/plugin/README.md): installing read-only routing hooks and subagents through the plugin marketplace
30
+ - [Proof](https://github.com/aunysillyme/model-orchestrator/blob/main/proof/README.md): dated measurements, methods, sample sizes, expiry and reproduction scripts
31
+ - [Security review history](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/security-review-history.md): completed review rounds, incident summaries and regression evidence
32
+ - [Changelog](https://github.com/aunysillyme/model-orchestrator/blob/main/CHANGELOG.md): release changes and upgrade notes
33
+ - [Agent instructions](https://github.com/aunysillyme/model-orchestrator/blob/main/AGENTS.md): headless setup and contributor checks
36
34
 
37
35
  ## Optional
38
36
 
39
- - [Security policy](https://github.com/aunysillyme/model-orchestrator/blob/main/SECURITY.md)
40
- - [Audit brief](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/audit-brief.md): the security notes and what has already been security-reviewed
37
+ - [Security policy](https://github.com/aunysillyme/model-orchestrator/blob/main/SECURITY.md): private vulnerability reporting
38
+ - [agent-personalizer](https://github.com/aunysillyme/agent-personalizer): one interview writes the profile and rules every AI reads, kept in sync from one source
39
+ - [website-build-skill](https://github.com/aunysillyme/website-build-skill): current website-building expertise for your AI, from research and design through accessibility and security
package/package.json CHANGED
@@ -1,23 +1,25 @@
1
1
  {
2
2
  "name": "model-orchestrator",
3
- "version": "0.1.35",
4
- "description": "Model orchestrator: rules, subagents and a lane runner tell agents which model to use, so small work goes to cheap tiers and fewer tokens go to frontier models.",
3
+ "version": "1.0.0",
4
+ "description": "Model router for AI coding agents: installs routing rules, 8 subagents, hooks and a CLI runner so your AI picks model and effort per task and saves tokens",
5
5
  "type": "module",
6
6
  "bin": {
7
- "model-orchestrator": "bin/cli.js"
7
+ "model-orchestrator": "bin/cli.js",
8
+ "aunx": "bin/aunx.js"
8
9
  },
9
10
  "files": [
10
11
  "bin",
11
12
  "src",
12
13
  "templates",
13
14
  "docs",
15
+ "!docs/router-trailer.gif",
14
16
  "README.md",
15
17
  "LICENSE",
16
18
  "CHANGELOG.md",
17
19
  "SECURITY.md",
18
- "scripts",
19
20
  "llms.txt",
20
- "AGENTS.md"
21
+ "AGENTS.md",
22
+ "proof"
21
23
  ],
22
24
  "scripts": {
23
25
  "start": "node bin/cli.js",
@@ -25,7 +27,10 @@
25
27
  "prepublishOnly": "npm test",
26
28
  "dry-run": "node bin/cli.js --yes --level 2 --ais claude-code,codex,grok --dir ./tmp-dry-run --dry",
27
29
  "gen:catalog": "node scripts/gen-catalog.js",
28
- "gen:plugin": "node scripts/gen-plugin.js"
30
+ "gen:plugin": "node scripts/gen-plugin.js",
31
+ "proof:measure": "node proof/scripts/measure.js",
32
+ "proof:render": "node proof/scripts/render.js",
33
+ "proof:demo": "node proof/scripts/record-gate.js"
29
34
  },
30
35
  "engines": {
31
36
  "node": ">=18"
@@ -81,8 +86,10 @@
81
86
  "prompt-routing",
82
87
  "coding-agent",
83
88
  "ai-coding",
84
- "coding-agents"
89
+ "coding-agents",
90
+ "aunx"
85
91
  ],
86
92
  "author": "aunysillyme (https://github.com/aunysillyme)",
87
- "license": "MIT"
93
+ "license": "MIT",
94
+ "funding": "https://github.com/sponsors/aunysillyme"
88
95
  }
@@ -0,0 +1,100 @@
1
+ # Reproduce the measurements
2
+
3
+ Generated from [results.json](results.json). Each figure has a method, sample size, measurement date and expiry. Run the scripts on your own machine to compare.
4
+
5
+ Environment: Node v22.22.3, darwin arm64. Timing varies with startup caches and other work on the machine. Synthetic cases show what those fixtures exercise.
6
+
7
+ | Measurement | Result | Sample size | Measured | Expires | Reproduce |
8
+ |---|---|---|---|---|---|
9
+ | Dry install wall time | 42.22 ms median | 7 | 2026-09-27 | 2026-10-11 | [script](../proof/scripts/install-time.js) |
10
+ | Lane runner overhead | 63.74 ms median difference | 7 | 2026-09-27 | 2026-10-11 | [script](../proof/scripts/runner-overhead.js) |
11
+ | Empty results flagged | 10 fixtures rejected | 10 | 2026-09-27 | 2026-10-11 | [script](../proof/scripts/missing-results.js) |
12
+ | Acceptance failures blocked | 4 fixtures rejected | 4 | 2026-09-27 | 2026-10-11 | [script](../proof/scripts/check-gate.js) |
13
+ | Main conversation browser tokens | 408147 tokens per browser step (median) | 9775 | 2026-09-26 | 2026-10-27 | author setup, re-measured locally |
14
+ | Small browser subagent tokens | 17197 tokens per step (highest of 5 runs) | 5 | 2026-09-27 | 2026-10-27 | author setup, re-measured locally |
15
+
16
+ ## Run the proof scripts
17
+
18
+ ```sh
19
+ node proof/scripts/measure.js
20
+ node proof/scripts/render.js
21
+ npm test
22
+ ```
23
+
24
+ The measurement command refreshes reproducible entries and preserves separately sourced author-setup entries. The test suite rejects expired, future-dated or incomplete entries and checks this page against the data. The weekly [refresh workflow](../.github/workflows/proof.yml) reruns the scripts and commits their data and generated page.
25
+
26
+ ## Try the acceptance gate
27
+
28
+ ```sh
29
+ aunx checks ACCEPTANCE_CHECKS.json
30
+ aunx checks run ACCEPTANCE_CHECKS.json
31
+ ```
32
+
33
+ The scaffold starts red. Replace the sample with commands that prove your requirements, then put `aunx checks run ACCEPTANCE_CHECKS.json && <your-release-command>` in your own release sequence. Commands are local code you review before running. Manual evidence stays UNVERIFIED and blocks the gate.
34
+
35
+ ![Acceptance gate rejects a missing output, then passes after the output exists](gate-demo.gif)
36
+
37
+ The [recording script](scripts/record-gate.js) captures real command output into an asciicast, then renders it with an already installed agg. Companion tools are installed by their users.
38
+
39
+ ## Measurement methods
40
+
41
+ ### Dry install wall time
42
+
43
+ Spawn a fresh Node installer process per sample; level 2, Claude Code + Codex, no companions, --dry. Includes Node startup and planning; writes no install files. Isolated home and PATH, no real vendors.
44
+
45
+ Kind: reproducible local measurement. Sample size: 7. Measured: 2026-09-27. Expires: 2026-10-11.
46
+
47
+ Source: [proof/scripts/install-time.js](../proof/scripts/install-time.js).
48
+
49
+ ### Lane runner overhead
50
+
51
+ Paired fresh processes: direct Node stub versus cli-run hermes with the same stub. Alternates pair order. Includes wrapper startup, validation and local log writes; excludes vendor/network/model time.
52
+
53
+ Kind: reproducible local measurement. Sample size: 7. Measured: 2026-09-27. Expires: 2026-10-11.
54
+
55
+ Source: [proof/scripts/runner-overhead.js](../proof/scripts/runner-overhead.js).
56
+
57
+ ### Empty results flagged
58
+
59
+ Run cli-run against an exit-0 stub for every supported lane, once with empty stdout and once with an empty native final result. Count exit 10/11 only. A successful Hermes response is the positive control. Synthetic fixtures measure these shapes only.
60
+
61
+ Kind: reproducible local measurement. Sample size: 10. Measured: 2026-09-27. Expires: 2026-10-11.
62
+
63
+ Source: [proof/scripts/missing-results.js](../proof/scripts/missing-results.js).
64
+
65
+ ### Acceptance failures blocked
66
+
67
+ Run aunx checks run against nonzero, manual, missing-program and timeout fixtures. Each must exit 1; a passing command must exit 0. This is a local command gate, activated by the user in their release sequence.
68
+
69
+ Kind: reproducible local measurement. Sample size: 4. Measured: 2026-09-27. Expires: 2026-10-11.
70
+
71
+ Source: [proof/scripts/check-gate.js](../proof/scripts/check-gate.js).
72
+
73
+ ### Main conversation browser tokens
74
+
75
+ Measured on the author's Claude Code sessions: every browser tool call in the transcripts, counting tokens re-read by the main conversation per step. Median: 408,147 tokens per browser step across 145 sessions and 9,775 browser steps. Sample size counts browser steps.
76
+
77
+ Kind: measured on the author's setup. Sample size: 9775. Measured: 2026-09-26. Expires: 2026-10-27.
78
+
79
+ Source: author setup, re-measured locally.
80
+
81
+ ### Small browser subagent tokens
82
+
83
+ At least 23x fewer tokens per step in this sample: the main conversation median of 408,147 divided by the highest subagent run of 17,197 is 23.73x. Measured on the author's Claude Code sessions, with the same kind of work handed to a small browser subagent. Five runs, tokens divided by steps or tool calls per run: 17,197 (206,369 tokens / 12 steps, 2026-09-26); 4,077 (93,773 / 23), 4,201 (105,034 / 25), 4,585 (91,709 / 20), and 4,489 (94,263 / 21), all four on 2026-09-27. Range: 4,077 to 17,197; median of 5 runs: 4,489. Typical context, using the median of 5 runs: about 91x fewer tokens per step. These are measurements from the author's own sessions, not a controlled comparison or a guarantee for other setups. The four 2026-09-27 runs shared one browser pane, so some steps were spent recovering a drifting tab, which raises the step count and lowers per-step tokens. The first run counted steps; the later runs counted tool calls. Sample size counts runs.
84
+
85
+ Kind: measured on the author's setup. Sample size: 5. Measured: 2026-09-27. Expires: 2026-10-27.
86
+
87
+ Source: author setup, re-measured locally.
88
+
89
+ ## Operation and verification
90
+
91
+ - **What and why:** executable measurements keep public figures traceable to current output.
92
+ - **Trigger:** weekly schedule, workflow dispatch, or `node proof/scripts/measure.js`.
93
+ - **Invocation chain:** workflow -> measurement functions -> isolated Node fixtures -> results.json -> this page -> npm test.
94
+ - **Dependencies:** Node and the repository. The optional GIF recorder uses agg from the asciinema project.
95
+ - **Reads:** package scripts, the installer, runner and acceptance-check runner. Fixture tests use an isolated home and PATH.
96
+ - **Writes:** results.json, this generated page, temporary fixture directories and local fixture logs. The recorder writes gate-demo.cast and gate-demo.gif.
97
+ - **Closed loop:** the workflow fails when measurement or tests fail. GitHub Actions records the failure; repository notification settings decide who receives it. No separate alert service is configured.
98
+ - **Failure modes:** runner behavior changes, an expired catalog snapshot, missing runtime, unavailable write permission, or timing noise. Review the failed job, rerun locally, and send a reproducible issue to the repository maintainers.
99
+ - **Run and verify:** run the commands above, inspect sample arrays and fixture exit codes in results.json, and require npm test to pass. A future-time unit test proves expiry can fail.
100
+ - **Source of truth:** results.json and the scripts it names. Author-setup evidence is added separately by its owner.
@@ -0,0 +1,9 @@
1
+ {"version":2,"width":86,"height":15,"title":"Acceptance checks: red, then green","env":{"TERM":"xterm-256color"}}
2
+ [0,"o","\u001b[1;32m$\u001b[0m aunx checks run checks.json\r\n"]
3
+ [0.6,"o","\u001b[31mFAIL output-exists: exit 1\r\n\u001b[0m"]
4
+ [1.2,"o","$ echo $?\r\n1\r\n"]
5
+ [3,"o","\r\n$ node -e 'require(\"node:fs\").writeFileSync(\"output.txt\",\"verified output\")'\r\n"]
6
+ [4.5,"o","\r\n\u001b[1;32m$\u001b[0m aunx checks run checks.json\r\n"]
7
+ [5,"o","\u001b[32mPASS output-exists: exit 0\r\n\u001b[0m"]
8
+ [5.5,"o","$ echo $?\r\n0\r\n"]
9
+ [7,"o","\r\nThe exit code gates the next command in your release sequence.\r\n"]
Binary file