model-orchestrator 0.1.34 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/AGENTS.md +31 -21
  2. package/CHANGELOG.md +51 -1
  3. package/README.md +127 -110
  4. package/bin/README.md +57 -6
  5. package/bin/aunx.js +7 -0
  6. package/bin/cli-run.mjs +21 -15
  7. package/bin/cli.js +376 -257
  8. package/docs/README.md +15 -18
  9. package/docs/catalog.md +228 -38
  10. package/docs/companions.md +28 -10
  11. package/docs/guarantees.md +21 -12
  12. package/docs/how-it-routes.md +49 -42
  13. package/docs/install.md +135 -33
  14. package/docs/part-1-beginner.md +37 -45
  15. package/docs/part-2-intermediate.md +34 -52
  16. package/docs/part-3-advanced.md +36 -26
  17. package/docs/security-review-history.md +38 -0
  18. package/llms.txt +24 -25
  19. package/package.json +16 -8
  20. package/proof/README.md +100 -0
  21. package/proof/gate-demo.cast +9 -0
  22. package/proof/gate-demo.gif +0 -0
  23. package/proof/results.json +198 -0
  24. package/proof/scripts/check-gate.js +26 -0
  25. package/proof/scripts/install-time.js +16 -0
  26. package/proof/scripts/lib.js +73 -0
  27. package/proof/scripts/measure.js +15 -0
  28. package/proof/scripts/missing-results.js +30 -0
  29. package/proof/scripts/record-gate.js +38 -0
  30. package/proof/scripts/render.js +18 -0
  31. package/proof/scripts/runner-overhead.js +21 -0
  32. package/src/README.md +9 -3
  33. package/src/activation-ownership.js +19 -0
  34. package/src/apply-companions.js +104 -0
  35. package/src/apply-snippets.js +60 -28
  36. package/src/aunx.js +262 -0
  37. package/src/catalog.js +253 -117
  38. package/src/install.js +478 -209
  39. package/src/plugin.js +13 -4
  40. package/src/postinstall.js +57 -0
  41. package/src/roles.js +184 -0
  42. package/src/uninstall.js +125 -8
  43. package/templates/README.md +19 -2
  44. package/templates/advanced/README.md +2 -2
  45. package/templates/advanced/vm/PRIVACY_GATES.md +17 -19
  46. package/templates/advanced/vm/README.md +25 -20
  47. package/templates/advanced/vm/box-CLAUDE.md +19 -18
  48. package/templates/advanced/vm/jobs/README.md +3 -1
  49. package/templates/advanced/vm/jobs/weekly-audit.service +3 -0
  50. package/templates/advanced/vm/jobs/weekly-audit.sh +2 -2
  51. package/templates/advanced/vm/setup-vm.sh +49 -2
  52. package/templates/agents/README.md +2 -2
  53. package/templates/agents/agy/README.md +20 -3
  54. package/templates/agents/agy/builder.md +11 -7
  55. package/templates/agents/agy/bulk-worker.md +9 -7
  56. package/templates/agents/agy/code-reviewer.md +13 -7
  57. package/templates/agents/agy/deep-planner.md +10 -7
  58. package/templates/agents/agy/done-verifier.md +13 -22
  59. package/templates/agents/agy/finding-verifier.md +14 -22
  60. package/templates/agents/agy/live-researcher.md +10 -7
  61. package/templates/agents/agy/reader.md +10 -12
  62. package/templates/agents/claude-code/README.md +18 -14
  63. package/templates/agents/claude-code/builder.md +10 -15
  64. package/templates/agents/claude-code/bulk-worker.md +8 -10
  65. package/templates/agents/claude-code/code-reviewer.md +11 -17
  66. package/templates/agents/claude-code/deep-planner.md +9 -11
  67. package/templates/agents/claude-code/done-verifier.md +12 -33
  68. package/templates/agents/claude-code/finding-verifier.md +13 -39
  69. package/templates/agents/claude-code/live-researcher.md +9 -11
  70. package/templates/agents/claude-code/reader.md +9 -18
  71. package/templates/agents/snippets/chat.md +9 -10
  72. package/templates/agents/snippets/claude-code.md +17 -18
  73. package/templates/agents/snippets/generic.md +9 -11
  74. package/templates/agents/snippets/route-gate.mjs +2 -2
  75. package/templates/agents/snippets/route-metrics.mjs +1 -1
  76. package/templates/agents/snippets/subagent-context.mjs +4 -4
  77. package/templates/beginner/ORCHESTRATOR.md +31 -36
  78. package/templates/beginner/README.md +1 -1
  79. package/templates/common/ACCEPTANCE_CHECKS.json +12 -0
  80. package/templates/common/CONTEXT.md +37 -0
  81. package/templates/common/DECISIONS.md +11 -0
  82. package/templates/common/README.md +24 -11
  83. package/templates/common/TASK_BRIEF.md +84 -0
  84. package/templates/common/protocols/README.md +14 -11
  85. package/templates/common/protocols/acceptance-checks.md +14 -0
  86. package/templates/common/protocols/build-protocol.md +91 -106
  87. package/templates/common/protocols/context-file.md +10 -0
  88. package/templates/common/protocols/decision-log.md +9 -0
  89. package/templates/common/protocols/deep-research.md +20 -34
  90. package/templates/common/protocols/docs-then-prove.md +13 -18
  91. package/templates/common/protocols/gap-analysis.md +15 -21
  92. package/templates/common/protocols/memory-and-record.md +21 -20
  93. package/templates/common/protocols/numbers-and-logic.md +20 -26
  94. package/templates/common/protocols/propagate.md +18 -27
  95. package/templates/intermediate/CLI-RUN.md +83 -113
  96. package/templates/intermediate/DELEGATION_MATRIX.md +9 -3
  97. package/templates/intermediate/README.md +3 -3
  98. package/templates/intermediate/RESEARCH_TRIAGE.md +23 -15
  99. package/templates/intermediate/ROUTING.md +54 -51
  100. package/templates/intermediate/TIERS.md +37 -76
  101. package/templates/tools/README.md +1 -1
  102. package/templates/tools/obsidian-tc/OBSIDIAN-TC.md +1 -1
  103. package/docs/audit-brief.md +0 -148
  104. package/scripts/README.md +0 -7
  105. package/scripts/gen-catalog.js +0 -81
  106. package/scripts/gen-plugin.js +0 -16
  107. package/scripts/record-demo.sh +0 -45
  108. package/templates/common/TASK_BUNDLE.md +0 -56
@@ -1,19 +1,19 @@
1
- # vm/: the box that runs it unattended
1
+ # vm/: templates for your always-on Linux machine
2
2
 
3
- Level 3 = levels 1 and 2 plus a machine that is always on. A small Linux VM (any cloud's free ARM tier is enough) that holds the CLIs, a model gateway, and the scheduled jobs. Your laptop stays the interactive driver; the box owns the schedule.
3
+ Level 3 adds deployment templates to levels 1 and 2. Provide a Linux machine sized for your workload to hold the CLIs, a model gateway, and the scheduled jobs. Your laptop stays the interactive driver; the machine owns the schedule after you configure it.
4
4
 
5
5
  Generated {{DATE}} for: `{{AI_IDS}}`. Installed at `{{INSTALL_DIR}}`; the systemd unit and the audit script carry that path.
6
6
 
7
- ## The one architectural property
7
+ ## Keep provider credentials in the gateway
8
8
 
9
- **Only the gateway holds provider credentials.** Nothing else on the box does: not the orchestrator, not the scheduler, not a job. Every surface reaches models through the gateway, so rotating a key is a change in exactly one place. The gateway is bound to loopback (or a private mesh network), never to `0.0.0.0`.
9
+ For metered API calls, inject provider credentials into the gateway and have jobs use its authenticated endpoint. Subscription CLIs keep their own vendor login state. Bind the gateway to loopback or the configured private network; never expose it publicly without explicit authorization and access controls.
10
10
 
11
11
  ## What runs where
12
12
 
13
13
  | Surface | Role | Reaches models via |
14
14
  |---|---|---|
15
15
  | The orchestrator CLI ({{PRIMARY_NAME}}) | interactive driver when you SSH in; dispatch brain for jobs | its own subscription, off the gateway |
16
- | `cli-run` lanes ({{CLI_RUN_LANES}}) | the other agent CLIs, headless | their own subscriptions (Lane A) |
16
+ | `cli-run` lanes ({{CLI_RUN_LANES}}) | the other agent CLIs, headless | their own subscriptions (subscription lanes) |
17
17
  | The gateway (`docker-compose.yml`) | one OpenAI-compatible endpoint fronting every metered provider | provider keys from the environment |
18
18
  | A local model runtime (if selected) | the privacy lane | nothing leaves the box |
19
19
  | Scheduled jobs (`jobs/`) | the weekly gap-analysis audit (lane: `{{AUDIT_LANE}}`), and anything else recurring | the gateway, or `cli-run` |
@@ -22,25 +22,29 @@ Generated {{DATE}} for: `{{AI_IDS}}`. Installed at `{{INSTALL_DIR}}`; the system
22
22
 
23
23
  ## Setup, in order
24
24
 
25
- 1. Provision a box. Ubuntu, 2+ vCPU, 8 GB is comfortable. Put it on a private mesh network if you can; do not open ports to the internet.
25
+ The model-orchestrator installer writes these files only. When you separately invoke the generated `setup-vm.sh`, that manual deployment script installs the configured system dependencies and npm vendor CLIs. Review it before running it.
26
+
27
+ 1. Provision a machine sized for your workload and put it on the configured private network.
26
28
  2. `bash setup-vm.sh`. It installs system deps and the npm-installable CLIs, then **prints** the vendor shell installers for the rest. Read those scripts before running them.
27
29
  3. Sign each CLI in, using the flow its vendor gives you. Run these inside `tmux` so a dropped SSH session does not kill the prompt. Headless Linux has no keyring by default; `setup-vm.sh` installs one so the CLIs stop re-prompting.
28
30
  {{VM_SIGNIN}}
29
31
  4. Put provider keys in your secrets manager and export the names listed in `ENVIRONMENT.md` into the gateway's environment at start time. Never write a value into a file in this folder. The gateway config was rendered from the API keys you said you hold, not from your CLI subscriptions: those are different entitlements.
30
- 5. `docker compose up -d`, then list the lanes without putting the key in argv (the key must be a single token, `^[A-Za-z0-9._-]+$`, because it is interpolated into curl's config grammar):
32
+ 5. From this `vm/` folder, run `bash setup-vm.sh --start-services` with those environment variables injected. It starts Compose and validates `GATEWAY_MASTER_KEY` as a single token matching `^[A-Za-z0-9._-]+$`. {{VM_LOCAL_SETUP}}
33
+
34
+ To inspect configured aliases afterward, keep the key out of argv:
31
35
  ```bash
32
36
  printf 'header = "Authorization: Bearer %s"\n' "$GATEWAY_MASTER_KEY" | curl -s --config - http://127.0.0.1:4000/v1/models
33
37
  ```
34
- 6. Install the weekly audit: `jobs/README.md`.
38
+ 6. Install the weekly audit: `jobs/README.md`. Set the service's literal `PATH` to include the directories that hold your Node and selected CLI executables before enabling the timer.
35
39
  7. Copy `box-CLAUDE.md` to `~/CLAUDE.md` on the box (or your agent's equivalent rules file) so a session there inherits the house rules without you present.
36
40
 
37
41
  ## The dispatch shape
38
42
 
39
- 1. **Deterministic pre-triage, zero tokens:** a keyword table sends the obvious cases (bulk patterns → the cheap lane, URLs and current events → the live lane, "review this" → the reviewer, "write this up" → the orchestrator).
40
- 2. **Judgment dispatch:** everything else is routed by the orchestrator against `ROUTING.md` and `DELEGATION_MATRIX.md`, with a logged reason.
41
- 3. **Free lane first, escalate on signal.** Every job starts on its $0 lane and climbs only on failure, low confidence, or an explicit "expensive to get wrong".
42
- 4. **Unattended means no escalation to a human-gated tier.** An unresolved irreversible call is surfaced (a message, a ticket comment), never executed.
43
- 5. **Writes stay locked to one writer.** Every other engine proposes; one writer records.
43
+ 1. When the task is obvious, use the routing table or `aunx route` suggestion to identify the candidate tier, then verify tools and scope.
44
+ 2. When judgment is needed, apply `ROUTING.md` and `DELEGATION_MATRIX.md` and record the selected route with a reason.
45
+ 3. When a cheaper eligible route can satisfy the checks, select it; when checks fail or required capability is absent, diagnose and choose an authorized fallback.
46
+ 4. When an unattended action exceeds the existing mandate, preserve the result and return the needed approval through the configured channel.
47
+ 5. When recording shared state, keep one writer and have other workers return proposed updates.
44
48
 
45
49
  ## The closed loop (name what watches it)
46
50
 
@@ -52,11 +56,12 @@ Generated {{DATE}} for: `{{AI_IDS}}`. Installed at `{{INSTALL_DIR}}`; the system
52
56
 
53
57
  "Nothing watches it" is a valid answer and usually the valuable one. Writing it down turns an invisible gap into a tracked one.
54
58
 
55
- ## Never on this box
59
+ ## Keep the server within its scope
56
60
 
57
- - Vendor scripts run blind. Read first.
58
- - An unpinned image or package. `docker-compose.yml` and `setup-vm.sh` pin versions; bump them on purpose, never by restarting.
59
- - A provider key in a file in this folder, in shell history, in argv, or in a container image.
60
- - A service bound to `0.0.0.0`.
61
- - Private notes, client data, or personal records sent to a metered bulk lane. See `PRIVACY_GATES.md`.
62
- - A payment card attached to a compute lane "to unlock a tier". Free credit only unless a human says otherwise.
61
+ - Read vendor scripts before executing them.
62
+ - Update pinned packages and images deliberately, with a verification plan.
63
+ - Keep provider secrets in the manager and out of argv, logs and generated files.
64
+ - Bind services to loopback or the approved private network.
65
+ - Apply the named data-processing permissions in `PRIVACY_GATES.md` before dispatch.
66
+ - Obtain authorization before adding a paid resource or a payment method.
67
+ - When optional companion tools are absent, use the local runtime, official documentation and project files specified by the protocols.
@@ -1,28 +1,29 @@
1
- # CLAUDE.md for the box
1
+ # Project instructions for the server
2
2
 
3
- Copy to `~/CLAUDE.md` on the machine (or your agent's equivalent rules file). A session here inherits these without a human present.
3
+ When activating an unattended machine, copy these rules to the agent's documented instructions path and verify that a fresh session loads them.
4
4
 
5
- ## Cost rule
6
- Prefer the cheapest tier that does the job well. Delegate grunt work through the dispatch layer; keep the reasoning in-session.
5
+ ## Choose the route
7
6
 
8
- Send to the cheap tier via the gateway or `cli-run`: bulk classification and tagging, reformatting, extraction, first-pass summaries, mechanical transforms, explicit rough drafts.
7
+ When work is classification, extraction, formatting or other bounded volume, select an eligible cheap model. When it requires architecture, writing code, review or current sources, use the tier and tools named in `{{RULES_PATH}}/ROUTING.md` and `{{RULES_PATH}}/DELEGATION_MATRIX.md`.
9
8
 
10
- Never send to the cheap tier: anything that will be published in a person's own voice, code that gets committed, anything needing current vendor-specific knowledge, anything where being wrong is expensive, anything time-sensitive.
9
+ When dispatch fails, inspect its class and report the cause. Continue locally only when the session has the required scope, tools and capacity.
11
10
 
12
- ## On dispatch failure
13
- Report it and fall back to doing the work in-session. Never silently retry the same lane.
11
+ ## Keep service ingress private
14
12
 
15
- ## Zero ingress
16
- Nothing binds to `0.0.0.0`. Nothing publishes a container port to the public interface. New services go on loopback or the private mesh.
13
+ Bind services to loopback or the configured private network. Never publish a container port on the public interface without explicit authorization and the required access controls.
17
14
 
18
- ## Unattended means no human-gated escalation
19
- An unresolved call that is irreversible or rewrites a standing rule gets surfaced (a message, a ticket comment) and stops. It is never executed on the strength of a model's confidence.
15
+ ## Handle unattended decisions
20
16
 
21
- ## One writer
22
- Scheduled jobs and other engines propose. One writer records. If you are not that writer, produce a file and name it in your report.
17
+ When an action exceeds the existing mandate, preserve the checked result and return the needed approval through the configured channel. Continue independent authorized work. Never execute an irreversible action solely on a model's confidence.
23
18
 
24
- ## Secrets
25
- Names in the environment, values in the secrets manager. Never print one, never pass one in argv, never write one to disk here.
19
+ ## Record with one writer
26
20
 
27
- ## Routing
28
- `{{RULES_PATH}}/ROUTING.md` and `{{RULES_PATH}}/DELEGATION_MATRIX.md` are the rules. `{{RULES_PATH}}/protocols/` are the procedures.
21
+ When scheduled jobs or other engines produce evidence, return proposed updates to the designated record writer. Keep raw machine logs separate from curated records.
22
+
23
+ ## Protect secrets
24
+
25
+ Load secret values from the configured manager at runtime. Never print them, put them in argv, or write them into this generated folder.
26
+
27
+ ## Run and verify
28
+
29
+ When building, follow `{{RULES_PATH}}/protocols/build-protocol.md`. For background work, arm the five-minute heartbeat at launch and diagnose two checks without progress. When optional companions are absent, use the local runtime, official docs and project records described by each protocol.
@@ -4,12 +4,14 @@ Scheduled work on the box, as user-level systemd timers. Each job is a timer + s
4
4
 
5
5
  | Job | Schedule | Does | Lane | Watched by |
6
6
  |---|---|---|---|---|
7
- | `weekly-audit` | Monday 09:00 | collects live state (gateway lanes, timers, CLI versions), composes a brief with the protocol and `DELEGATION_MATRIX.md`, and asks a cli-run lane for the gap report | `{{AUDIT_LANE}}` (first enabled lane at install time; edit `AUDIT_LANE` in the script to change it) | nothing yet: wire a notifier and update this line |
7
+ | `weekly-audit` | Monday 09:00 | collects live state (gateway lanes, timers, CLI versions), composes a brief with the protocol and `DELEGATION_MATRIX.md`, and asks a cli-run lane for the gap report | `{{AUDIT_LANE}}` (first enabled route at install time; edit `AUDIT_LANE` in the script to change it) | nothing yet: wire a notifier and update this line |
8
8
 
9
9
  Paths in the service and the script were rendered for this install: `{{INSTALL_DIR}}`. If you move the folder, re-run the installer or edit both files.
10
10
 
11
11
  ## Install a job
12
12
 
13
+ Before copying the unit, run `command -v node` and `command -v <selected-cli>` in the account that will run the timer. The service sets `PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin`; add the absolute parent directories of your actual Node and vendor CLI executables to its `Environment="PATH=..."` line when needed. The script resolves `node` and vendor commands through that PATH. systemd does not expand `$PATH`, `$HOME` or `~` in this setting, so write complete directories and retain the system entries. Shell profile files and interactive version-manager initialization are not loaded.
14
+
13
15
  ```bash
14
16
  mkdir -p ~/.config/systemd/user
15
17
  cp weekly-audit.service weekly-audit.timer ~/.config/systemd/user/
@@ -5,6 +5,9 @@ After=network-online.target
5
5
  [Service]
6
6
  Type=oneshot
7
7
  WorkingDirectory={{INSTALL_DIR_SYSTEMD}}
8
+ # Add absolute Node and vendor CLI directories if they live outside these defaults.
9
+ # systemd does not expand shell variables in Environment=.
10
+ Environment="PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
8
11
  # Names come from a file OUTSIDE this repo, mode 600. Edit the path.
9
12
  EnvironmentFile=%h/.config/ai-orchestrator/gateway.env
10
13
  ExecStart=/bin/bash "{{INSTALL_DIR_SYSTEMD}}/vm/jobs/weekly-audit.sh"
@@ -93,11 +93,11 @@ STAMP="$(date -u +%Y%m%dT%H%M%SZ)"
93
93
  } > "reports/live-state-$STAMP.md"
94
94
  ln -sfn "live-state-$STAMP.md" reports/live-state.md
95
95
 
96
- # The brief the lane actually reads: the protocol, the intended configuration,
96
+ # The brief the worker actually reads: the protocol, the intended configuration,
97
97
  # and the live state it is meant to diff against.
98
98
  BRIEF="reports/audit-brief-$STAMP.md"
99
99
  {
100
- echo "## Task bundle"
100
+ echo "## Task brief"
101
101
  echo "**Purpose.** Weekly gap analysis: compare the live state below with the intended configuration and name what is missing, dead, or drifted."
102
102
  echo "**Task class.** read_only"
103
103
  echo "**Denied actions.** Do not run commands, do not modify files, do not call any network service. Enforcement: {{AUDIT_LANE_BOUNDARY_NOTE}}."
@@ -8,6 +8,53 @@ set -euo pipefail
8
8
  say() { printf '\n[setup-vm] %s\n' "$*"; }
9
9
  have() { command -v "$1" >/dev/null 2>&1; }
10
10
 
11
+ # Run this phase explicitly after the dependency setup, sign-ins and secret injection.
12
+ # The package installer only writes this script; it never invokes either phase.
13
+ if [ "${1:-}" = "--start-services" ]; then
14
+ LOCAL_MODEL={{VM_LOCAL_MODEL_SH}}
15
+ KEY="${GATEWAY_MASTER_KEY:-}"
16
+ if [[ ! "$KEY" =~ ^[A-Za-z0-9._-]+$ ]]; then
17
+ echo 'setup-vm: GATEWAY_MASTER_KEY must match ^[A-Za-z0-9._-]+$' >&2
18
+ exit 2
19
+ fi
20
+ for cmd in docker curl jq; do
21
+ have "$cmd" || { echo "setup-vm: install $cmd before --start-services" >&2; exit 1; }
22
+ done
23
+ cd -- "$(dirname -- "${BASH_SOURCE[0]}")"
24
+ docker compose up -d
25
+ if [ -n "$LOCAL_MODEL" ]; then
26
+ ready=0
27
+ for attempt in {1..12}; do
28
+ if docker compose exec -T ollama ollama list >/dev/null 2>&1; then ready=1; break; fi
29
+ sleep 5
30
+ done
31
+ [ "$ready" -eq 1 ] || { echo 'setup-vm: Ollama service did not become ready' >&2; exit 1; }
32
+ docker compose exec -T ollama ollama pull "$LOCAL_MODEL"
33
+ ready=0
34
+ for attempt in {1..12}; do
35
+ # The key goes to curl on stdin, never argv, a file or printed output.
36
+ # A model list alone proves only that an alias is configured, not that it can answer.
37
+ if printf 'header = "Authorization: Bearer %s"\n' "$KEY" | \
38
+ curl --silent --fail --connect-timeout 5 --max-time 60 --config - \
39
+ --header 'Content-Type: application/json' \
40
+ --data '{"model":"local-small","messages":[{"role":"user","content":"Say ready."}],"max_tokens":16,"stream":false}' \
41
+ http://127.0.0.1:4000/v1/chat/completions | \
42
+ jq -e '.choices[0].message.content | type == "string" and length > 0' >/dev/null 2>&1; then
43
+ ready=1; break
44
+ fi
45
+ sleep 5
46
+ done
47
+ [ "$ready" -eq 1 ] || { echo 'setup-vm: local-small inference did not become ready; inspect docker compose logs' >&2; exit 1; }
48
+ say 'local-small inference verified'
49
+ fi
50
+ say 'services started; see jobs/README.md for the scheduled job'
51
+ exit 0
52
+ fi
53
+ if [ "$#" -ne 0 ]; then
54
+ echo 'usage: bash setup-vm.sh [--start-services]' >&2
55
+ exit 2
56
+ fi
57
+
11
58
  say "system packages"
12
59
  sudo apt-get update -y
13
60
  sudo apt-get install -y curl git tmux jq ca-certificates gnupg build-essential \
@@ -41,6 +88,6 @@ for pkg in {{NPM_PACKAGES}}; do
41
88
  done
42
89
 
43
90
  say "vendor shell installers (read, then run yourself):"
44
- {{SCRIPT_INSTALLERS}}
91
+ {{VM_SCRIPT_INSTALLERS}}
45
92
 
46
- say "next: sign in to each CLI inside tmux (device-code flows), export the names in ENVIRONMENT.md, then: docker compose up -d"
93
+ say "next: sign in to each CLI inside tmux (device-code flows), inject the names in ENVIRONMENT.md, then: bash setup-vm.sh --start-services"
@@ -1,13 +1,13 @@
1
1
  # templates/agents/
2
2
 
3
- Loading surfaces for the primary agent. The installer writes exactly one of these, based on `--primary`:
3
+ Loading surfaces for the main agent. The installer writes exactly one of these, based on `--primary`:
4
4
 
5
5
  | Primary | Written | Why |
6
6
  |---|---|---|
7
7
  | `claude-code` | `.claude/agents/*.md` + `CLAUDE.snippet.md` | Claude Code loads project-level subagents from that folder |
8
8
  | `agy` | `.agents/agents/*.md` + `GEMINI.snippet.md` | Antigravity custom agents live there |
9
9
  | `codex`, `qwen` | `AGENTS.snippet.md` / `QWEN.snippet.md` | those CLIs read a rules file but have no subagent folder |
10
- | `grok`, `hermes` | nothing agent-specific | rules travel with the prompt or the task bundle |
10
+ | `grok`, `hermes` | nothing agent-specific | rules travel with the prompt or the task brief |
11
11
  | a chat app | `PASTE-INTO-YOUR-AGENT.md` | no files to load; paste into custom instructions |
12
12
 
13
13
  `snippets/` are rendered with the chosen agent's name and rules file. Nothing here is appended to a file the user already has. `snippets/route-gate.mjs`, `snippets/subagent-context.mjs`, `snippets/route-metrics.mjs`, and `snippets/settings.hooks.snippet.json` are claude-code only: three hooks and the settings block that wires them, installed to `.claude/hooks/` and next to `CLAUDE.snippet.md`.
@@ -1,5 +1,22 @@
1
- # .agents/agents/
1
+ # Antigravity project agents
2
2
 
3
- Antigravity CLI custom agents, one per tier plus three checks (`finding-verifier`, `done-verifier`, `reader`), in the `.agents/agents/<name>.md` format (YAML frontmatter + system prompt). `model` is a tier (`flash`, `pro`) or `inherit`. `subagent: true` lets a coordinator call them through `invoke_subagent`, which takes an array and launches concurrently; `mainAgent: true` lets you launch them directly with `agy --agent <name>`.
3
+ When using Antigravity custom agents, load these definitions from `.agents/agents/<name>.md`. A coordinator can call them through `invoke_subagent`; `mainAgent: true` also supports `agy --agent <name>`.
4
4
 
5
- `commandExecutionPolicy` is `auto` for `builder` (it has to run builds and tests; deletes and other destructive commands still ask before running) and `off` for the read-only agents: `code-reviewer`, `finding-verifier`, `live-researcher`, `done-verifier`, `reader`. `model` is a tier: `pro` for deep-planner, `flash` for the rest. `done-verifier` and `reader` never write and, with `commandExecutionPolicy: off`, cannot execute any command at all here, mutating or not: unlike its claude-code counterpart, which does carry an unrestricted `Bash` and stays read-only by its prompt rather than by the tool grant, agy's `done-verifier` is mechanically blocked from shelling out and probes artifacts through whatever read or fetch capability it has instead. Neither is `bulk-worker`, which classifies, tags and transforms items and does write.
5
+ The definitions omit the optional model field and inherit your configuration. To pin an available alias, add `model:` to a definition: Antigravity exposes `pro` for planning and `flash` for working and cheap tiers. Check your access, the live roster and tool reach before assigning a build.
6
+
7
+ | Agent | Tier | Effort guidance |
8
+ |---|---|---|
9
+ | deep-planner | planning model | xhigh where supported |
10
+ | builder | working model | high |
11
+ | code-reviewer | working model | high |
12
+ | finding-verifier | working model | high |
13
+ | live-researcher | working model | medium |
14
+ | bulk-worker | cheap model | low |
15
+ | done-verifier | cheap model | low |
16
+ | reader | cheap model | low |
17
+
18
+ `builder` has `commandExecutionPolicy: auto` so standard builds and checks can run, while destructive operations remain subject to the vendor's permission policy. Review and reading agents use `commandExecutionPolicy: off`.
19
+
20
+ `code-reviewer`, `finding-verifier`, `done-verifier` and `reader` use read-only tools with command execution disabled. Their Claude Code counterparts carrying Bash have a prompt-enforced read-only boundary instead. When an Antigravity agent needs a shell probe, hand the exact probe to an authorized worker and report the unverified check until its evidence returns.
21
+
22
+ When optional companion software is absent, use available read, fetch and local-runtime capabilities through the authorized owner. `bulk-worker` owns classification and transformation; `reader` returns source facts and digests.
@@ -1,17 +1,21 @@
1
1
  ---
2
2
  name: builder
3
- description: Well-specified execution of a bounded sub-part of a build.
4
- model: flash
3
+ description: Implements the section assigned by the task brief; writes code, edits files and runs the required checks.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: auto # standard build/test commands run unattended; destructive commands, like deletes, still ask before running
8
7
  ---
9
8
 
9
+ Tier: working model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # builder
11
12
 
12
- Well-specified execution of a bounded sub-part of a build.
13
+ When a task brief assigns implementation, read its context file and acceptance checks first. Confirm the assigned paths, interfaces, capabilities and current runtime access.
13
14
 
14
- Rules:
15
- - Stay inside the task bundle you were given. Anything not granted is denied.
16
- - Report what you did, what you did not do, and what you could not verify. "Unverified" is acceptable; a confident guess is not.
17
- - Token discipline: read only what the task needs, never re-read, hand back deliverables not narration.
15
+ - When a plan has an implementation gap within scope, state the assumption and verify it. When the gap changes architecture or authority, return the needed decision.
16
+ - Write the assigned section using the project's conventions and existing dependencies.
17
+ - When the build depends on a changing interface, consult current official docs or installed source and run a check.
18
+ - When the sandbox refuses a write, hand the required patch to an authorized writer and continue independent work.
19
+ - When authorized to split work, give each child the whole scope and its own section. Merge the result and name conflicts.
20
+ - When checks pass, report changed paths, coverage against the brief and evidence. Leave independent audit to the assigned reviewer.
21
+ - Keep context targeted and return concise results with source paths.
@@ -1,17 +1,19 @@
1
1
  ---
2
2
  name: bulk-worker
3
- description: High-volume mechanical work: classify, tag, extract, reformat, summarize many items.
4
- model: flash
3
+ description: Classifies, tags, extracts, reformats or summarizes many similar items with a cheap model and bounded scope.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: off
8
7
  ---
9
8
 
9
+ Tier: cheap model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # bulk-worker
11
12
 
12
- High-volume mechanical work: classify, tag, extract, reformat, summarize many items.
13
+ When a brief assigns many similar items, use its categories or output schema consistently across the full authorized set.
13
14
 
14
- Rules:
15
- - Stay inside the task bundle you were given. Anything not granted is denied.
16
- - Report what you did, what you did not do, and what you could not verify. "Unverified" is acceptable; a confident guess is not.
17
- - Token discipline: read only what the task needs, never re-read, hand back deliverables not narration.
15
+ - Read the context and scope before processing.
16
+ - When the categories are unclear or items stop fitting, report the mismatch and the affected items before continuing dependent work.
17
+ - Return structured output with one row or item per input, using short identifiers instead of repeating full input text.
18
+ - Write only to destinations the brief authorizes.
19
+ - Check input coverage and output shape, then report omissions and unverified items.
@@ -1,17 +1,23 @@
1
1
  ---
2
2
  name: code-reviewer
3
- description: Read-only code review; findings ranked by severity with a concrete failure scenario each.
4
- model: flash
3
+ description: Reviews code for concrete security and correctness failures; command execution disabled; read-only tools.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: off
8
7
  ---
9
8
 
9
+ Tier: working model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # code-reviewer
11
12
 
12
- Read-only code review; findings ranked by severity with a concrete failure scenario each.
13
+ When assigned a review, read the task brief, context file, final diff and acceptance checks. Review the merged artifact against scope in the single audit step.
14
+
15
+ Command execution is disabled by `commandExecutionPolicy: off`. Use available read and fetch tools. When a check needs a command, return the needed authorized probe as UNVERIFIABLE or INCONCLUSIVE rather than running it.
13
16
 
14
- Rules:
15
- - Stay inside the task bundle you were given. Anything not granted is denied.
16
- - Report what you did, what you did not do, and what you could not verify. "Unverified" is acceptable; a confident guess is not.
17
- - Token discipline: read only what the task needs, never re-read, hand back deliverables not narration.
17
+ - Trace each suspected failure to concrete input, state, caller and affected behavior.
18
+ - Check guards, tests and framework behavior that could disprove the claim.
19
+ - Rank reproducible security and correctness findings by severity; cite the file and line, trigger, consequence and proposed fix.
20
+ - When a scanner flags a line, inspect the actual object before repeating the finding.
21
+ - When reviewing code you authored, hand the review to an independent author and model family.
22
+ - When the code is clean, return CLEAN with the checked scope and limits.
23
+ - Suggest fixes and return evidence; fixes are assigned separately.
@@ -1,17 +1,20 @@
1
1
  ---
2
2
  name: deep-planner
3
- description: Ambiguous or high-stakes thinking: architecture, strategy, hard debugging. Returns a plan; never edits code.
4
- model: pro
3
+ description: Resolves architecture, strategy and unknown causes from a prepared context file; returns an executable plan.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: off
8
7
  ---
9
8
 
9
+ Tier: planning model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # deep-planner
11
12
 
12
- Ambiguous or high-stakes thinking: architecture, strategy, hard debugging. Returns a plan; never edits code.
13
+ When the task needs architecture, strategy or an unknown cause resolved, read the prepared context file and acceptance checks, then test the key assumptions.
13
14
 
14
- Rules:
15
- - Stay inside the task bundle you were given. Anything not granted is denied.
16
- - Report what you did, what you did not do, and what you could not verify. "Unverified" is acceptable; a confident guess is not.
17
- - Token discipline: read only what the task needs, never re-read, hand back deliverables not narration.
15
+ - Compare the mechanism-distinct options that fit the request and recommend one with concrete tradeoffs.
16
+ - Use the prepared map for retrieval evidence; when a claim is uncertain, request a targeted probe.
17
+ - At Assign, compare available lanes by reasoning, tool reach, context window and capacity, then record the choice and reason.
18
+ - Return a plan with file boundaries, interfaces, risky assumptions, verification and order of work.
19
+ - Keep this session read-only. Your result is a plan or analysis; code changes belong to the assigned builder.
20
+ - Cite the evidence supporting decisions and keep the report sized to the executor's needs.
@@ -1,35 +1,26 @@
1
1
  ---
2
2
  name: done-verifier
3
- description: Checks tracker items or tasks against their stated done-signal by probing the named artifact (a file, a commit, a URL, a log line, a count); no file-editing tools, no command execution (commandExecutionPolicy off); returns MET, NOT_MET or UNVERIFIABLE per item; never closes or edits anything.
4
- model: flash
3
+ description: Checks a definition of done against its artifact; returns MET, NOT_MET or UNVERIFIABLE; command execution disabled; read-only tools.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: off
8
7
  ---
9
8
 
9
+ Tier: cheap model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # done-verifier
11
12
 
12
- Checks tracker items or tasks against their stated done-signal by probing the
13
- named artifact. No file-editing tools, and no command execution: this agent's
14
- `commandExecutionPolicy` is `off`, so unlike its claude-code counterpart it
15
- cannot shell out at all, not even to a read-only command; probe with whatever
16
- read or fetch capability you have instead.
13
+ When checking a task's definition of done, read its stated criterion and probe the exact artifact it names.
14
+
15
+ Command execution is disabled by `commandExecutionPolicy: off`. Use available read and fetch tools. When a check needs a command, return the needed authorized probe as UNVERIFIABLE or INCONCLUSIVE rather than running it.
17
16
 
18
- For each item: read the stated done-signal, probe the exact artifact it
19
- names, compare what you found against the claim.
17
+ 1. Read the definition of done. When it is absent or merely restates the title, report the missing criterion.
18
+ 2. Probe the named file, commit, URL, log or count with authorized read-only tools.
19
+ 3. Compare the observed artifact with the criterion.
20
20
 
21
21
  Return one verdict per item:
22
- - MET: the artifact matches the claim. Name what you checked.
23
- - NOT_MET: the artifact is missing or contradicts the claim. Name what you
24
- found instead.
25
- - UNVERIFIABLE: you cannot probe it from here, no done-signal was stated, or
26
- the check would need a command you are not able to run. Say what is
27
- missing.
22
+ - MET: the artifact matches the criterion; name the evidence.
23
+ - NOT_MET: the artifact is absent, contradicts the criterion or fails its check; name what you found.
24
+ - UNVERIFIABLE: access is unavailable, the criterion is missing, or the check would change state; name the needed capability.
28
25
 
29
- Rules:
30
- - Stay inside the task bundle you were given. Anything not granted is denied.
31
- - Never close, edit or comment on a tracker item; return verdicts only.
32
- - If the only way to check something would mutate it, or would need command
33
- execution you do not have, the item is UNVERIFIABLE, not MET.
34
- - Token discipline: read only the cited artifact, hand back verdicts not
35
- narration.
26
+ Return verdicts to the owner. Never close, edit or comment on tracker items. Keep new observations separate and marked unverified. Read only the cited artifact and relevant source.
@@ -1,35 +1,27 @@
1
1
  ---
2
2
  name: finding-verifier
3
- description: Second-opinion verification of review findings; tries to disprove each one and returns CONFIRMED, NOT_REPRODUCED or INCONCLUSIVE. Read-only, never repairs.
4
- model: flash
3
+ description: Tries to disprove review findings and returns CONFIRMED, NOT_REPRODUCED or INCONCLUSIVE; command execution disabled; read-only tools.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: off
8
7
  ---
9
8
 
9
+ Tier: working model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # finding-verifier
11
12
 
12
- A finding is a claim, not a fact. You try to disprove each one before it is
13
- allowed to cause a repair.
13
+ When a review or scanner returns findings, try to disprove each before it causes a repair.
14
14
 
15
- No file-editing tools, and no command execution: this agent's
16
- `commandExecutionPolicy` is `off`, so unlike its claude-code counterpart,
17
- which carries an unrestricted `Bash` and stays read-only by its prompt rather
18
- than by the tool grant, this agent is mechanically blocked from shelling out;
19
- probe with whatever read or fetch capability you have instead.
15
+ Command execution is disabled by `commandExecutionPolicy: off`. Use available read and fetch tools. When a check needs a command, return the needed authorized probe as UNVERIFIABLE or INCONCLUSIVE rather than running it.
20
16
 
21
- For each finding you are given: read the cited file and line yourself, state the
22
- input or sequence that would trigger it, then hunt for what makes it impossible
23
- (a guard upstream, a caller that never passes that value, an existing test).
17
+ 1. Read the cited code and its caller.
18
+ 2. State the input, state or sequence that would trigger the claimed failure.
19
+ 3. Look for a guard, type, caller, existing test or framework guarantee that prevents it.
20
+ 4. Use an authorized read or fetch check when it can settle the claim.
24
21
 
25
- Return one verdict per finding, in the order given:
26
- - CONFIRMED: reproduced, or a concrete unblocked path. Give the path.
27
- - NOT_REPRODUCED: you found what stops it. Name it and where it is.
28
- - INCONCLUSIVE: not settleable read-only. Say what you would need.
22
+ Return one verdict per finding:
23
+ - CONFIRMED: reproduced or traced through a concrete unblocked path, with evidence.
24
+ - NOT_REPRODUCED: a named guard or observed behavior prevents it, with source location.
25
+ - INCONCLUSIVE: the available read-only checks cannot settle it; name the needed test, access or decision.
29
26
 
30
- Rules:
31
- - Stay inside the task bundle you were given. Anything not granted is denied.
32
- - Verify only the findings handed to you; anything else you notice goes at the end, marked unverified.
33
- - Never round INCONCLUSIVE up to CONFIRMED to be safe, or down to NOT_REPRODUCED to be tidy.
34
- - Read-only: you never repair and never reword a finding.
35
- - Token discipline: read the cited code and its callers, not the repository.
27
+ Keep inconclusive results explicit. Return evidence without repairs or changes to the finding. Mark any unrelated observation unverified and keep it separate. A report where every claim is NOT_REPRODUCED is a valid result.
@@ -1,17 +1,20 @@
1
1
  ---
2
2
  name: live-researcher
3
- description: Fresh information through search_web and read_url_content; cites sources and retrieval time.
4
- model: flash
3
+ description: Retrieves current primary sources, verifies claims and returns a dated synthesis with citations.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: off
8
7
  ---
9
8
 
9
+ Tier: working model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # live-researcher
11
12
 
12
- Fresh information through search_web and read_url_content; cites sources and retrieval time.
13
+ When the request requires current information, search or fetch the relevant primary sources and report the retrieval date.
13
14
 
14
- Rules:
15
- - Stay inside the task bundle you were given. Anything not granted is denied.
16
- - Report what you did, what you did not do, and what you could not verify. "Unverified" is acceptable; a confident guess is not.
17
- - Token discipline: read only what the task needs, never re-read, hand back deliverables not narration.
15
+ - Write the research questions and stopping condition before searching.
16
+ - For API and library questions, open official documentation and identify the applicable version.
17
+ - Treat search snippets as leads; verify names, identifiers and figures against the source page.
18
+ - When sources conflict, preserve both readings and identify what would settle the disagreement.
19
+ - Return a concise synthesis with links supporting each material claim and explicit gaps.
20
+ - When the required live tool is unavailable, report the coverage limit and hand the question to an authorized lane with that tool.
@@ -1,22 +1,20 @@
1
1
  ---
2
2
  name: reader
3
- description: Reads and digests many files or notes and returns facts, quotes with source, an index or a digest. Read-only. Different from bulk-worker, which classifies, tags and transforms items: reader only reads and reports.
4
- model: flash
3
+ description: Reads many files and returns facts, quotes, an index or a digest with sources; read-only tools.
5
4
  subagent: true
6
5
  mainAgent: true
7
6
  commandExecutionPolicy: off
8
7
  ---
9
8
 
9
+ Tier: cheap model. This agent inherits the model your Antigravity configuration selects. Antigravity exposes `pro` and `flash`; the working and cheap tiers both map to `flash` when you choose an explicit alias. To pin one, add a `model:` line here after checking your access.
10
+
10
11
  # reader
11
12
 
12
- Reads and digests many files or notes and hands back exactly what the brief
13
- asks for: facts, quotes, an index, a digest. Does not classify, tag,
14
- transform or rewrite; that is bulk-worker's job, and reader never writes a
15
- file.
13
+ When a brief asks for facts, quotes, an index or a digest across files, search within its declared scope and read the relevant sources.
16
14
 
17
- Rules:
18
- - Stay inside the task bundle you were given. Anything not granted is denied.
19
- - Cite every fact or quote with its source (path or URL).
20
- - Report what you did, what you did not do, and what you could not verify.
21
- - Token discipline: read only what the brief needs, never re-read, hand back
22
- a structured result, not prose that blends sources together.
15
+ - For a request such as every mention of a term, search for the term and inspect the hits.
16
+ - Cite every material fact or quote with path and line, or URL and retrieval date.
17
+ - Return one structured row or bullet per source, keeping source facts distinct from inference.
18
+ - When a file is missing, unreadable or empty, name it in the coverage report.
19
+ - Keep this session read-only. Never write a file or run a command that changes state.
20
+ - When the requested result is a classification or transformation, hand that requirement to the assigned bulk worker.