model-orchestrator 1.0.2 → 1.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +28 -1
- package/README.md +1 -1
- package/docs/README.md +1 -0
- package/docs/catalog-advisories.md +116 -0
- package/docs/catalog-advisory-exceptions.json +1019 -0
- package/docs/catalog.md +1 -1
- package/docs/part-2-intermediate.md +1 -1
- package/llms.txt +1 -0
- package/package.json +1 -1
- package/src/README.md +1 -1
- package/src/catalog.js +6 -3
- package/templates/advanced/vm/ENVIRONMENT.md +9 -3
- package/templates/advanced/vm/jobs/README.md +22 -3
- package/templates/advanced/vm/jobs/weekly-audit.service +2 -1
- package/templates/advanced/vm/jobs/weekly-audit.sh +29 -7
- package/templates/agents/snippets/route-metrics.mjs +43 -0
package/docs/catalog.md
CHANGED
|
@@ -207,7 +207,7 @@ Generated from `src/catalog.js`. Do not hand-edit; `npm run gen:catalog` rewrite
|
|
|
207
207
|
- **What it is:** A local model runtime; work sent here stays on the machine
|
|
208
208
|
- **Install:** https://ollama.com/download (or `brew install ollama`)
|
|
209
209
|
- **Sign in:** none
|
|
210
|
-
- **Built against:** 0.
|
|
210
|
+
- **Built against:** 0.34.4
|
|
211
211
|
|
|
212
212
|
**Capability facts.** An unverified value needs a current capability check before use. Role assignments come from the selected stack, using these facts.
|
|
213
213
|
|
|
@@ -50,7 +50,7 @@ Optional companions can help: codecalc for execution and calculations, obsidian-
|
|
|
50
50
|
|
|
51
51
|
## Measure your own routing
|
|
52
52
|
|
|
53
|
-
`aunx route-metrics --summary` reads your local Claude Code routing log. It reports where work went, route-marker coverage and
|
|
53
|
+
`aunx route-metrics --summary` reads your local Claude Code routing log. It reports where work went, route-marker coverage, subagent durations, and whether the route each reply named matches the delegation that followed. Your own measurements are the basis for changing assignments and checking whether the rules are being followed.
|
|
54
54
|
|
|
55
55
|
## What the installer gives you at this level
|
|
56
56
|
|
package/llms.txt
CHANGED
|
@@ -29,6 +29,7 @@ Pick a model proxy (LiteLLM, Portkey, OpenRouter, claude-code-router) for per-re
|
|
|
29
29
|
- [Claude Code plugin](https://github.com/aunysillyme/model-orchestrator/blob/main/plugin/README.md): installing read-only routing hooks and subagents through the plugin marketplace
|
|
30
30
|
- [Proof](https://github.com/aunysillyme/model-orchestrator/blob/main/proof/README.md): dated measurements, methods, sample sizes, expiry and reproduction scripts
|
|
31
31
|
- [Security review history](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/security-review-history.md): completed review rounds, incident summaries and regression evidence
|
|
32
|
+
- [Catalog advisory checks](https://github.com/aunysillyme/model-orchestrator/blob/main/docs/catalog-advisories.md): package and container inventory, coverage, report statuses, exceptions and manual verification
|
|
32
33
|
- [Changelog](https://github.com/aunysillyme/model-orchestrator/blob/main/CHANGELOG.md): release changes and upgrade notes
|
|
33
34
|
- [Agent instructions](https://github.com/aunysillyme/model-orchestrator/blob/main/AGENTS.md): headless setup and contributor checks
|
|
34
35
|
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "model-orchestrator",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.4",
|
|
4
4
|
"description": "Model router for AI coding agents: installs routing rules, 8 subagents, hooks and a CLI runner so your AI picks model and effort per task and saves tokens",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
package/src/README.md
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
| File | Job |
|
|
4
4
|
|---|---|
|
|
5
5
|
| `bounded-file.js` | shared regular-file reader for manifests and check configuration: no-follow/nonblocking open, identity checks and a fixed byte cap even if a file grows. Unsafe files are refused; callers decide how to handle missing or malformed data. |
|
|
6
|
-
| `catalog.js` | the single list of levels and AIs. Add an AI here and the prompts, docs tables, delegation matrix, gateway config and installer all pick it up. Nothing else lists AIs. |
|
|
6
|
+
| `catalog.js` | the single list of levels and AIs. Add an AI here and the prompts, docs tables, delegation matrix, gateway config and installer all pick it up. Nothing else lists AIs. Package pins, companion advisory metadata and image references also supply the repository's advisory inventory. |
|
|
7
7
|
| `roles.js` | pure role assignment from selected catalog capability facts, billing and selection order. Renders the stack table and manifest roles, and infers the main agent from its supported surfaces. Unknown facts remain unverified; review requires a known different model family and private work requires local execution. |
|
|
8
8
|
| `aunx.js` | command dispatch for briefs, context, checks, routing and runner calls. Route suggestions read manifest roles through a capped regular-file JSON reader; symlinks and malformed files are ignored. Route lookup executes no project code. A project's runner requires explicit `--dir`. |
|
|
9
9
|
| `detect.js` | PATH lookup for a binary, plus the few places vendor installers drop binaries without touching PATH. No shell-outs. |
|
package/src/catalog.js
CHANGED
|
@@ -300,7 +300,7 @@ export const AIS = [
|
|
|
300
300
|
summary: 'A local model runtime; work sent here stays on the machine',
|
|
301
301
|
minLevel: 2,
|
|
302
302
|
install: { url: 'https://ollama.com/download', brew: 'ollama' },
|
|
303
|
-
builtAgainst: '0.
|
|
303
|
+
builtAgainst: '0.34.4',
|
|
304
304
|
auth: 'none',
|
|
305
305
|
rulesFile: null,
|
|
306
306
|
},
|
|
@@ -402,6 +402,7 @@ export const TOOLS = [
|
|
|
402
402
|
role: 'exact arithmetic, code execution in 31 languages, SMT logic checks, complexity and equivalence proofs; offline, no key, no telemetry',
|
|
403
403
|
get install() { return `uvx 'codecalc[full]==${this.pin}' setup --write`; },
|
|
404
404
|
pin: '0.5.0',
|
|
405
|
+
advisory: { ecosystem: 'PyPI', package: 'codecalc', extras: ['full'] },
|
|
405
406
|
mcpSnippets: { 'claude-code': 'mcp/mcpServers.json', codex: 'mcp/codex.config.toml', agy: 'mcp/agy.mcp_config.json', qwen: 'mcp/mcpServers.json' },
|
|
406
407
|
requires: 'uv (https://docs.astral.sh/uv/) and Python 3.10+',
|
|
407
408
|
autoClients: ['Claude Code', 'Claude Desktop', 'Cursor', 'VS Code', 'Zed'],
|
|
@@ -415,6 +416,7 @@ export const TOOLS = [
|
|
|
415
416
|
role: 'durable memory and record for your agents: hybrid retrieval (BM25 + dense + link graph), backlinks, compare-and-swap writes with a confirmation gate, folder ACLs, a poison scan on inferred writes; 163 tools, local by default',
|
|
416
417
|
get install() { return `npm install -g obsidian-tc@${this.pin} && obsidian-tc /path/to/your/vault`; },
|
|
417
418
|
pin: '1.26.0',
|
|
419
|
+
advisory: { ecosystem: 'npm', package: 'obsidian-tc' },
|
|
418
420
|
mcpSnippets: { 'claude-code': 'mcp/obsidian-tc.mcpServers.json', codex: 'mcp/obsidian-tc.codex.config.toml', agy: 'mcp/obsidian-tc.agy.mcp_config.json', qwen: 'mcp/obsidian-tc.mcpServers.json' },
|
|
419
421
|
requires: 'an Obsidian vault folder (the Obsidian app itself is only needed for live plugin bridges); Node 24+ or Bun 1.1+ (stricter than this installer); Ollama with `nomic-embed-text` for local embeddings, or a cloud embeddings key; the Local REST API plugin only for bridge tools',
|
|
420
422
|
autoClients: ['Cursor', 'VS Code'],
|
|
@@ -428,6 +430,7 @@ export const TOOLS = [
|
|
|
428
430
|
role: 'up-to-date, version-specific documentation and code examples for libraries, SDKs, APIs and CLIs, pulled into the prompt; tells the agent what the code is SUPPOSED to do. Paired with codecalc, which runs the code and proves what it actually does: docs never stand as proof, and where they disagree the run wins',
|
|
429
431
|
get install() { return `npx -y @upstash/context7-mcp@${this.pin}`; },
|
|
430
432
|
pin: '4.1.1',
|
|
433
|
+
advisory: { ecosystem: 'npm', package: '@upstash/context7-mcp' },
|
|
431
434
|
mcpSnippets: { 'claude-code': 'mcp/context7.claude-code.mcp.json', codex: 'mcp/context7.codex.config.toml', agy: 'mcp/context7.agy.mcp_config.json', qwen: 'mcp/context7.qwen.settings.json' },
|
|
432
435
|
requires: 'Node.js 18+ for the local server or the ctx7 CLI; a free CONTEXT7_API_KEY is optional, for higher rate limits (it works anonymously at the base rate)',
|
|
433
436
|
autoClients: [], // The pinned MCP server does not register itself; merge its snippets.
|
|
@@ -454,8 +457,8 @@ export const providerById = Object.fromEntries(PROVIDERS.map((p) => [p.id, p]));
|
|
|
454
457
|
// Container image pins for the level 3 templates. Bump deliberately; a
|
|
455
458
|
// reviewed box should not change underneath the user on a restart.
|
|
456
459
|
export const IMAGES = {
|
|
457
|
-
litellm: 'ghcr.io/berriai/litellm:v1.
|
|
458
|
-
ollama: 'ollama/ollama:0.
|
|
460
|
+
litellm: 'ghcr.io/berriai/litellm:v1.100.3',
|
|
461
|
+
ollama: 'ollama/ollama:0.34.4'
|
|
459
462
|
};
|
|
460
463
|
|
|
461
464
|
export const byId = Object.fromEntries(AIS.map((a) => [a.id, a]));
|
|
@@ -12,11 +12,17 @@ These are variable **names**. The values live in a secrets manager and are injec
|
|
|
12
12
|
|
|
13
13
|
## Gateway and scheduled audit environments
|
|
14
14
|
|
|
15
|
-
Inject the provider names above only into the environment used to launch the gateway with Compose. The weekly audit service instead reads `~/.config/ai-orchestrator/weekly-audit.env`, outside the installation and mode 600, containing
|
|
15
|
+
Inject the provider names above only into the environment used to launch the gateway with Compose. The weekly audit service instead reads `~/.config/ai-orchestrator/weekly-audit.env`, outside the installation and mode 600, containing `GATEWAY_MASTER_KEY` and optional documented runtime/location settings. Do not point that service at the gateway's provider-key file. Existing installations must update and reload their copied systemd unit when adopting this template.
|
|
16
16
|
|
|
17
|
-
The audit script
|
|
17
|
+
The audit script builds an explicit allowed environment with shell builtins before its first external command. It supplies the gateway header to curl through stdin, without a credential temp file or an argv value. Newlines anywhere in the key are rejected before collection. All provider keys, unrelated exported names, token variables and exported shell functions are absent from child environments.
|
|
18
18
|
|
|
19
|
-
|
|
19
|
+
The allowed runtime names are `HOME`, `PATH`, `USER`, `LOGNAME`, `SHELL`; `LANG`, `LANGUAGE`, `TZ`; `LC_ALL`, `LC_CTYPE`, `LC_COLLATE`, `LC_MESSAGES`, `LC_MONETARY`, `LC_NUMERIC`, `LC_TIME`, `LC_ADDRESS`, `LC_IDENTIFICATION`, `LC_MEASUREMENT`, `LC_NAME`, `LC_PAPER`, `LC_TELEPHONE`; `TMPDIR`, `TMP`, `TEMP`; `XDG_CONFIG_HOME`, `XDG_DATA_HOME`, `XDG_STATE_HOME`, `XDG_CACHE_HOME`, `XDG_RUNTIME_DIR`, `DBUS_SESSION_BUS_ADDRESS`; and the Windows shell runtime names `SystemRoot`, `SYSTEMROOT`, `WINDIR`. User-service access and stored sign-ins use those same locations.
|
|
20
|
+
|
|
21
|
+
Only the selected report worker additionally receives its location override: `CODEX_HOME` for `codex`, `HERMES_HOME` for `hermes`. These names are removed from collection/version probes. `agy`, `grok` and `qwen` use stored sign-ins in the common user/configuration locations, with no additional exported authentication name. Configure the selected vendor's sign-in in the systemd user's account, such as `hermes auth add <provider>` for Hermes. API-only setups that depend on exported provider keys need vendor-supported stored authentication before this job can run. Claude Code is not a cli-run worker in this catalog, so its session-token export is removed as well.
|
|
22
|
+
|
|
23
|
+
`PROBE_SECS` and `RUNNER_SECS` remain shell-only deadline overrides. Set documented runtime and location overrides in the service's `Environment=` settings or audit-only environment file. The key and any selected worker settings stay out of command arguments and logs. There is no arbitrary variable pass-through option.
|
|
24
|
+
|
|
25
|
+
When migrating, use `--update-docs` for unchanged managed documents, review preserved edited files, apply the new script and docs, update the copied service, then reload systemd. Reconfigure any sign-in that depended on an unrelated export and manually start the service. [The jobs README](jobs/README.md) owns the invocation chain, exact environment contract, migration and verification steps. The environment boundary preserves access to files under HOME/XDG; it does not isolate those files or undo Bash startup files. Real vendor sign-in and systemd behavior remain unverified until the manual run.
|
|
20
26
|
|
|
21
27
|
## Rules
|
|
22
28
|
|
|
@@ -21,9 +21,28 @@ systemctl --user list-timers # it should be listed with a next-run time
|
|
|
21
21
|
loginctl enable-linger "$USER" # so user timers run without a login session
|
|
22
22
|
```
|
|
23
23
|
|
|
24
|
-
The service reads
|
|
24
|
+
The service reads `GATEWAY_MASTER_KEY` and optional documented runtime/location settings from `~/.config/ai-orchestrator/weekly-audit.env`, outside this folder with mode 600. Provision that audit-only file from your secrets manager and point `EnvironmentFile=` at it before installing. Keep the gateway's provider-key environment separate. Existing installs must replace or edit their copied service, then run `systemctl --user daemon-reload`; updating the source template alone does not update the installed unit. The key must be a single token matching `^[A-Za-z0-9._-]+$`; any embedded or trailing newline is refused.
|
|
25
25
|
|
|
26
|
-
Sign the selected vendor CLI in under the same user before enabling the timer.
|
|
26
|
+
Sign the selected vendor CLI in under the same user before enabling the timer. Before its first external command, the script uses shell builtins to remove every export except the runtime settings below. It also removes exported shell functions and disables inherited tracing and automatic export. Collection commands and version probes receive this same allowed environment.
|
|
27
|
+
|
|
28
|
+
| Purpose | Allowed names |
|
|
29
|
+
|---|---|
|
|
30
|
+
| User and executable lookup | `HOME`, `PATH`, `USER`, `LOGNAME`, `SHELL` |
|
|
31
|
+
| Locale and time | `LANG`, `LANGUAGE`, `TZ`, `LC_ALL`, `LC_CTYPE`, `LC_COLLATE`, `LC_MESSAGES`, `LC_MONETARY`, `LC_NUMERIC`, `LC_TIME`, `LC_ADDRESS`, `LC_IDENTIFICATION`, `LC_MEASUREMENT`, `LC_NAME`, `LC_PAPER`, `LC_TELEPHONE` |
|
|
32
|
+
| Temporary directories | `TMPDIR`, `TMP`, `TEMP` |
|
|
33
|
+
| Stored configuration and sign-ins | `XDG_CONFIG_HOME`, `XDG_DATA_HOME`, `XDG_STATE_HOME`, `XDG_CACHE_HOME` |
|
|
34
|
+
| User service and keyring session | `XDG_RUNTIME_DIR`, `DBUS_SESSION_BUS_ADDRESS` |
|
|
35
|
+
| Windows shell runtime compatibility | `SystemRoot`, `SYSTEMROOT`, `WINDIR` |
|
|
36
|
+
|
|
37
|
+
The report worker additionally receives `CODEX_HOME` when the selected worker is `codex`, or `HERMES_HOME` when it is `hermes`, if that name was set. These location overrides are absent from collection and version probes. The `agy`, `grok` and `qwen` workers use their stored sign-ins under the common user/configuration directories. No token or provider-key variable is allowed for any current worker; `CLAUDE_CODE_OAUTH_TOKEN` is also removed because Claude Code is not a supported cli-run worker in this catalog. Configure authentication with the selected vendor's sign-in flow, such as `codex login`, `grok login`, `hermes auth add <provider>`, `agy`, or Qwen's `/auth`.
|
|
38
|
+
|
|
39
|
+
`GATEWAY_MASTER_KEY` stays in a private shell variable until the stdin probe finishes and is then removed. `PROBE_SECS` and `RUNNER_SECS` override the shell's deadlines without becoming child exports. Configure the allowed names and worker location overrides through the service's `Environment=` settings or its audit-only environment file. Keep that file limited to the gateway key and the documented runtime/location settings; keep provider credentials in the gateway launch environment. The script provides no arbitrary extra-variable override.
|
|
40
|
+
|
|
41
|
+
### Migrate an existing job
|
|
42
|
+
|
|
43
|
+
Preview your existing selection and paths with `--update-docs --dry-run`, then apply the update. Unchanged managed runtime files upgrade automatically; `--update-docs` refreshes unchanged documents. Review any edited files the installer preserves and merge the environment change into those copies, or use `--upgrade-runtime` when you intend to replace edited runtime files. Apply the new script and both environment documents to your installation, then replace or edit the copied service and run `systemctl --user daemon-reload`. Move sign-in setups that depend on other exported names to the vendor's stored sign-in flow under the service user. Reapply only the documented runtime/location overrides, confirm the selected CLI is on the service's explicit PATH, then perform the manual service check below. Changing `AUDIT_LANE` also requires matching that worker's runner configuration and flags.
|
|
44
|
+
|
|
45
|
+
This boundary limits child environment inheritance. Files under `HOME` and XDG directories remain readable, including stored sign-ins, and Bash startup files run before the script can filter exports. Live vendor authentication must be verified in the service user's account.
|
|
27
46
|
|
|
28
47
|
## Invocation, dependencies, reads and writes
|
|
29
48
|
|
|
@@ -59,7 +78,7 @@ For a manual check, start `systemctl --user start weekly-audit.service`, then ru
|
|
|
59
78
|
- **Previous report preserved:** output goes to a temp file and is renamed over `audit-<date>.md` only on a clean, non-empty run. A failed run leaves `failed-audit-<stamp>-rc<N>.md` beside it and the last good report untouched.
|
|
60
79
|
- **Boundary:** the lane runs with the strongest restriction it offers ({{AUDIT_LANE_BOUNDARY_NOTE}}). The brief's denied-actions list is an instruction, not an enforcement, for lanes without a sandbox flag.
|
|
61
80
|
- **Honest unknowns:** a probe that times out writes an `UNVERIFIED` line, which the brief tells the lane to treat as unknown, never clean.
|
|
62
|
-
- **Credential separation:** the gateway bearer is never written to a temp file or passed on argv
|
|
81
|
+
- **Credential separation:** the gateway bearer is never written to a temp file or passed on argv. Only the named runtime environment reaches probes, and only the selected worker receives its documented location override. Stored vendor sign-ins remain available.
|
|
63
82
|
|
|
64
83
|
A timer that has never been seen to fire is not known to work. Run `systemctl --user start weekly-audit.service` once by hand and read the journal before trusting the schedule.
|
|
65
84
|
|
|
@@ -8,7 +8,8 @@ WorkingDirectory={{INSTALL_DIR_SYSTEMD}}
|
|
|
8
8
|
# Add absolute Node and vendor CLI directories if they live outside these defaults.
|
|
9
9
|
# systemd does not expand shell variables in Environment=.
|
|
10
10
|
Environment="PATH=/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin"
|
|
11
|
-
# Audit-only file OUTSIDE this repo, mode 600: GATEWAY_MASTER_KEY
|
|
11
|
+
# Audit-only file OUTSIDE this repo, mode 600: GATEWAY_MASTER_KEY and optional
|
|
12
|
+
# runtime/location settings from jobs/README.md. Other exports are removed.
|
|
12
13
|
# Provider credentials stay in the separate gateway/Compose environment.
|
|
13
14
|
# Vendor CLIs use this user's stored sign-in state. Edit the path if needed.
|
|
14
15
|
EnvironmentFile=%h/.config/ai-orchestrator/weekly-audit.env
|
|
@@ -8,18 +8,32 @@
|
|
|
8
8
|
# renamed into place only on a clean exit; failed output is kept beside it for diagnosis
|
|
9
9
|
# - the lane runs with the strongest boundary it offers ({{AUDIT_LANE_BOUNDARY_NOTE}})
|
|
10
10
|
# - any nonzero rc from cli-run (10 to 18) means no report was produced; the timer's journal shows it
|
|
11
|
+
# Disable inherited tracing and automatic export before handling credentials.
|
|
12
|
+
set +a +x +v
|
|
11
13
|
set -uo pipefail
|
|
12
14
|
INSTALL_DIR={{INSTALL_DIR_SH}}
|
|
13
15
|
AUDIT_LANE="{{AUDIT_LANE}}"
|
|
14
16
|
AUDIT_LANE_FLAGS="{{AUDIT_LANE_FLAGS}}"
|
|
15
17
|
PROBE_SECS="${PROBE_SECS:-10}" # per collection probe
|
|
16
18
|
RUNNER_SECS="${RUNNER_SECS:-600}" # the model call; TimeoutStartSec in the unit covers the whole job
|
|
17
|
-
#
|
|
18
|
-
#
|
|
19
|
-
#
|
|
20
|
-
KEY="${GATEWAY_MASTER_KEY:-}"
|
|
19
|
+
# Builtins only until the exported environment is reduced to these named runtime
|
|
20
|
+
# settings. De-exporting also handles shell-owned readonly variables. Keep home
|
|
21
|
+
# overrides private until the selected worker runs; stored sign-ins stay on disk.
|
|
21
22
|
export -n KEY
|
|
22
|
-
|
|
23
|
+
KEY="${GATEWAY_MASTER_KEY:-}"
|
|
24
|
+
while IFS= read -r name; do
|
|
25
|
+
case "$name" in
|
|
26
|
+
HOME|PATH|USER|LOGNAME|SHELL|LANG|LANGUAGE|TZ|TMPDIR|TMP|TEMP|\
|
|
27
|
+
LC_ALL|LC_CTYPE|LC_COLLATE|LC_MESSAGES|LC_MONETARY|LC_NUMERIC|LC_TIME|\
|
|
28
|
+
LC_ADDRESS|LC_IDENTIFICATION|LC_MEASUREMENT|LC_NAME|LC_PAPER|LC_TELEPHONE|\
|
|
29
|
+
XDG_CONFIG_HOME|XDG_DATA_HOME|XDG_STATE_HOME|XDG_CACHE_HOME|\
|
|
30
|
+
XDG_RUNTIME_DIR|DBUS_SESSION_BUS_ADDRESS|SystemRoot|SYSTEMROOT|WINDIR) ;;
|
|
31
|
+
*) export -n "$name" ;;
|
|
32
|
+
esac
|
|
33
|
+
done < <(compgen -e)
|
|
34
|
+
# Exported shell functions are also outside the child environment contract.
|
|
35
|
+
while IFS= read -r name; do export -nf "$name"; done < <(compgen -A function)
|
|
36
|
+
unset name GATEWAY_MASTER_KEY LITELLM_MASTER_KEY ANTHROPIC_API_KEY OPENAI_API_KEY GEMINI_API_KEY XAI_API_KEY OPENROUTER_API_KEY
|
|
23
37
|
|
|
24
38
|
# A shell pattern checks the whole value, including embedded/trailing newlines.
|
|
25
39
|
# Line-oriented grep accepts a valid line even when another line is malformed.
|
|
@@ -118,8 +132,16 @@ BRIEF="reports/audit-brief-$STAMP.md"
|
|
|
118
132
|
# Write to a temp file; the dated report is replaced only by a clean, non-empty run.
|
|
119
133
|
FINAL="reports/audit-$DATE.md"
|
|
120
134
|
TMP="$(mktemp "reports/.audit-$STAMP-XXXXXX")"
|
|
121
|
-
|
|
122
|
-
|
|
135
|
+
(
|
|
136
|
+
# These are configuration locations, not provider keys. All current scheduled
|
|
137
|
+
# lanes use stored sign-ins; no token variable is part of this contract.
|
|
138
|
+
case "$AUDIT_LANE" in
|
|
139
|
+
codex) if [ "${CODEX_HOME+x}" ]; then export CODEX_HOME; fi ;;
|
|
140
|
+
hermes) if [ "${HERMES_HOME+x}" ]; then export HERMES_HOME; fi ;;
|
|
141
|
+
esac
|
|
142
|
+
# shellcheck disable=SC2086
|
|
143
|
+
exec node bin/cli-run.mjs "$AUDIT_LANE" $AUDIT_LANE_FLAGS --brief "$BRIEF" --timeout "$RUNNER_SECS" --quiet
|
|
144
|
+
) < /dev/null > "$TMP"
|
|
123
145
|
rc=$?
|
|
124
146
|
if [ "$rc" -eq 0 ] && [ -s "$TMP" ]; then
|
|
125
147
|
mv -f "$TMP" "$FINAL"
|
|
@@ -342,6 +342,40 @@ function runSummary(args) {
|
|
|
342
342
|
const starts = records.filter((r) => r.event === 'start').length;
|
|
343
343
|
const noMatchingStart = Math.max(0, dispatches.length - starts);
|
|
344
344
|
|
|
345
|
+
// Reconciliation: does the marker's claim match what the session actually did?
|
|
346
|
+
// Both halves are already in this log -- the Stop marker says which lane the turn
|
|
347
|
+
// DECLARED, and the PreToolUse event says what was actually DISPATCHED -- and until
|
|
348
|
+
// now nothing compared them. A declared lane is a claim; a dispatch is the act.
|
|
349
|
+
// Grouped by session because a marker is written once per turn while a dispatch can
|
|
350
|
+
// land on any turn of the same session, so turn-level pairing would report drift
|
|
351
|
+
// that is only ordering.
|
|
352
|
+
// PRESENCE, not identity, and the name says so: this asks whether a session that
|
|
353
|
+
// NAMED a lane went on to dispatch at all, not whether it dispatched the lane it named. A
|
|
354
|
+
// session that named builder and dispatched reader matches here. Lane identity is
|
|
355
|
+
// not recoverable from this log, because the marker names a lane from the user's own
|
|
356
|
+
// ROUTING.md while a dispatch names a subagent_type, and the two vocabularies do not
|
|
357
|
+
// have to line up.
|
|
358
|
+
const sessions = new Map();
|
|
359
|
+
for (const r of records) {
|
|
360
|
+
if (!r.session_id) continue;
|
|
361
|
+
if (!sessions.has(r.session_id)) sessions.set(r.session_id, { declaredOff: false, dispatched: false, sawLane: false });
|
|
362
|
+
const acc = sessions.get(r.session_id);
|
|
363
|
+
if (r.event === 'dispatch') acc.dispatched = true;
|
|
364
|
+
if (r.event === 'route' && Array.isArray(r.lane)) {
|
|
365
|
+
// 'missing' is what the hook writes when a turn carried NO marker, so it is the
|
|
366
|
+
// absence of a claim, not a claim of inline. Counting it here would double-report
|
|
367
|
+
// the gap the coverage line above already reports.
|
|
368
|
+
const named = r.lane.filter((l) => l !== 'missing');
|
|
369
|
+
if (named.length > 0) acc.sawLane = true;
|
|
370
|
+
if (named.some((l) => !inlineNames.has(String(l).trim().toLowerCase()))) acc.declaredOff = true;
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
const reconcilable = [...sessions.values()].filter((a) => a.sawLane);
|
|
374
|
+
const claimedNotDone = reconcilable.filter((a) => a.declaredOff && !a.dispatched).length;
|
|
375
|
+
const doneNotClaimed = reconcilable.filter((a) => !a.declaredOff && a.dispatched).length;
|
|
376
|
+
const agreed = reconcilable.length - claimedNotDone - doneNotClaimed;
|
|
377
|
+
const agreedPct = reconcilable.length > 0 ? (agreed / reconcilable.length) * 100 : null;
|
|
378
|
+
|
|
345
379
|
const ends = records.filter((r) => r.event === 'end' && r.agent_type && typeof r.duration_s === 'number');
|
|
346
380
|
const durationsByType = new Map();
|
|
347
381
|
for (const r of ends) {
|
|
@@ -366,6 +400,15 @@ function runSummary(args) {
|
|
|
366
400
|
if (dispatchCounts.size === 0) lines.push(' (none)');
|
|
367
401
|
for (const [type, count] of [...dispatchCounts.entries()].sort((a, b) => b[1] - a[1])) lines.push(' ' + type + ': ' + count);
|
|
368
402
|
lines.push('dispatches with no matching start: ' + noMatchingStart + ' (a hook or guard blocked them before launch)');
|
|
403
|
+
lines.push(
|
|
404
|
+
'delegation claimed vs observed: ' +
|
|
405
|
+
(agreedPct === null ? 'no session named a lane yet' : formatNumber(agreedPct) + '% of sessions match') +
|
|
406
|
+
' (' + agreed + '/' + reconcilable.length + ' sessions that named a lane)'
|
|
407
|
+
);
|
|
408
|
+
lines.push(' named a lane, no dispatch in this window: ' + claimedNotDone);
|
|
409
|
+
lines.push(' dispatched, but every named lane was inline: ' + doneNotClaimed);
|
|
410
|
+
lines.push(' presence only: a session that named one lane, then dispatched a different one, still counts as matching,');
|
|
411
|
+
lines.push(' and --since or a rotated log can split a session so one half lands in the counts above.');
|
|
369
412
|
lines.push('duration by agent_type (mean / max, seconds):');
|
|
370
413
|
if (durationsByType.size === 0) lines.push(' (none)');
|
|
371
414
|
for (const [type, durs] of durationsByType) {
|