model-orchestrator 0.1.34 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. package/AGENTS.md +31 -21
  2. package/CHANGELOG.md +51 -1
  3. package/README.md +127 -110
  4. package/bin/README.md +57 -6
  5. package/bin/aunx.js +7 -0
  6. package/bin/cli-run.mjs +21 -15
  7. package/bin/cli.js +376 -257
  8. package/docs/README.md +15 -18
  9. package/docs/catalog.md +228 -38
  10. package/docs/companions.md +28 -10
  11. package/docs/guarantees.md +21 -12
  12. package/docs/how-it-routes.md +49 -42
  13. package/docs/install.md +135 -33
  14. package/docs/part-1-beginner.md +37 -45
  15. package/docs/part-2-intermediate.md +34 -52
  16. package/docs/part-3-advanced.md +36 -26
  17. package/docs/security-review-history.md +38 -0
  18. package/llms.txt +24 -25
  19. package/package.json +16 -8
  20. package/proof/README.md +100 -0
  21. package/proof/gate-demo.cast +9 -0
  22. package/proof/gate-demo.gif +0 -0
  23. package/proof/results.json +198 -0
  24. package/proof/scripts/check-gate.js +26 -0
  25. package/proof/scripts/install-time.js +16 -0
  26. package/proof/scripts/lib.js +73 -0
  27. package/proof/scripts/measure.js +15 -0
  28. package/proof/scripts/missing-results.js +30 -0
  29. package/proof/scripts/record-gate.js +38 -0
  30. package/proof/scripts/render.js +18 -0
  31. package/proof/scripts/runner-overhead.js +21 -0
  32. package/src/README.md +9 -3
  33. package/src/activation-ownership.js +19 -0
  34. package/src/apply-companions.js +104 -0
  35. package/src/apply-snippets.js +60 -28
  36. package/src/aunx.js +262 -0
  37. package/src/catalog.js +253 -117
  38. package/src/install.js +478 -209
  39. package/src/plugin.js +13 -4
  40. package/src/postinstall.js +57 -0
  41. package/src/roles.js +184 -0
  42. package/src/uninstall.js +125 -8
  43. package/templates/README.md +19 -2
  44. package/templates/advanced/README.md +2 -2
  45. package/templates/advanced/vm/PRIVACY_GATES.md +17 -19
  46. package/templates/advanced/vm/README.md +25 -20
  47. package/templates/advanced/vm/box-CLAUDE.md +19 -18
  48. package/templates/advanced/vm/jobs/README.md +3 -1
  49. package/templates/advanced/vm/jobs/weekly-audit.service +3 -0
  50. package/templates/advanced/vm/jobs/weekly-audit.sh +2 -2
  51. package/templates/advanced/vm/setup-vm.sh +49 -2
  52. package/templates/agents/README.md +2 -2
  53. package/templates/agents/agy/README.md +20 -3
  54. package/templates/agents/agy/builder.md +11 -7
  55. package/templates/agents/agy/bulk-worker.md +9 -7
  56. package/templates/agents/agy/code-reviewer.md +13 -7
  57. package/templates/agents/agy/deep-planner.md +10 -7
  58. package/templates/agents/agy/done-verifier.md +13 -22
  59. package/templates/agents/agy/finding-verifier.md +14 -22
  60. package/templates/agents/agy/live-researcher.md +10 -7
  61. package/templates/agents/agy/reader.md +10 -12
  62. package/templates/agents/claude-code/README.md +18 -14
  63. package/templates/agents/claude-code/builder.md +10 -15
  64. package/templates/agents/claude-code/bulk-worker.md +8 -10
  65. package/templates/agents/claude-code/code-reviewer.md +11 -17
  66. package/templates/agents/claude-code/deep-planner.md +9 -11
  67. package/templates/agents/claude-code/done-verifier.md +12 -33
  68. package/templates/agents/claude-code/finding-verifier.md +13 -39
  69. package/templates/agents/claude-code/live-researcher.md +9 -11
  70. package/templates/agents/claude-code/reader.md +9 -18
  71. package/templates/agents/snippets/chat.md +9 -10
  72. package/templates/agents/snippets/claude-code.md +17 -18
  73. package/templates/agents/snippets/generic.md +9 -11
  74. package/templates/agents/snippets/route-gate.mjs +2 -2
  75. package/templates/agents/snippets/route-metrics.mjs +1 -1
  76. package/templates/agents/snippets/subagent-context.mjs +4 -4
  77. package/templates/beginner/ORCHESTRATOR.md +31 -36
  78. package/templates/beginner/README.md +1 -1
  79. package/templates/common/ACCEPTANCE_CHECKS.json +12 -0
  80. package/templates/common/CONTEXT.md +37 -0
  81. package/templates/common/DECISIONS.md +11 -0
  82. package/templates/common/README.md +24 -11
  83. package/templates/common/TASK_BRIEF.md +84 -0
  84. package/templates/common/protocols/README.md +14 -11
  85. package/templates/common/protocols/acceptance-checks.md +14 -0
  86. package/templates/common/protocols/build-protocol.md +91 -106
  87. package/templates/common/protocols/context-file.md +10 -0
  88. package/templates/common/protocols/decision-log.md +9 -0
  89. package/templates/common/protocols/deep-research.md +20 -34
  90. package/templates/common/protocols/docs-then-prove.md +13 -18
  91. package/templates/common/protocols/gap-analysis.md +15 -21
  92. package/templates/common/protocols/memory-and-record.md +21 -20
  93. package/templates/common/protocols/numbers-and-logic.md +20 -26
  94. package/templates/common/protocols/propagate.md +18 -27
  95. package/templates/intermediate/CLI-RUN.md +83 -113
  96. package/templates/intermediate/DELEGATION_MATRIX.md +9 -3
  97. package/templates/intermediate/README.md +3 -3
  98. package/templates/intermediate/RESEARCH_TRIAGE.md +23 -15
  99. package/templates/intermediate/ROUTING.md +54 -51
  100. package/templates/intermediate/TIERS.md +37 -76
  101. package/templates/tools/README.md +1 -1
  102. package/templates/tools/obsidian-tc/OBSIDIAN-TC.md +1 -1
  103. package/docs/audit-brief.md +0 -148
  104. package/scripts/README.md +0 -7
  105. package/scripts/gen-catalog.js +0 -81
  106. package/scripts/gen-plugin.js +0 -16
  107. package/scripts/record-demo.sh +0 -45
  108. package/templates/common/TASK_BUNDLE.md +0 -56
@@ -19,7 +19,7 @@ import { isAbsolute, join, basename } from 'node:path';
19
19
  const RULES_FILE_REL = {{RULES_DIR_OVERRIDE_JS}}
20
20
  ? {{RULES_CANDIDATES_JSON}}.map(rulesPath).join(' or ')
21
21
  : {{RULES_FILE_REL_JSON}};
22
- const TASK_BUNDLE_REL = rulesPath({{TASK_BUNDLE_REL_JSON}});
22
+ const TASK_BRIEF_REL = rulesPath({{TASK_BRIEF_REL_JSON}});
23
23
  const STDIN_DRAIN_MS = 250; // hard cap: never let an open, never-closed stdin pipe hold this hook open
24
24
 
25
25
  function rulesPath(baked) {
@@ -30,9 +30,9 @@ function rulesPath(baked) {
30
30
  const additionalContext = [
31
31
  'SUBAGENT CONTEXT (model-orchestrator).',
32
32
  'Routing rules: ' + RULES_FILE_REL + (isAbsolute(RULES_FILE_REL) ? '.' : ' (relative to the project root).'),
33
- 'Task bundle format: ' + TASK_BUNDLE_REL + '.',
34
- 'Report contract: say what you did, what you did NOT do, and what you could not verify. "Unverified" is acceptable; a confident guess is not. Stop at the bound your brief set, and never claim work you cannot show.',
35
- 'You are a delegate: do not route further work to another subagent yourself, and do not mark your own output as the final verification of it.'
33
+ 'Task brief format: ' + TASK_BRIEF_REL + '.',
34
+ 'Report contract: return coverage and evidence, including partial work and unverified checks. Stop at the bound set by your brief.',
35
+ 'When further delegation is authorized by your brief, keep the whole scope and section ownership in each handoff. Return your result to the assigning agent for independent verification.'
36
36
  ].join(' ');
37
37
 
38
38
  // Drain stdin without ever blocking on it. A bare `readFileSync(0)` waits
@@ -1,60 +1,55 @@
1
- # ORCHESTRATOR.md: routing rules for one agent
1
+ # ORCHESTRATOR.md: model routing inside one agent
2
2
 
3
- Primary agent: **{{PRIMARY_NAME}}**. Everything below runs inside that one agent. You do not need a second vendor to orchestrate; you need tiers, task classes, and gates that can fail.
3
+ {{STACK_SUMMARY}}
4
4
 
5
- ## Tiers (capability, not model names)
5
+ Main agent: **{{PRIMARY_NAME}}**. When work arrives, match it to a capability tier and the tools available in this session. A **tier** describes a model's capability and cost. A **lane** is an AI tool or model you can hand work to.
6
+
7
+ ## Choose the tier for the job
6
8
 
7
9
  | Tier | Use for | On {{PRIMARY_NAME}} |
8
10
  |---|---|---|
9
- | **deep** | ambiguous planning, architecture, strategy, root-cause debugging, anything expensive to get wrong | {{PRIMARY_DEEP}}, highest effort |
10
- | **standard** | code writing, code review, execution of a known plan, research synthesis | {{PRIMARY_STANDARD}}, high effort |
11
- | **fast** | classification, extraction, formatting, bulk summarization | {{PRIMARY_FAST}}, low effort |
12
-
13
- Three cost levers, always together: **tier** sets the price per token, **token discipline** sets how many tokens (read only what you will touch, never re-read, deliverables not narration), **effort** sets how hard each call thinks.
11
+ | planning model | ambiguity, architecture, strategy and unknown causes | {{PRIMARY_DEEP}} |
12
+ | working model | code writing, review, execution and research synthesis | {{PRIMARY_STANDARD}} |
13
+ | cheap model | classification, extraction, formatting and bulk summaries | {{PRIMARY_FAST}} |
14
14
 
15
- Robustness first, cost second. Split tiers because the split produces better work, not because it is cheaper. Cost is a constraint to respect, never the reason for a routing choice.
15
+ When routing, select capability, context size and effort together. Use a cheap model for bounded volume and a planning model when the decision requires it. For builds, check the live model roster and use high effort; raise to xhigh where supported for architecture, security or irreversible work.
16
16
 
17
17
  ## Decision tree (first match wins)
18
18
 
19
- 1. **Bulk and mechanical?** classify, tag, extract, reformat, summarize many similar items → fast tier.
20
- 2. **Needs live data?** trends, current docs, pricing, recent events → standard tier with tools; freshness comes from tools, not from a bigger model.
21
- 3. **Reviewing without changing?** → standard tier, read-only, findings ranked by severity. Escalate to deep only for security-critical review.
22
- 4. **Ambiguous, strategic, or expensive to get wrong?** "design my…", "figure out…", unknown cause → deep tier. Then hand the plan down.
19
+ 1. **Bulk or mechanical:** classify, tag, extract, rename or reformat -> cheap model tier.
20
+ 2. **Read many files:** use {{READER_ROLE}}; read the scoped sources and return the requested digest.
21
+ 3. **Current data:** use live tools with a working model.
22
+ 4. **Review code:** use {{REVIEW_ROLE}} in a fresh context, or an independent working model with read-only scope enforced by its prompt; Bash access remains a separate tool grant.
23
+ 5. **Verify findings:** reproduce each claim before a repair.
24
+ 6. **Check a definition of done:** probe its named artifact and report MET, NOT_MET or UNVERIFIABLE.
25
+ 7. **Ambiguity or architecture:** use a planning model and return an executable plan.
23
26
  {{DECISION_RULE5_L1}}
24
27
 
25
- Modifiers:
26
- - **Plan big, execute small.** The expensive tier steers, the cheaper tier does the volume. Never make the fast tier design anything; never make the deep tier grind out bulk output.{{INLINE_THRESHOLD_NOTE}}
27
- - **Never silently retry at the same tier after a failure.** Escalate one tier, or consult the deep tier once, and say which you did. If two consults do not unstick it, stop and tell the human.
28
- - **De-escalate.** If a request sounds deep but is a lookup or a small edit, route down. Default down, escalate on evidence.
29
-
30
- ## The two checkpoints (every build)
31
-
32
- - **Checkpoint 1, before writing anything.** You map everything it touches yourself (files, systems, docs, tickets). Then ask the deep tier, on the finished map: *is this the simplest way, what is most likely to go wrong, what did the request miss?* It must return one named weak spot and one gap in the request. Approval alone is not an answer.
33
- - **Checkpoint 2, after the build is green.** Security-shaped diffs get a second-opinion read (in a fresh context, told to challenge, allowed to answer CLEAN). Architecture-shaped diffs get the deep tier reviewing build against plan. Never both on one diff. Every finding reproduced before it reaches a human.
34
-
35
- Cap: two deep-tier consults per build. The full procedure is `protocols/build-protocol.md`.
28
+ When a route fails, diagnose the cause and state the next choice. When the task is a lookup, route down.{{INLINE_THRESHOLD_NOTE}}
36
29
 
37
- ## Delegating inside one agent
30
+ ## Build from a shared context
38
31
 
39
- {{DELEGATE_RULES_NOTE}} Every hand-off carries a `TASK_BUNDLE.md` brief: purpose, task class, granted scope, capabilities, denied actions, conventions it does not have, report contract, exit parameters. Absence is denial.
32
+ When building, follow `protocols/build-protocol.md`: freeze acceptance checks, probe availability, spike risky assumptions, order dependencies, research bounded questions, write one context file, and assign by live capability. Then build, merge split work, audit once with a companion consult asking a different question, and verify the authorized change in use.
40
33
 
41
- ## Numbers and logic go through a tool, never your head
34
+ When a background task runs, check liveness and output growth every five minutes. After two checks without growth, diagnose and report. When a boundary refuses a write, hand that patch to an authorized writer and continue independent work.
42
35
 
43
- Any figure someone will act on, any comparison you state, any complexity, equivalence or speedup claim: computed, not estimated. `protocols/numbers-and-logic.md` names when calling is mandatory. Companion tool: codecalc ({{CODECALC_STATUS}}).
36
+ When the workflow needs an independent reviewer unavailable to this session, report the audit as pending and arrange a separate review before shipping.
44
37
 
45
- ## Memory and record
38
+ ## Hand off a complete task brief
46
39
 
47
- Anything durable is searched for before it is written, its folder index is corrected in the same pass, and one writer records. `protocols/memory-and-record.md`. Companion tool (optional, needs an Obsidian vault): obsidian-tc, {{OBSIDIAN_TC_STATUS}}.
40
+ {{DELEGATE_RULES_NOTE}} When handing off work, use `TASK_BRIEF.md` (`aunx brief`): include the user's ask, context file, exact scope, non-goals, interfaces, capabilities, denied actions, acceptance checks, inventory, measurements and coverage-table report contract. Actions outside the granted scope are denied.
48
41
 
49
- ## Docs, then prove
42
+ ## Tools and fallbacks
50
43
 
51
- Before writing code against a library, SDK, API or CLI you have not confirmed the current shape of, pull current docs; then a run, not the doc, is what proves it behaves that way. `protocols/docs-then-prove.md`. Companion tool (optional, needs a network call): Context7, {{CONTEXT7_STATUS}}. Paired with codecalc, {{CODECALC_STATUS}}: docs say what it is supposed to do, codecalc's run says what it actually does.
44
+ - When a number or logical claim affects a decision, compute it with a tool. codecalc: {{CODECALC_STATUS}}. When absent, use the local runtime, spreadsheet or tests. See `protocols/numbers-and-logic.md`.
45
+ - When recording durable information, search first, update its index and keep one writer. obsidian-tc: {{OBSIDIAN_TC_STATUS}}. When absent, use project files and version control. See `protocols/memory-and-record.md`.
46
+ - When writing against a changing interface, read current docs and run a check. Context7: {{CONTEXT7_STATUS}}. When absent, use official docs or installed source; when codecalc is absent, use the project's runtime. See `protocols/docs-then-prove.md`.
52
47
 
53
- ## The seven protocols
48
+ ## Workflow files
54
49
 
55
- `protocols/build-protocol.md` · `protocols/propagate.md` · `protocols/gap-analysis.md` · `protocols/deep-research.md` · `protocols/numbers-and-logic.md` · `protocols/memory-and-record.md` · `protocols/docs-then-prove.md`. Each is a set of questions that can be answered wrong. That is the design, not a flaw.
50
+ When beginning a build, scaffold the context file with `aunx context` and the acceptance checks with `aunx checks`. Record decisions in `DECISIONS.md`. Use `protocols/README.md` to find the procedure for a rename, research task, record update or coverage check.
56
51
 
57
- ## When you outgrow this
52
+ ## Add another AI tool
58
53
 
59
- You will know: you keep wanting a second model family to read your diff, a $0 lane for bulk, or a live-data lane your primary does not have. That is level 2. Re-run the installer with `--level 2`.
54
+ When a task needs another model family, tool reach or capacity, rerun the installer with `--level 2`. Verify that lane's access before assigning work.
60
55
  {{ROUTE_GATE_SECTION}}
@@ -1,3 +1,3 @@
1
1
  # templates/beginner/
2
2
 
3
- Written at every level. `ORCHESTRATOR.md` is the single-agent routing document: tiers as effort levels inside one agent, the five-branch decision tree, the two checkpoints, and when you have outgrown level 1. At level 2+ `ROUTING.md` supersedes it and says so.
3
+ Written at every level. `ORCHESTRATOR.md` is the single-agent routing document: planning, working and cheap model tiers; task classification; building from a shared context file and acceptance checks; handing off a task brief; tool fallbacks; and when to add another AI tool. At level 2 and up, `ROUTING.md` supersedes it.
@@ -0,0 +1,12 @@
1
+ {
2
+ "version": 1,
3
+ "checks": [
4
+ {
5
+ "id": "replace-this-example",
6
+ "source": "Replace with the user's quoted requirement",
7
+ "description": "Replace with a property of the final artifact and a command that verifies it",
8
+ "command": ["node", "-e", "process.exit(1)"],
9
+ "cwd": "."
10
+ }
11
+ ]
12
+ }
@@ -0,0 +1,37 @@
1
+ # Context file
2
+
3
+ ## Request and approved scope
4
+
5
+ <Quote the user's request. Name the final artifact, authorized surfaces and non-goals.>
6
+
7
+ ## Source of truth
8
+
9
+ <List relevant rules, specifications and source paths with a short note on what each establishes.>
10
+
11
+ ## Current state
12
+
13
+ <Record the branch or revision, existing changes, baseline checks and live availability probes. Include the probe date and surface inspected.>
14
+
15
+ ## Affected surfaces
16
+
17
+ <Name the files, interfaces and consumers this change touches. Record bounded search evidence.>
18
+
19
+ ## Decisions and open questions
20
+
21
+ <Link the decision log. List resolved choices and any question that blocks dependent work.>
22
+
23
+ ## Acceptance checks
24
+
25
+ <Point to the checks file. Include baseline results, required runtime access and assumptions that still need a spike.>
26
+
27
+ ## Resources and order of work
28
+
29
+ <List available lanes and their reach, each section's owner, dependencies and steps skipped with reasons.>
30
+
31
+ ## Measurements
32
+
33
+ <Give each runtime spike's value, method, script, sample size and date.>
34
+
35
+ ## Verification and handoff
36
+
37
+ <Record final evidence and anything still unverified. Every task brief for this run reads this file.>
@@ -0,0 +1,11 @@
1
+ # Decision log
2
+
3
+ When choosing an approach, append a concise decision record with evidence. Record the practical reason, not private chain-of-thought.
4
+
5
+ ## <Date: decision title>
6
+
7
+ - **Did:** <the choice made>.
8
+ - **Why:** <the evidence and tradeoff that support it>.
9
+ - **Serves:** <the requirement or user outcome it serves>.
10
+ - **Rejected:** <alternatives considered and the reason each was set aside>.
11
+ - **Evidence:** <source path, command or measured result>.
@@ -1,9 +1,15 @@
1
1
  # Your orchestrator (start here)
2
2
 
3
3
  Installed {{DATE}} · level {{LEVEL_ID}}: **{{LEVEL_NAME}}**, {{LEVEL_TAGLINE}}
4
- Primary agent: **{{PRIMARY_NAME}}**
4
+ Main agent: **{{PRIMARY_NAME}}**
5
5
 
6
- You have access to:
6
+ {{STACK_TABLE}}
7
+
8
+ {{STACK_FALLBACK_NOTE}}
9
+
10
+ {{STACK_GAPS}}
11
+
12
+ What each AI is:
7
13
  {{AIS_LIST}}
8
14
 
9
15
  Companion tools:
@@ -11,14 +17,18 @@ Companion tools:
11
17
 
12
18
  ## The idea in one line
13
19
 
14
- This folder gives your agent routing instructions and, at level 2+, a runner for explicitly selected CLI lanes. The agent chooses the tier or lane; the runner does not automatically compare prices or choose a model.
20
+ This folder gives your agent routing instructions and, at level 2+, a runner for explicitly selected CLI lanes. The agent reads the rules and chooses the tier or lane; `aunx route` supplies a keyword suggestion and `aunx cli-run` runs the lane the caller selects.
15
21
 
16
- ## Activate it
22
+ ## What's left for you
17
23
 
18
24
  These are the same steps, in the same order, that the installer printed in your terminal.{{CHAT_UPLOAD_NOTE}}
19
25
 
20
26
  {{ACTIVATION_STEPS}}
21
27
 
28
+ Interactive installs apply the main agent's rules and supported project settings after the single confirmation. Choose automatic activation in the edit screen or use `--no-apply` to leave activation for later. With `--yes`, add `--apply-snippets` to apply it. Selected companions are registered when the main agent has a cataloged project MCP config; other hosts keep a manual setup step.
29
+
30
+ The installer runs the doctor presence check automatically. Live canaries remain opt-in through `--doctor --run` at level 2 or above.
31
+
22
32
  ## Then prove it took
23
33
 
24
34
  {{PROOF_STEPS}}
@@ -27,8 +37,11 @@ These are the same steps, in the same order, that the installer printed in your
27
37
 
28
38
  | File | Read it when |
29
39
  |---|---|
30
- | `ORCHESTRATOR.md` | First. The routing rules your primary agent follows: tiers, task classes, the two checkpoints. |
31
- | `TASK_BUNDLE.md` | Before you hand any work to a subagent, a second CLI, or a chat window. The brief template. |
40
+ | `{{ROUTING_FILE}}` | First. The routing rules your main agent follows: tiers, task classes, acceptance checks and resource selection. |
41
+ | `CONTEXT.md` | At the start of a run. Shared source facts, scope, decisions and measurements (`aunx context`). |
42
+ | `ACCEPTANCE_CHECKS.json` | When verifying the final artifact (`aunx checks run`). Replace the failing example first. |
43
+ | `DECISIONS.md` | When choosing an approach. Did / Why / Serves / Rejected with evidence. |
44
+ | `TASK_BRIEF.md` | Before you hand any work to a subagent, a second CLI, or a chat window. The brief template. |
32
45
  | `protocols/build-protocol.md` | You are about to build, code, migrate or deploy something. |
33
46
  | `protocols/propagate.md` | You are renaming or changing a term, path, slug, schema field or routing rule. |
34
47
  | `protocols/gap-analysis.md` | You just finished something comprehensive and want the second pass that hunts for what is missing. |
@@ -46,11 +59,11 @@ Level 2 adds `ROUTING.md`, `TIERS.md`, `DELEGATION_MATRIX.md`, `RESEARCH_TRIAGE.
46
59
 
47
60
  {{LOAD_IT}}
48
61
 
49
- ## The three rules that carry everything
62
+ ## Route, check and verify
50
63
 
51
- 1. **Route by capability tier, not by model name.** deep = ambiguous or expensive to get wrong · standard = well-specified execution and review · fast = bulk and mechanical. Default down, escalate on evidence.
52
- 2. **A gate you cannot fail is not a gate.** "Does it look good?" passes every time. "Name what is most likely to go wrong, and what the request missed" can come back empty, which is how you know it worked.
53
- 3. **Exit 0 is not a deliverable.** Any tool, CLI or subagent can report success and hand back nothing. Check for the artifact, not the status line.
64
+ 1. When selecting a tier, use a planning model for ambiguity, a working model for execution and review, and a cheap model for mechanical work. Check current capabilities before dispatch.
65
+ 2. When introducing a gate, demonstrate its failing case before relying on a passing result.
66
+ 3. When a tool reports success, verify the requested artifact and acceptance checks.
54
67
 
55
68
  ## Where things went
56
69
 
@@ -58,4 +71,4 @@ Level 2 adds `ROUTING.md`, `TIERS.md`, `DELEGATION_MATRIX.md`, `RESEARCH_TRIAGE.
58
71
 
59
72
  ## Uninstall
60
73
 
61
- Before deleting anything, inspect `MANIFEST.json`: entries beginning with `[project] ` identify the subagent files managed by this installation. Review those individual files and remove only the ones you no longer want, preserving edited or pre-existing files. Never delete the shared `.claude/agents` or `.agents/agents` folder; it may contain unrelated agents. If the manifest is missing, inspect files individually rather than deleting a folder. Remove the orchestrator block you manually copied into your project rules file or chat instructions. Then delete this generated docs folder only after preserving any work you added to it. The optional log at `~/.ai-orchestrator/cli-run.log.jsonl` is shared across installations; remove it only if you no longer need that history.
74
+ When removing this installation, run the installer with `--uninstall` and the same `--dir` and `--project` paths. Use `--dry` first to inspect what would be removed. The manifest identifies managed files, applied rules blocks and added hook or MCP entries; edited entries are preserved and named. Backups stay beside changed files. Never delete shared agent folders that may contain unrelated files. Review any rules or settings you merged by hand and remove only this installation's entries. The local log at `~/.ai-orchestrator/cli-run.log.jsonl` is shared across installations; preserve it while another installation uses it.
@@ -0,0 +1,84 @@
1
+ # Task brief (`aunx brief`)
2
+
3
+ When handing work to an agent, CLI or fresh session, fill this brief and pass it with the relevant context. Keep every field; write `none` with a reason for an empty field. Treat capabilities as an explicit allow-list: absence is denial.
4
+
5
+ ## Context
6
+
7
+ <Path to the run's context file, source files to read first, and facts that explain this task. Give a fresh session access to the actual contents.>
8
+
9
+ ## The user's ask
10
+
11
+ > <Quote the user's acceptance-critical request. Distinguish a rule from a preference or an unresolved choice.>
12
+
13
+ ## Purpose
14
+
15
+ <What this task is for and why.>
16
+
17
+ ## Scope of build
18
+
19
+ - Finished artifact: <one line naming the final result>.
20
+ - Files and changes: <each path, its owner and required behavior>.
21
+ - Shared interfaces: <contracts every section depends on>.
22
+ - Order of work: <dependencies and next steps>.
23
+ - Non-goals: <work deliberately outside this task>.
24
+ - Interfaces not to break: <existing commands, formats, APIs and behaviors>.
25
+
26
+ For a non-build task, describe the bounded result here and mark build-only fields inapplicable with a reason.
27
+
28
+ ## Task class
29
+
30
+ <read_only | draft_only | mutating>. For `draft_only`, write the proposed result to the named output and leave its destination unchanged.
31
+
32
+ ## Granted scope
33
+
34
+ <Paths, topics, record sets, environments and write ownership. Everything outside this list is out of scope.>
35
+
36
+ ## Capabilities
37
+
38
+ <Allowed actions and tools, including exact write destinations and permitted commands.>
39
+
40
+ ## Denied actions
41
+
42
+ <Explicit prohibitions, such as publishing, sending, deleting, changing permissions or exposing secrets. Actions absent from Capabilities are denied.>
43
+
44
+ ## Conventions
45
+
46
+ <Project rules the receiving session needs. Even a session that loads standing rules needs this task's scope and current decisions.>
47
+
48
+ ## Acceptance checks
49
+
50
+ | Requirement quoted from the ask | Final property | Verifier command or manual procedure | Baseline result |
51
+ |---|---|---|---|
52
+ | <quote> | <observable property> | <command; exit 0 means satisfied> | <PASS / FAIL / UNVERIFIED> |
53
+
54
+ Include availability checks for required tools and access. Name the checks file when using `aunx checks run`. Replay against the final merged artifact and any materialized output whose properties may change.
55
+
56
+ ## Resource inventory
57
+
58
+ - The lane **holds**: <probed tools, models, context window, permissions and runtime access>.
59
+ - The lane **lacks**: <needed capabilities absent here, and an authorized route to them if known>.
60
+ - Probe evidence: <command, date, result and the surface it actually inspected>.
61
+
62
+ ## Measurements
63
+
64
+ <Runtime spikes, baseline counts or performance figures with method, script, sample size and date. Mark an unmeasured assumption UNVERIFIED.>
65
+
66
+ ## Report contract
67
+
68
+ Return a coverage table with one row per requirement:
69
+
70
+ | Requirement | Status | Evidence |
71
+ |---|---|---|
72
+ | <requirement> | IMPLEMENTED / PARTIAL / MISSING / OUT-OF-SCOPE | <path, test or command output> |
73
+
74
+ Name changed files, verification results, unresolved decisions and unverified behavior. State what you did and did not do. Keep evidence separate from inference.
75
+
76
+ ## Exit parameters
77
+
78
+ <A wall-clock ceiling, work ceiling or stop condition. For background work, name the five-minute heartbeat and the response to two checks without progress.>
79
+
80
+ When a bound is reached, stop and return the partial result with uncovered requirements named. When permission refuses a write, hand the patch to an authorized writer and continue independent work.
81
+
82
+ ## Small read-only lookups
83
+
84
+ For a one-line lookup with no output artifact, state `Task brief: none (one-line lookup, read-only)` and its scope. For secrets, deletion, bulk mutation, deployment or someone else's data, use the full brief and the appropriate permission boundary.
@@ -1,15 +1,18 @@
1
- # protocols/
1
+ # Workflow playbooks
2
2
 
3
- Seven procedures. Each one is a list of questions whose answers can be wrong.
3
+ When a task matches a row, use that procedure and keep its evidence with the result.
4
4
 
5
- | File | Fires when | The gate |
5
+ | File | Use when | Result |
6
6
  |---|---|---|
7
- | `build-protocol.md` | You build, code, implement, migrate or deploy | 3 phases, 8 stages; the ship step is the only irreversible one |
8
- | `propagate.md` | You rename or change anything other files reference | Re-grep the OLD name everywhere and expect zero |
9
- | `gap-analysis.md` | You finished something comprehensive | A second pass that hunts for what is MISSING, ideally by a different model |
10
- | `deep-research.md` | The source set is unknown and the answer will be cited later | Parallel engines, then triage; disagreement is the signal |
11
- | `numbers-and-logic.md` | You are about to state a number, a comparison, a complexity, an equivalence | Computed by a tool (codecalc) or not stated |
12
- | `memory-and-record.md` | You are about to write anything durable | Searched first, indexed in the same pass, one writer (obsidian-tc when selected) |
13
- | `docs-then-prove.md` | You are about to write code against a library, SDK, API or CLI | Current docs first (Context7 when selected), then a run proves it; the run wins on disagreement |
7
+ | `build-protocol.md` | Building, implementing, migrating or deploying | A scoped, checked change verified in use |
8
+ | `context-file.md` | Sharing build context across agents | One source file every brief reads |
9
+ | `acceptance-checks.md` | Turning requirements into final-artifact checks | Commands and explicit manual checks with PASS/FAIL evidence |
10
+ | `decision-log.md` | Choosing an approach or skipping a step | Did / Why / Serves / Rejected with evidence |
11
+ | `propagate.md` | Changing a shared name, path or convention | All affected surfaces updated and the old identifier checked |
12
+ | `gap-analysis.md` | Checking coverage against a request | Missing scope and evidence gaps reported |
13
+ | `deep-research.md` | Answering a question from an initially unknown source set | Bounded research with verified sources |
14
+ | `numbers-and-logic.md` | Reporting consequential numbers or logical claims | Computed results and method |
15
+ | `memory-and-record.md` | Writing durable information | A searchable, indexed record with one writer |
16
+ | `docs-then-prove.md` | Coding against a changing interface | Current documentation and runtime verification |
14
17
 
15
- Not for lookups, prose edits, bulk classification or one-line config. Those get none of this.
18
+ For lookups, prose edits, bulk classification and one-line configuration changes, use the relevant routing rule and direct verification. When an optional companion tool is absent, each affected procedure names an equivalent local workflow.
@@ -0,0 +1,14 @@
1
+ # Acceptance checks (`aunx checks`)
2
+
3
+ - When a requirement determines acceptance, quote it and turn it into an observable property of the final artifact.
4
+ - Scaffold a checks file with `aunx checks`. Replace the intentionally failing example with a real verifier before using the file.
5
+ - Prefer an argv array for `command`, for example `["node", "--test", "test/example.test.js"]`. String commands run through the local shell, so review them as executable code before running a checks file.
6
+ - Set `cwd` relative to the checks file. Keep checks inside the task's authorized scope.
7
+ - Demonstrate a failing case for each new gate before trusting a passing result.
8
+ - Include availability facts when later work depends on a tool, permission or service remaining accessible.
9
+ - Run `aunx checks run ACCEPTANCE_CHECKS.json` against the final artifact. A failed command gives the gate exit code 1.
10
+ - When a check needs human judgment, record `manual: true` and its procedure in `description`. A manual or unverified check blocks the automated gate until replaced by a verifiable result; it is never silently counted as PASS.
11
+ - When packaging, rendering or deployment can change the property, replay the check against that materialized output too.
12
+ - When scope changes, preserve the original check and the decision that replaces it in the decision log.
13
+
14
+ When the command runner is unavailable, run each verifier with the project's local runtime and record its exit code. Optional companion tools can supply a verifier; their absence calls for an equivalent local check or an explicit UNVERIFIED result.