model-orchestrator 0.1.34 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +31 -21
- package/CHANGELOG.md +51 -1
- package/README.md +127 -110
- package/bin/README.md +57 -6
- package/bin/aunx.js +7 -0
- package/bin/cli-run.mjs +21 -15
- package/bin/cli.js +376 -257
- package/docs/README.md +15 -18
- package/docs/catalog.md +228 -38
- package/docs/companions.md +28 -10
- package/docs/guarantees.md +21 -12
- package/docs/how-it-routes.md +49 -42
- package/docs/install.md +135 -33
- package/docs/part-1-beginner.md +37 -45
- package/docs/part-2-intermediate.md +34 -52
- package/docs/part-3-advanced.md +36 -26
- package/docs/security-review-history.md +38 -0
- package/llms.txt +24 -25
- package/package.json +16 -8
- package/proof/README.md +100 -0
- package/proof/gate-demo.cast +9 -0
- package/proof/gate-demo.gif +0 -0
- package/proof/results.json +198 -0
- package/proof/scripts/check-gate.js +26 -0
- package/proof/scripts/install-time.js +16 -0
- package/proof/scripts/lib.js +73 -0
- package/proof/scripts/measure.js +15 -0
- package/proof/scripts/missing-results.js +30 -0
- package/proof/scripts/record-gate.js +38 -0
- package/proof/scripts/render.js +18 -0
- package/proof/scripts/runner-overhead.js +21 -0
- package/src/README.md +9 -3
- package/src/activation-ownership.js +19 -0
- package/src/apply-companions.js +104 -0
- package/src/apply-snippets.js +60 -28
- package/src/aunx.js +262 -0
- package/src/catalog.js +253 -117
- package/src/install.js +478 -209
- package/src/plugin.js +13 -4
- package/src/postinstall.js +57 -0
- package/src/roles.js +184 -0
- package/src/uninstall.js +125 -8
- package/templates/README.md +19 -2
- package/templates/advanced/README.md +2 -2
- package/templates/advanced/vm/PRIVACY_GATES.md +17 -19
- package/templates/advanced/vm/README.md +25 -20
- package/templates/advanced/vm/box-CLAUDE.md +19 -18
- package/templates/advanced/vm/jobs/README.md +3 -1
- package/templates/advanced/vm/jobs/weekly-audit.service +3 -0
- package/templates/advanced/vm/jobs/weekly-audit.sh +2 -2
- package/templates/advanced/vm/setup-vm.sh +49 -2
- package/templates/agents/README.md +2 -2
- package/templates/agents/agy/README.md +20 -3
- package/templates/agents/agy/builder.md +11 -7
- package/templates/agents/agy/bulk-worker.md +9 -7
- package/templates/agents/agy/code-reviewer.md +13 -7
- package/templates/agents/agy/deep-planner.md +10 -7
- package/templates/agents/agy/done-verifier.md +13 -22
- package/templates/agents/agy/finding-verifier.md +14 -22
- package/templates/agents/agy/live-researcher.md +10 -7
- package/templates/agents/agy/reader.md +10 -12
- package/templates/agents/claude-code/README.md +18 -14
- package/templates/agents/claude-code/builder.md +10 -15
- package/templates/agents/claude-code/bulk-worker.md +8 -10
- package/templates/agents/claude-code/code-reviewer.md +11 -17
- package/templates/agents/claude-code/deep-planner.md +9 -11
- package/templates/agents/claude-code/done-verifier.md +12 -33
- package/templates/agents/claude-code/finding-verifier.md +13 -39
- package/templates/agents/claude-code/live-researcher.md +9 -11
- package/templates/agents/claude-code/reader.md +9 -18
- package/templates/agents/snippets/chat.md +9 -10
- package/templates/agents/snippets/claude-code.md +17 -18
- package/templates/agents/snippets/generic.md +9 -11
- package/templates/agents/snippets/route-gate.mjs +2 -2
- package/templates/agents/snippets/route-metrics.mjs +1 -1
- package/templates/agents/snippets/subagent-context.mjs +4 -4
- package/templates/beginner/ORCHESTRATOR.md +31 -36
- package/templates/beginner/README.md +1 -1
- package/templates/common/ACCEPTANCE_CHECKS.json +12 -0
- package/templates/common/CONTEXT.md +37 -0
- package/templates/common/DECISIONS.md +11 -0
- package/templates/common/README.md +24 -11
- package/templates/common/TASK_BRIEF.md +84 -0
- package/templates/common/protocols/README.md +14 -11
- package/templates/common/protocols/acceptance-checks.md +14 -0
- package/templates/common/protocols/build-protocol.md +91 -106
- package/templates/common/protocols/context-file.md +10 -0
- package/templates/common/protocols/decision-log.md +9 -0
- package/templates/common/protocols/deep-research.md +20 -34
- package/templates/common/protocols/docs-then-prove.md +13 -18
- package/templates/common/protocols/gap-analysis.md +15 -21
- package/templates/common/protocols/memory-and-record.md +21 -20
- package/templates/common/protocols/numbers-and-logic.md +20 -26
- package/templates/common/protocols/propagate.md +18 -27
- package/templates/intermediate/CLI-RUN.md +83 -113
- package/templates/intermediate/DELEGATION_MATRIX.md +9 -3
- package/templates/intermediate/README.md +3 -3
- package/templates/intermediate/RESEARCH_TRIAGE.md +23 -15
- package/templates/intermediate/ROUTING.md +54 -51
- package/templates/intermediate/TIERS.md +37 -76
- package/templates/tools/README.md +1 -1
- package/templates/tools/obsidian-tc/OBSIDIAN-TC.md +1 -1
- package/docs/audit-brief.md +0 -148
- package/scripts/README.md +0 -7
- package/scripts/gen-catalog.js +0 -81
- package/scripts/gen-plugin.js +0 -16
- package/scripts/record-demo.sh +0 -45
- package/templates/common/TASK_BUNDLE.md +0 -56
|
@@ -19,7 +19,7 @@ import { isAbsolute, join, basename } from 'node:path';
|
|
|
19
19
|
const RULES_FILE_REL = {{RULES_DIR_OVERRIDE_JS}}
|
|
20
20
|
? {{RULES_CANDIDATES_JSON}}.map(rulesPath).join(' or ')
|
|
21
21
|
: {{RULES_FILE_REL_JSON}};
|
|
22
|
-
const
|
|
22
|
+
const TASK_BRIEF_REL = rulesPath({{TASK_BRIEF_REL_JSON}});
|
|
23
23
|
const STDIN_DRAIN_MS = 250; // hard cap: never let an open, never-closed stdin pipe hold this hook open
|
|
24
24
|
|
|
25
25
|
function rulesPath(baked) {
|
|
@@ -30,9 +30,9 @@ function rulesPath(baked) {
|
|
|
30
30
|
const additionalContext = [
|
|
31
31
|
'SUBAGENT CONTEXT (model-orchestrator).',
|
|
32
32
|
'Routing rules: ' + RULES_FILE_REL + (isAbsolute(RULES_FILE_REL) ? '.' : ' (relative to the project root).'),
|
|
33
|
-
'Task
|
|
34
|
-
'Report contract:
|
|
35
|
-
'
|
|
33
|
+
'Task brief format: ' + TASK_BRIEF_REL + '.',
|
|
34
|
+
'Report contract: return coverage and evidence, including partial work and unverified checks. Stop at the bound set by your brief.',
|
|
35
|
+
'When further delegation is authorized by your brief, keep the whole scope and section ownership in each handoff. Return your result to the assigning agent for independent verification.'
|
|
36
36
|
].join(' ');
|
|
37
37
|
|
|
38
38
|
// Drain stdin without ever blocking on it. A bare `readFileSync(0)` waits
|
|
@@ -1,60 +1,55 @@
|
|
|
1
|
-
# ORCHESTRATOR.md: routing
|
|
1
|
+
# ORCHESTRATOR.md: model routing inside one agent
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
{{STACK_SUMMARY}}
|
|
4
4
|
|
|
5
|
-
|
|
5
|
+
Main agent: **{{PRIMARY_NAME}}**. When work arrives, match it to a capability tier and the tools available in this session. A **tier** describes a model's capability and cost. A **lane** is an AI tool or model you can hand work to.
|
|
6
|
+
|
|
7
|
+
## Choose the tier for the job
|
|
6
8
|
|
|
7
9
|
| Tier | Use for | On {{PRIMARY_NAME}} |
|
|
8
10
|
|---|---|---|
|
|
9
|
-
|
|
|
10
|
-
|
|
|
11
|
-
|
|
|
12
|
-
|
|
13
|
-
Three cost levers, always together: **tier** sets the price per token, **token discipline** sets how many tokens (read only what you will touch, never re-read, deliverables not narration), **effort** sets how hard each call thinks.
|
|
11
|
+
| planning model | ambiguity, architecture, strategy and unknown causes | {{PRIMARY_DEEP}} |
|
|
12
|
+
| working model | code writing, review, execution and research synthesis | {{PRIMARY_STANDARD}} |
|
|
13
|
+
| cheap model | classification, extraction, formatting and bulk summaries | {{PRIMARY_FAST}} |
|
|
14
14
|
|
|
15
|
-
|
|
15
|
+
When routing, select capability, context size and effort together. Use a cheap model for bounded volume and a planning model when the decision requires it. For builds, check the live model roster and use high effort; raise to xhigh where supported for architecture, security or irreversible work.
|
|
16
16
|
|
|
17
17
|
## Decision tree (first match wins)
|
|
18
18
|
|
|
19
|
-
1. **Bulk
|
|
20
|
-
2. **
|
|
21
|
-
3. **
|
|
22
|
-
4. **
|
|
19
|
+
1. **Bulk or mechanical:** classify, tag, extract, rename or reformat -> cheap model tier.
|
|
20
|
+
2. **Read many files:** use {{READER_ROLE}}; read the scoped sources and return the requested digest.
|
|
21
|
+
3. **Current data:** use live tools with a working model.
|
|
22
|
+
4. **Review code:** use {{REVIEW_ROLE}} in a fresh context, or an independent working model with read-only scope enforced by its prompt; Bash access remains a separate tool grant.
|
|
23
|
+
5. **Verify findings:** reproduce each claim before a repair.
|
|
24
|
+
6. **Check a definition of done:** probe its named artifact and report MET, NOT_MET or UNVERIFIABLE.
|
|
25
|
+
7. **Ambiguity or architecture:** use a planning model and return an executable plan.
|
|
23
26
|
{{DECISION_RULE5_L1}}
|
|
24
27
|
|
|
25
|
-
|
|
26
|
-
- **Plan big, execute small.** The expensive tier steers, the cheaper tier does the volume. Never make the fast tier design anything; never make the deep tier grind out bulk output.{{INLINE_THRESHOLD_NOTE}}
|
|
27
|
-
- **Never silently retry at the same tier after a failure.** Escalate one tier, or consult the deep tier once, and say which you did. If two consults do not unstick it, stop and tell the human.
|
|
28
|
-
- **De-escalate.** If a request sounds deep but is a lookup or a small edit, route down. Default down, escalate on evidence.
|
|
29
|
-
|
|
30
|
-
## The two checkpoints (every build)
|
|
31
|
-
|
|
32
|
-
- **Checkpoint 1, before writing anything.** You map everything it touches yourself (files, systems, docs, tickets). Then ask the deep tier, on the finished map: *is this the simplest way, what is most likely to go wrong, what did the request miss?* It must return one named weak spot and one gap in the request. Approval alone is not an answer.
|
|
33
|
-
- **Checkpoint 2, after the build is green.** Security-shaped diffs get a second-opinion read (in a fresh context, told to challenge, allowed to answer CLEAN). Architecture-shaped diffs get the deep tier reviewing build against plan. Never both on one diff. Every finding reproduced before it reaches a human.
|
|
34
|
-
|
|
35
|
-
Cap: two deep-tier consults per build. The full procedure is `protocols/build-protocol.md`.
|
|
28
|
+
When a route fails, diagnose the cause and state the next choice. When the task is a lookup, route down.{{INLINE_THRESHOLD_NOTE}}
|
|
36
29
|
|
|
37
|
-
##
|
|
30
|
+
## Build from a shared context
|
|
38
31
|
|
|
39
|
-
|
|
32
|
+
When building, follow `protocols/build-protocol.md`: freeze acceptance checks, probe availability, spike risky assumptions, order dependencies, research bounded questions, write one context file, and assign by live capability. Then build, merge split work, audit once with a companion consult asking a different question, and verify the authorized change in use.
|
|
40
33
|
|
|
41
|
-
|
|
34
|
+
When a background task runs, check liveness and output growth every five minutes. After two checks without growth, diagnose and report. When a boundary refuses a write, hand that patch to an authorized writer and continue independent work.
|
|
42
35
|
|
|
43
|
-
|
|
36
|
+
When the workflow needs an independent reviewer unavailable to this session, report the audit as pending and arrange a separate review before shipping.
|
|
44
37
|
|
|
45
|
-
##
|
|
38
|
+
## Hand off a complete task brief
|
|
46
39
|
|
|
47
|
-
|
|
40
|
+
{{DELEGATE_RULES_NOTE}} When handing off work, use `TASK_BRIEF.md` (`aunx brief`): include the user's ask, context file, exact scope, non-goals, interfaces, capabilities, denied actions, acceptance checks, inventory, measurements and coverage-table report contract. Actions outside the granted scope are denied.
|
|
48
41
|
|
|
49
|
-
##
|
|
42
|
+
## Tools and fallbacks
|
|
50
43
|
|
|
51
|
-
|
|
44
|
+
- When a number or logical claim affects a decision, compute it with a tool. codecalc: {{CODECALC_STATUS}}. When absent, use the local runtime, spreadsheet or tests. See `protocols/numbers-and-logic.md`.
|
|
45
|
+
- When recording durable information, search first, update its index and keep one writer. obsidian-tc: {{OBSIDIAN_TC_STATUS}}. When absent, use project files and version control. See `protocols/memory-and-record.md`.
|
|
46
|
+
- When writing against a changing interface, read current docs and run a check. Context7: {{CONTEXT7_STATUS}}. When absent, use official docs or installed source; when codecalc is absent, use the project's runtime. See `protocols/docs-then-prove.md`.
|
|
52
47
|
|
|
53
|
-
##
|
|
48
|
+
## Workflow files
|
|
54
49
|
|
|
55
|
-
|
|
50
|
+
When beginning a build, scaffold the context file with `aunx context` and the acceptance checks with `aunx checks`. Record decisions in `DECISIONS.md`. Use `protocols/README.md` to find the procedure for a rename, research task, record update or coverage check.
|
|
56
51
|
|
|
57
|
-
##
|
|
52
|
+
## Add another AI tool
|
|
58
53
|
|
|
59
|
-
|
|
54
|
+
When a task needs another model family, tool reach or capacity, rerun the installer with `--level 2`. Verify that lane's access before assigning work.
|
|
60
55
|
{{ROUTE_GATE_SECTION}}
|
|
@@ -1,3 +1,3 @@
|
|
|
1
1
|
# templates/beginner/
|
|
2
2
|
|
|
3
|
-
Written at every level. `ORCHESTRATOR.md` is the single-agent routing document: tiers
|
|
3
|
+
Written at every level. `ORCHESTRATOR.md` is the single-agent routing document: planning, working and cheap model tiers; task classification; building from a shared context file and acceptance checks; handing off a task brief; tool fallbacks; and when to add another AI tool. At level 2 and up, `ROUTING.md` supersedes it.
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"checks": [
|
|
4
|
+
{
|
|
5
|
+
"id": "replace-this-example",
|
|
6
|
+
"source": "Replace with the user's quoted requirement",
|
|
7
|
+
"description": "Replace with a property of the final artifact and a command that verifies it",
|
|
8
|
+
"command": ["node", "-e", "process.exit(1)"],
|
|
9
|
+
"cwd": "."
|
|
10
|
+
}
|
|
11
|
+
]
|
|
12
|
+
}
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# Context file
|
|
2
|
+
|
|
3
|
+
## Request and approved scope
|
|
4
|
+
|
|
5
|
+
<Quote the user's request. Name the final artifact, authorized surfaces and non-goals.>
|
|
6
|
+
|
|
7
|
+
## Source of truth
|
|
8
|
+
|
|
9
|
+
<List relevant rules, specifications and source paths with a short note on what each establishes.>
|
|
10
|
+
|
|
11
|
+
## Current state
|
|
12
|
+
|
|
13
|
+
<Record the branch or revision, existing changes, baseline checks and live availability probes. Include the probe date and surface inspected.>
|
|
14
|
+
|
|
15
|
+
## Affected surfaces
|
|
16
|
+
|
|
17
|
+
<Name the files, interfaces and consumers this change touches. Record bounded search evidence.>
|
|
18
|
+
|
|
19
|
+
## Decisions and open questions
|
|
20
|
+
|
|
21
|
+
<Link the decision log. List resolved choices and any question that blocks dependent work.>
|
|
22
|
+
|
|
23
|
+
## Acceptance checks
|
|
24
|
+
|
|
25
|
+
<Point to the checks file. Include baseline results, required runtime access and assumptions that still need a spike.>
|
|
26
|
+
|
|
27
|
+
## Resources and order of work
|
|
28
|
+
|
|
29
|
+
<List available lanes and their reach, each section's owner, dependencies and steps skipped with reasons.>
|
|
30
|
+
|
|
31
|
+
## Measurements
|
|
32
|
+
|
|
33
|
+
<Give each runtime spike's value, method, script, sample size and date.>
|
|
34
|
+
|
|
35
|
+
## Verification and handoff
|
|
36
|
+
|
|
37
|
+
<Record final evidence and anything still unverified. Every task brief for this run reads this file.>
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# Decision log
|
|
2
|
+
|
|
3
|
+
When choosing an approach, append a concise decision record with evidence. Record the practical reason, not private chain-of-thought.
|
|
4
|
+
|
|
5
|
+
## <Date: decision title>
|
|
6
|
+
|
|
7
|
+
- **Did:** <the choice made>.
|
|
8
|
+
- **Why:** <the evidence and tradeoff that support it>.
|
|
9
|
+
- **Serves:** <the requirement or user outcome it serves>.
|
|
10
|
+
- **Rejected:** <alternatives considered and the reason each was set aside>.
|
|
11
|
+
- **Evidence:** <source path, command or measured result>.
|
|
@@ -1,9 +1,15 @@
|
|
|
1
1
|
# Your orchestrator (start here)
|
|
2
2
|
|
|
3
3
|
Installed {{DATE}} · level {{LEVEL_ID}}: **{{LEVEL_NAME}}**, {{LEVEL_TAGLINE}}
|
|
4
|
-
|
|
4
|
+
Main agent: **{{PRIMARY_NAME}}**
|
|
5
5
|
|
|
6
|
-
|
|
6
|
+
{{STACK_TABLE}}
|
|
7
|
+
|
|
8
|
+
{{STACK_FALLBACK_NOTE}}
|
|
9
|
+
|
|
10
|
+
{{STACK_GAPS}}
|
|
11
|
+
|
|
12
|
+
What each AI is:
|
|
7
13
|
{{AIS_LIST}}
|
|
8
14
|
|
|
9
15
|
Companion tools:
|
|
@@ -11,14 +17,18 @@ Companion tools:
|
|
|
11
17
|
|
|
12
18
|
## The idea in one line
|
|
13
19
|
|
|
14
|
-
This folder gives your agent routing instructions and, at level 2+, a runner for explicitly selected CLI lanes. The agent chooses the tier or lane;
|
|
20
|
+
This folder gives your agent routing instructions and, at level 2+, a runner for explicitly selected CLI lanes. The agent reads the rules and chooses the tier or lane; `aunx route` supplies a keyword suggestion and `aunx cli-run` runs the lane the caller selects.
|
|
15
21
|
|
|
16
|
-
##
|
|
22
|
+
## What's left for you
|
|
17
23
|
|
|
18
24
|
These are the same steps, in the same order, that the installer printed in your terminal.{{CHAT_UPLOAD_NOTE}}
|
|
19
25
|
|
|
20
26
|
{{ACTIVATION_STEPS}}
|
|
21
27
|
|
|
28
|
+
Interactive installs apply the main agent's rules and supported project settings after the single confirmation. Choose automatic activation in the edit screen or use `--no-apply` to leave activation for later. With `--yes`, add `--apply-snippets` to apply it. Selected companions are registered when the main agent has a cataloged project MCP config; other hosts keep a manual setup step.
|
|
29
|
+
|
|
30
|
+
The installer runs the doctor presence check automatically. Live canaries remain opt-in through `--doctor --run` at level 2 or above.
|
|
31
|
+
|
|
22
32
|
## Then prove it took
|
|
23
33
|
|
|
24
34
|
{{PROOF_STEPS}}
|
|
@@ -27,8 +37,11 @@ These are the same steps, in the same order, that the installer printed in your
|
|
|
27
37
|
|
|
28
38
|
| File | Read it when |
|
|
29
39
|
|---|---|
|
|
30
|
-
| `
|
|
31
|
-
| `
|
|
40
|
+
| `{{ROUTING_FILE}}` | First. The routing rules your main agent follows: tiers, task classes, acceptance checks and resource selection. |
|
|
41
|
+
| `CONTEXT.md` | At the start of a run. Shared source facts, scope, decisions and measurements (`aunx context`). |
|
|
42
|
+
| `ACCEPTANCE_CHECKS.json` | When verifying the final artifact (`aunx checks run`). Replace the failing example first. |
|
|
43
|
+
| `DECISIONS.md` | When choosing an approach. Did / Why / Serves / Rejected with evidence. |
|
|
44
|
+
| `TASK_BRIEF.md` | Before you hand any work to a subagent, a second CLI, or a chat window. The brief template. |
|
|
32
45
|
| `protocols/build-protocol.md` | You are about to build, code, migrate or deploy something. |
|
|
33
46
|
| `protocols/propagate.md` | You are renaming or changing a term, path, slug, schema field or routing rule. |
|
|
34
47
|
| `protocols/gap-analysis.md` | You just finished something comprehensive and want the second pass that hunts for what is missing. |
|
|
@@ -46,11 +59,11 @@ Level 2 adds `ROUTING.md`, `TIERS.md`, `DELEGATION_MATRIX.md`, `RESEARCH_TRIAGE.
|
|
|
46
59
|
|
|
47
60
|
{{LOAD_IT}}
|
|
48
61
|
|
|
49
|
-
##
|
|
62
|
+
## Route, check and verify
|
|
50
63
|
|
|
51
|
-
1.
|
|
52
|
-
2.
|
|
53
|
-
3.
|
|
64
|
+
1. When selecting a tier, use a planning model for ambiguity, a working model for execution and review, and a cheap model for mechanical work. Check current capabilities before dispatch.
|
|
65
|
+
2. When introducing a gate, demonstrate its failing case before relying on a passing result.
|
|
66
|
+
3. When a tool reports success, verify the requested artifact and acceptance checks.
|
|
54
67
|
|
|
55
68
|
## Where things went
|
|
56
69
|
|
|
@@ -58,4 +71,4 @@ Level 2 adds `ROUTING.md`, `TIERS.md`, `DELEGATION_MATRIX.md`, `RESEARCH_TRIAGE.
|
|
|
58
71
|
|
|
59
72
|
## Uninstall
|
|
60
73
|
|
|
61
|
-
|
|
74
|
+
When removing this installation, run the installer with `--uninstall` and the same `--dir` and `--project` paths. Use `--dry` first to inspect what would be removed. The manifest identifies managed files, applied rules blocks and added hook or MCP entries; edited entries are preserved and named. Backups stay beside changed files. Never delete shared agent folders that may contain unrelated files. Review any rules or settings you merged by hand and remove only this installation's entries. The local log at `~/.ai-orchestrator/cli-run.log.jsonl` is shared across installations; preserve it while another installation uses it.
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
# Task brief (`aunx brief`)
|
|
2
|
+
|
|
3
|
+
When handing work to an agent, CLI or fresh session, fill this brief and pass it with the relevant context. Keep every field; write `none` with a reason for an empty field. Treat capabilities as an explicit allow-list: absence is denial.
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
|
|
7
|
+
<Path to the run's context file, source files to read first, and facts that explain this task. Give a fresh session access to the actual contents.>
|
|
8
|
+
|
|
9
|
+
## The user's ask
|
|
10
|
+
|
|
11
|
+
> <Quote the user's acceptance-critical request. Distinguish a rule from a preference or an unresolved choice.>
|
|
12
|
+
|
|
13
|
+
## Purpose
|
|
14
|
+
|
|
15
|
+
<What this task is for and why.>
|
|
16
|
+
|
|
17
|
+
## Scope of build
|
|
18
|
+
|
|
19
|
+
- Finished artifact: <one line naming the final result>.
|
|
20
|
+
- Files and changes: <each path, its owner and required behavior>.
|
|
21
|
+
- Shared interfaces: <contracts every section depends on>.
|
|
22
|
+
- Order of work: <dependencies and next steps>.
|
|
23
|
+
- Non-goals: <work deliberately outside this task>.
|
|
24
|
+
- Interfaces not to break: <existing commands, formats, APIs and behaviors>.
|
|
25
|
+
|
|
26
|
+
For a non-build task, describe the bounded result here and mark build-only fields inapplicable with a reason.
|
|
27
|
+
|
|
28
|
+
## Task class
|
|
29
|
+
|
|
30
|
+
<read_only | draft_only | mutating>. For `draft_only`, write the proposed result to the named output and leave its destination unchanged.
|
|
31
|
+
|
|
32
|
+
## Granted scope
|
|
33
|
+
|
|
34
|
+
<Paths, topics, record sets, environments and write ownership. Everything outside this list is out of scope.>
|
|
35
|
+
|
|
36
|
+
## Capabilities
|
|
37
|
+
|
|
38
|
+
<Allowed actions and tools, including exact write destinations and permitted commands.>
|
|
39
|
+
|
|
40
|
+
## Denied actions
|
|
41
|
+
|
|
42
|
+
<Explicit prohibitions, such as publishing, sending, deleting, changing permissions or exposing secrets. Actions absent from Capabilities are denied.>
|
|
43
|
+
|
|
44
|
+
## Conventions
|
|
45
|
+
|
|
46
|
+
<Project rules the receiving session needs. Even a session that loads standing rules needs this task's scope and current decisions.>
|
|
47
|
+
|
|
48
|
+
## Acceptance checks
|
|
49
|
+
|
|
50
|
+
| Requirement quoted from the ask | Final property | Verifier command or manual procedure | Baseline result |
|
|
51
|
+
|---|---|---|---|
|
|
52
|
+
| <quote> | <observable property> | <command; exit 0 means satisfied> | <PASS / FAIL / UNVERIFIED> |
|
|
53
|
+
|
|
54
|
+
Include availability checks for required tools and access. Name the checks file when using `aunx checks run`. Replay against the final merged artifact and any materialized output whose properties may change.
|
|
55
|
+
|
|
56
|
+
## Resource inventory
|
|
57
|
+
|
|
58
|
+
- The lane **holds**: <probed tools, models, context window, permissions and runtime access>.
|
|
59
|
+
- The lane **lacks**: <needed capabilities absent here, and an authorized route to them if known>.
|
|
60
|
+
- Probe evidence: <command, date, result and the surface it actually inspected>.
|
|
61
|
+
|
|
62
|
+
## Measurements
|
|
63
|
+
|
|
64
|
+
<Runtime spikes, baseline counts or performance figures with method, script, sample size and date. Mark an unmeasured assumption UNVERIFIED.>
|
|
65
|
+
|
|
66
|
+
## Report contract
|
|
67
|
+
|
|
68
|
+
Return a coverage table with one row per requirement:
|
|
69
|
+
|
|
70
|
+
| Requirement | Status | Evidence |
|
|
71
|
+
|---|---|---|
|
|
72
|
+
| <requirement> | IMPLEMENTED / PARTIAL / MISSING / OUT-OF-SCOPE | <path, test or command output> |
|
|
73
|
+
|
|
74
|
+
Name changed files, verification results, unresolved decisions and unverified behavior. State what you did and did not do. Keep evidence separate from inference.
|
|
75
|
+
|
|
76
|
+
## Exit parameters
|
|
77
|
+
|
|
78
|
+
<A wall-clock ceiling, work ceiling or stop condition. For background work, name the five-minute heartbeat and the response to two checks without progress.>
|
|
79
|
+
|
|
80
|
+
When a bound is reached, stop and return the partial result with uncovered requirements named. When permission refuses a write, hand the patch to an authorized writer and continue independent work.
|
|
81
|
+
|
|
82
|
+
## Small read-only lookups
|
|
83
|
+
|
|
84
|
+
For a one-line lookup with no output artifact, state `Task brief: none (one-line lookup, read-only)` and its scope. For secrets, deletion, bulk mutation, deployment or someone else's data, use the full brief and the appropriate permission boundary.
|
|
@@ -1,15 +1,18 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Workflow playbooks
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
When a task matches a row, use that procedure and keep its evidence with the result.
|
|
4
4
|
|
|
5
|
-
| File |
|
|
5
|
+
| File | Use when | Result |
|
|
6
6
|
|---|---|---|
|
|
7
|
-
| `build-protocol.md` |
|
|
8
|
-
| `
|
|
9
|
-
| `
|
|
10
|
-
| `
|
|
11
|
-
| `
|
|
12
|
-
| `
|
|
13
|
-
| `
|
|
7
|
+
| `build-protocol.md` | Building, implementing, migrating or deploying | A scoped, checked change verified in use |
|
|
8
|
+
| `context-file.md` | Sharing build context across agents | One source file every brief reads |
|
|
9
|
+
| `acceptance-checks.md` | Turning requirements into final-artifact checks | Commands and explicit manual checks with PASS/FAIL evidence |
|
|
10
|
+
| `decision-log.md` | Choosing an approach or skipping a step | Did / Why / Serves / Rejected with evidence |
|
|
11
|
+
| `propagate.md` | Changing a shared name, path or convention | All affected surfaces updated and the old identifier checked |
|
|
12
|
+
| `gap-analysis.md` | Checking coverage against a request | Missing scope and evidence gaps reported |
|
|
13
|
+
| `deep-research.md` | Answering a question from an initially unknown source set | Bounded research with verified sources |
|
|
14
|
+
| `numbers-and-logic.md` | Reporting consequential numbers or logical claims | Computed results and method |
|
|
15
|
+
| `memory-and-record.md` | Writing durable information | A searchable, indexed record with one writer |
|
|
16
|
+
| `docs-then-prove.md` | Coding against a changing interface | Current documentation and runtime verification |
|
|
14
17
|
|
|
15
|
-
|
|
18
|
+
For lookups, prose edits, bulk classification and one-line configuration changes, use the relevant routing rule and direct verification. When an optional companion tool is absent, each affected procedure names an equivalent local workflow.
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# Acceptance checks (`aunx checks`)
|
|
2
|
+
|
|
3
|
+
- When a requirement determines acceptance, quote it and turn it into an observable property of the final artifact.
|
|
4
|
+
- Scaffold a checks file with `aunx checks`. Replace the intentionally failing example with a real verifier before using the file.
|
|
5
|
+
- Prefer an argv array for `command`, for example `["node", "--test", "test/example.test.js"]`. String commands run through the local shell, so review them as executable code before running a checks file.
|
|
6
|
+
- Set `cwd` relative to the checks file. Keep checks inside the task's authorized scope.
|
|
7
|
+
- Demonstrate a failing case for each new gate before trusting a passing result.
|
|
8
|
+
- Include availability facts when later work depends on a tool, permission or service remaining accessible.
|
|
9
|
+
- Run `aunx checks run ACCEPTANCE_CHECKS.json` against the final artifact. A failed command gives the gate exit code 1.
|
|
10
|
+
- When a check needs human judgment, record `manual: true` and its procedure in `description`. A manual or unverified check blocks the automated gate until replaced by a verifiable result; it is never silently counted as PASS.
|
|
11
|
+
- When packaging, rendering or deployment can change the property, replay the check against that materialized output too.
|
|
12
|
+
- When scope changes, preserve the original check and the decision that replaces it in the decision log.
|
|
13
|
+
|
|
14
|
+
When the command runner is unavailable, run each verifier with the project's local runtime and record its exit code. Optional companion tools can supply a verifier; their absence calls for an equivalent local check or an explicit UNVERIFIED result.
|