@codemeall/agent-fleet 0.1.0-preview.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/.claude-plugin/plugin.json +14 -0
  2. package/CHANGELOG.md +23 -0
  3. package/LICENSE +21 -0
  4. package/README.md +229 -0
  5. package/bin/build.js +5 -0
  6. package/bin/fleet.js +27 -0
  7. package/bin/setup.js +138 -0
  8. package/config.example.toml +21 -0
  9. package/dist/codex/agent-fleet/.codex-plugin/plugin.json +23 -0
  10. package/dist/codex/agent-fleet/skills/fleet/LICENSE +21 -0
  11. package/dist/codex/agent-fleet/skills/fleet/SKILL.md +82 -0
  12. package/dist/codex/agent-fleet/skills/fleet/agents/openai.yaml +6 -0
  13. package/dist/codex/agent-fleet/skills/fleet/bin/fleet +1028 -0
  14. package/dist/codex/agent-fleet/skills/fleet/providers.toml +106 -0
  15. package/dist/codex/agent-fleet/skills/fleet/references/harnesses.md +24 -0
  16. package/dist/codex/agent-fleet/skills/fleet/references/providers.md +31 -0
  17. package/dist/codex/agent-fleet/skills/fleet/references/routing.md +40 -0
  18. package/dist/codex/agent-fleet/skills/fleet/templates/preamble.md +16 -0
  19. package/dist/codex/agent-fleet/skills/fleet/templates/report.md +26 -0
  20. package/dist/codex/agent-fleet/skills/fleet/templates/review-prompt.md +14 -0
  21. package/dist/codex/agent-fleet/skills/fleet/templates/ticket-prompt.md +18 -0
  22. package/examples/README.md +58 -0
  23. package/examples/plan.json +12 -0
  24. package/examples/rules.md +25 -0
  25. package/examples/tickets/glossary.md +12 -0
  26. package/examples/tickets/guide.md +13 -0
  27. package/examples/tickets/overview.md +13 -0
  28. package/install.sh +9 -0
  29. package/package.json +63 -0
  30. package/plugin-manifests/codex.json +23 -0
  31. package/skill/LICENSE +21 -0
  32. package/skill/SKILL.md +83 -0
  33. package/skill/agents/openai.yaml +6 -0
  34. package/skill/bin/fleet +1028 -0
  35. package/skill/providers.toml +106 -0
  36. package/skill/references/harnesses.md +24 -0
  37. package/skill/references/providers.md +31 -0
  38. package/skill/references/routing.md +40 -0
  39. package/skill/templates/preamble.md +16 -0
  40. package/skill/templates/report.md +26 -0
  41. package/skill/templates/review-prompt.md +14 -0
  42. package/skill/templates/ticket-prompt.md +18 -0
@@ -0,0 +1,106 @@
1
+ # Provider adapters: the single source for how each agent CLI is launched,
2
+ # checked and stopped. ~/.config/agent-fleet/config.toml overrides any field.
3
+ #
4
+ # launch placeholders: {bin} is trusted shell setup; model/effort/prompt are quoted data.
5
+ # Shipped model IDs are account-dependent examples; verify them before routing.
6
+ # quit: sequence sent to the pane; "enter" and "ctrl+c" are keys, anything else is text.
7
+ # login: a cheap command that needs no model call; ready when its output contains login_ok.
8
+ # tiers: model + effort per ticket weight (see references/routing.md).
9
+ # family: model family, used to pick a reviewer from a different family.
10
+
11
+ [providers.claude]
12
+ enabled = true
13
+ bin = "claude"
14
+ launch = "{bin} --model {model} --effort {effort} {prompt}"
15
+ quit = ["/exit", "enter"]
16
+ login = "claude auth status"
17
+ login_ok = '"loggedIn": true'
18
+ family = "anthropic"
19
+ plugins = true
20
+ max = 2
21
+ tiers.heavy = { model = "opus", effort = "high" }
22
+ tiers.standard = { model = "sonnet", effort = "high" }
23
+ tiers.light = { model = "sonnet", effort = "medium" }
24
+ tiers.review = { model = "opus", effort = "high" }
25
+
26
+ [providers.claude-co]
27
+ enabled = false
28
+ bin = "CLAUDE_CONFIG_DIR=$HOME/.claude-co claude"
29
+ launch = "{bin} --model {model} --effort {effort} {prompt}"
30
+ quit = ["/exit", "enter"]
31
+ login = "CLAUDE_CONFIG_DIR=$HOME/.claude-co claude auth status"
32
+ login_ok = '"loggedIn": true'
33
+ family = "anthropic"
34
+ plugins = false
35
+ max = 2
36
+ tiers.heavy = { model = "opus", effort = "high" }
37
+ tiers.standard = { model = "sonnet", effort = "high" }
38
+ tiers.light = { model = "sonnet", effort = "medium" }
39
+ tiers.review = { model = "opus", effort = "high" }
40
+
41
+ [providers.codex]
42
+ enabled = true
43
+ bin = "codex"
44
+ launch = "{bin} -m {model} -c model_reasoning_effort={effort} -s workspace-write -a on-request {prompt}"
45
+ quit = ["/quit", "enter"]
46
+ login = "codex login status"
47
+ login_ok = "Logged in"
48
+ family = "openai"
49
+ plugins = false
50
+ max = 2
51
+ tiers.heavy = { model = "gpt-6-sol", effort = "high" }
52
+ tiers.standard = { model = "gpt-6-sol", effort = "medium" }
53
+ tiers.light = { model = "gpt-6-luna", effort = "high" }
54
+ tiers.review = { model = "gpt-6-sol", effort = "high" }
55
+
56
+ [providers.cursor]
57
+ enabled = true
58
+ bin = "cursor-agent"
59
+ # Cursor has no effort flag: effort is part of the model id.
60
+ launch = "{bin} --model {model} --trust {prompt}"
61
+ quit = ["/quit", "enter"]
62
+ login = "cursor-agent status"
63
+ login_ok = "Logged in"
64
+ models = "cursor-agent --list-models"
65
+ family = "mixed"
66
+ plugins = false
67
+ max = 2
68
+ tiers.heavy = { model = "grok-4.7-high", family = "xai" }
69
+ tiers.standard = { model = "kimi-k3-high", family = "moonshot" }
70
+ tiers.light = { model = "grok-4.7-high-fast", family = "xai" }
71
+ tiers.review = { model = "muse-spark-1.3-max", family = "meta" }
72
+
73
+ [providers.agy]
74
+ enabled = false
75
+ bin = "agy"
76
+ launch = "{bin} --model {model} --effort {effort} --mode accept-edits -i {prompt}"
77
+ quit = ["ctrl+c", "ctrl+c"]
78
+ login = "agy models"
79
+ login_ok = "gemini"
80
+ models = "agy models"
81
+ family = "mixed"
82
+ plugins = false
83
+ max = 1
84
+ tiers.heavy = { model = "gemini-3.1-pro-high", effort = "high", family = "google" }
85
+ tiers.standard = { model = "gemini-3.8-flash-high", effort = "high", family = "google" }
86
+ tiers.light = { model = "gemini-3.8-flash-medium", effort = "medium", family = "google" }
87
+ tiers.review = { model = "gemini-3.1-pro-high", effort = "high", family = "google" }
88
+
89
+ # Planned: turned on once the CLI is set up and its adapter is checked.
90
+ [providers.grok]
91
+ enabled = false
92
+ bin = "grok"
93
+ launch = "{bin} --model {model} {prompt}"
94
+ quit = ["/quit", "enter"]
95
+ family = "xai"
96
+ plugins = false
97
+ max = 1
98
+
99
+ [providers.muse]
100
+ enabled = false
101
+ bin = "muse"
102
+ launch = "{bin} --model {model} {prompt}"
103
+ quit = ["/quit", "enter"]
104
+ family = "meta"
105
+ plugins = false
106
+ max = 1
@@ -0,0 +1,24 @@
1
+ # Lead host requirements
2
+
3
+ The lead needs this skill, a local shell, access to the intended Git checkout, and permission to reach the cmux socket. Run `fleet doctor` in that same environment; a terminal outside the host is not proof that the host's sandbox can reach cmux.
4
+
5
+ | Host | Standalone invocation | User installation |
6
+ | --- | --- | --- |
7
+ | Claude Code | `/fleet …` | `~/.claude/skills/fleet` |
8
+ | Codex CLI / desktop | `$fleet …` or skill picker | `~/.agents/skills/fleet` |
9
+ | Cursor CLI / IDE | `/fleet …` or skill discovery | `~/.cursor/skills/fleet` |
10
+ | ChatGPT with local execution | Host-supported skill invocation | Requires a connection to the local runtime |
11
+
12
+ Install with `npx skills add codemeall/agent-fleet`, `npx @codemeall/agent-fleet@preview setup --harness <host>` or `./install.sh` from a clone; Claude Code can alternatively use the plugin marketplace (`/plugin marketplace add codemeall/agent-fleet`). Project setup uses the corresponding `.claude/skills`, `.agents/skills` or `.cursor/skills` beneath the repository. Codex's legacy `~/.codex/skills` and custom skill directories may still contain an older copy: remove stale duplicate registrations deliberately rather than installing multiple copies. An npm executable alone does not register the skill. A plugin may namespace the command; use the name shown by the host (Claude example: `/agent-fleet:fleet`).
13
+
14
+ ## Permission and connectivity checks
15
+
16
+ 1. Confirm Python 3.11+, Git, cmux and selected worker CLIs are available in the lead's shell.
17
+ 2. Confirm cmux is running and the intended workspace exists. Inside cmux the current workspace can be used; otherwise supply `fleet init <run> --workspace <ref>` explicitly.
18
+ 3. Run `fleet doctor`; diagnose its actual socket, executable or authentication error.
19
+ 4. If the host blocks a required operation, use its normal approval mechanism or have the owner configure a narrow allowance. Consult documentation for that installed host version. Do not disable the sandbox, switch permission modes or route to another harness to evade a rejection.
20
+ 5. Launch one small, scoped ticket and inspect it with `fleet peek` before starting a larger wave. Trust, authentication and model selection may still need owner input.
21
+
22
+ Claude permission classifiers, Codex sandbox profiles, Cursor allowances and remote-session topology vary by version. There is no universal setup flag that safely fixes all of them. Fleet's bundled launch configuration does not override host restrictions. A cloud-only chat can help plan the work but requires an explicit local execution connection to operate this fleet.
23
+
24
+ `fleet wait` defaults to 45 seconds and accepts at most 60 seconds. Use shorter waits when the host's command tool has a lower timeout. Provide progress updates between waits. A worker may use the same CLI as the lead; each tab is a distinct process and session.
@@ -0,0 +1,31 @@
1
+ # Provider adapters
2
+
3
+ `providers.toml` is the shipped adapter configuration; `fleet providers` prints it merged with personal overrides. Adapter command templates are trusted local executable configuration, not safe inputs from tickets. Model, effort and prompt substitutions are shell-quoted by the runtime. Keep account setup separate from model data.
4
+
5
+ All model IDs and efforts are examples tied to accounts and CLI versions. Before routing, check the local CLI's model list/help and authenticate through the owner's normal process. `doctor` checks binary presence and login signals, including the login command's exit status; it does not spend a model call or guarantee model access. A passing doctor is a preflight check, not an end-to-end compatibility claim.
6
+
7
+ ## Claude Code
8
+
9
+ The default `claude` adapter uses the normal account and interactive permission controls. `opus` and `sonnet` are example aliases. Effort support depends on the installed CLI and model. Do not respond to a permission denial by switching account or bypassing the host's controls.
10
+
11
+ `claude-co` is a disabled example of a second account using `CLAUDE_CONFIG_DIR=$HOME/.claude-co`. Enable it only after explicitly configuring and authenticating that account. It must receive a self-contained prompt; do not assume the same plugins exist under both configurations. Both accounts use Anthropic-family models, so switching between them does not satisfy cross-family review.
12
+
13
+ ## Codex
14
+
15
+ The example adapter uses `workspace-write` with `on-request` approval. Effective network and filesystem permissions still come from the host configuration; neither the adapter nor the preamble guarantees OS isolation. Fleet's worker contract prohibits network calls and package installation regardless of those capabilities. Request owner help for a necessary permission instead of weakening the sandbox.
16
+
17
+ Use model IDs and effort levels supported by the authenticated account. The shipped OpenAI IDs are examples. Model changes must be reflected in the plan and recorded worker metadata; do not silently switch a running worker's model in its UI and leave routing records stale.
18
+
19
+ ## Cursor
20
+
21
+ The default launches `cursor-agent` with the chosen model and repository trust, retaining interactive permission handling; it does not use `--force`. A trusted workspace is not permission for an arbitrary command. Inspect trust or command prompts as part of launch verification.
22
+
23
+ Cursor can serve several families. The example tiers identify xAI (`grok-…`), Moonshot (`kimi-…`) and Meta (`muse-…`) separately. Use `cursor-agent --list-models` to confirm your account's exact IDs; parameterized IDs must remain a single argument. A model override needs accurate family metadata, especially for cross-family review. A different CLI name alone is not evidence of a different family.
24
+
25
+ ## Optional and experimental adapters
26
+
27
+ Antigravity (`agy`) is disabled by default. Its example tiers describe Google models, but the adapter itself may serve multiple families. Validate its binary, authentication, model availability, command permissions and shutdown sequence before enabling it. No live compatibility claim is made.
28
+
29
+ The disabled `grok` and `muse` entries are incomplete placeholders, not supported providers. Verify that a binary with the expected name is actually the intended product; fill in a no-model authentication check, known family/tier metadata and tested launch/shutdown commands before use.
30
+
31
+ For every new adapter, test a harmless local ticket, multiple waves, a required review at capacity, process exit confirmation and recovery. Keep unavailable adapters out of auto routing. Login expiry or quota exhaustion can require owner intervention; they are not reasons to ignore the worker contract.
@@ -0,0 +1,40 @@
1
+ # Routing and review
2
+
3
+ For parameterized model IDs containing commas, select `single:<provider>` and set `model` in the JSON plan. The conversational `agents=` shorthand is comma-separated and cannot represent commas inside an ID. Quote the same full ID when passing `launch --model`.
4
+
5
+ The lead chooses assignments; Fleet enforces the saved plan. Persist the routing mode with `init --routing` and every ticket assignment with `plan --file`. Use the same provider, tier and explicit overrides at launch.
6
+
7
+ | Mode | Pool |
8
+ | --- | --- |
9
+ | `auto` | Enabled, authenticated adapters in `defaults.prefer` order |
10
+ | `single:<provider>` | Only the named adapter; tiers still select models |
11
+ | `agents=<provider>[:<model>],...` | Only listed adapters, with the listed model overriding tier models |
12
+
13
+ An unavailable explicit choice is a blocker to explain, not permission for a silent substitution. `prefer` is an editable ordering, not a claim about price, subscription quotas or model quality. All selected models must be checked against the account's availability.
14
+
15
+ ## Ticket tiers
16
+
17
+ | Tier | Use |
18
+ | --- | --- |
19
+ | `heavy` | Cross-module work, authentication, money, data integrity, schemas, new shared abstractions or consequential judgment |
20
+ | `standard` | A feature slice with clear criteria and established patterns |
21
+ | `light` | Mechanical, tightly scoped changes such as documentation or isolated repairs |
22
+ | `review` | Dedicated read-only review of a writer's captured diff |
23
+
24
+ Take consequential work first, balance ready providers without exceeding global or provider caps, and queue remaining tickets. Never raise caps merely to avoid waiting. Waves require satisfied blockers and disjoint exact file scopes. A `needs-verification` report does not free a slot: the process must exit.
25
+
26
+ ## Model family and review policy
27
+
28
+ `cross-heavy` (default) requires a different-family reviewer for each heavy implementation ticket. `cross-all` requires one for every ticket; `off` means lead verification alone. Record the chosen mode at run creation.
29
+
30
+ Family follows the resolved model, not the CLI. Use exact model-family mappings or matching tier metadata; a known single-family adapter can supply its family as a fallback. Mixed adapters require known model metadata. Use an explicit, accurate `family` in the plan or `--family` where necessary. Never invent a family solely to pass the gate.
31
+
32
+ If the allowed pool cannot provide a known different-family reviewer, tell the owner before implementation. Expand the pool under their routing instruction or obtain an explicit change to `review=off`; never silently downgrade the review policy. In particular, a second Claude account remains Anthropic, while two Cursor models may belong to different families.
33
+
34
+ Stop the writer before generating its scoped diff. Give the reviewer a separate ID, the original ticket and rules, that exact immutable diff, and only its report as a writable path. Review-only is a prompt contract in a shared checkout, not a security sandbox. Inspect for unauthorized changes. Stop and verify the reviewer before verifying the writer; dependents remain blocked throughout. If the writer's diff changes, redo review against the new diff.
35
+
36
+ ## Failures and plan changes
37
+
38
+ Before any worker launches, record an unavailable adapter and any owner-authorized substitution in the plan's decisions and update assignments explicitly. After the first launch the plan is immutable: use a new follow-up run for routing or scope changes, stopping all overlapping workers first and carrying forward pending dependencies, partial edits and context. A same-assignment repair may be relaunched only after the old process has exited. A missing tab is insufficient proof. Use `resume` to reconcile receipts, normal `stop` where possible, and evidence-backed `recover` only after independently verifying the process is gone.
39
+
40
+ Never change providers to circumvent a permission denial. Work requiring installs, secrets, network access or infrastructure changes falls outside the default worker contract; isolate and ask the owner to handle or explicitly authorize it separately.
@@ -0,0 +1,16 @@
1
+ # Fleet worker {{id}}
2
+
3
+ You are an implementation worker in the shared checkout at `{{repo}}`. The lead owns planning and verification, may message this pane, and will stop your process when your work is ready for review. The owner controls commits and publishing.
4
+
5
+ ## Hard rules
6
+
7
+ - Edit only the exact files listed in your scope and your assigned report. Other workers may be active. Preserve existing changes; do not overwrite or revert them. If you need another file, report the blocker and wait for a revised assignment.
8
+ - Use read-only Git (`status`, `diff`, `log`, `show`). No add, commit, stash, checkout, reset, branch or push. Do not invoke an upstream implementation workflow that includes those actions.
9
+ - Do not run build, dev, start or preview servers, or change shared build output. The lead handles builds using the repository's isolation rules.
10
+ - Never read `.env*` or other secrets. No cloud/database/deploy/network calls, migrations, schema generation, package installs or lockfile changes.
11
+ - Follow the host's approval boundaries. Do not bypass a denied command, weaken a sandbox, switch accounts or ask another worker to evade a denial.
12
+ - If blocked or unsure, write `Status: blocked` and the concrete question in your report, then stop making changes. Wait for the lead's decision.
13
+
14
+ ## Completion
15
+
16
+ Write `{{report}}` using the format below with exactly one `Status:` line. Use `needs-verification` when ready, then stop editing and wait. Do not mark yourself verified or resolve the source ticket. A report does not indicate process exit or lead acceptance.
@@ -0,0 +1,26 @@
1
+ ```markdown
2
+ # Report {{id}}
3
+
4
+ Status: needs-verification
5
+
6
+ ## Work or review performed
7
+ - Implementation: exact files changed and why; review: captured diff inspected.
8
+
9
+ ## Acceptance criteria
10
+ - [x] criterion: concrete test or inspection evidence
11
+ - [ ] criterion: remaining gap or why it could not be checked
12
+
13
+ ## Checks and evidence
14
+ - `command`: actual result; state when a check was not run
15
+
16
+ ## Findings
17
+ - Severity, file/location, failure scenario and suggested correction; or no actionable findings with review limits.
18
+
19
+ ## Decisions and blockers
20
+ - Decisions made, questions for the lead and any permission or scope blocker.
21
+
22
+ ## Out-of-scope observations
23
+ - Relevant issues left untouched.
24
+ ```
25
+
26
+ Use one status only: `needs-verification` or `blocked`. The lead may request changes, but only `fleet verify` records acceptance. Do not copy the illustrative checkboxes as evidence.
@@ -0,0 +1,14 @@
1
+ # Fleet reviewer {{id}}
2
+
3
+ Review writer `{{review_of}}` in `{{repo}}` against the local ticket `{{ticket}}` and the exact captured diff at `{{diff}}`.
4
+
5
+ ## Review contract
6
+
7
+ - You are a reviewer. Your only writable file is your report: `{{report}}`. Do not edit source, tickets, prompts, the captured diff, other reports or run state.
8
+ - Read the complete diff, ticket criteria, repository rules and relevant context. Evaluate the captured change; note any difference you observe between the checkout and the reviewed diff.
9
+ - Use only read-only inspection and explicitly authorized non-mutating checks. Do not run checks that generate files or caches. No Git mutations, package installs, network calls, secrets, deployments or permission bypasses.
10
+ - The shared checkout is not an isolated review sandbox. These are instruction boundaries; report unexpected concurrent changes and stop if they prevent reliable review.
11
+ - Report actionable findings with severity, file/location, concrete failure scenario and suggested correction. Separate confirmed defects from uncertainty. Do not fix findings yourself.
12
+ - Cover the ticket's acceptance criteria and testing gaps. If there are no actionable findings, say so and record what you inspected and any limitations; do not claim unrun checks passed.
13
+
14
+ Write the report below with `Status: needs-verification` when complete, or `Status: blocked` with the concrete blocker. Stop editing and wait for the lead. Only the lead can verify your report or accept the original ticket.
@@ -0,0 +1,18 @@
1
+ ## Your ticket
2
+
3
+ Read the local ticket: `{{ticket}}`.
4
+
5
+ <!-- LEAD: fill all fields before launch; preserve the exact saved scope. -->
6
+ **Files in scope:** {{files}}
7
+
8
+ **Context and decisions:** <!-- agreed spec and testing decisions; actual glossary/ADR paths; existing edits to preserve -->
9
+
10
+ **Checks to run:** <!-- exact bounded commands, working directory and success criteria; say explicitly if manual prose review is sufficient -->
11
+
12
+ ## How to work
13
+
14
+ 1. Read the ticket, project instructions, scoped code and the specified context. Resolve glossary paths from project configuration: current upstream conventions use `GLOSSARY.md` and optional `GLOSSARY-MAP.md`; older projects may use `CONTEXT.md`. Read relevant ADRs if they exist.
15
+ 2. Follow the agreed testing approach. For behavior changes, test meaningful acceptance criteria through public behavior where practical; do not invent redundant tests for prose or mechanical edits. Record visual/manual criteria honestly.
16
+ 3. Match existing naming and patterns. Keep changes inside scope and preserve pre-existing owner or other-worker edits.
17
+ 4. Run the exact permitted checks. Report failures and distinguish a demonstrated pre-existing failure from an assumption. Ask the lead about checks requiring forbidden operations.
18
+ 5. Review your scoped diff against every criterion and rule. Write your report, then stop editing. Fleet owns implementation under a no-commit contract; do not invoke upstream `/implement` unchanged.