@heretek-ai/epistemic-swarm 0.5.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/factory/SKILL.md +13 -0
- package/.omp/commands/domainexpansion.md +9 -0
- package/.omp/commands/factory.md +9 -0
- package/README.md +9 -0
- package/bin/cli.js +21 -0
- package/config/opencode-snippet.json +88 -1
- package/package.json +5 -3
- package/plugins/antigravity/skills/factory/SKILL.md +13 -0
- package/plugins/codex/skills/factory/SKILL.md +13 -0
- package/plugins/gemini/commands/domainexpansion.toml +7 -0
- package/plugins/gemini/commands/factory.toml +8 -0
- package/plugins/gemini/skills/factory/SKILL.md +13 -0
- package/plugins/opencode/index.js +35 -0
- package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
- package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
- package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
- package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
- package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
- package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
- package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
- package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
- package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
- package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
- package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
- package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
- package/runner/tests/test_factory.py +199 -0
- package/runner/tests/test_opencode_ux.py +8 -0
- package/runner/tests/test_swarm.py +2 -0
- package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
- package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
- package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
- package/scripts/build_adapters.py +7 -0
- package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
- package/skills/factory/SKILL.md +51 -0
- package/skills/factory/scripts/factory.py +212 -0
- package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
- package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Factory (thin adapter stub)
|
|
2
|
+
|
|
3
|
+
This file is a POINTER, not the implementation. It exists so harness skill
|
|
4
|
+
discovery finds an entry; the real skill lives in the IUMBTEMS repo.
|
|
5
|
+
|
|
6
|
+
- Canonical prose & scripts: `skills/factory/`
|
|
7
|
+
- Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
|
|
8
|
+
or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
|
|
9
|
+
- MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
|
|
10
|
+
|
|
11
|
+
Epistemic rules apply regardless of harness: tag claims as
|
|
12
|
+
`[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
|
|
13
|
+
`[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Autonomous agent-guided self-improvement loop (count-flagged)
|
|
3
|
+
---
|
|
4
|
+
|
|
5
|
+
Run the IUMBTEMS domain-expansion loop for `$1` loops (max 10):
|
|
6
|
+
|
|
7
|
+
Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill.
|
|
8
|
+
Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
|
|
9
|
+
Enforce via `python3 skills/factory/scripts/factory.py expansion --run <run> --loops $1`.
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Coding-factory Manager loop with grill-gated phased builds
|
|
3
|
+
---
|
|
4
|
+
|
|
5
|
+
Run the IUMBTEMS coding-factory Manager loop for `$1`:
|
|
6
|
+
|
|
7
|
+
1. Grill until `.factory/frontier.json` is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
|
|
8
|
+
2. Per gate: `python3 runner/research_swarm.py --mode brainstorm` plus `--mode darkharvest` (mock-first), then synthesize `.roadmap/<phase>/` GOAL.md + dossier.json.
|
|
9
|
+
3. Programmer subagent per phase; qa-a plus qa-b per phase; retries tracked via `python3 skills/factory/scripts/factory.py` (3 failures escalate).
|
package/README.md
CHANGED
|
@@ -149,6 +149,13 @@ iumbtems brainstorm "Where do we go from here?"
|
|
|
149
149
|
# Product competitor teardown with per-feature harvest verdicts
|
|
150
150
|
iumbtems darkharvest "Paseo-class agent harness competitor" --seeds https://github.com/a/b,https://github.com/c/d --max-repos 6 --mock-claude
|
|
151
151
|
|
|
152
|
+
# Coding-factory run-state helper (init, phase-add, qa-record, expansion, stop)
|
|
153
|
+
iumbtems factory init --run arena
|
|
154
|
+
iumbtems factory phase-add --run arena --phase 01-handoff --goal "Session handoff" --accept "round-trips;STOP kills loop"
|
|
155
|
+
iumbtems factory expansion --run arena --loops 10
|
|
156
|
+
|
|
157
|
+
# OpenCode slash commands (also Pi/OMP/Gemini): /factory, /domainexpansion, /darkharvest, /scout, /audit, /grill, /swarm
|
|
158
|
+
|
|
152
159
|
# Run Socratic grilling and decision frontier calculation
|
|
153
160
|
iumbtems grill --objective "L1 vs L2 state verification trade-offs"
|
|
154
161
|
|
|
@@ -164,6 +171,8 @@ iumbtems test
|
|
|
164
171
|
- **`/code-audit`**: Dialectic codebase review pairing a Structural Architect (thesis) with a Vulnerability Red-Teamer (antithesis) enforcing line-number proofs (`file:///path#L10-25`).
|
|
165
172
|
- **`/oss-scout`**: Evaluates GitHub repositories, package ecosystems (npm, crates.io, PyPI), license contamination (GPL/AGPL copyleft vs MIT/Apache), and outputs clean-room re-implementation blueprints.
|
|
166
173
|
- **`/darkharvest`**: Product competitor teardown (seed inspirations + prompt, expand to adjacents). Competitor × capability matrix, both-ways white-space gaps, per-feature `depend|vendor|clean-room-rebuild|skip` verdicts with SPDX attribution. Permissive-only vendoring; GPL/AGPL spec-rebuild only.
|
|
174
|
+
- **`/factory`**: Coding-factory Manager loop — grill-gated phased build (manager profile), per-phase programmer spawns, dual QA (3 retries then escalate), explicit sign-off per phase.
|
|
175
|
+
- **`/domainexpansion`**: Autonomous agent-guided self-improvement loop (`/domainexpansion <n>`, max 10); bypasses gates, stops on count OR `.factory/STOP` OR user kill.
|
|
167
176
|
- **`/grilling`**: Socratic assumption-inversion and Matt Pocock-style design tree frontier discovery.
|
|
168
177
|
- **`epistemic_search`**: Zero-key DuckDuckGo Lite search and content-addressed fetch with automatic SHA-256 caching.
|
|
169
178
|
|
package/bin/cli.js
CHANGED
|
@@ -50,6 +50,7 @@ Commands:
|
|
|
50
50
|
run "<objective>" Run the dialectic multi-agent research swarm
|
|
51
51
|
brainstorm "<prompt>" Run lateral brainstorming (feature vectors + spikes)
|
|
52
52
|
darkharvest "<arena>" Product competitor teardown with harvest verdicts
|
|
53
|
+
factory <subcommand> Factory run-state helper (init, phase-add, qa-record, expansion, stop)
|
|
53
54
|
grill Launch interactive Socratic decision tree framing
|
|
54
55
|
adapters Rebuild harness adapter mirrors (skills -> plugins/*, .agents)
|
|
55
56
|
install Install skills & MCP servers into ~/.claude/
|
|
@@ -82,6 +83,7 @@ Examples:
|
|
|
82
83
|
iumbtems run "Verify sub-millisecond ZK prover latency"
|
|
83
84
|
iumbtems brainstorm "Where do we go from here?"
|
|
84
85
|
iumbtems darkharvest "Paseo-class agent harness competitor" --seeds https://github.com/a/b,https://github.com/c/d --max-repos 6 --mock-claude
|
|
86
|
+
iumbtems factory init --run arena && iumbtems factory phase-add --run arena --phase 01-x --goal "..." --accept "a;b"
|
|
85
87
|
iumbtems grill --objective "Rollup architecture trade-offs"
|
|
86
88
|
iumbtems doctor
|
|
87
89
|
`);
|
|
@@ -174,6 +176,25 @@ switch (command) {
|
|
|
174
176
|
break;
|
|
175
177
|
}
|
|
176
178
|
|
|
179
|
+
case 'factory': {
|
|
180
|
+
// Factory run-state helper: forward subcommands to skills/factory/scripts/factory.py
|
|
181
|
+
// e.g. iumbtems factory init --run arena
|
|
182
|
+
// iumbtems factory qa-record --run arena --phase 01-x --seat qa-a --verdict pass
|
|
183
|
+
if (args[1] === '--help' || !args[1]) {
|
|
184
|
+
console.log([
|
|
185
|
+
'Usage: iumbtems factory <init|phase-add|qa-record|expansion|stop> [options]',
|
|
186
|
+
' init --run <name>',
|
|
187
|
+
' phase-add --run <name> --phase <id> --goal "<goal>" --accept "a;b"',
|
|
188
|
+
' qa-record --run <name> --phase <id> --seat <qa-a|qa-b> --verdict <pass|fail|conditional> [--reason "..."]',
|
|
189
|
+
' expansion --run <name> --loops <n> [--max-loops 10]',
|
|
190
|
+
' stop --run <name> (writes .factory/STOP kill-file)',
|
|
191
|
+
].join('\n'));
|
|
192
|
+
break;
|
|
193
|
+
}
|
|
194
|
+
runPython('skills/factory/scripts/factory.py', args.slice(1));
|
|
195
|
+
break;
|
|
196
|
+
}
|
|
197
|
+
|
|
177
198
|
case 'grill': {
|
|
178
199
|
runPython('skills/grilling/socratic_tree.py', args.slice(1));
|
|
179
200
|
break;
|
|
@@ -99,6 +99,92 @@
|
|
|
99
99
|
"iumbtems_verify_quote": true,
|
|
100
100
|
"iumbtems_config": true
|
|
101
101
|
}
|
|
102
|
+
},
|
|
103
|
+
"manager": {
|
|
104
|
+
"description": "IUMBTEMS Factory Manager: grill-gated phased builds. Owns gates, swarm dispatch, roadmap synthesis, QA tiebreaks. Never writes code.",
|
|
105
|
+
"mode": "primary",
|
|
106
|
+
"temperature": 0.2,
|
|
107
|
+
"permission": {
|
|
108
|
+
"edit": "deny",
|
|
109
|
+
"bash": "ask",
|
|
110
|
+
"task": {
|
|
111
|
+
"*": "deny",
|
|
112
|
+
"factory-*": "allow",
|
|
113
|
+
"programmer": "allow",
|
|
114
|
+
"qa-a": "allow",
|
|
115
|
+
"qa-b": "allow",
|
|
116
|
+
"brainstormer": "allow",
|
|
117
|
+
"darkharvester": "allow"
|
|
118
|
+
}
|
|
119
|
+
},
|
|
120
|
+
"tools": {
|
|
121
|
+
"read": true,
|
|
122
|
+
"write": true,
|
|
123
|
+
"grep": true,
|
|
124
|
+
"glob": true,
|
|
125
|
+
"task": true,
|
|
126
|
+
"iumbtems_brainstorm": true,
|
|
127
|
+
"iumbtems_darkharvest": true,
|
|
128
|
+
"iumbtems_verify_quote": true,
|
|
129
|
+
"iumbtems_socratic_frontier": true,
|
|
130
|
+
"iumbtems_config": true
|
|
131
|
+
}
|
|
132
|
+
},
|
|
133
|
+
"programmer": {
|
|
134
|
+
"description": "IUMBTEMS Factory Programmer: implements exactly one phase brief per spawn. Cites phase evidence hashes. Never invokes swarms or other programmers.",
|
|
135
|
+
"mode": "subagent",
|
|
136
|
+
"temperature": 0.3,
|
|
137
|
+
"permission": {
|
|
138
|
+
"edit": "allow",
|
|
139
|
+
"bash": "ask",
|
|
140
|
+
"task": {
|
|
141
|
+
"*": "deny"
|
|
142
|
+
}
|
|
143
|
+
},
|
|
144
|
+
"tools": {
|
|
145
|
+
"read": true,
|
|
146
|
+
"write": true,
|
|
147
|
+
"bash": true,
|
|
148
|
+
"grep": true,
|
|
149
|
+
"glob": true,
|
|
150
|
+
"iumbtems_verify_quote": true
|
|
151
|
+
}
|
|
152
|
+
},
|
|
153
|
+
"qa-a": {
|
|
154
|
+
"description": "IUMBTEMS Factory QA (functional): verifies phase acceptance criteria pass on the real surface. Read-only plus test execution. Diverged prompt from qa-b.",
|
|
155
|
+
"mode": "subagent",
|
|
156
|
+
"temperature": 0.1,
|
|
157
|
+
"permission": {
|
|
158
|
+
"edit": "deny",
|
|
159
|
+
"bash": "ask",
|
|
160
|
+
"task": {
|
|
161
|
+
"*": "deny"
|
|
162
|
+
}
|
|
163
|
+
},
|
|
164
|
+
"tools": {
|
|
165
|
+
"read": true,
|
|
166
|
+
"bash": true,
|
|
167
|
+
"grep": true,
|
|
168
|
+
"glob": true
|
|
169
|
+
}
|
|
170
|
+
},
|
|
171
|
+
"qa-b": {
|
|
172
|
+
"description": "IUMBTEMS Factory QA (adversarial): hunts edge cases, regressions, and acceptance loopholes the functional pass missed. Read-only plus test execution. Diverged prompt from qa-a.",
|
|
173
|
+
"mode": "subagent",
|
|
174
|
+
"temperature": 0.4,
|
|
175
|
+
"permission": {
|
|
176
|
+
"edit": "deny",
|
|
177
|
+
"bash": "ask",
|
|
178
|
+
"task": {
|
|
179
|
+
"*": "deny"
|
|
180
|
+
}
|
|
181
|
+
},
|
|
182
|
+
"tools": {
|
|
183
|
+
"read": true,
|
|
184
|
+
"bash": true,
|
|
185
|
+
"grep": true,
|
|
186
|
+
"glob": true
|
|
187
|
+
}
|
|
102
188
|
}
|
|
103
189
|
},
|
|
104
190
|
"skills": {
|
|
@@ -110,7 +196,8 @@
|
|
|
110
196
|
"./skills/code_audit",
|
|
111
197
|
"./skills/oss_scout",
|
|
112
198
|
"./skills/brainstorming",
|
|
113
|
-
"./skills/darkharvest"
|
|
199
|
+
"./skills/darkharvest",
|
|
200
|
+
"./skills/factory"
|
|
114
201
|
]
|
|
115
202
|
}
|
|
116
203
|
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@heretek-ai/epistemic-swarm",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.6.0",
|
|
4
4
|
"description": "IUMBTEMS: I Use My Brain To Express My Self — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi, OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
|
|
5
5
|
"main": "bin/cli.js",
|
|
6
6
|
"bin": {
|
|
@@ -80,7 +80,8 @@
|
|
|
80
80
|
"./skills/code_audit",
|
|
81
81
|
"./skills/oss_scout",
|
|
82
82
|
"./skills/brainstorming",
|
|
83
|
-
"./skills/darkharvest"
|
|
83
|
+
"./skills/darkharvest",
|
|
84
|
+
"./skills/factory"
|
|
84
85
|
],
|
|
85
86
|
"prompts": [
|
|
86
87
|
"./prompts/*.md"
|
|
@@ -98,7 +99,8 @@
|
|
|
98
99
|
"./skills/code_audit",
|
|
99
100
|
"./skills/oss_scout",
|
|
100
101
|
"./skills/brainstorming",
|
|
101
|
-
"./skills/darkharvest"
|
|
102
|
+
"./skills/darkharvest",
|
|
103
|
+
"./skills/factory"
|
|
102
104
|
],
|
|
103
105
|
"prompts": [
|
|
104
106
|
"./prompts/*.md"
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Factory (thin adapter stub)
|
|
2
|
+
|
|
3
|
+
This file is a POINTER, not the implementation. It exists so harness skill
|
|
4
|
+
discovery finds an entry; the real skill lives in the IUMBTEMS repo.
|
|
5
|
+
|
|
6
|
+
- Canonical prose & scripts: `skills/factory/`
|
|
7
|
+
- Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
|
|
8
|
+
or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
|
|
9
|
+
- MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
|
|
10
|
+
|
|
11
|
+
Epistemic rules apply regardless of harness: tag claims as
|
|
12
|
+
`[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
|
|
13
|
+
`[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Factory (thin adapter stub)
|
|
2
|
+
|
|
3
|
+
This file is a POINTER, not the implementation. It exists so harness skill
|
|
4
|
+
discovery finds an entry; the real skill lives in the IUMBTEMS repo.
|
|
5
|
+
|
|
6
|
+
- Canonical prose & scripts: `skills/factory/`
|
|
7
|
+
- Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
|
|
8
|
+
or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
|
|
9
|
+
- MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
|
|
10
|
+
|
|
11
|
+
Epistemic rules apply regardless of harness: tag claims as
|
|
12
|
+
`[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
|
|
13
|
+
`[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
description = "Autonomous agent-guided self-improvement loop (count-flagged)"
|
|
2
|
+
prompt = """Run the IUMBTEMS domain-expansion loop for {{args}} loops (max 10).
|
|
3
|
+
|
|
4
|
+
Bypasses per-loop gates; stops on count OR .factory/STOP file OR user kill.
|
|
5
|
+
Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
|
|
6
|
+
Enforce via: python3 skills/factory/scripts/factory.py expansion --run <run> --loops {{args}}.
|
|
7
|
+
"""
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
description = "Coding-factory Manager loop with grill-gated phased builds"
|
|
2
|
+
prompt = """Run the IUMBTEMS coding-factory Manager loop for: {{args}}.
|
|
3
|
+
|
|
4
|
+
1. Grill until .factory/frontier.json is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
|
|
5
|
+
2. Per gate: python3 runner/research_swarm.py --mode brainstorm plus --mode darkharvest (mock-first), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json.
|
|
6
|
+
3. Programmer subagent per phase; qa-a plus qa-b per phase; retries via python3 skills/factory/scripts/factory.py (3 failures escalate).
|
|
7
|
+
Report: .roadmap/ phases plus .factory/state.json.
|
|
8
|
+
"""
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
# Factory (thin adapter stub)
|
|
2
|
+
|
|
3
|
+
This file is a POINTER, not the implementation. It exists so harness skill
|
|
4
|
+
discovery finds an entry; the real skill lives in the IUMBTEMS repo.
|
|
5
|
+
|
|
6
|
+
- Canonical prose & scripts: `skills/factory/`
|
|
7
|
+
- Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
|
|
8
|
+
or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
|
|
9
|
+
- MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
|
|
10
|
+
|
|
11
|
+
Epistemic rules apply regardless of harness: tag claims as
|
|
12
|
+
`[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
|
|
13
|
+
`[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
|
|
@@ -512,6 +512,37 @@ export const OPENCODE_COMMANDS = [
|
|
|
512
512
|
'3. Legal rule: permissive-only vendor; GPL/AGPL clean-room-rebuild only; workflows clonable, assets never.',
|
|
513
513
|
].join('\n'),
|
|
514
514
|
},
|
|
515
|
+
{
|
|
516
|
+
name: 'factory',
|
|
517
|
+
description: 'Coding-factory Manager loop: grill-gated phased build with programmer spawns and dual QA',
|
|
518
|
+
usage: '/factory <product-arena>',
|
|
519
|
+
agent: 'manager',
|
|
520
|
+
subtask: false,
|
|
521
|
+
template: [
|
|
522
|
+
'Run the IUMBTEMS coding-factory Manager loop as the manager agent.',
|
|
523
|
+
'Arena: $ARGUMENTS',
|
|
524
|
+
'If $ARGUMENTS is empty, ask the user what to build first; never proceed on placeholder input.',
|
|
525
|
+
'1. Grill the user until .factory/frontier.json is settled (max 5 brainstorm+darkharvest swarm cycles per gate); explicit user approve advances each gate.',
|
|
526
|
+
'2. Per gate run iumbtems_brainstorm and iumbtems_darkharvest (mock_mode only for dry runs), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json (goal/evidence/acceptance/brief/verdict/hashes; every claim needs a VERIFIED hash).',
|
|
527
|
+
'3. Spawn the programmer subagent per phase with the phase dossier (cite phase hashes); run qa-a and qa-b (diverged prompts) per phase; track retries with skills/factory/scripts/factory.py (3 failures escalate to manager).',
|
|
528
|
+
'4. Manager tiebreaks QA disagreements; explicit user sign-off closes each phase.',
|
|
529
|
+
].join('\n'),
|
|
530
|
+
},
|
|
531
|
+
{
|
|
532
|
+
name: 'domainexpansion',
|
|
533
|
+
description: 'Autonomous agent-guided self-improvement loop over the codebase (count-flagged)',
|
|
534
|
+
usage: '/domainexpansion <n>',
|
|
535
|
+
agent: 'manager',
|
|
536
|
+
subtask: false,
|
|
537
|
+
template: [
|
|
538
|
+
'Run the IUMBTEMS domain-expansion loop as the manager agent.',
|
|
539
|
+
'Loops: $ARGUMENTS (integer count, max 10)',
|
|
540
|
+
'If $ARGUMENTS is not a positive integer, ask the user for the loop count first.',
|
|
541
|
+
'1. Bypass per-loop gates; stop on count OR .factory/STOP file OR user kill, whichever first (enforce via skills/factory/scripts/factory.py expansion).',
|
|
542
|
+
'2. Each loop: agents propose direction, quick iumbtems_brainstorm/iumbtems_darkharvest check, implement via programmer spawn, dual-QA verify.',
|
|
543
|
+
'3. All expansion proposals carry the strict VERIFIED evidence bar; log every loop to .factory/state.json.',
|
|
544
|
+
].join('\n'),
|
|
545
|
+
},
|
|
515
546
|
];
|
|
516
547
|
|
|
517
548
|
/**
|
|
@@ -527,6 +558,8 @@ export function commandCatalog() {
|
|
|
527
558
|
for (const cmd of OPENCODE_COMMANDS) {
|
|
528
559
|
if (!cmd?.name) continue;
|
|
529
560
|
out[cmd.name] = { description: cmd.description, template: cmd.template };
|
|
561
|
+
if (cmd.agent) out[cmd.name].agent = cmd.agent;
|
|
562
|
+
if (cmd.subtask !== undefined) out[cmd.name].subtask = cmd.subtask;
|
|
530
563
|
}
|
|
531
564
|
return out;
|
|
532
565
|
}
|
|
@@ -785,6 +818,8 @@ async function registerHostCommands(host) {
|
|
|
785
818
|
draft.add({
|
|
786
819
|
name: cmd.name,
|
|
787
820
|
description: cmd.description,
|
|
821
|
+
...(cmd.agent ? { agent: cmd.agent } : {}),
|
|
822
|
+
...(cmd.subtask !== undefined ? { subtask: cmd.subtask } : {}),
|
|
788
823
|
execute: async (input) => {
|
|
789
824
|
const args = input?.prompt?.text || '';
|
|
790
825
|
const prompt = (typeof input?.prompt === 'object' && input?.prompt !== null) ? input.prompt : {};
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Tests for the factory loop: run-state helper, QA retry bounds, plugin surface."""
|
|
3
|
+
|
|
4
|
+
import json
|
|
5
|
+
import subprocess
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
import unittest
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
|
|
12
|
+
if str(PROJECT_ROOT) not in sys.path:
|
|
13
|
+
sys.path.insert(0, str(PROJECT_ROOT))
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def run_node(code):
|
|
17
|
+
return subprocess.run(
|
|
18
|
+
["node", "--input-type=module", "-e", code],
|
|
19
|
+
capture_output=True,
|
|
20
|
+
text=True,
|
|
21
|
+
cwd=str(PROJECT_ROOT),
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def last_json_object(stdout):
|
|
26
|
+
lines = [l.strip() for l in stdout.strip().split("\n") if l.strip().startswith("{")]
|
|
27
|
+
return json.loads(lines[-1])
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class TestFactoryHelper(unittest.TestCase):
|
|
31
|
+
def test_qa_retry_escalates_after_three(self):
|
|
32
|
+
import shutil
|
|
33
|
+
|
|
34
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
35
|
+
# Redirect factory state + roadmap into tmp via cwd-independent paths:
|
|
36
|
+
# factory.py anchors to PROJECT_ROOT, so sandbox by copying is
|
|
37
|
+
# overkill — exercise the pure bound logic through CLI on a run
|
|
38
|
+
# name, then clean up the created dirs.
|
|
39
|
+
run = "test-escalation-run"
|
|
40
|
+
try:
|
|
41
|
+
subprocess.run(
|
|
42
|
+
[
|
|
43
|
+
sys.executable,
|
|
44
|
+
"skills/factory/scripts/factory.py",
|
|
45
|
+
"init",
|
|
46
|
+
"--run",
|
|
47
|
+
run,
|
|
48
|
+
],
|
|
49
|
+
check=True,
|
|
50
|
+
capture_output=True,
|
|
51
|
+
cwd=str(PROJECT_ROOT),
|
|
52
|
+
)
|
|
53
|
+
subprocess.run(
|
|
54
|
+
[
|
|
55
|
+
sys.executable,
|
|
56
|
+
"skills/factory/scripts/factory.py",
|
|
57
|
+
"phase-add",
|
|
58
|
+
"--run",
|
|
59
|
+
run,
|
|
60
|
+
"--phase",
|
|
61
|
+
"01-x",
|
|
62
|
+
"--goal",
|
|
63
|
+
"g",
|
|
64
|
+
"--accept",
|
|
65
|
+
"a",
|
|
66
|
+
],
|
|
67
|
+
check=True,
|
|
68
|
+
capture_output=True,
|
|
69
|
+
cwd=str(PROJECT_ROOT),
|
|
70
|
+
)
|
|
71
|
+
codes = []
|
|
72
|
+
for _ in range(3):
|
|
73
|
+
r = subprocess.run(
|
|
74
|
+
[
|
|
75
|
+
sys.executable,
|
|
76
|
+
"skills/factory/scripts/factory.py",
|
|
77
|
+
"qa-record",
|
|
78
|
+
"--run",
|
|
79
|
+
run,
|
|
80
|
+
"--phase",
|
|
81
|
+
"01-x",
|
|
82
|
+
"--seat",
|
|
83
|
+
"qa-a",
|
|
84
|
+
"--verdict",
|
|
85
|
+
"fail",
|
|
86
|
+
"--reason",
|
|
87
|
+
"t",
|
|
88
|
+
],
|
|
89
|
+
capture_output=True,
|
|
90
|
+
cwd=str(PROJECT_ROOT),
|
|
91
|
+
)
|
|
92
|
+
codes.append(r.returncode)
|
|
93
|
+
self.assertEqual(codes, [0, 0, 2]) # 3rd failure escalates
|
|
94
|
+
state = json.loads(
|
|
95
|
+
(PROJECT_ROOT / ".factory" / run / "state.json").read_text()
|
|
96
|
+
)
|
|
97
|
+
self.assertEqual(state["phases"]["01-x"]["status"], "escalated")
|
|
98
|
+
finally:
|
|
99
|
+
shutil.rmtree(PROJECT_ROOT / ".factory" / run, ignore_errors=True)
|
|
100
|
+
shutil.rmtree(PROJECT_ROOT / ".roadmap" / "01-x", ignore_errors=True)
|
|
101
|
+
|
|
102
|
+
def test_expansion_respects_stop_file(self):
|
|
103
|
+
import shutil
|
|
104
|
+
|
|
105
|
+
run = "test-expansion-run"
|
|
106
|
+
try:
|
|
107
|
+
subprocess.run(
|
|
108
|
+
[
|
|
109
|
+
sys.executable,
|
|
110
|
+
"skills/factory/scripts/factory.py",
|
|
111
|
+
"init",
|
|
112
|
+
"--run",
|
|
113
|
+
run,
|
|
114
|
+
],
|
|
115
|
+
check=True,
|
|
116
|
+
capture_output=True,
|
|
117
|
+
cwd=str(PROJECT_ROOT),
|
|
118
|
+
)
|
|
119
|
+
subprocess.run(
|
|
120
|
+
[
|
|
121
|
+
sys.executable,
|
|
122
|
+
"skills/factory/scripts/factory.py",
|
|
123
|
+
"stop",
|
|
124
|
+
"--run",
|
|
125
|
+
run,
|
|
126
|
+
],
|
|
127
|
+
check=True,
|
|
128
|
+
capture_output=True,
|
|
129
|
+
cwd=str(PROJECT_ROOT),
|
|
130
|
+
)
|
|
131
|
+
r = subprocess.run(
|
|
132
|
+
[
|
|
133
|
+
sys.executable,
|
|
134
|
+
"skills/factory/scripts/factory.py",
|
|
135
|
+
"expansion",
|
|
136
|
+
"--run",
|
|
137
|
+
run,
|
|
138
|
+
"--loops",
|
|
139
|
+
"10",
|
|
140
|
+
],
|
|
141
|
+
capture_output=True,
|
|
142
|
+
text=True,
|
|
143
|
+
cwd=str(PROJECT_ROOT),
|
|
144
|
+
)
|
|
145
|
+
self.assertEqual(r.returncode, 0)
|
|
146
|
+
self.assertIn("halting after 0/10", r.stdout)
|
|
147
|
+
finally:
|
|
148
|
+
shutil.rmtree(PROJECT_ROOT / ".factory" / run, ignore_errors=True)
|
|
149
|
+
|
|
150
|
+
def test_skill_and_commands_exist(self):
|
|
151
|
+
self.assertTrue((PROJECT_ROOT / "skills" / "factory" / "SKILL.md").exists())
|
|
152
|
+
self.assertTrue(
|
|
153
|
+
(PROJECT_ROOT / "skills" / "factory" / "scripts" / "factory.py").exists()
|
|
154
|
+
)
|
|
155
|
+
self.assertTrue((PROJECT_ROOT / ".omp" / "commands" / "factory.md").exists())
|
|
156
|
+
self.assertTrue(
|
|
157
|
+
(PROJECT_ROOT / ".omp" / "commands" / "domainexpansion.md").exists()
|
|
158
|
+
)
|
|
159
|
+
self.assertTrue(
|
|
160
|
+
(PROJECT_ROOT / "plugins" / "gemini" / "commands" / "factory.toml").exists()
|
|
161
|
+
)
|
|
162
|
+
with open(PROJECT_ROOT / "package.json") as f:
|
|
163
|
+
pkg = json.load(f)
|
|
164
|
+
self.assertIn("./skills/factory", pkg["pi"]["skills"])
|
|
165
|
+
self.assertIn("./skills/factory", pkg["omp"]["skills"])
|
|
166
|
+
|
|
167
|
+
def test_snippet_factory_roster(self):
|
|
168
|
+
with open(PROJECT_ROOT / "config" / "opencode-snippet.json") as f:
|
|
169
|
+
snippet = json.load(f)
|
|
170
|
+
for agent in ("manager", "programmer", "qa-a", "qa-b"):
|
|
171
|
+
self.assertIn(agent, snippet["agent"])
|
|
172
|
+
manager = snippet["agent"]["manager"]
|
|
173
|
+
self.assertEqual(manager["mode"], "primary")
|
|
174
|
+
task = manager["permission"]["task"]
|
|
175
|
+
self.assertEqual(task["*"], "deny")
|
|
176
|
+
for allowed in ("programmer", "qa-a", "qa-b", "brainstormer", "darkharvester"):
|
|
177
|
+
self.assertEqual(task[allowed], "allow")
|
|
178
|
+
self.assertEqual(snippet["agent"]["qa-a"]["permission"]["edit"], "deny")
|
|
179
|
+
self.assertEqual(snippet["agent"]["qa-b"]["permission"]["edit"], "deny")
|
|
180
|
+
|
|
181
|
+
def test_factory_commands_carry_agent_subtask(self):
|
|
182
|
+
res = run_node(
|
|
183
|
+
"""
|
|
184
|
+
import { OPENCODE_COMMANDS, commandCatalog } from "./plugins/opencode/index.js";
|
|
185
|
+
const catalog = commandCatalog();
|
|
186
|
+
console.log(JSON.stringify({
|
|
187
|
+
factory: catalog.factory, expansion: catalog.domainexpansion
|
|
188
|
+
}));
|
|
189
|
+
"""
|
|
190
|
+
)
|
|
191
|
+
self.assertEqual(res.returncode, 0, res.stderr)
|
|
192
|
+
data = last_json_object(res.stdout)
|
|
193
|
+
self.assertEqual(data["factory"]["agent"], "manager")
|
|
194
|
+
self.assertIn("$ARGUMENTS", data["factory"]["template"])
|
|
195
|
+
self.assertEqual(data["expansion"]["agent"], "manager")
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
if __name__ == "__main__":
|
|
199
|
+
unittest.main()
|
|
@@ -26,6 +26,8 @@ EXPECTED_COMMANDS = [
|
|
|
26
26
|
"brainstorming",
|
|
27
27
|
"brainstorm",
|
|
28
28
|
"darkharvest",
|
|
29
|
+
"factory",
|
|
30
|
+
"domainexpansion",
|
|
29
31
|
]
|
|
30
32
|
|
|
31
33
|
|
|
@@ -282,6 +284,10 @@ class TestInstallOpenCode(unittest.TestCase):
|
|
|
282
284
|
"epistemic-auditor",
|
|
283
285
|
"brainstormer",
|
|
284
286
|
"darkharvester",
|
|
287
|
+
"manager",
|
|
288
|
+
"programmer",
|
|
289
|
+
"qa-a",
|
|
290
|
+
"qa-b",
|
|
285
291
|
]:
|
|
286
292
|
self.assertIn(agent, cfg.get("agent", {}))
|
|
287
293
|
self.assertIn("iumbtems", cfg.get("mcp", {}))
|
|
@@ -606,6 +612,8 @@ class TestOpenCodeV2Transforms(unittest.TestCase):
|
|
|
606
612
|
"brainstorming",
|
|
607
613
|
"brainstorm",
|
|
608
614
|
"darkharvest",
|
|
615
|
+
"factory",
|
|
616
|
+
"domainexpansion",
|
|
609
617
|
]:
|
|
610
618
|
self.assertIn(c, data["commands"])
|
|
611
619
|
self.assertNotIn("goal", data["commands"])
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
@@ -40,6 +40,7 @@ CANONICAL_SKILLS = [
|
|
|
40
40
|
"oss_scout",
|
|
41
41
|
"brainstorming",
|
|
42
42
|
"darkharvest",
|
|
43
|
+
"factory",
|
|
43
44
|
]
|
|
44
45
|
|
|
45
46
|
# Skill -> the MCP tool(s) that now carry its programmatic surface.
|
|
@@ -52,6 +53,11 @@ SKILL_TOOLS = {
|
|
|
52
53
|
"oss_scout": ["iumbtems_oss_scout"],
|
|
53
54
|
"brainstorming": ["iumbtems_brainstorm"],
|
|
54
55
|
"darkharvest": ["iumbtems_darkharvest"],
|
|
56
|
+
"factory": [
|
|
57
|
+
"iumbtems_brainstorm",
|
|
58
|
+
"iumbtems_darkharvest",
|
|
59
|
+
"iumbtems_socratic_frontier",
|
|
60
|
+
],
|
|
55
61
|
}
|
|
56
62
|
|
|
57
63
|
SKILL_TITLES = {
|
|
@@ -63,6 +69,7 @@ SKILL_TITLES = {
|
|
|
63
69
|
"oss_scout": "OSS Scout",
|
|
64
70
|
"brainstorming": "Brainstorming",
|
|
65
71
|
"darkharvest": "Darkharvest",
|
|
72
|
+
"factory": "Factory",
|
|
66
73
|
}
|
|
67
74
|
|
|
68
75
|
# (target dir relative to root, mode). All targets are stubs since A2b.
|
|
Binary file
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: factory
|
|
3
|
+
description: Coding-factory Manager loop. Use when user invokes /factory or /domainexpansion, or wants grill-gated phased builds with programmer spawns and dual QA. Manager grills until frontier settled, runs brainstorm plus darkharvest swarms per gate, synthesizes .roadmap phases, spawns programmer per phase, and enforces dual-QA retry bounds. Never writes code itself; never bypasses explicit user sign-off (except inside /domainexpansion count).
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# Factory — Manager Loop & Domain Expansion
|
|
7
|
+
|
|
8
|
+
## 1. Roles (OpenCode profiles in `config/opencode-snippet.json`)
|
|
9
|
+
|
|
10
|
+
- **manager** (primary): owns gates, grilling, swarm dispatch, roadmap synthesis, tiebreaks.
|
|
11
|
+
Task allowlist: `factory-*`, `programmer`, `qa-a`, `qa-b`, `brainstormer`, `darkharvester` deny `*` otherwise.
|
|
12
|
+
- **programmer** (subagent): implements exactly one phase brief. May call `iumbtems_verify_quote`; may NOT invoke swarms or other programmers.
|
|
13
|
+
- **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
|
|
14
|
+
- **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
|
|
15
|
+
|
|
16
|
+
## 2. Gate protocol (max 5 swarm cycles per gate)
|
|
17
|
+
|
|
18
|
+
1. Grill until `.factory/frontier.json` settled (grilling skill).
|
|
19
|
+
2. Run brainstorm + darkharvest swarms (mock-first on fixtures).
|
|
20
|
+
3. Manager synthesizes `.roadmap/<phase>/` (GOAL.md + dossier.json).
|
|
21
|
+
4. Explicit user `approve` advances; anything else regrills (cycle counter in `.factory/state.json`).
|
|
22
|
+
|
|
23
|
+
## 3. Phase contract (`.roadmap/<phase>/`)
|
|
24
|
+
|
|
25
|
+
- `GOAL.md`: human-readable goal + acceptance criteria.
|
|
26
|
+
- `dossier.json`: `{phase, goal, evidence:[{hash, quote}], acceptance[], brief, verdict, hashes}`. Every harvest/claim entry needs a SHA-256 source hash or `file://` pointer or it is purged to NEGATIVE_KNOWLEDGE.
|
|
27
|
+
- Programmer receives the phase dossier (Manager chooses freeform vs strict brief but MUST cite phase hashes).
|
|
28
|
+
|
|
29
|
+
## 4. QA protocol (3 retries, then escalate)
|
|
30
|
+
|
|
31
|
+
- Both QA seats run per phase; disagreements go to manager tiebreak.
|
|
32
|
+
- `skills/factory/scripts/factory.py` tracks `qa_retries` per phase in `.factory/state.json`. On 3rd rejection: halt phase, return to manager with both QA reports (retry loop per plan; manager may regrill scope or escalate to user).
|
|
33
|
+
- QA verdicts: `pass | fail(reason) | conditional(note)`.
|
|
34
|
+
|
|
35
|
+
## 5. Domain expansion (`/domainexpansion <n>`)
|
|
36
|
+
|
|
37
|
+
- Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill, whichever first.
|
|
38
|
+
- Each loop: agents propose direction → quick swarm check → implement → QA → next.
|
|
39
|
+
- `factory.py` enforces: refuse `n < 1`, cap `n` at `--max-loops` default 10, check STOP file before every loop.
|
|
40
|
+
|
|
41
|
+
## 6. State layout (three dirs, distinct jobs)
|
|
42
|
+
|
|
43
|
+
- `.factory/`: run state (`state.json`, `frontier.json`, `STOP` kill-file). Gitignored runtime state.
|
|
44
|
+
- `.roadmap/`: output (phase dirs). Committed.
|
|
45
|
+
- `.research/`: evidence (swarm dossiers, source cache). Gitignored (existing rule).
|
|
46
|
+
|
|
47
|
+
## 7. Anti-patterns
|
|
48
|
+
|
|
49
|
+
- No code writes by manager; no swarm invocation by programmer; no edits by QA.
|
|
50
|
+
- No gate bypass outside `/domainexpansion`; no uncapped loops.
|
|
51
|
+
- No VERIFIED claims without hashes, even in expansion proposals.
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""
|
|
3
|
+
Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
|
|
4
|
+
|
|
5
|
+
All state lives under .factory/ (gitignored runtime state). Phase output goes
|
|
6
|
+
to .roadmap/<phase>/. Evidence stays in .research/. Read-only w.r.t. repo code.
|
|
7
|
+
|
|
8
|
+
Usage:
|
|
9
|
+
python3 skills/factory/scripts/factory.py init --run <name>
|
|
10
|
+
python3 skills/factory/scripts/factory.py phase-add --run <name> --phase 01-auth --goal "..." --accept "..."
|
|
11
|
+
python3 skills/factory/scripts/factory.py qa-record --run <name> --phase 01-auth --seat qa-a --verdict fail --reason "..."
|
|
12
|
+
python3 skills/factory/scripts/factory.py expansion --run <name> --loops 10 [--max-loops 10]
|
|
13
|
+
python3 skills/factory/scripts/factory.py stop --run <name> # write STOP kill-file
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
import argparse
|
|
17
|
+
import json
|
|
18
|
+
import sys
|
|
19
|
+
from datetime import datetime, timezone
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
|
|
23
|
+
FACTORY_DIR = PROJECT_ROOT / ".factory"
|
|
24
|
+
ROADMAP_DIR = PROJECT_ROOT / ".roadmap"
|
|
25
|
+
|
|
26
|
+
MAX_QA_RETRIES = 3
|
|
27
|
+
MAX_EXPANSION_LOOPS = 10
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def now():
|
|
31
|
+
return datetime.now(timezone.utc).isoformat()
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def run_dir(run):
|
|
35
|
+
return FACTORY_DIR / run
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def load_state(run):
|
|
39
|
+
p = run_dir(run) / "state.json"
|
|
40
|
+
if not p.exists():
|
|
41
|
+
raise SystemExit(f"No factory run '{run}'. Run `factory.py init` first.")
|
|
42
|
+
return json.loads(p.read_text(encoding="utf-8"))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def save_state(run, state):
|
|
46
|
+
p = run_dir(run) / "state.json"
|
|
47
|
+
p.write_text(json.dumps(state, indent=2), encoding="utf-8")
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def cmd_init(args):
|
|
51
|
+
d = run_dir(args.run)
|
|
52
|
+
d.mkdir(parents=True, exist_ok=True)
|
|
53
|
+
state = {
|
|
54
|
+
"run": args.run,
|
|
55
|
+
"created_at": now(),
|
|
56
|
+
"gate_cycles": 0,
|
|
57
|
+
"phases": {},
|
|
58
|
+
"expansion": {"loops_done": 0, "loops_planned": 0},
|
|
59
|
+
}
|
|
60
|
+
save_state(args.run, state)
|
|
61
|
+
print(f"✅ Factory run '{args.run}' initialised at {d}")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def cmd_phase_add(args):
|
|
65
|
+
state = load_state(args.run)
|
|
66
|
+
if args.phase in state["phases"]:
|
|
67
|
+
raise SystemExit(f"Phase '{args.phase}' already exists.")
|
|
68
|
+
state["phases"][args.phase] = {
|
|
69
|
+
"goal": args.goal,
|
|
70
|
+
"acceptance": [a.strip() for a in args.accept.split(";") if a.strip()],
|
|
71
|
+
"status": "briefed",
|
|
72
|
+
"qa_retries": 0,
|
|
73
|
+
"qa_reports": [],
|
|
74
|
+
}
|
|
75
|
+
save_state(args.run, state)
|
|
76
|
+
# Phase output: GOAL.md + dossier.json skeleton (evidence filled by manager).
|
|
77
|
+
phase_dir = ROADMAP_DIR / args.phase
|
|
78
|
+
phase_dir.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
(phase_dir / "GOAL.md").write_text(
|
|
80
|
+
f"# {args.phase}: {args.goal}\n\n## Acceptance\n"
|
|
81
|
+
+ "".join(f"- [ ] {a}\n" for a in state["phases"][args.phase]["acceptance"]),
|
|
82
|
+
encoding="utf-8",
|
|
83
|
+
)
|
|
84
|
+
(phase_dir / "dossier.json").write_text(
|
|
85
|
+
json.dumps(
|
|
86
|
+
{
|
|
87
|
+
"phase": args.phase,
|
|
88
|
+
"goal": args.goal,
|
|
89
|
+
"evidence": [],
|
|
90
|
+
"acceptance": state["phases"][args.phase]["acceptance"],
|
|
91
|
+
"brief": "",
|
|
92
|
+
"verdict": "briefed",
|
|
93
|
+
"hashes": [],
|
|
94
|
+
},
|
|
95
|
+
indent=2,
|
|
96
|
+
),
|
|
97
|
+
encoding="utf-8",
|
|
98
|
+
)
|
|
99
|
+
print(f"✅ Phase '{args.phase}' briefed → {phase_dir}")
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def cmd_qa_record(args):
|
|
103
|
+
state = load_state(args.run)
|
|
104
|
+
phase = state["phases"].get(args.phase)
|
|
105
|
+
if phase is None:
|
|
106
|
+
raise SystemExit(f"Unknown phase '{args.phase}'.")
|
|
107
|
+
if args.verdict not in ("pass", "fail", "conditional"):
|
|
108
|
+
raise SystemExit("verdict must be pass|fail|conditional.")
|
|
109
|
+
phase["qa_reports"].append(
|
|
110
|
+
{
|
|
111
|
+
"seat": args.seat,
|
|
112
|
+
"verdict": args.verdict,
|
|
113
|
+
"reason": args.reason or "",
|
|
114
|
+
"at": now(),
|
|
115
|
+
}
|
|
116
|
+
)
|
|
117
|
+
if args.verdict == "fail":
|
|
118
|
+
phase["qa_retries"] += 1
|
|
119
|
+
if phase["qa_retries"] >= MAX_QA_RETRIES:
|
|
120
|
+
phase["status"] = "escalated"
|
|
121
|
+
save_state(args.run, state)
|
|
122
|
+
print(
|
|
123
|
+
f"🛑 Phase '{args.phase}' ESCALATED after "
|
|
124
|
+
f"{MAX_QA_RETRIES} QA failures — back to manager."
|
|
125
|
+
)
|
|
126
|
+
return 2
|
|
127
|
+
phase["status"] = "retrying"
|
|
128
|
+
elif args.verdict == "pass":
|
|
129
|
+
phase["status"] = "signed-off"
|
|
130
|
+
else:
|
|
131
|
+
phase["status"] = "conditional"
|
|
132
|
+
save_state(args.run, state)
|
|
133
|
+
print(
|
|
134
|
+
f"✅ QA recorded: {args.phase} [{args.seat}] → {args.verdict} "
|
|
135
|
+
f"(status={phase['status']}, retries={phase['qa_retries']})"
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def stop_requested(run):
|
|
140
|
+
return (run_dir(run) / "STOP").exists()
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def cmd_expansion(args):
|
|
144
|
+
state = load_state(args.run)
|
|
145
|
+
n = args.loops
|
|
146
|
+
if n < 1:
|
|
147
|
+
raise SystemExit("loops must be >= 1.")
|
|
148
|
+
cap = args.max_loops or MAX_EXPANSION_LOOPS
|
|
149
|
+
n = min(n, cap)
|
|
150
|
+
state["expansion"]["loops_planned"] = n
|
|
151
|
+
save_state(args.run, state)
|
|
152
|
+
done = 0
|
|
153
|
+
for i in range(1, n + 1):
|
|
154
|
+
if stop_requested(args.run):
|
|
155
|
+
print(f"🛑 STOP file present — halting after {done}/{n} loops.")
|
|
156
|
+
break
|
|
157
|
+
done += 1
|
|
158
|
+
print(
|
|
159
|
+
f"🔁 Expansion loop {done}/{n}: agents propose → swarm check → "
|
|
160
|
+
f"implement → QA (Manager drives each step)."
|
|
161
|
+
)
|
|
162
|
+
state = load_state(args.run)
|
|
163
|
+
state["expansion"]["loops_done"] += done
|
|
164
|
+
save_state(args.run, state)
|
|
165
|
+
print(f"✅ Expansion finished {done} loop(s).")
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def cmd_stop(args):
|
|
169
|
+
(run_dir(args.run)).mkdir(parents=True, exist_ok=True)
|
|
170
|
+
(run_dir(args.run) / "STOP").write_text(
|
|
171
|
+
f"stop requested at {now()}\n", encoding="utf-8"
|
|
172
|
+
)
|
|
173
|
+
print(f"🛑 STOP file written for run '{args.run}'.")
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def main():
|
|
177
|
+
ap = argparse.ArgumentParser(description="Factory run-state helper")
|
|
178
|
+
sub = ap.add_subparsers(dest="command", required=True)
|
|
179
|
+
|
|
180
|
+
p = sub.add_parser("init")
|
|
181
|
+
p.add_argument("--run", required=True)
|
|
182
|
+
p = sub.add_parser("phase-add")
|
|
183
|
+
p.add_argument("--run", required=True)
|
|
184
|
+
p.add_argument("--phase", required=True)
|
|
185
|
+
p.add_argument("--goal", required=True)
|
|
186
|
+
p.add_argument("--accept", default="")
|
|
187
|
+
p = sub.add_parser("qa-record")
|
|
188
|
+
p.add_argument("--run", required=True)
|
|
189
|
+
p.add_argument("--phase", required=True)
|
|
190
|
+
p.add_argument("--seat", required=True)
|
|
191
|
+
p.add_argument("--verdict", required=True)
|
|
192
|
+
p.add_argument("--reason", default="")
|
|
193
|
+
p = sub.add_parser("expansion")
|
|
194
|
+
p.add_argument("--run", required=True)
|
|
195
|
+
p.add_argument("--loops", type=int, required=True)
|
|
196
|
+
p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
|
|
197
|
+
p = sub.add_parser("stop")
|
|
198
|
+
p.add_argument("--run", required=True)
|
|
199
|
+
|
|
200
|
+
args = ap.parse_args()
|
|
201
|
+
code = {
|
|
202
|
+
"init": cmd_init,
|
|
203
|
+
"phase-add": cmd_phase_add,
|
|
204
|
+
"qa-record": cmd_qa_record,
|
|
205
|
+
"expansion": cmd_expansion,
|
|
206
|
+
"stop": cmd_stop,
|
|
207
|
+
}[args.command](args)
|
|
208
|
+
sys.exit(code if isinstance(code, int) else 0)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
if __name__ == "__main__":
|
|
212
|
+
main()
|
|
Binary file
|
|
Binary file
|
|
Binary file
|
|
Binary file
|