@heretek-ai/epistemic-swarm 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. package/.agents/skills/factory/SKILL.md +13 -0
  2. package/.omp/commands/domainexpansion.md +9 -0
  3. package/.omp/commands/factory.md +9 -0
  4. package/README.md +9 -0
  5. package/bin/cli.js +21 -0
  6. package/config/opencode-snippet.json +88 -1
  7. package/package.json +5 -3
  8. package/plugins/antigravity/skills/factory/SKILL.md +13 -0
  9. package/plugins/codex/skills/factory/SKILL.md +13 -0
  10. package/plugins/gemini/commands/domainexpansion.toml +7 -0
  11. package/plugins/gemini/commands/factory.toml +8 -0
  12. package/plugins/gemini/skills/factory/SKILL.md +13 -0
  13. package/plugins/opencode/index.js +35 -0
  14. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  15. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  16. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  17. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  18. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  19. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  20. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  21. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  22. package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
  23. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  24. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  25. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  26. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  27. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  28. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  29. package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
  30. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  31. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  32. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  33. package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
  34. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  35. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  36. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  37. package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
  38. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  39. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  40. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  41. package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
  42. package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
  43. package/runner/tests/test_factory.py +199 -0
  44. package/runner/tests/test_opencode_ux.py +8 -0
  45. package/runner/tests/test_swarm.py +2 -0
  46. package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
  47. package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
  48. package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
  49. package/scripts/build_adapters.py +7 -0
  50. package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
  51. package/skills/factory/SKILL.md +51 -0
  52. package/skills/factory/scripts/factory.py +212 -0
  53. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  54. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  55. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  56. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
@@ -0,0 +1,13 @@
1
+ # Factory (thin adapter stub)
2
+
3
+ This file is a POINTER, not the implementation. It exists so harness skill
4
+ discovery finds an entry; the real skill lives in the IUMBTEMS repo.
5
+
6
+ - Canonical prose & scripts: `skills/factory/`
7
+ - Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
8
+ or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
9
+ - MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
10
+
11
+ Epistemic rules apply regardless of harness: tag claims as
12
+ `[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
13
+ `[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
@@ -0,0 +1,9 @@
1
+ ---
2
+ description: Autonomous agent-guided self-improvement loop (count-flagged)
3
+ ---
4
+
5
+ Run the IUMBTEMS domain-expansion loop for `$1` loops (max 10):
6
+
7
+ Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill.
8
+ Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
9
+ Enforce via `python3 skills/factory/scripts/factory.py expansion --run <run> --loops $1`.
@@ -0,0 +1,9 @@
1
+ ---
2
+ description: Coding-factory Manager loop with grill-gated phased builds
3
+ ---
4
+
5
+ Run the IUMBTEMS coding-factory Manager loop for `$1`:
6
+
7
+ 1. Grill until `.factory/frontier.json` is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
8
+ 2. Per gate: `python3 runner/research_swarm.py --mode brainstorm` plus `--mode darkharvest` (mock-first), then synthesize `.roadmap/<phase>/` GOAL.md + dossier.json.
9
+ 3. Programmer subagent per phase; qa-a plus qa-b per phase; retries tracked via `python3 skills/factory/scripts/factory.py` (3 failures escalate).
package/README.md CHANGED
@@ -149,6 +149,13 @@ iumbtems brainstorm "Where do we go from here?"
149
149
  # Product competitor teardown with per-feature harvest verdicts
150
150
  iumbtems darkharvest "Paseo-class agent harness competitor" --seeds https://github.com/a/b,https://github.com/c/d --max-repos 6 --mock-claude
151
151
 
152
+ # Coding-factory run-state helper (init, phase-add, qa-record, expansion, stop)
153
+ iumbtems factory init --run arena
154
+ iumbtems factory phase-add --run arena --phase 01-handoff --goal "Session handoff" --accept "round-trips;STOP kills loop"
155
+ iumbtems factory expansion --run arena --loops 10
156
+
157
+ # OpenCode slash commands (also Pi/OMP/Gemini): /factory, /domainexpansion, /darkharvest, /scout, /audit, /grill, /swarm
158
+
152
159
  # Run Socratic grilling and decision frontier calculation
153
160
  iumbtems grill --objective "L1 vs L2 state verification trade-offs"
154
161
 
@@ -164,6 +171,8 @@ iumbtems test
164
171
  - **`/code-audit`**: Dialectic codebase review pairing a Structural Architect (thesis) with a Vulnerability Red-Teamer (antithesis) enforcing line-number proofs (`file:///path#L10-25`).
165
172
  - **`/oss-scout`**: Evaluates GitHub repositories, package ecosystems (npm, crates.io, PyPI), license contamination (GPL/AGPL copyleft vs MIT/Apache), and outputs clean-room re-implementation blueprints.
166
173
  - **`/darkharvest`**: Product competitor teardown (seed inspirations + prompt, expand to adjacents). Competitor × capability matrix, both-ways white-space gaps, per-feature `depend|vendor|clean-room-rebuild|skip` verdicts with SPDX attribution. Permissive-only vendoring; GPL/AGPL spec-rebuild only.
174
+ - **`/factory`**: Coding-factory Manager loop — grill-gated phased build (manager profile), per-phase programmer spawns, dual QA (3 retries then escalate), explicit sign-off per phase.
175
+ - **`/domainexpansion`**: Autonomous agent-guided self-improvement loop (`/domainexpansion <n>`, max 10); bypasses gates, stops on count OR `.factory/STOP` OR user kill.
167
176
  - **`/grilling`**: Socratic assumption-inversion and Matt Pocock-style design tree frontier discovery.
168
177
  - **`epistemic_search`**: Zero-key DuckDuckGo Lite search and content-addressed fetch with automatic SHA-256 caching.
169
178
 
package/bin/cli.js CHANGED
@@ -50,6 +50,7 @@ Commands:
50
50
  run "<objective>" Run the dialectic multi-agent research swarm
51
51
  brainstorm "<prompt>" Run lateral brainstorming (feature vectors + spikes)
52
52
  darkharvest "<arena>" Product competitor teardown with harvest verdicts
53
+ factory <subcommand> Factory run-state helper (init, phase-add, qa-record, expansion, stop)
53
54
  grill Launch interactive Socratic decision tree framing
54
55
  adapters Rebuild harness adapter mirrors (skills -> plugins/*, .agents)
55
56
  install Install skills & MCP servers into ~/.claude/
@@ -82,6 +83,7 @@ Examples:
82
83
  iumbtems run "Verify sub-millisecond ZK prover latency"
83
84
  iumbtems brainstorm "Where do we go from here?"
84
85
  iumbtems darkharvest "Paseo-class agent harness competitor" --seeds https://github.com/a/b,https://github.com/c/d --max-repos 6 --mock-claude
86
+ iumbtems factory init --run arena && iumbtems factory phase-add --run arena --phase 01-x --goal "..." --accept "a;b"
85
87
  iumbtems grill --objective "Rollup architecture trade-offs"
86
88
  iumbtems doctor
87
89
  `);
@@ -174,6 +176,25 @@ switch (command) {
174
176
  break;
175
177
  }
176
178
 
179
+ case 'factory': {
180
+ // Factory run-state helper: forward subcommands to skills/factory/scripts/factory.py
181
+ // e.g. iumbtems factory init --run arena
182
+ // iumbtems factory qa-record --run arena --phase 01-x --seat qa-a --verdict pass
183
+ if (args[1] === '--help' || !args[1]) {
184
+ console.log([
185
+ 'Usage: iumbtems factory <init|phase-add|qa-record|expansion|stop> [options]',
186
+ ' init --run <name>',
187
+ ' phase-add --run <name> --phase <id> --goal "<goal>" --accept "a;b"',
188
+ ' qa-record --run <name> --phase <id> --seat <qa-a|qa-b> --verdict <pass|fail|conditional> [--reason "..."]',
189
+ ' expansion --run <name> --loops <n> [--max-loops 10]',
190
+ ' stop --run <name> (writes .factory/STOP kill-file)',
191
+ ].join('\n'));
192
+ break;
193
+ }
194
+ runPython('skills/factory/scripts/factory.py', args.slice(1));
195
+ break;
196
+ }
197
+
177
198
  case 'grill': {
178
199
  runPython('skills/grilling/socratic_tree.py', args.slice(1));
179
200
  break;
@@ -99,6 +99,92 @@
99
99
  "iumbtems_verify_quote": true,
100
100
  "iumbtems_config": true
101
101
  }
102
+ },
103
+ "manager": {
104
+ "description": "IUMBTEMS Factory Manager: grill-gated phased builds. Owns gates, swarm dispatch, roadmap synthesis, QA tiebreaks. Never writes code.",
105
+ "mode": "primary",
106
+ "temperature": 0.2,
107
+ "permission": {
108
+ "edit": "deny",
109
+ "bash": "ask",
110
+ "task": {
111
+ "*": "deny",
112
+ "factory-*": "allow",
113
+ "programmer": "allow",
114
+ "qa-a": "allow",
115
+ "qa-b": "allow",
116
+ "brainstormer": "allow",
117
+ "darkharvester": "allow"
118
+ }
119
+ },
120
+ "tools": {
121
+ "read": true,
122
+ "write": true,
123
+ "grep": true,
124
+ "glob": true,
125
+ "task": true,
126
+ "iumbtems_brainstorm": true,
127
+ "iumbtems_darkharvest": true,
128
+ "iumbtems_verify_quote": true,
129
+ "iumbtems_socratic_frontier": true,
130
+ "iumbtems_config": true
131
+ }
132
+ },
133
+ "programmer": {
134
+ "description": "IUMBTEMS Factory Programmer: implements exactly one phase brief per spawn. Cites phase evidence hashes. Never invokes swarms or other programmers.",
135
+ "mode": "subagent",
136
+ "temperature": 0.3,
137
+ "permission": {
138
+ "edit": "allow",
139
+ "bash": "ask",
140
+ "task": {
141
+ "*": "deny"
142
+ }
143
+ },
144
+ "tools": {
145
+ "read": true,
146
+ "write": true,
147
+ "bash": true,
148
+ "grep": true,
149
+ "glob": true,
150
+ "iumbtems_verify_quote": true
151
+ }
152
+ },
153
+ "qa-a": {
154
+ "description": "IUMBTEMS Factory QA (functional): verifies phase acceptance criteria pass on the real surface. Read-only plus test execution. Diverged prompt from qa-b.",
155
+ "mode": "subagent",
156
+ "temperature": 0.1,
157
+ "permission": {
158
+ "edit": "deny",
159
+ "bash": "ask",
160
+ "task": {
161
+ "*": "deny"
162
+ }
163
+ },
164
+ "tools": {
165
+ "read": true,
166
+ "bash": true,
167
+ "grep": true,
168
+ "glob": true
169
+ }
170
+ },
171
+ "qa-b": {
172
+ "description": "IUMBTEMS Factory QA (adversarial): hunts edge cases, regressions, and acceptance loopholes the functional pass missed. Read-only plus test execution. Diverged prompt from qa-a.",
173
+ "mode": "subagent",
174
+ "temperature": 0.4,
175
+ "permission": {
176
+ "edit": "deny",
177
+ "bash": "ask",
178
+ "task": {
179
+ "*": "deny"
180
+ }
181
+ },
182
+ "tools": {
183
+ "read": true,
184
+ "bash": true,
185
+ "grep": true,
186
+ "glob": true
187
+ }
102
188
  }
103
189
  },
104
190
  "skills": {
@@ -110,7 +196,8 @@
110
196
  "./skills/code_audit",
111
197
  "./skills/oss_scout",
112
198
  "./skills/brainstorming",
113
- "./skills/darkharvest"
199
+ "./skills/darkharvest",
200
+ "./skills/factory"
114
201
  ]
115
202
  }
116
203
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@heretek-ai/epistemic-swarm",
3
- "version": "0.5.0",
3
+ "version": "0.6.0",
4
4
  "description": "IUMBTEMS: I Use My Brain To Express My Self — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi, OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
5
5
  "main": "bin/cli.js",
6
6
  "bin": {
@@ -80,7 +80,8 @@
80
80
  "./skills/code_audit",
81
81
  "./skills/oss_scout",
82
82
  "./skills/brainstorming",
83
- "./skills/darkharvest"
83
+ "./skills/darkharvest",
84
+ "./skills/factory"
84
85
  ],
85
86
  "prompts": [
86
87
  "./prompts/*.md"
@@ -98,7 +99,8 @@
98
99
  "./skills/code_audit",
99
100
  "./skills/oss_scout",
100
101
  "./skills/brainstorming",
101
- "./skills/darkharvest"
102
+ "./skills/darkharvest",
103
+ "./skills/factory"
102
104
  ],
103
105
  "prompts": [
104
106
  "./prompts/*.md"
@@ -0,0 +1,13 @@
1
+ # Factory (thin adapter stub)
2
+
3
+ This file is a POINTER, not the implementation. It exists so harness skill
4
+ discovery finds an entry; the real skill lives in the IUMBTEMS repo.
5
+
6
+ - Canonical prose & scripts: `skills/factory/`
7
+ - Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
8
+ or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
9
+ - MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
10
+
11
+ Epistemic rules apply regardless of harness: tag claims as
12
+ `[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
13
+ `[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
@@ -0,0 +1,13 @@
1
+ # Factory (thin adapter stub)
2
+
3
+ This file is a POINTER, not the implementation. It exists so harness skill
4
+ discovery finds an entry; the real skill lives in the IUMBTEMS repo.
5
+
6
+ - Canonical prose & scripts: `skills/factory/`
7
+ - Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
8
+ or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
9
+ - MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
10
+
11
+ Epistemic rules apply regardless of harness: tag claims as
12
+ `[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
13
+ `[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
@@ -0,0 +1,7 @@
1
+ description = "Autonomous agent-guided self-improvement loop (count-flagged)"
2
+ prompt = """Run the IUMBTEMS domain-expansion loop for {{args}} loops (max 10).
3
+
4
+ Bypasses per-loop gates; stops on count OR .factory/STOP file OR user kill.
5
+ Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
6
+ Enforce via: python3 skills/factory/scripts/factory.py expansion --run <run> --loops {{args}}.
7
+ """
@@ -0,0 +1,8 @@
1
+ description = "Coding-factory Manager loop with grill-gated phased builds"
2
+ prompt = """Run the IUMBTEMS coding-factory Manager loop for: {{args}}.
3
+
4
+ 1. Grill until .factory/frontier.json is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
5
+ 2. Per gate: python3 runner/research_swarm.py --mode brainstorm plus --mode darkharvest (mock-first), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json.
6
+ 3. Programmer subagent per phase; qa-a plus qa-b per phase; retries via python3 skills/factory/scripts/factory.py (3 failures escalate).
7
+ Report: .roadmap/ phases plus .factory/state.json.
8
+ """
@@ -0,0 +1,13 @@
1
+ # Factory (thin adapter stub)
2
+
3
+ This file is a POINTER, not the implementation. It exists so harness skill
4
+ discovery finds an entry; the real skill lives in the IUMBTEMS repo.
5
+
6
+ - Canonical prose & scripts: `skills/factory/`
7
+ - Canonical programmatic surface: `python3 runner/mcp_server.py` (stdio MCP),
8
+ or one-shot: `python3 runner/mcp_server.py call <tool> '{...json...}'`
9
+ - MCP tools for this skill: `iumbtems_brainstorm`, `iumbtems_darkharvest`, `iumbtems_socratic_frontier`
10
+
11
+ Epistemic rules apply regardless of harness: tag claims as
12
+ `[VERIFIED: <hash>]`, `[INFERRED: <reasoning>]`, `[HYPOTHESIS: <test>]`, or
13
+ `[NEGATIVE_KNOWLEDGE: <query>]`. Writes go only to `.research/`.
@@ -512,6 +512,37 @@ export const OPENCODE_COMMANDS = [
512
512
  '3. Legal rule: permissive-only vendor; GPL/AGPL clean-room-rebuild only; workflows clonable, assets never.',
513
513
  ].join('\n'),
514
514
  },
515
+ {
516
+ name: 'factory',
517
+ description: 'Coding-factory Manager loop: grill-gated phased build with programmer spawns and dual QA',
518
+ usage: '/factory <product-arena>',
519
+ agent: 'manager',
520
+ subtask: false,
521
+ template: [
522
+ 'Run the IUMBTEMS coding-factory Manager loop as the manager agent.',
523
+ 'Arena: $ARGUMENTS',
524
+ 'If $ARGUMENTS is empty, ask the user what to build first; never proceed on placeholder input.',
525
+ '1. Grill the user until .factory/frontier.json is settled (max 5 brainstorm+darkharvest swarm cycles per gate); explicit user approve advances each gate.',
526
+ '2. Per gate run iumbtems_brainstorm and iumbtems_darkharvest (mock_mode only for dry runs), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json (goal/evidence/acceptance/brief/verdict/hashes; every claim needs a VERIFIED hash).',
527
+ '3. Spawn the programmer subagent per phase with the phase dossier (cite phase hashes); run qa-a and qa-b (diverged prompts) per phase; track retries with skills/factory/scripts/factory.py (3 failures escalate to manager).',
528
+ '4. Manager tiebreaks QA disagreements; explicit user sign-off closes each phase.',
529
+ ].join('\n'),
530
+ },
531
+ {
532
+ name: 'domainexpansion',
533
+ description: 'Autonomous agent-guided self-improvement loop over the codebase (count-flagged)',
534
+ usage: '/domainexpansion <n>',
535
+ agent: 'manager',
536
+ subtask: false,
537
+ template: [
538
+ 'Run the IUMBTEMS domain-expansion loop as the manager agent.',
539
+ 'Loops: $ARGUMENTS (integer count, max 10)',
540
+ 'If $ARGUMENTS is not a positive integer, ask the user for the loop count first.',
541
+ '1. Bypass per-loop gates; stop on count OR .factory/STOP file OR user kill, whichever first (enforce via skills/factory/scripts/factory.py expansion).',
542
+ '2. Each loop: agents propose direction, quick iumbtems_brainstorm/iumbtems_darkharvest check, implement via programmer spawn, dual-QA verify.',
543
+ '3. All expansion proposals carry the strict VERIFIED evidence bar; log every loop to .factory/state.json.',
544
+ ].join('\n'),
545
+ },
515
546
  ];
516
547
 
517
548
  /**
@@ -527,6 +558,8 @@ export function commandCatalog() {
527
558
  for (const cmd of OPENCODE_COMMANDS) {
528
559
  if (!cmd?.name) continue;
529
560
  out[cmd.name] = { description: cmd.description, template: cmd.template };
561
+ if (cmd.agent) out[cmd.name].agent = cmd.agent;
562
+ if (cmd.subtask !== undefined) out[cmd.name].subtask = cmd.subtask;
530
563
  }
531
564
  return out;
532
565
  }
@@ -785,6 +818,8 @@ async function registerHostCommands(host) {
785
818
  draft.add({
786
819
  name: cmd.name,
787
820
  description: cmd.description,
821
+ ...(cmd.agent ? { agent: cmd.agent } : {}),
822
+ ...(cmd.subtask !== undefined ? { subtask: cmd.subtask } : {}),
788
823
  execute: async (input) => {
789
824
  const args = input?.prompt?.text || '';
790
825
  const prompt = (typeof input?.prompt === 'object' && input?.prompt !== null) ? input.prompt : {};
@@ -0,0 +1,199 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for the factory loop: run-state helper, QA retry bounds, plugin surface."""
3
+
4
+ import json
5
+ import subprocess
6
+ import sys
7
+ import tempfile
8
+ import unittest
9
+ from pathlib import Path
10
+
11
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
12
+ if str(PROJECT_ROOT) not in sys.path:
13
+ sys.path.insert(0, str(PROJECT_ROOT))
14
+
15
+
16
+ def run_node(code):
17
+ return subprocess.run(
18
+ ["node", "--input-type=module", "-e", code],
19
+ capture_output=True,
20
+ text=True,
21
+ cwd=str(PROJECT_ROOT),
22
+ )
23
+
24
+
25
+ def last_json_object(stdout):
26
+ lines = [l.strip() for l in stdout.strip().split("\n") if l.strip().startswith("{")]
27
+ return json.loads(lines[-1])
28
+
29
+
30
+ class TestFactoryHelper(unittest.TestCase):
31
+ def test_qa_retry_escalates_after_three(self):
32
+ import shutil
33
+
34
+ with tempfile.TemporaryDirectory() as tmp:
35
+ # Redirect factory state + roadmap into tmp via cwd-independent paths:
36
+ # factory.py anchors to PROJECT_ROOT, so sandbox by copying is
37
+ # overkill — exercise the pure bound logic through CLI on a run
38
+ # name, then clean up the created dirs.
39
+ run = "test-escalation-run"
40
+ try:
41
+ subprocess.run(
42
+ [
43
+ sys.executable,
44
+ "skills/factory/scripts/factory.py",
45
+ "init",
46
+ "--run",
47
+ run,
48
+ ],
49
+ check=True,
50
+ capture_output=True,
51
+ cwd=str(PROJECT_ROOT),
52
+ )
53
+ subprocess.run(
54
+ [
55
+ sys.executable,
56
+ "skills/factory/scripts/factory.py",
57
+ "phase-add",
58
+ "--run",
59
+ run,
60
+ "--phase",
61
+ "01-x",
62
+ "--goal",
63
+ "g",
64
+ "--accept",
65
+ "a",
66
+ ],
67
+ check=True,
68
+ capture_output=True,
69
+ cwd=str(PROJECT_ROOT),
70
+ )
71
+ codes = []
72
+ for _ in range(3):
73
+ r = subprocess.run(
74
+ [
75
+ sys.executable,
76
+ "skills/factory/scripts/factory.py",
77
+ "qa-record",
78
+ "--run",
79
+ run,
80
+ "--phase",
81
+ "01-x",
82
+ "--seat",
83
+ "qa-a",
84
+ "--verdict",
85
+ "fail",
86
+ "--reason",
87
+ "t",
88
+ ],
89
+ capture_output=True,
90
+ cwd=str(PROJECT_ROOT),
91
+ )
92
+ codes.append(r.returncode)
93
+ self.assertEqual(codes, [0, 0, 2]) # 3rd failure escalates
94
+ state = json.loads(
95
+ (PROJECT_ROOT / ".factory" / run / "state.json").read_text()
96
+ )
97
+ self.assertEqual(state["phases"]["01-x"]["status"], "escalated")
98
+ finally:
99
+ shutil.rmtree(PROJECT_ROOT / ".factory" / run, ignore_errors=True)
100
+ shutil.rmtree(PROJECT_ROOT / ".roadmap" / "01-x", ignore_errors=True)
101
+
102
+ def test_expansion_respects_stop_file(self):
103
+ import shutil
104
+
105
+ run = "test-expansion-run"
106
+ try:
107
+ subprocess.run(
108
+ [
109
+ sys.executable,
110
+ "skills/factory/scripts/factory.py",
111
+ "init",
112
+ "--run",
113
+ run,
114
+ ],
115
+ check=True,
116
+ capture_output=True,
117
+ cwd=str(PROJECT_ROOT),
118
+ )
119
+ subprocess.run(
120
+ [
121
+ sys.executable,
122
+ "skills/factory/scripts/factory.py",
123
+ "stop",
124
+ "--run",
125
+ run,
126
+ ],
127
+ check=True,
128
+ capture_output=True,
129
+ cwd=str(PROJECT_ROOT),
130
+ )
131
+ r = subprocess.run(
132
+ [
133
+ sys.executable,
134
+ "skills/factory/scripts/factory.py",
135
+ "expansion",
136
+ "--run",
137
+ run,
138
+ "--loops",
139
+ "10",
140
+ ],
141
+ capture_output=True,
142
+ text=True,
143
+ cwd=str(PROJECT_ROOT),
144
+ )
145
+ self.assertEqual(r.returncode, 0)
146
+ self.assertIn("halting after 0/10", r.stdout)
147
+ finally:
148
+ shutil.rmtree(PROJECT_ROOT / ".factory" / run, ignore_errors=True)
149
+
150
+ def test_skill_and_commands_exist(self):
151
+ self.assertTrue((PROJECT_ROOT / "skills" / "factory" / "SKILL.md").exists())
152
+ self.assertTrue(
153
+ (PROJECT_ROOT / "skills" / "factory" / "scripts" / "factory.py").exists()
154
+ )
155
+ self.assertTrue((PROJECT_ROOT / ".omp" / "commands" / "factory.md").exists())
156
+ self.assertTrue(
157
+ (PROJECT_ROOT / ".omp" / "commands" / "domainexpansion.md").exists()
158
+ )
159
+ self.assertTrue(
160
+ (PROJECT_ROOT / "plugins" / "gemini" / "commands" / "factory.toml").exists()
161
+ )
162
+ with open(PROJECT_ROOT / "package.json") as f:
163
+ pkg = json.load(f)
164
+ self.assertIn("./skills/factory", pkg["pi"]["skills"])
165
+ self.assertIn("./skills/factory", pkg["omp"]["skills"])
166
+
167
+ def test_snippet_factory_roster(self):
168
+ with open(PROJECT_ROOT / "config" / "opencode-snippet.json") as f:
169
+ snippet = json.load(f)
170
+ for agent in ("manager", "programmer", "qa-a", "qa-b"):
171
+ self.assertIn(agent, snippet["agent"])
172
+ manager = snippet["agent"]["manager"]
173
+ self.assertEqual(manager["mode"], "primary")
174
+ task = manager["permission"]["task"]
175
+ self.assertEqual(task["*"], "deny")
176
+ for allowed in ("programmer", "qa-a", "qa-b", "brainstormer", "darkharvester"):
177
+ self.assertEqual(task[allowed], "allow")
178
+ self.assertEqual(snippet["agent"]["qa-a"]["permission"]["edit"], "deny")
179
+ self.assertEqual(snippet["agent"]["qa-b"]["permission"]["edit"], "deny")
180
+
181
+ def test_factory_commands_carry_agent_subtask(self):
182
+ res = run_node(
183
+ """
184
+ import { OPENCODE_COMMANDS, commandCatalog } from "./plugins/opencode/index.js";
185
+ const catalog = commandCatalog();
186
+ console.log(JSON.stringify({
187
+ factory: catalog.factory, expansion: catalog.domainexpansion
188
+ }));
189
+ """
190
+ )
191
+ self.assertEqual(res.returncode, 0, res.stderr)
192
+ data = last_json_object(res.stdout)
193
+ self.assertEqual(data["factory"]["agent"], "manager")
194
+ self.assertIn("$ARGUMENTS", data["factory"]["template"])
195
+ self.assertEqual(data["expansion"]["agent"], "manager")
196
+
197
+
198
+ if __name__ == "__main__":
199
+ unittest.main()
@@ -26,6 +26,8 @@ EXPECTED_COMMANDS = [
26
26
  "brainstorming",
27
27
  "brainstorm",
28
28
  "darkharvest",
29
+ "factory",
30
+ "domainexpansion",
29
31
  ]
30
32
 
31
33
 
@@ -282,6 +284,10 @@ class TestInstallOpenCode(unittest.TestCase):
282
284
  "epistemic-auditor",
283
285
  "brainstormer",
284
286
  "darkharvester",
287
+ "manager",
288
+ "programmer",
289
+ "qa-a",
290
+ "qa-b",
285
291
  ]:
286
292
  self.assertIn(agent, cfg.get("agent", {}))
287
293
  self.assertIn("iumbtems", cfg.get("mcp", {}))
@@ -606,6 +612,8 @@ class TestOpenCodeV2Transforms(unittest.TestCase):
606
612
  "brainstorming",
607
613
  "brainstorm",
608
614
  "darkharvest",
615
+ "factory",
616
+ "domainexpansion",
609
617
  ]:
610
618
  self.assertIn(c, data["commands"])
611
619
  self.assertNotIn("goal", data["commands"])
@@ -402,6 +402,8 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
402
402
  "brainstorming",
403
403
  "brainstorm",
404
404
  "darkharvest",
405
+ "factory",
406
+ "domainexpansion",
405
407
  ]:
406
408
  self.assertIn(c, data_oc["commands"])
407
409
 
@@ -40,6 +40,7 @@ CANONICAL_SKILLS = [
40
40
  "oss_scout",
41
41
  "brainstorming",
42
42
  "darkharvest",
43
+ "factory",
43
44
  ]
44
45
 
45
46
  # Skill -> the MCP tool(s) that now carry its programmatic surface.
@@ -52,6 +53,11 @@ SKILL_TOOLS = {
52
53
  "oss_scout": ["iumbtems_oss_scout"],
53
54
  "brainstorming": ["iumbtems_brainstorm"],
54
55
  "darkharvest": ["iumbtems_darkharvest"],
56
+ "factory": [
57
+ "iumbtems_brainstorm",
58
+ "iumbtems_darkharvest",
59
+ "iumbtems_socratic_frontier",
60
+ ],
55
61
  }
56
62
 
57
63
  SKILL_TITLES = {
@@ -63,6 +69,7 @@ SKILL_TITLES = {
63
69
  "oss_scout": "OSS Scout",
64
70
  "brainstorming": "Brainstorming",
65
71
  "darkharvest": "Darkharvest",
72
+ "factory": "Factory",
66
73
  }
67
74
 
68
75
  # (target dir relative to root, mode). All targets are stubs since A2b.
@@ -0,0 +1,51 @@
1
+ ---
2
+ name: factory
3
+ description: Coding-factory Manager loop. Use when user invokes /factory or /domainexpansion, or wants grill-gated phased builds with programmer spawns and dual QA. Manager grills until frontier settled, runs brainstorm plus darkharvest swarms per gate, synthesizes .roadmap phases, spawns programmer per phase, and enforces dual-QA retry bounds. Never writes code itself; never bypasses explicit user sign-off (except inside /domainexpansion count).
4
+ ---
5
+
6
+ # Factory — Manager Loop & Domain Expansion
7
+
8
+ ## 1. Roles (OpenCode profiles in `config/opencode-snippet.json`)
9
+
10
+ - **manager** (primary): owns gates, grilling, swarm dispatch, roadmap synthesis, tiebreaks.
11
+ Task allowlist: `factory-*`, `programmer`, `qa-a`, `qa-b`, `brainstormer`, `darkharvester` deny `*` otherwise.
12
+ - **programmer** (subagent): implements exactly one phase brief. May call `iumbtems_verify_quote`; may NOT invoke swarms or other programmers.
13
+ - **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
14
+ - **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
15
+
16
+ ## 2. Gate protocol (max 5 swarm cycles per gate)
17
+
18
+ 1. Grill until `.factory/frontier.json` settled (grilling skill).
19
+ 2. Run brainstorm + darkharvest swarms (mock-first on fixtures).
20
+ 3. Manager synthesizes `.roadmap/<phase>/` (GOAL.md + dossier.json).
21
+ 4. Explicit user `approve` advances; anything else regrills (cycle counter in `.factory/state.json`).
22
+
23
+ ## 3. Phase contract (`.roadmap/<phase>/`)
24
+
25
+ - `GOAL.md`: human-readable goal + acceptance criteria.
26
+ - `dossier.json`: `{phase, goal, evidence:[{hash, quote}], acceptance[], brief, verdict, hashes}`. Every harvest/claim entry needs a SHA-256 source hash or `file://` pointer or it is purged to NEGATIVE_KNOWLEDGE.
27
+ - Programmer receives the phase dossier (Manager chooses freeform vs strict brief but MUST cite phase hashes).
28
+
29
+ ## 4. QA protocol (3 retries, then escalate)
30
+
31
+ - Both QA seats run per phase; disagreements go to manager tiebreak.
32
+ - `skills/factory/scripts/factory.py` tracks `qa_retries` per phase in `.factory/state.json`. On 3rd rejection: halt phase, return to manager with both QA reports (retry loop per plan; manager may regrill scope or escalate to user).
33
+ - QA verdicts: `pass | fail(reason) | conditional(note)`.
34
+
35
+ ## 5. Domain expansion (`/domainexpansion <n>`)
36
+
37
+ - Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill, whichever first.
38
+ - Each loop: agents propose direction → quick swarm check → implement → QA → next.
39
+ - `factory.py` enforces: refuse `n < 1`, cap `n` at `--max-loops` default 10, check STOP file before every loop.
40
+
41
+ ## 6. State layout (three dirs, distinct jobs)
42
+
43
+ - `.factory/`: run state (`state.json`, `frontier.json`, `STOP` kill-file). Gitignored runtime state.
44
+ - `.roadmap/`: output (phase dirs). Committed.
45
+ - `.research/`: evidence (swarm dossiers, source cache). Gitignored (existing rule).
46
+
47
+ ## 7. Anti-patterns
48
+
49
+ - No code writes by manager; no swarm invocation by programmer; no edits by QA.
50
+ - No gate bypass outside `/domainexpansion`; no uncapped loops.
51
+ - No VERIFIED claims without hashes, even in expansion proposals.
@@ -0,0 +1,212 @@
1
+ #!/usr/bin/env python3
2
+ """
3
+ Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
4
+
5
+ All state lives under .factory/ (gitignored runtime state). Phase output goes
6
+ to .roadmap/<phase>/. Evidence stays in .research/. Read-only w.r.t. repo code.
7
+
8
+ Usage:
9
+ python3 skills/factory/scripts/factory.py init --run <name>
10
+ python3 skills/factory/scripts/factory.py phase-add --run <name> --phase 01-auth --goal "..." --accept "..."
11
+ python3 skills/factory/scripts/factory.py qa-record --run <name> --phase 01-auth --seat qa-a --verdict fail --reason "..."
12
+ python3 skills/factory/scripts/factory.py expansion --run <name> --loops 10 [--max-loops 10]
13
+ python3 skills/factory/scripts/factory.py stop --run <name> # write STOP kill-file
14
+ """
15
+
16
+ import argparse
17
+ import json
18
+ import sys
19
+ from datetime import datetime, timezone
20
+ from pathlib import Path
21
+
22
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
23
+ FACTORY_DIR = PROJECT_ROOT / ".factory"
24
+ ROADMAP_DIR = PROJECT_ROOT / ".roadmap"
25
+
26
+ MAX_QA_RETRIES = 3
27
+ MAX_EXPANSION_LOOPS = 10
28
+
29
+
30
+ def now():
31
+ return datetime.now(timezone.utc).isoformat()
32
+
33
+
34
+ def run_dir(run):
35
+ return FACTORY_DIR / run
36
+
37
+
38
+ def load_state(run):
39
+ p = run_dir(run) / "state.json"
40
+ if not p.exists():
41
+ raise SystemExit(f"No factory run '{run}'. Run `factory.py init` first.")
42
+ return json.loads(p.read_text(encoding="utf-8"))
43
+
44
+
45
+ def save_state(run, state):
46
+ p = run_dir(run) / "state.json"
47
+ p.write_text(json.dumps(state, indent=2), encoding="utf-8")
48
+
49
+
50
+ def cmd_init(args):
51
+ d = run_dir(args.run)
52
+ d.mkdir(parents=True, exist_ok=True)
53
+ state = {
54
+ "run": args.run,
55
+ "created_at": now(),
56
+ "gate_cycles": 0,
57
+ "phases": {},
58
+ "expansion": {"loops_done": 0, "loops_planned": 0},
59
+ }
60
+ save_state(args.run, state)
61
+ print(f"✅ Factory run '{args.run}' initialised at {d}")
62
+
63
+
64
+ def cmd_phase_add(args):
65
+ state = load_state(args.run)
66
+ if args.phase in state["phases"]:
67
+ raise SystemExit(f"Phase '{args.phase}' already exists.")
68
+ state["phases"][args.phase] = {
69
+ "goal": args.goal,
70
+ "acceptance": [a.strip() for a in args.accept.split(";") if a.strip()],
71
+ "status": "briefed",
72
+ "qa_retries": 0,
73
+ "qa_reports": [],
74
+ }
75
+ save_state(args.run, state)
76
+ # Phase output: GOAL.md + dossier.json skeleton (evidence filled by manager).
77
+ phase_dir = ROADMAP_DIR / args.phase
78
+ phase_dir.mkdir(parents=True, exist_ok=True)
79
+ (phase_dir / "GOAL.md").write_text(
80
+ f"# {args.phase}: {args.goal}\n\n## Acceptance\n"
81
+ + "".join(f"- [ ] {a}\n" for a in state["phases"][args.phase]["acceptance"]),
82
+ encoding="utf-8",
83
+ )
84
+ (phase_dir / "dossier.json").write_text(
85
+ json.dumps(
86
+ {
87
+ "phase": args.phase,
88
+ "goal": args.goal,
89
+ "evidence": [],
90
+ "acceptance": state["phases"][args.phase]["acceptance"],
91
+ "brief": "",
92
+ "verdict": "briefed",
93
+ "hashes": [],
94
+ },
95
+ indent=2,
96
+ ),
97
+ encoding="utf-8",
98
+ )
99
+ print(f"✅ Phase '{args.phase}' briefed → {phase_dir}")
100
+
101
+
102
+ def cmd_qa_record(args):
103
+ state = load_state(args.run)
104
+ phase = state["phases"].get(args.phase)
105
+ if phase is None:
106
+ raise SystemExit(f"Unknown phase '{args.phase}'.")
107
+ if args.verdict not in ("pass", "fail", "conditional"):
108
+ raise SystemExit("verdict must be pass|fail|conditional.")
109
+ phase["qa_reports"].append(
110
+ {
111
+ "seat": args.seat,
112
+ "verdict": args.verdict,
113
+ "reason": args.reason or "",
114
+ "at": now(),
115
+ }
116
+ )
117
+ if args.verdict == "fail":
118
+ phase["qa_retries"] += 1
119
+ if phase["qa_retries"] >= MAX_QA_RETRIES:
120
+ phase["status"] = "escalated"
121
+ save_state(args.run, state)
122
+ print(
123
+ f"🛑 Phase '{args.phase}' ESCALATED after "
124
+ f"{MAX_QA_RETRIES} QA failures — back to manager."
125
+ )
126
+ return 2
127
+ phase["status"] = "retrying"
128
+ elif args.verdict == "pass":
129
+ phase["status"] = "signed-off"
130
+ else:
131
+ phase["status"] = "conditional"
132
+ save_state(args.run, state)
133
+ print(
134
+ f"✅ QA recorded: {args.phase} [{args.seat}] → {args.verdict} "
135
+ f"(status={phase['status']}, retries={phase['qa_retries']})"
136
+ )
137
+
138
+
139
+ def stop_requested(run):
140
+ return (run_dir(run) / "STOP").exists()
141
+
142
+
143
+ def cmd_expansion(args):
144
+ state = load_state(args.run)
145
+ n = args.loops
146
+ if n < 1:
147
+ raise SystemExit("loops must be >= 1.")
148
+ cap = args.max_loops or MAX_EXPANSION_LOOPS
149
+ n = min(n, cap)
150
+ state["expansion"]["loops_planned"] = n
151
+ save_state(args.run, state)
152
+ done = 0
153
+ for i in range(1, n + 1):
154
+ if stop_requested(args.run):
155
+ print(f"🛑 STOP file present — halting after {done}/{n} loops.")
156
+ break
157
+ done += 1
158
+ print(
159
+ f"🔁 Expansion loop {done}/{n}: agents propose → swarm check → "
160
+ f"implement → QA (Manager drives each step)."
161
+ )
162
+ state = load_state(args.run)
163
+ state["expansion"]["loops_done"] += done
164
+ save_state(args.run, state)
165
+ print(f"✅ Expansion finished {done} loop(s).")
166
+
167
+
168
+ def cmd_stop(args):
169
+ (run_dir(args.run)).mkdir(parents=True, exist_ok=True)
170
+ (run_dir(args.run) / "STOP").write_text(
171
+ f"stop requested at {now()}\n", encoding="utf-8"
172
+ )
173
+ print(f"🛑 STOP file written for run '{args.run}'.")
174
+
175
+
176
+ def main():
177
+ ap = argparse.ArgumentParser(description="Factory run-state helper")
178
+ sub = ap.add_subparsers(dest="command", required=True)
179
+
180
+ p = sub.add_parser("init")
181
+ p.add_argument("--run", required=True)
182
+ p = sub.add_parser("phase-add")
183
+ p.add_argument("--run", required=True)
184
+ p.add_argument("--phase", required=True)
185
+ p.add_argument("--goal", required=True)
186
+ p.add_argument("--accept", default="")
187
+ p = sub.add_parser("qa-record")
188
+ p.add_argument("--run", required=True)
189
+ p.add_argument("--phase", required=True)
190
+ p.add_argument("--seat", required=True)
191
+ p.add_argument("--verdict", required=True)
192
+ p.add_argument("--reason", default="")
193
+ p = sub.add_parser("expansion")
194
+ p.add_argument("--run", required=True)
195
+ p.add_argument("--loops", type=int, required=True)
196
+ p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
197
+ p = sub.add_parser("stop")
198
+ p.add_argument("--run", required=True)
199
+
200
+ args = ap.parse_args()
201
+ code = {
202
+ "init": cmd_init,
203
+ "phase-add": cmd_phase_add,
204
+ "qa-record": cmd_qa_record,
205
+ "expansion": cmd_expansion,
206
+ "stop": cmd_stop,
207
+ }[args.command](args)
208
+ sys.exit(code if isinstance(code, int) else 0)
209
+
210
+
211
+ if __name__ == "__main__":
212
+ main()