@heretek-ai/epistemic-swarm 0.7.2 → 0.7.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/.omp/commands/domainexpansion.md +1 -1
  3. package/.omp/commands/factory.md +1 -1
  4. package/extensions/pi/index.js +28 -2
  5. package/package.json +1 -1
  6. package/plugins/factory/skills/factory/SKILL.md +10 -0
  7. package/plugins/factory/skills/factory/scripts/factory.py +39 -5
  8. package/plugins/gemini/commands/domainexpansion.toml +1 -1
  9. package/plugins/gemini/commands/factory.toml +1 -1
  10. package/plugins/opencode/index.js +53 -7
  11. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  12. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  13. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  14. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  15. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  16. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  17. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  18. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  19. package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
  20. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  21. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  22. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  23. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  24. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  25. package/runner/mcp_server.py +121 -0
  26. package/runner/research_swarm.py +36 -23
  27. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  28. package/runner/tests/__pycache__/test_backends.cpython-311.pyc +0 -0
  29. package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
  30. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  31. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  32. package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
  33. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  34. package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
  35. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  36. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  37. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  38. package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
  39. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  40. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  41. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  42. package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
  43. package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
  44. package/runner/tests/test_backends.py +49 -4
  45. package/runner/tests/test_factory.py +47 -0
  46. package/runner/tests/test_mcp_server.py +1 -0
  47. package/runner/tests/test_opencode_ux.py +2 -2
  48. package/runner/tests/test_swarm.py +1 -0
  49. package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
  50. package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
  51. package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
  52. package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
  53. package/skills/factory/SKILL.md +10 -0
  54. package/skills/factory/scripts/factory.py +39 -5
  55. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  56. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  57. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  58. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "epistemic-swarm",
3
- "version": "0.7.2",
3
+ "version": "0.7.4",
4
4
  "description": "High-Integrity Dialectic Research Agent Harness for Claude Code enforcing empirical evidence over parametric hallucination.",
5
5
  "author": {
6
6
  "name": "Heretek AI",
@@ -6,4 +6,4 @@ Run the IUMBTEMS domain-expansion loop for `$1` loops (max 10):
6
6
 
7
7
  Bypasses per-loop gates; stops on count OR `.factory/STOP` file OR user kill.
8
8
  Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
9
- Enforce via `python3 skills/factory/scripts/factory.py expansion --run <run> --loops $1`.
9
+ Enforce via the `iumbtems_factory` tool (expansion command with loops/max_loops).
@@ -6,4 +6,4 @@ Run the IUMBTEMS coding-factory Manager loop for `$1`:
6
6
 
7
7
  1. Grill until `.factory/frontier.json` is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
8
8
  2. Per gate: `python3 runner/research_swarm.py --mode brainstorm` plus `--mode darkharvest` (mock-first), then synthesize `.roadmap/<phase>/` GOAL.md + dossier.json.
9
- 3. Programmer subagent per phase; qa-a plus qa-b per phase; retries tracked via `python3 skills/factory/scripts/factory.py` (3 failures escalate).
9
+ 3. Programmer subagent per phase; qa-a plus qa-b per phase; retries tracked via the `iumbtems_factory` tool (phase-add / qa-record; 3 failures escalate). Never invoke factory helper scripts by relative path.
@@ -210,8 +210,7 @@ export default function initPiExtension(pi) {
210
210
  } catch { /* alias is best-effort across pi/omp versions */ }
211
211
 
212
212
  // /darkharvest: Product competitor teardown with harvest verdicts
213
- pi.registerCommand('darkharvest', {
214
- description: 'Product competitor teardown: seed inspirations, expand to adjacents, emit harvest verdicts',
213
+ pi.registerCommand('darkharvest', { description: 'Product competitor teardown: seed inspirations, expand to adjacents, emit harvest verdicts',
215
214
  usage: '/darkharvest <product-arena> [--seeds <urls>]',
216
215
  handler: async (args, ctx) => {
217
216
  const objective = args.trim();
@@ -316,6 +315,33 @@ export default function initPiExtension(pi) {
316
315
  return { content: [{ type: 'text', text: r.text }] };
317
316
  }
318
317
  });
318
+
319
+ // Tool: iumbtems_factory (run-state helper, no script paths)
320
+ pi.registerTool({
321
+ name: 'iumbtems_factory',
322
+ description: 'Drive factory run state: init / phase-add / qa-record / expansion / stop',
323
+ parameters: {
324
+ type: 'object',
325
+ properties: {
326
+ command: { type: 'string', enum: ['init', 'phase-add', 'qa-record', 'expansion', 'stop'] },
327
+ run: { type: 'string', description: 'Factory run name' },
328
+ phase: { type: 'string' },
329
+ goal: { type: 'string' },
330
+ accept: { type: 'string' },
331
+ seat: { type: 'string' },
332
+ verdict: { type: 'string', enum: ['pass', 'fail', 'conditional'] },
333
+ reason: { type: 'string' },
334
+ loops: { type: 'integer' },
335
+ max_loops: { type: 'integer' },
336
+ project_dir: { type: 'string' }
337
+ },
338
+ required: ['command']
339
+ },
340
+ execute: async (args = {}) => {
341
+ const r = callMcp('iumbtems_factory', args, {});
342
+ return { content: [{ type: 'text', text: r.text }] };
343
+ }
344
+ });
319
345
  }
320
346
  }
321
347
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@heretek-ai/epistemic-swarm",
3
- "version": "0.7.2",
3
+ "version": "0.7.4",
4
4
  "description": "IUMBTEMS: I Use My Brain To Express My Self — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi, OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
5
5
  "main": "bin/cli.js",
6
6
  "bin": {
@@ -13,6 +13,16 @@ description: Coding-factory Manager loop. Use when user invokes /factory or /dom
13
13
  - **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
14
14
  - **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
15
15
 
16
+ ## 1b. Driving run state (no filesystem paths)
17
+
18
+ Use the **`iumbtems_factory` MCP tool** for all run-state changes — never a
19
+ relative `skills/factory/scripts/factory.py` path (the toolchain lives in the
20
+ npm cache in consuming projects, and relative paths broke live: the manager
21
+ agent ran `find / -name factory.py`). The tool wraps the same helper and
22
+ resolves the project via `project_dir` argument, `IUMBTEMS_PROJECT_DIR`, or
23
+ the session cwd. It supports `init`, `phase-add`, `qa-record`, `expansion`,
24
+ `stop`, and returns `status: escalated` (exit 2) on the 3rd QA failure.
25
+
16
26
  ## 2. Gate protocol (max 5 swarm cycles per gate)
17
27
 
18
28
  1. Grill until `.factory/frontier.json` settled (grilling skill).
@@ -2,8 +2,13 @@
2
2
  """
3
3
  Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
4
4
 
5
- All state lives under .factory/ (gitignored runtime state). Phase output goes
6
- to .roadmap/<phase>/. Evidence stays in .research/. Read-only w.r.t. repo code.
5
+ All state lives under <project>/.factory/ (gitignored runtime state). Phase
6
+ output goes to <project>/.roadmap/<phase>/. Evidence stays in <project>/
7
+ .research/. Read-only w.r.t. repo code.
8
+
9
+ Project directory resolution: --project-dir > IUMBTEMS_PROJECT_DIR env > the
10
+ toolchain repo root (local-dev default). A consuming project must never leak
11
+ state into the IUMBTEMS checkout.
7
12
 
8
13
  Usage:
9
14
  python3 skills/factory/scripts/factory.py init --run <name>
@@ -15,13 +20,25 @@ Usage:
15
20
 
16
21
  import argparse
17
22
  import json
23
+ import os
18
24
  import sys
19
25
  from datetime import datetime, timezone
20
26
  from pathlib import Path
21
27
 
22
- PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
23
- FACTORY_DIR = PROJECT_ROOT / ".factory"
24
- ROADMAP_DIR = PROJECT_ROOT / ".roadmap"
28
+ REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent
29
+
30
+
31
+ def resolve_project_root(cli_value=None):
32
+ """--project-dir > IUMBTEMS_PROJECT_DIR > repo root (local-dev default)."""
33
+ for candidate in (cli_value, os.environ.get("IUMBTEMS_PROJECT_DIR")):
34
+ if candidate and Path(candidate).is_dir():
35
+ return Path(candidate).resolve()
36
+ return REPO_ROOT
37
+
38
+
39
+ PROJECT_DIR = resolve_project_root()
40
+ FACTORY_DIR = PROJECT_DIR / ".factory"
41
+ ROADMAP_DIR = PROJECT_DIR / ".roadmap"
25
42
 
26
43
  MAX_QA_RETRIES = 3
27
44
  MAX_EXPANSION_LOOPS = 10
@@ -173,31 +190,48 @@ def cmd_stop(args):
173
190
  print(f"🛑 STOP file written for run '{args.run}'.")
174
191
 
175
192
 
193
+ def _add_common(parser):
194
+ parser.add_argument(
195
+ "--project-dir",
196
+ default=None,
197
+ help="Project the factory state belongs to (default: IUMBTEMS_PROJECT_DIR env, else repo root)",
198
+ )
199
+
200
+
176
201
  def main():
202
+ global FACTORY_DIR, ROADMAP_DIR
177
203
  ap = argparse.ArgumentParser(description="Factory run-state helper")
178
204
  sub = ap.add_subparsers(dest="command", required=True)
179
205
 
180
206
  p = sub.add_parser("init")
181
207
  p.add_argument("--run", required=True)
208
+ _add_common(p)
182
209
  p = sub.add_parser("phase-add")
183
210
  p.add_argument("--run", required=True)
184
211
  p.add_argument("--phase", required=True)
185
212
  p.add_argument("--goal", required=True)
186
213
  p.add_argument("--accept", default="")
214
+ _add_common(p)
187
215
  p = sub.add_parser("qa-record")
188
216
  p.add_argument("--run", required=True)
189
217
  p.add_argument("--phase", required=True)
190
218
  p.add_argument("--seat", required=True)
191
219
  p.add_argument("--verdict", required=True)
192
220
  p.add_argument("--reason", default="")
221
+ _add_common(p)
193
222
  p = sub.add_parser("expansion")
194
223
  p.add_argument("--run", required=True)
195
224
  p.add_argument("--loops", type=int, required=True)
196
225
  p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
226
+ _add_common(p)
197
227
  p = sub.add_parser("stop")
198
228
  p.add_argument("--run", required=True)
229
+ _add_common(p)
199
230
 
200
231
  args = ap.parse_args()
232
+ root = resolve_project_root(args.project_dir)
233
+ FACTORY_DIR = root / ".factory"
234
+ ROADMAP_DIR = root / ".roadmap"
201
235
  code = {
202
236
  "init": cmd_init,
203
237
  "phase-add": cmd_phase_add,
@@ -3,5 +3,5 @@ prompt = """Run the IUMBTEMS domain-expansion loop for {{args}} loops (max 10).
3
3
 
4
4
  Bypasses per-loop gates; stops on count OR .factory/STOP file OR user kill.
5
5
  Each loop: agents propose direction, quick swarm check, implement, dual-QA verify.
6
- Enforce via: python3 skills/factory/scripts/factory.py expansion --run <run> --loops {{args}}.
6
+ Enforce via the iumbtems_factory tool (command: expansion, with loops/max_loops).
7
7
  """
@@ -3,6 +3,6 @@ prompt = """Run the IUMBTEMS coding-factory Manager loop for: {{args}}.
3
3
 
4
4
  1. Grill until .factory/frontier.json is settled (max 5 swarm cycles per gate); explicit user approve advances each gate.
5
5
  2. Per gate: python3 runner/research_swarm.py --mode brainstorm plus --mode darkharvest (mock-first), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json.
6
- 3. Programmer subagent per phase; qa-a plus qa-b per phase; retries via python3 skills/factory/scripts/factory.py (3 failures escalate).
6
+ 3. Programmer subagent per phase; qa-a plus qa-b per phase; retries via the iumbtems_factory tool (phase-add / qa-record; 3 failures escalate). Never invoke factory helper scripts by relative path.
7
7
  Report: .roadmap/ phases plus .factory/state.json.
8
8
  """
@@ -102,6 +102,7 @@ export const IUMBTEMS_TOOL_NAMES = [
102
102
  'iumbtems_oss_scout',
103
103
  'iumbtems_brainstorm',
104
104
  'iumbtems_darkharvest',
105
+ 'iumbtems_factory',
105
106
  'iumbtems_verify_quote',
106
107
  'iumbtems_socratic_frontier',
107
108
  'iumbtems_reindex_claims',
@@ -291,6 +292,31 @@ const TOOL_CATALOG = [
291
292
  required: ['objective'],
292
293
  },
293
294
  },
295
+ {
296
+ name: 'iumbtems_factory',
297
+ description:
298
+ 'Drive factory run state: init / phase-add / qa-record / expansion / stop. State goes to <project>/.factory and <project>/.roadmap; no helper-script path needed.',
299
+ input: {
300
+ type: 'object',
301
+ properties: {
302
+ command: {
303
+ type: 'string',
304
+ enum: ['init', 'phase-add', 'qa-record', 'expansion', 'stop'],
305
+ },
306
+ run: { type: 'string', description: 'Factory run name' },
307
+ phase: { type: 'string', description: 'Phase id (e.g. 01-auth)' },
308
+ goal: { type: 'string' },
309
+ accept: { type: 'string', description: 'Semicolon-separated acceptance criteria' },
310
+ seat: { type: 'string', description: 'QA seat (qa-a | qa-b)' },
311
+ verdict: { type: 'string', enum: ['pass', 'fail', 'conditional'] },
312
+ reason: { type: 'string' },
313
+ loops: { type: 'integer', description: 'Expansion loop count (1-10)' },
314
+ max_loops: { type: 'integer' },
315
+ project_dir: { type: 'string', description: 'Project root (default: IUMBTEMS_PROJECT_DIR or cwd)' },
316
+ },
317
+ required: ['command'],
318
+ },
319
+ },
294
320
  {
295
321
  name: 'iumbtems_verify_quote',
296
322
  description:
@@ -558,7 +584,7 @@ export const OPENCODE_COMMANDS = [
558
584
  'If $ARGUMENTS is empty, ask the user what to build first; never proceed on placeholder input.',
559
585
  '1. Grill the user until .factory/frontier.json is settled (max 5 brainstorm+darkharvest swarm cycles per gate); explicit user approve advances each gate.',
560
586
  '2. Per gate run iumbtems_brainstorm and iumbtems_darkharvest (mock_mode only for dry runs), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json (goal/evidence/acceptance/brief/verdict/hashes; every claim needs a VERIFIED hash).',
561
- '3. Spawn the programmer subagent per phase with the phase dossier (cite phase hashes); run qa-a and qa-b (diverged prompts) per phase; track retries with skills/factory/scripts/factory.py (3 failures escalate to manager).',
587
+ '3. Spawn the programmer subagent per phase with the phase dossier (cite phase hashes); run qa-a and qa-b (diverged prompts) per phase; track retries with the iumbtems_factory tool (command: phase-add / qa-record; 3 failures escalate to manager). Never invoke factory helper scripts by relative path.',
562
588
  '4. Manager tiebreaks QA disagreements; explicit user sign-off closes each phase.',
563
589
  ].join('\n'),
564
590
  },
@@ -572,7 +598,7 @@ export const OPENCODE_COMMANDS = [
572
598
  'Run the IUMBTEMS domain-expansion loop as the manager agent.',
573
599
  'Loops: $ARGUMENTS (integer count, max 10)',
574
600
  'If $ARGUMENTS is not a positive integer, ask the user for the loop count first.',
575
- '1. Bypass per-loop gates; stop on count OR .factory/STOP file OR user kill, whichever first (enforce via skills/factory/scripts/factory.py expansion).',
601
+ '1. Bypass per-loop gates; stop on count OR .factory/STOP file OR user kill, whichever first (enforce via the iumbtems_factory tool, command: expansion, with loops/max_loops).',
576
602
  '2. Each loop: agents propose direction, quick iumbtems_brainstorm/iumbtems_darkharvest check, implement via programmer spawn, dual-QA verify.',
577
603
  '3. All expansion proposals carry the strict VERIFIED evidence bar; log every loop to .factory/state.json.',
578
604
  ].join('\n'),
@@ -598,8 +624,14 @@ export function commandCatalog() {
598
624
  return out;
599
625
  }
600
626
 
601
- /** Tool map (object form) for the server hook; array catalog stays canonical. */
602
- function buildToolMap() {
627
+ /** Tool map (object form) for the server hook; array catalog stays canonical.
628
+ *
629
+ * `hostRoot` is the host-reported project directory (host.location.directory).
630
+ * The tool context's own cwd is absent on some host builds — observed live:
631
+ * every dispatch fell back to process.cwd() (/home/john) and evidence escaped
632
+ * the project tree. Prefer toolContext.cwd, then hostRoot, then process.cwd().
633
+ */
634
+ function buildToolMap(hostRoot = undefined) {
603
635
  return Object.fromEntries(
604
636
  TOOL_CATALOG.map((tool) => [
605
637
  tool.name,
@@ -607,7 +639,11 @@ function buildToolMap() {
607
639
  ...tool,
608
640
  options: { codemode: false },
609
641
  execute: async (args = {}, toolContext = undefined) =>
610
- callMcp(tool.name, normalizeArgs(tool.name, args), toolContext?.cwd),
642
+ callMcp(
643
+ tool.name,
644
+ normalizeArgs(tool.name, args),
645
+ toolContext?.cwd || hostRoot
646
+ ),
611
647
  },
612
648
  ])
613
649
  );
@@ -892,15 +928,25 @@ async function registerHostTools(host) {
892
928
  // cannot be intercepted the way Claude Code's hooks/hooks.json does.
893
929
  // Steering lives in command templates (epistemic search first); a host API
894
930
  // for pre-execution guards would close this for real.
931
+ //
932
+ // hostRoot: the host's project directory. toolContext.cwd is absent on
933
+ // some host builds (observed live: dispatches fell back to process.cwd()
934
+ // = /home/john and evidence escaped the project tree), so capture the
935
+ // host-reported directory here as the fallback.
936
+ const hostRoot = host?.location?.directory || undefined;
895
937
  const registration = await host.tool.transform((draft) => {
896
- for (const [name, spec] of Object.entries(buildToolMap())) {
938
+ for (const [name, spec] of Object.entries(buildToolMap(hostRoot))) {
897
939
  draft.add({
898
940
  name,
899
941
  description: spec.description,
900
942
  input: spec.input,
901
943
  options: { codemode: false },
902
944
  execute: async (args = {}, toolContext = undefined) => {
903
- const r = await callMcp(name, normalizeArgs(name, args), toolContext?.cwd);
945
+ const r = await callMcp(
946
+ name,
947
+ normalizeArgs(name, args),
948
+ toolContext?.cwd || hostRoot
949
+ );
904
950
  return { content: r.content };
905
951
  },
906
952
  });
@@ -192,6 +192,87 @@ def _handle_darkharvest(args: Dict[str, Any]) -> str:
192
192
  return _run_swarm_mode("darkharvest", args)
193
193
 
194
194
 
195
+ def _handle_factory(args: Dict[str, Any]) -> str:
196
+ """Drive factory run state (init / phase-add / qa-record / expansion / stop).
197
+
198
+ Wraps skills/factory/scripts/factory.py so agents never need a filesystem
199
+ path to the helper: the npm-installed toolchain lives outside the project,
200
+ and relative `skills/...` paths broke in consuming projects (observed
201
+ live: the manager agent ran `find / -name factory.py`).
202
+ """
203
+ import subprocess
204
+
205
+ command = str(args.get("command") or "").strip()
206
+ if command not in ("init", "phase-add", "qa-record", "expansion", "stop"):
207
+ raise ValueError(
208
+ "command must be one of: init, phase-add, qa-record, expansion, stop"
209
+ )
210
+ script = Path(PROJECT_ROOT) / "skills" / "factory" / "scripts" / "factory.py"
211
+ if not script.is_file():
212
+ raise FileNotFoundError(f"factory helper not found at {script}")
213
+
214
+ project = (
215
+ args.get("project_dir") or os.environ.get("IUMBTEMS_PROJECT_DIR") or os.getcwd()
216
+ )
217
+ cmd = [sys.executable, str(script), command, "--project-dir", str(project)]
218
+ if command == "init":
219
+ cmd += ["--run", str(args.get("run") or "")]
220
+ elif command == "phase-add":
221
+ cmd += [
222
+ "--run",
223
+ str(args.get("run") or ""),
224
+ "--phase",
225
+ str(args.get("phase") or ""),
226
+ "--goal",
227
+ str(args.get("goal") or ""),
228
+ "--accept",
229
+ str(args.get("accept") or ""),
230
+ ]
231
+ elif command == "qa-record":
232
+ cmd += [
233
+ "--run",
234
+ str(args.get("run") or ""),
235
+ "--phase",
236
+ str(args.get("phase") or ""),
237
+ "--seat",
238
+ str(args.get("seat") or ""),
239
+ "--verdict",
240
+ str(args.get("verdict") or ""),
241
+ ]
242
+ if args.get("reason"):
243
+ cmd += ["--reason", str(args["reason"])]
244
+ elif command == "expansion":
245
+ cmd += [
246
+ "--run",
247
+ str(args.get("run") or ""),
248
+ "--loops",
249
+ str(int(args.get("loops") or 1)),
250
+ ]
251
+ if args.get("max_loops"):
252
+ cmd += ["--max-loops", str(int(args["max_loops"]))]
253
+ elif command == "stop":
254
+ cmd += ["--run", str(args.get("run") or "")]
255
+
256
+ proc = subprocess.run(
257
+ cmd,
258
+ capture_output=True,
259
+ text=True,
260
+ shell=False,
261
+ stdin=subprocess.DEVNULL,
262
+ cwd=str(project),
263
+ )
264
+ payload = {
265
+ "status": "ok" if proc.returncode == 0 else "error",
266
+ "command": command,
267
+ "returncode": proc.returncode,
268
+ "project_dir": str(project),
269
+ "output": (proc.stdout or proc.stderr or "").strip()[-4000:],
270
+ }
271
+ if proc.returncode == 2:
272
+ payload["status"] = "escalated"
273
+ return _tool_text(payload)
274
+
275
+
195
276
  def _handle_verify_quote(args: Dict[str, Any]) -> str:
196
277
  """Verify a verbatim quote against the content-addressed source cache."""
197
278
  from skills.research_cache.hasher import SourceHasher
@@ -541,6 +622,46 @@ def build_tools() -> List[ToolSpec]:
541
622
  },
542
623
  handler=_handle_darkharvest,
543
624
  ),
625
+ ToolSpec(
626
+ name="iumbtems_factory",
627
+ description="Drive factory run state: init / phase-add / qa-record / expansion / stop. State goes to <project>/.factory and <project>/.roadmap; no filesystem path to helper scripts required.",
628
+ input_schema={
629
+ "type": "object",
630
+ "properties": {
631
+ "command": {
632
+ "type": "string",
633
+ "enum": ["init", "phase-add", "qa-record", "expansion", "stop"],
634
+ },
635
+ "run": {"type": "string", "description": "Factory run name"},
636
+ "phase": {
637
+ "type": "string",
638
+ "description": "Phase id (e.g. 01-auth)",
639
+ },
640
+ "goal": {"type": "string"},
641
+ "accept": {
642
+ "type": "string",
643
+ "description": "Semicolon-separated acceptance criteria",
644
+ },
645
+ "seat": {"type": "string", "description": "QA seat (qa-a | qa-b)"},
646
+ "verdict": {
647
+ "type": "string",
648
+ "enum": ["pass", "fail", "conditional"],
649
+ },
650
+ "reason": {"type": "string"},
651
+ "loops": {
652
+ "type": "integer",
653
+ "description": "Expansion loop count (1-10)",
654
+ },
655
+ "max_loops": {"type": "integer"},
656
+ "project_dir": {
657
+ "type": "string",
658
+ "description": "Project root (default: IUMBTEMS_PROJECT_DIR or cwd)",
659
+ },
660
+ },
661
+ "required": ["command"],
662
+ },
663
+ handler=_handle_factory,
664
+ ),
544
665
  ToolSpec(
545
666
  name="iumbtems_verify_quote",
546
667
  description="Audit a verbatim citation against the SHA-256 source cache (.research/sources/<hash>.md).",
@@ -43,17 +43,6 @@ OPENCODE_RUN_BASE = ["opencode", "run"]
43
43
  # Fenced JSON block marker shared by orchestrator/dossier stdout parsers.
44
44
  _JSON_FENCE = "```json"
45
45
 
46
- # Swarm mode -> OpenCode agent carrying the equivalent system prompt
47
- # (`opencode run` has no --system-prompt flag; the prompt rides on --agent).
48
- MODE_OPENCODE_AGENT = {
49
- "research": "alpha-thesis",
50
- "audit": "code-auditor",
51
- "scout": "oss-scout",
52
- "hybrid": "alpha-thesis",
53
- "brainstorm": "brainstormer",
54
- "darkharvest": "darkharvester",
55
- }
56
-
57
46
 
58
47
  def _default_backend_cmd(config_host: Optional[str] = None) -> List[str]:
59
48
  """Host-native default backend argv.
@@ -99,7 +88,17 @@ _TEXT_EVENT_MARKERS = ("message", "text", "result", "output", "content")
99
88
 
100
89
 
101
90
  def _event_text(obj: Dict[str, Any]) -> Optional[str]:
102
- """Text payload of one parsed event line, or None."""
91
+ """Text payload of one parsed event line, or None.
92
+
93
+ Handles the real `opencode run --format json` shape
94
+ ({"type":"text", "part":{"type":"text","text":"..."}}) plus top-level
95
+ variants for tolerance.
96
+ """
97
+ part = obj.get("part")
98
+ if isinstance(part, dict):
99
+ val = part.get("text")
100
+ if isinstance(val, str) and val.strip():
101
+ return val
103
102
  kind = str(obj.get("type", "")).lower()
104
103
  if kind and not any(m in kind for m in _TEXT_EVENT_MARKERS):
105
104
  return None
@@ -237,14 +236,16 @@ class SwarmRunner:
237
236
  return list(backend), model
238
237
 
239
238
  def _resolve_opencode_agent(self, role: str) -> Optional[str]:
240
- """OpenCode agent carrying the system prompt for this mode/role."""
241
- agents_cfg = self.config.get("agents") or {}
242
- role_cfg = agents_cfg.get(role) or {}
243
- if role_cfg.get("opencode_agent"):
244
- return role_cfg["opencode_agent"]
245
- if self.mode == "research" and role == "beta":
246
- return "beta-redteam"
247
- return MODE_OPENCODE_AGENT.get(self.mode)
239
+ """Explicitly configured OpenCode agent for this role, else None.
240
+
241
+ Default is None: the system prompt is inlined into the message because
242
+ `--agent <name>` fails hard ("Agent not found") unless the user
243
+ installed the snippet agent profiles — observed live, so the backend
244
+ must not depend on them. Set `agents.<role>.opencode_agent` (or the
245
+ top-level `opencode_agent` config key) to opt in.
246
+ """
247
+ role_cfg = (self.config.get("agents") or {}).get(role) or {}
248
+ return role_cfg.get("opencode_agent") or self.config.get("opencode_agent")
248
249
 
249
250
  def build_agent_cmd(
250
251
  self,
@@ -270,9 +271,16 @@ class SwarmRunner:
270
271
  "backend (CLI flag injection)"
271
272
  )
272
273
  if _backend_family(backend) == "opencode":
273
- return _build_opencode_cmd(
274
- backend, prompt, model, self._resolve_opencode_agent(role)
275
- )
274
+ agent = self._resolve_opencode_agent(role)
275
+ # No --system-prompt flag exists on `opencode run`, and --agent
276
+ # only works when the user installed the profile. Inline the
277
+ # system prompt into the message so runs never depend on
278
+ # host-side agent configuration.
279
+ if system_prompt_file and system_prompt_file.exists():
280
+ prompt = (
281
+ system_prompt_file.read_text(encoding="utf-8") + "\n\n" + prompt
282
+ )
283
+ return _build_opencode_cmd(backend, prompt, model, agent)
276
284
  cmd = list(backend) + [prompt, "--tools", tools]
277
285
  if model:
278
286
  cmd.extend(["--model", str(model)])
@@ -303,6 +311,11 @@ class SwarmRunner:
303
311
  text=True,
304
312
  check=True,
305
313
  cwd=str(PROJECT_ROOT),
314
+ # stdin MUST be DEVNULL: `opencode run` reads piped stdin to
315
+ # EOF before starting, and the MCP server's inherited stdin
316
+ # pipe is held open by the harness — every agent hung forever
317
+ # (observed live: 10+ min, 1s CPU, no network I/O).
318
+ stdin=subprocess.DEVNULL,
306
319
  shell=False,
307
320
  )
308
321
  if family == "opencode":
@@ -103,13 +103,29 @@ class TestRunnerBackendWiring(unittest.TestCase):
103
103
  r = self._runner(mode="scout")
104
104
  cmd = r.build_agent_cmd("obj")
105
105
  self.assertEqual(cmd[:2], ["opencode", "run"])
106
- self.assertIn("oss-scout", cmd)
106
+ # No --agent by default: named agents fail hard unless the user
107
+ # installed the profile ("Agent not found", observed live).
108
+ self.assertNotIn("--agent", cmd)
109
+ self.assertIn("--format", cmd)
110
+
111
+ def test_explicit_opencode_agent_config_opt_in(self):
112
+ with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "opencode"}):
113
+ r = self._runner(agent_overrides=None)
114
+ r.config["opencode_agent"] = "brainstormer"
115
+ cmd = r.build_agent_cmd("obj")
116
+ self.assertIn("--agent", cmd)
117
+ self.assertIn("brainstormer", cmd)
118
+
119
+ def test_system_prompt_inlined_for_opencode(self):
120
+ import tempfile as _tf
107
121
 
108
- def test_research_beta_maps_to_beta_redteam(self):
109
122
  with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "opencode"}):
110
123
  r = self._runner(mode="research")
111
- self.assertIn("beta-redteam", r.build_agent_cmd("obj", role="beta"))
112
- self.assertIn("alpha-thesis", r.build_agent_cmd("obj", role="alpha"))
124
+ spf = Path(_tf.mkdtemp()) / "sys.md"
125
+ spf.write_text("SYSTEM PROMPT BODY", encoding="utf-8")
126
+ cmd = r.build_agent_cmd("obj", system_prompt_file=spf)
127
+ self.assertIn("SYSTEM PROMPT BODY", cmd[2])
128
+ self.assertNotIn("--system-prompt", cmd)
113
129
 
114
130
  def test_explicit_override_beats_host_env(self):
115
131
  with mock.patch.dict(os.environ, {"IUMBTEMS_HOST": "opencode"}):
@@ -119,6 +135,17 @@ class TestRunnerBackendWiring(unittest.TestCase):
119
135
 
120
136
 
121
137
  class TestOpencodeOutputParsing(unittest.TestCase):
138
+ def test_extracts_real_part_text_events(self):
139
+ """Live shape: {"type":"text","part":{"type":"text","text":"..."}}."""
140
+ raw = "\n".join(
141
+ [
142
+ '{"type":"step_start","timestamp":1,"sessionID":"s","part":{"type":"step-start"}}',
143
+ '{"type":"text","timestamp":2,"sessionID":"s","part":{"id":"p1","type":"text","text":"pong"}}',
144
+ '{"type":"text","timestamp":3,"sessionID":"s","part":{"id":"p2","type":"text","text":"second"}}',
145
+ ]
146
+ )
147
+ self.assertEqual(_extract_opencode_text(raw), "pong\nsecond")
148
+
122
149
  def test_extracts_text_events(self):
123
150
  raw = "\n".join(
124
151
  [
@@ -139,6 +166,24 @@ class TestOpencodeOutputParsing(unittest.TestCase):
139
166
  self.assertEqual(_extract_opencode_text(raw), "verdict: clean-room")
140
167
 
141
168
 
169
+ class TestSpawnHardening(unittest.TestCase):
170
+ def test_stdin_is_devnull_for_backend_spawns(self):
171
+ """Regression: held-open stdin pipe made `opencode run` hang forever."""
172
+ import tempfile
173
+ from runner.research_swarm import SwarmRunner
174
+
175
+ r = SwarmRunner(
176
+ base_dir=Path(tempfile.mkdtemp()), mock_mode=False, mode="research"
177
+ )
178
+ fake = mock.Mock()
179
+ fake.stdout = "ok"
180
+ with mock.patch("subprocess.run", return_value=fake) as run_mock:
181
+ r.run_claude_process("objective text")
182
+ kwargs = run_mock.call_args.kwargs
183
+ self.assertEqual(kwargs.get("stdin"), subprocess.DEVNULL)
184
+ self.assertEqual(kwargs.get("shell"), False)
185
+
186
+
142
187
  class TestDossierStdoutFallback(unittest.TestCase):
143
188
  def _runner(self):
144
189
  import tempfile
@@ -2,6 +2,7 @@
2
2
  """Tests for the factory loop: run-state helper, QA retry bounds, plugin surface."""
3
3
 
4
4
  import json
5
+ import os
5
6
  import subprocess
6
7
  import sys
7
8
  import tempfile
@@ -164,6 +165,52 @@ class TestFactoryHelper(unittest.TestCase):
164
165
  self.assertIn("./skills/factory", pkg["pi"]["skills"])
165
166
  self.assertIn("./skills/factory", pkg["omp"]["skills"])
166
167
 
168
+ def test_project_dir_env_resolution(self):
169
+ """Factory state must land in the target project, never the checkout."""
170
+ with tempfile.TemporaryDirectory() as tmp:
171
+ env = dict(os.environ, IUMBTEMS_PROJECT_DIR=tmp)
172
+ r = subprocess.run(
173
+ [
174
+ sys.executable,
175
+ "skills/factory/scripts/factory.py",
176
+ "init",
177
+ "--run",
178
+ "env-run",
179
+ ],
180
+ capture_output=True,
181
+ text=True,
182
+ cwd=str(PROJECT_ROOT),
183
+ env=env,
184
+ )
185
+ self.assertEqual(r.returncode, 0, r.stderr)
186
+ self.assertTrue(
187
+ (Path(tmp) / ".factory" / "env-run" / "state.json").exists()
188
+ )
189
+
190
+ def test_mcp_factory_tool_project_dir(self):
191
+ """iumbtems_factory drives state in the given project (no script path)."""
192
+ with tempfile.TemporaryDirectory() as tmp:
193
+ r = subprocess.run(
194
+ [
195
+ sys.executable,
196
+ "runner/mcp_server.py",
197
+ "call",
198
+ "iumbtems_factory",
199
+ json.dumps(
200
+ {"command": "init", "run": "mcp-run", "project_dir": tmp}
201
+ ),
202
+ ],
203
+ capture_output=True,
204
+ text=True,
205
+ cwd=str(PROJECT_ROOT),
206
+ )
207
+ self.assertEqual(r.returncode, 0, r.stderr)
208
+ payload = json.loads(r.stdout)
209
+ self.assertEqual(payload["status"], "ok")
210
+ self.assertTrue(
211
+ (Path(tmp) / ".factory" / "mcp-run" / "state.json").exists()
212
+ )
213
+
167
214
  def test_snippet_factory_roster(self):
168
215
  with open(PROJECT_ROOT / "config" / "opencode-snippet.json") as f:
169
216
  snippet = json.load(f)
@@ -28,6 +28,7 @@ EXPECTED_TOOLS = [
28
28
  "iumbtems_oss_scout",
29
29
  "iumbtems_brainstorm",
30
30
  "iumbtems_darkharvest",
31
+ "iumbtems_factory",
31
32
  "iumbtems_verify_quote",
32
33
  "iumbtems_socratic_frontier",
33
34
  "iumbtems_reindex_claims",
@@ -214,7 +214,7 @@ class TestOpenCodeCommandCatalog(unittest.TestCase):
214
214
  self.assertEqual(res.returncode, 0, f"tool map test failed: {res.stderr}")
215
215
  data = last_json_object(res.stdout)
216
216
  self.assertEqual(sorted(data["keys"]), sorted(data["canonical"]))
217
- self.assertEqual(len(data["keys"]), 14)
217
+ self.assertEqual(len(data["keys"]), 15)
218
218
  self.assertTrue(all(data["ok"]))
219
219
 
220
220
 
@@ -618,7 +618,7 @@ class TestOpenCodeV2Transforms(unittest.TestCase):
618
618
  self.assertIn(c, data["commands"])
619
619
  self.assertNotIn("goal", data["commands"])
620
620
  self.assertEqual(data["cmdExec"], "function")
621
- self.assertEqual(len(data["tools"]), 14)
621
+ self.assertEqual(len(data["tools"]), 15)
622
622
  self.assertIn("iumbtems_brainstorm", data["tools"])
623
623
  self.assertIn("iumbtems_darkharvest", data["tools"])
624
624
  self.assertEqual(data["toolExec"], "function")
@@ -381,6 +381,7 @@ In our experiments, the 70B parameter model was trained on 15.0 trillion tokens.
381
381
  "iumbtems_oss_scout",
382
382
  "iumbtems_brainstorm",
383
383
  "iumbtems_darkharvest",
384
+ "iumbtems_factory",
384
385
  "iumbtems_verify_quote",
385
386
  "iumbtems_socratic_frontier",
386
387
  "iumbtems_reindex_claims",
@@ -13,6 +13,16 @@ description: Coding-factory Manager loop. Use when user invokes /factory or /dom
13
13
  - **qa-a / qa-b** (subagents): same model, DIVERGED prompts (functional-correctness vs adversarial edge-case). Read-only plus test execution; never edit.
14
14
  - **researcher** = existing `iumbtems_brainstorm` + `iumbtems_darkharvest` swarms (no new profile).
15
15
 
16
+ ## 1b. Driving run state (no filesystem paths)
17
+
18
+ Use the **`iumbtems_factory` MCP tool** for all run-state changes — never a
19
+ relative `skills/factory/scripts/factory.py` path (the toolchain lives in the
20
+ npm cache in consuming projects, and relative paths broke live: the manager
21
+ agent ran `find / -name factory.py`). The tool wraps the same helper and
22
+ resolves the project via `project_dir` argument, `IUMBTEMS_PROJECT_DIR`, or
23
+ the session cwd. It supports `init`, `phase-add`, `qa-record`, `expansion`,
24
+ `stop`, and returns `status: escalated` (exit 2) on the 3rd QA failure.
25
+
16
26
  ## 2. Gate protocol (max 5 swarm cycles per gate)
17
27
 
18
28
  1. Grill until `.factory/frontier.json` settled (grilling skill).
@@ -2,8 +2,13 @@
2
2
  """
3
3
  Factory run-state helper: phase dossiers, QA retry bounds, expansion loop guard.
4
4
 
5
- All state lives under .factory/ (gitignored runtime state). Phase output goes
6
- to .roadmap/<phase>/. Evidence stays in .research/. Read-only w.r.t. repo code.
5
+ All state lives under <project>/.factory/ (gitignored runtime state). Phase
6
+ output goes to <project>/.roadmap/<phase>/. Evidence stays in <project>/
7
+ .research/. Read-only w.r.t. repo code.
8
+
9
+ Project directory resolution: --project-dir > IUMBTEMS_PROJECT_DIR env > the
10
+ toolchain repo root (local-dev default). A consuming project must never leak
11
+ state into the IUMBTEMS checkout.
7
12
 
8
13
  Usage:
9
14
  python3 skills/factory/scripts/factory.py init --run <name>
@@ -15,13 +20,25 @@ Usage:
15
20
 
16
21
  import argparse
17
22
  import json
23
+ import os
18
24
  import sys
19
25
  from datetime import datetime, timezone
20
26
  from pathlib import Path
21
27
 
22
- PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent.parent
23
- FACTORY_DIR = PROJECT_ROOT / ".factory"
24
- ROADMAP_DIR = PROJECT_ROOT / ".roadmap"
28
+ REPO_ROOT = Path(__file__).resolve().parent.parent.parent.parent
29
+
30
+
31
+ def resolve_project_root(cli_value=None):
32
+ """--project-dir > IUMBTEMS_PROJECT_DIR > repo root (local-dev default)."""
33
+ for candidate in (cli_value, os.environ.get("IUMBTEMS_PROJECT_DIR")):
34
+ if candidate and Path(candidate).is_dir():
35
+ return Path(candidate).resolve()
36
+ return REPO_ROOT
37
+
38
+
39
+ PROJECT_DIR = resolve_project_root()
40
+ FACTORY_DIR = PROJECT_DIR / ".factory"
41
+ ROADMAP_DIR = PROJECT_DIR / ".roadmap"
25
42
 
26
43
  MAX_QA_RETRIES = 3
27
44
  MAX_EXPANSION_LOOPS = 10
@@ -173,31 +190,48 @@ def cmd_stop(args):
173
190
  print(f"🛑 STOP file written for run '{args.run}'.")
174
191
 
175
192
 
193
+ def _add_common(parser):
194
+ parser.add_argument(
195
+ "--project-dir",
196
+ default=None,
197
+ help="Project the factory state belongs to (default: IUMBTEMS_PROJECT_DIR env, else repo root)",
198
+ )
199
+
200
+
176
201
  def main():
202
+ global FACTORY_DIR, ROADMAP_DIR
177
203
  ap = argparse.ArgumentParser(description="Factory run-state helper")
178
204
  sub = ap.add_subparsers(dest="command", required=True)
179
205
 
180
206
  p = sub.add_parser("init")
181
207
  p.add_argument("--run", required=True)
208
+ _add_common(p)
182
209
  p = sub.add_parser("phase-add")
183
210
  p.add_argument("--run", required=True)
184
211
  p.add_argument("--phase", required=True)
185
212
  p.add_argument("--goal", required=True)
186
213
  p.add_argument("--accept", default="")
214
+ _add_common(p)
187
215
  p = sub.add_parser("qa-record")
188
216
  p.add_argument("--run", required=True)
189
217
  p.add_argument("--phase", required=True)
190
218
  p.add_argument("--seat", required=True)
191
219
  p.add_argument("--verdict", required=True)
192
220
  p.add_argument("--reason", default="")
221
+ _add_common(p)
193
222
  p = sub.add_parser("expansion")
194
223
  p.add_argument("--run", required=True)
195
224
  p.add_argument("--loops", type=int, required=True)
196
225
  p.add_argument("--max-loops", type=int, default=MAX_EXPANSION_LOOPS)
226
+ _add_common(p)
197
227
  p = sub.add_parser("stop")
198
228
  p.add_argument("--run", required=True)
229
+ _add_common(p)
199
230
 
200
231
  args = ap.parse_args()
232
+ root = resolve_project_root(args.project_dir)
233
+ FACTORY_DIR = root / ".factory"
234
+ ROADMAP_DIR = root / ".roadmap"
201
235
  code = {
202
236
  "init": cmd_init,
203
237
  "phase-add": cmd_phase_add,