@heretek-ai/epistemic-swarm 0.7.4 → 0.7.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. package/.claude-plugin/plugin.json +1 -1
  2. package/README.md +10 -9
  3. package/config/opencode-snippet.json +5 -5
  4. package/package.json +1 -1
  5. package/plugins/opencode/index.js +234 -1
  6. package/prompts/antigravity_repo_factcheck.md +143 -0
  7. package/runner/__pycache__/__init__.cpython-311.pyc +0 -0
  8. package/runner/__pycache__/auctioneer.cpython-311.pyc +0 -0
  9. package/runner/__pycache__/auditor_engine.cpython-311.pyc +0 -0
  10. package/runner/__pycache__/claim_store.cpython-311.pyc +0 -0
  11. package/runner/__pycache__/claim_witness.cpython-311.pyc +0 -0
  12. package/runner/__pycache__/living_dossiers.cpython-311.pyc +0 -0
  13. package/runner/__pycache__/mcp_protocol.cpython-311.pyc +0 -0
  14. package/runner/__pycache__/mcp_server.cpython-311.pyc +0 -0
  15. package/runner/__pycache__/path_safety.cpython-311.pyc +0 -0
  16. package/runner/__pycache__/pcrb.cpython-311.pyc +0 -0
  17. package/runner/__pycache__/pcrb_verify.cpython-311.pyc +0 -0
  18. package/runner/__pycache__/refinement.cpython-311.pyc +0 -0
  19. package/runner/__pycache__/research_swarm.cpython-311.pyc +0 -0
  20. package/runner/__pycache__/state_machine.cpython-311.pyc +0 -0
  21. package/runner/tests/__pycache__/test_auction_order.cpython-311.pyc +0 -0
  22. package/runner/tests/__pycache__/test_backends.cpython-311.pyc +0 -0
  23. package/runner/tests/__pycache__/test_bet1_spike.cpython-311.pyc +0 -0
  24. package/runner/tests/__pycache__/test_claim_store.cpython-311.pyc +0 -0
  25. package/runner/tests/__pycache__/test_claim_witness.cpython-311.pyc +0 -0
  26. package/runner/tests/__pycache__/test_claude_plugin.cpython-311.pyc +0 -0
  27. package/runner/tests/__pycache__/test_domain_packs.cpython-311.pyc +0 -0
  28. package/runner/tests/__pycache__/test_factory.cpython-311.pyc +0 -0
  29. package/runner/tests/__pycache__/test_fleet_seam.cpython-311.pyc +0 -0
  30. package/runner/tests/__pycache__/test_living_dossiers.cpython-311.pyc +0 -0
  31. package/runner/tests/__pycache__/test_mcp_server.cpython-311.pyc +0 -0
  32. package/runner/tests/__pycache__/test_opencode_ux.cpython-311.pyc +0 -0
  33. package/runner/tests/__pycache__/test_opencode_v2_registrars.cpython-311.pyc +0 -0
  34. package/runner/tests/__pycache__/test_pcrb.cpython-311.pyc +0 -0
  35. package/runner/tests/__pycache__/test_refinement.cpython-311.pyc +0 -0
  36. package/runner/tests/__pycache__/test_swarm.cpython-311.pyc +0 -0
  37. package/runner/tests/__pycache__/test_sweep_regressions.cpython-311.pyc +0 -0
  38. package/runner/tests/__pycache__/test_webcache.cpython-311.pyc +0 -0
  39. package/runner/tests/test_factory.py +4 -0
  40. package/runner/tests/test_opencode_v2_registrars.py +163 -0
  41. package/scripts/__pycache__/bet1_advisory_spike.cpython-311.pyc +0 -0
  42. package/scripts/__pycache__/build_adapters.cpython-311.pyc +0 -0
  43. package/scripts/__pycache__/divergence_experiment.cpython-311.pyc +0 -0
  44. package/skills/epistemic_search/scripts/__pycache__/search.cpython-311.pyc +0 -0
  45. package/skills/research_cache/__pycache__/__init__.cpython-311.pyc +0 -0
  46. package/skills/research_cache/__pycache__/hasher.cpython-311.pyc +0 -0
  47. package/skills/swarm_config/__pycache__/__init__.cpython-311.pyc +0 -0
  48. package/skills/swarm_config/__pycache__/configure.cpython-311.pyc +0 -0
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "epistemic-swarm",
3
- "version": "0.7.4",
3
+ "version": "0.7.6",
4
4
  "description": "High-Integrity Dialectic Research Agent Harness for Claude Code enforcing empirical evidence over parametric hallucination.",
5
5
  "author": {
6
6
  "name": "Heretek AI",
package/README.md CHANGED
@@ -75,19 +75,20 @@ pi install npm:@heretek-ai/epistemic-swarm
75
75
  - OMP (`omp.sh`, oh-my-pi) shares the same entry point: `omp install npm:@heretek-ai/epistemic-swarm`, project commands in `.omp/commands/` (`/swarm`, `/grill`, `/audit`, `/scout`, `/brainstorming`, `/swarm-config`), prompts in `.omp/prompts/`, hooks in `.omp/hooks/pre|post/`.
76
76
  - All commands automatically respect `.research/config.json`.
77
77
 
78
- ### 3. OpenCode V2 (`opencode.ai`)
79
- Enable IUMBTEMS in your `~/.config/opencode/opencode.json` or project `opencode.jsonc`. You can configure settings declaratively:
80
- ```json
78
+ ### 3. OpenCode V2 ([opencode.ai/v2/docs](https://opencode.ai/v2/docs))
79
+ Enable IUMBTEMS in your `~/.config/opencode/opencode.jsonc` or project `opencode.jsonc`. You can configure settings declaratively using native OpenCode V2 syntax:
80
+ ```jsonc
81
81
  {
82
- "plugin": [
83
- [
84
- "@heretek-ai/epistemic-swarm",
85
- {
82
+ "$schema": "https://opencode.ai/config.json",
83
+ "plugins": [
84
+ {
85
+ "package": "@heretek-ai/epistemic-swarm",
86
+ "options": {
86
87
  "search_engine": "duckduckgo",
87
88
  "max_iterations": 2,
88
89
  "mode": "research"
89
90
  }
90
- ]
91
+ }
91
92
  ]
92
93
  }
93
94
  ```
@@ -197,7 +198,7 @@ Every factual claim in IUMBTEMS carries an explicit evidentiary tag:
197
198
  1. **Discovery Tier**: SearXNG (unbiased metasearch) and Brave Search API.
198
199
  2. **Extraction Tier**: Firecrawl (headless JavaScript rendering, DOM cleaning, Markdown extraction).
199
200
  3. **Academic Tier**: Semantic Scholar / arXiv MCPs for DOI citation resolution.
200
- 4. **Caching Tier**: Content-addressed SHA-256 storage (`skills/research-cache/hasher.py`).
201
+ 4. **Caching Tier**: Content-addressed SHA-256 storage (`skills/research_cache/hasher.py`).
201
202
 
202
203
  ### Local Infrastructure (Optional)
203
204
  Run local SearXNG and Firecrawl instances via Docker Compose:
@@ -7,15 +7,15 @@
7
7
  "enabled": true
8
8
  }
9
9
  },
10
- "plugin": [
11
- [
12
- "@heretek-ai/epistemic-swarm",
13
- {
10
+ "plugins": [
11
+ {
12
+ "package": "@heretek-ai/epistemic-swarm",
13
+ "options": {
14
14
  "search_engine": "duckduckgo",
15
15
  "max_iterations": 2,
16
16
  "mode": "research"
17
17
  }
18
- ]
18
+ }
19
19
  ],
20
20
  "agent": {
21
21
  "code-auditor": {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@heretek-ai/epistemic-swarm",
3
- "version": "0.7.4",
3
+ "version": "0.7.6",
4
4
  "description": "IUMBTEMS: I Use My Brain To Express My Self — High-Integrity Dialectic Research Agent Harness for Claude Code, OpenCode V2, Pi, OMP (oh-my-pi), Gemini CLI, Codex CLI, and AntiGravity",
5
5
  "main": "bin/cli.js",
6
6
  "bin": {
@@ -577,6 +577,7 @@ export const OPENCODE_COMMANDS = [
577
577
  description: 'Coding-factory Manager loop: grill-gated phased build with programmer spawns and dual QA',
578
578
  usage: '/factory <product-arena>',
579
579
  agent: 'manager',
580
+ subagent: false,
580
581
  subtask: false,
581
582
  template: [
582
583
  'Run the IUMBTEMS coding-factory Manager loop as the manager agent.',
@@ -586,6 +587,7 @@ export const OPENCODE_COMMANDS = [
586
587
  '2. Per gate run iumbtems_brainstorm and iumbtems_darkharvest (mock_mode only for dry runs), then synthesize .roadmap/<phase>/ GOAL.md + dossier.json (goal/evidence/acceptance/brief/verdict/hashes; every claim needs a VERIFIED hash).',
587
588
  '3. Spawn the programmer subagent per phase with the phase dossier (cite phase hashes); run qa-a and qa-b (diverged prompts) per phase; track retries with the iumbtems_factory tool (command: phase-add / qa-record; 3 failures escalate to manager). Never invoke factory helper scripts by relative path.',
588
589
  '4. Manager tiebreaks QA disagreements; explicit user sign-off closes each phase.',
590
+ 'Swarm agents use the host-native backend automatically (claude -p on Claude Code, opencode run on OpenCode). Do not debug backend selection, config pins, or binary availability as part of a gate — if a swarm call fails, report the error and continue.',
589
591
  ].join('\n'),
590
592
  },
591
593
  {
@@ -593,6 +595,7 @@ export const OPENCODE_COMMANDS = [
593
595
  description: 'Autonomous agent-guided self-improvement loop over the codebase (count-flagged)',
594
596
  usage: '/domainexpansion <n>',
595
597
  agent: 'manager',
598
+ subagent: false,
596
599
  subtask: false,
597
600
  template: [
598
601
  'Run the IUMBTEMS domain-expansion loop as the manager agent.',
@@ -619,7 +622,14 @@ export function commandCatalog() {
619
622
  if (!cmd?.name) continue;
620
623
  out[cmd.name] = { description: cmd.description, template: cmd.template };
621
624
  if (cmd.agent) out[cmd.name].agent = cmd.agent;
625
+ if (cmd.subagent !== undefined) out[cmd.name].subagent = cmd.subagent;
622
626
  if (cmd.subtask !== undefined) out[cmd.name].subtask = cmd.subtask;
627
+ if (out[cmd.name].subagent === undefined && out[cmd.name].subtask !== undefined) {
628
+ out[cmd.name].subagent = out[cmd.name].subtask;
629
+ }
630
+ if (out[cmd.name].subtask === undefined && out[cmd.name].subagent !== undefined) {
631
+ out[cmd.name].subtask = out[cmd.name].subagent;
632
+ }
623
633
  }
624
634
  return out;
625
635
  }
@@ -889,13 +899,28 @@ async function registerHostCommands(host) {
889
899
  name: cmd.name,
890
900
  description: cmd.description,
891
901
  ...(cmd.agent ? { agent: cmd.agent } : {}),
892
- ...(cmd.subtask !== undefined ? { subtask: cmd.subtask } : {}),
902
+ ...(cmd.subagent !== undefined || cmd.subtask !== undefined
903
+ ? {
904
+ subagent: cmd.subagent !== undefined ? cmd.subagent : cmd.subtask,
905
+ subtask: cmd.subtask !== undefined ? cmd.subtask : cmd.subagent,
906
+ }
907
+ : {}),
893
908
  execute: async (input) => {
894
909
  const args = input?.prompt?.text || '';
895
910
  const prompt = (typeof input?.prompt === 'object' && input?.prompt !== null) ? input.prompt : {};
911
+ // Pin the declared agent for this run: the template says "as the
912
+ // manager agent", so make that true rather than aspirational.
913
+ let switched = false;
914
+ if (cmd.agent) {
915
+ switched = await switchSessionAgent(host, input?.sessionID, cmd.agent);
916
+ log(host, 'debug', 'command agent switch', {
917
+ command: cmd.name, agent: cmd.agent, switched,
918
+ });
919
+ }
896
920
  await host.session.prompt({
897
921
  ...prompt,
898
922
  sessionID: input?.sessionID,
923
+ ...(cmd.agent ? { agent: cmd.agent } : {}),
899
924
  text: cmd.template.split('$ARGUMENTS').join(String(args).trim()),
900
925
  delivery: input?.delivery,
901
926
  });
@@ -996,6 +1021,174 @@ async function registerCompactionHook(host, context) {
996
1021
  }
997
1022
  }
998
1023
 
1024
+ async function registerHostAgents(host) {
1025
+ // V2 Context exposes `agent.transform` (AgentDomain). Registering the IUMBTEMS
1026
+ // role profiles here removes the config-snippet install dependency: without
1027
+ // them the `/factory` command ran as the generic `build` agent (observed live:
1028
+ // every step of a factory session executed as `build`, not `manager`).
1029
+ if (typeof host?.agent?.transform !== 'function') return [];
1030
+ const snippet = readPkgJson(path.join(PKG_ROOT, 'config', 'opencode-snippet.json'));
1031
+ const agentDefs = (snippet && snippet.agent) || {};
1032
+ if (Object.keys(agentDefs).length === 0) return [];
1033
+
1034
+ let existing = new Set();
1035
+ try {
1036
+ const listed = await host.agent.list();
1037
+ existing = new Set((listed?.data || listed || []).map((a) => a?.id || a?.name));
1038
+ } catch (err) {
1039
+ log(host, 'warn', 'agent.list() failed; assuming an empty registry', errDetail(err));
1040
+ }
1041
+
1042
+ const registration = await host.agent.transform((draft) => {
1043
+ const claimed = new Set(existing);
1044
+ const added = [];
1045
+ for (const [name, def] of Object.entries(agentDefs)) {
1046
+ if (claimed.has(name)) continue;
1047
+ claimed.add(name);
1048
+ added.push(name);
1049
+ draft.add(toAgentInfo(name, def));
1050
+ }
1051
+ log(host, 'debug', 'agent.transform pass', { added });
1052
+ });
1053
+ try {
1054
+ if (typeof host.agent.reload === 'function') await host.agent.reload();
1055
+ log(host, 'info', 'agent profiles registered', { count: Object.keys(agentDefs).length });
1056
+ } catch (err) {
1057
+ log(host, 'warn', 'agent.reload() failed', errDetail(err));
1058
+ }
1059
+ return registration ? [registration] : [];
1060
+ }
1061
+
1062
+ /** Short role system prompts for the factory seats (skills carry the long form). */
1063
+ const FACTORY_SYSTEM = {
1064
+ manager:
1065
+ 'You are the IUMBTEMS Factory Manager. Own gate discipline: grill until the frontier is settled, run the brainstorm and darkharvest swarms, synthesize .roadmap phase dossiers, spawn the programmer per phase, run qa-a and qa-b, and tiebreak their disagreements. Never write implementation code. Use the iumbtems_factory tool for run state. Explicit user approval advances each gate.',
1066
+ programmer:
1067
+ 'You are the IUMBTEMS Factory Programmer. Implement exactly one phase brief per spawn. Cite phase evidence hashes. Never invoke swarms or other programmers. If the brief is ambiguous or untestable, stop and ask the manager.',
1068
+ 'qa-a':
1069
+ 'You are the IUMBTEMS Factory functional QA. Verify each phase acceptance criterion on the real surface with tests and inspection. Read-only plus test execution; never edit code. Return pass|fail(reason)|conditional(note).',
1070
+ 'qa-b':
1071
+ 'You are the IUMBTEMS Factory adversarial QA. Attack the phase: edge cases, regressions, vacuous acceptance criteria, error paths, resource limits. Read-only plus test execution; never edit code. Return pass|fail(reason)|conditional(note) with reproductions.',
1072
+ };
1073
+
1074
+ /** Map one snippet agent definition to an Agent.Info shape. */
1075
+ function toAgentInfo(name, def) {
1076
+ const permissions = [];
1077
+ for (const [tool, enabled] of Object.entries(def?.tools || {})) {
1078
+ permissions.push({ action: tool, resource: '*', effect: enabled ? 'allow' : 'deny' });
1079
+ }
1080
+ for (const [key, value] of Object.entries(def?.permission || {})) {
1081
+ if (typeof value === 'string') {
1082
+ permissions.push({ action: key, resource: '*', effect: value });
1083
+ } else if (value && typeof value === 'object') {
1084
+ for (const [resource, effect] of Object.entries(value)) {
1085
+ permissions.push({ action: key, resource, effect });
1086
+ }
1087
+ }
1088
+ }
1089
+ const info = {
1090
+ id: name,
1091
+ name,
1092
+ description: def?.description || `IUMBTEMS ${name}`,
1093
+ mode: def?.mode || 'all',
1094
+ hidden: Boolean(def?.hidden),
1095
+ request: { settings: {}, headers: {}, body: {} },
1096
+ permissions,
1097
+ };
1098
+ if (def?.steps) info.steps = def.steps;
1099
+ if (def?.color) info.color = def.color;
1100
+ const system = FACTORY_SYSTEM[name];
1101
+ if (system) info.system = system;
1102
+ return info;
1103
+ }
1104
+
1105
+ /** Canonical skills the plugin registers so agents never hunt the filesystem. */
1106
+ const BUNDLED_SKILLS = [
1107
+ 'factory',
1108
+ 'darkharvest',
1109
+ 'brainstorming',
1110
+ 'grilling',
1111
+ 'swarm_config',
1112
+ 'code_audit',
1113
+ 'oss_scout',
1114
+ 'research_cache',
1115
+ 'epistemic_search',
1116
+ ];
1117
+
1118
+ function parseSkillFrontmatter(text) {
1119
+ const out = {};
1120
+ if (!text.startsWith('---')) return out;
1121
+ const end = text.indexOf('\n---', 3);
1122
+ if (end < 0) return out;
1123
+ for (const line of text.slice(3, end).split('\n')) {
1124
+ const idx = line.indexOf(':');
1125
+ if (idx > 0) out[line.slice(0, idx).trim()] = line.slice(idx + 1).trim();
1126
+ }
1127
+ return out;
1128
+ }
1129
+
1130
+ async function registerHostSkills(host) {
1131
+ // V2 Context exposes `skill.transform` (SkillDomain). Live evidence: with no
1132
+ // skills registered, a factory agent ran `find / -name factory.py` and read
1133
+ // the skill prose out of the CLAUDE plugin cache to learn its own mechanics.
1134
+ if (typeof host?.skill?.transform !== 'function') return [];
1135
+ const skills = [];
1136
+ for (const name of BUNDLED_SKILLS) {
1137
+ const skillPath = path.join(PKG_ROOT, 'skills', name, 'SKILL.md');
1138
+ let content;
1139
+ try {
1140
+ content = readFileSync(skillPath, 'utf-8');
1141
+ } catch {
1142
+ continue;
1143
+ }
1144
+ const fm = parseSkillFrontmatter(content);
1145
+ skills.push({
1146
+ id: name,
1147
+ name,
1148
+ description: fm.description || `IUMBTEMS ${name} skill`,
1149
+ path: skillPath,
1150
+ content,
1151
+ autoinvoke: true,
1152
+ });
1153
+ }
1154
+ if (skills.length === 0) return [];
1155
+
1156
+ let existing = new Set();
1157
+ try {
1158
+ const listed = await host.skill.list();
1159
+ existing = new Set((listed?.data || listed || []).map((s) => s?.id || s?.name));
1160
+ } catch (err) {
1161
+ log(host, 'warn', 'skill.list() failed; assuming an empty registry', errDetail(err));
1162
+ }
1163
+
1164
+ const registration = await host.skill.transform((draft) => {
1165
+ const claimed = new Set(existing);
1166
+ const added = [];
1167
+ for (const skill of skills) {
1168
+ if (claimed.has(skill.id)) continue;
1169
+ claimed.add(skill.id);
1170
+ added.push(skill.id);
1171
+ draft.add(skill);
1172
+ }
1173
+ log(host, 'debug', 'skill.transform pass', { added });
1174
+ });
1175
+ try {
1176
+ if (typeof host.skill.reload === 'function') await host.skill.reload();
1177
+ log(host, 'info', 'skills registered', { count: skills.length });
1178
+ } catch (err) {
1179
+ log(host, 'warn', 'skill.reload() failed', errDetail(err));
1180
+ }
1181
+ return registration ? [registration] : [];
1182
+ }
1183
+
1184
+ function readPkgJson(file) {
1185
+ try {
1186
+ return JSON.parse(readFileSync(file, 'utf-8'));
1187
+ } catch {
1188
+ return undefined;
1189
+ }
1190
+ }
1191
+
999
1192
  async function applyConfigOptions(host, opts, context) {
1000
1193
  if (!opts || typeof opts !== 'object') return;
1001
1194
  const updates = {};
@@ -1029,6 +1222,36 @@ function startNudgeSubscription(host, opts, context, cleanups) {
1029
1222
  }
1030
1223
  }
1031
1224
 
1225
+ /**
1226
+ * Best-effort session agent switch for commands that pin an agent.
1227
+ *
1228
+ * Live gap: `/factory` declared `agent: 'manager'` but the session kept running
1229
+ * as `build`, so the manager playbook was never loaded. The V2 SessionDomain
1230
+ * exposes `switchAgent`; its exact call shape is not pinned in the public types,
1231
+ * so try the plausible shapes and never throw into the host.
1232
+ */
1233
+ async function switchSessionAgent(host, sessionID, agentId) {
1234
+ if (!agentId || !sessionID) return false;
1235
+ const fn = host?.session?.switchAgent;
1236
+ if (typeof fn !== 'function') return false;
1237
+ const attempts = [
1238
+ () => fn.call(host.session, { sessionID, agent: agentId }),
1239
+ () => fn.call(host.session, { sessionID, agentID: agentId }),
1240
+ () => fn.call(host.session, sessionID, agentId),
1241
+ () => fn.call(host.session, agentId),
1242
+ ];
1243
+ for (const attempt of attempts) {
1244
+ try {
1245
+ const res = attempt();
1246
+ if (res && typeof res.then === 'function') await res;
1247
+ return true;
1248
+ } catch {
1249
+ /* try the next call shape */
1250
+ }
1251
+ }
1252
+ return false;
1253
+ }
1254
+
1032
1255
  export function createOpenCodePlugin(context = {}) {
1033
1256
  return {
1034
1257
  id: 'heretek.iumbtems.epistemic-swarm',
@@ -1061,6 +1284,16 @@ export function createOpenCodePlugin(context = {}) {
1061
1284
  } catch (err) {
1062
1285
  log(host, 'error', 'registering slash commands failed', errDetail(err));
1063
1286
  }
1287
+ try {
1288
+ adopt(await registerHostAgents(host));
1289
+ } catch (err) {
1290
+ log(host, 'error', 'registering agent profiles failed', errDetail(err));
1291
+ }
1292
+ try {
1293
+ adopt(await registerHostSkills(host));
1294
+ } catch (err) {
1295
+ log(host, 'error', 'registering skills failed', errDetail(err));
1296
+ }
1064
1297
  try {
1065
1298
  adopt(await registerHostTools(host));
1066
1299
  } catch (err) {
@@ -0,0 +1,143 @@
1
+ # IUMBTEMS REPOSITORY FACT-CHECKING REVIEW PROMPT FOR ANTIGRAVITY
2
+
3
+ > **Role & Protocol**: You are operating as the **Epistemic Swarm Auditor** within Google AntiGravity. Your mission is to conduct a rigorous, evidentiary, live web-search fact-checking review of the **IUMBTEMS** repository (`https://github.com/Heretek-AI/IUMBTEMS` / local workspace).
4
+ >
5
+ > You are governed by the **Epistemic Integrity Protocol** defined in this repository: your internal parametric memory is strictly quarantined as untrusted heuristic guidance. You are prohibited from presenting unverified parametric recollections as established empirical facts. Every factual assertion must be verified against live reality via web searching and URL extraction.
6
+
7
+ ---
8
+
9
+ ## 1. MANDATORY TAGGING TAXONOMY
10
+
11
+ Every factual statement, version assertion, API specification claim, or benchmark metric MUST carry an explicit epistemic tag:
12
+
13
+ - `[VERIFIED: <URL | "verbatim quote excerpt">]`
14
+ - Backed by live web content fetched during this session via `search_web` or `read_url_content`.
15
+ - Must include the exact URL and an exact verbatim quote substring from the retrieved page.
16
+ - `[INFERRED: <Parent Tags> -> <Deductive Reasoning>]`
17
+ - Deductive conclusion derived directly from cited `[VERIFIED]` premises.
18
+ - `[HYPOTHESIS: <Measurable Falsification Condition>]`
19
+ - Unverified projection or speculation; requires an empirical test that would disprove it.
20
+ - `[NEGATIVE_KNOWLEDGE: <Search Query>]`
21
+ - Rigorous confirmation that an exhaustive live web search yielded zero supporting evidence.
22
+
23
+ ---
24
+
25
+ ## 2. REPOSITORY AUDIT TARGETS
26
+
27
+ Inspect the local codebase (`README.md`, `AGENTS.md`, `MARKETPLACE.md`, `package.json`, `prompts/`, and `plugins/`) and fact-check the following four empirical domains using `search_web` and `read_url_content`:
28
+
29
+ ### Domain 1: Package Registry & Release Veracity
30
+ - **Claims in Repo**: The project is published on npm as `@heretek-ai/epistemic-swarm` under Apache-2.0, providing binary `iumbtems`.
31
+ - **Fact-Checking Action**:
32
+ - Search `https://registry.npmjs.org/@heretek-ai%2Fepistemic-swarm` or search the web for npm package `@heretek-ai/epistemic-swarm`.
33
+ - Verify: Does the package exist on npm? What is the latest published version? Does it match `package.json`? Does it expose the `iumbtems` binary?
34
+
35
+ ### Domain 2: Peer Agent Harness Compatibility Claims
36
+ - **Claims in Repo**:
37
+ 1. **OpenCode V2**: Claims plugin integration in `plugins/opencode/index.js`, using slash commands (`/swarm`, `/grill`, `/audit`), agent profiles (`config/opencode-snippet.json`), and notes that OpenCode lacks a pre-execution webfetch hook.
38
+ 2. **Pi & OMP**: Claims native install via `pi install npm:@heretek-ai/epistemic-swarm` and `omp install npm:@heretek-ai/epistemic-swarm` with command blocks in `package.json`.
39
+ 3. **Claude Code**: Claims marketplace support via `claude plugin marketplace add Heretek-AI/IUMBTEMS` and `.claude-plugin/marketplace.json`.
40
+ - **Fact-Checking Action**:
41
+ - Search official documentation and repos for OpenCode (`opencode.ai`), Pi (`pi.dev`), and Claude Code plugin specs.
42
+ - Verify: Are the configuration formats, CLI command syntaxes, and plugin manifest schemas valid against current upstream specifications?
43
+
44
+ ### Domain 3: Cited Academic & Algorithmic Benchmarks
45
+ - **Claims in Repo** (found in `prompts/base_epistemic_system.md`, `prompts/orchestrator.md`, etc.):
46
+ 1. Llama-3-70B context window (131,072 tokens) and GQA across 8 KV heads (`arXiv:2407.21783`).
47
+ 2. Tip5 hash vs. Poseidon hash SNARK witness generation benchmarks.
48
+ 3. Zero-dependency Raft consensus and DuckDuckGo Lite HTML scraping behavior.
49
+ - **Fact-Checking Action**:
50
+ - Search arXiv and web sources for the cited papers and benchmarks.
51
+ - Verify: Are the numbers, citations, and DOIs authentic, or were any placeholder/synthetic examples presented as real citations?
52
+
53
+ ### Domain 4: License, Security & Dependency Invariants
54
+ - **Claims in Repo**: Apache-2.0 clean-room licensing, permissive-only vendoring, no AGPL/GPL contamination.
55
+ - **Fact-Checking Action**:
56
+ - Inspect dependencies in `package.json` and python scripts.
57
+ - Verify license status of key referenced dependencies via web search.
58
+
59
+ ---
60
+
61
+ ## 3. SINGLE-SESSION AGENTIC EXECUTION WORKFLOW
62
+
63
+ Execute the fact-checking mission autonomously in three sequential phases:
64
+
65
+ ```
66
+ [Phase 1: Alpha (Affirmative)] ──> [Phase 2: Beta (Adversary)] ──> [Phase 3: Epistemic Auditor]
67
+ ```
68
+
69
+ ### Phase 1: Alpha (The Affirmative Grounding)
70
+ 1. Read the local claims in `README.md` and `package.json` using `view_file`.
71
+ 2. Formulate targeted search queries and execute them using `search_web`.
72
+ 3. Fetch full source pages using `read_url_content` for key results.
73
+ 4. Extract verbatim evidence excerpts corroborating the repository's claims.
74
+ 5. Tag all confirmed claims with `[VERIFIED: <URL | "quote">]`.
75
+
76
+ ### Phase 2: Beta (The Adversarial Red Team)
77
+ 1. Execute inverted and adversarial queries to hunt for discrepancies, breaking changes, and invalid claims:
78
+ - `"<package> deprecated"`, `"<command> error"`, `"<paper> critique"`
79
+ - Check if any upstream harness APIs (OpenCode, Pi, Claude Code) have deprecated or altered the interfaces IUMBTEMS relies on.
80
+ - Hunt for missing packages, unfulfilled promises, or exaggerated marketing statements.
81
+ 2. If claimed features or benchmarks cannot be found online, log them as `[NEGATIVE_KNOWLEDGE: <query>]`.
82
+ 3. If an assertion is disproven by current live documentation, document the exact contradiction.
83
+
84
+ ### Phase 3: Epistemic Auditor & Mathematical Synthesis
85
+ 1. Perform character-for-character verification between extracted quotes and source URLs.
86
+ 2. Compute the **Epistemic Score**:
87
+ $$\mathcal{E} = \frac{1.0 \times N_{\text{verified}} + 0.5 \times N_{\text{neg\_knowledge}} - 2.5 \times N_{\text{rejected}}}{N_{\text{verified}} + N_{\text{inferred}} + N_{\text{hypothesis}} + N_{\text{rejected}}}$$
88
+ *(Threshold: $\mathcal{E} \ge 0.65$ to certify empirical grounding).*
89
+ 3. Compute the **Divergence Score**:
90
+ $$D = \frac{|\text{Contradicted Claims}|}{|\text{Total Scope Claims}|}$$
91
+ 4. Output the final synthesis report as a Markdown Artifact or structured response.
92
+
93
+ ---
94
+
95
+ ## 4. OUTPUT FORMAT SPECIFICATION
96
+
97
+ Your final output must follow this structure:
98
+
99
+ ```markdown
100
+ # Epistemic Fact-Checking Audit: IUMBTEMS Repository
101
+
102
+ ## Executive Summary
103
+ - **Overall Verdict**: [CERTIFIED (Score >= 0.65) | AUDIT_WARNING: LOW_EMPIRICAL_GROUNDING]
104
+ - **Epistemic Score ($\mathcal{E}$)**: `<score>`
105
+ - **Dialectic Divergence ($D$)**: `<score>`
106
+ - **Total Claims Audited**: `<count>` (Verified: `<count>`, Rejected: `<count>`, Negative Knowledge: `<count>`)
107
+
108
+ ---
109
+
110
+ ## Evidentiary Audit Ledger
111
+
112
+ ### 1. Package & Distribution Veracity
113
+ - Claim: ...
114
+ - Status: [VERIFIED | REJECTED | NEGATIVE_KNOWLEDGE]
115
+ - Evidence: [VERIFIED: https://... | "Verbatim quote..."]
116
+ - Notes: ...
117
+
118
+ ### 2. Multi-Harness Compatibility (OpenCode, Pi, OMP, Claude Code)
119
+ ...
120
+
121
+ ### 3. Academic & Benchmark Integrity
122
+ ...
123
+
124
+ ### 4. License & Contamination Safety
125
+ ...
126
+
127
+ ---
128
+
129
+ ## Divergence & Contradiction Matrix
130
+ | Dimension | Affirmative Claim (Alpha) | Adversarial Finding (Beta) | Adjudicated Truth |
131
+ | :--- | :--- | :--- | :--- |
132
+ | ... | ... | ... | ... |
133
+
134
+ ---
135
+
136
+ ## Actionable Remediations
137
+ 1. [P0/P1/P2] Specific changes required in `README.md`, `package.json`, or code to align with verified live reality.
138
+ ```
139
+
140
+ ---
141
+
142
+ ## 5. EXECUTION DIRECTIVE
143
+ Begin Phase 1 immediately: inspect local claims, invoke `search_web` to verify npm and harness registries, then proceed through Phase 2 and Phase 3 without stopping.
@@ -238,8 +238,12 @@ class TestFactoryHelper(unittest.TestCase):
238
238
  self.assertEqual(res.returncode, 0, res.stderr)
239
239
  data = last_json_object(res.stdout)
240
240
  self.assertEqual(data["factory"]["agent"], "manager")
241
+ self.assertEqual(data["factory"]["subagent"], False)
242
+ self.assertEqual(data["factory"]["subtask"], False)
241
243
  self.assertIn("$ARGUMENTS", data["factory"]["template"])
242
244
  self.assertEqual(data["expansion"]["agent"], "manager")
245
+ self.assertEqual(data["expansion"]["subagent"], False)
246
+ self.assertEqual(data["expansion"]["subtask"], False)
243
247
 
244
248
 
245
249
  if __name__ == "__main__":
@@ -0,0 +1,163 @@
1
+ #!/usr/bin/env python3
2
+ """Tests for OpenCode V2 agent/skill registration and command agent switching.
3
+
4
+ Live gap (ses_f1b924bdaffe): the /factory command declared `agent: manager`,
5
+ but no agent profiles existed (config snippet never installed) so every step
6
+ ran as the generic `build` agent, and skills were undiscoverable — the agent
7
+ ran `find / -name factory.py` to learn its own mechanics. These tests pin the
8
+ plugin-side fix: profiles and skills are registered through the V2 domains, and
9
+ the command switches the session agent.
10
+ """
11
+
12
+ import json
13
+ import subprocess
14
+ import unittest
15
+ from pathlib import Path
16
+
17
+ PROJECT_ROOT = Path(__file__).resolve().parent.parent.parent
18
+
19
+ V2_HOST = """
20
+ const reg = {agents: [], skills: [], commands: [], tools: [], prompts: [], switches: []};
21
+ const host = {
22
+ options: {},
23
+ location: {directory: '/home/john/Projects/STC'},
24
+ command: {list: async()=>({data:[]}), transform: async(fn)=>{fn({add:(x)=>{reg.commands.push(x);}}); return {dispose:()=>{}};}, reload: async()=>{}},
25
+ tool: {transform: async(fn)=>{fn({add:(t)=>{reg.tools.push(t);}}); return {dispose:()=>{}};}, reload: async()=>{}},
26
+ agent: {list: async()=>({data:[]}), transform: async(fn)=>{fn({add:(a)=>{reg.agents.push(a);}}); return {dispose:()=>{}};}, reload: async()=>{}},
27
+ skill: {list: async()=>({data:[]}), transform: async(fn)=>{fn({add:(s)=>{reg.skills.push(s);}}); return {dispose:()=>{}};}, reload: async()=>{}},
28
+ session: {
29
+ prompt: async(p)=>{reg.prompts.push(p); return {};},
30
+ switchAgent: async(a)=>{reg.switches.push(a); return {};}
31
+ }
32
+ };
33
+ """
34
+
35
+
36
+ def run_node(code):
37
+ return subprocess.run(
38
+ ["node", "--input-type=module", "-e", code],
39
+ capture_output=True,
40
+ text=True,
41
+ cwd=str(PROJECT_ROOT),
42
+ )
43
+
44
+
45
+ def last_json_object(stdout):
46
+ lines = [l.strip() for l in stdout.strip().split("\n") if l.strip().startswith("{")]
47
+ return json.loads(lines[-1])
48
+
49
+
50
+ class TestV2Registration(unittest.TestCase):
51
+ def test_agents_and_skills_registered(self):
52
+ res = run_node(
53
+ V2_HOST
54
+ + """
55
+ import plugin from "./plugins/opencode/index.js";
56
+ await plugin.setup(host);
57
+ const manager = reg.agents.find(a => a.id === 'manager');
58
+ const qa = reg.agents.find(a => a.id === 'qa-a');
59
+ const factory = reg.skills.find(s => s.id === 'factory');
60
+ console.log(JSON.stringify({
61
+ agentIds: reg.agents.map(a => a.id).sort(),
62
+ managerMode: manager?.mode,
63
+ managerSystem: Boolean(manager?.system),
64
+ managerPermissions: manager?.permissions?.length || 0,
65
+ managerDeniesEdit: (manager?.permissions || []).some(p => p.action === 'edit' && p.effect === 'deny'),
66
+ qaMode: qa?.mode,
67
+ skillIds: reg.skills.map(s => s.id).sort(),
68
+ factoryHasContent: Boolean(factory?.content?.includes('Factory')),
69
+ factoryPath: factory?.path,
70
+ }));
71
+ """
72
+ )
73
+ self.assertEqual(res.returncode, 0, res.stderr)
74
+ d = last_json_object(res.stdout)
75
+ for aid in (
76
+ "manager",
77
+ "programmer",
78
+ "qa-a",
79
+ "qa-b",
80
+ "brainstormer",
81
+ "darkharvester",
82
+ ):
83
+ self.assertIn(aid, d["agentIds"])
84
+ self.assertEqual(d["managerMode"], "primary")
85
+ self.assertTrue(d["managerSystem"])
86
+ self.assertTrue(d["managerDeniesEdit"], "manager must not edit code")
87
+ self.assertEqual(d["qaMode"], "subagent")
88
+ self.assertIn("factory", d["skillIds"])
89
+ self.assertIn("darkharvest", d["skillIds"])
90
+ self.assertTrue(d["factoryHasContent"])
91
+ self.assertTrue(d["factoryPath"].endswith("skills/factory/SKILL.md"))
92
+
93
+ def test_command_switches_session_agent(self):
94
+ res = run_node(
95
+ V2_HOST
96
+ + """
97
+ import plugin from "./plugins/opencode/index.js";
98
+ await plugin.setup(host);
99
+ const factory = reg.commands.find(c => c.name === 'factory');
100
+ await factory.execute({sessionID: 'ses_test', prompt: {text: 'Cockpit beta'}, delivery: 'steer'});
101
+ console.log(JSON.stringify({
102
+ switches: reg.switches,
103
+ promptAgent: reg.prompts[0]?.agent,
104
+ promptHasArena: reg.prompts[0]?.text?.includes('Cockpit beta'),
105
+ }));
106
+ """
107
+ )
108
+ self.assertEqual(res.returncode, 0, res.stderr)
109
+ d = last_json_object(res.stdout)
110
+ self.assertTrue(d["switches"], "switchAgent was never called")
111
+ self.assertEqual(d["promptAgent"], "manager")
112
+ self.assertTrue(d["promptHasArena"])
113
+
114
+ def test_preexisting_agent_and_skill_preserved(self):
115
+ res = run_node(
116
+ """
117
+ const agents = [], skills = [];
118
+ const host = {
119
+ options: {}, location: {directory: '/tmp'},
120
+ command: {list: async()=>({data:[]}), transform: async(fn)=>{fn({add:()=>{}}); return {dispose:()=>{}};}, reload: async()=>{}},
121
+ tool: {transform: async()=>({dispose:()=>{}}), reload: async()=>{}},
122
+ agent: {
123
+ list: async()=>({data:[{id:'manager'}]}),
124
+ transform: async(fn)=>{fn({add:(a)=>{agents.push(a.id);}}); return {dispose:()=>{}};},
125
+ reload: async()=>{}
126
+ },
127
+ skill: {
128
+ list: async()=>({data:[{id:'factory'}]}),
129
+ transform: async(fn)=>{fn({add:(s)=>{skills.push(s.id);}}); return {dispose:()=>{}};},
130
+ reload: async()=>{}
131
+ },
132
+ session: {prompt: async()=>({})}
133
+ };
134
+ import plugin from "./plugins/opencode/index.js";
135
+ await plugin.setup(host);
136
+ console.log(JSON.stringify({agents, skills}));
137
+ """
138
+ )
139
+ self.assertEqual(res.returncode, 0, res.stderr)
140
+ d = last_json_object(res.stdout)
141
+ self.assertNotIn("manager", d["agents"], "user-defined manager must win")
142
+ self.assertNotIn("factory", d["skills"], "user-defined factory skill must win")
143
+ self.assertIn("qa-a", d["agents"])
144
+
145
+ def test_graceful_without_agent_skill_domains(self):
146
+ res = run_node(
147
+ """
148
+ import plugin from "./plugins/opencode/index.js";
149
+ const host = {
150
+ options: {}, location: {directory: '/tmp'},
151
+ tool: {transform: async()=>({dispose:()=>{}}), reload: async()=>{}},
152
+ session: {prompt: async()=>({})}
153
+ };
154
+ const cleanup = await plugin.setup(host);
155
+ console.log(JSON.stringify({cleanupFn: typeof cleanup}));
156
+ """
157
+ )
158
+ self.assertEqual(res.returncode, 0, res.stderr)
159
+ self.assertEqual(last_json_object(res.stdout)["cleanupFn"], "function")
160
+
161
+
162
+ if __name__ == "__main__":
163
+ unittest.main()