@claude-flow/cli 3.38.12 → 3.38.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/.claude/.proven-config-version +1 -0
  2. package/.claude/helpers/.helpers-version +1 -1
  3. package/.claude/helpers/helpers.manifest.json +2 -2
  4. package/.claude/helpers/statusline.cjs +0 -0
  5. package/.claude/proven-config.json +42 -0
  6. package/catalog-manifest.json +4 -4
  7. package/dist/src/mcp-tools/hooks-tools.js +6 -1
  8. package/dist/src/ruvector/lattice-wasm.d.ts +14 -0
  9. package/dist/src/ruvector/lattice-wasm.js +144 -0
  10. package/dist/src/services/flywheel-receipt.d.ts +10 -0
  11. package/dist/src/services/flywheel-receipt.js +82 -7
  12. package/dist/src/services/flywheel-transaction.js +10 -1
  13. package/node_modules/@claude-flow/codex/dist/cli.js +0 -0
  14. package/node_modules/@claude-flow/plugin-agent-federation/dist/bin.js +0 -0
  15. package/node_modules/@claude-flow/security/dist/input-validator.d.ts +6 -6
  16. package/package.json +1 -1
  17. package/plugins/ruflo-metaharness/.claude-flow/daemon-state.json +178 -0
  18. package/plugins/ruflo-metaharness/.claude-flow/daemon.pid +1 -0
  19. package/plugins/ruflo-metaharness/.claude-flow/data/pending-insights.jsonl +5 -0
  20. package/plugins/ruflo-metaharness/.claude-flow/logs/daemon.log +269 -0
  21. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783604774864_ozbujc_prompt.log +19 -0
  22. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783604774864_ozbujc_result.log +108 -0
  23. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783605513587_ulvmpb_prompt.log +19 -0
  24. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783605513587_ulvmpb_result.log +209 -0
  25. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783606368867_ahysui_prompt.log +19 -0
  26. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783606368867_ahysui_result.log +192 -0
  27. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783607120257_lh05rb_prompt.log +19 -0
  28. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783607120257_lh05rb_result.log +13 -0
  29. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608020362_j3096j_prompt.log +19 -0
  30. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608020362_j3096j_result.log +120 -0
  31. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608776347_261b61_prompt.log +19 -0
  32. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608776347_261b61_result.log +85 -0
  33. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783609621359_s5i6ye_prompt.log +19 -0
  34. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783609621359_s5i6ye_result.log +13 -0
  35. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610087998_qihv9v_prompt.log +19 -0
  36. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610087998_qihv9v_result.log +17 -0
  37. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610773920_qlzmxo_prompt.log +19 -0
  38. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610773920_qlzmxo_result.log +17 -0
  39. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783611376090_xpqf1z_prompt.log +19 -0
  40. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783611376090_xpqf1z_result.log +16 -0
  41. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612097184_9rqfor_prompt.log +19 -0
  42. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612097184_9rqfor_result.log +138 -0
  43. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612811574_4u602j_prompt.log +19 -0
  44. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612811574_4u602j_result.log +16 -0
  45. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783613750487_a46ttn_prompt.log +19 -0
  46. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783613750487_a46ttn_result.log +107 -0
  47. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783614289360_figwc0_prompt.log +19 -0
  48. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783614289360_figwc0_result.log +200 -0
  49. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615067640_l7tm6a_prompt.log +19 -0
  50. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615067640_l7tm6a_result.log +54 -0
  51. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615825308_44nor5_prompt.log +19 -0
  52. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615825308_44nor5_result.log +85 -0
  53. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783616524771_ut1ftw_prompt.log +19 -0
  54. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783616524771_ut1ftw_result.log +266 -0
  55. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783617323039_fs3x5a_prompt.log +19 -0
  56. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783617323039_fs3x5a_result.log +56 -0
  57. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618049184_1f4yah_prompt.log +19 -0
  58. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618049184_1f4yah_result.log +96 -0
  59. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618793925_4ee7tf_prompt.log +19 -0
  60. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618793925_4ee7tf_result.log +481 -0
  61. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783619642574_yjr4mm_prompt.log +19 -0
  62. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783619642574_yjr4mm_result.log +104 -0
  63. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783620392771_oduto0_prompt.log +19 -0
  64. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783620392771_oduto0_result.log +148 -0
  65. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621183670_gd0p1x_prompt.log +19 -0
  66. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621183670_gd0p1x_result.log +111 -0
  67. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621738388_z4k48b_prompt.log +19 -0
  68. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621738388_z4k48b_result.log +89 -0
  69. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783622493677_zwc35w_prompt.log +19 -0
  70. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783622493677_zwc35w_result.log +207 -0
  71. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783604894861_v6n3ut_prompt.log +14 -0
  72. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783604894861_v6n3ut_result.log +66 -0
  73. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783605934532_9h8ikb_prompt.log +14 -0
  74. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783605934532_9h8ikb_result.log +68 -0
  75. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783607181736_t12f4y_prompt.log +14 -0
  76. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783607181736_t12f4y_result.log +78 -0
  77. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783608341266_zhk0fl_prompt.log +14 -0
  78. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783608341266_zhk0fl_result.log +68 -0
  79. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783609420180_7xw817_prompt.log +14 -0
  80. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783609420180_7xw817_result.log +72 -0
  81. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783610535307_2pxofp_prompt.log +14 -0
  82. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783610535307_2pxofp_result.log +17 -0
  83. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783611437445_ofwnpb_prompt.log +14 -0
  84. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783611437445_ofwnpb_result.log +60 -0
  85. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783612556827_1cb112_prompt.log +14 -0
  86. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783612556827_1cb112_result.log +56 -0
  87. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783613579531_18xiax_prompt.log +14 -0
  88. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783613579531_18xiax_result.log +74 -0
  89. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783614650481_2zdz7w_prompt.log +14 -0
  90. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783614650481_2zdz7w_result.log +68 -0
  91. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783615742328_rjj69d_prompt.log +14 -0
  92. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783615742328_rjj69d_result.log +65 -0
  93. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783616767230_1iad99_prompt.log +14 -0
  94. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783616767230_1iad99_result.log +68 -0
  95. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783617783537_rqagku_prompt.log +14 -0
  96. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783617783537_rqagku_result.log +56 -0
  97. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783618817343_r1lbhr_prompt.log +14 -0
  98. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783618817343_r1lbhr_result.log +65 -0
  99. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783619907773_8msfw3_prompt.log +14 -0
  100. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783619907773_8msfw3_result.log +77 -0
  101. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783621009302_hybdum_prompt.log +14 -0
  102. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783621009302_hybdum_result.log +64 -0
  103. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783622083681_nznpnx_prompt.log +14 -0
  104. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783622083681_nznpnx_result.log +56 -0
  105. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783623208340_olsbaw_prompt.log +14 -0
  106. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783605134860_9jssz9_prompt.log +14 -0
  107. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783605134860_9jssz9_result.log +69 -0
  108. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783606516743_zftbaa_prompt.log +14 -0
  109. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783606516743_zftbaa_result.log +92 -0
  110. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783608038317_jrd66e_prompt.log +14 -0
  111. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783608038317_jrd66e_result.log +92 -0
  112. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783609388416_mc3zoe_prompt.log +14 -0
  113. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783609388416_mc3zoe_result.log +82 -0
  114. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783610821384_tzqvlb_prompt.log +14 -0
  115. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783610821384_tzqvlb_result.log +17 -0
  116. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783612023285_buygpo_prompt.log +14 -0
  117. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783612023285_buygpo_result.log +57 -0
  118. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783613599301_6f78cw_prompt.log +14 -0
  119. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783613599301_6f78cw_result.log +60 -0
  120. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783615000844_v95ues_prompt.log +14 -0
  121. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783615000844_v95ues_result.log +69 -0
  122. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783616341693_bl5d9o_prompt.log +14 -0
  123. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783616341693_bl5d9o_result.log +64 -0
  124. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783617832831_ha6s8d_prompt.log +14 -0
  125. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783617832831_ha6s8d_result.log +42 -0
  126. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783619384959_s5iiwf_prompt.log +14 -0
  127. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783619384959_s5iiwf_result.log +47 -0
  128. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783620946263_d7ovai_prompt.log +14 -0
  129. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783620946263_d7ovai_result.log +52 -0
  130. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783622473846_t839e5_prompt.log +14 -0
  131. package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783622473846_t839e5_result.log +96 -0
  132. package/plugins/ruflo-metaharness/.claude-flow/metrics/backup.json +7 -0
  133. package/plugins/ruflo-metaharness/.claude-flow/metrics/codebase-map.json +11 -0
  134. package/plugins/ruflo-metaharness/.claude-flow/metrics/consolidation.json +16 -0
  135. package/plugins/ruflo-metaharness/.claude-flow/metrics/harness-loop.json +83 -0
  136. package/plugins/ruflo-metaharness/.claude-flow/metrics/performance.json +55 -0
  137. package/plugins/ruflo-metaharness/.claude-flow/metrics/security-audit.json +206 -0
  138. package/plugins/ruflo-metaharness/.claude-flow/metrics/test-gaps.json +95 -0
  139. package/plugins/ruflo-metaharness/.claude-flow/neural/stats.json +6 -0
  140. package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +1 -1
@@ -0,0 +1,96 @@
1
+ [2026-07-09T18:43:49.366Z] RESULT
2
+ ============================================================
3
+ {
4
+ "success": true,
5
+ "output": "## Test Coverage Analysis — `plugins/ruflo-metaharness`\n\n**Testing model:** no framework (no Jest/Vitest/Mocha, no `package.json` in this directory) — coverage is via hand-rolled `scripts/test-*.mjs` node scripts that assert and exit non-zero on failure. 34 `.mjs` files (~6,900 lines): 6 shared library modules (`_*.mjs`), 22 CLI wrappers, 6 test scripts.\n\n### What's actually covered today\n\n| Test file | Covers |\n|---|---|\n| `test-similarity.mjs` | `_similarity.mjs` pure functions (`projectToVec`, `cosine`, `categoricalAgreement`, `jaccard`, `similarity`) + `_harness.mjs`'s `SEVERITY_RANK`/`rankSeverity`/`parseMcpScanText` |\n| `test-graceful-degradation.mjs` | **Degraded-path only** (network blocked) for `score`, `genome`, `mcp-scan`, `threat-model`, `oia-audit`, `audit-list`, `audit-trend`, `mint` — asserts exit 0 + `\"degraded\":true` |\n| `test-mcp-tools.mjs` | MCP tool registry handler contract (callable, returns `{success,data,degraded,exitCode}`, never throws) |\n| `test-pipeline-roundtrip.mjs` | E2E `oia-audit` → `audit-trend` chain against the real repo |\n| `test-parallel-pipeline.mjs` | `recordPair`/`recordPairOutcome` (TS, elsewhere) → `.jsonl` → `router-parallel-analyze.mjs` composition |\n| `test-with-openrouter.mjs` | Costed E2E: scaffold via `mint` + `doctor`/`validate`/`score`/`genome` lifecycle |\n\nEvery one of these is an **integration/degraded-path/contract** test. None of them is a true unit test of a CLI wrapper's own logic (arg parsing, safety checks, output shaping) on its **success path** with a **fake/mocked upstream**.\n\n---\n\n### Gap 1 — `_invoke.mjs`: zero direct unit tests (highest priority)\n\nThis is the shared plumbing layer imported by ~15 scripts (`classifyDegraded`, `injectJson`, `parseTrailingJson`, `satisfiesTildeRange`, `findLocalPackageDir`, `ensureCachedInstall`, `importOptionalLibrary`, `makeDegradedEmitter`). It's pure/near-pure and currently only exercised transitively through e2e degraded-path runs — no test asserts its behavior directly.\n\n**Untested edge cases:**\n- `classifyDegraded`: `exitCode === null` → `-timeout` reason vs `DEGRADED_RX` match → `-not-available` vs neither → `{degraded:false}`. The regex itself (`could not determine executable|404|not installed|MODULE_NOT_FOUND|ENOTFOUND|getaddrinfo|ECONNREFUSED|ETIMEDOUT|npm ERR`) has no per-token test.\n- `parseTrailingJson`: multiple `{...}` blocks (must pick **last** parseable one, not first — this was called out in the file header as a fixed bug), nested objects that break the lazy regex (falls back to greedy whole-span match), no JSON present → `null`.\n- `injectJson`: `args` already contains `--json` → no duplicate; `wantJson=false` → untouched copy (not mutated).\n- `satisfiesTildeRange`: same major.minor + patch ≥ pin → true; different minor → false; malformed version strings → false (no throw).\n- `findLocalPackageDir`: `RUFLO_METAHARNESS_SKIP_LOCAL=1` short-circuits; pin-version filtering skips a stale ancestor install.\n\n```javascript\n// scripts/test-invoke.mjs — unit tests for _invoke.mjs (skeleton)\nimport {\n classifyDegraded, injectJson, parseTrailingJson, satisfiesTildeRange,\n} from './_invoke.mjs';\n\nlet failures = 0;\nfunction assertEq(actual, expected, label) {\n const ok = JSON.stringify(actual) === JSON.stringify(expected);\n if (!ok) { failures++; console.error(`FAIL ${label}: got ${JSON.stringify(actual)}, want ${JSON.stringify(expected)}`); }\n}\n\n// classifyDegraded\nassertEq(classifyDegraded('', null, 'x'), { degraded: true, reason: 'x-timeout' }, 'timeout');\nassertEq(classifyDegraded('npm ERR! 404', 1, 'x'), { degraded: true, reason: 'x-not-available' }, 'npm-404');\nassertEq(classifyDegraded('normal output', 0, 'x'), { degraded: false }, 'healthy');\n\n// injectJson\nassertEq(injectJson(['a'], true), ['a', '--json'], 'inject-json');\nassertEq(injectJson(['a', '--json'], true), ['a', '--json'], 'no-dup');\nassertEq(injectJson(['a'], false), ['a'], 'opt-out');\n\n// parseTrailingJson — LAST block wins (regression guard for the documented bug)\nassertEq(parseTrailingJson('progress {\"partial\":1}\\nfinal: {\"result\":\"ok\"}'), { result: 'ok' }, 'last-block');\nassertEq(parseTrailingJson('no json here'), null, 'no-json');\n\n// satisfiesTildeRange\nassertEq(satisfiesTildeRange('0.3.2', '~0.3.0'), true, 'patch-ok');\nassertEq(satisfiesTildeRange('0.4.0', '~0.3.0'), false, 'minor-mismatch');\nassertEq(satisfiesTildeRange('garbage', '~0.3.0'), false, 'malformed');\n\nconsole.log(failures === 0 ? 'PASS' : `${failures} FAILED`);\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n### Gap 2 — `_darwin.mjs`: zero coverage (not even degraded-path)\n\n`test-graceful-degradation.mjs` explicitly lists 8 covered skills; **none of `evolve`, `bench`, `security-bench`, `gepa` (all `_darwin.mjs` consumers) are in that list.** So the entire darwin invocation path — sync (`runDarwin`) and async streaming (`runDarwinAsync`, `onProgress` callback, `signal` abort, timeout kill) — has no test at all, degraded or otherwise.\n\n```javascript\n// scripts/test-darwin-degraded.mjs (skeleton) — extend test-graceful-degradation.mjs's\n// pattern (unresolvable registry) to cover the darwin-backed skills.\nimport { spawnSync } from 'node:child_process';\n\nconst DARWIN_SKILLS = [\n { name: 'bench', args: ['scripts/bench.mjs', '--op', 'verify', '--suite', '/tmp/nonexistent.json'] },\n { name: 'evolve', args: ['scripts/evolve.mjs', '--repo', '.', '--confirm'] },\n { name: 'security-bench', args: ['scripts/security-bench.mjs'] },\n { name: 'gepa', args: ['scripts/gepa.mjs', '--op', 'genome'] },\n];\n\nlet failures = 0;\nfor (const skill of DARWIN_SKILLS) {\n const r = spawnSync('node', skill.args, {\n encoding: 'utf-8',\n env: { ...process.env, npm_config_registry: 'http://127.0.0.1:1' },\n });\n const ok = r.status === 0 && /\"degraded\"\\s*:\\s*true/.test(r.stdout || '');\n if (!ok) { failures++; console.error(`FAIL ${skill.name}: exit=${r.status}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n### Gap 3 — `evolve.mjs`: safety-critical logic, zero tests\n\nThis is the only script that mutates a target repo and has a documented safety model (ADR-153). Its **pure logic is fully unit-testable without spawning darwin** but isn't tested:\n- `safetyChecks()`: `--generations` 0/51 rejected, `--children` 0/21 rejected, `--concurrency` 0/9 rejected, invalid `--sandbox`/`--selection`/`--mutator` rejected, nonexistent `--repo` → exit 2.\n- `defaultTimeoutMs()`: arithmetic for `mock` vs `real` sandbox, `Math.ceil(variants/parallelism)` bucketing.\n- `--confirm` omitted → dry-run plan printed, exit 0, **no subprocess spawned** (this is the load-bearing safety guarantee and has no regression test).\n- Exit code 99 propagation (`safetyDisqualified`) — never exercised.\n- `looksLikeGepaTranscript` / `extractGepaTranscripts` / `summarizeTraces` — pure functions handling malformed/absent `.metaharness/runs/*.json`, never tested with synthetic fixtures.\n\n```javascript\n// scripts/test-evolve-safety.mjs (skeleton)\nimport { spawnSync } from 'node:child_process';\n\nlet failures = 0;\nfunction expectExit(args, wantCode, label) {\n const r = spawnSync('node', ['scripts/evolve.mjs', ...args], { encoding: 'utf-8' });\n if (r.status !== wantCode) { failures++; console.error(`FAIL ${label}: exit ${r.status}, want ${wantCode}`); }\n return r;\n}\n\nexpectExit(['--repo', '.', '--generations', '0'], 2, 'generations-too-low');\nexpectExit(['--repo', '.', '--generations', '51'], 2, 'generations-too-high');\nexpectExit(['--repo', '.', '--children', '21'], 2, 'children-cap');\nexpectExit(['--repo', '.', '--sandbox', 'bogus'], 2, 'bad-sandbox');\nexpectExit(['--repo', '/nonexistent/path'], 2, 'bad-repo');\n\n// dry-run: no --confirm → must never spawn darwin, must exit 0 with a plan\nconst dry = expectExit(['--repo', '.'], 0, 'dry-run-plan');\nif (!/\"dryRun\"\\s*:\\s*true/.test(dry.stdout)) { failures++; console.error('FAIL dry-run-plan: missing dryRun:true'); }\n\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n### Gap 4 — `mint.mjs`: path-traversal safety check untested\n\n`safetyChecks()` refuses to write to the project root or inside the calling repo — this is a directory-traversal guard with no negative test:\n\n```javascript\n// scripts/test-mint-safety.mjs (skeleton)\nimport { spawnSync } from 'node:child_process';\n\nlet failures = 0;\nfunction expectRejected(target, label) {\n const r = spawnSync('node', ['scripts/mint.mjs', '--name', 'x', '--template', 'minimal', '--target', target], { encoding: 'utf-8' });\n if (r.status !== 2) { failures++; console.error(`FAIL ${label}: expected exit 2, got ${r.status}`); }\n}\n\nexpectRejected(process.cwd(), 'refuses-repo-root');\nexpectRejected(process.cwd() + '/subdir', 'refuses-inside-repo');\n\n// missing --name / --template\nfor (const args of [['--template', 'minimal'], ['--name', 'x']]) {\n const r = spawnSync('node', ['scripts/mint.mjs', ...args], { encoding: 'utf-8' });\n if (r.status !== 2) { failures++; console.error(`FAIL missing-required-arg ${args}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n### Gap 5 — `drift-from-history.mjs`: composed pipeline, zero tests\n\n404-line orchestrator (3-way fast-path selection: slow / `--baseline-key` / `--baseline-file`) with no test at any level. Highest-value untested branches:\n- `--baseline-file` path skips memory entirely (`skippedAuditList=true, usedBaselineFile=true`) — the `timing.path === 'file'` label is asserted nowhere.\n- Empty `audit-list` result (no history) → exit 2, distinguished from `metaharness` absence → exit 3 (uses `auditResult.json?.degraded` from the *other* parallel branch — subtle coupling, easy to regress).\n- `--alert-on-new-severity` orthogonal gate: fires independent of `--threshold`, and the two reasons must both appear in `alert.reasons` when both trigger.\n- `runScriptJsonAsync` error path (`spawn` failure → `exitCode:127`, `stderr:'spawn-failed'`) — never simulated.\n\n```javascript\n// scripts/test-drift-from-history.mjs (skeleton, fixture-driven)\nimport { spawnSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\n\nlet failures = 0;\nconst dir = mkdtempSync(join(tmpdir(), 'drift-test-'));\n\n// Fixture: two near-identical oia-audit-shaped records with fingerprints.\nconst baseline = { startedAt: '2026-01-01T00:00:00Z', composite: { worst: 'clean' },\n fingerprint: { score: { dims: [1,1,1] }, genome: { tags: ['a'] } } };\nwriteFileSync(join(dir, 'baseline.json'), JSON.stringify(baseline));\n\n// --baseline-file path: assert timing.path === 'file' and no audit-list call\nconst r = spawnSync('node', [\n 'scripts/drift-from-history.mjs',\n '--baseline-file', join(dir, 'baseline.json'),\n '--format', 'json',\n], { encoding: 'utf-8' });\nconst json = JSON.parse((r.stdout.match(/\\{[\\s\\S]*\\}/) || ['{}'])[0]);\nif (json.timing?.path !== 'file') { failures++; console.error('FAIL baseline-file: wrong fast-path label'); }\n\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n### Gap 6 — Untested error/malformed-input handling across CLI wrappers\n\nNone of the following are exercised anywhere:\n- `audit-trend.mjs` / `similarity.mjs`: malformed JSON file for `--baseline`/`--a`/`--b` (both have a try/catch around `JSON.parse` → exit 2, but no test); missing `composite` key in a supplied record (`audit-trend.mjs:104-107`); memory-namespace key-not-found path (`memRetrieve` returns `null`).\n- `mcp-scan.mjs`: invalid `--fail-on` value → exit 2 (untested); `rankSeverity` applied to an unknown/undefined severity in `findings` (regression the code comment explicitly calls out as a past bug — iter 63 — but has no test guarding it).\n- `redblue.mjs`: unknown subcommand → exit 2; `attack` without a valid family → exit 2; `report` without `--in` → exit 2; unknown `--flag` passthrough forwarding (`_passthrough` accumulation logic, including the \"peek next token, only consume if not another flag\" branch) — no test.\n- `gepa.mjs`: invalid `--op` → exit 2; `--transcript` pointing at a non-array JSON → exit 2; `readJsonFile` on missing file → exit 2.\n- `learn.mjs`: `CHECKOUT_RX` detection path (`requires a metaharness repo checkout` in combined stdout/stderr) → `{status:'checkout-required'}` — this is a load-bearing distinct-from-`degraded` state with no test.\n\n```javascript\n// scripts/test-cli-arg-validation.mjs (skeleton — one file to batch cheap negative-path checks)\nimport { spawnSync } from 'node:child_process';\n\nconst CASES = [\n { args: ['scripts/mcp-scan.mjs', '--fail-on', 'bogus'], code: 2, label: 'mcp-scan bad --fail-on' },\n { args: ['scripts/redblue.mjs', 'nonsense'], code: 2, label: 'redblue unknown subcommand' },\n { args: ['scripts/redblue.mjs', 'attack', 'bogus-family'], code: 2, label: 'redblue bad attack family' },\n { args: ['scripts/redblue.mjs', 'report'], code: 2, label: 'redblue report missing --in' },\n { args: ['scripts/gepa.mjs', '--op', 'bogus'], code: 2, label: 'gepa bad --op' },\n { args: ['scripts/bench.mjs', '--op', 'bogus'], code: 2, label: 'bench bad --op' },\n { args: ['scripts/bench.mjs', '--op', 'verify'], code: 2, label: 'bench verify missing --suite' },\n];\n\nlet failures = 0;\nfor (const c of CASES) {\n const r = spawnSync('node', c.args, { encoding: 'utf-8' });\n if (r.status !== c.code) { failures++; console.error(`FAIL ${c.label}: exit ${r.status}, want ${c.code}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n### Gap 7 — `router-parallel-analyze.mjs`: promotion-criteria math untested in isolation\n\nCovered only via `test-parallel-pipeline.mjs`'s full e2e (recorder → jsonl → analyzer). The three-criteria AND-gate arithmetic (`median`, `pctile`, `mean`, the >2%/<1%/<5% thresholds) has no direct unit test with a synthetic `.jsonl` fixture that pins one criterion failing while the other two pass — the case most likely to regress silently (OR-vs-AND was literally the bug this file's header warns about).\n\n```javascript\n// scripts/test-router-parallel-analyze.mjs (skeleton)\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nimport { spawnSync } from 'node:child_process';\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-test-'));\nconst file = join(dir, 'router-parallel.jsonl');\n\n// Fixture: quality +5% (passes >2%), cost +2% (FAILS <1%), latency +1% (passes <5%)\n// → overall must NOT be promotable even though 2/3 criteria pass (AND, not OR).\nconst rows = Array.from({ length: 20 }, (_, i) => JSON.stringify({\n ts: new Date(0).toISOString(),\n bandit: { pick: 'a', predictedQuality: 0.80, predictedCostUsd: 0.0100 },\n ser: { pick: 'b', predictedQuality: 0.84, predictedCostUsd: 0.0102 },\n outcome: { actualModel: 'b', actualQuality: 0.84, actualUsd: 0.0102, actualLatencyMs: 1010 },\n}));\nwriteFileSync(file, rows.join('\\n'));\n\nconst r = spawnSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file, '--strict', '--format', 'json'], { encoding: 'utf-8' });\nlet failures = 0;\nif (r.status !== 1) { failures++; console.error('FAIL: cost regression should block promotion under --strict (AND-gate)'); }\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n---\n\n### Summary table\n\n| File | Lines | Existing coverage | Gap severity |\n|---|---:|---|---|\n| `_invoke.mjs` | 232 | none (transitive only) | **High** — shared by ~15 files |\n| `_darwin.mjs` | 211 | none | **High** — powers evolve/bench/security-bench/gepa |\n| `evolve.mjs` | 405 | none | **High** — mutates repos, safety-critical |\n| `drift-from-history.mjs` | 364 | none | **High** — 3-way fast-path branching |\n| `gepa.mjs` | 154 | none | Medium |\n| `learn.mjs` | 128 | none | Medium |\n| `security-bench.mjs` | 175 | none (incl. markdown parser `parseSecurityBenchMarkdown`) | Medium |\n| `bench.mjs` | 96 | none | Medium |\n| `redblue.mjs` / `_redblue.mjs` | 286 / 144 | none | Medium |\n| `mint.mjs` | 127 | degraded-path only | Medium — success + safety-check paths untested |\n| `similarity.mjs` (CLI) | 159 | `_similarity.mjs` internals tested; CLI wrapper (arg parsing, mem lookup, `--alert-below`) not | Low-Medium |\n| `audit-trend.mjs` | 273 | degraded-path + e2e roundtrip only | Low-Medium — malformed-record/missing-key branches untested |\n| `mcp-scan.mjs` | 111 | `parseMcpScanText` tested via `_harness.mjs`; CLI arg validation + severity-filter edge cases not | Low |\n| `router-parallel-analyze.mjs` | 250 | e2e only | Low — pure math untested in isolation |\n\n**Recommendation:** prioritize `test-invoke.mjs` (Gap 1) — it's pure, fast, and de-risks every other script that imports it — then `test-evolve-safety.mjs` and `test-mint-safety.mjs` (Gaps 3–4) since those guard destructive/safety-critical behavior with zero regression protection today.\n",
6
+ "parsedOutput": {
7
+ "sections": [
8
+ {
9
+ "title": "Test Coverage Analysis — `plugins/ruflo-metaharness`",
10
+ "content": "\n**Testing model:** no framework (no Jest/Vitest/Mocha, no `package.json` in this directory) — coverage is via hand-rolled `scripts/test-*.mjs` node scripts that assert and exit non-zero on failure. 34 `.mjs` files (~6,900 lines): 6 shared library modules (`_*.mjs`), 22 CLI wrappers, 6 test scripts.\n\n",
11
+ "level": 2
12
+ },
13
+ {
14
+ "title": "What's actually covered today",
15
+ "content": "\n| Test file | Covers |\n|---|---|\n| `test-similarity.mjs` | `_similarity.mjs` pure functions (`projectToVec`, `cosine`, `categoricalAgreement`, `jaccard`, `similarity`) + `_harness.mjs`'s `SEVERITY_RANK`/`rankSeverity`/`parseMcpScanText` |\n| `test-graceful-degradation.mjs` | **Degraded-path only** (network blocked) for `score`, `genome`, `mcp-scan`, `threat-model`, `oia-audit`, `audit-list`, `audit-trend`, `mint` — asserts exit 0 + `\"degraded\":true` |\n| `test-mcp-tools.mjs` | MCP tool registry handler contract (callable, returns `{success,data,degraded,exitCode}`, never throws) |\n| `test-pipeline-roundtrip.mjs` | E2E `oia-audit` → `audit-trend` chain against the real repo |\n| `test-parallel-pipeline.mjs` | `recordPair`/`recordPairOutcome` (TS, elsewhere) → `.jsonl` → `router-parallel-analyze.mjs` composition |\n| `test-with-openrouter.mjs` | Costed E2E: scaffold via `mint` + `doctor`/`validate`/`score`/`genome` lifecycle |\n\nEvery one of these is an **integration/degraded-path/contract** test. None of them is a true unit test of a CLI wrapper's own logic (arg parsing, safety checks, output shaping) on its **success path** with a **fake/mocked upstream**.\n\n---\n\n",
16
+ "level": 3
17
+ },
18
+ {
19
+ "title": "Gap 1 — `_invoke.mjs`: zero direct unit tests (highest priority)",
20
+ "content": "\nThis is the shared plumbing layer imported by ~15 scripts (`classifyDegraded`, `injectJson`, `parseTrailingJson`, `satisfiesTildeRange`, `findLocalPackageDir`, `ensureCachedInstall`, `importOptionalLibrary`, `makeDegradedEmitter`). It's pure/near-pure and currently only exercised transitively through e2e degraded-path runs — no test asserts its behavior directly.\n\n**Untested edge cases:**\n- `classifyDegraded`: `exitCode === null` → `-timeout` reason vs `DEGRADED_RX` match → `-not-available` vs neither → `{degraded:false}`. The regex itself (`could not determine executable|404|not installed|MODULE_NOT_FOUND|ENOTFOUND|getaddrinfo|ECONNREFUSED|ETIMEDOUT|npm ERR`) has no per-token test.\n- `parseTrailingJson`: multiple `{...}` blocks (must pick **last** parseable one, not first — this was called out in the file header as a fixed bug), nested objects that break the lazy regex (falls back to greedy whole-span match), no JSON present → `null`.\n- `injectJson`: `args` already contains `--json` → no duplicate; `wantJson=false` → untouched copy (not mutated).\n- `satisfiesTildeRange`: same major.minor + patch ≥ pin → true; different minor → false; malformed version strings → false (no throw).\n- `findLocalPackageDir`: `RUFLO_METAHARNESS_SKIP_LOCAL=1` short-circuits; pin-version filtering skips a stale ancestor install.\n\n```javascript\n// scripts/test-invoke.mjs — unit tests for _invoke.mjs (skeleton)\nimport {\n classifyDegraded, injectJson, parseTrailingJson, satisfiesTildeRange,\n} from './_invoke.mjs';\n\nlet failures = 0;\nfunction assertEq(actual, expected, label) {\n const ok = JSON.stringify(actual) === JSON.stringify(expected);\n if (!ok) { failures++; console.error(`FAIL ${label}: got ${JSON.stringify(actual)}, want ${JSON.stringify(expected)}`); }\n}\n\n// classifyDegraded\nassertEq(classifyDegraded('', null, 'x'), { degraded: true, reason: 'x-timeout' }, 'timeout');\nassertEq(classifyDegraded('npm ERR! 404', 1, 'x'), { degraded: true, reason: 'x-not-available' }, 'npm-404');\nassertEq(classifyDegraded('normal output', 0, 'x'), { degraded: false }, 'healthy');\n\n// injectJson\nassertEq(injectJson(['a'], true), ['a', '--json'], 'inject-json');\nassertEq(injectJson(['a', '--json'], true), ['a', '--json'], 'no-dup');\nassertEq(injectJson(['a'], false), ['a'], 'opt-out');\n\n// parseTrailingJson — LAST block wins (regression guard for the documented bug)\nassertEq(parseTrailingJson('progress {\"partial\":1}\\nfinal: {\"result\":\"ok\"}'), { result: 'ok' }, 'last-block');\nassertEq(parseTrailingJson('no json here'), null, 'no-json');\n\n// satisfiesTildeRange\nassertEq(satisfiesTildeRange('0.3.2', '~0.3.0'), true, 'patch-ok');\nassertEq(satisfiesTildeRange('0.4.0', '~0.3.0'), false, 'minor-mismatch');\nassertEq(satisfiesTildeRange('garbage', '~0.3.0'), false, 'malformed');\n\nconsole.log(failures === 0 ? 'PASS' : `${failures} FAILED`);\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n",
21
+ "level": 3
22
+ },
23
+ {
24
+ "title": "Gap 2 — `_darwin.mjs`: zero coverage (not even degraded-path)",
25
+ "content": "\n`test-graceful-degradation.mjs` explicitly lists 8 covered skills; **none of `evolve`, `bench`, `security-bench`, `gepa` (all `_darwin.mjs` consumers) are in that list.** So the entire darwin invocation path — sync (`runDarwin`) and async streaming (`runDarwinAsync`, `onProgress` callback, `signal` abort, timeout kill) — has no test at all, degraded or otherwise.\n\n```javascript\n// scripts/test-darwin-degraded.mjs (skeleton) — extend test-graceful-degradation.mjs's\n// pattern (unresolvable registry) to cover the darwin-backed skills.\nimport { spawnSync } from 'node:child_process';\n\nconst DARWIN_SKILLS = [\n { name: 'bench', args: ['scripts/bench.mjs', '--op', 'verify', '--suite', '/tmp/nonexistent.json'] },\n { name: 'evolve', args: ['scripts/evolve.mjs', '--repo', '.', '--confirm'] },\n { name: 'security-bench', args: ['scripts/security-bench.mjs'] },\n { name: 'gepa', args: ['scripts/gepa.mjs', '--op', 'genome'] },\n];\n\nlet failures = 0;\nfor (const skill of DARWIN_SKILLS) {\n const r = spawnSync('node', skill.args, {\n encoding: 'utf-8',\n env: { ...process.env, npm_config_registry: 'http://127.0.0.1:1' },\n });\n const ok = r.status === 0 && /\"degraded\"\\s*:\\s*true/.test(r.stdout || '');\n if (!ok) { failures++; console.error(`FAIL ${skill.name}: exit=${r.status}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n",
26
+ "level": 3
27
+ },
28
+ {
29
+ "title": "Gap 3 — `evolve.mjs`: safety-critical logic, zero tests",
30
+ "content": "\nThis is the only script that mutates a target repo and has a documented safety model (ADR-153). Its **pure logic is fully unit-testable without spawning darwin** but isn't tested:\n- `safetyChecks()`: `--generations` 0/51 rejected, `--children` 0/21 rejected, `--concurrency` 0/9 rejected, invalid `--sandbox`/`--selection`/`--mutator` rejected, nonexistent `--repo` → exit 2.\n- `defaultTimeoutMs()`: arithmetic for `mock` vs `real` sandbox, `Math.ceil(variants/parallelism)` bucketing.\n- `--confirm` omitted → dry-run plan printed, exit 0, **no subprocess spawned** (this is the load-bearing safety guarantee and has no regression test).\n- Exit code 99 propagation (`safetyDisqualified`) — never exercised.\n- `looksLikeGepaTranscript` / `extractGepaTranscripts` / `summarizeTraces` — pure functions handling malformed/absent `.metaharness/runs/*.json`, never tested with synthetic fixtures.\n\n```javascript\n// scripts/test-evolve-safety.mjs (skeleton)\nimport { spawnSync } from 'node:child_process';\n\nlet failures = 0;\nfunction expectExit(args, wantCode, label) {\n const r = spawnSync('node', ['scripts/evolve.mjs', ...args], { encoding: 'utf-8' });\n if (r.status !== wantCode) { failures++; console.error(`FAIL ${label}: exit ${r.status}, want ${wantCode}`); }\n return r;\n}\n\nexpectExit(['--repo', '.', '--generations', '0'], 2, 'generations-too-low');\nexpectExit(['--repo', '.', '--generations', '51'], 2, 'generations-too-high');\nexpectExit(['--repo', '.', '--children', '21'], 2, 'children-cap');\nexpectExit(['--repo', '.', '--sandbox', 'bogus'], 2, 'bad-sandbox');\nexpectExit(['--repo', '/nonexistent/path'], 2, 'bad-repo');\n\n// dry-run: no --confirm → must never spawn darwin, must exit 0 with a plan\nconst dry = expectExit(['--repo', '.'], 0, 'dry-run-plan');\nif (!/\"dryRun\"\\s*:\\s*true/.test(dry.stdout)) { failures++; console.error('FAIL dry-run-plan: missing dryRun:true'); }\n\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n",
31
+ "level": 3
32
+ },
33
+ {
34
+ "title": "Gap 4 — `mint.mjs`: path-traversal safety check untested",
35
+ "content": "\n`safetyChecks()` refuses to write to the project root or inside the calling repo — this is a directory-traversal guard with no negative test:\n\n```javascript\n// scripts/test-mint-safety.mjs (skeleton)\nimport { spawnSync } from 'node:child_process';\n\nlet failures = 0;\nfunction expectRejected(target, label) {\n const r = spawnSync('node', ['scripts/mint.mjs', '--name', 'x', '--template', 'minimal', '--target', target], { encoding: 'utf-8' });\n if (r.status !== 2) { failures++; console.error(`FAIL ${label}: expected exit 2, got ${r.status}`); }\n}\n\nexpectRejected(process.cwd(), 'refuses-repo-root');\nexpectRejected(process.cwd() + '/subdir', 'refuses-inside-repo');\n\n// missing --name / --template\nfor (const args of [['--template', 'minimal'], ['--name', 'x']]) {\n const r = spawnSync('node', ['scripts/mint.mjs', ...args], { encoding: 'utf-8' });\n if (r.status !== 2) { failures++; console.error(`FAIL missing-required-arg ${args}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n",
36
+ "level": 3
37
+ },
38
+ {
39
+ "title": "Gap 5 — `drift-from-history.mjs`: composed pipeline, zero tests",
40
+ "content": "\n404-line orchestrator (3-way fast-path selection: slow / `--baseline-key` / `--baseline-file`) with no test at any level. Highest-value untested branches:\n- `--baseline-file` path skips memory entirely (`skippedAuditList=true, usedBaselineFile=true`) — the `timing.path === 'file'` label is asserted nowhere.\n- Empty `audit-list` result (no history) → exit 2, distinguished from `metaharness` absence → exit 3 (uses `auditResult.json?.degraded` from the *other* parallel branch — subtle coupling, easy to regress).\n- `--alert-on-new-severity` orthogonal gate: fires independent of `--threshold`, and the two reasons must both appear in `alert.reasons` when both trigger.\n- `runScriptJsonAsync` error path (`spawn` failure → `exitCode:127`, `stderr:'spawn-failed'`) — never simulated.\n\n```javascript\n// scripts/test-drift-from-history.mjs (skeleton, fixture-driven)\nimport { spawnSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\n\nlet failures = 0;\nconst dir = mkdtempSync(join(tmpdir(), 'drift-test-'));\n\n// Fixture: two near-identical oia-audit-shaped records with fingerprints.\nconst baseline = { startedAt: '2026-01-01T00:00:00Z', composite: { worst: 'clean' },\n fingerprint: { score: { dims: [1,1,1] }, genome: { tags: ['a'] } } };\nwriteFileSync(join(dir, 'baseline.json'), JSON.stringify(baseline));\n\n// --baseline-file path: assert timing.path === 'file' and no audit-list call\nconst r = spawnSync('node', [\n 'scripts/drift-from-history.mjs',\n '--baseline-file', join(dir, 'baseline.json'),\n '--format', 'json',\n], { encoding: 'utf-8' });\nconst json = JSON.parse((r.stdout.match(/\\{[\\s\\S]*\\}/) || ['{}'])[0]);\nif (json.timing?.path !== 'file') { failures++; console.error('FAIL baseline-file: wrong fast-path label'); }\n\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n",
41
+ "level": 3
42
+ },
43
+ {
44
+ "title": "Gap 6 — Untested error/malformed-input handling across CLI wrappers",
45
+ "content": "\nNone of the following are exercised anywhere:\n- `audit-trend.mjs` / `similarity.mjs`: malformed JSON file for `--baseline`/`--a`/`--b` (both have a try/catch around `JSON.parse` → exit 2, but no test); missing `composite` key in a supplied record (`audit-trend.mjs:104-107`); memory-namespace key-not-found path (`memRetrieve` returns `null`).\n- `mcp-scan.mjs`: invalid `--fail-on` value → exit 2 (untested); `rankSeverity` applied to an unknown/undefined severity in `findings` (regression the code comment explicitly calls out as a past bug — iter 63 — but has no test guarding it).\n- `redblue.mjs`: unknown subcommand → exit 2; `attack` without a valid family → exit 2; `report` without `--in` → exit 2; unknown `--flag` passthrough forwarding (`_passthrough` accumulation logic, including the \"peek next token, only consume if not another flag\" branch) — no test.\n- `gepa.mjs`: invalid `--op` → exit 2; `--transcript` pointing at a non-array JSON → exit 2; `readJsonFile` on missing file → exit 2.\n- `learn.mjs`: `CHECKOUT_RX` detection path (`requires a metaharness repo checkout` in combined stdout/stderr) → `{status:'checkout-required'}` — this is a load-bearing distinct-from-`degraded` state with no test.\n\n```javascript\n// scripts/test-cli-arg-validation.mjs (skeleton — one file to batch cheap negative-path checks)\nimport { spawnSync } from 'node:child_process';\n\nconst CASES = [\n { args: ['scripts/mcp-scan.mjs', '--fail-on', 'bogus'], code: 2, label: 'mcp-scan bad --fail-on' },\n { args: ['scripts/redblue.mjs', 'nonsense'], code: 2, label: 'redblue unknown subcommand' },\n { args: ['scripts/redblue.mjs', 'attack', 'bogus-family'], code: 2, label: 'redblue bad attack family' },\n { args: ['scripts/redblue.mjs', 'report'], code: 2, label: 'redblue report missing --in' },\n { args: ['scripts/gepa.mjs', '--op', 'bogus'], code: 2, label: 'gepa bad --op' },\n { args: ['scripts/bench.mjs', '--op', 'bogus'], code: 2, label: 'bench bad --op' },\n { args: ['scripts/bench.mjs', '--op', 'verify'], code: 2, label: 'bench verify missing --suite' },\n];\n\nlet failures = 0;\nfor (const c of CASES) {\n const r = spawnSync('node', c.args, { encoding: 'utf-8' });\n if (r.status !== c.code) { failures++; console.error(`FAIL ${c.label}: exit ${r.status}, want ${c.code}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n",
46
+ "level": 3
47
+ },
48
+ {
49
+ "title": "Gap 7 — `router-parallel-analyze.mjs`: promotion-criteria math untested in isolation",
50
+ "content": "\nCovered only via `test-parallel-pipeline.mjs`'s full e2e (recorder → jsonl → analyzer). The three-criteria AND-gate arithmetic (`median`, `pctile`, `mean`, the >2%/<1%/<5% thresholds) has no direct unit test with a synthetic `.jsonl` fixture that pins one criterion failing while the other two pass — the case most likely to regress silently (OR-vs-AND was literally the bug this file's header warns about).\n\n```javascript\n// scripts/test-router-parallel-analyze.mjs (skeleton)\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nimport { spawnSync } from 'node:child_process';\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-test-'));\nconst file = join(dir, 'router-parallel.jsonl');\n\n// Fixture: quality +5% (passes >2%), cost +2% (FAILS <1%), latency +1% (passes <5%)\n// → overall must NOT be promotable even though 2/3 criteria pass (AND, not OR).\nconst rows = Array.from({ length: 20 }, (_, i) => JSON.stringify({\n ts: new Date(0).toISOString(),\n bandit: { pick: 'a', predictedQuality: 0.80, predictedCostUsd: 0.0100 },\n ser: { pick: 'b', predictedQuality: 0.84, predictedCostUsd: 0.0102 },\n outcome: { actualModel: 'b', actualQuality: 0.84, actualUsd: 0.0102, actualLatencyMs: 1010 },\n}));\nwriteFileSync(file, rows.join('\\n'));\n\nconst r = spawnSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file, '--strict', '--format', 'json'], { encoding: 'utf-8' });\nlet failures = 0;\nif (r.status !== 1) { failures++; console.error('FAIL: cost regression should block promotion under --strict (AND-gate)'); }\nprocess.exit(failures === 0 ? 0 : 1);\n```\n\n---\n\n",
51
+ "level": 3
52
+ },
53
+ {
54
+ "title": "Summary table",
55
+ "content": "| File | Lines | Existing coverage | Gap severity |\n|---|---:|---|---|\n| `_invoke.mjs` | 232 | none (transitive only) | **High** — shared by ~15 files |\n| `_darwin.mjs` | 211 | none | **High** — powers evolve/bench/security-bench/gepa |\n| `evolve.mjs` | 405 | none | **High** — mutates repos, safety-critical |\n| `drift-from-history.mjs` | 364 | none | **High** — 3-way fast-path branching |\n| `gepa.mjs` | 154 | none | Medium |\n| `learn.mjs` | 128 | none | Medium |\n| `security-bench.mjs` | 175 | none (incl. markdown parser `parseSecurityBenchMarkdown`) | Medium |\n| `bench.mjs` | 96 | none | Medium |\n| `redblue.mjs` / `_redblue.mjs` | 286 / 144 | none | Medium |\n| `mint.mjs` | 127 | degraded-path only | Medium — success + safety-check paths untested |\n| `similarity.mjs` (CLI) | 159 | `_similarity.mjs` internals tested; CLI wrapper (arg parsing, mem lookup, `--alert-below`) not | Low-Medium |\n| `audit-trend.mjs` | 273 | degraded-path + e2e roundtrip only | Low-Medium — malformed-record/missing-key branches untested |\n| `mcp-scan.mjs` | 111 | `parseMcpScanText` tested via `_harness.mjs`; CLI arg validation + severity-filter edge cases not | Low |\n| `router-parallel-analyze.mjs` | 250 | e2e only | Low — pure math untested in isolation |\n\n**Recommendation:** prioritize `test-invoke.mjs` (Gap 1) — it's pure, fast, and de-risks every other script that imports it — then `test-evolve-safety.mjs` and `test-mint-safety.mjs` (Gaps 3–4) since those guard destructive/safety-critical behavior with zero regression protection today.",
56
+ "level": 3
57
+ }
58
+ ],
59
+ "codeBlocks": [
60
+ {
61
+ "language": "javascript",
62
+ "code": "// scripts/test-invoke.mjs — unit tests for _invoke.mjs (skeleton)\nimport {\n classifyDegraded, injectJson, parseTrailingJson, satisfiesTildeRange,\n} from './_invoke.mjs';\n\nlet failures = 0;\nfunction assertEq(actual, expected, label) {\n const ok = JSON.stringify(actual) === JSON.stringify(expected);\n if (!ok) { failures++; console.error(`FAIL ${label}: got ${JSON.stringify(actual)}, want ${JSON.stringify(expected)}`); }\n}\n\n// classifyDegraded\nassertEq(classifyDegraded('', null, 'x'), { degraded: true, reason: 'x-timeout' }, 'timeout');\nassertEq(classifyDegraded('npm ERR! 404', 1, 'x'), { degraded: true, reason: 'x-not-available' }, 'npm-404');\nassertEq(classifyDegraded('normal output', 0, 'x'), { degraded: false }, 'healthy');\n\n// injectJson\nassertEq(injectJson(['a'], true), ['a', '--json'], 'inject-json');\nassertEq(injectJson(['a', '--json'], true), ['a', '--json'], 'no-dup');\nassertEq(injectJson(['a'], false), ['a'], 'opt-out');\n\n// parseTrailingJson — LAST block wins (regression guard for the documented bug)\nassertEq(parseTrailingJson('progress {\"partial\":1}\\nfinal: {\"result\":\"ok\"}'), { result: 'ok' }, 'last-block');\nassertEq(parseTrailingJson('no json here'), null, 'no-json');\n\n// satisfiesTildeRange\nassertEq(satisfiesTildeRange('0.3.2', '~0.3.0'), true, 'patch-ok');\nassertEq(satisfiesTildeRange('0.4.0', '~0.3.0'), false, 'minor-mismatch');\nassertEq(satisfiesTildeRange('garbage', '~0.3.0'), false, 'malformed');\n\nconsole.log(failures === 0 ? 'PASS' : `${failures} FAILED`);\nprocess.exit(failures === 0 ? 0 : 1);"
63
+ },
64
+ {
65
+ "language": "javascript",
66
+ "code": "// scripts/test-darwin-degraded.mjs (skeleton) — extend test-graceful-degradation.mjs's\n// pattern (unresolvable registry) to cover the darwin-backed skills.\nimport { spawnSync } from 'node:child_process';\n\nconst DARWIN_SKILLS = [\n { name: 'bench', args: ['scripts/bench.mjs', '--op', 'verify', '--suite', '/tmp/nonexistent.json'] },\n { name: 'evolve', args: ['scripts/evolve.mjs', '--repo', '.', '--confirm'] },\n { name: 'security-bench', args: ['scripts/security-bench.mjs'] },\n { name: 'gepa', args: ['scripts/gepa.mjs', '--op', 'genome'] },\n];\n\nlet failures = 0;\nfor (const skill of DARWIN_SKILLS) {\n const r = spawnSync('node', skill.args, {\n encoding: 'utf-8',\n env: { ...process.env, npm_config_registry: 'http://127.0.0.1:1' },\n });\n const ok = r.status === 0 && /\"degraded\"\\s*:\\s*true/.test(r.stdout || '');\n if (!ok) { failures++; console.error(`FAIL ${skill.name}: exit=${r.status}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);"
67
+ },
68
+ {
69
+ "language": "javascript",
70
+ "code": "// scripts/test-evolve-safety.mjs (skeleton)\nimport { spawnSync } from 'node:child_process';\n\nlet failures = 0;\nfunction expectExit(args, wantCode, label) {\n const r = spawnSync('node', ['scripts/evolve.mjs', ...args], { encoding: 'utf-8' });\n if (r.status !== wantCode) { failures++; console.error(`FAIL ${label}: exit ${r.status}, want ${wantCode}`); }\n return r;\n}\n\nexpectExit(['--repo', '.', '--generations', '0'], 2, 'generations-too-low');\nexpectExit(['--repo', '.', '--generations', '51'], 2, 'generations-too-high');\nexpectExit(['--repo', '.', '--children', '21'], 2, 'children-cap');\nexpectExit(['--repo', '.', '--sandbox', 'bogus'], 2, 'bad-sandbox');\nexpectExit(['--repo', '/nonexistent/path'], 2, 'bad-repo');\n\n// dry-run: no --confirm → must never spawn darwin, must exit 0 with a plan\nconst dry = expectExit(['--repo', '.'], 0, 'dry-run-plan');\nif (!/\"dryRun\"\\s*:\\s*true/.test(dry.stdout)) { failures++; console.error('FAIL dry-run-plan: missing dryRun:true'); }\n\nprocess.exit(failures === 0 ? 0 : 1);"
71
+ },
72
+ {
73
+ "language": "javascript",
74
+ "code": "// scripts/test-mint-safety.mjs (skeleton)\nimport { spawnSync } from 'node:child_process';\n\nlet failures = 0;\nfunction expectRejected(target, label) {\n const r = spawnSync('node', ['scripts/mint.mjs', '--name', 'x', '--template', 'minimal', '--target', target], { encoding: 'utf-8' });\n if (r.status !== 2) { failures++; console.error(`FAIL ${label}: expected exit 2, got ${r.status}`); }\n}\n\nexpectRejected(process.cwd(), 'refuses-repo-root');\nexpectRejected(process.cwd() + '/subdir', 'refuses-inside-repo');\n\n// missing --name / --template\nfor (const args of [['--template', 'minimal'], ['--name', 'x']]) {\n const r = spawnSync('node', ['scripts/mint.mjs', ...args], { encoding: 'utf-8' });\n if (r.status !== 2) { failures++; console.error(`FAIL missing-required-arg ${args}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);"
75
+ },
76
+ {
77
+ "language": "javascript",
78
+ "code": "// scripts/test-drift-from-history.mjs (skeleton, fixture-driven)\nimport { spawnSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\n\nlet failures = 0;\nconst dir = mkdtempSync(join(tmpdir(), 'drift-test-'));\n\n// Fixture: two near-identical oia-audit-shaped records with fingerprints.\nconst baseline = { startedAt: '2026-01-01T00:00:00Z', composite: { worst: 'clean' },\n fingerprint: { score: { dims: [1,1,1] }, genome: { tags: ['a'] } } };\nwriteFileSync(join(dir, 'baseline.json'), JSON.stringify(baseline));\n\n// --baseline-file path: assert timing.path === 'file' and no audit-list call\nconst r = spawnSync('node', [\n 'scripts/drift-from-history.mjs',\n '--baseline-file', join(dir, 'baseline.json'),\n '--format', 'json',\n], { encoding: 'utf-8' });\nconst json = JSON.parse((r.stdout.match(/\\{[\\s\\S]*\\}/) || ['{}'])[0]);\nif (json.timing?.path !== 'file') { failures++; console.error('FAIL baseline-file: wrong fast-path label'); }\n\nprocess.exit(failures === 0 ? 0 : 1);"
79
+ },
80
+ {
81
+ "language": "javascript",
82
+ "code": "// scripts/test-cli-arg-validation.mjs (skeleton — one file to batch cheap negative-path checks)\nimport { spawnSync } from 'node:child_process';\n\nconst CASES = [\n { args: ['scripts/mcp-scan.mjs', '--fail-on', 'bogus'], code: 2, label: 'mcp-scan bad --fail-on' },\n { args: ['scripts/redblue.mjs', 'nonsense'], code: 2, label: 'redblue unknown subcommand' },\n { args: ['scripts/redblue.mjs', 'attack', 'bogus-family'], code: 2, label: 'redblue bad attack family' },\n { args: ['scripts/redblue.mjs', 'report'], code: 2, label: 'redblue report missing --in' },\n { args: ['scripts/gepa.mjs', '--op', 'bogus'], code: 2, label: 'gepa bad --op' },\n { args: ['scripts/bench.mjs', '--op', 'bogus'], code: 2, label: 'bench bad --op' },\n { args: ['scripts/bench.mjs', '--op', 'verify'], code: 2, label: 'bench verify missing --suite' },\n];\n\nlet failures = 0;\nfor (const c of CASES) {\n const r = spawnSync('node', c.args, { encoding: 'utf-8' });\n if (r.status !== c.code) { failures++; console.error(`FAIL ${c.label}: exit ${r.status}, want ${c.code}`); }\n}\nprocess.exit(failures === 0 ? 0 : 1);"
83
+ },
84
+ {
85
+ "language": "javascript",
86
+ "code": "// scripts/test-router-parallel-analyze.mjs (skeleton)\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nimport { spawnSync } from 'node:child_process';\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-test-'));\nconst file = join(dir, 'router-parallel.jsonl');\n\n// Fixture: quality +5% (passes >2%), cost +2% (FAILS <1%), latency +1% (passes <5%)\n// → overall must NOT be promotable even though 2/3 criteria pass (AND, not OR).\nconst rows = Array.from({ length: 20 }, (_, i) => JSON.stringify({\n ts: new Date(0).toISOString(),\n bandit: { pick: 'a', predictedQuality: 0.80, predictedCostUsd: 0.0100 },\n ser: { pick: 'b', predictedQuality: 0.84, predictedCostUsd: 0.0102 },\n outcome: { actualModel: 'b', actualQuality: 0.84, actualUsd: 0.0102, actualLatencyMs: 1010 },\n}));\nwriteFileSync(file, rows.join('\\n'));\n\nconst r = spawnSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file, '--strict', '--format', 'json'], { encoding: 'utf-8' });\nlet failures = 0;\nif (r.status !== 1) { failures++; console.error('FAIL: cost regression should block promotion under --strict (AND-gate)'); }\nprocess.exit(failures === 0 ? 0 : 1);"
87
+ }
88
+ ]
89
+ },
90
+ "durationMs": 155519,
91
+ "model": "sonnet",
92
+ "sandboxMode": "permissive",
93
+ "workerType": "testgaps",
94
+ "timestamp": "2026-07-09T18:43:49.365Z",
95
+ "executionId": "testgaps_1783622473846_t839e5"
96
+ }
@@ -0,0 +1,7 @@
1
+ {
2
+ "timestamp": "2026-07-09T13:54:14.860Z",
3
+ "backedUp": false,
4
+ "sizeBytes": 0,
5
+ "rotatedAway": 0,
6
+ "skipped": "no-db"
7
+ }
@@ -0,0 +1,11 @@
1
+ {
2
+ "timestamp": "2026-07-09T18:44:15.091Z",
3
+ "projectRoot": "/Users/cohen/Projects/ruflo/plugins/ruflo-metaharness",
4
+ "structure": {
5
+ "hasPackageJson": false,
6
+ "hasTsConfig": false,
7
+ "hasClaudeConfig": false,
8
+ "hasClaudeFlow": true
9
+ },
10
+ "scannedAt": 1783622655091
11
+ }
@@ -0,0 +1,16 @@
1
+ {
2
+ "timestamp": "2026-07-09T18:50:15.052Z",
3
+ "distillationEnabled": true,
4
+ "patternsConsolidated": 0,
5
+ "memoryCleaned": 0,
6
+ "duplicatesRemoved": 0,
7
+ "episodes": 0,
8
+ "patternEmbeddings": 0,
9
+ "causalEdges": 0,
10
+ "promoted": 0,
11
+ "byProvenance": {},
12
+ "namespaces": [],
13
+ "dryRun": false,
14
+ "corrupt": false,
15
+ "skipped": "no-db"
16
+ }
@@ -0,0 +1,83 @@
1
+ {
2
+ "timestamp": "2026-07-09T13:56:14.863Z",
3
+ "flywheel": {
4
+ "ran": false,
5
+ "reason": "opt-in required (RUFLO_HARNESS_LOOP=1)",
6
+ "generation": 0
7
+ },
8
+ "lineage": {
9
+ "generations": 0,
10
+ "attempts": 0,
11
+ "lineage": {
12
+ "generations": 0,
13
+ "candidatesEvaluated": 0,
14
+ "promotions": 0,
15
+ "rejections": 0,
16
+ "cumulativeHeldOutImprovement": 0,
17
+ "rootHash": null,
18
+ "branches": [],
19
+ "lineageIntact": false,
20
+ "allReplayable": true,
21
+ "nodes": [],
22
+ "problems": [
23
+ "expected exactly one immutable root, found 0"
24
+ ]
25
+ },
26
+ "plateau": {
27
+ "status": "insufficient-data",
28
+ "window": 5,
29
+ "medianImprovement": 0,
30
+ "promotionRate": 0,
31
+ "varianceShrinking": false,
32
+ "candidateVariance": 0,
33
+ "rationale": "need 5 generations, have 0"
34
+ },
35
+ "mutation": [],
36
+ "axisEffectiveness": [
37
+ {
38
+ "axis": "alpha",
39
+ "promotions": 0,
40
+ "meanDelta": 0
41
+ },
42
+ {
43
+ "axis": "subjectWeight",
44
+ "promotions": 0,
45
+ "meanDelta": 0
46
+ },
47
+ {
48
+ "axis": "mmrLambda",
49
+ "promotions": 0,
50
+ "meanDelta": 0
51
+ },
52
+ {
53
+ "axis": "bodyWeight",
54
+ "promotions": 0,
55
+ "meanDelta": 0
56
+ },
57
+ {
58
+ "axis": "typePenaltyFactor",
59
+ "promotions": 0,
60
+ "meanDelta": 0
61
+ }
62
+ ],
63
+ "cumulativeBenchmarkDelta": 0,
64
+ "cumulativeHumanRelevanceDelta": 0,
65
+ "humanEvalHash": null,
66
+ "served": {
67
+ "championHash": null,
68
+ "config": null,
69
+ "servedAt": null,
70
+ "fromGeneration": null
71
+ },
72
+ "champion": {
73
+ "config": {
74
+ "alpha": 0.5,
75
+ "subjectWeight": 2,
76
+ "mmrLambda": 0.7,
77
+ "bodyWeight": 1,
78
+ "typePenaltyFactor": 1
79
+ },
80
+ "hash": null
81
+ }
82
+ }
83
+ }
@@ -0,0 +1,55 @@
1
+ {
2
+ "timestamp": "2026-07-09T18:38:28.325Z",
3
+ "mode": "headless",
4
+ "workerType": "optimize",
5
+ "model": "sonnet",
6
+ "durationMs": 224644,
7
+ "executionId": "optimize_1783622083681_nznpnx",
8
+ "success": true,
9
+ "findings": {
10
+ "sections": [
11
+ {
12
+ "title": "Performance Analysis — `plugins/ruflo-metaharness`",
13
+ "content": "\nThis directory has no React/frontend code and no SQL database — it's a Claude Code plugin made of markdown-based skills/agents/commands plus 32 Node `.mjs` CLI scripts under `scripts/`. So \"N+1 queries\" and \"React re-renders\" don't apply literally; I translated them to the closest real equivalents: repeated subprocess spawns and redundant CLI invocations.\n\n",
14
+ "level": 2
15
+ },
16
+ {
17
+ "title": "1. Sequential subprocess spawns (the N+1 equivalent) — real, verified",
18
+ "content": "\n**`scripts/audit-list.mjs:100-113`** — for every audit record it wants to display, the script spawns a brand-new `npx @claude-flow/cli memory retrieve` process **synchronously, one at a time**:\n\n```js\n// current — sequential, blocking, one process per record\nfor (const key of slice) {\n const rec = memRetrieve(key); // spawnSync — full npx/CLI/ONNX cold-start each time\n if (!rec) continue;\n rows.push({ key, startedAt: rec.startedAt, ... });\n}\n```\n\nWith the default `--limit 20` (and up to 50 when called from `drift-from-history.mjs:199`), that's 20-50 sequential cold `npx` process starts — each carrying multi-second CLI/ONNX startup cost. Fix: switch to the async `spawn` pattern this same codebase already uses elsewhere (`runScriptJsonAsync` in `drift-from-history.mjs:130`, `execBinAsync` in `_harness.mjs:139`) and fan out with `Promise.all`:\n\n```js\nimport { spawn } from 'node:child_process';\n\nfunction memRetrieveAsync(key) {\n return new Promise((resolve) => {\n const p = spawn('npx', [CLI_PKG, 'memory', 'retrieve', '--namespace', NS, '--key', key],\n { stdio: ['ignore', 'pipe', 'pipe'], shell: process.platform === 'win32' });\n let out = '';\n p.stdout.on('data', (d) => { out += d; });\n p.on('close', () => {\n const m = /\\{[\\s\\S]*\\}/.exec(out);\n resolve(m ? (() => { try { return JSON.parse(m[0]); } catch { return null; } })() : null);\n });\n });\n}\n\nconst rows = (await Promise.all(slice.map(memRetrieveAsync)))\n .filter(Boolean)\n .map((rec, i) => ({ key: slice[i], startedAt: rec.startedAt, ... }));\n```\n\nThis is the highest-impact fix — it's hit on every `audit-list` and `drift-from-history` invocation, not just in CI.\n\n**`scripts/test-mcp-tools.mjs:130`** and **`scripts/test-graceful-degradation.mjs:136`** — same sequential-await-in-a-loop pattern over independent subprocess calls (up to 180s timeout each), but these only cost CI wall-clock time (worst case ~24 min for the 8-skill degradation drill), not user-facing latency. Same `Promise.allSettled` fix applies — `oia-audit.mjs` already has a `runAllParallel()` helper you can reuse as the template.\n\n",
19
+ "level": 3
20
+ },
21
+ {
22
+ "title": "2. Caching opportunities",
23
+ "content": "None outstanding — the codebase already does this well: `_harness.mjs` memoizes binary resolution in a module-level var, and `_invoke.mjs`'s `ensureCachedInstall` does one-time versioned npm-package caching.\n\n",
24
+ "level": 3
25
+ },
26
+ {
27
+ "title": "3. Memory leaks",
28
+ "content": "None found — subprocess handlers clean up timers/listeners, temp dirs are removed in `try/finally` blocks with `rmSync`.\n\n",
29
+ "level": 3
30
+ },
31
+ {
32
+ "title": "4. Redundant computations",
33
+ "content": "None found — `similarity()` / `parseMcpScanText()` are pure, sub-microsecond per the project's own micro-benchmarks (`bench-similarity.mjs`), and called at most once per invocation.\n\n",
34
+ "level": 3
35
+ },
36
+ {
37
+ "title": "5. React re-renders / N+1 DB queries",
38
+ "content": "N/A — no frontend or database code exists in this plugin.\n\n---\n\n**Bottom line:** this is a well-optimized codebase already (parallelized `oia-audit.mjs`, async `drift-from-history.mjs`, memoized installs) — the one real gap is `audit-list.mjs`'s sequential `memRetrieve` loop, worth fixing since it's on the hot path for both direct CLI use and the `drift-from-history` composite command.",
39
+ "level": 3
40
+ }
41
+ ],
42
+ "codeBlocks": [
43
+ {
44
+ "language": "js",
45
+ "code": "// current — sequential, blocking, one process per record\nfor (const key of slice) {\n const rec = memRetrieve(key); // spawnSync — full npx/CLI/ONNX cold-start each time\n if (!rec) continue;\n rows.push({ key, startedAt: rec.startedAt, ... });\n}"
46
+ },
47
+ {
48
+ "language": "js",
49
+ "code": "import { spawn } from 'node:child_process';\n\nfunction memRetrieveAsync(key) {\n return new Promise((resolve) => {\n const p = spawn('npx', [CLI_PKG, 'memory', 'retrieve', '--namespace', NS, '--key', key],\n { stdio: ['ignore', 'pipe', 'pipe'], shell: process.platform === 'win32' });\n let out = '';\n p.stdout.on('data', (d) => { out += d; });\n p.on('close', () => {\n const m = /\\{[\\s\\S]*\\}/.exec(out);\n resolve(m ? (() => { try { return JSON.parse(m[0]); } catch { return null; } })() : null);\n });\n });\n}\n\nconst rows = (await Promise.all(slice.map(memRetrieveAsync)))\n .filter(Boolean)\n .map((rec, i) => ({ key: slice[i], startedAt: rec.startedAt, ... }));"
50
+ }
51
+ ]
52
+ },
53
+ "rawOutputPreview": "Confirmed. I have a verified, concrete finding. Here's the analysis:\n\n## Performance Analysis — `plugins/ruflo-metaharness`\n\nThis directory has no React/frontend code and no SQL database — it's a Claude Code plugin made of markdown-based skills/agents/commands plus 32 Node `.mjs` CLI scripts under `scripts/`. So \"N+1 queries\" and \"React re-renders\" don't apply literally; I translated them to the closest real equivalents: repeated subprocess spawns and redundant CLI invocations.\n\n### 1. Sequential subprocess spawns (the N+1 equivalent) — real, verified\n\n**`scripts/audit-list.mjs:100-113`** — for every audit record it wants to display, the script spawns a brand-new `npx @claude-flow/cli memory retrieve` process **synchronously, one at a time**:\n\n```js\n// current — sequential, blocking, one process per record\nfor (const key of slice) {\n const rec = memRetrieve(key); // spawnSync — full npx/CLI/ONNX cold-start each time\n if (!rec) continue;\n rows.push({ key, startedAt: rec.startedAt, ... });\n}\n```\n\nWith the default `--limit 20` (and up to 50 when called from `drift-from-history.mjs:199`), that's 20-50 sequential cold `npx` process starts — each carrying multi-second CLI/ONNX startup cost. Fix: switch to the async `spawn` pattern this same codebase already uses elsewhere (`runScriptJsonAsync` in `drift-from-history.mjs:130`, `execBinAsync` in `_harness.mjs:139`) and fan out with `Promise.all`:\n\n```js\nimport { spawn } from 'node:child_process';\n\nfunction memRetrieveAsync(key) {\n return new Promise((resolve) => {\n const p = spawn('npx', [CLI_PKG, 'memory', 'retrieve', '--namespace', NS, '--key', key],\n { stdio: ['ignore', 'pipe', 'pipe'], shell: process.platform === 'win32' });\n let out = '';\n p.stdout.on('data', (d) => { out += d; });\n p.on('close', () => {\n const m = /\\{[\\s\\S]*\\}/.exec(out);\n resolve(m ? (() => { try { return JSON.parse(m[0]); } catch { return null; } })() : null);\n });\n });\n}\n\nconst rows = (await Promise.all(slice.ma",
54
+ "rawOutputLength": 3699
55
+ }
@@ -0,0 +1,206 @@
1
+ {
2
+ "timestamp": "2026-07-09T18:43:44.379Z",
3
+ "mode": "headless",
4
+ "workerType": "audit",
5
+ "model": "haiku",
6
+ "durationMs": 130702,
7
+ "executionId": "audit_1783622493677_zwc35w",
8
+ "success": true,
9
+ "findings": {
10
+ "vulnerabilities": [
11
+ {
12
+ "severity": "high",
13
+ "file": "scripts/test-with-openrouter.mjs",
14
+ "line": 65,
15
+ "category": "command-injection",
16
+ "title": "Shell Command Injection via String Interpolation",
17
+ "description": "The fetchSecretFromGcp() function interpolates the secretName parameter directly into an execSync() command without escaping or validation. A malicious secretName like 'openrouter_api_key; rm -rf /' would execute arbitrary shell commands.",
18
+ "code": "const out = execSync(`gcloud secrets versions access latest --secret=${secretName}`, ...)",
19
+ "impact": "Arbitrary code execution with the privileges of the running process",
20
+ "fix": "Use child_process.execFile() with array arguments instead of string interpolation"
21
+ },
22
+ {
23
+ "severity": "high",
24
+ "file": "scripts/test-with-openrouter.mjs",
25
+ "line": 109,
26
+ "category": "supply-chain",
27
+ "title": "Use of @latest Package Tag in Test Suite",
28
+ "description": "Uses 'metaharness@latest' which bypasses version pinning. A compromised upstream publish could execute arbitrary code on the next test run.",
29
+ "impact": "Supply chain attack - malicious upstream package could execute arbitrary code"
30
+ },
31
+ {
32
+ "severity": "high",
33
+ "file": "scripts/test-with-openrouter.mjs",
34
+ "line": 118,
35
+ "category": "supply-chain",
36
+ "title": "Use of @latest Package Tag for CLI",
37
+ "description": "Uses 'metaharness@latest' creating supply chain vulnerability",
38
+ "impact": "Supply chain attack vector"
39
+ },
40
+ {
41
+ "severity": "medium",
42
+ "file": "scripts/audit-list.mjs",
43
+ "line": 24,
44
+ "category": "supply-chain",
45
+ "title": "Conditional Use of @latest for CLI Package",
46
+ "description": "Uses '@claude-flow/cli@latest' when CLI_CORE env var is not set"
47
+ },
48
+ {
49
+ "severity": "medium",
50
+ "file": "scripts/audit-trend.mjs",
51
+ "line": 41,
52
+ "category": "supply-chain",
53
+ "title": "Conditional Use of @latest for CLI Package"
54
+ },
55
+ {
56
+ "severity": "medium",
57
+ "file": "scripts/drift-from-history.mjs",
58
+ "line": 47,
59
+ "category": "supply-chain",
60
+ "title": "Conditional Use of @latest for CLI Package"
61
+ },
62
+ {
63
+ "severity": "medium",
64
+ "file": "scripts/oia-audit.mjs",
65
+ "line": 38,
66
+ "category": "supply-chain",
67
+ "title": "Conditional Use of @latest for CLI Package"
68
+ },
69
+ {
70
+ "severity": "medium",
71
+ "file": "scripts/similarity.mjs",
72
+ "line": 33,
73
+ "category": "supply-chain",
74
+ "title": "Conditional Use of @latest for CLI Package"
75
+ },
76
+ {
77
+ "severity": "medium",
78
+ "file": "scripts/test-with-openrouter.mjs",
79
+ "line": 132,
80
+ "category": "information-disclosure",
81
+ "title": "Unvalidated Shell Command Execution",
82
+ "description": "execSync is used without proper error handling for shell commands. The gcloud account query could fail with sensitive information in error messages."
83
+ },
84
+ {
85
+ "severity": "low",
86
+ "file": "scripts/_invoke.mjs",
87
+ "line": 89,
88
+ "category": "environment-variable-validation",
89
+ "title": "Environment Variable Without Default Validation",
90
+ "description": "RUFLO_METAHARNESS_CACHE_BASE used for path construction without validation"
91
+ },
92
+ {
93
+ "severity": "low",
94
+ "file": "scripts/_invoke.mjs",
95
+ "line": 78,
96
+ "category": "regex-parsing",
97
+ "title": "Regex-based JSON Extraction",
98
+ "description": "Using regex to parse JSON from mixed output is fragile and could potentially be exploited"
99
+ },
100
+ {
101
+ "severity": "low",
102
+ "file": "scripts/audit-list.mjs",
103
+ "line": 21,
104
+ "category": "environment-variable-validation",
105
+ "title": "Unvalidated Namespace from Environment Variable"
106
+ },
107
+ {
108
+ "severity": "low",
109
+ "file": "scripts/test-with-openrouter.mjs",
110
+ "line": 40,
111
+ "category": "information-disclosure",
112
+ "title": "Temporary Directory Creation Without Secure Options"
113
+ }
114
+ ],
115
+ "riskScore": 68,
116
+ "summary": {
117
+ "critical": 0,
118
+ "high": 3,
119
+ "medium": 6,
120
+ "low": 4,
121
+ "total": 13
122
+ },
123
+ "recommendations": [
124
+ {
125
+ "priority": "CRITICAL",
126
+ "title": "Fix Command Injection in fetchSecretFromGcp()",
127
+ "files": [
128
+ "scripts/test-with-openrouter.mjs"
129
+ ],
130
+ "effort": "Low",
131
+ "fix": "Replace string interpolation with child_process.execFile(['gcloud', 'secrets', 'versions', 'access', 'latest', '--secret=' + secretName])"
132
+ },
133
+ {
134
+ "priority": "CRITICAL",
135
+ "title": "Remove @latest Package Tags from All Scripts",
136
+ "files": [
137
+ "scripts/test-with-openrouter.mjs",
138
+ "scripts/audit-list.mjs",
139
+ "scripts/audit-trend.mjs",
140
+ "scripts/drift-from-history.mjs",
141
+ "scripts/oia-audit.mjs",
142
+ "scripts/similarity.mjs"
143
+ ],
144
+ "effort": "Low",
145
+ "fix": "Replace '@claude-flow/cli@latest' with '@claude-flow/cli@~3.7.0' and 'metaharness@latest' with 'metaharness@~0.3.0'"
146
+ },
147
+ {
148
+ "priority": "HIGH",
149
+ "title": "Standardize Environment Variable Validation",
150
+ "effort": "Medium",
151
+ "fix": "Add validation for all environment variables used for paths, namespaces, and configuration"
152
+ },
153
+ {
154
+ "priority": "HIGH",
155
+ "title": "Replace String-based execSync with Array-based execFile",
156
+ "files": [
157
+ "scripts/test-with-openrouter.mjs",
158
+ "scripts/_invoke.mjs",
159
+ "scripts/_harness.mjs",
160
+ "scripts/_darwin.mjs",
161
+ "scripts/_redblue.mjs"
162
+ ],
163
+ "effort": "Medium",
164
+ "benefit": "Eliminates shell injection vulnerabilities"
165
+ },
166
+ {
167
+ "priority": "MEDIUM",
168
+ "title": "Add Security-focused Input Validation Module",
169
+ "effort": "Medium",
170
+ "benefit": "Reusable validation across all scripts"
171
+ },
172
+ {
173
+ "priority": "LOW",
174
+ "title": "Audit Dependencies for Known Vulnerabilities",
175
+ "command": "npm audit --production"
176
+ }
177
+ ],
178
+ "positiveSecurityFindings": [
179
+ {
180
+ "aspect": "Version Pinning in Production",
181
+ "description": "Main invocation helpers correctly use pinned version ranges (~0.3.0) instead of @latest",
182
+ "file": "scripts/_harness.mjs"
183
+ },
184
+ {
185
+ "aspect": "Graceful Degradation",
186
+ "description": "Proper graceful degradation when upstream packages are unavailable",
187
+ "file": "scripts/_invoke.mjs"
188
+ },
189
+ {
190
+ "aspect": "Secret Handling Awareness",
191
+ "description": "Echoes only length+prefix of API keys rather than full values",
192
+ "file": "scripts/test-with-openrouter.mjs"
193
+ },
194
+ {
195
+ "aspect": "Error Handling",
196
+ "description": "Proper try-catch blocks around JSON parsing and file operations"
197
+ },
198
+ {
199
+ "aspect": "Cleanup",
200
+ "description": "Temporary files properly cleaned up with recursive rmSync"
201
+ }
202
+ ]
203
+ },
204
+ "rawOutputPreview": "I've completed a comprehensive security analysis of the ruflo-metaharness codebase. Here's the detailed security audit report in JSON format:\n\n```json\n{\n \"vulnerabilities\": [\n {\n \"severity\": \"high\",\n \"file\": \"scripts/test-with-openrouter.mjs\",\n \"line\": 65,\n \"category\": \"command-injection\",\n \"title\": \"Shell Command Injection via String Interpolation\",\n \"description\": \"The fetchSecretFromGcp() function interpolates the secretName parameter directly into an execSync() command without escaping or validation. A malicious secretName like 'openrouter_api_key; rm -rf /' would execute arbitrary shell commands.\",\n \"code\": \"const out = execSync(`gcloud secrets versions access latest --secret=${secretName}`, ...)\",\n \"impact\": \"Arbitrary code execution with the privileges of the running process\",\n \"fix\": \"Use child_process.execFile() with array arguments instead of string interpolation\"\n },\n {\n \"severity\": \"high\",\n \"file\": \"scripts/test-with-openrouter.mjs\",\n \"line\": 109,\n \"category\": \"supply-chain\",\n \"title\": \"Use of @latest Package Tag in Test Suite\",\n \"description\": \"Uses 'metaharness@latest' which bypasses version pinning. A compromised upstream publish could execute arbitrary code on the next test run.\",\n \"impact\": \"Supply chain attack - malicious upstream package could execute arbitrary code\"\n },\n {\n \"severity\": \"high\",\n \"file\": \"scripts/test-with-openrouter.mjs\",\n \"line\": 118,\n \"category\": \"supply-chain\",\n \"title\": \"Use of @latest Package Tag for CLI\",\n \"description\": \"Uses 'metaharness@latest' creating supply chain vulnerability\",\n \"impact\": \"Supply chain attack vector\"\n },\n {\n \"severity\": \"medium\",\n \"file\": \"scripts/audit-list.mjs\",\n \"line\": 24,\n \"category\": \"supply-chain\",\n \"title\": \"Conditional Use of @latest for CLI Package\",\n \"description\": \"Uses '@claude-flow/cli@latest' when CLI_CORE env var is not s",
205
+ "rawOutputLength": 7864
206
+ }