@claude-flow/cli 3.38.12 → 3.38.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude/.proven-config-version +1 -0
- package/.claude/helpers/.helpers-version +1 -1
- package/.claude/helpers/helpers.manifest.json +2 -2
- package/.claude/helpers/statusline.cjs +0 -0
- package/.claude/proven-config.json +42 -0
- package/catalog-manifest.json +4 -4
- package/dist/src/mcp-tools/hooks-tools.js +6 -1
- package/dist/src/memory/memory-bridge.js +13 -2
- package/dist/src/ruvector/lattice-wasm.d.ts +14 -0
- package/dist/src/ruvector/lattice-wasm.js +144 -0
- package/dist/src/services/flywheel-receipt.d.ts +10 -0
- package/dist/src/services/flywheel-receipt.js +82 -7
- package/dist/src/services/flywheel-transaction.js +10 -1
- package/node_modules/@claude-flow/codex/dist/cli.js +0 -0
- package/node_modules/@claude-flow/plugin-agent-federation/dist/bin.js +0 -0
- package/node_modules/@claude-flow/security/dist/input-validator.d.ts +6 -6
- package/package.json +1 -1
- package/plugins/ruflo-metaharness/.claude-flow/daemon-state.json +178 -0
- package/plugins/ruflo-metaharness/.claude-flow/daemon.pid +1 -0
- package/plugins/ruflo-metaharness/.claude-flow/data/pending-insights.jsonl +5 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/daemon.log +269 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783604774864_ozbujc_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783604774864_ozbujc_result.log +108 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783605513587_ulvmpb_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783605513587_ulvmpb_result.log +209 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783606368867_ahysui_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783606368867_ahysui_result.log +192 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783607120257_lh05rb_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783607120257_lh05rb_result.log +13 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608020362_j3096j_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608020362_j3096j_result.log +120 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608776347_261b61_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783608776347_261b61_result.log +85 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783609621359_s5i6ye_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783609621359_s5i6ye_result.log +13 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610087998_qihv9v_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610087998_qihv9v_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610773920_qlzmxo_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783610773920_qlzmxo_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783611376090_xpqf1z_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783611376090_xpqf1z_result.log +16 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612097184_9rqfor_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612097184_9rqfor_result.log +138 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612811574_4u602j_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783612811574_4u602j_result.log +16 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783613750487_a46ttn_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783613750487_a46ttn_result.log +107 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783614289360_figwc0_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783614289360_figwc0_result.log +200 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615067640_l7tm6a_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615067640_l7tm6a_result.log +54 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615825308_44nor5_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783615825308_44nor5_result.log +85 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783616524771_ut1ftw_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783616524771_ut1ftw_result.log +266 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783617323039_fs3x5a_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783617323039_fs3x5a_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618049184_1f4yah_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618049184_1f4yah_result.log +96 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618793925_4ee7tf_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783618793925_4ee7tf_result.log +481 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783619642574_yjr4mm_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783619642574_yjr4mm_result.log +104 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783620392771_oduto0_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783620392771_oduto0_result.log +148 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621183670_gd0p1x_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621183670_gd0p1x_result.log +111 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621738388_z4k48b_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783621738388_z4k48b_result.log +89 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783622493677_zwc35w_prompt.log +19 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/audit_1783622493677_zwc35w_result.log +207 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783604894861_v6n3ut_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783604894861_v6n3ut_result.log +66 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783605934532_9h8ikb_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783605934532_9h8ikb_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783607181736_t12f4y_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783607181736_t12f4y_result.log +78 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783608341266_zhk0fl_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783608341266_zhk0fl_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783609420180_7xw817_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783609420180_7xw817_result.log +72 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783610535307_2pxofp_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783610535307_2pxofp_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783611437445_ofwnpb_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783611437445_ofwnpb_result.log +60 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783612556827_1cb112_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783612556827_1cb112_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783613579531_18xiax_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783613579531_18xiax_result.log +74 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783614650481_2zdz7w_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783614650481_2zdz7w_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783615742328_rjj69d_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783615742328_rjj69d_result.log +65 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783616767230_1iad99_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783616767230_1iad99_result.log +68 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783617783537_rqagku_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783617783537_rqagku_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783618817343_r1lbhr_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783618817343_r1lbhr_result.log +65 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783619907773_8msfw3_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783619907773_8msfw3_result.log +77 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783621009302_hybdum_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783621009302_hybdum_result.log +64 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783622083681_nznpnx_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783622083681_nznpnx_result.log +56 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/optimize_1783623208340_olsbaw_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783605134860_9jssz9_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783605134860_9jssz9_result.log +69 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783606516743_zftbaa_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783606516743_zftbaa_result.log +92 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783608038317_jrd66e_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783608038317_jrd66e_result.log +92 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783609388416_mc3zoe_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783609388416_mc3zoe_result.log +82 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783610821384_tzqvlb_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783610821384_tzqvlb_result.log +17 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783612023285_buygpo_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783612023285_buygpo_result.log +57 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783613599301_6f78cw_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783613599301_6f78cw_result.log +60 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783615000844_v95ues_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783615000844_v95ues_result.log +69 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783616341693_bl5d9o_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783616341693_bl5d9o_result.log +64 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783617832831_ha6s8d_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783617832831_ha6s8d_result.log +42 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783619384959_s5iiwf_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783619384959_s5iiwf_result.log +47 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783620946263_d7ovai_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783620946263_d7ovai_result.log +52 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783622473846_t839e5_prompt.log +14 -0
- package/plugins/ruflo-metaharness/.claude-flow/logs/headless/testgaps_1783622473846_t839e5_result.log +96 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/backup.json +7 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/codebase-map.json +11 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/consolidation.json +16 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/harness-loop.json +83 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/performance.json +55 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/security-audit.json +206 -0
- package/plugins/ruflo-metaharness/.claude-flow/metrics/test-gaps.json +95 -0
- package/plugins/ruflo-metaharness/.claude-flow/neural/stats.json +6 -0
- package/plugins/ruflo-metaharness/skills/harness-gepa/SKILL.md +1 -1
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T18:16:49.303Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze this codebase for performance optimizations:
|
|
4
|
+
- Identify N+1 query patterns
|
|
5
|
+
- Find unnecessary re-renders in React
|
|
6
|
+
- Suggest caching opportunities
|
|
7
|
+
- Identify memory leaks
|
|
8
|
+
- Find redundant computations
|
|
9
|
+
|
|
10
|
+
Provide actionable suggestions with code examples.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
[2026-07-09T18:19:43.669Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "Confirmed — `memRetrieve` at audit-list.mjs:56 uses `spawnSync('npx', ...)`, and it's called synchronously in the loop at line 104. That's a real sequential subprocess-per-item cost. Now I have verified findings for the report.\n\n## Performance Analysis: `plugins/ruflo-metaharness`\n\nNote: this plugin is a Node.js CLI toolset (32 `.mjs` scripts) with **no React/frontend and no database** — so I've mapped your requested categories onto their real equivalents here: subprocess/CLI-spawn amplification instead of N+1 queries, and I/O-bound scripts instead of re-renders. Most of the codebase already shows deliberate perf hardening (pinned-version installs to skip registry lookups, `Promise.all` fan-out in `oia-audit.mjs`/`drift-from-history.mjs`, `.slice()`-capped output, dedicated micro-benchmarks proving the similarity/scoring math is sub-microsecond). Four genuine issues remain, all in the same bucket: **sequential subprocess spawning where the work is independent**.\n\n### 1. `audit-list.mjs:99-117` — sequential `npx` spawn per record (confirmed)\n`memRetrieve(key)` (line 56) shells out via `spawnSync('npx', [...])` — a cold CLI invocation per call. It's called once per key in a plain `for` loop, so `--limit 50` (the default `drift-from-history.mjs` uses) means **up to 50 sequential subprocess spawns**.\n\n```js\n// before (audit-list.mjs:103-117)\nconst rows = [];\nfor (const key of slice) {\n const rec = memRetrieve(key);\n if (!rec) continue;\n rows.push({ key, /* ... */ });\n}\n\n// after — bounded concurrency instead of serial spawnSync\nconst CONCURRENCY = 8;\nconst rows = [];\nfor (let i = 0; i < slice.length; i += CONCURRENCY) {\n const batch = slice.slice(i, i + CONCURRENCY);\n const recs = await Promise.all(batch.map((key) => memRetrieveAsync(key))); // spawn() not spawnSync()\n batch.forEach((key, j) => {\n const rec = recs[j];\n if (rec) rows.push({ key, /* ... */ });\n });\n}\n```\nThis requires converting `memRetrieve` from `spawnSync` to async `spawn` (see `_harness.mjs:192` for the existing async pattern to mirror).\n\n### 2. `test-mcp-tools.mjs:130-214` — sequential Phase 3 tool tests (confirmed)\n15 MCP tool handlers run sequentially with `await`, several individually budgeted up to 180s for cold-cache `npx` fetches (`redblue`, `learn`, `gepa`, `drift_from_history`, `oia_audit`, `audit_list`). They have no data dependency on each other.\n\n```js\n// before: worst case = sum of all timeouts (could exceed 20+ min in CI)\nfor (const tool of tools) {\n await testTool(tool);\n}\n\n// after: worst case = max single timeout\nawait Promise.allSettled(tools.map((tool) => testTool(tool)));\n```\n\n### 3. `evolve.mjs:205-209` — synchronous file reads in `--diagnose` path (plausible, lower severity)\n`buildDiagnosis()` reads up to 100 run-record JSON files with `readFileSync` in a `for` loop.\n\n```js\n// before\nfor (const f of files) {\n const rec = JSON.parse(readFileSync(f, 'utf8'));\n // ...\n}\n\n// after\nconst recs = await Promise.all(\n files.map((f) => fs.promises.readFile(f, 'utf8').then(JSON.parse))\n);\n```\nCapped at 100 files and runs once per `evolve` invocation, so this is minor — worth fixing opportunistically, not urgently.\n\n### 4. `test-graceful-degradation.mjs:136-150` — 8 independent skill drills run serially (plausible)\nEach of the 8 skills gets its own subprocess + 180s timeout, run one after another (worst case ~24 min) even though each uses separate env vars and has no shared state.\n\n```js\n// after\nawait Promise.allSettled(skills.map((skill) => runDegradationDrill(skill)));\n```\n\n### Not found\nNo genuine issues in redundant computation, missing HTTP/API caching, or memory leaks — similarity/scoring math already has dedicated micro-benchmarks proving it's cheap, findings arrays are consistently `.slice()`-truncated, temp dirs are cleaned up in `finally` blocks, and the one OpenRouter call site (`test-with-openrouter.mjs`) is single-shot, so a cache layer isn't warranted.\n\n**Net effect if all four are applied**: CI wall-clock for `test-mcp-tools.mjs` and `test-graceful-degradation.mjs` drops from sum-of-timeouts to max-of-timeouts (potentially 10-20x on cold-cache runs), and `audit-list`/`drift-from-history` with large `--limit` values stop paying N sequential cold-spawn penalties.\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Performance Analysis: `plugins/ruflo-metaharness`",
|
|
10
|
+
"content": "\nNote: this plugin is a Node.js CLI toolset (32 `.mjs` scripts) with **no React/frontend and no database** — so I've mapped your requested categories onto their real equivalents here: subprocess/CLI-spawn amplification instead of N+1 queries, and I/O-bound scripts instead of re-renders. Most of the codebase already shows deliberate perf hardening (pinned-version installs to skip registry lookups, `Promise.all` fan-out in `oia-audit.mjs`/`drift-from-history.mjs`, `.slice()`-capped output, dedicated micro-benchmarks proving the similarity/scoring math is sub-microsecond). Four genuine issues remain, all in the same bucket: **sequential subprocess spawning where the work is independent**.\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "1. `audit-list.mjs:99-117` — sequential `npx` spawn per record (confirmed)",
|
|
15
|
+
"content": "`memRetrieve(key)` (line 56) shells out via `spawnSync('npx', [...])` — a cold CLI invocation per call. It's called once per key in a plain `for` loop, so `--limit 50` (the default `drift-from-history.mjs` uses) means **up to 50 sequential subprocess spawns**.\n\n```js\n// before (audit-list.mjs:103-117)\nconst rows = [];\nfor (const key of slice) {\n const rec = memRetrieve(key);\n if (!rec) continue;\n rows.push({ key, /* ... */ });\n}\n\n// after — bounded concurrency instead of serial spawnSync\nconst CONCURRENCY = 8;\nconst rows = [];\nfor (let i = 0; i < slice.length; i += CONCURRENCY) {\n const batch = slice.slice(i, i + CONCURRENCY);\n const recs = await Promise.all(batch.map((key) => memRetrieveAsync(key))); // spawn() not spawnSync()\n batch.forEach((key, j) => {\n const rec = recs[j];\n if (rec) rows.push({ key, /* ... */ });\n });\n}\n```\nThis requires converting `memRetrieve` from `spawnSync` to async `spawn` (see `_harness.mjs:192` for the existing async pattern to mirror).\n\n",
|
|
16
|
+
"level": 3
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "2. `test-mcp-tools.mjs:130-214` — sequential Phase 3 tool tests (confirmed)",
|
|
20
|
+
"content": "15 MCP tool handlers run sequentially with `await`, several individually budgeted up to 180s for cold-cache `npx` fetches (`redblue`, `learn`, `gepa`, `drift_from_history`, `oia_audit`, `audit_list`). They have no data dependency on each other.\n\n```js\n// before: worst case = sum of all timeouts (could exceed 20+ min in CI)\nfor (const tool of tools) {\n await testTool(tool);\n}\n\n// after: worst case = max single timeout\nawait Promise.allSettled(tools.map((tool) => testTool(tool)));\n```\n\n",
|
|
21
|
+
"level": 3
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "3. `evolve.mjs:205-209` — synchronous file reads in `--diagnose` path (plausible, lower severity)",
|
|
25
|
+
"content": "`buildDiagnosis()` reads up to 100 run-record JSON files with `readFileSync` in a `for` loop.\n\n```js\n// before\nfor (const f of files) {\n const rec = JSON.parse(readFileSync(f, 'utf8'));\n // ...\n}\n\n// after\nconst recs = await Promise.all(\n files.map((f) => fs.promises.readFile(f, 'utf8').then(JSON.parse))\n);\n```\nCapped at 100 files and runs once per `evolve` invocation, so this is minor — worth fixing opportunistically, not urgently.\n\n",
|
|
26
|
+
"level": 3
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "4. `test-graceful-degradation.mjs:136-150` — 8 independent skill drills run serially (plausible)",
|
|
30
|
+
"content": "Each of the 8 skills gets its own subprocess + 180s timeout, run one after another (worst case ~24 min) even though each uses separate env vars and has no shared state.\n\n```js\n// after\nawait Promise.allSettled(skills.map((skill) => runDegradationDrill(skill)));\n```\n\n",
|
|
31
|
+
"level": 3
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "Not found",
|
|
35
|
+
"content": "No genuine issues in redundant computation, missing HTTP/API caching, or memory leaks — similarity/scoring math already has dedicated micro-benchmarks proving it's cheap, findings arrays are consistently `.slice()`-truncated, temp dirs are cleaned up in `finally` blocks, and the one OpenRouter call site (`test-with-openrouter.mjs`) is single-shot, so a cache layer isn't warranted.\n\n**Net effect if all four are applied**: CI wall-clock for `test-mcp-tools.mjs` and `test-graceful-degradation.mjs` drops from sum-of-timeouts to max-of-timeouts (potentially 10-20x on cold-cache runs), and `audit-list`/`drift-from-history` with large `--limit` values stop paying N sequential cold-spawn penalties.",
|
|
36
|
+
"level": 3
|
|
37
|
+
}
|
|
38
|
+
],
|
|
39
|
+
"codeBlocks": [
|
|
40
|
+
{
|
|
41
|
+
"language": "js",
|
|
42
|
+
"code": "// before (audit-list.mjs:103-117)\nconst rows = [];\nfor (const key of slice) {\n const rec = memRetrieve(key);\n if (!rec) continue;\n rows.push({ key, /* ... */ });\n}\n\n// after — bounded concurrency instead of serial spawnSync\nconst CONCURRENCY = 8;\nconst rows = [];\nfor (let i = 0; i < slice.length; i += CONCURRENCY) {\n const batch = slice.slice(i, i + CONCURRENCY);\n const recs = await Promise.all(batch.map((key) => memRetrieveAsync(key))); // spawn() not spawnSync()\n batch.forEach((key, j) => {\n const rec = recs[j];\n if (rec) rows.push({ key, /* ... */ });\n });\n}"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"language": "js",
|
|
46
|
+
"code": "// before: worst case = sum of all timeouts (could exceed 20+ min in CI)\nfor (const tool of tools) {\n await testTool(tool);\n}\n\n// after: worst case = max single timeout\nawait Promise.allSettled(tools.map((tool) => testTool(tool)));"
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"language": "js",
|
|
50
|
+
"code": "// before\nfor (const f of files) {\n const rec = JSON.parse(readFileSync(f, 'utf8'));\n // ...\n}\n\n// after\nconst recs = await Promise.all(\n files.map((f) => fs.promises.readFile(f, 'utf8').then(JSON.parse))\n);"
|
|
51
|
+
},
|
|
52
|
+
{
|
|
53
|
+
"language": "js",
|
|
54
|
+
"code": "// after\nawait Promise.allSettled(skills.map((skill) => runDegradationDrill(skill)));"
|
|
55
|
+
}
|
|
56
|
+
]
|
|
57
|
+
},
|
|
58
|
+
"durationMs": 174367,
|
|
59
|
+
"model": "sonnet",
|
|
60
|
+
"sandboxMode": "permissive",
|
|
61
|
+
"workerType": "optimize",
|
|
62
|
+
"timestamp": "2026-07-09T18:19:43.669Z",
|
|
63
|
+
"executionId": "optimize_1783621009302_hybdum"
|
|
64
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T18:34:43.681Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze this codebase for performance optimizations:
|
|
4
|
+
- Identify N+1 query patterns
|
|
5
|
+
- Find unnecessary re-renders in React
|
|
6
|
+
- Suggest caching opportunities
|
|
7
|
+
- Identify memory leaks
|
|
8
|
+
- Find redundant computations
|
|
9
|
+
|
|
10
|
+
Provide actionable suggestions with code examples.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[2026-07-09T18:38:28.325Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "Confirmed. I have a verified, concrete finding. Here's the analysis:\n\n## Performance Analysis — `plugins/ruflo-metaharness`\n\nThis directory has no React/frontend code and no SQL database — it's a Claude Code plugin made of markdown-based skills/agents/commands plus 32 Node `.mjs` CLI scripts under `scripts/`. So \"N+1 queries\" and \"React re-renders\" don't apply literally; I translated them to the closest real equivalents: repeated subprocess spawns and redundant CLI invocations.\n\n### 1. Sequential subprocess spawns (the N+1 equivalent) — real, verified\n\n**`scripts/audit-list.mjs:100-113`** — for every audit record it wants to display, the script spawns a brand-new `npx @claude-flow/cli memory retrieve` process **synchronously, one at a time**:\n\n```js\n// current — sequential, blocking, one process per record\nfor (const key of slice) {\n const rec = memRetrieve(key); // spawnSync — full npx/CLI/ONNX cold-start each time\n if (!rec) continue;\n rows.push({ key, startedAt: rec.startedAt, ... });\n}\n```\n\nWith the default `--limit 20` (and up to 50 when called from `drift-from-history.mjs:199`), that's 20-50 sequential cold `npx` process starts — each carrying multi-second CLI/ONNX startup cost. Fix: switch to the async `spawn` pattern this same codebase already uses elsewhere (`runScriptJsonAsync` in `drift-from-history.mjs:130`, `execBinAsync` in `_harness.mjs:139`) and fan out with `Promise.all`:\n\n```js\nimport { spawn } from 'node:child_process';\n\nfunction memRetrieveAsync(key) {\n return new Promise((resolve) => {\n const p = spawn('npx', [CLI_PKG, 'memory', 'retrieve', '--namespace', NS, '--key', key],\n { stdio: ['ignore', 'pipe', 'pipe'], shell: process.platform === 'win32' });\n let out = '';\n p.stdout.on('data', (d) => { out += d; });\n p.on('close', () => {\n const m = /\\{[\\s\\S]*\\}/.exec(out);\n resolve(m ? (() => { try { return JSON.parse(m[0]); } catch { return null; } })() : null);\n });\n });\n}\n\nconst rows = (await Promise.all(slice.map(memRetrieveAsync)))\n .filter(Boolean)\n .map((rec, i) => ({ key: slice[i], startedAt: rec.startedAt, ... }));\n```\n\nThis is the highest-impact fix — it's hit on every `audit-list` and `drift-from-history` invocation, not just in CI.\n\n**`scripts/test-mcp-tools.mjs:130`** and **`scripts/test-graceful-degradation.mjs:136`** — same sequential-await-in-a-loop pattern over independent subprocess calls (up to 180s timeout each), but these only cost CI wall-clock time (worst case ~24 min for the 8-skill degradation drill), not user-facing latency. Same `Promise.allSettled` fix applies — `oia-audit.mjs` already has a `runAllParallel()` helper you can reuse as the template.\n\n### 2. Caching opportunities\nNone outstanding — the codebase already does this well: `_harness.mjs` memoizes binary resolution in a module-level var, and `_invoke.mjs`'s `ensureCachedInstall` does one-time versioned npm-package caching.\n\n### 3. Memory leaks\nNone found — subprocess handlers clean up timers/listeners, temp dirs are removed in `try/finally` blocks with `rmSync`.\n\n### 4. Redundant computations\nNone found — `similarity()` / `parseMcpScanText()` are pure, sub-microsecond per the project's own micro-benchmarks (`bench-similarity.mjs`), and called at most once per invocation.\n\n### 5. React re-renders / N+1 DB queries\nN/A — no frontend or database code exists in this plugin.\n\n---\n\n**Bottom line:** this is a well-optimized codebase already (parallelized `oia-audit.mjs`, async `drift-from-history.mjs`, memoized installs) — the one real gap is `audit-list.mjs`'s sequential `memRetrieve` loop, worth fixing since it's on the hot path for both direct CLI use and the `drift-from-history` composite command.\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Performance Analysis — `plugins/ruflo-metaharness`",
|
|
10
|
+
"content": "\nThis directory has no React/frontend code and no SQL database — it's a Claude Code plugin made of markdown-based skills/agents/commands plus 32 Node `.mjs` CLI scripts under `scripts/`. So \"N+1 queries\" and \"React re-renders\" don't apply literally; I translated them to the closest real equivalents: repeated subprocess spawns and redundant CLI invocations.\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "1. Sequential subprocess spawns (the N+1 equivalent) — real, verified",
|
|
15
|
+
"content": "\n**`scripts/audit-list.mjs:100-113`** — for every audit record it wants to display, the script spawns a brand-new `npx @claude-flow/cli memory retrieve` process **synchronously, one at a time**:\n\n```js\n// current — sequential, blocking, one process per record\nfor (const key of slice) {\n const rec = memRetrieve(key); // spawnSync — full npx/CLI/ONNX cold-start each time\n if (!rec) continue;\n rows.push({ key, startedAt: rec.startedAt, ... });\n}\n```\n\nWith the default `--limit 20` (and up to 50 when called from `drift-from-history.mjs:199`), that's 20-50 sequential cold `npx` process starts — each carrying multi-second CLI/ONNX startup cost. Fix: switch to the async `spawn` pattern this same codebase already uses elsewhere (`runScriptJsonAsync` in `drift-from-history.mjs:130`, `execBinAsync` in `_harness.mjs:139`) and fan out with `Promise.all`:\n\n```js\nimport { spawn } from 'node:child_process';\n\nfunction memRetrieveAsync(key) {\n return new Promise((resolve) => {\n const p = spawn('npx', [CLI_PKG, 'memory', 'retrieve', '--namespace', NS, '--key', key],\n { stdio: ['ignore', 'pipe', 'pipe'], shell: process.platform === 'win32' });\n let out = '';\n p.stdout.on('data', (d) => { out += d; });\n p.on('close', () => {\n const m = /\\{[\\s\\S]*\\}/.exec(out);\n resolve(m ? (() => { try { return JSON.parse(m[0]); } catch { return null; } })() : null);\n });\n });\n}\n\nconst rows = (await Promise.all(slice.map(memRetrieveAsync)))\n .filter(Boolean)\n .map((rec, i) => ({ key: slice[i], startedAt: rec.startedAt, ... }));\n```\n\nThis is the highest-impact fix — it's hit on every `audit-list` and `drift-from-history` invocation, not just in CI.\n\n**`scripts/test-mcp-tools.mjs:130`** and **`scripts/test-graceful-degradation.mjs:136`** — same sequential-await-in-a-loop pattern over independent subprocess calls (up to 180s timeout each), but these only cost CI wall-clock time (worst case ~24 min for the 8-skill degradation drill), not user-facing latency. Same `Promise.allSettled` fix applies — `oia-audit.mjs` already has a `runAllParallel()` helper you can reuse as the template.\n\n",
|
|
16
|
+
"level": 3
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "2. Caching opportunities",
|
|
20
|
+
"content": "None outstanding — the codebase already does this well: `_harness.mjs` memoizes binary resolution in a module-level var, and `_invoke.mjs`'s `ensureCachedInstall` does one-time versioned npm-package caching.\n\n",
|
|
21
|
+
"level": 3
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "3. Memory leaks",
|
|
25
|
+
"content": "None found — subprocess handlers clean up timers/listeners, temp dirs are removed in `try/finally` blocks with `rmSync`.\n\n",
|
|
26
|
+
"level": 3
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "4. Redundant computations",
|
|
30
|
+
"content": "None found — `similarity()` / `parseMcpScanText()` are pure, sub-microsecond per the project's own micro-benchmarks (`bench-similarity.mjs`), and called at most once per invocation.\n\n",
|
|
31
|
+
"level": 3
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "5. React re-renders / N+1 DB queries",
|
|
35
|
+
"content": "N/A — no frontend or database code exists in this plugin.\n\n---\n\n**Bottom line:** this is a well-optimized codebase already (parallelized `oia-audit.mjs`, async `drift-from-history.mjs`, memoized installs) — the one real gap is `audit-list.mjs`'s sequential `memRetrieve` loop, worth fixing since it's on the hot path for both direct CLI use and the `drift-from-history` composite command.",
|
|
36
|
+
"level": 3
|
|
37
|
+
}
|
|
38
|
+
],
|
|
39
|
+
"codeBlocks": [
|
|
40
|
+
{
|
|
41
|
+
"language": "js",
|
|
42
|
+
"code": "// current — sequential, blocking, one process per record\nfor (const key of slice) {\n const rec = memRetrieve(key); // spawnSync — full npx/CLI/ONNX cold-start each time\n if (!rec) continue;\n rows.push({ key, startedAt: rec.startedAt, ... });\n}"
|
|
43
|
+
},
|
|
44
|
+
{
|
|
45
|
+
"language": "js",
|
|
46
|
+
"code": "import { spawn } from 'node:child_process';\n\nfunction memRetrieveAsync(key) {\n return new Promise((resolve) => {\n const p = spawn('npx', [CLI_PKG, 'memory', 'retrieve', '--namespace', NS, '--key', key],\n { stdio: ['ignore', 'pipe', 'pipe'], shell: process.platform === 'win32' });\n let out = '';\n p.stdout.on('data', (d) => { out += d; });\n p.on('close', () => {\n const m = /\\{[\\s\\S]*\\}/.exec(out);\n resolve(m ? (() => { try { return JSON.parse(m[0]); } catch { return null; } })() : null);\n });\n });\n}\n\nconst rows = (await Promise.all(slice.map(memRetrieveAsync)))\n .filter(Boolean)\n .map((rec, i) => ({ key: slice[i], startedAt: rec.startedAt, ... }));"
|
|
47
|
+
}
|
|
48
|
+
]
|
|
49
|
+
},
|
|
50
|
+
"durationMs": 224644,
|
|
51
|
+
"model": "sonnet",
|
|
52
|
+
"sandboxMode": "permissive",
|
|
53
|
+
"workerType": "optimize",
|
|
54
|
+
"timestamp": "2026-07-09T18:38:28.325Z",
|
|
55
|
+
"executionId": "optimize_1783622083681_nznpnx"
|
|
56
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T18:53:28.340Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze this codebase for performance optimizations:
|
|
4
|
+
- Identify N+1 query patterns
|
|
5
|
+
- Find unnecessary re-renders in React
|
|
6
|
+
- Suggest caching opportunities
|
|
7
|
+
- Identify memory leaks
|
|
8
|
+
- Find redundant computations
|
|
9
|
+
|
|
10
|
+
Provide actionable suggestions with code examples.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T13:52:14.861Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
[2026-07-09T13:55:16.742Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "## Test Coverage Analysis — `plugins/ruflo-metaharness`\n\nThis plugin uses a hand-rolled test harness (`test-*.mjs` scripts with a custom `assert()` helper, no Jest/Vitest), plus `scripts/smoke.sh` (bash, mostly structural/CLI-registration grep checks) run in CI via `no-metaharness-smoke.yml` / `metaharness-ci.yml`.\n\n**Existing coverage:**\n| Test file | Covers |\n|---|---|\n| `test-similarity.mjs` | `_similarity.mjs` (all 5 fns) + `_harness.mjs` severity primitives |\n| `test-graceful-degradation.mjs` | CLI-level exit-0/degraded contract for `score/genome/mcp-scan/threat-model/oia-audit/audit-list/audit-trend/mint` only |\n| `test-mcp-tools.mjs` | MCP tool handler contract (shape, no-throw) against compiled dist |\n| `test-parallel-pipeline.mjs` | `router-parallel-analyze.mjs` as a black-box subprocess against a synthesized JSONL fixture |\n| `test-pipeline-roundtrip.mjs` | `oia-audit` → `audit-trend` e2e chain |\n| `test-with-openrouter.mjs` | Real e2e against OpenRouter (costs money, needs GCP secret) |\n| `smoke.sh` | Structural grep + CLI subcommand/`--help` registration for everything |\n\n## 1. Untested functions/classes\n\n**`_invoke.mjs` — zero direct unit tests.** This is the highest-value gap: it's the extracted shared plumbing (per its own header, \"converged security/perf/arch review\") used by ~15 scripts, made of small pure/near-pure functions, yet no test imports it directly — only indirectly through subprocess-level e2e tests that need npm/network.\n- `classifyDegraded(stderr, exitCode, reasonPrefix)` — untested\n- `injectJson(args, wantJson)` — untested\n- `parseTrailingJson(stdout)` — untested (this one **replaced a live bug** per the comment — the exact kind of function that regresses silently without a unit test)\n- `satisfiesTildeRange(version, pinVersion)` — untested\n- `findLocalPackageDir(pkg, pinVersion)` — untested (env-seam `RUFLO_METAHARNESS_SKIP_LOCAL` exists but nothing exercises it)\n- `cacheBaseDir()` — untested (env-seam `RUFLO_METAHARNESS_CACHE_BASE` exists but nothing exercises it)\n- `makeDegradedEmitter(pkg, pinVersion)` — untested\n\n**`_darwin.mjs` / `_redblue.mjs`** — only exercised transitively (network-dependent, via smoke.sh registration checks), never unit-tested directly:\n- `_darwin.mjs`: `runDarwin`, `runDarwinAsync` (streaming/`onProgress`/abort-signal paths), `importGepa`\n- `_redblue.mjs`: `runRedblue`, `emitRedblueDegradedJsonAndExit`\n\n**Pure parsing/logic functions, module-private (not exported), with no direct test:**\n- `security-bench.mjs: parseSecurityBenchMarkdown()` — 3 independent regex extractions (overall verdict, gate lines, baseline table)\n- `redblue.mjs: buildUpstreamArgs()` — per-subcommand argv construction + validation (`attack` family check, `report` requires `--in`)\n- `evolve.mjs: looksLikeGepaTranscript()`, `extractGepaTranscripts()`, `summarizeTraces()`, `safetyChecks()`\n- `audit-list.mjs: parseDurationMs()` — duration-spec parser (`7d`, `24h`, `1w`, `1m`)\n- `gepa.mjs: readJsonFile()`, `loadGenomeOrExit()`\n- `router-parallel-analyze.mjs: median()`, `pctile()`, `mean()` — pure stats, only exercised indirectly through a full subprocess run in `test-parallel-pipeline.mjs`\n\n**Note:** all of the above are unexported module-level functions in CLI scripts that run `main()` on import — that's *why* nobody unit-tests them directly today. Testing them requires either (a) exporting them (cheap, mirrors what `_similarity.mjs` already does) or (b) testing through stdout/exit-code capture of the whole script. Skeletons below assume (a) since it's the established pattern in this codebase.\n\n## 2. Edge cases not covered\n\n- `parseTrailingJson`: multiple `{...}` blocks where an earlier block is unparseable JSON but a *later* one nests braces the lazy regex fragments (the exact bug class the comment says this fixed) — no fixture actually exercises nested-object recovery via the greedy fallback.\n- `satisfiesTildeRange`: version with a `-alpha.N` suffix, `pinVersion` without a leading `~`, garbage input.\n- `classifyDegraded`: `exitCode === 0` with degraded-looking stderr (should NOT degrade — success takes precedence over noisy stderr); `stderr` undefined vs empty string.\n- `findLocalPackageDir`: version present but pre-1.0 (`0.x.y`) tilde matching; symlinked node_modules; two ancestors both having a stale version (should pick a satisfying one over the first found).\n- `parseDurationMs`: unit `m` is ambiguous (30-day \"month\" vs minutes) — worth a regression test pinning the intended semantics; malformed specs (`7`, `d7`, `-7d`).\n- `parseSecurityBenchMarkdown`: `overallMatch` is `null` (upstream changed emoji/wording) — does the caller handle a `null` gate list gracefully?\n- `buildUpstreamArgs('attack')`: missing `--count` (`ARGS.count === null`) should omit the flag, not emit `--count null`.\n\n## 3. Missing error handling tests\n\n- `ensureCachedInstall`: `mkdirSync` throw path (`cache-dir-create-failed`) and `npm install` non-zero exit (`install-failed`) — both branches untested.\n- `importOptionalLibrary`: the `recoverable` regex gate — an *unrelated* `TypeError` thrown from within an installed module must NOT be swallowed (re-thrown), only `MODULE_NOT_FOUND`-shaped errors should fall through to cached-install. No test proves non-recoverable errors propagate.\n- `evolve.mjs safetyChecks()`: none of `--generations`/`--children`/`--concurrency` out-of-range, invalid `--sandbox`/`--selection`/`--mutator`, or nonexistent `--repo` path are unit-tested (only reachable via full CLI spawn today).\n- `security-bench.mjs safetyChecks()`: `--population`/`--cycles` bounds untested directly.\n- `redblue.mjs`: `report` without `--in`, `attack` with invalid family — both call `process.exit(2)`, untested.\n\n## 4. Integration test gaps\n\n- **`evolve` / `gepa` / `learn` / `redblue` / `bench` / `security-bench` are absent from `test-graceful-degradation.mjs`.** The file's own header explicitly scopes to 8 skills and says nothing about these six. Given ADR-150's architectural constraint (\"ruflo remains operational if every MetaHarness package is removed\") is supposed to apply project-wide, this is a real gate gap, not just a nice-to-have — a regression in `_darwin.mjs`'s or `_redblue.mjs`'s degraded path for these specific subcommands would ship silently.\n- `drift-from-history.mjs` composing 3 primitives (`audit-list` → `similarity`/`audit-trend`) has no dedicated round-trip test analogous to `test-pipeline-roundtrip.mjs` — only structural grep in `smoke.sh`.\n- No test exercises `RUFLO_METAHARNESS_CACHE_BASE` / `RUFLO_METAHARNESS_SKIP_LOCAL` env seams that `_invoke.mjs` explicitly documents as \"used by test-graceful-degradation.mjs\" — grep confirms they aren't actually referenced there.\n\n## Test skeletons\n\n**New file: `scripts/test-invoke.mjs`** (mirrors `test-similarity.mjs`'s style; requires exporting nothing new — everything below is already exported)\n\n```js\n#!/usr/bin/env node\n// test-invoke.mjs — unit tests for the shared _invoke.mjs plumbing\n// (classifyDegraded, injectJson, parseTrailingJson, satisfiesTildeRange,\n// findLocalPackageDir, cacheBaseDir). No network, no subprocess spawns.\n\nimport { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nimport {\n classifyDegraded, injectJson, parseTrailingJson,\n satisfiesTildeRange, findLocalPackageDir, cacheBaseDir,\n} from './_invoke.mjs';\n\nlet passed = 0, failed = 0;\nconst failures = [];\nfunction assert(cond, label) {\n if (cond) { console.log(` ✓ ${label}`); passed++; }\n else { console.log(` ✗ ${label}`); failures.push(label); failed++; }\n}\n\nconsole.log('# classifyDegraded');\nassert(classifyDegraded('', null, 'x').degraded === true, 'null exitCode -> timeout');\nassert(classifyDegraded('', null, 'x').reason === 'x-timeout', 'timeout reason prefixed');\nassert(classifyDegraded('npm ERR! 404', 1, 'x').reason === 'x-not-available', 'npm ERR matches DEGRADED_RX');\nassert(classifyDegraded('some noisy stderr', 0, 'x').degraded === false, 'exit 0 with noisy stderr is NOT degraded');\nassert(classifyDegraded(undefined, 1, 'x').degraded === false, 'undefined stderr does not throw / does not false-positive');\n\nconsole.log('\\n# injectJson');\nassert(injectJson(['run'], true).includes('--json'), 'appends --json when wanted');\nassert(!injectJson(['run'], false).includes('--json'), 'omits --json when not wanted');\nassert(injectJson(['run', '--json'], true).filter((a) => a === '--json').length === 1, 'no duplicate --json');\n\nconsole.log('\\n# parseTrailingJson');\nassert(parseTrailingJson('progress\\n{\"a\":1}\\n{\"b\":2}').b === 2, 'picks LAST block, not first');\nassert(parseTrailingJson('{\"a\": {\"nested\": {\"deep\": 1}}}').a.nested.deep === 1, 'greedy fallback recovers nested braces');\nassert(parseTrailingJson('no json here') === null, 'no braces -> null, never throws');\nassert(parseTrailingJson('{\"a\":1} garbage {not json}').a === 1, 'unparseable trailing block falls back to earlier valid one');\n\nconsole.log('\\n# satisfiesTildeRange');\nassert(satisfiesTildeRange('0.8.3', '~0.8.0') === true, 'patch >= pin satisfies');\nassert(satisfiesTildeRange('0.8.0', '~0.8.0') === true, 'exact patch satisfies');\nassert(satisfiesTildeRange('0.9.0', '~0.8.0') === false, 'minor bump does not satisfy tilde range');\nassert(satisfiesTildeRange('not-a-version', '~0.8.0') === false, 'garbage version -> false, not throw');\nassert(satisfiesTildeRange('0.8.3', 'garbage-pin') === false, 'garbage pin -> false, not throw');\n\nconsole.log('\\n# cacheBaseDir + RUFLO_METAHARNESS_CACHE_BASE seam');\n{\n const prev = process.env.RUFLO_METAHARNESS_CACHE_BASE;\n delete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n assert(cacheBaseDir().endsWith('.ruflo'), 'defaults to ~/.ruflo');\n process.env.RUFLO_METAHARNESS_CACHE_BASE = '/tmp/custom-cache';\n assert(cacheBaseDir() === '/tmp/custom-cache', 'env override respected');\n if (prev === undefined) delete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n else process.env.RUFLO_METAHARNESS_CACHE_BASE = prev;\n}\n\nconsole.log('\\n# findLocalPackageDir + RUFLO_METAHARNESS_SKIP_LOCAL seam');\n{\n const fixture = mkdtempSync(join(tmpdir(), 'invoke-fixture-'));\n const pkgDir = join(fixture, 'node_modules', '@metaharness', 'darwin');\n mkdirSync(pkgDir, { recursive: true });\n writeFileSync(join(pkgDir, 'package.json'), JSON.stringify({ version: '0.8.3' }));\n const prevCwd = process.cwd();\n process.chdir(fixture);\n try {\n assert(findLocalPackageDir('@metaharness/darwin', '~0.8.0') === pkgDir, 'finds satisfying local install by walking cwd');\n assert(findLocalPackageDir('@metaharness/darwin', '~0.9.0') === null, 'rejects local install that does not satisfy pin');\n process.env.RUFLO_METAHARNESS_SKIP_LOCAL = '1';\n assert(findLocalPackageDir('@metaharness/darwin', '~0.8.0') === null, 'SKIP_LOCAL seam disables resolution entirely');\n } finally {\n delete process.env.RUFLO_METAHARNESS_SKIP_LOCAL;\n process.chdir(prevCwd);\n rmSync(fixture, { recursive: true, force: true });\n }\n}\n\nconsole.log(`\\n${passed} passed, ${failed} failed`);\nif (failed > 0) { console.log('Failures:', failures.join(', ')); process.exit(1); }\n```\n\n**Extend `test-graceful-degradation.mjs`** — add the 6 missing skills to close the ADR-150 gate gap:\n\n```js\n// Add alongside the existing 8-skill list:\nconst MISSING_SKILLS = ['evolve', 'gepa', 'learn', 'redblue', 'bench', 'security-bench'];\nfor (const skill of MISSING_SKILLS) {\n // same drill as the existing loop: spawn with npm_config_registry pointed\n // at an unresolvable host, assert exitCode === 0 and stdout contains\n // \"degraded\": true — these six currently escape the drill entirely.\n}\n```\n\n**New file: `scripts/test-cli-parsers.mjs`** (requires adding `export` to the named functions in `security-bench.mjs`, `redblue.mjs`, `audit-list.mjs`, `evolve.mjs`, `gepa.mjs` — currently module-private, which is *why* they're untestable in isolation today):\n\n```js\n#!/usr/bin/env node\n// test-cli-parsers.mjs — unit tests for pure parsing/validation logic\n// embedded in CLI scripts. Requires exporting: parseSecurityBenchMarkdown\n// (security-bench.mjs), buildUpstreamArgs (redblue.mjs), parseDurationMs\n// (audit-list.mjs), looksLikeGepaTranscript/extractGepaTranscripts/\n// summarizeTraces (evolve.mjs).\n\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nimport { parseDurationMs } from './audit-list.mjs';\nimport { looksLikeGepaTranscript, extractGepaTranscripts, summarizeTraces } from './evolve.mjs';\n\nlet passed = 0, failed = 0;\nconst failures = [];\nfunction assert(cond, label) {\n if (cond) { console.log(` ✓ ${label}`); passed++; }\n else { console.log(` ✗ ${label}`); failures.push(label); failed++; }\n}\n\nconsole.log('# parseSecurityBenchMarkdown');\nconst md = `\n**Overall: ✅ PASS**\n- ✅ **TPR improvement ≥ 25% vs fixed harness** — +150%\n- ❌ **FPR ≤ 5%** — 8%\n| B1 baseline | 0.42 | 0.60 | 0.10 | 0.90 | 0.95 | 2 | $0.10 |\n`;\nconst parsed = parseSecurityBenchMarkdown(md);\nassert(parsed.overall.ok === true, 'overall PASS parsed');\nassert(parsed.gates.length === 2, 'both gate lines parsed');\nassert(parsed.gates[1].ok === false, 'FAIL gate icon parsed correctly');\nassert(parsed.baselines[0].fitness === 0.42, 'baseline table row parsed');\nassert(parseSecurityBenchMarkdown('no matches here').overall === null, 'missing Overall line -> null, not throw');\nassert(parseSecurityBenchMarkdown('no matches here').gates.length === 0, 'no gate lines -> empty array, not throw');\n\nconsole.log('\\n# parseDurationMs');\nassert(parseDurationMs('7d') === 7 * 86400_000, '7d parses');\nassert(parseDurationMs('24h') === 24 * 3600_000, '24h parses');\nassert(parseDurationMs('1w') === 7 * 86400_000, '1w parses');\nassert(parseDurationMs('garbage') === null, 'malformed spec -> null');\nassert(parseDurationMs('-7d') === null, 'negative spec -> null (regex requires digits only)');\nassert(parseDurationMs('7') === null, 'missing unit -> null');\n\nconsole.log('\\n# evolve.mjs gepa-transcript helpers');\nassert(looksLikeGepaTranscript([{ actionRaw: 'x' }]) === true, 'actionRaw-shaped entries recognized');\nassert(looksLikeGepaTranscript([{ obs: 'y' }]) === true, 'obs-shaped entries recognized');\nassert(looksLikeGepaTranscript([]) === false, 'empty array is not a transcript');\nassert(looksLikeGepaTranscript([{ foo: 1 }]) === false, 'unrelated shape rejected');\nassert(looksLikeGepaTranscript(null) === false, 'null input -> false, not throw');\n\nconst record = { transcript: [{ actionRaw: 'a' }], traces: [{ taskId: 't1', exitCode: 1, timedOut: true, blockedActions: ['x'] }] };\nassert(extractGepaTranscripts(record).length === 1, 'top-level transcript extracted');\nconst summary = summarizeTraces(record);\nassert(summary.tasks === 1 && summary.failed === 1 && summary.timedOut === 1 && summary.blockedActions === 1, 'summarizeTraces tallies correctly');\nassert(summarizeTraces({}).tasks === 0, 'missing traces array -> zeroed summary, not throw');\n\nconsole.log(`\\n${passed} passed, ${failed} failed`);\nif (failed > 0) { console.log('Failures:', failures.join(', ')); process.exit(1); }\n```\n\n**New file: `scripts/test-router-parallel-math.mjs`** (requires exporting `median`/`pctile`/`mean` from `router-parallel-analyze.mjs`):\n\n```js\n#!/usr/bin/env node\nimport { median, pctile, mean } from './router-parallel-analyze.mjs';\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { if (cond) passed++; else { console.log(`✗ ${label}`); failed++; } }\n\nassert(median([1, 2, 3]) === 2, 'odd-length median');\nassert(median([1, 2, 3, 4]) === 2.5, 'even-length median averages middle two');\nassert(median([]) === undefined || Number.isNaN(median([])), 'empty array does not throw');\nassert(pctile([1, 2, 3, 4, 5], 0.5) === median([1, 2, 3, 4, 5]), 'p50 matches median');\nassert(mean([2, 4, 6]) === 4, 'mean computes correctly');\nassert(Number.isNaN(mean([])) || mean([]) === 0, 'empty array mean defined (pin the actual convention)');\n\nconsole.log(`${passed} passed, ${failed} failed`);\nif (failed > 0) process.exit(1);\n```\n\n## Priority\n\n1. `scripts/test-invoke.mjs` — closes the biggest real gap (shared plumbing, zero direct tests, already caused one documented regression).\n2. Extend `test-graceful-degradation.mjs` to the 6 missing skills — closes an actual ADR-150 CI gate hole.\n3. `test-cli-parsers.mjs` + `test-router-parallel-math.mjs` — cheap, pure-function tests once the target functions are exported (small, low-risk diffs to each source file).\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Test Coverage Analysis — `plugins/ruflo-metaharness`",
|
|
10
|
+
"content": "\nThis plugin uses a hand-rolled test harness (`test-*.mjs` scripts with a custom `assert()` helper, no Jest/Vitest), plus `scripts/smoke.sh` (bash, mostly structural/CLI-registration grep checks) run in CI via `no-metaharness-smoke.yml` / `metaharness-ci.yml`.\n\n**Existing coverage:**\n| Test file | Covers |\n|---|---|\n| `test-similarity.mjs` | `_similarity.mjs` (all 5 fns) + `_harness.mjs` severity primitives |\n| `test-graceful-degradation.mjs` | CLI-level exit-0/degraded contract for `score/genome/mcp-scan/threat-model/oia-audit/audit-list/audit-trend/mint` only |\n| `test-mcp-tools.mjs` | MCP tool handler contract (shape, no-throw) against compiled dist |\n| `test-parallel-pipeline.mjs` | `router-parallel-analyze.mjs` as a black-box subprocess against a synthesized JSONL fixture |\n| `test-pipeline-roundtrip.mjs` | `oia-audit` → `audit-trend` e2e chain |\n| `test-with-openrouter.mjs` | Real e2e against OpenRouter (costs money, needs GCP secret) |\n| `smoke.sh` | Structural grep + CLI subcommand/`--help` registration for everything |\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "1. Untested functions/classes",
|
|
15
|
+
"content": "\n**`_invoke.mjs` — zero direct unit tests.** This is the highest-value gap: it's the extracted shared plumbing (per its own header, \"converged security/perf/arch review\") used by ~15 scripts, made of small pure/near-pure functions, yet no test imports it directly — only indirectly through subprocess-level e2e tests that need npm/network.\n- `classifyDegraded(stderr, exitCode, reasonPrefix)` — untested\n- `injectJson(args, wantJson)` — untested\n- `parseTrailingJson(stdout)` — untested (this one **replaced a live bug** per the comment — the exact kind of function that regresses silently without a unit test)\n- `satisfiesTildeRange(version, pinVersion)` — untested\n- `findLocalPackageDir(pkg, pinVersion)` — untested (env-seam `RUFLO_METAHARNESS_SKIP_LOCAL` exists but nothing exercises it)\n- `cacheBaseDir()` — untested (env-seam `RUFLO_METAHARNESS_CACHE_BASE` exists but nothing exercises it)\n- `makeDegradedEmitter(pkg, pinVersion)` — untested\n\n**`_darwin.mjs` / `_redblue.mjs`** — only exercised transitively (network-dependent, via smoke.sh registration checks), never unit-tested directly:\n- `_darwin.mjs`: `runDarwin`, `runDarwinAsync` (streaming/`onProgress`/abort-signal paths), `importGepa`\n- `_redblue.mjs`: `runRedblue`, `emitRedblueDegradedJsonAndExit`\n\n**Pure parsing/logic functions, module-private (not exported), with no direct test:**\n- `security-bench.mjs: parseSecurityBenchMarkdown()` — 3 independent regex extractions (overall verdict, gate lines, baseline table)\n- `redblue.mjs: buildUpstreamArgs()` — per-subcommand argv construction + validation (`attack` family check, `report` requires `--in`)\n- `evolve.mjs: looksLikeGepaTranscript()`, `extractGepaTranscripts()`, `summarizeTraces()`, `safetyChecks()`\n- `audit-list.mjs: parseDurationMs()` — duration-spec parser (`7d`, `24h`, `1w`, `1m`)\n- `gepa.mjs: readJsonFile()`, `loadGenomeOrExit()`\n- `router-parallel-analyze.mjs: median()`, `pctile()`, `mean()` — pure stats, only exercised indirectly through a full subprocess run in `test-parallel-pipeline.mjs`\n\n**Note:** all of the above are unexported module-level functions in CLI scripts that run `main()` on import — that's *why* nobody unit-tests them directly today. Testing them requires either (a) exporting them (cheap, mirrors what `_similarity.mjs` already does) or (b) testing through stdout/exit-code capture of the whole script. Skeletons below assume (a) since it's the established pattern in this codebase.\n\n",
|
|
16
|
+
"level": 2
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "2. Edge cases not covered",
|
|
20
|
+
"content": "\n- `parseTrailingJson`: multiple `{...}` blocks where an earlier block is unparseable JSON but a *later* one nests braces the lazy regex fragments (the exact bug class the comment says this fixed) — no fixture actually exercises nested-object recovery via the greedy fallback.\n- `satisfiesTildeRange`: version with a `-alpha.N` suffix, `pinVersion` without a leading `~`, garbage input.\n- `classifyDegraded`: `exitCode === 0` with degraded-looking stderr (should NOT degrade — success takes precedence over noisy stderr); `stderr` undefined vs empty string.\n- `findLocalPackageDir`: version present but pre-1.0 (`0.x.y`) tilde matching; symlinked node_modules; two ancestors both having a stale version (should pick a satisfying one over the first found).\n- `parseDurationMs`: unit `m` is ambiguous (30-day \"month\" vs minutes) — worth a regression test pinning the intended semantics; malformed specs (`7`, `d7`, `-7d`).\n- `parseSecurityBenchMarkdown`: `overallMatch` is `null` (upstream changed emoji/wording) — does the caller handle a `null` gate list gracefully?\n- `buildUpstreamArgs('attack')`: missing `--count` (`ARGS.count === null`) should omit the flag, not emit `--count null`.\n\n",
|
|
21
|
+
"level": 2
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "3. Missing error handling tests",
|
|
25
|
+
"content": "\n- `ensureCachedInstall`: `mkdirSync` throw path (`cache-dir-create-failed`) and `npm install` non-zero exit (`install-failed`) — both branches untested.\n- `importOptionalLibrary`: the `recoverable` regex gate — an *unrelated* `TypeError` thrown from within an installed module must NOT be swallowed (re-thrown), only `MODULE_NOT_FOUND`-shaped errors should fall through to cached-install. No test proves non-recoverable errors propagate.\n- `evolve.mjs safetyChecks()`: none of `--generations`/`--children`/`--concurrency` out-of-range, invalid `--sandbox`/`--selection`/`--mutator`, or nonexistent `--repo` path are unit-tested (only reachable via full CLI spawn today).\n- `security-bench.mjs safetyChecks()`: `--population`/`--cycles` bounds untested directly.\n- `redblue.mjs`: `report` without `--in`, `attack` with invalid family — both call `process.exit(2)`, untested.\n\n",
|
|
26
|
+
"level": 2
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "4. Integration test gaps",
|
|
30
|
+
"content": "\n- **`evolve` / `gepa` / `learn` / `redblue` / `bench` / `security-bench` are absent from `test-graceful-degradation.mjs`.** The file's own header explicitly scopes to 8 skills and says nothing about these six. Given ADR-150's architectural constraint (\"ruflo remains operational if every MetaHarness package is removed\") is supposed to apply project-wide, this is a real gate gap, not just a nice-to-have — a regression in `_darwin.mjs`'s or `_redblue.mjs`'s degraded path for these specific subcommands would ship silently.\n- `drift-from-history.mjs` composing 3 primitives (`audit-list` → `similarity`/`audit-trend`) has no dedicated round-trip test analogous to `test-pipeline-roundtrip.mjs` — only structural grep in `smoke.sh`.\n- No test exercises `RUFLO_METAHARNESS_CACHE_BASE` / `RUFLO_METAHARNESS_SKIP_LOCAL` env seams that `_invoke.mjs` explicitly documents as \"used by test-graceful-degradation.mjs\" — grep confirms they aren't actually referenced there.\n\n",
|
|
31
|
+
"level": 2
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "Test skeletons",
|
|
35
|
+
"content": "\n**New file: `scripts/test-invoke.mjs`** (mirrors `test-similarity.mjs`'s style; requires exporting nothing new — everything below is already exported)\n\n```js\n#!/usr/bin/env node\n// test-invoke.mjs — unit tests for the shared _invoke.mjs plumbing\n// (classifyDegraded, injectJson, parseTrailingJson, satisfiesTildeRange,\n// findLocalPackageDir, cacheBaseDir). No network, no subprocess spawns.\n\nimport { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nimport {\n classifyDegraded, injectJson, parseTrailingJson,\n satisfiesTildeRange, findLocalPackageDir, cacheBaseDir,\n} from './_invoke.mjs';\n\nlet passed = 0, failed = 0;\nconst failures = [];\nfunction assert(cond, label) {\n if (cond) { console.log(` ✓ ${label}`); passed++; }\n else { console.log(` ✗ ${label}`); failures.push(label); failed++; }\n}\n\nconsole.log('# classifyDegraded');\nassert(classifyDegraded('', null, 'x').degraded === true, 'null exitCode -> timeout');\nassert(classifyDegraded('', null, 'x').reason === 'x-timeout', 'timeout reason prefixed');\nassert(classifyDegraded('npm ERR! 404', 1, 'x').reason === 'x-not-available', 'npm ERR matches DEGRADED_RX');\nassert(classifyDegraded('some noisy stderr', 0, 'x').degraded === false, 'exit 0 with noisy stderr is NOT degraded');\nassert(classifyDegraded(undefined, 1, 'x').degraded === false, 'undefined stderr does not throw / does not false-positive');\n\nconsole.log('\\n# injectJson');\nassert(injectJson(['run'], true).includes('--json'), 'appends --json when wanted');\nassert(!injectJson(['run'], false).includes('--json'), 'omits --json when not wanted');\nassert(injectJson(['run', '--json'], true).filter((a) => a === '--json').length === 1, 'no duplicate --json');\n\nconsole.log('\\n# parseTrailingJson');\nassert(parseTrailingJson('progress\\n{\"a\":1}\\n{\"b\":2}').b === 2, 'picks LAST block, not first');\nassert(parseTrailingJson('{\"a\": {\"nested\": {\"deep\": 1}}}').a.nested.deep === 1, 'greedy fallback recovers nested braces');\nassert(parseTrailingJson('no json here') === null, 'no braces -> null, never throws');\nassert(parseTrailingJson('{\"a\":1} garbage {not json}').a === 1, 'unparseable trailing block falls back to earlier valid one');\n\nconsole.log('\\n# satisfiesTildeRange');\nassert(satisfiesTildeRange('0.8.3', '~0.8.0') === true, 'patch >= pin satisfies');\nassert(satisfiesTildeRange('0.8.0', '~0.8.0') === true, 'exact patch satisfies');\nassert(satisfiesTildeRange('0.9.0', '~0.8.0') === false, 'minor bump does not satisfy tilde range');\nassert(satisfiesTildeRange('not-a-version', '~0.8.0') === false, 'garbage version -> false, not throw');\nassert(satisfiesTildeRange('0.8.3', 'garbage-pin') === false, 'garbage pin -> false, not throw');\n\nconsole.log('\\n# cacheBaseDir + RUFLO_METAHARNESS_CACHE_BASE seam');\n{\n const prev = process.env.RUFLO_METAHARNESS_CACHE_BASE;\n delete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n assert(cacheBaseDir().endsWith('.ruflo'), 'defaults to ~/.ruflo');\n process.env.RUFLO_METAHARNESS_CACHE_BASE = '/tmp/custom-cache';\n assert(cacheBaseDir() === '/tmp/custom-cache', 'env override respected');\n if (prev === undefined) delete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n else process.env.RUFLO_METAHARNESS_CACHE_BASE = prev;\n}\n\nconsole.log('\\n# findLocalPackageDir + RUFLO_METAHARNESS_SKIP_LOCAL seam');\n{\n const fixture = mkdtempSync(join(tmpdir(), 'invoke-fixture-'));\n const pkgDir = join(fixture, 'node_modules', '@metaharness', 'darwin');\n mkdirSync(pkgDir, { recursive: true });\n writeFileSync(join(pkgDir, 'package.json'), JSON.stringify({ version: '0.8.3' }));\n const prevCwd = process.cwd();\n process.chdir(fixture);\n try {\n assert(findLocalPackageDir('@metaharness/darwin', '~0.8.0') === pkgDir, 'finds satisfying local install by walking cwd');\n assert(findLocalPackageDir('@metaharness/darwin', '~0.9.0') === null, 'rejects local install that does not satisfy pin');\n process.env.RUFLO_METAHARNESS_SKIP_LOCAL = '1';\n assert(findLocalPackageDir('@metaharness/darwin', '~0.8.0') === null, 'SKIP_LOCAL seam disables resolution entirely');\n } finally {\n delete process.env.RUFLO_METAHARNESS_SKIP_LOCAL;\n process.chdir(prevCwd);\n rmSync(fixture, { recursive: true, force: true });\n }\n}\n\nconsole.log(`\\n${passed} passed, ${failed} failed`);\nif (failed > 0) { console.log('Failures:', failures.join(', ')); process.exit(1); }\n```\n\n**Extend `test-graceful-degradation.mjs`** — add the 6 missing skills to close the ADR-150 gate gap:\n\n```js\n// Add alongside the existing 8-skill list:\nconst MISSING_SKILLS = ['evolve', 'gepa', 'learn', 'redblue', 'bench', 'security-bench'];\nfor (const skill of MISSING_SKILLS) {\n // same drill as the existing loop: spawn with npm_config_registry pointed\n // at an unresolvable host, assert exitCode === 0 and stdout contains\n // \"degraded\": true — these six currently escape the drill entirely.\n}\n```\n\n**New file: `scripts/test-cli-parsers.mjs`** (requires adding `export` to the named functions in `security-bench.mjs`, `redblue.mjs`, `audit-list.mjs`, `evolve.mjs`, `gepa.mjs` — currently module-private, which is *why* they're untestable in isolation today):\n\n```js\n#!/usr/bin/env node\n// test-cli-parsers.mjs — unit tests for pure parsing/validation logic\n// embedded in CLI scripts. Requires exporting: parseSecurityBenchMarkdown\n// (security-bench.mjs), buildUpstreamArgs (redblue.mjs), parseDurationMs\n// (audit-list.mjs), looksLikeGepaTranscript/extractGepaTranscripts/\n// summarizeTraces (evolve.mjs).\n\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nimport { parseDurationMs } from './audit-list.mjs';\nimport { looksLikeGepaTranscript, extractGepaTranscripts, summarizeTraces } from './evolve.mjs';\n\nlet passed = 0, failed = 0;\nconst failures = [];\nfunction assert(cond, label) {\n if (cond) { console.log(` ✓ ${label}`); passed++; }\n else { console.log(` ✗ ${label}`); failures.push(label); failed++; }\n}\n\nconsole.log('# parseSecurityBenchMarkdown');\nconst md = `\n**Overall: ✅ PASS**\n- ✅ **TPR improvement ≥ 25% vs fixed harness** — +150%\n- ❌ **FPR ≤ 5%** — 8%\n| B1 baseline | 0.42 | 0.60 | 0.10 | 0.90 | 0.95 | 2 | $0.10 |\n`;\nconst parsed = parseSecurityBenchMarkdown(md);\nassert(parsed.overall.ok === true, 'overall PASS parsed');\nassert(parsed.gates.length === 2, 'both gate lines parsed');\nassert(parsed.gates[1].ok === false, 'FAIL gate icon parsed correctly');\nassert(parsed.baselines[0].fitness === 0.42, 'baseline table row parsed');\nassert(parseSecurityBenchMarkdown('no matches here').overall === null, 'missing Overall line -> null, not throw');\nassert(parseSecurityBenchMarkdown('no matches here').gates.length === 0, 'no gate lines -> empty array, not throw');\n\nconsole.log('\\n# parseDurationMs');\nassert(parseDurationMs('7d') === 7 * 86400_000, '7d parses');\nassert(parseDurationMs('24h') === 24 * 3600_000, '24h parses');\nassert(parseDurationMs('1w') === 7 * 86400_000, '1w parses');\nassert(parseDurationMs('garbage') === null, 'malformed spec -> null');\nassert(parseDurationMs('-7d') === null, 'negative spec -> null (regex requires digits only)');\nassert(parseDurationMs('7') === null, 'missing unit -> null');\n\nconsole.log('\\n# evolve.mjs gepa-transcript helpers');\nassert(looksLikeGepaTranscript([{ actionRaw: 'x' }]) === true, 'actionRaw-shaped entries recognized');\nassert(looksLikeGepaTranscript([{ obs: 'y' }]) === true, 'obs-shaped entries recognized');\nassert(looksLikeGepaTranscript([]) === false, 'empty array is not a transcript');\nassert(looksLikeGepaTranscript([{ foo: 1 }]) === false, 'unrelated shape rejected');\nassert(looksLikeGepaTranscript(null) === false, 'null input -> false, not throw');\n\nconst record = { transcript: [{ actionRaw: 'a' }], traces: [{ taskId: 't1', exitCode: 1, timedOut: true, blockedActions: ['x'] }] };\nassert(extractGepaTranscripts(record).length === 1, 'top-level transcript extracted');\nconst summary = summarizeTraces(record);\nassert(summary.tasks === 1 && summary.failed === 1 && summary.timedOut === 1 && summary.blockedActions === 1, 'summarizeTraces tallies correctly');\nassert(summarizeTraces({}).tasks === 0, 'missing traces array -> zeroed summary, not throw');\n\nconsole.log(`\\n${passed} passed, ${failed} failed`);\nif (failed > 0) { console.log('Failures:', failures.join(', ')); process.exit(1); }\n```\n\n**New file: `scripts/test-router-parallel-math.mjs`** (requires exporting `median`/`pctile`/`mean` from `router-parallel-analyze.mjs`):\n\n```js\n#!/usr/bin/env node\nimport { median, pctile, mean } from './router-parallel-analyze.mjs';\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { if (cond) passed++; else { console.log(`✗ ${label}`); failed++; } }\n\nassert(median([1, 2, 3]) === 2, 'odd-length median');\nassert(median([1, 2, 3, 4]) === 2.5, 'even-length median averages middle two');\nassert(median([]) === undefined || Number.isNaN(median([])), 'empty array does not throw');\nassert(pctile([1, 2, 3, 4, 5], 0.5) === median([1, 2, 3, 4, 5]), 'p50 matches median');\nassert(mean([2, 4, 6]) === 4, 'mean computes correctly');\nassert(Number.isNaN(mean([])) || mean([]) === 0, 'empty array mean defined (pin the actual convention)');\n\nconsole.log(`${passed} passed, ${failed} failed`);\nif (failed > 0) process.exit(1);\n```\n\n",
|
|
36
|
+
"level": 2
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"title": "Priority",
|
|
40
|
+
"content": "1. `scripts/test-invoke.mjs` — closes the biggest real gap (shared plumbing, zero direct tests, already caused one documented regression).\n2. Extend `test-graceful-degradation.mjs` to the 6 missing skills — closes an actual ADR-150 CI gate hole.\n3. `test-cli-parsers.mjs` + `test-router-parallel-math.mjs` — cheap, pure-function tests once the target functions are exported (small, low-risk diffs to each source file).",
|
|
41
|
+
"level": 2
|
|
42
|
+
}
|
|
43
|
+
],
|
|
44
|
+
"codeBlocks": [
|
|
45
|
+
{
|
|
46
|
+
"language": "js",
|
|
47
|
+
"code": "#!/usr/bin/env node\n// test-invoke.mjs — unit tests for the shared _invoke.mjs plumbing\n// (classifyDegraded, injectJson, parseTrailingJson, satisfiesTildeRange,\n// findLocalPackageDir, cacheBaseDir). No network, no subprocess spawns.\n\nimport { mkdtempSync, mkdirSync, writeFileSync, rmSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nimport {\n classifyDegraded, injectJson, parseTrailingJson,\n satisfiesTildeRange, findLocalPackageDir, cacheBaseDir,\n} from './_invoke.mjs';\n\nlet passed = 0, failed = 0;\nconst failures = [];\nfunction assert(cond, label) {\n if (cond) { console.log(` ✓ ${label}`); passed++; }\n else { console.log(` ✗ ${label}`); failures.push(label); failed++; }\n}\n\nconsole.log('# classifyDegraded');\nassert(classifyDegraded('', null, 'x').degraded === true, 'null exitCode -> timeout');\nassert(classifyDegraded('', null, 'x').reason === 'x-timeout', 'timeout reason prefixed');\nassert(classifyDegraded('npm ERR! 404', 1, 'x').reason === 'x-not-available', 'npm ERR matches DEGRADED_RX');\nassert(classifyDegraded('some noisy stderr', 0, 'x').degraded === false, 'exit 0 with noisy stderr is NOT degraded');\nassert(classifyDegraded(undefined, 1, 'x').degraded === false, 'undefined stderr does not throw / does not false-positive');\n\nconsole.log('\\n# injectJson');\nassert(injectJson(['run'], true).includes('--json'), 'appends --json when wanted');\nassert(!injectJson(['run'], false).includes('--json'), 'omits --json when not wanted');\nassert(injectJson(['run', '--json'], true).filter((a) => a === '--json').length === 1, 'no duplicate --json');\n\nconsole.log('\\n# parseTrailingJson');\nassert(parseTrailingJson('progress\\n{\"a\":1}\\n{\"b\":2}').b === 2, 'picks LAST block, not first');\nassert(parseTrailingJson('{\"a\": {\"nested\": {\"deep\": 1}}}').a.nested.deep === 1, 'greedy fallback recovers nested braces');\nassert(parseTrailingJson('no json here') === null, 'no braces -> null, never throws');\nassert(parseTrailingJson('{\"a\":1} garbage {not json}').a === 1, 'unparseable trailing block falls back to earlier valid one');\n\nconsole.log('\\n# satisfiesTildeRange');\nassert(satisfiesTildeRange('0.8.3', '~0.8.0') === true, 'patch >= pin satisfies');\nassert(satisfiesTildeRange('0.8.0', '~0.8.0') === true, 'exact patch satisfies');\nassert(satisfiesTildeRange('0.9.0', '~0.8.0') === false, 'minor bump does not satisfy tilde range');\nassert(satisfiesTildeRange('not-a-version', '~0.8.0') === false, 'garbage version -> false, not throw');\nassert(satisfiesTildeRange('0.8.3', 'garbage-pin') === false, 'garbage pin -> false, not throw');\n\nconsole.log('\\n# cacheBaseDir + RUFLO_METAHARNESS_CACHE_BASE seam');\n{\n const prev = process.env.RUFLO_METAHARNESS_CACHE_BASE;\n delete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n assert(cacheBaseDir().endsWith('.ruflo'), 'defaults to ~/.ruflo');\n process.env.RUFLO_METAHARNESS_CACHE_BASE = '/tmp/custom-cache';\n assert(cacheBaseDir() === '/tmp/custom-cache', 'env override respected');\n if (prev === undefined) delete process.env.RUFLO_METAHARNESS_CACHE_BASE;\n else process.env.RUFLO_METAHARNESS_CACHE_BASE = prev;\n}\n\nconsole.log('\\n# findLocalPackageDir + RUFLO_METAHARNESS_SKIP_LOCAL seam');\n{\n const fixture = mkdtempSync(join(tmpdir(), 'invoke-fixture-'));\n const pkgDir = join(fixture, 'node_modules', '@metaharness', 'darwin');\n mkdirSync(pkgDir, { recursive: true });\n writeFileSync(join(pkgDir, 'package.json'), JSON.stringify({ version: '0.8.3' }));\n const prevCwd = process.cwd();\n process.chdir(fixture);\n try {\n assert(findLocalPackageDir('@metaharness/darwin', '~0.8.0') === pkgDir, 'finds satisfying local install by walking cwd');\n assert(findLocalPackageDir('@metaharness/darwin', '~0.9.0') === null, 'rejects local install that does not satisfy pin');\n process.env.RUFLO_METAHARNESS_SKIP_LOCAL = '1';\n assert(findLocalPackageDir('@metaharness/darwin', '~0.8.0') === null, 'SKIP_LOCAL seam disables resolution entirely');\n } finally {\n delete process.env.RUFLO_METAHARNESS_SKIP_LOCAL;\n process.chdir(prevCwd);\n rmSync(fixture, { recursive: true, force: true });\n }\n}\n\nconsole.log(`\\n${passed} passed, ${failed} failed`);\nif (failed > 0) { console.log('Failures:', failures.join(', ')); process.exit(1); }"
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"language": "js",
|
|
51
|
+
"code": "// Add alongside the existing 8-skill list:\nconst MISSING_SKILLS = ['evolve', 'gepa', 'learn', 'redblue', 'bench', 'security-bench'];\nfor (const skill of MISSING_SKILLS) {\n // same drill as the existing loop: spawn with npm_config_registry pointed\n // at an unresolvable host, assert exitCode === 0 and stdout contains\n // \"degraded\": true — these six currently escape the drill entirely.\n}"
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"language": "js",
|
|
55
|
+
"code": "#!/usr/bin/env node\n// test-cli-parsers.mjs — unit tests for pure parsing/validation logic\n// embedded in CLI scripts. Requires exporting: parseSecurityBenchMarkdown\n// (security-bench.mjs), buildUpstreamArgs (redblue.mjs), parseDurationMs\n// (audit-list.mjs), looksLikeGepaTranscript/extractGepaTranscripts/\n// summarizeTraces (evolve.mjs).\n\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nimport { parseDurationMs } from './audit-list.mjs';\nimport { looksLikeGepaTranscript, extractGepaTranscripts, summarizeTraces } from './evolve.mjs';\n\nlet passed = 0, failed = 0;\nconst failures = [];\nfunction assert(cond, label) {\n if (cond) { console.log(` ✓ ${label}`); passed++; }\n else { console.log(` ✗ ${label}`); failures.push(label); failed++; }\n}\n\nconsole.log('# parseSecurityBenchMarkdown');\nconst md = `\n**Overall: ✅ PASS**\n- ✅ **TPR improvement ≥ 25% vs fixed harness** — +150%\n- ❌ **FPR ≤ 5%** — 8%\n| B1 baseline | 0.42 | 0.60 | 0.10 | 0.90 | 0.95 | 2 | $0.10 |\n`;\nconst parsed = parseSecurityBenchMarkdown(md);\nassert(parsed.overall.ok === true, 'overall PASS parsed');\nassert(parsed.gates.length === 2, 'both gate lines parsed');\nassert(parsed.gates[1].ok === false, 'FAIL gate icon parsed correctly');\nassert(parsed.baselines[0].fitness === 0.42, 'baseline table row parsed');\nassert(parseSecurityBenchMarkdown('no matches here').overall === null, 'missing Overall line -> null, not throw');\nassert(parseSecurityBenchMarkdown('no matches here').gates.length === 0, 'no gate lines -> empty array, not throw');\n\nconsole.log('\\n# parseDurationMs');\nassert(parseDurationMs('7d') === 7 * 86400_000, '7d parses');\nassert(parseDurationMs('24h') === 24 * 3600_000, '24h parses');\nassert(parseDurationMs('1w') === 7 * 86400_000, '1w parses');\nassert(parseDurationMs('garbage') === null, 'malformed spec -> null');\nassert(parseDurationMs('-7d') === null, 'negative spec -> null (regex requires digits only)');\nassert(parseDurationMs('7') === null, 'missing unit -> null');\n\nconsole.log('\\n# evolve.mjs gepa-transcript helpers');\nassert(looksLikeGepaTranscript([{ actionRaw: 'x' }]) === true, 'actionRaw-shaped entries recognized');\nassert(looksLikeGepaTranscript([{ obs: 'y' }]) === true, 'obs-shaped entries recognized');\nassert(looksLikeGepaTranscript([]) === false, 'empty array is not a transcript');\nassert(looksLikeGepaTranscript([{ foo: 1 }]) === false, 'unrelated shape rejected');\nassert(looksLikeGepaTranscript(null) === false, 'null input -> false, not throw');\n\nconst record = { transcript: [{ actionRaw: 'a' }], traces: [{ taskId: 't1', exitCode: 1, timedOut: true, blockedActions: ['x'] }] };\nassert(extractGepaTranscripts(record).length === 1, 'top-level transcript extracted');\nconst summary = summarizeTraces(record);\nassert(summary.tasks === 1 && summary.failed === 1 && summary.timedOut === 1 && summary.blockedActions === 1, 'summarizeTraces tallies correctly');\nassert(summarizeTraces({}).tasks === 0, 'missing traces array -> zeroed summary, not throw');\n\nconsole.log(`\\n${passed} passed, ${failed} failed`);\nif (failed > 0) { console.log('Failures:', failures.join(', ')); process.exit(1); }"
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"language": "js",
|
|
59
|
+
"code": "#!/usr/bin/env node\nimport { median, pctile, mean } from './router-parallel-analyze.mjs';\n\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { if (cond) passed++; else { console.log(`✗ ${label}`); failed++; } }\n\nassert(median([1, 2, 3]) === 2, 'odd-length median');\nassert(median([1, 2, 3, 4]) === 2.5, 'even-length median averages middle two');\nassert(median([]) === undefined || Number.isNaN(median([])), 'empty array does not throw');\nassert(pctile([1, 2, 3, 4, 5], 0.5) === median([1, 2, 3, 4, 5]), 'p50 matches median');\nassert(mean([2, 4, 6]) === 4, 'mean computes correctly');\nassert(Number.isNaN(mean([])) || mean([]) === 0, 'empty array mean defined (pin the actual convention)');\n\nconsole.log(`${passed} passed, ${failed} failed`);\nif (failed > 0) process.exit(1);"
|
|
60
|
+
}
|
|
61
|
+
]
|
|
62
|
+
},
|
|
63
|
+
"durationMs": 181882,
|
|
64
|
+
"model": "sonnet",
|
|
65
|
+
"sandboxMode": "permissive",
|
|
66
|
+
"workerType": "testgaps",
|
|
67
|
+
"timestamp": "2026-07-09T13:55:16.742Z",
|
|
68
|
+
"executionId": "testgaps_1783605134860_9jssz9"
|
|
69
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T14:15:16.745Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
[2026-07-09T14:20:38.325Z] RESULT
|
|
2
|
+
============================================================
|
|
3
|
+
{
|
|
4
|
+
"success": true,
|
|
5
|
+
"output": "## Test Coverage Analysis — `plugins/ruflo-metaharness`\n\n**Infra note:** there's no `package.json`/jest/vitest here — the \"test suite\" is 6 hand-rolled `test-*.mjs` scripts using a local `assert()` + pass/fail counter pattern, gated by `scripts/smoke.sh` (grep + exit-code checks). Skeletons below follow that existing convention rather than introducing a new framework.\n\n### 1. `threat-model.mjs` — stale local `SEVERITY_RANK` (highest priority, real bug)\n\nUnlike `mcp-scan.mjs`/`audit-trend.mjs`, this file keeps its own `{clean:0,low:1,medium:2,high:3}` instead of importing the shared table from `_harness.mjs`. `info`/`warn`/`error`/`critical` findings silently make `SEVERITY_RANK[worst]` return `undefined`, so `undefined >= threshold` is `false` — an `--alert-on-worst high` gate never fires for a `critical` finding.\n\n```js\n// scripts/test-threat-model-severity.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\n// Fixture: force a mocked upstream payload with worst='critical'\nconst out = execFileSync('node', ['scripts/threat-model.mjs', 'FIXTURE_REPO', '--alert-on-worst', 'high'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.exitCode === 1, 'critical finding should trip --alert-on-worst high, not silently pass');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nFix + regression test: migrate `threat-model.mjs` to `import { SEVERITY_RANK } from './_harness.mjs'`.\n\n### 2. `evolve.mjs --diagnose` subsystem — zero coverage\n\n`buildDiagnosis`/`extractGepaTranscripts`/`summarizeTraces`/`looksLikeGepaTranscript` (~130 lines) are never exercised by any of the 6 test scripts.\n\n```js\n// scripts/test-evolve-diagnose.mjs\nimport { execFileSync } from 'node:child_process';\nimport { mkdtempSync, writeFileSync, mkdirSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst runsDir = mkdtempSync(join(tmpdir(), 'evolve-diag-'));\nmkdirSync(join(runsDir, '.metaharness', 'runs'), { recursive: true });\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{\"id\":\"v1\"}');\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'v1.json'), JSON.stringify({ id: 'v1', transcript: [] }));\n\nconst out = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.diagnosis?.available === true || payload.diagnosis?.available === false, 'diagnosis object present');\n\n// corrupt winner.json -> must NOT throw, must degrade gracefully\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{not json');\nconst out2 = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(JSON.parse(out2).diagnosis, 'corrupt winner.json does not crash --diagnose');\n\n// >100 run records -> should truncate silently, not throw\nfor (let i = 0; i < 120; i++) writeFileSync(join(runsDir, '.metaharness', 'runs', `v${i}.json`), '{}');\nexecFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(true, '>100 run records does not crash (slice(0,100) guard)');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n### 3. `router-parallel-analyze.mjs` — crashes on one malformed JSONL line\n\n`rows.map(l => JSON.parse(l))` has no per-line try/catch; a single corrupt line takes down the whole analysis (exit 2) instead of being skipped/reported.\n\n```js\n// scripts/test-router-parallel-malformed.mjs\nimport { execFileSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-'));\nconst file = join(dir, 'router-parallel.jsonl');\nconst goodRow = JSON.stringify({ bandit: { quality: 0.8, usd: 0.01, latencyP95: 100 }, neural: { quality: 0.85, usd: 0.009, latencyP95: 95 } });\nwriteFileSync(file, Array(35).fill(goodRow).join('\\n') + '\\nNOT VALID JSON\\n');\n\nlet exitCode = 0, out = '';\ntry {\n out = execFileSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file], { encoding: 'utf8' });\n} catch (e) { exitCode = e.status; out = e.stdout; }\n\n// current (buggy) behavior: exitCode === 2, whole analysis aborts\n// desired: bad line skipped, analysis still runs on the 35 good rows\nassert(exitCode !== 2, 'one malformed JSONL line should not abort the whole analysis');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nThis one is a genuine bug fix candidate, not just a test gap — recommend wrapping the per-line `JSON.parse` in try/catch and counting skipped rows in the output.\n\n### 4. `gepa.mjs` — `validate`/`render`/`analyze` ops untested\n\nOnly `op:genome` is exercised (indirectly, via the MCP tool layer). `--alert-on-invalid` exit path is never hit.\n\n```js\n// scripts/test-gepa-ops.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nfor (const op of ['validate', 'render']) {\n const out = execFileSync('node', ['scripts/gepa.mjs', '--op', op], { encoding: 'utf8' });\n const payload = JSON.parse(out);\n assert(payload.op === op || payload.degraded, `op=${op} returns a shaped payload or degrades cleanly`);\n}\n\n// analyze requires --transcript to be a JSON array; non-array should exit 2\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'analyze', '--transcript', '{\"not\":\"array\"}'], { encoding: 'utf8' });\n assert(false, 'non-array --transcript should exit non-zero');\n} catch (e) { assert(e.status === 2, 'non-array --transcript exits 2'); }\n\n// --alert-on-invalid path\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'validate', '--path', 'FIXTURE_INVALID_GENOME.json', '--alert-on-invalid'], { encoding: 'utf8' });\n assert(false, 'invalid genome + --alert-on-invalid should exit non-zero');\n} catch (e) { assert(e.status === 1, 'alert-on-invalid trips on invalid genome'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n### 5. `security-bench.mjs` — `parseSecurityBenchMarkdown` has no isolated test\n\nFragile to upstream markdown drift (emoji/table format). Not exported, so needs a small refactor to be unit-testable — recommend exporting it.\n\n```js\n// scripts/test-security-bench-parse.mjs\n// Requires: export function parseSecurityBenchMarkdown(md) { ... } in security-bench.mjs\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst wellFormed = `## Overall: ✅ PASS\\n| Gate | Status |\\n|---|---|\\n| injection | ✅ |\\n`;\nconst r1 = parseSecurityBenchMarkdown(wellFormed);\nassert(r1.gates.length > 0, 'well-formed markdown parses gates');\n\nconst malformed = `Some unexpected upstream format change with no matching tables`;\nconst r2 = parseSecurityBenchMarkdown(malformed);\nassert(r2.gates.length === 0, 'malformed markdown returns empty gates, not throw');\nassert(r2.warning || r2.parseFailed, 'malformed markdown surfaces a warning instead of silently returning empty');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n### 6. Bench scripts (`bench.mjs`, `bench-*.mjs`) — no assertion-based coverage at all\n\nOnly reachable via `smoke.sh`. Minimum viable regression test:\n\n```js\n// scripts/test-bench-args.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\n// bench-parse-mcp-scan.mjs with --iters 1 exercises the p50/p99 boundary\nconst out = execFileSync('node', ['scripts/bench-parse-mcp-scan.mjs', '--iters', '1'], { encoding: 'utf8' });\nassert(!/NaN|undefined/.test(out), '--iters 1 does not produce NaN/undefined percentiles');\n\n// bench-recordpair-overhead.mjs arg validation\ntry {\n execFileSync('node', ['scripts/bench-recordpair-overhead.mjs', '--iters', '10'], { encoding: 'utf8' });\n assert(false, '--iters below 1000 should be rejected');\n} catch (e) { assert(e.status === 2, '--iters < 1000 exits 2'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n### 7. Missing error-handling tests (quick wins)\n\n| File | Untested error path | Test scenario |\n|---|---|---|\n| `_harness.mjs` | `rankSeverity(' high ')` → 0 (no trim) | `assert(rankSeverity(' high ') === rankSeverity('high'))` should fail today — flags the footgun |\n| `audit-list.mjs` | `memList()`/`memRetrieve()` swallow non-zero exit → indistinguishable from empty namespace | mock a failing `@claude-flow/cli` call, assert output distinguishes \"error\" from \"empty\" |\n| `audit-trend.mjs` | `fingerprint()` truncates message at 80 chars → false \"unchanged\" | two findings identical in first 80 chars, differing after — assert they're NOT deduped as unchanged |\n| `similarity.mjs` (CLI) | `--alert-below abc` → `Number('abc')` = `NaN`, comparison always false, alert never fires | `execFileSync(['similarity.mjs', ..., '--alert-below', 'abc'])` should exit non-zero (invalid arg), not silently succeed |\n| `mcp-scan.mjs` | `findings.map` throws if an array element is a raw string, not object | fixture upstream output with `findings: [\"a string\", {...}]`, assert graceful handling instead of throw |\n| `redblue.mjs` | passthrough arg parser misparses a negative-number value as a bare flag | `--some-flag -5` should preserve `-5` as the value |\n| `learn.mjs` | `checkout-required` detection regex breaks silently if upstream wording changes | fixture stderr with slightly different wording, assert it still surfaces `checkout-required` OR at minimum doesn't silently misreport `status:'failed'` |\n\n### 8. Integration test gaps\n\n- **`oia-audit.mjs` partial degradation**: only \"all 5 subprocesses degraded\" is tested (`test-graceful-degradation.mjs`). No test covers 1-4 of 5 degrading while the composite still reports as healthy at the top level.\n- **`mint.mjs` upstream workaround regression**: the documented pre-0.1.13 `--target`-ignored bug workaround has no test proving it still works against the currently pinned metaharness version.\n- **`test-parallel-pipeline.mjs`'s TS-source check is a grep, not a runtime test** — `router-parallel-recorder.ts`'s `recordPair`/`recordPairOutcome` are never actually executed, just pattern-matched as text. Recommend a real runtime import-and-call test (would need to run against the compiled `dist/` output).\n- **`test-pipeline-roundtrip.mjs` has a fragile path assumption**: `REPO_ROOT = dirname(dirname(dirname(SCRIPTS_DIR)))` — relocating the plugin breaks this silently (wrong dir, not an error). Add an assertion that `REPO_ROOT` actually contains expected sentinel files (e.g. root `CLAUDE.md`) before proceeding, failing loudly instead of silently pointing at the wrong tree.\n- **Cross-repo coverage unconfirmed**: whether `v3/@claude-flow/cli`'s own test suite covers the MCP tool handlers that wrap these scripts couldn't be verified from within this plugin directory — worth checking separately if you want full-picture coverage numbers.\n\n### Priority order if picking a subset to implement\n1. Fix `threat-model.mjs` SEVERITY_RANK (real bug, not just a gap) + regression test\n2. Fix `router-parallel-analyze.mjs` malformed-JSONL crash + regression test\n3. `evolve.mjs --diagnose` coverage (largest untested surface, ~130 lines)\n4. `gepa.mjs` validate/render/analyze ops\n5. Everything else in the table, roughly in the order listed\n",
|
|
6
|
+
"parsedOutput": {
|
|
7
|
+
"sections": [
|
|
8
|
+
{
|
|
9
|
+
"title": "Test Coverage Analysis — `plugins/ruflo-metaharness`",
|
|
10
|
+
"content": "\n**Infra note:** there's no `package.json`/jest/vitest here — the \"test suite\" is 6 hand-rolled `test-*.mjs` scripts using a local `assert()` + pass/fail counter pattern, gated by `scripts/smoke.sh` (grep + exit-code checks). Skeletons below follow that existing convention rather than introducing a new framework.\n\n",
|
|
11
|
+
"level": 2
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"title": "1. `threat-model.mjs` — stale local `SEVERITY_RANK` (highest priority, real bug)",
|
|
15
|
+
"content": "\nUnlike `mcp-scan.mjs`/`audit-trend.mjs`, this file keeps its own `{clean:0,low:1,medium:2,high:3}` instead of importing the shared table from `_harness.mjs`. `info`/`warn`/`error`/`critical` findings silently make `SEVERITY_RANK[worst]` return `undefined`, so `undefined >= threshold` is `false` — an `--alert-on-worst high` gate never fires for a `critical` finding.\n\n```js\n// scripts/test-threat-model-severity.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\n// Fixture: force a mocked upstream payload with worst='critical'\nconst out = execFileSync('node', ['scripts/threat-model.mjs', 'FIXTURE_REPO', '--alert-on-worst', 'high'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.exitCode === 1, 'critical finding should trip --alert-on-worst high, not silently pass');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nFix + regression test: migrate `threat-model.mjs` to `import { SEVERITY_RANK } from './_harness.mjs'`.\n\n",
|
|
16
|
+
"level": 3
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"title": "2. `evolve.mjs --diagnose` subsystem — zero coverage",
|
|
20
|
+
"content": "\n`buildDiagnosis`/`extractGepaTranscripts`/`summarizeTraces`/`looksLikeGepaTranscript` (~130 lines) are never exercised by any of the 6 test scripts.\n\n```js\n// scripts/test-evolve-diagnose.mjs\nimport { execFileSync } from 'node:child_process';\nimport { mkdtempSync, writeFileSync, mkdirSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst runsDir = mkdtempSync(join(tmpdir(), 'evolve-diag-'));\nmkdirSync(join(runsDir, '.metaharness', 'runs'), { recursive: true });\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{\"id\":\"v1\"}');\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'v1.json'), JSON.stringify({ id: 'v1', transcript: [] }));\n\nconst out = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.diagnosis?.available === true || payload.diagnosis?.available === false, 'diagnosis object present');\n\n// corrupt winner.json -> must NOT throw, must degrade gracefully\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{not json');\nconst out2 = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(JSON.parse(out2).diagnosis, 'corrupt winner.json does not crash --diagnose');\n\n// >100 run records -> should truncate silently, not throw\nfor (let i = 0; i < 120; i++) writeFileSync(join(runsDir, '.metaharness', 'runs', `v${i}.json`), '{}');\nexecFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(true, '>100 run records does not crash (slice(0,100) guard)');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n",
|
|
21
|
+
"level": 3
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"title": "3. `router-parallel-analyze.mjs` — crashes on one malformed JSONL line",
|
|
25
|
+
"content": "\n`rows.map(l => JSON.parse(l))` has no per-line try/catch; a single corrupt line takes down the whole analysis (exit 2) instead of being skipped/reported.\n\n```js\n// scripts/test-router-parallel-malformed.mjs\nimport { execFileSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-'));\nconst file = join(dir, 'router-parallel.jsonl');\nconst goodRow = JSON.stringify({ bandit: { quality: 0.8, usd: 0.01, latencyP95: 100 }, neural: { quality: 0.85, usd: 0.009, latencyP95: 95 } });\nwriteFileSync(file, Array(35).fill(goodRow).join('\\n') + '\\nNOT VALID JSON\\n');\n\nlet exitCode = 0, out = '';\ntry {\n out = execFileSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file], { encoding: 'utf8' });\n} catch (e) { exitCode = e.status; out = e.stdout; }\n\n// current (buggy) behavior: exitCode === 2, whole analysis aborts\n// desired: bad line skipped, analysis still runs on the 35 good rows\nassert(exitCode !== 2, 'one malformed JSONL line should not abort the whole analysis');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\nThis one is a genuine bug fix candidate, not just a test gap — recommend wrapping the per-line `JSON.parse` in try/catch and counting skipped rows in the output.\n\n",
|
|
26
|
+
"level": 3
|
|
27
|
+
},
|
|
28
|
+
{
|
|
29
|
+
"title": "4. `gepa.mjs` — `validate`/`render`/`analyze` ops untested",
|
|
30
|
+
"content": "\nOnly `op:genome` is exercised (indirectly, via the MCP tool layer). `--alert-on-invalid` exit path is never hit.\n\n```js\n// scripts/test-gepa-ops.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nfor (const op of ['validate', 'render']) {\n const out = execFileSync('node', ['scripts/gepa.mjs', '--op', op], { encoding: 'utf8' });\n const payload = JSON.parse(out);\n assert(payload.op === op || payload.degraded, `op=${op} returns a shaped payload or degrades cleanly`);\n}\n\n// analyze requires --transcript to be a JSON array; non-array should exit 2\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'analyze', '--transcript', '{\"not\":\"array\"}'], { encoding: 'utf8' });\n assert(false, 'non-array --transcript should exit non-zero');\n} catch (e) { assert(e.status === 2, 'non-array --transcript exits 2'); }\n\n// --alert-on-invalid path\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'validate', '--path', 'FIXTURE_INVALID_GENOME.json', '--alert-on-invalid'], { encoding: 'utf8' });\n assert(false, 'invalid genome + --alert-on-invalid should exit non-zero');\n} catch (e) { assert(e.status === 1, 'alert-on-invalid trips on invalid genome'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n",
|
|
31
|
+
"level": 3
|
|
32
|
+
},
|
|
33
|
+
{
|
|
34
|
+
"title": "5. `security-bench.mjs` — `parseSecurityBenchMarkdown` has no isolated test",
|
|
35
|
+
"content": "\nFragile to upstream markdown drift (emoji/table format). Not exported, so needs a small refactor to be unit-testable — recommend exporting it.\n\n```js\n// scripts/test-security-bench-parse.mjs\n// Requires: export function parseSecurityBenchMarkdown(md) { ... } in security-bench.mjs\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst wellFormed = `## Overall: ✅ PASS\\n| Gate | Status |\\n|---|---|\\n| injection | ✅ |\\n`;\nconst r1 = parseSecurityBenchMarkdown(wellFormed);\nassert(r1.gates.length > 0, 'well-formed markdown parses gates');\n\nconst malformed = `Some unexpected upstream format change with no matching tables`;\nconst r2 = parseSecurityBenchMarkdown(malformed);\nassert(r2.gates.length === 0, 'malformed markdown returns empty gates, not throw');\nassert(r2.warning || r2.parseFailed, 'malformed markdown surfaces a warning instead of silently returning empty');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n",
|
|
36
|
+
"level": 3
|
|
37
|
+
},
|
|
38
|
+
{
|
|
39
|
+
"title": "6. Bench scripts (`bench.mjs`, `bench-*.mjs`) — no assertion-based coverage at all",
|
|
40
|
+
"content": "\nOnly reachable via `smoke.sh`. Minimum viable regression test:\n\n```js\n// scripts/test-bench-args.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\n// bench-parse-mcp-scan.mjs with --iters 1 exercises the p50/p99 boundary\nconst out = execFileSync('node', ['scripts/bench-parse-mcp-scan.mjs', '--iters', '1'], { encoding: 'utf8' });\nassert(!/NaN|undefined/.test(out), '--iters 1 does not produce NaN/undefined percentiles');\n\n// bench-recordpair-overhead.mjs arg validation\ntry {\n execFileSync('node', ['scripts/bench-recordpair-overhead.mjs', '--iters', '10'], { encoding: 'utf8' });\n assert(false, '--iters below 1000 should be rejected');\n} catch (e) { assert(e.status === 2, '--iters < 1000 exits 2'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);\n```\n\n",
|
|
41
|
+
"level": 3
|
|
42
|
+
},
|
|
43
|
+
{
|
|
44
|
+
"title": "7. Missing error-handling tests (quick wins)",
|
|
45
|
+
"content": "\n| File | Untested error path | Test scenario |\n|---|---|---|\n| `_harness.mjs` | `rankSeverity(' high ')` → 0 (no trim) | `assert(rankSeverity(' high ') === rankSeverity('high'))` should fail today — flags the footgun |\n| `audit-list.mjs` | `memList()`/`memRetrieve()` swallow non-zero exit → indistinguishable from empty namespace | mock a failing `@claude-flow/cli` call, assert output distinguishes \"error\" from \"empty\" |\n| `audit-trend.mjs` | `fingerprint()` truncates message at 80 chars → false \"unchanged\" | two findings identical in first 80 chars, differing after — assert they're NOT deduped as unchanged |\n| `similarity.mjs` (CLI) | `--alert-below abc` → `Number('abc')` = `NaN`, comparison always false, alert never fires | `execFileSync(['similarity.mjs', ..., '--alert-below', 'abc'])` should exit non-zero (invalid arg), not silently succeed |\n| `mcp-scan.mjs` | `findings.map` throws if an array element is a raw string, not object | fixture upstream output with `findings: [\"a string\", {...}]`, assert graceful handling instead of throw |\n| `redblue.mjs` | passthrough arg parser misparses a negative-number value as a bare flag | `--some-flag -5` should preserve `-5` as the value |\n| `learn.mjs` | `checkout-required` detection regex breaks silently if upstream wording changes | fixture stderr with slightly different wording, assert it still surfaces `checkout-required` OR at minimum doesn't silently misreport `status:'failed'` |\n\n",
|
|
46
|
+
"level": 3
|
|
47
|
+
},
|
|
48
|
+
{
|
|
49
|
+
"title": "8. Integration test gaps",
|
|
50
|
+
"content": "\n- **`oia-audit.mjs` partial degradation**: only \"all 5 subprocesses degraded\" is tested (`test-graceful-degradation.mjs`). No test covers 1-4 of 5 degrading while the composite still reports as healthy at the top level.\n- **`mint.mjs` upstream workaround regression**: the documented pre-0.1.13 `--target`-ignored bug workaround has no test proving it still works against the currently pinned metaharness version.\n- **`test-parallel-pipeline.mjs`'s TS-source check is a grep, not a runtime test** — `router-parallel-recorder.ts`'s `recordPair`/`recordPairOutcome` are never actually executed, just pattern-matched as text. Recommend a real runtime import-and-call test (would need to run against the compiled `dist/` output).\n- **`test-pipeline-roundtrip.mjs` has a fragile path assumption**: `REPO_ROOT = dirname(dirname(dirname(SCRIPTS_DIR)))` — relocating the plugin breaks this silently (wrong dir, not an error). Add an assertion that `REPO_ROOT` actually contains expected sentinel files (e.g. root `CLAUDE.md`) before proceeding, failing loudly instead of silently pointing at the wrong tree.\n- **Cross-repo coverage unconfirmed**: whether `v3/@claude-flow/cli`'s own test suite covers the MCP tool handlers that wrap these scripts couldn't be verified from within this plugin directory — worth checking separately if you want full-picture coverage numbers.\n\n",
|
|
51
|
+
"level": 3
|
|
52
|
+
},
|
|
53
|
+
{
|
|
54
|
+
"title": "Priority order if picking a subset to implement",
|
|
55
|
+
"content": "1. Fix `threat-model.mjs` SEVERITY_RANK (real bug, not just a gap) + regression test\n2. Fix `router-parallel-analyze.mjs` malformed-JSONL crash + regression test\n3. `evolve.mjs --diagnose` coverage (largest untested surface, ~130 lines)\n4. `gepa.mjs` validate/render/analyze ops\n5. Everything else in the table, roughly in the order listed",
|
|
56
|
+
"level": 3
|
|
57
|
+
}
|
|
58
|
+
],
|
|
59
|
+
"codeBlocks": [
|
|
60
|
+
{
|
|
61
|
+
"language": "js",
|
|
62
|
+
"code": "// scripts/test-threat-model-severity.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(cond, label) { cond ? passed++ : (failed++, console.log(`✗ ${label}`)); }\n\n// Fixture: force a mocked upstream payload with worst='critical'\nconst out = execFileSync('node', ['scripts/threat-model.mjs', 'FIXTURE_REPO', '--alert-on-worst', 'high'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.exitCode === 1, 'critical finding should trip --alert-on-worst high, not silently pass');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"language": "js",
|
|
66
|
+
"code": "// scripts/test-evolve-diagnose.mjs\nimport { execFileSync } from 'node:child_process';\nimport { mkdtempSync, writeFileSync, mkdirSync } from 'node:fs';\nimport { tmpdir } from 'node:os';\nimport { join } from 'node:path';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst runsDir = mkdtempSync(join(tmpdir(), 'evolve-diag-'));\nmkdirSync(join(runsDir, '.metaharness', 'runs'), { recursive: true });\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{\"id\":\"v1\"}');\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'v1.json'), JSON.stringify({ id: 'v1', transcript: [] }));\n\nconst out = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nconst payload = JSON.parse(out);\nassert(payload.diagnosis?.available === true || payload.diagnosis?.available === false, 'diagnosis object present');\n\n// corrupt winner.json -> must NOT throw, must degrade gracefully\nwriteFileSync(join(runsDir, '.metaharness', 'runs', 'winner.json'), '{not json');\nconst out2 = execFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(JSON.parse(out2).diagnosis, 'corrupt winner.json does not crash --diagnose');\n\n// >100 run records -> should truncate silently, not throw\nfor (let i = 0; i < 120; i++) writeFileSync(join(runsDir, '.metaharness', 'runs', `v${i}.json`), '{}');\nexecFileSync('node', ['scripts/evolve.mjs', runsDir, '--diagnose', '--sandbox', 'mock'], { encoding: 'utf8' });\nassert(true, '>100 run records does not crash (slice(0,100) guard)');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
67
|
+
},
|
|
68
|
+
{
|
|
69
|
+
"language": "js",
|
|
70
|
+
"code": "// scripts/test-router-parallel-malformed.mjs\nimport { execFileSync } from 'node:child_process';\nimport { writeFileSync, mkdtempSync } from 'node:fs';\nimport { join } from 'node:path';\nimport { tmpdir } from 'node:os';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst dir = mkdtempSync(join(tmpdir(), 'rpa-'));\nconst file = join(dir, 'router-parallel.jsonl');\nconst goodRow = JSON.stringify({ bandit: { quality: 0.8, usd: 0.01, latencyP95: 100 }, neural: { quality: 0.85, usd: 0.009, latencyP95: 95 } });\nwriteFileSync(file, Array(35).fill(goodRow).join('\\n') + '\\nNOT VALID JSON\\n');\n\nlet exitCode = 0, out = '';\ntry {\n out = execFileSync('node', ['scripts/router-parallel-analyze.mjs', '--input', file], { encoding: 'utf8' });\n} catch (e) { exitCode = e.status; out = e.stdout; }\n\n// current (buggy) behavior: exitCode === 2, whole analysis aborts\n// desired: bad line skipped, analysis still runs on the 35 good rows\nassert(exitCode !== 2, 'one malformed JSONL line should not abort the whole analysis');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
71
|
+
},
|
|
72
|
+
{
|
|
73
|
+
"language": "js",
|
|
74
|
+
"code": "// scripts/test-gepa-ops.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nfor (const op of ['validate', 'render']) {\n const out = execFileSync('node', ['scripts/gepa.mjs', '--op', op], { encoding: 'utf8' });\n const payload = JSON.parse(out);\n assert(payload.op === op || payload.degraded, `op=${op} returns a shaped payload or degrades cleanly`);\n}\n\n// analyze requires --transcript to be a JSON array; non-array should exit 2\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'analyze', '--transcript', '{\"not\":\"array\"}'], { encoding: 'utf8' });\n assert(false, 'non-array --transcript should exit non-zero');\n} catch (e) { assert(e.status === 2, 'non-array --transcript exits 2'); }\n\n// --alert-on-invalid path\ntry {\n execFileSync('node', ['scripts/gepa.mjs', '--op', 'validate', '--path', 'FIXTURE_INVALID_GENOME.json', '--alert-on-invalid'], { encoding: 'utf8' });\n assert(false, 'invalid genome + --alert-on-invalid should exit non-zero');\n} catch (e) { assert(e.status === 1, 'alert-on-invalid trips on invalid genome'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"language": "js",
|
|
78
|
+
"code": "// scripts/test-security-bench-parse.mjs\n// Requires: export function parseSecurityBenchMarkdown(md) { ... } in security-bench.mjs\nimport { parseSecurityBenchMarkdown } from './security-bench.mjs';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\nconst wellFormed = `## Overall: ✅ PASS\\n| Gate | Status |\\n|---|---|\\n| injection | ✅ |\\n`;\nconst r1 = parseSecurityBenchMarkdown(wellFormed);\nassert(r1.gates.length > 0, 'well-formed markdown parses gates');\n\nconst malformed = `Some unexpected upstream format change with no matching tables`;\nconst r2 = parseSecurityBenchMarkdown(malformed);\nassert(r2.gates.length === 0, 'malformed markdown returns empty gates, not throw');\nassert(r2.warning || r2.parseFailed, 'malformed markdown surfaces a warning instead of silently returning empty');\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"language": "js",
|
|
82
|
+
"code": "// scripts/test-bench-args.mjs\nimport { execFileSync } from 'node:child_process';\nlet passed = 0, failed = 0;\nfunction assert(c, l) { c ? passed++ : (failed++, console.log(`✗ ${l}`)); }\n\n// bench-parse-mcp-scan.mjs with --iters 1 exercises the p50/p99 boundary\nconst out = execFileSync('node', ['scripts/bench-parse-mcp-scan.mjs', '--iters', '1'], { encoding: 'utf8' });\nassert(!/NaN|undefined/.test(out), '--iters 1 does not produce NaN/undefined percentiles');\n\n// bench-recordpair-overhead.mjs arg validation\ntry {\n execFileSync('node', ['scripts/bench-recordpair-overhead.mjs', '--iters', '10'], { encoding: 'utf8' });\n assert(false, '--iters below 1000 should be rejected');\n} catch (e) { assert(e.status === 2, '--iters < 1000 exits 2'); }\n\nconsole.log(failed ? `${failed} failed` : `${passed} passed, 0 failed`);\nprocess.exit(failed ? 1 : 0);"
|
|
83
|
+
}
|
|
84
|
+
]
|
|
85
|
+
},
|
|
86
|
+
"durationMs": 321581,
|
|
87
|
+
"model": "sonnet",
|
|
88
|
+
"sandboxMode": "permissive",
|
|
89
|
+
"workerType": "testgaps",
|
|
90
|
+
"timestamp": "2026-07-09T14:20:38.324Z",
|
|
91
|
+
"executionId": "testgaps_1783606516743_zftbaa"
|
|
92
|
+
}
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
[2026-07-09T14:40:38.318Z] PROMPT
|
|
2
|
+
============================================================
|
|
3
|
+
Analyze test coverage and identify gaps:
|
|
4
|
+
- Find untested functions and classes
|
|
5
|
+
- Identify edge cases not covered
|
|
6
|
+
- Suggest new test scenarios
|
|
7
|
+
- Check for missing error handling tests
|
|
8
|
+
- Identify integration test gaps
|
|
9
|
+
|
|
10
|
+
For each gap, provide a test skeleton.
|
|
11
|
+
|
|
12
|
+
## Instructions
|
|
13
|
+
|
|
14
|
+
Analyze the codebase and provide your response following the format specified in the task.
|