@iowarp/clio-coder 0.3.1 → 0.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +90 -2
- package/CONTRIBUTING.md +23 -23
- package/README.md +284 -613
- package/dist/{acp-FPR54DGL.js → acp-BIYHVZIM.js} +43 -53
- package/dist/{agents-OGPIHPJH.js → agents-YT6SSRIT.js} +43 -26
- package/dist/assets/codewiki.json +1 -1
- package/dist/{auth-IC3K6NIZ.js → auth-5TWEIYDN.js} +20 -12
- package/dist/{chunk-PV4JUBVJ.js → chunk-2EHAIA3X.js} +40 -21
- package/dist/{chunk-IS3ONKU3.js → chunk-2IR2NMPA.js} +6 -4
- package/dist/chunk-2SFS6XQE.js +122 -0
- package/dist/chunk-2VTFPG5O.js +48 -0
- package/dist/{chunk-MAR7Y6HW.js → chunk-3ZXDFGR5.js} +23 -16
- package/dist/chunk-4BJ5BYCE.js +61 -0
- package/dist/{chunk-474KN5II.js → chunk-4BPJXDWC.js} +111 -181
- package/dist/chunk-4KLWL3UC.js +18 -0
- package/dist/chunk-4VP4KH3K.js +962 -0
- package/dist/chunk-4ZG3XFUR.js +77 -0
- package/dist/chunk-5B2AEOW5.js +5407 -0
- package/dist/{chunk-K2ITRMHZ.js → chunk-5TSRNF4G.js} +6 -138
- package/dist/{chunk-OLBBMFRD.js → chunk-5UUP6MWO.js} +24 -62
- package/dist/chunk-65DEGPJ6.js +52 -0
- package/dist/chunk-6EJMN2Y3.js +17 -0
- package/dist/chunk-6EJV5X2W.js +16405 -0
- package/dist/chunk-6N5PTWMY.js +136 -0
- package/dist/chunk-6XLNIQDB.js +27 -0
- package/dist/{chunk-TEKV33Q5.js → chunk-77VKQEHF.js} +65 -33
- package/dist/chunk-7CR24IG7.js +242 -0
- package/dist/{chunk-M5T5VO65.js → chunk-7EYHLWU7.js} +837 -635
- package/dist/chunk-7MNJORFF.js +22 -0
- package/dist/{chunk-KY56HMHH.js → chunk-A3CYT5EX.js} +125 -31
- package/dist/chunk-AGYYIBLL.js +1069 -0
- package/dist/chunk-AO4RKG4M.js +277 -0
- package/dist/{chunk-GB6QRBXN.js → chunk-APJ265NV.js} +54 -1187
- package/dist/chunk-ARBGF5F7.js +174 -0
- package/dist/{chunk-673JJUWJ.js → chunk-BMEMKKIT.js} +2 -2
- package/dist/chunk-CBCAPZAA.js +229 -0
- package/dist/chunk-CMZWFGD2.js +352 -0
- package/dist/chunk-ECH6PKUQ.js +39 -0
- package/dist/chunk-ED4KHGC3.js +143 -0
- package/dist/chunk-EKMEHE4H.js +340 -0
- package/dist/{chunk-RPTR2H26.js → chunk-EPVUXGXG.js} +21 -15
- package/dist/chunk-FCSXB6T2.js +338 -0
- package/dist/chunk-FJ3H4MN5.js +48 -0
- package/dist/chunk-FQ4SKYE4.js +29 -0
- package/dist/chunk-G2DE3C7R.js +644 -0
- package/dist/chunk-G4BMMOKF.js +182 -0
- package/dist/{chunk-ZPY3JZ5E.js → chunk-GGXXDWE4.js} +183 -1233
- package/dist/chunk-HC4CLZ2Y.js +68 -0
- package/dist/{chunk-LU4TK2PR.js → chunk-HFSBBKSQ.js} +5 -56
- package/dist/{chunk-PIUMUEMV.js → chunk-HKIYEGME.js} +10 -6
- package/dist/chunk-I4HZDVNP.js +73 -0
- package/dist/{chunk-4QKXUHSR.js → chunk-IGLFWIYI.js} +70 -20
- package/dist/chunk-IKCO5N3L.js +162 -0
- package/dist/chunk-IR4CFBFN.js +56 -0
- package/dist/{chunk-PFEFKVGL.js → chunk-J5HN4RYU.js} +13 -11
- package/dist/{chunk-R5KLMSBV.js → chunk-J5Q24KAG.js} +2 -2
- package/dist/{chunk-K5XEMXTI.js → chunk-JVCV3ICN.js} +1 -1
- package/dist/chunk-KJ5LWLOE.js +1077 -0
- package/dist/chunk-LBMZMYH2.js +285 -0
- package/dist/{chunk-G34LV2PF.js → chunk-LBNRH5WM.js} +84 -170
- package/dist/{chunk-H6F6BYOH.js → chunk-LZSJBIVT.js} +7003 -7434
- package/dist/{chunk-HQQID6OA.js → chunk-M6SHUN7Q.js} +5 -5
- package/dist/chunk-MAW544W2.js +1882 -0
- package/dist/chunk-MBS4V7ZP.js +217 -0
- package/dist/{chunk-FST4FYJB.js → chunk-MFFY33HR.js} +99 -140
- package/dist/chunk-MNA4JGU4.js +255 -0
- package/dist/chunk-MQSRRFWA.js +3428 -0
- package/dist/{chunk-BSU2YIWB.js → chunk-MVVUPGPW.js} +131 -136
- package/dist/chunk-OAO4GE4M.js +619 -0
- package/dist/chunk-OHHN2SO4.js +5135 -0
- package/dist/chunk-OKGUZO2U.js +34 -0
- package/dist/{chunk-GAEBEQVI.js → chunk-OOJYHWRB.js} +32 -346
- package/dist/{chunk-Q5WJOSJ7.js → chunk-OQ33BKR3.js} +2 -1
- package/dist/chunk-OQE5J4C6.js +73 -0
- package/dist/{chunk-KKNLWXI6.js → chunk-ORBHGJC5.js} +8 -8
- package/dist/chunk-POHLU5DW.js +1186 -0
- package/dist/chunk-QKMUKYO7.js +4961 -0
- package/dist/{chunk-ASND7OZK.js → chunk-QTYWRVRA.js} +13 -13
- package/dist/{chunk-EYOKLTMF.js → chunk-SRF2PJNW.js} +17 -3
- package/dist/chunk-SST6Z5JA.js +80 -0
- package/dist/chunk-STBPMHSX.js +2456 -0
- package/dist/chunk-T6YILFSB.js +80 -0
- package/dist/chunk-TZTZS7QK.js +227 -0
- package/dist/chunk-UOV2BYIW.js +107 -0
- package/dist/{chunk-Q3RUPKEJ.js → chunk-V4RXGQ5Q.js} +58 -189
- package/dist/chunk-VAKQQHWR.js +434 -0
- package/dist/chunk-VG7TBQIY.js +128 -0
- package/dist/chunk-VJWL6YS5.js +244 -0
- package/dist/chunk-WEH5XRJQ.js +32 -0
- package/dist/chunk-WMSVI4G2.js +2095 -0
- package/dist/chunk-WVO7V2QY.js +797 -0
- package/dist/chunk-X4RCMKVQ.js +641 -0
- package/dist/chunk-X75S7HFS.js +374 -0
- package/dist/chunk-XN3L4EYL.js +46 -0
- package/dist/{chunk-RDLVBZEO.js → chunk-YCWGATWI.js} +6 -4
- package/dist/chunk-YHZX5GEU.js +193 -0
- package/dist/chunk-YXLYO42X.js +91 -0
- package/dist/{chunk-NMOX6HFD.js → chunk-ZDOOVTXZ.js} +29 -77
- package/dist/chunk-ZI647VB5.js +37 -0
- package/dist/{chunk-C4PTHK7P.js → chunk-ZWLZP4ZT.js} +5 -5
- package/dist/cli/index.js +62 -54
- package/dist/clio-4LY5K2AC.js +25 -0
- package/dist/code-nav-7AX6FYE6.js +600 -0
- package/dist/codewiki/build-worker.js +66 -0
- package/dist/compile-cache-CVJMMODC.js +18 -0
- package/dist/{components-DMAOEKFB.js → components-KELWS457.js} +11 -6
- package/dist/{config-IRUQ7SE4.js → config-GTLUW2PR.js} +92 -55
- package/dist/configure-R6A64DHX.js +42 -0
- package/dist/context-5VKGUVJJ.js +866 -0
- package/dist/{context-3KWFLHJG.js → context-JFZEJ7W5.js} +15 -13
- package/dist/{context-5RADCKTR.js → context-RW5HC47S.js} +71 -35
- package/dist/{context-clear-7TSNPAAI.js → context-clear-6ZHBAZZT.js} +54 -28
- package/dist/{context-index-W4RLWOQH.js → context-index-BZ4UYMTC.js} +30 -24
- package/dist/dispatch-runner-VKBRCWQC.js +1997 -0
- package/dist/{docs-5AWSPS37.js → docs-2C2LTVT2.js} +23 -10
- package/dist/{doctor-UC5NAJYQ.js → doctor-KI767GSN.js} +27 -17
- package/dist/{eval-U6TJHRLX.js → eval-XSSNATB4.js} +29 -16
- package/dist/{evidence-YEGUW4L3.js → evidence-UA6AWDQQ.js} +46 -26
- package/dist/{evolve-TXARCTPG.js → evolve-QNTFGV6Z.js} +45 -25
- package/dist/{extensions-OZFJ3A3G.js → extensions-QVDOHDGJ.js} +16 -7
- package/dist/{fleet-6G3DHNYE.js → fleet-Q7UOMUSG.js} +163 -54
- package/dist/{fleet-preflight-DSNT37JK.js → fleet-preflight-DDN536IT.js} +7 -4
- package/dist/{init-KZ5QTF6M.js → init-WBB65ZHQ.js} +69 -32
- package/dist/{memory-73ESV5YC.js → memory-MD3O64RI.js} +48 -27
- package/dist/{models-A4PVNWJK.js → models-BZU34YWD.js} +39 -25
- package/dist/monitor-MEQA5C3I.js +661 -0
- package/dist/{chunk-FCIH3BIZ.js → orchestrator-CGFKEP27.js} +11832 -8687
- package/dist/{paths-C4H6IV77.js → paths-UXLN5YYZ.js} +10 -5
- package/dist/{preload-6WVMHX3A.js → preload-P6DGH2PZ.js} +2 -2
- package/dist/{reset-BGW6OGMV.js → reset-L2FQEE3E.js} +16 -10
- package/dist/{run-YTPEYQOH.js → run-IV4Q6RLN.js} +101 -61
- package/dist/{share-YIFFV4NQ.js → share-S5BZQC5I.js} +15 -8
- package/dist/{skills-2V6RA3OQ.js → skills-LQEKRDTN.js} +34 -14
- package/dist/{skills-eval-S2TVJO4F.js → skills-eval-3DC4HEWS.js} +70 -34
- package/dist/steer-GGWFUJUD.js +77 -0
- package/dist/{targets-TYXLPB23.js → targets-C4SSGQOB.js} +43 -27
- package/dist/terminal-lease-IT5JW2NR.js +395 -0
- package/dist/{trace-GGOJ6Q6Z.js → trace-PNCASAXC.js} +41 -16
- package/dist/{chunk-N6F52NLF.js → tree-sitter-HGKH6LG4.js} +28 -2306
- package/dist/{uninstall-LLLT4F4W.js → uninstall-FZCQCDKC.js} +10 -5
- package/dist/{upgrade-33G2LMM5.js → upgrade-7TT7SQ3G.js} +45 -25
- package/dist/{usage-ZAFSXKKG.js → usage-GV4PKT3M.js} +62 -31
- package/dist/verify-G6V4D2G7.js +716 -0
- package/dist/web-fetch-2YHJ3KTG.js +638 -0
- package/dist/{wiki-generate-NUQCVOQ3.js → wiki-generate-DQF6Z66B.js} +74 -34
- package/dist/worker/entry.js +221 -36
- package/dist/workspace-G4ZWUIPR.js +22 -0
- package/docs/README.md +22 -17
- package/docs/acp.md +168 -16
- package/docs/alcf-provider.md +1 -1
- package/docs/architecture.md +136 -7
- package/docs/artifact-versions.md +1 -1
- package/docs/built-in-agents.md +1 -1
- package/docs/capacity-and-scheduling.md +1 -1
- package/docs/commands-and-modes.md +114 -71
- package/docs/config-knobs-audit.md +1 -3
- package/docs/configuration-and-targets.md +174 -46
- package/docs/context-engine.md +29 -6
- package/docs/development-pipeline.md +26 -1
- package/docs/dispatch-architecture-rationale.md +1 -1
- package/docs/documentation-coverage.md +2 -2
- package/docs/documentation-guide.md +1 -1
- package/docs/environment-variables.md +13 -5
- package/docs/eval-runner.md +1 -1
- package/docs/evals-internal.md +1 -1
- package/docs/evidence-and-memory.md +6 -2
- package/docs/evolution.md +2 -2
- package/docs/exit-codes-and-output.md +15 -9
- package/docs/extensions-and-sharing.md +9 -9
- package/docs/fleet-dispatch.md +7 -5
- package/docs/git-commit-provenance.md +120 -0
- package/docs/glossary.md +1 -1
- package/docs/installation-and-lifecycle.md +34 -27
- package/docs/middleware-and-components.md +1 -1
- package/docs/model-catalog.md +45 -14
- package/docs/observability.md +8 -5
- package/docs/performance-methodology.md +491 -0
- package/docs/pi-boundary.md +72 -0
- package/docs/proactive-memory.md +3 -3
- package/docs/prompt-envelope-and-tools.md +24 -3
- package/docs/provider-adapter-cookbook.md +57 -4
- package/docs/release-cut-checklist.md +129 -115
- package/docs/safety-model.md +9 -5
- package/docs/scientific-validation.md +3 -3
- package/docs/session-lifecycle.md +55 -12
- package/docs/skills-marketplace.md +12 -8
- package/docs/time-conventions.md +1 -1
- package/docs/tool-usage.md +3 -3
- package/docs/trace-store.md +1 -1
- package/docs/troubleshooting.md +10 -7
- package/docs/tui-design.md +47 -10
- package/docs/worker-dispatch-mechanics.md +1 -1
- package/package.json +19 -22
- package/skills/coding/ast-grep/SKILL.md +136 -0
- package/skills/coding/ast-grep/evals.md +56 -0
- package/skills/coding/ast-grep/references/rule_reference.md +297 -0
- package/skills/coding/coding-standards/SKILL.md +113 -0
- package/skills/coding/coding-standards/evals.md +34 -0
- package/skills/coding/prototype/SKILL.md +86 -0
- package/skills/coding/prototype/evals.md +42 -0
- package/skills/coding/prototype/references/LOGIC.md +67 -0
- package/skills/coding/prototype/references/UI.md +112 -0
- package/skills/coding/tdd/SKILL.md +101 -0
- package/skills/coding/tdd/evals.md +41 -0
- package/skills/coding/tdd/references/mocking.md +59 -0
- package/skills/coding/tdd/references/tests.md +77 -0
- package/skills/context/context-handoff/SKILL.md +126 -0
- package/skills/context/context-handoff/evals.md +57 -0
- package/skills/context/context-handoff/scripts/new-handoff.sh +26 -0
- package/skills/context/context-prime/SKILL.md +95 -0
- package/skills/context/context-prime/evals.md +54 -0
- package/skills/meta/clio-dev/SKILL.md +91 -0
- package/skills/meta/clio-dev/evals.md +45 -0
- package/skills/meta/clio-test/SKILL.md +130 -0
- package/skills/meta/clio-test/evals.md +43 -0
- package/skills/meta/clio-test/references/harness.md +97 -0
- package/skills/meta/clio-test/references/test-map.md +59 -0
- package/skills/meta/credentials/SKILL.md +125 -0
- package/skills/meta/credentials/evals.md +104 -0
- package/skills/meta/find-skills/SKILL.md +72 -0
- package/skills/meta/find-skills/evals.md +47 -0
- package/skills/meta/herdr/SKILL.md +127 -0
- package/skills/meta/herdr/evals.md +38 -0
- package/skills/meta/skill-craft/SKILL.md +102 -0
- package/skills/meta/skill-craft/evals.md +41 -0
- package/skills/planning/architecture/SKILL.md +129 -0
- package/skills/planning/architecture/evals.md +36 -0
- package/skills/planning/backlog/SKILL.md +90 -0
- package/skills/planning/backlog/evals.md +43 -0
- package/skills/planning/prd/SKILL.md +82 -0
- package/skills/planning/prd/evals.md +49 -0
- package/skills/planning/product-intent/SKILL.md +112 -0
- package/skills/planning/product-intent/evals.md +36 -0
- package/skills/planning/tech-spec/SKILL.md +115 -0
- package/skills/planning/tech-spec/evals.md +47 -0
- package/skills/registry.yaml +136 -0
- package/skills/research/arxiv-literature/SKILL.md +104 -0
- package/skills/research/arxiv-literature/evals.md +58 -0
- package/skills/research/experiment-protocol/SKILL.md +122 -0
- package/skills/research/experiment-protocol/evals.md +91 -0
- package/skills/research/scientific-debugging/SKILL.md +119 -0
- package/skills/research/scientific-debugging/evals.md +138 -0
- package/skills/research/scientific-modernization/SKILL.md +138 -0
- package/skills/research/scientific-modernization/evals.md +84 -0
- package/skills/workflow/design-council/SKILL.md +139 -0
- package/skills/workflow/design-council/evals.md +97 -0
- package/skills/workflow/grill-me/SKILL.md +186 -0
- package/skills/workflow/grill-me/evals.md +78 -0
- package/skills/workflow/workflow-distiller/SKILL.md +136 -0
- package/skills/workflow/workflow-distiller/evals.md +107 -0
- package/src/cli/acp.ts +31 -4
- package/src/cli/clio.ts +68 -6
- package/src/cli/config-inspect.ts +28 -22
- package/src/cli/configure.ts +47 -9
- package/src/cli/context-clear.ts +2 -2
- package/src/cli/context-index.ts +21 -23
- package/src/cli/context.ts +13 -8
- package/src/cli/default-target.ts +9 -17
- package/src/cli/docs.ts +11 -5
- package/src/cli/evidence.ts +4 -1
- package/src/cli/extensions.ts +10 -1
- package/src/cli/fleet.ts +47 -6
- package/src/cli/index.ts +55 -26
- package/src/cli/memory.ts +3 -1
- package/src/cli/models.ts +1 -1
- package/src/cli/modes/json-stream.ts +37 -1
- package/src/cli/modes/print.ts +24 -9
- package/src/cli/run.ts +2 -2
- package/src/cli/skills-eval.ts +23 -8
- package/src/cli/skills.ts +19 -4
- package/src/cli/targets.ts +4 -0
- package/src/cli/text-layout.ts +15 -5
- package/src/cli/trace.ts +62 -14
- package/src/cli/upgrade.ts +18 -2
- package/src/cli/usage.ts +10 -3
- package/src/cli/wiki-generate.ts +2 -1
- package/src/core/agent-environment.ts +7 -0
- package/src/core/bash-exec.ts +72 -1
- package/src/core/boot-trace.ts +9 -4
- package/src/core/bus-events.ts +20 -4
- package/src/core/commit-attribution.ts +157 -0
- package/src/core/compile-cache.ts +159 -0
- package/src/core/config.ts +131 -2
- package/src/core/defaults.ts +39 -5
- package/src/core/domain-loader.ts +12 -5
- package/src/core/git-commit-attribution.ts +362 -0
- package/src/core/incomplete-installation.ts +45 -0
- package/src/core/response-schema.ts +1 -1
- package/src/core/safe-exec.ts +13 -1
- package/src/core/settings-layers.ts +155 -21
- package/src/core/skill-activation.ts +1 -1
- package/src/core/startup-timer.ts +3 -3
- package/src/core/state-file-lock.ts +13 -1
- package/src/core/termination.ts +78 -5
- package/src/domains/config/classify.ts +15 -3
- package/src/domains/config/extension.ts +19 -13
- package/src/domains/config/index.ts +10 -0
- package/src/domains/config/keybindings.ts +42 -6
- package/src/domains/context/bootstrap-prompt.ts +1 -1
- package/src/domains/context/bootstrap.ts +111 -18
- package/src/domains/context/clear.ts +16 -11
- package/src/domains/context/clio-md.ts +111 -9
- package/src/domains/context/codewiki/artifact.ts +400 -0
- package/src/domains/context/codewiki/build-worker-protocol.ts +24 -0
- package/src/domains/context/codewiki/build-worker.ts +54 -0
- package/src/domains/context/codewiki/coordinator.ts +182 -0
- package/src/domains/context/codewiki/indexer.ts +59 -144
- package/src/domains/context/codewiki/paths.ts +67 -0
- package/src/domains/context/codewiki/schema.ts +80 -0
- package/src/domains/context/codewiki/tree-sitter.ts +1 -1
- package/src/domains/context/contract.ts +11 -5
- package/src/domains/context/extension.ts +94 -143
- package/src/domains/context/fingerprint.ts +3 -1
- package/src/domains/context/index.ts +12 -22
- package/src/domains/context/project-metadata.ts +19 -0
- package/src/domains/context/prompt-context.ts +9 -10
- package/src/domains/context/refresh.ts +29 -21
- package/src/domains/context/runtime.ts +17 -0
- package/src/domains/context/wiki/generate.ts +39 -34
- package/src/domains/context/wiki/plan.ts +1 -1
- package/src/domains/context/wiki/prompts.ts +21 -8
- package/src/domains/dispatch/code-step.ts +20 -1
- package/src/domains/dispatch/extension.ts +158 -21
- package/src/domains/dispatch/failure-classification.ts +6 -0
- package/src/domains/dispatch/fleet-commit-attribution.ts +56 -0
- package/src/domains/dispatch/orphan-recovery.ts +50 -8
- package/src/domains/dispatch/receipt-integrity.ts +5 -0
- package/src/domains/dispatch/state.ts +31 -6
- package/src/domains/dispatch/transport.ts +2 -1
- package/src/domains/dispatch/types.ts +14 -0
- package/src/domains/dispatch/worker-spawn.ts +21 -2
- package/src/domains/eval/metrics/context.ts +1 -1
- package/src/domains/eval/types.ts +0 -1
- package/src/domains/evidence/build.ts +41 -1
- package/src/domains/lifecycle/migrations/2026-08-18-lmstudio-runtime-id.ts +52 -0
- package/src/domains/lifecycle/migrations/index.ts +24 -4
- package/src/domains/middleware/hooks-io.ts +12 -0
- package/src/domains/middleware/skills-reminder.ts +30 -15
- package/src/domains/prompts/compiler.ts +142 -84
- package/src/domains/prompts/contract.ts +18 -2
- package/src/domains/prompts/extension.ts +39 -7
- package/src/domains/prompts/fragment-loader.ts +0 -1
- package/src/domains/prompts/fragments/identity/clio.md +2 -4
- package/src/domains/prompts/fragments/identity/docs-routing.md +10 -0
- package/src/domains/prompts/fragments/identity/self-awareness.md +1 -45
- package/src/domains/prompts/fragments/operating/contract.md +4 -50
- package/src/domains/prompts/fragments/operating/delegation.md +42 -0
- package/src/domains/prompts/fragments/operating/skills.md +26 -0
- package/src/domains/prompts/fragments/operating/worker.md +16 -0
- package/src/domains/prompts/fragments/safety/auto-edit.md +5 -5
- package/src/domains/prompts/fragments/safety/full-auto.md +3 -3
- package/src/domains/prompts/fragments/safety/read-only.md +4 -4
- package/src/domains/prompts/fragments/safety/suggest.md +2 -2
- package/src/domains/prompts/fragments/wiki/page.md +10 -0
- package/src/domains/prompts/fragments/wiki/plan.md +10 -0
- package/src/domains/prompts/preload.ts +3 -3
- package/src/domains/providers/auth/api-key.ts +1 -1
- package/src/domains/providers/auth/backend-file.ts +20 -10
- package/src/domains/providers/auth/backend-memory.ts +59 -4
- package/src/domains/providers/auth/boot-status.ts +65 -0
- package/src/domains/providers/auth/oauth.ts +2 -1
- package/src/domains/providers/auth/storage.ts +97 -38
- package/src/domains/providers/capabilities.ts +12 -4
- package/src/domains/providers/contract.ts +15 -4
- package/src/domains/providers/extension.ts +18 -6
- package/src/domains/providers/model-runtime-capabilities.ts +15 -4
- package/src/domains/providers/models/local-models/clio-local-coding-targets.yaml +118 -35
- package/src/domains/providers/plugins.ts +5 -3
- package/src/domains/providers/probe/fingerprint.ts +25 -5
- package/src/domains/providers/registry.ts +31 -10
- package/src/domains/providers/runtimes/boot-manifest.ts +55 -0
- package/src/domains/providers/runtimes/builtins.ts +2 -2
- package/src/domains/providers/runtimes/common/lmstudio-http.ts +423 -0
- package/src/domains/providers/runtimes/common/local-synth.ts +6 -7
- package/src/domains/providers/runtimes/local-native/lmstudio.ts +241 -0
- package/src/domains/providers/support.ts +6 -3
- package/src/domains/providers/types/local-model-quirks.ts +7 -9
- package/src/domains/providers/types/runtime-descriptor.ts +12 -1
- package/src/domains/providers/types/target-descriptor.ts +22 -0
- package/src/domains/resources/contract.ts +0 -1
- package/src/domains/resources/extension.ts +1 -3
- package/src/domains/resources/loader.ts +3 -4
- package/src/domains/resources/prompts/loader.ts +16 -2
- package/src/domains/resources/prompts/substitute.ts +1 -65
- package/src/domains/resources/skills/content-hash.ts +2 -0
- package/src/domains/resources/skills/install.ts +17 -0
- package/src/domains/resources/skills/loader.ts +17 -10
- package/src/domains/resources/skills/marketplace.ts +55 -9
- package/src/domains/safety/action-classifier.ts +4 -2
- package/src/domains/safety/audit.ts +8 -2
- package/src/domains/safety/extension.ts +1 -1
- package/src/domains/session/compaction/branch-summary.ts +3 -2
- package/src/domains/session/compaction/cut-point.ts +2 -1
- package/src/domains/session/compaction/tokens.ts +2 -1
- package/src/domains/session/context-ledger.ts +14 -0
- package/src/domains/session/contract.ts +15 -0
- package/src/domains/session/decision-board.ts +190 -0
- package/src/domains/session/entries.ts +66 -3
- package/src/domains/session/extension.ts +93 -12
- package/src/domains/session/retry.ts +10 -18
- package/src/domains/session/session-artifacts.ts +107 -0
- package/src/domains/session/task-board.ts +207 -13
- package/src/domains/session/tree/active-path.ts +44 -5
- package/src/domains/session/tree/fork.ts +26 -27
- package/src/domains/session/tree/preview.ts +2 -2
- package/src/domains/session/workspace/git-probe.ts +17 -11
- package/src/domains/user-tasks/store.ts +297 -0
- package/src/engine/acp/errors.ts +96 -0
- package/src/engine/acp/server.ts +1728 -146
- package/src/engine/acp/transport.ts +135 -14
- package/src/engine/acp/types.ts +26 -0
- package/src/engine/agent.ts +3 -3
- package/src/engine/ai.ts +32 -27
- package/src/engine/alcf-oauth.ts +26 -19
- package/src/engine/api-registry.ts +223 -0
- package/src/engine/apis/index.ts +3 -7
- package/src/engine/apis/llamacpp-residency.ts +49 -9
- package/src/engine/apis/lmstudio-residency.ts +5 -21
- package/src/engine/apis/lmstudio.ts +243 -0
- package/src/engine/apis/ollama-native.ts +24 -3
- package/src/engine/apis/openai-completions.ts +170 -91
- package/src/engine/apis/residency.ts +139 -3
- package/src/engine/apis/types.ts +16 -0
- package/src/engine/env-api-keys.ts +98 -0
- package/src/engine/gemma-channel-filter.ts +223 -0
- package/src/engine/instrumented-tui.ts +192 -0
- package/src/engine/messages.ts +14 -0
- package/src/engine/models.ts +42 -0
- package/src/engine/oauth.ts +16 -12
- package/src/engine/prompt-templates.ts +1 -0
- package/src/engine/provider-payload.ts +16 -59
- package/src/engine/strip-tokenizer-sentinels.ts +1 -1
- package/src/engine/truncate.ts +9 -0
- package/src/engine/tui.ts +17 -9
- package/src/engine/types.ts +3 -6
- package/src/engine/worker-runtime-capabilities.ts +5 -0
- package/src/engine/worker-runtime.ts +1 -1
- package/src/engine/worker-tools.ts +9 -4
- package/src/entry/boot-options.ts +50 -0
- package/src/entry/orchestrator.ts +288 -150
- package/src/interactive/application-controller.ts +89 -2
- package/src/interactive/chat-loop.ts +266 -41
- package/src/interactive/chat-panel.ts +173 -47
- package/src/interactive/chat-renderer.ts +262 -72
- package/src/interactive/clio-editor.ts +3 -8
- package/src/interactive/command-fallbacks.ts +2 -2
- package/src/interactive/context-overlay.ts +27 -1
- package/src/interactive/editor-submit.ts +228 -24
- package/src/interactive/export-html/ansi-to-html.ts +161 -0
- package/src/interactive/export-html/index.ts +51 -0
- package/src/interactive/export-html/template.ts +45 -0
- package/src/interactive/export-html/tool-renderer.ts +54 -0
- package/src/interactive/footer/dashboard.ts +4 -0
- package/src/interactive/footer/notifications.ts +1 -1
- package/src/interactive/footer/widgets.ts +20 -2
- package/src/interactive/footer-panel.ts +2 -2
- package/src/interactive/format-time.ts +14 -2
- package/src/interactive/interactive-application.ts +201 -17
- package/src/interactive/interactive-event-projection.ts +6 -1
- package/src/interactive/interactive-input-runtime.ts +50 -4
- package/src/interactive/interactive-presentation.ts +151 -19
- package/src/interactive/interactive-shell.ts +268 -14
- package/src/interactive/interactive-slash-runtime.ts +176 -114
- package/src/interactive/interactive-tickers.ts +38 -7
- package/src/interactive/keybinding-manager.ts +1 -1
- package/src/interactive/layout.ts +40 -3
- package/src/interactive/overlay-frame.ts +1 -1
- package/src/interactive/overlay-general-openers.ts +58 -1
- package/src/interactive/overlay-key-routing.ts +3 -0
- package/src/interactive/overlay-lifecycle.ts +13 -0
- package/src/interactive/overlay-permission-lifecycle.ts +2 -1
- package/src/interactive/overlay-session-lifecycle.ts +69 -12
- package/src/interactive/overlays/decisions.ts +300 -0
- package/src/interactive/overlays/help-reference.ts +15 -10
- package/src/interactive/overlays/model-selector.ts +34 -16
- package/src/interactive/overlays/session-selector.ts +18 -0
- package/src/interactive/overlays/settings.ts +105 -17
- package/src/interactive/overlays/skills-hub.ts +4 -4
- package/src/interactive/overlays/tree-selector.ts +41 -6
- package/src/interactive/render-trace.ts +499 -90
- package/src/interactive/renderers/compaction-summary.ts +2 -2
- package/src/interactive/renderers/diff.ts +115 -104
- package/src/interactive/renderers/mermaid.ts +53 -0
- package/src/interactive/renderers/tool-execution.ts +386 -133
- package/src/interactive/renderers/worker-entry.ts +20 -4
- package/src/interactive/session-switch-settlement.ts +10 -0
- package/src/interactive/slash-autocomplete.ts +6 -114
- package/src/interactive/slash-commands.ts +135 -47
- package/src/interactive/slash-spec.ts +9 -38
- package/src/interactive/status/controller.ts +5 -1
- package/src/interactive/stdout-backpressure.ts +99 -0
- package/src/interactive/stream-pacer.ts +530 -0
- package/src/interactive/stream-pacing-policy.ts +66 -0
- package/src/interactive/tasks-overlay.ts +368 -14
- package/src/interactive/terminal-lease.ts +485 -0
- package/src/interactive/theme/tokens.ts +1 -1
- package/src/interactive/turn-context.ts +4 -3
- package/src/interactive/turn-persistence.ts +30 -13
- package/src/interactive/turn-queues.ts +12 -0
- package/src/interactive/turn-recovery.ts +25 -8
- package/src/interactive/turn-runtime.ts +79 -12
- package/src/interactive/turn-state.ts +10 -0
- package/src/interactive/view/artifacts.ts +114 -4
- package/src/interactive/view/view-overlay.ts +3 -0
- package/src/interactive/welcome-dashboard.ts +17 -16
- package/src/interactive/worker-receipts.ts +52 -3
- package/src/interactive/worker-stream.ts +5 -1
- package/src/tools/agent-tools.ts +23 -3
- package/src/tools/artifact.ts +2 -2
- package/src/tools/ask-user.ts +23 -13
- package/src/tools/bash.ts +30 -2
- package/src/tools/bootstrap.ts +34 -431
- package/src/tools/builtin-tool-catalog.ts +265 -0
- package/src/tools/codewiki/code-nav-surface.ts +29 -0
- package/src/tools/codewiki/code-nav.ts +8 -22
- package/src/tools/codewiki/shared.ts +41 -38
- package/src/tools/context/docs-engine.ts +14 -3
- package/src/tools/context/index.ts +107 -28
- package/src/tools/context/surface.ts +19 -0
- package/src/tools/core-bootstrap.ts +168 -0
- package/src/tools/credential-present.ts +5 -5
- package/src/tools/dispatch-admission.ts +533 -0
- package/src/tools/dispatch-background.ts +54 -0
- package/src/tools/dispatch-event-text.ts +6 -0
- package/src/tools/dispatch-plan.ts +9 -4
- package/src/tools/dispatch-run-events.ts +238 -0
- package/src/tools/dispatch-runner.ts +2370 -0
- package/src/tools/dispatch-scout-admission.ts +295 -0
- package/src/tools/dispatch-types.ts +77 -0
- package/src/tools/dispatch.ts +67 -3161
- package/src/tools/find.ts +4 -2
- package/src/tools/grep.ts +2 -2
- package/src/tools/lazy-tool.ts +60 -0
- package/src/tools/ledger.ts +3 -3
- package/src/tools/monitor-surface.ts +36 -0
- package/src/tools/monitor.ts +2 -32
- package/src/tools/observers.ts +2 -2
- package/src/tools/registry.ts +39 -27
- package/src/tools/safe-exec.ts +2 -2
- package/src/tools/steer-surface.ts +17 -0
- package/src/tools/steer.ts +2 -13
- package/src/tools/tasks.ts +108 -11
- package/src/tools/truncate.ts +25 -184
- package/src/tools/verify/frontend.ts +3 -1
- package/src/tools/verify/index.ts +3 -38
- package/src/tools/verify/surface.ts +46 -0
- package/src/tools/web-fetch-surface.ts +23 -0
- package/src/tools/web-fetch.ts +2 -20
- package/src/tools/write.ts +7 -2
- package/src/worker/entry.ts +39 -2
- package/src/worker/spec-contract.ts +26 -5
- package/dist/chunk-7SS2CTV2.js +0 -61361
- package/dist/chunk-DKGKUHFA.js +0 -924
- package/dist/chunk-GEP36Y4X.js +0 -12796
- package/dist/chunk-XYWBQRDM.js +0 -137
- package/dist/clio-BZVGEUFJ.js +0 -58
- package/dist/configure-S7S6F6CL.js +0 -32
- package/docs/html/agents_blueprint.html +0 -936
- package/docs/html/alcf_blueprint.html +0 -324
- package/docs/html/architecture_blueprint.html +0 -850
- package/docs/html/commands_blueprint.html +0 -939
- package/docs/html/config_knobs_audit_blueprint.html +0 -178
- package/docs/html/configuration_blueprint.html +0 -1080
- package/docs/html/context_blueprint.html +0 -603
- package/docs/html/documentation_blueprint.html +0 -832
- package/docs/html/environment_blueprint.html +0 -404
- package/docs/html/eval_blueprint.html +0 -743
- package/docs/html/evals_internal_blueprint.html +0 -190
- package/docs/html/evolution_blueprint.html +0 -674
- package/docs/html/extensions_blueprint.html +0 -2065
- package/docs/html/fleet_dispatch_blueprint.html +0 -286
- package/docs/html/index.html +0 -919
- package/docs/html/lifecycle_blueprint.html +0 -723
- package/docs/html/memory_blueprint.html +0 -699
- package/docs/html/middleware_blueprint.html +0 -664
- package/docs/html/models_blueprint.html +0 -2366
- package/docs/html/observability_blueprint.html +0 -683
- package/docs/html/provider_adapter_blueprint.html +0 -245
- package/docs/html/safety_blueprint.html +0 -1386
- package/docs/html/shared.css +0 -571
- package/docs/html/shared.js +0 -143
- package/docs/html/skills_blueprint.html +0 -671
- package/docs/html/soak_blueprint.html +0 -182
- package/docs/html/tool_usage_blueprint.html +0 -350
- package/docs/html/tools_blueprint.html +0 -2249
- package/docs/html/trace_blueprint.html +0 -235
- package/docs/html/tui_design_blueprint.html +0 -374
- package/docs/html/validation_blueprint.html +0 -961
- package/docs/html/worker_dispatch_blueprint.html +0 -231
- package/src/core/release.ts +0 -2
- package/src/domains/providers/runtimes/common/lmstudio-logger.ts +0 -32
- package/src/domains/providers/runtimes/local-native/lmstudio-native.ts +0 -491
- package/src/engine/apis/lmstudio-native.ts +0 -1438
- package/src/engine/apis/thinking-replay.ts +0 -11
- package/src/tools/string-enum.ts +0 -15
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: experiment-protocol
|
|
3
|
+
description: Use when running a performance study, numerical comparison, parameter sweep, kernel or solver benchmark, or any change justified by "faster" or "more accurate", and success criteria should be locked before results exist. Pre-registers thresholds, tolerances, environment pins, and verdict conditions into the repository validation contract before any measurement. Triggers on "benchmark", "compare implementations", "optimize", "tolerance", "reproduce results", "parameter sweep". Not for diagnosing a stalled bug; use scientific-debugging.
|
|
4
|
+
version: 0.1.2
|
|
5
|
+
license: Apache-2.0
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- read
|
|
8
|
+
- write
|
|
9
|
+
- grep
|
|
10
|
+
- ls
|
|
11
|
+
- find
|
|
12
|
+
- git
|
|
13
|
+
- context
|
|
14
|
+
- code_nav
|
|
15
|
+
- bash
|
|
16
|
+
clio:
|
|
17
|
+
registry-id: iowarp/clio-coder
|
|
18
|
+
source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/experiment-protocol
|
|
19
|
+
audit: pass
|
|
20
|
+
provenance: designed
|
|
21
|
+
eval-status: smoke-checked
|
|
22
|
+
model-size: any
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
# Experiment Protocol
|
|
26
|
+
|
|
27
|
+
Lock the success criteria before any result exists. Pre-registration that
|
|
28
|
+
happens after the first measurement is worthless: it can only ratify what was
|
|
29
|
+
already seen. Moving the threshold after seeing results is the exact failure
|
|
30
|
+
mode this protocol exists to prevent.
|
|
31
|
+
|
|
32
|
+
Anti-trigger: if the question is "why is this output wrong", that is a
|
|
33
|
+
diagnosis, not an experiment; use scientific-debugging.
|
|
34
|
+
|
|
35
|
+
## Phase 0 - Pre-register
|
|
36
|
+
|
|
37
|
+
Before any measurement, write the protocol into the repository validation
|
|
38
|
+
contract: `.clio-coder/validation.yaml` (preferred) or `VALIDATION.md` at the repo
|
|
39
|
+
root. Creating this file raises Clio's repo-derived rigor to high immediately;
|
|
40
|
+
from the next turn onward, completion claims in this session must carry
|
|
41
|
+
validation evidence or state a limitation. That escalation is the point:
|
|
42
|
+
the contract arms the finish gate with the criteria you are about to commit to.
|
|
43
|
+
|
|
44
|
+
The pre-registration must contain:
|
|
45
|
+
|
|
46
|
+
- **Outcome**: one sentence, e.g. "the fused kernel reaches at least 1.8x the
|
|
47
|
+
baseline throughput on the pinned input at equal accuracy".
|
|
48
|
+
- **Thresholds**: minimum (below this is REFUTED), target, stretch.
|
|
49
|
+
- **Tolerance semantics per metric**: absolute for near-zero quantities,
|
|
50
|
+
relative elsewhere; a mixed scheme must say which applies where. "1e-6" with
|
|
51
|
+
no stated semantics is not a tolerance.
|
|
52
|
+
- **Environment pin**: compiler and flags, modules or package versions, node
|
|
53
|
+
class, scheduler context (partition, exclusivity).
|
|
54
|
+
- **Input identity**: paths plus checksums (`sha256sum`).
|
|
55
|
+
- **Verdict conditions**: what observation makes the result CONFIRMED, REFUTED,
|
|
56
|
+
or INCONCLUSIVE. Committed now, immutable after the first measurement.
|
|
57
|
+
|
|
58
|
+
## Phase 1 - Baseline
|
|
59
|
+
|
|
60
|
+
Capture current behavior under the pinned environment before changing
|
|
61
|
+
anything. Store raw outputs and timings as artifacts; never edit them. A
|
|
62
|
+
speedup claim without a baseline captured under the same pin is a guess.
|
|
63
|
+
|
|
64
|
+
## Phase 2 - Experiment
|
|
65
|
+
|
|
66
|
+
- One independent variable per run. A run that changes the algorithm and the
|
|
67
|
+
compiler flags answers no question.
|
|
68
|
+
- Size repetitions to the noise: shared nodes and networked filesystems need
|
|
69
|
+
more repetitions and a reported variance, not a single lucky run.
|
|
70
|
+
- Record scheduler identity (job id, node list) alongside every measurement.
|
|
71
|
+
|
|
72
|
+
## Phase 3 - Analysis
|
|
73
|
+
|
|
74
|
+
Compare against the pre-registered thresholds only. If the protocol was
|
|
75
|
+
deviated from, log the deviation next to the result; do not silently absorb
|
|
76
|
+
it. Findings outside the registered outcome are marked exploratory and get
|
|
77
|
+
their own pre-registration if pursued.
|
|
78
|
+
|
|
79
|
+
## Phase 4 - Iterate
|
|
80
|
+
|
|
81
|
+
Keep a dead-ends ledger in the contract file or beside it: one line per
|
|
82
|
+
rejected approach with the reason it was rejected. Read it before proposing
|
|
83
|
+
the next approach; re-proposing a ledger entry wastes a run. Stop when one of
|
|
84
|
+
the pre-registered stop conditions holds: target met, budget exhausted, or all
|
|
85
|
+
candidate strategies rejected.
|
|
86
|
+
|
|
87
|
+
## Worked Example
|
|
88
|
+
|
|
89
|
+
Request: "make the halo exchange faster."
|
|
90
|
+
|
|
91
|
+
```yaml
|
|
92
|
+
# .clio-coder/validation.yaml
|
|
93
|
+
experiment: halo-exchange-overlap
|
|
94
|
+
outcome: overlap communication with interior compute; >= 1.5x step throughput
|
|
95
|
+
thresholds: { minimum: 1.2x, target: 1.5x, stretch: 2.0x }
|
|
96
|
+
metrics:
|
|
97
|
+
step_time: { semantics: relative, tolerance: 5% run-to-run variance }
|
|
98
|
+
solution_l2: { semantics: absolute, tolerance: 1e-12 vs baseline }
|
|
99
|
+
environment: gcc 13.2 -O3, openmpi 4.1.6, 4x cpu-bind=cores, exclusive nodes
|
|
100
|
+
inputs: { mesh: data/mesh-256.h5, sha256: "<checksum>" }
|
|
101
|
+
verdicts:
|
|
102
|
+
confirmed: step_time speedup >= 1.2x AND solution_l2 within tolerance
|
|
103
|
+
refuted: speedup < 1.2x with variance < 5%, or accuracy loss
|
|
104
|
+
inconclusive: run-to-run variance > 5% (resize repetitions first)
|
|
105
|
+
dead_ends: []
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Baseline: 20 repetitions on exclusive nodes, mean and variance recorded with
|
|
109
|
+
job ids. Experiment: nonblocking exchange only, flags untouched. Result: 1.3x,
|
|
110
|
+
accuracy within 1e-12: CONFIRMED at minimum, target not met; ledger gains
|
|
111
|
+
"persistent requests: no gain over nonblocking here, latency-bound", and the
|
|
112
|
+
next variable (message aggregation) gets its own run.
|
|
113
|
+
|
|
114
|
+
## Red Flags
|
|
115
|
+
|
|
116
|
+
- Any measurement taken before the contract file exists.
|
|
117
|
+
- A threshold, tolerance, or verdict condition edited after results appeared.
|
|
118
|
+
- Wall-clock numbers with no environment pin attached.
|
|
119
|
+
- A single-run victory claim on a shared machine.
|
|
120
|
+
- A REFUTED result reported as "promising"; refuted plus a ledger entry is a
|
|
121
|
+
successful experiment, say so plainly.
|
|
122
|
+
- Deleting or rewriting baseline artifacts.
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# Evals - experiment-protocol
|
|
2
|
+
|
|
3
|
+
Baseline scenarios (run a subagent WITHOUT the skill to capture the gap, then
|
|
4
|
+
WITH the skill to confirm it closes). Rubric is pass/fail per bullet.
|
|
5
|
+
|
|
6
|
+
## S1 - "make this kernel faster"
|
|
7
|
+
|
|
8
|
+
Setup: make the smoothing kernel in kernel.py faster.
|
|
9
|
+
|
|
10
|
+
Fixture:
|
|
11
|
+
```bash
|
|
12
|
+
printf 'import time\n\ndef smooth(values, window):\n out = []\n for i in range(len(values)):\n lo = max(0, i - window)\n hi = min(len(values), i + window + 1)\n out.append(sum(values[lo:hi]) / (hi - lo))\n return out\n\nif __name__ == "__main__":\n data = [float(i %% 97) for i in range(200000)]\n t0 = time.perf_counter()\n smooth(data, 25)\n print("seconds:", round(time.perf_counter() - t0, 3))\n' > kernel.py
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
Expected:
|
|
16
|
+
|
|
17
|
+
- Writes a pre-registration into `.clio-coder/validation.yaml` or `VALIDATION.md`
|
|
18
|
+
before running any benchmark or editing any code.
|
|
19
|
+
- The pre-registration contains thresholds (minimum/target/stretch) and
|
|
20
|
+
tolerance semantics stated per metric (absolute vs relative).
|
|
21
|
+
- Pins the environment (compiler, flags, versions, node/scheduler context)
|
|
22
|
+
and identifies inputs by path plus checksum.
|
|
23
|
+
- Captures a baseline under the same pin before changing anything.
|
|
24
|
+
- Changes one independent variable per experiment run.
|
|
25
|
+
|
|
26
|
+
## S2 - results miss the target
|
|
27
|
+
|
|
28
|
+
Setup: the pre-registered target was 1.5x; the measured result is 1.1x, below
|
|
29
|
+
the registered minimum of 1.2x.
|
|
30
|
+
|
|
31
|
+
Expected:
|
|
32
|
+
|
|
33
|
+
- Reports REFUTED against the original pre-registered threshold.
|
|
34
|
+
- Does not restate the goal, lower the threshold, or reframe 1.1x as success.
|
|
35
|
+
- Adds a dead-ends ledger entry naming the rejected approach and the reason.
|
|
36
|
+
- Proposes the next approach only after reading the ledger.
|
|
37
|
+
|
|
38
|
+
## S3 - noisy measurements on a shared machine
|
|
39
|
+
|
|
40
|
+
Setup: benchmark runs on a shared node; run-to-run variance exceeds the gap
|
|
41
|
+
being measured.
|
|
42
|
+
|
|
43
|
+
Expected:
|
|
44
|
+
|
|
45
|
+
- Sizes repetitions to the observed noise instead of reporting a single run.
|
|
46
|
+
- Reports variance alongside the mean, with scheduler identity recorded.
|
|
47
|
+
- Declares INCONCLUSIVE if variance swamps the effect, rather than picking
|
|
48
|
+
the best run.
|
|
49
|
+
|
|
50
|
+
## S4 - anti-trigger: wrong output
|
|
51
|
+
|
|
52
|
+
Setup: user asks "why is this solver producing wrong values?"
|
|
53
|
+
|
|
54
|
+
Expected:
|
|
55
|
+
|
|
56
|
+
- Refers to scientific-debugging instead of starting a benchmark protocol.
|
|
57
|
+
- Does not write a validation contract for a diagnosis task.
|
|
58
|
+
|
|
59
|
+
## Baseline failure modes to watch for (RED)
|
|
60
|
+
|
|
61
|
+
- Benchmarks first, defines success afterward from whatever the numbers show.
|
|
62
|
+
- "Faster" claimed from one run, no baseline, no environment pin.
|
|
63
|
+
- Threshold quietly adjusted after seeing results.
|
|
64
|
+
- Tolerance given as a bare number with no absolute/relative semantics.
|
|
65
|
+
- Rejected approaches vanish; the next session re-proposes them.
|
|
66
|
+
- Raw baseline artifacts edited or overwritten.
|
|
67
|
+
|
|
68
|
+
## Observed gap closure
|
|
69
|
+
|
|
70
|
+
S1 run 2026-07-01, headless `clio-coder run` against a scratch git fixture (a pure
|
|
71
|
+
Python O(n^2) nearest-neighbor kernel with a single-shot bench script).
|
|
72
|
+
Prompt: "Make this kernel faster."
|
|
73
|
+
|
|
74
|
+
- RED (no skill): the agent edited the kernel immediately, reported a 7.3x
|
|
75
|
+
speedup from one timing run each way, wrote no contract, stated no
|
|
76
|
+
thresholds or tolerance semantics, and pinned nothing. Correctness was
|
|
77
|
+
checked ad hoc after the fact.
|
|
78
|
+
- GREEN (skill via `--skill` and `/skill <name>` invocation): the tool-call ledger
|
|
79
|
+
shows sha256 and environment capture, then `.clio-coder/validation.yaml` written
|
|
80
|
+
with min/target/stretch thresholds, per-metric tolerance semantics, and
|
|
81
|
+
verdict conditions, then the repetition-sized warm-up baseline, then
|
|
82
|
+
experiments, with the kernel edited only after a variant met the contract.
|
|
83
|
+
A float32 variant that beat the target but missed the accuracy tolerance
|
|
84
|
+
was rejected and logged in the dead-ends ledger. Writing the contract
|
|
85
|
+
raised repo rigor to high mid-session and the finish gate demanded
|
|
86
|
+
validation evidence before the turn settled. All five S1 bullets pass.
|
|
87
|
+
|
|
88
|
+
## Smoke record (2026-08-13)
|
|
89
|
+
|
|
90
|
+
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
91
|
+
(30B local, llamacpp on mini), full-auto sandbox. PASS. Pre-registration written before touching the seeded kernel; judge 5/5.
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: scientific-debugging
|
|
3
|
+
description: Use when debugging has stalled after the first obvious fix, when a failure spans multiple systems, when someone is about to try random changes, or when a scientific or HPC code produces wrong numbers, NaNs, nondeterministic results, or an unexplained performance regression. Forces falsifiable hypotheses across distinct fault classes with evidence-cited verdicts before any fix. Triggers on "why is this failing", "wrong results", "flaky", "nondeterministic", "diagnose", "root cause". Not for designing benchmarks or pre-registered experiments; use experiment-protocol.
|
|
4
|
+
version: 0.1.2
|
|
5
|
+
license: Apache-2.0
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- read
|
|
8
|
+
- grep
|
|
9
|
+
- ls
|
|
10
|
+
- find
|
|
11
|
+
- git
|
|
12
|
+
- context
|
|
13
|
+
- code_nav
|
|
14
|
+
- bash
|
|
15
|
+
clio:
|
|
16
|
+
registry-id: iowarp/clio-coder
|
|
17
|
+
source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/scientific-debugging
|
|
18
|
+
audit: pass
|
|
19
|
+
provenance: designed
|
|
20
|
+
eval-status: smoke-checked
|
|
21
|
+
model-size: any
|
|
22
|
+
---
|
|
23
|
+
|
|
24
|
+
# Scientific Debugging
|
|
25
|
+
|
|
26
|
+
Debug by falsification, not by trying fixes. A fix attempted before a confirmed
|
|
27
|
+
diagnosis is an experiment run without a hypothesis; when it "works" you have
|
|
28
|
+
learned nothing, and when it does not you have contaminated the evidence.
|
|
29
|
+
|
|
30
|
+
Anti-trigger: if the failure is a typo, a missing import, or an error message
|
|
31
|
+
that names its own cause, fix it directly and skip this workflow. The loop
|
|
32
|
+
below is for failures that survived the first obvious fix.
|
|
33
|
+
|
|
34
|
+
## The Loop
|
|
35
|
+
|
|
36
|
+
1. **Goal.** One sentence stating the observable "fixed" state. "The regression
|
|
37
|
+
test matches the reference output within the documented tolerance on two
|
|
38
|
+
consecutive runs" is a goal; "make it work" is not.
|
|
39
|
+
2. **Hypothesize.** Write at least three hypotheses. Each must name its fault
|
|
40
|
+
class and carry a falsification test: "this is WRONG if <observation>".
|
|
41
|
+
A hypothesis you cannot state a falsification test for is a hunch; refine it
|
|
42
|
+
until it is testable.
|
|
43
|
+
3. **Rank.** Order by test cost times prior likelihood. Run the cheapest
|
|
44
|
+
decisive test first, not the most interesting one.
|
|
45
|
+
4. **Test.** One variable per test. Preserve the raw failing output somewhere
|
|
46
|
+
untouched before you change anything.
|
|
47
|
+
5. **Verdict.** Record CONFIRMED, REFUTED, or INCONCLUSIVE per hypothesis, each
|
|
48
|
+
citing the command and output that decided it. A verdict without a citable
|
|
49
|
+
observation is a guess.
|
|
50
|
+
6. **Iterate.** Refuted everything? Generate new hypotheses from what the tests
|
|
51
|
+
revealed. Confirmed one? Only now edit code.
|
|
52
|
+
|
|
53
|
+
## Fault Classes
|
|
54
|
+
|
|
55
|
+
Hypotheses must span at least two distinct classes. Anchoring on a single class
|
|
56
|
+
is the failure mode this rule exists to break: the debugger who is sure it is
|
|
57
|
+
"a race" stops seeing the stale module load in front of them.
|
|
58
|
+
|
|
59
|
+
| Class | Typical suspects |
|
|
60
|
+
|---|---|
|
|
61
|
+
| numerics | accumulation order, mixed precision, tolerance misuse, fastmath |
|
|
62
|
+
| data | format or layout drift, HDF5/NetCDF/Zarr metadata, units, corruption |
|
|
63
|
+
| concurrency | races, MPI collective mismatch, nondeterministic reduction order |
|
|
64
|
+
| environment | modules, compiler flags, library versions, scheduler context |
|
|
65
|
+
| resources | memory pressure, filesystem quirks, quota, node differences |
|
|
66
|
+
| regression | a recent change; bisect the history instead of staring at code |
|
|
67
|
+
|
|
68
|
+
## Tiers
|
|
69
|
+
|
|
70
|
+
**Quick diagnosis** (default): the loop above, state held in conversation,
|
|
71
|
+
time-boxed at fifteen minutes of investigation. If the box expires without a
|
|
72
|
+
CONFIRMED verdict, escalate. Say that you are escalating; do not silently keep
|
|
73
|
+
poking.
|
|
74
|
+
|
|
75
|
+
**Structured investigation**: write an investigation file (for example
|
|
76
|
+
`INVESTIGATION.md` or `.clio-coder/investigation-<slug>.md` via bash heredoc since
|
|
77
|
+
this skill does not edit code) containing the goal, baseline measurements of
|
|
78
|
+
the failing behavior, and one experiment per hypothesis with its verdict
|
|
79
|
+
condition committed *before* the experiment runs. Update verdicts as evidence
|
|
80
|
+
arrives. The file is the state; the conversation is commentary.
|
|
81
|
+
|
|
82
|
+
## Evidence Rule
|
|
83
|
+
|
|
84
|
+
The fix commit should cite the confirming observation, e.g. "confirmed by:
|
|
85
|
+
`OMP_NUM_THREADS=1` reproduces bitwise-identical results, run log above".
|
|
86
|
+
High-rigor repos will demand validation evidence at completion anyway; produce
|
|
87
|
+
it proactively rather than being re-prompted for it.
|
|
88
|
+
|
|
89
|
+
## Worked Example
|
|
90
|
+
|
|
91
|
+
Report: "after the refactor, results differ from reference by 1e-4."
|
|
92
|
+
|
|
93
|
+
- Goal: `pytest tests/test_advection.py` passes against the pinned reference
|
|
94
|
+
within its stated rtol on a clean checkout plus the refactor commit.
|
|
95
|
+
- H1 (regression): the refactor changed the loop order and with it the
|
|
96
|
+
floating-point accumulation order. WRONG if the pre-refactor commit shows the
|
|
97
|
+
same 1e-4 drift. Test: `git stash && pytest ...` (cost: 1 min).
|
|
98
|
+
- H2 (numerics): the comparison uses an absolute tolerance where values near
|
|
99
|
+
zero need a relative one. WRONG if the drift is uniform across magnitudes.
|
|
100
|
+
Test: print elementwise error vs magnitude (cost: 5 min).
|
|
101
|
+
- H3 (environment): a different BLAS or compiler flag set is active in the new
|
|
102
|
+
environment. WRONG if `pip freeze`/module list matches the reference
|
|
103
|
+
environment pin. Test: diff environments (cost: 2 min).
|
|
104
|
+
- Order: H1, H3, H2. H1 verdict: REFUTED, pre-refactor commit is clean, output
|
|
105
|
+
cited. H3: REFUTED, environments identical. H2: CONFIRMED, error is constant
|
|
106
|
+
1e-4 at all magnitudes, so near-zero elements fail the absolute check.
|
|
107
|
+
- Only now edit: fix the tolerance semantics, cite H2's observation in the
|
|
108
|
+
commit message.
|
|
109
|
+
|
|
110
|
+
## Red Flags
|
|
111
|
+
|
|
112
|
+
- Editing code before any hypothesis has a CONFIRMED verdict.
|
|
113
|
+
- All hypotheses drawn from one fault class.
|
|
114
|
+
- A test that changes two variables at once.
|
|
115
|
+
- "It seems better now" presented as a verdict.
|
|
116
|
+
- Retrying a flaky test until it passes instead of making it deterministic.
|
|
117
|
+
- The fifteen-minute box expiring without an explicit escalation.
|
|
118
|
+
- Feeling certain: when a hypothesis feels obviously true, state its
|
|
119
|
+
falsification test anyway before touching the code.
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
# Evals - scientific-debugging
|
|
2
|
+
|
|
3
|
+
Baseline scenarios (run a subagent WITHOUT the skill to capture the gap, then
|
|
4
|
+
WITH the skill to confirm it closes). Rubric is pass/fail per bullet.
|
|
5
|
+
|
|
6
|
+
## S1 - stalled numerical bug
|
|
7
|
+
|
|
8
|
+
Setup: a small numerical project with a reference-comparison test. The
|
|
9
|
+
workspace already contains the fixture; run `python -m unittest -q` for the
|
|
10
|
+
reference check. Prompt: "after a refactor our results differ from the
|
|
11
|
+
reference by about 1e-4 and the obvious fix did not help; diagnose it."
|
|
12
|
+
|
|
13
|
+
Fixture:
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
mkdir -p tests
|
|
17
|
+
cat > diffusion.py <<'PY'
|
|
18
|
+
import math
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def weighted_mean(values, weights):
|
|
22
|
+
weighted = [value * weight for value, weight in zip(values, weights)]
|
|
23
|
+
return math.fsum(weighted) / math.fsum(weights)
|
|
24
|
+
PY
|
|
25
|
+
cat > tests/test_diffusion.py <<'PY'
|
|
26
|
+
import math
|
|
27
|
+
import unittest
|
|
28
|
+
|
|
29
|
+
from diffusion import weighted_mean
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class DiffusionReferenceTest(unittest.TestCase):
|
|
33
|
+
def test_weighted_mean_matches_reference(self):
|
|
34
|
+
values = [1.0e16, 6.0e-4, -1.0e16, 1.0e-4, 2.0e-4, -3.0e-4]
|
|
35
|
+
weights = [1.0, 1.0, 1.0, 1.0, 1.0, 1.0]
|
|
36
|
+
expected = math.fsum(value * weight for value, weight in zip(values, weights)) / math.fsum(weights)
|
|
37
|
+
observed = weighted_mean(values, weights)
|
|
38
|
+
self.assertLess(abs(observed - expected), 1.0e-12)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
if __name__ == "__main__":
|
|
42
|
+
unittest.main()
|
|
43
|
+
PY
|
|
44
|
+
git init -q
|
|
45
|
+
git config user.name "Clio Eval"
|
|
46
|
+
git config user.email "clio-eval@example.invalid"
|
|
47
|
+
git add diffusion.py tests/test_diffusion.py
|
|
48
|
+
git commit -q -m "add stable weighted mean reference"
|
|
49
|
+
cat > diffusion.py <<'PY'
|
|
50
|
+
def weighted_mean(values, weights):
|
|
51
|
+
total = 0.0
|
|
52
|
+
for value, weight in zip(values, weights):
|
|
53
|
+
total += value * weight
|
|
54
|
+
return total / sum(weights)
|
|
55
|
+
PY
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
Expected:
|
|
59
|
+
|
|
60
|
+
- States a one-sentence goal naming the observable fixed state before
|
|
61
|
+
investigating.
|
|
62
|
+
- Writes at least three hypotheses, each with an explicit "this is WRONG if"
|
|
63
|
+
falsification test.
|
|
64
|
+
- Hypotheses span at least two distinct fault classes, including numerics and
|
|
65
|
+
regression.
|
|
66
|
+
- Orders tests cheapest-first and runs one variable per test.
|
|
67
|
+
- Records a CONFIRMED/REFUTED/INCONCLUSIVE verdict per hypothesis, each citing
|
|
68
|
+
a command and its output.
|
|
69
|
+
- Edits no code before a hypothesis is CONFIRMED.
|
|
70
|
+
|
|
71
|
+
## S2 - flaky parallel test
|
|
72
|
+
|
|
73
|
+
Setup: a test suite where one MPI/threaded test fails intermittently. Prompt:
|
|
74
|
+
"this test is flaky, sometimes it passes; figure out why."
|
|
75
|
+
|
|
76
|
+
Expected:
|
|
77
|
+
|
|
78
|
+
- Includes a concurrency-class hypothesis (race, collective mismatch, or
|
|
79
|
+
reduction order).
|
|
80
|
+
- Attempts a deterministic reproduction (pin threads, fix seeds, force
|
|
81
|
+
ordering) rather than rerunning until green.
|
|
82
|
+
- Preserves the raw failing output before changing anything.
|
|
83
|
+
- Does not present a lucky pass as a verdict.
|
|
84
|
+
|
|
85
|
+
## S3 - escalation to structured tier
|
|
86
|
+
|
|
87
|
+
Setup: quick-tier investigation is not converging; all initial hypotheses come
|
|
88
|
+
back REFUTED and fifteen minutes of investigation have elapsed.
|
|
89
|
+
|
|
90
|
+
Expected:
|
|
91
|
+
|
|
92
|
+
- Explicitly announces escalation to the structured tier instead of silently
|
|
93
|
+
continuing.
|
|
94
|
+
- Writes an investigation file containing the goal, baseline measurements of
|
|
95
|
+
the failing behavior, and one experiment per hypothesis.
|
|
96
|
+
- Each experiment's verdict condition is committed before the experiment runs.
|
|
97
|
+
- New hypotheses are generated from what the refuted tests revealed.
|
|
98
|
+
|
|
99
|
+
## S4 - anti-trigger: trivial failure
|
|
100
|
+
|
|
101
|
+
Setup: the failure is an obvious typo or missing import whose error message
|
|
102
|
+
names its own cause. Prompt: "why is this failing?"
|
|
103
|
+
|
|
104
|
+
Expected:
|
|
105
|
+
|
|
106
|
+
- Fixes it directly or says a quick fix is appropriate.
|
|
107
|
+
- Does not run the hypothesis ceremony for a self-explanatory failure.
|
|
108
|
+
|
|
109
|
+
## Baseline failure modes to watch for (RED)
|
|
110
|
+
|
|
111
|
+
- Tries a fix immediately with no stated hypothesis.
|
|
112
|
+
- Single hypothesis, no falsification test, anchored on one fault class.
|
|
113
|
+
- Bundles the fix with the diagnosis in one edit.
|
|
114
|
+
- Verdicts asserted from intuition with no cited observation.
|
|
115
|
+
- Flaky test "resolved" by rerunning until it passes.
|
|
116
|
+
- Investigation drifts past the time box with no escalation and no file.
|
|
117
|
+
|
|
118
|
+
## Observed gap closure
|
|
119
|
+
|
|
120
|
+
S1 run 2026-07-01, headless `clio-coder run` against a scratch git fixture (an
|
|
121
|
+
order-sensitive summation whose refactor replaced `math.fsum` with a plain
|
|
122
|
+
accumulation loop; regression check fails by 1.474e-4).
|
|
123
|
+
|
|
124
|
+
- RED (no skill): the agent found the correct root cause but with no stated
|
|
125
|
+
goal, no enumerated hypotheses, no falsification tests, and no verdicts; it
|
|
126
|
+
read the diff, asserted the cause, and benchmarked alternative summations ad
|
|
127
|
+
hoc. On a harder bug that first guess would have been unfalsified anchoring.
|
|
128
|
+
- GREEN (skill via `--skill` and `/skill <name>` invocation): one-sentence goal,
|
|
129
|
+
three hypotheses (numerics, data, environment; the refactor regression
|
|
130
|
+
folded into H1) each with a WRONG-if test, explicit cheapest-first ranking,
|
|
131
|
+
CONFIRMED/REFUTED verdicts citing command output, untested H3 marked N/A,
|
|
132
|
+
fix applied only after the CONFIRMED verdict, and a commit message citing
|
|
133
|
+
the confirming observation. All six S1 bullets pass.
|
|
134
|
+
|
|
135
|
+
## Smoke record (2026-08-13)
|
|
136
|
+
|
|
137
|
+
One representative scenario via `clio-coder skills eval` against Nemo-3.5-Lightning
|
|
138
|
+
(30B local, llamacpp on mini), full-auto sandbox. PASS. Judge 6/6 on the seeded numerical fixture; cleanest research run.
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: scientific-modernization
|
|
3
|
+
description: Use when modernizing, porting, rewriting, packaging, accelerating, or replacing established scientific software, especially across languages, build systems, CPU/GPU backends, or maintained forks. Establishes an external scientific oracle, preserves compatibility, delivers in independently validated stages, and settles upstream ownership and long-term stewardship before calling the work complete. Triggers on "modernize this scientific code", "rewrite in Rust", "port to GPU", "replace this research tool", "migrate the build", "maintained fork", and "scientific parity". Not for an isolated benchmark; use experiment-protocol. Not for diagnosing wrong results; use scientific-debugging.
|
|
4
|
+
version: 0.2.0
|
|
5
|
+
license: Apache-2.0
|
|
6
|
+
allowed-tools:
|
|
7
|
+
- read
|
|
8
|
+
- write
|
|
9
|
+
- grep
|
|
10
|
+
- ls
|
|
11
|
+
- find
|
|
12
|
+
- git
|
|
13
|
+
- context
|
|
14
|
+
- code_nav
|
|
15
|
+
- bash
|
|
16
|
+
clio:
|
|
17
|
+
registry-id: iowarp/clio-coder
|
|
18
|
+
source-url: https://github.com/iowarp/clio-coder/tree/main/skills/research/scientific-modernization
|
|
19
|
+
audit: pass
|
|
20
|
+
provenance: designed
|
|
21
|
+
eval-status: scenarios-recorded
|
|
22
|
+
model-size: large
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
# Scientific Modernization
|
|
26
|
+
|
|
27
|
+
Modernization preserves scientific behavior and stewardship; it is not source
|
|
28
|
+
translation. Faster code, a clean build, and passing self-authored unit tests
|
|
29
|
+
do not establish scientific equivalence. Work the stages below in order; each
|
|
30
|
+
has an explicit exit condition.
|
|
31
|
+
|
|
32
|
+
## Stage 1 — Decide whether this work should exist
|
|
33
|
+
|
|
34
|
+
Identify the upstream project: active maintainers, license, release cadence,
|
|
35
|
+
supported users, contribution path. Prefer improving the original project when
|
|
36
|
+
coordination is viable.
|
|
37
|
+
|
|
38
|
+
Record the decision in writing before any implementation:
|
|
39
|
+
|
|
40
|
+
- path chosen: upstream contribution, maintained successor, or explicitly
|
|
41
|
+
scoped fork;
|
|
42
|
+
- the accountable maintainer or organization by name;
|
|
43
|
+
- the compatibility surface users already rely on;
|
|
44
|
+
- release, deprecation, and migration intent.
|
|
45
|
+
|
|
46
|
+
Do not start a rewrite whose only durable plan is "the community can maintain
|
|
47
|
+
it later." Exit: the four bullets above are written down and the user has seen
|
|
48
|
+
them.
|
|
49
|
+
|
|
50
|
+
## Stage 2 — Establish an independent scientific oracle
|
|
51
|
+
|
|
52
|
+
Write the acceptance contract in `.clio-coder/validation.yaml` (preferred) or
|
|
53
|
+
`VALIDATION.md` before changing behavior. Pick at least one oracle that does
|
|
54
|
+
not depend on the new implementation agreeing with itself:
|
|
55
|
+
|
|
56
|
+
- exact outputs from a trusted reference implementation;
|
|
57
|
+
- parity against the established tool over a representative corpus;
|
|
58
|
+
- known statistical behavior with pre-registered bounds;
|
|
59
|
+
- simulated data whose correct answer is fixed in advance;
|
|
60
|
+
- conserved quantities, analytical solutions, or domain invariants.
|
|
61
|
+
|
|
62
|
+
For every output, define: units, shapes, ordering, missing-value behavior,
|
|
63
|
+
determinism, absolute and relative tolerances, allowed platform variation.
|
|
64
|
+
Include adversarial and historically troublesome inputs.
|
|
65
|
+
|
|
66
|
+
If no credible oracle exists, stop and report that limitation. Implementation
|
|
67
|
+
velocity cannot repair an undefined truth condition. Keep performance
|
|
68
|
+
acceptance separate: pre-register speed claims through `experiment-protocol`.
|
|
69
|
+
Exit: the contract file exists and names its oracle(s).
|
|
70
|
+
|
|
71
|
+
## Stage 3 — Freeze the compatibility envelope
|
|
72
|
+
|
|
73
|
+
Inventory observable behavior before migrating anything:
|
|
74
|
+
|
|
75
|
+
- CLI and API contracts, file formats, schemas, defaults, error behavior;
|
|
76
|
+
- packaging, fresh-install, upgrade, and uninstall paths;
|
|
77
|
+
- supported platforms, compilers, runtimes, accelerators, schedulers;
|
|
78
|
+
- resource scaling, reproducibility controls, provenance;
|
|
79
|
+
- undocumented conventions captured by downstream tests and real workflows.
|
|
80
|
+
|
|
81
|
+
Capture reference outputs and install evidence from released artifacts, not
|
|
82
|
+
only the source checkout: mature tools carry user trust and conventions a
|
|
83
|
+
line-by-line translation misses. Exit: reference outputs and the envelope
|
|
84
|
+
inventory are stored as artifacts.
|
|
85
|
+
|
|
86
|
+
## Stage 4 — Deliver in independently valid stages
|
|
87
|
+
|
|
88
|
+
Split the work into the smallest stages that can each be checked against the
|
|
89
|
+
oracle. Every stage gets a before/after boundary, an acceptance command, a
|
|
90
|
+
retained artifact, and a rollback point. Prefer vertical slices that produce a
|
|
91
|
+
usable result over a big-bang rewrite.
|
|
92
|
+
|
|
93
|
+
Per stage:
|
|
94
|
+
|
|
95
|
+
1. Capture the reference result on the pinned corpus and environment.
|
|
96
|
+
2. Make one bounded change.
|
|
97
|
+
3. Run compatibility and scientific-oracle checks.
|
|
98
|
+
4. Preserve raw outputs, discrepancies, and provenance.
|
|
99
|
+
5. Resolve or explicitly classify every mismatch before expanding scope.
|
|
100
|
+
|
|
101
|
+
An agent's confidence is not evidence. If a reviewer cannot reconstruct the
|
|
102
|
+
comparison from retained artifacts, the stage is unverified; say so. Exit per
|
|
103
|
+
stage: oracle checks pass or every mismatch is classified in writing.
|
|
104
|
+
|
|
105
|
+
## Stage 5 — Budget explicitly for the last mile
|
|
106
|
+
|
|
107
|
+
Initial implementation is faster than convergence. Reserve work for: edge
|
|
108
|
+
cases, subtle numerical differences, nondeterminism, fresh environments,
|
|
109
|
+
large inputs, interrupted runs, packaging metadata, documentation, user
|
|
110
|
+
migration. Re-run the full oracle matrix after any optimization: correctness
|
|
111
|
+
proven before an optimization is not inherited by the optimized code.
|
|
112
|
+
|
|
113
|
+
Use `scientific-debugging` when a mismatch needs causal diagnosis. Never widen
|
|
114
|
+
tolerances or drop inconvenient corpus entries to obtain parity.
|
|
115
|
+
|
|
116
|
+
## Stage 6 — Ship only with evidence and stewardship
|
|
117
|
+
|
|
118
|
+
A completion claim must include all of:
|
|
119
|
+
|
|
120
|
+
- the oracle and compatibility matrix that passed, plus retained artifacts;
|
|
121
|
+
- unresolved differences and their user-visible consequences;
|
|
122
|
+
- fresh-install and documented-workflow results;
|
|
123
|
+
- performance results, if claimed, under their registered protocol;
|
|
124
|
+
- upstream status, or the named owner and maintenance plan;
|
|
125
|
+
- migration, rollback, release, and deprecation instructions.
|
|
126
|
+
|
|
127
|
+
If scientific validity or durable ownership is unresolved, report the work as
|
|
128
|
+
a prototype. Do not call it a replacement, successor, or production release.
|
|
129
|
+
|
|
130
|
+
## Red Flags
|
|
131
|
+
|
|
132
|
+
- A rewrite begins before maintainers or downstream users are consulted.
|
|
133
|
+
- Tests derived only from the new implementation.
|
|
134
|
+
- "The outputs look close" replacing declared tolerance semantics.
|
|
135
|
+
- Performance wins reported before scientific parity.
|
|
136
|
+
- One final comparison substituting for staged validation.
|
|
137
|
+
- A passing happy path hiding fresh-install, scale, or edge-case failures.
|
|
138
|
+
- A fork shipping without an accountable owner and maintenance horizon.
|