scip-query 0.14.0 → 0.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (321) hide show
  1. package/CHANGELOG.md +14 -2
  2. package/README.md +158 -62
  3. package/dist/augment-vue-worker.js +1 -1
  4. package/dist/chunk-2EU6WVR3.js +3 -0
  5. package/dist/chunk-2HFXPIBT.js +2 -0
  6. package/dist/chunk-2LGPMYS2.js +35 -0
  7. package/dist/chunk-2Y7YX54K.js +16 -0
  8. package/dist/chunk-363TZ64I.js +2 -0
  9. package/dist/chunk-3I34FZ6R.js +2 -0
  10. package/dist/chunk-3LAKH6BV.js +2 -0
  11. package/dist/chunk-3M2UCPFS.js +3 -0
  12. package/dist/chunk-3OPIMH42.js +2 -0
  13. package/dist/{chunk-QZOAOLLZ.js → chunk-3YAZH3JU.js} +2 -2
  14. package/dist/chunk-5P26M63Y.js +2 -0
  15. package/dist/{chunk-72FR7KEZ.js → chunk-5TVNLPUM.js} +14 -12
  16. package/dist/chunk-5UKDCXVC.js +5 -0
  17. package/dist/{chunk-R7U2I55B.js → chunk-5XDNFOAH.js} +2 -2
  18. package/dist/chunk-62Y3PC27.js +4 -0
  19. package/dist/chunk-6DNE65QS.js +40 -0
  20. package/dist/{chunk-N2OTJWC2.js → chunk-6EVORGCB.js} +2 -2
  21. package/dist/chunk-6GZ7LM7S.js +2 -0
  22. package/dist/chunk-753JWTUH.js +3 -0
  23. package/dist/chunk-7II7NFRV.js +4 -0
  24. package/dist/{chunk-M7EEHYSU.js → chunk-7RAS466A.js} +2 -2
  25. package/dist/chunk-A46JTKTM.js +5 -0
  26. package/dist/{chunk-2XRXEIRK.js → chunk-AAMZRDJT.js} +2 -2
  27. package/dist/chunk-AMPN2VMF.js +3 -0
  28. package/dist/{chunk-6XA4LDHY.js → chunk-B3X6KU75.js} +2 -2
  29. package/dist/chunk-BFWVHD2P.js +16 -0
  30. package/dist/chunk-BVHP2B4D.js +61 -0
  31. package/dist/{chunk-BUNDZ6K2.js → chunk-BVUA3GZC.js} +2 -2
  32. package/dist/{chunk-EDSD3VJK.js → chunk-CKIAEVGM.js} +2 -2
  33. package/dist/chunk-CM4EZLW4.js +2 -0
  34. package/dist/chunk-CS3FXTUC.js +2 -0
  35. package/dist/chunk-CVBS2FD4.js +2 -0
  36. package/dist/chunk-D5OXYOJP.js +7 -0
  37. package/dist/chunk-DA34GNNO.js +29 -0
  38. package/dist/chunk-DKK2H4D7.js +7 -0
  39. package/dist/chunk-EBSSCVR4.js +5 -0
  40. package/dist/{chunk-OXTM2RX3.js → chunk-EVCSXWER.js} +2 -2
  41. package/dist/chunk-FFIYGYQR.js +67 -0
  42. package/dist/chunk-G4KQ7S74.js +20 -0
  43. package/dist/{chunk-44P4SJFN.js → chunk-GWWYLYD3.js} +2 -2
  44. package/dist/chunk-HGUNWXJX.js +2 -0
  45. package/dist/chunk-HLRWOFXC.js +9 -0
  46. package/dist/{chunk-YQKECTPV.js → chunk-IRIM3SUR.js} +2 -2
  47. package/dist/chunk-JHK2DMUQ.js +4 -0
  48. package/dist/chunk-JQ2PCFZW.js +3 -0
  49. package/dist/chunk-JRFCWCOD.js +16 -0
  50. package/dist/chunk-JXAJDINF.js +2 -0
  51. package/dist/chunk-K32ZOW3P.js +2 -0
  52. package/dist/{chunk-3H5OSDF7.js → chunk-KI4WPR7Q.js} +2 -2
  53. package/dist/{chunk-43GKP6LI.js → chunk-KP2DJY7D.js} +2 -2
  54. package/dist/{chunk-YTVKPJXW.js → chunk-L27UIF3E.js} +2 -2
  55. package/dist/chunk-LT6AHYVY.js +18 -0
  56. package/dist/chunk-MCNWCJYL.js +11 -0
  57. package/dist/{chunk-K6UI6EBZ.js → chunk-MZB3DTHW.js} +2 -2
  58. package/dist/chunk-NLRWGGPY.js +8 -0
  59. package/dist/chunk-NLVKD45E.js +13 -0
  60. package/dist/chunk-OO6S4ECV.js +112 -0
  61. package/dist/{chunk-QYOPMTZ6.js → chunk-OYGKBDAD.js} +2 -2
  62. package/dist/chunk-P5QZILZ4.js +2 -0
  63. package/dist/chunk-P5XHIJHI.js +2 -0
  64. package/dist/chunk-PCCMMI2M.js +3 -0
  65. package/dist/{chunk-2TPCIPYA.js → chunk-Q65QYKU7.js} +2 -2
  66. package/dist/{chunk-K7SPOJHZ.js → chunk-QB7NRST4.js} +2 -2
  67. package/dist/{chunk-7JZRFDCU.js → chunk-QSNG3NHQ.js} +2 -2
  68. package/dist/chunk-QVZSWLLZ.js +2 -0
  69. package/dist/chunk-RLB7FZKR.js +10 -0
  70. package/dist/{chunk-KYB7FPXT.js → chunk-RPMT6MTX.js} +2 -2
  71. package/dist/chunk-RQWNF2RM.js +2 -0
  72. package/dist/chunk-RYFIRFYJ.js +4 -0
  73. package/dist/{chunk-QARYU7R3.js → chunk-S2FZILCF.js} +2 -2
  74. package/dist/{chunk-HV27H2V2.js → chunk-S3IXCXW2.js} +2 -2
  75. package/dist/{chunk-TYCZ7S4I.js → chunk-S5MTSES5.js} +2 -2
  76. package/dist/chunk-S5RZQA6V.js +2 -0
  77. package/dist/{chunk-PEHDARNC.js → chunk-SEEJPIWG.js} +2 -2
  78. package/dist/chunk-SOCS3467.js +2 -0
  79. package/dist/chunk-U6BJVXWQ.js +2 -0
  80. package/dist/chunk-UO4IADXJ.js +38 -0
  81. package/dist/{chunk-ZEO4CXQR.js → chunk-VLSI5MYU.js} +2 -2
  82. package/dist/chunk-VSQRU3AP.js +2 -0
  83. package/dist/chunk-VYQY75NA.js +2 -0
  84. package/dist/{chunk-NOKUZU56.js → chunk-VZCR3VQY.js} +2 -2
  85. package/dist/chunk-W3ZVRLCH.js +2 -0
  86. package/dist/chunk-WA64GKWB.js +20 -0
  87. package/dist/chunk-WFRZGMWJ.js +71 -0
  88. package/dist/chunk-WN6XEYVX.js +2 -0
  89. package/dist/chunk-XAEU4ROB.js +101 -0
  90. package/dist/chunk-XBPAJASV.js +2 -0
  91. package/dist/{chunk-GY3HAWUA.js → chunk-XC6PUFNN.js} +2 -2
  92. package/dist/chunk-Y42DMR6F.js +6 -0
  93. package/dist/chunk-YCVYOV4A.js +7 -0
  94. package/dist/{chunk-WPZPB42I.js → chunk-YD37KSEK.js} +2 -2
  95. package/dist/{chunk-V5W7H7V4.js → chunk-YFEJIYWH.js} +2 -2
  96. package/dist/chunk-YSUD2S7T.js +3 -0
  97. package/dist/{chunk-KJCDEDQW.js → chunk-ZEKBAMCN.js} +2 -2
  98. package/dist/chunk-ZHGUZKJL.js +2 -0
  99. package/dist/cli.js +587 -305
  100. package/dist/{config-types-BrHl3Bge.d.ts → config-types-SVkUIufk.d.ts} +10 -2
  101. package/dist/{db-_Bdx0E1W.d.ts → db-CS0TuVH6.d.ts} +1 -1
  102. package/dist/{git-history-Dao3_Pu9.d.ts → git-history-ChNHbfZQ.d.ts} +2 -1
  103. package/dist/{health-BEZ1Rt0S.d.ts → health-Btal3mQi.d.ts} +4 -4
  104. package/dist/index.d.ts +7 -5
  105. package/dist/index.js +1 -19
  106. package/dist/queries/affected.d.ts +2 -2
  107. package/dist/queries/affected.js +1 -1
  108. package/dist/queries/bottlenecks.d.ts +8 -2
  109. package/dist/queries/bottlenecks.js +1 -1
  110. package/dist/queries/by-kind.d.ts +2 -2
  111. package/dist/queries/by-kind.js +1 -1
  112. package/dist/queries/call-graph.d.ts +2 -2
  113. package/dist/queries/call-graph.js +1 -1
  114. package/dist/queries/change-surface.d.ts +2 -2
  115. package/dist/queries/change-surface.js +1 -1
  116. package/dist/queries/cleanup-plan.d.ts +2 -2
  117. package/dist/queries/cleanup-plan.js +1 -1
  118. package/dist/queries/co-change.d.ts +4 -3
  119. package/dist/queries/co-change.js +1 -1
  120. package/dist/queries/code.d.ts +2 -2
  121. package/dist/queries/code.js +1 -1
  122. package/dist/queries/complexity-hotspots.d.ts +3 -3
  123. package/dist/queries/complexity-hotspots.js +1 -1
  124. package/dist/queries/complexity.d.ts +5 -4
  125. package/dist/queries/complexity.js +1 -1
  126. package/dist/queries/convergence.d.ts +2 -2
  127. package/dist/queries/convergence.js +1 -1
  128. package/dist/queries/coupling.d.ts +2 -2
  129. package/dist/queries/coupling.js +1 -1
  130. package/dist/queries/cycles.d.ts +2 -2
  131. package/dist/queries/cycles.js +1 -1
  132. package/dist/queries/dataflow.d.ts +2 -2
  133. package/dist/queries/dataflow.js +1 -1
  134. package/dist/queries/dead.d.ts +12 -3
  135. package/dist/queries/dead.js +1 -1
  136. package/dist/queries/decorative-checkers.d.ts +2 -2
  137. package/dist/queries/decorative-checkers.js +1 -1
  138. package/dist/queries/deep-chains.d.ts +8 -3
  139. package/dist/queries/deep-chains.js +1 -1
  140. package/dist/queries/deps.d.ts +2 -2
  141. package/dist/queries/deps.js +1 -1
  142. package/dist/queries/diff-gate.d.ts +5 -4
  143. package/dist/queries/diff-gate.js +1 -1
  144. package/dist/queries/diff-impact.d.ts +3 -3
  145. package/dist/queries/diff-impact.js +1 -1
  146. package/dist/queries/doc-drift.d.ts +4 -2
  147. package/dist/queries/doc-drift.js +1 -1
  148. package/dist/queries/drift.d.ts +2 -2
  149. package/dist/queries/drift.js +1 -1
  150. package/dist/queries/duplicate-bodies.d.ts +3 -3
  151. package/dist/queries/duplicate-bodies.js +1 -1
  152. package/dist/queries/extract-candidates.d.ts +2 -2
  153. package/dist/queries/extract-candidates.js +1 -1
  154. package/dist/queries/fan.d.ts +11 -5
  155. package/dist/queries/fan.js +1 -1
  156. package/dist/queries/files.d.ts +2 -2
  157. package/dist/queries/health.d.ts +3 -3
  158. package/dist/queries/health.js +1 -1
  159. package/dist/queries/hierarchy.d.ts +2 -2
  160. package/dist/queries/hierarchy.js +1 -1
  161. package/dist/queries/hotspots.d.ts +2 -2
  162. package/dist/queries/hotspots.js +1 -1
  163. package/dist/queries/imports.d.ts +2 -2
  164. package/dist/queries/imports.js +1 -1
  165. package/dist/queries/incomplete-migration.d.ts +3 -3
  166. package/dist/queries/incomplete-migration.js +1 -1
  167. package/dist/queries/index.d.ts +5 -5
  168. package/dist/queries/index.js +1 -1
  169. package/dist/queries/isolated.d.ts +2 -2
  170. package/dist/queries/isolated.js +1 -1
  171. package/dist/queries/locality-candidates.d.ts +2 -2
  172. package/dist/queries/locality-candidates.js +1 -1
  173. package/dist/queries/members.d.ts +2 -2
  174. package/dist/queries/members.js +1 -1
  175. package/dist/queries/methods.d.ts +2 -2
  176. package/dist/queries/methods.js +1 -1
  177. package/dist/queries/not-implemented.d.ts +2 -2
  178. package/dist/queries/not-implemented.js +1 -1
  179. package/dist/queries/outline.d.ts +2 -2
  180. package/dist/queries/outline.js +1 -1
  181. package/dist/queries/passthrough-candidates.d.ts +3 -3
  182. package/dist/queries/passthrough-candidates.js +1 -1
  183. package/dist/queries/plan-context.d.ts +4 -4
  184. package/dist/queries/plan-context.js +1 -1
  185. package/dist/queries/react-component-duplicates.d.ts +4 -2
  186. package/dist/queries/react-component-duplicates.js +1 -1
  187. package/dist/queries/react-hook-candidates.d.ts +4 -2
  188. package/dist/queries/react-hook-candidates.js +1 -1
  189. package/dist/queries/react-large-component-pressure.d.ts +2 -2
  190. package/dist/queries/react-large-component-pressure.js +1 -1
  191. package/dist/queries/recent-duplicates.d.ts +4 -2
  192. package/dist/queries/recent-duplicates.js +1 -1
  193. package/dist/queries/redundant-reexports.d.ts +2 -2
  194. package/dist/queries/redundant-reexports.js +1 -1
  195. package/dist/queries/refs.d.ts +2 -2
  196. package/dist/queries/refs.js +1 -1
  197. package/dist/queries/self-audit.d.ts +2 -2
  198. package/dist/queries/self-audit.js +1 -1
  199. package/dist/queries/similar-chains.d.ts +2 -2
  200. package/dist/queries/similar-chains.js +1 -1
  201. package/dist/queries/similar-files.d.ts +2 -2
  202. package/dist/queries/similar-files.js +1 -1
  203. package/dist/queries/similar-signatures.d.ts +2 -2
  204. package/dist/queries/similar-signatures.js +1 -1
  205. package/dist/queries/similar.d.ts +2 -2
  206. package/dist/queries/similar.js +1 -1
  207. package/dist/queries/slice.d.ts +2 -2
  208. package/dist/queries/slice.js +1 -1
  209. package/dist/queries/stale-abstractions.d.ts +2 -2
  210. package/dist/queries/stale-abstractions.js +1 -1
  211. package/dist/queries/stats.d.ts +2 -2
  212. package/dist/queries/surface.d.ts +2 -2
  213. package/dist/queries/surface.js +1 -1
  214. package/dist/queries/symbols.d.ts +2 -2
  215. package/dist/queries/symbols.js +1 -1
  216. package/dist/queries/system.d.ts +2 -2
  217. package/dist/queries/system.js +1 -1
  218. package/dist/queries/test-quality.d.ts +2 -2
  219. package/dist/queries/test-quality.js +1 -1
  220. package/dist/queries/trace.d.ts +2 -2
  221. package/dist/queries/trace.js +1 -1
  222. package/dist/queries/twin-ab.d.ts +2 -2
  223. package/dist/queries/twin-ab.js +1 -1
  224. package/dist/queries/twin-drift.d.ts +2 -2
  225. package/dist/queries/twin-drift.js +1 -1
  226. package/dist/queries/unused-imports.d.ts +2 -2
  227. package/dist/queries/unused-imports.js +1 -1
  228. package/dist/queries/unused-params.d.ts +2 -2
  229. package/dist/queries/unused-params.js +1 -1
  230. package/dist/queries/vue-component-duplicates.d.ts +4 -2
  231. package/dist/queries/vue-component-duplicates.js +1 -1
  232. package/dist/queries/vue-composable-candidates.d.ts +4 -2
  233. package/dist/queries/vue-composable-candidates.js +1 -1
  234. package/dist/queries/vue-large-view-pressure.d.ts +2 -2
  235. package/dist/queries/vue-large-view-pressure.js +1 -1
  236. package/dist/queries/wrapper-candidates.d.ts +2 -2
  237. package/dist/queries/wrapper-candidates.js +1 -1
  238. package/dist/reindex-worker.js +256 -12
  239. package/dist/reindex.d.ts +29 -3
  240. package/dist/reindex.js +251 -22
  241. package/dist/runtime.d.ts +8 -4
  242. package/dist/runtime.js +2 -2
  243. package/dist/rust-semantic-session-server.js +2 -0
  244. package/dist/rust-semantic-session-worker.js +2 -0
  245. package/dist/rust-semantic-worker.js +2 -0
  246. package/dist/{scip-cli-C7cg4ZHR.d.ts → scip-cli-BKL-26TZ.d.ts} +2 -2
  247. package/dist/{symbol-types-DaoeXKUt.d.ts → symbol-types-BgWU6lhL.d.ts} +2 -0
  248. package/dist/watch-server.js +42 -0
  249. package/docs/AGENT_GUIDE.md +110 -34
  250. package/docs/AI_FAILURE_MODES.md +8 -3
  251. package/docs/COMMAND_REFERENCE.md +11 -6
  252. package/docs/REGEX_POLICY.md +3 -1
  253. package/docs/accuracy-audit-checklist.md +300 -0
  254. package/docs/accuracy-hardening-goal.md +493 -54
  255. package/docs/analyzer-inventory.md +31 -23
  256. package/docs/analyzer-validation-ledger.md +51 -21
  257. package/package.json +2 -1
  258. package/scripts/build-scip-windows.mjs +7 -7
  259. package/skills/_shared/SKILL.md +21 -18
  260. package/skills/scip-calibrate/SKILL.md +9 -3
  261. package/skills/scip-hyper-optimization/SKILL.md +5 -1
  262. package/skills/scip-query/SKILL.md +1 -1
  263. package/skills/scip-setup/SKILL.md +104 -4
  264. package/skills/scip-verify/SKILL.md +24 -13
  265. package/dist/chunk-27HMTUJE.js +0 -3
  266. package/dist/chunk-27XEGMCT.js +0 -7
  267. package/dist/chunk-3AMWF6V5.js +0 -3
  268. package/dist/chunk-5AAAEZ2Z.js +0 -2
  269. package/dist/chunk-5JRMZMLM.js +0 -2
  270. package/dist/chunk-5VY3RIEI.js +0 -38
  271. package/dist/chunk-6JUAGXR4.js +0 -4
  272. package/dist/chunk-6KDOYKCB.js +0 -40
  273. package/dist/chunk-6KMTPNZL.js +0 -3
  274. package/dist/chunk-7COVQPQQ.js +0 -9
  275. package/dist/chunk-7S5E7KWT.js +0 -2
  276. package/dist/chunk-7TVFPOD7.js +0 -102
  277. package/dist/chunk-A6KSEIRN.js +0 -61
  278. package/dist/chunk-AAMURYOK.js +0 -4
  279. package/dist/chunk-BP7S6V3M.js +0 -20
  280. package/dist/chunk-BRBF3E5X.js +0 -35
  281. package/dist/chunk-BZXE3HO4.js +0 -6
  282. package/dist/chunk-CXQQOXSI.js +0 -10
  283. package/dist/chunk-EVZSAR3S.js +0 -2
  284. package/dist/chunk-GMOEX7EV.js +0 -2
  285. package/dist/chunk-GNG622H3.js +0 -3
  286. package/dist/chunk-GUDKRIGW.js +0 -7
  287. package/dist/chunk-GULVO6NF.js +0 -43
  288. package/dist/chunk-HZZQKNVE.js +0 -2
  289. package/dist/chunk-I4R3SNMR.js +0 -2
  290. package/dist/chunk-IMQFJ53K.js +0 -2
  291. package/dist/chunk-ISLWJ4PY.js +0 -10
  292. package/dist/chunk-IUSSGHJO.js +0 -2
  293. package/dist/chunk-J5WVNZ6O.js +0 -2
  294. package/dist/chunk-JJUX2OAP.js +0 -2
  295. package/dist/chunk-JZBGTSAB.js +0 -13
  296. package/dist/chunk-K5N5RNGM.js +0 -2
  297. package/dist/chunk-KSJ544IV.js +0 -5
  298. package/dist/chunk-LY7NMOBW.js +0 -2
  299. package/dist/chunk-N23FT4EZ.js +0 -2
  300. package/dist/chunk-NCVFKOLB.js +0 -2
  301. package/dist/chunk-PESEPWZO.js +0 -2
  302. package/dist/chunk-PMY4UFGV.js +0 -2
  303. package/dist/chunk-Q7PWDXL5.js +0 -5
  304. package/dist/chunk-QVGFVRKD.js +0 -2
  305. package/dist/chunk-R5336VHZ.js +0 -2
  306. package/dist/chunk-RLB6SG25.js +0 -2
  307. package/dist/chunk-SFDVBARK.js +0 -2
  308. package/dist/chunk-TG7QSYCJ.js +0 -2
  309. package/dist/chunk-U3VMUDXJ.js +0 -2
  310. package/dist/chunk-UHPWZCYQ.js +0 -3
  311. package/dist/chunk-URSSPS5H.js +0 -2
  312. package/dist/chunk-WIBFXSYB.js +0 -6
  313. package/dist/chunk-WMO2VZZK.js +0 -3
  314. package/dist/chunk-WS3Z6W3M.js +0 -65
  315. package/dist/chunk-X3DTIEMU.js +0 -18
  316. package/dist/chunk-X5C43YVI.js +0 -2
  317. package/dist/chunk-X7CHWTZ2.js +0 -72
  318. package/dist/chunk-XC2PJOVM.js +0 -3
  319. package/dist/chunk-XDESMVOL.js +0 -4
  320. package/dist/chunk-XZBLZHXL.js +0 -16
  321. package/dist/chunk-ZMGBWSFZ.js +0 -2
@@ -1,54 +1,493 @@
1
- # Accuracy Hardening Goal
2
-
3
- Make scip-query's command answers reliable across languages and repositories by separating smoke tests from source-backed accuracy checks, fixing lookup/call-graph gaps found by real repos, and labeling heuristic results as candidates rather than exact facts.
4
-
5
- ## Scope
6
-
7
- - Keep CI changes out of scope for now.
8
- - Add committed test coverage that runs locally with `npm test`.
9
- - Keep real-repo calibration optional because those repositories exist only on some machines.
10
- - Prefer fixes that improve the shared indexing/query machinery over tuning output for one repository.
11
-
12
- ## Deliverables
13
-
14
- 1. Source-backed command accuracy coverage:
15
- - Add fixture/oracle tests for exact commands such as `symbols`, `code`, `refs`, `trace`, `call-graph`, `complexity`, `dataflow`, and `slice`.
16
- - Cover TypeScript, Python, Rust, and mixed-index behavior where practical.
17
- - Assert against source facts: definition text, expected files, expected callees/callers, and line-bounded code snippets.
18
-
19
- 2. Optional real-repo calibration harness:
20
- - Add a script that can run against local repositories when they exist.
21
- - Reindex each repo into a temporary cache.
22
- - Run source-backed checks for selected known symbols.
23
- - Print clear PASS/FAIL output with the command and evidence.
24
- - Never require these private/local repositories in normal test runs.
25
-
26
- 3. Indexer reliability taxonomy:
27
- - Preserve the existing fail-closed behavior for failed languages.
28
- - Keep `--allow-partial` as the explicit opt-in for incomplete mixed-language indexes.
29
- - Repair malformed SCIP definition occurrences before SQLite conversion when that can be done without inventing symbol metadata.
30
- - Keep conversion failures actionable and specific.
31
-
32
- 4. Lookup and call-graph accuracy:
33
- - Fix path-qualified lookup ranking such as `src/app.rs/run` so it prefers symbols defined in that file.
34
- - Keep exact call graph facts precise while still recovering real callable edges from SCIP mentions when AST extraction misses them.
35
- - Add regression tests for Rust qualified path calls.
36
-
37
- 5. Heuristic output labeling:
38
- - Label heuristic commands as candidates in CLI output.
39
- - Make the distinction visible for `similar`, `similar-files`, `similar-chains`, `extract-candidates`, `wrapper-candidates`, `passthrough-candidates`, `stale-abstractions`, `complexity-hotspots`, and `drift`.
40
- - Avoid implying those commands are exact compiler facts.
41
-
42
- 6. Performance guardrails:
43
- - Add an optional local performance/calibration path that records durations and index sizes.
44
- - Do not fail normal tests on private-repo timing.
45
-
46
- ## Acceptance Criteria
47
-
48
- - `npm test` passes.
49
- - `npm run typecheck` passes.
50
- - `npm run build` passes.
51
- - The optional calibration script passes on at least one TypeScript repo, one Python repo, and one Rust repo when those repos exist locally.
52
- - Path-qualified Rust lookup resolves `src/app.rs/run` to `app:run()`, not an unrelated `run`-named test.
53
- - `call-graph main` in the Rust fixture reports the qualified `app:run()` callee.
54
- - Heuristic command output includes explicit candidate/disclaimer language.
1
+ # Accuracy Hardening and Health Certification Roadmap
2
+
3
+ Date: 2026-07-10
4
+ Status: Complete for private-shadow readiness; public leaderboard remains a separate program
5
+
6
+ Current execution plan:
7
+ [`2026-07-11-remaining-accuracy-verification-program.md`](./plans/2026-07-11-remaining-accuracy-verification-program.md).
8
+ Closure certificate:
9
+ [`2026-07-11-accuracy-program-closure.md`](./validation/2026-07-11-accuracy-program-closure.md).
10
+
11
+ ## Goal
12
+
13
+ Make scip-query's reported health findings trustworthy enough to publish for
14
+ real open-source repositories. Trust is earned per detector, language, and
15
+ evidence capability; it is not inherited from the composite health score or
16
+ from calibration on a different ecosystem.
17
+
18
+ This program uses existing repositories as its primary evidence. Small
19
+ fixtures remain regression tests, while historical fixes and temporary
20
+ mutations in disposable worktrees provide known-positive cases for measuring
21
+ what a detector misses.
22
+
23
+ ## Accuracy Contract
24
+
25
+ Precision is the share of reported findings that satisfy the detector's truth
26
+ rule after the cited code is inspected. Recall is the share of known real
27
+ problems that the detector reports. Precision protects users from false
28
+ accusations; recall measures the problems the tool fails to show. Neither can
29
+ substitute for the other.
30
+
31
+ A truth rule is the repeatable decision procedure that makes two reviewers
32
+ classify the same candidate consistently. Every detector must define its rule
33
+ before its findings are sampled.
34
+
35
+ ### Certification levels
36
+
37
+ | Level | Evidence requirement | Public treatment |
38
+ | --------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------ |
39
+ | Certified | Observed precision at least 95%, conservative 95% confidence lower bound at least 90%, evidence from at least three repositories, and known-positive recall coverage | Actionable finding |
40
+ | Qualified | Observed precision at least 90% but not all certified gates are met | Investigation signal |
41
+ | Experimental | Observed precision below 90% or a known unresolved noise archetype can produce finding walls | Omit from public results; retain for development |
42
+ | Unsupported | Required parser, semantic provider, checker, framework model, or repository input is unavailable | Report `not analyzed`, never zero |
43
+ | Insufficient evidence | Too few reviewed findings or known-positive cases to estimate performance | Report uncertified status, never imply accuracy |
44
+
45
+ The conservative lower bound is the plausible floor after accounting for
46
+ sample size. For orientation, 95 valid rows out of 100 is 95% observed
47
+ precision but does not establish a 90% lower bound; about 97 out of 100 or 190
48
+ out of 200 does.
49
+
50
+ Certification is recorded independently for each detector-language pair and,
51
+ where framework behavior changes reachability, for the applicable framework
52
+ cohort. An aggregate percentage may summarize certified rows, but it must
53
+ never hide an uncertified combination.
54
+
55
+ ## Publication Contract
56
+
57
+ Until certification is complete, the composite health score is a local
58
+ prioritization aid and is not eligible for a public cross-repository
59
+ leaderboard.
60
+
61
+ Public output must:
62
+
63
+ - separate derived measurements, compiler/graph-backed candidates, and
64
+ investigation signals;
65
+ - identify language, detector version, indexer or semantic provider, commit,
66
+ dependency-install state, and capability status;
67
+ - report applicable production files or lines, excluded candidates, and
68
+ exclusion reasons;
69
+ - distinguish a supported zero from `unsupported`, `degraded`, and `not
70
+ analyzed`;
71
+ - normalize applicable findings by production lines or applicable files, not
72
+ by every indexer-emitted symbol;
73
+ - keep React, Vue, Rust macro/trait, Python framework, and other stack-specific
74
+ cohorts out of universal comparisons unless applicability is normalized;
75
+ - disclose Git-history coverage and commit filters for history-derived
76
+ measurements; and
77
+ - preserve the raw result and analysis version so a published result can be
78
+ reproduced after detector changes.
79
+
80
+ Facts and recommendations must remain separate. Identical bodies, dependency
81
+ cycles, branch counts, or Git change counts can be reproducible facts; a claim
82
+ that code should be consolidated, split, or deleted requires additional
83
+ evidence.
84
+
85
+ ## Real-Repository Corpus
86
+
87
+ The first calibration corpus uses repositories already available locally.
88
+ They remain read-only; any deletion test or planted known-positive case runs
89
+ in an isolated worktree or temporary clone.
90
+
91
+ | Cohort | Initial repositories | Coverage purpose |
92
+ | ---------- | ----------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------- |
93
+ | TypeScript | Vega_2.0, openwork, Stable_Management, traceroot; Element Plus and PrimeVue for Vue framework breadth | React, Vue, workspaces, packages, backend code, documentation/showcase families, and mixed-language boundaries |
94
+ | Rust | openai/codex, SynthRunnerRust | Large workspace plus small crate; traits, derives, macros, generated code, async runtimes, and public APIs |
95
+ | Python | scip-python, traceroot, Python files in openai/codex | Large Python codebase, mixed repositories, decorators, framework registration, and current semantic-capability gaps |
96
+
97
+ Additional public repositories may be added to cover a missing framework or
98
+ repository shape. Corpus additions must address a named gap rather than merely
99
+ increase volume.
100
+
101
+ ## Detector Truth Rules
102
+
103
+ Each detector gets a more detailed rule in its calibration manifest. The
104
+ starting rules are:
105
+
106
+ - **Dead code:** no production, public API, framework, generated, reflective,
107
+ configured, FFI, macro, or test-required consumer exists, and removal passes
108
+ every available checker covering the file. A missing checker prevents a
109
+ certified deletion claim.
110
+ - **Dependency cycle:** the reported files and dependency edges exist in the
111
+ accepted index. The cycle is factual; architectural harm is a separate
112
+ signal.
113
+ - **Exact duplicate body:** normalized callable bodies are equal. The equality
114
+ is factual; consolidation is a recommendation.
115
+ - **Complexity measurement:** source lines, branches, and cyclomatic estimate
116
+ match the stated AST or fallback basis. The composite hotspot product is a
117
+ prioritization signal, not cyclomatic or cognitive complexity.
118
+ - **Similarity, twins, wrappers, passthroughs, stale abstractions, extraction,
119
+ frontend pressure, and hidden coupling:** the emitted evidence must match
120
+ the code, but actionability is an investigation verdict until the
121
+ detector-language pair earns certification.
122
+
123
+ ## Roadmap
124
+
125
+ ### Phase 0 — Freeze the calibration and publication schemas
126
+
127
+ - Define the machine-readable calibration manifest, reviewed-row schema,
128
+ detector truth rule, verdict vocabulary, and noise-archetype vocabulary.
129
+ - Record repository, commit, tool version, detector version, language,
130
+ framework cohort, evidence provider, capability state, finding ID, file and
131
+ line, exclusion reasons, reviewer verdict, checker result, and review notes.
132
+ - Add Wilson confidence intervals and per-detector-language certification
133
+ status. Do not calculate one global accuracy number as the primary result.
134
+
135
+ Exit: the same reviewed rows always produce the same precision, confidence,
136
+ and certification decision.
137
+
138
+ ### Phase 1 — Extend the real-repository calibration harness
139
+
140
+ The current `scripts/accuracy-calibration.mjs` harness proves selected
141
+ navigation answers against known source facts and records timing. Extend it
142
+ without making local repositories a normal test dependency.
143
+
144
+ - Run health detectors in full, uncapped mode against temporary indexes.
145
+ - Select deterministic random and stratified samples instead of only top
146
+ findings.
147
+ - Produce a review ledger with `valid`, `invalid`, and `uncertain` verdicts.
148
+ - Preserve candidate evidence and source locations needed for code review.
149
+ - Group invalid findings by named noise archetype.
150
+ - Support a fresh holdout sample that was not used while fixing the detector.
151
+ - Replay selected historical commits and temporary known-positive mutations
152
+ in isolated worktrees for recall.
153
+
154
+ Exit: one command can generate a reproducible calibration packet without
155
+ modifying any corpus repository.
156
+
157
+ ### Phase 2 — Certify TypeScript dead-code findings
158
+
159
+ - Review approximately 25 deterministic candidates each from Vega_2.0,
160
+ openwork, Stable_Management, and traceroot for a 100-row baseline.
161
+ - Verify public exports, ambient declarations, decorators, framework
162
+ registration, workspace/barrel consumers, generated entry points, and test
163
+ boundaries.
164
+ - Fix shared false-positive archetypes rather than suppressing individual
165
+ examples or raising thresholds until output looks clean.
166
+ - Re-run the original sample and a fresh holdout sample.
167
+ - Exercise known positives from historical deletions or isolated worktree
168
+ mutations to establish recall coverage.
169
+
170
+ Exit: TypeScript dead code is certified, qualified, or explicitly retained as
171
+ experimental with every failed criterion recorded.
172
+
173
+ ### Phase 3 — Certify TypeScript measurements and remaining detector families
174
+
175
+ Work in evidence order:
176
+
177
+ 1. cycles, exact duplicates, raw complexity measurements, unused imports, and
178
+ unused parameters;
179
+ 2. recent duplicates, similarity, co-change, and doc drift;
180
+ 3. twins, wrappers, passthroughs, stale abstractions, extraction candidates,
181
+ React/Vue duplication and pressure, and hidden coupling.
182
+
183
+ Detectors below 90% remain absent from public output even when useful during
184
+ interactive exploration.
185
+
186
+ Exit: every TypeScript health family has a current certification state and a
187
+ published list of unresolved noise archetypes.
188
+
189
+ ### Phase 4 — Certify Rust and replace silent bias with visible applicability
190
+
191
+ Completed 2026-07-11 with qualified, insufficient, or unsupported verdicts;
192
+ no Rust row met the evidence threshold for promotion to certified. The renewal
193
+ used three pinned broad-replay repositories plus the prior six-repository dead
194
+ expansion, corrected seven trait/convention/import/signature advice archetypes,
195
+ and retained ordinary relationships on an untouched Vega holdout. Evidence and
196
+ per-command boundaries are recorded in
197
+ [`2026-07-11-rust-detector-certification.md`](./validation/2026-07-11-rust-detector-certification.md).
198
+
199
+ - Calibrate on openai/codex and SynthRunnerRust, then add at least one more
200
+ public Rust repository before awarding certified status.
201
+ - Measure trait, trait-implementation, derive, attribute-macro, test,
202
+ generated-code, async-runtime, ABI/export, and public-library behavior.
203
+ - Replace silent broad exclusions with reported applicability states and
204
+ exclusion counts wherever the tool cannot prove runtime reachability.
205
+ - Prefer rust-analyzer or SCIP evidence over framework-name rules. A
206
+ conservative exclusion may protect precision, but it must not make a
207
+ macro-heavy repository appear objectively healthier through invisible
208
+ omissions.
209
+ - Verify candidate deletion with `cargo check` when available and report when
210
+ build scripts, features, targets, or missing dependencies limit coverage.
211
+
212
+ Exit: each Rust detector is certified, qualified, experimental, unsupported,
213
+ or insufficiently evidenced, with exclusion coverage visible.
214
+
215
+ ### Phase 5 — Build and certify Python semantic and framework coverage
216
+
217
+ Status: Complete with explicit withholding. The current implementation has a
218
+ Python indexer and source facts but no Python semantic provider or type-aware
219
+ checker. The three-repository audit corrected dunder, transitive model,
220
+ protocol-hook, and runtime export false positives, then assigned every Python
221
+ row a qualified, insufficient, or unsupported verdict. Evidence:
222
+ [`2026-07-11-python-detector-certification.md`](./validation/2026-07-11-python-detector-certification.md).
223
+
224
+ - Establish the current indexing, semantic, and checker capability separately
225
+ for each corpus repository.
226
+ - Add or integrate semantic reference evidence before making strong dead-code
227
+ claims.
228
+ - Model decorator and registration behavior for Django, FastAPI, Flask,
229
+ pytest fixtures, Click/Typer, dataclasses, and Pydantic as corpus evidence
230
+ requires.
231
+ - Treat unsupported semantic or checker paths as `not analyzed`, not clean.
232
+ - Calibrate on scip-python, traceroot, and at least one additional public
233
+ Python repository with a different framework shape.
234
+
235
+ Exit: Python findings meet the same thresholds as TypeScript and Rust, or the
236
+ public capability matrix clearly withholds unsupported families.
237
+
238
+ ### Phase 6 — Replace the leaderboard score with certified evidence views
239
+
240
+ - Publish per-language and per-framework finding families, coverage, and
241
+ certification status before considering a composite score.
242
+ - Normalize only within comparable cohorts and applicable production code.
243
+ - Version methodology and retain historical results under the detector version
244
+ that produced them.
245
+ - Ensure health actions use calibrated language: facts, candidates, signals,
246
+ and unsupported analyses cannot share defect wording.
247
+
248
+ Exit: two repositories cannot change relative rank merely because one uses a
249
+ framework-specific detector or contains more excluded/test/generated symbols.
250
+
251
+ ### Phase 7 — Run a private cloud shadow leaderboard
252
+
253
+ - Analyze immutable clean checkouts for new commits with reproducible
254
+ dependency setup or an explicit degraded-mode disclosure.
255
+ - Reuse parent-commit indexes and semantic evidence through the completed
256
+ automatic incremental indexing system.
257
+ - Run for several weeks across diverse public repositories, auditing top
258
+ findings, commit-to-commit stability, coverage changes, and detector-version
259
+ rank changes.
260
+ - Publish only after certified detector output remains stable and
261
+ reproducible; keep experimental detectors private.
262
+
263
+ Exit: the service can reproduce any result from repository commit, analysis
264
+ version, configuration, capability record, and preserved raw output.
265
+
266
+ ## Completed Foundation
267
+
268
+ The previous accuracy-hardening program established:
269
+
270
+ - source-backed fixture/oracle checks for navigation commands;
271
+ - an optional real-repository calibration script using temporary caches;
272
+ - fail-closed mixed-language indexing with explicit partial opt-in;
273
+ - Rust path-qualified lookup and call-graph regression coverage;
274
+ - candidate/disclaimer language for heuristic commands; and
275
+ - optional timing and index-size metadata.
276
+
277
+ Those checks prove that selected commands retrieve expected source and graph
278
+ facts. They do not certify health-finding precision or recall, which is the
279
+ active program above.
280
+
281
+ ## Program Closure
282
+
283
+ Language detectors, composite workflows, navigation answers, and operational
284
+ state transitions have terminal verdicts under their current capabilities.
285
+ The final full repository verification passed, and certified evidence views are
286
+ ready for a capability-aware private cloud shadow. The aggregate health score
287
+ remains experimental and must not drive a public cross-language leaderboard.
288
+
289
+ ## Progress — 2026-07-10
290
+
291
+ The first TypeScript dead-code baseline is complete:
292
+ [`2026-07-10-typescript-dead-certification-baseline.md`](./validation/2026-07-10-typescript-dead-certification-baseline.md).
293
+ Across 78 deterministic rows from Vega_2.0, openwork, Stable_Management, and
294
+ traceroot, 25 were valid and 53 invalid: 32.1% observed precision with a 95%
295
+ Wilson interval of 22.7%–43.0%. The detector remains experimental.
296
+
297
+ Those five causes were fixed, and a second review exposed React lifecycle,
298
+ implemented-protocol, and nested Next.js proxy roots. The final certificate is
299
+ recorded in
300
+ [`2026-07-10-typescript-dead-certification.md`](./validation/2026-07-10-typescript-dead-certification.md).
301
+ The final fixed-seed sample contained 43 valid findings and zero invalid
302
+ findings across four repositories: 100% observed precision with a 91.8% 95%
303
+ Wilson lower bound, plus three positive-recall fixtures. TypeScript `dead` is
304
+ therefore certified under its repository-dead truth rule. This does not certify
305
+ Rust, Python, other TypeScript detectors, or the aggregate health score.
306
+
307
+ The Rust `dead` baseline and hardening replay are also complete:
308
+ [`2026-07-10-rust-dead-certification-baseline.md`](./validation/2026-07-10-rust-dead-certification-baseline.md)
309
+ and
310
+ [`2026-07-10-rust-dead-certification.md`](./validation/2026-07-10-rust-dead-certification.md).
311
+ The baseline found 1 valid and 51 invalid rows. Cargo library rooting and an
312
+ explicit `implicit-usage` signal tier removed every reviewed false positive;
313
+ the pinned replay produced three valid findings and zero invalid findings.
314
+ That is 100% observed precision, but the 43.8% Wilson lower bound and
315
+ one-repository finding sample are too small for certification. Rust `dead`
316
+ therefore remains insufficiently evidenced and must not be presented as a
317
+ public actionable metric. Three positive fixtures protect private-library and
318
+ binary-only recall while the corpus is expanded.
319
+
320
+ The TypeScript factual-detector campaign is complete:
321
+ [`2026-07-10-typescript-factual-detectors.md`](./validation/2026-07-10-typescript-factual-detectors.md).
322
+ `unused-imports` (59/59), `duplicate-bodies` (40/40), raw `complexity`
323
+ (40/40), and `redundant-reexports` (40/40) are certified under their narrow
324
+ fact rules. `unused-params`, `cycles`, `isolated`, `not-implemented`,
325
+ `decorative-checkers`, and `test-quality` were audited and hardened but remain
326
+ insufficiently evidenced because their surviving real-repository frames lack
327
+ the required row count or repository breadth. The campaign removed import-use,
328
+ ambient dependency-edge, framework-contract, re-export-attribution,
329
+ delegated-checker, and implicit-test-assertion false-positive archetypes.
330
+
331
+ The TypeScript general-similarity campaign is also complete:
332
+ [`2026-07-10-typescript-similarity-detectors.md`](./validation/2026-07-10-typescript-similarity-detectors.md).
333
+ The `similar`, `similar-files`, and `similar-signatures` relationship claims are
334
+ certified at 36/36, 40/40, and 40/40 valid rows across four repositories.
335
+ `similar-chains` is qualified because its correct 40/40 sample came from a
336
+ bounded 500-path input frame, and none of its sampled consolidation advice was
337
+ actionable. `twin-drift` is qualified at 37/40 (92.5%) after constant,
338
+ near-name, test-helper, and convention-only filtering. `recent-duplicates` is
339
+ 8/8 after generic React-plumbing hardening but remains insufficiently
340
+ evidenced. The campaign now records factual relationship validity separately
341
+ from recommendation utility.
342
+
343
+ The initial TypeScript architecture/history campaign is complete:
344
+ [`2026-07-10-typescript-architecture-history-detectors.md`](./validation/2026-07-10-typescript-architecture-history-detectors.md).
345
+ The `co-change`, `doc-drift`, and `stale-abstractions` relationship claims are
346
+ certified at 40/40 valid rows across four repositories. `drift` is qualified at
347
+ 40/40 because the population was 39 pattern deviations, one inferred layer,
348
+ and a supported zero for unused-import rows; none of the sampled drift advice
349
+ was actionable without ownership evidence. The initial packet qualified
350
+ `wrapper-candidates` at 30/30 and left `passthrough-candidates` insufficient at
351
+ 21/21. The 2026-07-11 renewal expanded those samples to 60/60 and 41/41 valid
352
+ relationships across three repositories, certifying both narrow facts. Removal
353
+ utility remains contextual. The campaign removed ambient-declaration,
354
+ generated co-change state, and structural-token feature-label noise.
355
+
356
+ The initial TypeScript extraction and graph-risk campaign is complete:
357
+ [`2026-07-10-typescript-extraction-graph-risk-detectors.md`](./validation/2026-07-10-typescript-extraction-graph-risk-detectors.md).
358
+ `extract-candidates`, `locality-candidates`, `coupling`, `deep-chains`,
359
+ `complexity-hotspots`, `hotspots`, `fan-in`, and `fan-out` are certified on
360
+ 43-48 valid sampled rows across four repositories. `bottlenecks` is qualified
361
+ at 33/33 because its 89.6% confidence floor remained just below the certified
362
+ threshold. The 69,942-row uncapped frame exposed and removed ambiguous
363
+ shortened fan-in identities and cycle-inflated dependency depth. These results
364
+ certify the disclosed measurements, not automatic refactoring actionability.
365
+
366
+ The bottleneck population renewal is complete:
367
+ [`2026-07-11-typescript-bottlenecks-certification.md`](./validation/2026-07-11-typescript-bottlenecks-certification.md).
368
+ The public result now discloses the deterministic caller-file and
369
+ external-callee sets behind its totals. All 39 reviewed rows passed arithmetic
370
+ and pinned-source checks across four repositories, producing a 91.0% Wilson
371
+ lower bound and certifying the narrow graph fact. Centrality remains a review
372
+ signal rather than evidence that refactoring is beneficial.
373
+
374
+ The React and Vue framework-detector campaign is complete:
375
+ [`2026-07-11-react-vue-detectors.md`](./validation/2026-07-11-react-vue-detectors.md).
376
+ React component duplication, hook candidates, and component pressure are each
377
+ certified at 48/48 valid rows across three repositories. Vue component
378
+ duplication is certified at 41/41 across four repositories, composable
379
+ candidates at 48/48 across three repositories with findings, and view pressure
380
+ at 64/64 across four repositories. The 54,586-row combined uncapped frame
381
+ removed unqualified dominant-pressure axes, nested-component route-page
382
+ misclassification, and unreproducible similarity-score totals. All sampled
383
+ refactoring advice remained non-actionable without local design context.
384
+
385
+ The same campaign's initial `augment-vue` runs reported 66,396 SFC mentions in
386
+ Stable_Management and 13,700 in on_main_mvp. Those totals established the
387
+ operational path but were not an accuracy certificate. Missing project files
388
+ no longer abort the whole run, and the reliable single-context resolver is the
389
+ default after parallel worker stalls reproduced at two, four, and eight
390
+ workers. Ordinary `reindex` does not run this optional augmentation.
391
+
392
+ The completed Vue reference-completeness audit is recorded in
393
+ [`2026-07-11-vue-reference-completeness.md`](./validation/2026-07-11-vue-reference-completeness.md).
394
+ It found that the earlier totals collapsed local SFC bindings onto component
395
+ symbols, attributed properties to nearby callables, and counted module-path
396
+ fragments. After hardening, an exact import/component oracle matched all 841
397
+ reviewed mentions across Stable_Management, on_main_mvp, and agent_chat. The
398
+ command is qualified for cross-file component identity; broader local and
399
+ property identity is withheld and explicitly unsupported.
400
+
401
+ The `twin-ab` generated-scaffold audit is complete:
402
+ [`2026-07-11-twin-ab-generated-scaffold.md`](./validation/2026-07-11-twin-ab-generated-scaffold.md).
403
+ Three pinned TypeScript repositories produced importable real-callable
404
+ scaffolds with zero scaffold-local compiler diagnostics. The audit found and
405
+ fixed a shared export-classification defect where an adjacent exported
406
+ declaration made a following private callable look exported; the two real
407
+ private pairs now refuse, and focused regression coverage preserves that
408
+ boundary.
409
+
410
+ The low-population TypeScript factual renewal is also complete:
411
+ [`2026-07-11-typescript-factual-evidence-renewal.md`](./validation/2026-07-11-typescript-factual-evidence-renewal.md).
412
+ All 25 emitted rows were valid across the four-repository replay, including four
413
+ newly reviewed mock-echo facts. `unused-params`, `cycles`, `isolated`,
414
+ `not-implemented`, `decorative-checkers`, and `test-quality` remain
415
+ insufficient because their natural populations still fail sample-size or
416
+ repository-breadth gates; they should be renewed only when a named new corpus
417
+ adds the missing population.
418
+
419
+ The wrapper/passthrough population renewal is complete:
420
+ [`2026-07-11-typescript-wrapper-passthrough-certification.md`](./validation/2026-07-11-typescript-wrapper-passthrough-certification.md).
421
+ Independent unbounded reference checks validated all 60 wrapper rows, and
422
+ source review validated all 41 passthrough rows. Both relationship claims are
423
+ now certified; automatic removal remains unsupported without local boundary
424
+ evidence.
425
+
426
+ The recent-duplicates population expansion is complete:
427
+ [`2026-07-11-typescript-recent-duplicates-expansion.md`](./validation/2026-07-11-typescript-recent-duplicates-expansion.md).
428
+ All eight natural findings passed pinned source, shared-evidence, and
429
+ file-addition-history checks across four repositories with findings. Two more
430
+ repositories produced supported zeros, and one repository's indexing failure
431
+ remained explicit rather than becoming a zero. The 67.6% confidence floor is
432
+ still insufficient, so renewal is terminal for this named corpus and should
433
+ resume only when another corpus or historical snapshot adds findings.
434
+
435
+ The widened Rust dead-code population replay is complete:
436
+ [`2026-07-11-rust-dead-expansion.md`](./validation/2026-07-11-rust-dead-expansion.md).
437
+ Five of six pinned repositories completed with indexing, source, semantic, and
438
+ checker capability; one indexing failure stayed explicit. The only three
439
+ natural findings again came from VegaAssistant and were all valid. Four
440
+ supported zeros cannot raise the 43.9% confidence floor, so Rust `dead` remains
441
+ insufficient and is terminal for this named corpus.
442
+
443
+ The Python detector campaign is complete under the current capability boundary:
444
+ [`2026-07-11-python-detector-certification.md`](./validation/2026-07-11-python-detector-certification.md).
445
+ `scip-python-plus` and source facts were available, but no Python semantic
446
+ provider or type-aware checker existed. The renewal corrected convention
447
+ dunders, transitive Pydantic model liveness, runtime protocol hooks, and
448
+ literal `__all__` export/import behavior. The Flask holdout retained 29
449
+ file-internal and two signature relationships while reporting zero
450
+ repository-dead rows. Every Python matrix row is now qualified, insufficient,
451
+ or unsupported; none is certified for public actionable scoring.
452
+
453
+ The composite workflow campaign is complete:
454
+ [`2026-07-11-composite-workflow-certification.md`](./validation/2026-07-11-composite-workflow-certification.md).
455
+ All 166 targeted cleanup, diff-gate, health, effectiveness, and impact probes
456
+ passed. A mixed Python replay exposed that health omitted capability state and
457
+ turned contextual relationships into action language. Health now carries the
458
+ live capability matrix, marks its score experimental and non-comparable across
459
+ languages, distinguishes syntax-only verification, and describes candidates as
460
+ investigations rather than automatic refactorings. Diff-gate and impact
461
+ workflows are qualified under their per-check evidence; only exact configured
462
+ coverage-contract comparison is certified in this slice.
463
+
464
+ The navigation and graph-answer campaign is complete:
465
+ [`2026-07-11-navigation-graph-certification.md`](./validation/2026-07-11-navigation-graph-certification.md).
466
+ All 107 exact-answer probes passed across 14 fixture files, and a live replay
467
+ kept definition ranges, imports, references, dependencies, calls, slices,
468
+ outlines, and counts mutually consistent. The audit corrected reference-only
469
+ imports, type owners, and module owners appearing as callable callers. It also
470
+ records that hierarchy is lexical, dataflow is reference-level, overload
471
+ selection and inheritance are unsupported, and semantic cross-language edges
472
+ exist only when an upstream indexer emits them.
473
+
474
+ ## Program Acceptance Criteria
475
+
476
+ - Every health detector-language pair has a visible certification state.
477
+ - Public actionable findings meet the certified threshold; public signals
478
+ meet the qualified threshold; lower-precision rows remain private.
479
+ - Precision samples span at least three repositories, and recall has
480
+ known-positive evidence.
481
+ - False positives are tracked as named archetypes with regression coverage,
482
+ not hidden by unreasoned suppressions.
483
+ - Unsupported and degraded analyses cannot appear as zero findings.
484
+ - Rust exclusions and framework-specific applicability are reported rather
485
+ than silently improving comparative results.
486
+ - Python does not make strong semantic health claims until the required
487
+ provider and checker evidence exists.
488
+ - Public comparisons are reproducible, versioned, coverage-aware, and limited
489
+ to comparable cohorts.
490
+ - `npm test`, `npm run typecheck`, and `npm run build` pass after each
491
+ implementation slice.
492
+ - Completed code or documentation changes pass `scip-query reindex` and
493
+ `scip-query diff-gate` before being declared finished.