kenaz 0.3.3 → 0.3.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (451) hide show
  1. package/core/.agents/skills/typesafe-ai/LICENSE +21 -21
  2. package/core/.agents/skills/typesafe-ai/SKILL.md +149 -149
  3. package/core/.claude/commands/Ansuz.md +15 -2
  4. package/core/.claude/hooks/__tests__/task-2395-project-registration.test.js +84 -0
  5. package/core/.claude/hooks/handoff-staleness.js +2 -2
  6. package/core/.claude/hooks/session-init.ps1 +49 -49
  7. package/core/.claude/hooks/turn-context.js +50 -2
  8. package/core/.claude/skills/ui-ux-pro-max/data/charts.csv +26 -26
  9. package/core/.claude/skills/ui-ux-pro-max/data/colors.csv +97 -97
  10. package/core/.claude/skills/ui-ux-pro-max/data/icons.csv +101 -101
  11. package/core/.claude/skills/ui-ux-pro-max/data/landing.csv +31 -31
  12. package/core/.claude/skills/ui-ux-pro-max/data/products.csv +96 -96
  13. package/core/.claude/skills/ui-ux-pro-max/data/react-performance.csv +45 -45
  14. package/core/.claude/skills/ui-ux-pro-max/data/stacks/astro.csv +54 -54
  15. package/core/.claude/skills/ui-ux-pro-max/data/stacks/flutter.csv +53 -53
  16. package/core/.claude/skills/ui-ux-pro-max/data/stacks/html-tailwind.csv +56 -56
  17. package/core/.claude/skills/ui-ux-pro-max/data/stacks/jetpack-compose.csv +53 -53
  18. package/core/.claude/skills/ui-ux-pro-max/data/stacks/nextjs.csv +53 -53
  19. package/core/.claude/skills/ui-ux-pro-max/data/stacks/nuxt-ui.csv +51 -51
  20. package/core/.claude/skills/ui-ux-pro-max/data/stacks/nuxtjs.csv +59 -59
  21. package/core/.claude/skills/ui-ux-pro-max/data/stacks/react-native.csv +52 -52
  22. package/core/.claude/skills/ui-ux-pro-max/data/stacks/react.csv +54 -54
  23. package/core/.claude/skills/ui-ux-pro-max/data/stacks/shadcn.csv +61 -61
  24. package/core/.claude/skills/ui-ux-pro-max/data/stacks/svelte.csv +54 -54
  25. package/core/.claude/skills/ui-ux-pro-max/data/stacks/swiftui.csv +51 -51
  26. package/core/.claude/skills/ui-ux-pro-max/data/stacks/vue.csv +50 -50
  27. package/core/.claude/skills/ui-ux-pro-max/data/styles.csv +68 -68
  28. package/core/.claude/skills/ui-ux-pro-max/data/typography.csv +57 -57
  29. package/core/.claude/skills/ui-ux-pro-max/data/ui-reasoning.csv +101 -101
  30. package/core/.claude/skills/ui-ux-pro-max/data/ux-guidelines.csv +99 -99
  31. package/core/.claude/skills/ui-ux-pro-max/data/web-interface.csv +31 -31
  32. package/core/.claude/skills/ui-ux-pro-max/scripts/core.py +253 -253
  33. package/core/.claude/skills/ui-ux-pro-max/scripts/design_system.py +1067 -1067
  34. package/core/.claude/skills/ui-ux-pro-max/scripts/search.py +106 -106
  35. package/core/.collaboration/core/task-api.js +1 -4
  36. package/core/.formation/README.md +21 -146
  37. package/core/.formation/bifrost/codex-doctor.js +37 -2
  38. package/core/.formation/bifrost/codex-doctor.test.js +18 -0
  39. package/core/.formation/bifrost/launch.js +1 -1
  40. package/core/.formation/bifrost/openai/codex-hook-shim.js +6 -0
  41. package/core/.formation/bifrost/openai/codex-hook-shim.test.js +11 -1
  42. package/core/.formation/bifrost/setup-codex.js +3 -0
  43. package/core/.formation/native/claude/README.md +1 -1
  44. package/core/.formation/role-map.yaml +3 -3
  45. package/core/.formation/spec/task_operations.parity.test.js +0 -1
  46. package/core/.formation/spec/task_operations.yaml +3 -3
  47. package/core/.formation/tests/hello.txt +2 -2
  48. package/core/.formation/tools/__tests__/subagent-bifrost-guard.test.js +4 -4
  49. package/core/.formation/tools/dashboard-models-chain.test.js +0 -6
  50. package/core/.formation/tools/dispatch-detail/cross-project-worktree.md +1 -9
  51. package/core/.formation/tools/dispatch-detail/worktree-subagent-mode.md +2 -2
  52. package/core/.formation/tools/dispatch.js +33 -554
  53. package/core/.formation/tools/mcp-allowlist.test.js +0 -1
  54. package/core/.formation/tools/post-complete/task2345.test.js +0 -5
  55. package/core/.formation/tools/pre-dispatch-snapshot.test.js +0 -100
  56. package/core/.formation/tools/prompt.js +33 -1623
  57. package/core/.formation/tools/prompt.test.js +26 -594
  58. package/core/.formation/unit/do.yaml +2 -4
  59. package/core/.kenaz/knowledge/ladybug_2c51da1e_shadow_file_frame_group.patch +163 -163
  60. package/core/.kenaz/knowledge/ladybug_30cf85690e_shadow_page_read_during_checkpoint.patch +393 -393
  61. package/core/.kenaz/knowledge/ladybug_7a17858d3a_node_group_delete_lock.patch +102 -102
  62. package/core/.kenaz/knowledge/ladybug_926a0ba1_column_checkpoint.patch +412 -412
  63. package/core/.kenaz/knowledge/ladybug_a346c4c4b6_pk_index_header_pages.patch +144 -144
  64. package/core/.kenaz/knowledge/ladybug_ba5f38815b_csr_node_group.patch +86 -86
  65. package/core/.kenaz/muninn_episode_write.ps1 +17 -17
  66. package/core/.kenaz/specialist/field_report_watermark.json +1 -1
  67. package/core/.kenaz/specialist/memory_guard_fired.json +1 -1
  68. package/core/.kenaz/specialist/memory_guard_trigger_cache.json +1 -1
  69. package/core/.kenaz/specialist/t1777_probe.txt +1 -1
  70. package/core/.matrix/index.matrix +5 -5
  71. package/core/.rule/general/agent-workflow.yaml +4 -4
  72. package/core/.rule/general/file-size-threshold.yaml +16 -4
  73. package/core/.rule/general/rust-verify-target.yaml +0 -1
  74. package/core/.rule/general/system-invariants.yaml +8 -25
  75. package/core/.specialist/Ansuz/SYSTEM_DOC.md +17 -42
  76. package/core/.specialist/Ansuz/capability.yaml +7 -14
  77. package/core/.specialist/Ansuz/concepts/formation_quick_ref.yaml +4 -11
  78. package/core/.specialist/Ansuz/elixir.yaml +3 -4
  79. package/core/.specialist/Ansuz/identity_contract.yaml +1 -1
  80. package/core/.specialist/Ansuz/knowledge.yaml +7 -10
  81. package/core/.specialist/Ansuz/rule.yaml +2 -2
  82. package/core/.specialist/Ansuz/tools/verify-and-land.js +22 -0
  83. package/core/.specialist/Ansuz/tools/verify-and-land.test.js +59 -5
  84. package/core/.specialist/Ansuz/workflow.yaml +7 -11
  85. package/core/.specialist/Forseti/core.yaml +1 -1
  86. package/core/.specialist/Heimdall/SKILL.md +1 -1
  87. package/core/.specialist/Heimdall/core.yaml +2 -2
  88. package/core/.specialist/Huginn/rule.yaml +1 -1
  89. package/core/.specialist/tools/README.md +2 -2
  90. package/core/.specialist/tools/__tests__/freya-cli-contract.test.js +0 -1
  91. package/core/.system/collaboration/README.md +1 -1
  92. package/core/.system/formation/LEGACY_SKILLS.md +2 -0
  93. package/core/.system/formation/POST_COMPLETE.md +2 -0
  94. package/core/.system/formation/README.md +2 -0
  95. package/core/KENAZ_CORE_VERSION +1 -1
  96. package/core/dev/ANSUZ_STARTUP_INJECTION_REVIEW.md +355 -355
  97. package/core/dev/JEV_INTEGRATION_PLAN.md +189 -189
  98. package/core/dev/design/matrix_format_EXAMPLE_memory.matrix +13 -13
  99. package/core/dev/design/matrix_format_EXAMPLE_project_status.matrix +9 -9
  100. package/core/dev/design/matrix_format_EXAMPLE_tasks.matrix +14 -14
  101. package/core/dev/design/matrix_v2_EXAMPLE.matrix +67 -67
  102. package/core/dev/jev-eval/ACTIVE_CORRECTION.md +52 -52
  103. package/core/dev/jev-eval/CONTEXT_REVIEW.md +31 -31
  104. package/core/dev/jev-eval/EVALUATION_STATUS.md +39 -39
  105. package/core/dev/jev-eval/HISTORICAL_CORPUS.md +33 -33
  106. package/core/dev/jev-eval/README.md +99 -99
  107. package/core/dev/jev-eval/THALAMUS_RANKING.md +58 -58
  108. package/core/dev/jev-eval/reports/annotation-pilot-2026-09-20.json +32 -32
  109. package/core/dev/jev-eval/reports/context-review-2026-09-20.json +48 -48
  110. package/core/dev/jev-eval/reports/history-corpus-2026-09-20.json +52 -52
  111. package/core/dev/pipeline/SELF_ENHANCEMENT_STATUS.json.bak +47 -47
  112. package/core/dev/pipeline/tasks/.gitkeep +1 -1
  113. package/core/dev/pipeline/test_results/TASK_TEST_PARALLEL_002_result.txt +27 -27
  114. package/core/dev/scripts/check-file-size.js +44 -17
  115. package/core/dev/scripts/init_git.sh +0 -0
  116. package/core/dev/scripts/init_project.sh +0 -0
  117. package/core/dev/scripts/legacy/dev_agent_scheduler.ps1.bak +536 -536
  118. package/core/dev/scripts/legacy/multi_agent_scheduler.ps1.bak +1978 -1978
  119. package/core/dev/scripts/matrix/src/cli/commands/list-tags.ts.wip +140 -140
  120. package/core/dev/scripts/matrix/src/cli/commands/manage-tags.ts.wip +173 -173
  121. package/core/dev/scripts/matrix/src/cli/commands/stats.ts.wip +168 -168
  122. package/core/dev/scripts/matrix/src/cli/commands/tokens.ts.wip +126 -126
  123. package/core/dev/scripts/multi_agent_scheduler.sh +0 -0
  124. package/core/dev/scripts/run_once.sh +0 -0
  125. package/core/dev/scripts/setup_mcp.sh +0 -0
  126. package/core/dev/scripts/setup_rag.sh +0 -0
  127. package/core/dev/scripts/start_dev_agent.sh +0 -0
  128. package/core/dev/scripts/start_multi_agent.sh +0 -0
  129. package/core/dev/scripts/sync_task_queue.sh +0 -0
  130. package/core/k-cli/README.md +1 -94
  131. package/core/k-cli/test/TEST_SPEC.md +5 -22
  132. package/core/k-cli/test/run.sh +1 -67
  133. package/core/plugins/kenaz/commands/Amelia.md +1 -1
  134. package/core/plugins/kenaz/commands/Ansuz.md +16 -3
  135. package/core/plugins/kenaz/scripts/kenaz-mode.js +11 -1
  136. package/core/plugins/kenaz/scripts/sync.js +6 -2
  137. package/core/plugins/kenaz/scripts/ygg.js +13 -1
  138. package/core/plugins/kenaz/skills/product_manager/SKILL.md +12 -12
  139. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/charts.csv +26 -26
  140. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/colors.csv +97 -97
  141. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/icons.csv +101 -101
  142. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/landing.csv +31 -31
  143. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/products.csv +96 -96
  144. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/react-performance.csv +45 -45
  145. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/astro.csv +54 -54
  146. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/flutter.csv +53 -53
  147. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/html-tailwind.csv +56 -56
  148. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/jetpack-compose.csv +53 -53
  149. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/nextjs.csv +53 -53
  150. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/nuxt-ui.csv +51 -51
  151. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/nuxtjs.csv +59 -59
  152. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/react-native.csv +52 -52
  153. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/react.csv +54 -54
  154. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/shadcn.csv +61 -61
  155. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/svelte.csv +54 -54
  156. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/swiftui.csv +51 -51
  157. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/stacks/vue.csv +50 -50
  158. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/styles.csv +68 -68
  159. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/typography.csv +57 -57
  160. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/ui-reasoning.csv +101 -101
  161. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/ux-guidelines.csv +99 -99
  162. package/core/plugins/kenaz/skills/ui-ux-pro-max/data/web-interface.csv +31 -31
  163. package/core/plugins/kenaz/skills/ui-ux-pro-max/scripts/core.py +253 -253
  164. package/core/plugins/kenaz/skills/ui-ux-pro-max/scripts/design_system.py +1067 -1067
  165. package/core/plugins/kenaz/skills/ui-ux-pro-max/scripts/search.py +106 -106
  166. package/lib/cli.js +11 -4
  167. package/package.json +11 -7
  168. package/vendor/install-kenaz-cli.js +18 -7
  169. package/core/.formation/legion/forge.yaml +0 -299
  170. package/core/.formation/legion/hive.yaml +0 -238
  171. package/core/.formation/squad/blitz.yaml +0 -192
  172. package/core/.formation/squad/strategy-tribunal.yaml +0 -354
  173. package/core/.formation/squad/strategy.yaml +0 -349
  174. package/core/.formation/squad/sweep.yaml +0 -208
  175. package/core/.formation/tools/broadcast.js +0 -103
  176. package/core/.formation/tools/dispatch-dryrun.test.js +0 -54
  177. package/core/.formation/tools/dispatch-notifier.js +0 -84
  178. package/core/.formation/tools/formations.js +0 -1203
  179. package/core/.formation/tools/headless-branch-recovery.test.js +0 -64
  180. package/core/.formation/tools/headless-dryrun.test.js +0 -79
  181. package/core/.formation/tools/headless-v2.js +0 -659
  182. package/core/.formation/tools/headless.js +0 -1071
  183. package/core/.formation/tools/legacy-scheduler.js +0 -1994
  184. package/core/.formation/tools/logger.js +0 -273
  185. package/core/.formation/tools/memory-loader.js +0 -422
  186. package/core/.formation/tools/process-state.js +0 -33
  187. package/core/.formation/tools/remote-conflict.js +0 -98
  188. package/core/.formation/tools/runner.js +0 -1332
  189. package/core/.formation/tools/task-utils.js +0 -344
  190. package/core/.formation/tools/task2028-headless-worktree.test.js +0 -112
  191. package/core/.formation/tools/tests/headless-mcp-args-plumbing.test.js +0 -123
  192. package/core/.formation/tools/worktree-leak.test.js +0 -162
  193. package/core/k-cli/dist/alaya/run.js +0 -91
  194. package/core/k-cli/dist/core/output.js +0 -24
  195. package/core/k-cli/dist/core/process.js +0 -38
  196. package/core/k-cli/dist/core/project.js +0 -80
  197. package/core/k-cli/dist/core/spawn.js +0 -268
  198. package/core/k-cli/dist/core/types.js +0 -2
  199. package/core/k-cli/dist/dashboard/update.js +0 -53
  200. package/core/k-cli/dist/dashboard/version.js +0 -20
  201. package/core/k-cli/dist/dispatch/run.js +0 -241
  202. package/core/k-cli/dist/dispatch/status.js +0 -103
  203. package/core/k-cli/dist/index.js +0 -283
  204. package/core/k-cli/dist/project/initialize.js +0 -301
  205. package/core/k-cli/dist/project/install.js +0 -89
  206. package/core/k-cli/dist/project/status.js +0 -138
  207. package/core/k-cli/dist/specialist/list.js +0 -100
  208. package/core/k-cli/dist/specialist/run.js +0 -93
  209. package/core/k-cli/dist/task/create.js +0 -194
  210. package/core/k-cli/dist/task/delete.js +0 -98
  211. package/core/k-cli/dist/task/get.js +0 -55
  212. package/core/k-cli/dist/task/list.js +0 -58
  213. package/core/k-cli/dist/task/submit.js +0 -156
  214. package/core/k-cli/dist/ygg/run.js +0 -51
  215. package/core/k-cli/node_modules/.bin/yaml +0 -16
  216. package/core/k-cli/node_modules/.bin/yaml.cmd +0 -17
  217. package/core/k-cli/node_modules/.bin/yaml.ps1 +0 -28
  218. package/core/k-cli/node_modules/.package-lock.json +0 -23
  219. package/core/k-cli/node_modules/yaml/LICENSE +0 -13
  220. package/core/k-cli/node_modules/yaml/README.md +0 -172
  221. package/core/k-cli/node_modules/yaml/bin.mjs +0 -11
  222. package/core/k-cli/node_modules/yaml/browser/dist/compose/compose-collection.js +0 -88
  223. package/core/k-cli/node_modules/yaml/browser/dist/compose/compose-doc.js +0 -43
  224. package/core/k-cli/node_modules/yaml/browser/dist/compose/compose-node.js +0 -109
  225. package/core/k-cli/node_modules/yaml/browser/dist/compose/compose-scalar.js +0 -86
  226. package/core/k-cli/node_modules/yaml/browser/dist/compose/composer.js +0 -217
  227. package/core/k-cli/node_modules/yaml/browser/dist/compose/resolve-block-map.js +0 -115
  228. package/core/k-cli/node_modules/yaml/browser/dist/compose/resolve-block-scalar.js +0 -198
  229. package/core/k-cli/node_modules/yaml/browser/dist/compose/resolve-block-seq.js +0 -49
  230. package/core/k-cli/node_modules/yaml/browser/dist/compose/resolve-end.js +0 -37
  231. package/core/k-cli/node_modules/yaml/browser/dist/compose/resolve-flow-collection.js +0 -207
  232. package/core/k-cli/node_modules/yaml/browser/dist/compose/resolve-flow-scalar.js +0 -223
  233. package/core/k-cli/node_modules/yaml/browser/dist/compose/resolve-props.js +0 -146
  234. package/core/k-cli/node_modules/yaml/browser/dist/compose/util-contains-newline.js +0 -34
  235. package/core/k-cli/node_modules/yaml/browser/dist/compose/util-empty-scalar-position.js +0 -26
  236. package/core/k-cli/node_modules/yaml/browser/dist/compose/util-flow-indent-check.js +0 -15
  237. package/core/k-cli/node_modules/yaml/browser/dist/compose/util-map-includes.js +0 -13
  238. package/core/k-cli/node_modules/yaml/browser/dist/doc/Document.js +0 -335
  239. package/core/k-cli/node_modules/yaml/browser/dist/doc/anchors.js +0 -71
  240. package/core/k-cli/node_modules/yaml/browser/dist/doc/applyReviver.js +0 -55
  241. package/core/k-cli/node_modules/yaml/browser/dist/doc/createNode.js +0 -88
  242. package/core/k-cli/node_modules/yaml/browser/dist/doc/directives.js +0 -176
  243. package/core/k-cli/node_modules/yaml/browser/dist/errors.js +0 -57
  244. package/core/k-cli/node_modules/yaml/browser/dist/index.js +0 -17
  245. package/core/k-cli/node_modules/yaml/browser/dist/log.js +0 -11
  246. package/core/k-cli/node_modules/yaml/browser/dist/nodes/Alias.js +0 -114
  247. package/core/k-cli/node_modules/yaml/browser/dist/nodes/Collection.js +0 -147
  248. package/core/k-cli/node_modules/yaml/browser/dist/nodes/Node.js +0 -38
  249. package/core/k-cli/node_modules/yaml/browser/dist/nodes/Pair.js +0 -36
  250. package/core/k-cli/node_modules/yaml/browser/dist/nodes/Scalar.js +0 -24
  251. package/core/k-cli/node_modules/yaml/browser/dist/nodes/YAMLMap.js +0 -144
  252. package/core/k-cli/node_modules/yaml/browser/dist/nodes/YAMLSeq.js +0 -113
  253. package/core/k-cli/node_modules/yaml/browser/dist/nodes/addPairToJSMap.js +0 -63
  254. package/core/k-cli/node_modules/yaml/browser/dist/nodes/identity.js +0 -36
  255. package/core/k-cli/node_modules/yaml/browser/dist/nodes/toJS.js +0 -37
  256. package/core/k-cli/node_modules/yaml/browser/dist/parse/cst-scalar.js +0 -214
  257. package/core/k-cli/node_modules/yaml/browser/dist/parse/cst-stringify.js +0 -61
  258. package/core/k-cli/node_modules/yaml/browser/dist/parse/cst-visit.js +0 -97
  259. package/core/k-cli/node_modules/yaml/browser/dist/parse/cst.js +0 -98
  260. package/core/k-cli/node_modules/yaml/browser/dist/parse/lexer.js +0 -717
  261. package/core/k-cli/node_modules/yaml/browser/dist/parse/line-counter.js +0 -39
  262. package/core/k-cli/node_modules/yaml/browser/dist/parse/parser.js +0 -967
  263. package/core/k-cli/node_modules/yaml/browser/dist/public-api.js +0 -102
  264. package/core/k-cli/node_modules/yaml/browser/dist/schema/Schema.js +0 -37
  265. package/core/k-cli/node_modules/yaml/browser/dist/schema/common/map.js +0 -17
  266. package/core/k-cli/node_modules/yaml/browser/dist/schema/common/null.js +0 -15
  267. package/core/k-cli/node_modules/yaml/browser/dist/schema/common/seq.js +0 -17
  268. package/core/k-cli/node_modules/yaml/browser/dist/schema/common/string.js +0 -14
  269. package/core/k-cli/node_modules/yaml/browser/dist/schema/core/bool.js +0 -19
  270. package/core/k-cli/node_modules/yaml/browser/dist/schema/core/float.js +0 -43
  271. package/core/k-cli/node_modules/yaml/browser/dist/schema/core/int.js +0 -38
  272. package/core/k-cli/node_modules/yaml/browser/dist/schema/core/schema.js +0 -23
  273. package/core/k-cli/node_modules/yaml/browser/dist/schema/json/schema.js +0 -62
  274. package/core/k-cli/node_modules/yaml/browser/dist/schema/tags.js +0 -96
  275. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/binary.js +0 -58
  276. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/bool.js +0 -26
  277. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/float.js +0 -46
  278. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/int.js +0 -71
  279. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/merge.js +0 -64
  280. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/omap.js +0 -74
  281. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/pairs.js +0 -78
  282. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/schema.js +0 -39
  283. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/set.js +0 -93
  284. package/core/k-cli/node_modules/yaml/browser/dist/schema/yaml-1.1/timestamp.js +0 -101
  285. package/core/k-cli/node_modules/yaml/browser/dist/stringify/foldFlowLines.js +0 -146
  286. package/core/k-cli/node_modules/yaml/browser/dist/stringify/stringify.js +0 -129
  287. package/core/k-cli/node_modules/yaml/browser/dist/stringify/stringifyCollection.js +0 -153
  288. package/core/k-cli/node_modules/yaml/browser/dist/stringify/stringifyComment.js +0 -20
  289. package/core/k-cli/node_modules/yaml/browser/dist/stringify/stringifyDocument.js +0 -85
  290. package/core/k-cli/node_modules/yaml/browser/dist/stringify/stringifyNumber.js +0 -24
  291. package/core/k-cli/node_modules/yaml/browser/dist/stringify/stringifyPair.js +0 -150
  292. package/core/k-cli/node_modules/yaml/browser/dist/stringify/stringifyString.js +0 -336
  293. package/core/k-cli/node_modules/yaml/browser/dist/util.js +0 -11
  294. package/core/k-cli/node_modules/yaml/browser/dist/visit.js +0 -233
  295. package/core/k-cli/node_modules/yaml/browser/index.js +0 -5
  296. package/core/k-cli/node_modules/yaml/browser/package.json +0 -3
  297. package/core/k-cli/node_modules/yaml/dist/cli.d.ts +0 -8
  298. package/core/k-cli/node_modules/yaml/dist/cli.mjs +0 -201
  299. package/core/k-cli/node_modules/yaml/dist/compose/compose-collection.d.ts +0 -11
  300. package/core/k-cli/node_modules/yaml/dist/compose/compose-collection.js +0 -90
  301. package/core/k-cli/node_modules/yaml/dist/compose/compose-doc.d.ts +0 -7
  302. package/core/k-cli/node_modules/yaml/dist/compose/compose-doc.js +0 -45
  303. package/core/k-cli/node_modules/yaml/dist/compose/compose-node.d.ts +0 -29
  304. package/core/k-cli/node_modules/yaml/dist/compose/compose-node.js +0 -112
  305. package/core/k-cli/node_modules/yaml/dist/compose/compose-scalar.d.ts +0 -5
  306. package/core/k-cli/node_modules/yaml/dist/compose/compose-scalar.js +0 -88
  307. package/core/k-cli/node_modules/yaml/dist/compose/composer.d.ts +0 -63
  308. package/core/k-cli/node_modules/yaml/dist/compose/composer.js +0 -222
  309. package/core/k-cli/node_modules/yaml/dist/compose/resolve-block-map.d.ts +0 -6
  310. package/core/k-cli/node_modules/yaml/dist/compose/resolve-block-map.js +0 -117
  311. package/core/k-cli/node_modules/yaml/dist/compose/resolve-block-scalar.d.ts +0 -11
  312. package/core/k-cli/node_modules/yaml/dist/compose/resolve-block-scalar.js +0 -200
  313. package/core/k-cli/node_modules/yaml/dist/compose/resolve-block-seq.d.ts +0 -6
  314. package/core/k-cli/node_modules/yaml/dist/compose/resolve-block-seq.js +0 -51
  315. package/core/k-cli/node_modules/yaml/dist/compose/resolve-end.d.ts +0 -6
  316. package/core/k-cli/node_modules/yaml/dist/compose/resolve-end.js +0 -39
  317. package/core/k-cli/node_modules/yaml/dist/compose/resolve-flow-collection.d.ts +0 -7
  318. package/core/k-cli/node_modules/yaml/dist/compose/resolve-flow-collection.js +0 -209
  319. package/core/k-cli/node_modules/yaml/dist/compose/resolve-flow-scalar.d.ts +0 -10
  320. package/core/k-cli/node_modules/yaml/dist/compose/resolve-flow-scalar.js +0 -225
  321. package/core/k-cli/node_modules/yaml/dist/compose/resolve-props.d.ts +0 -23
  322. package/core/k-cli/node_modules/yaml/dist/compose/resolve-props.js +0 -148
  323. package/core/k-cli/node_modules/yaml/dist/compose/util-contains-newline.d.ts +0 -2
  324. package/core/k-cli/node_modules/yaml/dist/compose/util-contains-newline.js +0 -36
  325. package/core/k-cli/node_modules/yaml/dist/compose/util-empty-scalar-position.d.ts +0 -2
  326. package/core/k-cli/node_modules/yaml/dist/compose/util-empty-scalar-position.js +0 -28
  327. package/core/k-cli/node_modules/yaml/dist/compose/util-flow-indent-check.d.ts +0 -3
  328. package/core/k-cli/node_modules/yaml/dist/compose/util-flow-indent-check.js +0 -17
  329. package/core/k-cli/node_modules/yaml/dist/compose/util-map-includes.d.ts +0 -4
  330. package/core/k-cli/node_modules/yaml/dist/compose/util-map-includes.js +0 -15
  331. package/core/k-cli/node_modules/yaml/dist/doc/Document.d.ts +0 -141
  332. package/core/k-cli/node_modules/yaml/dist/doc/Document.js +0 -337
  333. package/core/k-cli/node_modules/yaml/dist/doc/anchors.d.ts +0 -24
  334. package/core/k-cli/node_modules/yaml/dist/doc/anchors.js +0 -76
  335. package/core/k-cli/node_modules/yaml/dist/doc/applyReviver.d.ts +0 -9
  336. package/core/k-cli/node_modules/yaml/dist/doc/applyReviver.js +0 -57
  337. package/core/k-cli/node_modules/yaml/dist/doc/createNode.d.ts +0 -17
  338. package/core/k-cli/node_modules/yaml/dist/doc/createNode.js +0 -90
  339. package/core/k-cli/node_modules/yaml/dist/doc/directives.d.ts +0 -49
  340. package/core/k-cli/node_modules/yaml/dist/doc/directives.js +0 -178
  341. package/core/k-cli/node_modules/yaml/dist/errors.d.ts +0 -21
  342. package/core/k-cli/node_modules/yaml/dist/errors.js +0 -62
  343. package/core/k-cli/node_modules/yaml/dist/index.d.ts +0 -25
  344. package/core/k-cli/node_modules/yaml/dist/index.js +0 -50
  345. package/core/k-cli/node_modules/yaml/dist/log.d.ts +0 -3
  346. package/core/k-cli/node_modules/yaml/dist/log.js +0 -19
  347. package/core/k-cli/node_modules/yaml/dist/nodes/Alias.d.ts +0 -29
  348. package/core/k-cli/node_modules/yaml/dist/nodes/Alias.js +0 -116
  349. package/core/k-cli/node_modules/yaml/dist/nodes/Collection.d.ts +0 -73
  350. package/core/k-cli/node_modules/yaml/dist/nodes/Collection.js +0 -151
  351. package/core/k-cli/node_modules/yaml/dist/nodes/Node.d.ts +0 -53
  352. package/core/k-cli/node_modules/yaml/dist/nodes/Node.js +0 -40
  353. package/core/k-cli/node_modules/yaml/dist/nodes/Pair.d.ts +0 -22
  354. package/core/k-cli/node_modules/yaml/dist/nodes/Pair.js +0 -39
  355. package/core/k-cli/node_modules/yaml/dist/nodes/Scalar.d.ts +0 -43
  356. package/core/k-cli/node_modules/yaml/dist/nodes/Scalar.js +0 -27
  357. package/core/k-cli/node_modules/yaml/dist/nodes/YAMLMap.d.ts +0 -53
  358. package/core/k-cli/node_modules/yaml/dist/nodes/YAMLMap.js +0 -147
  359. package/core/k-cli/node_modules/yaml/dist/nodes/YAMLSeq.d.ts +0 -60
  360. package/core/k-cli/node_modules/yaml/dist/nodes/YAMLSeq.js +0 -115
  361. package/core/k-cli/node_modules/yaml/dist/nodes/addPairToJSMap.d.ts +0 -4
  362. package/core/k-cli/node_modules/yaml/dist/nodes/addPairToJSMap.js +0 -65
  363. package/core/k-cli/node_modules/yaml/dist/nodes/identity.d.ts +0 -23
  364. package/core/k-cli/node_modules/yaml/dist/nodes/identity.js +0 -53
  365. package/core/k-cli/node_modules/yaml/dist/nodes/toJS.d.ts +0 -29
  366. package/core/k-cli/node_modules/yaml/dist/nodes/toJS.js +0 -39
  367. package/core/k-cli/node_modules/yaml/dist/options.d.ts +0 -350
  368. package/core/k-cli/node_modules/yaml/dist/parse/cst-scalar.d.ts +0 -64
  369. package/core/k-cli/node_modules/yaml/dist/parse/cst-scalar.js +0 -218
  370. package/core/k-cli/node_modules/yaml/dist/parse/cst-stringify.d.ts +0 -8
  371. package/core/k-cli/node_modules/yaml/dist/parse/cst-stringify.js +0 -63
  372. package/core/k-cli/node_modules/yaml/dist/parse/cst-visit.d.ts +0 -39
  373. package/core/k-cli/node_modules/yaml/dist/parse/cst-visit.js +0 -99
  374. package/core/k-cli/node_modules/yaml/dist/parse/cst.d.ts +0 -109
  375. package/core/k-cli/node_modules/yaml/dist/parse/cst.js +0 -112
  376. package/core/k-cli/node_modules/yaml/dist/parse/lexer.d.ts +0 -87
  377. package/core/k-cli/node_modules/yaml/dist/parse/lexer.js +0 -719
  378. package/core/k-cli/node_modules/yaml/dist/parse/line-counter.d.ts +0 -22
  379. package/core/k-cli/node_modules/yaml/dist/parse/line-counter.js +0 -41
  380. package/core/k-cli/node_modules/yaml/dist/parse/parser.d.ts +0 -84
  381. package/core/k-cli/node_modules/yaml/dist/parse/parser.js +0 -972
  382. package/core/k-cli/node_modules/yaml/dist/public-api.d.ts +0 -44
  383. package/core/k-cli/node_modules/yaml/dist/public-api.js +0 -107
  384. package/core/k-cli/node_modules/yaml/dist/schema/Schema.d.ts +0 -17
  385. package/core/k-cli/node_modules/yaml/dist/schema/Schema.js +0 -39
  386. package/core/k-cli/node_modules/yaml/dist/schema/common/map.d.ts +0 -2
  387. package/core/k-cli/node_modules/yaml/dist/schema/common/map.js +0 -19
  388. package/core/k-cli/node_modules/yaml/dist/schema/common/null.d.ts +0 -4
  389. package/core/k-cli/node_modules/yaml/dist/schema/common/null.js +0 -17
  390. package/core/k-cli/node_modules/yaml/dist/schema/common/seq.d.ts +0 -2
  391. package/core/k-cli/node_modules/yaml/dist/schema/common/seq.js +0 -19
  392. package/core/k-cli/node_modules/yaml/dist/schema/common/string.d.ts +0 -2
  393. package/core/k-cli/node_modules/yaml/dist/schema/common/string.js +0 -16
  394. package/core/k-cli/node_modules/yaml/dist/schema/core/bool.d.ts +0 -4
  395. package/core/k-cli/node_modules/yaml/dist/schema/core/bool.js +0 -21
  396. package/core/k-cli/node_modules/yaml/dist/schema/core/float.d.ts +0 -4
  397. package/core/k-cli/node_modules/yaml/dist/schema/core/float.js +0 -47
  398. package/core/k-cli/node_modules/yaml/dist/schema/core/int.d.ts +0 -4
  399. package/core/k-cli/node_modules/yaml/dist/schema/core/int.js +0 -42
  400. package/core/k-cli/node_modules/yaml/dist/schema/core/schema.d.ts +0 -1
  401. package/core/k-cli/node_modules/yaml/dist/schema/core/schema.js +0 -25
  402. package/core/k-cli/node_modules/yaml/dist/schema/json/schema.d.ts +0 -2
  403. package/core/k-cli/node_modules/yaml/dist/schema/json/schema.js +0 -64
  404. package/core/k-cli/node_modules/yaml/dist/schema/json-schema.d.ts +0 -69
  405. package/core/k-cli/node_modules/yaml/dist/schema/tags.d.ts +0 -48
  406. package/core/k-cli/node_modules/yaml/dist/schema/tags.js +0 -99
  407. package/core/k-cli/node_modules/yaml/dist/schema/types.d.ts +0 -92
  408. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/binary.d.ts +0 -2
  409. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/binary.js +0 -70
  410. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/bool.d.ts +0 -7
  411. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/bool.js +0 -29
  412. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/float.d.ts +0 -4
  413. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/float.js +0 -50
  414. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/int.d.ts +0 -5
  415. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/int.js +0 -76
  416. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/merge.d.ts +0 -9
  417. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/merge.js +0 -68
  418. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/omap.d.ts +0 -22
  419. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/omap.js +0 -77
  420. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/pairs.d.ts +0 -10
  421. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/pairs.js +0 -82
  422. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/schema.d.ts +0 -1
  423. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/schema.js +0 -41
  424. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/set.d.ts +0 -28
  425. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/set.js +0 -96
  426. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/timestamp.d.ts +0 -6
  427. package/core/k-cli/node_modules/yaml/dist/schema/yaml-1.1/timestamp.js +0 -105
  428. package/core/k-cli/node_modules/yaml/dist/stringify/foldFlowLines.d.ts +0 -34
  429. package/core/k-cli/node_modules/yaml/dist/stringify/foldFlowLines.js +0 -151
  430. package/core/k-cli/node_modules/yaml/dist/stringify/stringify.d.ts +0 -21
  431. package/core/k-cli/node_modules/yaml/dist/stringify/stringify.js +0 -132
  432. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyCollection.d.ts +0 -17
  433. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyCollection.js +0 -155
  434. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyComment.d.ts +0 -10
  435. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyComment.js +0 -24
  436. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyDocument.d.ts +0 -4
  437. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyDocument.js +0 -87
  438. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyNumber.d.ts +0 -2
  439. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyNumber.js +0 -26
  440. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyPair.d.ts +0 -3
  441. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyPair.js +0 -152
  442. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyString.d.ts +0 -9
  443. package/core/k-cli/node_modules/yaml/dist/stringify/stringifyString.js +0 -338
  444. package/core/k-cli/node_modules/yaml/dist/test-events.d.ts +0 -4
  445. package/core/k-cli/node_modules/yaml/dist/test-events.js +0 -134
  446. package/core/k-cli/node_modules/yaml/dist/util.d.ts +0 -16
  447. package/core/k-cli/node_modules/yaml/dist/util.js +0 -28
  448. package/core/k-cli/node_modules/yaml/dist/visit.d.ts +0 -102
  449. package/core/k-cli/node_modules/yaml/dist/visit.js +0 -236
  450. package/core/k-cli/node_modules/yaml/package.json +0 -97
  451. package/core/k-cli/node_modules/yaml/util.js +0 -2
@@ -1,101 +1,101 @@
1
- # Kenaz Jev 離線評測工具
1
+ # Kenaz Jev 離線評測工具
2
2
 
3
3
  > **LEGACY (TASK_1987, 2026-09-21)**: the Muninn observer this integration overlays (translateAndClassifyMessages via muninn-watcher) is retired from all default paths; the Dashboard *Jev-assisted judgments* switch stays OFF and reaches no live consumer. Evidence and decision: `dev/MUNINN_RETIREMENT.md`. The research harness below remains runnable offline.
4
-
5
- 這是 J0/J1 研究原型。用途是驗證資料格式、API adapter、回放與評分流程;尚未接入 Dashboard、Synapse 或 Muninn。`promotion_eligible` 固定為 `false`,不會因合成資料拿到高分就自動上線。
6
-
7
- 需要 Node.js 22 以上。使用內建模組,不需 npm install。以下命令從 KenazAI 根目錄執行。
8
-
9
- ## 離線操作
10
-
11
- ```powershell
12
- node --test dev/jev-eval/test/*.test.mjs
13
- node dev/jev-eval/cli.mjs validate --dataset dev/jev-eval/fixtures/synthetic.jsonl
14
- node dev/jev-eval/cli.mjs evaluate --dataset dev/jev-eval/fixtures/synthetic.jsonl --provider rules
15
- node dev/jev-eval/cli.mjs template
16
- node dev/jev-eval/cli.mjs preflight
17
- ```
18
-
19
- `rules` 是簡單、可重複的本地基準,不是 Jev 模型或其品質模擬。報告可能顯示錯誤,這是比較基準的觀察結果;工程測試綠燈只代表評測工具符合測試契約。
20
-
21
- CLI 的 JSON 輸出可供其他工具讀取。先檢查 exit code,再解析 stdout;錯誤不代表資料集得零分。不要把 live provider 的不可用/逾時樣本當作正確回答。
22
-
23
- ## 資料與回放
24
-
25
- 故障研究矩陣可離線執行:`node dev/jev-eval/resilience.mjs`。它用本機 HTTP fixture 驗證 off、shadow、evaluation-active,結果分開記錄 provider 失敗與 fallback 交付。shadow 透過 deliver callback 先交付基準,但函式回傳的 Promise 仍等待觀察結果,不能將 await 直接當作產品非阻塞介面。這不是 production provider,沒有安裝版驗收宣告。
26
-
27
- 人工盲審工具:
28
-
29
- ```powershell
30
- node dev/jev-eval/review.mjs export --dataset dev/jev-eval/fixtures/synthetic.jsonl --out review-new.json
31
- node dev/jev-eval/review.mjs apply --dataset dev/jev-eval/fixtures/synthetic.jsonl --reviews review-new.json --out reviewed-new.jsonl
32
- ```
33
-
34
- export 隱藏原始 gold 與模型結果;人工填寫 decision、reviewed、reviewer、reviewed_at(UTC ISO 時間)後才能 apply。apply 要求每筆完整、身分和 hash 相符,且輸出必須是新檔,不覆寫來源。先寫同目錄暫存檔再以 hard link 發布,檔案系統若不支援 hard link 會失敗,不降級成覆寫。審閱旗標是聲明,不是真人身分驗證。
35
-
36
- 本輪已準備 [6 筆真實片段待審檔](reports/conversation-review.pending.json) 與 [11 筆合成案例待審檔](reports/synthetic-review.pending.json),皆未填答案。來源、比較設計、剩餘缺口見 [VALIDATION_PROGRESS.md](VALIDATION_PROGRESS.md)。
37
-
38
- 資料使用 JSONL,一行一筆,欄位包括 schema_version、id、project_id、session_id、group_id、task_id、event_id、event_seq、task_revision、purpose、language、split、provenance、state。
39
-
40
- - correction:`labels.correction` 是 gold,`state.message` 是待判斷原文,`state.assumptions` 是必要前提。
41
- - relevance:`candidates` 含穩定 id、text 和 grade 0–3。grade 只用於評測,不送到 Jev。
42
- - provenance 明確區分 synthetic/real 與 agent_proposed/human_verified。審閱流程見 [ANNOTATION_GUIDE.md](ANNOTATION_GUIDE.md)。
43
- - 同 session/group 不能跨 development、calibration、held_out,避免資料洩漏。
44
-
45
- 回放 prediction 必須與 dataset 的 id/project/session/task/event/sequence/revision 及 state_hash 對應,並包含 model、question_set_version、status、failure_reason、http_status、usage、latency_ms,以及 correction_probability 或 scores。改了原文或候選後不能沿用舊結果。每筆資料必須有一筆 prediction;未回答使用明確失敗狀態及 null,不把缺失或非法分數當成 0。完整契約以 schema.mjs 的 validatePrediction 為準。
46
-
47
- ```powershell
48
- node dev/jev-eval/cli.mjs evaluate --dataset <dataset.jsonl> --provider replay --predictions <predictions.jsonl>
49
- ```
50
-
51
- 回放延遲是錄製資料,rules 延遲是本地計算時間,兩者都不是當下 Jev 網路延遲。`--split held_out` 用於凍結問法與門檻後的驗收,不得用它反覆調參。
52
-
53
- 修正分類以 0.5 為研究門檻;recall_including_abstentions 將未回答的正例計入漏失,Brier 僅涵蓋已回答案例,須搭配 coverage 判讀。重排只排序已回答的候選,但 NDCG 的理想排序包含所有候選;另列完整查詢限定的 NDCG 與候選覆蓋率。全部 grade=0 的查詢不納入 NDCG/Recall 平均,排除數會列出。這些數字不含信賴區間,不能直接當上線驗收。
54
-
55
- ## 真實 API 測試
56
-
57
- live 模式需要本機 `TYPESAFE_API_KEY`。請透過本機安全設定配置,不把 key 貼到聊天、JSONL、報告、git 或命令列參數。
58
-
59
- `preflight` 只回報金鑰是否存在與固定模型版本,不連網、不輸出金鑰。CLI 不提供 endpoint 或模型覆寫選項。
60
-
61
- ```powershell
62
- node dev/jev-eval/cli.mjs evaluate --dataset dev/jev-eval/fixtures/synthetic.jsonl --provider live --timeout-ms 800
63
- ```
64
-
65
- 這個命令會把指定資料集的必要 state/候選文字送到 TypeSafe API,因此僅使用已準備可外送的資料。預設固定模型版本 `jev-1.13.0`。工具不掃描環境檔案、不擷取 live session、不自動上傳 Synapse,也不在離線模式使用金鑰。
66
-
67
- 本原型沒有同步重試;總期限應涵蓋 response body。401/422、429/529、畸形回答、缺少題目、越界分数都需要在結果中辨認,不以原始 HTTP body 作錯誤訊息,避免供應商回顯的敏感資料流入 log。
68
-
69
- ## 評估邊界
70
-
71
- ### 2026-09-20 API 冒煙實測
72
-
73
- 已安裝並使用專案的 [TypeSafe skill](../../.agents/skills/typesafe-ai/SKILL.md),以官方即時 API/model 文件核對既有 adapter,無需更改請求格式。金鑰僅透過測試子程序環境變數使用,未保存於專案。
74
-
75
- [實測 JSON](reports/live-smoke-2026-09-20.json) 保留兩輪各 11 筆合成案例:10 秒期限輪全部成功,p50 273ms、最大 736ms;800ms 期限輪全部成功,p50 272ms、最大 653ms。更正辨識各為 8/8,兩筆可計算的排序查詢 NDCG@5 為 1;全無相關候選的第三筆不納入此平均。同期 rules 更正辨識為 4/8,NDCG@5 約 0.822。
76
-
77
- 共執行三輪 API 請求,中間一輪工具輸出被截斷,不計入保留的統計。兩輪使用相同小型合成資料;重複輸入的服務端快取狀態未知,p95/p99 因每輪僅 11 筆均等於最大值。這不是獨立真人品質評估、穩態尾延遲或 production SLO。下一步仍是人工覆核的真實評測集與現行分類/排序基準比較,production 未啟用 Jev。
78
-
79
- 目前附帶的是小型合成回歸集合,標籤由代理提出。它不能取代計畫中的 500 則訊息、100 組 recall 與 30 段交接情境,也不構成真人驗證的 held-out 效果。
80
-
81
- 正式報告需另補:繁中/英文分開標註、現行 Haiku 與 Synapse 的公平基準、延遲分布、人工複核,以及壓縮/交接後任務是否真的更少漂移。沒有供應商實測就不能宣稱 Jev 比現有流程快或準。
82
-
83
- ## 來源
84
-
85
- 現有 Muninn 分類基準的隔離執行與本機額度阻塞紀錄,見 [BASELINE_COMPARISON.md](BASELINE_COMPARISON.md)。`node dev/jev-eval/muninn-baseline.mjs` 預設只做離線契約檢查;live 須明確指定 CLI,且只接受合成資料。回傳報告的 status 也必須成功,不能只用 exit code 判定模型回答有效。
86
-
87
- - [完整導入規劃](../JEV_INTEGRATION_PLAN.md)
88
- - [TypeSafe API](https://docs.typesafe.ai/api)
89
- - [模型版本與語言限制](https://docs.typesafe.ai/models)
90
- - [機率與 confidence](https://docs.typesafe.ai/confidence)
91
- - [Jev 已知限制](https://docs.typesafe.ai/model-jaggedness/jev-1.13)
92
-
93
- 所有 production 功能開關保持既有狀態;本目錄的工具不會啟用模型、修改任務狀態、寫入記憶權重或啟動額外常駐服務。
94
-
95
- ## Local historical corpus preparation
96
-
97
- The separate opt-in `corpus.mjs` tool inventories authorized local history without network calls. See [HISTORICAL_CORPUS.md](HISTORICAL_CORPUS.md): 458 unreviewed candidates, 42 short of target, not a labelled evaluation set. Existing evaluation CLI does not automatically collect or upload history.
98
-
99
- Local observed-context restoration and read-only review packets: [CONTEXT_REVIEW.md](CONTEXT_REVIEW.md). 419 partial, 23 missing, 16 rejected; 0 human-reviewed.
100
-
101
- Current consolidated findings and local annotation pilot: [EVALUATION_STATUS.md](EVALUATION_STATUS.md).
4
+
5
+ 這是 J0/J1 研究原型。用途是驗證資料格式、API adapter、回放與評分流程;尚未接入 Dashboard、Synapse 或 Muninn。`promotion_eligible` 固定為 `false`,不會因合成資料拿到高分就自動上線。
6
+
7
+ 需要 Node.js 22 以上。使用內建模組,不需 npm install。以下命令從 KenazAI 根目錄執行。
8
+
9
+ ## 離線操作
10
+
11
+ ```powershell
12
+ node --test dev/jev-eval/test/*.test.mjs
13
+ node dev/jev-eval/cli.mjs validate --dataset dev/jev-eval/fixtures/synthetic.jsonl
14
+ node dev/jev-eval/cli.mjs evaluate --dataset dev/jev-eval/fixtures/synthetic.jsonl --provider rules
15
+ node dev/jev-eval/cli.mjs template
16
+ node dev/jev-eval/cli.mjs preflight
17
+ ```
18
+
19
+ `rules` 是簡單、可重複的本地基準,不是 Jev 模型或其品質模擬。報告可能顯示錯誤,這是比較基準的觀察結果;工程測試綠燈只代表評測工具符合測試契約。
20
+
21
+ CLI 的 JSON 輸出可供其他工具讀取。先檢查 exit code,再解析 stdout;錯誤不代表資料集得零分。不要把 live provider 的不可用/逾時樣本當作正確回答。
22
+
23
+ ## 資料與回放
24
+
25
+ 故障研究矩陣可離線執行:`node dev/jev-eval/resilience.mjs`。它用本機 HTTP fixture 驗證 off、shadow、evaluation-active,結果分開記錄 provider 失敗與 fallback 交付。shadow 透過 deliver callback 先交付基準,但函式回傳的 Promise 仍等待觀察結果,不能將 await 直接當作產品非阻塞介面。這不是 production provider,沒有安裝版驗收宣告。
26
+
27
+ 人工盲審工具:
28
+
29
+ ```powershell
30
+ node dev/jev-eval/review.mjs export --dataset dev/jev-eval/fixtures/synthetic.jsonl --out review-new.json
31
+ node dev/jev-eval/review.mjs apply --dataset dev/jev-eval/fixtures/synthetic.jsonl --reviews review-new.json --out reviewed-new.jsonl
32
+ ```
33
+
34
+ export 隱藏原始 gold 與模型結果;人工填寫 decision、reviewed、reviewer、reviewed_at(UTC ISO 時間)後才能 apply。apply 要求每筆完整、身分和 hash 相符,且輸出必須是新檔,不覆寫來源。先寫同目錄暫存檔再以 hard link 發布,檔案系統若不支援 hard link 會失敗,不降級成覆寫。審閱旗標是聲明,不是真人身分驗證。
35
+
36
+ 本輪已準備 [6 筆真實片段待審檔](reports/conversation-review.pending.json) 與 [11 筆合成案例待審檔](reports/synthetic-review.pending.json),皆未填答案。來源、比較設計、剩餘缺口見 [VALIDATION_PROGRESS.md](VALIDATION_PROGRESS.md)。
37
+
38
+ 資料使用 JSONL,一行一筆,欄位包括 schema_version、id、project_id、session_id、group_id、task_id、event_id、event_seq、task_revision、purpose、language、split、provenance、state。
39
+
40
+ - correction:`labels.correction` 是 gold,`state.message` 是待判斷原文,`state.assumptions` 是必要前提。
41
+ - relevance:`candidates` 含穩定 id、text 和 grade 0–3。grade 只用於評測,不送到 Jev。
42
+ - provenance 明確區分 synthetic/real 與 agent_proposed/human_verified。審閱流程見 [ANNOTATION_GUIDE.md](ANNOTATION_GUIDE.md)。
43
+ - 同 session/group 不能跨 development、calibration、held_out,避免資料洩漏。
44
+
45
+ 回放 prediction 必須與 dataset 的 id/project/session/task/event/sequence/revision 及 state_hash 對應,並包含 model、question_set_version、status、failure_reason、http_status、usage、latency_ms,以及 correction_probability 或 scores。改了原文或候選後不能沿用舊結果。每筆資料必須有一筆 prediction;未回答使用明確失敗狀態及 null,不把缺失或非法分數當成 0。完整契約以 schema.mjs 的 validatePrediction 為準。
46
+
47
+ ```powershell
48
+ node dev/jev-eval/cli.mjs evaluate --dataset <dataset.jsonl> --provider replay --predictions <predictions.jsonl>
49
+ ```
50
+
51
+ 回放延遲是錄製資料,rules 延遲是本地計算時間,兩者都不是當下 Jev 網路延遲。`--split held_out` 用於凍結問法與門檻後的驗收,不得用它反覆調參。
52
+
53
+ 修正分類以 0.5 為研究門檻;recall_including_abstentions 將未回答的正例計入漏失,Brier 僅涵蓋已回答案例,須搭配 coverage 判讀。重排只排序已回答的候選,但 NDCG 的理想排序包含所有候選;另列完整查詢限定的 NDCG 與候選覆蓋率。全部 grade=0 的查詢不納入 NDCG/Recall 平均,排除數會列出。這些數字不含信賴區間,不能直接當上線驗收。
54
+
55
+ ## 真實 API 測試
56
+
57
+ live 模式需要本機 `TYPESAFE_API_KEY`。請透過本機安全設定配置,不把 key 貼到聊天、JSONL、報告、git 或命令列參數。
58
+
59
+ `preflight` 只回報金鑰是否存在與固定模型版本,不連網、不輸出金鑰。CLI 不提供 endpoint 或模型覆寫選項。
60
+
61
+ ```powershell
62
+ node dev/jev-eval/cli.mjs evaluate --dataset dev/jev-eval/fixtures/synthetic.jsonl --provider live --timeout-ms 800
63
+ ```
64
+
65
+ 這個命令會把指定資料集的必要 state/候選文字送到 TypeSafe API,因此僅使用已準備可外送的資料。預設固定模型版本 `jev-1.13.0`。工具不掃描環境檔案、不擷取 live session、不自動上傳 Synapse,也不在離線模式使用金鑰。
66
+
67
+ 本原型沒有同步重試;總期限應涵蓋 response body。401/422、429/529、畸形回答、缺少題目、越界分数都需要在結果中辨認,不以原始 HTTP body 作錯誤訊息,避免供應商回顯的敏感資料流入 log。
68
+
69
+ ## 評估邊界
70
+
71
+ ### 2026-09-20 API 冒煙實測
72
+
73
+ 已安裝並使用專案的 [TypeSafe skill](../../.agents/skills/typesafe-ai/SKILL.md),以官方即時 API/model 文件核對既有 adapter,無需更改請求格式。金鑰僅透過測試子程序環境變數使用,未保存於專案。
74
+
75
+ [實測 JSON](reports/live-smoke-2026-09-20.json) 保留兩輪各 11 筆合成案例:10 秒期限輪全部成功,p50 273ms、最大 736ms;800ms 期限輪全部成功,p50 272ms、最大 653ms。更正辨識各為 8/8,兩筆可計算的排序查詢 NDCG@5 為 1;全無相關候選的第三筆不納入此平均。同期 rules 更正辨識為 4/8,NDCG@5 約 0.822。
76
+
77
+ 共執行三輪 API 請求,中間一輪工具輸出被截斷,不計入保留的統計。兩輪使用相同小型合成資料;重複輸入的服務端快取狀態未知,p95/p99 因每輪僅 11 筆均等於最大值。這不是獨立真人品質評估、穩態尾延遲或 production SLO。下一步仍是人工覆核的真實評測集與現行分類/排序基準比較,production 未啟用 Jev。
78
+
79
+ 目前附帶的是小型合成回歸集合,標籤由代理提出。它不能取代計畫中的 500 則訊息、100 組 recall 與 30 段交接情境,也不構成真人驗證的 held-out 效果。
80
+
81
+ 正式報告需另補:繁中/英文分開標註、現行 Haiku 與 Synapse 的公平基準、延遲分布、人工複核,以及壓縮/交接後任務是否真的更少漂移。沒有供應商實測就不能宣稱 Jev 比現有流程快或準。
82
+
83
+ ## 來源
84
+
85
+ 現有 Muninn 分類基準的隔離執行與本機額度阻塞紀錄,見 [BASELINE_COMPARISON.md](BASELINE_COMPARISON.md)。`node dev/jev-eval/muninn-baseline.mjs` 預設只做離線契約檢查;live 須明確指定 CLI,且只接受合成資料。回傳報告的 status 也必須成功,不能只用 exit code 判定模型回答有效。
86
+
87
+ - [完整導入規劃](../JEV_INTEGRATION_PLAN.md)
88
+ - [TypeSafe API](https://docs.typesafe.ai/api)
89
+ - [模型版本與語言限制](https://docs.typesafe.ai/models)
90
+ - [機率與 confidence](https://docs.typesafe.ai/confidence)
91
+ - [Jev 已知限制](https://docs.typesafe.ai/model-jaggedness/jev-1.13)
92
+
93
+ 所有 production 功能開關保持既有狀態;本目錄的工具不會啟用模型、修改任務狀態、寫入記憶權重或啟動額外常駐服務。
94
+
95
+ ## Local historical corpus preparation
96
+
97
+ The separate opt-in `corpus.mjs` tool inventories authorized local history without network calls. See [HISTORICAL_CORPUS.md](HISTORICAL_CORPUS.md): 458 unreviewed candidates, 42 short of target, not a labelled evaluation set. Existing evaluation CLI does not automatically collect or upload history.
98
+
99
+ Local observed-context restoration and read-only review packets: [CONTEXT_REVIEW.md](CONTEXT_REVIEW.md). 419 partial, 23 missing, 16 rejected; 0 human-reviewed.
100
+
101
+ Current consolidated findings and local annotation pilot: [EVALUATION_STATUS.md](EVALUATION_STATUS.md).
@@ -1,58 +1,58 @@
1
- # Jev Thalamus context ranking
2
-
3
- Status: local implementation verified, 2026-09-20. No installed release or real relevance improvement claimed.
4
-
5
- ## Delivery contract
6
-
7
- The single Dashboard **Jev-assisted judgments** switch (TASK_1970) controls both existing project-local fields, `jevThalamusEnabled` and `jevCorrectionEnabled`, in one atomic settings request. Both default to false. Runtime fields remain separate for compatibility, while the Dashboard presents one control. The server reads `TYPESAFE_API_KEY` from its environment; no developer key is embedded or returned.
8
-
9
- The shared Rust service receives the current task and a bounded shortlist, asks one TypeSafe Score question per candidate, and returns a stable descending permutation. Same-score ties retain input order. It preserves all candidates; existing delivery budgets decide what is shown. It does not alter Thalamus potential, firing thresholds, decay, propagation, or cascade chain order.
10
-
11
- POST `/api/synapse/jev-thalamus-rank` accepts `{projectPath, taskContext, candidates:[{id,text}]}`. The project must be an explicit existing absolute directory with its own opt-in flag. The response is `{orderedIds,status,reason}`; only `status: ranked` may change delivery. Unknown or duplicated returned IDs must never be applied by consumers.
12
-
13
- Limits: at most eight candidates; task 4096 UTF-8 bytes; candidate text 2048 bytes and ID 512 bytes; request 32 KiB; provider response 64 KiB. The fixed provider is `https://api.typesafe.ai/v1/systemone`, model `jev-1.13.0`. One 800 ms deadline covers provider transport and body, with no retry or redirect. Config reads have separate bounds, so 800 ms is not an end-to-end workflow latency promise.
14
-
15
- Missing/false/malformed config, missing key, unusable input, sensitive-text patterns, timeout, HTTP failure or invalid Score envelope preserve baseline order. Disabling the flag while a request runs discards its result. Sensitive-text checks are conservative heuristics, not a guarantee that arbitrary text contains no secrets. No raw-text decision telemetry is written by this service.
16
-
17
- ## Consumers
18
-
19
- - Native Claude UserPromptSubmit: TASK_44 wiring orders recalled Active Sensing entries before the existing 800-character rendering budget. It sends the current bounded prompt and existing candidate summaries, not the full transcript. The hook retains original context, DDx/reflex messages, and Alaya output.
20
- - MCP `thalamus_context`: optional `taskContext` enables ranking of delivered near-threshold/recent-fire/cluster-symbol lists. Names, paths, and docstrings come from the project's existing read-only search index. Missing context or candidates retains baseline. A 200 ms caller deadline bounds enrichment waiting; the blocking SQLite worker may finish later. This is not hard cancellation of database computation.
21
- - Bifrost does not invoke native `.claude/hooks/` automatically. This implementation does not claim automatic Bifrost injection; its callers need the explicit MCP context argument.
22
-
23
- TASK_44 corrected the existing native hook recall mismatch (`query` instead of required `q`) and added explicit `projectPath` to recall and turn-context requests. Its separate existing seconds/milliseconds cooldown mismatch is outside this change.
24
-
25
- ## Settings and feedback
26
-
27
- TASK_1970 supersedes the separate UI introduced in TASK_1967 with one Default Models switch and one overall feedback area. GET status uses `feature=all`, derives both flags from the same config snapshot, and reports mixed legacy settings explicitly. A toggle sends both flags with the same value in one request and requires both acknowledgements. New overall feedback is tagged `all`; previous correction/Thalamus records retain their classifications, and legacy records without a discriminator belong to correction. Feedback is overall experience, not labeled ground truth.
28
-
29
- ## Verification and rollout
30
-
31
- Parent verified `node --test .claude/hooks/jev-thalamus-context.test.js`: 7/7 passed. This executes the real hook in a fixture VM and checks rendered selection before truncation, exact OFF/failure fallback, disabled-in-flight results, request bounds, and preserved safety/Alaya content. `node dev/scripts/run-rust-tests.js jev_thalamus` passed 12/12 tests (parent inspected the canonical runner log). Coverage includes shared scoring, read-only enrichment, actual route OFF/body limit, and MCP no-context/OFF behavior. The positive MCP path is assembled from those tested components; no live provider-to-MCP run was performed. Parent reran the two focused Jest files: 17/17 passed. Full Dashboard TypeScript checking passed with --noEmit --incremental false. Final combined `node dev/scripts/run-rust-tests.js jev` passed 23/23 (12 ranking/route/MCP + 11 settings). Together with 77 Node and 17 Jest tests, 117 engineering checks passed. Existing correction/research regression: `node --test .formation/tools/jev-correction.test.js dev/jev-eval/test/*.test.mjs` passed 70 tests during this work.
32
-
33
- Real Jev quality, Chinese-task relevance, representative end-to-end latency and comparison against baseline remain unmeasured for this feature. Offline mocked ranking validates delivery and fallback, not model accuracy. No historical session data was uploaded during this implementation.
34
-
35
- The core stager includes tracked `.claude` files only. A new untracked hook helper is omitted from an actual package until it is included in the tracked release candidate. No installer build, merge, or release has been performed.
36
-
37
- References: [TypeSafe API](https://docs.typesafe.ai/api), [Score](https://docs.typesafe.ai/primitives/score), [reranking cookbook](https://docs.typesafe.ai/cookbooks/rerank_typesafe). Official examples motivate the mechanism; they are not Kenaz measurements.
38
-
39
- ## Task tracking migration
40
-
41
- The user explicitly restored KenazAI as the tracking project. Existing implementations were not redispatched. Old quotation cards retain their history and forward references, with duplicate tracking closed as superseded; this does not cancel the implementation.
42
-
43
- | quotation history | KenazAI authoritative task | Scope |
44
- | --- | --- | --- |
45
- | TASK_43 | TASK_1965 | Shared Rust ranker and MCP/HTTP delivery |
46
- | TASK_44 | TASK_1966 | Native hook delivery |
47
- | TASK_45 | TASK_1967 | Independent settings and feedback |
48
- | TASK_46 | TASK_1968 | Verification documentation |
49
-
50
- TASK_1970 UI update: parent verified focused Jest 23/23, including actual panel SSR with exactly one checkbox, mixed legacy state, one-request dual-flag persistence, and partial-acknowledgement rejection. Canonical Rust filter `jev` passed 26/26; full Dashboard TypeScript and diff checks passed. Live combined status returned feature=all, enabled=true, mixed=false, both flags=true; runtime keyConfigured remained false, so no provider probe was sent.
51
-
52
- TASK_1971 follow-up: Windows startup falls back to HKCU Environment TYPESAFE_API_KEY
53
- only when absent from the inherited process environment. Six startup tests and
54
- 12 Jev regression tests passed. After natural dev reload, combined status reported
55
- keyConfigured=true; one synthetic live request returned ranked, orderedIds
56
- [startup, docx], reason=null. Client elapsed time was 2726ms including localhost
57
- request overhead; this is not a provider-only latency measurement or quality eval.
58
- No real transcript was sent, and installed release packaging remains unverified.
1
+ # Jev Thalamus context ranking
2
+
3
+ Status: local implementation verified, 2026-09-20. No installed release or real relevance improvement claimed.
4
+
5
+ ## Delivery contract
6
+
7
+ The single Dashboard **Jev-assisted judgments** switch (TASK_1970) controls both existing project-local fields, `jevThalamusEnabled` and `jevCorrectionEnabled`, in one atomic settings request. Both default to false. Runtime fields remain separate for compatibility, while the Dashboard presents one control. The server reads `TYPESAFE_API_KEY` from its environment; no developer key is embedded or returned.
8
+
9
+ The shared Rust service receives the current task and a bounded shortlist, asks one TypeSafe Score question per candidate, and returns a stable descending permutation. Same-score ties retain input order. It preserves all candidates; existing delivery budgets decide what is shown. It does not alter Thalamus potential, firing thresholds, decay, propagation, or cascade chain order.
10
+
11
+ POST `/api/synapse/jev-thalamus-rank` accepts `{projectPath, taskContext, candidates:[{id,text}]}`. The project must be an explicit existing absolute directory with its own opt-in flag. The response is `{orderedIds,status,reason}`; only `status: ranked` may change delivery. Unknown or duplicated returned IDs must never be applied by consumers.
12
+
13
+ Limits: at most eight candidates; task 4096 UTF-8 bytes; candidate text 2048 bytes and ID 512 bytes; request 32 KiB; provider response 64 KiB. The fixed provider is `https://api.typesafe.ai/v1/systemone`, model `jev-1.13.0`. One 800 ms deadline covers provider transport and body, with no retry or redirect. Config reads have separate bounds, so 800 ms is not an end-to-end workflow latency promise.
14
+
15
+ Missing/false/malformed config, missing key, unusable input, sensitive-text patterns, timeout, HTTP failure or invalid Score envelope preserve baseline order. Disabling the flag while a request runs discards its result. Sensitive-text checks are conservative heuristics, not a guarantee that arbitrary text contains no secrets. No raw-text decision telemetry is written by this service.
16
+
17
+ ## Consumers
18
+
19
+ - Native Claude UserPromptSubmit: TASK_44 wiring orders recalled Active Sensing entries before the existing 800-character rendering budget. It sends the current bounded prompt and existing candidate summaries, not the full transcript. The hook retains original context, DDx/reflex messages, and Alaya output.
20
+ - MCP `thalamus_context`: optional `taskContext` enables ranking of delivered near-threshold/recent-fire/cluster-symbol lists. Names, paths, and docstrings come from the project's existing read-only search index. Missing context or candidates retains baseline. A 200 ms caller deadline bounds enrichment waiting; the blocking SQLite worker may finish later. This is not hard cancellation of database computation.
21
+ - Bifrost does not invoke native `.claude/hooks/` automatically. This implementation does not claim automatic Bifrost injection; its callers need the explicit MCP context argument.
22
+
23
+ TASK_44 corrected the existing native hook recall mismatch (`query` instead of required `q`) and added explicit `projectPath` to recall and turn-context requests. Its separate existing seconds/milliseconds cooldown mismatch is outside this change.
24
+
25
+ ## Settings and feedback
26
+
27
+ TASK_1970 supersedes the separate UI introduced in TASK_1967 with one Default Models switch and one overall feedback area. GET status uses `feature=all`, derives both flags from the same config snapshot, and reports mixed legacy settings explicitly. A toggle sends both flags with the same value in one request and requires both acknowledgements. New overall feedback is tagged `all`; previous correction/Thalamus records retain their classifications, and legacy records without a discriminator belong to correction. Feedback is overall experience, not labeled ground truth.
28
+
29
+ ## Verification and rollout
30
+
31
+ Parent verified `node --test .claude/hooks/jev-thalamus-context.test.js`: 7/7 passed. This executes the real hook in a fixture VM and checks rendered selection before truncation, exact OFF/failure fallback, disabled-in-flight results, request bounds, and preserved safety/Alaya content. `node dev/scripts/run-rust-tests.js jev_thalamus` passed 12/12 tests (parent inspected the canonical runner log). Coverage includes shared scoring, read-only enrichment, actual route OFF/body limit, and MCP no-context/OFF behavior. The positive MCP path is assembled from those tested components; no live provider-to-MCP run was performed. Parent reran the two focused Jest files: 17/17 passed. Full Dashboard TypeScript checking passed with --noEmit --incremental false. Final combined `node dev/scripts/run-rust-tests.js jev` passed 23/23 (12 ranking/route/MCP + 11 settings). Together with 77 Node and 17 Jest tests, 117 engineering checks passed. Existing correction/research regression: `node --test .formation/tools/jev-correction.test.js dev/jev-eval/test/*.test.mjs` passed 70 tests during this work.
32
+
33
+ Real Jev quality, Chinese-task relevance, representative end-to-end latency and comparison against baseline remain unmeasured for this feature. Offline mocked ranking validates delivery and fallback, not model accuracy. No historical session data was uploaded during this implementation.
34
+
35
+ The core stager includes tracked `.claude` files only. A new untracked hook helper is omitted from an actual package until it is included in the tracked release candidate. No installer build, merge, or release has been performed.
36
+
37
+ References: [TypeSafe API](https://docs.typesafe.ai/api), [Score](https://docs.typesafe.ai/primitives/score), [reranking cookbook](https://docs.typesafe.ai/cookbooks/rerank_typesafe). Official examples motivate the mechanism; they are not Kenaz measurements.
38
+
39
+ ## Task tracking migration
40
+
41
+ The user explicitly restored KenazAI as the tracking project. Existing implementations were not redispatched. Old quotation cards retain their history and forward references, with duplicate tracking closed as superseded; this does not cancel the implementation.
42
+
43
+ | quotation history | KenazAI authoritative task | Scope |
44
+ | --- | --- | --- |
45
+ | TASK_43 | TASK_1965 | Shared Rust ranker and MCP/HTTP delivery |
46
+ | TASK_44 | TASK_1966 | Native hook delivery |
47
+ | TASK_45 | TASK_1967 | Independent settings and feedback |
48
+ | TASK_46 | TASK_1968 | Verification documentation |
49
+
50
+ TASK_1970 UI update: parent verified focused Jest 23/23, including actual panel SSR with exactly one checkbox, mixed legacy state, one-request dual-flag persistence, and partial-acknowledgement rejection. Canonical Rust filter `jev` passed 26/26; full Dashboard TypeScript and diff checks passed. Live combined status returned feature=all, enabled=true, mixed=false, both flags=true; runtime keyConfigured remained false, so no provider probe was sent.
51
+
52
+ TASK_1971 follow-up: Windows startup falls back to HKCU Environment TYPESAFE_API_KEY
53
+ only when absent from the inherited process environment. Six startup tests and
54
+ 12 Jev regression tests passed. After natural dev reload, combined status reported
55
+ keyConfigured=true; one synthetic live request returned ranked, orderedIds
56
+ [startup, docx], reason=null. Client elapsed time was 2726ms including localhost
57
+ request overhead; this is not a provider-only latency measurement or quality eval.
58
+ No real transcript was sent, and installed release packaging remains unverified.
@@ -1,32 +1,32 @@
1
- {
2
- "schema_version": 1,
3
- "date": "2026-09-20",
4
- "count": 12,
5
- "sample_groups": 12,
6
- "sampling": "12-case short-context convenience pilot; <=1800 characters total; up to4 each targetlength0-15,16-60,61+; sorted id, uniquegroups; not representative or held-out",
7
- "separate_agent_reviews": 2,
8
- "reviewers_independent_humans": false,
9
- "same_label_including_null": 10,
10
- "non_null_agreed": 9,
11
- "agreed_positive": 2,
12
- "agreed_negative": 7,
13
- "both_unresolved": 1,
14
- "disagreements": 2,
15
- "needs_adjudication": 3,
16
- "human_verified": 0,
17
- "jev_real_case_predictions": 0,
18
- "promotion_eligible": false,
19
- "evidence_substrings_verified": 49,
20
- "private_artifact_hashes": {
21
- "pilot.json": "36de481aa3d231084e8d130d2836d34f124e8cea234b4d1edd1b0fdbdbec0781",
22
- "reviewer-a.json": "b25b74d8d6bcd13acf9040630121d4976eadba4f09aa6b4cff5e4fa29df21e95",
23
- "reviewer-b.json": "a4c73c239136dd2a83db1f03a4bb4689aecaf07a98faee24c6b2028988b7adc3",
24
- "comparison.json": "44768dedd253618c41ae7366ee0a850b9d6ab5948292a72b4e7de3ee5293d287"
25
- },
26
- "limitations": [
27
- "Same model-family agent proposals are correlated; agreement is not accuracy or independent human verification.",
28
- "Convenience short-context pilot, no held-out or population inference.",
29
- "Two disagreements preserved without forced consensus; one shared null also unresolved.",
30
- "No Jev or Haiku request made for these real cases."
31
- ]
32
- }
1
+ {
2
+ "schema_version": 1,
3
+ "date": "2026-09-20",
4
+ "count": 12,
5
+ "sample_groups": 12,
6
+ "sampling": "12-case short-context convenience pilot; <=1800 characters total; up to4 each targetlength0-15,16-60,61+; sorted id, uniquegroups; not representative or held-out",
7
+ "separate_agent_reviews": 2,
8
+ "reviewers_independent_humans": false,
9
+ "same_label_including_null": 10,
10
+ "non_null_agreed": 9,
11
+ "agreed_positive": 2,
12
+ "agreed_negative": 7,
13
+ "both_unresolved": 1,
14
+ "disagreements": 2,
15
+ "needs_adjudication": 3,
16
+ "human_verified": 0,
17
+ "jev_real_case_predictions": 0,
18
+ "promotion_eligible": false,
19
+ "evidence_substrings_verified": 49,
20
+ "private_artifact_hashes": {
21
+ "pilot.json": "36de481aa3d231084e8d130d2836d34f124e8cea234b4d1edd1b0fdbdbec0781",
22
+ "reviewer-a.json": "b25b74d8d6bcd13acf9040630121d4976eadba4f09aa6b4cff5e4fa29df21e95",
23
+ "reviewer-b.json": "a4c73c239136dd2a83db1f03a4bb4689aecaf07a98faee24c6b2028988b7adc3",
24
+ "comparison.json": "44768dedd253618c41ae7366ee0a850b9d6ab5948292a72b4e7de3ee5293d287"
25
+ },
26
+ "limitations": [
27
+ "Same model-family agent proposals are correlated; agreement is not accuracy or independent human verification.",
28
+ "Convenience short-context pilot, no held-out or population inference.",
29
+ "Two disagreements preserved without forced consensus; one shared null also unresolved.",
30
+ "No Jev or Haiku request made for these real cases."
31
+ ]
32
+ }
@@ -1,48 +1,48 @@
1
- {
2
- "schema_version": 1,
3
- "count": 458,
4
- "local_only": true,
5
- "promotion_eligible": false,
6
- "gold_labels": 0,
7
- "status_counts": {
8
- "restored": 0,
9
- "partial": 419,
10
- "missing": 23,
11
- "rejected": 16
12
- },
13
- "limitations": [
14
- "Observed text only; no claims that assistant statements are true.",
15
- "Prefix hashes describe the current source snapshot; original corpus had no prefix hash, so historical immutability is not established.",
16
- "Missing/truncated context must be reviewed before annotation.",
17
- "No model calls or inferred goals/assumptions/labels."
18
- ],
19
- "verification": {
20
- "tests_passed": 54,
21
- "context_tests_passed": 10,
22
- "source_text_items_verified": 2352,
23
- "prefix_hashes_verified": 442,
24
- "ids_preserved": 458,
25
- "html_articles": 458,
26
- "active_html_elements": 0,
27
- "detected_credential_patterns_in_export": 0,
28
- "external_model_requests": 0,
29
- "human_verified": 0,
30
- "input_pending_sha256": "5f5a10efc646863feae904df71493854f25a3d8f82993c1b066ae6ddea9b9a0d",
31
- "review_packet_sha256": "0afb91d29be4dc004b5fd2fdefba98237cd6e060f8c0cd501415fa8bdc5baefa",
32
- "source_code_sha256": "fde7ed3ba6a8071c6b51a84ca3ca0d1e994670663e619d2473bd81dbe20edea3"
33
- },
34
- "gap_counts": {
35
- "missing_parent": 3,
36
- "credential_context": 123,
37
- "generated_context": 221,
38
- "history_window": 126,
39
- "turn_limit": 357,
40
- "compaction_or_rollback": 95,
41
- "ancestor_scope_mismatch": 26,
42
- "text_truncated": 23,
43
- "nonvisible_message": 10,
44
- "unparsed_record": 5,
45
- "missing_ancestor": 3
46
- },
47
- "rejection_notes": "16 cases in one Claude source: repeated attachment UUIDs with differing parentUuid make ancestry ambiguous. Metadata-only slug changes alone are tolerated."
48
- }
1
+ {
2
+ "schema_version": 1,
3
+ "count": 458,
4
+ "local_only": true,
5
+ "promotion_eligible": false,
6
+ "gold_labels": 0,
7
+ "status_counts": {
8
+ "restored": 0,
9
+ "partial": 419,
10
+ "missing": 23,
11
+ "rejected": 16
12
+ },
13
+ "limitations": [
14
+ "Observed text only; no claims that assistant statements are true.",
15
+ "Prefix hashes describe the current source snapshot; original corpus had no prefix hash, so historical immutability is not established.",
16
+ "Missing/truncated context must be reviewed before annotation.",
17
+ "No model calls or inferred goals/assumptions/labels."
18
+ ],
19
+ "verification": {
20
+ "tests_passed": 54,
21
+ "context_tests_passed": 10,
22
+ "source_text_items_verified": 2352,
23
+ "prefix_hashes_verified": 442,
24
+ "ids_preserved": 458,
25
+ "html_articles": 458,
26
+ "active_html_elements": 0,
27
+ "detected_credential_patterns_in_export": 0,
28
+ "external_model_requests": 0,
29
+ "human_verified": 0,
30
+ "input_pending_sha256": "5f5a10efc646863feae904df71493854f25a3d8f82993c1b066ae6ddea9b9a0d",
31
+ "review_packet_sha256": "0afb91d29be4dc004b5fd2fdefba98237cd6e060f8c0cd501415fa8bdc5baefa",
32
+ "source_code_sha256": "fde7ed3ba6a8071c6b51a84ca3ca0d1e994670663e619d2473bd81dbe20edea3"
33
+ },
34
+ "gap_counts": {
35
+ "missing_parent": 3,
36
+ "credential_context": 123,
37
+ "generated_context": 221,
38
+ "history_window": 126,
39
+ "turn_limit": 357,
40
+ "compaction_or_rollback": 95,
41
+ "ancestor_scope_mismatch": 26,
42
+ "text_truncated": 23,
43
+ "nonvisible_message": 10,
44
+ "unparsed_record": 5,
45
+ "missing_ancestor": 3
46
+ },
47
+ "rejection_notes": "16 cases in one Claude source: repeated attachment UUIDs with differing parentUuid make ancestry ambiguous. Metadata-only slug changes alone are tolerated."
48
+ }
@@ -1,52 +1,52 @@
1
- {
2
- "schema_version": 1,
3
- "local_only": true,
4
- "promotion_eligible": false,
5
- "gold_labels": 0,
6
- "target": 500,
7
- "selected": 458,
8
- "shortfall": 42,
9
- "files_discovered": 638,
10
- "candidate_count": 580,
11
- "unique_count": 458,
12
- "duplicate_count": 122,
13
- "family_groups": 68,
14
- "selected_sessions": 68,
15
- "selected_groups": 68,
16
- "exclusions": {
17
- "records": 149355,
18
- "automation_seed": 314,
19
- "out_of_scope_or_subagent": 403,
20
- "bytes_read": 402005275,
21
- "excluded_files": 335,
22
- "generated_message": 230,
23
- "credential_message": 498,
24
- "adjacent_duplicate": 11,
25
- "oversize_line": 10,
26
- "response_fallback_discarded": 371,
27
- "response_fallback_candidates": 35,
28
- "session_identity_change": 7,
29
- "cross_file_automation_candidates": 0
30
- },
31
- "limitations": [
32
- "Pending annotation only; no task goals or gold labels inferred.",
33
- "Context contains preceding eligible user message only, not assistant context.",
34
- "Secret detection is conservative heuristic, not a guarantee; review locally before reuse.",
35
- "Explicit lineage and duplicate messages >=160 characters merge families; short duplicates are removed without merging; no split assigned."
36
- ],
37
- "verification": {
38
- "tests_passed": 44,
39
- "corpus_tests_passed": 11,
40
- "source_counts": {
41
- "claude": 160,
42
- "codex": 298
43
- },
44
- "context_missing": 57,
45
- "pending_sha256": "5f5a10efc646863feae904df71493854f25a3d8f82993c1b066ae6ddea9b9a0d",
46
- "source_sha256": "61c69be358f848ee88fff2ec9d31975b91bd73fda16d8c243af2195db774575f",
47
- "parent_checked_unique_ids_messages_manifest": true,
48
- "human_verified": 0,
49
- "external_requests": 0,
50
- "first_export_rejected": true
51
- }
52
- }
1
+ {
2
+ "schema_version": 1,
3
+ "local_only": true,
4
+ "promotion_eligible": false,
5
+ "gold_labels": 0,
6
+ "target": 500,
7
+ "selected": 458,
8
+ "shortfall": 42,
9
+ "files_discovered": 638,
10
+ "candidate_count": 580,
11
+ "unique_count": 458,
12
+ "duplicate_count": 122,
13
+ "family_groups": 68,
14
+ "selected_sessions": 68,
15
+ "selected_groups": 68,
16
+ "exclusions": {
17
+ "records": 149355,
18
+ "automation_seed": 314,
19
+ "out_of_scope_or_subagent": 403,
20
+ "bytes_read": 402005275,
21
+ "excluded_files": 335,
22
+ "generated_message": 230,
23
+ "credential_message": 498,
24
+ "adjacent_duplicate": 11,
25
+ "oversize_line": 10,
26
+ "response_fallback_discarded": 371,
27
+ "response_fallback_candidates": 35,
28
+ "session_identity_change": 7,
29
+ "cross_file_automation_candidates": 0
30
+ },
31
+ "limitations": [
32
+ "Pending annotation only; no task goals or gold labels inferred.",
33
+ "Context contains preceding eligible user message only, not assistant context.",
34
+ "Secret detection is conservative heuristic, not a guarantee; review locally before reuse.",
35
+ "Explicit lineage and duplicate messages >=160 characters merge families; short duplicates are removed without merging; no split assigned."
36
+ ],
37
+ "verification": {
38
+ "tests_passed": 44,
39
+ "corpus_tests_passed": 11,
40
+ "source_counts": {
41
+ "claude": 160,
42
+ "codex": 298
43
+ },
44
+ "context_missing": 57,
45
+ "pending_sha256": "5f5a10efc646863feae904df71493854f25a3d8f82993c1b066ae6ddea9b9a0d",
46
+ "source_sha256": "61c69be358f848ee88fff2ec9d31975b91bd73fda16d8c243af2195db774575f",
47
+ "parent_checked_unique_ids_messages_manifest": true,
48
+ "human_verified": 0,
49
+ "external_requests": 0,
50
+ "first_export_rejected": true
51
+ }
52
+ }