akm-cli 0.9.17-alpha.2 → 0.9.17-alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (343) hide show
  1. package/CHANGELOG.md +756 -0
  2. package/dist/akm +94 -196
  3. package/dist/cli/shared.js +6 -2
  4. package/dist/cli.js +22 -9
  5. package/dist/commands/agent/agent-dispatch.js +1 -1
  6. package/dist/commands/command/command-execution.js +24 -62
  7. package/dist/commands/feedback-cli.js +0 -1
  8. package/dist/commands/health/accept-rate.js +2 -2
  9. package/dist/commands/health/checks.js +30 -75
  10. package/dist/commands/health/config-skew.js +38 -0
  11. package/dist/commands/health/egress.js +54 -0
  12. package/dist/commands/health/html-report.js +0 -38
  13. package/dist/commands/health/improve-metrics.js +123 -562
  14. package/dist/commands/health/plugin-staleness.js +53 -3
  15. package/dist/commands/health/renderers.js +12 -4
  16. package/dist/commands/health/report-view-model.js +11 -106
  17. package/dist/commands/health/types-improve.js +4 -19
  18. package/dist/commands/health/windows.js +64 -73
  19. package/dist/commands/health.js +122 -143
  20. package/dist/commands/improve/consolidate/chunking.js +25 -100
  21. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  22. package/dist/commands/improve/consolidate.js +538 -1075
  23. package/dist/commands/improve/content-hash.js +16 -24
  24. package/dist/commands/improve/distill/content-repair.js +18 -100
  25. package/dist/commands/improve/distill-guards.js +20 -81
  26. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  27. package/dist/commands/improve/distill.js +608 -1075
  28. package/dist/commands/improve/eligibility.js +126 -400
  29. package/dist/commands/improve/execution.js +3 -5
  30. package/dist/commands/improve/extract.js +487 -1046
  31. package/dist/commands/improve/feedback-valence.js +0 -25
  32. package/dist/commands/improve/improve-cli.js +29 -166
  33. package/dist/commands/improve/improve-result-file.js +10 -66
  34. package/dist/commands/improve/improve-strategies.js +12 -7
  35. package/dist/commands/improve/improve-usage-report.js +18 -64
  36. package/dist/commands/improve/improve.js +443 -1063
  37. package/dist/commands/improve/ledger.js +114 -0
  38. package/dist/commands/improve/locks.js +2 -8
  39. package/dist/commands/improve/loop-stages.js +459 -1172
  40. package/dist/commands/improve/memory/derived-ref.js +12 -77
  41. package/dist/commands/improve/memory/memory-belief.js +14 -118
  42. package/dist/commands/improve/memory/memory-improve.js +4 -3
  43. package/dist/commands/improve/outcome-loop.js +28 -156
  44. package/dist/commands/improve/planner.js +5 -10
  45. package/dist/commands/improve/preparation.js +851 -2339
  46. package/dist/commands/improve/proactive-maintenance.js +34 -101
  47. package/dist/commands/improve/reflect-noise.js +104 -280
  48. package/dist/commands/improve/reflect.js +621 -1367
  49. package/dist/commands/improve/salience.js +46 -232
  50. package/dist/commands/improve/session-asset.js +19 -100
  51. package/dist/commands/improve/stage.js +323 -0
  52. package/dist/commands/proposal/drain.js +251 -644
  53. package/dist/commands/proposal/proposal-cli.js +3 -18
  54. package/dist/commands/proposal/proposal-types.js +20 -41
  55. package/dist/commands/proposal/proposal.js +1 -2
  56. package/dist/commands/proposal/propose.js +134 -160
  57. package/dist/commands/proposal/repository.js +502 -1487
  58. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  59. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  60. package/dist/commands/proposal/validators/proposals.js +13 -89
  61. package/dist/commands/read/curate.js +63 -413
  62. package/dist/commands/read/search-cli.js +16 -33
  63. package/dist/commands/read/search.js +17 -23
  64. package/dist/commands/read/show.js +2 -13
  65. package/dist/commands/sources/bundle-cli.js +25 -2
  66. package/dist/commands/sources/bundle-config-ops.js +7 -0
  67. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  68. package/dist/commands/sources/info.js +2 -11
  69. package/dist/commands/sources/installed-stashes.js +197 -746
  70. package/dist/commands/sources/schema-repair.js +98 -129
  71. package/dist/commands/sources/source-add.js +62 -12
  72. package/dist/commands/sources/stash-cli.js +1 -1
  73. package/dist/commands/tasks/explain.js +10 -13
  74. package/dist/commands/tasks/tasks-cli.js +9 -8
  75. package/dist/commands/tasks/tasks.js +326 -930
  76. package/dist/commands/tasks/validate.js +42 -21
  77. package/dist/commands/workflow/plan.js +22 -29
  78. package/dist/commands/workflow-cli.js +4 -4
  79. package/dist/core/adapter/adapters/akm-adapter.js +0 -1
  80. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  81. package/dist/core/adapter/adapters/akm-metadata.js +11 -12
  82. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  83. package/dist/core/adapter/execution-source.js +17 -29
  84. package/dist/core/asset/resolve-ref.js +1 -1
  85. package/dist/core/bundle-id.js +42 -5
  86. package/dist/core/bundle-rename.js +291 -0
  87. package/dist/core/config/config-io.js +1 -2
  88. package/dist/core/config/config-schema.js +1 -33
  89. package/dist/core/config/config-walker.js +1 -1
  90. package/dist/core/config/config.js +163 -68
  91. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  92. package/dist/core/config/schema/embedding.js +20 -5
  93. package/dist/core/config/schema/engines.js +5 -0
  94. package/dist/core/config/schema/execution.js +1 -1
  95. package/dist/core/config/schema/experimental.js +1 -1
  96. package/dist/core/config/schema/improve-processes.js +21 -95
  97. package/dist/core/config/schema/improve.js +4 -42
  98. package/dist/core/config/schema/scheduler.js +12 -12
  99. package/dist/core/config/schema/search.js +6 -22
  100. package/dist/core/env-secret-ref.js +0 -1
  101. package/dist/core/errors.js +8 -9
  102. package/dist/core/file-lock.js +76 -173
  103. package/dist/core/logs-db.js +2 -2
  104. package/dist/core/paths.js +0 -27
  105. package/dist/core/redaction.js +109 -2
  106. package/dist/core/run-lock.js +2 -5
  107. package/dist/core/spawn-env.js +1 -1
  108. package/dist/core/state/migrations.js +108 -61
  109. package/dist/core/state-db-scope.js +2 -4
  110. package/dist/core/state-db.js +126 -692
  111. package/dist/core/type-presentation.js +1 -9
  112. package/dist/core/write-source.js +293 -1012
  113. package/dist/execution/input-contract.js +1 -1
  114. package/dist/execution/resolved-request.js +135 -689
  115. package/dist/execution/source.js +63 -257
  116. package/dist/execution/target-ref.js +1 -1
  117. package/dist/indexer/bundle-identity-guard.js +2 -2
  118. package/dist/indexer/db/graph-db.js +106 -46
  119. package/dist/indexer/ensure-index.js +44 -85
  120. package/dist/indexer/graph/graph-extraction.js +340 -562
  121. package/dist/indexer/graph/graph-related.js +130 -0
  122. package/dist/indexer/index-rebuild-lock.js +3 -11
  123. package/dist/indexer/index-writer-lock.js +8 -17
  124. package/dist/indexer/index-written-assets.js +139 -151
  125. package/dist/indexer/indexer.js +524 -846
  126. package/dist/indexer/materialize-embeddings.js +60 -397
  127. package/dist/indexer/passes/memory-inference.js +81 -90
  128. package/dist/indexer/passes/metadata.js +132 -200
  129. package/dist/indexer/read-preflight.js +0 -7
  130. package/dist/indexer/scan/doc-to-entry.js +1 -3
  131. package/dist/indexer/scan/drain-dir.js +1 -1
  132. package/dist/indexer/search/db-search.js +181 -590
  133. package/dist/indexer/search/fts-query.js +30 -41
  134. package/dist/indexer/search/ranking.js +28 -154
  135. package/dist/indexer/search/search-attribution.js +12 -32
  136. package/dist/indexer/search/search-fields.js +11 -15
  137. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  138. package/dist/indexer/search/search-source.js +1 -4
  139. package/dist/indexer/usage/usage-events.js +2 -7
  140. package/dist/integrations/agent/engine-fallback.js +23 -40
  141. package/dist/integrations/agent/engine-resolution.js +93 -183
  142. package/dist/integrations/agent/execution.js +507 -0
  143. package/dist/integrations/agent/model-map.js +28 -156
  144. package/dist/integrations/agent/request-lowering.js +66 -141
  145. package/dist/integrations/agent/runner-dispatch.js +143 -321
  146. package/dist/integrations/agent/runner.js +54 -14
  147. package/dist/integrations/lockfile.js +53 -101
  148. package/dist/llm/embedders/deterministic.js +2 -3
  149. package/dist/llm/embedders/profile.js +71 -0
  150. package/dist/llm/embedders/remote.js +10 -15
  151. package/dist/llm/graph-extract.js +3 -12
  152. package/dist/llm/index-passes.js +3 -5
  153. package/dist/llm/memory-infer.js +1 -2
  154. package/dist/llm/metadata-enhance.js +1 -2
  155. package/dist/llm/structured-call.js +5 -24
  156. package/dist/output/generic-render.js +23 -11
  157. package/dist/output/html-render.js +13 -10
  158. package/dist/output/render-registry.js +3 -32
  159. package/dist/output/shapes/helpers.js +2 -34
  160. package/dist/output/shapes/passthrough.js +1 -9
  161. package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
  162. package/dist/output/text/command-format.js +60 -23
  163. package/dist/output/text/helpers.js +1 -1
  164. package/dist/output/text/migrate.js +5 -14
  165. package/dist/output/text/proposal-format.js +1 -2
  166. package/dist/output/text/workflow-format.js +0 -32
  167. package/dist/output/text.js +2 -0
  168. package/dist/registry/factory.js +4 -19
  169. package/dist/registry/network.js +66 -220
  170. package/dist/registry/providers/index.js +0 -2
  171. package/dist/registry/providers/skills-sh.js +3 -14
  172. package/dist/registry/providers/static-index.js +24 -26
  173. package/dist/registry/resolve.js +55 -131
  174. package/dist/scripts/akm-migrate-node.js +43937 -93313
  175. package/dist/scripts/akm-migrate.js +43697 -93071
  176. package/dist/setup/registry-stash-loader.js +4 -13
  177. package/dist/setup/semantic-assets.js +3 -44
  178. package/dist/setup/setup.js +1 -1
  179. package/dist/setup/steps/tasks.js +25 -15
  180. package/dist/sources/provider-factory.js +17 -18
  181. package/dist/sources/providers/filesystem.js +2 -3
  182. package/dist/sources/providers/git-install.js +7 -1
  183. package/dist/sources/providers/git-provider.js +0 -3
  184. package/dist/sources/providers/git-stash.js +0 -17
  185. package/dist/sources/providers/npm.js +2 -4
  186. package/dist/sources/providers/provider-utils.js +5 -10
  187. package/dist/sources/providers/website.js +0 -2
  188. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  189. package/dist/sources/website-url.js +2 -2
  190. package/dist/storage/database.js +9 -35
  191. package/dist/storage/repositories/improve-ledger-repository.js +168 -0
  192. package/dist/storage/repositories/index-connection.js +34 -70
  193. package/dist/storage/repositories/index-entries-repository.js +69 -111
  194. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  195. package/dist/storage/repositories/index-entry-schema.js +83 -269
  196. package/dist/storage/repositories/index-fts-repository.js +86 -256
  197. package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
  198. package/dist/storage/repositories/index-meta-repository.js +6 -4
  199. package/dist/storage/repositories/index-schema.js +192 -220
  200. package/dist/storage/repositories/index-utility-repository.js +8 -29
  201. package/dist/storage/repositories/index-vec-repository.js +133 -414
  202. package/dist/storage/repositories/outcome-repository.js +2 -1
  203. package/dist/storage/repositories/proposals-repository.js +35 -0
  204. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  205. package/dist/storage/repositories/task-history-repository.js +26 -4
  206. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  207. package/dist/storage/sqlite-migrations.js +136 -0
  208. package/dist/storage/sqlite-pragmas.js +11 -9
  209. package/dist/storage/sqlite-transaction.js +170 -0
  210. package/dist/storage/state-db-integrity.js +34 -27
  211. package/dist/tasks/activation-config.js +134 -62
  212. package/dist/tasks/backends/cron.js +129 -277
  213. package/dist/tasks/backends/exec-utils.js +2 -5
  214. package/dist/tasks/backends/launchd.js +125 -745
  215. package/dist/tasks/backends/schtasks.js +101 -620
  216. package/dist/tasks/prepare/prepare-support.js +5 -15
  217. package/dist/tasks/prepare/prepare.js +0 -2
  218. package/dist/tasks/resolve-akm-bin.js +20 -79
  219. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  220. package/dist/tasks/scheduler-binding.js +18 -238
  221. package/dist/tasks/scheduler-invocation.js +52 -52
  222. package/dist/tasks/scheduler-lock.js +53 -0
  223. package/dist/tasks/scheduler-sync.js +363 -679
  224. package/dist/tasks/source/parse-task-source.js +160 -10
  225. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  226. package/dist/tasks/source/task-to-v4.js +2 -2
  227. package/dist/workflows/authoring/authoring.js +3 -12
  228. package/dist/workflows/compile.js +211 -0
  229. package/dist/workflows/concurrency-policy.js +13 -74
  230. package/dist/workflows/exec/child-invocation.js +3 -17
  231. package/dist/workflows/exec/child-workflow.js +32 -141
  232. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  233. package/dist/workflows/exec/environment.js +98 -0
  234. package/dist/workflows/exec/exec-unit.js +33 -140
  235. package/dist/workflows/exec/frozen-judge.js +7 -59
  236. package/dist/workflows/exec/native-executor.js +82 -341
  237. package/dist/workflows/exec/param-secrets.js +29 -47
  238. package/dist/workflows/exec/run-workflow.js +154 -387
  239. package/dist/workflows/exec/scheduler.js +9 -36
  240. package/dist/workflows/exec/step-work.js +127 -430
  241. package/dist/workflows/exec/unit-dispatch.js +11 -63
  242. package/dist/workflows/exec/unit-writer.js +8 -52
  243. package/dist/workflows/exec/worktree.js +39 -273
  244. package/dist/workflows/freeze/child-output-references.js +4 -15
  245. package/dist/workflows/freeze/environment.js +99 -92
  246. package/dist/workflows/freeze/freeze.js +172 -0
  247. package/dist/workflows/freeze/step-values.js +19 -21
  248. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  249. package/dist/workflows/freeze/targets/command.js +10 -33
  250. package/dist/workflows/freeze/targets/script.js +5 -12
  251. package/dist/workflows/freeze/targets/shell.js +3 -6
  252. package/dist/workflows/freeze/targets/task.js +25 -80
  253. package/dist/workflows/freeze/task-bindings.js +20 -67
  254. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  255. package/dist/workflows/ir/params.js +6 -51
  256. package/dist/workflows/ir/plan-hash.js +2 -34
  257. package/dist/workflows/parser.js +140 -43
  258. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  259. package/dist/workflows/renderer.js +36 -69
  260. package/dist/workflows/resource-limits.js +12 -120
  261. package/dist/workflows/runtime/agent-identity.js +8 -40
  262. package/dist/workflows/runtime/run-outputs.js +3 -6
  263. package/dist/workflows/runtime/run-plan.js +316 -0
  264. package/dist/workflows/runtime/runs.js +48 -200
  265. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  266. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  267. package/dist/workflows/validate-summary.js +2 -7
  268. package/docs/integration/bundling-akm.md +49 -42
  269. package/docs/migration/README.md +1 -0
  270. package/docs/migration/release-notes/0.9.17.md +41 -0
  271. package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
  272. package/docs/reference/cli.md +182 -125
  273. package/docs/reference/configuration.md +49 -56
  274. package/docs/reference/data-and-telemetry.md +19 -20
  275. package/docs/reference/tasks.md +86 -38
  276. package/docs/reference/workflow-schema.md +14 -18
  277. package/docs/reference/workflows.md +6 -9
  278. package/package.json +1 -1
  279. package/schemas/akm-config.json +87 -406
  280. package/dist/commands/health/advisories.js +0 -150
  281. package/dist/commands/health/metrics.js +0 -329
  282. package/dist/commands/health/surfaces.js +0 -102
  283. package/dist/commands/improve/anti-collapse.js +0 -83
  284. package/dist/commands/improve/collapse-detector.js +0 -432
  285. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  286. package/dist/commands/improve/consolidate/merge.js +0 -146
  287. package/dist/commands/improve/distill/promote-memory.js +0 -329
  288. package/dist/commands/improve/distill/quality-gate.js +0 -500
  289. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  290. package/dist/commands/improve/proposal-envelope.js +0 -31
  291. package/dist/commands/improve/run-context.js +0 -123
  292. package/dist/commands/improve/shared.js +0 -21
  293. package/dist/commands/improve/source-identity.js +0 -28
  294. package/dist/commands/improve/triage.js +0 -96
  295. package/dist/commands/proposal/drain-policies.js +0 -151
  296. package/dist/commands/sources/update-transaction.js +0 -220
  297. package/dist/core/action-contributors.js +0 -28
  298. package/dist/core/config/config-version-shim.js +0 -101
  299. package/dist/core/config/retired-experimental-keys-shim.js +0 -62
  300. package/dist/core/fs-txn.js +0 -405
  301. package/dist/core/lexical-score.js +0 -25
  302. package/dist/core/maintenance-barrier.js +0 -167
  303. package/dist/execution/executable-identity.js +0 -105
  304. package/dist/execution/guarded-source.js +0 -427
  305. package/dist/indexer/graph/graph-boost.js +0 -427
  306. package/dist/indexer/graph/graph-dedup.js +0 -95
  307. package/dist/indexer/search/name-match.js +0 -35
  308. package/dist/indexer/search/ranking-contributors.js +0 -515
  309. package/dist/indexer/walk/project-context.js +0 -192
  310. package/dist/integrations/agent/execution-cascade.js +0 -566
  311. package/dist/integrations/agent/execution-definitions.js +0 -202
  312. package/dist/integrations/agent/execution-lowering.js +0 -841
  313. package/dist/integrations/agent/execution-preparation.js +0 -98
  314. package/dist/integrations/agent/inline-execution.js +0 -74
  315. package/dist/registry/create-provider-registry.js +0 -29
  316. package/dist/registry/pinned-request-helper.js +0 -247
  317. package/dist/registry/pinned-transport.js +0 -717
  318. package/dist/sources/providers/index.js +0 -14
  319. package/dist/storage/engines/sqlite-migrations.js +0 -271
  320. package/dist/storage/repositories/canaries-repository.js +0 -107
  321. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  322. package/dist/storage/repositories/registry-cache.js +0 -113
  323. package/dist/tasks/scheduler-sync-preview.js +0 -52
  324. package/dist/workflows/freeze/resolve-steps.js +0 -86
  325. package/dist/workflows/freeze/source-freeze.js +0 -64
  326. package/dist/workflows/ir/compile.js +0 -321
  327. package/dist/workflows/ir/environment-v4.js +0 -330
  328. package/dist/workflows/ir/freeze-v4.js +0 -153
  329. package/dist/workflows/ir/schema-v4.js +0 -745
  330. package/dist/workflows/ir/schema.js +0 -354
  331. package/dist/workflows/program/schema.js +0 -78
  332. package/dist/workflows/runtime/checkin.js +0 -57
  333. package/dist/workflows/runtime/plan-classifier.js +0 -196
  334. package/dist/workflows/runtime/unit-checkin.js +0 -45
  335. package/dist/workflows/runtime/unit-phases.js +0 -20
  336. package/dist/workflows/schema.js +0 -4
  337. package/dist/workflows/source-ir/compile.js +0 -200
  338. package/dist/workflows/source-ir/program.js +0 -50
  339. package/dist/workflows/source-ir/result.js +0 -26
  340. package/dist/workflows/source-ir/schema.js +0 -786
  341. package/dist/workflows/source-ir/triggers.js +0 -79
  342. package/dist/workflows/source-ir/uses.js +0 -40
  343. package/dist/workflows/validator.js +0 -60
@@ -1,23 +1,9 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- /**
5
- * Shared memory-content hashing primitives, extracted from the deleted
6
- * `dedup.ts` (#617 dedup pre-pass, removed WI-7.3) so `consolidate.ts` /
7
- * `consolidate/chunking.ts` / `distill.ts` keep a stable, dependency-free home
8
- * for the case-preserving stripped-body hash they use for the body-embedding
9
- * cache and (formerly) the fidelity-check body comparison.
10
- *
11
- * @module content-hash
12
- */
13
4
  import { createHash } from "node:crypto";
14
- import { parseFrontmatter } from "../../core/asset/frontmatter.js";
15
- /**
16
- * Strip frontmatter from raw memory content, returning the body text trimmed.
17
- * Case and whitespace are preserved. Falls back to `raw.trim()` on
18
- * unparseable frontmatter (consistent with the pre-existing load-time hot
19
- * guard).
20
- */
5
+ import { computeNormalizedContentHash, parseFrontmatter } from "../../core/asset/frontmatter.js";
6
+ /** The markdown body with its frontmatter removed, trimmed (the raw text when it does not parse). */
21
7
  export function stripFrontmatterBody(raw) {
22
8
  try {
23
9
  return parseFrontmatter(raw).content.trim();
@@ -27,13 +13,19 @@ export function stripFrontmatterBody(raw) {
27
13
  }
28
14
  }
29
15
  /**
30
- * Hash used for change-detection and the body-embedding cache: case-/
31
- * whitespace-preserving stripped body. Two memories with the same wording
32
- * but different casing produce DIFFERENT hashes here, which is intentional —
33
- * we embed the exact text and cache by its precise content.
34
- *
35
- * This is the `content_hash` stored in `body_embeddings`.
16
+ * The one "is this the same content?" hash for improve and the proposal queue
17
+ * (sha256, hex):
18
+ * - `raw`: the exact bytes — proposal before/after and judged-content hashes,
19
+ * session transcripts, cache keys for plain text.
20
+ * - `body`: the body without frontmatter, case and wording preserved — memory
21
+ * and knowledge dedup and the body-embedding cache.
22
+ * - `normalized`: the whole asset minus akm's bookkeeping frontmatter
23
+ * (`BOOKKEEPING_FRONTMATTER_KEYS`), keys sorted — proposal freshness, so a
24
+ * salience or inference rewrite of the target never stales a proposal.
36
25
  */
37
- export function cacheHash(raw) {
38
- return createHash("sha256").update(stripFrontmatterBody(raw), "utf8").digest("hex");
26
+ export function contentHash(content, mode = "raw") {
27
+ if (mode === "raw")
28
+ return createHash("sha256").update(content).digest("hex");
29
+ const text = typeof content === "string" ? content : Buffer.from(content).toString("utf8");
30
+ return mode === "body" ? contentHash(stripFrontmatterBody(text)) : computeNormalizedContentHash(text);
39
31
  }
@@ -1,33 +1,16 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- /**
5
- * Pure content-repair + quality-validation stages for `akm distill`.
6
- *
7
- * Extracted verbatim from the inline body of `akmDistill` so each normalization
8
- * pass is an independently testable unit. Every function is a pure transform of
9
- * `(content, inputRef) → content | findings` with no I/O — logic is
10
- * byte-identical to the pre-extraction inline code. The lesson-path guard
11
- * (`effectiveProposalKind !== "knowledge"`) stays in the caller; these helpers
12
- * assume the lesson path.
13
- */
4
+ /** Deterministic lesson repairs and checks for `akm distill` (pure; the caller keeps the lesson-path guard). */
14
5
  import { assembleAssetFromString, serializeFrontmatterQuoted } from "../../../core/asset/asset-serialize.js";
15
6
  import { parseFrontmatter } from "../../../core/asset/frontmatter.js";
16
7
  import { repairTruncatedDescription } from "../../../core/text-truncation.js";
17
- import { detectDoubleFrontmatter, isValidDescription, isValidWhenToUse, } from "../../proposal/validators/proposal-quality-validators.js";
8
+ import { isValidDescription, isValidWhenToUse, lessonQualityIssues, } from "../../proposal/validators/proposal-quality-validators.js";
18
9
  /**
19
- * Auto-repair missing frontmatter fields before hard-failing. Small models
20
- * frequently produce a good lesson body but omit the YAML header entirely.
21
- * Rather than discarding valid content, we extract description/when_to_use
22
- * from the body and prepend the required frontmatter block.
23
- *
24
- * IMPORTANT: We do NOT synthesise placeholder strings here. If the body
25
- * does not contain text that passes the post-LLM validators
26
- * (`isValidDescription` / `isValidWhenToUse`), we leave the field missing
27
- * and let the lesson lint reject the proposal as `validation_failed`.
28
- * Emitting placeholders like `"Lesson distilled from <ref>"` or
29
- * `"When working with <slug>"` is what produced the systematic broken
30
- * proposals observed across 323 archived rejections.
10
+ * Fill a missing description / when_to_use from body lines that pass their
11
+ * validators — small models often write a good body with no header. Never a
12
+ * placeholder: those produced hundreds of broken proposals; a field nothing
13
+ * qualifies for stays missing for lint to reject.
31
14
  */
32
15
  export function autoRepairLessonFrontmatter(content, inputRef) {
33
16
  const parsed = parseFrontmatter(content);
@@ -37,7 +20,6 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
37
20
  if (!missingDesc && !missingWtu)
38
21
  return content;
39
22
  const body = parsed.content.trim();
40
- // Strip markdown formatting tokens from a line so extracted text is clean.
41
23
  const stripMd = (l) => l
42
24
  .replace(/\*\*([^*]+)\*\*/g, "$1")
43
25
  .replace(/\*([^*]+)\*/g, "$1")
@@ -45,14 +27,9 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
45
27
  .replace(/^[#*\->_]+\s*/, "")
46
28
  .replace(/:\s*$/, "")
47
29
  .trim();
48
- // Skip lines that look like YAML field assignments (key: value) or frontmatter delimiters.
49
- // These appear when the LLM leaks frontmatter content into the body, causing
50
- // auto-repair to produce description: "description: Key Takeaways".
30
+ // Leaked frontmatter lines in the body would yield `description: "description: …"`.
51
31
  const isYamlLike = (l) => /^---/.test(l) || /^[a-z_]+:\s/i.test(l);
52
32
  const bodyLines = body.split("\n").map(stripMd);
53
- // Extract description: first body line that BOTH looks like prose AND
54
- // passes isValidDescription. If nothing qualifies, leave the field
55
- // missing — the lint pass will reject the proposal cleanly.
56
33
  let descLine;
57
34
  for (const l of bodyLines) {
58
35
  if (isYamlLike(l))
@@ -64,8 +41,6 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
64
41
  break;
65
42
  }
66
43
  }
67
- // Extract when_to_use: a line starting with "When" / "Use when" / "Apply when"
68
- // that ALSO passes isValidWhenToUse (rejects circular fallbacks).
69
44
  let wtuLine;
70
45
  for (const l of bodyLines) {
71
46
  if (!/^(when |use when|apply when)/i.test(l))
@@ -83,23 +58,15 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
83
58
  ...(missingWtu && wtuLine ? { when_to_use: wtuLine } : {}),
84
59
  };
85
60
  const fmLines = serializeFrontmatterQuoted(repairedFm);
86
- // Only rewrite content if we actually have at least one field to write.
87
- // Otherwise leave the original content for the lint pass to reject.
88
61
  if (Object.keys(repairedFm).length > 0) {
89
62
  return assembleAssetFromString(fmLines, body);
90
63
  }
91
64
  return content;
92
65
  }
93
66
  /**
94
- * Description ↔ when_to_use auto-swap normalization (recover ~93% of
95
- * qwen-9b's `^when\b/i` rejections at zero LLM cost). When the LLM emits
96
- * a conditional-framed description ("When X happens, do Y") and the
97
- * when_to_use field looks like a declarative description (or is empty),
98
- * the two fields are mis-fielded — exactly what `isValidDescription`'s
99
- * error message says ("that pattern belongs in when_to_use"). We swap
100
- * them and revalidate; the swap is committed only if BOTH fields pass
101
- * their respective validators afterwards. If revalidation still fails,
102
- * we fall through returning the original content (swapped: 0).
67
+ * Swap a conditional description ("When X, do Y") with a declarative
68
+ * when_to_use — mis-fielded, as the description validator says — when both
69
+ * then pass; this recovers most `^when` rejections at no LLM cost.
103
70
  */
104
71
  export function autoSwapDescriptionWhenToUse(content, inputRef) {
105
72
  const parsedSwap = parseFrontmatter(content);
@@ -109,9 +76,6 @@ export function autoSwapDescriptionWhenToUse(content, inputRef) {
109
76
  const descStartsConditional = /^(when|if)\b/i.test(descRaw);
110
77
  const wtuStartsConditional = /^(when|if)\b/i.test(wtuRaw);
111
78
  if (descStartsConditional && !wtuStartsConditional && wtuRaw.length > 0) {
112
- // Try the swap and revalidate. The when_to_use validator requires the
113
- // value not match `/^when working with\b/i` (the circular fallback) —
114
- // a real description rarely does, so this usually passes.
115
79
  const swappedDescCheck = isValidDescription(wtuRaw, inputRef);
116
80
  const swappedWtuCheck = isValidWhenToUse(descRaw, inputRef);
117
81
  if (swappedDescCheck.ok && swappedWtuCheck.ok) {
@@ -126,13 +90,7 @@ export function autoSwapDescriptionWhenToUse(content, inputRef) {
126
90
  }
127
91
  return { content, swapped: 0 };
128
92
  }
129
- /**
130
- * Post-generation truncation repair (#556): if the LLM sliced the
131
- * description mid-sentence, deterministically complete it from its own text
132
- * / the lesson body BEFORE the lint + quality validators run. No-op
133
- * (byte-identical) for already-complete descriptions, so this never alters
134
- * a valid proposal.
135
- */
93
+ /** Complete a description cut mid-sentence from its own text or the body (#556); a complete one is untouched. */
136
94
  export function repairLessonDescriptionTruncation(content) {
137
95
  const parsedRepair = parseFrontmatter(content);
138
96
  const fmRepair = (parsedRepair.data ?? {});
@@ -145,52 +103,12 @@ export function repairLessonDescriptionTruncation(content) {
145
103
  const repairedFmLines = serializeFrontmatterQuoted({ ...fmRepair, description: repaired });
146
104
  return assembleAssetFromString(repairedFmLines, parsedRepair.content);
147
105
  }
148
- /**
149
- * Additional quality validators that run only on lessons whose lesson-lint
150
- * pass was clean. lesson-lint checks "field is present and non-empty"; these
151
- * reject the systematic failure modes observed across 323 archived rejected
152
- * proposals:
153
- * - description is a body fragment, section heading, or placeholder
154
- * - when_to_use is the circular "When working with <ref>" fallback
155
- * - description == when_to_use (LLM duplicated a single sentence)
156
- * - body contains a second pseudo-frontmatter block
157
- */
106
+ /** The shared lesson quality checks, for a lesson whose lint pass was clean. */
158
107
  export function collectLessonQualityFindings(content, inputRef) {
159
- const findings = [];
160
- const parsedQC = parseFrontmatter(content);
161
- const fmQC = (parsedQC.data ?? {});
162
- const descCheck = isValidDescription(fmQC.description, inputRef);
163
- if (!descCheck.ok) {
164
- findings.push({
165
- kind: "invalid-description",
166
- field: "description",
167
- message: `Distilled lesson for ${inputRef} has an invalid description: ${descCheck.reason}.`,
168
- });
169
- }
170
- const wtuCheck = isValidWhenToUse(fmQC.when_to_use, inputRef);
171
- if (!wtuCheck.ok) {
172
- findings.push({
173
- kind: "invalid-when_to_use",
174
- field: "when_to_use",
175
- message: `Distilled lesson for ${inputRef} has an invalid when_to_use: ${wtuCheck.reason}.`,
176
- });
177
- }
178
- // description and when_to_use must say different things.
179
- if (descCheck.ok &&
180
- wtuCheck.ok &&
181
- typeof fmQC.description === "string" &&
182
- typeof fmQC.when_to_use === "string" &&
183
- fmQC.description.trim().toLowerCase() === fmQC.when_to_use.trim().toLowerCase()) {
184
- findings.push({
185
- kind: "description-equals-when_to_use",
186
- field: "description",
187
- message: `Distilled lesson for ${inputRef} has identical description and when_to_use.`,
188
- });
189
- }
190
- // Double-frontmatter / pseudo-frontmatter pollution in the body.
191
- const dfm = detectDoubleFrontmatter(content);
192
- if (dfm) {
193
- findings.push({ kind: dfm.kind, field: "body", message: `Distilled lesson for ${inputRef}: ${dfm.message}` });
194
- }
195
- return findings;
108
+ const fm = (parseFrontmatter(content).data ?? {});
109
+ return lessonQualityIssues(fm, content, inputRef).map((issue) => ({
110
+ kind: issue.kind,
111
+ field: issue.field,
112
+ message: `Distilled lesson for ${inputRef}${issue.text}`,
113
+ }));
196
114
  }
@@ -2,34 +2,12 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /**
5
- * WS-3b distill-stage guards.
6
- *
7
- * **CLS interleaving (step 9)**
8
- * distill/memoryInference prompts include embedding-retrieved adjacent
9
- * lessons/knowledge so the pipeline doesn't overwrite prior generalizations.
10
- *
11
- * **Distill→source fidelity (step 10)**
12
- * After a distill proposal, check it against cited source memories; a
13
- * contradiction flag routes to human review.
14
- *
15
- * @module distill-guards
5
+ * Distill guards: related lessons/knowledge shown to the model so it does not
6
+ * overwrite prior generalizations (CLS context), and a cheap check that a
7
+ * proposal does not contradict the memories it came from.
16
8
  */
17
- // ── CLS adjacent lesson context (step 9) ─────────────────────────────────────
18
- /** Default number of adjacent lessons/knowledge for CLS interleaving. */
19
9
  export const DEFAULT_CLS_ADJACENT_COUNT = 3;
20
- /**
21
- * Build a CLS (Complementary Learning System) context snippet for injection
22
- * into distill/memoryInference prompts.
23
- *
24
- * Given a list of embedding-retrieved adjacent lessons/knowledge, formats them
25
- * as a markdown section to append to the prompt so the LLM avoids overwriting
26
- * prior generalizations.
27
- *
28
- * Returns an empty string when CLS is disabled or no adjacent items are found.
29
- *
30
- * @param adjacentItems - Top-N adjacent lessons/knowledge retrieved by embedding.
31
- * @param config - CLS config.
32
- */
10
+ /** The CLS prompt section (each entry capped at 400 chars); empty when disabled or nothing is related. */
33
11
  export function buildClsContext(adjacentItems, config) {
34
12
  if (!config.enabled || adjacentItems.length === 0)
35
13
  return "";
@@ -41,54 +19,23 @@ export function buildClsContext(adjacentItems, config) {
41
19
  "disagree with one, flag it as contradicted (do not ignore it).",
42
20
  "",
43
21
  ];
44
- for (const item of adjacentItems) {
45
- lines.push(`### ${item.ref}`);
46
- // Truncate to 400 chars to keep the prompt size reasonable.
47
- lines.push(item.content.trim().slice(0, 400));
48
- lines.push("");
49
- }
22
+ for (const item of adjacentItems)
23
+ lines.push(`### ${item.ref}`, item.content.trim().slice(0, 400), "");
50
24
  return lines.join("\n");
51
25
  }
52
26
  /**
53
- * Check a distill proposal against its cited source memories for contradictions.
54
- *
55
- * Uses a simple heuristic: looks for explicit negation of key claims in the
56
- * proposal body that appear in the source bodies. A full LLM-based
57
- * contradiction check is expensive (one LLM call per proposal); this cheap
58
- * heuristic catches the most obvious cases and flags them for human review.
59
- *
60
- * When `fidelityCheck.enabled` is false, returns `{ contradictionDetected: false }`
61
- * immediately (no work done).
62
- *
63
- * @param proposalBody - The stripped body of the distill proposal.
64
- * @param sourceBodies - The stripped bodies of the cited source memories.
65
- * @param config - Fidelity check config.
27
+ * Flag a proposal whose "always/must X" (or "never/must not X") claim meets
28
+ * the opposite claim about X in a source. Deliberately conservative: a flag
29
+ * only costs a human review, while a model call per proposal is expensive.
66
30
  */
67
31
  export function checkDistillFidelity(proposalBody, sourceBodies, config) {
68
- if (!config.enabled || sourceBodies.length === 0) {
69
- return { contradictionDetected: false };
70
- }
71
- // Heuristic: detect explicit negation of "never" / "always" / "must" claims.
72
- // A proposal that says "always X" while the source says "never X" (or vice
73
- // versa) is a clear contradiction worth flagging.
74
- //
75
- // This is intentionally conservative: it only flags when both the proposal
76
- // AND the source contain the opposing polarity of the same key term. False
77
- // negatives (missed contradictions) are preferred over false positives
78
- // (blocking valid proposals) since the consequence of a false positive is
79
- // a human review request, while the cost of a false negative is a slightly
80
- // degraded stash.
81
- const proposalLow = proposalBody.toLowerCase();
82
- // Extract "always/never/must/must not" claims from the proposal.
83
- const strongClaims = extractStrongClaims(proposalLow);
84
- if (strongClaims.length === 0)
32
+ if (!config.enabled || sourceBodies.length === 0)
85
33
  return { contradictionDetected: false };
34
+ const strongClaims = extractStrongClaims(proposalBody.toLowerCase());
86
35
  for (const sourceBody of sourceBodies) {
87
36
  const sourceLow = sourceBody.toLowerCase();
88
37
  for (const { polarity, term } of strongClaims) {
89
- const oppositePolarity = polarity === "positive" ? "negative" : "positive";
90
- const sourceHasOpposite = hasStrongClaim(sourceLow, term, oppositePolarity);
91
- if (sourceHasOpposite) {
38
+ if (hasStrongClaim(sourceLow, term, polarity === "positive" ? "negative" : "positive")) {
92
39
  return {
93
40
  contradictionDetected: true,
94
41
  reason: `Proposal makes a ${polarity} strong claim about "${term}" that conflicts with an opposing claim in a cited source. Route to human review.`,
@@ -96,32 +43,24 @@ export function checkDistillFidelity(proposalBody, sourceBodies, config) {
96
43
  }
97
44
  }
98
45
  }
99
- // Also flag proposals whose xrefs are empty (broken provenance).
100
- // This is a degradation signal, not a contradiction, but worth surfacing.
101
46
  return { contradictionDetected: false };
102
47
  }
48
+ const CLAIM_PATTERNS = [
49
+ { polarity: "positive", re: /\b(?:always|must)\s+(\w+)/g },
50
+ { polarity: "negative", re: /\b(?:never|must\s+not|should\s+not)\s+(\w+)/g },
51
+ ];
103
52
  function extractStrongClaims(text) {
104
53
  const claims = [];
105
- // Match "always <term>", "never <term>", "must <term>", "must not <term>".
106
- const patterns = [
107
- { polarity: "positive", re: /\b(?:always|must)\s+(\w+)/g },
108
- { polarity: "negative", re: /\b(?:never|must\s+not|should\s+not)\s+(\w+)/g },
109
- ];
110
- for (const { polarity, re } of patterns) {
111
- re.lastIndex = 0;
112
- let m = re.exec(text);
113
- while (m !== null) {
54
+ for (const { polarity, re } of CLAIM_PATTERNS) {
55
+ for (const m of text.matchAll(re)) {
114
56
  const term = m[1];
115
57
  if (term && term.length > 2)
116
58
  claims.push({ polarity, term });
117
- m = re.exec(text);
118
59
  }
119
60
  }
120
61
  return claims;
121
62
  }
122
63
  function hasStrongClaim(text, term, polarity) {
123
- if (polarity === "positive") {
124
- return /\b(?:always|must)\s/.test(text) && text.includes(term);
125
- }
126
- return /\b(?:never|must\s+not|should\s+not)\s/.test(text) && text.includes(term);
64
+ const marker = polarity === "positive" ? /\b(?:always|must)\s/ : /\b(?:never|must\s+not|should\s+not)\s/;
65
+ return marker.test(text) && text.includes(term);
127
66
  }
@@ -216,249 +216,29 @@ function assessWithWeightedModel(input, model, threshold) {
216
216
  modelName: model.name,
217
217
  };
218
218
  }
219
- function precision(tp, fp) {
220
- return tp + fp === 0 ? 1 : tp / (tp + fp);
221
- }
222
- function recall(tp, fn) {
223
- return tp + fn === 0 ? 1 : tp / (tp + fn);
224
- }
225
- function f1Score(p, r) {
226
- return p + r === 0 ? 0 : (2 * p * r) / (p + r);
227
- }
228
- function casePromoteValue(testCase) {
229
- return testCase.promoteValue ?? 3;
230
- }
231
- function caseFalsePromoteCost(testCase) {
232
- return testCase.falsePromoteCost ?? 4;
233
- }
234
- function caseMissedPromoteCost(testCase) {
235
- return testCase.missedPromoteCost ?? 2;
236
- }
237
- export function evaluateMemoryPromotionBenchmark(cases, policy = DEFAULT_PROMOTION_POLICY) {
238
- const results = cases.map((fixture) => {
239
- const assessment = policy.assess(fixture.input);
240
- const passed = assessment.promote === fixture.expectPromote;
241
- return {
242
- fixture,
243
- name: fixture.name,
244
- expectPromote: fixture.expectPromote,
245
- assessment,
246
- passed,
247
- };
248
- });
249
- const truePositives = results.filter((result) => result.assessment.promote && result.expectPromote).length;
250
- const trueNegatives = results.filter((result) => !result.assessment.promote && !result.expectPromote).length;
251
- const falsePositives = results.filter((result) => result.assessment.promote && !result.expectPromote).length;
252
- const falseNegatives = results.filter((result) => !result.assessment.promote && result.expectPromote).length;
253
- const correct = truePositives + trueNegatives;
254
- const p = precision(truePositives, falsePositives);
255
- const r = recall(truePositives, falseNegatives);
256
- let netOutcomeScore = 0;
257
- let capturedPromoteValue = 0;
258
- let preventedFalsePromotionCost = 0;
259
- for (const result of results) {
260
- if (result.expectPromote && result.assessment.promote) {
261
- const value = casePromoteValue(result.fixture);
262
- netOutcomeScore += value;
263
- capturedPromoteValue += value;
264
- }
265
- else if (result.expectPromote && !result.assessment.promote) {
266
- netOutcomeScore -= caseMissedPromoteCost(result.fixture);
267
- }
268
- else if (!result.expectPromote && result.assessment.promote) {
269
- netOutcomeScore -= caseFalsePromoteCost(result.fixture);
270
- }
271
- else {
272
- preventedFalsePromotionCost += caseFalsePromoteCost(result.fixture);
273
- }
274
- }
275
- return {
276
- total: results.length,
277
- correct,
278
- falsePositives,
279
- falseNegatives,
280
- accuracy: results.length === 0 ? 1 : correct / results.length,
281
- precision: p,
282
- recall: r,
283
- f1: f1Score(p, r),
284
- truePositives,
285
- trueNegatives,
286
- netOutcomeScore,
287
- capturedPromoteValue,
288
- preventedFalsePromotionCost,
289
- results: results.map(({ name, expectPromote, assessment, passed }) => ({
290
- name,
291
- expectPromote,
292
- assessment,
293
- passed,
294
- })),
295
- };
296
- }
297
- function thresholdCandidates() {
298
- const values = [];
299
- for (let value = 2.4; value <= 4.2; value += 0.2) {
300
- values.push(Number(value.toFixed(1)));
301
- }
302
- return values;
303
- }
304
- const POSITIVE_FEEDBACK_BASELINE = {
305
- name: "baseline-positive-feedback",
306
- threshold: 2,
307
- assess(input) {
308
- const knowledgeRef = deriveKnowledgeRef(input.inputRef);
309
- const featureState = collectPromotionFeatures(input);
310
- if (featureState.blockedBy.length > 0) {
311
- return {
312
- applicable: !featureState.blockedBy.includes("not-memory"),
313
- promote: false,
314
- score: 0,
315
- threshold: 2,
316
- knowledgeRef,
317
- blockedBy: featureState.blockedBy,
318
- positiveSignals: [],
319
- negativeSignals: [],
320
- modelName: "baseline-positive-feedback",
321
- };
322
- }
323
- const features = featureState.features;
324
- const promote = features.positiveFeedback >= 2;
325
- return {
326
- applicable: true,
327
- promote,
328
- score: features.positiveFeedback,
329
- threshold: 2,
330
- knowledgeRef,
331
- ...(promote ? { content: buildKnowledgeContent(input) } : {}),
332
- blockedBy: [],
333
- positiveSignals: promote ? ["baseline positive feedback rule"] : [],
334
- negativeSignals: promote ? [] : ["baseline positive feedback rule not met"],
335
- modelName: "baseline-positive-feedback",
336
- };
337
- },
338
- };
339
- const METADATA_BASELINE = {
340
- name: "baseline-metadata",
341
- threshold: 2,
342
- assess(input) {
343
- const knowledgeRef = deriveKnowledgeRef(input.inputRef);
344
- const featureState = collectPromotionFeatures(input);
345
- if (featureState.blockedBy.length > 0) {
346
- return {
347
- applicable: !featureState.blockedBy.includes("not-memory"),
348
- promote: false,
349
- score: 0,
350
- threshold: 2,
351
- knowledgeRef,
352
- blockedBy: featureState.blockedBy,
353
- positiveSignals: [],
354
- negativeSignals: [],
355
- modelName: "baseline-metadata",
356
- };
357
- }
358
- const features = featureState.features;
359
- const metadataScore = (features.hasSource ? 1 : 0) + (features.hasObservedAt ? 1 : 0);
360
- const promote = metadataScore >= 2;
361
- return {
362
- applicable: true,
363
- promote,
364
- score: metadataScore,
365
- threshold: 3,
366
- knowledgeRef,
367
- ...(promote ? { content: buildKnowledgeContent(input) } : {}),
368
- blockedBy: [],
369
- positiveSignals: promote ? ["baseline metadata rule"] : [],
370
- negativeSignals: promote ? [] : ["baseline metadata rule not met"],
371
- modelName: "baseline-metadata",
372
- };
373
- },
374
- };
375
- export function selectPromotionPolicy(corpus, candidates) {
376
- const trainingCases = corpus.filter((testCase) => (testCase.split ?? "train") === "train");
377
- const heldOutCases = corpus.filter((testCase) => (testCase.split ?? "train") === "heldout");
378
- let bestPolicy;
379
- let bestTraining;
380
- for (const model of candidates) {
381
- for (const threshold of thresholdCandidates()) {
382
- const policy = {
383
- name: model.name,
384
- threshold,
385
- assess: (input) => assessWithWeightedModel(input, model, threshold),
386
- };
387
- const training = evaluateMemoryPromotionBenchmark(trainingCases, policy);
388
- if (!bestTraining) {
389
- bestTraining = training;
390
- bestPolicy = policy;
391
- continue;
392
- }
393
- const trainingWins = training.f1 > bestTraining.f1 ||
394
- (training.f1 === bestTraining.f1 && training.netOutcomeScore > bestTraining.netOutcomeScore) ||
395
- (training.f1 === bestTraining.f1 &&
396
- training.netOutcomeScore === bestTraining.netOutcomeScore &&
397
- training.accuracy > bestTraining.accuracy);
398
- if (trainingWins) {
399
- bestTraining = training;
400
- bestPolicy = policy;
401
- }
402
- }
403
- }
404
- const selectedPolicy = bestPolicy;
405
- const selectedTraining = bestTraining;
406
- const heldOut = evaluateMemoryPromotionBenchmark(heldOutCases, selectedPolicy);
407
- const baselines = [POSITIVE_FEEDBACK_BASELINE, METADATA_BASELINE].map((policy) => {
408
- const baselineHeldOut = evaluateMemoryPromotionBenchmark(heldOutCases, policy);
409
- const noWorseThanSelected = heldOut.f1 >= baselineHeldOut.f1 && heldOut.netOutcomeScore >= baselineHeldOut.netOutcomeScore;
410
- const strictWinMetrics = [];
411
- if (heldOut.f1 > baselineHeldOut.f1)
412
- strictWinMetrics.push("f1");
413
- if (heldOut.netOutcomeScore > baselineHeldOut.netOutcomeScore)
414
- strictWinMetrics.push("netOutcomeScore");
415
- if (heldOut.accuracy > baselineHeldOut.accuracy)
416
- strictWinMetrics.push("accuracy");
417
- return {
418
- name: policy.name,
419
- heldOut: baselineHeldOut,
420
- noWorseThanSelected,
421
- strictWin: noWorseThanSelected && strictWinMetrics.length > 0,
422
- strictWinMetrics,
423
- };
424
- });
425
- const strictlyBeatsBaselines = baselines.every((baseline) => baseline.strictWin);
426
- return {
427
- corpusSize: corpus.length,
428
- trainingSize: trainingCases.length,
429
- heldOutSize: heldOutCases.length,
430
- selectedModel: { name: selectedPolicy.name, threshold: selectedPolicy.threshold },
431
- training: selectedTraining,
432
- heldOut,
433
- baselines,
434
- strictlyBeatsBaselines,
435
- };
436
- }
437
- export const DEFAULT_PROMOTION_POLICY_SELECTION = {
438
- selectedModel: {
439
- name: "balanced-evidence",
440
- positiveWeight: 0.8,
441
- repeatedPositiveWeight: 0.65,
442
- noPositivePenalty: 0.9,
443
- singlePositivePenalty: 0.7,
444
- negativeWeight: 2.0,
445
- curatedWeight: 0.55,
446
- confidenceWeight: 0.7,
447
- sourceWeight: 0.4,
448
- observedAtWeight: 0.4,
449
- descriptionWeight: 0.2,
450
- tagWeight: 0.15,
451
- substantiveBodyWeight: 0.15,
452
- tentativePenalty: 1.1,
453
- },
454
- threshold: 3.8,
455
- };
456
- const SELECTED_MODEL = DEFAULT_PROMOTION_POLICY_SELECTION.selectedModel;
457
- export const DEFAULT_PROMOTION_POLICY = {
458
- name: SELECTED_MODEL.name,
459
- threshold: DEFAULT_PROMOTION_POLICY_SELECTION.threshold,
460
- assess: (input) => assessWithWeightedModel(input, SELECTED_MODEL, DEFAULT_PROMOTION_POLICY_SELECTION.threshold),
219
+ /**
220
+ * The memory → knowledge promotion model: a weighted score over feedback
221
+ * reinforcement and memory metadata, promoted at or above the threshold. The
222
+ * weights were chosen by a grid search over a labelled corpus; they are a
223
+ * plain constant now.
224
+ */
225
+ const PROMOTION_MODEL = {
226
+ name: "balanced-evidence",
227
+ positiveWeight: 0.8,
228
+ repeatedPositiveWeight: 0.65,
229
+ noPositivePenalty: 0.9,
230
+ singlePositivePenalty: 0.7,
231
+ negativeWeight: 2.0,
232
+ curatedWeight: 0.55,
233
+ confidenceWeight: 0.7,
234
+ sourceWeight: 0.4,
235
+ observedAtWeight: 0.4,
236
+ descriptionWeight: 0.2,
237
+ tagWeight: 0.15,
238
+ substantiveBodyWeight: 0.15,
239
+ tentativePenalty: 1.1,
461
240
  };
241
+ const PROMOTION_THRESHOLD = 3.8;
462
242
  export function assessMemoryKnowledgePromotionCandidate(input) {
463
- return DEFAULT_PROMOTION_POLICY.assess(input);
243
+ return assessWithWeightedModel(input, PROMOTION_MODEL, PROMOTION_THRESHOLD);
464
244
  }