akm-cli 0.9.17-alpha.3 → 0.9.17-alpha.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (346) hide show
  1. package/CHANGELOG.md +760 -0
  2. package/dist/akm +94 -196
  3. package/dist/cli/shared.js +6 -2
  4. package/dist/cli.js +22 -9
  5. package/dist/commands/agent/agent-dispatch.js +1 -1
  6. package/dist/commands/command/command-execution.js +24 -62
  7. package/dist/commands/feedback-cli.js +0 -1
  8. package/dist/commands/health/accept-rate.js +2 -2
  9. package/dist/commands/health/checks.js +30 -75
  10. package/dist/commands/health/config-skew.js +38 -0
  11. package/dist/commands/health/egress.js +54 -0
  12. package/dist/commands/health/html-report.js +0 -38
  13. package/dist/commands/health/improve-metrics.js +123 -562
  14. package/dist/commands/health/plugin-staleness.js +53 -3
  15. package/dist/commands/health/renderers.js +12 -4
  16. package/dist/commands/health/report-view-model.js +11 -106
  17. package/dist/commands/health/types-improve.js +4 -19
  18. package/dist/commands/health/windows.js +64 -73
  19. package/dist/commands/health.js +122 -143
  20. package/dist/commands/improve/consolidate/chunking.js +25 -100
  21. package/dist/commands/improve/consolidate/sanitize.js +54 -149
  22. package/dist/commands/improve/consolidate.js +538 -1075
  23. package/dist/commands/improve/content-hash.js +16 -24
  24. package/dist/commands/improve/distill/content-repair.js +18 -100
  25. package/dist/commands/improve/distill-guards.js +20 -81
  26. package/dist/commands/improve/distill-promotion-policy.js +23 -243
  27. package/dist/commands/improve/distill.js +608 -1075
  28. package/dist/commands/improve/eligibility.js +126 -400
  29. package/dist/commands/improve/execution.js +3 -5
  30. package/dist/commands/improve/extract.js +487 -1046
  31. package/dist/commands/improve/feedback-valence.js +0 -25
  32. package/dist/commands/improve/improve-cli.js +29 -166
  33. package/dist/commands/improve/improve-result-file.js +10 -66
  34. package/dist/commands/improve/improve-strategies.js +12 -7
  35. package/dist/commands/improve/improve-usage-report.js +18 -64
  36. package/dist/commands/improve/improve.js +443 -1063
  37. package/dist/commands/improve/ledger.js +114 -0
  38. package/dist/commands/improve/locks.js +2 -8
  39. package/dist/commands/improve/loop-stages.js +459 -1172
  40. package/dist/commands/improve/memory/derived-ref.js +12 -77
  41. package/dist/commands/improve/memory/memory-belief.js +14 -118
  42. package/dist/commands/improve/memory/memory-improve.js +4 -3
  43. package/dist/commands/improve/outcome-loop.js +28 -156
  44. package/dist/commands/improve/planner.js +5 -10
  45. package/dist/commands/improve/preparation.js +851 -2339
  46. package/dist/commands/improve/proactive-maintenance.js +34 -101
  47. package/dist/commands/improve/reflect-noise.js +104 -280
  48. package/dist/commands/improve/reflect.js +621 -1367
  49. package/dist/commands/improve/salience.js +46 -232
  50. package/dist/commands/improve/session-asset.js +19 -100
  51. package/dist/commands/improve/stage.js +323 -0
  52. package/dist/commands/lint/base-linter.js +19 -5
  53. package/dist/commands/proposal/drain.js +251 -644
  54. package/dist/commands/proposal/proposal-cli.js +3 -18
  55. package/dist/commands/proposal/proposal-types.js +20 -41
  56. package/dist/commands/proposal/proposal.js +1 -2
  57. package/dist/commands/proposal/propose.js +134 -160
  58. package/dist/commands/proposal/repository.js +502 -1487
  59. package/dist/commands/proposal/validators/proposal-quality-validators.js +71 -174
  60. package/dist/commands/proposal/validators/proposal-validators.js +1 -1
  61. package/dist/commands/proposal/validators/proposals.js +13 -89
  62. package/dist/commands/read/curate.js +63 -413
  63. package/dist/commands/read/search-cli.js +16 -33
  64. package/dist/commands/read/search.js +17 -23
  65. package/dist/commands/read/show.js +2 -13
  66. package/dist/commands/sources/bundle-cli.js +25 -2
  67. package/dist/commands/sources/bundle-config-ops.js +4 -0
  68. package/dist/commands/sources/dangerous-env-audit.js +1 -2
  69. package/dist/commands/sources/info.js +2 -11
  70. package/dist/commands/sources/installed-stashes.js +197 -746
  71. package/dist/commands/sources/schema-repair.js +98 -129
  72. package/dist/commands/sources/source-add.js +62 -12
  73. package/dist/commands/sources/source-manage.js +9 -2
  74. package/dist/commands/sources/stash-cli.js +1 -1
  75. package/dist/commands/tasks/explain.js +10 -13
  76. package/dist/commands/tasks/tasks-cli.js +9 -8
  77. package/dist/commands/tasks/tasks.js +326 -930
  78. package/dist/commands/tasks/validate.js +42 -21
  79. package/dist/commands/workflow/plan.js +22 -29
  80. package/dist/commands/workflow-cli.js +4 -4
  81. package/dist/core/adapter/adapters/akm-adapter.js +0 -1
  82. package/dist/core/adapter/adapters/akm-lint.js +2 -3
  83. package/dist/core/adapter/adapters/akm-metadata.js +11 -12
  84. package/dist/core/adapter/adapters/akm-workflow-adapter.js +1 -1
  85. package/dist/core/adapter/execution-source.js +17 -29
  86. package/dist/core/asset/asset-placement.js +4 -13
  87. package/dist/core/asset/resolve-ref.js +1 -1
  88. package/dist/core/bundle-id.js +42 -5
  89. package/dist/core/bundle-rename.js +291 -0
  90. package/dist/core/config/config-io.js +1 -2
  91. package/dist/core/config/config-schema.js +1 -33
  92. package/dist/core/config/config-walker.js +1 -1
  93. package/dist/core/config/config.js +163 -68
  94. package/dist/core/config/legacy-source-shape-shim.js +38 -9
  95. package/dist/core/config/schema/embedding.js +20 -5
  96. package/dist/core/config/schema/engines.js +5 -0
  97. package/dist/core/config/schema/execution.js +1 -1
  98. package/dist/core/config/schema/experimental.js +1 -1
  99. package/dist/core/config/schema/improve-processes.js +21 -95
  100. package/dist/core/config/schema/improve.js +4 -42
  101. package/dist/core/config/schema/scheduler.js +12 -12
  102. package/dist/core/config/schema/search.js +6 -22
  103. package/dist/core/env-secret-ref.js +0 -1
  104. package/dist/core/errors.js +8 -9
  105. package/dist/core/file-lock.js +76 -173
  106. package/dist/core/logs-db.js +2 -2
  107. package/dist/core/paths.js +0 -27
  108. package/dist/core/redaction.js +109 -2
  109. package/dist/core/run-lock.js +2 -5
  110. package/dist/core/spawn-env.js +1 -1
  111. package/dist/core/state/migrations.js +108 -61
  112. package/dist/core/state-db-scope.js +2 -4
  113. package/dist/core/state-db.js +126 -692
  114. package/dist/core/type-presentation.js +1 -9
  115. package/dist/core/write-source.js +293 -1012
  116. package/dist/execution/input-contract.js +1 -1
  117. package/dist/execution/resolved-request.js +135 -689
  118. package/dist/execution/source.js +63 -257
  119. package/dist/execution/target-ref.js +1 -1
  120. package/dist/indexer/bundle-identity-guard.js +2 -2
  121. package/dist/indexer/db/graph-db.js +106 -46
  122. package/dist/indexer/ensure-index.js +44 -85
  123. package/dist/indexer/graph/graph-extraction.js +340 -562
  124. package/dist/indexer/graph/graph-related.js +130 -0
  125. package/dist/indexer/index-rebuild-lock.js +3 -11
  126. package/dist/indexer/index-writer-lock.js +8 -17
  127. package/dist/indexer/index-written-assets.js +139 -151
  128. package/dist/indexer/indexer.js +524 -846
  129. package/dist/indexer/materialize-embeddings.js +60 -397
  130. package/dist/indexer/passes/memory-inference.js +81 -90
  131. package/dist/indexer/passes/metadata.js +132 -200
  132. package/dist/indexer/read-preflight.js +0 -7
  133. package/dist/indexer/scan/doc-to-entry.js +1 -3
  134. package/dist/indexer/scan/drain-dir.js +1 -1
  135. package/dist/indexer/search/db-search.js +181 -590
  136. package/dist/indexer/search/fts-query.js +30 -41
  137. package/dist/indexer/search/ranking.js +28 -154
  138. package/dist/indexer/search/search-attribution.js +12 -32
  139. package/dist/indexer/search/search-fields.js +11 -15
  140. package/dist/indexer/search/search-hit-enrichers.js +54 -85
  141. package/dist/indexer/search/search-source.js +1 -4
  142. package/dist/indexer/usage/usage-events.js +2 -7
  143. package/dist/integrations/agent/engine-fallback.js +23 -40
  144. package/dist/integrations/agent/engine-resolution.js +93 -183
  145. package/dist/integrations/agent/execution.js +507 -0
  146. package/dist/integrations/agent/model-map.js +28 -156
  147. package/dist/integrations/agent/request-lowering.js +66 -141
  148. package/dist/integrations/agent/runner-dispatch.js +143 -321
  149. package/dist/integrations/agent/runner.js +54 -14
  150. package/dist/integrations/lockfile.js +53 -101
  151. package/dist/llm/embedders/deterministic.js +2 -3
  152. package/dist/llm/embedders/profile.js +71 -0
  153. package/dist/llm/embedders/remote.js +10 -15
  154. package/dist/llm/graph-extract.js +3 -12
  155. package/dist/llm/index-passes.js +3 -5
  156. package/dist/llm/memory-infer.js +1 -2
  157. package/dist/llm/metadata-enhance.js +1 -2
  158. package/dist/llm/structured-call.js +5 -24
  159. package/dist/output/generic-render.js +23 -11
  160. package/dist/output/html-render.js +13 -10
  161. package/dist/output/render-registry.js +3 -32
  162. package/dist/output/shapes/helpers.js +2 -34
  163. package/dist/output/shapes/passthrough.js +1 -9
  164. package/dist/{indexer/search/ranking-types.js → output/text/bundle-rename.js} +4 -1
  165. package/dist/output/text/command-format.js +60 -23
  166. package/dist/output/text/helpers.js +1 -1
  167. package/dist/output/text/migrate.js +5 -14
  168. package/dist/output/text/proposal-format.js +1 -2
  169. package/dist/output/text/workflow-format.js +0 -32
  170. package/dist/output/text.js +2 -0
  171. package/dist/registry/factory.js +4 -19
  172. package/dist/registry/network.js +66 -220
  173. package/dist/registry/providers/index.js +0 -2
  174. package/dist/registry/providers/skills-sh.js +3 -14
  175. package/dist/registry/providers/static-index.js +24 -26
  176. package/dist/registry/resolve.js +55 -131
  177. package/dist/scripts/akm-migrate-node.js +43940 -93320
  178. package/dist/scripts/akm-migrate.js +43700 -93078
  179. package/dist/setup/registry-stash-loader.js +4 -13
  180. package/dist/setup/semantic-assets.js +3 -44
  181. package/dist/setup/setup.js +1 -1
  182. package/dist/setup/steps/tasks.js +25 -15
  183. package/dist/sources/provider-factory.js +17 -18
  184. package/dist/sources/providers/filesystem.js +2 -3
  185. package/dist/sources/providers/git-install.js +7 -1
  186. package/dist/sources/providers/git-provider.js +0 -3
  187. package/dist/sources/providers/git-stash.js +0 -17
  188. package/dist/sources/providers/npm.js +2 -4
  189. package/dist/sources/providers/provider-utils.js +5 -10
  190. package/dist/sources/providers/website.js +0 -2
  191. package/dist/sources/snapshot-fetchers/website-ingest.js +1 -1
  192. package/dist/sources/website-url.js +2 -2
  193. package/dist/storage/database.js +9 -35
  194. package/dist/storage/repositories/improve-ledger-repository.js +168 -0
  195. package/dist/storage/repositories/index-connection.js +34 -70
  196. package/dist/storage/repositories/index-entries-repository.js +69 -111
  197. package/dist/storage/repositories/index-entry-mapper.js +1 -2
  198. package/dist/storage/repositories/index-entry-schema.js +83 -269
  199. package/dist/storage/repositories/index-fts-repository.js +86 -256
  200. package/dist/storage/repositories/index-llm-cache-repository.js +17 -0
  201. package/dist/storage/repositories/index-meta-repository.js +6 -4
  202. package/dist/storage/repositories/index-schema.js +192 -220
  203. package/dist/storage/repositories/index-utility-repository.js +8 -29
  204. package/dist/storage/repositories/index-vec-repository.js +133 -414
  205. package/dist/storage/repositories/outcome-repository.js +2 -1
  206. package/dist/storage/repositories/proposals-repository.js +35 -0
  207. package/dist/storage/repositories/registry-index-cache-repository.js +100 -0
  208. package/dist/storage/repositories/task-history-repository.js +26 -4
  209. package/dist/storage/repositories/workflow-runs-repository.js +53 -244
  210. package/dist/storage/sqlite-migrations.js +136 -0
  211. package/dist/storage/sqlite-pragmas.js +11 -9
  212. package/dist/storage/sqlite-transaction.js +170 -0
  213. package/dist/storage/state-db-integrity.js +34 -27
  214. package/dist/tasks/activation-config.js +134 -62
  215. package/dist/tasks/backends/cron.js +129 -277
  216. package/dist/tasks/backends/exec-utils.js +2 -5
  217. package/dist/tasks/backends/launchd.js +125 -745
  218. package/dist/tasks/backends/schtasks.js +101 -620
  219. package/dist/tasks/prepare/prepare-support.js +5 -15
  220. package/dist/tasks/prepare/prepare.js +0 -2
  221. package/dist/tasks/resolve-akm-bin.js +20 -79
  222. package/dist/tasks/run/attempt-lifecycle.js +0 -1
  223. package/dist/tasks/scheduler-binding.js +18 -238
  224. package/dist/tasks/scheduler-invocation.js +52 -52
  225. package/dist/tasks/scheduler-lock.js +53 -0
  226. package/dist/tasks/scheduler-sync.js +361 -751
  227. package/dist/tasks/source/parse-task-source.js +160 -10
  228. package/dist/tasks/source/task-source-v3-frozen.js +3 -4
  229. package/dist/tasks/source/task-to-v4.js +2 -2
  230. package/dist/workflows/authoring/authoring.js +3 -12
  231. package/dist/workflows/compile.js +211 -0
  232. package/dist/workflows/concurrency-policy.js +13 -74
  233. package/dist/workflows/exec/child-invocation.js +3 -17
  234. package/dist/workflows/exec/child-workflow.js +32 -141
  235. package/dist/workflows/exec/dispatch-redaction.js +13 -53
  236. package/dist/workflows/exec/environment.js +98 -0
  237. package/dist/workflows/exec/exec-unit.js +33 -140
  238. package/dist/workflows/exec/frozen-judge.js +7 -59
  239. package/dist/workflows/exec/native-executor.js +82 -341
  240. package/dist/workflows/exec/param-secrets.js +29 -47
  241. package/dist/workflows/exec/run-workflow.js +154 -387
  242. package/dist/workflows/exec/scheduler.js +9 -36
  243. package/dist/workflows/exec/step-work.js +127 -430
  244. package/dist/workflows/exec/unit-dispatch.js +11 -63
  245. package/dist/workflows/exec/unit-writer.js +8 -52
  246. package/dist/workflows/exec/worktree.js +39 -273
  247. package/dist/workflows/freeze/child-output-references.js +4 -15
  248. package/dist/workflows/freeze/environment.js +99 -92
  249. package/dist/workflows/freeze/freeze.js +172 -0
  250. package/dist/workflows/freeze/step-values.js +19 -21
  251. package/dist/workflows/freeze/targets/child-workflow.js +23 -92
  252. package/dist/workflows/freeze/targets/command.js +10 -33
  253. package/dist/workflows/freeze/targets/script.js +5 -12
  254. package/dist/workflows/freeze/targets/shell.js +3 -6
  255. package/dist/workflows/freeze/targets/task.js +25 -80
  256. package/dist/workflows/freeze/task-bindings.js +20 -67
  257. package/dist/workflows/{source-ir/github-yaml.js → github-yaml.js} +88 -206
  258. package/dist/workflows/ir/params.js +6 -51
  259. package/dist/workflows/ir/plan-hash.js +2 -34
  260. package/dist/workflows/parser.js +140 -43
  261. package/dist/{commands/improve/consolidate/types.js → workflows/plan.js} +2 -1
  262. package/dist/workflows/renderer.js +36 -69
  263. package/dist/workflows/resource-limits.js +12 -120
  264. package/dist/workflows/runtime/agent-identity.js +8 -40
  265. package/dist/workflows/runtime/run-outputs.js +3 -6
  266. package/dist/workflows/runtime/run-plan.js +316 -0
  267. package/dist/workflows/runtime/runs.js +48 -200
  268. package/dist/workflows/runtime/workflow-asset-loader.js +24 -57
  269. package/dist/workflows/{source-ir/semantics.js → source-semantics.js} +16 -20
  270. package/dist/workflows/validate-summary.js +2 -7
  271. package/docs/integration/bundling-akm.md +49 -42
  272. package/docs/migration/README.md +1 -0
  273. package/docs/migration/release-notes/0.9.17.md +41 -0
  274. package/docs/migration/v0.9.1-to-v0.9.2.md +19 -7
  275. package/docs/reference/cli.md +182 -125
  276. package/docs/reference/configuration.md +49 -56
  277. package/docs/reference/data-and-telemetry.md +19 -20
  278. package/docs/reference/tasks.md +86 -38
  279. package/docs/reference/workflow-schema.md +14 -18
  280. package/docs/reference/workflows.md +6 -9
  281. package/package.json +1 -1
  282. package/schemas/akm-config.json +87 -406
  283. package/dist/commands/health/advisories.js +0 -150
  284. package/dist/commands/health/metrics.js +0 -329
  285. package/dist/commands/health/surfaces.js +0 -102
  286. package/dist/commands/improve/anti-collapse.js +0 -83
  287. package/dist/commands/improve/collapse-detector.js +0 -432
  288. package/dist/commands/improve/consolidate/eligibility.js +0 -48
  289. package/dist/commands/improve/consolidate/merge.js +0 -146
  290. package/dist/commands/improve/distill/promote-memory.js +0 -329
  291. package/dist/commands/improve/distill/quality-gate.js +0 -500
  292. package/dist/commands/improve/memory/memory-contradiction-detect.js +0 -291
  293. package/dist/commands/improve/proposal-envelope.js +0 -31
  294. package/dist/commands/improve/run-context.js +0 -123
  295. package/dist/commands/improve/shared.js +0 -21
  296. package/dist/commands/improve/source-identity.js +0 -28
  297. package/dist/commands/improve/triage.js +0 -96
  298. package/dist/commands/proposal/drain-policies.js +0 -151
  299. package/dist/commands/sources/update-transaction.js +0 -220
  300. package/dist/core/action-contributors.js +0 -28
  301. package/dist/core/config/config-version-shim.js +0 -101
  302. package/dist/core/config/retired-experimental-keys-shim.js +0 -62
  303. package/dist/core/fs-txn.js +0 -405
  304. package/dist/core/lexical-score.js +0 -25
  305. package/dist/core/maintenance-barrier.js +0 -167
  306. package/dist/execution/executable-identity.js +0 -105
  307. package/dist/execution/guarded-source.js +0 -441
  308. package/dist/indexer/graph/graph-boost.js +0 -427
  309. package/dist/indexer/graph/graph-dedup.js +0 -95
  310. package/dist/indexer/search/name-match.js +0 -35
  311. package/dist/indexer/search/ranking-contributors.js +0 -515
  312. package/dist/indexer/walk/project-context.js +0 -192
  313. package/dist/integrations/agent/execution-cascade.js +0 -566
  314. package/dist/integrations/agent/execution-definitions.js +0 -202
  315. package/dist/integrations/agent/execution-lowering.js +0 -841
  316. package/dist/integrations/agent/execution-preparation.js +0 -98
  317. package/dist/integrations/agent/inline-execution.js +0 -74
  318. package/dist/registry/create-provider-registry.js +0 -29
  319. package/dist/registry/pinned-request-helper.js +0 -247
  320. package/dist/registry/pinned-transport.js +0 -717
  321. package/dist/sources/providers/index.js +0 -14
  322. package/dist/storage/engines/sqlite-migrations.js +0 -271
  323. package/dist/storage/repositories/canaries-repository.js +0 -107
  324. package/dist/storage/repositories/embedding-salvage-repository.js +0 -184
  325. package/dist/storage/repositories/registry-cache.js +0 -113
  326. package/dist/tasks/scheduler-sync-preview.js +0 -52
  327. package/dist/workflows/freeze/resolve-steps.js +0 -86
  328. package/dist/workflows/freeze/source-freeze.js +0 -64
  329. package/dist/workflows/ir/compile.js +0 -321
  330. package/dist/workflows/ir/environment-v4.js +0 -330
  331. package/dist/workflows/ir/freeze-v4.js +0 -153
  332. package/dist/workflows/ir/schema-v4.js +0 -745
  333. package/dist/workflows/ir/schema.js +0 -354
  334. package/dist/workflows/program/schema.js +0 -78
  335. package/dist/workflows/runtime/checkin.js +0 -57
  336. package/dist/workflows/runtime/plan-classifier.js +0 -196
  337. package/dist/workflows/runtime/unit-checkin.js +0 -45
  338. package/dist/workflows/runtime/unit-phases.js +0 -20
  339. package/dist/workflows/schema.js +0 -4
  340. package/dist/workflows/source-ir/compile.js +0 -200
  341. package/dist/workflows/source-ir/program.js +0 -50
  342. package/dist/workflows/source-ir/result.js +0 -26
  343. package/dist/workflows/source-ir/schema.js +0 -786
  344. package/dist/workflows/source-ir/triggers.js +0 -79
  345. package/dist/workflows/source-ir/uses.js +0 -40
  346. package/dist/workflows/validator.js +0 -60
@@ -1,23 +1,9 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- /**
5
- * Shared memory-content hashing primitives, extracted from the deleted
6
- * `dedup.ts` (#617 dedup pre-pass, removed WI-7.3) so `consolidate.ts` /
7
- * `consolidate/chunking.ts` / `distill.ts` keep a stable, dependency-free home
8
- * for the case-preserving stripped-body hash they use for the body-embedding
9
- * cache and (formerly) the fidelity-check body comparison.
10
- *
11
- * @module content-hash
12
- */
13
4
  import { createHash } from "node:crypto";
14
- import { parseFrontmatter } from "../../core/asset/frontmatter.js";
15
- /**
16
- * Strip frontmatter from raw memory content, returning the body text trimmed.
17
- * Case and whitespace are preserved. Falls back to `raw.trim()` on
18
- * unparseable frontmatter (consistent with the pre-existing load-time hot
19
- * guard).
20
- */
5
+ import { computeNormalizedContentHash, parseFrontmatter } from "../../core/asset/frontmatter.js";
6
+ /** The markdown body with its frontmatter removed, trimmed (the raw text when it does not parse). */
21
7
  export function stripFrontmatterBody(raw) {
22
8
  try {
23
9
  return parseFrontmatter(raw).content.trim();
@@ -27,13 +13,19 @@ export function stripFrontmatterBody(raw) {
27
13
  }
28
14
  }
29
15
  /**
30
- * Hash used for change-detection and the body-embedding cache: case-/
31
- * whitespace-preserving stripped body. Two memories with the same wording
32
- * but different casing produce DIFFERENT hashes here, which is intentional —
33
- * we embed the exact text and cache by its precise content.
34
- *
35
- * This is the `content_hash` stored in `body_embeddings`.
16
+ * The one "is this the same content?" hash for improve and the proposal queue
17
+ * (sha256, hex):
18
+ * - `raw`: the exact bytes — proposal before/after and judged-content hashes,
19
+ * session transcripts, cache keys for plain text.
20
+ * - `body`: the body without frontmatter, case and wording preserved — memory
21
+ * and knowledge dedup and the body-embedding cache.
22
+ * - `normalized`: the whole asset minus akm's bookkeeping frontmatter
23
+ * (`BOOKKEEPING_FRONTMATTER_KEYS`), keys sorted — proposal freshness, so a
24
+ * salience or inference rewrite of the target never stales a proposal.
36
25
  */
37
- export function cacheHash(raw) {
38
- return createHash("sha256").update(stripFrontmatterBody(raw), "utf8").digest("hex");
26
+ export function contentHash(content, mode = "raw") {
27
+ if (mode === "raw")
28
+ return createHash("sha256").update(content).digest("hex");
29
+ const text = typeof content === "string" ? content : Buffer.from(content).toString("utf8");
30
+ return mode === "body" ? contentHash(stripFrontmatterBody(text)) : computeNormalizedContentHash(text);
39
31
  }
@@ -1,33 +1,16 @@
1
1
  // This Source Code Form is subject to the terms of the Mozilla Public
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
- /**
5
- * Pure content-repair + quality-validation stages for `akm distill`.
6
- *
7
- * Extracted verbatim from the inline body of `akmDistill` so each normalization
8
- * pass is an independently testable unit. Every function is a pure transform of
9
- * `(content, inputRef) → content | findings` with no I/O — logic is
10
- * byte-identical to the pre-extraction inline code. The lesson-path guard
11
- * (`effectiveProposalKind !== "knowledge"`) stays in the caller; these helpers
12
- * assume the lesson path.
13
- */
4
+ /** Deterministic lesson repairs and checks for `akm distill` (pure; the caller keeps the lesson-path guard). */
14
5
  import { assembleAssetFromString, serializeFrontmatterQuoted } from "../../../core/asset/asset-serialize.js";
15
6
  import { parseFrontmatter } from "../../../core/asset/frontmatter.js";
16
7
  import { repairTruncatedDescription } from "../../../core/text-truncation.js";
17
- import { detectDoubleFrontmatter, isValidDescription, isValidWhenToUse, } from "../../proposal/validators/proposal-quality-validators.js";
8
+ import { isValidDescription, isValidWhenToUse, lessonQualityIssues, } from "../../proposal/validators/proposal-quality-validators.js";
18
9
  /**
19
- * Auto-repair missing frontmatter fields before hard-failing. Small models
20
- * frequently produce a good lesson body but omit the YAML header entirely.
21
- * Rather than discarding valid content, we extract description/when_to_use
22
- * from the body and prepend the required frontmatter block.
23
- *
24
- * IMPORTANT: We do NOT synthesise placeholder strings here. If the body
25
- * does not contain text that passes the post-LLM validators
26
- * (`isValidDescription` / `isValidWhenToUse`), we leave the field missing
27
- * and let the lesson lint reject the proposal as `validation_failed`.
28
- * Emitting placeholders like `"Lesson distilled from <ref>"` or
29
- * `"When working with <slug>"` is what produced the systematic broken
30
- * proposals observed across 323 archived rejections.
10
+ * Fill a missing description / when_to_use from body lines that pass their
11
+ * validators — small models often write a good body with no header. Never a
12
+ * placeholder: those produced hundreds of broken proposals; a field nothing
13
+ * qualifies for stays missing for lint to reject.
31
14
  */
32
15
  export function autoRepairLessonFrontmatter(content, inputRef) {
33
16
  const parsed = parseFrontmatter(content);
@@ -37,7 +20,6 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
37
20
  if (!missingDesc && !missingWtu)
38
21
  return content;
39
22
  const body = parsed.content.trim();
40
- // Strip markdown formatting tokens from a line so extracted text is clean.
41
23
  const stripMd = (l) => l
42
24
  .replace(/\*\*([^*]+)\*\*/g, "$1")
43
25
  .replace(/\*([^*]+)\*/g, "$1")
@@ -45,14 +27,9 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
45
27
  .replace(/^[#*\->_]+\s*/, "")
46
28
  .replace(/:\s*$/, "")
47
29
  .trim();
48
- // Skip lines that look like YAML field assignments (key: value) or frontmatter delimiters.
49
- // These appear when the LLM leaks frontmatter content into the body, causing
50
- // auto-repair to produce description: "description: Key Takeaways".
30
+ // Leaked frontmatter lines in the body would yield `description: "description: …"`.
51
31
  const isYamlLike = (l) => /^---/.test(l) || /^[a-z_]+:\s/i.test(l);
52
32
  const bodyLines = body.split("\n").map(stripMd);
53
- // Extract description: first body line that BOTH looks like prose AND
54
- // passes isValidDescription. If nothing qualifies, leave the field
55
- // missing — the lint pass will reject the proposal cleanly.
56
33
  let descLine;
57
34
  for (const l of bodyLines) {
58
35
  if (isYamlLike(l))
@@ -64,8 +41,6 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
64
41
  break;
65
42
  }
66
43
  }
67
- // Extract when_to_use: a line starting with "When" / "Use when" / "Apply when"
68
- // that ALSO passes isValidWhenToUse (rejects circular fallbacks).
69
44
  let wtuLine;
70
45
  for (const l of bodyLines) {
71
46
  if (!/^(when |use when|apply when)/i.test(l))
@@ -83,23 +58,15 @@ export function autoRepairLessonFrontmatter(content, inputRef) {
83
58
  ...(missingWtu && wtuLine ? { when_to_use: wtuLine } : {}),
84
59
  };
85
60
  const fmLines = serializeFrontmatterQuoted(repairedFm);
86
- // Only rewrite content if we actually have at least one field to write.
87
- // Otherwise leave the original content for the lint pass to reject.
88
61
  if (Object.keys(repairedFm).length > 0) {
89
62
  return assembleAssetFromString(fmLines, body);
90
63
  }
91
64
  return content;
92
65
  }
93
66
  /**
94
- * Description ↔ when_to_use auto-swap normalization (recover ~93% of
95
- * qwen-9b's `^when\b/i` rejections at zero LLM cost). When the LLM emits
96
- * a conditional-framed description ("When X happens, do Y") and the
97
- * when_to_use field looks like a declarative description (or is empty),
98
- * the two fields are mis-fielded — exactly what `isValidDescription`'s
99
- * error message says ("that pattern belongs in when_to_use"). We swap
100
- * them and revalidate; the swap is committed only if BOTH fields pass
101
- * their respective validators afterwards. If revalidation still fails,
102
- * we fall through returning the original content (swapped: 0).
67
+ * Swap a conditional description ("When X, do Y") with a declarative
68
+ * when_to_use — mis-fielded, as the description validator says — when both
69
+ * then pass; this recovers most `^when` rejections at no LLM cost.
103
70
  */
104
71
  export function autoSwapDescriptionWhenToUse(content, inputRef) {
105
72
  const parsedSwap = parseFrontmatter(content);
@@ -109,9 +76,6 @@ export function autoSwapDescriptionWhenToUse(content, inputRef) {
109
76
  const descStartsConditional = /^(when|if)\b/i.test(descRaw);
110
77
  const wtuStartsConditional = /^(when|if)\b/i.test(wtuRaw);
111
78
  if (descStartsConditional && !wtuStartsConditional && wtuRaw.length > 0) {
112
- // Try the swap and revalidate. The when_to_use validator requires the
113
- // value not match `/^when working with\b/i` (the circular fallback) —
114
- // a real description rarely does, so this usually passes.
115
79
  const swappedDescCheck = isValidDescription(wtuRaw, inputRef);
116
80
  const swappedWtuCheck = isValidWhenToUse(descRaw, inputRef);
117
81
  if (swappedDescCheck.ok && swappedWtuCheck.ok) {
@@ -126,13 +90,7 @@ export function autoSwapDescriptionWhenToUse(content, inputRef) {
126
90
  }
127
91
  return { content, swapped: 0 };
128
92
  }
129
- /**
130
- * Post-generation truncation repair (#556): if the LLM sliced the
131
- * description mid-sentence, deterministically complete it from its own text
132
- * / the lesson body BEFORE the lint + quality validators run. No-op
133
- * (byte-identical) for already-complete descriptions, so this never alters
134
- * a valid proposal.
135
- */
93
+ /** Complete a description cut mid-sentence from its own text or the body (#556); a complete one is untouched. */
136
94
  export function repairLessonDescriptionTruncation(content) {
137
95
  const parsedRepair = parseFrontmatter(content);
138
96
  const fmRepair = (parsedRepair.data ?? {});
@@ -145,52 +103,12 @@ export function repairLessonDescriptionTruncation(content) {
145
103
  const repairedFmLines = serializeFrontmatterQuoted({ ...fmRepair, description: repaired });
146
104
  return assembleAssetFromString(repairedFmLines, parsedRepair.content);
147
105
  }
148
- /**
149
- * Additional quality validators that run only on lessons whose lesson-lint
150
- * pass was clean. lesson-lint checks "field is present and non-empty"; these
151
- * reject the systematic failure modes observed across 323 archived rejected
152
- * proposals:
153
- * - description is a body fragment, section heading, or placeholder
154
- * - when_to_use is the circular "When working with <ref>" fallback
155
- * - description == when_to_use (LLM duplicated a single sentence)
156
- * - body contains a second pseudo-frontmatter block
157
- */
106
+ /** The shared lesson quality checks, for a lesson whose lint pass was clean. */
158
107
  export function collectLessonQualityFindings(content, inputRef) {
159
- const findings = [];
160
- const parsedQC = parseFrontmatter(content);
161
- const fmQC = (parsedQC.data ?? {});
162
- const descCheck = isValidDescription(fmQC.description, inputRef);
163
- if (!descCheck.ok) {
164
- findings.push({
165
- kind: "invalid-description",
166
- field: "description",
167
- message: `Distilled lesson for ${inputRef} has an invalid description: ${descCheck.reason}.`,
168
- });
169
- }
170
- const wtuCheck = isValidWhenToUse(fmQC.when_to_use, inputRef);
171
- if (!wtuCheck.ok) {
172
- findings.push({
173
- kind: "invalid-when_to_use",
174
- field: "when_to_use",
175
- message: `Distilled lesson for ${inputRef} has an invalid when_to_use: ${wtuCheck.reason}.`,
176
- });
177
- }
178
- // description and when_to_use must say different things.
179
- if (descCheck.ok &&
180
- wtuCheck.ok &&
181
- typeof fmQC.description === "string" &&
182
- typeof fmQC.when_to_use === "string" &&
183
- fmQC.description.trim().toLowerCase() === fmQC.when_to_use.trim().toLowerCase()) {
184
- findings.push({
185
- kind: "description-equals-when_to_use",
186
- field: "description",
187
- message: `Distilled lesson for ${inputRef} has identical description and when_to_use.`,
188
- });
189
- }
190
- // Double-frontmatter / pseudo-frontmatter pollution in the body.
191
- const dfm = detectDoubleFrontmatter(content);
192
- if (dfm) {
193
- findings.push({ kind: dfm.kind, field: "body", message: `Distilled lesson for ${inputRef}: ${dfm.message}` });
194
- }
195
- return findings;
108
+ const fm = (parseFrontmatter(content).data ?? {});
109
+ return lessonQualityIssues(fm, content, inputRef).map((issue) => ({
110
+ kind: issue.kind,
111
+ field: issue.field,
112
+ message: `Distilled lesson for ${inputRef}${issue.text}`,
113
+ }));
196
114
  }
@@ -2,34 +2,12 @@
2
2
  // License, v. 2.0. If a copy of the MPL was not distributed with this
3
3
  // file, You can obtain one at https://mozilla.org/MPL/2.0/.
4
4
  /**
5
- * WS-3b distill-stage guards.
6
- *
7
- * **CLS interleaving (step 9)**
8
- * distill/memoryInference prompts include embedding-retrieved adjacent
9
- * lessons/knowledge so the pipeline doesn't overwrite prior generalizations.
10
- *
11
- * **Distill→source fidelity (step 10)**
12
- * After a distill proposal, check it against cited source memories; a
13
- * contradiction flag routes to human review.
14
- *
15
- * @module distill-guards
5
+ * Distill guards: related lessons/knowledge shown to the model so it does not
6
+ * overwrite prior generalizations (CLS context), and a cheap check that a
7
+ * proposal does not contradict the memories it came from.
16
8
  */
17
- // ── CLS adjacent lesson context (step 9) ─────────────────────────────────────
18
- /** Default number of adjacent lessons/knowledge for CLS interleaving. */
19
9
  export const DEFAULT_CLS_ADJACENT_COUNT = 3;
20
- /**
21
- * Build a CLS (Complementary Learning System) context snippet for injection
22
- * into distill/memoryInference prompts.
23
- *
24
- * Given a list of embedding-retrieved adjacent lessons/knowledge, formats them
25
- * as a markdown section to append to the prompt so the LLM avoids overwriting
26
- * prior generalizations.
27
- *
28
- * Returns an empty string when CLS is disabled or no adjacent items are found.
29
- *
30
- * @param adjacentItems - Top-N adjacent lessons/knowledge retrieved by embedding.
31
- * @param config - CLS config.
32
- */
10
+ /** The CLS prompt section (each entry capped at 400 chars); empty when disabled or nothing is related. */
33
11
  export function buildClsContext(adjacentItems, config) {
34
12
  if (!config.enabled || adjacentItems.length === 0)
35
13
  return "";
@@ -41,54 +19,23 @@ export function buildClsContext(adjacentItems, config) {
41
19
  "disagree with one, flag it as contradicted (do not ignore it).",
42
20
  "",
43
21
  ];
44
- for (const item of adjacentItems) {
45
- lines.push(`### ${item.ref}`);
46
- // Truncate to 400 chars to keep the prompt size reasonable.
47
- lines.push(item.content.trim().slice(0, 400));
48
- lines.push("");
49
- }
22
+ for (const item of adjacentItems)
23
+ lines.push(`### ${item.ref}`, item.content.trim().slice(0, 400), "");
50
24
  return lines.join("\n");
51
25
  }
52
26
  /**
53
- * Check a distill proposal against its cited source memories for contradictions.
54
- *
55
- * Uses a simple heuristic: looks for explicit negation of key claims in the
56
- * proposal body that appear in the source bodies. A full LLM-based
57
- * contradiction check is expensive (one LLM call per proposal); this cheap
58
- * heuristic catches the most obvious cases and flags them for human review.
59
- *
60
- * When `fidelityCheck.enabled` is false, returns `{ contradictionDetected: false }`
61
- * immediately (no work done).
62
- *
63
- * @param proposalBody - The stripped body of the distill proposal.
64
- * @param sourceBodies - The stripped bodies of the cited source memories.
65
- * @param config - Fidelity check config.
27
+ * Flag a proposal whose "always/must X" (or "never/must not X") claim meets
28
+ * the opposite claim about X in a source. Deliberately conservative: a flag
29
+ * only costs a human review, while a model call per proposal is expensive.
66
30
  */
67
31
  export function checkDistillFidelity(proposalBody, sourceBodies, config) {
68
- if (!config.enabled || sourceBodies.length === 0) {
69
- return { contradictionDetected: false };
70
- }
71
- // Heuristic: detect explicit negation of "never" / "always" / "must" claims.
72
- // A proposal that says "always X" while the source says "never X" (or vice
73
- // versa) is a clear contradiction worth flagging.
74
- //
75
- // This is intentionally conservative: it only flags when both the proposal
76
- // AND the source contain the opposing polarity of the same key term. False
77
- // negatives (missed contradictions) are preferred over false positives
78
- // (blocking valid proposals) since the consequence of a false positive is
79
- // a human review request, while the cost of a false negative is a slightly
80
- // degraded stash.
81
- const proposalLow = proposalBody.toLowerCase();
82
- // Extract "always/never/must/must not" claims from the proposal.
83
- const strongClaims = extractStrongClaims(proposalLow);
84
- if (strongClaims.length === 0)
32
+ if (!config.enabled || sourceBodies.length === 0)
85
33
  return { contradictionDetected: false };
34
+ const strongClaims = extractStrongClaims(proposalBody.toLowerCase());
86
35
  for (const sourceBody of sourceBodies) {
87
36
  const sourceLow = sourceBody.toLowerCase();
88
37
  for (const { polarity, term } of strongClaims) {
89
- const oppositePolarity = polarity === "positive" ? "negative" : "positive";
90
- const sourceHasOpposite = hasStrongClaim(sourceLow, term, oppositePolarity);
91
- if (sourceHasOpposite) {
38
+ if (hasStrongClaim(sourceLow, term, polarity === "positive" ? "negative" : "positive")) {
92
39
  return {
93
40
  contradictionDetected: true,
94
41
  reason: `Proposal makes a ${polarity} strong claim about "${term}" that conflicts with an opposing claim in a cited source. Route to human review.`,
@@ -96,32 +43,24 @@ export function checkDistillFidelity(proposalBody, sourceBodies, config) {
96
43
  }
97
44
  }
98
45
  }
99
- // Also flag proposals whose xrefs are empty (broken provenance).
100
- // This is a degradation signal, not a contradiction, but worth surfacing.
101
46
  return { contradictionDetected: false };
102
47
  }
48
+ const CLAIM_PATTERNS = [
49
+ { polarity: "positive", re: /\b(?:always|must)\s+(\w+)/g },
50
+ { polarity: "negative", re: /\b(?:never|must\s+not|should\s+not)\s+(\w+)/g },
51
+ ];
103
52
  function extractStrongClaims(text) {
104
53
  const claims = [];
105
- // Match "always <term>", "never <term>", "must <term>", "must not <term>".
106
- const patterns = [
107
- { polarity: "positive", re: /\b(?:always|must)\s+(\w+)/g },
108
- { polarity: "negative", re: /\b(?:never|must\s+not|should\s+not)\s+(\w+)/g },
109
- ];
110
- for (const { polarity, re } of patterns) {
111
- re.lastIndex = 0;
112
- let m = re.exec(text);
113
- while (m !== null) {
54
+ for (const { polarity, re } of CLAIM_PATTERNS) {
55
+ for (const m of text.matchAll(re)) {
114
56
  const term = m[1];
115
57
  if (term && term.length > 2)
116
58
  claims.push({ polarity, term });
117
- m = re.exec(text);
118
59
  }
119
60
  }
120
61
  return claims;
121
62
  }
122
63
  function hasStrongClaim(text, term, polarity) {
123
- if (polarity === "positive") {
124
- return /\b(?:always|must)\s/.test(text) && text.includes(term);
125
- }
126
- return /\b(?:never|must\s+not|should\s+not)\s/.test(text) && text.includes(term);
64
+ const marker = polarity === "positive" ? /\b(?:always|must)\s/ : /\b(?:never|must\s+not|should\s+not)\s/;
65
+ return marker.test(text) && text.includes(term);
127
66
  }
@@ -216,249 +216,29 @@ function assessWithWeightedModel(input, model, threshold) {
216
216
  modelName: model.name,
217
217
  };
218
218
  }
219
- function precision(tp, fp) {
220
- return tp + fp === 0 ? 1 : tp / (tp + fp);
221
- }
222
- function recall(tp, fn) {
223
- return tp + fn === 0 ? 1 : tp / (tp + fn);
224
- }
225
- function f1Score(p, r) {
226
- return p + r === 0 ? 0 : (2 * p * r) / (p + r);
227
- }
228
- function casePromoteValue(testCase) {
229
- return testCase.promoteValue ?? 3;
230
- }
231
- function caseFalsePromoteCost(testCase) {
232
- return testCase.falsePromoteCost ?? 4;
233
- }
234
- function caseMissedPromoteCost(testCase) {
235
- return testCase.missedPromoteCost ?? 2;
236
- }
237
- export function evaluateMemoryPromotionBenchmark(cases, policy = DEFAULT_PROMOTION_POLICY) {
238
- const results = cases.map((fixture) => {
239
- const assessment = policy.assess(fixture.input);
240
- const passed = assessment.promote === fixture.expectPromote;
241
- return {
242
- fixture,
243
- name: fixture.name,
244
- expectPromote: fixture.expectPromote,
245
- assessment,
246
- passed,
247
- };
248
- });
249
- const truePositives = results.filter((result) => result.assessment.promote && result.expectPromote).length;
250
- const trueNegatives = results.filter((result) => !result.assessment.promote && !result.expectPromote).length;
251
- const falsePositives = results.filter((result) => result.assessment.promote && !result.expectPromote).length;
252
- const falseNegatives = results.filter((result) => !result.assessment.promote && result.expectPromote).length;
253
- const correct = truePositives + trueNegatives;
254
- const p = precision(truePositives, falsePositives);
255
- const r = recall(truePositives, falseNegatives);
256
- let netOutcomeScore = 0;
257
- let capturedPromoteValue = 0;
258
- let preventedFalsePromotionCost = 0;
259
- for (const result of results) {
260
- if (result.expectPromote && result.assessment.promote) {
261
- const value = casePromoteValue(result.fixture);
262
- netOutcomeScore += value;
263
- capturedPromoteValue += value;
264
- }
265
- else if (result.expectPromote && !result.assessment.promote) {
266
- netOutcomeScore -= caseMissedPromoteCost(result.fixture);
267
- }
268
- else if (!result.expectPromote && result.assessment.promote) {
269
- netOutcomeScore -= caseFalsePromoteCost(result.fixture);
270
- }
271
- else {
272
- preventedFalsePromotionCost += caseFalsePromoteCost(result.fixture);
273
- }
274
- }
275
- return {
276
- total: results.length,
277
- correct,
278
- falsePositives,
279
- falseNegatives,
280
- accuracy: results.length === 0 ? 1 : correct / results.length,
281
- precision: p,
282
- recall: r,
283
- f1: f1Score(p, r),
284
- truePositives,
285
- trueNegatives,
286
- netOutcomeScore,
287
- capturedPromoteValue,
288
- preventedFalsePromotionCost,
289
- results: results.map(({ name, expectPromote, assessment, passed }) => ({
290
- name,
291
- expectPromote,
292
- assessment,
293
- passed,
294
- })),
295
- };
296
- }
297
- function thresholdCandidates() {
298
- const values = [];
299
- for (let value = 2.4; value <= 4.2; value += 0.2) {
300
- values.push(Number(value.toFixed(1)));
301
- }
302
- return values;
303
- }
304
- const POSITIVE_FEEDBACK_BASELINE = {
305
- name: "baseline-positive-feedback",
306
- threshold: 2,
307
- assess(input) {
308
- const knowledgeRef = deriveKnowledgeRef(input.inputRef);
309
- const featureState = collectPromotionFeatures(input);
310
- if (featureState.blockedBy.length > 0) {
311
- return {
312
- applicable: !featureState.blockedBy.includes("not-memory"),
313
- promote: false,
314
- score: 0,
315
- threshold: 2,
316
- knowledgeRef,
317
- blockedBy: featureState.blockedBy,
318
- positiveSignals: [],
319
- negativeSignals: [],
320
- modelName: "baseline-positive-feedback",
321
- };
322
- }
323
- const features = featureState.features;
324
- const promote = features.positiveFeedback >= 2;
325
- return {
326
- applicable: true,
327
- promote,
328
- score: features.positiveFeedback,
329
- threshold: 2,
330
- knowledgeRef,
331
- ...(promote ? { content: buildKnowledgeContent(input) } : {}),
332
- blockedBy: [],
333
- positiveSignals: promote ? ["baseline positive feedback rule"] : [],
334
- negativeSignals: promote ? [] : ["baseline positive feedback rule not met"],
335
- modelName: "baseline-positive-feedback",
336
- };
337
- },
338
- };
339
- const METADATA_BASELINE = {
340
- name: "baseline-metadata",
341
- threshold: 2,
342
- assess(input) {
343
- const knowledgeRef = deriveKnowledgeRef(input.inputRef);
344
- const featureState = collectPromotionFeatures(input);
345
- if (featureState.blockedBy.length > 0) {
346
- return {
347
- applicable: !featureState.blockedBy.includes("not-memory"),
348
- promote: false,
349
- score: 0,
350
- threshold: 2,
351
- knowledgeRef,
352
- blockedBy: featureState.blockedBy,
353
- positiveSignals: [],
354
- negativeSignals: [],
355
- modelName: "baseline-metadata",
356
- };
357
- }
358
- const features = featureState.features;
359
- const metadataScore = (features.hasSource ? 1 : 0) + (features.hasObservedAt ? 1 : 0);
360
- const promote = metadataScore >= 2;
361
- return {
362
- applicable: true,
363
- promote,
364
- score: metadataScore,
365
- threshold: 3,
366
- knowledgeRef,
367
- ...(promote ? { content: buildKnowledgeContent(input) } : {}),
368
- blockedBy: [],
369
- positiveSignals: promote ? ["baseline metadata rule"] : [],
370
- negativeSignals: promote ? [] : ["baseline metadata rule not met"],
371
- modelName: "baseline-metadata",
372
- };
373
- },
374
- };
375
- export function selectPromotionPolicy(corpus, candidates) {
376
- const trainingCases = corpus.filter((testCase) => (testCase.split ?? "train") === "train");
377
- const heldOutCases = corpus.filter((testCase) => (testCase.split ?? "train") === "heldout");
378
- let bestPolicy;
379
- let bestTraining;
380
- for (const model of candidates) {
381
- for (const threshold of thresholdCandidates()) {
382
- const policy = {
383
- name: model.name,
384
- threshold,
385
- assess: (input) => assessWithWeightedModel(input, model, threshold),
386
- };
387
- const training = evaluateMemoryPromotionBenchmark(trainingCases, policy);
388
- if (!bestTraining) {
389
- bestTraining = training;
390
- bestPolicy = policy;
391
- continue;
392
- }
393
- const trainingWins = training.f1 > bestTraining.f1 ||
394
- (training.f1 === bestTraining.f1 && training.netOutcomeScore > bestTraining.netOutcomeScore) ||
395
- (training.f1 === bestTraining.f1 &&
396
- training.netOutcomeScore === bestTraining.netOutcomeScore &&
397
- training.accuracy > bestTraining.accuracy);
398
- if (trainingWins) {
399
- bestTraining = training;
400
- bestPolicy = policy;
401
- }
402
- }
403
- }
404
- const selectedPolicy = bestPolicy;
405
- const selectedTraining = bestTraining;
406
- const heldOut = evaluateMemoryPromotionBenchmark(heldOutCases, selectedPolicy);
407
- const baselines = [POSITIVE_FEEDBACK_BASELINE, METADATA_BASELINE].map((policy) => {
408
- const baselineHeldOut = evaluateMemoryPromotionBenchmark(heldOutCases, policy);
409
- const noWorseThanSelected = heldOut.f1 >= baselineHeldOut.f1 && heldOut.netOutcomeScore >= baselineHeldOut.netOutcomeScore;
410
- const strictWinMetrics = [];
411
- if (heldOut.f1 > baselineHeldOut.f1)
412
- strictWinMetrics.push("f1");
413
- if (heldOut.netOutcomeScore > baselineHeldOut.netOutcomeScore)
414
- strictWinMetrics.push("netOutcomeScore");
415
- if (heldOut.accuracy > baselineHeldOut.accuracy)
416
- strictWinMetrics.push("accuracy");
417
- return {
418
- name: policy.name,
419
- heldOut: baselineHeldOut,
420
- noWorseThanSelected,
421
- strictWin: noWorseThanSelected && strictWinMetrics.length > 0,
422
- strictWinMetrics,
423
- };
424
- });
425
- const strictlyBeatsBaselines = baselines.every((baseline) => baseline.strictWin);
426
- return {
427
- corpusSize: corpus.length,
428
- trainingSize: trainingCases.length,
429
- heldOutSize: heldOutCases.length,
430
- selectedModel: { name: selectedPolicy.name, threshold: selectedPolicy.threshold },
431
- training: selectedTraining,
432
- heldOut,
433
- baselines,
434
- strictlyBeatsBaselines,
435
- };
436
- }
437
- export const DEFAULT_PROMOTION_POLICY_SELECTION = {
438
- selectedModel: {
439
- name: "balanced-evidence",
440
- positiveWeight: 0.8,
441
- repeatedPositiveWeight: 0.65,
442
- noPositivePenalty: 0.9,
443
- singlePositivePenalty: 0.7,
444
- negativeWeight: 2.0,
445
- curatedWeight: 0.55,
446
- confidenceWeight: 0.7,
447
- sourceWeight: 0.4,
448
- observedAtWeight: 0.4,
449
- descriptionWeight: 0.2,
450
- tagWeight: 0.15,
451
- substantiveBodyWeight: 0.15,
452
- tentativePenalty: 1.1,
453
- },
454
- threshold: 3.8,
455
- };
456
- const SELECTED_MODEL = DEFAULT_PROMOTION_POLICY_SELECTION.selectedModel;
457
- export const DEFAULT_PROMOTION_POLICY = {
458
- name: SELECTED_MODEL.name,
459
- threshold: DEFAULT_PROMOTION_POLICY_SELECTION.threshold,
460
- assess: (input) => assessWithWeightedModel(input, SELECTED_MODEL, DEFAULT_PROMOTION_POLICY_SELECTION.threshold),
219
+ /**
220
+ * The memory → knowledge promotion model: a weighted score over feedback
221
+ * reinforcement and memory metadata, promoted at or above the threshold. The
222
+ * weights were chosen by a grid search over a labelled corpus; they are a
223
+ * plain constant now.
224
+ */
225
+ const PROMOTION_MODEL = {
226
+ name: "balanced-evidence",
227
+ positiveWeight: 0.8,
228
+ repeatedPositiveWeight: 0.65,
229
+ noPositivePenalty: 0.9,
230
+ singlePositivePenalty: 0.7,
231
+ negativeWeight: 2.0,
232
+ curatedWeight: 0.55,
233
+ confidenceWeight: 0.7,
234
+ sourceWeight: 0.4,
235
+ observedAtWeight: 0.4,
236
+ descriptionWeight: 0.2,
237
+ tagWeight: 0.15,
238
+ substantiveBodyWeight: 0.15,
239
+ tentativePenalty: 1.1,
461
240
  };
241
+ const PROMOTION_THRESHOLD = 3.8;
462
242
  export function assessMemoryKnowledgePromotionCandidate(input) {
463
- return DEFAULT_PROMOTION_POLICY.assess(input);
243
+ return assessWithWeightedModel(input, PROMOTION_MODEL, PROMOTION_THRESHOLD);
464
244
  }