enigma-memory 0.1.17 → 0.1.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (290) hide show
  1. package/README.md +85 -27
  2. package/apps/cli/bin/enigma-desktop.mjs +140 -0
  3. package/apps/cli/bin/enigma-terminal.mjs +78 -0
  4. package/apps/cli/bin/enigma.mjs +4979 -3263
  5. package/apps/desktop/electron-main.cjs +217 -0
  6. package/apps/desktop/package.json +12 -0
  7. package/apps/desktop/src/app.js +264 -7
  8. package/apps/desktop/src/index.html +3514 -1373
  9. package/apps/desktop/src/launch-electron.mjs +51 -0
  10. package/apps/desktop/src/server.mjs +2914 -0
  11. package/apps/desktop/src/styles.css +2972 -260
  12. package/apps/desktop/src/zk-browser-prove.mjs +53 -0
  13. package/apps/desktop/src/zk-state.mjs +1789 -0
  14. package/apps/gateway/bin/enigma-gateway.mjs +102 -5
  15. package/apps/gateway/src/server.mjs +271 -8
  16. package/apps/ios/EnigmaCore/Package.swift +12 -0
  17. package/apps/ios/EnigmaCore/Sources/EnigmaCore/EnigmaAPIClient.swift +227 -0
  18. package/apps/ios/EnigmaCore/Sources/EnigmaCore/Models.swift +278 -0
  19. package/apps/ios/EnigmaCore/Sources/EnigmaCore/PKCE.swift +96 -0
  20. package/apps/ios/EnigmaCore/Sources/EnigmaCore/PrivacyMinimizer.swift +187 -0
  21. package/apps/ios/EnigmaCore/Sources/EnigmaCore/ToolModels.swift +129 -0
  22. package/apps/ios/EnigmaCore/Tests/EnigmaCoreTests/EnigmaCoreTests.swift +42 -0
  23. package/apps/ios/EnigmaIOS/Enigma/AppModel.swift +346 -0
  24. package/apps/ios/EnigmaIOS/Enigma/Assets.xcassets/AccentColor.colorset/Contents.json +12 -0
  25. package/apps/ios/EnigmaIOS/Enigma/Assets.xcassets/AppIcon.appiconset/Contents.json +11 -0
  26. package/apps/ios/EnigmaIOS/Enigma/Assets.xcassets/AppIcon.appiconset/EnigmaAppIcon.png +0 -0
  27. package/apps/ios/EnigmaIOS/Enigma/Assets.xcassets/Contents.json +3 -0
  28. package/apps/ios/EnigmaIOS/Enigma/Assets.xcassets/LaunchBackground.colorset/Contents.json +12 -0
  29. package/apps/ios/EnigmaIOS/Enigma/ChatView.swift +181 -0
  30. package/apps/ios/EnigmaIOS/Enigma/CouncilView.swift +78 -0
  31. package/apps/ios/EnigmaIOS/Enigma/CreateView.swift +152 -0
  32. package/apps/ios/EnigmaIOS/Enigma/EnigmaApp.swift +52 -0
  33. package/apps/ios/EnigmaIOS/Enigma/Info.plist +52 -0
  34. package/apps/ios/EnigmaIOS/Enigma/NaturalLanguagePrivacyTagger.swift +26 -0
  35. package/apps/ios/EnigmaIOS/Enigma/OAuthClient.swift +321 -0
  36. package/apps/ios/EnigmaIOS/Enigma/OnboardingView.swift +105 -0
  37. package/apps/ios/EnigmaIOS/Enigma/PrivateVaultView.swift +275 -0
  38. package/apps/ios/EnigmaIOS/Enigma/SecureStore.swift +76 -0
  39. package/apps/ios/EnigmaIOS/Enigma/SettingsView.swift +60 -0
  40. package/apps/ios/EnigmaIOS/Enigma/Theme.swift +80 -0
  41. package/apps/ios/EnigmaIOS/EnigmaIOS.xcodeproj/project.pbxproj +211 -0
  42. package/apps/ios/EnigmaIOS/EnigmaIOS.xcodeproj/xcshareddata/xcschemes/Enigma.xcscheme +23 -0
  43. package/apps/native-host/README.md +19 -8
  44. package/apps/native-host/bin/enigma-native-host.mjs +229 -13
  45. package/apps/relay/bin/enigma-relay.mjs +103 -5
  46. package/apps/relay/src/federation-runtime.mjs +618 -0
  47. package/apps/relay/src/server.mjs +310 -9
  48. package/apps/verifier/bin/enigma-verify.mjs +327 -11
  49. package/cortex-v3/circuits/build/intent_vk_bytes.json +35 -0
  50. package/cortex-v3/circuits/build/sale_vk_bytes.json +35 -0
  51. package/cortex-v3/circuits/build/vk_bytes.json +32 -0
  52. package/cortex-v3/proving-assets.json +64 -0
  53. package/cortex-v3/zk/BUILD-CONTRACT.md +87 -0
  54. package/cortex-v3/zk/action-transition-vk.json +119 -0
  55. package/cortex-v3/zk/alias-adversarial.test.mjs +220 -0
  56. package/cortex-v3/zk/groth16-verify-child.mjs +17 -0
  57. package/cortex-v3/zk/intent-witness.mjs +365 -0
  58. package/cortex-v3/zk/intent-witness.test.mjs +485 -0
  59. package/cortex-v3/zk/proving-assets.mjs +203 -0
  60. package/cortex-v3/zk/sale-witness.mjs +783 -0
  61. package/cortex-v3/zk/sale-witness.test.mjs +784 -0
  62. package/cortex-v3/zk/sealed-sale-release-vk.json +119 -0
  63. package/cortex-v3/zk/settlement-evidence.mjs +722 -0
  64. package/cortex-v3/zk/setup-intent.mjs +688 -0
  65. package/cortex-v3/zk/setup-sale.mjs +666 -0
  66. package/cortex-v3/zk/setup.mjs +594 -0
  67. package/cortex-v3/zk/witness.mjs +184 -0
  68. package/cortex-v3/zk/zk-codec.mjs +232 -0
  69. package/cortex-v3/zk/zk-codec.test.mjs +293 -0
  70. package/cortex-v3/zk/zk-settle.mjs +370 -0
  71. package/cortex-v3/zk/zk-tree.mjs +256 -0
  72. package/cortex-v3/zk/zk-tree.test.mjs +419 -0
  73. package/deploy/SIMULATION.md +14 -9
  74. package/deploy/docker-compose.local-production-simulation.yml +54 -12
  75. package/docs/benchmark-attestation-network.md +487 -487
  76. package/docs/benchmark-reproducibility.md +289 -289
  77. package/docs/blockchain-only-mechanisms.md +400 -400
  78. package/docs/browser-extension-install.md +8 -6
  79. package/docs/client-connectors.md +16 -12
  80. package/docs/demo-proof-network.md +275 -275
  81. package/docs/developer-ecosystem.md +15 -13
  82. package/docs/developer-proof-quickstart.md +325 -325
  83. package/docs/enigma-memory-ready-conformance.md +378 -376
  84. package/docs/install-anywhere.md +64 -30
  85. package/docs/installers-and-desktop.md +8 -7
  86. package/docs/memory-benchmarks.md +1 -1
  87. package/docs/memory-drive-health-model.md +690 -690
  88. package/docs/novelty-invention-candidates.md +161 -161
  89. package/docs/proof-network-build-notes.md +240 -240
  90. package/docs/proof-network-claim-boundaries.md +320 -318
  91. package/docs/proof-network.md +339 -339
  92. package/docs/sdk-api.md +324 -324
  93. package/docs/solana-proof-rail.md +453 -453
  94. package/examples/01-quickstart-agent/index.mjs +49 -0
  95. package/examples/01_agent_memory_quickstart.mjs +57 -0
  96. package/examples/02-multi-agent-swarm/index.mjs +57 -0
  97. package/examples/02_cross_model_passport.mjs +64 -0
  98. package/examples/03-langchain-memory/index.mjs +41 -0
  99. package/examples/03_poseidon_commitment_verification.mjs +71 -0
  100. package/examples/04-python-trading-agent/trader.py +49 -0
  101. package/examples/README.md +27 -0
  102. package/examples/ci/github-actions.yml +7 -2
  103. package/package.json +410 -278
  104. package/packages/adapters/PACKAGE_CONTRACT.md +1 -1
  105. package/packages/connectors/src/index.js +196 -4
  106. package/packages/connectors/swarm-router.mjs +168 -0
  107. package/packages/core/src/index.js +249 -2
  108. package/packages/core/src/version.mjs +7 -0
  109. package/packages/dev-tools/package.json +19 -0
  110. package/packages/dev-tools/src/index.js +4 -0
  111. package/packages/dev-tools/src/memory-benchmark-suite.js +112 -0
  112. package/packages/dev-tools/src/swarm-simulator.js +101 -0
  113. package/packages/dev-tools/src/vault-inspector.js +114 -0
  114. package/packages/dev-tools/src/vector-benchmark.js +100 -0
  115. package/packages/developer-platform/src/access-credentials.js +341 -0
  116. package/packages/developer-platform/src/http.js +132 -0
  117. package/packages/developer-platform/src/index.js +4 -0
  118. package/packages/developer-platform/src/usage-http.js +60 -0
  119. package/packages/developer-platform/src/usage.js +295 -0
  120. package/packages/enclave-runtime/attestation.mjs +159 -0
  121. package/packages/enclave-runtime/index.mjs +47 -0
  122. package/packages/enclave-runtime/session-manager.mjs +253 -0
  123. package/packages/enclave-runtime/zeroization-proof.mjs +227 -0
  124. package/packages/enigma-reflex/package.json +14 -0
  125. package/packages/enigma-reflex/src/index.js +204 -0
  126. package/packages/enigma-reflex/training/generate-dataset.mjs +40 -0
  127. package/packages/enigma-reflex/training/requirements.txt +8 -0
  128. package/packages/enigma-reflex/training/train.py +314 -0
  129. package/packages/enigma-weave/LICENSE +22 -0
  130. package/packages/enigma-weave/UPSTREAM.json +21 -0
  131. package/packages/enigma-weave/package.json +14 -0
  132. package/packages/enigma-weave/src/index.js +286 -0
  133. package/packages/hosted-cloud/src/index.js +80 -5
  134. package/packages/importers/src/index.js +432 -0
  135. package/packages/inference-runtime/src/browser.js +401 -0
  136. package/packages/inference-runtime/src/chat.js +265 -0
  137. package/packages/inference-runtime/src/code.js +407 -0
  138. package/packages/inference-runtime/src/contracts.js +162 -0
  139. package/packages/inference-runtime/src/http.js +232 -0
  140. package/packages/inference-runtime/src/image.js +186 -0
  141. package/packages/inference-runtime/src/index.js +10 -0
  142. package/packages/inference-runtime/src/model-router.js +320 -0
  143. package/packages/inference-runtime/src/platform.js +125 -0
  144. package/packages/inference-runtime/src/privacy.js +400 -0
  145. package/packages/inference-runtime/src/video.js +253 -0
  146. package/packages/mcp-server/README.md +22 -6
  147. package/packages/mcp-server/bin/enigma-mcp.mjs +2 -1
  148. package/packages/mcp-server/src/index.js +2498 -1185
  149. package/packages/mcp-server/src/oauth.js +561 -0
  150. package/packages/mcp-server/src/private-handoff.js +84 -0
  151. package/packages/mcp-server/src/remote-http.js +273 -0
  152. package/packages/mcp-server/src/remote-policy.js +72 -0
  153. package/packages/mcp-server/swarm-bridge.mjs +361 -0
  154. package/packages/mesh/index.d.ts +283 -0
  155. package/packages/mesh/package.json +23 -0
  156. package/packages/mesh/src/crypto.js +189 -0
  157. package/packages/mesh/src/federation-packets.js +353 -0
  158. package/packages/mesh/src/gossip.js +311 -0
  159. package/packages/mesh/src/index.js +6 -0
  160. package/packages/mesh/src/protocol.js +255 -0
  161. package/packages/mesh/src/router.js +279 -0
  162. package/packages/mesh/src/transport.js +306 -0
  163. package/packages/passport/src/index.js +436 -7
  164. package/packages/private-economy/src/credits-http.js +100 -0
  165. package/packages/private-economy/src/credits.js +447 -0
  166. package/packages/private-economy/src/index.js +5 -0
  167. package/packages/private-economy/src/payments-http.js +120 -0
  168. package/packages/private-economy/src/payments.js +509 -0
  169. package/packages/private-economy/src/x402.js +346 -0
  170. package/packages/proof-network/PACKAGE_CONTRACT.md +21 -0
  171. package/packages/rag/index.d.ts +182 -0
  172. package/packages/rag/models/Xenova/all-MiniLM-L6-v2/THIRD_PARTY_LICENSES.txt +207 -0
  173. package/packages/rag/models/Xenova/all-MiniLM-L6-v2/config.json +25 -0
  174. package/packages/rag/models/Xenova/all-MiniLM-L6-v2/onnx/model_quantized.onnx +0 -0
  175. package/packages/rag/models/Xenova/all-MiniLM-L6-v2/sha256-manifest.json +28 -0
  176. package/packages/rag/models/Xenova/all-MiniLM-L6-v2/tokenizer.json +30686 -0
  177. package/packages/rag/models/Xenova/all-MiniLM-L6-v2/tokenizer_config.json +15 -0
  178. package/packages/rag/package.json +27 -0
  179. package/packages/rag/src/blinded-search.js +109 -0
  180. package/packages/rag/src/bm25.js +169 -0
  181. package/packages/rag/src/embeddings.js +459 -0
  182. package/packages/rag/src/hybrid.js +76 -0
  183. package/packages/rag/src/index.js +38 -0
  184. package/packages/rag/src/reranker.js +61 -0
  185. package/packages/rag/src/research.js +107 -0
  186. package/packages/rag/src/vector-store.js +430 -0
  187. package/packages/rag/src/verify-model-artifacts.mjs +4 -0
  188. package/packages/sdk/index.d.ts +760 -0
  189. package/packages/sdk/package.json +33 -0
  190. package/packages/sdk/python/README.md +24 -0
  191. package/packages/sdk/python/enigma_sdk.py +250 -0
  192. package/packages/sdk/python/pyproject.toml +34 -0
  193. package/packages/sdk/python/requirements.txt +1 -0
  194. package/packages/sdk/python/setup.py +20 -0
  195. package/packages/sdk/src/federation/capability-grant.js +389 -0
  196. package/packages/sdk/src/federation/federation-router.js +360 -0
  197. package/packages/sdk/src/federation/ghostmesh-bridge.js +497 -0
  198. package/packages/sdk/src/federation/index.js +3 -0
  199. package/packages/sdk/src/index.js +1796 -0
  200. package/packages/sdk/src/intelligence/contradiction.js +337 -0
  201. package/packages/sdk/src/intelligence/decision-engine.js +155 -0
  202. package/packages/sdk/src/intelligence/index.js +4 -0
  203. package/packages/sdk/src/intelligence/ontology.js +122 -0
  204. package/packages/sdk/src/intelligence/temporal.js +123 -0
  205. package/packages/sdk/src/market-client.js +142 -0
  206. package/packages/sdk/src/mesh-client.js +110 -0
  207. package/packages/sdk/src/middleware/index.js +3 -0
  208. package/packages/sdk/src/middleware/langchain.js +159 -0
  209. package/packages/sdk/src/middleware/llamaindex.js +101 -0
  210. package/packages/sdk/src/middleware/vercel-ai.js +112 -0
  211. package/packages/sdk/src/rag-client.js +85 -0
  212. package/packages/sdk/src/swarm-orchestrator.js +260 -0
  213. package/packages/settlement/PACKAGE_CONTRACT.md +1 -1
  214. package/packages/snapcompact/THIRD_PARTY_LICENSES.txt +40 -0
  215. package/packages/snapcompact/assets/8x13-latin1.bdf +3837 -0
  216. package/packages/snapcompact/index.d.ts +284 -0
  217. package/packages/snapcompact/package.json +25 -0
  218. package/packages/snapcompact/src/index.js +716 -0
  219. package/packages/storage/PACKAGE_CONTRACT.md +1 -1
  220. package/packages/terminal-console/animations.mjs +240 -0
  221. package/packages/terminal-console/auto-anchor.mjs +220 -0
  222. package/packages/terminal-console/banner.mjs +91 -0
  223. package/packages/terminal-console/commands.mjs +459 -0
  224. package/packages/terminal-console/delegation.mjs +152 -0
  225. package/packages/terminal-console/index.mjs +5 -0
  226. package/packages/terminal-console/outbox.mjs +143 -0
  227. package/packages/terminal-console/phantom-bridge.mjs +637 -0
  228. package/packages/terminal-console/repl.mjs +136 -0
  229. package/packages/terminal-console/signer-store.mjs +130 -0
  230. package/packages/terminal-console/solana-rpc.mjs +214 -0
  231. package/packages/terminal-console/solana-transport.mjs +189 -0
  232. package/packages/terminal-tui/dashboard.mjs +214 -0
  233. package/packages/terminal-tui/index.mjs +28 -0
  234. package/packages/terminal-tui/merkle-tree-renderer.mjs +268 -0
  235. package/packages/terminal-tui/telemetry-hud.mjs +137 -0
  236. package/packages/vault/index.d.ts +449 -0
  237. package/packages/vault/package.json +27 -0
  238. package/packages/vault/src/e2ee.mjs +393 -0
  239. package/packages/vault/src/enclave.js +481 -0
  240. package/packages/vault/src/erasure.js +207 -0
  241. package/packages/vault/src/index.js +1150 -125
  242. package/packages/vault/src/persistence.js +307 -0
  243. package/packages/vault/src/poseidon.js +354 -0
  244. package/packages/vault/src/receipt.js +459 -0
  245. package/scripts/benchmark-optical-context.mjs +166 -0
  246. package/scripts/bootstrap-enigma.mjs +502 -0
  247. package/scripts/build-edge-backend-workers.mjs +20 -5
  248. package/scripts/build-goal-completion-audit.mjs +72 -25
  249. package/scripts/build-hosted-api-key-lifecycle.mjs +292 -274
  250. package/scripts/build-hosted-customer-lifecycle.mjs +493 -476
  251. package/scripts/build-hosted-probe-worker.mjs +19 -4
  252. package/scripts/build-installer-assets.mjs +409 -389
  253. package/scripts/build-operator-evidence-starter.mjs +59 -1
  254. package/scripts/build-production-backend-env-kit.mjs +2 -0
  255. package/scripts/build-production-unblocker.mjs +4 -1
  256. package/scripts/build-proof-network-packet.mjs +213 -213
  257. package/scripts/check.mjs +25 -4
  258. package/scripts/collect-hosted-backend-live-evidence.mjs +49 -12
  259. package/scripts/install-enigma-local.mjs +18 -5
  260. package/scripts/release-audit.mjs +74 -115
  261. package/scripts/release-provenance.mjs +12 -2
  262. package/scripts/run-backend-readiness-smoke.mjs +112 -10
  263. package/scripts/run-standard-memory-benchmarks.mjs +1354 -1352
  264. package/scripts/scan-secrets.mjs +178 -0
  265. package/scripts/simulate-production-env.mjs +71 -10
  266. package/scripts/validate-hosted-backend-live.mjs +112 -1
  267. package/specs/antibody-pack-v1.schema.json +95 -0
  268. package/specs/antigen-envelope-v1.schema.json +81 -0
  269. package/specs/boundary-manifest-v1.schema.json +35 -35
  270. package/specs/capsule-v1.schema.json +55 -55
  271. package/specs/claim-boundary-manifest-v1.schema.json +22 -22
  272. package/specs/claim-ledger-v1.schema.json +291 -0
  273. package/specs/context-passport-v1.schema.json +59 -0
  274. package/specs/deletion-tombstone-v1.schema.json +26 -26
  275. package/specs/evidence-packet-v1.schema.json +177 -0
  276. package/specs/hosted-backend-live-evidence-v1.schema.json +72 -3
  277. package/specs/immune-scan-report-v1.schema.json +112 -0
  278. package/specs/lifecycle-receipt-log-v1.schema.json +67 -0
  279. package/specs/memory-atom-v1.schema.json +59 -0
  280. package/specs/memory-event-v1.schema.json +42 -42
  281. package/specs/passport-v1.schema.json +50 -50
  282. package/specs/proof-of-non-use-v1.schema.json +65 -0
  283. package/specs/quarantine-record-v1.schema.json +126 -0
  284. package/specs/receipt-v1.schema.json +61 -61
  285. package/specs/state-checkpoint-v1.schema.json +37 -37
  286. package/specs/trust-bundle-v1.schema.json +56 -56
  287. package/specs/trust-card-v1.schema.json +119 -0
  288. package/docs/proof-network-launch-plan.md +0 -421
  289. package/packages/metering/PACKAGE_CONTRACT.md +0 -20
  290. package/scripts/build-ai-orchestration-plan.mjs +0 -248
@@ -1,1352 +1,1354 @@
1
- #!/usr/bin/env node
2
- import { createReadStream } from 'node:fs';
3
- import { mkdir, readFile, writeFile } from 'node:fs/promises';
4
- import { basename, dirname, resolve } from 'node:path';
5
- import { fileURLToPath } from 'node:url';
6
- import { createHash } from 'node:crypto';
7
- import { performance } from 'node:perf_hooks';
8
- import { StringDecoder } from 'node:string_decoder';
9
- import { estimateTextTokens } from '../packages/optimizer/src/index.js';
10
-
11
- export const STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA = 'enigma.standard_memory_benchmark_suite.v1';
12
-
13
- export const STANDARD_MEMORY_BENCHMARK_METHODS = Object.freeze([
14
- Object.freeze({
15
- id: 'full_context',
16
- label: 'Full context',
17
- boundary: 'Supplies every parsed memory record for each query; no provider API or model answer generation.',
18
- uses_top_k: false,
19
- }),
20
- Object.freeze({
21
- id: 'recency_last_n',
22
- label: 'Recency last N',
23
- boundary: 'Supplies the most recent local memory records up to --top-k.',
24
- uses_top_k: true,
25
- }),
26
- Object.freeze({
27
- id: 'keyword_filter',
28
- label: 'Keyword filter',
29
- boundary: 'Supplies local memory records whose public-safe deterministic tokens overlap the query, capped by --top-k.',
30
- uses_top_k: true,
31
- }),
32
- Object.freeze({
33
- id: 'enigma_relevance',
34
- label: 'Enigma relevance',
35
- boundary: 'Uses deterministic query-aware relevance features over local public-safe memory metadata and content tokens, then ranks locally without provider APIs.',
36
- uses_top_k: true,
37
- }),
38
- ]);
39
-
40
- export const STANDARD_EXTERNAL_COMPETITOR_ADAPTERS = Object.freeze([
41
- Object.freeze({
42
- id: 'mem0',
43
- name: 'Mem0',
44
- status: 'not_run_requires_credentials_or_runtime',
45
- target_type: 'external_adapter',
46
- can_run_in_this_harness: false,
47
- scores_included: false,
48
- required_artifacts: Object.freeze([
49
- 'Mem0 platform credentials or open-source runtime',
50
- 'Pinned Mem0 SDK/package versions',
51
- 'Fixed extraction, update, retrieval, reset, model, and tool policy',
52
- 'Same reviewed dataset manifest, split, top-k, and scorer as Enigma rows',
53
- ]),
54
- official_doc: 'https://docs.mem0.ai/',
55
- boundary_reason: 'The standard runner has no Mem0 credentials, SDK/runtime, fixed memory loop, reset policy, model/tool environment, or reviewed adapter scorer, so no Mem0 score is produced.',
56
- }),
57
- ]);
58
-
59
- const LOCOMO_SOURCE_URL = 'https://raw.githubusercontent.com/snap-research/locomo/main/data/locomo10.json';
60
- const LONGMEMEVAL_SOURCE_URLS = Object.freeze([
61
- 'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_oracle.json',
62
- 'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_s_cleaned.json',
63
- 'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_m_cleaned.json',
64
- ]);
65
- const LOCOMO_TASK_CATEGORIES = Object.freeze(['multi-session QA', 'event summarization', 'multimodal generation over long conversations']);
66
- const LONGMEMEVAL_TASK_CATEGORIES = Object.freeze(['information extraction', 'multi-session reasoning', 'temporal reasoning', 'knowledge updates', 'abstention']);
67
- const DATASET_TASK_CATEGORIES = Object.freeze({ locomo: LOCOMO_TASK_CATEGORIES, longmemeval: LONGMEMEVAL_TASK_CATEGORIES });
68
- export const STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA = 'enigma.standard_memory_benchmark_protocol_plan.v1';
69
- const PROTOCOL_REF_RE = /^[a-z0-9][a-z0-9._:/@+-]{2,191}$/u;
70
- const DEFAULT_ANSWERER_MODEL_REF = 'model:answerer-not-selected';
71
- const DEFAULT_JUDGE_MODEL_REF = 'model:judge-not-selected';
72
- const DEFAULT_ANSWER_PROMPT_REF = 'prompt:standard-answer@not-pinned';
73
- const DEFAULT_JUDGE_PROMPT_REF = 'prompt:standard-judge@not-pinned';
74
- const DEFAULT_PROTOCOL_REF = 'protocol:apples-to-apples-full-answer@not-pinned';
75
-
76
- const QUERY_RELEVANCE_STOPWORDS = new Set([
77
- 'about',
78
- 'after',
79
- 'again',
80
- 'against',
81
- 'also',
82
- 'and',
83
- 'any',
84
- 'are',
85
- 'assistant',
86
- 'because',
87
- 'been',
88
- 'before',
89
- 'being',
90
- 'between',
91
- 'can',
92
- 'could',
93
- 'current',
94
- 'does',
95
- 'from',
96
- 'has',
97
- 'have',
98
- 'how',
99
- 'into',
100
- 'its',
101
- 'latest',
102
- 'more',
103
- 'most',
104
- 'number',
105
- 'own',
106
- 'owns',
107
- 'please',
108
- 'should',
109
- 'that',
110
- 'the',
111
- 'their',
112
- 'then',
113
- 'there',
114
- 'these',
115
- 'they',
116
- 'this',
117
- 'use',
118
- 'using',
119
- 'was',
120
- 'what',
121
- 'when',
122
- 'where',
123
- 'which',
124
- 'who',
125
- 'whose',
126
- 'why',
127
- 'with',
128
- 'would',
129
- ]);
130
-
131
- function parseArgs(argv = process.argv.slice(2)) {
132
- const options = { top_k: 5 };
133
- for (let index = 0; index < argv.length; index += 1) {
134
- const arg = argv[index];
135
- if (arg === '--locomo') {
136
- options.locomo = requiredFlagValue(argv, index, arg);
137
- index += 1;
138
- } else if (arg === '--longmemeval') {
139
- options.longmemeval = requiredFlagValue(argv, index, arg);
140
- index += 1;
141
- } else if (arg === '--max-locomo-qa') {
142
- options.max_locomo_qa = positiveInteger(requiredFlagValue(argv, index, arg), arg);
143
- index += 1;
144
- } else if (arg === '--max-longmemeval-items') {
145
- options.max_longmemeval_items = positiveInteger(requiredFlagValue(argv, index, arg), arg);
146
- index += 1;
147
- } else if (arg === '--top-k') {
148
- options.top_k = positiveInteger(requiredFlagValue(argv, index, arg), arg);
149
- index += 1;
150
- } else if (arg === '--out') {
151
- options.out = requiredFlagValue(argv, index, arg);
152
- index += 1;
153
- } else if (arg === '--protocol-plan') {
154
- options.protocol_plan = true;
155
- } else if (arg === '--answerer-ref') {
156
- options.answerer_ref = requiredFlagValue(argv, index, arg);
157
- index += 1;
158
- } else if (arg === '--judge-ref') {
159
- options.judge_ref = requiredFlagValue(argv, index, arg);
160
- index += 1;
161
- } else if (arg === '--answer-prompt-ref') {
162
- options.answer_prompt_ref = requiredFlagValue(argv, index, arg);
163
- index += 1;
164
- } else if (arg === '--judge-prompt-ref') {
165
- options.judge_prompt_ref = requiredFlagValue(argv, index, arg);
166
- index += 1;
167
- } else if (arg === '--protocol-ref') {
168
- options.protocol_ref = requiredFlagValue(argv, index, arg);
169
- index += 1;
170
- } else if (arg === '--dry-run') {
171
- options.dry_run = true;
172
- } else if (arg === '--help' || arg === '-h') {
173
- options.help = true;
174
- } else {
175
- throw new Error(`Unknown option ${arg}`);
176
- }
177
- }
178
- return options;
179
- }
180
-
181
- function requiredFlagValue(argv, index, flag) {
182
- const value = argv[index + 1];
183
- if (value === undefined || value.startsWith('--')) throw new Error(`${flag} requires a value`);
184
- return value;
185
- }
186
-
187
- function positiveInteger(value, name) {
188
- const number = Number(value);
189
- if (!Number.isInteger(number) || number <= 0) throw new Error(`${name} must be a positive integer`);
190
- return number;
191
- }
192
-
193
- function optionalPositiveInteger(value, name) {
194
- if (value === undefined || value === null) return undefined;
195
- return positiveInteger(value, name);
196
- }
197
-
198
- function datasetPlanRows(options) {
199
- const rows = [];
200
- if (options.locomo !== undefined || options.locomoPath !== undefined) {
201
- rows.push({
202
- id: 'locomo',
203
- label: 'LoCoMo',
204
- local_file_name: publicFileName(options.locomo ?? options.locomoPath),
205
- source_url: LOCOMO_SOURCE_URL,
206
- license: 'CC BY-NC 4.0',
207
- sample_limit: optionalPositiveInteger(options.max_locomo_qa ?? options.maxLocomoQa, 'max_locomo_qa') ?? null,
208
- parser: 'conversation session turns as memory records; qa evidence labels score support only',
209
- });
210
- }
211
- if (options.longmemeval !== undefined || options.longmemevalPath !== undefined || options.longMemEvalPath !== undefined) {
212
- rows.push({
213
- id: 'longmemeval',
214
- label: 'LongMemEval',
215
- local_file_name: publicFileName(options.longmemeval ?? options.longmemevalPath ?? options.longMemEvalPath),
216
- source_url: LONGMEMEVAL_SOURCE_URLS,
217
- license: 'Review upstream Hugging Face dataset card and LongMemEval repository terms.',
218
- sample_limit: optionalPositiveInteger(options.max_longmemeval_items ?? options.maxLongMemEvalItems, 'max_longmemeval_items') ?? null,
219
- parser: 'haystack_sessions turns as memory records; answer-session labels score support only',
220
- });
221
- }
222
- return rows;
223
- }
224
-
225
- function offlineCommandBoundaries({ scoresIncluded, datasetFilesRead }) {
226
- return {
227
- deterministic_offline_runner: true,
228
- dataset_files_read_from_local_disk: datasetFilesRead,
229
- network_calls_made: false,
230
- provider_api_calls_made: false,
231
- api_spend_possible: false,
232
- hosted_memory_service_called: false,
233
- external_competitor_adapters_run: false,
234
- mem0_adapter_run: false,
235
- llm_used: false,
236
- llm_answer_accuracy_scored: false,
237
- retrieval_evidence_proxy_scored: scoresIncluded,
238
- benchmark_scores_included: scoresIncluded,
239
- raw_question_text_included: false,
240
- raw_answer_text_included: false,
241
- raw_conversation_text_included: false,
242
- gold_labels_used_for_retrieval: false,
243
- gold_labels_used_for_scoring: scoresIncluded,
244
- };
245
- }
246
-
247
- function applesToApplesControls(topK) {
248
- return {
249
- same_top_k_for_all_methods: true,
250
- top_k: topK,
251
- same_parser_per_dataset: true,
252
- same_local_records_per_dataset: true,
253
- same_gold_evidence_labels_per_dataset_for_scoring_only: true,
254
- local_deterministic_methods_only: true,
255
- provider_runtime_fixed: false,
256
- competitor_runtime_fixed: false,
257
- answer_generator_fixed: false,
258
- evaluator_model_fixed: false,
259
- };
260
- }
261
-
262
- export function buildStandardBenchmarkDryRunPlan(options = {}) {
263
- const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
264
- const datasets = datasetPlanRows(options);
265
- if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
266
- return {
267
- schema: 'enigma.standard_memory_benchmark_plan.v1',
268
- generated_at: options.generated_at ?? new Date().toISOString(),
269
- package: {
270
- name: 'enigma-memory',
271
- version: '0.1.17',
272
- },
273
- public_safe: true,
274
- dry_run: true,
275
- top_k: topK,
276
- datasets_planned: datasets,
277
- local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
278
- external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
279
- ...adapter,
280
- required_artifacts: [...adapter.required_artifacts],
281
- })),
282
- command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
283
- apples_to_apples_controls: applesToApplesControls(topK),
284
- non_claims: [
285
- 'This dry run does not read dataset files and produces no benchmark score.',
286
- 'No provider APIs, hosted memory services, Mem0 runtime, competitor SDKs, LLM generators, or evaluator models are called.',
287
- 'A scored report requires a separate non-dry-run command against the exact local dataset files and hashes.',
288
- ],
289
- };
290
- }
291
- function protocolRef(value, label, fallback) {
292
- if (value === undefined || value === null) return fallback;
293
- const normalized = typeof value === 'string' ? value.trim() : String(value);
294
- if (normalized === '') throw new Error(`${label} must be a non-empty public ref`);
295
- if (!PROTOCOL_REF_RE.test(normalized)) throw new Error(`${label} must be a lowercase public ref using letters, numbers, . _ : / @ + or -`);
296
- return normalized;
297
- }
298
-
299
- function protocolPlanCategorySet(datasets) {
300
- const categorySet = [];
301
- const seen = new Set();
302
- for (const row of datasets) {
303
- for (const category of DATASET_TASK_CATEGORIES[row.id] ?? []) {
304
- if (!seen.has(category)) {
305
- seen.add(category);
306
- categorySet.push(category);
307
- }
308
- }
309
- }
310
- return categorySet;
311
- }
312
-
313
- export function buildStandardBenchmarkProtocolPlan(options = {}) {
314
- const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
315
- const datasets = datasetPlanRows(options);
316
- if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
317
- const answererProvided = (options.answerer_ref ?? options.answererModelRef) !== undefined;
318
- const judgeProvided = (options.judge_ref ?? options.judgeModelRef) !== undefined;
319
- const answererRef = protocolRef(options.answerer_ref ?? options.answererModelRef, 'answerer_ref', DEFAULT_ANSWERER_MODEL_REF);
320
- const judgeRef = protocolRef(options.judge_ref ?? options.judgeModelRef, 'judge_ref', DEFAULT_JUDGE_MODEL_REF);
321
- const answerPromptRef = protocolRef(options.answer_prompt_ref ?? options.answerPromptRef, 'answer_prompt_ref', DEFAULT_ANSWER_PROMPT_REF);
322
- const judgePromptRef = protocolRef(options.judge_prompt_ref ?? options.judgePromptRef, 'judge_prompt_ref', DEFAULT_JUDGE_PROMPT_REF);
323
- const protocolRefValue = protocolRef(options.protocol_ref ?? options.protocolRef, 'protocol_ref', DEFAULT_PROTOCOL_REF);
324
- const promptsFixed = answererProvided && judgeProvided;
325
- return {
326
- schema: STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA,
327
- generated_at: options.generated_at ?? new Date().toISOString(),
328
- package: {
329
- name: 'enigma-memory',
330
- version: '0.1.17',
331
- },
332
- public_safe: true,
333
- protocol_plan: true,
334
- dry_run: true,
335
- top_k: topK,
336
- category_set: protocolPlanCategorySet(datasets),
337
- datasets_planned: datasets,
338
- answerer: {
339
- model_ref: answererRef,
340
- temperature: 0,
341
- max_tokens: 1024,
342
- fixed: answererProvided,
343
- },
344
- judge: {
345
- model_ref: judgeRef,
346
- kind: 'llm-as-judge-or-exact-match-not-selected',
347
- temperature: 0,
348
- max_tokens: 512,
349
- fixed: judgeProvided,
350
- },
351
- prompt_refs: [answerPromptRef, judgePromptRef],
352
- protocol_refs: [protocolRefValue],
353
- competitor_adapter_refs: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => `adapter:${adapter.id}@not-pinned`),
354
- external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
355
- ...adapter,
356
- required_artifacts: [...adapter.required_artifacts],
357
- adapter_ref: `adapter:${adapter.id}@not-pinned`,
358
- })),
359
- apples_to_apples_controls: applesToApplesControls(topK),
360
- protocol_controls: {
361
- same_answerer_model_for_all_rows: answererProvided,
362
- same_judge_model_for_all_rows: judgeProvided,
363
- same_prompts_for_all_rows: promptsFixed,
364
- same_competitor_adapters_for_enigma_and_baselines: false,
365
- answerer_model_fixed: answererProvided,
366
- judge_model_fixed: judgeProvided,
367
- prompts_fixed: promptsFixed,
368
- temperature_fixed: false,
369
- budget_caps_set: false,
370
- },
371
- cost_estimate_inputs: {
372
- dataset_sample_limits: datasets.map((row) => ({ dataset: row.id, sample_limit: row.sample_limit })),
373
- answerer_temperature: 0,
374
- answerer_max_tokens: 1024,
375
- judge_temperature: 0,
376
- judge_max_tokens: 512,
377
- max_retries: 0,
378
- request_timeout_ms: null,
379
- budget_cap_required_before_run: true,
380
- budget_cap_set: false,
381
- },
382
- benchmark_boundaries: {
383
- official_dataset_files_required: true,
384
- credentials_required: false,
385
- external_provider_calls: false,
386
- llm_answer_accuracy_scored: false,
387
- retrieval_evidence_proxy_scored: false,
388
- raw_question_text_included: false,
389
- raw_answer_text_included: false,
390
- raw_conversation_text_included: false,
391
- provider_deletion_claim: false,
392
- model_forgetting_claim: false,
393
- roi_or_provider_invoice_savings_claim: false,
394
- compliance_certification_claim: false,
395
- benchmark_leadership_claim: false,
396
- },
397
- protocol_boundaries: {
398
- network_required: false,
399
- provider_calls_made: false,
400
- answers_generated: false,
401
- judged: false,
402
- competitor_adapters_run: false,
403
- api_spend_possible: false,
404
- },
405
- command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
406
- review_rules: [
407
- 'report hash format: sha256:<hex> over the public-safe protocol-plan JSON bytes',
408
- 'dataset ref, runner ref, package ref, environment ref, and verifier ref formats are pinned before a live run',
409
- 'metric scope: this artifact plans the full-answer protocol only; it is not a score',
410
- 'limitation: no provider answer-accuracy, competitor-performance, benchmark-leadership, ROI, provider-deletion, model-forgetting, or compliance claim is made',
411
- ],
412
- non_claims: [
413
- 'This protocol plan is a readiness artifact: it records the planned full-answer protocol dimensions and does not execute it.',
414
- 'No network is used, no provider APIs are called, no answers are generated, no answers are judged, and no competitor adapters are run.',
415
- 'Answerer/judge model refs, prompt refs, and competitor adapter refs are public-safe references only; they are not credentials, model calls, prompts, or scores.',
416
- 'A scored full-answer run requires a separate credentialed command with frozen models, prompts, evaluator, dataset manifest, and budget caps.',
417
- ],
418
- };
419
- }
420
-
421
- function publicFileName(path) {
422
- return path === undefined || path === null ? undefined : basename(String(path));
423
- }
424
-
425
- async function readJsonWithSha256(path) {
426
- const raw = await readFile(path, 'utf8');
427
- return {
428
- data: JSON.parse(raw),
429
- sha256: createHash('sha256').update(raw).digest('hex'),
430
- };
431
- }
432
-
433
- function isJsonWhitespace(char) {
434
- const code = char.charCodeAt(0);
435
- return code === 0x20 || code === 0x0a || code === 0x0d || code === 0x09 || code === 0xfeff;
436
- }
437
-
438
- function createTopLevelArraySampler(maxItems, name) {
439
- const items = [];
440
- let started = false;
441
- let closed = false;
442
- let collecting = false;
443
- let doneCollecting = false;
444
- let expectSeparator = false;
445
- let depth = 0;
446
- let inString = false;
447
- let escaped = false;
448
- let current = '';
449
-
450
- function fail(message) {
451
- throw new Error(`${name} sample-mode JSON parse failed: ${message}`);
452
- }
453
-
454
- function finishItem() {
455
- items.push(current);
456
- current = '';
457
- collecting = false;
458
- expectSeparator = true;
459
- if (items.length >= maxItems) doneCollecting = true;
460
- }
461
-
462
- return {
463
- write(text) {
464
- for (const char of text) {
465
- if (!started) {
466
- if (isJsonWhitespace(char)) continue;
467
- if (char !== '[') fail('expected a top-level array');
468
- started = true;
469
- continue;
470
- }
471
-
472
- if (doneCollecting) continue;
473
-
474
- if (closed) {
475
- if (!isJsonWhitespace(char)) fail('found trailing data after the top-level array');
476
- continue;
477
- }
478
-
479
- if (!collecting) {
480
- if (isJsonWhitespace(char)) continue;
481
- if (expectSeparator) {
482
- if (char === ',') {
483
- expectSeparator = false;
484
- continue;
485
- }
486
- if (char === ']') {
487
- closed = true;
488
- continue;
489
- }
490
- fail('expected a comma or closing bracket between items');
491
- }
492
- if (char === ']') {
493
- closed = true;
494
- continue;
495
- }
496
- if (char !== '{' && char !== '[') fail('expected each sampled item to be an object or array');
497
- collecting = true;
498
- current = char;
499
- depth = 1;
500
- continue;
501
- }
502
-
503
- current += char;
504
- if (inString) {
505
- if (escaped) {
506
- escaped = false;
507
- } else if (char === '\\') {
508
- escaped = true;
509
- } else if (char === '"') {
510
- inString = false;
511
- }
512
- continue;
513
- }
514
- if (char === '"') {
515
- inString = true;
516
- } else if (char === '{' || char === '[') {
517
- depth += 1;
518
- } else if (char === '}' || char === ']') {
519
- depth -= 1;
520
- if (depth < 0) fail('encountered an unmatched closing bracket');
521
- if (depth === 0) finishItem();
522
- }
523
- }
524
- },
525
- get done() {
526
- return doneCollecting;
527
- },
528
- finish() {
529
- if (!started) fail('empty input');
530
- if (!doneCollecting && collecting) fail('ended inside a sampled item');
531
- if (!doneCollecting && inString) fail('ended inside a string');
532
- if (!doneCollecting && !closed) fail('ended before the top-level array closed');
533
- return JSON.parse(`[${items.join(',')}]`);
534
- },
535
- };
536
- }
537
-
538
- async function readJsonArraySampleWithSha256(path, maxItems, name) {
539
- const hash = createHash('sha256');
540
- const decoder = new StringDecoder('utf8');
541
- const sampler = createTopLevelArraySampler(maxItems, name);
542
- for await (const chunk of createReadStream(path)) {
543
- hash.update(chunk);
544
- if (!sampler.done) sampler.write(decoder.write(chunk));
545
- }
546
- if (!sampler.done) {
547
- const tail = decoder.end();
548
- if (tail.length > 0) sampler.write(tail);
549
- }
550
- return {
551
- data: sampler.finish(),
552
- sha256: hash.digest('hex'),
553
- };
554
- }
555
-
556
- async function readLongMemEvalJsonWithSha256(path, maxItems) {
557
- if (maxItems === undefined) return readJsonWithSha256(path);
558
- try {
559
- return await readJsonArraySampleWithSha256(path, maxItems, 'LongMemEval');
560
- } catch (error) {
561
- if (error instanceof Error && error.message === 'LongMemEval sample-mode JSON parse failed: expected a top-level array') {
562
- return readJsonWithSha256(path);
563
- }
564
- throw error;
565
- }
566
- }
567
-
568
- function normalizeDatasetArray(data, name) {
569
- if (Array.isArray(data)) return data;
570
- if (data && typeof data === 'object') {
571
- for (const key of ['data', 'items', 'examples', 'samples']) {
572
- if (Array.isArray(data[key])) return data[key];
573
- }
574
- }
575
- throw new TypeError(`${name} dataset must be a JSON array or object containing an array`);
576
- }
577
-
578
- function addMeaningfulToken(tokens, token) {
579
- if (token.length < 3) return;
580
- if (!/[a-z]/u.test(token)) return;
581
- if (QUERY_RELEVANCE_STOPWORDS.has(token)) return;
582
- tokens.add(token);
583
- }
584
-
585
- function stemToken(token) {
586
- if (token.length > 5 && token.endsWith('ing')) {
587
- let stem = token.slice(0, -3);
588
- if (stem.length > 3 && stem.at(-1) === stem.at(-2)) stem = stem.slice(0, -1);
589
- return stem;
590
- }
591
- if (token.length > 4 && token.endsWith('ed')) {
592
- let stem = token.slice(0, -2);
593
- if (stem.length > 3 && stem.at(-1) === stem.at(-2)) stem = stem.slice(0, -1);
594
- return stem;
595
- }
596
- if (token.length > 4 && token.endsWith('ies')) return `${token.slice(0, -3)}y`;
597
- if (token.length > 4 && token.endsWith('es')) return token.slice(0, -2);
598
- if (token.length > 3 && token.endsWith('s') && !token.endsWith('ss')) return token.slice(0, -1);
599
- return token;
600
- }
601
-
602
- function addStemmedMeaningfulToken(tokens, token) {
603
- addMeaningfulToken(tokens, token);
604
- const stem = stemToken(token);
605
- addMeaningfulToken(tokens, stem);
606
- if (token.endsWith('ed') || token.endsWith('ing')) addMeaningfulToken(tokens, `${stem}e`);
607
- }
608
-
609
- function meaningfulTokensFrom(value) {
610
- const tokens = new Set();
611
- if (value === undefined || value === null) return tokens;
612
- for (const match of String(value).toLowerCase().matchAll(/[a-z0-9]+(?:[-_][a-z0-9]+)*/gu)) {
613
- const token = match[0];
614
- addMeaningfulToken(tokens, token);
615
- if (token.includes('-') || token.includes('_')) {
616
- for (const part of token.split(/[-_]+/u)) addMeaningfulToken(tokens, part);
617
- }
618
- }
619
- return tokens;
620
- }
621
-
622
- function stemmedMeaningfulTokensFrom(value) {
623
- const tokens = new Set();
624
- if (value === undefined || value === null) return tokens;
625
- for (const match of String(value).toLowerCase().matchAll(/[a-z0-9]+(?:[-_][a-z0-9]+)*/gu)) {
626
- const token = match[0];
627
- addStemmedMeaningfulToken(tokens, token);
628
- if (token.includes('-') || token.includes('_')) {
629
- for (const part of token.split(/[-_]+/u)) addStemmedMeaningfulToken(tokens, part);
630
- }
631
- }
632
- return tokens;
633
- }
634
-
635
- function stemmedTokenSequenceFrom(value) {
636
- const sequence = [];
637
- if (value === undefined || value === null) return sequence;
638
- for (const match of String(value).toLowerCase().matchAll(/[a-z0-9]+(?:[-_][a-z0-9]+)*/gu)) {
639
- const token = match[0];
640
- const parts = token.includes('-') || token.includes('_') ? token.split(/[-_]+/u) : [token];
641
- for (const part of parts) {
642
- const stem = stemToken(part);
643
- if (stem.length >= 3 && /[a-z]/u.test(stem) && !QUERY_RELEVANCE_STOPWORDS.has(stem)) sequence.push(stem);
644
- }
645
- }
646
- return sequence;
647
- }
648
-
649
- function tokenOverlapScore(queryTokens, record) {
650
- if (queryTokens.size === 0) return 0;
651
- const recordTokens = meaningfulTokensFrom(record.content);
652
- let score = 0;
653
- for (const token of queryTokens) if (recordTokens.has(token)) score += 1;
654
- return score;
655
- }
656
-
657
- function normalizedTagSessionToken(value) {
658
- return String(value ?? '').toLowerCase().replace(/^session[-_:]?/u, '').replace(/[^a-z0-9]+/gu, '');
659
- }
660
-
661
- function roleHintsFrom(value) {
662
- const hints = new Set();
663
- const text = String(value ?? '').toLowerCase();
664
- for (const match of text.matchAll(/\b(?:assistant|user|system|human|agent|speaker[-_\s]?[a-z0-9]+)\b/gu)) {
665
- const compact = match[0].replace(/\s+/gu, '_');
666
- hints.add(compact);
667
- for (const token of stemmedMeaningfulTokensFrom(compact)) hints.add(token);
668
- }
669
- return hints;
670
- }
671
-
672
- function sessionHintsFrom(value) {
673
- const hints = new Set();
674
- const text = String(value ?? '').toLowerCase();
675
- for (const match of text.matchAll(/\b(?:session|sess)\s*[-_:]?\s*([a-z0-9]+)\b/gu)) hints.add(match[1]);
676
- for (const match of text.matchAll(/\bd\s*[-_:]?\s*(\d+)\b/gu)) hints.add(`d${Number(match[1])}`);
677
- for (const match of text.matchAll(/\bsession[-_]([a-z0-9]+)\b/gu)) hints.add(match[1]);
678
- return hints;
679
- }
680
-
681
- function dateHintsFrom(value) {
682
- const hints = new Set();
683
- const text = String(value ?? '').toLowerCase();
684
- for (const match of text.matchAll(/\b(?:19|20)\d{2}\b/gu)) hints.add(match[0]);
685
- for (const match of text.matchAll(/\b\d{4}[-/]\d{1,2}(?:[-/]\d{1,2})?\b/gu)) hints.add(match[0].replace(/\D+/gu, '-'));
686
- for (const match of text.matchAll(/\b(?:jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|jun(?:e)?|jul(?:y)?|aug(?:ust)?|sep(?:t(?:ember)?)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)\b/gu)) hints.add(match[0].slice(0, 3));
687
- for (const match of text.matchAll(/\b(?:today|yesterday|tomorrow|recent|recently|latest|newest|current|previous|last|earliest|oldest)\b/gu)) hints.add(match[0]);
688
- return hints;
689
- }
690
-
691
- function recordSessionTokens(record) {
692
- const tokens = new Set();
693
- const values = [record.session_id, record.dialog_id, record.turn_id, ...(record.tags ?? [])];
694
- for (const value of values) {
695
- if (value === undefined || value === null) continue;
696
- const normalized = normalizedTagSessionToken(value);
697
- if (normalized) tokens.add(normalized);
698
- for (const hint of sessionHintsFrom(value)) tokens.add(normalizedTagSessionToken(hint));
699
- }
700
- return tokens;
701
- }
702
-
703
- function countSetIntersection(left, right) {
704
- let count = 0;
705
- for (const value of left) if (right.has(value)) count += 1;
706
- return count;
707
- }
708
-
709
- function phraseAndProximityScore(querySequence, recordSequence) {
710
- if (querySequence.length < 2 || recordSequence.length < 2) return 0;
711
- const recordBigrams = new Set();
712
- const positions = new Map();
713
- for (let index = 0; index < recordSequence.length; index += 1) {
714
- const token = recordSequence[index];
715
- if (!positions.has(token)) positions.set(token, []);
716
- positions.get(token).push(index);
717
- if (index > 0) recordBigrams.add(`${recordSequence[index - 1]}\u0000${token}`);
718
- }
719
- let score = 0;
720
- for (let index = 1; index < querySequence.length; index += 1) {
721
- const previous = querySequence[index - 1];
722
- const current = querySequence[index];
723
- if (previous === current) continue;
724
- if (recordBigrams.has(`${previous}\u0000${current}`)) {
725
- score += 10;
726
- continue;
727
- }
728
- const leftPositions = positions.get(previous);
729
- const rightPositions = positions.get(current);
730
- if (!leftPositions || !rightPositions) continue;
731
- let near = false;
732
- for (const left of leftPositions) {
733
- for (const right of rightPositions) {
734
- if (Math.abs(left - right) <= 6) {
735
- near = true;
736
- break;
737
- }
738
- }
739
- if (near) break;
740
- }
741
- if (near) score += 4;
742
- }
743
- return score;
744
- }
745
-
746
- function hasRecencyIntent(hints) {
747
- for (const hint of hints) {
748
- if (hint === 'recent' || hint === 'recently' || hint === 'latest' || hint === 'newest' || hint === 'current' || hint === 'previous' || hint === 'last') return true;
749
- }
750
- return false;
751
- }
752
-
753
- function enigmaRelevanceScore(query, record) {
754
- const question = String(query.question ?? '');
755
- const queryTokens = stemmedMeaningfulTokensFrom(question);
756
- const categoryTokens = stemmedMeaningfulTokensFrom(`${query.category ?? ''} ${query.question_type ?? ''}`);
757
- const contentTokens = stemmedMeaningfulTokensFrom(record.content);
758
- const metadataTokens = stemmedMeaningfulTokensFrom(`${record.kind ?? ''} ${(record.tags ?? []).join(' ')}`);
759
- for (const token of stemmedMeaningfulTokensFrom(`${record.role ?? ''} ${record.session_id ?? ''} ${record.dialog_id ?? ''} ${record.turn_id ?? ''}`)) {
760
- metadataTokens.add(token);
761
- }
762
-
763
- const contentMatches = countSetIntersection(queryTokens, contentTokens);
764
- const metadataMatches = countSetIntersection(queryTokens, metadataTokens);
765
- const categoryMatches = countSetIntersection(categoryTokens, metadataTokens) + countSetIntersection(categoryTokens, contentTokens);
766
- const roleMatches = countSetIntersection(roleHintsFrom(question), roleHintsFrom(record.role));
767
- const querySessionHints = sessionHintsFrom(question);
768
- const recordSessions = recordSessionTokens(record);
769
- const sessionMatches = countSetIntersection(querySessionHints, recordSessions);
770
- const queryDateHints = dateHintsFrom(question);
771
- const temporalMatches = countSetIntersection(queryDateHints, dateHintsFrom(`${record.content} ${(record.tags ?? []).join(' ')}`));
772
- const temporalRecencyScore = hasRecencyIntent(queryDateHints) ? Math.min(6, Math.log2(record.ordinal + 2)) : 0;
773
- const phraseScore = phraseAndProximityScore(stemmedTokenSequenceFrom(question), stemmedTokenSequenceFrom(record.content));
774
-
775
- return {
776
- record,
777
- score: (contentMatches * 12)
778
- + (metadataMatches * 4)
779
- + (categoryMatches * 3)
780
- + (roleMatches * 18)
781
- + (sessionMatches * 16)
782
- + (temporalMatches * 8)
783
- + temporalRecencyScore
784
- + phraseScore,
785
- contentMatches,
786
- metadataMatches,
787
- categoryMatches,
788
- roleMatches,
789
- sessionMatches,
790
- temporalMatches,
791
- temporalRecencyScore,
792
- phraseScore,
793
- };
794
- }
795
-
796
- function compareRecordId(left, right) {
797
- return String(left.id).localeCompare(String(right.id));
798
- }
799
-
800
- function compareRecencyDesc(left, right) {
801
- if (left.ordinal !== right.ordinal) return right.ordinal - left.ordinal;
802
- return compareRecordId(left, right);
803
- }
804
-
805
- function rankedByOverlap(records, query) {
806
- const queryTokens = meaningfulTokensFrom(query);
807
- if (queryTokens.size === 0) return [];
808
- const scored = [];
809
- for (const record of records) {
810
- const score = tokenOverlapScore(queryTokens, record);
811
- if (score > 0) scored.push({ record, score });
812
- }
813
- scored.sort((left, right) => {
814
- if (left.score !== right.score) return right.score - left.score;
815
- return compareRecencyDesc(left.record, right.record);
816
- });
817
- return scored.map((item) => item.record);
818
- }
819
-
820
- function rankedByEnigmaRelevance(records, query) {
821
- const scored = [];
822
- for (const record of records) {
823
- const item = enigmaRelevanceScore(query, record);
824
- if (item.score > 0) scored.push(item);
825
- }
826
- if (scored.length === 0) return [...records].sort(compareRecencyDesc);
827
- scored.sort((left, right) => {
828
- if (left.score !== right.score) return right.score - left.score;
829
- if (left.phraseScore !== right.phraseScore) return right.phraseScore - left.phraseScore;
830
- if (left.contentMatches !== right.contentMatches) return right.contentMatches - left.contentMatches;
831
- if (left.roleMatches !== right.roleMatches) return right.roleMatches - left.roleMatches;
832
- if (left.sessionMatches !== right.sessionMatches) return right.sessionMatches - left.sessionMatches;
833
- if (left.temporalMatches !== right.temporalMatches) return right.temporalMatches - left.temporalMatches;
834
- if (left.temporalRecencyScore !== right.temporalRecencyScore) return right.temporalRecencyScore - left.temporalRecencyScore;
835
- if (left.metadataMatches !== right.metadataMatches) return right.metadataMatches - left.metadataMatches;
836
- return compareRecencyDesc(left.record, right.record);
837
- });
838
- return scored.map((item) => item.record);
839
- }
840
-
841
- function selectRecords(methodId, records, query, topK) {
842
- if (methodId === 'full_context') return records;
843
- if (methodId === 'recency_last_n') return [...records].sort(compareRecencyDesc).slice(0, topK);
844
- if (methodId === 'keyword_filter') return rankedByOverlap(records, query.question).slice(0, topK);
845
- if (methodId === 'enigma_relevance') return rankedByEnigmaRelevance(records, query).slice(0, topK);
846
- throw new Error(`Unknown method ${methodId}`);
847
- }
848
-
849
- function estimatePromptTokens(question, selectedRecords) {
850
- let tokens = estimateTextTokens(question);
851
- for (const record of selectedRecords) tokens += record.estimated_tokens;
852
- return tokens;
853
- }
854
-
855
- function percentile(values, ratio) {
856
- if (values.length === 0) return 0;
857
- const sorted = [...values].sort((left, right) => left - right);
858
- const index = Math.min(sorted.length - 1, Math.max(0, Math.ceil(sorted.length * ratio) - 1));
859
- return sorted[index];
860
- }
861
-
862
- function latencySummary(values) {
863
- return {
864
- samples: values.length,
865
- p50_ms: round6(percentile(values, 0.5)),
866
- p95_ms: round6(percentile(values, 0.95)),
867
- min_ms: round6(values.length === 0 ? 0 : Math.min(...values)),
868
- max_ms: round6(values.length === 0 ? 0 : Math.max(...values)),
869
- };
870
- }
871
-
872
- function rate(numerator, denominator) {
873
- if (denominator === 0) return null;
874
- return round6(numerator / denominator);
875
- }
876
-
877
- function round6(value) {
878
- return Number(value.toFixed(6));
879
- }
880
-
881
- function mean(total, count) {
882
- return count === 0 ? 0 : round6(total / count);
883
- }
884
-
885
- function recordFromContent(args) {
886
- const content = String(args.content ?? '');
887
- return {
888
- id: args.id,
889
- dataset_item_id: args.dataset_item_id,
890
- session_id: args.session_id,
891
- turn_id: args.turn_id,
892
- dialog_id: args.dialog_id,
893
- role: args.role,
894
- kind: args.kind,
895
- tags: args.tags ?? [],
896
- has_answer: args.has_answer === true,
897
- ordinal: args.ordinal,
898
- content,
899
- estimated_tokens: estimateTextTokens(content),
900
- };
901
- }
902
-
903
- function parseLocomoEvidenceLabels(evidence) {
904
- const labels = new Set();
905
- const stack = Array.isArray(evidence) ? [...evidence] : [evidence];
906
- while (stack.length > 0) {
907
- const value = stack.shift();
908
- if (Array.isArray(value)) {
909
- stack.push(...value);
910
- continue;
911
- }
912
- if (value === undefined || value === null) continue;
913
- for (const match of String(value).matchAll(/D\s*(\d+)\s*:\s*(\d+)/giu)) {
914
- labels.add(`D${Number(match[1])}:${Number(match[2])}`);
915
- }
916
- }
917
- return labels;
918
- }
919
-
920
- function dialogIdForTurn(sessionNumber, turn, turnIndex) {
921
- const raw = turn?.dia_id ?? turn?.dialog_id ?? turn?.turn_id ?? turn?.id ?? turnIndex + 1;
922
- const text = String(raw);
923
- const match = text.match(/^D\s*(\d+)\s*:\s*(\d+)$/iu);
924
- if (match) return `D${Number(match[1])}:${Number(match[2])}`;
925
- const numeric = text.match(/\d+/u)?.[0] ?? String(turnIndex + 1);
926
- return `D${sessionNumber}:${Number(numeric)}`;
927
- }
928
-
929
- export function parseLocomoDataset(data, options = {}) {
930
- const maxQa = optionalPositiveInteger(options.max_qa ?? options.maxQa, 'max_locomo_qa');
931
- const rows = normalizeDatasetArray(data, 'LoCoMo');
932
- const records = [];
933
- const queries = [];
934
- let ordinal = 0;
935
- for (let sampleIndex = 0; sampleIndex < rows.length; sampleIndex += 1) {
936
- const sample = rows[sampleIndex] ?? {};
937
- const itemId = String(sample.sample_id ?? sample.id ?? `sample_${sampleIndex + 1}`);
938
- const conversation = sample.conversation ?? {};
939
- const sessionNames = Object.keys(conversation)
940
- .map((key) => {
941
- const match = key.match(/^session_(\d+)$/u);
942
- return match ? { key, sessionNumber: Number(match[1]) } : null;
943
- })
944
- .filter(Boolean)
945
- .sort((left, right) => left.sessionNumber - right.sessionNumber);
946
- for (const { key, sessionNumber } of sessionNames) {
947
- const session = conversation[key];
948
- if (!Array.isArray(session)) continue;
949
- for (let turnIndex = 0; turnIndex < session.length; turnIndex += 1) {
950
- const turn = session[turnIndex] ?? {};
951
- const dialogId = dialogIdForTurn(sessionNumber, turn, turnIndex);
952
- records.push(recordFromContent({
953
- id: `locomo:${itemId}:${dialogId}`,
954
- dataset_item_id: itemId,
955
- session_id: `D${sessionNumber}`,
956
- turn_id: dialogId,
957
- dialog_id: dialogId,
958
- role: turn.speaker ?? turn.role ?? undefined,
959
- kind: 'locomo_conversation_turn',
960
- tags: ['locomo', `session_${sessionNumber}`],
961
- ordinal,
962
- content: turn.text ?? turn.content ?? turn.message ?? '',
963
- }));
964
- ordinal += 1;
965
- }
966
- }
967
- const qaRows = Array.isArray(sample.qa) ? sample.qa : [];
968
- for (let qaIndex = 0; qaIndex < qaRows.length; qaIndex += 1) {
969
- if (maxQa !== undefined && queries.length >= maxQa) break;
970
- const qa = qaRows[qaIndex] ?? {};
971
- const evidenceDialogIds = parseLocomoEvidenceLabels(qa.evidence);
972
- queries.push({
973
- id: `locomo:${itemId}:qa_${qaIndex + 1}`,
974
- dataset_item_id: itemId,
975
- question: String(qa.question ?? ''),
976
- category: qa.category === undefined ? undefined : String(qa.category),
977
- evidence_dialog_ids: evidenceDialogIds,
978
- abstention: evidenceDialogIds.size === 0,
979
- });
980
- }
981
- if (maxQa !== undefined && queries.length >= maxQa) break;
982
- }
983
- return {
984
- id: 'locomo',
985
- label: 'LoCoMo',
986
- source_url: LOCOMO_SOURCE_URL,
987
- license: 'CC BY-NC 4.0',
988
- parser: 'conversation session turns as memory records; qa evidence labels mapped to dialog ids such as D1:3 and semicolon-separated labels',
989
- task_categories: [...LOCOMO_TASK_CATEGORIES],
990
- records,
991
- queries,
992
- };
993
- }
994
-
995
- function sessionIdAt(ids, index) {
996
- if (Array.isArray(ids) && ids[index] !== undefined && ids[index] !== null) return String(ids[index]);
997
- return String(index);
998
- }
999
-
1000
- function answerSessionIds(item) {
1001
- if (!Array.isArray(item.answer_session_ids)) return new Set();
1002
- return new Set(item.answer_session_ids.map((id) => String(id)));
1003
- }
1004
-
1005
- function isLongMemEvalAbstention(item, answerSessions) {
1006
- const id = String(item.question_id ?? item.id ?? '');
1007
- if (id.endsWith('_abs') || id.includes('_abs_')) return true;
1008
- if (String(item.question_type ?? '').toLowerCase().includes('abst')) return true;
1009
- return answerSessions.size === 0;
1010
- }
1011
-
1012
- export function parseLongMemEvalDataset(data, options = {}) {
1013
- const maxItems = optionalPositiveInteger(options.max_items ?? options.maxItems, 'max_longmemeval_items');
1014
- const rows = normalizeDatasetArray(data, 'LongMemEval');
1015
- const records = [];
1016
- const queries = [];
1017
- let ordinal = 0;
1018
- const limit = maxItems === undefined ? rows.length : Math.min(rows.length, maxItems);
1019
- for (let itemIndex = 0; itemIndex < limit; itemIndex += 1) {
1020
- const item = rows[itemIndex] ?? {};
1021
- const itemId = String(item.question_id ?? item.id ?? `item_${itemIndex + 1}`);
1022
- const sessions = Array.isArray(item.haystack_sessions) ? item.haystack_sessions : [];
1023
- const sessionIds = item.haystack_session_ids;
1024
- const evidenceTurnIds = new Set();
1025
- for (let sessionIndex = 0; sessionIndex < sessions.length; sessionIndex += 1) {
1026
- const session = sessions[sessionIndex];
1027
- if (!Array.isArray(session)) continue;
1028
- const sessionId = sessionIdAt(sessionIds, sessionIndex);
1029
- for (let turnIndex = 0; turnIndex < session.length; turnIndex += 1) {
1030
- const turn = session[turnIndex] ?? {};
1031
- const turnId = `${sessionId}:${turnIndex}`;
1032
- if (turn.has_answer === true) evidenceTurnIds.add(turnId);
1033
- records.push(recordFromContent({
1034
- id: `longmemeval:${itemId}:${turnId}`,
1035
- dataset_item_id: itemId,
1036
- session_id: sessionId,
1037
- turn_id: turnId,
1038
- role: turn.role ?? undefined,
1039
- kind: 'longmemeval_haystack_turn',
1040
- tags: ['longmemeval', String(item.question_type ?? ''), `session_${sessionId}`],
1041
- has_answer: turn.has_answer === true,
1042
- ordinal,
1043
- content: turn.content ?? turn.text ?? turn.message ?? '',
1044
- }));
1045
- ordinal += 1;
1046
- }
1047
- }
1048
- const answerSessions = answerSessionIds(item);
1049
- queries.push({
1050
- id: `longmemeval:${itemId}`,
1051
- dataset_item_id: itemId,
1052
- question: String(item.question ?? ''),
1053
- question_type: item.question_type === undefined ? undefined : String(item.question_type),
1054
- evidence_turn_ids: evidenceTurnIds,
1055
- evidence_session_ids: answerSessions,
1056
- abstention: isLongMemEvalAbstention(item, answerSessions),
1057
- });
1058
- }
1059
- return {
1060
- id: 'longmemeval',
1061
- label: 'LongMemEval',
1062
- source_url: LONGMEMEVAL_SOURCE_URLS,
1063
- license: 'See Hugging Face dataset card and upstream LongMemEval repository for the selected cleaned file.',
1064
- parser: 'haystack_sessions turns as memory records; has_answer:true turns and answer_session_ids are used as evidence labels; _abs ids are evaluated as abstention cases',
1065
- task_categories: [...LONGMEMEVAL_TASK_CATEGORIES],
1066
- records,
1067
- queries,
1068
- };
1069
- }
1070
-
1071
- function scoreLocomoMethod(method, dataset, topK) {
1072
- const latencies = [];
1073
- let evidenceQuestions = 0;
1074
- let hits = 0;
1075
- let exactCoverage = 0;
1076
- let totalTokens = 0;
1077
- let selectedTotal = 0;
1078
- for (const query of dataset.queries) {
1079
- const start = performance.now();
1080
- const records = dataset.records.filter((record) => record.dataset_item_id === query.dataset_item_id);
1081
- const selected = selectRecords(method.id, records, query, topK);
1082
- latencies.push(performance.now() - start);
1083
- const selectedDialogs = new Set(selected.map((record) => record.dialog_id));
1084
- selectedTotal += selected.length;
1085
- totalTokens += estimatePromptTokens(query.question, selected);
1086
- if (query.evidence_dialog_ids.size > 0) {
1087
- evidenceQuestions += 1;
1088
- let covered = 0;
1089
- for (const evidenceId of query.evidence_dialog_ids) if (selectedDialogs.has(evidenceId)) covered += 1;
1090
- if (covered > 0) hits += 1;
1091
- if (covered === query.evidence_dialog_ids.size) exactCoverage += 1;
1092
- }
1093
- }
1094
- return {
1095
- id: method.id,
1096
- method: method.id,
1097
- local_method_only: true,
1098
- external_provider_called: false,
1099
- retrieval_proxy_only: true,
1100
- uses_top_k: method.uses_top_k,
1101
- top_k: method.uses_top_k ? topK : null,
1102
- question_count: dataset.queries.length,
1103
- evidence_question_count: evidenceQuestions,
1104
- evidence_hit_at_k: rate(hits, evidenceQuestions),
1105
- exact_evidence_coverage: rate(exactCoverage, evidenceQuestions),
1106
- estimated_prompt_tokens: {
1107
- total: totalTokens,
1108
- mean_per_question: mean(totalTokens, dataset.queries.length),
1109
- estimator: 'estimateTextTokens deterministic local estimator',
1110
- },
1111
- selected_memory_count: {
1112
- total: selectedTotal,
1113
- mean_per_question: mean(selectedTotal, dataset.queries.length),
1114
- },
1115
- latency: latencySummary(latencies),
1116
- public_question_text_included: false,
1117
- public_answer_text_included: false,
1118
- raw_conversation_text_included: false,
1119
- };
1120
- }
1121
-
1122
- function scoreLongMemEvalMethod(method, dataset, topK) {
1123
- const latencies = [];
1124
- let turnEvidenceQuestions = 0;
1125
- let turnHits = 0;
1126
- let exactTurnCoverage = 0;
1127
- let sessionEvidenceQuestions = 0;
1128
- let sessionHits = 0;
1129
- let exactSessionCoverage = 0;
1130
- let abstentionQuestions = 0;
1131
- let abstentionCorrect = 0;
1132
- let totalTokens = 0;
1133
- let selectedTotal = 0;
1134
- for (const query of dataset.queries) {
1135
- const start = performance.now();
1136
- const records = dataset.records.filter((record) => record.dataset_item_id === query.dataset_item_id);
1137
- const selected = selectRecords(method.id, records, query, topK);
1138
- latencies.push(performance.now() - start);
1139
- const selectedTurns = new Set(selected.map((record) => record.turn_id));
1140
- const selectedSessions = new Set(selected.map((record) => record.session_id));
1141
- selectedTotal += selected.length;
1142
- totalTokens += estimatePromptTokens(query.question, selected);
1143
-
1144
- if (query.abstention) {
1145
- abstentionQuestions += 1;
1146
- let selectedGold = false;
1147
- for (const turnId of query.evidence_turn_ids) if (selectedTurns.has(turnId)) selectedGold = true;
1148
- for (const sessionId of query.evidence_session_ids) if (selectedSessions.has(sessionId)) selectedGold = true;
1149
- if (!selectedGold) abstentionCorrect += 1;
1150
- continue;
1151
- }
1152
-
1153
- if (query.evidence_turn_ids.size > 0) {
1154
- turnEvidenceQuestions += 1;
1155
- let covered = 0;
1156
- for (const turnId of query.evidence_turn_ids) if (selectedTurns.has(turnId)) covered += 1;
1157
- if (covered > 0) turnHits += 1;
1158
- if (covered === query.evidence_turn_ids.size) exactTurnCoverage += 1;
1159
- }
1160
- if (query.evidence_session_ids.size > 0) {
1161
- sessionEvidenceQuestions += 1;
1162
- let covered = 0;
1163
- for (const sessionId of query.evidence_session_ids) if (selectedSessions.has(sessionId)) covered += 1;
1164
- if (covered > 0) sessionHits += 1;
1165
- if (covered === query.evidence_session_ids.size) exactSessionCoverage += 1;
1166
- }
1167
- }
1168
- return {
1169
- id: method.id,
1170
- method: method.id,
1171
- local_method_only: true,
1172
- external_provider_called: false,
1173
- retrieval_proxy_only: true,
1174
- uses_top_k: method.uses_top_k,
1175
- top_k: method.uses_top_k ? topK : null,
1176
- item_count: dataset.queries.length,
1177
- turn_evidence_question_count: turnEvidenceQuestions,
1178
- turn_evidence_hit_at_k: rate(turnHits, turnEvidenceQuestions),
1179
- exact_turn_evidence_coverage: rate(exactTurnCoverage, turnEvidenceQuestions),
1180
- session_evidence_question_count: sessionEvidenceQuestions,
1181
- session_evidence_hit_at_k: rate(sessionHits, sessionEvidenceQuestions),
1182
- exact_session_evidence_coverage: rate(exactSessionCoverage, sessionEvidenceQuestions),
1183
- abstention_questions: abstentionQuestions,
1184
- abstention_correct: abstentionCorrect,
1185
- abstention_correctness: rate(abstentionCorrect, abstentionQuestions),
1186
- estimated_prompt_tokens: {
1187
- total: totalTokens,
1188
- mean_per_item: mean(totalTokens, dataset.queries.length),
1189
- estimator: 'estimateTextTokens deterministic local estimator',
1190
- },
1191
- selected_memory_count: {
1192
- total: selectedTotal,
1193
- mean_per_item: mean(selectedTotal, dataset.queries.length),
1194
- },
1195
- latency: latencySummary(latencies),
1196
- public_question_text_included: false,
1197
- public_answer_text_included: false,
1198
- raw_conversation_text_included: false,
1199
- };
1200
- }
1201
-
1202
- function scoreDataset(dataset, topK) {
1203
- const methodRows = STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => (
1204
- dataset.id === 'locomo' ? scoreLocomoMethod(method, dataset, topK) : scoreLongMemEvalMethod(method, dataset, topK)
1205
- ));
1206
- return {
1207
- id: dataset.id,
1208
- dataset: dataset.id,
1209
- label: dataset.label,
1210
- source_url: dataset.source_url,
1211
- license: dataset.license,
1212
- parser: dataset.parser,
1213
- task_categories: dataset.task_categories,
1214
- record_count: dataset.records.length,
1215
- question_count: dataset.queries.length,
1216
- item_count: dataset.queries.length,
1217
- raw_question_text_included: false,
1218
- raw_answer_text_included: false,
1219
- raw_conversation_text_included: false,
1220
- methods: methodRows,
1221
- };
1222
- }
1223
-
1224
- export function runStandardMemoryBenchmarkSuite(options = {}) {
1225
- const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
1226
- const datasetRows = [];
1227
- if (options.locomoData !== undefined) {
1228
- datasetRows.push(scoreDataset(parseLocomoDataset(options.locomoData, { max_qa: options.max_locomo_qa ?? options.maxLocomoQa }), topK));
1229
- }
1230
- if (options.longMemEvalData !== undefined || options.longmemevalData !== undefined) {
1231
- datasetRows.push(scoreDataset(parseLongMemEvalDataset(options.longMemEvalData ?? options.longmemevalData, { max_items: options.max_longmemeval_items ?? options.maxLongMemEvalItems }), topK));
1232
- }
1233
- if (datasetRows.length === 0) throw new Error('At least one standard dataset is required: provide locomoData and/or longMemEvalData');
1234
- return buildSuiteReport(datasetRows, topK, options);
1235
- }
1236
-
1237
- export async function runStandardMemoryBenchmarkSuiteFromFiles(options = {}) {
1238
- const datasetRows = [];
1239
- const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
1240
- if (options.locomo !== undefined || options.locomoPath !== undefined) {
1241
- const path = options.locomo ?? options.locomoPath;
1242
- const loaded = await readJsonWithSha256(path);
1243
- datasetRows.push({
1244
- ...scoreDataset(parseLocomoDataset(loaded.data, { max_qa: options.max_locomo_qa ?? options.maxLocomoQa }), topK),
1245
- local_file_name: publicFileName(path),
1246
- input_sha256: loaded.sha256,
1247
- });
1248
- }
1249
- if (options.longmemeval !== undefined || options.longmemevalPath !== undefined || options.longMemEvalPath !== undefined) {
1250
- const path = options.longmemeval ?? options.longmemevalPath ?? options.longMemEvalPath;
1251
- const maxItems = optionalPositiveInteger(options.max_longmemeval_items ?? options.maxLongMemEvalItems, 'max_longmemeval_items');
1252
- const loaded = await readLongMemEvalJsonWithSha256(path, maxItems);
1253
- datasetRows.push({
1254
- ...scoreDataset(parseLongMemEvalDataset(loaded.data, { max_items: maxItems }), topK),
1255
- local_file_name: publicFileName(path),
1256
- input_sha256: loaded.sha256,
1257
- });
1258
- }
1259
- if (datasetRows.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
1260
- return buildSuiteReport(datasetRows, topK, options);
1261
- }
1262
-
1263
- function buildSuiteReport(datasetRows, topK, options) {
1264
- return {
1265
- schema: STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA,
1266
- generated_at: options.generated_at ?? new Date().toISOString(),
1267
- package: {
1268
- name: 'enigma-memory',
1269
- version: '0.1.17',
1270
- },
1271
- public_safe: true,
1272
- top_k: topK,
1273
- source_urls: {
1274
- locomo: LOCOMO_SOURCE_URL,
1275
- longmemeval: LONGMEMEVAL_SOURCE_URLS,
1276
- },
1277
- license_and_boundary_notes: [
1278
- 'LoCoMo source data is CC BY-NC 4.0; keep local dataset files and raw conversations out of public reports unless separately reviewed.',
1279
- 'LongMemEval cleaned files are operator-supplied local JSON files from the upstream Hugging Face dataset repository.',
1280
- 'Scores are retrieval/evidence proxy metrics over official dataset labels, not LLM-generated answer accuracy.',
1281
- 'No provider APIs, hosted runtimes, competitor SDKs, or external accounts are called by this runner.',
1282
- 'Rows are local deterministic methods only; no third-party competitor scores or benchmark-leadership claims are emitted.',
1283
- ],
1284
- command_boundaries: offlineCommandBoundaries({ scoresIncluded: true, datasetFilesRead: true }),
1285
- apples_to_apples_controls: applesToApplesControls(topK),
1286
- benchmark_boundaries: {
1287
- official_dataset_files_required: true,
1288
- credentials_required: false,
1289
- external_provider_calls: false,
1290
- llm_answer_accuracy_scored: false,
1291
- retrieval_evidence_proxy_scored: true,
1292
- raw_question_text_included: false,
1293
- raw_answer_text_included: false,
1294
- raw_conversation_text_included: false,
1295
- provider_deletion_claim: false,
1296
- model_forgetting_claim: false,
1297
- roi_or_provider_invoice_savings_claim: false,
1298
- compliance_certification_claim: false,
1299
- benchmark_leadership_claim: false,
1300
- },
1301
- relevance_logic: {
1302
- token_extraction: 'keyword_filter uses basic lowercase /[a-z0-9]+(?:[-_][a-z0-9]+)*/ overlap after stopword removal; enigma_relevance additionally applies deterministic suffix stemming for ing, ed, and plural s forms',
1303
- production_alignment: 'public-safe local approximation of production query-aware retrieval using query/content stems, role/session/kind/tag hints, category/task hints, temporal/date hints, and phrase/proximity boosts; no private memory is emitted',
1304
- keyword_filter_fallback: 'empty result when no query/content token overlap exists',
1305
- enigma_relevance_fallback: 'falls back to all local candidates only when no enhanced relevance signal exists, then applies deterministic local ranking and --top-k',
1306
- provider_api_used: false,
1307
- llm_used: false,
1308
- gold_labels_used_for_retrieval: false,
1309
- },
1310
- local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
1311
- external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
1312
- ...adapter,
1313
- required_artifacts: [...adapter.required_artifacts],
1314
- })),
1315
- datasets: datasetRows,
1316
- dataset_rows: datasetRows,
1317
- };
1318
- }
1319
-
1320
- function usage() {
1321
- return `Usage: node scripts/run-standard-memory-benchmarks.mjs [--locomo <path>] [--longmemeval <path>] [--max-locomo-qa <n>] [--max-longmemeval-items <n>] [--top-k <n>] [--out <path>] [--dry-run] [--protocol-plan [--answerer-ref <ref>] [--judge-ref <ref>] [--answer-prompt-ref <ref>] [--judge-prompt-ref <ref>] [--protocol-ref <ref>]]\n\nProduces schema ${STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA}. Raw question, answer, and conversation text are never written to the report. With --longmemeval and --max-longmemeval-items, the local top-level JSON array is streamed for hashing and only the requested sample items are parsed. Use --dry-run to print a public-safe offline execution plan without reading dataset files or producing scores. Use --protocol-plan to print a public-safe full-answer benchmark PROTOCOL plan (schema ${STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA}) recording the planned category set, top-k, answerer/judge model refs, prompt/protocol refs, competitor adapter refs, and cost-estimate inputs, with explicit network_required:false, provider_calls_made:false, answers_generated:false, and judged:false boundaries; it does not call providers, generate answers, judge answers, or run competitor adapters.`;
1322
- }
1323
-
1324
- async function main() {
1325
- const options = parseArgs();
1326
- if (options.help) {
1327
- console.log(usage());
1328
- return;
1329
- }
1330
- const report = options.protocol_plan
1331
- ? buildStandardBenchmarkProtocolPlan(options)
1332
- : options.dry_run
1333
- ? buildStandardBenchmarkDryRunPlan(options)
1334
- : await runStandardMemoryBenchmarkSuiteFromFiles(options);
1335
- const serialized = `${JSON.stringify(report, null, 2)}\n`;
1336
- if (options.out) {
1337
- const outPath = resolve(options.out);
1338
- await mkdir(dirname(outPath), { recursive: true });
1339
- await writeFile(outPath, serialized);
1340
- } else {
1341
- process.stdout.write(serialized);
1342
- }
1343
- }
1344
-
1345
- const invokedPath = process.argv[1] ? resolve(process.argv[1]) : '';
1346
- const modulePath = fileURLToPath(import.meta.url);
1347
- if (invokedPath === modulePath) {
1348
- main().catch((error) => {
1349
- console.error(error instanceof Error ? error.message : String(error));
1350
- process.exitCode = 1;
1351
- });
1352
- }
1
+ #!/usr/bin/env node
2
+ import { createReadStream } from 'node:fs';
3
+ import { mkdir, readFile, writeFile } from 'node:fs/promises';
4
+ import { basename, dirname, resolve } from 'node:path';
5
+ import { fileURLToPath } from 'node:url';
6
+ import { createHash } from 'node:crypto';
7
+ import { performance } from 'node:perf_hooks';
8
+ import { StringDecoder } from 'node:string_decoder';
9
+ import { estimateTextTokens } from '../packages/optimizer/src/index.js';
10
+
11
+ export const STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA = 'enigma.standard_memory_benchmark_suite.v1';
12
+
13
+ export const STANDARD_MEMORY_BENCHMARK_METHODS = Object.freeze([
14
+ Object.freeze({
15
+ id: 'full_context',
16
+ label: 'Full context',
17
+ boundary: 'Supplies every parsed memory record for each query; no provider API or model answer generation.',
18
+ uses_top_k: false,
19
+ }),
20
+ Object.freeze({
21
+ id: 'recency_last_n',
22
+ label: 'Recency last N',
23
+ boundary: 'Supplies the most recent local memory records up to --top-k.',
24
+ uses_top_k: true,
25
+ }),
26
+ Object.freeze({
27
+ id: 'keyword_filter',
28
+ label: 'Keyword filter',
29
+ boundary: 'Supplies local memory records whose public-safe deterministic tokens overlap the query, capped by --top-k.',
30
+ uses_top_k: true,
31
+ }),
32
+ Object.freeze({
33
+ id: 'enigma_relevance',
34
+ label: 'Enigma relevance',
35
+ boundary: 'Uses deterministic query-aware relevance features over local public-safe memory metadata and content tokens, then ranks locally without provider APIs.',
36
+ uses_top_k: true,
37
+ }),
38
+ ]);
39
+
40
+ export const STANDARD_EXTERNAL_COMPETITOR_ADAPTERS = Object.freeze([
41
+ Object.freeze({
42
+ id: 'mem0',
43
+ name: 'Mem0',
44
+ status: 'not_run_requires_credentials_or_runtime',
45
+ target_type: 'external_adapter',
46
+ can_run_in_this_harness: false,
47
+ scores_included: false,
48
+ required_artifacts: Object.freeze([
49
+ 'Mem0 platform credentials or open-source runtime',
50
+ 'Pinned Mem0 SDK/package versions',
51
+ 'Fixed extraction, update, retrieval, reset, model, and tool policy',
52
+ 'Same reviewed dataset manifest, split, top-k, and scorer as Enigma rows',
53
+ ]),
54
+ official_doc: 'https://docs.mem0.ai/',
55
+ boundary_reason: 'The standard runner has no Mem0 credentials, SDK/runtime, fixed memory loop, reset policy, model/tool environment, or reviewed adapter scorer, so no Mem0 score is produced.',
56
+ }),
57
+ ]);
58
+
59
+ const LOCOMO_SOURCE_URL = 'https://raw.githubusercontent.com/snap-research/locomo/main/data/locomo10.json';
60
+ const LONGMEMEVAL_SOURCE_URLS = Object.freeze([
61
+ 'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_oracle.json',
62
+ 'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_s_cleaned.json',
63
+ 'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_m_cleaned.json',
64
+ ]);
65
+ const LOCOMO_TASK_CATEGORIES = Object.freeze(['multi-session QA', 'event summarization', 'multimodal generation over long conversations']);
66
+ const LONGMEMEVAL_TASK_CATEGORIES = Object.freeze(['information extraction', 'multi-session reasoning', 'temporal reasoning', 'knowledge updates', 'abstention']);
67
+ const DATASET_TASK_CATEGORIES = Object.freeze({ locomo: LOCOMO_TASK_CATEGORIES, longmemeval: LONGMEMEVAL_TASK_CATEGORIES });
68
+ export const STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA = 'enigma.standard_memory_benchmark_protocol_plan.v1';
69
+ const PROTOCOL_REF_RE = /^[a-z0-9][a-z0-9._:/@+-]{2,191}$/u;
70
+ const DEFAULT_ANSWERER_MODEL_REF = 'model:answerer-not-selected';
71
+ const DEFAULT_JUDGE_MODEL_REF = 'model:judge-not-selected';
72
+ const DEFAULT_ANSWER_PROMPT_REF = 'prompt:standard-answer@not-pinned';
73
+ const DEFAULT_JUDGE_PROMPT_REF = 'prompt:standard-judge@not-pinned';
74
+ const DEFAULT_PROTOCOL_REF = 'protocol:apples-to-apples-full-answer@not-pinned';
75
+
76
+ const QUERY_RELEVANCE_STOPWORDS = new Set([
77
+ 'about',
78
+ 'after',
79
+ 'again',
80
+ 'against',
81
+ 'also',
82
+ 'and',
83
+ 'any',
84
+ 'are',
85
+ 'assistant',
86
+ 'because',
87
+ 'been',
88
+ 'before',
89
+ 'being',
90
+ 'between',
91
+ 'can',
92
+ 'could',
93
+ 'current',
94
+ 'does',
95
+ 'from',
96
+ 'has',
97
+ 'have',
98
+ 'how',
99
+ 'into',
100
+ 'its',
101
+ 'latest',
102
+ 'more',
103
+ 'most',
104
+ 'number',
105
+ 'own',
106
+ 'owns',
107
+ 'please',
108
+ 'should',
109
+ 'that',
110
+ 'the',
111
+ 'their',
112
+ 'then',
113
+ 'there',
114
+ 'these',
115
+ 'they',
116
+ 'this',
117
+ 'use',
118
+ 'using',
119
+ 'was',
120
+ 'what',
121
+ 'when',
122
+ 'where',
123
+ 'which',
124
+ 'who',
125
+ 'whose',
126
+ 'why',
127
+ 'with',
128
+ 'would',
129
+ ]);
130
+
131
+ function parseArgs(argv = process.argv.slice(2)) {
132
+ const options = { top_k: 5 };
133
+ for (let index = 0; index < argv.length; index += 1) {
134
+ const arg = argv[index];
135
+ if (arg === '--locomo') {
136
+ options.locomo = requiredFlagValue(argv, index, arg);
137
+ index += 1;
138
+ } else if (arg === '--longmemeval') {
139
+ options.longmemeval = requiredFlagValue(argv, index, arg);
140
+ index += 1;
141
+ } else if (arg === '--max-locomo-qa') {
142
+ options.max_locomo_qa = positiveInteger(requiredFlagValue(argv, index, arg), arg);
143
+ index += 1;
144
+ } else if (arg === '--max-longmemeval-items') {
145
+ options.max_longmemeval_items = positiveInteger(requiredFlagValue(argv, index, arg), arg);
146
+ index += 1;
147
+ } else if (arg === '--top-k') {
148
+ options.top_k = positiveInteger(requiredFlagValue(argv, index, arg), arg);
149
+ index += 1;
150
+ } else if (arg === '--out') {
151
+ options.out = requiredFlagValue(argv, index, arg);
152
+ index += 1;
153
+ } else if (arg === '--protocol-plan') {
154
+ options.protocol_plan = true;
155
+ } else if (arg === '--answerer-ref') {
156
+ options.answerer_ref = requiredFlagValue(argv, index, arg);
157
+ index += 1;
158
+ } else if (arg === '--judge-ref') {
159
+ options.judge_ref = requiredFlagValue(argv, index, arg);
160
+ index += 1;
161
+ } else if (arg === '--answer-prompt-ref') {
162
+ options.answer_prompt_ref = requiredFlagValue(argv, index, arg);
163
+ index += 1;
164
+ } else if (arg === '--judge-prompt-ref') {
165
+ options.judge_prompt_ref = requiredFlagValue(argv, index, arg);
166
+ index += 1;
167
+ } else if (arg === '--protocol-ref') {
168
+ options.protocol_ref = requiredFlagValue(argv, index, arg);
169
+ index += 1;
170
+ } else if (arg === '--dry-run') {
171
+ options.dry_run = true;
172
+ } else if (arg === '--help' || arg === '-h') {
173
+ options.help = true;
174
+ } else {
175
+ throw new Error(`Unknown option ${arg}`);
176
+ }
177
+ }
178
+ return options;
179
+ }
180
+
181
+ function requiredFlagValue(argv, index, flag) {
182
+ const value = argv[index + 1];
183
+ if (value === undefined || value.startsWith('--')) throw new Error(`${flag} requires a value`);
184
+ return value;
185
+ }
186
+
187
+ function positiveInteger(value, name) {
188
+ const number = Number(value);
189
+ if (!Number.isInteger(number) || number <= 0) throw new Error(`${name} must be a positive integer`);
190
+ return number;
191
+ }
192
+
193
+ function optionalPositiveInteger(value, name) {
194
+ if (value === undefined || value === null) return undefined;
195
+ return positiveInteger(value, name);
196
+ }
197
+
198
+ function datasetPlanRows(options) {
199
+ const rows = [];
200
+ if (options.locomo !== undefined || options.locomoPath !== undefined) {
201
+ rows.push({
202
+ id: 'locomo',
203
+ label: 'LoCoMo',
204
+ local_file_name: publicFileName(options.locomo ?? options.locomoPath),
205
+ source_url: LOCOMO_SOURCE_URL,
206
+ license: 'CC BY-NC 4.0',
207
+ sample_limit: optionalPositiveInteger(options.max_locomo_qa ?? options.maxLocomoQa, 'max_locomo_qa') ?? null,
208
+ parser: 'conversation session turns as memory records; qa evidence labels score support only',
209
+ });
210
+ }
211
+ if (options.longmemeval !== undefined || options.longmemevalPath !== undefined || options.longMemEvalPath !== undefined) {
212
+ rows.push({
213
+ id: 'longmemeval',
214
+ label: 'LongMemEval',
215
+ local_file_name: publicFileName(options.longmemeval ?? options.longmemevalPath ?? options.longMemEvalPath),
216
+ source_url: LONGMEMEVAL_SOURCE_URLS,
217
+ license: 'Review upstream Hugging Face dataset card and LongMemEval repository terms.',
218
+ sample_limit: optionalPositiveInteger(options.max_longmemeval_items ?? options.maxLongMemEvalItems, 'max_longmemeval_items') ?? null,
219
+ parser: 'haystack_sessions turns as memory records; answer-session labels score support only',
220
+ });
221
+ }
222
+ return rows;
223
+ }
224
+
225
+ function offlineCommandBoundaries({ scoresIncluded, datasetFilesRead }) {
226
+ return {
227
+ deterministic_offline_runner: true,
228
+ dataset_files_read_from_local_disk: datasetFilesRead,
229
+ network_calls_made: false,
230
+ provider_api_calls_made: false,
231
+ api_spend_possible: false,
232
+ hosted_memory_service_called: false,
233
+ external_competitor_adapters_run: false,
234
+ mem0_adapter_run: false,
235
+ llm_used: false,
236
+ llm_answer_accuracy_scored: false,
237
+ retrieval_evidence_proxy_scored: scoresIncluded,
238
+ benchmark_scores_included: scoresIncluded,
239
+ raw_question_text_included: false,
240
+ raw_answer_text_included: false,
241
+ raw_conversation_text_included: false,
242
+ gold_labels_used_for_retrieval: false,
243
+ gold_labels_used_for_scoring: scoresIncluded,
244
+ };
245
+ }
246
+
247
+ function applesToApplesControls(topK) {
248
+ return {
249
+ same_top_k_for_all_methods: true,
250
+ top_k: topK,
251
+ same_parser_per_dataset: true,
252
+ same_local_records_per_dataset: true,
253
+ same_gold_evidence_labels_per_dataset_for_scoring_only: true,
254
+ local_deterministic_methods_only: true,
255
+ provider_runtime_fixed: false,
256
+ competitor_runtime_fixed: false,
257
+ answer_generator_fixed: false,
258
+ evaluator_model_fixed: false,
259
+ };
260
+ }
261
+
262
+ export function buildStandardBenchmarkDryRunPlan(options = {}) {
263
+ const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
264
+ const datasets = datasetPlanRows(options);
265
+ if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
266
+ return {
267
+ schema: 'enigma.standard_memory_benchmark_plan.v1',
268
+ generated_at: options.generated_at ?? new Date().toISOString(),
269
+ package: {
270
+ name: 'enigma-memory',
271
+ version: '0.1.17',
272
+ },
273
+ public_safe: true,
274
+ dry_run: true,
275
+ top_k: topK,
276
+ datasets_planned: datasets,
277
+ local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
278
+ external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
279
+ ...adapter,
280
+ required_artifacts: [...adapter.required_artifacts],
281
+ })),
282
+ command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
283
+ apples_to_apples_controls: applesToApplesControls(topK),
284
+ non_claims: [
285
+ 'This dry run does not read dataset files and produces no benchmark score.',
286
+ 'No provider APIs, hosted memory services, Mem0 runtime, competitor SDKs, LLM generators, or evaluator models are called.',
287
+ 'A scored report requires a separate non-dry-run command against the exact local dataset files and hashes.',
288
+ ],
289
+ };
290
+ }
291
+ function protocolRef(value, label, fallback) {
292
+ if (value === undefined || value === null) return fallback;
293
+ const normalized = typeof value === 'string' ? value.trim() : String(value);
294
+ if (normalized === '') throw new Error(`${label} must be a non-empty public ref`);
295
+ if (!PROTOCOL_REF_RE.test(normalized)) throw new Error(`${label} must be a lowercase public ref using letters, numbers, . _ : / @ + or -`);
296
+ return normalized;
297
+ }
298
+
299
+ function protocolPlanCategorySet(datasets) {
300
+ const categorySet = [];
301
+ const seen = new Set();
302
+ for (const row of datasets) {
303
+ for (const category of DATASET_TASK_CATEGORIES[row.id] ?? []) {
304
+ if (!seen.has(category)) {
305
+ seen.add(category);
306
+ categorySet.push(category);
307
+ }
308
+ }
309
+ }
310
+ return categorySet;
311
+ }
312
+
313
+ export function buildStandardBenchmarkProtocolPlan(options = {}) {
314
+ const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
315
+ const datasets = datasetPlanRows(options);
316
+ if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
317
+ const answererProvided = (options.answerer_ref ?? options.answererModelRef) !== undefined;
318
+ const judgeProvided = (options.judge_ref ?? options.judgeModelRef) !== undefined;
319
+ const answererRef = protocolRef(options.answerer_ref ?? options.answererModelRef, 'answerer_ref', DEFAULT_ANSWERER_MODEL_REF);
320
+ const judgeRef = protocolRef(options.judge_ref ?? options.judgeModelRef, 'judge_ref', DEFAULT_JUDGE_MODEL_REF);
321
+ const answerPromptRef = protocolRef(options.answer_prompt_ref ?? options.answerPromptRef, 'answer_prompt_ref', DEFAULT_ANSWER_PROMPT_REF);
322
+ const judgePromptRef = protocolRef(options.judge_prompt_ref ?? options.judgePromptRef, 'judge_prompt_ref', DEFAULT_JUDGE_PROMPT_REF);
323
+ const protocolRefValue = protocolRef(options.protocol_ref ?? options.protocolRef, 'protocol_ref', DEFAULT_PROTOCOL_REF);
324
+ const promptsFixed = answererProvided && judgeProvided;
325
+ return {
326
+ schema: STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA,
327
+ generated_at: options.generated_at ?? new Date().toISOString(),
328
+ package: {
329
+ name: 'enigma-memory',
330
+ version: '0.1.17',
331
+ },
332
+ public_safe: true,
333
+ protocol_plan: true,
334
+ dry_run: true,
335
+ top_k: topK,
336
+ category_set: protocolPlanCategorySet(datasets),
337
+ datasets_planned: datasets,
338
+ answerer: {
339
+ model_ref: answererRef,
340
+ temperature: 0,
341
+ max_tokens: 1024,
342
+ fixed: answererProvided,
343
+ },
344
+ judge: {
345
+ model_ref: judgeRef,
346
+ kind: 'llm-as-judge-or-exact-match-not-selected',
347
+ temperature: 0,
348
+ max_tokens: 512,
349
+ fixed: judgeProvided,
350
+ },
351
+ prompt_refs: [answerPromptRef, judgePromptRef],
352
+ protocol_refs: [protocolRefValue],
353
+ competitor_adapter_refs: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => `adapter:${adapter.id}@not-pinned`),
354
+ external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
355
+ ...adapter,
356
+ required_artifacts: [...adapter.required_artifacts],
357
+ adapter_ref: `adapter:${adapter.id}@not-pinned`,
358
+ })),
359
+ apples_to_apples_controls: applesToApplesControls(topK),
360
+ protocol_controls: {
361
+ same_answerer_model_for_all_rows: answererProvided,
362
+ same_judge_model_for_all_rows: judgeProvided,
363
+ same_prompts_for_all_rows: promptsFixed,
364
+ same_competitor_adapters_for_enigma_and_baselines: false,
365
+ answerer_model_fixed: answererProvided,
366
+ judge_model_fixed: judgeProvided,
367
+ prompts_fixed: promptsFixed,
368
+ temperature_fixed: false,
369
+ budget_caps_set: false,
370
+ },
371
+ cost_estimate_inputs: {
372
+ dataset_sample_limits: datasets.map((row) => ({ dataset: row.id, sample_limit: row.sample_limit })),
373
+ answerer_temperature: 0,
374
+ answerer_max_tokens: 1024,
375
+ judge_temperature: 0,
376
+ judge_max_tokens: 512,
377
+ max_retries: 0,
378
+ request_timeout_ms: null,
379
+ budget_cap_required_before_run: true,
380
+ budget_cap_set: false,
381
+ },
382
+ benchmark_boundaries: {
383
+ official_dataset_files_required: true,
384
+ credentials_required: false,
385
+ external_provider_calls: false,
386
+ llm_answer_accuracy_scored: false,
387
+ retrieval_evidence_proxy_scored: false,
388
+ raw_question_text_included: false,
389
+ raw_answer_text_included: false,
390
+ raw_conversation_text_included: false,
391
+ provider_deletion_claim: false,
392
+ model_forgetting_claim: false,
393
+ roi_or_provider_invoice_savings_claim: false,
394
+ compliance_certification_claim: false,
395
+ benchmark_leadership_claim: false,
396
+ },
397
+ protocol_boundaries: {
398
+ network_required: false,
399
+ provider_calls_made: false,
400
+ answers_generated: false,
401
+ judged: false,
402
+ competitor_adapters_run: false,
403
+ api_spend_possible: false,
404
+ },
405
+ command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
406
+ review_rules: [
407
+ 'report hash format: sha256:<hex> over the public-safe protocol-plan JSON bytes',
408
+ 'dataset ref, runner ref, package ref, environment ref, and verifier ref formats are pinned before a live run',
409
+ 'metric scope: this artifact plans the full-answer protocol only; it is not a score',
410
+ 'limitation: no provider answer-accuracy, competitor-performance, benchmark-leadership, ROI, provider-deletion, model-forgetting, or compliance claim is made',
411
+ ],
412
+ non_claims: [
413
+ 'This protocol plan is a readiness artifact: it records the planned full-answer protocol dimensions and does not execute it.',
414
+ 'No network is used, no provider APIs are called, no answers are generated, no answers are judged, and no competitor adapters are run.',
415
+ 'Answerer/judge model refs, prompt refs, and competitor adapter refs are public-safe references only; they are not credentials, model calls, prompts, or scores.',
416
+ 'A scored full-answer run requires a separate credentialed command with frozen models, prompts, evaluator, dataset manifest, and budget caps.',
417
+ ],
418
+ };
419
+ }
420
+
421
+ function publicFileName(path) {
422
+ return path === undefined || path === null ? undefined : basename(String(path));
423
+ }
424
+
425
+ async function readJsonWithSha256(path) {
426
+ const raw = await readFile(path, 'utf8');
427
+ return {
428
+ data: JSON.parse(raw),
429
+ sha256: createHash('sha256').update(raw).digest('hex'),
430
+ };
431
+ }
432
+
433
+ function isJsonWhitespace(char) {
434
+ const code = char.charCodeAt(0);
435
+ return code === 0x20 || code === 0x0a || code === 0x0d || code === 0x09 || code === 0xfeff;
436
+ }
437
+
438
+ function createTopLevelArraySampler(maxItems, name) {
439
+ const items = [];
440
+ let started = false;
441
+ let closed = false;
442
+ let collecting = false;
443
+ let doneCollecting = false;
444
+ let expectSeparator = false;
445
+ let depth = 0;
446
+ let inString = false;
447
+ let escaped = false;
448
+ let current = '';
449
+
450
+ function fail(message) {
451
+ throw new Error(`${name} sample-mode JSON parse failed: ${message}`);
452
+ }
453
+
454
+ function finishItem() {
455
+ items.push(current);
456
+ current = '';
457
+ collecting = false;
458
+ expectSeparator = true;
459
+ if (items.length >= maxItems) doneCollecting = true;
460
+ }
461
+
462
+ return {
463
+ write(text) {
464
+ for (const char of text) {
465
+ if (!started) {
466
+ if (isJsonWhitespace(char)) continue;
467
+ if (char !== '[') fail('expected a top-level array');
468
+ started = true;
469
+ continue;
470
+ }
471
+
472
+ if (doneCollecting) continue;
473
+
474
+ if (closed) {
475
+ if (!isJsonWhitespace(char)) fail('found trailing data after the top-level array');
476
+ continue;
477
+ }
478
+
479
+ if (!collecting) {
480
+ if (isJsonWhitespace(char)) continue;
481
+ if (expectSeparator) {
482
+ if (char === ',') {
483
+ expectSeparator = false;
484
+ continue;
485
+ }
486
+ if (char === ']') {
487
+ closed = true;
488
+ continue;
489
+ }
490
+ fail('expected a comma or closing bracket between items');
491
+ }
492
+ if (char === ']') {
493
+ closed = true;
494
+ continue;
495
+ }
496
+ if (char !== '{' && char !== '[') fail('expected each sampled item to be an object or array');
497
+ collecting = true;
498
+ current = char;
499
+ depth = 1;
500
+ continue;
501
+ }
502
+
503
+ current += char;
504
+ if (inString) {
505
+ if (escaped) {
506
+ escaped = false;
507
+ } else if (char === '\\') {
508
+ escaped = true;
509
+ } else if (char === '"') {
510
+ inString = false;
511
+ }
512
+ continue;
513
+ }
514
+ if (char === '"') {
515
+ inString = true;
516
+ } else if (char === '{' || char === '[') {
517
+ depth += 1;
518
+ } else if (char === '}' || char === ']') {
519
+ depth -= 1;
520
+ if (depth < 0) fail('encountered an unmatched closing bracket');
521
+ if (depth === 0) finishItem();
522
+ }
523
+ }
524
+ },
525
+ get done() {
526
+ return doneCollecting;
527
+ },
528
+ finish() {
529
+ if (!started) fail('empty input');
530
+ if (!doneCollecting && collecting) fail('ended inside a sampled item');
531
+ if (!doneCollecting && inString) fail('ended inside a string');
532
+ if (!doneCollecting && !closed) fail('ended before the top-level array closed');
533
+ return JSON.parse(`[${items.join(',')}]`);
534
+ },
535
+ };
536
+ }
537
+
538
+ async function readJsonArraySampleWithSha256(path, maxItems, name) {
539
+ const hash = createHash('sha256');
540
+ const decoder = new StringDecoder('utf8');
541
+ const sampler = createTopLevelArraySampler(maxItems, name);
542
+ for await (const chunk of createReadStream(path)) {
543
+ hash.update(chunk);
544
+ if (!sampler.done) sampler.write(decoder.write(chunk));
545
+ }
546
+ if (!sampler.done) {
547
+ const tail = decoder.end();
548
+ if (tail.length > 0) sampler.write(tail);
549
+ }
550
+ return {
551
+ data: sampler.finish(),
552
+ sha256: hash.digest('hex'),
553
+ };
554
+ }
555
+
556
+ async function readLongMemEvalJsonWithSha256(path, maxItems) {
557
+ if (maxItems === undefined) return readJsonWithSha256(path);
558
+ try {
559
+ return await readJsonArraySampleWithSha256(path, maxItems, 'LongMemEval');
560
+ } catch (error) {
561
+ if (error instanceof Error && error.message === 'LongMemEval sample-mode JSON parse failed: expected a top-level array') {
562
+ return readJsonWithSha256(path);
563
+ }
564
+ throw error;
565
+ }
566
+ }
567
+
568
+ function normalizeDatasetArray(data, name) {
569
+ if (Array.isArray(data)) return data;
570
+ if (data && typeof data === 'object') {
571
+ for (const key of ['data', 'items', 'examples', 'samples']) {
572
+ if (Array.isArray(data[key])) return data[key];
573
+ }
574
+ }
575
+ throw new TypeError(`${name} dataset must be a JSON array or object containing an array`);
576
+ }
577
+
578
+ function addMeaningfulToken(tokens, token) {
579
+ if (token.length < 3) return;
580
+ if (!/[a-z]/u.test(token)) return;
581
+ if (QUERY_RELEVANCE_STOPWORDS.has(token)) return;
582
+ tokens.add(token);
583
+ }
584
+
585
+ function stemToken(token) {
586
+ if (token.length > 5 && token.endsWith('ing')) {
587
+ let stem = token.slice(0, -3);
588
+ if (stem.length > 3 && stem.at(-1) === stem.at(-2)) stem = stem.slice(0, -1);
589
+ return stem;
590
+ }
591
+ if (token.length > 4 && token.endsWith('ed')) {
592
+ let stem = token.slice(0, -2);
593
+ if (stem.length > 3 && stem.at(-1) === stem.at(-2)) stem = stem.slice(0, -1);
594
+ return stem;
595
+ }
596
+ if (token.length > 4 && token.endsWith('ies')) return `${token.slice(0, -3)}y`;
597
+ if (token.length > 4 && token.endsWith('es')) return token.slice(0, -2);
598
+ if (token.length > 3 && token.endsWith('s') && !token.endsWith('ss')) return token.slice(0, -1);
599
+ return token;
600
+ }
601
+
602
+ function addStemmedMeaningfulToken(tokens, token) {
603
+ addMeaningfulToken(tokens, token);
604
+ const stem = stemToken(token);
605
+ addMeaningfulToken(tokens, stem);
606
+ if (token.endsWith('ed') || token.endsWith('ing')) addMeaningfulToken(tokens, `${stem}e`);
607
+ }
608
+
609
+ function meaningfulTokensFrom(value) {
610
+ const tokens = new Set();
611
+ if (value === undefined || value === null) return tokens;
612
+ for (const match of String(value).toLowerCase().matchAll(/[a-z0-9]+(?:[-_][a-z0-9]+)*/gu)) {
613
+ const token = match[0];
614
+ addMeaningfulToken(tokens, token);
615
+ if (token.includes('-') || token.includes('_')) {
616
+ for (const part of token.split(/[-_]+/u)) addMeaningfulToken(tokens, part);
617
+ }
618
+ }
619
+ return tokens;
620
+ }
621
+
622
+ function stemmedMeaningfulTokensFrom(value) {
623
+ const tokens = new Set();
624
+ if (value === undefined || value === null) return tokens;
625
+ for (const match of String(value).toLowerCase().matchAll(/[a-z0-9]+(?:[-_][a-z0-9]+)*/gu)) {
626
+ const token = match[0];
627
+ addStemmedMeaningfulToken(tokens, token);
628
+ if (token.includes('-') || token.includes('_')) {
629
+ for (const part of token.split(/[-_]+/u)) addStemmedMeaningfulToken(tokens, part);
630
+ }
631
+ }
632
+ return tokens;
633
+ }
634
+
635
+ function stemmedTokenSequenceFrom(value) {
636
+ const sequence = [];
637
+ if (value === undefined || value === null) return sequence;
638
+ for (const match of String(value).toLowerCase().matchAll(/[a-z0-9]+(?:[-_][a-z0-9]+)*/gu)) {
639
+ const token = match[0];
640
+ const parts = token.includes('-') || token.includes('_') ? token.split(/[-_]+/u) : [token];
641
+ for (const part of parts) {
642
+ const stem = stemToken(part);
643
+ if (stem.length >= 3 && /[a-z]/u.test(stem) && !QUERY_RELEVANCE_STOPWORDS.has(stem)) sequence.push(stem);
644
+ }
645
+ }
646
+ return sequence;
647
+ }
648
+
649
+ function tokenOverlapScore(queryTokens, record) {
650
+ if (queryTokens.size === 0) return 0;
651
+ const recordTokens = meaningfulTokensFrom(record.content);
652
+ let score = 0;
653
+ for (const token of queryTokens) if (recordTokens.has(token)) score += 1;
654
+ return score;
655
+ }
656
+
657
+ function normalizedTagSessionToken(value) {
658
+ return String(value ?? '').toLowerCase().replace(/^session[-_:]?/u, '').replace(/[^a-z0-9]+/gu, '');
659
+ }
660
+
661
+ function roleHintsFrom(value) {
662
+ const hints = new Set();
663
+ const text = String(value ?? '').toLowerCase();
664
+ for (const match of text.matchAll(/\b(?:assistant|user|system|human|agent|speaker[-_\s]?[a-z0-9]+)\b/gu)) {
665
+ const compact = match[0].replace(/\s+/gu, '_');
666
+ hints.add(compact);
667
+ for (const token of stemmedMeaningfulTokensFrom(compact)) hints.add(token);
668
+ }
669
+ return hints;
670
+ }
671
+
672
+ function sessionHintsFrom(value) {
673
+ const hints = new Set();
674
+ const text = String(value ?? '').toLowerCase();
675
+ for (const match of text.matchAll(/\b(?:session|sess)\s*[-_:]?\s*([a-z0-9]+)\b/gu)) hints.add(match[1]);
676
+ for (const match of text.matchAll(/\bd\s*[-_:]?\s*(\d+)\b/gu)) hints.add(`d${Number(match[1])}`);
677
+ for (const match of text.matchAll(/\bsession[-_]([a-z0-9]+)\b/gu)) hints.add(match[1]);
678
+ return hints;
679
+ }
680
+
681
+ function dateHintsFrom(value) {
682
+ const hints = new Set();
683
+ const text = String(value ?? '').toLowerCase();
684
+ for (const match of text.matchAll(/\b(?:19|20)\d{2}\b/gu)) hints.add(match[0]);
685
+ for (const match of text.matchAll(/\b\d{4}[-/]\d{1,2}(?:[-/]\d{1,2})?\b/gu)) hints.add(match[0].replace(/\D+/gu, '-'));
686
+ for (const match of text.matchAll(/\b(?:jan(?:uary)?|feb(?:ruary)?|mar(?:ch)?|apr(?:il)?|may|jun(?:e)?|jul(?:y)?|aug(?:ust)?|sep(?:t(?:ember)?)?|oct(?:ober)?|nov(?:ember)?|dec(?:ember)?)\b/gu)) hints.add(match[0].slice(0, 3));
687
+ for (const match of text.matchAll(/\b(?:today|yesterday|tomorrow|recent|recently|latest|newest|current|previous|last|earliest|oldest)\b/gu)) hints.add(match[0]);
688
+ return hints;
689
+ }
690
+
691
+ function recordSessionTokens(record) {
692
+ const tokens = new Set();
693
+ const values = [record.session_id, record.dialog_id, record.turn_id, ...(record.tags ?? [])];
694
+ for (const value of values) {
695
+ if (value === undefined || value === null) continue;
696
+ const normalized = normalizedTagSessionToken(value);
697
+ if (normalized) tokens.add(normalized);
698
+ for (const hint of sessionHintsFrom(value)) tokens.add(normalizedTagSessionToken(hint));
699
+ }
700
+ return tokens;
701
+ }
702
+
703
+ function countSetIntersection(left, right) {
704
+ let count = 0;
705
+ for (const value of left) if (right.has(value)) count += 1;
706
+ return count;
707
+ }
708
+
709
+ function phraseAndProximityScore(querySequence, recordSequence) {
710
+ if (querySequence.length < 2 || recordSequence.length < 2) return 0;
711
+ const recordBigrams = new Set();
712
+ const positions = new Map();
713
+ for (let index = 0; index < recordSequence.length; index += 1) {
714
+ const token = recordSequence[index];
715
+ if (!positions.has(token)) positions.set(token, []);
716
+ positions.get(token).push(index);
717
+ if (index > 0) recordBigrams.add(`${recordSequence[index - 1]}\u0000${token}`);
718
+ }
719
+ let score = 0;
720
+ for (let index = 1; index < querySequence.length; index += 1) {
721
+ const previous = querySequence[index - 1];
722
+ const current = querySequence[index];
723
+ if (previous === current) continue;
724
+ if (recordBigrams.has(`${previous}\u0000${current}`)) {
725
+ score += 10;
726
+ continue;
727
+ }
728
+ const leftPositions = positions.get(previous);
729
+ const rightPositions = positions.get(current);
730
+ if (!leftPositions || !rightPositions) continue;
731
+ let near = false;
732
+ for (const left of leftPositions) {
733
+ for (const right of rightPositions) {
734
+ if (Math.abs(left - right) <= 6) {
735
+ near = true;
736
+ break;
737
+ }
738
+ }
739
+ if (near) break;
740
+ }
741
+ if (near) score += 4;
742
+ }
743
+ return score;
744
+ }
745
+
746
+ function hasRecencyIntent(hints) {
747
+ for (const hint of hints) {
748
+ if (hint === 'recent' || hint === 'recently' || hint === 'latest' || hint === 'newest' || hint === 'current' || hint === 'previous' || hint === 'last') return true;
749
+ }
750
+ return false;
751
+ }
752
+
753
+ function enigmaRelevanceScore(query, record) {
754
+ const question = String(query.question ?? '');
755
+ const queryTokens = stemmedMeaningfulTokensFrom(question);
756
+ const categoryTokens = stemmedMeaningfulTokensFrom(`${query.category ?? ''} ${query.question_type ?? ''}`);
757
+ const contentTokens = stemmedMeaningfulTokensFrom(record.content);
758
+ const metadataTokens = stemmedMeaningfulTokensFrom(`${record.kind ?? ''} ${(record.tags ?? []).join(' ')}`);
759
+ for (const token of stemmedMeaningfulTokensFrom(`${record.role ?? ''} ${record.session_id ?? ''} ${record.dialog_id ?? ''} ${record.turn_id ?? ''}`)) {
760
+ metadataTokens.add(token);
761
+ }
762
+
763
+ const contentMatches = countSetIntersection(queryTokens, contentTokens);
764
+ const metadataMatches = countSetIntersection(queryTokens, metadataTokens);
765
+ const categoryMatches = countSetIntersection(categoryTokens, metadataTokens) + countSetIntersection(categoryTokens, contentTokens);
766
+ const roleMatches = countSetIntersection(roleHintsFrom(question), roleHintsFrom(record.role));
767
+ const querySessionHints = sessionHintsFrom(question);
768
+ const recordSessions = recordSessionTokens(record);
769
+ const sessionMatches = countSetIntersection(querySessionHints, recordSessions);
770
+ const queryDateHints = dateHintsFrom(question);
771
+ const temporalMatches = countSetIntersection(queryDateHints, dateHintsFrom(`${record.content} ${(record.tags ?? []).join(' ')}`));
772
+ const temporalRecencyScore = hasRecencyIntent(queryDateHints) ? Math.min(6, Math.log2(record.ordinal + 2)) : 0;
773
+ const phraseScore = phraseAndProximityScore(stemmedTokenSequenceFrom(question), stemmedTokenSequenceFrom(record.content));
774
+
775
+ return {
776
+ record,
777
+ score: (contentMatches * 12)
778
+ + (metadataMatches * 4)
779
+ + (categoryMatches * 3)
780
+ + (roleMatches * 18)
781
+ + (sessionMatches * 16)
782
+ + (temporalMatches * 8)
783
+ + temporalRecencyScore
784
+ + phraseScore,
785
+ contentMatches,
786
+ metadataMatches,
787
+ categoryMatches,
788
+ roleMatches,
789
+ sessionMatches,
790
+ temporalMatches,
791
+ temporalRecencyScore,
792
+ phraseScore,
793
+ };
794
+ }
795
+
796
+ function compareRecordId(left, right) {
797
+ return String(left.id).localeCompare(String(right.id));
798
+ }
799
+
800
+ function compareRecencyDesc(left, right) {
801
+ if (left.ordinal !== right.ordinal) return right.ordinal - left.ordinal;
802
+ return compareRecordId(left, right);
803
+ }
804
+
805
+ function rankedByOverlap(records, query) {
806
+ const queryTokens = meaningfulTokensFrom(query);
807
+ if (queryTokens.size === 0) return [];
808
+ const scored = [];
809
+ for (const record of records) {
810
+ const score = tokenOverlapScore(queryTokens, record);
811
+ if (score > 0) scored.push({ record, score });
812
+ }
813
+ scored.sort((left, right) => {
814
+ if (left.score !== right.score) return right.score - left.score;
815
+ return compareRecencyDesc(left.record, right.record);
816
+ });
817
+ return scored.map((item) => item.record);
818
+ }
819
+
820
+ function rankedByEnigmaRelevance(records, query) {
821
+ const scored = [];
822
+ for (const record of records) {
823
+ const item = enigmaRelevanceScore(query, record);
824
+ if (item.score > 0) scored.push(item);
825
+ }
826
+ if (scored.length === 0) return [...records].sort(compareRecencyDesc);
827
+ scored.sort((left, right) => {
828
+ if (left.score !== right.score) return right.score - left.score;
829
+ if (left.phraseScore !== right.phraseScore) return right.phraseScore - left.phraseScore;
830
+ if (left.contentMatches !== right.contentMatches) return right.contentMatches - left.contentMatches;
831
+ if (left.roleMatches !== right.roleMatches) return right.roleMatches - left.roleMatches;
832
+ if (left.sessionMatches !== right.sessionMatches) return right.sessionMatches - left.sessionMatches;
833
+ if (left.temporalMatches !== right.temporalMatches) return right.temporalMatches - left.temporalMatches;
834
+ if (left.temporalRecencyScore !== right.temporalRecencyScore) return right.temporalRecencyScore - left.temporalRecencyScore;
835
+ if (left.metadataMatches !== right.metadataMatches) return right.metadataMatches - left.metadataMatches;
836
+ return compareRecencyDesc(left.record, right.record);
837
+ });
838
+ return scored.map((item) => item.record);
839
+ }
840
+
841
+ function selectRecords(methodId, records, query, topK) {
842
+ if (methodId === 'full_context') return records;
843
+ if (methodId === 'recency_last_n') return [...records].sort(compareRecencyDesc).slice(0, topK);
844
+ if (methodId === 'keyword_filter') return rankedByOverlap(records, query.question).slice(0, topK);
845
+ if (methodId === 'enigma_relevance') return rankedByEnigmaRelevance(records, query).slice(0, topK);
846
+ throw new Error(`Unknown method ${methodId}`);
847
+ }
848
+
849
+ function estimatePromptTokens(question, selectedRecords) {
850
+ let tokens = estimateTextTokens(question);
851
+ for (const record of selectedRecords) tokens += record.estimated_tokens;
852
+ return tokens;
853
+ }
854
+
855
+ function percentile(values, ratio) {
856
+ if (values.length === 0) return 0;
857
+ const sorted = [...values].sort((left, right) => left - right);
858
+ const index = Math.min(sorted.length - 1, Math.max(0, Math.ceil(sorted.length * ratio) - 1));
859
+ return sorted[index];
860
+ }
861
+
862
+ function latencySummary(values) {
863
+ return {
864
+ samples: values.length,
865
+ p50_ms: round6(percentile(values, 0.5)),
866
+ p95_ms: round6(percentile(values, 0.95)),
867
+ min_ms: round6(values.length === 0 ? 0 : Math.min(...values)),
868
+ max_ms: round6(values.length === 0 ? 0 : Math.max(...values)),
869
+ };
870
+ }
871
+
872
+ function rate(numerator, denominator) {
873
+ if (denominator === 0) return null;
874
+ return round6(numerator / denominator);
875
+ }
876
+
877
+ function round6(value) {
878
+ return Number(value.toFixed(6));
879
+ }
880
+
881
+ function mean(total, count) {
882
+ return count === 0 ? 0 : round6(total / count);
883
+ }
884
+
885
+ function recordFromContent(args) {
886
+ const content = String(args.content ?? '');
887
+ return {
888
+ id: args.id,
889
+ dataset_item_id: args.dataset_item_id,
890
+ session_id: args.session_id,
891
+ turn_id: args.turn_id,
892
+ dialog_id: args.dialog_id,
893
+ role: args.role,
894
+ kind: args.kind,
895
+ tags: args.tags ?? [],
896
+ has_answer: args.has_answer === true,
897
+ ordinal: args.ordinal,
898
+ content,
899
+ estimated_tokens: estimateTextTokens(content),
900
+ };
901
+ }
902
+
903
+ function parseLocomoEvidenceLabels(evidence) {
904
+ const labels = new Set();
905
+ const stack = Array.isArray(evidence) ? [...evidence] : [evidence];
906
+ while (stack.length > 0) {
907
+ const value = stack.shift();
908
+ if (Array.isArray(value)) {
909
+ stack.push(...value);
910
+ continue;
911
+ }
912
+ if (value === undefined || value === null) continue;
913
+ for (const match of String(value).matchAll(/D\s*(\d+)\s*:\s*(\d+)/giu)) {
914
+ labels.add(`D${Number(match[1])}:${Number(match[2])}`);
915
+ }
916
+ }
917
+ return labels;
918
+ }
919
+
920
+ function dialogIdForTurn(sessionNumber, turn, turnIndex) {
921
+ const raw = turn?.dia_id ?? turn?.dialog_id ?? turn?.turn_id ?? turn?.id ?? turnIndex + 1;
922
+ const text = String(raw);
923
+ const match = text.match(/^D\s*(\d+)\s*:\s*(\d+)$/iu);
924
+ if (match) return `D${Number(match[1])}:${Number(match[2])}`;
925
+ const numeric = text.match(/\d+/u)?.[0] ?? String(turnIndex + 1);
926
+ return `D${sessionNumber}:${Number(numeric)}`;
927
+ }
928
+
929
+ export function parseLocomoDataset(data, options = {}) {
930
+ const maxQa = optionalPositiveInteger(options.max_qa ?? options.maxQa, 'max_locomo_qa');
931
+ const rows = normalizeDatasetArray(data, 'LoCoMo');
932
+ const records = [];
933
+ const queries = [];
934
+ let ordinal = 0;
935
+ for (let sampleIndex = 0; sampleIndex < rows.length; sampleIndex += 1) {
936
+ const sample = rows[sampleIndex] ?? {};
937
+ const itemId = String(sample.sample_id ?? sample.id ?? `sample_${sampleIndex + 1}`);
938
+ const conversation = sample.conversation ?? {};
939
+ const sessionNames = Object.keys(conversation)
940
+ .map((key) => {
941
+ const match = key.match(/^session_(\d+)$/u);
942
+ return match ? { key, sessionNumber: Number(match[1]) } : null;
943
+ })
944
+ .filter(Boolean)
945
+ .sort((left, right) => left.sessionNumber - right.sessionNumber);
946
+ for (const { key, sessionNumber } of sessionNames) {
947
+ const session = conversation[key];
948
+ if (!Array.isArray(session)) continue;
949
+ for (let turnIndex = 0; turnIndex < session.length; turnIndex += 1) {
950
+ const turn = session[turnIndex] ?? {};
951
+ const dialogId = dialogIdForTurn(sessionNumber, turn, turnIndex);
952
+ records.push(recordFromContent({
953
+ id: `locomo:${itemId}:${dialogId}`,
954
+ dataset_item_id: itemId,
955
+ session_id: `D${sessionNumber}`,
956
+ turn_id: dialogId,
957
+ dialog_id: dialogId,
958
+ role: turn.speaker ?? turn.role ?? undefined,
959
+ kind: 'locomo_conversation_turn',
960
+ tags: ['locomo', `session_${sessionNumber}`],
961
+ ordinal,
962
+ content: turn.text ?? turn.content ?? turn.message ?? '',
963
+ }));
964
+ ordinal += 1;
965
+ }
966
+ }
967
+ const qaRows = Array.isArray(sample.qa) ? sample.qa : [];
968
+ for (let qaIndex = 0; qaIndex < qaRows.length; qaIndex += 1) {
969
+ if (maxQa !== undefined && queries.length >= maxQa) break;
970
+ const qa = qaRows[qaIndex] ?? {};
971
+ const evidenceDialogIds = parseLocomoEvidenceLabels(qa.evidence);
972
+ queries.push({
973
+ id: `locomo:${itemId}:qa_${qaIndex + 1}`,
974
+ dataset_item_id: itemId,
975
+ question: String(qa.question ?? ''),
976
+ category: qa.category === undefined ? undefined : String(qa.category),
977
+ evidence_dialog_ids: evidenceDialogIds,
978
+ abstention: evidenceDialogIds.size === 0,
979
+ });
980
+ }
981
+ if (maxQa !== undefined && queries.length >= maxQa) break;
982
+ }
983
+ return {
984
+ id: 'locomo',
985
+ label: 'LoCoMo',
986
+ source_url: LOCOMO_SOURCE_URL,
987
+ license: 'CC BY-NC 4.0',
988
+ parser: 'conversation session turns as memory records; qa evidence labels mapped to dialog ids such as D1:3 and semicolon-separated labels',
989
+ task_categories: [...LOCOMO_TASK_CATEGORIES],
990
+ records,
991
+ queries,
992
+ };
993
+ }
994
+
995
+ function sessionIdAt(ids, index) {
996
+ if (Array.isArray(ids) && ids[index] !== undefined && ids[index] !== null) return String(ids[index]);
997
+ return String(index);
998
+ }
999
+
1000
+ function answerSessionIds(item) {
1001
+ if (!Array.isArray(item.answer_session_ids)) return new Set();
1002
+ return new Set(item.answer_session_ids.map((id) => String(id)));
1003
+ }
1004
+
1005
+ function isLongMemEvalAbstention(item, answerSessions) {
1006
+ const id = String(item.question_id ?? item.id ?? '');
1007
+ if (id.endsWith('_abs') || id.includes('_abs_')) return true;
1008
+ if (String(item.question_type ?? '').toLowerCase().includes('abst')) return true;
1009
+ return answerSessions.size === 0;
1010
+ }
1011
+
1012
+ export function parseLongMemEvalDataset(data, options = {}) {
1013
+ const maxItems = optionalPositiveInteger(options.max_items ?? options.maxItems, 'max_longmemeval_items');
1014
+ const rows = normalizeDatasetArray(data, 'LongMemEval');
1015
+ const records = [];
1016
+ const queries = [];
1017
+ let ordinal = 0;
1018
+ const limit = maxItems === undefined ? rows.length : Math.min(rows.length, maxItems);
1019
+ for (let itemIndex = 0; itemIndex < limit; itemIndex += 1) {
1020
+ const item = rows[itemIndex] ?? {};
1021
+ const itemId = String(item.question_id ?? item.id ?? `item_${itemIndex + 1}`);
1022
+ const sessions = Array.isArray(item.haystack_sessions) ? item.haystack_sessions : [];
1023
+ const sessionIds = item.haystack_session_ids;
1024
+ const evidenceTurnIds = new Set();
1025
+ for (let sessionIndex = 0; sessionIndex < sessions.length; sessionIndex += 1) {
1026
+ const session = sessions[sessionIndex];
1027
+ if (!Array.isArray(session)) continue;
1028
+ const sessionId = sessionIdAt(sessionIds, sessionIndex);
1029
+ for (let turnIndex = 0; turnIndex < session.length; turnIndex += 1) {
1030
+ const turn = session[turnIndex] ?? {};
1031
+ const turnId = `${sessionId}:${turnIndex}`;
1032
+ if (turn.has_answer === true) evidenceTurnIds.add(turnId);
1033
+ records.push(recordFromContent({
1034
+ id: `longmemeval:${itemId}:${turnId}`,
1035
+ dataset_item_id: itemId,
1036
+ session_id: sessionId,
1037
+ turn_id: turnId,
1038
+ role: turn.role ?? undefined,
1039
+ kind: 'longmemeval_haystack_turn',
1040
+ tags: ['longmemeval', String(item.question_type ?? ''), `session_${sessionId}`],
1041
+ has_answer: turn.has_answer === true,
1042
+ ordinal,
1043
+ content: turn.content ?? turn.text ?? turn.message ?? '',
1044
+ }));
1045
+ ordinal += 1;
1046
+ }
1047
+ }
1048
+ const answerSessions = answerSessionIds(item);
1049
+ queries.push({
1050
+ id: `longmemeval:${itemId}`,
1051
+ dataset_item_id: itemId,
1052
+ question: String(item.question ?? ''),
1053
+ question_type: item.question_type === undefined ? undefined : String(item.question_type),
1054
+ evidence_turn_ids: evidenceTurnIds,
1055
+ evidence_session_ids: answerSessions,
1056
+ abstention: isLongMemEvalAbstention(item, answerSessions),
1057
+ });
1058
+ }
1059
+ return {
1060
+ id: 'longmemeval',
1061
+ label: 'LongMemEval',
1062
+ source_url: LONGMEMEVAL_SOURCE_URLS,
1063
+ license: 'See Hugging Face dataset card and upstream LongMemEval repository for the selected cleaned file.',
1064
+ parser: 'haystack_sessions turns as memory records; has_answer:true turns and answer_session_ids are used as evidence labels; _abs ids are evaluated as abstention cases',
1065
+ task_categories: [...LONGMEMEVAL_TASK_CATEGORIES],
1066
+ records,
1067
+ queries,
1068
+ };
1069
+ }
1070
+
1071
+ function scoreLocomoMethod(method, dataset, topK) {
1072
+ const latencies = [];
1073
+ let evidenceQuestions = 0;
1074
+ let hits = 0;
1075
+ let exactCoverage = 0;
1076
+ let totalTokens = 0;
1077
+ let selectedTotal = 0;
1078
+ for (const query of dataset.queries) {
1079
+ const start = performance.now();
1080
+ const records = dataset.records.filter((record) => record.dataset_item_id === query.dataset_item_id);
1081
+ const selected = selectRecords(method.id, records, query, topK);
1082
+ latencies.push(performance.now() - start);
1083
+ const selectedDialogs = new Set(selected.map((record) => record.dialog_id));
1084
+ selectedTotal += selected.length;
1085
+ totalTokens += estimatePromptTokens(query.question, selected);
1086
+ if (query.evidence_dialog_ids.size > 0) {
1087
+ evidenceQuestions += 1;
1088
+ let covered = 0;
1089
+ for (const evidenceId of query.evidence_dialog_ids) if (selectedDialogs.has(evidenceId)) covered += 1;
1090
+ if (covered > 0) hits += 1;
1091
+ if (covered === query.evidence_dialog_ids.size) exactCoverage += 1;
1092
+ }
1093
+ }
1094
+ return {
1095
+ id: method.id,
1096
+ method: method.id,
1097
+ local_method_only: true,
1098
+ external_provider_called: false,
1099
+ retrieval_proxy_only: true,
1100
+ uses_top_k: method.uses_top_k,
1101
+ top_k: method.uses_top_k ? topK : null,
1102
+ question_count: dataset.queries.length,
1103
+ evidence_question_count: evidenceQuestions,
1104
+ evidence_hit_at_k: rate(hits, evidenceQuestions),
1105
+ exact_evidence_coverage: rate(exactCoverage, evidenceQuestions),
1106
+ estimated_prompt_tokens: {
1107
+ total: totalTokens,
1108
+ mean_per_question: mean(totalTokens, dataset.queries.length),
1109
+ estimator: 'estimateTextTokens deterministic local estimator',
1110
+ },
1111
+ selected_memory_count: {
1112
+ total: selectedTotal,
1113
+ mean_per_question: mean(selectedTotal, dataset.queries.length),
1114
+ },
1115
+ latency: latencySummary(latencies),
1116
+ public_question_text_included: false,
1117
+ public_answer_text_included: false,
1118
+ raw_conversation_text_included: false,
1119
+ };
1120
+ }
1121
+
1122
+ function scoreLongMemEvalMethod(method, dataset, topK) {
1123
+ const latencies = [];
1124
+ let turnEvidenceQuestions = 0;
1125
+ let turnHits = 0;
1126
+ let exactTurnCoverage = 0;
1127
+ let sessionEvidenceQuestions = 0;
1128
+ let sessionHits = 0;
1129
+ let exactSessionCoverage = 0;
1130
+ let abstentionQuestions = 0;
1131
+ let abstentionCorrect = 0;
1132
+ let totalTokens = 0;
1133
+ let selectedTotal = 0;
1134
+ for (const query of dataset.queries) {
1135
+ const start = performance.now();
1136
+ const records = dataset.records.filter((record) => record.dataset_item_id === query.dataset_item_id);
1137
+ const selected = selectRecords(method.id, records, query, topK);
1138
+ latencies.push(performance.now() - start);
1139
+ const selectedTurns = new Set(selected.map((record) => record.turn_id));
1140
+ const selectedSessions = new Set(selected.map((record) => record.session_id));
1141
+ selectedTotal += selected.length;
1142
+ totalTokens += estimatePromptTokens(query.question, selected);
1143
+
1144
+ if (query.abstention) {
1145
+ abstentionQuestions += 1;
1146
+ let selectedGold = false;
1147
+ for (const turnId of query.evidence_turn_ids) if (selectedTurns.has(turnId)) selectedGold = true;
1148
+ for (const sessionId of query.evidence_session_ids) if (selectedSessions.has(sessionId)) selectedGold = true;
1149
+ if (!selectedGold) abstentionCorrect += 1;
1150
+ continue;
1151
+ }
1152
+
1153
+ if (query.evidence_turn_ids.size > 0) {
1154
+ turnEvidenceQuestions += 1;
1155
+ let covered = 0;
1156
+ for (const turnId of query.evidence_turn_ids) if (selectedTurns.has(turnId)) covered += 1;
1157
+ if (covered > 0) turnHits += 1;
1158
+ if (covered === query.evidence_turn_ids.size) exactTurnCoverage += 1;
1159
+ }
1160
+ if (query.evidence_session_ids.size > 0) {
1161
+ sessionEvidenceQuestions += 1;
1162
+ let covered = 0;
1163
+ for (const sessionId of query.evidence_session_ids) if (selectedSessions.has(sessionId)) covered += 1;
1164
+ if (covered > 0) sessionHits += 1;
1165
+ if (covered === query.evidence_session_ids.size) exactSessionCoverage += 1;
1166
+ }
1167
+ }
1168
+ return {
1169
+ id: method.id,
1170
+ method: method.id,
1171
+ local_method_only: true,
1172
+ external_provider_called: false,
1173
+ retrieval_proxy_only: true,
1174
+ uses_top_k: method.uses_top_k,
1175
+ top_k: method.uses_top_k ? topK : null,
1176
+ item_count: dataset.queries.length,
1177
+ turn_evidence_question_count: turnEvidenceQuestions,
1178
+ turn_evidence_hit_at_k: rate(turnHits, turnEvidenceQuestions),
1179
+ exact_turn_evidence_coverage: rate(exactTurnCoverage, turnEvidenceQuestions),
1180
+ session_evidence_question_count: sessionEvidenceQuestions,
1181
+ session_evidence_hit_at_k: rate(sessionHits, sessionEvidenceQuestions),
1182
+ exact_session_evidence_coverage: rate(exactSessionCoverage, sessionEvidenceQuestions),
1183
+ abstention_questions: abstentionQuestions,
1184
+ abstention_correct: abstentionCorrect,
1185
+ abstention_correctness: rate(abstentionCorrect, abstentionQuestions),
1186
+ estimated_prompt_tokens: {
1187
+ total: totalTokens,
1188
+ mean_per_item: mean(totalTokens, dataset.queries.length),
1189
+ estimator: 'estimateTextTokens deterministic local estimator',
1190
+ },
1191
+ selected_memory_count: {
1192
+ total: selectedTotal,
1193
+ mean_per_item: mean(selectedTotal, dataset.queries.length),
1194
+ },
1195
+ latency: latencySummary(latencies),
1196
+ public_question_text_included: false,
1197
+ public_answer_text_included: false,
1198
+ raw_conversation_text_included: false,
1199
+ };
1200
+ }
1201
+
1202
+ function scoreDataset(dataset, topK) {
1203
+ const methodRows = STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => (
1204
+ dataset.id === 'locomo' ? scoreLocomoMethod(method, dataset, topK) : scoreLongMemEvalMethod(method, dataset, topK)
1205
+ ));
1206
+ return {
1207
+ id: dataset.id,
1208
+ dataset: dataset.id,
1209
+ label: dataset.label,
1210
+ source_url: dataset.source_url,
1211
+ license: dataset.license,
1212
+ parser: dataset.parser,
1213
+ task_categories: dataset.task_categories,
1214
+ record_count: dataset.records.length,
1215
+ question_count: dataset.queries.length,
1216
+ item_count: dataset.queries.length,
1217
+ raw_question_text_included: false,
1218
+ raw_answer_text_included: false,
1219
+ raw_conversation_text_included: false,
1220
+ methods: methodRows,
1221
+ };
1222
+ }
1223
+
1224
+ export function runStandardMemoryBenchmarkSuite(options = {}) {
1225
+ const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
1226
+ const datasetRows = [];
1227
+ if (options.locomoData !== undefined) {
1228
+ datasetRows.push(scoreDataset(parseLocomoDataset(options.locomoData, { max_qa: options.max_locomo_qa ?? options.maxLocomoQa }), topK));
1229
+ }
1230
+ if (options.longMemEvalData !== undefined || options.longmemevalData !== undefined) {
1231
+ datasetRows.push(scoreDataset(parseLongMemEvalDataset(options.longMemEvalData ?? options.longmemevalData, { max_items: options.max_longmemeval_items ?? options.maxLongMemEvalItems }), topK));
1232
+ }
1233
+ if (datasetRows.length === 0) throw new Error('At least one standard dataset is required: provide locomoData and/or longMemEvalData');
1234
+ return buildSuiteReport(datasetRows, topK, options);
1235
+ }
1236
+
1237
+ export async function runStandardMemoryBenchmarkSuiteFromFiles(options = {}) {
1238
+ const datasetRows = [];
1239
+ const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
1240
+ if (options.locomo !== undefined || options.locomoPath !== undefined) {
1241
+ const path = options.locomo ?? options.locomoPath;
1242
+ const loaded = await readJsonWithSha256(path);
1243
+ datasetRows.push({
1244
+ ...scoreDataset(parseLocomoDataset(loaded.data, { max_qa: options.max_locomo_qa ?? options.maxLocomoQa }), topK),
1245
+ local_file_name: publicFileName(path),
1246
+ input_sha256: loaded.sha256,
1247
+ });
1248
+ }
1249
+ if (options.longmemeval !== undefined || options.longmemevalPath !== undefined || options.longMemEvalPath !== undefined) {
1250
+ const path = options.longmemeval ?? options.longmemevalPath ?? options.longMemEvalPath;
1251
+ const maxItems = optionalPositiveInteger(options.max_longmemeval_items ?? options.maxLongMemEvalItems, 'max_longmemeval_items');
1252
+ const loaded = await readLongMemEvalJsonWithSha256(path, maxItems);
1253
+ datasetRows.push({
1254
+ ...scoreDataset(parseLongMemEvalDataset(loaded.data, { max_items: maxItems }), topK),
1255
+ local_file_name: publicFileName(path),
1256
+ input_sha256: loaded.sha256,
1257
+ });
1258
+ }
1259
+ if (datasetRows.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
1260
+ return buildSuiteReport(datasetRows, topK, options);
1261
+ }
1262
+
1263
+ function buildSuiteReport(datasetRows, topK, options) {
1264
+ return {
1265
+ schema: STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA,
1266
+ generated_at: options.generated_at ?? new Date().toISOString(),
1267
+ package: {
1268
+ name: 'enigma-memory',
1269
+ version: '0.1.17',
1270
+ },
1271
+ public_safe: true,
1272
+ top_k: topK,
1273
+ source_urls: {
1274
+ locomo: LOCOMO_SOURCE_URL,
1275
+ longmemeval: LONGMEMEVAL_SOURCE_URLS,
1276
+ },
1277
+ license_and_boundary_notes: [
1278
+ 'LoCoMo source data is CC BY-NC 4.0; keep local dataset files and raw conversations out of public reports unless separately reviewed.',
1279
+ 'LongMemEval cleaned files are operator-supplied local JSON files from the upstream Hugging Face dataset repository.',
1280
+ 'Scores are retrieval/evidence proxy metrics over official dataset labels, not LLM-generated answer accuracy.',
1281
+ 'No provider APIs, hosted runtimes, competitor SDKs, or external accounts are called by this runner.',
1282
+ 'Rows are local deterministic methods only; no third-party competitor scores or benchmark-leadership claims are emitted.',
1283
+ 'Product runtime boundary: the shipped enigma context and search commands use keyword token overlap only. The richer stem, role/session/kind/tag, category/task, temporal/date, and phrase/proximity signals below are benchmark-only local approximations and are not used by the product runtime.',
1284
+ ],
1285
+ command_boundaries: offlineCommandBoundaries({ scoresIncluded: true, datasetFilesRead: true }),
1286
+ apples_to_apples_controls: applesToApplesControls(topK),
1287
+ benchmark_boundaries: {
1288
+ official_dataset_files_required: true,
1289
+ credentials_required: false,
1290
+ external_provider_calls: false,
1291
+ llm_answer_accuracy_scored: false,
1292
+ retrieval_evidence_proxy_scored: true,
1293
+ raw_question_text_included: false,
1294
+ raw_answer_text_included: false,
1295
+ raw_conversation_text_included: false,
1296
+ provider_deletion_claim: false,
1297
+ model_forgetting_claim: false,
1298
+ roi_or_provider_invoice_savings_claim: false,
1299
+ compliance_certification_claim: false,
1300
+ benchmark_leadership_claim: false,
1301
+ },
1302
+ relevance_logic: {
1303
+ token_extraction: 'keyword_filter uses basic lowercase /[a-z0-9]+(?:[-_][a-z0-9]+)*/ overlap after stopword removal; enigma_relevance additionally applies deterministic suffix stemming for ing, ed, and plural s forms',
1304
+ production_alignment: 'public-safe local approximation of production query-aware retrieval using query/content stems, role/session/kind/tag hints, category/task hints, temporal/date hints, and phrase/proximity boosts; no private memory is emitted',
1305
+ product_runtime_boundary: 'The shipped enigma context and search product runtime uses keyword token overlap only. Stem, role/session/kind/tag, category/task, temporal/date, and phrase/proximity signals are benchmark-only local approximations and are not used by the product runtime.',
1306
+ keyword_filter_fallback: 'empty result when no query/content token overlap exists',
1307
+ enigma_relevance_fallback: 'falls back to all local candidates only when no enhanced relevance signal exists, then applies deterministic local ranking and --top-k',
1308
+ provider_api_used: false,
1309
+ llm_used: false,
1310
+ gold_labels_used_for_retrieval: false,
1311
+ },
1312
+ local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
1313
+ external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
1314
+ ...adapter,
1315
+ required_artifacts: [...adapter.required_artifacts],
1316
+ })),
1317
+ datasets: datasetRows,
1318
+ dataset_rows: datasetRows,
1319
+ };
1320
+ }
1321
+
1322
+ function usage() {
1323
+ return `Usage: node scripts/run-standard-memory-benchmarks.mjs [--locomo <path>] [--longmemeval <path>] [--max-locomo-qa <n>] [--max-longmemeval-items <n>] [--top-k <n>] [--out <path>] [--dry-run] [--protocol-plan [--answerer-ref <ref>] [--judge-ref <ref>] [--answer-prompt-ref <ref>] [--judge-prompt-ref <ref>] [--protocol-ref <ref>]]\n\nProduces schema ${STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA}. Raw question, answer, and conversation text are never written to the report. With --longmemeval and --max-longmemeval-items, the local top-level JSON array is streamed for hashing and only the requested sample items are parsed. Use --dry-run to print a public-safe offline execution plan without reading dataset files or producing scores. Use --protocol-plan to print a public-safe full-answer benchmark PROTOCOL plan (schema ${STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA}) recording the planned category set, top-k, answerer/judge model refs, prompt/protocol refs, competitor adapter refs, and cost-estimate inputs, with explicit network_required:false, provider_calls_made:false, answers_generated:false, and judged:false boundaries; it does not call providers, generate answers, judge answers, or run competitor adapters.`;
1324
+ }
1325
+
1326
+ async function main() {
1327
+ const options = parseArgs();
1328
+ if (options.help) {
1329
+ console.log(usage());
1330
+ return;
1331
+ }
1332
+ const report = options.protocol_plan
1333
+ ? buildStandardBenchmarkProtocolPlan(options)
1334
+ : options.dry_run
1335
+ ? buildStandardBenchmarkDryRunPlan(options)
1336
+ : await runStandardMemoryBenchmarkSuiteFromFiles(options);
1337
+ const serialized = `${JSON.stringify(report, null, 2)}\n`;
1338
+ if (options.out) {
1339
+ const outPath = resolve(options.out);
1340
+ await mkdir(dirname(outPath), { recursive: true });
1341
+ await writeFile(outPath, serialized);
1342
+ } else {
1343
+ process.stdout.write(serialized);
1344
+ }
1345
+ }
1346
+
1347
+ const invokedPath = process.argv[1] ? resolve(process.argv[1]) : '';
1348
+ const modulePath = fileURLToPath(import.meta.url);
1349
+ if (invokedPath === modulePath) {
1350
+ main().catch((error) => {
1351
+ console.error(error instanceof Error ? error.message : String(error));
1352
+ process.exitCode = 1;
1353
+ });
1354
+ }