enigma-memory 0.1.15 → 0.1.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/README.md +70 -86
  2. package/apps/cli/bin/enigma.mjs +152 -15
  3. package/apps/native-host/README.md +20 -0
  4. package/deploy/SIMULATION.md +34 -38
  5. package/docs/benchmark-attestation-network.md +2 -2
  6. package/docs/benchmark-reproducibility.md +70 -9
  7. package/docs/blockchain-only-mechanisms.md +12 -0
  8. package/docs/browser-extension-install.md +9 -6
  9. package/docs/client-connectors.md +29 -55
  10. package/docs/demo-proof-network.md +3 -3
  11. package/docs/developer-ecosystem.md +207 -223
  12. package/docs/developer-proof-quickstart.md +3 -3
  13. package/docs/enigma-memory-ready-conformance.md +1 -1
  14. package/docs/hosted-cloud-product.md +31 -0
  15. package/docs/install-anywhere.md +68 -70
  16. package/docs/memory-benchmarks.md +21 -3
  17. package/docs/memory-drive-health-model.md +41 -0
  18. package/docs/proof-network-build-notes.md +2 -2
  19. package/docs/proof-network.md +90 -8
  20. package/docs/sdk-api.md +1 -1
  21. package/docs/solana-devnet-acceptance.md +1 -1
  22. package/docs/solana-proof-rail.md +1 -1
  23. package/package.json +7 -1
  24. package/packages/connectors/src/index.js +13 -0
  25. package/packages/hosted-cloud/src/index.js +470 -2
  26. package/packages/mcp-server/README.md +22 -0
  27. package/packages/mcp-server/src/index.js +1 -1
  28. package/packages/passport/src/index.js +730 -0
  29. package/packages/proof-network/src/index.js +139 -0
  30. package/scripts/build-benchmark-proof-release.mjs +119 -6
  31. package/scripts/build-cloudflare-token-policy.mjs +6 -2
  32. package/scripts/build-hosted-api-key-lifecycle.mjs +1 -1
  33. package/scripts/build-hosted-customer-lifecycle.mjs +23 -3
  34. package/scripts/build-installer-assets.mjs +1 -1
  35. package/scripts/build-production-handoff-packet.mjs +1 -1
  36. package/scripts/build-production-unblocker.mjs +409 -409
  37. package/scripts/build-production-workplan.mjs +3 -1
  38. package/scripts/build-proof-network-packet.mjs +1 -1
  39. package/scripts/check.mjs +3 -1
  40. package/scripts/cloudflare-ops.mjs +35 -0
  41. package/scripts/collect-hosted-backend-live-evidence.mjs +44 -2
  42. package/scripts/run-memory-benchmarks.mjs +5 -0
  43. package/scripts/run-standard-memory-benchmarks.mjs +287 -5
  44. package/scripts/stage-cloudflare-pages-artifact.mjs +145 -0
@@ -234,6 +234,8 @@ export function buildProductionWorkplan(inputs = {}, options = {}) {
234
234
  const endpointRefs = missingHostedEndpointRefs(hosted);
235
235
  const stateBlockers = hostedStateBlockers(hosted);
236
236
 
237
+ const finalDependencyCommand = 'npm run production:dependencies -- --goal-audit .enigma/goal-audit-current.json --release-audit .enigma/release-audit-current.json --worker-inspect .enigma/worker-inspect-result-current.json --whitepaper .enigma/whitepaper-claims-current.json --cloudflare-credentials .enigma/cloudflare-credentials-current.json --edge-deploy .enigma/edge-backend-deployment-current.json --edge-live .enigma/edge-backend-bootstrap-live-current.json --storage-bootstrap .enigma/cloudflare-storage-bootstrap-current.json';
238
+
237
239
  const phases = [
238
240
  makePhase({
239
241
  id: 'cloudflare_credentials',
@@ -309,7 +311,7 @@ export function buildProductionWorkplan(inputs = {}, options = {}) {
309
311
  owner: 'operator-or-reviewer',
310
312
  prerequisites: ['cloudflare_credentials', 'cloudflare_worker_permission', 'hosted_backend_refs', 'operator_acceptance', 'release_gates'],
311
313
  blockers: dependencyReport.launch_ready === true ? [] : ['launch_ready is false'],
312
- commands: [release.next_command, staticSite.next_command, whitepaper.next_command, 'npm run production:goal-audit -- --site <public-site-dir> --domain enigmamemory.com --release-audit .enigma/release-audit-current.json', 'npm run production:dependencies -- --goal-audit .enigma/goal-audit-current.json --release-audit .enigma/release-audit-current.json --worker-inspect .enigma/worker-inspect-validation-current.json --whitepaper .enigma/whitepaper-claims-current.json --cloudflare-credentials .enigma/cloudflare-credentials-current.json --edge-deploy .enigma/edge-backend-deployment-current.json --edge-live .enigma/edge-backend-bootstrap-live-current.json --storage-bootstrap .enigma/cloudflare-storage-bootstrap-current.json'],
314
+ commands: [release.next_command, staticSite.next_command, whitepaper.next_command, 'npm run production:goal-audit -- --site <public-site-dir> --domain enigmamemory.com --release-audit .enigma/release-audit-current.json', finalDependencyCommand],
313
315
  evidence: [...release.evidence, ...staticSite.evidence, ...whitepaper.evidence],
314
316
  details: { goal_complete: dependencyReport.goal_complete === true, launch_ready: dependencyReport.launch_ready === true },
315
317
  }),
@@ -15,7 +15,7 @@ import {
15
15
  validateProofNetworkPacket,
16
16
  } from '../packages/proof-network/src/index.js';
17
17
 
18
- export const PROOF_NETWORK_PACKET_RELEASE_TARGET = '0.1.15';
18
+ export const PROOF_NETWORK_PACKET_RELEASE_TARGET = '0.1.17';
19
19
 
20
20
  const HASH_RE = /^(?:sha256:)?[a-f0-9]{64}$/iu;
21
21
  const SECRET_VALUE_RE = /(?:Bearer\s+[A-Za-z0-9._~+/=-]{12,}|Basic\s+[A-Za-z0-9+/=-]{12,}|-----BEGIN [A-Z ]*PRIVATE KEY-----|https?:\/\/[^\s/@]+:[^\s/@]+@|sk-[A-Za-z0-9_-]{16,}|AKIA[0-9A-Z]{16}|\b(?:raw[\s_-]*memory|plaintext[\s_-]*prompts?|plain[\s_-]*text[\s_-]*prompts?|private[\s_-]*prompts?|provider[\s_-]*responses?|full[\s_-]*transcript|decrypted[\s_-]*memory|credentials?|secrets?|passwords?|private[\s_-]*keys?|api[\s_-]*key[\s_-]*(?:secret|material|value)|api[\s_-]*secrets?|access[\s_-]*tokens?|refresh[\s_-]*tokens?|token[\s_-]*values?|credential[\s_-]*material|tenant[\s_-]*names?)\b)/iu;
package/scripts/check.mjs CHANGED
@@ -3,7 +3,9 @@ import path from 'node:path';
3
3
  import { fileURLToPath, pathToFileURL } from 'node:url';
4
4
 
5
5
  const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
6
- const packageJsonPath = path.join(root, 'package.json');
6
+ const packageJsonPath = process.env.ENIGMA_CHECK_PACKAGE_JSON_OVERRIDE
7
+ ? path.resolve(process.env.ENIGMA_CHECK_PACKAGE_JSON_OVERRIDE)
8
+ : path.join(root, 'package.json');
7
9
  const packageJson = JSON.parse(fs.readFileSync(packageJsonPath, 'utf8'));
8
10
  const productionRoots = [
9
11
  'packages/adapters/src/',
@@ -6,6 +6,7 @@ import { dirname, join } from 'node:path';
6
6
  import { pathToFileURL } from 'node:url';
7
7
  import { promisify } from 'node:util';
8
8
  import { applyCloudflareSecretEnvFileFromArgv, CloudflareSecretEnvError } from './cloudflare-secret-env.mjs';
9
+ import { PUBLIC_SITE_SECURITY_RESULT_SCHEMA, validatePublicSiteSecurity } from './validate-public-site-security.mjs';
9
10
 
10
11
  const execFile = promisify(execFileCallback);
11
12
 
@@ -46,6 +47,8 @@ Commands:
46
47
 
47
48
  pages deploy --site <dir> --project-name <name> [--execute]
48
49
  Without --execute, prints the exact Wrangler deploy plan only.
50
+ Dry-run output includes local public-site security validation.
51
+ --execute refuses artifacts with local security blockers before invoking Wrangler.
49
52
  With --execute, runs Wrangler through npm exec/npx without printing the token.
50
53
 
51
54
  pages verify --url <https-url> --project-name <name> [--domain <host>] \\
@@ -264,6 +267,35 @@ function redactPlanOutput(plan) {
264
267
  return redactOperationalPayload(plan);
265
268
  }
266
269
 
270
+ function redactPublicSiteSecurityBlocker(entry) {
271
+ return {
272
+ message: redactOperationalText(entry?.message ?? ''),
273
+ ...(entry?.path === undefined ? {} : { path: redactOperationalText(String(entry.path)) }),
274
+ };
275
+ }
276
+
277
+ function publicSiteSecurityDeploySummary(result) {
278
+ return {
279
+ schema: PUBLIC_SITE_SECURITY_RESULT_SCHEMA,
280
+ ok: result?.ok === true,
281
+ status: result?.status ?? 'blocked',
282
+ blocker_count: Array.isArray(result?.blockers) ? result.blockers.length : 0,
283
+ blockers: Array.isArray(result?.blockers) ? result.blockers.map(redactPublicSiteSecurityBlocker) : [],
284
+ checked: redactOperationalPayload(result?.checked ?? {}),
285
+ claimBoundary: result?.claim_boundary ?? [],
286
+ };
287
+ }
288
+
289
+ function publicSiteSecurityErrorMessage(summary) {
290
+ const blockerText = summary.blockers
291
+ .slice(0, 5)
292
+ .map((entry) => (entry.path ? `${entry.path}: ${entry.message}` : entry.message))
293
+ .join('; ');
294
+ return blockerText.length > 0
295
+ ? `public site security validation blocked Pages deploy: ${blockerText}`
296
+ : 'public site security validation blocked Pages deploy';
297
+ }
298
+
267
299
  function parsePositiveInteger(value, name) {
268
300
  const text = requireNonEmptyString(name, value);
269
301
  if (!/^\d+$/.test(text)) throw new UsageError(`${name} must be a positive integer`);
@@ -1518,6 +1550,7 @@ export async function runCloudflareOpsCommand(command, {
1518
1550
 
1519
1551
  if (command.kind === 'pages.deploy') {
1520
1552
  const plan = buildWranglerPagesDeployPlan(command);
1553
+ const siteSecurity = publicSiteSecurityDeploySummary(await validatePublicSiteSecurity({ site: command.site }));
1521
1554
  if (!command.execute) {
1522
1555
  return {
1523
1556
  json: {
@@ -1526,10 +1559,12 @@ export async function runCloudflareOpsCommand(command, {
1526
1559
  dryRun: true,
1527
1560
  execute: false,
1528
1561
  plan: redactPlanOutput(plan),
1562
+ siteSecurity,
1529
1563
  claimBoundary: 'Plan only; no Cloudflare Pages deployment was executed.',
1530
1564
  },
1531
1565
  };
1532
1566
  }
1567
+ if (!siteSecurity.ok) throw new UsageError(publicSiteSecurityErrorMessage(siteSecurity));
1533
1568
  const result = await execFileImpl(plan.command, plan.args, { shell: plan.usesShell === true, windowsHide: true, maxBuffer: 10 * 1024 * 1024, env });
1534
1569
  return {
1535
1570
  json: {
@@ -1,4 +1,5 @@
1
1
  #!/usr/bin/env node
2
+ import https from 'node:https';
2
3
  import { createHash } from 'node:crypto';
3
4
  import { mkdir, readFile, writeFile } from 'node:fs/promises';
4
5
  import { dirname, resolve } from 'node:path';
@@ -81,6 +82,42 @@ function redactProbeBody(value, path = 'probe.body') {
81
82
  return value;
82
83
  }
83
84
 
85
+ export function localSimulationLoopbackFetch(url, init = {}) {
86
+ const parsed = new URL(url);
87
+ const host = parsed.hostname.toLowerCase();
88
+ if (parsed.protocol !== 'https:' || (host !== 'sim.enigmamemory.com' && !host.endsWith('.sim.enigmamemory.com'))) {
89
+ throw new Error('--local-simulation-loopback only supports https://*.sim.enigmamemory.com simulation probes');
90
+ }
91
+ const request = {
92
+ hostname: '127.0.0.1',
93
+ port: parsed.port || 443,
94
+ path: `${parsed.pathname}${parsed.search}`,
95
+ method: init.method || 'GET',
96
+ headers: init.headers,
97
+ rejectUnauthorized: false,
98
+ servername: parsed.hostname,
99
+ };
100
+ return new Promise((resolve, reject) => {
101
+ const req = https.request(request, (res) => {
102
+ const chunks = [];
103
+ res.on('data', (chunk) => chunks.push(chunk));
104
+ res.on('end', () => {
105
+ const text = Buffer.concat(chunks).toString('utf8');
106
+ resolve({
107
+ ok: res.statusCode >= 200 && res.statusCode < 300,
108
+ status: res.statusCode,
109
+ statusText: res.statusMessage || '',
110
+ url,
111
+ redirected: false,
112
+ text: async () => text,
113
+ });
114
+ });
115
+ });
116
+ req.on('error', reject);
117
+ req.end();
118
+ });
119
+ }
120
+
84
121
  async function fetchProbe(url, { fetchImpl = globalThis.fetch, observedAt }) {
85
122
  if (typeof fetchImpl !== 'function') throw new Error('global fetch is not available in this Node runtime');
86
123
  const response = await fetchImpl(url, {
@@ -189,6 +226,10 @@ function parseArgs(argv) {
189
226
  }
190
227
  if (!arg.startsWith('--')) throw new Error(`Unexpected argument: ${arg}`);
191
228
  const name = arg.slice(2);
229
+ if (name === 'local-simulation-loopback') {
230
+ flags.set(name, true);
231
+ continue;
232
+ }
192
233
  const value = argv[index + 1];
193
234
  if (!value || value.startsWith('--')) throw new Error(`${arg} requires a value`);
194
235
  flags.set(name, value);
@@ -198,7 +239,7 @@ function parseArgs(argv) {
198
239
  }
199
240
 
200
241
  function usage() {
201
- return `Usage: node scripts/collect-hosted-backend-live-evidence.mjs --relay-url <https-base> --gateway-url <https-base> --refs-json <refs.json> --domain <domain> --environment-id <id> --cloud-provider <provider> --region <region> --owner <owner> --operator-decision go --operator-packet-ref <ref> --operator-approved-at <iso> --operator-approved-by <name> [--out <collection.json>] [--evidence-out <evidence.json>]\n\nCollects public HTTPS /livez and /readyz evidence for relay and gateway, then validates it with validate-hosted-backend-live. It never sends credentials and does not deploy infrastructure.\n`;
242
+ return `Usage: node scripts/collect-hosted-backend-live-evidence.mjs --relay-url <https-base> --gateway-url <https-base> --refs-json <refs.json> --domain <domain> --environment-id <id> --cloud-provider <provider> --region <region> --owner <owner> --operator-decision go --operator-packet-ref <ref> --operator-approved-at <iso> --operator-approved-by <name> [--out <collection.json>] [--evidence-out <evidence.json>] [--local-simulation-loopback]\n\nCollects public HTTPS /livez and /readyz evidence for relay and gateway, then validates it with validate-hosted-backend-live. It never sends credentials and does not deploy infrastructure. The --local-simulation-loopback flag is restricted to https://*.sim.enigmamemory.com local simulation probes with self-signed TLS and must not be used as production evidence.\n`;
202
243
  }
203
244
 
204
245
  async function runCli(argv = process.argv.slice(2), { fetchImpl = globalThis.fetch } = {}) {
@@ -212,6 +253,7 @@ async function runCli(argv = process.argv.slice(2), { fetchImpl = globalThis.fet
212
253
  const refs = await readJsonFile(refsPath);
213
254
  const environment = await maybeReadJsonFile(readFlag(flags, 'environment-json'));
214
255
  const operatorAcceptance = await maybeReadJsonFile(readFlag(flags, 'operator-acceptance-json'));
256
+ const selectedFetchImpl = flags.get('local-simulation-loopback') === true ? localSimulationLoopbackFetch : fetchImpl;
215
257
  const collection = await collectHostedBackendLiveEvidence({
216
258
  relayBaseUrl: readFlag(flags, 'relay-url'),
217
259
  gatewayBaseUrl: readFlag(flags, 'gateway-url'),
@@ -233,7 +275,7 @@ async function runCli(argv = process.argv.slice(2), { fetchImpl = globalThis.fet
233
275
  operatorApprovedAt: readFlag(flags, 'operator-approved-at'),
234
276
  operatorApprovedBy: readFlag(flags, 'operator-approved-by'),
235
277
  observed_at: readFlag(flags, 'observed-at') ?? new Date().toISOString(),
236
- fetchImpl,
278
+ fetchImpl: selectedFetchImpl,
237
279
  });
238
280
  const collectionJson = `${JSON.stringify(collection, null, 2)}\n`;
239
281
  const evidenceJson = `${JSON.stringify(collection.evidence, null, 2)}\n`;
@@ -832,6 +832,11 @@ export function runMemoryBenchmarkSuite(options = {}) {
832
832
  credentials_required: false,
833
833
  external_downloads_required: false,
834
834
  external_provider_calls: false,
835
+ llm_answer_accuracy_scored: false,
836
+ provider_api_calls_made: false,
837
+ api_spend_possible: false,
838
+ mem0_adapter_run: false,
839
+ external_competitor_adapters_run: false,
835
840
  raw_private_memory_plaintext_included: false,
836
841
  provider_deletion_claim: false,
837
842
  model_forgetting_claim: false,
@@ -37,12 +37,41 @@ export const STANDARD_MEMORY_BENCHMARK_METHODS = Object.freeze([
37
37
  }),
38
38
  ]);
39
39
 
40
+ export const STANDARD_EXTERNAL_COMPETITOR_ADAPTERS = Object.freeze([
41
+ Object.freeze({
42
+ id: 'mem0',
43
+ name: 'Mem0',
44
+ status: 'not_run_requires_credentials_or_runtime',
45
+ target_type: 'external_adapter',
46
+ can_run_in_this_harness: false,
47
+ scores_included: false,
48
+ required_artifacts: Object.freeze([
49
+ 'Mem0 platform credentials or open-source runtime',
50
+ 'Pinned Mem0 SDK/package versions',
51
+ 'Fixed extraction, update, retrieval, reset, model, and tool policy',
52
+ 'Same reviewed dataset manifest, split, top-k, and scorer as Enigma rows',
53
+ ]),
54
+ official_doc: 'https://docs.mem0.ai/',
55
+ boundary_reason: 'The standard runner has no Mem0 credentials, SDK/runtime, fixed memory loop, reset policy, model/tool environment, or reviewed adapter scorer, so no Mem0 score is produced.',
56
+ }),
57
+ ]);
58
+
40
59
  const LOCOMO_SOURCE_URL = 'https://raw.githubusercontent.com/snap-research/locomo/main/data/locomo10.json';
41
60
  const LONGMEMEVAL_SOURCE_URLS = Object.freeze([
42
61
  'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_oracle.json',
43
62
  'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_s_cleaned.json',
44
63
  'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_m_cleaned.json',
45
64
  ]);
65
+ const LOCOMO_TASK_CATEGORIES = Object.freeze(['multi-session QA', 'event summarization', 'multimodal generation over long conversations']);
66
+ const LONGMEMEVAL_TASK_CATEGORIES = Object.freeze(['information extraction', 'multi-session reasoning', 'temporal reasoning', 'knowledge updates', 'abstention']);
67
+ const DATASET_TASK_CATEGORIES = Object.freeze({ locomo: LOCOMO_TASK_CATEGORIES, longmemeval: LONGMEMEVAL_TASK_CATEGORIES });
68
+ export const STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA = 'enigma.standard_memory_benchmark_protocol_plan.v1';
69
+ const PROTOCOL_REF_RE = /^[a-z0-9][a-z0-9._:/@+-]{2,191}$/u;
70
+ const DEFAULT_ANSWERER_MODEL_REF = 'model:answerer-not-selected';
71
+ const DEFAULT_JUDGE_MODEL_REF = 'model:judge-not-selected';
72
+ const DEFAULT_ANSWER_PROMPT_REF = 'prompt:standard-answer@not-pinned';
73
+ const DEFAULT_JUDGE_PROMPT_REF = 'prompt:standard-judge@not-pinned';
74
+ const DEFAULT_PROTOCOL_REF = 'protocol:apples-to-apples-full-answer@not-pinned';
46
75
 
47
76
  const QUERY_RELEVANCE_STOPWORDS = new Set([
48
77
  'about',
@@ -121,6 +150,25 @@ function parseArgs(argv = process.argv.slice(2)) {
121
150
  } else if (arg === '--out') {
122
151
  options.out = requiredFlagValue(argv, index, arg);
123
152
  index += 1;
153
+ } else if (arg === '--protocol-plan') {
154
+ options.protocol_plan = true;
155
+ } else if (arg === '--answerer-ref') {
156
+ options.answerer_ref = requiredFlagValue(argv, index, arg);
157
+ index += 1;
158
+ } else if (arg === '--judge-ref') {
159
+ options.judge_ref = requiredFlagValue(argv, index, arg);
160
+ index += 1;
161
+ } else if (arg === '--answer-prompt-ref') {
162
+ options.answer_prompt_ref = requiredFlagValue(argv, index, arg);
163
+ index += 1;
164
+ } else if (arg === '--judge-prompt-ref') {
165
+ options.judge_prompt_ref = requiredFlagValue(argv, index, arg);
166
+ index += 1;
167
+ } else if (arg === '--protocol-ref') {
168
+ options.protocol_ref = requiredFlagValue(argv, index, arg);
169
+ index += 1;
170
+ } else if (arg === '--dry-run') {
171
+ options.dry_run = true;
124
172
  } else if (arg === '--help' || arg === '-h') {
125
173
  options.help = true;
126
174
  } else {
@@ -147,6 +195,229 @@ function optionalPositiveInteger(value, name) {
147
195
  return positiveInteger(value, name);
148
196
  }
149
197
 
198
+ function datasetPlanRows(options) {
199
+ const rows = [];
200
+ if (options.locomo !== undefined || options.locomoPath !== undefined) {
201
+ rows.push({
202
+ id: 'locomo',
203
+ label: 'LoCoMo',
204
+ local_file_name: publicFileName(options.locomo ?? options.locomoPath),
205
+ source_url: LOCOMO_SOURCE_URL,
206
+ license: 'CC BY-NC 4.0',
207
+ sample_limit: optionalPositiveInteger(options.max_locomo_qa ?? options.maxLocomoQa, 'max_locomo_qa') ?? null,
208
+ parser: 'conversation session turns as memory records; qa evidence labels score support only',
209
+ });
210
+ }
211
+ if (options.longmemeval !== undefined || options.longmemevalPath !== undefined || options.longMemEvalPath !== undefined) {
212
+ rows.push({
213
+ id: 'longmemeval',
214
+ label: 'LongMemEval',
215
+ local_file_name: publicFileName(options.longmemeval ?? options.longmemevalPath ?? options.longMemEvalPath),
216
+ source_url: LONGMEMEVAL_SOURCE_URLS,
217
+ license: 'Review upstream Hugging Face dataset card and LongMemEval repository terms.',
218
+ sample_limit: optionalPositiveInteger(options.max_longmemeval_items ?? options.maxLongMemEvalItems, 'max_longmemeval_items') ?? null,
219
+ parser: 'haystack_sessions turns as memory records; answer-session labels score support only',
220
+ });
221
+ }
222
+ return rows;
223
+ }
224
+
225
+ function offlineCommandBoundaries({ scoresIncluded, datasetFilesRead }) {
226
+ return {
227
+ deterministic_offline_runner: true,
228
+ dataset_files_read_from_local_disk: datasetFilesRead,
229
+ network_calls_made: false,
230
+ provider_api_calls_made: false,
231
+ api_spend_possible: false,
232
+ hosted_memory_service_called: false,
233
+ external_competitor_adapters_run: false,
234
+ mem0_adapter_run: false,
235
+ llm_used: false,
236
+ llm_answer_accuracy_scored: false,
237
+ retrieval_evidence_proxy_scored: scoresIncluded,
238
+ benchmark_scores_included: scoresIncluded,
239
+ raw_question_text_included: false,
240
+ raw_answer_text_included: false,
241
+ raw_conversation_text_included: false,
242
+ gold_labels_used_for_retrieval: false,
243
+ gold_labels_used_for_scoring: scoresIncluded,
244
+ };
245
+ }
246
+
247
+ function applesToApplesControls(topK) {
248
+ return {
249
+ same_top_k_for_all_methods: true,
250
+ top_k: topK,
251
+ same_parser_per_dataset: true,
252
+ same_local_records_per_dataset: true,
253
+ same_gold_evidence_labels_per_dataset_for_scoring_only: true,
254
+ local_deterministic_methods_only: true,
255
+ provider_runtime_fixed: false,
256
+ competitor_runtime_fixed: false,
257
+ answer_generator_fixed: false,
258
+ evaluator_model_fixed: false,
259
+ };
260
+ }
261
+
262
+ export function buildStandardBenchmarkDryRunPlan(options = {}) {
263
+ const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
264
+ const datasets = datasetPlanRows(options);
265
+ if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
266
+ return {
267
+ schema: 'enigma.standard_memory_benchmark_plan.v1',
268
+ generated_at: options.generated_at ?? new Date().toISOString(),
269
+ package: {
270
+ name: 'enigma-memory',
271
+ version: '0.1.17',
272
+ },
273
+ public_safe: true,
274
+ dry_run: true,
275
+ top_k: topK,
276
+ datasets_planned: datasets,
277
+ local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
278
+ external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
279
+ ...adapter,
280
+ required_artifacts: [...adapter.required_artifacts],
281
+ })),
282
+ command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
283
+ apples_to_apples_controls: applesToApplesControls(topK),
284
+ non_claims: [
285
+ 'This dry run does not read dataset files and produces no benchmark score.',
286
+ 'No provider APIs, hosted memory services, Mem0 runtime, competitor SDKs, LLM generators, or evaluator models are called.',
287
+ 'A scored report requires a separate non-dry-run command against the exact local dataset files and hashes.',
288
+ ],
289
+ };
290
+ }
291
+ function protocolRef(value, label, fallback) {
292
+ if (value === undefined || value === null) return fallback;
293
+ const normalized = typeof value === 'string' ? value.trim() : String(value);
294
+ if (normalized === '') throw new Error(`${label} must be a non-empty public ref`);
295
+ if (!PROTOCOL_REF_RE.test(normalized)) throw new Error(`${label} must be a lowercase public ref using letters, numbers, . _ : / @ + or -`);
296
+ return normalized;
297
+ }
298
+
299
+ function protocolPlanCategorySet(datasets) {
300
+ const categorySet = [];
301
+ const seen = new Set();
302
+ for (const row of datasets) {
303
+ for (const category of DATASET_TASK_CATEGORIES[row.id] ?? []) {
304
+ if (!seen.has(category)) {
305
+ seen.add(category);
306
+ categorySet.push(category);
307
+ }
308
+ }
309
+ }
310
+ return categorySet;
311
+ }
312
+
313
+ export function buildStandardBenchmarkProtocolPlan(options = {}) {
314
+ const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
315
+ const datasets = datasetPlanRows(options);
316
+ if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
317
+ const answererProvided = (options.answerer_ref ?? options.answererModelRef) !== undefined;
318
+ const judgeProvided = (options.judge_ref ?? options.judgeModelRef) !== undefined;
319
+ const answererRef = protocolRef(options.answerer_ref ?? options.answererModelRef, 'answerer_ref', DEFAULT_ANSWERER_MODEL_REF);
320
+ const judgeRef = protocolRef(options.judge_ref ?? options.judgeModelRef, 'judge_ref', DEFAULT_JUDGE_MODEL_REF);
321
+ const answerPromptRef = protocolRef(options.answer_prompt_ref ?? options.answerPromptRef, 'answer_prompt_ref', DEFAULT_ANSWER_PROMPT_REF);
322
+ const judgePromptRef = protocolRef(options.judge_prompt_ref ?? options.judgePromptRef, 'judge_prompt_ref', DEFAULT_JUDGE_PROMPT_REF);
323
+ const protocolRefValue = protocolRef(options.protocol_ref ?? options.protocolRef, 'protocol_ref', DEFAULT_PROTOCOL_REF);
324
+ const promptsFixed = answererProvided && judgeProvided;
325
+ return {
326
+ schema: STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA,
327
+ generated_at: options.generated_at ?? new Date().toISOString(),
328
+ package: {
329
+ name: 'enigma-memory',
330
+ version: '0.1.17',
331
+ },
332
+ public_safe: true,
333
+ protocol_plan: true,
334
+ dry_run: true,
335
+ top_k: topK,
336
+ category_set: protocolPlanCategorySet(datasets),
337
+ datasets_planned: datasets,
338
+ answerer: {
339
+ model_ref: answererRef,
340
+ temperature: 0,
341
+ max_tokens: 1024,
342
+ fixed: answererProvided,
343
+ },
344
+ judge: {
345
+ model_ref: judgeRef,
346
+ kind: 'llm-as-judge-or-exact-match-not-selected',
347
+ temperature: 0,
348
+ max_tokens: 512,
349
+ fixed: judgeProvided,
350
+ },
351
+ prompt_refs: [answerPromptRef, judgePromptRef],
352
+ protocol_refs: [protocolRefValue],
353
+ competitor_adapter_refs: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => `adapter:${adapter.id}@not-pinned`),
354
+ external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
355
+ ...adapter,
356
+ required_artifacts: [...adapter.required_artifacts],
357
+ adapter_ref: `adapter:${adapter.id}@not-pinned`,
358
+ })),
359
+ apples_to_apples_controls: applesToApplesControls(topK),
360
+ protocol_controls: {
361
+ same_answerer_model_for_all_rows: answererProvided,
362
+ same_judge_model_for_all_rows: judgeProvided,
363
+ same_prompts_for_all_rows: promptsFixed,
364
+ same_competitor_adapters_for_enigma_and_baselines: false,
365
+ answerer_model_fixed: answererProvided,
366
+ judge_model_fixed: judgeProvided,
367
+ prompts_fixed: promptsFixed,
368
+ temperature_fixed: false,
369
+ budget_caps_set: false,
370
+ },
371
+ cost_estimate_inputs: {
372
+ dataset_sample_limits: datasets.map((row) => ({ dataset: row.id, sample_limit: row.sample_limit })),
373
+ answerer_temperature: 0,
374
+ answerer_max_tokens: 1024,
375
+ judge_temperature: 0,
376
+ judge_max_tokens: 512,
377
+ max_retries: 0,
378
+ request_timeout_ms: null,
379
+ budget_cap_required_before_run: true,
380
+ budget_cap_set: false,
381
+ },
382
+ benchmark_boundaries: {
383
+ official_dataset_files_required: true,
384
+ credentials_required: false,
385
+ external_provider_calls: false,
386
+ llm_answer_accuracy_scored: false,
387
+ retrieval_evidence_proxy_scored: false,
388
+ raw_question_text_included: false,
389
+ raw_answer_text_included: false,
390
+ raw_conversation_text_included: false,
391
+ provider_deletion_claim: false,
392
+ model_forgetting_claim: false,
393
+ roi_or_provider_invoice_savings_claim: false,
394
+ compliance_certification_claim: false,
395
+ benchmark_leadership_claim: false,
396
+ },
397
+ protocol_boundaries: {
398
+ network_required: false,
399
+ provider_calls_made: false,
400
+ answers_generated: false,
401
+ judged: false,
402
+ competitor_adapters_run: false,
403
+ api_spend_possible: false,
404
+ },
405
+ command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
406
+ review_rules: [
407
+ 'report hash format: sha256:<hex> over the public-safe protocol-plan JSON bytes',
408
+ 'dataset ref, runner ref, package ref, environment ref, and verifier ref formats are pinned before a live run',
409
+ 'metric scope: this artifact plans the full-answer protocol only; it is not a score',
410
+ 'limitation: no provider answer-accuracy, competitor-performance, benchmark-leadership, ROI, provider-deletion, model-forgetting, or compliance claim is made',
411
+ ],
412
+ non_claims: [
413
+ 'This protocol plan is a readiness artifact: it records the planned full-answer protocol dimensions and does not execute it.',
414
+ 'No network is used, no provider APIs are called, no answers are generated, no answers are judged, and no competitor adapters are run.',
415
+ 'Answerer/judge model refs, prompt refs, and competitor adapter refs are public-safe references only; they are not credentials, model calls, prompts, or scores.',
416
+ 'A scored full-answer run requires a separate credentialed command with frozen models, prompts, evaluator, dataset manifest, and budget caps.',
417
+ ],
418
+ };
419
+ }
420
+
150
421
  function publicFileName(path) {
151
422
  return path === undefined || path === null ? undefined : basename(String(path));
152
423
  }
@@ -715,7 +986,7 @@ export function parseLocomoDataset(data, options = {}) {
715
986
  source_url: LOCOMO_SOURCE_URL,
716
987
  license: 'CC BY-NC 4.0',
717
988
  parser: 'conversation session turns as memory records; qa evidence labels mapped to dialog ids such as D1:3 and semicolon-separated labels',
718
- task_categories: ['multi-session QA', 'event summarization', 'multimodal generation over long conversations'],
989
+ task_categories: [...LOCOMO_TASK_CATEGORIES],
719
990
  records,
720
991
  queries,
721
992
  };
@@ -791,7 +1062,7 @@ export function parseLongMemEvalDataset(data, options = {}) {
791
1062
  source_url: LONGMEMEVAL_SOURCE_URLS,
792
1063
  license: 'See Hugging Face dataset card and upstream LongMemEval repository for the selected cleaned file.',
793
1064
  parser: 'haystack_sessions turns as memory records; has_answer:true turns and answer_session_ids are used as evidence labels; _abs ids are evaluated as abstention cases',
794
- task_categories: ['information extraction', 'multi-session reasoning', 'temporal reasoning', 'knowledge updates', 'abstention'],
1065
+ task_categories: [...LONGMEMEVAL_TASK_CATEGORIES],
795
1066
  records,
796
1067
  queries,
797
1068
  };
@@ -995,7 +1266,7 @@ function buildSuiteReport(datasetRows, topK, options) {
995
1266
  generated_at: options.generated_at ?? new Date().toISOString(),
996
1267
  package: {
997
1268
  name: 'enigma-memory',
998
- version: '0.1.15',
1269
+ version: '0.1.17',
999
1270
  },
1000
1271
  public_safe: true,
1001
1272
  top_k: topK,
@@ -1010,6 +1281,8 @@ function buildSuiteReport(datasetRows, topK, options) {
1010
1281
  'No provider APIs, hosted runtimes, competitor SDKs, or external accounts are called by this runner.',
1011
1282
  'Rows are local deterministic methods only; no third-party competitor scores or benchmark-leadership claims are emitted.',
1012
1283
  ],
1284
+ command_boundaries: offlineCommandBoundaries({ scoresIncluded: true, datasetFilesRead: true }),
1285
+ apples_to_apples_controls: applesToApplesControls(topK),
1013
1286
  benchmark_boundaries: {
1014
1287
  official_dataset_files_required: true,
1015
1288
  credentials_required: false,
@@ -1032,15 +1305,20 @@ function buildSuiteReport(datasetRows, topK, options) {
1032
1305
  enigma_relevance_fallback: 'falls back to all local candidates only when no enhanced relevance signal exists, then applies deterministic local ranking and --top-k',
1033
1306
  provider_api_used: false,
1034
1307
  llm_used: false,
1308
+ gold_labels_used_for_retrieval: false,
1035
1309
  },
1036
1310
  local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
1311
+ external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
1312
+ ...adapter,
1313
+ required_artifacts: [...adapter.required_artifacts],
1314
+ })),
1037
1315
  datasets: datasetRows,
1038
1316
  dataset_rows: datasetRows,
1039
1317
  };
1040
1318
  }
1041
1319
 
1042
1320
  function usage() {
1043
- return `Usage: node scripts/run-standard-memory-benchmarks.mjs [--locomo <path>] [--longmemeval <path>] [--max-locomo-qa <n>] [--max-longmemeval-items <n>] [--top-k <n>] [--out <path>]\n\nProduces schema ${STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA}. Raw question, answer, and conversation text are never written to the report. With --longmemeval and --max-longmemeval-items, the local top-level JSON array is streamed for hashing and only the requested sample items are parsed.`;
1321
+ return `Usage: node scripts/run-standard-memory-benchmarks.mjs [--locomo <path>] [--longmemeval <path>] [--max-locomo-qa <n>] [--max-longmemeval-items <n>] [--top-k <n>] [--out <path>] [--dry-run] [--protocol-plan [--answerer-ref <ref>] [--judge-ref <ref>] [--answer-prompt-ref <ref>] [--judge-prompt-ref <ref>] [--protocol-ref <ref>]]\n\nProduces schema ${STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA}. Raw question, answer, and conversation text are never written to the report. With --longmemeval and --max-longmemeval-items, the local top-level JSON array is streamed for hashing and only the requested sample items are parsed. Use --dry-run to print a public-safe offline execution plan without reading dataset files or producing scores. Use --protocol-plan to print a public-safe full-answer benchmark PROTOCOL plan (schema ${STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA}) recording the planned category set, top-k, answerer/judge model refs, prompt/protocol refs, competitor adapter refs, and cost-estimate inputs, with explicit network_required:false, provider_calls_made:false, answers_generated:false, and judged:false boundaries; it does not call providers, generate answers, judge answers, or run competitor adapters.`;
1044
1322
  }
1045
1323
 
1046
1324
  async function main() {
@@ -1049,7 +1327,11 @@ async function main() {
1049
1327
  console.log(usage());
1050
1328
  return;
1051
1329
  }
1052
- const report = await runStandardMemoryBenchmarkSuiteFromFiles(options);
1330
+ const report = options.protocol_plan
1331
+ ? buildStandardBenchmarkProtocolPlan(options)
1332
+ : options.dry_run
1333
+ ? buildStandardBenchmarkDryRunPlan(options)
1334
+ : await runStandardMemoryBenchmarkSuiteFromFiles(options);
1053
1335
  const serialized = `${JSON.stringify(report, null, 2)}\n`;
1054
1336
  if (options.out) {
1055
1337
  const outPath = resolve(options.out);