enigma-memory 0.1.15 → 0.1.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +70 -86
- package/apps/cli/bin/enigma.mjs +152 -15
- package/apps/native-host/README.md +20 -0
- package/deploy/SIMULATION.md +34 -38
- package/docs/benchmark-attestation-network.md +2 -2
- package/docs/benchmark-reproducibility.md +70 -9
- package/docs/blockchain-only-mechanisms.md +12 -0
- package/docs/browser-extension-install.md +9 -6
- package/docs/client-connectors.md +29 -55
- package/docs/demo-proof-network.md +3 -3
- package/docs/developer-ecosystem.md +207 -223
- package/docs/developer-proof-quickstart.md +3 -3
- package/docs/enigma-memory-ready-conformance.md +1 -1
- package/docs/hosted-cloud-product.md +31 -0
- package/docs/install-anywhere.md +68 -70
- package/docs/memory-benchmarks.md +21 -3
- package/docs/memory-drive-health-model.md +41 -0
- package/docs/proof-network-build-notes.md +2 -2
- package/docs/proof-network.md +90 -8
- package/docs/sdk-api.md +1 -1
- package/docs/solana-devnet-acceptance.md +1 -1
- package/docs/solana-proof-rail.md +1 -1
- package/package.json +7 -1
- package/packages/connectors/src/index.js +13 -0
- package/packages/hosted-cloud/src/index.js +470 -2
- package/packages/mcp-server/README.md +22 -0
- package/packages/mcp-server/src/index.js +1 -1
- package/packages/passport/src/index.js +730 -0
- package/packages/proof-network/src/index.js +139 -0
- package/scripts/build-benchmark-proof-release.mjs +119 -6
- package/scripts/build-cloudflare-token-policy.mjs +6 -2
- package/scripts/build-hosted-api-key-lifecycle.mjs +1 -1
- package/scripts/build-hosted-customer-lifecycle.mjs +23 -3
- package/scripts/build-installer-assets.mjs +1 -1
- package/scripts/build-production-handoff-packet.mjs +1 -1
- package/scripts/build-production-unblocker.mjs +409 -409
- package/scripts/build-production-workplan.mjs +3 -1
- package/scripts/build-proof-network-packet.mjs +1 -1
- package/scripts/check.mjs +3 -1
- package/scripts/cloudflare-ops.mjs +35 -0
- package/scripts/collect-hosted-backend-live-evidence.mjs +44 -2
- package/scripts/run-memory-benchmarks.mjs +5 -0
- package/scripts/run-standard-memory-benchmarks.mjs +287 -5
- package/scripts/stage-cloudflare-pages-artifact.mjs +145 -0
|
@@ -234,6 +234,8 @@ export function buildProductionWorkplan(inputs = {}, options = {}) {
|
|
|
234
234
|
const endpointRefs = missingHostedEndpointRefs(hosted);
|
|
235
235
|
const stateBlockers = hostedStateBlockers(hosted);
|
|
236
236
|
|
|
237
|
+
const finalDependencyCommand = 'npm run production:dependencies -- --goal-audit .enigma/goal-audit-current.json --release-audit .enigma/release-audit-current.json --worker-inspect .enigma/worker-inspect-result-current.json --whitepaper .enigma/whitepaper-claims-current.json --cloudflare-credentials .enigma/cloudflare-credentials-current.json --edge-deploy .enigma/edge-backend-deployment-current.json --edge-live .enigma/edge-backend-bootstrap-live-current.json --storage-bootstrap .enigma/cloudflare-storage-bootstrap-current.json';
|
|
238
|
+
|
|
237
239
|
const phases = [
|
|
238
240
|
makePhase({
|
|
239
241
|
id: 'cloudflare_credentials',
|
|
@@ -309,7 +311,7 @@ export function buildProductionWorkplan(inputs = {}, options = {}) {
|
|
|
309
311
|
owner: 'operator-or-reviewer',
|
|
310
312
|
prerequisites: ['cloudflare_credentials', 'cloudflare_worker_permission', 'hosted_backend_refs', 'operator_acceptance', 'release_gates'],
|
|
311
313
|
blockers: dependencyReport.launch_ready === true ? [] : ['launch_ready is false'],
|
|
312
|
-
commands: [release.next_command, staticSite.next_command, whitepaper.next_command, 'npm run production:goal-audit -- --site <public-site-dir> --domain enigmamemory.com --release-audit .enigma/release-audit-current.json',
|
|
314
|
+
commands: [release.next_command, staticSite.next_command, whitepaper.next_command, 'npm run production:goal-audit -- --site <public-site-dir> --domain enigmamemory.com --release-audit .enigma/release-audit-current.json', finalDependencyCommand],
|
|
313
315
|
evidence: [...release.evidence, ...staticSite.evidence, ...whitepaper.evidence],
|
|
314
316
|
details: { goal_complete: dependencyReport.goal_complete === true, launch_ready: dependencyReport.launch_ready === true },
|
|
315
317
|
}),
|
|
@@ -15,7 +15,7 @@ import {
|
|
|
15
15
|
validateProofNetworkPacket,
|
|
16
16
|
} from '../packages/proof-network/src/index.js';
|
|
17
17
|
|
|
18
|
-
export const PROOF_NETWORK_PACKET_RELEASE_TARGET = '0.1.
|
|
18
|
+
export const PROOF_NETWORK_PACKET_RELEASE_TARGET = '0.1.17';
|
|
19
19
|
|
|
20
20
|
const HASH_RE = /^(?:sha256:)?[a-f0-9]{64}$/iu;
|
|
21
21
|
const SECRET_VALUE_RE = /(?:Bearer\s+[A-Za-z0-9._~+/=-]{12,}|Basic\s+[A-Za-z0-9+/=-]{12,}|-----BEGIN [A-Z ]*PRIVATE KEY-----|https?:\/\/[^\s/@]+:[^\s/@]+@|sk-[A-Za-z0-9_-]{16,}|AKIA[0-9A-Z]{16}|\b(?:raw[\s_-]*memory|plaintext[\s_-]*prompts?|plain[\s_-]*text[\s_-]*prompts?|private[\s_-]*prompts?|provider[\s_-]*responses?|full[\s_-]*transcript|decrypted[\s_-]*memory|credentials?|secrets?|passwords?|private[\s_-]*keys?|api[\s_-]*key[\s_-]*(?:secret|material|value)|api[\s_-]*secrets?|access[\s_-]*tokens?|refresh[\s_-]*tokens?|token[\s_-]*values?|credential[\s_-]*material|tenant[\s_-]*names?)\b)/iu;
|
package/scripts/check.mjs
CHANGED
|
@@ -3,7 +3,9 @@ import path from 'node:path';
|
|
|
3
3
|
import { fileURLToPath, pathToFileURL } from 'node:url';
|
|
4
4
|
|
|
5
5
|
const root = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '..');
|
|
6
|
-
const packageJsonPath =
|
|
6
|
+
const packageJsonPath = process.env.ENIGMA_CHECK_PACKAGE_JSON_OVERRIDE
|
|
7
|
+
? path.resolve(process.env.ENIGMA_CHECK_PACKAGE_JSON_OVERRIDE)
|
|
8
|
+
: path.join(root, 'package.json');
|
|
7
9
|
const packageJson = JSON.parse(fs.readFileSync(packageJsonPath, 'utf8'));
|
|
8
10
|
const productionRoots = [
|
|
9
11
|
'packages/adapters/src/',
|
|
@@ -6,6 +6,7 @@ import { dirname, join } from 'node:path';
|
|
|
6
6
|
import { pathToFileURL } from 'node:url';
|
|
7
7
|
import { promisify } from 'node:util';
|
|
8
8
|
import { applyCloudflareSecretEnvFileFromArgv, CloudflareSecretEnvError } from './cloudflare-secret-env.mjs';
|
|
9
|
+
import { PUBLIC_SITE_SECURITY_RESULT_SCHEMA, validatePublicSiteSecurity } from './validate-public-site-security.mjs';
|
|
9
10
|
|
|
10
11
|
const execFile = promisify(execFileCallback);
|
|
11
12
|
|
|
@@ -46,6 +47,8 @@ Commands:
|
|
|
46
47
|
|
|
47
48
|
pages deploy --site <dir> --project-name <name> [--execute]
|
|
48
49
|
Without --execute, prints the exact Wrangler deploy plan only.
|
|
50
|
+
Dry-run output includes local public-site security validation.
|
|
51
|
+
--execute refuses artifacts with local security blockers before invoking Wrangler.
|
|
49
52
|
With --execute, runs Wrangler through npm exec/npx without printing the token.
|
|
50
53
|
|
|
51
54
|
pages verify --url <https-url> --project-name <name> [--domain <host>] \\
|
|
@@ -264,6 +267,35 @@ function redactPlanOutput(plan) {
|
|
|
264
267
|
return redactOperationalPayload(plan);
|
|
265
268
|
}
|
|
266
269
|
|
|
270
|
+
function redactPublicSiteSecurityBlocker(entry) {
|
|
271
|
+
return {
|
|
272
|
+
message: redactOperationalText(entry?.message ?? ''),
|
|
273
|
+
...(entry?.path === undefined ? {} : { path: redactOperationalText(String(entry.path)) }),
|
|
274
|
+
};
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
function publicSiteSecurityDeploySummary(result) {
|
|
278
|
+
return {
|
|
279
|
+
schema: PUBLIC_SITE_SECURITY_RESULT_SCHEMA,
|
|
280
|
+
ok: result?.ok === true,
|
|
281
|
+
status: result?.status ?? 'blocked',
|
|
282
|
+
blocker_count: Array.isArray(result?.blockers) ? result.blockers.length : 0,
|
|
283
|
+
blockers: Array.isArray(result?.blockers) ? result.blockers.map(redactPublicSiteSecurityBlocker) : [],
|
|
284
|
+
checked: redactOperationalPayload(result?.checked ?? {}),
|
|
285
|
+
claimBoundary: result?.claim_boundary ?? [],
|
|
286
|
+
};
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
function publicSiteSecurityErrorMessage(summary) {
|
|
290
|
+
const blockerText = summary.blockers
|
|
291
|
+
.slice(0, 5)
|
|
292
|
+
.map((entry) => (entry.path ? `${entry.path}: ${entry.message}` : entry.message))
|
|
293
|
+
.join('; ');
|
|
294
|
+
return blockerText.length > 0
|
|
295
|
+
? `public site security validation blocked Pages deploy: ${blockerText}`
|
|
296
|
+
: 'public site security validation blocked Pages deploy';
|
|
297
|
+
}
|
|
298
|
+
|
|
267
299
|
function parsePositiveInteger(value, name) {
|
|
268
300
|
const text = requireNonEmptyString(name, value);
|
|
269
301
|
if (!/^\d+$/.test(text)) throw new UsageError(`${name} must be a positive integer`);
|
|
@@ -1518,6 +1550,7 @@ export async function runCloudflareOpsCommand(command, {
|
|
|
1518
1550
|
|
|
1519
1551
|
if (command.kind === 'pages.deploy') {
|
|
1520
1552
|
const plan = buildWranglerPagesDeployPlan(command);
|
|
1553
|
+
const siteSecurity = publicSiteSecurityDeploySummary(await validatePublicSiteSecurity({ site: command.site }));
|
|
1521
1554
|
if (!command.execute) {
|
|
1522
1555
|
return {
|
|
1523
1556
|
json: {
|
|
@@ -1526,10 +1559,12 @@ export async function runCloudflareOpsCommand(command, {
|
|
|
1526
1559
|
dryRun: true,
|
|
1527
1560
|
execute: false,
|
|
1528
1561
|
plan: redactPlanOutput(plan),
|
|
1562
|
+
siteSecurity,
|
|
1529
1563
|
claimBoundary: 'Plan only; no Cloudflare Pages deployment was executed.',
|
|
1530
1564
|
},
|
|
1531
1565
|
};
|
|
1532
1566
|
}
|
|
1567
|
+
if (!siteSecurity.ok) throw new UsageError(publicSiteSecurityErrorMessage(siteSecurity));
|
|
1533
1568
|
const result = await execFileImpl(plan.command, plan.args, { shell: plan.usesShell === true, windowsHide: true, maxBuffer: 10 * 1024 * 1024, env });
|
|
1534
1569
|
return {
|
|
1535
1570
|
json: {
|
|
@@ -1,4 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
+
import https from 'node:https';
|
|
2
3
|
import { createHash } from 'node:crypto';
|
|
3
4
|
import { mkdir, readFile, writeFile } from 'node:fs/promises';
|
|
4
5
|
import { dirname, resolve } from 'node:path';
|
|
@@ -81,6 +82,42 @@ function redactProbeBody(value, path = 'probe.body') {
|
|
|
81
82
|
return value;
|
|
82
83
|
}
|
|
83
84
|
|
|
85
|
+
export function localSimulationLoopbackFetch(url, init = {}) {
|
|
86
|
+
const parsed = new URL(url);
|
|
87
|
+
const host = parsed.hostname.toLowerCase();
|
|
88
|
+
if (parsed.protocol !== 'https:' || (host !== 'sim.enigmamemory.com' && !host.endsWith('.sim.enigmamemory.com'))) {
|
|
89
|
+
throw new Error('--local-simulation-loopback only supports https://*.sim.enigmamemory.com simulation probes');
|
|
90
|
+
}
|
|
91
|
+
const request = {
|
|
92
|
+
hostname: '127.0.0.1',
|
|
93
|
+
port: parsed.port || 443,
|
|
94
|
+
path: `${parsed.pathname}${parsed.search}`,
|
|
95
|
+
method: init.method || 'GET',
|
|
96
|
+
headers: init.headers,
|
|
97
|
+
rejectUnauthorized: false,
|
|
98
|
+
servername: parsed.hostname,
|
|
99
|
+
};
|
|
100
|
+
return new Promise((resolve, reject) => {
|
|
101
|
+
const req = https.request(request, (res) => {
|
|
102
|
+
const chunks = [];
|
|
103
|
+
res.on('data', (chunk) => chunks.push(chunk));
|
|
104
|
+
res.on('end', () => {
|
|
105
|
+
const text = Buffer.concat(chunks).toString('utf8');
|
|
106
|
+
resolve({
|
|
107
|
+
ok: res.statusCode >= 200 && res.statusCode < 300,
|
|
108
|
+
status: res.statusCode,
|
|
109
|
+
statusText: res.statusMessage || '',
|
|
110
|
+
url,
|
|
111
|
+
redirected: false,
|
|
112
|
+
text: async () => text,
|
|
113
|
+
});
|
|
114
|
+
});
|
|
115
|
+
});
|
|
116
|
+
req.on('error', reject);
|
|
117
|
+
req.end();
|
|
118
|
+
});
|
|
119
|
+
}
|
|
120
|
+
|
|
84
121
|
async function fetchProbe(url, { fetchImpl = globalThis.fetch, observedAt }) {
|
|
85
122
|
if (typeof fetchImpl !== 'function') throw new Error('global fetch is not available in this Node runtime');
|
|
86
123
|
const response = await fetchImpl(url, {
|
|
@@ -189,6 +226,10 @@ function parseArgs(argv) {
|
|
|
189
226
|
}
|
|
190
227
|
if (!arg.startsWith('--')) throw new Error(`Unexpected argument: ${arg}`);
|
|
191
228
|
const name = arg.slice(2);
|
|
229
|
+
if (name === 'local-simulation-loopback') {
|
|
230
|
+
flags.set(name, true);
|
|
231
|
+
continue;
|
|
232
|
+
}
|
|
192
233
|
const value = argv[index + 1];
|
|
193
234
|
if (!value || value.startsWith('--')) throw new Error(`${arg} requires a value`);
|
|
194
235
|
flags.set(name, value);
|
|
@@ -198,7 +239,7 @@ function parseArgs(argv) {
|
|
|
198
239
|
}
|
|
199
240
|
|
|
200
241
|
function usage() {
|
|
201
|
-
return `Usage: node scripts/collect-hosted-backend-live-evidence.mjs --relay-url <https-base> --gateway-url <https-base> --refs-json <refs.json> --domain <domain> --environment-id <id> --cloud-provider <provider> --region <region> --owner <owner> --operator-decision go --operator-packet-ref <ref> --operator-approved-at <iso> --operator-approved-by <name> [--out <collection.json>] [--evidence-out <evidence.json>]\n\nCollects public HTTPS /livez and /readyz evidence for relay and gateway, then validates it with validate-hosted-backend-live. It never sends credentials and does not deploy infrastructure.\n`;
|
|
242
|
+
return `Usage: node scripts/collect-hosted-backend-live-evidence.mjs --relay-url <https-base> --gateway-url <https-base> --refs-json <refs.json> --domain <domain> --environment-id <id> --cloud-provider <provider> --region <region> --owner <owner> --operator-decision go --operator-packet-ref <ref> --operator-approved-at <iso> --operator-approved-by <name> [--out <collection.json>] [--evidence-out <evidence.json>] [--local-simulation-loopback]\n\nCollects public HTTPS /livez and /readyz evidence for relay and gateway, then validates it with validate-hosted-backend-live. It never sends credentials and does not deploy infrastructure. The --local-simulation-loopback flag is restricted to https://*.sim.enigmamemory.com local simulation probes with self-signed TLS and must not be used as production evidence.\n`;
|
|
202
243
|
}
|
|
203
244
|
|
|
204
245
|
async function runCli(argv = process.argv.slice(2), { fetchImpl = globalThis.fetch } = {}) {
|
|
@@ -212,6 +253,7 @@ async function runCli(argv = process.argv.slice(2), { fetchImpl = globalThis.fet
|
|
|
212
253
|
const refs = await readJsonFile(refsPath);
|
|
213
254
|
const environment = await maybeReadJsonFile(readFlag(flags, 'environment-json'));
|
|
214
255
|
const operatorAcceptance = await maybeReadJsonFile(readFlag(flags, 'operator-acceptance-json'));
|
|
256
|
+
const selectedFetchImpl = flags.get('local-simulation-loopback') === true ? localSimulationLoopbackFetch : fetchImpl;
|
|
215
257
|
const collection = await collectHostedBackendLiveEvidence({
|
|
216
258
|
relayBaseUrl: readFlag(flags, 'relay-url'),
|
|
217
259
|
gatewayBaseUrl: readFlag(flags, 'gateway-url'),
|
|
@@ -233,7 +275,7 @@ async function runCli(argv = process.argv.slice(2), { fetchImpl = globalThis.fet
|
|
|
233
275
|
operatorApprovedAt: readFlag(flags, 'operator-approved-at'),
|
|
234
276
|
operatorApprovedBy: readFlag(flags, 'operator-approved-by'),
|
|
235
277
|
observed_at: readFlag(flags, 'observed-at') ?? new Date().toISOString(),
|
|
236
|
-
fetchImpl,
|
|
278
|
+
fetchImpl: selectedFetchImpl,
|
|
237
279
|
});
|
|
238
280
|
const collectionJson = `${JSON.stringify(collection, null, 2)}\n`;
|
|
239
281
|
const evidenceJson = `${JSON.stringify(collection.evidence, null, 2)}\n`;
|
|
@@ -832,6 +832,11 @@ export function runMemoryBenchmarkSuite(options = {}) {
|
|
|
832
832
|
credentials_required: false,
|
|
833
833
|
external_downloads_required: false,
|
|
834
834
|
external_provider_calls: false,
|
|
835
|
+
llm_answer_accuracy_scored: false,
|
|
836
|
+
provider_api_calls_made: false,
|
|
837
|
+
api_spend_possible: false,
|
|
838
|
+
mem0_adapter_run: false,
|
|
839
|
+
external_competitor_adapters_run: false,
|
|
835
840
|
raw_private_memory_plaintext_included: false,
|
|
836
841
|
provider_deletion_claim: false,
|
|
837
842
|
model_forgetting_claim: false,
|
|
@@ -37,12 +37,41 @@ export const STANDARD_MEMORY_BENCHMARK_METHODS = Object.freeze([
|
|
|
37
37
|
}),
|
|
38
38
|
]);
|
|
39
39
|
|
|
40
|
+
export const STANDARD_EXTERNAL_COMPETITOR_ADAPTERS = Object.freeze([
|
|
41
|
+
Object.freeze({
|
|
42
|
+
id: 'mem0',
|
|
43
|
+
name: 'Mem0',
|
|
44
|
+
status: 'not_run_requires_credentials_or_runtime',
|
|
45
|
+
target_type: 'external_adapter',
|
|
46
|
+
can_run_in_this_harness: false,
|
|
47
|
+
scores_included: false,
|
|
48
|
+
required_artifacts: Object.freeze([
|
|
49
|
+
'Mem0 platform credentials or open-source runtime',
|
|
50
|
+
'Pinned Mem0 SDK/package versions',
|
|
51
|
+
'Fixed extraction, update, retrieval, reset, model, and tool policy',
|
|
52
|
+
'Same reviewed dataset manifest, split, top-k, and scorer as Enigma rows',
|
|
53
|
+
]),
|
|
54
|
+
official_doc: 'https://docs.mem0.ai/',
|
|
55
|
+
boundary_reason: 'The standard runner has no Mem0 credentials, SDK/runtime, fixed memory loop, reset policy, model/tool environment, or reviewed adapter scorer, so no Mem0 score is produced.',
|
|
56
|
+
}),
|
|
57
|
+
]);
|
|
58
|
+
|
|
40
59
|
const LOCOMO_SOURCE_URL = 'https://raw.githubusercontent.com/snap-research/locomo/main/data/locomo10.json';
|
|
41
60
|
const LONGMEMEVAL_SOURCE_URLS = Object.freeze([
|
|
42
61
|
'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_oracle.json',
|
|
43
62
|
'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_s_cleaned.json',
|
|
44
63
|
'https://huggingface.co/datasets/xiaowu0162/longmemeval-cleaned/resolve/main/longmemeval_m_cleaned.json',
|
|
45
64
|
]);
|
|
65
|
+
const LOCOMO_TASK_CATEGORIES = Object.freeze(['multi-session QA', 'event summarization', 'multimodal generation over long conversations']);
|
|
66
|
+
const LONGMEMEVAL_TASK_CATEGORIES = Object.freeze(['information extraction', 'multi-session reasoning', 'temporal reasoning', 'knowledge updates', 'abstention']);
|
|
67
|
+
const DATASET_TASK_CATEGORIES = Object.freeze({ locomo: LOCOMO_TASK_CATEGORIES, longmemeval: LONGMEMEVAL_TASK_CATEGORIES });
|
|
68
|
+
export const STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA = 'enigma.standard_memory_benchmark_protocol_plan.v1';
|
|
69
|
+
const PROTOCOL_REF_RE = /^[a-z0-9][a-z0-9._:/@+-]{2,191}$/u;
|
|
70
|
+
const DEFAULT_ANSWERER_MODEL_REF = 'model:answerer-not-selected';
|
|
71
|
+
const DEFAULT_JUDGE_MODEL_REF = 'model:judge-not-selected';
|
|
72
|
+
const DEFAULT_ANSWER_PROMPT_REF = 'prompt:standard-answer@not-pinned';
|
|
73
|
+
const DEFAULT_JUDGE_PROMPT_REF = 'prompt:standard-judge@not-pinned';
|
|
74
|
+
const DEFAULT_PROTOCOL_REF = 'protocol:apples-to-apples-full-answer@not-pinned';
|
|
46
75
|
|
|
47
76
|
const QUERY_RELEVANCE_STOPWORDS = new Set([
|
|
48
77
|
'about',
|
|
@@ -121,6 +150,25 @@ function parseArgs(argv = process.argv.slice(2)) {
|
|
|
121
150
|
} else if (arg === '--out') {
|
|
122
151
|
options.out = requiredFlagValue(argv, index, arg);
|
|
123
152
|
index += 1;
|
|
153
|
+
} else if (arg === '--protocol-plan') {
|
|
154
|
+
options.protocol_plan = true;
|
|
155
|
+
} else if (arg === '--answerer-ref') {
|
|
156
|
+
options.answerer_ref = requiredFlagValue(argv, index, arg);
|
|
157
|
+
index += 1;
|
|
158
|
+
} else if (arg === '--judge-ref') {
|
|
159
|
+
options.judge_ref = requiredFlagValue(argv, index, arg);
|
|
160
|
+
index += 1;
|
|
161
|
+
} else if (arg === '--answer-prompt-ref') {
|
|
162
|
+
options.answer_prompt_ref = requiredFlagValue(argv, index, arg);
|
|
163
|
+
index += 1;
|
|
164
|
+
} else if (arg === '--judge-prompt-ref') {
|
|
165
|
+
options.judge_prompt_ref = requiredFlagValue(argv, index, arg);
|
|
166
|
+
index += 1;
|
|
167
|
+
} else if (arg === '--protocol-ref') {
|
|
168
|
+
options.protocol_ref = requiredFlagValue(argv, index, arg);
|
|
169
|
+
index += 1;
|
|
170
|
+
} else if (arg === '--dry-run') {
|
|
171
|
+
options.dry_run = true;
|
|
124
172
|
} else if (arg === '--help' || arg === '-h') {
|
|
125
173
|
options.help = true;
|
|
126
174
|
} else {
|
|
@@ -147,6 +195,229 @@ function optionalPositiveInteger(value, name) {
|
|
|
147
195
|
return positiveInteger(value, name);
|
|
148
196
|
}
|
|
149
197
|
|
|
198
|
+
function datasetPlanRows(options) {
|
|
199
|
+
const rows = [];
|
|
200
|
+
if (options.locomo !== undefined || options.locomoPath !== undefined) {
|
|
201
|
+
rows.push({
|
|
202
|
+
id: 'locomo',
|
|
203
|
+
label: 'LoCoMo',
|
|
204
|
+
local_file_name: publicFileName(options.locomo ?? options.locomoPath),
|
|
205
|
+
source_url: LOCOMO_SOURCE_URL,
|
|
206
|
+
license: 'CC BY-NC 4.0',
|
|
207
|
+
sample_limit: optionalPositiveInteger(options.max_locomo_qa ?? options.maxLocomoQa, 'max_locomo_qa') ?? null,
|
|
208
|
+
parser: 'conversation session turns as memory records; qa evidence labels score support only',
|
|
209
|
+
});
|
|
210
|
+
}
|
|
211
|
+
if (options.longmemeval !== undefined || options.longmemevalPath !== undefined || options.longMemEvalPath !== undefined) {
|
|
212
|
+
rows.push({
|
|
213
|
+
id: 'longmemeval',
|
|
214
|
+
label: 'LongMemEval',
|
|
215
|
+
local_file_name: publicFileName(options.longmemeval ?? options.longmemevalPath ?? options.longMemEvalPath),
|
|
216
|
+
source_url: LONGMEMEVAL_SOURCE_URLS,
|
|
217
|
+
license: 'Review upstream Hugging Face dataset card and LongMemEval repository terms.',
|
|
218
|
+
sample_limit: optionalPositiveInteger(options.max_longmemeval_items ?? options.maxLongMemEvalItems, 'max_longmemeval_items') ?? null,
|
|
219
|
+
parser: 'haystack_sessions turns as memory records; answer-session labels score support only',
|
|
220
|
+
});
|
|
221
|
+
}
|
|
222
|
+
return rows;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
function offlineCommandBoundaries({ scoresIncluded, datasetFilesRead }) {
|
|
226
|
+
return {
|
|
227
|
+
deterministic_offline_runner: true,
|
|
228
|
+
dataset_files_read_from_local_disk: datasetFilesRead,
|
|
229
|
+
network_calls_made: false,
|
|
230
|
+
provider_api_calls_made: false,
|
|
231
|
+
api_spend_possible: false,
|
|
232
|
+
hosted_memory_service_called: false,
|
|
233
|
+
external_competitor_adapters_run: false,
|
|
234
|
+
mem0_adapter_run: false,
|
|
235
|
+
llm_used: false,
|
|
236
|
+
llm_answer_accuracy_scored: false,
|
|
237
|
+
retrieval_evidence_proxy_scored: scoresIncluded,
|
|
238
|
+
benchmark_scores_included: scoresIncluded,
|
|
239
|
+
raw_question_text_included: false,
|
|
240
|
+
raw_answer_text_included: false,
|
|
241
|
+
raw_conversation_text_included: false,
|
|
242
|
+
gold_labels_used_for_retrieval: false,
|
|
243
|
+
gold_labels_used_for_scoring: scoresIncluded,
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
|
|
247
|
+
function applesToApplesControls(topK) {
|
|
248
|
+
return {
|
|
249
|
+
same_top_k_for_all_methods: true,
|
|
250
|
+
top_k: topK,
|
|
251
|
+
same_parser_per_dataset: true,
|
|
252
|
+
same_local_records_per_dataset: true,
|
|
253
|
+
same_gold_evidence_labels_per_dataset_for_scoring_only: true,
|
|
254
|
+
local_deterministic_methods_only: true,
|
|
255
|
+
provider_runtime_fixed: false,
|
|
256
|
+
competitor_runtime_fixed: false,
|
|
257
|
+
answer_generator_fixed: false,
|
|
258
|
+
evaluator_model_fixed: false,
|
|
259
|
+
};
|
|
260
|
+
}
|
|
261
|
+
|
|
262
|
+
export function buildStandardBenchmarkDryRunPlan(options = {}) {
|
|
263
|
+
const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
|
|
264
|
+
const datasets = datasetPlanRows(options);
|
|
265
|
+
if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
|
|
266
|
+
return {
|
|
267
|
+
schema: 'enigma.standard_memory_benchmark_plan.v1',
|
|
268
|
+
generated_at: options.generated_at ?? new Date().toISOString(),
|
|
269
|
+
package: {
|
|
270
|
+
name: 'enigma-memory',
|
|
271
|
+
version: '0.1.17',
|
|
272
|
+
},
|
|
273
|
+
public_safe: true,
|
|
274
|
+
dry_run: true,
|
|
275
|
+
top_k: topK,
|
|
276
|
+
datasets_planned: datasets,
|
|
277
|
+
local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
|
|
278
|
+
external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
|
|
279
|
+
...adapter,
|
|
280
|
+
required_artifacts: [...adapter.required_artifacts],
|
|
281
|
+
})),
|
|
282
|
+
command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
|
|
283
|
+
apples_to_apples_controls: applesToApplesControls(topK),
|
|
284
|
+
non_claims: [
|
|
285
|
+
'This dry run does not read dataset files and produces no benchmark score.',
|
|
286
|
+
'No provider APIs, hosted memory services, Mem0 runtime, competitor SDKs, LLM generators, or evaluator models are called.',
|
|
287
|
+
'A scored report requires a separate non-dry-run command against the exact local dataset files and hashes.',
|
|
288
|
+
],
|
|
289
|
+
};
|
|
290
|
+
}
|
|
291
|
+
function protocolRef(value, label, fallback) {
|
|
292
|
+
if (value === undefined || value === null) return fallback;
|
|
293
|
+
const normalized = typeof value === 'string' ? value.trim() : String(value);
|
|
294
|
+
if (normalized === '') throw new Error(`${label} must be a non-empty public ref`);
|
|
295
|
+
if (!PROTOCOL_REF_RE.test(normalized)) throw new Error(`${label} must be a lowercase public ref using letters, numbers, . _ : / @ + or -`);
|
|
296
|
+
return normalized;
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
function protocolPlanCategorySet(datasets) {
|
|
300
|
+
const categorySet = [];
|
|
301
|
+
const seen = new Set();
|
|
302
|
+
for (const row of datasets) {
|
|
303
|
+
for (const category of DATASET_TASK_CATEGORIES[row.id] ?? []) {
|
|
304
|
+
if (!seen.has(category)) {
|
|
305
|
+
seen.add(category);
|
|
306
|
+
categorySet.push(category);
|
|
307
|
+
}
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
return categorySet;
|
|
311
|
+
}
|
|
312
|
+
|
|
313
|
+
export function buildStandardBenchmarkProtocolPlan(options = {}) {
|
|
314
|
+
const topK = optionalPositiveInteger(options.top_k ?? options.topK, 'top_k') ?? 5;
|
|
315
|
+
const datasets = datasetPlanRows(options);
|
|
316
|
+
if (datasets.length === 0) throw new Error('Provide --locomo <path> and/or --longmemeval <path>');
|
|
317
|
+
const answererProvided = (options.answerer_ref ?? options.answererModelRef) !== undefined;
|
|
318
|
+
const judgeProvided = (options.judge_ref ?? options.judgeModelRef) !== undefined;
|
|
319
|
+
const answererRef = protocolRef(options.answerer_ref ?? options.answererModelRef, 'answerer_ref', DEFAULT_ANSWERER_MODEL_REF);
|
|
320
|
+
const judgeRef = protocolRef(options.judge_ref ?? options.judgeModelRef, 'judge_ref', DEFAULT_JUDGE_MODEL_REF);
|
|
321
|
+
const answerPromptRef = protocolRef(options.answer_prompt_ref ?? options.answerPromptRef, 'answer_prompt_ref', DEFAULT_ANSWER_PROMPT_REF);
|
|
322
|
+
const judgePromptRef = protocolRef(options.judge_prompt_ref ?? options.judgePromptRef, 'judge_prompt_ref', DEFAULT_JUDGE_PROMPT_REF);
|
|
323
|
+
const protocolRefValue = protocolRef(options.protocol_ref ?? options.protocolRef, 'protocol_ref', DEFAULT_PROTOCOL_REF);
|
|
324
|
+
const promptsFixed = answererProvided && judgeProvided;
|
|
325
|
+
return {
|
|
326
|
+
schema: STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA,
|
|
327
|
+
generated_at: options.generated_at ?? new Date().toISOString(),
|
|
328
|
+
package: {
|
|
329
|
+
name: 'enigma-memory',
|
|
330
|
+
version: '0.1.17',
|
|
331
|
+
},
|
|
332
|
+
public_safe: true,
|
|
333
|
+
protocol_plan: true,
|
|
334
|
+
dry_run: true,
|
|
335
|
+
top_k: topK,
|
|
336
|
+
category_set: protocolPlanCategorySet(datasets),
|
|
337
|
+
datasets_planned: datasets,
|
|
338
|
+
answerer: {
|
|
339
|
+
model_ref: answererRef,
|
|
340
|
+
temperature: 0,
|
|
341
|
+
max_tokens: 1024,
|
|
342
|
+
fixed: answererProvided,
|
|
343
|
+
},
|
|
344
|
+
judge: {
|
|
345
|
+
model_ref: judgeRef,
|
|
346
|
+
kind: 'llm-as-judge-or-exact-match-not-selected',
|
|
347
|
+
temperature: 0,
|
|
348
|
+
max_tokens: 512,
|
|
349
|
+
fixed: judgeProvided,
|
|
350
|
+
},
|
|
351
|
+
prompt_refs: [answerPromptRef, judgePromptRef],
|
|
352
|
+
protocol_refs: [protocolRefValue],
|
|
353
|
+
competitor_adapter_refs: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => `adapter:${adapter.id}@not-pinned`),
|
|
354
|
+
external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
|
|
355
|
+
...adapter,
|
|
356
|
+
required_artifacts: [...adapter.required_artifacts],
|
|
357
|
+
adapter_ref: `adapter:${adapter.id}@not-pinned`,
|
|
358
|
+
})),
|
|
359
|
+
apples_to_apples_controls: applesToApplesControls(topK),
|
|
360
|
+
protocol_controls: {
|
|
361
|
+
same_answerer_model_for_all_rows: answererProvided,
|
|
362
|
+
same_judge_model_for_all_rows: judgeProvided,
|
|
363
|
+
same_prompts_for_all_rows: promptsFixed,
|
|
364
|
+
same_competitor_adapters_for_enigma_and_baselines: false,
|
|
365
|
+
answerer_model_fixed: answererProvided,
|
|
366
|
+
judge_model_fixed: judgeProvided,
|
|
367
|
+
prompts_fixed: promptsFixed,
|
|
368
|
+
temperature_fixed: false,
|
|
369
|
+
budget_caps_set: false,
|
|
370
|
+
},
|
|
371
|
+
cost_estimate_inputs: {
|
|
372
|
+
dataset_sample_limits: datasets.map((row) => ({ dataset: row.id, sample_limit: row.sample_limit })),
|
|
373
|
+
answerer_temperature: 0,
|
|
374
|
+
answerer_max_tokens: 1024,
|
|
375
|
+
judge_temperature: 0,
|
|
376
|
+
judge_max_tokens: 512,
|
|
377
|
+
max_retries: 0,
|
|
378
|
+
request_timeout_ms: null,
|
|
379
|
+
budget_cap_required_before_run: true,
|
|
380
|
+
budget_cap_set: false,
|
|
381
|
+
},
|
|
382
|
+
benchmark_boundaries: {
|
|
383
|
+
official_dataset_files_required: true,
|
|
384
|
+
credentials_required: false,
|
|
385
|
+
external_provider_calls: false,
|
|
386
|
+
llm_answer_accuracy_scored: false,
|
|
387
|
+
retrieval_evidence_proxy_scored: false,
|
|
388
|
+
raw_question_text_included: false,
|
|
389
|
+
raw_answer_text_included: false,
|
|
390
|
+
raw_conversation_text_included: false,
|
|
391
|
+
provider_deletion_claim: false,
|
|
392
|
+
model_forgetting_claim: false,
|
|
393
|
+
roi_or_provider_invoice_savings_claim: false,
|
|
394
|
+
compliance_certification_claim: false,
|
|
395
|
+
benchmark_leadership_claim: false,
|
|
396
|
+
},
|
|
397
|
+
protocol_boundaries: {
|
|
398
|
+
network_required: false,
|
|
399
|
+
provider_calls_made: false,
|
|
400
|
+
answers_generated: false,
|
|
401
|
+
judged: false,
|
|
402
|
+
competitor_adapters_run: false,
|
|
403
|
+
api_spend_possible: false,
|
|
404
|
+
},
|
|
405
|
+
command_boundaries: offlineCommandBoundaries({ scoresIncluded: false, datasetFilesRead: false }),
|
|
406
|
+
review_rules: [
|
|
407
|
+
'report hash format: sha256:<hex> over the public-safe protocol-plan JSON bytes',
|
|
408
|
+
'dataset ref, runner ref, package ref, environment ref, and verifier ref formats are pinned before a live run',
|
|
409
|
+
'metric scope: this artifact plans the full-answer protocol only; it is not a score',
|
|
410
|
+
'limitation: no provider answer-accuracy, competitor-performance, benchmark-leadership, ROI, provider-deletion, model-forgetting, or compliance claim is made',
|
|
411
|
+
],
|
|
412
|
+
non_claims: [
|
|
413
|
+
'This protocol plan is a readiness artifact: it records the planned full-answer protocol dimensions and does not execute it.',
|
|
414
|
+
'No network is used, no provider APIs are called, no answers are generated, no answers are judged, and no competitor adapters are run.',
|
|
415
|
+
'Answerer/judge model refs, prompt refs, and competitor adapter refs are public-safe references only; they are not credentials, model calls, prompts, or scores.',
|
|
416
|
+
'A scored full-answer run requires a separate credentialed command with frozen models, prompts, evaluator, dataset manifest, and budget caps.',
|
|
417
|
+
],
|
|
418
|
+
};
|
|
419
|
+
}
|
|
420
|
+
|
|
150
421
|
function publicFileName(path) {
|
|
151
422
|
return path === undefined || path === null ? undefined : basename(String(path));
|
|
152
423
|
}
|
|
@@ -715,7 +986,7 @@ export function parseLocomoDataset(data, options = {}) {
|
|
|
715
986
|
source_url: LOCOMO_SOURCE_URL,
|
|
716
987
|
license: 'CC BY-NC 4.0',
|
|
717
988
|
parser: 'conversation session turns as memory records; qa evidence labels mapped to dialog ids such as D1:3 and semicolon-separated labels',
|
|
718
|
-
task_categories: [
|
|
989
|
+
task_categories: [...LOCOMO_TASK_CATEGORIES],
|
|
719
990
|
records,
|
|
720
991
|
queries,
|
|
721
992
|
};
|
|
@@ -791,7 +1062,7 @@ export function parseLongMemEvalDataset(data, options = {}) {
|
|
|
791
1062
|
source_url: LONGMEMEVAL_SOURCE_URLS,
|
|
792
1063
|
license: 'See Hugging Face dataset card and upstream LongMemEval repository for the selected cleaned file.',
|
|
793
1064
|
parser: 'haystack_sessions turns as memory records; has_answer:true turns and answer_session_ids are used as evidence labels; _abs ids are evaluated as abstention cases',
|
|
794
|
-
task_categories: [
|
|
1065
|
+
task_categories: [...LONGMEMEVAL_TASK_CATEGORIES],
|
|
795
1066
|
records,
|
|
796
1067
|
queries,
|
|
797
1068
|
};
|
|
@@ -995,7 +1266,7 @@ function buildSuiteReport(datasetRows, topK, options) {
|
|
|
995
1266
|
generated_at: options.generated_at ?? new Date().toISOString(),
|
|
996
1267
|
package: {
|
|
997
1268
|
name: 'enigma-memory',
|
|
998
|
-
version: '0.1.
|
|
1269
|
+
version: '0.1.17',
|
|
999
1270
|
},
|
|
1000
1271
|
public_safe: true,
|
|
1001
1272
|
top_k: topK,
|
|
@@ -1010,6 +1281,8 @@ function buildSuiteReport(datasetRows, topK, options) {
|
|
|
1010
1281
|
'No provider APIs, hosted runtimes, competitor SDKs, or external accounts are called by this runner.',
|
|
1011
1282
|
'Rows are local deterministic methods only; no third-party competitor scores or benchmark-leadership claims are emitted.',
|
|
1012
1283
|
],
|
|
1284
|
+
command_boundaries: offlineCommandBoundaries({ scoresIncluded: true, datasetFilesRead: true }),
|
|
1285
|
+
apples_to_apples_controls: applesToApplesControls(topK),
|
|
1013
1286
|
benchmark_boundaries: {
|
|
1014
1287
|
official_dataset_files_required: true,
|
|
1015
1288
|
credentials_required: false,
|
|
@@ -1032,15 +1305,20 @@ function buildSuiteReport(datasetRows, topK, options) {
|
|
|
1032
1305
|
enigma_relevance_fallback: 'falls back to all local candidates only when no enhanced relevance signal exists, then applies deterministic local ranking and --top-k',
|
|
1033
1306
|
provider_api_used: false,
|
|
1034
1307
|
llm_used: false,
|
|
1308
|
+
gold_labels_used_for_retrieval: false,
|
|
1035
1309
|
},
|
|
1036
1310
|
local_methods: STANDARD_MEMORY_BENCHMARK_METHODS.map((method) => ({ ...method })),
|
|
1311
|
+
external_competitor_adapters: STANDARD_EXTERNAL_COMPETITOR_ADAPTERS.map((adapter) => ({
|
|
1312
|
+
...adapter,
|
|
1313
|
+
required_artifacts: [...adapter.required_artifacts],
|
|
1314
|
+
})),
|
|
1037
1315
|
datasets: datasetRows,
|
|
1038
1316
|
dataset_rows: datasetRows,
|
|
1039
1317
|
};
|
|
1040
1318
|
}
|
|
1041
1319
|
|
|
1042
1320
|
function usage() {
|
|
1043
|
-
return `Usage: node scripts/run-standard-memory-benchmarks.mjs [--locomo <path>] [--longmemeval <path>] [--max-locomo-qa <n>] [--max-longmemeval-items <n>] [--top-k <n>] [--out <path>]\n\nProduces schema ${STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA}. Raw question, answer, and conversation text are never written to the report. With --longmemeval and --max-longmemeval-items, the local top-level JSON array is streamed for hashing and only the requested sample items are parsed.`;
|
|
1321
|
+
return `Usage: node scripts/run-standard-memory-benchmarks.mjs [--locomo <path>] [--longmemeval <path>] [--max-locomo-qa <n>] [--max-longmemeval-items <n>] [--top-k <n>] [--out <path>] [--dry-run] [--protocol-plan [--answerer-ref <ref>] [--judge-ref <ref>] [--answer-prompt-ref <ref>] [--judge-prompt-ref <ref>] [--protocol-ref <ref>]]\n\nProduces schema ${STANDARD_MEMORY_BENCHMARK_SUITE_SCHEMA}. Raw question, answer, and conversation text are never written to the report. With --longmemeval and --max-longmemeval-items, the local top-level JSON array is streamed for hashing and only the requested sample items are parsed. Use --dry-run to print a public-safe offline execution plan without reading dataset files or producing scores. Use --protocol-plan to print a public-safe full-answer benchmark PROTOCOL plan (schema ${STANDARD_MEMORY_BENCHMARK_PROTOCOL_PLAN_SCHEMA}) recording the planned category set, top-k, answerer/judge model refs, prompt/protocol refs, competitor adapter refs, and cost-estimate inputs, with explicit network_required:false, provider_calls_made:false, answers_generated:false, and judged:false boundaries; it does not call providers, generate answers, judge answers, or run competitor adapters.`;
|
|
1044
1322
|
}
|
|
1045
1323
|
|
|
1046
1324
|
async function main() {
|
|
@@ -1049,7 +1327,11 @@ async function main() {
|
|
|
1049
1327
|
console.log(usage());
|
|
1050
1328
|
return;
|
|
1051
1329
|
}
|
|
1052
|
-
const report =
|
|
1330
|
+
const report = options.protocol_plan
|
|
1331
|
+
? buildStandardBenchmarkProtocolPlan(options)
|
|
1332
|
+
: options.dry_run
|
|
1333
|
+
? buildStandardBenchmarkDryRunPlan(options)
|
|
1334
|
+
: await runStandardMemoryBenchmarkSuiteFromFiles(options);
|
|
1053
1335
|
const serialized = `${JSON.stringify(report, null, 2)}\n`;
|
|
1054
1336
|
if (options.out) {
|
|
1055
1337
|
const outPath = resolve(options.out);
|