codemem 0.39.1 → 0.40.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,12 +1,12 @@
1
1
  #!/usr/bin/env node
2
- import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, REF_BACKFILL_JOB, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildAuthHeaders, buildBaseUrl, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, compareMemoryRoleReports, connect, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getExtractionBenchmarkProfile, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, initDatabase, isEmbeddingDisabled, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, planReplicationOpsAgePrune, projectMatchesFilter, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, setPeerProjectFilter, stripJsonComments, stripPrivateObj, stripTrailingCommas, syncPassPreflight, updatePeerAddresses, vacuumDatabase, writeCodememConfigFile } from "@codemem/core";
2
+ import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, REF_BACKFILL_JOB, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildAuthHeaders, buildBaseUrl, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, clonePromptPackAttempt, compareMemoryRoleReports, connect, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, estimateExtractionModelCost, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getAttributionDiagnostics, getExtractionBenchmarkProfile, getExtractionModelPricing, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, initDatabase, isEmbeddingDisabled, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, planReplicationOpsAgePrune, projectMatchesFilter, promptPackArtifactFingerprint, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, recordPromptPackArtifacts, recordPromptPackTerminal, recordRetrievalSurface, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, resolveRetrievalSession, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, scoreExtractionBenchmarkOutput, setPeerProjectFilter, stripJsonComments, stripPrivateObj, stripTrailingCommas, syncPassPreflight, tryUpdateRetrievalDelivery, updatePeerAddresses, vacuumDatabase, writeCodememConfigFile } from "@codemem/core";
3
3
  import { Command, Option } from "commander";
4
4
  import omelette from "omelette";
5
+ import { randomInt, randomUUID } from "node:crypto";
5
6
  import { appendFileSync, copyFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, rmdirSync, statSync, unlinkSync, writeFileSync } from "node:fs";
6
7
  import { homedir, networkInterfaces } from "node:os";
7
- import { dirname, isAbsolute, join, relative, resolve, sep } from "node:path";
8
+ import { dirname, isAbsolute, join, posix, relative, resolve, sep, win32 } from "node:path";
8
9
  import { styleText } from "node:util";
9
- import { randomInt, randomUUID } from "node:crypto";
10
10
  import * as p from "@clack/prompts";
11
11
  import { serve } from "@hono/node-server";
12
12
  import { execFileSync, spawn, spawnSync } from "node:child_process";
@@ -287,11 +287,77 @@ function queryByFile(dbPath, relativePath, project, limit) {
287
287
  function resolveProject$1(payload) {
288
288
  return resolveHookProject(typeof payload.cwd === "string" ? payload.cwd : null, payload.project);
289
289
  }
290
+ function sourceSessionId(payload) {
291
+ const value = payload.session_id;
292
+ return typeof value === "string" && value.trim() ? value.trim() : null;
293
+ }
290
294
  async function buildClaudeFileContext(payload, opts, deps = {}) {
291
295
  if (envTruthy$3(process.env.CODEMEM_PLUGIN_IGNORE)) return continueResult$2();
292
- if (!envNotDisabled$2(process.env.CODEMEM_FILE_CONTEXT || "1")) return continueResult$2();
293
296
  const filePath = extractFilePath(payload);
294
297
  if (!filePath) return continueResult$2();
298
+ const startedAt = (deps.now ?? (() => /* @__PURE__ */ new Date()))();
299
+ const attemptId = (deps.createAttemptId ?? randomUUID)();
300
+ const resolveDb = deps.resolveDb ?? resolveDbPath;
301
+ let resolvedDbPath = null;
302
+ let activeStore = null;
303
+ const getStore = () => {
304
+ resolvedDbPath ??= resolveDb(resolveDbOpt(opts));
305
+ activeStore ??= (deps.createStore ?? ((dbPath) => new MemoryStore(dbPath)))(resolvedDbPath);
306
+ return activeStore;
307
+ };
308
+ const finish = (result) => {
309
+ try {
310
+ activeStore?.close();
311
+ } catch {}
312
+ activeStore = null;
313
+ return result;
314
+ };
315
+ const record = (attemptInput) => {
316
+ if (!envNotDisabled$2(process.env.CODEMEM_RETRIEVAL_LEDGER || "1")) return;
317
+ try {
318
+ resolvedDbPath ??= resolveDb(resolveDbOpt(opts));
319
+ const completedAt = (deps.now ?? (() => /* @__PURE__ */ new Date()))();
320
+ const ledgerInput = {
321
+ ...attemptInput,
322
+ attemptId,
323
+ surface: "file_context",
324
+ trigger: "automatic",
325
+ startedAt: startedAt.toISOString(),
326
+ completedAt: completedAt.toISOString(),
327
+ latencyMs: Math.max(0, completedAt.getTime() - startedAt.getTime()),
328
+ recorderVersion: "claude-file-context-v1",
329
+ source: "claude",
330
+ streamId: sourceSessionId(payload),
331
+ sourceSessionId: sourceSessionId(payload),
332
+ mode: "claude_pre_tool_use_read"
333
+ };
334
+ if (deps.recordAttempt) deps.recordAttempt(resolvedDbPath, ledgerInput);
335
+ else {
336
+ const store = getStore();
337
+ recordRetrievalSurface(store.db, {
338
+ ...ledgerInput,
339
+ sessionId: resolveRetrievalSession(store.db, "claude", ledgerInput.sourceSessionId)
340
+ });
341
+ }
342
+ } catch {}
343
+ };
344
+ const updateDelivery = (status) => {
345
+ if (!envNotDisabled$2(process.env.CODEMEM_RETRIEVAL_LEDGER || "1")) return;
346
+ try {
347
+ resolvedDbPath ??= resolveDb(resolveDbOpt(opts));
348
+ if (deps.updateDelivery) deps.updateDelivery(resolvedDbPath, attemptId, status);
349
+ else tryUpdateRetrievalDelivery(getStore().db, attemptId, status);
350
+ } catch {}
351
+ };
352
+ if (!envNotDisabled$2(process.env.CODEMEM_FILE_CONTEXT || "1")) {
353
+ record({
354
+ retrievalStatus: "skipped",
355
+ deliveryStatus: "not_attempted",
356
+ failureCode: "file_context_disabled",
357
+ failureStage: "configuration"
358
+ });
359
+ return finish(continueResult$2());
360
+ }
295
361
  const cwd = typeof payload.cwd === "string" && payload.cwd.trim() ? payload.cwd : process.cwd();
296
362
  const expandedPath = expandHome$3(filePath);
297
363
  const absolutePath = isAbsolute(expandedPath) ? expandedPath : resolve(cwd, expandedPath);
@@ -299,38 +365,89 @@ async function buildClaudeFileContext(payload, opts, deps = {}) {
299
365
  const escapesCwd = relativePath === ".." || relativePath.startsWith("../") || isAbsolute(relativePath);
300
366
  if (!relativePath || escapesCwd) {
301
367
  logHookEvent(`file_context.skip reason=outside_cwd path=${JSON.stringify(filePath)} cwd=${JSON.stringify(cwd)}`);
302
- return continueResult$2();
368
+ record({
369
+ retrievalStatus: "skipped",
370
+ deliveryStatus: "not_attempted",
371
+ failureCode: "outside_cwd",
372
+ failureStage: "path_validation"
373
+ });
374
+ return finish(continueResult$2());
303
375
  }
304
376
  const minBytes = Number.parseInt(process.env.CODEMEM_FILE_CONTEXT_MIN_BYTES ?? `${FILE_GATE_MIN_BYTES}`, 10);
305
377
  const minBytesEffective = Number.isFinite(minBytes) && minBytes >= 0 ? minBytes : FILE_GATE_MIN_BYTES;
306
378
  const stat = (deps.statFile ?? statFile)(absolutePath);
307
379
  if (!stat) {
308
380
  logHookEvent(`file_context.skip reason=stat_failed path=${JSON.stringify(relativePath)}`);
309
- return continueResult$2();
381
+ record({
382
+ retrievalStatus: "skipped",
383
+ deliveryStatus: "not_attempted",
384
+ failureCode: "stat_failed",
385
+ failureStage: "file_access",
386
+ repositoryPaths: [relativePath]
387
+ });
388
+ return finish(continueResult$2());
310
389
  }
311
390
  const bypassSizeGate = SMALL_FILE_BYPASS_PATTERNS.some((p) => p.test(relativePath));
312
391
  if (stat.sizeBytes < minBytesEffective && !bypassSizeGate) {
313
392
  logHookEvent(`file_context.skip reason=below_size_gate path=${JSON.stringify(relativePath)} size=${stat.sizeBytes} gate=${minBytesEffective}`);
314
- return continueResult$2();
393
+ record({
394
+ retrievalStatus: "skipped",
395
+ deliveryStatus: "not_attempted",
396
+ failureCode: "below_size_gate",
397
+ failureStage: "size_gate",
398
+ repositoryPaths: [relativePath]
399
+ });
400
+ return finish(continueResult$2());
315
401
  }
316
402
  const project = resolveProject$1(payload);
317
- const resolveDb = deps.resolveDb ?? resolveDbPath;
318
403
  const queryFn = deps.queryByFile ?? queryByFile;
319
404
  let rows = [];
320
405
  try {
321
- rows = queryFn(resolveDb(resolveDbOpt(opts)), relativePath, project, FETCH_LIMIT);
406
+ resolvedDbPath ??= resolveDb(resolveDbOpt(opts));
407
+ rows = deps.queryByFile ? queryFn(resolvedDbPath, relativePath, project, FETCH_LIMIT) : getStore().findByFile(relativePath, {
408
+ limit: FETCH_LIMIT,
409
+ ...project ? { project } : {}
410
+ });
322
411
  } catch (err) {
323
412
  logHookEvent(`codemem claude-hook-file-context query failed: ${err instanceof Error ? err.message : String(err)}`);
324
- return continueResult$2();
413
+ record({
414
+ retrievalStatus: "failed",
415
+ deliveryStatus: "not_attempted",
416
+ failureCode: "query_failed",
417
+ failureStage: "retrieval",
418
+ project,
419
+ filters: project ? { project } : void 0,
420
+ repositoryPaths: [relativePath]
421
+ });
422
+ return finish(continueResult$2());
325
423
  }
326
424
  if (rows.length === 0) {
327
425
  logHookEvent(`file_context.skip reason=no_observations path=${JSON.stringify(relativePath)} project=${JSON.stringify(project ?? "")}`);
328
- return continueResult$2();
426
+ record({
427
+ retrievalStatus: "no_results",
428
+ deliveryStatus: "not_attempted",
429
+ project,
430
+ filters: project ? { project } : void 0,
431
+ repositoryPaths: [relativePath]
432
+ });
433
+ return finish(continueResult$2());
329
434
  }
330
435
  const top = scoreAndDedupe(rows, relativePath, DISPLAY_LIMIT);
331
436
  if (top.length === 0) {
332
437
  logHookEvent(`file_context.skip reason=no_top_after_dedupe path=${JSON.stringify(relativePath)} candidates=${rows.length}`);
333
- return continueResult$2();
438
+ record({
439
+ retrievalStatus: "succeeded",
440
+ deliveryStatus: "not_attempted",
441
+ candidateIds: rows.map((row) => row.id),
442
+ candidateCount: rows.length,
443
+ selectedIds: [],
444
+ failureCode: "no_top_after_dedupe",
445
+ failureStage: "selection",
446
+ project,
447
+ filters: project ? { project } : void 0,
448
+ repositoryPaths: [relativePath]
449
+ });
450
+ return finish(continueResult$2());
334
451
  }
335
452
  let staleness = null;
336
453
  if (stat.mtimeMs > 0) {
@@ -343,13 +460,30 @@ async function buildClaudeFileContext(payload, opts, deps = {}) {
343
460
  newestObservationMs
344
461
  };
345
462
  }
346
- const timeline = formatTimeline(top, relativePath, staleness);
463
+ record({
464
+ retrievalStatus: "succeeded",
465
+ deliveryStatus: "not_attempted",
466
+ candidateIds: rows.map((row) => row.id),
467
+ candidateCount: rows.length,
468
+ selectedIds: top.map((row) => row.id),
469
+ project,
470
+ filters: project ? { project } : void 0,
471
+ repositoryPaths: [relativePath]
472
+ });
473
+ let timeline;
474
+ try {
475
+ timeline = formatTimeline(top, relativePath, staleness);
476
+ } catch {
477
+ updateDelivery("failed");
478
+ return finish(continueResult$2());
479
+ }
347
480
  logHookEvent(`file_context.ok path=${JSON.stringify(relativePath)} candidates=${rows.length} surfaced=${top.length} project=${JSON.stringify(project ?? "")} stale=${staleness ? "true" : "false"}`);
348
- return { hookSpecificOutput: {
481
+ updateDelivery("handed_off");
482
+ return finish({ hookSpecificOutput: {
349
483
  hookEventName: "PreToolUse",
350
484
  permissionDecision: "allow",
351
485
  additionalContext: timeline
352
- } };
486
+ } });
353
487
  }
354
488
  var claudeHookFileContextCmd = new Command("claude-hook-file-context").configureHelp(helpStyle).description("Return Claude PreToolUse:Read additionalContext from per-file observation timeline");
355
489
  addDbOption(claudeHookFileContextCmd);
@@ -2666,12 +2800,13 @@ function buildCoordinatorCommand() {
2666
2800
  }
2667
2801
  });
2668
2802
  cmd.addCommand(listScopeMembersCmd);
2669
- const grantScopeMemberCmd = new Command("grant-scope-member").configureHelp(helpStyle).description("Grant a device explicit access to a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").option("--role <role>", "membership role").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2803
+ const grantScopeMemberCmd = new Command("grant-scope-member").configureHelp(helpStyle).description("Grant a device explicit access to a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").requiredOption("--effect-id <id>", "deterministic mutation effect id").option("--role <role>", "membership role").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2670
2804
  addDbOption(grantScopeMemberCmd);
2671
2805
  addJsonOption(grantScopeMemberCmd);
2672
2806
  grantScopeMemberCmd.action(async (groupId, scopeId, deviceId, opts) => {
2673
2807
  try {
2674
2808
  const membership = await coordinatorGrantScopeMembershipAction({
2809
+ effectId: opts.effectId,
2675
2810
  groupId,
2676
2811
  scopeId,
2677
2812
  deviceId,
@@ -2699,12 +2834,13 @@ function buildCoordinatorCommand() {
2699
2834
  }
2700
2835
  });
2701
2836
  cmd.addCommand(grantScopeMemberCmd);
2702
- const revokeScopeMemberCmd = new Command("revoke-scope-member").configureHelp(helpStyle).description("Revoke a device from a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2837
+ const revokeScopeMemberCmd = new Command("revoke-scope-member").configureHelp(helpStyle).description("Revoke a device from a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").requiredOption("--effect-id <id>", "deterministic mutation effect id").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2703
2838
  addDbOption(revokeScopeMemberCmd);
2704
2839
  addJsonOption(revokeScopeMemberCmd);
2705
2840
  revokeScopeMemberCmd.action(async (groupId, scopeId, deviceId, opts) => {
2706
2841
  try {
2707
2842
  if (!await coordinatorRevokeScopeMembershipAction({
2843
+ effectId: opts.effectId,
2708
2844
  groupId,
2709
2845
  scopeId,
2710
2846
  deviceId,
@@ -4634,6 +4770,9 @@ function parseStrictPositiveId(value) {
4634
4770
  const n = Number(value.trim());
4635
4771
  return Number.isFinite(n) && n >= 1 && Number.isInteger(n) ? n : null;
4636
4772
  }
4773
+ function resolveOpenAIResponsesOverride(cliEnabled, configured) {
4774
+ return cliEnabled === true ? true : configured;
4775
+ }
4637
4776
  function showMemoryAction(idStr, opts) {
4638
4777
  const memoryId = parseStrictPositiveId(idStr);
4639
4778
  if (memoryId === null) {
@@ -5061,10 +5200,11 @@ function createMemoryExtractionReplayCommand() {
5061
5200
  const observerConfigWithOverrides = {
5062
5201
  ...observerConfig,
5063
5202
  observerTemperature: observerTemperature ?? observerConfig.observerTemperature,
5064
- observerOpenAIUseResponses: opts.openaiResponses === true,
5065
- observerReasoningEffort: opts.reasoningEffort?.trim() || null,
5066
- observerReasoningSummary: opts.reasoningSummary?.trim() || null,
5067
- observerMaxOutputTokens: maxOutputTokens ?? observerConfig.observerMaxTokens
5203
+ observerOpenAIUseResponses: resolveOpenAIResponsesOverride(opts.openaiResponses, observerConfig.observerOpenAIUseResponses),
5204
+ observerReasoningEffort: opts.reasoningEffort === void 0 ? observerConfig.observerReasoningEffort : opts.reasoningEffort.trim() || null,
5205
+ observerReasoningSummary: opts.reasoningSummary === void 0 ? observerConfig.observerReasoningSummary : opts.reasoningSummary.trim() || null,
5206
+ observerMaxOutputTokens: maxOutputTokens ?? observerConfig.observerMaxOutputTokens ?? observerConfig.observerMaxTokens,
5207
+ observerExplicitConfigKeys: maxOutputTokens === null ? observerConfig.observerExplicitConfigKeys : [...new Set([...observerConfig.observerExplicitConfigKeys ?? [], "observerMaxOutputTokens"])]
5068
5208
  };
5069
5209
  const observer = new ObserverClient(observerConfigWithOverrides);
5070
5210
  const result = opts.observerTierRouting === true ? await replayBatchExtractionWithTierRouting(resolveDbOpt(opts), observerConfigWithOverrides, {
@@ -5117,8 +5257,37 @@ function createMemoryExtractionReplayCommand() {
5117
5257
  });
5118
5258
  return cmd;
5119
5259
  }
5260
+ function reconcileExtractionBenchmarkStatus(input) {
5261
+ const quality = input.finalQuality;
5262
+ let status = input.classification.status;
5263
+ let reason = input.classification.reason;
5264
+ if (input.purpose === "shape_quality" && quality && status !== "observer_no_output") {
5265
+ if (quality.summaryDisposition.score === 0) {
5266
+ status = "shape_fail";
5267
+ reason = `summary disposition ${quality.summaryDisposition.actual} does not satisfy expected ${quality.summaryDisposition.expected}`;
5268
+ } else if (status === "shape_fail" && quality.summaryDisposition.actual === "skip" && input.finalFailureReasons.length > 0 && input.finalFailureReasons.every((failure) => failure.startsWith("summary count "))) {
5269
+ status = "pass";
5270
+ reason = "valid low-signal skip satisfies benchmark disposition";
5271
+ }
5272
+ }
5273
+ return {
5274
+ status,
5275
+ reason,
5276
+ quality,
5277
+ initialQuality: input.initialQuality
5278
+ };
5279
+ }
5280
+ function summarizeBenchmarkReasoning(runs, fallback) {
5281
+ const source = runs[0] ?? fallback;
5282
+ const reasoningEfforts = new Set(runs.map((run) => run.reasoningEffort));
5283
+ const reasoningSummaries = new Set(runs.map((run) => run.reasoningSummary));
5284
+ return {
5285
+ reasoningEffort: reasoningEfforts.size > 1 ? "mixed" : source.reasoningEffort,
5286
+ reasoningSummary: reasoningSummaries.size > 1 ? "mixed" : source.reasoningSummary
5287
+ };
5288
+ }
5120
5289
  function createMemoryExtractionBenchmarkCommand() {
5121
- const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run");
5290
+ const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run").option("--repetitions <n>", "run every benchmark batch 1-10 times to measure model stability", "1");
5122
5291
  addDbOption(cmd);
5123
5292
  addJsonOption(cmd);
5124
5293
  cmd.action(async (opts) => {
@@ -5139,20 +5308,24 @@ function createMemoryExtractionBenchmarkCommand() {
5139
5308
  const maxOutputTokensInput = opts.maxOutputTokens?.trim() ?? "";
5140
5309
  const maxOutputTokens = maxOutputTokensInput.length > 0 ? parseStrictPositiveId(maxOutputTokensInput) : null;
5141
5310
  if (maxOutputTokensInput.length > 0 && maxOutputTokens === null) throw new Error(`Invalid max output tokens: ${maxOutputTokensInput || opts.maxOutputTokens}`);
5311
+ const repetitionsInput = opts.repetitions?.trim() ?? "1";
5312
+ const repetitions = parseStrictPositiveId(repetitionsInput);
5313
+ if (repetitions === null || repetitions > 10) throw new Error(`Invalid repetitions: ${repetitionsInput || opts.repetitions}`);
5142
5314
  const observerConfig = loadObserverConfig();
5143
5315
  const observerConfigWithOverrides = {
5144
5316
  ...observerConfig,
5145
5317
  observerProvider: opts.observerProvider?.trim() || observerConfig.observerProvider,
5146
5318
  observerModel: opts.observerModel?.trim() || observerConfig.observerModel,
5147
5319
  observerTemperature: observerTemperature ?? observerConfig.observerTemperature,
5148
- observerOpenAIUseResponses: opts.openaiResponses === true,
5149
- observerReasoningEffort: opts.reasoningEffort?.trim() || null,
5150
- observerReasoningSummary: opts.reasoningSummary?.trim() || null,
5151
- observerMaxOutputTokens: maxOutputTokens ?? observerConfig.observerMaxTokens
5320
+ observerOpenAIUseResponses: resolveOpenAIResponsesOverride(opts.openaiResponses, observerConfig.observerOpenAIUseResponses),
5321
+ observerReasoningEffort: opts.reasoningEffort === void 0 ? observerConfig.observerReasoningEffort : opts.reasoningEffort.trim() || null,
5322
+ observerReasoningSummary: opts.reasoningSummary === void 0 ? observerConfig.observerReasoningSummary : opts.reasoningSummary.trim() || null,
5323
+ observerMaxOutputTokens: maxOutputTokens ?? observerConfig.observerMaxOutputTokens ?? observerConfig.observerMaxTokens,
5324
+ observerExplicitConfigKeys: maxOutputTokens === null ? observerConfig.observerExplicitConfigKeys : [...new Set([...observerConfig.observerExplicitConfigKeys ?? [], "observerMaxOutputTokens"])]
5152
5325
  };
5153
5326
  const observer = new ObserverClient(observerConfigWithOverrides);
5154
5327
  const runs = [];
5155
- for (const batch of benchmark.batches) {
5328
+ for (let iteration = 1; iteration <= repetitions; iteration += 1) for (const batch of benchmark.batches) {
5156
5329
  const scenarioId = batch.scenarioId ?? benchmark.scenarioId;
5157
5330
  const result = opts.observerTierRouting === true ? await replayBatchExtractionWithTierRouting(resolveDbOpt(opts), observerConfigWithOverrides, {
5158
5331
  batchId: batch.batchId,
@@ -5163,7 +5336,51 @@ function createMemoryExtractionBenchmarkCommand() {
5163
5336
  scenarioId,
5164
5337
  transcriptBudget: transcriptBudget ?? void 0
5165
5338
  });
5339
+ const costModel = result.observer.modelFallbackApplied ? result.observer.resolvedModel : result.observer.resolvedModel ?? result.observer.model;
5340
+ const initialCost = costModel ? estimateExtractionModelCost(costModel, result.observer.initialUsage) : null;
5341
+ const repairCost = costModel ? estimateExtractionModelCost(costModel, result.observer.repairedUsage) : null;
5342
+ const totalCost = costModel ? estimateExtractionModelCost(costModel, result.observer.totalUsage) : null;
5343
+ const pricing = costModel ? getExtractionModelPricing(costModel) : null;
5344
+ const costUnavailableReason = totalCost ? null : result.observer.modelFallbackApplied && !result.observer.resolvedModel ? "model_fallback_unresolved" : result.observer.totalUsage == null ? "missing_usage" : "unknown_model_pricing";
5345
+ const initialQuality = result.observer.initialDiagnostics ? scoreExtractionBenchmarkOutput({
5346
+ parsed: result.observer.initialParsed,
5347
+ diagnostics: result.observer.initialDiagnostics,
5348
+ review: batch.review ?? {
5349
+ status: "unreviewed",
5350
+ reviewerNotes: "No durable-fact review has been recorded for this batch."
5351
+ },
5352
+ estimatedCostUsd: initialCost?.totalCostUsd ?? null,
5353
+ expectedSummaryDisposition: batch.expectedSummaryDisposition
5354
+ }) : null;
5355
+ const repairQuality = result.observer.repairedParsed && result.observer.repairedDiagnostics ? scoreExtractionBenchmarkOutput({
5356
+ parsed: result.observer.repairedParsed,
5357
+ diagnostics: result.observer.repairedDiagnostics,
5358
+ review: batch.review ?? {
5359
+ status: "unreviewed",
5360
+ reviewerNotes: "No durable-fact review has been recorded for this batch."
5361
+ },
5362
+ estimatedCostUsd: repairCost?.totalCostUsd ?? null,
5363
+ expectedSummaryDisposition: batch.expectedSummaryDisposition
5364
+ }) : null;
5365
+ const finalQuality = result.observer.diagnostics ? scoreExtractionBenchmarkOutput({
5366
+ parsed: result.observer.parsed,
5367
+ diagnostics: result.observer.diagnostics,
5368
+ review: batch.review ?? {
5369
+ status: "unreviewed",
5370
+ reviewerNotes: "No durable-fact review has been recorded for this batch."
5371
+ },
5372
+ estimatedCostUsd: totalCost?.totalCostUsd ?? null,
5373
+ expectedSummaryDisposition: batch.expectedSummaryDisposition
5374
+ }) : null;
5375
+ const reconciled = reconcileExtractionBenchmarkStatus({
5376
+ purpose: batch.purpose,
5377
+ classification: result.classification,
5378
+ finalFailureReasons: result.evaluation.failureReasons,
5379
+ initialQuality,
5380
+ finalQuality
5381
+ });
5166
5382
  runs.push({
5383
+ iteration,
5167
5384
  batchId: batch.batchId,
5168
5385
  sessionId: batch.sessionId,
5169
5386
  label: batch.label,
@@ -5171,17 +5388,23 @@ function createMemoryExtractionBenchmarkCommand() {
5171
5388
  complexity: batch.complexity,
5172
5389
  scenarioId,
5173
5390
  expectedTier: batch.expectedTier ?? null,
5391
+ expectedSummaryDisposition: batch.expectedSummaryDisposition,
5174
5392
  analysis: {
5175
5393
  eventSpan: result.analysis.eventSpan,
5176
5394
  promptCount: result.analysis.promptCount,
5177
5395
  toolCount: result.analysis.toolCount,
5178
5396
  transcriptLength: result.analysis.transcriptLength
5179
5397
  },
5180
- status: result.classification.status,
5181
- reason: result.classification.reason,
5398
+ status: reconciled.status,
5399
+ reason: reconciled.reason,
5182
5400
  tier: result.observer.tier ?? "manual",
5183
5401
  provider: result.observer.provider,
5184
5402
  model: result.observer.model,
5403
+ transport: result.observer.transport,
5404
+ requestedModel: result.observer.requestedModel,
5405
+ resolvedModel: result.observer.resolvedModel,
5406
+ modelFallbackApplied: result.observer.modelFallbackApplied,
5407
+ modelFallbackReason: result.observer.modelFallbackReason,
5185
5408
  openaiUseResponses: result.observer.openaiUseResponses,
5186
5409
  reasoningEffort: result.observer.reasoningEffort,
5187
5410
  reasoningSummary: result.observer.reasoningSummary,
@@ -5189,39 +5412,111 @@ function createMemoryExtractionBenchmarkCommand() {
5189
5412
  temperature: result.observer.temperature,
5190
5413
  summaries: result.evaluation.counts.summaries,
5191
5414
  observations: result.evaluation.counts.observations,
5192
- repairApplied: result.observer.repairApplied
5415
+ repairApplied: result.observer.repairApplied,
5416
+ initial: {
5417
+ raw: result.observer.initialRaw,
5418
+ status: result.initialClassification.status,
5419
+ reason: result.initialClassification.reason,
5420
+ pass: result.initialEvaluation.pass,
5421
+ failureReasons: result.initialEvaluation.failureReasons,
5422
+ summaries: result.initialEvaluation.counts.summaries,
5423
+ observations: result.initialEvaluation.counts.observations,
5424
+ diagnostics: result.observer.initialDiagnostics,
5425
+ elapsedMs: result.observer.initialElapsedMs,
5426
+ usage: result.observer.initialUsage,
5427
+ quality: reconciled.initialQuality
5428
+ },
5429
+ repair: {
5430
+ applied: result.observer.repairApplied,
5431
+ raw: result.observer.repairedRaw,
5432
+ status: result.repairedClassification?.status ?? null,
5433
+ reason: result.repairedClassification?.reason ?? null,
5434
+ pass: result.repairedEvaluation?.pass ?? null,
5435
+ failureReasons: result.repairedEvaluation?.failureReasons ?? [],
5436
+ summaries: result.repairedEvaluation?.counts.summaries ?? null,
5437
+ observations: result.repairedEvaluation?.counts.observations ?? null,
5438
+ diagnostics: result.observer.repairedDiagnostics,
5439
+ elapsedMs: result.observer.repairedElapsedMs,
5440
+ usage: result.observer.repairedUsage,
5441
+ quality: repairQuality
5442
+ },
5443
+ telemetry: {
5444
+ totalElapsedMs: result.observer.totalElapsedMs,
5445
+ totalUsage: result.observer.totalUsage
5446
+ },
5447
+ pricing,
5448
+ cost: {
5449
+ initial: initialCost,
5450
+ repair: repairCost,
5451
+ total: totalCost,
5452
+ unavailableReason: costUnavailableReason
5453
+ },
5454
+ quality: reconciled.quality
5193
5455
  });
5194
5456
  }
5457
+ const reviewedQualityRuns = runs.filter((run) => run.quality?.weightedQualityScore != null);
5458
+ const knownCostRuns = runs.filter((run) => run.cost.total != null);
5459
+ const knownElapsedRuns = runs.filter((run) => run.telemetry.totalElapsedMs != null);
5195
5460
  const summary = {
5461
+ repetitions,
5196
5462
  total: runs.length,
5197
5463
  shapeQualityTotal: runs.filter((run) => run.purpose === "shape_quality").length,
5198
5464
  shapeQualityPasses: runs.filter((run) => run.purpose === "shape_quality" && run.status === "pass").length,
5199
5465
  shapeQualityFails: runs.filter((run) => run.purpose === "shape_quality" && run.status === "shape_fail").length,
5200
5466
  expectedTierTotal: runs.filter((run) => run.expectedTier != null).length,
5201
5467
  expectedTierMatches: runs.filter((run) => run.expectedTier != null && run.expectedTier === run.tier).length,
5202
- robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length
5468
+ robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length,
5469
+ summaryDispositionTotal: runs.filter((run) => run.quality != null).length,
5470
+ summaryDispositionMatches: runs.filter((run) => run.quality?.summaryDisposition.score === 1).length,
5471
+ reviewedQualityRuns: reviewedQualityRuns.length,
5472
+ knownCostRuns: knownCostRuns.length,
5473
+ unknownCostRuns: runs.length - knownCostRuns.length,
5474
+ missingUsageRuns: runs.filter((run) => run.cost.unavailableReason === "missing_usage").length,
5475
+ unknownPricingRuns: runs.filter((run) => run.cost.unavailableReason === "unknown_model_pricing").length,
5476
+ fallbackUnresolvedRuns: runs.filter((run) => run.cost.unavailableReason === "model_fallback_unresolved").length,
5477
+ totalKnownCostUsd: knownCostRuns.reduce((sum, run) => sum + (run.cost.total?.totalCostUsd ?? 0), 0),
5478
+ knownElapsedRuns: knownElapsedRuns.length,
5479
+ totalKnownElapsedMs: knownElapsedRuns.reduce((sum, run) => sum + (run.telemetry.totalElapsedMs ?? 0), 0),
5480
+ perBatchStability: benchmark.batches.map((batch) => {
5481
+ const batchRuns = runs.filter((run) => run.batchId === batch.batchId);
5482
+ const passes = batchRuns.filter((run) => run.status === "pass").length;
5483
+ return {
5484
+ batchId: batch.batchId,
5485
+ purpose: batch.purpose,
5486
+ passes,
5487
+ total: batchRuns.length,
5488
+ passRate: batchRuns.length > 0 ? passes / batchRuns.length : null,
5489
+ statuses: batchRuns.map((run) => run.status)
5490
+ };
5491
+ })
5203
5492
  };
5204
- const uniqueObserverKeys = Array.from(new Set(runs.map((run) => `${run.provider}::${run.model}::${run.openaiUseResponses ? "responses" : "chat"}`)));
5493
+ const uniqueObserverKeys = Array.from(new Set(runs.map((run) => `${run.provider}::${run.model}::${run.transport}`)));
5494
+ const benchmarkReasoning = summarizeBenchmarkReasoning(runs, {
5495
+ reasoningEffort: observer.reasoningEffort,
5496
+ reasoningSummary: observer.reasoningSummary
5497
+ });
5205
5498
  const observerSummary = opts.observerTierRouting === true ? {
5206
5499
  provider: uniqueObserverKeys.length === 1 ? runs[0]?.provider ?? observer.provider : "mixed",
5207
5500
  model: uniqueObserverKeys.length === 1 ? runs[0]?.model ?? observer.model : "mixed",
5501
+ transport: uniqueObserverKeys.length === 1 ? runs[0]?.transport ?? "unknown" : "mixed",
5208
5502
  tierRouting: true,
5209
5503
  openaiUseResponses: uniqueObserverKeys.length === 1 ? runs[0]?.openaiUseResponses ?? observer.openaiUseResponses : null,
5210
- reasoningEffort: uniqueObserverKeys.length === 1 ? runs[0]?.reasoningEffort ?? observer.reasoningEffort : "mixed",
5211
- reasoningSummary: uniqueObserverKeys.length === 1 ? runs[0]?.reasoningSummary ?? observer.reasoningSummary : "mixed",
5212
- maxOutputTokens: uniqueObserverKeys.length === 1 ? runs[0]?.maxOutputTokens ?? observer.maxOutputTokens : null,
5213
- temperature: uniqueObserverKeys.length === 1 ? runs[0]?.temperature ?? observer.temperature : null,
5504
+ reasoningEffort: benchmarkReasoning.reasoningEffort,
5505
+ reasoningSummary: benchmarkReasoning.reasoningSummary,
5506
+ maxOutputTokens: uniqueObserverKeys.length === 1 ? runs[0]?.maxOutputTokens ?? null : null,
5507
+ temperature: uniqueObserverKeys.length === 1 ? runs[0]?.temperature ?? null : null,
5214
5508
  transcriptBudget: transcriptBudget ?? null,
5215
5509
  selectedObservers: uniqueObserverKeys
5216
5510
  } : {
5217
5511
  provider: observer.provider,
5218
5512
  model: observer.model,
5513
+ transport: runs[0]?.transport ?? observer.getStatus().runtime,
5219
5514
  tierRouting: false,
5220
5515
  openaiUseResponses: observer.openaiUseResponses,
5221
- reasoningEffort: observer.reasoningEffort,
5222
- reasoningSummary: observer.reasoningSummary,
5223
- maxOutputTokens: observer.maxOutputTokens,
5224
- temperature: observer.temperature,
5516
+ reasoningEffort: benchmarkReasoning.reasoningEffort,
5517
+ reasoningSummary: benchmarkReasoning.reasoningSummary,
5518
+ maxOutputTokens: runs[0]?.maxOutputTokens ?? null,
5519
+ temperature: runs[0]?.temperature ?? null,
5225
5520
  transcriptBudget: transcriptBudget ?? null,
5226
5521
  selectedObservers: uniqueObserverKeys
5227
5522
  };
@@ -5229,7 +5524,8 @@ function createMemoryExtractionBenchmarkCommand() {
5229
5524
  benchmark: {
5230
5525
  id: benchmark.id,
5231
5526
  title: benchmark.title,
5232
- scenarioId: benchmark.scenarioId
5527
+ scenarioId: benchmark.scenarioId,
5528
+ modelCandidates: benchmark.modelCandidates
5233
5529
  },
5234
5530
  observer: observerSummary,
5235
5531
  summary,
@@ -5243,19 +5539,31 @@ function createMemoryExtractionBenchmarkCommand() {
5243
5539
  p.log.info([
5244
5540
  `Benchmark: ${benchmark.id} — ${benchmark.title}`,
5245
5541
  `Observer: ${observerSummary.provider}/${observerSummary.model}`,
5542
+ `Transport: ${observerSummary.transport}`,
5246
5543
  `Tier routing: ${opts.observerTierRouting === true ? "yes" : "no"}`,
5247
5544
  `OpenAI Responses: ${observerSummary.openaiUseResponses === null ? "mixed" : observerSummary.openaiUseResponses ? "yes" : "no"}`,
5248
- `Reasoning effort: ${observerSummary.reasoningEffort ?? "none"}`,
5249
- `Reasoning summary: ${observerSummary.reasoningSummary ?? "none"}`,
5250
- `Max output tokens: ${observerSummary.maxOutputTokens ?? "mixed"}`,
5251
- `Temperature: ${observerSummary.temperature ?? "mixed"}`,
5545
+ `Reasoning effort: ${observerSummary.reasoningEffort ?? "not transmitted"}`,
5546
+ `Reasoning summary: ${observerSummary.reasoningSummary ?? "not transmitted"}`,
5547
+ `Max output tokens: ${observerSummary.transport === "codex_consumer" ? "not transmitted" : observerSummary.maxOutputTokens ?? "mixed"}`,
5548
+ `Temperature: ${observerSummary.transport === "mixed" ? "mixed" : observerSummary.temperature ?? "not transmitted"}`,
5252
5549
  `Transcript budget override: ${transcriptBudget ?? "default"}`,
5550
+ `Repetitions: ${summary.repetitions}`,
5253
5551
  `Shape-quality passes: ${summary.shapeQualityPasses}/${summary.shapeQualityTotal}`,
5254
5552
  `Shape-quality fails: ${summary.shapeQualityFails}`,
5255
5553
  `Expected-tier matches: ${summary.expectedTierMatches}/${summary.expectedTierTotal}`,
5256
- `Observer no-output cases: ${summary.robustnessNoOutput}`
5554
+ `Observer no-output cases: ${summary.robustnessNoOutput}`,
5555
+ `Summary disposition matches: ${summary.summaryDispositionMatches}/${summary.summaryDispositionTotal}`,
5556
+ `Reviewed quality runs: ${summary.reviewedQualityRuns} (compare per-run dimensions; scores are fixture-specific)`,
5557
+ `Known estimated cost: $${summary.totalKnownCostUsd.toFixed(6)} (${summary.knownCostRuns}/${summary.total}; missing usage=${summary.missingUsageRuns}, unknown pricing=${summary.unknownPricingRuns}, unresolved fallback=${summary.fallbackUnresolvedRuns})`,
5558
+ `Known elapsed time: ${summary.totalKnownElapsedMs}ms (${summary.knownElapsedRuns}/${summary.total} run(s))`
5257
5559
  ].join("\n"));
5258
- for (const run of runs) p.log.message(` [${run.batchId}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model}${run.openaiUseResponses ? " [responses]" : ""} summaries=${run.summaries} observations=${run.observations} repair=${run.repairApplied ? "yes" : "no"} — ${run.label}`);
5560
+ for (const run of runs) {
5561
+ const qualityLabel = run.quality?.weightedQualityScore == null ? "n/a" : run.quality.weightedQualityScore.toFixed(3);
5562
+ const costLabel = run.cost.total == null ? "n/a" : `$${run.cost.total.totalCostUsd.toFixed(6)}`;
5563
+ const latencyLabel = run.telemetry.totalElapsedMs == null ? "n/a" : `${run.telemetry.totalElapsedMs}ms`;
5564
+ const missingRequired = run.quality?.requiredRecall.missingLabelIds.join(",") || "none";
5565
+ p.log.message(` [${run.batchId}#${run.iteration}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} disposition=${run.quality?.summaryDisposition.actual ?? "n/a"}/${run.expectedSummaryDisposition} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model} [${run.transport}] initial=${run.initial.summaries}s/${run.initial.observations}o final=${run.summaries}s/${run.observations}o quality=${qualityLabel} coverage=${run.quality?.weightedQualityCoverage?.toFixed(3) ?? "n/a"} required_missing=${missingRequired} cost=${costLabel} latency=${latencyLabel} schema_loss=${run.initial.diagnostics?.dataLoss === true ? "yes" : "no"} fallback=${run.modelFallbackApplied ? "yes" : "no"} repair=${run.repairApplied ? "yes" : "no"} — ${run.label}`);
5566
+ }
5259
5567
  p.outro("done");
5260
5568
  } catch (error) {
5261
5569
  const message = error instanceof Error ? error.message : "Extraction benchmark failed";
@@ -5355,6 +5663,127 @@ memoryCommand.addCommand(createMemoryRelinkReportCommand());
5355
5663
  memoryCommand.addCommand(createMemoryRelinkPlanCommand());
5356
5664
  //#endregion
5357
5665
  //#region src/commands/pack.ts
5666
+ var MAX_INTERNAL_LEDGER_INPUT_BYTES = 16 * 1024;
5667
+ var INTERNAL_LEDGER_INPUT_TIMEOUT_MS = 1e3;
5668
+ var FORBIDDEN_INTERNAL_KEYS = new Set([
5669
+ "body",
5670
+ "context",
5671
+ "pack",
5672
+ "pack_text",
5673
+ "path",
5674
+ "preview",
5675
+ "prompt",
5676
+ "query",
5677
+ "raw_prompt",
5678
+ "title"
5679
+ ]);
5680
+ var INTERNAL_LEDGER_KEYS = new Set([
5681
+ "action",
5682
+ "attempt_id",
5683
+ "started_at",
5684
+ "source",
5685
+ "stream_id",
5686
+ "source_session_id",
5687
+ "prompt_number",
5688
+ "request_id",
5689
+ "retrieval_status",
5690
+ "delivery_status",
5691
+ "failure_code",
5692
+ "failure_stage",
5693
+ "original_attempt_id"
5694
+ ]);
5695
+ var INTERNAL_LEDGER_ACTIONS = new Set([
5696
+ "record",
5697
+ "delivery",
5698
+ "cache_reuse"
5699
+ ]);
5700
+ var INTERNAL_RETRIEVAL_STATUSES = new Set(["skipped", "failed"]);
5701
+ var INTERNAL_DELIVERY_STATUSES = new Set([
5702
+ "handed_off",
5703
+ "failed",
5704
+ "unknown"
5705
+ ]);
5706
+ async function readInternalLedgerPayload() {
5707
+ const stdin = process.stdin;
5708
+ if (stdin.isTTY) throw new PackUsageError("internal ledger metadata requires piped stdin JSON");
5709
+ return parseInternalLedgerPayload(await new Promise((resolve, reject) => {
5710
+ let value = "";
5711
+ const cleanup = () => {
5712
+ clearTimeout(timer);
5713
+ stdin.off("data", onData);
5714
+ stdin.off("end", onEnd);
5715
+ stdin.off("error", onError);
5716
+ };
5717
+ const fail = (error) => {
5718
+ cleanup();
5719
+ stdin.pause();
5720
+ reject(error);
5721
+ };
5722
+ const onData = (chunk) => {
5723
+ value += String(chunk);
5724
+ if (Buffer.byteLength(value, "utf8") > MAX_INTERNAL_LEDGER_INPUT_BYTES) fail(new PackUsageError("internal ledger metadata exceeds 16384 bytes"));
5725
+ };
5726
+ const onEnd = () => {
5727
+ cleanup();
5728
+ resolve(value);
5729
+ };
5730
+ const onError = (error) => fail(error);
5731
+ const timer = setTimeout(() => fail(new PackUsageError("internal ledger metadata read timed out")), INTERNAL_LEDGER_INPUT_TIMEOUT_MS);
5732
+ stdin.on("data", onData);
5733
+ stdin.once("end", onEnd);
5734
+ stdin.once("error", onError);
5735
+ }));
5736
+ }
5737
+ function parseInternalLedgerPayload(raw) {
5738
+ if (Buffer.byteLength(raw, "utf8") > MAX_INTERNAL_LEDGER_INPUT_BYTES) throw new PackUsageError("internal ledger metadata exceeds 16384 bytes");
5739
+ let value;
5740
+ try {
5741
+ value = JSON.parse(raw);
5742
+ } catch {
5743
+ throw new PackUsageError("internal ledger metadata must be valid JSON");
5744
+ }
5745
+ if (value == null || typeof value !== "object" || Array.isArray(value)) throw new PackUsageError("internal ledger metadata must be an object");
5746
+ for (const key of Object.keys(value)) {
5747
+ if (FORBIDDEN_INTERNAL_KEYS.has(key)) throw new PackUsageError(`internal ledger metadata rejects sensitive field: ${key}`);
5748
+ if (!INTERNAL_LEDGER_KEYS.has(key)) throw new PackUsageError(`internal ledger metadata contains unsupported field: ${key}`);
5749
+ }
5750
+ const payload = value;
5751
+ if (typeof payload.attempt_id !== "string") throw new PackUsageError("internal ledger metadata requires attempt_id");
5752
+ if (payload.action != null && !INTERNAL_LEDGER_ACTIONS.has(payload.action)) throw new PackUsageError("internal ledger action is invalid");
5753
+ if (payload.retrieval_status != null && !INTERNAL_RETRIEVAL_STATUSES.has(payload.retrieval_status)) throw new PackUsageError("internal ledger retrieval_status is invalid");
5754
+ if (payload.delivery_status != null && !INTERNAL_DELIVERY_STATUSES.has(payload.delivery_status)) throw new PackUsageError("internal ledger delivery_status is invalid");
5755
+ if (payload.prompt_number != null && (!Number.isInteger(payload.prompt_number) || payload.prompt_number < 0)) throw new PackUsageError("internal ledger prompt_number must be a non-negative integer");
5756
+ for (const key of [
5757
+ "started_at",
5758
+ "source",
5759
+ "stream_id",
5760
+ "source_session_id",
5761
+ "request_id",
5762
+ "failure_code",
5763
+ "failure_stage",
5764
+ "original_attempt_id"
5765
+ ]) {
5766
+ const field = payload[key];
5767
+ if (field != null && (typeof field !== "string" || field.length > 512)) throw new PackUsageError(`internal ledger metadata field ${key} is invalid`);
5768
+ if (typeof field === "string" && (posix.isAbsolute(field) || win32.isAbsolute(field))) throw new PackUsageError(`internal ledger metadata rejects absolute paths in field: ${key}`);
5769
+ }
5770
+ return payload;
5771
+ }
5772
+ function attemptMetadata(payload) {
5773
+ return {
5774
+ attemptId: payload.attempt_id,
5775
+ startedAt: payload.started_at ?? (/* @__PURE__ */ new Date()).toISOString(),
5776
+ completedAt: (/* @__PURE__ */ new Date()).toISOString(),
5777
+ source: payload.source ?? "opencode",
5778
+ streamId: payload.stream_id ?? null,
5779
+ sourceSessionId: payload.source_session_id ?? null,
5780
+ promptNumber: payload.prompt_number ?? null,
5781
+ requestId: payload.request_id ?? null
5782
+ };
5783
+ }
5784
+ function handleInstrumentedPackLedger(db, payload, context, filters, artifacts) {
5785
+ return recordPromptPackArtifacts(db, attemptMetadata(payload), context, filters, artifacts);
5786
+ }
5358
5787
  function describeCandidate(candidate) {
5359
5788
  const scoreParts = [
5360
5789
  candidate.scores.combined_score != null ? `combined=${candidate.scores.combined_score.toFixed(2)}` : null,
@@ -5428,24 +5857,46 @@ async function withStore(opts, errorCode, run) {
5428
5857
  async function packAction(context, opts) {
5429
5858
  await withStore(opts, "pack_failed", async (store) => {
5430
5859
  const { limit, budget, filters, renderOptions } = buildPackRequestOptions(opts, { envProject: process.env.CODEMEM_PROJECT });
5431
- const result = await store.buildMemoryPackAsync(context, limit, budget, filters, renderOptions);
5432
- if (opts.json) {
5433
- console.log(JSON.stringify(result, null, 2));
5434
- return;
5435
- }
5436
- p.intro(`Memory pack for "${context}"`);
5437
- if (result.items.length === 0) {
5438
- p.log.warn("No relevant memories found.");
5439
- p.outro("done");
5860
+ let result;
5861
+ if (opts.internalLedger) {
5862
+ const artifacts = await store.buildMemoryPackWithTraceAsync(context, limit, budget, filters, renderOptions);
5863
+ result = artifacts.response;
5864
+ let artifactFingerprint;
5865
+ try {
5866
+ artifactFingerprint = promptPackArtifactFingerprint(store.db, context, filters, artifacts);
5867
+ } catch {}
5868
+ let ledgerOutcome;
5869
+ try {
5870
+ const ledgerPayload = await readInternalLedgerPayload();
5871
+ ledgerOutcome = handleInstrumentedPackLedger(store.db, ledgerPayload, context, filters, artifacts);
5872
+ } catch {}
5873
+ emitPackResult(context, opts, result, artifactFingerprint, ledgerOutcome?.ok === false && ledgerOutcome.reason === "idempotency_conflict" ? ledgerOutcome : void 0);
5440
5874
  return;
5441
- }
5442
- const metrics = result.metrics;
5443
- p.log.info(`${metrics.total_items} items, ~${metrics.pack_tokens} tokens` + (metrics.fallback_used ? " (fallback)" : "") + ` [fts:${metrics.sources.fts} sem:${metrics.sources.semantic} fuzzy:${metrics.sources.fuzzy}]`);
5444
- for (const item of result.items) p.log.step(`#${item.id} ${item.kind} ${item.title}`);
5445
- p.note(result.pack_text, "pack_text");
5446
- p.outro("done");
5875
+ } else result = await store.buildMemoryPackAsync(context, limit, budget, filters, renderOptions);
5876
+ emitPackResult(context, opts, result);
5447
5877
  });
5448
5878
  }
5879
+ function emitPackResult(context, opts, result, ledgerArtifactFingerprint, ledgerOutcome) {
5880
+ if (opts.json) {
5881
+ console.log(JSON.stringify({
5882
+ ...result,
5883
+ ...ledgerArtifactFingerprint ? { ledger_artifact_fingerprint: ledgerArtifactFingerprint } : {},
5884
+ ...ledgerOutcome ? { ledger_outcome: ledgerOutcome } : {}
5885
+ }, null, 2));
5886
+ return;
5887
+ }
5888
+ p.intro(`Memory pack for "${context}"`);
5889
+ if (result.items.length === 0) {
5890
+ p.log.warn("No relevant memories found.");
5891
+ p.outro("done");
5892
+ return;
5893
+ }
5894
+ const metrics = result.metrics;
5895
+ p.log.info(`${metrics.total_items} items, ~${metrics.pack_tokens} tokens` + (metrics.fallback_used ? " (fallback)" : "") + ` [fts:${metrics.sources.fts} sem:${metrics.sources.semantic} fuzzy:${metrics.sources.fuzzy}]`);
5896
+ for (const item of result.items) p.log.step(`#${item.id} ${item.kind} ${item.title}`);
5897
+ p.note(result.pack_text, "pack_text");
5898
+ p.outro("done");
5899
+ }
5449
5900
  async function traceAction(context, opts) {
5450
5901
  await withStore(opts, "pack_trace_failed", async (store) => {
5451
5902
  const { limit, budget, filters, renderOptions } = buildPackRequestOptions(opts, { envProject: process.env.CODEMEM_PROJECT });
@@ -5458,6 +5909,7 @@ async function traceAction(context, opts) {
5458
5909
  });
5459
5910
  }
5460
5911
  var packCmd = addPackRequestOptions(new Command("pack").enablePositionalOptions().configureHelp(helpStyle).description("Build a context-aware memory pack").argument("<context>", "context string to search for"));
5912
+ packCmd.addOption(new Option("--internal-ledger").hideHelp());
5461
5913
  addDbOption(packCmd);
5462
5914
  addJsonOption(packCmd);
5463
5915
  packCmd.action(packAction);
@@ -5467,6 +5919,38 @@ addJsonOption(traceCmd);
5467
5919
  traceCmd.action(traceAction);
5468
5920
  packCmd.addCommand(traceCmd);
5469
5921
  var packCommand = packCmd;
5922
+ function handlePromptPackLedger(db, payload) {
5923
+ const metadata = attemptMetadata(payload);
5924
+ if (payload.action === "delivery") {
5925
+ const status = payload.delivery_status;
5926
+ if (!status) throw new PackUsageError("delivery action requires delivery_status");
5927
+ const outcome = tryUpdateRetrievalDelivery(db, payload.attempt_id, status);
5928
+ if (!outcome.ok) throw new Error(outcome.reason);
5929
+ return outcome.value;
5930
+ }
5931
+ if (payload.action === "cache_reuse") {
5932
+ if (!payload.original_attempt_id) throw new PackUsageError("cache_reuse action requires original_attempt_id");
5933
+ const outcome = clonePromptPackAttempt(db, payload.original_attempt_id, metadata);
5934
+ if (!outcome.ok) throw new Error(outcome.reason);
5935
+ return outcome.value;
5936
+ }
5937
+ if (payload.action === "record") {
5938
+ if (!payload.retrieval_status || !payload.failure_code || !payload.failure_stage) throw new PackUsageError("record action requires retrieval_status, failure_code, and failure_stage");
5939
+ const outcome = recordPromptPackTerminal(db, metadata, payload.retrieval_status, payload.failure_code, payload.failure_stage);
5940
+ if (!outcome.ok) throw new Error(outcome.reason);
5941
+ return outcome.value;
5942
+ }
5943
+ throw new PackUsageError("internal ledger action is invalid");
5944
+ }
5945
+ async function promptPackLedgerAction(opts) {
5946
+ await withStore(opts, "prompt_pack_ledger_failed", async (store) => {
5947
+ const payload = await readInternalLedgerPayload();
5948
+ handlePromptPackLedger(store.db, payload);
5949
+ });
5950
+ }
5951
+ var ledgerCmd = new Command("prompt-pack-ledger").description("Record internal prompt-pack lifecycle metadata").action(promptPackLedgerAction);
5952
+ addDbOption(ledgerCmd);
5953
+ var promptPackLedgerCommand = ledgerCmd;
5470
5954
  //#endregion
5471
5955
  //#region src/commands/recent.ts
5472
5956
  var cmd = new Command("recent").configureHelp(helpStyle).description("Show recent memories").option("--limit <n>", "max results", "5").option("--project <project>", "project identifier (defaults to git repo root)").option("--all-projects", "search across all projects").option("--kind <kind>", "filter by memory kind");
@@ -5976,6 +6460,24 @@ function sqliteVecFailureDiagnostics(error, dbPath) {
5976
6460
  `error=${message}`
5977
6461
  ];
5978
6462
  }
6463
+ async function runServeCoordinatorMaintenance(store, dependencies) {
6464
+ const projectShares = await dependencies.advancePendingProjectShares(store, { limit: 3 });
6465
+ const coordinatorEnrollment = await dependencies.reconcileConfiguredCoordinatorEnrollment(store);
6466
+ const enrollmentIssues = coordinatorEnrollment.issues ?? 0;
6467
+ const recipientPolicies = await dependencies.reconcileRecipientPolicyProjects(store, { limit: 3 });
6468
+ if (projectShares.failed > 0) throw new Error(`share operation maintenance failed for ${projectShares.failed} of ${projectShares.processed} operations`);
6469
+ if (coordinatorEnrollment.failedGroups > 0 || enrollmentIssues > 0) {
6470
+ const groupLabel = coordinatorEnrollment.failedGroups === 1 ? "group" : "groups";
6471
+ const issueLabel = enrollmentIssues === 1 ? "issue" : "issues";
6472
+ throw new Error(`coordinator enrollment maintenance failed for ${coordinatorEnrollment.failedGroups} ${groupLabel} with ${enrollmentIssues} reconciliation ${issueLabel}`);
6473
+ }
6474
+ if (recipientPolicies.failed > 0) throw new Error(`recipient policy maintenance failed for ${recipientPolicies.failed} of ${recipientPolicies.processed} projects`);
6475
+ return {
6476
+ projectShares,
6477
+ coordinatorEnrollment,
6478
+ recipientPolicies
6479
+ };
6480
+ }
5979
6481
  async function startBackgroundViewer(invocation) {
5980
6482
  warnIfViewerExposed(invocation.host, invocation.port);
5981
6483
  if (await isPortOpen(invocation.host, invocation.port)) {
@@ -6008,7 +6510,7 @@ async function startBackgroundViewer(invocation) {
6008
6510
  p.outro(`Viewer started in background (pid ${child.pid}) at http://${invocation.host}:${invocation.port}`);
6009
6511
  }
6010
6512
  async function startForegroundViewer(invocation) {
6011
- const { createApp, createSyncApp, closeStore, getStore } = await import("@codemem/server");
6513
+ const { advancePendingProjectShares, createApp, createSyncApp, closeStore, getStore, reconcileConfiguredCoordinatorEnrollment, reconcileRecipientPolicyProjects } = await import("@codemem/server");
6012
6514
  const { serve } = await import("@hono/node-server");
6013
6515
  if (invocation.dbPath) process.env.CODEMEM_DB = invocation.dbPath;
6014
6516
  if (invocation.configPath) process.env.CODEMEM_CONFIG = invocation.configPath;
@@ -6099,6 +6601,13 @@ async function startForegroundViewer(invocation) {
6099
6601
  port: syncConfig.syncPort,
6100
6602
  signal: syncAbort.signal,
6101
6603
  scanner: store.scanner,
6604
+ onAfterCoordinatorRefresh: async () => {
6605
+ await runServeCoordinatorMaintenance(store, {
6606
+ advancePendingProjectShares,
6607
+ reconcileConfiguredCoordinatorEnrollment,
6608
+ reconcileRecipientPolicyProjects
6609
+ });
6610
+ },
6102
6611
  onPhaseChange: (phase) => {
6103
6612
  if (phase === "running") {
6104
6613
  syncRuntimeStatus.phase = null;
@@ -6705,9 +7214,38 @@ function fmtTokens(n) {
6705
7214
  var statsCmd = new Command("stats").configureHelp(helpStyle).description("Show database statistics");
6706
7215
  addDbOption(statsCmd);
6707
7216
  addJsonOption(statsCmd);
7217
+ statsCmd.addOption(new Option("-a, --attribution", "show local retrieval attribution diagnostics"));
7218
+ function renderAttributionDiagnostics(report) {
7219
+ const evidence = report.evidenceCompleteness;
7220
+ return [
7221
+ "Retrieval attribution diagnostics (local, bounded)",
7222
+ `Lifecycle: ${report.lifecycle.requestedAttempts} requested, ${report.lifecycle.selectedAttempts} selected, ${report.lifecycle.handedOffAttempts} handed off`,
7223
+ `Exposures: ${report.lifecycle.selectedExposures} selected, ${report.lifecycle.handedOffExposures} handed off`,
7224
+ `Evidence: ${[
7225
+ `${evidence.assessedAttempts} assessed`,
7226
+ `${evidence.unassessedAttempts} unassessed`,
7227
+ `${evidence.assessedKnownAttempts} known`,
7228
+ `${evidence.assessedUnknownAttempts} unknown`,
7229
+ `${evidence.assessmentStatusIndeterminateAttempts} status indeterminate`,
7230
+ `${evidence.assessmentDetailsIncompleteAttempts} details incomplete`,
7231
+ `${evidence.assessmentRowsInvalid} invalid rows`,
7232
+ `${evidence.assessmentRowsOmittedByLimit} omitted by limit`
7233
+ ].join(", ")}`,
7234
+ `Latency: ${report.overhead.latencySampleCount} samples, ${report.overhead.averageLatencyMs ?? "unknown"} ms average`,
7235
+ `Steering: ${report.sourceLocationSteering.assessmentCount} source-location assessments, ${report.sourceLocationSteering.matchedPathCount} matched paths`,
7236
+ `Findings: ${report.findings.stale} stale, ${report.findings.harmful} harmful`,
7237
+ "Limitations:",
7238
+ ...report.limitations.map((limitation) => `- ${limitation}`)
7239
+ ].join("\n");
7240
+ }
6708
7241
  var statsCommand = statsCmd.action((opts) => {
6709
7242
  const store = new MemoryStore(resolveDbPath(resolveDbOpt(opts)));
6710
7243
  try {
7244
+ if (opts.attribution) {
7245
+ const report = getAttributionDiagnostics(store.db);
7246
+ console.log(opts.json ? JSON.stringify(report, null, 2) : renderAttributionDiagnostics(report));
7247
+ return;
7248
+ }
6711
7249
  const result = store.stats();
6712
7250
  if (opts.json) {
6713
7251
  console.log(JSON.stringify(result, null, 2));
@@ -7811,6 +8349,7 @@ program.addCommand(embedCommand);
7811
8349
  program.addCommand(recentCommand);
7812
8350
  program.addCommand(searchCommand);
7813
8351
  program.addCommand(packCommand);
8352
+ program.addCommand(promptPackLedgerCommand, { hidden: true });
7814
8353
  program.addCommand(showMemoryCommand, { hidden: true });
7815
8354
  program.addCommand(forgetMemoryCommand, { hidden: true });
7816
8355
  program.addCommand(rememberMemoryCommand, { hidden: true });