codemem 0.44.0-alpha.1 → 0.44.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, DeviceIdentityError, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, PROMPT_TRANSPORT_PROTOCOL_RANGE, REF_BACKFILL_JOB, RawEventIngestValidationError, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, TRUSTED_HOOK_MAPPER_OPTIONS, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, arePromptTransportProtocolRangesCompatible, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildBaseUrl, buildDirectPeerAuthHeaders, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, buildViewerIdentityTarget, classifyPromptTransportFailure, clonePromptPackAttempt, collectOperationalStatus, compareMemoryRoleReports, connect, connectReadOnly, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, detectInstallKind, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, estimateExtractionModelCost, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getAttributionDiagnostics, getExtractionBenchmarkProfile, getExtractionModelPricing, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getUpdateStatus, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, ingestRawEvents, initDatabase, isEmbeddingDisabled, isStableReleaseVersion, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, normalizePromptTransportProtocolRange, planReplicationOpsAgePrune, probeCodememViewerLiveness, probeCodememViewerLiveness as probeCodememViewerLiveness$1, projectMatchesFilter, promptPackArtifactFingerprint, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, recordPromptPackArtifacts, recordPromptPackTerminal, recordRetrievalSurface, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, resolveRetrievalSession, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, scoreExtractionBenchmarkOutput, setPeerProjectFilter, stripJsonComments, stripTrailingCommas, syncPassPreflight, tryUpdateRetrievalDelivery, updatePeerAddresses, vacuumDatabase, viewerUrl, writeCodememConfigFile } from "@codemem/core";
2
+ import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, DeviceIdentityError, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, ObserverOutputError, ObserverOutputTransportError, PROMPT_TRANSPORT_PROTOCOL_RANGE, REF_BACKFILL_JOB, RawEventIngestValidationError, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, TRUSTED_HOOK_MAPPER_OPTIONS, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, arePromptTransportProtocolRangesCompatible, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildBaseUrl, buildDirectPeerAuthHeaders, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, buildViewerIdentityTarget, classifyPromptTransportFailure, clonePromptPackAttempt, collectOperationalStatus, compareMemoryRoleReports, connect, connectReadOnly, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, detectInstallKind, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, estimateExtractionModelCost, exportMemories, extractApplyPatchPaths, extractionReplayObserverIdentity, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getAttributionDiagnostics, getExtractionBenchmarkProfile, getExtractionModelPricing, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getUpdateStatus, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, ingestRawEvents, initDatabase, isAutomaticRecallMeasurement, isEmbeddingDisabled, isReleaseVersionForChannel, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, normalizePromptTransportProtocolRange, planReplicationOpsAgePrune, probeCodememViewerLiveness, probeCodememViewerLiveness as probeCodememViewerLiveness$1, probeRequiredNativeRuntime, projectMatchesFilter, promptPackArtifactFingerprint, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, recordAutomaticRecall, recordPromptPackArtifacts, recordPromptPackTerminal, recordRetrievalSurface, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, resolveRetrievalSession, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, scoreExtractionBenchmarkOutput, setPeerProjectFilter, stripJsonComments, stripTrailingCommas, syncPassPreflight, tryUpdateRetrievalDelivery, updatePeerAddresses, vacuumDatabase, viewerUrl, writeCodememConfigFile } from "@codemem/core";
3
3
  import { Command, Option } from "commander";
4
4
  import omelette from "omelette";
5
5
  import { randomInt, randomUUID } from "node:crypto";
@@ -4397,7 +4397,7 @@ function startMaintenanceWorkerRuntime(options = {}) {
4397
4397
  dbPath,
4398
4398
  signal: options.signal,
4399
4399
  batchSize: 10,
4400
- idleIntervalMs: 5e3
4400
+ idleIntervalMs: 6e4
4401
4401
  }));
4402
4402
  runners.push(createSequentialBackfillCoordinator(store, [
4403
4403
  {
@@ -5217,6 +5217,163 @@ function summarizeBenchmarkReasoning(runs, fallback) {
5217
5217
  reasoningSummary: reasoningSummaries.size > 1 ? "mixed" : source.reasoningSummary
5218
5218
  };
5219
5219
  }
5220
+ function benchmarkSecondAttemptValue(attempted, value) {
5221
+ return attempted ? value : null;
5222
+ }
5223
+ function benchmarkSecondAttemptList(attempted, value) {
5224
+ return attempted ? value : [];
5225
+ }
5226
+ function buildBenchmarkSecondAttempt(result, input) {
5227
+ return {
5228
+ attempted: input.attempted,
5229
+ reason: input.reason,
5230
+ raw: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedRaw),
5231
+ status: benchmarkSecondAttemptValue(input.attempted, result.repairedClassification?.status ?? null),
5232
+ resultReason: benchmarkSecondAttemptValue(input.attempted, result.repairedClassification?.reason ?? null),
5233
+ pass: benchmarkSecondAttemptValue(input.attempted, result.repairedEvaluation?.pass ?? null),
5234
+ failureReasons: benchmarkSecondAttemptList(input.attempted, result.repairedEvaluation?.failureReasons ?? []),
5235
+ summaries: benchmarkSecondAttemptValue(input.attempted, result.repairedEvaluation?.counts.summaries ?? null),
5236
+ observations: benchmarkSecondAttemptValue(input.attempted, result.repairedEvaluation?.counts.observations ?? null),
5237
+ diagnostics: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedDiagnostics),
5238
+ elapsedMs: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedElapsedMs),
5239
+ usage: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedUsage),
5240
+ quality: benchmarkSecondAttemptValue(input.attempted, input.quality)
5241
+ };
5242
+ }
5243
+ function estimateBenchmarkAttemptCost(model, usage) {
5244
+ if (model === null) return {
5245
+ total: null,
5246
+ unavailableReason: "model_fallback_unresolved"
5247
+ };
5248
+ const total = estimateExtractionModelCost(model, usage);
5249
+ if (total) return {
5250
+ total,
5251
+ unavailableReason: null
5252
+ };
5253
+ if (usage === null) return {
5254
+ total: null,
5255
+ unavailableReason: "missing_usage"
5256
+ };
5257
+ return {
5258
+ total: null,
5259
+ unavailableReason: "unknown_model_pricing"
5260
+ };
5261
+ }
5262
+ function rate(numerator, denominator) {
5263
+ return denominator > 0 ? numerator / denominator : null;
5264
+ }
5265
+ function mean(values) {
5266
+ return values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : null;
5267
+ }
5268
+ function countOutputModes(runs, key) {
5269
+ const counts = {};
5270
+ for (const run of runs) counts[run[key]] = (counts[run[key]] ?? 0) + 1;
5271
+ return counts;
5272
+ }
5273
+ function collectBenchmarkMetricSamples(runs) {
5274
+ return {
5275
+ latencies: runs.flatMap((run) => run.telemetry.totalElapsedMs == null ? [] : [run.telemetry.totalElapsedMs]),
5276
+ usages: runs.flatMap((run) => run.telemetry.totalUsage == null ? [] : [run.telemetry.totalUsage]),
5277
+ summaryScores: runs.flatMap((run) => run.quality == null ? [] : [run.quality.summaryDisposition.score]),
5278
+ structuralScores: runs.flatMap((run) => {
5279
+ if (run.outputValidation === "invalid") return [0];
5280
+ if (run.outputValidation === "valid") return [1];
5281
+ return run.quality == null ? [] : [run.quality.schemaCompliance.score];
5282
+ }),
5283
+ semanticScores: runs.flatMap((run) => run.quality == null ? [] : semanticQualityScore(run.quality))
5284
+ };
5285
+ }
5286
+ function semanticQualityScore(quality) {
5287
+ if (quality.reviewStatus !== "reviewed") return [];
5288
+ const available = [
5289
+ [quality.summaryDisposition.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.summaryDisposition],
5290
+ [quality.requiredRecall?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.requiredRecall],
5291
+ [quality.optionalRecall?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.optionalRecall],
5292
+ [quality.worthinessPrecision?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.worthinessPrecision],
5293
+ [quality.summaryBreadth?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.summaryBreadth],
5294
+ [quality.observationSignals?.redundancyAvoidance, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.redundancyAvoidance],
5295
+ [quality.observationSignals?.segmentation, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.segmentation]
5296
+ ].filter((entry) => entry[0] != null);
5297
+ const weight = available.reduce((sum, [, dimensionWeight]) => sum + dimensionWeight, 0);
5298
+ if (weight === 0) return [];
5299
+ return [available.reduce((sum, [score, dimensionWeight]) => sum + score * dimensionWeight, 0) / weight];
5300
+ }
5301
+ function benchmarkModeMetrics(runs) {
5302
+ const validOutputs = runs.filter((run) => run.outputValidation === "valid" || run.quality?.schemaCompliance.compliant === true).length;
5303
+ const repairAttempts = runs.filter((run) => run.repairAttempted).length;
5304
+ const retryAttempts = runs.filter((run) => run.retryAttempted).length;
5305
+ const repairOrRetryAttempts = runs.filter((run) => run.repairAttempted || run.retryAttempted).length;
5306
+ const { latencies, usages, summaryScores, structuralScores, semanticScores } = collectBenchmarkMetricSamples(runs);
5307
+ const retainedObservations = runs.reduce((sum, run) => sum + run.observations, 0);
5308
+ const matchingSummaryDispositions = summaryScores.filter((score) => score === 1).length;
5309
+ return {
5310
+ runs: runs.length,
5311
+ validOutputs,
5312
+ validOutputRate: rate(validOutputs, runs.length),
5313
+ repairAttempts,
5314
+ repairRate: rate(repairAttempts, runs.length),
5315
+ retryAttempts,
5316
+ retryRate: rate(retryAttempts, runs.length),
5317
+ repairOrRetryAttempts,
5318
+ repairOrRetryRate: rate(repairOrRetryAttempts, runs.length),
5319
+ knownLatencyRuns: latencies.length,
5320
+ totalLatencyMs: latencies.reduce((sum, value) => sum + value, 0),
5321
+ meanLatencyMs: mean(latencies),
5322
+ knownTokenRuns: usages.length,
5323
+ inputTokens: usages.reduce((sum, usage) => sum + usage.inputTokens, 0),
5324
+ outputTokens: usages.reduce((sum, usage) => sum + usage.outputTokens, 0),
5325
+ totalTokens: usages.reduce((sum, usage) => sum + (usage.totalTokens ?? usage.inputTokens + usage.outputTokens), 0),
5326
+ retainedObservations,
5327
+ meanRetainedObservations: rate(retainedObservations, runs.length),
5328
+ summaryDispositionEvaluated: summaryScores.length,
5329
+ summaryDispositionMatches: matchingSummaryDispositions,
5330
+ summaryDispositionMatchRate: rate(matchingSummaryDispositions, summaryScores.length),
5331
+ structuralQualityEvaluated: structuralScores.length,
5332
+ contractIntegrityRate: mean(structuralScores),
5333
+ semanticQualityEvaluated: semanticScores.length,
5334
+ meanSemanticQuality: mean(semanticScores)
5335
+ };
5336
+ }
5337
+ function summarizeExtractionBenchmarkReporting(runs) {
5338
+ const byActualOutputMode = {};
5339
+ for (const mode of [
5340
+ "json_schema",
5341
+ "forced_tool",
5342
+ "legacy_xml"
5343
+ ]) {
5344
+ const modeRuns = runs.filter((run) => run.actualOutputMode === mode);
5345
+ if (modeRuns.length > 0) byActualOutputMode[mode] = benchmarkModeMetrics(modeRuns);
5346
+ }
5347
+ return {
5348
+ requestedOutputModes: countOutputModes(runs, "requestedOutputMode"),
5349
+ actualOutputModes: countOutputModes(runs, "actualOutputMode"),
5350
+ overall: benchmarkModeMetrics(runs),
5351
+ byActualOutputMode
5352
+ };
5353
+ }
5354
+ function summarizeExtractionBenchmarkAttempts(attempts, batches) {
5355
+ return {
5356
+ total: attempts.length,
5357
+ shapeQualityTotal: attempts.filter((attempt) => attempt.purpose === "shape_quality").length,
5358
+ shapeQualityPasses: attempts.filter((attempt) => attempt.purpose === "shape_quality" && attempt.status === "pass").length,
5359
+ shapeQualityFails: attempts.filter((attempt) => attempt.purpose === "shape_quality" && attempt.status !== "pass").length,
5360
+ expectedTierTotal: attempts.filter((attempt) => attempt.expectedTier != null).length,
5361
+ expectedTierMatches: attempts.filter((attempt) => attempt.expectedTier != null && attempt.expectedTier === attempt.tier).length,
5362
+ robustnessNoOutput: attempts.filter((attempt) => attempt.purpose === "replay_robustness" && (attempt.status === "observer_no_output" || attempt.status === "output_failure")).length,
5363
+ perBatchStability: batches.map((batch) => {
5364
+ const batchAttempts = attempts.filter((attempt) => attempt.batchId === batch.batchId).toSorted((left, right) => left.iteration - right.iteration);
5365
+ const passes = batchAttempts.filter((attempt) => attempt.status === "pass").length;
5366
+ return {
5367
+ batchId: batch.batchId,
5368
+ purpose: batch.purpose,
5369
+ passes,
5370
+ total: batchAttempts.length,
5371
+ passRate: rate(passes, batchAttempts.length),
5372
+ statuses: batchAttempts.map((attempt) => attempt.status)
5373
+ };
5374
+ })
5375
+ };
5376
+ }
5220
5377
  function createMemoryExtractionBenchmarkCommand() {
5221
5378
  const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run").option("--repetitions <n>", "run every benchmark batch 1-10 times to measure model stability", "1");
5222
5379
  addDbOption(cmd);
@@ -5255,21 +5412,54 @@ function createMemoryExtractionBenchmarkCommand() {
5255
5412
  observerExplicitConfigKeys: maxOutputTokens === null ? observerConfig.observerExplicitConfigKeys : [.../* @__PURE__ */ new Set([...observerConfig.observerExplicitConfigKeys ?? [], "observerMaxOutputTokens"])]
5256
5413
  };
5257
5414
  const observer = new ObserverClient(observerConfigWithOverrides);
5415
+ const outputFailures = [];
5258
5416
  const runs = [];
5259
5417
  for (let iteration = 1; iteration <= repetitions; iteration += 1) for (const batch of benchmark.batches) {
5260
5418
  const scenarioId = batch.scenarioId ?? benchmark.scenarioId;
5261
- const result = opts.observerTierRouting === true ? await replayBatchExtractionWithTierRouting(resolveDbOpt(opts), observerConfigWithOverrides, {
5262
- batchId: batch.batchId,
5263
- scenarioId,
5264
- transcriptBudget: transcriptBudget ?? void 0
5265
- }) : await replayBatchExtraction(resolveDbOpt(opts), observer, {
5266
- batchId: batch.batchId,
5267
- scenarioId,
5268
- transcriptBudget: transcriptBudget ?? void 0
5269
- });
5419
+ let selectedObserver = extractionReplayObserverIdentity(observer, null);
5420
+ let result;
5421
+ try {
5422
+ result = opts.observerTierRouting === true ? await replayBatchExtractionWithTierRouting(resolveDbOpt(opts), observerConfigWithOverrides, {
5423
+ batchId: batch.batchId,
5424
+ scenarioId,
5425
+ transcriptBudget: transcriptBudget ?? void 0,
5426
+ onOutputFailure: (context) => {
5427
+ selectedObserver = context;
5428
+ }
5429
+ }) : await replayBatchExtraction(resolveDbOpt(opts), observer, {
5430
+ batchId: batch.batchId,
5431
+ scenarioId,
5432
+ transcriptBudget: transcriptBudget ?? void 0
5433
+ });
5434
+ } catch (error) {
5435
+ if (!(error instanceof ObserverOutputError) && !(error instanceof ObserverOutputTransportError)) throw error;
5436
+ if (opts.observerTierRouting !== true) selectedObserver = extractionReplayObserverIdentity(observer, null);
5437
+ const failureCost = estimateBenchmarkAttemptCost(selectedObserver.model, error.telemetry.totalUsage);
5438
+ outputFailures.push({
5439
+ iteration,
5440
+ batchId: batch.batchId,
5441
+ purpose: batch.purpose,
5442
+ status: "output_failure",
5443
+ reason: error instanceof ObserverOutputError ? error.reason : error.code,
5444
+ expectedTier: batch.expectedTier ?? null,
5445
+ ...selectedObserver,
5446
+ requestedOutputMode: error.diagnostics.requestedMode,
5447
+ actualOutputMode: error.diagnostics.actualMode,
5448
+ outputValidation: "invalid",
5449
+ repairAttempted: error.diagnostics.repairAttempted,
5450
+ retryAttempted: error.diagnostics.retryAttempted,
5451
+ observations: 0,
5452
+ telemetry: error.telemetry,
5453
+ cost: failureCost,
5454
+ quality: null
5455
+ });
5456
+ continue;
5457
+ }
5270
5458
  const costModel = result.observer.modelFallbackApplied ? result.observer.resolvedModel : result.observer.resolvedModel ?? result.observer.model;
5271
5459
  const initialCost = costModel ? estimateExtractionModelCost(costModel, result.observer.initialUsage) : null;
5272
- const repairCost = costModel ? estimateExtractionModelCost(costModel, result.observer.repairedUsage) : null;
5460
+ const secondAttemptCost = costModel ? estimateExtractionModelCost(costModel, result.observer.repairedUsage) : null;
5461
+ const repairCost = benchmarkSecondAttemptValue(result.observer.repairAttempted, secondAttemptCost);
5462
+ const retryCost = benchmarkSecondAttemptValue(result.observer.retryAttempted, secondAttemptCost);
5273
5463
  const totalCost = costModel ? estimateExtractionModelCost(costModel, result.observer.totalUsage) : null;
5274
5464
  const pricing = costModel ? getExtractionModelPricing(costModel) : null;
5275
5465
  const costUnavailableReason = totalCost ? null : result.observer.modelFallbackApplied && !result.observer.resolvedModel ? "model_fallback_unresolved" : result.observer.totalUsage == null ? "missing_usage" : "unknown_model_pricing";
@@ -5283,14 +5473,14 @@ function createMemoryExtractionBenchmarkCommand() {
5283
5473
  estimatedCostUsd: initialCost?.totalCostUsd ?? null,
5284
5474
  expectedSummaryDisposition: batch.expectedSummaryDisposition
5285
5475
  }) : null;
5286
- const repairQuality = result.observer.repairedParsed && result.observer.repairedDiagnostics ? scoreExtractionBenchmarkOutput({
5476
+ const secondAttemptQuality = result.observer.repairedParsed && result.observer.repairedDiagnostics ? scoreExtractionBenchmarkOutput({
5287
5477
  parsed: result.observer.repairedParsed,
5288
5478
  diagnostics: result.observer.repairedDiagnostics,
5289
5479
  review: batch.review ?? {
5290
5480
  status: "unreviewed",
5291
5481
  reviewerNotes: "No durable-fact review has been recorded for this batch."
5292
5482
  },
5293
- estimatedCostUsd: repairCost?.totalCostUsd ?? null,
5483
+ estimatedCostUsd: secondAttemptCost?.totalCostUsd ?? null,
5294
5484
  expectedSummaryDisposition: batch.expectedSummaryDisposition
5295
5485
  }) : null;
5296
5486
  const finalQuality = result.observer.diagnostics ? scoreExtractionBenchmarkOutput({
@@ -5310,6 +5500,16 @@ function createMemoryExtractionBenchmarkCommand() {
5310
5500
  initialQuality,
5311
5501
  finalQuality
5312
5502
  });
5503
+ const repairAttempt = buildBenchmarkSecondAttempt(result, {
5504
+ attempted: result.observer.repairAttempted,
5505
+ reason: null,
5506
+ quality: secondAttemptQuality
5507
+ });
5508
+ const retryAttempt = buildBenchmarkSecondAttempt(result, {
5509
+ attempted: result.observer.retryAttempted,
5510
+ reason: result.observer.retryReason,
5511
+ quality: secondAttemptQuality
5512
+ });
5313
5513
  runs.push({
5314
5514
  iteration,
5315
5515
  batchId: batch.batchId,
@@ -5336,6 +5536,9 @@ function createMemoryExtractionBenchmarkCommand() {
5336
5536
  resolvedModel: result.observer.resolvedModel,
5337
5537
  modelFallbackApplied: result.observer.modelFallbackApplied,
5338
5538
  modelFallbackReason: result.observer.modelFallbackReason,
5539
+ requestedOutputMode: result.observer.requestedOutputMode,
5540
+ actualOutputMode: result.observer.actualOutputMode,
5541
+ outputValidation: result.observer.outputValidation,
5339
5542
  openaiUseResponses: result.observer.openaiUseResponses,
5340
5543
  reasoningEffort: result.observer.reasoningEffort,
5341
5544
  reasoningSummary: result.observer.reasoningSummary,
@@ -5344,6 +5547,9 @@ function createMemoryExtractionBenchmarkCommand() {
5344
5547
  summaries: result.evaluation.counts.summaries,
5345
5548
  observations: result.evaluation.counts.observations,
5346
5549
  repairApplied: result.observer.repairApplied,
5550
+ repairAttempted: result.observer.repairAttempted,
5551
+ retryAttempted: result.observer.retryAttempted,
5552
+ retryReason: result.observer.retryReason,
5347
5553
  initial: {
5348
5554
  raw: result.observer.initialRaw,
5349
5555
  status: result.initialClassification.status,
@@ -5359,18 +5565,19 @@ function createMemoryExtractionBenchmarkCommand() {
5359
5565
  },
5360
5566
  repair: {
5361
5567
  applied: result.observer.repairApplied,
5362
- raw: result.observer.repairedRaw,
5363
- status: result.repairedClassification?.status ?? null,
5364
- reason: result.repairedClassification?.reason ?? null,
5365
- pass: result.repairedEvaluation?.pass ?? null,
5366
- failureReasons: result.repairedEvaluation?.failureReasons ?? [],
5367
- summaries: result.repairedEvaluation?.counts.summaries ?? null,
5368
- observations: result.repairedEvaluation?.counts.observations ?? null,
5369
- diagnostics: result.observer.repairedDiagnostics,
5370
- elapsedMs: result.observer.repairedElapsedMs,
5371
- usage: result.observer.repairedUsage,
5372
- quality: repairQuality
5568
+ raw: repairAttempt.raw,
5569
+ status: repairAttempt.status,
5570
+ reason: repairAttempt.resultReason,
5571
+ pass: repairAttempt.pass,
5572
+ failureReasons: repairAttempt.failureReasons,
5573
+ summaries: repairAttempt.summaries,
5574
+ observations: repairAttempt.observations,
5575
+ diagnostics: repairAttempt.diagnostics,
5576
+ elapsedMs: repairAttempt.elapsedMs,
5577
+ usage: repairAttempt.usage,
5578
+ quality: repairAttempt.quality
5373
5579
  },
5580
+ retry: retryAttempt,
5374
5581
  telemetry: {
5375
5582
  totalElapsedMs: result.observer.totalElapsedMs,
5376
5583
  totalUsage: result.observer.totalUsage
@@ -5379,6 +5586,7 @@ function createMemoryExtractionBenchmarkCommand() {
5379
5586
  cost: {
5380
5587
  initial: initialCost,
5381
5588
  repair: repairCost,
5589
+ retry: retryCost,
5382
5590
  total: totalCost,
5383
5591
  unavailableReason: costUnavailableReason
5384
5592
  },
@@ -5386,68 +5594,55 @@ function createMemoryExtractionBenchmarkCommand() {
5386
5594
  });
5387
5595
  }
5388
5596
  const reviewedQualityRuns = runs.filter((run) => run.quality?.weightedQualityScore != null);
5389
- const knownCostRuns = runs.filter((run) => run.cost.total != null);
5390
- const knownElapsedRuns = runs.filter((run) => run.telemetry.totalElapsedMs != null);
5597
+ const attempts = [...runs, ...outputFailures];
5598
+ const knownCostRuns = attempts.filter((run) => run.cost.total != null);
5599
+ const attemptSummary = summarizeExtractionBenchmarkAttempts(attempts, benchmark.batches);
5600
+ const knownElapsedRuns = attempts.filter((run) => run.telemetry.totalElapsedMs != null);
5391
5601
  const summary = {
5392
5602
  repetitions,
5393
- total: runs.length,
5394
- shapeQualityTotal: runs.filter((run) => run.purpose === "shape_quality").length,
5395
- shapeQualityPasses: runs.filter((run) => run.purpose === "shape_quality" && run.status === "pass").length,
5396
- shapeQualityFails: runs.filter((run) => run.purpose === "shape_quality" && run.status === "shape_fail").length,
5397
- expectedTierTotal: runs.filter((run) => run.expectedTier != null).length,
5398
- expectedTierMatches: runs.filter((run) => run.expectedTier != null && run.expectedTier === run.tier).length,
5399
- robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length,
5603
+ ...attemptSummary,
5400
5604
  summaryDispositionTotal: runs.filter((run) => run.quality != null).length,
5401
5605
  summaryDispositionMatches: runs.filter((run) => run.quality?.summaryDisposition.score === 1).length,
5402
5606
  reviewedQualityRuns: reviewedQualityRuns.length,
5403
5607
  knownCostRuns: knownCostRuns.length,
5404
- unknownCostRuns: runs.length - knownCostRuns.length,
5405
- missingUsageRuns: runs.filter((run) => run.cost.unavailableReason === "missing_usage").length,
5406
- unknownPricingRuns: runs.filter((run) => run.cost.unavailableReason === "unknown_model_pricing").length,
5407
- fallbackUnresolvedRuns: runs.filter((run) => run.cost.unavailableReason === "model_fallback_unresolved").length,
5608
+ unknownCostRuns: attempts.length - knownCostRuns.length,
5609
+ missingUsageRuns: attempts.filter((run) => run.cost.unavailableReason === "missing_usage").length,
5610
+ unknownPricingRuns: attempts.filter((run) => run.cost.unavailableReason === "unknown_model_pricing").length,
5611
+ fallbackUnresolvedRuns: attempts.filter((run) => run.cost.unavailableReason === "model_fallback_unresolved").length,
5408
5612
  totalKnownCostUsd: knownCostRuns.reduce((sum, run) => sum + (run.cost.total?.totalCostUsd ?? 0), 0),
5409
5613
  knownElapsedRuns: knownElapsedRuns.length,
5410
5614
  totalKnownElapsedMs: knownElapsedRuns.reduce((sum, run) => sum + (run.telemetry.totalElapsedMs ?? 0), 0),
5411
- perBatchStability: benchmark.batches.map((batch) => {
5412
- const batchRuns = runs.filter((run) => run.batchId === batch.batchId);
5413
- const passes = batchRuns.filter((run) => run.status === "pass").length;
5414
- return {
5415
- batchId: batch.batchId,
5416
- purpose: batch.purpose,
5417
- passes,
5418
- total: batchRuns.length,
5419
- passRate: batchRuns.length > 0 ? passes / batchRuns.length : null,
5420
- statuses: batchRuns.map((run) => run.status)
5421
- };
5422
- })
5615
+ output: summarizeExtractionBenchmarkReporting(attempts),
5616
+ outputFailures
5423
5617
  };
5424
- const uniqueObserverKeys = Array.from(new Set(runs.map((run) => `${run.provider}::${run.model}::${run.transport}`)));
5425
- const benchmarkReasoning = summarizeBenchmarkReasoning(runs, {
5618
+ const uniqueObserverKeys = Array.from(new Set(attempts.map((run) => `${run.provider}::${run.model}::${run.transport}`)));
5619
+ const firstObserver = attempts[0];
5620
+ const benchmarkReasoning = summarizeBenchmarkReasoning(attempts, {
5426
5621
  reasoningEffort: observer.reasoningEffort,
5427
5622
  reasoningSummary: observer.reasoningSummary
5428
5623
  });
5429
5624
  const observerSummary = opts.observerTierRouting === true ? {
5430
- provider: uniqueObserverKeys.length === 1 ? runs[0]?.provider ?? observer.provider : "mixed",
5431
- model: uniqueObserverKeys.length === 1 ? runs[0]?.model ?? observer.model : "mixed",
5432
- transport: uniqueObserverKeys.length === 1 ? runs[0]?.transport ?? "unknown" : "mixed",
5625
+ provider: uniqueObserverKeys.length === 1 ? firstObserver?.provider ?? observer.provider : "mixed",
5626
+ model: uniqueObserverKeys.length === 1 ? firstObserver?.model ?? observer.model : "mixed",
5627
+ transport: uniqueObserverKeys.length === 1 ? firstObserver?.transport ?? "unknown" : "mixed",
5433
5628
  tierRouting: true,
5434
- openaiUseResponses: uniqueObserverKeys.length === 1 ? runs[0]?.openaiUseResponses ?? observer.openaiUseResponses : null,
5629
+ openaiUseResponses: uniqueObserverKeys.length === 1 ? firstObserver?.openaiUseResponses ?? observer.openaiUseResponses : null,
5435
5630
  reasoningEffort: benchmarkReasoning.reasoningEffort,
5436
5631
  reasoningSummary: benchmarkReasoning.reasoningSummary,
5437
- maxOutputTokens: uniqueObserverKeys.length === 1 ? runs[0]?.maxOutputTokens ?? null : null,
5438
- temperature: uniqueObserverKeys.length === 1 ? runs[0]?.temperature ?? null : null,
5632
+ maxOutputTokens: uniqueObserverKeys.length === 1 ? firstObserver?.maxOutputTokens ?? null : null,
5633
+ temperature: uniqueObserverKeys.length === 1 ? firstObserver?.temperature ?? null : null,
5439
5634
  transcriptBudget: transcriptBudget ?? null,
5440
5635
  selectedObservers: uniqueObserverKeys
5441
5636
  } : {
5442
5637
  provider: observer.provider,
5443
5638
  model: observer.model,
5444
- transport: runs[0]?.transport ?? observer.getStatus().runtime,
5639
+ transport: firstObserver?.transport ?? observer.getStatus().runtime,
5445
5640
  tierRouting: false,
5446
5641
  openaiUseResponses: observer.openaiUseResponses,
5447
5642
  reasoningEffort: benchmarkReasoning.reasoningEffort,
5448
5643
  reasoningSummary: benchmarkReasoning.reasoningSummary,
5449
- maxOutputTokens: runs[0]?.maxOutputTokens ?? null,
5450
- temperature: runs[0]?.temperature ?? null,
5644
+ maxOutputTokens: firstObserver?.maxOutputTokens ?? null,
5645
+ temperature: firstObserver?.temperature ?? null,
5451
5646
  transcriptBudget: transcriptBudget ?? null,
5452
5647
  selectedObservers: uniqueObserverKeys
5453
5648
  };
@@ -5486,14 +5681,23 @@ function createMemoryExtractionBenchmarkCommand() {
5486
5681
  `Summary disposition matches: ${summary.summaryDispositionMatches}/${summary.summaryDispositionTotal}`,
5487
5682
  `Reviewed quality runs: ${summary.reviewedQualityRuns} (compare per-run dimensions; scores are fixture-specific)`,
5488
5683
  `Known estimated cost: $${summary.totalKnownCostUsd.toFixed(6)} (${summary.knownCostRuns}/${summary.total}; missing usage=${summary.missingUsageRuns}, unknown pricing=${summary.unknownPricingRuns}, unresolved fallback=${summary.fallbackUnresolvedRuns})`,
5489
- `Known elapsed time: ${summary.totalKnownElapsedMs}ms (${summary.knownElapsedRuns}/${summary.total} run(s))`
5684
+ `Known elapsed time: ${summary.totalKnownElapsedMs}ms (${summary.knownElapsedRuns}/${summary.total} run(s))`,
5685
+ `Output modes requested/actual: ${JSON.stringify(summary.output.requestedOutputModes)} / ${JSON.stringify(summary.output.actualOutputModes)}`,
5686
+ `Valid output: ${summary.output.overall.validOutputs}/${summary.output.overall.runs}; repair/retry: ${summary.output.overall.repairAttempts}/${summary.output.overall.retryAttempts}; mean latency: ${summary.output.overall.meanLatencyMs ?? "n/a"}ms`,
5687
+ `Tokens input/output: ${summary.output.overall.inputTokens}/${summary.output.overall.outputTokens} (${summary.output.overall.knownTokenRuns} measured run(s)); retained observations: ${summary.output.overall.retainedObservations}`,
5688
+ `Summary disposition matches: ${summary.output.overall.summaryDispositionMatches}/${summary.output.overall.summaryDispositionEvaluated}; contract integrity/semantic quality: ${summary.output.overall.contractIntegrityRate?.toFixed(3) ?? "n/a"}/${summary.output.overall.meanSemanticQuality?.toFixed(3) ?? "n/a"}`
5490
5689
  ].join("\n"));
5491
5690
  for (const run of runs) {
5492
5691
  const qualityLabel = run.quality?.weightedQualityScore == null ? "n/a" : run.quality.weightedQualityScore.toFixed(3);
5493
5692
  const costLabel = run.cost.total == null ? "n/a" : `$${run.cost.total.totalCostUsd.toFixed(6)}`;
5494
5693
  const latencyLabel = run.telemetry.totalElapsedMs == null ? "n/a" : `${run.telemetry.totalElapsedMs}ms`;
5495
5694
  const missingRequired = run.quality?.requiredRecall.missingLabelIds.join(",") || "none";
5496
- p.log.message(` [${run.batchId}#${run.iteration}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} disposition=${run.quality?.summaryDisposition.actual ?? "n/a"}/${run.expectedSummaryDisposition} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model} [${run.transport}] initial=${run.initial.summaries}s/${run.initial.observations}o final=${run.summaries}s/${run.observations}o quality=${qualityLabel} coverage=${run.quality?.weightedQualityCoverage?.toFixed(3) ?? "n/a"} required_missing=${missingRequired} cost=${costLabel} latency=${latencyLabel} schema_loss=${run.initial.diagnostics?.dataLoss === true ? "yes" : "no"} fallback=${run.modelFallbackApplied ? "yes" : "no"} repair=${run.repairApplied ? "yes" : "no"} — ${run.label}`);
5695
+ p.log.message(` [${run.batchId}#${run.iteration}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} mode=${run.requestedOutputMode}/${run.actualOutputMode} valid=${run.outputValidation} disposition=${run.quality?.summaryDisposition.actual ?? "n/a"}/${run.expectedSummaryDisposition} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model} [${run.transport}] initial=${run.initial.summaries}s/${run.initial.observations}o final=${run.summaries}s/${run.observations}o quality=${qualityLabel} coverage=${run.quality?.weightedQualityCoverage?.toFixed(3) ?? "n/a"} required_missing=${missingRequired} cost=${costLabel} latency=${latencyLabel} schema_loss=${run.initial.diagnostics?.dataLoss === true ? "yes" : "no"} fallback=${run.modelFallbackApplied ? "yes" : "no"} repair=${run.repairApplied ? "yes" : "no"} retry=${run.retryAttempted ? run.retryReason ?? "yes" : "no"} — ${run.label}`);
5696
+ }
5697
+ for (const failure of outputFailures.toSorted((left, right) => left.iteration - right.iteration || left.batchId - right.batchId)) {
5698
+ const costLabel = failure.cost.total == null ? "n/a" : `$${failure.cost.total.totalCostUsd.toFixed(6)}`;
5699
+ const latencyLabel = failure.telemetry.totalElapsedMs == null ? "n/a" : `${failure.telemetry.totalElapsedMs}ms`;
5700
+ p.log.error(` [${failure.batchId}#${failure.iteration}] output_failure tier=${failure.tier ?? "n/a"} expected=${failure.expectedTier ?? "n/a"} mode=${failure.requestedOutputMode}/${failure.actualOutputMode} model=${failure.model ?? "unresolved"} cost=${costLabel} latency=${latencyLabel} — ${failure.reason}`);
5497
5701
  }
5498
5702
  p.outro("done");
5499
5703
  } catch (error) {
@@ -5609,6 +5813,8 @@ var FORBIDDEN_INTERNAL_KEYS = /* @__PURE__ */ new Set([
5609
5813
  "title"
5610
5814
  ]);
5611
5815
  var INTERNAL_LEDGER_KEYS = /* @__PURE__ */ new Set([
5816
+ "automatic_recall",
5817
+ "evaluation_key",
5612
5818
  "action",
5613
5819
  "attempt_id",
5614
5820
  "started_at",
@@ -5626,7 +5832,8 @@ var INTERNAL_LEDGER_KEYS = /* @__PURE__ */ new Set([
5626
5832
  var INTERNAL_LEDGER_ACTIONS = /* @__PURE__ */ new Set([
5627
5833
  "record",
5628
5834
  "delivery",
5629
- "cache_reuse"
5835
+ "cache_reuse",
5836
+ "recall"
5630
5837
  ]);
5631
5838
  var INTERNAL_RETRIEVAL_STATUSES = /* @__PURE__ */ new Set(["skipped", "failed"]);
5632
5839
  var INTERNAL_DELIVERY_STATUSES = /* @__PURE__ */ new Set([
@@ -5785,12 +5992,28 @@ async function withStore(opts, errorCode, run) {
5785
5992
  store?.close();
5786
5993
  }
5787
5994
  }
5995
+ function ledgerAutomaticContext(payload) {
5996
+ if (payload?.source?.trim().toLowerCase() !== "opencode") return null;
5997
+ if (!payload.source_session_id) return null;
5998
+ return {
5999
+ source: "opencode",
6000
+ hostSessionId: payload.source_session_id
6001
+ };
6002
+ }
6003
+ async function readOptionalLedgerPayload() {
6004
+ try {
6005
+ return await readInternalLedgerPayload();
6006
+ } catch {
6007
+ return;
6008
+ }
6009
+ }
5788
6010
  async function packAction(context, opts) {
5789
6011
  await withStore(opts, "pack_failed", async (store) => {
5790
6012
  const { limit, budget, filters, renderOptions } = buildPackRequestOptions(opts, { envProject: process.env.CODEMEM_PROJECT });
5791
6013
  let result;
5792
6014
  if (opts.internalLedger) {
5793
- const artifacts = await store.buildMemoryPackWithTraceAsync(context, limit, budget, filters, renderOptions);
6015
+ const ledgerPayload = await readOptionalLedgerPayload();
6016
+ const artifacts = await store.buildMemoryPackWithTraceAsync(context, limit, budget, filters, renderOptions, ledgerAutomaticContext(ledgerPayload));
5794
6017
  result = artifacts.response;
5795
6018
  let artifactFingerprint;
5796
6019
  try {
@@ -5798,12 +6021,13 @@ async function packAction(context, opts) {
5798
6021
  } catch {}
5799
6022
  let ledgerOutcome;
5800
6023
  try {
5801
- const ledgerPayload = await readInternalLedgerPayload();
6024
+ if (!ledgerPayload) throw new Error("internal ledger metadata unavailable");
5802
6025
  ledgerOutcome = handleInstrumentedPackLedger(store.db, ledgerPayload, context, filters, artifacts);
5803
6026
  } catch {}
5804
6027
  emitPackResult(context, opts, result, artifactFingerprint, ledgerOutcome?.ok === false && ledgerOutcome.reason === "idempotency_conflict" ? ledgerOutcome : void 0);
5805
6028
  return;
5806
- } else result = await store.buildMemoryPackAsync(context, limit, budget, filters, renderOptions);
6029
+ }
6030
+ result = await store.buildMemoryPackAsync(context, limit, budget, filters, renderOptions);
5807
6031
  emitPackResult(context, opts, result);
5808
6032
  });
5809
6033
  }
@@ -5850,7 +6074,30 @@ addJsonOption(traceCmd);
5850
6074
  traceCmd.action(traceAction);
5851
6075
  packCmd.addCommand(traceCmd);
5852
6076
  var packCommand = packCmd;
6077
+ function handleRecallMeasurement(db, payload) {
6078
+ if (payload.automatic_recall === void 0 && payload.action !== "recall") return null;
6079
+ const isValidMeasurement = isAutomaticRecallMeasurement(payload.automatic_recall) && typeof payload.evaluation_key === "string" && /^[a-f0-9]{64}$/.test(payload.evaluation_key);
6080
+ if (payload.action === "delivery") {
6081
+ if (!isValidMeasurement) return null;
6082
+ try {
6083
+ recordAutomaticRecall(db, payload.attempt_id, payload.evaluation_key, payload.automatic_recall);
6084
+ } catch {}
6085
+ return null;
6086
+ }
6087
+ if (payload.action !== "recall" || !isValidMeasurement) throw new PackUsageError("invalid automatic recall measurement");
6088
+ const outcome = recordAutomaticRecall(db, payload.attempt_id, payload.evaluation_key, payload.automatic_recall);
6089
+ if (!outcome.ok) throw new PackUsageError(outcome.reason);
6090
+ return outcome.value;
6091
+ }
5853
6092
  function handlePromptPackLedger(db, payload) {
6093
+ if (payload.action === "delivery") {
6094
+ const outcome = handlePromptPackLifecycle(db, payload);
6095
+ handleRecallMeasurement(db, payload);
6096
+ return outcome;
6097
+ }
6098
+ return handleRecallMeasurement(db, payload) ?? handlePromptPackLifecycle(db, payload);
6099
+ }
6100
+ function handlePromptPackLifecycle(db, payload) {
5854
6101
  const metadata = attemptMetadata(payload);
5855
6102
  if (payload.action === "delivery") {
5856
6103
  const status = payload.delivery_status;
@@ -6460,6 +6707,51 @@ function sqliteVecFailureDiagnostics(error, dbPath) {
6460
6707
  `error=${message}`
6461
6708
  ];
6462
6709
  }
6710
+ function requiredNativeRuntimeFailureDiagnostics(error, context = {
6711
+ version: VERSION,
6712
+ nodeVersion: process.version,
6713
+ platform: process.platform,
6714
+ arch: process.arch
6715
+ }) {
6716
+ const detail = error instanceof Error ? error.message : String(error);
6717
+ return [
6718
+ "Required SQLite native binding is unavailable.",
6719
+ `Runtime: Node ${context.nodeVersion} on ${context.platform}/${context.arch}.`,
6720
+ `Details: ${detail}`,
6721
+ `Reinstall this exact codemem version: npm install -g codemem@${context.version}`,
6722
+ "Use Node 24.15+ on a supported 64-bit macOS, Linux, or Windows target; see the native install matrix for current platform coverage.",
6723
+ "This required better-sqlite3 failure is separate from optional sqlite-vec loading, which can fall back to lexical search."
6724
+ ];
6725
+ }
6726
+ async function preflightServeNativeRuntime(invocation, dependencies = {
6727
+ isPortOpen,
6728
+ probe: probeRequiredNativeRuntime
6729
+ }) {
6730
+ if (invocation.mode === "start" && await dependencies.isPortOpen(invocation.host, invocation.port)) return { state: "already_running" };
6731
+ try {
6732
+ dependencies.probe();
6733
+ return { state: "ready" };
6734
+ } catch (error) {
6735
+ return {
6736
+ state: "unavailable",
6737
+ diagnostics: requiredNativeRuntimeFailureDiagnostics(error)
6738
+ };
6739
+ }
6740
+ }
6741
+ async function nativeRuntimeAllowsServeStart(invocation) {
6742
+ if (invocation.mode !== "start" && invocation.mode !== "restart") return true;
6743
+ const preflight = await preflightServeNativeRuntime(invocation);
6744
+ if (preflight.state === "already_running") {
6745
+ p.log.warn(`Viewer already running at http://${invocation.host}:${invocation.port}`);
6746
+ if (!invocation.background) process.exitCode = 1;
6747
+ return false;
6748
+ }
6749
+ if (preflight.state === "ready") return true;
6750
+ p.intro("codemem viewer");
6751
+ for (const line of preflight.diagnostics) p.log.error(line);
6752
+ process.exitCode = 1;
6753
+ return false;
6754
+ }
6463
6755
  async function runServeCoordinatorMaintenance(store, dependencies) {
6464
6756
  const projectShares = await dependencies.advancePendingProjectShares(store, { limit: 3 });
6465
6757
  const coordinatorEnrollment = await dependencies.reconcileConfiguredCoordinatorEnrollment(store);
@@ -6518,11 +6810,11 @@ async function startForegroundViewer(invocation) {
6518
6810
  if (invocation.dbPath) process.env.CODEMEM_DB = invocation.dbPath;
6519
6811
  if (invocation.configPath) process.env.CODEMEM_CONFIG = invocation.configPath;
6520
6812
  warnIfViewerExposed(invocation.host, invocation.port);
6521
- if (await isPortOpen(invocation.host, invocation.port)) {
6522
- p.log.warn(`Viewer already running at http://${invocation.host}:${invocation.port}`);
6523
- process.exitCode = 1;
6524
- return;
6525
- }
6813
+ const dbPath = resolveDbPath(invocation.dbPath ?? void 0);
6814
+ if (!await terminateTrustedMaintenanceWorker(dbPath, {
6815
+ gracefulMs: 1e3,
6816
+ forceMs: 5e3
6817
+ })) p.log.warn("Existing maintenance worker is not trusted or did not stop; starting viewer anyway");
6526
6818
  const preparedDb = prepareViewerDatabase(invocation.dbPath);
6527
6819
  const observer = new ObserverClient();
6528
6820
  let store;
@@ -6543,11 +6835,6 @@ async function startForegroundViewer(invocation) {
6543
6835
  const config = invocation.configPath ? readCodememConfigFileAtPath(invocation.configPath) : readCodememConfigFile();
6544
6836
  const syncConfig = readCoordinatorSyncConfig(config);
6545
6837
  const syncEnabled = syncConfig.syncEnabled;
6546
- const dbPath = resolveDbPath(invocation.dbPath ?? void 0);
6547
- if (!await terminateTrustedMaintenanceWorker(dbPath, {
6548
- gracefulMs: 1e3,
6549
- forceMs: 5e3
6550
- })) p.log.warn("Existing maintenance worker is not trusted or did not stop; starting viewer anyway");
6551
6838
  const syncRuntimeStatus = {
6552
6839
  phase: syncEnabled ? "starting" : "disabled",
6553
6840
  detail: syncEnabled ? "Waiting for viewer startup to finish" : "Sync is disabled"
@@ -6683,7 +6970,10 @@ async function startForegroundViewer(invocation) {
6683
6970
  shutdown();
6684
6971
  });
6685
6972
  }
6686
- async function runServeInvocation(invocation) {
6973
+ async function runServeInvocation(invocation, dependencies = {
6974
+ nativeRuntimeAllowsServeStart,
6975
+ stopExistingViewer
6976
+ }) {
6687
6977
  const dbPath = resolveDbPath(invocation.dbPath ?? void 0);
6688
6978
  const runtimeConflict = await findRuntimeViewerConflict(dbPath, {
6689
6979
  host: invocation.host,
@@ -6696,8 +6986,9 @@ async function runServeInvocation(invocation) {
6696
6986
  process.exitCode = 1;
6697
6987
  return;
6698
6988
  }
6989
+ if (!await dependencies.nativeRuntimeAllowsServeStart(invocation)) return;
6699
6990
  if (invocation.mode === "stop" || invocation.mode === "restart") {
6700
- const result = await stopExistingViewer(dbPath, {
6991
+ const result = await dependencies.stopExistingViewer(dbPath, {
6701
6992
  host: invocation.host,
6702
6993
  port: invocation.port
6703
6994
  });
@@ -7337,7 +7628,7 @@ function installCodex(force) {
7337
7628
  }
7338
7629
  if (codememOnPath(true)) p.log.info("Codex hooks will call `codemem` directly (found on PATH).");
7339
7630
  else {
7340
- const globalInstallCommand = process.platform === "linux" ? "env ONNXRUNTIME_NODE_INSTALL=skip npm install -g codemem @codemem/embeddings" : "npm install -g codemem @codemem/embeddings";
7631
+ const globalInstallCommand = process.platform === "linux" ? "env ONNXRUNTIME_NODE_INSTALL=skip npm install -g codemem" : "npm install -g codemem";
7341
7632
  p.log.info(`\`codemem\` is not on PATH, so Codex hooks will run via \`npx -y --package codemem --package @codemem/embeddings codemem\` (works without a global install). For lower hook latency: ${globalInstallCommand}`);
7342
7633
  }
7343
7634
  let ok = true;
@@ -8970,10 +9261,10 @@ function renderStatusWarning(error) {
8970
9261
  function renderHumanStatus(status) {
8971
9262
  const warning = renderStatusWarning(status.error);
8972
9263
  if (status.update_available) return `Update available${cacheQualifier(status.stale)}: ${status.current_version} → ${status.latest_version}. ${status.recommended_action}${warning}`;
8973
- if (!isStableReleaseVersion(status.current_version)) return `Unable to compare current version ${status.current_version} with ${status.latest_version}. ${status.recommended_action}${warning}`;
9264
+ if (!status.channel || !isReleaseVersionForChannel(status.current_version, status.channel)) return `Unable to compare current version ${status.current_version} with ${status.latest_version}. ${status.recommended_action}${warning}`;
8974
9265
  return `${status.current_version} is up to date${cacheQualifier(status.stale)}.${warning}`;
8975
9266
  }
8976
- var checkCommand = addJsonOption(new Command("check").description("Check for a newer stable codemem release")).addOption(new Option("-r, --refresh", "bypass the six-hour release cache")).configureHelp(helpStyle).action(async (options) => {
9267
+ var checkCommand = addJsonOption(new Command("check").description("Check for a newer codemem release on the installed channel")).addOption(new Option("-r, --refresh", "bypass the six-hour release cache")).configureHelp(helpStyle).action(async (options) => {
8977
9268
  try {
8978
9269
  const installKind = detectInstallKind({
8979
9270
  entryPath: process.argv[1] ?? "",
@@ -9037,8 +9328,8 @@ async function installUpdate(options) {
9037
9328
  return;
9038
9329
  }
9039
9330
  const targetVersion = status.latest_version;
9040
- if (!isStableReleaseVersion(targetVersion)) {
9041
- failInstall(options, "update_install_refused", "release version is not a stable semantic version");
9331
+ if (!status.channel || !isReleaseVersionForChannel(targetVersion, status.channel)) {
9332
+ failInstall(options, "update_install_refused", "release version does not match the installed channel");
9042
9333
  return;
9043
9334
  }
9044
9335
  releaseInstallLock = await acquireInstallLock();
@@ -9076,7 +9367,7 @@ async function installUpdate(options) {
9076
9367
  await releaseInstallLock?.();
9077
9368
  }
9078
9369
  }
9079
- var installCommand = addJsonOption(new Command("install").description("Install an eligible stable codemem update")).configureHelp(helpStyle).action(installUpdate);
9370
+ var installCommand = addJsonOption(new Command("install").description("Install an eligible codemem update on the installed channel")).configureHelp(helpStyle).action(installUpdate);
9080
9371
  var updateCommand = new Command("update").description("Inspect and manage codemem updates").enablePositionalOptions().configureHelp(helpStyle).addCommand(checkCommand).addCommand(installCommand);
9081
9372
  //#endregion
9082
9373
  //#region src/commands/version.ts