codemem 0.39.0 → 0.40.0-alpha.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -1,5 +1,5 @@
1
1
  #!/usr/bin/env node
2
- import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, REF_BACKFILL_JOB, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildAuthHeaders, buildBaseUrl, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, compareMemoryRoleReports, connect, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getExtractionBenchmarkProfile, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, initDatabase, isEmbeddingDisabled, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, planReplicationOpsAgePrune, projectMatchesFilter, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, setPeerProjectFilter, stripJsonComments, stripPrivateObj, stripTrailingCommas, syncPassPreflight, updatePeerAddresses, vacuumDatabase, writeCodememConfigFile } from "@codemem/core";
2
+ import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, REF_BACKFILL_JOB, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildAuthHeaders, buildBaseUrl, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, compareMemoryRoleReports, connect, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, estimateExtractionModelCost, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getExtractionBenchmarkProfile, getExtractionModelPricing, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, initDatabase, isEmbeddingDisabled, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, planReplicationOpsAgePrune, projectMatchesFilter, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, scoreExtractionBenchmarkOutput, setPeerProjectFilter, stripJsonComments, stripPrivateObj, stripTrailingCommas, syncPassPreflight, updatePeerAddresses, vacuumDatabase, writeCodememConfigFile } from "@codemem/core";
3
3
  import { Command, Option } from "commander";
4
4
  import omelette from "omelette";
5
5
  import { appendFileSync, copyFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, rmdirSync, statSync, unlinkSync, writeFileSync } from "node:fs";
@@ -2666,12 +2666,13 @@ function buildCoordinatorCommand() {
2666
2666
  }
2667
2667
  });
2668
2668
  cmd.addCommand(listScopeMembersCmd);
2669
- const grantScopeMemberCmd = new Command("grant-scope-member").configureHelp(helpStyle).description("Grant a device explicit access to a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").option("--role <role>", "membership role").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2669
+ const grantScopeMemberCmd = new Command("grant-scope-member").configureHelp(helpStyle).description("Grant a device explicit access to a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").requiredOption("--effect-id <id>", "deterministic mutation effect id").option("--role <role>", "membership role").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2670
2670
  addDbOption(grantScopeMemberCmd);
2671
2671
  addJsonOption(grantScopeMemberCmd);
2672
2672
  grantScopeMemberCmd.action(async (groupId, scopeId, deviceId, opts) => {
2673
2673
  try {
2674
2674
  const membership = await coordinatorGrantScopeMembershipAction({
2675
+ effectId: opts.effectId,
2675
2676
  groupId,
2676
2677
  scopeId,
2677
2678
  deviceId,
@@ -2699,12 +2700,13 @@ function buildCoordinatorCommand() {
2699
2700
  }
2700
2701
  });
2701
2702
  cmd.addCommand(grantScopeMemberCmd);
2702
- const revokeScopeMemberCmd = new Command("revoke-scope-member").configureHelp(helpStyle).description("Revoke a device from a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2703
+ const revokeScopeMemberCmd = new Command("revoke-scope-member").configureHelp(helpStyle).description("Revoke a device from a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").requiredOption("--effect-id <id>", "deterministic mutation effect id").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
2703
2704
  addDbOption(revokeScopeMemberCmd);
2704
2705
  addJsonOption(revokeScopeMemberCmd);
2705
2706
  revokeScopeMemberCmd.action(async (groupId, scopeId, deviceId, opts) => {
2706
2707
  try {
2707
2708
  if (!await coordinatorRevokeScopeMembershipAction({
2709
+ effectId: opts.effectId,
2708
2710
  groupId,
2709
2711
  scopeId,
2710
2712
  deviceId,
@@ -5117,8 +5119,28 @@ function createMemoryExtractionReplayCommand() {
5117
5119
  });
5118
5120
  return cmd;
5119
5121
  }
5122
+ function reconcileExtractionBenchmarkStatus(input) {
5123
+ const quality = input.finalQuality;
5124
+ let status = input.classification.status;
5125
+ let reason = input.classification.reason;
5126
+ if (input.purpose === "shape_quality" && quality && status !== "observer_no_output") {
5127
+ if (quality.summaryDisposition.score === 0) {
5128
+ status = "shape_fail";
5129
+ reason = `summary disposition ${quality.summaryDisposition.actual} does not satisfy expected ${quality.summaryDisposition.expected}`;
5130
+ } else if (status === "shape_fail" && quality.summaryDisposition.actual === "skip" && input.finalFailureReasons.length > 0 && input.finalFailureReasons.every((failure) => failure.startsWith("summary count "))) {
5131
+ status = "pass";
5132
+ reason = "valid low-signal skip satisfies benchmark disposition";
5133
+ }
5134
+ }
5135
+ return {
5136
+ status,
5137
+ reason,
5138
+ quality,
5139
+ initialQuality: input.initialQuality
5140
+ };
5141
+ }
5120
5142
  function createMemoryExtractionBenchmarkCommand() {
5121
- const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run");
5143
+ const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run").option("--repetitions <n>", "run every benchmark batch 1-10 times to measure model stability", "1");
5122
5144
  addDbOption(cmd);
5123
5145
  addJsonOption(cmd);
5124
5146
  cmd.action(async (opts) => {
@@ -5139,6 +5161,9 @@ function createMemoryExtractionBenchmarkCommand() {
5139
5161
  const maxOutputTokensInput = opts.maxOutputTokens?.trim() ?? "";
5140
5162
  const maxOutputTokens = maxOutputTokensInput.length > 0 ? parseStrictPositiveId(maxOutputTokensInput) : null;
5141
5163
  if (maxOutputTokensInput.length > 0 && maxOutputTokens === null) throw new Error(`Invalid max output tokens: ${maxOutputTokensInput || opts.maxOutputTokens}`);
5164
+ const repetitionsInput = opts.repetitions?.trim() ?? "1";
5165
+ const repetitions = parseStrictPositiveId(repetitionsInput);
5166
+ if (repetitions === null || repetitions > 10) throw new Error(`Invalid repetitions: ${repetitionsInput || opts.repetitions}`);
5142
5167
  const observerConfig = loadObserverConfig();
5143
5168
  const observerConfigWithOverrides = {
5144
5169
  ...observerConfig,
@@ -5152,7 +5177,7 @@ function createMemoryExtractionBenchmarkCommand() {
5152
5177
  };
5153
5178
  const observer = new ObserverClient(observerConfigWithOverrides);
5154
5179
  const runs = [];
5155
- for (const batch of benchmark.batches) {
5180
+ for (let iteration = 1; iteration <= repetitions; iteration += 1) for (const batch of benchmark.batches) {
5156
5181
  const scenarioId = batch.scenarioId ?? benchmark.scenarioId;
5157
5182
  const result = opts.observerTierRouting === true ? await replayBatchExtractionWithTierRouting(resolveDbOpt(opts), observerConfigWithOverrides, {
5158
5183
  batchId: batch.batchId,
@@ -5163,7 +5188,51 @@ function createMemoryExtractionBenchmarkCommand() {
5163
5188
  scenarioId,
5164
5189
  transcriptBudget: transcriptBudget ?? void 0
5165
5190
  });
5191
+ const costModel = result.observer.modelFallbackApplied ? result.observer.resolvedModel : result.observer.resolvedModel ?? result.observer.model;
5192
+ const initialCost = costModel ? estimateExtractionModelCost(costModel, result.observer.initialUsage) : null;
5193
+ const repairCost = costModel ? estimateExtractionModelCost(costModel, result.observer.repairedUsage) : null;
5194
+ const totalCost = costModel ? estimateExtractionModelCost(costModel, result.observer.totalUsage) : null;
5195
+ const pricing = costModel ? getExtractionModelPricing(costModel) : null;
5196
+ const costUnavailableReason = totalCost ? null : result.observer.modelFallbackApplied && !result.observer.resolvedModel ? "model_fallback_unresolved" : result.observer.totalUsage == null ? "missing_usage" : "unknown_model_pricing";
5197
+ const initialQuality = result.observer.initialDiagnostics ? scoreExtractionBenchmarkOutput({
5198
+ parsed: result.observer.initialParsed,
5199
+ diagnostics: result.observer.initialDiagnostics,
5200
+ review: batch.review ?? {
5201
+ status: "unreviewed",
5202
+ reviewerNotes: "No durable-fact review has been recorded for this batch."
5203
+ },
5204
+ estimatedCostUsd: initialCost?.totalCostUsd ?? null,
5205
+ expectedSummaryDisposition: batch.expectedSummaryDisposition
5206
+ }) : null;
5207
+ const repairQuality = result.observer.repairedParsed && result.observer.repairedDiagnostics ? scoreExtractionBenchmarkOutput({
5208
+ parsed: result.observer.repairedParsed,
5209
+ diagnostics: result.observer.repairedDiagnostics,
5210
+ review: batch.review ?? {
5211
+ status: "unreviewed",
5212
+ reviewerNotes: "No durable-fact review has been recorded for this batch."
5213
+ },
5214
+ estimatedCostUsd: repairCost?.totalCostUsd ?? null,
5215
+ expectedSummaryDisposition: batch.expectedSummaryDisposition
5216
+ }) : null;
5217
+ const finalQuality = result.observer.diagnostics ? scoreExtractionBenchmarkOutput({
5218
+ parsed: result.observer.parsed,
5219
+ diagnostics: result.observer.diagnostics,
5220
+ review: batch.review ?? {
5221
+ status: "unreviewed",
5222
+ reviewerNotes: "No durable-fact review has been recorded for this batch."
5223
+ },
5224
+ estimatedCostUsd: totalCost?.totalCostUsd ?? null,
5225
+ expectedSummaryDisposition: batch.expectedSummaryDisposition
5226
+ }) : null;
5227
+ const reconciled = reconcileExtractionBenchmarkStatus({
5228
+ purpose: batch.purpose,
5229
+ classification: result.classification,
5230
+ finalFailureReasons: result.evaluation.failureReasons,
5231
+ initialQuality,
5232
+ finalQuality
5233
+ });
5166
5234
  runs.push({
5235
+ iteration,
5167
5236
  batchId: batch.batchId,
5168
5237
  sessionId: batch.sessionId,
5169
5238
  label: batch.label,
@@ -5171,17 +5240,23 @@ function createMemoryExtractionBenchmarkCommand() {
5171
5240
  complexity: batch.complexity,
5172
5241
  scenarioId,
5173
5242
  expectedTier: batch.expectedTier ?? null,
5243
+ expectedSummaryDisposition: batch.expectedSummaryDisposition,
5174
5244
  analysis: {
5175
5245
  eventSpan: result.analysis.eventSpan,
5176
5246
  promptCount: result.analysis.promptCount,
5177
5247
  toolCount: result.analysis.toolCount,
5178
5248
  transcriptLength: result.analysis.transcriptLength
5179
5249
  },
5180
- status: result.classification.status,
5181
- reason: result.classification.reason,
5250
+ status: reconciled.status,
5251
+ reason: reconciled.reason,
5182
5252
  tier: result.observer.tier ?? "manual",
5183
5253
  provider: result.observer.provider,
5184
5254
  model: result.observer.model,
5255
+ transport: result.observer.transport,
5256
+ requestedModel: result.observer.requestedModel,
5257
+ resolvedModel: result.observer.resolvedModel,
5258
+ modelFallbackApplied: result.observer.modelFallbackApplied,
5259
+ modelFallbackReason: result.observer.modelFallbackReason,
5185
5260
  openaiUseResponses: result.observer.openaiUseResponses,
5186
5261
  reasoningEffort: result.observer.reasoningEffort,
5187
5262
  reasoningSummary: result.observer.reasoningSummary,
@@ -5189,22 +5264,89 @@ function createMemoryExtractionBenchmarkCommand() {
5189
5264
  temperature: result.observer.temperature,
5190
5265
  summaries: result.evaluation.counts.summaries,
5191
5266
  observations: result.evaluation.counts.observations,
5192
- repairApplied: result.observer.repairApplied
5267
+ repairApplied: result.observer.repairApplied,
5268
+ initial: {
5269
+ raw: result.observer.initialRaw,
5270
+ status: result.initialClassification.status,
5271
+ reason: result.initialClassification.reason,
5272
+ pass: result.initialEvaluation.pass,
5273
+ failureReasons: result.initialEvaluation.failureReasons,
5274
+ summaries: result.initialEvaluation.counts.summaries,
5275
+ observations: result.initialEvaluation.counts.observations,
5276
+ diagnostics: result.observer.initialDiagnostics,
5277
+ elapsedMs: result.observer.initialElapsedMs,
5278
+ usage: result.observer.initialUsage,
5279
+ quality: reconciled.initialQuality
5280
+ },
5281
+ repair: {
5282
+ applied: result.observer.repairApplied,
5283
+ raw: result.observer.repairedRaw,
5284
+ status: result.repairedClassification?.status ?? null,
5285
+ reason: result.repairedClassification?.reason ?? null,
5286
+ pass: result.repairedEvaluation?.pass ?? null,
5287
+ failureReasons: result.repairedEvaluation?.failureReasons ?? [],
5288
+ summaries: result.repairedEvaluation?.counts.summaries ?? null,
5289
+ observations: result.repairedEvaluation?.counts.observations ?? null,
5290
+ diagnostics: result.observer.repairedDiagnostics,
5291
+ elapsedMs: result.observer.repairedElapsedMs,
5292
+ usage: result.observer.repairedUsage,
5293
+ quality: repairQuality
5294
+ },
5295
+ telemetry: {
5296
+ totalElapsedMs: result.observer.totalElapsedMs,
5297
+ totalUsage: result.observer.totalUsage
5298
+ },
5299
+ pricing,
5300
+ cost: {
5301
+ initial: initialCost,
5302
+ repair: repairCost,
5303
+ total: totalCost,
5304
+ unavailableReason: costUnavailableReason
5305
+ },
5306
+ quality: reconciled.quality
5193
5307
  });
5194
5308
  }
5309
+ const reviewedQualityRuns = runs.filter((run) => run.quality?.weightedQualityScore != null);
5310
+ const knownCostRuns = runs.filter((run) => run.cost.total != null);
5311
+ const knownElapsedRuns = runs.filter((run) => run.telemetry.totalElapsedMs != null);
5195
5312
  const summary = {
5313
+ repetitions,
5196
5314
  total: runs.length,
5197
5315
  shapeQualityTotal: runs.filter((run) => run.purpose === "shape_quality").length,
5198
5316
  shapeQualityPasses: runs.filter((run) => run.purpose === "shape_quality" && run.status === "pass").length,
5199
5317
  shapeQualityFails: runs.filter((run) => run.purpose === "shape_quality" && run.status === "shape_fail").length,
5200
5318
  expectedTierTotal: runs.filter((run) => run.expectedTier != null).length,
5201
5319
  expectedTierMatches: runs.filter((run) => run.expectedTier != null && run.expectedTier === run.tier).length,
5202
- robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length
5320
+ robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length,
5321
+ summaryDispositionTotal: runs.filter((run) => run.quality != null).length,
5322
+ summaryDispositionMatches: runs.filter((run) => run.quality?.summaryDisposition.score === 1).length,
5323
+ reviewedQualityRuns: reviewedQualityRuns.length,
5324
+ knownCostRuns: knownCostRuns.length,
5325
+ unknownCostRuns: runs.length - knownCostRuns.length,
5326
+ missingUsageRuns: runs.filter((run) => run.cost.unavailableReason === "missing_usage").length,
5327
+ unknownPricingRuns: runs.filter((run) => run.cost.unavailableReason === "unknown_model_pricing").length,
5328
+ fallbackUnresolvedRuns: runs.filter((run) => run.cost.unavailableReason === "model_fallback_unresolved").length,
5329
+ totalKnownCostUsd: knownCostRuns.reduce((sum, run) => sum + (run.cost.total?.totalCostUsd ?? 0), 0),
5330
+ knownElapsedRuns: knownElapsedRuns.length,
5331
+ totalKnownElapsedMs: knownElapsedRuns.reduce((sum, run) => sum + (run.telemetry.totalElapsedMs ?? 0), 0),
5332
+ perBatchStability: benchmark.batches.map((batch) => {
5333
+ const batchRuns = runs.filter((run) => run.batchId === batch.batchId);
5334
+ const passes = batchRuns.filter((run) => run.status === "pass").length;
5335
+ return {
5336
+ batchId: batch.batchId,
5337
+ purpose: batch.purpose,
5338
+ passes,
5339
+ total: batchRuns.length,
5340
+ passRate: batchRuns.length > 0 ? passes / batchRuns.length : null,
5341
+ statuses: batchRuns.map((run) => run.status)
5342
+ };
5343
+ })
5203
5344
  };
5204
- const uniqueObserverKeys = Array.from(new Set(runs.map((run) => `${run.provider}::${run.model}::${run.openaiUseResponses ? "responses" : "chat"}`)));
5345
+ const uniqueObserverKeys = Array.from(new Set(runs.map((run) => `${run.provider}::${run.model}::${run.transport}`)));
5205
5346
  const observerSummary = opts.observerTierRouting === true ? {
5206
5347
  provider: uniqueObserverKeys.length === 1 ? runs[0]?.provider ?? observer.provider : "mixed",
5207
5348
  model: uniqueObserverKeys.length === 1 ? runs[0]?.model ?? observer.model : "mixed",
5349
+ transport: uniqueObserverKeys.length === 1 ? runs[0]?.transport ?? "unknown" : "mixed",
5208
5350
  tierRouting: true,
5209
5351
  openaiUseResponses: uniqueObserverKeys.length === 1 ? runs[0]?.openaiUseResponses ?? observer.openaiUseResponses : null,
5210
5352
  reasoningEffort: uniqueObserverKeys.length === 1 ? runs[0]?.reasoningEffort ?? observer.reasoningEffort : "mixed",
@@ -5216,6 +5358,7 @@ function createMemoryExtractionBenchmarkCommand() {
5216
5358
  } : {
5217
5359
  provider: observer.provider,
5218
5360
  model: observer.model,
5361
+ transport: runs[0]?.transport ?? observer.getStatus().runtime,
5219
5362
  tierRouting: false,
5220
5363
  openaiUseResponses: observer.openaiUseResponses,
5221
5364
  reasoningEffort: observer.reasoningEffort,
@@ -5229,7 +5372,8 @@ function createMemoryExtractionBenchmarkCommand() {
5229
5372
  benchmark: {
5230
5373
  id: benchmark.id,
5231
5374
  title: benchmark.title,
5232
- scenarioId: benchmark.scenarioId
5375
+ scenarioId: benchmark.scenarioId,
5376
+ modelCandidates: benchmark.modelCandidates
5233
5377
  },
5234
5378
  observer: observerSummary,
5235
5379
  summary,
@@ -5243,6 +5387,7 @@ function createMemoryExtractionBenchmarkCommand() {
5243
5387
  p.log.info([
5244
5388
  `Benchmark: ${benchmark.id} — ${benchmark.title}`,
5245
5389
  `Observer: ${observerSummary.provider}/${observerSummary.model}`,
5390
+ `Transport: ${observerSummary.transport}`,
5246
5391
  `Tier routing: ${opts.observerTierRouting === true ? "yes" : "no"}`,
5247
5392
  `OpenAI Responses: ${observerSummary.openaiUseResponses === null ? "mixed" : observerSummary.openaiUseResponses ? "yes" : "no"}`,
5248
5393
  `Reasoning effort: ${observerSummary.reasoningEffort ?? "none"}`,
@@ -5250,12 +5395,23 @@ function createMemoryExtractionBenchmarkCommand() {
5250
5395
  `Max output tokens: ${observerSummary.maxOutputTokens ?? "mixed"}`,
5251
5396
  `Temperature: ${observerSummary.temperature ?? "mixed"}`,
5252
5397
  `Transcript budget override: ${transcriptBudget ?? "default"}`,
5398
+ `Repetitions: ${summary.repetitions}`,
5253
5399
  `Shape-quality passes: ${summary.shapeQualityPasses}/${summary.shapeQualityTotal}`,
5254
5400
  `Shape-quality fails: ${summary.shapeQualityFails}`,
5255
5401
  `Expected-tier matches: ${summary.expectedTierMatches}/${summary.expectedTierTotal}`,
5256
- `Observer no-output cases: ${summary.robustnessNoOutput}`
5402
+ `Observer no-output cases: ${summary.robustnessNoOutput}`,
5403
+ `Summary disposition matches: ${summary.summaryDispositionMatches}/${summary.summaryDispositionTotal}`,
5404
+ `Reviewed quality runs: ${summary.reviewedQualityRuns} (compare per-run dimensions; scores are fixture-specific)`,
5405
+ `Known estimated cost: $${summary.totalKnownCostUsd.toFixed(6)} (${summary.knownCostRuns}/${summary.total}; missing usage=${summary.missingUsageRuns}, unknown pricing=${summary.unknownPricingRuns}, unresolved fallback=${summary.fallbackUnresolvedRuns})`,
5406
+ `Known elapsed time: ${summary.totalKnownElapsedMs}ms (${summary.knownElapsedRuns}/${summary.total} run(s))`
5257
5407
  ].join("\n"));
5258
- for (const run of runs) p.log.message(` [${run.batchId}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model}${run.openaiUseResponses ? " [responses]" : ""} summaries=${run.summaries} observations=${run.observations} repair=${run.repairApplied ? "yes" : "no"} — ${run.label}`);
5408
+ for (const run of runs) {
5409
+ const qualityLabel = run.quality?.weightedQualityScore == null ? "n/a" : run.quality.weightedQualityScore.toFixed(3);
5410
+ const costLabel = run.cost.total == null ? "n/a" : `$${run.cost.total.totalCostUsd.toFixed(6)}`;
5411
+ const latencyLabel = run.telemetry.totalElapsedMs == null ? "n/a" : `${run.telemetry.totalElapsedMs}ms`;
5412
+ const missingRequired = run.quality?.requiredRecall.missingLabelIds.join(",") || "none";
5413
+ p.log.message(` [${run.batchId}#${run.iteration}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} disposition=${run.quality?.summaryDisposition.actual ?? "n/a"}/${run.expectedSummaryDisposition} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model} [${run.transport}] initial=${run.initial.summaries}s/${run.initial.observations}o final=${run.summaries}s/${run.observations}o quality=${qualityLabel} coverage=${run.quality?.weightedQualityCoverage?.toFixed(3) ?? "n/a"} required_missing=${missingRequired} cost=${costLabel} latency=${latencyLabel} schema_loss=${run.initial.diagnostics?.dataLoss === true ? "yes" : "no"} fallback=${run.modelFallbackApplied ? "yes" : "no"} repair=${run.repairApplied ? "yes" : "no"} — ${run.label}`);
5414
+ }
5259
5415
  p.outro("done");
5260
5416
  } catch (error) {
5261
5417
  const message = error instanceof Error ? error.message : "Extraction benchmark failed";
@@ -5976,6 +6132,14 @@ function sqliteVecFailureDiagnostics(error, dbPath) {
5976
6132
  `error=${message}`
5977
6133
  ];
5978
6134
  }
6135
+ async function runServeCoordinatorMaintenance(store, dependencies) {
6136
+ const projectShares = await dependencies.advancePendingProjectShares(store, { limit: 3 });
6137
+ if (projectShares.failed > 0) throw new Error(`share operation maintenance failed for ${projectShares.failed} of ${projectShares.processed} operations`);
6138
+ return {
6139
+ projectShares,
6140
+ recipientPolicies: await dependencies.reconcileRecipientPolicyProjects(store, { limit: 3 })
6141
+ };
6142
+ }
5979
6143
  async function startBackgroundViewer(invocation) {
5980
6144
  warnIfViewerExposed(invocation.host, invocation.port);
5981
6145
  if (await isPortOpen(invocation.host, invocation.port)) {
@@ -6008,7 +6172,7 @@ async function startBackgroundViewer(invocation) {
6008
6172
  p.outro(`Viewer started in background (pid ${child.pid}) at http://${invocation.host}:${invocation.port}`);
6009
6173
  }
6010
6174
  async function startForegroundViewer(invocation) {
6011
- const { createApp, createSyncApp, closeStore, getStore } = await import("@codemem/server");
6175
+ const { advancePendingProjectShares, createApp, createSyncApp, closeStore, getStore, reconcileRecipientPolicyProjects } = await import("@codemem/server");
6012
6176
  const { serve } = await import("@hono/node-server");
6013
6177
  if (invocation.dbPath) process.env.CODEMEM_DB = invocation.dbPath;
6014
6178
  if (invocation.configPath) process.env.CODEMEM_CONFIG = invocation.configPath;
@@ -6099,6 +6263,12 @@ async function startForegroundViewer(invocation) {
6099
6263
  port: syncConfig.syncPort,
6100
6264
  signal: syncAbort.signal,
6101
6265
  scanner: store.scanner,
6266
+ onAfterCoordinatorRefresh: async () => {
6267
+ await runServeCoordinatorMaintenance(store, {
6268
+ advancePendingProjectShares,
6269
+ reconcileRecipientPolicyProjects
6270
+ });
6271
+ },
6102
6272
  onPhaseChange: (phase) => {
6103
6273
  if (phase === "running") {
6104
6274
  syncRuntimeStatus.phase = null;