codemem 0.44.0-alpha.1 → 0.44.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -5
- package/dist/index.js +376 -85
- package/dist/index.js.map +1 -1
- package/package.json +7 -4
package/dist/index.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, DeviceIdentityError, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, PROMPT_TRANSPORT_PROTOCOL_RANGE, REF_BACKFILL_JOB, RawEventIngestValidationError, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, TRUSTED_HOOK_MAPPER_OPTIONS, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, arePromptTransportProtocolRangesCompatible, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildBaseUrl, buildDirectPeerAuthHeaders, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, buildViewerIdentityTarget, classifyPromptTransportFailure, clonePromptPackAttempt, collectOperationalStatus, compareMemoryRoleReports, connect, connectReadOnly, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, detectInstallKind, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, estimateExtractionModelCost, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getAttributionDiagnostics, getExtractionBenchmarkProfile, getExtractionModelPricing, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getUpdateStatus, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, ingestRawEvents, initDatabase, isEmbeddingDisabled,
|
|
2
|
+
import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, DeviceIdentityError, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, ObserverOutputError, ObserverOutputTransportError, PROMPT_TRANSPORT_PROTOCOL_RANGE, REF_BACKFILL_JOB, RawEventIngestValidationError, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, TRUSTED_HOOK_MAPPER_OPTIONS, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, arePromptTransportProtocolRangesCompatible, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildBaseUrl, buildDirectPeerAuthHeaders, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, buildViewerIdentityTarget, classifyPromptTransportFailure, clonePromptPackAttempt, collectOperationalStatus, compareMemoryRoleReports, connect, connectReadOnly, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, detectInstallKind, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, estimateExtractionModelCost, exportMemories, extractApplyPatchPaths, extractionReplayObserverIdentity, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getAttributionDiagnostics, getExtractionBenchmarkProfile, getExtractionModelPricing, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getUpdateStatus, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, ingestRawEvents, initDatabase, isAutomaticRecallMeasurement, isEmbeddingDisabled, isReleaseVersionForChannel, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, normalizePromptTransportProtocolRange, planReplicationOpsAgePrune, probeCodememViewerLiveness, probeCodememViewerLiveness as probeCodememViewerLiveness$1, probeRequiredNativeRuntime, projectMatchesFilter, promptPackArtifactFingerprint, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, recordAutomaticRecall, recordPromptPackArtifacts, recordPromptPackTerminal, recordRetrievalSurface, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, resolveRetrievalSession, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, scoreExtractionBenchmarkOutput, setPeerProjectFilter, stripJsonComments, stripTrailingCommas, syncPassPreflight, tryUpdateRetrievalDelivery, updatePeerAddresses, vacuumDatabase, viewerUrl, writeCodememConfigFile } from "@codemem/core";
|
|
3
3
|
import { Command, Option } from "commander";
|
|
4
4
|
import omelette from "omelette";
|
|
5
5
|
import { randomInt, randomUUID } from "node:crypto";
|
|
@@ -4397,7 +4397,7 @@ function startMaintenanceWorkerRuntime(options = {}) {
|
|
|
4397
4397
|
dbPath,
|
|
4398
4398
|
signal: options.signal,
|
|
4399
4399
|
batchSize: 10,
|
|
4400
|
-
idleIntervalMs:
|
|
4400
|
+
idleIntervalMs: 6e4
|
|
4401
4401
|
}));
|
|
4402
4402
|
runners.push(createSequentialBackfillCoordinator(store, [
|
|
4403
4403
|
{
|
|
@@ -5217,6 +5217,163 @@ function summarizeBenchmarkReasoning(runs, fallback) {
|
|
|
5217
5217
|
reasoningSummary: reasoningSummaries.size > 1 ? "mixed" : source.reasoningSummary
|
|
5218
5218
|
};
|
|
5219
5219
|
}
|
|
5220
|
+
function benchmarkSecondAttemptValue(attempted, value) {
|
|
5221
|
+
return attempted ? value : null;
|
|
5222
|
+
}
|
|
5223
|
+
function benchmarkSecondAttemptList(attempted, value) {
|
|
5224
|
+
return attempted ? value : [];
|
|
5225
|
+
}
|
|
5226
|
+
function buildBenchmarkSecondAttempt(result, input) {
|
|
5227
|
+
return {
|
|
5228
|
+
attempted: input.attempted,
|
|
5229
|
+
reason: input.reason,
|
|
5230
|
+
raw: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedRaw),
|
|
5231
|
+
status: benchmarkSecondAttemptValue(input.attempted, result.repairedClassification?.status ?? null),
|
|
5232
|
+
resultReason: benchmarkSecondAttemptValue(input.attempted, result.repairedClassification?.reason ?? null),
|
|
5233
|
+
pass: benchmarkSecondAttemptValue(input.attempted, result.repairedEvaluation?.pass ?? null),
|
|
5234
|
+
failureReasons: benchmarkSecondAttemptList(input.attempted, result.repairedEvaluation?.failureReasons ?? []),
|
|
5235
|
+
summaries: benchmarkSecondAttemptValue(input.attempted, result.repairedEvaluation?.counts.summaries ?? null),
|
|
5236
|
+
observations: benchmarkSecondAttemptValue(input.attempted, result.repairedEvaluation?.counts.observations ?? null),
|
|
5237
|
+
diagnostics: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedDiagnostics),
|
|
5238
|
+
elapsedMs: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedElapsedMs),
|
|
5239
|
+
usage: benchmarkSecondAttemptValue(input.attempted, result.observer.repairedUsage),
|
|
5240
|
+
quality: benchmarkSecondAttemptValue(input.attempted, input.quality)
|
|
5241
|
+
};
|
|
5242
|
+
}
|
|
5243
|
+
function estimateBenchmarkAttemptCost(model, usage) {
|
|
5244
|
+
if (model === null) return {
|
|
5245
|
+
total: null,
|
|
5246
|
+
unavailableReason: "model_fallback_unresolved"
|
|
5247
|
+
};
|
|
5248
|
+
const total = estimateExtractionModelCost(model, usage);
|
|
5249
|
+
if (total) return {
|
|
5250
|
+
total,
|
|
5251
|
+
unavailableReason: null
|
|
5252
|
+
};
|
|
5253
|
+
if (usage === null) return {
|
|
5254
|
+
total: null,
|
|
5255
|
+
unavailableReason: "missing_usage"
|
|
5256
|
+
};
|
|
5257
|
+
return {
|
|
5258
|
+
total: null,
|
|
5259
|
+
unavailableReason: "unknown_model_pricing"
|
|
5260
|
+
};
|
|
5261
|
+
}
|
|
5262
|
+
function rate(numerator, denominator) {
|
|
5263
|
+
return denominator > 0 ? numerator / denominator : null;
|
|
5264
|
+
}
|
|
5265
|
+
function mean(values) {
|
|
5266
|
+
return values.length > 0 ? values.reduce((sum, value) => sum + value, 0) / values.length : null;
|
|
5267
|
+
}
|
|
5268
|
+
function countOutputModes(runs, key) {
|
|
5269
|
+
const counts = {};
|
|
5270
|
+
for (const run of runs) counts[run[key]] = (counts[run[key]] ?? 0) + 1;
|
|
5271
|
+
return counts;
|
|
5272
|
+
}
|
|
5273
|
+
function collectBenchmarkMetricSamples(runs) {
|
|
5274
|
+
return {
|
|
5275
|
+
latencies: runs.flatMap((run) => run.telemetry.totalElapsedMs == null ? [] : [run.telemetry.totalElapsedMs]),
|
|
5276
|
+
usages: runs.flatMap((run) => run.telemetry.totalUsage == null ? [] : [run.telemetry.totalUsage]),
|
|
5277
|
+
summaryScores: runs.flatMap((run) => run.quality == null ? [] : [run.quality.summaryDisposition.score]),
|
|
5278
|
+
structuralScores: runs.flatMap((run) => {
|
|
5279
|
+
if (run.outputValidation === "invalid") return [0];
|
|
5280
|
+
if (run.outputValidation === "valid") return [1];
|
|
5281
|
+
return run.quality == null ? [] : [run.quality.schemaCompliance.score];
|
|
5282
|
+
}),
|
|
5283
|
+
semanticScores: runs.flatMap((run) => run.quality == null ? [] : semanticQualityScore(run.quality))
|
|
5284
|
+
};
|
|
5285
|
+
}
|
|
5286
|
+
function semanticQualityScore(quality) {
|
|
5287
|
+
if (quality.reviewStatus !== "reviewed") return [];
|
|
5288
|
+
const available = [
|
|
5289
|
+
[quality.summaryDisposition.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.summaryDisposition],
|
|
5290
|
+
[quality.requiredRecall?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.requiredRecall],
|
|
5291
|
+
[quality.optionalRecall?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.optionalRecall],
|
|
5292
|
+
[quality.worthinessPrecision?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.worthinessPrecision],
|
|
5293
|
+
[quality.summaryBreadth?.score, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.summaryBreadth],
|
|
5294
|
+
[quality.observationSignals?.redundancyAvoidance, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.redundancyAvoidance],
|
|
5295
|
+
[quality.observationSignals?.segmentation, EXTRACTION_BENCHMARK_QUALITY_WEIGHTS.segmentation]
|
|
5296
|
+
].filter((entry) => entry[0] != null);
|
|
5297
|
+
const weight = available.reduce((sum, [, dimensionWeight]) => sum + dimensionWeight, 0);
|
|
5298
|
+
if (weight === 0) return [];
|
|
5299
|
+
return [available.reduce((sum, [score, dimensionWeight]) => sum + score * dimensionWeight, 0) / weight];
|
|
5300
|
+
}
|
|
5301
|
+
function benchmarkModeMetrics(runs) {
|
|
5302
|
+
const validOutputs = runs.filter((run) => run.outputValidation === "valid" || run.quality?.schemaCompliance.compliant === true).length;
|
|
5303
|
+
const repairAttempts = runs.filter((run) => run.repairAttempted).length;
|
|
5304
|
+
const retryAttempts = runs.filter((run) => run.retryAttempted).length;
|
|
5305
|
+
const repairOrRetryAttempts = runs.filter((run) => run.repairAttempted || run.retryAttempted).length;
|
|
5306
|
+
const { latencies, usages, summaryScores, structuralScores, semanticScores } = collectBenchmarkMetricSamples(runs);
|
|
5307
|
+
const retainedObservations = runs.reduce((sum, run) => sum + run.observations, 0);
|
|
5308
|
+
const matchingSummaryDispositions = summaryScores.filter((score) => score === 1).length;
|
|
5309
|
+
return {
|
|
5310
|
+
runs: runs.length,
|
|
5311
|
+
validOutputs,
|
|
5312
|
+
validOutputRate: rate(validOutputs, runs.length),
|
|
5313
|
+
repairAttempts,
|
|
5314
|
+
repairRate: rate(repairAttempts, runs.length),
|
|
5315
|
+
retryAttempts,
|
|
5316
|
+
retryRate: rate(retryAttempts, runs.length),
|
|
5317
|
+
repairOrRetryAttempts,
|
|
5318
|
+
repairOrRetryRate: rate(repairOrRetryAttempts, runs.length),
|
|
5319
|
+
knownLatencyRuns: latencies.length,
|
|
5320
|
+
totalLatencyMs: latencies.reduce((sum, value) => sum + value, 0),
|
|
5321
|
+
meanLatencyMs: mean(latencies),
|
|
5322
|
+
knownTokenRuns: usages.length,
|
|
5323
|
+
inputTokens: usages.reduce((sum, usage) => sum + usage.inputTokens, 0),
|
|
5324
|
+
outputTokens: usages.reduce((sum, usage) => sum + usage.outputTokens, 0),
|
|
5325
|
+
totalTokens: usages.reduce((sum, usage) => sum + (usage.totalTokens ?? usage.inputTokens + usage.outputTokens), 0),
|
|
5326
|
+
retainedObservations,
|
|
5327
|
+
meanRetainedObservations: rate(retainedObservations, runs.length),
|
|
5328
|
+
summaryDispositionEvaluated: summaryScores.length,
|
|
5329
|
+
summaryDispositionMatches: matchingSummaryDispositions,
|
|
5330
|
+
summaryDispositionMatchRate: rate(matchingSummaryDispositions, summaryScores.length),
|
|
5331
|
+
structuralQualityEvaluated: structuralScores.length,
|
|
5332
|
+
contractIntegrityRate: mean(structuralScores),
|
|
5333
|
+
semanticQualityEvaluated: semanticScores.length,
|
|
5334
|
+
meanSemanticQuality: mean(semanticScores)
|
|
5335
|
+
};
|
|
5336
|
+
}
|
|
5337
|
+
function summarizeExtractionBenchmarkReporting(runs) {
|
|
5338
|
+
const byActualOutputMode = {};
|
|
5339
|
+
for (const mode of [
|
|
5340
|
+
"json_schema",
|
|
5341
|
+
"forced_tool",
|
|
5342
|
+
"legacy_xml"
|
|
5343
|
+
]) {
|
|
5344
|
+
const modeRuns = runs.filter((run) => run.actualOutputMode === mode);
|
|
5345
|
+
if (modeRuns.length > 0) byActualOutputMode[mode] = benchmarkModeMetrics(modeRuns);
|
|
5346
|
+
}
|
|
5347
|
+
return {
|
|
5348
|
+
requestedOutputModes: countOutputModes(runs, "requestedOutputMode"),
|
|
5349
|
+
actualOutputModes: countOutputModes(runs, "actualOutputMode"),
|
|
5350
|
+
overall: benchmarkModeMetrics(runs),
|
|
5351
|
+
byActualOutputMode
|
|
5352
|
+
};
|
|
5353
|
+
}
|
|
5354
|
+
function summarizeExtractionBenchmarkAttempts(attempts, batches) {
|
|
5355
|
+
return {
|
|
5356
|
+
total: attempts.length,
|
|
5357
|
+
shapeQualityTotal: attempts.filter((attempt) => attempt.purpose === "shape_quality").length,
|
|
5358
|
+
shapeQualityPasses: attempts.filter((attempt) => attempt.purpose === "shape_quality" && attempt.status === "pass").length,
|
|
5359
|
+
shapeQualityFails: attempts.filter((attempt) => attempt.purpose === "shape_quality" && attempt.status !== "pass").length,
|
|
5360
|
+
expectedTierTotal: attempts.filter((attempt) => attempt.expectedTier != null).length,
|
|
5361
|
+
expectedTierMatches: attempts.filter((attempt) => attempt.expectedTier != null && attempt.expectedTier === attempt.tier).length,
|
|
5362
|
+
robustnessNoOutput: attempts.filter((attempt) => attempt.purpose === "replay_robustness" && (attempt.status === "observer_no_output" || attempt.status === "output_failure")).length,
|
|
5363
|
+
perBatchStability: batches.map((batch) => {
|
|
5364
|
+
const batchAttempts = attempts.filter((attempt) => attempt.batchId === batch.batchId).toSorted((left, right) => left.iteration - right.iteration);
|
|
5365
|
+
const passes = batchAttempts.filter((attempt) => attempt.status === "pass").length;
|
|
5366
|
+
return {
|
|
5367
|
+
batchId: batch.batchId,
|
|
5368
|
+
purpose: batch.purpose,
|
|
5369
|
+
passes,
|
|
5370
|
+
total: batchAttempts.length,
|
|
5371
|
+
passRate: rate(passes, batchAttempts.length),
|
|
5372
|
+
statuses: batchAttempts.map((attempt) => attempt.status)
|
|
5373
|
+
};
|
|
5374
|
+
})
|
|
5375
|
+
};
|
|
5376
|
+
}
|
|
5220
5377
|
function createMemoryExtractionBenchmarkCommand() {
|
|
5221
5378
|
const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run").option("--repetitions <n>", "run every benchmark batch 1-10 times to measure model stability", "1");
|
|
5222
5379
|
addDbOption(cmd);
|
|
@@ -5255,21 +5412,54 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5255
5412
|
observerExplicitConfigKeys: maxOutputTokens === null ? observerConfig.observerExplicitConfigKeys : [.../* @__PURE__ */ new Set([...observerConfig.observerExplicitConfigKeys ?? [], "observerMaxOutputTokens"])]
|
|
5256
5413
|
};
|
|
5257
5414
|
const observer = new ObserverClient(observerConfigWithOverrides);
|
|
5415
|
+
const outputFailures = [];
|
|
5258
5416
|
const runs = [];
|
|
5259
5417
|
for (let iteration = 1; iteration <= repetitions; iteration += 1) for (const batch of benchmark.batches) {
|
|
5260
5418
|
const scenarioId = batch.scenarioId ?? benchmark.scenarioId;
|
|
5261
|
-
|
|
5262
|
-
|
|
5263
|
-
|
|
5264
|
-
|
|
5265
|
-
|
|
5266
|
-
|
|
5267
|
-
|
|
5268
|
-
|
|
5269
|
-
|
|
5419
|
+
let selectedObserver = extractionReplayObserverIdentity(observer, null);
|
|
5420
|
+
let result;
|
|
5421
|
+
try {
|
|
5422
|
+
result = opts.observerTierRouting === true ? await replayBatchExtractionWithTierRouting(resolveDbOpt(opts), observerConfigWithOverrides, {
|
|
5423
|
+
batchId: batch.batchId,
|
|
5424
|
+
scenarioId,
|
|
5425
|
+
transcriptBudget: transcriptBudget ?? void 0,
|
|
5426
|
+
onOutputFailure: (context) => {
|
|
5427
|
+
selectedObserver = context;
|
|
5428
|
+
}
|
|
5429
|
+
}) : await replayBatchExtraction(resolveDbOpt(opts), observer, {
|
|
5430
|
+
batchId: batch.batchId,
|
|
5431
|
+
scenarioId,
|
|
5432
|
+
transcriptBudget: transcriptBudget ?? void 0
|
|
5433
|
+
});
|
|
5434
|
+
} catch (error) {
|
|
5435
|
+
if (!(error instanceof ObserverOutputError) && !(error instanceof ObserverOutputTransportError)) throw error;
|
|
5436
|
+
if (opts.observerTierRouting !== true) selectedObserver = extractionReplayObserverIdentity(observer, null);
|
|
5437
|
+
const failureCost = estimateBenchmarkAttemptCost(selectedObserver.model, error.telemetry.totalUsage);
|
|
5438
|
+
outputFailures.push({
|
|
5439
|
+
iteration,
|
|
5440
|
+
batchId: batch.batchId,
|
|
5441
|
+
purpose: batch.purpose,
|
|
5442
|
+
status: "output_failure",
|
|
5443
|
+
reason: error instanceof ObserverOutputError ? error.reason : error.code,
|
|
5444
|
+
expectedTier: batch.expectedTier ?? null,
|
|
5445
|
+
...selectedObserver,
|
|
5446
|
+
requestedOutputMode: error.diagnostics.requestedMode,
|
|
5447
|
+
actualOutputMode: error.diagnostics.actualMode,
|
|
5448
|
+
outputValidation: "invalid",
|
|
5449
|
+
repairAttempted: error.diagnostics.repairAttempted,
|
|
5450
|
+
retryAttempted: error.diagnostics.retryAttempted,
|
|
5451
|
+
observations: 0,
|
|
5452
|
+
telemetry: error.telemetry,
|
|
5453
|
+
cost: failureCost,
|
|
5454
|
+
quality: null
|
|
5455
|
+
});
|
|
5456
|
+
continue;
|
|
5457
|
+
}
|
|
5270
5458
|
const costModel = result.observer.modelFallbackApplied ? result.observer.resolvedModel : result.observer.resolvedModel ?? result.observer.model;
|
|
5271
5459
|
const initialCost = costModel ? estimateExtractionModelCost(costModel, result.observer.initialUsage) : null;
|
|
5272
|
-
const
|
|
5460
|
+
const secondAttemptCost = costModel ? estimateExtractionModelCost(costModel, result.observer.repairedUsage) : null;
|
|
5461
|
+
const repairCost = benchmarkSecondAttemptValue(result.observer.repairAttempted, secondAttemptCost);
|
|
5462
|
+
const retryCost = benchmarkSecondAttemptValue(result.observer.retryAttempted, secondAttemptCost);
|
|
5273
5463
|
const totalCost = costModel ? estimateExtractionModelCost(costModel, result.observer.totalUsage) : null;
|
|
5274
5464
|
const pricing = costModel ? getExtractionModelPricing(costModel) : null;
|
|
5275
5465
|
const costUnavailableReason = totalCost ? null : result.observer.modelFallbackApplied && !result.observer.resolvedModel ? "model_fallback_unresolved" : result.observer.totalUsage == null ? "missing_usage" : "unknown_model_pricing";
|
|
@@ -5283,14 +5473,14 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5283
5473
|
estimatedCostUsd: initialCost?.totalCostUsd ?? null,
|
|
5284
5474
|
expectedSummaryDisposition: batch.expectedSummaryDisposition
|
|
5285
5475
|
}) : null;
|
|
5286
|
-
const
|
|
5476
|
+
const secondAttemptQuality = result.observer.repairedParsed && result.observer.repairedDiagnostics ? scoreExtractionBenchmarkOutput({
|
|
5287
5477
|
parsed: result.observer.repairedParsed,
|
|
5288
5478
|
diagnostics: result.observer.repairedDiagnostics,
|
|
5289
5479
|
review: batch.review ?? {
|
|
5290
5480
|
status: "unreviewed",
|
|
5291
5481
|
reviewerNotes: "No durable-fact review has been recorded for this batch."
|
|
5292
5482
|
},
|
|
5293
|
-
estimatedCostUsd:
|
|
5483
|
+
estimatedCostUsd: secondAttemptCost?.totalCostUsd ?? null,
|
|
5294
5484
|
expectedSummaryDisposition: batch.expectedSummaryDisposition
|
|
5295
5485
|
}) : null;
|
|
5296
5486
|
const finalQuality = result.observer.diagnostics ? scoreExtractionBenchmarkOutput({
|
|
@@ -5310,6 +5500,16 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5310
5500
|
initialQuality,
|
|
5311
5501
|
finalQuality
|
|
5312
5502
|
});
|
|
5503
|
+
const repairAttempt = buildBenchmarkSecondAttempt(result, {
|
|
5504
|
+
attempted: result.observer.repairAttempted,
|
|
5505
|
+
reason: null,
|
|
5506
|
+
quality: secondAttemptQuality
|
|
5507
|
+
});
|
|
5508
|
+
const retryAttempt = buildBenchmarkSecondAttempt(result, {
|
|
5509
|
+
attempted: result.observer.retryAttempted,
|
|
5510
|
+
reason: result.observer.retryReason,
|
|
5511
|
+
quality: secondAttemptQuality
|
|
5512
|
+
});
|
|
5313
5513
|
runs.push({
|
|
5314
5514
|
iteration,
|
|
5315
5515
|
batchId: batch.batchId,
|
|
@@ -5336,6 +5536,9 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5336
5536
|
resolvedModel: result.observer.resolvedModel,
|
|
5337
5537
|
modelFallbackApplied: result.observer.modelFallbackApplied,
|
|
5338
5538
|
modelFallbackReason: result.observer.modelFallbackReason,
|
|
5539
|
+
requestedOutputMode: result.observer.requestedOutputMode,
|
|
5540
|
+
actualOutputMode: result.observer.actualOutputMode,
|
|
5541
|
+
outputValidation: result.observer.outputValidation,
|
|
5339
5542
|
openaiUseResponses: result.observer.openaiUseResponses,
|
|
5340
5543
|
reasoningEffort: result.observer.reasoningEffort,
|
|
5341
5544
|
reasoningSummary: result.observer.reasoningSummary,
|
|
@@ -5344,6 +5547,9 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5344
5547
|
summaries: result.evaluation.counts.summaries,
|
|
5345
5548
|
observations: result.evaluation.counts.observations,
|
|
5346
5549
|
repairApplied: result.observer.repairApplied,
|
|
5550
|
+
repairAttempted: result.observer.repairAttempted,
|
|
5551
|
+
retryAttempted: result.observer.retryAttempted,
|
|
5552
|
+
retryReason: result.observer.retryReason,
|
|
5347
5553
|
initial: {
|
|
5348
5554
|
raw: result.observer.initialRaw,
|
|
5349
5555
|
status: result.initialClassification.status,
|
|
@@ -5359,18 +5565,19 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5359
5565
|
},
|
|
5360
5566
|
repair: {
|
|
5361
5567
|
applied: result.observer.repairApplied,
|
|
5362
|
-
raw:
|
|
5363
|
-
status:
|
|
5364
|
-
reason:
|
|
5365
|
-
pass:
|
|
5366
|
-
failureReasons:
|
|
5367
|
-
summaries:
|
|
5368
|
-
observations:
|
|
5369
|
-
diagnostics:
|
|
5370
|
-
elapsedMs:
|
|
5371
|
-
usage:
|
|
5372
|
-
quality:
|
|
5568
|
+
raw: repairAttempt.raw,
|
|
5569
|
+
status: repairAttempt.status,
|
|
5570
|
+
reason: repairAttempt.resultReason,
|
|
5571
|
+
pass: repairAttempt.pass,
|
|
5572
|
+
failureReasons: repairAttempt.failureReasons,
|
|
5573
|
+
summaries: repairAttempt.summaries,
|
|
5574
|
+
observations: repairAttempt.observations,
|
|
5575
|
+
diagnostics: repairAttempt.diagnostics,
|
|
5576
|
+
elapsedMs: repairAttempt.elapsedMs,
|
|
5577
|
+
usage: repairAttempt.usage,
|
|
5578
|
+
quality: repairAttempt.quality
|
|
5373
5579
|
},
|
|
5580
|
+
retry: retryAttempt,
|
|
5374
5581
|
telemetry: {
|
|
5375
5582
|
totalElapsedMs: result.observer.totalElapsedMs,
|
|
5376
5583
|
totalUsage: result.observer.totalUsage
|
|
@@ -5379,6 +5586,7 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5379
5586
|
cost: {
|
|
5380
5587
|
initial: initialCost,
|
|
5381
5588
|
repair: repairCost,
|
|
5589
|
+
retry: retryCost,
|
|
5382
5590
|
total: totalCost,
|
|
5383
5591
|
unavailableReason: costUnavailableReason
|
|
5384
5592
|
},
|
|
@@ -5386,68 +5594,55 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5386
5594
|
});
|
|
5387
5595
|
}
|
|
5388
5596
|
const reviewedQualityRuns = runs.filter((run) => run.quality?.weightedQualityScore != null);
|
|
5389
|
-
const
|
|
5390
|
-
const
|
|
5597
|
+
const attempts = [...runs, ...outputFailures];
|
|
5598
|
+
const knownCostRuns = attempts.filter((run) => run.cost.total != null);
|
|
5599
|
+
const attemptSummary = summarizeExtractionBenchmarkAttempts(attempts, benchmark.batches);
|
|
5600
|
+
const knownElapsedRuns = attempts.filter((run) => run.telemetry.totalElapsedMs != null);
|
|
5391
5601
|
const summary = {
|
|
5392
5602
|
repetitions,
|
|
5393
|
-
|
|
5394
|
-
shapeQualityTotal: runs.filter((run) => run.purpose === "shape_quality").length,
|
|
5395
|
-
shapeQualityPasses: runs.filter((run) => run.purpose === "shape_quality" && run.status === "pass").length,
|
|
5396
|
-
shapeQualityFails: runs.filter((run) => run.purpose === "shape_quality" && run.status === "shape_fail").length,
|
|
5397
|
-
expectedTierTotal: runs.filter((run) => run.expectedTier != null).length,
|
|
5398
|
-
expectedTierMatches: runs.filter((run) => run.expectedTier != null && run.expectedTier === run.tier).length,
|
|
5399
|
-
robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length,
|
|
5603
|
+
...attemptSummary,
|
|
5400
5604
|
summaryDispositionTotal: runs.filter((run) => run.quality != null).length,
|
|
5401
5605
|
summaryDispositionMatches: runs.filter((run) => run.quality?.summaryDisposition.score === 1).length,
|
|
5402
5606
|
reviewedQualityRuns: reviewedQualityRuns.length,
|
|
5403
5607
|
knownCostRuns: knownCostRuns.length,
|
|
5404
|
-
unknownCostRuns:
|
|
5405
|
-
missingUsageRuns:
|
|
5406
|
-
unknownPricingRuns:
|
|
5407
|
-
fallbackUnresolvedRuns:
|
|
5608
|
+
unknownCostRuns: attempts.length - knownCostRuns.length,
|
|
5609
|
+
missingUsageRuns: attempts.filter((run) => run.cost.unavailableReason === "missing_usage").length,
|
|
5610
|
+
unknownPricingRuns: attempts.filter((run) => run.cost.unavailableReason === "unknown_model_pricing").length,
|
|
5611
|
+
fallbackUnresolvedRuns: attempts.filter((run) => run.cost.unavailableReason === "model_fallback_unresolved").length,
|
|
5408
5612
|
totalKnownCostUsd: knownCostRuns.reduce((sum, run) => sum + (run.cost.total?.totalCostUsd ?? 0), 0),
|
|
5409
5613
|
knownElapsedRuns: knownElapsedRuns.length,
|
|
5410
5614
|
totalKnownElapsedMs: knownElapsedRuns.reduce((sum, run) => sum + (run.telemetry.totalElapsedMs ?? 0), 0),
|
|
5411
|
-
|
|
5412
|
-
|
|
5413
|
-
const passes = batchRuns.filter((run) => run.status === "pass").length;
|
|
5414
|
-
return {
|
|
5415
|
-
batchId: batch.batchId,
|
|
5416
|
-
purpose: batch.purpose,
|
|
5417
|
-
passes,
|
|
5418
|
-
total: batchRuns.length,
|
|
5419
|
-
passRate: batchRuns.length > 0 ? passes / batchRuns.length : null,
|
|
5420
|
-
statuses: batchRuns.map((run) => run.status)
|
|
5421
|
-
};
|
|
5422
|
-
})
|
|
5615
|
+
output: summarizeExtractionBenchmarkReporting(attempts),
|
|
5616
|
+
outputFailures
|
|
5423
5617
|
};
|
|
5424
|
-
const uniqueObserverKeys = Array.from(new Set(
|
|
5425
|
-
const
|
|
5618
|
+
const uniqueObserverKeys = Array.from(new Set(attempts.map((run) => `${run.provider}::${run.model}::${run.transport}`)));
|
|
5619
|
+
const firstObserver = attempts[0];
|
|
5620
|
+
const benchmarkReasoning = summarizeBenchmarkReasoning(attempts, {
|
|
5426
5621
|
reasoningEffort: observer.reasoningEffort,
|
|
5427
5622
|
reasoningSummary: observer.reasoningSummary
|
|
5428
5623
|
});
|
|
5429
5624
|
const observerSummary = opts.observerTierRouting === true ? {
|
|
5430
|
-
provider: uniqueObserverKeys.length === 1 ?
|
|
5431
|
-
model: uniqueObserverKeys.length === 1 ?
|
|
5432
|
-
transport: uniqueObserverKeys.length === 1 ?
|
|
5625
|
+
provider: uniqueObserverKeys.length === 1 ? firstObserver?.provider ?? observer.provider : "mixed",
|
|
5626
|
+
model: uniqueObserverKeys.length === 1 ? firstObserver?.model ?? observer.model : "mixed",
|
|
5627
|
+
transport: uniqueObserverKeys.length === 1 ? firstObserver?.transport ?? "unknown" : "mixed",
|
|
5433
5628
|
tierRouting: true,
|
|
5434
|
-
openaiUseResponses: uniqueObserverKeys.length === 1 ?
|
|
5629
|
+
openaiUseResponses: uniqueObserverKeys.length === 1 ? firstObserver?.openaiUseResponses ?? observer.openaiUseResponses : null,
|
|
5435
5630
|
reasoningEffort: benchmarkReasoning.reasoningEffort,
|
|
5436
5631
|
reasoningSummary: benchmarkReasoning.reasoningSummary,
|
|
5437
|
-
maxOutputTokens: uniqueObserverKeys.length === 1 ?
|
|
5438
|
-
temperature: uniqueObserverKeys.length === 1 ?
|
|
5632
|
+
maxOutputTokens: uniqueObserverKeys.length === 1 ? firstObserver?.maxOutputTokens ?? null : null,
|
|
5633
|
+
temperature: uniqueObserverKeys.length === 1 ? firstObserver?.temperature ?? null : null,
|
|
5439
5634
|
transcriptBudget: transcriptBudget ?? null,
|
|
5440
5635
|
selectedObservers: uniqueObserverKeys
|
|
5441
5636
|
} : {
|
|
5442
5637
|
provider: observer.provider,
|
|
5443
5638
|
model: observer.model,
|
|
5444
|
-
transport:
|
|
5639
|
+
transport: firstObserver?.transport ?? observer.getStatus().runtime,
|
|
5445
5640
|
tierRouting: false,
|
|
5446
5641
|
openaiUseResponses: observer.openaiUseResponses,
|
|
5447
5642
|
reasoningEffort: benchmarkReasoning.reasoningEffort,
|
|
5448
5643
|
reasoningSummary: benchmarkReasoning.reasoningSummary,
|
|
5449
|
-
maxOutputTokens:
|
|
5450
|
-
temperature:
|
|
5644
|
+
maxOutputTokens: firstObserver?.maxOutputTokens ?? null,
|
|
5645
|
+
temperature: firstObserver?.temperature ?? null,
|
|
5451
5646
|
transcriptBudget: transcriptBudget ?? null,
|
|
5452
5647
|
selectedObservers: uniqueObserverKeys
|
|
5453
5648
|
};
|
|
@@ -5486,14 +5681,23 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5486
5681
|
`Summary disposition matches: ${summary.summaryDispositionMatches}/${summary.summaryDispositionTotal}`,
|
|
5487
5682
|
`Reviewed quality runs: ${summary.reviewedQualityRuns} (compare per-run dimensions; scores are fixture-specific)`,
|
|
5488
5683
|
`Known estimated cost: $${summary.totalKnownCostUsd.toFixed(6)} (${summary.knownCostRuns}/${summary.total}; missing usage=${summary.missingUsageRuns}, unknown pricing=${summary.unknownPricingRuns}, unresolved fallback=${summary.fallbackUnresolvedRuns})`,
|
|
5489
|
-
`Known elapsed time: ${summary.totalKnownElapsedMs}ms (${summary.knownElapsedRuns}/${summary.total} run(s))
|
|
5684
|
+
`Known elapsed time: ${summary.totalKnownElapsedMs}ms (${summary.knownElapsedRuns}/${summary.total} run(s))`,
|
|
5685
|
+
`Output modes requested/actual: ${JSON.stringify(summary.output.requestedOutputModes)} / ${JSON.stringify(summary.output.actualOutputModes)}`,
|
|
5686
|
+
`Valid output: ${summary.output.overall.validOutputs}/${summary.output.overall.runs}; repair/retry: ${summary.output.overall.repairAttempts}/${summary.output.overall.retryAttempts}; mean latency: ${summary.output.overall.meanLatencyMs ?? "n/a"}ms`,
|
|
5687
|
+
`Tokens input/output: ${summary.output.overall.inputTokens}/${summary.output.overall.outputTokens} (${summary.output.overall.knownTokenRuns} measured run(s)); retained observations: ${summary.output.overall.retainedObservations}`,
|
|
5688
|
+
`Summary disposition matches: ${summary.output.overall.summaryDispositionMatches}/${summary.output.overall.summaryDispositionEvaluated}; contract integrity/semantic quality: ${summary.output.overall.contractIntegrityRate?.toFixed(3) ?? "n/a"}/${summary.output.overall.meanSemanticQuality?.toFixed(3) ?? "n/a"}`
|
|
5490
5689
|
].join("\n"));
|
|
5491
5690
|
for (const run of runs) {
|
|
5492
5691
|
const qualityLabel = run.quality?.weightedQualityScore == null ? "n/a" : run.quality.weightedQualityScore.toFixed(3);
|
|
5493
5692
|
const costLabel = run.cost.total == null ? "n/a" : `$${run.cost.total.totalCostUsd.toFixed(6)}`;
|
|
5494
5693
|
const latencyLabel = run.telemetry.totalElapsedMs == null ? "n/a" : `${run.telemetry.totalElapsedMs}ms`;
|
|
5495
5694
|
const missingRequired = run.quality?.requiredRecall.missingLabelIds.join(",") || "none";
|
|
5496
|
-
p.log.message(` [${run.batchId}#${run.iteration}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} disposition=${run.quality?.summaryDisposition.actual ?? "n/a"}/${run.expectedSummaryDisposition} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model} [${run.transport}] initial=${run.initial.summaries}s/${run.initial.observations}o final=${run.summaries}s/${run.observations}o quality=${qualityLabel} coverage=${run.quality?.weightedQualityCoverage?.toFixed(3) ?? "n/a"} required_missing=${missingRequired} cost=${costLabel} latency=${latencyLabel} schema_loss=${run.initial.diagnostics?.dataLoss === true ? "yes" : "no"} fallback=${run.modelFallbackApplied ? "yes" : "no"} repair=${run.repairApplied ? "yes" : "no"} — ${run.label}`);
|
|
5695
|
+
p.log.message(` [${run.batchId}#${run.iteration}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} mode=${run.requestedOutputMode}/${run.actualOutputMode} valid=${run.outputValidation} disposition=${run.quality?.summaryDisposition.actual ?? "n/a"}/${run.expectedSummaryDisposition} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model} [${run.transport}] initial=${run.initial.summaries}s/${run.initial.observations}o final=${run.summaries}s/${run.observations}o quality=${qualityLabel} coverage=${run.quality?.weightedQualityCoverage?.toFixed(3) ?? "n/a"} required_missing=${missingRequired} cost=${costLabel} latency=${latencyLabel} schema_loss=${run.initial.diagnostics?.dataLoss === true ? "yes" : "no"} fallback=${run.modelFallbackApplied ? "yes" : "no"} repair=${run.repairApplied ? "yes" : "no"} retry=${run.retryAttempted ? run.retryReason ?? "yes" : "no"} — ${run.label}`);
|
|
5696
|
+
}
|
|
5697
|
+
for (const failure of outputFailures.toSorted((left, right) => left.iteration - right.iteration || left.batchId - right.batchId)) {
|
|
5698
|
+
const costLabel = failure.cost.total == null ? "n/a" : `$${failure.cost.total.totalCostUsd.toFixed(6)}`;
|
|
5699
|
+
const latencyLabel = failure.telemetry.totalElapsedMs == null ? "n/a" : `${failure.telemetry.totalElapsedMs}ms`;
|
|
5700
|
+
p.log.error(` [${failure.batchId}#${failure.iteration}] output_failure tier=${failure.tier ?? "n/a"} expected=${failure.expectedTier ?? "n/a"} mode=${failure.requestedOutputMode}/${failure.actualOutputMode} model=${failure.model ?? "unresolved"} cost=${costLabel} latency=${latencyLabel} — ${failure.reason}`);
|
|
5497
5701
|
}
|
|
5498
5702
|
p.outro("done");
|
|
5499
5703
|
} catch (error) {
|
|
@@ -5609,6 +5813,8 @@ var FORBIDDEN_INTERNAL_KEYS = /* @__PURE__ */ new Set([
|
|
|
5609
5813
|
"title"
|
|
5610
5814
|
]);
|
|
5611
5815
|
var INTERNAL_LEDGER_KEYS = /* @__PURE__ */ new Set([
|
|
5816
|
+
"automatic_recall",
|
|
5817
|
+
"evaluation_key",
|
|
5612
5818
|
"action",
|
|
5613
5819
|
"attempt_id",
|
|
5614
5820
|
"started_at",
|
|
@@ -5626,7 +5832,8 @@ var INTERNAL_LEDGER_KEYS = /* @__PURE__ */ new Set([
|
|
|
5626
5832
|
var INTERNAL_LEDGER_ACTIONS = /* @__PURE__ */ new Set([
|
|
5627
5833
|
"record",
|
|
5628
5834
|
"delivery",
|
|
5629
|
-
"cache_reuse"
|
|
5835
|
+
"cache_reuse",
|
|
5836
|
+
"recall"
|
|
5630
5837
|
]);
|
|
5631
5838
|
var INTERNAL_RETRIEVAL_STATUSES = /* @__PURE__ */ new Set(["skipped", "failed"]);
|
|
5632
5839
|
var INTERNAL_DELIVERY_STATUSES = /* @__PURE__ */ new Set([
|
|
@@ -5785,12 +5992,28 @@ async function withStore(opts, errorCode, run) {
|
|
|
5785
5992
|
store?.close();
|
|
5786
5993
|
}
|
|
5787
5994
|
}
|
|
5995
|
+
function ledgerAutomaticContext(payload) {
|
|
5996
|
+
if (payload?.source?.trim().toLowerCase() !== "opencode") return null;
|
|
5997
|
+
if (!payload.source_session_id) return null;
|
|
5998
|
+
return {
|
|
5999
|
+
source: "opencode",
|
|
6000
|
+
hostSessionId: payload.source_session_id
|
|
6001
|
+
};
|
|
6002
|
+
}
|
|
6003
|
+
async function readOptionalLedgerPayload() {
|
|
6004
|
+
try {
|
|
6005
|
+
return await readInternalLedgerPayload();
|
|
6006
|
+
} catch {
|
|
6007
|
+
return;
|
|
6008
|
+
}
|
|
6009
|
+
}
|
|
5788
6010
|
async function packAction(context, opts) {
|
|
5789
6011
|
await withStore(opts, "pack_failed", async (store) => {
|
|
5790
6012
|
const { limit, budget, filters, renderOptions } = buildPackRequestOptions(opts, { envProject: process.env.CODEMEM_PROJECT });
|
|
5791
6013
|
let result;
|
|
5792
6014
|
if (opts.internalLedger) {
|
|
5793
|
-
const
|
|
6015
|
+
const ledgerPayload = await readOptionalLedgerPayload();
|
|
6016
|
+
const artifacts = await store.buildMemoryPackWithTraceAsync(context, limit, budget, filters, renderOptions, ledgerAutomaticContext(ledgerPayload));
|
|
5794
6017
|
result = artifacts.response;
|
|
5795
6018
|
let artifactFingerprint;
|
|
5796
6019
|
try {
|
|
@@ -5798,12 +6021,13 @@ async function packAction(context, opts) {
|
|
|
5798
6021
|
} catch {}
|
|
5799
6022
|
let ledgerOutcome;
|
|
5800
6023
|
try {
|
|
5801
|
-
|
|
6024
|
+
if (!ledgerPayload) throw new Error("internal ledger metadata unavailable");
|
|
5802
6025
|
ledgerOutcome = handleInstrumentedPackLedger(store.db, ledgerPayload, context, filters, artifacts);
|
|
5803
6026
|
} catch {}
|
|
5804
6027
|
emitPackResult(context, opts, result, artifactFingerprint, ledgerOutcome?.ok === false && ledgerOutcome.reason === "idempotency_conflict" ? ledgerOutcome : void 0);
|
|
5805
6028
|
return;
|
|
5806
|
-
}
|
|
6029
|
+
}
|
|
6030
|
+
result = await store.buildMemoryPackAsync(context, limit, budget, filters, renderOptions);
|
|
5807
6031
|
emitPackResult(context, opts, result);
|
|
5808
6032
|
});
|
|
5809
6033
|
}
|
|
@@ -5850,7 +6074,30 @@ addJsonOption(traceCmd);
|
|
|
5850
6074
|
traceCmd.action(traceAction);
|
|
5851
6075
|
packCmd.addCommand(traceCmd);
|
|
5852
6076
|
var packCommand = packCmd;
|
|
6077
|
+
function handleRecallMeasurement(db, payload) {
|
|
6078
|
+
if (payload.automatic_recall === void 0 && payload.action !== "recall") return null;
|
|
6079
|
+
const isValidMeasurement = isAutomaticRecallMeasurement(payload.automatic_recall) && typeof payload.evaluation_key === "string" && /^[a-f0-9]{64}$/.test(payload.evaluation_key);
|
|
6080
|
+
if (payload.action === "delivery") {
|
|
6081
|
+
if (!isValidMeasurement) return null;
|
|
6082
|
+
try {
|
|
6083
|
+
recordAutomaticRecall(db, payload.attempt_id, payload.evaluation_key, payload.automatic_recall);
|
|
6084
|
+
} catch {}
|
|
6085
|
+
return null;
|
|
6086
|
+
}
|
|
6087
|
+
if (payload.action !== "recall" || !isValidMeasurement) throw new PackUsageError("invalid automatic recall measurement");
|
|
6088
|
+
const outcome = recordAutomaticRecall(db, payload.attempt_id, payload.evaluation_key, payload.automatic_recall);
|
|
6089
|
+
if (!outcome.ok) throw new PackUsageError(outcome.reason);
|
|
6090
|
+
return outcome.value;
|
|
6091
|
+
}
|
|
5853
6092
|
function handlePromptPackLedger(db, payload) {
|
|
6093
|
+
if (payload.action === "delivery") {
|
|
6094
|
+
const outcome = handlePromptPackLifecycle(db, payload);
|
|
6095
|
+
handleRecallMeasurement(db, payload);
|
|
6096
|
+
return outcome;
|
|
6097
|
+
}
|
|
6098
|
+
return handleRecallMeasurement(db, payload) ?? handlePromptPackLifecycle(db, payload);
|
|
6099
|
+
}
|
|
6100
|
+
function handlePromptPackLifecycle(db, payload) {
|
|
5854
6101
|
const metadata = attemptMetadata(payload);
|
|
5855
6102
|
if (payload.action === "delivery") {
|
|
5856
6103
|
const status = payload.delivery_status;
|
|
@@ -6460,6 +6707,51 @@ function sqliteVecFailureDiagnostics(error, dbPath) {
|
|
|
6460
6707
|
`error=${message}`
|
|
6461
6708
|
];
|
|
6462
6709
|
}
|
|
6710
|
+
function requiredNativeRuntimeFailureDiagnostics(error, context = {
|
|
6711
|
+
version: VERSION,
|
|
6712
|
+
nodeVersion: process.version,
|
|
6713
|
+
platform: process.platform,
|
|
6714
|
+
arch: process.arch
|
|
6715
|
+
}) {
|
|
6716
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
6717
|
+
return [
|
|
6718
|
+
"Required SQLite native binding is unavailable.",
|
|
6719
|
+
`Runtime: Node ${context.nodeVersion} on ${context.platform}/${context.arch}.`,
|
|
6720
|
+
`Details: ${detail}`,
|
|
6721
|
+
`Reinstall this exact codemem version: npm install -g codemem@${context.version}`,
|
|
6722
|
+
"Use Node 24.15+ on a supported 64-bit macOS, Linux, or Windows target; see the native install matrix for current platform coverage.",
|
|
6723
|
+
"This required better-sqlite3 failure is separate from optional sqlite-vec loading, which can fall back to lexical search."
|
|
6724
|
+
];
|
|
6725
|
+
}
|
|
6726
|
+
async function preflightServeNativeRuntime(invocation, dependencies = {
|
|
6727
|
+
isPortOpen,
|
|
6728
|
+
probe: probeRequiredNativeRuntime
|
|
6729
|
+
}) {
|
|
6730
|
+
if (invocation.mode === "start" && await dependencies.isPortOpen(invocation.host, invocation.port)) return { state: "already_running" };
|
|
6731
|
+
try {
|
|
6732
|
+
dependencies.probe();
|
|
6733
|
+
return { state: "ready" };
|
|
6734
|
+
} catch (error) {
|
|
6735
|
+
return {
|
|
6736
|
+
state: "unavailable",
|
|
6737
|
+
diagnostics: requiredNativeRuntimeFailureDiagnostics(error)
|
|
6738
|
+
};
|
|
6739
|
+
}
|
|
6740
|
+
}
|
|
6741
|
+
async function nativeRuntimeAllowsServeStart(invocation) {
|
|
6742
|
+
if (invocation.mode !== "start" && invocation.mode !== "restart") return true;
|
|
6743
|
+
const preflight = await preflightServeNativeRuntime(invocation);
|
|
6744
|
+
if (preflight.state === "already_running") {
|
|
6745
|
+
p.log.warn(`Viewer already running at http://${invocation.host}:${invocation.port}`);
|
|
6746
|
+
if (!invocation.background) process.exitCode = 1;
|
|
6747
|
+
return false;
|
|
6748
|
+
}
|
|
6749
|
+
if (preflight.state === "ready") return true;
|
|
6750
|
+
p.intro("codemem viewer");
|
|
6751
|
+
for (const line of preflight.diagnostics) p.log.error(line);
|
|
6752
|
+
process.exitCode = 1;
|
|
6753
|
+
return false;
|
|
6754
|
+
}
|
|
6463
6755
|
async function runServeCoordinatorMaintenance(store, dependencies) {
|
|
6464
6756
|
const projectShares = await dependencies.advancePendingProjectShares(store, { limit: 3 });
|
|
6465
6757
|
const coordinatorEnrollment = await dependencies.reconcileConfiguredCoordinatorEnrollment(store);
|
|
@@ -6518,11 +6810,11 @@ async function startForegroundViewer(invocation) {
|
|
|
6518
6810
|
if (invocation.dbPath) process.env.CODEMEM_DB = invocation.dbPath;
|
|
6519
6811
|
if (invocation.configPath) process.env.CODEMEM_CONFIG = invocation.configPath;
|
|
6520
6812
|
warnIfViewerExposed(invocation.host, invocation.port);
|
|
6521
|
-
|
|
6522
|
-
|
|
6523
|
-
|
|
6524
|
-
|
|
6525
|
-
}
|
|
6813
|
+
const dbPath = resolveDbPath(invocation.dbPath ?? void 0);
|
|
6814
|
+
if (!await terminateTrustedMaintenanceWorker(dbPath, {
|
|
6815
|
+
gracefulMs: 1e3,
|
|
6816
|
+
forceMs: 5e3
|
|
6817
|
+
})) p.log.warn("Existing maintenance worker is not trusted or did not stop; starting viewer anyway");
|
|
6526
6818
|
const preparedDb = prepareViewerDatabase(invocation.dbPath);
|
|
6527
6819
|
const observer = new ObserverClient();
|
|
6528
6820
|
let store;
|
|
@@ -6543,11 +6835,6 @@ async function startForegroundViewer(invocation) {
|
|
|
6543
6835
|
const config = invocation.configPath ? readCodememConfigFileAtPath(invocation.configPath) : readCodememConfigFile();
|
|
6544
6836
|
const syncConfig = readCoordinatorSyncConfig(config);
|
|
6545
6837
|
const syncEnabled = syncConfig.syncEnabled;
|
|
6546
|
-
const dbPath = resolveDbPath(invocation.dbPath ?? void 0);
|
|
6547
|
-
if (!await terminateTrustedMaintenanceWorker(dbPath, {
|
|
6548
|
-
gracefulMs: 1e3,
|
|
6549
|
-
forceMs: 5e3
|
|
6550
|
-
})) p.log.warn("Existing maintenance worker is not trusted or did not stop; starting viewer anyway");
|
|
6551
6838
|
const syncRuntimeStatus = {
|
|
6552
6839
|
phase: syncEnabled ? "starting" : "disabled",
|
|
6553
6840
|
detail: syncEnabled ? "Waiting for viewer startup to finish" : "Sync is disabled"
|
|
@@ -6683,7 +6970,10 @@ async function startForegroundViewer(invocation) {
|
|
|
6683
6970
|
shutdown();
|
|
6684
6971
|
});
|
|
6685
6972
|
}
|
|
6686
|
-
async function runServeInvocation(invocation
|
|
6973
|
+
async function runServeInvocation(invocation, dependencies = {
|
|
6974
|
+
nativeRuntimeAllowsServeStart,
|
|
6975
|
+
stopExistingViewer
|
|
6976
|
+
}) {
|
|
6687
6977
|
const dbPath = resolveDbPath(invocation.dbPath ?? void 0);
|
|
6688
6978
|
const runtimeConflict = await findRuntimeViewerConflict(dbPath, {
|
|
6689
6979
|
host: invocation.host,
|
|
@@ -6696,8 +6986,9 @@ async function runServeInvocation(invocation) {
|
|
|
6696
6986
|
process.exitCode = 1;
|
|
6697
6987
|
return;
|
|
6698
6988
|
}
|
|
6989
|
+
if (!await dependencies.nativeRuntimeAllowsServeStart(invocation)) return;
|
|
6699
6990
|
if (invocation.mode === "stop" || invocation.mode === "restart") {
|
|
6700
|
-
const result = await stopExistingViewer(dbPath, {
|
|
6991
|
+
const result = await dependencies.stopExistingViewer(dbPath, {
|
|
6701
6992
|
host: invocation.host,
|
|
6702
6993
|
port: invocation.port
|
|
6703
6994
|
});
|
|
@@ -7337,7 +7628,7 @@ function installCodex(force) {
|
|
|
7337
7628
|
}
|
|
7338
7629
|
if (codememOnPath(true)) p.log.info("Codex hooks will call `codemem` directly (found on PATH).");
|
|
7339
7630
|
else {
|
|
7340
|
-
const globalInstallCommand = process.platform === "linux" ? "env ONNXRUNTIME_NODE_INSTALL=skip npm install -g codemem
|
|
7631
|
+
const globalInstallCommand = process.platform === "linux" ? "env ONNXRUNTIME_NODE_INSTALL=skip npm install -g codemem" : "npm install -g codemem";
|
|
7341
7632
|
p.log.info(`\`codemem\` is not on PATH, so Codex hooks will run via \`npx -y --package codemem --package @codemem/embeddings codemem\` (works without a global install). For lower hook latency: ${globalInstallCommand}`);
|
|
7342
7633
|
}
|
|
7343
7634
|
let ok = true;
|
|
@@ -8970,10 +9261,10 @@ function renderStatusWarning(error) {
|
|
|
8970
9261
|
function renderHumanStatus(status) {
|
|
8971
9262
|
const warning = renderStatusWarning(status.error);
|
|
8972
9263
|
if (status.update_available) return `Update available${cacheQualifier(status.stale)}: ${status.current_version} → ${status.latest_version}. ${status.recommended_action}${warning}`;
|
|
8973
|
-
if (!
|
|
9264
|
+
if (!status.channel || !isReleaseVersionForChannel(status.current_version, status.channel)) return `Unable to compare current version ${status.current_version} with ${status.latest_version}. ${status.recommended_action}${warning}`;
|
|
8974
9265
|
return `${status.current_version} is up to date${cacheQualifier(status.stale)}.${warning}`;
|
|
8975
9266
|
}
|
|
8976
|
-
var checkCommand = addJsonOption(new Command("check").description("Check for a newer
|
|
9267
|
+
var checkCommand = addJsonOption(new Command("check").description("Check for a newer codemem release on the installed channel")).addOption(new Option("-r, --refresh", "bypass the six-hour release cache")).configureHelp(helpStyle).action(async (options) => {
|
|
8977
9268
|
try {
|
|
8978
9269
|
const installKind = detectInstallKind({
|
|
8979
9270
|
entryPath: process.argv[1] ?? "",
|
|
@@ -9037,8 +9328,8 @@ async function installUpdate(options) {
|
|
|
9037
9328
|
return;
|
|
9038
9329
|
}
|
|
9039
9330
|
const targetVersion = status.latest_version;
|
|
9040
|
-
if (!
|
|
9041
|
-
failInstall(options, "update_install_refused", "release version
|
|
9331
|
+
if (!status.channel || !isReleaseVersionForChannel(targetVersion, status.channel)) {
|
|
9332
|
+
failInstall(options, "update_install_refused", "release version does not match the installed channel");
|
|
9042
9333
|
return;
|
|
9043
9334
|
}
|
|
9044
9335
|
releaseInstallLock = await acquireInstallLock();
|
|
@@ -9076,7 +9367,7 @@ async function installUpdate(options) {
|
|
|
9076
9367
|
await releaseInstallLock?.();
|
|
9077
9368
|
}
|
|
9078
9369
|
}
|
|
9079
|
-
var installCommand = addJsonOption(new Command("install").description("Install an eligible
|
|
9370
|
+
var installCommand = addJsonOption(new Command("install").description("Install an eligible codemem update on the installed channel")).configureHelp(helpStyle).action(installUpdate);
|
|
9080
9371
|
var updateCommand = new Command("update").description("Inspect and manage codemem updates").enablePositionalOptions().configureHelp(helpStyle).addCommand(checkCommand).addCommand(installCommand);
|
|
9081
9372
|
//#endregion
|
|
9082
9373
|
//#region src/commands/version.ts
|