codemem 0.38.0 → 0.40.0-alpha.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +184 -14
- package/dist/index.js.map +1 -1
- package/package.json +4 -4
package/dist/index.js
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
|
-
import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, REF_BACKFILL_JOB, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildAuthHeaders, buildBaseUrl, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, compareMemoryRoleReports, connect, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getExtractionBenchmarkProfile, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, initDatabase, isEmbeddingDisabled, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, planReplicationOpsAgePrune, projectMatchesFilter, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, setPeerProjectFilter, stripJsonComments, stripPrivateObj, stripTrailingCommas, syncPassPreflight, updatePeerAddresses, vacuumDatabase, writeCodememConfigFile } from "@codemem/core";
|
|
2
|
+
import { DEDUP_KEY_BACKFILL_JOB, DEFAULT_COORDINATOR_DB_PATH, DedupKeyBackfillRunner, MUTATING_TOOL_NAMES, MemoryStore, ObserverClient, REF_BACKFILL_JOB, RawEventSweeper, RefBackfillRunner, SCOPE_BACKFILL_JOB, SESSION_CONTEXT_BACKFILL_JOB, SUMMARY_DEDUP_BACKFILL_JOB, ScopeBackfillRunner, SessionContextBackfillRunner, SummaryDedupBackfillRunner, SyncRetentionRunner, VERSION, VectorModelMigrationRunner, aiBackfillStructuredContent, applyBootstrapSnapshot, applyDistillRule, backfillMemoryDedupKeys, backfillNarrativeFromBody, backfillTagsText, backfillVectors, buildAuthHeaders, buildBaseUrl, buildDistillReport, buildRawEventEnvelopeFromCodexHook, buildRawEventEnvelopeFromHook, compareMemoryRoleReports, connect, coordinatorCreateGroupAction, coordinatorCreateInviteAction, coordinatorCreateScopeAction, coordinatorDisableDeviceAction, coordinatorEnrollDeviceAction, coordinatorGrantScopeMembershipAction, coordinatorImportInviteAction, coordinatorListBootstrapGrantsAction, coordinatorListDevicesAction, coordinatorListGroupsAction, coordinatorListJoinRequestsAction, coordinatorListScopeMembershipsAction, coordinatorListScopesAction, coordinatorRemoveDeviceAction, coordinatorRenameDeviceAction, coordinatorReviewJoinRequestAction, coordinatorRevokeBootstrapGrantAction, coordinatorRevokeScopeMembershipAction, coordinatorUpdateScopeAction, createBetterSqliteCoordinatorApp, deactivateLowSignalMemories, deactivateLowSignalObservations, dedupNearDuplicateMemories, draftDistillRule, ensureDeviceIdentity, ensureSchemaBootstrapped, estimateExtractionModelCost, exportMemories, extractApplyPatchPaths, fetchAllSnapshotPages, fingerprintPublicKey, flushRawEvents, formatHostPort, getExtractionBenchmarkProfile, getExtractionModelPricing, getInjectionEvalScenarioPack, getInjectionEvalScenarioPrompts, getMaintenanceJob, getMemoryArtifactReport, getMemoryRoleReport, getRawEventRelinkPlan, getRawEventRelinkReport, getRawEventStatus, getSemanticIndexDiagnostics, getSessionExtractionEval, getSessionExtractionEvalScenario, getWorkspaceCodememConfigPath, hasPendingDedupKeyBackfill, hasPendingRefBackfill, hasPendingScopeBackfill, hasPendingSessionContextBackfill, hasPendingSummaryDedupBackfill, hasUnsyncedSharedMemoryChanges, importMemories, initDatabase, isEmbeddingDisabled, judgeDistillReport, listMaintenanceJobs, listPerPeerScopeSyncState, listRetentionScopeIds, loadObserverConfig, loadPublicKey, loadSqliteVec, mdnsEnabled, planReplicationOpsAgePrune, projectMatchesFilter, pruneReplicationOpsUntilCaughtUp, rawEventsGate, readCodememConfigFile, readCodememConfigFileAtPath, readCoordinatorSyncConfig, readImportPayload, renderUnifiedDiff, replayBatchExtraction, replayBatchExtractionWithTierRouting, requestJson, resolveCodememConfigPath, resolveDbPath, resolveHookProject, resolveProject, resolveProjectRoot, retryRawEventFailures, runSyncDaemon, runSyncPass, scanSecretsRetroactive, schema, scoreExtractionBenchmarkOutput, setPeerProjectFilter, stripJsonComments, stripPrivateObj, stripTrailingCommas, syncPassPreflight, updatePeerAddresses, vacuumDatabase, writeCodememConfigFile } from "@codemem/core";
|
|
3
3
|
import { Command, Option } from "commander";
|
|
4
4
|
import omelette from "omelette";
|
|
5
5
|
import { appendFileSync, copyFileSync, existsSync, mkdirSync, readFileSync, readdirSync, renameSync, rmSync, rmdirSync, statSync, unlinkSync, writeFileSync } from "node:fs";
|
|
@@ -2666,12 +2666,13 @@ function buildCoordinatorCommand() {
|
|
|
2666
2666
|
}
|
|
2667
2667
|
});
|
|
2668
2668
|
cmd.addCommand(listScopeMembersCmd);
|
|
2669
|
-
const grantScopeMemberCmd = new Command("grant-scope-member").configureHelp(helpStyle).description("Grant a device explicit access to a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").option("--role <role>", "membership role").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
|
|
2669
|
+
const grantScopeMemberCmd = new Command("grant-scope-member").configureHelp(helpStyle).description("Grant a device explicit access to a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").requiredOption("--effect-id <id>", "deterministic mutation effect id").option("--role <role>", "membership role").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
|
|
2670
2670
|
addDbOption(grantScopeMemberCmd);
|
|
2671
2671
|
addJsonOption(grantScopeMemberCmd);
|
|
2672
2672
|
grantScopeMemberCmd.action(async (groupId, scopeId, deviceId, opts) => {
|
|
2673
2673
|
try {
|
|
2674
2674
|
const membership = await coordinatorGrantScopeMembershipAction({
|
|
2675
|
+
effectId: opts.effectId,
|
|
2675
2676
|
groupId,
|
|
2676
2677
|
scopeId,
|
|
2677
2678
|
deviceId,
|
|
@@ -2699,12 +2700,13 @@ function buildCoordinatorCommand() {
|
|
|
2699
2700
|
}
|
|
2700
2701
|
});
|
|
2701
2702
|
cmd.addCommand(grantScopeMemberCmd);
|
|
2702
|
-
const revokeScopeMemberCmd = new Command("revoke-scope-member").configureHelp(helpStyle).description("Revoke a device from a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
|
|
2703
|
+
const revokeScopeMemberCmd = new Command("revoke-scope-member").configureHelp(helpStyle).description("Revoke a device from a Sharing domain").argument("<group>", "group id").argument("<scope-id>", "Sharing domain scope_id").argument("<device-id>", "device id").requiredOption("--effect-id <id>", "deterministic mutation effect id").option("--membership-epoch <epoch>", "membership epoch").option("--manifest-hash <hash>", "membership manifest hash").option("--remote-url <url>", "remote coordinator URL override").option("--admin-secret <secret>", "remote coordinator admin secret override");
|
|
2703
2704
|
addDbOption(revokeScopeMemberCmd);
|
|
2704
2705
|
addJsonOption(revokeScopeMemberCmd);
|
|
2705
2706
|
revokeScopeMemberCmd.action(async (groupId, scopeId, deviceId, opts) => {
|
|
2706
2707
|
try {
|
|
2707
2708
|
if (!await coordinatorRevokeScopeMembershipAction({
|
|
2709
|
+
effectId: opts.effectId,
|
|
2708
2710
|
groupId,
|
|
2709
2711
|
scopeId,
|
|
2710
2712
|
deviceId,
|
|
@@ -5117,8 +5119,28 @@ function createMemoryExtractionReplayCommand() {
|
|
|
5117
5119
|
});
|
|
5118
5120
|
return cmd;
|
|
5119
5121
|
}
|
|
5122
|
+
function reconcileExtractionBenchmarkStatus(input) {
|
|
5123
|
+
const quality = input.finalQuality;
|
|
5124
|
+
let status = input.classification.status;
|
|
5125
|
+
let reason = input.classification.reason;
|
|
5126
|
+
if (input.purpose === "shape_quality" && quality && status !== "observer_no_output") {
|
|
5127
|
+
if (quality.summaryDisposition.score === 0) {
|
|
5128
|
+
status = "shape_fail";
|
|
5129
|
+
reason = `summary disposition ${quality.summaryDisposition.actual} does not satisfy expected ${quality.summaryDisposition.expected}`;
|
|
5130
|
+
} else if (status === "shape_fail" && quality.summaryDisposition.actual === "skip" && input.finalFailureReasons.length > 0 && input.finalFailureReasons.every((failure) => failure.startsWith("summary count "))) {
|
|
5131
|
+
status = "pass";
|
|
5132
|
+
reason = "valid low-signal skip satisfies benchmark disposition";
|
|
5133
|
+
}
|
|
5134
|
+
}
|
|
5135
|
+
return {
|
|
5136
|
+
status,
|
|
5137
|
+
reason,
|
|
5138
|
+
quality,
|
|
5139
|
+
initialQuality: input.initialQuality
|
|
5140
|
+
};
|
|
5141
|
+
}
|
|
5120
5142
|
function createMemoryExtractionBenchmarkCommand() {
|
|
5121
|
-
const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run");
|
|
5143
|
+
const cmd = new Command("extraction-benchmark").configureHelp(helpStyle).description("Run the formal extraction replay benchmark set and print a cost/quality scoreboard").requiredOption("--benchmark <id>", "benchmark profile id").option("--observer-provider <provider>", "override observer provider for this benchmark run").option("--observer-model <model>", "override observer model for this benchmark run").option("--observer-tier-routing", "use replay-only benchmark-backed observer tier routing").option("--openai-responses", "use OpenAI Responses API for this benchmark run").option("--reasoning-effort <level>", "set OpenAI reasoning.effort for this benchmark run (responses path)").option("--reasoning-summary <mode>", "set OpenAI reasoning.summary for this benchmark run (responses path)").option("--max-output-tokens <n>", "override OpenAI max_output_tokens for this benchmark run (responses path)").option("--observer-temperature <value>", "override observer temperature for this benchmark run").option("--transcript-budget <chars>", "override replay transcript budget in characters for this benchmark run").option("--repetitions <n>", "run every benchmark batch 1-10 times to measure model stability", "1");
|
|
5122
5144
|
addDbOption(cmd);
|
|
5123
5145
|
addJsonOption(cmd);
|
|
5124
5146
|
cmd.action(async (opts) => {
|
|
@@ -5139,6 +5161,9 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5139
5161
|
const maxOutputTokensInput = opts.maxOutputTokens?.trim() ?? "";
|
|
5140
5162
|
const maxOutputTokens = maxOutputTokensInput.length > 0 ? parseStrictPositiveId(maxOutputTokensInput) : null;
|
|
5141
5163
|
if (maxOutputTokensInput.length > 0 && maxOutputTokens === null) throw new Error(`Invalid max output tokens: ${maxOutputTokensInput || opts.maxOutputTokens}`);
|
|
5164
|
+
const repetitionsInput = opts.repetitions?.trim() ?? "1";
|
|
5165
|
+
const repetitions = parseStrictPositiveId(repetitionsInput);
|
|
5166
|
+
if (repetitions === null || repetitions > 10) throw new Error(`Invalid repetitions: ${repetitionsInput || opts.repetitions}`);
|
|
5142
5167
|
const observerConfig = loadObserverConfig();
|
|
5143
5168
|
const observerConfigWithOverrides = {
|
|
5144
5169
|
...observerConfig,
|
|
@@ -5152,7 +5177,7 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5152
5177
|
};
|
|
5153
5178
|
const observer = new ObserverClient(observerConfigWithOverrides);
|
|
5154
5179
|
const runs = [];
|
|
5155
|
-
for (const batch of benchmark.batches) {
|
|
5180
|
+
for (let iteration = 1; iteration <= repetitions; iteration += 1) for (const batch of benchmark.batches) {
|
|
5156
5181
|
const scenarioId = batch.scenarioId ?? benchmark.scenarioId;
|
|
5157
5182
|
const result = opts.observerTierRouting === true ? await replayBatchExtractionWithTierRouting(resolveDbOpt(opts), observerConfigWithOverrides, {
|
|
5158
5183
|
batchId: batch.batchId,
|
|
@@ -5163,7 +5188,51 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5163
5188
|
scenarioId,
|
|
5164
5189
|
transcriptBudget: transcriptBudget ?? void 0
|
|
5165
5190
|
});
|
|
5191
|
+
const costModel = result.observer.modelFallbackApplied ? result.observer.resolvedModel : result.observer.resolvedModel ?? result.observer.model;
|
|
5192
|
+
const initialCost = costModel ? estimateExtractionModelCost(costModel, result.observer.initialUsage) : null;
|
|
5193
|
+
const repairCost = costModel ? estimateExtractionModelCost(costModel, result.observer.repairedUsage) : null;
|
|
5194
|
+
const totalCost = costModel ? estimateExtractionModelCost(costModel, result.observer.totalUsage) : null;
|
|
5195
|
+
const pricing = costModel ? getExtractionModelPricing(costModel) : null;
|
|
5196
|
+
const costUnavailableReason = totalCost ? null : result.observer.modelFallbackApplied && !result.observer.resolvedModel ? "model_fallback_unresolved" : result.observer.totalUsage == null ? "missing_usage" : "unknown_model_pricing";
|
|
5197
|
+
const initialQuality = result.observer.initialDiagnostics ? scoreExtractionBenchmarkOutput({
|
|
5198
|
+
parsed: result.observer.initialParsed,
|
|
5199
|
+
diagnostics: result.observer.initialDiagnostics,
|
|
5200
|
+
review: batch.review ?? {
|
|
5201
|
+
status: "unreviewed",
|
|
5202
|
+
reviewerNotes: "No durable-fact review has been recorded for this batch."
|
|
5203
|
+
},
|
|
5204
|
+
estimatedCostUsd: initialCost?.totalCostUsd ?? null,
|
|
5205
|
+
expectedSummaryDisposition: batch.expectedSummaryDisposition
|
|
5206
|
+
}) : null;
|
|
5207
|
+
const repairQuality = result.observer.repairedParsed && result.observer.repairedDiagnostics ? scoreExtractionBenchmarkOutput({
|
|
5208
|
+
parsed: result.observer.repairedParsed,
|
|
5209
|
+
diagnostics: result.observer.repairedDiagnostics,
|
|
5210
|
+
review: batch.review ?? {
|
|
5211
|
+
status: "unreviewed",
|
|
5212
|
+
reviewerNotes: "No durable-fact review has been recorded for this batch."
|
|
5213
|
+
},
|
|
5214
|
+
estimatedCostUsd: repairCost?.totalCostUsd ?? null,
|
|
5215
|
+
expectedSummaryDisposition: batch.expectedSummaryDisposition
|
|
5216
|
+
}) : null;
|
|
5217
|
+
const finalQuality = result.observer.diagnostics ? scoreExtractionBenchmarkOutput({
|
|
5218
|
+
parsed: result.observer.parsed,
|
|
5219
|
+
diagnostics: result.observer.diagnostics,
|
|
5220
|
+
review: batch.review ?? {
|
|
5221
|
+
status: "unreviewed",
|
|
5222
|
+
reviewerNotes: "No durable-fact review has been recorded for this batch."
|
|
5223
|
+
},
|
|
5224
|
+
estimatedCostUsd: totalCost?.totalCostUsd ?? null,
|
|
5225
|
+
expectedSummaryDisposition: batch.expectedSummaryDisposition
|
|
5226
|
+
}) : null;
|
|
5227
|
+
const reconciled = reconcileExtractionBenchmarkStatus({
|
|
5228
|
+
purpose: batch.purpose,
|
|
5229
|
+
classification: result.classification,
|
|
5230
|
+
finalFailureReasons: result.evaluation.failureReasons,
|
|
5231
|
+
initialQuality,
|
|
5232
|
+
finalQuality
|
|
5233
|
+
});
|
|
5166
5234
|
runs.push({
|
|
5235
|
+
iteration,
|
|
5167
5236
|
batchId: batch.batchId,
|
|
5168
5237
|
sessionId: batch.sessionId,
|
|
5169
5238
|
label: batch.label,
|
|
@@ -5171,17 +5240,23 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5171
5240
|
complexity: batch.complexity,
|
|
5172
5241
|
scenarioId,
|
|
5173
5242
|
expectedTier: batch.expectedTier ?? null,
|
|
5243
|
+
expectedSummaryDisposition: batch.expectedSummaryDisposition,
|
|
5174
5244
|
analysis: {
|
|
5175
5245
|
eventSpan: result.analysis.eventSpan,
|
|
5176
5246
|
promptCount: result.analysis.promptCount,
|
|
5177
5247
|
toolCount: result.analysis.toolCount,
|
|
5178
5248
|
transcriptLength: result.analysis.transcriptLength
|
|
5179
5249
|
},
|
|
5180
|
-
status:
|
|
5181
|
-
reason:
|
|
5250
|
+
status: reconciled.status,
|
|
5251
|
+
reason: reconciled.reason,
|
|
5182
5252
|
tier: result.observer.tier ?? "manual",
|
|
5183
5253
|
provider: result.observer.provider,
|
|
5184
5254
|
model: result.observer.model,
|
|
5255
|
+
transport: result.observer.transport,
|
|
5256
|
+
requestedModel: result.observer.requestedModel,
|
|
5257
|
+
resolvedModel: result.observer.resolvedModel,
|
|
5258
|
+
modelFallbackApplied: result.observer.modelFallbackApplied,
|
|
5259
|
+
modelFallbackReason: result.observer.modelFallbackReason,
|
|
5185
5260
|
openaiUseResponses: result.observer.openaiUseResponses,
|
|
5186
5261
|
reasoningEffort: result.observer.reasoningEffort,
|
|
5187
5262
|
reasoningSummary: result.observer.reasoningSummary,
|
|
@@ -5189,22 +5264,89 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5189
5264
|
temperature: result.observer.temperature,
|
|
5190
5265
|
summaries: result.evaluation.counts.summaries,
|
|
5191
5266
|
observations: result.evaluation.counts.observations,
|
|
5192
|
-
repairApplied: result.observer.repairApplied
|
|
5267
|
+
repairApplied: result.observer.repairApplied,
|
|
5268
|
+
initial: {
|
|
5269
|
+
raw: result.observer.initialRaw,
|
|
5270
|
+
status: result.initialClassification.status,
|
|
5271
|
+
reason: result.initialClassification.reason,
|
|
5272
|
+
pass: result.initialEvaluation.pass,
|
|
5273
|
+
failureReasons: result.initialEvaluation.failureReasons,
|
|
5274
|
+
summaries: result.initialEvaluation.counts.summaries,
|
|
5275
|
+
observations: result.initialEvaluation.counts.observations,
|
|
5276
|
+
diagnostics: result.observer.initialDiagnostics,
|
|
5277
|
+
elapsedMs: result.observer.initialElapsedMs,
|
|
5278
|
+
usage: result.observer.initialUsage,
|
|
5279
|
+
quality: reconciled.initialQuality
|
|
5280
|
+
},
|
|
5281
|
+
repair: {
|
|
5282
|
+
applied: result.observer.repairApplied,
|
|
5283
|
+
raw: result.observer.repairedRaw,
|
|
5284
|
+
status: result.repairedClassification?.status ?? null,
|
|
5285
|
+
reason: result.repairedClassification?.reason ?? null,
|
|
5286
|
+
pass: result.repairedEvaluation?.pass ?? null,
|
|
5287
|
+
failureReasons: result.repairedEvaluation?.failureReasons ?? [],
|
|
5288
|
+
summaries: result.repairedEvaluation?.counts.summaries ?? null,
|
|
5289
|
+
observations: result.repairedEvaluation?.counts.observations ?? null,
|
|
5290
|
+
diagnostics: result.observer.repairedDiagnostics,
|
|
5291
|
+
elapsedMs: result.observer.repairedElapsedMs,
|
|
5292
|
+
usage: result.observer.repairedUsage,
|
|
5293
|
+
quality: repairQuality
|
|
5294
|
+
},
|
|
5295
|
+
telemetry: {
|
|
5296
|
+
totalElapsedMs: result.observer.totalElapsedMs,
|
|
5297
|
+
totalUsage: result.observer.totalUsage
|
|
5298
|
+
},
|
|
5299
|
+
pricing,
|
|
5300
|
+
cost: {
|
|
5301
|
+
initial: initialCost,
|
|
5302
|
+
repair: repairCost,
|
|
5303
|
+
total: totalCost,
|
|
5304
|
+
unavailableReason: costUnavailableReason
|
|
5305
|
+
},
|
|
5306
|
+
quality: reconciled.quality
|
|
5193
5307
|
});
|
|
5194
5308
|
}
|
|
5309
|
+
const reviewedQualityRuns = runs.filter((run) => run.quality?.weightedQualityScore != null);
|
|
5310
|
+
const knownCostRuns = runs.filter((run) => run.cost.total != null);
|
|
5311
|
+
const knownElapsedRuns = runs.filter((run) => run.telemetry.totalElapsedMs != null);
|
|
5195
5312
|
const summary = {
|
|
5313
|
+
repetitions,
|
|
5196
5314
|
total: runs.length,
|
|
5197
5315
|
shapeQualityTotal: runs.filter((run) => run.purpose === "shape_quality").length,
|
|
5198
5316
|
shapeQualityPasses: runs.filter((run) => run.purpose === "shape_quality" && run.status === "pass").length,
|
|
5199
5317
|
shapeQualityFails: runs.filter((run) => run.purpose === "shape_quality" && run.status === "shape_fail").length,
|
|
5200
5318
|
expectedTierTotal: runs.filter((run) => run.expectedTier != null).length,
|
|
5201
5319
|
expectedTierMatches: runs.filter((run) => run.expectedTier != null && run.expectedTier === run.tier).length,
|
|
5202
|
-
robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length
|
|
5320
|
+
robustnessNoOutput: runs.filter((run) => run.status === "observer_no_output").length,
|
|
5321
|
+
summaryDispositionTotal: runs.filter((run) => run.quality != null).length,
|
|
5322
|
+
summaryDispositionMatches: runs.filter((run) => run.quality?.summaryDisposition.score === 1).length,
|
|
5323
|
+
reviewedQualityRuns: reviewedQualityRuns.length,
|
|
5324
|
+
knownCostRuns: knownCostRuns.length,
|
|
5325
|
+
unknownCostRuns: runs.length - knownCostRuns.length,
|
|
5326
|
+
missingUsageRuns: runs.filter((run) => run.cost.unavailableReason === "missing_usage").length,
|
|
5327
|
+
unknownPricingRuns: runs.filter((run) => run.cost.unavailableReason === "unknown_model_pricing").length,
|
|
5328
|
+
fallbackUnresolvedRuns: runs.filter((run) => run.cost.unavailableReason === "model_fallback_unresolved").length,
|
|
5329
|
+
totalKnownCostUsd: knownCostRuns.reduce((sum, run) => sum + (run.cost.total?.totalCostUsd ?? 0), 0),
|
|
5330
|
+
knownElapsedRuns: knownElapsedRuns.length,
|
|
5331
|
+
totalKnownElapsedMs: knownElapsedRuns.reduce((sum, run) => sum + (run.telemetry.totalElapsedMs ?? 0), 0),
|
|
5332
|
+
perBatchStability: benchmark.batches.map((batch) => {
|
|
5333
|
+
const batchRuns = runs.filter((run) => run.batchId === batch.batchId);
|
|
5334
|
+
const passes = batchRuns.filter((run) => run.status === "pass").length;
|
|
5335
|
+
return {
|
|
5336
|
+
batchId: batch.batchId,
|
|
5337
|
+
purpose: batch.purpose,
|
|
5338
|
+
passes,
|
|
5339
|
+
total: batchRuns.length,
|
|
5340
|
+
passRate: batchRuns.length > 0 ? passes / batchRuns.length : null,
|
|
5341
|
+
statuses: batchRuns.map((run) => run.status)
|
|
5342
|
+
};
|
|
5343
|
+
})
|
|
5203
5344
|
};
|
|
5204
|
-
const uniqueObserverKeys = Array.from(new Set(runs.map((run) => `${run.provider}::${run.model}::${run.
|
|
5345
|
+
const uniqueObserverKeys = Array.from(new Set(runs.map((run) => `${run.provider}::${run.model}::${run.transport}`)));
|
|
5205
5346
|
const observerSummary = opts.observerTierRouting === true ? {
|
|
5206
5347
|
provider: uniqueObserverKeys.length === 1 ? runs[0]?.provider ?? observer.provider : "mixed",
|
|
5207
5348
|
model: uniqueObserverKeys.length === 1 ? runs[0]?.model ?? observer.model : "mixed",
|
|
5349
|
+
transport: uniqueObserverKeys.length === 1 ? runs[0]?.transport ?? "unknown" : "mixed",
|
|
5208
5350
|
tierRouting: true,
|
|
5209
5351
|
openaiUseResponses: uniqueObserverKeys.length === 1 ? runs[0]?.openaiUseResponses ?? observer.openaiUseResponses : null,
|
|
5210
5352
|
reasoningEffort: uniqueObserverKeys.length === 1 ? runs[0]?.reasoningEffort ?? observer.reasoningEffort : "mixed",
|
|
@@ -5216,6 +5358,7 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5216
5358
|
} : {
|
|
5217
5359
|
provider: observer.provider,
|
|
5218
5360
|
model: observer.model,
|
|
5361
|
+
transport: runs[0]?.transport ?? observer.getStatus().runtime,
|
|
5219
5362
|
tierRouting: false,
|
|
5220
5363
|
openaiUseResponses: observer.openaiUseResponses,
|
|
5221
5364
|
reasoningEffort: observer.reasoningEffort,
|
|
@@ -5229,7 +5372,8 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5229
5372
|
benchmark: {
|
|
5230
5373
|
id: benchmark.id,
|
|
5231
5374
|
title: benchmark.title,
|
|
5232
|
-
scenarioId: benchmark.scenarioId
|
|
5375
|
+
scenarioId: benchmark.scenarioId,
|
|
5376
|
+
modelCandidates: benchmark.modelCandidates
|
|
5233
5377
|
},
|
|
5234
5378
|
observer: observerSummary,
|
|
5235
5379
|
summary,
|
|
@@ -5243,6 +5387,7 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5243
5387
|
p.log.info([
|
|
5244
5388
|
`Benchmark: ${benchmark.id} — ${benchmark.title}`,
|
|
5245
5389
|
`Observer: ${observerSummary.provider}/${observerSummary.model}`,
|
|
5390
|
+
`Transport: ${observerSummary.transport}`,
|
|
5246
5391
|
`Tier routing: ${opts.observerTierRouting === true ? "yes" : "no"}`,
|
|
5247
5392
|
`OpenAI Responses: ${observerSummary.openaiUseResponses === null ? "mixed" : observerSummary.openaiUseResponses ? "yes" : "no"}`,
|
|
5248
5393
|
`Reasoning effort: ${observerSummary.reasoningEffort ?? "none"}`,
|
|
@@ -5250,12 +5395,23 @@ function createMemoryExtractionBenchmarkCommand() {
|
|
|
5250
5395
|
`Max output tokens: ${observerSummary.maxOutputTokens ?? "mixed"}`,
|
|
5251
5396
|
`Temperature: ${observerSummary.temperature ?? "mixed"}`,
|
|
5252
5397
|
`Transcript budget override: ${transcriptBudget ?? "default"}`,
|
|
5398
|
+
`Repetitions: ${summary.repetitions}`,
|
|
5253
5399
|
`Shape-quality passes: ${summary.shapeQualityPasses}/${summary.shapeQualityTotal}`,
|
|
5254
5400
|
`Shape-quality fails: ${summary.shapeQualityFails}`,
|
|
5255
5401
|
`Expected-tier matches: ${summary.expectedTierMatches}/${summary.expectedTierTotal}`,
|
|
5256
|
-
`Observer no-output cases: ${summary.robustnessNoOutput}
|
|
5402
|
+
`Observer no-output cases: ${summary.robustnessNoOutput}`,
|
|
5403
|
+
`Summary disposition matches: ${summary.summaryDispositionMatches}/${summary.summaryDispositionTotal}`,
|
|
5404
|
+
`Reviewed quality runs: ${summary.reviewedQualityRuns} (compare per-run dimensions; scores are fixture-specific)`,
|
|
5405
|
+
`Known estimated cost: $${summary.totalKnownCostUsd.toFixed(6)} (${summary.knownCostRuns}/${summary.total}; missing usage=${summary.missingUsageRuns}, unknown pricing=${summary.unknownPricingRuns}, unresolved fallback=${summary.fallbackUnresolvedRuns})`,
|
|
5406
|
+
`Known elapsed time: ${summary.totalKnownElapsedMs}ms (${summary.knownElapsedRuns}/${summary.total} run(s))`
|
|
5257
5407
|
].join("\n"));
|
|
5258
|
-
for (const run of runs)
|
|
5408
|
+
for (const run of runs) {
|
|
5409
|
+
const qualityLabel = run.quality?.weightedQualityScore == null ? "n/a" : run.quality.weightedQualityScore.toFixed(3);
|
|
5410
|
+
const costLabel = run.cost.total == null ? "n/a" : `$${run.cost.total.totalCostUsd.toFixed(6)}`;
|
|
5411
|
+
const latencyLabel = run.telemetry.totalElapsedMs == null ? "n/a" : `${run.telemetry.totalElapsedMs}ms`;
|
|
5412
|
+
const missingRequired = run.quality?.requiredRecall.missingLabelIds.join(",") || "none";
|
|
5413
|
+
p.log.message(` [${run.batchId}#${run.iteration}] ${run.status.padEnd(18)} ${run.complexity.padEnd(10)} tier=${run.tier.padEnd(6)} expected=${(run.expectedTier ?? "n/a").padEnd(6)} disposition=${run.quality?.summaryDisposition.actual ?? "n/a"}/${run.expectedSummaryDisposition} span=${String(run.analysis.eventSpan).padEnd(3)} prompts=${run.analysis.promptCount} tools=${String(run.analysis.toolCount).padEnd(2)} transcript=${run.analysis.transcriptLength} ${run.provider}/${run.model} [${run.transport}] initial=${run.initial.summaries}s/${run.initial.observations}o final=${run.summaries}s/${run.observations}o quality=${qualityLabel} coverage=${run.quality?.weightedQualityCoverage?.toFixed(3) ?? "n/a"} required_missing=${missingRequired} cost=${costLabel} latency=${latencyLabel} schema_loss=${run.initial.diagnostics?.dataLoss === true ? "yes" : "no"} fallback=${run.modelFallbackApplied ? "yes" : "no"} repair=${run.repairApplied ? "yes" : "no"} — ${run.label}`);
|
|
5414
|
+
}
|
|
5259
5415
|
p.outro("done");
|
|
5260
5416
|
} catch (error) {
|
|
5261
5417
|
const message = error instanceof Error ? error.message : "Extraction benchmark failed";
|
|
@@ -5976,6 +6132,14 @@ function sqliteVecFailureDiagnostics(error, dbPath) {
|
|
|
5976
6132
|
`error=${message}`
|
|
5977
6133
|
];
|
|
5978
6134
|
}
|
|
6135
|
+
async function runServeCoordinatorMaintenance(store, dependencies) {
|
|
6136
|
+
const projectShares = await dependencies.advancePendingProjectShares(store, { limit: 3 });
|
|
6137
|
+
if (projectShares.failed > 0) throw new Error(`share operation maintenance failed for ${projectShares.failed} of ${projectShares.processed} operations`);
|
|
6138
|
+
return {
|
|
6139
|
+
projectShares,
|
|
6140
|
+
recipientPolicies: await dependencies.reconcileRecipientPolicyProjects(store, { limit: 3 })
|
|
6141
|
+
};
|
|
6142
|
+
}
|
|
5979
6143
|
async function startBackgroundViewer(invocation) {
|
|
5980
6144
|
warnIfViewerExposed(invocation.host, invocation.port);
|
|
5981
6145
|
if (await isPortOpen(invocation.host, invocation.port)) {
|
|
@@ -6008,7 +6172,7 @@ async function startBackgroundViewer(invocation) {
|
|
|
6008
6172
|
p.outro(`Viewer started in background (pid ${child.pid}) at http://${invocation.host}:${invocation.port}`);
|
|
6009
6173
|
}
|
|
6010
6174
|
async function startForegroundViewer(invocation) {
|
|
6011
|
-
const { createApp, createSyncApp, closeStore, getStore } = await import("@codemem/server");
|
|
6175
|
+
const { advancePendingProjectShares, createApp, createSyncApp, closeStore, getStore, reconcileRecipientPolicyProjects } = await import("@codemem/server");
|
|
6012
6176
|
const { serve } = await import("@hono/node-server");
|
|
6013
6177
|
if (invocation.dbPath) process.env.CODEMEM_DB = invocation.dbPath;
|
|
6014
6178
|
if (invocation.configPath) process.env.CODEMEM_CONFIG = invocation.configPath;
|
|
@@ -6099,6 +6263,12 @@ async function startForegroundViewer(invocation) {
|
|
|
6099
6263
|
port: syncConfig.syncPort,
|
|
6100
6264
|
signal: syncAbort.signal,
|
|
6101
6265
|
scanner: store.scanner,
|
|
6266
|
+
onAfterCoordinatorRefresh: async () => {
|
|
6267
|
+
await runServeCoordinatorMaintenance(store, {
|
|
6268
|
+
advancePendingProjectShares,
|
|
6269
|
+
reconcileRecipientPolicyProjects
|
|
6270
|
+
});
|
|
6271
|
+
},
|
|
6102
6272
|
onPhaseChange: (phase) => {
|
|
6103
6273
|
if (phase === "running") {
|
|
6104
6274
|
syncRuntimeStatus.phase = null;
|