@dsh-enhanced/assistant-evaluation 0.1.7 → 0.1.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/store.js CHANGED
@@ -10,6 +10,18 @@ export class EvaluationStoreError extends Error {
10
10
  this.name = 'EvaluationStoreError';
11
11
  }
12
12
  }
13
+ function projectionState(row) {
14
+ return Object.freeze({
15
+ evaluationId: row.evaluation_id,
16
+ scope: Object.freeze({ workspace: row.workspace, preset: row.preset }),
17
+ status: row.status,
18
+ attemptCount: row.attempt_count,
19
+ nextAttemptAt: row.next_attempt_at,
20
+ ...(row.last_failure_code === null ? {} : { lastFailureCode: row.last_failure_code }),
21
+ createdAt: row.created_at,
22
+ updatedAt: row.updated_at,
23
+ });
24
+ }
13
25
  const standardIntegerMetrics = new Set([
14
26
  'costUsdMicros', 'latencyMs', 'inputTokens', 'outputTokens', 'toolCalls', 'retries',
15
27
  ]);
@@ -141,6 +153,54 @@ function stored(row) {
141
153
  evaluator: Object.freeze({ id: row.evaluator_id, version: row.evaluator_version }),
142
154
  });
143
155
  }
156
+ function trustedEvidence(row) {
157
+ return Object.freeze(JSON.parse(row.evidence_json)
158
+ .map(entry => Object.freeze({ ...entry })));
159
+ }
160
+ function executionComponent(row) {
161
+ if (row === undefined)
162
+ return undefined;
163
+ return Object.freeze({
164
+ outcomeId: row.id,
165
+ status: row.execution_status,
166
+ source: Object.freeze({ kind: row.source_kind, id: row.source_id }),
167
+ evidence: trustedEvidence(row),
168
+ occurredAt: row.occurred_at,
169
+ evaluator: Object.freeze({ id: row.evaluator_id, version: row.evaluator_version }),
170
+ });
171
+ }
172
+ function objectiveComponent(row) {
173
+ if (row === undefined)
174
+ return undefined;
175
+ return Object.freeze({
176
+ outcomeId: row.id,
177
+ status: row.objective_status,
178
+ source: Object.freeze({ kind: row.source_kind, id: row.source_id }),
179
+ evidence: trustedEvidence(row),
180
+ occurredAt: row.occurred_at,
181
+ evaluator: Object.freeze({ id: row.evaluator_id, version: row.evaluator_version }),
182
+ });
183
+ }
184
+ function projected(row) {
185
+ return Object.freeze({
186
+ ...stored(row),
187
+ projection: Object.freeze({
188
+ subjectKind: row.task_subject_kind,
189
+ subjectRef: row.task_subject_ref,
190
+ status: row.task_objective_conflicted === 1 ? 'objective-conflict' : 'ready',
191
+ primaryOutcomeId: row.task_primary_outcome_id,
192
+ ...(row.task_execution_outcome_id === null
193
+ ? {} : { executionOutcomeId: row.task_execution_outcome_id }),
194
+ ...(row.task_objective_outcome_id === null
195
+ ? {} : { objectiveOutcomeId: row.task_objective_outcome_id }),
196
+ ...(row.task_delivery_outcome_id === null
197
+ ? {} : { deliveryOutcomeId: row.task_delivery_outcome_id }),
198
+ learningVersion: row.task_learning_version,
199
+ learningDigest: row.task_learning_digest,
200
+ learningDisposition: row.task_learning_disposition,
201
+ }),
202
+ });
203
+ }
144
204
  function selfAssessment(row) {
145
205
  return Object.freeze({
146
206
  id: row.id,
@@ -162,6 +222,90 @@ function selfAssessment(row) {
162
222
  function digest(value) {
163
223
  return createHash('sha256').update(JSON.stringify(value)).digest('hex');
164
224
  }
225
+ /** Stable cross-package digest for one canonical task learning revision. */
226
+ export function evaluationLearningProjectionDigest(input) {
227
+ return digest([
228
+ 'evaluation-task-learning/v1',
229
+ input.scopeKey,
230
+ input.projection.subjectKind,
231
+ input.projection.subjectRef,
232
+ input.projection.disposition,
233
+ input.situation,
234
+ input.execution === undefined ? null : [
235
+ input.execution.outcomeId,
236
+ input.execution.status,
237
+ input.execution.source.kind,
238
+ input.execution.source.id,
239
+ input.execution.evidence.map(entry => [entry.kind, entry.ref, entry.digest ?? null]),
240
+ input.execution.occurredAt,
241
+ input.execution.evaluator.id,
242
+ input.execution.evaluator.version,
243
+ ],
244
+ input.objective === undefined ? null : [
245
+ input.objective.outcomeId,
246
+ input.objective.status,
247
+ input.objective.source.kind,
248
+ input.objective.source.id,
249
+ input.objective.evidence.map(entry => [entry.kind, entry.ref, entry.digest ?? null]),
250
+ input.objective.occurredAt,
251
+ input.objective.evaluator.id,
252
+ input.objective.evaluator.version,
253
+ ],
254
+ input.projection.evidenceOutcomeId ?? null,
255
+ ]);
256
+ }
257
+ function taskSubject(scopeKey, outcomeId, references) {
258
+ const automationRunRefs = new Set(references
259
+ .filter(reference => reference.kind === 'automation-run')
260
+ .map(reference => reference.ref));
261
+ if (automationRunRefs.size === 1) {
262
+ const ref = [...automationRunRefs][0];
263
+ return Object.freeze({
264
+ key: JSON.stringify([scopeKey, 'automation-run', ref]),
265
+ kind: 'automation-run',
266
+ ref,
267
+ });
268
+ }
269
+ return Object.freeze({
270
+ key: JSON.stringify([scopeKey, 'outcome', outcomeId]),
271
+ kind: 'outcome',
272
+ ref: outcomeId,
273
+ });
274
+ }
275
+ function newer(left, right) {
276
+ if (left.recorded_at !== right.recorded_at)
277
+ return left.recorded_at - right.recorded_at;
278
+ return left.id === right.id ? 0 : left.id > right.id ? 1 : -1;
279
+ }
280
+ function latest(rows) {
281
+ return rows.reduce((winner, row) => (winner === undefined || newer(row, winner) > 0 ? row : winner), undefined);
282
+ }
283
+ function containsEvidence(row, kind) {
284
+ try {
285
+ return JSON.parse(row.evidence_json).some(entry => (typeof entry === 'object' && entry !== null && !Array.isArray(entry)
286
+ && entry.kind === kind));
287
+ }
288
+ catch {
289
+ return false;
290
+ }
291
+ }
292
+ function isAuthoritativeAutomationTerminal(row) {
293
+ return row.trust === 'trusted'
294
+ && row.source_kind === 'automation'
295
+ && row.source_id === 'assistant-automations'
296
+ && row.evaluator_id === 'assistant-automations'
297
+ && /^(?:terminal|host-runbook)-v[1-9][0-9]*$/u.test(row.evaluator_version)
298
+ && containsEvidence(row, 'automation-run');
299
+ }
300
+ function isAuthenticatedOwnerFeedback(row) {
301
+ return row.trust === 'trusted'
302
+ && row.source_kind === 'user-feedback'
303
+ && row.source_id === 'assistant-delivery/typed-owner-feedback'
304
+ && row.evaluator_id === 'assistant-delivery-owner-feedback'
305
+ && row.evaluator_version === '2'
306
+ && containsEvidence(row, 'automation-run')
307
+ && containsEvidence(row, 'delivery-outbox');
308
+ }
165
309
  export class EvaluationStore {
166
310
  #database;
167
311
  #now;
@@ -195,6 +339,13 @@ export class EvaluationStore {
195
339
  this.#database.close();
196
340
  throw new EvaluationStoreError('invalid-input', 'default summary window exceeds the maximum window');
197
341
  }
342
+ try {
343
+ this.#rebuildTaskProjections();
344
+ }
345
+ catch (error) {
346
+ this.#database.close();
347
+ throw error;
348
+ }
198
349
  }
199
350
  close() { this.#database.close(); }
200
351
  getOutcome(scopeInput, outcomeIdInput) {
@@ -209,23 +360,125 @@ export class EvaluationStore {
209
360
  const normalized = this.#normalize(input);
210
361
  const payloadHash = digest(normalized);
211
362
  const id = `outcome-${randomUUID()}`;
363
+ const subject = taskSubject(normalized.scopeKey, id, normalized.evidence);
212
364
  const recordedAt = timestamp(this.#now(), 'recordedAt');
213
365
  const metric = normalized.metrics;
214
- this.#database.prepare(`
215
- INSERT INTO evaluation_outcomes(
216
- id, idempotency_key, payload_hash, scope_key, workspace, preset, situation,
217
- execution_status, objective_status, delivery_status, source_kind, source_id,
218
- trust, evidence_json, metrics_json, cost_usd_micros, latency_ms, input_tokens,
219
- output_tokens, tool_calls, occurred_at, recorded_at, evaluator_id, evaluator_version)
220
- VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
221
- ON CONFLICT(idempotency_key) DO NOTHING
222
- `).run(id, normalized.idempotencyKey, payloadHash, normalized.scopeKey, normalized.scope.workspace, normalized.scope.preset, normalized.situation, normalized.executionStatus, normalized.objectiveStatus, normalized.deliveryStatus, normalized.source.kind, normalized.source.id, normalized.trust, JSON.stringify(normalized.evidence), JSON.stringify(normalized.metrics), metric.costUsdMicros ?? null, metric.latencyMs ?? null, metric.inputTokens ?? null, metric.outputTokens ?? null, metric.toolCalls ?? null, normalized.occurredAt, recordedAt, normalized.evaluator.id, normalized.evaluator.version);
223
- const winner = this.#database.prepare('SELECT * FROM evaluation_outcomes WHERE idempotency_key = ?')
224
- .get(normalized.idempotencyKey);
225
- if (winner.payload_hash !== payloadHash) {
226
- throw new EvaluationStoreError('idempotency-conflict', 'evaluation outcome idempotency key was reused with different content');
366
+ this.#database.exec('BEGIN IMMEDIATE');
367
+ try {
368
+ this.#database.prepare(`
369
+ INSERT INTO evaluation_outcomes(
370
+ id, idempotency_key, payload_hash, scope_key, workspace, preset, situation,
371
+ execution_status, objective_status, delivery_status, source_kind, source_id,
372
+ trust, evidence_json, metrics_json, cost_usd_micros, latency_ms, input_tokens,
373
+ output_tokens, tool_calls, occurred_at, recorded_at, evaluator_id, evaluator_version,
374
+ task_subject_key, task_subject_kind, task_subject_ref)
375
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
376
+ ON CONFLICT(idempotency_key) DO NOTHING
377
+ `).run(id, normalized.idempotencyKey, payloadHash, normalized.scopeKey, normalized.scope.workspace, normalized.scope.preset, normalized.situation, normalized.executionStatus, normalized.objectiveStatus, normalized.deliveryStatus, normalized.source.kind, normalized.source.id, normalized.trust, JSON.stringify(normalized.evidence), JSON.stringify(normalized.metrics), metric.costUsdMicros ?? null, metric.latencyMs ?? null, metric.inputTokens ?? null, metric.outputTokens ?? null, metric.toolCalls ?? null, normalized.occurredAt, recordedAt, normalized.evaluator.id, normalized.evaluator.version, subject.key, subject.kind, subject.ref);
378
+ const winner = this.#database.prepare('SELECT * FROM evaluation_outcomes WHERE idempotency_key = ?')
379
+ .get(normalized.idempotencyKey);
380
+ if (winner.payload_hash !== payloadHash) {
381
+ throw new EvaluationStoreError('idempotency-conflict', 'evaluation outcome idempotency key was reused with different content');
382
+ }
383
+ const winnerSubject = winner.task_subject_key === null
384
+ ? taskSubject(winner.scope_key, winner.id, JSON.parse(winner.evidence_json))
385
+ : {
386
+ key: winner.task_subject_key,
387
+ kind: winner.task_subject_kind,
388
+ ref: winner.task_subject_ref,
389
+ };
390
+ this.#database.prepare(`
391
+ INSERT INTO evaluation_task_projections(
392
+ subject_key, scope_key, subject_kind, subject_ref, updated_at)
393
+ VALUES (?, ?, ?, ?, ?)
394
+ ON CONFLICT(subject_key) DO NOTHING
395
+ `).run(winnerSubject.key, winner.scope_key, winnerSubject.kind, winnerSubject.ref, winner.recorded_at);
396
+ const refreshed = this.#refreshTaskProjection(winnerSubject.key);
397
+ if (winner.trust === 'trusted' && refreshed.learningVersionChanged) {
398
+ this.#database.prepare(`
399
+ INSERT INTO evaluation_projection_outbox(
400
+ evaluation_id, status, attempt_count, next_attempt_at,
401
+ last_failure_at, last_failure_code, created_at, updated_at)
402
+ VALUES (?, 'pending', 0, ?, NULL, NULL, ?, ?)
403
+ ON CONFLICT(evaluation_id) DO NOTHING
404
+ `).run(winner.id, winner.recorded_at, winner.recorded_at, winner.recorded_at);
405
+ this.#advanceScopeWatermark(winner.scope_key, winner.recorded_at);
406
+ }
407
+ this.#database.exec('COMMIT');
408
+ return stored(winner);
409
+ }
410
+ catch (error) {
411
+ this.#database.exec('ROLLBACK');
412
+ throw error;
227
413
  }
228
- return stored(winner);
414
+ }
415
+ listPendingProjections(limitInput = 100, nowInput = this.#now()) {
416
+ const limit = timestamp(limitInput, 'projection limit');
417
+ const now = timestamp(nowInput, 'projection now');
418
+ if (limit < 1 || limit > 1_000) {
419
+ throw new EvaluationStoreError('invalid-input', 'projection limit must be between 1 and 1000');
420
+ }
421
+ const rows = this.#database.prepare(`
422
+ SELECT projection.*, outcome.workspace, outcome.preset
423
+ FROM evaluation_projection_outbox projection
424
+ JOIN evaluation_outcomes outcome ON outcome.id = projection.evaluation_id
425
+ WHERE projection.status = 'pending' AND projection.next_attempt_at <= ?
426
+ ORDER BY projection.next_attempt_at, projection.created_at, projection.evaluation_id
427
+ LIMIT ?
428
+ `).all(now, limit);
429
+ return rows.map(row => projectionState(row));
430
+ }
431
+ peekPendingProjection(scopeInput, nowInput = this.#now()) {
432
+ const { scopeKey } = canonicalEvaluationScope(scopeInput);
433
+ const now = timestamp(nowInput, 'projection now');
434
+ const row = this.#database.prepare(`
435
+ SELECT projection.*, outcome.workspace, outcome.preset
436
+ FROM evaluation_projection_outbox projection
437
+ JOIN evaluation_outcomes outcome ON outcome.id = projection.evaluation_id
438
+ WHERE projection.status = 'pending' AND projection.next_attempt_at <= ?
439
+ AND outcome.scope_key = ? AND outcome.trust = 'trusted'
440
+ ORDER BY projection.next_attempt_at, projection.created_at, projection.evaluation_id
441
+ LIMIT 1
442
+ `).get(now, scopeKey);
443
+ return row === undefined ? undefined : projectionState(row);
444
+ }
445
+ getProjection(scopeInput, evaluationIdInput) {
446
+ const { scopeKey } = canonicalEvaluationScope(scopeInput);
447
+ const evaluationId = boundedText(evaluationIdInput, 'evaluationId', 200);
448
+ const row = this.#database.prepare(`
449
+ SELECT projection.*, outcome.workspace, outcome.preset
450
+ FROM evaluation_projection_outbox projection
451
+ JOIN evaluation_outcomes outcome ON outcome.id = projection.evaluation_id
452
+ WHERE projection.evaluation_id = ? AND outcome.scope_key = ?
453
+ AND outcome.trust = 'trusted'
454
+ `).get(evaluationId, scopeKey);
455
+ return row === undefined ? undefined : projectionState(row);
456
+ }
457
+ completeProjection(input) {
458
+ const evaluationId = boundedText(input.evaluationId, 'evaluationId', 200);
459
+ const now = timestamp(input.now, 'projection completion time');
460
+ return this.#database.prepare(`
461
+ UPDATE evaluation_projection_outbox
462
+ SET status = 'recorded', updated_at = ?, last_failure_code = NULL
463
+ WHERE evaluation_id = ? AND status = 'pending'
464
+ `).run(now, evaluationId).changes === 1;
465
+ }
466
+ deferProjection(input) {
467
+ const evaluationId = boundedText(input.evaluationId, 'evaluationId', 200);
468
+ const now = timestamp(input.now, 'projection failure time');
469
+ const retryAt = timestamp(input.retryAt, 'projection retry time');
470
+ if (retryAt <= now)
471
+ throw new EvaluationStoreError('invalid-input', 'projection retry must be in the future');
472
+ const failureCode = boundedText(input.failureCode, 'projection failureCode', 64);
473
+ if (!/^[A-Za-z0-9][A-Za-z0-9._:-]{0,63}$/u.test(failureCode)) {
474
+ throw new EvaluationStoreError('invalid-input', 'projection failureCode is invalid');
475
+ }
476
+ return this.#database.prepare(`
477
+ UPDATE evaluation_projection_outbox
478
+ SET attempt_count = attempt_count + 1, next_attempt_at = ?,
479
+ last_failure_at = ?, last_failure_code = ?, updated_at = ?
480
+ WHERE evaluation_id = ? AND status = 'pending'
481
+ `).run(retryAt, now, failureCode, now, evaluationId).changes === 1;
229
482
  }
230
483
  /**
231
484
  * Append a self-reported objective judgement linked to an immutable Host
@@ -307,6 +560,181 @@ export class EvaluationStore {
307
560
  return assessment === undefined ? [] : [assessment];
308
561
  });
309
562
  }
563
+ /** Resolve the task projection containing one immutable audit outcome. */
564
+ getTaskProjection(scopeInput, outcomeIdInput) {
565
+ const { scopeKey } = canonicalEvaluationScope(scopeInput);
566
+ const outcomeId = boundedText(outcomeIdInput, 'outcomeId', 200);
567
+ const row = this.#database.prepare(`
568
+ SELECT task.*
569
+ FROM evaluation_outcomes audit
570
+ JOIN evaluation_task_projection_view task
571
+ ON task.task_subject_key = audit.task_subject_key
572
+ WHERE audit.id = ? AND audit.scope_key = ?
573
+ `).get(outcomeId, scopeKey);
574
+ return row === undefined ? undefined : projected(row);
575
+ }
576
+ /**
577
+ * Resolve an append-only outbox trigger to the latest canonical state of its
578
+ * task. The trigger may be arbitrarily old; version/digest always describe
579
+ * the current task projection.
580
+ */
581
+ getTaskLearningProjection(scopeInput, outcomeIdInput) {
582
+ const { scopeKey } = canonicalEvaluationScope(scopeInput);
583
+ const outcomeId = boundedText(outcomeIdInput, 'outcomeId', 200);
584
+ const row = this.#database.prepare(`
585
+ SELECT task.*
586
+ FROM evaluation_projection_outbox outbox
587
+ JOIN evaluation_outcomes audit ON audit.id = outbox.evaluation_id
588
+ JOIN evaluation_task_projection_view task
589
+ ON task.task_subject_key = audit.task_subject_key
590
+ WHERE audit.id = ? AND audit.scope_key = ? AND audit.trust = 'trusted'
591
+ `).get(outcomeId, scopeKey);
592
+ if (row === undefined || row.trust !== 'trusted')
593
+ return undefined;
594
+ const watermarkRow = this.#database.prepare(`
595
+ SELECT watermark FROM evaluation_scope_watermarks WHERE scope_key = ?
596
+ `).get(scopeKey);
597
+ if (watermarkRow === undefined || !Number.isSafeInteger(watermarkRow.watermark)
598
+ || watermarkRow.watermark < 1) {
599
+ throw new EvaluationStoreError('invalid-input', 'canonical scope watermark is unavailable');
600
+ }
601
+ const task = projected(row);
602
+ const selected = (id) => id === undefined
603
+ ? undefined
604
+ : this.#database.prepare('SELECT * FROM evaluation_outcomes WHERE id = ? AND scope_key = ?')
605
+ .get(id, scopeKey);
606
+ const execution = executionComponent(selected(task.projection.executionOutcomeId));
607
+ const objective = task.projection.status === 'objective-conflict'
608
+ ? undefined
609
+ : objectiveComponent(selected(task.projection.objectiveOutcomeId));
610
+ const projection = Object.freeze({
611
+ subjectKind: task.projection.subjectKind,
612
+ subjectRef: task.projection.subjectRef,
613
+ version: task.projection.learningVersion,
614
+ digest: task.projection.learningDigest,
615
+ disposition: task.projection.learningDisposition,
616
+ ...(task.projection.objectiveOutcomeId === undefined
617
+ ? {} : { evidenceOutcomeId: task.projection.objectiveOutcomeId }),
618
+ });
619
+ const receipt = Object.freeze({
620
+ triggerOutcomeId: outcomeId,
621
+ scope: Object.freeze({ ...task.scope }),
622
+ scopeKey: task.scopeKey,
623
+ scopeWatermark: watermarkRow.watermark,
624
+ situation: task.situation,
625
+ ...(execution === undefined ? {} : { execution }),
626
+ ...(objective === undefined ? {} : { objective }),
627
+ projection,
628
+ });
629
+ if (projection.version < 1 || !/^[a-f\d]{64}$/u.test(projection.digest)
630
+ || evaluationLearningProjectionDigest(receipt) !== projection.digest) {
631
+ throw new EvaluationStoreError('invalid-input', 'canonical task learning projection is corrupt');
632
+ }
633
+ return receipt;
634
+ }
635
+ /**
636
+ * Hold Evaluation's scope writer fence while a synchronous downstream
637
+ * callback acquires and commits its own writer transaction. The fixed lock
638
+ * order is Evaluation first, downstream second; a Promise-returning callback
639
+ * is rejected so the lock can never escape this stack frame.
640
+ */
641
+ withLearningWriterFence(scopeInput, fenceInput, callback) {
642
+ const { scopeKey } = canonicalEvaluationScope(scopeInput);
643
+ const fence = this.#normalizeLearningWriterFence(fenceInput);
644
+ this.#database.exec('BEGIN IMMEDIATE');
645
+ try {
646
+ const watermark = this.#database.prepare(`
647
+ SELECT watermark FROM evaluation_scope_watermarks WHERE scope_key = ?
648
+ `).get(scopeKey);
649
+ if (watermark?.watermark !== fence.scopeWatermark) {
650
+ this.#database.exec('COMMIT');
651
+ return Object.freeze({ matched: false, reason: 'watermark-changed' });
652
+ }
653
+ const pending = this.#database.prepare(`
654
+ SELECT 1 AS present
655
+ FROM evaluation_projection_outbox outbox
656
+ JOIN evaluation_outcomes outcome ON outcome.id = outbox.evaluation_id
657
+ WHERE outcome.scope_key = ? AND outcome.trust = 'trusted'
658
+ AND outbox.status = 'pending'
659
+ LIMIT 1
660
+ `).get(scopeKey);
661
+ if (pending !== undefined) {
662
+ this.#database.exec('COMMIT');
663
+ return Object.freeze({ matched: false, reason: 'projection-pending' });
664
+ }
665
+ const statement = this.#database.prepare(`
666
+ SELECT learning_version AS version, learning_digest AS digest,
667
+ learning_disposition AS disposition
668
+ FROM evaluation_task_projections
669
+ WHERE scope_key = ? AND subject_kind = ? AND subject_ref = ?
670
+ `);
671
+ for (const evidence of fence.evidence) {
672
+ const current = statement.get(scopeKey, evidence.subjectKind, evidence.subjectRef);
673
+ if (current === undefined || current.version !== evidence.version
674
+ || current.digest !== evidence.digest || current.disposition !== 'upsert') {
675
+ this.#database.exec('COMMIT');
676
+ return Object.freeze({ matched: false, reason: 'evidence-changed' });
677
+ }
678
+ }
679
+ const value = callback();
680
+ if (typeof value === 'object' && value !== null && 'then' in value
681
+ && typeof value.then === 'function') {
682
+ throw new EvaluationStoreError('invalid-input', 'learning writer fence callback must be synchronous');
683
+ }
684
+ this.#database.exec('COMMIT');
685
+ return Object.freeze({ matched: true, value });
686
+ }
687
+ catch (error) {
688
+ this.#database.exec('ROLLBACK');
689
+ throw error;
690
+ }
691
+ }
692
+ /** Query one deterministic latest row per task; raw query() remains the audit API. */
693
+ queryTasks(input) {
694
+ const { scopeKey } = canonicalEvaluationScope(input.scope);
695
+ const limit = input.limit ?? Math.min(50, this.#maxQueryLimit);
696
+ if (!Number.isSafeInteger(limit) || limit < 1 || limit > this.#maxQueryLimit) {
697
+ throw new EvaluationStoreError('invalid-input', `query limit must be between 1 and ${this.#maxQueryLimit}`);
698
+ }
699
+ const clauses = ['scope_key = ?'];
700
+ const parameters = [scopeKey];
701
+ const add = (column, value) => {
702
+ if (value === undefined)
703
+ return;
704
+ clauses.push(`${column} = ?`);
705
+ parameters.push(value);
706
+ };
707
+ add('situation', input.situation === undefined ? undefined : this.#situation(input.situation));
708
+ add('execution_status', input.executionStatus === undefined
709
+ ? undefined : oneOf(input.executionStatus, executionStatuses, 'executionStatus'));
710
+ add('objective_status', input.objectiveStatus === undefined
711
+ ? undefined : oneOf(input.objectiveStatus, objectiveStatuses, 'objectiveStatus'));
712
+ add('delivery_status', input.deliveryStatus === undefined
713
+ ? undefined : oneOf(input.deliveryStatus, deliveryStatuses, 'deliveryStatus'));
714
+ add('source_kind', input.sourceKind === undefined
715
+ ? undefined : oneOf(input.sourceKind, outcomeSourceKinds, 'sourceKind'));
716
+ add('trust', input.trust === undefined ? undefined : oneOf(input.trust, outcomeTrustLevels, 'trust'));
717
+ if (input.excludeSituationPrefix !== undefined) {
718
+ const prefix = this.#situation(input.excludeSituationPrefix);
719
+ clauses.push('substr(situation, 1, length(?)) <> ?');
720
+ parameters.push(prefix, prefix);
721
+ }
722
+ const [from, to] = this.#optionalRange(input.fromOccurredAt, input.toOccurredAt);
723
+ if (from !== undefined) {
724
+ clauses.push('occurred_at >= ?');
725
+ parameters.push(from);
726
+ }
727
+ if (to !== undefined) {
728
+ clauses.push('occurred_at <= ?');
729
+ parameters.push(to);
730
+ }
731
+ parameters.push(limit);
732
+ const rows = this.#database.prepare(`
733
+ SELECT * FROM evaluation_task_projection_view WHERE ${clauses.join(' AND ')}
734
+ ORDER BY occurred_at DESC, task_subject_key DESC LIMIT ?
735
+ `).all(...parameters);
736
+ return rows.map(row => projected(row));
737
+ }
310
738
  query(input) {
311
739
  const { scopeKey } = canonicalEvaluationScope(input.scope);
312
740
  const limit = input.limit ?? Math.min(50, this.#maxQueryLimit);
@@ -384,7 +812,7 @@ export class EvaluationStore {
384
812
  TOTAL(tool_calls) AS tool_calls,
385
813
  TOTAL(latency_ms) AS latency_total,
386
814
  COUNT(latency_ms) AS latency_count
387
- FROM evaluation_outcomes
815
+ FROM evaluation_task_projection_view
388
816
  WHERE scope_key = ? AND occurred_at >= ? AND occurred_at <= ?
389
817
  AND (? IS NULL OR situation = ?)
390
818
  AND (? IS NULL OR substr(situation, 1, length(?)) <> ?)
@@ -431,6 +859,17 @@ export class EvaluationStore {
431
859
  `).get();
432
860
  const assessmentRow = this.#database.prepare('SELECT COUNT(*) AS count FROM evaluation_self_assessments')
433
861
  .get();
862
+ const taskRow = this.#database.prepare(`
863
+ SELECT COUNT(*) AS count, SUM(objective_conflicted) AS conflicted
864
+ FROM evaluation_task_projections WHERE primary_outcome_id IS NOT NULL
865
+ `).get();
866
+ const projectionRow = this.#database.prepare(`
867
+ SELECT COUNT(*) AS pending,
868
+ SUM(attempt_count > 0) AS retrying,
869
+ TOTAL(attempt_count) AS attempts,
870
+ MIN(created_at) AS oldest
871
+ FROM evaluation_projection_outbox WHERE status = 'pending'
872
+ `).get();
434
873
  return Object.freeze({
435
874
  ready: true,
436
875
  schemaVersion: evaluationSchemaVersion,
@@ -439,9 +878,246 @@ export class EvaluationStore {
439
878
  selfReportedOutcomes: row.self_reported ?? 0,
440
879
  externalOutcomes: row.external ?? 0,
441
880
  selfAssessments: assessmentRow.count,
881
+ taskProjections: taskRow.count,
882
+ conflictedTaskProjections: taskRow.conflicted ?? 0,
883
+ pendingProjections: projectionRow.pending,
884
+ retryingProjections: projectionRow.retrying ?? 0,
885
+ projectionAttempts: projectionRow.attempts,
886
+ ...(projectionRow.oldest === null ? {} : { oldestPendingProjectionAt: projectionRow.oldest }),
442
887
  ...(row.latest === null ? {} : { latestOccurredAt: row.latest }),
443
888
  });
444
889
  }
890
+ #rebuildTaskProjections() {
891
+ this.#database.exec('BEGIN IMMEDIATE');
892
+ try {
893
+ const rows = this.#database.prepare(`
894
+ SELECT * FROM evaluation_outcomes ORDER BY recorded_at, id
895
+ `).all();
896
+ const updateOutcome = this.#database.prepare(`
897
+ UPDATE evaluation_outcomes
898
+ SET task_subject_key = ?, task_subject_kind = ?, task_subject_ref = ?
899
+ WHERE id = ?
900
+ `);
901
+ const insertSubject = this.#database.prepare(`
902
+ INSERT INTO evaluation_task_projections(
903
+ subject_key, scope_key, subject_kind, subject_ref, updated_at)
904
+ VALUES (?, ?, ?, ?, ?)
905
+ ON CONFLICT(subject_key) DO UPDATE SET
906
+ scope_key = excluded.scope_key,
907
+ subject_kind = excluded.subject_kind,
908
+ subject_ref = excluded.subject_ref
909
+ `);
910
+ const subjects = new Set();
911
+ for (const row of rows) {
912
+ const subject = row.task_subject_key === null
913
+ || row.task_subject_kind === null
914
+ || row.task_subject_ref === null
915
+ ? taskSubject(row.scope_key, row.id, JSON.parse(row.evidence_json))
916
+ : { key: row.task_subject_key, kind: row.task_subject_kind, ref: row.task_subject_ref };
917
+ if (row.task_subject_key !== subject.key
918
+ || row.task_subject_kind !== subject.kind
919
+ || row.task_subject_ref !== subject.ref) {
920
+ updateOutcome.run(subject.key, subject.kind, subject.ref, row.id);
921
+ }
922
+ insertSubject.run(subject.key, row.scope_key, subject.kind, subject.ref, row.recorded_at);
923
+ subjects.add(subject.key);
924
+ }
925
+ for (const subjectKey of [...subjects].sort()) {
926
+ const refreshed = this.#refreshTaskProjection(subjectKey);
927
+ if (!refreshed.learningVersionChanged)
928
+ continue;
929
+ const current = this.#database.prepare(`
930
+ SELECT projection.scope_key, projection.primary_outcome_id, projection.updated_at,
931
+ outcome.trust
932
+ FROM evaluation_task_projections projection
933
+ JOIN evaluation_outcomes outcome ON outcome.id = projection.primary_outcome_id
934
+ WHERE projection.subject_key = ?
935
+ `).get(subjectKey);
936
+ if (current.trust !== 'trusted')
937
+ continue;
938
+ this.#database.prepare(`
939
+ INSERT INTO evaluation_projection_outbox(
940
+ evaluation_id, status, attempt_count, next_attempt_at,
941
+ last_failure_at, last_failure_code, created_at, updated_at)
942
+ VALUES (?, 'pending', 0, ?, NULL, NULL, ?, ?)
943
+ ON CONFLICT(evaluation_id) DO UPDATE SET
944
+ status = 'pending', attempt_count = 0,
945
+ next_attempt_at = excluded.next_attempt_at,
946
+ last_failure_at = NULL, last_failure_code = NULL,
947
+ updated_at = excluded.updated_at
948
+ `).run(current.primary_outcome_id, current.updated_at, current.updated_at, current.updated_at);
949
+ this.#advanceScopeWatermark(current.scope_key, current.updated_at);
950
+ }
951
+ this.#database.exec('COMMIT');
952
+ }
953
+ catch (error) {
954
+ this.#database.exec('ROLLBACK');
955
+ throw error;
956
+ }
957
+ }
958
+ #refreshTaskProjection(subjectKey) {
959
+ const projection = this.#database.prepare(`
960
+ SELECT subject_kind, subject_ref, learning_version, learning_digest, learning_disposition
961
+ FROM evaluation_task_projections WHERE subject_key = ?
962
+ `).get(subjectKey);
963
+ if (projection === undefined) {
964
+ throw new EvaluationStoreError('not-found', 'task projection subject was not found');
965
+ }
966
+ const rows = this.#database.prepare(`
967
+ SELECT * FROM evaluation_outcomes WHERE task_subject_key = ?
968
+ ORDER BY recorded_at, id
969
+ `).all(subjectKey);
970
+ if (rows.length === 0)
971
+ return { learningVersionChanged: false };
972
+ let primary;
973
+ let execution;
974
+ let objective;
975
+ let delivery;
976
+ let objectiveConflicted = false;
977
+ if (projection.subject_kind === 'outcome') {
978
+ primary = latest(rows);
979
+ execution = primary;
980
+ objective = primary;
981
+ delivery = primary;
982
+ }
983
+ else {
984
+ const terminals = rows.filter(row => isAuthoritativeAutomationTerminal(row));
985
+ execution = latest(terminals);
986
+ const owners = rows.filter(row => isAuthenticatedOwnerFeedback(row)
987
+ && row.objective_status !== 'unknown');
988
+ const ownerStatuses = new Set(owners.map(row => row.objective_status));
989
+ if (ownerStatuses.size > 1)
990
+ objectiveConflicted = true;
991
+ else if (owners.length > 0)
992
+ objective = latest(owners);
993
+ else {
994
+ const trustedObjectives = rows.filter(row => row.trust === 'trusted'
995
+ && row.source_kind !== 'user-feedback'
996
+ && row.objective_status !== 'unknown');
997
+ const ranked = trustedObjectives.map(row => ({
998
+ row,
999
+ // Independent trusted evaluators supersede the terminal producer's
1000
+ // initial objective, while an owner judgement has already won above.
1001
+ rank: isAuthoritativeAutomationTerminal(row) ? 1 : 2,
1002
+ })).sort((left, right) => left.rank - right.rank || newer(left.row, right.row));
1003
+ objective = ranked.at(-1)?.row;
1004
+ }
1005
+ const ownerDelivery = latest(rows.filter(row => isAuthenticatedOwnerFeedback(row)
1006
+ && row.delivery_status === 'delivered'));
1007
+ const trustedDelivery = latest(rows.filter(row => row.trust === 'trusted'
1008
+ && row.source_kind === 'delivery' && row.delivery_status !== 'unknown'));
1009
+ delivery = ownerDelivery ?? trustedDelivery ?? execution;
1010
+ const rankedPrimary = rows.map(row => ({
1011
+ row,
1012
+ rank: isAuthoritativeAutomationTerminal(row)
1013
+ ? 4
1014
+ : isAuthenticatedOwnerFeedback(row)
1015
+ ? 3
1016
+ : row.trust === 'trusted'
1017
+ ? 2
1018
+ : row.trust === 'external' ? 1 : 0,
1019
+ })).sort((left, right) => left.rank - right.rank || newer(left.row, right.row));
1020
+ primary = rankedPrimary.at(-1).row;
1021
+ }
1022
+ const objectiveStatus = objectiveConflicted ? 'unknown' : objective?.objective_status ?? 'unknown';
1023
+ const objectiveSituationMismatch = objective !== undefined && objective.situation !== primary.situation;
1024
+ const hostRunbook = execution !== undefined && isAuthoritativeAutomationTerminal(execution)
1025
+ && /^host-runbook-v[1-9][0-9]*$/u.test(execution.evaluator_version);
1026
+ const learningDisposition = !objectiveConflicted
1027
+ && !objectiveSituationMismatch
1028
+ && objective?.trust === 'trusted'
1029
+ && (objectiveStatus === 'achieved' || objectiveStatus === 'not-achieved')
1030
+ && (projection.subject_kind !== 'automation-run' || execution !== undefined)
1031
+ && !hostRunbook
1032
+ ? 'upsert'
1033
+ : 'retract';
1034
+ const evidenceOutcomeId = objectiveConflicted ? undefined : objective?.id;
1035
+ const learningExecution = executionComponent(execution);
1036
+ const learningObjective = objectiveConflicted ? undefined : objectiveComponent(objective);
1037
+ const learningDigest = evaluationLearningProjectionDigest({
1038
+ scopeKey: primary.scope_key,
1039
+ situation: primary.situation,
1040
+ ...(learningExecution === undefined ? {} : { execution: learningExecution }),
1041
+ ...(learningObjective === undefined ? {} : { objective: learningObjective }),
1042
+ projection: {
1043
+ subjectKind: projection.subject_kind,
1044
+ subjectRef: projection.subject_ref,
1045
+ disposition: learningDisposition,
1046
+ ...(evidenceOutcomeId === undefined ? {} : { evidenceOutcomeId }),
1047
+ },
1048
+ });
1049
+ const learningVersionChanged = projection.learning_digest !== learningDigest
1050
+ || projection.learning_disposition !== learningDisposition;
1051
+ const learningVersion = projection.learning_version === 0
1052
+ ? 1
1053
+ : learningVersionChanged
1054
+ ? projection.learning_version + 1
1055
+ : projection.learning_version;
1056
+ if (!Number.isSafeInteger(learningVersion)) {
1057
+ throw new EvaluationStoreError('invalid-input', 'task learning projection version overflow');
1058
+ }
1059
+ const updatedAt = Math.max(...rows.map(row => row.recorded_at));
1060
+ this.#database.prepare(`
1061
+ UPDATE evaluation_task_projections
1062
+ SET primary_outcome_id = ?, execution_outcome_id = ?, objective_outcome_id = ?,
1063
+ delivery_outcome_id = ?, objective_conflicted = ?, learning_version = ?,
1064
+ learning_digest = ?, learning_disposition = ?, updated_at = ?
1065
+ WHERE subject_key = ?
1066
+ `).run(primary.id, execution?.id ?? null, objectiveConflicted ? null : objective?.id ?? null, delivery?.id ?? null, objectiveConflicted ? 1 : 0, learningVersion, learningDigest, learningDisposition, updatedAt, subjectKey);
1067
+ return { learningVersionChanged: projection.learning_version === 0 || learningVersionChanged };
1068
+ }
1069
+ #advanceScopeWatermark(scopeKey, updatedAt) {
1070
+ this.#database.prepare(`
1071
+ INSERT INTO evaluation_scope_watermarks(scope_key, watermark, updated_at)
1072
+ VALUES (?, 1, ?)
1073
+ ON CONFLICT(scope_key) DO UPDATE SET
1074
+ watermark = evaluation_scope_watermarks.watermark + 1,
1075
+ updated_at = excluded.updated_at
1076
+ `).run(scopeKey, updatedAt);
1077
+ const row = this.#database.prepare(`
1078
+ SELECT watermark FROM evaluation_scope_watermarks WHERE scope_key = ?
1079
+ `).get(scopeKey);
1080
+ if (!Number.isSafeInteger(row.watermark) || row.watermark < 1) {
1081
+ throw new EvaluationStoreError('invalid-input', 'canonical scope watermark overflow');
1082
+ }
1083
+ return row.watermark;
1084
+ }
1085
+ #normalizeLearningWriterFence(input) {
1086
+ if (typeof input !== 'object' || input === null || Array.isArray(input)
1087
+ || !Number.isSafeInteger(input.scopeWatermark) || input.scopeWatermark < 1
1088
+ || !Array.isArray(input.evidence) || input.evidence.length < 1
1089
+ || input.evidence.length > 10_000) {
1090
+ throw new EvaluationStoreError('invalid-input', 'learning writer fence is invalid');
1091
+ }
1092
+ const seen = new Set();
1093
+ const entries = input.evidence.map((raw, index) => {
1094
+ if (typeof raw !== 'object' || raw === null || Array.isArray(raw)
1095
+ || raw.disposition !== 'upsert'
1096
+ || (raw.subjectKind !== 'automation-run' && raw.subjectKind !== 'outcome')
1097
+ || !Number.isSafeInteger(raw.version) || raw.version < 1
1098
+ || raw.version > 1_000_000_000
1099
+ || typeof raw.digest !== 'string' || !/^[a-f\d]{64}$/u.test(raw.digest)) {
1100
+ throw new EvaluationStoreError('invalid-input', `learning writer fence evidence[${index}] is invalid`);
1101
+ }
1102
+ const subjectRef = boundedText(raw.subjectRef, `evidence[${index}].subjectRef`, 1_000);
1103
+ const identity = JSON.stringify([raw.subjectKind, subjectRef]);
1104
+ if (seen.has(identity)) {
1105
+ throw new EvaluationStoreError('invalid-input', 'learning writer fence contains duplicate evidence');
1106
+ }
1107
+ seen.add(identity);
1108
+ return Object.freeze({
1109
+ subjectKind: raw.subjectKind,
1110
+ subjectRef,
1111
+ version: raw.version,
1112
+ digest: raw.digest,
1113
+ disposition: 'upsert',
1114
+ });
1115
+ });
1116
+ return Object.freeze({
1117
+ scopeWatermark: input.scopeWatermark,
1118
+ evidence: Object.freeze(entries),
1119
+ });
1120
+ }
445
1121
  #normalize(input) {
446
1122
  const { scope, scopeKey } = canonicalEvaluationScope(input.scope);
447
1123
  const situation = this.#situation(input.situation);