@dsh-enhanced/assistant-evaluation 0.1.23 → 0.1.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +109 -3
- package/bin/dsh-benchmark.js +21 -0
- package/lib/benchmark/cli.d.ts +6 -0
- package/lib/benchmark/cli.d.ts.map +1 -0
- package/lib/benchmark/cli.js +332 -0
- package/lib/benchmark/cli.js.map +1 -0
- package/lib/benchmark/corpus.d.ts +28 -0
- package/lib/benchmark/corpus.d.ts.map +1 -0
- package/lib/benchmark/corpus.js +173 -0
- package/lib/benchmark/corpus.js.map +1 -0
- package/lib/benchmark/deepseek.d.ts +15 -0
- package/lib/benchmark/deepseek.d.ts.map +1 -0
- package/lib/benchmark/deepseek.js +143 -0
- package/lib/benchmark/deepseek.js.map +1 -0
- package/lib/benchmark/holdout-evidence.d.ts +126 -0
- package/lib/benchmark/holdout-evidence.d.ts.map +1 -0
- package/lib/benchmark/holdout-evidence.js +466 -0
- package/lib/benchmark/holdout-evidence.js.map +1 -0
- package/lib/benchmark/holdout-protocol.d.ts +68 -0
- package/lib/benchmark/holdout-protocol.d.ts.map +1 -0
- package/lib/benchmark/holdout-protocol.js +271 -0
- package/lib/benchmark/holdout-protocol.js.map +1 -0
- package/lib/benchmark/holdout-provider.d.ts +26 -0
- package/lib/benchmark/holdout-provider.d.ts.map +1 -0
- package/lib/benchmark/holdout-provider.js +382 -0
- package/lib/benchmark/holdout-provider.js.map +1 -0
- package/lib/benchmark/holdout.d.ts +83 -0
- package/lib/benchmark/holdout.d.ts.map +1 -0
- package/lib/benchmark/holdout.js +391 -0
- package/lib/benchmark/holdout.js.map +1 -0
- package/lib/benchmark/index.d.ts +9 -0
- package/lib/benchmark/index.d.ts.map +1 -0
- package/lib/benchmark/index.js +7 -0
- package/lib/benchmark/index.js.map +1 -0
- package/lib/benchmark/memory-corpus.d.ts +55 -0
- package/lib/benchmark/memory-corpus.d.ts.map +1 -0
- package/lib/benchmark/memory-corpus.js +153 -0
- package/lib/benchmark/memory-corpus.js.map +1 -0
- package/lib/benchmark/memory-runtime.d.ts +63 -0
- package/lib/benchmark/memory-runtime.d.ts.map +1 -0
- package/lib/benchmark/memory-runtime.js +75 -0
- package/lib/benchmark/memory-runtime.js.map +1 -0
- package/lib/benchmark/native.d.ts +46 -0
- package/lib/benchmark/native.d.ts.map +1 -0
- package/lib/benchmark/native.js +300 -0
- package/lib/benchmark/native.js.map +1 -0
- package/lib/benchmark/report.d.ts +4 -0
- package/lib/benchmark/report.d.ts.map +1 -0
- package/lib/benchmark/report.js +153 -0
- package/lib/benchmark/report.js.map +1 -0
- package/lib/benchmark/runner.d.ts +25 -0
- package/lib/benchmark/runner.d.ts.map +1 -0
- package/lib/benchmark/runner.js +119 -0
- package/lib/benchmark/runner.js.map +1 -0
- package/lib/benchmark/schema.d.ts +20 -0
- package/lib/benchmark/schema.d.ts.map +1 -0
- package/lib/benchmark/schema.js +183 -0
- package/lib/benchmark/schema.js.map +1 -0
- package/lib/benchmark/store.d.ts +18 -0
- package/lib/benchmark/store.d.ts.map +1 -0
- package/lib/benchmark/store.js +321 -0
- package/lib/benchmark/store.js.map +1 -0
- package/lib/benchmark/strategy-capabilities.d.ts +154 -0
- package/lib/benchmark/strategy-capabilities.d.ts.map +1 -0
- package/lib/benchmark/strategy-capabilities.js +300 -0
- package/lib/benchmark/strategy-capabilities.js.map +1 -0
- package/lib/benchmark/strategy-config.d.ts +26 -0
- package/lib/benchmark/strategy-config.d.ts.map +1 -0
- package/lib/benchmark/strategy-config.js +80 -0
- package/lib/benchmark/strategy-config.js.map +1 -0
- package/lib/benchmark/strategy-corpus.d.ts +21 -0
- package/lib/benchmark/strategy-corpus.d.ts.map +1 -0
- package/lib/benchmark/strategy-corpus.js +95 -0
- package/lib/benchmark/strategy-corpus.js.map +1 -0
- package/lib/benchmark/strategy-evidence.d.ts +161 -0
- package/lib/benchmark/strategy-evidence.d.ts.map +1 -0
- package/lib/benchmark/strategy-evidence.js +520 -0
- package/lib/benchmark/strategy-evidence.js.map +1 -0
- package/lib/benchmark/strategy-executor.d.ts +27 -0
- package/lib/benchmark/strategy-executor.d.ts.map +1 -0
- package/lib/benchmark/strategy-executor.js +178 -0
- package/lib/benchmark/strategy-executor.js.map +1 -0
- package/lib/benchmark/strategy-goal-runtime.d.ts +78 -0
- package/lib/benchmark/strategy-goal-runtime.d.ts.map +1 -0
- package/lib/benchmark/strategy-goal-runtime.js +324 -0
- package/lib/benchmark/strategy-goal-runtime.js.map +1 -0
- package/lib/benchmark/strategy-meter.d.ts +62 -0
- package/lib/benchmark/strategy-meter.d.ts.map +1 -0
- package/lib/benchmark/strategy-meter.js +236 -0
- package/lib/benchmark/strategy-meter.js.map +1 -0
- package/lib/benchmark/strategy-owner.d.ts +71 -0
- package/lib/benchmark/strategy-owner.d.ts.map +1 -0
- package/lib/benchmark/strategy-owner.js +282 -0
- package/lib/benchmark/strategy-owner.js.map +1 -0
- package/lib/benchmark/strategy-plan.d.ts +51 -0
- package/lib/benchmark/strategy-plan.d.ts.map +1 -0
- package/lib/benchmark/strategy-plan.js +71 -0
- package/lib/benchmark/strategy-plan.js.map +1 -0
- package/lib/benchmark/strategy-policy.d.ts +16 -0
- package/lib/benchmark/strategy-policy.d.ts.map +1 -0
- package/lib/benchmark/strategy-policy.js +77 -0
- package/lib/benchmark/strategy-policy.js.map +1 -0
- package/lib/benchmark/strategy.d.ts +17 -0
- package/lib/benchmark/strategy.d.ts.map +1 -0
- package/lib/benchmark/strategy.js +21 -0
- package/lib/benchmark/strategy.js.map +1 -0
- package/lib/benchmark/types.d.ts +133 -0
- package/lib/benchmark/types.d.ts.map +1 -0
- package/lib/benchmark/types.js +3 -0
- package/lib/benchmark/types.js.map +1 -0
- package/lib/benchmark/usage.d.ts +36 -0
- package/lib/benchmark/usage.d.ts.map +1 -0
- package/lib/benchmark/usage.js +84 -0
- package/lib/benchmark/usage.js.map +1 -0
- package/lib/service.d.ts +50 -1
- package/lib/service.d.ts.map +1 -1
- package/lib/service.js +556 -9
- package/lib/service.js.map +1 -1
- package/lib/sqlite.d.ts +1 -1
- package/lib/sqlite.d.ts.map +1 -1
- package/lib/sqlite.js +94 -2
- package/lib/sqlite.js.map +1 -1
- package/lib/store.d.ts +27 -5
- package/lib/store.d.ts.map +1 -1
- package/lib/store.js +324 -25
- package/lib/store.js.map +1 -1
- package/lib/types.d.ts +114 -5
- package/lib/types.d.ts.map +1 -1
- package/lib/types.js.map +1 -1
- package/lib/version.d.ts +1 -1
- package/lib/version.js +1 -1
- package/package.json +125 -18
package/lib/store.js
CHANGED
|
@@ -255,14 +255,18 @@ export function evaluationLearningProjectionDigest(input) {
|
|
|
255
255
|
]);
|
|
256
256
|
}
|
|
257
257
|
function taskSubject(scopeKey, outcomeId, references) {
|
|
258
|
-
const
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
258
|
+
const typed = references.flatMap(reference => (reference.kind === 'automation-run' || reference.kind === 'foreground-turn' || reference.kind === 'goal-step' || reference.kind === 'goal-outcome'
|
|
259
|
+
? [{ kind: reference.kind, ref: reference.ref }]
|
|
260
|
+
: []));
|
|
261
|
+
const identities = new Map(typed.map(reference => [`${reference.kind}\u0000${reference.ref}`, reference]));
|
|
262
|
+
// A row that names more than one task subject is intentionally quarantined
|
|
263
|
+
// to its own audit outcome. It must never bridge an Automation and a
|
|
264
|
+
// foreground turn (or two runs) into one learning subject.
|
|
265
|
+
if (identities.size === 1) {
|
|
266
|
+
const { kind, ref } = [...identities.values()][0];
|
|
263
267
|
return Object.freeze({
|
|
264
|
-
key: JSON.stringify([scopeKey,
|
|
265
|
-
kind
|
|
268
|
+
key: JSON.stringify([scopeKey, kind, ref]),
|
|
269
|
+
kind,
|
|
266
270
|
ref,
|
|
267
271
|
});
|
|
268
272
|
}
|
|
@@ -303,9 +307,24 @@ function isAuthenticatedOwnerFeedback(row) {
|
|
|
303
307
|
&& row.source_id === 'assistant-delivery/typed-owner-feedback'
|
|
304
308
|
&& row.evaluator_id === 'assistant-delivery-owner-feedback'
|
|
305
309
|
&& row.evaluator_version === '2'
|
|
306
|
-
&& containsEvidence(row, 'automation-run')
|
|
310
|
+
&& (containsEvidence(row, 'automation-run') || containsEvidence(row, 'foreground-turn')
|
|
311
|
+
|| containsEvidence(row, 'goal-outcome'))
|
|
307
312
|
&& containsEvidence(row, 'delivery-outbox');
|
|
308
313
|
}
|
|
314
|
+
function isTrustedVerifierReceipt(row) {
|
|
315
|
+
return row.trust === 'trusted'
|
|
316
|
+
&& row.source_kind === 'evaluator'
|
|
317
|
+
&& row.source_id === 'assistant-verifier'
|
|
318
|
+
&& row.evaluator_id === 'assistant-verifier'
|
|
319
|
+
&& row.evaluator_version === '1'
|
|
320
|
+
&& containsEvidence(row, 'acceptance-contract')
|
|
321
|
+
&& containsEvidence(row, 'verification-receipt');
|
|
322
|
+
}
|
|
323
|
+
function isAuthoritativeForegroundTerminal(row) {
|
|
324
|
+
return isTrustedVerifierReceipt(row)
|
|
325
|
+
&& row.source_kind === 'evaluator'
|
|
326
|
+
&& containsEvidence(row, 'foreground-turn');
|
|
327
|
+
}
|
|
309
328
|
export class EvaluationStore {
|
|
310
329
|
#database;
|
|
311
330
|
#now;
|
|
@@ -356,15 +375,240 @@ export class EvaluationStore {
|
|
|
356
375
|
`).get(outcomeId, scopeKey);
|
|
357
376
|
return row === undefined ? undefined : stored(row);
|
|
358
377
|
}
|
|
359
|
-
|
|
378
|
+
/** Adopt a pre-revision owner row only through the exact Host delivery capability. */
|
|
379
|
+
adoptLegacyOwnerFeedback(claims) {
|
|
380
|
+
if (claims.ownerCommand === undefined || claims.initialIdempotencyKey === undefined)
|
|
381
|
+
return;
|
|
382
|
+
// Pre-revision rows only represented Automation runs. Foreground turns were
|
|
383
|
+
// introduced with the typed subject claims and cannot be guessed from them.
|
|
384
|
+
if ((claims.subjectKind ?? 'automation-run') !== 'automation-run')
|
|
385
|
+
return;
|
|
386
|
+
const scopeKey = canonicalEvaluationScope(claims.scope).scopeKey;
|
|
387
|
+
this.#database.exec('BEGIN IMMEDIATE');
|
|
388
|
+
try {
|
|
389
|
+
const row = this.#database.prepare(`SELECT * FROM evaluation_outcomes WHERE idempotency_key = ? AND scope_key = ?`)
|
|
390
|
+
.get(claims.initialIdempotencyKey, scopeKey);
|
|
391
|
+
if (row !== undefined && isAuthenticatedOwnerFeedback(row)) {
|
|
392
|
+
const evidence = JSON.parse(row.evidence_json);
|
|
393
|
+
if (!evidence.some(ref => ref.kind === 'automation-run' && ref.ref === claims.runId)
|
|
394
|
+
|| !evidence.some(ref => ref.kind === 'delivery-outbox' && ref.ref === claims.outboxId)) {
|
|
395
|
+
throw new EvaluationStoreError('invalid-input', 'legacy feedback does not match exact delivered result');
|
|
396
|
+
}
|
|
397
|
+
this.#database.prepare(`INSERT INTO evaluation_owner_revisions
|
|
398
|
+
(outcome_id, subject_key, lineage, version, previous_outcome_id, action, command_json)
|
|
399
|
+
VALUES (?, ?, ?, 1, NULL, 'initial', ?) ON CONFLICT DO NOTHING`)
|
|
400
|
+
.run(row.id, row.task_subject_key, JSON.stringify([claims.ownerCommand.principalRecordId, claims.ownerCommand.principalVersion]), JSON.stringify({
|
|
401
|
+
...claims.ownerCommand, action: 'legacy-adoption', outboxId: claims.outboxId,
|
|
402
|
+
runId: claims.runId, bindingId: claims.bindingId, principalId: claims.principalId,
|
|
403
|
+
}));
|
|
404
|
+
}
|
|
405
|
+
this.#database.exec('COMMIT');
|
|
406
|
+
}
|
|
407
|
+
catch (error) {
|
|
408
|
+
this.#database.exec('ROLLBACK');
|
|
409
|
+
throw error;
|
|
410
|
+
}
|
|
411
|
+
}
|
|
412
|
+
/**
|
|
413
|
+
* Adopt the exact current Verifier whole-goal judgement as revision one for
|
|
414
|
+
* an authenticated owner lineage. The immutable Verifier outcome remains the
|
|
415
|
+
* authority row; the Goals proof is checked by the service and is not stored.
|
|
416
|
+
*/
|
|
417
|
+
adoptTrustedGoalOutcomeOwnerBaseline(claims, proof, verifierOutcome) {
|
|
418
|
+
if (claims.subjectKind !== 'goal-outcome' || claims.ownerCommand === undefined) {
|
|
419
|
+
throw new EvaluationStoreError('invalid-input', 'goal outcome owner baseline claims are invalid');
|
|
420
|
+
}
|
|
421
|
+
const { scopeKey } = canonicalEvaluationScope(claims.scope);
|
|
422
|
+
const subjectRef = boundedText(claims.subjectRef, 'goal outcome assessmentId', 1_000);
|
|
423
|
+
const subject = taskSubject(scopeKey, '', [{ kind: 'goal-outcome', ref: subjectRef }]);
|
|
424
|
+
const lineage = JSON.stringify([
|
|
425
|
+
claims.ownerCommand.principalRecordId,
|
|
426
|
+
claims.ownerCommand.principalVersion,
|
|
427
|
+
]);
|
|
428
|
+
const normalized = this.#normalize(verifierOutcome);
|
|
429
|
+
const payloadHash = digest(normalized);
|
|
430
|
+
const recordedAt = timestamp(this.#now(), 'recordedAt');
|
|
431
|
+
const outcomeId = `outcome-${randomUUID()}`;
|
|
432
|
+
const verifierSubject = taskSubject(normalized.scopeKey, outcomeId, normalized.evidence);
|
|
433
|
+
const exactEvidence = (entries, kind, ref, digestValue) => entries.some(item => (item.kind === kind && item.ref === ref
|
|
434
|
+
&& (digestValue === undefined || item.digest === digestValue)));
|
|
435
|
+
if (normalized.scopeKey !== scopeKey || normalized.situation !== claims.situation
|
|
436
|
+
|| normalized.objectiveStatus !== proof.receipt.objectiveStatus
|
|
437
|
+
|| normalized.deliveryStatus !== 'not-required' || normalized.trust !== 'trusted'
|
|
438
|
+
|| normalized.source.kind !== 'evaluator' || normalized.source.id !== 'assistant-verifier'
|
|
439
|
+
|| normalized.evaluator.id !== 'assistant-verifier' || normalized.evaluator.version !== '1'
|
|
440
|
+
|| normalized.occurredAt !== proof.receipt.completedAt
|
|
441
|
+
|| normalized.idempotencyKey !== `assistant-verifier:${proof.receipt.id}`
|
|
442
|
+
|| verifierSubject.kind !== 'goal-outcome' || verifierSubject.ref !== subjectRef
|
|
443
|
+
|| normalized.evidence.length !== 4
|
|
444
|
+
|| !exactEvidence(normalized.evidence, 'goal-outcome', subjectRef)
|
|
445
|
+
|| !exactEvidence(normalized.evidence, 'acceptance-contract', proof.contract.id, proof.contract.digest)
|
|
446
|
+
|| !exactEvidence(normalized.evidence, 'verification-receipt', proof.receipt.id, proof.receipt.digest)
|
|
447
|
+
|| normalized.evidence.filter(item => item.kind === 'execution' && item.ref === subjectRef).length !== 1) {
|
|
448
|
+
throw new EvaluationStoreError('invalid-input', 'trusted goal outcome baseline input is invalid');
|
|
449
|
+
}
|
|
450
|
+
let changed = false;
|
|
451
|
+
this.#database.exec('BEGIN IMMEDIATE');
|
|
452
|
+
try {
|
|
453
|
+
let row = this.#database.prepare(`
|
|
454
|
+
SELECT outcome.* FROM evaluation_outcomes outcome
|
|
455
|
+
WHERE outcome.task_subject_key = ? AND outcome.task_subject_kind = 'goal-outcome'
|
|
456
|
+
AND outcome.task_subject_ref = ? AND outcome.trust = 'trusted'
|
|
457
|
+
AND outcome.source_kind = 'evaluator' AND outcome.source_id = 'assistant-verifier'
|
|
458
|
+
AND outcome.evaluator_id = 'assistant-verifier' AND outcome.evaluator_version = '1'
|
|
459
|
+
ORDER BY outcome.recorded_at DESC, outcome.id DESC LIMIT 1
|
|
460
|
+
`).get(subject.key, subjectRef);
|
|
461
|
+
if (row === undefined) {
|
|
462
|
+
const metric = normalized.metrics;
|
|
463
|
+
this.#database.prepare(`
|
|
464
|
+
INSERT INTO evaluation_outcomes(
|
|
465
|
+
id, idempotency_key, payload_hash, scope_key, workspace, preset, situation,
|
|
466
|
+
execution_status, objective_status, delivery_status, source_kind, source_id,
|
|
467
|
+
trust, evidence_json, metrics_json, cost_usd_micros, latency_ms, input_tokens,
|
|
468
|
+
output_tokens, tool_calls, occurred_at, recorded_at, evaluator_id, evaluator_version,
|
|
469
|
+
task_subject_key, task_subject_kind, task_subject_ref)
|
|
470
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
471
|
+
ON CONFLICT(idempotency_key) DO NOTHING
|
|
472
|
+
`).run(outcomeId, normalized.idempotencyKey, payloadHash, normalized.scopeKey, normalized.scope.workspace, normalized.scope.preset, normalized.situation, normalized.executionStatus, normalized.objectiveStatus, normalized.deliveryStatus, normalized.source.kind, normalized.source.id, normalized.trust, JSON.stringify(normalized.evidence), JSON.stringify(normalized.metrics), metric.costUsdMicros ?? null, metric.latencyMs ?? null, metric.inputTokens ?? null, metric.outputTokens ?? null, metric.toolCalls ?? null, normalized.occurredAt, recordedAt, normalized.evaluator.id, normalized.evaluator.version, verifierSubject.key, verifierSubject.kind, verifierSubject.ref);
|
|
473
|
+
const winner = this.#database.prepare('SELECT * FROM evaluation_outcomes WHERE idempotency_key = ?')
|
|
474
|
+
.get(normalized.idempotencyKey);
|
|
475
|
+
if (winner.payload_hash !== payloadHash || winner.task_subject_key !== verifierSubject.key
|
|
476
|
+
|| winner.task_subject_kind !== verifierSubject.kind || winner.task_subject_ref !== verifierSubject.ref) {
|
|
477
|
+
throw new EvaluationStoreError('idempotency-conflict', 'trusted verifier outcome identity was reused');
|
|
478
|
+
}
|
|
479
|
+
this.#database.prepare(`
|
|
480
|
+
INSERT INTO evaluation_task_projections(
|
|
481
|
+
subject_key, scope_key, subject_kind, subject_ref, updated_at)
|
|
482
|
+
VALUES (?, ?, ?, ?, ?) ON CONFLICT(subject_key) DO NOTHING
|
|
483
|
+
`).run(verifierSubject.key, winner.scope_key, verifierSubject.kind, verifierSubject.ref, winner.recorded_at);
|
|
484
|
+
const refreshed = this.#refreshTaskProjection(verifierSubject.key);
|
|
485
|
+
if (refreshed.learningVersionChanged) {
|
|
486
|
+
this.#database.prepare(`
|
|
487
|
+
INSERT INTO evaluation_projection_outbox(
|
|
488
|
+
evaluation_id, status, attempt_count, next_attempt_at,
|
|
489
|
+
last_failure_at, last_failure_code, created_at, updated_at)
|
|
490
|
+
VALUES (?, 'pending', 0, ?, NULL, NULL, ?, ?)
|
|
491
|
+
ON CONFLICT(evaluation_id) DO NOTHING
|
|
492
|
+
`).run(winner.id, winner.recorded_at, winner.recorded_at, winner.recorded_at);
|
|
493
|
+
this.#advanceScopeWatermark(winner.scope_key, winner.recorded_at);
|
|
494
|
+
}
|
|
495
|
+
row = winner;
|
|
496
|
+
changed = true;
|
|
497
|
+
}
|
|
498
|
+
const storedEvidence = JSON.parse(row.evidence_json);
|
|
499
|
+
if (!isTrustedVerifierReceipt(row)
|
|
500
|
+
|| row.idempotency_key !== normalized.idempotencyKey || row.payload_hash !== payloadHash
|
|
501
|
+
|| (row.objective_status !== 'achieved' && row.objective_status !== 'not-achieved')
|
|
502
|
+
|| row.objective_status !== proof.receipt.objectiveStatus
|
|
503
|
+
|| row.situation !== claims.situation
|
|
504
|
+
|| !exactEvidence(storedEvidence, 'goal-outcome', subjectRef)
|
|
505
|
+
|| !exactEvidence(storedEvidence, 'acceptance-contract', proof.contract.id, proof.contract.digest)
|
|
506
|
+
|| !exactEvidence(storedEvidence, 'verification-receipt', proof.receipt.id, proof.receipt.digest)) {
|
|
507
|
+
throw new EvaluationStoreError('invalid-input', 'trusted goal outcome baseline does not match the current verifier receipt');
|
|
508
|
+
}
|
|
509
|
+
const existing = this.#database.prepare(`
|
|
510
|
+
SELECT outcome_id FROM evaluation_owner_revisions
|
|
511
|
+
WHERE subject_key = ? AND lineage = ? ORDER BY version DESC LIMIT 1
|
|
512
|
+
`).get(subject.key, lineage);
|
|
513
|
+
if (existing !== undefined) {
|
|
514
|
+
this.#database.exec('COMMIT');
|
|
515
|
+
return changed;
|
|
516
|
+
}
|
|
517
|
+
const foreignLineage = this.#database.prepare(`
|
|
518
|
+
SELECT lineage FROM evaluation_owner_revisions WHERE subject_key = ? LIMIT 1
|
|
519
|
+
`).get(subject.key);
|
|
520
|
+
if (foreignLineage !== undefined) {
|
|
521
|
+
throw new EvaluationStoreError('invalid-input', 'goal outcome baseline belongs to another owner lineage');
|
|
522
|
+
}
|
|
523
|
+
this.#database.prepare(`INSERT INTO evaluation_owner_revisions
|
|
524
|
+
(outcome_id, subject_key, lineage, version, previous_outcome_id, action, command_json)
|
|
525
|
+
VALUES (?, ?, ?, 1, NULL, 'initial', ?)`)
|
|
526
|
+
.run(row.id, subject.key, lineage, JSON.stringify({
|
|
527
|
+
action: 'goal-outcome-baseline-adoption',
|
|
528
|
+
assessmentId: subjectRef,
|
|
529
|
+
}));
|
|
530
|
+
changed = true;
|
|
531
|
+
this.#database.exec('COMMIT');
|
|
532
|
+
return changed;
|
|
533
|
+
}
|
|
534
|
+
catch (error) {
|
|
535
|
+
this.#database.exec('ROLLBACK');
|
|
536
|
+
throw error;
|
|
537
|
+
}
|
|
538
|
+
}
|
|
539
|
+
ownerObjectiveState(scope, subjectKindOrRef, subjectRefOrPrincipal, principalRecordIdOrVersion, principalVersion) {
|
|
540
|
+
// Arity, not the opaque legacy run id, selects the generic overload.
|
|
541
|
+
const generic = principalVersion !== undefined;
|
|
542
|
+
const subjectKind = generic ? subjectKindOrRef : 'automation-run';
|
|
543
|
+
const subjectRef = generic ? subjectRefOrPrincipal : subjectKindOrRef;
|
|
544
|
+
const principalRecordId = generic ? principalRecordIdOrVersion : subjectRefOrPrincipal;
|
|
545
|
+
const version = generic ? principalVersion : principalRecordIdOrVersion;
|
|
546
|
+
const subject = taskSubject(canonicalEvaluationScope(scope).scopeKey, '', [{ kind: subjectKind, ref: subjectRef }]);
|
|
547
|
+
const row = this.#database.prepare(`
|
|
548
|
+
SELECT revision.version, outcome.objective_status
|
|
549
|
+
FROM evaluation_owner_revisions revision JOIN evaluation_outcomes outcome ON outcome.id = revision.outcome_id
|
|
550
|
+
WHERE revision.subject_key = ? AND revision.lineage = ? ORDER BY revision.version DESC LIMIT 1
|
|
551
|
+
`).get(subject.key, JSON.stringify([principalRecordId, version]));
|
|
552
|
+
return row === undefined ? undefined : { version: row.version, objectiveStatus: row.objective_status };
|
|
553
|
+
}
|
|
554
|
+
ownerCommandResult(outcome) {
|
|
555
|
+
const row = this.#database.prepare(`SELECT version FROM evaluation_owner_revisions WHERE outcome_id = ?`)
|
|
556
|
+
.get(outcome.id);
|
|
557
|
+
return row === undefined ? outcome : { ...outcome, ownerFeedbackState: { version: row.version, objectiveStatus: outcome.objectiveStatus } };
|
|
558
|
+
}
|
|
559
|
+
append(input, ownerCommand) {
|
|
360
560
|
const normalized = this.#normalize(input);
|
|
361
561
|
const payloadHash = digest(normalized);
|
|
362
562
|
const id = `outcome-${randomUUID()}`;
|
|
363
563
|
const subject = taskSubject(normalized.scopeKey, id, normalized.evidence);
|
|
364
564
|
const recordedAt = timestamp(this.#now(), 'recordedAt');
|
|
365
565
|
const metric = normalized.metrics;
|
|
566
|
+
let committed = false;
|
|
366
567
|
this.#database.exec('BEGIN IMMEDIATE');
|
|
367
568
|
try {
|
|
569
|
+
let previous;
|
|
570
|
+
const commandHash = digest({ input: normalized, ownerCommand: ownerCommand ?? null });
|
|
571
|
+
if (ownerCommand !== undefined) {
|
|
572
|
+
const replay = this.#database.prepare(`SELECT * FROM evaluation_owner_commands WHERE operation_id = ?`)
|
|
573
|
+
.get(ownerCommand.operationId);
|
|
574
|
+
if (replay !== undefined) {
|
|
575
|
+
if (replay.payload_hash !== commandHash)
|
|
576
|
+
throw new EvaluationStoreError('idempotency-conflict', 'owner command identity reused');
|
|
577
|
+
if (replay.failure_code !== null)
|
|
578
|
+
throw new EvaluationStoreError(replay.failure_code === 'idempotency-conflict' ? 'idempotency-conflict' : 'version-conflict', 'owner judgement changed; inspect current feedback and retry');
|
|
579
|
+
const result = this.getOutcome(input.scope, replay.outcome_id);
|
|
580
|
+
this.#database.exec('COMMIT');
|
|
581
|
+
return this.ownerCommandResult(result);
|
|
582
|
+
}
|
|
583
|
+
previous = this.#database.prepare(`
|
|
584
|
+
SELECT revision.outcome_id, revision.version, outcome.objective_status
|
|
585
|
+
FROM evaluation_owner_revisions revision JOIN evaluation_outcomes outcome ON outcome.id = revision.outcome_id
|
|
586
|
+
WHERE revision.subject_key = ? AND revision.lineage = ? ORDER BY revision.version DESC LIMIT 1
|
|
587
|
+
`).get(subject.key, JSON.stringify([ownerCommand.principalRecordId, ownerCommand.principalVersion]));
|
|
588
|
+
if (ownerCommand.action !== 'initial' && (previous === undefined
|
|
589
|
+
|| previous.version !== ownerCommand.expectedVersion || previous.objective_status !== ownerCommand.previousStatus)) {
|
|
590
|
+
this.#database.prepare(`INSERT INTO evaluation_owner_commands VALUES (?, ?, NULL, 'version-conflict')`)
|
|
591
|
+
.run(ownerCommand.operationId, commandHash);
|
|
592
|
+
this.#database.exec('COMMIT');
|
|
593
|
+
committed = true;
|
|
594
|
+
throw new EvaluationStoreError('version-conflict', 'owner judgement changed; inspect current feedback and retry');
|
|
595
|
+
}
|
|
596
|
+
if (previous !== undefined && (ownerCommand.action === 'initial'
|
|
597
|
+
|| previous.objective_status === normalized.objectiveStatus)) {
|
|
598
|
+
if (previous.objective_status !== normalized.objectiveStatus) {
|
|
599
|
+
this.#database.prepare(`INSERT INTO evaluation_owner_commands VALUES (?, ?, NULL, 'idempotency-conflict')`)
|
|
600
|
+
.run(ownerCommand.operationId, commandHash);
|
|
601
|
+
this.#database.exec('COMMIT');
|
|
602
|
+
committed = true;
|
|
603
|
+
throw new EvaluationStoreError('idempotency-conflict', 'initial judgement already exists; use an explicit correction');
|
|
604
|
+
}
|
|
605
|
+
this.#database.prepare(`INSERT INTO evaluation_owner_commands VALUES (?, ?, ?, NULL)`)
|
|
606
|
+
.run(ownerCommand.operationId, commandHash, previous.outcome_id);
|
|
607
|
+
const result = this.getOutcome(input.scope, previous.outcome_id);
|
|
608
|
+
this.#database.exec('COMMIT');
|
|
609
|
+
return this.ownerCommandResult(result);
|
|
610
|
+
}
|
|
611
|
+
}
|
|
368
612
|
this.#database.prepare(`
|
|
369
613
|
INSERT INTO evaluation_outcomes(
|
|
370
614
|
id, idempotency_key, payload_hash, scope_key, workspace, preset, situation,
|
|
@@ -393,6 +637,12 @@ export class EvaluationStore {
|
|
|
393
637
|
VALUES (?, ?, ?, ?, ?)
|
|
394
638
|
ON CONFLICT(subject_key) DO NOTHING
|
|
395
639
|
`).run(winnerSubject.key, winner.scope_key, winnerSubject.kind, winnerSubject.ref, winner.recorded_at);
|
|
640
|
+
if (ownerCommand !== undefined) {
|
|
641
|
+
this.#database.prepare(`INSERT INTO evaluation_owner_revisions VALUES (?, ?, ?, ?, ?, ?, ?)`)
|
|
642
|
+
.run(winner.id, winnerSubject.key, JSON.stringify([ownerCommand.principalRecordId, ownerCommand.principalVersion]), (previous?.version ?? 0) + 1, previous?.outcome_id ?? null, ownerCommand.action, JSON.stringify(ownerCommand));
|
|
643
|
+
this.#database.prepare(`INSERT INTO evaluation_owner_commands VALUES (?, ?, ?, NULL)`)
|
|
644
|
+
.run(ownerCommand.operationId, commandHash, winner.id);
|
|
645
|
+
}
|
|
396
646
|
const refreshed = this.#refreshTaskProjection(winnerSubject.key);
|
|
397
647
|
if (winner.trust === 'trusted' && refreshed.learningVersionChanged) {
|
|
398
648
|
this.#database.prepare(`
|
|
@@ -405,10 +655,11 @@ export class EvaluationStore {
|
|
|
405
655
|
this.#advanceScopeWatermark(winner.scope_key, winner.recorded_at);
|
|
406
656
|
}
|
|
407
657
|
this.#database.exec('COMMIT');
|
|
408
|
-
return stored(winner);
|
|
658
|
+
return ownerCommand === undefined ? stored(winner) : this.ownerCommandResult(stored(winner));
|
|
409
659
|
}
|
|
410
660
|
catch (error) {
|
|
411
|
-
|
|
661
|
+
if (!committed)
|
|
662
|
+
this.#database.exec('ROLLBACK');
|
|
412
663
|
throw error;
|
|
413
664
|
}
|
|
414
665
|
}
|
|
@@ -573,6 +824,36 @@ export class EvaluationStore {
|
|
|
573
824
|
`).get(outcomeId, scopeKey);
|
|
574
825
|
return row === undefined ? undefined : projected(row);
|
|
575
826
|
}
|
|
827
|
+
/** Exact run lookup without an audit query limit or a second success definition. */
|
|
828
|
+
getAutomationRunLearningProjection(scopeInput, runIdInput) {
|
|
829
|
+
const { scopeKey } = canonicalEvaluationScope(scopeInput);
|
|
830
|
+
const runId = boundedText(runIdInput, 'runId', 200);
|
|
831
|
+
const row = this.#database.prepare(`
|
|
832
|
+
SELECT task.* FROM evaluation_task_projection_view task
|
|
833
|
+
WHERE task.scope_key = ? AND task.task_subject_kind = 'automation-run'
|
|
834
|
+
AND task.task_subject_ref = ?
|
|
835
|
+
`).get(scopeKey, runId);
|
|
836
|
+
if (row === undefined)
|
|
837
|
+
return undefined;
|
|
838
|
+
const task = projected(row);
|
|
839
|
+
if (task.projection.status !== 'ready')
|
|
840
|
+
return undefined;
|
|
841
|
+
return this.getTaskLearningProjection(scopeInput, task.projection.primaryOutcomeId);
|
|
842
|
+
}
|
|
843
|
+
/** Exact whole-goal lookup by the verifier assessment identity. */
|
|
844
|
+
getGoalOutcomeLearningProjection(scopeInput, assessmentIdInput) {
|
|
845
|
+
const { scopeKey } = canonicalEvaluationScope(scopeInput);
|
|
846
|
+
const assessmentId = boundedText(assessmentIdInput, 'assessmentId', 1_000);
|
|
847
|
+
const row = this.#database.prepare(`
|
|
848
|
+
SELECT task.* FROM evaluation_task_projection_view task
|
|
849
|
+
WHERE task.scope_key = ? AND task.task_subject_kind = 'goal-outcome'
|
|
850
|
+
AND task.task_subject_ref = ?
|
|
851
|
+
`).get(scopeKey, assessmentId);
|
|
852
|
+
if (row === undefined)
|
|
853
|
+
return undefined;
|
|
854
|
+
const task = projected(row);
|
|
855
|
+
return this.getTaskLearningProjection(scopeInput, task.projection.primaryOutcomeId);
|
|
856
|
+
}
|
|
576
857
|
/**
|
|
577
858
|
* Resolve an append-only outbox trigger to the latest canonical state of its
|
|
578
859
|
* task. The trigger may be arbitrarily old; version/digest always describe
|
|
@@ -638,9 +919,21 @@ export class EvaluationStore {
|
|
|
638
919
|
* order is Evaluation first, downstream second; a Promise-returning callback
|
|
639
920
|
* is rejected so the lock can never escape this stack frame.
|
|
640
921
|
*/
|
|
641
|
-
withLearningWriterFence(scopeInput, fenceInput, callback) {
|
|
922
|
+
withLearningWriterFence(scopeInput, fenceInput, callback, options = {}) {
|
|
923
|
+
const { scopeKey } = canonicalEvaluationScope(scopeInput);
|
|
924
|
+
const fence = this.#normalizeLearningWriterFence(fenceInput, false);
|
|
925
|
+
return this.#withLearningWriterFence(scopeKey, fence, callback, options);
|
|
926
|
+
}
|
|
927
|
+
/**
|
|
928
|
+
* Fence exact current canonical revisions, including retractions, without
|
|
929
|
+
* waiting for optional Evolution projection delivery.
|
|
930
|
+
*/
|
|
931
|
+
withCanonicalTaskWriterFence(scopeInput, fenceInput, callback) {
|
|
642
932
|
const { scopeKey } = canonicalEvaluationScope(scopeInput);
|
|
643
|
-
const fence = this.#normalizeLearningWriterFence(fenceInput);
|
|
933
|
+
const fence = this.#normalizeLearningWriterFence(fenceInput, true);
|
|
934
|
+
return this.#withLearningWriterFence(scopeKey, fence, callback, { requireProjectionDelivery: false });
|
|
935
|
+
}
|
|
936
|
+
#withLearningWriterFence(scopeKey, fence, callback, options) {
|
|
644
937
|
this.#database.exec('BEGIN IMMEDIATE');
|
|
645
938
|
try {
|
|
646
939
|
const watermark = this.#database.prepare(`
|
|
@@ -650,7 +943,7 @@ export class EvaluationStore {
|
|
|
650
943
|
this.#database.exec('COMMIT');
|
|
651
944
|
return Object.freeze({ matched: false, reason: 'watermark-changed' });
|
|
652
945
|
}
|
|
653
|
-
const pending = this.#database.prepare(`
|
|
946
|
+
const pending = options.requireProjectionDelivery === false ? undefined : this.#database.prepare(`
|
|
654
947
|
SELECT 1 AS present
|
|
655
948
|
FROM evaluation_projection_outbox outbox
|
|
656
949
|
JOIN evaluation_outcomes outcome ON outcome.id = outbox.evaluation_id
|
|
@@ -671,7 +964,7 @@ export class EvaluationStore {
|
|
|
671
964
|
for (const evidence of fence.evidence) {
|
|
672
965
|
const current = statement.get(scopeKey, evidence.subjectKind, evidence.subjectRef);
|
|
673
966
|
if (current === undefined || current.version !== evidence.version
|
|
674
|
-
|| current.digest !== evidence.digest || current.disposition !==
|
|
967
|
+
|| current.digest !== evidence.digest || current.disposition !== evidence.disposition) {
|
|
675
968
|
this.#database.exec('COMMIT');
|
|
676
969
|
return Object.freeze({ matched: false, reason: 'evidence-changed' });
|
|
677
970
|
}
|
|
@@ -982,9 +1275,13 @@ export class EvaluationStore {
|
|
|
982
1275
|
}
|
|
983
1276
|
else {
|
|
984
1277
|
const terminals = rows.filter(row => isAuthoritativeAutomationTerminal(row));
|
|
985
|
-
execution =
|
|
986
|
-
|
|
987
|
-
|
|
1278
|
+
execution = projection.subject_kind === 'automation-run'
|
|
1279
|
+
? latest(terminals)
|
|
1280
|
+
: latest(rows.filter(row => isAuthoritativeForegroundTerminal(row)));
|
|
1281
|
+
const superseded = new Set(this.#database.prepare(`
|
|
1282
|
+
SELECT previous_outcome_id FROM evaluation_owner_revisions WHERE subject_key = ? AND previous_outcome_id IS NOT NULL
|
|
1283
|
+
`).all(subjectKey).map(row => row.previous_outcome_id));
|
|
1284
|
+
const owners = rows.filter(row => isAuthenticatedOwnerFeedback(row) && !superseded.has(row.id));
|
|
988
1285
|
const ownerStatuses = new Set(owners.map(row => row.objective_status));
|
|
989
1286
|
if (ownerStatuses.size > 1)
|
|
990
1287
|
objectiveConflicted = true;
|
|
@@ -993,12 +1290,14 @@ export class EvaluationStore {
|
|
|
993
1290
|
else {
|
|
994
1291
|
const trustedObjectives = rows.filter(row => row.trust === 'trusted'
|
|
995
1292
|
&& row.source_kind !== 'user-feedback'
|
|
996
|
-
|
|
1293
|
+
// The newest verifier receipt is authoritative even when it is
|
|
1294
|
+
// unknown: it retracts a stale achieved proof after expiry/recheck.
|
|
1295
|
+
&& (row.objective_status !== 'unknown' || isTrustedVerifierReceipt(row)));
|
|
997
1296
|
const ranked = trustedObjectives.map(row => ({
|
|
998
1297
|
row,
|
|
999
1298
|
// Independent trusted evaluators supersede the terminal producer's
|
|
1000
1299
|
// initial objective, while an owner judgement has already won above.
|
|
1001
|
-
rank: isAuthoritativeAutomationTerminal(row) ? 1 : 2,
|
|
1300
|
+
rank: isAuthoritativeAutomationTerminal(row) ? 1 : isTrustedVerifierReceipt(row) ? 3 : 2,
|
|
1002
1301
|
})).sort((left, right) => left.rank - right.rank || newer(left.row, right.row));
|
|
1003
1302
|
objective = ranked.at(-1)?.row;
|
|
1004
1303
|
}
|
|
@@ -1009,7 +1308,7 @@ export class EvaluationStore {
|
|
|
1009
1308
|
delivery = ownerDelivery ?? trustedDelivery ?? execution;
|
|
1010
1309
|
const rankedPrimary = rows.map(row => ({
|
|
1011
1310
|
row,
|
|
1012
|
-
rank: isAuthoritativeAutomationTerminal(row)
|
|
1311
|
+
rank: isAuthoritativeAutomationTerminal(row) || isAuthoritativeForegroundTerminal(row)
|
|
1013
1312
|
? 4
|
|
1014
1313
|
: isAuthenticatedOwnerFeedback(row)
|
|
1015
1314
|
? 3
|
|
@@ -1082,7 +1381,7 @@ export class EvaluationStore {
|
|
|
1082
1381
|
}
|
|
1083
1382
|
return row.watermark;
|
|
1084
1383
|
}
|
|
1085
|
-
#normalizeLearningWriterFence(input) {
|
|
1384
|
+
#normalizeLearningWriterFence(input, allowRetract) {
|
|
1086
1385
|
if (typeof input !== 'object' || input === null || Array.isArray(input)
|
|
1087
1386
|
|| !Number.isSafeInteger(input.scopeWatermark) || input.scopeWatermark < 1
|
|
1088
1387
|
|| !Array.isArray(input.evidence) || input.evidence.length < 1
|
|
@@ -1092,8 +1391,8 @@ export class EvaluationStore {
|
|
|
1092
1391
|
const seen = new Set();
|
|
1093
1392
|
const entries = input.evidence.map((raw, index) => {
|
|
1094
1393
|
if (typeof raw !== 'object' || raw === null || Array.isArray(raw)
|
|
1095
|
-
|| raw.disposition !== 'upsert'
|
|
1096
|
-
|| (raw.subjectKind !== 'automation-run' && raw.subjectKind !== 'outcome')
|
|
1394
|
+
|| (raw.disposition !== 'upsert' && (!allowRetract || raw.disposition !== 'retract'))
|
|
1395
|
+
|| (raw.subjectKind !== 'automation-run' && raw.subjectKind !== 'foreground-turn' && raw.subjectKind !== 'goal-step' && raw.subjectKind !== 'goal-outcome' && raw.subjectKind !== 'outcome')
|
|
1097
1396
|
|| !Number.isSafeInteger(raw.version) || raw.version < 1
|
|
1098
1397
|
|| raw.version > 1_000_000_000
|
|
1099
1398
|
|| typeof raw.digest !== 'string' || !/^[a-f\d]{64}$/u.test(raw.digest)) {
|
|
@@ -1110,7 +1409,7 @@ export class EvaluationStore {
|
|
|
1110
1409
|
subjectRef,
|
|
1111
1410
|
version: raw.version,
|
|
1112
1411
|
digest: raw.digest,
|
|
1113
|
-
disposition:
|
|
1412
|
+
disposition: raw.disposition,
|
|
1114
1413
|
});
|
|
1115
1414
|
});
|
|
1116
1415
|
return Object.freeze({
|