@ngockhoale/ukit 3.3.3 → 3.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. package/CHANGELOG.md +40 -0
  2. package/manifests/engineConformance.yaml +17 -1
  3. package/manifests/hostCapabilities.yaml +68 -1
  4. package/manifests/platform.full.yaml +138 -0
  5. package/manifests/platform.user.yaml +255 -3
  6. package/package.json +1 -1
  7. package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
  8. package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
  9. package/scripts/probe/codex-capability-probe.mjs +169 -0
  10. package/src/cli/commands/doctor.js +168 -0
  11. package/src/cli/commands/indexTools.js +7 -0
  12. package/src/cli/commands/metrics.js +66 -2
  13. package/src/cli/commands/playbook.js +4 -4
  14. package/src/cli/commands/vm.js +49 -8
  15. package/src/core/agentRuntime/adapters.js +328 -27
  16. package/src/core/agentRuntime/artifacts.js +89 -0
  17. package/src/core/agentRuntime/context.js +345 -1
  18. package/src/core/agentRuntime/contract.js +296 -0
  19. package/src/core/agentRuntime/eventStore.js +176 -0
  20. package/src/core/agentRuntime/shadowRun.js +481 -5
  21. package/src/core/agentRuntime/telemetry.js +121 -0
  22. package/src/core/observability/emit/lifecycle.js +68 -1
  23. package/src/core/observability/emit/sessionBoot.js +393 -0
  24. package/src/core/observability/privacy/allowlist.js +10 -1
  25. package/src/core/observability/schema/registry.js +10 -0
  26. package/src/core/runtimeConfig.js +133 -0
  27. package/src/core/userPlaybooks.js +18 -3
  28. package/src/decision/registry.js +19 -0
  29. package/src/diagnostics/feedbackEvents.js +7 -4
  30. package/src/diagnostics/routeOutcomes.js +51 -6
  31. package/src/diagnostics/skillAccuracy.js +43 -3
  32. package/src/index/crossCheckMatrix.js +412 -0
  33. package/src/index/fixLoopEscalation.js +453 -0
  34. package/src/index/playbookRegistry.js +691 -0
  35. package/src/index/reviewPolicy.js +368 -0
  36. package/src/index/routeResolver.js +915 -0
  37. package/src/index/sessionHistoryExtractor.js +359 -0
  38. package/src/index/taskRouting.js +764 -581
  39. package/src/index/tierSelection.js +308 -0
  40. package/src/index/verificationMap.js +404 -0
  41. package/template_project/.claude/hooks/observability-emit.mjs +14 -0
  42. package/template_project/.claude/hooks/record-execution.mjs +19 -1
  43. package/template_project/.claude/hooks/skill-router.sh +691 -25
  44. package/template_project/.claude/hooks/verification-guard.sh +230 -1
  45. package/template_project/.claude/settings.json +2 -2
  46. package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
  47. package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
  48. package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
  49. package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
  50. package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
  51. package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
  52. package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
  53. package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
  54. package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
  55. package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
  56. package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
  57. package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
  58. package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
  59. package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
  60. package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
  61. package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
  62. package/template_project/.codex/README.md +8 -0
  63. package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
  64. package/template_project/ukit/README.md +1 -1
  65. package/template_project/ukit/storage/config.json +20 -0
  66. package/template_user/playbooks/architecture-decision.md +28 -0
  67. package/template_user/playbooks/autonomous-run.md +43 -0
  68. package/template_user/playbooks/autopilot-full.md +59 -0
  69. package/template_user/playbooks/autopilot-stack.md +54 -0
  70. package/template_user/playbooks/babysit.md +39 -0
  71. package/template_user/playbooks/bug-fix.md +3 -1
  72. package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
  73. package/template_user/playbooks/hillclimb.md +44 -0
  74. package/template_user/playbooks/investigation.md +21 -0
  75. package/template_user/playbooks/migration.md +21 -0
  76. package/template_user/playbooks/open-pr.md +48 -0
  77. package/template_user/playbooks/orchestrate.md +45 -0
  78. package/template_user/playbooks/performance.md +33 -0
  79. package/template_user/playbooks/prototype.md +28 -0
  80. package/template_user/playbooks/refactor.md +19 -0
  81. package/template_user/playbooks/release.md +28 -0
  82. package/template_user/playbooks/runtime-forensics.md +23 -0
  83. package/template_user/playbooks/session-pickup.md +31 -0
  84. package/template_user/playbooks/shipping.md +53 -0
  85. package/template_user/playbooks/skill-evaluation.md +48 -0
  86. package/template_user/playbooks/small-feature.md +20 -0
  87. package/template_user/playbooks/verification-map.json +153 -0
  88. package/template_user/playbooks/verification.md +22 -0
  89. package/template_user/playbooks/worktree-cleanup.md +37 -0
@@ -62,7 +62,7 @@ STATE_FILE="$PROJECT_ROOT/.claude/ukit/skill-router-state.json"
62
62
  ROUTE_CACHE_FILE="$PROJECT_ROOT/.claude/ukit/route-cache.json"
63
63
  PROGRESS_FILE="$PROJECT_ROOT/.claude/ukit/verification-progress.json"
64
64
 
65
- INPUT_FILE="$UKIT_INPUT_FILE" STATE_FILE="$STATE_FILE" ROUTE_CACHE_FILE="$ROUTE_CACHE_FILE" PROGRESS_FILE="$PROGRESS_FILE" UKIT_RUNTIME_DIR="$SCRIPT_DIR/../ukit/runtime" node <<'NODE'
65
+ INPUT_FILE="$UKIT_INPUT_FILE" STATE_FILE="$STATE_FILE" ROUTE_CACHE_FILE="$ROUTE_CACHE_FILE" PROGRESS_FILE="$PROGRESS_FILE" UKIT_RUNTIME_DIR="$SCRIPT_DIR/../ukit/runtime" PROJECT_ROOT="$PROJECT_ROOT" node <<'NODE'
66
66
  const HOOK_DEADLINE_MS = Number.parseInt(process.env.UKIT_HOOK_DEADLINE_MS || '', 10) || 3000;
67
67
  const LOCK_STARTED_AT = Date.now();
68
68
  // SPEC §8: a timed-out gate silently passes work it never evaluated — announce the
@@ -88,6 +88,7 @@ setTimeout(() => {
88
88
  const fs = require('fs').promises;
89
89
  const path = require('path');
90
90
  const { pathToFileURL } = require('url');
91
+ const { execFileSync } = require('child_process');
91
92
 
92
93
  async function readJson(filePath, fallback = null) {
93
94
  try {
@@ -403,6 +404,234 @@ if (typeof state?.ts === 'number' && stateAgeMs > STATE_FRESH_MS) {
403
404
  process.exit(0);
404
405
  }
405
406
 
407
+ // --- BL-010 (TASK-008): small-feature done-criteria receipt classes ------------
408
+ // On a `fastPathKind: small-feature` route the guard observes the two
409
+ // done-criteria receipt classes (wired-into-usage, surface-check) and records
410
+ // the latest status as an advisory `playbook-finding` receipt in the
411
+ // exec-ledger. Findings never block (P2 GO gate `false-block = 0`) — docs-only
412
+ // or non-feature routes skip the whole block.
413
+ const runtimeDir = process.env.UKIT_RUNTIME_DIR || '';
414
+ let ledgerModulePromise = null;
415
+ function loadLedgerModule() {
416
+ if (!ledgerModulePromise) {
417
+ ledgerModulePromise = (async () => {
418
+ const modulePath = runtimeDir ? path.join(runtimeDir, 'execution-ledger.mjs') : '';
419
+ if (!modulePath) return null;
420
+ try {
421
+ await fs.access(modulePath);
422
+ return await import(pathToFileURL(modulePath).href);
423
+ } catch {
424
+ return null;
425
+ }
426
+ })();
427
+ }
428
+ return ledgerModulePromise;
429
+ }
430
+
431
+ const NEW_SOURCE_FILE_RE = /\.(jsx?|tsx?|mjs|cjs|vue|svelte|css|scss|sass|less|html)$/i;
432
+
433
+ // `git status --porcelain` positions: untracked entries start '??', index-staged
434
+ // adds start 'A'. Both are "component-create" candidates for the wiring check.
435
+ function readGitChangedPaths(projectRoot) {
436
+ try {
437
+ const out = execFileSync('git', ['status', '--porcelain', '--untracked-files=all'], {
438
+ cwd: projectRoot,
439
+ encoding: 'utf8',
440
+ timeout: 1500,
441
+ stdio: ['ignore', 'pipe', 'ignore'],
442
+ });
443
+ return out.split('\n').map((line) => line.trim()).filter(Boolean);
444
+ } catch {
445
+ return null;
446
+ }
447
+ }
448
+
449
+ async function readSessionLedger(projectRoot, payload, ledgerModule) {
450
+ try {
451
+ if (ledgerModule && typeof ledgerModule.readExecutionLedger === 'function') {
452
+ return await ledgerModule.readExecutionLedger(projectRoot, payload);
453
+ }
454
+ } catch {}
455
+ try {
456
+ const sessionId = String(payload?.session_id || payload?.sessionId || 'default')
457
+ .trim().replace(/[^a-zA-Z0-9._-]/g, '_').slice(0, 96) || 'default';
458
+ return await readJson(path.join(
459
+ projectRoot, '.ukit', 'storage', 'cache', 'exec-ledger', `${sessionId}.json`,
460
+ ), null);
461
+ } catch {
462
+ return null;
463
+ }
464
+ }
465
+
466
+ async function observeSmallFeatureDoneCriteria({ projectRoot, payload, state }) {
467
+ const route = state?.routeSummary;
468
+ if (!route || (route.fastPathKind !== 'small-feature' && route.playbookId !== 'small-feature')) {
469
+ return;
470
+ }
471
+ const ledgerModule = await loadLedgerModule();
472
+ const ledger = await readSessionLedger(projectRoot, payload, ledgerModule);
473
+ const gitLines = readGitChangedPaths(projectRoot);
474
+ const newSourceFiles = gitLines === null ? [] : gitLines
475
+ .filter((line) => (line.startsWith('??') || line.startsWith('A')) && NEW_SOURCE_FILE_RE.test(line))
476
+ .map((line) => line.replace(/^(..)\s+/, ''))
477
+ // Rename lines carry "old -> new"; keep the destination only.
478
+ .map((entry) => entry.split(' -> ').pop());
479
+ const existingFileChanges = gitLines === null ? [] : gitLines
480
+ .filter((line) => !line.startsWith('??') && /^[MDRCTU]/.test(line));
481
+ const hasBuildEvidence = ledger?.writeAttempted === true || newSourceFiles.length > 0;
482
+ if (!hasBuildEvidence) return;
483
+
484
+ const observations = [];
485
+ // wired-into-usage: a created component with no change to any tracked file is
486
+ // the named cheat — nothing reaches it. Only observed when git answers; an
487
+ // absent repo yields no finding rather than a fabricated one.
488
+ if (gitLines !== null) {
489
+ if (newSourceFiles.length > 0 && existingFileChanges.length === 0) {
490
+ observations.push({
491
+ class: 'wired-into-usage',
492
+ status: 'missing',
493
+ detail: `new source file ${newSourceFiles[0]} with no tracked-file wiring edit`,
494
+ file: newSourceFiles[0],
495
+ });
496
+ } else if (newSourceFiles.length > 0) {
497
+ observations.push({
498
+ class: 'wired-into-usage',
499
+ status: 'satisfied',
500
+ detail: `new source file ${newSourceFiles[0]} plus ${existingFileChanges.length} tracked change(s)`,
501
+ file: newSourceFiles[0],
502
+ });
503
+ }
504
+ }
505
+ // surface-check: build-pass proves nothing — a verification receipt must land.
506
+ observations.push({
507
+ class: 'surface-check',
508
+ status: ledger?.verificationAttempted === true ? 'satisfied' : 'missing',
509
+ detail: ledger?.verificationAttempted === true
510
+ ? 'verification attempted on this ledger'
511
+ : 'no verification receipt recorded yet',
512
+ });
513
+
514
+ if (!ledgerModule || typeof ledgerModule.recordLedgerEvent !== 'function') return;
515
+ for (const observation of observations) {
516
+ try {
517
+ await ledgerModule.recordLedgerEvent(
518
+ {
519
+ type: 'receipt',
520
+ receipt: {
521
+ ts: Date.now(),
522
+ kind: 'playbook-finding',
523
+ success: true,
524
+ class: observation.class,
525
+ status: observation.status,
526
+ detail: observation.detail,
527
+ file: observation.file || null,
528
+ },
529
+ },
530
+ {
531
+ projectRoot,
532
+ payload,
533
+ harness: process.env.UKIT_HARNESS || 'claude',
534
+ routeState: state,
535
+ deadlineMs: Math.min(800, Math.max(200, HOOK_DEADLINE_MS - 400)),
536
+ },
537
+ );
538
+ } catch {
539
+ // Advisory finding — a failed write must never affect the command path.
540
+ }
541
+ }
542
+ }
543
+
544
+ await observeSmallFeatureDoneCriteria({
545
+ projectRoot: process.env.PROJECT_ROOT || process.cwd(),
546
+ payload,
547
+ state,
548
+ });
549
+
550
+ // --- BL-014 (TASK-C85-014): verification-map recipe findings ---------------
551
+ // The guard consults the same verification-map the stop gate consumes: the
552
+ // artifact class of the current change resolves a recipe whose
553
+ // evidenceRequired[] classes become `recipe-evidence` playbook-finding
554
+ // receipts (latest status banked per class). Unclassifiable diffs and
555
+ // missing/malformed maps yield NO recipe → existing behavior unchanged.
556
+ // Findings stay advisory — the guard never hard-blocks.
557
+ const UKIT_INDEX_DIR = path.join(runtimeDir || '.', '..', 'index');
558
+ let verificationMapModulePromise = null;
559
+ function loadVerificationMapModule() {
560
+ if (!verificationMapModulePromise) {
561
+ verificationMapModulePromise = (async () => {
562
+ const modulePath = path.join(UKIT_INDEX_DIR, 'verification-map.mjs');
563
+ try {
564
+ await fs.access(modulePath);
565
+ return await import(pathToFileURL(modulePath).href);
566
+ } catch {
567
+ return null;
568
+ }
569
+ })();
570
+ }
571
+ return verificationMapModulePromise;
572
+ }
573
+
574
+ async function observeVerificationRecipe({ projectRoot, payload, state }) {
575
+ const route = state?.routeSummary;
576
+ const playbookId = route?.playbookId || null;
577
+ if (!playbookId) return; // no playbook route → no recipe consult at all
578
+ const mapModule = await loadVerificationMapModule();
579
+ if (!mapModule) return;
580
+
581
+ const gitLines = readGitChangedPaths(projectRoot);
582
+ if (gitLines === null) return; // no git answer → no classification, no recipe
583
+ const changedPaths = gitLines.map((line) => line.replace(/^(..)\s+/, ''));
584
+ const artifactClass = mapModule.classifyArtifactClass(changedPaths);
585
+ if (!artifactClass) return; // unknown class → pre-map behavior (case 5)
586
+
587
+ const map = await mapModule.loadVerificationMap({ projectRoot });
588
+ if (!map) return; // absent/malformed map → fail closed, no recipe
589
+ const recipe = await mapModule.recipeForArtifact({ playbookId, artifactClass, map });
590
+ if (!recipe || !Array.isArray(recipe.evidenceRequired) || recipe.evidenceRequired.length === 0) {
591
+ return; // recipe resolved but demands nothing (e.g. docs)
592
+ }
593
+
594
+ const ledgerModule = await loadLedgerModule();
595
+ const ledger = await readSessionLedger(projectRoot, payload, ledgerModule);
596
+ if (!ledgerModule || typeof ledgerModule.recordLedgerEvent !== 'function') return;
597
+
598
+ for (const requiredClass of recipe.evidenceRequired) {
599
+ const satisfied = mapModule.receiptEvidenceSatisfied(requiredClass, ledger || {});
600
+ try {
601
+ await ledgerModule.recordLedgerEvent(
602
+ {
603
+ type: 'receipt',
604
+ receipt: {
605
+ ts: Date.now(),
606
+ kind: 'playbook-finding',
607
+ success: true,
608
+ class: requiredClass,
609
+ status: satisfied ? 'satisfied' : 'missing',
610
+ detail: satisfied
611
+ ? `recipe ${artifactClass}: receipt class ${requiredClass} satisfied`
612
+ : `recipe ${artifactClass}: receipt class ${requiredClass} not yet evidenced — check: ${recipe.check || 'run the recipe check'}${recipe.fallback ? `; fallback: ${recipe.fallback}` : ''}`,
613
+ },
614
+ },
615
+ {
616
+ projectRoot,
617
+ payload,
618
+ harness: process.env.UKIT_HARNESS || 'claude',
619
+ routeState: state,
620
+ deadlineMs: Math.min(800, Math.max(200, HOOK_DEADLINE_MS - 400)),
621
+ },
622
+ );
623
+ } catch {
624
+ // Advisory finding — a failed write must never affect the command path.
625
+ }
626
+ }
627
+ }
628
+
629
+ await observeVerificationRecipe({
630
+ projectRoot: process.env.PROJECT_ROOT || process.cwd(),
631
+ payload,
632
+ state,
633
+ });
634
+
406
635
  const primaryCommands = unique((
407
636
  routeSummary?.primaryCommands?.length
408
637
  ? routeSummary.primaryCommands
@@ -160,8 +160,8 @@
160
160
  "hooks": [
161
161
  {
162
162
  "type": "command",
163
- "command": "node \"$CLAUDE_PROJECT_DIR/.claude/ukit/runtime/hook-chain-runner.mjs\" --emit-verdict - \"$CLAUDE_PROJECT_DIR/.claude/hooks/project-important.sh\":8 \"$CLAUDE_PROJECT_DIR/.claude/hooks/auto-prune-bash.sh\":12 \"$CLAUDE_PROJECT_DIR/.claude/hooks/reset-compact-pressure.sh\":12 \"$CLAUDE_PROJECT_DIR/.claude/hooks/handoff-resume.sh\":8",
164
- "timeout": 50
163
+ "command": "node \"$CLAUDE_PROJECT_DIR/.claude/ukit/runtime/hook-chain-runner.mjs\" --emit-verdict - \"$CLAUDE_PROJECT_DIR/.claude/hooks/project-important.sh\":8 \"$CLAUDE_PROJECT_DIR/.claude/hooks/auto-prune-bash.sh\":12 \"$CLAUDE_PROJECT_DIR/.claude/hooks/reset-compact-pressure.sh\":12 \"$CLAUDE_PROJECT_DIR/.claude/hooks/handoff-resume.sh\":8 \"$CLAUDE_PROJECT_DIR/.claude/hooks/observability-emit.mjs\":8",
164
+ "timeout": 60
165
165
  }
166
166
  ]
167
167
  }
@@ -0,0 +1,415 @@
1
+ // cross-check-matrix.mjs — parity-locked mirror of src/index/crossCheckMatrix.js
2
+ // (BL-016 / SPEC FR-004 / ARCH §Adaptive Cross-Check). Mirrors never import
3
+ // src/; tests/consistency/crossCheckParity.test.js locks the export surface,
4
+ // the matrix rows, and identical {depth, reasons[]} on shared fixtures.
5
+ // Keep logic identical.
6
+ //
7
+ // Verification depth scales with risk instead of being uniform — a separate
8
+ // dimension from implementation depth (fast path vs full workflow). This
9
+ // module is deterministic and pure: it resolves
10
+ //
11
+ // change-risk × implementer-reliability × evidence-quality
12
+ // → none | runnable | review-round | escalate
13
+ //
14
+ // and the result feeds the review policy's `escalationDepth` input slot
15
+ // (src/index/reviewPolicy.js), which maps depth → action floor.
16
+ //
17
+ // Contract (SPEC §8):
18
+ // resolveCrossCheckDepth({riskFloor?, rigor?, implementerReliability?,
19
+ // evidenceQuality?, reviewer?, implementer?, runnable?, role?, model?,
20
+ // project?, overrides?}) → {depth, reasons[]}
21
+ //
22
+ // Invariants from ARCH §Adaptive Cross-Check:
23
+ // * high change-risk overrides implementer reliability — a strong model on
24
+ // an auth/data-path change still escalates to mandatory review.
25
+ // * a reviewer of the SAME model family is not independent review — a
26
+ // same-family re-read never satisfies a review-round; the depth demotes
27
+ // to `runnable` with the limit stated in reasons[].
28
+ // * no reviewer → `runnable` where possible + honest stated limit; the
29
+ // caller reports INCONCLUSIVE when nothing is runnable (the depth enum
30
+ // cannot return INCONCLUSIVE — it names a depth, not a verdict).
31
+ // * implementer-reliability is a joined outcome signal, never a model-name
32
+ // verdict — absent or unrecognized input resolves to baseline-neutral
33
+ // 'unknown' (GAP M13).
34
+ // * `crossCheck.{role|model}×{project}` maintainer overrides may tighten or
35
+ // loosen within the enum; malformed overrides are ignored, never fatal.
36
+ //
37
+ // Parity: tests/consistency/crossCheckParity.test.js locks this module and
38
+ // the installed mirror to identical exports, identical matrix rows, and
39
+ // identical {depth, reasons[]} on shared fixtures.
40
+
41
+ // Ordered by severity — override clamping and floor comparisons rely on the
42
+ // index order, so consumers must never reorder this list.
43
+ export const CROSS_CHECK_DEPTHS = Object.freeze([
44
+ 'none',
45
+ 'runnable',
46
+ 'review-round',
47
+ 'escalate',
48
+ ]);
49
+
50
+ export const CROSS_CHECK_RISK_LEVELS = Object.freeze(['low', 'medium', 'high']);
51
+
52
+ export const CROSS_CHECK_RELIABILITY_LEVELS = Object.freeze([
53
+ 'trusted',
54
+ 'needs-oversight',
55
+ 'unknown',
56
+ ]);
57
+
58
+ export const CROSS_CHECK_EVIDENCE_QUALITIES = Object.freeze([
59
+ 'strong',
60
+ 'thin',
61
+ 'missing-hooks',
62
+ ]);
63
+
64
+ // The ARCH §Adaptive Cross-Check matrix verbatim (7 rows). 'any' is a
65
+ // wildcard. First matching row wins, which gives the two 'any'-wildcard rows
66
+ // their precedence: high-risk escalate beats everything; missing-hooks
67
+ // degrade applies to whatever remains.
68
+ export const CROSS_CHECK_MATRIX = Object.freeze([
69
+ { risk: 'low', reliability: 'trusted', evidence: 'strong', depth: 'none' },
70
+ { risk: 'low', reliability: 'needs-oversight', evidence: 'thin', depth: 'review-round' },
71
+ { risk: 'low', reliability: 'unknown', evidence: 'strong', depth: 'runnable' },
72
+ { risk: 'medium', reliability: 'trusted', evidence: 'strong', depth: 'runnable' },
73
+ { risk: 'medium', reliability: 'needs-oversight', evidence: 'any', depth: 'review-round' },
74
+ { risk: 'high', reliability: 'any', evidence: 'any', depth: 'escalate' },
75
+ { risk: 'any', reliability: 'any', evidence: 'missing-hooks', depth: 'runnable' },
76
+ ]);
77
+
78
+ // Cells the matrix does not list resolve conservatively: needs-oversight on
79
+ // a low/medium-risk change is the row-2/row-5 signal family, so it still
80
+ // earns a review round; trusted and baseline-neutral reliability never drop
81
+ // below `runnable` on unlisted cells — thin evidence can never justify `none`.
82
+ const RELIABILITY_RESIDUAL = Object.freeze({
83
+ 'needs-oversight': 'review-round',
84
+ });
85
+
86
+ // Bounded maintainer override shapes: a bound leaf carries only these keys
87
+ // (validated in src/core/runtimeConfig.js); all other top-level keys of the
88
+ // section are reserved.
89
+ const BOUND_KEYS = new Set(['depth', 'minDepth', 'maxDepth']);
90
+ const CROSS_CHECK_SECTION_KEYS = new Set(['enabled', 'byRole', 'byModel']);
91
+
92
+ const MAX_ID_LENGTH = 96;
93
+
94
+ const RISK_ALIASES = Object.freeze({
95
+ high: 'high',
96
+ 'high-risk': 'high',
97
+ critical: 'high',
98
+ medium: 'medium',
99
+ moderate: 'medium',
100
+ low: 'low',
101
+ none: 'low',
102
+ trivial: 'low',
103
+ });
104
+
105
+ const RELIABILITY_ALIASES = Object.freeze({
106
+ trusted: 'trusted',
107
+ 'needs-oversight': 'needs-oversight',
108
+ unknown: 'unknown',
109
+ baseline: 'unknown',
110
+ 'baseline-neutral': 'unknown',
111
+ });
112
+
113
+ // Receipt classes attesting real-run/surface evidence vs build-only or
114
+ // claimed-only work (ARCH: evidence-quality is about receipt classes present,
115
+ // never about what the agent asserts).
116
+ const STRONG_EVIDENCE_TOKENS = new Set([
117
+ 'strong',
118
+ 'surface-check',
119
+ 'render-observation',
120
+ 'io-run',
121
+ 'run-receipt',
122
+ 'real-run',
123
+ 'verification-evidence',
124
+ 'write-evidence',
125
+ ]);
126
+ const THIN_EVIDENCE_TOKENS = new Set([
127
+ 'thin',
128
+ 'build-only',
129
+ 'claimed',
130
+ 'claimed-only',
131
+ 'log',
132
+ ]);
133
+ const MISSING_HOOKS_TOKENS = new Set([
134
+ 'missing-hooks',
135
+ 'unavailable',
136
+ 'cannot-produce',
137
+ ]);
138
+
139
+ function isPlainObject(value) {
140
+ return Boolean(value) && typeof value === 'object' && !Array.isArray(value);
141
+ }
142
+
143
+ function depthIndex(depth) {
144
+ return CROSS_CHECK_DEPTHS.indexOf(depth);
145
+ }
146
+
147
+ function boundedId(value) {
148
+ return typeof value === 'string' && value.trim()
149
+ ? value.trim().slice(0, MAX_ID_LENGTH)
150
+ : null;
151
+ }
152
+
153
+ // ─── input normalization ────────────────────────────────────────────────────
154
+
155
+ // riskFloor arrives as the route record's {floor:'none'|'high-risk', codes[]},
156
+ // a bare enum/code string, or a level name. `rigor` (R0-R4) only ever RAISES
157
+ // the floor: R3+ ceremony on a non-floor route still counts as medium risk —
158
+ // the router asked for deep work, so verification follows.
159
+ function normalizeRisk(riskFloor, rigor) {
160
+ let risk = 'low';
161
+ if (typeof riskFloor === 'string') {
162
+ risk = RISK_ALIASES[riskFloor.trim().toLowerCase()] ?? 'low';
163
+ } else if (isPlainObject(riskFloor)) {
164
+ const floor = typeof riskFloor.floor === 'string'
165
+ ? riskFloor.floor.trim().toLowerCase()
166
+ : null;
167
+ if (RISK_ALIASES[floor]) {
168
+ risk = RISK_ALIASES[floor];
169
+ } else if (typeof riskFloor.risk === 'string') {
170
+ risk = RISK_ALIASES[riskFloor.risk.trim().toLowerCase()] ?? 'low';
171
+ }
172
+ }
173
+ if (risk === 'low' && typeof rigor === 'string') {
174
+ const level = /^r(\d)$/i.exec(rigor.trim());
175
+ if (level && Number(level[1]) >= 3) risk = 'medium';
176
+ }
177
+ return risk;
178
+ }
179
+
180
+ // Reliability is joined outcome data, never a model-name verdict: anything
181
+ // absent or unrecognized collapses to baseline-neutral 'unknown'.
182
+ function normalizeReliability(value) {
183
+ if (typeof value !== 'string') return 'unknown';
184
+ return RELIABILITY_ALIASES[value.trim().toLowerCase()] ?? 'unknown';
185
+ }
186
+
187
+ function normalizeEvidenceQuality(value) {
188
+ if (value == null) return 'thin';
189
+ if (Array.isArray(value)) {
190
+ if (value.length === 0) return 'thin';
191
+ const tokens = value
192
+ .filter((token) => typeof token === 'string' && token.trim())
193
+ .map((token) => token.trim().toLowerCase());
194
+ if (tokens.some((token) => MISSING_HOOKS_TOKENS.has(token))) return 'missing-hooks';
195
+ if (tokens.some((token) => STRONG_EVIDENCE_TOKENS.has(token))) return 'strong';
196
+ return 'thin';
197
+ }
198
+ if (typeof value === 'string') {
199
+ const token = value.trim().toLowerCase();
200
+ if (MISSING_HOOKS_TOKENS.has(token)) return 'missing-hooks';
201
+ if (STRONG_EVIDENCE_TOKENS.has(token)) return 'strong';
202
+ return 'thin';
203
+ }
204
+ if (isPlainObject(value)) {
205
+ if (typeof value.quality === 'string') return normalizeEvidenceQuality(value.quality);
206
+ if (Array.isArray(value.classes)) return normalizeEvidenceQuality(value.classes);
207
+ }
208
+ return 'thin';
209
+ }
210
+
211
+ // ─── reviewer independence ──────────────────────────────────────────────────
212
+ // A review-round only counts when a reviewer of a DIFFERENT model family
213
+ // exists — same-family re-reads and self re-reads are not independent review
214
+ // (SPEC FR-004). A reviewer descriptor that supplies no family cannot be
215
+ // proven same or different; it is treated as independent but the assumption
216
+ // is named in the reasons so the receipt stays honest.
217
+ function reviewerIndependence(reviewer, implementer) {
218
+ if (!isPlainObject(reviewer)) {
219
+ return { independent: false, reason: 'no reviewer available' };
220
+ }
221
+ const reviewerFamily = boundedId(reviewer.modelFamily) ?? boundedId(reviewer.family);
222
+ const reviewerModel = boundedId(reviewer.model) ?? boundedId(reviewer.id) ?? boundedId(reviewer.name);
223
+ const implementerFamily = isPlainObject(implementer)
224
+ ? (boundedId(implementer.modelFamily) ?? boundedId(implementer.family))
225
+ : null;
226
+ const implementerModel = isPlainObject(implementer)
227
+ ? (boundedId(implementer.model) ?? boundedId(implementer.id) ?? boundedId(implementer.name))
228
+ : null;
229
+ if (reviewerModel && implementerModel && reviewerModel === implementerModel) {
230
+ return { independent: false, reason: 'reviewer is the implementer re-reading its own diff — not independent' };
231
+ }
232
+ if (reviewerFamily && implementerFamily && reviewerFamily === implementerFamily) {
233
+ return { independent: false, reason: 'same-model-family reviewer is not independent review' };
234
+ }
235
+ return {
236
+ independent: true,
237
+ reason: reviewerFamily
238
+ ? 'different-model-family reviewer available'
239
+ : 'reviewer family unproven — assuming independent',
240
+ };
241
+ }
242
+
243
+ // ─── maintainer overrides ───────────────────────────────────────────────────
244
+ // `crossCheck.{role|model}×{project}` bounded overrides, per ARCH §Config:
245
+ // enum-only, shipped ON. A bound leaf is {depth|minDepth|maxDepth}; an entry
246
+ // may also be a project map {projectId|'*': bound}. byRole wins over byModel;
247
+ // a project-named leaf wins over the '*' leaf of the same entry. Unknown
248
+ // shapes are ignored with a reason — a maintainer typo must never crash the
249
+ // route nor invent a depth outside the enum.
250
+ function boundLeaf(entry) {
251
+ if (!isPlainObject(entry)) return null;
252
+ if (Object.keys(entry).some((key) => BOUND_KEYS.has(key))) return entry;
253
+ return null;
254
+ }
255
+
256
+ function resolveOverrideBound(overrides, { role = null, model = null, project = null }) {
257
+ if (!isPlainObject(overrides) || overrides.enabled === false) return null;
258
+ const maps = [
259
+ ['role', role, overrides.byRole],
260
+ ['model', model, overrides.byModel],
261
+ ];
262
+ for (const [kind, name, map] of maps) {
263
+ if (!name || !isPlainObject(map)) continue;
264
+ const entry = map[name];
265
+ if (entry === undefined) continue;
266
+ const direct = boundLeaf(entry);
267
+ if (direct) {
268
+ return { bound: direct, label: `${kind} "${name}"` };
269
+ }
270
+ if (isPlainObject(entry)) {
271
+ const projectLeaf = project ? boundLeaf(entry[project]) : null;
272
+ const starLeaf = boundLeaf(entry['*']);
273
+ const picked = projectLeaf ?? starLeaf;
274
+ if (picked) {
275
+ return {
276
+ bound: picked,
277
+ label: `${kind} "${name}"${projectLeaf ? ` project "${project}"` : ''}`,
278
+ };
279
+ }
280
+ }
281
+ return { bound: null, label: `${kind} "${name}"` };
282
+ }
283
+ // A bare bound ({depth:…} / {minDepth:…} / {maxDepth:…}) applies directly.
284
+ const direct = boundLeaf(overrides);
285
+ if (direct) return { bound: direct, label: 'crossCheck' };
286
+ return null;
287
+ }
288
+
289
+ // ─── resolver ───────────────────────────────────────────────────────────────
290
+
291
+ export function resolveCrossCheckDepth({
292
+ riskFloor = null,
293
+ rigor = null,
294
+ implementerReliability = null,
295
+ evidenceQuality = null,
296
+ reviewer = undefined,
297
+ implementer = null,
298
+ runnable = true,
299
+ role = null,
300
+ model = null,
301
+ project = null,
302
+ overrides = null,
303
+ } = {}) {
304
+ const reasons = [];
305
+ const risk = normalizeRisk(riskFloor, rigor);
306
+ const reliability = normalizeReliability(implementerReliability);
307
+ const evidence = normalizeEvidenceQuality(evidenceQuality);
308
+
309
+ if (implementerReliability == null
310
+ || (typeof implementerReliability === 'string'
311
+ && RELIABILITY_ALIASES[implementerReliability.trim().toLowerCase()] === undefined)) {
312
+ reasons.push(
313
+ 'implementerReliability unresolved — defaulting to project-baseline neutral '
314
+ + '("unknown"); reliability is a joined outcome signal, never a model-name verdict',
315
+ );
316
+ }
317
+
318
+ let depth = null;
319
+ let decidingRow = null;
320
+ for (const row of CROSS_CHECK_MATRIX) {
321
+ if ((row.risk === 'any' || row.risk === risk)
322
+ && (row.reliability === 'any' || row.reliability === reliability)
323
+ && (row.evidence === 'any' || row.evidence === evidence)) {
324
+ depth = row.depth;
325
+ decidingRow = row;
326
+ break;
327
+ }
328
+ }
329
+
330
+ if (depth) {
331
+ const index = CROSS_CHECK_MATRIX.indexOf(decidingRow) + 1;
332
+ reasons.push(
333
+ `matrix row ${index}: risk=${decidingRow.risk === 'any' ? risk : decidingRow.risk} × `
334
+ + `reliability=${decidingRow.reliability === 'any' ? reliability : decidingRow.reliability} × `
335
+ + `evidence=${decidingRow.evidence === 'any' ? evidence : decidingRow.evidence} → ${decidingRow.depth}`,
336
+ );
337
+ if (decidingRow.risk === 'high') {
338
+ reasons.push('high change-risk overrides implementer reliability — mandatory review regardless of track record');
339
+ }
340
+ if (decidingRow.evidence === 'missing-hooks') {
341
+ reasons.push(
342
+ 'engine cannot produce a needed receipt class (missing hooks) — runnable '
343
+ + 'checks where possible + honest stated limit; INCONCLUSIVE if nothing runnable',
344
+ );
345
+ }
346
+ } else {
347
+ depth = RELIABILITY_RESIDUAL[reliability] ?? 'runnable';
348
+ reasons.push(
349
+ `no verbatim matrix row for risk=${risk} × reliability=${reliability} × evidence=${evidence} — `
350
+ + `residual rule resolves ${depth} (needs-oversight still earns a review round; `
351
+ + 'trusted/baseline never drops below runnable)',
352
+ );
353
+ }
354
+
355
+ // Reviewer independence only demotes the review-round depth: the matrix
356
+ // chose it for the signal, but without a different-family reviewer a
357
+ // same-family or self re-read is not independent review. 'escalate' is a
358
+ // mandatory hand-off — it never silently downgrades; the reason names the
359
+ // missing reviewer so the consumer escalates to a human or another lane.
360
+ const independence = reviewerIndependence(reviewer, implementer);
361
+ if (depth === 'review-round' && !independence.independent) {
362
+ reasons.push(`${independence.reason} — review-round demoted to runnable checks + stated limit`);
363
+ depth = 'runnable';
364
+ if (runnable === false) {
365
+ reasons.push('nothing runnable — report INCONCLUSIVE');
366
+ }
367
+ } else if (depth === 'review-round') {
368
+ reasons.push(independence.reason);
369
+ } else if (depth === 'escalate' && !independence.independent) {
370
+ reasons.push(
371
+ `${independence.reason} — escalate stands: mandatory review requires a `
372
+ + 'different-family reviewer or human sign-off, never a silent downgrade',
373
+ );
374
+ }
375
+
376
+ // Maintainer overrides bound the resolved depth last. A 'depth' pin wins;
377
+ // minDepth/maxDepth clamp within the enum. Invalid bound values are ignored
378
+ // with a named reason — schema validation in runtimeConfig.js rejects them
379
+ // at parse time; this guard covers programmatic callers.
380
+ const resolved = resolveOverrideBound(overrides, { role, model, project });
381
+ if (resolved && resolved.bound) {
382
+ const bound = resolved.bound;
383
+ let applied = null;
384
+ if (typeof bound.depth === 'string' && CROSS_CHECK_DEPTHS.includes(bound.depth)) {
385
+ depth = bound.depth;
386
+ applied = `depth=${bound.depth}`;
387
+ } else {
388
+ if (typeof bound.maxDepth === 'string'
389
+ && CROSS_CHECK_DEPTHS.includes(bound.maxDepth)
390
+ && depthIndex(depth) > depthIndex(bound.maxDepth)) {
391
+ depth = bound.maxDepth;
392
+ applied = `maxDepth=${bound.maxDepth}`;
393
+ }
394
+ if (typeof bound.minDepth === 'string'
395
+ && CROSS_CHECK_DEPTHS.includes(bound.minDepth)
396
+ && depthIndex(depth) < depthIndex(bound.minDepth)) {
397
+ depth = bound.minDepth;
398
+ applied = applied ? `${applied} + minDepth=${bound.minDepth}` : `minDepth=${bound.minDepth}`;
399
+ }
400
+ }
401
+ if (applied) {
402
+ reasons.push(`crossCheck override (${resolved.label}) applied: ${applied} → ${depth}`);
403
+ } else {
404
+ reasons.push(`crossCheck override (${resolved.label}) present but no enum bound applied`);
405
+ }
406
+ } else if (resolved) {
407
+ reasons.push(`crossCheck override (${resolved.label}) has no valid enum bound — ignored`);
408
+ } else if (isPlainObject(overrides) && Object.keys(overrides).some(
409
+ (key) => !CROSS_CHECK_SECTION_KEYS.has(key) && !BOUND_KEYS.has(key),
410
+ )) {
411
+ reasons.push('crossCheck override keys unrecognized — ignored');
412
+ }
413
+
414
+ return { depth, reasons };
415
+ }