@ngockhoale/ukit 3.0.1 → 3.0.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -70,6 +70,7 @@ export const ROUTE_MUTABILITIES = new Set(['read-only', 'mutating', 'mixed']);
70
70
  export const ROUTE_RIGOR_LEVELS = new Set(['R0', 'R1', 'R2', 'R3', 'R4']);
71
71
  export const ROUTE_MODEL_TIERS = new Set(['lite', 'code', 'smart']);
72
72
  export const ROUTE_RISK_FLOORS = new Set(['none', 'high-risk']);
73
+ export const ROUTE_EFFORTS = new Set(['low', 'medium', 'high']);
73
74
  // SPEC §5 FR-003: fixed table order — codes are emitted and printed in this order.
74
75
  // The first six raise the floor to 'high-risk'; the last two are informational only.
75
76
  export const ROUTE_RISK_REASON_CODES = Object.freeze([
@@ -354,6 +355,37 @@ function formatRiskFloorSegment(riskFloor = null) {
354
355
  return printed.length > 0 ? `risk=${riskFloor.floor}(${printed.join(',')})` : null;
355
356
  }
356
357
 
358
+ // FR-001 (M01.2' limits fragment): compile the contract's numeric budget keys
359
+ // into an advisory map. Only finite numbers survive — a non-numeric or missing
360
+ // key is simply absent. Zero is a real budget ("no read passes"), never
361
+ // filtered. Same key set as the ceremonyBudget.limits builder below.
362
+ export function deriveCeremonyLimits(executionContract = null) {
363
+ if (executionContract === null || typeof executionContract !== 'object') {
364
+ return {};
365
+ }
366
+ return Object.fromEntries(
367
+ ['maxReadPasses', 'maxContextPulls', 'maxReadPassesBeforeReassess']
368
+ .filter((key) => Number.isFinite(executionContract[key]))
369
+ .map((key) => [key, executionContract[key]]),
370
+ );
371
+ }
372
+
373
+ // Route-line segment (FR-002): fixed reads,ctx,reassess order; only present
374
+ // keys print. Empty/absent map → null so the segment never appears.
375
+ export function formatLimitsSegment(limits = null) {
376
+ if (limits === null || typeof limits !== 'object') {
377
+ return null;
378
+ }
379
+ const parts = [
380
+ ['reads', 'maxReadPasses'],
381
+ ['ctx', 'maxContextPulls'],
382
+ ['reassess', 'maxReadPassesBeforeReassess'],
383
+ ]
384
+ .filter(([, key]) => Number.isFinite(limits[key]))
385
+ .map(([label, key]) => `${label}:${limits[key]}`);
386
+ return parts.length > 0 ? `limits=${parts.join(',')}` : null;
387
+ }
388
+
357
389
  // FR-004 (M01.3'): Fast Path eligibility predicate. Returns null when no riskFloor
358
390
  // was supplied — eligibility must never be derived without the floor check, so a
359
391
  // missing floor means "not computed", not "none". Eligible iff the lane is
@@ -460,6 +492,10 @@ function buildResolvedRouteFields({
460
492
  completionState = null,
461
493
  riskFloor = null,
462
494
  } = {}) {
495
+ // Decision table v2 (FR-001/FR-002): tier + effort resolve together from the
496
+ // contract lane and the additive riskFloor. The router cannot observe host
497
+ // binding capabilities, so the emitted pair is advisory text by definition.
498
+ const tierDecision = resolveModelTier({ executionMode, riskFloor });
463
499
  const signalText = buildNormalizedRouteSignalText(routingContext.promptText, routingContext.commandText);
464
500
  const deliveryOnly = isDeliveryOnlyRequest({
465
501
  signalText,
@@ -484,7 +520,8 @@ function buildResolvedRouteFields({
484
520
  riskFloor: riskFloor?.floor ?? null,
485
521
  phase: null,
486
522
  contractVersion: ROUTE_CONTRACT_VERSION,
487
- modelTier: executionContract?.modelTier ?? null,
523
+ modelTier: tierDecision.tier,
524
+ effort: tierDecision.effort,
488
525
  },
489
526
  evidence: {
490
527
  observations: [],
@@ -567,6 +604,10 @@ export function validateResolvedRoute(route = null) {
567
604
  if (route.execution.modelTier !== null && !ROUTE_MODEL_TIERS.has(route.execution.modelTier)) {
568
605
  errors.push(`execution.modelTier must be null or one of: ${[...ROUTE_MODEL_TIERS].join(', ')}.`);
569
606
  }
607
+ if (route.execution.effort !== null && route.execution.effort !== undefined
608
+ && !ROUTE_EFFORTS.has(route.execution.effort)) {
609
+ errors.push(`execution.effort must be null or one of: ${[...ROUTE_EFFORTS].join(', ')}.`);
610
+ }
570
611
  }
571
612
  if (!isObject(route.evidence)) {
572
613
  errors.push('evidence must be an object.');
@@ -2879,6 +2920,12 @@ function buildRouteSummary({
2879
2920
  })
2880
2921
  : null;
2881
2922
  const riskSegment = formatRiskFloorSegment(riskFloor);
2923
+ // FR-003 (M01.2' limits fragment): the rigor advisory — emitted whenever
2924
+ // rigor.stage is on, independent of fastPath/escalation. Stage off → no
2925
+ // segment, byte-identical route.
2926
+ const limitsSegment = resolveRouteStage(runtimeConfig, 'rigor') !== 'off'
2927
+ ? formatLimitsSegment(deriveCeremonyLimits(executionContract))
2928
+ : null;
2882
2929
  // FR-004 (M01.3'): fastPath is emitted whenever its own stage is on — the field
2883
2930
  // is always set then (eligible or not) so telemetry/harness can read it; the
2884
2931
  // route-line segment prints only for eligible routes.
@@ -2951,6 +2998,7 @@ function buildRouteSummary({
2951
2998
  formatCompactSegment('styles', styleFiles),
2952
2999
  editGuardHint ? `editGuard=${editGuardHint}` : null,
2953
3000
  riskSegment,
3001
+ limitsSegment,
2954
3002
  fastPathSegment,
2955
3003
  delegationRecommendation?.hint ? `delegate=${delegationRecommendation.hint}` : null,
2956
3004
  policyMode ? `policy=${policyMode}` : null,
@@ -3442,6 +3490,45 @@ function buildExecutionContract(executionMode = null) {
3442
3490
  return contracts[executionMode] ? { ...contracts[executionMode] } : null;
3443
3491
  }
3444
3492
 
3493
+ // Decision table v2 (V3_RESHAPE §5): the contracts table's modelTier column is the
3494
+ // base row; resolveModelTier adds the (riskFloor, hostCapabilities) columns.
3495
+ // Deterministic — no registry, no leases, no outbound capability calls.
3496
+ // Literal mirror of src/core/executionContracts.js resolveModelTier; parity is
3497
+ // locked by tests/consistency/executionContractSync.test.js.
3498
+ const MODEL_TIER_ORDER = ['lite', 'code', 'smart'];
3499
+ const MODEL_TIER_EFFORT = {
3500
+ lite: 'low',
3501
+ code: 'medium',
3502
+ smart: 'high',
3503
+ };
3504
+
3505
+ /**
3506
+ * Resolves {tier, effort, advisoryOnly} for a route. A 'high-risk' floor escalates
3507
+ * the tier one band (lite→code→smart, capped at smart) and forces effort 'high'.
3508
+ * Unknown mode → {tier:null, effort:null, advisoryOnly:true}. Missing/unknown
3509
+ * riskFloor → 'none'. Missing hostCapabilities → advisoryOnly:true (the route text
3510
+ * is all the host gets when it cannot bind model and effort).
3511
+ */
3512
+ export function resolveModelTier({
3513
+ executionMode = null,
3514
+ riskFloor = null,
3515
+ hostCapabilities = null,
3516
+ } = {}) {
3517
+ const baseTier = buildExecutionContract(executionMode)?.modelTier ?? null;
3518
+ if (!baseTier) {
3519
+ return { tier: null, effort: null, advisoryOnly: true };
3520
+ }
3521
+ const highRisk = riskFloor?.floor === 'high-risk';
3522
+ const tier = highRisk
3523
+ ? MODEL_TIER_ORDER[Math.min(MODEL_TIER_ORDER.indexOf(baseTier) + 1, MODEL_TIER_ORDER.length - 1)]
3524
+ : baseTier;
3525
+ return {
3526
+ tier,
3527
+ effort: highRisk ? 'high' : MODEL_TIER_EFFORT[tier],
3528
+ advisoryOnly: !(hostCapabilities?.canBindModel === true && hostCapabilities?.canBindEffort === true),
3529
+ };
3530
+ }
3531
+
3445
3532
  function buildApproachSelectorResult({
3446
3533
  executionMode = null,
3447
3534
  executionScores = null,
@@ -3726,7 +3813,8 @@ export function computeRiskEscalation({
3726
3813
  previousRouteSummary = null,
3727
3814
  config = null,
3728
3815
  } = {}) {
3729
- if (resolveRouteStage(config, 'escalation') === 'off') {
3816
+ const stage = resolveRouteStage(config, 'escalation');
3817
+ if (stage === 'off') {
3730
3818
  return null;
3731
3819
  }
3732
3820
  const sourceFlags = config?.routing?.escalation?.sources ?? {};
@@ -3747,7 +3835,7 @@ export function computeRiskEscalation({
3747
3835
  const level = RISK_ESCALATION_LEVELS[derivedLevel] >= previousLevel
3748
3836
  ? derivedLevel
3749
3837
  : previous.level;
3750
- return { level, sources };
3838
+ return { level, sources, stage };
3751
3839
  }
3752
3840
 
3753
3841
  // FR-005: buildRouteSummary runs before riskEscalation exists, so the
@@ -1634,7 +1634,10 @@ function requiredEvidence(state = {}) {
1634
1634
  // FR-006 (M01.4' tail): a route persisted with a high riskEscalation must re-verify at
1635
1635
  // Stop even when its own contract only demands write evidence. The route state is the
1636
1636
  // carrier — no config read here; absent/malformed/non-high fields are a no-op.
1637
- if (routeSummary.riskEscalation?.level === 'high') {
1637
+ // C48 FR-004: a record stamped `stage: 'shadow'` is advisory-only and does not demand
1638
+ // verification-evidence; a missing or malformed stage fails closed (enforcing) so
1639
+ // pre-C48 persisted records keep re-verifying until re-stamped.
1640
+ if (routeSummary.riskEscalation?.level === 'high' && routeSummary.riskEscalation?.stage !== 'shadow') {
1638
1641
  required.push('verification-evidence');
1639
1642
  }
1640
1643
  return [...new Set(required)];
@@ -16,10 +16,14 @@
16
16
  * `docs/AI_HANDOFF/RUN.md` carries `Phase:` other than `done`/`blocked`, a
17
17
  * handoff-fullstack run is still in flight and the stop is bounced back with
18
18
  * the cursor's `Next:` step — this is what keeps an overnight run moving
19
- * after a recap instead of stalling idle. A stalled-cursor breaker releases
20
- * the stop if the cursor has not advanced across `stopGateMaxStalledBlocks`
21
- * consecutive blocked stops (same liveness shape as the ledger's
22
- * noProgressCount breaker);
19
+ * after a recap instead of stalling idle. When RUN.md carries an
20
+ * `ExitPredicate:` line (mechanic #7), `Phase: done` is accepted only if the
21
+ * predicate evaluates true — `all(index:all-done, git:clean,
22
+ * file-exists:<relpath>)` terms checked mechanically, no eval, no shell-out.
23
+ * A stalled-cursor breaker releases the stop if the cursor has not advanced
24
+ * across `stopGateMaxStalledBlocks` consecutive blocked stops (same liveness
25
+ * shape as the ledger's noProgressCount breaker), and its advisory reports
26
+ * the predicate state so a plateau is visible;
23
27
  * - it merges the three evaluator results with a deterministic owner order —
24
28
  * fail-closed completion failure > completion block > handoff-cursor block >
25
29
  * watchdog block > merged advisory > silent release — and emits AT MOST ONE
@@ -71,7 +75,7 @@
71
75
  import fs from 'node:fs/promises';
72
76
  import fsSync from 'node:fs';
73
77
  import path from 'node:path';
74
- import { spawn } from 'node:child_process';
78
+ import { spawn, spawnSync } from 'node:child_process';
75
79
 
76
80
  import { withAsyncLock } from './async-lock.mjs';
77
81
 
@@ -447,9 +451,141 @@ function parseRunCursor(text) {
447
451
  phase: field('Phase'),
448
452
  cursor: field('Cursor'),
449
453
  next: field('Next'),
454
+ exitPredicate: field('ExitPredicate'),
450
455
  };
451
456
  }
452
457
 
458
+ // ─── Exit predicate (mechanic #7 — exit-predicate-first loop) ─────────────
459
+ //
460
+ // The planner writes one `ExitPredicate:` line into RUN.md at P2 and the gate
461
+ // evaluates it mechanically when the cursor claims `Phase: done`. The grammar
462
+ // is deliberately tiny — `all(<term>[,<term>...])` or a single term, where a
463
+ // term is `index:all-done`, `git:clean`, or `file-exists:<relpath>` — so the
464
+ // whole check stays inside the hook deadline: ≤3 file reads + one
465
+ // `git status --porcelain`, no eval, no shell-out of predicate text.
466
+
467
+ /**
468
+ * Parse the ExitPredicate expression into a term list. Returns null when the
469
+ * expression is malformed — a malformed predicate is unsatisfiable, so a run
470
+ * that wrote garbage cannot claim `done` through it.
471
+ */
472
+ export function parseExitPredicate(expr) {
473
+ const trimmed = String(expr || '').trim();
474
+ if (!trimmed) return null;
475
+ const inner = trimmed.match(/^all\((.*)\)$/s)?.[1] ?? trimmed;
476
+ const terms = inner
477
+ .split(',')
478
+ .map((t) => t.trim())
479
+ .filter(Boolean);
480
+ if (terms.length === 0) return null;
481
+ const TERM_RE = /^(index:all-done|git:clean|file-exists:[^\s,]+)$/;
482
+ return terms.every((t) => TERM_RE.test(t)) ? terms : null;
483
+ }
484
+
485
+ /**
486
+ * `index:all-done` — every TASK-xxx row in docs/AI_HANDOFF/INDEX.md is in a
487
+ * terminal-success status (`done` or `cancelled_superseded`). An unreadable or
488
+ * taskless INDEX cannot prove all-done, so the term fails.
489
+ */
490
+ async function evalIndexAllDone(projectRoot) {
491
+ let text;
492
+ try {
493
+ text = await fs.readFile(path.join(projectRoot, 'docs', 'AI_HANDOFF', 'INDEX.md'), 'utf8');
494
+ } catch {
495
+ return { pass: false, detail: 'INDEX.md unreadable' };
496
+ }
497
+ const rows = String(text)
498
+ .split('\n')
499
+ .map((line) => line.trim())
500
+ .filter((line) => /^\|TASK-/i.test(line));
501
+ if (rows.length === 0) return { pass: false, detail: 'INDEX.md has no task rows' };
502
+ const failing = [];
503
+ for (const row of rows) {
504
+ const cells = row.split('|').map((c) => c.trim()).filter(Boolean);
505
+ const status = (cells[4] || '').toLowerCase();
506
+ if (status !== 'done' && status !== 'cancelled_superseded') {
507
+ failing.push(`${cells[0]}=${cells[4] || '?'}`);
508
+ }
509
+ }
510
+ return failing.length === 0
511
+ ? { pass: true, detail: `${rows.length} task row(s) done` }
512
+ : { pass: false, detail: `not done: ${failing.join(', ')}` };
513
+ }
514
+
515
+ /**
516
+ * `git:clean` — `git status --porcelain` is empty. A non-git directory (or a
517
+ * missing git binary) cannot prove clean, so the term fails without throwing.
518
+ */
519
+ function evalGitClean(projectRoot) {
520
+ try {
521
+ if (!fsSync.existsSync(path.join(projectRoot, '.git'))) {
522
+ return { pass: false, detail: 'not a git worktree' };
523
+ }
524
+ const res = spawnSync('git', ['status', '--porcelain'], {
525
+ cwd: projectRoot,
526
+ encoding: 'utf8',
527
+ timeout: 2000,
528
+ });
529
+ if (res.error || res.status !== 0) {
530
+ return { pass: false, detail: 'git status failed' };
531
+ }
532
+ return String(res.stdout).trim() === ''
533
+ ? { pass: true, detail: 'clean' }
534
+ : { pass: false, detail: 'dirty worktree' };
535
+ } catch {
536
+ return { pass: false, detail: 'git status failed' };
537
+ }
538
+ }
539
+
540
+ /**
541
+ * `file-exists:<relpath>` — the repo-relative path exists. Absolute paths and
542
+ * `..` segments are rejected outright: the predicate may only look inside the
543
+ * project root.
544
+ */
545
+ async function evalFileExists(projectRoot, rel) {
546
+ const segments = String(rel).split(/[\\/]/);
547
+ if (path.isAbsolute(rel) || segments.includes('..')) {
548
+ return { pass: false, detail: 'path escapes project root' };
549
+ }
550
+ try {
551
+ await fs.stat(path.join(projectRoot, rel));
552
+ return { pass: true, detail: 'exists' };
553
+ } catch {
554
+ return { pass: false, detail: 'missing' };
555
+ }
556
+ }
557
+
558
+ /**
559
+ * Evaluate the RUN.md `ExitPredicate:` expression. Returns
560
+ * `{ ok, terms: [{term, pass, detail}] }`; `ok` is false when the expression is
561
+ * malformed (unsatisfiable). Advisory like the rest of the lane — a thrown
562
+ * error becomes a failed term, never an exception.
563
+ */
564
+ async function evaluateExitPredicate(projectRoot, expr) {
565
+ const terms = parseExitPredicate(expr);
566
+ if (!terms) {
567
+ return { ok: false, terms: [{ term: String(expr || '').trim() || '(empty)', pass: false, detail: 'malformed predicate' }] };
568
+ }
569
+ const results = [];
570
+ for (const term of terms) {
571
+ let r;
572
+ try {
573
+ if (term === 'index:all-done') r = await evalIndexAllDone(projectRoot);
574
+ else if (term === 'git:clean') r = evalGitClean(projectRoot);
575
+ else r = await evalFileExists(projectRoot, term.slice('file-exists:'.length));
576
+ } catch (err) {
577
+ r = { pass: false, detail: `evaluator error: ${err?.message || err}` };
578
+ }
579
+ results.push({ term, pass: r.pass, detail: r.detail });
580
+ }
581
+ return { ok: results.every((r) => r.pass), terms: results };
582
+ }
583
+
584
+ /** One-line per-term predicate state for block reasons and stall advisories. */
585
+ function formatPredicateState(result) {
586
+ return result.terms.map((t) => `${t.term}=${t.pass ? 'pass' : `FAIL (${t.detail})`}`).join(', ');
587
+ }
588
+
453
589
  /**
454
590
  * Mutate `.ukit/storage/cache/stop-coordinator/state.json`'s `handoff` slot:
455
591
  * { signature, count }. Same signature as last time → count+1; a moved cursor
@@ -506,6 +642,13 @@ async function bumpHandoffStallCount({ projectRoot, signature, lockBudgetMs }) {
506
642
  * un-advancing cursor has already bounced `maxStalledBlocks` stops in a row
507
643
  * (liveness breaker — the gate must never be an unstoppable loop).
508
644
  *
645
+ * Exit predicate (mechanic #7): when RUN.md carries an `ExitPredicate:` line,
646
+ * `Phase: done` is only accepted if the predicate evaluates true — a `done`
647
+ * claim with failing terms is bounced back with the per-term state. A missing
648
+ * predicate keeps the legacy phase-only release; `Phase: blocked` stays
649
+ * terminal either way. The predicate is never relaxed: the stalled-cursor
650
+ * advisory reports its state so a plateau is visible, not silent.
651
+ *
509
652
  * Advisory lane: EVERY failure path returns { kind: 'none' }, never throws.
510
653
  */
511
654
  export async function evaluateHandoffCursor({ projectRoot, now = Date.now(), lockBudgetMs = LOCK_BUDGET_MS } = {}) {
@@ -518,21 +661,57 @@ export async function evaluateHandoffCursor({ projectRoot, now = Date.now(), loc
518
661
  } catch {
519
662
  return { kind: 'none' };
520
663
  }
521
- const { goal, phase, cursor, next } = parseRunCursor(text);
522
- if (!phase || /^done\b/i.test(phase) || /^blocked\b/i.test(phase)) {
664
+ const { goal, phase, cursor, next, exitPredicate } = parseRunCursor(text);
665
+ // Shared terminal predicate — tests/handoff/cycle21 greps this exact shape across
666
+ // all four RUN.md consumers; keep `^(done|blocked)` as one expression.
667
+ const terminal = String(phase).match(/^(done|blocked)\b/i)?.[1]?.toLowerCase() || '';
668
+ if (!phase || terminal === 'blocked') {
523
669
  return { kind: 'none' };
524
670
  }
671
+ if (terminal === 'done') {
672
+ if (!exitPredicate) return { kind: 'none' }; // legacy phase-only release
673
+ let predicate;
674
+ try {
675
+ predicate = await evaluateExitPredicate(projectRoot, exitPredicate);
676
+ } catch {
677
+ return { kind: 'none' }; // advisory lane: evaluator failure never wedges a stop
678
+ }
679
+ if (predicate.ok) return { kind: 'none' };
680
+ const lines = [
681
+ 'UKit handoff run claims `Phase: done` but its ExitPredicate is not satisfied — this',
682
+ 'stop is refused. "Done" means the predicate the planner wrote at P2 evaluates true;',
683
+ 'a plateau is not a stop and the predicate is never relaxed.',
684
+ '',
685
+ ` Goal: ${goal || '(not recorded)'}`,
686
+ ` ExitPredicate: ${exitPredicate}`,
687
+ ` Predicate state: ${formatPredicateState(predicate)}`,
688
+ '',
689
+ 'CONTINUE IMMEDIATELY: read docs/AI_HANDOFF/RUN.md and docs/AI_HANDOFF/INDEX.md, finish',
690
+ 'the work the failing terms name (or record the real external blocker and set',
691
+ '`Phase: blocked`), then rewrite the run cursor.',
692
+ ];
693
+ return { kind: 'block', reason: lines.join('\n') };
694
+ }
525
695
 
526
696
  const signature = `${phase}|${cursor}|${next}`;
527
697
  const streak = await bumpHandoffStallCount({ projectRoot, signature, lockBudgetMs });
528
698
  if (streak !== null && streak > gate.maxStalledBlocks) {
699
+ let predicateState = 'no ExitPredicate recorded';
700
+ if (exitPredicate) {
701
+ try {
702
+ predicateState = formatPredicateState(await evaluateExitPredicate(projectRoot, exitPredicate));
703
+ } catch {
704
+ predicateState = 'predicate evaluation failed';
705
+ }
706
+ }
529
707
  return {
530
708
  kind: 'advisory',
531
709
  systemMessage:
532
710
  `[ukit-stop-coordinator] handoff-cursor released this stop: the run cursor ` +
533
711
  `("${signature}") has not advanced across ${streak - 1} consecutive blocked stops ` +
534
712
  `(cap ${gate.maxStalledBlocks}), so the gate treats the run as stalled rather than ` +
535
- `looping forever. Inspect docs/AI_HANDOFF/RUN.md and docs/AI_HANDOFF/INDEX.md; ` +
713
+ `looping forever. ExitPredicate state: ${predicateState}. ` +
714
+ `Inspect docs/AI_HANDOFF/RUN.md and docs/AI_HANDOFF/INDEX.md; ` +
536
715
  `the run can be resumed with /ukit:handoff-fullstack or cleared with /ukit:handoff-clear.`,
537
716
  };
538
717
  }
@@ -56,6 +56,8 @@ If any input is missing, return `CHANGES-REQUESTED` with reason "incomplete hand
56
56
  VERDICT: APPROVED | APPROVED-WITH-MINOR | CHANGES-REQUESTED | CRITICAL
57
57
  REVIEWER_MODEL: [model name actually used]
58
58
  EXECUTOR_MODEL: [from executor report]
59
+ PANEL_MEMBER: [panel tier — `@slow`/`@default`/`@smol`, or `solo` when the panel degenerates; `-` outside panel runs]
60
+ AGREEMENT_MAP: [lead member only — finding → panel members reporting it, from review-panel-aggregate.mjs output; `-` otherwise]
59
61
  VERIFICATION_RERUN:
60
62
  command: [exact command]
61
63
  result: [N pass / M fail]
@@ -120,6 +122,31 @@ Same model is the most common silent failure. Do not skip this check.
120
122
  - Keep the verdict block <= 30 lines. Findings are bullet points, not essays.
121
123
  - The same-model refusal above is non-negotiable: bypassing it defeats the entire Quality Gate.
122
124
 
125
+
126
+ ## Panel review (diversity review)
127
+
128
+ When the orchestrator spawns you as one member of the `modelRoles['review-panel']`
129
+ panel (`.ukit/storage/config.json` → `modelRoles`, default `['smart','code','lite']`
130
+ — the omp tier aliases `@slow`/`@default`/`@smol`), everything above applies
131
+ unchanged — same Review order, same severity ladder, same verdict block — with these
132
+ additions:
133
+
134
+ - Set `PANEL_MEMBER: <tier>` in the verdict to your panel tier (e.g. `@slow`,
135
+ `@default`, `@smol`). Every member runs the same rubric independently; do not read
136
+ or wait for other members' verdicts.
137
+ - Leave `AGREEMENT_MAP` as `-` unless you are the **lead member** (the first entry in
138
+ `modelRoles['review-panel']` = the judgment tier). After the orchestrator runs
139
+ `node .claude/ukit/index/review-panel-aggregate.mjs <TASK-xxx.md...>` and hands you
140
+ the output, the lead fills `AGREEMENT_MAP` with the emitted finding → members map
141
+ and applies the lead-judgment buckets to every finding: **Act on** / **Consider** /
142
+ **Noted** / **Dismissed**.
143
+ - `consensus≥2 identical findings = high signal`: a finding reported by two or more
144
+ panel members is high-signal and must not be bucketed below **Consider** without a
145
+ stated reason.
146
+ - Degenerate panel: if `modelRoles['review-panel']` is empty or has one entry, the
147
+ panel collapses to a single reviewer — run the normal solo review, set
148
+ `PANEL_MEMBER: solo`, and note the fallback in `NOTES`.
149
+
123
150
  ## Spec/Plan Review (REVIEW_TARGET_TYPE=spec|plan)
124
151
 
125
152
  ### Inputs you expect
@@ -195,6 +195,21 @@ Status: planning_done — ready for executor
195
195
  ```
196
196
  Wave structure is NOT stored here — inferred from task `Dependencies` fields at runtime.
197
197
 
198
+ **RUN.md** — write the run's `ExitPredicate:` line before iteration 1 (the orchestrator
199
+ keeps it verbatim on every cursor rewrite):
200
+
201
+ ```
202
+ ExitPredicate: all(index:all-done, git:clean)
203
+ ```
204
+
205
+ Grammar: `all(<term>[,<term>...])` or a single term; terms are `index:all-done`,
206
+ `git:clean`, `file-exists:<relpath>` (repo-relative, no `..` segments). Default to
207
+ `all(index:all-done, git:clean)`; add `file-exists:` terms only for artifacts the plan
208
+ itself must produce. The stop gate evaluates this mechanically when the cursor claims
209
+ `Phase: done` — a `done` claim with failing terms is bounced back with the per-term
210
+ state. **Never relax the predicate mid-run**: if the run cannot satisfy it, the plan is
211
+ wrong — fix the plan, not the gate.
212
+
198
213
  ## Self-Audit — run before reporting, every time
199
214
 
200
215
  The downstream pipeline is unattended: an incomplete or wrong plan is not caught by a human
@@ -85,9 +85,9 @@
85
85
  },
86
86
  "routing": {
87
87
  "routeSchema": { "stage": "off" },
88
- "rigor": { "stage": "off" },
89
- "fastPath": { "stage": "off" },
90
- "escalation": { "stage": "off" }
88
+ "rigor": { "stage": "shadow" },
89
+ "fastPath": { "stage": "shadow" },
90
+ "escalation": { "stage": "shadow" }
91
91
  },
92
92
  "modelRoles": {
93
93
  "code": "code",