@ngockhoale/ukit 3.3.2 → 3.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +44 -0
- package/manifests/engineConformance.yaml +17 -1
- package/manifests/hostCapabilities.yaml +68 -1
- package/manifests/platform.full.yaml +138 -0
- package/manifests/platform.user.yaml +255 -3
- package/package.json +1 -1
- package/scripts/bench/subagent-orchestrator-corpus.mjs +275 -0
- package/scripts/bench/subagent-orchestrator-eval.mjs +565 -0
- package/scripts/probe/codex-capability-probe.mjs +169 -0
- package/src/cli/commands/doctor.js +168 -0
- package/src/cli/commands/indexTools.js +7 -0
- package/src/cli/commands/metrics.js +66 -2
- package/src/cli/commands/playbook.js +4 -4
- package/src/cli/commands/vm.js +49 -8
- package/src/core/agentRuntime/adapters.js +328 -27
- package/src/core/agentRuntime/artifacts.js +89 -0
- package/src/core/agentRuntime/context.js +345 -1
- package/src/core/agentRuntime/contract.js +296 -0
- package/src/core/agentRuntime/eventStore.js +176 -0
- package/src/core/agentRuntime/shadowRun.js +481 -5
- package/src/core/agentRuntime/telemetry.js +121 -0
- package/src/core/observability/emit/lifecycle.js +68 -1
- package/src/core/observability/emit/sessionBoot.js +393 -0
- package/src/core/observability/privacy/allowlist.js +10 -1
- package/src/core/observability/schema/registry.js +10 -0
- package/src/core/runtimeConfig.js +133 -0
- package/src/core/userPlaybooks.js +18 -3
- package/src/decision/registry.js +19 -0
- package/src/diagnostics/feedbackEvents.js +7 -4
- package/src/diagnostics/routeOutcomes.js +51 -6
- package/src/diagnostics/skillAccuracy.js +43 -3
- package/src/index/crossCheckMatrix.js +412 -0
- package/src/index/fixLoopEscalation.js +453 -0
- package/src/index/playbookRegistry.js +691 -0
- package/src/index/reviewPolicy.js +368 -0
- package/src/index/routeResolver.js +915 -0
- package/src/index/sessionHistoryExtractor.js +359 -0
- package/src/index/taskRouting.js +764 -581
- package/src/index/tierSelection.js +308 -0
- package/src/index/verificationMap.js +404 -0
- package/template_project/.claude/hooks/observability-emit.mjs +14 -0
- package/template_project/.claude/hooks/record-execution.mjs +19 -1
- package/template_project/.claude/hooks/skill-router.sh +691 -25
- package/template_project/.claude/hooks/verification-guard.sh +230 -1
- package/template_project/.claude/settings.json +2 -2
- package/template_project/.claude/ukit/index/cross-check-matrix.mjs +415 -0
- package/template_project/.claude/ukit/index/fix-loop-escalation.mjs +456 -0
- package/template_project/.claude/ukit/index/playbook-registry.mjs +690 -0
- package/template_project/.claude/ukit/index/review-panel-aggregate.mjs +20 -2
- package/template_project/.claude/ukit/index/review-policy.mjs +376 -0
- package/template_project/.claude/ukit/index/route-resolver.mjs +1059 -0
- package/template_project/.claude/ukit/index/route-task.mjs +1253 -846
- package/template_project/.claude/ukit/index/session-history-extractor.mjs +362 -0
- package/template_project/.claude/ukit/index/tier-selection.mjs +309 -0
- package/template_project/.claude/ukit/index/verification-map.mjs +403 -0
- package/template_project/.claude/ukit/index/worktree-sweep.mjs +195 -0
- package/template_project/.claude/ukit/runtime/execution-ledger.mjs +789 -11
- package/template_project/.claude/ukit/runtime/observability-emit.mjs +1102 -0
- package/template_project/.claude/ukit/runtime/reinject-context.mjs +9 -1
- package/template_project/.claude/ukit/runtime/resumable-run.mjs +149 -5
- package/template_project/.claude/ukit/runtime/stop-coordinator.mjs +323 -6
- package/template_project/.codex/README.md +8 -0
- package/template_project/.omp/hooks/pre/ukit-bridge.js +8 -1
- package/template_project/ukit/README.md +1 -1
- package/template_project/ukit/storage/config.json +20 -0
- package/template_user/playbooks/architecture-decision.md +28 -0
- package/template_user/playbooks/autonomous-run.md +43 -0
- package/template_user/playbooks/autopilot-full.md +59 -0
- package/template_user/playbooks/autopilot-stack.md +54 -0
- package/template_user/playbooks/babysit.md +39 -0
- package/template_user/playbooks/bug-fix.md +3 -1
- package/template_user/playbooks/{issue-implementation.md → feature-implementation.md} +4 -2
- package/template_user/playbooks/hillclimb.md +44 -0
- package/template_user/playbooks/investigation.md +21 -0
- package/template_user/playbooks/migration.md +21 -0
- package/template_user/playbooks/open-pr.md +48 -0
- package/template_user/playbooks/orchestrate.md +45 -0
- package/template_user/playbooks/performance.md +33 -0
- package/template_user/playbooks/prototype.md +28 -0
- package/template_user/playbooks/refactor.md +19 -0
- package/template_user/playbooks/release.md +28 -0
- package/template_user/playbooks/runtime-forensics.md +23 -0
- package/template_user/playbooks/session-pickup.md +31 -0
- package/template_user/playbooks/shipping.md +53 -0
- package/template_user/playbooks/skill-evaluation.md +48 -0
- package/template_user/playbooks/small-feature.md +20 -0
- package/template_user/playbooks/verification-map.json +153 -0
- package/template_user/playbooks/verification.md +22 -0
- package/template_user/playbooks/worktree-cleanup.md +37 -0
|
@@ -100,8 +100,16 @@ async function buildResumableRunLines(projectRoot, state, config) {
|
|
|
100
100
|
const record = resumed.record;
|
|
101
101
|
if (!record) return [];
|
|
102
102
|
const lines = [];
|
|
103
|
+
// BL-024 (session-pickup): the state-verify receipt — record-vs-worktree
|
|
104
|
+
// fingerprint agreement (or the divergent keys) announced before any
|
|
105
|
+
// resumed plan lines, so a cold pickup never trusts the record silently.
|
|
103
106
|
if (resumed.status === 'stale') {
|
|
104
|
-
|
|
107
|
+
const keys = Array.isArray(resumed.changed) && resumed.changed.length > 0
|
|
108
|
+
? resumed.changed.join(', ')
|
|
109
|
+
: 'source/index';
|
|
110
|
+
lines.push(`- Resumed state verify: stale — recorded fingerprints diverge (${keys}); dependent plans were invalidated, completed receipts kept.`);
|
|
111
|
+
} else {
|
|
112
|
+
lines.push('- Resumed state verify: fresh — recorded fingerprints match the live index/config.');
|
|
105
113
|
}
|
|
106
114
|
const phase = String(record.workflow?.phase ?? '').trim();
|
|
107
115
|
if (phase) lines.push(`- Resumed run phase: ${phase}`);
|
|
@@ -445,7 +445,10 @@ function applyInvalidationCodes(record, codes) {
|
|
|
445
445
|
* Read the C10 record for a task. Never throws on corrupt state.
|
|
446
446
|
*
|
|
447
447
|
* @returns {Promise<{status:'fresh'|'stale'|'invalid'|'absent', record:object|null,
|
|
448
|
-
* invalidated?:string[], warnings:string[]}>}
|
|
448
|
+
* invalidated?:string[], changed?:string[], warnings:string[]}>}
|
|
449
|
+
* `changed` lists the source/index/config snapshot keys that diverged from the
|
|
450
|
+
* live fingerprint — the state-verify evidence the session-pickup playbook
|
|
451
|
+
* consumes (BL-024). Only present once the fingerprint check ran.
|
|
449
452
|
*/
|
|
450
453
|
export async function readResumableRun(projectRoot, {
|
|
451
454
|
taskId,
|
|
@@ -500,11 +503,93 @@ export async function readResumableRun(projectRoot, {
|
|
|
500
503
|
status: 'stale',
|
|
501
504
|
record,
|
|
502
505
|
invalidated,
|
|
506
|
+
changed,
|
|
503
507
|
warnings: [`source/index fingerprint changed (${changed.join(', ')}); dependent plans invalidated`],
|
|
504
508
|
};
|
|
505
509
|
}
|
|
506
510
|
|
|
507
|
-
return { status: 'fresh', record: parsed, warnings: [] };
|
|
511
|
+
return { status: 'fresh', record: parsed, changed, warnings: [] };
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
// --- suspend (TASK-C86-024, BL-024, CENSUS PB-U20) ----------------------------
|
|
515
|
+
|
|
516
|
+
/**
|
|
517
|
+
* The suspend half of the session-pickup contract: checkpoint progress into a
|
|
518
|
+
* record the pickup `read` consumes. Same merge semantics as route finalize —
|
|
519
|
+
* the just-completed phase lands in workflow.completedBlocks, plans and
|
|
520
|
+
* completed receipts carry forward, and a fresh live fingerprint is stamped so
|
|
521
|
+
* the next resume starts clean. An omitted taskBoundary preserves the existing
|
|
522
|
+
* record's boundary; an explicit new one starts a fresh record (a different
|
|
523
|
+
* logical task). Stage 'off' stays fail-closed — suspending writes nothing.
|
|
524
|
+
*/
|
|
525
|
+
export async function suspendResumableRun(projectRoot, {
|
|
526
|
+
taskId,
|
|
527
|
+
taskBoundary = null,
|
|
528
|
+
phase = 'suspended',
|
|
529
|
+
nextAction = null,
|
|
530
|
+
completedBlocks = [],
|
|
531
|
+
signal,
|
|
532
|
+
deadlineMs = LOCK_MAX_SLICE_MS,
|
|
533
|
+
} = {}) {
|
|
534
|
+
const boundaryProvided = typeof taskBoundary === 'string' && taskBoundary.length > 0;
|
|
535
|
+
const resumed = boundaryProvided
|
|
536
|
+
? await readResumableRun(projectRoot, { taskId, taskBoundary })
|
|
537
|
+
: await readResumableRun(projectRoot, { taskId });
|
|
538
|
+
const previous = resumed.status === 'fresh' || resumed.status === 'stale'
|
|
539
|
+
? resumed.record
|
|
540
|
+
: null;
|
|
541
|
+
const boundary = boundaryProvided
|
|
542
|
+
? taskBoundary
|
|
543
|
+
: (previous?.taskBoundary ?? 'suspended');
|
|
544
|
+
|
|
545
|
+
const carried = [...(previous?.workflow?.completedBlocks ?? [])];
|
|
546
|
+
const previousPhase = previous?.workflow?.phase;
|
|
547
|
+
if (previousPhase && previousPhase !== phase && !carried.includes(previousPhase)) {
|
|
548
|
+
carried.push(previousPhase);
|
|
549
|
+
}
|
|
550
|
+
const checkpoint = Array.isArray(completedBlocks) ? completedBlocks : [];
|
|
551
|
+
const merged = [...new Set([...carried, ...checkpoint])];
|
|
552
|
+
|
|
553
|
+
const fingerprint = await resumableRunSourceFingerprint(projectRoot);
|
|
554
|
+
const record = {
|
|
555
|
+
schemaVersion: RESUMABLE_RUN_SCHEMA_VERSION,
|
|
556
|
+
taskId,
|
|
557
|
+
taskBoundary: boundary,
|
|
558
|
+
route: previous?.route ?? {
|
|
559
|
+
routeVersion: 'suspend-cli',
|
|
560
|
+
mode: 'suspend',
|
|
561
|
+
rigor: 'unresolved',
|
|
562
|
+
contractVersion: 'suspend-cli',
|
|
563
|
+
},
|
|
564
|
+
budget: previous?.budget ?? { policyVersion: 'suspend-cli', consumed: 0, remaining: 0 },
|
|
565
|
+
workflow: {
|
|
566
|
+
workflowId: previous?.workflow?.workflowId ?? 'session-pickup',
|
|
567
|
+
workflowVersion: previous?.workflow?.workflowVersion ?? 'v1',
|
|
568
|
+
phase: String(phase ?? 'suspended'),
|
|
569
|
+
completedBlocks: merged,
|
|
570
|
+
},
|
|
571
|
+
invariants: previous?.invariants ?? [],
|
|
572
|
+
hypotheses: previous?.hypotheses ?? [],
|
|
573
|
+
decisions: previous?.decisions ?? [],
|
|
574
|
+
evidenceRefs: previous?.evidenceRefs ?? [],
|
|
575
|
+
escalationHistory: previous?.escalationHistory ?? [],
|
|
576
|
+
unresolvedFailures: previous?.unresolvedFailures ?? [],
|
|
577
|
+
nextAction: typeof nextAction === 'string' && nextAction !== ''
|
|
578
|
+
? nextAction.slice(0, MAX_STRING_LENGTH)
|
|
579
|
+
: null,
|
|
580
|
+
sourceSnapshot: { ...(previous?.sourceSnapshot ?? {}), ...fingerprint },
|
|
581
|
+
};
|
|
582
|
+
|
|
583
|
+
const write = await writeResumableRun(projectRoot, record, { signal, deadlineMs });
|
|
584
|
+
return {
|
|
585
|
+
ok: write.ok === true,
|
|
586
|
+
status: write.ok ? 'written' : (write.journaled ? 'journaled' : 'skipped'),
|
|
587
|
+
reason: write.ok || write.journaled ? null : (write.reason ?? 'unknown'),
|
|
588
|
+
errors: write.errors,
|
|
589
|
+
journaled: write.journaled === true,
|
|
590
|
+
path: write.ok ? write.path : resumableRunPath(projectRoot, taskId),
|
|
591
|
+
record,
|
|
592
|
+
};
|
|
508
593
|
}
|
|
509
594
|
|
|
510
595
|
/**
|
|
@@ -584,15 +669,74 @@ export async function resumableRunSourceFingerprint(projectRoot, { indexGenerate
|
|
|
584
669
|
|
|
585
670
|
// --- CLI (diagnostics) ---------------------------------------------------------
|
|
586
671
|
|
|
672
|
+
function parseCliFlags(argv) {
|
|
673
|
+
const flags = {};
|
|
674
|
+
for (let i = 0; i < argv.length; i += 1) {
|
|
675
|
+
const match = /^--([\w-]+)$/.exec(argv[i]);
|
|
676
|
+
if (match) {
|
|
677
|
+
flags[match[1]] = argv[i + 1] !== undefined ? argv[i + 1] : '';
|
|
678
|
+
i += 1;
|
|
679
|
+
}
|
|
680
|
+
}
|
|
681
|
+
return flags;
|
|
682
|
+
}
|
|
683
|
+
|
|
587
684
|
async function main() {
|
|
588
|
-
const [command, rootArg, taskId] = process.argv.slice(2);
|
|
685
|
+
const [command, rootArg, taskId, ...rest] = process.argv.slice(2);
|
|
589
686
|
const projectRoot = path.resolve(rootArg || process.cwd());
|
|
687
|
+
const flags = parseCliFlags(rest);
|
|
590
688
|
if (command === 'read' && taskId) {
|
|
591
|
-
|
|
689
|
+
// BL-024 (session-pickup): the pickup path's state-verify receipt — compare
|
|
690
|
+
// the recorded fingerprints against the live index/config so a stale
|
|
691
|
+
// record reports the divergence instead of being trusted silently.
|
|
692
|
+
const sourceFingerprint = await resumableRunSourceFingerprint(projectRoot);
|
|
693
|
+
const result = await readResumableRun(projectRoot, {
|
|
694
|
+
taskId,
|
|
695
|
+
taskBoundary: typeof flags.boundary === 'string' && flags.boundary !== ''
|
|
696
|
+
? flags.boundary
|
|
697
|
+
: undefined,
|
|
698
|
+
sourceFingerprint,
|
|
699
|
+
});
|
|
700
|
+
const output = {
|
|
701
|
+
...result,
|
|
702
|
+
stateVerify: {
|
|
703
|
+
result: result.status === 'fresh' ? 'pass' : 'fail',
|
|
704
|
+
changed: result.changed ?? [],
|
|
705
|
+
cursor: result.record
|
|
706
|
+
? {
|
|
707
|
+
phase: result.record.workflow?.phase ?? null,
|
|
708
|
+
completedBlocks: result.record.workflow?.completedBlocks ?? [],
|
|
709
|
+
nextAction: result.record.nextAction ?? null,
|
|
710
|
+
}
|
|
711
|
+
: null,
|
|
712
|
+
},
|
|
713
|
+
};
|
|
714
|
+
process.stdout.write(`${JSON.stringify(output, null, 2)}\n`);
|
|
715
|
+
return;
|
|
716
|
+
}
|
|
717
|
+
if (command === 'suspend' && taskId) {
|
|
718
|
+
// BL-024 (PB-U20 suspend half): checkpoint progress into a record the
|
|
719
|
+
// pickup `read` above consumes — persist + report, never write unlocked.
|
|
720
|
+
const result = await suspendResumableRun(projectRoot, {
|
|
721
|
+
taskId,
|
|
722
|
+
taskBoundary: typeof flags.boundary === 'string' && flags.boundary !== ''
|
|
723
|
+
? flags.boundary
|
|
724
|
+
: null,
|
|
725
|
+
phase: flags.phase,
|
|
726
|
+
nextAction: typeof flags.next === 'string' && flags.next !== '' ? flags.next : null,
|
|
727
|
+
completedBlocks: typeof flags.completed === 'string' && flags.completed !== ''
|
|
728
|
+
? flags.completed.split(',').map((entry) => entry.trim()).filter(Boolean)
|
|
729
|
+
: [],
|
|
730
|
+
});
|
|
592
731
|
process.stdout.write(`${JSON.stringify(result, null, 2)}\n`);
|
|
732
|
+
if (!result.ok && !result.journaled) process.exitCode = 3;
|
|
593
733
|
return;
|
|
594
734
|
}
|
|
595
|
-
process.stderr.write(
|
|
735
|
+
process.stderr.write(
|
|
736
|
+
'usage: resumable-run.mjs read <projectRoot> <taskId> [--boundary <id>]\n'
|
|
737
|
+
+ ' resumable-run.mjs suspend <projectRoot> <taskId> [--boundary <id>] '
|
|
738
|
+
+ '[--phase <phase>] [--next <action>] [--completed <a,b,...>]\n',
|
|
739
|
+
);
|
|
596
740
|
process.exitCode = 2;
|
|
597
741
|
}
|
|
598
742
|
|
|
@@ -78,6 +78,18 @@ import path from 'node:path';
|
|
|
78
78
|
import { spawn, spawnSync } from 'node:child_process';
|
|
79
79
|
|
|
80
80
|
import { withAsyncLock } from './async-lock.mjs';
|
|
81
|
+
// TASK-005 (FR-005): the outcome seam loads LAZILY — a partial install missing
|
|
82
|
+
// observability-emit.mjs must degrade to "no outcome record" (telemetry is
|
|
83
|
+
// advisory), never crash the coordinator at module load.
|
|
84
|
+
let outcomeEmitModulePromise = null;
|
|
85
|
+
function loadOutcomeEmitModule() {
|
|
86
|
+
if (!outcomeEmitModulePromise) {
|
|
87
|
+
outcomeEmitModulePromise = import(
|
|
88
|
+
new URL('./observability-emit.mjs', import.meta.url).href
|
|
89
|
+
).catch(() => null);
|
|
90
|
+
}
|
|
91
|
+
return outcomeEmitModulePromise;
|
|
92
|
+
}
|
|
81
93
|
|
|
82
94
|
// Orphan-leak guard (2.4.1 / TASK-025): completion-gate.sh spawns this as
|
|
83
95
|
// `node "$SCRIPT"`, so a host that kills the bash wrapper would reparent a
|
|
@@ -357,11 +369,21 @@ export function coordinateStopDecisions({ completion = null, watchdog = null, ha
|
|
|
357
369
|
}
|
|
358
370
|
|
|
359
371
|
// ─── Completion evaluator (spawn once) ────────────────────────────────────
|
|
360
|
-
function runEvaluatorChild(scriptPath, input, budgetMs) {
|
|
372
|
+
function runEvaluatorChild(scriptPath, input, budgetMs, { cwd = null } = {}) {
|
|
361
373
|
return new Promise((resolve, reject) => {
|
|
362
374
|
const child = spawn(process.execPath, [scriptPath, '--evaluate-stop'], {
|
|
363
|
-
// The child gets the same environment plus its own (smaller) self-kill
|
|
364
|
-
|
|
375
|
+
// The child gets the same environment plus its own (smaller) self-kill
|
|
376
|
+
// budget, and the caller's project root is pinned twice — cwd AND an
|
|
377
|
+
// explicit CLAUDE_PROJECT_DIR — because the evaluator resolves its own
|
|
378
|
+
// root from `CLAUDE_PROJECT_DIR || process.cwd()`: an in-proc caller
|
|
379
|
+
// (bridge, tests) whose cwd is another project used to evaluate THAT
|
|
380
|
+
// project's state and silently release the wrong stop.
|
|
381
|
+
cwd: cwd || undefined,
|
|
382
|
+
env: {
|
|
383
|
+
...process.env,
|
|
384
|
+
CLAUDE_PROJECT_DIR: cwd || process.env.CLAUDE_PROJECT_DIR || '',
|
|
385
|
+
UKIT_HOOK_DEADLINE_MS: String(Math.max(250, budgetMs)),
|
|
386
|
+
},
|
|
365
387
|
stdio: ['pipe', 'pipe', 'pipe'],
|
|
366
388
|
});
|
|
367
389
|
// Tracked so the hook-deadline path can reap a still-running evaluator instead of
|
|
@@ -416,10 +438,237 @@ export async function evaluateCompletionPolicy({ projectRoot, rawInput, budgetMs
|
|
|
416
438
|
} catch {
|
|
417
439
|
throw new Error('completion evaluator unavailable: execution-ledger.mjs is missing (run: ukit install)');
|
|
418
440
|
}
|
|
419
|
-
|
|
441
|
+
// TASK-C85-014 (BL-014): consult the verification map BEFORE spawning the
|
|
442
|
+
// evaluator — the resolved recipe is stamped onto the route state the child
|
|
443
|
+
// re-reads, so the single authority (the ledger's evaluateCompletion) owns
|
|
444
|
+
// the verdict either way. Advisory: any failure leaves the gate as before.
|
|
445
|
+
try {
|
|
446
|
+
await stampVerificationRecipe({ projectRoot });
|
|
447
|
+
} catch {}
|
|
448
|
+
const { stdout } = await runEvaluatorChild(ledgerPath, String(rawInput || ''), budgetMs, { cwd: projectRoot });
|
|
420
449
|
return normalizeEvaluatorStdout(stdout);
|
|
421
450
|
}
|
|
422
451
|
|
|
452
|
+
// ─── Verification-map recipe consult (C85 TASK-014, BL-014) ────────────────
|
|
453
|
+
// The resolved verification recipe is an advisory INPUT to the completion
|
|
454
|
+
// evaluator, not a second gate: the coordinator reads the route state, loads
|
|
455
|
+
// verification-map.json through the shared index reader, classifies the
|
|
456
|
+
// artifact from the change set, and stamps `routeSummary.verificationRecipe`
|
|
457
|
+
// (+ `routeSummary.artifactClass`) back onto the state file before the
|
|
458
|
+
// evaluator child re-reads it. The stamp persists deliberately — the
|
|
459
|
+
// resume-intent lane (`hasUnfinishedCompletion`) then counts recipe evidence
|
|
460
|
+
// identically across compactions. Map absent/malformed or class unknown →
|
|
461
|
+
// no recipe, and a stale stamp is cleared so behavior returns to pre-map.
|
|
462
|
+
let verificationMapModulePromise = null;
|
|
463
|
+
function loadVerificationMapModule() {
|
|
464
|
+
if (!verificationMapModulePromise) {
|
|
465
|
+
verificationMapModulePromise = import(
|
|
466
|
+
new URL('../index/verification-map.mjs', import.meta.url).href
|
|
467
|
+
).catch(() => null);
|
|
468
|
+
}
|
|
469
|
+
return verificationMapModulePromise;
|
|
470
|
+
}
|
|
471
|
+
|
|
472
|
+
function readChangedPathsForRecipe(projectRoot) {
|
|
473
|
+
try {
|
|
474
|
+
// -z NUL-terminates entries verbatim — the C-quoted porcelain form
|
|
475
|
+
// ("src/My Widget.jsx") would misclassify to null downstream (fix round:
|
|
476
|
+
// review minor). Rename/copy entries emit two fields under -z: `XY to`
|
|
477
|
+
// then the bare `from` path — the from field is skipped positionally.
|
|
478
|
+
const out = spawnSync('git', ['status', '--porcelain=v1', '-z', '--untracked-files=all'], {
|
|
479
|
+
cwd: projectRoot,
|
|
480
|
+
encoding: 'utf8',
|
|
481
|
+
timeout: 800,
|
|
482
|
+
stdio: ['ignore', 'pipe', 'ignore'],
|
|
483
|
+
});
|
|
484
|
+
if (out.error || out.status !== 0) return [];
|
|
485
|
+
const fields = out.stdout.split('\0');
|
|
486
|
+
const entries = [];
|
|
487
|
+
for (let i = 0; i < fields.length; i += 1) {
|
|
488
|
+
const field = fields[i];
|
|
489
|
+
if (!field || !/^.. /.test(field)) continue;
|
|
490
|
+
entries.push(field); // raw "XY path" — callers strip the status prefix
|
|
491
|
+
if (/^[RC]/.test(field)) i += 1; // consume the rename/copy source field
|
|
492
|
+
}
|
|
493
|
+
return entries;
|
|
494
|
+
} catch {
|
|
495
|
+
return [];
|
|
496
|
+
}
|
|
497
|
+
}
|
|
498
|
+
|
|
499
|
+
async function writeJsonAtomicLocal(filePath, value) {
|
|
500
|
+
await fs.mkdir(path.dirname(filePath), { recursive: true });
|
|
501
|
+
const tempPath = `${filePath}.tmp-${process.pid}-${Date.now()}-${Math.random().toString(16).slice(2)}`;
|
|
502
|
+
try {
|
|
503
|
+
await fs.writeFile(tempPath, `${JSON.stringify(value, null, 2)}\n`, 'utf8');
|
|
504
|
+
await fs.rename(tempPath, filePath);
|
|
505
|
+
} catch (error) {
|
|
506
|
+
try { await fs.rm(tempPath, { force: true }); } catch {}
|
|
507
|
+
throw error;
|
|
508
|
+
}
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
// The stamped recipe must round-trip through JSON: a plain data subset, never
|
|
512
|
+
// the reader's full object (which carries no functions anyway, but this keeps
|
|
513
|
+
// the on-disk contract explicit).
|
|
514
|
+
function serializeVerificationRecipe(recipe, artifactClass) {
|
|
515
|
+
if (!recipe) return null;
|
|
516
|
+
return {
|
|
517
|
+
artifactClass,
|
|
518
|
+
name: recipe.name || null,
|
|
519
|
+
check: recipe.check || null,
|
|
520
|
+
harness: recipe.harness || null,
|
|
521
|
+
fallback: recipe.fallback || null,
|
|
522
|
+
evidenceRequired: Array.isArray(recipe.evidenceRequired) ? recipe.evidenceRequired : [],
|
|
523
|
+
};
|
|
524
|
+
}
|
|
525
|
+
|
|
526
|
+
async function stampVerificationRecipe({ projectRoot }) {
|
|
527
|
+
const mapModule = await loadVerificationMapModule();
|
|
528
|
+
if (!mapModule) return; // reader absent (partial install): pre-map behavior
|
|
529
|
+
const statePath = path.join(projectRoot, '.claude', 'ukit', 'skill-router-state.json');
|
|
530
|
+
let state;
|
|
531
|
+
try {
|
|
532
|
+
state = JSON.parse(await fs.readFile(statePath, 'utf8'));
|
|
533
|
+
} catch {
|
|
534
|
+
return; // no route state → nothing to stamp
|
|
535
|
+
}
|
|
536
|
+
const routeSummary = state?.routeSummary;
|
|
537
|
+
if (!routeSummary || typeof routeSummary !== 'object') return;
|
|
538
|
+
|
|
539
|
+
const changedPaths = readChangedPathsForRecipe(projectRoot)
|
|
540
|
+
.map((line) => line.replace(/^(..)\s+/, ''));
|
|
541
|
+
const artifactClass = mapModule.classifyArtifactClass(changedPaths);
|
|
542
|
+
const map = await mapModule.loadVerificationMap({ projectRoot });
|
|
543
|
+
const recipe = artifactClass && map
|
|
544
|
+
? await mapModule.recipeForArtifact({
|
|
545
|
+
playbookId: routeSummary.playbookId || null,
|
|
546
|
+
artifactClass,
|
|
547
|
+
map,
|
|
548
|
+
})
|
|
549
|
+
: null;
|
|
550
|
+
const serialized = serializeVerificationRecipe(recipe, artifactClass);
|
|
551
|
+
|
|
552
|
+
const before = JSON.stringify(routeSummary.verificationRecipe ?? null);
|
|
553
|
+
const after = JSON.stringify(serialized);
|
|
554
|
+
if (before === after) return; // no stamp change → no rewrite
|
|
555
|
+
routeSummary.verificationRecipe = serialized;
|
|
556
|
+
if (artifactClass) routeSummary.artifactClass = artifactClass;
|
|
557
|
+
else delete routeSummary.artifactClass;
|
|
558
|
+
await writeJsonAtomicLocal(statePath, state);
|
|
559
|
+
}
|
|
560
|
+
|
|
561
|
+
// ─── Review-policy advisory lane (C85 TASK-015, BL-015) ────────────────────
|
|
562
|
+
// "Is one more verification round needed?" becomes a signal decision at the
|
|
563
|
+
// pre-Stop gate. The lane is advisory: the verdict + namedCheck are recorded
|
|
564
|
+
// on the coordinator output and a `kind:'review-policy'` decision receipt is
|
|
565
|
+
// appended for calibration — the completion gate's own block logic and the
|
|
566
|
+
// merged decision are untouched (single Stop authority preserved).
|
|
567
|
+
let reviewPolicyModulePromise = null;
|
|
568
|
+
function loadReviewPolicyModule() {
|
|
569
|
+
if (!reviewPolicyModulePromise) {
|
|
570
|
+
reviewPolicyModulePromise = import(
|
|
571
|
+
new URL('../index/review-policy.mjs', import.meta.url).href
|
|
572
|
+
).catch(() => null);
|
|
573
|
+
}
|
|
574
|
+
return reviewPolicyModulePromise;
|
|
575
|
+
}
|
|
576
|
+
|
|
577
|
+
let ledgerReceiptModulePromise = null;
|
|
578
|
+
function loadLedgerReceiptModule() {
|
|
579
|
+
if (!ledgerReceiptModulePromise) {
|
|
580
|
+
ledgerReceiptModulePromise = import(
|
|
581
|
+
new URL('./execution-ledger.mjs', import.meta.url).href
|
|
582
|
+
).catch(() => null);
|
|
583
|
+
}
|
|
584
|
+
return ledgerReceiptModulePromise;
|
|
585
|
+
}
|
|
586
|
+
|
|
587
|
+
/**
|
|
588
|
+
* Evaluate the review decision policy once per stop when a route record
|
|
589
|
+
* exists, append the calibration receipt, and stamp reviewRounds +
|
|
590
|
+
* lastReviewPolicy back onto the route state for the next stop's round
|
|
591
|
+
* accounting. Returns the decision {action, namedCheck?, signalsHit[],
|
|
592
|
+
* reviewRounds, ...} or null when there is no route record / the module is
|
|
593
|
+
* absent (partial install). Never throws — an advisory lane must not wedge
|
|
594
|
+
* the coordinator.
|
|
595
|
+
*
|
|
596
|
+
* `completionKind` ('block'|'advisory'|'complete'|null) is the completion
|
|
597
|
+
* evaluator's verdict, used only to mark which done criteria are evidenced.
|
|
598
|
+
*/
|
|
599
|
+
export async function evaluateStopReviewPolicy({ projectRoot, completionKind = null } = {}) {
|
|
600
|
+
try {
|
|
601
|
+
const policyModule = await loadReviewPolicyModule();
|
|
602
|
+
if (!policyModule || typeof policyModule.evaluateReviewPolicy !== 'function') return null;
|
|
603
|
+
|
|
604
|
+
const statePath = path.join(projectRoot, '.claude', 'ukit', 'skill-router-state.json');
|
|
605
|
+
let state;
|
|
606
|
+
try {
|
|
607
|
+
state = JSON.parse(await fs.readFile(statePath, 'utf8'));
|
|
608
|
+
} catch {
|
|
609
|
+
return null; // no route record → nothing to evaluate
|
|
610
|
+
}
|
|
611
|
+
const routeSummary = state?.routeSummary;
|
|
612
|
+
if (!routeSummary || typeof routeSummary !== 'object') return null;
|
|
613
|
+
|
|
614
|
+
// A gate-released stop means the contract's evidence landed ('verified');
|
|
615
|
+
// otherwise the criteria are unproven — the gate still owns the block, the
|
|
616
|
+
// policy only reports what the next round would have to check.
|
|
617
|
+
const evidenceSatisfied = completionKind === 'complete';
|
|
618
|
+
const requiredTokens = [
|
|
619
|
+
...(Array.isArray(routeSummary?.executionContract?.completionEvidence)
|
|
620
|
+
? routeSummary.executionContract.completionEvidence
|
|
621
|
+
: []),
|
|
622
|
+
...(Array.isArray(routeSummary?.verificationRecipe?.evidenceRequired)
|
|
623
|
+
? routeSummary.verificationRecipe.evidenceRequired
|
|
624
|
+
: []),
|
|
625
|
+
].filter((token) => typeof token === 'string' && token.trim());
|
|
626
|
+
const doneCriteria = [...new Set(requiredTokens)].map((name) => ({
|
|
627
|
+
name,
|
|
628
|
+
status: evidenceSatisfied ? 'verified' : 'missing',
|
|
629
|
+
}));
|
|
630
|
+
|
|
631
|
+
const historySignals = routeSummary.historySignals && typeof routeSummary.historySignals === 'object'
|
|
632
|
+
? routeSummary.historySignals
|
|
633
|
+
: {};
|
|
634
|
+
const round = Number.isFinite(routeSummary.reviewRounds)
|
|
635
|
+
? routeSummary.reviewRounds
|
|
636
|
+
: 0;
|
|
637
|
+
|
|
638
|
+
const ledgerModule = await loadLedgerReceiptModule();
|
|
639
|
+
const appendDecisionReceipt = ledgerModule?.appendDecisionReceipt;
|
|
640
|
+
|
|
641
|
+
const decision = policyModule.evaluateReviewPolicy({
|
|
642
|
+
signals: routeSummary.suspicionSignals ?? {},
|
|
643
|
+
doneCriteria,
|
|
644
|
+
evidenceClasses: [],
|
|
645
|
+
fixLoopCount: Number.isFinite(historySignals.fixLoopCount) ? historySignals.fixLoopCount : 0,
|
|
646
|
+
round,
|
|
647
|
+
cap: policyModule.DEFAULT_REVIEW_ROUND_CAP,
|
|
648
|
+
projectRoot,
|
|
649
|
+
appendDecisionReceipt,
|
|
650
|
+
});
|
|
651
|
+
if (!decision || typeof decision !== 'object') return null;
|
|
652
|
+
|
|
653
|
+
// Persist the round counter + last verdict for the next stop (same atomic
|
|
654
|
+
// write as the recipe stamp). A stale verdict is honest — it is the last
|
|
655
|
+
// recorded policy decision, keyed by its action.
|
|
656
|
+
routeSummary.reviewRounds = Number.isFinite(decision.reviewRounds) ? decision.reviewRounds : round;
|
|
657
|
+
routeSummary.lastReviewPolicy = {
|
|
658
|
+
action: decision.action,
|
|
659
|
+
namedCheck: decision.namedCheck ?? null,
|
|
660
|
+
signalsHit: Array.isArray(decision.signalsHit) ? decision.signalsHit : [],
|
|
661
|
+
at: new Date().toISOString(),
|
|
662
|
+
};
|
|
663
|
+
await writeJsonAtomicLocal(statePath, state);
|
|
664
|
+
|
|
665
|
+
return decision;
|
|
666
|
+
} catch {
|
|
667
|
+
return null; // advisory lane: never wedge the coordinator
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
|
|
671
|
+
|
|
423
672
|
// ─── Handoff-cursor evaluator (in-process, advisory-on-failure) ───────────
|
|
424
673
|
/**
|
|
425
674
|
* Read `.ukit/storage/config.json` → `handoff.fullstack.stopGate*` merged over
|
|
@@ -813,7 +1062,6 @@ export async function runStopCoordinator({
|
|
|
813
1062
|
failures.watchdog = error?.message || String(error);
|
|
814
1063
|
process.stderr.write(`[ukit-stop-coordinator] task-watchdog evaluator failed: ${failures.watchdog}\n`);
|
|
815
1064
|
}
|
|
816
|
-
|
|
817
1065
|
let handoff = null;
|
|
818
1066
|
try {
|
|
819
1067
|
handoff = await evaluateHandoffCursor({ projectRoot, now, lockBudgetMs });
|
|
@@ -822,7 +1070,76 @@ export async function runStopCoordinator({
|
|
|
822
1070
|
process.stderr.write(`[ukit-stop-coordinator] handoff-cursor evaluator failed: ${failures.handoff}\n`);
|
|
823
1071
|
}
|
|
824
1072
|
|
|
825
|
-
|
|
1073
|
+
const merged = coordinateStopDecisions({ completion, watchdog, handoff, failures });
|
|
1074
|
+
|
|
1075
|
+
// C85 TASK-015 (BL-015): advisory review-policy verdict. The evaluator runs
|
|
1076
|
+
// only when a route record exists; its action + namedCheck are recorded on
|
|
1077
|
+
// the output and a review-policy receipt lands in decisions.tsv. The gate's
|
|
1078
|
+
// own block logic is unchanged — a non-finish verdict only surfaces as a
|
|
1079
|
+
// systemMessage note, never a block.
|
|
1080
|
+
try {
|
|
1081
|
+
const reviewPolicy = await evaluateStopReviewPolicy({
|
|
1082
|
+
projectRoot,
|
|
1083
|
+
completionKind: completion?.kind ?? null,
|
|
1084
|
+
});
|
|
1085
|
+
if (reviewPolicy) {
|
|
1086
|
+
merged.reviewPolicy = reviewPolicy;
|
|
1087
|
+
if (reviewPolicy.action !== 'finish') {
|
|
1088
|
+
const note = `[ukit-stop-coordinator] review policy (advisory): ${reviewPolicy.action}`
|
|
1089
|
+
+ (reviewPolicy.namedCheck ? ` — ${reviewPolicy.namedCheck}` : '')
|
|
1090
|
+
+ (Array.isArray(reviewPolicy.signalsHit) && reviewPolicy.signalsHit.length
|
|
1091
|
+
? ` (signals: ${reviewPolicy.signalsHit.join(', ')})` : '');
|
|
1092
|
+
merged.systemMessage = merged.systemMessage
|
|
1093
|
+
? `${merged.systemMessage}\n${note}`
|
|
1094
|
+
: note;
|
|
1095
|
+
}
|
|
1096
|
+
}
|
|
1097
|
+
} catch { /* advisory verdicts never affect the stop decision */ }
|
|
1098
|
+
|
|
1099
|
+
// TASK-005 (FR-005): the coordinator's terminal verdict lands ONE
|
|
1100
|
+
// `outcome.observed` record per coordinated stop — the merged evaluator
|
|
1101
|
+
// decision is the session's honest outcome at this stop. Fire-and-forget:
|
|
1102
|
+
// awaited inside the emit module's own 250ms bound; any failure returns
|
|
1103
|
+
// null and never changes the decision the caller is about to emit.
|
|
1104
|
+
{
|
|
1105
|
+
const stopOutcomeVerdict = merged.decision === 'block' ? 'BLOCKED'
|
|
1106
|
+
: typeof merged.systemMessage === 'string' && merged.systemMessage ? 'INCONCLUSIVE'
|
|
1107
|
+
: 'VERIFIED';
|
|
1108
|
+
// runId joins ledger-emitted outcomes on the same session identity the
|
|
1109
|
+
// ledger derives (run-<session>); transcript-only payloads hash the path
|
|
1110
|
+
// instead of leaking it; 'no-session' stays null (missing-data contract).
|
|
1111
|
+
const outcomePayload = sessionKey === 'no-session'
|
|
1112
|
+
? payload
|
|
1113
|
+
: (typeof payload?.session_id === 'string' && payload.session_id
|
|
1114
|
+
? { session_id: sessionKey }
|
|
1115
|
+
: { transcript_path: sessionKey });
|
|
1116
|
+
try {
|
|
1117
|
+
const emitModule = await loadOutcomeEmitModule();
|
|
1118
|
+
if (!emitModule || typeof emitModule.buildOutcomeObservedRecord !== 'function'
|
|
1119
|
+
|| typeof emitModule.emitOutcomeObserved !== 'function') {
|
|
1120
|
+
return { skip: null, merged };
|
|
1121
|
+
}
|
|
1122
|
+
const outcomeRecord = emitModule.buildOutcomeObservedRecord({
|
|
1123
|
+
outcome: {
|
|
1124
|
+
verdict: stopOutcomeVerdict,
|
|
1125
|
+
source: 'stop-coordinator',
|
|
1126
|
+
routeId: null,
|
|
1127
|
+
routeFingerprint: null,
|
|
1128
|
+
ledgerKey: null,
|
|
1129
|
+
runId: emitModule.outcomeRunId(outcomePayload),
|
|
1130
|
+
evidenceIds: [],
|
|
1131
|
+
cost: { latencyMs: null, rounds: null },
|
|
1132
|
+
reviewActions: [],
|
|
1133
|
+
},
|
|
1134
|
+
projectRoot,
|
|
1135
|
+
payload: typeof payload?.session_id === 'string' && payload.session_id ? payload : outcomePayload,
|
|
1136
|
+
env: process.env,
|
|
1137
|
+
});
|
|
1138
|
+
await emitModule.emitOutcomeObserved({ projectRoot, record: outcomeRecord, deadlineMs: 200 });
|
|
1139
|
+
} catch { /* telemetry never blocks or delays the stop decision */ }
|
|
1140
|
+
}
|
|
1141
|
+
|
|
1142
|
+
return { skip: null, merged };
|
|
826
1143
|
}
|
|
827
1144
|
|
|
828
1145
|
// ─── CLI ──────────────────────────────────────────────────────────────────
|
|
@@ -80,6 +80,14 @@ For clearly non-code specialist lanes (docs-only, status, task queue), skip the
|
|
|
80
80
|
- If verification is needed, prefer `node .codex/ukit/index/verify-context.mjs ...`.
|
|
81
81
|
- **Treat helper commands as internal orchestration.** Run them yourself when needed; never ask end users to run them.
|
|
82
82
|
|
|
83
|
+
## Enforcement Ceiling + Playbooks (Codex contract)
|
|
84
|
+
|
|
85
|
+
- Codex exposes **no hook surface** — routing, gates, and playbooks reach you only as instructions in this file. Every playbook-covered ask resolves to either a measured live path or this advisory fallback; nothing here is hard-blocked.
|
|
86
|
+
- Natural prompts route through `node .codex/ukit/index/route-task.mjs "<prompt>"` — the same route record as the other engines, invoked by instruction, not by hook.
|
|
87
|
+
- When a route returns a `playbookId`, read the playbook file `~/.ukit/playbooks/<id>.md` before implementing, and follow its steps.
|
|
88
|
+
- Outcome records for codex-engine runs carry `enforcement: advisory` — the honest marker that nothing was enforced.
|
|
89
|
+
- Honest ceiling: the codex lane caps at **SHADOW/advisory**. Report advisory, never a live-enforcement claim.
|
|
90
|
+
|
|
83
91
|
## UKit Shared Runtime
|
|
84
92
|
|
|
85
93
|
- Shared runtime state lives in `.ukit/storage/`.
|
|
@@ -63,7 +63,11 @@ export const HOOK_EVENT_MAP = {
|
|
|
63
63
|
},
|
|
64
64
|
before_agent_start: ['sensitive-data-guard.mjs', 'skill-router.sh', 'vision-router.sh', 'context-window-guard.sh'],
|
|
65
65
|
'session.compacting': ['reinject-context.sh'],
|
|
66
|
-
|
|
66
|
+
// TASK-003 (FR-003): the in-proc emit step runs LAST — project-important.sh
|
|
67
|
+
// owns the mandate-first lead (TASK-042 contract); the emit is advisory,
|
|
68
|
+
// <250ms, and writes the session's `execution.started` record after the
|
|
69
|
+
// context-producing scripts.
|
|
70
|
+
session_start: ['project-important.sh', 'auto-prune-bash.sh', 'reset-compact-pressure.sh', 'handoff-resume.sh', 'observability-emit.mjs'],
|
|
67
71
|
// TASK-002 (cycle 45): omp emits `session_stop` (no `session_end`); the chain
|
|
68
72
|
// runs only on RELEASED stops inside runSessionStop — never on blocked or
|
|
69
73
|
// dedupe-skipped stops, since `ukit memory episode` dedupes on meta.ledgerKey
|
|
@@ -135,6 +139,9 @@ export const ADVISORY_SCRIPTS = new Set([
|
|
|
135
139
|
'reset-compact-pressure.sh',
|
|
136
140
|
'handoff-resume.sh',
|
|
137
141
|
'session-episode.sh',
|
|
142
|
+
// TASK-003 (FR-003): session-start emit step, advisory — a telemetry fault
|
|
143
|
+
// can never block or delay session start.
|
|
144
|
+
'observability-emit.mjs',
|
|
138
145
|
]);
|
|
139
146
|
|
|
140
147
|
function classifyFailure(scriptName) {
|
|
@@ -23,7 +23,7 @@ UKit state lives at two levels:
|
|
|
23
23
|
`ukit install` and **never overwritten** afterwards — every user item is
|
|
24
24
|
`mergeStrategy: skip`. Per `manifests/platform.user.yaml` the seeds are:
|
|
25
25
|
`README.md`, `storage/config.json`, `playbooks/bug-fix.md`,
|
|
26
|
-
`playbooks/
|
|
26
|
+
`playbooks/feature-implementation.md`.
|
|
27
27
|
|
|
28
28
|
Precedence (per key / per id):
|
|
29
29
|
|
|
@@ -97,6 +97,11 @@
|
|
|
97
97
|
"stage": "default"
|
|
98
98
|
}
|
|
99
99
|
},
|
|
100
|
+
"crossCheck": {
|
|
101
|
+
"enabled": true,
|
|
102
|
+
"byRole": {},
|
|
103
|
+
"byModel": {}
|
|
104
|
+
},
|
|
100
105
|
"decisionPlane": {
|
|
101
106
|
"enabled": true,
|
|
102
107
|
"stage": "default",
|
|
@@ -229,6 +234,7 @@
|
|
|
229
234
|
"escalation": {
|
|
230
235
|
"enabled": true,
|
|
231
236
|
"debugLoopThreshold": 2,
|
|
237
|
+
"laneDeepening": true,
|
|
232
238
|
"tierOrder": [
|
|
233
239
|
"lite",
|
|
234
240
|
"code",
|
|
@@ -422,6 +428,20 @@
|
|
|
422
428
|
"stage": "default"
|
|
423
429
|
}
|
|
424
430
|
},
|
|
431
|
+
"subagentOrchestrator": {
|
|
432
|
+
"telemetry": {
|
|
433
|
+
"stage": "off"
|
|
434
|
+
},
|
|
435
|
+
"referenceHandoff": {
|
|
436
|
+
"stage": "off"
|
|
437
|
+
},
|
|
438
|
+
"roleAdvice": {
|
|
439
|
+
"stage": "off"
|
|
440
|
+
},
|
|
441
|
+
"contextSelection": {
|
|
442
|
+
"stage": "off"
|
|
443
|
+
}
|
|
444
|
+
},
|
|
425
445
|
"safePatch": {
|
|
426
446
|
"enabled": true,
|
|
427
447
|
"strictSharedRisk": true,
|