@ngockhoale/ukit 2.7.13 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/manifests/documentation.yaml +11 -0
- package/manifests/platform.full.yaml +182 -0
- package/manifests/platform.user.yaml +53 -0
- package/package.json +3 -1
- package/src/cli/commands/diff.js +4 -2
- package/src/cli/commands/doctor.js +22 -1
- package/src/cli/commands/install.js +10 -0
- package/src/cli/commands/memory.js +142 -3
- package/src/cli/commands/playbook.js +53 -0
- package/src/cli/index.js +7 -0
- package/src/core/memory/recordStore.js +81 -0
- package/src/core/memory/storeV2.js +16 -52
- package/src/core/memory/userMemory.js +111 -0
- package/src/core/paths.js +1 -0
- package/src/core/runInstallPipeline.js +96 -3
- package/src/core/runtimeConfig.js +170 -5
- package/src/core/userPaths.js +21 -0
- package/src/core/userPlaybooks.js +185 -0
- package/src/index/taskRouting.js +422 -21
- package/src/index/verificationPlan.js +17 -0
- package/src/manifest/validateManifest.js +19 -0
- package/templates/.claude/config/providers.md +1 -3
- package/templates/.claude/skills/principle-attack-the-premise/SKILL.md +16 -0
- package/templates/.claude/skills/principle-boundary-discipline/SKILL.md +16 -0
- package/templates/.claude/skills/principle-encode-lessons-in-structure/SKILL.md +16 -0
- package/templates/.claude/skills/principle-fix-root-causes/SKILL.md +18 -0
- package/templates/.claude/skills/principle-foundational-thinking/SKILL.md +17 -0
- package/templates/.claude/skills/principle-guard-the-context-window/SKILL.md +16 -0
- package/templates/.claude/skills/principle-laziness-protocol/SKILL.md +17 -0
- package/templates/.claude/skills/principle-migrate-callers-then-delete-legacy-apis/SKILL.md +16 -0
- package/templates/.claude/skills/principle-minimize-reader-load/SKILL.md +17 -0
- package/templates/.claude/skills/principle-model-the-domain/SKILL.md +16 -0
- package/templates/.claude/skills/principle-never-block-on-the-human/SKILL.md +16 -0
- package/templates/.claude/skills/principle-prove-it-works/SKILL.md +18 -0
- package/templates/.claude/skills/principle-sequence-verifiable-units/SKILL.md +16 -0
- package/templates/.claude/skills/principle-subtract-before-you-add/SKILL.md +16 -0
- package/templates/.claude/skills/principle-test-behavior-not-implementation/SKILL.md +18 -0
- package/templates/.claude/ukit/index/route-task.mjs +652 -28
- package/templates/.claude/ukit/runtime/execution-ledger.mjs +238 -9
- package/templates/ukit/README.md +31 -0
- package/templates/ukit/storage/config.json +10 -0
- package/templates/user/README.md +21 -0
- package/templates/user/playbooks/bug-fix.md +18 -0
- package/templates/user/playbooks/issue-implementation.md +14 -0
- package/templates/user/storage/config.json +16 -0
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
3
|
import crypto from 'node:crypto';
|
|
4
|
+
import { execFileSync } from 'node:child_process';
|
|
4
5
|
import fs from 'node:fs/promises';
|
|
5
6
|
import fsSync from 'node:fs';
|
|
6
7
|
import path from 'node:path';
|
|
@@ -482,7 +483,7 @@ function compactReceipt(receipt) {
|
|
|
482
483
|
kind: receipt.kind,
|
|
483
484
|
success: receipt.success,
|
|
484
485
|
};
|
|
485
|
-
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error']) {
|
|
486
|
+
for (const key of ['toolName', 'toolUseId', 'file', 'command', 'exitCode', 'scope', 'error', 'verdict']) {
|
|
486
487
|
if (receipt[key] !== undefined && receipt[key] !== null && receipt[key] !== '') {
|
|
487
488
|
compact[key] = receipt[key];
|
|
488
489
|
}
|
|
@@ -490,6 +491,139 @@ function compactReceipt(receipt) {
|
|
|
490
491
|
return compact;
|
|
491
492
|
}
|
|
492
493
|
|
|
494
|
+
// --- Typed verdicts (SPEC-typed-verdicts §2) -----------------------------------
|
|
495
|
+
// One optional verdict record rides the ledger entry — additive, no new store.
|
|
496
|
+
// kind: 'live-verified' | 'test-verified' | 'check-only' | 'blocked' | 'failed'.
|
|
497
|
+
// Currency is proven by headSha + baseSha + patch-id, not by commit messages or a
|
|
498
|
+
// green check from an older SHA.
|
|
499
|
+
const VERDICT_KINDS = new Set(['live-verified', 'test-verified', 'check-only', 'blocked', 'failed']);
|
|
500
|
+
const VERDICT_SATISFYING_KINDS = new Set(['test-verified', 'live-verified']);
|
|
501
|
+
const MAX_VERDICT_HISTORY = 8;
|
|
502
|
+
|
|
503
|
+
function gitOutput(args, cwd) {
|
|
504
|
+
try {
|
|
505
|
+
const out = execFileSync('git', args, {
|
|
506
|
+
cwd,
|
|
507
|
+
encoding: 'utf8',
|
|
508
|
+
stdio: ['ignore', 'pipe', 'ignore'],
|
|
509
|
+
timeout: 5000,
|
|
510
|
+
});
|
|
511
|
+
return String(out).trim() || null;
|
|
512
|
+
} catch {
|
|
513
|
+
return null;
|
|
514
|
+
}
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
// A diff whose changed paths are all tests/docs/lint-format config is
|
|
518
|
+
// stale-but-recoverable: one rebuild at current head restores currency.
|
|
519
|
+
const RECOVERABLE_DIFF_PATH = /(^|\/)(docs\/|[^/]*\.md$)|\.(test|spec)\.[cm]?[jt]sx?$|(^|\/)(tests?|specs?|__tests__)\/|(?:^|\/)\.?eslint|\.prettierrc|biome\.json|\.editorconfig|tsconfig.*\.json$/i;
|
|
520
|
+
|
|
521
|
+
function isRecoverableDiffPath(filePath) {
|
|
522
|
+
return RECOVERABLE_DIFF_PATH.test(String(filePath || '').replace(/\\/g, '/'));
|
|
523
|
+
}
|
|
524
|
+
|
|
525
|
+
// `git diff` names changed files in `diff --git a/<old> b/<new>` headers — parse them
|
|
526
|
+
// instead of spending a third git call on --name-only.
|
|
527
|
+
function diffChangedPaths(diffText) {
|
|
528
|
+
const paths = new Set();
|
|
529
|
+
for (const line of String(diffText || '').split('\n')) {
|
|
530
|
+
const match = line.match(/^diff --git a\/(.+) b\/(.+)$/);
|
|
531
|
+
if (match) {
|
|
532
|
+
paths.add(match[1]);
|
|
533
|
+
paths.add(match[2]);
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
return [...paths];
|
|
537
|
+
}
|
|
538
|
+
|
|
539
|
+
function gitPatchId(diffText) {
|
|
540
|
+
if (!diffText) return null;
|
|
541
|
+
try {
|
|
542
|
+
const out = execFileSync('git', ['patch-id', '--stable'], {
|
|
543
|
+
input: diffText,
|
|
544
|
+
encoding: 'utf8',
|
|
545
|
+
stdio: ['pipe', 'pipe', 'ignore'],
|
|
546
|
+
timeout: 5000,
|
|
547
|
+
});
|
|
548
|
+
return String(out).trim().split(/\s+/)[0] || null;
|
|
549
|
+
} catch {
|
|
550
|
+
return null;
|
|
551
|
+
}
|
|
552
|
+
}
|
|
553
|
+
|
|
554
|
+
// SPEC §2.2 — is this verdict still valid at the current head?
|
|
555
|
+
// Returns { status, current, headSha } where status is one of:
|
|
556
|
+
// 'current' | 'stale' | 'stale-recoverable' | 'untracked' (no git / no sha data).
|
|
557
|
+
// Cost: one `rev-parse` fast path; when head moved, one `git diff` + one `patch-id`.
|
|
558
|
+
export function verdictCurrent(verdict, { cwd = process.cwd() } = {}) {
|
|
559
|
+
if (!verdict || typeof verdict !== 'object') {
|
|
560
|
+
return { status: 'untracked', current: false, headSha: null };
|
|
561
|
+
}
|
|
562
|
+
const currentHead = gitOutput(['rev-parse', 'HEAD'], cwd);
|
|
563
|
+
if (!currentHead) {
|
|
564
|
+
// No git repo (or git unavailable): currency cannot be proven either way —
|
|
565
|
+
// null-safe per spec non-goals; do not invent staleness on lanes without git.
|
|
566
|
+
return { status: 'untracked', current: true, headSha: null };
|
|
567
|
+
}
|
|
568
|
+
if (!verdict.headSha) {
|
|
569
|
+
return { status: 'untracked', current: true, headSha: currentHead };
|
|
570
|
+
}
|
|
571
|
+
if (verdict.headSha === currentHead) {
|
|
572
|
+
return { status: 'current', current: true, headSha: currentHead };
|
|
573
|
+
}
|
|
574
|
+
// Head moved. With no base/patch-id recorded there is nothing to compare — stale.
|
|
575
|
+
if (!verdict.baseSha || !verdict.patchId) {
|
|
576
|
+
return { status: 'stale', current: false, headSha: currentHead };
|
|
577
|
+
}
|
|
578
|
+
const diff = gitOutput(['diff', `${verdict.baseSha}..${currentHead}`], cwd);
|
|
579
|
+
const patchId = gitPatchId(diff);
|
|
580
|
+
if (patchId && patchId === verdict.patchId) {
|
|
581
|
+
// Rebase/reorder only — same patch, still current.
|
|
582
|
+
return { status: 'current', current: true, headSha: currentHead };
|
|
583
|
+
}
|
|
584
|
+
// Recoverability is judged on the delta SINCE the verdict (headSha..HEAD): when
|
|
585
|
+
// only tests/docs/lint-config moved, one rebuild at current head restores currency.
|
|
586
|
+
const delta = gitOutput(['diff', `${verdict.headSha}..${currentHead}`], cwd);
|
|
587
|
+
const changed = diffChangedPaths(delta);
|
|
588
|
+
if (changed.length > 0 && changed.every(isRecoverableDiffPath)) {
|
|
589
|
+
return { status: 'stale-recoverable', current: false, headSha: currentHead };
|
|
590
|
+
}
|
|
591
|
+
return { status: 'stale', current: false, headSha: currentHead };
|
|
592
|
+
}
|
|
593
|
+
|
|
594
|
+
// Mint the verdict record for a verification receipt. Explicit payload fields win
|
|
595
|
+
// (a hook or subagent can attest its own kind/verifier); otherwise the kind is
|
|
596
|
+
// derived from the command shape — test runners prove `test-verified`, a run that
|
|
597
|
+
// only type-checks/lints is honestly `check-only`.
|
|
598
|
+
const TEST_RUNNER_COMMAND = /(?:^|\s)(?:vitest|jest|mocha|ava|pytest|py\.test)(?:\s|$)|(?:npm|pnpm|yarn|bun)(?:\s+run)?\s+test(?:\s|$)/i;
|
|
599
|
+
|
|
600
|
+
function mintVerificationVerdict(receipt, payload, projectRoot) {
|
|
601
|
+
const supplied = payload?.verdict && typeof payload.verdict === 'object' ? payload.verdict : {};
|
|
602
|
+
const cwd = projectRoot || payload?.cwd || process.cwd();
|
|
603
|
+
const headSha = supplied.headSha || gitOutput(['rev-parse', 'HEAD'], cwd);
|
|
604
|
+
const baseSha = supplied.baseSha !== undefined
|
|
605
|
+
? supplied.baseSha
|
|
606
|
+
: (headSha ? gitOutput(['merge-base', 'HEAD', '@{upstream}'], cwd) : null);
|
|
607
|
+
let patchId = supplied.patchId || null;
|
|
608
|
+
if (!patchId && headSha && baseSha) {
|
|
609
|
+
patchId = gitPatchId(gitOutput(['diff', `${baseSha}..${headSha}`], cwd));
|
|
610
|
+
}
|
|
611
|
+
let kind = supplied.kind;
|
|
612
|
+
if (!VERDICT_KINDS.has(kind)) {
|
|
613
|
+
if (receipt.success !== true) kind = 'failed';
|
|
614
|
+
else kind = TEST_RUNNER_COMMAND.test(receipt.command || '') ? 'test-verified' : 'check-only';
|
|
615
|
+
}
|
|
616
|
+
return {
|
|
617
|
+
kind,
|
|
618
|
+
evidence: supplied.evidence || receipt.command || null,
|
|
619
|
+
verifier: supplied.verifier || 'self',
|
|
620
|
+
headSha: headSha || null,
|
|
621
|
+
baseSha: baseSha || null,
|
|
622
|
+
patchId,
|
|
623
|
+
ts: supplied.ts || new Date().toISOString(),
|
|
624
|
+
};
|
|
625
|
+
}
|
|
626
|
+
|
|
493
627
|
/**
|
|
494
628
|
* Target-aware matching between a receipt's file and a routed expected file. Both sides
|
|
495
629
|
* may disagree about absolute vs relative spelling (hook payloads carry absolute paths,
|
|
@@ -735,6 +869,11 @@ function carriedEvidenceLedger(fresh, current) {
|
|
|
735
869
|
verificationAttempted: fresh.verificationAttempted || current.verificationAttempted === true,
|
|
736
870
|
verificationSucceeded: fresh.verificationSucceeded || current.verificationSucceeded === true,
|
|
737
871
|
verificationFailed: fresh.verificationFailed || current.verificationFailed === true,
|
|
872
|
+
// Typed verdicts belong to the same logical request: a re-key must keep the latest
|
|
873
|
+
// verdict and its history or the gate would re-demand verification mid-request.
|
|
874
|
+
verdict: fresh.verdict || current.verdict || null,
|
|
875
|
+
verdictHistory: [...(current.verdictHistory || []), ...(fresh.verdictHistory || [])]
|
|
876
|
+
.slice(-MAX_VERDICT_HISTORY),
|
|
738
877
|
receipts: [...(current.receipts || [])].slice(-MAX_RECEIPTS),
|
|
739
878
|
// The continuation budget belongs to the same logical request (promptKey), so a re-key
|
|
740
879
|
// must keep counting toward the cap. Resetting it here made the cap unreachable and the
|
|
@@ -800,6 +939,8 @@ function freshLedger(payload, routeState, harness) {
|
|
|
800
939
|
verificationFailed: false,
|
|
801
940
|
receipts: [],
|
|
802
941
|
blocker: null,
|
|
942
|
+
verdict: null,
|
|
943
|
+
verdictHistory: [],
|
|
803
944
|
continuationCount: 0,
|
|
804
945
|
noProgressCount: 0,
|
|
805
946
|
lastProgressDigest: null,
|
|
@@ -1181,6 +1322,13 @@ function applyReceiptToLedger(ledger, receipt, { vibecode = false } = {}) {
|
|
|
1181
1322
|
);
|
|
1182
1323
|
if (banked) next.bankedVerifications = banked;
|
|
1183
1324
|
}
|
|
1325
|
+
// SPEC-typed-verdicts §2.1: the latest verdict is the ledger's verdict; superseded
|
|
1326
|
+
// ones are kept in a bounded history so "one rebuild restores currency" has data.
|
|
1327
|
+
if (receipt.verdict && typeof receipt.verdict === 'object') {
|
|
1328
|
+
const history = Array.isArray(next.verdictHistory) ? next.verdictHistory : [];
|
|
1329
|
+
next.verdictHistory = [...history, ...(next.verdict ? [next.verdict] : [])].slice(-MAX_VERDICT_HISTORY);
|
|
1330
|
+
next.verdict = receipt.verdict;
|
|
1331
|
+
}
|
|
1184
1332
|
|
|
1185
1333
|
// Verification-loop tracking. `terminalShellCommandUnit` gives the loop identity:
|
|
1186
1334
|
// the same failing check rerun — `setup && yarn test` and `yarn test 2>&1 | tail`
|
|
@@ -1420,6 +1568,10 @@ export async function recordExecutionReceipt({
|
|
|
1420
1568
|
const matched = routedCommands.find((cmd) => matchesRoutedCommand(receipt.command, cmd));
|
|
1421
1569
|
receipt.scope = matched ? 'targeted' : 'broad';
|
|
1422
1570
|
}
|
|
1571
|
+
// SPEC-typed-verdicts §2.1: every verification receipt mints a typed verdict record
|
|
1572
|
+
// (kind/evidence/verifier/headSha/baseSha/patchId/ts). Explicit payload.verdict
|
|
1573
|
+
// fields win so a hook or subagent verifier can attest its own kind and identity.
|
|
1574
|
+
receipt.verdict = mintVerificationVerdict(receipt, payload, projectRoot);
|
|
1423
1575
|
} else {
|
|
1424
1576
|
// Untracked tool: no event, no lock, no write — same as the old unlocked early return.
|
|
1425
1577
|
return { rejected: true, eventId: null };
|
|
@@ -1482,10 +1634,21 @@ function requiredEvidence(state = {}) {
|
|
|
1482
1634
|
return [...new Set(routeSummary.completionState?.missingEvidence || [])];
|
|
1483
1635
|
}
|
|
1484
1636
|
|
|
1485
|
-
function evidenceSatisfied(evidence, ledger = {}, state = {}) {
|
|
1637
|
+
function evidenceSatisfied(evidence, ledger = {}, state = {}, { cwd } = {}) {
|
|
1486
1638
|
const routeSummary = state?.routeSummary || {};
|
|
1487
1639
|
if (evidence === 'write-evidence') return ledger.writeSucceeded === true;
|
|
1488
1640
|
if (evidence === 'verification-evidence') {
|
|
1641
|
+
// SPEC-typed-verdicts §2.3: when a verdict record exists it governs — only a
|
|
1642
|
+
// CURRENT test-verified/live-verified verdict satisfies the gate. A stale or
|
|
1643
|
+
// check-only verdict does not, no matter what the legacy booleans say. With no
|
|
1644
|
+
// verdict record the boolean path below is unchanged (backward compatible).
|
|
1645
|
+
const verdicts = [ledger.verdict, ...(ledger.verdictHistory || [])].filter(Boolean);
|
|
1646
|
+
if (verdicts.length > 0) {
|
|
1647
|
+
return verdicts.some(
|
|
1648
|
+
(verdict) => VERDICT_SATISFYING_KINDS.has(verdict.kind)
|
|
1649
|
+
&& verdictCurrent(verdict, { cwd }).current === true,
|
|
1650
|
+
);
|
|
1651
|
+
}
|
|
1489
1652
|
// WS-C: when the route names concrete verification commands, only a receipt that ran
|
|
1490
1653
|
// one of them counts — an unrelated `yarn test` no longer satisfies the gate.
|
|
1491
1654
|
// F-15: unless the user explicitly requested/approved a broad suite — the same
|
|
@@ -1513,7 +1676,7 @@ function evidenceSatisfied(evidence, ledger = {}, state = {}) {
|
|
|
1513
1676
|
return false;
|
|
1514
1677
|
}
|
|
1515
1678
|
|
|
1516
|
-
function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}) {
|
|
1679
|
+
function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}, { cwd } = {}) {
|
|
1517
1680
|
let instruction = null;
|
|
1518
1681
|
if (missingEvidence.includes('write-evidence')) {
|
|
1519
1682
|
if (!ledger.sourceSucceeded) {
|
|
@@ -1524,9 +1687,23 @@ function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}) {
|
|
|
1524
1687
|
instruction = 'Make the smallest correct Edit/Write now; do not end after more read-only analysis.';
|
|
1525
1688
|
}
|
|
1526
1689
|
} else if (missingEvidence.includes('verification-evidence')) {
|
|
1527
|
-
|
|
1690
|
+
// SPEC-typed-verdicts §2.3: a verdict record names its own recovery — stale gets
|
|
1691
|
+
// the stale-specific instruction, check-only names the missing surface. These
|
|
1692
|
+
// take precedence over the generic "run verification" wording below.
|
|
1693
|
+
if (!instruction && ledger.verdict && typeof ledger.verdict === 'object') {
|
|
1694
|
+
if (ledger.verdict.kind === 'check-only') {
|
|
1695
|
+
instruction = 'check-only does not satisfy this lane; verify on the real surface (run the feature / inspect the artifact).';
|
|
1696
|
+
} else {
|
|
1697
|
+
const currency = verdictCurrent(ledger.verdict, { cwd });
|
|
1698
|
+
if (currency.current !== true) {
|
|
1699
|
+
const sha = ledger.verdict.headSha ? String(ledger.verdict.headSha).slice(0, 12) : 'unknown';
|
|
1700
|
+
instruction = `Verification verdict is stale (head moved since ${sha}). Re-run the routed verification at current head.`;
|
|
1701
|
+
}
|
|
1702
|
+
}
|
|
1703
|
+
}
|
|
1704
|
+
if (!instruction && ledger.verificationFailed) {
|
|
1528
1705
|
instruction = 'Use the latest failed verification output, fix the failure, and rerun targeted verification.';
|
|
1529
|
-
} else {
|
|
1706
|
+
} else if (!instruction) {
|
|
1530
1707
|
const routedCommands = routedVerificationCommands(routeSummary);
|
|
1531
1708
|
if (routedCommands.length > 0 && ledger.verificationSucceeded && !ledger.targetedVerificationSucceeded) {
|
|
1532
1709
|
instruction = `Verification ran but did not match the routed plan (${routedCommands.slice(0, 3).join('; ')}). Run one of those and inspect its result before stopping.`;
|
|
@@ -1549,7 +1726,7 @@ function recoveryInstruction(missingEvidence, ledger = {}, routeSummary = {}) {
|
|
|
1549
1726
|
return instruction ?? 'Complete the current routed milestone before stopping.';
|
|
1550
1727
|
}
|
|
1551
1728
|
|
|
1552
|
-
export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
1729
|
+
export function evaluateCompletion({ state = {}, ledger = {}, cwd } = {}) {
|
|
1553
1730
|
const routeSummary = state?.routeSummary || {};
|
|
1554
1731
|
// Explicit vibecode autonomy: the user asked UKit to run one prompt to a finished result,
|
|
1555
1732
|
// so the completion gate keeps pushing instead of releasing at the continuation cap.
|
|
@@ -1620,7 +1797,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
1620
1797
|
|| ledger.requestKey === state.requestKey
|
|
1621
1798
|
|| (ledger.promptKey && evidencePromptKey(state) === ledger.promptKey);
|
|
1622
1799
|
const effectiveLedger = sameRequest ? ledger : {};
|
|
1623
|
-
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, state));
|
|
1800
|
+
const missingEvidence = evidence.filter((item) => !evidenceSatisfied(item, effectiveLedger, state, { cwd }));
|
|
1624
1801
|
if (missingEvidence.length === 0) {
|
|
1625
1802
|
// Silent success: the route is present and every required evidence is satisfied. Marked
|
|
1626
1803
|
// `complete` so the CLI dispatch recognizes it BEFORE the loud final else — otherwise a
|
|
@@ -1742,7 +1919,7 @@ export function evaluateCompletion({ state = {}, ledger = {} } = {}) {
|
|
|
1742
1919
|
}
|
|
1743
1920
|
|
|
1744
1921
|
const finalAttempt = !vibecode && continuationCount === MAX_CONTINUATIONS - 1;
|
|
1745
|
-
const instruction = recoveryInstruction(missingEvidence, effectiveLedger, routeSummary);
|
|
1922
|
+
const instruction = recoveryInstruction(missingEvidence, effectiveLedger, routeSummary, { cwd });
|
|
1746
1923
|
return {
|
|
1747
1924
|
continue: true,
|
|
1748
1925
|
missingEvidence,
|
|
@@ -1886,7 +2063,7 @@ async function runEvaluateStop() {
|
|
|
1886
2063
|
return;
|
|
1887
2064
|
}
|
|
1888
2065
|
|
|
1889
|
-
const result = evaluateCompletion({ state, ledger });
|
|
2066
|
+
const result = evaluateCompletion({ state, ledger, cwd: projectRoot });
|
|
1890
2067
|
|
|
1891
2068
|
// A reentrant Stop (Claude Code re-fires Stop after this hook already blocked once) is
|
|
1892
2069
|
// gated exactly like any other Stop: while evidence is still missing it blocks again with
|
|
@@ -1937,6 +2114,58 @@ async function main() {
|
|
|
1937
2114
|
// Compare real paths: a project under a symlinked root (macOS /tmp -> /private/tmp, or a
|
|
1938
2115
|
// symlinked work directory) makes import.meta.url resolve to the real path while argv[1]
|
|
1939
2116
|
// keeps the symlinked spelling. Without realpath the guard silently never runs the CLI.
|
|
2117
|
+
// --- decision log (v3, pstack show-me-your-work port) -------------------------
|
|
2118
|
+
// Append-only TSV at .ukit/storage/decisions.tsv. Columns: ts / phase / decision /
|
|
2119
|
+
// why / evidence / result. Evidence is a pointer (SHA, file:line, artifact path),
|
|
2120
|
+
// never prose. A wrong call is a NEW row whose result supersedes — there is no
|
|
2121
|
+
// update/delete path by design.
|
|
2122
|
+
|
|
2123
|
+
const DECISION_LOG_HEADER = 'ts\tphase\tdecision\twhy\tevidence\tresult';
|
|
2124
|
+
|
|
2125
|
+
function decisionLogPath(projectRoot) {
|
|
2126
|
+
return path.join(projectRoot, '.ukit', 'storage', 'decisions.tsv');
|
|
2127
|
+
}
|
|
2128
|
+
|
|
2129
|
+
// Cells may carry attacker-influenceable strings (PR titles, filenames). A cell
|
|
2130
|
+
// starting with = + - @ executes as a formula when a reviewer opens the TSV in a
|
|
2131
|
+
// spreadsheet — prefix with ' (pstack log.sh does the same).
|
|
2132
|
+
function sanitizeDecisionCell(value) {
|
|
2133
|
+
let cell = String(value ?? '').replace(/[\t\n\r]+/g, ' ').trim();
|
|
2134
|
+
if (/^[=+\-@]/.test(cell)) cell = `'${cell}`;
|
|
2135
|
+
return cell;
|
|
2136
|
+
}
|
|
2137
|
+
|
|
2138
|
+
export async function appendDecision(projectRoot, {
|
|
2139
|
+
phase = '',
|
|
2140
|
+
decision = '',
|
|
2141
|
+
why = '',
|
|
2142
|
+
evidence = '',
|
|
2143
|
+
result = '',
|
|
2144
|
+
now = Date.now(),
|
|
2145
|
+
} = {}) {
|
|
2146
|
+
if (!projectRoot || !decision) return { appended: false, reason: 'invalid-decision' };
|
|
2147
|
+
const target = decisionLogPath(projectRoot);
|
|
2148
|
+
const row = [
|
|
2149
|
+
new Date(now).toISOString(),
|
|
2150
|
+
phase,
|
|
2151
|
+
decision,
|
|
2152
|
+
why,
|
|
2153
|
+
evidence,
|
|
2154
|
+
result,
|
|
2155
|
+
].map(sanitizeDecisionCell).join('\t');
|
|
2156
|
+
try {
|
|
2157
|
+
await fs.mkdir(path.dirname(target), { recursive: true });
|
|
2158
|
+
let needsHeader = true;
|
|
2159
|
+
try {
|
|
2160
|
+
needsHeader = (await fs.stat(target)).size === 0;
|
|
2161
|
+
} catch {}
|
|
2162
|
+
await fs.appendFile(target, `${needsHeader ? `${DECISION_LOG_HEADER}\n` : ''}${row}\n`, 'utf8');
|
|
2163
|
+
return { appended: true, path: target };
|
|
2164
|
+
} catch (error) {
|
|
2165
|
+
return { appended: false, reason: error?.message ?? String(error) };
|
|
2166
|
+
}
|
|
2167
|
+
}
|
|
2168
|
+
|
|
1940
2169
|
function isDirectRun() {
|
|
1941
2170
|
const argvPath = process.argv[1];
|
|
1942
2171
|
if (!argvPath) return false;
|
package/templates/ukit/README.md
CHANGED
|
@@ -12,3 +12,34 @@ The compact-pressure state is used to trigger conservative auto-compact behavior
|
|
|
12
12
|
- preserve active task, rules, decisions, and current code focus first
|
|
13
13
|
|
|
14
14
|
This folder is generated by `ukit install` and is safe to keep local-only.
|
|
15
|
+
|
|
16
|
+
## Two layers
|
|
17
|
+
|
|
18
|
+
UKit state lives at two levels:
|
|
19
|
+
|
|
20
|
+
- **Project level** — `<project>/.ukit/` (this folder). Refreshed by every
|
|
21
|
+
`ukit install` / `ukit update`; treat it as generated.
|
|
22
|
+
- **User level** — `~/.ukit/` (your home directory). Seeded once by
|
|
23
|
+
`ukit install` and **never overwritten** afterwards — every user item is
|
|
24
|
+
`mergeStrategy: skip`. Per `manifests/platform.user.yaml` the seeds are:
|
|
25
|
+
`README.md`, `storage/config.json`, `playbooks/bug-fix.md`,
|
|
26
|
+
`playbooks/issue-implementation.md`.
|
|
27
|
+
|
|
28
|
+
Precedence (per key / per id):
|
|
29
|
+
|
|
30
|
+
- **Config**: built-in defaults < `~/.ukit/storage/config.json` <
|
|
31
|
+
`.ukit/storage/config.json` — project wins every key. `version` and `agent`
|
|
32
|
+
are project-managed and ignored in the user file.
|
|
33
|
+
- **Playbooks**: built-in < `~/.ukit/playbooks/` < `.ukit/playbooks/` — a
|
|
34
|
+
project playbook with the same id overrides the user's for that project.
|
|
35
|
+
- **Memory**: project records ++ user records — the user record wins on id
|
|
36
|
+
collision.
|
|
37
|
+
|
|
38
|
+
CLI surface:
|
|
39
|
+
|
|
40
|
+
- `ukit memory --user [v2] <list|add|get|update|forget|stats>` — operate on the
|
|
41
|
+
user-level memory store (`~/.ukit/storage/memory/v2/records.json`).
|
|
42
|
+
- `ukit playbook list` / `ukit playbook show <id>` — inspect resolved playbooks
|
|
43
|
+
across all three levels.
|
|
44
|
+
- `ukit doctor` — prints a `User layer:` health line (presence, config state,
|
|
45
|
+
playbook and memory-record counts).
|
|
@@ -83,6 +83,16 @@
|
|
|
83
83
|
"enabled": true,
|
|
84
84
|
"defaultModel": "claude-sonnet-5"
|
|
85
85
|
},
|
|
86
|
+
"routing": {
|
|
87
|
+
"routeSchema": { "stage": "off" }
|
|
88
|
+
},
|
|
89
|
+
"modelRoles": {
|
|
90
|
+
"code": "code",
|
|
91
|
+
"judgment": "smart",
|
|
92
|
+
"review-panel": ["smart", "code", "lite"],
|
|
93
|
+
"fast-worker": "lite",
|
|
94
|
+
"vision": "vision"
|
|
95
|
+
},
|
|
86
96
|
"orchestration": {
|
|
87
97
|
"enabled": true,
|
|
88
98
|
"orchestratorModel": "claude-sonnet-5",
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# UKit User Layer (~/.ukit)
|
|
2
|
+
|
|
3
|
+
This folder is **yours — ukit never overwrites it**. `ukit install` / `ukit update`
|
|
4
|
+
seed missing files only (`mergeStrategy: skip`); anything already here is left
|
|
5
|
+
untouched, so edits, additions, and deletions are safe.
|
|
6
|
+
|
|
7
|
+
## Layout
|
|
8
|
+
|
|
9
|
+
- `storage/config.json` — personal defaults merged **under** the project config
|
|
10
|
+
(`.ukit/storage/config.json`). Project keys win; `version` and `agent` are
|
|
11
|
+
project-managed and stripped from this file at load.
|
|
12
|
+
- `playbooks/<id>.md` — personal workflow policies. Frontmatter `id` + `lanes`,
|
|
13
|
+
then the markdown body.
|
|
14
|
+
- `storage/memory/` — cross-project memory (`v2/records.json`), written by
|
|
15
|
+
`ukit memory --user`.
|
|
16
|
+
|
|
17
|
+
## Precedence
|
|
18
|
+
|
|
19
|
+
Playbooks: built-in < user (`~/.ukit/playbooks/`) < project (`.ukit/playbooks/`).
|
|
20
|
+
Config: user defaults < project config (per-key merge).
|
|
21
|
+
Memory: project records ++ user records (user wins on id collision).
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
---
|
|
2
|
+
id: bug-fix
|
|
3
|
+
lanes: [find-cause]
|
|
4
|
+
---
|
|
5
|
+
You own this bug. Reproduce, root-cause, fix, verify on the same surface.
|
|
6
|
+
If the user says "new task", re-route — do not treat the message as the next step.
|
|
7
|
+
1. Reproduce it yourself on the matching surface — ask the user only with a stated
|
|
8
|
+
reason the surface cannot reach the target.
|
|
9
|
+
2. Binary-search the cause: form candidate hypotheses, rule them out until one
|
|
10
|
+
survives; confirm the surviving mechanism with runtime evidence.
|
|
11
|
+
3. Make the smallest fix that kills the mechanism — belt-and-suspenders that
|
|
12
|
+
"might help" is a hypothesis, not a fix; it does not ship.
|
|
13
|
+
4. Verify on the same surface: the original repro now passes. "Inconclusive" or
|
|
14
|
+
wrong-surface is not a pass. A unit test shows branch behavior, not bug absence.
|
|
15
|
+
5. Keep the rejected hypotheses — one line each, why ruled out.
|
|
16
|
+
Reply: what was broken, root cause, fix, how verified — paste failing-then-passing
|
|
17
|
+
repro output verbatim.
|
|
18
|
+
Ask the human only for: irreversible writes, a genuine preference call no experiment settles, or a real dead end. Everything else: do it, report it.
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
---
|
|
2
|
+
id: issue-implementation
|
|
3
|
+
lanes: [local-build, shared-edit, map-impact]
|
|
4
|
+
---
|
|
5
|
+
You own this task. Normalize the goal, build, verify.
|
|
6
|
+
If the user says "new task", re-route — do not treat the message as the next step.
|
|
7
|
+
1. State the done condition as a checkable predicate before writing code.
|
|
8
|
+
2. Find the established analog — follow it unless you name why it does not fit.
|
|
9
|
+
3. Name the data shape and its organizing structure before writing logic.
|
|
10
|
+
4. Implement the smallest change satisfying the predicate.
|
|
11
|
+
5. Verify against the predicate on the real artifact — not "it compiles".
|
|
12
|
+
6. Widen once: check the impact surface the route named, no broader.
|
|
13
|
+
Reply: what changed, the predicate, the evidence it now holds.
|
|
14
|
+
Ask the human only for: irreversible writes, a genuine preference call no experiment settles, or a real dead end. Everything else: do it, report it.
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
{
|
|
2
|
+
"modelRoles": {
|
|
3
|
+
"code": "code",
|
|
4
|
+
"judgment": "smart",
|
|
5
|
+
"review-panel": ["smart", "code", "lite"],
|
|
6
|
+
"fast-worker": "lite",
|
|
7
|
+
"vision": "vision"
|
|
8
|
+
},
|
|
9
|
+
"autonomy": {
|
|
10
|
+
"level": "balanced"
|
|
11
|
+
},
|
|
12
|
+
"compact": {
|
|
13
|
+
"tokenThreshold": 150000,
|
|
14
|
+
"hardCapTokens": 500000
|
|
15
|
+
}
|
|
16
|
+
}
|