@dzhechkov/harness-cli 0.3.253 → 0.3.256
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.dz-manifest.json +15 -7
- package/README.md +177 -2
- package/dist/cli.d.ts.map +1 -1
- package/dist/cli.js +768 -28
- package/dist/cli.js.map +1 -1
- package/package.json +2 -2
- package/sbom.json +26 -6
- package/src/cli.ts +747 -24
package/dist/cli.js
CHANGED
|
@@ -6,10 +6,14 @@
|
|
|
6
6
|
import { chmodSync, existsSync, lstatSync, mkdirSync, mkdtempSync, readFileSync, readdirSync, readlinkSync, realpathSync, renameSync, rmdirSync, rmSync, statSync, symlinkSync, writeFileSync } from 'node:fs';
|
|
7
7
|
import { basename, dirname, isAbsolute, join, relative, resolve, sep } from 'node:path';
|
|
8
8
|
import { fileURLToPath } from 'node:url';
|
|
9
|
-
import { execSync, spawn } from 'node:child_process';
|
|
9
|
+
import { execFileSync, execSync, spawn } from 'node:child_process';
|
|
10
10
|
import { homedir, tmpdir } from 'node:os';
|
|
11
11
|
import { createRequire } from 'node:module';
|
|
12
|
-
import { createSkill, getSkillInfo, getWorkflow, isTargetName, listSkills, runDoctor, runInit, benchmarkSkill, benchmarkSkills, scanMcp, reconcileCapabilities, RECONCILE_BANNER, buildRegistry, discoverSkillPackDirs, checkUpstream, compareSkills, checkAllUpstream, sweepSkillDrift, syncCanonicalSkill, checkUpgrades, discoverPackages, discoverSourcePackages, fetchAllDownloads, filterByCategory, pretrain, recommend, generatePlugin, publishPackages, runSetup, runMigrate, searchRegistry, runSync, runVerify, runInitAgentsMd, runInitGeminiMd, TARGET_NAMES, buildParityMatrix, TARGET_CAPABILITIES, TARGET_SHORT_LABELS, WORKFLOW_NAMES, importEcc, recordPattern, resolveLearningBackend, storeStats, consolidateSessions, pruneNoisePatterns, lessonDeltaReport, removePatternsByIds, snapshotStore, recallHybrid, teachGuard, mirrorPatternsToVector, mirrorEntriesToVector, patternVectorEntry, readMemoryLearningConfig, promotePatterns, quarantineExpiryCandidates, pruneQuarantinePatterns, clearAgentdbQuarantine, vectorMirrorEnabled, vectorTierStatus, resolveVectorEngine, reindexVectorStore, harmonizeVectorStore, importRvfCheckpoint, statuslineData, writeFeatureAdrState, computeUsage, deriveUsageCalibration, normalizeClaudeUsageModelKey, readUsageLimits, claimCheck, summarize, queryBookKnowledge, loadStorePatternsSync, patternRecordId, loadStoreRecords, recordToPattern, bundleSkills, brainHome, listBrain, promoteProjectToBrain, updateBrainSource, queryBrain, groundPrompt, expandKu, reindexBrainVectors, buildPrimer, exportBrainSlice, importBrainSlice, registerKusToBrain, RECALL_USAGE_LOG_RELATIVE, RECALL_USAGE_LOG_MAX_BYTES, parseRecallUsageLog, buildRecallUsageReport, buildManifest, buildSbom, resolveTrustRoot, decideVerifyPolicy, generateSigningKeypair, evaluateGuard, resolveRules, auditRecord, guardExitCode, DEFAULT_RULES, parsePnpmLockImporters,
|
|
12
|
+
import { createSkill, getSkillInfo, getWorkflow, isTargetName, listSkills, runDoctor, runInit, benchmarkSkill, benchmarkSkills, scanMcp, reconcileCapabilities, RECONCILE_BANNER, buildRegistry, discoverSkillPackDirs, checkUpstream, compareSkills, checkAllUpstream, sweepSkillDrift, syncCanonicalSkill, checkUpgrades, discoverPackages, discoverSourcePackages, fetchAllDownloads, filterByCategory, pretrain, recommend, generatePlugin, publishPackages, runSetup, runMigrate, searchRegistry, runSync, runVerify, runInitAgentsMd, runInitGeminiMd, TARGET_NAMES, buildParityMatrix, TARGET_CAPABILITIES, TARGET_SHORT_LABELS, WORKFLOW_NAMES, importEcc, recordPattern, resolveLearningBackend, storeStats, consolidateSessions, pruneNoisePatterns, lessonDeltaReport, removePatternsByIds, snapshotStore, recallHybrid, teachGuard, mirrorPatternsToVector, mirrorEntriesToVector, patternVectorEntry, readMemoryLearningConfig, promotePatterns, quarantineExpiryCandidates, pruneQuarantinePatterns, clearAgentdbQuarantine, vectorMirrorEnabled, vectorTierStatus, resolveVectorEngine, reindexVectorStore, harmonizeVectorStore, importRvfCheckpoint, statuslineData, writeFeatureAdrState, computeUsage, deriveUsageCalibration, normalizeClaudeUsageModelKey, readUsageLimits, claimCheck, summarize, queryBookKnowledge, loadStorePatternsSync, patternRecordId, loadStoreRecords, recordToPattern, bundleSkills, brainHome, listBrain, promoteProjectToBrain, updateBrainSource, queryBrain, groundPrompt, expandKu, reindexBrainVectors, buildPrimer, exportBrainSlice, importBrainSlice, registerKusToBrain, RECALL_USAGE_LOG_RELATIVE, RECALL_USAGE_LOG_MAX_BYTES, parseRecallUsageLog, buildRecallUsageReport, buildManifest, buildSbom, resolveTrustRoot, decideVerifyPolicy, generateSigningKeypair, evaluateGuard, resolveRules, auditRecord, guardExitCode, DEFAULT_RULES, parsePnpmLockImporters,
|
|
13
|
+
// guard-promotion (feature guard-promotion, scout idea #1)
|
|
14
|
+
assembleCandidates, renderPromotionReport, renderPromotionAdr, normalizePromotionState, nextPromotionState, globMatch, promotionAdrRelPath, DEFAULT_WINDOW_DAYS, DEFAULT_PERIODS, MAX_CONTENT_FETCHES, BUILTIN_COVERAGE, decideProvenance, isInsideTree, signManifest, verifyManifest, listSignablePackFiles, assertKeyOutsideTree, decidePublishGate, collectPackageFacts, planReleaseGates, selectAffectedPackages, classifyGateExecutions, buildFailureIssue, buildReleaseNotes, releaseTagName, firstOutputLine, formatPublishError, MANIFEST_NAME, SBOM_NAME, buildArchitectureMap, renderMapHuman, findArchitectureDrift, renderDriftReport, scanWorkspacePackages, loadSubsystemManifest, loadProductVision, checkFeatureAgainstArchitecture, renderArchCheck, planProjectSkills, guidanceForStage, renderInjectionReport, analyzeCorpus, renderRakeReport, renderCriticSection, rakeAsLesson, rakeReward, DEFAULT_RAKE_THRESHOLDS, streamSessionEvents, findLatestTranscript, detectProcessRakes, buildRetro, renderRetro, retroLessonText, PROCESS_SIGNATURES, RETRO_DOMAIN, scanForSetup, buildSetupPlan, scaffoldFromSpec, renderScaffoldPreview, readExistingForScaffold, assembleChallengeContext, buildChallengeBrief, planDiscriminationCheck, classifyDiscrimination, pickAdversaryModel, CHALLENGE_QUESTIONS, loadOutcomes, renderOutcomes, statsForKey, selectAutoCost, recordProvisional, finalizeOutcome, COST_LADDER, splitScenarios, budgetPlan, selectWinner, proseScopeOk, renderProseDiff, readScenarioIds, DEFAULT_MAX_JUDGE_RUNS, collectDeliveryFacts, planDeliveryCheck, renderDeliveryBrief, classifyDelivery, isUsablePlaneResult, renderDeliveryReview, scanSkillsLayout, parseInitFacts, verifyRegistration, buildContentProbePrompt, classifyContentProbe, renderContentProbe, findNonRegistrableSkillDirs, assembleCompoundingReport,
|
|
15
|
+
// Cold-vs-warm EPOCH RUNNER (feature epoch-replay) — orchestrates + scores, never calls a model.
|
|
16
|
+
replayableInstances, buildWorkOrder, buildJudgePrompts, unblindJudgments, verifyWorkOrder, isValidMargin, DIGEST_HONEST_SCOPE, scoreEpochReplay, generateMockOutcomes, renderEpochReplayResult, renderWorkOrderSummary, renderJudgePromptsSummary, WORK_ORDER_KIND, DEFAULT_MOCK_N, DEFAULT_MOCK_SEED, scoreRun, renderScorecard, renderCompoundingReport, readReinforcementState, readQuarantineState, registrationExitCode, renderRegistrationReport,
|
|
13
17
|
// Smart Backlog (feature smart-backlog) — goal-directed idea pipeline over the Brain vector engine.
|
|
14
18
|
readBacklogConfig, readIdeas, writeIdeas, ideaId, dedupIdea, readGoalMap, readGoalMapDetailed, parseEffort, ensureBacklogGitignored, isSafeId, alignIdea, mirrorIdeaVector, snapshotIdeas, spinRoulette, rankRoulette, seededRng, eligibleIdeas, stageEnrichment, buildJiraDraft, resolveJiraAdapter, makeBacklogIO, harmonizeBacklog, BACKLOG_BACKENDS, } from '@dzhechkov/harness-core';
|
|
15
19
|
import { getPreset, PRESET_NAMES } from '@dzhechkov/harness-presets';
|
|
@@ -39,6 +43,10 @@ Usage:
|
|
|
39
43
|
dz delivery-check --slug <slug> [--context-only] [--findings <f.json>] [--strict] [--author <model>] [--json] (portable Step-10 Delivery Gate: prints the 4-plane review brief + artifact probes; --findings classifies a fed-back review into a fail-closed ready|blocked hand-off and writes features/<slug>/10_delivery_review.md; --strict exits 1 on blocked)
|
|
40
44
|
dz skills-verify [--dir <project>] [--expect a,b] [--static] [--strict] [--json] (does .claude/skills/ actually REGISTER? --static = instant layout scan for CI; default reads the authoritative system/init listing from a real session. exit 0 pass / 1 fail / 2 inconclusive — never a false pass)
|
|
41
45
|
dz compounding [--project <dir>] [--json] (honest learning-loop payoff report: pool write-only ratio, guard repeat-violation trajectory, cold-vs-warm replay readiness, instrumentation health — a gate without enough data says INSUFFICIENT_DATA, never a fake verdict)
|
|
46
|
+
dz epoch-replay --mock [--n <N>] [--effect <-1..1>] [--tie-rate <0..1>] [--seed <N>] [--slice <name>] [--json] ($0 synthetic run — exercises the verdict math, NOT evidence)
|
|
47
|
+
dz epoch-replay --emit [--project <dir>] [--limit <N>] [--seed <N>] [--out <file>] (cold-vs-warm work order: instances + PRE-REGISTERED blind A/B assignment; the runner never calls a model)
|
|
48
|
+
dz epoch-replay --judge <filled-work-order.json> [--out <file>] (blind judge prompts from the filled plans)
|
|
49
|
+
dz epoch-replay --score <judgments.json> --work-order <file> [--slice <name>] [--json] (un-blind against the pre-registered assignment → SUPPORTED only when the two 95% Wilson CIs are DISJOINT, else FALSIFIED / INCONCLUSIVE)
|
|
42
50
|
dz score --slug <feature> [--project <dir>] [--json] (process scorecard for ONE feature-adr run, from its artifacts: ADR confirmation, discrimination, cross-model QE grade, live verification, README-first, learning loop, amendments — descriptive-only, a low score exits 0)
|
|
43
51
|
dz backlog add "<idea>" [--effort 1-5] [--proposal <text>] [--dry-run] [--project <dir>] [--json] (capture an idea: semantic dedup against existing ideas via the Brain vector engine (DUPLICATE>=0.92 merges, RELATED links, NEW creates) + GoalMap alignment; --dry-run classifies without writing)
|
|
44
52
|
dz backlog list [--status <s>] [--goal <id>] [--project <dir>] [--json] (list captured ideas, filterable by status/goal)
|
|
@@ -4599,6 +4607,47 @@ function gatherGuardFacts(op, root, text, storeCap) {
|
|
|
4599
4607
|
catch {
|
|
4600
4608
|
facts['lockfile'] = { parsed: false }; /* no lockfile (not a pnpm workspace) — rule stays silent */
|
|
4601
4609
|
}
|
|
4610
|
+
// change: the working-tree diff, for PROMOTED (template) rules. Without this fact a rule written
|
|
4611
|
+
// by `dz guard promote --apply` would be INERT — present in the config and enforcing nothing.
|
|
4612
|
+
// Contents are read only for the globs an active `format-match` rule actually asks about.
|
|
4613
|
+
try {
|
|
4614
|
+
const status = execSync('git status --porcelain', { cwd: root, encoding: 'utf-8' });
|
|
4615
|
+
const files = status
|
|
4616
|
+
.split('\n')
|
|
4617
|
+
.map((l) => l.slice(3).trim())
|
|
4618
|
+
.map((p) => (p.includes(' -> ') ? p.split(' -> ')[1].trim() : p)) // renames: the destination is the changed path
|
|
4619
|
+
.filter((p) => p !== '');
|
|
4620
|
+
const formatGlobs = (Array.isArray(loadGuardConfig(root).rules) ? loadGuardConfig(root).rules : [])
|
|
4621
|
+
.filter((r) => r?.template === 'format-match' && typeof r?.params?.file === 'string')
|
|
4622
|
+
.map((r) => r.params.file);
|
|
4623
|
+
const contents = {};
|
|
4624
|
+
if (formatGlobs.length > 0) {
|
|
4625
|
+
for (const f of files) {
|
|
4626
|
+
if (!formatGlobs.some((g) => globMatch(g, f)))
|
|
4627
|
+
continue;
|
|
4628
|
+
const abs = resolve(root, f);
|
|
4629
|
+
// Containment: a `git status` path is repo-relative, but `..` in one must never let the
|
|
4630
|
+
// LIVE reader step outside the repo the HISTORICAL reader is confined to.
|
|
4631
|
+
if (abs !== root && !abs.startsWith(root + sep))
|
|
4632
|
+
continue;
|
|
4633
|
+
try {
|
|
4634
|
+
// lstat, NOT stat (Codex QE MED-3). `git show <sha>:<path>` yields the SYMLINK TARGET
|
|
4635
|
+
// TEXT, never the file it points at, so a live reader that follows links answers a
|
|
4636
|
+
// different question than the replay — and `/dev/zero` behind a symlink hangs the read.
|
|
4637
|
+
// Skipping non-regular files restores replay/live equivalence and closes the DoS.
|
|
4638
|
+
const st = lstatSync(abs);
|
|
4639
|
+
if (!st.isFile())
|
|
4640
|
+
continue;
|
|
4641
|
+
if (st.size > MAX_CONTENT_BYTES)
|
|
4642
|
+
continue; // too large to be a spec file — undecidable, never guessed
|
|
4643
|
+
contents[f] = readFileSync(abs, 'utf8');
|
|
4644
|
+
}
|
|
4645
|
+
catch { /* deleted — leave it undecidable, never guess */ }
|
|
4646
|
+
}
|
|
4647
|
+
}
|
|
4648
|
+
facts['change'] = { files, ...(Object.keys(contents).length > 0 ? { contents } : {}) };
|
|
4649
|
+
}
|
|
4650
|
+
catch { /* not a git repo — every template rule stays silent (fail-open) */ }
|
|
4602
4651
|
}
|
|
4603
4652
|
if (op === 'consolidate') {
|
|
4604
4653
|
try {
|
|
@@ -4642,6 +4691,373 @@ function runGuardEvaluation(root, op, text, overrideReason) {
|
|
|
4642
4691
|
catch { /* audit is best-effort, never blocks the verdict */ }
|
|
4643
4692
|
return result;
|
|
4644
4693
|
}
|
|
4694
|
+
// ── `dz guard promote` (feature guard-promotion, scout idea #1) ─────────────────────────────────
|
|
4695
|
+
const PROMOTIONS_DIR = join('features', 'guard-promotion', 'promotions');
|
|
4696
|
+
const PROMOTION_STATE_FILE = join('.dz', 'promotion-state.json');
|
|
4697
|
+
/**
|
|
4698
|
+
* Ceiling on a single file read, applied IDENTICALLY to the historical (`git show`) and live
|
|
4699
|
+
* (working-tree) readers. A `format-match` target is a spec/manifest file; anything larger is not
|
|
4700
|
+
* one, and an unbounded read of a symlinked `/dev/zero` is a hang, not a measurement.
|
|
4701
|
+
*/
|
|
4702
|
+
const MAX_CONTENT_BYTES = 1024 * 1024;
|
|
4703
|
+
const GUARD_PROMOTE_USAGE = [
|
|
4704
|
+
'dz guard promote [--project <dir>] [--json] [--dry-run | --apply]',
|
|
4705
|
+
' [--window-days <N>] [--periods <N>] [--limit <N>]',
|
|
4706
|
+
].join('\n ');
|
|
4707
|
+
/**
|
|
4708
|
+
* Read real commit history as {@link ChangeSet}s — the shadow-replay corpus. `--name-only` gives the
|
|
4709
|
+
* change shape every v1 template consumes. Merges are excluded (their file list is a union of the
|
|
4710
|
+
* branches, not a decision anyone made).
|
|
4711
|
+
*
|
|
4712
|
+
* A missing/failing `git` yields `[]`, which makes every candidate `insufficient-data` — the command
|
|
4713
|
+
* still exits 0 and says why. No history is not a promotion.
|
|
4714
|
+
*/
|
|
4715
|
+
function readGitChanges(root, sinceIso) {
|
|
4716
|
+
let out = '';
|
|
4717
|
+
try {
|
|
4718
|
+
// execFileSync (argv form), NOT a shell string: `--name-only` paths come straight from the repo,
|
|
4719
|
+
// and a filename containing `$(…)` or a backtick would EXPAND inside a double-quoted shell
|
|
4720
|
+
// argument. `-z` is not used because the pretty header needs line framing; the argv form removes
|
|
4721
|
+
// the shell entirely instead.
|
|
4722
|
+
out = execFileSync('git', ['log', `--since=${sinceIso}`, '--no-merges', '--name-only', '--pretty=format:%x01%H%x09%cI'], {
|
|
4723
|
+
cwd: root,
|
|
4724
|
+
encoding: 'utf-8',
|
|
4725
|
+
maxBuffer: 64 * 1024 * 1024,
|
|
4726
|
+
});
|
|
4727
|
+
}
|
|
4728
|
+
catch {
|
|
4729
|
+
return [];
|
|
4730
|
+
}
|
|
4731
|
+
const changes = [];
|
|
4732
|
+
let current = null;
|
|
4733
|
+
for (const raw of out.split('\n')) {
|
|
4734
|
+
if (raw.startsWith('\x01')) {
|
|
4735
|
+
if (current)
|
|
4736
|
+
changes.push(current);
|
|
4737
|
+
const [id, ts] = raw.slice(1).split('\t');
|
|
4738
|
+
current = { id: (id ?? '').slice(0, 12), ts: ts ?? '', files: [] };
|
|
4739
|
+
continue;
|
|
4740
|
+
}
|
|
4741
|
+
const line = raw.trim();
|
|
4742
|
+
if (line === '' || current === null)
|
|
4743
|
+
continue;
|
|
4744
|
+
current.files.push(line);
|
|
4745
|
+
}
|
|
4746
|
+
if (current)
|
|
4747
|
+
changes.push(current);
|
|
4748
|
+
return changes;
|
|
4749
|
+
}
|
|
4750
|
+
/**
|
|
4751
|
+
* Attach file text at each historical commit for the `format-match` candidates that need it.
|
|
4752
|
+
*
|
|
4753
|
+
* HARD CAP (`MAX_CONTENT_FETCHES`): over it we STOP fetching, which leaves those changes without
|
|
4754
|
+
* contents, which makes `templateFires` return `undecidable`, which makes the candidate
|
|
4755
|
+
* `insufficient-data`. What we deliberately do NOT do is fall back to the file's CURRENT content —
|
|
4756
|
+
* evaluating a historical commit against today's file is exactly the fabricated-win shape ADR-002
|
|
4757
|
+
* refused for `presence-check`.
|
|
4758
|
+
*/
|
|
4759
|
+
function attachChangeContents(root, changes, globs) {
|
|
4760
|
+
if (globs.length === 0)
|
|
4761
|
+
return [...changes];
|
|
4762
|
+
let budget = MAX_CONTENT_FETCHES;
|
|
4763
|
+
return changes.map((c) => {
|
|
4764
|
+
const wanted = c.files.filter((f) => globs.some((g) => globMatch(g, f)));
|
|
4765
|
+
if (wanted.length === 0)
|
|
4766
|
+
return c;
|
|
4767
|
+
const contents = {};
|
|
4768
|
+
for (const f of wanted) {
|
|
4769
|
+
if (budget <= 0)
|
|
4770
|
+
return c; // over cap ⇒ leave this change undecidable, never guess
|
|
4771
|
+
budget -= 1;
|
|
4772
|
+
try {
|
|
4773
|
+
// argv form, NOT a shell string: a repo path is untrusted input and `$(…)`/backticks would
|
|
4774
|
+
// expand inside a quoted shell argument. No `--` terminator — `git show -- <rev>:<path>`
|
|
4775
|
+
// exits 0 with EMPTY output (the terminator turns the rev-with-path into a pathspec). The
|
|
4776
|
+
// option-smuggling risk it would have covered is absent anyway: the argument always begins
|
|
4777
|
+
// with a 12-hex sha.
|
|
4778
|
+
const text = execFileSync('git', ['show', `${c.id}:${f}`], { cwd: root, encoding: 'utf-8', maxBuffer: MAX_CONTENT_BYTES });
|
|
4779
|
+
// Same size ceiling as the live reader, so replay and live agree on what is too big to judge.
|
|
4780
|
+
if (text.length > MAX_CONTENT_BYTES)
|
|
4781
|
+
return c;
|
|
4782
|
+
// EMPTY OUTPUT IS NOT CONTENT. git reported this path as changed in this commit, so an
|
|
4783
|
+
// empty body is far more likely a failed lookup than a genuinely empty file — and treating
|
|
4784
|
+
// it as content is the worst possible failure: `''.includes(x)` is false, so the rule would
|
|
4785
|
+
// FIRE on every single file and fabricate a clean sweep of wins. Undecidable instead.
|
|
4786
|
+
if (text === '')
|
|
4787
|
+
return c;
|
|
4788
|
+
contents[f] = text;
|
|
4789
|
+
}
|
|
4790
|
+
catch {
|
|
4791
|
+
return c; // deleted/renamed at that commit — undecidable, not clean
|
|
4792
|
+
}
|
|
4793
|
+
}
|
|
4794
|
+
return { ...c, contents };
|
|
4795
|
+
});
|
|
4796
|
+
}
|
|
4797
|
+
/** Every rule the engine would run: the built-ins plus any template rules already in `.dz/guard.json`. */
|
|
4798
|
+
function existingRuleViews(root) {
|
|
4799
|
+
const cfg = loadGuardConfig(root);
|
|
4800
|
+
const configRules = Array.isArray(cfg.rules) ? cfg.rules : [];
|
|
4801
|
+
const disabled = new Set(configRules.filter((r) => r?.enabled === false && typeof r.id === 'string').map((r) => r.id));
|
|
4802
|
+
// A built-in the operator has DISABLED does not cover anything — otherwise the promoter would
|
|
4803
|
+
// refuse a candidate as a duplicate of a rule that is not running, and the gap would stay open.
|
|
4804
|
+
const views = DEFAULT_RULES.filter((r) => !disabled.has(r.id)).map((r) => ({ id: r.id }));
|
|
4805
|
+
for (const o of configRules) {
|
|
4806
|
+
if (typeof o?.id !== 'string' || o.enabled === false)
|
|
4807
|
+
continue;
|
|
4808
|
+
// A rule op-scoped AWAY from publish covers nothing a change-shaped promotion targets — letting
|
|
4809
|
+
// it suppress a candidate as a "duplicate" keeps the gap open (Codex re-QE MED, mirror of the
|
|
4810
|
+
// disabled-builtin rationale above).
|
|
4811
|
+
const ops = o.ops;
|
|
4812
|
+
if (Array.isArray(ops) && !ops.includes('publish'))
|
|
4813
|
+
continue;
|
|
4814
|
+
if (views.some((v) => v.id === o.id))
|
|
4815
|
+
continue;
|
|
4816
|
+
views.push({ id: o.id, ...(typeof o.template === 'string' ? { template: o.template } : {}), ...(o.params && typeof o.params === 'object' ? { params: o.params } : {}) });
|
|
4817
|
+
}
|
|
4818
|
+
return views;
|
|
4819
|
+
}
|
|
4820
|
+
/** Atomic JSON write — tmp + rename, so a crash mid-write never leaves a half-parsed state file. */
|
|
4821
|
+
function writeJsonAtomic(path, value) {
|
|
4822
|
+
mkdirSync(dirname(path), { recursive: true });
|
|
4823
|
+
const tmp = `${path}.tmp.${process.pid}`;
|
|
4824
|
+
writeFileSync(tmp, `${JSON.stringify(value, null, 2)}\n`, 'utf-8');
|
|
4825
|
+
renameSync(tmp, path);
|
|
4826
|
+
}
|
|
4827
|
+
/** The rolled-up refusal record (ADR-004): one file, regenerated in place, one row per lesson. */
|
|
4828
|
+
function renderNotPromotableRollup(report, nowTs) {
|
|
4829
|
+
const refused = report.candidates.filter((c) => c.verdict === 'not-promotable');
|
|
4830
|
+
const out = [];
|
|
4831
|
+
out.push('# 000 — Not promotable (rolled-up refusal record)');
|
|
4832
|
+
out.push('');
|
|
4833
|
+
out.push(`**Decision:** REFUSED · **Date:** ${nowTs} · **Count:** ${refused.length} of ${report.totalLessons} lesson(s)`);
|
|
4834
|
+
out.push('');
|
|
4835
|
+
out.push('These lessons do not reduce to a v1 `dz guard promote` rule template. Rule code is NEVER');
|
|
4836
|
+
out.push('synthesised from lesson text (ADR-002), so a lesson that fits no template is refused aloud');
|
|
4837
|
+
out.push('rather than force-fitted. This file is regenerated in place on every run.');
|
|
4838
|
+
out.push('');
|
|
4839
|
+
out.push('| lesson | reason | first 90 chars |');
|
|
4840
|
+
out.push('|---|---|---|');
|
|
4841
|
+
for (const c of refused) {
|
|
4842
|
+
const t = c.lessonText.replace(/\|/g, '\\|').replace(/\n/g, ' ').slice(0, 90);
|
|
4843
|
+
out.push(`| \`${c.lessonId}\` | ${c.reason.replace(/\|/g, '\\|')} | ${t} |`);
|
|
4844
|
+
}
|
|
4845
|
+
return out.join('\n') + '\n';
|
|
4846
|
+
}
|
|
4847
|
+
/**
|
|
4848
|
+
* `dz guard promote` — lesson → guard-rule promotion with a "win twice to promote" gate.
|
|
4849
|
+
*
|
|
4850
|
+
* Thin by design: gather (lessons, existing rules, real commit history, state) → the PURE
|
|
4851
|
+
* `assembleCandidates` → render → write. Default PROPOSES (documents + journal only); `--dry-run`
|
|
4852
|
+
* writes nothing at all; `--apply` is the only path that touches `.dz/guard.json`, always SOFT.
|
|
4853
|
+
*/
|
|
4854
|
+
function cmdGuardPromote(options, flags, root, write) {
|
|
4855
|
+
const json = flags.has('json');
|
|
4856
|
+
const fail = (msg) => {
|
|
4857
|
+
write(json ? JSON.stringify({ error: msg, exitCode: 1 }) : `dz guard promote: ${msg}\n usage: ${GUARD_PROMOTE_USAGE}`);
|
|
4858
|
+
return 1;
|
|
4859
|
+
};
|
|
4860
|
+
for (const f of flags)
|
|
4861
|
+
if (!['json', 'dry-run', 'apply', 'help'].includes(f))
|
|
4862
|
+
return fail(`unknown option --${f}`);
|
|
4863
|
+
for (const k of options.keys()) {
|
|
4864
|
+
if (k === '_positional_0')
|
|
4865
|
+
continue; // the `promote` subcommand token itself
|
|
4866
|
+
if (k.startsWith('_positional_'))
|
|
4867
|
+
return fail(`unexpected argument "${options.get(k)}"`);
|
|
4868
|
+
if (!['project', 'window-days', 'periods', 'limit'].includes(k))
|
|
4869
|
+
return fail(`unknown option --${k}`);
|
|
4870
|
+
}
|
|
4871
|
+
if (flags.has('help')) {
|
|
4872
|
+
write(`dz guard promote — promote a learned lesson to a deterministic guard rule\n usage: ${GUARD_PROMOTE_USAGE}`);
|
|
4873
|
+
write(' A candidate must SHADOW-WIN twice consecutively over real commit history before it is proposed.');
|
|
4874
|
+
write(' Default: writes proposal/refusal documents only. --dry-run: writes nothing. --apply: writes the SOFT rule into .dz/guard.json.');
|
|
4875
|
+
return 0;
|
|
4876
|
+
}
|
|
4877
|
+
const dryRun = flags.has('dry-run');
|
|
4878
|
+
const apply = flags.has('apply');
|
|
4879
|
+
// NOT a silent precedence: two contradictory intents is an error, not a coin flip.
|
|
4880
|
+
if (dryRun && apply)
|
|
4881
|
+
return fail('--dry-run and --apply are mutually exclusive');
|
|
4882
|
+
const num = (key, dflt, lo, hi) => {
|
|
4883
|
+
const raw = options.get(key);
|
|
4884
|
+
if (raw === undefined)
|
|
4885
|
+
return dflt;
|
|
4886
|
+
const n = Number(raw);
|
|
4887
|
+
if (!Number.isFinite(n) || !Number.isInteger(n) || n < lo || n > hi)
|
|
4888
|
+
return null;
|
|
4889
|
+
return n;
|
|
4890
|
+
};
|
|
4891
|
+
const windowDays = num('window-days', DEFAULT_WINDOW_DAYS, 1, 365);
|
|
4892
|
+
if (windowDays === null)
|
|
4893
|
+
return fail('--window-days expects an integer in [1, 365]');
|
|
4894
|
+
const periods = num('periods', DEFAULT_PERIODS, 2, 52);
|
|
4895
|
+
if (periods === null)
|
|
4896
|
+
return fail('--periods expects an integer in [2, 52]');
|
|
4897
|
+
const limit = num('limit', 15, 1, 1000);
|
|
4898
|
+
if (limit === null)
|
|
4899
|
+
return fail('--limit expects an integer in [1, 1000]');
|
|
4900
|
+
const nowTs = new Date().toISOString();
|
|
4901
|
+
const sinceIso = new Date(Date.now() - windowDays * periods * 86_400_000).toISOString();
|
|
4902
|
+
// lessons — the SAME readers `dz compounding` uses; no second store.
|
|
4903
|
+
const lessons = loadStoreRecords(root).map((r) => ({
|
|
4904
|
+
dzId: r.id,
|
|
4905
|
+
text: typeof r.text === 'string' ? r.text : '',
|
|
4906
|
+
quarantined: readQuarantineState(r).quarantined,
|
|
4907
|
+
uses: readReinforcementState(r).uses,
|
|
4908
|
+
}));
|
|
4909
|
+
const existingRules = existingRuleViews(root);
|
|
4910
|
+
const changes = readGitChanges(root, sinceIso);
|
|
4911
|
+
// State is read on EVERY run, including --dry-run, because it carries the LOCAL-clock `firstSeen`
|
|
4912
|
+
// the elapsed gate reads (MED-7). --dry-run still writes nothing — which is exactly why a
|
|
4913
|
+
// dry-run-only workflow never starts that clock, and the wait reason says so.
|
|
4914
|
+
const state = normalizePromotionState((() => {
|
|
4915
|
+
try {
|
|
4916
|
+
return JSON.parse(readFileSync(join(root, PROMOTION_STATE_FILE), 'utf-8'));
|
|
4917
|
+
}
|
|
4918
|
+
catch {
|
|
4919
|
+
return null;
|
|
4920
|
+
}
|
|
4921
|
+
})());
|
|
4922
|
+
const firstSeen = {};
|
|
4923
|
+
for (const [id, e] of Object.entries(state.entries))
|
|
4924
|
+
if (e.firstSeenTs !== '')
|
|
4925
|
+
firstSeen[id] = e.firstSeenTs;
|
|
4926
|
+
// Pass 1 discovers which `format-match` candidates exist, so contents are fetched only for the
|
|
4927
|
+
// globs that actually need them (and only up to the cap).
|
|
4928
|
+
const pass1 = assembleCandidates({ lessons, existingRules, changes, nowTs, windowDays, periods, firstSeen });
|
|
4929
|
+
const formatGlobs = [...new Set(pass1.candidates.filter((c) => c.template === 'format-match' && typeof c.params?.file === 'string').map((c) => c.params.file))];
|
|
4930
|
+
const report = formatGlobs.length === 0
|
|
4931
|
+
? pass1
|
|
4932
|
+
: assembleCandidates({ lessons, existingRules, changes: attachChangeContents(root, changes, formatGlobs), nowTs, windowDays, periods, firstSeen });
|
|
4933
|
+
// ── write side ────────────────────────────────────────────────────────────────────────────────
|
|
4934
|
+
const written = [];
|
|
4935
|
+
const applied = [];
|
|
4936
|
+
const conflicts = [];
|
|
4937
|
+
if (!dryRun) {
|
|
4938
|
+
const adrSeqs = {};
|
|
4939
|
+
let seq = state.nextAdrSeq;
|
|
4940
|
+
let allocated = 0;
|
|
4941
|
+
for (const c of report.candidates) {
|
|
4942
|
+
if (c.verdict !== 'promote' && c.verdict !== 'duplicate')
|
|
4943
|
+
continue;
|
|
4944
|
+
const key = c.ruleId;
|
|
4945
|
+
// ONE document per CANDIDATE, not per run — but the reused thing is the integer SEQUENCE, never
|
|
4946
|
+
// a path. State is attacker-shaped input (a JSON file anyone can corrupt), and a path taken
|
|
4947
|
+
// from it and handed to writeFileSync overwrites whatever it names — with no `--apply`, no
|
|
4948
|
+
// promotion, and no way to notice. The path is DERIVED from the validated id + that integer.
|
|
4949
|
+
const existingSeq = state.entries[key]?.adrSeq;
|
|
4950
|
+
const useSeq = existingSeq ?? seq;
|
|
4951
|
+
const rel = promotionAdrRelPath(key, useSeq);
|
|
4952
|
+
if (rel === null)
|
|
4953
|
+
continue; // an id or sequence that fails validation writes NOTHING
|
|
4954
|
+
if (existingSeq === undefined) {
|
|
4955
|
+
seq += 1;
|
|
4956
|
+
allocated += 1;
|
|
4957
|
+
}
|
|
4958
|
+
// Belt to the derivation's braces: resolve and assert containment before writing. A derivation
|
|
4959
|
+
// that is correct today is not a substitute for checking the thing you are about to write.
|
|
4960
|
+
const abs = resolve(root, rel);
|
|
4961
|
+
const dir = resolve(root, PROMOTIONS_DIR);
|
|
4962
|
+
if (abs !== dir && !abs.startsWith(dir + sep))
|
|
4963
|
+
continue;
|
|
4964
|
+
try {
|
|
4965
|
+
mkdirSync(dirname(abs), { recursive: true });
|
|
4966
|
+
// Lexical containment is not PHYSICAL containment (Codex re-QE HIGH): a symlinked
|
|
4967
|
+
// promotions/ directory (or a symlink planted at the ADR leaf) redirects the write outside
|
|
4968
|
+
// the repo while every string check passes. realpath the directory that actually exists on
|
|
4969
|
+
// disk and require it to be the real promotions dir under the real root; refuse a leaf that
|
|
4970
|
+
// is a symlink.
|
|
4971
|
+
const realDir = realpathSync(dirname(abs));
|
|
4972
|
+
const expectedReal = join(realpathSync(root), PROMOTIONS_DIR.split('/').join(sep));
|
|
4973
|
+
if (realDir !== expectedReal)
|
|
4974
|
+
continue;
|
|
4975
|
+
if (existsSync(abs) && lstatSync(abs).isSymbolicLink())
|
|
4976
|
+
continue;
|
|
4977
|
+
writeFileSync(abs, renderPromotionAdr(c, useSeq, nowTs), 'utf-8');
|
|
4978
|
+
adrSeqs[key] = useSeq;
|
|
4979
|
+
written.push(rel);
|
|
4980
|
+
}
|
|
4981
|
+
catch { /* a document we cannot write must not lose the verdict */ }
|
|
4982
|
+
}
|
|
4983
|
+
if (report.candidates.some((c) => c.verdict === 'not-promotable')) {
|
|
4984
|
+
const rel = join(PROMOTIONS_DIR, '000-not-promotable.md');
|
|
4985
|
+
try {
|
|
4986
|
+
mkdirSync(join(root, PROMOTIONS_DIR), { recursive: true });
|
|
4987
|
+
writeFileSync(join(root, rel), renderNotPromotableRollup(report, nowTs), 'utf-8');
|
|
4988
|
+
written.push(rel);
|
|
4989
|
+
}
|
|
4990
|
+
catch { /* best-effort */ }
|
|
4991
|
+
}
|
|
4992
|
+
if (apply) {
|
|
4993
|
+
const cfg = loadGuardConfig(root);
|
|
4994
|
+
const rules = Array.isArray(cfg.rules) ? [...cfg.rules] : [];
|
|
4995
|
+
for (const c of report.candidates) {
|
|
4996
|
+
if (c.verdict !== 'promote' || c.proposedRule === null)
|
|
4997
|
+
continue;
|
|
4998
|
+
const want = c.proposedRule;
|
|
4999
|
+
const clash = rules.find((r) => r?.id === want.id);
|
|
5000
|
+
if (clash !== undefined) {
|
|
5001
|
+
// ID EQUALITY IS NOT IDEMPOTENCE (Codex QE MED-5). A rule that merely SHARES the id — a
|
|
5002
|
+
// hand-written bare entry, or a same-id rule with different params — is not the rule we
|
|
5003
|
+
// are promoting. Skipping it silently reports "applied" while installing nothing (the bare
|
|
5004
|
+
// entry does not even enforce, since resolveRules drops an unknown id with no template).
|
|
5005
|
+
// Same id + same BODY is genuine idempotence; same id + different body is a conflict, and a
|
|
5006
|
+
// conflict is refused out loud rather than resolved by guessing which side to keep.
|
|
5007
|
+
const sameBody = clash.template === want.template && JSON.stringify(clash.params ?? null) === JSON.stringify(want.params);
|
|
5008
|
+
// Same body but DISABLED (or op-scoped away from publish) is NOT idempotence: the rule
|
|
5009
|
+
// exists on paper and enforces nothing — "already installed" would be a false success
|
|
5010
|
+
// (Codex re-QE MED). Refuse loudly so the operator re-enables or removes it.
|
|
5011
|
+
const clashEnabled = clash.enabled !== false;
|
|
5012
|
+
const clashOps = clash.ops;
|
|
5013
|
+
const clashCoversPublish = !Array.isArray(clashOps) || clashOps.includes('publish');
|
|
5014
|
+
if (sameBody && clashEnabled && clashCoversPublish)
|
|
5015
|
+
continue; // already installed AND active — genuine idempotence
|
|
5016
|
+
if (sameBody) {
|
|
5017
|
+
conflicts.push(`${want.id}: an identical rule exists in .dz/guard.json but is ${clashEnabled ? 'op-scoped away from publish' : 'DISABLED'} — it enforces nothing; re-enable it (or remove it and re-run --apply) instead of trusting a rule that is not running`);
|
|
5018
|
+
continue;
|
|
5019
|
+
}
|
|
5020
|
+
conflicts.push(`${want.id}: an existing .dz/guard.json rule shares this id but has a different body (existing template=${JSON.stringify(clash.template ?? null)} params=${JSON.stringify(clash.params ?? null)}; promoted template=${JSON.stringify(want.template)} params=${JSON.stringify(want.params)}) — refusing to overwrite or to claim success; rename or remove the existing rule`);
|
|
5021
|
+
continue;
|
|
5022
|
+
}
|
|
5023
|
+
rules.push(want);
|
|
5024
|
+
applied.push(want.id);
|
|
5025
|
+
}
|
|
5026
|
+
if (applied.length > 0)
|
|
5027
|
+
writeJsonAtomic(join(root, '.dz', 'guard.json'), { ...cfg, rules });
|
|
5028
|
+
}
|
|
5029
|
+
const next = nextPromotionState(state, report, nowTs, adrSeqs, allocated);
|
|
5030
|
+
const withApplied = applied.length === 0 ? next : {
|
|
5031
|
+
...next,
|
|
5032
|
+
entries: Object.fromEntries(Object.entries(next.entries).map(([k, v]) => (applied.includes(k) ? [k, { ...v, appliedTs: nowTs }] : [k, v]))),
|
|
5033
|
+
};
|
|
5034
|
+
try {
|
|
5035
|
+
writeJsonAtomic(join(root, PROMOTION_STATE_FILE), withApplied);
|
|
5036
|
+
}
|
|
5037
|
+
catch { /* best-effort */ }
|
|
5038
|
+
}
|
|
5039
|
+
// A refused conflict means the requested apply did NOT fully happen — exit non-zero rather than
|
|
5040
|
+
// let a zero exit report success for work that was deliberately not done.
|
|
5041
|
+
const exitCode = conflicts.length > 0 ? 1 : 0;
|
|
5042
|
+
if (json) {
|
|
5043
|
+
write(JSON.stringify({ ...report, mode: dryRun ? 'dry-run' : apply ? 'apply' : 'propose', written, applied, conflicts, exitCode }, null, 2));
|
|
5044
|
+
return exitCode;
|
|
5045
|
+
}
|
|
5046
|
+
write(renderPromotionReport(report, limit));
|
|
5047
|
+
write('');
|
|
5048
|
+
if (dryRun)
|
|
5049
|
+
write(' MODE: --dry-run — nothing was written (not even .dz/promotion-state.json)');
|
|
5050
|
+
else {
|
|
5051
|
+
write(` WROTE: ${written.length === 0 ? '(no decisions to record)' : written.join(', ')}`);
|
|
5052
|
+
if (apply)
|
|
5053
|
+
write(` APPLIED to .dz/guard.json: ${applied.length === 0 ? '(none)' : applied.join(', ')} — SOFT severity, always`);
|
|
5054
|
+
else
|
|
5055
|
+
write(' Nothing was written to .dz/guard.json — re-run with --apply to install the promoted rule(s).');
|
|
5056
|
+
}
|
|
5057
|
+
for (const c of conflicts)
|
|
5058
|
+
write(` ✗ CONFLICT — ${c}`);
|
|
5059
|
+
return exitCode;
|
|
5060
|
+
}
|
|
4645
5061
|
/**
|
|
4646
5062
|
* `dz guard` — the declarative constraint layer that refuses a self-mutating op when a HARD invariant is
|
|
4647
5063
|
* violated. Simple outside: `dz guard check --op publish` works with zero config (built-in defaults).
|
|
@@ -4697,8 +5113,13 @@ function cmdGuard(options, flags, cwd, write) {
|
|
|
4697
5113
|
}
|
|
4698
5114
|
return 0;
|
|
4699
5115
|
}
|
|
5116
|
+
// `promote` — lesson → guard-rule promotion. It lives HERE, not as a top-level `dz promote`,
|
|
5117
|
+
// because its object IS a guard rule: its evidence is .dz/guard-audit.jsonl + real commit history
|
|
5118
|
+
// and its write target is .dz/guard.json (ADR-001).
|
|
5119
|
+
if (sub === 'promote')
|
|
5120
|
+
return cmdGuardPromote(options, flags, options.get('project') !== undefined ? resolve(cwd, options.get('project')) : root, write);
|
|
4700
5121
|
if (sub !== 'check') {
|
|
4701
|
-
write(`dz guard: unknown subcommand '${sub}' — use: check --op <op> | --init | log`);
|
|
5122
|
+
write(`dz guard: unknown subcommand '${sub}' — use: check --op <op> | promote | --init | log`);
|
|
4702
5123
|
return 1;
|
|
4703
5124
|
}
|
|
4704
5125
|
// check
|
|
@@ -5495,6 +5916,46 @@ function cmdScore(options, flags, cwd, write) {
|
|
|
5495
5916
|
write(renderScorecard(card));
|
|
5496
5917
|
return 0;
|
|
5497
5918
|
}
|
|
5919
|
+
/**
|
|
5920
|
+
* Read the apply-leg usage log into the shape `compounding.ts` / `epoch-replay.ts` expect.
|
|
5921
|
+
*
|
|
5922
|
+
* ONE reader, because there were two and they disagreed: the `dz compounding` fact-gatherer used
|
|
5923
|
+
* to drop `eventId` and `queryTruncated`, so the readiness gate counted 48 "replayable pairs" over
|
|
5924
|
+
* a log whose honest count is 25 (MEASURED on this repo, 2026-07-29). Both defences documented in
|
|
5925
|
+
* `assembleCompoundingReport` — "one prompt = one pair" (Codex #1) and "a truncated query is a
|
|
5926
|
+
* prefix, not the prompt" (Codex #3) — were correct in the pure function and DEAD at its only
|
|
5927
|
+
* caller, because the caller never passed the fields they read.
|
|
5928
|
+
*/
|
|
5929
|
+
function readRecallUsageEvents(root) {
|
|
5930
|
+
const usage = [];
|
|
5931
|
+
try {
|
|
5932
|
+
const text = readFileSync(join(root, '.dz', 'recall-usage.jsonl'), 'utf-8');
|
|
5933
|
+
for (const line of text.split('\n')) {
|
|
5934
|
+
if (line.trim() === '')
|
|
5935
|
+
continue;
|
|
5936
|
+
try {
|
|
5937
|
+
const o = JSON.parse(line);
|
|
5938
|
+
if (typeof o.dzId === 'string' && typeof o.ts === 'string' && o.kind !== 'aggregate') {
|
|
5939
|
+
usage.push({
|
|
5940
|
+
dzId: o.dzId,
|
|
5941
|
+
ts: o.ts,
|
|
5942
|
+
...(typeof o.query === 'string' ? { query: o.query } : {}),
|
|
5943
|
+
...(typeof o.runId === 'string' ? { runId: o.runId } : {}),
|
|
5944
|
+
...(typeof o.eventId === 'string' ? { eventId: o.eventId } : {}),
|
|
5945
|
+
...(o.queryTruncated === true ? { queryTruncated: true } : {}),
|
|
5946
|
+
});
|
|
5947
|
+
}
|
|
5948
|
+
}
|
|
5949
|
+
catch {
|
|
5950
|
+
/* one bad line must not kill the read */
|
|
5951
|
+
}
|
|
5952
|
+
}
|
|
5953
|
+
}
|
|
5954
|
+
catch {
|
|
5955
|
+
/* no log yet — callers report the absence */
|
|
5956
|
+
}
|
|
5957
|
+
return usage;
|
|
5958
|
+
}
|
|
5498
5959
|
/**
|
|
5499
5960
|
* `dz compounding` — does the learning loop actually PAY? (feature compounding, scout C2.)
|
|
5500
5961
|
* Gathers the facts (store rows, apply-leg usage log, guard audit) and hands them to the PURE
|
|
@@ -5532,31 +5993,7 @@ function cmdCompounding(options, flags, cwd, write) {
|
|
|
5532
5993
|
reward: typeof r.score === 'number' ? r.score : null,
|
|
5533
5994
|
}));
|
|
5534
5995
|
// apply-leg usage events (read records only; aggregate rows carry no query by construction)
|
|
5535
|
-
const usage =
|
|
5536
|
-
try {
|
|
5537
|
-
const text = readFileSync(join(root, '.dz', 'recall-usage.jsonl'), 'utf-8');
|
|
5538
|
-
for (const line of text.split('\n')) {
|
|
5539
|
-
if (line.trim() === '')
|
|
5540
|
-
continue;
|
|
5541
|
-
try {
|
|
5542
|
-
const o = JSON.parse(line);
|
|
5543
|
-
if (typeof o.dzId === 'string' && typeof o.ts === 'string' && o.kind !== 'aggregate') {
|
|
5544
|
-
usage.push({
|
|
5545
|
-
dzId: o.dzId,
|
|
5546
|
-
ts: o.ts,
|
|
5547
|
-
...(typeof o.query === 'string' ? { query: o.query } : {}),
|
|
5548
|
-
...(typeof o.runId === 'string' ? { runId: o.runId } : {}),
|
|
5549
|
-
});
|
|
5550
|
-
}
|
|
5551
|
-
}
|
|
5552
|
-
catch {
|
|
5553
|
-
/* one bad line must not kill the report */
|
|
5554
|
-
}
|
|
5555
|
-
}
|
|
5556
|
-
}
|
|
5557
|
-
catch {
|
|
5558
|
-
/* no log yet — the report says so */
|
|
5559
|
-
}
|
|
5996
|
+
const usage = readRecallUsageEvents(root);
|
|
5560
5997
|
// guard audit events
|
|
5561
5998
|
const guard = [];
|
|
5562
5999
|
try {
|
|
@@ -5589,6 +6026,307 @@ function cmdCompounding(options, flags, cwd, write) {
|
|
|
5589
6026
|
write(renderCompoundingReport(report));
|
|
5590
6027
|
return 0;
|
|
5591
6028
|
}
|
|
6029
|
+
// ── `dz epoch-replay` (feature epoch-replay) ────────────────────────────────────────────────────
|
|
6030
|
+
const EPOCH_REPLAY_DIR = join('.dz', 'epoch-replay');
|
|
6031
|
+
const EPOCH_REPLAY_USAGE = [
|
|
6032
|
+
'dz epoch-replay --mock [--n <N>] [--effect <-1..1>] [--tie-rate <0..1>] [--seed <N>] [--slice <name>] [--margin <0..1>] [--json]',
|
|
6033
|
+
'dz epoch-replay --emit [--project <dir>] [--limit <N>] [--seed <N>] [--margin <0..0.5>] [--out <file>] [--json]',
|
|
6034
|
+
'dz epoch-replay --judge <filled-work-order.json> [--out <file>] [--json]',
|
|
6035
|
+
'dz epoch-replay --score <judgments.json> --work-order <file> [--slice <name>] [--json] (margin comes from the work order)',
|
|
6036
|
+
].join('\n ');
|
|
6037
|
+
/** Read a JSON file into an object, or return a parse/IO error string. */
|
|
6038
|
+
function readJsonFile(path) {
|
|
6039
|
+
try {
|
|
6040
|
+
return { value: JSON.parse(readFileSync(path, 'utf-8')) };
|
|
6041
|
+
}
|
|
6042
|
+
catch (e) {
|
|
6043
|
+
return { error: `cannot read ${path}: ${e instanceof Error ? e.message : String(e)}` };
|
|
6044
|
+
}
|
|
6045
|
+
}
|
|
6046
|
+
/**
|
|
6047
|
+
* Integrity-check a parsed work order. Checking `kind` + `Array.isArray(items)` was VACUOUS: a
|
|
6048
|
+
* hand-written file with those two fields and a fabricated `warmIsA` bought whatever verdict its
|
|
6049
|
+
* author wanted (Codex QE HIGH-2). `verifyWorkOrder` recomputes the digest AND re-derives every
|
|
6050
|
+
* assignment from the stated seed.
|
|
6051
|
+
*/
|
|
6052
|
+
function asVerifiedWorkOrder(value) {
|
|
6053
|
+
const v = verifyWorkOrder(value);
|
|
6054
|
+
if (!v.ok)
|
|
6055
|
+
return { problems: v.problems };
|
|
6056
|
+
return { order: value };
|
|
6057
|
+
}
|
|
6058
|
+
function writeJsonOut(path, value) {
|
|
6059
|
+
mkdirSync(dirname(path), { recursive: true });
|
|
6060
|
+
writeFileSync(path, `${JSON.stringify(value, null, 2)}\n`, 'utf-8');
|
|
6061
|
+
}
|
|
6062
|
+
/** Numeric option parsing that never silently accepts garbage. */
|
|
6063
|
+
function numOpt(options, key) {
|
|
6064
|
+
const raw = options.get(key);
|
|
6065
|
+
if (raw === undefined)
|
|
6066
|
+
return null;
|
|
6067
|
+
const n = Number(raw);
|
|
6068
|
+
if (!Number.isFinite(n))
|
|
6069
|
+
return { error: `--${key} expects a finite number, got ${JSON.stringify(raw)}` };
|
|
6070
|
+
return { value: n };
|
|
6071
|
+
}
|
|
6072
|
+
/**
|
|
6073
|
+
* `dz epoch-replay` — the executable cold-vs-warm epoch runner (feature epoch-replay, scout #4).
|
|
6074
|
+
*
|
|
6075
|
+
* `dz compounding` says whether a replay CAN be run; this says what it FOUND. The runner never
|
|
6076
|
+
* calls a model: real mode emits a work order, renders blind judge prompts, and scores filled
|
|
6077
|
+
* judgments; `--mock` exercises the same verdict math on seeded synthetic outcomes at $0.
|
|
6078
|
+
*/
|
|
6079
|
+
function cmdEpochReplay(options, flags, cwd, write) {
|
|
6080
|
+
const json = flags.has('json');
|
|
6081
|
+
const fail = (msg) => {
|
|
6082
|
+
write(json ? JSON.stringify({ error: msg, exitCode: 1 }) : `dz epoch-replay: ${msg}\n usage:\n ${EPOCH_REPLAY_USAGE}`);
|
|
6083
|
+
return 1;
|
|
6084
|
+
};
|
|
6085
|
+
if (flags.has('help')) {
|
|
6086
|
+
write(`dz epoch-replay — cold (epoch 0) vs warm (epoch 1), Wilson-CI three-valued verdict\n ${EPOCH_REPLAY_USAGE}`);
|
|
6087
|
+
write('');
|
|
6088
|
+
write(' ONE binomial over DECISIVE pairs (ties carry no direction and are excluded from the test).');
|
|
6089
|
+
write(' SUPPORTED only when the paired lift interval lies ENTIRELY above zero.');
|
|
6090
|
+
write(' FALSIFIED only on HARM (entirely below zero), or on a passed NON-SUPERIORITY test (the');
|
|
6091
|
+
write(' lift upper bound below the PRE-REGISTERED margin, default 0.05). Otherwise INCONCLUSIVE —');
|
|
6092
|
+
write(' a first-class honest outcome; a tie is UNDER-POWERED, never "refuted".');
|
|
6093
|
+
write(' The margin is pre-registered at --emit and stored in the work order; --score reads it there.');
|
|
6094
|
+
write(' This runner ORCHESTRATES and SCORES; it never calls a model. The judge-facing file holds');
|
|
6095
|
+
write(' {id, prompt} and nothing else; --score refuses any work order whose digest or seed-derived');
|
|
6096
|
+
write(' assignment does not check out, and refuses duplicate judgement ids.');
|
|
6097
|
+
return 0;
|
|
6098
|
+
}
|
|
6099
|
+
const allowedFlags = new Set(['mock', 'emit', 'json', 'help']);
|
|
6100
|
+
for (const flag of flags) {
|
|
6101
|
+
if (!allowedFlags.has(flag))
|
|
6102
|
+
return fail(`unknown option --${flag}`);
|
|
6103
|
+
}
|
|
6104
|
+
const allowedOptions = new Set(['judge', 'score', 'work-order', 'project', 'out', 'n', 'effect', 'tie-rate', 'seed', 'slice', 'limit', 'margin']);
|
|
6105
|
+
for (const key of options.keys()) {
|
|
6106
|
+
if (key.startsWith('_positional_'))
|
|
6107
|
+
return fail(`unexpected argument "${options.get(key)}"`);
|
|
6108
|
+
if (!allowedOptions.has(key))
|
|
6109
|
+
return fail(`unknown option --${key}`);
|
|
6110
|
+
}
|
|
6111
|
+
// Exactly ONE mode. A command whose default mode is a report can silently swallow a typo'd mode
|
|
6112
|
+
// flag and print something that reads like a result — so there is NO default mode here.
|
|
6113
|
+
const modes = [
|
|
6114
|
+
flags.has('mock') ? 'mock' : null,
|
|
6115
|
+
flags.has('emit') ? 'emit' : null,
|
|
6116
|
+
options.has('judge') ? 'judge' : null,
|
|
6117
|
+
options.has('score') ? 'score' : null,
|
|
6118
|
+
].filter((m) => m !== null);
|
|
6119
|
+
if (modes.length === 0)
|
|
6120
|
+
return fail('pick exactly one mode: --mock | --emit | --judge <file> | --score <file>');
|
|
6121
|
+
if (modes.length > 1)
|
|
6122
|
+
return fail(`modes are exclusive, got: ${modes.join(', ')}`);
|
|
6123
|
+
const mode = modes[0];
|
|
6124
|
+
const num = (key) => {
|
|
6125
|
+
const r = numOpt(options, key);
|
|
6126
|
+
if (r === null)
|
|
6127
|
+
return undefined;
|
|
6128
|
+
if ('error' in r)
|
|
6129
|
+
return r;
|
|
6130
|
+
return r.value;
|
|
6131
|
+
};
|
|
6132
|
+
// ── --mock: seeded synthetic outcomes, $0, exercises the real verdict math ──
|
|
6133
|
+
if (mode === 'mock') {
|
|
6134
|
+
for (const key of ['judge', 'score', 'work-order', 'project', 'out', 'limit']) {
|
|
6135
|
+
if (options.has(key))
|
|
6136
|
+
return fail(`--${key} is not valid with --mock`);
|
|
6137
|
+
}
|
|
6138
|
+
const parsed = {};
|
|
6139
|
+
for (const key of ['n', 'effect', 'tie-rate', 'seed']) {
|
|
6140
|
+
const v = num(key);
|
|
6141
|
+
if (typeof v === 'object' && v !== null)
|
|
6142
|
+
return fail(v.error);
|
|
6143
|
+
parsed[key] = v;
|
|
6144
|
+
}
|
|
6145
|
+
const outcomes = generateMockOutcomes({
|
|
6146
|
+
n: parsed.n ?? DEFAULT_MOCK_N,
|
|
6147
|
+
effect: parsed.effect ?? 0,
|
|
6148
|
+
tieRate: parsed['tie-rate'] ?? 0,
|
|
6149
|
+
seed: parsed.seed ?? DEFAULT_MOCK_SEED,
|
|
6150
|
+
});
|
|
6151
|
+
const marginOpt = numOpt(options, 'margin');
|
|
6152
|
+
if (marginOpt !== null && 'error' in marginOpt)
|
|
6153
|
+
return fail(marginOpt.error);
|
|
6154
|
+
const result = scoreEpochReplay(outcomes, {
|
|
6155
|
+
slice: options.get('slice') ?? 'all',
|
|
6156
|
+
...(marginOpt !== null ? { margin: marginOpt.value } : {}),
|
|
6157
|
+
});
|
|
6158
|
+
if (result.refusal !== null)
|
|
6159
|
+
return fail(result.refusal);
|
|
6160
|
+
if (json) {
|
|
6161
|
+
write(JSON.stringify({ mode: 'mock', synthetic: true, ...result, exitCode: 0 }, null, 2));
|
|
6162
|
+
}
|
|
6163
|
+
else {
|
|
6164
|
+
write(renderEpochReplayResult(result));
|
|
6165
|
+
write('');
|
|
6166
|
+
write(` SYNTHETIC (--mock): outcomes generated with seed ${parsed.seed ?? DEFAULT_MOCK_SEED} at a TRUE effect of ${parsed.effect ?? 0}. ` +
|
|
6167
|
+
'This exercises the protocol, it is NOT evidence about the learning loop.');
|
|
6168
|
+
}
|
|
6169
|
+
return 0;
|
|
6170
|
+
}
|
|
6171
|
+
// ── --emit: the generation work order (real mode, stage 1) ──
|
|
6172
|
+
if (mode === 'emit') {
|
|
6173
|
+
for (const key of ['judge', 'score', 'work-order', 'n', 'effect', 'tie-rate', 'slice']) {
|
|
6174
|
+
if (options.has(key))
|
|
6175
|
+
return fail(`--${key} is not valid with --emit`);
|
|
6176
|
+
}
|
|
6177
|
+
const root = resolve(cwd, options.get('project') ?? '.');
|
|
6178
|
+
const seed = num('seed');
|
|
6179
|
+
if (typeof seed === 'object' && seed !== null)
|
|
6180
|
+
return fail(seed.error);
|
|
6181
|
+
const limit = num('limit');
|
|
6182
|
+
if (typeof limit === 'object' && limit !== null)
|
|
6183
|
+
return fail(limit.error);
|
|
6184
|
+
// HIGH-B: the non-superiority margin is PRE-REGISTERED here, digest-covered, and read back by
|
|
6185
|
+
// --score. Out of range is refused, never clamped — `--margin 99` must not buy FALSIFIED.
|
|
6186
|
+
const emitMargin = num('margin');
|
|
6187
|
+
if (typeof emitMargin === 'object' && emitMargin !== null)
|
|
6188
|
+
return fail(emitMargin.error);
|
|
6189
|
+
if (typeof emitMargin === 'number' && !isValidMargin(emitMargin)) {
|
|
6190
|
+
return fail(`--margin ${emitMargin} must be in (0, 0.5] — refused, not clamped: an oversized margin buys FALSIFIED`);
|
|
6191
|
+
}
|
|
6192
|
+
const lessonText = new Map();
|
|
6193
|
+
for (const r of loadStoreRecords(root)) {
|
|
6194
|
+
if (typeof r.text === 'string' && r.text.trim() !== '')
|
|
6195
|
+
lessonText.set(r.id, r.text);
|
|
6196
|
+
}
|
|
6197
|
+
// The SAME reader `dz compounding` uses — readiness and the runner must see one corpus.
|
|
6198
|
+
const instances = replayableInstances(readRecallUsageEvents(root), lessonText);
|
|
6199
|
+
const order = buildWorkOrder(instances, {
|
|
6200
|
+
...(typeof seed === 'number' ? { seed } : {}),
|
|
6201
|
+
...(typeof limit === 'number' ? { limit } : {}),
|
|
6202
|
+
...(typeof emitMargin === 'number' ? { margin: emitMargin } : {}),
|
|
6203
|
+
});
|
|
6204
|
+
const outPath = resolve(cwd, options.get('out') ?? join(root, EPOCH_REPLAY_DIR, 'work-order.json'));
|
|
6205
|
+
try {
|
|
6206
|
+
writeJsonOut(outPath, order);
|
|
6207
|
+
}
|
|
6208
|
+
catch (e) {
|
|
6209
|
+
return fail(`cannot write ${outPath}: ${e instanceof Error ? e.message : String(e)}`);
|
|
6210
|
+
}
|
|
6211
|
+
if (json) {
|
|
6212
|
+
write(JSON.stringify({ mode: 'emit', out: outPath, instances: order.items.length, seed: order.seed, margin: order.margin, digest: order.digest, corpusFingerprint: order.corpusFingerprint, emittedAt: order.emittedAt, exitCode: 0 }, null, 2));
|
|
6213
|
+
}
|
|
6214
|
+
else
|
|
6215
|
+
write(renderWorkOrderSummary(order, outPath));
|
|
6216
|
+
return 0;
|
|
6217
|
+
}
|
|
6218
|
+
// ── --judge: blind judge prompts from a FILLED work order (real mode, stage 2) ──
|
|
6219
|
+
if (mode === 'judge') {
|
|
6220
|
+
for (const key of ['score', 'work-order', 'n', 'effect', 'tie-rate', 'slice', 'limit', 'seed', 'margin']) {
|
|
6221
|
+
if (options.has(key))
|
|
6222
|
+
return fail(`--${key} is not valid with --judge`);
|
|
6223
|
+
}
|
|
6224
|
+
const inPath = resolve(cwd, options.get('judge'));
|
|
6225
|
+
const read = readJsonFile(inPath);
|
|
6226
|
+
if ('error' in read)
|
|
6227
|
+
return fail(read.error);
|
|
6228
|
+
const verified = asVerifiedWorkOrder(read.value);
|
|
6229
|
+
if ('problems' in verified) {
|
|
6230
|
+
return fail(`${inPath} is not a verifiable ${WORK_ORDER_KIND}: ${verified.problems.join('; ')} (emit one with \`dz epoch-replay --emit\`)`);
|
|
6231
|
+
}
|
|
6232
|
+
const result = buildJudgePrompts(verified.order);
|
|
6233
|
+
const outPath = resolve(cwd, options.get('out') ?? join(dirname(inPath), 'judge-prompts.json'));
|
|
6234
|
+
try {
|
|
6235
|
+
// The JUDGE-FACING artifact. Its whole content is {id, prompt} per item — no `warmIsA`, no
|
|
6236
|
+
// arm names, no path back to the work order, and NOT the `skipped` list (whose reasons name
|
|
6237
|
+
// arms). Anything else here hands the judge the answer key (Codex QE CRITICAL-1).
|
|
6238
|
+
writeJsonOut(outPath, {
|
|
6239
|
+
kind: 'dz-epoch-replay-judge-prompts',
|
|
6240
|
+
version: 1,
|
|
6241
|
+
prompts: result.prompts.map((p) => ({ id: p.id, prompt: p.prompt })),
|
|
6242
|
+
});
|
|
6243
|
+
}
|
|
6244
|
+
catch (e) {
|
|
6245
|
+
return fail(`cannot write ${outPath}: ${e instanceof Error ? e.message : String(e)}`);
|
|
6246
|
+
}
|
|
6247
|
+
// `skipped` is OPERATOR-facing only — stdout / --json, never the file.
|
|
6248
|
+
if (json)
|
|
6249
|
+
write(JSON.stringify({ mode: 'judge', out: outPath, prompts: result.prompts.length, skipped: result.skipped, exitCode: 0 }, null, 2));
|
|
6250
|
+
else
|
|
6251
|
+
write(renderJudgePromptsSummary(result, outPath));
|
|
6252
|
+
return 0;
|
|
6253
|
+
}
|
|
6254
|
+
// ── --score: un-blind + verdict (real mode, stage 3) ──
|
|
6255
|
+
for (const key of ['n', 'effect', 'tie-rate', 'limit', 'seed', 'out', 'project']) {
|
|
6256
|
+
if (options.has(key))
|
|
6257
|
+
return fail(`--${key} is not valid with --score`);
|
|
6258
|
+
}
|
|
6259
|
+
// HIGH-B: a margin chosen once the counts are visible is not a pre-registration — and `--margin 99`
|
|
6260
|
+
// at scoring time would simply buy FALSIFIED. Real mode reads it from the work order, full stop.
|
|
6261
|
+
if (options.has('margin')) {
|
|
6262
|
+
return fail('--margin is not valid with --score: the non-superiority margin is PRE-REGISTERED at --emit and stored in the work order (re-emit to change it)');
|
|
6263
|
+
}
|
|
6264
|
+
const orderPath = options.get('work-order');
|
|
6265
|
+
if (orderPath === undefined) {
|
|
6266
|
+
return fail('--score requires --work-order <file>: un-blinding must use the PRE-REGISTERED assignment, not a label in the judgments file');
|
|
6267
|
+
}
|
|
6268
|
+
const orderRead = readJsonFile(resolve(cwd, orderPath));
|
|
6269
|
+
if ('error' in orderRead)
|
|
6270
|
+
return fail(orderRead.error);
|
|
6271
|
+
const verifiedOrder = asVerifiedWorkOrder(orderRead.value);
|
|
6272
|
+
if ('problems' in verifiedOrder) {
|
|
6273
|
+
return fail(`${resolve(cwd, orderPath)} is not a verifiable ${WORK_ORDER_KIND} — refusing to un-blind against it: ${verifiedOrder.problems.join('; ')}`);
|
|
6274
|
+
}
|
|
6275
|
+
const order = verifiedOrder.order;
|
|
6276
|
+
const judgePath = resolve(cwd, options.get('score'));
|
|
6277
|
+
const judgeRead = readJsonFile(judgePath);
|
|
6278
|
+
if ('error' in judgeRead)
|
|
6279
|
+
return fail(judgeRead.error);
|
|
6280
|
+
const rawRows = Array.isArray(judgeRead.value)
|
|
6281
|
+
? judgeRead.value
|
|
6282
|
+
: typeof judgeRead.value === 'object' && judgeRead.value !== null && Array.isArray(judgeRead.value.judgments)
|
|
6283
|
+
? judgeRead.value.judgments
|
|
6284
|
+
: null;
|
|
6285
|
+
if (rawRows === null)
|
|
6286
|
+
return fail(`${judgePath} must be an array of {id, winner} rows (or {"judgments": [...]})`);
|
|
6287
|
+
const unblind = unblindJudgments(order, rawRows.map((r) => {
|
|
6288
|
+
const o = (typeof r === 'object' && r !== null ? r : {});
|
|
6289
|
+
return { id: typeof o.id === 'string' ? o.id : '', winner: typeof o.winner === 'string' ? o.winner : '' };
|
|
6290
|
+
}));
|
|
6291
|
+
// A duplicated judgement id is corrupt input, not a skippable row — refuse loudly.
|
|
6292
|
+
if (!unblind.ok)
|
|
6293
|
+
return fail(unblind.error ?? 'judgments refused');
|
|
6294
|
+
const { outcomes, skipped } = unblind;
|
|
6295
|
+
const result = scoreEpochReplay(outcomes, {
|
|
6296
|
+
slice: options.get('slice') ?? 'all',
|
|
6297
|
+
margin: order.margin, // PRE-REGISTERED in the work order, verified by the digest
|
|
6298
|
+
});
|
|
6299
|
+
if (result.refusal !== null)
|
|
6300
|
+
return fail(result.refusal);
|
|
6301
|
+
if (json) {
|
|
6302
|
+
write(JSON.stringify({
|
|
6303
|
+
mode: 'score',
|
|
6304
|
+
workOrder: resolve(cwd, orderPath),
|
|
6305
|
+
judgments: judgePath,
|
|
6306
|
+
scored: outcomes.length,
|
|
6307
|
+
skipped,
|
|
6308
|
+
provenance: { seed: order.seed, margin: order.margin, emittedAt: order.emittedAt, digest: order.digest, corpusFingerprint: order.corpusFingerprint, digestScope: DIGEST_HONEST_SCOPE },
|
|
6309
|
+
...result,
|
|
6310
|
+
exitCode: 0,
|
|
6311
|
+
}, null, 2));
|
|
6312
|
+
}
|
|
6313
|
+
else {
|
|
6314
|
+
write(renderEpochReplayResult(result));
|
|
6315
|
+
if (skipped.length > 0) {
|
|
6316
|
+
write('');
|
|
6317
|
+
write(` ${skipped.length} judgment(s) SKIPPED (never guessed):`);
|
|
6318
|
+
for (const s of skipped)
|
|
6319
|
+
write(` · ${s.id}: ${s.reason}`);
|
|
6320
|
+
}
|
|
6321
|
+
// Provenance, so a reviewer can ask for the original emitted file and compare.
|
|
6322
|
+
write('');
|
|
6323
|
+
write(` WORK ORDER: seed ${order.seed} · margin ${order.margin} (pre-registered) · emitted ${order.emittedAt}`);
|
|
6324
|
+
write(` digest ${order.digest}`);
|
|
6325
|
+
write(` corpus ${order.corpusFingerprint}`);
|
|
6326
|
+
write(` ${DIGEST_HONEST_SCOPE}`);
|
|
6327
|
+
}
|
|
6328
|
+
return 0;
|
|
6329
|
+
}
|
|
5592
6330
|
/**
|
|
5593
6331
|
* Env vars an INHERITED Claude session leaks into a child. Left in place, the probe can silently
|
|
5594
6332
|
* read the parent's project instead of the target — the exact confound that made a hand-rolled
|
|
@@ -6756,6 +7494,8 @@ export async function runCli(argv, io = {}) {
|
|
|
6756
7494
|
return cmdSkillsVerify(options, flags, cwd, write);
|
|
6757
7495
|
case 'compounding':
|
|
6758
7496
|
return cmdCompounding(options, flags, cwd, write);
|
|
7497
|
+
case 'epoch-replay':
|
|
7498
|
+
return cmdEpochReplay(options, flags, cwd, write);
|
|
6759
7499
|
case 'score':
|
|
6760
7500
|
return cmdScore(options, flags, cwd, write);
|
|
6761
7501
|
case 'backlog':
|