@ngockhoale/ukit 3.0.7 → 3.0.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (105) hide show
  1. package/CHANGELOG.md +18 -1
  2. package/manifests/documentation.yaml +11 -0
  3. package/package.json +1 -1
  4. package/scripts/audit/decision-coverage.mjs +29 -2
  5. package/scripts/bench/data-foundation.mjs +52 -3
  6. package/scripts/bench/decision-runtime-baseline.mjs +427 -0
  7. package/scripts/bench/decision-runtime-metrics.mjs +67 -0
  8. package/scripts/bench/decision-runtime-variant.mjs +626 -0
  9. package/scripts/bench/memory-ablation.mjs +495 -0
  10. package/scripts/bench/memory-baseline.mjs +596 -0
  11. package/scripts/bench/memory-bench.mjs +661 -0
  12. package/scripts/bench/memory-canary.mjs +321 -0
  13. package/scripts/bench/memory-corpus.mjs +354 -0
  14. package/scripts/bench/memory-gate.mjs +389 -0
  15. package/scripts/bench/memory-metrics.mjs +179 -0
  16. package/scripts/bench/parallel-agents.mjs +33 -11
  17. package/scripts/bench/recorder-overhead.mjs +204 -0
  18. package/scripts/bench/sqlite-spike.mjs +451 -0
  19. package/scripts/measure-decision-gateway.mjs +306 -0
  20. package/scripts/perf/audit-perf.mjs +35 -17
  21. package/src/bug/triageBug.js +4 -3
  22. package/src/cli/commands/memory.js +357 -63
  23. package/src/context/detectProjectContext.js +11 -1
  24. package/src/core/agentRuntime/adapters.js +254 -0
  25. package/src/core/agentRuntime/artifacts.js +192 -0
  26. package/src/core/agentRuntime/completionGate.js +176 -0
  27. package/src/core/agentRuntime/context.js +149 -0
  28. package/src/core/agentRuntime/contract.js +247 -0
  29. package/src/core/agentRuntime/diagnostics.js +244 -0
  30. package/src/core/agentRuntime/evaluation.js +163 -0
  31. package/src/core/agentRuntime/eventStore.js +404 -0
  32. package/src/core/agentRuntime/liveness.js +60 -0
  33. package/src/core/agentRuntime/planCompiler.js +322 -0
  34. package/src/core/agentRuntime/promotion.js +53 -0
  35. package/src/core/agentRuntime/qualityComparison.js +112 -0
  36. package/src/core/agentRuntime/recovery.js +266 -0
  37. package/src/core/agentRuntime/resourcePolicy.js +78 -0
  38. package/src/core/agentRuntime/runtimeSupport.js +237 -0
  39. package/src/core/agentRuntime/supervisor.js +565 -0
  40. package/src/core/agentRuntime/vmEngine.js +621 -0
  41. package/src/core/codeintel/analogy.js +3 -2
  42. package/src/core/experiments/dynamicWorkflow.js +17 -2
  43. package/src/core/fileOps.js +21 -3
  44. package/src/core/memory/deltaOverlays.js +75 -30
  45. package/src/core/memory/learningCandidates.js +93 -48
  46. package/src/core/memory/memoryFlags.js +83 -0
  47. package/src/core/memory/memoryFreshness.js +190 -0
  48. package/src/core/memory/memoryHit.js +144 -0
  49. package/src/core/memory/migrate.js +69 -189
  50. package/src/core/memory/migrateMapping.js +232 -0
  51. package/src/core/memory/mutateMemory.js +323 -0
  52. package/src/core/memory/policy.js +96 -0
  53. package/src/core/memory/projectIdentity.js +266 -0
  54. package/src/core/memory/recordIndex.js +178 -0
  55. package/src/core/memory/recordStore.js +133 -20
  56. package/src/core/memory/records.js +144 -6
  57. package/src/core/memory/retrieval.js +259 -125
  58. package/src/core/memory/store.js +16 -5
  59. package/src/core/memory/storeBackup.js +226 -0
  60. package/src/core/memory/storeV2.js +63 -26
  61. package/src/core/memory/storeV2Loader.js +30 -12
  62. package/src/core/memory/userMemory.js +38 -20
  63. package/src/core/memory/writeClassification.js +161 -0
  64. package/src/core/memory/writeGuard.js +129 -0
  65. package/src/core/observability/adapters/hookTelemetryAdapter.js +90 -0
  66. package/src/core/observability/analytics/cohorts.js +148 -0
  67. package/src/core/observability/analytics/storeDigest.js +163 -0
  68. package/src/core/observability/evaluation/experimentPlan.js +95 -0
  69. package/src/core/observability/evaluation/findings.js +99 -0
  70. package/src/core/observability/evaluation/optimizationKnowledge.js +10 -1
  71. package/src/core/observability/evaluation/perturbation.js +273 -0
  72. package/src/core/observability/evaluation/replay.js +7 -1
  73. package/src/core/observability/evaluation/scorecard.js +23 -3
  74. package/src/core/observability/rollout.js +11 -7
  75. package/src/core/observability/schema/compatibility.js +135 -0
  76. package/src/core/observability/schema/registry.js +99 -0
  77. package/src/core/observability/schema/validate.js +7 -0
  78. package/src/core/observability/support/import.js +53 -9
  79. package/src/core/observability/support/paths.js +13 -3
  80. package/src/core/observability/support/projector.js +148 -12
  81. package/src/core/output/index.js +12 -2
  82. package/src/core/runtimeConfig.js +83 -0
  83. package/src/core/runtimePaths.js +3 -0
  84. package/src/core/sensitiveValueScanner.js +40 -0
  85. package/src/core/token/index.js +40 -3
  86. package/src/decision/client.js +37 -13
  87. package/src/decision/protocol.js +1 -1
  88. package/src/decision/registry.js +5 -3
  89. package/src/decision/runtimeDecide.js +242 -0
  90. package/src/decision/runtimeFilter.js +150 -0
  91. package/src/decision/runtimeScheduler.js +239 -0
  92. package/src/index/buildIndex.js +13 -12
  93. package/src/index/queryIndex.js +35 -14
  94. package/src/index/relatedTests.js +50 -8
  95. package/src/index/resolveContext.js +9 -4
  96. package/src/manifest/selectItems.js +7 -3
  97. package/src/render/instructionRenderer.js +17 -5
  98. package/template_project/.claude/ukit/index/lib/index-core.mjs +94 -39
  99. package/template_project/.claude/ukit/index/route-task.mjs +121 -19
  100. package/template_project/.claude/ukit/index/unic-decision.mjs +28 -13
  101. package/template_project/.claude/ukit/runtime/memory-flags.mjs +51 -0
  102. package/template_project/.claude/ukit/runtime/memory-freshness.mjs +155 -0
  103. package/template_project/.claude/ukit/runtime/memory-policy.mjs +286 -0
  104. package/template_project/.claude/ukit/runtime/output-compression.mjs +3 -0
  105. package/template_project/.claude/ukit/runtime/reinject-context.mjs +145 -14
@@ -0,0 +1,95 @@
1
+ /**
2
+ * experimentPlan — pure staged controlled-experiment planner (GAP-c).
3
+ *
4
+ * Maps a KB optimization entry + a `computeGateComparison` GateComparison to
5
+ * staged rollout ACTIONS over the seam ladder. The planner EMITS intent only:
6
+ * it never mutates config, never calls killSwitch/resolveSeamConfig, and never
7
+ * touches fs. Execution (if any) is a separate operator-owned act.
8
+ *
9
+ * Rules:
10
+ * - gate must be a real GateComparison (verdict in pass|non-inferior|unknown);
11
+ * anything else → TypeError (a fabricated gate can never plan a promotion).
12
+ * - abort flag (operator rollback_criterion met) → kill 'abort-requested'.
13
+ * - gate.forbiddenFailures > 0 → kill 'forbidden-failures' prepended.
14
+ * - verdict !== 'pass' → hold (reason = verdict). 'non-inferior' means the
15
+ * variant FAILED the pre-registered gate.
16
+ * - verdict 'pass' + non-empty controlled_rollout.evidence_refs → one promote
17
+ * action per seam in stage-rank order (shadow → canary → default), each
18
+ * carrying requires_evidence_refs:true. Missing refs → hold
19
+ * 'missing-controlled-rollout-evidence': promotion never fires on the
20
+ * quality signal alone.
21
+ */
22
+
23
+ import { SEAM_MIN_STAGE } from '../rollout.js';
24
+
25
+ export const ROLLOUT_ACTIONS = Object.freeze(['promote', 'hold', 'kill']);
26
+
27
+ const GATE_VERDICTS = new Set(['pass', 'non-inferior', 'unknown']);
28
+
29
+ const STAGE_RANK = { shadow: 1, canary: 2, default: 3 };
30
+
31
+ function isPlainObject(value) {
32
+ return value !== null && typeof value === 'object' && !Array.isArray(value);
33
+ }
34
+
35
+ function isNonEmptyStringList(value) {
36
+ return (
37
+ Array.isArray(value) &&
38
+ value.length > 0 &&
39
+ value.every((v) => typeof v === 'string' && v.length > 0)
40
+ );
41
+ }
42
+
43
+ /**
44
+ * @param {{optimization?: object, gate?: object, abort?: boolean}} input
45
+ * @returns {{actions: object[]}} frozen action list
46
+ */
47
+ export function planControlledRollout(input = {}) {
48
+ const { optimization, gate, abort } = isPlainObject(input) ? input : {};
49
+
50
+ if (!isPlainObject(gate) || !GATE_VERDICTS.has(gate.verdict)) {
51
+ throw new TypeError(
52
+ 'experimentPlan: gate must be a GateComparison with verdict in pass|non-inferior|unknown',
53
+ );
54
+ }
55
+
56
+ const actions = [];
57
+
58
+ if (abort === true) {
59
+ actions.push({ type: 'kill', reason: 'abort-requested' });
60
+ }
61
+ if (gate.forbiddenFailures > 0) {
62
+ actions.push({ type: 'kill', reason: 'forbidden-failures' });
63
+ }
64
+
65
+ if (gate.verdict !== 'pass') {
66
+ actions.push({ type: 'hold', reason: gate.verdict });
67
+ return { actions: Object.freeze(actions) };
68
+ }
69
+
70
+ if (actions.length > 0) {
71
+ // pass + forbiddenFailures is unreachable from real computeGateComparison
72
+ // output (forbidden → non-inferior hard gate); defend the shape anyway.
73
+ return { actions: Object.freeze(actions) };
74
+ }
75
+
76
+ const rollout = isPlainObject(optimization) ? optimization.controlled_rollout : undefined;
77
+ const evidenceRefs = isPlainObject(rollout) ? rollout.evidence_refs : undefined;
78
+ if (!isNonEmptyStringList(evidenceRefs)) {
79
+ return {
80
+ actions: Object.freeze([{ type: 'hold', reason: 'missing-controlled-rollout-evidence' }]),
81
+ };
82
+ }
83
+
84
+ // Order from the shared seam contract (read-only): each seam promotes at the
85
+ // stage it unlocks, lowest stage rank first.
86
+ const seams = Object.entries(SEAM_MIN_STAGE)
87
+ .map(([seam, stage]) => ({ seam, stage }))
88
+ .sort((a, b) => STAGE_RANK[a.stage] - STAGE_RANK[b.stage] || a.seam.localeCompare(b.seam));
89
+
90
+ for (const { seam, stage } of seams) {
91
+ actions.push({ type: 'promote', seam, stage, requires_evidence_refs: true });
92
+ }
93
+
94
+ return { actions: Object.freeze(actions) };
95
+ }
@@ -0,0 +1,99 @@
1
+ /**
2
+ * findings.js (TASK-002, SPEC §2 G6-FR04) — evaluator findings validator.
3
+ *
4
+ * validateEvaluatorFindings(payload) → { findings[], rejected[] }
5
+ *
6
+ * Advisory-only boundary between the optional AI evaluator and the rest
7
+ * of the pipeline: an evaluator's opinion can never be promoted to fact
8
+ * or policy. Every accepted finding is re-shaped by construction to
9
+ * { claim, evidence_refs, confidence, alternatives?, decision:
10
+ * 'unsupported', applied: false }
11
+ * and frozen — caller-injected `decision`/`applied`/`promoted` fields are
12
+ * never propagated.
13
+ *
14
+ * Evidence gate: a candidate without a non-empty `claim` (non-empty
15
+ * string) or a non-empty `evidence_refs` (all-string array) is moved to
16
+ * `rejected[]` with `{ candidate_index, reason }` — never silently
17
+ * dropped. Out-of-enum confidence coerces to 'unknown'; `alternatives`
18
+ * carries over as advisory hints (strings only). Malformed payloads
19
+ * (non-object, non-array, or a non-array `findings` field) reject the
20
+ * whole payload with `candidate_index: null`.
21
+ *
22
+ * Pure module: no fs, no clock, no model calls; never throws on data.
23
+ */
24
+
25
+ export const FINDING_CONFIDENCE = Object.freeze([
26
+ 'high',
27
+ 'medium',
28
+ 'low',
29
+ 'unknown',
30
+ ]);
31
+
32
+ function isPlainObject(value) {
33
+ return value !== null && typeof value === 'object' && !Array.isArray(value);
34
+ }
35
+
36
+ function isNonEmptyString(value) {
37
+ return typeof value === 'string' && value.length > 0;
38
+ }
39
+
40
+ function isStringList(value) {
41
+ return Array.isArray(value) && value.length > 0 && value.every(isNonEmptyString);
42
+ }
43
+
44
+ function candidates(payload) {
45
+ if (Array.isArray(payload)) return payload;
46
+ if (isPlainObject(payload) && Array.isArray(payload.findings)) {
47
+ return payload.findings;
48
+ }
49
+ return null;
50
+ }
51
+
52
+ /**
53
+ * Validate evaluator-returned findings. `payload` may be either
54
+ * `{ findings: [...] }` or a bare findings array — both accept the same
55
+ * candidate shape.
56
+ */
57
+ export function validateEvaluatorFindings(payload) {
58
+ const list = candidates(payload);
59
+ if (list === null) {
60
+ return {
61
+ findings: [],
62
+ rejected: [{ candidate_index: null, reason: 'malformed-payload' }],
63
+ };
64
+ }
65
+
66
+ const findings = [];
67
+ const rejected = [];
68
+
69
+ for (let i = 0; i < list.length; i += 1) {
70
+ const candidate = isPlainObject(list[i]) ? list[i] : {};
71
+ if (!isNonEmptyString(candidate.claim)) {
72
+ rejected.push({ candidate_index: i, reason: 'missing-claim' });
73
+ continue;
74
+ }
75
+ if (!isStringList(candidate.evidence_refs)) {
76
+ rejected.push({ candidate_index: i, reason: 'missing-evidence-refs' });
77
+ continue;
78
+ }
79
+
80
+ const finding = {
81
+ claim: candidate.claim,
82
+ evidence_refs: [...candidate.evidence_refs],
83
+ confidence: FINDING_CONFIDENCE.includes(candidate.confidence)
84
+ ? candidate.confidence
85
+ : 'unknown',
86
+ };
87
+ if (Array.isArray(candidate.alternatives)) {
88
+ const alternatives = candidate.alternatives.filter(isNonEmptyString);
89
+ if (alternatives.length > 0) finding.alternatives = alternatives;
90
+ }
91
+ // Advisory-only by construction: evaluator decision/applied fields
92
+ // are never read; the output shape is fixed.
93
+ finding.decision = 'unsupported';
94
+ finding.applied = false;
95
+ findings.push(Object.freeze(finding));
96
+ }
97
+
98
+ return { findings, rejected };
99
+ }
@@ -18,7 +18,8 @@
18
18
  *
19
19
  * Invariants (SPEC §12, DF-FR12):
20
20
  * - `decision: 'promoted'` or `applied: true` REQUIRE
21
- * `controlled_rollout.evidence_refs` — no promotion from token
21
+ * `controlled_rollout.evidence_refs` AND a non-empty
22
+ * `alternative_explanations` list — no promotion from token
22
23
  * reduction alone; violating entries throw TypeError.
23
24
  * - `decision: 'unsupported'` records can never carry `applied: true`.
24
25
  * - This module writes ONLY inside `root` — no runtime config, no
@@ -110,6 +111,13 @@ export function createOptimizationStore({ root, clock } = {}) {
110
111
  throw new TypeError(
111
112
  'recordOptimization: decision promoted / applied requires controlled_rollout.evidence_refs',
112
113
  );
114
+ } else if (decision === 'promoted' || applied) {
115
+ const alternatives = stringList(entry.alternative_explanations);
116
+ if (alternatives.length === 0) {
117
+ throw new TypeError(
118
+ 'recordOptimization: decision promoted / applied requires alternative_explanations',
119
+ );
120
+ }
113
121
  }
114
122
 
115
123
  const core = {
@@ -127,6 +135,7 @@ export function createOptimizationStore({ root, clock } = {}) {
127
135
  rollback_criterion:
128
136
  typeof entry.rollback_criterion === 'string' ? entry.rollback_criterion : null,
129
137
  supersedes: stringList(entry.supersedes),
138
+ alternative_explanations: stringList(entry.alternative_explanations),
130
139
  created_at: now(),
131
140
  };
132
141
 
@@ -0,0 +1,273 @@
1
+ /**
2
+ * perturbation.js (TASK-002, SPEC §2 G5-FR06/G5-FR07) — synthetic mutation harness.
3
+ *
4
+ * mutateSummaries(summaries, { seed, plan? }) → { summaries, manifest }
5
+ * detectPerturbations(mutatedSummaries, baselineSummaries, manifest)
6
+ * → { injected, detected, missed[], details[] }
7
+ *
8
+ * Pure module: no fs, no clock, no unseeded randomness (mulberry32, same
9
+ * construction as the bench corpus generator). Mutations edit TraceSummary
10
+ * structures so the downstream pipeline sees a genuinely different trace;
11
+ * detection is measured through the REAL computeFingerprint /
12
+ * detectOpportunities path — never from the manifest alone.
13
+ *
14
+ * Mutation kinds (MUTATION_KINDS):
15
+ * drop_record — removes a span and raises drops.dropped_count>0
16
+ * retry_inflation — adds an attempt_index>1 model span and raises retries
17
+ * cache_miss_spike — adds cache misses (cache.observed, hit_rate recomputed)
18
+ * orphan_span — rewrites parent_span_id to an absent id
19
+ * telemetry_incomplete — flips telemetry_complete=false
20
+ *
21
+ * Detection rule (per manifest entry): the mutated trace is detected when its
22
+ * fingerprint kind/count diverges from its own baseline, OR — once the trace
23
+ * is proven changed (hash divergence) — its opportunity cluster reports
24
+ * `inconclusive` (the honesty path telemetry_incomplete relies on) or a
25
+ * non-clean pattern. An unmutated trace claimed in the manifest cannot be
26
+ * detected: hash-identical traces have nothing to report.
27
+ */
28
+
29
+ import { computeFingerprint } from '../analytics/fingerprints.js';
30
+ import { detectOpportunities } from '../analytics/opportunities.js';
31
+
32
+ export const MUTATION_KINDS = Object.freeze([
33
+ 'drop_record',
34
+ 'retry_inflation',
35
+ 'cache_miss_spike',
36
+ 'orphan_span',
37
+ 'telemetry_incomplete',
38
+ ]);
39
+
40
+ // Deterministic PRNG (mulberry32) — identical construction to the bench corpus.
41
+ function mulberry32(seed) {
42
+ let a = seed >>> 0;
43
+ return function next() {
44
+ a |= 0;
45
+ a = (a + 0x6d2b79f5) | 0;
46
+ let t = Math.imul(a ^ (a >>> 15), 1 | a);
47
+ t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
48
+ return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
49
+ };
50
+ }
51
+
52
+ function isPlainObject(value) {
53
+ return value !== null && typeof value === 'object' && !Array.isArray(value);
54
+ }
55
+
56
+ function num(value) {
57
+ return typeof value === 'number' && Number.isFinite(value) ? value : 0;
58
+ }
59
+
60
+ /**
61
+ * Candidate indices for a kind — a mutation must produce a genuinely
62
+ * different trace, so kinds with preconditions filter to eligible traces.
63
+ * Prefers traces not already mutated by this call.
64
+ */
65
+ function candidateIndices(mutated, kind, used) {
66
+ const eligible = mutated
67
+ .map((s, idx) => ({ s, idx }))
68
+ .filter(({ s }) => {
69
+ if (!isPlainObject(s) || !Array.isArray(s.spans) || s.spans.length === 0) return false;
70
+ if (kind === 'telemetry_incomplete') return s.telemetry_complete === true;
71
+ return true;
72
+ })
73
+ .map(({ idx }) => idx);
74
+ const fresh = eligible.filter((idx) => !used.has(idx));
75
+ return fresh.length > 0 ? fresh : eligible;
76
+ }
77
+
78
+ function ensureCounters(summary) {
79
+ if (!isPlainObject(summary.retries)) summary.retries = {};
80
+ if (!isPlainObject(summary.drops)) summary.drops = {};
81
+ if (!isPlainObject(summary.drops.by_reason)) summary.drops.by_reason = {};
82
+ if (!isPlainObject(summary.cache)) summary.cache = {};
83
+ if (!isPlainObject(summary.coverage)) summary.coverage = {};
84
+ }
85
+ const MUTATORS = {
86
+ drop_record(summary) {
87
+ const idx = summary.spans.findIndex(
88
+ (s) => typeof s.parent_span_id === 'string' && s.parent_span_id.length > 0,
89
+ );
90
+ const i = idx >= 0 ? idx : 0;
91
+ const removed = summary.spans.splice(i, 1)[0] || null;
92
+ summary.drops.events = num(summary.drops.events) + 1;
93
+ summary.drops.dropped_count = num(summary.drops.dropped_count) + 1;
94
+ summary.drops.by_reason.synthetic_drop = num(summary.drops.by_reason.synthetic_drop) + 1;
95
+ summary.telemetry_complete = false;
96
+ summary.coverage.spans = summary.spans.length;
97
+ return { removed_span_id: removed ? removed.span_id : null };
98
+ },
99
+
100
+ retry_inflation(summary) {
101
+ const model =
102
+ summary.spans.find((s) => s.family === 'model') ||
103
+ summary.spans[summary.spans.length - 1] ||
104
+ null;
105
+ const maxAttempt = summary.spans.reduce(
106
+ (m, s) => (Number.isInteger(s.attempt_index) ? Math.max(m, s.attempt_index) : m),
107
+ 0,
108
+ );
109
+ const attemptIndex = Math.max(2, maxAttempt + 1);
110
+ const clone = {
111
+ ...(model || {
112
+ parent_span_id: null,
113
+ family: 'model',
114
+ operation: null,
115
+ status: 'completed',
116
+ duration_ms: 0,
117
+ }),
118
+ span_id: `${model ? model.span_id : 'span'}-retry${attemptIndex}`,
119
+ attempt_index: attemptIndex,
120
+ status: 'completed',
121
+ start_record_id: `rec-${summary.trace_id}-retry${attemptIndex}-start`,
122
+ end_record_id: `rec-${summary.trace_id}-retry${attemptIndex}-end`,
123
+ };
124
+ summary.spans.push(clone);
125
+ summary.retries.retries = num(summary.retries.retries) + 1;
126
+ summary.retries.model_attempts = num(summary.retries.model_attempts) + 1;
127
+ summary.coverage.spans = summary.spans.length;
128
+ return { added_attempt_span: clone.span_id, attempt_index: attemptIndex };
129
+ },
130
+
131
+ cache_miss_spike(summary, rand) {
132
+ const added = 2 + Math.floor(rand() * 3); // 2..4 misses
133
+ summary.cache.observed = true;
134
+ summary.cache.hits = num(summary.cache.hits);
135
+ summary.cache.misses = num(summary.cache.misses) + added;
136
+ const lookups = summary.cache.hits + summary.cache.misses;
137
+ summary.cache.hit_rate = lookups > 0 ? summary.cache.hits / lookups : null;
138
+ return { misses_added: added };
139
+ },
140
+
141
+ orphan_span(summary) {
142
+ const idx = summary.spans.findIndex(
143
+ (s) => typeof s.parent_span_id === 'string' && s.parent_span_id.length > 0,
144
+ );
145
+ const i = idx >= 0 ? idx : 0;
146
+ const target = summary.spans[i];
147
+ const previous = target.parent_span_id ?? null;
148
+ target.parent_span_id = `absent-${target.span_id}`;
149
+ summary.coverage.orphan_spans = num(summary.coverage.orphan_spans) + 1;
150
+ return { span_id: target.span_id, previous_parent: previous };
151
+ },
152
+
153
+ telemetry_incomplete(summary) {
154
+ const previous = summary.telemetry_complete;
155
+ summary.telemetry_complete = false;
156
+ return { previous_telemetry_complete: previous };
157
+ },
158
+ };
159
+
160
+ /**
161
+ * @param {Array<object>} summaries — baseline TraceSummary objects (not mutated).
162
+ * @param {{ seed?: number, plan?: Array<{kind:string}> }} [options]
163
+ * plan defaults to one mutation per MUTATION_KINDS entry.
164
+ * @returns {{ summaries: Array<object>, manifest: Array<object> }}
165
+ * manifest entries: { mutation_id, trace_id, kind, detail }.
166
+ */
167
+ export function mutateSummaries(summaries, { seed = 42, plan } = {}) {
168
+ if (!Array.isArray(summaries)) {
169
+ throw new TypeError('mutateSummaries: summaries must be an array of TraceSummary');
170
+ }
171
+ const effectivePlan = Array.isArray(plan) ? plan : MUTATION_KINDS.map((kind) => ({ kind }));
172
+ for (const entry of effectivePlan) {
173
+ if (!isPlainObject(entry) || !MUTATION_KINDS.includes(entry.kind)) {
174
+ throw new TypeError(
175
+ `mutateSummaries: plan entries need kind ∈ {${MUTATION_KINDS.join(', ')}}`,
176
+ );
177
+ }
178
+ }
179
+
180
+ const rand = mulberry32(seed);
181
+ const mutated = summaries.map((s) => JSON.parse(JSON.stringify(s)));
182
+ const manifest = [];
183
+ const used = new Set();
184
+
185
+ effectivePlan.forEach((entry, i) => {
186
+ const pickFrom = candidateIndices(mutated, entry.kind, used);
187
+ if (pickFrom.length === 0) return;
188
+ const targetIdx = pickFrom[Math.floor(rand() * pickFrom.length)];
189
+ used.add(targetIdx);
190
+ const summary = mutated[targetIdx];
191
+ ensureCounters(summary);
192
+ const detail = MUTATORS[entry.kind](summary, rand);
193
+ manifest.push({
194
+ mutation_id: `mut-${String(i + 1).padStart(3, '0')}`,
195
+ trace_id: typeof summary.trace_id === 'string' ? summary.trace_id : `idx-${targetIdx}`,
196
+ kind: entry.kind,
197
+ detail,
198
+ });
199
+ });
200
+
201
+ return { summaries: mutated, manifest };
202
+ }
203
+
204
+ /**
205
+ * Measure detection through the real pipeline: computeFingerprint for
206
+ * per-trace divergence vs baseline and detectOpportunities for cluster
207
+ * status over the mutated set.
208
+ *
209
+ * @param {Array<object>} mutatedSummaries
210
+ * @param {Array<object>} baselineSummaries
211
+ * @param {Array<object>} manifest — entries from mutateSummaries.
212
+ * @returns {{ injected:number, detected:number, missed:Array<object>, details:Array<object> }}
213
+ */
214
+ export function detectPerturbations(mutatedSummaries, baselineSummaries, manifest) {
215
+ if (!Array.isArray(mutatedSummaries) || !Array.isArray(baselineSummaries) || !Array.isArray(manifest)) {
216
+ throw new TypeError('detectPerturbations: summaries and manifest must be arrays');
217
+ }
218
+
219
+ const baselineByTrace = new Map();
220
+ for (const s of baselineSummaries) {
221
+ if (isPlainObject(s) && typeof s.trace_id === 'string') baselineByTrace.set(s.trace_id, s);
222
+ }
223
+ const mutatedByTrace = new Map();
224
+ for (const s of mutatedSummaries) {
225
+ if (isPlainObject(s) && typeof s.trace_id === 'string') mutatedByTrace.set(s.trace_id, s);
226
+ }
227
+
228
+ // Real pipeline — never read the manifest as evidence.
229
+ const opportunities = detectOpportunities(mutatedSummaries);
230
+ const oppByTrace = new Map();
231
+ for (const opp of opportunities) {
232
+ for (const ref of opp.trace_refs || []) {
233
+ if (!oppByTrace.has(ref)) oppByTrace.set(ref, opp);
234
+ }
235
+ }
236
+
237
+ const details = [];
238
+ for (const entry of manifest) {
239
+ const mutated = mutatedByTrace.get(entry.trace_id);
240
+ const baseline = baselineByTrace.get(entry.trace_id);
241
+ const via = [];
242
+ let opportunity = null;
243
+ if (mutated && baseline) {
244
+ const fm = computeFingerprint(mutated);
245
+ const fb = computeFingerprint(baseline);
246
+ const changed = fm.hash !== fb.hash;
247
+ if (fm.kind !== fb.kind || fm.count !== fb.count) via.push('fingerprint');
248
+ opportunity = oppByTrace.get(entry.trace_id) || null;
249
+ if (changed && opportunity) {
250
+ if (opportunity.status === 'inconclusive') via.push('inconclusive');
251
+ else if (opportunity.fingerprint && opportunity.fingerprint.kind !== 'clean') {
252
+ via.push('opportunity');
253
+ }
254
+ }
255
+ }
256
+ details.push({
257
+ mutation_id: entry.mutation_id,
258
+ trace_id: entry.trace_id,
259
+ kind: entry.kind,
260
+ detected: via.length > 0,
261
+ via,
262
+ opportunity,
263
+ });
264
+ }
265
+
266
+ const missed = details.filter((d) => !d.detected);
267
+ return {
268
+ injected: manifest.length,
269
+ detected: details.length - missed.length,
270
+ missed,
271
+ details,
272
+ };
273
+ }
@@ -25,6 +25,8 @@
25
25
  * deterministically; explicit canonical values always win.
26
26
  */
27
27
 
28
+ import { SEMANTIC_REGISTRY } from '../schema/registry.js';
29
+
28
30
  export const REPLAY_TAGS = Object.freeze(['POLICY', 'CONTEXT', 'SIMULATED']);
29
31
 
30
32
  export const REPLAY_LIMITATIONS = Object.freeze({
@@ -92,7 +94,11 @@ function materializeRecord(raw, ctx, index) {
92
94
  }
93
95
  if (typeof record.monotonic_ns !== 'number') record.monotonic_ns = dtMs * 1e6;
94
96
  if (typeof record.importance !== 'string') record.importance = 'normal';
95
- if (typeof record.privacy_class !== 'string') record.privacy_class = 'public';
97
+ if (typeof record.privacy_class !== 'string') {
98
+ // Default to the semantic registry's privacy floor so materialized
99
+ // records satisfy the contract; unregistered names fall back to public.
100
+ record.privacy_class = SEMANTIC_REGISTRY[record.semantic_name]?.privacy_class ?? 'public';
101
+ }
96
102
  if (!isPlainObject(record.payload)) record.payload = {};
97
103
  return record;
98
104
  }
@@ -16,6 +16,10 @@
16
16
  * forbidden, bounds }, latency, tokens, cohorts, provenance, fidelity,
17
17
  * sample_size, verdict: 'pass'|'fail'|'inconclusive', cases }
18
18
  *
19
+ * `cohorts` blocks are keyed by `task_type` first (falling back to the
20
+ * case's `cohort` string, then 'unknown') so a mixed-task corpus can never
21
+ * hide a per-task regression inside a pooled aggregate (Simpson's-paradox
22
+ * guard, G5-FR05).
19
23
  * Everything here is offline and deterministic — no model calls, no IO, no
20
24
  * clock, no randomness outside the seeded bootstrap in compareScorecards.
21
25
  */
@@ -129,6 +133,9 @@ function evaluateCase(goldenCase, policyId) {
129
133
  const base = {
130
134
  case_id: replay.case_id,
131
135
  cohort: typeof goldenCase.cohort === 'string' ? goldenCase.cohort : 'default',
136
+ task_type: typeof goldenCase.task_type === 'string' && goldenCase.task_type.length > 0
137
+ ? goldenCase.task_type
138
+ : 'unknown',
132
139
  fidelity: { tag: replay.tag, limitations: replay.limitations },
133
140
  };
134
141
  if (!replay.ok) {
@@ -241,6 +248,18 @@ function cohortVerdict(results, minN) {
241
248
  return 'pass';
242
249
  }
243
250
 
251
+ /**
252
+ * Stratification block key: `task_type` is the first-class axis; cases
253
+ * without one fall back to their declared `cohort` string, then 'unknown'.
254
+ * Mixed-task corpora never pool into a bare global block.
255
+ */
256
+ function blockKeyOf(result) {
257
+ return typeof result.task_type === 'string' && result.task_type !== 'unknown'
258
+ ? result.task_type
259
+ : (typeof result.cohort === 'string' && result.cohort.length > 0
260
+ ? result.cohort
261
+ : 'unknown');
262
+ }
244
263
  /**
245
264
  * @param {object} corpus — `{ cases: GoldenCase[], pre_registered?, metric_version? }`.
246
265
  * @param {string|object} policy — policy id (resolved against
@@ -275,8 +294,9 @@ export function evaluateVariant(corpus, policy) {
275
294
  if (typeof result.metrics.tokens_to_success === 'number') {
276
295
  tokenValues.push(result.metrics.tokens_to_success);
277
296
  }
278
- if (!cohorts.has(result.cohort)) cohorts.set(result.cohort, []);
279
- cohorts.get(result.cohort).push(result);
297
+ const blockKey = blockKeyOf(result);
298
+ if (!cohorts.has(blockKey)) cohorts.set(blockKey, []);
299
+ cohorts.get(blockKey).push(result);
280
300
  }
281
301
 
282
302
  const cohortReport = {};
@@ -396,7 +416,7 @@ export function compareScorecards(a, b, { seed = 42 } = {}) {
396
416
  const pairs = [];
397
417
  for (const [id, ca] of aById) {
398
418
  const cb = bById.get(id);
399
- if (cb && ca.cohort === key && cb.cohort === key && ca.ok && cb.ok) pairs.push([ca, cb]);
419
+ if (cb && blockKeyOf(ca) === key && blockKeyOf(cb) === key && ca.ok && cb.ok) pairs.push([ca, cb]);
400
420
  }
401
421
  const tokens = metricDelta(pairs, 'tokens_to_success', seed);
402
422
  const time = metricDelta(pairs, 'time_to_success_ms', seed + 1);
@@ -1,17 +1,20 @@
1
1
  /**
2
2
  * rollout.js (TASK-015, SPEC §5 DF-FR13 / §13) — per-seam staged rollout.
3
3
  *
4
- * SEAM_MIN_STAGE = { recorder: 'shadow', projector: 'canary', evaluation: 'default' }
5
- * resolveSeamStages(config) → { recorder, projector, evaluation }
4
+ * SEAM_MIN_STAGE = { recorder: 'shadow', analytics: 'shadow',
5
+ * projector: 'canary', evaluation: 'default' }
6
+ * resolveSeamStages(config) → { recorder, analytics, projector, evaluation }
6
7
  * resolveSeamConfig(config, seam) → live config view bound to that seam
7
8
  * killSwitch(config, seam?) → config
8
9
  *
9
10
  * The single global flag `observability.stage` promotes off → shadow →
10
11
  * canary → default; each seam activates at its own minimum stage, so the
11
- * ladder is: shadow records only, canary adds the support projection,
12
- * default adds the evaluation surface. A seam whose effective stage falls
13
- * below its minimum resolves to 'off' — the value existing consumers
14
- * (createRecorder, projectSupport) already treat as their kill switch.
12
+ * ladder is: shadow records only (and unlocks the derived analytics lane —
13
+ * rebuildIndex/renderStoreDigest over retained segments), canary adds the
14
+ * support projection, default adds the evaluation surface. A seam whose
15
+ * effective stage falls below its minimum resolves to 'off' — the value
16
+ * existing consumers (createRecorder, projectSupport) already treat as
17
+ * their kill switch.
15
18
  *
16
19
  * Per-seam overrides live at `observability.seams.<seam>.stage` and are
17
20
  * RESTRICTIVE-ONLY: the effective stage is min(global, override) clamped
@@ -38,10 +41,11 @@ import { resolveStage } from './emit/config.js';
38
41
 
39
42
  const STAGE_RANK = { off: 0, shadow: 1, canary: 2, default: 3 };
40
43
 
41
- export const SEAM_NAMES = Object.freeze(['recorder', 'projector', 'evaluation']);
44
+ export const SEAM_NAMES = Object.freeze(['recorder', 'analytics', 'projector', 'evaluation']);
42
45
 
43
46
  export const SEAM_MIN_STAGE = Object.freeze({
44
47
  recorder: 'shadow',
48
+ analytics: 'shadow',
45
49
  projector: 'canary',
46
50
  evaluation: 'default',
47
51
  });