@ngockhoale/ukit 3.0.7 → 3.0.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -1
- package/manifests/documentation.yaml +11 -0
- package/package.json +1 -1
- package/scripts/audit/decision-coverage.mjs +29 -2
- package/scripts/bench/data-foundation.mjs +52 -3
- package/scripts/bench/decision-runtime-baseline.mjs +427 -0
- package/scripts/bench/decision-runtime-metrics.mjs +67 -0
- package/scripts/bench/decision-runtime-variant.mjs +626 -0
- package/scripts/bench/memory-ablation.mjs +495 -0
- package/scripts/bench/memory-baseline.mjs +596 -0
- package/scripts/bench/memory-bench.mjs +661 -0
- package/scripts/bench/memory-canary.mjs +321 -0
- package/scripts/bench/memory-corpus.mjs +354 -0
- package/scripts/bench/memory-gate.mjs +389 -0
- package/scripts/bench/memory-metrics.mjs +179 -0
- package/scripts/bench/parallel-agents.mjs +33 -11
- package/scripts/bench/recorder-overhead.mjs +204 -0
- package/scripts/bench/sqlite-spike.mjs +451 -0
- package/scripts/measure-decision-gateway.mjs +306 -0
- package/scripts/perf/audit-perf.mjs +35 -17
- package/src/bug/triageBug.js +4 -3
- package/src/cli/commands/memory.js +357 -63
- package/src/context/detectProjectContext.js +11 -1
- package/src/core/agentRuntime/adapters.js +254 -0
- package/src/core/agentRuntime/artifacts.js +192 -0
- package/src/core/agentRuntime/completionGate.js +176 -0
- package/src/core/agentRuntime/context.js +149 -0
- package/src/core/agentRuntime/contract.js +247 -0
- package/src/core/agentRuntime/diagnostics.js +244 -0
- package/src/core/agentRuntime/evaluation.js +163 -0
- package/src/core/agentRuntime/eventStore.js +404 -0
- package/src/core/agentRuntime/liveness.js +60 -0
- package/src/core/agentRuntime/planCompiler.js +322 -0
- package/src/core/agentRuntime/promotion.js +53 -0
- package/src/core/agentRuntime/qualityComparison.js +112 -0
- package/src/core/agentRuntime/recovery.js +266 -0
- package/src/core/agentRuntime/resourcePolicy.js +78 -0
- package/src/core/agentRuntime/runtimeSupport.js +237 -0
- package/src/core/agentRuntime/supervisor.js +565 -0
- package/src/core/agentRuntime/vmEngine.js +621 -0
- package/src/core/codeintel/analogy.js +3 -2
- package/src/core/experiments/dynamicWorkflow.js +17 -2
- package/src/core/fileOps.js +21 -3
- package/src/core/memory/deltaOverlays.js +75 -30
- package/src/core/memory/learningCandidates.js +93 -48
- package/src/core/memory/memoryFlags.js +83 -0
- package/src/core/memory/memoryFreshness.js +190 -0
- package/src/core/memory/memoryHit.js +144 -0
- package/src/core/memory/migrate.js +69 -189
- package/src/core/memory/migrateMapping.js +232 -0
- package/src/core/memory/mutateMemory.js +323 -0
- package/src/core/memory/policy.js +96 -0
- package/src/core/memory/projectIdentity.js +266 -0
- package/src/core/memory/recordIndex.js +178 -0
- package/src/core/memory/recordStore.js +133 -20
- package/src/core/memory/records.js +144 -6
- package/src/core/memory/retrieval.js +259 -125
- package/src/core/memory/store.js +16 -5
- package/src/core/memory/storeBackup.js +226 -0
- package/src/core/memory/storeV2.js +63 -26
- package/src/core/memory/storeV2Loader.js +30 -12
- package/src/core/memory/userMemory.js +38 -20
- package/src/core/memory/writeClassification.js +161 -0
- package/src/core/memory/writeGuard.js +129 -0
- package/src/core/observability/adapters/hookTelemetryAdapter.js +90 -0
- package/src/core/observability/analytics/cohorts.js +148 -0
- package/src/core/observability/analytics/storeDigest.js +163 -0
- package/src/core/observability/evaluation/experimentPlan.js +95 -0
- package/src/core/observability/evaluation/findings.js +99 -0
- package/src/core/observability/evaluation/optimizationKnowledge.js +10 -1
- package/src/core/observability/evaluation/perturbation.js +273 -0
- package/src/core/observability/evaluation/replay.js +7 -1
- package/src/core/observability/evaluation/scorecard.js +23 -3
- package/src/core/observability/rollout.js +11 -7
- package/src/core/observability/schema/compatibility.js +135 -0
- package/src/core/observability/schema/registry.js +99 -0
- package/src/core/observability/schema/validate.js +7 -0
- package/src/core/observability/support/import.js +53 -9
- package/src/core/observability/support/paths.js +13 -3
- package/src/core/observability/support/projector.js +148 -12
- package/src/core/output/index.js +12 -2
- package/src/core/runtimeConfig.js +83 -0
- package/src/core/runtimePaths.js +3 -0
- package/src/core/sensitiveValueScanner.js +40 -0
- package/src/core/token/index.js +40 -3
- package/src/decision/client.js +37 -13
- package/src/decision/protocol.js +1 -1
- package/src/decision/registry.js +5 -3
- package/src/decision/runtimeDecide.js +242 -0
- package/src/decision/runtimeFilter.js +150 -0
- package/src/decision/runtimeScheduler.js +239 -0
- package/src/index/buildIndex.js +13 -12
- package/src/index/queryIndex.js +35 -14
- package/src/index/relatedTests.js +50 -8
- package/src/index/resolveContext.js +9 -4
- package/src/manifest/selectItems.js +7 -3
- package/src/render/instructionRenderer.js +17 -5
- package/template_project/.claude/ukit/index/lib/index-core.mjs +94 -39
- package/template_project/.claude/ukit/index/route-task.mjs +121 -19
- package/template_project/.claude/ukit/index/unic-decision.mjs +28 -13
- package/template_project/.claude/ukit/runtime/memory-flags.mjs +51 -0
- package/template_project/.claude/ukit/runtime/memory-freshness.mjs +155 -0
- package/template_project/.claude/ukit/runtime/memory-policy.mjs +286 -0
- package/template_project/.claude/ukit/runtime/output-compression.mjs +3 -0
- package/template_project/.claude/ukit/runtime/reinject-context.mjs +145 -14
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* experimentPlan — pure staged controlled-experiment planner (GAP-c).
|
|
3
|
+
*
|
|
4
|
+
* Maps a KB optimization entry + a `computeGateComparison` GateComparison to
|
|
5
|
+
* staged rollout ACTIONS over the seam ladder. The planner EMITS intent only:
|
|
6
|
+
* it never mutates config, never calls killSwitch/resolveSeamConfig, and never
|
|
7
|
+
* touches fs. Execution (if any) is a separate operator-owned act.
|
|
8
|
+
*
|
|
9
|
+
* Rules:
|
|
10
|
+
* - gate must be a real GateComparison (verdict in pass|non-inferior|unknown);
|
|
11
|
+
* anything else → TypeError (a fabricated gate can never plan a promotion).
|
|
12
|
+
* - abort flag (operator rollback_criterion met) → kill 'abort-requested'.
|
|
13
|
+
* - gate.forbiddenFailures > 0 → kill 'forbidden-failures' prepended.
|
|
14
|
+
* - verdict !== 'pass' → hold (reason = verdict). 'non-inferior' means the
|
|
15
|
+
* variant FAILED the pre-registered gate.
|
|
16
|
+
* - verdict 'pass' + non-empty controlled_rollout.evidence_refs → one promote
|
|
17
|
+
* action per seam in stage-rank order (shadow → canary → default), each
|
|
18
|
+
* carrying requires_evidence_refs:true. Missing refs → hold
|
|
19
|
+
* 'missing-controlled-rollout-evidence': promotion never fires on the
|
|
20
|
+
* quality signal alone.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
import { SEAM_MIN_STAGE } from '../rollout.js';
|
|
24
|
+
|
|
25
|
+
export const ROLLOUT_ACTIONS = Object.freeze(['promote', 'hold', 'kill']);
|
|
26
|
+
|
|
27
|
+
const GATE_VERDICTS = new Set(['pass', 'non-inferior', 'unknown']);
|
|
28
|
+
|
|
29
|
+
const STAGE_RANK = { shadow: 1, canary: 2, default: 3 };
|
|
30
|
+
|
|
31
|
+
function isPlainObject(value) {
|
|
32
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function isNonEmptyStringList(value) {
|
|
36
|
+
return (
|
|
37
|
+
Array.isArray(value) &&
|
|
38
|
+
value.length > 0 &&
|
|
39
|
+
value.every((v) => typeof v === 'string' && v.length > 0)
|
|
40
|
+
);
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* @param {{optimization?: object, gate?: object, abort?: boolean}} input
|
|
45
|
+
* @returns {{actions: object[]}} frozen action list
|
|
46
|
+
*/
|
|
47
|
+
export function planControlledRollout(input = {}) {
|
|
48
|
+
const { optimization, gate, abort } = isPlainObject(input) ? input : {};
|
|
49
|
+
|
|
50
|
+
if (!isPlainObject(gate) || !GATE_VERDICTS.has(gate.verdict)) {
|
|
51
|
+
throw new TypeError(
|
|
52
|
+
'experimentPlan: gate must be a GateComparison with verdict in pass|non-inferior|unknown',
|
|
53
|
+
);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
const actions = [];
|
|
57
|
+
|
|
58
|
+
if (abort === true) {
|
|
59
|
+
actions.push({ type: 'kill', reason: 'abort-requested' });
|
|
60
|
+
}
|
|
61
|
+
if (gate.forbiddenFailures > 0) {
|
|
62
|
+
actions.push({ type: 'kill', reason: 'forbidden-failures' });
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
if (gate.verdict !== 'pass') {
|
|
66
|
+
actions.push({ type: 'hold', reason: gate.verdict });
|
|
67
|
+
return { actions: Object.freeze(actions) };
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
if (actions.length > 0) {
|
|
71
|
+
// pass + forbiddenFailures is unreachable from real computeGateComparison
|
|
72
|
+
// output (forbidden → non-inferior hard gate); defend the shape anyway.
|
|
73
|
+
return { actions: Object.freeze(actions) };
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
const rollout = isPlainObject(optimization) ? optimization.controlled_rollout : undefined;
|
|
77
|
+
const evidenceRefs = isPlainObject(rollout) ? rollout.evidence_refs : undefined;
|
|
78
|
+
if (!isNonEmptyStringList(evidenceRefs)) {
|
|
79
|
+
return {
|
|
80
|
+
actions: Object.freeze([{ type: 'hold', reason: 'missing-controlled-rollout-evidence' }]),
|
|
81
|
+
};
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
// Order from the shared seam contract (read-only): each seam promotes at the
|
|
85
|
+
// stage it unlocks, lowest stage rank first.
|
|
86
|
+
const seams = Object.entries(SEAM_MIN_STAGE)
|
|
87
|
+
.map(([seam, stage]) => ({ seam, stage }))
|
|
88
|
+
.sort((a, b) => STAGE_RANK[a.stage] - STAGE_RANK[b.stage] || a.seam.localeCompare(b.seam));
|
|
89
|
+
|
|
90
|
+
for (const { seam, stage } of seams) {
|
|
91
|
+
actions.push({ type: 'promote', seam, stage, requires_evidence_refs: true });
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
return { actions: Object.freeze(actions) };
|
|
95
|
+
}
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* findings.js (TASK-002, SPEC §2 G6-FR04) — evaluator findings validator.
|
|
3
|
+
*
|
|
4
|
+
* validateEvaluatorFindings(payload) → { findings[], rejected[] }
|
|
5
|
+
*
|
|
6
|
+
* Advisory-only boundary between the optional AI evaluator and the rest
|
|
7
|
+
* of the pipeline: an evaluator's opinion can never be promoted to fact
|
|
8
|
+
* or policy. Every accepted finding is re-shaped by construction to
|
|
9
|
+
* { claim, evidence_refs, confidence, alternatives?, decision:
|
|
10
|
+
* 'unsupported', applied: false }
|
|
11
|
+
* and frozen — caller-injected `decision`/`applied`/`promoted` fields are
|
|
12
|
+
* never propagated.
|
|
13
|
+
*
|
|
14
|
+
* Evidence gate: a candidate without a non-empty `claim` (non-empty
|
|
15
|
+
* string) or a non-empty `evidence_refs` (all-string array) is moved to
|
|
16
|
+
* `rejected[]` with `{ candidate_index, reason }` — never silently
|
|
17
|
+
* dropped. Out-of-enum confidence coerces to 'unknown'; `alternatives`
|
|
18
|
+
* carries over as advisory hints (strings only). Malformed payloads
|
|
19
|
+
* (non-object, non-array, or a non-array `findings` field) reject the
|
|
20
|
+
* whole payload with `candidate_index: null`.
|
|
21
|
+
*
|
|
22
|
+
* Pure module: no fs, no clock, no model calls; never throws on data.
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
export const FINDING_CONFIDENCE = Object.freeze([
|
|
26
|
+
'high',
|
|
27
|
+
'medium',
|
|
28
|
+
'low',
|
|
29
|
+
'unknown',
|
|
30
|
+
]);
|
|
31
|
+
|
|
32
|
+
function isPlainObject(value) {
|
|
33
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
function isNonEmptyString(value) {
|
|
37
|
+
return typeof value === 'string' && value.length > 0;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function isStringList(value) {
|
|
41
|
+
return Array.isArray(value) && value.length > 0 && value.every(isNonEmptyString);
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
function candidates(payload) {
|
|
45
|
+
if (Array.isArray(payload)) return payload;
|
|
46
|
+
if (isPlainObject(payload) && Array.isArray(payload.findings)) {
|
|
47
|
+
return payload.findings;
|
|
48
|
+
}
|
|
49
|
+
return null;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Validate evaluator-returned findings. `payload` may be either
|
|
54
|
+
* `{ findings: [...] }` or a bare findings array — both accept the same
|
|
55
|
+
* candidate shape.
|
|
56
|
+
*/
|
|
57
|
+
export function validateEvaluatorFindings(payload) {
|
|
58
|
+
const list = candidates(payload);
|
|
59
|
+
if (list === null) {
|
|
60
|
+
return {
|
|
61
|
+
findings: [],
|
|
62
|
+
rejected: [{ candidate_index: null, reason: 'malformed-payload' }],
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const findings = [];
|
|
67
|
+
const rejected = [];
|
|
68
|
+
|
|
69
|
+
for (let i = 0; i < list.length; i += 1) {
|
|
70
|
+
const candidate = isPlainObject(list[i]) ? list[i] : {};
|
|
71
|
+
if (!isNonEmptyString(candidate.claim)) {
|
|
72
|
+
rejected.push({ candidate_index: i, reason: 'missing-claim' });
|
|
73
|
+
continue;
|
|
74
|
+
}
|
|
75
|
+
if (!isStringList(candidate.evidence_refs)) {
|
|
76
|
+
rejected.push({ candidate_index: i, reason: 'missing-evidence-refs' });
|
|
77
|
+
continue;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
const finding = {
|
|
81
|
+
claim: candidate.claim,
|
|
82
|
+
evidence_refs: [...candidate.evidence_refs],
|
|
83
|
+
confidence: FINDING_CONFIDENCE.includes(candidate.confidence)
|
|
84
|
+
? candidate.confidence
|
|
85
|
+
: 'unknown',
|
|
86
|
+
};
|
|
87
|
+
if (Array.isArray(candidate.alternatives)) {
|
|
88
|
+
const alternatives = candidate.alternatives.filter(isNonEmptyString);
|
|
89
|
+
if (alternatives.length > 0) finding.alternatives = alternatives;
|
|
90
|
+
}
|
|
91
|
+
// Advisory-only by construction: evaluator decision/applied fields
|
|
92
|
+
// are never read; the output shape is fixed.
|
|
93
|
+
finding.decision = 'unsupported';
|
|
94
|
+
finding.applied = false;
|
|
95
|
+
findings.push(Object.freeze(finding));
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
return { findings, rejected };
|
|
99
|
+
}
|
|
@@ -18,7 +18,8 @@
|
|
|
18
18
|
*
|
|
19
19
|
* Invariants (SPEC §12, DF-FR12):
|
|
20
20
|
* - `decision: 'promoted'` or `applied: true` REQUIRE
|
|
21
|
-
* `controlled_rollout.evidence_refs`
|
|
21
|
+
* `controlled_rollout.evidence_refs` AND a non-empty
|
|
22
|
+
* `alternative_explanations` list — no promotion from token
|
|
22
23
|
* reduction alone; violating entries throw TypeError.
|
|
23
24
|
* - `decision: 'unsupported'` records can never carry `applied: true`.
|
|
24
25
|
* - This module writes ONLY inside `root` — no runtime config, no
|
|
@@ -110,6 +111,13 @@ export function createOptimizationStore({ root, clock } = {}) {
|
|
|
110
111
|
throw new TypeError(
|
|
111
112
|
'recordOptimization: decision promoted / applied requires controlled_rollout.evidence_refs',
|
|
112
113
|
);
|
|
114
|
+
} else if (decision === 'promoted' || applied) {
|
|
115
|
+
const alternatives = stringList(entry.alternative_explanations);
|
|
116
|
+
if (alternatives.length === 0) {
|
|
117
|
+
throw new TypeError(
|
|
118
|
+
'recordOptimization: decision promoted / applied requires alternative_explanations',
|
|
119
|
+
);
|
|
120
|
+
}
|
|
113
121
|
}
|
|
114
122
|
|
|
115
123
|
const core = {
|
|
@@ -127,6 +135,7 @@ export function createOptimizationStore({ root, clock } = {}) {
|
|
|
127
135
|
rollback_criterion:
|
|
128
136
|
typeof entry.rollback_criterion === 'string' ? entry.rollback_criterion : null,
|
|
129
137
|
supersedes: stringList(entry.supersedes),
|
|
138
|
+
alternative_explanations: stringList(entry.alternative_explanations),
|
|
130
139
|
created_at: now(),
|
|
131
140
|
};
|
|
132
141
|
|
|
@@ -0,0 +1,273 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* perturbation.js (TASK-002, SPEC §2 G5-FR06/G5-FR07) — synthetic mutation harness.
|
|
3
|
+
*
|
|
4
|
+
* mutateSummaries(summaries, { seed, plan? }) → { summaries, manifest }
|
|
5
|
+
* detectPerturbations(mutatedSummaries, baselineSummaries, manifest)
|
|
6
|
+
* → { injected, detected, missed[], details[] }
|
|
7
|
+
*
|
|
8
|
+
* Pure module: no fs, no clock, no unseeded randomness (mulberry32, same
|
|
9
|
+
* construction as the bench corpus generator). Mutations edit TraceSummary
|
|
10
|
+
* structures so the downstream pipeline sees a genuinely different trace;
|
|
11
|
+
* detection is measured through the REAL computeFingerprint /
|
|
12
|
+
* detectOpportunities path — never from the manifest alone.
|
|
13
|
+
*
|
|
14
|
+
* Mutation kinds (MUTATION_KINDS):
|
|
15
|
+
* drop_record — removes a span and raises drops.dropped_count>0
|
|
16
|
+
* retry_inflation — adds an attempt_index>1 model span and raises retries
|
|
17
|
+
* cache_miss_spike — adds cache misses (cache.observed, hit_rate recomputed)
|
|
18
|
+
* orphan_span — rewrites parent_span_id to an absent id
|
|
19
|
+
* telemetry_incomplete — flips telemetry_complete=false
|
|
20
|
+
*
|
|
21
|
+
* Detection rule (per manifest entry): the mutated trace is detected when its
|
|
22
|
+
* fingerprint kind/count diverges from its own baseline, OR — once the trace
|
|
23
|
+
* is proven changed (hash divergence) — its opportunity cluster reports
|
|
24
|
+
* `inconclusive` (the honesty path telemetry_incomplete relies on) or a
|
|
25
|
+
* non-clean pattern. An unmutated trace claimed in the manifest cannot be
|
|
26
|
+
* detected: hash-identical traces have nothing to report.
|
|
27
|
+
*/
|
|
28
|
+
|
|
29
|
+
import { computeFingerprint } from '../analytics/fingerprints.js';
|
|
30
|
+
import { detectOpportunities } from '../analytics/opportunities.js';
|
|
31
|
+
|
|
32
|
+
export const MUTATION_KINDS = Object.freeze([
|
|
33
|
+
'drop_record',
|
|
34
|
+
'retry_inflation',
|
|
35
|
+
'cache_miss_spike',
|
|
36
|
+
'orphan_span',
|
|
37
|
+
'telemetry_incomplete',
|
|
38
|
+
]);
|
|
39
|
+
|
|
40
|
+
// Deterministic PRNG (mulberry32) — identical construction to the bench corpus.
|
|
41
|
+
function mulberry32(seed) {
|
|
42
|
+
let a = seed >>> 0;
|
|
43
|
+
return function next() {
|
|
44
|
+
a |= 0;
|
|
45
|
+
a = (a + 0x6d2b79f5) | 0;
|
|
46
|
+
let t = Math.imul(a ^ (a >>> 15), 1 | a);
|
|
47
|
+
t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t;
|
|
48
|
+
return ((t ^ (t >>> 14)) >>> 0) / 4294967296;
|
|
49
|
+
};
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
function isPlainObject(value) {
|
|
53
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
function num(value) {
|
|
57
|
+
return typeof value === 'number' && Number.isFinite(value) ? value : 0;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* Candidate indices for a kind — a mutation must produce a genuinely
|
|
62
|
+
* different trace, so kinds with preconditions filter to eligible traces.
|
|
63
|
+
* Prefers traces not already mutated by this call.
|
|
64
|
+
*/
|
|
65
|
+
function candidateIndices(mutated, kind, used) {
|
|
66
|
+
const eligible = mutated
|
|
67
|
+
.map((s, idx) => ({ s, idx }))
|
|
68
|
+
.filter(({ s }) => {
|
|
69
|
+
if (!isPlainObject(s) || !Array.isArray(s.spans) || s.spans.length === 0) return false;
|
|
70
|
+
if (kind === 'telemetry_incomplete') return s.telemetry_complete === true;
|
|
71
|
+
return true;
|
|
72
|
+
})
|
|
73
|
+
.map(({ idx }) => idx);
|
|
74
|
+
const fresh = eligible.filter((idx) => !used.has(idx));
|
|
75
|
+
return fresh.length > 0 ? fresh : eligible;
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
function ensureCounters(summary) {
|
|
79
|
+
if (!isPlainObject(summary.retries)) summary.retries = {};
|
|
80
|
+
if (!isPlainObject(summary.drops)) summary.drops = {};
|
|
81
|
+
if (!isPlainObject(summary.drops.by_reason)) summary.drops.by_reason = {};
|
|
82
|
+
if (!isPlainObject(summary.cache)) summary.cache = {};
|
|
83
|
+
if (!isPlainObject(summary.coverage)) summary.coverage = {};
|
|
84
|
+
}
|
|
85
|
+
const MUTATORS = {
|
|
86
|
+
drop_record(summary) {
|
|
87
|
+
const idx = summary.spans.findIndex(
|
|
88
|
+
(s) => typeof s.parent_span_id === 'string' && s.parent_span_id.length > 0,
|
|
89
|
+
);
|
|
90
|
+
const i = idx >= 0 ? idx : 0;
|
|
91
|
+
const removed = summary.spans.splice(i, 1)[0] || null;
|
|
92
|
+
summary.drops.events = num(summary.drops.events) + 1;
|
|
93
|
+
summary.drops.dropped_count = num(summary.drops.dropped_count) + 1;
|
|
94
|
+
summary.drops.by_reason.synthetic_drop = num(summary.drops.by_reason.synthetic_drop) + 1;
|
|
95
|
+
summary.telemetry_complete = false;
|
|
96
|
+
summary.coverage.spans = summary.spans.length;
|
|
97
|
+
return { removed_span_id: removed ? removed.span_id : null };
|
|
98
|
+
},
|
|
99
|
+
|
|
100
|
+
retry_inflation(summary) {
|
|
101
|
+
const model =
|
|
102
|
+
summary.spans.find((s) => s.family === 'model') ||
|
|
103
|
+
summary.spans[summary.spans.length - 1] ||
|
|
104
|
+
null;
|
|
105
|
+
const maxAttempt = summary.spans.reduce(
|
|
106
|
+
(m, s) => (Number.isInteger(s.attempt_index) ? Math.max(m, s.attempt_index) : m),
|
|
107
|
+
0,
|
|
108
|
+
);
|
|
109
|
+
const attemptIndex = Math.max(2, maxAttempt + 1);
|
|
110
|
+
const clone = {
|
|
111
|
+
...(model || {
|
|
112
|
+
parent_span_id: null,
|
|
113
|
+
family: 'model',
|
|
114
|
+
operation: null,
|
|
115
|
+
status: 'completed',
|
|
116
|
+
duration_ms: 0,
|
|
117
|
+
}),
|
|
118
|
+
span_id: `${model ? model.span_id : 'span'}-retry${attemptIndex}`,
|
|
119
|
+
attempt_index: attemptIndex,
|
|
120
|
+
status: 'completed',
|
|
121
|
+
start_record_id: `rec-${summary.trace_id}-retry${attemptIndex}-start`,
|
|
122
|
+
end_record_id: `rec-${summary.trace_id}-retry${attemptIndex}-end`,
|
|
123
|
+
};
|
|
124
|
+
summary.spans.push(clone);
|
|
125
|
+
summary.retries.retries = num(summary.retries.retries) + 1;
|
|
126
|
+
summary.retries.model_attempts = num(summary.retries.model_attempts) + 1;
|
|
127
|
+
summary.coverage.spans = summary.spans.length;
|
|
128
|
+
return { added_attempt_span: clone.span_id, attempt_index: attemptIndex };
|
|
129
|
+
},
|
|
130
|
+
|
|
131
|
+
cache_miss_spike(summary, rand) {
|
|
132
|
+
const added = 2 + Math.floor(rand() * 3); // 2..4 misses
|
|
133
|
+
summary.cache.observed = true;
|
|
134
|
+
summary.cache.hits = num(summary.cache.hits);
|
|
135
|
+
summary.cache.misses = num(summary.cache.misses) + added;
|
|
136
|
+
const lookups = summary.cache.hits + summary.cache.misses;
|
|
137
|
+
summary.cache.hit_rate = lookups > 0 ? summary.cache.hits / lookups : null;
|
|
138
|
+
return { misses_added: added };
|
|
139
|
+
},
|
|
140
|
+
|
|
141
|
+
orphan_span(summary) {
|
|
142
|
+
const idx = summary.spans.findIndex(
|
|
143
|
+
(s) => typeof s.parent_span_id === 'string' && s.parent_span_id.length > 0,
|
|
144
|
+
);
|
|
145
|
+
const i = idx >= 0 ? idx : 0;
|
|
146
|
+
const target = summary.spans[i];
|
|
147
|
+
const previous = target.parent_span_id ?? null;
|
|
148
|
+
target.parent_span_id = `absent-${target.span_id}`;
|
|
149
|
+
summary.coverage.orphan_spans = num(summary.coverage.orphan_spans) + 1;
|
|
150
|
+
return { span_id: target.span_id, previous_parent: previous };
|
|
151
|
+
},
|
|
152
|
+
|
|
153
|
+
telemetry_incomplete(summary) {
|
|
154
|
+
const previous = summary.telemetry_complete;
|
|
155
|
+
summary.telemetry_complete = false;
|
|
156
|
+
return { previous_telemetry_complete: previous };
|
|
157
|
+
},
|
|
158
|
+
};
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* @param {Array<object>} summaries — baseline TraceSummary objects (not mutated).
|
|
162
|
+
* @param {{ seed?: number, plan?: Array<{kind:string}> }} [options]
|
|
163
|
+
* plan defaults to one mutation per MUTATION_KINDS entry.
|
|
164
|
+
* @returns {{ summaries: Array<object>, manifest: Array<object> }}
|
|
165
|
+
* manifest entries: { mutation_id, trace_id, kind, detail }.
|
|
166
|
+
*/
|
|
167
|
+
export function mutateSummaries(summaries, { seed = 42, plan } = {}) {
|
|
168
|
+
if (!Array.isArray(summaries)) {
|
|
169
|
+
throw new TypeError('mutateSummaries: summaries must be an array of TraceSummary');
|
|
170
|
+
}
|
|
171
|
+
const effectivePlan = Array.isArray(plan) ? plan : MUTATION_KINDS.map((kind) => ({ kind }));
|
|
172
|
+
for (const entry of effectivePlan) {
|
|
173
|
+
if (!isPlainObject(entry) || !MUTATION_KINDS.includes(entry.kind)) {
|
|
174
|
+
throw new TypeError(
|
|
175
|
+
`mutateSummaries: plan entries need kind ∈ {${MUTATION_KINDS.join(', ')}}`,
|
|
176
|
+
);
|
|
177
|
+
}
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
const rand = mulberry32(seed);
|
|
181
|
+
const mutated = summaries.map((s) => JSON.parse(JSON.stringify(s)));
|
|
182
|
+
const manifest = [];
|
|
183
|
+
const used = new Set();
|
|
184
|
+
|
|
185
|
+
effectivePlan.forEach((entry, i) => {
|
|
186
|
+
const pickFrom = candidateIndices(mutated, entry.kind, used);
|
|
187
|
+
if (pickFrom.length === 0) return;
|
|
188
|
+
const targetIdx = pickFrom[Math.floor(rand() * pickFrom.length)];
|
|
189
|
+
used.add(targetIdx);
|
|
190
|
+
const summary = mutated[targetIdx];
|
|
191
|
+
ensureCounters(summary);
|
|
192
|
+
const detail = MUTATORS[entry.kind](summary, rand);
|
|
193
|
+
manifest.push({
|
|
194
|
+
mutation_id: `mut-${String(i + 1).padStart(3, '0')}`,
|
|
195
|
+
trace_id: typeof summary.trace_id === 'string' ? summary.trace_id : `idx-${targetIdx}`,
|
|
196
|
+
kind: entry.kind,
|
|
197
|
+
detail,
|
|
198
|
+
});
|
|
199
|
+
});
|
|
200
|
+
|
|
201
|
+
return { summaries: mutated, manifest };
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
/**
|
|
205
|
+
* Measure detection through the real pipeline: computeFingerprint for
|
|
206
|
+
* per-trace divergence vs baseline and detectOpportunities for cluster
|
|
207
|
+
* status over the mutated set.
|
|
208
|
+
*
|
|
209
|
+
* @param {Array<object>} mutatedSummaries
|
|
210
|
+
* @param {Array<object>} baselineSummaries
|
|
211
|
+
* @param {Array<object>} manifest — entries from mutateSummaries.
|
|
212
|
+
* @returns {{ injected:number, detected:number, missed:Array<object>, details:Array<object> }}
|
|
213
|
+
*/
|
|
214
|
+
export function detectPerturbations(mutatedSummaries, baselineSummaries, manifest) {
|
|
215
|
+
if (!Array.isArray(mutatedSummaries) || !Array.isArray(baselineSummaries) || !Array.isArray(manifest)) {
|
|
216
|
+
throw new TypeError('detectPerturbations: summaries and manifest must be arrays');
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
const baselineByTrace = new Map();
|
|
220
|
+
for (const s of baselineSummaries) {
|
|
221
|
+
if (isPlainObject(s) && typeof s.trace_id === 'string') baselineByTrace.set(s.trace_id, s);
|
|
222
|
+
}
|
|
223
|
+
const mutatedByTrace = new Map();
|
|
224
|
+
for (const s of mutatedSummaries) {
|
|
225
|
+
if (isPlainObject(s) && typeof s.trace_id === 'string') mutatedByTrace.set(s.trace_id, s);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
// Real pipeline — never read the manifest as evidence.
|
|
229
|
+
const opportunities = detectOpportunities(mutatedSummaries);
|
|
230
|
+
const oppByTrace = new Map();
|
|
231
|
+
for (const opp of opportunities) {
|
|
232
|
+
for (const ref of opp.trace_refs || []) {
|
|
233
|
+
if (!oppByTrace.has(ref)) oppByTrace.set(ref, opp);
|
|
234
|
+
}
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
const details = [];
|
|
238
|
+
for (const entry of manifest) {
|
|
239
|
+
const mutated = mutatedByTrace.get(entry.trace_id);
|
|
240
|
+
const baseline = baselineByTrace.get(entry.trace_id);
|
|
241
|
+
const via = [];
|
|
242
|
+
let opportunity = null;
|
|
243
|
+
if (mutated && baseline) {
|
|
244
|
+
const fm = computeFingerprint(mutated);
|
|
245
|
+
const fb = computeFingerprint(baseline);
|
|
246
|
+
const changed = fm.hash !== fb.hash;
|
|
247
|
+
if (fm.kind !== fb.kind || fm.count !== fb.count) via.push('fingerprint');
|
|
248
|
+
opportunity = oppByTrace.get(entry.trace_id) || null;
|
|
249
|
+
if (changed && opportunity) {
|
|
250
|
+
if (opportunity.status === 'inconclusive') via.push('inconclusive');
|
|
251
|
+
else if (opportunity.fingerprint && opportunity.fingerprint.kind !== 'clean') {
|
|
252
|
+
via.push('opportunity');
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
details.push({
|
|
257
|
+
mutation_id: entry.mutation_id,
|
|
258
|
+
trace_id: entry.trace_id,
|
|
259
|
+
kind: entry.kind,
|
|
260
|
+
detected: via.length > 0,
|
|
261
|
+
via,
|
|
262
|
+
opportunity,
|
|
263
|
+
});
|
|
264
|
+
}
|
|
265
|
+
|
|
266
|
+
const missed = details.filter((d) => !d.detected);
|
|
267
|
+
return {
|
|
268
|
+
injected: manifest.length,
|
|
269
|
+
detected: details.length - missed.length,
|
|
270
|
+
missed,
|
|
271
|
+
details,
|
|
272
|
+
};
|
|
273
|
+
}
|
|
@@ -25,6 +25,8 @@
|
|
|
25
25
|
* deterministically; explicit canonical values always win.
|
|
26
26
|
*/
|
|
27
27
|
|
|
28
|
+
import { SEMANTIC_REGISTRY } from '../schema/registry.js';
|
|
29
|
+
|
|
28
30
|
export const REPLAY_TAGS = Object.freeze(['POLICY', 'CONTEXT', 'SIMULATED']);
|
|
29
31
|
|
|
30
32
|
export const REPLAY_LIMITATIONS = Object.freeze({
|
|
@@ -92,7 +94,11 @@ function materializeRecord(raw, ctx, index) {
|
|
|
92
94
|
}
|
|
93
95
|
if (typeof record.monotonic_ns !== 'number') record.monotonic_ns = dtMs * 1e6;
|
|
94
96
|
if (typeof record.importance !== 'string') record.importance = 'normal';
|
|
95
|
-
if (typeof record.privacy_class !== 'string')
|
|
97
|
+
if (typeof record.privacy_class !== 'string') {
|
|
98
|
+
// Default to the semantic registry's privacy floor so materialized
|
|
99
|
+
// records satisfy the contract; unregistered names fall back to public.
|
|
100
|
+
record.privacy_class = SEMANTIC_REGISTRY[record.semantic_name]?.privacy_class ?? 'public';
|
|
101
|
+
}
|
|
96
102
|
if (!isPlainObject(record.payload)) record.payload = {};
|
|
97
103
|
return record;
|
|
98
104
|
}
|
|
@@ -16,6 +16,10 @@
|
|
|
16
16
|
* forbidden, bounds }, latency, tokens, cohorts, provenance, fidelity,
|
|
17
17
|
* sample_size, verdict: 'pass'|'fail'|'inconclusive', cases }
|
|
18
18
|
*
|
|
19
|
+
* `cohorts` blocks are keyed by `task_type` first (falling back to the
|
|
20
|
+
* case's `cohort` string, then 'unknown') so a mixed-task corpus can never
|
|
21
|
+
* hide a per-task regression inside a pooled aggregate (Simpson's-paradox
|
|
22
|
+
* guard, G5-FR05).
|
|
19
23
|
* Everything here is offline and deterministic — no model calls, no IO, no
|
|
20
24
|
* clock, no randomness outside the seeded bootstrap in compareScorecards.
|
|
21
25
|
*/
|
|
@@ -129,6 +133,9 @@ function evaluateCase(goldenCase, policyId) {
|
|
|
129
133
|
const base = {
|
|
130
134
|
case_id: replay.case_id,
|
|
131
135
|
cohort: typeof goldenCase.cohort === 'string' ? goldenCase.cohort : 'default',
|
|
136
|
+
task_type: typeof goldenCase.task_type === 'string' && goldenCase.task_type.length > 0
|
|
137
|
+
? goldenCase.task_type
|
|
138
|
+
: 'unknown',
|
|
132
139
|
fidelity: { tag: replay.tag, limitations: replay.limitations },
|
|
133
140
|
};
|
|
134
141
|
if (!replay.ok) {
|
|
@@ -241,6 +248,18 @@ function cohortVerdict(results, minN) {
|
|
|
241
248
|
return 'pass';
|
|
242
249
|
}
|
|
243
250
|
|
|
251
|
+
/**
|
|
252
|
+
* Stratification block key: `task_type` is the first-class axis; cases
|
|
253
|
+
* without one fall back to their declared `cohort` string, then 'unknown'.
|
|
254
|
+
* Mixed-task corpora never pool into a bare global block.
|
|
255
|
+
*/
|
|
256
|
+
function blockKeyOf(result) {
|
|
257
|
+
return typeof result.task_type === 'string' && result.task_type !== 'unknown'
|
|
258
|
+
? result.task_type
|
|
259
|
+
: (typeof result.cohort === 'string' && result.cohort.length > 0
|
|
260
|
+
? result.cohort
|
|
261
|
+
: 'unknown');
|
|
262
|
+
}
|
|
244
263
|
/**
|
|
245
264
|
* @param {object} corpus — `{ cases: GoldenCase[], pre_registered?, metric_version? }`.
|
|
246
265
|
* @param {string|object} policy — policy id (resolved against
|
|
@@ -275,8 +294,9 @@ export function evaluateVariant(corpus, policy) {
|
|
|
275
294
|
if (typeof result.metrics.tokens_to_success === 'number') {
|
|
276
295
|
tokenValues.push(result.metrics.tokens_to_success);
|
|
277
296
|
}
|
|
278
|
-
|
|
279
|
-
cohorts.
|
|
297
|
+
const blockKey = blockKeyOf(result);
|
|
298
|
+
if (!cohorts.has(blockKey)) cohorts.set(blockKey, []);
|
|
299
|
+
cohorts.get(blockKey).push(result);
|
|
280
300
|
}
|
|
281
301
|
|
|
282
302
|
const cohortReport = {};
|
|
@@ -396,7 +416,7 @@ export function compareScorecards(a, b, { seed = 42 } = {}) {
|
|
|
396
416
|
const pairs = [];
|
|
397
417
|
for (const [id, ca] of aById) {
|
|
398
418
|
const cb = bById.get(id);
|
|
399
|
-
if (cb && ca
|
|
419
|
+
if (cb && blockKeyOf(ca) === key && blockKeyOf(cb) === key && ca.ok && cb.ok) pairs.push([ca, cb]);
|
|
400
420
|
}
|
|
401
421
|
const tokens = metricDelta(pairs, 'tokens_to_success', seed);
|
|
402
422
|
const time = metricDelta(pairs, 'time_to_success_ms', seed + 1);
|
|
@@ -1,17 +1,20 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* rollout.js (TASK-015, SPEC §5 DF-FR13 / §13) — per-seam staged rollout.
|
|
3
3
|
*
|
|
4
|
-
* SEAM_MIN_STAGE = { recorder: 'shadow',
|
|
5
|
-
*
|
|
4
|
+
* SEAM_MIN_STAGE = { recorder: 'shadow', analytics: 'shadow',
|
|
5
|
+
* projector: 'canary', evaluation: 'default' }
|
|
6
|
+
* resolveSeamStages(config) → { recorder, analytics, projector, evaluation }
|
|
6
7
|
* resolveSeamConfig(config, seam) → live config view bound to that seam
|
|
7
8
|
* killSwitch(config, seam?) → config
|
|
8
9
|
*
|
|
9
10
|
* The single global flag `observability.stage` promotes off → shadow →
|
|
10
11
|
* canary → default; each seam activates at its own minimum stage, so the
|
|
11
|
-
* ladder is: shadow records only
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
12
|
+
* ladder is: shadow records only (and unlocks the derived analytics lane —
|
|
13
|
+
* rebuildIndex/renderStoreDigest over retained segments), canary adds the
|
|
14
|
+
* support projection, default adds the evaluation surface. A seam whose
|
|
15
|
+
* effective stage falls below its minimum resolves to 'off' — the value
|
|
16
|
+
* existing consumers (createRecorder, projectSupport) already treat as
|
|
17
|
+
* their kill switch.
|
|
15
18
|
*
|
|
16
19
|
* Per-seam overrides live at `observability.seams.<seam>.stage` and are
|
|
17
20
|
* RESTRICTIVE-ONLY: the effective stage is min(global, override) clamped
|
|
@@ -38,10 +41,11 @@ import { resolveStage } from './emit/config.js';
|
|
|
38
41
|
|
|
39
42
|
const STAGE_RANK = { off: 0, shadow: 1, canary: 2, default: 3 };
|
|
40
43
|
|
|
41
|
-
export const SEAM_NAMES = Object.freeze(['recorder', 'projector', 'evaluation']);
|
|
44
|
+
export const SEAM_NAMES = Object.freeze(['recorder', 'analytics', 'projector', 'evaluation']);
|
|
42
45
|
|
|
43
46
|
export const SEAM_MIN_STAGE = Object.freeze({
|
|
44
47
|
recorder: 'shadow',
|
|
48
|
+
analytics: 'shadow',
|
|
45
49
|
projector: 'canary',
|
|
46
50
|
evaluation: 'default',
|
|
47
51
|
});
|