@ngockhoale/ukit 3.0.12 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +1 -0
- package/manifests/documentation.yaml +12 -0
- package/package.json +1 -1
- package/scripts/bench/data-foundation.mjs +368 -50
- package/src/cli/commands/doctor.js +232 -3
- package/src/cli/commands/feedback.js +64 -1
- package/src/cli/commands/install.js +18 -0
- package/src/cli/commands/memory.js +42 -37
- package/src/cli/commands/telemetry.js +460 -0
- package/src/cli/index.js +7 -0
- package/src/core/agentRuntime/adapters.js +83 -2
- package/src/core/agentRuntime/diagnostics.js +104 -0
- package/src/core/agentRuntime/supervisor.js +137 -0
- package/src/core/agentRuntime/telemetry.js +204 -0
- package/src/core/memory/memoryEmit.js +131 -0
- package/src/core/memory/memoryHit.js +1 -1
- package/src/core/memory/migrate.js +18 -11
- package/src/core/memory/migrateMapping.js +15 -7
- package/src/core/memory/mutateMemory.js +22 -4
- package/src/core/memory/recordIndex.js +10 -3
- package/src/core/memory/recordStore.js +28 -3
- package/src/core/memory/retrieval.js +79 -38
- package/src/core/memory/store.js +37 -37
- package/src/core/memory/storeV2.js +28 -25
- package/src/core/memory/storeV2Loader.js +2 -2
- package/src/core/observability/adapters/ingest.js +576 -0
- package/src/core/observability/analytics/anomalies.js +415 -0
- package/src/core/observability/analytics/summary.js +16 -1
- package/src/core/observability/emit/config.js +69 -1
- package/src/core/observability/emit/crash.js +434 -0
- package/src/core/observability/emit/lifecycle.js +349 -0
- package/src/core/observability/emit/recorder.js +135 -9
- package/src/core/observability/evaluation/aiPacket.js +52 -10
- package/src/core/observability/evaluation/outcomes.js +95 -0
- package/src/core/observability/evaluation/runner.js +225 -0
- package/src/core/observability/privacy/allowlist.js +23 -3
- package/src/core/observability/schema/compatibility.js +48 -3
- package/src/core/observability/schema/constants.js +5 -0
- package/src/core/observability/schema/registry.js +57 -0
- package/src/core/observability/schema/validate.js +68 -6
- package/src/core/observability/segments/internal.js +42 -8
- package/src/core/observability/segments/readSegments.js +35 -1
- package/src/core/observability/segments/recovery.js +3 -2
- package/src/core/observability/segments/retention.js +137 -33
- package/src/core/observability/support/projector.js +88 -18
- package/src/core/observability/support/provision.js +160 -0
- package/src/core/observability/support/renderer.js +2 -2
- package/src/core/observability/support/schedule.js +174 -0
- package/template_project/.claude/hooks/auto-allow-bash.sh +7 -1
- package/template_project/.claude/hooks/auto-prune-bash.sh +16 -7
- package/template_project/.claude/hooks/verification-guard.sh +13 -4
- package/template_project/.claude/ukit/runtime/async-lock.mjs +26 -0
|
@@ -2,17 +2,19 @@
|
|
|
2
2
|
* aiPacket.js (TASK-013, SPEC §5 DF-FR12, §8) — compact offline evaluator
|
|
3
3
|
* packet.
|
|
4
4
|
*
|
|
5
|
-
* buildEvaluatorPacket(summaries): SafePacket
|
|
5
|
+
* buildEvaluatorPacket(summaries, extras?): SafePacket
|
|
6
6
|
*
|
|
7
7
|
* Builds the bounded, sanitized packet an offline AI evaluator consumes.
|
|
8
8
|
* The packet carries ONLY the evaluator's reading sequence — summary →
|
|
9
|
-
* anomalies → representative evidence refs — derived from TraceSummaries
|
|
9
|
+
* anomalies → representative evidence refs — derived from TraceSummaries,
|
|
10
|
+
* plus (when `extras` is given) the derived `anomalies` and
|
|
11
|
+
* `opportunities` sections from the analytics lane (TASK-014, DF2-FR14).
|
|
10
12
|
* It never contains raw records, prompts, tee output, or payload content.
|
|
11
13
|
*
|
|
12
14
|
* Contract:
|
|
13
15
|
* - Hard cap: serialized packet never exceeds PACKET_MAX_BYTES (16KB).
|
|
14
|
-
* Overflow
|
|
15
|
-
* `
|
|
16
|
+
* Overflow sheds whole traces first, then `opportunities`, then
|
|
17
|
+
* `anomalies` — each section omission is marked in `omitted_sections`.
|
|
16
18
|
* - Sanitized: every string passes redactString (redaction rules +
|
|
17
19
|
* secret scanner + length cap); unknown fields are dropped, not
|
|
18
20
|
* carried through.
|
|
@@ -175,8 +177,8 @@ function traceEntryOf(summary) {
|
|
|
175
177
|
};
|
|
176
178
|
}
|
|
177
179
|
|
|
178
|
-
function buildPacket(traces, totalCount, truncated, omitted) {
|
|
179
|
-
|
|
180
|
+
function buildPacket(traces, totalCount, truncated, omitted, sections = null, omittedSections = []) {
|
|
181
|
+
const packet = {
|
|
180
182
|
packet_version: PACKET_VERSION,
|
|
181
183
|
generated_by: 'buildEvaluatorPacket',
|
|
182
184
|
metric_versions: [...new Set(traces.map((t) => t.metric_version).filter(Boolean))].sort(),
|
|
@@ -190,17 +192,37 @@ function buildPacket(traces, totalCount, truncated, omitted) {
|
|
|
190
192
|
},
|
|
191
193
|
traces,
|
|
192
194
|
};
|
|
195
|
+
// Extras keys exist only when the caller supplied them — the unpinned
|
|
196
|
+
// trace-only shape (no anomalies/opportunities keys) stays intact.
|
|
197
|
+
if (sections !== null) {
|
|
198
|
+
packet.anomalies = sections.anomalies;
|
|
199
|
+
packet.opportunities = sections.opportunities;
|
|
200
|
+
packet.omitted_sections = omittedSections;
|
|
201
|
+
}
|
|
202
|
+
return packet;
|
|
193
203
|
}
|
|
194
204
|
|
|
195
205
|
function packetBytes(packet) {
|
|
196
206
|
return Buffer.byteLength(JSON.stringify(packet), 'utf8');
|
|
197
207
|
}
|
|
198
208
|
|
|
199
|
-
export function buildEvaluatorPacket(summaries) {
|
|
209
|
+
export function buildEvaluatorPacket(summaries, extras = {}) {
|
|
200
210
|
if (!Array.isArray(summaries)) {
|
|
201
211
|
throw new TypeError('buildEvaluatorPacket: summaries must be an array of TraceSummary');
|
|
202
212
|
}
|
|
203
213
|
|
|
214
|
+
// DF2-FR14: the runner passes the derived analytics sections alongside
|
|
215
|
+
// the trace summaries. Both go through the same sanitizer as trace
|
|
216
|
+
// content — no caller field survives untouched.
|
|
217
|
+
const hasExtras =
|
|
218
|
+
isPlainObject(extras) &&
|
|
219
|
+
(extras.anomalies !== undefined || extras.opportunities !== undefined);
|
|
220
|
+
const anomalies =
|
|
221
|
+
hasExtras && Array.isArray(extras.anomalies) ? sanitizeValue(extras.anomalies, 0) : [];
|
|
222
|
+
const opportunities =
|
|
223
|
+
hasExtras && Array.isArray(extras.opportunities) ? sanitizeValue(extras.opportunities, 0) : [];
|
|
224
|
+
const sections = hasExtras ? { anomalies, opportunities } : null;
|
|
225
|
+
|
|
204
226
|
const traces = summaries
|
|
205
227
|
.map(traceEntryOf)
|
|
206
228
|
.map((entry) => sanitizeValue(entry, 0))
|
|
@@ -211,7 +233,7 @@ export function buildEvaluatorPacket(summaries) {
|
|
|
211
233
|
});
|
|
212
234
|
|
|
213
235
|
const total = traces.length;
|
|
214
|
-
let packet = buildPacket(traces, total, false, 0);
|
|
236
|
+
let packet = buildPacket(traces, total, false, 0, sections);
|
|
215
237
|
if (packetBytes(packet) <= PACKET_MAX_BYTES) return packet;
|
|
216
238
|
|
|
217
239
|
// Deterministic bound: keep the largest trace prefix that fits the cap.
|
|
@@ -219,12 +241,32 @@ export function buildEvaluatorPacket(summaries) {
|
|
|
219
241
|
let hi = traces.length;
|
|
220
242
|
while (lo < hi) {
|
|
221
243
|
const mid = (lo + hi + 1) >> 1;
|
|
222
|
-
if (
|
|
244
|
+
if (
|
|
245
|
+
packetBytes(buildPacket(traces.slice(0, mid), total, true, total - mid, sections)) <=
|
|
246
|
+
PACKET_MAX_BYTES
|
|
247
|
+
) {
|
|
223
248
|
lo = mid;
|
|
224
249
|
} else {
|
|
225
250
|
hi = mid - 1;
|
|
226
251
|
}
|
|
227
252
|
}
|
|
228
|
-
packet = buildPacket(traces.slice(0, lo), total, true, total - lo);
|
|
253
|
+
packet = buildPacket(traces.slice(0, lo), total, true, total - lo, sections);
|
|
254
|
+
if (packetBytes(packet) <= PACKET_MAX_BYTES) return packet;
|
|
255
|
+
|
|
256
|
+
// Traces alone no longer fit: shed the derived sections next —
|
|
257
|
+
// opportunities before anomalies — each omission recorded explicitly.
|
|
258
|
+
const omittedSections = [];
|
|
259
|
+
for (const name of ['opportunities', 'anomalies']) {
|
|
260
|
+
const kept =
|
|
261
|
+
name === 'opportunities'
|
|
262
|
+
? { anomalies, opportunities: [] }
|
|
263
|
+
: { anomalies: [], opportunities: [] };
|
|
264
|
+
packet = buildPacket(traces.slice(0, lo), total, true, total - lo, kept, [
|
|
265
|
+
...omittedSections,
|
|
266
|
+
name,
|
|
267
|
+
]);
|
|
268
|
+
omittedSections.push(name);
|
|
269
|
+
if (packetBytes(packet) <= PACKET_MAX_BYTES) return packet;
|
|
270
|
+
}
|
|
229
271
|
return packet;
|
|
230
272
|
}
|
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
// outcomes.js (TASK-013, SPEC §5 DF2-FR13, §7, §8) — canonical outcome
|
|
2
|
+
// signals. One write-time seam:
|
|
3
|
+
//
|
|
4
|
+
// recordOutcome({recorder, ref?, verdict, kind, source?, sessionId?,
|
|
5
|
+
// evidenceRefs?}) → EmitResult
|
|
6
|
+
//
|
|
7
|
+
// EmitResult is the recorder's own { status: 'accepted'|'dropped', reason? }
|
|
8
|
+
// so callers see the true fate; only caller-side validation failures are
|
|
9
|
+
// invented here: bad verdict → dropped/invalid-verdict, bad kind →
|
|
10
|
+
// dropped/invalid-kind, missing recorder → dropped/recorder-missing.
|
|
11
|
+
// Off-stage → the recorder's own dropped/stage-off (this module never
|
|
12
|
+
// bypasses the gate, and never throws — a caller can always log the result
|
|
13
|
+
// and move on).
|
|
14
|
+
//
|
|
15
|
+
// Payload (SPEC §7): {verdict, kind, source, trace_ref|execution_ref,
|
|
16
|
+
// evidence_refs[], ref:{kind:'trace'|'execution'|'unresolved', value?}}.
|
|
17
|
+
// `ref` is honest joining: a resolved ref lands BOTH as a typed
|
|
18
|
+
// {kind,value} and as the SPEC-shaped trace_ref/execution_ref field; a
|
|
19
|
+
// missing or ambiguous ref emits ref:{kind:'unresolved'} — never guessed,
|
|
20
|
+
// never fabricated (SPEC §8 honesty + task test 3). trace_ref/execution_ref
|
|
21
|
+
// are omitted when unresolved rather than filled with a placeholder.
|
|
22
|
+
//
|
|
23
|
+
// Payload discipline: metadata only. Free-form feedback text is denied at
|
|
24
|
+
// the sanitize boundary anyway; this module never puts it in a payload —
|
|
25
|
+
// pass ids via sessionId/evidenceRefs instead.
|
|
26
|
+
|
|
27
|
+
const VERDICTS = new Set(['accepted', 'rejected', 'revised', 'unknown']);
|
|
28
|
+
const KINDS = new Set(['user-correction', 'acceptance', 'verification', 'manual']);
|
|
29
|
+
const REF_KINDS = new Set(['trace', 'execution']);
|
|
30
|
+
const EVIDENCE_REFS_MAX = 32;
|
|
31
|
+
|
|
32
|
+
function isPlainObject(value) {
|
|
33
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
function isNonEmptyString(value) {
|
|
37
|
+
return typeof value === 'string' && value.length > 0;
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
// One honest resolution: 'trace'|'execution' → {kind,value}; anything else
|
|
41
|
+
// → {kind:'unresolved'}. Accepts {kind,value}, {trace_ref}, {execution_ref},
|
|
42
|
+
// {traceRef}, {executionRef} so callers pass whichever shape their record
|
|
43
|
+
// already carries; a bare string is ambiguous between the two ref kinds and
|
|
44
|
+
// resolves honestly to unresolved rather than guessing the join key.
|
|
45
|
+
function resolveRef(ref) {
|
|
46
|
+
if (!isPlainObject(ref)) return { kind: 'unresolved' };
|
|
47
|
+
if (REF_KINDS.has(ref.kind) && isNonEmptyString(ref.value)) {
|
|
48
|
+
return { kind: ref.kind, value: ref.value };
|
|
49
|
+
}
|
|
50
|
+
const trace = ref.trace_ref ?? ref.traceRef;
|
|
51
|
+
if (isNonEmptyString(trace)) return { kind: 'trace', value: trace };
|
|
52
|
+
const execution = ref.execution_ref ?? ref.executionRef;
|
|
53
|
+
if (isNonEmptyString(execution)) return { kind: 'execution', value: execution };
|
|
54
|
+
return { kind: 'unresolved' };
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function cleanEvidenceRefs(refs) {
|
|
58
|
+
if (!Array.isArray(refs)) return [];
|
|
59
|
+
return refs.filter(isNonEmptyString).slice(0, EVIDENCE_REFS_MAX);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
export function recordOutcome({ recorder, ref, verdict, kind, source, sessionId, evidenceRefs } = {}) {
|
|
63
|
+
if (!VERDICTS.has(verdict)) {
|
|
64
|
+
return { status: 'dropped', reason: 'invalid-verdict' };
|
|
65
|
+
}
|
|
66
|
+
if (!KINDS.has(kind)) {
|
|
67
|
+
return { status: 'dropped', reason: 'invalid-kind' };
|
|
68
|
+
}
|
|
69
|
+
if (!recorder || typeof recorder.emit !== 'function') {
|
|
70
|
+
return { status: 'dropped', reason: 'recorder-missing' };
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const resolved = resolveRef(ref);
|
|
74
|
+
const payload = {
|
|
75
|
+
verdict,
|
|
76
|
+
kind,
|
|
77
|
+
source: isNonEmptyString(source) ? source : 'unknown',
|
|
78
|
+
ref: resolved,
|
|
79
|
+
evidence_refs: cleanEvidenceRefs(evidenceRefs),
|
|
80
|
+
};
|
|
81
|
+
if (resolved.kind === 'trace') payload.trace_ref = resolved.value;
|
|
82
|
+
if (resolved.kind === 'execution') payload.execution_ref = resolved.value;
|
|
83
|
+
|
|
84
|
+
const record = { semantic_name: 'outcome.observed', payload };
|
|
85
|
+
if (isNonEmptyString(sessionId)) record.session_id = sessionId;
|
|
86
|
+
|
|
87
|
+
// emit() is already never-throw and returns a typed EmitResult; the guard
|
|
88
|
+
// keeps this seam honest even if a caller hands it a hand-rolled recorder
|
|
89
|
+
// that violates that contract.
|
|
90
|
+
try {
|
|
91
|
+
return recorder.emit(record);
|
|
92
|
+
} catch {
|
|
93
|
+
return { status: 'dropped', reason: 'recorder-error' };
|
|
94
|
+
}
|
|
95
|
+
}
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* runner.js (TASK-014, SPEC §5 DF2-FR14, §6, §8) — the AI evaluation lane.
|
|
3
|
+
*
|
|
4
|
+
* runEvaluation({ root, config, clock?, provider? })
|
|
5
|
+
* → Promise<{ status: 'ran'|'skipped'|'degraded', reason?,
|
|
6
|
+
* packet_bytes, findings_accepted, findings_rejected,
|
|
7
|
+
* kb_ids[] }>
|
|
8
|
+
* proposeAdaptation({ root?, clock?, proposal? })
|
|
9
|
+
* → { status: 'unsupported', reason, kb_id? } (P3 seam — never applies)
|
|
10
|
+
*
|
|
11
|
+
* `root` is the observability storage root: retained records live under
|
|
12
|
+
* `root/segments` and the optimization KB (optimization-kb.jsonl) is
|
|
13
|
+
* appended directly under `root`, alongside crashes/ and
|
|
14
|
+
* last-projection.json.
|
|
15
|
+
*
|
|
16
|
+
* Pipeline: rebuildIndex → detectAnomalies + detectOpportunities →
|
|
17
|
+
* buildEvaluatorPacket → injected provider → findings.js validation →
|
|
18
|
+
* append accepted findings to the KB as `evaluation` records.
|
|
19
|
+
*
|
|
20
|
+
* Provider seam: the evaluator transport is out of scope (planner decision
|
|
21
|
+
* — HTTP/UNIC plumbing is a later cycle's wiring). The provider is an
|
|
22
|
+
* injected async function `(packetJson) → { findings: [...] }`; dispatch
|
|
23
|
+
* only happens when BOTH sides exist: the operator opted in via
|
|
24
|
+
* `observability.evaluator.provider` and a callable `provider` was
|
|
25
|
+
* injected. Configured-but-unwired reports `provider_unavailable` so the
|
|
26
|
+
* CLI can distinguish "not configured" from "transport not wired yet".
|
|
27
|
+
* Without the config opt-in nothing ever runs — the deterministic
|
|
28
|
+
* analytics lane stays the default.
|
|
29
|
+
*
|
|
30
|
+
* Contract:
|
|
31
|
+
* - never throws on data; every failure path is a typed status.
|
|
32
|
+
* - `skipped` reasons: no-provider | provider_unavailable | no-data
|
|
33
|
+
* - `degraded` reasons: invalid_root | rebuild_failed |
|
|
34
|
+
* provider_timeout | provider_error | provider_malformed_response
|
|
35
|
+
* - KB entries carry provenance.source 'evaluated_judgment' + evidence
|
|
36
|
+
* refs; findings.js re-shapes them so evaluator opinion is recorded
|
|
37
|
+
* with decision 'unsupported', applied false — never promoted.
|
|
38
|
+
*/
|
|
39
|
+
|
|
40
|
+
import path from 'node:path';
|
|
41
|
+
|
|
42
|
+
import { rebuildIndex } from '../analytics/rebuild.js';
|
|
43
|
+
import { detectAnomalies } from '../analytics/anomalies.js';
|
|
44
|
+
import { detectOpportunities } from '../analytics/opportunities.js';
|
|
45
|
+
import { buildEvaluatorPacket } from './aiPacket.js';
|
|
46
|
+
import { validateEvaluatorFindings } from './findings.js';
|
|
47
|
+
import { createOptimizationStore } from './optimizationKnowledge.js';
|
|
48
|
+
|
|
49
|
+
const SEGMENTS_DIR = 'segments';
|
|
50
|
+
const DEFAULT_TIMEOUT_MS = 30_000;
|
|
51
|
+
const MAX_TIMEOUT_MS = 10 * 60_000;
|
|
52
|
+
|
|
53
|
+
function isPlainObject(value) {
|
|
54
|
+
return value !== null && typeof value === 'object' && !Array.isArray(value);
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
function result(status, extra = {}) {
|
|
58
|
+
return {
|
|
59
|
+
status,
|
|
60
|
+
packet_bytes: 0,
|
|
61
|
+
findings_accepted: 0,
|
|
62
|
+
findings_rejected: 0,
|
|
63
|
+
kb_ids: [],
|
|
64
|
+
...extra,
|
|
65
|
+
};
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
/**
|
|
69
|
+
* Read `observability.evaluator` from the runtime config. Fail-safe like
|
|
70
|
+
* resolveStage: absent or malformed node → null (lane off); a present
|
|
71
|
+
* node still requires a non-empty `provider` name. `timeoutMs` clamps to
|
|
72
|
+
* (0, 10min], defaulting to 30s.
|
|
73
|
+
*/
|
|
74
|
+
function resolveEvaluatorConfig(config) {
|
|
75
|
+
const node = isPlainObject(config) ? config.observability : undefined;
|
|
76
|
+
if (!isPlainObject(node)) return null;
|
|
77
|
+
const evaluator = node.evaluator;
|
|
78
|
+
if (!isPlainObject(evaluator)) return null;
|
|
79
|
+
if (typeof evaluator.provider !== 'string' || evaluator.provider.length === 0) return null;
|
|
80
|
+
const raw = evaluator.timeoutMs;
|
|
81
|
+
const timeoutMs =
|
|
82
|
+
typeof raw === 'number' && Number.isFinite(raw) && raw > 0
|
|
83
|
+
? Math.min(raw, MAX_TIMEOUT_MS)
|
|
84
|
+
: DEFAULT_TIMEOUT_MS;
|
|
85
|
+
return {
|
|
86
|
+
provider: evaluator.provider,
|
|
87
|
+
endpoint: typeof evaluator.endpoint === 'string' ? evaluator.endpoint : null,
|
|
88
|
+
model: typeof evaluator.model === 'string' ? evaluator.model : null,
|
|
89
|
+
timeoutMs,
|
|
90
|
+
};
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
function timeoutError() {
|
|
94
|
+
const err = new Error('evaluator provider timed out');
|
|
95
|
+
err.code = 'UKIT_EVALUATOR_TIMEOUT';
|
|
96
|
+
return err;
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
/** Race the injected provider against timeoutMs; the loser is dropped. */
|
|
100
|
+
function callProvider(provider, packetJson, timeoutMs) {
|
|
101
|
+
return new Promise((resolve, reject) => {
|
|
102
|
+
const timer = setTimeout(() => reject(timeoutError()), timeoutMs);
|
|
103
|
+
if (typeof timer.unref === 'function') timer.unref();
|
|
104
|
+
Promise.resolve()
|
|
105
|
+
.then(() => provider(packetJson))
|
|
106
|
+
.then(resolve, reject)
|
|
107
|
+
.finally(() => clearTimeout(timer));
|
|
108
|
+
});
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
/**
|
|
112
|
+
* Record one validated finding as an optimization-KB `evaluation` entry.
|
|
113
|
+
* findings.js already guarantees claim + evidence_refs and pins
|
|
114
|
+
* decision 'unsupported' / applied false — the store re-checks that
|
|
115
|
+
* contract itself, so an evaluator opinion can never be promoted here.
|
|
116
|
+
*/
|
|
117
|
+
function toKbEntry(finding, evaluator) {
|
|
118
|
+
return {
|
|
119
|
+
hypothesis: finding.claim,
|
|
120
|
+
evidence_refs: finding.evidence_refs,
|
|
121
|
+
decision: 'unsupported',
|
|
122
|
+
applied: false,
|
|
123
|
+
provenance: { source: 'evaluated_judgment' },
|
|
124
|
+
alternative_explanations: Array.isArray(finding.alternatives) ? finding.alternatives : [],
|
|
125
|
+
versions: {
|
|
126
|
+
confidence: finding.confidence,
|
|
127
|
+
evaluator_provider: evaluator.provider,
|
|
128
|
+
evaluator_model: evaluator.model,
|
|
129
|
+
},
|
|
130
|
+
};
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
export async function runEvaluation({ root, config, clock, provider } = {}) {
|
|
134
|
+
if (typeof root !== 'string' || root.length === 0) {
|
|
135
|
+
return result('degraded', { reason: 'invalid_root' });
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
// Provider gate first: no evaluator config → the deterministic lane is
|
|
139
|
+
// all that ever runs; never build or dispatch a packet the operator
|
|
140
|
+
// did not ask to send.
|
|
141
|
+
const evaluator = resolveEvaluatorConfig(config);
|
|
142
|
+
if (evaluator === null) return result('skipped', { reason: 'no-provider' });
|
|
143
|
+
if (typeof provider !== 'function') {
|
|
144
|
+
return result('skipped', { reason: 'provider_unavailable' });
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
let rebuilt;
|
|
148
|
+
try {
|
|
149
|
+
rebuilt = await rebuildIndex(path.join(root, SEGMENTS_DIR));
|
|
150
|
+
} catch {
|
|
151
|
+
return result('degraded', { reason: 'rebuild_failed' });
|
|
152
|
+
}
|
|
153
|
+
const summaries = Array.isArray(rebuilt && rebuilt.summaries) ? rebuilt.summaries : [];
|
|
154
|
+
if (summaries.length === 0) return result('skipped', { reason: 'no-data' });
|
|
155
|
+
|
|
156
|
+
const anomalies = detectAnomalies(summaries);
|
|
157
|
+
const opportunities = detectOpportunities(summaries);
|
|
158
|
+
const packet = buildEvaluatorPacket(summaries, { anomalies, opportunities });
|
|
159
|
+
const packetJson = JSON.stringify(packet);
|
|
160
|
+
const packetBytes = Buffer.byteLength(packetJson, 'utf8');
|
|
161
|
+
|
|
162
|
+
let payload;
|
|
163
|
+
try {
|
|
164
|
+
payload = await callProvider(provider, packetJson, evaluator.timeoutMs);
|
|
165
|
+
} catch (err) {
|
|
166
|
+
return result('degraded', {
|
|
167
|
+
reason: err && err.code === 'UKIT_EVALUATOR_TIMEOUT' ? 'provider_timeout' : 'provider_error',
|
|
168
|
+
packet_bytes: packetBytes,
|
|
169
|
+
});
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
const { findings, rejected } = validateEvaluatorFindings(payload);
|
|
173
|
+
// candidate_index null means the whole payload was malformed — that is
|
|
174
|
+
// a provider contract violation, reported as degraded rather than a
|
|
175
|
+
// clean run with zero findings.
|
|
176
|
+
if (rejected.some((entry) => entry && entry.candidate_index === null)) {
|
|
177
|
+
return result('degraded', {
|
|
178
|
+
reason: 'provider_malformed_response',
|
|
179
|
+
packet_bytes: packetBytes,
|
|
180
|
+
findings_rejected: rejected.length,
|
|
181
|
+
});
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
const store = createOptimizationStore({ root, clock });
|
|
185
|
+
const kbIds = [];
|
|
186
|
+
for (const finding of findings) {
|
|
187
|
+
const { id } = store.recordOptimization(toKbEntry(finding, evaluator));
|
|
188
|
+
kbIds.push(id);
|
|
189
|
+
}
|
|
190
|
+
|
|
191
|
+
return result('ran', {
|
|
192
|
+
packet_bytes: packetBytes,
|
|
193
|
+
findings_accepted: findings.length,
|
|
194
|
+
findings_rejected: rejected.length,
|
|
195
|
+
kb_ids: kbIds,
|
|
196
|
+
});
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* P3 adaptive-actuation seam — interface only. Proposals are recorded in
|
|
201
|
+
* the optimization KB as `unsupported`/unapplied evidence (never applied,
|
|
202
|
+
* never gated on), and the caller always gets the same typed answer: this
|
|
203
|
+
* lane does not actuate.
|
|
204
|
+
*/
|
|
205
|
+
export function proposeAdaptation({ root, clock, proposal } = {}) {
|
|
206
|
+
let kbId = null;
|
|
207
|
+
if (typeof root === 'string' && root.length > 0) {
|
|
208
|
+
const store = createOptimizationStore({ root, clock });
|
|
209
|
+
kbId = store.recordOptimization({
|
|
210
|
+
hypothesis: typeof proposal === 'string' && proposal.length > 0
|
|
211
|
+
? proposal
|
|
212
|
+
: 'adaptive actuation proposal (unnamed)',
|
|
213
|
+
evidence_refs: [],
|
|
214
|
+
decision: 'unsupported',
|
|
215
|
+
applied: false,
|
|
216
|
+
provenance: { source: 'optimization_decision' },
|
|
217
|
+
versions: { interface: 'proposeAdaptation', interface_only: true },
|
|
218
|
+
}).id;
|
|
219
|
+
}
|
|
220
|
+
return {
|
|
221
|
+
status: 'unsupported',
|
|
222
|
+
reason: 'adaptive actuation is interface-only (P3); proposals are recorded, never applied',
|
|
223
|
+
...(kbId ? { kb_id: kbId } : {}),
|
|
224
|
+
};
|
|
225
|
+
}
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
* redaction rules change so downstream readers can detect stale records.
|
|
12
12
|
*/
|
|
13
13
|
|
|
14
|
-
export const REDACTION_VERSION = 'df-redact-
|
|
14
|
+
export const REDACTION_VERSION = 'df-redact-5';
|
|
15
15
|
|
|
16
16
|
/**
|
|
17
17
|
* Envelope fields permitted on a persisted record (DF-FR01) plus the
|
|
@@ -39,6 +39,11 @@ export const ALLOWED_FIELDS = new Set([
|
|
|
39
39
|
'payload',
|
|
40
40
|
'origin',
|
|
41
41
|
'redaction_version',
|
|
42
|
+
// DF2-FR01 v1.1 additive envelope fields: multi-agent correlation ids and
|
|
43
|
+
// the recorder's sampling policy stamp — additive, never required.
|
|
44
|
+
'agent_id',
|
|
45
|
+
'project_instance_id',
|
|
46
|
+
'sampling',
|
|
42
47
|
]);
|
|
43
48
|
|
|
44
49
|
/**
|
|
@@ -130,7 +135,7 @@ export const ALLOWED_SUPPORT_PAYLOAD_FIELDS = new Set([
|
|
|
130
135
|
'exit_code',
|
|
131
136
|
'signal',
|
|
132
137
|
'attempt',
|
|
133
|
-
'
|
|
138
|
+
'attempt_index',
|
|
134
139
|
'retry_count',
|
|
135
140
|
'duration_ms',
|
|
136
141
|
'elapsed_ms',
|
|
@@ -140,6 +145,7 @@ export const ALLOWED_SUPPORT_PAYLOAD_FIELDS = new Set([
|
|
|
140
145
|
'count',
|
|
141
146
|
'total',
|
|
142
147
|
'dropped',
|
|
148
|
+
'dropped_count',
|
|
143
149
|
'sampled',
|
|
144
150
|
'truncated',
|
|
145
151
|
'cache',
|
|
@@ -157,8 +163,22 @@ export const ALLOWED_SUPPORT_PAYLOAD_FIELDS = new Set([
|
|
|
157
163
|
'p95',
|
|
158
164
|
'p99',
|
|
159
165
|
'min',
|
|
160
|
-
'max',
|
|
161
166
|
'mean',
|
|
167
|
+
// DF2-FR01 v1.1: metadata-only payload keys for the new semantic names —
|
|
168
|
+
// counts, kinds, opaque refs, verdict, detector, evidence_refs, crash
|
|
169
|
+
// codes/digest, and the sampling stamp shape. Never content fields.
|
|
170
|
+
'item_count',
|
|
171
|
+
'ref',
|
|
172
|
+
'trace_ref',
|
|
173
|
+
'execution_ref',
|
|
174
|
+
'verdict',
|
|
175
|
+
'detector',
|
|
176
|
+
'evidence_refs',
|
|
177
|
+
'cohort_key',
|
|
178
|
+
'codes',
|
|
179
|
+
'digest',
|
|
180
|
+
'policy_version',
|
|
181
|
+
'rate',
|
|
162
182
|
]);
|
|
163
183
|
|
|
164
184
|
// --- size caps (prototype defaults per SPEC §14; freeze after measurement) ---
|
|
@@ -109,9 +109,9 @@ export function checkRegistryCompatibility(prev, next, options = {}) {
|
|
|
109
109
|
}
|
|
110
110
|
|
|
111
111
|
/**
|
|
112
|
-
* Gate a semantic envelope on its declared schema major.
|
|
113
|
-
*
|
|
114
|
-
* `schema_version` (
|
|
112
|
+
* Gate a semantic envelope on its declared schema major. The version gate
|
|
113
|
+
* itself never inspects fields — compatibility is judged ONLY on
|
|
114
|
+
* `schema_version` (envelope-field strictness lives in
|
|
115
115
|
* validateSemanticRecord).
|
|
116
116
|
*
|
|
117
117
|
* @param {object} record — candidate semantic envelope.
|
|
@@ -133,3 +133,48 @@ export function validateEnvelopeSchemaVersion(record) {
|
|
|
133
133
|
}
|
|
134
134
|
return { ok: true };
|
|
135
135
|
}
|
|
136
|
+
|
|
137
|
+
/**
|
|
138
|
+
* Reader tolerance check (DF2-FR01): a v1 reader can consume records that
|
|
139
|
+
* carry v1.1 additive envelope fields (agent_id, project_instance_id,
|
|
140
|
+
* sampling) because the additive contract keeps schema_version at 1.
|
|
141
|
+
* The check gates on schema_version and REPORTS which fields were newer
|
|
142
|
+
* than the reader's known set — tolerance is observed, not assumed.
|
|
143
|
+
*
|
|
144
|
+
* @param {object} record — candidate semantic envelope.
|
|
145
|
+
* @returns {{ ok: true, newerFields: string[] } | { ok: false, reason: string }}
|
|
146
|
+
*/
|
|
147
|
+
export function checkReaderTolerance(record) {
|
|
148
|
+
const gate = validateEnvelopeSchemaVersion(record);
|
|
149
|
+
if (!gate.ok) return gate;
|
|
150
|
+
const newerFields = Object.keys(record).filter((key) => !V1_0_ENVELOPE_FIELDS.has(key));
|
|
151
|
+
return { ok: true, newerFields };
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
// The v1.0 baseline envelope — what a reader frozen before the v1.1
|
|
155
|
+
// additive release knows about. `redaction_version` is the stamp
|
|
156
|
+
// sanitizeObserved adds before persistence; v1.1's agent_id,
|
|
157
|
+
// project_instance_id and sampling are intentionally ABSENT so the check
|
|
158
|
+
// reports them as tolerated-but-newer.
|
|
159
|
+
const V1_0_ENVELOPE_FIELDS = new Set([
|
|
160
|
+
'record_type',
|
|
161
|
+
'semantic_name',
|
|
162
|
+
'schema_version',
|
|
163
|
+
'record_id',
|
|
164
|
+
'trace_id',
|
|
165
|
+
'span_id',
|
|
166
|
+
'parent_span_id',
|
|
167
|
+
'execution_id',
|
|
168
|
+
'session_id',
|
|
169
|
+
'project_ref',
|
|
170
|
+
'boot_id',
|
|
171
|
+
'writer_id',
|
|
172
|
+
'sequence',
|
|
173
|
+
'wall_time_utc',
|
|
174
|
+
'monotonic_ns',
|
|
175
|
+
'importance',
|
|
176
|
+
'privacy_class',
|
|
177
|
+
'origin',
|
|
178
|
+
'payload',
|
|
179
|
+
'redaction_version',
|
|
180
|
+
]);
|
|
@@ -26,6 +26,11 @@ export const ENVELOPE_FIELDS = Object.freeze([
|
|
|
26
26
|
'privacy_class',
|
|
27
27
|
'origin',
|
|
28
28
|
'payload',
|
|
29
|
+
// DF2-FR01 v1.1 additive optional fields (introduced_version '1.1') —
|
|
30
|
+
// multi-agent correlation ids and the recorder's sampling policy stamp.
|
|
31
|
+
'agent_id',
|
|
32
|
+
'project_instance_id',
|
|
33
|
+
'sampling',
|
|
29
34
|
]);
|
|
30
35
|
|
|
31
36
|
export const REQUIRED_ENVELOPE_FIELDS = Object.freeze([
|
|
@@ -14,6 +14,7 @@
|
|
|
14
14
|
import { PRIVACY_CLASSES } from './constants.js';
|
|
15
15
|
|
|
16
16
|
const V1 = '1';
|
|
17
|
+
const V1_1 = '1.1';
|
|
17
18
|
|
|
18
19
|
export const SEMANTIC_REGISTRY = Object.freeze({
|
|
19
20
|
'execution.started': {
|
|
@@ -160,6 +161,55 @@ export const SEMANTIC_REGISTRY = Object.freeze({
|
|
|
160
161
|
introduced_version: V1,
|
|
161
162
|
deprecated: null,
|
|
162
163
|
},
|
|
164
|
+
// --- schema v1.1 additions (DF2-FR01, append-only) ---
|
|
165
|
+
'memory.write': {
|
|
166
|
+
definition: 'A memory item was written to the store. Payload carries counts, kinds and an opaque ref — never content.',
|
|
167
|
+
unit: 'items',
|
|
168
|
+
privacy_class: 'sensitive',
|
|
169
|
+
retention_hint: 'aggregate',
|
|
170
|
+
introduced_version: V1_1,
|
|
171
|
+
deprecated: null,
|
|
172
|
+
},
|
|
173
|
+
'memory.retrieved': {
|
|
174
|
+
definition: 'Memory items were retrieved for a task. Payload carries counts, kinds and an opaque ref — never content.',
|
|
175
|
+
unit: 'items',
|
|
176
|
+
privacy_class: 'sensitive',
|
|
177
|
+
retention_hint: 'aggregate',
|
|
178
|
+
introduced_version: V1_1,
|
|
179
|
+
deprecated: null,
|
|
180
|
+
},
|
|
181
|
+
'retrieval.query': {
|
|
182
|
+
definition: 'A retrieval query ran against memory or context stores. Payload carries the query kind, result count and duration — never query text.',
|
|
183
|
+
unit: 'ms',
|
|
184
|
+
privacy_class: 'internal',
|
|
185
|
+
retention_hint: 'aggregate',
|
|
186
|
+
introduced_version: V1_1,
|
|
187
|
+
deprecated: null,
|
|
188
|
+
},
|
|
189
|
+
'outcome.observed': {
|
|
190
|
+
definition: 'A user or system outcome signal was observed and linked to a trace or execution ref. Payload carries verdict, kind, source and evidence_refs.',
|
|
191
|
+
unit: 'event',
|
|
192
|
+
privacy_class: 'internal',
|
|
193
|
+
retention_hint: 'retain',
|
|
194
|
+
introduced_version: V1_1,
|
|
195
|
+
deprecated: null,
|
|
196
|
+
},
|
|
197
|
+
'anomaly.detected': {
|
|
198
|
+
definition: 'A deterministic detector flagged an anomalous trace. Payload carries detector, kind, evidence_refs, typed confidence and cohort_key.',
|
|
199
|
+
unit: 'event',
|
|
200
|
+
privacy_class: 'internal',
|
|
201
|
+
retention_hint: 'retain',
|
|
202
|
+
introduced_version: V1_1,
|
|
203
|
+
deprecated: null,
|
|
204
|
+
},
|
|
205
|
+
'telemetry.crashed': {
|
|
206
|
+
definition: 'The host process crashed or exited abnormally; capture wrote a bounded crash file. Payload carries signal/error kind codes and a digest — never stacks or paths.',
|
|
207
|
+
unit: 'event',
|
|
208
|
+
privacy_class: 'internal',
|
|
209
|
+
retention_hint: 'retain',
|
|
210
|
+
introduced_version: V1_1,
|
|
211
|
+
deprecated: null,
|
|
212
|
+
},
|
|
163
213
|
});
|
|
164
214
|
|
|
165
215
|
// Error/wait/reason vocabulary: codes + definitions, no free-text grouping.
|
|
@@ -312,6 +362,13 @@ export const REASON_CODES = Object.freeze({
|
|
|
312
362
|
deprecated: null,
|
|
313
363
|
privacy_class: 'internal',
|
|
314
364
|
},
|
|
365
|
+
CONFIG_INVALID: {
|
|
366
|
+
definition: 'Sampling configuration was malformed; sampling disabled for this boot.',
|
|
367
|
+
unit: null,
|
|
368
|
+
introduced_version: V1_1,
|
|
369
|
+
deprecated: null,
|
|
370
|
+
privacy_class: 'internal',
|
|
371
|
+
},
|
|
315
372
|
});
|
|
316
373
|
|
|
317
374
|
// Guard: registry privacy classes must stay inside the frozen vocabulary.
|