@ngockhoale/ukit 3.0.12 → 3.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +1 -0
- package/manifests/documentation.yaml +12 -0
- package/package.json +1 -1
- package/scripts/bench/data-foundation.mjs +368 -50
- package/src/cli/commands/doctor.js +232 -3
- package/src/cli/commands/feedback.js +64 -1
- package/src/cli/commands/install.js +18 -0
- package/src/cli/commands/memory.js +42 -37
- package/src/cli/commands/telemetry.js +460 -0
- package/src/cli/index.js +7 -0
- package/src/core/agentRuntime/adapters.js +83 -2
- package/src/core/agentRuntime/diagnostics.js +104 -0
- package/src/core/agentRuntime/supervisor.js +137 -0
- package/src/core/agentRuntime/telemetry.js +204 -0
- package/src/core/memory/memoryEmit.js +131 -0
- package/src/core/memory/memoryHit.js +1 -1
- package/src/core/memory/migrate.js +18 -11
- package/src/core/memory/migrateMapping.js +15 -7
- package/src/core/memory/mutateMemory.js +22 -4
- package/src/core/memory/recordIndex.js +10 -3
- package/src/core/memory/recordStore.js +28 -3
- package/src/core/memory/retrieval.js +79 -38
- package/src/core/memory/store.js +37 -37
- package/src/core/memory/storeV2.js +28 -25
- package/src/core/memory/storeV2Loader.js +2 -2
- package/src/core/observability/adapters/ingest.js +576 -0
- package/src/core/observability/analytics/anomalies.js +415 -0
- package/src/core/observability/analytics/summary.js +16 -1
- package/src/core/observability/emit/config.js +69 -1
- package/src/core/observability/emit/crash.js +434 -0
- package/src/core/observability/emit/lifecycle.js +349 -0
- package/src/core/observability/emit/recorder.js +135 -9
- package/src/core/observability/evaluation/aiPacket.js +52 -10
- package/src/core/observability/evaluation/outcomes.js +95 -0
- package/src/core/observability/evaluation/runner.js +225 -0
- package/src/core/observability/privacy/allowlist.js +23 -3
- package/src/core/observability/schema/compatibility.js +48 -3
- package/src/core/observability/schema/constants.js +5 -0
- package/src/core/observability/schema/registry.js +57 -0
- package/src/core/observability/schema/validate.js +68 -6
- package/src/core/observability/segments/internal.js +42 -8
- package/src/core/observability/segments/readSegments.js +35 -1
- package/src/core/observability/segments/recovery.js +3 -2
- package/src/core/observability/segments/retention.js +137 -33
- package/src/core/observability/support/projector.js +88 -18
- package/src/core/observability/support/provision.js +160 -0
- package/src/core/observability/support/renderer.js +2 -2
- package/src/core/observability/support/schedule.js +174 -0
- package/template_project/.claude/hooks/auto-allow-bash.sh +7 -1
- package/template_project/.claude/hooks/auto-prune-bash.sh +16 -7
- package/template_project/.claude/hooks/verification-guard.sh +13 -4
- package/template_project/.claude/ukit/runtime/async-lock.mjs +26 -0
|
@@ -0,0 +1,576 @@
|
|
|
1
|
+
// ingest.js (TASK-006, SPEC §5 DF2-FR06, §8) — the collect lane that turns
|
|
2
|
+
// already-stored `.ukit/storage/` telemetry into canonical SemanticRecords.
|
|
3
|
+
// Sources (all read-only, never throws):
|
|
4
|
+
//
|
|
5
|
+
// hook_latency .ukit/storage/cache/hook-latency/*.jsonl
|
|
6
|
+
// rows from hook-telemetry.mjs → adaptHookTelemetryEvent
|
|
7
|
+
// routes .ukit/storage/cache/route-audit.json joined to
|
|
8
|
+
// cache/exec-ledger/*.json on requestKey
|
|
9
|
+
// (same join discipline as collectRouteOutcomes — that
|
|
10
|
+
// owner exports aggregates only, so the row join lives
|
|
11
|
+
// here) → adaptRouteOutcomes
|
|
12
|
+
// decisions .ukit/storage/memory/rollout-receipts.jsonl
|
|
13
|
+
// bounded receipts from memoryFlags.appendRolloutReceipt
|
|
14
|
+
// → mapped onto the decisionAdapter shape
|
|
15
|
+
// context .ukit/storage/cache/retriever-lanes.jsonl
|
|
16
|
+
// each event's mergedPaths become injected context items
|
|
17
|
+
// → adaptContextItems
|
|
18
|
+
//
|
|
19
|
+
// ingestStoredTelemetry({ projectRoot, recorder, deadlineMs?, sources?, now? })
|
|
20
|
+
// → Promise<{ ok, status:'complete'|'partial'|'degraded',
|
|
21
|
+
// coverage: { <source>: { read, emitted, rejected, skipped, reason? } },
|
|
22
|
+
// last_cursors }>
|
|
23
|
+
//
|
|
24
|
+
// Counters: `read` rows consumed by the adapter lane; `emitted` records
|
|
25
|
+
// accepted (or queued-at-full-queue) by recorder.emit; `rejected` covers
|
|
26
|
+
// malformed lines, null adapter output, and recorder drops — one logical
|
|
27
|
+
// row never lands twice. `skipped` counts rows left unprocessed when the
|
|
28
|
+
// deadline fired mid-source.
|
|
29
|
+
//
|
|
30
|
+
// Cursors live at `.ukit/storage/observability/ingest-cursors.json`
|
|
31
|
+
// (tmp+rename atomic): per-file byte offsets for append-only JSONL, a
|
|
32
|
+
// bounded seen-set for route join rows. Cursor loss or a cursor write
|
|
33
|
+
// failure re-reads rows it already emitted — bounded telemetry accepts a
|
|
34
|
+
// rare duplicate over blocking a collect. `feedback-manual.jsonl` is
|
|
35
|
+
// deliberately NOT read: TASK-013 emits `outcome.observed` at write-time;
|
|
36
|
+
// ingesting that file would double-count the same event.
|
|
37
|
+
|
|
38
|
+
import crypto from 'node:crypto';
|
|
39
|
+
import fs from 'node:fs/promises';
|
|
40
|
+
import path from 'node:path';
|
|
41
|
+
|
|
42
|
+
import { adaptHookTelemetryEvent } from './hookTelemetryAdapter.js';
|
|
43
|
+
import { adaptRouteOutcomes } from './routeAdapter.js';
|
|
44
|
+
import { adaptDecisionReceipts } from './decisionAdapter.js';
|
|
45
|
+
import { adaptContextItems } from './contextAdapter.js';
|
|
46
|
+
import { createAdapterContext, isPlainObject } from './common.js';
|
|
47
|
+
import { listLedgerFiles } from '../../../diagnostics/ledgerFiles.js';
|
|
48
|
+
|
|
49
|
+
const STORAGE_REL = path.join('.ukit', 'storage');
|
|
50
|
+
const CACHE_REL = path.join(STORAGE_REL, 'cache');
|
|
51
|
+
const HOOK_LATENCY_DIR_REL = path.join(CACHE_REL, 'hook-latency');
|
|
52
|
+
const ROUTE_AUDIT_REL = path.join(CACHE_REL, 'route-audit.json');
|
|
53
|
+
const EXEC_LEDGER_DIR_REL = path.join(CACHE_REL, 'exec-ledger');
|
|
54
|
+
const ROLLOUT_RECEIPTS_REL = path.join(STORAGE_REL, 'memory', 'rollout-receipts.jsonl');
|
|
55
|
+
const RETRIEVER_LANES_REL = path.join(CACHE_REL, 'retriever-lanes.jsonl');
|
|
56
|
+
const CURSOR_REL = path.join(STORAGE_REL, 'observability', 'ingest-cursors.json');
|
|
57
|
+
|
|
58
|
+
// Bounded cursor state: hook-latency keeps ≤200 files by owner sweep and
|
|
59
|
+
// the route seen-set only needs to outlive the audit window plus the
|
|
60
|
+
// ledger dir cap — these caps cover both with headroom.
|
|
61
|
+
const MAX_FILE_CURSORS = 512;
|
|
62
|
+
const MAX_ROUTE_SEEN = 4096;
|
|
63
|
+
const MAX_HOOK_FILES = 256;
|
|
64
|
+
const DEFAULT_LIMIT_LEDGERS = 500;
|
|
65
|
+
|
|
66
|
+
const LF = 0x0a;
|
|
67
|
+
const UTF8_BOM = '\uFEFF';
|
|
68
|
+
|
|
69
|
+
function zeroCoverage() {
|
|
70
|
+
return { read: 0, emitted: 0, rejected: 0, skipped: 0 };
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
function emptyCursors() {
|
|
74
|
+
return { version: 1, sources: {} };
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function boundedMapSet(map, key, value) {
|
|
78
|
+
map.delete(key); // refresh insertion order so pruning is oldest-first
|
|
79
|
+
map.set(key, value);
|
|
80
|
+
while (map.size > MAX_FILE_CURSORS) map.delete(map.keys().next().value);
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
async function readJsonSafe(filePath) {
|
|
84
|
+
try {
|
|
85
|
+
return { doc: JSON.parse(await fs.readFile(filePath, 'utf8')), error: null };
|
|
86
|
+
} catch (error) {
|
|
87
|
+
return { doc: null, error: error?.code === 'ENOENT' ? null : 'malformed_doc' };
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
async function loadCursors(projectRoot) {
|
|
92
|
+
const file = path.join(projectRoot, CURSOR_REL);
|
|
93
|
+
try {
|
|
94
|
+
const doc = JSON.parse(await fs.readFile(file, 'utf8'));
|
|
95
|
+
if (!isPlainObject(doc) || !isPlainObject(doc.sources)) {
|
|
96
|
+
return { cursors: emptyCursors(), reset: true };
|
|
97
|
+
}
|
|
98
|
+
return { cursors: { version: 1, sources: doc.sources }, reset: false };
|
|
99
|
+
} catch (error) {
|
|
100
|
+
return { cursors: emptyCursors(), reset: error?.code !== 'ENOENT' };
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
async function saveCursors(projectRoot, cursors) {
|
|
105
|
+
const file = path.join(projectRoot, CURSOR_REL);
|
|
106
|
+
const tmp = `${file}.tmp-${process.pid}-${Math.random().toString(16).slice(2)}`;
|
|
107
|
+
await fs.mkdir(path.dirname(file), { recursive: true });
|
|
108
|
+
await fs.writeFile(tmp, `${JSON.stringify(cursors)}\n`, 'utf8');
|
|
109
|
+
try {
|
|
110
|
+
await fs.rename(tmp, file);
|
|
111
|
+
} catch (error) {
|
|
112
|
+
await fs.rm(tmp, { force: true }).catch(() => {});
|
|
113
|
+
throw error;
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
// Split a buffer tail into lines with byte-accurate next-offsets. A
|
|
118
|
+
// trailing partial line (torn write) is returned too — like the segment
|
|
119
|
+
// reader it is a permanent reject, so the cursor may advance past it
|
|
120
|
+
// rather than blocking forever on a dead byte tail.
|
|
121
|
+
function splitLines(buf, start) {
|
|
122
|
+
const lines = [];
|
|
123
|
+
let pos = start;
|
|
124
|
+
while (pos < buf.length) {
|
|
125
|
+
const nl = buf.indexOf(LF, pos);
|
|
126
|
+
const end = nl === -1 ? buf.length : nl;
|
|
127
|
+
lines.push({ offset: pos, next: nl === -1 ? buf.length : nl + 1, raw: buf.subarray(pos, end) });
|
|
128
|
+
pos = lines[lines.length - 1].next;
|
|
129
|
+
}
|
|
130
|
+
return lines;
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
/**
|
|
134
|
+
* Consume an append-only JSONL file from its stored byte cursor.
|
|
135
|
+
* `rowFn(line)` adapts+emits one parsed row and updates `cov` itself;
|
|
136
|
+
* the consumer owns `read`, malformed-line `rejected`, deadline `skipped`,
|
|
137
|
+
* and cursor advancement. Rotation/truncation (file smaller than the
|
|
138
|
+
* stored offset) resets the cursor to 0 — the kept lines re-emit as
|
|
139
|
+
* bounded duplicates rather than silently vanishing.
|
|
140
|
+
*/
|
|
141
|
+
async function consumeJsonlFile(filePath, cursorEntry, cov, rowFn, expired) {
|
|
142
|
+
let stat;
|
|
143
|
+
try {
|
|
144
|
+
stat = await fs.stat(filePath);
|
|
145
|
+
} catch (error) {
|
|
146
|
+
return error?.code === 'ENOENT' ? {} : { error: 'io_stat' };
|
|
147
|
+
}
|
|
148
|
+
let offset = cursorEntry?.offset ?? 0;
|
|
149
|
+
if (!Number.isInteger(offset) || offset < 0 || offset > stat.size) offset = 0;
|
|
150
|
+
if (offset === stat.size) return {}; // no new bytes
|
|
151
|
+
|
|
152
|
+
let buf;
|
|
153
|
+
try {
|
|
154
|
+
buf = await fs.readFile(filePath);
|
|
155
|
+
} catch {
|
|
156
|
+
return { error: 'io_read' };
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
const lines = splitLines(buf, offset);
|
|
160
|
+
let consumed = offset;
|
|
161
|
+
let partial = false;
|
|
162
|
+
for (let i = 0; i < lines.length; i += 1) {
|
|
163
|
+
const line = lines[i];
|
|
164
|
+
if (expired()) {
|
|
165
|
+
cov.skipped += lines.length - i;
|
|
166
|
+
partial = true;
|
|
167
|
+
break;
|
|
168
|
+
}
|
|
169
|
+
cov.read += 1;
|
|
170
|
+
let text = line.raw.toString('utf8').trim();
|
|
171
|
+
if (line.offset === 0 && text.startsWith(UTF8_BOM)) text = text.slice(1);
|
|
172
|
+
if (text.length > 0) {
|
|
173
|
+
let parsed = null;
|
|
174
|
+
try {
|
|
175
|
+
parsed = JSON.parse(text);
|
|
176
|
+
} catch {
|
|
177
|
+
/* malformed line → permanent reject below */
|
|
178
|
+
}
|
|
179
|
+
if (parsed === null) cov.rejected += 1;
|
|
180
|
+
else if (rowFn(parsed) === 'skipped') {
|
|
181
|
+
cov.skipped += lines.length - i;
|
|
182
|
+
partial = true;
|
|
183
|
+
break;
|
|
184
|
+
}
|
|
185
|
+
}
|
|
186
|
+
// Blank and malformed lines are dead space — the cursor advances past
|
|
187
|
+
// them; only a rowFn('skipped') stop leaves the cursor behind.
|
|
188
|
+
consumed = line.next;
|
|
189
|
+
}
|
|
190
|
+
return { partial, entry: { offset: consumed, size: stat.size } };
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// --- source: routes -----------------------------------------------------------
|
|
194
|
+
|
|
195
|
+
// Join route-audit entries to exec-ledger docs on requestKey — identical
|
|
196
|
+
// discipline to collectRouteOutcomes (last audit entry per key wins,
|
|
197
|
+
// ledgers ordered mtime-desc, non-ledger filenames excluded), but kept at
|
|
198
|
+
// row granularity: adaptRouteOutcomes consumes {audit, ledger} pairs and
|
|
199
|
+
// the owner's public surface returns aggregates only.
|
|
200
|
+
|
|
201
|
+
// Deterministic serialization for the dedup fingerprint: object keys sort
|
|
202
|
+
// so a same-content write in a different key order cannot masquerade as a
|
|
203
|
+
// mutation. Values are ledger doc fields only — JSON.parse output, never
|
|
204
|
+
// cyclic.
|
|
205
|
+
function stableValue(value) {
|
|
206
|
+
if (Array.isArray(value)) return value.map(stableValue);
|
|
207
|
+
if (isPlainObject(value)) {
|
|
208
|
+
const out = {};
|
|
209
|
+
for (const key of Object.keys(value).sort()) out[key] = stableValue(value[key]);
|
|
210
|
+
return out;
|
|
211
|
+
}
|
|
212
|
+
return value;
|
|
213
|
+
}
|
|
214
|
+
|
|
215
|
+
// Ledger docs are mutated in place (execution-ledger.mjs bumps updatedAt on
|
|
216
|
+
// every mutation and appends receipts). The dedup key must reflect the doc
|
|
217
|
+
// version, not just its identity: updatedAt is the fast signal, and a
|
|
218
|
+
// digest of the stabilized doc catches mutations that reuse or lack a
|
|
219
|
+
// distinct timestamp (same-millisecond writes, hand-edited docs). An absent
|
|
220
|
+
// ledger fingerprints as '0'.
|
|
221
|
+
function ledgerFingerprint(doc) {
|
|
222
|
+
if (!isPlainObject(doc)) return '0';
|
|
223
|
+
const digest = crypto
|
|
224
|
+
.createHash('sha1')
|
|
225
|
+
.update(JSON.stringify(stableValue(doc)))
|
|
226
|
+
.digest('hex')
|
|
227
|
+
.slice(0, 12);
|
|
228
|
+
return `${Number.isFinite(doc.updatedAt) ? doc.updatedAt : 0}:${digest}`;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
async function collectRouteRows(projectRoot, { limitLedgers, errors }) {
|
|
232
|
+
const { doc: auditDoc, error: auditError } = await readJsonSafe(
|
|
233
|
+
path.join(projectRoot, ROUTE_AUDIT_REL),
|
|
234
|
+
);
|
|
235
|
+
if (auditError) errors.push('malformed_audit');
|
|
236
|
+
const auditEntries = Array.isArray(auditDoc?.entries) ? auditDoc.entries : [];
|
|
237
|
+
|
|
238
|
+
const ledgerDir = path.join(projectRoot, EXEC_LEDGER_DIR_REL);
|
|
239
|
+
const names = await listLedgerFiles(ledgerDir, limitLedgers);
|
|
240
|
+
const ledgers = [];
|
|
241
|
+
for (const name of names) {
|
|
242
|
+
const { doc, error } = await readJsonSafe(path.join(ledgerDir, name));
|
|
243
|
+
if (error) {
|
|
244
|
+
errors.push('malformed_ledger');
|
|
245
|
+
continue;
|
|
246
|
+
}
|
|
247
|
+
if (isPlainObject(doc)) ledgers.push({ name, doc });
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
const auditByKey = new Map();
|
|
251
|
+
for (const entry of auditEntries) {
|
|
252
|
+
if (isPlainObject(entry) && typeof entry.requestKey === 'string' && entry.requestKey) {
|
|
253
|
+
auditByKey.set(entry.requestKey, entry);
|
|
254
|
+
}
|
|
255
|
+
}
|
|
256
|
+
const ledgerByKey = new Map();
|
|
257
|
+
for (const { name, doc } of ledgers) {
|
|
258
|
+
if (typeof doc.requestKey === 'string' && doc.requestKey && !ledgerByKey.has(doc.requestKey)) {
|
|
259
|
+
ledgerByKey.set(doc.requestKey, { name, doc });
|
|
260
|
+
}
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
const rows = [];
|
|
264
|
+
const matched = new Set();
|
|
265
|
+
for (const audit of auditByKey.values()) {
|
|
266
|
+
const hit = ledgerByKey.get(audit.requestKey);
|
|
267
|
+
if (hit) matched.add(hit.name);
|
|
268
|
+
// The dedup key embeds the ledger fingerprint (updatedAt + digest of
|
|
269
|
+
// the doc), so a ledger arriving after its audit — or mutated in place
|
|
270
|
+
// (execution-ledger.mjs rewrites the same file as the run progresses) —
|
|
271
|
+
// re-emits as the newer row instead of deduping against the stale key.
|
|
272
|
+
rows.push({
|
|
273
|
+
key: `a|${audit.ts ?? ''}|${audit.requestKey}|${hit ? ledgerFingerprint(hit.doc) : 0}`,
|
|
274
|
+
row: { audit, ledger: hit?.doc ?? {} },
|
|
275
|
+
});
|
|
276
|
+
}
|
|
277
|
+
for (const { name, doc } of ledgers) {
|
|
278
|
+
if (matched.has(name)) continue;
|
|
279
|
+
if (typeof doc.requestKey !== 'string' || !doc.requestKey) continue;
|
|
280
|
+
rows.push({
|
|
281
|
+
key: `l|${name}|${doc.requestKey}|${ledgerFingerprint(doc)}`,
|
|
282
|
+
row: { audit: {}, ledger: doc },
|
|
283
|
+
});
|
|
284
|
+
}
|
|
285
|
+
return rows;
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
function decisionReceiptShape(line) {
|
|
289
|
+
// rollout-receipts.jsonl rows carry {ts, plane, stage, outcome, code,
|
|
290
|
+
// latencyBand} (memoryFlags.RECEIPT_FIELDS). Map onto the adapter's
|
|
291
|
+
// provenance shape: plane → experiment, outcome → outcomeClass,
|
|
292
|
+
// code → fallbackCode, latencyBand → latencyClass.
|
|
293
|
+
if (!isPlainObject(line)) return null;
|
|
294
|
+
return {
|
|
295
|
+
ts: line.ts,
|
|
296
|
+
stage: line.stage,
|
|
297
|
+
experiment: line.plane,
|
|
298
|
+
outcomeClass: line.outcome,
|
|
299
|
+
fallbackCode: line.code,
|
|
300
|
+
latencyClass: line.latencyBand,
|
|
301
|
+
};
|
|
302
|
+
}
|
|
303
|
+
|
|
304
|
+
function contextItemShapes(line) {
|
|
305
|
+
// A retriever-lane event's mergedPaths are the items that entered the
|
|
306
|
+
// episode context. `referenced` stays false — nothing on disk observes a
|
|
307
|
+
// later reference, and the adapter marks the record a heuristic signal
|
|
308
|
+
// (evaluator_caveat), never a verdict.
|
|
309
|
+
if (!isPlainObject(line) || !Array.isArray(line.mergedPaths)) return [];
|
|
310
|
+
const items = [];
|
|
311
|
+
for (const ref of line.mergedPaths.slice(0, 32)) {
|
|
312
|
+
if (typeof ref !== 'string' || ref.length === 0) continue;
|
|
313
|
+
items.push({
|
|
314
|
+
kind: 'retrieval',
|
|
315
|
+
ref,
|
|
316
|
+
ts: line.ts,
|
|
317
|
+
selected: true,
|
|
318
|
+
injected: true,
|
|
319
|
+
referenced: false,
|
|
320
|
+
});
|
|
321
|
+
}
|
|
322
|
+
return items;
|
|
323
|
+
}
|
|
324
|
+
|
|
325
|
+
/**
|
|
326
|
+
* Emit one adapted record through the recorder and count it. The recorder
|
|
327
|
+
* contract: 'accepted' queues the record; 'queue-full' still queues the
|
|
328
|
+
* incoming record (the OLDEST queued one is evicted — the new row lands);
|
|
329
|
+
* every other status is a drop this lane counts as rejected, keeping the
|
|
330
|
+
* recorder's typed reason (e.g. 'stage-off', 'sampled-out') on coverage.
|
|
331
|
+
*/
|
|
332
|
+
function emitCounted(recorder, cov, record) {
|
|
333
|
+
let res = null;
|
|
334
|
+
try {
|
|
335
|
+
res = recorder.emit(record);
|
|
336
|
+
} catch {
|
|
337
|
+
res = { status: 'dropped', reason: 'emit_threw' };
|
|
338
|
+
}
|
|
339
|
+
if (res && (res.status === 'accepted' || res.status === 'queue-full')) {
|
|
340
|
+
cov.emitted += 1;
|
|
341
|
+
return true;
|
|
342
|
+
}
|
|
343
|
+
cov.rejected += 1;
|
|
344
|
+
cov.last_drop_reason = res?.reason ?? 'emit_rejected';
|
|
345
|
+
return false;
|
|
346
|
+
}
|
|
347
|
+
|
|
348
|
+
export async function ingestStoredTelemetry({
|
|
349
|
+
projectRoot,
|
|
350
|
+
recorder,
|
|
351
|
+
deadlineMs,
|
|
352
|
+
sources,
|
|
353
|
+
now,
|
|
354
|
+
limitLedgers,
|
|
355
|
+
} = {}) {
|
|
356
|
+
if (
|
|
357
|
+
typeof projectRoot !== 'string' ||
|
|
358
|
+
projectRoot.length === 0 ||
|
|
359
|
+
!recorder ||
|
|
360
|
+
typeof recorder.emit !== 'function'
|
|
361
|
+
) {
|
|
362
|
+
return { ok: false, reason: 'invalid_args' };
|
|
363
|
+
}
|
|
364
|
+
|
|
365
|
+
const nowFn = typeof now === 'function' ? now : () => Date.now();
|
|
366
|
+
const start = nowFn();
|
|
367
|
+
const deadline = Number.isFinite(deadlineMs) ? start + deadlineMs : null;
|
|
368
|
+
const expired = () => deadline !== null && nowFn() >= deadline;
|
|
369
|
+
const wanted = Array.isArray(sources) && sources.length > 0 ? new Set(sources) : null;
|
|
370
|
+
const enabled = (name) => wanted === null || wanted.has(name);
|
|
371
|
+
|
|
372
|
+
const { cursors, reset: cursorReset } = await loadCursors(projectRoot);
|
|
373
|
+
const coverage = {};
|
|
374
|
+
let degraded = false;
|
|
375
|
+
let partial = false;
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
const runSource = async (name, fn) => {
|
|
379
|
+
const cov = zeroCoverage();
|
|
380
|
+
coverage[name] = cov;
|
|
381
|
+
if (!enabled(name)) {
|
|
382
|
+
cov.reason = 'not_requested';
|
|
383
|
+
return;
|
|
384
|
+
}
|
|
385
|
+
if (expired()) {
|
|
386
|
+
partial = true;
|
|
387
|
+
cov.reason = 'deadline';
|
|
388
|
+
return;
|
|
389
|
+
}
|
|
390
|
+
try {
|
|
391
|
+
await fn(cov);
|
|
392
|
+
} catch {
|
|
393
|
+
degraded = true;
|
|
394
|
+
cov.errors = ['source_error'];
|
|
395
|
+
cov.reason = 'source_error';
|
|
396
|
+
}
|
|
397
|
+
if (cov.deadlineHit) {
|
|
398
|
+
partial = true;
|
|
399
|
+
cov.reason = 'deadline';
|
|
400
|
+
}
|
|
401
|
+
if (cov.errors && !cov.reason) cov.reason = cov.errors[0];
|
|
402
|
+
if (cov.last_drop_reason && !cov.reason) cov.reason = cov.last_drop_reason;
|
|
403
|
+
};
|
|
404
|
+
|
|
405
|
+
// --- hook_latency ---
|
|
406
|
+
await runSource('hook_latency', async (cov) => {
|
|
407
|
+
const dir = path.join(projectRoot, HOOK_LATENCY_DIR_REL);
|
|
408
|
+
let entries;
|
|
409
|
+
try {
|
|
410
|
+
entries = await fs.readdir(dir, { withFileTypes: true });
|
|
411
|
+
} catch {
|
|
412
|
+
return; // missing/unreadable dir → zero coverage, not an error
|
|
413
|
+
}
|
|
414
|
+
const files = entries
|
|
415
|
+
.filter((e) => e.isFile() && e.name.endsWith('.jsonl'))
|
|
416
|
+
.map((e) => e.name)
|
|
417
|
+
.sort()
|
|
418
|
+
.slice(0, MAX_HOOK_FILES);
|
|
419
|
+
const stored = cursors.sources.hook_latency;
|
|
420
|
+
const fileCursors = new Map(
|
|
421
|
+
Object.entries(isPlainObject(stored?.files) ? stored.files : {}),
|
|
422
|
+
);
|
|
423
|
+
const ctx = createAdapterContext({ writerId: 'ingest-hook-latency' });
|
|
424
|
+
for (const name of files) {
|
|
425
|
+
if (expired()) {
|
|
426
|
+
cov.deadlineHit = true;
|
|
427
|
+
break;
|
|
428
|
+
}
|
|
429
|
+
const outcome = await consumeJsonlFile(
|
|
430
|
+
path.join(dir, name),
|
|
431
|
+
fileCursors.get(name),
|
|
432
|
+
cov,
|
|
433
|
+
(row) => {
|
|
434
|
+
const record = adaptHookTelemetryEvent(row, ctx);
|
|
435
|
+
if (!record) {
|
|
436
|
+
cov.rejected += 1;
|
|
437
|
+
return 'consumed';
|
|
438
|
+
}
|
|
439
|
+
emitCounted(recorder, cov, record);
|
|
440
|
+
return 'consumed';
|
|
441
|
+
},
|
|
442
|
+
expired,
|
|
443
|
+
);
|
|
444
|
+
if (outcome.error) {
|
|
445
|
+
degraded = true;
|
|
446
|
+
(cov.errors ??= []).push(outcome.error);
|
|
447
|
+
}
|
|
448
|
+
if (outcome.entry) boundedMapSet(fileCursors, name, outcome.entry);
|
|
449
|
+
if (outcome.partial) {
|
|
450
|
+
cov.deadlineHit = true;
|
|
451
|
+
break;
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
cursors.sources.hook_latency = { files: Object.fromEntries(fileCursors) };
|
|
455
|
+
});
|
|
456
|
+
|
|
457
|
+
// --- routes ---
|
|
458
|
+
await runSource('routes', async (cov) => {
|
|
459
|
+
const joinErrors = [];
|
|
460
|
+
const rows = await collectRouteRows(projectRoot, {
|
|
461
|
+
limitLedgers:
|
|
462
|
+
Number.isInteger(limitLedgers) && limitLedgers > 0 ? limitLedgers : DEFAULT_LIMIT_LEDGERS,
|
|
463
|
+
errors: joinErrors,
|
|
464
|
+
});
|
|
465
|
+
if (joinErrors.length > 0) {
|
|
466
|
+
degraded = true;
|
|
467
|
+
cov.errors = joinErrors;
|
|
468
|
+
}
|
|
469
|
+
const stored = cursors.sources.routes;
|
|
470
|
+
const seen = new Set(Array.isArray(stored?.seen) ? stored.seen : []);
|
|
471
|
+
const ctxOpts = { writerId: 'ingest-routes' };
|
|
472
|
+
for (let i = 0; i < rows.length; i += 1) {
|
|
473
|
+
if (expired()) {
|
|
474
|
+
cov.skipped += rows.length - i;
|
|
475
|
+
cov.deadlineHit = true;
|
|
476
|
+
break;
|
|
477
|
+
}
|
|
478
|
+
const { key, row } = rows[i];
|
|
479
|
+
if (seen.has(key)) continue;
|
|
480
|
+
cov.read += 1;
|
|
481
|
+
const records = adaptRouteOutcomes([row], ctxOpts);
|
|
482
|
+
if (records.length === 0) {
|
|
483
|
+
cov.rejected += 1;
|
|
484
|
+
}
|
|
485
|
+
for (const record of records) {
|
|
486
|
+
emitCounted(recorder, cov, record);
|
|
487
|
+
}
|
|
488
|
+
seen.add(key);
|
|
489
|
+
}
|
|
490
|
+
// Bounded tail: dedup only needs to outlive the input window.
|
|
491
|
+
const all = [...seen];
|
|
492
|
+
cursors.sources.routes = {
|
|
493
|
+
seen: all.length > MAX_ROUTE_SEEN ? all.slice(-MAX_ROUTE_SEEN) : all,
|
|
494
|
+
...(all.length > MAX_ROUTE_SEEN ? { seen_truncated: true } : {}),
|
|
495
|
+
};
|
|
496
|
+
});
|
|
497
|
+
|
|
498
|
+
// --- decisions ---
|
|
499
|
+
await runSource('decisions', async (cov) => {
|
|
500
|
+
const stored = cursors.sources.decisions;
|
|
501
|
+
const ctxOpts = { writerId: 'ingest-decisions' };
|
|
502
|
+
const outcome = await consumeJsonlFile(
|
|
503
|
+
path.join(projectRoot, ROLLOUT_RECEIPTS_REL),
|
|
504
|
+
isPlainObject(stored?.entry) ? stored.entry : null,
|
|
505
|
+
cov,
|
|
506
|
+
(line) => {
|
|
507
|
+
const shaped = decisionReceiptShape(line);
|
|
508
|
+
const records = shaped ? adaptDecisionReceipts([shaped], ctxOpts) : [];
|
|
509
|
+
if (records.length === 0) {
|
|
510
|
+
cov.rejected += 1;
|
|
511
|
+
return 'consumed';
|
|
512
|
+
}
|
|
513
|
+
for (const record of records) {
|
|
514
|
+
emitCounted(recorder, cov, record);
|
|
515
|
+
}
|
|
516
|
+
return 'consumed';
|
|
517
|
+
},
|
|
518
|
+
expired,
|
|
519
|
+
);
|
|
520
|
+
if (outcome.error) {
|
|
521
|
+
degraded = true;
|
|
522
|
+
(cov.errors ??= []).push(outcome.error);
|
|
523
|
+
}
|
|
524
|
+
if (outcome.entry) cursors.sources.decisions = { entry: outcome.entry };
|
|
525
|
+
if (outcome.partial) cov.deadlineHit = true;
|
|
526
|
+
});
|
|
527
|
+
|
|
528
|
+
// --- context ---
|
|
529
|
+
await runSource('context', async (cov) => {
|
|
530
|
+
const stored = cursors.sources.context;
|
|
531
|
+
const ctxOpts = { writerId: 'ingest-context' };
|
|
532
|
+
const outcome = await consumeJsonlFile(
|
|
533
|
+
path.join(projectRoot, RETRIEVER_LANES_REL),
|
|
534
|
+
isPlainObject(stored?.entry) ? stored.entry : null,
|
|
535
|
+
cov,
|
|
536
|
+
(line) => {
|
|
537
|
+
const records = adaptContextItems(contextItemShapes(line), ctxOpts);
|
|
538
|
+
if (records.length === 0) {
|
|
539
|
+
cov.rejected += 1;
|
|
540
|
+
return 'consumed';
|
|
541
|
+
}
|
|
542
|
+
for (const record of records) {
|
|
543
|
+
emitCounted(recorder, cov, record);
|
|
544
|
+
}
|
|
545
|
+
return 'consumed';
|
|
546
|
+
},
|
|
547
|
+
expired,
|
|
548
|
+
);
|
|
549
|
+
if (outcome.error) {
|
|
550
|
+
degraded = true;
|
|
551
|
+
(cov.errors ??= []).push(outcome.error);
|
|
552
|
+
}
|
|
553
|
+
if (outcome.entry) cursors.sources.context = { entry: outcome.entry };
|
|
554
|
+
if (outcome.partial) cov.deadlineHit = true;
|
|
555
|
+
});
|
|
556
|
+
|
|
557
|
+
// Cursor persist is the atomic commit point (tmp+rename). A failed write
|
|
558
|
+
// leaves the LAST good cursor on disk — the next collect re-emits those
|
|
559
|
+
// rows as bounded duplicates rather than claiming delivery it never made.
|
|
560
|
+
let cursorError = null;
|
|
561
|
+
try {
|
|
562
|
+
await saveCursors(projectRoot, cursors);
|
|
563
|
+
} catch (error) {
|
|
564
|
+
cursorError = error?.code ?? 'cursor_persist_failed';
|
|
565
|
+
degraded = true;
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
return {
|
|
569
|
+
ok: true,
|
|
570
|
+
status: degraded ? 'degraded' : partial ? 'partial' : 'complete',
|
|
571
|
+
coverage,
|
|
572
|
+
last_cursors: cursors,
|
|
573
|
+
...(cursorReset ? { cursor_reset: true } : {}),
|
|
574
|
+
...(cursorError ? { cursor_error: cursorError } : {}),
|
|
575
|
+
};
|
|
576
|
+
}
|