@descryy/runtime-orchestrator 0.4.8 → 0.4.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (49) hide show
  1. package/package.json +6 -6
  2. package/dist/attach-to-running-jvm-process.d.ts +0 -86
  3. package/dist/attach-to-running-jvm-process.d.ts.map +0 -1
  4. package/dist/attach-to-running-jvm-process.js +0 -408
  5. package/dist/attach-to-running-jvm-process.js.map +0 -1
  6. package/dist/attach-to-running-node-process.d.ts +0 -80
  7. package/dist/attach-to-running-node-process.d.ts.map +0 -1
  8. package/dist/attach-to-running-node-process.js +0 -290
  9. package/dist/attach-to-running-node-process.js.map +0 -1
  10. package/dist/attach-to-running-process.d.ts +0 -83
  11. package/dist/attach-to-running-process.d.ts.map +0 -1
  12. package/dist/attach-to-running-process.js +0 -86
  13. package/dist/attach-to-running-process.js.map +0 -1
  14. package/dist/cdp-client.d.ts +0 -27
  15. package/dist/cdp-client.d.ts.map +0 -1
  16. package/dist/cdp-client.js +0 -98
  17. package/dist/cdp-client.js.map +0 -1
  18. package/dist/cgroup-partial-restriction.d.ts +0 -80
  19. package/dist/cgroup-partial-restriction.d.ts.map +0 -1
  20. package/dist/cgroup-partial-restriction.js +0 -189
  21. package/dist/cgroup-partial-restriction.js.map +0 -1
  22. package/dist/collector-version.d.ts +0 -7
  23. package/dist/collector-version.d.ts.map +0 -1
  24. package/dist/collector-version.js +0 -9
  25. package/dist/collector-version.js.map +0 -1
  26. package/dist/index.d.ts +0 -19
  27. package/dist/index.d.ts.map +0 -1
  28. package/dist/index.js +0 -10
  29. package/dist/index.js.map +0 -1
  30. package/dist/instrumented-execution.d.ts +0 -135
  31. package/dist/instrumented-execution.d.ts.map +0 -1
  32. package/dist/instrumented-execution.js +0 -696
  33. package/dist/instrumented-execution.js.map +0 -1
  34. package/dist/jvm-agent/build.d.ts +0 -48
  35. package/dist/jvm-agent/build.d.ts.map +0 -1
  36. package/dist/jvm-agent/build.js +0 -129
  37. package/dist/jvm-agent/build.js.map +0 -1
  38. package/dist/polling-output-source.d.ts +0 -43
  39. package/dist/polling-output-source.d.ts.map +0 -1
  40. package/dist/polling-output-source.js +0 -64
  41. package/dist/polling-output-source.js.map +0 -1
  42. package/dist/profile-backend-observation.d.ts +0 -67
  43. package/dist/profile-backend-observation.d.ts.map +0 -1
  44. package/dist/profile-backend-observation.js +0 -107
  45. package/dist/profile-backend-observation.js.map +0 -1
  46. package/dist/respawn-and-supervise.d.ts +0 -86
  47. package/dist/respawn-and-supervise.d.ts.map +0 -1
  48. package/dist/respawn-and-supervise.js +0 -171
  49. package/dist/respawn-and-supervise.js.map +0 -1
@@ -1,696 +0,0 @@
1
- /**
2
- * The assembly wiring the runtime's pieces together in production code (previously
3
- * only connected ad hoc by tests). Runs an `ExecutionController` over a `services`
4
- * configuration; wraps each spawned process in a live `ProcessOutputSource`
5
- * (polls the buffer, yields each line once); builds a `CollectorContext` with a
6
- * real `resolveSourceRoot` closed over the execution's own processes; starts,
7
- * drains and stops the adapter's collectors; writes every emitted item through
8
- * `EvidenceStore` so redaction runs (plan §28).
9
- *
10
- * Deliberately does not decide whether a finding exists, correlate to graph
11
- * nodes, or drive a browser — separate stages, separate contracts.
12
- */
13
- import { delimiter } from "node:path";
14
- import { resolveServiceRootForOrigin } from "@descryy/runtime-contracts";
15
- import { createInboundProxy } from "@descryy/runtime-backend-observation";
16
- import { createExternalRequestCollector } from "@descryy/runtime-external-service-observation";
17
- import { ExecutionController, checkDependencyVersions, checkEnvironmentVersion, runPreflight, portFreeCheck, readinessTargetCheck, createProcessOutputReader, } from "@descryy/runtime-controller";
18
- import { EvidenceStore } from "@descryy/runtime-evidence-store";
19
- import { createPollingProcessOutputSource } from "./polling-output-source.js";
20
- import { COLLECTOR_VERSION } from "./collector-version.js";
21
- /**
22
- * Every spawned service's captured output, read off the controller at the
23
- * moment the result is built. Keyed by service name, which is also the only
24
- * thing that can be joined back to `ExecutionConfiguration.services`; a handle
25
- * with no service name has nothing to key on and is skipped.
26
- */
27
- function captureServiceOutput(controller) {
28
- const output = {};
29
- for (const handle of controller.execution.processes) {
30
- if (handle.serviceName === null)
31
- continue;
32
- output[handle.serviceName] = controller.readProcessOutput(handle.serviceName);
33
- }
34
- return output;
35
- }
36
- /**
37
- * Builds the origin → source-root lookup a collector reads through
38
- * `CollectorContext.resolveSourceRoot`. Closes over the execution's own
39
- * processes, which is what makes the port→service join possible at all — under
40
- * ephemeral ports only the running process knows its own port.
41
- */
42
- export function createSourceRootResolver(processes, configuration) {
43
- return (origin) => resolveServiceRootForOrigin(origin, { processes, configuration });
44
- }
45
- /**
46
- * Rewrites every spawnable service so it launches with outbound
47
- * instrumentation in place.
48
- *
49
- * The declaration comes from the adapter and names no language; this function
50
- * applies it and also names none. `attach`-mode services are skipped: their
51
- * process is already running, so there is no launch left to influence, and
52
- * pretending otherwise would claim instrumentation that is not there.
53
- */
54
- export function applyOutboundLaunch(execution, launch) {
55
- const services = {};
56
- for (const [name, service] of Object.entries(execution.configuration.services)) {
57
- if (service.command === undefined) {
58
- services[name] = service;
59
- continue;
60
- }
61
- const prepend = new Set(launch.prependToExisting ?? []);
62
- const env = { ...service.env };
63
- for (const [key, value] of Object.entries(launch.env ?? {})) {
64
- // extend, don't replace, a variable the target relies on (PYTHONPATH is the
65
- // live case) -- replacing it breaks the app's own module resolution
66
- const existing = prepend.has(key) ? (service.env?.[key] ?? process.env[key] ?? "") : "";
67
- env[key] = existing === "" ? value : `${value}${delimiter}${existing}`;
68
- }
69
- services[name] = {
70
- ...service,
71
- ...(Object.keys(env).length === 0 ? {} : { env }),
72
- ...(launch.interpreterArgs === undefined || launch.interpreterArgs.length === 0
73
- ? {}
74
- : { interpreterArgs: [...(service.interpreterArgs ?? []), ...launch.interpreterArgs] }),
75
- };
76
- }
77
- return { ...execution, configuration: { ...execution.configuration, services } };
78
- }
79
- /**
80
- * RT-263: the checks this run's own preflight can answer before anything is spawned or
81
- * attached to — see `@descryy/runtime-controller`'s preflight.ts for the engine. Scoped
82
- * deliberately narrow: a command-mode (boot) service's declared readiness target is not
83
- * yet reachable by definition (nothing has started), so only its configured fixed port
84
- * being free is checked, the same precondition RT-024's ephemeral-port default exists to
85
- * avoid needing at all. An attach-mode service is expected to already be running, so its
86
- * declared readiness checks are probed directly, single-shot — filtered to `http`/
87
- * `tcp-port`, the only kinds answerable without a live process's own output (a
88
- * `log-pattern` or `command` check would misreport "broken" before this run has started
89
- * anything to produce output for it to read).
90
- */
91
- function buildPreflightChecks(configuration, runOptions) {
92
- const checks = [];
93
- for (const [name, service] of Object.entries(configuration.services)) {
94
- if (service.attach !== undefined) {
95
- const readinessSpec = runOptions.readiness[name];
96
- if (readinessSpec === undefined)
97
- continue;
98
- const declared = readinessSpec.checks({ port: service.port ?? 0, outputReader: createProcessOutputReader(() => "") });
99
- const singleShotTestable = declared.filter((check) => check.kind === "http" || check.kind === "tcp-port");
100
- if (singleShotTestable.length > 0) {
101
- checks.push(readinessTargetCheck(`service "${name}" (attach) declared readiness target`, singleShotTestable));
102
- }
103
- }
104
- else if (service.command !== undefined && service.port !== undefined) {
105
- checks.push(portFreeCheck("127.0.0.1", service.port, `service "${name}"'s configured port ${String(service.port)} is free`));
106
- }
107
- }
108
- return checks;
109
- }
110
- export async function runInstrumentedExecution(input) {
111
- // outbound observation needs two things: a collector to read the marker lines,
112
- // and a launch that installs client instrumentation before app code can capture
113
- // an unpatched client. The launch is adapter-declared and language-specific; an
114
- // adapter that declares nothing gets neither.
115
- //
116
- // budget clock starts here, not at controller.run() -- marginally earlier than
117
- // the controller's own timer, so a caller never waits past its declared bound
118
- const budgetStartedAt = Date.now();
119
- const outboundLaunch = input.adapter.outboundHttpLaunch?.() ?? null;
120
- const launchedExecution = outboundLaunch === null ? input.execution : applyOutboundLaunch(input.execution, outboundLaunch);
121
- const configuration = launchedExecution.configuration;
122
- // closure, not repeated per-return: there are four returns, and one quietly
123
- // reporting a different number is the bug this field exists to catch
124
- const budgetOutcome = (observedForMs) => {
125
- const elapsedMs = Date.now() - budgetStartedAt;
126
- return {
127
- declaredMs: configuration.timeoutMs,
128
- elapsedMs,
129
- requestedObservationMs: input.observeForMs,
130
- observedForMs,
131
- exceeded: elapsedMs > configuration.timeoutMs,
132
- };
133
- };
134
- const controller = ExecutionController.create(launchedExecution);
135
- // B7: the execution row, minted here and nowhere else in production.
136
- // `EvidenceStore.recordExecution` was complete and tested since the
137
- // `executions` table was added -- schema, redaction, environment capture,
138
- // upsert -- but this repo's and descry-core's own sources together held
139
- // exactly two callers of it outside its own tests, both in
140
- // evidence-store's test suite. Every evidence row this run is about to
141
- // write carries `controller.execution.executionId`, so the parent row
142
- // must exist before the first of them does; `controller.execution` is
143
- // real the moment `create()` returns (it mints the id synchronously),
144
- // which is before `controller.run()` -- and therefore before this
145
- // function's own first `store.write()`, in the catch block just below.
146
- input.store.recordExecution({
147
- executionId: controller.execution.executionId,
148
- application: controller.execution.application,
149
- repository: controller.execution.repository,
150
- commit: controller.execution.commit,
151
- configuration: controller.execution.configuration,
152
- });
153
- // RT-263: read-only ready check, before anything is spawned or attached to. Reuses
154
- // exactly the preconditions the run below depends on, so a broken setup is named with
155
- // its fix in seconds rather than after the full readiness wait (up to each service's
156
- // own `timeoutMs`) has already failed. Never mutates anything -- see preflight.ts.
157
- const preflight = await runPreflight(buildPreflightChecks(configuration, input.runOptions));
158
- if (preflight.status === "broken") {
159
- return {
160
- execution: controller.execution,
161
- evidence: [],
162
- collectorCapabilities: [],
163
- validationError: "preflight failed before this run started anything: " +
164
- preflight.checks
165
- .filter((check) => check.required && !check.ok)
166
- .map((check) => `${check.label} — ${check.message}`)
167
- .join("; "),
168
- budget: budgetOutcome(0),
169
- // Refused before spawning, so nothing can have been said -- empty is the true
170
- // answer here, not a default standing in for one.
171
- serviceOutput: {},
172
- };
173
- }
174
- // RT-219: controller.run()'s one throw path (spawnProcess catch/rethrow, after
175
- // recording the failure in serviceStartResults) used to reject this whole
176
- // function uncaught -- zero evidence trail. Caught here and written through
177
- // EvidenceStore tagged "harness", same "resolves normally" contract RT-219
178
- // established for collector.start(). controller.execution/serviceStartResults
179
- // are both already real at this point (run() mutates state before it throws).
180
- let execution;
181
- try {
182
- execution = await controller.run(input.runOptions);
183
- }
184
- catch (error) {
185
- const detail = error instanceof Error ? `${error.name}: ${error.message}` : String(error);
186
- const harnessEvidence = input.store.write({
187
- executionId: controller.execution.executionId,
188
- timestamp: new Date().toISOString(),
189
- source: "harness",
190
- service: null,
191
- process: null,
192
- eventType: "COLLECTOR_ERROR",
193
- payload: {
194
- raw: `ExecutionController.run() threw instead of resolving: ${detail}`,
195
- error: detail,
196
- },
197
- traceId: null,
198
- requestId: null,
199
- correlationId: null,
200
- graphNodeId: null,
201
- sourceLocation: null,
202
- stackTrace: null,
203
- confidence: 1,
204
- collectorVersion: COLLECTOR_VERSION,
205
- });
206
- return {
207
- execution: controller.execution,
208
- evidence: [harnessEvidence],
209
- collectorCapabilities: [],
210
- validationError: `ExecutionController.run() threw: ${detail}`,
211
- budget: budgetOutcome(0),
212
- // The throw path is where output matters most: whatever the service
213
- // printed before the run came apart is the only account of it there is.
214
- serviceOutput: captureServiceOutput(controller),
215
- };
216
- }
217
- if (controller.validationError !== null) {
218
- return {
219
- execution,
220
- evidence: [],
221
- collectorCapabilities: [],
222
- validationError: controller.validationError,
223
- budget: budgetOutcome(0),
224
- // Refused before spawning, so nothing can have been said. Empty here is
225
- // the true answer, not a default standing in for one.
226
- serviceOutput: {},
227
- };
228
- }
229
- // a handle with no serviceName can't be attributed to a config entry -- skipped
230
- // declared before the sources close over it: the flip lets stop() finish rather
231
- // than wait forever on a source still watching a live buffer
232
- let finished = false;
233
- const sources = [];
234
- // second, independent source per process -- NOT shared with the backend
235
- // collector's: createPollingProcessOutputSource yields each line exactly once,
236
- // so sharing means two collectors race for it. Measured: sharing made the
237
- // backend log collector eat every DESCRY_EXTERNAL_REQUEST line and the outbound
238
- // collector report nothing, reading as "no outbound calls".
239
- const namedSources = [];
240
- for (const handle of execution.processes) {
241
- if (handle.serviceName === null)
242
- continue;
243
- const serviceName = handle.serviceName;
244
- // stamped here, at construction -- by the time evidence reaches
245
- // correlateExecution the handle is out of scope and this association is gone
246
- sources.push(createPollingProcessOutputSource({
247
- processId: handle.processId,
248
- serviceName,
249
- readOutput: () => controller.readProcessOutput(serviceName),
250
- isFinished: () => finished,
251
- }));
252
- namedSources.push({
253
- serviceName,
254
- source: createPollingProcessOutputSource({
255
- processId: handle.processId,
256
- serviceName,
257
- readOutput: () => controller.readProcessOutput(serviceName),
258
- isFinished: () => finished,
259
- }),
260
- });
261
- }
262
- const backendCollectors = input.adapter.createBackendCollectors(sources);
263
- // only started when the launch above actually installed instrumentation --
264
- // otherwise it would report "no outbound calls" for a service making them,
265
- // worse than declining to look. createExternalRequestCollector names no
266
- // language itself; the adapter's stack parser is the only language-specific bit.
267
- const outboundCollectors = outboundLaunch === null
268
- ? []
269
- : namedSources.map(({ source, serviceName }) => createExternalRequestCollector({
270
- source,
271
- service: serviceName,
272
- stackTraceParser: input.adapter.stackTraceParser,
273
- }));
274
- const emitted = [];
275
- // RT-261: bounds on the evidence THIS run accumulates in memory, on top of
276
- // (not instead of) each collector's own bounds -- a noisy app that clears
277
- // every per-collector cap by fanning across many sources could still grow
278
- // this array without limit. Starting values, not measurements.
279
- //
280
- // Per-source: caps how many events any single `source` (roughly, one
281
- // collector's worth) contributes before the rest are dropped -- disclosed
282
- // once per source, not once per dropped event, so the disclosure itself
283
- // can't become the thing that blows the budget.
284
- const MAX_EVENTS_PER_SOURCE = 10_000;
285
- // Whole-run: caps total evidence payload bytes across every source. Past
286
- // it, a payload that would grow the total further is kept as a summary
287
- // instead of dropped outright -- the row (and its eventType/source/ids)
288
- // still exists, only its body is what got cut.
289
- const MAX_RUN_EVIDENCE_BYTES = 50 * 1024 * 1024;
290
- const eventCountBySource = new Map();
291
- const sourcesAtCap = new Set();
292
- let totalEvidenceBytes = 0;
293
- let runCapDisclosed = false;
294
- function approximatePayloadBytes(payload) {
295
- try {
296
- return Buffer.byteLength(JSON.stringify(payload ?? {}), "utf8");
297
- }
298
- catch {
299
- // Circular or otherwise unstringifiable -- treat as unknown-but-nonzero
300
- // rather than crash the collector that produced it.
301
- return 0;
302
- }
303
- }
304
- function harnessDisclosure(raw, collectorId, error) {
305
- return input.store.write({
306
- executionId: execution.executionId,
307
- timestamp: new Date().toISOString(),
308
- source: "harness",
309
- service: null,
310
- process: null,
311
- eventType: "COLLECTOR_ERROR",
312
- payload: error === undefined ? { raw, collectorId } : { raw, collectorId, error },
313
- traceId: null,
314
- requestId: null,
315
- correlationId: null,
316
- graphNodeId: null,
317
- sourceLocation: null,
318
- stackTrace: null,
319
- confidence: 1,
320
- collectorVersion: COLLECTOR_VERSION,
321
- });
322
- }
323
- // dependency/environment version-mismatch checks (DEC-NEXT-fault-layer-empirical-
324
- // confirmation.md): tagged "dependency-check"/"environment-check" rather than the
325
- // service's own source, so a reader can tell version drift from an app
326
- // observation. Run once per service (own cwd/lockfile/engines.node),
327
- // unconditionally, but only emitted as evidence on an actual mismatch.
328
- //
329
- // observedNodeVersion is process.version captured once during controller.run();
330
- // never re-probed here.
331
- const observedNodeVersion = execution.environmentMetadata?.nodeVersion ?? process.version;
332
- for (const handle of execution.processes) {
333
- if (handle.serviceName === null)
334
- continue;
335
- const serviceConfig = configuration.services[handle.serviceName];
336
- if (serviceConfig === undefined)
337
- continue;
338
- const dependencyResult = checkDependencyVersions(serviceConfig.cwd);
339
- for (const mismatch of dependencyResult.mismatches) {
340
- emitted.push(input.store.write({
341
- executionId: execution.executionId,
342
- timestamp: new Date().toISOString(),
343
- source: "dependency-check",
344
- service: handle.serviceName,
345
- process: handle.processId,
346
- eventType: "DEPENDENCY_VERSION_MISMATCH",
347
- payload: {
348
- raw: `dependency "${mismatch.name}": lockfile declares ${mismatch.declaredVersion}, resolved ${mismatch.resolvedVersion}`,
349
- name: mismatch.name,
350
- declaredVersion: mismatch.declaredVersion,
351
- resolvedVersion: mismatch.resolvedVersion,
352
- },
353
- traceId: null,
354
- requestId: null,
355
- correlationId: null,
356
- graphNodeId: null,
357
- sourceLocation: null,
358
- stackTrace: null,
359
- confidence: 1,
360
- collectorVersion: COLLECTOR_VERSION,
361
- }));
362
- }
363
- const environmentResult = checkEnvironmentVersion(serviceConfig.cwd, observedNodeVersion);
364
- if (environmentResult.mismatch) {
365
- emitted.push(input.store.write({
366
- executionId: execution.executionId,
367
- timestamp: new Date().toISOString(),
368
- source: "environment-check",
369
- service: handle.serviceName,
370
- process: handle.processId,
371
- eventType: "ENVIRONMENT_VERSION_MISMATCH",
372
- payload: {
373
- raw: `declared node version ${environmentResult.declaredVersion ?? "unknown"} (${environmentResult.declaredSource ?? "unknown"}), observed ${environmentResult.observedVersion}`,
374
- declaredVersion: environmentResult.declaredVersion,
375
- declaredSource: environmentResult.declaredSource,
376
- observedVersion: environmentResult.observedVersion,
377
- },
378
- traceId: null,
379
- requestId: null,
380
- correlationId: null,
381
- graphNodeId: null,
382
- sourceLocation: null,
383
- stackTrace: null,
384
- confidence: 1,
385
- collectorVersion: COLLECTOR_VERSION,
386
- }));
387
- }
388
- }
389
- // opt-in InboundProxy wiring (RT-192, DEC-271), see frontWithInboundProxy's doc.
390
- // Built after controller.run() returns, so readiness already exists in
391
- // serviceStartResults -- this loop reads that answer, never re-derives it.
392
- const proxies = [];
393
- if (input.frontWithInboundProxy !== undefined) {
394
- for (const handle of execution.processes) {
395
- const serviceName = handle.serviceName;
396
- if (serviceName === null)
397
- continue;
398
- if (input.frontWithInboundProxy[serviceName] !== true)
399
- continue;
400
- const readinessOutcome = controller.serviceStartResults.find((result) => result.serviceName === serviceName && result.stage === "readiness");
401
- if (readinessOutcome?.succeeded !== true || handle.port === null) {
402
- // requested but never proven live -- reported as COLLECTOR_ERROR, not a
403
- // silent no-proxy fallback (plan §17: "not observed" vs "nothing there")
404
- emitted.push(input.store.write({
405
- executionId: execution.executionId,
406
- timestamp: new Date().toISOString(),
407
- source: "backend-process",
408
- service: serviceName,
409
- process: handle.processId,
410
- eventType: "COLLECTOR_ERROR",
411
- payload: {
412
- raw: `service "${serviceName}" was flagged frontWithInboundProxy but its own readiness never succeeded -- never fronting a process that was not proven live`,
413
- collectorId: `inbound-proxy:${handle.processId}`,
414
- },
415
- traceId: null,
416
- requestId: null,
417
- correlationId: null,
418
- graphNodeId: null,
419
- sourceLocation: null,
420
- stackTrace: null,
421
- confidence: 1,
422
- collectorVersion: COLLECTOR_VERSION,
423
- }));
424
- continue;
425
- }
426
- proxies.push({
427
- serviceName,
428
- proxy: createInboundProxy({
429
- targetHost: "127.0.0.1",
430
- targetPort: handle.port,
431
- processId: handle.processId,
432
- service: serviceName,
433
- }),
434
- });
435
- }
436
- }
437
- const collectors = [...backendCollectors, ...outboundCollectors, ...proxies.map((entry) => entry.proxy.collector)];
438
- const context = {
439
- executionId: execution.executionId,
440
- configuration,
441
- // redaction runs here, on the way in -- EvidenceStore.write() is the only
442
- // path that sets redactionStatus (plan §28)
443
- //
444
- // RT-261: bounded before it reaches the store, not after -- see the
445
- // MAX_EVENTS_PER_SOURCE/MAX_RUN_EVIDENCE_BYTES block above this function.
446
- emit: (evidenceInput) => {
447
- const { source } = evidenceInput;
448
- const countSoFar = eventCountBySource.get(source) ?? 0;
449
- if (countSoFar >= MAX_EVENTS_PER_SOURCE) {
450
- if (!sourcesAtCap.has(source)) {
451
- sourcesAtCap.add(source);
452
- emitted.push(harnessDisclosure(`source "${source}" reached this run's ${String(MAX_EVENTS_PER_SOURCE)}-event cap -- further events from it are dropped ` +
453
- "for the rest of this run, disclosed once here rather than per dropped event.", source));
454
- }
455
- return;
456
- }
457
- eventCountBySource.set(source, countSoFar + 1);
458
- const payloadBytes = approximatePayloadBytes(evidenceInput.payload);
459
- if (totalEvidenceBytes + payloadBytes > MAX_RUN_EVIDENCE_BYTES) {
460
- if (!runCapDisclosed) {
461
- runCapDisclosed = true;
462
- emitted.push(harnessDisclosure(`this run's whole-execution evidence cap (${String(MAX_RUN_EVIDENCE_BYTES)} bytes) was reached -- further large ` +
463
- "payloads are kept as a summary (eventType/source/ids only) instead of dropped outright.", "orchestrator:evidence-cap"));
464
- }
465
- const summarized = input.store.write({
466
- ...evidenceInput,
467
- executionId: execution.executionId,
468
- payload: {
469
- raw: `payload dropped -- this run's whole-execution evidence cap (${String(MAX_RUN_EVIDENCE_BYTES)} bytes) was already reached`,
470
- originalEventType: evidenceInput.eventType,
471
- },
472
- });
473
- emitted.push(summarized);
474
- totalEvidenceBytes += approximatePayloadBytes(summarized.payload);
475
- return;
476
- }
477
- totalEvidenceBytes += payloadBytes;
478
- emitted.push(input.store.write({ ...evidenceInput, executionId: execution.executionId }));
479
- },
480
- resolveSourceRoot: createSourceRootResolver(execution.processes, configuration),
481
- };
482
- // RT-219: collector.start()'s documented failure path resolves {available:
483
- // false, reason}, handled below. A collector that instead THROWS out of
484
- // start() is a different claim -- the harness's own machinery breaking, not
485
- // "I looked and could not" -- so it's caught here and tagged "harness" rather
486
- // than the collector's domain source, instead of aborting this Promise.all
487
- // uncaught with zero evidence trail.
488
- const startOutcomes = await Promise.all(collectors.map(async (collector) => {
489
- try {
490
- return { collector, result: await collector.start(context), threw: false };
491
- }
492
- catch (error) {
493
- const detail = error instanceof Error ? `${error.name}: ${error.message}` : String(error);
494
- emitted.push(input.store.write({
495
- executionId: execution.executionId,
496
- timestamp: new Date().toISOString(),
497
- source: "harness",
498
- service: null,
499
- process: null,
500
- eventType: "COLLECTOR_ERROR",
501
- payload: {
502
- raw: `collector "${collector.collectorId}" threw during start() instead of resolving {available:false}: ${detail}`,
503
- collectorId: collector.collectorId,
504
- error: detail,
505
- },
506
- traceId: null,
507
- requestId: null,
508
- correlationId: null,
509
- graphNodeId: null,
510
- sourceLocation: null,
511
- stackTrace: null,
512
- confidence: 1,
513
- collectorVersion: COLLECTOR_VERSION,
514
- }));
515
- return { collector, result: { available: false, reason: `start() threw: ${detail}` }, threw: true };
516
- }
517
- }));
518
- // a thrown start() was already reported above under "harness" -- excluded here
519
- // to avoid double-reporting the same failure under "backend-process"
520
- const unavailable = startOutcomes.filter((entry) => !entry.threw && !entry.result.available);
521
- // reported, never silently absent (plan §17: "not observed" vs "nothing there")
522
- for (const entry of unavailable) {
523
- emitted.push(input.store.write({
524
- executionId: execution.executionId,
525
- timestamp: new Date().toISOString(),
526
- source: "backend-process",
527
- service: null,
528
- process: null,
529
- eventType: "COLLECTOR_ERROR",
530
- payload: {
531
- raw: `collector "${entry.collector.collectorId}" reported unavailable at start: ${entry.result.available ? "" : (entry.result.reason ?? "no reason given")}`,
532
- collectorId: entry.collector.collectorId,
533
- },
534
- traceId: null,
535
- requestId: null,
536
- correlationId: null,
537
- graphNodeId: null,
538
- sourceLocation: null,
539
- stackTrace: null,
540
- confidence: 1,
541
- collectorVersion: COLLECTOR_VERSION,
542
- }));
543
- }
544
- // proxies with available:false above never reach here -- port stayed null
545
- const proxyPorts = {};
546
- for (const entry of proxies) {
547
- const port = entry.proxy.port();
548
- if (port !== null)
549
- proxyPorts[entry.serviceName] = port;
550
- }
551
- // Wrapped because a throw above teardown used to leak the whole execution.
552
- // onReady is caller code, documented not to be caught -- but "not caught" and
553
- // "not cleaned up" are different promises: when onReady threw, finished stayed
554
- // false, so collector.stop()/controller.stop() were skipped, sources kept
555
- // polling, the spawned child was never killed, and the timer held the event
556
- // loop open. Measured: rejects in 270ms, process then never exits. Exception
557
- // still propagates unchanged to the caller; only the leak goes away.
558
- //
559
- // Found the hard way: a worktree missing node_modules made a test's onReady
560
- // throw after a correctly bounded readiness wait, then the suite hung for
561
- // twenty minutes.
562
- let observationMs = 0;
563
- let bodyFailure = null;
564
- let teardownFailure = null;
565
- try {
566
- await input.onReady?.({ execution, proxyPorts });
567
- // observation window is bounded by the declared budget, not just observeForMs
568
- // -- unconditionally sleeping the full window used to leave timeoutMs bounding
569
- // only the service, not the wait; measured, a 1500ms budget with a 12000ms
570
- // window returned after 12288ms with zero evidence (services dead the whole
571
- // remainder). Clamped rather than aborted: teardown still owns OS resources
572
- // and must run; the overshoot is reported via budget.exceeded, not hidden.
573
- const remainingMs = budgetStartedAt + configuration.timeoutMs - Date.now();
574
- observationMs = Math.max(0, Math.min(input.observeForMs, remainingMs));
575
- if (observationMs > 0) {
576
- await new Promise((resolve) => setTimeout(resolve, observationMs));
577
- }
578
- // rule 7: cut-short window is a labelled degradation. execution.state's
579
- // TIMED_OUT says the services stopped early, not that observation did --
580
- // tagged "harness"/COLLECTOR_ERROR to keep "found nothing" apart from
581
- // "stopped looking".
582
- if (observationMs < input.observeForMs) {
583
- emitted.push(input.store.write({
584
- executionId: execution.executionId,
585
- timestamp: new Date().toISOString(),
586
- source: "harness",
587
- service: null,
588
- process: null,
589
- eventType: "COLLECTOR_ERROR",
590
- payload: {
591
- raw: `the observation window was cut short by the declared whole-execution budget: ` +
592
- `observed for ${String(observationMs)}ms of the ${String(input.observeForMs)}ms requested, ` +
593
- `against a ${String(configuration.timeoutMs)}ms budget. Evidence from this run is partial — ` +
594
- `absence of an event is not evidence the application did not produce one.`,
595
- collectorId: "orchestrator:execution-budget",
596
- },
597
- traceId: null,
598
- requestId: null,
599
- correlationId: null,
600
- graphNodeId: null,
601
- sourceLocation: null,
602
- stackTrace: null,
603
- confidence: 1,
604
- collectorVersion: COLLECTOR_VERSION,
605
- }));
606
- }
607
- }
608
- catch (error) {
609
- // held, not handled -- rethrown below, unchanged, once teardown has run
610
- bodyFailure = { error };
611
- }
612
- finally {
613
- // order matters: sources must be told the process is done before stop()
614
- // awaits their drain, or it waits forever on a generator polling a live buffer
615
- finished = true;
616
- // RT-262: before this, `Promise.all` meant one collector's stop() throwing
617
- // aborted the group immediately -- collectors after it in the array never
618
- // got their own stop() called, and (worse) the app-process kill below sat
619
- // inside the same try, so it never ran either: a single misbehaving
620
- // collector both dropped other collectors' teardown AND leaked the real
621
- // spawned process. `allSettled` runs every collector's stop() to
622
- // completion regardless of any other's outcome; a rejection becomes a
623
- // disclosed COLLECTOR_ERROR (its own evidence survives this run, it just
624
- // named what went wrong) instead of an exception that skips the rest.
625
- const stopOutcomes = await Promise.allSettled(collectors.map((collector) => collector.stop()));
626
- for (let index = 0; index < stopOutcomes.length; index++) {
627
- const outcome = stopOutcomes[index];
628
- if (outcome?.status !== "rejected")
629
- continue;
630
- const collector = collectors[index];
631
- const detail = outcome.reason instanceof Error ? `${outcome.reason.name}: ${outcome.reason.message}` : String(outcome.reason);
632
- emitted.push(harnessDisclosure(`collector "${collector?.collectorId ?? "unknown"}" threw during stop(): ${detail}`, collector?.collectorId ?? "unknown", detail));
633
- }
634
- try {
635
- // Always runs, even if one or more collectors above threw during
636
- // stop() -- killing the real app process must never depend on every
637
- // collector's own teardown succeeding. This is the one failure RT-262
638
- // still escalates: a process Descry cannot confirm it killed is a
639
- // leaked process, not a degraded observation.
640
- if (observationMs < input.observeForMs) {
641
- await controller.timeout();
642
- }
643
- else {
644
- await controller.stop();
645
- }
646
- }
647
- catch (teardownError) {
648
- // held and rethrown below rather than thrown inside finally, which can
649
- // discard an in-flight exception
650
- if (bodyFailure === null) {
651
- teardownFailure = { error: teardownError };
652
- }
653
- else {
654
- // caller's error is the actionable one; a teardown failure must not
655
- // silently replace it -- written to the evidence store instead (tagged
656
- // "harness", RT-219) rather than thrown. Not in the returned array, but
657
- // the store is durable so the record survives the throw.
658
- input.store.write({
659
- executionId: execution.executionId,
660
- timestamp: new Date().toISOString(),
661
- source: "harness",
662
- service: null,
663
- process: null,
664
- eventType: "COLLECTOR_ERROR",
665
- payload: {
666
- raw: "teardown also failed while unwinding from an earlier failure, and is recorded rather than " +
667
- `thrown so it does not mask it: ${teardownError instanceof Error ? `${teardownError.name}: ${teardownError.message}` : String(teardownError)}`,
668
- collectorId: "orchestrator:teardown",
669
- },
670
- traceId: null,
671
- requestId: null,
672
- correlationId: null,
673
- graphNodeId: null,
674
- sourceLocation: null,
675
- stackTrace: null,
676
- confidence: 1,
677
- collectorVersion: COLLECTOR_VERSION,
678
- });
679
- }
680
- }
681
- }
682
- // rethrown untouched, same object/stack; caller's failure wins over teardown's
683
- if (bodyFailure !== null)
684
- throw bodyFailure.error;
685
- if (teardownFailure !== null)
686
- throw teardownFailure.error;
687
- return {
688
- execution: controller.execution,
689
- evidence: emitted,
690
- collectorCapabilities: collectors.map((collector) => ({ collectorId: collector.collectorId, capabilities: collector.capabilities() })),
691
- validationError: null,
692
- budget: budgetOutcome(observationMs),
693
- serviceOutput: captureServiceOutput(controller),
694
- };
695
- }
696
- //# sourceMappingURL=instrumented-execution.js.map