@descryy/runtime-orchestrator 0.4.8 → 0.4.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +6 -6
- package/dist/attach-to-running-jvm-process.d.ts +0 -86
- package/dist/attach-to-running-jvm-process.d.ts.map +0 -1
- package/dist/attach-to-running-jvm-process.js +0 -408
- package/dist/attach-to-running-jvm-process.js.map +0 -1
- package/dist/attach-to-running-node-process.d.ts +0 -80
- package/dist/attach-to-running-node-process.d.ts.map +0 -1
- package/dist/attach-to-running-node-process.js +0 -290
- package/dist/attach-to-running-node-process.js.map +0 -1
- package/dist/attach-to-running-process.d.ts +0 -83
- package/dist/attach-to-running-process.d.ts.map +0 -1
- package/dist/attach-to-running-process.js +0 -86
- package/dist/attach-to-running-process.js.map +0 -1
- package/dist/cdp-client.d.ts +0 -27
- package/dist/cdp-client.d.ts.map +0 -1
- package/dist/cdp-client.js +0 -98
- package/dist/cdp-client.js.map +0 -1
- package/dist/cgroup-partial-restriction.d.ts +0 -80
- package/dist/cgroup-partial-restriction.d.ts.map +0 -1
- package/dist/cgroup-partial-restriction.js +0 -189
- package/dist/cgroup-partial-restriction.js.map +0 -1
- package/dist/collector-version.d.ts +0 -7
- package/dist/collector-version.d.ts.map +0 -1
- package/dist/collector-version.js +0 -9
- package/dist/collector-version.js.map +0 -1
- package/dist/index.d.ts +0 -19
- package/dist/index.d.ts.map +0 -1
- package/dist/index.js +0 -10
- package/dist/index.js.map +0 -1
- package/dist/instrumented-execution.d.ts +0 -135
- package/dist/instrumented-execution.d.ts.map +0 -1
- package/dist/instrumented-execution.js +0 -696
- package/dist/instrumented-execution.js.map +0 -1
- package/dist/jvm-agent/build.d.ts +0 -48
- package/dist/jvm-agent/build.d.ts.map +0 -1
- package/dist/jvm-agent/build.js +0 -129
- package/dist/jvm-agent/build.js.map +0 -1
- package/dist/polling-output-source.d.ts +0 -43
- package/dist/polling-output-source.d.ts.map +0 -1
- package/dist/polling-output-source.js +0 -64
- package/dist/polling-output-source.js.map +0 -1
- package/dist/profile-backend-observation.d.ts +0 -67
- package/dist/profile-backend-observation.d.ts.map +0 -1
- package/dist/profile-backend-observation.js +0 -107
- package/dist/profile-backend-observation.js.map +0 -1
- package/dist/respawn-and-supervise.d.ts +0 -86
- package/dist/respawn-and-supervise.d.ts.map +0 -1
- package/dist/respawn-and-supervise.js +0 -171
- package/dist/respawn-and-supervise.js.map +0 -1
|
@@ -1,696 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* The assembly wiring the runtime's pieces together in production code (previously
|
|
3
|
-
* only connected ad hoc by tests). Runs an `ExecutionController` over a `services`
|
|
4
|
-
* configuration; wraps each spawned process in a live `ProcessOutputSource`
|
|
5
|
-
* (polls the buffer, yields each line once); builds a `CollectorContext` with a
|
|
6
|
-
* real `resolveSourceRoot` closed over the execution's own processes; starts,
|
|
7
|
-
* drains and stops the adapter's collectors; writes every emitted item through
|
|
8
|
-
* `EvidenceStore` so redaction runs (plan §28).
|
|
9
|
-
*
|
|
10
|
-
* Deliberately does not decide whether a finding exists, correlate to graph
|
|
11
|
-
* nodes, or drive a browser — separate stages, separate contracts.
|
|
12
|
-
*/
|
|
13
|
-
import { delimiter } from "node:path";
|
|
14
|
-
import { resolveServiceRootForOrigin } from "@descryy/runtime-contracts";
|
|
15
|
-
import { createInboundProxy } from "@descryy/runtime-backend-observation";
|
|
16
|
-
import { createExternalRequestCollector } from "@descryy/runtime-external-service-observation";
|
|
17
|
-
import { ExecutionController, checkDependencyVersions, checkEnvironmentVersion, runPreflight, portFreeCheck, readinessTargetCheck, createProcessOutputReader, } from "@descryy/runtime-controller";
|
|
18
|
-
import { EvidenceStore } from "@descryy/runtime-evidence-store";
|
|
19
|
-
import { createPollingProcessOutputSource } from "./polling-output-source.js";
|
|
20
|
-
import { COLLECTOR_VERSION } from "./collector-version.js";
|
|
21
|
-
/**
|
|
22
|
-
* Every spawned service's captured output, read off the controller at the
|
|
23
|
-
* moment the result is built. Keyed by service name, which is also the only
|
|
24
|
-
* thing that can be joined back to `ExecutionConfiguration.services`; a handle
|
|
25
|
-
* with no service name has nothing to key on and is skipped.
|
|
26
|
-
*/
|
|
27
|
-
function captureServiceOutput(controller) {
|
|
28
|
-
const output = {};
|
|
29
|
-
for (const handle of controller.execution.processes) {
|
|
30
|
-
if (handle.serviceName === null)
|
|
31
|
-
continue;
|
|
32
|
-
output[handle.serviceName] = controller.readProcessOutput(handle.serviceName);
|
|
33
|
-
}
|
|
34
|
-
return output;
|
|
35
|
-
}
|
|
36
|
-
/**
|
|
37
|
-
* Builds the origin → source-root lookup a collector reads through
|
|
38
|
-
* `CollectorContext.resolveSourceRoot`. Closes over the execution's own
|
|
39
|
-
* processes, which is what makes the port→service join possible at all — under
|
|
40
|
-
* ephemeral ports only the running process knows its own port.
|
|
41
|
-
*/
|
|
42
|
-
export function createSourceRootResolver(processes, configuration) {
|
|
43
|
-
return (origin) => resolveServiceRootForOrigin(origin, { processes, configuration });
|
|
44
|
-
}
|
|
45
|
-
/**
|
|
46
|
-
* Rewrites every spawnable service so it launches with outbound
|
|
47
|
-
* instrumentation in place.
|
|
48
|
-
*
|
|
49
|
-
* The declaration comes from the adapter and names no language; this function
|
|
50
|
-
* applies it and also names none. `attach`-mode services are skipped: their
|
|
51
|
-
* process is already running, so there is no launch left to influence, and
|
|
52
|
-
* pretending otherwise would claim instrumentation that is not there.
|
|
53
|
-
*/
|
|
54
|
-
export function applyOutboundLaunch(execution, launch) {
|
|
55
|
-
const services = {};
|
|
56
|
-
for (const [name, service] of Object.entries(execution.configuration.services)) {
|
|
57
|
-
if (service.command === undefined) {
|
|
58
|
-
services[name] = service;
|
|
59
|
-
continue;
|
|
60
|
-
}
|
|
61
|
-
const prepend = new Set(launch.prependToExisting ?? []);
|
|
62
|
-
const env = { ...service.env };
|
|
63
|
-
for (const [key, value] of Object.entries(launch.env ?? {})) {
|
|
64
|
-
// extend, don't replace, a variable the target relies on (PYTHONPATH is the
|
|
65
|
-
// live case) -- replacing it breaks the app's own module resolution
|
|
66
|
-
const existing = prepend.has(key) ? (service.env?.[key] ?? process.env[key] ?? "") : "";
|
|
67
|
-
env[key] = existing === "" ? value : `${value}${delimiter}${existing}`;
|
|
68
|
-
}
|
|
69
|
-
services[name] = {
|
|
70
|
-
...service,
|
|
71
|
-
...(Object.keys(env).length === 0 ? {} : { env }),
|
|
72
|
-
...(launch.interpreterArgs === undefined || launch.interpreterArgs.length === 0
|
|
73
|
-
? {}
|
|
74
|
-
: { interpreterArgs: [...(service.interpreterArgs ?? []), ...launch.interpreterArgs] }),
|
|
75
|
-
};
|
|
76
|
-
}
|
|
77
|
-
return { ...execution, configuration: { ...execution.configuration, services } };
|
|
78
|
-
}
|
|
79
|
-
/**
|
|
80
|
-
* RT-263: the checks this run's own preflight can answer before anything is spawned or
|
|
81
|
-
* attached to — see `@descryy/runtime-controller`'s preflight.ts for the engine. Scoped
|
|
82
|
-
* deliberately narrow: a command-mode (boot) service's declared readiness target is not
|
|
83
|
-
* yet reachable by definition (nothing has started), so only its configured fixed port
|
|
84
|
-
* being free is checked, the same precondition RT-024's ephemeral-port default exists to
|
|
85
|
-
* avoid needing at all. An attach-mode service is expected to already be running, so its
|
|
86
|
-
* declared readiness checks are probed directly, single-shot — filtered to `http`/
|
|
87
|
-
* `tcp-port`, the only kinds answerable without a live process's own output (a
|
|
88
|
-
* `log-pattern` or `command` check would misreport "broken" before this run has started
|
|
89
|
-
* anything to produce output for it to read).
|
|
90
|
-
*/
|
|
91
|
-
function buildPreflightChecks(configuration, runOptions) {
|
|
92
|
-
const checks = [];
|
|
93
|
-
for (const [name, service] of Object.entries(configuration.services)) {
|
|
94
|
-
if (service.attach !== undefined) {
|
|
95
|
-
const readinessSpec = runOptions.readiness[name];
|
|
96
|
-
if (readinessSpec === undefined)
|
|
97
|
-
continue;
|
|
98
|
-
const declared = readinessSpec.checks({ port: service.port ?? 0, outputReader: createProcessOutputReader(() => "") });
|
|
99
|
-
const singleShotTestable = declared.filter((check) => check.kind === "http" || check.kind === "tcp-port");
|
|
100
|
-
if (singleShotTestable.length > 0) {
|
|
101
|
-
checks.push(readinessTargetCheck(`service "${name}" (attach) declared readiness target`, singleShotTestable));
|
|
102
|
-
}
|
|
103
|
-
}
|
|
104
|
-
else if (service.command !== undefined && service.port !== undefined) {
|
|
105
|
-
checks.push(portFreeCheck("127.0.0.1", service.port, `service "${name}"'s configured port ${String(service.port)} is free`));
|
|
106
|
-
}
|
|
107
|
-
}
|
|
108
|
-
return checks;
|
|
109
|
-
}
|
|
110
|
-
export async function runInstrumentedExecution(input) {
|
|
111
|
-
// outbound observation needs two things: a collector to read the marker lines,
|
|
112
|
-
// and a launch that installs client instrumentation before app code can capture
|
|
113
|
-
// an unpatched client. The launch is adapter-declared and language-specific; an
|
|
114
|
-
// adapter that declares nothing gets neither.
|
|
115
|
-
//
|
|
116
|
-
// budget clock starts here, not at controller.run() -- marginally earlier than
|
|
117
|
-
// the controller's own timer, so a caller never waits past its declared bound
|
|
118
|
-
const budgetStartedAt = Date.now();
|
|
119
|
-
const outboundLaunch = input.adapter.outboundHttpLaunch?.() ?? null;
|
|
120
|
-
const launchedExecution = outboundLaunch === null ? input.execution : applyOutboundLaunch(input.execution, outboundLaunch);
|
|
121
|
-
const configuration = launchedExecution.configuration;
|
|
122
|
-
// closure, not repeated per-return: there are four returns, and one quietly
|
|
123
|
-
// reporting a different number is the bug this field exists to catch
|
|
124
|
-
const budgetOutcome = (observedForMs) => {
|
|
125
|
-
const elapsedMs = Date.now() - budgetStartedAt;
|
|
126
|
-
return {
|
|
127
|
-
declaredMs: configuration.timeoutMs,
|
|
128
|
-
elapsedMs,
|
|
129
|
-
requestedObservationMs: input.observeForMs,
|
|
130
|
-
observedForMs,
|
|
131
|
-
exceeded: elapsedMs > configuration.timeoutMs,
|
|
132
|
-
};
|
|
133
|
-
};
|
|
134
|
-
const controller = ExecutionController.create(launchedExecution);
|
|
135
|
-
// B7: the execution row, minted here and nowhere else in production.
|
|
136
|
-
// `EvidenceStore.recordExecution` was complete and tested since the
|
|
137
|
-
// `executions` table was added -- schema, redaction, environment capture,
|
|
138
|
-
// upsert -- but this repo's and descry-core's own sources together held
|
|
139
|
-
// exactly two callers of it outside its own tests, both in
|
|
140
|
-
// evidence-store's test suite. Every evidence row this run is about to
|
|
141
|
-
// write carries `controller.execution.executionId`, so the parent row
|
|
142
|
-
// must exist before the first of them does; `controller.execution` is
|
|
143
|
-
// real the moment `create()` returns (it mints the id synchronously),
|
|
144
|
-
// which is before `controller.run()` -- and therefore before this
|
|
145
|
-
// function's own first `store.write()`, in the catch block just below.
|
|
146
|
-
input.store.recordExecution({
|
|
147
|
-
executionId: controller.execution.executionId,
|
|
148
|
-
application: controller.execution.application,
|
|
149
|
-
repository: controller.execution.repository,
|
|
150
|
-
commit: controller.execution.commit,
|
|
151
|
-
configuration: controller.execution.configuration,
|
|
152
|
-
});
|
|
153
|
-
// RT-263: read-only ready check, before anything is spawned or attached to. Reuses
|
|
154
|
-
// exactly the preconditions the run below depends on, so a broken setup is named with
|
|
155
|
-
// its fix in seconds rather than after the full readiness wait (up to each service's
|
|
156
|
-
// own `timeoutMs`) has already failed. Never mutates anything -- see preflight.ts.
|
|
157
|
-
const preflight = await runPreflight(buildPreflightChecks(configuration, input.runOptions));
|
|
158
|
-
if (preflight.status === "broken") {
|
|
159
|
-
return {
|
|
160
|
-
execution: controller.execution,
|
|
161
|
-
evidence: [],
|
|
162
|
-
collectorCapabilities: [],
|
|
163
|
-
validationError: "preflight failed before this run started anything: " +
|
|
164
|
-
preflight.checks
|
|
165
|
-
.filter((check) => check.required && !check.ok)
|
|
166
|
-
.map((check) => `${check.label} — ${check.message}`)
|
|
167
|
-
.join("; "),
|
|
168
|
-
budget: budgetOutcome(0),
|
|
169
|
-
// Refused before spawning, so nothing can have been said -- empty is the true
|
|
170
|
-
// answer here, not a default standing in for one.
|
|
171
|
-
serviceOutput: {},
|
|
172
|
-
};
|
|
173
|
-
}
|
|
174
|
-
// RT-219: controller.run()'s one throw path (spawnProcess catch/rethrow, after
|
|
175
|
-
// recording the failure in serviceStartResults) used to reject this whole
|
|
176
|
-
// function uncaught -- zero evidence trail. Caught here and written through
|
|
177
|
-
// EvidenceStore tagged "harness", same "resolves normally" contract RT-219
|
|
178
|
-
// established for collector.start(). controller.execution/serviceStartResults
|
|
179
|
-
// are both already real at this point (run() mutates state before it throws).
|
|
180
|
-
let execution;
|
|
181
|
-
try {
|
|
182
|
-
execution = await controller.run(input.runOptions);
|
|
183
|
-
}
|
|
184
|
-
catch (error) {
|
|
185
|
-
const detail = error instanceof Error ? `${error.name}: ${error.message}` : String(error);
|
|
186
|
-
const harnessEvidence = input.store.write({
|
|
187
|
-
executionId: controller.execution.executionId,
|
|
188
|
-
timestamp: new Date().toISOString(),
|
|
189
|
-
source: "harness",
|
|
190
|
-
service: null,
|
|
191
|
-
process: null,
|
|
192
|
-
eventType: "COLLECTOR_ERROR",
|
|
193
|
-
payload: {
|
|
194
|
-
raw: `ExecutionController.run() threw instead of resolving: ${detail}`,
|
|
195
|
-
error: detail,
|
|
196
|
-
},
|
|
197
|
-
traceId: null,
|
|
198
|
-
requestId: null,
|
|
199
|
-
correlationId: null,
|
|
200
|
-
graphNodeId: null,
|
|
201
|
-
sourceLocation: null,
|
|
202
|
-
stackTrace: null,
|
|
203
|
-
confidence: 1,
|
|
204
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
205
|
-
});
|
|
206
|
-
return {
|
|
207
|
-
execution: controller.execution,
|
|
208
|
-
evidence: [harnessEvidence],
|
|
209
|
-
collectorCapabilities: [],
|
|
210
|
-
validationError: `ExecutionController.run() threw: ${detail}`,
|
|
211
|
-
budget: budgetOutcome(0),
|
|
212
|
-
// The throw path is where output matters most: whatever the service
|
|
213
|
-
// printed before the run came apart is the only account of it there is.
|
|
214
|
-
serviceOutput: captureServiceOutput(controller),
|
|
215
|
-
};
|
|
216
|
-
}
|
|
217
|
-
if (controller.validationError !== null) {
|
|
218
|
-
return {
|
|
219
|
-
execution,
|
|
220
|
-
evidence: [],
|
|
221
|
-
collectorCapabilities: [],
|
|
222
|
-
validationError: controller.validationError,
|
|
223
|
-
budget: budgetOutcome(0),
|
|
224
|
-
// Refused before spawning, so nothing can have been said. Empty here is
|
|
225
|
-
// the true answer, not a default standing in for one.
|
|
226
|
-
serviceOutput: {},
|
|
227
|
-
};
|
|
228
|
-
}
|
|
229
|
-
// a handle with no serviceName can't be attributed to a config entry -- skipped
|
|
230
|
-
// declared before the sources close over it: the flip lets stop() finish rather
|
|
231
|
-
// than wait forever on a source still watching a live buffer
|
|
232
|
-
let finished = false;
|
|
233
|
-
const sources = [];
|
|
234
|
-
// second, independent source per process -- NOT shared with the backend
|
|
235
|
-
// collector's: createPollingProcessOutputSource yields each line exactly once,
|
|
236
|
-
// so sharing means two collectors race for it. Measured: sharing made the
|
|
237
|
-
// backend log collector eat every DESCRY_EXTERNAL_REQUEST line and the outbound
|
|
238
|
-
// collector report nothing, reading as "no outbound calls".
|
|
239
|
-
const namedSources = [];
|
|
240
|
-
for (const handle of execution.processes) {
|
|
241
|
-
if (handle.serviceName === null)
|
|
242
|
-
continue;
|
|
243
|
-
const serviceName = handle.serviceName;
|
|
244
|
-
// stamped here, at construction -- by the time evidence reaches
|
|
245
|
-
// correlateExecution the handle is out of scope and this association is gone
|
|
246
|
-
sources.push(createPollingProcessOutputSource({
|
|
247
|
-
processId: handle.processId,
|
|
248
|
-
serviceName,
|
|
249
|
-
readOutput: () => controller.readProcessOutput(serviceName),
|
|
250
|
-
isFinished: () => finished,
|
|
251
|
-
}));
|
|
252
|
-
namedSources.push({
|
|
253
|
-
serviceName,
|
|
254
|
-
source: createPollingProcessOutputSource({
|
|
255
|
-
processId: handle.processId,
|
|
256
|
-
serviceName,
|
|
257
|
-
readOutput: () => controller.readProcessOutput(serviceName),
|
|
258
|
-
isFinished: () => finished,
|
|
259
|
-
}),
|
|
260
|
-
});
|
|
261
|
-
}
|
|
262
|
-
const backendCollectors = input.adapter.createBackendCollectors(sources);
|
|
263
|
-
// only started when the launch above actually installed instrumentation --
|
|
264
|
-
// otherwise it would report "no outbound calls" for a service making them,
|
|
265
|
-
// worse than declining to look. createExternalRequestCollector names no
|
|
266
|
-
// language itself; the adapter's stack parser is the only language-specific bit.
|
|
267
|
-
const outboundCollectors = outboundLaunch === null
|
|
268
|
-
? []
|
|
269
|
-
: namedSources.map(({ source, serviceName }) => createExternalRequestCollector({
|
|
270
|
-
source,
|
|
271
|
-
service: serviceName,
|
|
272
|
-
stackTraceParser: input.adapter.stackTraceParser,
|
|
273
|
-
}));
|
|
274
|
-
const emitted = [];
|
|
275
|
-
// RT-261: bounds on the evidence THIS run accumulates in memory, on top of
|
|
276
|
-
// (not instead of) each collector's own bounds -- a noisy app that clears
|
|
277
|
-
// every per-collector cap by fanning across many sources could still grow
|
|
278
|
-
// this array without limit. Starting values, not measurements.
|
|
279
|
-
//
|
|
280
|
-
// Per-source: caps how many events any single `source` (roughly, one
|
|
281
|
-
// collector's worth) contributes before the rest are dropped -- disclosed
|
|
282
|
-
// once per source, not once per dropped event, so the disclosure itself
|
|
283
|
-
// can't become the thing that blows the budget.
|
|
284
|
-
const MAX_EVENTS_PER_SOURCE = 10_000;
|
|
285
|
-
// Whole-run: caps total evidence payload bytes across every source. Past
|
|
286
|
-
// it, a payload that would grow the total further is kept as a summary
|
|
287
|
-
// instead of dropped outright -- the row (and its eventType/source/ids)
|
|
288
|
-
// still exists, only its body is what got cut.
|
|
289
|
-
const MAX_RUN_EVIDENCE_BYTES = 50 * 1024 * 1024;
|
|
290
|
-
const eventCountBySource = new Map();
|
|
291
|
-
const sourcesAtCap = new Set();
|
|
292
|
-
let totalEvidenceBytes = 0;
|
|
293
|
-
let runCapDisclosed = false;
|
|
294
|
-
function approximatePayloadBytes(payload) {
|
|
295
|
-
try {
|
|
296
|
-
return Buffer.byteLength(JSON.stringify(payload ?? {}), "utf8");
|
|
297
|
-
}
|
|
298
|
-
catch {
|
|
299
|
-
// Circular or otherwise unstringifiable -- treat as unknown-but-nonzero
|
|
300
|
-
// rather than crash the collector that produced it.
|
|
301
|
-
return 0;
|
|
302
|
-
}
|
|
303
|
-
}
|
|
304
|
-
function harnessDisclosure(raw, collectorId, error) {
|
|
305
|
-
return input.store.write({
|
|
306
|
-
executionId: execution.executionId,
|
|
307
|
-
timestamp: new Date().toISOString(),
|
|
308
|
-
source: "harness",
|
|
309
|
-
service: null,
|
|
310
|
-
process: null,
|
|
311
|
-
eventType: "COLLECTOR_ERROR",
|
|
312
|
-
payload: error === undefined ? { raw, collectorId } : { raw, collectorId, error },
|
|
313
|
-
traceId: null,
|
|
314
|
-
requestId: null,
|
|
315
|
-
correlationId: null,
|
|
316
|
-
graphNodeId: null,
|
|
317
|
-
sourceLocation: null,
|
|
318
|
-
stackTrace: null,
|
|
319
|
-
confidence: 1,
|
|
320
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
321
|
-
});
|
|
322
|
-
}
|
|
323
|
-
// dependency/environment version-mismatch checks (DEC-NEXT-fault-layer-empirical-
|
|
324
|
-
// confirmation.md): tagged "dependency-check"/"environment-check" rather than the
|
|
325
|
-
// service's own source, so a reader can tell version drift from an app
|
|
326
|
-
// observation. Run once per service (own cwd/lockfile/engines.node),
|
|
327
|
-
// unconditionally, but only emitted as evidence on an actual mismatch.
|
|
328
|
-
//
|
|
329
|
-
// observedNodeVersion is process.version captured once during controller.run();
|
|
330
|
-
// never re-probed here.
|
|
331
|
-
const observedNodeVersion = execution.environmentMetadata?.nodeVersion ?? process.version;
|
|
332
|
-
for (const handle of execution.processes) {
|
|
333
|
-
if (handle.serviceName === null)
|
|
334
|
-
continue;
|
|
335
|
-
const serviceConfig = configuration.services[handle.serviceName];
|
|
336
|
-
if (serviceConfig === undefined)
|
|
337
|
-
continue;
|
|
338
|
-
const dependencyResult = checkDependencyVersions(serviceConfig.cwd);
|
|
339
|
-
for (const mismatch of dependencyResult.mismatches) {
|
|
340
|
-
emitted.push(input.store.write({
|
|
341
|
-
executionId: execution.executionId,
|
|
342
|
-
timestamp: new Date().toISOString(),
|
|
343
|
-
source: "dependency-check",
|
|
344
|
-
service: handle.serviceName,
|
|
345
|
-
process: handle.processId,
|
|
346
|
-
eventType: "DEPENDENCY_VERSION_MISMATCH",
|
|
347
|
-
payload: {
|
|
348
|
-
raw: `dependency "${mismatch.name}": lockfile declares ${mismatch.declaredVersion}, resolved ${mismatch.resolvedVersion}`,
|
|
349
|
-
name: mismatch.name,
|
|
350
|
-
declaredVersion: mismatch.declaredVersion,
|
|
351
|
-
resolvedVersion: mismatch.resolvedVersion,
|
|
352
|
-
},
|
|
353
|
-
traceId: null,
|
|
354
|
-
requestId: null,
|
|
355
|
-
correlationId: null,
|
|
356
|
-
graphNodeId: null,
|
|
357
|
-
sourceLocation: null,
|
|
358
|
-
stackTrace: null,
|
|
359
|
-
confidence: 1,
|
|
360
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
361
|
-
}));
|
|
362
|
-
}
|
|
363
|
-
const environmentResult = checkEnvironmentVersion(serviceConfig.cwd, observedNodeVersion);
|
|
364
|
-
if (environmentResult.mismatch) {
|
|
365
|
-
emitted.push(input.store.write({
|
|
366
|
-
executionId: execution.executionId,
|
|
367
|
-
timestamp: new Date().toISOString(),
|
|
368
|
-
source: "environment-check",
|
|
369
|
-
service: handle.serviceName,
|
|
370
|
-
process: handle.processId,
|
|
371
|
-
eventType: "ENVIRONMENT_VERSION_MISMATCH",
|
|
372
|
-
payload: {
|
|
373
|
-
raw: `declared node version ${environmentResult.declaredVersion ?? "unknown"} (${environmentResult.declaredSource ?? "unknown"}), observed ${environmentResult.observedVersion}`,
|
|
374
|
-
declaredVersion: environmentResult.declaredVersion,
|
|
375
|
-
declaredSource: environmentResult.declaredSource,
|
|
376
|
-
observedVersion: environmentResult.observedVersion,
|
|
377
|
-
},
|
|
378
|
-
traceId: null,
|
|
379
|
-
requestId: null,
|
|
380
|
-
correlationId: null,
|
|
381
|
-
graphNodeId: null,
|
|
382
|
-
sourceLocation: null,
|
|
383
|
-
stackTrace: null,
|
|
384
|
-
confidence: 1,
|
|
385
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
386
|
-
}));
|
|
387
|
-
}
|
|
388
|
-
}
|
|
389
|
-
// opt-in InboundProxy wiring (RT-192, DEC-271), see frontWithInboundProxy's doc.
|
|
390
|
-
// Built after controller.run() returns, so readiness already exists in
|
|
391
|
-
// serviceStartResults -- this loop reads that answer, never re-derives it.
|
|
392
|
-
const proxies = [];
|
|
393
|
-
if (input.frontWithInboundProxy !== undefined) {
|
|
394
|
-
for (const handle of execution.processes) {
|
|
395
|
-
const serviceName = handle.serviceName;
|
|
396
|
-
if (serviceName === null)
|
|
397
|
-
continue;
|
|
398
|
-
if (input.frontWithInboundProxy[serviceName] !== true)
|
|
399
|
-
continue;
|
|
400
|
-
const readinessOutcome = controller.serviceStartResults.find((result) => result.serviceName === serviceName && result.stage === "readiness");
|
|
401
|
-
if (readinessOutcome?.succeeded !== true || handle.port === null) {
|
|
402
|
-
// requested but never proven live -- reported as COLLECTOR_ERROR, not a
|
|
403
|
-
// silent no-proxy fallback (plan §17: "not observed" vs "nothing there")
|
|
404
|
-
emitted.push(input.store.write({
|
|
405
|
-
executionId: execution.executionId,
|
|
406
|
-
timestamp: new Date().toISOString(),
|
|
407
|
-
source: "backend-process",
|
|
408
|
-
service: serviceName,
|
|
409
|
-
process: handle.processId,
|
|
410
|
-
eventType: "COLLECTOR_ERROR",
|
|
411
|
-
payload: {
|
|
412
|
-
raw: `service "${serviceName}" was flagged frontWithInboundProxy but its own readiness never succeeded -- never fronting a process that was not proven live`,
|
|
413
|
-
collectorId: `inbound-proxy:${handle.processId}`,
|
|
414
|
-
},
|
|
415
|
-
traceId: null,
|
|
416
|
-
requestId: null,
|
|
417
|
-
correlationId: null,
|
|
418
|
-
graphNodeId: null,
|
|
419
|
-
sourceLocation: null,
|
|
420
|
-
stackTrace: null,
|
|
421
|
-
confidence: 1,
|
|
422
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
423
|
-
}));
|
|
424
|
-
continue;
|
|
425
|
-
}
|
|
426
|
-
proxies.push({
|
|
427
|
-
serviceName,
|
|
428
|
-
proxy: createInboundProxy({
|
|
429
|
-
targetHost: "127.0.0.1",
|
|
430
|
-
targetPort: handle.port,
|
|
431
|
-
processId: handle.processId,
|
|
432
|
-
service: serviceName,
|
|
433
|
-
}),
|
|
434
|
-
});
|
|
435
|
-
}
|
|
436
|
-
}
|
|
437
|
-
const collectors = [...backendCollectors, ...outboundCollectors, ...proxies.map((entry) => entry.proxy.collector)];
|
|
438
|
-
const context = {
|
|
439
|
-
executionId: execution.executionId,
|
|
440
|
-
configuration,
|
|
441
|
-
// redaction runs here, on the way in -- EvidenceStore.write() is the only
|
|
442
|
-
// path that sets redactionStatus (plan §28)
|
|
443
|
-
//
|
|
444
|
-
// RT-261: bounded before it reaches the store, not after -- see the
|
|
445
|
-
// MAX_EVENTS_PER_SOURCE/MAX_RUN_EVIDENCE_BYTES block above this function.
|
|
446
|
-
emit: (evidenceInput) => {
|
|
447
|
-
const { source } = evidenceInput;
|
|
448
|
-
const countSoFar = eventCountBySource.get(source) ?? 0;
|
|
449
|
-
if (countSoFar >= MAX_EVENTS_PER_SOURCE) {
|
|
450
|
-
if (!sourcesAtCap.has(source)) {
|
|
451
|
-
sourcesAtCap.add(source);
|
|
452
|
-
emitted.push(harnessDisclosure(`source "${source}" reached this run's ${String(MAX_EVENTS_PER_SOURCE)}-event cap -- further events from it are dropped ` +
|
|
453
|
-
"for the rest of this run, disclosed once here rather than per dropped event.", source));
|
|
454
|
-
}
|
|
455
|
-
return;
|
|
456
|
-
}
|
|
457
|
-
eventCountBySource.set(source, countSoFar + 1);
|
|
458
|
-
const payloadBytes = approximatePayloadBytes(evidenceInput.payload);
|
|
459
|
-
if (totalEvidenceBytes + payloadBytes > MAX_RUN_EVIDENCE_BYTES) {
|
|
460
|
-
if (!runCapDisclosed) {
|
|
461
|
-
runCapDisclosed = true;
|
|
462
|
-
emitted.push(harnessDisclosure(`this run's whole-execution evidence cap (${String(MAX_RUN_EVIDENCE_BYTES)} bytes) was reached -- further large ` +
|
|
463
|
-
"payloads are kept as a summary (eventType/source/ids only) instead of dropped outright.", "orchestrator:evidence-cap"));
|
|
464
|
-
}
|
|
465
|
-
const summarized = input.store.write({
|
|
466
|
-
...evidenceInput,
|
|
467
|
-
executionId: execution.executionId,
|
|
468
|
-
payload: {
|
|
469
|
-
raw: `payload dropped -- this run's whole-execution evidence cap (${String(MAX_RUN_EVIDENCE_BYTES)} bytes) was already reached`,
|
|
470
|
-
originalEventType: evidenceInput.eventType,
|
|
471
|
-
},
|
|
472
|
-
});
|
|
473
|
-
emitted.push(summarized);
|
|
474
|
-
totalEvidenceBytes += approximatePayloadBytes(summarized.payload);
|
|
475
|
-
return;
|
|
476
|
-
}
|
|
477
|
-
totalEvidenceBytes += payloadBytes;
|
|
478
|
-
emitted.push(input.store.write({ ...evidenceInput, executionId: execution.executionId }));
|
|
479
|
-
},
|
|
480
|
-
resolveSourceRoot: createSourceRootResolver(execution.processes, configuration),
|
|
481
|
-
};
|
|
482
|
-
// RT-219: collector.start()'s documented failure path resolves {available:
|
|
483
|
-
// false, reason}, handled below. A collector that instead THROWS out of
|
|
484
|
-
// start() is a different claim -- the harness's own machinery breaking, not
|
|
485
|
-
// "I looked and could not" -- so it's caught here and tagged "harness" rather
|
|
486
|
-
// than the collector's domain source, instead of aborting this Promise.all
|
|
487
|
-
// uncaught with zero evidence trail.
|
|
488
|
-
const startOutcomes = await Promise.all(collectors.map(async (collector) => {
|
|
489
|
-
try {
|
|
490
|
-
return { collector, result: await collector.start(context), threw: false };
|
|
491
|
-
}
|
|
492
|
-
catch (error) {
|
|
493
|
-
const detail = error instanceof Error ? `${error.name}: ${error.message}` : String(error);
|
|
494
|
-
emitted.push(input.store.write({
|
|
495
|
-
executionId: execution.executionId,
|
|
496
|
-
timestamp: new Date().toISOString(),
|
|
497
|
-
source: "harness",
|
|
498
|
-
service: null,
|
|
499
|
-
process: null,
|
|
500
|
-
eventType: "COLLECTOR_ERROR",
|
|
501
|
-
payload: {
|
|
502
|
-
raw: `collector "${collector.collectorId}" threw during start() instead of resolving {available:false}: ${detail}`,
|
|
503
|
-
collectorId: collector.collectorId,
|
|
504
|
-
error: detail,
|
|
505
|
-
},
|
|
506
|
-
traceId: null,
|
|
507
|
-
requestId: null,
|
|
508
|
-
correlationId: null,
|
|
509
|
-
graphNodeId: null,
|
|
510
|
-
sourceLocation: null,
|
|
511
|
-
stackTrace: null,
|
|
512
|
-
confidence: 1,
|
|
513
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
514
|
-
}));
|
|
515
|
-
return { collector, result: { available: false, reason: `start() threw: ${detail}` }, threw: true };
|
|
516
|
-
}
|
|
517
|
-
}));
|
|
518
|
-
// a thrown start() was already reported above under "harness" -- excluded here
|
|
519
|
-
// to avoid double-reporting the same failure under "backend-process"
|
|
520
|
-
const unavailable = startOutcomes.filter((entry) => !entry.threw && !entry.result.available);
|
|
521
|
-
// reported, never silently absent (plan §17: "not observed" vs "nothing there")
|
|
522
|
-
for (const entry of unavailable) {
|
|
523
|
-
emitted.push(input.store.write({
|
|
524
|
-
executionId: execution.executionId,
|
|
525
|
-
timestamp: new Date().toISOString(),
|
|
526
|
-
source: "backend-process",
|
|
527
|
-
service: null,
|
|
528
|
-
process: null,
|
|
529
|
-
eventType: "COLLECTOR_ERROR",
|
|
530
|
-
payload: {
|
|
531
|
-
raw: `collector "${entry.collector.collectorId}" reported unavailable at start: ${entry.result.available ? "" : (entry.result.reason ?? "no reason given")}`,
|
|
532
|
-
collectorId: entry.collector.collectorId,
|
|
533
|
-
},
|
|
534
|
-
traceId: null,
|
|
535
|
-
requestId: null,
|
|
536
|
-
correlationId: null,
|
|
537
|
-
graphNodeId: null,
|
|
538
|
-
sourceLocation: null,
|
|
539
|
-
stackTrace: null,
|
|
540
|
-
confidence: 1,
|
|
541
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
542
|
-
}));
|
|
543
|
-
}
|
|
544
|
-
// proxies with available:false above never reach here -- port stayed null
|
|
545
|
-
const proxyPorts = {};
|
|
546
|
-
for (const entry of proxies) {
|
|
547
|
-
const port = entry.proxy.port();
|
|
548
|
-
if (port !== null)
|
|
549
|
-
proxyPorts[entry.serviceName] = port;
|
|
550
|
-
}
|
|
551
|
-
// Wrapped because a throw above teardown used to leak the whole execution.
|
|
552
|
-
// onReady is caller code, documented not to be caught -- but "not caught" and
|
|
553
|
-
// "not cleaned up" are different promises: when onReady threw, finished stayed
|
|
554
|
-
// false, so collector.stop()/controller.stop() were skipped, sources kept
|
|
555
|
-
// polling, the spawned child was never killed, and the timer held the event
|
|
556
|
-
// loop open. Measured: rejects in 270ms, process then never exits. Exception
|
|
557
|
-
// still propagates unchanged to the caller; only the leak goes away.
|
|
558
|
-
//
|
|
559
|
-
// Found the hard way: a worktree missing node_modules made a test's onReady
|
|
560
|
-
// throw after a correctly bounded readiness wait, then the suite hung for
|
|
561
|
-
// twenty minutes.
|
|
562
|
-
let observationMs = 0;
|
|
563
|
-
let bodyFailure = null;
|
|
564
|
-
let teardownFailure = null;
|
|
565
|
-
try {
|
|
566
|
-
await input.onReady?.({ execution, proxyPorts });
|
|
567
|
-
// observation window is bounded by the declared budget, not just observeForMs
|
|
568
|
-
// -- unconditionally sleeping the full window used to leave timeoutMs bounding
|
|
569
|
-
// only the service, not the wait; measured, a 1500ms budget with a 12000ms
|
|
570
|
-
// window returned after 12288ms with zero evidence (services dead the whole
|
|
571
|
-
// remainder). Clamped rather than aborted: teardown still owns OS resources
|
|
572
|
-
// and must run; the overshoot is reported via budget.exceeded, not hidden.
|
|
573
|
-
const remainingMs = budgetStartedAt + configuration.timeoutMs - Date.now();
|
|
574
|
-
observationMs = Math.max(0, Math.min(input.observeForMs, remainingMs));
|
|
575
|
-
if (observationMs > 0) {
|
|
576
|
-
await new Promise((resolve) => setTimeout(resolve, observationMs));
|
|
577
|
-
}
|
|
578
|
-
// rule 7: cut-short window is a labelled degradation. execution.state's
|
|
579
|
-
// TIMED_OUT says the services stopped early, not that observation did --
|
|
580
|
-
// tagged "harness"/COLLECTOR_ERROR to keep "found nothing" apart from
|
|
581
|
-
// "stopped looking".
|
|
582
|
-
if (observationMs < input.observeForMs) {
|
|
583
|
-
emitted.push(input.store.write({
|
|
584
|
-
executionId: execution.executionId,
|
|
585
|
-
timestamp: new Date().toISOString(),
|
|
586
|
-
source: "harness",
|
|
587
|
-
service: null,
|
|
588
|
-
process: null,
|
|
589
|
-
eventType: "COLLECTOR_ERROR",
|
|
590
|
-
payload: {
|
|
591
|
-
raw: `the observation window was cut short by the declared whole-execution budget: ` +
|
|
592
|
-
`observed for ${String(observationMs)}ms of the ${String(input.observeForMs)}ms requested, ` +
|
|
593
|
-
`against a ${String(configuration.timeoutMs)}ms budget. Evidence from this run is partial — ` +
|
|
594
|
-
`absence of an event is not evidence the application did not produce one.`,
|
|
595
|
-
collectorId: "orchestrator:execution-budget",
|
|
596
|
-
},
|
|
597
|
-
traceId: null,
|
|
598
|
-
requestId: null,
|
|
599
|
-
correlationId: null,
|
|
600
|
-
graphNodeId: null,
|
|
601
|
-
sourceLocation: null,
|
|
602
|
-
stackTrace: null,
|
|
603
|
-
confidence: 1,
|
|
604
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
605
|
-
}));
|
|
606
|
-
}
|
|
607
|
-
}
|
|
608
|
-
catch (error) {
|
|
609
|
-
// held, not handled -- rethrown below, unchanged, once teardown has run
|
|
610
|
-
bodyFailure = { error };
|
|
611
|
-
}
|
|
612
|
-
finally {
|
|
613
|
-
// order matters: sources must be told the process is done before stop()
|
|
614
|
-
// awaits their drain, or it waits forever on a generator polling a live buffer
|
|
615
|
-
finished = true;
|
|
616
|
-
// RT-262: before this, `Promise.all` meant one collector's stop() throwing
|
|
617
|
-
// aborted the group immediately -- collectors after it in the array never
|
|
618
|
-
// got their own stop() called, and (worse) the app-process kill below sat
|
|
619
|
-
// inside the same try, so it never ran either: a single misbehaving
|
|
620
|
-
// collector both dropped other collectors' teardown AND leaked the real
|
|
621
|
-
// spawned process. `allSettled` runs every collector's stop() to
|
|
622
|
-
// completion regardless of any other's outcome; a rejection becomes a
|
|
623
|
-
// disclosed COLLECTOR_ERROR (its own evidence survives this run, it just
|
|
624
|
-
// named what went wrong) instead of an exception that skips the rest.
|
|
625
|
-
const stopOutcomes = await Promise.allSettled(collectors.map((collector) => collector.stop()));
|
|
626
|
-
for (let index = 0; index < stopOutcomes.length; index++) {
|
|
627
|
-
const outcome = stopOutcomes[index];
|
|
628
|
-
if (outcome?.status !== "rejected")
|
|
629
|
-
continue;
|
|
630
|
-
const collector = collectors[index];
|
|
631
|
-
const detail = outcome.reason instanceof Error ? `${outcome.reason.name}: ${outcome.reason.message}` : String(outcome.reason);
|
|
632
|
-
emitted.push(harnessDisclosure(`collector "${collector?.collectorId ?? "unknown"}" threw during stop(): ${detail}`, collector?.collectorId ?? "unknown", detail));
|
|
633
|
-
}
|
|
634
|
-
try {
|
|
635
|
-
// Always runs, even if one or more collectors above threw during
|
|
636
|
-
// stop() -- killing the real app process must never depend on every
|
|
637
|
-
// collector's own teardown succeeding. This is the one failure RT-262
|
|
638
|
-
// still escalates: a process Descry cannot confirm it killed is a
|
|
639
|
-
// leaked process, not a degraded observation.
|
|
640
|
-
if (observationMs < input.observeForMs) {
|
|
641
|
-
await controller.timeout();
|
|
642
|
-
}
|
|
643
|
-
else {
|
|
644
|
-
await controller.stop();
|
|
645
|
-
}
|
|
646
|
-
}
|
|
647
|
-
catch (teardownError) {
|
|
648
|
-
// held and rethrown below rather than thrown inside finally, which can
|
|
649
|
-
// discard an in-flight exception
|
|
650
|
-
if (bodyFailure === null) {
|
|
651
|
-
teardownFailure = { error: teardownError };
|
|
652
|
-
}
|
|
653
|
-
else {
|
|
654
|
-
// caller's error is the actionable one; a teardown failure must not
|
|
655
|
-
// silently replace it -- written to the evidence store instead (tagged
|
|
656
|
-
// "harness", RT-219) rather than thrown. Not in the returned array, but
|
|
657
|
-
// the store is durable so the record survives the throw.
|
|
658
|
-
input.store.write({
|
|
659
|
-
executionId: execution.executionId,
|
|
660
|
-
timestamp: new Date().toISOString(),
|
|
661
|
-
source: "harness",
|
|
662
|
-
service: null,
|
|
663
|
-
process: null,
|
|
664
|
-
eventType: "COLLECTOR_ERROR",
|
|
665
|
-
payload: {
|
|
666
|
-
raw: "teardown also failed while unwinding from an earlier failure, and is recorded rather than " +
|
|
667
|
-
`thrown so it does not mask it: ${teardownError instanceof Error ? `${teardownError.name}: ${teardownError.message}` : String(teardownError)}`,
|
|
668
|
-
collectorId: "orchestrator:teardown",
|
|
669
|
-
},
|
|
670
|
-
traceId: null,
|
|
671
|
-
requestId: null,
|
|
672
|
-
correlationId: null,
|
|
673
|
-
graphNodeId: null,
|
|
674
|
-
sourceLocation: null,
|
|
675
|
-
stackTrace: null,
|
|
676
|
-
confidence: 1,
|
|
677
|
-
collectorVersion: COLLECTOR_VERSION,
|
|
678
|
-
});
|
|
679
|
-
}
|
|
680
|
-
}
|
|
681
|
-
}
|
|
682
|
-
// rethrown untouched, same object/stack; caller's failure wins over teardown's
|
|
683
|
-
if (bodyFailure !== null)
|
|
684
|
-
throw bodyFailure.error;
|
|
685
|
-
if (teardownFailure !== null)
|
|
686
|
-
throw teardownFailure.error;
|
|
687
|
-
return {
|
|
688
|
-
execution: controller.execution,
|
|
689
|
-
evidence: emitted,
|
|
690
|
-
collectorCapabilities: collectors.map((collector) => ({ collectorId: collector.collectorId, capabilities: collector.capabilities() })),
|
|
691
|
-
validationError: null,
|
|
692
|
-
budget: budgetOutcome(observationMs),
|
|
693
|
-
serviceOutput: captureServiceOutput(controller),
|
|
694
|
-
};
|
|
695
|
-
}
|
|
696
|
-
//# sourceMappingURL=instrumented-execution.js.map
|