@juspay/neurolink 10.12.4 → 10.12.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/browser/neurolink.min.js +420 -420
- package/dist/cli/commands/proxy.js +13 -17
- package/dist/cli/commands/proxyAnalyze.js +8 -3
- package/dist/lib/processors/archive/ArchiveProcessor.js +13 -67
- package/dist/lib/processors/archive/zipEntryReader.d.ts +51 -0
- package/dist/lib/processors/archive/zipEntryReader.js +81 -0
- package/dist/lib/processors/document/OpenDocumentProcessor.js +13 -1
- package/dist/lib/processors/document/PptxProcessor.js +58 -12
- package/dist/lib/proxy/logCleanupScheduler.d.ts +12 -0
- package/dist/lib/proxy/logCleanupScheduler.js +74 -0
- package/dist/lib/proxy/logCleanupWorkerEntry.d.ts +1 -0
- package/dist/lib/proxy/logCleanupWorkerEntry.js +15 -0
- package/dist/lib/proxy/proxyAnalysis.js +6 -0
- package/dist/lib/proxy/requestLogger.d.ts +6 -0
- package/dist/lib/proxy/requestLogger.js +57 -47
- package/dist/lib/proxy/rollingWorkerSupervisor.d.ts +2 -0
- package/dist/lib/proxy/rollingWorkerSupervisor.js +40 -2
- package/dist/lib/server/routes/claudeProxyRoutes.d.ts +2 -0
- package/dist/lib/server/routes/claudeProxyRoutes.js +33 -2
- package/dist/lib/types/processor.d.ts +15 -0
- package/dist/lib/types/proxy.d.ts +32 -1
- package/dist/processors/archive/ArchiveProcessor.js +13 -67
- package/dist/processors/archive/zipEntryReader.d.ts +51 -0
- package/dist/processors/archive/zipEntryReader.js +80 -0
- package/dist/processors/document/OpenDocumentProcessor.js +13 -1
- package/dist/processors/document/PptxProcessor.js +58 -12
- package/dist/proxy/logCleanupScheduler.d.ts +12 -0
- package/dist/proxy/logCleanupScheduler.js +73 -0
- package/dist/proxy/logCleanupWorkerEntry.d.ts +1 -0
- package/dist/proxy/logCleanupWorkerEntry.js +14 -0
- package/dist/proxy/proxyAnalysis.js +6 -0
- package/dist/proxy/requestLogger.d.ts +6 -0
- package/dist/proxy/requestLogger.js +57 -47
- package/dist/proxy/rollingWorkerSupervisor.d.ts +2 -0
- package/dist/proxy/rollingWorkerSupervisor.js +40 -2
- package/dist/server/routes/claudeProxyRoutes.d.ts +2 -0
- package/dist/server/routes/claudeProxyRoutes.js +33 -2
- package/dist/types/processor.d.ts +15 -0
- package/dist/types/proxy.d.ts +32 -1
- package/package.json +5 -3
|
@@ -22,6 +22,7 @@ import { withTimeout } from "../../lib/utils/async/withTimeout.js";
|
|
|
22
22
|
import { formatUptime, isProcessRunning, StateFileManager, } from "../utils/serverUtils.js";
|
|
23
23
|
import { configureProxyKeepAliveDispatcher } from "../../lib/proxy/proxyDispatcher.js";
|
|
24
24
|
import { ProxyRuntimeConfigStore } from "../../lib/proxy/runtimeConfig.js";
|
|
25
|
+
import { startProxyLogCleanupScheduler } from "../../lib/proxy/logCleanupScheduler.js";
|
|
25
26
|
import { anthropicAccountKeysEqual, createAccountAllowlist, ENV_ANTHROPIC_ACCOUNT_KEY, isAccountAllowed, LEGACY_ANTHROPIC_ACCOUNT_KEY, normalizeAnthropicAccountKey, shouldLoadFallbackCredential, } from "../../lib/proxy/accountSelection.js";
|
|
26
27
|
import { beginProxyRequest, getProxyActivitySnapshot, trackProxyResponse, } from "../../lib/proxy/proxyActivity.js";
|
|
27
28
|
import { flushProxyLifecycleEvents, getProxyLifecycleLoggerSnapshot, hashProxyLifecycleSessionId, logProxyLifecycleEvent, } from "../../lib/proxy/proxyLifecycle.js";
|
|
@@ -1208,10 +1209,12 @@ async function createProxyNeurolinkRuntime(logsDir) {
|
|
|
1208
1209
|
process.env.NEUROLINK_SKIP_MCP = "true";
|
|
1209
1210
|
const { NeuroLink } = await import("../../lib/neurolink.js");
|
|
1210
1211
|
const neurolink = new NeuroLink();
|
|
1211
|
-
const { initRequestLogger
|
|
1212
|
+
const { initRequestLogger } = await import("../../lib/proxy/requestLogger.js");
|
|
1212
1213
|
initRequestLogger(true, logsDir);
|
|
1213
|
-
|
|
1214
|
-
|
|
1214
|
+
return {
|
|
1215
|
+
neurolink,
|
|
1216
|
+
logsDir: logsDir ?? join(homedir(), ".neurolink", "logs"),
|
|
1217
|
+
};
|
|
1215
1218
|
}
|
|
1216
1219
|
function registerProxyRequestTracking(app, requestMetadata, readiness) {
|
|
1217
1220
|
app.use("/v1/*", async (c, next) => {
|
|
@@ -2158,7 +2161,7 @@ async function refreshProxyTokensInBackground(accountAllowlist) {
|
|
|
2158
2161
|
// non-fatal
|
|
2159
2162
|
}
|
|
2160
2163
|
}
|
|
2161
|
-
function startProxyBackgroundMaintenance(
|
|
2164
|
+
function startProxyBackgroundMaintenance(logsDir, getAccountAllowlist) {
|
|
2162
2165
|
const refreshInterval = setInterval(() => {
|
|
2163
2166
|
if (backgroundRefreshInProgress) {
|
|
2164
2167
|
return;
|
|
@@ -2172,15 +2175,8 @@ function startProxyBackgroundMaintenance(cleanupLogs, getAccountAllowlist) {
|
|
|
2172
2175
|
backgroundRefreshInProgress = false;
|
|
2173
2176
|
});
|
|
2174
2177
|
}, 30_000);
|
|
2175
|
-
const
|
|
2176
|
-
|
|
2177
|
-
cleanupLogs(7, 500);
|
|
2178
|
-
}
|
|
2179
|
-
catch (error) {
|
|
2180
|
-
logger.debug(`[proxy] background log cleanup failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
2181
|
-
}
|
|
2182
|
-
}, 60 * 60 * 1000);
|
|
2183
|
-
return { refreshInterval, logCleanupInterval };
|
|
2178
|
+
const logCleanupScheduler = startProxyLogCleanupScheduler({ logsDir });
|
|
2179
|
+
return { refreshInterval, logCleanupScheduler };
|
|
2184
2180
|
}
|
|
2185
2181
|
function registerProxyShutdownHandlers(params) {
|
|
2186
2182
|
let shutdownStarted = false;
|
|
@@ -2220,7 +2216,7 @@ function registerProxyShutdownHandlers(params) {
|
|
|
2220
2216
|
}
|
|
2221
2217
|
shutdownStarted = true;
|
|
2222
2218
|
clearInterval(params.refreshInterval);
|
|
2223
|
-
|
|
2219
|
+
await params.logCleanupScheduler.stop();
|
|
2224
2220
|
params.updaterSupervisor?.stop();
|
|
2225
2221
|
params.stopRuntimeConfig?.();
|
|
2226
2222
|
logger.always(`\nShutting down proxy (${signal})...`);
|
|
@@ -2524,7 +2520,7 @@ async function startProxyRuntime(params) {
|
|
|
2524
2520
|
else {
|
|
2525
2521
|
logger.always(chalk.dim(" ⊘ Dev mode: skipping client auto-configuration"));
|
|
2526
2522
|
}
|
|
2527
|
-
const maintenance = startProxyBackgroundMaintenance(params.
|
|
2523
|
+
const maintenance = startProxyBackgroundMaintenance(params.logsDir, () => params.runtimeConfigStore
|
|
2528
2524
|
? params.runtimeConfigStore.getSnapshot().accountAllowlist
|
|
2529
2525
|
: params.accountAllowlist);
|
|
2530
2526
|
const shutdown = registerProxyShutdownHandlers({
|
|
@@ -2719,7 +2715,7 @@ async function startProxyCommandHandler(argv) {
|
|
|
2719
2715
|
// of opening a new flow per request — cuts outbound flow churn through host
|
|
2720
2716
|
// content-filters. Runs once, after env load so it can be tuned via env.
|
|
2721
2717
|
configureProxyKeepAliveDispatcher();
|
|
2722
|
-
const { neurolink,
|
|
2718
|
+
const { neurolink, logsDir } = await createProxyNeurolinkRuntime(devPaths?.logsDir);
|
|
2723
2719
|
const configPath = argv.config
|
|
2724
2720
|
? resolve(argv.config)
|
|
2725
2721
|
: join(homedir(), ".neurolink", "proxy-config.yaml");
|
|
@@ -2770,7 +2766,7 @@ async function startProxyCommandHandler(argv) {
|
|
|
2770
2766
|
accountAllowlist,
|
|
2771
2767
|
loadedEnvFile,
|
|
2772
2768
|
passthrough,
|
|
2773
|
-
|
|
2769
|
+
logsDir,
|
|
2774
2770
|
runtimeConfigStore,
|
|
2775
2771
|
});
|
|
2776
2772
|
}
|
|
@@ -18,9 +18,11 @@ function printAnalysis(report) {
|
|
|
18
18
|
logger.always(report.coverage.finalRequests
|
|
19
19
|
? ` Completed: ${report.requests.completed} (${chalk.green(`${report.requests.success} success`)}, ${chalk.red(`${report.requests.errors} errors`)})`
|
|
20
20
|
: chalk.yellow(" Completed: unavailable (no final request logs)"));
|
|
21
|
-
logger.always(report.coverage.attempts &&
|
|
21
|
+
logger.always(report.coverage.attempts &&
|
|
22
|
+
report.coverage.finalRequests &&
|
|
23
|
+
report.coverage.comparableRequestAttempts
|
|
22
24
|
? ` Recovered after retry: ${report.requests.recoveredAfterRetry}`
|
|
23
|
-
: chalk.yellow(" Recovered after retry: unavailable"));
|
|
25
|
+
: chalk.yellow(" Recovered after retry: unavailable (request/attempt windows are incomplete or non-comparable)"));
|
|
24
26
|
if (report.requests.errors > 0) {
|
|
25
27
|
logger.always(` Final error types: ${JSON.stringify(report.requests.errorTypes)}`);
|
|
26
28
|
}
|
|
@@ -79,11 +81,14 @@ function printAnalysis(report) {
|
|
|
79
81
|
}
|
|
80
82
|
logger.always("");
|
|
81
83
|
logger.always(chalk.bold(" Data Quality"));
|
|
84
|
+
if (!report.coverage.comparableRequestAttempts) {
|
|
85
|
+
logger.always(chalk.yellow(" WARNING: request and attempt totals do not cover a comparable full window; do not reconcile them as one cohort"));
|
|
86
|
+
}
|
|
82
87
|
logger.always(` ${report.dataQuality.linesRead} lines scanned, ${report.dataQuality.malformedLines} malformed, ${report.dataQuality.unsupportedLifecycleLines} unsupported lifecycle, ${report.dataQuality.lifecycleSequenceGaps} sequence gaps, ${report.dataQuality.lifecycleSequenceDuplicates} duplicates`);
|
|
83
88
|
logger.always(` Routing decisions: ${report.dataQuality.routingDecisions.valid} valid, ${report.dataQuality.routingDecisions.invalid} invalid, ${report.dataQuality.routingDecisions.absent} absent`);
|
|
84
89
|
for (const [stream, range] of Object.entries(report.dataQuality.streams)) {
|
|
85
90
|
if (range.observedFrom) {
|
|
86
|
-
logger.always(` ${stream}: ${range.observedFrom} to ${range.observedTo}${range.
|
|
91
|
+
logger.always(` ${stream}: ${range.observedFrom} to ${range.observedTo}${range.completeWindow ? "" : chalk.yellow(" (partial: starts after requested window)")}`);
|
|
87
92
|
}
|
|
88
93
|
}
|
|
89
94
|
const artifacts = report.dataQuality.bodyArtifacts;
|
|
@@ -37,6 +37,7 @@
|
|
|
37
37
|
*/
|
|
38
38
|
import * as path from "path";
|
|
39
39
|
import { BaseFileProcessor } from "../base/BaseFileProcessor.js";
|
|
40
|
+
import { isDecompressionBoundExceeded, readZipEntryWithinLimit, } from "./zipEntryReader.js";
|
|
40
41
|
import { SIZE_LIMITS_MB } from "../config/index.js";
|
|
41
42
|
import { FileErrorCode } from "../errors/index.js";
|
|
42
43
|
// =============================================================================
|
|
@@ -159,71 +160,6 @@ const SINGLE_STREAM_TOOLS = {
|
|
|
159
160
|
xz: "xz",
|
|
160
161
|
zst: "zstd",
|
|
161
162
|
};
|
|
162
|
-
/**
|
|
163
|
-
* Read one ZIP entry's bytes without trusting the size it declares.
|
|
164
|
-
*
|
|
165
|
-
* `entry.getData()` cannot be used for this. It sizes its output buffer from
|
|
166
|
-
* the central-directory `size` field, which the archive author chooses, and
|
|
167
|
-
* adm-zip only arms its own guard when that field is positive:
|
|
168
|
-
*
|
|
169
|
-
* const option = version >= 15 && expectedLength > 0
|
|
170
|
-
* ? { maxOutputLength: expectedLength } : {};
|
|
171
|
-
*
|
|
172
|
-
* So an entry declaring 0 disables the bound and the caller's `size > maxSize`
|
|
173
|
-
* check in one move — `0 > 5MB` is false, and the inflate then runs uncapped.
|
|
174
|
-
* The declared size is the attack, so nothing here may depend on it: the cap
|
|
175
|
-
* comes from our own limit and is handed to the decoder.
|
|
176
|
-
*
|
|
177
|
-
* CRC is verified on both paths rather than dropped, so bypassing `getData()`
|
|
178
|
-
* does not also quietly lose its corruption check — a STORED entry is copied
|
|
179
|
-
* out rather than decoded, but it can be damaged just the same. It detects
|
|
180
|
-
* damage, not malice — the CRC field is attacker-controlled too.
|
|
181
|
-
*/
|
|
182
|
-
function readZipEntryWithinLimit(entry, maxBytes, zlibModule) {
|
|
183
|
-
const compressed = entry.getCompressedData();
|
|
184
|
-
const matchesCrc = (data) => (zlibModule.crc32(data) >>> 0) === (entry.header.crc >>> 0);
|
|
185
|
-
// STORED: the bytes are already the payload, so its own length is the bound.
|
|
186
|
-
if (entry.header.method === ZIP_METHOD_STORED) {
|
|
187
|
-
if (compressed.length > maxBytes) {
|
|
188
|
-
return { status: "too-large" };
|
|
189
|
-
}
|
|
190
|
-
return matchesCrc(compressed)
|
|
191
|
-
? { status: "ok", buffer: compressed }
|
|
192
|
-
: { status: "corrupt" };
|
|
193
|
-
}
|
|
194
|
-
if (entry.header.method !== ZIP_METHOD_DEFLATED) {
|
|
195
|
-
return { status: "unsupported-method" };
|
|
196
|
-
}
|
|
197
|
-
let inflated;
|
|
198
|
-
try {
|
|
199
|
-
inflated = zlibModule.inflateRawSync(compressed, {
|
|
200
|
-
maxOutputLength: maxBytes,
|
|
201
|
-
});
|
|
202
|
-
}
|
|
203
|
-
catch (error) {
|
|
204
|
-
if (isDecompressionBoundExceeded(error)) {
|
|
205
|
-
return { status: "too-large" };
|
|
206
|
-
}
|
|
207
|
-
return { status: "corrupt" };
|
|
208
|
-
}
|
|
209
|
-
return matchesCrc(inflated)
|
|
210
|
-
? { status: "ok", buffer: inflated }
|
|
211
|
-
: { status: "corrupt" };
|
|
212
|
-
}
|
|
213
|
-
/** ZIP compression methods this reader handles (APPNOTE 4.4.5). */
|
|
214
|
-
const ZIP_METHOD_STORED = 0;
|
|
215
|
-
const ZIP_METHOD_DEFLATED = 8;
|
|
216
|
-
/**
|
|
217
|
-
* Whether a zlib rejection is the output bound firing rather than bad input.
|
|
218
|
-
*
|
|
219
|
-
* `maxOutputLength` aborts an inflate the moment its output would pass the cap,
|
|
220
|
-
* which is the whole point — but it surfaces as a plain `RangeError`, and a
|
|
221
|
-
* bomb reported as "failed to decompress" reads as a corrupt upload and invites
|
|
222
|
-
* the user to send it again. It will fail identically every time.
|
|
223
|
-
*
|
|
224
|
-
* Keyed on `code`, not the message: the message embeds a byte count.
|
|
225
|
-
*/
|
|
226
|
-
const isDecompressionBoundExceeded = (error) => error?.code === "ERR_BUFFER_TOO_LARGE";
|
|
227
163
|
/** File extensions recognized as archive formats */
|
|
228
164
|
const SUPPORTED_ARCHIVE_EXTENSIONS = [".zip", ".tar", ".gz", ".tgz", ".bz2", ".tbz2", ".jar", ".xz", ".txz", ".zst", ".tzst"];
|
|
229
165
|
// =============================================================================
|
|
@@ -1419,6 +1355,7 @@ export class ArchiveProcessor extends BaseFileProcessor {
|
|
|
1419
1355
|
.sort((a, b) => a.uncompressedSize - b.uncompressedSize);
|
|
1420
1356
|
let totalExtracted = 0;
|
|
1421
1357
|
let extractCount = 0;
|
|
1358
|
+
const zlibModule = await import("zlib");
|
|
1422
1359
|
for (const entry of candidates) {
|
|
1423
1360
|
if (extractCount >= ARCHIVE_CONFIG.MAX_EXTRACT_ENTRIES) {
|
|
1424
1361
|
break;
|
|
@@ -1431,8 +1368,17 @@ export class ArchiveProcessor extends BaseFileProcessor {
|
|
|
1431
1368
|
if (!zipEntry) {
|
|
1432
1369
|
continue;
|
|
1433
1370
|
}
|
|
1434
|
-
|
|
1435
|
-
|
|
1371
|
+
// This path was already bounded, but only by coincidence: entries
|
|
1372
|
+
// declaring 0 are dropped above, oversized declarations are dropped
|
|
1373
|
+
// above, and adm-zip caps everything else at the size it declares.
|
|
1374
|
+
// That leaves the ceiling resting on library internals we do not
|
|
1375
|
+
// trust anywhere else in this file, so it is spelled out here.
|
|
1376
|
+
const read = readZipEntryWithinLimit(zipEntry, Math.min(ARCHIVE_CONFIG.MAX_EXTRACT_ENTRY_SIZE, ARCHIVE_CONFIG.MAX_TOTAL_EXTRACT_SIZE - totalExtracted), zlibModule);
|
|
1377
|
+
if (read.status !== "ok") {
|
|
1378
|
+
continue;
|
|
1379
|
+
}
|
|
1380
|
+
const data = read.buffer;
|
|
1381
|
+
if (data.length === 0) {
|
|
1436
1382
|
continue;
|
|
1437
1383
|
}
|
|
1438
1384
|
// Simple binary detection: check for null bytes in first 512 bytes
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bounded ZIP entry reading, shared by every processor that opens a ZIP.
|
|
3
|
+
*
|
|
4
|
+
* Extracted from ArchiveProcessor because the Office formats are ZIPs too:
|
|
5
|
+
* .pptx and .odt read their entries directly and so can be handed the same
|
|
6
|
+
* bomb, and need the same refusal. One implementation rather than three means
|
|
7
|
+
* a correction to the guard lands everywhere at once.
|
|
8
|
+
*
|
|
9
|
+
* .docx and .xlsx deliberately do NOT use this. mammoth and exceljs unzip for
|
|
10
|
+
* themselves and were measured refusing a 400MB bomb at 46MB and 53MB peak,
|
|
11
|
+
* so wrapping them in a pre-scan bought no safety and roughly tripled the cost
|
|
12
|
+
* of every ordinary document.
|
|
13
|
+
*
|
|
14
|
+
* @module processors/archive/zipEntryReader
|
|
15
|
+
*/
|
|
16
|
+
import type { ArchiveEntryReadResult, BoundedZipEntry } from "../../types/index.js";
|
|
17
|
+
/** ZIP compression methods this reader handles (APPNOTE 4.4.5). */
|
|
18
|
+
export declare const ZIP_METHOD_STORED = 0;
|
|
19
|
+
export declare const ZIP_METHOD_DEFLATED = 8;
|
|
20
|
+
/**
|
|
21
|
+
* Whether a zlib rejection is the output bound firing rather than bad input.
|
|
22
|
+
*
|
|
23
|
+
* `maxOutputLength` aborts an inflate the moment its output would pass the cap,
|
|
24
|
+
* which is the whole point — but it surfaces as a plain `RangeError`, and a
|
|
25
|
+
* bomb reported as "failed to decompress" reads as a corrupt upload and invites
|
|
26
|
+
* the user to send it again. It will fail identically every time.
|
|
27
|
+
*
|
|
28
|
+
* Keyed on `code`, not the message: the message embeds a byte count.
|
|
29
|
+
*/
|
|
30
|
+
export declare const isDecompressionBoundExceeded: (error: unknown) => boolean;
|
|
31
|
+
/**
|
|
32
|
+
* Read one ZIP entry's bytes without trusting the size it declares.
|
|
33
|
+
*
|
|
34
|
+
* `entry.getData()` cannot be used for this. It sizes its output buffer from
|
|
35
|
+
* the central-directory `size` field, which the archive author chooses, and
|
|
36
|
+
* adm-zip only arms its own guard when that field is positive:
|
|
37
|
+
*
|
|
38
|
+
* const option = version >= 15 && expectedLength > 0
|
|
39
|
+
* ? { maxOutputLength: expectedLength } : {};
|
|
40
|
+
*
|
|
41
|
+
* So an entry declaring 0 disables the bound and the caller's `size > maxSize`
|
|
42
|
+
* check in one move — `0 > 5MB` is false, and the inflate then runs uncapped.
|
|
43
|
+
* The declared size is the attack, so nothing here may depend on it: the cap
|
|
44
|
+
* comes from our own limit and is handed to the decoder.
|
|
45
|
+
*
|
|
46
|
+
* CRC is verified on both paths rather than dropped, so bypassing `getData()`
|
|
47
|
+
* does not also quietly lose its corruption check — a STORED entry is copied
|
|
48
|
+
* out rather than decoded, but it can be damaged just the same. It detects
|
|
49
|
+
* damage, not malice — the CRC field is attacker-controlled too.
|
|
50
|
+
*/
|
|
51
|
+
export declare function readZipEntryWithinLimit(entry: BoundedZipEntry, maxBytes: number, zlibModule: typeof import("zlib")): ArchiveEntryReadResult;
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Bounded ZIP entry reading, shared by every processor that opens a ZIP.
|
|
3
|
+
*
|
|
4
|
+
* Extracted from ArchiveProcessor because the Office formats are ZIPs too:
|
|
5
|
+
* .pptx and .odt read their entries directly and so can be handed the same
|
|
6
|
+
* bomb, and need the same refusal. One implementation rather than three means
|
|
7
|
+
* a correction to the guard lands everywhere at once.
|
|
8
|
+
*
|
|
9
|
+
* .docx and .xlsx deliberately do NOT use this. mammoth and exceljs unzip for
|
|
10
|
+
* themselves and were measured refusing a 400MB bomb at 46MB and 53MB peak,
|
|
11
|
+
* so wrapping them in a pre-scan bought no safety and roughly tripled the cost
|
|
12
|
+
* of every ordinary document.
|
|
13
|
+
*
|
|
14
|
+
* @module processors/archive/zipEntryReader
|
|
15
|
+
*/
|
|
16
|
+
/** ZIP compression methods this reader handles (APPNOTE 4.4.5). */
|
|
17
|
+
export const ZIP_METHOD_STORED = 0;
|
|
18
|
+
export const ZIP_METHOD_DEFLATED = 8;
|
|
19
|
+
/**
|
|
20
|
+
* Whether a zlib rejection is the output bound firing rather than bad input.
|
|
21
|
+
*
|
|
22
|
+
* `maxOutputLength` aborts an inflate the moment its output would pass the cap,
|
|
23
|
+
* which is the whole point — but it surfaces as a plain `RangeError`, and a
|
|
24
|
+
* bomb reported as "failed to decompress" reads as a corrupt upload and invites
|
|
25
|
+
* the user to send it again. It will fail identically every time.
|
|
26
|
+
*
|
|
27
|
+
* Keyed on `code`, not the message: the message embeds a byte count.
|
|
28
|
+
*/
|
|
29
|
+
export const isDecompressionBoundExceeded = (error) => error?.code === "ERR_BUFFER_TOO_LARGE";
|
|
30
|
+
/**
|
|
31
|
+
* Read one ZIP entry's bytes without trusting the size it declares.
|
|
32
|
+
*
|
|
33
|
+
* `entry.getData()` cannot be used for this. It sizes its output buffer from
|
|
34
|
+
* the central-directory `size` field, which the archive author chooses, and
|
|
35
|
+
* adm-zip only arms its own guard when that field is positive:
|
|
36
|
+
*
|
|
37
|
+
* const option = version >= 15 && expectedLength > 0
|
|
38
|
+
* ? { maxOutputLength: expectedLength } : {};
|
|
39
|
+
*
|
|
40
|
+
* So an entry declaring 0 disables the bound and the caller's `size > maxSize`
|
|
41
|
+
* check in one move — `0 > 5MB` is false, and the inflate then runs uncapped.
|
|
42
|
+
* The declared size is the attack, so nothing here may depend on it: the cap
|
|
43
|
+
* comes from our own limit and is handed to the decoder.
|
|
44
|
+
*
|
|
45
|
+
* CRC is verified on both paths rather than dropped, so bypassing `getData()`
|
|
46
|
+
* does not also quietly lose its corruption check — a STORED entry is copied
|
|
47
|
+
* out rather than decoded, but it can be damaged just the same. It detects
|
|
48
|
+
* damage, not malice — the CRC field is attacker-controlled too.
|
|
49
|
+
*/
|
|
50
|
+
export function readZipEntryWithinLimit(entry, maxBytes, zlibModule) {
|
|
51
|
+
const compressed = entry.getCompressedData();
|
|
52
|
+
const matchesCrc = (data) => (zlibModule.crc32(data) >>> 0) === (entry.header.crc >>> 0);
|
|
53
|
+
// STORED: the bytes are already the payload, so its own length is the bound.
|
|
54
|
+
if (entry.header.method === ZIP_METHOD_STORED) {
|
|
55
|
+
if (compressed.length > maxBytes) {
|
|
56
|
+
return { status: "too-large" };
|
|
57
|
+
}
|
|
58
|
+
return matchesCrc(compressed)
|
|
59
|
+
? { status: "ok", buffer: compressed }
|
|
60
|
+
: { status: "corrupt" };
|
|
61
|
+
}
|
|
62
|
+
if (entry.header.method !== ZIP_METHOD_DEFLATED) {
|
|
63
|
+
return { status: "unsupported-method" };
|
|
64
|
+
}
|
|
65
|
+
let inflated;
|
|
66
|
+
try {
|
|
67
|
+
inflated = zlibModule.inflateRawSync(compressed, {
|
|
68
|
+
maxOutputLength: maxBytes,
|
|
69
|
+
});
|
|
70
|
+
}
|
|
71
|
+
catch (error) {
|
|
72
|
+
if (isDecompressionBoundExceeded(error)) {
|
|
73
|
+
return { status: "too-large" };
|
|
74
|
+
}
|
|
75
|
+
return { status: "corrupt" };
|
|
76
|
+
}
|
|
77
|
+
return matchesCrc(inflated)
|
|
78
|
+
? { status: "ok", buffer: inflated }
|
|
79
|
+
: { status: "corrupt" };
|
|
80
|
+
}
|
|
81
|
+
//# sourceMappingURL=zipEntryReader.js.map
|
|
@@ -7,7 +7,9 @@
|
|
|
7
7
|
* @module processors/document/OpenDocumentProcessor
|
|
8
8
|
*/
|
|
9
9
|
import { createRequire } from "node:module";
|
|
10
|
+
import * as zlib from "node:zlib";
|
|
10
11
|
import { BaseFileProcessor } from "../base/BaseFileProcessor.js";
|
|
12
|
+
import { readZipEntryWithinLimit } from "../archive/zipEntryReader.js";
|
|
11
13
|
import { SIZE_LIMITS } from "../config/index.js";
|
|
12
14
|
const require = createRequire(import.meta.url);
|
|
13
15
|
// Re-import for local use within this file
|
|
@@ -64,7 +66,17 @@ export class OpenDocumentProcessor extends BaseFileProcessor {
|
|
|
64
66
|
// Try to get content.xml
|
|
65
67
|
const contentEntry = zip.getEntry("content.xml");
|
|
66
68
|
if (contentEntry) {
|
|
67
|
-
|
|
69
|
+
// Bounded rather than `getData()`: an ODF file is a ZIP, so its
|
|
70
|
+
// content.xml can declare any uncompressed size it likes, and
|
|
71
|
+
// `maxSizeMB` only ever saw the compressed archive on the way in.
|
|
72
|
+
const read = readZipEntryWithinLimit(contentEntry, SIZE_LIMITS.DOCUMENT_MAX_MB * 1024 * 1024, zlib);
|
|
73
|
+
if (read.status === "too-large") {
|
|
74
|
+
throw new Error(`content.xml exceeds the ${SIZE_LIMITS.DOCUMENT_MAX_MB}MB limit for OpenDocument content`);
|
|
75
|
+
}
|
|
76
|
+
if (read.status !== "ok") {
|
|
77
|
+
throw new Error("content.xml could not be read from the archive");
|
|
78
|
+
}
|
|
79
|
+
const xmlContent = read.buffer.toString("utf-8");
|
|
68
80
|
const extracted = this.extractTextFromXml(xmlContent);
|
|
69
81
|
textContent = extracted.text;
|
|
70
82
|
paragraphCount = extracted.paragraphCount;
|
|
@@ -27,6 +27,47 @@
|
|
|
27
27
|
* ```
|
|
28
28
|
*/
|
|
29
29
|
import AdmZip from "adm-zip";
|
|
30
|
+
import * as zlib from "node:zlib";
|
|
31
|
+
import { readZipEntryWithinLimit } from "../archive/zipEntryReader.js";
|
|
32
|
+
import { SIZE_LIMITS } from "../config/index.js";
|
|
33
|
+
/**
|
|
34
|
+
* Total budget for everything read out of one presentation.
|
|
35
|
+
*
|
|
36
|
+
* This class is a static utility with no `BaseFileProcessor` behind it, so
|
|
37
|
+
* there is no `maxSizeMB` to inherit and nothing else was bounding these
|
|
38
|
+
* reads at all — not even a late check. A .pptx is a ZIP, so any entry in it
|
|
39
|
+
* can declare whatever uncompressed size its author likes.
|
|
40
|
+
*/
|
|
41
|
+
const PPTX_MAX_TOTAL_BYTES = SIZE_LIMITS.DOCUMENT_MAX_MB * 1024 * 1024;
|
|
42
|
+
/**
|
|
43
|
+
* Read entries out of one presentation, refusing once the total passes budget.
|
|
44
|
+
*
|
|
45
|
+
* The budget spans the call rather than each entry: a deck of two hundred
|
|
46
|
+
* slides that each sit just under a per-entry limit is the obvious way around
|
|
47
|
+
* one. Returns null for an absent or unreadable entry, which every call site
|
|
48
|
+
* already treats as "no content here".
|
|
49
|
+
*/
|
|
50
|
+
function createBoundedEntryReader() {
|
|
51
|
+
let spent = 0;
|
|
52
|
+
return (entry) => {
|
|
53
|
+
if (!entry) {
|
|
54
|
+
return null;
|
|
55
|
+
}
|
|
56
|
+
const remaining = PPTX_MAX_TOTAL_BYTES - spent;
|
|
57
|
+
if (remaining <= 0) {
|
|
58
|
+
throw new Error(`PPTX content exceeds the ${SIZE_LIMITS.DOCUMENT_MAX_MB}MB limit`);
|
|
59
|
+
}
|
|
60
|
+
const read = readZipEntryWithinLimit(entry, remaining, zlib);
|
|
61
|
+
if (read.status === "too-large") {
|
|
62
|
+
throw new Error(`PPTX content exceeds the ${SIZE_LIMITS.DOCUMENT_MAX_MB}MB limit`);
|
|
63
|
+
}
|
|
64
|
+
if (read.status !== "ok") {
|
|
65
|
+
return null;
|
|
66
|
+
}
|
|
67
|
+
spent += read.buffer.length;
|
|
68
|
+
return read.buffer.toString("utf-8");
|
|
69
|
+
};
|
|
70
|
+
}
|
|
30
71
|
/**
|
|
31
72
|
* Regex to match text content within PowerPoint XML `<a:t>` elements.
|
|
32
73
|
* These elements contain the actual visible text on slides.
|
|
@@ -63,14 +104,17 @@ export class PptxProcessor {
|
|
|
63
104
|
static async extractText(content) {
|
|
64
105
|
const zip = new AdmZip(content);
|
|
65
106
|
const entries = zip.getEntries();
|
|
107
|
+
const readEntry = createBoundedEntryReader();
|
|
66
108
|
// Collect slide entries with their slide numbers for sorting
|
|
67
109
|
const slides = [];
|
|
68
110
|
for (const entry of entries) {
|
|
69
111
|
const match = entry.entryName.match(SLIDE_ENTRY_REGEX);
|
|
70
112
|
if (match) {
|
|
71
113
|
const slideNumber = parseInt(match[1], 10);
|
|
72
|
-
const xmlContent = entry
|
|
73
|
-
|
|
114
|
+
const xmlContent = readEntry(entry);
|
|
115
|
+
if (xmlContent !== null) {
|
|
116
|
+
slides.push({ slideNumber, xml: xmlContent });
|
|
117
|
+
}
|
|
74
118
|
}
|
|
75
119
|
}
|
|
76
120
|
// Sort slides by number (slide1, slide2, ...)
|
|
@@ -82,7 +126,7 @@ export class PptxProcessor {
|
|
|
82
126
|
parts.push(`Presentation: ${slides.length} slide(s)\n`);
|
|
83
127
|
for (const slide of slides) {
|
|
84
128
|
const texts = PptxProcessor.extractTextFromXml(slide.xml);
|
|
85
|
-
const notes = PptxProcessor.extractNotesForSlide(zip, slide.slideNumber);
|
|
129
|
+
const notes = PptxProcessor.extractNotesForSlide(zip, slide.slideNumber, readEntry);
|
|
86
130
|
// Emit a slide section when it has either body text or speaker notes.
|
|
87
131
|
if (texts.length > 0 || notes) {
|
|
88
132
|
parts.push(`### Slide ${slide.slideNumber}`);
|
|
@@ -111,12 +155,11 @@ export class PptxProcessor {
|
|
|
111
155
|
* @param slideNumber - 1-indexed slide number
|
|
112
156
|
* @returns The notes text (runs joined by a space), or null when absent
|
|
113
157
|
*/
|
|
114
|
-
static extractNotesForSlide(zip, slideNumber) {
|
|
115
|
-
const
|
|
116
|
-
if (
|
|
158
|
+
static extractNotesForSlide(zip, slideNumber, readEntry) {
|
|
159
|
+
const relsXml = readEntry(zip.getEntry(SLIDE_RELS_NAME(slideNumber)));
|
|
160
|
+
if (relsXml === null) {
|
|
117
161
|
return null;
|
|
118
162
|
}
|
|
119
|
-
const relsXml = relsEntry.getData().toString("utf-8");
|
|
120
163
|
let notesTarget = null;
|
|
121
164
|
RELATIONSHIP_TAG_REGEX.lastIndex = 0;
|
|
122
165
|
for (let match = RELATIONSHIP_TAG_REGEX.exec(relsXml); match !== null; match = RELATIONSHIP_TAG_REGEX.exec(relsXml)) {
|
|
@@ -134,11 +177,11 @@ export class PptxProcessor {
|
|
|
134
177
|
const normalized = notesTarget
|
|
135
178
|
.replace(/^\.\.\//, "ppt/")
|
|
136
179
|
.replace(/^\/+/, "");
|
|
137
|
-
const
|
|
138
|
-
if (
|
|
180
|
+
const notesXml = readEntry(zip.getEntry(normalized));
|
|
181
|
+
if (notesXml === null) {
|
|
139
182
|
return null;
|
|
140
183
|
}
|
|
141
|
-
const notes = PptxProcessor.extractTextFromXml(
|
|
184
|
+
const notes = PptxProcessor.extractTextFromXml(notesXml).join(" ");
|
|
142
185
|
return notes.trim() || null;
|
|
143
186
|
}
|
|
144
187
|
/**
|
|
@@ -175,6 +218,7 @@ export class PptxProcessor {
|
|
|
175
218
|
static async extractSlides(content, slideNumbers) {
|
|
176
219
|
const zip = new AdmZip(content);
|
|
177
220
|
const entries = zip.getEntries();
|
|
221
|
+
const readEntry = createBoundedEntryReader();
|
|
178
222
|
// Collect all slides
|
|
179
223
|
const slides = [];
|
|
180
224
|
for (const entry of entries) {
|
|
@@ -182,8 +226,10 @@ export class PptxProcessor {
|
|
|
182
226
|
if (match) {
|
|
183
227
|
const slideNumber = parseInt(match[1], 10);
|
|
184
228
|
if (slideNumbers.includes(slideNumber)) {
|
|
185
|
-
const xmlContent = entry
|
|
186
|
-
|
|
229
|
+
const xmlContent = readEntry(entry);
|
|
230
|
+
if (xmlContent !== null) {
|
|
231
|
+
slides.push({ slideNumber, xml: xmlContent });
|
|
232
|
+
}
|
|
187
233
|
}
|
|
188
234
|
}
|
|
189
235
|
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { ProxyLogCleanupScheduler } from "../types/index.js";
|
|
2
|
+
/**
|
|
3
|
+
* Runs retention in a worker so large log trees cannot delay readiness or
|
|
4
|
+
* block active streams. Concurrent runs are coalesced into the active scan.
|
|
5
|
+
*/
|
|
6
|
+
export declare function startProxyLogCleanupScheduler(params: {
|
|
7
|
+
logsDir: string;
|
|
8
|
+
maxAgeDays?: number;
|
|
9
|
+
maxSizeMb?: number;
|
|
10
|
+
initialDelayMs?: number;
|
|
11
|
+
intervalMs?: number;
|
|
12
|
+
}): ProxyLogCleanupScheduler;
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
import { Worker } from "node:worker_threads";
|
|
2
|
+
import { withTimeout } from "../utils/async/withTimeout.js";
|
|
3
|
+
import { logger } from "../utils/logger.js";
|
|
4
|
+
const DEFAULT_INITIAL_DELAY_MS = 30_000;
|
|
5
|
+
const DEFAULT_INTERVAL_MS = 60 * 60 * 1000;
|
|
6
|
+
const WORKER_TERMINATION_TIMEOUT_MS = 5_000;
|
|
7
|
+
/**
|
|
8
|
+
* Runs retention in a worker so large log trees cannot delay readiness or
|
|
9
|
+
* block active streams. Concurrent runs are coalesced into the active scan.
|
|
10
|
+
*/
|
|
11
|
+
export function startProxyLogCleanupScheduler(params) {
|
|
12
|
+
const maxAgeDays = params.maxAgeDays ?? 7;
|
|
13
|
+
const maxSizeMb = params.maxSizeMb ?? 500;
|
|
14
|
+
let activeWorker;
|
|
15
|
+
let stopped = false;
|
|
16
|
+
const trigger = () => {
|
|
17
|
+
if (stopped || activeWorker) {
|
|
18
|
+
return false;
|
|
19
|
+
}
|
|
20
|
+
let worker;
|
|
21
|
+
try {
|
|
22
|
+
worker = new Worker(params.workerUrl ??
|
|
23
|
+
new URL("./logCleanupWorkerEntry.js", import.meta.url), {
|
|
24
|
+
execArgv: process.execArgv.filter((argument) => !argument.startsWith("--input-type")),
|
|
25
|
+
workerData: {
|
|
26
|
+
logsDir: params.logsDir,
|
|
27
|
+
maxAgeDays,
|
|
28
|
+
maxSizeMb,
|
|
29
|
+
},
|
|
30
|
+
});
|
|
31
|
+
}
|
|
32
|
+
catch (error) {
|
|
33
|
+
logger.debug(`[proxy] could not start background log cleanup: ${error instanceof Error ? error.message : String(error)}`);
|
|
34
|
+
return false;
|
|
35
|
+
}
|
|
36
|
+
activeWorker = worker;
|
|
37
|
+
worker.unref();
|
|
38
|
+
worker.once("error", (error) => {
|
|
39
|
+
logger.debug(`[proxy] background log cleanup failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
40
|
+
});
|
|
41
|
+
worker.once("exit", (code) => {
|
|
42
|
+
if (activeWorker === worker) {
|
|
43
|
+
activeWorker = undefined;
|
|
44
|
+
}
|
|
45
|
+
if (code !== 0 && !stopped) {
|
|
46
|
+
logger.debug(`[proxy] background log cleanup exited with code ${code}`);
|
|
47
|
+
}
|
|
48
|
+
});
|
|
49
|
+
return true;
|
|
50
|
+
};
|
|
51
|
+
const initialTimer = setTimeout(trigger, params.initialDelayMs ?? DEFAULT_INITIAL_DELAY_MS);
|
|
52
|
+
initialTimer.unref();
|
|
53
|
+
const interval = setInterval(trigger, params.intervalMs ?? DEFAULT_INTERVAL_MS);
|
|
54
|
+
interval.unref();
|
|
55
|
+
return {
|
|
56
|
+
trigger,
|
|
57
|
+
stop: async () => {
|
|
58
|
+
stopped = true;
|
|
59
|
+
clearTimeout(initialTimer);
|
|
60
|
+
clearInterval(interval);
|
|
61
|
+
const worker = activeWorker;
|
|
62
|
+
activeWorker = undefined;
|
|
63
|
+
if (worker) {
|
|
64
|
+
try {
|
|
65
|
+
await withTimeout(worker.terminate(), WORKER_TERMINATION_TIMEOUT_MS, "Timed out terminating the proxy log cleanup worker");
|
|
66
|
+
}
|
|
67
|
+
catch (error) {
|
|
68
|
+
logger.debug(`[proxy] background log cleanup termination failed: ${error instanceof Error ? error.message : String(error)}`);
|
|
69
|
+
}
|
|
70
|
+
}
|
|
71
|
+
},
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
//# sourceMappingURL=logCleanupScheduler.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
import { parentPort, workerData } from "node:worker_threads";
|
|
2
|
+
import { cleanupLogsAt } from "./requestLogger.js";
|
|
3
|
+
const data = workerData;
|
|
4
|
+
try {
|
|
5
|
+
cleanupLogsAt(data.logsDir, data.maxAgeDays, data.maxSizeMb);
|
|
6
|
+
parentPort?.postMessage({ ok: true });
|
|
7
|
+
}
|
|
8
|
+
catch (error) {
|
|
9
|
+
parentPort?.postMessage({
|
|
10
|
+
ok: false,
|
|
11
|
+
error: error instanceof Error ? error.message : String(error),
|
|
12
|
+
});
|
|
13
|
+
process.exitCode = 1;
|
|
14
|
+
}
|
|
15
|
+
//# sourceMappingURL=logCleanupWorkerEntry.js.map
|
|
@@ -737,6 +737,10 @@ export async function analyzeProxyLogs(options) {
|
|
|
737
737
|
const artifactsReferenced = artifactsPresent + artifactsMissing;
|
|
738
738
|
const finalSummary = summarizeFinalRequests(finalRequests, terminalStreamErrors, attemptsByRequest, accounts);
|
|
739
739
|
const routingSummary = summarizeRouting(finalRequests);
|
|
740
|
+
const streamComplete = (stream) => {
|
|
741
|
+
const range = observedRanges[stream];
|
|
742
|
+
return range.from !== null && range.from <= sinceMs;
|
|
743
|
+
};
|
|
740
744
|
return {
|
|
741
745
|
generatedAt: new Date(nowMs).toISOString(),
|
|
742
746
|
since: new Date(sinceMs).toISOString(),
|
|
@@ -755,6 +759,7 @@ export async function analyzeProxyLogs(options) {
|
|
|
755
759
|
attemptLatency: attemptLatency.length > 0,
|
|
756
760
|
cacheUsage: finalSummary.cache.requestsWithUsage > 0,
|
|
757
761
|
routingDecisions: routingSummary.totalRecords > 0,
|
|
762
|
+
comparableRequestAttempts: streamComplete("requests") && streamComplete("attempts"),
|
|
758
763
|
},
|
|
759
764
|
dataQuality: {
|
|
760
765
|
linesRead,
|
|
@@ -768,6 +773,7 @@ export async function analyzeProxyLogs(options) {
|
|
|
768
773
|
observedFrom: range.from === null ? null : new Date(range.from).toISOString(),
|
|
769
774
|
observedTo: range.to === null ? null : new Date(range.to).toISOString(),
|
|
770
775
|
startsAtOrBeforeRequestedWindow: range.from !== null && range.from <= sinceMs,
|
|
776
|
+
completeWindow: range.from !== null && range.from <= sinceMs,
|
|
771
777
|
},
|
|
772
778
|
])),
|
|
773
779
|
bodyArtifacts: {
|
|
@@ -70,3 +70,9 @@ export declare function logStreamError(entry: {
|
|
|
70
70
|
* Non-fatal — proxy keeps working even if cleanup fails.
|
|
71
71
|
*/
|
|
72
72
|
export declare function cleanupLogs(maxAgeDays?: number, maxSizeMb?: number): void;
|
|
73
|
+
/**
|
|
74
|
+
* Path-scoped retention implementation used by the proxy cleanup worker.
|
|
75
|
+
* This function is intentionally synchronous: callers must run it outside the
|
|
76
|
+
* request-serving process when the directory can contain many artifacts.
|
|
77
|
+
*/
|
|
78
|
+
export declare function cleanupLogsAt(activeLogDir: string, maxAgeDays?: number, maxSizeMb?: number): void;
|