infracensus-collector 1.4.1 → 1.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@ import {
2
2
  extensionKind,
3
3
  extractOpenXml,
4
4
  extractPlainText
5
- } from "./chunk-C5MDCEUD.js";
5
+ } from "./chunk-O3Z4CFCD.js";
6
6
  import {
7
7
  checkEntryPath,
8
8
  checkEntrySize
@@ -67,4 +67,4 @@ function extractArchive(bytes) {
67
67
  export {
68
68
  extractArchive
69
69
  };
70
- //# sourceMappingURL=archive-RCFXOPI6.js.map
70
+ //# sourceMappingURL=archive-O5IFKUA6.js.map
@@ -0,0 +1,7 @@
1
+ {
2
+ "version": 3,
3
+ "sources": ["../src/sensitiveData/archive.ts"],
4
+ "sourcesContent": ["import { unzipSync } from \"fflate\";\r\nimport { checkEntryPath, checkEntrySize, type SizeBudget } from \"@netdoc/config-ir\";\r\nimport { extensionKind, extractOpenXml, extractPlainText, type ExtractOutcome } from \"./extract.js\";\r\n\r\n/**\r\n * Archives (PLAN-010 Phase 6).\r\n *\r\n * \u2500\u2500 WHY A ZIP IS WORTH OPENING AT ALL \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * \"Records_2019.zip\" on a share is the single most likely place an old export\r\n * of a whole roster is sitting. It is what somebody made when they cleared a\r\n * folder, and it is exactly the material that outlives the retention policy\r\n * everybody has forgotten. Leaving archives unread means the module's coverage\r\n * number is honest and its FINDINGS are systematically missing the worst files.\r\n *\r\n * \u2500\u2500 AND WHY IT IS THE MOST DANGEROUS THING THIS MODULE DOES \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * An archive is attacker-controlled input that expands. Anyone who can write to\r\n * a share can leave a 42 KB file that becomes several petabytes, and the scan\r\n * runs unattended, at night, on a machine inside the customer's network. So\r\n * nothing here is written fresh: it reuses `@netdoc/config-ir`'s archive guards,\r\n * which already carry the per-file cap, the entry cap, the total-expansion cap\r\n * and the compression-ratio ceiling, threaded through one budget so an archive\r\n * is refused the MOMENT it crosses a limit rather than after it has expanded.\r\n *\r\n * `checkEntryPath` is the other half: an entry named `..\\..\\windows\\system32`\r\n * is a zip-slip attempt, and this never writes to disk \u2014 but the path becomes a\r\n * REPORTED path, so a crafted name would land in an audit row and an export.\r\n *\r\n * \u2500\u2500 WHAT IS NOT DONE, AND SAID RATHER THAN IMPLIED \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * Nested archives are NOT opened. One level only. A zip inside a zip is the\r\n * classic amplification shape, and the guards above bound each level but not\r\n * their product \u2014 an archive of a thousand archives of a thousand archives is\r\n * within every per-archive limit and is unbounded overall. Depth is the cheap,\r\n * total answer, and the inner archive is reported as an unread entry rather\r\n * than silently ignored.\r\n *\r\n * Encrypted zips are refused as `encrypted`, not as corrupt. \"1,900\r\n * password-protected archives\" is a finding; \"1,900 corrupt files\" is noise.\r\n */\r\n\r\n\r\n/** Entries read per archive, over and above the shared budget's entry cap. */\r\nconst MAX_TEXT_ENTRIES = 500;\r\n\r\nexport interface ArchiveEntry {\r\n /** Path INSIDE the archive, as reported. Never written to disk. */\r\n path: string;\r\n text: string;\r\n}\r\n\r\nexport interface ArchiveOutcome {\r\n entries: ArchiveEntry[];\r\n /** Entries that were not read, by reason, so the gap is countable. */\r\n unread: Record<string, number>;\r\n /** True when a cap stopped the walk of this archive before it finished. */\r\n truncated: boolean;\r\n}\r\n\r\nconst note = (unread: Record<string, number>, reason: string): void => {\r\n unread[reason] = (unread[reason] ?? 0) + 1;\r\n};\r\n\r\n/**\r\n * Read the text entries of one archive.\r\n *\r\n * Returns entries rather than one concatenated blob so the caller can attribute\r\n * a finding to `Records_2019.zip \u2192 registrar/2018.csv`. A single joined string\r\n * would report the archive as the file holding an SSN, which is true and\r\n * useless: nobody can act on it without opening the archive themselves.\r\n */\r\nexport function extractArchive(bytes: Uint8Array): ArchiveOutcome {\r\n const entries: ArchiveEntry[] = [];\r\n const unread: Record<string, number> = {};\r\n let truncated = false;\r\n\r\n // The shared budget, threaded through every entry. This is what makes the\r\n // guard incremental rather than a post-hoc check on something already\r\n // expanded in memory.\r\n const budget: SizeBudget = { entries: 0, uncompressedBytes: 0 };\r\n\r\n let files: Record<string, Uint8Array>;\r\n try {\r\n files = unzipSync(bytes);\r\n } catch (err) {\r\n const message = err instanceof Error ? err.message : String(err);\r\n // fflate has no encrypted-zip support and fails on the flag. Distinguished\r\n // because a password-protected archive is a finding an operator can act on\r\n // and a corrupt one is not.\r\n note(unread, /encrypt|password/i.test(message) ? \"encrypted\" : \"corrupt\");\r\n return { entries, unread, truncated };\r\n }\r\n\r\n for (const [rawPath, content] of Object.entries(files)) {\r\n // Directory entries. Not a gap; nothing was ever going to be read from one.\r\n if (rawPath.endsWith(\"/\") || content.length === 0) continue;\r\n\r\n if (entries.length >= MAX_TEXT_ENTRIES) {\r\n truncated = true;\r\n break;\r\n }\r\n\r\n // Zip-slip. Never written to disk here \u2014 but the path is REPORTED, and a\r\n // crafted name would land in an audit row and an export.\r\n const pathCheck = checkEntryPath(rawPath);\r\n if (!pathCheck.ok) {\r\n note(unread, pathCheck.reason ?? \"unsafe-path\");\r\n continue;\r\n }\r\n\r\n // The compressed size is not exposed per entry by `unzipSync`, so the\r\n // ratio check cannot run per entry here. The per-file, entry-count and\r\n // total-expansion caps all still apply \u2014 and the whole archive was already\r\n // bounded by the walk's own `maxFileBytes` before it was ever read.\r\n const sizeCheck = checkEntrySize(content.length, undefined, budget);\r\n if (!sizeCheck.ok) {\r\n note(unread, sizeCheck.reason ?? \"too-large\");\r\n // A budget failure is about the ARCHIVE, not this entry: once it is\r\n // exceeded every later entry fails too. Stop, and say the read was\r\n // truncated rather than counting a thousand identical refusals.\r\n truncated = true;\r\n break;\r\n }\r\n\r\n const extension = rawPath.slice(rawPath.lastIndexOf(\".\") + 1).toLowerCase();\r\n const kind = extensionKind(extension);\r\n\r\n if (kind === \"ignore\") continue;\r\n\r\n if (extension === \"zip\") {\r\n // ALWAYS refused, never conditionally. This started as a depth counter\r\n // and the counter was the bug: at the only depth it is ever called with\r\n // it took the other branch, so a zip inside a zip was reported as an\r\n // unsupported format rather than as the nested archive it is. A knob\r\n // with one possible value is worse than no knob \u2014 it reads as a policy\r\n // that can be tuned and behaves as one that cannot.\r\n //\r\n // The policy itself is in the header: the guards bound each LEVEL, not\r\n // their product, so an archive of archives is within every per-archive\r\n // limit and unbounded overall. One level is the cheap total answer.\r\n note(unread, \"nested-archive\");\r\n continue;\r\n }\r\n\r\n // PDF inside an archive is skipped rather than parsed: `extractPdf` is\r\n // async and everything here is synchronous, and making this async to reach\r\n // it would put an await inside the bomb guard's loop for a case that is\r\n // rare. Counted, so the gap is visible rather than assumed away.\r\n if (kind === \"pdf\" || kind === \"unsupported\") {\r\n note(unread, \"unsupported-format\");\r\n continue;\r\n }\r\n\r\n const outcome: ExtractOutcome =\r\n kind === \"openxml\" ? extractOpenXml(content, extension) : extractPlainText(content);\r\n\r\n if (outcome.kind === \"text\") {\r\n entries.push({ path: rawPath, text: outcome.text });\r\n } else if (outcome.kind === \"unreadable\") {\r\n note(unread, outcome.reason);\r\n }\r\n }\r\n\r\n return { entries, unread, truncated };\r\n}\r\n"],
5
+ "mappings": ";;;;;;;;;;;AAAA,SAAS,iBAAiB;AA4C1B,IAAM,mBAAmB;AAgBzB,IAAM,OAAO,CAAC,QAAgC,WAAyB;AACrE,SAAO,MAAM,KAAK,OAAO,MAAM,KAAK,KAAK;AAC3C;AAUO,SAAS,eAAe,OAAmC;AAChE,QAAM,UAA0B,CAAC;AACjC,QAAM,SAAiC,CAAC;AACxC,MAAI,YAAY;AAKhB,QAAM,SAAqB,EAAE,SAAS,GAAG,mBAAmB,EAAE;AAE9D,MAAI;AACJ,MAAI;AACF,YAAQ,UAAU,KAAK;AAAA,EACzB,SAAS,KAAK;AACZ,UAAM,UAAU,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AAI/D,SAAK,QAAQ,oBAAoB,KAAK,OAAO,IAAI,cAAc,SAAS;AACxE,WAAO,EAAE,SAAS,QAAQ,UAAU;AAAA,EACtC;AAEA,aAAW,CAAC,SAAS,OAAO,KAAK,OAAO,QAAQ,KAAK,GAAG;AAEtD,QAAI,QAAQ,SAAS,GAAG,KAAK,QAAQ,WAAW,EAAG;AAEnD,QAAI,QAAQ,UAAU,kBAAkB;AACtC,kBAAY;AACZ;AAAA,IACF;AAIA,UAAM,YAAY,eAAe,OAAO;AACxC,QAAI,CAAC,UAAU,IAAI;AACjB,WAAK,QAAQ,UAAU,UAAU,aAAa;AAC9C;AAAA,IACF;AAMA,UAAM,YAAY,eAAe,QAAQ,QAAQ,QAAW,MAAM;AAClE,QAAI,CAAC,UAAU,IAAI;AACjB,WAAK,QAAQ,UAAU,UAAU,WAAW;AAI5C,kBAAY;AACZ;AAAA,IACF;AAEA,UAAM,YAAY,QAAQ,MAAM,QAAQ,YAAY,GAAG,IAAI,CAAC,EAAE,YAAY;AAC1E,UAAM,OAAO,cAAc,SAAS;AAEpC,QAAI,SAAS,SAAU;AAEvB,QAAI,cAAc,OAAO;AAWvB,WAAK,QAAQ,gBAAgB;AAC7B;AAAA,IACF;AAMA,QAAI,SAAS,SAAS,SAAS,eAAe;AAC5C,WAAK,QAAQ,oBAAoB;AACjC;AAAA,IACF;AAEA,UAAM,UACJ,SAAS,YAAY,eAAe,SAAS,SAAS,IAAI,iBAAiB,OAAO;AAEpF,QAAI,QAAQ,SAAS,QAAQ;AAC3B,cAAQ,KAAK,EAAE,MAAM,SAAS,MAAM,QAAQ,KAAK,CAAC;AAAA,IACpD,WAAW,QAAQ,SAAS,cAAc;AACxC,WAAK,QAAQ,QAAQ,MAAM;AAAA,IAC7B;AAAA,EACF;AAEA,SAAO,EAAE,SAAS,QAAQ,UAAU;AACtC;",
6
+ "names": []
7
+ }
@@ -155,11 +155,11 @@ async function extractFile(path, extension) {
155
155
  }
156
156
  if (bytes.length === 0) return { kind: "skipped", reason: "zero-length" };
157
157
  if (kind === "pdf") {
158
- const { extractPdf } = await import("./pdf-7V5KSOI3.js");
158
+ const { extractPdf } = await import("./pdf-JFKD73IQ.js");
159
159
  return extractPdf(bytes);
160
160
  }
161
161
  if (kind === "archive") {
162
- const { extractArchive } = await import("./archive-RCFXOPI6.js");
162
+ const { extractArchive } = await import("./archive-O5IFKUA6.js");
163
163
  const out = extractArchive(bytes);
164
164
  return out.entries.length > 0 ? { kind: "archive", entries: out.entries, unread: out.unread, truncated: out.truncated } : {
165
165
  kind: "unreadable",
@@ -360,4 +360,4 @@ export {
360
360
  extractOpenXml,
361
361
  csvCells
362
362
  };
363
- //# sourceMappingURL=chunk-C5MDCEUD.js.map
363
+ //# sourceMappingURL=chunk-O3Z4CFCD.js.map
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "version": 3,
3
3
  "sources": ["../src/sensitiveData/extract.ts", "../src/sensitiveData/encoding.ts"],
4
- "sourcesContent": ["import { readFile } from \"node:fs/promises\";\r\nimport { unzipSync } from \"fflate\";\r\nimport { decodeText, decodedLooksLikeText } from \"./encoding.js\";\r\n\r\n/**\r\n * Getting text out of a file (PLAN-010 Phase 3).\r\n *\r\n * \u2500\u2500 EVERY OUTCOME IS NAMED \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * There is no silent skip in this file. A file is text, or it is deliberately\r\n * ignored, or it is `unreadable` WITH A REASON \u2014 and the reason is counted and\r\n * reported per share. \"1,900 password-protected spreadsheets\" and \"1,900 corrupt\r\n * files\" and \"1,900 scanned PDFs with no text layer\" are three different facts\r\n * about an estate, and only one of them is somebody's problem to fix.\r\n *\r\n * Every competing tool collapses all three into \"skipped\". A report that quietly\r\n * omitted a third of the estate is worse than one that says which third, which\r\n * is why the unreadable table is a first-class output rather than a debug log.\r\n *\r\n * \u2500\u2500 WHY `fflate` AND NOT A HAND-ROLLED ZIP READER \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * An OpenXML document is a ZIP, and reading one looks like 250 lines over\r\n * `node:zlib`. It is not: a real implementation must survive zip64, mismatched\r\n * local and central headers, and malformed central directories \u2014 on input that\r\n * arrives from a file share and is therefore attacker-influenced. That is a\r\n * security liability written to save a dependency. `fflate` is ~8 KB, has no\r\n * dependencies of its own, and is runtime-portable, which keeps the option of\r\n * running this same extractor in a Worker later.\r\n */\r\n\r\nexport type UnreadableReason =\r\n | \"encrypted\"\r\n | \"corrupt\"\r\n | \"timeout\"\r\n | \"access-denied\"\r\n | \"in-use\"\r\n | \"no-text-layer\"\r\n | \"unsupported-format\"\r\n | \"too-large\"\r\n | \"binary\";\r\n\r\nexport interface TabularCell {\r\n sheet: string;\r\n row: number;\r\n column: number;\r\n header: string | null;\r\n value: string;\r\n}\r\n\r\nexport type ExtractOutcome =\r\n | {\r\n kind: \"text\";\r\n text: string;\r\n /** True when the extracted text hit the cap. Reported, never silent. */\r\n truncated: boolean;\r\n /** Populated for csv/xlsx. The single biggest precision lever \u2014 see below. */\r\n cells?: TabularCell[];\r\n }\r\n /**\r\n * An archive, as its READABLE MEMBERS rather than as one blob.\r\n *\r\n * Separate from `text` because a finding has to be able to say\r\n * `Records_2019.zip -> registrar/2018.csv`. Concatenating the members would\r\n * report the archive as the file holding an SSN, which is true and useless:\r\n * nobody can act on it without opening the archive themselves.\r\n */\r\n | {\r\n kind: \"archive\";\r\n entries: Array<{ path: string; text: string }>;\r\n /** Members that could not be read, by reason, so the gap is countable. */\r\n unread: Record<string, number>;\r\n truncated: boolean;\r\n }\r\n | { kind: \"skipped\"; reason: \"ignored-extension\" | \"zero-length\" }\r\n | { kind: \"unreadable\"; reason: UnreadableReason; detail: string };\r\n\r\n/** Extracted text is capped; past this the file is classified on its head. */\r\nexport const MAX_TEXT_CHARS = 20 * 1024 * 1024;\r\n\r\nconst PLAIN_TEXT = new Set([\r\n \"txt\", \"csv\", \"tsv\", \"log\", \"json\", \"xml\", \"sql\", \"md\", \"htm\", \"html\", \"ini\", \"cfg\",\r\n \"conf\", \"yaml\", \"yml\", \"ps1\", \"bat\", \"cmd\", \"sh\", \"py\", \"js\", \"ts\", \"css\", \"rtf\",\r\n]);\r\nconst OPENXML = new Set([\"docx\", \"xlsx\", \"pptx\", \"docm\", \"xlsm\", \"pptm\"]);\r\n/**\r\n * Recognised, deliberately not supported yet, and COUNTED rather than ignored.\r\n *\r\n * `pdf` left this set in Phase 6 and now has its own kind. It is worth saying\r\n * why it was the one that earned the work: a scanned PDF extracts to nothing,\r\n * and \"extracted nothing\" and \"contains nothing\" are the same value with\r\n * opposite meanings. Every other format here fails loudly; that one would have\r\n * failed as a clean result. See pdf.ts.\r\n */\r\nconst KNOWN_UNSUPPORTED = new Set([\"doc\", \"xls\", \"ppt\", \"msg\", \"pst\", \"7z\", \"rar\", \"eml\"]);\r\n\r\n/**\r\n * The OLE compound-file signature.\r\n *\r\n * An ECMA-376 encrypted OpenXML document is NOT a ZIP \u2014 it is an OLE container\r\n * holding the encrypted package. Recognising it is the difference between\r\n * reporting \"1,900 password-protected spreadsheets\", which is a finding, and\r\n * \"1,900 corrupt files\", which is noise.\r\n */\r\nconst OLE_MAGIC = [0xd0, 0xcf, 0x11, 0xe0, 0xa1, 0xb1, 0x1a, 0xe1];\r\n\r\nconst isOle = (b: Uint8Array): boolean =>\r\n b.length >= 8 && OLE_MAGIC.every((v, i) => b[i] === v);\r\n\r\nconst isZip = (b: Uint8Array): boolean =>\r\n b.length >= 4 && b[0] === 0x50 && b[1] === 0x4b && (b[2] === 0x03 || b[2] === 0x05 || b[2] === 0x07);\r\n\r\nexport function extensionKind(\r\n ext: string,\r\n): \"text\" | \"openxml\" | \"pdf\" | \"archive\" | \"unsupported\" | \"ignore\" {\r\n const e = ext.toLowerCase().replace(/^\\./, \"\");\r\n if (PLAIN_TEXT.has(e)) return \"text\";\r\n if (OPENXML.has(e)) return \"openxml\";\r\n if (e === \"pdf\") return \"pdf\";\r\n if (e === \"zip\") return \"archive\";\r\n if (KNOWN_UNSUPPORTED.has(e)) return \"unsupported\";\r\n return \"ignore\";\r\n}\r\n\r\nexport async function extractFile(path: string, extension: string): Promise<ExtractOutcome> {\r\n const kind = extensionKind(extension);\r\n if (kind === \"ignore\") return { kind: \"skipped\", reason: \"ignored-extension\" };\r\n if (kind === \"unsupported\") {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"unsupported-format\",\r\n detail: `.${extension} is recognised but not yet extracted \u2014 counted so the coverage gap is visible`,\r\n };\r\n }\r\n\r\n let bytes: Uint8Array;\r\n try {\r\n bytes = new Uint8Array(await readFile(path));\r\n } catch (err) {\r\n const code = (err as NodeJS.ErrnoException | undefined)?.code ?? \"\";\r\n const detail = err instanceof Error ? err.message : String(err);\r\n if (code === \"EACCES\" || code === \"EPERM\") return { kind: \"unreadable\", reason: \"access-denied\", detail };\r\n if (code === \"EBUSY\") return { kind: \"unreadable\", reason: \"in-use\", detail };\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail };\r\n }\r\n\r\n if (bytes.length === 0) return { kind: \"skipped\", reason: \"zero-length\" };\r\n if (kind === \"pdf\") {\r\n // Imported lazily inside the extractor, so a run over a share with no PDFs\r\n // never loads several megabytes of parser. See pdf.ts.\r\n const { extractPdf } = await import(\"./pdf.js\");\r\n return extractPdf(bytes);\r\n }\r\n if (kind === \"archive\") {\r\n // Guarded by @netdoc/config-ir's archive budget \u2014 an archive is\r\n // attacker-controlled input that expands, and is the most dangerous thing\r\n // this module reads. See archive.ts.\r\n const { extractArchive } = await import(\"./archive.js\");\r\n const out = extractArchive(bytes);\r\n // Nothing readable inside is not the same as a corrupt archive, and both\r\n // differ from an archive that was refused by the bomb guard. The reasons\r\n // travel in `unread`; this only decides whether there is text to classify.\r\n return out.entries.length > 0\r\n ? { kind: \"archive\", entries: out.entries, unread: out.unread, truncated: out.truncated }\r\n : {\r\n kind: \"unreadable\",\r\n reason: firstReason(out.unread),\r\n detail: describeUnread(out.unread),\r\n };\r\n }\r\n return kind === \"openxml\" ? extractOpenXml(bytes, extension) : extractPlainText(bytes);\r\n}\r\n\r\nexport function extractPlainText(bytes: Uint8Array): ExtractOutcome {\r\n const decoded = decodeText(bytes);\r\n if (decoded === null) {\r\n return { kind: \"unreadable\", reason: \"binary\", detail: \"not text in any recognised encoding\" };\r\n }\r\n if (!decodedLooksLikeText(decoded.text)) {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"binary\",\r\n detail: `decoded as ${decoded.encoding} but is mostly replacement characters`,\r\n };\r\n }\r\n const truncated = decoded.text.length > MAX_TEXT_CHARS;\r\n return {\r\n kind: \"text\",\r\n text: truncated ? decoded.text.slice(0, MAX_TEXT_CHARS) : decoded.text,\r\n truncated,\r\n };\r\n}\r\n\r\nexport function extractOpenXml(bytes: Uint8Array, extension: string): ExtractOutcome {\r\n if (isOle(bytes)) {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"encrypted\",\r\n detail: \"an OLE container \u2014 an ECMA-376 password-protected document\",\r\n };\r\n }\r\n if (!isZip(bytes)) {\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail: \"not a ZIP container\" };\r\n }\r\n\r\n let files: Record<string, Uint8Array>;\r\n try {\r\n files = unzipSync(bytes);\r\n } catch (err) {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"corrupt\",\r\n detail: err instanceof Error ? err.message : \"could not read the ZIP container\",\r\n };\r\n }\r\n\r\n const e = extension.toLowerCase();\r\n if (e.startsWith(\"xls\")) return extractXlsx(files);\r\n\r\n // docx/pptx: the text runs, in document order. Small enough that a\r\n // whole-document scan is fine \u2014 unlike a worksheet, which is not.\r\n const parts = Object.keys(files).filter((n) =>\r\n e.startsWith(\"doc\") ? n === \"word/document.xml\" : /^ppt\\/slides\\/slide\\d+\\.xml$/.test(n),\r\n );\r\n if (parts.length === 0) {\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail: \"no document part inside the container\" };\r\n }\r\n\r\n let text = \"\";\r\n for (const part of parts.sort()) {\r\n text += `${textRuns(new TextDecoder().decode(files[part]!))}\\n`;\r\n if (text.length > MAX_TEXT_CHARS) break;\r\n }\r\n const truncated = text.length > MAX_TEXT_CHARS;\r\n return { kind: \"text\", text: truncated ? text.slice(0, MAX_TEXT_CHARS) : text, truncated };\r\n}\r\n\r\n/** `<w:t>` / `<a:t>` runs, joined. A regex, because a DOM here buys nothing. */\r\nfunction textRuns(xml: string): string {\r\n const out: string[] = [];\r\n for (const m of xml.matchAll(/<(?:w|a):t(?:\\s[^>]*)?>([\\s\\S]*?)<\\/(?:w|a):t>/g)) {\r\n out.push(unescapeXml(m[1] ?? \"\"));\r\n }\r\n return out.join(\" \");\r\n}\r\n\r\nfunction unescapeXml(s: string): string {\r\n return s\r\n .replace(/&lt;/g, \"<\")\r\n .replace(/&gt;/g, \">\")\r\n .replace(/&quot;/g, '\"')\r\n .replace(/&apos;/g, \"'\")\r\n .replace(/&#(\\d+);/g, (_, d: string) => String.fromCodePoint(Number(d)))\r\n .replace(/&amp;/g, \"&\");\r\n}\r\n\r\n/**\r\n * Worksheets, cell by cell, WITHOUT building a DOM.\r\n *\r\n * \u2500\u2500 WHY THIS IS HAND-WRITTEN AND NOT `fast-xml-parser` \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * A worksheet's `sheet1.xml` is routinely tens or hundreds of megabytes \u2014 it is\r\n * one XML element per cell. Handing that to a DOM parser allocates an object\r\n * graph many times the file size and will OOM the collector on the exact files\r\n * that matter most, which are the big ones. A linear scan for `<c \u2026>\u2026</c>` costs\r\n * one pass and constant memory.\r\n *\r\n * \u2500\u2500 AND WHY CELLS AND NOT JUST TEXT \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * The COLUMN HEADER is the single strongest precision signal the classifier has.\r\n * Most real SSN volume on a share is in a spreadsheet column headed `SSN`, and a\r\n * header match lifts a bare nine-digit value \u2014 which the catalog otherwise drops\r\n * outright \u2014 to high confidence. In the other direction it vetoes a Luhn-passing\r\n * value under `Invoice`. Flattening a worksheet to a blob of text throws that\r\n * away, so the cells are carried structurally.\r\n */\r\nfunction extractXlsx(files: Record<string, Uint8Array>): ExtractOutcome {\r\n const sharedStrings = files[\"xl/sharedStrings.xml\"]\r\n ? [...new TextDecoder().decode(files[\"xl/sharedStrings.xml\"]).matchAll(/<si>([\\s\\S]*?)<\\/si>/g)].map(\r\n (m) => textRunsGeneric(m[1] ?? \"\"),\r\n )\r\n : [];\r\n\r\n const sheetNames = sheetNameMap(files);\r\n const cells: TabularCell[] = [];\r\n let text = \"\";\r\n\r\n const sheetParts = Object.keys(files)\r\n .filter((n) => /^xl\\/worksheets\\/sheet\\d+\\.xml$/.test(n))\r\n .sort();\r\n\r\n if (sheetParts.length === 0) {\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail: \"no worksheet part inside the container\" };\r\n }\r\n\r\n for (const part of sheetParts) {\r\n const index = Number(/sheet(\\d+)\\.xml$/.exec(part)?.[1] ?? \"1\");\r\n const sheet = sheetNames[index - 1] ?? `Sheet${index}`;\r\n const xml = new TextDecoder().decode(files[part]!);\r\n\r\n /** Row 1's values, keyed by column index \u2014 the header row. */\r\n const headers = new Map<number, string>();\r\n\r\n for (const m of xml.matchAll(/<c\\s+r=\"([A-Z]+)(\\d+)\"([^>]*)>([\\s\\S]*?)<\\/c>/g)) {\r\n const column = columnIndex(m[1]!);\r\n const row = Number(m[2]);\r\n const attrs = m[3] ?? \"\";\r\n const inner = m[4] ?? \"\";\r\n\r\n const raw = /<v>([\\s\\S]*?)<\\/v>/.exec(inner)?.[1] ?? \"\";\r\n const isShared = /\\bt=\"s\"/.test(attrs);\r\n const isInline = /\\bt=\"inlineStr\"/.test(attrs);\r\n\r\n const value = isShared\r\n ? (sharedStrings[Number(raw)] ?? \"\")\r\n : isInline\r\n ? textRunsGeneric(inner)\r\n : unescapeXml(raw);\r\n\r\n if (value.length === 0) continue;\r\n\r\n if (row === 1) {\r\n headers.set(column, value);\r\n continue;\r\n }\r\n\r\n cells.push({ sheet, row, column, header: headers.get(column) ?? null, value });\r\n text += `${value}\\n`;\r\n if (text.length > MAX_TEXT_CHARS) break;\r\n }\r\n if (text.length > MAX_TEXT_CHARS) break;\r\n }\r\n\r\n const truncated = text.length > MAX_TEXT_CHARS;\r\n return {\r\n kind: \"text\",\r\n text: truncated ? text.slice(0, MAX_TEXT_CHARS) : text,\r\n truncated,\r\n cells,\r\n };\r\n}\r\n\r\n/** `<t>` runs without a namespace prefix, as used inside sharedStrings. */\r\nfunction textRunsGeneric(xml: string): string {\r\n const out: string[] = [];\r\n for (const m of xml.matchAll(/<t(?:\\s[^>]*)?>([\\s\\S]*?)<\\/t>/g)) out.push(unescapeXml(m[1] ?? \"\"));\r\n return out.join(\"\");\r\n}\r\n\r\n/** Sheet display names in workbook order, so a hint reads `Payroll!C12`. */\r\nfunction sheetNameMap(files: Record<string, Uint8Array>): string[] {\r\n const wb = files[\"xl/workbook.xml\"];\r\n if (!wb) return [];\r\n return [...new TextDecoder().decode(wb).matchAll(/<sheet\\b[^>]*\\bname=\"([^\"]*)\"/g)].map((m) =>\r\n unescapeXml(m[1] ?? \"\"),\r\n );\r\n}\r\n\r\n/** `A` -> 0, `AA` -> 26. */\r\nexport function columnIndex(label: string): number {\r\n let n = 0;\r\n for (const c of label) n = n * 26 + (c.charCodeAt(0) - 64);\r\n return n - 1;\r\n}\r\n\r\n/**\r\n * Split CSV/TSV text into cells, so a delimited file gets the same column-header\r\n * treatment a spreadsheet does.\r\n *\r\n * Deliberately not a full RFC 4180 parser \u2014 the classifier needs cell VALUES and\r\n * their headers, not a faithful round-trip, and a mis-split cell costs a little\r\n * context rather than a wrong answer. `physicalWrite.ts` makes the same call for\r\n * the same reason.\r\n */\r\nexport function csvCells(text: string, delimiter = \",\"): TabularCell[] {\r\n const lines = text.split(/\\r?\\n/).filter((l) => l.length > 0);\r\n if (lines.length < 2) return [];\r\n const headers = splitLine(lines[0]!, delimiter);\r\n const cells: TabularCell[] = [];\r\n for (let r = 1; r < lines.length; r++) {\r\n const values = splitLine(lines[r]!, delimiter);\r\n for (let c = 0; c < values.length; c++) {\r\n const value = values[c]!.trim();\r\n if (value.length === 0) continue;\r\n cells.push({ sheet: \"\", row: r + 1, column: c, header: headers[c]?.trim() ?? null, value });\r\n }\r\n }\r\n return cells;\r\n}\r\n\r\nfunction splitLine(line: string, delimiter: string): string[] {\r\n const out: string[] = [];\r\n let cur = \"\";\r\n let quoted = false;\r\n for (let i = 0; i < line.length; i++) {\r\n const ch = line[i]!;\r\n if (ch === '\"') {\r\n if (quoted && line[i + 1] === '\"') {\r\n cur += '\"';\r\n i++;\r\n } else quoted = !quoted;\r\n } else if (ch === delimiter && !quoted) {\r\n out.push(cur);\r\n cur = \"\";\r\n } else cur += ch;\r\n }\r\n out.push(cur);\r\n return out;\r\n}\r\n\n/**\n * The reason to report for an archive nothing could be read from.\n *\n * Picks the MOST COMMON, not the first: an archive of four hundred images and\n * one corrupt entry is an archive of images, and reporting it as corrupt sends\n * an operator looking for damage that is not there. Ties break toward the\n * alphabetically first reason so the answer is stable between runs.\n */\nfunction firstReason(unread: Record<string, number>): UnreadableReason {\n const ranked = Object.entries(unread).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]));\n const top = ranked[0]?.[0];\n // Everything `archive.ts` records is either an UnreadableReason already or a\n // path/size rejection, which is a refusal to read rather than a failure to.\n const known: readonly string[] = [\n \"encrypted\",\n \"corrupt\",\n \"timeout\",\n \"access-denied\",\n \"in-use\",\n \"no-text-layer\",\n \"unsupported-format\",\n \"too-large\",\n \"offline-stub\",\n ];\n return (top && known.includes(top) ? top : \"unsupported-format\") as UnreadableReason;\n}\n\n/** Every reason and its count, for the detail line an operator reads. */\nfunction describeUnread(unread: Record<string, number>): string {\n const parts = Object.entries(unread)\n .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))\n .map(([reason, n]) => `${n} ${reason}`);\n return parts.length > 0 ? `nothing readable inside: ${parts.join(\", \")}` : \"archive holds no readable files\";\n}\n", "/**\r\n * Deciding what bytes actually say (PLAN-010 Phase 3).\r\n *\r\n * \u2500\u2500 THE BUG THIS FILE EXISTS TO PREVENT \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * `looksBinary()` in `@netdoc/config-ir` returns true on the first NUL byte, and\r\n * that is exactly right for the job it was written for: a Cisco configuration is\r\n * ASCII, and a NUL means somebody handed the importer a `.pcap`.\r\n *\r\n * It is exactly wrong here. **UTF-16LE text is roughly half NUL bytes** \u2014 every\r\n * ASCII character is stored as `XX 00` \u2014 and UTF-16LE is what `Out-File`\r\n * produces by default, what `Set-Content` produced for years, what Notepad\r\n * writes when you pick \"Unicode\", and what a large share of Windows log and\r\n * report exporters emit. Reusing `looksBinary` on raw bytes would have silently\r\n * skipped a substantial fraction of exactly the Windows text files this module\r\n * exists to read, with no error and no count \u2014 the files would simply never\r\n * appear in any total.\r\n *\r\n * So the order is: sniff, decode, and only THEN ask whether the result looks\r\n * like text. Judging encoded bytes by a rule written for ASCII is the mistake;\r\n * judging the decoded string is the fix.\r\n *\r\n * `config-ir`'s `looksBinary` is deliberately not changed \u2014 its callers want the\r\n * old behaviour, and widening it would loosen a guard that is correct where it\r\n * is used.\r\n */\r\n\r\nexport type Encoding = \"utf-8\" | \"utf-16le\" | \"utf-16be\" | \"windows-1252\" | \"binary\";\r\n\r\nexport interface SniffResult {\r\n encoding: Encoding;\r\n /** Bytes to skip: the byte-order mark, when one was present. */\r\n bomLength: number;\r\n /** How the decision was reached, for the unreadable-file report. */\r\n reason: string;\r\n}\r\n\r\n/** Bytes inspected when guessing. Enough to be confident, small enough to be free. */\r\nconst SAMPLE_BYTES = 4096;\r\n\r\n/**\r\n * Identify the encoding of a byte buffer.\r\n *\r\n * A BOM is decisive \u2014 it is the file saying what it is, and second-guessing it\r\n * is how a correctly-labelled file gets mangled. Everything after that is a\r\n * heuristic, and each one is written to fail toward \"text\" rather than\r\n * \"binary\": a file wrongly treated as text yields garbage the classifier finds\r\n * nothing in, which costs one wasted read. A file wrongly treated as binary is\r\n * never examined at all and never appears in any count, which is the failure\r\n * mode that makes a compliance report quietly incomplete.\r\n */\r\nexport function sniffEncoding(bytes: Uint8Array): SniffResult {\r\n if (bytes.length === 0) return { encoding: \"utf-8\", bomLength: 0, reason: \"empty file\" };\r\n\r\n if (bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf) {\r\n return { encoding: \"utf-8\", bomLength: 3, reason: \"UTF-8 byte-order mark\" };\r\n }\r\n if (bytes.length >= 2 && bytes[0] === 0xff && bytes[1] === 0xfe) {\r\n return { encoding: \"utf-16le\", bomLength: 2, reason: \"UTF-16LE byte-order mark\" };\r\n }\r\n if (bytes.length >= 2 && bytes[0] === 0xfe && bytes[1] === 0xff) {\r\n return { encoding: \"utf-16be\", bomLength: 2, reason: \"UTF-16BE byte-order mark\" };\r\n }\r\n\r\n const sample = bytes.subarray(0, Math.min(SAMPLE_BYTES, bytes.length));\r\n\r\n // BOM-less UTF-16. PowerShell's redirection operators have emitted this, and\r\n // so do plenty of exporters. The signature is unmistakable once you look for\r\n // it rather than for \"contains NUL\": NULs land on alternating offsets, and on\r\n // only ONE of the two parities.\r\n const parity = nulParity(sample);\r\n if (parity !== null) {\r\n return {\r\n encoding: parity === \"odd\" ? \"utf-16le\" : \"utf-16be\",\r\n bomLength: 0,\r\n reason: `NUL bytes on ${parity} offsets only \u2014 BOM-less UTF-16`,\r\n };\r\n }\r\n\r\n // Now, and only now, is a NUL evidence of binary content.\r\n let nuls = 0;\r\n let controls = 0;\r\n for (const b of sample) {\r\n if (b === 0) nuls++;\r\n // Control characters that are not tab, newline or carriage return.\r\n else if (b < 0x09 || (b > 0x0d && b < 0x20)) controls++;\r\n }\r\n if (nuls > 0) {\r\n return { encoding: \"binary\", bomLength: 0, reason: \"NUL bytes at both parities\" };\r\n }\r\n if (controls / sample.length > 0.1) {\r\n return { encoding: \"binary\", bomLength: 0, reason: \"more than 10% control bytes\" };\r\n }\r\n\r\n // UTF-8 unless the bytes disprove it. `windows-1252` is the fallback because\r\n // it is what a Windows estate produces when it is not producing UTF-8, and\r\n // decoding those bytes as UTF-8 turns every accented name into a replacement\r\n // character \u2014 which matters when the name is the thing being searched for.\r\n return isValidUtf8(sample)\r\n ? { encoding: \"utf-8\", bomLength: 0, reason: \"valid UTF-8 byte sequences\" }\r\n : { encoding: \"windows-1252\", bomLength: 0, reason: \"invalid UTF-8 \u2014 assuming Windows-1252\" };\r\n}\r\n\r\n/**\r\n * Which offset parity the NUL bytes sit on, or null if they are on both / too\r\n * few to judge.\r\n *\r\n * UTF-16LE stores ASCII as `XX 00`, so the NULs are at odd offsets; UTF-16BE is\r\n * `00 XX`, so they are at even ones. A file with NULs at both parities is not\r\n * UTF-16 text \u2014 it is binary that happens to contain zeroes.\r\n */\r\nfunction nulParity(sample: Uint8Array): \"odd\" | \"even\" | null {\r\n let odd = 0;\r\n let even = 0;\r\n for (let i = 0; i < sample.length; i++) {\r\n if (sample[i] !== 0) continue;\r\n if (i % 2 === 0) even++;\r\n else odd++;\r\n }\r\n const total = odd + even;\r\n // A third of the sample being NUL is the signature of half-ASCII UTF-16. A\r\n // handful of stray zeroes is not, and guessing UTF-16 on those would mangle an\r\n // ordinary file.\r\n if (total < sample.length / 4) return null;\r\n if (odd > 0 && even === 0) return \"odd\";\r\n if (even > 0 && odd === 0) return \"even\";\r\n return null;\r\n}\r\n\r\n/** Strict UTF-8 validation \u2014 the continuation bytes have to be right. */\r\nfunction isValidUtf8(bytes: Uint8Array): boolean {\r\n let i = 0;\r\n while (i < bytes.length) {\r\n const b = bytes[i]!;\r\n let extra: number;\r\n if (b < 0x80) extra = 0;\r\n else if ((b & 0xe0) === 0xc0) extra = 1;\r\n else if ((b & 0xf0) === 0xe0) extra = 2;\r\n else if ((b & 0xf8) === 0xf0) extra = 3;\r\n else return false;\r\n\r\n // A truncated sequence at the very end of the SAMPLE is not evidence of\r\n // anything \u2014 the file continues past where we stopped looking.\r\n if (i + extra >= bytes.length) return true;\r\n for (let k = 1; k <= extra; k++) {\r\n if ((bytes[i + k]! & 0xc0) !== 0x80) return false;\r\n }\r\n i += extra + 1;\r\n }\r\n return true;\r\n}\r\n\r\n/**\r\n * Decode to a string, or null when the bytes are not text.\r\n *\r\n * `fatal: false` throughout: a single malformed sequence in a large document\r\n * should cost that character, not the whole file. A report that dropped a\r\n * 40 MB spreadsheet because one cell held a stray byte would be exactly the\r\n * kind of silent incompleteness this module is built to avoid.\r\n */\r\nexport function decodeText(bytes: Uint8Array): { text: string; encoding: Encoding } | null {\r\n const sniff = sniffEncoding(bytes);\r\n if (sniff.encoding === \"binary\") return null;\r\n\r\n const body = sniff.bomLength > 0 ? bytes.subarray(sniff.bomLength) : bytes;\r\n try {\r\n const text = new TextDecoder(sniff.encoding, { fatal: false }).decode(body);\r\n return { text, encoding: sniff.encoding };\r\n } catch {\r\n // An unsupported label on some runtime. Fall back rather than lose the file.\r\n return { text: new TextDecoder(\"utf-8\", { fatal: false }).decode(body), encoding: \"utf-8\" };\r\n }\r\n}\r\n\r\n/**\r\n * Whether a DECODED string reads as text.\r\n *\r\n * The counterpart to `looksBinary`, applied on the right side of the decode.\r\n * A high proportion of replacement characters means the bytes were not what the\r\n * sniff concluded \u2014 which is a real outcome for a file with no BOM and no valid\r\n * UTF-8, and the honest response is to stop rather than classify noise.\r\n */\r\nexport function decodedLooksLikeText(text: string): boolean {\r\n if (text.length === 0) return true;\r\n let replacements = 0;\r\n const limit = Math.min(text.length, 4096);\r\n for (let i = 0; i < limit; i++) {\r\n if (text.charCodeAt(i) === 0xfffd) replacements++;\r\n }\r\n return replacements / limit < 0.05;\r\n}\r\n"],
4
+ "sourcesContent": ["import { readFile } from \"node:fs/promises\";\r\nimport { unzipSync } from \"fflate\";\r\nimport { decodeText, decodedLooksLikeText } from \"./encoding.js\";\r\n\r\n/**\r\n * Getting text out of a file (PLAN-010 Phase 3).\r\n *\r\n * \u2500\u2500 EVERY OUTCOME IS NAMED \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * There is no silent skip in this file. A file is text, or it is deliberately\r\n * ignored, or it is `unreadable` WITH A REASON \u2014 and the reason is counted and\r\n * reported per share. \"1,900 password-protected spreadsheets\" and \"1,900 corrupt\r\n * files\" and \"1,900 scanned PDFs with no text layer\" are three different facts\r\n * about an estate, and only one of them is somebody's problem to fix.\r\n *\r\n * Every competing tool collapses all three into \"skipped\". A report that quietly\r\n * omitted a third of the estate is worse than one that says which third, which\r\n * is why the unreadable table is a first-class output rather than a debug log.\r\n *\r\n * \u2500\u2500 WHY `fflate` AND NOT A HAND-ROLLED ZIP READER \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * An OpenXML document is a ZIP, and reading one looks like 250 lines over\r\n * `node:zlib`. It is not: a real implementation must survive zip64, mismatched\r\n * local and central headers, and malformed central directories \u2014 on input that\r\n * arrives from a file share and is therefore attacker-influenced. That is a\r\n * security liability written to save a dependency. `fflate` is ~8 KB, has no\r\n * dependencies of its own, and is runtime-portable, which keeps the option of\r\n * running this same extractor in a Worker later.\r\n */\r\n\r\nexport type UnreadableReason =\r\n | \"encrypted\"\r\n | \"corrupt\"\r\n | \"timeout\"\r\n | \"access-denied\"\r\n | \"in-use\"\r\n | \"no-text-layer\"\r\n | \"unsupported-format\"\r\n | \"too-large\"\r\n | \"binary\";\r\n\r\nexport interface TabularCell {\r\n sheet: string;\r\n row: number;\r\n column: number;\r\n header: string | null;\r\n value: string;\r\n}\r\n\r\nexport type ExtractOutcome =\r\n | {\r\n kind: \"text\";\r\n text: string;\r\n /** True when the extracted text hit the cap. Reported, never silent. */\r\n truncated: boolean;\r\n /** Populated for csv/xlsx. The single biggest precision lever \u2014 see below. */\r\n cells?: TabularCell[];\r\n }\r\n /**\r\n * An archive, as its READABLE MEMBERS rather than as one blob.\r\n *\r\n * Separate from `text` because a finding has to be able to say\r\n * `Records_2019.zip -> registrar/2018.csv`. Concatenating the members would\r\n * report the archive as the file holding an SSN, which is true and useless:\r\n * nobody can act on it without opening the archive themselves.\r\n */\r\n | {\r\n kind: \"archive\";\r\n entries: Array<{ path: string; text: string }>;\r\n /** Members that could not be read, by reason, so the gap is countable. */\r\n unread: Record<string, number>;\r\n truncated: boolean;\r\n }\r\n | { kind: \"skipped\"; reason: \"ignored-extension\" | \"zero-length\" }\r\n | { kind: \"unreadable\"; reason: UnreadableReason; detail: string };\r\n\r\n/** Extracted text is capped; past this the file is classified on its head. */\r\nexport const MAX_TEXT_CHARS = 20 * 1024 * 1024;\r\n\r\nconst PLAIN_TEXT = new Set([\r\n \"txt\", \"csv\", \"tsv\", \"log\", \"json\", \"xml\", \"sql\", \"md\", \"htm\", \"html\", \"ini\", \"cfg\",\r\n \"conf\", \"yaml\", \"yml\", \"ps1\", \"bat\", \"cmd\", \"sh\", \"py\", \"js\", \"ts\", \"css\", \"rtf\",\r\n]);\r\nconst OPENXML = new Set([\"docx\", \"xlsx\", \"pptx\", \"docm\", \"xlsm\", \"pptm\"]);\r\n/**\r\n * Recognised, deliberately not supported yet, and COUNTED rather than ignored.\r\n *\r\n * `pdf` left this set in Phase 6 and now has its own kind. It is worth saying\r\n * why it was the one that earned the work: a scanned PDF extracts to nothing,\r\n * and \"extracted nothing\" and \"contains nothing\" are the same value with\r\n * opposite meanings. Every other format here fails loudly; that one would have\r\n * failed as a clean result. See pdf.ts.\r\n */\r\nconst KNOWN_UNSUPPORTED = new Set([\"doc\", \"xls\", \"ppt\", \"msg\", \"pst\", \"7z\", \"rar\", \"eml\"]);\r\n\r\n/**\r\n * The OLE compound-file signature.\r\n *\r\n * An ECMA-376 encrypted OpenXML document is NOT a ZIP \u2014 it is an OLE container\r\n * holding the encrypted package. Recognising it is the difference between\r\n * reporting \"1,900 password-protected spreadsheets\", which is a finding, and\r\n * \"1,900 corrupt files\", which is noise.\r\n */\r\nconst OLE_MAGIC = [0xd0, 0xcf, 0x11, 0xe0, 0xa1, 0xb1, 0x1a, 0xe1];\r\n\r\nconst isOle = (b: Uint8Array): boolean =>\r\n b.length >= 8 && OLE_MAGIC.every((v, i) => b[i] === v);\r\n\r\nconst isZip = (b: Uint8Array): boolean =>\r\n b.length >= 4 && b[0] === 0x50 && b[1] === 0x4b && (b[2] === 0x03 || b[2] === 0x05 || b[2] === 0x07);\r\n\r\nexport function extensionKind(\r\n ext: string,\r\n): \"text\" | \"openxml\" | \"pdf\" | \"archive\" | \"unsupported\" | \"ignore\" {\r\n const e = ext.toLowerCase().replace(/^\\./, \"\");\r\n if (PLAIN_TEXT.has(e)) return \"text\";\r\n if (OPENXML.has(e)) return \"openxml\";\r\n if (e === \"pdf\") return \"pdf\";\r\n if (e === \"zip\") return \"archive\";\r\n if (KNOWN_UNSUPPORTED.has(e)) return \"unsupported\";\r\n return \"ignore\";\r\n}\r\n\r\nexport async function extractFile(path: string, extension: string): Promise<ExtractOutcome> {\r\n const kind = extensionKind(extension);\r\n if (kind === \"ignore\") return { kind: \"skipped\", reason: \"ignored-extension\" };\r\n if (kind === \"unsupported\") {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"unsupported-format\",\r\n detail: `.${extension} is recognised but not yet extracted \u2014 counted so the coverage gap is visible`,\r\n };\r\n }\r\n\r\n let bytes: Uint8Array;\r\n try {\r\n bytes = new Uint8Array(await readFile(path));\r\n } catch (err) {\r\n const code = (err as NodeJS.ErrnoException | undefined)?.code ?? \"\";\r\n const detail = err instanceof Error ? err.message : String(err);\r\n if (code === \"EACCES\" || code === \"EPERM\") return { kind: \"unreadable\", reason: \"access-denied\", detail };\r\n if (code === \"EBUSY\") return { kind: \"unreadable\", reason: \"in-use\", detail };\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail };\r\n }\r\n\r\n if (bytes.length === 0) return { kind: \"skipped\", reason: \"zero-length\" };\r\n if (kind === \"pdf\") {\r\n // Imported lazily inside the extractor, so a run over a share with no PDFs\r\n // never loads several megabytes of parser. See pdf.ts.\r\n const { extractPdf } = await import(\"./pdf.js\");\r\n return extractPdf(bytes);\r\n }\r\n if (kind === \"archive\") {\r\n // Guarded by @netdoc/config-ir's archive budget \u2014 an archive is\r\n // attacker-controlled input that expands, and is the most dangerous thing\r\n // this module reads. See archive.ts.\r\n const { extractArchive } = await import(\"./archive.js\");\r\n const out = extractArchive(bytes);\r\n // Nothing readable inside is not the same as a corrupt archive, and both\r\n // differ from an archive that was refused by the bomb guard. The reasons\r\n // travel in `unread`; this only decides whether there is text to classify.\r\n return out.entries.length > 0\r\n ? { kind: \"archive\", entries: out.entries, unread: out.unread, truncated: out.truncated }\r\n : {\r\n kind: \"unreadable\",\r\n reason: firstReason(out.unread),\r\n detail: describeUnread(out.unread),\r\n };\r\n }\r\n return kind === \"openxml\" ? extractOpenXml(bytes, extension) : extractPlainText(bytes);\r\n}\r\n\r\nexport function extractPlainText(bytes: Uint8Array): ExtractOutcome {\r\n const decoded = decodeText(bytes);\r\n if (decoded === null) {\r\n return { kind: \"unreadable\", reason: \"binary\", detail: \"not text in any recognised encoding\" };\r\n }\r\n if (!decodedLooksLikeText(decoded.text)) {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"binary\",\r\n detail: `decoded as ${decoded.encoding} but is mostly replacement characters`,\r\n };\r\n }\r\n const truncated = decoded.text.length > MAX_TEXT_CHARS;\r\n return {\r\n kind: \"text\",\r\n text: truncated ? decoded.text.slice(0, MAX_TEXT_CHARS) : decoded.text,\r\n truncated,\r\n };\r\n}\r\n\r\nexport function extractOpenXml(bytes: Uint8Array, extension: string): ExtractOutcome {\r\n if (isOle(bytes)) {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"encrypted\",\r\n detail: \"an OLE container \u2014 an ECMA-376 password-protected document\",\r\n };\r\n }\r\n if (!isZip(bytes)) {\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail: \"not a ZIP container\" };\r\n }\r\n\r\n let files: Record<string, Uint8Array>;\r\n try {\r\n files = unzipSync(bytes);\r\n } catch (err) {\r\n return {\r\n kind: \"unreadable\",\r\n reason: \"corrupt\",\r\n detail: err instanceof Error ? err.message : \"could not read the ZIP container\",\r\n };\r\n }\r\n\r\n const e = extension.toLowerCase();\r\n if (e.startsWith(\"xls\")) return extractXlsx(files);\r\n\r\n // docx/pptx: the text runs, in document order. Small enough that a\r\n // whole-document scan is fine \u2014 unlike a worksheet, which is not.\r\n const parts = Object.keys(files).filter((n) =>\r\n e.startsWith(\"doc\") ? n === \"word/document.xml\" : /^ppt\\/slides\\/slide\\d+\\.xml$/.test(n),\r\n );\r\n if (parts.length === 0) {\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail: \"no document part inside the container\" };\r\n }\r\n\r\n let text = \"\";\r\n for (const part of parts.sort()) {\r\n text += `${textRuns(new TextDecoder().decode(files[part]!))}\\n`;\r\n if (text.length > MAX_TEXT_CHARS) break;\r\n }\r\n const truncated = text.length > MAX_TEXT_CHARS;\r\n return { kind: \"text\", text: truncated ? text.slice(0, MAX_TEXT_CHARS) : text, truncated };\r\n}\r\n\r\n/** `<w:t>` / `<a:t>` runs, joined. A regex, because a DOM here buys nothing. */\r\nfunction textRuns(xml: string): string {\r\n const out: string[] = [];\r\n for (const m of xml.matchAll(/<(?:w|a):t(?:\\s[^>]*)?>([\\s\\S]*?)<\\/(?:w|a):t>/g)) {\r\n out.push(unescapeXml(m[1] ?? \"\"));\r\n }\r\n return out.join(\" \");\r\n}\r\n\r\nfunction unescapeXml(s: string): string {\r\n return s\r\n .replace(/&lt;/g, \"<\")\r\n .replace(/&gt;/g, \">\")\r\n .replace(/&quot;/g, '\"')\r\n .replace(/&apos;/g, \"'\")\r\n .replace(/&#(\\d+);/g, (_, d: string) => String.fromCodePoint(Number(d)))\r\n .replace(/&amp;/g, \"&\");\r\n}\r\n\r\n/**\r\n * Worksheets, cell by cell, WITHOUT building a DOM.\r\n *\r\n * \u2500\u2500 WHY THIS IS HAND-WRITTEN AND NOT `fast-xml-parser` \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * A worksheet's `sheet1.xml` is routinely tens or hundreds of megabytes \u2014 it is\r\n * one XML element per cell. Handing that to a DOM parser allocates an object\r\n * graph many times the file size and will OOM the collector on the exact files\r\n * that matter most, which are the big ones. A linear scan for `<c \u2026>\u2026</c>` costs\r\n * one pass and constant memory.\r\n *\r\n * \u2500\u2500 AND WHY CELLS AND NOT JUST TEXT \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * The COLUMN HEADER is the single strongest precision signal the classifier has.\r\n * Most real SSN volume on a share is in a spreadsheet column headed `SSN`, and a\r\n * header match lifts a bare nine-digit value \u2014 which the catalog otherwise drops\r\n * outright \u2014 to high confidence. In the other direction it vetoes a Luhn-passing\r\n * value under `Invoice`. Flattening a worksheet to a blob of text throws that\r\n * away, so the cells are carried structurally.\r\n */\r\nfunction extractXlsx(files: Record<string, Uint8Array>): ExtractOutcome {\r\n const sharedStrings = files[\"xl/sharedStrings.xml\"]\r\n ? [...new TextDecoder().decode(files[\"xl/sharedStrings.xml\"]).matchAll(/<si>([\\s\\S]*?)<\\/si>/g)].map(\r\n (m) => textRunsGeneric(m[1] ?? \"\"),\r\n )\r\n : [];\r\n\r\n const sheetNames = sheetNameMap(files);\r\n const cells: TabularCell[] = [];\r\n let text = \"\";\r\n\r\n const sheetParts = Object.keys(files)\r\n .filter((n) => /^xl\\/worksheets\\/sheet\\d+\\.xml$/.test(n))\r\n .sort();\r\n\r\n if (sheetParts.length === 0) {\r\n return { kind: \"unreadable\", reason: \"corrupt\", detail: \"no worksheet part inside the container\" };\r\n }\r\n\r\n for (const part of sheetParts) {\r\n const index = Number(/sheet(\\d+)\\.xml$/.exec(part)?.[1] ?? \"1\");\r\n const sheet = sheetNames[index - 1] ?? `Sheet${index}`;\r\n const xml = new TextDecoder().decode(files[part]!);\r\n\r\n /** Row 1's values, keyed by column index \u2014 the header row. */\r\n const headers = new Map<number, string>();\r\n\r\n for (const m of xml.matchAll(/<c\\s+r=\"([A-Z]+)(\\d+)\"([^>]*)>([\\s\\S]*?)<\\/c>/g)) {\r\n const column = columnIndex(m[1]!);\r\n const row = Number(m[2]);\r\n const attrs = m[3] ?? \"\";\r\n const inner = m[4] ?? \"\";\r\n\r\n const raw = /<v>([\\s\\S]*?)<\\/v>/.exec(inner)?.[1] ?? \"\";\r\n const isShared = /\\bt=\"s\"/.test(attrs);\r\n const isInline = /\\bt=\"inlineStr\"/.test(attrs);\r\n\r\n const value = isShared\r\n ? (sharedStrings[Number(raw)] ?? \"\")\r\n : isInline\r\n ? textRunsGeneric(inner)\r\n : unescapeXml(raw);\r\n\r\n if (value.length === 0) continue;\r\n\r\n if (row === 1) {\r\n headers.set(column, value);\r\n continue;\r\n }\r\n\r\n cells.push({ sheet, row, column, header: headers.get(column) ?? null, value });\r\n text += `${value}\\n`;\r\n if (text.length > MAX_TEXT_CHARS) break;\r\n }\r\n if (text.length > MAX_TEXT_CHARS) break;\r\n }\r\n\r\n const truncated = text.length > MAX_TEXT_CHARS;\r\n return {\r\n kind: \"text\",\r\n text: truncated ? text.slice(0, MAX_TEXT_CHARS) : text,\r\n truncated,\r\n cells,\r\n };\r\n}\r\n\r\n/** `<t>` runs without a namespace prefix, as used inside sharedStrings. */\r\nfunction textRunsGeneric(xml: string): string {\r\n const out: string[] = [];\r\n for (const m of xml.matchAll(/<t(?:\\s[^>]*)?>([\\s\\S]*?)<\\/t>/g)) out.push(unescapeXml(m[1] ?? \"\"));\r\n return out.join(\"\");\r\n}\r\n\r\n/** Sheet display names in workbook order, so a hint reads `Payroll!C12`. */\r\nfunction sheetNameMap(files: Record<string, Uint8Array>): string[] {\r\n const wb = files[\"xl/workbook.xml\"];\r\n if (!wb) return [];\r\n return [...new TextDecoder().decode(wb).matchAll(/<sheet\\b[^>]*\\bname=\"([^\"]*)\"/g)].map((m) =>\r\n unescapeXml(m[1] ?? \"\"),\r\n );\r\n}\r\n\r\n/** `A` -> 0, `AA` -> 26. */\r\nexport function columnIndex(label: string): number {\r\n let n = 0;\r\n for (const c of label) n = n * 26 + (c.charCodeAt(0) - 64);\r\n return n - 1;\r\n}\r\n\r\n/**\r\n * Split CSV/TSV text into cells, so a delimited file gets the same column-header\r\n * treatment a spreadsheet does.\r\n *\r\n * Deliberately not a full RFC 4180 parser \u2014 the classifier needs cell VALUES and\r\n * their headers, not a faithful round-trip, and a mis-split cell costs a little\r\n * context rather than a wrong answer. `physicalWrite.ts` makes the same call for\r\n * the same reason.\r\n */\r\nexport function csvCells(text: string, delimiter = \",\"): TabularCell[] {\r\n const lines = text.split(/\\r?\\n/).filter((l) => l.length > 0);\r\n if (lines.length < 2) return [];\r\n const headers = splitLine(lines[0]!, delimiter);\r\n const cells: TabularCell[] = [];\r\n for (let r = 1; r < lines.length; r++) {\r\n const values = splitLine(lines[r]!, delimiter);\r\n for (let c = 0; c < values.length; c++) {\r\n const value = values[c]!.trim();\r\n if (value.length === 0) continue;\r\n cells.push({ sheet: \"\", row: r + 1, column: c, header: headers[c]?.trim() ?? null, value });\r\n }\r\n }\r\n return cells;\r\n}\r\n\r\nfunction splitLine(line: string, delimiter: string): string[] {\r\n const out: string[] = [];\r\n let cur = \"\";\r\n let quoted = false;\r\n for (let i = 0; i < line.length; i++) {\r\n const ch = line[i]!;\r\n if (ch === '\"') {\r\n if (quoted && line[i + 1] === '\"') {\r\n cur += '\"';\r\n i++;\r\n } else quoted = !quoted;\r\n } else if (ch === delimiter && !quoted) {\r\n out.push(cur);\r\n cur = \"\";\r\n } else cur += ch;\r\n }\r\n out.push(cur);\r\n return out;\r\n}\r\n\r\n/**\r\n * The reason to report for an archive nothing could be read from.\r\n *\r\n * Picks the MOST COMMON, not the first: an archive of four hundred images and\r\n * one corrupt entry is an archive of images, and reporting it as corrupt sends\r\n * an operator looking for damage that is not there. Ties break toward the\r\n * alphabetically first reason so the answer is stable between runs.\r\n */\r\nfunction firstReason(unread: Record<string, number>): UnreadableReason {\r\n const ranked = Object.entries(unread).sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]));\r\n const top = ranked[0]?.[0];\r\n // Everything `archive.ts` records is either an UnreadableReason already or a\r\n // path/size rejection, which is a refusal to read rather than a failure to.\r\n const known: readonly string[] = [\r\n \"encrypted\",\r\n \"corrupt\",\r\n \"timeout\",\r\n \"access-denied\",\r\n \"in-use\",\r\n \"no-text-layer\",\r\n \"unsupported-format\",\r\n \"too-large\",\r\n \"offline-stub\",\r\n ];\r\n return (top && known.includes(top) ? top : \"unsupported-format\") as UnreadableReason;\r\n}\r\n\r\n/** Every reason and its count, for the detail line an operator reads. */\r\nfunction describeUnread(unread: Record<string, number>): string {\r\n const parts = Object.entries(unread)\r\n .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))\r\n .map(([reason, n]) => `${n} ${reason}`);\r\n return parts.length > 0 ? `nothing readable inside: ${parts.join(\", \")}` : \"archive holds no readable files\";\r\n}\r\n", "/**\r\n * Deciding what bytes actually say (PLAN-010 Phase 3).\r\n *\r\n * \u2500\u2500 THE BUG THIS FILE EXISTS TO PREVENT \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\r\n *\r\n * `looksBinary()` in `@netdoc/config-ir` returns true on the first NUL byte, and\r\n * that is exactly right for the job it was written for: a Cisco configuration is\r\n * ASCII, and a NUL means somebody handed the importer a `.pcap`.\r\n *\r\n * It is exactly wrong here. **UTF-16LE text is roughly half NUL bytes** \u2014 every\r\n * ASCII character is stored as `XX 00` \u2014 and UTF-16LE is what `Out-File`\r\n * produces by default, what `Set-Content` produced for years, what Notepad\r\n * writes when you pick \"Unicode\", and what a large share of Windows log and\r\n * report exporters emit. Reusing `looksBinary` on raw bytes would have silently\r\n * skipped a substantial fraction of exactly the Windows text files this module\r\n * exists to read, with no error and no count \u2014 the files would simply never\r\n * appear in any total.\r\n *\r\n * So the order is: sniff, decode, and only THEN ask whether the result looks\r\n * like text. Judging encoded bytes by a rule written for ASCII is the mistake;\r\n * judging the decoded string is the fix.\r\n *\r\n * `config-ir`'s `looksBinary` is deliberately not changed \u2014 its callers want the\r\n * old behaviour, and widening it would loosen a guard that is correct where it\r\n * is used.\r\n */\r\n\r\nexport type Encoding = \"utf-8\" | \"utf-16le\" | \"utf-16be\" | \"windows-1252\" | \"binary\";\r\n\r\nexport interface SniffResult {\r\n encoding: Encoding;\r\n /** Bytes to skip: the byte-order mark, when one was present. */\r\n bomLength: number;\r\n /** How the decision was reached, for the unreadable-file report. */\r\n reason: string;\r\n}\r\n\r\n/** Bytes inspected when guessing. Enough to be confident, small enough to be free. */\r\nconst SAMPLE_BYTES = 4096;\r\n\r\n/**\r\n * Identify the encoding of a byte buffer.\r\n *\r\n * A BOM is decisive \u2014 it is the file saying what it is, and second-guessing it\r\n * is how a correctly-labelled file gets mangled. Everything after that is a\r\n * heuristic, and each one is written to fail toward \"text\" rather than\r\n * \"binary\": a file wrongly treated as text yields garbage the classifier finds\r\n * nothing in, which costs one wasted read. A file wrongly treated as binary is\r\n * never examined at all and never appears in any count, which is the failure\r\n * mode that makes a compliance report quietly incomplete.\r\n */\r\nexport function sniffEncoding(bytes: Uint8Array): SniffResult {\r\n if (bytes.length === 0) return { encoding: \"utf-8\", bomLength: 0, reason: \"empty file\" };\r\n\r\n if (bytes.length >= 3 && bytes[0] === 0xef && bytes[1] === 0xbb && bytes[2] === 0xbf) {\r\n return { encoding: \"utf-8\", bomLength: 3, reason: \"UTF-8 byte-order mark\" };\r\n }\r\n if (bytes.length >= 2 && bytes[0] === 0xff && bytes[1] === 0xfe) {\r\n return { encoding: \"utf-16le\", bomLength: 2, reason: \"UTF-16LE byte-order mark\" };\r\n }\r\n if (bytes.length >= 2 && bytes[0] === 0xfe && bytes[1] === 0xff) {\r\n return { encoding: \"utf-16be\", bomLength: 2, reason: \"UTF-16BE byte-order mark\" };\r\n }\r\n\r\n const sample = bytes.subarray(0, Math.min(SAMPLE_BYTES, bytes.length));\r\n\r\n // BOM-less UTF-16. PowerShell's redirection operators have emitted this, and\r\n // so do plenty of exporters. The signature is unmistakable once you look for\r\n // it rather than for \"contains NUL\": NULs land on alternating offsets, and on\r\n // only ONE of the two parities.\r\n const parity = nulParity(sample);\r\n if (parity !== null) {\r\n return {\r\n encoding: parity === \"odd\" ? \"utf-16le\" : \"utf-16be\",\r\n bomLength: 0,\r\n reason: `NUL bytes on ${parity} offsets only \u2014 BOM-less UTF-16`,\r\n };\r\n }\r\n\r\n // Now, and only now, is a NUL evidence of binary content.\r\n let nuls = 0;\r\n let controls = 0;\r\n for (const b of sample) {\r\n if (b === 0) nuls++;\r\n // Control characters that are not tab, newline or carriage return.\r\n else if (b < 0x09 || (b > 0x0d && b < 0x20)) controls++;\r\n }\r\n if (nuls > 0) {\r\n return { encoding: \"binary\", bomLength: 0, reason: \"NUL bytes at both parities\" };\r\n }\r\n if (controls / sample.length > 0.1) {\r\n return { encoding: \"binary\", bomLength: 0, reason: \"more than 10% control bytes\" };\r\n }\r\n\r\n // UTF-8 unless the bytes disprove it. `windows-1252` is the fallback because\r\n // it is what a Windows estate produces when it is not producing UTF-8, and\r\n // decoding those bytes as UTF-8 turns every accented name into a replacement\r\n // character \u2014 which matters when the name is the thing being searched for.\r\n return isValidUtf8(sample)\r\n ? { encoding: \"utf-8\", bomLength: 0, reason: \"valid UTF-8 byte sequences\" }\r\n : { encoding: \"windows-1252\", bomLength: 0, reason: \"invalid UTF-8 \u2014 assuming Windows-1252\" };\r\n}\r\n\r\n/**\r\n * Which offset parity the NUL bytes sit on, or null if they are on both / too\r\n * few to judge.\r\n *\r\n * UTF-16LE stores ASCII as `XX 00`, so the NULs are at odd offsets; UTF-16BE is\r\n * `00 XX`, so they are at even ones. A file with NULs at both parities is not\r\n * UTF-16 text \u2014 it is binary that happens to contain zeroes.\r\n */\r\nfunction nulParity(sample: Uint8Array): \"odd\" | \"even\" | null {\r\n let odd = 0;\r\n let even = 0;\r\n for (let i = 0; i < sample.length; i++) {\r\n if (sample[i] !== 0) continue;\r\n if (i % 2 === 0) even++;\r\n else odd++;\r\n }\r\n const total = odd + even;\r\n // A third of the sample being NUL is the signature of half-ASCII UTF-16. A\r\n // handful of stray zeroes is not, and guessing UTF-16 on those would mangle an\r\n // ordinary file.\r\n if (total < sample.length / 4) return null;\r\n if (odd > 0 && even === 0) return \"odd\";\r\n if (even > 0 && odd === 0) return \"even\";\r\n return null;\r\n}\r\n\r\n/** Strict UTF-8 validation \u2014 the continuation bytes have to be right. */\r\nfunction isValidUtf8(bytes: Uint8Array): boolean {\r\n let i = 0;\r\n while (i < bytes.length) {\r\n const b = bytes[i]!;\r\n let extra: number;\r\n if (b < 0x80) extra = 0;\r\n else if ((b & 0xe0) === 0xc0) extra = 1;\r\n else if ((b & 0xf0) === 0xe0) extra = 2;\r\n else if ((b & 0xf8) === 0xf0) extra = 3;\r\n else return false;\r\n\r\n // A truncated sequence at the very end of the SAMPLE is not evidence of\r\n // anything \u2014 the file continues past where we stopped looking.\r\n if (i + extra >= bytes.length) return true;\r\n for (let k = 1; k <= extra; k++) {\r\n if ((bytes[i + k]! & 0xc0) !== 0x80) return false;\r\n }\r\n i += extra + 1;\r\n }\r\n return true;\r\n}\r\n\r\n/**\r\n * Decode to a string, or null when the bytes are not text.\r\n *\r\n * `fatal: false` throughout: a single malformed sequence in a large document\r\n * should cost that character, not the whole file. A report that dropped a\r\n * 40 MB spreadsheet because one cell held a stray byte would be exactly the\r\n * kind of silent incompleteness this module is built to avoid.\r\n */\r\nexport function decodeText(bytes: Uint8Array): { text: string; encoding: Encoding } | null {\r\n const sniff = sniffEncoding(bytes);\r\n if (sniff.encoding === \"binary\") return null;\r\n\r\n const body = sniff.bomLength > 0 ? bytes.subarray(sniff.bomLength) : bytes;\r\n try {\r\n const text = new TextDecoder(sniff.encoding, { fatal: false }).decode(body);\r\n return { text, encoding: sniff.encoding };\r\n } catch {\r\n // An unsupported label on some runtime. Fall back rather than lose the file.\r\n return { text: new TextDecoder(\"utf-8\", { fatal: false }).decode(body), encoding: \"utf-8\" };\r\n }\r\n}\r\n\r\n/**\r\n * Whether a DECODED string reads as text.\r\n *\r\n * The counterpart to `looksBinary`, applied on the right side of the decode.\r\n * A high proportion of replacement characters means the bytes were not what the\r\n * sniff concluded \u2014 which is a real outcome for a file with no BOM and no valid\r\n * UTF-8, and the honest response is to stop rather than classify noise.\r\n */\r\nexport function decodedLooksLikeText(text: string): boolean {\r\n if (text.length === 0) return true;\r\n let replacements = 0;\r\n const limit = Math.min(text.length, 4096);\r\n for (let i = 0; i < limit; i++) {\r\n if (text.charCodeAt(i) === 0xfffd) replacements++;\r\n }\r\n return replacements / limit < 0.05;\r\n}\r\n"],
5
5
  "mappings": ";AAAA,SAAS,gBAAgB;AACzB,SAAS,iBAAiB;;;ACqC1B,IAAM,eAAe;AAad,SAAS,cAAc,OAAgC;AAC5D,MAAI,MAAM,WAAW,EAAG,QAAO,EAAE,UAAU,SAAS,WAAW,GAAG,QAAQ,aAAa;AAEvF,MAAI,MAAM,UAAU,KAAK,MAAM,CAAC,MAAM,OAAQ,MAAM,CAAC,MAAM,OAAQ,MAAM,CAAC,MAAM,KAAM;AACpF,WAAO,EAAE,UAAU,SAAS,WAAW,GAAG,QAAQ,wBAAwB;AAAA,EAC5E;AACA,MAAI,MAAM,UAAU,KAAK,MAAM,CAAC,MAAM,OAAQ,MAAM,CAAC,MAAM,KAAM;AAC/D,WAAO,EAAE,UAAU,YAAY,WAAW,GAAG,QAAQ,2BAA2B;AAAA,EAClF;AACA,MAAI,MAAM,UAAU,KAAK,MAAM,CAAC,MAAM,OAAQ,MAAM,CAAC,MAAM,KAAM;AAC/D,WAAO,EAAE,UAAU,YAAY,WAAW,GAAG,QAAQ,2BAA2B;AAAA,EAClF;AAEA,QAAM,SAAS,MAAM,SAAS,GAAG,KAAK,IAAI,cAAc,MAAM,MAAM,CAAC;AAMrE,QAAM,SAAS,UAAU,MAAM;AAC/B,MAAI,WAAW,MAAM;AACnB,WAAO;AAAA,MACL,UAAU,WAAW,QAAQ,aAAa;AAAA,MAC1C,WAAW;AAAA,MACX,QAAQ,gBAAgB,MAAM;AAAA,IAChC;AAAA,EACF;AAGA,MAAI,OAAO;AACX,MAAI,WAAW;AACf,aAAW,KAAK,QAAQ;AACtB,QAAI,MAAM,EAAG;AAAA,aAEJ,IAAI,KAAS,IAAI,MAAQ,IAAI,GAAO;AAAA,EAC/C;AACA,MAAI,OAAO,GAAG;AACZ,WAAO,EAAE,UAAU,UAAU,WAAW,GAAG,QAAQ,6BAA6B;AAAA,EAClF;AACA,MAAI,WAAW,OAAO,SAAS,KAAK;AAClC,WAAO,EAAE,UAAU,UAAU,WAAW,GAAG,QAAQ,8BAA8B;AAAA,EACnF;AAMA,SAAO,YAAY,MAAM,IACrB,EAAE,UAAU,SAAS,WAAW,GAAG,QAAQ,6BAA6B,IACxE,EAAE,UAAU,gBAAgB,WAAW,GAAG,QAAQ,6CAAwC;AAChG;AAUA,SAAS,UAAU,QAA2C;AAC5D,MAAI,MAAM;AACV,MAAI,OAAO;AACX,WAAS,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK;AACtC,QAAI,OAAO,CAAC,MAAM,EAAG;AACrB,QAAI,IAAI,MAAM,EAAG;AAAA,QACZ;AAAA,EACP;AACA,QAAM,QAAQ,MAAM;AAIpB,MAAI,QAAQ,OAAO,SAAS,EAAG,QAAO;AACtC,MAAI,MAAM,KAAK,SAAS,EAAG,QAAO;AAClC,MAAI,OAAO,KAAK,QAAQ,EAAG,QAAO;AAClC,SAAO;AACT;AAGA,SAAS,YAAY,OAA4B;AAC/C,MAAI,IAAI;AACR,SAAO,IAAI,MAAM,QAAQ;AACvB,UAAM,IAAI,MAAM,CAAC;AACjB,QAAI;AACJ,QAAI,IAAI,IAAM,SAAQ;AAAA,cACZ,IAAI,SAAU,IAAM,SAAQ;AAAA,cAC5B,IAAI,SAAU,IAAM,SAAQ;AAAA,cAC5B,IAAI,SAAU,IAAM,SAAQ;AAAA,QACjC,QAAO;AAIZ,QAAI,IAAI,SAAS,MAAM,OAAQ,QAAO;AACtC,aAAS,IAAI,GAAG,KAAK,OAAO,KAAK;AAC/B,WAAK,MAAM,IAAI,CAAC,IAAK,SAAU,IAAM,QAAO;AAAA,IAC9C;AACA,SAAK,QAAQ;AAAA,EACf;AACA,SAAO;AACT;AAUO,SAAS,WAAW,OAAgE;AACzF,QAAM,QAAQ,cAAc,KAAK;AACjC,MAAI,MAAM,aAAa,SAAU,QAAO;AAExC,QAAM,OAAO,MAAM,YAAY,IAAI,MAAM,SAAS,MAAM,SAAS,IAAI;AACrE,MAAI;AACF,UAAM,OAAO,IAAI,YAAY,MAAM,UAAU,EAAE,OAAO,MAAM,CAAC,EAAE,OAAO,IAAI;AAC1E,WAAO,EAAE,MAAM,UAAU,MAAM,SAAS;AAAA,EAC1C,QAAQ;AAEN,WAAO,EAAE,MAAM,IAAI,YAAY,SAAS,EAAE,OAAO,MAAM,CAAC,EAAE,OAAO,IAAI,GAAG,UAAU,QAAQ;AAAA,EAC5F;AACF;AAUO,SAAS,qBAAqB,MAAuB;AAC1D,MAAI,KAAK,WAAW,EAAG,QAAO;AAC9B,MAAI,eAAe;AACnB,QAAM,QAAQ,KAAK,IAAI,KAAK,QAAQ,IAAI;AACxC,WAAS,IAAI,GAAG,IAAI,OAAO,KAAK;AAC9B,QAAI,KAAK,WAAW,CAAC,MAAM,MAAQ;AAAA,EACrC;AACA,SAAO,eAAe,QAAQ;AAChC;;;ADjHO,IAAM,iBAAiB,KAAK,OAAO;AAE1C,IAAM,aAAa,oBAAI,IAAI;AAAA,EACzB;AAAA,EAAO;AAAA,EAAO;AAAA,EAAO;AAAA,EAAO;AAAA,EAAQ;AAAA,EAAO;AAAA,EAAO;AAAA,EAAM;AAAA,EAAO;AAAA,EAAQ;AAAA,EAAO;AAAA,EAC9E;AAAA,EAAQ;AAAA,EAAQ;AAAA,EAAO;AAAA,EAAO;AAAA,EAAO;AAAA,EAAO;AAAA,EAAM;AAAA,EAAM;AAAA,EAAM;AAAA,EAAM;AAAA,EAAO;AAC7E,CAAC;AACD,IAAM,UAAU,oBAAI,IAAI,CAAC,QAAQ,QAAQ,QAAQ,QAAQ,QAAQ,MAAM,CAAC;AAUxE,IAAM,oBAAoB,oBAAI,IAAI,CAAC,OAAO,OAAO,OAAO,OAAO,OAAO,MAAM,OAAO,KAAK,CAAC;AAUzF,IAAM,YAAY,CAAC,KAAM,KAAM,IAAM,KAAM,KAAM,KAAM,IAAM,GAAI;AAEjE,IAAM,QAAQ,CAAC,MACb,EAAE,UAAU,KAAK,UAAU,MAAM,CAAC,GAAG,MAAM,EAAE,CAAC,MAAM,CAAC;AAEvD,IAAM,QAAQ,CAAC,MACb,EAAE,UAAU,KAAK,EAAE,CAAC,MAAM,MAAQ,EAAE,CAAC,MAAM,OAAS,EAAE,CAAC,MAAM,KAAQ,EAAE,CAAC,MAAM,KAAQ,EAAE,CAAC,MAAM;AAE1F,SAAS,cACd,KACmE;AACnE,QAAM,IAAI,IAAI,YAAY,EAAE,QAAQ,OAAO,EAAE;AAC7C,MAAI,WAAW,IAAI,CAAC,EAAG,QAAO;AAC9B,MAAI,QAAQ,IAAI,CAAC,EAAG,QAAO;AAC3B,MAAI,MAAM,MAAO,QAAO;AACxB,MAAI,MAAM,MAAO,QAAO;AACxB,MAAI,kBAAkB,IAAI,CAAC,EAAG,QAAO;AACrC,SAAO;AACT;AAEA,eAAsB,YAAY,MAAc,WAA4C;AAC1F,QAAM,OAAO,cAAc,SAAS;AACpC,MAAI,SAAS,SAAU,QAAO,EAAE,MAAM,WAAW,QAAQ,oBAAoB;AAC7E,MAAI,SAAS,eAAe;AAC1B,WAAO;AAAA,MACL,MAAM;AAAA,MACN,QAAQ;AAAA,MACR,QAAQ,IAAI,SAAS;AAAA,IACvB;AAAA,EACF;AAEA,MAAI;AACJ,MAAI;AACF,YAAQ,IAAI,WAAW,MAAM,SAAS,IAAI,CAAC;AAAA,EAC7C,SAAS,KAAK;AACZ,UAAM,OAAQ,KAA2C,QAAQ;AACjE,UAAM,SAAS,eAAe,QAAQ,IAAI,UAAU,OAAO,GAAG;AAC9D,QAAI,SAAS,YAAY,SAAS,QAAS,QAAO,EAAE,MAAM,cAAc,QAAQ,iBAAiB,OAAO;AACxG,QAAI,SAAS,QAAS,QAAO,EAAE,MAAM,cAAc,QAAQ,UAAU,OAAO;AAC5E,WAAO,EAAE,MAAM,cAAc,QAAQ,WAAW,OAAO;AAAA,EACzD;AAEA,MAAI,MAAM,WAAW,EAAG,QAAO,EAAE,MAAM,WAAW,QAAQ,cAAc;AACxE,MAAI,SAAS,OAAO;AAGlB,UAAM,EAAE,WAAW,IAAI,MAAM,OAAO,mBAAU;AAC9C,WAAO,WAAW,KAAK;AAAA,EACzB;AACA,MAAI,SAAS,WAAW;AAItB,UAAM,EAAE,eAAe,IAAI,MAAM,OAAO,uBAAc;AACtD,UAAM,MAAM,eAAe,KAAK;AAIhC,WAAO,IAAI,QAAQ,SAAS,IACxB,EAAE,MAAM,WAAW,SAAS,IAAI,SAAS,QAAQ,IAAI,QAAQ,WAAW,IAAI,UAAU,IACtF;AAAA,MACE,MAAM;AAAA,MACN,QAAQ,YAAY,IAAI,MAAM;AAAA,MAC9B,QAAQ,eAAe,IAAI,MAAM;AAAA,IACnC;AAAA,EACN;AACA,SAAO,SAAS,YAAY,eAAe,OAAO,SAAS,IAAI,iBAAiB,KAAK;AACvF;AAEO,SAAS,iBAAiB,OAAmC;AAClE,QAAM,UAAU,WAAW,KAAK;AAChC,MAAI,YAAY,MAAM;AACpB,WAAO,EAAE,MAAM,cAAc,QAAQ,UAAU,QAAQ,sCAAsC;AAAA,EAC/F;AACA,MAAI,CAAC,qBAAqB,QAAQ,IAAI,GAAG;AACvC,WAAO;AAAA,MACL,MAAM;AAAA,MACN,QAAQ;AAAA,MACR,QAAQ,cAAc,QAAQ,QAAQ;AAAA,IACxC;AAAA,EACF;AACA,QAAM,YAAY,QAAQ,KAAK,SAAS;AACxC,SAAO;AAAA,IACL,MAAM;AAAA,IACN,MAAM,YAAY,QAAQ,KAAK,MAAM,GAAG,cAAc,IAAI,QAAQ;AAAA,IAClE;AAAA,EACF;AACF;AAEO,SAAS,eAAe,OAAmB,WAAmC;AACnF,MAAI,MAAM,KAAK,GAAG;AAChB,WAAO;AAAA,MACL,MAAM;AAAA,MACN,QAAQ;AAAA,MACR,QAAQ;AAAA,IACV;AAAA,EACF;AACA,MAAI,CAAC,MAAM,KAAK,GAAG;AACjB,WAAO,EAAE,MAAM,cAAc,QAAQ,WAAW,QAAQ,sBAAsB;AAAA,EAChF;AAEA,MAAI;AACJ,MAAI;AACF,YAAQ,UAAU,KAAK;AAAA,EACzB,SAAS,KAAK;AACZ,WAAO;AAAA,MACL,MAAM;AAAA,MACN,QAAQ;AAAA,MACR,QAAQ,eAAe,QAAQ,IAAI,UAAU;AAAA,IAC/C;AAAA,EACF;AAEA,QAAM,IAAI,UAAU,YAAY;AAChC,MAAI,EAAE,WAAW,KAAK,EAAG,QAAO,YAAY,KAAK;AAIjD,QAAM,QAAQ,OAAO,KAAK,KAAK,EAAE;AAAA,IAAO,CAAC,MACvC,EAAE,WAAW,KAAK,IAAI,MAAM,sBAAsB,+BAA+B,KAAK,CAAC;AAAA,EACzF;AACA,MAAI,MAAM,WAAW,GAAG;AACtB,WAAO,EAAE,MAAM,cAAc,QAAQ,WAAW,QAAQ,wCAAwC;AAAA,EAClG;AAEA,MAAI,OAAO;AACX,aAAW,QAAQ,MAAM,KAAK,GAAG;AAC/B,YAAQ,GAAG,SAAS,IAAI,YAAY,EAAE,OAAO,MAAM,IAAI,CAAE,CAAC,CAAC;AAAA;AAC3D,QAAI,KAAK,SAAS,eAAgB;AAAA,EACpC;AACA,QAAM,YAAY,KAAK,SAAS;AAChC,SAAO,EAAE,MAAM,QAAQ,MAAM,YAAY,KAAK,MAAM,GAAG,cAAc,IAAI,MAAM,UAAU;AAC3F;AAGA,SAAS,SAAS,KAAqB;AACrC,QAAM,MAAgB,CAAC;AACvB,aAAW,KAAK,IAAI,SAAS,iDAAiD,GAAG;AAC/E,QAAI,KAAK,YAAY,EAAE,CAAC,KAAK,EAAE,CAAC;AAAA,EAClC;AACA,SAAO,IAAI,KAAK,GAAG;AACrB;AAEA,SAAS,YAAY,GAAmB;AACtC,SAAO,EACJ,QAAQ,SAAS,GAAG,EACpB,QAAQ,SAAS,GAAG,EACpB,QAAQ,WAAW,GAAG,EACtB,QAAQ,WAAW,GAAG,EACtB,QAAQ,aAAa,CAAC,GAAG,MAAc,OAAO,cAAc,OAAO,CAAC,CAAC,CAAC,EACtE,QAAQ,UAAU,GAAG;AAC1B;AAsBA,SAAS,YAAY,OAAmD;AACtE,QAAM,gBAAgB,MAAM,sBAAsB,IAC9C,CAAC,GAAG,IAAI,YAAY,EAAE,OAAO,MAAM,sBAAsB,CAAC,EAAE,SAAS,uBAAuB,CAAC,EAAE;AAAA,IAC7F,CAAC,MAAM,gBAAgB,EAAE,CAAC,KAAK,EAAE;AAAA,EACnC,IACA,CAAC;AAEL,QAAM,aAAa,aAAa,KAAK;AACrC,QAAM,QAAuB,CAAC;AAC9B,MAAI,OAAO;AAEX,QAAM,aAAa,OAAO,KAAK,KAAK,EACjC,OAAO,CAAC,MAAM,kCAAkC,KAAK,CAAC,CAAC,EACvD,KAAK;AAER,MAAI,WAAW,WAAW,GAAG;AAC3B,WAAO,EAAE,MAAM,cAAc,QAAQ,WAAW,QAAQ,yCAAyC;AAAA,EACnG;AAEA,aAAW,QAAQ,YAAY;AAC7B,UAAM,QAAQ,OAAO,mBAAmB,KAAK,IAAI,IAAI,CAAC,KAAK,GAAG;AAC9D,UAAM,QAAQ,WAAW,QAAQ,CAAC,KAAK,QAAQ,KAAK;AACpD,UAAM,MAAM,IAAI,YAAY,EAAE,OAAO,MAAM,IAAI,CAAE;AAGjD,UAAM,UAAU,oBAAI,IAAoB;AAExC,eAAW,KAAK,IAAI,SAAS,gDAAgD,GAAG;AAC9E,YAAM,SAAS,YAAY,EAAE,CAAC,CAAE;AAChC,YAAM,MAAM,OAAO,EAAE,CAAC,CAAC;AACvB,YAAM,QAAQ,EAAE,CAAC,KAAK;AACtB,YAAM,QAAQ,EAAE,CAAC,KAAK;AAEtB,YAAM,MAAM,qBAAqB,KAAK,KAAK,IAAI,CAAC,KAAK;AACrD,YAAM,WAAW,UAAU,KAAK,KAAK;AACrC,YAAM,WAAW,kBAAkB,KAAK,KAAK;AAE7C,YAAM,QAAQ,WACT,cAAc,OAAO,GAAG,CAAC,KAAK,KAC/B,WACE,gBAAgB,KAAK,IACrB,YAAY,GAAG;AAErB,UAAI,MAAM,WAAW,EAAG;AAExB,UAAI,QAAQ,GAAG;AACb,gBAAQ,IAAI,QAAQ,KAAK;AACzB;AAAA,MACF;AAEA,YAAM,KAAK,EAAE,OAAO,KAAK,QAAQ,QAAQ,QAAQ,IAAI,MAAM,KAAK,MAAM,MAAM,CAAC;AAC7E,cAAQ,GAAG,KAAK;AAAA;AAChB,UAAI,KAAK,SAAS,eAAgB;AAAA,IACpC;AACA,QAAI,KAAK,SAAS,eAAgB;AAAA,EACpC;AAEA,QAAM,YAAY,KAAK,SAAS;AAChC,SAAO;AAAA,IACL,MAAM;AAAA,IACN,MAAM,YAAY,KAAK,MAAM,GAAG,cAAc,IAAI;AAAA,IAClD;AAAA,IACA;AAAA,EACF;AACF;AAGA,SAAS,gBAAgB,KAAqB;AAC5C,QAAM,MAAgB,CAAC;AACvB,aAAW,KAAK,IAAI,SAAS,iCAAiC,EAAG,KAAI,KAAK,YAAY,EAAE,CAAC,KAAK,EAAE,CAAC;AACjG,SAAO,IAAI,KAAK,EAAE;AACpB;AAGA,SAAS,aAAa,OAA6C;AACjE,QAAM,KAAK,MAAM,iBAAiB;AAClC,MAAI,CAAC,GAAI,QAAO,CAAC;AACjB,SAAO,CAAC,GAAG,IAAI,YAAY,EAAE,OAAO,EAAE,EAAE,SAAS,gCAAgC,CAAC,EAAE;AAAA,IAAI,CAAC,MACvF,YAAY,EAAE,CAAC,KAAK,EAAE;AAAA,EACxB;AACF;AAGO,SAAS,YAAY,OAAuB;AACjD,MAAI,IAAI;AACR,aAAW,KAAK,MAAO,KAAI,IAAI,MAAM,EAAE,WAAW,CAAC,IAAI;AACvD,SAAO,IAAI;AACb;AAWO,SAAS,SAAS,MAAc,YAAY,KAAoB;AACrE,QAAM,QAAQ,KAAK,MAAM,OAAO,EAAE,OAAO,CAAC,MAAM,EAAE,SAAS,CAAC;AAC5D,MAAI,MAAM,SAAS,EAAG,QAAO,CAAC;AAC9B,QAAM,UAAU,UAAU,MAAM,CAAC,GAAI,SAAS;AAC9C,QAAM,QAAuB,CAAC;AAC9B,WAAS,IAAI,GAAG,IAAI,MAAM,QAAQ,KAAK;AACrC,UAAM,SAAS,UAAU,MAAM,CAAC,GAAI,SAAS;AAC7C,aAAS,IAAI,GAAG,IAAI,OAAO,QAAQ,KAAK;AACtC,YAAM,QAAQ,OAAO,CAAC,EAAG,KAAK;AAC9B,UAAI,MAAM,WAAW,EAAG;AACxB,YAAM,KAAK,EAAE,OAAO,IAAI,KAAK,IAAI,GAAG,QAAQ,GAAG,QAAQ,QAAQ,CAAC,GAAG,KAAK,KAAK,MAAM,MAAM,CAAC;AAAA,IAC5F;AAAA,EACF;AACA,SAAO;AACT;AAEA,SAAS,UAAU,MAAc,WAA6B;AAC5D,QAAM,MAAgB,CAAC;AACvB,MAAI,MAAM;AACV,MAAI,SAAS;AACb,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,KAAK,KAAK,CAAC;AACjB,QAAI,OAAO,KAAK;AACd,UAAI,UAAU,KAAK,IAAI,CAAC,MAAM,KAAK;AACjC,eAAO;AACP;AAAA,MACF,MAAO,UAAS,CAAC;AAAA,IACnB,WAAW,OAAO,aAAa,CAAC,QAAQ;AACtC,UAAI,KAAK,GAAG;AACZ,YAAM;AAAA,IACR,MAAO,QAAO;AAAA,EAChB;AACA,MAAI,KAAK,GAAG;AACZ,SAAO;AACT;AAUA,SAAS,YAAY,QAAkD;AACrE,QAAM,SAAS,OAAO,QAAQ,MAAM,EAAE,KAAK,CAAC,GAAG,MAAM,EAAE,CAAC,IAAI,EAAE,CAAC,KAAK,EAAE,CAAC,EAAE,cAAc,EAAE,CAAC,CAAC,CAAC;AAC5F,QAAM,MAAM,OAAO,CAAC,IAAI,CAAC;AAGzB,QAAM,QAA2B;AAAA,IAC/B;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,IACA;AAAA,EACF;AACA,SAAQ,OAAO,MAAM,SAAS,GAAG,IAAI,MAAM;AAC7C;AAGA,SAAS,eAAe,QAAwC;AAC9D,QAAM,QAAQ,OAAO,QAAQ,MAAM,EAChC,KAAK,CAAC,GAAG,MAAM,EAAE,CAAC,IAAI,EAAE,CAAC,KAAK,EAAE,CAAC,EAAE,cAAc,EAAE,CAAC,CAAC,CAAC,EACtD,IAAI,CAAC,CAAC,QAAQ,CAAC,MAAM,GAAG,CAAC,IAAI,MAAM,EAAE;AACxC,SAAO,MAAM,SAAS,IAAI,4BAA4B,MAAM,KAAK,IAAI,CAAC,KAAK;AAC7E;",
6
6
  "names": []
7
7
  }
@@ -1263,4 +1263,4 @@ export {
1263
1263
  plaintextKeyWarning,
1264
1264
  collectorVersion
1265
1265
  };
1266
- //# sourceMappingURL=chunk-7T2LVP6E.js.map
1266
+ //# sourceMappingURL=chunk-YIKST7R3.js.map