supply-chain-guard 6.5.0 → 6.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +61 -14
- package/action.yml +10 -3
- package/dist/archive-extractor.d.ts +28 -5
- package/dist/archive-extractor.d.ts.map +1 -1
- package/dist/archive-extractor.js +230 -56
- package/dist/archive-extractor.js.map +1 -1
- package/dist/cache-dir.d.ts +26 -0
- package/dist/cache-dir.d.ts.map +1 -1
- package/dist/cache-dir.js +79 -1
- package/dist/cache-dir.js.map +1 -1
- package/dist/catalog-digest.d.ts +4 -4
- package/dist/catalog-digest.js +4 -4
- package/dist/cli.js +9 -3
- package/dist/cli.js.map +1 -1
- package/dist/dependency-confusion.js +1 -1
- package/dist/dependency-governance.d.ts +1 -1
- package/dist/dependency-governance.d.ts.map +1 -1
- package/dist/dependency-governance.js +48 -20
- package/dist/dependency-governance.js.map +1 -1
- package/dist/diff-scanner.d.ts +7 -0
- package/dist/diff-scanner.d.ts.map +1 -1
- package/dist/diff-scanner.js +33 -11
- package/dist/diff-scanner.js.map +1 -1
- package/dist/external-threat-intel.d.ts.map +1 -1
- package/dist/external-threat-intel.js +53 -5
- package/dist/external-threat-intel.js.map +1 -1
- package/dist/extracted-file-walker.d.ts +5 -0
- package/dist/extracted-file-walker.d.ts.map +1 -1
- package/dist/extracted-file-walker.js +4 -1
- package/dist/extracted-file-walker.js.map +1 -1
- package/dist/feed-signing-key.d.ts +34 -0
- package/dist/feed-signing-key.d.ts.map +1 -0
- package/dist/feed-signing-key.js +76 -0
- package/dist/feed-signing-key.js.map +1 -0
- package/dist/feed.d.ts +66 -4
- package/dist/feed.d.ts.map +1 -1
- package/dist/feed.js +299 -46
- package/dist/feed.js.map +1 -1
- package/dist/git-remote-url.d.ts.map +1 -1
- package/dist/git-remote-url.js +53 -1
- package/dist/git-remote-url.js.map +1 -1
- package/dist/github-trust-scanner.js +7 -7
- package/dist/github-trust-scanner.js.map +1 -1
- package/dist/index.d.ts +1 -1
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +4 -3
- package/dist/index.js.map +1 -1
- package/dist/install-guard.d.ts.map +1 -1
- package/dist/install-guard.js +105 -17
- package/dist/install-guard.js.map +1 -1
- package/dist/install-hook-scanner.d.ts.map +1 -1
- package/dist/install-hook-scanner.js +2 -1
- package/dist/install-hook-scanner.js.map +1 -1
- package/dist/internal-disclosure.d.ts.map +1 -1
- package/dist/internal-disclosure.js +17 -4
- package/dist/internal-disclosure.js.map +1 -1
- package/dist/ioc-blocklist.d.ts.map +1 -1
- package/dist/ioc-blocklist.js +28 -1
- package/dist/ioc-blocklist.js.map +1 -1
- package/dist/json-utils.d.ts +6 -0
- package/dist/json-utils.d.ts.map +1 -1
- package/dist/json-utils.js +10 -1
- package/dist/json-utils.js.map +1 -1
- package/dist/lockfile-checker.d.ts.map +1 -1
- package/dist/lockfile-checker.js +141 -20
- package/dist/lockfile-checker.js.map +1 -1
- package/dist/mcp-scanner.d.ts.map +1 -1
- package/dist/mcp-scanner.js +548 -59
- package/dist/mcp-scanner.js.map +1 -1
- package/dist/mcp-server.d.ts +37 -1
- package/dist/mcp-server.d.ts.map +1 -1
- package/dist/mcp-server.js +220 -25
- package/dist/mcp-server.js.map +1 -1
- package/dist/npm-scanner.d.ts.map +1 -1
- package/dist/npm-scanner.js +31 -6
- package/dist/npm-scanner.js.map +1 -1
- package/dist/org-scanner.js +2 -2
- package/dist/org-scanner.js.map +1 -1
- package/dist/pattern-applicability.d.ts +13 -1
- package/dist/pattern-applicability.d.ts.map +1 -1
- package/dist/pattern-applicability.js +73 -3
- package/dist/pattern-applicability.js.map +1 -1
- package/dist/pattern-scanner.d.ts +16 -1
- package/dist/pattern-scanner.d.ts.map +1 -1
- package/dist/pattern-scanner.js +91 -1
- package/dist/pattern-scanner.js.map +1 -1
- package/dist/patterns.d.ts.map +1 -1
- package/dist/patterns.js +83 -77
- package/dist/patterns.js.map +1 -1
- package/dist/policy-engine.d.ts +4 -1
- package/dist/policy-engine.d.ts.map +1 -1
- package/dist/policy-engine.js +13 -4
- package/dist/policy-engine.js.map +1 -1
- package/dist/pypi-scanner.d.ts +9 -0
- package/dist/pypi-scanner.d.ts.map +1 -1
- package/dist/pypi-scanner.js +57 -16
- package/dist/pypi-scanner.js.map +1 -1
- package/dist/regex-complexity.d.ts +16 -0
- package/dist/regex-complexity.d.ts.map +1 -1
- package/dist/regex-complexity.js +73 -0
- package/dist/regex-complexity.js.map +1 -1
- package/dist/reporter.d.ts.map +1 -1
- package/dist/reporter.js +23 -14
- package/dist/reporter.js.map +1 -1
- package/dist/safe-exec.d.ts +33 -0
- package/dist/safe-exec.d.ts.map +1 -0
- package/dist/safe-exec.js +132 -0
- package/dist/safe-exec.js.map +1 -0
- package/dist/sbom-generator.js +1 -1
- package/dist/scan-exclusions.d.ts +61 -0
- package/dist/scan-exclusions.d.ts.map +1 -0
- package/dist/scan-exclusions.js +375 -0
- package/dist/scan-exclusions.js.map +1 -0
- package/dist/scanner.d.ts.map +1 -1
- package/dist/scanner.js +292 -47
- package/dist/scanner.js.map +1 -1
- package/dist/script-language.d.ts +7 -0
- package/dist/script-language.d.ts.map +1 -1
- package/dist/script-language.js +31 -0
- package/dist/script-language.js.map +1 -1
- package/dist/self-scan-files.json +27 -1
- package/dist/slsa-verifier.d.ts +10 -0
- package/dist/slsa-verifier.d.ts.map +1 -1
- package/dist/slsa-verifier.js +144 -8
- package/dist/slsa-verifier.js.map +1 -1
- package/dist/text-decoding.d.ts +20 -0
- package/dist/text-decoding.d.ts.map +1 -0
- package/dist/text-decoding.js +71 -0
- package/dist/text-decoding.js.map +1 -0
- package/dist/threat-intel.d.ts +39 -13
- package/dist/threat-intel.d.ts.map +1 -1
- package/dist/threat-intel.js +295 -59
- package/dist/threat-intel.js.map +1 -1
- package/dist/types.d.ts +17 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/types.js.map +1 -1
- package/dist/vscode-scanner.d.ts.map +1 -1
- package/dist/vscode-scanner.js +6 -4
- package/dist/vscode-scanner.js.map +1 -1
- package/package.json +3 -3
- package/policy-schema.json +1 -1
- package/self-scan-manifest.json +39 -13
package/README.md
CHANGED
|
@@ -30,7 +30,7 @@ Gate every pull request:
|
|
|
30
30
|
|
|
31
31
|
```yaml
|
|
32
32
|
- uses: actions/checkout@v4
|
|
33
|
-
- uses: homeofe/supply-chain-guard@v6.5.
|
|
33
|
+
- uses: homeofe/supply-chain-guard@v6.5.2
|
|
34
34
|
```
|
|
35
35
|
|
|
36
36
|
Let your AI coding agent check a package before it installs it (MCP):
|
|
@@ -204,7 +204,7 @@ Run the scanner as a [pre-commit](https://pre-commit.com) hook (Python-ecosystem
|
|
|
204
204
|
```yaml
|
|
205
205
|
repos:
|
|
206
206
|
- repo: https://github.com/homeofe/supply-chain-guard
|
|
207
|
-
rev: v6.5.
|
|
207
|
+
rev: v6.5.2
|
|
208
208
|
hooks:
|
|
209
209
|
- id: supply-chain-guard
|
|
210
210
|
```
|
|
@@ -236,7 +236,7 @@ The hook scans the repository root on every commit and fails on high or critical
|
|
|
236
236
|
Run the scanner without a Node toolchain via the official multi-arch image (linux/amd64, linux/arm64), published to GHCR on every release tag:
|
|
237
237
|
|
|
238
238
|
```bash
|
|
239
|
-
docker run --rm -v ${PWD}:/scan ghcr.io/homeofe/supply-chain-guard:6.5.
|
|
239
|
+
docker run --rm -v ${PWD}:/scan ghcr.io/homeofe/supply-chain-guard:6.5.2 scan /scan
|
|
240
240
|
```
|
|
241
241
|
|
|
242
242
|
`${PWD}` works in bash, zsh, and PowerShell; in cmd.exe use `%cd%` instead.
|
|
@@ -245,8 +245,8 @@ The image keeps the catalog cache in `/cache` (`SCG_CACHE_DIR`), so it is gone
|
|
|
245
245
|
after `--rm`. To keep it across runs, mount a named volume there:
|
|
246
246
|
|
|
247
247
|
```bash
|
|
248
|
-
docker run --rm -v scg-cache:/cache ghcr.io/homeofe/supply-chain-guard:6.5.
|
|
249
|
-
docker run --rm -v scg-cache:/cache -v ${PWD}:/scan ghcr.io/homeofe/supply-chain-guard:6.5.
|
|
248
|
+
docker run --rm -v scg-cache:/cache ghcr.io/homeofe/supply-chain-guard:6.5.2 feed refresh
|
|
249
|
+
docker run --rm -v scg-cache:/cache -v ${PWD}:/scan ghcr.io/homeofe/supply-chain-guard:6.5.2 scan /scan
|
|
250
250
|
```
|
|
251
251
|
|
|
252
252
|
## Quickstart
|
|
@@ -532,7 +532,7 @@ The external file is one entry per line, `#` for comments, `sha256:<digest>` for
|
|
|
532
532
|
**The two sources are not equally trusted, and the difference is deliberate.** `SCG_INTERNAL_DISCLOSURE_FILE` is set by whoever runs the scan, so it may name any path on the machine and carry any pattern. `internalDisclosure.externalFile` and `internalDisclosure.patterns` live in the committed policy file, which travels inside the repository being scanned, and scanning a repository you do not own is the ordinary case for this tool. Entries from there are therefore bounded:
|
|
533
533
|
|
|
534
534
|
- `externalFile` must stay inside the scanned directory. An absolute path is refused, a relative path that climbs out with `..` is refused, and so is one that leaves through a symbolic link. The file is not opened, so nothing about a path outside the tree reaches the report. The bound is the scanned directory and nothing narrower: a path that stays inside it is still read, `.git/config` included, so a committed `externalFile` can still point at whatever your runner wrote into the workspace. Matches from it stay redacted.
|
|
535
|
-
- A regular expression from `patterns`, **or from an `externalFile` that is inside the tree**, is capped at 200 characters and
|
|
535
|
+
- A regular expression from `patterns`, **or from an `externalFile` that is inside the tree**, is capped at 200 characters and limited to a safe subset: no quantified group at all (`(a+)+`, `(a|a)+`, `(a|ab)*`, `(a+){2,30}`), no backreference and no lookaround. Literals, unquantified groups and quantified single characters or classes (`[a-z0-9.-]+`, `ab[0-9]{3}`) are accepted. A quantified group can take exponential time to report no match, so one committed line would otherwise occupy a runner until the workflow times out. A pattern from the operator's own file or environment variable is not restricted.
|
|
536
536
|
- Whatever survives those checks runs under a wall-clock budget for the whole scan. On overrun the file reports `INTERNAL_DISCLOSURE_TRUNCATED` rather than running on.
|
|
537
537
|
|
|
538
538
|
A refusal is an `INTERNAL_DENYLIST_REFUSED` finding at `medium` severity, and like every other coverage finding it marks the scan partial rather than passing quietly. In the published Action a partial scan exits 1 on its own, independently of `fail-on`. None of this applies to the environment-variable source.
|
|
@@ -612,13 +612,21 @@ baseline:
|
|
|
612
612
|
|
|
613
613
|
Findings can also be suppressed inline with a comment on the line directly
|
|
614
614
|
above them: `// scg-ignore-next-line RULE reason` (JS/TS) or
|
|
615
|
-
`# scg-ignore-next-line RULE` (Python/YAML/shell).
|
|
615
|
+
`# scg-ignore-next-line RULE` (Python/YAML/shell). Inline comments apply to
|
|
616
|
+
local directory scans only, and never hide a `high` or `critical` finding in a
|
|
617
|
+
cloned GitHub repository, an MCP `scan_directory` call, or an npm, PyPI or
|
|
618
|
+
VS Code package scan.
|
|
616
619
|
|
|
617
620
|
### Where the policy is read from, and what that means on a pull request
|
|
618
621
|
|
|
619
|
-
|
|
620
|
-
else. There is no flag, environment variable or
|
|
621
|
-
scanner at a policy outside the scan target.
|
|
622
|
+
For a local directory scan the policy file is read **from the directory being
|
|
623
|
+
scanned**, and from nowhere else. There is no flag, environment variable or
|
|
624
|
+
Action input that points the scanner at a policy outside the scan target.
|
|
625
|
+
|
|
626
|
+
A scan whose target is a cloned GitHub repository URL, and the MCP
|
|
627
|
+
`scan_directory` tool, **never load a policy file from the scanned tree**
|
|
628
|
+
(library callers can set `trustTargetPolicy: false` to get the same). The tree
|
|
629
|
+
there belongs to someone else.
|
|
622
630
|
|
|
623
631
|
On a `pull_request` event the checkout materialises the **head of the proposing
|
|
624
632
|
branch**, so the policy that governs the scan is the one on the branch under
|
|
@@ -639,6 +647,13 @@ than left to be discovered. What it is **not** is silent:
|
|
|
639
647
|
(`POLICY_DISABLE_NO_REASON`, `POLICY_IGNORE_NO_REASON`,
|
|
640
648
|
`POLICY_SUPPRESSION_NO_REASON`), so an undocumented exclusion costs a line in
|
|
641
649
|
the report rather than nothing.
|
|
650
|
+
- When the policy, or an inline `scg-ignore-next-line` comment, removes,
|
|
651
|
+
ignores or downgrades a `high` or `critical` finding, the scan runs a second
|
|
652
|
+
pass without them and the report gains a medium `POLICY_SUPPRESSED_SEVERE`
|
|
653
|
+
finding that lists the rule ids and counts, plus `riskLevelBeforePolicy` and
|
|
654
|
+
`maxSeverityBeforePolicy` (JSON and the "RISK BEFORE POLICY" text line). The
|
|
655
|
+
exit code is not changed by default, because suppressing a false positive is
|
|
656
|
+
the intended use of the mechanism. A scan with neither stays single-pass.
|
|
642
657
|
|
|
643
658
|
If your threat model includes an untrusted proposer, the controls that actually
|
|
644
659
|
hold are outside this tool: require review on `.supply-chain-guard.yml` through
|
|
@@ -1001,7 +1016,7 @@ jobs:
|
|
|
1001
1016
|
runs-on: ubuntu-latest
|
|
1002
1017
|
steps:
|
|
1003
1018
|
- uses: actions/checkout@v4
|
|
1004
|
-
- uses: homeofe/supply-chain-guard@v6.5.
|
|
1019
|
+
- uses: homeofe/supply-chain-guard@v6.5.2
|
|
1005
1020
|
with:
|
|
1006
1021
|
fail-on: critical
|
|
1007
1022
|
comment-on-pr: true
|
|
@@ -1089,7 +1104,7 @@ two are never confused. If a deliberately frozen rule set is the intent, exclude
|
|
|
1089
1104
|
the rule by name:
|
|
1090
1105
|
|
|
1091
1106
|
```yaml
|
|
1092
|
-
- uses: homeofe/supply-chain-guard@v6.5.
|
|
1107
|
+
- uses: homeofe/supply-chain-guard@v6.5.2
|
|
1093
1108
|
with:
|
|
1094
1109
|
exclude-rules: THREAT_FEED_STALE
|
|
1095
1110
|
```
|
|
@@ -1112,6 +1127,20 @@ scans, in the per-user cache directory unless `--cache-dir` or `SCG_CACHE_DIR`
|
|
|
1112
1127
|
names another (see [Catalog cache](#catalog-cache)). The cache belongs to one
|
|
1113
1128
|
release, so an upgrade needs a new refresh.
|
|
1114
1129
|
|
|
1130
|
+
A catalog that was never downloaded is the ordinary state of a fresh install, so
|
|
1131
|
+
it stays an informational note and does not mark the scan partial. A catalog
|
|
1132
|
+
that IS on disk but cannot be used (unreadable or truncated, or not matching
|
|
1133
|
+
the digest this release pins) is reported as
|
|
1134
|
+
`THREAT_FEED_CATALOG_UNAVAILABLE` instead: the scan is marked partial and exits
|
|
1135
|
+
non-zero, because the historical indicators were silently not consulted. Run
|
|
1136
|
+
`supply-chain-guard feed refresh` to replace it. A catalog that `feed refresh`
|
|
1137
|
+
installed for this release and that has since been deleted counts the same way:
|
|
1138
|
+
the refresh leaves a small `catalog-installed.json` marker beside it, and a
|
|
1139
|
+
missing catalog with that marker present is `THREAT_FEED_CATALOG_UNAVAILABLE`
|
|
1140
|
+
(high, partial), not a fresh install. The marker only ever raises the finding; a
|
|
1141
|
+
missing, malformed or older-release marker reads as never installed.
|
|
1142
|
+
`catalog: required` raises both rules to critical.
|
|
1143
|
+
|
|
1115
1144
|
#### Catalog cache
|
|
1116
1145
|
|
|
1117
1146
|
`feed refresh` and every scan resolve the same directory, in this order:
|
|
@@ -1192,7 +1221,7 @@ preceding `feed refresh` in the workflow does not count. `catalog: required`
|
|
|
1192
1221
|
on the Action therefore needs:
|
|
1193
1222
|
|
|
1194
1223
|
```yaml
|
|
1195
|
-
- uses: homeofe/supply-chain-guard@v6.5.
|
|
1224
|
+
- uses: homeofe/supply-chain-guard@v6.5.2
|
|
1196
1225
|
with:
|
|
1197
1226
|
refresh-catalog: true
|
|
1198
1227
|
```
|
|
@@ -1203,7 +1232,7 @@ setting rather than firing on every scan. If scanning against the bundled set
|
|
|
1203
1232
|
alone is the intent, exclude the rule by name:
|
|
1204
1233
|
|
|
1205
1234
|
```yaml
|
|
1206
|
-
- uses: homeofe/supply-chain-guard@v6.5.
|
|
1235
|
+
- uses: homeofe/supply-chain-guard@v6.5.2
|
|
1207
1236
|
with:
|
|
1208
1237
|
exclude-rules: THREAT_FEED_CATALOG_MISSING
|
|
1209
1238
|
```
|
|
@@ -1327,6 +1356,24 @@ supply-chain-guard feed osv # export malicious-package IOCs as OSV records
|
|
|
1327
1356
|
|
|
1328
1357
|
A refreshed feed is merged into every scan for the next 24 hours automatically.
|
|
1329
1358
|
|
|
1359
|
+
**Feed integrity:** the default source is the `feed.json` attached to the latest
|
|
1360
|
+
GitHub Release, with `feed.json.sig` beside it, an Ed25519 signature over the
|
|
1361
|
+
exact bytes of the feed that the release job makes. `feed refresh` verifies it
|
|
1362
|
+
against the public key compiled into the installed package (fingerprint in
|
|
1363
|
+
[SECURITY.md](SECURITY.md#feed-integrity)) before parsing anything, and refuses
|
|
1364
|
+
a missing, malformed or wrong signature with a non-zero exit, leaving the
|
|
1365
|
+
previous cache in place. The signed bytes and signature are kept beside the
|
|
1366
|
+
cache and checked again on every scan, so a cache edited or planted afterwards
|
|
1367
|
+
is reported as `THREAT_FEED_CACHE_UNREADABLE` and the scan is partial.
|
|
1368
|
+
|
|
1369
|
+
A self-hosted mirror passed with `--url` must serve `feed.json.sig` next to the
|
|
1370
|
+
feed, signed by the same key. A mirror that cannot be signed can be accepted
|
|
1371
|
+
explicitly with `feed refresh --url <url> --allow-unsigned-feed`; that prints a
|
|
1372
|
+
warning, records in the cache that the feed was not verified, and trusts the
|
|
1373
|
+
feed as downloaded. There is no environment variable for it. The feed is
|
|
1374
|
+
published with each release, so a refresh picks up the indicators of the latest
|
|
1375
|
+
release rather than of the latest commit.
|
|
1376
|
+
|
|
1330
1377
|
**Rule-set age:** `feed stats` reports two ages, the one bundled with the
|
|
1331
1378
|
installed version and the effective one at scan time, and marks either `[STALE]`
|
|
1332
1379
|
past 30 days. `--format json` returns the same values as `bundledFreshness` and
|
package/action.yml
CHANGED
|
@@ -18,8 +18,14 @@ branding:
|
|
|
18
18
|
# suppressed by the loaded config is named in the scan report in every output
|
|
19
19
|
# format, including the markdown body of the pull request comment this Action
|
|
20
20
|
# posts by default, and a narrowing with no written reason is reported as a
|
|
21
|
-
# finding.
|
|
22
|
-
#
|
|
21
|
+
# finding. When the config (or an inline scg-ignore-next-line comment) hides a
|
|
22
|
+
# high or critical finding, the report adds POLICY_SUPPRESSED_SEVERE and
|
|
23
|
+
# riskLevelBeforePolicy, the risk level the scan would have had without it.
|
|
24
|
+
# A policy file and inline suppression comments are NOT honoured for a
|
|
25
|
+
# high or critical finding when the scan target is a cloned GitHub repository
|
|
26
|
+
# URL or an MCP scan_directory call. See the "Where the policy is read from"
|
|
27
|
+
# section of the README for the controls that apply when the proposer is
|
|
28
|
+
# untrusted.
|
|
23
29
|
inputs:
|
|
24
30
|
path:
|
|
25
31
|
description: "Path to scan (defaults to repository root)"
|
|
@@ -99,7 +105,7 @@ runs:
|
|
|
99
105
|
- name: Install supply-chain-guard
|
|
100
106
|
shell: bash
|
|
101
107
|
env:
|
|
102
|
-
SCG_VERSION: "6.5.
|
|
108
|
+
SCG_VERSION: "6.5.2"
|
|
103
109
|
SCG_EXPECTED_REPOSITORY: "homeofe/supply-chain-guard"
|
|
104
110
|
SCG_EXPECTED_REPOSITORY_ID: "1185867580"
|
|
105
111
|
SCG_EXPECTED_WORKFLOW: ".github/workflows/ci.yml"
|
|
@@ -253,6 +259,7 @@ runs:
|
|
|
253
259
|
or . == "INTERNAL_DENYLIST_REFUSED"
|
|
254
260
|
or . == "POLICY_INVALID_INTERNAL_TERM"
|
|
255
261
|
or . == "THREAT_FEED_CACHE_UNREADABLE"
|
|
262
|
+
or . == "THREAT_FEED_CATALOG_UNAVAILABLE"
|
|
256
263
|
or . == "DIFF_BASE_UNAVAILABLE"
|
|
257
264
|
or . == "RISK_HISTORY_UNREADABLE"
|
|
258
265
|
or . == "TRIAGE_STORE_UNREADABLE"
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Archive extraction helpers. Every member is validated
|
|
3
|
-
* extractor
|
|
2
|
+
* Archive extraction helpers. Every member is validated and then written by
|
|
3
|
+
* this module itself (no extractor process), so an archive cannot write outside
|
|
4
|
+
* the extraction root and the validated path is the written path.
|
|
4
5
|
*/
|
|
5
6
|
export declare const ARCHIVE_MAX_ENTRIES = 100000;
|
|
6
7
|
export declare const ARCHIVE_MAX_INPUT_BYTES: number;
|
|
@@ -11,7 +12,29 @@ export declare class ArchiveSecurityError extends Error {
|
|
|
11
12
|
}
|
|
12
13
|
export declare function preflightZipArchive(archivePath: string): void;
|
|
13
14
|
export declare function preflightTarArchive(archivePath: string): void;
|
|
14
|
-
export
|
|
15
|
-
|
|
16
|
-
|
|
15
|
+
export interface TarExtractionResult {
|
|
16
|
+
/**
|
|
17
|
+
* Symlink and hardlink members whose content was NOT written. A link that
|
|
18
|
+
* resolves to a regular file in the archive is written as a copy of that file
|
|
19
|
+
* instead; only links to directories (or whose path a directory already
|
|
20
|
+
* occupies) land here. No link is ever created on disk. Callers report each
|
|
21
|
+
* as an incomplete path so the result is partial, not silently clean.
|
|
22
|
+
*/
|
|
23
|
+
skippedLinks: string[];
|
|
24
|
+
}
|
|
25
|
+
/**
|
|
26
|
+
* Extract a ZIP by writing the members the parser validated, never by spawning
|
|
27
|
+
* `unzip` or bsdtar: a tool may read names, extra fields or entry types
|
|
28
|
+
* differently from what was validated. Default keeps files that already exist;
|
|
29
|
+
* `overwrite` replaces an existing regular file (never a link or directory).
|
|
30
|
+
*/
|
|
31
|
+
export declare function extractZip(archivePath: string, extractDir: string, overwrite?: boolean): TarExtractionResult;
|
|
32
|
+
/**
|
|
33
|
+
* Extract a tar (gzip, bzip2 and xz are autodetected) by writing the members
|
|
34
|
+
* the parser validated, never by spawning the PATH `tar`. This keeps the
|
|
35
|
+
* validated path identical to the written path and works the same on every
|
|
36
|
+
* host, including Windows machines whose PATH puts GNU tar before System32.
|
|
37
|
+
*/
|
|
38
|
+
export declare function extractTar(archivePath: string, extractDir: string): TarExtractionResult;
|
|
39
|
+
export declare function extractTarGz(archivePath: string, extractDir: string): TarExtractionResult;
|
|
17
40
|
//# sourceMappingURL=archive-extractor.d.ts.map
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"archive-extractor.d.ts","sourceRoot":"","sources":["../src/archive-extractor.ts"],"names":[],"mappings":"AAAA
|
|
1
|
+
{"version":3,"file":"archive-extractor.d.ts","sourceRoot":"","sources":["../src/archive-extractor.ts"],"names":[],"mappings":"AAAA;;;;GAIG;AAQH,eAAO,MAAM,mBAAmB,SAAU,CAAC;AAC3C,eAAO,MAAM,uBAAuB,QAAoB,CAAC;AACzD,eAAO,MAAM,0BAA0B,QAAoB,CAAC;AAC5D,eAAO,MAAM,2BAA2B,KAAK,CAAC;AAkC9C,qBAAa,oBAAqB,SAAQ,KAAK;IAC7C,YAAY,OAAO,EAAE,MAAM,EAG1B;CACF;AA4fD,wBAAgB,mBAAmB,CAAC,WAAW,EAAE,MAAM,GAAG,IAAI,CAE7D;AAsOD,wBAAgB,mBAAmB,CAAC,WAAW,EAAE,MAAM,GAAG,IAAI,CAE7D;AAED,MAAM,WAAW,mBAAmB;IAClC;;;;;;OAMG;IACH,YAAY,EAAE,MAAM,EAAE,CAAC;CACxB;AA2GD;;;;;GAKG;AACH,wBAAgB,UAAU,CAAC,WAAW,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,EAAE,SAAS,UAAQ,GAAG,mBAAmB,CAM1G;AAED;;;;;GAKG;AACH,wBAAgB,UAAU,CAAC,WAAW,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,GAAG,mBAAmB,CAEvF;AAED,wBAAgB,YAAY,CAAC,WAAW,EAAE,MAAM,EAAE,UAAU,EAAE,MAAM,GAAG,mBAAmB,CAEzF"}
|
|
@@ -1,7 +1,8 @@
|
|
|
1
1
|
"use strict";
|
|
2
2
|
/**
|
|
3
|
-
* Archive extraction helpers. Every member is validated
|
|
4
|
-
* extractor
|
|
3
|
+
* Archive extraction helpers. Every member is validated and then written by
|
|
4
|
+
* this module itself (no extractor process), so an archive cannot write outside
|
|
5
|
+
* the extraction root and the validated path is the written path.
|
|
5
6
|
*/
|
|
6
7
|
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
7
8
|
if (k2 === undefined) k2 = k;
|
|
@@ -43,9 +44,10 @@ exports.preflightTarArchive = preflightTarArchive;
|
|
|
43
44
|
exports.extractZip = extractZip;
|
|
44
45
|
exports.extractTar = extractTar;
|
|
45
46
|
exports.extractTarGz = extractTarGz;
|
|
46
|
-
const
|
|
47
|
+
const safe_exec_js_1 = require("./safe-exec.js");
|
|
47
48
|
const fs = __importStar(require("node:fs"));
|
|
48
49
|
const path = __importStar(require("node:path"));
|
|
50
|
+
const zlib = __importStar(require("node:zlib"));
|
|
49
51
|
const node_zlib_1 = require("node:zlib");
|
|
50
52
|
exports.ARCHIVE_MAX_ENTRIES = 100_000;
|
|
51
53
|
exports.ARCHIVE_MAX_INPUT_BYTES = 128 * 1024 * 1024;
|
|
@@ -492,7 +494,28 @@ function decodeZipPayload(entry, compressed) {
|
|
|
492
494
|
unsafe(`member "${entry.path || "."}" payload cannot be decompressed safely`);
|
|
493
495
|
}
|
|
494
496
|
}
|
|
495
|
-
|
|
497
|
+
/** CRC-32 (IEEE 802.3), as ZIP stores it. `zlib.crc32` exists from Node 22.2; engines allows 22.0. */
|
|
498
|
+
let crcTable;
|
|
499
|
+
function crc32Of(data) {
|
|
500
|
+
const native = zlib.crc32;
|
|
501
|
+
if (typeof native === "function")
|
|
502
|
+
return native(data) >>> 0;
|
|
503
|
+
if (crcTable === undefined) {
|
|
504
|
+
crcTable = new Uint32Array(256);
|
|
505
|
+
for (let n = 0; n < 256; n++) {
|
|
506
|
+
let c = n;
|
|
507
|
+
for (let k = 0; k < 8; k++)
|
|
508
|
+
c = (c & 1) !== 0 ? 0xedb88320 ^ (c >>> 1) : c >>> 1;
|
|
509
|
+
crcTable[n] = c >>> 0;
|
|
510
|
+
}
|
|
511
|
+
}
|
|
512
|
+
let crc = 0xffffffff;
|
|
513
|
+
for (let index = 0; index < data.length; index++)
|
|
514
|
+
crc = crcTable[(crc ^ data[index]) & 0xff] ^ (crc >>> 8);
|
|
515
|
+
return (crc ^ 0xffffffff) >>> 0;
|
|
516
|
+
}
|
|
517
|
+
/** Validate a ZIP and return its members with the payloads that were validated. */
|
|
518
|
+
function parseZipArchive(archivePath) {
|
|
496
519
|
const content = readBoundedArchive(path.resolve(archivePath));
|
|
497
520
|
const eocdOffset = findZipEndOfCentralDirectory(content);
|
|
498
521
|
const location = parseZipDirectoryLocation(content, eocdOffset);
|
|
@@ -568,6 +591,10 @@ function preflightZipArchive(archivePath) {
|
|
|
568
591
|
for (const entry of entries) {
|
|
569
592
|
const local = zipLocalPayload(content, entry, location.centralOffset);
|
|
570
593
|
const expanded = decodeZipPayload(entry, local.compressed);
|
|
594
|
+
// The central directory's CRC is what every other reader trusts; a payload
|
|
595
|
+
// that disagrees was altered or corrupted after it was described.
|
|
596
|
+
if (crc32Of(expanded) !== entry.crc32)
|
|
597
|
+
unsafe(`member "${entry.path || "."}" fails its CRC-32 check`);
|
|
571
598
|
if (entry.kind === "directory" && expanded.length !== 0)
|
|
572
599
|
unsafe(`directory "${entry.path || "."}" has a payload`);
|
|
573
600
|
if (entry.kind === "symlink") {
|
|
@@ -575,6 +602,8 @@ function preflightZipArchive(archivePath) {
|
|
|
575
602
|
unsafe(`link "${entry.path}" target is too large`);
|
|
576
603
|
entry.target = resolveRelativeTarget(entry.path, decodeZipPortableString(expanded, entry.flags, `ZIP link "${entry.path}" target`));
|
|
577
604
|
}
|
|
605
|
+
else if (entry.kind === "file")
|
|
606
|
+
entry.data = expanded;
|
|
578
607
|
localRegions.push({ start: local.start, end: local.end, path: entry.path || "." });
|
|
579
608
|
}
|
|
580
609
|
localRegions.sort((left, right) => left.start - right.start || left.end - right.end);
|
|
@@ -583,7 +612,12 @@ function preflightZipArchive(archivePath) {
|
|
|
583
612
|
unsafe(`local ZIP regions for "${localRegions[index - 1].path}" and "${localRegions[index].path}" overlap`);
|
|
584
613
|
}
|
|
585
614
|
}
|
|
586
|
-
|
|
615
|
+
const members = entries.filter((entry) => entry.path.length > 0);
|
|
616
|
+
validateArchiveGraph(members);
|
|
617
|
+
return members;
|
|
618
|
+
}
|
|
619
|
+
function preflightZipArchive(archivePath) {
|
|
620
|
+
parseZipArchive(archivePath);
|
|
587
621
|
}
|
|
588
622
|
function parseTarNumber(field, label) {
|
|
589
623
|
if (field.length === 0)
|
|
@@ -683,7 +717,7 @@ function decompressTarArchive(archivePath, content) {
|
|
|
683
717
|
const isXz = content.length >= 6 && content.subarray(0, 6).equals(Buffer.from([0xfd, 0x37, 0x7a, 0x58, 0x5a, 0x00]));
|
|
684
718
|
if (isBzip2 || isXz) {
|
|
685
719
|
try {
|
|
686
|
-
return (0,
|
|
720
|
+
return (0, safe_exec_js_1.execToolSync)(isBzip2 ? "bzip2" : "xz", ["-dc", archivePath], {
|
|
687
721
|
encoding: "buffer", maxBuffer: exports.ARCHIVE_MAX_EXPANDED_BYTES + 1, stdio: ["ignore", "pipe", "pipe"],
|
|
688
722
|
});
|
|
689
723
|
}
|
|
@@ -693,7 +727,20 @@ function decompressTarArchive(archivePath, content) {
|
|
|
693
727
|
}
|
|
694
728
|
return content;
|
|
695
729
|
}
|
|
696
|
-
|
|
730
|
+
/** A validated tar member; `data` is set for regular files only. */
|
|
731
|
+
/**
|
|
732
|
+
* Parse and validate every tar header, returning the members exactly as the
|
|
733
|
+
* extractor will write them. Metadata precedence (one rule, documented here
|
|
734
|
+
* because GNU tar and bsdtar disagree on several corners):
|
|
735
|
+
* - a PAX `x` record applies to the next member and wins over GNU `L`/`K`
|
|
736
|
+
* long-name records, whatever order they appear in;
|
|
737
|
+
* - a GNU `L`/`K` record wins over the header fields;
|
|
738
|
+
* - a PAX `g` record may carry neither `path` nor `linkpath` (the two
|
|
739
|
+
* extractors disagree on it, so the archive is refused).
|
|
740
|
+
* Nothing here is delegated to a system `tar`, so no dialect can resolve a
|
|
741
|
+
* member to a different path than the one validated.
|
|
742
|
+
*/
|
|
743
|
+
function parseTarArchive(archivePath) {
|
|
697
744
|
const resolvedArchivePath = path.resolve(archivePath);
|
|
698
745
|
const compressed = readBoundedArchive(resolvedArchivePath);
|
|
699
746
|
const content = decompressTarArchive(resolvedArchivePath, compressed);
|
|
@@ -706,8 +753,10 @@ function preflightTarArchive(archivePath) {
|
|
|
706
753
|
let headerCount = 0;
|
|
707
754
|
let cursor = 0;
|
|
708
755
|
let sawEndMarker = false;
|
|
709
|
-
let
|
|
756
|
+
let paxOverrides = {};
|
|
757
|
+
let gnuOverrides = {};
|
|
710
758
|
let globalOverrides = {};
|
|
759
|
+
const pendingMetadata = () => Object.keys(paxOverrides).length > 0 || Object.keys(gnuOverrides).length > 0;
|
|
711
760
|
while (cursor + TAR_BLOCK_SIZE <= content.length) {
|
|
712
761
|
const header = content.subarray(cursor, cursor + TAR_BLOCK_SIZE);
|
|
713
762
|
if (header.every((byte) => byte === 0)) {
|
|
@@ -722,20 +771,20 @@ function preflightTarArchive(archivePath) {
|
|
|
722
771
|
unsafe(`entry count exceeds ${exports.ARCHIVE_MAX_ENTRIES}`);
|
|
723
772
|
verifyTarChecksum(header);
|
|
724
773
|
const headerNameBytes = tarFieldBytes(header, 0, 100);
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
774
|
+
// Any `ustar\0` magic is POSIX ustar whatever the two version bytes say:
|
|
775
|
+
// GNU tar and bsdtar both honour the prefix field then.
|
|
776
|
+
const posixUstar = header.subarray(257, 263).equals(Buffer.from([0x75, 0x73, 0x74, 0x61, 0x72, 0x00]));
|
|
728
777
|
// Bytes 345..499 are a pathname prefix only in POSIX ustar/PAX. Old-GNU
|
|
729
|
-
// uses the same region for timestamps and sparse metadata; V7
|
|
730
|
-
// outside the header. Treating either as a prefix
|
|
731
|
-
// extractor never writes.
|
|
778
|
+
// (`ustar `) uses the same region for timestamps and sparse metadata; V7
|
|
779
|
+
// leaves it outside the header. Treating either as a prefix would validate
|
|
780
|
+
// a path the extractor never writes.
|
|
732
781
|
const prefixBytes = posixUstar
|
|
733
782
|
? tarFieldBytes(header, 345, 155)
|
|
734
783
|
: Buffer.alloc(0);
|
|
735
784
|
const rawHeaderLinkBytes = tarFieldBytes(header, 157, 100);
|
|
736
785
|
const typeFlag = header[156] === 0 ? "0" : String.fromCharCode(header[156]);
|
|
737
786
|
const headerSize = parseTarNumber(header.subarray(124, 136), "tar entry size");
|
|
738
|
-
const overrides = { ...globalOverrides, ...
|
|
787
|
+
const overrides = { ...globalOverrides, ...gnuOverrides, ...paxOverrides };
|
|
739
788
|
const metadataRecord = typeFlag === "x" || typeFlag === "g" ||
|
|
740
789
|
typeFlag === "L" || typeFlag === "K";
|
|
741
790
|
const size = metadataRecord ? headerSize : (overrides.size ?? headerSize);
|
|
@@ -750,10 +799,14 @@ function preflightTarArchive(archivePath) {
|
|
|
750
799
|
unsafe("tar entry padding is truncated");
|
|
751
800
|
if (typeFlag === "x" || typeFlag === "g") {
|
|
752
801
|
const parsed = parsePax(payload);
|
|
753
|
-
if (typeFlag === "g")
|
|
802
|
+
if (typeFlag === "g") {
|
|
803
|
+
if (parsed.path !== undefined || parsed.linkpath !== undefined) {
|
|
804
|
+
unsafe("a PAX global record carries path or linkpath, which extractors apply inconsistently");
|
|
805
|
+
}
|
|
754
806
|
globalOverrides = { ...globalOverrides, ...parsed };
|
|
807
|
+
}
|
|
755
808
|
else
|
|
756
|
-
|
|
809
|
+
paxOverrides = { ...paxOverrides, ...parsed };
|
|
757
810
|
cursor = nextCursor;
|
|
758
811
|
continue;
|
|
759
812
|
}
|
|
@@ -762,13 +815,14 @@ function preflightTarArchive(archivePath) {
|
|
|
762
815
|
unsafe("GNU tar long-name record is too large");
|
|
763
816
|
const nul = payload.indexOf(0);
|
|
764
817
|
const value = decodeLegacyTarString(nul === -1 ? payload : payload.subarray(0, nul), "GNU tar long-name record");
|
|
765
|
-
|
|
766
|
-
? { ...
|
|
767
|
-
: { ...
|
|
818
|
+
gnuOverrides = typeFlag === "L"
|
|
819
|
+
? { ...gnuOverrides, path: value }
|
|
820
|
+
: { ...gnuOverrides, linkpath: value };
|
|
768
821
|
cursor = nextCursor;
|
|
769
822
|
continue;
|
|
770
823
|
}
|
|
771
|
-
|
|
824
|
+
// PAX `x` over GNU `L`/`K`, independent of record order (see above).
|
|
825
|
+
const selectedOverrides = { ...gnuOverrides, ...paxOverrides };
|
|
772
826
|
let rawPath = selectedOverrides.path;
|
|
773
827
|
if (rawPath === undefined) {
|
|
774
828
|
const headerName = decodeLegacyTarString(headerNameBytes, "tar header name");
|
|
@@ -779,12 +833,20 @@ function preflightTarArchive(archivePath) {
|
|
|
779
833
|
if (rawLink === undefined && (typeFlag === "1" || typeFlag === "2")) {
|
|
780
834
|
rawLink = decodeLegacyTarString(rawHeaderLinkBytes, "tar header link target");
|
|
781
835
|
}
|
|
836
|
+
const carriesLinkpath = rawLink !== undefined;
|
|
782
837
|
rawLink ??= "";
|
|
783
|
-
|
|
838
|
+
paxOverrides = {};
|
|
839
|
+
gnuOverrides = {};
|
|
784
840
|
const normalizedPath = normalizeMemberPath(rawPath, "entry path", typeFlag === "5");
|
|
841
|
+
// bsdtar turns a regular file that carries a linkpath into a hardlink while
|
|
842
|
+
// GNU tar writes a regular file, so the member type would depend on the
|
|
843
|
+
// extractor. Refuse it.
|
|
844
|
+
if (carriesLinkpath && (typeFlag === "0" || typeFlag === "7" || typeFlag === "5")) {
|
|
845
|
+
unsafe(`member "${normalizedPath}" is not a link but carries a linkpath`);
|
|
846
|
+
}
|
|
785
847
|
let entry;
|
|
786
848
|
if (typeFlag === "0" || typeFlag === "7")
|
|
787
|
-
entry = { path: normalizedPath, kind: "file" };
|
|
849
|
+
entry = { path: normalizedPath, kind: "file", data: payload };
|
|
788
850
|
else if (typeFlag === "5") {
|
|
789
851
|
if (size !== 0)
|
|
790
852
|
unsafe(`directory "${normalizedPath}" has a data payload`);
|
|
@@ -811,46 +873,158 @@ function preflightTarArchive(archivePath) {
|
|
|
811
873
|
unsafe("tar end marker is missing");
|
|
812
874
|
if (!content.subarray(cursor).every((byte) => byte === 0))
|
|
813
875
|
unsafe("tar contains non-zero data after its end marker");
|
|
814
|
-
if (
|
|
876
|
+
if (pendingMetadata())
|
|
815
877
|
unsafe("tar ends with metadata that does not describe a member");
|
|
816
878
|
validateArchiveGraph(entries.filter((entry) => entry.path.length > 0));
|
|
879
|
+
return entries;
|
|
880
|
+
}
|
|
881
|
+
function preflightTarArchive(archivePath) {
|
|
882
|
+
parseTarArchive(archivePath);
|
|
883
|
+
}
|
|
884
|
+
/** Create `rel` below `root` one component at a time, refusing anything that is not a plain directory. */
|
|
885
|
+
function ensureContainedDirectory(root, rootReal, rel, verified) {
|
|
886
|
+
let current = root;
|
|
887
|
+
let built = "";
|
|
888
|
+
for (const component of rel === "" ? [] : rel.split("/")) {
|
|
889
|
+
built = built === "" ? component : `${built}/${component}`;
|
|
890
|
+
current = path.join(current, component);
|
|
891
|
+
let stat;
|
|
892
|
+
try {
|
|
893
|
+
stat = fs.lstatSync(current);
|
|
894
|
+
}
|
|
895
|
+
catch {
|
|
896
|
+
stat = undefined;
|
|
897
|
+
}
|
|
898
|
+
if (stat === undefined)
|
|
899
|
+
fs.mkdirSync(current);
|
|
900
|
+
else if (!stat.isDirectory())
|
|
901
|
+
unsafe(`"${built}" already exists and is not a directory`);
|
|
902
|
+
}
|
|
903
|
+
if (!verified.has(rel)) {
|
|
904
|
+
const real = fs.realpathSync(current);
|
|
905
|
+
if (real !== rootReal && !real.startsWith(rootReal + path.sep)) {
|
|
906
|
+
unsafe(`"${rel || "."}" resolves outside the extraction root`);
|
|
907
|
+
}
|
|
908
|
+
verified.add(rel);
|
|
909
|
+
}
|
|
910
|
+
return current;
|
|
911
|
+
}
|
|
912
|
+
/**
|
|
913
|
+
* Write the validated members ourselves: regular files and directories only,
|
|
914
|
+
* every file created with `wx` so nothing existing is followed or replaced.
|
|
915
|
+
*/
|
|
916
|
+
function writeArchiveMembers(members, extractDir, existing = "refuse") {
|
|
917
|
+
fs.mkdirSync(extractDir, { recursive: true });
|
|
918
|
+
const rootReal = fs.realpathSync(extractDir);
|
|
919
|
+
const verified = new Set();
|
|
920
|
+
const skippedLinks = [];
|
|
921
|
+
const writeFile = (memberPath, data) => {
|
|
922
|
+
const slash = memberPath.lastIndexOf("/");
|
|
923
|
+
const parentRel = slash === -1 ? "" : memberPath.slice(0, slash);
|
|
924
|
+
const parent = ensureContainedDirectory(extractDir, rootReal, parentRel, verified);
|
|
925
|
+
const target = path.join(parent, memberPath.slice(slash + 1));
|
|
926
|
+
let fd;
|
|
927
|
+
try {
|
|
928
|
+
fd = fs.openSync(target, "wx", 0o644);
|
|
929
|
+
}
|
|
930
|
+
catch { /* handled below */ }
|
|
931
|
+
if (fd === undefined) {
|
|
932
|
+
let stat;
|
|
933
|
+
try {
|
|
934
|
+
stat = fs.lstatSync(target);
|
|
935
|
+
}
|
|
936
|
+
catch {
|
|
937
|
+
stat = undefined;
|
|
938
|
+
}
|
|
939
|
+
if (stat === undefined || existing === "refuse")
|
|
940
|
+
unsafe(`"${memberPath}" cannot be created exclusively`);
|
|
941
|
+
// Keep: an existing entry of any kind stays as it is. Replace: only a
|
|
942
|
+
// plain regular file is replaced, never a link or directory.
|
|
943
|
+
if (existing === "keep")
|
|
944
|
+
return;
|
|
945
|
+
if (!stat.isFile())
|
|
946
|
+
unsafe(`"${memberPath}" already exists and is not a regular file`);
|
|
947
|
+
try {
|
|
948
|
+
fs.unlinkSync(target);
|
|
949
|
+
fd = fs.openSync(target, "wx", 0o644);
|
|
950
|
+
}
|
|
951
|
+
catch {
|
|
952
|
+
unsafe(`"${memberPath}" cannot be replaced`);
|
|
953
|
+
}
|
|
954
|
+
}
|
|
955
|
+
try {
|
|
956
|
+
let written = 0;
|
|
957
|
+
while (written < data.length)
|
|
958
|
+
written += fs.writeSync(fd, data, written, data.length - written);
|
|
959
|
+
}
|
|
960
|
+
finally {
|
|
961
|
+
fs.closeSync(fd);
|
|
962
|
+
}
|
|
963
|
+
};
|
|
964
|
+
const links = [];
|
|
965
|
+
for (const member of members) {
|
|
966
|
+
if (member.path.length === 0)
|
|
967
|
+
continue;
|
|
968
|
+
if (member.kind === "symlink" || member.kind === "hardlink") {
|
|
969
|
+
links.push(member);
|
|
970
|
+
continue;
|
|
971
|
+
}
|
|
972
|
+
if (member.kind === "directory") {
|
|
973
|
+
ensureContainedDirectory(extractDir, rootReal, member.path, verified);
|
|
974
|
+
continue;
|
|
975
|
+
}
|
|
976
|
+
writeFile(member.path, member.data ?? Buffer.alloc(0));
|
|
977
|
+
}
|
|
978
|
+
// Links are never created on disk. A link the preflight resolved to a
|
|
979
|
+
// regular FILE inside the archive is written as a copy of that file's bytes,
|
|
980
|
+
// so the scan reads exactly what the link would show without a path anything
|
|
981
|
+
// could follow out of the root; a symlinked LICENSE or README in an sdist must
|
|
982
|
+
// not make the whole scan partial. A link to a directory, or one whose path is
|
|
983
|
+
// already taken by a directory, is reported to the caller as skipped.
|
|
984
|
+
const byPath = new Map();
|
|
985
|
+
for (const member of members) {
|
|
986
|
+
if (member.path.length > 0)
|
|
987
|
+
byPath.set(portableMemberPathKey(member.path), member);
|
|
988
|
+
}
|
|
989
|
+
const context = { resolvedPaths: new Map(), work: 0 };
|
|
990
|
+
for (const link of links) {
|
|
991
|
+
const resolved = link.kind === "hardlink"
|
|
992
|
+
? resolveVirtualPath(link.target ?? "", byPath, true, context)
|
|
993
|
+
: resolveVirtualPath(link.path, byPath, true, context);
|
|
994
|
+
const target = byPath.get(portableMemberPathKey(resolved));
|
|
995
|
+
let taken = false;
|
|
996
|
+
try {
|
|
997
|
+
fs.lstatSync(path.join(extractDir, link.path));
|
|
998
|
+
taken = true;
|
|
999
|
+
}
|
|
1000
|
+
catch { /* free */ }
|
|
1001
|
+
if (target?.kind !== "file" || (taken && existing === "refuse")) {
|
|
1002
|
+
skippedLinks.push(link.path);
|
|
1003
|
+
continue;
|
|
1004
|
+
}
|
|
1005
|
+
writeFile(link.path, target.data ?? Buffer.alloc(0));
|
|
1006
|
+
}
|
|
1007
|
+
return { skippedLinks };
|
|
817
1008
|
}
|
|
1009
|
+
/**
|
|
1010
|
+
* Extract a ZIP by writing the members the parser validated, never by spawning
|
|
1011
|
+
* `unzip` or bsdtar: a tool may read names, extra fields or entry types
|
|
1012
|
+
* differently from what was validated. Default keeps files that already exist;
|
|
1013
|
+
* `overwrite` replaces an existing regular file (never a link or directory).
|
|
1014
|
+
*/
|
|
818
1015
|
function extractZip(archivePath, extractDir, overwrite = false) {
|
|
819
|
-
|
|
820
|
-
const resolvedExtractDir = path.resolve(extractDir);
|
|
821
|
-
preflightZipArchive(resolvedArchivePath);
|
|
822
|
-
fs.mkdirSync(resolvedExtractDir, { recursive: true });
|
|
823
|
-
if (process.platform === "win32") {
|
|
824
|
-
// Modern Windows ships bsdtar in System32. It auto-detects ZIP and avoids
|
|
825
|
-
// making runtime extraction depend on a separately installed Info-ZIP.
|
|
826
|
-
const windowsRoot = process.env.SystemRoot || "C:\\Windows";
|
|
827
|
-
const tarPath = path.win32.join(windowsRoot, "System32", "tar.exe");
|
|
828
|
-
if (!fs.existsSync(tarPath)) {
|
|
829
|
-
throw new Error(`ZIP extraction backend is unavailable: expected Windows bsdtar at ${tarPath}`);
|
|
830
|
-
}
|
|
831
|
-
const args = ["-x"];
|
|
832
|
-
if (!overwrite)
|
|
833
|
-
args.push("-k");
|
|
834
|
-
args.push("-f", resolvedArchivePath, "-C", resolvedExtractDir);
|
|
835
|
-
(0, node_child_process_1.execFileSync)(tarPath, args, { stdio: "pipe" });
|
|
836
|
-
return;
|
|
837
|
-
}
|
|
838
|
-
const args = ["-q"];
|
|
839
|
-
if (overwrite)
|
|
840
|
-
args.push("-o");
|
|
841
|
-
args.push(resolvedArchivePath, "-d", resolvedExtractDir);
|
|
842
|
-
(0, node_child_process_1.execFileSync)("unzip", args, { stdio: "pipe" });
|
|
1016
|
+
return writeArchiveMembers(parseZipArchive(archivePath), path.resolve(extractDir), overwrite ? "replace" : "keep");
|
|
843
1017
|
}
|
|
1018
|
+
/**
|
|
1019
|
+
* Extract a tar (gzip, bzip2 and xz are autodetected) by writing the members
|
|
1020
|
+
* the parser validated, never by spawning the PATH `tar`. This keeps the
|
|
1021
|
+
* validated path identical to the written path and works the same on every
|
|
1022
|
+
* host, including Windows machines whose PATH puts GNU tar before System32.
|
|
1023
|
+
*/
|
|
844
1024
|
function extractTar(archivePath, extractDir) {
|
|
845
|
-
|
|
846
|
-
const resolvedExtractDir = path.resolve(extractDir);
|
|
847
|
-
preflightTarArchive(resolvedArchivePath);
|
|
848
|
-
(0, node_child_process_1.execFileSync)("tar", ["xf", resolvedArchivePath, "-C", resolvedExtractDir], { stdio: "pipe" });
|
|
1025
|
+
return writeArchiveMembers(parseTarArchive(archivePath), path.resolve(extractDir));
|
|
849
1026
|
}
|
|
850
1027
|
function extractTarGz(archivePath, extractDir) {
|
|
851
|
-
|
|
852
|
-
const resolvedExtractDir = path.resolve(extractDir);
|
|
853
|
-
preflightTarArchive(resolvedArchivePath);
|
|
854
|
-
(0, node_child_process_1.execFileSync)("tar", ["xzf", resolvedArchivePath, "-C", resolvedExtractDir], { stdio: "pipe" });
|
|
1028
|
+
return extractTar(archivePath, extractDir);
|
|
855
1029
|
}
|
|
856
1030
|
//# sourceMappingURL=archive-extractor.js.map
|