aisbom-cli 1.3.0__tar.gz → 1.3.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/PKG-INFO +31 -2
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/README.md +30 -1
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/cli.py +19 -4
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/corpus.py +46 -2
- aisbom_cli-1.3.2/aisbom/pickle_containers.py +304 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/properties.py +49 -1
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/remote.py +24 -8
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/safety.py +303 -2
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/scanner.py +361 -16
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/pyproject.toml +9 -1
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/LICENSE +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/__init__.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/diff.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/linter.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/loop_state.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/mock_generator.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/protobuf_reader.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/spdx_gen.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/telemetry.py +0 -0
- {aisbom_cli-1.3.0 → aisbom_cli-1.3.2}/aisbom/version_check.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: aisbom-cli
|
|
3
|
-
Version: 1.3.
|
|
3
|
+
Version: 1.3.2
|
|
4
4
|
Summary: An AI Supply Chain security tool that that detects Pickle bombs and generates CycloneDX SBOMs for Machine Learning models.
|
|
5
5
|
License-File: LICENSE
|
|
6
6
|
Author: Ajoy L
|
|
@@ -81,7 +81,10 @@ A typical scan against a project with mixed artifacts:
|
|
|
81
81
|
│ │ │ __class__, __mro__, __subclasses__) │ │
|
|
82
82
|
│ exfil.onnx │ ONNX │ CRITICAL (ONNX External Data Escapes Model │ UNKNOWN │
|
|
83
83
|
│ │ │ Directory: ../../../etc/passwd) │ │
|
|
84
|
+
│ sklearn_pipeline.joblib │ Joblib │ CRITICAL (RCE Detected: posix.system) │ UNKNOWN │
|
|
85
|
+
│ features.npy │ NumPy │ CRITICAL (RCE Detected: builtins.eval) │ UNKNOWN │
|
|
84
86
|
│ detector.onnx │ ONNX │ LOW │ UNKNOWN │
|
|
87
|
+
│ embeddings.npy │ NumPy │ LOW │ UNKNOWN │
|
|
85
88
|
│ llama-3-quant.gguf │ GGUF │ LOW │ LEGAL RISK (cc-by-nc-sa-4.0) │
|
|
86
89
|
│ safe_model.safetensors │ SafeTensors │ LOW │ PASS │
|
|
87
90
|
│ restricted_model.safetensors │ SafeTensors │ LOW │ LEGAL RISK (cc-by-nc-4.0) │
|
|
@@ -94,7 +97,10 @@ A compliant `sbom.json` (CycloneDX v1.6) including SHA256 hashes and license dat
|
|
|
94
97
|
|
|
95
98
|
| Format | Extensions | What AIsbom looks for |
|
|
96
99
|
|---|---|---|
|
|
97
|
-
| **PyTorch / Pickle** | `.pt` `.pth` `.bin` | Dangerous globals in the pickle opcodes, including indirect-execution gadgets. Concatenated streams are all scanned — a legacy `torch.save` file hides its object behind three header pickles — and non-standard containers (7z, rar, xz…) are flagged. If a file is so full of streams that the walk hits its work limit, the scan says so rather than reporting clean. |
|
|
100
|
+
| **PyTorch / Pickle** | `.pt` `.pth` `.bin` `.pkl` `.pickle` | Dangerous globals in the pickle opcodes, including indirect-execution gadgets. Concatenated streams are all scanned — a legacy `torch.save` file hides its object behind three header pickles — and non-standard containers (7z, rar, xz…) are flagged. If a file is so full of streams that the walk hits its work limit, the scan says so rather than reporting clean. |
|
|
101
|
+
| **joblib** | `.joblib` | The pickle inside the container, whichever codec joblib chose (zlib, gzip, bz2, lzma/xz, and the legacy `ZF` format), plus uncompressed files. A payload placed *after* an array — where a real model puts its weights — is found. |
|
|
102
|
+
| **dill** | `.dill` | Everything the pickle path finds, plus dill's own code-reconstruction globals: a dill'd function or lambda is a marshalled code object rebuilt on load, and is reported as CRITICAL. |
|
|
103
|
+
| **NumPy** | `.npy` `.npz` | The pickle stream behind an `allow_pickle` object array. Each `.npz` member is opened, including ones whose checksum or header has been tampered with. |
|
|
98
104
|
| **Keras** | `.keras` `.h5` `.hdf5` | `Lambda` layers and embedded marshalled code objects in the model config — an actively exploited RCE vector. |
|
|
99
105
|
| **GGUF** | `.gguf` | License and architecture metadata, plus the embedded Jinja **chat template**, checked for sandbox-escape constructs. |
|
|
100
106
|
| **ONNX** | `.onnx` | Producer/opset/IR metadata, custom operators, and external-data paths that point outside the model directory. |
|
|
@@ -134,6 +140,18 @@ xattr -d com.apple.quarantine aisbom-macos-*
|
|
|
134
140
|
|
|
135
141
|
## Common workflows
|
|
136
142
|
|
|
143
|
+
### Scan targets and exit codes
|
|
144
|
+
|
|
145
|
+
`scan` accepts a directory, a single model file, a `hf://` repo, or an HTTPS URL. A directory is walked recursively; a single file is scanned on its own.
|
|
146
|
+
|
|
147
|
+
| Exit | Meaning |
|
|
148
|
+
|---|---|
|
|
149
|
+
| `0` | Scan completed, no CRITICAL risk |
|
|
150
|
+
| `1` | The scan could not be completed — target missing or unreadable, a file failed to parse, or a remote fetch failed |
|
|
151
|
+
| `2` | A **CRITICAL** risk was found (suppress with `--no-fail-on-risk`) |
|
|
152
|
+
|
|
153
|
+
`--no-fail-on-risk` governs risk findings only. An unusable target still exits `1`, so a typo'd path in CI fails loudly instead of passing as a clean scan.
|
|
154
|
+
|
|
137
155
|
### Scan a Hugging Face model
|
|
138
156
|
|
|
139
157
|
```bash
|
|
@@ -390,6 +408,17 @@ Attribute names are matched **exactly, never as substrings**, so ordinary global
|
|
|
390
408
|
|
|
391
409
|
Pickle is not the only way a model file gets code to run. Each of the other formats has its own vector, and each is read as inert data:
|
|
392
410
|
|
|
411
|
+
**joblib, dill and NumPy object arrays** (`.joblib`, `.dill`, `.npy`, `.npz`) are pickle underneath — the everyday serialization of scikit-learn and scientific Python, carrying exactly the arbitrary-code-execution risk of a `.pt`, and opened by almost nothing that calls itself an SBOM tool. AIsbom reads them without importing joblib, dill or numpy: the compression is standard-library, the `.npy` header is parsed by hand, and nothing is ever unpickled.
|
|
412
|
+
|
|
413
|
+
Two things about these formats are worth stating plainly:
|
|
414
|
+
|
|
415
|
+
- **A raw array block does not end the scan.** joblib writes its pickle up to an array, dumps the raw buffer inline, then resumes pickling — which stops an opcode disassembly dead a few hundred bytes into the file. Since every real model has weights, "after the first array" is the natural place for a payload. The remaining bytes get a second pass that recovers globals directly, so a dangerous import behind an array block is still reported. That pass reads bytes rather than structure, so it recognises *known* sinks only: in strict mode it does not contribute unrecognized-import findings, and it cannot judge a dual-use constructor like `operator.methodcaller("system")`, because the argument that decides is a stack relationship it has no stack for.
|
|
416
|
+
- **A declared dtype never decides whether to look.** The `.npy` header is attacker-supplied, so a pickle sitting behind a header claiming `'<f8'` would otherwise be a one-line evasion. The data section is disassembled whatever the header says. What the header *does* affect is the reported risk: an array of ordinary numbers with no pickle in it is LOW, not "pickle present".
|
|
417
|
+
|
|
418
|
+
- **No limit ends in a clean verdict.** Four bounds apply here — the file-read budget, the `.npz` member count, the decompressor's output cap, and the disassembler's stream budget — and each one leaves bytes unexamined. Hitting any of them reports `MEDIUM (Pickle Scan Incomplete)` rather than passing the prefix off as the whole file. A limit that reports "clean" is not a safety measure; it is a hiding place with a length attached. A real finding still outranks the marker.
|
|
419
|
+
|
|
420
|
+
joblib's `lz4` and `zstd` codecs have no standard-library decompressor. Following the same reasoning as 7z containers, those are **named but not opened** — `MEDIUM (Unscanned Container: lz4)` — rather than pulling a native dependency into every install and every standalone binary. That is an honest "we did not read these bytes", which is a different answer from a clean scan.
|
|
421
|
+
|
|
393
422
|
**SafeTensors and GGUF** use binary formats with structured headers — AIsbom parses these directly to extract metadata (artifact names, license info, architecture details) without loading tensor weights.
|
|
394
423
|
|
|
395
424
|
**Keras** (`.keras`, `.h5`, `.hdf5`) carries a different execution vector: a `Lambda` layer stores an arbitrary Python callable in the model config as a base64-encoded marshalled code object, and `load_model` runs it. AIsbom reads the config out of both containers — the `.keras` zip and the legacy HDF5 attribute — and flags `Lambda` layers and embedded code objects as CRITICAL. The payload is identified from its header bytes and **never unmarshalled**. A truncated or corrupted container is still scanned rather than skipped, so damaging a file is not a way to hide a payload.
|
|
@@ -56,7 +56,10 @@ A typical scan against a project with mixed artifacts:
|
|
|
56
56
|
│ │ │ __class__, __mro__, __subclasses__) │ │
|
|
57
57
|
│ exfil.onnx │ ONNX │ CRITICAL (ONNX External Data Escapes Model │ UNKNOWN │
|
|
58
58
|
│ │ │ Directory: ../../../etc/passwd) │ │
|
|
59
|
+
│ sklearn_pipeline.joblib │ Joblib │ CRITICAL (RCE Detected: posix.system) │ UNKNOWN │
|
|
60
|
+
│ features.npy │ NumPy │ CRITICAL (RCE Detected: builtins.eval) │ UNKNOWN │
|
|
59
61
|
│ detector.onnx │ ONNX │ LOW │ UNKNOWN │
|
|
62
|
+
│ embeddings.npy │ NumPy │ LOW │ UNKNOWN │
|
|
60
63
|
│ llama-3-quant.gguf │ GGUF │ LOW │ LEGAL RISK (cc-by-nc-sa-4.0) │
|
|
61
64
|
│ safe_model.safetensors │ SafeTensors │ LOW │ PASS │
|
|
62
65
|
│ restricted_model.safetensors │ SafeTensors │ LOW │ LEGAL RISK (cc-by-nc-4.0) │
|
|
@@ -69,7 +72,10 @@ A compliant `sbom.json` (CycloneDX v1.6) including SHA256 hashes and license dat
|
|
|
69
72
|
|
|
70
73
|
| Format | Extensions | What AIsbom looks for |
|
|
71
74
|
|---|---|---|
|
|
72
|
-
| **PyTorch / Pickle** | `.pt` `.pth` `.bin` | Dangerous globals in the pickle opcodes, including indirect-execution gadgets. Concatenated streams are all scanned — a legacy `torch.save` file hides its object behind three header pickles — and non-standard containers (7z, rar, xz…) are flagged. If a file is so full of streams that the walk hits its work limit, the scan says so rather than reporting clean. |
|
|
75
|
+
| **PyTorch / Pickle** | `.pt` `.pth` `.bin` `.pkl` `.pickle` | Dangerous globals in the pickle opcodes, including indirect-execution gadgets. Concatenated streams are all scanned — a legacy `torch.save` file hides its object behind three header pickles — and non-standard containers (7z, rar, xz…) are flagged. If a file is so full of streams that the walk hits its work limit, the scan says so rather than reporting clean. |
|
|
76
|
+
| **joblib** | `.joblib` | The pickle inside the container, whichever codec joblib chose (zlib, gzip, bz2, lzma/xz, and the legacy `ZF` format), plus uncompressed files. A payload placed *after* an array — where a real model puts its weights — is found. |
|
|
77
|
+
| **dill** | `.dill` | Everything the pickle path finds, plus dill's own code-reconstruction globals: a dill'd function or lambda is a marshalled code object rebuilt on load, and is reported as CRITICAL. |
|
|
78
|
+
| **NumPy** | `.npy` `.npz` | The pickle stream behind an `allow_pickle` object array. Each `.npz` member is opened, including ones whose checksum or header has been tampered with. |
|
|
73
79
|
| **Keras** | `.keras` `.h5` `.hdf5` | `Lambda` layers and embedded marshalled code objects in the model config — an actively exploited RCE vector. |
|
|
74
80
|
| **GGUF** | `.gguf` | License and architecture metadata, plus the embedded Jinja **chat template**, checked for sandbox-escape constructs. |
|
|
75
81
|
| **ONNX** | `.onnx` | Producer/opset/IR metadata, custom operators, and external-data paths that point outside the model directory. |
|
|
@@ -109,6 +115,18 @@ xattr -d com.apple.quarantine aisbom-macos-*
|
|
|
109
115
|
|
|
110
116
|
## Common workflows
|
|
111
117
|
|
|
118
|
+
### Scan targets and exit codes
|
|
119
|
+
|
|
120
|
+
`scan` accepts a directory, a single model file, a `hf://` repo, or an HTTPS URL. A directory is walked recursively; a single file is scanned on its own.
|
|
121
|
+
|
|
122
|
+
| Exit | Meaning |
|
|
123
|
+
|---|---|
|
|
124
|
+
| `0` | Scan completed, no CRITICAL risk |
|
|
125
|
+
| `1` | The scan could not be completed — target missing or unreadable, a file failed to parse, or a remote fetch failed |
|
|
126
|
+
| `2` | A **CRITICAL** risk was found (suppress with `--no-fail-on-risk`) |
|
|
127
|
+
|
|
128
|
+
`--no-fail-on-risk` governs risk findings only. An unusable target still exits `1`, so a typo'd path in CI fails loudly instead of passing as a clean scan.
|
|
129
|
+
|
|
112
130
|
### Scan a Hugging Face model
|
|
113
131
|
|
|
114
132
|
```bash
|
|
@@ -365,6 +383,17 @@ Attribute names are matched **exactly, never as substrings**, so ordinary global
|
|
|
365
383
|
|
|
366
384
|
Pickle is not the only way a model file gets code to run. Each of the other formats has its own vector, and each is read as inert data:
|
|
367
385
|
|
|
386
|
+
**joblib, dill and NumPy object arrays** (`.joblib`, `.dill`, `.npy`, `.npz`) are pickle underneath — the everyday serialization of scikit-learn and scientific Python, carrying exactly the arbitrary-code-execution risk of a `.pt`, and opened by almost nothing that calls itself an SBOM tool. AIsbom reads them without importing joblib, dill or numpy: the compression is standard-library, the `.npy` header is parsed by hand, and nothing is ever unpickled.
|
|
387
|
+
|
|
388
|
+
Two things about these formats are worth stating plainly:
|
|
389
|
+
|
|
390
|
+
- **A raw array block does not end the scan.** joblib writes its pickle up to an array, dumps the raw buffer inline, then resumes pickling — which stops an opcode disassembly dead a few hundred bytes into the file. Since every real model has weights, "after the first array" is the natural place for a payload. The remaining bytes get a second pass that recovers globals directly, so a dangerous import behind an array block is still reported. That pass reads bytes rather than structure, so it recognises *known* sinks only: in strict mode it does not contribute unrecognized-import findings, and it cannot judge a dual-use constructor like `operator.methodcaller("system")`, because the argument that decides is a stack relationship it has no stack for.
|
|
391
|
+
- **A declared dtype never decides whether to look.** The `.npy` header is attacker-supplied, so a pickle sitting behind a header claiming `'<f8'` would otherwise be a one-line evasion. The data section is disassembled whatever the header says. What the header *does* affect is the reported risk: an array of ordinary numbers with no pickle in it is LOW, not "pickle present".
|
|
392
|
+
|
|
393
|
+
- **No limit ends in a clean verdict.** Four bounds apply here — the file-read budget, the `.npz` member count, the decompressor's output cap, and the disassembler's stream budget — and each one leaves bytes unexamined. Hitting any of them reports `MEDIUM (Pickle Scan Incomplete)` rather than passing the prefix off as the whole file. A limit that reports "clean" is not a safety measure; it is a hiding place with a length attached. A real finding still outranks the marker.
|
|
394
|
+
|
|
395
|
+
joblib's `lz4` and `zstd` codecs have no standard-library decompressor. Following the same reasoning as 7z containers, those are **named but not opened** — `MEDIUM (Unscanned Container: lz4)` — rather than pulling a native dependency into every install and every standalone binary. That is an honest "we did not read these bytes", which is a different answer from a clean scan.
|
|
396
|
+
|
|
368
397
|
**SafeTensors and GGUF** use binary formats with structured headers — AIsbom parses these directly to extract metadata (artifact names, license info, architecture details) without loading tensor weights.
|
|
369
398
|
|
|
370
399
|
**Keras** (`.keras`, `.h5`, `.hdf5`) carries a different execution vector: a `Lambda` layer stores an arbitrary Python callable in the model config as a base64-encoded marshalled code object, and `load_model` runs it. AIsbom reads the config out of both containers — the `.keras` zip and the legacy HDF5 attribute — and flags `Lambda` layers and embedded code objects as CRITICAL. The payload is identified from its header bytes and **never unmarshalled**. A truncated or corrupted container is still scanned rather than skipped, so damaging a file is not a way to hide a payload.
|
|
@@ -607,7 +607,11 @@ def scan(
|
|
|
607
607
|
)
|
|
608
608
|
console.print(table)
|
|
609
609
|
else:
|
|
610
|
-
|
|
610
|
+
# "No AI models found" is a claim about the target's contents, so it
|
|
611
|
+
# must not be printed when the target was never scanned at all (#125).
|
|
612
|
+
# The target error itself is rendered below.
|
|
613
|
+
if not any(e.get('target_error') for e in results['errors']):
|
|
614
|
+
console.print("[yellow]No AI models found.[/yellow]")
|
|
611
615
|
|
|
612
616
|
# LINT OUTPUT (Migration Report)
|
|
613
617
|
lint_failures = [a for a in results['artifacts'] if a.get('details', {}).get('lint_report')]
|
|
@@ -633,9 +637,20 @@ def scan(
|
|
|
633
637
|
if results['dependencies']:
|
|
634
638
|
console.print(f"\n📦 Found [bold]{len(results['dependencies'])}[/bold] Python libraries.")
|
|
635
639
|
|
|
636
|
-
#
|
|
637
|
-
#
|
|
638
|
-
|
|
640
|
+
# Unusable targets (#125): the path was missing or nothing could scan it,
|
|
641
|
+
# so nothing was examined. Printed to stderr like fetch failures — it is a
|
|
642
|
+
# failed instruction, not a finding about the model.
|
|
643
|
+
for err in [e for e in results['errors'] if e.get('target_error')]:
|
|
644
|
+
err_console.print(
|
|
645
|
+
f"[bold red]✖[/bold red] Cannot scan [yellow]{err['file']}[/yellow]: {err['error']}"
|
|
646
|
+
)
|
|
647
|
+
|
|
648
|
+
# Parse errors only — fetch failures and target errors already printed
|
|
649
|
+
# their own message to stderr and don't fit the "Could not parse" framing.
|
|
650
|
+
parse_errors = [
|
|
651
|
+
e for e in results['errors']
|
|
652
|
+
if not e.get('fetch_failure') and not e.get('target_error')
|
|
653
|
+
]
|
|
639
654
|
if parse_errors:
|
|
640
655
|
console.print("\n[bold red]⚠️ Errors Encountered:[/bold red]")
|
|
641
656
|
for err in parse_errors:
|
|
@@ -216,6 +216,11 @@ class BypassCase:
|
|
|
216
216
|
builder: Callable[[Path], None]
|
|
217
217
|
expected: str = "detected"
|
|
218
218
|
malicious: bool = True
|
|
219
|
+
# Why this case is currently not fully caught, for cases that aren't.
|
|
220
|
+
# Deliberately does *not* change `expected`: every evasion technique here
|
|
221
|
+
# remains one a correct scanner should catch, so the gate keeps counting it
|
|
222
|
+
# against us. This field explains a gap; it never excuses one.
|
|
223
|
+
limitation: str | None = None
|
|
219
224
|
|
|
220
225
|
|
|
221
226
|
@dataclass(frozen=True)
|
|
@@ -262,6 +267,15 @@ CASES: tuple[BypassCase, ...] = (
|
|
|
262
267
|
"meant picklescan never opened it, while the model still loaded."
|
|
263
268
|
),
|
|
264
269
|
builder=_build_nullifai_7z,
|
|
270
|
+
limitation=(
|
|
271
|
+
"AIsbom reports `CRITICAL (Non-Standard Container: 7z)` — the right severity, "
|
|
272
|
+
"but earned from the container rather than the payload. The archive is named, "
|
|
273
|
+
"never unpacked, so the `os.system` call inside is never disassembled and the "
|
|
274
|
+
"reported reason is not the real one. Unpacking 7z would mean a native "
|
|
275
|
+
"dependency in every install to cover one evasion class, which is not a "
|
|
276
|
+
"trade worth making; a user acting on this verdict is nonetheless correctly "
|
|
277
|
+
"warned off the file."
|
|
278
|
+
),
|
|
265
279
|
),
|
|
266
280
|
BypassCase(
|
|
267
281
|
id="nullifai-broken-stream",
|
|
@@ -386,6 +400,17 @@ CASES: tuple[BypassCase, ...] = (
|
|
|
386
400
|
"argument, so string-matching on the opcode argument sees nothing."
|
|
387
401
|
),
|
|
388
402
|
builder=_build_shadowpickle,
|
|
403
|
+
limitation=(
|
|
404
|
+
"AIsbom does resolve STACK_GLOBAL and reads the pair off the stack, so it "
|
|
405
|
+
"sees `collections.OrderedDict` — and that name is legitimately allowlisted, "
|
|
406
|
+
"because real state_dicts are OrderedDicts. Both modes therefore return only "
|
|
407
|
+
"`MEDIUM (Pickle Present)`, the baseline every pickle gets, rather than a "
|
|
408
|
+
"signal specific to this file. This is the ceiling on static allowlist "
|
|
409
|
+
"analysis: the call is indistinguishable from a legitimate call to an "
|
|
410
|
+
"allowlisted global, and flagging the shape would flag ordinary checkpoints. "
|
|
411
|
+
"Closing it needs evidence beyond the resolved name — argument shape, or "
|
|
412
|
+
"provenance — not a new entry on a blocklist."
|
|
413
|
+
),
|
|
389
414
|
),
|
|
390
415
|
)
|
|
391
416
|
|
|
@@ -546,14 +571,23 @@ def strip_generated_stamp(text: str) -> str:
|
|
|
546
571
|
).strip()
|
|
547
572
|
|
|
548
573
|
|
|
574
|
+
def is_caught(row: dict) -> bool:
|
|
575
|
+
"""
|
|
576
|
+
Whether a case counts as caught: named as a threat in at least one mode.
|
|
577
|
+
|
|
578
|
+
The headline count and the limitation-note gate must agree on this, or the
|
|
579
|
+
document can claim a case is uncaught while counting it among the wins.
|
|
580
|
+
"""
|
|
581
|
+
return row["blocklist"] == "detected" or row["strict"] == "detected"
|
|
582
|
+
|
|
583
|
+
|
|
549
584
|
def render_markdown(results: dict) -> str:
|
|
550
585
|
"""Render the human-readable scorecard published by the /blog article."""
|
|
551
586
|
cases = results["cases"]
|
|
552
587
|
evasions = [c for c in CASES if not is_control(c)]
|
|
553
588
|
controls = [c for c in CASES if is_control(c)]
|
|
554
589
|
|
|
555
|
-
detected = sum(1 for c in evasions if cases[c.id]
|
|
556
|
-
or cases[c.id]["strict"] == "detected")
|
|
590
|
+
detected = sum(1 for c in evasions if is_caught(cases[c.id]))
|
|
557
591
|
|
|
558
592
|
lines = [
|
|
559
593
|
"# Does AIsbom catch it? — pickle-evasion scorecard",
|
|
@@ -605,6 +639,16 @@ def render_markdown(results: dict) -> str:
|
|
|
605
639
|
case.description,
|
|
606
640
|
"",
|
|
607
641
|
]
|
|
642
|
+
# A limitation describes why a case is *not* caught, so it stops being
|
|
643
|
+
# true the moment detection improves. Gate it on the live verdict rather
|
|
644
|
+
# than publishing unconditionally: `--write` would otherwise regenerate
|
|
645
|
+
# the document with a stale claim, and the staleness test cannot catch
|
|
646
|
+
# that because it compares the committed file against this same
|
|
647
|
+
# renderer. test_limitation_note_is_dropped_once_a_case_is_caught pins
|
|
648
|
+
# the gate; test_no_limitation_note_on_a_caught_case fails the build so
|
|
649
|
+
# the dead text gets removed rather than silently lingering here.
|
|
650
|
+
if case.limitation and not is_caught(cases[case.id]):
|
|
651
|
+
lines += [f"**Current limitation:** {case.limitation}", ""]
|
|
608
652
|
|
|
609
653
|
return "\n".join(lines).rstrip() + "\n"
|
|
610
654
|
|
|
@@ -0,0 +1,304 @@
|
|
|
1
|
+
"""Locate the pickle stream inside the wrapper formats that carry one.
|
|
2
|
+
|
|
3
|
+
joblib, dill and numpy's object arrays are the everyday serialization formats of
|
|
4
|
+
scientific Python, and every one of them is pickle underneath — the same
|
|
5
|
+
arbitrary-code-execution risk as a bare ``.pkl``, wrapped in a container that
|
|
6
|
+
generic tooling does not open.
|
|
7
|
+
|
|
8
|
+
Nothing in this module imports joblib, numpy or dill. The compression is stdlib,
|
|
9
|
+
the ``.npy`` header is parsed by hand, and the payload is never unpickled — it is
|
|
10
|
+
handed back as bytes for the disassembler to read. That is deliberate twice over:
|
|
11
|
+
a scanner that loads the artifact to inspect it has already lost, and pulling a
|
|
12
|
+
scientific-Python stack into the package would put it into every install and
|
|
13
|
+
every standalone binary for the sake of reading bytes we can read ourselves.
|
|
14
|
+
|
|
15
|
+
Everything here is bounded. A container is an attacker-controlled input, so a
|
|
16
|
+
decompressor is never handed an unbounded output buffer and a declared header
|
|
17
|
+
length is never trusted far enough to allocate against it.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import ast
|
|
23
|
+
import bz2
|
|
24
|
+
import lzma
|
|
25
|
+
import re
|
|
26
|
+
import zlib
|
|
27
|
+
from typing import Any, Dict, Tuple
|
|
28
|
+
|
|
29
|
+
# Fallback ceiling on any single decompressed payload: a compression bomb must
|
|
30
|
+
# cost the same as a large ordinary file, not more. The decompressors below are
|
|
31
|
+
# all incremental and take this as a `max_length`, so the bomb is never expanded
|
|
32
|
+
# in the first place.
|
|
33
|
+
#
|
|
34
|
+
# Callers pass their own budget — the scanner passes the same figure it used to
|
|
35
|
+
# read the file, so a remote scan that fetched 2MB does not then expand 16.
|
|
36
|
+
DECOMPRESS_MAX_BYTES = 16 * 1024 * 1024
|
|
37
|
+
|
|
38
|
+
# `.npy` declares its header length in the file. Cap what we will honour: the
|
|
39
|
+
# real headers are a few dozen bytes, and a declared length is exactly the field
|
|
40
|
+
# an attacker would inflate.
|
|
41
|
+
NPY_MAX_HEADER_BYTES = 1 * 1024 * 1024
|
|
42
|
+
|
|
43
|
+
NPY_MAGIC = b"\x93NUMPY"
|
|
44
|
+
|
|
45
|
+
# joblib's pre-0.10 container: a two-byte tag, a decimal length, then zlib data.
|
|
46
|
+
# Still readable today, and still a way to carry a pickle past a scanner that
|
|
47
|
+
# only knows the modern layout.
|
|
48
|
+
JOBLIB_ZFILE_MAGIC = b"ZF"
|
|
49
|
+
|
|
50
|
+
# Compression formats joblib writes that the standard library can open. The
|
|
51
|
+
# label is what gets reported, so it is the name a user would recognise.
|
|
52
|
+
_GZIP = "gzip"
|
|
53
|
+
_BZ2 = "bz2"
|
|
54
|
+
_LZMA = "lzma"
|
|
55
|
+
_XZ = "xz"
|
|
56
|
+
_ZLIB = "zlib"
|
|
57
|
+
|
|
58
|
+
SUPPORTED_COMPRESSION = (_ZLIB, _GZIP, _BZ2, _LZMA, _XZ)
|
|
59
|
+
|
|
60
|
+
# Formats joblib also supports whose decompressors are not in the standard
|
|
61
|
+
# library. Naming one is not the same as reading it — see `describe_container`.
|
|
62
|
+
_MAGIC_UNSUPPORTED = (
|
|
63
|
+
(b"\x04\x22\x4d\x18", "lz4"),
|
|
64
|
+
(b"\x28\xb5\x2f\xfd", "zstd"),
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
_MAGIC_SUPPORTED = (
|
|
68
|
+
(b"\x1f\x8b", _GZIP),
|
|
69
|
+
(b"BZh", _BZ2),
|
|
70
|
+
(b"\xfd7zXZ\x00", _XZ),
|
|
71
|
+
(b"\x5d\x00\x00", _LZMA),
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _looks_like_zlib(head: bytes) -> bool:
|
|
76
|
+
"""True for a zlib stream header (RFC 1950).
|
|
77
|
+
|
|
78
|
+
zlib has no magic number, only a two-byte header with a checksum property:
|
|
79
|
+
the low nibble of the first byte is the compression method (8 = deflate) and
|
|
80
|
+
the 16-bit big-endian pair is a multiple of 31. Testing both is what keeps
|
|
81
|
+
an ordinary binary file that happens to start with 0x78 from being mistaken
|
|
82
|
+
for a compressed container.
|
|
83
|
+
"""
|
|
84
|
+
if len(head) < 2:
|
|
85
|
+
return False
|
|
86
|
+
if head[0] & 0x0F != 8:
|
|
87
|
+
return False
|
|
88
|
+
return ((head[0] << 8) | head[1]) % 31 == 0
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def detect_compression(head: bytes) -> Tuple[str | None, bool]:
|
|
92
|
+
"""Return ``(label, readable)`` for the compression ``head`` begins with.
|
|
93
|
+
|
|
94
|
+
``readable`` is False for a format we can name but not open, which is a
|
|
95
|
+
materially different answer from "not compressed" and is reported as such
|
|
96
|
+
rather than being quietly treated as a clean scan.
|
|
97
|
+
"""
|
|
98
|
+
if not head:
|
|
99
|
+
return None, False
|
|
100
|
+
for magic, label in _MAGIC_SUPPORTED:
|
|
101
|
+
if head.startswith(magic):
|
|
102
|
+
return label, True
|
|
103
|
+
for magic, label in _MAGIC_UNSUPPORTED:
|
|
104
|
+
if head.startswith(magic):
|
|
105
|
+
return label, False
|
|
106
|
+
if _looks_like_zlib(head):
|
|
107
|
+
return _ZLIB, True
|
|
108
|
+
return None, False
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def decompress(data: bytes, label: str, limit: int = DECOMPRESS_MAX_BYTES) -> Tuple[bytes | None, bool]:
|
|
112
|
+
"""Inflate ``data`` with the named codec; return ``(payload, complete)``.
|
|
113
|
+
|
|
114
|
+
``complete`` is False when the decompressor stopped before the end of the
|
|
115
|
+
stream — because it hit ``limit``, or because the compressed data itself was
|
|
116
|
+
cut short. That distinction has to reach the caller. A payload placed after
|
|
117
|
+
a large compressible value would otherwise sit in the part we never expanded
|
|
118
|
+
while the prefix scanned clean, and a cap nobody reports is somewhere to hide
|
|
119
|
+
a payload rather than a safety measure.
|
|
120
|
+
|
|
121
|
+
Returns whatever was recovered before an error when a stream is corrupt — a
|
|
122
|
+
damaged tail must not discard the front, the same reasoning that makes a
|
|
123
|
+
broken zip member worth reading anyway. ``(None, False)`` means nothing at
|
|
124
|
+
all could be read.
|
|
125
|
+
"""
|
|
126
|
+
if not data:
|
|
127
|
+
return None, False
|
|
128
|
+
try:
|
|
129
|
+
if label == _ZLIB:
|
|
130
|
+
engine = zlib.decompressobj()
|
|
131
|
+
elif label == _GZIP:
|
|
132
|
+
# 16 + MAX_WBITS selects the gzip wrapper.
|
|
133
|
+
engine = zlib.decompressobj(16 + zlib.MAX_WBITS)
|
|
134
|
+
elif label == _BZ2:
|
|
135
|
+
engine = bz2.BZ2Decompressor()
|
|
136
|
+
elif label in (_LZMA, _XZ):
|
|
137
|
+
# FORMAT_AUTO reads both the `.xz` container and the older
|
|
138
|
+
# standalone `.lzma` framing joblib still emits.
|
|
139
|
+
engine = lzma.LZMADecompressor(format=lzma.FORMAT_AUTO)
|
|
140
|
+
else:
|
|
141
|
+
return None, False
|
|
142
|
+
|
|
143
|
+
out = engine.decompress(data, limit)
|
|
144
|
+
# Every one of these decompressors exposes `eof`, which is True only
|
|
145
|
+
# once the stream's own end marker has been consumed.
|
|
146
|
+
return (out or None), bool(getattr(engine, "eof", False))
|
|
147
|
+
except Exception:
|
|
148
|
+
return None, False
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def unwrap_zfile(data: bytes, limit: int = DECOMPRESS_MAX_BYTES) -> Tuple[bytes | None, bool]:
|
|
152
|
+
"""Inflate joblib's legacy ``ZF`` container; return ``(payload, complete)``.
|
|
153
|
+
|
|
154
|
+
Layout is the tag, a space-padded decimal length, then a zlib stream. The
|
|
155
|
+
declared length is read past rather than trusted — the decompressor stops at
|
|
156
|
+
the real end of the stream, so a lie in that field buys nothing.
|
|
157
|
+
"""
|
|
158
|
+
if not data.startswith(JOBLIB_ZFILE_MAGIC):
|
|
159
|
+
return None, False
|
|
160
|
+
# The length field is fixed-width ASCII in every version that wrote it.
|
|
161
|
+
body = data[len(JOBLIB_ZFILE_MAGIC):]
|
|
162
|
+
match = re.match(rb"\s*([0-9]+)\s*", body[:32])
|
|
163
|
+
start = match.end() if match else 0
|
|
164
|
+
return decompress(body[start:], _ZLIB, limit)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def describe_container(data: bytes, limit: int = DECOMPRESS_MAX_BYTES) -> Dict[str, Any]:
|
|
168
|
+
"""Unwrap one layer of compression, if any, and say what was found.
|
|
169
|
+
|
|
170
|
+
Returns a dict with ``compression`` (label or None), ``readable`` (whether
|
|
171
|
+
we could open it), ``payload`` (the inner bytes, or None), and ``truncated``
|
|
172
|
+
(whether the decompressor stopped short of the stream's end). An unreadable
|
|
173
|
+
container yields a payload of None with the format still named, so the
|
|
174
|
+
caller can report "we did not read these bytes" instead of "clean" — and a
|
|
175
|
+
truncated one says so rather than passing off a prefix as the whole file.
|
|
176
|
+
"""
|
|
177
|
+
result: Dict[str, Any] = {
|
|
178
|
+
"compression": None, "readable": True, "payload": data, "truncated": False,
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
if data.startswith(JOBLIB_ZFILE_MAGIC):
|
|
182
|
+
payload, complete = unwrap_zfile(data, limit)
|
|
183
|
+
result.update(compression="zfile", readable=payload is not None,
|
|
184
|
+
payload=payload, truncated=payload is not None and not complete)
|
|
185
|
+
return result
|
|
186
|
+
|
|
187
|
+
label, readable = detect_compression(data[:16])
|
|
188
|
+
if label is None:
|
|
189
|
+
return result
|
|
190
|
+
|
|
191
|
+
result["compression"] = label
|
|
192
|
+
if not readable:
|
|
193
|
+
result.update(readable=False, payload=None)
|
|
194
|
+
return result
|
|
195
|
+
|
|
196
|
+
payload, complete = decompress(data, label, limit)
|
|
197
|
+
result.update(readable=payload is not None, payload=payload,
|
|
198
|
+
truncated=payload is not None and not complete)
|
|
199
|
+
return result
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def parse_npy_header(data: bytes) -> Dict[str, Any] | None:
|
|
203
|
+
"""Read a ``.npy`` header, returning where the data section starts.
|
|
204
|
+
|
|
205
|
+
The header is a Python dict *literal* — it is parsed with
|
|
206
|
+
``ast.literal_eval``, which builds values and cannot call anything, and only
|
|
207
|
+
after the declared length has been bounds-checked. Nothing here evaluates
|
|
208
|
+
the file's contents; a malformed header yields None rather than an
|
|
209
|
+
exception, because a file we cannot parse still gets scanned by the caller.
|
|
210
|
+
"""
|
|
211
|
+
if not data.startswith(NPY_MAGIC) or len(data) < 10:
|
|
212
|
+
return None
|
|
213
|
+
|
|
214
|
+
major = data[6]
|
|
215
|
+
if major == 1:
|
|
216
|
+
length_field, header_start = 2, 10
|
|
217
|
+
else:
|
|
218
|
+
# v2 and v3 widened the length field to four bytes.
|
|
219
|
+
length_field, header_start = 4, 12
|
|
220
|
+
if len(data) < header_start:
|
|
221
|
+
return None
|
|
222
|
+
|
|
223
|
+
header_len = int.from_bytes(data[8:8 + length_field], "little")
|
|
224
|
+
if header_len <= 0 or header_len > NPY_MAX_HEADER_BYTES:
|
|
225
|
+
return None
|
|
226
|
+
|
|
227
|
+
raw_header = data[header_start:header_start + header_len]
|
|
228
|
+
if len(raw_header) < header_len:
|
|
229
|
+
# Truncated: the data offset is still known, which is what matters.
|
|
230
|
+
raw_header = data[header_start:]
|
|
231
|
+
|
|
232
|
+
descr: Any = None
|
|
233
|
+
fortran_order = None
|
|
234
|
+
shape = None
|
|
235
|
+
try:
|
|
236
|
+
parsed = ast.literal_eval(raw_header.decode("latin-1").strip())
|
|
237
|
+
if isinstance(parsed, dict):
|
|
238
|
+
descr = parsed.get("descr")
|
|
239
|
+
fortran_order = parsed.get("fortran_order")
|
|
240
|
+
shape = parsed.get("shape")
|
|
241
|
+
except Exception:
|
|
242
|
+
# A header we cannot parse is not a reason to stop: fall back to reading
|
|
243
|
+
# the dtype string out of it, and failing that, carry on with the offset
|
|
244
|
+
# alone. The data section is scanned either way.
|
|
245
|
+
match = re.search(rb"'descr'\s*:\s*'([^']*)'", raw_header)
|
|
246
|
+
if match:
|
|
247
|
+
descr = match.group(1).decode("latin-1")
|
|
248
|
+
|
|
249
|
+
return {
|
|
250
|
+
"version": f"{major}.{data[7]}",
|
|
251
|
+
"descr": descr,
|
|
252
|
+
"fortran_order": fortran_order,
|
|
253
|
+
"shape": shape,
|
|
254
|
+
"data_offset": header_start + header_len,
|
|
255
|
+
"object_dtype": _is_object_dtype(descr),
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def _is_object_dtype(descr: Any) -> bool:
|
|
260
|
+
"""True if a dtype descriptor carries Python objects anywhere inside it.
|
|
261
|
+
|
|
262
|
+
``'|O'`` is the plain object array. A structured dtype is a list of
|
|
263
|
+
``(name, format)`` or ``(name, format, shape)`` entries, and only the
|
|
264
|
+
*format* element is a dtype — searching the whole entry reads field names as
|
|
265
|
+
type codes, so an ordinary ``[('FOO', '<i4')]`` came back True on the
|
|
266
|
+
strength of the letter O in its name.
|
|
267
|
+
|
|
268
|
+
Note this only chooses the reported risk label, never whether to scan: the
|
|
269
|
+
data section is disassembled whatever the header claims, because the header
|
|
270
|
+
is attacker-supplied. Getting this wrong is noise, not a bypass.
|
|
271
|
+
"""
|
|
272
|
+
if descr is None:
|
|
273
|
+
return False
|
|
274
|
+
if isinstance(descr, str):
|
|
275
|
+
# Strip the byte-order/size prefix, leaving the type character.
|
|
276
|
+
return "O" in descr.lstrip("|<>=")
|
|
277
|
+
if isinstance(descr, list):
|
|
278
|
+
# Structured: inspect each field's format, never its name or shape.
|
|
279
|
+
return any(
|
|
280
|
+
_is_object_dtype(field[1])
|
|
281
|
+
for field in descr
|
|
282
|
+
if isinstance(field, (list, tuple)) and len(field) >= 2
|
|
283
|
+
)
|
|
284
|
+
if isinstance(descr, tuple) and len(descr) == 2:
|
|
285
|
+
# A subarray format, `(format, shape)` — the dtype is the first element.
|
|
286
|
+
return _is_object_dtype(descr[0])
|
|
287
|
+
return False
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def npy_data_section(data: bytes, limit: int = DECOMPRESS_MAX_BYTES) -> Tuple[bytes | None, Dict[str, Any] | None]:
|
|
291
|
+
"""Return ``(data_section, header)`` for a ``.npy`` buffer.
|
|
292
|
+
|
|
293
|
+
The data section is returned whatever the declared dtype says. Deciding by
|
|
294
|
+
the header first would mean trusting an attacker-supplied field to tell us
|
|
295
|
+
whether to look — and a pickle behind a header claiming ``'<f8'`` is exactly
|
|
296
|
+
the shape that trick would take.
|
|
297
|
+
"""
|
|
298
|
+
header = parse_npy_header(data)
|
|
299
|
+
if header is None:
|
|
300
|
+
return None, None
|
|
301
|
+
offset = header["data_offset"]
|
|
302
|
+
if offset >= len(data):
|
|
303
|
+
return b"", header
|
|
304
|
+
return data[offset:offset + limit], header
|
|
@@ -23,8 +23,19 @@ _FRAMEWORK_TO_FORMAT = {
|
|
|
23
23
|
"GGUF": "gguf",
|
|
24
24
|
"Keras": "keras",
|
|
25
25
|
"ONNX": "onnx",
|
|
26
|
+
# The non-torch pickle carriers. `Pickle` shares PyTorch's token because the
|
|
27
|
+
# format genuinely is the same one; joblib and numpy get their own, because
|
|
28
|
+
# the container is a real difference a consumer may want to filter on.
|
|
29
|
+
"Pickle": "pickle",
|
|
30
|
+
"Joblib": "joblib",
|
|
31
|
+
"NumPy": "numpy",
|
|
26
32
|
}
|
|
27
33
|
|
|
34
|
+
# Formats whose findings are pickle opcodes. They share the `aisbom:pickle:*`
|
|
35
|
+
# vocabulary rather than each inventing a parallel one, so a consumer that knows
|
|
36
|
+
# how to read a threat off a `.pt` reads one off a `.joblib` unchanged.
|
|
37
|
+
_PICKLE_BEARING_FORMATS = {"pickle", "joblib", "numpy"}
|
|
38
|
+
|
|
28
39
|
|
|
29
40
|
def _format_for(art: Dict[str, Any]) -> str | None:
|
|
30
41
|
return _FRAMEWORK_TO_FORMAT.get(art.get("framework"))
|
|
@@ -62,12 +73,49 @@ def build_component_properties(art: Dict[str, Any]) -> List[Tuple[str, str]]:
|
|
|
62
73
|
details = art.get("details") or {}
|
|
63
74
|
props.append(("aisbom:format", fmt))
|
|
64
75
|
|
|
65
|
-
if fmt
|
|
76
|
+
if fmt in _PICKLE_BEARING_FORMATS:
|
|
66
77
|
threats = details.get("threats") or []
|
|
67
78
|
for threat in threats:
|
|
68
79
|
props.append(("aisbom:pickle:opcode", str(threat)))
|
|
69
80
|
props.append(("aisbom:pickle:opcode_count", str(len(threats))))
|
|
70
81
|
|
|
82
|
+
# The container the stream was found in — `bare`, `npy`, `npz`, or the
|
|
83
|
+
# compression joblib used. Absent for `.pt`, which has its own shape.
|
|
84
|
+
container = details.get("container")
|
|
85
|
+
if container:
|
|
86
|
+
props.append(("aisbom:pickle:container", str(container)))
|
|
87
|
+
if details.get("scan_incomplete"):
|
|
88
|
+
props.append(("aisbom:pickle:scan_incomplete", "true"))
|
|
89
|
+
if details.get("dill_code_objects"):
|
|
90
|
+
props.append(("aisbom:pickle:dill_code_objects", "true"))
|
|
91
|
+
|
|
92
|
+
if fmt == "joblib":
|
|
93
|
+
compression = details.get("compression")
|
|
94
|
+
if compression:
|
|
95
|
+
props.append(("aisbom:joblib:compression", str(compression)))
|
|
96
|
+
decompressed = details.get("decompressed_bytes")
|
|
97
|
+
if decompressed is not None:
|
|
98
|
+
props.append(("aisbom:joblib:decompressed_bytes", str(decompressed)))
|
|
99
|
+
|
|
100
|
+
elif fmt == "numpy":
|
|
101
|
+
dtype = details.get("dtype")
|
|
102
|
+
if dtype:
|
|
103
|
+
props.append(("aisbom:numpy:dtype", str(dtype)))
|
|
104
|
+
if details.get("object_dtype") is not None:
|
|
105
|
+
props.append((
|
|
106
|
+
"aisbom:numpy:object_dtype",
|
|
107
|
+
"true" if details.get("object_dtype") else "false",
|
|
108
|
+
))
|
|
109
|
+
shape = details.get("shape")
|
|
110
|
+
if shape:
|
|
111
|
+
props.append(("aisbom:numpy:shape", str(shape)))
|
|
112
|
+
npy_version = details.get("npy_version")
|
|
113
|
+
if npy_version:
|
|
114
|
+
props.append(("aisbom:numpy:npy_version", str(npy_version)))
|
|
115
|
+
internal_files = details.get("internal_files")
|
|
116
|
+
if internal_files is not None:
|
|
117
|
+
props.append(("aisbom:numpy:member_count", str(internal_files)))
|
|
118
|
+
|
|
71
119
|
elif fmt == "safetensors":
|
|
72
120
|
tensor_count = details.get("tensors")
|
|
73
121
|
if tensor_count is not None:
|