aisbom-cli 1.3.2__tar.gz → 1.3.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/PKG-INFO +1 -1
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/safety.py +76 -3
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/scanner.py +17 -9
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/pyproject.toml +1 -1
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/LICENSE +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/README.md +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/__init__.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/cli.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/corpus.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/diff.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/linter.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/loop_state.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/mock_generator.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/pickle_containers.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/properties.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/protobuf_reader.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/remote.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/spdx_gen.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/telemetry.py +0 -0
- {aisbom_cli-1.3.2 → aisbom_cli-1.3.3}/aisbom/version_check.py +0 -0
|
@@ -1158,6 +1158,77 @@ class _NullWriter:
|
|
|
1158
1158
|
pass
|
|
1159
1159
|
|
|
1160
1160
|
|
|
1161
|
+
# Opcodes whose argument runs to a newline and can plausibly *start* a real
|
|
1162
|
+
# pickle while carrying a lot of bytes. Each maps to the characters its
|
|
1163
|
+
# argument legitimately begins with, so a file is only re-read when its second
|
|
1164
|
+
# byte is consistent with the opcode its first byte claims to be.
|
|
1165
|
+
#
|
|
1166
|
+
# This guard is why a parquet file is not re-read: it opens `PAR1`, and while
|
|
1167
|
+
# `P` is the PERSID opcode, PERSID is not on this list — no real pickle starts
|
|
1168
|
+
# with a persistent id, and accepting it would mean re-reading every parquet,
|
|
1169
|
+
# ORC and arrow file in a tree up to the sniff cap.
|
|
1170
|
+
_NEWLINE_ARG_STARTS = {
|
|
1171
|
+
"S": b"'\"", # STRING — always quoted
|
|
1172
|
+
"V": None, # UNICODE — raw text, no reliable lead byte
|
|
1173
|
+
"I": b"0123456789+-", # INT
|
|
1174
|
+
"L": b"0123456789+-", # LONG
|
|
1175
|
+
"F": b"0123456789+-.", # FLOAT
|
|
1176
|
+
}
|
|
1177
|
+
|
|
1178
|
+
|
|
1179
|
+
def _first_argument_overruns(data: bytes) -> bool:
|
|
1180
|
+
"""Does `data`'s first opcode declare an argument longer than `data`?
|
|
1181
|
+
|
|
1182
|
+
Discovery reads a bounded head and only re-reads when the head looked like
|
|
1183
|
+
an unfinished pickle. "At least one opcode parsed" is the usual signal, but
|
|
1184
|
+
a pickle that opens with a single huge literal completes *no* opcodes — so
|
|
1185
|
+
without this check a payload hidden behind one 64KB literal is never found,
|
|
1186
|
+
while the documented limit claims 16MB. Measured before it was fixed: a
|
|
1187
|
+
65,000-byte pad was caught and a 70,000-byte pad was not.
|
|
1188
|
+
|
|
1189
|
+
Length-prefixed arguments are read exactly. Newline-terminated ones cannot
|
|
1190
|
+
be measured without the newline, so they are admitted only for the handful
|
|
1191
|
+
of opcodes that can really begin a pickle, and only when the following byte
|
|
1192
|
+
matches what that opcode's argument must start with.
|
|
1193
|
+
"""
|
|
1194
|
+
if not data:
|
|
1195
|
+
return False
|
|
1196
|
+
op = pickletools.code2op.get(chr(data[0]))
|
|
1197
|
+
if op is None or op.arg is None:
|
|
1198
|
+
return False
|
|
1199
|
+
|
|
1200
|
+
n = op.arg.n
|
|
1201
|
+
if n >= 0:
|
|
1202
|
+
# A fixed-width argument. These are a handful of bytes; they cannot be
|
|
1203
|
+
# what overran a 64KB read.
|
|
1204
|
+
return False
|
|
1205
|
+
|
|
1206
|
+
if n == pickletools.UP_TO_NEWLINE:
|
|
1207
|
+
if op.code not in _NEWLINE_ARG_STARTS:
|
|
1208
|
+
return False
|
|
1209
|
+
if b"\n" in data:
|
|
1210
|
+
# The terminator is already in view, so the argument is not what
|
|
1211
|
+
# ran off the end — something else failed, and re-reading will not
|
|
1212
|
+
# change that.
|
|
1213
|
+
return False
|
|
1214
|
+
lead = _NEWLINE_ARG_STARTS[op.code]
|
|
1215
|
+
return lead is None or (len(data) > 1 and data[1] in lead)
|
|
1216
|
+
|
|
1217
|
+
# Length-prefixed: read the declared size and believe it only far enough to
|
|
1218
|
+
# decide whether a bigger read would reach the end of the argument.
|
|
1219
|
+
widths = {
|
|
1220
|
+
pickletools.TAKEN_FROM_ARGUMENT1: 1,
|
|
1221
|
+
pickletools.TAKEN_FROM_ARGUMENT4: 4,
|
|
1222
|
+
pickletools.TAKEN_FROM_ARGUMENT4U: 4,
|
|
1223
|
+
pickletools.TAKEN_FROM_ARGUMENT8U: 8,
|
|
1224
|
+
}
|
|
1225
|
+
width = widths.get(n)
|
|
1226
|
+
if width is None or len(data) < 1 + width:
|
|
1227
|
+
return False
|
|
1228
|
+
declared = int.from_bytes(data[1 : 1 + width], "little")
|
|
1229
|
+
return 1 + width + declared > len(data)
|
|
1230
|
+
|
|
1231
|
+
|
|
1161
1232
|
def head_looks_like_pickle(data: bytes) -> tuple[bool, bool]:
|
|
1162
1233
|
"""Does the head of a file begin with a genuinely valid pickle?
|
|
1163
1234
|
|
|
@@ -1201,9 +1272,11 @@ def head_looks_like_pickle(data: bytes) -> tuple[bool, bool]:
|
|
|
1201
1272
|
pass
|
|
1202
1273
|
|
|
1203
1274
|
if stop_at is None:
|
|
1204
|
-
# No complete pickle in view. Worth a bigger read
|
|
1205
|
-
#
|
|
1206
|
-
|
|
1275
|
+
# No complete pickle in view. Worth a bigger read if something parsed
|
|
1276
|
+
# — or if nothing parsed *because* the very first opcode carries an
|
|
1277
|
+
# argument that runs off the end of the buffer, which is the one way a
|
|
1278
|
+
# genuine pickle yields no opcodes at all.
|
|
1279
|
+
return (False, parsed > 0 or _first_argument_overruns(data))
|
|
1207
1280
|
|
|
1208
1281
|
try:
|
|
1209
1282
|
pickletools.dis(io.BytesIO(data[:stop_at]), out=_NullWriter())
|
|
@@ -354,15 +354,23 @@ class DeepScanner:
|
|
|
354
354
|
def _sniff_is_pickle(self, full_path: Path) -> bool:
|
|
355
355
|
"""Decide by content whether an unclaimed file is a pickle.
|
|
356
356
|
|
|
357
|
-
Reads a small head first and
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
357
|
+
Reads a small head first and re-reads with a larger budget only when
|
|
358
|
+
that head looked like an unfinished pickle, which is what keeps a walk
|
|
359
|
+
over a tree full of parquet and PNGs cheap: those settle on the first
|
|
360
|
+
read and are never opened a second time.
|
|
361
|
+
|
|
362
|
+
Two things can leave a head unfinished, and both must count. Opcodes
|
|
363
|
+
may parse cleanly without reaching STOP — an ordinary large pickle. Or
|
|
364
|
+
*no* opcode completes because the first one carries an argument running
|
|
365
|
+
past the buffer; `_first_argument_overruns` recognizes that case. The
|
|
366
|
+
original rule counted only the first, so one 64KB literal ahead of a
|
|
367
|
+
payload hid it from discovery entirely while the documented limit
|
|
368
|
+
claimed 16MB.
|
|
369
|
+
|
|
370
|
+
Known limit, stated rather than hidden: past PICKLE_SNIFF_MAX_BYTES we
|
|
371
|
+
stop looking, so a payload behind a literal larger than that is not
|
|
372
|
+
discovered by content. Files carrying a recognized model extension are
|
|
373
|
+
unaffected — they never reach this path.
|
|
366
374
|
"""
|
|
367
375
|
try:
|
|
368
376
|
size = full_path.stat().st_size
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "aisbom-cli"
|
|
3
|
-
version = "1.3.
|
|
3
|
+
version = "1.3.3"
|
|
4
4
|
description = "An AI Supply Chain security tool that that detects Pickle bombs and generates CycloneDX SBOMs for Machine Learning models."
|
|
5
5
|
authors = ["Ajoy L <lab700xdev@gmail.com>"]
|
|
6
6
|
readme = "README.md"
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|