wasm-tools 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
wasm_tools/__init__.py ADDED
@@ -0,0 +1,11 @@
1
+ from .api import (
2
+ parse_wasm_bytes,
3
+ parse_wasm_bytes_json,
4
+ parse_wasm_file,
5
+ parse_wasm_file_json,
6
+ )
7
+ from .models import ObjdumpMode, ObjdumpOptions, ObjdumpState
8
+ from .parser import BinaryReader
9
+ from .visitor import BinaryReaderObjdumpDisassemble, BinaryReaderObjdumpPrepass
10
+
11
+ __version__ = "1.0.0"
wasm_tools/api.py ADDED
@@ -0,0 +1,617 @@
1
+ """Library-first JSON-friendly API for WebAssembly parsing results."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from typing import Any
7
+
8
+ from .models import ObjdumpMode, ObjdumpOptions, ObjdumpState, SECTION_NAMES
9
+ from .parser import BinaryReader
10
+ from .visitor import BinaryReaderNop, BinaryReaderObjdumpPrepass
11
+
12
+
13
+ _HIGH_RISK_FINDING_WEIGHTS = {
14
+ "WASM-CAP-001": 30,
15
+ "WASM-CFG-002": 25,
16
+ "WASM-DOS-003": 30,
17
+ "WASM-LOOP-004": 15,
18
+ }
19
+
20
+
21
+ def _tier_from_score(score: int) -> str:
22
+ if score >= 70:
23
+ return "high"
24
+ if score >= 40:
25
+ return "medium"
26
+ if score > 0:
27
+ return "low"
28
+ return "none"
29
+
30
+
31
+ class _BinaryReaderJsonCollector(BinaryReaderNop):
32
+ """Collect parser callbacks into a structured report dictionary."""
33
+
34
+ def __init__(self, filename: str, state: ObjdumpState) -> None:
35
+ """Initialize per-module collection state."""
36
+ self.filename = filename
37
+ self.objdump_state = state
38
+ self.module_version: int | None = None
39
+ self.offset = 0
40
+ self.current_opcode: Any | None = None
41
+ self.errors: list[str] = []
42
+ self.sections: list[dict[str, Any]] = []
43
+ self._functions: dict[int, dict[str, Any]] = {}
44
+ self._current_function: dict[str, Any] | None = None
45
+ self._pending_instruction: dict[str, Any] | None = None
46
+
47
+ def begin_module(self, version: int) -> None:
48
+ """Capture the module version from the binary header."""
49
+ self.module_version = version
50
+
51
+ def begin_section(self, section_index: int, section_code: int, size: int) -> None:
52
+ """Record section metadata and byte offsets."""
53
+ self.sections.append(
54
+ {
55
+ "index": section_index,
56
+ "id": section_code,
57
+ "name": SECTION_NAMES.get(section_code, "Unknown"),
58
+ "size": size,
59
+ "offset": self.offset,
60
+ }
61
+ )
62
+
63
+ def begin_custom_section(
64
+ self, section_index: int, size: int, section_name: str
65
+ ) -> None:
66
+ """Attach custom section names to the last-recorded section."""
67
+ if self.sections:
68
+ self.sections[-1]["name"] = section_name
69
+
70
+ def begin_function_body(self, index: int, size: int) -> None:
71
+ """Start collecting instructions for a function body."""
72
+ self._current_function = {
73
+ "index": index,
74
+ "name": self.objdump_state.function_names.get(index, ""),
75
+ "signature_index": self.objdump_state.function_types.get(index),
76
+ "offset": self.offset,
77
+ "body_size": size,
78
+ "instructions": [],
79
+ }
80
+ self._functions[index] = self._current_function
81
+
82
+ def end_function_body(self, index: int) -> None:
83
+ """Close function collection and flush unfinished instruction records."""
84
+ if self._pending_instruction and self._current_function:
85
+ self._pending_instruction["decode_incomplete"] = True
86
+ self._current_function["instructions"].append(self._pending_instruction)
87
+ self._pending_instruction = None
88
+ self._current_function = None
89
+
90
+ def on_opcode(self, opcode: Any) -> None:
91
+ """Create a pending instruction that later callbacks will complete."""
92
+ self.current_opcode = opcode
93
+ self._pending_instruction = {
94
+ "offset": self.offset,
95
+ "opcode": getattr(opcode, "name", "unknown"),
96
+ "immediates": [],
97
+ }
98
+
99
+ def on_opcode_bare(self) -> None:
100
+ """Finalize an opcode that has no immediate operands."""
101
+ self._finalize_instruction()
102
+
103
+ def on_end_expr(self) -> None:
104
+ """Finalize an `end` opcode that closes an expression."""
105
+ self._finalize_instruction()
106
+
107
+ def on_opcode_block_sig(self, sig: int) -> None:
108
+ """Finalize block-like opcodes with signature immediates."""
109
+ self._append_immediates(sig)
110
+
111
+ def on_opcode_index(self, idx: int) -> None:
112
+ """Finalize index-based opcodes."""
113
+ self._append_immediates(idx)
114
+
115
+ def on_opcode_uint32(self, val: int) -> None:
116
+ """Finalize opcodes with one 32-bit integer immediate."""
117
+ self._append_immediates(val)
118
+
119
+ def on_opcode_uint64(self, val: int) -> None:
120
+ """Finalize opcodes with one 64-bit integer immediate."""
121
+ self._append_immediates(val)
122
+
123
+ def on_opcode_f32(self, val: float) -> None:
124
+ """Finalize opcodes with one 32-bit float immediate."""
125
+ self._append_immediates(val)
126
+
127
+ def on_opcode_f64(self, val: float) -> None:
128
+ """Finalize opcodes with one 64-bit float immediate."""
129
+ self._append_immediates(val)
130
+
131
+ def on_opcode_uint32_uint32(self, v1: int, v2: int) -> None:
132
+ """Finalize opcodes with two integer immediates."""
133
+ self._append_immediates(v1, v2)
134
+
135
+ def on_call_indirect_expr(self, sig: int, tab: int) -> None:
136
+ """Finalize call_indirect with signature and table indices."""
137
+ self._append_immediates(sig, tab)
138
+
139
+ def on_error(self, message: str) -> None:
140
+ """Capture parser errors for programmatic callers."""
141
+ self.errors.append(message)
142
+
143
+ def _append_immediates(self, *values: Any) -> None:
144
+ """Append immediate values and finalize the current instruction."""
145
+ if self._pending_instruction is None:
146
+ return
147
+ self._pending_instruction["immediates"].extend(values)
148
+ self._finalize_instruction()
149
+
150
+ def _finalize_instruction(self) -> None:
151
+ """Add the pending instruction to the active function."""
152
+ if self._pending_instruction and self._current_function:
153
+ self._current_function["instructions"].append(self._pending_instruction)
154
+ self._pending_instruction = None
155
+
156
+ def build_report(self) -> dict[str, Any]:
157
+ """Build the final JSON-ready report for this module."""
158
+ functions = [self._functions[idx] for idx in sorted(self._functions)]
159
+ for fn in functions:
160
+ fn["instruction_count"] = len(fn["instructions"])
161
+
162
+ state = self.objdump_state
163
+
164
+ types = [
165
+ {"index": i, "params": list(t.params), "results": list(t.results)}
166
+ for i, t in enumerate(state.types)
167
+ if t is not None
168
+ ]
169
+
170
+ imports = []
171
+ for imp in state.imports:
172
+ if imp is None:
173
+ continue
174
+ rec: dict[str, Any] = {
175
+ "index": imp.index,
176
+ "module": imp.module,
177
+ "name": imp.name,
178
+ "kind": imp.kind,
179
+ }
180
+ if imp.kind == "func":
181
+ rec["type_index"] = imp.type_index
182
+ elif imp.kind == "table":
183
+ rec["ref_type"] = imp.table_ref_type
184
+ if imp.table_limits:
185
+ rec["limits"] = {
186
+ "min": imp.table_limits.minimum,
187
+ "max": imp.table_limits.maximum,
188
+ }
189
+ elif imp.kind == "memory":
190
+ if imp.mem_limits:
191
+ rec["limits"] = {
192
+ "min": imp.mem_limits.minimum,
193
+ "max": imp.mem_limits.maximum,
194
+ "is_64": imp.mem_limits.is_64,
195
+ }
196
+ elif imp.kind == "global":
197
+ rec["valtype"] = imp.global_valtype
198
+ rec["mutable"] = imp.global_mutable
199
+ elif imp.kind == "tag":
200
+ rec["type_index"] = imp.tag_type_index
201
+ imports.append(rec)
202
+
203
+ exports = [
204
+ {
205
+ "index": exp.index,
206
+ "name": exp.name,
207
+ "kind": exp.kind,
208
+ "ref_index": exp.ref_index,
209
+ }
210
+ for exp in state.exports
211
+ if exp is not None
212
+ ]
213
+
214
+ globals_ = [
215
+ {
216
+ "index": g.index,
217
+ "valtype": g.valtype,
218
+ "mutable": g.mutable,
219
+ "init": g.init_expr,
220
+ }
221
+ for g in state.globals
222
+ if g is not None
223
+ ]
224
+
225
+ tables = [
226
+ {
227
+ "index": t.index,
228
+ "ref_type": t.ref_type,
229
+ "limits": {"min": t.limits.minimum, "max": t.limits.maximum},
230
+ }
231
+ for t in state.tables
232
+ if t is not None
233
+ ]
234
+
235
+ memories = [
236
+ {
237
+ "index": m.index,
238
+ "limits": {
239
+ "min": m.limits.minimum,
240
+ "max": m.limits.maximum,
241
+ "is_64": m.limits.is_64,
242
+ },
243
+ }
244
+ for m in state.memories
245
+ if m is not None
246
+ ]
247
+
248
+ data_segments = [
249
+ {
250
+ "index": d.index,
251
+ "mode": d.mode,
252
+ "memory_index": d.memory_index,
253
+ "offset": d.offset_expr,
254
+ "size": d.size,
255
+ }
256
+ for d in state.data_segments
257
+ if d is not None
258
+ ]
259
+
260
+ elements = [
261
+ {
262
+ "index": e.index,
263
+ "mode": e.mode,
264
+ "ref_type": e.ref_type,
265
+ "table_index": e.table_index,
266
+ "offset": e.offset_expr,
267
+ "count": e.count,
268
+ "func_indices": list(e.func_indices),
269
+ }
270
+ for e in state.elements
271
+ if e is not None
272
+ ]
273
+
274
+ tags = [
275
+ {"index": tg.index, "type_index": tg.type_index}
276
+ for tg in state.tags
277
+ if tg is not None
278
+ ]
279
+
280
+ analysis = self._build_analysis(functions, imports, data_segments)
281
+
282
+ return {
283
+ "file": self.filename,
284
+ "module_version": self.module_version,
285
+ "section_count": len(self.sections),
286
+ "sections": self.sections,
287
+ "function_count": len(functions),
288
+ "functions": functions,
289
+ "types": types,
290
+ "imports": imports,
291
+ "exports": exports,
292
+ "globals": globals_,
293
+ "tables": tables,
294
+ "memories": memories,
295
+ "data_segments": data_segments,
296
+ "elements": elements,
297
+ "tags": tags,
298
+ "start_function": state.start_function,
299
+ "analysis": analysis,
300
+ "errors": self.errors,
301
+ }
302
+
303
+ def _build_analysis(
304
+ self,
305
+ functions: list[dict[str, Any]],
306
+ imports: list[dict[str, Any]],
307
+ data_segments: list[dict[str, Any]],
308
+ ) -> dict[str, Any]:
309
+ op_counts: dict[str, int] = {}
310
+ loop_max_depth = 0
311
+ loop_memory_ops = 0
312
+ loop_branch_ops = 0
313
+ indirect_call_ops = 0
314
+ table_mutation_ops = 0
315
+ memory_grow_ops = 0
316
+ bulk_memory_ops = 0
317
+ memory_access_ops = 0
318
+ dynamic_funcs: set[int] = set()
319
+ table_mutation_funcs: set[int] = set()
320
+ loop_memory_funcs: set[int] = set()
321
+
322
+ block_openers = {"block", "loop", "if", "try_table"}
323
+ table_mutators = {
324
+ "table.set",
325
+ "table.grow",
326
+ "table.fill",
327
+ "table.copy",
328
+ "table.init",
329
+ "elem.drop",
330
+ }
331
+ bulk_memory = {"memory.init", "memory.copy", "memory.fill", "data.drop"}
332
+
333
+ for fn in functions:
334
+ active_loops = 0
335
+ block_stack: list[str] = []
336
+ for ins in fn.get("instructions", []):
337
+ op = ins.get("opcode", "")
338
+ op_counts[op] = op_counts.get(op, 0) + 1
339
+
340
+ if op in block_openers:
341
+ block_stack.append(op)
342
+ if op == "loop":
343
+ active_loops += 1
344
+ loop_max_depth = max(loop_max_depth, active_loops)
345
+ elif op == "end" and block_stack:
346
+ if block_stack.pop() == "loop":
347
+ active_loops = max(0, active_loops - 1)
348
+
349
+ is_memory_access = (
350
+ op.startswith("i32.load")
351
+ or op.startswith("i64.load")
352
+ or op.startswith("f32.load")
353
+ or op.startswith("f64.load")
354
+ or op.startswith("v128.load")
355
+ or op.startswith("i32.store")
356
+ or op.startswith("i64.store")
357
+ or op.startswith("f32.store")
358
+ or op.startswith("f64.store")
359
+ or op.startswith("v128.store")
360
+ or op.startswith("memory.")
361
+ )
362
+ if is_memory_access:
363
+ memory_access_ops += 1
364
+ if active_loops > 0:
365
+ loop_memory_ops += 1
366
+ loop_memory_funcs.add(fn.get("index", -1))
367
+
368
+ if op in {"br", "br_if", "br_table"} and active_loops > 0:
369
+ loop_branch_ops += 1
370
+
371
+ if op == "memory.grow":
372
+ memory_grow_ops += 1
373
+
374
+ if op in bulk_memory:
375
+ bulk_memory_ops += 1
376
+
377
+ if op in {
378
+ "call_indirect",
379
+ "return_call_indirect",
380
+ "call_ref",
381
+ "return_call_ref",
382
+ }:
383
+ indirect_call_ops += 1
384
+ dynamic_funcs.add(fn.get("index", -1))
385
+
386
+ if op in table_mutators:
387
+ table_mutation_ops += 1
388
+ table_mutation_funcs.add(fn.get("index", -1))
389
+
390
+ capabilities = sorted(self._capabilities_from_imports(imports))
391
+ findings = self._build_findings(
392
+ capabilities=capabilities,
393
+ memory_grow_ops=memory_grow_ops,
394
+ bulk_memory_ops=bulk_memory_ops,
395
+ memory_access_ops=memory_access_ops,
396
+ loop_max_depth=loop_max_depth,
397
+ loop_memory_ops=loop_memory_ops,
398
+ loop_branch_ops=loop_branch_ops,
399
+ indirect_call_ops=indirect_call_ops,
400
+ table_mutation_ops=table_mutation_ops,
401
+ dynamic_funcs=dynamic_funcs,
402
+ table_mutation_funcs=table_mutation_funcs,
403
+ loop_memory_funcs=loop_memory_funcs,
404
+ )
405
+
406
+ cap_risk = {
407
+ "fs.path": 10,
408
+ "fs.io": 8,
409
+ "network": 12,
410
+ "process.terminate": 8,
411
+ "crypto.random": 2,
412
+ "clock.high_res": 4,
413
+ "host.logging": 2,
414
+ "host.memory": 3,
415
+ "host.table": 3,
416
+ "host.global": 2,
417
+ "host.tag": 2,
418
+ "js.host": 6,
419
+ }
420
+ capability_score = sum(cap_risk.get(cap, 0) for cap in capabilities)
421
+ findings_score = sum(
422
+ _HIGH_RISK_FINDING_WEIGHTS.get(f["id"], 5) for f in findings
423
+ )
424
+ risk_score = min(100, capability_score + findings_score)
425
+
426
+ return {
427
+ "summary": {
428
+ "risk_score": risk_score,
429
+ "risk_tier": _tier_from_score(risk_score),
430
+ "finding_count": len(findings),
431
+ },
432
+ "capabilities": capabilities,
433
+ "profiles": {
434
+ "memory": {
435
+ "memory_access_ops": memory_access_ops,
436
+ "memory_grow_ops": memory_grow_ops,
437
+ "bulk_memory_ops": bulk_memory_ops,
438
+ "data_segment_total_bytes": sum(
439
+ ds.get("size", 0) for ds in data_segments
440
+ ),
441
+ },
442
+ "control_flow": {
443
+ "indirect_call_ops": indirect_call_ops,
444
+ "table_mutation_ops": table_mutation_ops,
445
+ },
446
+ "compute": {
447
+ "max_loop_depth": loop_max_depth,
448
+ "loop_memory_ops": loop_memory_ops,
449
+ "loop_branch_ops": loop_branch_ops,
450
+ },
451
+ },
452
+ "findings": findings,
453
+ }
454
+
455
+ def _capabilities_from_imports(self, imports: list[dict[str, Any]]) -> set[str]:
456
+ capabilities: set[str] = set()
457
+ for imp in imports:
458
+ module = str(imp.get("module", ""))
459
+ name = str(imp.get("name", ""))
460
+ kind = str(imp.get("kind", ""))
461
+
462
+ if kind == "memory":
463
+ capabilities.add("host.memory")
464
+ elif kind == "table":
465
+ capabilities.add("host.table")
466
+ elif kind == "global":
467
+ capabilities.add("host.global")
468
+ elif kind == "tag":
469
+ capabilities.add("host.tag")
470
+
471
+ if module == "wasi_snapshot_preview1":
472
+ if name.startswith("path_"):
473
+ capabilities.add("fs.path")
474
+ if name.startswith("fd_"):
475
+ capabilities.add("fs.io")
476
+ if name.startswith("sock_"):
477
+ capabilities.add("network")
478
+ if name == "random_get":
479
+ capabilities.add("crypto.random")
480
+ if name == "proc_exit":
481
+ capabilities.add("process.terminate")
482
+ if name in {"clock_time_get", "poll_oneoff"}:
483
+ capabilities.add("clock.high_res")
484
+
485
+ if module in {"env", "wbg", "js"}:
486
+ if any(token in name for token in ("log", "print")):
487
+ capabilities.add("host.logging")
488
+ if "abort" in name or name == "proc_exit":
489
+ capabilities.add("process.terminate")
490
+ if module in {"wbg", "js"}:
491
+ capabilities.add("js.host")
492
+ return capabilities
493
+
494
+ def _build_findings(self, **kwargs: Any) -> list[dict[str, Any]]:
495
+ capabilities: list[str] = kwargs["capabilities"]
496
+ indirect_call_ops: int = kwargs["indirect_call_ops"]
497
+ table_mutation_ops: int = kwargs["table_mutation_ops"]
498
+ dynamic_funcs: set[int] = kwargs["dynamic_funcs"]
499
+ table_mutation_funcs: set[int] = kwargs["table_mutation_funcs"]
500
+ memory_grow_ops: int = kwargs["memory_grow_ops"]
501
+ loop_memory_ops: int = kwargs["loop_memory_ops"]
502
+ loop_memory_funcs: set[int] = kwargs["loop_memory_funcs"]
503
+ loop_max_depth: int = kwargs["loop_max_depth"]
504
+
505
+ findings: list[dict[str, Any]] = []
506
+
507
+ if {"fs.path", "network"}.issubset(set(capabilities)):
508
+ findings.append(
509
+ {
510
+ "id": "WASM-CAP-001",
511
+ "title": "Filesystem and network host capabilities are both imported",
512
+ "severity": "high",
513
+ "confidence": "high",
514
+ "evidence": {"capabilities": ["fs.path", "network"]},
515
+ "remediation": "Review sandbox policy and restrict host imports to least privilege.",
516
+ }
517
+ )
518
+
519
+ if indirect_call_ops > 0 and table_mutation_ops > 0:
520
+ findings.append(
521
+ {
522
+ "id": "WASM-CFG-002",
523
+ "title": "Dynamic dispatch surface includes mutable table operations",
524
+ "severity": "high",
525
+ "confidence": "medium",
526
+ "evidence": {
527
+ "indirect_call_ops": indirect_call_ops,
528
+ "table_mutation_ops": table_mutation_ops,
529
+ "dynamic_funcs": sorted(i for i in dynamic_funcs if i >= 0),
530
+ "table_mutation_funcs": sorted(
531
+ i for i in table_mutation_funcs if i >= 0
532
+ ),
533
+ },
534
+ "remediation": "Prefer immutable dispatch tables or add strict index validation around table mutations.",
535
+ }
536
+ )
537
+
538
+ if memory_grow_ops > 0 and loop_memory_ops > 0:
539
+ findings.append(
540
+ {
541
+ "id": "WASM-DOS-003",
542
+ "title": "Memory growth occurs in loop context",
543
+ "severity": "high",
544
+ "confidence": "medium",
545
+ "evidence": {
546
+ "memory_grow_ops": memory_grow_ops,
547
+ "loop_memory_ops": loop_memory_ops,
548
+ "functions": sorted(i for i in loop_memory_funcs if i >= 0),
549
+ },
550
+ "remediation": "Apply growth limits and add explicit loop bounds when executing untrusted inputs.",
551
+ }
552
+ )
553
+
554
+ if loop_max_depth >= 3:
555
+ findings.append(
556
+ {
557
+ "id": "WASM-LOOP-004",
558
+ "title": "Deep loop nesting increases computational amplification risk",
559
+ "severity": "medium",
560
+ "confidence": "medium",
561
+ "evidence": {"max_loop_depth": loop_max_depth},
562
+ "remediation": "Add runtime fuel/step limits or watchdog timeouts for untrusted modules.",
563
+ }
564
+ )
565
+
566
+ return findings
567
+
568
+
569
+ def parse_wasm_bytes(data: bytes, filename: str = "<memory>") -> dict[str, Any]:
570
+ """Parse raw wasm bytes and return a structured report dictionary."""
571
+ options = ObjdumpOptions(mode=ObjdumpMode.RAW_DATA, filename=filename)
572
+ state = ObjdumpState()
573
+
574
+ # Pass 1: gather names, types, and all section metadata into shared state.
575
+ BinaryReader(data, BinaryReaderObjdumpPrepass(data, options, state)).read_module()
576
+
577
+ # Pass 2: collect function bodies and instruction streams.
578
+ collector = _BinaryReaderJsonCollector(filename, state)
579
+ BinaryReader(data, collector).read_module()
580
+ return collector.build_report()
581
+
582
+
583
+ def parse_wasm_file(path: str) -> dict[str, Any]:
584
+ """Read and parse a wasm file path into a structured report dictionary."""
585
+ try:
586
+ with open(path, "rb") as wasm_file:
587
+ return parse_wasm_bytes(wasm_file.read(), filename=path)
588
+ except OSError as exc:
589
+ return {
590
+ "file": path,
591
+ "module_version": None,
592
+ "section_count": 0,
593
+ "sections": [],
594
+ "function_count": 0,
595
+ "functions": [],
596
+ "analysis": {
597
+ "summary": {"risk_score": 0, "risk_tier": "none", "finding_count": 0},
598
+ "capabilities": [],
599
+ "profiles": {},
600
+ "findings": [],
601
+ },
602
+ "errors": [f"Error reading {path}: {exc}"],
603
+ }
604
+
605
+
606
+ def parse_wasm_bytes_json(
607
+ data: bytes, filename: str = "<memory>", indent: int = 2
608
+ ) -> str:
609
+ """Parse wasm bytes and serialize the structured report as JSON text."""
610
+ report = parse_wasm_bytes(data, filename=filename)
611
+ return json.dumps(report, ensure_ascii=False, indent=indent)
612
+
613
+
614
+ def parse_wasm_file_json(path: str, indent: int = 2) -> str:
615
+ """Parse a wasm file and serialize the structured report as JSON text."""
616
+ report = parse_wasm_file(path)
617
+ return json.dumps(report, ensure_ascii=False, indent=indent)