pwasm 0.1a0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
pwasm/__init__.py ADDED
@@ -0,0 +1,44 @@
1
+ """Pure Python WebAssembly Runtime.
2
+
3
+ A WebAssembly runtime implemented in pure Python with no dependencies.
4
+ """
5
+
6
+ from .decoder import (
7
+ decode_module,
8
+ BinaryReader,
9
+ decode_unsigned_leb128,
10
+ decode_signed_leb128,
11
+ )
12
+ from .errors import WasmError, DecodeError, ValidationError, TrapError, LinkError
13
+ from .types import Module, FuncType, Function, Export, Import, Instruction
14
+ from .executor import instantiate, Instance
15
+
16
+ __version__ = "0.1.0"
17
+
18
+ __all__ = [
19
+ # Main API
20
+ "decode_module",
21
+ "instantiate",
22
+ "Instance",
23
+ # Decoder internals (for testing)
24
+ "BinaryReader",
25
+ "decode_unsigned_leb128",
26
+ "decode_signed_leb128",
27
+ # Types
28
+ "Module",
29
+ "FuncType",
30
+ "Function",
31
+ "Export",
32
+ "Import",
33
+ "Instruction",
34
+ # Errors
35
+ "WasmError",
36
+ "DecodeError",
37
+ "ValidationError",
38
+ "TrapError",
39
+ "LinkError",
40
+ ]
41
+
42
+
43
+ def hello() -> str:
44
+ return "Hello from pwasm!"
pwasm/decoder.py ADDED
@@ -0,0 +1,533 @@
1
+ """WebAssembly binary format decoder."""
2
+
3
+ import struct
4
+ from typing import BinaryIO
5
+ from pathlib import Path
6
+
7
+ from .errors import DecodeError
8
+ from .types import (
9
+ Module,
10
+ FuncType,
11
+ Function,
12
+ Export,
13
+ Import,
14
+ Memory,
15
+ Table,
16
+ Global,
17
+ GlobalType,
18
+ Element,
19
+ Data,
20
+ Limits,
21
+ Instruction,
22
+ VALTYPE_ENCODING,
23
+ EXPORT_KIND_ENCODING,
24
+ )
25
+ from . import opcodes
26
+
27
+
28
+ # WASM magic number and version
29
+ WASM_MAGIC = b"\x00asm"
30
+ WASM_VERSION = 1
31
+
32
+ # Section IDs
33
+ SECTION_CUSTOM = 0
34
+ SECTION_TYPE = 1
35
+ SECTION_IMPORT = 2
36
+ SECTION_FUNCTION = 3
37
+ SECTION_TABLE = 4
38
+ SECTION_MEMORY = 5
39
+ SECTION_GLOBAL = 6
40
+ SECTION_EXPORT = 7
41
+ SECTION_START = 8
42
+ SECTION_ELEMENT = 9
43
+ SECTION_CODE = 10
44
+ SECTION_DATA = 11
45
+ SECTION_DATA_COUNT = 12
46
+
47
+ # Block type encoding
48
+ BLOCK_TYPE_EMPTY = 0x40
49
+
50
+
51
+ class BinaryReader:
52
+ """A reader for binary data with position tracking."""
53
+
54
+ def __init__(self, data: bytes) -> None:
55
+ self.data = data
56
+ self.position = 0
57
+
58
+ def read_byte(self) -> int:
59
+ """Read a single byte."""
60
+ if self.position >= len(self.data):
61
+ raise DecodeError(f"Unexpected end of data at position {self.position}")
62
+ byte = self.data[self.position]
63
+ self.position += 1
64
+ return byte
65
+
66
+ def read_bytes(self, n: int) -> bytes:
67
+ """Read n bytes."""
68
+ if self.position + n > len(self.data):
69
+ raise DecodeError(
70
+ f"Unexpected end of data: wanted {n} bytes at position {self.position}"
71
+ )
72
+ result = self.data[self.position : self.position + n]
73
+ self.position += n
74
+ return result
75
+
76
+ def eof(self) -> bool:
77
+ """Check if at end of data."""
78
+ return self.position >= len(self.data)
79
+
80
+ def remaining(self) -> int:
81
+ """Return number of remaining bytes."""
82
+ return len(self.data) - self.position
83
+
84
+
85
+ def decode_unsigned_leb128(reader: BinaryReader, max_bits: int = 32) -> int:
86
+ """Decode an unsigned LEB128 integer."""
87
+ result = 0
88
+ shift = 0
89
+ while True:
90
+ byte = reader.read_byte()
91
+ result |= (byte & 0x7F) << shift
92
+ if (byte & 0x80) == 0:
93
+ break
94
+ shift += 7
95
+ if shift >= max_bits + 7: # Allow some slack for encoding
96
+ raise DecodeError("LEB128 integer too long")
97
+ return result
98
+
99
+
100
+ def decode_signed_leb128(reader: BinaryReader, max_bits: int = 32) -> int:
101
+ """Decode a signed LEB128 integer."""
102
+ result = 0
103
+ shift = 0
104
+ byte = 0
105
+ while True:
106
+ byte = reader.read_byte()
107
+ result |= (byte & 0x7F) << shift
108
+ shift += 7
109
+ if (byte & 0x80) == 0:
110
+ break
111
+ if shift >= max_bits + 7:
112
+ raise DecodeError("LEB128 integer too long")
113
+
114
+ # Sign extend if the sign bit (bit 6 of the last byte) is set
115
+ if byte & 0x40:
116
+ result |= -(1 << shift)
117
+
118
+ return result
119
+
120
+
121
+ def decode_name(reader: BinaryReader) -> str:
122
+ """Decode a UTF-8 name (length-prefixed byte vector)."""
123
+ length = decode_unsigned_leb128(reader)
124
+ data = reader.read_bytes(length)
125
+ try:
126
+ return data.decode("utf-8")
127
+ except UnicodeDecodeError as e:
128
+ raise DecodeError(f"Invalid UTF-8 in name: {e}") from e
129
+
130
+
131
+ def decode_valtype(reader: BinaryReader) -> str:
132
+ """Decode a value type."""
133
+ byte = reader.read_byte()
134
+ if byte not in VALTYPE_ENCODING:
135
+ raise DecodeError(f"Unknown value type: 0x{byte:02x}")
136
+ return VALTYPE_ENCODING[byte]
137
+
138
+
139
+ def decode_limits(reader: BinaryReader) -> Limits:
140
+ """Decode limits (min, optional max)."""
141
+ flags = reader.read_byte()
142
+ min_val = decode_unsigned_leb128(reader)
143
+ max_val = None
144
+ if flags & 0x01:
145
+ max_val = decode_unsigned_leb128(reader)
146
+ return Limits(min=min_val, max=max_val)
147
+
148
+
149
+ def decode_blocktype(reader: BinaryReader) -> tuple | str | int:
150
+ """Decode a block type (empty, valtype, or type index)."""
151
+ byte = reader.read_byte()
152
+ if byte == BLOCK_TYPE_EMPTY:
153
+ return () # Empty result
154
+ if byte in VALTYPE_ENCODING:
155
+ return (VALTYPE_ENCODING[byte],) # Single result type
156
+ # Otherwise it's a signed type index (for multi-value)
157
+ # Need to put the byte back and read as signed LEB128
158
+ reader.position -= 1
159
+ return decode_signed_leb128(reader)
160
+
161
+
162
+ def decode_instruction(reader: BinaryReader) -> Instruction:
163
+ """Decode a single instruction."""
164
+ opcode = reader.read_byte()
165
+
166
+ # Get opcode name
167
+ if opcode not in opcodes.OPCODE_NAMES:
168
+ raise DecodeError(f"Unknown opcode: 0x{opcode:02x}")
169
+ name = opcodes.OPCODE_NAMES[opcode]
170
+
171
+ # Handle different immediate types
172
+ if opcode in opcodes.NO_IMMEDIATE:
173
+ return Instruction(name)
174
+
175
+ if opcode in opcodes.U32_IMMEDIATE:
176
+ operand = decode_unsigned_leb128(reader)
177
+ return Instruction(name, operand)
178
+
179
+ if opcode in opcodes.I32_IMMEDIATE:
180
+ operand = decode_signed_leb128(reader, 32)
181
+ return Instruction(name, operand)
182
+
183
+ if opcode in opcodes.I64_IMMEDIATE:
184
+ operand = decode_signed_leb128(reader, 64)
185
+ return Instruction(name, operand)
186
+
187
+ if opcode in opcodes.F32_IMMEDIATE:
188
+ data = reader.read_bytes(4)
189
+ operand = struct.unpack("<f", data)[0]
190
+ return Instruction(name, operand)
191
+
192
+ if opcode in opcodes.F64_IMMEDIATE:
193
+ data = reader.read_bytes(8)
194
+ operand = struct.unpack("<d", data)[0]
195
+ return Instruction(name, operand)
196
+
197
+ if opcode in opcodes.MEMORY_IMMEDIATE:
198
+ align = decode_unsigned_leb128(reader)
199
+ offset = decode_unsigned_leb128(reader)
200
+ return Instruction(name, (align, offset))
201
+
202
+ if opcode in opcodes.BLOCK_TYPE:
203
+ blocktype = decode_blocktype(reader)
204
+ return Instruction(name, blocktype)
205
+
206
+ if opcode == opcodes.BR_TABLE:
207
+ # Vector of labels + default label
208
+ count = decode_unsigned_leb128(reader)
209
+ labels = [decode_unsigned_leb128(reader) for _ in range(count)]
210
+ default = decode_unsigned_leb128(reader)
211
+ return Instruction(name, (labels, default))
212
+
213
+ if opcode == opcodes.CALL_INDIRECT:
214
+ type_idx = decode_unsigned_leb128(reader)
215
+ table_idx = decode_unsigned_leb128(reader)
216
+ return Instruction(name, (type_idx, table_idx))
217
+
218
+ if opcode == opcodes.MEMORY_SIZE or opcode == opcodes.MEMORY_GROW:
219
+ # Memory index (always 0 in MVP)
220
+ _ = reader.read_byte() # Reserved byte, must be 0
221
+ return Instruction(name)
222
+
223
+ if opcode == opcodes.SELECT_T:
224
+ # Typed select with result types
225
+ count = decode_unsigned_leb128(reader)
226
+ types = tuple(decode_valtype(reader) for _ in range(count))
227
+ return Instruction(name, types)
228
+
229
+ if opcode == opcodes.REF_NULL:
230
+ reftype = decode_valtype(reader)
231
+ return Instruction(name, reftype)
232
+
233
+ raise DecodeError(f"Unhandled opcode: 0x{opcode:02x} ({name})")
234
+
235
+
236
+ def decode_expr(reader: BinaryReader) -> list[Instruction]:
237
+ """Decode an expression (instruction sequence ending with END)."""
238
+ instructions = []
239
+ depth = 0
240
+ while True:
241
+ instr = decode_instruction(reader)
242
+ instructions.append(instr)
243
+
244
+ # Track nesting for block/loop/if
245
+ if instr.opcode in ("block", "loop", "if"):
246
+ depth += 1
247
+ elif instr.opcode == "end":
248
+ if depth == 0:
249
+ break
250
+ depth -= 1
251
+
252
+ return instructions
253
+
254
+
255
+ def decode_func_type(reader: BinaryReader) -> FuncType:
256
+ """Decode a function type."""
257
+ marker = reader.read_byte()
258
+ if marker != 0x60:
259
+ raise DecodeError(f"Expected function type marker 0x60, got 0x{marker:02x}")
260
+
261
+ # Parameters
262
+ param_count = decode_unsigned_leb128(reader)
263
+ params = tuple(decode_valtype(reader) for _ in range(param_count))
264
+
265
+ # Results
266
+ result_count = decode_unsigned_leb128(reader)
267
+ results = tuple(decode_valtype(reader) for _ in range(result_count))
268
+
269
+ return FuncType(params, results)
270
+
271
+
272
+ def decode_type_section(reader: BinaryReader, module: Module) -> None:
273
+ """Decode the type section."""
274
+ count = decode_unsigned_leb128(reader)
275
+ for _ in range(count):
276
+ module.types.append(decode_func_type(reader))
277
+
278
+
279
+ def decode_import_section(reader: BinaryReader, module: Module) -> None:
280
+ """Decode the import section."""
281
+ count = decode_unsigned_leb128(reader)
282
+ for _ in range(count):
283
+ mod_name = decode_name(reader)
284
+ name = decode_name(reader)
285
+ kind = reader.read_byte()
286
+
287
+ if kind == 0x00: # func
288
+ type_idx = decode_unsigned_leb128(reader)
289
+ module.imports.append(Import(mod_name, name, "func", type_idx))
290
+ elif kind == 0x01: # table
291
+ elem_type = decode_valtype(reader)
292
+ limits = decode_limits(reader)
293
+ module.imports.append(Import(mod_name, name, "table", (elem_type, limits)))
294
+ elif kind == 0x02: # memory
295
+ limits = decode_limits(reader)
296
+ module.imports.append(Import(mod_name, name, "memory", limits))
297
+ elif kind == 0x03: # global
298
+ valtype = decode_valtype(reader)
299
+ mutable = reader.read_byte() != 0
300
+ module.imports.append(
301
+ Import(mod_name, name, "global", GlobalType(valtype, mutable))
302
+ )
303
+ else:
304
+ raise DecodeError(f"Unknown import kind: {kind}")
305
+
306
+
307
+ def decode_function_section(reader: BinaryReader, module: Module) -> None:
308
+ """Decode the function section (just type indices)."""
309
+ count = decode_unsigned_leb128(reader)
310
+ # We store the type indices temporarily; bodies come from code section
311
+ module._func_type_indices = [decode_unsigned_leb128(reader) for _ in range(count)]
312
+
313
+
314
+ def decode_table_section(reader: BinaryReader, module: Module) -> None:
315
+ """Decode the table section."""
316
+ count = decode_unsigned_leb128(reader)
317
+ for _ in range(count):
318
+ elem_type = decode_valtype(reader)
319
+ limits = decode_limits(reader)
320
+ module.tables.append(Table(elem_type, limits))
321
+
322
+
323
+ def decode_memory_section(reader: BinaryReader, module: Module) -> None:
324
+ """Decode the memory section."""
325
+ count = decode_unsigned_leb128(reader)
326
+ for _ in range(count):
327
+ limits = decode_limits(reader)
328
+ module.mems.append(Memory(limits))
329
+
330
+
331
+ def decode_global_section(reader: BinaryReader, module: Module) -> None:
332
+ """Decode the global section."""
333
+ count = decode_unsigned_leb128(reader)
334
+ for _ in range(count):
335
+ valtype = decode_valtype(reader)
336
+ mutable = reader.read_byte() != 0
337
+ init = decode_expr(reader)
338
+ module.globals.append(Global(GlobalType(valtype, mutable), init))
339
+
340
+
341
+ def decode_export_section(reader: BinaryReader, module: Module) -> None:
342
+ """Decode the export section."""
343
+ count = decode_unsigned_leb128(reader)
344
+ for _ in range(count):
345
+ name = decode_name(reader)
346
+ kind_byte = reader.read_byte()
347
+ if kind_byte not in EXPORT_KIND_ENCODING:
348
+ raise DecodeError(f"Unknown export kind: {kind_byte}")
349
+ kind = EXPORT_KIND_ENCODING[kind_byte]
350
+ index = decode_unsigned_leb128(reader)
351
+ module.exports.append(Export(name, kind, index))
352
+
353
+
354
+ def decode_start_section(reader: BinaryReader, module: Module) -> None:
355
+ """Decode the start section."""
356
+ module.start = decode_unsigned_leb128(reader)
357
+
358
+
359
+ def decode_element_section(reader: BinaryReader, module: Module) -> None:
360
+ """Decode the element section."""
361
+ count = decode_unsigned_leb128(reader)
362
+ for _ in range(count):
363
+ # Simple active element segment (MVP)
364
+ flags = decode_unsigned_leb128(reader)
365
+
366
+ if flags == 0:
367
+ # Active segment for table 0
368
+ offset = decode_expr(reader)
369
+ func_count = decode_unsigned_leb128(reader)
370
+ func_indices = [decode_unsigned_leb128(reader) for _ in range(func_count)]
371
+ module.elem.append(Element(0, offset, func_indices))
372
+ else:
373
+ # More complex element segment types (post-MVP)
374
+ raise DecodeError(f"Unsupported element segment flags: {flags}")
375
+
376
+
377
+ def decode_code_section(reader: BinaryReader, module: Module) -> None:
378
+ """Decode the code section."""
379
+ count = decode_unsigned_leb128(reader)
380
+
381
+ if not hasattr(module, "_func_type_indices"):
382
+ raise DecodeError("Code section without function section")
383
+
384
+ if count != len(module._func_type_indices):
385
+ raise DecodeError(
386
+ f"Code section count ({count}) != function section count "
387
+ f"({len(module._func_type_indices)})"
388
+ )
389
+
390
+ for i in range(count):
391
+ body_size = decode_unsigned_leb128(reader)
392
+ body_start = reader.position
393
+
394
+ # Local declarations
395
+ local_count = decode_unsigned_leb128(reader)
396
+ locals_list: list[str] = []
397
+ for _ in range(local_count):
398
+ n = decode_unsigned_leb128(reader)
399
+ valtype = decode_valtype(reader)
400
+ locals_list.extend([valtype] * n)
401
+
402
+ # Instructions
403
+ body = decode_expr(reader)
404
+
405
+ # Verify we consumed exactly body_size bytes
406
+ consumed = reader.position - body_start
407
+ if consumed != body_size:
408
+ raise DecodeError(
409
+ f"Function body size mismatch: expected {body_size}, got {consumed}"
410
+ )
411
+
412
+ module.funcs.append(
413
+ Function(
414
+ type_idx=module._func_type_indices[i],
415
+ locals=tuple(locals_list),
416
+ body=body,
417
+ )
418
+ )
419
+
420
+
421
+ def decode_data_section(reader: BinaryReader, module: Module) -> None:
422
+ """Decode the data section."""
423
+ count = decode_unsigned_leb128(reader)
424
+ for _ in range(count):
425
+ flags = decode_unsigned_leb128(reader)
426
+
427
+ if flags == 0:
428
+ # Active segment for memory 0
429
+ offset = decode_expr(reader)
430
+ length = decode_unsigned_leb128(reader)
431
+ init = reader.read_bytes(length)
432
+ module.data.append(Data(0, offset, init))
433
+ elif flags == 1:
434
+ # Passive segment
435
+ length = decode_unsigned_leb128(reader)
436
+ init = reader.read_bytes(length)
437
+ module.data.append(Data(-1, [], init)) # -1 indicates passive
438
+ elif flags == 2:
439
+ # Active segment with explicit memory index
440
+ mem_idx = decode_unsigned_leb128(reader)
441
+ offset = decode_expr(reader)
442
+ length = decode_unsigned_leb128(reader)
443
+ init = reader.read_bytes(length)
444
+ module.data.append(Data(mem_idx, offset, init))
445
+ else:
446
+ raise DecodeError(f"Unsupported data segment flags: {flags}")
447
+
448
+
449
+ def decode_section(reader: BinaryReader, module: Module) -> None:
450
+ """Decode a single section."""
451
+ section_id = reader.read_byte()
452
+ section_size = decode_unsigned_leb128(reader)
453
+ section_end = reader.position + section_size
454
+
455
+ # Create a sub-reader for the section content
456
+ section_data = reader.read_bytes(section_size)
457
+ section_reader = BinaryReader(section_data)
458
+
459
+ if section_id == SECTION_CUSTOM:
460
+ # Skip custom sections for now (could parse names section later)
461
+ pass
462
+ elif section_id == SECTION_TYPE:
463
+ decode_type_section(section_reader, module)
464
+ elif section_id == SECTION_IMPORT:
465
+ decode_import_section(section_reader, module)
466
+ elif section_id == SECTION_FUNCTION:
467
+ decode_function_section(section_reader, module)
468
+ elif section_id == SECTION_TABLE:
469
+ decode_table_section(section_reader, module)
470
+ elif section_id == SECTION_MEMORY:
471
+ decode_memory_section(section_reader, module)
472
+ elif section_id == SECTION_GLOBAL:
473
+ decode_global_section(section_reader, module)
474
+ elif section_id == SECTION_EXPORT:
475
+ decode_export_section(section_reader, module)
476
+ elif section_id == SECTION_START:
477
+ decode_start_section(section_reader, module)
478
+ elif section_id == SECTION_ELEMENT:
479
+ decode_element_section(section_reader, module)
480
+ elif section_id == SECTION_CODE:
481
+ decode_code_section(section_reader, module)
482
+ elif section_id == SECTION_DATA:
483
+ decode_data_section(section_reader, module)
484
+ elif section_id == SECTION_DATA_COUNT:
485
+ # Data count section (for bulk memory)
486
+ _ = decode_unsigned_leb128(section_reader) # Just read and ignore for now
487
+ else:
488
+ raise DecodeError(f"Unknown section id: {section_id}")
489
+
490
+
491
+ def decode_module(source: bytes | BinaryIO | Path) -> Module:
492
+ """Decode a WebAssembly module from binary format.
493
+
494
+ Args:
495
+ source: WASM bytes, file-like object, or path to .wasm file
496
+
497
+ Returns:
498
+ Decoded Module object
499
+
500
+ Raises:
501
+ DecodeError: If the binary format is invalid
502
+ """
503
+ # Handle different source types
504
+ if isinstance(source, Path):
505
+ data = source.read_bytes()
506
+ elif isinstance(source, bytes):
507
+ data = source
508
+ else:
509
+ # Assume file-like object
510
+ data = source.read()
511
+
512
+ reader = BinaryReader(data)
513
+
514
+ # Check magic number
515
+ magic = reader.read_bytes(4)
516
+ if magic != WASM_MAGIC:
517
+ raise DecodeError(
518
+ f"Invalid WASM magic number: expected {WASM_MAGIC!r}, got {magic!r}"
519
+ )
520
+
521
+ # Check version
522
+ version_bytes = reader.read_bytes(4)
523
+ version = int.from_bytes(version_bytes, "little")
524
+ if version != WASM_VERSION:
525
+ raise DecodeError(f"Unsupported WASM version: {version}")
526
+
527
+ # Create module and decode sections
528
+ module = Module()
529
+
530
+ while not reader.eof():
531
+ decode_section(reader, module)
532
+
533
+ return module
pwasm/errors.py ADDED
@@ -0,0 +1,31 @@
1
+ """Exception classes for the WebAssembly runtime."""
2
+
3
+
4
+ class WasmError(Exception):
5
+ """Base class for all WebAssembly runtime errors."""
6
+
7
+ pass
8
+
9
+
10
+ class DecodeError(WasmError):
11
+ """Error during binary format decoding."""
12
+
13
+ pass
14
+
15
+
16
+ class ValidationError(WasmError):
17
+ """Error during module validation."""
18
+
19
+ pass
20
+
21
+
22
+ class TrapError(WasmError):
23
+ """Runtime trap (division by zero, unreachable, etc.)."""
24
+
25
+ pass
26
+
27
+
28
+ class LinkError(WasmError):
29
+ """Error during module instantiation/linking."""
30
+
31
+ pass