zipwire 0.3.0__tar.gz → 0.4.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {zipwire-0.3.0 → zipwire-0.4.0}/.github/dependabot.yml +2 -0
  2. {zipwire-0.3.0 → zipwire-0.4.0}/.github/workflows/ci.yml +8 -8
  3. {zipwire-0.3.0 → zipwire-0.4.0}/.github/workflows/codeql.yml +3 -3
  4. {zipwire-0.3.0 → zipwire-0.4.0}/.github/workflows/release.yml +4 -4
  5. {zipwire-0.3.0 → zipwire-0.4.0}/.github/workflows/scorecard.yml +3 -3
  6. {zipwire-0.3.0 → zipwire-0.4.0}/PKG-INFO +49 -4
  7. {zipwire-0.3.0 → zipwire-0.4.0}/README.md +47 -2
  8. {zipwire-0.3.0 → zipwire-0.4.0}/docs/api.rst +8 -0
  9. {zipwire-0.3.0 → zipwire-0.4.0}/docs/backends.rst +61 -0
  10. {zipwire-0.3.0 → zipwire-0.4.0}/docs/quickstart.rst +40 -0
  11. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/__init__.py +17 -0
  12. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_version.py +2 -2
  13. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/backends/__init__.py +4 -0
  14. zipwire-0.4.0/src/zipwire/backends/_file.py +206 -0
  15. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_backends.py +149 -2
  16. {zipwire-0.3.0 → zipwire-0.4.0}/uv.lock +106 -106
  17. {zipwire-0.3.0 → zipwire-0.4.0}/.gitignore +0 -0
  18. {zipwire-0.3.0 → zipwire-0.4.0}/.readthedocs.yaml +0 -0
  19. {zipwire-0.3.0 → zipwire-0.4.0}/AGENTS.md +0 -0
  20. {zipwire-0.3.0 → zipwire-0.4.0}/CLAUDE.md +0 -0
  21. {zipwire-0.3.0 → zipwire-0.4.0}/LICENSE +0 -0
  22. {zipwire-0.3.0 → zipwire-0.4.0}/SECURITY.md +0 -0
  23. {zipwire-0.3.0 → zipwire-0.4.0}/docs/Makefile +0 -0
  24. {zipwire-0.3.0 → zipwire-0.4.0}/docs/conf.py +0 -0
  25. {zipwire-0.3.0 → zipwire-0.4.0}/docs/index.rst +0 -0
  26. {zipwire-0.3.0 → zipwire-0.4.0}/docs/requirements.txt +0 -0
  27. {zipwire-0.3.0 → zipwire-0.4.0}/docs/security.rst +0 -0
  28. {zipwire-0.3.0 → zipwire-0.4.0}/pyproject.toml +0 -0
  29. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/__main__.py +0 -0
  30. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_async.py +0 -0
  31. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_base.py +0 -0
  32. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_constants.py +0 -0
  33. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_decompress.py +0 -0
  34. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_errors.py +0 -0
  35. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_parser.py +0 -0
  36. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_sync.py +0 -0
  37. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_types.py +0 -0
  38. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_wheel.py +0 -0
  39. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/_zipinfo.py +0 -0
  40. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/backends/_aiohttp.py +0 -0
  41. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/backends/_httpx2.py +0 -0
  42. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/backends/_requests.py +0 -0
  43. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/backends/_urllib3.py +0 -0
  44. {zipwire-0.3.0 → zipwire-0.4.0}/src/zipwire/py.typed +0 -0
  45. {zipwire-0.3.0 → zipwire-0.4.0}/tests/__init__.py +0 -0
  46. {zipwire-0.3.0 → zipwire-0.4.0}/tests/conftest.py +0 -0
  47. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_async.py +0 -0
  48. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_cli.py +0 -0
  49. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_decompress.py +0 -0
  50. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_integration.py +0 -0
  51. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_parser.py +0 -0
  52. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_sync.py +0 -0
  53. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_wheel.py +0 -0
  54. {zipwire-0.3.0 → zipwire-0.4.0}/tests/test_zipinfo.py +0 -0
  55. {zipwire-0.3.0 → zipwire-0.4.0}/tox.ini +0 -0
@@ -8,3 +8,5 @@ updates:
8
8
  codeql:
9
9
  patterns:
10
10
  - "github/codeql-action/*"
11
+ cooldown:
12
+ default-days: 7
@@ -18,10 +18,10 @@ jobs:
18
18
  permissions:
19
19
  contents: read
20
20
  steps:
21
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
21
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
22
22
  with:
23
23
  persist-credentials: false
24
- - uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2
24
+ - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
25
25
  with:
26
26
  enable-cache: true
27
27
  - run: uvx ruff check src tests
@@ -37,10 +37,10 @@ jobs:
37
37
  python-version: ["3.11", "3.12", "3.13", "3.14", "3.15"]
38
38
  continue-on-error: ${{ matrix.python-version == '3.15' }}
39
39
  steps:
40
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
40
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
41
41
  with:
42
42
  persist-credentials: false
43
- - uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2
43
+ - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
44
44
  with:
45
45
  enable-cache: true
46
46
  - run: uv tool install --python ${{ matrix.python-version }} tox --with tox-uv --with tox-gh
@@ -58,10 +58,10 @@ jobs:
58
58
  permissions:
59
59
  contents: read
60
60
  steps:
61
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
61
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
62
62
  with:
63
63
  persist-credentials: false
64
- - uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2
64
+ - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
65
65
  with:
66
66
  enable-cache: true
67
67
  - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1
@@ -78,10 +78,10 @@ jobs:
78
78
  permissions:
79
79
  contents: read
80
80
  steps:
81
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
81
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
82
82
  with:
83
83
  persist-credentials: false
84
- - uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2
84
+ - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
85
85
  with:
86
86
  enable-cache: true
87
87
  - run: uv sync --group docs
@@ -16,10 +16,10 @@ jobs:
16
16
  permissions:
17
17
  security-events: write
18
18
  steps:
19
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
19
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
20
20
  with:
21
21
  persist-credentials: false
22
- - uses: github/codeql-action/init@7188fc363630916deb702c7fdcf4e481b751f97a # v4.37.1
22
+ - uses: github/codeql-action/init@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7
23
23
  with:
24
24
  languages: python
25
- - uses: github/codeql-action/analyze@7188fc363630916deb702c7fdcf4e481b751f97a # v4.37.1
25
+ - uses: github/codeql-action/analyze@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7
@@ -12,12 +12,12 @@ jobs:
12
12
  permissions:
13
13
  contents: read
14
14
  steps:
15
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
15
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
16
16
  with:
17
17
  persist-credentials: false
18
- - uses: astral-sh/setup-uv@11f9893b081a58869d3b5fccaea48c9e9e46f990 # v8.3.2
18
+ - uses: astral-sh/setup-uv@20cfd1bf945f4377ade1205e4dbc17946fc9a30d # v10.0.1
19
19
  with:
20
- enable-cache: true
20
+ enable-cache: false
21
21
  - run: uv build
22
22
  - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
23
23
  with:
@@ -36,6 +36,6 @@ jobs:
36
36
  with:
37
37
  name: dist
38
38
  path: dist/
39
- - uses: pypa/gh-action-pypi-publish@cef221092ed1bacb1cc03d23a2d87d1d172e277b # v1.14.0
39
+ - uses: pypa/gh-action-pypi-publish@dc37677b2e1c63e2034f94d8a5b11f265b73ba33 # v1.14.2
40
40
  with:
41
41
  attestations: true
@@ -15,10 +15,10 @@ jobs:
15
15
  security-events: write
16
16
  id-token: write
17
17
  steps:
18
- - uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
18
+ - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1
19
19
  with:
20
20
  persist-credentials: false
21
- - uses: ossf/scorecard-action@4eaacf0543bb3f2c246792bd56e8cdeffafb205a # v2.4.3
21
+ - uses: ossf/scorecard-action@2d1146689b8cda280b9bc96326124645441f03bc # v2.4.4
22
22
  with:
23
23
  results_file: results.sarif
24
24
  results_format: sarif
@@ -28,6 +28,6 @@ jobs:
28
28
  name: scorecard-results
29
29
  path: results.sarif
30
30
  retention-days: 5
31
- - uses: github/codeql-action/upload-sarif@7188fc363630916deb702c7fdcf4e481b751f97a # v4.37.1
31
+ - uses: github/codeql-action/upload-sarif@ff2f1c621b7f889edc0d3c761ac2e6a3f8cdb0dd # v4.37.7
32
32
  with:
33
33
  sarif_file: results.sarif
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: zipwire
3
- Version: 0.3.0
3
+ Version: 0.4.0
4
4
  Summary: Read and extract files from remote ZIP archives over HTTP range requests
5
5
  Project-URL: Homepage, https://github.com/tiran/zipwire
6
6
  Project-URL: Documentation, https://zipwire.readthedocs.io/
@@ -60,8 +60,13 @@ CDNs, object stores, and static file servers do.
60
60
  memory usage low even for large entries.
61
61
  - **Sync and async** - `SyncRemoteZip` for synchronous code,
62
62
  `AsyncRemoteZip` with `await`/`async with` for asyncio.
63
+ - **Wheel metadata** - `SyncRemoteWheel` / `AsyncRemoteWheel` read a Python
64
+ wheel's `.dist-info` (METADATA, WHEEL, RECORD) straight from PyPI in a single
65
+ adaptive tail request, without downloading the wheel.
63
66
  - **ZIP64** - supports archives and entries larger than 4 GiB.
64
67
  - **Pluggable backends** - bring your own HTTP library (see below).
68
+ - **Local files** - `FileReader` / `AsyncFileReader` open an archive on disk
69
+ through the exact same API, no HTTP server required.
65
70
 
66
71
  ## Installation and backends
67
72
 
@@ -81,11 +86,36 @@ pip install zipwire[httpx2]
81
86
  | requests | `RequestsReader` | sync | 1.1 | `requests` |
82
87
  | aiohttp | `AiohttpReader` | async | 1.1 | `aiohttp` |
83
88
 
84
- Every backend accepts an optional pre-configured client or session so you can
85
- share connection pools, authentication, and retry configuration.
89
+ Every HTTP backend accepts an optional pre-configured client or session so you
90
+ can share connection pools, authentication, and retry configuration.
91
+
92
+ For archives that already live on the local filesystem, `FileReader` and
93
+ `AsyncFileReader` (in `zipwire.backends`, no extra dependency) satisfy the same
94
+ reader protocols - see the local-file example below.
86
95
 
87
96
  ## Examples
88
97
 
98
+ ### Read Python wheel metadata without downloading the wheel
99
+
100
+ A common use case: fetch a wheel's `METADATA`, `WHEEL`, or `RECORD` from PyPI
101
+ without downloading the (often huge) wheel itself. `SyncRemoteWheel` and
102
+ `AsyncRemoteWheel` parse the wheel URL to locate the `.dist-info` directory and
103
+ fetch an adaptive tail, so metadata entries are served from memory without
104
+ extra HTTP requests:
105
+
106
+ ```python
107
+ from zipwire import SyncRemoteWheel
108
+ from zipwire.backends import Urllib3Reader
109
+
110
+ url = "https://files.pythonhosted.org/.../requests-2.32.3-py3-none-any.whl"
111
+ with SyncRemoteWheel(Urllib3Reader(url)) as whl:
112
+ print(whl.read(whl.metadata_name).decode())
113
+ ```
114
+
115
+ This optimization relies on the [recommended wheel layout](https://packaging.python.org/en/latest/specifications/binary-distribution-format/#recommended-archiver-features)
116
+ of placing `.dist-info` at the end of the archive; wheels built otherwise still
117
+ work, falling back to a normal range request per entry.
118
+
89
119
  ### Sync - list files and read one
90
120
 
91
121
  ```python
@@ -130,6 +160,21 @@ async def main():
130
160
  asyncio.run(main())
131
161
  ```
132
162
 
163
+ ### Local archive - same API, no HTTP
164
+
165
+ `FileReader` opens an archive from disk through the same interface, accepting a
166
+ path or a `file://` URI. `AsyncFileReader` is the async counterpart. This lets
167
+ code that already uses zipwire handle local and remote archives the same way,
168
+ without special-casing either.
169
+
170
+ ```python
171
+ from zipwire import SyncRemoteZip
172
+ from zipwire.backends import FileReader
173
+
174
+ with SyncRemoteZip(FileReader("/path/to/archive.zip")) as rz:
175
+ data = rz.read("path/to/file.txt")
176
+ ```
177
+
133
178
  ## License
134
179
 
135
180
  Apache-2.0
@@ -25,8 +25,13 @@ CDNs, object stores, and static file servers do.
25
25
  memory usage low even for large entries.
26
26
  - **Sync and async** - `SyncRemoteZip` for synchronous code,
27
27
  `AsyncRemoteZip` with `await`/`async with` for asyncio.
28
+ - **Wheel metadata** - `SyncRemoteWheel` / `AsyncRemoteWheel` read a Python
29
+ wheel's `.dist-info` (METADATA, WHEEL, RECORD) straight from PyPI in a single
30
+ adaptive tail request, without downloading the wheel.
28
31
  - **ZIP64** - supports archives and entries larger than 4 GiB.
29
32
  - **Pluggable backends** - bring your own HTTP library (see below).
33
+ - **Local files** - `FileReader` / `AsyncFileReader` open an archive on disk
34
+ through the exact same API, no HTTP server required.
30
35
 
31
36
  ## Installation and backends
32
37
 
@@ -46,11 +51,36 @@ pip install zipwire[httpx2]
46
51
  | requests | `RequestsReader` | sync | 1.1 | `requests` |
47
52
  | aiohttp | `AiohttpReader` | async | 1.1 | `aiohttp` |
48
53
 
49
- Every backend accepts an optional pre-configured client or session so you can
50
- share connection pools, authentication, and retry configuration.
54
+ Every HTTP backend accepts an optional pre-configured client or session so you
55
+ can share connection pools, authentication, and retry configuration.
56
+
57
+ For archives that already live on the local filesystem, `FileReader` and
58
+ `AsyncFileReader` (in `zipwire.backends`, no extra dependency) satisfy the same
59
+ reader protocols - see the local-file example below.
51
60
 
52
61
  ## Examples
53
62
 
63
+ ### Read Python wheel metadata without downloading the wheel
64
+
65
+ A common use case: fetch a wheel's `METADATA`, `WHEEL`, or `RECORD` from PyPI
66
+ without downloading the (often huge) wheel itself. `SyncRemoteWheel` and
67
+ `AsyncRemoteWheel` parse the wheel URL to locate the `.dist-info` directory and
68
+ fetch an adaptive tail, so metadata entries are served from memory without
69
+ extra HTTP requests:
70
+
71
+ ```python
72
+ from zipwire import SyncRemoteWheel
73
+ from zipwire.backends import Urllib3Reader
74
+
75
+ url = "https://files.pythonhosted.org/.../requests-2.32.3-py3-none-any.whl"
76
+ with SyncRemoteWheel(Urllib3Reader(url)) as whl:
77
+ print(whl.read(whl.metadata_name).decode())
78
+ ```
79
+
80
+ This optimization relies on the [recommended wheel layout](https://packaging.python.org/en/latest/specifications/binary-distribution-format/#recommended-archiver-features)
81
+ of placing `.dist-info` at the end of the archive; wheels built otherwise still
82
+ work, falling back to a normal range request per entry.
83
+
54
84
  ### Sync - list files and read one
55
85
 
56
86
  ```python
@@ -95,6 +125,21 @@ async def main():
95
125
  asyncio.run(main())
96
126
  ```
97
127
 
128
+ ### Local archive - same API, no HTTP
129
+
130
+ `FileReader` opens an archive from disk through the same interface, accepting a
131
+ path or a `file://` URI. `AsyncFileReader` is the async counterpart. This lets
132
+ code that already uses zipwire handle local and remote archives the same way,
133
+ without special-casing either.
134
+
135
+ ```python
136
+ from zipwire import SyncRemoteZip
137
+ from zipwire.backends import FileReader
138
+
139
+ with SyncRemoteZip(FileReader("/path/to/archive.zip")) as rz:
140
+ data = rz.read("path/to/file.txt")
141
+ ```
142
+
98
143
  ## License
99
144
 
100
145
  Apache-2.0
@@ -78,6 +78,14 @@ Backends
78
78
  .. autoclass:: zipwire.backends._aiohttp.AiohttpReader
79
79
  :members:
80
80
 
81
+ .. autoclass:: zipwire.backends._file.FileReader
82
+ :members:
83
+ :special-members: __enter__, __exit__
84
+
85
+ .. autoclass:: zipwire.backends._file.AsyncFileReader
86
+ :members:
87
+ :special-members: __aenter__, __aexit__
88
+
81
89
  Exceptions
82
90
  ----------
83
91
 
@@ -43,6 +43,16 @@ Available backends
43
43
  - async
44
44
  - 1.1
45
45
  - ``aiohttp``
46
+ * - local files
47
+ - :class:`~zipwire.backends.FileReader`
48
+ - sync
49
+ - --
50
+ - *(included)*
51
+ * - local files
52
+ - :class:`~zipwire.backends.AsyncFileReader`
53
+ - async
54
+ - --
55
+ - *(included)*
46
56
 
47
57
  Choosing a backend
48
58
  ------------------
@@ -60,6 +70,57 @@ Choosing a backend
60
70
  - **Requests integration** - use :class:`~zipwire.backends.RequestsReader`
61
71
  if your project already uses ``requests`` and you want to share sessions,
62
72
  authentication, or retry configuration.
73
+ - **Local archives** - use :class:`~zipwire.backends.FileReader` or
74
+ :class:`~zipwire.backends.AsyncFileReader` to read a ZIP or wheel that
75
+ already lives on the local filesystem. No extra dependency is required.
76
+
77
+ Reading local files
78
+ -------------------
79
+
80
+ :class:`~zipwire.backends.FileReader` and
81
+ :class:`~zipwire.backends.AsyncFileReader` back the same
82
+ :class:`~zipwire.SyncRemoteZip` / :class:`~zipwire.AsyncRemoteZip` API with
83
+ local file IO, so identical code opens a local or a remote archive. Instead
84
+ of an HTTP round-trip, :meth:`~zipwire.backends.FileReader.head` synthesises
85
+ ``Content-Length`` / ``Accept-Ranges`` from ``os.fstat`` on the open handle
86
+ (the size is cached, assuming the file does not change), and range reads
87
+ seek into a lazily-opened file handle. A missing path raises
88
+ :exc:`FileNotFoundError`, mirroring how a network 404 surfaces as
89
+ :exc:`OSError`.
90
+
91
+ Both accept either a path (``str`` / :class:`os.PathLike`) or a ``file://``
92
+ URI via :meth:`~zipwire.backends.FileReader.from_uri` (percent-decoded, with
93
+ an empty or ``localhost`` host).
94
+
95
+ .. code-block:: python
96
+
97
+ from zipwire import SyncRemoteZip
98
+ from zipwire.backends import FileReader
99
+
100
+ with SyncRemoteZip(FileReader("/path/to/archive.zip")) as rz:
101
+ data = rz.read("file.txt")
102
+
103
+ reader = FileReader.from_uri("file:///path/to/archive.zip")
104
+
105
+ Regular files do not support true non-blocking IO, so
106
+ :class:`~zipwire.backends.AsyncFileReader` offloads every blocking call to a
107
+ worker thread via :func:`asyncio.to_thread` -- the same approach libraries
108
+ such as ``aiofiles`` use internally, without the extra dependency.
109
+
110
+ .. code-block:: python
111
+
112
+ import asyncio
113
+
114
+ from zipwire import AsyncRemoteZip
115
+ from zipwire.backends import AsyncFileReader
116
+
117
+
118
+ async def main():
119
+ async with AsyncRemoteZip(AsyncFileReader("/path/to/archive.zip")) as rz:
120
+ print(await rz.read("file.txt"))
121
+
122
+
123
+ asyncio.run(main())
63
124
 
64
125
  Passing an existing client
65
126
  --------------------------
@@ -46,6 +46,25 @@ entries:
46
46
  with open("output.bin", "wb") as f:
47
47
  rz.read_into("big-file.bin", f)
48
48
 
49
+ Read a local archive
50
+ ^^^^^^^^^^^^^^^^^^^^
51
+
52
+ :class:`~zipwire.backends.FileReader` opens an archive on the local
53
+ filesystem through the same interface, so the same code works for local and
54
+ remote archives. It accepts a path or a ``file://`` URI and needs no extra
55
+ dependency:
56
+
57
+ .. code-block:: python
58
+
59
+ from zipwire import SyncRemoteZip
60
+ from zipwire.backends import FileReader
61
+
62
+ with SyncRemoteZip(FileReader("/path/to/archive.zip")) as rz:
63
+ data = rz.read("path/to/file.txt")
64
+
65
+ # Or from a file:// URI
66
+ reader = FileReader.from_uri("file:///path/to/archive.zip")
67
+
49
68
  Asynchronous usage
50
69
  ------------------
51
70
 
@@ -89,6 +108,27 @@ Async with httpx2
89
108
 
90
109
  asyncio.run(main())
91
110
 
111
+ Async local archive
112
+ ^^^^^^^^^^^^^^^^^^^
113
+
114
+ :class:`~zipwire.backends.AsyncFileReader` reads a local archive and offloads
115
+ blocking file IO to a worker thread so it never stalls the event loop:
116
+
117
+ .. code-block:: python
118
+
119
+ import asyncio
120
+
121
+ from zipwire import AsyncRemoteZip
122
+ from zipwire.backends import AsyncFileReader
123
+
124
+
125
+ async def main():
126
+ async with AsyncRemoteZip(AsyncFileReader("/path/to/archive.zip")) as rz:
127
+ print(await rz.read("path/to/file.txt"))
128
+
129
+
130
+ asyncio.run(main())
131
+
92
132
  Async streaming to disk
93
133
  ^^^^^^^^^^^^^^^^^^^^^^^^
94
134
 
@@ -59,6 +59,23 @@ Asynchronous:
59
59
  - ``Httpx2AsyncReader`` -- uses *httpx2*, supports HTTP/2
60
60
  (``pip install zipwire[httpx2]``)
61
61
 
62
+ Local files:
63
+ - ``FileReader`` / ``AsyncFileReader`` -- read an archive from the local
64
+ filesystem through the same API (no extra dependency). Useful for
65
+ testing or for handling ``file://`` URIs and local paths uniformly
66
+ with HTTP URLs.
67
+
68
+ ::
69
+
70
+ from zipwire import SyncRemoteZip
71
+ from zipwire.backends import FileReader
72
+
73
+ with SyncRemoteZip(FileReader("/path/to/archive.zip")) as rz:
74
+ data = rz.read("path/inside/archive.txt")
75
+
76
+ # From a file:// URI
77
+ reader = FileReader.from_uri("file:///path/to/archive.zip")
78
+
62
79
  Wheel subclasses
63
80
  ----------------
64
81
 
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '0.3.0'
22
- __version_tuple__ = version_tuple = (0, 3, 0)
21
+ __version__ = version = '0.4.0'
22
+ __version_tuple__ = version_tuple = (0, 4, 0)
23
23
 
24
24
  __commit_id__ = commit_id = None
@@ -14,6 +14,8 @@ from typing import TYPE_CHECKING
14
14
 
15
15
  if TYPE_CHECKING:
16
16
  from zipwire.backends._aiohttp import AiohttpReader as AiohttpReader
17
+ from zipwire.backends._file import AsyncFileReader as AsyncFileReader
18
+ from zipwire.backends._file import FileReader as FileReader
17
19
  from zipwire.backends._httpx2 import Httpx2AsyncReader as Httpx2AsyncReader
18
20
  from zipwire.backends._httpx2 import Httpx2SyncReader as Httpx2SyncReader
19
21
  from zipwire.backends._requests import RequestsReader as RequestsReader
@@ -25,6 +27,8 @@ _LAZY_IMPORTS: dict[str, tuple[str, str]] = {
25
27
  "AiohttpReader": ("zipwire.backends._aiohttp", "AiohttpReader"),
26
28
  "Urllib3Reader": ("zipwire.backends._urllib3", "Urllib3Reader"),
27
29
  "RequestsReader": ("zipwire.backends._requests", "RequestsReader"),
30
+ "FileReader": ("zipwire.backends._file", "FileReader"),
31
+ "AsyncFileReader": ("zipwire.backends._file", "AsyncFileReader"),
28
32
  }
29
33
 
30
34
  __all__ = list(_LAZY_IMPORTS)
@@ -0,0 +1,206 @@
1
+ """Local-file readers that satisfy the SyncReader / AsyncReader protocols.
2
+
3
+ ``FileReader`` and ``AsyncFileReader`` back the same
4
+ :class:`~zipwire.SyncRemoteZip` / :class:`~zipwire.AsyncRemoteZip` API with
5
+ local file IO, so an archive on disk opens exactly like a remote one. They
6
+ are useful for testing, for treating ``file://`` URIs uniformly with HTTP
7
+ URLs, and for reading a wheel or ZIP that already lives on the local
8
+ filesystem without a running HTTP server.
9
+
10
+ Unlike the HTTP backends there is no network round-trip: :meth:`head`
11
+ synthesises headers from ``os.fstat`` on the open handle (the size is
12
+ cached, assuming the file does not change), and range reads seek into a
13
+ lazily-opened file handle.
14
+
15
+ Regular files do not support true non-blocking IO (``epoll``/``select`` and
16
+ ``O_NONBLOCK`` do not apply to them), so ``AsyncFileReader`` offloads every
17
+ blocking call to a worker thread via :func:`asyncio.to_thread` -- the same
18
+ strategy libraries such as ``aiofiles`` use internally, without the extra
19
+ dependency.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ import asyncio
25
+ import os
26
+ import typing
27
+ from urllib.parse import urlparse
28
+ from urllib.request import url2pathname
29
+
30
+ from zipwire._constants import STREAM_CHUNK_SIZE
31
+
32
+ if typing.TYPE_CHECKING:
33
+ from collections.abc import AsyncIterator, Iterator
34
+
35
+ from zipwire._types import Headers
36
+
37
+
38
+ def _path_from_file_uri(uri: str) -> str:
39
+ """Convert a ``file://`` URI to a local filesystem path.
40
+
41
+ Accepts an empty or ``localhost`` host and percent-decodes the path
42
+ via :func:`urllib.request.url2pathname`.
43
+
44
+ Raises:
45
+ ValueError: If *uri* is not a ``file://`` URI or names a remote host.
46
+ """
47
+ parsed = urlparse(uri)
48
+ if parsed.scheme != "file":
49
+ raise ValueError(f"Not a file:// URI: {uri!r}")
50
+ if parsed.netloc not in ("", "localhost"):
51
+ raise ValueError(f"Non-local file URI host {parsed.netloc!r} in {uri!r}")
52
+ return url2pathname(parsed.path)
53
+
54
+
55
+ def _stat_headers(size: int) -> dict[str, str]:
56
+ """Synthesise HTTP-style headers describing a local file of *size* bytes."""
57
+ return {"content-length": str(size), "accept-ranges": "bytes"}
58
+
59
+
60
+ class FileReader:
61
+ """SyncReader implementation backed by a local file.
62
+
63
+ Opens the file lazily on the first read and keeps a single handle
64
+ open until :meth:`close` (or context-manager exit). A missing path
65
+ surfaces as :exc:`FileNotFoundError`, mirroring how an HTTP 404
66
+ surfaces as :exc:`OSError` in the network backends.
67
+ """
68
+
69
+ def __init__(self, path: str | os.PathLike[str], *, url: str | None = None) -> None:
70
+ self._path = os.fspath(path)
71
+ self._url = url if url is not None else self._path
72
+ self._handle: typing.BinaryIO | None = None
73
+ self._size: int | None = None
74
+
75
+ @classmethod
76
+ def from_uri(cls, uri: str) -> FileReader:
77
+ """Create a :class:`FileReader` from a ``file://`` URI."""
78
+ return cls(_path_from_file_uri(uri), url=uri)
79
+
80
+ @property
81
+ def url(self) -> str:
82
+ """The original path or ``file://`` URI for this reader."""
83
+ return self._url
84
+
85
+ def _open(self) -> typing.BinaryIO:
86
+ if self._handle is None:
87
+ self._handle = open(self._path, "rb") # noqa: SIM115
88
+ return self._handle
89
+
90
+ def _file_size(self) -> int:
91
+ # Cached from fstat on the open handle. We assume the file does
92
+ # not change for the lifetime of the reader.
93
+ if self._size is None:
94
+ self._size = os.fstat(self._open().fileno()).st_size
95
+ return self._size
96
+
97
+ def head(self) -> Headers:
98
+ return _stat_headers(self._file_size())
99
+
100
+ def read_range(self, offset: int, length: int) -> tuple[bytes, Headers]:
101
+ fh = self._open()
102
+ fh.seek(offset)
103
+ data = fh.read(length)
104
+ return data, _stat_headers(self._file_size())
105
+
106
+ def stream_range(self, offset: int, length: int) -> Iterator[bytes]:
107
+ fh = self._open()
108
+ fh.seek(offset)
109
+ remaining = length
110
+ while remaining > 0:
111
+ chunk = fh.read(min(STREAM_CHUNK_SIZE, remaining))
112
+ if not chunk:
113
+ break
114
+ remaining -= len(chunk)
115
+ yield chunk
116
+
117
+ def close(self) -> None:
118
+ if self._handle is not None:
119
+ self._handle.close()
120
+ self._handle = None
121
+ self._size = None
122
+
123
+ def __enter__(self) -> FileReader:
124
+ return self
125
+
126
+ def __exit__(self, *exc: object) -> None:
127
+ self.close()
128
+
129
+
130
+ class AsyncFileReader:
131
+ """AsyncReader implementation backed by a local file.
132
+
133
+ Regular files cannot be read without blocking, so every blocking
134
+ call is offloaded to a worker thread with :func:`asyncio.to_thread`,
135
+ keeping the event loop responsive. A missing path surfaces as
136
+ :exc:`FileNotFoundError`, mirroring how an HTTP 404 surfaces as
137
+ :exc:`OSError` in the network backends.
138
+ """
139
+
140
+ def __init__(self, path: str | os.PathLike[str], *, url: str | None = None) -> None:
141
+ self._path = os.fspath(path)
142
+ self._url = url if url is not None else self._path
143
+ self._handle: typing.BinaryIO | None = None
144
+ self._size: int | None = None
145
+
146
+ @classmethod
147
+ def from_uri(cls, uri: str) -> AsyncFileReader:
148
+ """Create an :class:`AsyncFileReader` from a ``file://`` URI."""
149
+ return cls(_path_from_file_uri(uri), url=uri)
150
+
151
+ @property
152
+ def url(self) -> str:
153
+ """The original path or ``file://`` URI for this reader."""
154
+ return self._url
155
+
156
+ def _open(self) -> typing.BinaryIO:
157
+ if self._handle is None:
158
+ self._handle = open(self._path, "rb") # noqa: SIM115
159
+ return self._handle
160
+
161
+ def _file_size(self) -> int:
162
+ # Cached from fstat on the open handle. We assume the file does
163
+ # not change for the lifetime of the reader.
164
+ if self._size is None:
165
+ self._size = os.fstat(self._open().fileno()).st_size
166
+ return self._size
167
+
168
+ async def head(self) -> Headers:
169
+ size = await asyncio.to_thread(self._file_size)
170
+ return _stat_headers(size)
171
+
172
+ async def read_range(self, offset: int, length: int) -> tuple[bytes, Headers]:
173
+ def _read() -> tuple[bytes, int]:
174
+ fh = self._open()
175
+ fh.seek(offset)
176
+ return fh.read(length), self._file_size()
177
+
178
+ data, size = await asyncio.to_thread(_read)
179
+ return data, _stat_headers(size)
180
+
181
+ async def stream_range(self, offset: int, length: int) -> AsyncIterator[bytes]:
182
+ def _open_seek() -> typing.BinaryIO:
183
+ fh = self._open()
184
+ fh.seek(offset)
185
+ return fh
186
+
187
+ fh = await asyncio.to_thread(_open_seek)
188
+ remaining = length
189
+ while remaining > 0:
190
+ chunk = await asyncio.to_thread(fh.read, min(STREAM_CHUNK_SIZE, remaining))
191
+ if not chunk:
192
+ break
193
+ remaining -= len(chunk)
194
+ yield chunk
195
+
196
+ async def close(self) -> None:
197
+ if self._handle is not None:
198
+ await asyncio.to_thread(self._handle.close)
199
+ self._handle = None
200
+ self._size = None
201
+
202
+ async def __aenter__(self) -> AsyncFileReader:
203
+ return self
204
+
205
+ async def __aexit__(self, *exc: object) -> None:
206
+ await self.close()