awrtifact 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. {awrtifact-0.2.0 → awrtifact-0.2.2}/PKG-INFO +1 -1
  2. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/__init__.py +1 -1
  3. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/workflows/mirror-to-release.yml +1 -1
  4. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/worker_template.py +123 -23
  5. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact.egg-info/PKG-INFO +1 -1
  6. {awrtifact-0.2.0 → awrtifact-0.2.2}/pyproject.toml +1 -1
  7. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_fetch.py +12 -8
  8. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_serve_spec.py +13 -0
  9. {awrtifact-0.2.0 → awrtifact-0.2.2}/README.md +0 -0
  10. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/__main__.py +0 -0
  11. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/_doctor.py +0 -0
  12. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/backup.py +0 -0
  13. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/cli.py +0 -0
  14. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/gobbonet/__init__.py +0 -0
  15. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/gobbonet/backup-gate.html +0 -0
  16. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/gobbonet/gobbonet-backup.js +0 -0
  17. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/workflows/__init__.py +0 -0
  18. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/workflows/hash-release-object.yml +0 -0
  19. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/workflows/mirror-hf-set.yml +0 -0
  20. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/data/workflows/mirror-hf-to-release.yml +0 -0
  21. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/fetch.py +0 -0
  22. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/fetchset.py +0 -0
  23. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/gh.py +0 -0
  24. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/ghapi.py +0 -0
  25. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/hashes.py +0 -0
  26. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/hf.py +0 -0
  27. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/manifest.py +0 -0
  28. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/mirror.py +0 -0
  29. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/mirrorset.py +0 -0
  30. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/plan.py +0 -0
  31. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/provision.py +0 -0
  32. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/sealing.py +0 -0
  33. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/serve_spec.py +0 -0
  34. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/spec.py +0 -0
  35. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/split.py +0 -0
  36. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/upload.py +0 -0
  37. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact/verify.py +0 -0
  38. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact.egg-info/SOURCES.txt +0 -0
  39. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact.egg-info/dependency_links.txt +0 -0
  40. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact.egg-info/entry_points.txt +0 -0
  41. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact.egg-info/requires.txt +0 -0
  42. {awrtifact-0.2.0 → awrtifact-0.2.2}/awrtifact.egg-info/top_level.txt +0 -0
  43. {awrtifact-0.2.0 → awrtifact-0.2.2}/setup.cfg +0 -0
  44. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_fetchset.py +0 -0
  45. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_ghapi.py +0 -0
  46. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_hf.py +0 -0
  47. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_manifest.py +0 -0
  48. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_mirror_hf.py +0 -0
  49. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_mirrorset.py +0 -0
  50. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_provision.py +0 -0
  51. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_sealing.py +0 -0
  52. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_split_verify.py +0 -0
  53. {awrtifact-0.2.0 → awrtifact-0.2.2}/tests/test_upload.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: awrtifact
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Aither World Artifact — deliberately chunk artifacts into GitHub release assets and fetch them back byte-verified. The productized aitherkvcache mirror lane.
5
5
  License: Apache-2.0
6
6
  Project-URL: Homepage, https://github.com/Aitherium/awrtifact
@@ -9,4 +9,4 @@ fetched back byte-verified.
9
9
  The core is stdlib-only; the spec-shaped commands need PyYAML (extra: `spec`).
10
10
  """
11
11
 
12
- __version__ = "0.2.0"
12
+ __version__ = "0.2.2"
@@ -132,7 +132,7 @@ jobs:
132
132
  gh release upload "$RELEASE" "$PART_FILE" --repo "$GITHUB_REPOSITORY" --clobber
133
133
  echo "uploaded $PART_FILE ($SIZE bytes)"
134
134
 
135
- # The release-manifest contract (D-2261): every release carries its
135
+ # The release-manifest contract: every release carries its
136
136
  # own <name>.manifest.json. The manifest job needs this part's hash —
137
137
  # ship it as a tiny artifact rather than re-downloading the part. The
138
138
  # inner filename is UNIQUE per row: merge-multiple with a shared
@@ -124,20 +124,12 @@ async function serveChunked(request, spec, name) {
124
124
  if (partStart > end) break;
125
125
  const subStart = Math.max(0, start - partStart);
126
126
  const subEnd = Math.min(part.size - 1, end - partStart);
127
- const upstreamResp = await fetch(spec.upstream + part.name, {
128
- headers: { Range: `bytes=${subStart}-${subEnd}` },
129
- redirect: 'follow',
130
- });
131
- if (upstreamResp.status !== 206 && upstreamResp.status !== 200) {
132
- throw new Error(`upstream ${part.name} -> ${upstreamResp.status}`);
133
- }
134
- const reader = upstreamResp.body.getReader();
135
- // eslint-disable-next-line no-constant-condition
136
- while (true) {
137
- const { done, value } = await reader.read();
138
- if (done) break;
139
- await writer.write(value);
140
- }
127
+ await streamUpstream(
128
+ spec.upstream + part.name,
129
+ { Range: `bytes=${subStart}-${subEnd}` },
130
+ writer,
131
+ part.name,
132
+ );
141
133
  partStart += part.size;
142
134
  }
143
135
  await writer.close();
@@ -148,6 +140,64 @@ async function serveChunked(request, spec, name) {
148
140
  return new Response(readable, { status, headers });
149
141
  }
150
142
 
143
+ // Stream one upstream body into the writer, RETRYING an EMPTY body. A
144
+ // Cloudflare→GitHub fetch can return a valid status with ZERO bytes
145
+ // (measured 2026-08-28: 206/0 at a part seam — the client then sees
146
+ // Content-Length promise a full range and receive nothing, which its
147
+ // truncation check reads as corruption; the AW002 seam gate caught it as
148
+ // 'does not stitch'). Streaming-safe: only the FIRST chunk is awaited to
149
+ // decide, so large bodies are never buffered.
150
+ async function streamUpstream(url, headers, writer, what) {
151
+ for (let attempt = 1; attempt <= 5; attempt++) {
152
+ const resp = await fetch(url, { headers, redirect: 'follow' });
153
+ if (resp.status === 429 || resp.status >= 500) {
154
+ // TRANSIENT upstream state (GitHub rate-limits the shared egress IP —
155
+ // measured 2026-08-28: a burst of fresh-fetch probes tripped it, and
156
+ // throwing here killed the stream into a 206/0 to the client). Retry,
157
+ // never fail the stream on a 429/5xx.
158
+ continue;
159
+ }
160
+ if (resp.status !== 206 && resp.status !== 200) {
161
+ throw new Error(`upstream ${what} -> ${resp.status}`);
162
+ }
163
+ const reader = resp.body.getReader();
164
+ let first;
165
+ try {
166
+ first = await reader.read();
167
+ } catch (_e) {
168
+ // The connection died between the response headers and the first body
169
+ // chunk — measured 2026-09-01 at ~40% of cold Cloudflare->GitHub
170
+ // fetches of the 90MB part, and the deployed empty-chunk retry did NOT
171
+ // move the client-visible rate (3/10 pre -> 4/10 post): the read
172
+ // THROWS here, it does not return empty, so no retry loop ever saw it.
173
+ // Uncaught, it aborts the writer AFTER the 206 status was sent — the
174
+ // client-visible 206/0. Same treatment as an empty body: back off and
175
+ // retry the fetch.
176
+ await new Promise((r) => setTimeout(r, 500));
177
+ continue;
178
+ }
179
+ if (!first.value) {
180
+ // 0-length FIRST chunk — cold upstream connections (Cloudflare -> GitHub)
181
+ // can deliver valid 206/200 headers with an empty body, and the first
182
+ // chunk is NOT always final (measured 2026-09-01: ~30-40% of cold first
183
+ // fetches). Cancel the reader so the connection drains, back off, and
184
+ // retry the fetch.
185
+ await reader.cancel();
186
+ await new Promise((r) => setTimeout(r, 500));
187
+ continue;
188
+ }
189
+ await writer.write(first.value);
190
+ // eslint-disable-next-line no-constant-condition
191
+ while (true) {
192
+ const { done, value } = await reader.read();
193
+ if (done) break;
194
+ await writer.write(value);
195
+ }
196
+ return;
197
+ }
198
+ throw new Error(`upstream ${what} failed after 5 attempts`);
199
+ }
200
+
151
201
  /**
152
202
  * R2 first. Everything below is unchanged and stays as the fallback.
153
203
  *
@@ -241,21 +291,71 @@ export default {
241
291
  // R2 before everything, INCLUDING the chunked path: an object uploaded whole
242
292
  // to R2 makes its `.partN` manifest irrelevant, and checking after would keep
243
293
  // serving the stitched copy of a file that no longer needs stitching.
244
- const fromR2 = await serveFromR2(request, env, name);
294
+ // A PREFIXED request names one release, so a same-named object from another
295
+ // release must not answer it. Measured 2026-09-03: /microembedder-v2/config.json
296
+ // served microembedder-v1's 650-byte config (and its 22,972,370-byte ONNX) because
297
+ // R2 and the chunked map are keyed by bare name and were consulted BEFORE the
298
+ // prefix. R2 is skipped for prefixed paths (its keys carry no release); chunked
299
+ // entries answer only when their upstream IS the requested release.
300
+ const fromR2 = baseOverride ? null : await serveFromR2(request, env, name);
245
301
  if (fromR2) return fromR2;
246
302
  // Virtual chunked asset (>2 GiB source, split at upload).
247
- if (CHUNKED[name]) return serveChunked(request, CHUNKED[name], name);
303
+ if (CHUNKED[name] && (!baseOverride || CHUNKED[name].upstream === baseOverride)) {
304
+ return serveChunked(request, CHUNKED[name], name);
305
+ }
248
306
  // Try each upstream until one has the file. A GitHub release 404s fast for a
249
307
  // missing asset, so the fallback cost is one small miss per unknown name.
250
308
  // A prefixed path has exactly ONE candidate (its own release).
251
309
  const candidates = baseOverride ? [baseOverride] : UPSTREAMS;
252
310
  for (const base of candidates) {
253
- const upstream = await fetch(base + name, {
254
- method: request.method === 'HEAD' ? 'HEAD' : 'GET',
255
- headers: request.headers.has('Range') ? { Range: request.headers.get('Range') } : {},
256
- redirect: 'follow',
257
- });
258
- if (upstream.status !== 404 && upstream.status !== 410) {
311
+ if (request.method === 'HEAD') {
312
+ const upstream = await fetch(base + name, { method: 'HEAD', redirect: 'follow' });
313
+ if (upstream.status === 404 || upstream.status === 410) continue;
314
+ const headers = new Headers(upstream.headers);
315
+ for (const [k, v] of Object.entries(cors)) headers.set(k, v);
316
+ headers.set('Cache-Control', 'public, max-age=31536000, immutable');
317
+ headers.set('Content-Type', contentTypeFor(name));
318
+ return new Response(null, { status: upstream.status, headers });
319
+ }
320
+ // GET with retry on an EMPTY body (the 206/0 class, measured 2026-08-28).
321
+ // Streaming-safe: only the first chunk is awaited to decide.
322
+ for (let attempt = 1; attempt <= 3; attempt++) {
323
+ const upstream = await fetch(base + name, {
324
+ method: 'GET',
325
+ headers: request.headers.has('Range') ? { Range: request.headers.get('Range') } : {},
326
+ redirect: 'follow',
327
+ });
328
+ if (upstream.status === 404 || upstream.status === 410) break; // next upstream
329
+ if (upstream.status === 429 || upstream.status >= 500) {
330
+ // transient (rate limit) — retry, never serve the error as bytes
331
+ continue;
332
+ }
333
+ const reader = upstream.body.getReader();
334
+ let first;
335
+ try {
336
+ first = await reader.read();
337
+ } catch (_e) {
338
+ // The connection died between headers and the first chunk — the
339
+ // same class as streamUpstream (measured 2026-09-01 at ~40% of cold
340
+ // Cloudflare->GitHub fetches); a throw here is NOT the empty-body
341
+ // check below, and uncaught it aborts the response after the status
342
+ // was sent. Back off and retry the fetch.
343
+ await new Promise((r) => setTimeout(r, 500));
344
+ continue;
345
+ }
346
+ if (!first.value) continue; // empty first chunk (0-length or done) — retry
347
+ const body = new ReadableStream({
348
+ async start(controller) {
349
+ if (first.value) controller.enqueue(first.value);
350
+ // eslint-disable-next-line no-constant-condition
351
+ while (true) {
352
+ const { done, value } = await reader.read();
353
+ if (done) break;
354
+ controller.enqueue(value);
355
+ }
356
+ controller.close();
357
+ },
358
+ });
259
359
  const headers = new Headers(upstream.headers);
260
360
  for (const [k, v] of Object.entries(cors)) headers.set(k, v);
261
361
  headers.set('Cache-Control', 'public, max-age=31536000, immutable');
@@ -263,7 +363,7 @@ export default {
263
363
  // `import()` refuses that MIME (measured 2026-08-26). The upstream
264
364
  // headers are copied for Content-Length/Range, but the TYPE is always ours.
265
365
  headers.set('Content-Type', contentTypeFor(name));
266
- return new Response(upstream.body, { status: upstream.status, headers });
366
+ return new Response(body, { status: upstream.status, headers });
267
367
  }
268
368
  }
269
369
  const headers = new Headers(cors);
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: awrtifact
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Aither World Artifact — deliberately chunk artifacts into GitHub release assets and fetch them back byte-verified. The productized aitherkvcache mirror lane.
5
5
  License: Apache-2.0
6
6
  Project-URL: Homepage, https://github.com/Aitherium/awrtifact
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "awrtifact"
7
- version = "0.2.0"
7
+ version = "0.2.2"
8
8
  description = "Aither World Artifact — deliberately chunk artifacts into GitHub release assets and fetch them back byte-verified. The productized aitherkvcache mirror lane."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -1,9 +1,13 @@
1
1
  """awrtifact fetch: base-URL and full-URL forms both land the named asset; a 404 is a
2
2
  refusal with a non-zero exit, never a silent success.
3
3
 
4
- Measured 2026-09-02 from inside a fleet container: `awrtifact fetch NAME --url
5
- https://artifact.aitherium.com/` (the README's documented form) fetched the BASE
6
- verbatim, got the worker's 404 for the root, and the caller saw an empty output dir.
4
+ Measured 2026-09-02 from inside a fleet container: `awrtifact fetch NAME --url https://<store>/`
5
+ (the README's documented form) fetched the BASE verbatim, got the worker's 404 for
6
+ the root, and the caller saw an empty output dir.
7
+
8
+ The asset here is a neutral placeholder on purpose: this file SHIPS in the sdist,
9
+ and the moat rule (AWRF005) is that a published package must not advertise which
10
+ models we serve. The behaviour under test is the URL join, not the name.
7
11
  """
8
12
  from __future__ import annotations
9
13
 
@@ -30,7 +34,7 @@ class _Handler(http.server.BaseHTTPRequestHandler):
30
34
  pass
31
35
 
32
36
  def _serve(self, head_only: bool) -> None:
33
- if self.path != "/aither-code-embed.config.json":
37
+ if self.path != "/example-model.config.json":
34
38
  self.send_response(404)
35
39
  self.end_headers()
36
40
  return
@@ -68,18 +72,18 @@ def server():
68
72
 
69
73
 
70
74
  def test_base_url_form_lands_the_named_asset(server, tmp_path):
71
- r = fetch_mod.fetch("aither-code-embed.config.json", server, tmp_path, expected=len(BODY),
75
+ r = fetch_mod.fetch("example-model.config.json", server, tmp_path, expected=len(BODY),
72
76
  lockfile=tmp_path / "lock.json")
73
77
  assert r["status"] == "fetched"
74
- assert (tmp_path / "aither-code-embed.config.json").read_bytes() == BODY
78
+ assert (tmp_path / "example-model.config.json").read_bytes() == BODY
75
79
  assert r["sha256"] == hashlib.sha256(BODY).hexdigest()
76
80
 
77
81
 
78
82
  def test_full_url_form_still_works(server, tmp_path):
79
- r = fetch_mod.fetch("aither-code-embed.config.json", server + "aither-code-embed.config.json",
83
+ r = fetch_mod.fetch("example-model.config.json", server + "example-model.config.json",
80
84
  tmp_path, expected=len(BODY), lockfile=tmp_path / "lock.json")
81
85
  assert r["status"] == "fetched"
82
- assert (tmp_path / "aither-code-embed.config.json").read_bytes() == BODY
86
+ assert (tmp_path / "example-model.config.json").read_bytes() == BODY
83
87
 
84
88
 
85
89
  def test_missing_asset_is_a_refusal_not_an_empty_dir(server, tmp_path):
@@ -157,3 +157,16 @@ def test_prefix_route_namespaces_duplicate_names(tmp_path):
157
157
  base["artifacts"] = list(base["artifacts"]) + [same]
158
158
  with pytest.raises(ValueError, match="same name"):
159
159
  spec_mod.validate(base)
160
+
161
+
162
+ def test_prefix_route_is_a_namespace_for_chunked_and_r2():
163
+ """2026-09-03: /microembedder-v2/config.json served microembedder-v1's 650-byte config
164
+ because R2 and CHUNKED were consulted by bare name BEFORE the release prefix. The
165
+ generated worker must skip R2 for prefixed paths and answer a chunked entry only
166
+ when its upstream IS the requested release."""
167
+ from awrtifact import worker_template
168
+
169
+ src = worker_template.__file__
170
+ text = open(src, encoding="utf-8").read()
171
+ assert "const fromR2 = baseOverride ? null : await serveFromR2(request, env, name);" in text
172
+ assert "CHUNKED[name].upstream === baseOverride" in text
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes