awrtifact 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- awrtifact-0.1.0/PKG-INFO +82 -0
- awrtifact-0.1.0/README.md +68 -0
- awrtifact-0.1.0/awrtifact/__init__.py +12 -0
- awrtifact-0.1.0/awrtifact/__main__.py +4 -0
- awrtifact-0.1.0/awrtifact/backup.py +92 -0
- awrtifact-0.1.0/awrtifact/cli.py +253 -0
- awrtifact-0.1.0/awrtifact/fetch.py +174 -0
- awrtifact-0.1.0/awrtifact/gh.py +132 -0
- awrtifact-0.1.0/awrtifact/hashes.py +50 -0
- awrtifact-0.1.0/awrtifact/manifest.py +107 -0
- awrtifact-0.1.0/awrtifact/mirror.py +169 -0
- awrtifact-0.1.0/awrtifact/plan.py +40 -0
- awrtifact-0.1.0/awrtifact/serve_spec.py +124 -0
- awrtifact-0.1.0/awrtifact/spec.py +232 -0
- awrtifact-0.1.0/awrtifact/split.py +97 -0
- awrtifact-0.1.0/awrtifact/upload.py +83 -0
- awrtifact-0.1.0/awrtifact/verify.py +72 -0
- awrtifact-0.1.0/awrtifact/worker_template.py +280 -0
- awrtifact-0.1.0/awrtifact.egg-info/PKG-INFO +82 -0
- awrtifact-0.1.0/awrtifact.egg-info/SOURCES.txt +27 -0
- awrtifact-0.1.0/awrtifact.egg-info/dependency_links.txt +1 -0
- awrtifact-0.1.0/awrtifact.egg-info/entry_points.txt +2 -0
- awrtifact-0.1.0/awrtifact.egg-info/requires.txt +3 -0
- awrtifact-0.1.0/awrtifact.egg-info/top_level.txt +1 -0
- awrtifact-0.1.0/pyproject.toml +30 -0
- awrtifact-0.1.0/setup.cfg +4 -0
- awrtifact-0.1.0/tests/test_manifest.py +60 -0
- awrtifact-0.1.0/tests/test_serve_spec.py +127 -0
- awrtifact-0.1.0/tests/test_split_verify.py +87 -0
awrtifact-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: awrtifact
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Aither World Artifact — deliberately chunk artifacts into GitHub release assets and fetch them back byte-verified. The productized aitherkvcache mirror lane.
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Project-URL: Homepage, https://github.com/Aitherium/awrtifact
|
|
7
|
+
Project-URL: Documentation, https://github.com/Aitherium/awrtifact#readme
|
|
8
|
+
Project-URL: Repository, https://github.com/Aitherium/awrtifact.git
|
|
9
|
+
Project-URL: Issues, https://github.com/Aitherium/awrtifact/issues
|
|
10
|
+
Requires-Python: >=3.10
|
|
11
|
+
Description-Content-Type: text/markdown
|
|
12
|
+
Provides-Extra: spec
|
|
13
|
+
Requires-Dist: PyYAML>=6.0; extra == "spec"
|
|
14
|
+
|
|
15
|
+
# awrtifact — Aither World Artifact
|
|
16
|
+
|
|
17
|
+
Deliberately chunk artifacts into GitHub release assets and fetch them back
|
|
18
|
+
byte-verified. The productized aitherkvcache mirror lane: any artifact (model
|
|
19
|
+
weights, datasets, builds, backups) → `.partN` slices under GitHub's 2 GiB
|
|
20
|
+
per-asset cap → versioned release → served by a generated Cloudflare Worker
|
|
21
|
+
(CORS + HTTP Range + stitching) → fetched back with size and sha256 checks.
|
|
22
|
+
|
|
23
|
+
## The loop
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
# 1. Split a local artifact into .partN slices + manifest.json
|
|
27
|
+
awrtifact split DeepSeek-V4-Flash.gguf --out parts/
|
|
28
|
+
|
|
29
|
+
# 2. What is missing from the release? (resumable — re-runs upload only gaps)
|
|
30
|
+
awrtifact plan parts/manifest.json --repo Aitherium/aitherkvcache --release fleet-v1
|
|
31
|
+
|
|
32
|
+
# 3. Upload the missing parts (size-checked before upload, --create makes the release)
|
|
33
|
+
awrtifact upload parts/manifest.json --repo Aitherium/aitherkvcache \
|
|
34
|
+
--release fleet-v1 --dir parts --parallel 8 --create
|
|
35
|
+
|
|
36
|
+
# 4. Prove the local parts match the manifest byte-for-byte
|
|
37
|
+
awrtifact verify parts/manifest.json --dir parts
|
|
38
|
+
|
|
39
|
+
# 5. Fetch it back anywhere (resume + size check + sha256 TOFU lockfile)
|
|
40
|
+
awrtifact fetch DeepSeek-V4-Flash.gguf --url https://weights.example.com/ \
|
|
41
|
+
--out /srv/models --expected 49999999999
|
|
42
|
+
|
|
43
|
+
# 6. Or let the spec drive everything: generate the serving worker from it
|
|
44
|
+
awrtifact serve-spec awrtifact.yaml --check # drift gate
|
|
45
|
+
awrtifact serve-spec awrtifact.yaml # write the generated worker
|
|
46
|
+
|
|
47
|
+
# 7. Bring the backup up deliberately (dispatch mirror runs for gaps)
|
|
48
|
+
awrtifact backup-catalog awrtifact.yaml --dry-run
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Why the checks exist
|
|
52
|
+
|
|
53
|
+
- **2 GiB cap.** GitHub release assets top out at 2 GiB; larger artifacts are
|
|
54
|
+
split at 1.9 GB. The worker stitches `.partN` slices behind the original
|
|
55
|
+
filename, so clients ask for the name, never the parts.
|
|
56
|
+
- **Size is the truncation detector.** A short file is not a download error —
|
|
57
|
+
the loader reports a corrupt artifact. Every step verifies sizes; the
|
|
58
|
+
manifest's `total` is the number the client checks.
|
|
59
|
+
- **sha256 per part and whole.** Split records both; verify proves each slice
|
|
60
|
+
and that the slices stitch back into the original.
|
|
61
|
+
- **Resume everywhere.** Uploads skip present parts; fetches resume from the
|
|
62
|
+
byte count on disk via Range.
|
|
63
|
+
|
|
64
|
+
## Spec (the declarative store)
|
|
65
|
+
|
|
66
|
+
`awrtifact.yaml` names the store's repo, releases, allowlist and artifacts
|
|
67
|
+
(see `awrtifact/spec.py` for the full shape). The worker's data sections are
|
|
68
|
+
GENERATED from it — adding an artifact is a spec edit + regenerate, never a
|
|
69
|
+
worker code edit. `serve-spec --check` is the drift gate.
|
|
70
|
+
|
|
71
|
+
## Install
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install awrtifact # core (stdlib-only)
|
|
75
|
+
pip install 'awrtifact[spec]' # + spec commands (PyYAML)
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
Requires the `gh` CLI for plan/upload/backup commands (your existing auth).
|
|
79
|
+
|
|
80
|
+
## License
|
|
81
|
+
|
|
82
|
+
Apache-2.0
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# awrtifact — Aither World Artifact
|
|
2
|
+
|
|
3
|
+
Deliberately chunk artifacts into GitHub release assets and fetch them back
|
|
4
|
+
byte-verified. The productized aitherkvcache mirror lane: any artifact (model
|
|
5
|
+
weights, datasets, builds, backups) → `.partN` slices under GitHub's 2 GiB
|
|
6
|
+
per-asset cap → versioned release → served by a generated Cloudflare Worker
|
|
7
|
+
(CORS + HTTP Range + stitching) → fetched back with size and sha256 checks.
|
|
8
|
+
|
|
9
|
+
## The loop
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
# 1. Split a local artifact into .partN slices + manifest.json
|
|
13
|
+
awrtifact split DeepSeek-V4-Flash.gguf --out parts/
|
|
14
|
+
|
|
15
|
+
# 2. What is missing from the release? (resumable — re-runs upload only gaps)
|
|
16
|
+
awrtifact plan parts/manifest.json --repo Aitherium/aitherkvcache --release fleet-v1
|
|
17
|
+
|
|
18
|
+
# 3. Upload the missing parts (size-checked before upload, --create makes the release)
|
|
19
|
+
awrtifact upload parts/manifest.json --repo Aitherium/aitherkvcache \
|
|
20
|
+
--release fleet-v1 --dir parts --parallel 8 --create
|
|
21
|
+
|
|
22
|
+
# 4. Prove the local parts match the manifest byte-for-byte
|
|
23
|
+
awrtifact verify parts/manifest.json --dir parts
|
|
24
|
+
|
|
25
|
+
# 5. Fetch it back anywhere (resume + size check + sha256 TOFU lockfile)
|
|
26
|
+
awrtifact fetch DeepSeek-V4-Flash.gguf --url https://weights.example.com/ \
|
|
27
|
+
--out /srv/models --expected 49999999999
|
|
28
|
+
|
|
29
|
+
# 6. Or let the spec drive everything: generate the serving worker from it
|
|
30
|
+
awrtifact serve-spec awrtifact.yaml --check # drift gate
|
|
31
|
+
awrtifact serve-spec awrtifact.yaml # write the generated worker
|
|
32
|
+
|
|
33
|
+
# 7. Bring the backup up deliberately (dispatch mirror runs for gaps)
|
|
34
|
+
awrtifact backup-catalog awrtifact.yaml --dry-run
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Why the checks exist
|
|
38
|
+
|
|
39
|
+
- **2 GiB cap.** GitHub release assets top out at 2 GiB; larger artifacts are
|
|
40
|
+
split at 1.9 GB. The worker stitches `.partN` slices behind the original
|
|
41
|
+
filename, so clients ask for the name, never the parts.
|
|
42
|
+
- **Size is the truncation detector.** A short file is not a download error —
|
|
43
|
+
the loader reports a corrupt artifact. Every step verifies sizes; the
|
|
44
|
+
manifest's `total` is the number the client checks.
|
|
45
|
+
- **sha256 per part and whole.** Split records both; verify proves each slice
|
|
46
|
+
and that the slices stitch back into the original.
|
|
47
|
+
- **Resume everywhere.** Uploads skip present parts; fetches resume from the
|
|
48
|
+
byte count on disk via Range.
|
|
49
|
+
|
|
50
|
+
## Spec (the declarative store)
|
|
51
|
+
|
|
52
|
+
`awrtifact.yaml` names the store's repo, releases, allowlist and artifacts
|
|
53
|
+
(see `awrtifact/spec.py` for the full shape). The worker's data sections are
|
|
54
|
+
GENERATED from it — adding an artifact is a spec edit + regenerate, never a
|
|
55
|
+
worker code edit. `serve-spec --check` is the drift gate.
|
|
56
|
+
|
|
57
|
+
## Install
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
pip install awrtifact # core (stdlib-only)
|
|
61
|
+
pip install 'awrtifact[spec]' # + spec commands (PyYAML)
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Requires the `gh` CLI for plan/upload/backup commands (your existing auth).
|
|
65
|
+
|
|
66
|
+
## License
|
|
67
|
+
|
|
68
|
+
Apache-2.0
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""awrtifact — deliberately chunk artifacts into GitHub release assets.
|
|
2
|
+
|
|
3
|
+
Aither World Artifact is the productized aitherkvcache mirror lane: ANY artifact
|
|
4
|
+
(model weights, datasets, builds, backups) is split into `.partN` slices under
|
|
5
|
+
GitHub's 2 GiB per-asset cap, stored as a versioned GitHub release, served by a
|
|
6
|
+
generated Cloudflare Worker (CORS + HTTP Range + `.partN` stitching), and
|
|
7
|
+
fetched back byte-verified.
|
|
8
|
+
|
|
9
|
+
The core is stdlib-only; the spec-shaped commands need PyYAML (extra: `spec`).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""`awrtifact backup-catalog` — dispatch the mirror workflow for gaps.
|
|
2
|
+
|
|
3
|
+
The point of the precaution: a spec with `source_url` per artifact is the
|
|
4
|
+
backup MANIFEST — every artifact listed has a named, size-checked origin and
|
|
5
|
+
a target release. This command turns the spec into workflow dispatches for
|
|
6
|
+
artifacts whose parts are NOT yet in the release (the resumable plan check),
|
|
7
|
+
so the mirror can be brought up deliberately and capped (the workflow's own
|
|
8
|
+
≤20-runner fan-out; the origin box's uplink is never involved).
|
|
9
|
+
|
|
10
|
+
`--dry-run` prints the dispatches; without it, dispatches via `gh workflow
|
|
11
|
+
run`. The workflow must exist in the store's repo — this checks, because a
|
|
12
|
+
dispatch to a missing workflow fails as a SILENCE (gh prints a warning and
|
|
13
|
+
exits 0 — measured class; `gh workflow run` on an unknown workflow does not
|
|
14
|
+
fail the caller).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import sys
|
|
20
|
+
|
|
21
|
+
from . import gh
|
|
22
|
+
from . import spec as spec_mod
|
|
23
|
+
|
|
24
|
+
DEFAULT_WORKFLOW = "mirror-hf-to-release.yml"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def backup_catalog(
|
|
28
|
+
spec: dict,
|
|
29
|
+
workflow: str = DEFAULT_WORKFLOW,
|
|
30
|
+
dry_run: bool = False,
|
|
31
|
+
) -> dict:
|
|
32
|
+
"""Dispatch mirror runs for artifacts missing from their releases."""
|
|
33
|
+
repo = spec["store"]["repo"]
|
|
34
|
+
if not dry_run and not gh.workflow_exists(repo, workflow):
|
|
35
|
+
raise ValueError(
|
|
36
|
+
f"workflow {workflow} does not exist in {repo} — cannot dispatch; "
|
|
37
|
+
f"re-push the workflow first"
|
|
38
|
+
)
|
|
39
|
+
planned: list[dict] = []
|
|
40
|
+
local_only: list[str] = []
|
|
41
|
+
for art in spec.get("artifacts") or []:
|
|
42
|
+
name = art["name"]
|
|
43
|
+
if not art.get("source_url"):
|
|
44
|
+
# Local-only artifact: no cloud lane can mirror it; it is a
|
|
45
|
+
# `awrtifact mirror <local-path>` candidate. A None source_url
|
|
46
|
+
# reaching the dispatch would upload parts named "None.partN"
|
|
47
|
+
# (the measured 2026-08-27 incident) — skipped and reported.
|
|
48
|
+
local_only.append(name)
|
|
49
|
+
continue
|
|
50
|
+
release = spec_mod.artifact_release(spec, art)
|
|
51
|
+
existing = gh.release_assets(repo, release)
|
|
52
|
+
parts = spec_mod.chunked_map(spec).get(name)
|
|
53
|
+
if not parts:
|
|
54
|
+
# Whole-file artifact: present when the asset itself exists.
|
|
55
|
+
if name in existing:
|
|
56
|
+
continue
|
|
57
|
+
inputs = {
|
|
58
|
+
"hf_url": art["source_url"],
|
|
59
|
+
"name": name,
|
|
60
|
+
"total_bytes": str(art["total"]),
|
|
61
|
+
"release": release,
|
|
62
|
+
}
|
|
63
|
+
planned.append({"artifact": name, "inputs": inputs})
|
|
64
|
+
continue
|
|
65
|
+
missing = [p["name"] for p in parts["parts"] if p["name"] not in existing]
|
|
66
|
+
if not missing:
|
|
67
|
+
continue
|
|
68
|
+
inputs = {
|
|
69
|
+
"hf_url": art["source_url"],
|
|
70
|
+
"name": name,
|
|
71
|
+
"total_bytes": str(art["total"]),
|
|
72
|
+
"release": release,
|
|
73
|
+
"part_size": str(art.get("part_size") or 1900000000),
|
|
74
|
+
}
|
|
75
|
+
planned.append(
|
|
76
|
+
{"artifact": name, "inputs": inputs, "missing_parts": len(missing)}
|
|
77
|
+
)
|
|
78
|
+
if dry_run:
|
|
79
|
+
for item in planned:
|
|
80
|
+
print(f"dispatch {item['artifact']}: {item['inputs']}", file=sys.stderr)
|
|
81
|
+
else:
|
|
82
|
+
for item in planned:
|
|
83
|
+
gh.workflow_dispatch(repo, workflow, item["inputs"])
|
|
84
|
+
print(f"dispatched {item['artifact']} -> {workflow} in {repo}")
|
|
85
|
+
return {
|
|
86
|
+
"dispatched": len(planned),
|
|
87
|
+
"local_only": local_only,
|
|
88
|
+
"already_present": len(
|
|
89
|
+
[a for a in spec.get("artifacts") or [] if a["name"] not in
|
|
90
|
+
{i["artifact"] for i in planned} and a["name"] not in local_only]
|
|
91
|
+
),
|
|
92
|
+
}
|
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
"""awrtifact CLI — the deliberate chunk-and-release loop.
|
|
2
|
+
|
|
3
|
+
awrtifact split FILE [--part-size N] [--out DIR]
|
|
4
|
+
awrtifact plan MANIFEST --repo OWNER/REPO --release TAG
|
|
5
|
+
awrtifact upload MANIFEST --repo OWNER/REPO --release TAG [--dir DIR] [--parallel N] [--create]
|
|
6
|
+
awrtifact verify MANIFEST [--dir DIR]
|
|
7
|
+
awrtifact fetch NAME --url BASE --out DIR [--expected N] [--verify-only]
|
|
8
|
+
awrtifact serve-spec SPEC [--emit-dir DIR] [--check]
|
|
9
|
+
awrtifact backup-catalog SPEC [--workflow W] [--dry-run]
|
|
10
|
+
|
|
11
|
+
Exit codes: 0 ok · 1 operational failure (upload failed, fetch short, drift) ·
|
|
12
|
+
2 usage or data error (bad manifest/spec, missing dependency).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from . import __version__, gh
|
|
22
|
+
from . import backup as backup_mod
|
|
23
|
+
from . import fetch as fetch_mod
|
|
24
|
+
from . import manifest as manifest_mod
|
|
25
|
+
from . import mirror as mirror_mod
|
|
26
|
+
from . import plan as plan_mod
|
|
27
|
+
from . import serve_spec as serve_spec_mod
|
|
28
|
+
from . import split as split_mod
|
|
29
|
+
from . import upload as upload_mod
|
|
30
|
+
from . import verify as verify_mod
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _die(message: str, code: int = 2) -> int:
|
|
34
|
+
print(f"awrtifact: {message}", file=sys.stderr)
|
|
35
|
+
return code
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _cmd_split(args: argparse.Namespace) -> int:
|
|
39
|
+
try:
|
|
40
|
+
m = split_mod.split_file(Path(args.file), args.part_size,
|
|
41
|
+
Path(args.out) if args.out else None)
|
|
42
|
+
except (ValueError, OSError) as exc:
|
|
43
|
+
return _die(str(exc))
|
|
44
|
+
out = Path(args.out) if args.out else Path(args.file).parent
|
|
45
|
+
manifest_path = out / "manifest.json"
|
|
46
|
+
manifest_mod.write(m, manifest_path)
|
|
47
|
+
print(f"split {m['name']}: {m['total']} bytes -> "
|
|
48
|
+
f"{len(m['parts'])} parts, manifest {manifest_path}")
|
|
49
|
+
return 0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _cmd_plan(args: argparse.Namespace) -> int:
|
|
53
|
+
try:
|
|
54
|
+
m = manifest_mod.load(Path(args.manifest))
|
|
55
|
+
planned = plan_mod.plan_parts(m, args.repo, args.release)
|
|
56
|
+
except (ValueError, OSError) as exc:
|
|
57
|
+
return _die(str(exc))
|
|
58
|
+
gh.print_json(planned)
|
|
59
|
+
return 0
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _cmd_upload(args: argparse.Namespace) -> int:
|
|
63
|
+
try:
|
|
64
|
+
m = manifest_mod.load(Path(args.manifest))
|
|
65
|
+
result = upload_mod.upload_manifest(
|
|
66
|
+
m, args.repo, args.release,
|
|
67
|
+
Path(args.dir) if args.dir else Path(args.manifest).parent,
|
|
68
|
+
parallel=args.parallel, create=args.create,
|
|
69
|
+
)
|
|
70
|
+
except (ValueError, OSError) as exc:
|
|
71
|
+
return _die(str(exc))
|
|
72
|
+
print(f"uploaded {len(result['uploaded'])} part(s); "
|
|
73
|
+
f"{len(result['skipped_present'])} already present")
|
|
74
|
+
if result["failed"]:
|
|
75
|
+
for f in result["failed"]:
|
|
76
|
+
print(f" FAILED: {f}", file=sys.stderr)
|
|
77
|
+
return 1
|
|
78
|
+
return 0
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _cmd_verify(args: argparse.Namespace) -> int:
|
|
82
|
+
try:
|
|
83
|
+
m = manifest_mod.load(Path(args.manifest))
|
|
84
|
+
report = verify_mod.verify_manifest(m, Path(args.dir) if args.dir
|
|
85
|
+
else Path(args.manifest).parent)
|
|
86
|
+
except (ValueError, OSError) as exc:
|
|
87
|
+
return _die(str(exc))
|
|
88
|
+
if args.json:
|
|
89
|
+
gh.print_json(report)
|
|
90
|
+
else:
|
|
91
|
+
for check in report["checks"]:
|
|
92
|
+
mark = "ok" if check["ok"] else "FAIL " + ", ".join(check["errors"])
|
|
93
|
+
print(f" {check['part']}: {mark}")
|
|
94
|
+
whole = report["whole"]
|
|
95
|
+
mark = "ok" if whole["ok"] else "FAIL " + ", ".join(whole["errors"])
|
|
96
|
+
print(f" whole-file: {mark}")
|
|
97
|
+
return 0 if report["ok"] else 1
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _cmd_fetch(args: argparse.Namespace) -> int:
|
|
101
|
+
try:
|
|
102
|
+
result = fetch_mod.fetch(
|
|
103
|
+
args.name, args.url, Path(args.out),
|
|
104
|
+
expected=args.expected, lockfile=Path(args.lock) if args.lock else None,
|
|
105
|
+
verify_only=args.verify_only,
|
|
106
|
+
)
|
|
107
|
+
except (ValueError, OSError, fetch_mod.FetchError) as exc:
|
|
108
|
+
return _die(str(exc), 1)
|
|
109
|
+
print(f"{result['status']}: {result['path']} ({result['bytes']} bytes, "
|
|
110
|
+
f"sha256 {result['sha256'][:16]}…)")
|
|
111
|
+
return 0 if result["status"] in ("fetched", "verified", "up-to-date") else 1
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _cmd_serve_spec(args: argparse.Namespace) -> int:
|
|
115
|
+
spec_path = Path(args.spec)
|
|
116
|
+
try:
|
|
117
|
+
dirs = serve_spec_mod.emit(spec_mod_load(spec_path), spec_path, check=args.check)
|
|
118
|
+
except ValueError as exc:
|
|
119
|
+
return _die(str(exc))
|
|
120
|
+
if args.check:
|
|
121
|
+
print(f"spec {spec_path}: generated workers are current")
|
|
122
|
+
else:
|
|
123
|
+
for d in dirs:
|
|
124
|
+
print(f"generated worker -> {d}")
|
|
125
|
+
return 0
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def spec_mod_load(path: Path) -> dict:
|
|
129
|
+
# Imported lazily so the core commands stay stdlib-only; the spec module
|
|
130
|
+
# raises a clear "pip install awrtifact[spec]" when PyYAML is absent.
|
|
131
|
+
from . import spec as spec_mod # noqa: PLC0415 — optional dependency
|
|
132
|
+
|
|
133
|
+
return spec_mod.load(path)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _cmd_mirror(args: argparse.Namespace) -> int:
|
|
137
|
+
try:
|
|
138
|
+
if mirror_mod.URL_RE.match(args.source):
|
|
139
|
+
result = mirror_mod.mirror_url(
|
|
140
|
+
args.source, args.name, args.release, args.repo,
|
|
141
|
+
total=args.total, workflow=args.workflow,
|
|
142
|
+
)
|
|
143
|
+
else:
|
|
144
|
+
result = mirror_mod.mirror_file(
|
|
145
|
+
Path(args.source), args.release, args.repo,
|
|
146
|
+
name=args.name, parallel=args.parallel,
|
|
147
|
+
)
|
|
148
|
+
except (ValueError, OSError) as exc:
|
|
149
|
+
return _die(str(exc))
|
|
150
|
+
print(f"mirror {result['lane']}: {result}")
|
|
151
|
+
return 0
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _cmd_backup(args: argparse.Namespace) -> int:
|
|
155
|
+
try:
|
|
156
|
+
spec = spec_mod_load(Path(args.spec))
|
|
157
|
+
result = backup_mod.backup_catalog(
|
|
158
|
+
spec, workflow=args.workflow, dry_run=args.dry_run
|
|
159
|
+
)
|
|
160
|
+
except (ValueError, OSError) as exc:
|
|
161
|
+
return _die(str(exc))
|
|
162
|
+
print(f"planned {result['dispatched']} dispatch(es); "
|
|
163
|
+
f"{result['already_present']} artifact(s) already in their releases")
|
|
164
|
+
if result.get("local_only"):
|
|
165
|
+
print("local-only (no cloud source — mirror from the local path):")
|
|
166
|
+
for name in result["local_only"]:
|
|
167
|
+
print(f" {name}")
|
|
168
|
+
return 0
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
172
|
+
parser = argparse.ArgumentParser(
|
|
173
|
+
prog="awrtifact",
|
|
174
|
+
description="Deliberately chunk artifacts into GitHub release assets "
|
|
175
|
+
"and fetch them back byte-verified.",
|
|
176
|
+
)
|
|
177
|
+
parser.add_argument("--version", action="version", version=__version__)
|
|
178
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
179
|
+
|
|
180
|
+
p = sub.add_parser("split", help="split a file into .partN slices + manifest")
|
|
181
|
+
p.add_argument("file")
|
|
182
|
+
p.add_argument("--part-size", type=int, default=1900000000)
|
|
183
|
+
p.add_argument("--out")
|
|
184
|
+
p.set_defaults(func=_cmd_split)
|
|
185
|
+
|
|
186
|
+
p = sub.add_parser("plan", help="which parts are missing from the release")
|
|
187
|
+
p.add_argument("manifest")
|
|
188
|
+
p.add_argument("--repo", required=True)
|
|
189
|
+
p.add_argument("--release", required=True)
|
|
190
|
+
p.set_defaults(func=_cmd_plan)
|
|
191
|
+
|
|
192
|
+
p = sub.add_parser("upload", help="upload missing parts (resumable)")
|
|
193
|
+
p.add_argument("manifest")
|
|
194
|
+
p.add_argument("--repo", required=True)
|
|
195
|
+
p.add_argument("--release", required=True)
|
|
196
|
+
p.add_argument("--dir")
|
|
197
|
+
p.add_argument("--parallel", type=int, default=1)
|
|
198
|
+
p.add_argument("--create", action="store_true")
|
|
199
|
+
p.set_defaults(func=_cmd_upload)
|
|
200
|
+
|
|
201
|
+
p = sub.add_parser("verify", help="verify local parts against the manifest")
|
|
202
|
+
p.add_argument("manifest")
|
|
203
|
+
p.add_argument("--dir")
|
|
204
|
+
p.add_argument("--json", action="store_true")
|
|
205
|
+
p.set_defaults(func=_cmd_verify)
|
|
206
|
+
|
|
207
|
+
p = sub.add_parser("fetch", help="resumable, verified download")
|
|
208
|
+
p.add_argument("name")
|
|
209
|
+
p.add_argument("--url", required=True)
|
|
210
|
+
p.add_argument("--out", required=True)
|
|
211
|
+
p.add_argument("--expected", type=int)
|
|
212
|
+
p.add_argument("--lock")
|
|
213
|
+
p.add_argument("--verify-only", action="store_true")
|
|
214
|
+
p.set_defaults(func=_cmd_fetch)
|
|
215
|
+
|
|
216
|
+
p = sub.add_parser("serve-spec", help="generate the worker from the spec")
|
|
217
|
+
p.add_argument("spec")
|
|
218
|
+
p.add_argument("--emit-dir")
|
|
219
|
+
p.add_argument("--check", action="store_true")
|
|
220
|
+
p.set_defaults(func=_cmd_serve_spec)
|
|
221
|
+
|
|
222
|
+
p = sub.add_parser(
|
|
223
|
+
"mirror",
|
|
224
|
+
help="feed it a URL or file — it mirrors to GitHub seamlessly",
|
|
225
|
+
)
|
|
226
|
+
p.add_argument("source", help="Range-serving URL or a local file path")
|
|
227
|
+
p.add_argument("--release", required=True, help="release tag to upload into")
|
|
228
|
+
p.add_argument("--repo", default="Aitherium/aitherkvcache")
|
|
229
|
+
p.add_argument("--name", help="served asset name (default: URL basename / filename)")
|
|
230
|
+
p.add_argument("--total", type=int, help="declared size; fail loud on mismatch")
|
|
231
|
+
p.add_argument("--workflow", default=mirror_mod.DEFAULT_WORKFLOW)
|
|
232
|
+
p.add_argument("--parallel", type=int, default=4)
|
|
233
|
+
p.set_defaults(func=_cmd_mirror)
|
|
234
|
+
|
|
235
|
+
p = sub.add_parser("backup-catalog", help="dispatch mirror runs for gaps")
|
|
236
|
+
p.add_argument("spec")
|
|
237
|
+
p.add_argument("--workflow", default=backup_mod.DEFAULT_WORKFLOW)
|
|
238
|
+
p.add_argument("--dry-run", action="store_true")
|
|
239
|
+
p.set_defaults(func=_cmd_backup)
|
|
240
|
+
|
|
241
|
+
return parser
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def main(argv: list[str] | None = None) -> int:
|
|
245
|
+
args = build_parser().parse_args(argv)
|
|
246
|
+
try:
|
|
247
|
+
return args.func(args)
|
|
248
|
+
except KeyboardInterrupt:
|
|
249
|
+
return 130
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
if __name__ == "__main__":
|
|
253
|
+
sys.exit(main())
|
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""`awrtifact fetch` — resumable, size-verified, TOFU-hashed download.
|
|
2
|
+
|
|
3
|
+
Ported from the tenant fetch lane (`.DEPLOYMENT/templates/tenant-repo/models/
|
|
4
|
+
fetch_models.py`) because that script's lessons are the point:
|
|
5
|
+
|
|
6
|
+
1. RESUME. These are multi-GB streams. A restarted download continues from
|
|
7
|
+
the byte count already on disk via `Range: bytes=<have>-`, and a server
|
|
8
|
+
that answers non-206 to a resume restarts from byte 0 explicitly.
|
|
9
|
+
2. SIZE VERIFICATION. A truncated artifact is not a download error — the
|
|
10
|
+
loader reports a corrupt model, which reads like a bad quant rather than
|
|
11
|
+
a short file. Every download is checked against the expected size before
|
|
12
|
+
it is accepted.
|
|
13
|
+
3. STITCHED ASSETS ARE TRANSPARENT. Files over GitHub's 2 GiB cap live
|
|
14
|
+
upstream as `.partN` slices; the worker reassembles them behind the
|
|
15
|
+
original filename. Ask for the original name and it works.
|
|
16
|
+
4. TOFU SHA256. The first successful fetch records the digest into the
|
|
17
|
+
lockfile; every later fetch (or --verify-only) compares. That makes a
|
|
18
|
+
silently-changed mirror asset visible on the second machine. It is
|
|
19
|
+
trust-on-first-use, not a signed digest, and the output says so.
|
|
20
|
+
|
|
21
|
+
MAX_STALLS is the real stop condition: attempts that make no progress.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import hashlib
|
|
27
|
+
import json
|
|
28
|
+
import urllib.error
|
|
29
|
+
import urllib.request
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
CHUNK = 8 * 1024 * 1024
|
|
33
|
+
MAX_ATTEMPTS = 200
|
|
34
|
+
MAX_STALLS = 4
|
|
35
|
+
UA = "awrtifact-fetch/1"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class FetchError(RuntimeError):
|
|
39
|
+
"""A download failed honestly (not a silent truncation)."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _resume_headers(have: int) -> dict[str, str]:
|
|
43
|
+
return {"Range": f"bytes={have}-", "User-Agent": UA} if have else {"User-Agent": UA}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _download_once(url: str, dest: Path, expected: int | None) -> int:
|
|
47
|
+
"""One attempt; returns bytes present on disk after the attempt."""
|
|
48
|
+
have = dest.stat().st_size if dest.exists() else 0
|
|
49
|
+
if expected is not None and have > expected:
|
|
50
|
+
raise FetchError(
|
|
51
|
+
f"on-disk {dest.name} is {have} bytes, larger than expected "
|
|
52
|
+
f"{expected} — remove it or fetch elsewhere"
|
|
53
|
+
)
|
|
54
|
+
headers = _resume_headers(have)
|
|
55
|
+
req = urllib.request.Request(url, headers=headers)
|
|
56
|
+
try:
|
|
57
|
+
with urllib.request.urlopen(req, timeout=60) as resp: # noqa: S310 — https checked by caller
|
|
58
|
+
status = getattr(resp, "status", 200)
|
|
59
|
+
if have and status != 206:
|
|
60
|
+
# Server ignored the resume; start over rather than append.
|
|
61
|
+
have = 0
|
|
62
|
+
if have:
|
|
63
|
+
with open(dest, "ab") as f:
|
|
64
|
+
while True:
|
|
65
|
+
chunk = resp.read(CHUNK)
|
|
66
|
+
if not chunk:
|
|
67
|
+
break
|
|
68
|
+
f.write(chunk)
|
|
69
|
+
else:
|
|
70
|
+
with open(dest, "wb") as f:
|
|
71
|
+
while True:
|
|
72
|
+
chunk = resp.read(CHUNK)
|
|
73
|
+
if not chunk:
|
|
74
|
+
break
|
|
75
|
+
f.write(chunk)
|
|
76
|
+
except urllib.error.HTTPError as exc:
|
|
77
|
+
if exc.code in (416,):
|
|
78
|
+
# Range not satisfiable: the server has more than we think.
|
|
79
|
+
raise FetchError(f"server 416 — {dest.name} may be complete here "
|
|
80
|
+
f"and short at the mirror") from exc
|
|
81
|
+
raise FetchError(f"HTTP {exc.code} fetching {dest.name}") from exc
|
|
82
|
+
except urllib.error.URLError as exc:
|
|
83
|
+
raise FetchError(f"network error fetching {dest.name}: {exc.reason}") from exc
|
|
84
|
+
return dest.stat().st_size
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _load_lock(path: Path) -> dict:
|
|
88
|
+
if not path.is_file():
|
|
89
|
+
return {}
|
|
90
|
+
try:
|
|
91
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
92
|
+
except (OSError, json.JSONDecodeError):
|
|
93
|
+
return {}
|
|
94
|
+
return raw if isinstance(raw, dict) else {}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _save_lock(path: Path, lock: dict) -> None:
|
|
98
|
+
path.write_text(json.dumps(lock, indent=2, sort_keys=True) + "\n", encoding="utf-8")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _sha256(path: Path) -> str:
|
|
102
|
+
h = hashlib.sha256()
|
|
103
|
+
with open(path, "rb") as f:
|
|
104
|
+
while True:
|
|
105
|
+
chunk = f.read(CHUNK)
|
|
106
|
+
if not chunk:
|
|
107
|
+
break
|
|
108
|
+
h.update(chunk)
|
|
109
|
+
return h.hexdigest()
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def fetch(
|
|
113
|
+
name: str,
|
|
114
|
+
url: str,
|
|
115
|
+
dest_dir: Path,
|
|
116
|
+
expected: int | None,
|
|
117
|
+
lockfile: Path | None = None,
|
|
118
|
+
verify_only: bool = False,
|
|
119
|
+
) -> dict:
|
|
120
|
+
"""Fetch `name` from `url` into `dest_dir`, size- and hash-verified.
|
|
121
|
+
|
|
122
|
+
`expected` is the manifest's total (or None to trust Content-Length).
|
|
123
|
+
Returns {"path", "bytes", "sha256", "status"} where status is one of
|
|
124
|
+
fetched / verified / up-to-date / sha-mismatch.
|
|
125
|
+
"""
|
|
126
|
+
dest_dir = Path(dest_dir)
|
|
127
|
+
dest_dir.mkdir(parents=True, exist_ok=True)
|
|
128
|
+
dest = dest_dir / name
|
|
129
|
+
lock_path = Path(lockfile) if lockfile else dest_dir / "awrtifact.lock.json"
|
|
130
|
+
lock = _load_lock(lock_path)
|
|
131
|
+
known = lock.get(name)
|
|
132
|
+
|
|
133
|
+
if verify_only and known and dest.is_file():
|
|
134
|
+
got = _sha256(dest)
|
|
135
|
+
if got == known["sha256"]:
|
|
136
|
+
return {"path": str(dest), "bytes": dest.stat().st_size,
|
|
137
|
+
"sha256": got, "status": "verified"}
|
|
138
|
+
return {"path": str(dest), "bytes": dest.stat().st_size,
|
|
139
|
+
"sha256": got, "status": "sha-mismatch"}
|
|
140
|
+
|
|
141
|
+
attempts = 0
|
|
142
|
+
stalls = 0
|
|
143
|
+
last = dest.stat().st_size if dest.exists() else 0
|
|
144
|
+
while attempts < MAX_ATTEMPTS and stalls < MAX_STALLS:
|
|
145
|
+
attempts += 1
|
|
146
|
+
now = _download_once(url, dest, expected)
|
|
147
|
+
if expected is not None and now > expected:
|
|
148
|
+
raise FetchError(
|
|
149
|
+
f"{name}: server delivered more than the expected {expected} "
|
|
150
|
+
f"bytes — the mirror is serving a different artifact"
|
|
151
|
+
)
|
|
152
|
+
if now == last:
|
|
153
|
+
stalls += 1
|
|
154
|
+
else:
|
|
155
|
+
stalls = 0
|
|
156
|
+
last = now
|
|
157
|
+
if expected is not None and now == expected:
|
|
158
|
+
break
|
|
159
|
+
if expected is not None and last != expected:
|
|
160
|
+
raise FetchError(
|
|
161
|
+
f"{name}: stopped at {last} of {expected} bytes — the bytes on "
|
|
162
|
+
f"disk are valid; re-run to resume from here"
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
got = _sha256(dest)
|
|
166
|
+
if known:
|
|
167
|
+
if got != known["sha256"]:
|
|
168
|
+
return {"path": str(dest), "bytes": last, "sha256": got,
|
|
169
|
+
"status": "sha-mismatch"}
|
|
170
|
+
lock[name]["bytes"] = last
|
|
171
|
+
else:
|
|
172
|
+
lock[name] = {"sha256": got, "bytes": last}
|
|
173
|
+
_save_lock(lock_path, lock)
|
|
174
|
+
return {"path": str(dest), "bytes": last, "sha256": got, "status": "fetched"}
|