awrtifact 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- awrtifact/__init__.py +12 -0
- awrtifact/__main__.py +4 -0
- awrtifact/backup.py +92 -0
- awrtifact/cli.py +253 -0
- awrtifact/fetch.py +174 -0
- awrtifact/gh.py +132 -0
- awrtifact/hashes.py +50 -0
- awrtifact/manifest.py +107 -0
- awrtifact/mirror.py +169 -0
- awrtifact/plan.py +40 -0
- awrtifact/serve_spec.py +124 -0
- awrtifact/spec.py +232 -0
- awrtifact/split.py +97 -0
- awrtifact/upload.py +83 -0
- awrtifact/verify.py +72 -0
- awrtifact/worker_template.py +280 -0
- awrtifact-0.1.0.dist-info/METADATA +82 -0
- awrtifact-0.1.0.dist-info/RECORD +21 -0
- awrtifact-0.1.0.dist-info/WHEEL +5 -0
- awrtifact-0.1.0.dist-info/entry_points.txt +2 -0
- awrtifact-0.1.0.dist-info/top_level.txt +1 -0
awrtifact/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""awrtifact — deliberately chunk artifacts into GitHub release assets.
|
|
2
|
+
|
|
3
|
+
Aither World Artifact is the productized aitherkvcache mirror lane: ANY artifact
|
|
4
|
+
(model weights, datasets, builds, backups) is split into `.partN` slices under
|
|
5
|
+
GitHub's 2 GiB per-asset cap, stored as a versioned GitHub release, served by a
|
|
6
|
+
generated Cloudflare Worker (CORS + HTTP Range + `.partN` stitching), and
|
|
7
|
+
fetched back byte-verified.
|
|
8
|
+
|
|
9
|
+
The core is stdlib-only; the spec-shaped commands need PyYAML (extra: `spec`).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
__version__ = "0.1.0"
|
awrtifact/__main__.py
ADDED
awrtifact/backup.py
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""`awrtifact backup-catalog` — dispatch the mirror workflow for gaps.
|
|
2
|
+
|
|
3
|
+
The point of the precaution: a spec with `source_url` per artifact is the
|
|
4
|
+
backup MANIFEST — every artifact listed has a named, size-checked origin and
|
|
5
|
+
a target release. This command turns the spec into workflow dispatches for
|
|
6
|
+
artifacts whose parts are NOT yet in the release (the resumable plan check),
|
|
7
|
+
so the mirror can be brought up deliberately and capped (the workflow's own
|
|
8
|
+
≤20-runner fan-out; the origin box's uplink is never involved).
|
|
9
|
+
|
|
10
|
+
`--dry-run` prints the dispatches; without it, dispatches via `gh workflow
|
|
11
|
+
run`. The workflow must exist in the store's repo — this checks, because a
|
|
12
|
+
dispatch to a missing workflow fails as a SILENCE (gh prints a warning and
|
|
13
|
+
exits 0 — measured class; `gh workflow run` on an unknown workflow does not
|
|
14
|
+
fail the caller).
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import sys
|
|
20
|
+
|
|
21
|
+
from . import gh
|
|
22
|
+
from . import spec as spec_mod
|
|
23
|
+
|
|
24
|
+
DEFAULT_WORKFLOW = "mirror-hf-to-release.yml"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def backup_catalog(
|
|
28
|
+
spec: dict,
|
|
29
|
+
workflow: str = DEFAULT_WORKFLOW,
|
|
30
|
+
dry_run: bool = False,
|
|
31
|
+
) -> dict:
|
|
32
|
+
"""Dispatch mirror runs for artifacts missing from their releases."""
|
|
33
|
+
repo = spec["store"]["repo"]
|
|
34
|
+
if not dry_run and not gh.workflow_exists(repo, workflow):
|
|
35
|
+
raise ValueError(
|
|
36
|
+
f"workflow {workflow} does not exist in {repo} — cannot dispatch; "
|
|
37
|
+
f"re-push the workflow first"
|
|
38
|
+
)
|
|
39
|
+
planned: list[dict] = []
|
|
40
|
+
local_only: list[str] = []
|
|
41
|
+
for art in spec.get("artifacts") or []:
|
|
42
|
+
name = art["name"]
|
|
43
|
+
if not art.get("source_url"):
|
|
44
|
+
# Local-only artifact: no cloud lane can mirror it; it is a
|
|
45
|
+
# `awrtifact mirror <local-path>` candidate. A None source_url
|
|
46
|
+
# reaching the dispatch would upload parts named "None.partN"
|
|
47
|
+
# (the measured 2026-08-27 incident) — skipped and reported.
|
|
48
|
+
local_only.append(name)
|
|
49
|
+
continue
|
|
50
|
+
release = spec_mod.artifact_release(spec, art)
|
|
51
|
+
existing = gh.release_assets(repo, release)
|
|
52
|
+
parts = spec_mod.chunked_map(spec).get(name)
|
|
53
|
+
if not parts:
|
|
54
|
+
# Whole-file artifact: present when the asset itself exists.
|
|
55
|
+
if name in existing:
|
|
56
|
+
continue
|
|
57
|
+
inputs = {
|
|
58
|
+
"hf_url": art["source_url"],
|
|
59
|
+
"name": name,
|
|
60
|
+
"total_bytes": str(art["total"]),
|
|
61
|
+
"release": release,
|
|
62
|
+
}
|
|
63
|
+
planned.append({"artifact": name, "inputs": inputs})
|
|
64
|
+
continue
|
|
65
|
+
missing = [p["name"] for p in parts["parts"] if p["name"] not in existing]
|
|
66
|
+
if not missing:
|
|
67
|
+
continue
|
|
68
|
+
inputs = {
|
|
69
|
+
"hf_url": art["source_url"],
|
|
70
|
+
"name": name,
|
|
71
|
+
"total_bytes": str(art["total"]),
|
|
72
|
+
"release": release,
|
|
73
|
+
"part_size": str(art.get("part_size") or 1900000000),
|
|
74
|
+
}
|
|
75
|
+
planned.append(
|
|
76
|
+
{"artifact": name, "inputs": inputs, "missing_parts": len(missing)}
|
|
77
|
+
)
|
|
78
|
+
if dry_run:
|
|
79
|
+
for item in planned:
|
|
80
|
+
print(f"dispatch {item['artifact']}: {item['inputs']}", file=sys.stderr)
|
|
81
|
+
else:
|
|
82
|
+
for item in planned:
|
|
83
|
+
gh.workflow_dispatch(repo, workflow, item["inputs"])
|
|
84
|
+
print(f"dispatched {item['artifact']} -> {workflow} in {repo}")
|
|
85
|
+
return {
|
|
86
|
+
"dispatched": len(planned),
|
|
87
|
+
"local_only": local_only,
|
|
88
|
+
"already_present": len(
|
|
89
|
+
[a for a in spec.get("artifacts") or [] if a["name"] not in
|
|
90
|
+
{i["artifact"] for i in planned} and a["name"] not in local_only]
|
|
91
|
+
),
|
|
92
|
+
}
|
awrtifact/cli.py
ADDED
|
@@ -0,0 +1,253 @@
|
|
|
1
|
+
"""awrtifact CLI — the deliberate chunk-and-release loop.
|
|
2
|
+
|
|
3
|
+
awrtifact split FILE [--part-size N] [--out DIR]
|
|
4
|
+
awrtifact plan MANIFEST --repo OWNER/REPO --release TAG
|
|
5
|
+
awrtifact upload MANIFEST --repo OWNER/REPO --release TAG [--dir DIR] [--parallel N] [--create]
|
|
6
|
+
awrtifact verify MANIFEST [--dir DIR]
|
|
7
|
+
awrtifact fetch NAME --url BASE --out DIR [--expected N] [--verify-only]
|
|
8
|
+
awrtifact serve-spec SPEC [--emit-dir DIR] [--check]
|
|
9
|
+
awrtifact backup-catalog SPEC [--workflow W] [--dry-run]
|
|
10
|
+
|
|
11
|
+
Exit codes: 0 ok · 1 operational failure (upload failed, fetch short, drift) ·
|
|
12
|
+
2 usage or data error (bad manifest/spec, missing dependency).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import sys
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from . import __version__, gh
|
|
22
|
+
from . import backup as backup_mod
|
|
23
|
+
from . import fetch as fetch_mod
|
|
24
|
+
from . import manifest as manifest_mod
|
|
25
|
+
from . import mirror as mirror_mod
|
|
26
|
+
from . import plan as plan_mod
|
|
27
|
+
from . import serve_spec as serve_spec_mod
|
|
28
|
+
from . import split as split_mod
|
|
29
|
+
from . import upload as upload_mod
|
|
30
|
+
from . import verify as verify_mod
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _die(message: str, code: int = 2) -> int:
|
|
34
|
+
print(f"awrtifact: {message}", file=sys.stderr)
|
|
35
|
+
return code
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _cmd_split(args: argparse.Namespace) -> int:
|
|
39
|
+
try:
|
|
40
|
+
m = split_mod.split_file(Path(args.file), args.part_size,
|
|
41
|
+
Path(args.out) if args.out else None)
|
|
42
|
+
except (ValueError, OSError) as exc:
|
|
43
|
+
return _die(str(exc))
|
|
44
|
+
out = Path(args.out) if args.out else Path(args.file).parent
|
|
45
|
+
manifest_path = out / "manifest.json"
|
|
46
|
+
manifest_mod.write(m, manifest_path)
|
|
47
|
+
print(f"split {m['name']}: {m['total']} bytes -> "
|
|
48
|
+
f"{len(m['parts'])} parts, manifest {manifest_path}")
|
|
49
|
+
return 0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _cmd_plan(args: argparse.Namespace) -> int:
|
|
53
|
+
try:
|
|
54
|
+
m = manifest_mod.load(Path(args.manifest))
|
|
55
|
+
planned = plan_mod.plan_parts(m, args.repo, args.release)
|
|
56
|
+
except (ValueError, OSError) as exc:
|
|
57
|
+
return _die(str(exc))
|
|
58
|
+
gh.print_json(planned)
|
|
59
|
+
return 0
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _cmd_upload(args: argparse.Namespace) -> int:
|
|
63
|
+
try:
|
|
64
|
+
m = manifest_mod.load(Path(args.manifest))
|
|
65
|
+
result = upload_mod.upload_manifest(
|
|
66
|
+
m, args.repo, args.release,
|
|
67
|
+
Path(args.dir) if args.dir else Path(args.manifest).parent,
|
|
68
|
+
parallel=args.parallel, create=args.create,
|
|
69
|
+
)
|
|
70
|
+
except (ValueError, OSError) as exc:
|
|
71
|
+
return _die(str(exc))
|
|
72
|
+
print(f"uploaded {len(result['uploaded'])} part(s); "
|
|
73
|
+
f"{len(result['skipped_present'])} already present")
|
|
74
|
+
if result["failed"]:
|
|
75
|
+
for f in result["failed"]:
|
|
76
|
+
print(f" FAILED: {f}", file=sys.stderr)
|
|
77
|
+
return 1
|
|
78
|
+
return 0
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _cmd_verify(args: argparse.Namespace) -> int:
|
|
82
|
+
try:
|
|
83
|
+
m = manifest_mod.load(Path(args.manifest))
|
|
84
|
+
report = verify_mod.verify_manifest(m, Path(args.dir) if args.dir
|
|
85
|
+
else Path(args.manifest).parent)
|
|
86
|
+
except (ValueError, OSError) as exc:
|
|
87
|
+
return _die(str(exc))
|
|
88
|
+
if args.json:
|
|
89
|
+
gh.print_json(report)
|
|
90
|
+
else:
|
|
91
|
+
for check in report["checks"]:
|
|
92
|
+
mark = "ok" if check["ok"] else "FAIL " + ", ".join(check["errors"])
|
|
93
|
+
print(f" {check['part']}: {mark}")
|
|
94
|
+
whole = report["whole"]
|
|
95
|
+
mark = "ok" if whole["ok"] else "FAIL " + ", ".join(whole["errors"])
|
|
96
|
+
print(f" whole-file: {mark}")
|
|
97
|
+
return 0 if report["ok"] else 1
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _cmd_fetch(args: argparse.Namespace) -> int:
|
|
101
|
+
try:
|
|
102
|
+
result = fetch_mod.fetch(
|
|
103
|
+
args.name, args.url, Path(args.out),
|
|
104
|
+
expected=args.expected, lockfile=Path(args.lock) if args.lock else None,
|
|
105
|
+
verify_only=args.verify_only,
|
|
106
|
+
)
|
|
107
|
+
except (ValueError, OSError, fetch_mod.FetchError) as exc:
|
|
108
|
+
return _die(str(exc), 1)
|
|
109
|
+
print(f"{result['status']}: {result['path']} ({result['bytes']} bytes, "
|
|
110
|
+
f"sha256 {result['sha256'][:16]}…)")
|
|
111
|
+
return 0 if result["status"] in ("fetched", "verified", "up-to-date") else 1
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _cmd_serve_spec(args: argparse.Namespace) -> int:
|
|
115
|
+
spec_path = Path(args.spec)
|
|
116
|
+
try:
|
|
117
|
+
dirs = serve_spec_mod.emit(spec_mod_load(spec_path), spec_path, check=args.check)
|
|
118
|
+
except ValueError as exc:
|
|
119
|
+
return _die(str(exc))
|
|
120
|
+
if args.check:
|
|
121
|
+
print(f"spec {spec_path}: generated workers are current")
|
|
122
|
+
else:
|
|
123
|
+
for d in dirs:
|
|
124
|
+
print(f"generated worker -> {d}")
|
|
125
|
+
return 0
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def spec_mod_load(path: Path) -> dict:
|
|
129
|
+
# Imported lazily so the core commands stay stdlib-only; the spec module
|
|
130
|
+
# raises a clear "pip install awrtifact[spec]" when PyYAML is absent.
|
|
131
|
+
from . import spec as spec_mod # noqa: PLC0415 — optional dependency
|
|
132
|
+
|
|
133
|
+
return spec_mod.load(path)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _cmd_mirror(args: argparse.Namespace) -> int:
|
|
137
|
+
try:
|
|
138
|
+
if mirror_mod.URL_RE.match(args.source):
|
|
139
|
+
result = mirror_mod.mirror_url(
|
|
140
|
+
args.source, args.name, args.release, args.repo,
|
|
141
|
+
total=args.total, workflow=args.workflow,
|
|
142
|
+
)
|
|
143
|
+
else:
|
|
144
|
+
result = mirror_mod.mirror_file(
|
|
145
|
+
Path(args.source), args.release, args.repo,
|
|
146
|
+
name=args.name, parallel=args.parallel,
|
|
147
|
+
)
|
|
148
|
+
except (ValueError, OSError) as exc:
|
|
149
|
+
return _die(str(exc))
|
|
150
|
+
print(f"mirror {result['lane']}: {result}")
|
|
151
|
+
return 0
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _cmd_backup(args: argparse.Namespace) -> int:
|
|
155
|
+
try:
|
|
156
|
+
spec = spec_mod_load(Path(args.spec))
|
|
157
|
+
result = backup_mod.backup_catalog(
|
|
158
|
+
spec, workflow=args.workflow, dry_run=args.dry_run
|
|
159
|
+
)
|
|
160
|
+
except (ValueError, OSError) as exc:
|
|
161
|
+
return _die(str(exc))
|
|
162
|
+
print(f"planned {result['dispatched']} dispatch(es); "
|
|
163
|
+
f"{result['already_present']} artifact(s) already in their releases")
|
|
164
|
+
if result.get("local_only"):
|
|
165
|
+
print("local-only (no cloud source — mirror from the local path):")
|
|
166
|
+
for name in result["local_only"]:
|
|
167
|
+
print(f" {name}")
|
|
168
|
+
return 0
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
172
|
+
parser = argparse.ArgumentParser(
|
|
173
|
+
prog="awrtifact",
|
|
174
|
+
description="Deliberately chunk artifacts into GitHub release assets "
|
|
175
|
+
"and fetch them back byte-verified.",
|
|
176
|
+
)
|
|
177
|
+
parser.add_argument("--version", action="version", version=__version__)
|
|
178
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
179
|
+
|
|
180
|
+
p = sub.add_parser("split", help="split a file into .partN slices + manifest")
|
|
181
|
+
p.add_argument("file")
|
|
182
|
+
p.add_argument("--part-size", type=int, default=1900000000)
|
|
183
|
+
p.add_argument("--out")
|
|
184
|
+
p.set_defaults(func=_cmd_split)
|
|
185
|
+
|
|
186
|
+
p = sub.add_parser("plan", help="which parts are missing from the release")
|
|
187
|
+
p.add_argument("manifest")
|
|
188
|
+
p.add_argument("--repo", required=True)
|
|
189
|
+
p.add_argument("--release", required=True)
|
|
190
|
+
p.set_defaults(func=_cmd_plan)
|
|
191
|
+
|
|
192
|
+
p = sub.add_parser("upload", help="upload missing parts (resumable)")
|
|
193
|
+
p.add_argument("manifest")
|
|
194
|
+
p.add_argument("--repo", required=True)
|
|
195
|
+
p.add_argument("--release", required=True)
|
|
196
|
+
p.add_argument("--dir")
|
|
197
|
+
p.add_argument("--parallel", type=int, default=1)
|
|
198
|
+
p.add_argument("--create", action="store_true")
|
|
199
|
+
p.set_defaults(func=_cmd_upload)
|
|
200
|
+
|
|
201
|
+
p = sub.add_parser("verify", help="verify local parts against the manifest")
|
|
202
|
+
p.add_argument("manifest")
|
|
203
|
+
p.add_argument("--dir")
|
|
204
|
+
p.add_argument("--json", action="store_true")
|
|
205
|
+
p.set_defaults(func=_cmd_verify)
|
|
206
|
+
|
|
207
|
+
p = sub.add_parser("fetch", help="resumable, verified download")
|
|
208
|
+
p.add_argument("name")
|
|
209
|
+
p.add_argument("--url", required=True)
|
|
210
|
+
p.add_argument("--out", required=True)
|
|
211
|
+
p.add_argument("--expected", type=int)
|
|
212
|
+
p.add_argument("--lock")
|
|
213
|
+
p.add_argument("--verify-only", action="store_true")
|
|
214
|
+
p.set_defaults(func=_cmd_fetch)
|
|
215
|
+
|
|
216
|
+
p = sub.add_parser("serve-spec", help="generate the worker from the spec")
|
|
217
|
+
p.add_argument("spec")
|
|
218
|
+
p.add_argument("--emit-dir")
|
|
219
|
+
p.add_argument("--check", action="store_true")
|
|
220
|
+
p.set_defaults(func=_cmd_serve_spec)
|
|
221
|
+
|
|
222
|
+
p = sub.add_parser(
|
|
223
|
+
"mirror",
|
|
224
|
+
help="feed it a URL or file — it mirrors to GitHub seamlessly",
|
|
225
|
+
)
|
|
226
|
+
p.add_argument("source", help="Range-serving URL or a local file path")
|
|
227
|
+
p.add_argument("--release", required=True, help="release tag to upload into")
|
|
228
|
+
p.add_argument("--repo", default="Aitherium/aitherkvcache")
|
|
229
|
+
p.add_argument("--name", help="served asset name (default: URL basename / filename)")
|
|
230
|
+
p.add_argument("--total", type=int, help="declared size; fail loud on mismatch")
|
|
231
|
+
p.add_argument("--workflow", default=mirror_mod.DEFAULT_WORKFLOW)
|
|
232
|
+
p.add_argument("--parallel", type=int, default=4)
|
|
233
|
+
p.set_defaults(func=_cmd_mirror)
|
|
234
|
+
|
|
235
|
+
p = sub.add_parser("backup-catalog", help="dispatch mirror runs for gaps")
|
|
236
|
+
p.add_argument("spec")
|
|
237
|
+
p.add_argument("--workflow", default=backup_mod.DEFAULT_WORKFLOW)
|
|
238
|
+
p.add_argument("--dry-run", action="store_true")
|
|
239
|
+
p.set_defaults(func=_cmd_backup)
|
|
240
|
+
|
|
241
|
+
return parser
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def main(argv: list[str] | None = None) -> int:
|
|
245
|
+
args = build_parser().parse_args(argv)
|
|
246
|
+
try:
|
|
247
|
+
return args.func(args)
|
|
248
|
+
except KeyboardInterrupt:
|
|
249
|
+
return 130
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
if __name__ == "__main__":
|
|
253
|
+
sys.exit(main())
|
awrtifact/fetch.py
ADDED
|
@@ -0,0 +1,174 @@
|
|
|
1
|
+
"""`awrtifact fetch` — resumable, size-verified, TOFU-hashed download.
|
|
2
|
+
|
|
3
|
+
Ported from the tenant fetch lane (`.DEPLOYMENT/templates/tenant-repo/models/
|
|
4
|
+
fetch_models.py`) because that script's lessons are the point:
|
|
5
|
+
|
|
6
|
+
1. RESUME. These are multi-GB streams. A restarted download continues from
|
|
7
|
+
the byte count already on disk via `Range: bytes=<have>-`, and a server
|
|
8
|
+
that answers non-206 to a resume restarts from byte 0 explicitly.
|
|
9
|
+
2. SIZE VERIFICATION. A truncated artifact is not a download error — the
|
|
10
|
+
loader reports a corrupt model, which reads like a bad quant rather than
|
|
11
|
+
a short file. Every download is checked against the expected size before
|
|
12
|
+
it is accepted.
|
|
13
|
+
3. STITCHED ASSETS ARE TRANSPARENT. Files over GitHub's 2 GiB cap live
|
|
14
|
+
upstream as `.partN` slices; the worker reassembles them behind the
|
|
15
|
+
original filename. Ask for the original name and it works.
|
|
16
|
+
4. TOFU SHA256. The first successful fetch records the digest into the
|
|
17
|
+
lockfile; every later fetch (or --verify-only) compares. That makes a
|
|
18
|
+
silently-changed mirror asset visible on the second machine. It is
|
|
19
|
+
trust-on-first-use, not a signed digest, and the output says so.
|
|
20
|
+
|
|
21
|
+
MAX_STALLS is the real stop condition: attempts that make no progress.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
import hashlib
|
|
27
|
+
import json
|
|
28
|
+
import urllib.error
|
|
29
|
+
import urllib.request
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
|
|
32
|
+
CHUNK = 8 * 1024 * 1024
|
|
33
|
+
MAX_ATTEMPTS = 200
|
|
34
|
+
MAX_STALLS = 4
|
|
35
|
+
UA = "awrtifact-fetch/1"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class FetchError(RuntimeError):
|
|
39
|
+
"""A download failed honestly (not a silent truncation)."""
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def _resume_headers(have: int) -> dict[str, str]:
|
|
43
|
+
return {"Range": f"bytes={have}-", "User-Agent": UA} if have else {"User-Agent": UA}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _download_once(url: str, dest: Path, expected: int | None) -> int:
|
|
47
|
+
"""One attempt; returns bytes present on disk after the attempt."""
|
|
48
|
+
have = dest.stat().st_size if dest.exists() else 0
|
|
49
|
+
if expected is not None and have > expected:
|
|
50
|
+
raise FetchError(
|
|
51
|
+
f"on-disk {dest.name} is {have} bytes, larger than expected "
|
|
52
|
+
f"{expected} — remove it or fetch elsewhere"
|
|
53
|
+
)
|
|
54
|
+
headers = _resume_headers(have)
|
|
55
|
+
req = urllib.request.Request(url, headers=headers)
|
|
56
|
+
try:
|
|
57
|
+
with urllib.request.urlopen(req, timeout=60) as resp: # noqa: S310 — https checked by caller
|
|
58
|
+
status = getattr(resp, "status", 200)
|
|
59
|
+
if have and status != 206:
|
|
60
|
+
# Server ignored the resume; start over rather than append.
|
|
61
|
+
have = 0
|
|
62
|
+
if have:
|
|
63
|
+
with open(dest, "ab") as f:
|
|
64
|
+
while True:
|
|
65
|
+
chunk = resp.read(CHUNK)
|
|
66
|
+
if not chunk:
|
|
67
|
+
break
|
|
68
|
+
f.write(chunk)
|
|
69
|
+
else:
|
|
70
|
+
with open(dest, "wb") as f:
|
|
71
|
+
while True:
|
|
72
|
+
chunk = resp.read(CHUNK)
|
|
73
|
+
if not chunk:
|
|
74
|
+
break
|
|
75
|
+
f.write(chunk)
|
|
76
|
+
except urllib.error.HTTPError as exc:
|
|
77
|
+
if exc.code in (416,):
|
|
78
|
+
# Range not satisfiable: the server has more than we think.
|
|
79
|
+
raise FetchError(f"server 416 — {dest.name} may be complete here "
|
|
80
|
+
f"and short at the mirror") from exc
|
|
81
|
+
raise FetchError(f"HTTP {exc.code} fetching {dest.name}") from exc
|
|
82
|
+
except urllib.error.URLError as exc:
|
|
83
|
+
raise FetchError(f"network error fetching {dest.name}: {exc.reason}") from exc
|
|
84
|
+
return dest.stat().st_size
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _load_lock(path: Path) -> dict:
|
|
88
|
+
if not path.is_file():
|
|
89
|
+
return {}
|
|
90
|
+
try:
|
|
91
|
+
raw = json.loads(path.read_text(encoding="utf-8"))
|
|
92
|
+
except (OSError, json.JSONDecodeError):
|
|
93
|
+
return {}
|
|
94
|
+
return raw if isinstance(raw, dict) else {}
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _save_lock(path: Path, lock: dict) -> None:
|
|
98
|
+
path.write_text(json.dumps(lock, indent=2, sort_keys=True) + "\n", encoding="utf-8")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _sha256(path: Path) -> str:
|
|
102
|
+
h = hashlib.sha256()
|
|
103
|
+
with open(path, "rb") as f:
|
|
104
|
+
while True:
|
|
105
|
+
chunk = f.read(CHUNK)
|
|
106
|
+
if not chunk:
|
|
107
|
+
break
|
|
108
|
+
h.update(chunk)
|
|
109
|
+
return h.hexdigest()
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def fetch(
|
|
113
|
+
name: str,
|
|
114
|
+
url: str,
|
|
115
|
+
dest_dir: Path,
|
|
116
|
+
expected: int | None,
|
|
117
|
+
lockfile: Path | None = None,
|
|
118
|
+
verify_only: bool = False,
|
|
119
|
+
) -> dict:
|
|
120
|
+
"""Fetch `name` from `url` into `dest_dir`, size- and hash-verified.
|
|
121
|
+
|
|
122
|
+
`expected` is the manifest's total (or None to trust Content-Length).
|
|
123
|
+
Returns {"path", "bytes", "sha256", "status"} where status is one of
|
|
124
|
+
fetched / verified / up-to-date / sha-mismatch.
|
|
125
|
+
"""
|
|
126
|
+
dest_dir = Path(dest_dir)
|
|
127
|
+
dest_dir.mkdir(parents=True, exist_ok=True)
|
|
128
|
+
dest = dest_dir / name
|
|
129
|
+
lock_path = Path(lockfile) if lockfile else dest_dir / "awrtifact.lock.json"
|
|
130
|
+
lock = _load_lock(lock_path)
|
|
131
|
+
known = lock.get(name)
|
|
132
|
+
|
|
133
|
+
if verify_only and known and dest.is_file():
|
|
134
|
+
got = _sha256(dest)
|
|
135
|
+
if got == known["sha256"]:
|
|
136
|
+
return {"path": str(dest), "bytes": dest.stat().st_size,
|
|
137
|
+
"sha256": got, "status": "verified"}
|
|
138
|
+
return {"path": str(dest), "bytes": dest.stat().st_size,
|
|
139
|
+
"sha256": got, "status": "sha-mismatch"}
|
|
140
|
+
|
|
141
|
+
attempts = 0
|
|
142
|
+
stalls = 0
|
|
143
|
+
last = dest.stat().st_size if dest.exists() else 0
|
|
144
|
+
while attempts < MAX_ATTEMPTS and stalls < MAX_STALLS:
|
|
145
|
+
attempts += 1
|
|
146
|
+
now = _download_once(url, dest, expected)
|
|
147
|
+
if expected is not None and now > expected:
|
|
148
|
+
raise FetchError(
|
|
149
|
+
f"{name}: server delivered more than the expected {expected} "
|
|
150
|
+
f"bytes — the mirror is serving a different artifact"
|
|
151
|
+
)
|
|
152
|
+
if now == last:
|
|
153
|
+
stalls += 1
|
|
154
|
+
else:
|
|
155
|
+
stalls = 0
|
|
156
|
+
last = now
|
|
157
|
+
if expected is not None and now == expected:
|
|
158
|
+
break
|
|
159
|
+
if expected is not None and last != expected:
|
|
160
|
+
raise FetchError(
|
|
161
|
+
f"{name}: stopped at {last} of {expected} bytes — the bytes on "
|
|
162
|
+
f"disk are valid; re-run to resume from here"
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
got = _sha256(dest)
|
|
166
|
+
if known:
|
|
167
|
+
if got != known["sha256"]:
|
|
168
|
+
return {"path": str(dest), "bytes": last, "sha256": got,
|
|
169
|
+
"status": "sha-mismatch"}
|
|
170
|
+
lock[name]["bytes"] = last
|
|
171
|
+
else:
|
|
172
|
+
lock[name] = {"sha256": got, "bytes": last}
|
|
173
|
+
_save_lock(lock_path, lock)
|
|
174
|
+
return {"path": str(dest), "bytes": last, "sha256": got, "status": "fetched"}
|
awrtifact/gh.py
ADDED
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
"""`gh` CLI wrapper — the smallest shell around GitHub's release API.
|
|
2
|
+
|
|
3
|
+
Deliberately shells the ambient `gh` CLI (like the lanes it productizes —
|
|
4
|
+
`seed-q1-mirror.ps1`, `build_webml_cdn.mjs`, `mirror-hf-to-release.yml` all
|
|
5
|
+
use it): the operator's existing auth is the auth, no token handling here.
|
|
6
|
+
Every call is size-bounded (release asset lists) and text-decoded explicitly.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import subprocess
|
|
13
|
+
from typing import Sequence
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
class GhError(RuntimeError):
|
|
17
|
+
"""A gh invocation failed — message carries gh's stderr."""
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _run(args: Sequence[str]) -> subprocess.CompletedProcess:
|
|
21
|
+
return subprocess.run(
|
|
22
|
+
["gh", *args],
|
|
23
|
+
capture_output=True,
|
|
24
|
+
text=True,
|
|
25
|
+
encoding="utf-8",
|
|
26
|
+
errors="replace",
|
|
27
|
+
check=False,
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def release_assets(repo: str, release: str) -> set[str]:
|
|
32
|
+
"""Asset names currently in the release. Empty set on a missing release."""
|
|
33
|
+
proc = _run(
|
|
34
|
+
[
|
|
35
|
+
"release",
|
|
36
|
+
"view",
|
|
37
|
+
release,
|
|
38
|
+
"--repo",
|
|
39
|
+
repo,
|
|
40
|
+
"--json",
|
|
41
|
+
"assets",
|
|
42
|
+
"--jq",
|
|
43
|
+
".assets[].name",
|
|
44
|
+
]
|
|
45
|
+
)
|
|
46
|
+
if proc.returncode != 0:
|
|
47
|
+
return set()
|
|
48
|
+
names = {line.strip() for line in proc.stdout.splitlines() if line.strip()}
|
|
49
|
+
return names
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def release_exists(repo: str, release: str) -> bool:
|
|
53
|
+
proc = _run(["release", "view", release, "--repo", repo])
|
|
54
|
+
return proc.returncode == 0
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def create_release(repo: str, release: str, title: str, notes: str) -> None:
|
|
58
|
+
proc = _run(
|
|
59
|
+
[
|
|
60
|
+
"release",
|
|
61
|
+
"create",
|
|
62
|
+
release,
|
|
63
|
+
"--repo",
|
|
64
|
+
repo,
|
|
65
|
+
"--title",
|
|
66
|
+
title,
|
|
67
|
+
"--notes",
|
|
68
|
+
notes,
|
|
69
|
+
]
|
|
70
|
+
)
|
|
71
|
+
if proc.returncode != 0:
|
|
72
|
+
raise GhError(f"gh release create {release}: {proc.stderr.strip()}")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def upload(repo: str, release: str, path: str) -> None:
|
|
76
|
+
"""Upload one asset with --clobber (idempotent re-upload)."""
|
|
77
|
+
proc = _run(
|
|
78
|
+
["release", "upload", release, path, "--repo", repo, "--clobber"]
|
|
79
|
+
)
|
|
80
|
+
if proc.returncode != 0:
|
|
81
|
+
raise GhError(f"gh release upload {path}: {proc.stderr.strip()}")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def workflow_dispatch(
|
|
85
|
+
repo: str, workflow: str, inputs: dict[str, str]
|
|
86
|
+
) -> None:
|
|
87
|
+
"""Fire a workflow_dispatch with string inputs (mirror lane)."""
|
|
88
|
+
args = ["workflow", "run", workflow, "--repo", repo]
|
|
89
|
+
for key, value in sorted(inputs.items()):
|
|
90
|
+
args.extend(["-f", f"{key}={value}"])
|
|
91
|
+
proc = _run(args)
|
|
92
|
+
if proc.returncode != 0:
|
|
93
|
+
raise GhError(f"gh workflow run {workflow}: {proc.stderr.strip()}")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def asset_sizes(repo: str, release: str) -> dict[str, int]:
|
|
97
|
+
"""name -> size for the release's assets (verify/plan use this)."""
|
|
98
|
+
proc = _run(
|
|
99
|
+
[
|
|
100
|
+
"release",
|
|
101
|
+
"view",
|
|
102
|
+
release,
|
|
103
|
+
"--repo",
|
|
104
|
+
repo,
|
|
105
|
+
"--json",
|
|
106
|
+
"assets",
|
|
107
|
+
"--jq",
|
|
108
|
+
'.assets[] | "\\(.name)\t\\(.size)"',
|
|
109
|
+
]
|
|
110
|
+
)
|
|
111
|
+
if proc.returncode != 0:
|
|
112
|
+
return {}
|
|
113
|
+
out: dict[str, int] = {}
|
|
114
|
+
for line in proc.stdout.splitlines():
|
|
115
|
+
if "\t" not in line:
|
|
116
|
+
continue
|
|
117
|
+
name, size = line.split("\t", 1)
|
|
118
|
+
try:
|
|
119
|
+
out[name] = int(size)
|
|
120
|
+
except ValueError:
|
|
121
|
+
continue
|
|
122
|
+
return out
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def workflow_exists(repo: str, workflow: str) -> bool:
|
|
126
|
+
"""Is the named workflow present in the repo (backup-catalog gate)?"""
|
|
127
|
+
proc = _run(["workflow", "view", workflow, "--repo", repo])
|
|
128
|
+
return proc.returncode == 0
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def print_json(payload: dict) -> None:
|
|
132
|
+
print(json.dumps(payload, indent=2, sort_keys=True))
|