glc-loader 1.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. glc_loader/NOTICE +4 -0
  2. glc_loader/__init__.py +190 -0
  3. glc_loader/__main__.py +446 -0
  4. glc_loader/_inspect.py +786 -0
  5. glc_loader/artifact.py +517 -0
  6. glc_loader/container.py +378 -0
  7. glc_loader/loader.py +611 -0
  8. glc_loader/metal/__init__.py +25 -0
  9. glc_loader/metal/ple_gather_mlx.py +424 -0
  10. glc_loader/metal/tbe_backbone_batched_mlx.py +513 -0
  11. glc_loader/metal/tbe_decode_mlx.py +409 -0
  12. glc_loader/metal/tbe_decode_mlx_v2.py +245 -0
  13. glc_loader/metal/tbe_decode_mlx_v3.py +313 -0
  14. glc_loader/metal/tbe_decode_mlx_v4.py +174 -0
  15. glc_loader/metal/tbe_gemv_fused_lut_mlx.py +682 -0
  16. glc_loader/metal/tbe_gemv_fused_mlx.py +819 -0
  17. glc_loader/metal/tbe_linear_mlx.py +94 -0
  18. glc_loader/metal/tbe_linear_mlx_v2.py +62 -0
  19. glc_loader/metal/tbe_linear_mlx_v3.py +64 -0
  20. glc_loader/metal/tbe_linear_mlx_v4.py +73 -0
  21. glc_loader/metal/tbe_mlx_model.py +448 -0
  22. glc_loader/metal/tbe_moe_batched_mlx.py +565 -0
  23. glc_loader/modules.py +317 -0
  24. glc_loader/tbe_anyrank.py +116 -0
  25. glc_loader/tbe_artifact.py +824 -0
  26. glc_loader/tbe_container.py +446 -0
  27. glc_loader/tbe_device_map.py +447 -0
  28. glc_loader/tbe_gather.py +270 -0
  29. glc_loader/tbe_mma.py +475 -0
  30. glc_loader/tbe_mma_kernel.cu +704 -0
  31. glc_loader/tbe_mma_kernels.py +125 -0
  32. glc_loader/tbe_modules.py +401 -0
  33. glc_loader/tbe_serve_portable.py +181 -0
  34. glc_loader/tbe_serving.py +835 -0
  35. glc_loader/tbe_stream_loader.py +772 -0
  36. glc_loader-1.1.0.dist-info/LICENSE +202 -0
  37. glc_loader-1.1.0.dist-info/METADATA +292 -0
  38. glc_loader-1.1.0.dist-info/NOTICE +4 -0
  39. glc_loader-1.1.0.dist-info/RECORD +76 -0
  40. glc_loader-1.1.0.dist-info/WHEEL +5 -0
  41. glc_loader-1.1.0.dist-info/entry_points.txt +3 -0
  42. glc_loader-1.1.0.dist-info/top_level.txt +2 -0
  43. glc_serve/__init__.py +29 -0
  44. glc_serve/__main__.py +4 -0
  45. glc_serve/bench.py +713 -0
  46. glc_serve/bundle.py +756 -0
  47. glc_serve/engine.py +633 -0
  48. glc_serve/espec.py +352 -0
  49. glc_serve/fastdec.py +640 -0
  50. glc_serve/fastdec_kernels.cu +512 -0
  51. glc_serve/fastserve.py +409 -0
  52. glc_serve/gate.py +342 -0
  53. glc_serve/loader.py +708 -0
  54. glc_serve/memcap.py +137 -0
  55. glc_serve/miv_gemv.py +890 -0
  56. glc_serve/miv_tbe.py +396 -0
  57. glc_serve/miv_tbe_csrc/miv_tbe.h +65 -0
  58. glc_serve/miv_tbe_csrc/miv_tbe_api.cu +76 -0
  59. glc_serve/miv_tbe_csrc/miv_tbe_kernels.cuh +497 -0
  60. glc_serve/miv_tbe_csrc/miv_tbe_m1.cu +3 -0
  61. glc_serve/miv_tbe_csrc/miv_tbe_m2.cu +3 -0
  62. glc_serve/miv_tbe_csrc/miv_tbe_m3.cu +3 -0
  63. glc_serve/miv_tbe_csrc/miv_tbe_m4.cu +3 -0
  64. glc_serve/miv_tbe_csrc/miv_tbe_m5.cu +3 -0
  65. glc_serve/miv_tbe_csrc/miv_tbe_m6.cu +3 -0
  66. glc_serve/miv_tbe_csrc/miv_tbe_m7.cu +3 -0
  67. glc_serve/miv_tbe_csrc/miv_tbe_m8.cu +3 -0
  68. glc_serve/miv_tbe_csrc/miv_tbe_torch.cpp +91 -0
  69. glc_serve/modules.py +535 -0
  70. glc_serve/mtp.py +337 -0
  71. glc_serve/pack.py +431 -0
  72. glc_serve/q8serve.py +494 -0
  73. glc_serve/server.py +500 -0
  74. glc_serve/tbe_desc.py +278 -0
  75. glc_serve/tunes/fastdec_sm120.json +133 -0
  76. glc_serve/vendor.py +76 -0
glc_loader/NOTICE ADDED
@@ -0,0 +1,4 @@
1
+ GeoRefine codec software
2
+ Copyright 2026 Tsotchke Corporation
3
+
4
+ This package is licensed under the Apache License, Version 2.0.
glc_loader/__init__.py ADDED
@@ -0,0 +1,190 @@
1
+ """``glc_loader`` -- read a GLC-RELEASE artifact with no build repo present.
2
+
3
+ import sys; sys.path.insert(0, "/path/to/artifact")
4
+ from glc_loader import load_model
5
+ model, tokenizer, receipt = load_model("/path/to/artifact", device="cuda")
6
+
7
+ or, without writing any code at all:
8
+
9
+ python -m glc_loader generate --artifact /path/to/artifact --prompt "hello"
10
+ python -m glc_loader expand --artifact /path/to/artifact --out ./dense
11
+
12
+ Third-party requirements: ``torch`` and ``safetensors`` for the container;
13
+ ``transformers`` additionally for :func:`load_model` (it supplies the model
14
+ class and the tokenizer, not the weights). ``triton`` is optional and only
15
+ selects a faster backend. Nothing here imports the repository that built the
16
+ artifact -- that is the point of the format, and it is enforced by a test.
17
+ """
18
+ from .artifact import (
19
+ Artifact,
20
+ FORMAT,
21
+ GLCArtifactError,
22
+ expand,
23
+ open_artifact,
24
+ verify,
25
+ )
26
+ from .container import (
27
+ CONTAINER_MAGIC,
28
+ FWP1Error,
29
+ FWP1Tensor,
30
+ decode_fwp1,
31
+ decode_fwp1_rows,
32
+ encode_fwp1,
33
+ fwp1_certify,
34
+ )
35
+ from .loader import (
36
+ CertificateError,
37
+ assert_certificate_usable,
38
+ load_model,
39
+ load_tokenizer,
40
+ )
41
+ from .modules import (
42
+ BACKENDS,
43
+ GLCBackendError,
44
+ GLCEmbedding,
45
+ GLCLinear,
46
+ GLCTiedLMHead,
47
+ )
48
+ from .tbe_artifact import (
49
+ ARTIFACT_FORMAT,
50
+ COMPRESSION_INFO_FILENAME,
51
+ COMPRESSION_SCHEMA_VERSION,
52
+ STANDALONE_LOADER,
53
+ TBEArtifact,
54
+ TBEArtifactError,
55
+ build_meta_skeleton_from_artifact,
56
+ detect_artifact_format,
57
+ is_legacy_v1_container,
58
+ iter_artifact_tensors,
59
+ load_standalone,
60
+ open_tbe_artifact,
61
+ verify_sha256sums,
62
+ verify_standalone_bit_exact,
63
+ write_sha256sums,
64
+ )
65
+ from .tbe_container import (
66
+ TBEError,
67
+ TBETensor,
68
+ decode_tbe,
69
+ encode_tbe,
70
+ tbe_certify,
71
+ )
72
+ from .tbe_serve_portable import PortableTBELinear, PortableTBEError, load_compressed_transformers
73
+ from .tbe_device_map import (
74
+ STRATEGY_BALANCED,
75
+ STRATEGY_SEQUENTIAL,
76
+ TBEDeviceMapError,
77
+ devices_in_map,
78
+ group_key,
79
+ manifest_group_bytes,
80
+ plan_device_map,
81
+ plan_device_map_from_manifest,
82
+ resolve_device,
83
+ single_device_map,
84
+ validate_device_map,
85
+ )
86
+ from .tbe_mma import (
87
+ TBEDevice,
88
+ TBEMMAError,
89
+ resolve_tbe_mma_arch,
90
+ tbe_mma_available,
91
+ upload_tbe,
92
+ )
93
+ from .tbe_modules import (
94
+ GLCTBEExpertBank,
95
+ GLCTBELinear,
96
+ TBEBackendError,
97
+ TBEPoolRegistry,
98
+ TBETransientPool,
99
+ )
100
+ from .tbe_serving import (
101
+ TBEServingError,
102
+ enable_hf_tbe_serving,
103
+ place_uncoded_tensors,
104
+ )
105
+ from .tbe_stream_loader import (
106
+ TBEStreamLoadError,
107
+ build_meta_skeleton,
108
+ iter_safetensors_shards,
109
+ stream_load_tbe_model,
110
+ stream_load_tbe_model_standalone,
111
+ stream_place_tensors,
112
+ )
113
+
114
+ __version__ = "1.1.0"
115
+
116
+ __all__ = [
117
+ "ARTIFACT_FORMAT",
118
+ "Artifact",
119
+ "BACKENDS",
120
+ "COMPRESSION_INFO_FILENAME",
121
+ "COMPRESSION_SCHEMA_VERSION",
122
+ "CONTAINER_MAGIC",
123
+ "CertificateError",
124
+ "FORMAT",
125
+ "FWP1Error",
126
+ "FWP1Tensor",
127
+ "GLCArtifactError",
128
+ "GLCBackendError",
129
+ "GLCEmbedding",
130
+ "GLCLinear",
131
+ "GLCTBEExpertBank",
132
+ "GLCTBELinear",
133
+ "GLCTiedLMHead",
134
+ "STANDALONE_LOADER",
135
+ "STRATEGY_BALANCED",
136
+ "STRATEGY_SEQUENTIAL",
137
+ "TBEArtifact",
138
+ "TBEArtifactError",
139
+ "TBEBackendError",
140
+ "TBEDevice",
141
+ "TBEDeviceMapError",
142
+ "TBEError",
143
+ "TBEMMAError",
144
+ "TBEPoolRegistry",
145
+ "TBEServingError",
146
+ "TBEStreamLoadError",
147
+ "TBETensor",
148
+ "TBETransientPool",
149
+ "__version__",
150
+ "assert_certificate_usable",
151
+ "build_meta_skeleton",
152
+ "build_meta_skeleton_from_artifact",
153
+ "decode_fwp1",
154
+ "decode_fwp1_rows",
155
+ "decode_tbe",
156
+ "detect_artifact_format",
157
+ "devices_in_map",
158
+ "enable_hf_tbe_serving",
159
+ "encode_fwp1",
160
+ "encode_tbe",
161
+ "expand",
162
+ "fwp1_certify",
163
+ "group_key",
164
+ "is_legacy_v1_container",
165
+ "iter_artifact_tensors",
166
+ "iter_safetensors_shards",
167
+ "load_model",
168
+ "load_standalone",
169
+ "load_tokenizer",
170
+ "manifest_group_bytes",
171
+ "open_artifact",
172
+ "open_tbe_artifact",
173
+ "place_uncoded_tensors",
174
+ "plan_device_map",
175
+ "plan_device_map_from_manifest",
176
+ "resolve_device",
177
+ "resolve_tbe_mma_arch",
178
+ "single_device_map",
179
+ "stream_load_tbe_model",
180
+ "stream_load_tbe_model_standalone",
181
+ "stream_place_tensors",
182
+ "tbe_certify",
183
+ "tbe_mma_available",
184
+ "upload_tbe",
185
+ "validate_device_map",
186
+ "verify",
187
+ "verify_sha256sums",
188
+ "verify_standalone_bit_exact",
189
+ "write_sha256sums",
190
+ ]
glc_loader/__main__.py ADDED
@@ -0,0 +1,446 @@
1
+ """Client-side command line: ``glc-loader <command>`` / ``python -m glc_loader``.
2
+
3
+ Runs from inside an artifact directory, or against one by path, with nothing
4
+ installed but this package and its three dependencies (torch, safetensors,
5
+ transformers). Nothing here imports the repository that built the artifact.
6
+
7
+ glc-loader info <artifact-dir> what it is, at which scopes, and what
8
+ is NOT certified
9
+ glc-loader verify <artifact-dir> integrity, with honest exit codes
10
+ glc-loader load <artifact-dir> smoke-load and generate a few tokens
11
+
12
+ The older ``--artifact``-flag spelling still works everywhere, so the
13
+ instructions printed inside already-released artifacts keep running.
14
+
15
+ EXIT CODES (verify). 0 verified; 1 an integrity check FAILED; 2 malformed,
16
+ unreadable, or not a GeoRefine container; 3 verified but the artifact
17
+ EXPANDS; 4 the format is recognised but this directory cannot be verified
18
+ from what it ships. 4 exists because the alternative is a misleading 2: a
19
+ v1 TBE transcode directory is a correct artifact with no self-contained
20
+ verifier, and ``experiments.georefine.lossless_audit`` on one exits 2 ("no
21
+ blobs/") purely because it is the wrong auditor for that codec.
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import argparse
26
+ import json
27
+ from pathlib import Path
28
+ import sys
29
+ import time
30
+ from typing import Any, Dict, Optional, Sequence
31
+
32
+ if __package__ in (None, ""): # pragma: no cover - direct-script execution
33
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
34
+ __package__ = "glc_loader"
35
+
36
+ from ._inspect import (
37
+ EXIT_MALFORMED,
38
+ EXIT_OK,
39
+ EXIT_UNVERIFIABLE,
40
+ KIND_GLC_RELEASE_V1,
41
+ KIND_TBE_V1_TRANSCODE,
42
+ KIND_TBE_V2,
43
+ InspectError,
44
+ certificate_summary,
45
+ detect,
46
+ not_certified,
47
+ scope_ratios,
48
+ source_identity,
49
+ verify_artifact,
50
+ )
51
+ from .artifact import GLCArtifactError, expand, open_artifact
52
+
53
+
54
+ def _default_artifact() -> str:
55
+ """The artifact this package was vendored into, if it was vendored."""
56
+ here = Path(__file__).resolve().parent.parent
57
+ return str(here)
58
+
59
+
60
+ def _target(args: argparse.Namespace) -> str:
61
+ """Positional path wins; ``--artifact`` is the compatible spelling."""
62
+ return getattr(args, "artifact_pos", None) or args.artifact
63
+
64
+
65
+ def _emit(obj: Dict[str, Any]) -> None:
66
+ print(json.dumps(obj, indent=2, sort_keys=True, default=str))
67
+
68
+
69
+ def _fail(reason: str, detail: str, code: int = EXIT_MALFORMED) -> int:
70
+ print(
71
+ json.dumps(
72
+ {"status": "FAIL", "reason": reason, "detail": detail,
73
+ "exit_code": code},
74
+ indent=2, sort_keys=True,
75
+ ),
76
+ file=sys.stderr,
77
+ )
78
+ return code
79
+
80
+
81
+ # ---------------------------------------------------------------------------
82
+ # info
83
+ # ---------------------------------------------------------------------------
84
+ def cmd_info(args: argparse.Namespace) -> int:
85
+ det = detect(_target(args))
86
+ out: Dict[str, Any] = {
87
+ "artifact": str(det.root),
88
+ "format": det.kind,
89
+ "source": source_identity(det),
90
+ "ratios": scope_ratios(det),
91
+ "ratio_note": (
92
+ "Three scopes, never conflated. stored = bytes on disk. "
93
+ "served_resident = bytes the device holds for weights. "
94
+ "whole_process = measured peak process VRAM, dense vs coded, and "
95
+ "it is the only one that includes activations, the KV cache and "
96
+ "allocator slack. A ratio reported at one scope says nothing "
97
+ "about the other two."
98
+ ),
99
+ "certificate": certificate_summary(det),
100
+ "not_certified": not_certified(det),
101
+ }
102
+ if det.kind == KIND_GLC_RELEASE_V1:
103
+ art = open_artifact(str(det.root), require_certificate=False)
104
+ out["summary"] = art.summary()
105
+ out["container"] = det.manifest.get("container")
106
+ out["effective_config"] = det.manifest.get("effective_config")
107
+ elif det.kind in (KIND_TBE_V2, KIND_TBE_V1_TRANSCODE):
108
+ out["container"] = det.manifest.get("container")
109
+ out["summary"] = det.manifest.get("summary")
110
+ out["escape_band_pct"] = det.manifest.get("escape_band_pct")
111
+ out["escape_over_band"] = [
112
+ o.get("name") for o in (det.manifest.get("escape_over_band") or [])
113
+ ]
114
+ if det.compression_info:
115
+ out["compression_info"] = {
116
+ k: v for k, v in det.compression_info.items()
117
+ if k != "sidecars"
118
+ }
119
+ if not det.understood:
120
+ out["status"] = "REFUSED"
121
+ out["reason"] = det.reason
122
+ out["detail"] = det.detail
123
+ _emit(out)
124
+ return EXIT_UNVERIFIABLE if det.kind == KIND_TBE_V1_TRANSCODE \
125
+ else EXIT_MALFORMED
126
+ _emit(out)
127
+ return EXIT_OK
128
+
129
+
130
+ # ---------------------------------------------------------------------------
131
+ # verify
132
+ # ---------------------------------------------------------------------------
133
+ def cmd_verify(args: argparse.Namespace) -> int:
134
+ det = detect(_target(args))
135
+ res = verify_artifact(
136
+ det, deep=args.deep, progress=args.progress, limit=args.limit,
137
+ )
138
+ res["certificate"] = certificate_summary(det)
139
+ res["not_certified"] = not_certified(det)
140
+ stream = sys.stdout if res["exit_code"] == EXIT_OK else sys.stderr
141
+ print(json.dumps(res, indent=2, sort_keys=True, default=str), file=stream)
142
+ return int(res["exit_code"])
143
+
144
+
145
+ # ---------------------------------------------------------------------------
146
+ # load
147
+ # ---------------------------------------------------------------------------
148
+ def _generate(model, tok, prompt: str, *, device: str, max_new_tokens: int,
149
+ chat: bool) -> Dict[str, Any]:
150
+ import torch
151
+
152
+ text = prompt
153
+ if chat and getattr(tok, "chat_template", None):
154
+ text = tok.apply_chat_template(
155
+ [{"role": "user", "content": prompt}],
156
+ tokenize=False, add_generation_prompt=True,
157
+ )
158
+ enc = tok(text, return_tensors="pt").to(device)
159
+ t0 = time.perf_counter()
160
+ with torch.no_grad():
161
+ out = model.generate(
162
+ **enc,
163
+ max_new_tokens=max_new_tokens,
164
+ do_sample=False,
165
+ pad_token_id=tok.pad_token_id or tok.eos_token_id,
166
+ )
167
+ dt = time.perf_counter() - t0
168
+ n_new = int(out.shape[-1] - enc["input_ids"].shape[-1])
169
+ return {
170
+ "prompt": prompt,
171
+ "completion": tok.decode(
172
+ out[0][enc["input_ids"].shape[-1]:], skip_special_tokens=True,
173
+ ),
174
+ "new_tokens": n_new,
175
+ "seconds": dt,
176
+ "tokens_per_second": (n_new / dt) if dt > 0 else None,
177
+ "throughput_note": (
178
+ "a single greedy run on whatever device this is; NOT a benchmark. "
179
+ "Decode speed depends on the engine, not the format. The earlier "
180
+ "decode path measured 0.94x dense on a Blackwell RTX PRO 6000 and "
181
+ "0.57-0.62x on an A100 (single runs, +/-0.05 spread). The GeoRefine "
182
+ "engine (whole-step CUDA graph + fused MIV-TBE kernel) measured "
183
+ "37.84 tok/s batch-1 AR on the Qwen3.8-27B TBE bundle, vs 28.29 for "
184
+ "the bf16 parent in the same engine and 27.4-27.6 for llama.cpp bf16 "
185
+ "on the same RTX PRO 6000, at 49.9-50.0 GB vs 65.4-72.8 GB peak VRAM "
186
+ "(KERNEL_SPEED_20260925.md)."
187
+ ),
188
+ }
189
+
190
+
191
+ def cmd_load(args: argparse.Namespace) -> int:
192
+ det = detect(_target(args))
193
+
194
+ if det.kind == KIND_GLC_RELEASE_V1:
195
+ from .loader import load_model
196
+
197
+ model, tok, receipt = load_model(
198
+ str(det.root),
199
+ device=args.device,
200
+ backend=args.backend,
201
+ trust_remote_code=args.trust_remote_code,
202
+ require_certificate=not args.no_certificate,
203
+ )
204
+ _emit({
205
+ "status": "LOADED",
206
+ "format": det.kind,
207
+ "serving_receipt": receipt,
208
+ "generation": _generate(
209
+ model, tok, args.prompt, device=args.device,
210
+ max_new_tokens=args.max_new_tokens, chat=args.chat,
211
+ ),
212
+ "certificate": certificate_summary(det),
213
+ })
214
+ return EXIT_OK
215
+
216
+ if det.kind == KIND_TBE_V2:
217
+ try:
218
+ from .tbe_artifact import TBEArtifactError, load_standalone
219
+ except ImportError as exc:
220
+ return _fail(
221
+ "tbe_artifact_module_unavailable",
222
+ f"this build of glc_loader cannot load {KIND_TBE_V2} "
223
+ f"artifacts ({exc}). Upgrade glc-loader.",
224
+ )
225
+ from .tbe_device_map import single_device_map
226
+
227
+ device_map = single_device_map(args.device)
228
+ try:
229
+ model, tok, receipt = load_standalone(
230
+ str(det.root), device_map=device_map,
231
+ )
232
+ except TBEArtifactError as exc:
233
+ return _fail(
234
+ getattr(exc, "reason", "load_failed"),
235
+ getattr(exc, "detail", str(exc)),
236
+ )
237
+ except RuntimeError as exc:
238
+ # Report the typed reason the serving stack raised, verbatim. Do
239
+ # not narrate a cause: the refusals on this path include a
240
+ # missing device map, an unsupported compute capability, a
241
+ # non-bf16 candidate AND a parameter the container never carried,
242
+ # and guessing between them in the CLI is how an accurate error
243
+ # becomes a misleading one.
244
+ return _fail(
245
+ getattr(exc, "reason", None) or f"load_refused_{type(exc).__name__}",
246
+ f"{getattr(exc, 'detail', None) or exc} "
247
+ "[glc-loader did not interpret this; it is the serving "
248
+ "stack's own typed refusal. The TBE serving path needs CUDA "
249
+ "of a supported compute capability, or Apple Silicon with the "
250
+ "[metal] extra; it never degrades to a slower or larger path.]",
251
+ )
252
+ if tok is None:
253
+ return _fail(
254
+ "no_tokenizer",
255
+ "the artifact carries no tokenizer, so text cannot go in or "
256
+ "come out. Re-issue it with its tokenizer sidecars.",
257
+ )
258
+ _emit({
259
+ "status": "LOADED",
260
+ "format": det.kind,
261
+ "serving_receipt": receipt,
262
+ "generation": _generate(
263
+ model, tok, args.prompt, device=args.device,
264
+ max_new_tokens=args.max_new_tokens, chat=args.chat,
265
+ ),
266
+ "certificate": certificate_summary(det),
267
+ })
268
+ return EXIT_OK
269
+
270
+ if det.kind == KIND_TBE_V1_TRANSCODE:
271
+ return _fail(det.reason, det.detail, EXIT_UNVERIFIABLE)
272
+ return _fail(det.reason or "not_a_georefine_artifact", det.detail)
273
+
274
+
275
+ # ---------------------------------------------------------------------------
276
+ # expand + generate: GLC-RELEASE/1 only, unchanged
277
+ # ---------------------------------------------------------------------------
278
+ def cmd_expand(args: argparse.Namespace) -> int:
279
+ res = expand(_target(args), args.out, progress=args.progress)
280
+ _emit(res)
281
+ return EXIT_OK
282
+
283
+
284
+ def cmd_generate(args: argparse.Namespace) -> int:
285
+ import torch
286
+
287
+ from .loader import load_model
288
+
289
+ model, tok, receipt = load_model(
290
+ _target(args),
291
+ device=args.device,
292
+ backend=args.backend,
293
+ trust_remote_code=args.trust_remote_code,
294
+ require_certificate=not args.no_certificate,
295
+ )
296
+ prompt = args.prompt
297
+ if args.chat and getattr(tok, "chat_template", None):
298
+ prompt = tok.apply_chat_template(
299
+ [{"role": "user", "content": args.prompt}],
300
+ tokenize=False, add_generation_prompt=True,
301
+ )
302
+ enc = tok(prompt, return_tensors="pt").to(args.device)
303
+ t0 = time.perf_counter()
304
+ with torch.no_grad():
305
+ out = model.generate(
306
+ **enc,
307
+ max_new_tokens=args.max_new_tokens,
308
+ do_sample=args.temperature > 0,
309
+ temperature=args.temperature if args.temperature > 0 else None,
310
+ top_p=args.top_p if args.temperature > 0 else None,
311
+ pad_token_id=tok.pad_token_id or tok.eos_token_id,
312
+ )
313
+ dt = time.perf_counter() - t0
314
+ n_new = int(out.shape[-1] - enc["input_ids"].shape[-1])
315
+ text = tok.decode(out[0][enc["input_ids"].shape[-1]:], skip_special_tokens=True)
316
+ _emit(
317
+ {
318
+ "receipt": receipt,
319
+ "prompt": args.prompt,
320
+ "completion": text,
321
+ "new_tokens": n_new,
322
+ "seconds": dt,
323
+ "tokens_per_second": (n_new / dt) if dt > 0 else None,
324
+ }
325
+ )
326
+ return EXIT_OK
327
+
328
+
329
+ # ---------------------------------------------------------------------------
330
+ def _parser() -> argparse.ArgumentParser:
331
+ p = argparse.ArgumentParser(
332
+ prog="glc-loader",
333
+ description=__doc__,
334
+ formatter_class=argparse.RawDescriptionHelpFormatter,
335
+ )
336
+ p.add_argument("--version", action="store_true",
337
+ help="print the package version and exit")
338
+ sub = p.add_subparsers(dest="command")
339
+
340
+ def common(sp):
341
+ sp.add_argument(
342
+ "artifact_pos", nargs="?", default=None, metavar="ARTIFACT-DIR",
343
+ help="the artifact directory (default: the directory this "
344
+ "package was vendored into)",
345
+ )
346
+ sp.add_argument("--artifact", default=_default_artifact(),
347
+ help=argparse.SUPPRESS)
348
+ return sp
349
+
350
+ i = common(sub.add_parser(
351
+ "info",
352
+ help="format, source, licence, ratios at three scopes, and what is "
353
+ "NOT certified",
354
+ ))
355
+ i.set_defaults(func=cmd_info)
356
+
357
+ v = common(sub.add_parser(
358
+ "verify", help="check the artifact against its own digests",
359
+ ))
360
+ v.add_argument(
361
+ "--deep", action="store_true",
362
+ help="decode every container and check it against the source digest "
363
+ "recorded before encoding; proves bit-exactness locally",
364
+ )
365
+ v.add_argument("--progress", action="store_true")
366
+ v.add_argument(
367
+ "--limit", type=int, default=None,
368
+ help="with --deep, check only the first N coded tensors (a smoke "
369
+ "check, not a verification -- the exit code says so)",
370
+ )
371
+ v.set_defaults(func=cmd_verify)
372
+
373
+ ld = common(sub.add_parser(
374
+ "load", help="smoke-load the artifact and generate a few tokens",
375
+ ))
376
+ ld.add_argument("--device", default="cpu")
377
+ ld.add_argument("--prompt", default="The capital of France is")
378
+ ld.add_argument("--max-new-tokens", type=int, default=16)
379
+ ld.add_argument("--chat", action="store_true")
380
+ ld.add_argument("--backend", default=None,
381
+ choices=[None, "materialize", "resident", "triton"])
382
+ ld.add_argument("--trust-remote-code", action="store_true")
383
+ ld.add_argument(
384
+ "--no-certificate", action="store_true",
385
+ help="load even if the certificate is absent or does not claim a pass",
386
+ )
387
+ ld.set_defaults(func=cmd_load)
388
+
389
+ e = common(sub.add_parser(
390
+ "expand",
391
+ help="write a plain HF checkpoint for any other engine "
392
+ "(GLC-RELEASE/1 only)",
393
+ ))
394
+ e.add_argument("--out", required=True)
395
+ e.add_argument("--progress", action="store_true")
396
+ e.set_defaults(func=cmd_expand)
397
+
398
+ g = common(sub.add_parser(
399
+ "generate", help="load and generate text (GLC-RELEASE/1 only)",
400
+ ))
401
+ g.add_argument("--prompt", required=True)
402
+ g.add_argument("--device", default="cpu")
403
+ g.add_argument("--backend", default=None, choices=[None, "materialize",
404
+ "resident", "triton"])
405
+ g.add_argument("--max-new-tokens", type=int, default=64)
406
+ g.add_argument("--temperature", type=float, default=0.0)
407
+ g.add_argument("--top-p", type=float, default=0.95)
408
+ g.add_argument("--chat", action="store_true")
409
+ g.add_argument("--trust-remote-code", action="store_true")
410
+ g.add_argument(
411
+ "--no-certificate", action="store_true",
412
+ help="load even if the certificate is absent or does not claim a pass",
413
+ )
414
+ g.set_defaults(func=cmd_generate)
415
+ return p
416
+
417
+
418
+ def main(argv: Optional[Sequence[str]] = None) -> int:
419
+ parser = _parser()
420
+ args = parser.parse_args(list(argv) if argv is not None else None)
421
+ if getattr(args, "version", False):
422
+ from . import __version__
423
+
424
+ print(__version__)
425
+ return EXIT_OK
426
+ if not getattr(args, "command", None):
427
+ parser.print_help()
428
+ return EXIT_MALFORMED
429
+ try:
430
+ return int(args.func(args))
431
+ except InspectError as exc:
432
+ return _fail(exc.reason, exc.detail, exc.exit_code)
433
+ except GLCArtifactError as exc:
434
+ return _fail(type(exc).__name__, str(exc))
435
+ except ModuleNotFoundError as exc:
436
+ return _fail(
437
+ "missing_dependency",
438
+ f"{exc}. glc-loader needs torch, safetensors and transformers; "
439
+ "the CUDA fast path additionally needs `pip install "
440
+ "glc-loader[cuda]` and the Apple Silicon path `pip install "
441
+ "glc-loader[metal]`.",
442
+ )
443
+
444
+
445
+ if __name__ == "__main__":
446
+ raise SystemExit(main())