glc-loader 1.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- glc_loader/NOTICE +4 -0
- glc_loader/__init__.py +190 -0
- glc_loader/__main__.py +446 -0
- glc_loader/_inspect.py +786 -0
- glc_loader/artifact.py +517 -0
- glc_loader/container.py +378 -0
- glc_loader/loader.py +611 -0
- glc_loader/metal/__init__.py +25 -0
- glc_loader/metal/ple_gather_mlx.py +424 -0
- glc_loader/metal/tbe_backbone_batched_mlx.py +513 -0
- glc_loader/metal/tbe_decode_mlx.py +409 -0
- glc_loader/metal/tbe_decode_mlx_v2.py +245 -0
- glc_loader/metal/tbe_decode_mlx_v3.py +313 -0
- glc_loader/metal/tbe_decode_mlx_v4.py +174 -0
- glc_loader/metal/tbe_gemv_fused_lut_mlx.py +682 -0
- glc_loader/metal/tbe_gemv_fused_mlx.py +819 -0
- glc_loader/metal/tbe_linear_mlx.py +94 -0
- glc_loader/metal/tbe_linear_mlx_v2.py +62 -0
- glc_loader/metal/tbe_linear_mlx_v3.py +64 -0
- glc_loader/metal/tbe_linear_mlx_v4.py +73 -0
- glc_loader/metal/tbe_mlx_model.py +448 -0
- glc_loader/metal/tbe_moe_batched_mlx.py +565 -0
- glc_loader/modules.py +317 -0
- glc_loader/tbe_anyrank.py +116 -0
- glc_loader/tbe_artifact.py +824 -0
- glc_loader/tbe_container.py +446 -0
- glc_loader/tbe_device_map.py +447 -0
- glc_loader/tbe_gather.py +270 -0
- glc_loader/tbe_mma.py +475 -0
- glc_loader/tbe_mma_kernel.cu +704 -0
- glc_loader/tbe_mma_kernels.py +125 -0
- glc_loader/tbe_modules.py +401 -0
- glc_loader/tbe_serve_portable.py +181 -0
- glc_loader/tbe_serving.py +835 -0
- glc_loader/tbe_stream_loader.py +772 -0
- glc_loader-1.1.0.dist-info/LICENSE +202 -0
- glc_loader-1.1.0.dist-info/METADATA +292 -0
- glc_loader-1.1.0.dist-info/NOTICE +4 -0
- glc_loader-1.1.0.dist-info/RECORD +76 -0
- glc_loader-1.1.0.dist-info/WHEEL +5 -0
- glc_loader-1.1.0.dist-info/entry_points.txt +3 -0
- glc_loader-1.1.0.dist-info/top_level.txt +2 -0
- glc_serve/__init__.py +29 -0
- glc_serve/__main__.py +4 -0
- glc_serve/bench.py +713 -0
- glc_serve/bundle.py +756 -0
- glc_serve/engine.py +633 -0
- glc_serve/espec.py +352 -0
- glc_serve/fastdec.py +640 -0
- glc_serve/fastdec_kernels.cu +512 -0
- glc_serve/fastserve.py +409 -0
- glc_serve/gate.py +342 -0
- glc_serve/loader.py +708 -0
- glc_serve/memcap.py +137 -0
- glc_serve/miv_gemv.py +890 -0
- glc_serve/miv_tbe.py +396 -0
- glc_serve/miv_tbe_csrc/miv_tbe.h +65 -0
- glc_serve/miv_tbe_csrc/miv_tbe_api.cu +76 -0
- glc_serve/miv_tbe_csrc/miv_tbe_kernels.cuh +497 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m1.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m2.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m3.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m4.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m5.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m6.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m7.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_m8.cu +3 -0
- glc_serve/miv_tbe_csrc/miv_tbe_torch.cpp +91 -0
- glc_serve/modules.py +535 -0
- glc_serve/mtp.py +337 -0
- glc_serve/pack.py +431 -0
- glc_serve/q8serve.py +494 -0
- glc_serve/server.py +500 -0
- glc_serve/tbe_desc.py +278 -0
- glc_serve/tunes/fastdec_sm120.json +133 -0
- glc_serve/vendor.py +76 -0
glc_loader/NOTICE
ADDED
glc_loader/__init__.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""``glc_loader`` -- read a GLC-RELEASE artifact with no build repo present.
|
|
2
|
+
|
|
3
|
+
import sys; sys.path.insert(0, "/path/to/artifact")
|
|
4
|
+
from glc_loader import load_model
|
|
5
|
+
model, tokenizer, receipt = load_model("/path/to/artifact", device="cuda")
|
|
6
|
+
|
|
7
|
+
or, without writing any code at all:
|
|
8
|
+
|
|
9
|
+
python -m glc_loader generate --artifact /path/to/artifact --prompt "hello"
|
|
10
|
+
python -m glc_loader expand --artifact /path/to/artifact --out ./dense
|
|
11
|
+
|
|
12
|
+
Third-party requirements: ``torch`` and ``safetensors`` for the container;
|
|
13
|
+
``transformers`` additionally for :func:`load_model` (it supplies the model
|
|
14
|
+
class and the tokenizer, not the weights). ``triton`` is optional and only
|
|
15
|
+
selects a faster backend. Nothing here imports the repository that built the
|
|
16
|
+
artifact -- that is the point of the format, and it is enforced by a test.
|
|
17
|
+
"""
|
|
18
|
+
from .artifact import (
|
|
19
|
+
Artifact,
|
|
20
|
+
FORMAT,
|
|
21
|
+
GLCArtifactError,
|
|
22
|
+
expand,
|
|
23
|
+
open_artifact,
|
|
24
|
+
verify,
|
|
25
|
+
)
|
|
26
|
+
from .container import (
|
|
27
|
+
CONTAINER_MAGIC,
|
|
28
|
+
FWP1Error,
|
|
29
|
+
FWP1Tensor,
|
|
30
|
+
decode_fwp1,
|
|
31
|
+
decode_fwp1_rows,
|
|
32
|
+
encode_fwp1,
|
|
33
|
+
fwp1_certify,
|
|
34
|
+
)
|
|
35
|
+
from .loader import (
|
|
36
|
+
CertificateError,
|
|
37
|
+
assert_certificate_usable,
|
|
38
|
+
load_model,
|
|
39
|
+
load_tokenizer,
|
|
40
|
+
)
|
|
41
|
+
from .modules import (
|
|
42
|
+
BACKENDS,
|
|
43
|
+
GLCBackendError,
|
|
44
|
+
GLCEmbedding,
|
|
45
|
+
GLCLinear,
|
|
46
|
+
GLCTiedLMHead,
|
|
47
|
+
)
|
|
48
|
+
from .tbe_artifact import (
|
|
49
|
+
ARTIFACT_FORMAT,
|
|
50
|
+
COMPRESSION_INFO_FILENAME,
|
|
51
|
+
COMPRESSION_SCHEMA_VERSION,
|
|
52
|
+
STANDALONE_LOADER,
|
|
53
|
+
TBEArtifact,
|
|
54
|
+
TBEArtifactError,
|
|
55
|
+
build_meta_skeleton_from_artifact,
|
|
56
|
+
detect_artifact_format,
|
|
57
|
+
is_legacy_v1_container,
|
|
58
|
+
iter_artifact_tensors,
|
|
59
|
+
load_standalone,
|
|
60
|
+
open_tbe_artifact,
|
|
61
|
+
verify_sha256sums,
|
|
62
|
+
verify_standalone_bit_exact,
|
|
63
|
+
write_sha256sums,
|
|
64
|
+
)
|
|
65
|
+
from .tbe_container import (
|
|
66
|
+
TBEError,
|
|
67
|
+
TBETensor,
|
|
68
|
+
decode_tbe,
|
|
69
|
+
encode_tbe,
|
|
70
|
+
tbe_certify,
|
|
71
|
+
)
|
|
72
|
+
from .tbe_serve_portable import PortableTBELinear, PortableTBEError, load_compressed_transformers
|
|
73
|
+
from .tbe_device_map import (
|
|
74
|
+
STRATEGY_BALANCED,
|
|
75
|
+
STRATEGY_SEQUENTIAL,
|
|
76
|
+
TBEDeviceMapError,
|
|
77
|
+
devices_in_map,
|
|
78
|
+
group_key,
|
|
79
|
+
manifest_group_bytes,
|
|
80
|
+
plan_device_map,
|
|
81
|
+
plan_device_map_from_manifest,
|
|
82
|
+
resolve_device,
|
|
83
|
+
single_device_map,
|
|
84
|
+
validate_device_map,
|
|
85
|
+
)
|
|
86
|
+
from .tbe_mma import (
|
|
87
|
+
TBEDevice,
|
|
88
|
+
TBEMMAError,
|
|
89
|
+
resolve_tbe_mma_arch,
|
|
90
|
+
tbe_mma_available,
|
|
91
|
+
upload_tbe,
|
|
92
|
+
)
|
|
93
|
+
from .tbe_modules import (
|
|
94
|
+
GLCTBEExpertBank,
|
|
95
|
+
GLCTBELinear,
|
|
96
|
+
TBEBackendError,
|
|
97
|
+
TBEPoolRegistry,
|
|
98
|
+
TBETransientPool,
|
|
99
|
+
)
|
|
100
|
+
from .tbe_serving import (
|
|
101
|
+
TBEServingError,
|
|
102
|
+
enable_hf_tbe_serving,
|
|
103
|
+
place_uncoded_tensors,
|
|
104
|
+
)
|
|
105
|
+
from .tbe_stream_loader import (
|
|
106
|
+
TBEStreamLoadError,
|
|
107
|
+
build_meta_skeleton,
|
|
108
|
+
iter_safetensors_shards,
|
|
109
|
+
stream_load_tbe_model,
|
|
110
|
+
stream_load_tbe_model_standalone,
|
|
111
|
+
stream_place_tensors,
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
__version__ = "1.1.0"
|
|
115
|
+
|
|
116
|
+
__all__ = [
|
|
117
|
+
"ARTIFACT_FORMAT",
|
|
118
|
+
"Artifact",
|
|
119
|
+
"BACKENDS",
|
|
120
|
+
"COMPRESSION_INFO_FILENAME",
|
|
121
|
+
"COMPRESSION_SCHEMA_VERSION",
|
|
122
|
+
"CONTAINER_MAGIC",
|
|
123
|
+
"CertificateError",
|
|
124
|
+
"FORMAT",
|
|
125
|
+
"FWP1Error",
|
|
126
|
+
"FWP1Tensor",
|
|
127
|
+
"GLCArtifactError",
|
|
128
|
+
"GLCBackendError",
|
|
129
|
+
"GLCEmbedding",
|
|
130
|
+
"GLCLinear",
|
|
131
|
+
"GLCTBEExpertBank",
|
|
132
|
+
"GLCTBELinear",
|
|
133
|
+
"GLCTiedLMHead",
|
|
134
|
+
"STANDALONE_LOADER",
|
|
135
|
+
"STRATEGY_BALANCED",
|
|
136
|
+
"STRATEGY_SEQUENTIAL",
|
|
137
|
+
"TBEArtifact",
|
|
138
|
+
"TBEArtifactError",
|
|
139
|
+
"TBEBackendError",
|
|
140
|
+
"TBEDevice",
|
|
141
|
+
"TBEDeviceMapError",
|
|
142
|
+
"TBEError",
|
|
143
|
+
"TBEMMAError",
|
|
144
|
+
"TBEPoolRegistry",
|
|
145
|
+
"TBEServingError",
|
|
146
|
+
"TBEStreamLoadError",
|
|
147
|
+
"TBETensor",
|
|
148
|
+
"TBETransientPool",
|
|
149
|
+
"__version__",
|
|
150
|
+
"assert_certificate_usable",
|
|
151
|
+
"build_meta_skeleton",
|
|
152
|
+
"build_meta_skeleton_from_artifact",
|
|
153
|
+
"decode_fwp1",
|
|
154
|
+
"decode_fwp1_rows",
|
|
155
|
+
"decode_tbe",
|
|
156
|
+
"detect_artifact_format",
|
|
157
|
+
"devices_in_map",
|
|
158
|
+
"enable_hf_tbe_serving",
|
|
159
|
+
"encode_fwp1",
|
|
160
|
+
"encode_tbe",
|
|
161
|
+
"expand",
|
|
162
|
+
"fwp1_certify",
|
|
163
|
+
"group_key",
|
|
164
|
+
"is_legacy_v1_container",
|
|
165
|
+
"iter_artifact_tensors",
|
|
166
|
+
"iter_safetensors_shards",
|
|
167
|
+
"load_model",
|
|
168
|
+
"load_standalone",
|
|
169
|
+
"load_tokenizer",
|
|
170
|
+
"manifest_group_bytes",
|
|
171
|
+
"open_artifact",
|
|
172
|
+
"open_tbe_artifact",
|
|
173
|
+
"place_uncoded_tensors",
|
|
174
|
+
"plan_device_map",
|
|
175
|
+
"plan_device_map_from_manifest",
|
|
176
|
+
"resolve_device",
|
|
177
|
+
"resolve_tbe_mma_arch",
|
|
178
|
+
"single_device_map",
|
|
179
|
+
"stream_load_tbe_model",
|
|
180
|
+
"stream_load_tbe_model_standalone",
|
|
181
|
+
"stream_place_tensors",
|
|
182
|
+
"tbe_certify",
|
|
183
|
+
"tbe_mma_available",
|
|
184
|
+
"upload_tbe",
|
|
185
|
+
"validate_device_map",
|
|
186
|
+
"verify",
|
|
187
|
+
"verify_sha256sums",
|
|
188
|
+
"verify_standalone_bit_exact",
|
|
189
|
+
"write_sha256sums",
|
|
190
|
+
]
|
glc_loader/__main__.py
ADDED
|
@@ -0,0 +1,446 @@
|
|
|
1
|
+
"""Client-side command line: ``glc-loader <command>`` / ``python -m glc_loader``.
|
|
2
|
+
|
|
3
|
+
Runs from inside an artifact directory, or against one by path, with nothing
|
|
4
|
+
installed but this package and its three dependencies (torch, safetensors,
|
|
5
|
+
transformers). Nothing here imports the repository that built the artifact.
|
|
6
|
+
|
|
7
|
+
glc-loader info <artifact-dir> what it is, at which scopes, and what
|
|
8
|
+
is NOT certified
|
|
9
|
+
glc-loader verify <artifact-dir> integrity, with honest exit codes
|
|
10
|
+
glc-loader load <artifact-dir> smoke-load and generate a few tokens
|
|
11
|
+
|
|
12
|
+
The older ``--artifact``-flag spelling still works everywhere, so the
|
|
13
|
+
instructions printed inside already-released artifacts keep running.
|
|
14
|
+
|
|
15
|
+
EXIT CODES (verify). 0 verified; 1 an integrity check FAILED; 2 malformed,
|
|
16
|
+
unreadable, or not a GeoRefine container; 3 verified but the artifact
|
|
17
|
+
EXPANDS; 4 the format is recognised but this directory cannot be verified
|
|
18
|
+
from what it ships. 4 exists because the alternative is a misleading 2: a
|
|
19
|
+
v1 TBE transcode directory is a correct artifact with no self-contained
|
|
20
|
+
verifier, and ``experiments.georefine.lossless_audit`` on one exits 2 ("no
|
|
21
|
+
blobs/") purely because it is the wrong auditor for that codec.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import argparse
|
|
26
|
+
import json
|
|
27
|
+
from pathlib import Path
|
|
28
|
+
import sys
|
|
29
|
+
import time
|
|
30
|
+
from typing import Any, Dict, Optional, Sequence
|
|
31
|
+
|
|
32
|
+
if __package__ in (None, ""): # pragma: no cover - direct-script execution
|
|
33
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
34
|
+
__package__ = "glc_loader"
|
|
35
|
+
|
|
36
|
+
from ._inspect import (
|
|
37
|
+
EXIT_MALFORMED,
|
|
38
|
+
EXIT_OK,
|
|
39
|
+
EXIT_UNVERIFIABLE,
|
|
40
|
+
KIND_GLC_RELEASE_V1,
|
|
41
|
+
KIND_TBE_V1_TRANSCODE,
|
|
42
|
+
KIND_TBE_V2,
|
|
43
|
+
InspectError,
|
|
44
|
+
certificate_summary,
|
|
45
|
+
detect,
|
|
46
|
+
not_certified,
|
|
47
|
+
scope_ratios,
|
|
48
|
+
source_identity,
|
|
49
|
+
verify_artifact,
|
|
50
|
+
)
|
|
51
|
+
from .artifact import GLCArtifactError, expand, open_artifact
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _default_artifact() -> str:
|
|
55
|
+
"""The artifact this package was vendored into, if it was vendored."""
|
|
56
|
+
here = Path(__file__).resolve().parent.parent
|
|
57
|
+
return str(here)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _target(args: argparse.Namespace) -> str:
|
|
61
|
+
"""Positional path wins; ``--artifact`` is the compatible spelling."""
|
|
62
|
+
return getattr(args, "artifact_pos", None) or args.artifact
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _emit(obj: Dict[str, Any]) -> None:
|
|
66
|
+
print(json.dumps(obj, indent=2, sort_keys=True, default=str))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _fail(reason: str, detail: str, code: int = EXIT_MALFORMED) -> int:
|
|
70
|
+
print(
|
|
71
|
+
json.dumps(
|
|
72
|
+
{"status": "FAIL", "reason": reason, "detail": detail,
|
|
73
|
+
"exit_code": code},
|
|
74
|
+
indent=2, sort_keys=True,
|
|
75
|
+
),
|
|
76
|
+
file=sys.stderr,
|
|
77
|
+
)
|
|
78
|
+
return code
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
# ---------------------------------------------------------------------------
|
|
82
|
+
# info
|
|
83
|
+
# ---------------------------------------------------------------------------
|
|
84
|
+
def cmd_info(args: argparse.Namespace) -> int:
|
|
85
|
+
det = detect(_target(args))
|
|
86
|
+
out: Dict[str, Any] = {
|
|
87
|
+
"artifact": str(det.root),
|
|
88
|
+
"format": det.kind,
|
|
89
|
+
"source": source_identity(det),
|
|
90
|
+
"ratios": scope_ratios(det),
|
|
91
|
+
"ratio_note": (
|
|
92
|
+
"Three scopes, never conflated. stored = bytes on disk. "
|
|
93
|
+
"served_resident = bytes the device holds for weights. "
|
|
94
|
+
"whole_process = measured peak process VRAM, dense vs coded, and "
|
|
95
|
+
"it is the only one that includes activations, the KV cache and "
|
|
96
|
+
"allocator slack. A ratio reported at one scope says nothing "
|
|
97
|
+
"about the other two."
|
|
98
|
+
),
|
|
99
|
+
"certificate": certificate_summary(det),
|
|
100
|
+
"not_certified": not_certified(det),
|
|
101
|
+
}
|
|
102
|
+
if det.kind == KIND_GLC_RELEASE_V1:
|
|
103
|
+
art = open_artifact(str(det.root), require_certificate=False)
|
|
104
|
+
out["summary"] = art.summary()
|
|
105
|
+
out["container"] = det.manifest.get("container")
|
|
106
|
+
out["effective_config"] = det.manifest.get("effective_config")
|
|
107
|
+
elif det.kind in (KIND_TBE_V2, KIND_TBE_V1_TRANSCODE):
|
|
108
|
+
out["container"] = det.manifest.get("container")
|
|
109
|
+
out["summary"] = det.manifest.get("summary")
|
|
110
|
+
out["escape_band_pct"] = det.manifest.get("escape_band_pct")
|
|
111
|
+
out["escape_over_band"] = [
|
|
112
|
+
o.get("name") for o in (det.manifest.get("escape_over_band") or [])
|
|
113
|
+
]
|
|
114
|
+
if det.compression_info:
|
|
115
|
+
out["compression_info"] = {
|
|
116
|
+
k: v for k, v in det.compression_info.items()
|
|
117
|
+
if k != "sidecars"
|
|
118
|
+
}
|
|
119
|
+
if not det.understood:
|
|
120
|
+
out["status"] = "REFUSED"
|
|
121
|
+
out["reason"] = det.reason
|
|
122
|
+
out["detail"] = det.detail
|
|
123
|
+
_emit(out)
|
|
124
|
+
return EXIT_UNVERIFIABLE if det.kind == KIND_TBE_V1_TRANSCODE \
|
|
125
|
+
else EXIT_MALFORMED
|
|
126
|
+
_emit(out)
|
|
127
|
+
return EXIT_OK
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
# ---------------------------------------------------------------------------
|
|
131
|
+
# verify
|
|
132
|
+
# ---------------------------------------------------------------------------
|
|
133
|
+
def cmd_verify(args: argparse.Namespace) -> int:
|
|
134
|
+
det = detect(_target(args))
|
|
135
|
+
res = verify_artifact(
|
|
136
|
+
det, deep=args.deep, progress=args.progress, limit=args.limit,
|
|
137
|
+
)
|
|
138
|
+
res["certificate"] = certificate_summary(det)
|
|
139
|
+
res["not_certified"] = not_certified(det)
|
|
140
|
+
stream = sys.stdout if res["exit_code"] == EXIT_OK else sys.stderr
|
|
141
|
+
print(json.dumps(res, indent=2, sort_keys=True, default=str), file=stream)
|
|
142
|
+
return int(res["exit_code"])
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
# ---------------------------------------------------------------------------
|
|
146
|
+
# load
|
|
147
|
+
# ---------------------------------------------------------------------------
|
|
148
|
+
def _generate(model, tok, prompt: str, *, device: str, max_new_tokens: int,
|
|
149
|
+
chat: bool) -> Dict[str, Any]:
|
|
150
|
+
import torch
|
|
151
|
+
|
|
152
|
+
text = prompt
|
|
153
|
+
if chat and getattr(tok, "chat_template", None):
|
|
154
|
+
text = tok.apply_chat_template(
|
|
155
|
+
[{"role": "user", "content": prompt}],
|
|
156
|
+
tokenize=False, add_generation_prompt=True,
|
|
157
|
+
)
|
|
158
|
+
enc = tok(text, return_tensors="pt").to(device)
|
|
159
|
+
t0 = time.perf_counter()
|
|
160
|
+
with torch.no_grad():
|
|
161
|
+
out = model.generate(
|
|
162
|
+
**enc,
|
|
163
|
+
max_new_tokens=max_new_tokens,
|
|
164
|
+
do_sample=False,
|
|
165
|
+
pad_token_id=tok.pad_token_id or tok.eos_token_id,
|
|
166
|
+
)
|
|
167
|
+
dt = time.perf_counter() - t0
|
|
168
|
+
n_new = int(out.shape[-1] - enc["input_ids"].shape[-1])
|
|
169
|
+
return {
|
|
170
|
+
"prompt": prompt,
|
|
171
|
+
"completion": tok.decode(
|
|
172
|
+
out[0][enc["input_ids"].shape[-1]:], skip_special_tokens=True,
|
|
173
|
+
),
|
|
174
|
+
"new_tokens": n_new,
|
|
175
|
+
"seconds": dt,
|
|
176
|
+
"tokens_per_second": (n_new / dt) if dt > 0 else None,
|
|
177
|
+
"throughput_note": (
|
|
178
|
+
"a single greedy run on whatever device this is; NOT a benchmark. "
|
|
179
|
+
"Decode speed depends on the engine, not the format. The earlier "
|
|
180
|
+
"decode path measured 0.94x dense on a Blackwell RTX PRO 6000 and "
|
|
181
|
+
"0.57-0.62x on an A100 (single runs, +/-0.05 spread). The GeoRefine "
|
|
182
|
+
"engine (whole-step CUDA graph + fused MIV-TBE kernel) measured "
|
|
183
|
+
"37.84 tok/s batch-1 AR on the Qwen3.8-27B TBE bundle, vs 28.29 for "
|
|
184
|
+
"the bf16 parent in the same engine and 27.4-27.6 for llama.cpp bf16 "
|
|
185
|
+
"on the same RTX PRO 6000, at 49.9-50.0 GB vs 65.4-72.8 GB peak VRAM "
|
|
186
|
+
"(KERNEL_SPEED_20260925.md)."
|
|
187
|
+
),
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def cmd_load(args: argparse.Namespace) -> int:
|
|
192
|
+
det = detect(_target(args))
|
|
193
|
+
|
|
194
|
+
if det.kind == KIND_GLC_RELEASE_V1:
|
|
195
|
+
from .loader import load_model
|
|
196
|
+
|
|
197
|
+
model, tok, receipt = load_model(
|
|
198
|
+
str(det.root),
|
|
199
|
+
device=args.device,
|
|
200
|
+
backend=args.backend,
|
|
201
|
+
trust_remote_code=args.trust_remote_code,
|
|
202
|
+
require_certificate=not args.no_certificate,
|
|
203
|
+
)
|
|
204
|
+
_emit({
|
|
205
|
+
"status": "LOADED",
|
|
206
|
+
"format": det.kind,
|
|
207
|
+
"serving_receipt": receipt,
|
|
208
|
+
"generation": _generate(
|
|
209
|
+
model, tok, args.prompt, device=args.device,
|
|
210
|
+
max_new_tokens=args.max_new_tokens, chat=args.chat,
|
|
211
|
+
),
|
|
212
|
+
"certificate": certificate_summary(det),
|
|
213
|
+
})
|
|
214
|
+
return EXIT_OK
|
|
215
|
+
|
|
216
|
+
if det.kind == KIND_TBE_V2:
|
|
217
|
+
try:
|
|
218
|
+
from .tbe_artifact import TBEArtifactError, load_standalone
|
|
219
|
+
except ImportError as exc:
|
|
220
|
+
return _fail(
|
|
221
|
+
"tbe_artifact_module_unavailable",
|
|
222
|
+
f"this build of glc_loader cannot load {KIND_TBE_V2} "
|
|
223
|
+
f"artifacts ({exc}). Upgrade glc-loader.",
|
|
224
|
+
)
|
|
225
|
+
from .tbe_device_map import single_device_map
|
|
226
|
+
|
|
227
|
+
device_map = single_device_map(args.device)
|
|
228
|
+
try:
|
|
229
|
+
model, tok, receipt = load_standalone(
|
|
230
|
+
str(det.root), device_map=device_map,
|
|
231
|
+
)
|
|
232
|
+
except TBEArtifactError as exc:
|
|
233
|
+
return _fail(
|
|
234
|
+
getattr(exc, "reason", "load_failed"),
|
|
235
|
+
getattr(exc, "detail", str(exc)),
|
|
236
|
+
)
|
|
237
|
+
except RuntimeError as exc:
|
|
238
|
+
# Report the typed reason the serving stack raised, verbatim. Do
|
|
239
|
+
# not narrate a cause: the refusals on this path include a
|
|
240
|
+
# missing device map, an unsupported compute capability, a
|
|
241
|
+
# non-bf16 candidate AND a parameter the container never carried,
|
|
242
|
+
# and guessing between them in the CLI is how an accurate error
|
|
243
|
+
# becomes a misleading one.
|
|
244
|
+
return _fail(
|
|
245
|
+
getattr(exc, "reason", None) or f"load_refused_{type(exc).__name__}",
|
|
246
|
+
f"{getattr(exc, 'detail', None) or exc} "
|
|
247
|
+
"[glc-loader did not interpret this; it is the serving "
|
|
248
|
+
"stack's own typed refusal. The TBE serving path needs CUDA "
|
|
249
|
+
"of a supported compute capability, or Apple Silicon with the "
|
|
250
|
+
"[metal] extra; it never degrades to a slower or larger path.]",
|
|
251
|
+
)
|
|
252
|
+
if tok is None:
|
|
253
|
+
return _fail(
|
|
254
|
+
"no_tokenizer",
|
|
255
|
+
"the artifact carries no tokenizer, so text cannot go in or "
|
|
256
|
+
"come out. Re-issue it with its tokenizer sidecars.",
|
|
257
|
+
)
|
|
258
|
+
_emit({
|
|
259
|
+
"status": "LOADED",
|
|
260
|
+
"format": det.kind,
|
|
261
|
+
"serving_receipt": receipt,
|
|
262
|
+
"generation": _generate(
|
|
263
|
+
model, tok, args.prompt, device=args.device,
|
|
264
|
+
max_new_tokens=args.max_new_tokens, chat=args.chat,
|
|
265
|
+
),
|
|
266
|
+
"certificate": certificate_summary(det),
|
|
267
|
+
})
|
|
268
|
+
return EXIT_OK
|
|
269
|
+
|
|
270
|
+
if det.kind == KIND_TBE_V1_TRANSCODE:
|
|
271
|
+
return _fail(det.reason, det.detail, EXIT_UNVERIFIABLE)
|
|
272
|
+
return _fail(det.reason or "not_a_georefine_artifact", det.detail)
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
# ---------------------------------------------------------------------------
|
|
276
|
+
# expand + generate: GLC-RELEASE/1 only, unchanged
|
|
277
|
+
# ---------------------------------------------------------------------------
|
|
278
|
+
def cmd_expand(args: argparse.Namespace) -> int:
|
|
279
|
+
res = expand(_target(args), args.out, progress=args.progress)
|
|
280
|
+
_emit(res)
|
|
281
|
+
return EXIT_OK
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def cmd_generate(args: argparse.Namespace) -> int:
|
|
285
|
+
import torch
|
|
286
|
+
|
|
287
|
+
from .loader import load_model
|
|
288
|
+
|
|
289
|
+
model, tok, receipt = load_model(
|
|
290
|
+
_target(args),
|
|
291
|
+
device=args.device,
|
|
292
|
+
backend=args.backend,
|
|
293
|
+
trust_remote_code=args.trust_remote_code,
|
|
294
|
+
require_certificate=not args.no_certificate,
|
|
295
|
+
)
|
|
296
|
+
prompt = args.prompt
|
|
297
|
+
if args.chat and getattr(tok, "chat_template", None):
|
|
298
|
+
prompt = tok.apply_chat_template(
|
|
299
|
+
[{"role": "user", "content": args.prompt}],
|
|
300
|
+
tokenize=False, add_generation_prompt=True,
|
|
301
|
+
)
|
|
302
|
+
enc = tok(prompt, return_tensors="pt").to(args.device)
|
|
303
|
+
t0 = time.perf_counter()
|
|
304
|
+
with torch.no_grad():
|
|
305
|
+
out = model.generate(
|
|
306
|
+
**enc,
|
|
307
|
+
max_new_tokens=args.max_new_tokens,
|
|
308
|
+
do_sample=args.temperature > 0,
|
|
309
|
+
temperature=args.temperature if args.temperature > 0 else None,
|
|
310
|
+
top_p=args.top_p if args.temperature > 0 else None,
|
|
311
|
+
pad_token_id=tok.pad_token_id or tok.eos_token_id,
|
|
312
|
+
)
|
|
313
|
+
dt = time.perf_counter() - t0
|
|
314
|
+
n_new = int(out.shape[-1] - enc["input_ids"].shape[-1])
|
|
315
|
+
text = tok.decode(out[0][enc["input_ids"].shape[-1]:], skip_special_tokens=True)
|
|
316
|
+
_emit(
|
|
317
|
+
{
|
|
318
|
+
"receipt": receipt,
|
|
319
|
+
"prompt": args.prompt,
|
|
320
|
+
"completion": text,
|
|
321
|
+
"new_tokens": n_new,
|
|
322
|
+
"seconds": dt,
|
|
323
|
+
"tokens_per_second": (n_new / dt) if dt > 0 else None,
|
|
324
|
+
}
|
|
325
|
+
)
|
|
326
|
+
return EXIT_OK
|
|
327
|
+
|
|
328
|
+
|
|
329
|
+
# ---------------------------------------------------------------------------
|
|
330
|
+
def _parser() -> argparse.ArgumentParser:
|
|
331
|
+
p = argparse.ArgumentParser(
|
|
332
|
+
prog="glc-loader",
|
|
333
|
+
description=__doc__,
|
|
334
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
335
|
+
)
|
|
336
|
+
p.add_argument("--version", action="store_true",
|
|
337
|
+
help="print the package version and exit")
|
|
338
|
+
sub = p.add_subparsers(dest="command")
|
|
339
|
+
|
|
340
|
+
def common(sp):
|
|
341
|
+
sp.add_argument(
|
|
342
|
+
"artifact_pos", nargs="?", default=None, metavar="ARTIFACT-DIR",
|
|
343
|
+
help="the artifact directory (default: the directory this "
|
|
344
|
+
"package was vendored into)",
|
|
345
|
+
)
|
|
346
|
+
sp.add_argument("--artifact", default=_default_artifact(),
|
|
347
|
+
help=argparse.SUPPRESS)
|
|
348
|
+
return sp
|
|
349
|
+
|
|
350
|
+
i = common(sub.add_parser(
|
|
351
|
+
"info",
|
|
352
|
+
help="format, source, licence, ratios at three scopes, and what is "
|
|
353
|
+
"NOT certified",
|
|
354
|
+
))
|
|
355
|
+
i.set_defaults(func=cmd_info)
|
|
356
|
+
|
|
357
|
+
v = common(sub.add_parser(
|
|
358
|
+
"verify", help="check the artifact against its own digests",
|
|
359
|
+
))
|
|
360
|
+
v.add_argument(
|
|
361
|
+
"--deep", action="store_true",
|
|
362
|
+
help="decode every container and check it against the source digest "
|
|
363
|
+
"recorded before encoding; proves bit-exactness locally",
|
|
364
|
+
)
|
|
365
|
+
v.add_argument("--progress", action="store_true")
|
|
366
|
+
v.add_argument(
|
|
367
|
+
"--limit", type=int, default=None,
|
|
368
|
+
help="with --deep, check only the first N coded tensors (a smoke "
|
|
369
|
+
"check, not a verification -- the exit code says so)",
|
|
370
|
+
)
|
|
371
|
+
v.set_defaults(func=cmd_verify)
|
|
372
|
+
|
|
373
|
+
ld = common(sub.add_parser(
|
|
374
|
+
"load", help="smoke-load the artifact and generate a few tokens",
|
|
375
|
+
))
|
|
376
|
+
ld.add_argument("--device", default="cpu")
|
|
377
|
+
ld.add_argument("--prompt", default="The capital of France is")
|
|
378
|
+
ld.add_argument("--max-new-tokens", type=int, default=16)
|
|
379
|
+
ld.add_argument("--chat", action="store_true")
|
|
380
|
+
ld.add_argument("--backend", default=None,
|
|
381
|
+
choices=[None, "materialize", "resident", "triton"])
|
|
382
|
+
ld.add_argument("--trust-remote-code", action="store_true")
|
|
383
|
+
ld.add_argument(
|
|
384
|
+
"--no-certificate", action="store_true",
|
|
385
|
+
help="load even if the certificate is absent or does not claim a pass",
|
|
386
|
+
)
|
|
387
|
+
ld.set_defaults(func=cmd_load)
|
|
388
|
+
|
|
389
|
+
e = common(sub.add_parser(
|
|
390
|
+
"expand",
|
|
391
|
+
help="write a plain HF checkpoint for any other engine "
|
|
392
|
+
"(GLC-RELEASE/1 only)",
|
|
393
|
+
))
|
|
394
|
+
e.add_argument("--out", required=True)
|
|
395
|
+
e.add_argument("--progress", action="store_true")
|
|
396
|
+
e.set_defaults(func=cmd_expand)
|
|
397
|
+
|
|
398
|
+
g = common(sub.add_parser(
|
|
399
|
+
"generate", help="load and generate text (GLC-RELEASE/1 only)",
|
|
400
|
+
))
|
|
401
|
+
g.add_argument("--prompt", required=True)
|
|
402
|
+
g.add_argument("--device", default="cpu")
|
|
403
|
+
g.add_argument("--backend", default=None, choices=[None, "materialize",
|
|
404
|
+
"resident", "triton"])
|
|
405
|
+
g.add_argument("--max-new-tokens", type=int, default=64)
|
|
406
|
+
g.add_argument("--temperature", type=float, default=0.0)
|
|
407
|
+
g.add_argument("--top-p", type=float, default=0.95)
|
|
408
|
+
g.add_argument("--chat", action="store_true")
|
|
409
|
+
g.add_argument("--trust-remote-code", action="store_true")
|
|
410
|
+
g.add_argument(
|
|
411
|
+
"--no-certificate", action="store_true",
|
|
412
|
+
help="load even if the certificate is absent or does not claim a pass",
|
|
413
|
+
)
|
|
414
|
+
g.set_defaults(func=cmd_generate)
|
|
415
|
+
return p
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def main(argv: Optional[Sequence[str]] = None) -> int:
|
|
419
|
+
parser = _parser()
|
|
420
|
+
args = parser.parse_args(list(argv) if argv is not None else None)
|
|
421
|
+
if getattr(args, "version", False):
|
|
422
|
+
from . import __version__
|
|
423
|
+
|
|
424
|
+
print(__version__)
|
|
425
|
+
return EXIT_OK
|
|
426
|
+
if not getattr(args, "command", None):
|
|
427
|
+
parser.print_help()
|
|
428
|
+
return EXIT_MALFORMED
|
|
429
|
+
try:
|
|
430
|
+
return int(args.func(args))
|
|
431
|
+
except InspectError as exc:
|
|
432
|
+
return _fail(exc.reason, exc.detail, exc.exit_code)
|
|
433
|
+
except GLCArtifactError as exc:
|
|
434
|
+
return _fail(type(exc).__name__, str(exc))
|
|
435
|
+
except ModuleNotFoundError as exc:
|
|
436
|
+
return _fail(
|
|
437
|
+
"missing_dependency",
|
|
438
|
+
f"{exc}. glc-loader needs torch, safetensors and transformers; "
|
|
439
|
+
"the CUDA fast path additionally needs `pip install "
|
|
440
|
+
"glc-loader[cuda]` and the Apple Silicon path `pip install "
|
|
441
|
+
"glc-loader[metal]`.",
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
if __name__ == "__main__":
|
|
446
|
+
raise SystemExit(main())
|