outo-models-cli 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/PKG-INFO +8 -2
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/README.md +7 -1
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/pyproject.toml +1 -1
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/__init__.py +1 -1
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/api/__init__.py +28 -0
- outo_models_cli-0.2.0/src/outo_models_cli/api/lfs.py +229 -0
- outo_models_cli-0.2.0/src/outo_models_cli/api/lfs_batch.py +162 -0
- outo_models_cli-0.2.0/src/outo_models_cli/api/lfs_upload.py +182 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/commands/_shared.py +34 -5
- outo_models_cli-0.2.0/src/outo_models_cli/commands/_upload_commit.py +169 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/commands/auth.py +1 -1
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/commands/download.py +13 -2
- outo_models_cli-0.2.0/src/outo_models_cli/commands/upload.py +233 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/config.py +52 -40
- outo_models_cli-0.2.0/src/outo_models_cli/http.py +146 -0
- outo_models_cli-0.2.0/tests/test_cli_upload_lfs.py +312 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_cli_xfer.py +25 -4
- outo_models_cli-0.2.0/tests/test_download_lfs.py +209 -0
- outo_models_cli-0.2.0/tests/test_lfs.py +260 -0
- outo_models_cli-0.2.0/tests/test_upload_commit.py +149 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/uv.lock +1 -1
- outo_models_cli-0.1.0/src/outo_models_cli/commands/upload.py +0 -118
- outo_models_cli-0.1.0/src/outo_models_cli/http.py +0 -73
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/.gitignore +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/LICENSE +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/__main__.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/api/auth.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/api/repos.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/api/upload.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/commands/__init__.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/commands/ls.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/commands/repo.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/downloader/__init__.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/downloader/transfer.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/downloader/walk.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/errors.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/main.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/src/outo_models_cli/matchers.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/conftest.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_api_crud.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_api_upload.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_cli_auth.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_cli_errors.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_cli_help.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_cli_repo.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_config.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_download_repo.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_download_transfer.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_download_walk.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_errors.py +0 -0
- {outo_models_cli-0.1.0 → outo_models_cli-0.2.0}/tests/test_matchers.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: outo-models-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Command-line client for self-hosted outo-models servers — auth, repo, download, and upload.
|
|
5
5
|
Project-URL: Homepage, https://github.com/outo-models/outo-models
|
|
6
6
|
Project-URL: Documentation, https://github.com/outo-models/outo-models/blob/main/omc/README.md
|
|
@@ -47,7 +47,7 @@ Both `omc` and `outo-models-cli` console scripts are installed.
|
|
|
47
47
|
|
|
48
48
|
```bash
|
|
49
49
|
# 1. Log in (prompts for a PAT, masked)
|
|
50
|
-
omc auth login --server
|
|
50
|
+
omc auth login --server https://models.example.com
|
|
51
51
|
|
|
52
52
|
# 2. Verify
|
|
53
53
|
omc auth whoami
|
|
@@ -65,6 +65,12 @@ omc download alice/my-model --local-dir ./my-model
|
|
|
65
65
|
omc upload alice/my-model ./checkpoints --path-in-repo weights --message "v1"
|
|
66
66
|
```
|
|
67
67
|
|
|
68
|
+
Files larger than 100 MiB are uploaded through Git LFS automatically;
|
|
69
|
+
the CLI partitions the input set, runs the LFS batch + PUT dance for
|
|
70
|
+
each large file, and commits the resulting pointer text alongside any
|
|
71
|
+
small files in a single commit. No `git lfs track` setup is required
|
|
72
|
+
on the user side.
|
|
73
|
+
|
|
68
74
|
## Commands
|
|
69
75
|
|
|
70
76
|
```
|
|
@@ -16,7 +16,7 @@ Both `omc` and `outo-models-cli` console scripts are installed.
|
|
|
16
16
|
|
|
17
17
|
```bash
|
|
18
18
|
# 1. Log in (prompts for a PAT, masked)
|
|
19
|
-
omc auth login --server
|
|
19
|
+
omc auth login --server https://models.example.com
|
|
20
20
|
|
|
21
21
|
# 2. Verify
|
|
22
22
|
omc auth whoami
|
|
@@ -34,6 +34,12 @@ omc download alice/my-model --local-dir ./my-model
|
|
|
34
34
|
omc upload alice/my-model ./checkpoints --path-in-repo weights --message "v1"
|
|
35
35
|
```
|
|
36
36
|
|
|
37
|
+
Files larger than 100 MiB are uploaded through Git LFS automatically;
|
|
38
|
+
the CLI partitions the input set, runs the LFS batch + PUT dance for
|
|
39
|
+
each large file, and commits the resulting pointer text alongside any
|
|
40
|
+
small files in a single commit. No `git lfs track` setup is required
|
|
41
|
+
on the user side.
|
|
42
|
+
|
|
37
43
|
## Commands
|
|
38
44
|
|
|
39
45
|
```
|
|
@@ -189,6 +189,22 @@ def display_filename(path_str: str, *, file_root: Path | None, idx: int) -> str:
|
|
|
189
189
|
# sites can write `api.me(...)` instead of `api.auth.me(...)`.
|
|
190
190
|
|
|
191
191
|
from outo_models_cli.api.auth import me # noqa: E402
|
|
192
|
+
from outo_models_cli.api.lfs import ( # noqa: E402
|
|
193
|
+
LFS_CONTENT_TYPE,
|
|
194
|
+
DedupMap,
|
|
195
|
+
LargeFile,
|
|
196
|
+
Partition,
|
|
197
|
+
dedupe_objects,
|
|
198
|
+
partition_files,
|
|
199
|
+
pointer_text,
|
|
200
|
+
pointers_for,
|
|
201
|
+
)
|
|
202
|
+
from outo_models_cli.api.lfs_batch import ( # noqa: E402
|
|
203
|
+
BatchAction,
|
|
204
|
+
BatchObjectError,
|
|
205
|
+
batch_upload,
|
|
206
|
+
)
|
|
207
|
+
from outo_models_cli.api.lfs_upload import upload_objects # noqa: E402
|
|
192
208
|
from outo_models_cli.api.repos import ( # noqa: E402
|
|
193
209
|
create_repo,
|
|
194
210
|
delete_repo,
|
|
@@ -201,23 +217,35 @@ from outo_models_cli.api.repos import ( # noqa: E402
|
|
|
201
217
|
from outo_models_cli.api.upload import upload # noqa: E402
|
|
202
218
|
|
|
203
219
|
__all__ = [
|
|
220
|
+
"LFS_CONTENT_TYPE",
|
|
221
|
+
"BatchAction",
|
|
222
|
+
"BatchObjectError",
|
|
223
|
+
"DedupMap",
|
|
204
224
|
"FileEntry",
|
|
225
|
+
"LargeFile",
|
|
226
|
+
"Partition",
|
|
205
227
|
"RepoDetail",
|
|
206
228
|
"RepoSummary",
|
|
207
229
|
"UploadResult",
|
|
208
230
|
"WhoAmI",
|
|
231
|
+
"batch_upload",
|
|
209
232
|
"create_repo",
|
|
233
|
+
"dedupe_objects",
|
|
210
234
|
"delete_repo",
|
|
211
235
|
"display_filename",
|
|
212
236
|
"get_repo",
|
|
213
237
|
"list_files",
|
|
214
238
|
"list_repos",
|
|
215
239
|
"me",
|
|
240
|
+
"partition_files",
|
|
241
|
+
"pointer_text",
|
|
242
|
+
"pointers_for",
|
|
216
243
|
"resolve_url",
|
|
217
244
|
"send",
|
|
218
245
|
"summary_from",
|
|
219
246
|
"unwrap",
|
|
220
247
|
"upload",
|
|
248
|
+
"upload_objects",
|
|
221
249
|
"walk_repo",
|
|
222
250
|
"with_client",
|
|
223
251
|
]
|
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""Git LFS upload protocol — partition, batch, stream-PUT, pointer text.
|
|
2
|
+
|
|
3
|
+
The CLI drives the four LFS endpoints the server implements:
|
|
4
|
+
|
|
5
|
+
POST /{owner}/{name}.git/info/lfs/objects/batch
|
|
6
|
+
PUT /{owner}/{name}.git/info/lfs/objects/{oid}
|
|
7
|
+
GET /{owner}/{name}.git/info/lfs/objects/{oid}
|
|
8
|
+
GET /{owner}/{name}/resolve/{revision}/{path:path} # redirects to the above for LFS blobs
|
|
9
|
+
|
|
10
|
+
The upload path runs at the boundary between two regimes:
|
|
11
|
+
|
|
12
|
+
* Files at or below the per-file multipart cap (100 MiB) ride the
|
|
13
|
+
existing `POST /api/repos/.../upload` endpoint unchanged.
|
|
14
|
+
* Files above the cap use the LFS batch + PUT dance, then are
|
|
15
|
+
committed to the repo as LFS pointer text. The commit happens via
|
|
16
|
+
the same multipart endpoint — the pointer bytes are tiny so they
|
|
17
|
+
fit the multipart limit with no special-casing.
|
|
18
|
+
|
|
19
|
+
The download path is the mirror image: `GET /resolve/...` returns a 302
|
|
20
|
+
to `/info/lfs/objects/{oid}` whenever the on-tree blob is an LFS
|
|
21
|
+
pointer, and the CLI's Basic-auth client (see `http.build_basic_async_client`)
|
|
22
|
+
follows that redirect transparently because httpx forwards the
|
|
23
|
+
`Authorization` header on same-origin redirects.
|
|
24
|
+
|
|
25
|
+
The wire format is the canonical git-lfs batch shape:
|
|
26
|
+
|
|
27
|
+
Request: {"operation": "upload", "objects": [{"oid", "size"}, ...]}
|
|
28
|
+
Response: {"objects": [{"oid", "size", "actions": {"upload": {"href", "header", "expires_in"}}}
|
|
29
|
+
| "error": {"code", "message"}]
|
|
30
|
+
| (no `actions` when the object is already on the server)}
|
|
31
|
+
|
|
32
|
+
Auth is HTTP Basic (`username:PAT`) — Bearer is not accepted on the LFS
|
|
33
|
+
surface, so this module assumes the client was built with
|
|
34
|
+
`build_basic_client` / `build_basic_async_client`.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
import hashlib
|
|
40
|
+
from collections.abc import Iterable
|
|
41
|
+
from dataclasses import dataclass, field
|
|
42
|
+
from pathlib import Path
|
|
43
|
+
from typing import Any
|
|
44
|
+
|
|
45
|
+
LFS_CONTENT_TYPE = "application/vnd.git-lfs+json"
|
|
46
|
+
|
|
47
|
+
#: Chunk size used when streaming a file into the LFS PUT body. Big
|
|
48
|
+
#: enough to keep the per-chunk syscalls off the hot path; small enough
|
|
49
|
+
#: to bound peak memory at O(chunks-in-flight) regardless of file size.
|
|
50
|
+
PUT_CHUNK_BYTES = 1024 * 1024
|
|
51
|
+
|
|
52
|
+
#: Same chunk size used when computing the sha256 alongside the PUT
|
|
53
|
+
#: stream. Identical to `PUT_CHUNK_BYTES` so a single read loop drives both.
|
|
54
|
+
HASH_CHUNK_BYTES = 1024 * 1024
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
# ---------------------------------------------------------------------------
|
|
58
|
+
# Pointer text
|
|
59
|
+
# ---------------------------------------------------------------------------
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def pointer_text(oid: str, size: int) -> bytes:
|
|
63
|
+
"""Render the canonical git-lfs pointer text for `(oid, size)`.
|
|
64
|
+
|
|
65
|
+
The server's resolve endpoint sniffs the leading `version https://git-lfs`
|
|
66
|
+
line and redirects to the object URL; both the line order and the
|
|
67
|
+
trailing newline are part of the wire contract and MUST match.
|
|
68
|
+
"""
|
|
69
|
+
return (f"version https://git-lfs.github.com/spec/v1\noid sha256:{oid}\nsize {size}\n").encode()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
# File partitioning + streaming sha256
|
|
74
|
+
# ---------------------------------------------------------------------------
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
@dataclass(frozen=True, slots=True)
|
|
78
|
+
class LargeFile:
|
|
79
|
+
"""A file that exceeds the multipart cap and must travel through LFS.
|
|
80
|
+
|
|
81
|
+
`oid` is computed by streaming the file once at partition time. The
|
|
82
|
+
CLI never loads the bytes into memory; both the digest and the
|
|
83
|
+
subsequent PUT body come from the same on-disk file via independent
|
|
84
|
+
read passes.
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
path: Path
|
|
88
|
+
oid: str
|
|
89
|
+
size: int
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@dataclass(frozen=True, slots=True)
|
|
93
|
+
class Partition:
|
|
94
|
+
"""Result of splitting a user-supplied file set into small / large buckets.
|
|
95
|
+
|
|
96
|
+
`small` keeps the original `Path` order so the on-disk filename
|
|
97
|
+
ordering round-trips into the multipart `files[]` list. `large` is
|
|
98
|
+
ordered by descending size so the biggest object (whose PUT takes
|
|
99
|
+
the longest) gets a tighter Rich progress ETA — UX nicety, not a
|
|
100
|
+
protocol requirement.
|
|
101
|
+
"""
|
|
102
|
+
|
|
103
|
+
small: list[Path]
|
|
104
|
+
large: list[LargeFile] = field(default_factory=list)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _sha256_of(path: Path) -> tuple[str, int]:
|
|
108
|
+
"""Stream `path` through sha256; return `(hex_digest, size)`.
|
|
109
|
+
|
|
110
|
+
A 1 MiB read buffer keeps memory bounded regardless of file size.
|
|
111
|
+
Errors from the underlying `open()` (file removed between the
|
|
112
|
+
partition step and the PUT) surface unchanged so the caller's
|
|
113
|
+
`OmcError` mapping sees the real cause.
|
|
114
|
+
"""
|
|
115
|
+
hasher = hashlib.sha256()
|
|
116
|
+
total = 0
|
|
117
|
+
with path.open("rb") as fp:
|
|
118
|
+
while True:
|
|
119
|
+
chunk = fp.read(HASH_CHUNK_BYTES)
|
|
120
|
+
if not chunk:
|
|
121
|
+
break
|
|
122
|
+
hasher.update(chunk)
|
|
123
|
+
total += len(chunk)
|
|
124
|
+
return hasher.hexdigest(), total
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def partition_files(files: Iterable[Path], *, cap_bytes: int) -> Partition:
|
|
128
|
+
"""Split `files` into small (multipart) and large (LFS) buckets.
|
|
129
|
+
|
|
130
|
+
The function streams each large file exactly once to compute its
|
|
131
|
+
sha256 — the same digest the server's LFS PUT handler will verify
|
|
132
|
+
at completion. Identical `(oid, size)` pairs are detected at batch
|
|
133
|
+
time, not here, so the caller still sees one entry per file path;
|
|
134
|
+
dedup happens against the wire-shape objects.
|
|
135
|
+
"""
|
|
136
|
+
from outo_models_cli.errors import BadResponseError
|
|
137
|
+
|
|
138
|
+
small: list[Path] = []
|
|
139
|
+
large: list[LargeFile] = []
|
|
140
|
+
for f in files:
|
|
141
|
+
size = f.stat().st_size
|
|
142
|
+
if size <= cap_bytes:
|
|
143
|
+
small.append(f)
|
|
144
|
+
continue
|
|
145
|
+
oid, observed = _sha256_of(f)
|
|
146
|
+
if observed != size:
|
|
147
|
+
# Defensive: `stat().st_size` could race with concurrent
|
|
148
|
+
# writers. Surface as `BadResponseError`-shaped failure so
|
|
149
|
+
# the user sees a clean English line, not a Python traceback.
|
|
150
|
+
raise BadResponseError(
|
|
151
|
+
f"Size of {f} changed during hashing ({observed} != {size}); retry.",
|
|
152
|
+
)
|
|
153
|
+
large.append(LargeFile(path=f, oid=oid, size=size))
|
|
154
|
+
large.sort(key=lambda lf: lf.size, reverse=True)
|
|
155
|
+
return Partition(small=small, large=large)
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# ---------------------------------------------------------------------------
|
|
159
|
+
# Deduplication — identical (oid, size) pairs collapse to one batch entry
|
|
160
|
+
# ---------------------------------------------------------------------------
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
@dataclass(frozen=True, slots=True)
|
|
164
|
+
class DedupMap:
|
|
165
|
+
"""Mapping from a deduped `(oid, size)` back to the files that need it.
|
|
166
|
+
|
|
167
|
+
The server's batch response carries one entry per `(oid, size)`. If
|
|
168
|
+
the user happens to have two different paths with identical content
|
|
169
|
+
(a common case in model repos that bundle the same checkpoint twice
|
|
170
|
+
under different names), only one PUT is required and the pointer
|
|
171
|
+
text is reused at commit time.
|
|
172
|
+
"""
|
|
173
|
+
|
|
174
|
+
key: tuple[str, int]
|
|
175
|
+
paths: tuple[Path, ...]
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def dedupe_objects(items: list[LargeFile]) -> tuple[list[dict[str, Any]], list[DedupMap]]:
|
|
179
|
+
"""Collapse duplicate `(oid, size)` items; return (batch_entries, path_maps).
|
|
180
|
+
|
|
181
|
+
`batch_entries` is the list of `{"oid", "size"}` dicts to send in
|
|
182
|
+
the LFS batch body — order matches the first occurrence in `items`
|
|
183
|
+
so the partition-time ordering survives. `path_maps` carries the
|
|
184
|
+
full set of paths that need each pointer text at commit time.
|
|
185
|
+
"""
|
|
186
|
+
seen: dict[tuple[str, int], int] = {}
|
|
187
|
+
entries: list[dict[str, Any]] = []
|
|
188
|
+
path_lists: list[list[Path]] = []
|
|
189
|
+
|
|
190
|
+
for item in items:
|
|
191
|
+
key = (item.oid, item.size)
|
|
192
|
+
idx = seen.get(key)
|
|
193
|
+
if idx is None:
|
|
194
|
+
seen[key] = len(entries)
|
|
195
|
+
entries.append({"oid": item.oid, "size": item.size})
|
|
196
|
+
path_lists.append([item.path])
|
|
197
|
+
else:
|
|
198
|
+
path_lists[idx].append(item.path)
|
|
199
|
+
|
|
200
|
+
maps = [DedupMap(key=key, paths=tuple(path_lists[idx])) for key, idx in seen.items()]
|
|
201
|
+
return entries, maps
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def pointers_for(large: list[LargeFile]) -> dict[str, bytes]:
|
|
205
|
+
"""Return a `{oid: pointer_text}` mapping for every deduped object.
|
|
206
|
+
|
|
207
|
+
The mapping is keyed on oid only because dedupe means a single
|
|
208
|
+
pointer text covers every file that hashes to that oid; the caller
|
|
209
|
+
expands it back to one entry per path when building the multipart
|
|
210
|
+
request.
|
|
211
|
+
"""
|
|
212
|
+
out: dict[str, bytes] = {}
|
|
213
|
+
for item in large:
|
|
214
|
+
out.setdefault(item.oid, pointer_text(item.oid, item.size))
|
|
215
|
+
return out
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
__all__ = [
|
|
219
|
+
"HASH_CHUNK_BYTES",
|
|
220
|
+
"LFS_CONTENT_TYPE",
|
|
221
|
+
"PUT_CHUNK_BYTES",
|
|
222
|
+
"DedupMap",
|
|
223
|
+
"LargeFile",
|
|
224
|
+
"Partition",
|
|
225
|
+
"dedupe_objects",
|
|
226
|
+
"partition_files",
|
|
227
|
+
"pointer_text",
|
|
228
|
+
"pointers_for",
|
|
229
|
+
]
|
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""LFS batch request/response parsing.
|
|
2
|
+
|
|
3
|
+
The server's `/info/lfs/objects/batch` endpoint accepts an
|
|
4
|
+
`{"operation": "upload", "objects": [{"oid", "size"}]}` body and returns
|
|
5
|
+
a response with one entry per object. Each entry has either an
|
|
6
|
+
`actions.upload` block (with an absolute `href` the client streams the
|
|
7
|
+
raw bytes to) or a per-object `error` block (413 size cap, 413 quota,
|
|
8
|
+
404 unknown object for the download operation). An object that is
|
|
9
|
+
already on the server comes back WITHOUT `actions` AND without `error`
|
|
10
|
+
— the client interprets that as "skip the PUT, write pointer text".
|
|
11
|
+
|
|
12
|
+
This module is the pure wire-format layer: it sends the request and
|
|
13
|
+
parses the response into typed dataclasses that the upload command can
|
|
14
|
+
drain with one loop. The actual streaming PUT lives in `lfs_upload.py`.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
|
|
21
|
+
import httpx
|
|
22
|
+
|
|
23
|
+
from outo_models_cli.api import send, unwrap
|
|
24
|
+
from outo_models_cli.api.lfs import LFS_CONTENT_TYPE
|
|
25
|
+
from outo_models_cli.errors import BadResponseError, map_response_error
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True, slots=True)
|
|
29
|
+
class BatchAction:
|
|
30
|
+
"""Wire-level action handed back by the server for one object."""
|
|
31
|
+
|
|
32
|
+
oid: str
|
|
33
|
+
size: int
|
|
34
|
+
href: str
|
|
35
|
+
header: dict[str, str] = field(default_factory=dict)
|
|
36
|
+
expires_in: int | None = None
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
@dataclass(frozen=True, slots=True)
|
|
40
|
+
class BatchObjectError:
|
|
41
|
+
"""Per-object failure surfaced by the LFS batch response."""
|
|
42
|
+
|
|
43
|
+
oid: str
|
|
44
|
+
size: int
|
|
45
|
+
code: int
|
|
46
|
+
message: str
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def batch_upload(
|
|
50
|
+
client: httpx.Client,
|
|
51
|
+
*,
|
|
52
|
+
owner: str,
|
|
53
|
+
name: str,
|
|
54
|
+
objects: list[dict[str, object]],
|
|
55
|
+
) -> tuple[list[BatchAction], list[BatchObjectError], list[str]]:
|
|
56
|
+
"""POST the LFS upload batch; split the response into actions / errors / present.
|
|
57
|
+
|
|
58
|
+
Returns:
|
|
59
|
+
* `actions` — objects the server wants uploaded, in input order.
|
|
60
|
+
* `errors` — per-object failures (413 size cap, 413 quota, ...).
|
|
61
|
+
* `present` — oids the server already has; the caller skips the PUT
|
|
62
|
+
and still writes the pointer text for these at commit time.
|
|
63
|
+
"""
|
|
64
|
+
body = {
|
|
65
|
+
"operation": "upload",
|
|
66
|
+
"transfers": ["basic"],
|
|
67
|
+
"objects": objects,
|
|
68
|
+
}
|
|
69
|
+
response = send(
|
|
70
|
+
client,
|
|
71
|
+
"POST",
|
|
72
|
+
f"/{owner}/{name}.git/info/lfs/objects/batch",
|
|
73
|
+
headers={"Accept": LFS_CONTENT_TYPE, "Content-Type": LFS_CONTENT_TYPE},
|
|
74
|
+
json=body,
|
|
75
|
+
)
|
|
76
|
+
if response.status_code == 404:
|
|
77
|
+
raise BadResponseError(
|
|
78
|
+
"Server does not support Git LFS at this endpoint. "
|
|
79
|
+
"Upgrade the server or contact the operator.",
|
|
80
|
+
)
|
|
81
|
+
if response.status_code == 406:
|
|
82
|
+
raise BadResponseError(
|
|
83
|
+
"Server rejected the LFS request (missing Accept: application/vnd.git-lfs+json).",
|
|
84
|
+
)
|
|
85
|
+
if response.status_code == 415:
|
|
86
|
+
raise BadResponseError(
|
|
87
|
+
"Server rejected the LFS request (wrong Content-Type).",
|
|
88
|
+
)
|
|
89
|
+
if response.status_code >= 400:
|
|
90
|
+
raise map_response_error(response)
|
|
91
|
+
payload = unwrap(response)
|
|
92
|
+
raw_objects = payload.get("objects")
|
|
93
|
+
if not isinstance(raw_objects, list):
|
|
94
|
+
raise BadResponseError("LFS batch response is missing `objects`.")
|
|
95
|
+
|
|
96
|
+
actions: list[BatchAction] = []
|
|
97
|
+
errors: list[BatchObjectError] = []
|
|
98
|
+
present: list[str] = []
|
|
99
|
+
|
|
100
|
+
for entry in raw_objects:
|
|
101
|
+
if not isinstance(entry, dict):
|
|
102
|
+
continue
|
|
103
|
+
oid = str(entry.get("oid", ""))
|
|
104
|
+
size_val = entry.get("size", 0)
|
|
105
|
+
size = int(size_val) if isinstance(size_val, (int, float)) else 0
|
|
106
|
+
error_raw = entry.get("error")
|
|
107
|
+
if isinstance(error_raw, dict):
|
|
108
|
+
code_raw = error_raw.get("code")
|
|
109
|
+
code = int(code_raw) if isinstance(code_raw, (int, float)) else 0
|
|
110
|
+
message = str(error_raw.get("message", ""))
|
|
111
|
+
errors.append(BatchObjectError(oid=oid, size=size, code=code, message=message))
|
|
112
|
+
continue
|
|
113
|
+
actions_raw = entry.get("actions")
|
|
114
|
+
if not isinstance(actions_raw, dict):
|
|
115
|
+
# No actions AND no error → object is already stored server-side.
|
|
116
|
+
present.append(oid)
|
|
117
|
+
continue
|
|
118
|
+
upload_raw = actions_raw.get("upload")
|
|
119
|
+
if not isinstance(upload_raw, dict):
|
|
120
|
+
errors.append(
|
|
121
|
+
BatchObjectError(
|
|
122
|
+
oid=oid,
|
|
123
|
+
size=size,
|
|
124
|
+
code=500,
|
|
125
|
+
message="LFS response missing `actions.upload`.",
|
|
126
|
+
)
|
|
127
|
+
)
|
|
128
|
+
continue
|
|
129
|
+
href = str(upload_raw.get("href", ""))
|
|
130
|
+
if not href:
|
|
131
|
+
errors.append(
|
|
132
|
+
BatchObjectError(
|
|
133
|
+
oid=oid,
|
|
134
|
+
size=size,
|
|
135
|
+
code=500,
|
|
136
|
+
message="LFS response missing upload `href`.",
|
|
137
|
+
)
|
|
138
|
+
)
|
|
139
|
+
continue
|
|
140
|
+
header_raw = upload_raw.get("header") or {}
|
|
141
|
+
header: dict[str, str] = {}
|
|
142
|
+
if isinstance(header_raw, dict):
|
|
143
|
+
for k, v in header_raw.items():
|
|
144
|
+
if isinstance(k, str) and isinstance(v, str):
|
|
145
|
+
header[k] = v
|
|
146
|
+
expires_in_raw = upload_raw.get("expires_in")
|
|
147
|
+
expires_in: int | None = None
|
|
148
|
+
if isinstance(expires_in_raw, (int, float)):
|
|
149
|
+
expires_in = int(expires_in_raw)
|
|
150
|
+
actions.append(
|
|
151
|
+
BatchAction(
|
|
152
|
+
oid=oid,
|
|
153
|
+
size=size,
|
|
154
|
+
href=href,
|
|
155
|
+
header=header,
|
|
156
|
+
expires_in=expires_in,
|
|
157
|
+
)
|
|
158
|
+
)
|
|
159
|
+
return actions, errors, present
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
__all__ = ["BatchAction", "BatchObjectError", "batch_upload"]
|