instance-repo 1.0.7.dev0__tar.gz → 1.0.9__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/PKG-INFO +1 -1
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/__init__.py +1 -1
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/cli.py +17 -4
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/instances.py +169 -55
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/versions.py +63 -7
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/content.py +31 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/layout.py +8 -2
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/models.py +3 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/repo.py +13 -3
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/store/oss.py +9 -13
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/store/registry.py +13 -10
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/validate.py +81 -8
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo.egg-info/PKG-INFO +1 -1
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/pyproject.toml +1 -1
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/README.md +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/_routing.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/_site_defaults.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/bootstrap.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/cache.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/__init__.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/config.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/datasets.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/images.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/reports.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/clients/scaffold.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/concurrency.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/endpoints.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/errors.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/image.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/loader.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/paths.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/release.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/retry.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/store/__init__.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/store/acr.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/store/base.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/tasktoml.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo/transport.py +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo.egg-info/SOURCES.txt +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo.egg-info/dependency_links.txt +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo.egg-info/entry_points.txt +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo.egg-info/requires.txt +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo.egg-info/top_level.txt +0 -0
- {instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/setup.cfg +0 -0
|
@@ -60,8 +60,8 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
60
60
|
p.add_argument("--no-image", action="store_true")
|
|
61
61
|
p.add_argument("--overwrite", action="store_true")
|
|
62
62
|
p.add_argument("--no-register", action="store_true",
|
|
63
|
-
help="
|
|
64
|
-
"
|
|
63
|
+
help="只传 OSS 不写库(默认上传后自动 ingest 写库;"
|
|
64
|
+
"置上则跳过 ingest,之后自行 ingest 扫描入库)")
|
|
65
65
|
p.add_argument("--no-verify", action="store_true",
|
|
66
66
|
help="跳过上传后的回读校验(默认校验 OSS 对象/ACR 镜像确已落库)")
|
|
67
67
|
|
|
@@ -76,7 +76,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
76
76
|
pm.add_argument("--no-image", action="store_true")
|
|
77
77
|
pm.add_argument("--overwrite", action="store_true")
|
|
78
78
|
pm.add_argument("--register", action="store_true",
|
|
79
|
-
help="
|
|
79
|
+
help="整批上传完成后一次 ingest 写库(缺省只传 OSS,不写库)")
|
|
80
80
|
pm.add_argument("--no-verify", action="store_true")
|
|
81
81
|
pm.add_argument("--concurrency", type=int, default=8)
|
|
82
82
|
pm.add_argument("--continue-on-error", action="store_true",
|
|
@@ -285,6 +285,8 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
285
285
|
help="create 用:oss://{bucket}/{prefix}")
|
|
286
286
|
ver.add_argument("--storage-type", choices=["oss"], default="oss")
|
|
287
287
|
ver.add_argument("--splits", default=None, help="逗号分隔 split 名")
|
|
288
|
+
ver.add_argument("--split", default=None,
|
|
289
|
+
help="status 用:查 split 级真实记录(split_first 发布状态/目标路径)")
|
|
288
290
|
ver.add_argument("--status", choices=_STATUS_CHOICES, default=None,
|
|
289
291
|
help="update 用:版本状态流转")
|
|
290
292
|
ver.add_argument("--run-type", choices=["train", "eval"], default=None,
|
|
@@ -454,6 +456,11 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
454
456
|
verify=not args.no_verify)
|
|
455
457
|
print(f"{res.instance_id} uploaded={res.uploaded} "
|
|
456
458
|
f"image_pushed={res.image_pushed}", file=out)
|
|
459
|
+
if res.ingest:
|
|
460
|
+
stats = res.ingest.get("stats", {}) or {}
|
|
461
|
+
print(f"ingest {res.ingest.get('ingest_id')} "
|
|
462
|
+
f"status={res.ingest.get('status')} "
|
|
463
|
+
f"instances={stats.get('instance_count')}", file=out)
|
|
457
464
|
return 0
|
|
458
465
|
|
|
459
466
|
if args.cmd == "push-many":
|
|
@@ -711,7 +718,13 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
711
718
|
return 0
|
|
712
719
|
dataset, version, _ = _resolve_target(args, 3)
|
|
713
720
|
if args.action == "status":
|
|
714
|
-
|
|
721
|
+
if args.split:
|
|
722
|
+
sp = r.versions.split_detail(dataset, version, args.split)
|
|
723
|
+
print(f"{sp.get('name', args.split)}\t{sp.get('status', '')}\t"
|
|
724
|
+
f"{sp.get('storage_path', '')}\t"
|
|
725
|
+
f"published_at={sp.get('published_at') or ''}", file=out)
|
|
726
|
+
else:
|
|
727
|
+
print(r.versions.status(dataset, version), file=out)
|
|
715
728
|
return 0
|
|
716
729
|
if args.action == "set-run-type":
|
|
717
730
|
r.versions.set_run_type(dataset, version,
|
|
@@ -15,25 +15,46 @@ from pathlib import Path
|
|
|
15
15
|
from .. import paths as P
|
|
16
16
|
from ..concurrency import map_concurrent
|
|
17
17
|
from ..content import pack
|
|
18
|
-
from ..errors import ImageMissing, TransportError
|
|
18
|
+
from ..errors import ImageMissing, SchemaError, TransportError
|
|
19
19
|
from ..image import dockerfile_from, rewrite_dockerfile
|
|
20
20
|
from ..layout import Layout
|
|
21
21
|
from ..models import (
|
|
22
22
|
DatasetInstance, ManifestEntry, OwnerReference, PushResult, Feedback,
|
|
23
23
|
)
|
|
24
|
-
from ..validate import check_layout, check_swe_json,
|
|
24
|
+
from ..validate import (check_layout, check_swe_json, check_custom,
|
|
25
|
+
detect_format, validate_instance_id,
|
|
26
|
+
CUSTOM_SCHEMA_VERSION)
|
|
25
27
|
|
|
26
28
|
logger = logging.getLogger(__name__)
|
|
27
29
|
|
|
28
30
|
|
|
29
31
|
class InstancesClient:
|
|
30
32
|
def __init__(self, transport, profile, content_store, image_store,
|
|
31
|
-
registry) -> None:
|
|
33
|
+
registry, ingester=None) -> None:
|
|
32
34
|
self._t = transport
|
|
33
35
|
self._p = profile
|
|
34
36
|
self._content = content_store
|
|
35
37
|
self._image = image_store
|
|
36
38
|
self._reg = registry
|
|
39
|
+
# register=True 时经 ingest 写库(apiserver :commit 端点缺失,三段式
|
|
40
|
+
# 收尾改为 ingest)。ingester 形如 (dataset, version) -> ingest job dict;
|
|
41
|
+
# 由 Repo 注入 versions.ingest。直接构造未注入时 register=True 会报错。
|
|
42
|
+
self._ingester = ingester
|
|
43
|
+
|
|
44
|
+
def _after_upload(self, register: bool, dataset: str,
|
|
45
|
+
version: str) -> dict | None:
|
|
46
|
+
"""register=True:上传成功后触发一次 ingest 写库,返回 ingest job 摘要。
|
|
47
|
+
|
|
48
|
+
register=False 返回 None(只传 OSS,由调用方决定何时 ingest)。
|
|
49
|
+
"""
|
|
50
|
+
if not register:
|
|
51
|
+
return None
|
|
52
|
+
if self._ingester is None:
|
|
53
|
+
raise SchemaError(
|
|
54
|
+
"register=True 需要 ingest 回调:请通过 Repo() 构造客户端"
|
|
55
|
+
"(已注入 versions.ingest),或显式传入 ingester=...;"
|
|
56
|
+
"直接构造 InstancesClient 且未注入时请用 register=False 只传 OSS")
|
|
57
|
+
return self._ingester(dataset, version)
|
|
37
58
|
|
|
38
59
|
def _layout(self, metadata_model: str | None = None) -> Layout:
|
|
39
60
|
"""按(显式或 profile 的)元数据模型构造布局策略。
|
|
@@ -47,10 +68,12 @@ class InstancesClient:
|
|
|
47
68
|
|
|
48
69
|
# ── 只读 ──
|
|
49
70
|
def validate(self, local_path):
|
|
50
|
-
"""校验实例:自动检测格式(目录→harbor,.json→swe)。"""
|
|
71
|
+
"""校验实例:自动检测格式(目录→harbor,dir-notask→custom,.json→swe)。"""
|
|
51
72
|
fmt = detect_format(local_path)
|
|
52
73
|
if fmt == "swe":
|
|
53
74
|
return check_swe_json(local_path)
|
|
75
|
+
if fmt == "custom":
|
|
76
|
+
return check_custom(local_path)
|
|
54
77
|
return check_layout(local_path)
|
|
55
78
|
|
|
56
79
|
def get(self, dataset: str, version: str, instance_id: str,
|
|
@@ -82,20 +105,32 @@ class InstancesClient:
|
|
|
82
105
|
harbor: validate → (可选)推镜像 → 上传 OSS(content.tgz + {id}.json)。
|
|
83
106
|
swe: validate JSON → 上传 OSS({id}.json only),不推镜像。
|
|
84
107
|
|
|
85
|
-
register=True
|
|
86
|
-
|
|
87
|
-
|
|
108
|
+
register=True(默认):上传成功后自动触发一次 versions.ingest 把元数据写库
|
|
109
|
+
(apiserver 的 :commit 端点缺失,三段式收尾由 ingest 替代;结果见
|
|
110
|
+
``PushResult.ingest``)。register=False:只传 OSS 不写库,之后由调用方
|
|
111
|
+
自行 versions.ingest() 扫描入库。
|
|
88
112
|
|
|
89
113
|
verify=True(默认)在真正发生上传后回读校验:确认 OSS 对象确实可 HEAD 到。
|
|
90
114
|
|
|
91
115
|
image_release_dst_acr(可选):目标(prod)ACR host(仅 harbor 格式有效)。
|
|
92
116
|
"""
|
|
93
117
|
local_path = Path(local_path)
|
|
118
|
+
if register and self._ingester is None:
|
|
119
|
+
# fail-fast:register=True 需要 ingest 回调,先于任何打包/上传报错
|
|
120
|
+
raise SchemaError(
|
|
121
|
+
"register=True 需要 ingest 回调:请通过 Repo() 构造客户端"
|
|
122
|
+
"(已注入 versions.ingest),或显式传入 ingester=...;"
|
|
123
|
+
"直接构造 InstancesClient 且未注入时请用 register=False 只传 OSS")
|
|
94
124
|
fmt = detect_format(local_path)
|
|
95
125
|
if fmt == "swe":
|
|
96
126
|
return self._push_swe(local_path, dataset, version, split=split,
|
|
97
127
|
overwrite=overwrite, register=register,
|
|
98
128
|
verify=verify)
|
|
129
|
+
if fmt == "custom":
|
|
130
|
+
return self._push_custom(local_path, dataset, version, split=split,
|
|
131
|
+
push_image=push_image,
|
|
132
|
+
overwrite=overwrite, register=register,
|
|
133
|
+
verify=verify)
|
|
99
134
|
# ── harbor 路径 ──
|
|
100
135
|
report = check_layout(local_path) # ① 本地校验(缺文件即抛)
|
|
101
136
|
instance_id = report.instance_id
|
|
@@ -147,23 +182,26 @@ class InstancesClient:
|
|
|
147
182
|
instance_status="draft",
|
|
148
183
|
)
|
|
149
184
|
uid = instance_id
|
|
150
|
-
if register:
|
|
151
|
-
registered = self._reg.register(meta) # 校验 schema(owner_reference 可选)
|
|
152
|
-
uid = registered.id or instance_id
|
|
153
|
-
|
|
154
|
-
image_pushed = False
|
|
155
|
-
if push_image and has_image:
|
|
156
|
-
# ③ 推 ACR(final 改写 Dockerfile)
|
|
157
|
-
self._image.push(image_tar, docker_ref, overwrite=overwrite)
|
|
158
|
-
rewrite_dockerfile(dockerfile, docker_ref, tar_role="final")
|
|
159
|
-
image_pushed = True
|
|
160
185
|
|
|
161
|
-
#
|
|
186
|
+
# 打包 + 构造 manifest(本地)→ register(pending) → 推 ACR → 上传 OSS。
|
|
187
|
+
# register 需要 manifest:apiserver 的 hasFile 校验要求至少一条
|
|
188
|
+
# content_type=file 项,manifest 缺失时 register 被 92005 拒(曾导致
|
|
189
|
+
# custom/harbor 主路径 push(register=True) 失败)。register 仍先于任何
|
|
190
|
+
# 远程写(元数据先行),manifest 在本地打包时即可算出 digest。
|
|
162
191
|
with tempfile.TemporaryDirectory(prefix="irepo_") as td:
|
|
163
192
|
tgz = Path(td) / "content.tgz"
|
|
164
193
|
digest = pack(local_path, tgz, exclude_image_tar=True)
|
|
165
194
|
meta.manifest = _manifest(self._p.oss_bucket, dataset, version,
|
|
166
195
|
instance_id, split, digest, lay)
|
|
196
|
+
|
|
197
|
+
image_pushed = False
|
|
198
|
+
if push_image and has_image:
|
|
199
|
+
# ③ 推 ACR(final 改写 Dockerfile)
|
|
200
|
+
self._image.push(image_tar, docker_ref, overwrite=overwrite)
|
|
201
|
+
rewrite_dockerfile(dockerfile, docker_ref, tar_role="final")
|
|
202
|
+
image_pushed = True
|
|
203
|
+
|
|
204
|
+
# ④ 上传 OSS(排除 image.tar.gz);{id}.json 供 pull 校验 / ingest 扫描
|
|
167
205
|
put = self._content.put(instance_id, dataset, version, split, tgz,
|
|
168
206
|
meta.to_canonical_json(), overwrite=overwrite,
|
|
169
207
|
layout=lay)
|
|
@@ -181,22 +219,15 @@ class InstancesClient:
|
|
|
181
219
|
raise TransportError(
|
|
182
220
|
f"{instance_id}: ACR 回读校验失败——推送后未 inspect 到镜像 "
|
|
183
221
|
f"{docker_ref}")
|
|
184
|
-
# ⑤
|
|
185
|
-
#
|
|
186
|
-
#
|
|
187
|
-
|
|
188
|
-
# digest,之后任何 pull(verify=True) 必报 E_DIGEST_MISMATCH。清 digests 还不够——
|
|
189
|
-
# pull 读的是 **manifest** 里的 content_digest,故 skip 分支必须同时摘掉它。
|
|
222
|
+
# ⑤ 写库:register=True 时经 ingest 扫描入库(替代缺失的 :commit 三段式收尾)。
|
|
223
|
+
# ingest 幂等(重扫 OSS),skip 与否都由服务端按 OSS 实际字节重算 digest,
|
|
224
|
+
# 不再有「本地 digest 覆盖别家生产者」的 skip 分支歧义。
|
|
225
|
+
ingest_result = self._after_upload(register, dataset, version)
|
|
190
226
|
digests = {"content.tgz": digest}
|
|
191
|
-
if register:
|
|
192
|
-
commit_digests = None if skipped else digests
|
|
193
|
-
commit_manifest = (_manifest_without_digest(meta.manifest) if skipped
|
|
194
|
-
else meta.manifest)
|
|
195
|
-
self._reg.commit(uid, commit_digests, manifest=commit_manifest)
|
|
196
227
|
return PushResult(instance_id=instance_id, uploaded=not skipped,
|
|
197
228
|
skipped=skipped, image_pushed=image_pushed,
|
|
198
229
|
docker_image=docker_ref or None,
|
|
199
|
-
digests=digests)
|
|
230
|
+
digests=digests, ingest=ingest_result)
|
|
200
231
|
|
|
201
232
|
def _push_swe(self, json_path: Path, dataset: str, version: str, *,
|
|
202
233
|
split: str = "default", overwrite: bool = False,
|
|
@@ -245,32 +276,105 @@ class InstancesClient:
|
|
|
245
276
|
raise TransportError(
|
|
246
277
|
f"{instance_id}: OSS 回读校验失败——上传后未在 OSS 找到 {idx_key}")
|
|
247
278
|
|
|
248
|
-
# ④
|
|
249
|
-
|
|
250
|
-
file_path=P.oss_uri(self._p.oss_bucket, idx_key),
|
|
251
|
-
content_type="file", content_format="json",
|
|
252
|
-
content_digest=digest)]
|
|
253
|
-
|
|
254
|
-
if register:
|
|
255
|
-
# 构造结构化元数据(不含 patch/test_patch 等大字段)
|
|
256
|
-
meta = DatasetInstance(
|
|
257
|
-
instance_id=instance_id, dataset=dataset,
|
|
258
|
-
dataset_version=version, split=split,
|
|
259
|
-
docker_image=docker_image,
|
|
260
|
-
manifest=manifest,
|
|
261
|
-
extend_meta={k: str(v) for k, v in report.raw_data.items()
|
|
262
|
-
if k in ("repo", "base_commit", "language")
|
|
263
|
-
and v is not None},
|
|
264
|
-
instance_status="draft",
|
|
265
|
-
)
|
|
266
|
-
registered = self._reg.register(meta)
|
|
267
|
-
uid = registered.id or instance_id
|
|
268
|
-
self._reg.commit(uid, {"instance.json": digest}, manifest=manifest)
|
|
279
|
+
# ④ 写库:register=True 时经 ingest 扫描入库(替代缺失的 :commit 收尾)
|
|
280
|
+
ingest_result = self._after_upload(register, dataset, version)
|
|
269
281
|
|
|
270
282
|
return PushResult(instance_id=instance_id, uploaded=not skipped,
|
|
271
283
|
skipped=skipped, image_pushed=False,
|
|
272
284
|
docker_image=docker_image or None,
|
|
273
|
-
digests={"instance.json": digest}
|
|
285
|
+
digests={"instance.json": digest},
|
|
286
|
+
ingest=ingest_result)
|
|
287
|
+
|
|
288
|
+
def _push_custom(self, local_path: Path, dataset: str, version: str, *,
|
|
289
|
+
split: str = "default", push_image: bool = True,
|
|
290
|
+
overwrite: bool = False, register: bool = True,
|
|
291
|
+
verify: bool = True) -> PushResult:
|
|
292
|
+
"""推一个 custom 格式实例:validate → 上传元数据 JSON → (目录输入时) 上传资产。
|
|
293
|
+
|
|
294
|
+
custom 格式的两种输入:
|
|
295
|
+
|
|
296
|
+
- **目录**:SDK 生成最小元数据 JSON(``instance_id`` / ``metadata_model`` /
|
|
297
|
+
``schema_version`` / ``created_at`` / ``updated_at``)上传到 ``{split}/{id}.json``,
|
|
298
|
+
目录内容**原样**逐文件上传到 ``{split}-assets/{id}/`` 前缀下(保持相对路径)。
|
|
299
|
+
- **``.json`` 文件**:校验后**原样**上传到 ``{split}/{id}.json``(无资产)。
|
|
300
|
+
|
|
301
|
+
custom 格式**不参与镜像流程**:``push_image=True`` 时告警并忽略(与 SWE 同口径)。
|
|
302
|
+
"""
|
|
303
|
+
import datetime as _dt
|
|
304
|
+
import warnings as _warnings
|
|
305
|
+
|
|
306
|
+
if push_image:
|
|
307
|
+
_warnings.warn(
|
|
308
|
+
"custom 格式不支持 ACR 镜像推送(push_image=True 已忽略);"
|
|
309
|
+
"如需镜像请用 harbor 格式",
|
|
310
|
+
UserWarning, stacklevel=3)
|
|
311
|
+
|
|
312
|
+
report = check_custom(local_path) # ① 校验(含 instance_id 安全检查)
|
|
313
|
+
instance_id = report.instance_id
|
|
314
|
+
lay = self._layout()
|
|
315
|
+
idx_key = lay.metadata_key(dataset, instance_id, version=version, split=split)
|
|
316
|
+
sts_prefix = lay.sts_prefix(dataset)
|
|
317
|
+
|
|
318
|
+
if report.source == "json":
|
|
319
|
+
# JSON 输入:原样上传
|
|
320
|
+
raw_bytes = local_path.read_bytes()
|
|
321
|
+
meta_digest = "sha256:" + hashlib.sha256(raw_bytes).hexdigest()
|
|
322
|
+
put = self._content.put_object(
|
|
323
|
+
idx_key, raw_bytes, overwrite=overwrite,
|
|
324
|
+
sts_prefix=sts_prefix, dataset=dataset)
|
|
325
|
+
skipped = bool(put.get("skipped"))
|
|
326
|
+
asset_digests: dict[str, str] = {}
|
|
327
|
+
else:
|
|
328
|
+
# 目录输入:生成最小元数据 JSON + 原样上传资产
|
|
329
|
+
now_iso = _dt.datetime.now(_dt.timezone.utc).isoformat(
|
|
330
|
+
timespec="seconds").replace("+00:00", "Z")
|
|
331
|
+
meta_doc = {
|
|
332
|
+
"instance_id": instance_id,
|
|
333
|
+
"metadata_model": "custom",
|
|
334
|
+
"schema_version": CUSTOM_SCHEMA_VERSION,
|
|
335
|
+
"created_at": now_iso,
|
|
336
|
+
"updated_at": now_iso,
|
|
337
|
+
}
|
|
338
|
+
raw_bytes = json.dumps(meta_doc, ensure_ascii=False).encode("utf-8")
|
|
339
|
+
meta_digest = "sha256:" + hashlib.sha256(raw_bytes).hexdigest()
|
|
340
|
+
put = self._content.put_object(
|
|
341
|
+
idx_key, raw_bytes, overwrite=overwrite,
|
|
342
|
+
sts_prefix=sts_prefix, dataset=dataset)
|
|
343
|
+
skipped = bool(put.get("skipped"))
|
|
344
|
+
|
|
345
|
+
# 原样上传目录内所有文件到 {split}-assets/{id}/<relpath>
|
|
346
|
+
asset_prefix = (f"{lay.dataset_root(dataset)}"
|
|
347
|
+
f"{split}-assets/{instance_id}/")
|
|
348
|
+
asset_digests = {}
|
|
349
|
+
for child in sorted(local_path.rglob("*")):
|
|
350
|
+
if not child.is_file():
|
|
351
|
+
continue
|
|
352
|
+
rel = child.relative_to(local_path).as_posix()
|
|
353
|
+
key = asset_prefix + rel
|
|
354
|
+
data = child.read_bytes()
|
|
355
|
+
digest = "sha256:" + hashlib.sha256(data).hexdigest()
|
|
356
|
+
asset_digests[rel] = digest
|
|
357
|
+
aput = self._content.put_object(
|
|
358
|
+
key, data, overwrite=overwrite,
|
|
359
|
+
sts_prefix=sts_prefix, dataset=dataset)
|
|
360
|
+
if not skipped and bool(aput.get("skipped")):
|
|
361
|
+
skipped = True # 任一被 skip 即整体视为 skip
|
|
362
|
+
|
|
363
|
+
# ③ 回读校验(防静默成功)
|
|
364
|
+
if verify and not skipped:
|
|
365
|
+
if not self._content.object_exists(idx_key, sts_prefix=sts_prefix,
|
|
366
|
+
dataset=dataset):
|
|
367
|
+
raise TransportError(
|
|
368
|
+
f"{instance_id}: OSS 回读校验失败——上传后未在 OSS 找到 {idx_key}")
|
|
369
|
+
|
|
370
|
+
# ④ 写库:register=True 时经 ingest 扫描入库(替代缺失的 :commit 收尾)
|
|
371
|
+
ingest_result = self._after_upload(register, dataset, version)
|
|
372
|
+
|
|
373
|
+
return PushResult(instance_id=instance_id, uploaded=not skipped,
|
|
374
|
+
skipped=skipped, image_pushed=False,
|
|
375
|
+
docker_image=None,
|
|
376
|
+
digests={"instance.json": meta_digest},
|
|
377
|
+
ingest=ingest_result)
|
|
274
378
|
|
|
275
379
|
# ── 上架前:把生产副本的 task.toml 指向生产镜像(本地重写,不触网)──────────
|
|
276
380
|
def retarget_prod_content(self, local_dir, dataset: str, version: str,
|
|
@@ -341,6 +445,10 @@ class InstancesClient:
|
|
|
341
445
|
置 True 则不抛,返回列表中失败项以 ``ItemError`` 占位、成功项为 ``PushResult``,
|
|
342
446
|
供上层(如 deliver)对成功子集继续 ingest 并单独汇报失败。结果按输入顺序。
|
|
343
447
|
|
|
448
|
+
``register=True``:整批上传完成后触发**一次** ingest 写库(绝不逐实例
|
|
449
|
+
ingest——N 个实例 N 次全量扫描是 O(N²) 浪费);ingest 结果不在返回列表里,
|
|
450
|
+
需要时由调用方自行调用 ``versions.ingest`` 获取(deliver/CLI 均如此)。
|
|
451
|
+
|
|
344
452
|
``owner_reference`` 整批共用同一来源引用(批量交付通常同源;需要逐实例不同来源时
|
|
345
453
|
请自行循环 ``push``)。CLI 的 push / push-many / deliver 都把它设为必填——交付
|
|
346
454
|
路径必须能追溯上游来源。
|
|
@@ -365,14 +473,20 @@ class InstancesClient:
|
|
|
365
473
|
pass
|
|
366
474
|
|
|
367
475
|
def _one(d):
|
|
476
|
+
# 逐实例强制 register=False:批量 ingest 只在整批结束后执行一次
|
|
368
477
|
return self.push(d, dataset, version,
|
|
369
478
|
owner_reference=owner_reference, split=split,
|
|
370
479
|
push_image=push_image, overwrite=overwrite,
|
|
371
|
-
register=
|
|
480
|
+
register=False, verify=verify,
|
|
372
481
|
image_release_dst_acr=image_release_dst_acr)
|
|
373
482
|
|
|
374
|
-
|
|
375
|
-
|
|
483
|
+
results = map_concurrent(_one, dirs, concurrency=concurrency,
|
|
484
|
+
return_exceptions=continue_on_error)
|
|
485
|
+
# continue_on_error=False 时 map_concurrent 已对失败项抛 ItemError,能走到
|
|
486
|
+
# 这里说明整批成功;True 时对成功子集 ingest(与 deliver 语义一致)。
|
|
487
|
+
if register:
|
|
488
|
+
self._after_upload(True, dataset, version)
|
|
489
|
+
return results
|
|
376
490
|
|
|
377
491
|
def pull(self, dataset: str, version: str, instance_id: str, dest_dir,
|
|
378
492
|
split: str = "default", *, verify: bool = True) -> Path:
|
|
@@ -216,10 +216,42 @@ class VersionsClient:
|
|
|
216
216
|
f"&version={quote(version, safe='')}{extra}")
|
|
217
217
|
return data.get("status", "")
|
|
218
218
|
|
|
219
|
+
# GET /apis/v1/datasets/splits/detail:split-first 叶详情(真实 split 记录)。
|
|
220
|
+
_SPLITS = "/apis/v1/datasets/splits"
|
|
221
|
+
|
|
222
|
+
def split_detail(self, dataset: str, version: str, split: str, *,
|
|
223
|
+
environment: str | None = None) -> dict:
|
|
224
|
+
"""split 级读 API(``GET /apis/v1/datasets/splits/detail``)。
|
|
225
|
+
|
|
226
|
+
直接读 split-first 记录:``status``/``published_at``/``storage_path``
|
|
227
|
+
都是真实值,不经过 version_first 兼容投影。版本级读接口
|
|
228
|
+
(:meth:`list`/:meth:`status`)只保证 version_first 语义;要看
|
|
229
|
+
split_first 的发布状态与目标路径请用本方法。
|
|
230
|
+
|
|
231
|
+
``environment`` 缺省由 profile 派生(与读路径对称);显式传入可覆盖,
|
|
232
|
+
例如发布后查 ``environment="online"`` 的 published 状态与目标路径。
|
|
233
|
+
"""
|
|
234
|
+
if not (split or "").strip():
|
|
235
|
+
raise SchemaError("split is required for split detail")
|
|
236
|
+
mm = normalize_metadata_model(
|
|
237
|
+
self._p.metadata_model if self._p is not None else None)
|
|
238
|
+
env = (environment or "").strip()
|
|
239
|
+
if not env and self._p is not None:
|
|
240
|
+
env = (getattr(self._p, "dataset_environment", "") or "").strip()
|
|
241
|
+
query = (f"dataset_name={quote(dataset, safe='')}"
|
|
242
|
+
f"&version={quote(version or '', safe='')}"
|
|
243
|
+
f"&split={quote(split, safe='')}"
|
|
244
|
+
f"&metadata_model={quote(mm, safe='')}")
|
|
245
|
+
if env:
|
|
246
|
+
query += f"&environment={quote(env, safe='')}"
|
|
247
|
+
data = self._t.get(f"{self._SPLITS}/detail?{query}")
|
|
248
|
+
return data if isinstance(data, dict) else {}
|
|
249
|
+
|
|
219
250
|
def create(self, dataset: str, version: str, *,
|
|
220
251
|
storage_path: str | None = None,
|
|
221
252
|
splits: list[str] | None = None, status: str = "draft",
|
|
222
|
-
storage_type: str = "oss"
|
|
253
|
+
storage_type: str = "oss",
|
|
254
|
+
environment: str | None = None) -> DatasetVersion:
|
|
223
255
|
"""显式创建 DatasetVersion(``POST /apis/v1/datasets/versions``)。
|
|
224
256
|
|
|
225
257
|
通常 version 由 ``ingest`` 扫描 OSS 时自动建;本方法用于需要**先建空版本壳**
|
|
@@ -232,20 +264,42 @@ class VersionsClient:
|
|
|
232
264
|
splits : 可选 split 名列表;缺省由 apiserver 按 storage_path 推导。
|
|
233
265
|
status : 版本状态,缺省 ``draft``(合法值见 VERSION_STATUS_VALUES)。
|
|
234
266
|
storage_type : 存储类型,缺省 ``oss``(当前仅支持)。
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
267
|
+
environment : 可选。显式传入时原样下发;缺省时由 storage_env 派生
|
|
268
|
+
(dev/internal-dev/external-dev→pre,prod→online)。profile 未显式配置
|
|
269
|
+
storage_env 但配置了 cluster 时,会经 repo-config 解析出绑定的
|
|
270
|
+
storage_env 再派生(与 ingest 对称);显式 environment 与派生值冲突时
|
|
271
|
+
前置拒绝。
|
|
239
272
|
|
|
240
273
|
run_type 不在建版本时设置(apiserver create 不收该字段):建好后用
|
|
241
274
|
``set_run_type`` / ``update`` 打标,或走 ``ingest(run_type=...)``。
|
|
242
275
|
"""
|
|
243
276
|
if not version:
|
|
244
277
|
raise SchemaError("version is required")
|
|
278
|
+
# 与 ingest 对称(问题1 彻底修复):storage_env 未在 profile 里显式配置时,
|
|
279
|
+
# 经 repo-config 解析出与 cluster 绑定的 storage_env 再派生 environment;
|
|
280
|
+
# 同时复用**同一次**响应里的 bucket/prefix 计算 storage_path,避免出现
|
|
281
|
+
# 「create 落 online、push/ingest 落 pre」的读写环境撕裂。
|
|
282
|
+
selected_storage_env = ""
|
|
283
|
+
bound_bucket = bound_prefix = ""
|
|
284
|
+
if self._p is not None:
|
|
285
|
+
_, selected_storage_env, bound_bucket, bound_prefix = \
|
|
286
|
+
self._ingest_selectors(None)
|
|
287
|
+
env = (environment or "").strip()
|
|
288
|
+
if not env and self._p is not None:
|
|
289
|
+
env = (getattr(self._p, "dataset_environment", "") or "").strip()
|
|
290
|
+
derived_env = dataset_environment_for(selected_storage_env)
|
|
291
|
+
if env and derived_env and env != derived_env:
|
|
292
|
+
raise SchemaError(
|
|
293
|
+
f"environment={env!r} 与 storage_env="
|
|
294
|
+
f"{selected_storage_env!r} 派生出的 {derived_env!r} 冲突:"
|
|
295
|
+
f"请只提供其中一个,或把 storage_env 改成与目标环境一致的值")
|
|
296
|
+
if not env:
|
|
297
|
+
env = derived_env
|
|
245
298
|
if storage_path is None and self._p is not None:
|
|
246
|
-
bucket = getattr(self._p, "oss_bucket", "") or ""
|
|
299
|
+
bucket = (getattr(self._p, "oss_bucket", "") or "") or bound_bucket
|
|
247
300
|
if bucket:
|
|
248
|
-
storage_path = self._layout(
|
|
301
|
+
storage_path = self._layout(
|
|
302
|
+
oss_prefix=bound_prefix or None).ingest_storage_path(
|
|
249
303
|
bucket, dataset, version)
|
|
250
304
|
if not storage_path or not storage_path.startswith("oss://"):
|
|
251
305
|
raise SchemaError(
|
|
@@ -264,6 +318,8 @@ class VersionsClient:
|
|
|
264
318
|
"storage_type": storage_type, "status": status}
|
|
265
319
|
if splits:
|
|
266
320
|
body["splits"] = splits
|
|
321
|
+
if env:
|
|
322
|
+
body["environment"] = env
|
|
267
323
|
data = self._t.post(
|
|
268
324
|
f"{self._BASE}?dataset_name={quote(dataset, safe='')}", body)
|
|
269
325
|
return DatasetVersion(**_pick(data))
|
|
@@ -8,11 +8,15 @@ from __future__ import annotations
|
|
|
8
8
|
|
|
9
9
|
import gzip
|
|
10
10
|
import hashlib
|
|
11
|
+
import re
|
|
11
12
|
import tarfile
|
|
12
13
|
from pathlib import Path
|
|
13
14
|
|
|
14
15
|
_EXCLUDE_REL = "environment/image.tar.gz"
|
|
15
16
|
|
|
17
|
+
# 32 位 hex(无前缀)= apiserver ingest 扫描写入的 OSS ETag(简单上传即 md5)。
|
|
18
|
+
_MD5_HEX_RE = re.compile(r"^[0-9a-fA-F]{32}$")
|
|
19
|
+
|
|
16
20
|
|
|
17
21
|
def sha256_file(path: str | Path) -> str:
|
|
18
22
|
"""流式计算文件 sha-256,返回 'sha256:<hex>'。"""
|
|
@@ -23,6 +27,33 @@ def sha256_file(path: str | Path) -> str:
|
|
|
23
27
|
return "sha256:" + h.hexdigest()
|
|
24
28
|
|
|
25
29
|
|
|
30
|
+
def md5_file(path: str | Path) -> str:
|
|
31
|
+
"""流式计算文件 md5,返回 32 位小写 hex。"""
|
|
32
|
+
h = hashlib.md5()
|
|
33
|
+
with open(path, "rb") as f:
|
|
34
|
+
for chunk in iter(lambda: f.read(1 << 20), b""):
|
|
35
|
+
h.update(chunk)
|
|
36
|
+
return h.hexdigest()
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def verify_file_digest(path: str | Path, expected: str | None) -> bool:
|
|
40
|
+
"""校验本地文件与 manifest ``content_digest`` 一致(两种生产者格式)。
|
|
41
|
+
|
|
42
|
+
- SDK push/commit 路径写 ``sha256:<hex>`` → 按 sha256 比对;
|
|
43
|
+
- apiserver ingest 扫描路径写 OSS ``ETag``(简单上传即 32 位 md5 hex,
|
|
44
|
+
无前缀)→ 按 md5 比对(对齐服务端契约,避免 pull(verify=True) 误报)。
|
|
45
|
+
空/缺失 digest(历史资产合并条目)视为跳过校验,返回 True。
|
|
46
|
+
"""
|
|
47
|
+
if not expected:
|
|
48
|
+
return True
|
|
49
|
+
expected = expected.strip().strip('"')
|
|
50
|
+
if not expected:
|
|
51
|
+
return True
|
|
52
|
+
if _MD5_HEX_RE.match(expected):
|
|
53
|
+
return md5_file(path) == expected.lower()
|
|
54
|
+
return sha256_file(path) == expected
|
|
55
|
+
|
|
56
|
+
|
|
26
57
|
def _normalize(ti: tarfile.TarInfo) -> tarfile.TarInfo:
|
|
27
58
|
"""归一化 tar 条目元数据,保证跨时间/机器可复现。"""
|
|
28
59
|
ti.mtime = 0
|
|
@@ -37,9 +37,13 @@ from .errors import SchemaError
|
|
|
37
37
|
|
|
38
38
|
MODEL_VERSION_FIRST = "version_first"
|
|
39
39
|
MODEL_SPLIT_FIRST = "split_first"
|
|
40
|
+
#: ``custom`` 是用户自定义格式的元数据模型:OSS 路径沿用 split_first 约定
|
|
41
|
+
#: ({split}/{id}.json 元数据 + {split}-assets/{id}/ 原样目录资产),SDK 不校验具体
|
|
42
|
+
#: 字段(仅强约束 instance_id),元数据 freeform。
|
|
43
|
+
MODEL_CUSTOM = "custom"
|
|
40
44
|
#: v0.8 起默认 split_first(v0.7.x 行为等价于显式传 version_first)。
|
|
41
45
|
DEFAULT_METADATA_MODEL = MODEL_SPLIT_FIRST
|
|
42
|
-
_MODELS = (MODEL_VERSION_FIRST, MODEL_SPLIT_FIRST)
|
|
46
|
+
_MODELS = (MODEL_VERSION_FIRST, MODEL_SPLIT_FIRST, MODEL_CUSTOM)
|
|
43
47
|
|
|
44
48
|
# ACR tag 合法字符集与长度(OCI/Docker 约定):首字符须为字母数字或下划线,
|
|
45
49
|
# 其余可含字母数字、下划线、点、连字符,总长 **< 105**(ACR tag 硬上限 128,
|
|
@@ -126,7 +130,9 @@ class Layout:
|
|
|
126
130
|
|
|
127
131
|
@property
|
|
128
132
|
def split_first(self) -> bool:
|
|
129
|
-
|
|
133
|
+
# custom 沿用 split_first 的路径约定({split}/{id}.json +
|
|
134
|
+
# {split}-assets/{id}/ 兄弟目录),故 path 构造上一视同仁。
|
|
135
|
+
return self.model in (MODEL_SPLIT_FIRST, MODEL_CUSTOM)
|
|
130
136
|
|
|
131
137
|
# ── 目录基座 ──
|
|
132
138
|
def dataset_root(self, dataset: str) -> str:
|
|
@@ -246,6 +246,9 @@ class PushResult:
|
|
|
246
246
|
image_pushed: bool = False
|
|
247
247
|
docker_image: str | None = None
|
|
248
248
|
digests: dict[str, str] = field(default_factory=dict)
|
|
249
|
+
# register=True 时经 ingest 写库:ingest job 摘要(ingest_id/status/stats);
|
|
250
|
+
# register=False(只传 OSS)时为 None。
|
|
251
|
+
ingest: dict | None = None
|
|
249
252
|
|
|
250
253
|
|
|
251
254
|
# ── RolloutReport(质检 report,服务下游 verl 训练;键名对齐 verl 无下划线口径) ──
|
|
@@ -75,9 +75,11 @@ class Repo:
|
|
|
75
75
|
metadata_model=normalize_metadata_model(self.profile.metadata_model))
|
|
76
76
|
self.datasets = DatasetsClient(self.transport)
|
|
77
77
|
self.versions = VersionsClient(self.transport, self.profile)
|
|
78
|
+
# push(register=True) 的上传后写库经 ingest 完成(:commit 端点缺失),
|
|
79
|
+
# 注入 versions.ingest 作为 ingester 回调。
|
|
78
80
|
self.instances = InstancesClient(
|
|
79
81
|
self.transport, self.profile, self._content, self._image,
|
|
80
|
-
self._registry)
|
|
82
|
+
self._registry, ingester=self.versions.ingest)
|
|
81
83
|
self.reports = ReportsClient(self._content, self.profile)
|
|
82
84
|
# 独立镜像能力(v0.8 新增):push / pull / copy / exists 不再依赖实例目录。
|
|
83
85
|
self.images = ImageClient(self.transport, self.profile, self._image)
|
|
@@ -134,8 +136,16 @@ class Repo:
|
|
|
134
136
|
failed = [r for r in results if isinstance(r, ItemError)]
|
|
135
137
|
job = None
|
|
136
138
|
if ingest and pushed: # 仅在有成功上传时才 ingest
|
|
137
|
-
|
|
138
|
-
|
|
139
|
+
# 透传 split 给 ingest:用户若未通过 ingest_kwargs 显式指定 splits/split,
|
|
140
|
+
# 且 split 非默认值,则自动透传,避免 ingest 扫描所有 split 目录时
|
|
141
|
+
# 对未交付的空 split 报 92005(split path has no instance files)。
|
|
142
|
+
ingest_split_kwargs = {}
|
|
143
|
+
if "splits" not in ingest_kwargs and "split" not in ingest_kwargs:
|
|
144
|
+
if split and split != "default":
|
|
145
|
+
ingest_split_kwargs["split"] = split
|
|
146
|
+
job = self.versions.ingest(
|
|
147
|
+
dataset, version, publish=publish,
|
|
148
|
+
**ingest_split_kwargs, **ingest_kwargs)
|
|
139
149
|
return {"pushed": pushed, "failed": failed, "ingest": job}
|
|
140
150
|
|
|
141
151
|
# ── 源到目标交付:上传 → 元数据入库 → 发布 workflow ────────────────────────
|
|
@@ -16,7 +16,7 @@ from .._routing import (
|
|
|
16
16
|
_is_reachability_error,
|
|
17
17
|
_plan_oss_route,
|
|
18
18
|
)
|
|
19
|
-
from ..content import sha256_file
|
|
19
|
+
from ..content import sha256_file, verify_file_digest
|
|
20
20
|
from ..errors import DigestMismatch, PrivateNetworkRequired, TransportError
|
|
21
21
|
from ..retry import retry_transient
|
|
22
22
|
from .base import BaseContentStore
|
|
@@ -315,12 +315,10 @@ class OssContentStore(BaseContentStore):
|
|
|
315
315
|
dataset,
|
|
316
316
|
lambda bucket: bucket.get_object_to_file(key, str(dest)),
|
|
317
317
|
)
|
|
318
|
-
if expected_digest is not None:
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
f"{key}: content digest mismatch "
|
|
323
|
-
f"(expected {expected_digest}, got {actual})")
|
|
318
|
+
if expected_digest is not None and not verify_file_digest(dest, expected_digest):
|
|
319
|
+
raise DigestMismatch(
|
|
320
|
+
f"{key}: content digest mismatch "
|
|
321
|
+
f"(expected {expected_digest}, got {sha256_file(dest)})")
|
|
324
322
|
return dest
|
|
325
323
|
|
|
326
324
|
def object_exists(self, key: str, *, sts_prefix: str | None = None,
|
|
@@ -413,10 +411,8 @@ class OssContentStore(BaseContentStore):
|
|
|
413
411
|
dataset,
|
|
414
412
|
lambda bucket: bucket.get_object_to_file(tgz_key, str(dest)),
|
|
415
413
|
)
|
|
416
|
-
if expected_digest is not None:
|
|
417
|
-
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
f"{instance_id}: content digest mismatch "
|
|
421
|
-
f"(expected {expected_digest}, got {actual})")
|
|
414
|
+
if expected_digest is not None and not verify_file_digest(dest, expected_digest):
|
|
415
|
+
raise DigestMismatch(
|
|
416
|
+
f"{instance_id}: content digest mismatch "
|
|
417
|
+
f"(expected {expected_digest}, got {sha256_file(dest)})")
|
|
422
418
|
return dest
|
|
@@ -37,17 +37,15 @@ def _unimplemented(endpoint: str, hint: str):
|
|
|
37
37
|
|
|
38
38
|
|
|
39
39
|
def _q(dataset: str, version: str, split: str | None = None,
|
|
40
|
-
instance_id: str | None = None) -> str:
|
|
40
|
+
instance_id: str | None = None, *, keep_default_split: bool = False) -> str:
|
|
41
41
|
# 注意:apiserver 端 datasetNameFromQuery 只认 dataset_name(不是 dataset);
|
|
42
|
-
# 用 dataset= 会 404
|
|
42
|
+
# 用 dataset= 会 404/"dataset_name is required"。此处务必对齐服务端契约。
|
|
43
43
|
parts = [f"dataset_name={quote(dataset, safe='')}",
|
|
44
44
|
f"version={quote(version, safe='')}"]
|
|
45
|
-
if split and split != "default":
|
|
46
|
-
# "default"
|
|
47
|
-
#
|
|
48
|
-
#
|
|
49
|
-
# 参数,让 apiserver 按"省略 split→跨 split 匹配(含空 split)"命中。否则默认
|
|
50
|
-
# split="default" 会原样发 split=default,与库里 split="" 不匹配→查不到(实测 0 命中)。
|
|
45
|
+
if split and (split != "default" or keep_default_split):
|
|
46
|
+
# version_first / 未声明 metadata_model:"default" 是"无 split"的哨兵 → 省略 split
|
|
47
|
+
# 参数让 apiserver 跨 split 匹配(含空 split)。split_first:"default" 是真实 split
|
|
48
|
+
# 名 → 必须显式下发,否则 apiserver 在 split_first 语义下报 92005 "split is required"。
|
|
51
49
|
parts.append(f"split={quote(split, safe='')}")
|
|
52
50
|
if instance_id:
|
|
53
51
|
parts.append(f"instance_id={quote(instance_id, safe='')}")
|
|
@@ -73,7 +71,8 @@ class ApiMetadataRegistry(BaseMetadataRegistry):
|
|
|
73
71
|
split_first,不显式下发就会用旧语义去查 split-first 数据。environment 为空时
|
|
74
72
|
省略(服务端按 online 缺省,保留未配置 storage_env 的旧行为)。
|
|
75
73
|
"""
|
|
76
|
-
base = _q(dataset, version, split, instance_id
|
|
74
|
+
base = _q(dataset, version, split, instance_id,
|
|
75
|
+
keep_default_split=(self._metadata_model == "split_first"))
|
|
77
76
|
extra = []
|
|
78
77
|
if self._metadata_model:
|
|
79
78
|
extra.append(f"metadata_model={quote(self._metadata_model, safe='')}")
|
|
@@ -99,7 +98,11 @@ class ApiMetadataRegistry(BaseMetadataRegistry):
|
|
|
99
98
|
据 caller 派生),故不进 body。
|
|
100
99
|
"""
|
|
101
100
|
meta.validate()
|
|
102
|
-
|
|
101
|
+
if self._metadata_model == "split_first":
|
|
102
|
+
# split_first: "default" 是真实 split 名,不作哨兵归一 → 必须显式下发
|
|
103
|
+
split = meta.split or ""
|
|
104
|
+
else:
|
|
105
|
+
split = "" if meta.split == "default" else (meta.split or "")
|
|
103
106
|
body: dict = {"instance_id": meta.instance_id, "split": split}
|
|
104
107
|
if meta.manifest:
|
|
105
108
|
body["manifest"] = [m.to_dict() for m in meta.manifest]
|
|
@@ -1,8 +1,10 @@
|
|
|
1
|
-
"""validate.py — Instance 本地校验(Layout / SWE JSON),纯本地无网络。
|
|
1
|
+
"""validate.py — Instance 本地校验(Layout / SWE JSON / Custom),纯本地无网络。
|
|
2
2
|
|
|
3
|
-
|
|
4
|
-
- harbor
|
|
3
|
+
三种格式自动检测(detect_format):
|
|
4
|
+
- harbor(目录 + task.toml):instruction.md · task.toml · environment/Dockerfile · tests/test.sh|test.ps1
|
|
5
5
|
- swe(单 JSON 文件):instance_id + docker_image 必备;数据和元数据合一
|
|
6
|
+
- custom(目录无 task.toml / 单 JSON 文件 freeform):仅强约束 instance_id;
|
|
7
|
+
元数据 freeform,push 时目录输入由 SDK 生成最小 JSON、JSON 输入原样上传
|
|
6
8
|
|
|
7
9
|
测试脚本兼容 Linux(tests/test.sh)与 Windows(tests/test.ps1),二者至少其一。
|
|
8
10
|
"""
|
|
@@ -50,21 +52,25 @@ def validate_instance_id(instance_id: str, source: str = "") -> None:
|
|
|
50
52
|
|
|
51
53
|
# ── 格式检测 ─────────────────────────────────────────────────────────────────
|
|
52
54
|
|
|
53
|
-
def detect_format(local_path: str | Path) -> Literal["harbor", "swe"]:
|
|
54
|
-
"""
|
|
55
|
+
def detect_format(local_path: str | Path) -> Literal["harbor", "swe", "custom"]:
|
|
56
|
+
"""自动检测输入格式:
|
|
55
57
|
|
|
56
|
-
|
|
58
|
+
- 目录 + 含 task.toml → harbor
|
|
59
|
+
- 目录 + 不含 task.toml → custom(用户自定义格式,SDK 不校验具体字段)
|
|
60
|
+
- .json 文件 → swe
|
|
61
|
+
|
|
62
|
+
不做内容校验(那是 check_layout / check_swe_json / check_custom 的责任)。
|
|
57
63
|
"""
|
|
58
64
|
p = Path(local_path)
|
|
59
65
|
if p.is_dir():
|
|
60
|
-
return "harbor"
|
|
66
|
+
return "harbor" if (p / "task.toml").is_file() else "custom"
|
|
61
67
|
if p.is_file() and p.suffix.lower() == ".json":
|
|
62
68
|
return "swe"
|
|
63
69
|
if not p.exists():
|
|
64
70
|
raise LayoutError(f"path does not exist: {p}")
|
|
65
71
|
raise SchemaError(
|
|
66
72
|
f"unrecognized instance format: {p} "
|
|
67
|
-
f"(expected a directory for harbor or a .json file for swe)")
|
|
73
|
+
f"(expected a directory for harbor/custom or a .json file for swe)")
|
|
68
74
|
|
|
69
75
|
|
|
70
76
|
# ── Harbor 格式校验 ──────────────────────────────────────────────────────────
|
|
@@ -204,3 +210,70 @@ def check_swe_json(json_path: str | Path) -> SWEValidationReport:
|
|
|
204
210
|
return SWEValidationReport(
|
|
205
211
|
instance_id=instance_id, ok=True, warnings=warnings,
|
|
206
212
|
metadata=metadata, raw_data=data)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
# ── Custom 格式校验(用户自定义,freeform) ─────────────────────────────────
|
|
216
|
+
|
|
217
|
+
CUSTOM_SCHEMA_VERSION = "1.0"
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
@dataclass
|
|
221
|
+
class CustomValidationReport:
|
|
222
|
+
"""Custom 格式校验结果。"""
|
|
223
|
+
instance_id: str
|
|
224
|
+
format: str = "custom"
|
|
225
|
+
ok: bool = True
|
|
226
|
+
metadata: dict = field(default_factory=dict) # 目录输入时 SDK 生成;JSON 输入时原样解析
|
|
227
|
+
raw_data: dict = field(default_factory=dict) # JSON 输入时的原始数据;目录输入时为空
|
|
228
|
+
source: str = "directory" # "directory" 或 "json"
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def check_custom(local_path: str | Path) -> CustomValidationReport:
|
|
232
|
+
"""校验 custom 格式输入。
|
|
233
|
+
|
|
234
|
+
两种输入形态:
|
|
235
|
+
|
|
236
|
+
- 目录:``instance_id`` = 目录名(强约束 charset),元数据由 SDK push 时生成
|
|
237
|
+
(最小 schema:``instance_id`` / ``metadata_model`` / ``schema_version`` /
|
|
238
|
+
``created_at`` / ``updated_at``)。
|
|
239
|
+
- ``.json`` 文件:必须含 ``instance_id`` 字段且与文件名 stem 一致;其余字段
|
|
240
|
+
freeform,push 时原样上传。
|
|
241
|
+
"""
|
|
242
|
+
p = Path(local_path)
|
|
243
|
+
if p.is_dir():
|
|
244
|
+
instance_id = p.name
|
|
245
|
+
validate_instance_id(instance_id, source=str(p))
|
|
246
|
+
return CustomValidationReport(
|
|
247
|
+
instance_id=instance_id, source="directory")
|
|
248
|
+
if p.is_file() and p.suffix.lower() == ".json":
|
|
249
|
+
try:
|
|
250
|
+
with open(p, "r", encoding="utf-8") as f:
|
|
251
|
+
data = json.load(f)
|
|
252
|
+
except UnicodeDecodeError as e:
|
|
253
|
+
raise SchemaError(
|
|
254
|
+
f"{p.name}: not valid UTF-8: {e}") from e
|
|
255
|
+
except json.JSONDecodeError as e:
|
|
256
|
+
raise SchemaError(f"{p.name}: invalid JSON: {e}") from e
|
|
257
|
+
if not isinstance(data, dict):
|
|
258
|
+
raise SchemaError(
|
|
259
|
+
f"{p.name}: JSON root must be an object, got {type(data).__name__}")
|
|
260
|
+
raw_id = data.get("instance_id")
|
|
261
|
+
if not raw_id or (isinstance(raw_id, str) and not raw_id.strip()):
|
|
262
|
+
raise SchemaError(f"{p.name}: required field 'instance_id' missing or empty")
|
|
263
|
+
if not isinstance(raw_id, str):
|
|
264
|
+
raise SchemaError(
|
|
265
|
+
f"{p.name}: instance_id must be a string, got {type(raw_id).__name__}")
|
|
266
|
+
instance_id = raw_id.strip()
|
|
267
|
+
validate_instance_id(instance_id, source=p.name)
|
|
268
|
+
stem = p.stem
|
|
269
|
+
if stem != instance_id:
|
|
270
|
+
raise SchemaError(
|
|
271
|
+
f"{p.name}: filename stem '{stem}' does not match instance_id "
|
|
272
|
+
f"'{instance_id}' — file must be named '{instance_id}.json'")
|
|
273
|
+
return CustomValidationReport(
|
|
274
|
+
instance_id=instance_id, source="json",
|
|
275
|
+
metadata=dict(data), raw_data=dict(data))
|
|
276
|
+
if not p.exists():
|
|
277
|
+
raise LayoutError(f"path does not exist: {p}")
|
|
278
|
+
raise SchemaError(
|
|
279
|
+
f"custom format expects a directory or .json file, got: {p}")
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{instance_repo-1.0.7.dev0 → instance_repo-1.0.9}/instance_repo.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|