instance-repo 1.1.1.dev0__tar.gz → 1.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/PKG-INFO +3 -2
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/README.md +2 -1
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/__init__.py +1 -1
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/artifacts.py +8 -4
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/cli.py +101 -26
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/instances.py +66 -11
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/splits.py +52 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/models.py +4 -3
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo.egg-info/PKG-INFO +3 -2
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/pyproject.toml +1 -1
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/_routing.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/_site_defaults.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/bootstrap.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/cache.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/__init__.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/benchmarks.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/config.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/datasets.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/images.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/reports.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/scaffold.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/versions.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/clients/workflows.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/concurrency.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/content.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/endpoints.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/errors.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/image.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/layout.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/loader.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/paths.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/release.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/repo.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/retry.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/store/__init__.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/store/acr.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/store/base.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/store/oss.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/store/registry.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/tasktoml.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/transport.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo/validate.py +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo.egg-info/SOURCES.txt +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo.egg-info/dependency_links.txt +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo.egg-info/entry_points.txt +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo.egg-info/requires.txt +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo.egg-info/top_level.txt +0 -0
- {instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: instance-repo
|
|
3
|
-
Version: 1.1.
|
|
3
|
+
Version: 1.1.2
|
|
4
4
|
Summary: InstanceRepo SDK — 评测 Instance 统一存储客户端(SDK-only;控制面由 apiserver 承载)
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -149,9 +149,10 @@ keys = repo.list_user_data(uid)
|
|
|
149
149
|
irepo --help
|
|
150
150
|
# 全局参数:--api-env / --api-base / --cluster / --storage-env / --metadata-model / --profile
|
|
151
151
|
# 无 --api-key:凭据只从 AP_API_KEY(旧别名 INSTANCEREPO_TOKEN)读取,避免进 shell history
|
|
152
|
-
#
|
|
152
|
+
# 20 个子命令:validate / push / push-many / deliver / pull / list / get
|
|
153
153
|
# publish / grant / feedback / report / whoami / user-data
|
|
154
154
|
# repo-config / create / claim / version / image
|
|
155
|
+
# split(list/get/set-run-type)/ dataset(list/get/update/access/revoke/benchmarks/workflows)
|
|
155
156
|
irepo image ref --dataset alibaba/mybench --instance-id inst-1 --split test
|
|
156
157
|
irepo image pull '<acr>/<ns>/mybench:test-inst-1' ./out.tar
|
|
157
158
|
```
|
|
@@ -136,9 +136,10 @@ keys = repo.list_user_data(uid)
|
|
|
136
136
|
irepo --help
|
|
137
137
|
# 全局参数:--api-env / --api-base / --cluster / --storage-env / --metadata-model / --profile
|
|
138
138
|
# 无 --api-key:凭据只从 AP_API_KEY(旧别名 INSTANCEREPO_TOKEN)读取,避免进 shell history
|
|
139
|
-
#
|
|
139
|
+
# 20 个子命令:validate / push / push-many / deliver / pull / list / get
|
|
140
140
|
# publish / grant / feedback / report / whoami / user-data
|
|
141
141
|
# repo-config / create / claim / version / image
|
|
142
|
+
# split(list/get/set-run-type)/ dataset(list/get/update/access/revoke/benchmarks/workflows)
|
|
142
143
|
irepo image ref --dataset alibaba/mybench --instance-id inst-1 --split test
|
|
143
144
|
irepo image pull '<acr>/<ns>/mybench:test-inst-1' ./out.tar
|
|
144
145
|
```
|
|
@@ -160,15 +160,19 @@ def fetch_job_info(api_key: str, job_id: str, cluster: str,
|
|
|
160
160
|
|
|
161
161
|
def fetch_group_job_ids(api_key: str, group_id: str, cluster: str,
|
|
162
162
|
ap_endpoint: str | None = None) -> list[str]:
|
|
163
|
-
"""调 AP GET /groups/{group_id}/jobs 分页收集全部 job_id。
|
|
164
|
-
|
|
165
|
-
|
|
163
|
+
"""调 AP GET /api/groups/{group_id}/jobs 分页收集全部 job_id。
|
|
164
|
+
|
|
165
|
+
endpoint 解析与 fetch_job_info 完全一致(复用 _ap_endpoint),
|
|
166
|
+
且 URL 必须带 /api 前缀:AP 网关的裸 /groups/{id}/jobs 路径返回
|
|
167
|
+
控制台 HTML 页(HTTP 200),json 解析会报错(真机实测)。
|
|
168
|
+
"""
|
|
169
|
+
endpoint = _ap_endpoint(ap_endpoint)
|
|
166
170
|
job_ids: list[str] = []
|
|
167
171
|
skip = 0
|
|
168
172
|
limit = 500
|
|
169
173
|
|
|
170
174
|
while True:
|
|
171
|
-
url = (f"{endpoint}/groups/{group_id}/jobs"
|
|
175
|
+
url = (f"{endpoint}/api/groups/{group_id}/jobs"
|
|
172
176
|
f"?skip={skip}&limit={limit}&include_post_process=false")
|
|
173
177
|
headers = {
|
|
174
178
|
"X-API-Key": api_key,
|
|
@@ -284,24 +284,27 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
284
284
|
gt.add_argument("--split", default=None,
|
|
285
285
|
help="实例所在 split;缺省为跨 split 检索")
|
|
286
286
|
|
|
287
|
-
ver = sub.add_parser("version", help="版本管理(create/list/status/update/set-run-type)")
|
|
287
|
+
ver = sub.add_parser("version", help="版本管理(create/list/status/update/set-run-type/ingest)")
|
|
288
288
|
ver.add_argument("action",
|
|
289
289
|
choices=["create", "list", "status", "update",
|
|
290
|
-
"set-run-type", "get"])
|
|
290
|
+
"set-run-type", "get", "ingest"])
|
|
291
291
|
ver.add_argument("ref", nargs="?",
|
|
292
|
-
help="create/list: {L1}/{L2};"
|
|
292
|
+
help="create/list/ingest: {L1}/{L2};"
|
|
293
293
|
"status/update/set-run-type: {L1}/{L2}/{version}"
|
|
294
294
|
"(或用 --dataset[/--version])")
|
|
295
295
|
ver.add_argument("--dataset", default=None, help="{L1}/{L2}")
|
|
296
296
|
ver.add_argument("--version", default=None,
|
|
297
|
-
help="create:
|
|
297
|
+
help="create/ingest: 要处理的版本号(split_first 可省略);"
|
|
298
298
|
"status/update/set-run-type: 目标版本(寻址用)")
|
|
299
299
|
ver.add_argument("--storage-path", default=None,
|
|
300
300
|
help="create 用:oss://{bucket}/{prefix}")
|
|
301
301
|
ver.add_argument("--storage-type", choices=["oss"], default="oss")
|
|
302
|
-
ver.add_argument("--splits", default=None,
|
|
302
|
+
ver.add_argument("--splits", default=None,
|
|
303
|
+
help="逗号分隔 split 名(create/ingest 用;ingest 与 --split 互斥)")
|
|
303
304
|
ver.add_argument("--split", default=None,
|
|
304
|
-
help="status 用:查 split
|
|
305
|
+
help="status 用:查 split 级真实记录;"
|
|
306
|
+
"ingest 用:只 ingest 该 split(与 --splits 互斥)——"
|
|
307
|
+
"version 下 split 很多时强烈建议传,避免全量扫描拖垮事务")
|
|
305
308
|
ver.add_argument("--status", choices=_STATUS_CHOICES, default=None,
|
|
306
309
|
help="update 用:版本状态流转")
|
|
307
310
|
ver.add_argument("--run-type", choices=["train", "eval"], default=None,
|
|
@@ -309,19 +312,36 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
309
312
|
ver.add_argument("--split-run-types", default=None,
|
|
310
313
|
help="split 级打标,格式 name=train,name2=eval(与 --run-type 互斥)")
|
|
311
314
|
ver.add_argument("--environment", default=None,
|
|
312
|
-
help="get 用:覆盖 profile 派生的 environment(pre/online)")
|
|
315
|
+
help="get/ingest 用:覆盖 profile 派生的 environment(pre/online)")
|
|
316
|
+
ver.add_argument("--metadata-model", default=None,
|
|
317
|
+
help="ingest 用:显式指定元数据模型(split_first/version_first)")
|
|
318
|
+
ver.add_argument("--dry-run", action="store_true",
|
|
319
|
+
help="ingest 用:只预览扫描结果不落库")
|
|
320
|
+
ver.add_argument("--publish", action="store_true",
|
|
321
|
+
help="ingest 用:完成后把 version 置 published")
|
|
322
|
+
ver.add_argument("--no-wait", action="store_true",
|
|
323
|
+
help="ingest 用:立即返回 job 不轮询终态")
|
|
324
|
+
ver.add_argument("--poll-sec", type=float, default=5.0, dest="poll_sec",
|
|
325
|
+
help="ingest 用:轮询间隔(秒)")
|
|
326
|
+
ver.add_argument("--timeout", type=float, default=1800.0, dest="timeout_sec",
|
|
327
|
+
help="ingest 用:轮询超时(秒)")
|
|
328
|
+
ver.add_argument("--allow-layout-inference", action="store_true",
|
|
329
|
+
help="ingest 用:放弃「split_first 无 version 必须给 splits」"
|
|
330
|
+
"前置检查,交给服务端推断布局")
|
|
331
|
+
ver.add_argument("--cluster", default=None,
|
|
332
|
+
help="ingest 用:运行时 cluster slug(覆盖 profile)")
|
|
313
333
|
|
|
314
334
|
# ── 1.0.10 控制面管理子命令 ─────────────────────────────────────────────────
|
|
315
|
-
sp = sub.add_parser("split", help="split 记录管理(list/get/set-run-type)")
|
|
316
|
-
sp.add_argument("action", choices=["list", "get", "set-run-type"])
|
|
335
|
+
sp = sub.add_parser("split", help="split 记录管理(list/get/create/set-run-type)")
|
|
336
|
+
sp.add_argument("action", choices=["list", "get", "create", "set-run-type"])
|
|
317
337
|
sp.add_argument("ref", nargs="?",
|
|
318
|
-
help="list: {L1}/{L2};get/set-run-type: {L1}/{L2}/{split}"
|
|
319
|
-
"(或用 --dataset/--split)")
|
|
338
|
+
help="list: {L1}/{L2};get/set-run-type: {L1}/{L2}/{split};"
|
|
339
|
+
"create: {L1}/{L2}/{split}(或用 --dataset/--split)")
|
|
320
340
|
sp.add_argument("--dataset", default=None, help="{L1}/{L2}")
|
|
321
341
|
sp.add_argument("--split", default=None,
|
|
322
342
|
help="目标 split 名(get/set-run-type 用)")
|
|
323
343
|
sp.add_argument("--version", default=None,
|
|
324
|
-
help="list 按 version 精确过滤;get
|
|
344
|
+
help="list 按 version 精确过滤;get 可选定位版本;create 必填")
|
|
325
345
|
sp.add_argument("--scope", choices=["unversioned", "exact", "all"], default=None,
|
|
326
346
|
help="list 跨版本策略:缺省无 version → all(SDK 取全量哲学),"
|
|
327
347
|
"有 version 省略(服务端只接受 exact)")
|
|
@@ -329,9 +349,13 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
329
349
|
help="list 按状态过滤")
|
|
330
350
|
sp.add_argument("--run-type", choices=["train", "eval", ""], default=None,
|
|
331
351
|
dest="run_type",
|
|
332
|
-
help="set-run-type 用:train/eval 或空串清除")
|
|
352
|
+
help="set-run-type/create 用:train/eval 或空串清除")
|
|
353
|
+
sp.add_argument("--storage-path", default=None, dest="storage_path",
|
|
354
|
+
help="create 用:split 壳的 storage_path(缺省自动计算)")
|
|
333
355
|
sp.add_argument("--page", type=int, default=1)
|
|
334
356
|
sp.add_argument("--page-size", type=int, default=50, dest="page_size")
|
|
357
|
+
sp.add_argument("--environment", default=None,
|
|
358
|
+
help="create 覆盖 profile 派生的 environment(pre/online)")
|
|
335
359
|
|
|
336
360
|
ds = sub.add_parser("dataset",
|
|
337
361
|
help="数据集/基准/工作流管理(list/get/update/access/"
|
|
@@ -449,7 +473,7 @@ def _resolve_split_ref(args) -> tuple[str, str]:
|
|
|
449
473
|
if args.ref:
|
|
450
474
|
segs = _split_ref(args.ref, 3)
|
|
451
475
|
return f"{segs[0]}/{segs[1]}", segs[2]
|
|
452
|
-
raise SystemExit("split get/set-run-type 需要 {L1}/{L2}/{split} 或 "
|
|
476
|
+
raise SystemExit("split get/create/set-run-type 需要 {L1}/{L2}/{split} 或 "
|
|
453
477
|
"--dataset/--split")
|
|
454
478
|
|
|
455
479
|
|
|
@@ -870,6 +894,17 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
870
894
|
return 0
|
|
871
895
|
|
|
872
896
|
if args.cmd == "version":
|
|
897
|
+
# 非法 flag 组合显式报错(此前静默忽略,用户传了却不生效极难自查):
|
|
898
|
+
# --splits 只有 create/ingest 消费;--split 只有 status/ingest 消费
|
|
899
|
+
# (ingest 的 --split/--splits 互斥由 SDK 层拦截)。
|
|
900
|
+
if args.splits is not None and args.action in (
|
|
901
|
+
"status", "get", "set-run-type", "update"):
|
|
902
|
+
raise SystemExit(
|
|
903
|
+
f"argument --splits: not allowed with version action {args.action}")
|
|
904
|
+
if args.split is not None and args.action in (
|
|
905
|
+
"create", "update", "list", "get", "set-run-type"):
|
|
906
|
+
raise SystemExit(
|
|
907
|
+
f"argument --split: not allowed with version action {args.action}")
|
|
873
908
|
srt = (_parse_split_run_types(args.split_run_types)
|
|
874
909
|
if args.split_run_types else None)
|
|
875
910
|
if args.action in ("create", "list"):
|
|
@@ -889,6 +924,32 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
889
924
|
print(f"created version {dv.dataset}/{dv.version} status={dv.status}",
|
|
890
925
|
file=out)
|
|
891
926
|
return 0
|
|
927
|
+
if args.action == "ingest":
|
|
928
|
+
# ingest 寻址与 create/list 同为 parts=2(split_first 下 version 可省略);
|
|
929
|
+
# --version 是业务参数不是寻址参数,故不能走 parts=3 的 _resolve_target。
|
|
930
|
+
dataset, _, _ = _resolve_target(args, 2)
|
|
931
|
+
splits = ([s.strip() for s in args.splits.split(",") if s.strip()]
|
|
932
|
+
if args.splits else None)
|
|
933
|
+
job = r.versions.ingest(
|
|
934
|
+
dataset, args.version or None,
|
|
935
|
+
storage_path=args.storage_path, splits=splits,
|
|
936
|
+
split=args.split, metadata_model=args.metadata_model,
|
|
937
|
+
environment=args.environment,
|
|
938
|
+
allow_layout_inference=args.allow_layout_inference,
|
|
939
|
+
publish=args.publish, dry_run=args.dry_run,
|
|
940
|
+
wait=not args.no_wait, poll_sec=args.poll_sec,
|
|
941
|
+
timeout_sec=args.timeout_sec, cluster=args.cluster)
|
|
942
|
+
if args.dry_run:
|
|
943
|
+
for dv in job.get("versions") or []:
|
|
944
|
+
print(f"version\t{dv.get('version', '')}\t"
|
|
945
|
+
f"instances={dv.get('instance_count') or 0}"
|
|
946
|
+
f"\tsplits={len(dv.get('splits') or [])}", file=out)
|
|
947
|
+
return 0
|
|
948
|
+
print(f"ingest {job.get('ingest_id', '')} "
|
|
949
|
+
f"status={job.get('status', '')} "
|
|
950
|
+
f"instances={job.get('stats', {}).get('instance_count') or 0}",
|
|
951
|
+
file=out)
|
|
952
|
+
return 0
|
|
892
953
|
dataset, version, _ = _resolve_target(args, 3)
|
|
893
954
|
if args.action == "status":
|
|
894
955
|
if args.split:
|
|
@@ -903,19 +964,17 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
903
964
|
# 1.0.10:取版本完整详情(splits 明细 + manifest + 时间戳)。
|
|
904
965
|
# split_first 分支时服务端聚合 legacy 投影。
|
|
905
966
|
dv = r.versions.get(dataset, version, environment=args.environment)
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
file=out)
|
|
911
|
-
print(f"dataset_version_id={dv.
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
print(f" split\t{s.get('name','')}\tstatus={s.get('status','')}"
|
|
916
|
-
f"\toss_prefix={s.get('oss_prefix','')}"
|
|
967
|
+
# versions.get 返回 DatasetVersion dataclass(字段访问,非 dict 取值)。
|
|
968
|
+
print(f"version\t{dv.version}\tstatus={dv.status}"
|
|
969
|
+
f"\tstorage_path={dv.storage_path or ''}", file=out)
|
|
970
|
+
print(f"published_at={dv.published_at or ''}"
|
|
971
|
+
f"\tlast_ingested_at={dv.last_ingested_at or ''}", file=out)
|
|
972
|
+
print(f"dataset_version_id={dv.dataset_version_id or ''}", file=out)
|
|
973
|
+
for s in dv.splits:
|
|
974
|
+
print(f" split\t{s.get('name', '')}\tstatus={s.get('status', '')}"
|
|
975
|
+
f"\toss_prefix={s.get('oss_prefix', '')}"
|
|
917
976
|
f"\tinstance_count={s.get('instance_count') or 0}", file=out)
|
|
918
|
-
manifest = dv.
|
|
977
|
+
manifest = dv.manifest or {}
|
|
919
978
|
print(f"manifest\tinstance_count={manifest.get('instance_count') or 0}"
|
|
920
979
|
f"\tfile_count={manifest.get('file_count') or 0}"
|
|
921
980
|
f"\tsize_bytes={manifest.get('size_bytes') or 0}", file=out)
|
|
@@ -952,6 +1011,22 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
952
1011
|
file=out)
|
|
953
1012
|
print(f"page={res.page}/{_page_count(res)} total={res.total}", file=out)
|
|
954
1013
|
return 0
|
|
1014
|
+
if args.action == "create":
|
|
1015
|
+
dataset, split_name = _resolve_split_ref(args)
|
|
1016
|
+
if not (args.version or "").strip():
|
|
1017
|
+
raise SystemExit(
|
|
1018
|
+
"split create 需要 --version(apiserver CreateVersion 硬约束)")
|
|
1019
|
+
sp = r.splits.create(dataset, split_name,
|
|
1020
|
+
version=args.version,
|
|
1021
|
+
run_type=args.run_type,
|
|
1022
|
+
storage_path=args.storage_path,
|
|
1023
|
+
environment=args.environment)
|
|
1024
|
+
src = "projected" if sp.projected else "record"
|
|
1025
|
+
print(f"created split\t{sp.name}\tversion={sp.version or ''}"
|
|
1026
|
+
f"\tstatus={sp.status}\trun_type={sp.run_type or ''}"
|
|
1027
|
+
f"\t[{src}]", file=out)
|
|
1028
|
+
print(f"storage_path={sp.storage_path}", file=out)
|
|
1029
|
+
return 0
|
|
955
1030
|
# get / set-run-type:需要定位到具体 split
|
|
956
1031
|
dataset, split_name = _resolve_split_ref(args)
|
|
957
1032
|
if args.action == "get":
|
|
@@ -39,6 +39,21 @@ def _registry_overrides(environment: str | None,
|
|
|
39
39
|
return overrides
|
|
40
40
|
|
|
41
41
|
|
|
42
|
+
def _validate_custom_asset_rel(rel: str) -> None:
|
|
43
|
+
"""白名单校验 custom 资产相对路径(防元数据被篡改造成 pull 路径穿越)。
|
|
44
|
+
|
|
45
|
+
push 侧用 ``Path.relative_to`` 生成 ``assets`` 清单是安全的;pull 侧**不能**
|
|
46
|
+
信任元数据:拒绝绝对路径、拒绝含 ``..`` 段、拒绝空。非法即抛 SchemaError,
|
|
47
|
+
不静默跳过。
|
|
48
|
+
"""
|
|
49
|
+
if not rel:
|
|
50
|
+
raise SchemaError("custom 资产清单含空路径")
|
|
51
|
+
if rel.startswith("/") or rel.startswith("\\"):
|
|
52
|
+
raise SchemaError(f"custom 资产清单含绝对路径: {rel!r}")
|
|
53
|
+
if ".." in rel.replace("\\", "/").split("/"):
|
|
54
|
+
raise SchemaError(f"custom 资产清单含 '..' 段: {rel!r}")
|
|
55
|
+
|
|
56
|
+
|
|
42
57
|
class InstancesClient:
|
|
43
58
|
# registry 的 BASE 同值;此处独立声明使 list_paged/update 等纯控制面方法
|
|
44
59
|
# 不依赖注入的 registry(直构客户端亦可调用)。
|
|
@@ -497,12 +512,19 @@ class InstancesClient:
|
|
|
497
512
|
# 目录输入:生成最小元数据 JSON + 原样上传资产
|
|
498
513
|
now_iso = _dt.datetime.now(_dt.timezone.utc).isoformat(
|
|
499
514
|
timespec="seconds").replace("+00:00", "Z")
|
|
515
|
+
# 先枚举资产相对路径并写入元数据 ``assets`` 字段:pull 还原时据此逐对象
|
|
516
|
+
# GetObject,免去 bucket LIST——普通用户的 dataset-RBAC STS 不含
|
|
517
|
+
# oss:ListObjects,靠 LIST 枚举资产会 403/DatasetIdentifierRequired。
|
|
518
|
+
asset_rels = sorted(
|
|
519
|
+
child.relative_to(local_path).as_posix()
|
|
520
|
+
for child in local_path.rglob("*") if child.is_file())
|
|
500
521
|
meta_doc = {
|
|
501
522
|
"instance_id": instance_id,
|
|
502
523
|
"metadata_model": "custom",
|
|
503
524
|
"schema_version": CUSTOM_SCHEMA_VERSION,
|
|
504
525
|
"created_at": now_iso,
|
|
505
526
|
"updated_at": now_iso,
|
|
527
|
+
"assets": asset_rels,
|
|
506
528
|
}
|
|
507
529
|
raw_bytes = json.dumps(meta_doc, ensure_ascii=False).encode("utf-8")
|
|
508
530
|
meta_digest = "sha256:" + hashlib.sha256(raw_bytes).hexdigest()
|
|
@@ -515,10 +537,8 @@ class InstancesClient:
|
|
|
515
537
|
asset_prefix = (f"{lay.dataset_root(dataset)}"
|
|
516
538
|
f"{split}-assets/{instance_id}/")
|
|
517
539
|
asset_digests = {}
|
|
518
|
-
for
|
|
519
|
-
|
|
520
|
-
continue
|
|
521
|
-
rel = child.relative_to(local_path).as_posix()
|
|
540
|
+
for rel in asset_rels:
|
|
541
|
+
child = local_path / rel
|
|
522
542
|
key = asset_prefix + rel
|
|
523
543
|
data = child.read_bytes()
|
|
524
544
|
digest = "sha256:" + hashlib.sha256(data).hexdigest()
|
|
@@ -764,9 +784,22 @@ class InstancesClient:
|
|
|
764
784
|
self._content.get_object(
|
|
765
785
|
idx_key, dest, expected_digest=expected,
|
|
766
786
|
sts_prefix=sts_prefix, dataset=dataset)
|
|
767
|
-
|
|
787
|
+
meta_bytes = dest.read_bytes()
|
|
788
|
+
# 从元数据 ``assets`` 字段还原资产清单,逐对象 GetObject——普通用户的
|
|
789
|
+
# dataset-RBAC STS 不含 oss:ListObjects,不能靠 LIST 枚举资产。
|
|
790
|
+
# 旧数据无 ``assets`` 字段时 asset_rels=None,由下游回退到 LIST。
|
|
791
|
+
asset_rels = None
|
|
792
|
+
if meta_bytes is not None:
|
|
793
|
+
try:
|
|
794
|
+
meta_doc = json.loads(meta_bytes)
|
|
795
|
+
except (json.JSONDecodeError, UnicodeDecodeError):
|
|
796
|
+
meta_doc = None
|
|
797
|
+
if isinstance(meta_doc, dict) and isinstance(
|
|
798
|
+
meta_doc.get("assets"), list):
|
|
799
|
+
asset_rels = [str(x) for x in meta_doc["assets"]]
|
|
768
800
|
self._download_custom_assets(
|
|
769
|
-
dataset, instance_id, split, dest_dir, lay, sts_prefix
|
|
801
|
+
dataset, instance_id, split, dest_dir, lay, sts_prefix,
|
|
802
|
+
asset_rels=asset_rels)
|
|
770
803
|
return dest
|
|
771
804
|
|
|
772
805
|
if fmt == "swe":
|
|
@@ -787,27 +820,49 @@ class InstancesClient:
|
|
|
787
820
|
expected_digest=expected, layout=lay)
|
|
788
821
|
|
|
789
822
|
def _download_custom_assets(self, dataset: str, instance_id: str,
|
|
790
|
-
split: str, dest_dir, lay, sts_prefix: str
|
|
823
|
+
split: str, dest_dir, lay, sts_prefix: str,
|
|
824
|
+
asset_rels: list[str] | None = None
|
|
791
825
|
) -> list[Path]:
|
|
792
826
|
"""下载 custom 格式的资产目录({split}-assets/{id}/ 前缀下的所有文件)。
|
|
793
827
|
|
|
794
828
|
资产文件按相对路径还原到 ``{dest_dir}/{instance_id}/`` 下。返回已下载
|
|
795
829
|
的本地文件路径列表。无资产时返回空列表(不报错)。
|
|
830
|
+
|
|
831
|
+
``asset_rels`` 为元数据 ``assets`` 字段记录的资产相对路径清单:非 None 时
|
|
832
|
+
逐对象 GetObject 还原(普通用户的 dataset-RBAC STS 不含 oss:ListObjects,
|
|
833
|
+
不能靠 LIST 枚举);为 None(旧数据无该字段)时回退到 LIST 枚举。
|
|
796
834
|
"""
|
|
797
835
|
asset_prefix = (f"{lay.dataset_root(dataset)}"
|
|
798
836
|
f"{split}-assets/{instance_id}/")
|
|
837
|
+
dest_root = Path(dest_dir) / instance_id
|
|
838
|
+
downloaded: list[Path] = []
|
|
839
|
+
|
|
840
|
+
if asset_rels is not None:
|
|
841
|
+
# 元数据记录了资产清单:逐对象下载,免 LIST。
|
|
842
|
+
for rel in asset_rels:
|
|
843
|
+
if rel.endswith("/"):
|
|
844
|
+
continue # 目录占位符(push 侧不生成;兼容历史数据)
|
|
845
|
+
_validate_custom_asset_rel(rel)
|
|
846
|
+
local_path = dest_root / rel
|
|
847
|
+
local_path.parent.mkdir(parents=True, exist_ok=True)
|
|
848
|
+
self._content.get_object(
|
|
849
|
+
asset_prefix + rel, local_path,
|
|
850
|
+
sts_prefix=sts_prefix, dataset=dataset)
|
|
851
|
+
downloaded.append(local_path)
|
|
852
|
+
return downloaded
|
|
853
|
+
|
|
854
|
+
# 回退:旧数据无 assets 清单,用 LIST 枚举(仅 system 调用方有 LIST 权限)。
|
|
799
855
|
list_fn = getattr(self._content, "list_objects", None)
|
|
800
856
|
if list_fn is None:
|
|
801
857
|
return []
|
|
802
858
|
keys = list_fn(asset_prefix)
|
|
803
859
|
if not keys:
|
|
804
860
|
return []
|
|
805
|
-
downloaded: list[Path] = []
|
|
806
|
-
dest_root = Path(dest_dir) / instance_id
|
|
807
861
|
for key in keys:
|
|
808
862
|
rel = key[len(asset_prefix):]
|
|
809
|
-
if
|
|
810
|
-
continue #
|
|
863
|
+
if rel.endswith("/"):
|
|
864
|
+
continue # 目录占位符(push 侧不生成;兼容历史数据)
|
|
865
|
+
_validate_custom_asset_rel(rel)
|
|
811
866
|
local_path = dest_root / rel
|
|
812
867
|
local_path.parent.mkdir(parents=True, exist_ok=True)
|
|
813
868
|
self._content.get_object(
|
|
@@ -222,6 +222,58 @@ class SplitsClient:
|
|
|
222
222
|
|
|
223
223
|
# ── 写 ──
|
|
224
224
|
|
|
225
|
+
def create(self, dataset: str, split: str, *,
|
|
226
|
+
version: str, run_type: str | None = None,
|
|
227
|
+
storage_path: str | None = None,
|
|
228
|
+
environment: str | None = None) -> DatasetSplit:
|
|
229
|
+
"""创建 split 记录——组合 versions.create(splits=[split]) 的便利封装。
|
|
230
|
+
|
|
231
|
+
apiserver **没有** split create 端点(``POST /datasets/splits`` 不存在):
|
|
232
|
+
split 记录由 ingest 扫描 OSS 或 versions.create/update(splits=[...])
|
|
233
|
+
间接产生(pre 集合 find-or-upsert ensure 真实 split record)。本方法是
|
|
234
|
+
"先建空 split 壳"的官方路径升级为 SDK 直接 API:
|
|
235
|
+
|
|
236
|
+
1. ``versions.create(dataset, version, splits=[split])``(version **必填**
|
|
237
|
+
——apiserver CreateVersion 硬约束;storage_path 缺省自动计算);
|
|
238
|
+
2. 可选 ``run_type``:create 后经 versions.update(split_run_types) 打标;
|
|
239
|
+
3. 读回 ``splits.get``(同 environment)返回记录。
|
|
240
|
+
|
|
241
|
+
参数:
|
|
242
|
+
dataset : dataset 名(必填)。
|
|
243
|
+
split : split 名(必填)。
|
|
244
|
+
version : version(必填——CreateVersion 硬约束)。
|
|
245
|
+
run_type : 可选,train/eval(create 后打标)。
|
|
246
|
+
storage_path : 可选,缺省按 profile 自动计算(与 versions.create 同源)。
|
|
247
|
+
environment : 可选,缺省由 profile 派生(pre/online)。
|
|
248
|
+
|
|
249
|
+
返回 :class:`~instance_repo.models.DatasetSplit`(读回;pre 集合 ensure
|
|
250
|
+
的 record 或 version 内嵌投影)。
|
|
251
|
+
"""
|
|
252
|
+
from .versions import VersionsClient
|
|
253
|
+
if not (dataset or "").strip():
|
|
254
|
+
raise SchemaError("dataset is required for splits.create")
|
|
255
|
+
if not (split or "").strip():
|
|
256
|
+
raise SchemaError("split is required for splits.create")
|
|
257
|
+
if not (version or "").strip():
|
|
258
|
+
raise SchemaError(
|
|
259
|
+
"version is required for splits.create (apiserver CreateVersion "
|
|
260
|
+
"requires version; split records are created via versions.create)")
|
|
261
|
+
rt = (run_type or "").strip()
|
|
262
|
+
if rt and rt not in RUN_TYPE_VALUES:
|
|
263
|
+
raise SchemaError(
|
|
264
|
+
f"invalid run_type {rt!r}; must be one of "
|
|
265
|
+
f"{', '.join(RUN_TYPE_VALUES)} or None")
|
|
266
|
+
env = self._env(environment)
|
|
267
|
+
vc = VersionsClient(self._t, self._p)
|
|
268
|
+
vc.create(dataset, version, splits=[split],
|
|
269
|
+
storage_path=storage_path, environment=env or None)
|
|
270
|
+
if rt:
|
|
271
|
+
# run_type 打标与 create 互斥字段:经 split_run_types 单独打(apiserver
|
|
272
|
+
# UpdateVersion 的 run_type/split_run_types/splits 两两互斥)。
|
|
273
|
+
vc.update(dataset, version, split_run_types={split: rt})
|
|
274
|
+
return self.get(dataset, split, version=version,
|
|
275
|
+
environment=env or None)
|
|
276
|
+
|
|
225
277
|
def update(self, dataset: str, split: str | None = None, *,
|
|
226
278
|
version: str | None = None, run_type: str = "",
|
|
227
279
|
dataset_split_id: str | None = None,
|
|
@@ -124,7 +124,7 @@ class DatasetSeries:
|
|
|
124
124
|
available_envs: list[str] = field(default_factory=list)
|
|
125
125
|
# 服务端 `DatasetSeriesStatus`("" 正常 / "deprecated");列表项
|
|
126
126
|
# `DatasetListItem.Status` 带 `omitempty`,正常态不出现。
|
|
127
|
-
|
|
127
|
+
status: str = ""
|
|
128
128
|
|
|
129
129
|
def validate(self) -> None:
|
|
130
130
|
if self.visibility not in VISIBILITY_VALUES:
|
|
@@ -139,11 +139,12 @@ class DatasetSeries:
|
|
|
139
139
|
"""create 请求体:剔除服务端派生/服务端分配的字段。
|
|
140
140
|
|
|
141
141
|
teacher_direction / teacher_type 由 apiserver 从 benchmark 派生(传了也被忽略),
|
|
142
|
-
dataset_id 由服务端分配;dataset_l1 / dataset_l2 由 dataset_name
|
|
142
|
+
dataset_id 由服务端分配;dataset_l1 / dataset_l2 由 dataset_name 推导。status
|
|
143
|
+
是服务端 DatasetSeriesStatus(deprecated 打标),create 不接受。都不发送,
|
|
143
144
|
避免让调用方误以为这些值能由客户端决定。
|
|
144
145
|
"""
|
|
145
146
|
drop = {"teacher_direction", "teacher_type", "dataset_id",
|
|
146
|
-
"dataset_l1", "dataset_l2"}
|
|
147
|
+
"dataset_l1", "dataset_l2", "status"}
|
|
147
148
|
return {k: v for k, v in self.to_dict().items() if k not in drop}
|
|
148
149
|
|
|
149
150
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: instance-repo
|
|
3
|
-
Version: 1.1.
|
|
3
|
+
Version: 1.1.2
|
|
4
4
|
Summary: InstanceRepo SDK — 评测 Instance 统一存储客户端(SDK-only;控制面由 apiserver 承载)
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
@@ -149,9 +149,10 @@ keys = repo.list_user_data(uid)
|
|
|
149
149
|
irepo --help
|
|
150
150
|
# 全局参数:--api-env / --api-base / --cluster / --storage-env / --metadata-model / --profile
|
|
151
151
|
# 无 --api-key:凭据只从 AP_API_KEY(旧别名 INSTANCEREPO_TOKEN)读取,避免进 shell history
|
|
152
|
-
#
|
|
152
|
+
# 20 个子命令:validate / push / push-many / deliver / pull / list / get
|
|
153
153
|
# publish / grant / feedback / report / whoami / user-data
|
|
154
154
|
# repo-config / create / claim / version / image
|
|
155
|
+
# split(list/get/set-run-type)/ dataset(list/get/update/access/revoke/benchmarks/workflows)
|
|
155
156
|
irepo image ref --dataset alibaba/mybench --instance-id inst-1 --split test
|
|
156
157
|
irepo image pull '<acr>/<ns>/mybench:test-inst-1' ./out.tar
|
|
157
158
|
```
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{instance_repo-1.1.1.dev0 → instance_repo-1.1.2}/instance_repo.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|