instance-repo 1.0.2__tar.gz → 1.0.7.dev0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/PKG-INFO +1 -1
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/__init__.py +1 -1
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/bootstrap.py +15 -2
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/cli.py +12 -7
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/config.py +111 -1
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/versions.py +81 -27
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/layout.py +4 -3
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/release.py +62 -6
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/repo.py +22 -24
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/store/registry.py +30 -1
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/transport.py +3 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo.egg-info/PKG-INFO +1 -1
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/pyproject.toml +1 -1
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/README.md +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/_routing.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/_site_defaults.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/cache.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/__init__.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/datasets.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/images.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/instances.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/reports.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/clients/scaffold.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/concurrency.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/content.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/endpoints.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/errors.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/image.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/loader.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/models.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/paths.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/retry.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/store/__init__.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/store/acr.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/store/base.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/store/oss.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/tasktoml.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo/validate.py +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo.egg-info/SOURCES.txt +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo.egg-info/dependency_links.txt +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo.egg-info/entry_points.txt +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo.egg-info/requires.txt +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo.egg-info/top_level.txt +0 -0
- {instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/setup.cfg +0 -0
|
@@ -19,6 +19,7 @@ registry/namespace),校验并缓存。
|
|
|
19
19
|
"""
|
|
20
20
|
from __future__ import annotations
|
|
21
21
|
|
|
22
|
+
import logging
|
|
22
23
|
import threading
|
|
23
24
|
import time
|
|
24
25
|
|
|
@@ -28,6 +29,8 @@ from .clients.config import (
|
|
|
28
29
|
)
|
|
29
30
|
from .errors import InstanceRepoError, SchemaError
|
|
30
31
|
|
|
32
|
+
_logger = logging.getLogger(__name__)
|
|
33
|
+
|
|
31
34
|
#: 自动发现结果的缓存存活时间(秒)。寻址配置变更频率极低,一小时足够;
|
|
32
35
|
#: 需要立刻生效时用 ``resolve_profile(..., refresh=True)`` 或 ``clear_cache()``。
|
|
33
36
|
DISCOVERY_TTL_SECONDS = 3600.0
|
|
@@ -82,7 +85,14 @@ def discover(transport, profile: Profile, *, refresh: bool = False) -> dict:
|
|
|
82
85
|
except (InstanceRepoError, OSError):
|
|
83
86
|
# 控制面错误以及 DNS/连接层 OSError 都按延迟失败处理:构造 Repo 不应阻断
|
|
84
87
|
# 纯控制面操作。记空结果避免每次数据面调用都重试,TTL 到期后仍会恢复。
|
|
85
|
-
|
|
88
|
+
#
|
|
89
|
+
# 不缓存空结果:缓存空结果会让"修复了集群名拼写"或"network 恢复"后仍需等 1 小时。
|
|
90
|
+
# 改为每次数据面操作都重试(apiserver 的 repo-config 是轻量 GET,无副作用)。
|
|
91
|
+
_logger.warning(
|
|
92
|
+
"repo-config 自动发现失败(selector=%s=%s),后续数据面操作会报 "
|
|
93
|
+
"oss_bucket is required。如非网络问题,请检查 cluster 名是否正确:"
|
|
94
|
+
"用 irepo repo-config --cluster <值> 可直接看到错误。",
|
|
95
|
+
selector_type, selector_value, exc_info=True)
|
|
86
96
|
return {}
|
|
87
97
|
_cache.put(key, dict(found), _expiry())
|
|
88
98
|
return dict(found)
|
|
@@ -120,7 +130,10 @@ def resolve_profile(transport, profile: Profile, *, refresh: bool = False
|
|
|
120
130
|
profile.sources["storage_env"] = "discovery"
|
|
121
131
|
# 无论 storage_env 是查询所得,还是显式值经本次查询确认一致,都记录二者绑定。
|
|
122
132
|
# ingest 只复用带这个证明的双 selector,避免方法级 cluster 覆盖后夹带旧环境。
|
|
123
|
-
|
|
133
|
+
# 绑定值取 repo-config 响应回传的**规范 Name**(cluster 入参为 id/url_cluster 时
|
|
134
|
+
# 已被 resolve_cluster 解析),与 ingest body 下发值同源,服务端按 Name 精确匹配。
|
|
135
|
+
profile.storage_env_cluster = (
|
|
136
|
+
found.get("cluster") or profile.cluster)
|
|
124
137
|
|
|
125
138
|
for field in DISCOVERABLE_FIELDS:
|
|
126
139
|
value = found.get(field)
|
|
@@ -122,9 +122,11 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
122
122
|
|
|
123
123
|
pub = sub.add_parser("publish", help="发布到运行时配置的目标数据面(benchmark-release)")
|
|
124
124
|
pub.add_argument("ref", nargs="?",
|
|
125
|
-
help="{L1}/{L2}/{version}
|
|
125
|
+
help="{L1}/{L2}/{version};split_first 无版本 split 发布用 {L1}/{L2}"
|
|
126
|
+
"(或用 --dataset/--version[--split])")
|
|
126
127
|
pub.add_argument("--dataset", default=None, help="{L1}/{L2}")
|
|
127
|
-
pub.add_argument("--version", default=None
|
|
128
|
+
pub.add_argument("--version", default=None,
|
|
129
|
+
help="版本号;split_first 的 split 发布可省略(省略时必须给 --split)")
|
|
128
130
|
pub.add_argument("--instances", default=None,
|
|
129
131
|
help="逗号分隔的 KEEP 实例 id;granularity=instance 或 images=on 时必填")
|
|
130
132
|
pub.add_argument("--target", default="",
|
|
@@ -515,11 +517,14 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
515
517
|
return 0
|
|
516
518
|
|
|
517
519
|
if args.cmd == "publish":
|
|
518
|
-
|
|
520
|
+
# split-first 的 split 发布:给了 --split 且没给 version 时允许只寻址 dataset
|
|
521
|
+
# ({L1}/{L2} 或 --dataset),version 置空走 dataset_distributions 协议。
|
|
522
|
+
need_version = not (args.split and not args.version)
|
|
523
|
+
dataset, version, _ = _resolve_target(args, 3 if need_version else 2)
|
|
519
524
|
if args.request_json:
|
|
520
525
|
with open(args.request_json, encoding="utf-8") as f:
|
|
521
526
|
body = json.load(f)
|
|
522
|
-
wf = r.versions.publish(dataset, version, request_body=body,
|
|
527
|
+
wf = r.versions.publish(dataset, version or "", request_body=body,
|
|
523
528
|
wait_approval=args.wait_approval)
|
|
524
529
|
else:
|
|
525
530
|
insts = ([s.strip() for s in args.instances.split(",") if s.strip()]
|
|
@@ -533,7 +538,7 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
533
538
|
if args.target_acr_namespace:
|
|
534
539
|
tgt_ovr["acr_namespace"] = args.target_acr_namespace
|
|
535
540
|
wf = r.versions.publish(
|
|
536
|
-
dataset, version, instances=insts, target=args.target,
|
|
541
|
+
dataset, version or "", instances=insts, target=args.target,
|
|
537
542
|
target_overrides=(tgt_ovr or None),
|
|
538
543
|
granularity=args.granularity, images=args.images,
|
|
539
544
|
overwrite=args.overwrite, wait_approval=args.wait_approval,
|
|
@@ -693,8 +698,8 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
693
698
|
for v in r.versions.list(dataset):
|
|
694
699
|
print(f"{v.version}\t{v.status}\t{v.storage_path}", file=out)
|
|
695
700
|
return 0
|
|
696
|
-
if not args.version
|
|
697
|
-
raise SystemExit("version create 需要 --version
|
|
701
|
+
if not args.version:
|
|
702
|
+
raise SystemExit("version create 需要 --version")
|
|
698
703
|
splits = ([s.strip() for s in args.splits.split(",") if s.strip()]
|
|
699
704
|
if args.splits else None)
|
|
700
705
|
dv = r.versions.create(dataset, args.version,
|
|
@@ -26,6 +26,7 @@ OSS bucket/endpoint/region 或 ACR registry/namespace,这些一律运行时取
|
|
|
26
26
|
"""
|
|
27
27
|
from __future__ import annotations
|
|
28
28
|
|
|
29
|
+
import logging
|
|
29
30
|
import os
|
|
30
31
|
import warnings
|
|
31
32
|
from dataclasses import dataclass, field
|
|
@@ -33,6 +34,8 @@ from urllib.parse import urlencode
|
|
|
33
34
|
|
|
34
35
|
from .. import _site_defaults
|
|
35
36
|
|
|
37
|
+
_logger = logging.getLogger(__name__)
|
|
38
|
+
|
|
36
39
|
# ── 规范参数与兼容别名 ──
|
|
37
40
|
AP_API_KEY = "AP_API_KEY"
|
|
38
41
|
AP_BASE_URL = "AP_BASE_URL"
|
|
@@ -277,6 +280,11 @@ def fetch_repo_config(transport, *, cluster: str = "", storage_env: str = "") ->
|
|
|
277
280
|
需要先有一个能发控制面请求的 transport(已配 api_base + token)。
|
|
278
281
|
cluster 与 storage_env 二选一;cluster 优先(apiserver 先由 cluster 解析出 storage_env)。
|
|
279
282
|
|
|
283
|
+
如果 cluster 是集群 ID(c 前缀 + 32 位 hex)或 url_cluster,会先通过
|
|
284
|
+
``GET /apis/v1/clusters`` 解析为集群 Name 再下发——apiserver 的 repo-config
|
|
285
|
+
只按 Name 精确匹配(大小写敏感),用户不需要区分 cluster name / cluster id /
|
|
286
|
+
url_cluster。
|
|
287
|
+
|
|
280
288
|
用法示例::
|
|
281
289
|
|
|
282
290
|
from instance_repo.clients.config import fetch_repo_config
|
|
@@ -287,6 +295,7 @@ def fetch_repo_config(transport, *, cluster: str = "", storage_env: str = "") ->
|
|
|
287
295
|
repo = Repo(profile_overrides=overrides)
|
|
288
296
|
"""
|
|
289
297
|
if cluster:
|
|
298
|
+
cluster = resolve_cluster(transport, cluster)
|
|
290
299
|
query = urlencode({"cluster": cluster})
|
|
291
300
|
elif storage_env:
|
|
292
301
|
query = urlencode({"storage_env": storage_env})
|
|
@@ -294,8 +303,12 @@ def fetch_repo_config(transport, *, cluster: str = "", storage_env: str = "") ->
|
|
|
294
303
|
raise ValueError("fetch_repo_config requires 'cluster' or 'storage_env'")
|
|
295
304
|
data = transport.get(f"/apis/v1/repo-config?{query}")
|
|
296
305
|
# 映射 apiserver 响应字段到 SDK profile_overrides 的 key;cluster 查询的顶层
|
|
297
|
-
# storage_env 是服务端解析结果,必须保留给 Profile 与 ingest 使用。
|
|
306
|
+
# storage_env 是服务端解析结果,必须保留给 Profile 与 ingest 使用。cluster 键回传
|
|
307
|
+
# 解析后的**规范 Name**(可能是 id/url_cluster 的解析结果),供 bootstrap 记录绑定
|
|
308
|
+
# 证明、ingest 下发 body——两个消费方都需要服务端按 Name 精确匹配的规范值。
|
|
298
309
|
result = {}
|
|
310
|
+
if cluster:
|
|
311
|
+
result["cluster"] = cluster
|
|
299
312
|
if cluster and data.get("storage_env"):
|
|
300
313
|
result["storage_env"] = normalize_storage_env(data["storage_env"])
|
|
301
314
|
if oss := data.get("oss"):
|
|
@@ -314,3 +327,100 @@ def fetch_repo_config(transport, *, cluster: str = "", storage_env: str = "") ->
|
|
|
314
327
|
if isinstance(api_env_raw, str) and api_env_raw.strip():
|
|
315
328
|
result["api_env"] = normalize_api_env(api_env_raw)
|
|
316
329
|
return result
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
# ── cluster ID / url_cluster → Name 解析 ──
|
|
333
|
+
|
|
334
|
+
#: 集群列表缓存时间(秒)。集群拓扑变更频率极低,1 小时足够。
|
|
335
|
+
_CLUSTER_MAP_TTL = 3600.0
|
|
336
|
+
|
|
337
|
+
#: 缓存按 transport 身份(api_base + Env 头)隔离:不同控制面部署的集群拓扑各自独立,
|
|
338
|
+
#: 共享一个进程里先后连两个 apiserver 时不会串味。
|
|
339
|
+
_cluster_map_cache: dict = {} # (api_base, env) → 映射表
|
|
340
|
+
_cluster_map_expiry: dict = {} # (api_base, env) → 过期时刻
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _cluster_cache_key(transport) -> tuple:
|
|
344
|
+
return (
|
|
345
|
+
str(getattr(transport, "base_url", "") or "").rstrip("/"),
|
|
346
|
+
str(getattr(transport, "env", "") or ""),
|
|
347
|
+
id(transport) if getattr(transport, "_custom_send", False) else 0,
|
|
348
|
+
)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _empty_cluster_map() -> dict:
|
|
352
|
+
return {"id_to_name": {}, "url_to_name": {}, "names": set()}
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def fetch_cluster_map(transport) -> dict:
|
|
356
|
+
"""从 ``GET /apis/v1/clusters`` 拉取集群列表,构建 ``id→Name`` 与 ``url_cluster→Name`` 映射。
|
|
357
|
+
|
|
358
|
+
apiserver 的 repo-config 只按集群 Name 精确匹配(大小写敏感),因此 SDK 把用户可能
|
|
359
|
+
提供的三种标识(cluster ID / url_cluster / Name)都规范化为 Name 再下发。结果缓存在
|
|
360
|
+
进程内存中(按 transport 身份隔离),TTL 与 discover 一致(3600 秒)。拉取失败不缓存
|
|
361
|
+
新结果:已有该 transport 的成功缓存时回退使用(stale 拓扑好过没有),否则返回空映射。
|
|
362
|
+
"""
|
|
363
|
+
import time
|
|
364
|
+
global _cluster_map_cache, _cluster_map_expiry
|
|
365
|
+
key = _cluster_cache_key(transport)
|
|
366
|
+
now = time.time()
|
|
367
|
+
cached = _cluster_map_cache.get(key)
|
|
368
|
+
if cached is not None and now < _cluster_map_expiry.get(key, 0.0):
|
|
369
|
+
return cached
|
|
370
|
+
try:
|
|
371
|
+
data = transport.get("/apis/v1/clusters")
|
|
372
|
+
clusters = data.get("clusters", data) if isinstance(data, dict) else (data or [])
|
|
373
|
+
id_to_name: dict[str, str] = {}
|
|
374
|
+
url_to_name: dict[str, str] = {}
|
|
375
|
+
names: set[str] = set()
|
|
376
|
+
for entry in (clusters or []):
|
|
377
|
+
if not isinstance(entry, dict):
|
|
378
|
+
continue
|
|
379
|
+
cid = str(entry.get("cluster_id") or "").strip()
|
|
380
|
+
url = str(entry.get("url_cluster") or "").strip()
|
|
381
|
+
name = str(entry.get("name") or "").strip()
|
|
382
|
+
if not name:
|
|
383
|
+
continue
|
|
384
|
+
names.add(name)
|
|
385
|
+
if cid and cid not in id_to_name:
|
|
386
|
+
id_to_name[cid] = name
|
|
387
|
+
if url and url not in url_to_name:
|
|
388
|
+
url_to_name[url] = name
|
|
389
|
+
fresh = {
|
|
390
|
+
"id_to_name": id_to_name,
|
|
391
|
+
"url_to_name": url_to_name,
|
|
392
|
+
"names": names,
|
|
393
|
+
}
|
|
394
|
+
_cluster_map_cache[key] = fresh
|
|
395
|
+
_cluster_map_expiry[key] = now + _CLUSTER_MAP_TTL
|
|
396
|
+
return fresh
|
|
397
|
+
except Exception:
|
|
398
|
+
_logger.debug("fetch_cluster_map failed", exc_info=True)
|
|
399
|
+
return cached if cached is not None else _empty_cluster_map()
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
def resolve_cluster(transport, cluster_value: str) -> str:
|
|
403
|
+
"""将 cluster ID 或 url_cluster 解析为集群 Name;已是 Name 则原样返回。
|
|
404
|
+
|
|
405
|
+
调用 ``GET /apis/v1/clusters`` 获取映射后解析。优先级:Name 本身 > cluster ID >
|
|
406
|
+
url_cluster。映射失败时保留原值(让 apiserver 返回明确的 "unknown cluster" 错误)。
|
|
407
|
+
入参先去除首尾空白(与 Go/Java 的 ResolveCluster 一致);纯空白直接返回,不发请求。
|
|
408
|
+
"""
|
|
409
|
+
value = str(cluster_value or "").strip()
|
|
410
|
+
if not value:
|
|
411
|
+
return value
|
|
412
|
+
cmap = fetch_cluster_map(transport)
|
|
413
|
+
if value in cmap["names"]:
|
|
414
|
+
return value
|
|
415
|
+
for key in ("id_to_name", "url_to_name"):
|
|
416
|
+
resolved = cmap[key].get(value)
|
|
417
|
+
if resolved:
|
|
418
|
+
return resolved
|
|
419
|
+
return value
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def clear_cluster_map_cache() -> None:
|
|
423
|
+
"""清空集群映射缓存(测试用)。"""
|
|
424
|
+
global _cluster_map_cache, _cluster_map_expiry
|
|
425
|
+
_cluster_map_cache = {}
|
|
426
|
+
_cluster_map_expiry = {}
|
|
@@ -108,6 +108,8 @@ class VersionsClient:
|
|
|
108
108
|
「cluster/storage_env 指向 A 数据面、storage_path 指向 B 桶」的错配请求;
|
|
109
109
|
响应缺 bucket/prefix 时与 Go/Java 一样在 POST 前 fail-fast,不静默回退到
|
|
110
110
|
共享 profile 的 OSS 路径。cluster 为空时沿用 profile 自身寻址。
|
|
111
|
+
返回的 cluster 是 repo-config 响应回传的**规范 Name**(id/url_cluster 会被
|
|
112
|
+
resolve_cluster 解析掉),ingest body 用它下发,服务端按 Name 精确匹配。
|
|
111
113
|
"""
|
|
112
114
|
profile_cluster = (getattr(self._p, "cluster", "") or "").strip()
|
|
113
115
|
method_cluster = (cluster or "").strip()
|
|
@@ -130,10 +132,14 @@ class VersionsClient:
|
|
|
130
132
|
profile_bucket, profile_prefix)
|
|
131
133
|
|
|
132
134
|
found = fetch_repo_config(self._t, cluster=effective_cluster)
|
|
135
|
+
# fetch_repo_config 已把 cluster 入参(可能是 id/url_cluster)解析为规范 Name 并
|
|
136
|
+
# 回传;ingest body 与 profile 绑定证明都用这个值,服务端按 Name 精确匹配。
|
|
137
|
+
# 报错文案也统一用规范 Name(与 Go/Java 的 binding.cluster 一致)。
|
|
138
|
+
resolved_cluster = (found.get("cluster") or effective_cluster).strip()
|
|
133
139
|
resolved_storage_env = normalize_storage_env(found.get("storage_env"))
|
|
134
140
|
if not resolved_storage_env:
|
|
135
141
|
raise SchemaError(
|
|
136
|
-
f"cluster={
|
|
142
|
+
f"cluster={resolved_cluster!r} 的 repo-config 响应缺少非空 "
|
|
137
143
|
"storage_env;已中止 ingest,避免发送未绑定的双 selector")
|
|
138
144
|
|
|
139
145
|
storage_source = (getattr(self._p, "sources", {}) or {}).get(
|
|
@@ -141,7 +147,7 @@ class VersionsClient:
|
|
|
141
147
|
if profile_storage_env and storage_source != "discovery" \
|
|
142
148
|
and profile_storage_env != resolved_storage_env:
|
|
143
149
|
raise SchemaError(
|
|
144
|
-
f"cluster={
|
|
150
|
+
f"cluster={resolved_cluster!r} 与显式 profile storage_env="
|
|
145
151
|
f"{profile_storage_env!r} 冲突:服务端解析为 "
|
|
146
152
|
f"{resolved_storage_env!r}")
|
|
147
153
|
|
|
@@ -150,7 +156,7 @@ class VersionsClient:
|
|
|
150
156
|
if not bound_bucket or not bound_prefix:
|
|
151
157
|
missing = "oss_bucket" if not bound_bucket else "oss_prefix"
|
|
152
158
|
raise SchemaError(
|
|
153
|
-
f"cluster={
|
|
159
|
+
f"cluster={resolved_cluster!r} 的 repo-config 响应缺少非空 "
|
|
154
160
|
f"{missing};已中止 ingest,避免回退到其他数据面的 OSS 路径")
|
|
155
161
|
|
|
156
162
|
# 只有解析的是 profile 自身 cluster 时才回填绑定;方法级覆盖不能篡改 profile
|
|
@@ -159,9 +165,9 @@ class VersionsClient:
|
|
|
159
165
|
if not profile_storage_env or storage_source == "discovery":
|
|
160
166
|
self._p.storage_env = resolved_storage_env
|
|
161
167
|
self._p.sources["storage_env"] = "discovery"
|
|
162
|
-
self._p.storage_env_cluster =
|
|
168
|
+
self._p.storage_env_cluster = resolved_cluster
|
|
163
169
|
|
|
164
|
-
return (
|
|
170
|
+
return (resolved_cluster, resolved_storage_env,
|
|
165
171
|
bound_bucket, bound_prefix)
|
|
166
172
|
|
|
167
173
|
def _recorded_images(self, dataset: str, version: str,
|
|
@@ -210,7 +216,8 @@ class VersionsClient:
|
|
|
210
216
|
f"&version={quote(version, safe='')}{extra}")
|
|
211
217
|
return data.get("status", "")
|
|
212
218
|
|
|
213
|
-
def create(self, dataset: str, version: str, *,
|
|
219
|
+
def create(self, dataset: str, version: str, *,
|
|
220
|
+
storage_path: str | None = None,
|
|
214
221
|
splits: list[str] | None = None, status: str = "draft",
|
|
215
222
|
storage_type: str = "oss") -> DatasetVersion:
|
|
216
223
|
"""显式创建 DatasetVersion(``POST /apis/v1/datasets/versions``)。
|
|
@@ -219,7 +226,9 @@ class VersionsClient:
|
|
|
219
226
|
(例如先建再 PATCH run_type、或先建再单独 ingest)的场景。
|
|
220
227
|
|
|
221
228
|
参数:
|
|
222
|
-
storage_path :
|
|
229
|
+
storage_path : 可选。缺省时按 profile 自动计算
|
|
230
|
+
(``oss://{bucket}/{prefix}/{L1}/{L2}/{version}/``,与 ingest 同源)。
|
|
231
|
+
显式传入时原样下发(特殊桶/自定义布局场景覆盖)。
|
|
223
232
|
splits : 可选 split 名列表;缺省由 apiserver 按 storage_path 推导。
|
|
224
233
|
status : 版本状态,缺省 ``draft``(合法值见 VERSION_STATUS_VALUES)。
|
|
225
234
|
storage_type : 存储类型,缺省 ``oss``(当前仅支持)。
|
|
@@ -233,9 +242,16 @@ class VersionsClient:
|
|
|
233
242
|
"""
|
|
234
243
|
if not version:
|
|
235
244
|
raise SchemaError("version is required")
|
|
245
|
+
if storage_path is None and self._p is not None:
|
|
246
|
+
bucket = getattr(self._p, "oss_bucket", "") or ""
|
|
247
|
+
if bucket:
|
|
248
|
+
storage_path = self._layout().ingest_storage_path(
|
|
249
|
+
bucket, dataset, version)
|
|
236
250
|
if not storage_path or not storage_path.startswith("oss://"):
|
|
237
251
|
raise SchemaError(
|
|
238
|
-
f"storage_path must be an oss:// URL, got {storage_path!r}"
|
|
252
|
+
f"storage_path must be an oss:// URL, got {storage_path!r}; "
|
|
253
|
+
f"未显式提供且 profile 无 oss_bucket 时无法自动计算,"
|
|
254
|
+
f"请配置 INSTANCE_REPO_STORAGE_ENV 或用 profile_overrides 提供")
|
|
239
255
|
if status not in VERSION_STATUS_VALUES:
|
|
240
256
|
raise SchemaError(
|
|
241
257
|
f"invalid status '{status}'; valid values: "
|
|
@@ -528,7 +544,7 @@ class VersionsClient:
|
|
|
528
544
|
wf = self.publish(dataset, version, instances=instances, images=images, **kw)
|
|
529
545
|
return wf, skipped
|
|
530
546
|
|
|
531
|
-
def publish(self, dataset: str, version: str, *,
|
|
547
|
+
def publish(self, dataset: str, version: str = "", *,
|
|
532
548
|
instances: list[str] | None = None,
|
|
533
549
|
target: str = "",
|
|
534
550
|
target_overrides: dict | None = None,
|
|
@@ -548,6 +564,15 @@ class VersionsClient:
|
|
|
548
564
|
sleep=time.sleep) -> dict:
|
|
549
565
|
"""提交 benchmark-release workflow(运行时源地址→目标地址)并轮询到终态。
|
|
550
566
|
|
|
567
|
+
按元数据模型选择发布协议(v1.0.5 起):
|
|
568
|
+
|
|
569
|
+
* **split_first 且给了 split**(granularity≠instance):走 apiserver master 的
|
|
570
|
+
``dataset_distributions`` 发布协议。split 必填、version 可选——无 version 即
|
|
571
|
+
无版本 split 发布,有 version 即版本化 split 发布;目标路径由服务端从 online
|
|
572
|
+
Dataset base 派生,发布成功后服务端只 ingest 该 split 并置 published。
|
|
573
|
+
* **其余情况**(version_first / 未给 split / granularity=instance):保留旧
|
|
574
|
+
``oss_distributions`` 纯前缀拷贝语义,行为与 v0.8~v1.0.4 一致。
|
|
575
|
+
|
|
551
576
|
默认按业务意图构造请求体(SDK 内部拼 workflow body,隐藏 distributions/幂等 key/
|
|
552
577
|
registry host 改写)。源 OSS/ACR 地址来自当前 Repo 的运行时环境或构造参数;目标
|
|
553
578
|
OSS/ACR 地址必须通过 ``target_overrides`` 显式提供。若地址来自 repo-config,调用方
|
|
@@ -556,32 +581,57 @@ class VersionsClient:
|
|
|
556
581
|
环境,也没有隐式目标地址。``request_body`` 为直接透传自定义 workflow body 的逃生舱。
|
|
557
582
|
|
|
558
583
|
参数:
|
|
584
|
+
dataset : {L1}/{L2}
|
|
585
|
+
version : 版本号。dataset_distributions 协议下可选(split_first 无版本
|
|
586
|
+
split 发布);旧协议下必填
|
|
559
587
|
instances : KEEP 实例 id;granularity='instance' 或 images='on' 时必填
|
|
560
588
|
target : 目标标签;只用于标识,不解析或选择数据面地址
|
|
561
589
|
target_overrides: 目标 oss_bucket/acr_host/acr_namespace 等运行时地址
|
|
562
|
-
granularity : 'dataset'
|
|
563
|
-
|
|
590
|
+
granularity : 旧协议专用('dataset' 整前缀一条 | 'instance' 逐实例枚举);
|
|
591
|
+
dataset_distributions 协议恒为整 split 目录
|
|
592
|
+
images : 'off'(仅 OSS) | 'on'(按实例枚举 ACR 镜像分发,可与
|
|
593
|
+
dataset_distributions 同单下发)
|
|
594
|
+
split : **必须与 push 时一致**;split_first 协议下必填
|
|
564
595
|
approval_mode : 'internal' 时 workflow 停在 awaiting_approval;wait_approval=False
|
|
565
596
|
即在该状态返回(交人工审批),True 则继续等到 succeeded/failed
|
|
566
|
-
split : **必须与 push 时一致**(默认 'default')。instance 粒度按
|
|
567
|
-
{version}-assets/{split}/{id}/… 拼源路径;传错会让上架 validate
|
|
568
|
-
报 source_path does not exist(真机踩过)。dataset 粒度不受影响。
|
|
569
597
|
"""
|
|
570
598
|
if request_body is None:
|
|
571
599
|
if self._p is None:
|
|
572
600
|
raise SchemaError("typed publish 需要源 profile(用 Repo().versions,"
|
|
573
601
|
"或改用 request_body 逃生舱)")
|
|
602
|
+
model = normalize_metadata_model(
|
|
603
|
+
metadata_model or (self._p.metadata_model if self._p else None))
|
|
604
|
+
use_dataset_protocol = (model == MODEL_SPLIT_FIRST
|
|
605
|
+
and bool(split.strip())
|
|
606
|
+
and granularity != "instance")
|
|
574
607
|
dst = load_profile(target or "target", overrides=target_overrides)
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
608
|
+
lay = self._layout(model)
|
|
609
|
+
dataset_dists: list[dict] = []
|
|
610
|
+
oss_dists: list[dict] = []
|
|
611
|
+
if use_dataset_protocol:
|
|
612
|
+
if not dst.oss_bucket:
|
|
613
|
+
raise SchemaError(
|
|
614
|
+
"publish 需要目标 oss_bucket;SDK 不内置任何环境的桶地址,请通过 "
|
|
615
|
+
"target_overrides={'oss_bucket': ..., 'acr_host': ...} 提供(见文档)")
|
|
616
|
+
dataset_dists = R.build_dataset_distributions(
|
|
617
|
+
dataset, split=split, version=version,
|
|
618
|
+
metadata_model=model,
|
|
619
|
+
src_bucket=self._p.oss_bucket,
|
|
620
|
+
overwrite=overwrite, layout=lay)
|
|
621
|
+
else:
|
|
622
|
+
if not version:
|
|
623
|
+
raise SchemaError(
|
|
624
|
+
f"version is required when metadata_model={model!r} "
|
|
625
|
+
f"且未指定 split;split_first 的 split 发布请显式传 split")
|
|
626
|
+
if not dst.oss_bucket:
|
|
627
|
+
raise SchemaError(
|
|
628
|
+
"publish 需要目标 oss_bucket;SDK 不内置任何环境的桶地址,请通过 "
|
|
629
|
+
"target_overrides={'oss_bucket': ..., 'acr_host': ...} 提供(见文档)")
|
|
630
|
+
oss_dists = R.build_oss_distributions(
|
|
631
|
+
dataset, version, instances or [],
|
|
632
|
+
src_bucket=self._p.oss_bucket, dst_bucket=dst.oss_bucket,
|
|
633
|
+
granularity=granularity, overwrite=overwrite, split=split,
|
|
634
|
+
instance_objects=instance_objects, layout=lay)
|
|
585
635
|
image_dists: list[dict] = []
|
|
586
636
|
if images == "on":
|
|
587
637
|
if not instances:
|
|
@@ -599,12 +649,16 @@ class VersionsClient:
|
|
|
599
649
|
layout=lay, split=split)
|
|
600
650
|
elif images != "off":
|
|
601
651
|
raise SchemaError(f"unknown images {images!r} (use 'off'|'on')")
|
|
602
|
-
scope = "
|
|
603
|
-
|
|
652
|
+
scope = ("split" if dataset_dists
|
|
653
|
+
else "dataset" if granularity == "dataset"
|
|
654
|
+
else f"n{len(instances or [])}")
|
|
655
|
+
idem = R.make_idempotency_key(dataset, version, scope,
|
|
656
|
+
oss_dists, image_dists, dataset_dists)
|
|
604
657
|
request_body = R.build_release_request(
|
|
605
658
|
idempotency_key=idem,
|
|
606
|
-
description=description or f"release {dataset}/{version} ({
|
|
659
|
+
description=description or f"release {dataset}/{version} ({scope})",
|
|
607
660
|
oss_dists=oss_dists, image_dists=image_dists,
|
|
661
|
+
dataset_dists=dataset_dists,
|
|
608
662
|
approval_mode=approval_mode, backend=backend)
|
|
609
663
|
|
|
610
664
|
wf = self._t.post(self._WORKFLOWS, request_body)
|
|
@@ -42,10 +42,11 @@ DEFAULT_METADATA_MODEL = MODEL_SPLIT_FIRST
|
|
|
42
42
|
_MODELS = (MODEL_VERSION_FIRST, MODEL_SPLIT_FIRST)
|
|
43
43
|
|
|
44
44
|
# ACR tag 合法字符集与长度(OCI/Docker 约定):首字符须为字母数字或下划线,
|
|
45
|
-
# 其余可含字母数字、下划线、点、连字符,总长
|
|
46
|
-
#
|
|
45
|
+
# 其余可含字母数字、下划线、点、连字符,总长 **< 105**(ACR tag 硬上限 128,
|
|
46
|
+
# 预留 {version}-{id} 拼接余量,超长片段由 sanitize 截断)。**大小写敏感**,
|
|
47
|
+
# 不做小写化——小写化会静默改变镜像身份,进而指向一个从未推送过的引用。
|
|
47
48
|
_TAG_RE = re.compile(r"^[A-Za-z0-9_][A-Za-z0-9_.\-]*$")
|
|
48
|
-
_TAG_MAX_LEN =
|
|
49
|
+
_TAG_MAX_LEN = 104
|
|
49
50
|
_TAG_BAD_CHARS = re.compile(r"[^A-Za-z0-9_.\-]+")
|
|
50
51
|
_TAG_BAD_HEAD = re.compile(r"^[^A-Za-z0-9_]+")
|
|
51
52
|
|
|
@@ -174,21 +174,75 @@ def build_image_distributions(dataset: str, version: str, instances: list[str],
|
|
|
174
174
|
return dists
|
|
175
175
|
|
|
176
176
|
|
|
177
|
+
def build_dataset_distributions(dataset: str, *,
|
|
178
|
+
split: str = "",
|
|
179
|
+
version: str = "",
|
|
180
|
+
metadata_model: str = "",
|
|
181
|
+
src_bucket: str,
|
|
182
|
+
overwrite: bool = False,
|
|
183
|
+
layout: "L.Layout | None" = None) -> list[dict]:
|
|
184
|
+
"""构造 ``dataset_distributions`` 数组(apiserver master 的 dataset release 协议)。
|
|
185
|
+
|
|
186
|
+
与 :func:`build_oss_distributions` 的纯前缀拷贝不同,本协议每项携带
|
|
187
|
+
``metadata_model/dataset/split/version/source_path``,服务端据此解析发布单元、
|
|
188
|
+
从 online Dataset base 派生目标路径(SDK 不提供 target_path),发布成功后以
|
|
189
|
+
``publish_after_validation=true`` 只 ingest 该单元并置 published。
|
|
190
|
+
|
|
191
|
+
模型约定(对齐 apiserver ``ResolveDatasetRelease``):
|
|
192
|
+
|
|
193
|
+
* ``split_first``:split 必填、version 可选。无 version 时源目录必须是整个
|
|
194
|
+
split 目录 ``swe/datasets/<L1>/<L2>/<split>/``;有 version 时是
|
|
195
|
+
``swe/datasets/<L1>/<L2>/<version>/<split>/``。
|
|
196
|
+
* ``version_first``:version 必填,源目录为整个版本目录
|
|
197
|
+
``swe/datasets/<L1>/<L2>/<version>/``。
|
|
198
|
+
|
|
199
|
+
返回单元素数组(一次发布一个 split/版本);多个单元多次调用后合并提交。
|
|
200
|
+
"""
|
|
201
|
+
lay = layout or L.Layout()
|
|
202
|
+
model = L.normalize_metadata_model(metadata_model) or lay.model
|
|
203
|
+
ver, spl = L.check_locator(model, version, split)
|
|
204
|
+
if model == L.MODEL_SPLIT_FIRST:
|
|
205
|
+
key = f"{lay.dataset_root(dataset)}{ver + '/' if ver else ''}{spl}/"
|
|
206
|
+
else:
|
|
207
|
+
key = f"{lay.dataset_root(dataset)}{ver}/"
|
|
208
|
+
return [{
|
|
209
|
+
"source_path": P.oss_uri(src_bucket, key),
|
|
210
|
+
"dataset": dataset,
|
|
211
|
+
"metadata_model": model,
|
|
212
|
+
"version": ver,
|
|
213
|
+
"split": spl if model == L.MODEL_SPLIT_FIRST else "",
|
|
214
|
+
"overwrite": overwrite,
|
|
215
|
+
}]
|
|
216
|
+
|
|
217
|
+
|
|
177
218
|
def make_idempotency_key(dataset: str, version: str, scope: str,
|
|
178
|
-
oss_dists: list[dict], image_dists: list[dict]
|
|
219
|
+
oss_dists: list[dict], image_dists: list[dict],
|
|
220
|
+
dataset_dists: list[dict] | None = None) -> str:
|
|
179
221
|
"""内容寻址幂等 key:相同内容→相同 key→平台复用已有 workflow(安全重试)。
|
|
180
|
-
形如 ``<sanitize(dataset)>-<version>-release-<scope>-<sha1[:10]>``。
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
222
|
+
形如 ``<sanitize(dataset)>-<version>-release-<scope>-<sha1[:10]>``。
|
|
223
|
+
|
|
224
|
+
``dataset_dists`` 为空时不进载荷——保持与三语言 conformance golden 的旧 key
|
|
225
|
+
逐字节一致(纯拷贝协议的既有调用方 key 不变);非空时才参与哈希,使两种协议
|
|
226
|
+
互不串用同一个 workflow。
|
|
227
|
+
"""
|
|
228
|
+
payload = {"oss": oss_dists, "image": image_dists}
|
|
229
|
+
if dataset_dists:
|
|
230
|
+
payload["dataset"] = dataset_dists
|
|
231
|
+
text = json.dumps(payload, sort_keys=True, ensure_ascii=False)
|
|
232
|
+
digest = hashlib.sha1(text.encode("utf-8")).hexdigest()[:10]
|
|
184
233
|
return f"{P.sanitize_repo(dataset)}-{version}-release-{scope}-{digest}"
|
|
185
234
|
|
|
186
235
|
|
|
187
236
|
def build_release_request(*, idempotency_key: str, description: str,
|
|
188
237
|
oss_dists: list[dict], image_dists: list[dict],
|
|
238
|
+
dataset_dists: list[dict] | None = None,
|
|
189
239
|
approval_mode: str = "internal",
|
|
190
240
|
backend: str = "acr") -> dict:
|
|
191
|
-
"""组装 benchmark-release workflow
|
|
241
|
+
"""组装 benchmark-release workflow 请求体;空的分发数组省略(均可选)。
|
|
242
|
+
|
|
243
|
+
``dataset_dists`` 走 apiserver master 的 ``dataset_distributions`` 发布协议
|
|
244
|
+
(split-first 的 split 发布 / version-first 的版本发布),与 ``oss_dists``
|
|
245
|
+
的纯前缀拷贝语义不同:前者发布成功后由服务端 ingest 并置 published。
|
|
192
246
|
|
|
193
247
|
``backend`` 自 v0.8 起默认 ``"acr"``(v0.7.x 为 ``"bp"``):镜像分发的真实执行后端就是
|
|
194
248
|
ACR 仓库同步,默认值应与之一致,避免调用方每次都要显式纠正。
|
|
@@ -196,6 +250,8 @@ def build_release_request(*, idempotency_key: str, description: str,
|
|
|
196
250
|
spec: dict = {"approval_mode": approval_mode, "backend": backend}
|
|
197
251
|
if image_dists:
|
|
198
252
|
spec["image_distributions"] = image_dists
|
|
253
|
+
if dataset_dists:
|
|
254
|
+
spec["dataset_distributions"] = dataset_dists
|
|
199
255
|
if oss_dists:
|
|
200
256
|
spec["oss_distributions"] = oss_dists
|
|
201
257
|
return {
|
|
@@ -568,33 +568,31 @@ class Repo:
|
|
|
568
568
|
def my_user_data_permissions(self, uid: str, subpath: str = "") -> list[str]:
|
|
569
569
|
"""返回当前用户对某用户数据资源(uid 级或子路径级)拥有的角色名列表。
|
|
570
570
|
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
571
|
+
经 GET /apis/v1/me/resource-permissions 自查有效动作(get=读、use=写、
|
|
572
|
+
assign=管 ACL),再映射回短名:assign→admin、use→writer、get→reader
|
|
573
|
+
(取能力最高档)。目录主人恒为 admin(与服务端 owner 快路径语义一致)。
|
|
574
|
+
|
|
575
|
+
历史实现走 GET /apis/v1/role-bindings 列表再客户端过滤,但该端点对非
|
|
576
|
+
owner 调用方有 assign 门控,被授权者查自己 principal 恒得空列表,故改用
|
|
577
|
+
/me/resource-permissions 自查端点。
|
|
575
578
|
"""
|
|
576
579
|
if not str(uid).strip():
|
|
577
580
|
raise SchemaError("uid must not be empty")
|
|
578
581
|
from urllib.parse import quote
|
|
582
|
+
# 先构造 resource_id:subpath 校验在任何网络请求之前完成(对称 Go/Java)。
|
|
579
583
|
resource_id = self._user_data_acl_resource_id(uid, subpath)
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
f"&
|
|
584
|
+
if str(uid).strip() == str(self.whoami()).strip():
|
|
585
|
+
return ["admin"]
|
|
586
|
+
path = (f"/apis/v1/me/resource-permissions?resource_type=user_data"
|
|
587
|
+
f"&resource_id={quote(resource_id, safe='')}")
|
|
584
588
|
data = self.transport.get(path)
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
matched: list[str] = []
|
|
596
|
-
for b, role_name in zip(bindings or [], roles):
|
|
597
|
-
rid = b.get("resource_id", "") if isinstance(b, dict) else ""
|
|
598
|
-
if rid == resource_id:
|
|
599
|
-
matched.append(role_name)
|
|
600
|
-
return matched
|
|
589
|
+
actions = data.get("actions", []) if isinstance(data, dict) else []
|
|
590
|
+
if not actions:
|
|
591
|
+
return []
|
|
592
|
+
if "assign" in actions:
|
|
593
|
+
return ["admin"]
|
|
594
|
+
if "use" in actions:
|
|
595
|
+
return ["writer"]
|
|
596
|
+
if "get" in actions:
|
|
597
|
+
return ["reader"]
|
|
598
|
+
return []
|
|
@@ -82,11 +82,40 @@ class ApiMetadataRegistry(BaseMetadataRegistry):
|
|
|
82
82
|
return base + ("&" + "&".join(extra) if extra else "")
|
|
83
83
|
|
|
84
84
|
def register(self, meta: DatasetInstance) -> DatasetInstance:
|
|
85
|
+
"""注册实例元数据(pending)。
|
|
86
|
+
|
|
87
|
+
apiserver 契约(POST /apis/v1/datasets/instances):
|
|
88
|
+
|
|
89
|
+
* dataset / version / split 走 **query string**(``dataset_name`` /
|
|
90
|
+
``version`` / ``split``),body 里没有 dataset 字段——handler 用
|
|
91
|
+
``DisallowUnknownFields`` 解码,发 ``dataset``/``dataset_version`` 会被
|
|
92
|
+
400 拒;query 缺 dataset_name 则在 split_first 定位时 92005。
|
|
93
|
+
* body 只接受 ``instance_id`` / ``split`` / ``manifest`` / ``tags`` /
|
|
94
|
+
``difficulty`` / ``docker_image`` / ``extend_meta``(
|
|
95
|
+
createDatasetInstanceRequest 白名单)。
|
|
96
|
+
* split 的 ``default`` 哨兵归一为空串(与写路径 _version_prefix 对称)。
|
|
97
|
+
|
|
98
|
+
owner_reference 等其余字段在 register 端点无承接位(owner 由 apiserver
|
|
99
|
+
据 caller 派生),故不进 body。
|
|
100
|
+
"""
|
|
85
101
|
meta.validate()
|
|
102
|
+
split = "" if meta.split == "default" else (meta.split or "")
|
|
103
|
+
body: dict = {"instance_id": meta.instance_id, "split": split}
|
|
104
|
+
if meta.manifest:
|
|
105
|
+
body["manifest"] = [m.to_dict() for m in meta.manifest]
|
|
106
|
+
if meta.tags:
|
|
107
|
+
body["tags"] = meta.tags
|
|
108
|
+
if meta.difficulty is not None:
|
|
109
|
+
body["difficulty"] = meta.difficulty
|
|
110
|
+
if meta.docker_image:
|
|
111
|
+
body["docker_image"] = meta.docker_image
|
|
112
|
+
if meta.extend_meta:
|
|
113
|
+
body["extend_meta"] = meta.extend_meta
|
|
114
|
+
path = f"{self._BASE}?{self._q(meta.dataset, meta.dataset_version, split)}"
|
|
86
115
|
with _unimplemented(
|
|
87
116
|
"POST /apis/v1/datasets/instances(实例注册)",
|
|
88
117
|
"当前可用路径:push(register=False) 只传 OSS,再用 versions.ingest() 扫描入库。"):
|
|
89
|
-
data = self._t.post(
|
|
118
|
+
data = self._t.post(path, body)
|
|
90
119
|
return DatasetInstance.from_dict(data)
|
|
91
120
|
|
|
92
121
|
def commit(self, instance_uid: str, digests: dict | None,
|
|
@@ -171,6 +171,9 @@ class Transport:
|
|
|
171
171
|
self.env = env
|
|
172
172
|
self.timeout = timeout
|
|
173
173
|
self._send = http_send or _KeepAliveSender()
|
|
174
|
+
# 自定义发送器身份参与缓存键(见 config._cluster_cache_key):注入过 fake/代理的
|
|
175
|
+
# transport 即使 base/env 相同也各自隔离,与 Go/Java 的 customSend 语义一致。
|
|
176
|
+
self._custom_send = http_send is not None
|
|
174
177
|
self._retry_attempts = retry_attempts
|
|
175
178
|
# STS 凭据缓存:key=("oss",bucket,prefix,duration) / ("acr",host,duration),
|
|
176
179
|
# 按签发返回的 Expiration 失效(留 60s 安全余量)。关缓存则每次现签。
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{instance_repo-1.0.2 → instance_repo-1.0.7.dev0}/instance_repo.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|