instance-repo 1.1.4__tar.gz → 1.3.0.dev0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/PKG-INFO +63 -4
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/README.md +61 -3
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/__init__.py +1 -1
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/artifacts.py +51 -125
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/cli.py +32 -9
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/splits.py +112 -58
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/versions.py +18 -10
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/repo.py +19 -7
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo.egg-info/PKG-INFO +63 -4
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo.egg-info/requires.txt +1 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/pyproject.toml +2 -1
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/_routing.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/_site_defaults.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/bootstrap.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/cache.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/__init__.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/benchmarks.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/config.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/datasets.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/images.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/instances.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/reports.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/scaffold.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/clients/workflows.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/concurrency.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/content.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/endpoints.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/errors.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/image.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/layout.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/loader.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/models.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/paths.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/release.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/retry.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/store/__init__.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/store/acr.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/store/base.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/store/oss.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/store/registry.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/tasktoml.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/transport.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo/validate.py +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo.egg-info/SOURCES.txt +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo.egg-info/dependency_links.txt +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo.egg-info/entry_points.txt +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo.egg-info/top_level.txt +0 -0
- {instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/setup.cfg +0 -0
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: instance-repo
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0.dev0
|
|
4
4
|
Summary: InstanceRepo SDK — 评测 Instance 统一存储客户端(SDK-only;控制面由 apiserver 承载)
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: irepo-artifact<0.2.0,>=0.1.1
|
|
7
8
|
Requires-Dist: tomli>=1.1.0; python_version < "3.11"
|
|
8
9
|
Provides-Extra: oss
|
|
9
10
|
Requires-Dist: oss2>=2.19; extra == "oss"
|
|
@@ -81,8 +82,9 @@ pip install "instance-repo[oss]" # 需要数据面 OSS 读写时,附带 os
|
|
|
81
82
|
不是 default)。旧数据集显式传 `Repo(metadata_model="version_first")`,其对象布局与镜像 tag
|
|
82
83
|
**逐字节不变**,存量数据无需迁移。
|
|
83
84
|
|
|
84
|
-
服务端把空 `metadata_model` 归一为 `version_first
|
|
85
|
-
|
|
85
|
+
服务端把空 `metadata_model` 归一为 `version_first`。数据面与读路径仍按 profile 的模型工作;
|
|
86
|
+
但 `repo.versions.create(...)` 的 `metadata_model` 仅在调用方显式传入非空值时下发,不从
|
|
87
|
+
profile 自动注入。显式 `split_first` 仍由服务端拒绝;创建 split 应使用 `repo.splits.create(...)`。
|
|
86
88
|
|
|
87
89
|
| | `split_first`(默认) | `version_first`(旧) |
|
|
88
90
|
| --- | --- | --- |
|
|
@@ -96,6 +98,29 @@ pip install "instance-repo[oss]" # 需要数据面 OSS 读写时,附带 os
|
|
|
96
98
|
92005 ambiguous。显式给 splits 时服务端只列举 `{split}/` 与 `{split}-assets/`,推断被完全绕过。
|
|
97
99
|
确需推断时传 `allow_layout_inference=True`。
|
|
98
100
|
|
|
101
|
+
## Split 创建
|
|
102
|
+
|
|
103
|
+
`repo.splits.create(dataset, split, *, version=None, run_type=None, storage_path=None,
|
|
104
|
+
environment=None)` 直接调用 `POST /apis/v1/datasets/splits`,body 固定为
|
|
105
|
+
`{"splits":["<split>"]}`。`version` 可选;省略即创建 unversioned split,query 中不会发送
|
|
106
|
+
`version=`。存储路径由服务端根据 dataset/version/split 推导。
|
|
107
|
+
|
|
108
|
+
`storage_path` 只为 1.2.1 调用兼容保留(计划 1.3.0 删除):空值允许,非空值在网络请求前抛
|
|
109
|
+
`SchemaError`。提供 `run_type` 时,创建后通过 `PATCH /apis/v1/datasets/splits/detail` 更新,
|
|
110
|
+
不会调用 Version PATCH;最后 GET 读回权威 Split 记录。
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
repo.splits.create("alibaba/mybench", "test")
|
|
114
|
+
repo.splits.create("alibaba/mybench", "test", version="v1", run_type="eval")
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
对应 CLI 的 `--version` 也是可选项;`--storage-path` 仅兼容旧脚本,非空值会失败:
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
irepo split create alibaba/mybench/test --run-type eval
|
|
121
|
+
irepo split create alibaba/mybench/test --version v1
|
|
122
|
+
```
|
|
123
|
+
|
|
99
124
|
## 独立镜像能力(`repo.images`,0.8 新增)
|
|
100
125
|
|
|
101
126
|
```python
|
|
@@ -152,7 +177,8 @@ irepo --help
|
|
|
152
177
|
# 20 个子命令:validate / push / push-many / deliver / pull / list / get
|
|
153
178
|
# publish / grant / feedback / report / whoami / user-data
|
|
154
179
|
# repo-config / create / claim / version / image
|
|
155
|
-
# split(list/get/set-run-type
|
|
180
|
+
# split(list/get/create/set-run-type;create 的 --version 可选,--storage-path 非空即拒绝)
|
|
181
|
+
# dataset(list/get/update/access/revoke/benchmarks/workflows)
|
|
156
182
|
irepo image ref --dataset alibaba/mybench --instance-id inst-1 --split test
|
|
157
183
|
irepo image pull '<acr>/<ns>/mybench:test-inst-1' ./out.tar
|
|
158
184
|
```
|
|
@@ -175,3 +201,36 @@ irepo publish alibaba/mybench/v1 --split test \
|
|
|
175
201
|
版本号在 `pyproject.toml` / `instance_repo.__version__` / Go `version.go` /
|
|
176
202
|
conformance golden 四处保持一致。各环境的真实配置取值、端到端示例与错误码对照,见团队内部
|
|
177
203
|
部署文档。
|
|
204
|
+
|
|
205
|
+
## 产物下载改造(开发版)
|
|
206
|
+
|
|
207
|
+
产物下载现在委托独立 `irepo-artifact` 包,通过 AP 签发的 HTTP 链接下载,
|
|
208
|
+
不再要求 OSS SDK、STS 或 artifact bucket/prefix 配置;其他存储功能的 OSS 依赖不变。
|
|
209
|
+
本地开发需同时安装两个包:
|
|
210
|
+
|
|
211
|
+
```bash
|
|
212
|
+
python -m pip install -e ./sdk/python-artifact -e ./sdk/python
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
保留 `irepo artifact download`、`download-many`、`download-group`。三个命令均支持
|
|
216
|
+
`--include-files`、`--exclude-files`(可重复、可逗号分隔)和 `--extract`。
|
|
217
|
+
|
|
218
|
+
```bash
|
|
219
|
+
irepo artifact download --job-id job-a --output-dir out/job-a
|
|
220
|
+
irepo artifact download-many --job-ids job-a,job-b --output-dir out --extract
|
|
221
|
+
irepo artifact download-group --group-id group-a --output-dir out \
|
|
222
|
+
--include-files logs,result.json --exclude-files debug.log,tmp
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
默认不解压;`--extract` 与文件筛选冲突。筛选通过 manifest 得到最终文件列表,再向 AP 获取
|
|
226
|
+
各文件 URL。不开启筛选时优先 OSS,网络不可达或中断时回退 dataplane。
|
|
227
|
+
|
|
228
|
+
**路径迁移:** 单 Job 的输出目录就是 Job 目录,包为 `<output_dir>/result.tgz`;
|
|
229
|
+
批量/Group 为 `<output_dir>/jobs/<job_id>/result.tgz`。解压或筛选内容位于 Job 目录的
|
|
230
|
+
`artifacts/output/`。旧 `<job_id>.tgz` 文件名统一为 `result.tgz`,旧脚本应使用 SDK 返回路径。
|
|
231
|
+
旧 SDK 单 Job 仍返回字符串,批量仍返回路径或异常字典;旧入口维持覆盖意图,新独立客户端
|
|
232
|
+
默认拒绝覆盖。请勿在旧目录尚有需保留文件时直接覆盖迁移。
|
|
233
|
+
|
|
234
|
+
权限由 AP artifact 接口统一校验,SDK 不再执行 Job 详情预检查。
|
|
235
|
+
更多行为与安装说明见 [独立包 README](../python-artifact/README.md)。
|
|
236
|
+
独立包尚为开发版本,正式发布完整 SDK 前必须先发布可安装的独立包版本。
|
|
@@ -68,8 +68,9 @@ pip install "instance-repo[oss]" # 需要数据面 OSS 读写时,附带 os
|
|
|
68
68
|
不是 default)。旧数据集显式传 `Repo(metadata_model="version_first")`,其对象布局与镜像 tag
|
|
69
69
|
**逐字节不变**,存量数据无需迁移。
|
|
70
70
|
|
|
71
|
-
服务端把空 `metadata_model` 归一为 `version_first
|
|
72
|
-
|
|
71
|
+
服务端把空 `metadata_model` 归一为 `version_first`。数据面与读路径仍按 profile 的模型工作;
|
|
72
|
+
但 `repo.versions.create(...)` 的 `metadata_model` 仅在调用方显式传入非空值时下发,不从
|
|
73
|
+
profile 自动注入。显式 `split_first` 仍由服务端拒绝;创建 split 应使用 `repo.splits.create(...)`。
|
|
73
74
|
|
|
74
75
|
| | `split_first`(默认) | `version_first`(旧) |
|
|
75
76
|
| --- | --- | --- |
|
|
@@ -83,6 +84,29 @@ pip install "instance-repo[oss]" # 需要数据面 OSS 读写时,附带 os
|
|
|
83
84
|
92005 ambiguous。显式给 splits 时服务端只列举 `{split}/` 与 `{split}-assets/`,推断被完全绕过。
|
|
84
85
|
确需推断时传 `allow_layout_inference=True`。
|
|
85
86
|
|
|
87
|
+
## Split 创建
|
|
88
|
+
|
|
89
|
+
`repo.splits.create(dataset, split, *, version=None, run_type=None, storage_path=None,
|
|
90
|
+
environment=None)` 直接调用 `POST /apis/v1/datasets/splits`,body 固定为
|
|
91
|
+
`{"splits":["<split>"]}`。`version` 可选;省略即创建 unversioned split,query 中不会发送
|
|
92
|
+
`version=`。存储路径由服务端根据 dataset/version/split 推导。
|
|
93
|
+
|
|
94
|
+
`storage_path` 只为 1.2.1 调用兼容保留(计划 1.3.0 删除):空值允许,非空值在网络请求前抛
|
|
95
|
+
`SchemaError`。提供 `run_type` 时,创建后通过 `PATCH /apis/v1/datasets/splits/detail` 更新,
|
|
96
|
+
不会调用 Version PATCH;最后 GET 读回权威 Split 记录。
|
|
97
|
+
|
|
98
|
+
```python
|
|
99
|
+
repo.splits.create("alibaba/mybench", "test")
|
|
100
|
+
repo.splits.create("alibaba/mybench", "test", version="v1", run_type="eval")
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
对应 CLI 的 `--version` 也是可选项;`--storage-path` 仅兼容旧脚本,非空值会失败:
|
|
104
|
+
|
|
105
|
+
```bash
|
|
106
|
+
irepo split create alibaba/mybench/test --run-type eval
|
|
107
|
+
irepo split create alibaba/mybench/test --version v1
|
|
108
|
+
```
|
|
109
|
+
|
|
86
110
|
## 独立镜像能力(`repo.images`,0.8 新增)
|
|
87
111
|
|
|
88
112
|
```python
|
|
@@ -139,7 +163,8 @@ irepo --help
|
|
|
139
163
|
# 20 个子命令:validate / push / push-many / deliver / pull / list / get
|
|
140
164
|
# publish / grant / feedback / report / whoami / user-data
|
|
141
165
|
# repo-config / create / claim / version / image
|
|
142
|
-
# split(list/get/set-run-type
|
|
166
|
+
# split(list/get/create/set-run-type;create 的 --version 可选,--storage-path 非空即拒绝)
|
|
167
|
+
# dataset(list/get/update/access/revoke/benchmarks/workflows)
|
|
143
168
|
irepo image ref --dataset alibaba/mybench --instance-id inst-1 --split test
|
|
144
169
|
irepo image pull '<acr>/<ns>/mybench:test-inst-1' ./out.tar
|
|
145
170
|
```
|
|
@@ -162,3 +187,36 @@ irepo publish alibaba/mybench/v1 --split test \
|
|
|
162
187
|
版本号在 `pyproject.toml` / `instance_repo.__version__` / Go `version.go` /
|
|
163
188
|
conformance golden 四处保持一致。各环境的真实配置取值、端到端示例与错误码对照,见团队内部
|
|
164
189
|
部署文档。
|
|
190
|
+
|
|
191
|
+
## 产物下载改造(开发版)
|
|
192
|
+
|
|
193
|
+
产物下载现在委托独立 `irepo-artifact` 包,通过 AP 签发的 HTTP 链接下载,
|
|
194
|
+
不再要求 OSS SDK、STS 或 artifact bucket/prefix 配置;其他存储功能的 OSS 依赖不变。
|
|
195
|
+
本地开发需同时安装两个包:
|
|
196
|
+
|
|
197
|
+
```bash
|
|
198
|
+
python -m pip install -e ./sdk/python-artifact -e ./sdk/python
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
保留 `irepo artifact download`、`download-many`、`download-group`。三个命令均支持
|
|
202
|
+
`--include-files`、`--exclude-files`(可重复、可逗号分隔)和 `--extract`。
|
|
203
|
+
|
|
204
|
+
```bash
|
|
205
|
+
irepo artifact download --job-id job-a --output-dir out/job-a
|
|
206
|
+
irepo artifact download-many --job-ids job-a,job-b --output-dir out --extract
|
|
207
|
+
irepo artifact download-group --group-id group-a --output-dir out \
|
|
208
|
+
--include-files logs,result.json --exclude-files debug.log,tmp
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
默认不解压;`--extract` 与文件筛选冲突。筛选通过 manifest 得到最终文件列表,再向 AP 获取
|
|
212
|
+
各文件 URL。不开启筛选时优先 OSS,网络不可达或中断时回退 dataplane。
|
|
213
|
+
|
|
214
|
+
**路径迁移:** 单 Job 的输出目录就是 Job 目录,包为 `<output_dir>/result.tgz`;
|
|
215
|
+
批量/Group 为 `<output_dir>/jobs/<job_id>/result.tgz`。解压或筛选内容位于 Job 目录的
|
|
216
|
+
`artifacts/output/`。旧 `<job_id>.tgz` 文件名统一为 `result.tgz`,旧脚本应使用 SDK 返回路径。
|
|
217
|
+
旧 SDK 单 Job 仍返回字符串,批量仍返回路径或异常字典;旧入口维持覆盖意图,新独立客户端
|
|
218
|
+
默认拒绝覆盖。请勿在旧目录尚有需保留文件时直接覆盖迁移。
|
|
219
|
+
|
|
220
|
+
权限由 AP artifact 接口统一校验,SDK 不再执行 Job 详情预检查。
|
|
221
|
+
更多行为与安装说明见 [独立包 README](../python-artifact/README.md)。
|
|
222
|
+
独立包尚为开发版本,正式发布完整 SDK 前必须先发布可安装的独立包版本。
|
|
@@ -1,19 +1,16 @@
|
|
|
1
|
-
"""artifacts.py — AP
|
|
1
|
+
"""artifacts.py — AP artifact HTTP 下载兼容入口。
|
|
2
2
|
|
|
3
3
|
调用方:普通用户 / UTS SDK,透传 apikey + job_id + cluster_id。
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
OSS:{ARTIFACT_PREFIX_ROOT}/{job_id}/result.tgz(前缀根经 _site_defaults 注入)。
|
|
4
|
+
鉴权:由 AP 产物接口负责,SDK 不再读取 Job 详情预检查。
|
|
5
|
+
下载委托独立 irepo-artifact 包;旧 bucket/key 工具保留但不参与下载。
|
|
7
6
|
"""
|
|
8
7
|
from __future__ import annotations
|
|
9
8
|
|
|
10
9
|
import json
|
|
11
10
|
import logging
|
|
12
11
|
import os
|
|
13
|
-
import posixpath
|
|
14
12
|
import urllib.error
|
|
15
13
|
import urllib.request
|
|
16
|
-
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
17
14
|
from typing import Any
|
|
18
15
|
|
|
19
16
|
from .errors import Forbidden, InstanceRepoError
|
|
@@ -216,136 +213,65 @@ def fetch_group_job_ids(api_key: str, group_id: str, cluster: str,
|
|
|
216
213
|
return job_ids
|
|
217
214
|
|
|
218
215
|
|
|
219
|
-
def
|
|
220
|
-
|
|
216
|
+
def _client(api_key: str, cluster: str, ap_endpoint: str | None):
|
|
217
|
+
from irepo_artifact import ArtifactClient
|
|
218
|
+
return ArtifactClient(ap_endpoint=_ap_endpoint(ap_endpoint), api_key=api_key, cluster=cluster)
|
|
221
219
|
|
|
222
|
-
artifact 文件名不固定:实测两种命名(result.tgz 与 {job_id}.tgz)。
|
|
223
|
-
先 list 前缀目录找非空的 .tgz 对象,兼容所有命名;找不到 → ArtifactNotFoundError。
|
|
224
220
|
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
221
|
+
def _legacy_error(error):
|
|
222
|
+
from irepo_artifact import AccessDenied, ArtifactNotFound, NoMatchingFiles
|
|
223
|
+
if isinstance(error, AccessDenied):
|
|
224
|
+
mapped = ArtifactPermissionError(str(error), cause=error)
|
|
225
|
+
elif isinstance(error, (ArtifactNotFound, NoMatchingFiles)):
|
|
226
|
+
mapped = ArtifactNotFoundError(str(error), cause=error)
|
|
227
|
+
else:
|
|
228
|
+
mapped = InstanceRepoError(f"{type(error).__name__}: {error}", cause=error)
|
|
229
|
+
mapped.artifact_error = error
|
|
230
|
+
mapped.error_type = type(error).__name__
|
|
231
|
+
mapped.status = getattr(error, 'status', None)
|
|
232
|
+
return mapped
|
|
230
233
|
|
|
231
|
-
# 获取 STS(同数据面多 job 共享缓存)
|
|
232
|
-
creds = transport.get_oss_credentials(bucket, prefix)
|
|
233
|
-
auth = oss2.StsAuth(
|
|
234
|
-
creds["access_key_id"],
|
|
235
|
-
creds["access_key_secret"],
|
|
236
|
-
creds["security_token"],
|
|
237
|
-
)
|
|
238
234
|
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
if not endpoint:
|
|
242
|
-
raise InstanceRepoError("STS response missing endpoint")
|
|
235
|
+
def _result_path(result):
|
|
236
|
+
return str(result.files_dir if result.files_dir is not None else result.archive_path)
|
|
243
237
|
|
|
244
|
-
bkt = oss2.Bucket(auth, endpoint, bucket)
|
|
245
|
-
|
|
246
|
-
# 1. list 前缀目录,找 artifact tgz 对象(跳过 size=0 的目录占位符)
|
|
247
|
-
key = None
|
|
248
|
-
for obj in oss2.ObjectIterator(bkt, prefix=prefix):
|
|
249
|
-
if obj.size > 0 and obj.key.endswith(".tgz"):
|
|
250
|
-
key = obj.key
|
|
251
|
-
break
|
|
252
|
-
if not key:
|
|
253
|
-
raise ArtifactNotFoundError(
|
|
254
|
-
f"no artifact tgz found under {prefix!r} for job {job_id!r}")
|
|
255
|
-
|
|
256
|
-
# 2. 本地路径(保留 OSS 上的真实文件名)
|
|
257
|
-
job_dir = os.path.join(output_dir, job_id)
|
|
258
|
-
os.makedirs(job_dir, exist_ok=True)
|
|
259
|
-
local_path = os.path.join(job_dir, os.path.basename(key.rstrip("/")))
|
|
260
|
-
|
|
261
|
-
# 3. 下载(带 OSS 限流退避)
|
|
262
|
-
wait = OSS_RETRY_INITIAL_WAIT
|
|
263
|
-
for attempt in range(1, OSS_RETRY_MAX_ATTEMPTS + 1):
|
|
264
|
-
try:
|
|
265
|
-
bkt.get_object_to_file(key, local_path)
|
|
266
|
-
logger.info("downloaded %s → %s", key, local_path)
|
|
267
|
-
return local_path
|
|
268
|
-
except oss2.exceptions.ServerError as e:
|
|
269
|
-
if e.status == 503 and attempt < OSS_RETRY_MAX_ATTEMPTS:
|
|
270
|
-
logger.warning("OSS 503 for %s, retry %d/%d in %.1fs",
|
|
271
|
-
job_id, attempt, OSS_RETRY_MAX_ATTEMPTS, wait)
|
|
272
|
-
import time
|
|
273
|
-
time.sleep(wait)
|
|
274
|
-
wait = min(wait * 2, OSS_RETRY_MAX_WAIT)
|
|
275
|
-
else:
|
|
276
|
-
raise
|
|
277
|
-
return local_path # unreachable, but keeps type checker happy
|
|
278
|
-
|
|
279
|
-
|
|
280
|
-
# ---------------------------------------------------------------------------
|
|
281
|
-
# 公开 API(给 Repo 类调用)
|
|
282
|
-
# ---------------------------------------------------------------------------
|
|
283
238
|
|
|
284
239
|
def download_artifact(transport, api_key: str, job_id: str,
|
|
285
240
|
cluster: str, output_dir: str = ".",
|
|
286
|
-
ap_endpoint: str | None = None
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
output_dir: 本地输出目录
|
|
295
|
-
ap_endpoint: AP 控制面 URL(默认从环境变量或常量取)
|
|
296
|
-
|
|
297
|
-
Returns:
|
|
298
|
-
下载文件的本地路径
|
|
241
|
+
ap_endpoint: str | None = None, *,
|
|
242
|
+
include_files: list[str] | None = None,
|
|
243
|
+
exclude_files: list[str] | None = None,
|
|
244
|
+
extract: bool = False) -> str:
|
|
245
|
+
"""Download through AP-issued HTTP links; transport retained for compatibility.
|
|
246
|
+
|
|
247
|
+
output_dir is the Job directory. AP artifact endpoints enforce access;
|
|
248
|
+
no Job precheck, STS or object-storage SDK is used.
|
|
299
249
|
"""
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
250
|
+
from irepo_artifact import ArtifactError
|
|
251
|
+
try:
|
|
252
|
+
result = _client(api_key, cluster, ap_endpoint).download(
|
|
253
|
+
job_id, output_dir, include_files=include_files,
|
|
254
|
+
exclude_files=exclude_files, extract=extract, overwrite=True)
|
|
255
|
+
return _result_path(result)
|
|
256
|
+
except ArtifactError as exc:
|
|
257
|
+
raise _legacy_error(exc) from exc
|
|
308
258
|
|
|
309
259
|
|
|
310
260
|
def download_artifacts(transport, api_key: str, job_ids: list[str],
|
|
311
261
|
cluster: str, output_dir: str = ".",
|
|
312
262
|
max_concurrent: int = DEFAULT_MAX_CONCURRENT,
|
|
313
|
-
ap_endpoint: str | None = None,
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
324
|
-
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
Returns:
|
|
329
|
-
{job_id: 本地文件路径 或 Exception}
|
|
330
|
-
"""
|
|
331
|
-
bucket = resolve_artifact_bucket(cluster)
|
|
332
|
-
results: dict[str, str | Exception] = {}
|
|
333
|
-
|
|
334
|
-
def _do(jid: str) -> tuple[str, str | Exception]:
|
|
335
|
-
try:
|
|
336
|
-
# AP workspace ACL 检查
|
|
337
|
-
fetch_job_info(api_key, jid, cluster, ap_endpoint)
|
|
338
|
-
# 下载
|
|
339
|
-
path = _download_one(transport, bucket, jid, output_dir)
|
|
340
|
-
return jid, path
|
|
341
|
-
except Exception as e:
|
|
342
|
-
logger.error("failed to download artifact for job %s: %s", jid, e)
|
|
343
|
-
return jid, e
|
|
344
|
-
|
|
345
|
-
with ThreadPoolExecutor(max_workers=max_concurrent) as pool:
|
|
346
|
-
futures = {pool.submit(_do, jid): jid for jid in job_ids}
|
|
347
|
-
for future in as_completed(futures):
|
|
348
|
-
jid, result = future.result()
|
|
349
|
-
results[jid] = result
|
|
350
|
-
|
|
351
|
-
return results
|
|
263
|
+
ap_endpoint: str | None = None, *,
|
|
264
|
+
include_files: list[str] | None = None,
|
|
265
|
+
exclude_files: list[str] | None = None,
|
|
266
|
+
extract: bool = False) -> dict[str, str | Exception]:
|
|
267
|
+
"""Download into output_dir/jobs/<job_id>; retain per-Job path/error results."""
|
|
268
|
+
from irepo_artifact import ArtifactError
|
|
269
|
+
try:
|
|
270
|
+
results = _client(api_key, cluster, ap_endpoint).download_many(
|
|
271
|
+
job_ids, output_dir, max_concurrent=max_concurrent,
|
|
272
|
+
include_files=include_files, exclude_files=exclude_files,
|
|
273
|
+
extract=extract, overwrite=True)
|
|
274
|
+
except ArtifactError as exc:
|
|
275
|
+
raise _legacy_error(exc) from exc
|
|
276
|
+
return {jid: (_legacy_error(result.error) if result.error is not None
|
|
277
|
+
else _result_path(result)) for jid, result in results.items()}
|
|
@@ -341,7 +341,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
341
341
|
sp.add_argument("--split", default=None,
|
|
342
342
|
help="目标 split 名(get/set-run-type 用)")
|
|
343
343
|
sp.add_argument("--version", default=None,
|
|
344
|
-
help="list 按 version 精确过滤;get
|
|
344
|
+
help="list 按 version 精确过滤;get/create 可选定位版本")
|
|
345
345
|
sp.add_argument("--scope", choices=["unversioned", "exact", "all"], default=None,
|
|
346
346
|
help="list 跨版本策略:缺省无 version → all(SDK 取全量哲学),"
|
|
347
347
|
"有 version 省略(服务端只接受 exact)")
|
|
@@ -351,7 +351,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
351
351
|
dest="run_type",
|
|
352
352
|
help="set-run-type/create 用:train/eval 或空串清除")
|
|
353
353
|
sp.add_argument("--storage-path", default=None, dest="storage_path",
|
|
354
|
-
help="create
|
|
354
|
+
help="create 的 1.2.1 兼容参数;非空值会被拒绝,路径由服务端推导")
|
|
355
355
|
sp.add_argument("--page", type=int, default=1)
|
|
356
356
|
sp.add_argument("--page-size", type=int, default=50, dest="page_size")
|
|
357
357
|
sp.add_argument("--environment", default=None,
|
|
@@ -407,7 +407,7 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
407
407
|
|
|
408
408
|
# ── Artifact 下载(STS 直连 OSS,workspace ACL 鉴权)──
|
|
409
409
|
art = sub.add_parser("artifact",
|
|
410
|
-
help="AP job artifact 下载(
|
|
410
|
+
help="AP job artifact 下载(AP HTTP 链接)")
|
|
411
411
|
asub = art.add_subparsers(dest="artifact_cmd", required=True)
|
|
412
412
|
|
|
413
413
|
adl = asub.add_parser("download", help="下载单个 job 的 artifact")
|
|
@@ -437,6 +437,16 @@ def build_parser() -> argparse.ArgumentParser:
|
|
|
437
437
|
adlg.add_argument("--concurrent", type=int, default=8)
|
|
438
438
|
adlg.add_argument("--ap-endpoint", default=None, dest="ap_endpoint")
|
|
439
439
|
|
|
440
|
+
for artifact_parser in (adl, adlm, adlg):
|
|
441
|
+
artifact_parser.add_argument(
|
|
442
|
+
"--include-files", action="append", default=None,
|
|
443
|
+
help="Only download matching artifact files or folders (repeatable; comma-separated)")
|
|
444
|
+
artifact_parser.add_argument(
|
|
445
|
+
"--exclude-files", action="append", default=None,
|
|
446
|
+
help="Skip matching artifact files or folders (repeatable; comma-separated)")
|
|
447
|
+
artifact_parser.add_argument("--extract", action="store_true",
|
|
448
|
+
help="Extract the complete archive (default: keep archive only)")
|
|
449
|
+
|
|
440
450
|
return ap
|
|
441
451
|
|
|
442
452
|
|
|
@@ -609,6 +619,16 @@ def _effective_cluster(args) -> str | None:
|
|
|
609
619
|
def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
610
620
|
args = _parse_args(argv)
|
|
611
621
|
out = out or sys.stdout
|
|
622
|
+
if args.cmd == "artifact":
|
|
623
|
+
for name in ("include_files", "exclude_files"):
|
|
624
|
+
raw = getattr(args, name)
|
|
625
|
+
values = list(dict.fromkeys(v.strip() for item in (raw or [])
|
|
626
|
+
for v in item.split(",") if v.strip()))
|
|
627
|
+
if raw is not None and not values:
|
|
628
|
+
raise InstanceRepoError(f"--{name.replace('_', '-')} 至少需要一个非空的文件或目录选择器")
|
|
629
|
+
setattr(args, name, values or None)
|
|
630
|
+
if args.extract and (args.include_files or args.exclude_files):
|
|
631
|
+
raise InstanceRepoError("--extract 不能与 --include-files 或 --exclude-files 同时使用")
|
|
612
632
|
cluster = _effective_cluster(args)
|
|
613
633
|
r = repo or Repo(profile=args.profile, api_env=getattr(args, "api_env", None),
|
|
614
634
|
api_base=getattr(args, "api_base", None), cluster=cluster,
|
|
@@ -1015,9 +1035,6 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
1015
1035
|
return 0
|
|
1016
1036
|
if args.action == "create":
|
|
1017
1037
|
dataset, split_name = _resolve_split_ref(args)
|
|
1018
|
-
if not (args.version or "").strip():
|
|
1019
|
-
raise SystemExit(
|
|
1020
|
-
"split create 需要 --version(apiserver CreateVersion 硬约束)")
|
|
1021
1038
|
sp = r.splits.create(dataset, split_name,
|
|
1022
1039
|
version=args.version,
|
|
1023
1040
|
run_type=args.run_type,
|
|
@@ -1172,7 +1189,9 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
1172
1189
|
path = r.download_artifact(
|
|
1173
1190
|
args.job_id, cluster=args.cluster,
|
|
1174
1191
|
output_dir=args.output_dir,
|
|
1175
|
-
ap_endpoint=args.ap_endpoint
|
|
1192
|
+
ap_endpoint=args.ap_endpoint,
|
|
1193
|
+
include_files=args.include_files, exclude_files=args.exclude_files,
|
|
1194
|
+
extract=args.extract)
|
|
1176
1195
|
print(str(path), file=out)
|
|
1177
1196
|
return 0
|
|
1178
1197
|
if args.artifact_cmd == "download-many":
|
|
@@ -1181,7 +1200,9 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
1181
1200
|
job_ids, cluster=args.cluster,
|
|
1182
1201
|
output_dir=args.output_dir,
|
|
1183
1202
|
max_concurrent=args.concurrent,
|
|
1184
|
-
ap_endpoint=args.ap_endpoint
|
|
1203
|
+
ap_endpoint=args.ap_endpoint,
|
|
1204
|
+
include_files=args.include_files, exclude_files=args.exclude_files,
|
|
1205
|
+
extract=args.extract)
|
|
1185
1206
|
ok, fail = 0, 0
|
|
1186
1207
|
for jid, res in results.items():
|
|
1187
1208
|
if isinstance(res, Exception):
|
|
@@ -1197,7 +1218,9 @@ def main(argv=None, *, repo: Repo | None = None, out=None) -> int:
|
|
|
1197
1218
|
args.group_id, cluster=args.cluster,
|
|
1198
1219
|
output_dir=args.output_dir,
|
|
1199
1220
|
max_concurrent=args.concurrent,
|
|
1200
|
-
ap_endpoint=args.ap_endpoint
|
|
1221
|
+
ap_endpoint=args.ap_endpoint,
|
|
1222
|
+
include_files=args.include_files, exclude_files=args.exclude_files,
|
|
1223
|
+
extract=args.extract)
|
|
1201
1224
|
ok, fail = 0, 0
|
|
1202
1225
|
for jid, res in results.items():
|
|
1203
1226
|
if isinstance(res, Exception):
|
|
@@ -5,11 +5,11 @@
|
|
|
5
5
|
version_scope=unversioned——只列无版本 split;跨全部版本须显式 version_scope=all);
|
|
6
6
|
status 过滤;environment(默认 online);分页默认 page=1/page_size=50(上限 200)。
|
|
7
7
|
* ``GET /datasets/splits/detail``:dataset_name+split 必填、version 可选;裸记录。
|
|
8
|
+
* ``POST /datasets/splits``:dataset_name 必填、version 可选;body 为
|
|
9
|
+
``{"splits": [...]}``,响应兼容 DatasetVersion,需再 GET split 详情确认。
|
|
8
10
|
* ``PATCH /datasets/splits/detail``:dataset_split_id 或 dataset_name+split(+version)
|
|
9
|
-
寻址;body 仅 run_type(必填,"train"/"eval"
|
|
10
|
-
|
|
11
|
-
* split 无 create/delete 端点:split 记录由 ingest 扫描 OSS 或
|
|
12
|
-
versions.create/update(splits=[...]) 间接产生——SDK 不伪造这些能力。
|
|
11
|
+
寻址;body 仅 run_type(必填,"train"/"eval",空串=清除);environment 与
|
|
12
|
+
POST/GET 保持一致;返回更新后的记录。
|
|
13
13
|
|
|
14
14
|
设计决策(方案 D1,已评审):``SplitsClient.list`` 在 version 为空且未显式传
|
|
15
15
|
version_scope 时**默认 all**(跨全部版本+无版本 split)。理由:SDK 的 list 语义
|
|
@@ -21,9 +21,10 @@ from __future__ import annotations
|
|
|
21
21
|
|
|
22
22
|
from urllib.parse import quote
|
|
23
23
|
|
|
24
|
-
from ..errors import SchemaError, TransportError
|
|
24
|
+
from ..errors import SchemaError, TransportError, WriteOutcomeUnknown
|
|
25
25
|
from ..models import (DatasetSplit, PagedResult, RUN_TYPE_VALUES,
|
|
26
26
|
SPLIT_STATUS_VALUES)
|
|
27
|
+
from ..retry import retry_transient
|
|
27
28
|
|
|
28
29
|
# 服务端 ListSplits 认可的 version_scope 取值(recordsvc)。
|
|
29
30
|
VERSION_SCOPE_VALUES = ("unversioned", "exact", "all")
|
|
@@ -222,57 +223,105 @@ class SplitsClient:
|
|
|
222
223
|
|
|
223
224
|
# ── 写 ──
|
|
224
225
|
|
|
226
|
+
def _post_create(self, dataset: str, split: str, version: str,
|
|
227
|
+
environment: str) -> None:
|
|
228
|
+
"""创建 split;POST 非幂等,响应是 DatasetVersion 兼容体,故不解析。"""
|
|
229
|
+
q = f"dataset_name={quote(dataset, safe='')}"
|
|
230
|
+
if version:
|
|
231
|
+
q += f"&version={quote(version, safe='')}"
|
|
232
|
+
if environment:
|
|
233
|
+
q += f"&environment={quote(environment, safe='')}"
|
|
234
|
+
self._t.post(f"{self._BASE}?{q}", {"splits": [split]},
|
|
235
|
+
idempotent=False)
|
|
236
|
+
|
|
225
237
|
def create(self, dataset: str, split: str, *,
|
|
226
|
-
version: str, run_type: str | None = None,
|
|
238
|
+
version: str | None = None, run_type: str | None = None,
|
|
227
239
|
storage_path: str | None = None,
|
|
228
240
|
environment: str | None = None) -> DatasetSplit:
|
|
229
|
-
"""
|
|
230
|
-
|
|
231
|
-
apiserver **没有** split create 端点(``POST /datasets/splits`` 不存在):
|
|
232
|
-
split 记录由 ingest 扫描 OSS 或 versions.create/update(splits=[...])
|
|
233
|
-
间接产生(pre 集合 find-or-upsert ensure 真实 split record)。本方法是
|
|
234
|
-
"先建空 split 壳"的官方路径升级为 SDK 直接 API:
|
|
241
|
+
"""经独立 split API 创建记录并从详情端点权威读回。
|
|
235
242
|
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
参数:
|
|
242
|
-
dataset : dataset 名(必填)。
|
|
243
|
-
split : split 名(必填)。
|
|
244
|
-
version : version(必填——CreateVersion 硬约束)。
|
|
245
|
-
run_type : 可选,train/eval(create 后打标)。
|
|
246
|
-
storage_path : 可选,缺省按 profile 自动计算(与 versions.create 同源)。
|
|
247
|
-
environment : 可选,缺省由 profile 派生(pre/online)。
|
|
248
|
-
|
|
249
|
-
返回 :class:`~instance_repo.models.DatasetSplit`(读回;pre 集合 ensure
|
|
250
|
-
的 record 或 version 内嵌投影)。
|
|
243
|
+
``version`` 可选;空值表示 unversioned split。``storage_path`` 仅为 1.2.1
|
|
244
|
+
源码兼容保留,独立端点不接受它,因此非空值会在网络调用前拒绝。
|
|
245
|
+
POST 响应是 DatasetVersion 兼容体,不能当 DatasetSplit 解析;可选 run_type
|
|
246
|
+
PATCH 完成后始终 GET 详情,并对创建后的短暂不可见做三次有界重试。
|
|
251
247
|
"""
|
|
252
|
-
|
|
253
|
-
|
|
248
|
+
dataset_name = (dataset or "").strip()
|
|
249
|
+
split_name = (split or "").strip()
|
|
250
|
+
version_name = (version or "").strip()
|
|
251
|
+
if not dataset_name:
|
|
254
252
|
raise SchemaError("dataset is required for splits.create")
|
|
255
|
-
if not
|
|
253
|
+
if not split_name:
|
|
256
254
|
raise SchemaError("split is required for splits.create")
|
|
257
|
-
if
|
|
255
|
+
if (storage_path or "").strip():
|
|
258
256
|
raise SchemaError(
|
|
259
|
-
"
|
|
260
|
-
"
|
|
257
|
+
"storage_path is deprecated for splits.create: split creation now "
|
|
258
|
+
"uses an independent endpoint that does not accept storage paths")
|
|
261
259
|
rt = (run_type or "").strip()
|
|
262
260
|
if rt and rt not in RUN_TYPE_VALUES:
|
|
263
261
|
raise SchemaError(
|
|
264
262
|
f"invalid run_type {rt!r}; must be one of "
|
|
265
263
|
f"{', '.join(RUN_TYPE_VALUES)} or None")
|
|
266
264
|
env = self._env(environment)
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
265
|
+
|
|
266
|
+
try:
|
|
267
|
+
self._post_create(dataset_name, split_name, version_name, env)
|
|
268
|
+
except Exception as exc:
|
|
269
|
+
raise TransportError(
|
|
270
|
+
f"split create POST failed before creation was confirmed: {exc}",
|
|
271
|
+
status=getattr(exc, "status", None),
|
|
272
|
+
biz_code=getattr(exc, "biz_code", None)) from exc
|
|
273
|
+
|
|
274
|
+
coordinates = _create_coordinates(
|
|
275
|
+
dataset_name, version_name, split_name, env)
|
|
270
276
|
if rt:
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
|
|
275
|
-
|
|
277
|
+
try:
|
|
278
|
+
self._update_with_environment(
|
|
279
|
+
dataset_name, split_name, version=version_name,
|
|
280
|
+
run_type=rt, status="", dataset_split_id=None,
|
|
281
|
+
environment=env)
|
|
282
|
+
except Exception as exc:
|
|
283
|
+
raise TransportError(
|
|
284
|
+
"split was created but run_type update failed; "
|
|
285
|
+
f"the split create is NOT rolled back; {coordinates}: {exc}",
|
|
286
|
+
status=getattr(exc, "status", None),
|
|
287
|
+
biz_code=getattr(exc, "biz_code", None)) from exc
|
|
288
|
+
|
|
289
|
+
def readback() -> DatasetSplit:
|
|
290
|
+
found = self._read_detail_once(
|
|
291
|
+
dataset_name, split_name, version_name, env)
|
|
292
|
+
if found is None or found.name != split_name:
|
|
293
|
+
# A successful create can be briefly invisible to the detail read.
|
|
294
|
+
# Mark only this post-write condition transient; retry_transient keeps
|
|
295
|
+
# deterministic 4xx/schema failures single-shot.
|
|
296
|
+
raise TransportError(
|
|
297
|
+
"created split is not yet visible in authoritative readback",
|
|
298
|
+
status=503)
|
|
299
|
+
return found
|
|
300
|
+
|
|
301
|
+
try:
|
|
302
|
+
return retry_transient(readback, attempts=3)
|
|
303
|
+
except Exception as exc:
|
|
304
|
+
raise WriteOutcomeUnknown(
|
|
305
|
+
"split create succeeded but authoritative readback could not be "
|
|
306
|
+
f"confirmed; {coordinates}: {exc}",
|
|
307
|
+
cause=exc) from exc
|
|
308
|
+
|
|
309
|
+
def _read_detail_once(self, dataset: str, split: str, version: str,
|
|
310
|
+
environment: str) -> DatasetSplit | None:
|
|
311
|
+
"""Read only split detail once; never retry or project legacy version data."""
|
|
312
|
+
q = (f"dataset_name={quote(dataset, safe='')}"
|
|
313
|
+
f"&split={quote(split, safe='')}")
|
|
314
|
+
if version:
|
|
315
|
+
q += f"&version={quote(version, safe='')}"
|
|
316
|
+
if environment:
|
|
317
|
+
q += f"&environment={quote(environment, safe='')}"
|
|
318
|
+
try:
|
|
319
|
+
data = self._t._request_once("GET", f"{self._BASE}/detail?{q}")
|
|
320
|
+
except TransportError as exc:
|
|
321
|
+
if exc.status == 404:
|
|
322
|
+
return None
|
|
323
|
+
raise
|
|
324
|
+
return DatasetSplit.from_dict(data) if isinstance(data, dict) and data else None
|
|
276
325
|
|
|
277
326
|
def update(self, dataset: str, split: str | None = None, *,
|
|
278
327
|
version: str | None = None, run_type: str = "",
|
|
@@ -290,19 +339,24 @@ class SplitsClient:
|
|
|
290
339
|
典型用法:锁定 split(``status="published"``)后 push(overwrite=True)
|
|
291
340
|
会被 instances 侧前置检查拒绝,解锁用 ``status="draft"``。
|
|
292
341
|
|
|
293
|
-
写端点**仅限 online**(apiserver 对显式 pre 返回 400;pre 环境 split 的
|
|
294
|
-
run_type 变更经 ``versions.update(split_run_types=...)`` 走集合路由)。
|
|
295
|
-
|
|
296
342
|
``environment`` 语义与 :meth:`get`/:meth:`list` 对齐:显式参数优先,
|
|
297
|
-
缺省由 profile
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
显式 ``"online"`` 会拼进 query,请求自文档化;为空(无 profile)时
|
|
301
|
-
不拼参,服务端默认 online(向后兼容)。
|
|
343
|
+
缺省由 profile 派生。任何非空有效环境均拼入 query,使独立 create 的
|
|
344
|
+
POST、可选 run_type PATCH 与权威 GET 始终访问同一集合;为空时沿用
|
|
345
|
+
服务端默认环境。
|
|
302
346
|
|
|
303
347
|
寻址二选一:``dataset_split_id``(优先,split-first 精确记录 id)或
|
|
304
348
|
``dataset``+``split``(+可选 ``version``)name 三元组。
|
|
305
349
|
"""
|
|
350
|
+
return self._update_with_environment(
|
|
351
|
+
dataset, split, version=version, run_type=run_type, status=status,
|
|
352
|
+
dataset_split_id=dataset_split_id,
|
|
353
|
+
environment=self._env(environment))
|
|
354
|
+
|
|
355
|
+
def _update_with_environment(self, dataset: str, split: str | None, *,
|
|
356
|
+
version: str | None, run_type: str, status: str,
|
|
357
|
+
dataset_split_id: str | None,
|
|
358
|
+
environment: str) -> DatasetSplit:
|
|
359
|
+
"""PATCH using an already-resolved environment, including frozen empty."""
|
|
306
360
|
if dataset_split_id:
|
|
307
361
|
q = f"dataset_split_id={quote(dataset_split_id, safe='')}"
|
|
308
362
|
else:
|
|
@@ -313,15 +367,8 @@ class SplitsClient:
|
|
|
313
367
|
f"&split={quote(split, safe='')}")
|
|
314
368
|
if (version or "").strip():
|
|
315
369
|
q += f"&version={quote(version, safe='')}"
|
|
316
|
-
|
|
317
|
-
|
|
318
|
-
raise SchemaError(
|
|
319
|
-
"splits.update is online-only: the split record write endpoint "
|
|
320
|
-
"(PATCH /datasets/splits/detail) only accepts the online "
|
|
321
|
-
"environment. To change the run_type of a pre split, use "
|
|
322
|
-
"versions.update(split_run_types={split: run_type}) instead.")
|
|
323
|
-
if env == "online":
|
|
324
|
-
q += "&environment=online"
|
|
370
|
+
if environment:
|
|
371
|
+
q += f"&environment={quote(environment, safe='')}"
|
|
325
372
|
rt = run_type or ""
|
|
326
373
|
if rt and rt not in RUN_TYPE_VALUES: # 空串=清除,合法
|
|
327
374
|
raise SchemaError(
|
|
@@ -347,6 +394,13 @@ class SplitsClient:
|
|
|
347
394
|
return env
|
|
348
395
|
|
|
349
396
|
|
|
397
|
+
def _create_coordinates(dataset: str, version: str, split: str,
|
|
398
|
+
environment: str) -> str:
|
|
399
|
+
"""Stable recovery coordinates for post-write partial-completion errors."""
|
|
400
|
+
return (f"dataset={dataset}, version={version or '<unversioned>'}, "
|
|
401
|
+
f"split={split}, environment={environment or '<server-default>'}")
|
|
402
|
+
|
|
403
|
+
|
|
350
404
|
def normalize_split_version_scope(version: str | None,
|
|
351
405
|
version_scope: str | None) -> str:
|
|
352
406
|
"""纯函数:供 conformance 跨语言比对 scope 缺省规则(version 空 → all)。"""
|
|
@@ -329,16 +329,21 @@ class VersionsClient:
|
|
|
329
329
|
storage_env 但配置了 cluster 时,会经 repo-config 解析出绑定的
|
|
330
330
|
storage_env 再派生(与 ingest 对称);显式 environment 与派生值冲突时
|
|
331
331
|
前置拒绝。
|
|
332
|
-
metadata_model:
|
|
333
|
-
``version_first
|
|
334
|
-
|
|
335
|
-
否则 apiserver 按 version_first 路由写入 legacy 集合。
|
|
332
|
+
metadata_model: 可选。显式非空值归一后下发(``split_first`` /
|
|
333
|
+
``version_first``);空值回退到 profile 的 ``metadata_model``
|
|
334
|
+
(再空取 SDK 默认 ``split_first``),与 ingest/list/update 一致。
|
|
336
335
|
|
|
337
336
|
run_type 不在建版本时设置(apiserver create 不收该字段):建好后用
|
|
338
337
|
``set_run_type`` / ``update`` 打标,或走 ``ingest(run_type=...)``。
|
|
339
338
|
"""
|
|
340
339
|
if not version:
|
|
341
340
|
raise SchemaError("version is required")
|
|
341
|
+
# metadata_model:显式非空值先校验(确保非法值在任何 repo-config 探测或
|
|
342
|
+
# POST 前失败);空值回退到 profile(与 ingest/list/update 一致)。
|
|
343
|
+
if metadata_model is not None and metadata_model.strip():
|
|
344
|
+
normalize_metadata_model(metadata_model) # early reject invalid
|
|
345
|
+
else:
|
|
346
|
+
metadata_model = None
|
|
342
347
|
# 与 ingest 对称(问题1 彻底修复):storage_env 未在 profile 里显式配置时,
|
|
343
348
|
# 经 repo-config 解析出与 cluster 绑定的 storage_env 再派生 environment;
|
|
344
349
|
# 同时复用**同一次**响应里的 bucket/prefix 计算 storage_path,避免出现
|
|
@@ -385,15 +390,18 @@ class VersionsClient:
|
|
|
385
390
|
# environment 必须放 query(apiserver createVersionRequest 无该字段,
|
|
386
391
|
# 只从 datasetEnvironmentFromQuery 读);放 body 会被静默丢弃导致
|
|
387
392
|
# 版本落到默认 online 而 list/get/update 走 profile 派生的 pre。
|
|
388
|
-
# metadata_model
|
|
393
|
+
# metadata_model 与 ingest/list/update 对齐:显式值优先,空值从 profile
|
|
394
|
+
# 派生(apiserver d7275ab7 起已放开 split_first Version 创建限制)。
|
|
395
|
+
mm = metadata_model
|
|
396
|
+
if mm is None and self._p is not None:
|
|
397
|
+
mm = normalize_metadata_model(
|
|
398
|
+
self._p.metadata_model if self._p else None)
|
|
399
|
+
if mm is None:
|
|
400
|
+
mm = normalize_metadata_model(None)
|
|
389
401
|
q = f"dataset_name={quote(dataset, safe='')}"
|
|
390
402
|
if env:
|
|
391
403
|
q += f"&environment={quote(env, safe='')}"
|
|
392
|
-
mm =
|
|
393
|
-
if mm is None and self._p is not None:
|
|
394
|
-
mm = normalize_metadata_model(self._p.metadata_model)
|
|
395
|
-
if mm:
|
|
396
|
-
q += f"&metadata_model={quote(mm, safe='')}"
|
|
404
|
+
q += f"&metadata_model={quote(mm, safe='')}"
|
|
397
405
|
data = self._t.post(f"{self._BASE}?{q}", body)
|
|
398
406
|
return DatasetVersion(**_pick(data))
|
|
399
407
|
|
|
@@ -462,14 +462,17 @@ class Repo:
|
|
|
462
462
|
pfx += self._clean_user_data_subpath(subpath)
|
|
463
463
|
return self._content.list_objects(pfx)
|
|
464
464
|
|
|
465
|
-
# ── Artifact 下载(
|
|
465
|
+
# ── Artifact 下载(AP HTTP 链接,由服务端执行权限校验)──
|
|
466
466
|
|
|
467
467
|
def download_artifact(self, job_id: str, cluster: str | None = None,
|
|
468
468
|
output_dir: str = ".", *,
|
|
469
|
-
ap_endpoint: str | None = None
|
|
469
|
+
ap_endpoint: str | None = None,
|
|
470
|
+
include_files: list[str] | None = None,
|
|
471
|
+
exclude_files: list[str] | None = None,
|
|
472
|
+
extract: bool = False) -> str:
|
|
470
473
|
"""下载单个 job 的 artifact (result.tgz) 到本地。
|
|
471
474
|
|
|
472
|
-
|
|
475
|
+
鉴权:由 AP artifact 接口执行,SDK 不再查询 Job 详情作预检查。
|
|
473
476
|
并发模型:单 job = 单文件单线程。
|
|
474
477
|
|
|
475
478
|
Args:
|
|
@@ -484,17 +487,21 @@ class Repo:
|
|
|
484
487
|
from .artifacts import download_artifact as _dl
|
|
485
488
|
cluster = cluster or self.profile.cluster or ""
|
|
486
489
|
return _dl(self.transport, self.transport.token, job_id,
|
|
487
|
-
cluster, output_dir, ap_endpoint
|
|
490
|
+
cluster, output_dir, ap_endpoint,
|
|
491
|
+
include_files=include_files, exclude_files=exclude_files, extract=extract)
|
|
488
492
|
|
|
489
493
|
def download_artifacts(self, job_ids: list[str],
|
|
490
494
|
cluster: str | None = None,
|
|
491
495
|
output_dir: str = ".", *,
|
|
492
496
|
max_concurrent: int = 8,
|
|
493
497
|
ap_endpoint: str | None = None,
|
|
498
|
+
include_files: list[str] | None = None,
|
|
499
|
+
exclude_files: list[str] | None = None,
|
|
500
|
+
extract: bool = False,
|
|
494
501
|
) -> dict[str, str | Exception]:
|
|
495
502
|
"""批量下载多个 job 的 artifact。
|
|
496
503
|
|
|
497
|
-
并发粒度是跨 job
|
|
504
|
+
并发粒度是跨 job(不是单文件内多线程)。下载使用 AP 返回的 HTTP 链接。
|
|
498
505
|
|
|
499
506
|
Args:
|
|
500
507
|
job_ids: AP job ID 列表
|
|
@@ -509,13 +516,17 @@ class Repo:
|
|
|
509
516
|
from .artifacts import download_artifacts as _dl_many
|
|
510
517
|
cluster = cluster or self.profile.cluster or ""
|
|
511
518
|
return _dl_many(self.transport, self.transport.token, job_ids,
|
|
512
|
-
cluster, output_dir, max_concurrent, ap_endpoint
|
|
519
|
+
cluster, output_dir, max_concurrent, ap_endpoint,
|
|
520
|
+
include_files=include_files, exclude_files=exclude_files, extract=extract)
|
|
513
521
|
|
|
514
522
|
def download_artifacts_from_group(self, group_id: str,
|
|
515
523
|
cluster: str | None = None,
|
|
516
524
|
output_dir: str = ".", *,
|
|
517
525
|
max_concurrent: int = 8,
|
|
518
526
|
ap_endpoint: str | None = None,
|
|
527
|
+
include_files: list[str] | None = None,
|
|
528
|
+
exclude_files: list[str] | None = None,
|
|
529
|
+
extract: bool = False,
|
|
519
530
|
) -> dict[str, str | Exception]:
|
|
520
531
|
"""下载某个 group 下全部 job 的 artifact。
|
|
521
532
|
|
|
@@ -539,7 +550,8 @@ class Repo:
|
|
|
539
550
|
logger.warning("group %s has no jobs", group_id)
|
|
540
551
|
return {}
|
|
541
552
|
return _dl_many(self.transport, self.transport.token, job_ids,
|
|
542
|
-
cluster, output_dir, max_concurrent, ap_endpoint
|
|
553
|
+
cluster, output_dir, max_concurrent, ap_endpoint,
|
|
554
|
+
include_files=include_files, exclude_files=exclude_files, extract=extract)
|
|
543
555
|
|
|
544
556
|
# ── 用户数据 ACL(role-bindings 接口,resource_type=user_data)──
|
|
545
557
|
_USER_DATA_ROLE_MAP = {
|
|
@@ -1,9 +1,10 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: instance-repo
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0.dev0
|
|
4
4
|
Summary: InstanceRepo SDK — 评测 Instance 统一存储客户端(SDK-only;控制面由 apiserver 承载)
|
|
5
5
|
Requires-Python: >=3.10
|
|
6
6
|
Description-Content-Type: text/markdown
|
|
7
|
+
Requires-Dist: irepo-artifact<0.2.0,>=0.1.1
|
|
7
8
|
Requires-Dist: tomli>=1.1.0; python_version < "3.11"
|
|
8
9
|
Provides-Extra: oss
|
|
9
10
|
Requires-Dist: oss2>=2.19; extra == "oss"
|
|
@@ -81,8 +82,9 @@ pip install "instance-repo[oss]" # 需要数据面 OSS 读写时,附带 os
|
|
|
81
82
|
不是 default)。旧数据集显式传 `Repo(metadata_model="version_first")`,其对象布局与镜像 tag
|
|
82
83
|
**逐字节不变**,存量数据无需迁移。
|
|
83
84
|
|
|
84
|
-
服务端把空 `metadata_model` 归一为 `version_first
|
|
85
|
-
|
|
85
|
+
服务端把空 `metadata_model` 归一为 `version_first`。数据面与读路径仍按 profile 的模型工作;
|
|
86
|
+
但 `repo.versions.create(...)` 的 `metadata_model` 仅在调用方显式传入非空值时下发,不从
|
|
87
|
+
profile 自动注入。显式 `split_first` 仍由服务端拒绝;创建 split 应使用 `repo.splits.create(...)`。
|
|
86
88
|
|
|
87
89
|
| | `split_first`(默认) | `version_first`(旧) |
|
|
88
90
|
| --- | --- | --- |
|
|
@@ -96,6 +98,29 @@ pip install "instance-repo[oss]" # 需要数据面 OSS 读写时,附带 os
|
|
|
96
98
|
92005 ambiguous。显式给 splits 时服务端只列举 `{split}/` 与 `{split}-assets/`,推断被完全绕过。
|
|
97
99
|
确需推断时传 `allow_layout_inference=True`。
|
|
98
100
|
|
|
101
|
+
## Split 创建
|
|
102
|
+
|
|
103
|
+
`repo.splits.create(dataset, split, *, version=None, run_type=None, storage_path=None,
|
|
104
|
+
environment=None)` 直接调用 `POST /apis/v1/datasets/splits`,body 固定为
|
|
105
|
+
`{"splits":["<split>"]}`。`version` 可选;省略即创建 unversioned split,query 中不会发送
|
|
106
|
+
`version=`。存储路径由服务端根据 dataset/version/split 推导。
|
|
107
|
+
|
|
108
|
+
`storage_path` 只为 1.2.1 调用兼容保留(计划 1.3.0 删除):空值允许,非空值在网络请求前抛
|
|
109
|
+
`SchemaError`。提供 `run_type` 时,创建后通过 `PATCH /apis/v1/datasets/splits/detail` 更新,
|
|
110
|
+
不会调用 Version PATCH;最后 GET 读回权威 Split 记录。
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
repo.splits.create("alibaba/mybench", "test")
|
|
114
|
+
repo.splits.create("alibaba/mybench", "test", version="v1", run_type="eval")
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
对应 CLI 的 `--version` 也是可选项;`--storage-path` 仅兼容旧脚本,非空值会失败:
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
irepo split create alibaba/mybench/test --run-type eval
|
|
121
|
+
irepo split create alibaba/mybench/test --version v1
|
|
122
|
+
```
|
|
123
|
+
|
|
99
124
|
## 独立镜像能力(`repo.images`,0.8 新增)
|
|
100
125
|
|
|
101
126
|
```python
|
|
@@ -152,7 +177,8 @@ irepo --help
|
|
|
152
177
|
# 20 个子命令:validate / push / push-many / deliver / pull / list / get
|
|
153
178
|
# publish / grant / feedback / report / whoami / user-data
|
|
154
179
|
# repo-config / create / claim / version / image
|
|
155
|
-
# split(list/get/set-run-type
|
|
180
|
+
# split(list/get/create/set-run-type;create 的 --version 可选,--storage-path 非空即拒绝)
|
|
181
|
+
# dataset(list/get/update/access/revoke/benchmarks/workflows)
|
|
156
182
|
irepo image ref --dataset alibaba/mybench --instance-id inst-1 --split test
|
|
157
183
|
irepo image pull '<acr>/<ns>/mybench:test-inst-1' ./out.tar
|
|
158
184
|
```
|
|
@@ -175,3 +201,36 @@ irepo publish alibaba/mybench/v1 --split test \
|
|
|
175
201
|
版本号在 `pyproject.toml` / `instance_repo.__version__` / Go `version.go` /
|
|
176
202
|
conformance golden 四处保持一致。各环境的真实配置取值、端到端示例与错误码对照,见团队内部
|
|
177
203
|
部署文档。
|
|
204
|
+
|
|
205
|
+
## 产物下载改造(开发版)
|
|
206
|
+
|
|
207
|
+
产物下载现在委托独立 `irepo-artifact` 包,通过 AP 签发的 HTTP 链接下载,
|
|
208
|
+
不再要求 OSS SDK、STS 或 artifact bucket/prefix 配置;其他存储功能的 OSS 依赖不变。
|
|
209
|
+
本地开发需同时安装两个包:
|
|
210
|
+
|
|
211
|
+
```bash
|
|
212
|
+
python -m pip install -e ./sdk/python-artifact -e ./sdk/python
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
保留 `irepo artifact download`、`download-many`、`download-group`。三个命令均支持
|
|
216
|
+
`--include-files`、`--exclude-files`(可重复、可逗号分隔)和 `--extract`。
|
|
217
|
+
|
|
218
|
+
```bash
|
|
219
|
+
irepo artifact download --job-id job-a --output-dir out/job-a
|
|
220
|
+
irepo artifact download-many --job-ids job-a,job-b --output-dir out --extract
|
|
221
|
+
irepo artifact download-group --group-id group-a --output-dir out \
|
|
222
|
+
--include-files logs,result.json --exclude-files debug.log,tmp
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
默认不解压;`--extract` 与文件筛选冲突。筛选通过 manifest 得到最终文件列表,再向 AP 获取
|
|
226
|
+
各文件 URL。不开启筛选时优先 OSS,网络不可达或中断时回退 dataplane。
|
|
227
|
+
|
|
228
|
+
**路径迁移:** 单 Job 的输出目录就是 Job 目录,包为 `<output_dir>/result.tgz`;
|
|
229
|
+
批量/Group 为 `<output_dir>/jobs/<job_id>/result.tgz`。解压或筛选内容位于 Job 目录的
|
|
230
|
+
`artifacts/output/`。旧 `<job_id>.tgz` 文件名统一为 `result.tgz`,旧脚本应使用 SDK 返回路径。
|
|
231
|
+
旧 SDK 单 Job 仍返回字符串,批量仍返回路径或异常字典;旧入口维持覆盖意图,新独立客户端
|
|
232
|
+
默认拒绝覆盖。请勿在旧目录尚有需保留文件时直接覆盖迁移。
|
|
233
|
+
|
|
234
|
+
权限由 AP artifact 接口统一校验,SDK 不再执行 Job 详情预检查。
|
|
235
|
+
更多行为与安装说明见 [独立包 README](../python-artifact/README.md)。
|
|
236
|
+
独立包尚为开发版本,正式发布完整 SDK 前必须先发布可安装的独立包版本。
|
|
@@ -4,12 +4,13 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "instance-repo"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.3.0.dev0"
|
|
8
8
|
description = "InstanceRepo SDK — 评测 Instance 统一存储客户端(SDK-only;控制面由 apiserver 承载)"
|
|
9
9
|
requires-python = ">=3.10"
|
|
10
10
|
readme = "README.md"
|
|
11
11
|
# tomllib 是 3.11 内置;3.10 上回退到 tomli(见 validate.py),故按 Python 版本条件依赖。
|
|
12
12
|
dependencies = [
|
|
13
|
+
"irepo-artifact>=0.1.1,<0.2.0",
|
|
13
14
|
"tomli>=1.1.0; python_version < '3.11'",
|
|
14
15
|
]
|
|
15
16
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{instance_repo-1.1.4 → instance_repo-1.3.0.dev0}/instance_repo.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|