pyattacker 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyattacker/__init__.py +198 -0
- pyattacker/__main__.py +12 -0
- pyattacker/algorithm.py +355 -0
- pyattacker/artifact.py +255 -0
- pyattacker/backends.py +228 -0
- pyattacker/cli.py +682 -0
- pyattacker/declarative.py +327 -0
- pyattacker/errors.py +223 -0
- pyattacker/export.py +291 -0
- pyattacker/merge.py +177 -0
- pyattacker/monitor.py +108 -0
- pyattacker/pipeline.py +241 -0
- pyattacker/plugins.py +244 -0
- pyattacker/resource.py +1018 -0
- pyattacker/runner.py +1406 -0
- pyattacker/scheduler.py +135 -0
- pyattacker/server.py +297 -0
- pyattacker/shard.py +107 -0
- pyattacker/store/__init__.py +32 -0
- pyattacker/store/base.py +351 -0
- pyattacker/store/memory.py +368 -0
- pyattacker/store/sqlite.py +734 -0
- pyattacker/store/writebehind.py +262 -0
- pyattacker/task.py +508 -0
- pyattacker/tasks/__init__.py +380 -0
- pyattacker-0.1.0.dist-info/METADATA +347 -0
- pyattacker-0.1.0.dist-info/RECORD +30 -0
- pyattacker-0.1.0.dist-info/WHEEL +4 -0
- pyattacker-0.1.0.dist-info/entry_points.txt +2 -0
- pyattacker-0.1.0.dist-info/licenses/LICENSE +21 -0
pyattacker/artifact.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""Artifact —— the persistent carrier of a task's state.
|
|
2
|
+
|
|
3
|
+
Design notes:
|
|
4
|
+
* Artifacts are **content-addressed** (blake2b digest), which gives deduplication and integrity checking for free.
|
|
5
|
+
* An artifact is persisted as soon as it is produced —— that is the task-level checkpoint and the entire source of resume capability.
|
|
6
|
+
* Encoding goes through a pluggable codec, JSON by default (supports dataclass / primitives / Enum / bytes).
|
|
7
|
+
* An artifact with ``seq == SEED_SEQ`` is the seed input of the whole pipeline (one row of the dataset).
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import base64
|
|
13
|
+
import dataclasses
|
|
14
|
+
import hashlib
|
|
15
|
+
import json
|
|
16
|
+
from collections.abc import Iterable
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from enum import Enum
|
|
19
|
+
from typing import Any, Protocol, runtime_checkable
|
|
20
|
+
|
|
21
|
+
from .errors import ArtifactCodecError
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"SEED_SEQ",
|
|
25
|
+
"SEED_TASK",
|
|
26
|
+
"Encoded",
|
|
27
|
+
"Artifact",
|
|
28
|
+
"Codec",
|
|
29
|
+
"JsonCodec",
|
|
30
|
+
"BytesCodec",
|
|
31
|
+
"CodecRegistry",
|
|
32
|
+
"DEFAULT_REGISTRY",
|
|
33
|
+
"canonical_json",
|
|
34
|
+
"digest_of",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
SEED_SEQ = -1
|
|
38
|
+
SEED_TASK = "__seed__"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _json_default(obj: Any) -> Any:
|
|
42
|
+
if dataclasses.is_dataclass(obj) and not isinstance(obj, type):
|
|
43
|
+
return dataclasses.asdict(obj)
|
|
44
|
+
if isinstance(obj, Enum):
|
|
45
|
+
return obj.value
|
|
46
|
+
if isinstance(obj, (set, frozenset)):
|
|
47
|
+
return sorted(obj, key=repr)
|
|
48
|
+
if isinstance(obj, (bytes, bytearray, memoryview)):
|
|
49
|
+
return {"__bytes__": base64.b64encode(bytes(obj)).decode("ascii")}
|
|
50
|
+
to_json = getattr(obj, "__json__", None)
|
|
51
|
+
if callable(to_json):
|
|
52
|
+
return to_json()
|
|
53
|
+
raise ArtifactCodecError(
|
|
54
|
+
f"artifact payload is not JSON-encodable: {type(obj).__name__}; "
|
|
55
|
+
f"register a codec for it (CodecRegistry.register)"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def canonical_json(obj: Any) -> str:
|
|
60
|
+
"""Stable serialization: sorted keys, no extra whitespace. Used for digest computation, not for display."""
|
|
61
|
+
return json.dumps(
|
|
62
|
+
obj, sort_keys=True, ensure_ascii=False, separators=(",", ":"), default=_json_default
|
|
63
|
+
)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def digest_of(data: bytes | str) -> str:
|
|
67
|
+
if isinstance(data, str):
|
|
68
|
+
data = data.encode("utf-8")
|
|
69
|
+
return hashlib.blake2b(data, digest_size=16).hexdigest()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True, slots=True)
|
|
73
|
+
class Encoded:
|
|
74
|
+
"""The result of one encoding pass. ``data`` is always bytes, ready to be written straight into a BLOB."""
|
|
75
|
+
|
|
76
|
+
type_name: str
|
|
77
|
+
codec: str
|
|
78
|
+
data: bytes
|
|
79
|
+
digest: str
|
|
80
|
+
size: int
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@runtime_checkable
|
|
84
|
+
class Codec(Protocol):
|
|
85
|
+
name: str
|
|
86
|
+
|
|
87
|
+
def can_encode(self, obj: Any) -> bool: ...
|
|
88
|
+
|
|
89
|
+
def dumps(self, obj: Any) -> bytes: ...
|
|
90
|
+
|
|
91
|
+
def loads(self, data: bytes) -> Any: ...
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class JsonCodec:
|
|
95
|
+
name = "json"
|
|
96
|
+
|
|
97
|
+
def can_encode(self, obj: Any) -> bool:
|
|
98
|
+
try:
|
|
99
|
+
canonical_json(obj)
|
|
100
|
+
except ArtifactCodecError:
|
|
101
|
+
return False
|
|
102
|
+
return True
|
|
103
|
+
|
|
104
|
+
def dumps(self, obj: Any) -> bytes:
|
|
105
|
+
return canonical_json(obj).encode("utf-8")
|
|
106
|
+
|
|
107
|
+
def loads(self, data: bytes) -> Any:
|
|
108
|
+
return json.loads(data.decode("utf-8"))
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
class BytesCodec:
|
|
112
|
+
"""Raw byte payloads (images, audio, log chunks, ...)."""
|
|
113
|
+
|
|
114
|
+
name = "bytes"
|
|
115
|
+
|
|
116
|
+
def can_encode(self, obj: Any) -> bool:
|
|
117
|
+
return isinstance(obj, (bytes, bytearray, memoryview))
|
|
118
|
+
|
|
119
|
+
def dumps(self, obj: Any) -> bytes:
|
|
120
|
+
return bytes(obj)
|
|
121
|
+
|
|
122
|
+
def loads(self, data: bytes) -> Any:
|
|
123
|
+
return data
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class CodecRegistry:
|
|
127
|
+
"""Registry mapping types → codecs; also responsible for restoring JSON payloads back into user types."""
|
|
128
|
+
|
|
129
|
+
def __init__(self) -> None:
|
|
130
|
+
self._codecs: dict[str, Codec] = {"json": JsonCodec(), "bytes": BytesCodec()}
|
|
131
|
+
self._by_type: dict[type, str] = {
|
|
132
|
+
bytes: "bytes",
|
|
133
|
+
bytearray: "bytes",
|
|
134
|
+
memoryview: "bytes",
|
|
135
|
+
}
|
|
136
|
+
self._rebuild: dict[str, type] = {}
|
|
137
|
+
|
|
138
|
+
def register(self, codec: Codec, *, for_types: Iterable[type] = (), name: str | None = None) -> None:
|
|
139
|
+
"""Register a codec. **Later registrations take precedence over earlier ones.**
|
|
140
|
+
|
|
141
|
+
``codec_for`` falls back to asking each registered codec in turn, and the built-in JSON
|
|
142
|
+
codec accepts almost anything, so insertion order decides whether a specialised codec ever
|
|
143
|
+
gets a chance. Moving each new registration to the front makes the rule explicit: if you
|
|
144
|
+
register a codec that claims a payload, it wins. An explicit ``for_types`` mapping still
|
|
145
|
+
beats the scan, because naming the type is the strongest statement of intent.
|
|
146
|
+
"""
|
|
147
|
+
codec_name = name or codec.name
|
|
148
|
+
self._codecs = {codec_name: codec, **{k: v for k, v in self._codecs.items() if k != codec_name}}
|
|
149
|
+
for tp in for_types:
|
|
150
|
+
self._by_type[tp] = codec_name
|
|
151
|
+
|
|
152
|
+
def register_type(self, cls: type) -> type:
|
|
153
|
+
"""Register a (dataclass) type; on restore the object is rebuilt as ``cls(**payload)``."""
|
|
154
|
+
self._rebuild[cls.__name__] = cls
|
|
155
|
+
return cls
|
|
156
|
+
|
|
157
|
+
def type_name_of(self, obj: Any) -> str:
|
|
158
|
+
return type(obj).__name__
|
|
159
|
+
|
|
160
|
+
def codec_for(self, obj: Any) -> Codec:
|
|
161
|
+
codec_name = self._by_type.get(type(obj))
|
|
162
|
+
if codec_name is not None:
|
|
163
|
+
return self._codecs[codec_name]
|
|
164
|
+
for base in type(obj).__mro__[1:]:
|
|
165
|
+
codec_name = self._by_type.get(base)
|
|
166
|
+
if codec_name is not None:
|
|
167
|
+
return self._codecs[codec_name]
|
|
168
|
+
for codec in self._codecs.values():
|
|
169
|
+
if codec.can_encode(obj):
|
|
170
|
+
return codec
|
|
171
|
+
raise ArtifactCodecError(f"no codec available for {type(obj).__name__}")
|
|
172
|
+
|
|
173
|
+
def dump(self, obj: Any) -> Encoded:
|
|
174
|
+
codec = self.codec_for(obj)
|
|
175
|
+
data = codec.dumps(obj)
|
|
176
|
+
return Encoded(
|
|
177
|
+
type_name=self.type_name_of(obj),
|
|
178
|
+
codec=codec.name,
|
|
179
|
+
data=data,
|
|
180
|
+
digest=digest_of(data),
|
|
181
|
+
size=len(data),
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
def load_raw(self, encoded: Encoded) -> Any:
|
|
185
|
+
codec = self._codecs.get(encoded.codec)
|
|
186
|
+
if codec is None:
|
|
187
|
+
raise ArtifactCodecError(f"unknown codec: {encoded.codec}")
|
|
188
|
+
return codec.loads(encoded.data)
|
|
189
|
+
|
|
190
|
+
def load(self, encoded: Encoded) -> Any:
|
|
191
|
+
"""Decode + best-effort restore of the user type (a registered dataclass)."""
|
|
192
|
+
value = self.load_raw(encoded)
|
|
193
|
+
cls = self._rebuild.get(encoded.type_name)
|
|
194
|
+
if cls is None:
|
|
195
|
+
return value
|
|
196
|
+
if dataclasses.is_dataclass(cls) and isinstance(value, dict):
|
|
197
|
+
try:
|
|
198
|
+
return cls(**value)
|
|
199
|
+
except TypeError as exc: # on field mismatch fall back to the raw dict, staying diagnosable
|
|
200
|
+
raise ArtifactCodecError(
|
|
201
|
+
f"cannot restore artifact as {cls.__name__}: {exc}"
|
|
202
|
+
) from exc
|
|
203
|
+
if isinstance(cls, type) and not isinstance(value, cls):
|
|
204
|
+
try:
|
|
205
|
+
return cls(value)
|
|
206
|
+
except Exception: # pragma: no cover - best effort
|
|
207
|
+
return value
|
|
208
|
+
return value
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
DEFAULT_REGISTRY = CodecRegistry()
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
@dataclass(frozen=True, slots=True)
|
|
215
|
+
class Artifact:
|
|
216
|
+
"""The persisted state produced by a task."""
|
|
217
|
+
|
|
218
|
+
id: str
|
|
219
|
+
pipeline_id: str
|
|
220
|
+
task_name: str
|
|
221
|
+
seq: int
|
|
222
|
+
type_name: str
|
|
223
|
+
codec: str
|
|
224
|
+
digest: str
|
|
225
|
+
size: int
|
|
226
|
+
payload: bytes | None
|
|
227
|
+
created_at: float
|
|
228
|
+
is_final: bool = False
|
|
229
|
+
blob_ref: str | None = None
|
|
230
|
+
meta: dict[str, Any] = field(default_factory=dict)
|
|
231
|
+
|
|
232
|
+
@property
|
|
233
|
+
def available(self) -> bool:
|
|
234
|
+
"""Whether the bytes are reachable right now.
|
|
235
|
+
|
|
236
|
+
Stores hydrate ``payload`` from the artifact backend on read, so this is False only when
|
|
237
|
+
the payload was deliberately dropped (``journal=summary``, a ``null`` backend) or is
|
|
238
|
+
genuinely missing — which is exactly the signal resume needs to rerun the pipeline.
|
|
239
|
+
"""
|
|
240
|
+
return self.payload is not None
|
|
241
|
+
|
|
242
|
+
def encoded(self) -> Encoded:
|
|
243
|
+
if self.payload is None:
|
|
244
|
+
raise ArtifactCodecError(f"artifact {self.id} has no stored payload (a limitation of journal mode)")
|
|
245
|
+
return Encoded(
|
|
246
|
+
type_name=self.type_name,
|
|
247
|
+
codec=self.codec,
|
|
248
|
+
data=self.payload,
|
|
249
|
+
digest=self.digest,
|
|
250
|
+
size=self.size,
|
|
251
|
+
)
|
|
252
|
+
|
|
253
|
+
@staticmethod
|
|
254
|
+
def build_id(pipeline_id: str, seq: int) -> str:
|
|
255
|
+
return f"{pipeline_id}:{seq}"
|
pyattacker/backends.py
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""Artifact backends: where payload bytes live.
|
|
2
|
+
|
|
3
|
+
By default a payload lives inline in the store (a SQLite BLOB), which is right for the common
|
|
4
|
+
case — a model response is a few kilobytes. It stops being right when an artifact is a rendered
|
|
5
|
+
image, an audio clip or a multi-megabyte transcript: the database doubles in size, every backup
|
|
6
|
+
copies it, and `export` has to materialize it.
|
|
7
|
+
|
|
8
|
+
A backend answers one question — *should these bytes go somewhere else?* — and then owns them:
|
|
9
|
+
|
|
10
|
+
* :class:`InlineBackend` (default): never spills, behaviour is exactly as before;
|
|
11
|
+
* :class:`FileBackend`: content-addressed files under a root directory, written atomically;
|
|
12
|
+
* :class:`NullBackend`: keeps the digest and drops the bytes (hash-only journals).
|
|
13
|
+
|
|
14
|
+
The contract with the store is deliberately small: the store still records `digest`/`size`/`codec`
|
|
15
|
+
in its own row, plus an opaque `blob_ref`. On read the store fills `payload` back in from the
|
|
16
|
+
backend, so an artifact stays a single object to everything upstream.
|
|
17
|
+
|
|
18
|
+
**Content addressing is what makes this safe**: the file name is the digest, so identical payloads
|
|
19
|
+
collapse into one file, a partially written file can never be mistaken for a complete one, and a
|
|
20
|
+
backend can be shared by many runs and many stores.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import contextlib
|
|
26
|
+
import json
|
|
27
|
+
import os
|
|
28
|
+
import tempfile
|
|
29
|
+
from dataclasses import dataclass
|
|
30
|
+
from typing import Any, Protocol, runtime_checkable
|
|
31
|
+
|
|
32
|
+
from .artifact import Artifact
|
|
33
|
+
from .errors import ConfigError, PyAttackerError
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"ArtifactBackend",
|
|
37
|
+
"InlineBackend",
|
|
38
|
+
"FileBackend",
|
|
39
|
+
"NullBackend",
|
|
40
|
+
"resolve_backend",
|
|
41
|
+
"BACKENDS",
|
|
42
|
+
]
|
|
43
|
+
|
|
44
|
+
DEFAULT_MIN_BYTES = 256 * 1024
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@runtime_checkable
|
|
48
|
+
class ArtifactBackend(Protocol):
|
|
49
|
+
"""Where artifact payloads are kept."""
|
|
50
|
+
|
|
51
|
+
name: str
|
|
52
|
+
|
|
53
|
+
def wants(self, artifact: Artifact) -> bool:
|
|
54
|
+
"""Should this artifact's payload be stored outside the database?"""
|
|
55
|
+
...
|
|
56
|
+
|
|
57
|
+
def put(self, artifact: Artifact) -> str:
|
|
58
|
+
"""Store the payload, returning an opaque reference the store will keep."""
|
|
59
|
+
...
|
|
60
|
+
|
|
61
|
+
def get(self, ref: str) -> bytes | None:
|
|
62
|
+
"""Fetch a payload by reference (``None`` when it is gone)."""
|
|
63
|
+
...
|
|
64
|
+
|
|
65
|
+
def delete(self, ref: str) -> bool:
|
|
66
|
+
"""Best-effort removal. Never called by the kernel; provided for operators."""
|
|
67
|
+
...
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@dataclass
|
|
71
|
+
class InlineBackend:
|
|
72
|
+
"""Keep everything in the store. The default, and the previous behaviour exactly."""
|
|
73
|
+
|
|
74
|
+
name: str = "inline"
|
|
75
|
+
|
|
76
|
+
def wants(self, artifact: Artifact) -> bool:
|
|
77
|
+
return False
|
|
78
|
+
|
|
79
|
+
def put(self, artifact: Artifact) -> str: # pragma: no cover - never called
|
|
80
|
+
raise PyAttackerError("InlineBackend never spills payloads")
|
|
81
|
+
|
|
82
|
+
def get(self, ref: str) -> bytes | None:
|
|
83
|
+
return None
|
|
84
|
+
|
|
85
|
+
def delete(self, ref: str) -> bool:
|
|
86
|
+
return False
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
@dataclass
|
|
90
|
+
class FileBackend:
|
|
91
|
+
"""Content-addressed files under ``root``: ``<root>/ab/cdef…``.
|
|
92
|
+
|
|
93
|
+
Writes go to a temporary file in the same directory and are then ``os.replace``-d into place,
|
|
94
|
+
so a crash mid-write cannot leave a truncated blob that looks valid.
|
|
95
|
+
"""
|
|
96
|
+
|
|
97
|
+
root: str
|
|
98
|
+
min_bytes: int = DEFAULT_MIN_BYTES
|
|
99
|
+
name: str = "file"
|
|
100
|
+
|
|
101
|
+
def __post_init__(self) -> None:
|
|
102
|
+
self.root = os.path.abspath(os.path.expanduser(str(self.root)))
|
|
103
|
+
os.makedirs(self.root, exist_ok=True)
|
|
104
|
+
|
|
105
|
+
def wants(self, artifact: Artifact) -> bool:
|
|
106
|
+
return artifact.payload is not None and artifact.size >= self.min_bytes
|
|
107
|
+
|
|
108
|
+
def path_for(self, digest: str) -> str:
|
|
109
|
+
return os.path.join(self.root, digest[:2], digest[2:])
|
|
110
|
+
|
|
111
|
+
def put(self, artifact: Artifact) -> str:
|
|
112
|
+
if artifact.payload is None:
|
|
113
|
+
raise PyAttackerError(f"artifact {artifact.id} has no payload to spill")
|
|
114
|
+
target = self.path_for(artifact.digest)
|
|
115
|
+
if not os.path.exists(target):
|
|
116
|
+
os.makedirs(os.path.dirname(target), exist_ok=True)
|
|
117
|
+
# Not a context manager on purpose: the file must outlive the block so it can be
|
|
118
|
+
# fsync-ed and then atomically renamed into place.
|
|
119
|
+
handle = tempfile.NamedTemporaryFile( # noqa: SIM115
|
|
120
|
+
dir=os.path.dirname(target), prefix=".tmp-", delete=False
|
|
121
|
+
)
|
|
122
|
+
try:
|
|
123
|
+
handle.write(artifact.payload)
|
|
124
|
+
handle.flush()
|
|
125
|
+
os.fsync(handle.fileno())
|
|
126
|
+
handle.close()
|
|
127
|
+
os.replace(handle.name, target)
|
|
128
|
+
except BaseException:
|
|
129
|
+
handle.close()
|
|
130
|
+
with contextlib.suppress(OSError): # pragma: no cover - best effort
|
|
131
|
+
os.unlink(handle.name)
|
|
132
|
+
raise
|
|
133
|
+
return f"file://{target}"
|
|
134
|
+
|
|
135
|
+
def resolve(self, ref: str) -> str:
|
|
136
|
+
"""Turn a reference into a path: ``file:///abs/path`` or a path relative to ``root``."""
|
|
137
|
+
path = ref[len("file://") :] if ref.startswith("file://") else ref
|
|
138
|
+
if not os.path.isabs(path):
|
|
139
|
+
path = os.path.join(self.root, path)
|
|
140
|
+
return path
|
|
141
|
+
|
|
142
|
+
def get(self, ref: str) -> bytes | None:
|
|
143
|
+
try:
|
|
144
|
+
with open(self.resolve(ref), "rb") as handle:
|
|
145
|
+
return handle.read()
|
|
146
|
+
except FileNotFoundError:
|
|
147
|
+
return None
|
|
148
|
+
|
|
149
|
+
def delete(self, ref: str) -> bool:
|
|
150
|
+
try:
|
|
151
|
+
os.unlink(self.resolve(ref))
|
|
152
|
+
return True
|
|
153
|
+
except OSError:
|
|
154
|
+
return False
|
|
155
|
+
|
|
156
|
+
def stats(self) -> dict[str, Any]:
|
|
157
|
+
"""How much the backend holds (for operators; walks the tree, so not for hot paths)."""
|
|
158
|
+
files = 0
|
|
159
|
+
total = 0
|
|
160
|
+
for dirpath, _, names in os.walk(self.root):
|
|
161
|
+
for name in names:
|
|
162
|
+
if name.startswith(".tmp-"):
|
|
163
|
+
continue
|
|
164
|
+
files += 1
|
|
165
|
+
try:
|
|
166
|
+
total += os.path.getsize(os.path.join(dirpath, name))
|
|
167
|
+
except OSError: # pragma: no cover
|
|
168
|
+
continue
|
|
169
|
+
return {"root": self.root, "files": files, "bytes": total}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
@dataclass
|
|
173
|
+
class NullBackend:
|
|
174
|
+
"""Keep the digest, drop the bytes: useful when you only care *that* an artifact existed."""
|
|
175
|
+
|
|
176
|
+
name: str = "null"
|
|
177
|
+
min_bytes: int = 0
|
|
178
|
+
_seen: int = 0
|
|
179
|
+
|
|
180
|
+
def wants(self, artifact: Artifact) -> bool:
|
|
181
|
+
return artifact.payload is not None
|
|
182
|
+
|
|
183
|
+
def put(self, artifact: Artifact) -> str:
|
|
184
|
+
self._seen += 1
|
|
185
|
+
return f"null:{artifact.digest}:{artifact.size}"
|
|
186
|
+
|
|
187
|
+
def get(self, ref: str) -> bytes | None:
|
|
188
|
+
return None
|
|
189
|
+
|
|
190
|
+
def delete(self, ref: str) -> bool:
|
|
191
|
+
return True
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
BACKENDS: dict[str, Any] = {"inline": InlineBackend, "file": FileBackend, "null": NullBackend}
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def resolve_backend(spec: Any) -> ArtifactBackend:
|
|
198
|
+
"""``None``/``"inline"``/``"file:///data/blobs"``/``{"kind": "file", "root": ...}``/an instance."""
|
|
199
|
+
if spec is None:
|
|
200
|
+
return InlineBackend()
|
|
201
|
+
if isinstance(spec, ArtifactBackend):
|
|
202
|
+
return spec
|
|
203
|
+
if isinstance(spec, str):
|
|
204
|
+
raw = spec.strip()
|
|
205
|
+
if raw in ("", "inline", "none"):
|
|
206
|
+
return InlineBackend()
|
|
207
|
+
if raw == "null":
|
|
208
|
+
return NullBackend()
|
|
209
|
+
if raw.startswith("file://"):
|
|
210
|
+
return FileBackend(root=raw[len("file://") :])
|
|
211
|
+
if raw.startswith("{"):
|
|
212
|
+
try:
|
|
213
|
+
return resolve_backend(json.loads(raw))
|
|
214
|
+
except json.JSONDecodeError as exc:
|
|
215
|
+
raise ConfigError(f"artifact backend is not valid JSON: {exc}") from exc
|
|
216
|
+
# a bare path is treated as a file backend root, which is what people mean by "--blobs /data"
|
|
217
|
+
return FileBackend(root=raw)
|
|
218
|
+
if isinstance(spec, dict):
|
|
219
|
+
params = dict(spec)
|
|
220
|
+
kind = str(params.pop("kind", params.pop("name", "file")))
|
|
221
|
+
factory = BACKENDS.get(kind)
|
|
222
|
+
if factory is None:
|
|
223
|
+
raise ConfigError(f"unknown artifact backend {kind!r}; available: {sorted(BACKENDS)}")
|
|
224
|
+
try:
|
|
225
|
+
return factory(**params)
|
|
226
|
+
except TypeError as exc:
|
|
227
|
+
raise ConfigError(f"cannot configure {kind!r} backend: {exc}") from exc
|
|
228
|
+
raise ConfigError(f"cannot resolve artifact backend from {spec!r}")
|