pipelex-api 0.71.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pipelex_api/__init__.py +0 -0
- pipelex_api/api.toml +21 -0
- pipelex_api/api_config.py +144 -0
- pipelex_api/bundle.py +243 -0
- pipelex_api/disclosure.py +43 -0
- pipelex_api/error_types.py +80 -0
- pipelex_api/error_uri.py +45 -0
- pipelex_api/errors.py +130 -0
- pipelex_api/exception_handlers.py +693 -0
- pipelex_api/json_body.py +182 -0
- pipelex_api/limits.py +67 -0
- pipelex_api/main.py +221 -0
- pipelex_api/method_cache.py +241 -0
- pipelex_api/method_source.py +215 -0
- pipelex_api/middleware.py +209 -0
- pipelex_api/openapi_responses.py +186 -0
- pipelex_api/openapi_schema.py +83 -0
- pipelex_api/problem_document.py +134 -0
- pipelex_api/py.typed +0 -0
- pipelex_api/routes/__init__.py +23 -0
- pipelex_api/routes/health.py +24 -0
- pipelex_api/routes/pipelex/__init__.py +21 -0
- pipelex_api/routes/pipelex/agent/__init__.py +11 -0
- pipelex_api/routes/pipelex/agent/concept.py +60 -0
- pipelex_api/routes/pipelex/agent/models.py +49 -0
- pipelex_api/routes/pipelex/agent/pipe_spec.py +59 -0
- pipelex_api/routes/pipelex/build/__init__.py +11 -0
- pipelex_api/routes/pipelex/build/inputs.py +192 -0
- pipelex_api/routes/pipelex/build/output.py +163 -0
- pipelex_api/routes/pipelex/build/runner.py +236 -0
- pipelex_api/routes/pipelex/codegen.py +164 -0
- pipelex_api/routes/pipelex/crate_ops.py +331 -0
- pipelex_api/routes/pipelex/pipe_io.py +186 -0
- pipelex_api/routes/pipelex/pipeline.py +938 -0
- pipelex_api/routes/pipelex/resolve.py +81 -0
- pipelex_api/routes/pipelex/tools.py +111 -0
- pipelex_api/routes/pipelex/utils.py +6 -0
- pipelex_api/routes/pipelex/validate.py +473 -0
- pipelex_api/routes/version.py +51 -0
- pipelex_api/schemas/__init__.py +0 -0
- pipelex_api/schemas/models.py +653 -0
- pipelex_api/security.py +284 -0
- pipelex_api-0.71.0.dist-info/METADATA +188 -0
- pipelex_api-0.71.0.dist-info/RECORD +46 -0
- pipelex_api-0.71.0.dist-info/WHEEL +4 -0
- pipelex_api-0.71.0.dist-info/licenses/LICENSE +95 -0
|
@@ -0,0 +1,241 @@
|
|
|
1
|
+
"""Per-instance on-disk clone cache for fetched method packages, keyed by resolved commit SHA.
|
|
2
|
+
|
|
3
|
+
A `method_ref` run fetches a public git repository. Cloning on every request would hammer
|
|
4
|
+
GitHub's anonymous-clone rate limits and add seconds of latency, so this server keeps the
|
|
5
|
+
clones on local disk — but **never keyed by tag**: tags can move, and a moved tag must
|
|
6
|
+
change what runs. Instead, each request first resolves the reference to its commit SHA with
|
|
7
|
+
a cheap `git ls-remote` (no clone), then looks the clone up by that SHA. A repointed tag or
|
|
8
|
+
an advanced default branch resolves to a new SHA and therefore a fresh clone; the cached
|
|
9
|
+
copy of the old SHA ages out on its own.
|
|
10
|
+
|
|
11
|
+
The cache is per server instance (a directory under the system temp dir by default,
|
|
12
|
+
`METHOD_CACHE_DIR` to override) and bounded three ways — clone count, total bytes, and age —
|
|
13
|
+
via the `MAX_METHOD_CACHE_*` knobs in `pipelex_api.limits`. Eviction runs on insert, oldest-first by
|
|
14
|
+
last use, and never evicts the most recently used entry. Package location, bounds, and the
|
|
15
|
+
structures scan are NOT cached: they re-run against the cached clone on every request, so a
|
|
16
|
+
deployment-mode change (or a ceiling change) applies immediately.
|
|
17
|
+
|
|
18
|
+
Concurrency: clones land in a per-request staging directory and are installed with an atomic
|
|
19
|
+
`rename`, so two concurrent requests for the same SHA cannot corrupt each other — the loser
|
|
20
|
+
of the rename race discards its staging copy and uses the winner's. The resolution seam
|
|
21
|
+
(`pipelex_api.method_source`) copies everything it needs out of the cached clone before the run
|
|
22
|
+
starts, so eviction can never pull a directory out from under a running pipeline.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
import os
|
|
26
|
+
import shutil
|
|
27
|
+
import subprocess
|
|
28
|
+
import tempfile
|
|
29
|
+
import threading
|
|
30
|
+
import time
|
|
31
|
+
import uuid
|
|
32
|
+
from functools import cache
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
|
|
35
|
+
from pipelex import log
|
|
36
|
+
from pipelex.methods.exceptions import MethodFetchError
|
|
37
|
+
from pipelex.methods.fetching import FetchedMethodPackage, ensure_package_within_bounds, fetch_method_package
|
|
38
|
+
from pipelex.methods.method_ref import MethodRef
|
|
39
|
+
from pipelex.methods.package_locator import locate_package_in_clone
|
|
40
|
+
from pipelex.system.environment import get_optional_env
|
|
41
|
+
|
|
42
|
+
from pipelex_api.limits import MAX_METHOD_CACHE_AGE_SECONDS, MAX_METHOD_CACHE_CLONES, MAX_METHOD_CACHE_TOTAL_BYTES
|
|
43
|
+
|
|
44
|
+
LS_REMOTE_TIMEOUT_SECONDS = 60
|
|
45
|
+
|
|
46
|
+
_STAGING_DIR_NAME = ".staging"
|
|
47
|
+
_DEFAULT_CACHE_DIR_NAME = "pipelex-api-method-cache"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _run_ls_remote(*, clone_url: str, ref_patterns: list[str], ref_str: str) -> list[tuple[str, str]]:
|
|
51
|
+
"""Run `git ls-remote` for the given ref patterns and return `(sha, ref_name)` pairs.
|
|
52
|
+
|
|
53
|
+
Raises:
|
|
54
|
+
MethodFetchError: If git is unavailable, the remote cannot be reached, or the call times out.
|
|
55
|
+
"""
|
|
56
|
+
try:
|
|
57
|
+
result = subprocess.run( # noqa: S603 — fixed argv, no shell; the URL is derived from the validated address grammar
|
|
58
|
+
["git", "ls-remote", clone_url, *ref_patterns], # noqa: S607
|
|
59
|
+
capture_output=True,
|
|
60
|
+
text=True,
|
|
61
|
+
check=True,
|
|
62
|
+
timeout=LS_REMOTE_TIMEOUT_SECONDS,
|
|
63
|
+
)
|
|
64
|
+
except FileNotFoundError as exc:
|
|
65
|
+
msg = "git is not installed or not found on PATH"
|
|
66
|
+
raise MethodFetchError(msg) from exc
|
|
67
|
+
except subprocess.CalledProcessError as exc:
|
|
68
|
+
msg = f"Failed to fetch method '{ref_str}': could not reach the repository ({exc.stderr.strip()})"
|
|
69
|
+
raise MethodFetchError(msg) from exc
|
|
70
|
+
except subprocess.TimeoutExpired as exc:
|
|
71
|
+
msg = f"Timed out resolving method '{ref_str}' against the remote repository"
|
|
72
|
+
raise MethodFetchError(msg) from exc
|
|
73
|
+
|
|
74
|
+
pairs: list[tuple[str, str]] = []
|
|
75
|
+
for line in result.stdout.strip().splitlines():
|
|
76
|
+
parts = line.split("\t")
|
|
77
|
+
if len(parts) == 2:
|
|
78
|
+
pairs.append((parts[0], parts[1]))
|
|
79
|
+
return pairs
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def resolve_remote_commit_sha(*, ref: MethodRef, clone_url: str | None = None) -> str:
|
|
83
|
+
"""Resolve a method reference to the commit SHA it currently points at, without cloning.
|
|
84
|
+
|
|
85
|
+
For a `@<tag>` reference, only `refs/tags/<tag>` is consulted — a branch of the same
|
|
86
|
+
name does not count, so `@main` is refused here, before any clone (the same tags-only
|
|
87
|
+
rule `pipelex.methods.fetching.ensure_cloned_at_tag` enforces after a clone). For an
|
|
88
|
+
annotated tag the peeled (`^{}`) commit SHA is returned, matching what `rev-parse HEAD`
|
|
89
|
+
reports inside a clone at that tag. A bare reference resolves the default branch `HEAD`.
|
|
90
|
+
|
|
91
|
+
Args:
|
|
92
|
+
ref: The parsed method reference.
|
|
93
|
+
clone_url: Override for the derived clone URL (tests, non-default remotes).
|
|
94
|
+
|
|
95
|
+
Returns:
|
|
96
|
+
The full commit SHA the reference resolves to right now.
|
|
97
|
+
|
|
98
|
+
Raises:
|
|
99
|
+
MethodFetchError: If the remote cannot be reached, or `@<tag>` does not name a tag.
|
|
100
|
+
"""
|
|
101
|
+
effective_clone_url = clone_url or ref.clone_url
|
|
102
|
+
if ref.tag:
|
|
103
|
+
pairs = _run_ls_remote(
|
|
104
|
+
clone_url=effective_clone_url,
|
|
105
|
+
ref_patterns=[f"refs/tags/{ref.tag}", f"refs/tags/{ref.tag}^{{}}"],
|
|
106
|
+
ref_str=ref.ref_str,
|
|
107
|
+
)
|
|
108
|
+
peeled = [sha for sha, name in pairs if name.endswith("^{}")]
|
|
109
|
+
if peeled:
|
|
110
|
+
return peeled[0]
|
|
111
|
+
direct = [sha for sha, name in pairs if name == f"refs/tags/{ref.tag}"]
|
|
112
|
+
if direct:
|
|
113
|
+
return direct[0]
|
|
114
|
+
msg = (
|
|
115
|
+
f"'{ref.tag}' in method reference '{ref.ref_str}' does not name a git tag on the repository — "
|
|
116
|
+
f"`@<tag>` pins a git tag (recommended form vX.Y.Z); branch names are not accepted."
|
|
117
|
+
)
|
|
118
|
+
raise MethodFetchError(msg)
|
|
119
|
+
pairs = _run_ls_remote(clone_url=effective_clone_url, ref_patterns=["HEAD"], ref_str=ref.ref_str)
|
|
120
|
+
head = [sha for sha, name in pairs if name == "HEAD"]
|
|
121
|
+
if not head:
|
|
122
|
+
msg = f"The repository behind method reference '{ref.ref_str}' has no HEAD (empty repository?)"
|
|
123
|
+
raise MethodFetchError(msg)
|
|
124
|
+
return head[0]
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _directory_size_bytes(directory: Path) -> int:
|
|
128
|
+
total = 0
|
|
129
|
+
for file_path in directory.rglob("*"):
|
|
130
|
+
if file_path.is_file() and not file_path.is_symlink():
|
|
131
|
+
total += file_path.stat().st_size
|
|
132
|
+
return total
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
class MethodCloneCache:
|
|
136
|
+
"""SHA-keyed, bounded, on-disk cache of method-package clones (see the module docstring)."""
|
|
137
|
+
|
|
138
|
+
def __init__(self, *, root_dir: Path) -> None:
|
|
139
|
+
self._root = root_dir
|
|
140
|
+
self._lock = threading.Lock()
|
|
141
|
+
|
|
142
|
+
def get_or_fetch(self, *, ref: MethodRef, clone_url: str | None = None) -> FetchedMethodPackage:
|
|
143
|
+
"""Return the package a reference points at, cloning only when its SHA is not cached.
|
|
144
|
+
|
|
145
|
+
Always re-runs package location and the bounds check against the (possibly cached)
|
|
146
|
+
clone — only the clone itself is cached, never a verdict about it.
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
ref: The parsed method reference.
|
|
150
|
+
clone_url: Override for the derived clone URL (tests, non-default remotes).
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
The fetched package, rooted inside the cache directory.
|
|
154
|
+
|
|
155
|
+
Raises:
|
|
156
|
+
MethodFetchError: If SHA resolution or the clone fails, or `@<tag>` is not a tag.
|
|
157
|
+
MethodPackageNotFoundError: If no package matches the requested address.
|
|
158
|
+
MethodPackageAmbiguityError: If more than one package matches.
|
|
159
|
+
MethodPackageTooLargeError: If the selected package exceeds the ceilings.
|
|
160
|
+
"""
|
|
161
|
+
commit_sha = resolve_remote_commit_sha(ref=ref, clone_url=clone_url)
|
|
162
|
+
clone_dir = self._root / commit_sha
|
|
163
|
+
with self._lock:
|
|
164
|
+
if clone_dir.is_dir():
|
|
165
|
+
os.utime(clone_dir) # LRU bookkeeping: mtime is the eviction order
|
|
166
|
+
return self._package_from_clone(ref=ref, clone_dir=clone_dir, commit_sha=commit_sha)
|
|
167
|
+
|
|
168
|
+
# Miss: clone into a per-request staging dir (outside the lock — clones are slow),
|
|
169
|
+
# then install atomically. A tag repointed between ls-remote and clone resolves to the
|
|
170
|
+
# clone's ACTUAL commit SHA — that is what gets recorded and keyed, never the stale one.
|
|
171
|
+
staging_root = self._root / _STAGING_DIR_NAME
|
|
172
|
+
staging_root.mkdir(parents=True, exist_ok=True)
|
|
173
|
+
staging_dir = staging_root / uuid.uuid4().hex
|
|
174
|
+
try:
|
|
175
|
+
fetched = fetch_method_package(ref=ref, dest_dir=staging_dir, clone_url=clone_url)
|
|
176
|
+
target_dir = self._root / fetched.commit_sha
|
|
177
|
+
with self._lock:
|
|
178
|
+
if not target_dir.exists():
|
|
179
|
+
try:
|
|
180
|
+
staging_dir.rename(target_dir)
|
|
181
|
+
except OSError:
|
|
182
|
+
# A concurrent request installed the same SHA between the check and the
|
|
183
|
+
# rename; its copy is identical (same commit), so use it.
|
|
184
|
+
log.debug(f"Method clone cache: lost the install race for {fetched.commit_sha}, reusing the winner's copy")
|
|
185
|
+
self._evict()
|
|
186
|
+
return self._package_from_clone(ref=ref, clone_dir=target_dir, commit_sha=fetched.commit_sha)
|
|
187
|
+
finally:
|
|
188
|
+
shutil.rmtree(staging_dir, ignore_errors=True)
|
|
189
|
+
|
|
190
|
+
def _package_from_clone(self, *, ref: MethodRef, clone_dir: Path, commit_sha: str) -> FetchedMethodPackage:
|
|
191
|
+
"""Locate + bound the requested package inside a cached clone (never cached themselves)."""
|
|
192
|
+
located = locate_package_in_clone(clone_root=clone_dir, requested_address=ref.address)
|
|
193
|
+
ensure_package_within_bounds(package_dir=located.package_dir, package_address=located.full_address)
|
|
194
|
+
return FetchedMethodPackage(
|
|
195
|
+
ref=ref,
|
|
196
|
+
full_address=located.full_address,
|
|
197
|
+
commit_sha=commit_sha,
|
|
198
|
+
clone_dir=clone_dir,
|
|
199
|
+
package_dir=located.package_dir,
|
|
200
|
+
manifest=located.manifest,
|
|
201
|
+
)
|
|
202
|
+
|
|
203
|
+
def _evict(self) -> None:
|
|
204
|
+
"""Enforce the age, count, and byte bounds — oldest-first, keeping the newest entry.
|
|
205
|
+
|
|
206
|
+
Called with the lock held, right after an install. Best-effort by design: a clone
|
|
207
|
+
directory another process already removed is simply skipped.
|
|
208
|
+
"""
|
|
209
|
+
entries = sorted(
|
|
210
|
+
(entry for entry in self._root.iterdir() if entry.is_dir() and entry.name != _STAGING_DIR_NAME),
|
|
211
|
+
key=lambda entry: entry.stat().st_mtime,
|
|
212
|
+
)
|
|
213
|
+
if not entries:
|
|
214
|
+
return
|
|
215
|
+
newest = entries[-1]
|
|
216
|
+
now = time.time()
|
|
217
|
+
survivors: list[Path] = []
|
|
218
|
+
for entry in entries:
|
|
219
|
+
if entry != newest and now - entry.stat().st_mtime > MAX_METHOD_CACHE_AGE_SECONDS:
|
|
220
|
+
shutil.rmtree(entry, ignore_errors=True)
|
|
221
|
+
else:
|
|
222
|
+
survivors.append(entry)
|
|
223
|
+
total_bytes = sum(_directory_size_bytes(entry) for entry in survivors)
|
|
224
|
+
while len(survivors) > 1 and (len(survivors) > MAX_METHOD_CACHE_CLONES or total_bytes > MAX_METHOD_CACHE_TOTAL_BYTES):
|
|
225
|
+
oldest = survivors.pop(0)
|
|
226
|
+
total_bytes -= _directory_size_bytes(oldest)
|
|
227
|
+
shutil.rmtree(oldest, ignore_errors=True)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
@cache
|
|
231
|
+
def get_method_clone_cache() -> MethodCloneCache:
|
|
232
|
+
"""The per-instance clone cache singleton.
|
|
233
|
+
|
|
234
|
+
The root directory is `METHOD_CACHE_DIR` when set, else a fixed directory under the
|
|
235
|
+
system temp dir — per instance, surviving requests but not the host, which is exactly
|
|
236
|
+
the cache's contract (a cold instance re-clones once per SHA).
|
|
237
|
+
"""
|
|
238
|
+
configured = get_optional_env("METHOD_CACHE_DIR")
|
|
239
|
+
root_dir = Path(configured) if configured else Path(tempfile.gettempdir()) / _DEFAULT_CACHE_DIR_NAME
|
|
240
|
+
root_dir.mkdir(parents=True, exist_ok=True)
|
|
241
|
+
return MethodCloneCache(root_dir=root_dir)
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
"""Resolve a `method_ref` into run/validate/tooling inputs: fetch → locate → refuse → materialize.
|
|
2
|
+
|
|
3
|
+
The wire field `method_ref` carries a globally resolvable address —
|
|
4
|
+
`github.com/<owner>/<repo>[/<selector>][@<tag>]` — and THIS runner is its resolver (the
|
|
5
|
+
layered extension policy's Rule 3: an address needs no catalog, so it is a layer-2 concept;
|
|
6
|
+
the hosted-only `method_id` never reaches this server). The grammar, the fetch-at-tag, the
|
|
7
|
+
manifest-identity package location, the bounds, and the structures check all live in
|
|
8
|
+
`pipelex.methods`; this module composes them behind the SHA-keyed clone cache
|
|
9
|
+
(`pipelex_api.method_cache`) and shapes the result for each route family:
|
|
10
|
+
|
|
11
|
+
- The run routes and `/validate` use :func:`fetched_method_source`: the package's `.mthds`
|
|
12
|
+
files travel as `mthds_contents` (paired with their real relative paths as
|
|
13
|
+
`mthds_sources`, so diagnostics carry true per-file labels), and only the non-`.mthds`
|
|
14
|
+
files are materialized into a temporary `library_dirs` entry — exactly the split the
|
|
15
|
+
method-bundle transport uses, so a fetched package runs the same proven path.
|
|
16
|
+
- The tooling routes (`/resolve`, `/codegen`, `/build/*`) use
|
|
17
|
+
:func:`fetch_method_mthds_files`: only the `.mthds` files, as `files[]` items, paired with
|
|
18
|
+
the manifest's `main_pipe` so the per-pipe projections default their selector exactly as a
|
|
19
|
+
run does. No Python ever loads there, so the execution-locus gate does not apply.
|
|
20
|
+
|
|
21
|
+
The security gate (packaging invariant 7 — execution locus decides): `.mthds` content is
|
|
22
|
+
data, always acceptable. On a deployment that is NOT sandbox-hosted, a fetched package
|
|
23
|
+
carrying ANY `.py` is refused with the same 403 the bundle transport uses
|
|
24
|
+
(`CustomCodeRequiresSandbox`) — running it would import customer code in-process. On a
|
|
25
|
+
sandbox-hosted deployment, PipeFunc `.py` is acceptable (captured as text, executed in the
|
|
26
|
+
network-blocked sandbox) but a package declaring `StructuredContent` subclasses is refused
|
|
27
|
+
loudly (`MethodStructuresRefusedError` → 403): structure classes would have to be imported into
|
|
28
|
+
the runner's own process, and the rule-naming error teaches authors to express types as MTHDS
|
|
29
|
+
concepts. The library load refuses the same classes again, for a bundle as for a package.
|
|
30
|
+
|
|
31
|
+
Everything a run needs is copied OUT of the cached clone before this module yields — the
|
|
32
|
+
`.mthds` text into memory, the rest into a per-request temp directory — so cache eviction
|
|
33
|
+
can never race a running pipeline.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from collections.abc import Generator
|
|
37
|
+
from contextlib import contextmanager
|
|
38
|
+
from pathlib import Path, PurePosixPath
|
|
39
|
+
from typing import NamedTuple
|
|
40
|
+
|
|
41
|
+
from mthds.package.discovery import MANIFEST_FILENAME
|
|
42
|
+
from pipelex import log
|
|
43
|
+
from pipelex.config import is_pipe_func_sandbox_hosted
|
|
44
|
+
from pipelex.methods.fetching import FetchedMethodPackage, MethodProvenance
|
|
45
|
+
from pipelex.methods.method_ref import parse_method_ref
|
|
46
|
+
from pipelex.methods.structures_check import ensure_no_structured_content_python
|
|
47
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
48
|
+
|
|
49
|
+
from pipelex_api.bundle import ParsedBundle, materialize_parsed
|
|
50
|
+
from pipelex_api.error_types import ErrorType
|
|
51
|
+
from pipelex_api.errors import raise_forbidden, raise_validation_error
|
|
52
|
+
from pipelex_api.limits import MAX_MTHDS_FILE_BYTES
|
|
53
|
+
from pipelex_api.method_cache import get_method_clone_cache
|
|
54
|
+
from pipelex_api.schemas.models import MthdsFileItem
|
|
55
|
+
|
|
56
|
+
_SKIPPED_DIR_NAMES = {".git", "__pycache__"}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class FetchedMethodSource(BaseModel):
|
|
60
|
+
"""A fetched package shaped for the run/validate path (see the module docstring)."""
|
|
61
|
+
|
|
62
|
+
model_config = ConfigDict(frozen=True)
|
|
63
|
+
|
|
64
|
+
mthds_contents: list[str]
|
|
65
|
+
mthds_sources: list[str]
|
|
66
|
+
library_dirs: list[str] | None = None
|
|
67
|
+
main_pipe: str | None = Field(default=None, description="The manifest's declared entry pipe; a request `pipe_code` overrides it.")
|
|
68
|
+
provenance: MethodProvenance
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
class FetchedMthdsFiles(NamedTuple):
|
|
72
|
+
"""A fetched package shaped for the tooling path: its `.mthds` files plus the manifest's entry pipe."""
|
|
73
|
+
|
|
74
|
+
files: list[MthdsFileItem]
|
|
75
|
+
"""The package's `.mthds` files as `files[]` items, each labelled with its real relative path."""
|
|
76
|
+
|
|
77
|
+
main_pipe: str | None
|
|
78
|
+
"""The manifest's declared `main_pipe` (a bare pipe code), or None when the manifest declares none."""
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _fetch_package(method_ref: str) -> FetchedMethodPackage:
|
|
82
|
+
"""Parse the reference and fetch its package through the SHA-keyed clone cache.
|
|
83
|
+
|
|
84
|
+
Every failure mode is a distinct pipelex `MethodRefError` subclass, rendered by the
|
|
85
|
+
global handler as RFC 7807 `problem+json` with the class name as `error_type` and the
|
|
86
|
+
status the API maps for it (`pipelex_api.exception_handlers._ERROR_TYPE_STATUS_OVERRIDES`).
|
|
87
|
+
"""
|
|
88
|
+
ref = parse_method_ref(method_ref)
|
|
89
|
+
package = get_method_clone_cache().get_or_fetch(ref=ref)
|
|
90
|
+
log.info(
|
|
91
|
+
f"Resolved method_ref '{ref.ref_str}': address={package.provenance.address} "
|
|
92
|
+
f"tag={package.provenance.tag} commit_sha={package.provenance.commit_sha}"
|
|
93
|
+
)
|
|
94
|
+
return package
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _package_files(package: FetchedMethodPackage) -> list[Path]:
|
|
98
|
+
"""The package's files, deterministically ordered, with VCS/tooling residue skipped."""
|
|
99
|
+
files: list[Path] = []
|
|
100
|
+
for file_path in sorted(package.package_dir.rglob("*")):
|
|
101
|
+
relative_parts = file_path.relative_to(package.package_dir).parts
|
|
102
|
+
if any(part in _SKIPPED_DIR_NAMES for part in relative_parts):
|
|
103
|
+
continue
|
|
104
|
+
if file_path.is_file():
|
|
105
|
+
files.append(file_path)
|
|
106
|
+
return files
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _read_mthds_text(file_path: Path, *, relative: str, package_address: str) -> str:
|
|
110
|
+
try:
|
|
111
|
+
content = file_path.read_text(encoding="utf-8")
|
|
112
|
+
except UnicodeDecodeError:
|
|
113
|
+
raise_validation_error(message=f"File '{relative}' in method package '{package_address}' is not valid UTF-8.")
|
|
114
|
+
if len(content.encode("utf-8")) > MAX_MTHDS_FILE_BYTES:
|
|
115
|
+
msg = f"File '{relative}' in method package '{package_address}' exceeds the {MAX_MTHDS_FILE_BYTES // 1024} KiB per-file limit."
|
|
116
|
+
raise_validation_error(message=msg)
|
|
117
|
+
return content
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _apply_execution_locus_gate(package: FetchedMethodPackage, *, python_relpaths: list[str]) -> None:
|
|
121
|
+
"""Refuse Python that would execute where it must not (see the module docstring)."""
|
|
122
|
+
if not python_relpaths:
|
|
123
|
+
return
|
|
124
|
+
if not is_pipe_func_sandbox_hosted():
|
|
125
|
+
msg = f"Method package '{package.full_address}' ships custom Python (.py); running it requires a sandbox-hosted deployment."
|
|
126
|
+
raise_forbidden(message=msg, error_type=ErrorType.CUSTOM_CODE_REQUIRES_SANDBOX)
|
|
127
|
+
ensure_no_structured_content_python(package_dir=package.package_dir, package_address=package.full_address)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@contextmanager
|
|
131
|
+
def fetched_method_source(method_ref: str) -> Generator[FetchedMethodSource, None, None]:
|
|
132
|
+
"""Fetch a `method_ref`'s package and shape it for the run/validate path.
|
|
133
|
+
|
|
134
|
+
Yields the package's `.mthds` files as `(mthds_contents, mthds_sources)` pairs and its
|
|
135
|
+
non-`.mthds` files (PipeFunc `.py`, `requirements.txt`, …) materialized into a temporary
|
|
136
|
+
`library_dirs` entry, cleaned up on exit. The execution-locus gate runs before anything
|
|
137
|
+
touches disk.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
method_ref: The raw `method_ref` string from the request.
|
|
141
|
+
|
|
142
|
+
Raises:
|
|
143
|
+
MethodRefError subclasses: parse, fetch, location, bounds, and structures failures —
|
|
144
|
+
each rendered as `problem+json` by the global handler.
|
|
145
|
+
ApiError: the 403 custom-code gate on a non-sandbox deployment, and 422s for a
|
|
146
|
+
package whose `.mthds` content this server cannot accept.
|
|
147
|
+
"""
|
|
148
|
+
package = _fetch_package(method_ref)
|
|
149
|
+
files = _package_files(package)
|
|
150
|
+
python_relpaths = [file_path.relative_to(package.package_dir).as_posix() for file_path in files if file_path.suffix == ".py"]
|
|
151
|
+
_apply_execution_locus_gate(package, python_relpaths=python_relpaths)
|
|
152
|
+
|
|
153
|
+
mthds_contents: list[str] = []
|
|
154
|
+
mthds_sources: list[str] = []
|
|
155
|
+
other_entries: list[tuple[PurePosixPath, bytes]] = []
|
|
156
|
+
for file_path in files:
|
|
157
|
+
relative = file_path.relative_to(package.package_dir).as_posix()
|
|
158
|
+
if file_path.suffix == ".mthds":
|
|
159
|
+
mthds_contents.append(_read_mthds_text(file_path, relative=relative, package_address=package.full_address))
|
|
160
|
+
mthds_sources.append(relative)
|
|
161
|
+
elif file_path.name != MANIFEST_FILENAME:
|
|
162
|
+
# The manifest is already consumed (identity + `main_pipe`); materializing it into
|
|
163
|
+
# the library dir would hand the local loader a package boundary it must not see.
|
|
164
|
+
other_entries.append((PurePosixPath(relative), file_path.read_bytes()))
|
|
165
|
+
if not mthds_contents:
|
|
166
|
+
raise_validation_error(message=f"Method package '{package.full_address}' contains no .mthds file.")
|
|
167
|
+
|
|
168
|
+
if not other_entries:
|
|
169
|
+
yield FetchedMethodSource(
|
|
170
|
+
mthds_contents=mthds_contents,
|
|
171
|
+
mthds_sources=mthds_sources,
|
|
172
|
+
library_dirs=None,
|
|
173
|
+
main_pipe=package.manifest.main_pipe,
|
|
174
|
+
provenance=package.provenance,
|
|
175
|
+
)
|
|
176
|
+
return
|
|
177
|
+
with materialize_parsed(ParsedBundle(entries=tuple(other_entries))) as bundle:
|
|
178
|
+
yield FetchedMethodSource(
|
|
179
|
+
mthds_contents=mthds_contents,
|
|
180
|
+
mthds_sources=mthds_sources,
|
|
181
|
+
library_dirs=[str(bundle.directory)],
|
|
182
|
+
main_pipe=package.manifest.main_pipe,
|
|
183
|
+
provenance=package.provenance,
|
|
184
|
+
)
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def fetch_method_mthds_files(method_ref: str) -> FetchedMthdsFiles:
|
|
188
|
+
"""Fetch a `method_ref`'s package and return its `.mthds` files as `files[]` items, plus its `main_pipe`.
|
|
189
|
+
|
|
190
|
+
The tooling-route shape: each item pairs the file's content with its real relative path
|
|
191
|
+
as `source`, so crate provenance and diagnostics carry true per-file labels. Only
|
|
192
|
+
`.mthds` data travels — the package's Python (if any) never loads on these routes, so
|
|
193
|
+
the execution-locus gate does not apply here. The manifest's `main_pipe` rides beside the
|
|
194
|
+
files so a per-pipe projection can default its selector the way a run does — the manifest
|
|
195
|
+
is the package author's declaration of the entry pipe, and dropping it here would make
|
|
196
|
+
`main_pipe` buy them nothing on the tooling routes.
|
|
197
|
+
|
|
198
|
+
Args:
|
|
199
|
+
method_ref: The raw `method_ref` string from the request.
|
|
200
|
+
|
|
201
|
+
Raises:
|
|
202
|
+
MethodRefError subclasses: parse, fetch, location, and bounds failures.
|
|
203
|
+
ApiError: 422 for a package with no `.mthds` file or one this server cannot accept.
|
|
204
|
+
"""
|
|
205
|
+
package = _fetch_package(method_ref)
|
|
206
|
+
items: list[MthdsFileItem] = []
|
|
207
|
+
for file_path in _package_files(package):
|
|
208
|
+
if file_path.suffix != ".mthds":
|
|
209
|
+
continue
|
|
210
|
+
relative = file_path.relative_to(package.package_dir).as_posix()
|
|
211
|
+
content = _read_mthds_text(file_path, relative=relative, package_address=package.full_address)
|
|
212
|
+
items.append(MthdsFileItem(content=content, source=relative))
|
|
213
|
+
if not items:
|
|
214
|
+
raise_validation_error(message=f"Method package '{package.full_address}' contains no .mthds file.")
|
|
215
|
+
return FetchedMthdsFiles(files=items, main_pipe=package.manifest.main_pipe)
|
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""ASGI middleware shared across the FastAPI app."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
import secrets
|
|
5
|
+
import time
|
|
6
|
+
from collections.abc import Awaitable, Callable
|
|
7
|
+
|
|
8
|
+
from fastapi import Request, Response
|
|
9
|
+
from fastapi.responses import JSONResponse
|
|
10
|
+
from pipelex import log
|
|
11
|
+
from starlette.datastructures import MutableHeaders
|
|
12
|
+
from starlette.types import ASGIApp, Message, Receive, Scope, Send
|
|
13
|
+
|
|
14
|
+
from pipelex_api.error_types import ErrorType
|
|
15
|
+
from pipelex_api.limits import MAX_REQUEST_BODY_BYTES, MAX_REQUEST_BODY_MIB
|
|
16
|
+
from pipelex_api.problem_document import PROBLEM_JSON_MEDIA_TYPE, build_problem_document_from_api_error
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def request_id_of(request: Request) -> str | None:
|
|
20
|
+
"""Return the correlation id `RequestIdMiddleware` stored on this request, or `None` when it did not run.
|
|
21
|
+
|
|
22
|
+
`getattr` rather than attribute access: a request that never went through the middleware — a unit
|
|
23
|
+
test issuing a bare ASGI call, a non-HTTP scope — simply has no id, and a reader of the id is not
|
|
24
|
+
the place to discover that. This is the one way the API reads the id back; the runtime's bound log
|
|
25
|
+
context carries the same value onto every record but is not a lookup table for anyone else.
|
|
26
|
+
"""
|
|
27
|
+
return getattr(request.state, "request_id", None)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _too_large_response(*, request: Request) -> JSONResponse:
|
|
31
|
+
"""Build the 413 RFC 7807 problem response for an over-limit request body.
|
|
32
|
+
|
|
33
|
+
The body-size check runs in middleware, before routing, so it cannot go
|
|
34
|
+
through the `pipelex_api.errors` helpers — a middleware must `return` a response,
|
|
35
|
+
not raise. It builds the same problem document directly, reading the request
|
|
36
|
+
context off the `Request` it was handed. `RequestIdMiddleware` runs outermost,
|
|
37
|
+
so the id is already on `request.state` by the time this is reached; that
|
|
38
|
+
middleware's `send` wrapper also stamps the `X-Request-ID` header onto this
|
|
39
|
+
response.
|
|
40
|
+
"""
|
|
41
|
+
document = build_problem_document_from_api_error(
|
|
42
|
+
ErrorType.PAYLOAD_TOO_LARGE,
|
|
43
|
+
f"Request body exceeds {MAX_REQUEST_BODY_MIB} MiB limit",
|
|
44
|
+
413,
|
|
45
|
+
instance=request.url.path,
|
|
46
|
+
request_id=request_id_of(request),
|
|
47
|
+
)
|
|
48
|
+
return JSONResponse(status_code=413, content=document, media_type=PROBLEM_JSON_MEDIA_TYPE)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
async def request_body_size_middleware(request: Request, call_next: Callable[[Request], Awaitable[Response]]) -> Response:
|
|
52
|
+
"""Reject requests whose body exceeds MAX_REQUEST_BODY_BYTES.
|
|
53
|
+
|
|
54
|
+
Two layers of defense:
|
|
55
|
+
1. Trust `Content-Length` header when present — fast reject before any body is read.
|
|
56
|
+
2. For chunked / missing-header requests, wrap `receive` so we count bytes as
|
|
57
|
+
they arrive. When the cumulative count crosses the cap, the over-limit
|
|
58
|
+
chunk is NOT forwarded: it is replaced with an end-of-stream marker so
|
|
59
|
+
`await request.body()` returns a bounded (often empty) body rather than
|
|
60
|
+
the full oversized payload, and any further read keeps returning the
|
|
61
|
+
same terminator (NOT `http.disconnect`, which would raise
|
|
62
|
+
`ClientDisconnect` and escape past the 413 override). The 413 response
|
|
63
|
+
then overrides whatever the route returned on its truncated input.
|
|
64
|
+
"""
|
|
65
|
+
content_length = request.headers.get("content-length")
|
|
66
|
+
if content_length is not None:
|
|
67
|
+
try:
|
|
68
|
+
declared = int(content_length)
|
|
69
|
+
except ValueError:
|
|
70
|
+
declared = -1
|
|
71
|
+
if declared > MAX_REQUEST_BODY_BYTES:
|
|
72
|
+
return _too_large_response(request=request)
|
|
73
|
+
|
|
74
|
+
original_receive = request.receive
|
|
75
|
+
bytes_seen = 0
|
|
76
|
+
too_large = False
|
|
77
|
+
|
|
78
|
+
async def counting_receive() -> Message:
|
|
79
|
+
nonlocal bytes_seen, too_large
|
|
80
|
+
if too_large:
|
|
81
|
+
# Idempotent end-of-stream on every subsequent read. Returning
|
|
82
|
+
# `http.disconnect` instead would raise `ClientDisconnect` in
|
|
83
|
+
# Starlette's body reader, which would escape `call_next` as an
|
|
84
|
+
# exception and bypass the 413 override below — the request would
|
|
85
|
+
# land as a sanitized 500 instead of 413.
|
|
86
|
+
return {"type": "http.request", "body": b"", "more_body": False}
|
|
87
|
+
message = await original_receive()
|
|
88
|
+
if message.get("type") == "http.request":
|
|
89
|
+
body = message.get("body", b"")
|
|
90
|
+
if isinstance(body, (bytes, bytearray)):
|
|
91
|
+
bytes_seen += len(body)
|
|
92
|
+
if bytes_seen > MAX_REQUEST_BODY_BYTES:
|
|
93
|
+
too_large = True
|
|
94
|
+
return {"type": "http.request", "body": b"", "more_body": False}
|
|
95
|
+
return message
|
|
96
|
+
|
|
97
|
+
request._receive = counting_receive # type: ignore[assignment] # noqa: SLF001
|
|
98
|
+
|
|
99
|
+
response = await call_next(request)
|
|
100
|
+
if too_large:
|
|
101
|
+
return _too_large_response(request=request)
|
|
102
|
+
return response
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
# --- Request correlation -----------------------------------------------------
|
|
106
|
+
|
|
107
|
+
REQUEST_ID_HEADER = "X-Request-ID"
|
|
108
|
+
|
|
109
|
+
# Crockford's Base32 alphabet — the ULID encoding. Excludes I, L, O, U so a
|
|
110
|
+
# request id stays unambiguous if a human transcribes it out of a log.
|
|
111
|
+
_CROCKFORD_BASE32 = "0123456789ABCDEFGHJKMNPQRSTVWXYZ"
|
|
112
|
+
_ULID_LENGTH = 26
|
|
113
|
+
_ULID_RANDOM_BITS = 80
|
|
114
|
+
_ULID_TIMESTAMP_MASK = (1 << 48) - 1
|
|
115
|
+
|
|
116
|
+
# An inbound X-Request-ID is reflected into response headers, logs, and error
|
|
117
|
+
# bodies, so a client-supplied value is never trusted verbatim: it must be a
|
|
118
|
+
# bounded run of injection-safe characters or it is discarded. ULIDs and UUIDs
|
|
119
|
+
# both satisfy this.
|
|
120
|
+
_REQUEST_ID_MAX_LENGTH = 128
|
|
121
|
+
_REQUEST_ID_CHARSET = re.compile(r"\A[A-Za-z0-9_-]+\Z")
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def generate_request_id() -> str:
|
|
125
|
+
"""Generate a fresh ULID: a 26-char, time-sortable, URL-safe identifier.
|
|
126
|
+
|
|
127
|
+
The high 48 bits encode a millisecond Unix timestamp, so ids sort by
|
|
128
|
+
creation time; the low 80 bits are cryptographically random. The 128-bit
|
|
129
|
+
value is rendered in Crockford Base32.
|
|
130
|
+
"""
|
|
131
|
+
timestamp_ms = (time.time_ns() // 1_000_000) & _ULID_TIMESTAMP_MASK
|
|
132
|
+
value = (timestamp_ms << _ULID_RANDOM_BITS) | secrets.randbits(_ULID_RANDOM_BITS)
|
|
133
|
+
digits = [""] * _ULID_LENGTH
|
|
134
|
+
for index in range(_ULID_LENGTH - 1, -1, -1):
|
|
135
|
+
value, remainder = divmod(value, 32)
|
|
136
|
+
digits[index] = _CROCKFORD_BASE32[remainder]
|
|
137
|
+
return "".join(digits)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _is_valid_request_id(candidate: str) -> bool:
|
|
141
|
+
"""Return whether a client-supplied request id is safe to reflect back verbatim."""
|
|
142
|
+
return 0 < len(candidate) <= _REQUEST_ID_MAX_LENGTH and _REQUEST_ID_CHARSET.match(candidate) is not None
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _resolve_request_id(scope: Scope) -> str:
|
|
146
|
+
"""Return the request id for this request.
|
|
147
|
+
|
|
148
|
+
Reuses a valid inbound `X-Request-ID` header; otherwise — absent,
|
|
149
|
+
malformed, or over-long — mints a fresh ULID.
|
|
150
|
+
"""
|
|
151
|
+
for name, value in scope.get("headers", []):
|
|
152
|
+
if name == b"x-request-id":
|
|
153
|
+
candidate: str = value.decode("latin-1").strip()
|
|
154
|
+
if _is_valid_request_id(candidate):
|
|
155
|
+
return candidate
|
|
156
|
+
break
|
|
157
|
+
return generate_request_id()
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
class RequestIdMiddleware:
|
|
161
|
+
"""Pure-ASGI middleware that assigns a correlation id to every HTTP request.
|
|
162
|
+
|
|
163
|
+
For each request it reuses a valid inbound `X-Request-ID` or mints a fresh
|
|
164
|
+
ULID, stores it on `request.state.request_id`, binds the runtime's log
|
|
165
|
+
context to that id for the duration of the request, and echoes
|
|
166
|
+
`X-Request-ID` on the response (success and error alike).
|
|
167
|
+
|
|
168
|
+
Binding the runtime's context rather than an API-owned contextvar is what
|
|
169
|
+
puts `request_id` on *every* record emitted underneath — the API's own
|
|
170
|
+
error lines, and equally the ones pipelex emits from inside a run — as an
|
|
171
|
+
attribute a structured sink indexes, with no call site having to pass it
|
|
172
|
+
and no message having to interpolate it. The binding is in-process only: a
|
|
173
|
+
run dispatched to a worker gets the id from its `RunMetadata`, where the
|
|
174
|
+
run routes put it by passing `request_id_of(request)` to the runner.
|
|
175
|
+
|
|
176
|
+
Applied in `pipelex_api.main` by wrapping the whole FastAPI app
|
|
177
|
+
(`app = RequestIdMiddleware(app)`), NOT via `app.add_middleware()`.
|
|
178
|
+
`add_middleware` always nests a middleware *inside* Starlette's
|
|
179
|
+
`ServerErrorMiddleware`, which would leave it unable to bind the context
|
|
180
|
+
for — or set a header on — the catch-all 500 that `ServerErrorMiddleware`
|
|
181
|
+
emits. Wrapping the app puts this middleware genuinely outermost, outside
|
|
182
|
+
`ServerErrorMiddleware`, so the context is bound and `X-Request-ID` is
|
|
183
|
+
echoed on every response, the catch-all 500 included. Raw ASGI (rather than
|
|
184
|
+
`BaseHTTPMiddleware`) keeps a single contextvar context across the whole
|
|
185
|
+
stack and lets the `send` wrapper inject the header on any response.
|
|
186
|
+
"""
|
|
187
|
+
|
|
188
|
+
def __init__(self, app: ASGIApp) -> None:
|
|
189
|
+
self.app = app
|
|
190
|
+
|
|
191
|
+
async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None:
|
|
192
|
+
if scope["type"] != "http":
|
|
193
|
+
await self.app(scope, receive, send)
|
|
194
|
+
return
|
|
195
|
+
|
|
196
|
+
request_id = _resolve_request_id(scope)
|
|
197
|
+
scope.setdefault("state", {})["request_id"] = request_id
|
|
198
|
+
|
|
199
|
+
async def send_with_request_id(message: Message) -> None:
|
|
200
|
+
if message["type"] == "http.response.start":
|
|
201
|
+
MutableHeaders(scope=message)[REQUEST_ID_HEADER] = request_id
|
|
202
|
+
await send(message)
|
|
203
|
+
|
|
204
|
+
# The runtime's own context, not an API-owned one: `request_id` is one of the three run
|
|
205
|
+
# identifiers it reserves, so every record emitted underneath carries it as an attribute.
|
|
206
|
+
# The route path is deliberately not bound here — it is not a run identifier, and the API
|
|
207
|
+
# ships it as a `route` field on the lines that want it (see `pipelex_api.exception_handlers`).
|
|
208
|
+
with log.context(request_id=request_id):
|
|
209
|
+
await self.app(scope, receive, send_with_request_id)
|