pipelex-api 0.71.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. pipelex_api/__init__.py +0 -0
  2. pipelex_api/api.toml +21 -0
  3. pipelex_api/api_config.py +144 -0
  4. pipelex_api/bundle.py +243 -0
  5. pipelex_api/disclosure.py +43 -0
  6. pipelex_api/error_types.py +80 -0
  7. pipelex_api/error_uri.py +45 -0
  8. pipelex_api/errors.py +130 -0
  9. pipelex_api/exception_handlers.py +693 -0
  10. pipelex_api/json_body.py +182 -0
  11. pipelex_api/limits.py +67 -0
  12. pipelex_api/main.py +221 -0
  13. pipelex_api/method_cache.py +241 -0
  14. pipelex_api/method_source.py +215 -0
  15. pipelex_api/middleware.py +209 -0
  16. pipelex_api/openapi_responses.py +186 -0
  17. pipelex_api/openapi_schema.py +83 -0
  18. pipelex_api/problem_document.py +134 -0
  19. pipelex_api/py.typed +0 -0
  20. pipelex_api/routes/__init__.py +23 -0
  21. pipelex_api/routes/health.py +24 -0
  22. pipelex_api/routes/pipelex/__init__.py +21 -0
  23. pipelex_api/routes/pipelex/agent/__init__.py +11 -0
  24. pipelex_api/routes/pipelex/agent/concept.py +60 -0
  25. pipelex_api/routes/pipelex/agent/models.py +49 -0
  26. pipelex_api/routes/pipelex/agent/pipe_spec.py +59 -0
  27. pipelex_api/routes/pipelex/build/__init__.py +11 -0
  28. pipelex_api/routes/pipelex/build/inputs.py +192 -0
  29. pipelex_api/routes/pipelex/build/output.py +163 -0
  30. pipelex_api/routes/pipelex/build/runner.py +236 -0
  31. pipelex_api/routes/pipelex/codegen.py +164 -0
  32. pipelex_api/routes/pipelex/crate_ops.py +331 -0
  33. pipelex_api/routes/pipelex/pipe_io.py +186 -0
  34. pipelex_api/routes/pipelex/pipeline.py +938 -0
  35. pipelex_api/routes/pipelex/resolve.py +81 -0
  36. pipelex_api/routes/pipelex/tools.py +111 -0
  37. pipelex_api/routes/pipelex/utils.py +6 -0
  38. pipelex_api/routes/pipelex/validate.py +473 -0
  39. pipelex_api/routes/version.py +51 -0
  40. pipelex_api/schemas/__init__.py +0 -0
  41. pipelex_api/schemas/models.py +653 -0
  42. pipelex_api/security.py +284 -0
  43. pipelex_api-0.71.0.dist-info/METADATA +188 -0
  44. pipelex_api-0.71.0.dist-info/RECORD +46 -0
  45. pipelex_api-0.71.0.dist-info/WHEEL +4 -0
  46. pipelex_api-0.71.0.dist-info/licenses/LICENSE +95 -0
@@ -0,0 +1,241 @@
1
+ """Per-instance on-disk clone cache for fetched method packages, keyed by resolved commit SHA.
2
+
3
+ A `method_ref` run fetches a public git repository. Cloning on every request would hammer
4
+ GitHub's anonymous-clone rate limits and add seconds of latency, so this server keeps the
5
+ clones on local disk — but **never keyed by tag**: tags can move, and a moved tag must
6
+ change what runs. Instead, each request first resolves the reference to its commit SHA with
7
+ a cheap `git ls-remote` (no clone), then looks the clone up by that SHA. A repointed tag or
8
+ an advanced default branch resolves to a new SHA and therefore a fresh clone; the cached
9
+ copy of the old SHA ages out on its own.
10
+
11
+ The cache is per server instance (a directory under the system temp dir by default,
12
+ `METHOD_CACHE_DIR` to override) and bounded three ways — clone count, total bytes, and age —
13
+ via the `MAX_METHOD_CACHE_*` knobs in `pipelex_api.limits`. Eviction runs on insert, oldest-first by
14
+ last use, and never evicts the most recently used entry. Package location, bounds, and the
15
+ structures scan are NOT cached: they re-run against the cached clone on every request, so a
16
+ deployment-mode change (or a ceiling change) applies immediately.
17
+
18
+ Concurrency: clones land in a per-request staging directory and are installed with an atomic
19
+ `rename`, so two concurrent requests for the same SHA cannot corrupt each other — the loser
20
+ of the rename race discards its staging copy and uses the winner's. The resolution seam
21
+ (`pipelex_api.method_source`) copies everything it needs out of the cached clone before the run
22
+ starts, so eviction can never pull a directory out from under a running pipeline.
23
+ """
24
+
25
+ import os
26
+ import shutil
27
+ import subprocess
28
+ import tempfile
29
+ import threading
30
+ import time
31
+ import uuid
32
+ from functools import cache
33
+ from pathlib import Path
34
+
35
+ from pipelex import log
36
+ from pipelex.methods.exceptions import MethodFetchError
37
+ from pipelex.methods.fetching import FetchedMethodPackage, ensure_package_within_bounds, fetch_method_package
38
+ from pipelex.methods.method_ref import MethodRef
39
+ from pipelex.methods.package_locator import locate_package_in_clone
40
+ from pipelex.system.environment import get_optional_env
41
+
42
+ from pipelex_api.limits import MAX_METHOD_CACHE_AGE_SECONDS, MAX_METHOD_CACHE_CLONES, MAX_METHOD_CACHE_TOTAL_BYTES
43
+
44
+ LS_REMOTE_TIMEOUT_SECONDS = 60
45
+
46
+ _STAGING_DIR_NAME = ".staging"
47
+ _DEFAULT_CACHE_DIR_NAME = "pipelex-api-method-cache"
48
+
49
+
50
+ def _run_ls_remote(*, clone_url: str, ref_patterns: list[str], ref_str: str) -> list[tuple[str, str]]:
51
+ """Run `git ls-remote` for the given ref patterns and return `(sha, ref_name)` pairs.
52
+
53
+ Raises:
54
+ MethodFetchError: If git is unavailable, the remote cannot be reached, or the call times out.
55
+ """
56
+ try:
57
+ result = subprocess.run( # noqa: S603 — fixed argv, no shell; the URL is derived from the validated address grammar
58
+ ["git", "ls-remote", clone_url, *ref_patterns], # noqa: S607
59
+ capture_output=True,
60
+ text=True,
61
+ check=True,
62
+ timeout=LS_REMOTE_TIMEOUT_SECONDS,
63
+ )
64
+ except FileNotFoundError as exc:
65
+ msg = "git is not installed or not found on PATH"
66
+ raise MethodFetchError(msg) from exc
67
+ except subprocess.CalledProcessError as exc:
68
+ msg = f"Failed to fetch method '{ref_str}': could not reach the repository ({exc.stderr.strip()})"
69
+ raise MethodFetchError(msg) from exc
70
+ except subprocess.TimeoutExpired as exc:
71
+ msg = f"Timed out resolving method '{ref_str}' against the remote repository"
72
+ raise MethodFetchError(msg) from exc
73
+
74
+ pairs: list[tuple[str, str]] = []
75
+ for line in result.stdout.strip().splitlines():
76
+ parts = line.split("\t")
77
+ if len(parts) == 2:
78
+ pairs.append((parts[0], parts[1]))
79
+ return pairs
80
+
81
+
82
+ def resolve_remote_commit_sha(*, ref: MethodRef, clone_url: str | None = None) -> str:
83
+ """Resolve a method reference to the commit SHA it currently points at, without cloning.
84
+
85
+ For a `@<tag>` reference, only `refs/tags/<tag>` is consulted — a branch of the same
86
+ name does not count, so `@main` is refused here, before any clone (the same tags-only
87
+ rule `pipelex.methods.fetching.ensure_cloned_at_tag` enforces after a clone). For an
88
+ annotated tag the peeled (`^{}`) commit SHA is returned, matching what `rev-parse HEAD`
89
+ reports inside a clone at that tag. A bare reference resolves the default branch `HEAD`.
90
+
91
+ Args:
92
+ ref: The parsed method reference.
93
+ clone_url: Override for the derived clone URL (tests, non-default remotes).
94
+
95
+ Returns:
96
+ The full commit SHA the reference resolves to right now.
97
+
98
+ Raises:
99
+ MethodFetchError: If the remote cannot be reached, or `@<tag>` does not name a tag.
100
+ """
101
+ effective_clone_url = clone_url or ref.clone_url
102
+ if ref.tag:
103
+ pairs = _run_ls_remote(
104
+ clone_url=effective_clone_url,
105
+ ref_patterns=[f"refs/tags/{ref.tag}", f"refs/tags/{ref.tag}^{{}}"],
106
+ ref_str=ref.ref_str,
107
+ )
108
+ peeled = [sha for sha, name in pairs if name.endswith("^{}")]
109
+ if peeled:
110
+ return peeled[0]
111
+ direct = [sha for sha, name in pairs if name == f"refs/tags/{ref.tag}"]
112
+ if direct:
113
+ return direct[0]
114
+ msg = (
115
+ f"'{ref.tag}' in method reference '{ref.ref_str}' does not name a git tag on the repository — "
116
+ f"`@<tag>` pins a git tag (recommended form vX.Y.Z); branch names are not accepted."
117
+ )
118
+ raise MethodFetchError(msg)
119
+ pairs = _run_ls_remote(clone_url=effective_clone_url, ref_patterns=["HEAD"], ref_str=ref.ref_str)
120
+ head = [sha for sha, name in pairs if name == "HEAD"]
121
+ if not head:
122
+ msg = f"The repository behind method reference '{ref.ref_str}' has no HEAD (empty repository?)"
123
+ raise MethodFetchError(msg)
124
+ return head[0]
125
+
126
+
127
+ def _directory_size_bytes(directory: Path) -> int:
128
+ total = 0
129
+ for file_path in directory.rglob("*"):
130
+ if file_path.is_file() and not file_path.is_symlink():
131
+ total += file_path.stat().st_size
132
+ return total
133
+
134
+
135
+ class MethodCloneCache:
136
+ """SHA-keyed, bounded, on-disk cache of method-package clones (see the module docstring)."""
137
+
138
+ def __init__(self, *, root_dir: Path) -> None:
139
+ self._root = root_dir
140
+ self._lock = threading.Lock()
141
+
142
+ def get_or_fetch(self, *, ref: MethodRef, clone_url: str | None = None) -> FetchedMethodPackage:
143
+ """Return the package a reference points at, cloning only when its SHA is not cached.
144
+
145
+ Always re-runs package location and the bounds check against the (possibly cached)
146
+ clone — only the clone itself is cached, never a verdict about it.
147
+
148
+ Args:
149
+ ref: The parsed method reference.
150
+ clone_url: Override for the derived clone URL (tests, non-default remotes).
151
+
152
+ Returns:
153
+ The fetched package, rooted inside the cache directory.
154
+
155
+ Raises:
156
+ MethodFetchError: If SHA resolution or the clone fails, or `@<tag>` is not a tag.
157
+ MethodPackageNotFoundError: If no package matches the requested address.
158
+ MethodPackageAmbiguityError: If more than one package matches.
159
+ MethodPackageTooLargeError: If the selected package exceeds the ceilings.
160
+ """
161
+ commit_sha = resolve_remote_commit_sha(ref=ref, clone_url=clone_url)
162
+ clone_dir = self._root / commit_sha
163
+ with self._lock:
164
+ if clone_dir.is_dir():
165
+ os.utime(clone_dir) # LRU bookkeeping: mtime is the eviction order
166
+ return self._package_from_clone(ref=ref, clone_dir=clone_dir, commit_sha=commit_sha)
167
+
168
+ # Miss: clone into a per-request staging dir (outside the lock — clones are slow),
169
+ # then install atomically. A tag repointed between ls-remote and clone resolves to the
170
+ # clone's ACTUAL commit SHA — that is what gets recorded and keyed, never the stale one.
171
+ staging_root = self._root / _STAGING_DIR_NAME
172
+ staging_root.mkdir(parents=True, exist_ok=True)
173
+ staging_dir = staging_root / uuid.uuid4().hex
174
+ try:
175
+ fetched = fetch_method_package(ref=ref, dest_dir=staging_dir, clone_url=clone_url)
176
+ target_dir = self._root / fetched.commit_sha
177
+ with self._lock:
178
+ if not target_dir.exists():
179
+ try:
180
+ staging_dir.rename(target_dir)
181
+ except OSError:
182
+ # A concurrent request installed the same SHA between the check and the
183
+ # rename; its copy is identical (same commit), so use it.
184
+ log.debug(f"Method clone cache: lost the install race for {fetched.commit_sha}, reusing the winner's copy")
185
+ self._evict()
186
+ return self._package_from_clone(ref=ref, clone_dir=target_dir, commit_sha=fetched.commit_sha)
187
+ finally:
188
+ shutil.rmtree(staging_dir, ignore_errors=True)
189
+
190
+ def _package_from_clone(self, *, ref: MethodRef, clone_dir: Path, commit_sha: str) -> FetchedMethodPackage:
191
+ """Locate + bound the requested package inside a cached clone (never cached themselves)."""
192
+ located = locate_package_in_clone(clone_root=clone_dir, requested_address=ref.address)
193
+ ensure_package_within_bounds(package_dir=located.package_dir, package_address=located.full_address)
194
+ return FetchedMethodPackage(
195
+ ref=ref,
196
+ full_address=located.full_address,
197
+ commit_sha=commit_sha,
198
+ clone_dir=clone_dir,
199
+ package_dir=located.package_dir,
200
+ manifest=located.manifest,
201
+ )
202
+
203
+ def _evict(self) -> None:
204
+ """Enforce the age, count, and byte bounds — oldest-first, keeping the newest entry.
205
+
206
+ Called with the lock held, right after an install. Best-effort by design: a clone
207
+ directory another process already removed is simply skipped.
208
+ """
209
+ entries = sorted(
210
+ (entry for entry in self._root.iterdir() if entry.is_dir() and entry.name != _STAGING_DIR_NAME),
211
+ key=lambda entry: entry.stat().st_mtime,
212
+ )
213
+ if not entries:
214
+ return
215
+ newest = entries[-1]
216
+ now = time.time()
217
+ survivors: list[Path] = []
218
+ for entry in entries:
219
+ if entry != newest and now - entry.stat().st_mtime > MAX_METHOD_CACHE_AGE_SECONDS:
220
+ shutil.rmtree(entry, ignore_errors=True)
221
+ else:
222
+ survivors.append(entry)
223
+ total_bytes = sum(_directory_size_bytes(entry) for entry in survivors)
224
+ while len(survivors) > 1 and (len(survivors) > MAX_METHOD_CACHE_CLONES or total_bytes > MAX_METHOD_CACHE_TOTAL_BYTES):
225
+ oldest = survivors.pop(0)
226
+ total_bytes -= _directory_size_bytes(oldest)
227
+ shutil.rmtree(oldest, ignore_errors=True)
228
+
229
+
230
+ @cache
231
+ def get_method_clone_cache() -> MethodCloneCache:
232
+ """The per-instance clone cache singleton.
233
+
234
+ The root directory is `METHOD_CACHE_DIR` when set, else a fixed directory under the
235
+ system temp dir — per instance, surviving requests but not the host, which is exactly
236
+ the cache's contract (a cold instance re-clones once per SHA).
237
+ """
238
+ configured = get_optional_env("METHOD_CACHE_DIR")
239
+ root_dir = Path(configured) if configured else Path(tempfile.gettempdir()) / _DEFAULT_CACHE_DIR_NAME
240
+ root_dir.mkdir(parents=True, exist_ok=True)
241
+ return MethodCloneCache(root_dir=root_dir)
@@ -0,0 +1,215 @@
1
+ """Resolve a `method_ref` into run/validate/tooling inputs: fetch → locate → refuse → materialize.
2
+
3
+ The wire field `method_ref` carries a globally resolvable address —
4
+ `github.com/<owner>/<repo>[/<selector>][@<tag>]` — and THIS runner is its resolver (the
5
+ layered extension policy's Rule 3: an address needs no catalog, so it is a layer-2 concept;
6
+ the hosted-only `method_id` never reaches this server). The grammar, the fetch-at-tag, the
7
+ manifest-identity package location, the bounds, and the structures check all live in
8
+ `pipelex.methods`; this module composes them behind the SHA-keyed clone cache
9
+ (`pipelex_api.method_cache`) and shapes the result for each route family:
10
+
11
+ - The run routes and `/validate` use :func:`fetched_method_source`: the package's `.mthds`
12
+ files travel as `mthds_contents` (paired with their real relative paths as
13
+ `mthds_sources`, so diagnostics carry true per-file labels), and only the non-`.mthds`
14
+ files are materialized into a temporary `library_dirs` entry — exactly the split the
15
+ method-bundle transport uses, so a fetched package runs the same proven path.
16
+ - The tooling routes (`/resolve`, `/codegen`, `/build/*`) use
17
+ :func:`fetch_method_mthds_files`: only the `.mthds` files, as `files[]` items, paired with
18
+ the manifest's `main_pipe` so the per-pipe projections default their selector exactly as a
19
+ run does. No Python ever loads there, so the execution-locus gate does not apply.
20
+
21
+ The security gate (packaging invariant 7 — execution locus decides): `.mthds` content is
22
+ data, always acceptable. On a deployment that is NOT sandbox-hosted, a fetched package
23
+ carrying ANY `.py` is refused with the same 403 the bundle transport uses
24
+ (`CustomCodeRequiresSandbox`) — running it would import customer code in-process. On a
25
+ sandbox-hosted deployment, PipeFunc `.py` is acceptable (captured as text, executed in the
26
+ network-blocked sandbox) but a package declaring `StructuredContent` subclasses is refused
27
+ loudly (`MethodStructuresRefusedError` → 403): structure classes would have to be imported into
28
+ the runner's own process, and the rule-naming error teaches authors to express types as MTHDS
29
+ concepts. The library load refuses the same classes again, for a bundle as for a package.
30
+
31
+ Everything a run needs is copied OUT of the cached clone before this module yields — the
32
+ `.mthds` text into memory, the rest into a per-request temp directory — so cache eviction
33
+ can never race a running pipeline.
34
+ """
35
+
36
+ from collections.abc import Generator
37
+ from contextlib import contextmanager
38
+ from pathlib import Path, PurePosixPath
39
+ from typing import NamedTuple
40
+
41
+ from mthds.package.discovery import MANIFEST_FILENAME
42
+ from pipelex import log
43
+ from pipelex.config import is_pipe_func_sandbox_hosted
44
+ from pipelex.methods.fetching import FetchedMethodPackage, MethodProvenance
45
+ from pipelex.methods.method_ref import parse_method_ref
46
+ from pipelex.methods.structures_check import ensure_no_structured_content_python
47
+ from pydantic import BaseModel, ConfigDict, Field
48
+
49
+ from pipelex_api.bundle import ParsedBundle, materialize_parsed
50
+ from pipelex_api.error_types import ErrorType
51
+ from pipelex_api.errors import raise_forbidden, raise_validation_error
52
+ from pipelex_api.limits import MAX_MTHDS_FILE_BYTES
53
+ from pipelex_api.method_cache import get_method_clone_cache
54
+ from pipelex_api.schemas.models import MthdsFileItem
55
+
56
+ _SKIPPED_DIR_NAMES = {".git", "__pycache__"}
57
+
58
+
59
+ class FetchedMethodSource(BaseModel):
60
+ """A fetched package shaped for the run/validate path (see the module docstring)."""
61
+
62
+ model_config = ConfigDict(frozen=True)
63
+
64
+ mthds_contents: list[str]
65
+ mthds_sources: list[str]
66
+ library_dirs: list[str] | None = None
67
+ main_pipe: str | None = Field(default=None, description="The manifest's declared entry pipe; a request `pipe_code` overrides it.")
68
+ provenance: MethodProvenance
69
+
70
+
71
+ class FetchedMthdsFiles(NamedTuple):
72
+ """A fetched package shaped for the tooling path: its `.mthds` files plus the manifest's entry pipe."""
73
+
74
+ files: list[MthdsFileItem]
75
+ """The package's `.mthds` files as `files[]` items, each labelled with its real relative path."""
76
+
77
+ main_pipe: str | None
78
+ """The manifest's declared `main_pipe` (a bare pipe code), or None when the manifest declares none."""
79
+
80
+
81
+ def _fetch_package(method_ref: str) -> FetchedMethodPackage:
82
+ """Parse the reference and fetch its package through the SHA-keyed clone cache.
83
+
84
+ Every failure mode is a distinct pipelex `MethodRefError` subclass, rendered by the
85
+ global handler as RFC 7807 `problem+json` with the class name as `error_type` and the
86
+ status the API maps for it (`pipelex_api.exception_handlers._ERROR_TYPE_STATUS_OVERRIDES`).
87
+ """
88
+ ref = parse_method_ref(method_ref)
89
+ package = get_method_clone_cache().get_or_fetch(ref=ref)
90
+ log.info(
91
+ f"Resolved method_ref '{ref.ref_str}': address={package.provenance.address} "
92
+ f"tag={package.provenance.tag} commit_sha={package.provenance.commit_sha}"
93
+ )
94
+ return package
95
+
96
+
97
+ def _package_files(package: FetchedMethodPackage) -> list[Path]:
98
+ """The package's files, deterministically ordered, with VCS/tooling residue skipped."""
99
+ files: list[Path] = []
100
+ for file_path in sorted(package.package_dir.rglob("*")):
101
+ relative_parts = file_path.relative_to(package.package_dir).parts
102
+ if any(part in _SKIPPED_DIR_NAMES for part in relative_parts):
103
+ continue
104
+ if file_path.is_file():
105
+ files.append(file_path)
106
+ return files
107
+
108
+
109
+ def _read_mthds_text(file_path: Path, *, relative: str, package_address: str) -> str:
110
+ try:
111
+ content = file_path.read_text(encoding="utf-8")
112
+ except UnicodeDecodeError:
113
+ raise_validation_error(message=f"File '{relative}' in method package '{package_address}' is not valid UTF-8.")
114
+ if len(content.encode("utf-8")) > MAX_MTHDS_FILE_BYTES:
115
+ msg = f"File '{relative}' in method package '{package_address}' exceeds the {MAX_MTHDS_FILE_BYTES // 1024} KiB per-file limit."
116
+ raise_validation_error(message=msg)
117
+ return content
118
+
119
+
120
+ def _apply_execution_locus_gate(package: FetchedMethodPackage, *, python_relpaths: list[str]) -> None:
121
+ """Refuse Python that would execute where it must not (see the module docstring)."""
122
+ if not python_relpaths:
123
+ return
124
+ if not is_pipe_func_sandbox_hosted():
125
+ msg = f"Method package '{package.full_address}' ships custom Python (.py); running it requires a sandbox-hosted deployment."
126
+ raise_forbidden(message=msg, error_type=ErrorType.CUSTOM_CODE_REQUIRES_SANDBOX)
127
+ ensure_no_structured_content_python(package_dir=package.package_dir, package_address=package.full_address)
128
+
129
+
130
+ @contextmanager
131
+ def fetched_method_source(method_ref: str) -> Generator[FetchedMethodSource, None, None]:
132
+ """Fetch a `method_ref`'s package and shape it for the run/validate path.
133
+
134
+ Yields the package's `.mthds` files as `(mthds_contents, mthds_sources)` pairs and its
135
+ non-`.mthds` files (PipeFunc `.py`, `requirements.txt`, …) materialized into a temporary
136
+ `library_dirs` entry, cleaned up on exit. The execution-locus gate runs before anything
137
+ touches disk.
138
+
139
+ Args:
140
+ method_ref: The raw `method_ref` string from the request.
141
+
142
+ Raises:
143
+ MethodRefError subclasses: parse, fetch, location, bounds, and structures failures —
144
+ each rendered as `problem+json` by the global handler.
145
+ ApiError: the 403 custom-code gate on a non-sandbox deployment, and 422s for a
146
+ package whose `.mthds` content this server cannot accept.
147
+ """
148
+ package = _fetch_package(method_ref)
149
+ files = _package_files(package)
150
+ python_relpaths = [file_path.relative_to(package.package_dir).as_posix() for file_path in files if file_path.suffix == ".py"]
151
+ _apply_execution_locus_gate(package, python_relpaths=python_relpaths)
152
+
153
+ mthds_contents: list[str] = []
154
+ mthds_sources: list[str] = []
155
+ other_entries: list[tuple[PurePosixPath, bytes]] = []
156
+ for file_path in files:
157
+ relative = file_path.relative_to(package.package_dir).as_posix()
158
+ if file_path.suffix == ".mthds":
159
+ mthds_contents.append(_read_mthds_text(file_path, relative=relative, package_address=package.full_address))
160
+ mthds_sources.append(relative)
161
+ elif file_path.name != MANIFEST_FILENAME:
162
+ # The manifest is already consumed (identity + `main_pipe`); materializing it into
163
+ # the library dir would hand the local loader a package boundary it must not see.
164
+ other_entries.append((PurePosixPath(relative), file_path.read_bytes()))
165
+ if not mthds_contents:
166
+ raise_validation_error(message=f"Method package '{package.full_address}' contains no .mthds file.")
167
+
168
+ if not other_entries:
169
+ yield FetchedMethodSource(
170
+ mthds_contents=mthds_contents,
171
+ mthds_sources=mthds_sources,
172
+ library_dirs=None,
173
+ main_pipe=package.manifest.main_pipe,
174
+ provenance=package.provenance,
175
+ )
176
+ return
177
+ with materialize_parsed(ParsedBundle(entries=tuple(other_entries))) as bundle:
178
+ yield FetchedMethodSource(
179
+ mthds_contents=mthds_contents,
180
+ mthds_sources=mthds_sources,
181
+ library_dirs=[str(bundle.directory)],
182
+ main_pipe=package.manifest.main_pipe,
183
+ provenance=package.provenance,
184
+ )
185
+
186
+
187
+ def fetch_method_mthds_files(method_ref: str) -> FetchedMthdsFiles:
188
+ """Fetch a `method_ref`'s package and return its `.mthds` files as `files[]` items, plus its `main_pipe`.
189
+
190
+ The tooling-route shape: each item pairs the file's content with its real relative path
191
+ as `source`, so crate provenance and diagnostics carry true per-file labels. Only
192
+ `.mthds` data travels — the package's Python (if any) never loads on these routes, so
193
+ the execution-locus gate does not apply here. The manifest's `main_pipe` rides beside the
194
+ files so a per-pipe projection can default its selector the way a run does — the manifest
195
+ is the package author's declaration of the entry pipe, and dropping it here would make
196
+ `main_pipe` buy them nothing on the tooling routes.
197
+
198
+ Args:
199
+ method_ref: The raw `method_ref` string from the request.
200
+
201
+ Raises:
202
+ MethodRefError subclasses: parse, fetch, location, and bounds failures.
203
+ ApiError: 422 for a package with no `.mthds` file or one this server cannot accept.
204
+ """
205
+ package = _fetch_package(method_ref)
206
+ items: list[MthdsFileItem] = []
207
+ for file_path in _package_files(package):
208
+ if file_path.suffix != ".mthds":
209
+ continue
210
+ relative = file_path.relative_to(package.package_dir).as_posix()
211
+ content = _read_mthds_text(file_path, relative=relative, package_address=package.full_address)
212
+ items.append(MthdsFileItem(content=content, source=relative))
213
+ if not items:
214
+ raise_validation_error(message=f"Method package '{package.full_address}' contains no .mthds file.")
215
+ return FetchedMthdsFiles(files=items, main_pipe=package.manifest.main_pipe)
@@ -0,0 +1,209 @@
1
+ """ASGI middleware shared across the FastAPI app."""
2
+
3
+ import re
4
+ import secrets
5
+ import time
6
+ from collections.abc import Awaitable, Callable
7
+
8
+ from fastapi import Request, Response
9
+ from fastapi.responses import JSONResponse
10
+ from pipelex import log
11
+ from starlette.datastructures import MutableHeaders
12
+ from starlette.types import ASGIApp, Message, Receive, Scope, Send
13
+
14
+ from pipelex_api.error_types import ErrorType
15
+ from pipelex_api.limits import MAX_REQUEST_BODY_BYTES, MAX_REQUEST_BODY_MIB
16
+ from pipelex_api.problem_document import PROBLEM_JSON_MEDIA_TYPE, build_problem_document_from_api_error
17
+
18
+
19
+ def request_id_of(request: Request) -> str | None:
20
+ """Return the correlation id `RequestIdMiddleware` stored on this request, or `None` when it did not run.
21
+
22
+ `getattr` rather than attribute access: a request that never went through the middleware — a unit
23
+ test issuing a bare ASGI call, a non-HTTP scope — simply has no id, and a reader of the id is not
24
+ the place to discover that. This is the one way the API reads the id back; the runtime's bound log
25
+ context carries the same value onto every record but is not a lookup table for anyone else.
26
+ """
27
+ return getattr(request.state, "request_id", None)
28
+
29
+
30
+ def _too_large_response(*, request: Request) -> JSONResponse:
31
+ """Build the 413 RFC 7807 problem response for an over-limit request body.
32
+
33
+ The body-size check runs in middleware, before routing, so it cannot go
34
+ through the `pipelex_api.errors` helpers — a middleware must `return` a response,
35
+ not raise. It builds the same problem document directly, reading the request
36
+ context off the `Request` it was handed. `RequestIdMiddleware` runs outermost,
37
+ so the id is already on `request.state` by the time this is reached; that
38
+ middleware's `send` wrapper also stamps the `X-Request-ID` header onto this
39
+ response.
40
+ """
41
+ document = build_problem_document_from_api_error(
42
+ ErrorType.PAYLOAD_TOO_LARGE,
43
+ f"Request body exceeds {MAX_REQUEST_BODY_MIB} MiB limit",
44
+ 413,
45
+ instance=request.url.path,
46
+ request_id=request_id_of(request),
47
+ )
48
+ return JSONResponse(status_code=413, content=document, media_type=PROBLEM_JSON_MEDIA_TYPE)
49
+
50
+
51
+ async def request_body_size_middleware(request: Request, call_next: Callable[[Request], Awaitable[Response]]) -> Response:
52
+ """Reject requests whose body exceeds MAX_REQUEST_BODY_BYTES.
53
+
54
+ Two layers of defense:
55
+ 1. Trust `Content-Length` header when present — fast reject before any body is read.
56
+ 2. For chunked / missing-header requests, wrap `receive` so we count bytes as
57
+ they arrive. When the cumulative count crosses the cap, the over-limit
58
+ chunk is NOT forwarded: it is replaced with an end-of-stream marker so
59
+ `await request.body()` returns a bounded (often empty) body rather than
60
+ the full oversized payload, and any further read keeps returning the
61
+ same terminator (NOT `http.disconnect`, which would raise
62
+ `ClientDisconnect` and escape past the 413 override). The 413 response
63
+ then overrides whatever the route returned on its truncated input.
64
+ """
65
+ content_length = request.headers.get("content-length")
66
+ if content_length is not None:
67
+ try:
68
+ declared = int(content_length)
69
+ except ValueError:
70
+ declared = -1
71
+ if declared > MAX_REQUEST_BODY_BYTES:
72
+ return _too_large_response(request=request)
73
+
74
+ original_receive = request.receive
75
+ bytes_seen = 0
76
+ too_large = False
77
+
78
+ async def counting_receive() -> Message:
79
+ nonlocal bytes_seen, too_large
80
+ if too_large:
81
+ # Idempotent end-of-stream on every subsequent read. Returning
82
+ # `http.disconnect` instead would raise `ClientDisconnect` in
83
+ # Starlette's body reader, which would escape `call_next` as an
84
+ # exception and bypass the 413 override below — the request would
85
+ # land as a sanitized 500 instead of 413.
86
+ return {"type": "http.request", "body": b"", "more_body": False}
87
+ message = await original_receive()
88
+ if message.get("type") == "http.request":
89
+ body = message.get("body", b"")
90
+ if isinstance(body, (bytes, bytearray)):
91
+ bytes_seen += len(body)
92
+ if bytes_seen > MAX_REQUEST_BODY_BYTES:
93
+ too_large = True
94
+ return {"type": "http.request", "body": b"", "more_body": False}
95
+ return message
96
+
97
+ request._receive = counting_receive # type: ignore[assignment] # noqa: SLF001
98
+
99
+ response = await call_next(request)
100
+ if too_large:
101
+ return _too_large_response(request=request)
102
+ return response
103
+
104
+
105
+ # --- Request correlation -----------------------------------------------------
106
+
107
+ REQUEST_ID_HEADER = "X-Request-ID"
108
+
109
+ # Crockford's Base32 alphabet — the ULID encoding. Excludes I, L, O, U so a
110
+ # request id stays unambiguous if a human transcribes it out of a log.
111
+ _CROCKFORD_BASE32 = "0123456789ABCDEFGHJKMNPQRSTVWXYZ"
112
+ _ULID_LENGTH = 26
113
+ _ULID_RANDOM_BITS = 80
114
+ _ULID_TIMESTAMP_MASK = (1 << 48) - 1
115
+
116
+ # An inbound X-Request-ID is reflected into response headers, logs, and error
117
+ # bodies, so a client-supplied value is never trusted verbatim: it must be a
118
+ # bounded run of injection-safe characters or it is discarded. ULIDs and UUIDs
119
+ # both satisfy this.
120
+ _REQUEST_ID_MAX_LENGTH = 128
121
+ _REQUEST_ID_CHARSET = re.compile(r"\A[A-Za-z0-9_-]+\Z")
122
+
123
+
124
+ def generate_request_id() -> str:
125
+ """Generate a fresh ULID: a 26-char, time-sortable, URL-safe identifier.
126
+
127
+ The high 48 bits encode a millisecond Unix timestamp, so ids sort by
128
+ creation time; the low 80 bits are cryptographically random. The 128-bit
129
+ value is rendered in Crockford Base32.
130
+ """
131
+ timestamp_ms = (time.time_ns() // 1_000_000) & _ULID_TIMESTAMP_MASK
132
+ value = (timestamp_ms << _ULID_RANDOM_BITS) | secrets.randbits(_ULID_RANDOM_BITS)
133
+ digits = [""] * _ULID_LENGTH
134
+ for index in range(_ULID_LENGTH - 1, -1, -1):
135
+ value, remainder = divmod(value, 32)
136
+ digits[index] = _CROCKFORD_BASE32[remainder]
137
+ return "".join(digits)
138
+
139
+
140
+ def _is_valid_request_id(candidate: str) -> bool:
141
+ """Return whether a client-supplied request id is safe to reflect back verbatim."""
142
+ return 0 < len(candidate) <= _REQUEST_ID_MAX_LENGTH and _REQUEST_ID_CHARSET.match(candidate) is not None
143
+
144
+
145
+ def _resolve_request_id(scope: Scope) -> str:
146
+ """Return the request id for this request.
147
+
148
+ Reuses a valid inbound `X-Request-ID` header; otherwise — absent,
149
+ malformed, or over-long — mints a fresh ULID.
150
+ """
151
+ for name, value in scope.get("headers", []):
152
+ if name == b"x-request-id":
153
+ candidate: str = value.decode("latin-1").strip()
154
+ if _is_valid_request_id(candidate):
155
+ return candidate
156
+ break
157
+ return generate_request_id()
158
+
159
+
160
+ class RequestIdMiddleware:
161
+ """Pure-ASGI middleware that assigns a correlation id to every HTTP request.
162
+
163
+ For each request it reuses a valid inbound `X-Request-ID` or mints a fresh
164
+ ULID, stores it on `request.state.request_id`, binds the runtime's log
165
+ context to that id for the duration of the request, and echoes
166
+ `X-Request-ID` on the response (success and error alike).
167
+
168
+ Binding the runtime's context rather than an API-owned contextvar is what
169
+ puts `request_id` on *every* record emitted underneath — the API's own
170
+ error lines, and equally the ones pipelex emits from inside a run — as an
171
+ attribute a structured sink indexes, with no call site having to pass it
172
+ and no message having to interpolate it. The binding is in-process only: a
173
+ run dispatched to a worker gets the id from its `RunMetadata`, where the
174
+ run routes put it by passing `request_id_of(request)` to the runner.
175
+
176
+ Applied in `pipelex_api.main` by wrapping the whole FastAPI app
177
+ (`app = RequestIdMiddleware(app)`), NOT via `app.add_middleware()`.
178
+ `add_middleware` always nests a middleware *inside* Starlette's
179
+ `ServerErrorMiddleware`, which would leave it unable to bind the context
180
+ for — or set a header on — the catch-all 500 that `ServerErrorMiddleware`
181
+ emits. Wrapping the app puts this middleware genuinely outermost, outside
182
+ `ServerErrorMiddleware`, so the context is bound and `X-Request-ID` is
183
+ echoed on every response, the catch-all 500 included. Raw ASGI (rather than
184
+ `BaseHTTPMiddleware`) keeps a single contextvar context across the whole
185
+ stack and lets the `send` wrapper inject the header on any response.
186
+ """
187
+
188
+ def __init__(self, app: ASGIApp) -> None:
189
+ self.app = app
190
+
191
+ async def __call__(self, scope: Scope, receive: Receive, send: Send) -> None:
192
+ if scope["type"] != "http":
193
+ await self.app(scope, receive, send)
194
+ return
195
+
196
+ request_id = _resolve_request_id(scope)
197
+ scope.setdefault("state", {})["request_id"] = request_id
198
+
199
+ async def send_with_request_id(message: Message) -> None:
200
+ if message["type"] == "http.response.start":
201
+ MutableHeaders(scope=message)[REQUEST_ID_HEADER] = request_id
202
+ await send(message)
203
+
204
+ # The runtime's own context, not an API-owned one: `request_id` is one of the three run
205
+ # identifiers it reserves, so every record emitted underneath carries it as an attribute.
206
+ # The route path is deliberately not bound here — it is not a run identifier, and the API
207
+ # ships it as a `route` field on the lines that want it (see `pipelex_api.exception_handlers`).
208
+ with log.context(request_id=request_id):
209
+ await self.app(scope, receive, send_with_request_id)