programasweights 0.4.4__tar.gz → 0.4.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. programasweights-0.4.6/.github/workflows/test.yml +82 -0
  2. {programasweights-0.4.4 → programasweights-0.4.6}/AGENTS.md +6 -0
  3. {programasweights-0.4.4 → programasweights-0.4.6}/CHANGELOG.md +22 -0
  4. {programasweights-0.4.4 → programasweights-0.4.6}/PKG-INFO +28 -3
  5. {programasweights-0.4.4 → programasweights-0.4.6}/PYPI_README.md +26 -1
  6. {programasweights-0.4.4 → programasweights-0.4.6}/README.md +26 -1
  7. {programasweights-0.4.4 → programasweights-0.4.6}/docs/api-reference/python-sdk.md +94 -6
  8. {programasweights-0.4.4 → programasweights-0.4.6}/docs/api-reference/rest-api.md +10 -1
  9. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/__init__.py +45 -22
  10. programasweights-0.4.6/programasweights/_program_reference.py +61 -0
  11. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/client.py +11 -7
  12. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/convert_peft_to_paw.py +12 -14
  13. programasweights-0.4.6/programasweights/errors.py +79 -0
  14. programasweights-0.4.6/programasweights/local_program.py +196 -0
  15. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime_llamacpp.py +38 -0
  16. {programasweights-0.4.4 → programasweights-0.4.6}/pyproject.toml +1 -1
  17. programasweights-0.4.6/tests/test_api_errors.py +222 -0
  18. {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_base_interpreter.py +204 -2
  19. programasweights-0.4.6/tests/test_compile_timeouts.py +129 -0
  20. programasweights-0.4.6/tests/test_local_program.py +676 -0
  21. programasweights-0.4.4/.github/workflows/test.yml +0 -43
  22. {programasweights-0.4.4 → programasweights-0.4.6}/.gitignore +0 -0
  23. {programasweights-0.4.4 → programasweights-0.4.6}/.readthedocs.yaml +0 -0
  24. {programasweights-0.4.4 → programasweights-0.4.6}/LICENSE +0 -0
  25. {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/001-llama-cpp-over-pytorch.md +0 -0
  26. {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/002-q4_0-adapter-format.md +0 -0
  27. {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/003-single-spec-field.md +0 -0
  28. {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/004-compiler-naming.md +0 -0
  29. {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/005-vllm-hidden-states.md +0 -0
  30. {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/006-email-api-key-auth.md +0 -0
  31. {programasweights-0.4.4 → programasweights-0.4.6}/docs/advanced/adrs.md +0 -0
  32. {programasweights-0.4.4 → programasweights-0.4.6}/docs/advanced/architecture.md +0 -0
  33. {programasweights-0.4.4 → programasweights-0.4.6}/docs/api-reference/cli.md +0 -0
  34. {programasweights-0.4.4 → programasweights-0.4.6}/docs/architecture.md +0 -0
  35. {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/alien-taboo.md +0 -0
  36. {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/log-monitoring.md +0 -0
  37. {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/semantic-search.md +0 -0
  38. {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/site-navigation.md +0 -0
  39. {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/tool-calling.md +0 -0
  40. {programasweights-0.4.4 → programasweights-0.4.6}/docs/getting-started/first-program.md +0 -0
  41. {programasweights-0.4.4 → programasweights-0.4.6}/docs/getting-started/installation.md +0 -0
  42. {programasweights-0.4.4 → programasweights-0.4.6}/docs/getting-started/naming-programs.md +0 -0
  43. {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/browser-inference.md +0 -0
  44. {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/how-it-works.md +0 -0
  45. {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/local-inference.md +0 -0
  46. {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/writing-good-specs.md +0 -0
  47. {programasweights-0.4.4 → programasweights-0.4.6}/docs/hub/browsing-programs.md +0 -0
  48. {programasweights-0.4.4 → programasweights-0.4.6}/docs/hub/feedback-cases.md +0 -0
  49. {programasweights-0.4.4 → programasweights-0.4.6}/docs/hub/publishing-programs.md +0 -0
  50. {programasweights-0.4.4 → programasweights-0.4.6}/docs/index.md +0 -0
  51. {programasweights-0.4.4 → programasweights-0.4.6}/docs/requirements.txt +0 -0
  52. {programasweights-0.4.4 → programasweights-0.4.6}/examples/flask_app.py +0 -0
  53. {programasweights-0.4.4 → programasweights-0.4.6}/examples/jupyter_notebook.py +0 -0
  54. {programasweights-0.4.4 → programasweights-0.4.6}/examples/langchain_integration.py +0 -0
  55. {programasweights-0.4.4 → programasweights-0.4.6}/examples/paw_monitor.py +0 -0
  56. {programasweights-0.4.4 → programasweights-0.4.6}/examples/replace_openai.py +0 -0
  57. {programasweights-0.4.4 → programasweights-0.4.6}/mkdocs.yml +0 -0
  58. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/_output.py +0 -0
  59. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/artifacts.py +0 -0
  60. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/cache.py +0 -0
  61. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/cli.py +0 -0
  62. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/compiler/__init__.py +0 -0
  63. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/compiler/dummy.py +0 -0
  64. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/config.py +0 -0
  65. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/paw_format.py +0 -0
  66. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime/__init__.py +0 -0
  67. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime/interpreter.py +0 -0
  68. {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime/interpreter_onnx.py +0 -0
  69. {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_cli_auth.py +0 -0
  70. {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_desktop_sdk.py +0 -0
  71. {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_offline_cache.py +0 -0
  72. {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_runtime_registry_sdk.py +0 -0
  73. {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_sdk.py +0 -0
  74. {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_sdk.sh +0 -0
@@ -0,0 +1,82 @@
1
+ name: tests
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ fail-fast: false
13
+ matrix:
14
+ python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+
18
+ - name: Set up Python ${{ matrix.python-version }}
19
+ uses: actions/setup-python@v5
20
+ with:
21
+ python-version: ${{ matrix.python-version }}
22
+
23
+ - name: Install (hermetic deps only)
24
+ # Install httpx + pytest and the package itself without pulling the heavy
25
+ # llama-cpp-python build. Runtime tests inject a fake llama_cpp module,
26
+ # so CI needs neither the native extension nor a model download.
27
+ run: |
28
+ python -m pip install --upgrade pip
29
+ python -m pip install httpx pytest
30
+ python -m pip install -e . --no-deps
31
+
32
+ - name: Run hermetic tests
33
+ # Scoped to tests that need no network, no model download, and no
34
+ # PAW_API_KEY. Auth tests (@needs_auth) auto-skip without a key; the
35
+ # network/model-download tests in test_sdk.py are excluded here and can
36
+ # be run separately against a live server.
37
+ run: |
38
+ pytest \
39
+ tests/test_api_errors.py \
40
+ tests/test_compile_timeouts.py \
41
+ tests/test_local_program.py \
42
+ tests/test_base_interpreter.py \
43
+ tests/test_cli_auth.py \
44
+ tests/test_desktop_sdk.py \
45
+ tests/test_runtime_registry_sdk.py \
46
+ tests/test_sdk.py::TestInstallAndImport
47
+
48
+ local-files-windows:
49
+ runs-on: windows-latest
50
+ strategy:
51
+ fail-fast: false
52
+ matrix:
53
+ python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
54
+ steps:
55
+ - uses: actions/checkout@v4
56
+ - uses: actions/setup-python@v5
57
+ with:
58
+ python-version: ${{ matrix.python-version }}
59
+ - name: Install hermetic test dependencies
60
+ run: |
61
+ python -m pip install httpx pytest
62
+ python -m pip install -e . --no-deps
63
+ - name: Test Windows local paths, cache locks, and compile errors
64
+ run: python -m pytest tests/test_api_errors.py tests/test_compile_timeouts.py tests/test_local_program.py --junitxml=test-results.xml
65
+ - name: Annotate Windows test failures
66
+ if: failure()
67
+ shell: python
68
+ run: |
69
+ from pathlib import Path
70
+ import xml.etree.ElementTree as ET
71
+
72
+ report = Path("test-results.xml")
73
+ if report.exists():
74
+ for case in ET.parse(report).iter("testcase"):
75
+ for result in case:
76
+ if result.tag in {"failure", "error"}:
77
+ # GitHub truncates annotations, so retain the actual
78
+ # exception at the end of long pytest tracebacks.
79
+ detail = result.text or result.get("message", "")
80
+ message = f"{case.get('name')}: {detail[-3500:]}"
81
+ message = message.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")
82
+ print(f"::error::{message}")
@@ -43,6 +43,8 @@ fn = paw.compile_and_load("Classify sentiment as positive or negative")
43
43
  fn("I love this!") # "positive"
44
44
  ```
45
45
 
46
+ Load a local `.paw` file with `paw.function("./classifier.paw")` (SDK 0.4.5+).
47
+
46
48
  If you want the smaller browser-compatible runtime explicitly, pass `compiler="paw-4b-gpt2"`. Otherwise, omit `compiler` and let the server default decide.
47
49
 
48
50
  ## Current Public Compilers
@@ -89,6 +91,8 @@ Output: delete
89
91
  - Spec + input + output share a ~2048 token context window. Inputs that exceed it will error.
90
92
  - `max_tokens` defaults to `None`: generation runs until EOS or the context limit.
91
93
  - Compile runs on the hosted PAW API. Inference should usually run locally through the SDK.
94
+ - Synchronous compile requests use a 40-minute read timeout.
95
+ - **Run local inference sequentially by default.** With PAW’s current llama.cpp backend, simultaneous inference calls often perform worse. Reuse loaded functions and process inputs one at a time; never call the same function instance concurrently.
92
96
  - **GPU acceleration** is enabled by default (`n_gpu_layers=-1`). Uses Metal on Mac, CUDA on Linux, and falls back to CPU automatically. If GPU causes issues, set `PAW_GPU_LAYERS=0` or pass `n_gpu_layers=0`.
93
97
  - **First call** is usually ~1-5s because it loads the base model. Subsequent calls are typically ~0.05-0.5s depending on input length and GPU availability.
94
98
  - **Base model files are shared** across programs on disk. Each Standard LoRA adapter is ~22 MB; each Compact LoRA adapter is ~5 MB.
@@ -102,6 +106,8 @@ Output: delete
102
106
 
103
107
  ## Common Errors
104
108
 
109
+ Compile API HTTP errors raise `paw.APIError`. Check `error.code` and `error.message` for details. The SDK does not retry automatically.
110
+
105
111
  | Error | Cause | Fix |
106
112
  |-------|-------|-----|
107
113
  | `RuntimeError: assets not ready` on download | Program is still generating after compile | The SDK polls automatically for up to 60s. If it still fails, retry shortly or recompile. |
@@ -1,5 +1,27 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.4.6 (2026-09-13)
4
+
5
+ - Add an optional `logits_processor` argument to a compiled or base program
6
+ call for caller-supplied token constraints in llama.cpp's sampler.
7
+ Defaults to `None`; sampling is unchanged when it is unset.
8
+ - Propagate processor failures to the caller instead of letting native
9
+ callback errors silently continue with unconstrained output.
10
+
11
+ ## 0.4.5 (2026-09-10)
12
+
13
+ - Expose structured compile API failures as `paw.APIError`, compatible with
14
+ `httpx.HTTPStatusError`, preserving server code, message, request ID, and the
15
+ original response. Transport errors propagate unchanged; no automatic retries.
16
+ - Allow synchronous compilation a 2,400-second read timeout while keeping
17
+ connect, write, and pool timeouts at 120 seconds. Async submission stays at
18
+ 30 seconds; precheck, status, and cancellation stay at 10 seconds.
19
+ - Load current GGUF ZIP `.paw` files through `paw.function` using explicit
20
+ local paths or `Path` objects. Validated imports use a separate SHA-256 cache
21
+ without replacing Hub caches or source files; invalid local files never fall
22
+ back to Hub lookup. Shared base-model assets may still need downloading unless
23
+ offline mode is requested. Legacy tensor-format `.paw` files are unsupported.
24
+
3
25
  ## 0.4.4 (2026-07-18)
4
26
 
5
27
  - Add desktop preparation and cache inspection APIs with structured progress:
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: programasweights
3
- Version: 0.4.4
3
+ Version: 0.4.6
4
4
  Summary: Compile natural language specifications into neural programs that run locally via llama.cpp.
5
5
  Project-URL: Homepage, https://programasweights.com
6
6
  Project-URL: Repository, https://github.com/programasweights/programasweights-python
@@ -84,6 +84,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
84
84
 
85
85
  GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
86
86
 
87
+ ## Constrained Decoding
88
+
89
+ In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
90
+
91
+ ```python
92
+ import llama_cpp
93
+ # my_processor is your compatible callable: (input_ids, scores) -> scores.
94
+ fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
95
+ ```
96
+
97
+ Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
98
+
87
99
  ## Desktop and Offline Workflows
88
100
 
89
101
  Prepare and inspect validated local assets without keeping a model loaded:
@@ -102,6 +114,19 @@ finetune compiles can be queued with
102
114
  `paw.compile_async(spec, compiler="paw-ft-bs48")`; an explicit finetune
103
115
  compiler is required.
104
116
 
117
+ Load a saved current GGUF ZIP `.paw` bundle directly (SDK 0.4.5+):
118
+
119
+ ```python
120
+ from pathlib import Path
121
+ fn = paw.function(Path("classifier.paw"))
122
+ ```
123
+
124
+ Local files are validated into a separate SHA-256 cache without changing the
125
+ source or falling back to Hub lookup. Runtime metadata or the shared base model
126
+ may still download; `offline=True` prohibits those requests. Legacy tensor-format
127
+ `.paw` files are unsupported. Local-file inputs are supported by `function`,
128
+ not `prepare_program` or `is_offline_ready`.
129
+
105
130
  Advanced adapter-free inference is available with
106
131
  `paw.function(None, interpreter="gpt2")`; see the Python API reference for
107
132
  its intentionally strict semantics.
@@ -179,4 +204,4 @@ paw login
179
204
 
180
205
  ## License
181
206
 
182
- MIT
207
+ MIT
@@ -53,6 +53,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
53
53
 
54
54
  GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
55
55
 
56
+ ## Constrained Decoding
57
+
58
+ In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
59
+
60
+ ```python
61
+ import llama_cpp
62
+ # my_processor is your compatible callable: (input_ids, scores) -> scores.
63
+ fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
64
+ ```
65
+
66
+ Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
67
+
56
68
  ## Desktop and Offline Workflows
57
69
 
58
70
  Prepare and inspect validated local assets without keeping a model loaded:
@@ -71,6 +83,19 @@ finetune compiles can be queued with
71
83
  `paw.compile_async(spec, compiler="paw-ft-bs48")`; an explicit finetune
72
84
  compiler is required.
73
85
 
86
+ Load a saved current GGUF ZIP `.paw` bundle directly (SDK 0.4.5+):
87
+
88
+ ```python
89
+ from pathlib import Path
90
+ fn = paw.function(Path("classifier.paw"))
91
+ ```
92
+
93
+ Local files are validated into a separate SHA-256 cache without changing the
94
+ source or falling back to Hub lookup. Runtime metadata or the shared base model
95
+ may still download; `offline=True` prohibits those requests. Legacy tensor-format
96
+ `.paw` files are unsupported. Local-file inputs are supported by `function`,
97
+ not `prepare_program` or `is_offline_ready`.
98
+
74
99
  Advanced adapter-free inference is available with
75
100
  `paw.function(None, interpreter="gpt2")`; see the Python API reference for
76
101
  its intentionally strict semantics.
@@ -148,4 +173,4 @@ paw login
148
173
 
149
174
  ## License
150
175
 
151
- MIT
176
+ MIT
@@ -53,6 +53,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
53
53
 
54
54
  GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
55
55
 
56
+ ## Constrained Decoding
57
+
58
+ In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
59
+
60
+ ```python
61
+ import llama_cpp
62
+ # my_processor is your compatible callable: (input_ids, scores) -> scores.
63
+ fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
64
+ ```
65
+
66
+ Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
67
+
56
68
  ## Desktop and Offline Workflows
57
69
 
58
70
  Prepare and inspect validated local assets without keeping a model loaded:
@@ -71,6 +83,19 @@ finetune compiles can be queued with
71
83
  `paw.compile_async(spec, compiler="paw-ft-bs48")`; an explicit finetune
72
84
  compiler is required.
73
85
 
86
+ Load a saved current GGUF ZIP `.paw` bundle directly (SDK 0.4.5+):
87
+
88
+ ```python
89
+ from pathlib import Path
90
+ fn = paw.function(Path("classifier.paw"))
91
+ ```
92
+
93
+ Local files are validated into a separate SHA-256 cache without changing the
94
+ source or falling back to Hub lookup. Runtime metadata or the shared base model
95
+ may still download; `offline=True` prohibits those requests. Legacy tensor-format
96
+ `.paw` files are unsupported. Local-file inputs are supported by `function`,
97
+ not `prepare_program` or `is_offline_ready`.
98
+
74
99
  Advanced adapter-free inference is available with
75
100
  `paw.function(None, interpreter="gpt2")`; see the
76
101
  [Python API reference](docs/api-reference/python-sdk.md#advanced-adapter-free-base-interpreter)
@@ -149,4 +174,4 @@ paw login
149
174
 
150
175
  ## License
151
176
 
152
- MIT
177
+ MIT
@@ -28,21 +28,23 @@ fn = paw.function(
28
28
  )
29
29
  ```
30
30
 
31
- Loads a compiled program and returns a callable. Downloads the program and base model on first use; cached locally after that. Works offline after first download.
31
+ Loads a compiled program and returns a callable. Hub references download the
32
+ program and base model on first use; local `.paw` files supply the program
33
+ bundle directly. Required runtime metadata and base models are cached for reuse.
32
34
 
33
35
  | Parameter | Description |
34
36
  |-----------|-------------|
35
- | `program_id` | Required. A `Program` object, hash ID (e.g. `a6b454023d41ac9ca845`), slug (e.g. `da03/my-classifier`), or official shorthand (e.g. `email-triage`). A `Program` resolves by immutable `id`, not its mutable slug. |
37
+ | `program_id` | Required. A `Program` object, hash ID (e.g. `a6b454023d41ac9ca845`), slug (e.g. `da03/my-classifier`), official shorthand (e.g. `email-triage`), or local `.paw` path (see below). A `Program` resolves by immutable `id`, not its mutable slug. |
36
38
  | `n_ctx` | Context length for the local runtime (default `2048`). |
37
39
  | `n_gpu_layers` | GPU layers to offload (`0` = CPU-only, `-1` = all). The default is `-1`, or `PAW_GPU_LAYERS` when set. |
38
40
  | `verbose` | Enable verbose logging (default `False`). |
39
- | `offline` | Require all program/runtime/model assets to already be cached and make zero network calls. `PAW_OFFLINE=1` has the same effect. |
41
+ | `offline` | Use only local files/cache and make zero network calls; fail if required validated assets are missing. `PAW_OFFLINE=1` has the same effect. |
40
42
  | `interpreter` | Advanced adapter-free mode only. Must be passed by keyword and only when `program_id` is explicitly `None`. Supported values are `Qwen/Qwen3-0.6B` and `gpt2`. |
41
43
 
42
44
  The returned callable:
43
45
 
44
46
  ```python
45
- output: str = fn(input_text, max_tokens=None, temperature=0.0)
47
+ output: str = fn(input_text, max_tokens=None, temperature=0.0, logits_processor=None)
46
48
  ```
47
49
 
48
50
  | Parameter | Description |
@@ -50,11 +52,14 @@ output: str = fn(input_text, max_tokens=None, temperature=0.0)
50
52
  | `input_text` | Input string for the program. |
51
53
  | `max_tokens` | Maximum tokens to generate. `None` (default) = use all remaining context window. |
52
54
  | `temperature` | Sampling temperature (default `0.0`). |
55
+ | `logits_processor` | SDK 0.4.6+. Optional `llama_cpp.LogitsProcessorList` of caller-supplied processors, applied at every generation step. `None` (default) keeps sampling unchanged. |
56
+
57
+ This advanced hook is not built-in regex or JSON-schema validation. Each processor takes `(input_ids, scores)` and returns modified scores. Its token history includes the full prompt (including any compiled prefix and suffix or base-model template) plus generated tokens. Create or reset stateful processors for each call; processor exceptions propagate to the caller. Token limits and the usual output whitespace trimming still apply, so validate the returned result.
53
58
 
54
59
  **Context limits:** Spec + input + output share a ~2048 token window. Inputs that exceed it will error. `max_tokens` defaults to `None`: generation runs until EOS or the context limit.
55
60
 
56
61
  Compiled mode is strict: the adapter, prompt template, matching metadata,
57
- runtime manifest, and runtime-compatible base-model file must all validate. Version 0.4.4
62
+ runtime manifest, and runtime-compatible base-model file must all validate. Version 0.4.5
58
63
  accepts runtime manifest version 1 with `adapter_format="gguf_lora"`.
59
64
  Built-in models are checked against pinned size/SHA-256 metadata and GGUF
60
65
  magic. Historical manifests for those known runtime IDs are normalized to the
@@ -62,6 +67,45 @@ same canonical integrity metadata, so missing server-side checksum fields
62
67
  cannot weaken validation. Missing or failed adapters raise an error; the SDK
63
68
  never silently falls back to an unadapted base model.
64
69
 
70
+ ### Loading a local `.paw` file
71
+
72
+ Version 0.4.5 adds local-file inputs to `paw.function`:
73
+
74
+ ```python
75
+ from pathlib import Path
76
+
77
+ fn = paw.function(Path("classifier.paw"))
78
+ # With the required runtime metadata and base model already available locally:
79
+ fn = paw.function("./classifier.paw", offline=True)
80
+ ```
81
+
82
+ Use a current GGUF ZIP `.paw` bundle, such as one downloaded from a hosted
83
+ compile. It must contain `meta.json`, `adapter.gguf`, and `prompt_template.txt`,
84
+ with only `pseudo_program.txt` allowed as an optional extra; serialized native
85
+ prefix state is not accepted from archives. Local inputs are selected
86
+ deterministically: a `Path`/`os.PathLike`
87
+ object, an explicit path such as `./classifier.paw` or an absolute path, or a
88
+ string ending in `.paw` (case-insensitive). Ordinary IDs and slugs such as
89
+ `da03/my-classifier` keep their existing Hub behavior even if a matching local
90
+ file exists. Use `Path(...)` or an explicit path for a filename without the
91
+ `.paw` suffix. URL inputs are unsupported.
92
+
93
+ The bundle is validated and imported under
94
+ `PAW_CACHE_DIR/local_programs/<archive-sha256>` (default cache root:
95
+ `~/.cache/programasweights`). Its source file is unchanged, and its metadata
96
+ cannot replace a Hub program-ID or slug cache. Missing or invalid local files
97
+ raise an error without falling back to a Hub lookup or program download.
98
+
99
+ A local program does not necessarily make the first load fully offline:
100
+ the existing runtime policy may fetch required runtime metadata from PAW and
101
+ download the shared base model. Pass `offline=True` or set `PAW_OFFLINE=1`
102
+ to prohibit all network access. Historical `PAW\x02` tensor containers,
103
+ including output from the legacy `convert_peft_to_paw` module, are not supported
104
+ by this loader; it does not convert PEFT tensors to GGUF.
105
+
106
+ Only `paw.function` gains local-file inputs. `prepare_program` and
107
+ `is_offline_ready` continue to accept Hub program references.
108
+
65
109
  ### Advanced: adapter-free base interpreter
66
110
 
67
111
  Pass explicit `None` plus an interpreter to run the supported base GGUF without a compiled PAW program:
@@ -155,11 +199,24 @@ Compiles a natural language spec on the server. Returns a `Program` object.
155
199
  |-----------|-------------|
156
200
  | `id` | Hash-based program identifier. Use with `paw.function(program.id)`. |
157
201
  | `slug` | Full slug handle (e.g. `da03/my-classifier`) if one was created, `None` otherwise. |
158
- | `status` | `"ready"` on success, `"failed"` on error. |
202
+ | `status` | Status returned by the server, normally `"ready"` on success. HTTP errors raise `APIError` instead of returning a failed `Program`. |
159
203
  | `compiler_snapshot` | Exact compiler version used. |
160
204
  | `timings` | Timing metadata from the server. |
161
205
  | `error` | Error message when compilation fails. |
162
206
 
207
+ ### Compile timeouts
208
+
209
+ Synchronous `compile` uses `httpx.Timeout(120.0, read=2400.0)`: connect, write,
210
+ and pool waits remain 120 seconds; the read timeout is 2,400 seconds. This
211
+ allows for the origin's 1,900-second provider wait plus up to 330 seconds of
212
+ artifact finalization. It is a timeout while waiting for response data, **not a
213
+ 40-minute total deadline or guarantee**; upstream services may fail earlier.
214
+ The same setting applies to the compile step of `compile_and_load`.
215
+
216
+ Async submission retains a 30-second timeout. Precheck, status polling, and
217
+ cancellation each retain a 10-second timeout. For long finetunes, prefer the
218
+ explicit async workflow below so you retain a job ID for later status checks.
219
+
163
220
  ## Long-running compile jobs
164
221
 
165
222
  The asynchronous compile endpoint is available through both `PAWClient` and top-level helpers:
@@ -183,6 +240,37 @@ Status and cancellation requests must use the same authenticated account as
183
240
  submission. Anonymous jobs are bound to the validated client IP that submitted
184
241
  them.
185
242
 
243
+ ### Compile API errors
244
+
245
+ `compile`, `precheck_compile`, `compile_async`, `get_compile_status`, and
246
+ `cancel_compile` raise `paw.APIError` for HTTP 4xx/5xx responses. It is a subclass
247
+ of `httpx.HTTPStatusError`, so existing handlers continue to work. When supplied
248
+ by the server, `code`, `message`, and `request_id` are available as attributes
249
+ and included in the exception text. Missing fields are `None`; the original
250
+ `request` and `response` remain available, including response headers and body.
251
+
252
+ ```python
253
+ try:
254
+ job = paw.compile_async(SPEC, compiler="paw-ft-bs48")
255
+ except paw.APIError as error:
256
+ print(error.code, error.message, error.request_id)
257
+ # error.response.status_code and error.response.headers are unchanged.
258
+ raise
259
+ ```
260
+
261
+ For example, a `durable_queue_unavailable` 503 reports that durable Redis must
262
+ be healthy before async compilation can proceed. That rejection occurs before
263
+ the job is accepted; the caller can submit again after service recovery.
264
+ The SDK does not automatically retry compilation requests: other failures may
265
+ occur after a job has already been recorded. Invalid/non-JSON error bodies
266
+ retain the ordinary HTTP error description rather than displaying raw content.
267
+
268
+ Transport errors such as `httpx.ReadTimeout` propagate unchanged, rather than
269
+ becoming `APIError`. A timeout does not prove the server rejected or cancelled
270
+ the work, so the SDK does not automatically resubmit it. Both `paw.compile`
271
+ and `paw.compile_and_load` propagate these errors; `compile_and_load` does not
272
+ attempt to load a function when compilation raises.
273
+
186
274
  ## `paw.compile_and_load`
187
275
 
188
276
  ```python
@@ -136,7 +136,16 @@ List available compiler models and identifiers for use with compile requests.
136
136
 
137
137
  ### `GET /health`
138
138
 
139
- Liveness or readiness style health check for the API service.
139
+ Returns HTTP 200 for API liveness. The JSON `status` is `healthy` only when
140
+ all enabled public compilers pass their provider checks; otherwise it is
141
+ `degraded`, with details in `warnings` and `gpu_services`, keyed by compiler
142
+ name. Finetune checks include its base compiler, durable Redis, and at least
143
+ one healthy worker. A healthy worker remains available while busy.
144
+
145
+ Checks run concurrently with a three-second timeout and share a five-second
146
+ cache. `queue_depth` counts waiting finetune jobs across distinct dispatchers,
147
+ not running jobs; it is `null` when a queue cannot be verified. This is a
148
+ readiness observation, not a guarantee that a new compilation will succeed.
140
149
 
141
150
  ## Errors
142
151
 
@@ -27,7 +27,7 @@ try:
27
27
  from importlib.metadata import version as _meta_version
28
28
  __version__ = _meta_version("programasweights")
29
29
  except Exception:
30
- __version__ = "0.4.4"
30
+ __version__ = "0.4.6"
31
31
 
32
32
  from ._output import ProgressCallback, ProgressEvent, report_progress
33
33
  from .cache import CachedProgram
@@ -39,6 +39,7 @@ from .client import (
39
39
  Program,
40
40
  )
41
41
  from .config import get_api_url, get_api_key, set_api_key
42
+ from .errors import APIError
42
43
 
43
44
 
44
45
  def compile(
@@ -400,18 +401,22 @@ def function(
400
401
  ):
401
402
  """Load a compiled program, or explicitly load a bare base interpreter.
402
403
 
403
- Downloads the .paw bundle and base model GGUF on first use.
404
- Subsequent calls use the local cache.
404
+ Hub references download the .paw bundle on first use; local paths supply
405
+ it directly. Required runtime metadata and base models may still download
406
+ unless offline mode is enabled. Subsequent calls reuse validated caches.
405
407
 
406
408
  Args:
407
409
  program_id: Program ID (str), slug (``da03/my-program``), pinned version
408
- (``da03/my-program@v3``), or a ``Program`` object from compile().
410
+ (``da03/my-program@v3``), a ``Program`` object from compile(), or a
411
+ local GGUF-based .paw bundle. PathLike objects and explicit path
412
+ strings (including strings ending in .paw) select local files.
409
413
  n_ctx: Context window size for llama.cpp.
410
414
  n_gpu_layers: GPU layers (-1 = all GPU, 0 = CPU only). Defaults to -1
411
415
  (auto-uses Metal/CUDA if available, safe fallback to CPU).
412
416
  Set ``PAW_GPU_LAYERS=0`` env var to force CPU-only.
413
417
  verbose: Print llama.cpp debug output.
414
- offline: Skip server check for slug resolution and use local cache only.
418
+ offline: Prohibit network access. Local bundles may be imported, but
419
+ their runtime and base model must already be available locally.
415
420
  Also set via ``PAW_OFFLINE=1`` env var.
416
421
  interpreter: Advanced adapter-free mode. This is only valid when
417
422
  ``program_id`` is explicitly ``None``. Initially supported values
@@ -427,6 +432,8 @@ def function(
427
432
 
428
433
  >>> fn = paw.function("da03/my-program@v2") # pinned version
429
434
 
435
+ >>> fn = paw.function("./classifier.paw") # local bundle
436
+
430
437
  >>> base = paw.function(None, interpreter="gpt2")
431
438
  """
432
439
  import os
@@ -452,7 +459,12 @@ def function(
452
459
  offline=offline,
453
460
  )
454
461
 
455
- program_reference = _coerce_program_reference(program_id)
462
+ from ._program_reference import local_program_path
463
+
464
+ local_path = local_program_path(program_id)
465
+ program_reference = (
466
+ _coerce_program_reference(program_id) if local_path is None else None
467
+ )
456
468
  if program_reference == "":
457
469
  raise ValueError(
458
470
  "program_id cannot be an empty string; pass explicit None with "
@@ -464,25 +476,35 @@ def function(
464
476
  "program_id=None to request adapter-free base mode."
465
477
  )
466
478
 
467
- from .runtime_llamacpp import PawFunction
479
+ if local_path is not None:
480
+ from .local_program import import_local_program
468
481
 
469
- resolved_id = _resolve_program_id(program_reference, offline=offline)
470
- if offline and not cache.has_valid_program_assets(resolved_id):
471
- raise RuntimeError(
472
- f"Program {resolved_id} is not fully cached; offline mode "
473
- "prohibits program downloads."
474
- )
475
- if not cache.has_valid_program_assets(resolved_id):
476
- from .client import PAWClient
482
+ # Validate explicit local input before importing the native runtime.
483
+ # A missing/corrupt file is never retried as a Hub ID or slug.
484
+ program_dir = import_local_program(local_path)
485
+ from .runtime_llamacpp import PawFunction
486
+ else:
487
+ # Preserve the existing Hub path's fail-fast dependency check before
488
+ # resolving slugs or downloading assets.
489
+ from .runtime_llamacpp import PawFunction
477
490
 
478
- client = PAWClient(api_url=get_api_url(), api_key=get_api_key())
479
- client.download_paw(resolved_id)
480
- if not cache.has_valid_program_assets(resolved_id):
481
- raise RuntimeError(
482
- f"Program {resolved_id} is missing valid compiled assets."
483
- )
491
+ resolved_id = _resolve_program_id(program_reference, offline=offline)
492
+ if offline and not cache.has_valid_program_assets(resolved_id):
493
+ raise RuntimeError(
494
+ f"Program {resolved_id} is not fully cached; offline mode "
495
+ "prohibits program downloads."
496
+ )
497
+ if not cache.has_valid_program_assets(resolved_id):
498
+ from .client import PAWClient
499
+
500
+ client = PAWClient(api_url=get_api_url(), api_key=get_api_key())
501
+ client.download_paw(resolved_id)
502
+ if not cache.has_valid_program_assets(resolved_id):
503
+ raise RuntimeError(
504
+ f"Program {resolved_id} is missing valid compiled assets."
505
+ )
506
+ program_dir = cache.get_program_dir(resolved_id)
484
507
 
485
- program_dir = cache.get_program_dir(resolved_id)
486
508
  return PawFunction(
487
509
  program_dir,
488
510
  n_ctx=n_ctx,
@@ -616,6 +638,7 @@ def list_compilers() -> list[dict]:
616
638
 
617
639
 
618
640
  __all__ = [
641
+ "APIError",
619
642
  "CachedProgram",
620
643
  "CompileCancellation",
621
644
  "CompileJob",
@@ -0,0 +1,61 @@
1
+ """Deterministic local-file intent, independent of filesystem contents."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import re
7
+ from pathlib import Path
8
+
9
+
10
+ _WINDOWS_DRIVE = re.compile(r"^[A-Za-z]:")
11
+ _URI_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:")
12
+
13
+
14
+ def local_program_path(reference: object) -> Path | None:
15
+ """Return an explicit local path, or None for an ID/slug/Program.
16
+
17
+ Path-like objects always mean files. Strings select files by syntax, never
18
+ by existence; in particular, an ``owner/slug`` remains a Hub reference.
19
+ """
20
+ is_pathlike = isinstance(reference, os.PathLike)
21
+ if is_pathlike:
22
+ value = os.fspath(reference)
23
+ if not isinstance(value, str):
24
+ raise TypeError("Local program paths must be text, not bytes.")
25
+ elif isinstance(reference, str):
26
+ value = reference
27
+ else:
28
+ return None
29
+
30
+ if "\x00" in value:
31
+ raise ValueError("Program references cannot contain null bytes.")
32
+ windows_drive = bool(_WINDOWS_DRIVE.match(value))
33
+ if not is_pathlike and not windows_drive and _URI_SCHEME.match(value):
34
+ raise ValueError(
35
+ "Program URLs are not supported. Download the .paw bundle first "
36
+ "and pass its local path."
37
+ )
38
+
39
+ explicit_path = value.startswith(
40
+ ("./", "../", "/", "~/", "\\", ".\\", "..\\", "~\\")
41
+ )
42
+ if not (
43
+ is_pathlike or windows_drive or explicit_path
44
+ or value.lower().endswith(".paw")
45
+ ):
46
+ return None
47
+ if not value:
48
+ raise ValueError("Local program path cannot be empty.")
49
+
50
+ # A Windows-looking string must never be sent to the Hub on another OS.
51
+ # Path objects, however, explicitly name a native path and may contain
52
+ # characters that would have a different meaning on another platform.
53
+ windows_path = windows_drive or value.startswith(
54
+ ("\\", ".\\", "..\\", "~\\")
55
+ )
56
+ if not is_pathlike and os.name != "nt" and windows_path:
57
+ raise ValueError(
58
+ "This is a Windows filesystem path, which cannot be opened on "
59
+ "this platform. Pass a native local path instead."
60
+ )
61
+ return Path(value).expanduser()