programasweights 0.4.4__tar.gz → 0.4.6__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- programasweights-0.4.6/.github/workflows/test.yml +82 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/AGENTS.md +6 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/CHANGELOG.md +22 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/PKG-INFO +28 -3
- {programasweights-0.4.4 → programasweights-0.4.6}/PYPI_README.md +26 -1
- {programasweights-0.4.4 → programasweights-0.4.6}/README.md +26 -1
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/api-reference/python-sdk.md +94 -6
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/api-reference/rest-api.md +10 -1
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/__init__.py +45 -22
- programasweights-0.4.6/programasweights/_program_reference.py +61 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/client.py +11 -7
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/convert_peft_to_paw.py +12 -14
- programasweights-0.4.6/programasweights/errors.py +79 -0
- programasweights-0.4.6/programasweights/local_program.py +196 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime_llamacpp.py +38 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/pyproject.toml +1 -1
- programasweights-0.4.6/tests/test_api_errors.py +222 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_base_interpreter.py +204 -2
- programasweights-0.4.6/tests/test_compile_timeouts.py +129 -0
- programasweights-0.4.6/tests/test_local_program.py +676 -0
- programasweights-0.4.4/.github/workflows/test.yml +0 -43
- {programasweights-0.4.4 → programasweights-0.4.6}/.gitignore +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/.readthedocs.yaml +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/LICENSE +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/001-llama-cpp-over-pytorch.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/002-q4_0-adapter-format.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/003-single-spec-field.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/004-compiler-naming.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/005-vllm-hidden-states.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/adr/006-email-api-key-auth.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/advanced/adrs.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/advanced/architecture.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/api-reference/cli.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/architecture.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/alien-taboo.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/log-monitoring.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/semantic-search.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/site-navigation.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/case-studies/tool-calling.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/getting-started/first-program.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/getting-started/installation.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/getting-started/naming-programs.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/browser-inference.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/how-it-works.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/local-inference.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/guide/writing-good-specs.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/hub/browsing-programs.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/hub/feedback-cases.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/hub/publishing-programs.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/index.md +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/docs/requirements.txt +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/examples/flask_app.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/examples/jupyter_notebook.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/examples/langchain_integration.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/examples/paw_monitor.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/examples/replace_openai.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/mkdocs.yml +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/_output.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/artifacts.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/cache.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/cli.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/compiler/__init__.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/compiler/dummy.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/config.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/paw_format.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime/__init__.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime/interpreter.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/programasweights/runtime/interpreter_onnx.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_cli_auth.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_desktop_sdk.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_offline_cache.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_runtime_registry_sdk.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_sdk.py +0 -0
- {programasweights-0.4.4 → programasweights-0.4.6}/tests/test_sdk.sh +0 -0
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
name: tests
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
test:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
strategy:
|
|
12
|
+
fail-fast: false
|
|
13
|
+
matrix:
|
|
14
|
+
python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
|
|
18
|
+
- name: Set up Python ${{ matrix.python-version }}
|
|
19
|
+
uses: actions/setup-python@v5
|
|
20
|
+
with:
|
|
21
|
+
python-version: ${{ matrix.python-version }}
|
|
22
|
+
|
|
23
|
+
- name: Install (hermetic deps only)
|
|
24
|
+
# Install httpx + pytest and the package itself without pulling the heavy
|
|
25
|
+
# llama-cpp-python build. Runtime tests inject a fake llama_cpp module,
|
|
26
|
+
# so CI needs neither the native extension nor a model download.
|
|
27
|
+
run: |
|
|
28
|
+
python -m pip install --upgrade pip
|
|
29
|
+
python -m pip install httpx pytest
|
|
30
|
+
python -m pip install -e . --no-deps
|
|
31
|
+
|
|
32
|
+
- name: Run hermetic tests
|
|
33
|
+
# Scoped to tests that need no network, no model download, and no
|
|
34
|
+
# PAW_API_KEY. Auth tests (@needs_auth) auto-skip without a key; the
|
|
35
|
+
# network/model-download tests in test_sdk.py are excluded here and can
|
|
36
|
+
# be run separately against a live server.
|
|
37
|
+
run: |
|
|
38
|
+
pytest \
|
|
39
|
+
tests/test_api_errors.py \
|
|
40
|
+
tests/test_compile_timeouts.py \
|
|
41
|
+
tests/test_local_program.py \
|
|
42
|
+
tests/test_base_interpreter.py \
|
|
43
|
+
tests/test_cli_auth.py \
|
|
44
|
+
tests/test_desktop_sdk.py \
|
|
45
|
+
tests/test_runtime_registry_sdk.py \
|
|
46
|
+
tests/test_sdk.py::TestInstallAndImport
|
|
47
|
+
|
|
48
|
+
local-files-windows:
|
|
49
|
+
runs-on: windows-latest
|
|
50
|
+
strategy:
|
|
51
|
+
fail-fast: false
|
|
52
|
+
matrix:
|
|
53
|
+
python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
|
|
54
|
+
steps:
|
|
55
|
+
- uses: actions/checkout@v4
|
|
56
|
+
- uses: actions/setup-python@v5
|
|
57
|
+
with:
|
|
58
|
+
python-version: ${{ matrix.python-version }}
|
|
59
|
+
- name: Install hermetic test dependencies
|
|
60
|
+
run: |
|
|
61
|
+
python -m pip install httpx pytest
|
|
62
|
+
python -m pip install -e . --no-deps
|
|
63
|
+
- name: Test Windows local paths, cache locks, and compile errors
|
|
64
|
+
run: python -m pytest tests/test_api_errors.py tests/test_compile_timeouts.py tests/test_local_program.py --junitxml=test-results.xml
|
|
65
|
+
- name: Annotate Windows test failures
|
|
66
|
+
if: failure()
|
|
67
|
+
shell: python
|
|
68
|
+
run: |
|
|
69
|
+
from pathlib import Path
|
|
70
|
+
import xml.etree.ElementTree as ET
|
|
71
|
+
|
|
72
|
+
report = Path("test-results.xml")
|
|
73
|
+
if report.exists():
|
|
74
|
+
for case in ET.parse(report).iter("testcase"):
|
|
75
|
+
for result in case:
|
|
76
|
+
if result.tag in {"failure", "error"}:
|
|
77
|
+
# GitHub truncates annotations, so retain the actual
|
|
78
|
+
# exception at the end of long pytest tracebacks.
|
|
79
|
+
detail = result.text or result.get("message", "")
|
|
80
|
+
message = f"{case.get('name')}: {detail[-3500:]}"
|
|
81
|
+
message = message.replace("%", "%25").replace("\r", "%0D").replace("\n", "%0A")
|
|
82
|
+
print(f"::error::{message}")
|
|
@@ -43,6 +43,8 @@ fn = paw.compile_and_load("Classify sentiment as positive or negative")
|
|
|
43
43
|
fn("I love this!") # "positive"
|
|
44
44
|
```
|
|
45
45
|
|
|
46
|
+
Load a local `.paw` file with `paw.function("./classifier.paw")` (SDK 0.4.5+).
|
|
47
|
+
|
|
46
48
|
If you want the smaller browser-compatible runtime explicitly, pass `compiler="paw-4b-gpt2"`. Otherwise, omit `compiler` and let the server default decide.
|
|
47
49
|
|
|
48
50
|
## Current Public Compilers
|
|
@@ -89,6 +91,8 @@ Output: delete
|
|
|
89
91
|
- Spec + input + output share a ~2048 token context window. Inputs that exceed it will error.
|
|
90
92
|
- `max_tokens` defaults to `None`: generation runs until EOS or the context limit.
|
|
91
93
|
- Compile runs on the hosted PAW API. Inference should usually run locally through the SDK.
|
|
94
|
+
- Synchronous compile requests use a 40-minute read timeout.
|
|
95
|
+
- **Run local inference sequentially by default.** With PAW’s current llama.cpp backend, simultaneous inference calls often perform worse. Reuse loaded functions and process inputs one at a time; never call the same function instance concurrently.
|
|
92
96
|
- **GPU acceleration** is enabled by default (`n_gpu_layers=-1`). Uses Metal on Mac, CUDA on Linux, and falls back to CPU automatically. If GPU causes issues, set `PAW_GPU_LAYERS=0` or pass `n_gpu_layers=0`.
|
|
93
97
|
- **First call** is usually ~1-5s because it loads the base model. Subsequent calls are typically ~0.05-0.5s depending on input length and GPU availability.
|
|
94
98
|
- **Base model files are shared** across programs on disk. Each Standard LoRA adapter is ~22 MB; each Compact LoRA adapter is ~5 MB.
|
|
@@ -102,6 +106,8 @@ Output: delete
|
|
|
102
106
|
|
|
103
107
|
## Common Errors
|
|
104
108
|
|
|
109
|
+
Compile API HTTP errors raise `paw.APIError`. Check `error.code` and `error.message` for details. The SDK does not retry automatically.
|
|
110
|
+
|
|
105
111
|
| Error | Cause | Fix |
|
|
106
112
|
|-------|-------|-----|
|
|
107
113
|
| `RuntimeError: assets not ready` on download | Program is still generating after compile | The SDK polls automatically for up to 60s. If it still fails, retry shortly or recompile. |
|
|
@@ -1,5 +1,27 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.4.6 (2026-09-13)
|
|
4
|
+
|
|
5
|
+
- Add an optional `logits_processor` argument to a compiled or base program
|
|
6
|
+
call for caller-supplied token constraints in llama.cpp's sampler.
|
|
7
|
+
Defaults to `None`; sampling is unchanged when it is unset.
|
|
8
|
+
- Propagate processor failures to the caller instead of letting native
|
|
9
|
+
callback errors silently continue with unconstrained output.
|
|
10
|
+
|
|
11
|
+
## 0.4.5 (2026-09-10)
|
|
12
|
+
|
|
13
|
+
- Expose structured compile API failures as `paw.APIError`, compatible with
|
|
14
|
+
`httpx.HTTPStatusError`, preserving server code, message, request ID, and the
|
|
15
|
+
original response. Transport errors propagate unchanged; no automatic retries.
|
|
16
|
+
- Allow synchronous compilation a 2,400-second read timeout while keeping
|
|
17
|
+
connect, write, and pool timeouts at 120 seconds. Async submission stays at
|
|
18
|
+
30 seconds; precheck, status, and cancellation stay at 10 seconds.
|
|
19
|
+
- Load current GGUF ZIP `.paw` files through `paw.function` using explicit
|
|
20
|
+
local paths or `Path` objects. Validated imports use a separate SHA-256 cache
|
|
21
|
+
without replacing Hub caches or source files; invalid local files never fall
|
|
22
|
+
back to Hub lookup. Shared base-model assets may still need downloading unless
|
|
23
|
+
offline mode is requested. Legacy tensor-format `.paw` files are unsupported.
|
|
24
|
+
|
|
3
25
|
## 0.4.4 (2026-07-18)
|
|
4
26
|
|
|
5
27
|
- Add desktop preparation and cache inspection APIs with structured progress:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: programasweights
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.6
|
|
4
4
|
Summary: Compile natural language specifications into neural programs that run locally via llama.cpp.
|
|
5
5
|
Project-URL: Homepage, https://programasweights.com
|
|
6
6
|
Project-URL: Repository, https://github.com/programasweights/programasweights-python
|
|
@@ -84,6 +84,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
|
|
|
84
84
|
|
|
85
85
|
GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
|
|
86
86
|
|
|
87
|
+
## Constrained Decoding
|
|
88
|
+
|
|
89
|
+
In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
import llama_cpp
|
|
93
|
+
# my_processor is your compatible callable: (input_ids, scores) -> scores.
|
|
94
|
+
fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
|
|
98
|
+
|
|
87
99
|
## Desktop and Offline Workflows
|
|
88
100
|
|
|
89
101
|
Prepare and inspect validated local assets without keeping a model loaded:
|
|
@@ -102,6 +114,19 @@ finetune compiles can be queued with
|
|
|
102
114
|
`paw.compile_async(spec, compiler="paw-ft-bs48")`; an explicit finetune
|
|
103
115
|
compiler is required.
|
|
104
116
|
|
|
117
|
+
Load a saved current GGUF ZIP `.paw` bundle directly (SDK 0.4.5+):
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from pathlib import Path
|
|
121
|
+
fn = paw.function(Path("classifier.paw"))
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Local files are validated into a separate SHA-256 cache without changing the
|
|
125
|
+
source or falling back to Hub lookup. Runtime metadata or the shared base model
|
|
126
|
+
may still download; `offline=True` prohibits those requests. Legacy tensor-format
|
|
127
|
+
`.paw` files are unsupported. Local-file inputs are supported by `function`,
|
|
128
|
+
not `prepare_program` or `is_offline_ready`.
|
|
129
|
+
|
|
105
130
|
Advanced adapter-free inference is available with
|
|
106
131
|
`paw.function(None, interpreter="gpt2")`; see the Python API reference for
|
|
107
132
|
its intentionally strict semantics.
|
|
@@ -179,4 +204,4 @@ paw login
|
|
|
179
204
|
|
|
180
205
|
## License
|
|
181
206
|
|
|
182
|
-
MIT
|
|
207
|
+
MIT
|
|
@@ -53,6 +53,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
|
|
|
53
53
|
|
|
54
54
|
GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
|
|
55
55
|
|
|
56
|
+
## Constrained Decoding
|
|
57
|
+
|
|
58
|
+
In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
import llama_cpp
|
|
62
|
+
# my_processor is your compatible callable: (input_ids, scores) -> scores.
|
|
63
|
+
fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
|
|
67
|
+
|
|
56
68
|
## Desktop and Offline Workflows
|
|
57
69
|
|
|
58
70
|
Prepare and inspect validated local assets without keeping a model loaded:
|
|
@@ -71,6 +83,19 @@ finetune compiles can be queued with
|
|
|
71
83
|
`paw.compile_async(spec, compiler="paw-ft-bs48")`; an explicit finetune
|
|
72
84
|
compiler is required.
|
|
73
85
|
|
|
86
|
+
Load a saved current GGUF ZIP `.paw` bundle directly (SDK 0.4.5+):
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from pathlib import Path
|
|
90
|
+
fn = paw.function(Path("classifier.paw"))
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Local files are validated into a separate SHA-256 cache without changing the
|
|
94
|
+
source or falling back to Hub lookup. Runtime metadata or the shared base model
|
|
95
|
+
may still download; `offline=True` prohibits those requests. Legacy tensor-format
|
|
96
|
+
`.paw` files are unsupported. Local-file inputs are supported by `function`,
|
|
97
|
+
not `prepare_program` or `is_offline_ready`.
|
|
98
|
+
|
|
74
99
|
Advanced adapter-free inference is available with
|
|
75
100
|
`paw.function(None, interpreter="gpt2")`; see the Python API reference for
|
|
76
101
|
its intentionally strict semantics.
|
|
@@ -148,4 +173,4 @@ paw login
|
|
|
148
173
|
|
|
149
174
|
## License
|
|
150
175
|
|
|
151
|
-
MIT
|
|
176
|
+
MIT
|
|
@@ -53,6 +53,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
|
|
|
53
53
|
|
|
54
54
|
GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
|
|
55
55
|
|
|
56
|
+
## Constrained Decoding
|
|
57
|
+
|
|
58
|
+
In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
|
|
59
|
+
|
|
60
|
+
```python
|
|
61
|
+
import llama_cpp
|
|
62
|
+
# my_processor is your compatible callable: (input_ids, scores) -> scores.
|
|
63
|
+
fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
|
|
67
|
+
|
|
56
68
|
## Desktop and Offline Workflows
|
|
57
69
|
|
|
58
70
|
Prepare and inspect validated local assets without keeping a model loaded:
|
|
@@ -71,6 +83,19 @@ finetune compiles can be queued with
|
|
|
71
83
|
`paw.compile_async(spec, compiler="paw-ft-bs48")`; an explicit finetune
|
|
72
84
|
compiler is required.
|
|
73
85
|
|
|
86
|
+
Load a saved current GGUF ZIP `.paw` bundle directly (SDK 0.4.5+):
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from pathlib import Path
|
|
90
|
+
fn = paw.function(Path("classifier.paw"))
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Local files are validated into a separate SHA-256 cache without changing the
|
|
94
|
+
source or falling back to Hub lookup. Runtime metadata or the shared base model
|
|
95
|
+
may still download; `offline=True` prohibits those requests. Legacy tensor-format
|
|
96
|
+
`.paw` files are unsupported. Local-file inputs are supported by `function`,
|
|
97
|
+
not `prepare_program` or `is_offline_ready`.
|
|
98
|
+
|
|
74
99
|
Advanced adapter-free inference is available with
|
|
75
100
|
`paw.function(None, interpreter="gpt2")`; see the
|
|
76
101
|
[Python API reference](docs/api-reference/python-sdk.md#advanced-adapter-free-base-interpreter)
|
|
@@ -149,4 +174,4 @@ paw login
|
|
|
149
174
|
|
|
150
175
|
## License
|
|
151
176
|
|
|
152
|
-
MIT
|
|
177
|
+
MIT
|
|
@@ -28,21 +28,23 @@ fn = paw.function(
|
|
|
28
28
|
)
|
|
29
29
|
```
|
|
30
30
|
|
|
31
|
-
Loads a compiled program and returns a callable.
|
|
31
|
+
Loads a compiled program and returns a callable. Hub references download the
|
|
32
|
+
program and base model on first use; local `.paw` files supply the program
|
|
33
|
+
bundle directly. Required runtime metadata and base models are cached for reuse.
|
|
32
34
|
|
|
33
35
|
| Parameter | Description |
|
|
34
36
|
|-----------|-------------|
|
|
35
|
-
| `program_id` | Required. A `Program` object, hash ID (e.g. `a6b454023d41ac9ca845`), slug (e.g. `da03/my-classifier`),
|
|
37
|
+
| `program_id` | Required. A `Program` object, hash ID (e.g. `a6b454023d41ac9ca845`), slug (e.g. `da03/my-classifier`), official shorthand (e.g. `email-triage`), or local `.paw` path (see below). A `Program` resolves by immutable `id`, not its mutable slug. |
|
|
36
38
|
| `n_ctx` | Context length for the local runtime (default `2048`). |
|
|
37
39
|
| `n_gpu_layers` | GPU layers to offload (`0` = CPU-only, `-1` = all). The default is `-1`, or `PAW_GPU_LAYERS` when set. |
|
|
38
40
|
| `verbose` | Enable verbose logging (default `False`). |
|
|
39
|
-
| `offline` |
|
|
41
|
+
| `offline` | Use only local files/cache and make zero network calls; fail if required validated assets are missing. `PAW_OFFLINE=1` has the same effect. |
|
|
40
42
|
| `interpreter` | Advanced adapter-free mode only. Must be passed by keyword and only when `program_id` is explicitly `None`. Supported values are `Qwen/Qwen3-0.6B` and `gpt2`. |
|
|
41
43
|
|
|
42
44
|
The returned callable:
|
|
43
45
|
|
|
44
46
|
```python
|
|
45
|
-
output: str = fn(input_text, max_tokens=None, temperature=0.0)
|
|
47
|
+
output: str = fn(input_text, max_tokens=None, temperature=0.0, logits_processor=None)
|
|
46
48
|
```
|
|
47
49
|
|
|
48
50
|
| Parameter | Description |
|
|
@@ -50,11 +52,14 @@ output: str = fn(input_text, max_tokens=None, temperature=0.0)
|
|
|
50
52
|
| `input_text` | Input string for the program. |
|
|
51
53
|
| `max_tokens` | Maximum tokens to generate. `None` (default) = use all remaining context window. |
|
|
52
54
|
| `temperature` | Sampling temperature (default `0.0`). |
|
|
55
|
+
| `logits_processor` | SDK 0.4.6+. Optional `llama_cpp.LogitsProcessorList` of caller-supplied processors, applied at every generation step. `None` (default) keeps sampling unchanged. |
|
|
56
|
+
|
|
57
|
+
This advanced hook is not built-in regex or JSON-schema validation. Each processor takes `(input_ids, scores)` and returns modified scores. Its token history includes the full prompt (including any compiled prefix and suffix or base-model template) plus generated tokens. Create or reset stateful processors for each call; processor exceptions propagate to the caller. Token limits and the usual output whitespace trimming still apply, so validate the returned result.
|
|
53
58
|
|
|
54
59
|
**Context limits:** Spec + input + output share a ~2048 token window. Inputs that exceed it will error. `max_tokens` defaults to `None`: generation runs until EOS or the context limit.
|
|
55
60
|
|
|
56
61
|
Compiled mode is strict: the adapter, prompt template, matching metadata,
|
|
57
|
-
runtime manifest, and runtime-compatible base-model file must all validate. Version 0.4.
|
|
62
|
+
runtime manifest, and runtime-compatible base-model file must all validate. Version 0.4.5
|
|
58
63
|
accepts runtime manifest version 1 with `adapter_format="gguf_lora"`.
|
|
59
64
|
Built-in models are checked against pinned size/SHA-256 metadata and GGUF
|
|
60
65
|
magic. Historical manifests for those known runtime IDs are normalized to the
|
|
@@ -62,6 +67,45 @@ same canonical integrity metadata, so missing server-side checksum fields
|
|
|
62
67
|
cannot weaken validation. Missing or failed adapters raise an error; the SDK
|
|
63
68
|
never silently falls back to an unadapted base model.
|
|
64
69
|
|
|
70
|
+
### Loading a local `.paw` file
|
|
71
|
+
|
|
72
|
+
Version 0.4.5 adds local-file inputs to `paw.function`:
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from pathlib import Path
|
|
76
|
+
|
|
77
|
+
fn = paw.function(Path("classifier.paw"))
|
|
78
|
+
# With the required runtime metadata and base model already available locally:
|
|
79
|
+
fn = paw.function("./classifier.paw", offline=True)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Use a current GGUF ZIP `.paw` bundle, such as one downloaded from a hosted
|
|
83
|
+
compile. It must contain `meta.json`, `adapter.gguf`, and `prompt_template.txt`,
|
|
84
|
+
with only `pseudo_program.txt` allowed as an optional extra; serialized native
|
|
85
|
+
prefix state is not accepted from archives. Local inputs are selected
|
|
86
|
+
deterministically: a `Path`/`os.PathLike`
|
|
87
|
+
object, an explicit path such as `./classifier.paw` or an absolute path, or a
|
|
88
|
+
string ending in `.paw` (case-insensitive). Ordinary IDs and slugs such as
|
|
89
|
+
`da03/my-classifier` keep their existing Hub behavior even if a matching local
|
|
90
|
+
file exists. Use `Path(...)` or an explicit path for a filename without the
|
|
91
|
+
`.paw` suffix. URL inputs are unsupported.
|
|
92
|
+
|
|
93
|
+
The bundle is validated and imported under
|
|
94
|
+
`PAW_CACHE_DIR/local_programs/<archive-sha256>` (default cache root:
|
|
95
|
+
`~/.cache/programasweights`). Its source file is unchanged, and its metadata
|
|
96
|
+
cannot replace a Hub program-ID or slug cache. Missing or invalid local files
|
|
97
|
+
raise an error without falling back to a Hub lookup or program download.
|
|
98
|
+
|
|
99
|
+
A local program does not necessarily make the first load fully offline:
|
|
100
|
+
the existing runtime policy may fetch required runtime metadata from PAW and
|
|
101
|
+
download the shared base model. Pass `offline=True` or set `PAW_OFFLINE=1`
|
|
102
|
+
to prohibit all network access. Historical `PAW\x02` tensor containers,
|
|
103
|
+
including output from the legacy `convert_peft_to_paw` module, are not supported
|
|
104
|
+
by this loader; it does not convert PEFT tensors to GGUF.
|
|
105
|
+
|
|
106
|
+
Only `paw.function` gains local-file inputs. `prepare_program` and
|
|
107
|
+
`is_offline_ready` continue to accept Hub program references.
|
|
108
|
+
|
|
65
109
|
### Advanced: adapter-free base interpreter
|
|
66
110
|
|
|
67
111
|
Pass explicit `None` plus an interpreter to run the supported base GGUF without a compiled PAW program:
|
|
@@ -155,11 +199,24 @@ Compiles a natural language spec on the server. Returns a `Program` object.
|
|
|
155
199
|
|-----------|-------------|
|
|
156
200
|
| `id` | Hash-based program identifier. Use with `paw.function(program.id)`. |
|
|
157
201
|
| `slug` | Full slug handle (e.g. `da03/my-classifier`) if one was created, `None` otherwise. |
|
|
158
|
-
| `status` | `"ready"` on success
|
|
202
|
+
| `status` | Status returned by the server, normally `"ready"` on success. HTTP errors raise `APIError` instead of returning a failed `Program`. |
|
|
159
203
|
| `compiler_snapshot` | Exact compiler version used. |
|
|
160
204
|
| `timings` | Timing metadata from the server. |
|
|
161
205
|
| `error` | Error message when compilation fails. |
|
|
162
206
|
|
|
207
|
+
### Compile timeouts
|
|
208
|
+
|
|
209
|
+
Synchronous `compile` uses `httpx.Timeout(120.0, read=2400.0)`: connect, write,
|
|
210
|
+
and pool waits remain 120 seconds; the read timeout is 2,400 seconds. This
|
|
211
|
+
allows for the origin's 1,900-second provider wait plus up to 330 seconds of
|
|
212
|
+
artifact finalization. It is a timeout while waiting for response data, **not a
|
|
213
|
+
40-minute total deadline or guarantee**; upstream services may fail earlier.
|
|
214
|
+
The same setting applies to the compile step of `compile_and_load`.
|
|
215
|
+
|
|
216
|
+
Async submission retains a 30-second timeout. Precheck, status polling, and
|
|
217
|
+
cancellation each retain a 10-second timeout. For long finetunes, prefer the
|
|
218
|
+
explicit async workflow below so you retain a job ID for later status checks.
|
|
219
|
+
|
|
163
220
|
## Long-running compile jobs
|
|
164
221
|
|
|
165
222
|
The asynchronous compile endpoint is available through both `PAWClient` and top-level helpers:
|
|
@@ -183,6 +240,37 @@ Status and cancellation requests must use the same authenticated account as
|
|
|
183
240
|
submission. Anonymous jobs are bound to the validated client IP that submitted
|
|
184
241
|
them.
|
|
185
242
|
|
|
243
|
+
### Compile API errors
|
|
244
|
+
|
|
245
|
+
`compile`, `precheck_compile`, `compile_async`, `get_compile_status`, and
|
|
246
|
+
`cancel_compile` raise `paw.APIError` for HTTP 4xx/5xx responses. It is a subclass
|
|
247
|
+
of `httpx.HTTPStatusError`, so existing handlers continue to work. When supplied
|
|
248
|
+
by the server, `code`, `message`, and `request_id` are available as attributes
|
|
249
|
+
and included in the exception text. Missing fields are `None`; the original
|
|
250
|
+
`request` and `response` remain available, including response headers and body.
|
|
251
|
+
|
|
252
|
+
```python
|
|
253
|
+
try:
|
|
254
|
+
job = paw.compile_async(SPEC, compiler="paw-ft-bs48")
|
|
255
|
+
except paw.APIError as error:
|
|
256
|
+
print(error.code, error.message, error.request_id)
|
|
257
|
+
# error.response.status_code and error.response.headers are unchanged.
|
|
258
|
+
raise
|
|
259
|
+
```
|
|
260
|
+
|
|
261
|
+
For example, a `durable_queue_unavailable` 503 reports that durable Redis must
|
|
262
|
+
be healthy before async compilation can proceed. That rejection occurs before
|
|
263
|
+
the job is accepted; the caller can submit again after service recovery.
|
|
264
|
+
The SDK does not automatically retry compilation requests: other failures may
|
|
265
|
+
occur after a job has already been recorded. Invalid/non-JSON error bodies
|
|
266
|
+
retain the ordinary HTTP error description rather than displaying raw content.
|
|
267
|
+
|
|
268
|
+
Transport errors such as `httpx.ReadTimeout` propagate unchanged, rather than
|
|
269
|
+
becoming `APIError`. A timeout does not prove the server rejected or cancelled
|
|
270
|
+
the work, so the SDK does not automatically resubmit it. Both `paw.compile`
|
|
271
|
+
and `paw.compile_and_load` propagate these errors; `compile_and_load` does not
|
|
272
|
+
attempt to load a function when compilation raises.
|
|
273
|
+
|
|
186
274
|
## `paw.compile_and_load`
|
|
187
275
|
|
|
188
276
|
```python
|
|
@@ -136,7 +136,16 @@ List available compiler models and identifiers for use with compile requests.
|
|
|
136
136
|
|
|
137
137
|
### `GET /health`
|
|
138
138
|
|
|
139
|
-
|
|
139
|
+
Returns HTTP 200 for API liveness. The JSON `status` is `healthy` only when
|
|
140
|
+
all enabled public compilers pass their provider checks; otherwise it is
|
|
141
|
+
`degraded`, with details in `warnings` and `gpu_services`, keyed by compiler
|
|
142
|
+
name. Finetune checks include its base compiler, durable Redis, and at least
|
|
143
|
+
one healthy worker. A healthy worker remains available while busy.
|
|
144
|
+
|
|
145
|
+
Checks run concurrently with a three-second timeout and share a five-second
|
|
146
|
+
cache. `queue_depth` counts waiting finetune jobs across distinct dispatchers,
|
|
147
|
+
not running jobs; it is `null` when a queue cannot be verified. This is a
|
|
148
|
+
readiness observation, not a guarantee that a new compilation will succeed.
|
|
140
149
|
|
|
141
150
|
## Errors
|
|
142
151
|
|
|
@@ -27,7 +27,7 @@ try:
|
|
|
27
27
|
from importlib.metadata import version as _meta_version
|
|
28
28
|
__version__ = _meta_version("programasweights")
|
|
29
29
|
except Exception:
|
|
30
|
-
__version__ = "0.4.
|
|
30
|
+
__version__ = "0.4.6"
|
|
31
31
|
|
|
32
32
|
from ._output import ProgressCallback, ProgressEvent, report_progress
|
|
33
33
|
from .cache import CachedProgram
|
|
@@ -39,6 +39,7 @@ from .client import (
|
|
|
39
39
|
Program,
|
|
40
40
|
)
|
|
41
41
|
from .config import get_api_url, get_api_key, set_api_key
|
|
42
|
+
from .errors import APIError
|
|
42
43
|
|
|
43
44
|
|
|
44
45
|
def compile(
|
|
@@ -400,18 +401,22 @@ def function(
|
|
|
400
401
|
):
|
|
401
402
|
"""Load a compiled program, or explicitly load a bare base interpreter.
|
|
402
403
|
|
|
403
|
-
|
|
404
|
-
|
|
404
|
+
Hub references download the .paw bundle on first use; local paths supply
|
|
405
|
+
it directly. Required runtime metadata and base models may still download
|
|
406
|
+
unless offline mode is enabled. Subsequent calls reuse validated caches.
|
|
405
407
|
|
|
406
408
|
Args:
|
|
407
409
|
program_id: Program ID (str), slug (``da03/my-program``), pinned version
|
|
408
|
-
(``da03/my-program@v3``),
|
|
410
|
+
(``da03/my-program@v3``), a ``Program`` object from compile(), or a
|
|
411
|
+
local GGUF-based .paw bundle. PathLike objects and explicit path
|
|
412
|
+
strings (including strings ending in .paw) select local files.
|
|
409
413
|
n_ctx: Context window size for llama.cpp.
|
|
410
414
|
n_gpu_layers: GPU layers (-1 = all GPU, 0 = CPU only). Defaults to -1
|
|
411
415
|
(auto-uses Metal/CUDA if available, safe fallback to CPU).
|
|
412
416
|
Set ``PAW_GPU_LAYERS=0`` env var to force CPU-only.
|
|
413
417
|
verbose: Print llama.cpp debug output.
|
|
414
|
-
offline:
|
|
418
|
+
offline: Prohibit network access. Local bundles may be imported, but
|
|
419
|
+
their runtime and base model must already be available locally.
|
|
415
420
|
Also set via ``PAW_OFFLINE=1`` env var.
|
|
416
421
|
interpreter: Advanced adapter-free mode. This is only valid when
|
|
417
422
|
``program_id`` is explicitly ``None``. Initially supported values
|
|
@@ -427,6 +432,8 @@ def function(
|
|
|
427
432
|
|
|
428
433
|
>>> fn = paw.function("da03/my-program@v2") # pinned version
|
|
429
434
|
|
|
435
|
+
>>> fn = paw.function("./classifier.paw") # local bundle
|
|
436
|
+
|
|
430
437
|
>>> base = paw.function(None, interpreter="gpt2")
|
|
431
438
|
"""
|
|
432
439
|
import os
|
|
@@ -452,7 +459,12 @@ def function(
|
|
|
452
459
|
offline=offline,
|
|
453
460
|
)
|
|
454
461
|
|
|
455
|
-
|
|
462
|
+
from ._program_reference import local_program_path
|
|
463
|
+
|
|
464
|
+
local_path = local_program_path(program_id)
|
|
465
|
+
program_reference = (
|
|
466
|
+
_coerce_program_reference(program_id) if local_path is None else None
|
|
467
|
+
)
|
|
456
468
|
if program_reference == "":
|
|
457
469
|
raise ValueError(
|
|
458
470
|
"program_id cannot be an empty string; pass explicit None with "
|
|
@@ -464,25 +476,35 @@ def function(
|
|
|
464
476
|
"program_id=None to request adapter-free base mode."
|
|
465
477
|
)
|
|
466
478
|
|
|
467
|
-
|
|
479
|
+
if local_path is not None:
|
|
480
|
+
from .local_program import import_local_program
|
|
468
481
|
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
|
|
476
|
-
from .
|
|
482
|
+
# Validate explicit local input before importing the native runtime.
|
|
483
|
+
# A missing/corrupt file is never retried as a Hub ID or slug.
|
|
484
|
+
program_dir = import_local_program(local_path)
|
|
485
|
+
from .runtime_llamacpp import PawFunction
|
|
486
|
+
else:
|
|
487
|
+
# Preserve the existing Hub path's fail-fast dependency check before
|
|
488
|
+
# resolving slugs or downloading assets.
|
|
489
|
+
from .runtime_llamacpp import PawFunction
|
|
477
490
|
|
|
478
|
-
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
491
|
+
resolved_id = _resolve_program_id(program_reference, offline=offline)
|
|
492
|
+
if offline and not cache.has_valid_program_assets(resolved_id):
|
|
493
|
+
raise RuntimeError(
|
|
494
|
+
f"Program {resolved_id} is not fully cached; offline mode "
|
|
495
|
+
"prohibits program downloads."
|
|
496
|
+
)
|
|
497
|
+
if not cache.has_valid_program_assets(resolved_id):
|
|
498
|
+
from .client import PAWClient
|
|
499
|
+
|
|
500
|
+
client = PAWClient(api_url=get_api_url(), api_key=get_api_key())
|
|
501
|
+
client.download_paw(resolved_id)
|
|
502
|
+
if not cache.has_valid_program_assets(resolved_id):
|
|
503
|
+
raise RuntimeError(
|
|
504
|
+
f"Program {resolved_id} is missing valid compiled assets."
|
|
505
|
+
)
|
|
506
|
+
program_dir = cache.get_program_dir(resolved_id)
|
|
484
507
|
|
|
485
|
-
program_dir = cache.get_program_dir(resolved_id)
|
|
486
508
|
return PawFunction(
|
|
487
509
|
program_dir,
|
|
488
510
|
n_ctx=n_ctx,
|
|
@@ -616,6 +638,7 @@ def list_compilers() -> list[dict]:
|
|
|
616
638
|
|
|
617
639
|
|
|
618
640
|
__all__ = [
|
|
641
|
+
"APIError",
|
|
619
642
|
"CachedProgram",
|
|
620
643
|
"CompileCancellation",
|
|
621
644
|
"CompileJob",
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
"""Deterministic local-file intent, independent of filesystem contents."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
_WINDOWS_DRIVE = re.compile(r"^[A-Za-z]:")
|
|
11
|
+
_URI_SCHEME = re.compile(r"^[A-Za-z][A-Za-z0-9+.-]*:")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def local_program_path(reference: object) -> Path | None:
|
|
15
|
+
"""Return an explicit local path, or None for an ID/slug/Program.
|
|
16
|
+
|
|
17
|
+
Path-like objects always mean files. Strings select files by syntax, never
|
|
18
|
+
by existence; in particular, an ``owner/slug`` remains a Hub reference.
|
|
19
|
+
"""
|
|
20
|
+
is_pathlike = isinstance(reference, os.PathLike)
|
|
21
|
+
if is_pathlike:
|
|
22
|
+
value = os.fspath(reference)
|
|
23
|
+
if not isinstance(value, str):
|
|
24
|
+
raise TypeError("Local program paths must be text, not bytes.")
|
|
25
|
+
elif isinstance(reference, str):
|
|
26
|
+
value = reference
|
|
27
|
+
else:
|
|
28
|
+
return None
|
|
29
|
+
|
|
30
|
+
if "\x00" in value:
|
|
31
|
+
raise ValueError("Program references cannot contain null bytes.")
|
|
32
|
+
windows_drive = bool(_WINDOWS_DRIVE.match(value))
|
|
33
|
+
if not is_pathlike and not windows_drive and _URI_SCHEME.match(value):
|
|
34
|
+
raise ValueError(
|
|
35
|
+
"Program URLs are not supported. Download the .paw bundle first "
|
|
36
|
+
"and pass its local path."
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
explicit_path = value.startswith(
|
|
40
|
+
("./", "../", "/", "~/", "\\", ".\\", "..\\", "~\\")
|
|
41
|
+
)
|
|
42
|
+
if not (
|
|
43
|
+
is_pathlike or windows_drive or explicit_path
|
|
44
|
+
or value.lower().endswith(".paw")
|
|
45
|
+
):
|
|
46
|
+
return None
|
|
47
|
+
if not value:
|
|
48
|
+
raise ValueError("Local program path cannot be empty.")
|
|
49
|
+
|
|
50
|
+
# A Windows-looking string must never be sent to the Hub on another OS.
|
|
51
|
+
# Path objects, however, explicitly name a native path and may contain
|
|
52
|
+
# characters that would have a different meaning on another platform.
|
|
53
|
+
windows_path = windows_drive or value.startswith(
|
|
54
|
+
("\\", ".\\", "..\\", "~\\")
|
|
55
|
+
)
|
|
56
|
+
if not is_pathlike and os.name != "nt" and windows_path:
|
|
57
|
+
raise ValueError(
|
|
58
|
+
"This is a Windows filesystem path, which cannot be opened on "
|
|
59
|
+
"this platform. Pass a native local path instead."
|
|
60
|
+
)
|
|
61
|
+
return Path(value).expanduser()
|