programasweights 0.4.5__tar.gz → 0.4.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- programasweights-0.4.7/.cursor/rules/releases.mdc +17 -0
- programasweights-0.4.7/.github/workflows/release.yml +72 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/.github/workflows/test.yml +13 -1
- {programasweights-0.4.5 → programasweights-0.4.7}/AGENTS.md +42 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/CHANGELOG.md +19 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/PKG-INFO +27 -1
- {programasweights-0.4.5 → programasweights-0.4.7}/PYPI_README.md +26 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/README.md +26 -0
- programasweights-0.4.7/RELEASING.md +82 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/api-reference/cli.md +7 -2
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/api-reference/python-sdk.md +40 -6
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/api-reference/rest-api.md +31 -2
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/getting-started/first-program.md +11 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/getting-started/installation.md +3 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/index.md +14 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/__init__.py +64 -8
- programasweights-0.4.7/programasweights/_remote.py +99 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/cli.py +27 -10
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/runtime_llamacpp.py +38 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/pyproject.toml +1 -1
- programasweights-0.4.7/scripts/release_metadata.py +156 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_base_interpreter.py +204 -2
- programasweights-0.4.7/tests/test_release_metadata.py +296 -0
- programasweights-0.4.7/tests/test_remote_inference.py +294 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/.gitignore +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/.readthedocs.yaml +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/LICENSE +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/adr/001-llama-cpp-over-pytorch.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/adr/002-q4_0-adapter-format.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/adr/003-single-spec-field.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/adr/004-compiler-naming.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/adr/005-vllm-hidden-states.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/adr/006-email-api-key-auth.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/advanced/adrs.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/advanced/architecture.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/architecture.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/case-studies/alien-taboo.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/case-studies/log-monitoring.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/case-studies/semantic-search.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/case-studies/site-navigation.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/case-studies/tool-calling.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/getting-started/naming-programs.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/guide/browser-inference.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/guide/how-it-works.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/guide/local-inference.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/guide/writing-good-specs.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/hub/browsing-programs.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/hub/feedback-cases.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/hub/publishing-programs.md +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/docs/requirements.txt +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/examples/flask_app.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/examples/jupyter_notebook.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/examples/langchain_integration.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/examples/paw_monitor.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/examples/replace_openai.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/mkdocs.yml +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/_output.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/_program_reference.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/artifacts.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/cache.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/client.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/compiler/__init__.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/compiler/dummy.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/config.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/convert_peft_to_paw.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/errors.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/local_program.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/paw_format.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/runtime/__init__.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/runtime/interpreter.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/programasweights/runtime/interpreter_onnx.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_api_errors.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_cli_auth.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_compile_timeouts.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_desktop_sdk.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_local_program.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_offline_cache.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_runtime_registry_sdk.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_sdk.py +0 -0
- {programasweights-0.4.5 → programasweights-0.4.7}/tests/test_sdk.sh +0 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
---
|
|
2
|
+
description: Publish and verify complete Python SDK releases
|
|
3
|
+
alwaysApply: true
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# SDK releases
|
|
7
|
+
|
|
8
|
+
For release or publication work, read `RELEASING.md` first.
|
|
9
|
+
|
|
10
|
+
- A release is complete only when its annotated remote tag, PyPI wheel and
|
|
11
|
+
source distribution, and published GitHub Release all exist and agree.
|
|
12
|
+
- Build from the clean, tested release commit; push its annotated version tag
|
|
13
|
+
before uploading the exact checked artifacts to PyPI.
|
|
14
|
+
- Verify the `GitHub release` workflow succeeded and the latest GitHub Release
|
|
15
|
+
agrees with PyPI. A tag alone is not a GitHub Release entry.
|
|
16
|
+
- Repair missing entries using the documented workflow dispatch. Never move a
|
|
17
|
+
published tag, rebuild an existing version, or bypass artifact verification.
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
name: GitHub release
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags: ['v*']
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
inputs:
|
|
8
|
+
tag:
|
|
9
|
+
description: 'Existing stable version tag, e.g. v0.4.6'
|
|
10
|
+
required: true
|
|
11
|
+
type: string
|
|
12
|
+
|
|
13
|
+
permissions:
|
|
14
|
+
contents: write
|
|
15
|
+
|
|
16
|
+
concurrency:
|
|
17
|
+
group: github-release-${{ inputs.tag || github.ref_name }}
|
|
18
|
+
cancel-in-progress: false
|
|
19
|
+
|
|
20
|
+
jobs:
|
|
21
|
+
release:
|
|
22
|
+
runs-on: ubuntu-latest
|
|
23
|
+
timeout-minutes: 15
|
|
24
|
+
env:
|
|
25
|
+
RELEASE_TAG: ${{ inputs.tag || github.ref_name }}
|
|
26
|
+
GH_TOKEN: ${{ github.token }}
|
|
27
|
+
GH_REPO: ${{ github.repository }}
|
|
28
|
+
steps:
|
|
29
|
+
- name: Set the verification output directory
|
|
30
|
+
run: echo "RELEASE_DIR=$RUNNER_TEMP/verified-release" >> "$GITHUB_ENV"
|
|
31
|
+
- uses: actions/checkout@v4
|
|
32
|
+
with:
|
|
33
|
+
fetch-depth: 0
|
|
34
|
+
- uses: actions/setup-python@v5
|
|
35
|
+
with:
|
|
36
|
+
python-version: '3.12'
|
|
37
|
+
- name: Verify the remote tag and both published PyPI artifacts
|
|
38
|
+
# Tags are pushed before uploading. Wait for the wheel AND sdist;
|
|
39
|
+
# never announce an unpublished version or rebuild release assets.
|
|
40
|
+
run: python scripts/release_metadata.py --tag "$RELEASE_TAG" --output-dir "$RELEASE_DIR" --wait-seconds 600
|
|
41
|
+
- name: Publish the missing GitHub release
|
|
42
|
+
shell: bash
|
|
43
|
+
run: |
|
|
44
|
+
set -euo pipefail
|
|
45
|
+
if gh release view "$RELEASE_TAG" --json tagName >/dev/null 2>&1; then
|
|
46
|
+
echo "Release already exists; preserving its notes and assets."
|
|
47
|
+
else
|
|
48
|
+
latest=$(jq -r .latest "$RELEASE_DIR/verified.json")
|
|
49
|
+
gh release create "$RELEASE_TAG" "$RELEASE_DIR"/artifacts/* \
|
|
50
|
+
--verify-tag --title "$RELEASE_TAG" \
|
|
51
|
+
--notes-file "$RELEASE_DIR/notes.md" --latest="$latest"
|
|
52
|
+
fi
|
|
53
|
+
gh release view "$RELEASE_TAG" --json tagName,isDraft,isPrerelease,url \
|
|
54
|
+
| jq -e --arg tag "$RELEASE_TAG" '.tagName == $tag and .isDraft == false and .isPrerelease == false'
|
|
55
|
+
- name: Verify the uploaded release assets and latest marker
|
|
56
|
+
shell: bash
|
|
57
|
+
run: |
|
|
58
|
+
set -euo pipefail
|
|
59
|
+
gh release download "$RELEASE_TAG" --pattern 'programasweights-*' \
|
|
60
|
+
--dir "$RELEASE_DIR/github-assets"
|
|
61
|
+
for artifact in "$RELEASE_DIR"/artifacts/*; do
|
|
62
|
+
cmp "$artifact" "$RELEASE_DIR/github-assets/$(basename "$artifact")"
|
|
63
|
+
done
|
|
64
|
+
if [ "$(jq -r .latest "$RELEASE_DIR/verified.json")" = true ]; then
|
|
65
|
+
gh api "repos/$GH_REPO/releases/latest" \
|
|
66
|
+
| jq -e --arg tag "$RELEASE_TAG" '.tag_name == $tag'
|
|
67
|
+
fi
|
|
68
|
+
- name: Retain the verification record
|
|
69
|
+
uses: actions/upload-artifact@v4
|
|
70
|
+
with:
|
|
71
|
+
name: release-verification-${{ inputs.tag || github.ref_name }}
|
|
72
|
+
path: ${{ runner.temp }}/verified-release/verified.json
|
|
@@ -6,6 +6,17 @@ on:
|
|
|
6
6
|
pull_request:
|
|
7
7
|
|
|
8
8
|
jobs:
|
|
9
|
+
release-metadata:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v4
|
|
13
|
+
- uses: actions/setup-python@v5
|
|
14
|
+
with:
|
|
15
|
+
python-version: '3.12'
|
|
16
|
+
- run: python -m pip install pytest
|
|
17
|
+
- name: Test release verification without network or publishing
|
|
18
|
+
run: python -m pytest tests/test_release_metadata.py
|
|
19
|
+
|
|
9
20
|
test:
|
|
10
21
|
runs-on: ubuntu-latest
|
|
11
22
|
strategy:
|
|
@@ -41,6 +52,7 @@ jobs:
|
|
|
41
52
|
tests/test_local_program.py \
|
|
42
53
|
tests/test_base_interpreter.py \
|
|
43
54
|
tests/test_cli_auth.py \
|
|
55
|
+
tests/test_remote_inference.py \
|
|
44
56
|
tests/test_desktop_sdk.py \
|
|
45
57
|
tests/test_runtime_registry_sdk.py \
|
|
46
58
|
tests/test_sdk.py::TestInstallAndImport
|
|
@@ -61,7 +73,7 @@ jobs:
|
|
|
61
73
|
python -m pip install httpx pytest
|
|
62
74
|
python -m pip install -e . --no-deps
|
|
63
75
|
- name: Test Windows local paths, cache locks, and compile errors
|
|
64
|
-
run: python -m pytest tests/test_api_errors.py tests/test_compile_timeouts.py tests/test_local_program.py --junitxml=test-results.xml
|
|
76
|
+
run: python -m pytest tests/test_api_errors.py tests/test_compile_timeouts.py tests/test_local_program.py tests/test_remote_inference.py --junitxml=test-results.xml
|
|
65
77
|
- name: Annotate Windows test failures
|
|
66
78
|
if: failure()
|
|
67
79
|
shell: python
|
|
@@ -43,8 +43,46 @@ fn = paw.compile_and_load("Classify sentiment as positive or negative")
|
|
|
43
43
|
fn("I love this!") # "positive"
|
|
44
44
|
```
|
|
45
45
|
|
|
46
|
+
Load a local `.paw` file with `paw.function("./classifier.paw")` (SDK 0.4.5+).
|
|
47
|
+
|
|
46
48
|
If you want the smaller browser-compatible runtime explicitly, pass `compiler="paw-4b-gpt2"`. Otherwise, omit `compiler` and let the server default decide.
|
|
47
49
|
|
|
50
|
+
## Remote inference (optional)
|
|
51
|
+
|
|
52
|
+
Use the hosted API for fast inference in around 150 ms, without a local model download.
|
|
53
|
+
|
|
54
|
+
### Python SDK
|
|
55
|
+
|
|
56
|
+
```python
|
|
57
|
+
import programasweights as paw
|
|
58
|
+
|
|
59
|
+
with paw.function("email-triage", remote=True) as remote_fn:
|
|
60
|
+
print(remote_fn("Urgent: the server is down!"))
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
For authenticated requests, set `PAW_API_KEY` or use `paw.login()`.
|
|
64
|
+
|
|
65
|
+
### Direct HTTP
|
|
66
|
+
|
|
67
|
+
Use `httpx` directly without installing the PAW SDK.
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
import httpx
|
|
71
|
+
|
|
72
|
+
with httpx.Client(timeout=60.0) as client:
|
|
73
|
+
response = client.post(
|
|
74
|
+
"https://programasweights.com/api/v1/infer",
|
|
75
|
+
json={
|
|
76
|
+
"program_id": "email-triage",
|
|
77
|
+
"input": "Urgent: server is down!"
|
|
78
|
+
},
|
|
79
|
+
)
|
|
80
|
+
response.raise_for_status()
|
|
81
|
+
print(response.json()["output"])
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
For authenticated requests, pass `headers={"X-API-Key": api_key}` to `client.post()`.
|
|
85
|
+
|
|
48
86
|
## Current Public Compilers
|
|
49
87
|
|
|
50
88
|
- **Standard** (`paw-4b-qwen3-0.6b`) — higher accuracy, 594 MB base + ~22 MB/program. This is the current server default.
|
|
@@ -89,6 +127,8 @@ Output: delete
|
|
|
89
127
|
- Spec + input + output share a ~2048 token context window. Inputs that exceed it will error.
|
|
90
128
|
- `max_tokens` defaults to `None`: generation runs until EOS or the context limit.
|
|
91
129
|
- Compile runs on the hosted PAW API. Inference should usually run locally through the SDK.
|
|
130
|
+
- Synchronous compile requests use a 40-minute read timeout.
|
|
131
|
+
- **Run local inference sequentially by default.** With PAW’s current llama.cpp backend, simultaneous inference calls often perform worse. Reuse loaded functions and process inputs one at a time; never call the same function instance concurrently.
|
|
92
132
|
- **GPU acceleration** is enabled by default (`n_gpu_layers=-1`). Uses Metal on Mac, CUDA on Linux, and falls back to CPU automatically. If GPU causes issues, set `PAW_GPU_LAYERS=0` or pass `n_gpu_layers=0`.
|
|
93
133
|
- **First call** is usually ~1-5s because it loads the base model. Subsequent calls are typically ~0.05-0.5s depending on input length and GPU availability.
|
|
94
134
|
- **Base model files are shared** across programs on disk. Each Standard LoRA adapter is ~22 MB; each Compact LoRA adapter is ~5 MB.
|
|
@@ -102,6 +142,8 @@ Output: delete
|
|
|
102
142
|
|
|
103
143
|
## Common Errors
|
|
104
144
|
|
|
145
|
+
Compile API HTTP errors raise `paw.APIError`. Check `error.code` and `error.message` for details. The SDK does not retry automatically.
|
|
146
|
+
|
|
105
147
|
| Error | Cause | Fix |
|
|
106
148
|
|-------|-------|-----|
|
|
107
149
|
| `RuntimeError: assets not ready` on download | Program is still generating after compile | The SDK polls automatically for up to 60s. If it still fails, retry shortly or recompile. |
|
|
@@ -1,5 +1,24 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.4.7 (2026-09-20)
|
|
4
|
+
|
|
5
|
+
- Add `remote=True` to `paw.function` and `paw.compile_and_load`, plus
|
|
6
|
+
`paw run --remote`, for hosted inference without downloading model assets
|
|
7
|
+
or loading the local runtime. Local inference remains the default.
|
|
8
|
+
- Use server generation defaults unless explicitly overridden. Reuse HTTP
|
|
9
|
+
connections, support `with` and `.close()`, and preserve structured
|
|
10
|
+
`paw.APIError` details.
|
|
11
|
+
- Reject remote inference combined with offline mode or incompatible local
|
|
12
|
+
runtime options.
|
|
13
|
+
|
|
14
|
+
## 0.4.6 (2026-09-13)
|
|
15
|
+
|
|
16
|
+
- Add an optional `logits_processor` argument to a compiled or base program
|
|
17
|
+
call for caller-supplied token constraints in llama.cpp's sampler.
|
|
18
|
+
Defaults to `None`; sampling is unchanged when it is unset.
|
|
19
|
+
- Propagate processor failures to the caller instead of letting native
|
|
20
|
+
callback errors silently continue with unconstrained output.
|
|
21
|
+
|
|
3
22
|
## 0.4.5 (2026-09-10)
|
|
4
23
|
|
|
5
24
|
- Expose structured compile API failures as `paw.APIError`, compatible with
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: programasweights
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.7
|
|
4
4
|
Summary: Compile natural language specifications into neural programs that run locally via llama.cpp.
|
|
5
5
|
Project-URL: Homepage, https://programasweights.com
|
|
6
6
|
Project-URL: Repository, https://github.com/programasweights/programasweights-python
|
|
@@ -66,6 +66,19 @@ fn("I love this!") # "positive"
|
|
|
66
66
|
|
|
67
67
|
If you specifically want the smaller browser-compatible runtime, pass `compiler="paw-4b-gpt2"`. Otherwise, omit `compiler` and let the server default decide.
|
|
68
68
|
|
|
69
|
+
## Remote inference (optional)
|
|
70
|
+
|
|
71
|
+
Use the hosted API for fast inference in around 150 ms, without a local model download.
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
import programasweights as paw
|
|
75
|
+
|
|
76
|
+
with paw.function("email-triage", remote=True) as remote_fn:
|
|
77
|
+
print(remote_fn("Urgent: the server is down!"))
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
For direct HTTP calls, see the [REST API reference](https://programasweights.readthedocs.io/en/latest/api-reference/rest-api/#post-infer).
|
|
81
|
+
|
|
69
82
|
## Current Public Compilers
|
|
70
83
|
|
|
71
84
|
|
|
@@ -84,6 +97,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
|
|
|
84
97
|
|
|
85
98
|
GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
|
|
86
99
|
|
|
100
|
+
## Constrained Decoding
|
|
101
|
+
|
|
102
|
+
In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
import llama_cpp
|
|
106
|
+
# my_processor is your compatible callable: (input_ids, scores) -> scores.
|
|
107
|
+
fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
|
|
111
|
+
|
|
87
112
|
## Desktop and Offline Workflows
|
|
88
113
|
|
|
89
114
|
Prepare and inspect validated local assets without keeping a model loaded:
|
|
@@ -177,6 +202,7 @@ Generate API keys at [programasweights.com/settings](https://programasweights.co
|
|
|
177
202
|
paw compile --spec "Extract error lines from logs" --json
|
|
178
203
|
paw run --program <program_id> --input "[ERROR] timeout" --json
|
|
179
204
|
paw run --program <program_id> --input "[ERROR] timeout" --offline --json
|
|
205
|
+
paw run --program <program_id> --input "[ERROR] timeout" --remote --json
|
|
180
206
|
paw login
|
|
181
207
|
```
|
|
182
208
|
|
|
@@ -35,6 +35,19 @@ fn("I love this!") # "positive"
|
|
|
35
35
|
|
|
36
36
|
If you specifically want the smaller browser-compatible runtime, pass `compiler="paw-4b-gpt2"`. Otherwise, omit `compiler` and let the server default decide.
|
|
37
37
|
|
|
38
|
+
## Remote inference (optional)
|
|
39
|
+
|
|
40
|
+
Use the hosted API for fast inference in around 150 ms, without a local model download.
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
import programasweights as paw
|
|
44
|
+
|
|
45
|
+
with paw.function("email-triage", remote=True) as remote_fn:
|
|
46
|
+
print(remote_fn("Urgent: the server is down!"))
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
For direct HTTP calls, see the [REST API reference](https://programasweights.readthedocs.io/en/latest/api-reference/rest-api/#post-infer).
|
|
50
|
+
|
|
38
51
|
## Current Public Compilers
|
|
39
52
|
|
|
40
53
|
|
|
@@ -53,6 +66,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
|
|
|
53
66
|
|
|
54
67
|
GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
|
|
55
68
|
|
|
69
|
+
## Constrained Decoding
|
|
70
|
+
|
|
71
|
+
In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
import llama_cpp
|
|
75
|
+
# my_processor is your compatible callable: (input_ids, scores) -> scores.
|
|
76
|
+
fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
|
|
80
|
+
|
|
56
81
|
## Desktop and Offline Workflows
|
|
57
82
|
|
|
58
83
|
Prepare and inspect validated local assets without keeping a model loaded:
|
|
@@ -146,6 +171,7 @@ Generate API keys at [programasweights.com/settings](https://programasweights.co
|
|
|
146
171
|
paw compile --spec "Extract error lines from logs" --json
|
|
147
172
|
paw run --program <program_id> --input "[ERROR] timeout" --json
|
|
148
173
|
paw run --program <program_id> --input "[ERROR] timeout" --offline --json
|
|
174
|
+
paw run --program <program_id> --input "[ERROR] timeout" --remote --json
|
|
149
175
|
paw login
|
|
150
176
|
```
|
|
151
177
|
|
|
@@ -35,6 +35,19 @@ fn("I love this!") # "positive"
|
|
|
35
35
|
|
|
36
36
|
If you specifically want the smaller browser-compatible runtime, pass `compiler="paw-4b-gpt2"`. Otherwise, omit `compiler` and let the server default decide.
|
|
37
37
|
|
|
38
|
+
## Remote inference (optional)
|
|
39
|
+
|
|
40
|
+
Use the hosted API for fast inference in around 150 ms, without a local model download.
|
|
41
|
+
|
|
42
|
+
```python
|
|
43
|
+
import programasweights as paw
|
|
44
|
+
|
|
45
|
+
with paw.function("email-triage", remote=True) as remote_fn:
|
|
46
|
+
print(remote_fn("Urgent: the server is down!"))
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
For direct HTTP calls, see the [REST API reference](docs/api-reference/rest-api.md#post-infer).
|
|
50
|
+
|
|
38
51
|
## Current Public Compilers
|
|
39
52
|
|
|
40
53
|
|
|
@@ -53,6 +66,18 @@ If you need to inspect available compiler aliases programmatically, use `paw.lis
|
|
|
53
66
|
|
|
54
67
|
GPU acceleration is enabled by default (Metal on Mac, CUDA on Linux, falls back to CPU). Set `PAW_GPU_LAYERS=0` to force CPU if GPU causes issues.
|
|
55
68
|
|
|
69
|
+
## Constrained Decoding
|
|
70
|
+
|
|
71
|
+
In SDK 0.4.6+, a call accepts an optional `logits_processor`: an advanced hook for caller-supplied llama.cpp-compatible token constraints, not built-in regex or JSON-schema validation. It runs at every generation step. The default, `None`, keeps sampling unchanged.
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
import llama_cpp
|
|
75
|
+
# my_processor is your compatible callable: (input_ids, scores) -> scores.
|
|
76
|
+
fn("Office line: +1-555-666-7777", logits_processor=llama_cpp.LogitsProcessorList([my_processor]))
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Processors see the full prompt and generated-token history, not just the output. Create or reset stateful processors for each call. Token limits and output whitespace trimming still apply, so validate the returned result.
|
|
80
|
+
|
|
56
81
|
## Desktop and Offline Workflows
|
|
57
82
|
|
|
58
83
|
Prepare and inspect validated local assets without keeping a model loaded:
|
|
@@ -147,6 +172,7 @@ Generate API keys at [programasweights.com/settings](https://programasweights.co
|
|
|
147
172
|
paw compile --spec "Extract error lines from logs" --json
|
|
148
173
|
paw run --program <program_id> --input "[ERROR] timeout" --json
|
|
149
174
|
paw run --program <program_id> --input "[ERROR] timeout" --offline --json
|
|
175
|
+
paw run --program <program_id> --input "[ERROR] timeout" --remote --json
|
|
150
176
|
paw login
|
|
151
177
|
```
|
|
152
178
|
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
# Releasing the Python SDK
|
|
2
|
+
|
|
3
|
+
A release has three separate records: an annotated Git tag identifies the source,
|
|
4
|
+
PyPI serves the installable wheel and source distribution, and a GitHub Release
|
|
5
|
+
shows the version and release notes on the repository. A tag or PyPI upload alone
|
|
6
|
+
does not create a GitHub Release. Finish and verify all three before reporting a
|
|
7
|
+
release complete.
|
|
8
|
+
|
|
9
|
+
## Publish a new version
|
|
10
|
+
|
|
11
|
+
1. Update `pyproject.toml`, the fallback `__version__` in
|
|
12
|
+
`programasweights/__init__.py`, and the matching section in `CHANGELOG.md`.
|
|
13
|
+
Commit and push these changes to `main` using the existing user Git identity.
|
|
14
|
+
Release from a clean checkout of that pushed commit; confirm `HEAD` equals
|
|
15
|
+
`origin/main` and the `tests` workflow is green for that exact commit.
|
|
16
|
+
2. Set the intended version, then build a wheel and source distribution into a
|
|
17
|
+
new, empty directory. Install `build` and `twine` in the release environment if
|
|
18
|
+
needed. Use the two exact filenames below, not a shared `dist/*` directory.
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
SDK_RELEASE_VERSION=0.4.7 # replace with the version being released
|
|
22
|
+
SDK_RELEASE_DIR=$(mktemp -d)
|
|
23
|
+
python -m build --outdir "$SDK_RELEASE_DIR"
|
|
24
|
+
python -m twine check \
|
|
25
|
+
"$SDK_RELEASE_DIR/programasweights-$SDK_RELEASE_VERSION-py3-none-any.whl" \
|
|
26
|
+
"$SDK_RELEASE_DIR/programasweights-$SDK_RELEASE_VERSION.tar.gz"
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
3. Create and push the annotated version tag **before uploading to PyPI**. Check
|
|
30
|
+
that the remote tag resolves to the intended commit and is an annotated tag.
|
|
31
|
+
Never replace an existing release tag with a new target.
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
git tag -a "v$SDK_RELEASE_VERSION" -m "Release v$SDK_RELEASE_VERSION"
|
|
35
|
+
git push origin "v$SDK_RELEASE_VERSION"
|
|
36
|
+
git ls-remote --tags origin "refs/tags/v$SDK_RELEASE_VERSION*"
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
4. Upload the exact wheel and source distribution that passed `twine check`:
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
python -m twine upload \
|
|
43
|
+
"$SDK_RELEASE_DIR/programasweights-$SDK_RELEASE_VERSION-py3-none-any.whl" \
|
|
44
|
+
"$SDK_RELEASE_DIR/programasweights-$SDK_RELEASE_VERSION.tar.gz"
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
5. The `release.yml` workflow starts on the `v*` tag push. It waits up to 600
|
|
48
|
+
seconds for PyPI, verifies the remote annotated tag, version metadata, and
|
|
49
|
+
published wheel/source distribution against the tagged source, then creates
|
|
50
|
+
the GitHub Release with notes from that tag's `CHANGELOG.md` and the published
|
|
51
|
+
artifacts. It preserves an existing release and marks a new release as latest
|
|
52
|
+
only when its version is the latest on PyPI. If upload takes longer than the
|
|
53
|
+
wait window, dispatch the workflow again after PyPI publication succeeds.
|
|
54
|
+
6. Verify the workflow succeeded, the GitHub Releases page has the expected entry
|
|
55
|
+
and notes, and the latest-release entry agrees with PyPI's latest version.
|
|
56
|
+
Compare both uploaded files' SHA-256 hashes with the version's PyPI JSON
|
|
57
|
+
(`https://pypi.org/pypi/programasweights/<version>/json`, `urls[].digests.sha256`)
|
|
58
|
+
and with the GitHub Release assets. Do not treat a tag listing as confirmation
|
|
59
|
+
that the release entry exists.
|
|
60
|
+
|
|
61
|
+
## Recover a missing GitHub Release
|
|
62
|
+
|
|
63
|
+
When a version already exists on PyPI, keep its existing tag and published files.
|
|
64
|
+
Do not rebuild or republish that version, and never move its tag. Verify the
|
|
65
|
+
remote annotated tag, matching version metadata and changelog section, and the
|
|
66
|
+
published artifacts first. `scripts/release_metadata.py` requires Python 3.11+
|
|
67
|
+
and performs those checks, preparing the notes and assets in a new directory:
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
SDK_RELEASE_RECOVERY_DIR=$(mktemp -d)
|
|
71
|
+
git fetch origin --tags
|
|
72
|
+
python scripts/release_metadata.py --tag v0.4.6 \
|
|
73
|
+
--output-dir "$SDK_RELEASE_RECOVERY_DIR/verified" --wait-seconds 600
|
|
74
|
+
gh workflow run release.yml --ref main -f tag=v0.4.6
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Replace `v0.4.6` with the verified existing tag to repair. Run this from current
|
|
78
|
+
`main`, which contains the recovery workflow and script; the release content is
|
|
79
|
+
read from the requested tag. The workflow is idempotent: an existing GitHub
|
|
80
|
+
Release is preserved. Recheck the release entry, latest-version status, and
|
|
81
|
+
artifact hashes after it finishes. A source or artifact mismatch needs
|
|
82
|
+
investigation; do not retag or overwrite published artifacts to make it pass.
|
|
@@ -33,10 +33,14 @@ paw compile --spec "Classify message urgency" [--compiler paw-4b-qwen3-0.6b] [--
|
|
|
33
33
|
|
|
34
34
|
Run inference locally against a compiled program (the normal mode), or
|
|
35
35
|
explicitly against a bare base interpreter (advanced mode).
|
|
36
|
+
Add `--remote` for hosted inference without downloading local model assets.
|
|
36
37
|
|
|
37
38
|
```bash
|
|
38
39
|
paw run --program <id_or_slug> --input "your text" [--offline] [--json]
|
|
39
40
|
|
|
41
|
+
# Optional hosted inference
|
|
42
|
+
paw run --program email-triage --input "Urgent: the server is down!" --remote --json
|
|
43
|
+
|
|
40
44
|
# Advanced adapter-free mode
|
|
41
45
|
paw run --base --interpreter gpt2 --input "raw prompt" [--offline] [--json]
|
|
42
46
|
```
|
|
@@ -47,10 +51,11 @@ paw run --base --interpreter gpt2 --input "raw prompt" [--offline] [--json]
|
|
|
47
51
|
| `--base` | Select adapter-free base mode. Mutually exclusive with `--program` and requires `--interpreter`. |
|
|
48
52
|
| `--interpreter` | Base interpreter: `Qwen/Qwen3-0.6B` or `gpt2`. Only valid with `--base`. |
|
|
49
53
|
| `--input` | Input text for the program. |
|
|
50
|
-
| `--max-tokens` | Maximum tokens to generate
|
|
51
|
-
| `--temperature` | Sampling temperature
|
|
54
|
+
| `--max-tokens` | Maximum tokens to generate. Local default: 512. Remote inference uses the server default when omitted. |
|
|
55
|
+
| `--temperature` | Sampling temperature. Local default: 0.0. Remote inference uses the server default when omitted. |
|
|
52
56
|
| `--verbose` | Print llama.cpp debug output. |
|
|
53
57
|
| `--offline` | Require all selected assets to already be cached and make zero network calls. |
|
|
58
|
+
| `--remote` | Use hosted inference for `--program`. Incompatible with `--base`, `--offline`, `--verbose`, and `PAW_OFFLINE=1`. |
|
|
54
59
|
| `--json` | JSON output with `mode`, `program`, `interpreter`, `input`, and `output`. |
|
|
55
60
|
|
|
56
61
|
Exactly one of `--program` and `--base` is required. An empty or
|
|
@@ -24,11 +24,13 @@ fn = paw.function(
|
|
|
24
24
|
verbose=False,
|
|
25
25
|
offline=False,
|
|
26
26
|
*,
|
|
27
|
+
remote=False,
|
|
27
28
|
interpreter=None,
|
|
28
29
|
)
|
|
29
30
|
```
|
|
30
31
|
|
|
31
|
-
|
|
32
|
+
Returns a callable for a compiled program. Local inference is the default.
|
|
33
|
+
Hub references download the
|
|
32
34
|
program and base model on first use; local `.paw` files supply the program
|
|
33
35
|
bundle directly. Required runtime metadata and base models are cached for reuse.
|
|
34
36
|
|
|
@@ -39,12 +41,13 @@ bundle directly. Required runtime metadata and base models are cached for reuse.
|
|
|
39
41
|
| `n_gpu_layers` | GPU layers to offload (`0` = CPU-only, `-1` = all). The default is `-1`, or `PAW_GPU_LAYERS` when set. |
|
|
40
42
|
| `verbose` | Enable verbose logging (default `False`). |
|
|
41
43
|
| `offline` | Use only local files/cache and make zero network calls; fail if required validated assets are missing. `PAW_OFFLINE=1` has the same effect. |
|
|
44
|
+
| `remote` | Run hosted inference without downloading model assets (default `False`). Accepts a `Program` object, ID, or slug. Cannot be combined with offline mode, local file paths, `interpreter`, or non-default local runtime options. |
|
|
42
45
|
| `interpreter` | Advanced adapter-free mode only. Must be passed by keyword and only when `program_id` is explicitly `None`. Supported values are `Qwen/Qwen3-0.6B` and `gpt2`. |
|
|
43
46
|
|
|
44
|
-
|
|
47
|
+
For local inference, the returned callable accepts:
|
|
45
48
|
|
|
46
49
|
```python
|
|
47
|
-
output: str = fn(input_text, max_tokens=None, temperature=0.0)
|
|
50
|
+
output: str = fn(input_text, max_tokens=None, temperature=0.0, logits_processor=None)
|
|
48
51
|
```
|
|
49
52
|
|
|
50
53
|
| Parameter | Description |
|
|
@@ -52,6 +55,9 @@ output: str = fn(input_text, max_tokens=None, temperature=0.0)
|
|
|
52
55
|
| `input_text` | Input string for the program. |
|
|
53
56
|
| `max_tokens` | Maximum tokens to generate. `None` (default) = use all remaining context window. |
|
|
54
57
|
| `temperature` | Sampling temperature (default `0.0`). |
|
|
58
|
+
| `logits_processor` | SDK 0.4.6+. Optional `llama_cpp.LogitsProcessorList` of caller-supplied processors, applied at every generation step. `None` (default) keeps sampling unchanged. |
|
|
59
|
+
|
|
60
|
+
This advanced hook is not built-in regex or JSON-schema validation. Each processor takes `(input_ids, scores)` and returns modified scores. Its token history includes the full prompt (including any compiled prefix and suffix or base-model template) plus generated tokens. Create or reset stateful processors for each call; processor exceptions propagate to the caller. Token limits and the usual output whitespace trimming still apply, so validate the returned result.
|
|
55
61
|
|
|
56
62
|
**Context limits:** Spec + input + output share a ~2048 token window. Inputs that exceed it will error. `max_tokens` defaults to `None`: generation runs until EOS or the context limit.
|
|
57
63
|
|
|
@@ -64,6 +70,25 @@ same canonical integrity metadata, so missing server-side checksum fields
|
|
|
64
70
|
cannot weaken validation. Missing or failed adapters raise an error; the SDK
|
|
65
71
|
never silently falls back to an unadapted base model.
|
|
66
72
|
|
|
73
|
+
### Remote inference
|
|
74
|
+
|
|
75
|
+
```python
|
|
76
|
+
with paw.function("email-triage", remote=True) as remote_fn:
|
|
77
|
+
output = remote_fn("Urgent: the server is down!")
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
The callable returns a string and accepts optional `max_tokens` and `temperature`.
|
|
81
|
+
Omitting either argument, or passing `None`, uses the server's default for that
|
|
82
|
+
setting. `logits_processor` is supported only for local inference.
|
|
83
|
+
|
|
84
|
+
Slugs are resolved once when the function is loaded. The `with` block closes
|
|
85
|
+
the HTTP client; otherwise, call `remote_fn.close()` when finished.
|
|
86
|
+
|
|
87
|
+
Remote calls use the existing [SDK configuration](#configuration) for the API
|
|
88
|
+
URL and API key. HTTP 4xx/5xx responses raise `paw.APIError`, preserving the
|
|
89
|
+
response status, headers, and structured error details. Transport errors
|
|
90
|
+
propagate unchanged.
|
|
91
|
+
|
|
67
92
|
### Loading a local `.paw` file
|
|
68
93
|
|
|
69
94
|
Version 0.4.5 adds local-file inputs to `paw.function`:
|
|
@@ -271,12 +296,21 @@ attempt to load a function when compilation raises.
|
|
|
271
296
|
## `paw.compile_and_load`
|
|
272
297
|
|
|
273
298
|
```python
|
|
274
|
-
fn = paw.compile_and_load(spec, compiler=
|
|
299
|
+
fn = paw.compile_and_load(spec, compiler=None, remote=False, **kwargs)
|
|
275
300
|
```
|
|
276
301
|
|
|
277
|
-
|
|
302
|
+
Compiles a spec and returns a callable. Inference runs locally by default.
|
|
303
|
+
Pass `remote=True` for hosted inference without downloading model assets.
|
|
304
|
+
|
|
305
|
+
Accepts the parameters of `paw.compile`, plus `n_ctx`, `n_gpu_layers`, `verbose`,
|
|
306
|
+
and `remote`. Local runtime options must retain their defaults with `remote=True`.
|
|
278
307
|
|
|
279
|
-
|
|
308
|
+
```python
|
|
309
|
+
with paw.compile_and_load(
|
|
310
|
+
"Classify sentiment as positive or negative", remote=True,
|
|
311
|
+
) as remote_fn:
|
|
312
|
+
output = remote_fn("I love this!")
|
|
313
|
+
```
|
|
280
314
|
|
|
281
315
|
## `paw.list_programs`
|
|
282
316
|
|
|
@@ -42,7 +42,10 @@ Compile a specification.
|
|
|
42
42
|
|
|
43
43
|
### `POST /infer`
|
|
44
44
|
|
|
45
|
-
|
|
45
|
+
Use the hosted API for fast inference in around 150 ms, without a local model download.
|
|
46
|
+
|
|
47
|
+
The Python SDK exposes hosted inference through `paw.function(..., remote=True)`.
|
|
48
|
+
See the [remote inference reference](python-sdk.md#remote-inference).
|
|
46
49
|
|
|
47
50
|
**Request body (JSON):**
|
|
48
51
|
|
|
@@ -55,6 +58,23 @@ Run inference for a compiled program (server-side execution).
|
|
|
55
58
|
|
|
56
59
|
**Response (JSON):** includes `output`, `tokens_generated`, and `latency_ms`.
|
|
57
60
|
|
|
61
|
+
**Example:**
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
import httpx
|
|
65
|
+
|
|
66
|
+
with httpx.Client(timeout=60.0) as client:
|
|
67
|
+
response = client.post(
|
|
68
|
+
"https://programasweights.com/api/v1/infer",
|
|
69
|
+
json={
|
|
70
|
+
"program_id": "email-triage",
|
|
71
|
+
"input": "Urgent: server is down!"
|
|
72
|
+
},
|
|
73
|
+
)
|
|
74
|
+
response.raise_for_status()
|
|
75
|
+
print(response.json()["output"])
|
|
76
|
+
```
|
|
77
|
+
|
|
58
78
|
### `GET /programs`
|
|
59
79
|
|
|
60
80
|
List or search programs.
|
|
@@ -136,7 +156,16 @@ List available compiler models and identifiers for use with compile requests.
|
|
|
136
156
|
|
|
137
157
|
### `GET /health`
|
|
138
158
|
|
|
139
|
-
|
|
159
|
+
Returns HTTP 200 for API liveness. The JSON `status` is `healthy` only when
|
|
160
|
+
all enabled public compilers pass their provider checks; otherwise it is
|
|
161
|
+
`degraded`, with details in `warnings` and `gpu_services`, keyed by compiler
|
|
162
|
+
name. Finetune checks include its base compiler, durable Redis, and at least
|
|
163
|
+
one healthy worker. A healthy worker remains available while busy.
|
|
164
|
+
|
|
165
|
+
Checks run concurrently with a three-second timeout and share a five-second
|
|
166
|
+
cache. `queue_depth` counts waiting finetune jobs across distinct dispatchers,
|
|
167
|
+
not running jobs; it is `null` when a queue cannot be verified. This is a
|
|
168
|
+
readiness observation, not a guarantee that a new compilation will succeed.
|
|
140
169
|
|
|
141
170
|
## Errors
|
|
142
171
|
|
|
@@ -20,6 +20,17 @@ print(result)
|
|
|
20
20
|
|
|
21
21
|
The first call may download the program and runtime assets; later calls use the local cache.
|
|
22
22
|
|
|
23
|
+
### Optional: remote inference
|
|
24
|
+
|
|
25
|
+
For fast inference without downloading model assets, pass `remote=True`:
|
|
26
|
+
|
|
27
|
+
```python
|
|
28
|
+
with paw.function("email-triage", remote=True) as remote_fn:
|
|
29
|
+
print(remote_fn("Urgent: the server is down!"))
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
For direct HTTP calls, see the [REST API reference](../api-reference/rest-api.md#post-infer).
|
|
33
|
+
|
|
23
34
|
## Step 2: Compile your own program
|
|
24
35
|
|
|
25
36
|
Describe the behavior you want in natural language, compile it, then load the result by `program_id`:
|
|
@@ -16,6 +16,9 @@ The `--extra-index-url` flag provides pre-built binaries for `llama-cpp-python`,
|
|
|
16
16
|
GPU offload defaults to all available layers (`n_gpu_layers=-1`). Set
|
|
17
17
|
`PAW_GPU_LAYERS=0` to force CPU-only execution.
|
|
18
18
|
|
|
19
|
+
The same installation supports [remote inference](../api-reference/python-sdk.md#remote-inference)
|
|
20
|
+
with `paw.function(..., remote=True)`, without downloading local model assets.
|
|
21
|
+
|
|
19
22
|
## Anaconda on Linux: OpenMP / libgomp errors
|
|
20
23
|
|
|
21
24
|
If the build fails with `libgomp`-related errors, disable OpenMP:
|