matrx-batch 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- matrx_batch-0.1.0/.gitignore +271 -0
- matrx_batch-0.1.0/CLAUDE.md +137 -0
- matrx_batch-0.1.0/PKG-INFO +88 -0
- matrx_batch-0.1.0/README.md +64 -0
- matrx_batch-0.1.0/matrx_batch/__init__.py +168 -0
- matrx_batch-0.1.0/matrx_batch/_ext.py +142 -0
- matrx_batch-0.1.0/matrx_batch/anthropic_batch.py +183 -0
- matrx_batch-0.1.0/matrx_batch/embedding_cache.py +180 -0
- matrx_batch-0.1.0/matrx_batch/openai_batch.py +213 -0
- matrx_batch-0.1.0/matrx_batch/router.py +295 -0
- matrx_batch-0.1.0/pyproject.toml +49 -0
- matrx_batch-0.1.0/tests/__init__.py +0 -0
- matrx_batch-0.1.0/tests/test_embedding_cache.py +280 -0
- matrx_batch-0.1.0/tests/test_router.py +446 -0
|
@@ -0,0 +1,271 @@
|
|
|
1
|
+
*.pyc
|
|
2
|
+
secrets/
|
|
3
|
+
ignore/
|
|
4
|
+
temp/
|
|
5
|
+
logs/
|
|
6
|
+
# The broad `logs/` rule above is for RUNTIME log output, but it also matched
|
|
7
|
+
# the dashboard's SOURCE directory and silently swallowed an entire feature's
|
|
8
|
+
# files (only the pre-existing index.tsx stayed tracked), breaking the prod
|
|
9
|
+
# Docker build with "Could not resolve ./structured-tab". Re-include the source.
|
|
10
|
+
!apps/dashboard/src/features/logs/
|
|
11
|
+
!apps/dashboard/src/features/logs/**
|
|
12
|
+
todo
|
|
13
|
+
text_notes/
|
|
14
|
+
aidream/secrets/2.env
|
|
15
|
+
automation_matrix/matrix_processing/temp/*
|
|
16
|
+
cd
|
|
17
|
+
# Byte-compiled / optimized / DLL files
|
|
18
|
+
__pycache__/
|
|
19
|
+
*.py[cod]
|
|
20
|
+
*$py.class
|
|
21
|
+
|
|
22
|
+
# C extensions
|
|
23
|
+
*.so
|
|
24
|
+
.venv/
|
|
25
|
+
|
|
26
|
+
# Distribution / packaging
|
|
27
|
+
.Python
|
|
28
|
+
build/
|
|
29
|
+
develop-eggs/
|
|
30
|
+
dist/
|
|
31
|
+
downloads/
|
|
32
|
+
eggs/
|
|
33
|
+
.eggs/
|
|
34
|
+
lib/
|
|
35
|
+
lib64/
|
|
36
|
+
# The blanket lib/ rule above is from the standard Python .gitignore template
|
|
37
|
+
# and was silently swallowing TS source under the SPA `src/lib/` folders.
|
|
38
|
+
# Re-allow them explicitly so frontend builds don't ship without their lib layer.
|
|
39
|
+
!apps/dashboard/src/lib/
|
|
40
|
+
!apps/dashboard/src/lib/**
|
|
41
|
+
!apps/workflow-studio/src/lib/
|
|
42
|
+
!apps/workflow-studio/src/lib/**
|
|
43
|
+
parts/
|
|
44
|
+
sdist/
|
|
45
|
+
var/
|
|
46
|
+
wheels/
|
|
47
|
+
share/python-wheels/
|
|
48
|
+
*.egg-info/
|
|
49
|
+
.installed.cfg
|
|
50
|
+
*.egg
|
|
51
|
+
MANIFEST
|
|
52
|
+
|
|
53
|
+
# PyInstaller
|
|
54
|
+
# Usually these files are written by a python script from a template
|
|
55
|
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
|
56
|
+
*.manifest
|
|
57
|
+
*.spec
|
|
58
|
+
|
|
59
|
+
# Installer logs
|
|
60
|
+
pip-log.txt
|
|
61
|
+
pip-delete-this-directory.txt
|
|
62
|
+
|
|
63
|
+
# Unit test / coverage reports
|
|
64
|
+
ai/tests/clean_response.json
|
|
65
|
+
ai/tests/cx_storage_response.json
|
|
66
|
+
ai/tests/execution_test.py
|
|
67
|
+
ai/tests/final_response.json
|
|
68
|
+
htmlcov/
|
|
69
|
+
.tox/
|
|
70
|
+
.nox/
|
|
71
|
+
.coverage
|
|
72
|
+
.coverage.*
|
|
73
|
+
.cache
|
|
74
|
+
nosetests.xml
|
|
75
|
+
coverage.xml
|
|
76
|
+
*.cover
|
|
77
|
+
*.py,cover
|
|
78
|
+
.hypothesis/
|
|
79
|
+
.pytest_cache/
|
|
80
|
+
cover/
|
|
81
|
+
|
|
82
|
+
# Translations
|
|
83
|
+
*.mo
|
|
84
|
+
*.pot
|
|
85
|
+
|
|
86
|
+
# Django stuff:
|
|
87
|
+
*.log
|
|
88
|
+
local_settings.py
|
|
89
|
+
db.sqlite3
|
|
90
|
+
db.sqlite3-journal
|
|
91
|
+
|
|
92
|
+
# Flask stuff:
|
|
93
|
+
instance/
|
|
94
|
+
.webassets-cache
|
|
95
|
+
|
|
96
|
+
# Scrapy stuff:
|
|
97
|
+
.scrapy
|
|
98
|
+
|
|
99
|
+
# Sphinx documentation
|
|
100
|
+
docs/_build/
|
|
101
|
+
|
|
102
|
+
# PyBuilder
|
|
103
|
+
.pybuilder/
|
|
104
|
+
target/
|
|
105
|
+
|
|
106
|
+
# Jupyter Notebook
|
|
107
|
+
.ipynb_checkpoints
|
|
108
|
+
|
|
109
|
+
# IPython
|
|
110
|
+
profile_default/
|
|
111
|
+
ipython_config.py
|
|
112
|
+
|
|
113
|
+
# pyenv
|
|
114
|
+
# For a library or package, you might want to ignore these files since the code is
|
|
115
|
+
# intended to run in multiple environments; otherwise, check them in:
|
|
116
|
+
# .python-version
|
|
117
|
+
|
|
118
|
+
# pipenv
|
|
119
|
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
|
120
|
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
|
121
|
+
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
|
122
|
+
# install all needed dependencies.
|
|
123
|
+
#Pipfile.lock
|
|
124
|
+
|
|
125
|
+
# poetry
|
|
126
|
+
# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
|
|
127
|
+
# This is especially recommended for binary packages to ensure reproducibility, and is more
|
|
128
|
+
# commonly ignored for libraries.
|
|
129
|
+
# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
|
|
130
|
+
|
|
131
|
+
# pdm
|
|
132
|
+
# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
|
|
133
|
+
#pdm.lock
|
|
134
|
+
# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
|
|
135
|
+
# in version control.
|
|
136
|
+
# https://pdm.fming.dev/#use-with-ide
|
|
137
|
+
.pdm.toml
|
|
138
|
+
|
|
139
|
+
# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
|
|
140
|
+
__pypackages__/
|
|
141
|
+
|
|
142
|
+
# Celery stuff
|
|
143
|
+
celerybeat-schedule
|
|
144
|
+
celerybeat.pid
|
|
145
|
+
|
|
146
|
+
# SageMath parsed files
|
|
147
|
+
*.sage.py
|
|
148
|
+
|
|
149
|
+
# Environments
|
|
150
|
+
.env
|
|
151
|
+
.env_remote
|
|
152
|
+
.venv
|
|
153
|
+
env/
|
|
154
|
+
venv/
|
|
155
|
+
ENV/
|
|
156
|
+
env.bak/
|
|
157
|
+
venv.bak/
|
|
158
|
+
.env.armanonly
|
|
159
|
+
|
|
160
|
+
# Spyder project settings
|
|
161
|
+
.spyderproject
|
|
162
|
+
.spyproject
|
|
163
|
+
|
|
164
|
+
# Rope project settings
|
|
165
|
+
.ropeproject
|
|
166
|
+
|
|
167
|
+
# mkdocs documentation
|
|
168
|
+
/site
|
|
169
|
+
|
|
170
|
+
# mypy
|
|
171
|
+
.mypy_cache/
|
|
172
|
+
.dmypy.json
|
|
173
|
+
dmypy.json
|
|
174
|
+
|
|
175
|
+
# Pyre type checker
|
|
176
|
+
.pyre/
|
|
177
|
+
|
|
178
|
+
# random armani files
|
|
179
|
+
/armani_dev/secrets/
|
|
180
|
+
/armani/
|
|
181
|
+
/_armani/
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
# pytype static type analyzer
|
|
186
|
+
.pytype/
|
|
187
|
+
|
|
188
|
+
# Cython debug symbols
|
|
189
|
+
cython_debug/
|
|
190
|
+
|
|
191
|
+
.idea/
|
|
192
|
+
.vscode/
|
|
193
|
+
/node_modules/
|
|
194
|
+
|
|
195
|
+
# Frontend pnpm workspace (apps/) — node_modules at the workspace root and any
|
|
196
|
+
# member, plus Vite caches and build output. The unified lockfile (apps/pnpm-lock.yaml)
|
|
197
|
+
# IS committed; everything below is regenerated.
|
|
198
|
+
node_modules/
|
|
199
|
+
apps/**/.vite/
|
|
200
|
+
apps/**/dist/
|
|
201
|
+
.vite/
|
|
202
|
+
|
|
203
|
+
dump.rdb
|
|
204
|
+
|
|
205
|
+
frontend/
|
|
206
|
+
|
|
207
|
+
# AME Temp Files and directory structure
|
|
208
|
+
# Ignore all files in the temp directory and its subdirectories
|
|
209
|
+
/temp/**/*
|
|
210
|
+
/tmp/**/*
|
|
211
|
+
|
|
212
|
+
# Allow .gitkeep files to retain directory structure
|
|
213
|
+
!/temp/**/.gitkeep
|
|
214
|
+
!/tmp/**/.gitkeep
|
|
215
|
+
|
|
216
|
+
# Armani
|
|
217
|
+
.history*
|
|
218
|
+
.history/
|
|
219
|
+
local_data/
|
|
220
|
+
local_reports_data/
|
|
221
|
+
webscraper/quick_scrapes/temp/
|
|
222
|
+
automation_matrix/ai_apis/fireworks/_dev/*
|
|
223
|
+
automation_matrix/ai_apis/fireworks/_dev/fireworks_sample.py
|
|
224
|
+
*.pdf
|
|
225
|
+
*.flac
|
|
226
|
+
*.mp3
|
|
227
|
+
*.wav
|
|
228
|
+
miniconda.sh
|
|
229
|
+
/database/python_sql/temp_data/
|
|
230
|
+
.history*
|
|
231
|
+
.history/
|
|
232
|
+
.history/
|
|
233
|
+
|
|
234
|
+
_dev/
|
|
235
|
+
/_dev/
|
|
236
|
+
requirements_filtered.txt
|
|
237
|
+
|
|
238
|
+
# matrx-dev-tools backups
|
|
239
|
+
.env-backups/
|
|
240
|
+
# Matrx Ship config (contains API key)
|
|
241
|
+
.matrx-ship.json
|
|
242
|
+
|
|
243
|
+
# Matrx config (contains API keys)
|
|
244
|
+
.matrx.json
|
|
245
|
+
.matrx-tools.conf
|
|
246
|
+
|
|
247
|
+
# Claude Code local worktrees and per-user settings
|
|
248
|
+
.claude/worktrees/
|
|
249
|
+
.claude/settings.local.json
|
|
250
|
+
|
|
251
|
+
# Append-only snapshots from matrx_utils.update_history (unbounded; do not commit)
|
|
252
|
+
common/utils/data_in_code/data_history.json
|
|
253
|
+
packages/matrx-utils/matrx_utils/data_in_code/data_history.json
|
|
254
|
+
|
|
255
|
+
# Tool-dispatch debug logs — one file per server start, never committed
|
|
256
|
+
.matrx-debug/
|
|
257
|
+
|
|
258
|
+
# macOS Finder metadata
|
|
259
|
+
.DS_Store
|
|
260
|
+
**/.DS_Store
|
|
261
|
+
|
|
262
|
+
# Environment files
|
|
263
|
+
.env
|
|
264
|
+
.env.*
|
|
265
|
+
*.env
|
|
266
|
+
*.env.*
|
|
267
|
+
|
|
268
|
+
# Keep safe templates trackable
|
|
269
|
+
!.env.example
|
|
270
|
+
!.env.sample
|
|
271
|
+
!.env.template
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
# CLAUDE.md — matrx-batch
|
|
2
|
+
|
|
3
|
+
> **Operating Principle: Build the platform, not the artifact.** Every task is a probe that exposes a missing capability — build it, then consume it. Full doctrine: [/PRINCIPLES.md](../../PRINCIPLES.md).
|
|
4
|
+
|
|
5
|
+
**Package:** `matrx-batch` (PyPI) — Python 3.13+ — currently v0.1.0
|
|
6
|
+
**Role in the graph:** One notch above `matrx-utils`. Depends on
|
|
7
|
+
`matrx-utils` only. Consumed by `matrx-rag` (embedding cache) and
|
|
8
|
+
`matrx-ai` (urgency router). MUST NOT depend on any other matrx-*
|
|
9
|
+
sibling at module-import time.
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
## What this package owns
|
|
14
|
+
|
|
15
|
+
Three primitives, one mission: make AI cost shape predictable.
|
|
16
|
+
|
|
17
|
+
1. **`openai_batch.py`** — async wrapper over OpenAI's Batch API.
|
|
18
|
+
`submit_batch(jobs)` → uploads JSONL via `files.create(purpose="batch")`,
|
|
19
|
+
then `batches.create()`. `poll_batch(id)` returns a typed status.
|
|
20
|
+
`fetch_results(id)` parses the output JSONL into typed results.
|
|
21
|
+
2. **`anthropic_batch.py`** — same shape over Anthropic Message Batches
|
|
22
|
+
(`messages.batches.create()`).
|
|
23
|
+
3. **`router.py`** — `BatchRouter.submit(job)` decides live vs batch
|
|
24
|
+
based on `job.urgency`. `live` → direct provider call. `batch` →
|
|
25
|
+
enqueue. `auto` → cost-based (default threshold: >5000 estimated
|
|
26
|
+
tokens → batch).
|
|
27
|
+
4. **`embedding_cache.py`** — `EmbeddingCache.get_or_compute(text,
|
|
28
|
+
model, organization_id=...)` reads/writes `rag.embedding_cache`.
|
|
29
|
+
SHA256(`f"{model}::{text}"`) keyed. Increments `hit_count` on every
|
|
30
|
+
reuse, on matrx-orm — see rule 4.
|
|
31
|
+
|
|
32
|
+
---
|
|
33
|
+
|
|
34
|
+
## The hard rules
|
|
35
|
+
|
|
36
|
+
1. **No imports from `aidream/` or from any other matrx-* sibling.** This
|
|
37
|
+
is the Package Independence Rule (root CLAUDE.md). If matrx-batch
|
|
38
|
+
needs ORM models, settings, or a context resolver, it accepts them
|
|
39
|
+
via `configure(...)`.
|
|
40
|
+
2. **No top-level provider SDK imports.** The `openai` / `anthropic`
|
|
41
|
+
imports happen lazily inside method bodies. Hosts that swap the
|
|
42
|
+
provider clients via `configure(...)` never pay the SDK import cost.
|
|
43
|
+
3. **No new asyncpg pool.** The package never opens its own connection.
|
|
44
|
+
4. **Every DB access goes through matrx-orm. There is no exception here.**
|
|
45
|
+
A package that touches a database depends on matrx-orm — always, no
|
|
46
|
+
package is exempt. "It's a sibling package, so it can't use the ORM"
|
|
47
|
+
is FALSE: models arrive by name through the injection seam
|
|
48
|
+
(`get_db_model(...)` / a `db/db_requirements.py` manifest), exactly as
|
|
49
|
+
matrx-rag and matrx-graph do. **Canonical doctrine — read it before
|
|
50
|
+
touching any DB code here:**
|
|
51
|
+
[`.claude/skills/eliminate-raw-sql/SKILL.md`](../../.claude/skills/eliminate-raw-sql/SKILL.md).
|
|
52
|
+
- `embedding_cache.py`'s `rag.embedding_cache` access is **fully
|
|
53
|
+
converted** — zero raw SQL remains in this package.
|
|
54
|
+
|
|
55
|
+
---
|
|
56
|
+
|
|
57
|
+
## The `configure()` injection pattern
|
|
58
|
+
|
|
59
|
+
```python
|
|
60
|
+
import matrx_batch
|
|
61
|
+
|
|
62
|
+
matrx_batch.configure(
|
|
63
|
+
openai_client=AsyncOpenAI(api_key=...), # provider client
|
|
64
|
+
anthropic_client=AsyncAnthropic(api_key=...),
|
|
65
|
+
pool_factory=async_callable_returning_asyncpg_pool,
|
|
66
|
+
cost_check=async_callable_for_budget_precheck, # optional
|
|
67
|
+
# auto_threshold_tokens=5000, # optional override
|
|
68
|
+
)
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Hosts that don't configure a slot get a sensible fallback or a
|
|
72
|
+
loud `BatchNotConfiguredError` at call time (never at import time —
|
|
73
|
+
`import matrx_batch` always succeeds in a minimal environment).
|
|
74
|
+
|
|
75
|
+
When adding a new injection point:
|
|
76
|
+
|
|
77
|
+
1. Add a new keyword arg to `configure(...)` in `__init__.py`.
|
|
78
|
+
2. Store it in `_ext.py`'s registry.
|
|
79
|
+
3. Access via `get_ext("name")` at the call site (NOT at module top).
|
|
80
|
+
4. Document the slot here and in the host's `package_integration.py`.
|
|
81
|
+
|
|
82
|
+
---
|
|
83
|
+
|
|
84
|
+
## Dependency rules specific to this package
|
|
85
|
+
|
|
86
|
+
- ✅ `from matrx_utils import ...` — declared workspace dep.
|
|
87
|
+
- ✅ `from openai import ...`, `from anthropic import ...` — declared
|
|
88
|
+
runtime deps; import lazily.
|
|
89
|
+
- ✅ `from matrx_orm import ...` — matrx-batch touches a database, so it
|
|
90
|
+
uses the ORM like every other package. Concrete host models resolve by
|
|
91
|
+
name through the injection seam, never by importing host model files.
|
|
92
|
+
- ❌ No hand-written SQL over a raw pool. (The legacy `asyncpg` pool in
|
|
93
|
+
`embedding_cache.py` is debt being burned down — do not add to it.)
|
|
94
|
+
- ❌ No `from matrx_rag import ...` or `from matrx_ai import ...`
|
|
95
|
+
(they consume us, not the other way around — would create a cycle).
|
|
96
|
+
- ❌ No `from aidream import ...` or any root-module imports.
|
|
97
|
+
|
|
98
|
+
---
|
|
99
|
+
|
|
100
|
+
## Testing this package in isolation
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
uv run pytest packages/matrx-batch/tests
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Tests run with mocked providers and a mocked / in-memory pool. No live
|
|
107
|
+
DB or provider keys required. Real-DB integration coverage lives in
|
|
108
|
+
aidream's test suite.
|
|
109
|
+
|
|
110
|
+
---
|
|
111
|
+
|
|
112
|
+
## What lives where
|
|
113
|
+
|
|
114
|
+
| File | Owns |
|
|
115
|
+
|---|---|
|
|
116
|
+
| `__init__.py` | Public API: `configure()`, `BatchRouter`, `BatchableJob`, `EmbeddingCache`, `get_embedding_cache()` |
|
|
117
|
+
| `_ext.py` | Injection registry — host config storage |
|
|
118
|
+
| `openai_batch.py` | OpenAI Batch submit / poll / fetch |
|
|
119
|
+
| `anthropic_batch.py` | Anthropic Message Batches submit / poll / fetch |
|
|
120
|
+
| `router.py` | Urgency-based routing decision |
|
|
121
|
+
| `embedding_cache.py` | `rag.embedding_cache` read/write + hit_count |
|
|
122
|
+
| `tests/test_embedding_cache.py` | Cache hit/miss/hit_count verification |
|
|
123
|
+
| `tests/test_router.py` | Urgency routing decision |
|
|
124
|
+
|
|
125
|
+
---
|
|
126
|
+
|
|
127
|
+
## Known gotchas
|
|
128
|
+
|
|
129
|
+
- Batch APIs have a 24h SLA. Never block inference on a batch
|
|
130
|
+
submission — the router always returns a `JobReceipt` immediately;
|
|
131
|
+
consumers poll the receipt or wire a webhook.
|
|
132
|
+
- The embedding cache is content-addressable. Two orgs that produce
|
|
133
|
+
identical text get one row. `organization_id` is stored for cleanup
|
|
134
|
+
queries but is NOT part of the cache key.
|
|
135
|
+
- `hit_count` is incremented in the SAME UPDATE that touches `created_at`,
|
|
136
|
+
so a high-traffic key is also a hot row. Acceptable: the index leaf
|
|
137
|
+
doesn't move because the PK is fixed.
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: matrx-batch
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Cost-shield foundation for AI workloads: OpenAI + Anthropic Batch APIs, live/batch urgency router, shared embedding cache.
|
|
5
|
+
Author-email: Matrx <admin@aimatrx.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: anthropic,batch,cost,embeddings,matrx,openai
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
13
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
14
|
+
Requires-Python: >=3.13
|
|
15
|
+
Requires-Dist: anthropic>=0.40
|
|
16
|
+
Requires-Dist: matrx-orm
|
|
17
|
+
Requires-Dist: matrx-utils
|
|
18
|
+
Requires-Dist: openai>=2.0
|
|
19
|
+
Requires-Dist: pydantic>=2.12
|
|
20
|
+
Provides-Extra: dev
|
|
21
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
22
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
|
|
25
|
+
# matrx-batch
|
|
26
|
+
|
|
27
|
+
Cost-shield foundation for AI workloads in the Matrx ecosystem.
|
|
28
|
+
|
|
29
|
+
Three primitives:
|
|
30
|
+
|
|
31
|
+
1. **OpenAI Batch API + Anthropic Message Batches** — submit JSONL, poll
|
|
32
|
+
status, fetch results. Async wrappers over the provider SDKs.
|
|
33
|
+
2. **Urgency router** — `BatchRouter.submit(job)` routes a job to live
|
|
34
|
+
inference, to the provider's Batch API (50% discount, 24h SLA), or
|
|
35
|
+
makes a cost-based decision automatically.
|
|
36
|
+
3. **Embedding cache** — content-addressable (SHA256 of `model || text`)
|
|
37
|
+
reuse of embedding vectors across the org. Backed by
|
|
38
|
+
`rag.embedding_cache` via a host-supplied asyncpg pool.
|
|
39
|
+
|
|
40
|
+
## Dependency posture
|
|
41
|
+
|
|
42
|
+
`matrx-batch` sits one notch above `matrx-utils` in the dependency graph
|
|
43
|
+
and is consumed by `matrx-rag` (embedding cache) and `matrx-ai` (urgency
|
|
44
|
+
router, future). It **MUST NOT** import from `aidream/` or from any other
|
|
45
|
+
`matrx-*` sibling — every cross-boundary dependency comes in via
|
|
46
|
+
`matrx_batch.configure(...)`.
|
|
47
|
+
|
|
48
|
+
## Usage (host-side wiring)
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
import matrx_batch
|
|
52
|
+
from openai import AsyncOpenAI
|
|
53
|
+
from anthropic import AsyncAnthropic
|
|
54
|
+
|
|
55
|
+
matrx_batch.configure(
|
|
56
|
+
openai_client=AsyncOpenAI(api_key=settings.OPENAI_API_KEY),
|
|
57
|
+
anthropic_client=AsyncAnthropic(api_key=settings.ANTHROPIC_API_KEY),
|
|
58
|
+
pool_factory=lambda: get_async_pg_pool(),
|
|
59
|
+
cost_check=_aidream_budget_precheck,
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
# Wire the cache into matrx-rag's embed() funnel
|
|
63
|
+
import matrx_rag
|
|
64
|
+
matrx_rag.configure(embedding_cache=matrx_batch.get_embedding_cache())
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
## Usage (consumer)
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
from matrx_batch import BatchRouter, BatchableJob
|
|
71
|
+
|
|
72
|
+
router = BatchRouter()
|
|
73
|
+
receipt = await router.submit(
|
|
74
|
+
BatchableJob(
|
|
75
|
+
kind="chat",
|
|
76
|
+
provider="anthropic",
|
|
77
|
+
urgency="auto",
|
|
78
|
+
payload={"model": "claude-haiku-4-5", "messages": [...]},
|
|
79
|
+
estimated_tokens_in=8200,
|
|
80
|
+
)
|
|
81
|
+
)
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
## See also
|
|
85
|
+
|
|
86
|
+
- `CLAUDE.md` — package rules and injection-point contract.
|
|
87
|
+
- `matrx_batch/embedding_cache.py` — chunk-level embedding reuse.
|
|
88
|
+
- `matrx_batch/router.py` — urgency routing decisions.
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# matrx-batch
|
|
2
|
+
|
|
3
|
+
Cost-shield foundation for AI workloads in the Matrx ecosystem.
|
|
4
|
+
|
|
5
|
+
Three primitives:
|
|
6
|
+
|
|
7
|
+
1. **OpenAI Batch API + Anthropic Message Batches** — submit JSONL, poll
|
|
8
|
+
status, fetch results. Async wrappers over the provider SDKs.
|
|
9
|
+
2. **Urgency router** — `BatchRouter.submit(job)` routes a job to live
|
|
10
|
+
inference, to the provider's Batch API (50% discount, 24h SLA), or
|
|
11
|
+
makes a cost-based decision automatically.
|
|
12
|
+
3. **Embedding cache** — content-addressable (SHA256 of `model || text`)
|
|
13
|
+
reuse of embedding vectors across the org. Backed by
|
|
14
|
+
`rag.embedding_cache` via a host-supplied asyncpg pool.
|
|
15
|
+
|
|
16
|
+
## Dependency posture
|
|
17
|
+
|
|
18
|
+
`matrx-batch` sits one notch above `matrx-utils` in the dependency graph
|
|
19
|
+
and is consumed by `matrx-rag` (embedding cache) and `matrx-ai` (urgency
|
|
20
|
+
router, future). It **MUST NOT** import from `aidream/` or from any other
|
|
21
|
+
`matrx-*` sibling — every cross-boundary dependency comes in via
|
|
22
|
+
`matrx_batch.configure(...)`.
|
|
23
|
+
|
|
24
|
+
## Usage (host-side wiring)
|
|
25
|
+
|
|
26
|
+
```python
|
|
27
|
+
import matrx_batch
|
|
28
|
+
from openai import AsyncOpenAI
|
|
29
|
+
from anthropic import AsyncAnthropic
|
|
30
|
+
|
|
31
|
+
matrx_batch.configure(
|
|
32
|
+
openai_client=AsyncOpenAI(api_key=settings.OPENAI_API_KEY),
|
|
33
|
+
anthropic_client=AsyncAnthropic(api_key=settings.ANTHROPIC_API_KEY),
|
|
34
|
+
pool_factory=lambda: get_async_pg_pool(),
|
|
35
|
+
cost_check=_aidream_budget_precheck,
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
# Wire the cache into matrx-rag's embed() funnel
|
|
39
|
+
import matrx_rag
|
|
40
|
+
matrx_rag.configure(embedding_cache=matrx_batch.get_embedding_cache())
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
## Usage (consumer)
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from matrx_batch import BatchRouter, BatchableJob
|
|
47
|
+
|
|
48
|
+
router = BatchRouter()
|
|
49
|
+
receipt = await router.submit(
|
|
50
|
+
BatchableJob(
|
|
51
|
+
kind="chat",
|
|
52
|
+
provider="anthropic",
|
|
53
|
+
urgency="auto",
|
|
54
|
+
payload={"model": "claude-haiku-4-5", "messages": [...]},
|
|
55
|
+
estimated_tokens_in=8200,
|
|
56
|
+
)
|
|
57
|
+
)
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## See also
|
|
61
|
+
|
|
62
|
+
- `CLAUDE.md` — package rules and injection-point contract.
|
|
63
|
+
- `matrx_batch/embedding_cache.py` — chunk-level embedding reuse.
|
|
64
|
+
- `matrx_batch/router.py` — urgency routing decisions.
|