matrx-batch 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,271 @@
1
+ *.pyc
2
+ secrets/
3
+ ignore/
4
+ temp/
5
+ logs/
6
+ # The broad `logs/` rule above is for RUNTIME log output, but it also matched
7
+ # the dashboard's SOURCE directory and silently swallowed an entire feature's
8
+ # files (only the pre-existing index.tsx stayed tracked), breaking the prod
9
+ # Docker build with "Could not resolve ./structured-tab". Re-include the source.
10
+ !apps/dashboard/src/features/logs/
11
+ !apps/dashboard/src/features/logs/**
12
+ todo
13
+ text_notes/
14
+ aidream/secrets/2.env
15
+ automation_matrix/matrix_processing/temp/*
16
+ cd
17
+ # Byte-compiled / optimized / DLL files
18
+ __pycache__/
19
+ *.py[cod]
20
+ *$py.class
21
+
22
+ # C extensions
23
+ *.so
24
+ .venv/
25
+
26
+ # Distribution / packaging
27
+ .Python
28
+ build/
29
+ develop-eggs/
30
+ dist/
31
+ downloads/
32
+ eggs/
33
+ .eggs/
34
+ lib/
35
+ lib64/
36
+ # The blanket lib/ rule above is from the standard Python .gitignore template
37
+ # and was silently swallowing TS source under the SPA `src/lib/` folders.
38
+ # Re-allow them explicitly so frontend builds don't ship without their lib layer.
39
+ !apps/dashboard/src/lib/
40
+ !apps/dashboard/src/lib/**
41
+ !apps/workflow-studio/src/lib/
42
+ !apps/workflow-studio/src/lib/**
43
+ parts/
44
+ sdist/
45
+ var/
46
+ wheels/
47
+ share/python-wheels/
48
+ *.egg-info/
49
+ .installed.cfg
50
+ *.egg
51
+ MANIFEST
52
+
53
+ # PyInstaller
54
+ # Usually these files are written by a python script from a template
55
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
56
+ *.manifest
57
+ *.spec
58
+
59
+ # Installer logs
60
+ pip-log.txt
61
+ pip-delete-this-directory.txt
62
+
63
+ # Unit test / coverage reports
64
+ ai/tests/clean_response.json
65
+ ai/tests/cx_storage_response.json
66
+ ai/tests/execution_test.py
67
+ ai/tests/final_response.json
68
+ htmlcov/
69
+ .tox/
70
+ .nox/
71
+ .coverage
72
+ .coverage.*
73
+ .cache
74
+ nosetests.xml
75
+ coverage.xml
76
+ *.cover
77
+ *.py,cover
78
+ .hypothesis/
79
+ .pytest_cache/
80
+ cover/
81
+
82
+ # Translations
83
+ *.mo
84
+ *.pot
85
+
86
+ # Django stuff:
87
+ *.log
88
+ local_settings.py
89
+ db.sqlite3
90
+ db.sqlite3-journal
91
+
92
+ # Flask stuff:
93
+ instance/
94
+ .webassets-cache
95
+
96
+ # Scrapy stuff:
97
+ .scrapy
98
+
99
+ # Sphinx documentation
100
+ docs/_build/
101
+
102
+ # PyBuilder
103
+ .pybuilder/
104
+ target/
105
+
106
+ # Jupyter Notebook
107
+ .ipynb_checkpoints
108
+
109
+ # IPython
110
+ profile_default/
111
+ ipython_config.py
112
+
113
+ # pyenv
114
+ # For a library or package, you might want to ignore these files since the code is
115
+ # intended to run in multiple environments; otherwise, check them in:
116
+ # .python-version
117
+
118
+ # pipenv
119
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
120
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
121
+ # having no cross-platform support, pipenv may install dependencies that don't work, or not
122
+ # install all needed dependencies.
123
+ #Pipfile.lock
124
+
125
+ # poetry
126
+ # Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control.
127
+ # This is especially recommended for binary packages to ensure reproducibility, and is more
128
+ # commonly ignored for libraries.
129
+ # https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control
130
+
131
+ # pdm
132
+ # Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control.
133
+ #pdm.lock
134
+ # pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it
135
+ # in version control.
136
+ # https://pdm.fming.dev/#use-with-ide
137
+ .pdm.toml
138
+
139
+ # PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm
140
+ __pypackages__/
141
+
142
+ # Celery stuff
143
+ celerybeat-schedule
144
+ celerybeat.pid
145
+
146
+ # SageMath parsed files
147
+ *.sage.py
148
+
149
+ # Environments
150
+ .env
151
+ .env_remote
152
+ .venv
153
+ env/
154
+ venv/
155
+ ENV/
156
+ env.bak/
157
+ venv.bak/
158
+ .env.armanonly
159
+
160
+ # Spyder project settings
161
+ .spyderproject
162
+ .spyproject
163
+
164
+ # Rope project settings
165
+ .ropeproject
166
+
167
+ # mkdocs documentation
168
+ /site
169
+
170
+ # mypy
171
+ .mypy_cache/
172
+ .dmypy.json
173
+ dmypy.json
174
+
175
+ # Pyre type checker
176
+ .pyre/
177
+
178
+ # random armani files
179
+ /armani_dev/secrets/
180
+ /armani/
181
+ /_armani/
182
+
183
+
184
+
185
+ # pytype static type analyzer
186
+ .pytype/
187
+
188
+ # Cython debug symbols
189
+ cython_debug/
190
+
191
+ .idea/
192
+ .vscode/
193
+ /node_modules/
194
+
195
+ # Frontend pnpm workspace (apps/) — node_modules at the workspace root and any
196
+ # member, plus Vite caches and build output. The unified lockfile (apps/pnpm-lock.yaml)
197
+ # IS committed; everything below is regenerated.
198
+ node_modules/
199
+ apps/**/.vite/
200
+ apps/**/dist/
201
+ .vite/
202
+
203
+ dump.rdb
204
+
205
+ frontend/
206
+
207
+ # AME Temp Files and directory structure
208
+ # Ignore all files in the temp directory and its subdirectories
209
+ /temp/**/*
210
+ /tmp/**/*
211
+
212
+ # Allow .gitkeep files to retain directory structure
213
+ !/temp/**/.gitkeep
214
+ !/tmp/**/.gitkeep
215
+
216
+ # Armani
217
+ .history*
218
+ .history/
219
+ local_data/
220
+ local_reports_data/
221
+ webscraper/quick_scrapes/temp/
222
+ automation_matrix/ai_apis/fireworks/_dev/*
223
+ automation_matrix/ai_apis/fireworks/_dev/fireworks_sample.py
224
+ *.pdf
225
+ *.flac
226
+ *.mp3
227
+ *.wav
228
+ miniconda.sh
229
+ /database/python_sql/temp_data/
230
+ .history*
231
+ .history/
232
+ .history/
233
+
234
+ _dev/
235
+ /_dev/
236
+ requirements_filtered.txt
237
+
238
+ # matrx-dev-tools backups
239
+ .env-backups/
240
+ # Matrx Ship config (contains API key)
241
+ .matrx-ship.json
242
+
243
+ # Matrx config (contains API keys)
244
+ .matrx.json
245
+ .matrx-tools.conf
246
+
247
+ # Claude Code local worktrees and per-user settings
248
+ .claude/worktrees/
249
+ .claude/settings.local.json
250
+
251
+ # Append-only snapshots from matrx_utils.update_history (unbounded; do not commit)
252
+ common/utils/data_in_code/data_history.json
253
+ packages/matrx-utils/matrx_utils/data_in_code/data_history.json
254
+
255
+ # Tool-dispatch debug logs — one file per server start, never committed
256
+ .matrx-debug/
257
+
258
+ # macOS Finder metadata
259
+ .DS_Store
260
+ **/.DS_Store
261
+
262
+ # Environment files
263
+ .env
264
+ .env.*
265
+ *.env
266
+ *.env.*
267
+
268
+ # Keep safe templates trackable
269
+ !.env.example
270
+ !.env.sample
271
+ !.env.template
@@ -0,0 +1,137 @@
1
+ # CLAUDE.md — matrx-batch
2
+
3
+ > **Operating Principle: Build the platform, not the artifact.** Every task is a probe that exposes a missing capability — build it, then consume it. Full doctrine: [/PRINCIPLES.md](../../PRINCIPLES.md).
4
+
5
+ **Package:** `matrx-batch` (PyPI) — Python 3.13+ — currently v0.1.0
6
+ **Role in the graph:** One notch above `matrx-utils`. Depends on
7
+ `matrx-utils` only. Consumed by `matrx-rag` (embedding cache) and
8
+ `matrx-ai` (urgency router). MUST NOT depend on any other matrx-*
9
+ sibling at module-import time.
10
+
11
+ ---
12
+
13
+ ## What this package owns
14
+
15
+ Three primitives, one mission: make AI cost shape predictable.
16
+
17
+ 1. **`openai_batch.py`** — async wrapper over OpenAI's Batch API.
18
+ `submit_batch(jobs)` → uploads JSONL via `files.create(purpose="batch")`,
19
+ then `batches.create()`. `poll_batch(id)` returns a typed status.
20
+ `fetch_results(id)` parses the output JSONL into typed results.
21
+ 2. **`anthropic_batch.py`** — same shape over Anthropic Message Batches
22
+ (`messages.batches.create()`).
23
+ 3. **`router.py`** — `BatchRouter.submit(job)` decides live vs batch
24
+ based on `job.urgency`. `live` → direct provider call. `batch` →
25
+ enqueue. `auto` → cost-based (default threshold: >5000 estimated
26
+ tokens → batch).
27
+ 4. **`embedding_cache.py`** — `EmbeddingCache.get_or_compute(text,
28
+ model, organization_id=...)` reads/writes `rag.embedding_cache`.
29
+ SHA256(`f"{model}::{text}"`) keyed. Increments `hit_count` on every
30
+ reuse, on matrx-orm — see rule 4.
31
+
32
+ ---
33
+
34
+ ## The hard rules
35
+
36
+ 1. **No imports from `aidream/` or from any other matrx-* sibling.** This
37
+ is the Package Independence Rule (root CLAUDE.md). If matrx-batch
38
+ needs ORM models, settings, or a context resolver, it accepts them
39
+ via `configure(...)`.
40
+ 2. **No top-level provider SDK imports.** The `openai` / `anthropic`
41
+ imports happen lazily inside method bodies. Hosts that swap the
42
+ provider clients via `configure(...)` never pay the SDK import cost.
43
+ 3. **No new asyncpg pool.** The package never opens its own connection.
44
+ 4. **Every DB access goes through matrx-orm. There is no exception here.**
45
+ A package that touches a database depends on matrx-orm — always, no
46
+ package is exempt. "It's a sibling package, so it can't use the ORM"
47
+ is FALSE: models arrive by name through the injection seam
48
+ (`get_db_model(...)` / a `db/db_requirements.py` manifest), exactly as
49
+ matrx-rag and matrx-graph do. **Canonical doctrine — read it before
50
+ touching any DB code here:**
51
+ [`.claude/skills/eliminate-raw-sql/SKILL.md`](../../.claude/skills/eliminate-raw-sql/SKILL.md).
52
+ - `embedding_cache.py`'s `rag.embedding_cache` access is **fully
53
+ converted** — zero raw SQL remains in this package.
54
+
55
+ ---
56
+
57
+ ## The `configure()` injection pattern
58
+
59
+ ```python
60
+ import matrx_batch
61
+
62
+ matrx_batch.configure(
63
+ openai_client=AsyncOpenAI(api_key=...), # provider client
64
+ anthropic_client=AsyncAnthropic(api_key=...),
65
+ pool_factory=async_callable_returning_asyncpg_pool,
66
+ cost_check=async_callable_for_budget_precheck, # optional
67
+ # auto_threshold_tokens=5000, # optional override
68
+ )
69
+ ```
70
+
71
+ Hosts that don't configure a slot get a sensible fallback or a
72
+ loud `BatchNotConfiguredError` at call time (never at import time —
73
+ `import matrx_batch` always succeeds in a minimal environment).
74
+
75
+ When adding a new injection point:
76
+
77
+ 1. Add a new keyword arg to `configure(...)` in `__init__.py`.
78
+ 2. Store it in `_ext.py`'s registry.
79
+ 3. Access via `get_ext("name")` at the call site (NOT at module top).
80
+ 4. Document the slot here and in the host's `package_integration.py`.
81
+
82
+ ---
83
+
84
+ ## Dependency rules specific to this package
85
+
86
+ - ✅ `from matrx_utils import ...` — declared workspace dep.
87
+ - ✅ `from openai import ...`, `from anthropic import ...` — declared
88
+ runtime deps; import lazily.
89
+ - ✅ `from matrx_orm import ...` — matrx-batch touches a database, so it
90
+ uses the ORM like every other package. Concrete host models resolve by
91
+ name through the injection seam, never by importing host model files.
92
+ - ❌ No hand-written SQL over a raw pool. (The legacy `asyncpg` pool in
93
+ `embedding_cache.py` is debt being burned down — do not add to it.)
94
+ - ❌ No `from matrx_rag import ...` or `from matrx_ai import ...`
95
+ (they consume us, not the other way around — would create a cycle).
96
+ - ❌ No `from aidream import ...` or any root-module imports.
97
+
98
+ ---
99
+
100
+ ## Testing this package in isolation
101
+
102
+ ```bash
103
+ uv run pytest packages/matrx-batch/tests
104
+ ```
105
+
106
+ Tests run with mocked providers and a mocked / in-memory pool. No live
107
+ DB or provider keys required. Real-DB integration coverage lives in
108
+ aidream's test suite.
109
+
110
+ ---
111
+
112
+ ## What lives where
113
+
114
+ | File | Owns |
115
+ |---|---|
116
+ | `__init__.py` | Public API: `configure()`, `BatchRouter`, `BatchableJob`, `EmbeddingCache`, `get_embedding_cache()` |
117
+ | `_ext.py` | Injection registry — host config storage |
118
+ | `openai_batch.py` | OpenAI Batch submit / poll / fetch |
119
+ | `anthropic_batch.py` | Anthropic Message Batches submit / poll / fetch |
120
+ | `router.py` | Urgency-based routing decision |
121
+ | `embedding_cache.py` | `rag.embedding_cache` read/write + hit_count |
122
+ | `tests/test_embedding_cache.py` | Cache hit/miss/hit_count verification |
123
+ | `tests/test_router.py` | Urgency routing decision |
124
+
125
+ ---
126
+
127
+ ## Known gotchas
128
+
129
+ - Batch APIs have a 24h SLA. Never block inference on a batch
130
+ submission — the router always returns a `JobReceipt` immediately;
131
+ consumers poll the receipt or wire a webhook.
132
+ - The embedding cache is content-addressable. Two orgs that produce
133
+ identical text get one row. `organization_id` is stored for cleanup
134
+ queries but is NOT part of the cache key.
135
+ - `hit_count` is incremented in the SAME UPDATE that touches `created_at`,
136
+ so a high-traffic key is also a hot row. Acceptable: the index leaf
137
+ doesn't move because the PK is fixed.
@@ -0,0 +1,88 @@
1
+ Metadata-Version: 2.4
2
+ Name: matrx-batch
3
+ Version: 0.1.0
4
+ Summary: Cost-shield foundation for AI workloads: OpenAI + Anthropic Batch APIs, live/batch urgency router, shared embedding cache.
5
+ Author-email: Matrx <admin@aimatrx.com>
6
+ License: MIT
7
+ Keywords: anthropic,batch,cost,embeddings,matrx,openai
8
+ Classifier: Development Status :: 3 - Alpha
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Programming Language :: Python :: 3.13
13
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
14
+ Requires-Python: >=3.13
15
+ Requires-Dist: anthropic>=0.40
16
+ Requires-Dist: matrx-orm
17
+ Requires-Dist: matrx-utils
18
+ Requires-Dist: openai>=2.0
19
+ Requires-Dist: pydantic>=2.12
20
+ Provides-Extra: dev
21
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
22
+ Requires-Dist: pytest>=8.0; extra == 'dev'
23
+ Description-Content-Type: text/markdown
24
+
25
+ # matrx-batch
26
+
27
+ Cost-shield foundation for AI workloads in the Matrx ecosystem.
28
+
29
+ Three primitives:
30
+
31
+ 1. **OpenAI Batch API + Anthropic Message Batches** — submit JSONL, poll
32
+ status, fetch results. Async wrappers over the provider SDKs.
33
+ 2. **Urgency router** — `BatchRouter.submit(job)` routes a job to live
34
+ inference, to the provider's Batch API (50% discount, 24h SLA), or
35
+ makes a cost-based decision automatically.
36
+ 3. **Embedding cache** — content-addressable (SHA256 of `model || text`)
37
+ reuse of embedding vectors across the org. Backed by
38
+ `rag.embedding_cache` via a host-supplied asyncpg pool.
39
+
40
+ ## Dependency posture
41
+
42
+ `matrx-batch` sits one notch above `matrx-utils` in the dependency graph
43
+ and is consumed by `matrx-rag` (embedding cache) and `matrx-ai` (urgency
44
+ router, future). It **MUST NOT** import from `aidream/` or from any other
45
+ `matrx-*` sibling — every cross-boundary dependency comes in via
46
+ `matrx_batch.configure(...)`.
47
+
48
+ ## Usage (host-side wiring)
49
+
50
+ ```python
51
+ import matrx_batch
52
+ from openai import AsyncOpenAI
53
+ from anthropic import AsyncAnthropic
54
+
55
+ matrx_batch.configure(
56
+ openai_client=AsyncOpenAI(api_key=settings.OPENAI_API_KEY),
57
+ anthropic_client=AsyncAnthropic(api_key=settings.ANTHROPIC_API_KEY),
58
+ pool_factory=lambda: get_async_pg_pool(),
59
+ cost_check=_aidream_budget_precheck,
60
+ )
61
+
62
+ # Wire the cache into matrx-rag's embed() funnel
63
+ import matrx_rag
64
+ matrx_rag.configure(embedding_cache=matrx_batch.get_embedding_cache())
65
+ ```
66
+
67
+ ## Usage (consumer)
68
+
69
+ ```python
70
+ from matrx_batch import BatchRouter, BatchableJob
71
+
72
+ router = BatchRouter()
73
+ receipt = await router.submit(
74
+ BatchableJob(
75
+ kind="chat",
76
+ provider="anthropic",
77
+ urgency="auto",
78
+ payload={"model": "claude-haiku-4-5", "messages": [...]},
79
+ estimated_tokens_in=8200,
80
+ )
81
+ )
82
+ ```
83
+
84
+ ## See also
85
+
86
+ - `CLAUDE.md` — package rules and injection-point contract.
87
+ - `matrx_batch/embedding_cache.py` — chunk-level embedding reuse.
88
+ - `matrx_batch/router.py` — urgency routing decisions.
@@ -0,0 +1,64 @@
1
+ # matrx-batch
2
+
3
+ Cost-shield foundation for AI workloads in the Matrx ecosystem.
4
+
5
+ Three primitives:
6
+
7
+ 1. **OpenAI Batch API + Anthropic Message Batches** — submit JSONL, poll
8
+ status, fetch results. Async wrappers over the provider SDKs.
9
+ 2. **Urgency router** — `BatchRouter.submit(job)` routes a job to live
10
+ inference, to the provider's Batch API (50% discount, 24h SLA), or
11
+ makes a cost-based decision automatically.
12
+ 3. **Embedding cache** — content-addressable (SHA256 of `model || text`)
13
+ reuse of embedding vectors across the org. Backed by
14
+ `rag.embedding_cache` via a host-supplied asyncpg pool.
15
+
16
+ ## Dependency posture
17
+
18
+ `matrx-batch` sits one notch above `matrx-utils` in the dependency graph
19
+ and is consumed by `matrx-rag` (embedding cache) and `matrx-ai` (urgency
20
+ router, future). It **MUST NOT** import from `aidream/` or from any other
21
+ `matrx-*` sibling — every cross-boundary dependency comes in via
22
+ `matrx_batch.configure(...)`.
23
+
24
+ ## Usage (host-side wiring)
25
+
26
+ ```python
27
+ import matrx_batch
28
+ from openai import AsyncOpenAI
29
+ from anthropic import AsyncAnthropic
30
+
31
+ matrx_batch.configure(
32
+ openai_client=AsyncOpenAI(api_key=settings.OPENAI_API_KEY),
33
+ anthropic_client=AsyncAnthropic(api_key=settings.ANTHROPIC_API_KEY),
34
+ pool_factory=lambda: get_async_pg_pool(),
35
+ cost_check=_aidream_budget_precheck,
36
+ )
37
+
38
+ # Wire the cache into matrx-rag's embed() funnel
39
+ import matrx_rag
40
+ matrx_rag.configure(embedding_cache=matrx_batch.get_embedding_cache())
41
+ ```
42
+
43
+ ## Usage (consumer)
44
+
45
+ ```python
46
+ from matrx_batch import BatchRouter, BatchableJob
47
+
48
+ router = BatchRouter()
49
+ receipt = await router.submit(
50
+ BatchableJob(
51
+ kind="chat",
52
+ provider="anthropic",
53
+ urgency="auto",
54
+ payload={"model": "claude-haiku-4-5", "messages": [...]},
55
+ estimated_tokens_in=8200,
56
+ )
57
+ )
58
+ ```
59
+
60
+ ## See also
61
+
62
+ - `CLAUDE.md` — package rules and injection-point contract.
63
+ - `matrx_batch/embedding_cache.py` — chunk-level embedding reuse.
64
+ - `matrx_batch/router.py` — urgency routing decisions.