carl-studio 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- carl_studio-0.3.0/.claude/skills/terminals/carl.md +132 -0
- carl_studio-0.3.0/.github/workflows/publish.yml +18 -0
- carl_studio-0.3.0/.gitignore +16 -0
- carl_studio-0.3.0/AGENTS.md +51 -0
- carl_studio-0.3.0/LICENSE +21 -0
- carl_studio-0.3.0/PKG-INFO +291 -0
- carl_studio-0.3.0/README.md +213 -0
- carl_studio-0.3.0/assets/carl-paradigm.gif +0 -0
- carl_studio-0.3.0/assets/carl-paradigm.svg +1708 -0
- carl_studio-0.3.0/carl.py +497 -0
- carl_studio-0.3.0/carl.yaml.example +33 -0
- carl_studio-0.3.0/llms.txt +31 -0
- carl_studio-0.3.0/paper/carl-paper.md +385 -0
- carl_studio-0.3.0/pyproject.toml +89 -0
- carl_studio-0.3.0/scripts/generate_paradigm_gif.py +519 -0
- carl_studio-0.3.0/src/carl_studio/__init__.py +99 -0
- carl_studio-0.3.0/src/carl_studio/align/__init__.py +34 -0
- carl_studio-0.3.0/src/carl_studio/align/pipeline.py +1233 -0
- carl_studio-0.3.0/src/carl_studio/bench/__init__.py +40 -0
- carl_studio-0.3.0/src/carl_studio/bench/probes.py +678 -0
- carl_studio-0.3.0/src/carl_studio/bench/suite.py +297 -0
- carl_studio-0.3.0/src/carl_studio/bundler.py +753 -0
- carl_studio-0.3.0/src/carl_studio/cli.py +2340 -0
- carl_studio-0.3.0/src/carl_studio/compute/__init__.py +55 -0
- carl_studio-0.3.0/src/carl_studio/compute/hf_jobs.py +648 -0
- carl_studio-0.3.0/src/carl_studio/compute/local.py +67 -0
- carl_studio-0.3.0/src/carl_studio/compute/prime.py +64 -0
- carl_studio-0.3.0/src/carl_studio/compute/protocol.py +37 -0
- carl_studio-0.3.0/src/carl_studio/compute/runpod.py +81 -0
- carl_studio-0.3.0/src/carl_studio/compute/ssh.py +76 -0
- carl_studio-0.3.0/src/carl_studio/compute/tinker.py +58 -0
- carl_studio-0.3.0/src/carl_studio/console.py +212 -0
- carl_studio-0.3.0/src/carl_studio/data/__init__.py +27 -0
- carl_studio-0.3.0/src/carl_studio/data/adapters/__init__.py +11 -0
- carl_studio-0.3.0/src/carl_studio/data/adapters/base.py +46 -0
- carl_studio-0.3.0/src/carl_studio/data/adapters/nemotron.py +65 -0
- carl_studio-0.3.0/src/carl_studio/data/adapters/tool_calling.py +38 -0
- carl_studio-0.3.0/src/carl_studio/data/curriculum.py +161 -0
- carl_studio-0.3.0/src/carl_studio/data/registry.py +106 -0
- carl_studio-0.3.0/src/carl_studio/data/sources.yaml +76 -0
- carl_studio-0.3.0/src/carl_studio/data/types.py +173 -0
- carl_studio-0.3.0/src/carl_studio/environments/__init__.py +46 -0
- carl_studio-0.3.0/src/carl_studio/environments/atropos.py +110 -0
- carl_studio-0.3.0/src/carl_studio/environments/builtins/__init__.py +4 -0
- carl_studio-0.3.0/src/carl_studio/environments/builtins/code_sandbox.py +241 -0
- carl_studio-0.3.0/src/carl_studio/environments/builtins/sql_sandbox.py +350 -0
- carl_studio-0.3.0/src/carl_studio/environments/prime_rl.py +37 -0
- carl_studio-0.3.0/src/carl_studio/environments/protocol.py +106 -0
- carl_studio-0.3.0/src/carl_studio/environments/registry.py +63 -0
- carl_studio-0.3.0/src/carl_studio/environments/validation.py +94 -0
- carl_studio-0.3.0/src/carl_studio/eval/__init__.py +23 -0
- carl_studio-0.3.0/src/carl_studio/eval/runner.py +1566 -0
- carl_studio-0.3.0/src/carl_studio/experiment/__init__.py +35 -0
- carl_studio-0.3.0/src/carl_studio/experiment/manager.py +210 -0
- carl_studio-0.3.0/src/carl_studio/experiment/types.py +224 -0
- carl_studio-0.3.0/src/carl_studio/hub/__init__.py +4 -0
- carl_studio-0.3.0/src/carl_studio/hub/models.py +104 -0
- carl_studio-0.3.0/src/carl_studio/learn/__init__.py +31 -0
- carl_studio-0.3.0/src/carl_studio/learn/ingest.py +303 -0
- carl_studio-0.3.0/src/carl_studio/learn/pipeline.py +243 -0
- carl_studio-0.3.0/src/carl_studio/learn/qa_gen.py +441 -0
- carl_studio-0.3.0/src/carl_studio/learn/quality.py +344 -0
- carl_studio-0.3.0/src/carl_studio/mcp/__init__.py +4 -0
- carl_studio-0.3.0/src/carl_studio/mcp/server.py +153 -0
- carl_studio-0.3.0/src/carl_studio/observe/__init__.py +12 -0
- carl_studio-0.3.0/src/carl_studio/observe/app.py +252 -0
- carl_studio-0.3.0/src/carl_studio/observe/data_source.py +243 -0
- carl_studio-0.3.0/src/carl_studio/paper/__init__.py +13 -0
- carl_studio-0.3.0/src/carl_studio/paper/experiments.py +170 -0
- carl_studio-0.3.0/src/carl_studio/paper/generator.py +434 -0
- carl_studio-0.3.0/src/carl_studio/primitives/__init__.py +21 -0
- carl_studio-0.3.0/src/carl_studio/primitives/coherence_observer.py +352 -0
- carl_studio-0.3.0/src/carl_studio/primitives/coherence_probe.py +173 -0
- carl_studio-0.3.0/src/carl_studio/primitives/coherence_trace.py +615 -0
- carl_studio-0.3.0/src/carl_studio/primitives/constants.py +19 -0
- carl_studio-0.3.0/src/carl_studio/primitives/frame_buffer.py +166 -0
- carl_studio-0.3.0/src/carl_studio/primitives/math.py +31 -0
- carl_studio-0.3.0/src/carl_studio/project.py +90 -0
- carl_studio-0.3.0/src/carl_studio/py.typed +0 -0
- carl_studio-0.3.0/src/carl_studio/settings.py +408 -0
- carl_studio-0.3.0/src/carl_studio/theme.py +273 -0
- carl_studio-0.3.0/src/carl_studio/tier.py +277 -0
- carl_studio-0.3.0/src/carl_studio/training/__init__.py +14 -0
- carl_studio-0.3.0/src/carl_studio/training/callbacks.py +127 -0
- carl_studio-0.3.0/src/carl_studio/training/cascade.py +182 -0
- carl_studio-0.3.0/src/carl_studio/training/lr_resonance.py +71 -0
- carl_studio-0.3.0/src/carl_studio/training/multi_env.py +150 -0
- carl_studio-0.3.0/src/carl_studio/training/pipeline.py +354 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/__init__.py +22 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/base.py +79 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/cloud.py +38 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/composite.py +224 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/discontinuity.py +54 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/multiscale.py +60 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/task.py +287 -0
- carl_studio-0.3.0/src/carl_studio/training/rewards/vlm.py +89 -0
- carl_studio-0.3.0/src/carl_studio/training/trace_callback.py +163 -0
- carl_studio-0.3.0/src/carl_studio/training/trainer.py +711 -0
- carl_studio-0.3.0/src/carl_studio/ttt/__init__.py +13 -0
- carl_studio-0.3.0/src/carl_studio/ttt/slot.py +87 -0
- carl_studio-0.3.0/src/carl_studio/types/__init__.py +23 -0
- carl_studio-0.3.0/src/carl_studio/types/config.py +214 -0
- carl_studio-0.3.0/src/carl_studio/types/reward.py +24 -0
- carl_studio-0.3.0/src/carl_studio/types/run.py +51 -0
- carl_studio-0.3.0/tests/__init__.py +0 -0
- carl_studio-0.3.0/tests/conftest.py +59 -0
- carl_studio-0.3.0/tests/test_align.py +205 -0
- carl_studio-0.3.0/tests/test_bench.py +304 -0
- carl_studio-0.3.0/tests/test_bundler.py +66 -0
- carl_studio-0.3.0/tests/test_cascade.py +70 -0
- carl_studio-0.3.0/tests/test_coherence_trace.py +331 -0
- carl_studio-0.3.0/tests/test_environments.py +355 -0
- carl_studio-0.3.0/tests/test_eval.py +253 -0
- carl_studio-0.3.0/tests/test_gate.py +116 -0
- carl_studio-0.3.0/tests/test_hf_jobs.py +655 -0
- carl_studio-0.3.0/tests/test_hub.py +30 -0
- carl_studio-0.3.0/tests/test_integration_seams.py +441 -0
- carl_studio-0.3.0/tests/test_learn.py +435 -0
- carl_studio-0.3.0/tests/test_lr_resonance.py +183 -0
- carl_studio-0.3.0/tests/test_mcp.py +7 -0
- carl_studio-0.3.0/tests/test_observer.py +139 -0
- carl_studio-0.3.0/tests/test_primitives.py +54 -0
- carl_studio-0.3.0/tests/test_rewards.py +100 -0
- carl_studio-0.3.0/tests/test_settings.py +441 -0
- carl_studio-0.3.0/tests/test_trainer.py +86 -0
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: carl
|
|
3
|
+
description: |
|
|
4
|
+
This skill should be used when the user asks to "train a model", "run CARL", "check coherence",
|
|
5
|
+
"submit a training job", "monitor training", "evaluate a checkpoint", "deploy the agent",
|
|
6
|
+
"carl train", "carl eval", "carl observe", or works with CARL-trained models, coherence metrics,
|
|
7
|
+
phase transitions, or the zero-rl-pipeline. The CARL training companion for carl-studio.
|
|
8
|
+
version: 0.3.0
|
|
9
|
+
---
|
|
10
|
+
|
|
11
|
+
# CARL Training Companion
|
|
12
|
+
|
|
13
|
+
Manage the full lifecycle of training VLMs with coherence-aware reward signals via carl-studio.
|
|
14
|
+
|
|
15
|
+
## Conservation Law
|
|
16
|
+
|
|
17
|
+
| Constant | Value | Meaning |
|
|
18
|
+
|----------|-------|---------|
|
|
19
|
+
| kappa | 64/3 | Conservation constant |
|
|
20
|
+
| sigma | 3/16 | Semantic quantum |
|
|
21
|
+
| kappa * sigma | 4 | Bits per embedding dimension |
|
|
22
|
+
| T* | kappa * d | Decompression boundary |
|
|
23
|
+
| Defect threshold | \|delta_Phi\| > 0.03 | Discontinuity detection |
|
|
24
|
+
|
|
25
|
+
Triadic dimensions (d = 3*2^k) give exact binary T*. Non-triadic incur ~33% context tax.
|
|
26
|
+
|
|
27
|
+
## Pipeline Architecture
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
Phase 1' (COMPLETE): VLM grounding — 94.6% click accuracy, 100% format compliance
|
|
31
|
+
Phase 2' (ACTIVE): Environment GRPO — tool-calling through real sandbox interaction
|
|
32
|
+
Phase 3 (BUILT): TTT — SLOT optimizer + LoRA micro-update for inference-time adaptation
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
### Phase 2' — Environment GRPO (Current)
|
|
36
|
+
|
|
37
|
+
The model enters a real coding sandbox (`CodingSandboxEnv`) and learns tool-calling through interaction. Four tools: `read_file`, `write_file`, `execute_code`, `run_shell`.
|
|
38
|
+
|
|
39
|
+
**Reward architecture (4 functions, 3 roles):**
|
|
40
|
+
1. `tool_engagement` (w=1.0) — dense: "did the model try?"
|
|
41
|
+
2. `task_completion` (w=3.0) — sparse binary: "did the code run?"
|
|
42
|
+
3. `gated_carl` (w=1.0) — quality: "was the reasoning clean?" (Stage B only)
|
|
43
|
+
4. `gr3_length_penalty` (w=2.0) — structural: shorter correct solutions score higher
|
|
44
|
+
|
|
45
|
+
**Cascade gating:** CARL activates when `task_completion` shows sustained capability. Self-calibrating — no hardcoded threshold.
|
|
46
|
+
|
|
47
|
+
## Training Operations
|
|
48
|
+
|
|
49
|
+
### Submit a Job
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from huggingface_hub import HfApi
|
|
53
|
+
api = HfApi(token=HF_TOKEN)
|
|
54
|
+
job = api.run_uv_job(
|
|
55
|
+
script='https://huggingface.co/datasets/wheattoast11/zero-rl-tool-calling-data/resolve/main/train_env_grpo.py',
|
|
56
|
+
flavor='a100-large',
|
|
57
|
+
timeout='2d',
|
|
58
|
+
secrets={'HF_TOKEN': HF_TOKEN},
|
|
59
|
+
)
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
### Monitor
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
j = api.inspect_job(job_id=JOB_ID) # Status
|
|
66
|
+
logs = list(api.fetch_job_logs(job_id=JOB_ID)) # Logs (generator)
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Watch Trackio dashboard for smoothed trends. Never judge from individual log entries.
|
|
70
|
+
|
|
71
|
+
### Key Health Signals
|
|
72
|
+
|
|
73
|
+
- **Length phase transition:** Rise-then-drop in mean_terminated_length = model learning. Monotonic growth = death.
|
|
74
|
+
- **Tool call frequency:** Should increase as model learns. 0 → 1-4 by step 30+.
|
|
75
|
+
- **Cascade gate fire:** Stage A → B when task_completion sustains above running distribution median.
|
|
76
|
+
- **Clipped ratio:** High (>50%) means completions hitting max_completion_length. Normal early, should decrease.
|
|
77
|
+
|
|
78
|
+
## Phase Transition Gate
|
|
79
|
+
|
|
80
|
+
`PhaseTransitionGate` monitors `mean_token_accuracy` during SFT. Windowed check: 3 of last 5 steps above 0.99 = crystallized. Robust to batch-level noise.
|
|
81
|
+
|
|
82
|
+
Structural signature: entropy spike (1.0 → 9.3 → 0.12) over ~40 steps. Token accuracy jumps from 3% to 99% during collapse. Consistent with Kuramoto synchronization.
|
|
83
|
+
|
|
84
|
+
## Coherence Trap
|
|
85
|
+
|
|
86
|
+
CARL rewards high Phi (coherence). Maximum Phi = delta function = mode collapse. Without cascade gating, CARL dominates sparse task signal and collapses tool-calling at step ~26 (observed in v1). Fix: `CascadeRewardManager` gates CARL behind task performance.
|
|
87
|
+
|
|
88
|
+
If GRPO produces identical completions (zero within-group variance): temperature >= 2.0 + top_k=50.
|
|
89
|
+
|
|
90
|
+
## Compute
|
|
91
|
+
|
|
92
|
+
| Phase | Flavor | VRAM | Price |
|
|
93
|
+
|-------|--------|------|-------|
|
|
94
|
+
| Phase 1' SFT/GRPO | l40sx1 | 48 GB | $1.80/hr |
|
|
95
|
+
| Phase 2' Env GRPO | a100-large | 80 GB | $2.50/hr |
|
|
96
|
+
| Eval | l4x1 | 24 GB | $0.80/hr |
|
|
97
|
+
|
|
98
|
+
**Required env var:** `PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True` — set in the training script. Prevents CUDA fragmentation OOM during multi-turn generation.
|
|
99
|
+
|
|
100
|
+
## CLI Reference
|
|
101
|
+
|
|
102
|
+
```
|
|
103
|
+
carl train Train with CARL rewards
|
|
104
|
+
carl train --send-it Full autonomous pipeline
|
|
105
|
+
carl eval Run eval gates
|
|
106
|
+
carl observe Live coherence monitoring
|
|
107
|
+
carl align Realign a drifted model
|
|
108
|
+
carl learn Ingest new knowledge
|
|
109
|
+
carl bench Coherence meta-benchmarks
|
|
110
|
+
carl status <id> Job status
|
|
111
|
+
carl push Push checkpoint to Hub
|
|
112
|
+
carl bundle Generate self-contained script
|
|
113
|
+
carl mcp Start MCP server
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
## File Layout
|
|
117
|
+
|
|
118
|
+
```
|
|
119
|
+
phase1/ # CARL infrastructure (ModelSpec, rewards, tool profiles)
|
|
120
|
+
phase2/ # Phase 2' environment GRPO (ACTIVE)
|
|
121
|
+
phase3/ # TTT infrastructure (SLOT, LoRA, continuous RL)
|
|
122
|
+
scripts/ # CLI tools, submission, eval, diagnostics
|
|
123
|
+
data/ # Training/eval data
|
|
124
|
+
legacy/ # Archived scripts (v1-v7 versions preserved)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
## Hub Models
|
|
128
|
+
|
|
129
|
+
| Model | Phase | Status |
|
|
130
|
+
|-------|-------|--------|
|
|
131
|
+
| wheattoast11/OmniCoder-9B-Zero-Phase2 | Phase 1' | PASS — 94.6% click accuracy |
|
|
132
|
+
| wheattoast11/OmniCoder-9B-Zero-Phase2Prime | Phase 2' | Training (v7) |
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
on:
|
|
3
|
+
release:
|
|
4
|
+
types: [published]
|
|
5
|
+
|
|
6
|
+
jobs:
|
|
7
|
+
publish:
|
|
8
|
+
runs-on: ubuntu-latest
|
|
9
|
+
permissions:
|
|
10
|
+
id-token: write
|
|
11
|
+
steps:
|
|
12
|
+
- uses: actions/checkout@v4
|
|
13
|
+
- uses: actions/setup-python@v5
|
|
14
|
+
with:
|
|
15
|
+
python-version: "3.12"
|
|
16
|
+
- run: pip install build
|
|
17
|
+
- run: python -m build
|
|
18
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# carl-studio
|
|
2
|
+
|
|
3
|
+
## Setup
|
|
4
|
+
```bash
|
|
5
|
+
uv pip install -e ".[dev]"
|
|
6
|
+
```
|
|
7
|
+
|
|
8
|
+
## Test
|
|
9
|
+
```bash
|
|
10
|
+
pytest tests/ -q --tb=short
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Type Check
|
|
14
|
+
```bash
|
|
15
|
+
pyright --strict src/carl_studio/types/
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
## Code Style
|
|
19
|
+
- Pydantic BaseModel for all configs and data structures
|
|
20
|
+
- Type hints on every function signature
|
|
21
|
+
- Constants are module-level in `primitives/constants.py`, never parameters
|
|
22
|
+
- `from __future__ import annotations` in every file
|
|
23
|
+
- Lazy imports for optional dependencies (torch, anthropic, runpod, etc.)
|
|
24
|
+
|
|
25
|
+
## Key Files
|
|
26
|
+
- `src/carl_studio/primitives/constants.py` — κ, σ, threshold (NEVER modify)
|
|
27
|
+
- `src/carl_studio/primitives/coherence_probe.py` — THE source of truth for coherence math
|
|
28
|
+
- `src/carl_studio/primitives/math.py` — shared compute_phi() (single source, used by all reward components)
|
|
29
|
+
- `src/carl_studio/primitives/coherence_observer.py` — Claude-in-the-loop diagnostics
|
|
30
|
+
- `src/carl_studio/types/config.py` — all config types (TrainingConfig is the root)
|
|
31
|
+
- `src/carl_studio/training/trainer.py` — CARLTrainer: async dispatch to remote or local
|
|
32
|
+
- `src/carl_studio/training/rewards/composite.py` — CARLReward + make_carl_reward factory
|
|
33
|
+
- `src/carl_studio/training/rewards/vlm.py` — VLM rewards (click_accuracy, coordinate_format, precision)
|
|
34
|
+
- `src/carl_studio/compute/protocol.py` — ComputeBackend interface
|
|
35
|
+
- `src/carl_studio/bundler.py` — self-contained HF Jobs script generator
|
|
36
|
+
- `src/carl_studio/mcp/server.py` — 9 MCP tools for AI agent consumption
|
|
37
|
+
- `paper/carl-paper.md` — formal research paper
|
|
38
|
+
|
|
39
|
+
## Boundaries
|
|
40
|
+
- Never modify constants (KAPPA, SIGMA, DEFECT_THRESHOLD)
|
|
41
|
+
- Bundled scripts (from bundler.py) must be self-contained — no carl_studio imports
|
|
42
|
+
- Backend implementations use lazy imports — don't require all SDKs installed
|
|
43
|
+
- `import carl_studio` must work with only pydantic + numpy + typer + anyio + pyyaml
|
|
44
|
+
- Crystal/coherence math uses numpy, never torch (convert at boundary)
|
|
45
|
+
- Qwen3.5 thinking mode must be disabled for short-output tasks (coordinates, etc.)
|
|
46
|
+
- GRPO num_generations=8 minimum — 4 causes zero-advantage zero-gradient
|
|
47
|
+
|
|
48
|
+
## Naming
|
|
49
|
+
- Public API: "coherence" (CoherenceProbe, CoherenceSnapshot, discontinuity)
|
|
50
|
+
- Internal math: physics names preserved (Φ, κ, σ, entropy, defect)
|
|
51
|
+
- The mapping: Crystal→Coherence, Defect→Discontinuity, Crystallization→Commitment, Melting→Dissolution
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Intuition Labs LLC
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,291 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: carl-studio
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: CARL (Coherence-Aware Reinforcement Learning) — information-theoretic reward signals for LLM training via token-level probability distributions
|
|
5
|
+
Project-URL: Homepage, https://github.com/wheattoast11/carl
|
|
6
|
+
Project-URL: Documentation, https://github.com/wheattoast11/carl#readme
|
|
7
|
+
Project-URL: Repository, https://github.com/wheattoast11/carl
|
|
8
|
+
Project-URL: Issues, https://github.com/wheattoast11/carl/issues
|
|
9
|
+
Author-email: Tej Desai <tej@terminals.tech>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: coherence,grpo,information-theory,llm,reinforcement-learning,reward-shaping,rl,training
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Environment :: GPU :: NVIDIA CUDA
|
|
15
|
+
Classifier: Intended Audience :: Developers
|
|
16
|
+
Classifier: Intended Audience :: Science/Research
|
|
17
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
18
|
+
Classifier: Operating System :: OS Independent
|
|
19
|
+
Classifier: Programming Language :: Python :: 3
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
24
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
25
|
+
Classifier: Typing :: Typed
|
|
26
|
+
Requires-Python: >=3.11
|
|
27
|
+
Requires-Dist: anyio>=4.0
|
|
28
|
+
Requires-Dist: numpy
|
|
29
|
+
Requires-Dist: pydantic-settings>=2.0
|
|
30
|
+
Requires-Dist: pydantic>=2.9
|
|
31
|
+
Requires-Dist: pyyaml>=6.0
|
|
32
|
+
Requires-Dist: rich>=14.0
|
|
33
|
+
Requires-Dist: typer>=0.12
|
|
34
|
+
Provides-Extra: all
|
|
35
|
+
Requires-Dist: anthropic>=0.40; extra == 'all'
|
|
36
|
+
Requires-Dist: bitsandbytes; extra == 'all'
|
|
37
|
+
Requires-Dist: datasets>=3.0; extra == 'all'
|
|
38
|
+
Requires-Dist: huggingface-hub>=0.25; extra == 'all'
|
|
39
|
+
Requires-Dist: mcp>=1.0; extra == 'all'
|
|
40
|
+
Requires-Dist: peft>=0.13; extra == 'all'
|
|
41
|
+
Requires-Dist: pyright; extra == 'all'
|
|
42
|
+
Requires-Dist: pytest; extra == 'all'
|
|
43
|
+
Requires-Dist: ruff; extra == 'all'
|
|
44
|
+
Requires-Dist: runpod>=1.7; extra == 'all'
|
|
45
|
+
Requires-Dist: textual-plotext>=0.3; extra == 'all'
|
|
46
|
+
Requires-Dist: textual>=1.0; extra == 'all'
|
|
47
|
+
Requires-Dist: tinker>=0.16; extra == 'all'
|
|
48
|
+
Requires-Dist: torch>=2.4; extra == 'all'
|
|
49
|
+
Requires-Dist: trackio; extra == 'all'
|
|
50
|
+
Requires-Dist: transformers>=5.0; extra == 'all'
|
|
51
|
+
Requires-Dist: trl>=0.12; extra == 'all'
|
|
52
|
+
Provides-Extra: dev
|
|
53
|
+
Requires-Dist: pyright; extra == 'dev'
|
|
54
|
+
Requires-Dist: pytest; extra == 'dev'
|
|
55
|
+
Requires-Dist: ruff; extra == 'dev'
|
|
56
|
+
Provides-Extra: hf
|
|
57
|
+
Requires-Dist: huggingface-hub>=0.25; extra == 'hf'
|
|
58
|
+
Provides-Extra: mcp
|
|
59
|
+
Requires-Dist: mcp>=1.0; extra == 'mcp'
|
|
60
|
+
Provides-Extra: observe
|
|
61
|
+
Requires-Dist: anthropic>=0.40; extra == 'observe'
|
|
62
|
+
Provides-Extra: runpod
|
|
63
|
+
Requires-Dist: runpod>=1.7; extra == 'runpod'
|
|
64
|
+
Provides-Extra: tinker
|
|
65
|
+
Requires-Dist: tinker>=0.16; extra == 'tinker'
|
|
66
|
+
Provides-Extra: training
|
|
67
|
+
Requires-Dist: bitsandbytes; extra == 'training'
|
|
68
|
+
Requires-Dist: datasets>=3.0; extra == 'training'
|
|
69
|
+
Requires-Dist: peft>=0.13; extra == 'training'
|
|
70
|
+
Requires-Dist: torch>=2.4; extra == 'training'
|
|
71
|
+
Requires-Dist: trackio; extra == 'training'
|
|
72
|
+
Requires-Dist: transformers>=5.0; extra == 'training'
|
|
73
|
+
Requires-Dist: trl>=0.12; extra == 'training'
|
|
74
|
+
Provides-Extra: tui
|
|
75
|
+
Requires-Dist: textual-plotext>=0.3; extra == 'tui'
|
|
76
|
+
Requires-Dist: textual>=1.0; extra == 'tui'
|
|
77
|
+
Description-Content-Type: text/markdown
|
|
78
|
+
|
|
79
|
+
# CARL Studio
|
|
80
|
+
|
|
81
|
+
**Coherence-Aware Reinforcement Learning**
|
|
82
|
+
|
|
83
|
+
> Models don't learn gradually -- they crystallize.
|
|
84
|
+
|
|
85
|
+

|
|
86
|
+
|
|
87
|
+
CARL adds information-theoretic reward signals to RL training that measure *how* a model generates, not just *what* it generates. One conservation law. Three reward components. Model-agnostic.
|
|
88
|
+
|
|
89
|
+
## Install
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
pip install carl-studio
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Quick Start
|
|
96
|
+
|
|
97
|
+
### Observe (zero friction -- point at any existing run)
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
carl observe --trackio https://your-space.hf.space
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
No training, no config, no GPU. See your model's learning geometry instantly.
|
|
104
|
+
|
|
105
|
+
### Train
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
from carl_studio import CARLTrainer, TrainingConfig
|
|
109
|
+
|
|
110
|
+
trainer = CARLTrainer(TrainingConfig(
|
|
111
|
+
run_name="my-first-carl",
|
|
112
|
+
base_model="Qwen/Qwen3.5-9B",
|
|
113
|
+
output_repo="your-username/my-model",
|
|
114
|
+
method="grpo",
|
|
115
|
+
dataset_repo="trl-lib/Capybara",
|
|
116
|
+
compute_target="l4x1",
|
|
117
|
+
))
|
|
118
|
+
run = await trainer.train()
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Or from the CLI:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
carl train --model Qwen/Qwen3.5-9B --method grpo --compute l4x1
|
|
125
|
+
carl train --send-it # full pipeline: SFT -> gate -> GRPO -> eval -> push
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
## Architecture
|
|
129
|
+
|
|
130
|
+
```
|
|
131
|
+
Layer 4 MCP Server 9 tools for AI agent consumption
|
|
132
|
+
──────────────────────────────────────────────────
|
|
133
|
+
Layer 3 CLI carl train | eval | observe | align | learn | bench | mcp
|
|
134
|
+
──────────────────────────────────────────────────
|
|
135
|
+
Layer 2 Training CARLTrainer, CascadeRewardManager, environments
|
|
136
|
+
──────────────────────────────────────────────────
|
|
137
|
+
Layer 1 SDK ModelSpec, TrainSpec, VRAMBudget, CoherenceProbe
|
|
138
|
+
──────────────────────────────────────────────────
|
|
139
|
+
Layer 0 Primitives compute_phi, kappa, sigma, PhaseTransitionGate
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
### The Reward
|
|
143
|
+
|
|
144
|
+
```
|
|
145
|
+
R_CARL = 0.50 * R_coherence + 0.30 * R_cloud + 0.20 * R_discontinuity
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
| Component | Measures |
|
|
149
|
+
|-----------|----------|
|
|
150
|
+
| Multiscale coherence | Phi consistency across dyadic block scales |
|
|
151
|
+
| Cloud quality | P(selected) * Phi -- confident AND correct |
|
|
152
|
+
| Discontinuity targeting | Sharp Phi transitions at structurally appropriate locations |
|
|
153
|
+
|
|
154
|
+
### Cascade Gating
|
|
155
|
+
|
|
156
|
+
CARL is length-biased -- a verbose, confident model scores high. Without gating, it dominates sparse task signal and causes mode collapse. The cascade solves this:
|
|
157
|
+
|
|
158
|
+
```
|
|
159
|
+
Stage A (early): task rewards only -- "learn to use tools"
|
|
160
|
+
Stage B (gated): task + CARL rewards -- "now do it coherently"
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
The gate self-calibrates from the task metric's running distribution. No hardcoded threshold.
|
|
164
|
+
|
|
165
|
+
### Order Parameter
|
|
166
|
+
|
|
167
|
+
```
|
|
168
|
+
Phi = 1 - H(P) / log|V|
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
0 = uniform (maximum uncertainty). 1 = delta (complete certainty).
|
|
172
|
+
|
|
173
|
+
## The Conservation Law
|
|
174
|
+
|
|
175
|
+
| Constant | Value | Meaning |
|
|
176
|
+
|----------|-------|---------|
|
|
177
|
+
| kappa | 64/3 | Conservation constant |
|
|
178
|
+
| sigma | 3/16 | Semantic quantum |
|
|
179
|
+
| kappa * sigma | **4** | Bits per embedding dimension |
|
|
180
|
+
| T* | kappa * d | Decompression boundary |
|
|
181
|
+
|
|
182
|
+
Derived in [Bounded Informational Time Crystals](https://doi.org/10.5281/zenodo.18906944). Validated across 6,244 trials in [Material Reality](https://doi.org/10.5281/zenodo.18992029). Formally proved in [Semantic Realizability](https://doi.org/10.5281/zenodo.18992031).
|
|
183
|
+
|
|
184
|
+
## Key Finding: Phase Transitions
|
|
185
|
+
|
|
186
|
+
During VLM SFT, the model exhibits a first-order phase transition:
|
|
187
|
+
|
|
188
|
+
| Steps | Phase | Accuracy | Entropy | What happens |
|
|
189
|
+
|-------|-------|----------|---------|--------------|
|
|
190
|
+
| 0-10 | Baseline | 3% | 1.0 | Pre-training distribution intact |
|
|
191
|
+
| 10-20 | Melting | 8% | **9.3** | Distribution destabilizes completely |
|
|
192
|
+
| 20-25 | **Transition** | **65%** | 4.1 | Accuracy jumps 57 points in 5 steps |
|
|
193
|
+
| 25-35 | Crystallization | 99% | 0.4 | Rapid convergence |
|
|
194
|
+
| 35-46 | Converged | **99.3%** | 0.12 | Fully crystallized |
|
|
195
|
+
|
|
196
|
+
Entropy spikes to near-maximum, then accuracy discontinuously jumps once the system passes the critical coupling threshold. Consistent with Kuramoto synchronization in coupled oscillator systems.
|
|
197
|
+
|
|
198
|
+
## CLI
|
|
199
|
+
|
|
200
|
+
**Core triad:**
|
|
201
|
+
```
|
|
202
|
+
carl observe See learning geometry on any run (no GPU required)
|
|
203
|
+
carl eval Pass/fail gate on a checkpoint
|
|
204
|
+
carl train Train with CARL rewards (SFT, GRPO, DPO, KTO, ORPO)
|
|
205
|
+
carl train --send-it Full autonomous pipeline: SFT -> gate -> GRPO -> eval -> push
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
**Operations:**
|
|
209
|
+
```
|
|
210
|
+
carl status <id> Job status
|
|
211
|
+
carl logs <id> Job logs
|
|
212
|
+
carl stop <id> Cancel a job
|
|
213
|
+
carl push Push checkpoint to Hub
|
|
214
|
+
carl bundle Generate self-contained training script
|
|
215
|
+
carl compute List GPU flavors and pricing
|
|
216
|
+
carl setup First-time setup
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
**Advanced** (experimental):
|
|
220
|
+
```
|
|
221
|
+
carl align Realign a drifted model
|
|
222
|
+
carl learn Ingest knowledge, generate data, train
|
|
223
|
+
carl bench Coherence meta-benchmarks
|
|
224
|
+
carl mcp Start MCP server (9 tools for AI agents)
|
|
225
|
+
carl dev Development utilities
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
## Model-Agnostic
|
|
229
|
+
|
|
230
|
+
`ModelSpec.from_pretrained()` auto-detects architecture, modality, thinking mode, quantization constraints, and LoRA targets from any HuggingFace config.json. No per-model branches.
|
|
231
|
+
|
|
232
|
+
| Model | Status |
|
|
233
|
+
|-------|--------|
|
|
234
|
+
| Qwen 3.5 9B VLM | Primary -- 94.6% click accuracy |
|
|
235
|
+
| Gemma 4 E4B | Planned |
|
|
236
|
+
| Gemma 4 31B | Planned (multi-GPU) |
|
|
237
|
+
|
|
238
|
+
## Compute Backends
|
|
239
|
+
|
|
240
|
+
| Backend | Flag |
|
|
241
|
+
|---------|------|
|
|
242
|
+
| HuggingFace Jobs | `--compute l4x1` / `a100-large` / `h200` |
|
|
243
|
+
| RunPod | `--compute runpod` |
|
|
244
|
+
| Tinker | `--compute tinker` |
|
|
245
|
+
| Prime Intellect | `--compute prime` |
|
|
246
|
+
| SSH | `--compute ssh` |
|
|
247
|
+
| Local | `--compute local` |
|
|
248
|
+
|
|
249
|
+
## Test-Time Training
|
|
250
|
+
|
|
251
|
+
CARL includes TTT mechanisms for post-deployment adaptation:
|
|
252
|
+
|
|
253
|
+
- **SLOT** -- hidden delta injection (8 Adam steps, architecture-agnostic)
|
|
254
|
+
- **LoRA micro-update** -- rank-1 online adaptation
|
|
255
|
+
|
|
256
|
+
## IP Boundaries
|
|
257
|
+
|
|
258
|
+
CARL Studio is MIT-licensed. The mathematics -- conservation law, order parameter, reward components -- are independently derivable from the three published papers (CC-BY-4.0).
|
|
259
|
+
|
|
260
|
+
This package is the *open training framework*. It does **not** include the runtime dynamics or autonomous orchestration:
|
|
261
|
+
|
|
262
|
+
| What | Where | License |
|
|
263
|
+
|------|-------|---------|
|
|
264
|
+
| Conservation law, Phi, rewards | **CARL Studio** (this repo) | MIT |
|
|
265
|
+
| Observe, eval, train CLI | **CARL Studio** (this repo) | MIT |
|
|
266
|
+
| Resonance LR modulation | **terminals-runtime** | BUSL-1.1 |
|
|
267
|
+
| SLOT / LoRA micro-update (TTT) | **terminals-runtime** | BUSL-1.1 |
|
|
268
|
+
| Kuramoto oscillator dynamics | **terminals-runtime** | BUSL-1.1 |
|
|
269
|
+
| Coherence diagnosis methodology | **terminals-runtime** | BUSL-1.1 |
|
|
270
|
+
| Audio coherence (CHORD) | Terminals Platform | BUSL-1.1 |
|
|
271
|
+
| Cross-substrate isomorphisms | Terminals Platform | BUSL-1.1 |
|
|
272
|
+
| Interactive Research Environment | Terminals Platform | BUSL-1.1 |
|
|
273
|
+
| Material Reality datasets | Zenodo | CC-BY-4.0 |
|
|
274
|
+
|
|
275
|
+
The bifurcation is deliberate: CARL Studio provides the full training loop (observe, eval, train) using published mathematics. The `terminals-runtime` package adds autonomous features (resonance-aware LR, test-time training, Claude-powered diagnosis) behind BUSL-1.1. Same conservation law, different sides of the boundary.
|
|
276
|
+
|
|
277
|
+
## Citation
|
|
278
|
+
|
|
279
|
+
```bibtex
|
|
280
|
+
@article{desai2026carl,
|
|
281
|
+
title = {Coherence-Aware Reinforcement Learning},
|
|
282
|
+
author = {Desai, Tej},
|
|
283
|
+
year = {2026},
|
|
284
|
+
url = {https://github.com/wheattoast11/carl},
|
|
285
|
+
note = {Intuition Labs LLC}
|
|
286
|
+
}
|
|
287
|
+
```
|
|
288
|
+
|
|
289
|
+
## License
|
|
290
|
+
|
|
291
|
+
MIT -- Intuition Labs LLC
|