carl-studio 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (125) hide show
  1. carl_studio-0.3.0/.claude/skills/terminals/carl.md +132 -0
  2. carl_studio-0.3.0/.github/workflows/publish.yml +18 -0
  3. carl_studio-0.3.0/.gitignore +16 -0
  4. carl_studio-0.3.0/AGENTS.md +51 -0
  5. carl_studio-0.3.0/LICENSE +21 -0
  6. carl_studio-0.3.0/PKG-INFO +291 -0
  7. carl_studio-0.3.0/README.md +213 -0
  8. carl_studio-0.3.0/assets/carl-paradigm.gif +0 -0
  9. carl_studio-0.3.0/assets/carl-paradigm.svg +1708 -0
  10. carl_studio-0.3.0/carl.py +497 -0
  11. carl_studio-0.3.0/carl.yaml.example +33 -0
  12. carl_studio-0.3.0/llms.txt +31 -0
  13. carl_studio-0.3.0/paper/carl-paper.md +385 -0
  14. carl_studio-0.3.0/pyproject.toml +89 -0
  15. carl_studio-0.3.0/scripts/generate_paradigm_gif.py +519 -0
  16. carl_studio-0.3.0/src/carl_studio/__init__.py +99 -0
  17. carl_studio-0.3.0/src/carl_studio/align/__init__.py +34 -0
  18. carl_studio-0.3.0/src/carl_studio/align/pipeline.py +1233 -0
  19. carl_studio-0.3.0/src/carl_studio/bench/__init__.py +40 -0
  20. carl_studio-0.3.0/src/carl_studio/bench/probes.py +678 -0
  21. carl_studio-0.3.0/src/carl_studio/bench/suite.py +297 -0
  22. carl_studio-0.3.0/src/carl_studio/bundler.py +753 -0
  23. carl_studio-0.3.0/src/carl_studio/cli.py +2340 -0
  24. carl_studio-0.3.0/src/carl_studio/compute/__init__.py +55 -0
  25. carl_studio-0.3.0/src/carl_studio/compute/hf_jobs.py +648 -0
  26. carl_studio-0.3.0/src/carl_studio/compute/local.py +67 -0
  27. carl_studio-0.3.0/src/carl_studio/compute/prime.py +64 -0
  28. carl_studio-0.3.0/src/carl_studio/compute/protocol.py +37 -0
  29. carl_studio-0.3.0/src/carl_studio/compute/runpod.py +81 -0
  30. carl_studio-0.3.0/src/carl_studio/compute/ssh.py +76 -0
  31. carl_studio-0.3.0/src/carl_studio/compute/tinker.py +58 -0
  32. carl_studio-0.3.0/src/carl_studio/console.py +212 -0
  33. carl_studio-0.3.0/src/carl_studio/data/__init__.py +27 -0
  34. carl_studio-0.3.0/src/carl_studio/data/adapters/__init__.py +11 -0
  35. carl_studio-0.3.0/src/carl_studio/data/adapters/base.py +46 -0
  36. carl_studio-0.3.0/src/carl_studio/data/adapters/nemotron.py +65 -0
  37. carl_studio-0.3.0/src/carl_studio/data/adapters/tool_calling.py +38 -0
  38. carl_studio-0.3.0/src/carl_studio/data/curriculum.py +161 -0
  39. carl_studio-0.3.0/src/carl_studio/data/registry.py +106 -0
  40. carl_studio-0.3.0/src/carl_studio/data/sources.yaml +76 -0
  41. carl_studio-0.3.0/src/carl_studio/data/types.py +173 -0
  42. carl_studio-0.3.0/src/carl_studio/environments/__init__.py +46 -0
  43. carl_studio-0.3.0/src/carl_studio/environments/atropos.py +110 -0
  44. carl_studio-0.3.0/src/carl_studio/environments/builtins/__init__.py +4 -0
  45. carl_studio-0.3.0/src/carl_studio/environments/builtins/code_sandbox.py +241 -0
  46. carl_studio-0.3.0/src/carl_studio/environments/builtins/sql_sandbox.py +350 -0
  47. carl_studio-0.3.0/src/carl_studio/environments/prime_rl.py +37 -0
  48. carl_studio-0.3.0/src/carl_studio/environments/protocol.py +106 -0
  49. carl_studio-0.3.0/src/carl_studio/environments/registry.py +63 -0
  50. carl_studio-0.3.0/src/carl_studio/environments/validation.py +94 -0
  51. carl_studio-0.3.0/src/carl_studio/eval/__init__.py +23 -0
  52. carl_studio-0.3.0/src/carl_studio/eval/runner.py +1566 -0
  53. carl_studio-0.3.0/src/carl_studio/experiment/__init__.py +35 -0
  54. carl_studio-0.3.0/src/carl_studio/experiment/manager.py +210 -0
  55. carl_studio-0.3.0/src/carl_studio/experiment/types.py +224 -0
  56. carl_studio-0.3.0/src/carl_studio/hub/__init__.py +4 -0
  57. carl_studio-0.3.0/src/carl_studio/hub/models.py +104 -0
  58. carl_studio-0.3.0/src/carl_studio/learn/__init__.py +31 -0
  59. carl_studio-0.3.0/src/carl_studio/learn/ingest.py +303 -0
  60. carl_studio-0.3.0/src/carl_studio/learn/pipeline.py +243 -0
  61. carl_studio-0.3.0/src/carl_studio/learn/qa_gen.py +441 -0
  62. carl_studio-0.3.0/src/carl_studio/learn/quality.py +344 -0
  63. carl_studio-0.3.0/src/carl_studio/mcp/__init__.py +4 -0
  64. carl_studio-0.3.0/src/carl_studio/mcp/server.py +153 -0
  65. carl_studio-0.3.0/src/carl_studio/observe/__init__.py +12 -0
  66. carl_studio-0.3.0/src/carl_studio/observe/app.py +252 -0
  67. carl_studio-0.3.0/src/carl_studio/observe/data_source.py +243 -0
  68. carl_studio-0.3.0/src/carl_studio/paper/__init__.py +13 -0
  69. carl_studio-0.3.0/src/carl_studio/paper/experiments.py +170 -0
  70. carl_studio-0.3.0/src/carl_studio/paper/generator.py +434 -0
  71. carl_studio-0.3.0/src/carl_studio/primitives/__init__.py +21 -0
  72. carl_studio-0.3.0/src/carl_studio/primitives/coherence_observer.py +352 -0
  73. carl_studio-0.3.0/src/carl_studio/primitives/coherence_probe.py +173 -0
  74. carl_studio-0.3.0/src/carl_studio/primitives/coherence_trace.py +615 -0
  75. carl_studio-0.3.0/src/carl_studio/primitives/constants.py +19 -0
  76. carl_studio-0.3.0/src/carl_studio/primitives/frame_buffer.py +166 -0
  77. carl_studio-0.3.0/src/carl_studio/primitives/math.py +31 -0
  78. carl_studio-0.3.0/src/carl_studio/project.py +90 -0
  79. carl_studio-0.3.0/src/carl_studio/py.typed +0 -0
  80. carl_studio-0.3.0/src/carl_studio/settings.py +408 -0
  81. carl_studio-0.3.0/src/carl_studio/theme.py +273 -0
  82. carl_studio-0.3.0/src/carl_studio/tier.py +277 -0
  83. carl_studio-0.3.0/src/carl_studio/training/__init__.py +14 -0
  84. carl_studio-0.3.0/src/carl_studio/training/callbacks.py +127 -0
  85. carl_studio-0.3.0/src/carl_studio/training/cascade.py +182 -0
  86. carl_studio-0.3.0/src/carl_studio/training/lr_resonance.py +71 -0
  87. carl_studio-0.3.0/src/carl_studio/training/multi_env.py +150 -0
  88. carl_studio-0.3.0/src/carl_studio/training/pipeline.py +354 -0
  89. carl_studio-0.3.0/src/carl_studio/training/rewards/__init__.py +22 -0
  90. carl_studio-0.3.0/src/carl_studio/training/rewards/base.py +79 -0
  91. carl_studio-0.3.0/src/carl_studio/training/rewards/cloud.py +38 -0
  92. carl_studio-0.3.0/src/carl_studio/training/rewards/composite.py +224 -0
  93. carl_studio-0.3.0/src/carl_studio/training/rewards/discontinuity.py +54 -0
  94. carl_studio-0.3.0/src/carl_studio/training/rewards/multiscale.py +60 -0
  95. carl_studio-0.3.0/src/carl_studio/training/rewards/task.py +287 -0
  96. carl_studio-0.3.0/src/carl_studio/training/rewards/vlm.py +89 -0
  97. carl_studio-0.3.0/src/carl_studio/training/trace_callback.py +163 -0
  98. carl_studio-0.3.0/src/carl_studio/training/trainer.py +711 -0
  99. carl_studio-0.3.0/src/carl_studio/ttt/__init__.py +13 -0
  100. carl_studio-0.3.0/src/carl_studio/ttt/slot.py +87 -0
  101. carl_studio-0.3.0/src/carl_studio/types/__init__.py +23 -0
  102. carl_studio-0.3.0/src/carl_studio/types/config.py +214 -0
  103. carl_studio-0.3.0/src/carl_studio/types/reward.py +24 -0
  104. carl_studio-0.3.0/src/carl_studio/types/run.py +51 -0
  105. carl_studio-0.3.0/tests/__init__.py +0 -0
  106. carl_studio-0.3.0/tests/conftest.py +59 -0
  107. carl_studio-0.3.0/tests/test_align.py +205 -0
  108. carl_studio-0.3.0/tests/test_bench.py +304 -0
  109. carl_studio-0.3.0/tests/test_bundler.py +66 -0
  110. carl_studio-0.3.0/tests/test_cascade.py +70 -0
  111. carl_studio-0.3.0/tests/test_coherence_trace.py +331 -0
  112. carl_studio-0.3.0/tests/test_environments.py +355 -0
  113. carl_studio-0.3.0/tests/test_eval.py +253 -0
  114. carl_studio-0.3.0/tests/test_gate.py +116 -0
  115. carl_studio-0.3.0/tests/test_hf_jobs.py +655 -0
  116. carl_studio-0.3.0/tests/test_hub.py +30 -0
  117. carl_studio-0.3.0/tests/test_integration_seams.py +441 -0
  118. carl_studio-0.3.0/tests/test_learn.py +435 -0
  119. carl_studio-0.3.0/tests/test_lr_resonance.py +183 -0
  120. carl_studio-0.3.0/tests/test_mcp.py +7 -0
  121. carl_studio-0.3.0/tests/test_observer.py +139 -0
  122. carl_studio-0.3.0/tests/test_primitives.py +54 -0
  123. carl_studio-0.3.0/tests/test_rewards.py +100 -0
  124. carl_studio-0.3.0/tests/test_settings.py +441 -0
  125. carl_studio-0.3.0/tests/test_trainer.py +86 -0
@@ -0,0 +1,132 @@
1
+ ---
2
+ name: carl
3
+ description: |
4
+ This skill should be used when the user asks to "train a model", "run CARL", "check coherence",
5
+ "submit a training job", "monitor training", "evaluate a checkpoint", "deploy the agent",
6
+ "carl train", "carl eval", "carl observe", or works with CARL-trained models, coherence metrics,
7
+ phase transitions, or the zero-rl-pipeline. The CARL training companion for carl-studio.
8
+ version: 0.3.0
9
+ ---
10
+
11
+ # CARL Training Companion
12
+
13
+ Manage the full lifecycle of training VLMs with coherence-aware reward signals via carl-studio.
14
+
15
+ ## Conservation Law
16
+
17
+ | Constant | Value | Meaning |
18
+ |----------|-------|---------|
19
+ | kappa | 64/3 | Conservation constant |
20
+ | sigma | 3/16 | Semantic quantum |
21
+ | kappa * sigma | 4 | Bits per embedding dimension |
22
+ | T* | kappa * d | Decompression boundary |
23
+ | Defect threshold | \|delta_Phi\| > 0.03 | Discontinuity detection |
24
+
25
+ Triadic dimensions (d = 3*2^k) give exact binary T*. Non-triadic incur ~33% context tax.
26
+
27
+ ## Pipeline Architecture
28
+
29
+ ```
30
+ Phase 1' (COMPLETE): VLM grounding — 94.6% click accuracy, 100% format compliance
31
+ Phase 2' (ACTIVE): Environment GRPO — tool-calling through real sandbox interaction
32
+ Phase 3 (BUILT): TTT — SLOT optimizer + LoRA micro-update for inference-time adaptation
33
+ ```
34
+
35
+ ### Phase 2' — Environment GRPO (Current)
36
+
37
+ The model enters a real coding sandbox (`CodingSandboxEnv`) and learns tool-calling through interaction. Four tools: `read_file`, `write_file`, `execute_code`, `run_shell`.
38
+
39
+ **Reward architecture (4 functions, 3 roles):**
40
+ 1. `tool_engagement` (w=1.0) — dense: "did the model try?"
41
+ 2. `task_completion` (w=3.0) — sparse binary: "did the code run?"
42
+ 3. `gated_carl` (w=1.0) — quality: "was the reasoning clean?" (Stage B only)
43
+ 4. `gr3_length_penalty` (w=2.0) — structural: shorter correct solutions score higher
44
+
45
+ **Cascade gating:** CARL activates when `task_completion` shows sustained capability. Self-calibrating — no hardcoded threshold.
46
+
47
+ ## Training Operations
48
+
49
+ ### Submit a Job
50
+
51
+ ```python
52
+ from huggingface_hub import HfApi
53
+ api = HfApi(token=HF_TOKEN)
54
+ job = api.run_uv_job(
55
+ script='https://huggingface.co/datasets/wheattoast11/zero-rl-tool-calling-data/resolve/main/train_env_grpo.py',
56
+ flavor='a100-large',
57
+ timeout='2d',
58
+ secrets={'HF_TOKEN': HF_TOKEN},
59
+ )
60
+ ```
61
+
62
+ ### Monitor
63
+
64
+ ```python
65
+ j = api.inspect_job(job_id=JOB_ID) # Status
66
+ logs = list(api.fetch_job_logs(job_id=JOB_ID)) # Logs (generator)
67
+ ```
68
+
69
+ Watch Trackio dashboard for smoothed trends. Never judge from individual log entries.
70
+
71
+ ### Key Health Signals
72
+
73
+ - **Length phase transition:** Rise-then-drop in mean_terminated_length = model learning. Monotonic growth = death.
74
+ - **Tool call frequency:** Should increase as model learns. 0 → 1-4 by step 30+.
75
+ - **Cascade gate fire:** Stage A → B when task_completion sustains above running distribution median.
76
+ - **Clipped ratio:** High (>50%) means completions hitting max_completion_length. Normal early, should decrease.
77
+
78
+ ## Phase Transition Gate
79
+
80
+ `PhaseTransitionGate` monitors `mean_token_accuracy` during SFT. Windowed check: 3 of last 5 steps above 0.99 = crystallized. Robust to batch-level noise.
81
+
82
+ Structural signature: entropy spike (1.0 → 9.3 → 0.12) over ~40 steps. Token accuracy jumps from 3% to 99% during collapse. Consistent with Kuramoto synchronization.
83
+
84
+ ## Coherence Trap
85
+
86
+ CARL rewards high Phi (coherence). Maximum Phi = delta function = mode collapse. Without cascade gating, CARL dominates sparse task signal and collapses tool-calling at step ~26 (observed in v1). Fix: `CascadeRewardManager` gates CARL behind task performance.
87
+
88
+ If GRPO produces identical completions (zero within-group variance): temperature >= 2.0 + top_k=50.
89
+
90
+ ## Compute
91
+
92
+ | Phase | Flavor | VRAM | Price |
93
+ |-------|--------|------|-------|
94
+ | Phase 1' SFT/GRPO | l40sx1 | 48 GB | $1.80/hr |
95
+ | Phase 2' Env GRPO | a100-large | 80 GB | $2.50/hr |
96
+ | Eval | l4x1 | 24 GB | $0.80/hr |
97
+
98
+ **Required env var:** `PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True` — set in the training script. Prevents CUDA fragmentation OOM during multi-turn generation.
99
+
100
+ ## CLI Reference
101
+
102
+ ```
103
+ carl train Train with CARL rewards
104
+ carl train --send-it Full autonomous pipeline
105
+ carl eval Run eval gates
106
+ carl observe Live coherence monitoring
107
+ carl align Realign a drifted model
108
+ carl learn Ingest new knowledge
109
+ carl bench Coherence meta-benchmarks
110
+ carl status <id> Job status
111
+ carl push Push checkpoint to Hub
112
+ carl bundle Generate self-contained script
113
+ carl mcp Start MCP server
114
+ ```
115
+
116
+ ## File Layout
117
+
118
+ ```
119
+ phase1/ # CARL infrastructure (ModelSpec, rewards, tool profiles)
120
+ phase2/ # Phase 2' environment GRPO (ACTIVE)
121
+ phase3/ # TTT infrastructure (SLOT, LoRA, continuous RL)
122
+ scripts/ # CLI tools, submission, eval, diagnostics
123
+ data/ # Training/eval data
124
+ legacy/ # Archived scripts (v1-v7 versions preserved)
125
+ ```
126
+
127
+ ## Hub Models
128
+
129
+ | Model | Phase | Status |
130
+ |-------|-------|--------|
131
+ | wheattoast11/OmniCoder-9B-Zero-Phase2 | Phase 1' | PASS — 94.6% click accuracy |
132
+ | wheattoast11/OmniCoder-9B-Zero-Phase2Prime | Phase 2' | Training (v7) |
@@ -0,0 +1,18 @@
1
+ name: Publish to PyPI
2
+ on:
3
+ release:
4
+ types: [published]
5
+
6
+ jobs:
7
+ publish:
8
+ runs-on: ubuntu-latest
9
+ permissions:
10
+ id-token: write
11
+ steps:
12
+ - uses: actions/checkout@v4
13
+ - uses: actions/setup-python@v5
14
+ with:
15
+ python-version: "3.12"
16
+ - run: pip install build
17
+ - run: python -m build
18
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,16 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *$py.class
4
+ *.egg-info/
5
+ dist/
6
+ build/
7
+ .eggs/
8
+ *.whl
9
+ .pytest_cache/
10
+ .mypy_cache/
11
+ .ruff_cache/
12
+ *.egg
13
+ .env
14
+ .venv/
15
+ venv/
16
+ env/
@@ -0,0 +1,51 @@
1
+ # carl-studio
2
+
3
+ ## Setup
4
+ ```bash
5
+ uv pip install -e ".[dev]"
6
+ ```
7
+
8
+ ## Test
9
+ ```bash
10
+ pytest tests/ -q --tb=short
11
+ ```
12
+
13
+ ## Type Check
14
+ ```bash
15
+ pyright --strict src/carl_studio/types/
16
+ ```
17
+
18
+ ## Code Style
19
+ - Pydantic BaseModel for all configs and data structures
20
+ - Type hints on every function signature
21
+ - Constants are module-level in `primitives/constants.py`, never parameters
22
+ - `from __future__ import annotations` in every file
23
+ - Lazy imports for optional dependencies (torch, anthropic, runpod, etc.)
24
+
25
+ ## Key Files
26
+ - `src/carl_studio/primitives/constants.py` — κ, σ, threshold (NEVER modify)
27
+ - `src/carl_studio/primitives/coherence_probe.py` — THE source of truth for coherence math
28
+ - `src/carl_studio/primitives/math.py` — shared compute_phi() (single source, used by all reward components)
29
+ - `src/carl_studio/primitives/coherence_observer.py` — Claude-in-the-loop diagnostics
30
+ - `src/carl_studio/types/config.py` — all config types (TrainingConfig is the root)
31
+ - `src/carl_studio/training/trainer.py` — CARLTrainer: async dispatch to remote or local
32
+ - `src/carl_studio/training/rewards/composite.py` — CARLReward + make_carl_reward factory
33
+ - `src/carl_studio/training/rewards/vlm.py` — VLM rewards (click_accuracy, coordinate_format, precision)
34
+ - `src/carl_studio/compute/protocol.py` — ComputeBackend interface
35
+ - `src/carl_studio/bundler.py` — self-contained HF Jobs script generator
36
+ - `src/carl_studio/mcp/server.py` — 9 MCP tools for AI agent consumption
37
+ - `paper/carl-paper.md` — formal research paper
38
+
39
+ ## Boundaries
40
+ - Never modify constants (KAPPA, SIGMA, DEFECT_THRESHOLD)
41
+ - Bundled scripts (from bundler.py) must be self-contained — no carl_studio imports
42
+ - Backend implementations use lazy imports — don't require all SDKs installed
43
+ - `import carl_studio` must work with only pydantic + numpy + typer + anyio + pyyaml
44
+ - Crystal/coherence math uses numpy, never torch (convert at boundary)
45
+ - Qwen3.5 thinking mode must be disabled for short-output tasks (coordinates, etc.)
46
+ - GRPO num_generations=8 minimum — 4 causes zero-advantage zero-gradient
47
+
48
+ ## Naming
49
+ - Public API: "coherence" (CoherenceProbe, CoherenceSnapshot, discontinuity)
50
+ - Internal math: physics names preserved (Φ, κ, σ, entropy, defect)
51
+ - The mapping: Crystal→Coherence, Defect→Discontinuity, Crystallization→Commitment, Melting→Dissolution
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Intuition Labs LLC
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,291 @@
1
+ Metadata-Version: 2.4
2
+ Name: carl-studio
3
+ Version: 0.3.0
4
+ Summary: CARL (Coherence-Aware Reinforcement Learning) — information-theoretic reward signals for LLM training via token-level probability distributions
5
+ Project-URL: Homepage, https://github.com/wheattoast11/carl
6
+ Project-URL: Documentation, https://github.com/wheattoast11/carl#readme
7
+ Project-URL: Repository, https://github.com/wheattoast11/carl
8
+ Project-URL: Issues, https://github.com/wheattoast11/carl/issues
9
+ Author-email: Tej Desai <tej@terminals.tech>
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: coherence,grpo,information-theory,llm,reinforcement-learning,reward-shaping,rl,training
13
+ Classifier: Development Status :: 4 - Beta
14
+ Classifier: Environment :: GPU :: NVIDIA CUDA
15
+ Classifier: Intended Audience :: Developers
16
+ Classifier: Intended Audience :: Science/Research
17
+ Classifier: License :: OSI Approved :: MIT License
18
+ Classifier: Operating System :: OS Independent
19
+ Classifier: Programming Language :: Python :: 3
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
24
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
25
+ Classifier: Typing :: Typed
26
+ Requires-Python: >=3.11
27
+ Requires-Dist: anyio>=4.0
28
+ Requires-Dist: numpy
29
+ Requires-Dist: pydantic-settings>=2.0
30
+ Requires-Dist: pydantic>=2.9
31
+ Requires-Dist: pyyaml>=6.0
32
+ Requires-Dist: rich>=14.0
33
+ Requires-Dist: typer>=0.12
34
+ Provides-Extra: all
35
+ Requires-Dist: anthropic>=0.40; extra == 'all'
36
+ Requires-Dist: bitsandbytes; extra == 'all'
37
+ Requires-Dist: datasets>=3.0; extra == 'all'
38
+ Requires-Dist: huggingface-hub>=0.25; extra == 'all'
39
+ Requires-Dist: mcp>=1.0; extra == 'all'
40
+ Requires-Dist: peft>=0.13; extra == 'all'
41
+ Requires-Dist: pyright; extra == 'all'
42
+ Requires-Dist: pytest; extra == 'all'
43
+ Requires-Dist: ruff; extra == 'all'
44
+ Requires-Dist: runpod>=1.7; extra == 'all'
45
+ Requires-Dist: textual-plotext>=0.3; extra == 'all'
46
+ Requires-Dist: textual>=1.0; extra == 'all'
47
+ Requires-Dist: tinker>=0.16; extra == 'all'
48
+ Requires-Dist: torch>=2.4; extra == 'all'
49
+ Requires-Dist: trackio; extra == 'all'
50
+ Requires-Dist: transformers>=5.0; extra == 'all'
51
+ Requires-Dist: trl>=0.12; extra == 'all'
52
+ Provides-Extra: dev
53
+ Requires-Dist: pyright; extra == 'dev'
54
+ Requires-Dist: pytest; extra == 'dev'
55
+ Requires-Dist: ruff; extra == 'dev'
56
+ Provides-Extra: hf
57
+ Requires-Dist: huggingface-hub>=0.25; extra == 'hf'
58
+ Provides-Extra: mcp
59
+ Requires-Dist: mcp>=1.0; extra == 'mcp'
60
+ Provides-Extra: observe
61
+ Requires-Dist: anthropic>=0.40; extra == 'observe'
62
+ Provides-Extra: runpod
63
+ Requires-Dist: runpod>=1.7; extra == 'runpod'
64
+ Provides-Extra: tinker
65
+ Requires-Dist: tinker>=0.16; extra == 'tinker'
66
+ Provides-Extra: training
67
+ Requires-Dist: bitsandbytes; extra == 'training'
68
+ Requires-Dist: datasets>=3.0; extra == 'training'
69
+ Requires-Dist: peft>=0.13; extra == 'training'
70
+ Requires-Dist: torch>=2.4; extra == 'training'
71
+ Requires-Dist: trackio; extra == 'training'
72
+ Requires-Dist: transformers>=5.0; extra == 'training'
73
+ Requires-Dist: trl>=0.12; extra == 'training'
74
+ Provides-Extra: tui
75
+ Requires-Dist: textual-plotext>=0.3; extra == 'tui'
76
+ Requires-Dist: textual>=1.0; extra == 'tui'
77
+ Description-Content-Type: text/markdown
78
+
79
+ # CARL Studio
80
+
81
+ **Coherence-Aware Reinforcement Learning**
82
+
83
+ > Models don't learn gradually -- they crystallize.
84
+
85
+ ![CARL Phase Transition](assets/carl-paradigm.gif)
86
+
87
+ CARL adds information-theoretic reward signals to RL training that measure *how* a model generates, not just *what* it generates. One conservation law. Three reward components. Model-agnostic.
88
+
89
+ ## Install
90
+
91
+ ```bash
92
+ pip install carl-studio
93
+ ```
94
+
95
+ ## Quick Start
96
+
97
+ ### Observe (zero friction -- point at any existing run)
98
+
99
+ ```bash
100
+ carl observe --trackio https://your-space.hf.space
101
+ ```
102
+
103
+ No training, no config, no GPU. See your model's learning geometry instantly.
104
+
105
+ ### Train
106
+
107
+ ```python
108
+ from carl_studio import CARLTrainer, TrainingConfig
109
+
110
+ trainer = CARLTrainer(TrainingConfig(
111
+ run_name="my-first-carl",
112
+ base_model="Qwen/Qwen3.5-9B",
113
+ output_repo="your-username/my-model",
114
+ method="grpo",
115
+ dataset_repo="trl-lib/Capybara",
116
+ compute_target="l4x1",
117
+ ))
118
+ run = await trainer.train()
119
+ ```
120
+
121
+ Or from the CLI:
122
+
123
+ ```bash
124
+ carl train --model Qwen/Qwen3.5-9B --method grpo --compute l4x1
125
+ carl train --send-it # full pipeline: SFT -> gate -> GRPO -> eval -> push
126
+ ```
127
+
128
+ ## Architecture
129
+
130
+ ```
131
+ Layer 4 MCP Server 9 tools for AI agent consumption
132
+ ──────────────────────────────────────────────────
133
+ Layer 3 CLI carl train | eval | observe | align | learn | bench | mcp
134
+ ──────────────────────────────────────────────────
135
+ Layer 2 Training CARLTrainer, CascadeRewardManager, environments
136
+ ──────────────────────────────────────────────────
137
+ Layer 1 SDK ModelSpec, TrainSpec, VRAMBudget, CoherenceProbe
138
+ ──────────────────────────────────────────────────
139
+ Layer 0 Primitives compute_phi, kappa, sigma, PhaseTransitionGate
140
+ ```
141
+
142
+ ### The Reward
143
+
144
+ ```
145
+ R_CARL = 0.50 * R_coherence + 0.30 * R_cloud + 0.20 * R_discontinuity
146
+ ```
147
+
148
+ | Component | Measures |
149
+ |-----------|----------|
150
+ | Multiscale coherence | Phi consistency across dyadic block scales |
151
+ | Cloud quality | P(selected) * Phi -- confident AND correct |
152
+ | Discontinuity targeting | Sharp Phi transitions at structurally appropriate locations |
153
+
154
+ ### Cascade Gating
155
+
156
+ CARL is length-biased -- a verbose, confident model scores high. Without gating, it dominates sparse task signal and causes mode collapse. The cascade solves this:
157
+
158
+ ```
159
+ Stage A (early): task rewards only -- "learn to use tools"
160
+ Stage B (gated): task + CARL rewards -- "now do it coherently"
161
+ ```
162
+
163
+ The gate self-calibrates from the task metric's running distribution. No hardcoded threshold.
164
+
165
+ ### Order Parameter
166
+
167
+ ```
168
+ Phi = 1 - H(P) / log|V|
169
+ ```
170
+
171
+ 0 = uniform (maximum uncertainty). 1 = delta (complete certainty).
172
+
173
+ ## The Conservation Law
174
+
175
+ | Constant | Value | Meaning |
176
+ |----------|-------|---------|
177
+ | kappa | 64/3 | Conservation constant |
178
+ | sigma | 3/16 | Semantic quantum |
179
+ | kappa * sigma | **4** | Bits per embedding dimension |
180
+ | T* | kappa * d | Decompression boundary |
181
+
182
+ Derived in [Bounded Informational Time Crystals](https://doi.org/10.5281/zenodo.18906944). Validated across 6,244 trials in [Material Reality](https://doi.org/10.5281/zenodo.18992029). Formally proved in [Semantic Realizability](https://doi.org/10.5281/zenodo.18992031).
183
+
184
+ ## Key Finding: Phase Transitions
185
+
186
+ During VLM SFT, the model exhibits a first-order phase transition:
187
+
188
+ | Steps | Phase | Accuracy | Entropy | What happens |
189
+ |-------|-------|----------|---------|--------------|
190
+ | 0-10 | Baseline | 3% | 1.0 | Pre-training distribution intact |
191
+ | 10-20 | Melting | 8% | **9.3** | Distribution destabilizes completely |
192
+ | 20-25 | **Transition** | **65%** | 4.1 | Accuracy jumps 57 points in 5 steps |
193
+ | 25-35 | Crystallization | 99% | 0.4 | Rapid convergence |
194
+ | 35-46 | Converged | **99.3%** | 0.12 | Fully crystallized |
195
+
196
+ Entropy spikes to near-maximum, then accuracy discontinuously jumps once the system passes the critical coupling threshold. Consistent with Kuramoto synchronization in coupled oscillator systems.
197
+
198
+ ## CLI
199
+
200
+ **Core triad:**
201
+ ```
202
+ carl observe See learning geometry on any run (no GPU required)
203
+ carl eval Pass/fail gate on a checkpoint
204
+ carl train Train with CARL rewards (SFT, GRPO, DPO, KTO, ORPO)
205
+ carl train --send-it Full autonomous pipeline: SFT -> gate -> GRPO -> eval -> push
206
+ ```
207
+
208
+ **Operations:**
209
+ ```
210
+ carl status <id> Job status
211
+ carl logs <id> Job logs
212
+ carl stop <id> Cancel a job
213
+ carl push Push checkpoint to Hub
214
+ carl bundle Generate self-contained training script
215
+ carl compute List GPU flavors and pricing
216
+ carl setup First-time setup
217
+ ```
218
+
219
+ **Advanced** (experimental):
220
+ ```
221
+ carl align Realign a drifted model
222
+ carl learn Ingest knowledge, generate data, train
223
+ carl bench Coherence meta-benchmarks
224
+ carl mcp Start MCP server (9 tools for AI agents)
225
+ carl dev Development utilities
226
+ ```
227
+
228
+ ## Model-Agnostic
229
+
230
+ `ModelSpec.from_pretrained()` auto-detects architecture, modality, thinking mode, quantization constraints, and LoRA targets from any HuggingFace config.json. No per-model branches.
231
+
232
+ | Model | Status |
233
+ |-------|--------|
234
+ | Qwen 3.5 9B VLM | Primary -- 94.6% click accuracy |
235
+ | Gemma 4 E4B | Planned |
236
+ | Gemma 4 31B | Planned (multi-GPU) |
237
+
238
+ ## Compute Backends
239
+
240
+ | Backend | Flag |
241
+ |---------|------|
242
+ | HuggingFace Jobs | `--compute l4x1` / `a100-large` / `h200` |
243
+ | RunPod | `--compute runpod` |
244
+ | Tinker | `--compute tinker` |
245
+ | Prime Intellect | `--compute prime` |
246
+ | SSH | `--compute ssh` |
247
+ | Local | `--compute local` |
248
+
249
+ ## Test-Time Training
250
+
251
+ CARL includes TTT mechanisms for post-deployment adaptation:
252
+
253
+ - **SLOT** -- hidden delta injection (8 Adam steps, architecture-agnostic)
254
+ - **LoRA micro-update** -- rank-1 online adaptation
255
+
256
+ ## IP Boundaries
257
+
258
+ CARL Studio is MIT-licensed. The mathematics -- conservation law, order parameter, reward components -- are independently derivable from the three published papers (CC-BY-4.0).
259
+
260
+ This package is the *open training framework*. It does **not** include the runtime dynamics or autonomous orchestration:
261
+
262
+ | What | Where | License |
263
+ |------|-------|---------|
264
+ | Conservation law, Phi, rewards | **CARL Studio** (this repo) | MIT |
265
+ | Observe, eval, train CLI | **CARL Studio** (this repo) | MIT |
266
+ | Resonance LR modulation | **terminals-runtime** | BUSL-1.1 |
267
+ | SLOT / LoRA micro-update (TTT) | **terminals-runtime** | BUSL-1.1 |
268
+ | Kuramoto oscillator dynamics | **terminals-runtime** | BUSL-1.1 |
269
+ | Coherence diagnosis methodology | **terminals-runtime** | BUSL-1.1 |
270
+ | Audio coherence (CHORD) | Terminals Platform | BUSL-1.1 |
271
+ | Cross-substrate isomorphisms | Terminals Platform | BUSL-1.1 |
272
+ | Interactive Research Environment | Terminals Platform | BUSL-1.1 |
273
+ | Material Reality datasets | Zenodo | CC-BY-4.0 |
274
+
275
+ The bifurcation is deliberate: CARL Studio provides the full training loop (observe, eval, train) using published mathematics. The `terminals-runtime` package adds autonomous features (resonance-aware LR, test-time training, Claude-powered diagnosis) behind BUSL-1.1. Same conservation law, different sides of the boundary.
276
+
277
+ ## Citation
278
+
279
+ ```bibtex
280
+ @article{desai2026carl,
281
+ title = {Coherence-Aware Reinforcement Learning},
282
+ author = {Desai, Tej},
283
+ year = {2026},
284
+ url = {https://github.com/wheattoast11/carl},
285
+ note = {Intuition Labs LLC}
286
+ }
287
+ ```
288
+
289
+ ## License
290
+
291
+ MIT -- Intuition Labs LLC