dual-loop-controller 2.0.0a1__tar.gz → 2.0.0a3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/PKG-INFO +29 -5
  2. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/README.md +136 -112
  3. dual_loop_controller-2.0.0a3/dual_loop/__init__.py +66 -0
  4. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/benchmarks/comprehensive_suite.py +201 -147
  5. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/benchmarks/graph_reasoning.py +42 -8
  6. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/benchmarks/halting_audit.py +99 -98
  7. dual_loop_controller-2.0.0a3/dual_loop/checkpoints/checkpoint_trained_dualloop.pt +0 -0
  8. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/decoder.py +53 -5
  9. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/halting.py +71 -71
  10. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop_controller.egg-info/PKG-INFO +29 -5
  11. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop_controller.egg-info/SOURCES.txt +1 -0
  12. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/pyproject.toml +56 -53
  13. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/tests/test_dual_loop.py +41 -0
  14. dual_loop_controller-2.0.0a1/dual_loop/__init__.py +0 -22
  15. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/LICENSE +0 -0
  16. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/adapters/__init__.py +0 -0
  17. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/adapters/latent_adapter.py +0 -0
  18. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/benchmarks/__init__.py +0 -0
  19. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/benchmarks/initiative_benchmark.py +0 -0
  20. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/controller.py +0 -0
  21. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop/memory.py +0 -0
  22. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop_controller.egg-info/dependency_links.txt +0 -0
  23. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop_controller.egg-info/requires.txt +0 -0
  24. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/dual_loop_controller.egg-info/top_level.txt +0 -0
  25. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/setup.cfg +0 -0
  26. {dual_loop_controller-2.0.0a1 → dual_loop_controller-2.0.0a3}/tests/test_adapter_integration.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: dual-loop-controller
3
- Version: 2.0.0a1
3
+ Version: 2.0.0a3
4
4
  Summary: A hardware-aligned, manifold-preserving latent deliberation framework for Transformers
5
5
  Author: Ch3nOff
6
6
  License-Expression: MIT
@@ -30,6 +30,7 @@ Dynamic: license-file
30
30
  # Dual-Loop Cognitive Controller v2.0
31
31
  > **A Hardware-Aligned Latent Deliberation Framework for Transformers: Architecture & Empirical Analysis**
32
32
 
33
+ [![PyPI](https://img.shields.io/pypi/v/dual-loop-controller.svg)](https://pypi.org/project/dual-loop-controller/)
33
34
  [![Tests](https://img.shields.io/badge/tests-passing-brightgreen.svg)](tests/)
34
35
  [![PyTorch](https://img.shields.io/badge/PyTorch-2.14%2B-ee4c2c.svg)](https://pytorch.org/)
35
36
  [![Status](https://img.shields.io/badge/status-empirical--audit-orange.svg)](#empirical-findings)
@@ -99,7 +100,14 @@ tau = 1.40 nats | 29.4% | 1.89 | 49.0% | 13.2% | 37
99
100
 
100
101
  **Justified Operating Point**:
101
102
  * **$\tau = 1.25 \dots 1.40\text{ nats}$** is the justifiable Pareto region: it achieves a **37% reduction in compute** (average **1.89 steps** vs. 3.00) while maintaining peak accuracy (**29.4%**), with a genuinely heterogeneous distribution across steps ($49\%$ at $K=1$, $13\%$ at $K=2$, $38\%$ at $K=3$).
102
- * Arbitrary default thresholds (like 0.5 or blindly using a batch-mean percentile) collapse execution to all-or-nothing extremes ($3.00$ or $1.00$). Dynamic halting must always be calibrated per-sample against empirical validation entropy.
103
+ ### 5. In-Distribution Memorization vs. Out-of-Distribution Generalization
104
+ A crucial empirical insight discovered during data isolation audits:
105
+ * **In-Distribution (Train Set, 500 seen graphs)**:
106
+ `K=0: 43.6% -> K=1: 51.4% -> K=2: 59.4% -> K=3: 63.2% (+19.6% monotonic test-time scaling)`
107
+ The recurrent latent controller successfully learns and memorizes multi-hop relational transitions for familiar graph topologies.
108
+ * **Out-of-Distribution (Held-Out Test Set, 500 unseen graphs)**:
109
+ `K=0: 28.6% -> K=1: 27.6% -> K=2: 28.0% -> K=3: 28.4% (Flat scaling / ~28-30%)`
110
+ Without discrete token anchors, continuous latent representations suffer from representational drift on novel graph structures at the 225K parameter regime.
103
111
 
104
112
  ---
105
113
 
@@ -118,22 +126,38 @@ Despite the scaling limits at small model regimes, the repository provides clean
118
126
 
119
127
  ### 1. Installation
120
128
  ```bash
129
+ # Install officially from PyPI:
130
+ pip install --pre dual-loop-controller
131
+ # or exact version: pip install dual-loop-controller==2.0.0a3
132
+
133
+ # Or install direct from GitHub release tag:
134
+ pip install git+https://github.com/Ch3nOff/dual-loop-controller.git@v2.0.0a3
135
+
136
+ # Or clone locally and install in editable mode:
121
137
  git clone https://github.com/Ch3nOff/dual-loop-controller.git
122
138
  cd dual-loop-controller
123
- python -m pip install torch numpy
139
+ pip install -e .
124
140
  ```
125
141
 
142
+ > [!NOTE]
143
+ > **Pretrained Weights Bundled**: A 225K parameter trained reference checkpoint (~912 KB) is bundled directly in `dual_loop/checkpoints/checkpoint_trained_dualloop.pt`. Fresh clones and pip installs run inference and audits out-of-the-box without requiring a training step first.
144
+
126
145
  ### 2. Running Component Tests (Verifying Shapes & Gradients)
127
146
  ```bash
128
147
  python -m unittest discover -s tests -p "test_*.py"
129
148
  ```
130
149
 
131
- ### 3. Running the Honest Benchmark Suite (Live Tensor Computations)
150
+ ### 3. Verifying Dynamic Halting & Pareto Calibration
151
+ ```bash
152
+ python verify_dynamic_inference.py
153
+ ```
154
+
155
+ ### 4. Running the Honest Benchmark Suite (Live Tensor Computations)
132
156
  ```bash
133
157
  python -m dual_loop.benchmarks.comprehensive_suite
134
158
  ```
135
159
 
136
- ### 4. Re-Training from Scratch
160
+ ### 5. Re-Training from Scratch
137
161
  ```bash
138
162
  python train.py --epochs 35 --hops 3 --k_steps 3 --d_model 64
139
163
  ```
@@ -1,112 +1,136 @@
1
- # Dual-Loop Cognitive Controller v2.0
2
- > **A Hardware-Aligned Latent Deliberation Framework for Transformers: Architecture & Empirical Analysis**
3
-
4
- [![Tests](https://img.shields.io/badge/tests-passing-brightgreen.svg)](tests/)
5
- [![PyTorch](https://img.shields.io/badge/PyTorch-2.14%2B-ee4c2c.svg)](https://pytorch.org/)
6
- [![Status](https://img.shields.io/badge/status-empirical--audit-orange.svg)](#empirical-findings)
7
- [![License](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
8
-
9
- Standard Autoregressive Transformers perform uniform $O(1)$ layer computation per token regardless of task complexity. While Chain-of-Thought (CoT) prompting allows multi-step reasoning, it expends significant output token bandwidth and introduces serial generation latency.
10
-
11
- The **Dual-Loop Cognitive Controller** investigates decoupling deliberation from token generation into two loops:
12
- 1. **Outer Loop (Executive Deliberation / System 2)**: Runs recursive state transitions in a continuous latent space without emitting intermediate tokens.
13
- 2. **Inner Loop (Language Generation / System 1)**: Reads the matured latent thoughts ($H_{\text{thought}}$) as a soft prefix to decode final text responses.
14
-
15
- ---
16
-
17
- ## Empirical Findings & Negative Results (The Unvarnished Truth)
18
-
19
- To maintain strict scientific integrity, this repository reports **the actual, measured behavior of the model trained end-to-end (225,959 parameters, 35 epochs, 3,500 samples, 16 nodes, chance baseline = 6.25%)**, rather than idealized projections.
20
-
21
- ### 1. The Model Learns Real Relational Signals
22
- * **Final Test Accuracy (3-Hop Graph Reasoning)**: **29.4%** vs. random chance **6.25%** (~4.7x better than random guessing).
23
- * This confirms that the weight-tied recurrent Transformer and CWM buffer are capable of gradient propagation and multi-step pattern learning.
24
-
25
- ### 2. The Absence of Monotonic Test-Time Compute Scaling
26
- A central theoretical hypothesis of recurrent latent pondering is that increasing inference steps ($K$) will progressively improve answer accuracy. **On this 225K parameter implementation, this claim does not hold**:
27
-
28
- ```text
29
- ========================================================================================
30
- EMPIRICAL TEST-TIME COMPUTE EVALUATION (Checkpoint: checkpoint_trained_dualloop.pt)
31
- ========================================================================================
32
- Ponder Steps (K) | Test Accuracy (500 samples) | Mean Predictive Entropy (nats)
33
- ----------------------------------------------------------------------------------------
34
- K = 0 (No Ponder)| 27.4% - 30.6% | 1.332 - 1.362 nats
35
- K = 1 | 28.2% | 1.370 nats
36
- K = 2 | 30.6% | 1.307 nats
37
- K = 3 (Trained) | 30.4% | 1.268 nats
38
- K = 4 | 30.0% | 1.268 nats
39
- K = 5 | 31.6% | 1.275 nats
40
- ========================================================================================
41
- ```
42
-
43
- **Scientific Diagnosis**:
44
- * **Flat/Noisy Trajectory**: $K=0$ (bypassing the Outer Loop entirely) performs at parity with or slightly exceeds intermediate $K$ values.
45
- * **Representational Drift**: Tracing individual predictions step-by-step reveals that while some cases improve with pondering, others degrade (e.g. correct at $K=0..1$, but diverging to incorrect candidates at $K=2..3$ due to distractor pull).
46
- * **Scale Artifact vs. Fundamental Limit**: At 225K parameters, the latent space lacks the geometric capacity to preserve stable multi-step deductions without explicit discrete token anchors. Pondering without token-level supervision introduces noise as much as refinement.
47
-
48
- ### 3. Degradation Under Context Distractors (Stress Test)
49
- When distractor edge count increases on 3-hop graphs, performance decays steadily:
50
- * **6 Edges**: 31.0%
51
- * **8 Edges**: 21.0%
52
- * **12 Edges**: 13.7%
53
- * **16 Edges**: 10.3%
54
-
55
- ### 4. Dynamic Halting Audit & The Pareto Trade-Off
56
- A naive threshold like `0.5 nats` fails because the model operates at `~1.25–1.40 nats` (resulting in static $K=3.00$). Evaluating per-sample dynamic halting across a threshold sweep reveals the true **Accuracy vs. Compute Pareto Frontier**:
57
-
58
- ```text
59
- ========================================================================================
60
- PER-SAMPLE DYNAMIC HALTING PARETO FRONTIER (500 Test Samples)
61
- ========================================================================================
62
- Entropy Threshold | Test Accuracy | Avg Steps | % Halt @ K=1 | % Halt @ K=2 | % Halt @ K=3
63
- ----------------------------------------------------------------------------------------
64
- tau = 0.80 nats | 28.0% | 2.81 | 7.6% | 3.8% | 88.6%
65
- tau = 1.15 nats | 28.0% | 2.42 | 24.2% | 9.6% | 66.2%
66
- tau = 1.25 nats | 28.8% | 2.23 | 32.2% | 12.2% | 55.6%
67
- tau = 1.40 nats | 29.4% | 1.89 | 49.0% | 13.2% | 37.8%
68
- ========================================================================================
69
- ```
70
-
71
- **Justified Operating Point**:
72
- * **$\tau = 1.25 \dots 1.40\text{ nats}$** is the justifiable Pareto region: it achieves a **37% reduction in compute** (average **1.89 steps** vs. 3.00) while maintaining peak accuracy (**29.4%**), with a genuinely heterogeneous distribution across steps ($49\%$ at $K=1$, $13\%$ at $K=2$, $38\%$ at $K=3$).
73
- * Arbitrary default thresholds (like 0.5 or blindly using a batch-mean percentile) collapse execution to all-or-nothing extremes ($3.00$ or $1.00$). Dynamic halting must always be calibrated per-sample against empirical validation entropy.
74
-
75
- ---
76
-
77
- ## Architectural Implementation
78
-
79
- Despite the scaling limits at small model regimes, the repository provides clean, production-grade PyTorch implementations of the core modules:
80
-
81
- * **Cognitive Working Memory (`dual_loop/memory.py`)**: Compresses context into $M \ll N$ slots in GPU SRAM/L2 cache to avoid HBM memory bandwidth roundtrips.
82
- * **Top-K Capacity Routing (`dual_loop/controller.py`)**: Enforces static tensor shapes $[B, K_{\text{cap}}, D]$ to eliminate CUDA warp divergence (MoD-style).
83
- * **Calibrated Entropy Halting (`dual_loop/halting.py`)**: Adaptive stopping based on predictive uncertainty and convergence delta.
84
- * **Latent Deliberation Adapter (`dual_loop/adapters/latent_adapter.py`)**: A plug-and-play mid-network adapter for pretrained LLMs (e.g., Llama, Qwen).
85
-
86
- ---
87
-
88
- ## Quickstart
89
-
90
- ### 1. Installation
91
- ```bash
92
- git clone https://github.com/Ch3nOff/dual-loop-controller.git
93
- cd dual-loop-controller
94
- python -m pip install torch numpy
95
- ```
96
-
97
- ### 2. Running Component Tests (Verifying Shapes & Gradients)
98
- ```bash
99
- python -m unittest discover -s tests -p "test_*.py"
100
- ```
101
-
102
- ### 3. Running the Honest Benchmark Suite (Live Tensor Computations)
103
- ```bash
104
- python -m dual_loop.benchmarks.comprehensive_suite
105
- ```
106
-
107
- ### 4. Re-Training from Scratch
108
- ```bash
109
- python train.py --epochs 35 --hops 3 --k_steps 3 --d_model 64
110
- ```
111
-
112
- For the complete technical paper and theoretical post-mortem, see [WHITEPAPER.md](WHITEPAPER.md).
1
+ # Dual-Loop Cognitive Controller v2.0
2
+ > **A Hardware-Aligned Latent Deliberation Framework for Transformers: Architecture & Empirical Analysis**
3
+
4
+ [![PyPI](https://img.shields.io/pypi/v/dual-loop-controller.svg)](https://pypi.org/project/dual-loop-controller/)
5
+ [![Tests](https://img.shields.io/badge/tests-passing-brightgreen.svg)](tests/)
6
+ [![PyTorch](https://img.shields.io/badge/PyTorch-2.14%2B-ee4c2c.svg)](https://pytorch.org/)
7
+ [![Status](https://img.shields.io/badge/status-empirical--audit-orange.svg)](#empirical-findings)
8
+ [![License](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
9
+
10
+ Standard Autoregressive Transformers perform uniform $O(1)$ layer computation per token regardless of task complexity. While Chain-of-Thought (CoT) prompting allows multi-step reasoning, it expends significant output token bandwidth and introduces serial generation latency.
11
+
12
+ The **Dual-Loop Cognitive Controller** investigates decoupling deliberation from token generation into two loops:
13
+ 1. **Outer Loop (Executive Deliberation / System 2)**: Runs recursive state transitions in a continuous latent space without emitting intermediate tokens.
14
+ 2. **Inner Loop (Language Generation / System 1)**: Reads the matured latent thoughts ($H_{\text{thought}}$) as a soft prefix to decode final text responses.
15
+
16
+ ---
17
+
18
+ ## Empirical Findings & Negative Results (The Unvarnished Truth)
19
+
20
+ To maintain strict scientific integrity, this repository reports **the actual, measured behavior of the model trained end-to-end (225,959 parameters, 35 epochs, 3,500 samples, 16 nodes, chance baseline = 6.25%)**, rather than idealized projections.
21
+
22
+ ### 1. The Model Learns Real Relational Signals
23
+ * **Final Test Accuracy (3-Hop Graph Reasoning)**: **29.4%** vs. random chance **6.25%** (~4.7x better than random guessing).
24
+ * This confirms that the weight-tied recurrent Transformer and CWM buffer are capable of gradient propagation and multi-step pattern learning.
25
+
26
+ ### 2. The Absence of Monotonic Test-Time Compute Scaling
27
+ A central theoretical hypothesis of recurrent latent pondering is that increasing inference steps ($K$) will progressively improve answer accuracy. **On this 225K parameter implementation, this claim does not hold**:
28
+
29
+ ```text
30
+ ========================================================================================
31
+ EMPIRICAL TEST-TIME COMPUTE EVALUATION (Checkpoint: checkpoint_trained_dualloop.pt)
32
+ ========================================================================================
33
+ Ponder Steps (K) | Test Accuracy (500 samples) | Mean Predictive Entropy (nats)
34
+ ----------------------------------------------------------------------------------------
35
+ K = 0 (No Ponder)| 27.4% - 30.6% | 1.332 - 1.362 nats
36
+ K = 1 | 28.2% | 1.370 nats
37
+ K = 2 | 30.6% | 1.307 nats
38
+ K = 3 (Trained) | 30.4% | 1.268 nats
39
+ K = 4 | 30.0% | 1.268 nats
40
+ K = 5 | 31.6% | 1.275 nats
41
+ ========================================================================================
42
+ ```
43
+
44
+ **Scientific Diagnosis**:
45
+ * **Flat/Noisy Trajectory**: $K=0$ (bypassing the Outer Loop entirely) performs at parity with or slightly exceeds intermediate $K$ values.
46
+ * **Representational Drift**: Tracing individual predictions step-by-step reveals that while some cases improve with pondering, others degrade (e.g. correct at $K=0..1$, but diverging to incorrect candidates at $K=2..3$ due to distractor pull).
47
+ * **Scale Artifact vs. Fundamental Limit**: At 225K parameters, the latent space lacks the geometric capacity to preserve stable multi-step deductions without explicit discrete token anchors. Pondering without token-level supervision introduces noise as much as refinement.
48
+
49
+ ### 3. Degradation Under Context Distractors (Stress Test)
50
+ When distractor edge count increases on 3-hop graphs, performance decays steadily:
51
+ * **6 Edges**: 31.0%
52
+ * **8 Edges**: 21.0%
53
+ * **12 Edges**: 13.7%
54
+ * **16 Edges**: 10.3%
55
+
56
+ ### 4. Dynamic Halting Audit & The Pareto Trade-Off
57
+ A naive threshold like `0.5 nats` fails because the model operates at `~1.25–1.40 nats` (resulting in static $K=3.00$). Evaluating per-sample dynamic halting across a threshold sweep reveals the true **Accuracy vs. Compute Pareto Frontier**:
58
+
59
+ ```text
60
+ ========================================================================================
61
+ PER-SAMPLE DYNAMIC HALTING PARETO FRONTIER (500 Test Samples)
62
+ ========================================================================================
63
+ Entropy Threshold | Test Accuracy | Avg Steps | % Halt @ K=1 | % Halt @ K=2 | % Halt @ K=3
64
+ ----------------------------------------------------------------------------------------
65
+ tau = 0.80 nats | 28.0% | 2.81 | 7.6% | 3.8% | 88.6%
66
+ tau = 1.15 nats | 28.0% | 2.42 | 24.2% | 9.6% | 66.2%
67
+ tau = 1.25 nats | 28.8% | 2.23 | 32.2% | 12.2% | 55.6%
68
+ tau = 1.40 nats | 29.4% | 1.89 | 49.0% | 13.2% | 37.8%
69
+ ========================================================================================
70
+ ```
71
+
72
+ **Justified Operating Point**:
73
+ * **$\tau = 1.25 \dots 1.40\text{ nats}$** is the justifiable Pareto region: it achieves a **37% reduction in compute** (average **1.89 steps** vs. 3.00) while maintaining peak accuracy (**29.4%**), with a genuinely heterogeneous distribution across steps ($49\%$ at $K=1$, $13\%$ at $K=2$, $38\%$ at $K=3$).
74
+ ### 5. In-Distribution Memorization vs. Out-of-Distribution Generalization
75
+ A crucial empirical insight discovered during data isolation audits:
76
+ * **In-Distribution (Train Set, 500 seen graphs)**:
77
+ `K=0: 43.6% -> K=1: 51.4% -> K=2: 59.4% -> K=3: 63.2% (+19.6% monotonic test-time scaling)`
78
+ The recurrent latent controller successfully learns and memorizes multi-hop relational transitions for familiar graph topologies.
79
+ * **Out-of-Distribution (Held-Out Test Set, 500 unseen graphs)**:
80
+ `K=0: 28.6% -> K=1: 27.6% -> K=2: 28.0% -> K=3: 28.4% (Flat scaling / ~28-30%)`
81
+ Without discrete token anchors, continuous latent representations suffer from representational drift on novel graph structures at the 225K parameter regime.
82
+
83
+ ---
84
+
85
+ ## Architectural Implementation
86
+
87
+ Despite the scaling limits at small model regimes, the repository provides clean, production-grade PyTorch implementations of the core modules:
88
+
89
+ * **Cognitive Working Memory (`dual_loop/memory.py`)**: Compresses context into $M \ll N$ slots in GPU SRAM/L2 cache to avoid HBM memory bandwidth roundtrips.
90
+ * **Top-K Capacity Routing (`dual_loop/controller.py`)**: Enforces static tensor shapes $[B, K_{\text{cap}}, D]$ to eliminate CUDA warp divergence (MoD-style).
91
+ * **Calibrated Entropy Halting (`dual_loop/halting.py`)**: Adaptive stopping based on predictive uncertainty and convergence delta.
92
+ * **Latent Deliberation Adapter (`dual_loop/adapters/latent_adapter.py`)**: A plug-and-play mid-network adapter for pretrained LLMs (e.g., Llama, Qwen).
93
+
94
+ ---
95
+
96
+ ## Quickstart
97
+
98
+ ### 1. Installation
99
+ ```bash
100
+ # Install officially from PyPI:
101
+ pip install --pre dual-loop-controller
102
+ # or exact version: pip install dual-loop-controller==2.0.0a3
103
+
104
+ # Or install direct from GitHub release tag:
105
+ pip install git+https://github.com/Ch3nOff/dual-loop-controller.git@v2.0.0a3
106
+
107
+ # Or clone locally and install in editable mode:
108
+ git clone https://github.com/Ch3nOff/dual-loop-controller.git
109
+ cd dual-loop-controller
110
+ pip install -e .
111
+ ```
112
+
113
+ > [!NOTE]
114
+ > **Pretrained Weights Bundled**: A 225K parameter trained reference checkpoint (~912 KB) is bundled directly in `dual_loop/checkpoints/checkpoint_trained_dualloop.pt`. Fresh clones and pip installs run inference and audits out-of-the-box without requiring a training step first.
115
+
116
+ ### 2. Running Component Tests (Verifying Shapes & Gradients)
117
+ ```bash
118
+ python -m unittest discover -s tests -p "test_*.py"
119
+ ```
120
+
121
+ ### 3. Verifying Dynamic Halting & Pareto Calibration
122
+ ```bash
123
+ python verify_dynamic_inference.py
124
+ ```
125
+
126
+ ### 4. Running the Honest Benchmark Suite (Live Tensor Computations)
127
+ ```bash
128
+ python -m dual_loop.benchmarks.comprehensive_suite
129
+ ```
130
+
131
+ ### 5. Re-Training from Scratch
132
+ ```bash
133
+ python train.py --epochs 35 --hops 3 --k_steps 3 --d_model 64
134
+ ```
135
+
136
+ For the complete technical paper and theoretical post-mortem, see [WHITEPAPER.md](WHITEPAPER.md).
@@ -0,0 +1,66 @@
1
+ """
2
+ Dual-Loop Cognitive Controller v2.0
3
+ ===================================
4
+ A hardware-aligned, manifold-preserving latent reasoning framework
5
+ for Transformer architectures.
6
+ """
7
+
8
+ from .memory import CognitiveWorkingMemory
9
+ from .halting import EntropyHaltingUnit
10
+ from .controller import RecurrentLatentController, TopKCapacityCrossAttention
11
+ from .decoder import DualLoopTransformer
12
+ from .adapters.latent_adapter import LatentDeliberationAdapter
13
+
14
+ import os
15
+ import torch
16
+ from typing import Optional
17
+
18
+ __version__ = "2.0.0a3"
19
+
20
+ def get_default_checkpoint_path() -> Optional[str]:
21
+ """
22
+ Resolves the pre-trained reference checkpoint path by checking:
23
+ 1. Local working directory: './checkpoint_trained_dualloop.pt'
24
+ 2. Bundled package data: 'dual_loop/checkpoints/checkpoint_trained_dualloop.pt'
25
+ """
26
+ local_path = "checkpoint_trained_dualloop.pt"
27
+ if os.path.exists(local_path):
28
+ return os.path.abspath(local_path)
29
+
30
+ pkg_path = os.path.join(os.path.dirname(__file__), "checkpoints", "checkpoint_trained_dualloop.pt")
31
+ if os.path.exists(pkg_path):
32
+ return os.path.abspath(pkg_path)
33
+
34
+ return None
35
+
36
+ def load_trained_checkpoint(model: Optional[DualLoopTransformer] = None, checkpoint_path: Optional[str] = None):
37
+ """
38
+ Loads pre-trained weights into the DualLoopTransformer model, automatically falling back
39
+ to the bundled package checkpoint if no path is provided.
40
+ """
41
+ resolved_path = checkpoint_path or get_default_checkpoint_path()
42
+ if resolved_path is None or not os.path.exists(resolved_path):
43
+ raise FileNotFoundError(
44
+ "Pretrained checkpoint not found in local directory or bundled package data.\n"
45
+ "To train the model from scratch, execute:\n"
46
+ " python train.py --epochs 35 --hops 3 --k_steps 3\n"
47
+ "or run:\n"
48
+ " python evaluate_real_behavior.py"
49
+ )
50
+
51
+ state_dict = torch.load(resolved_path, map_location="cpu", weights_only=True)
52
+ if model is not None:
53
+ model.load_state_dict(state_dict, strict=False)
54
+ return model, resolved_path
55
+ return state_dict
56
+
57
+ __all__ = [
58
+ "CognitiveWorkingMemory",
59
+ "EntropyHaltingUnit",
60
+ "RecurrentLatentController",
61
+ "TopKCapacityCrossAttention",
62
+ "DualLoopTransformer",
63
+ "LatentDeliberationAdapter",
64
+ "get_default_checkpoint_path",
65
+ "load_trained_checkpoint",
66
+ ]