dual-loop-controller 2.2.1__tar.gz → 2.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dual_loop_controller-2.2.1/dual_loop_controller.egg-info → dual_loop_controller-2.2.2}/PKG-INFO +50 -8
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/README.md +49 -7
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/__init__.py +1 -1
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2/dual_loop_controller.egg-info}/PKG-INFO +50 -8
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/pyproject.toml +1 -1
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/LICENSE +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/adapters/__init__.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/adapters/latent_adapter.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/adapters/qwen_adapter.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/__init__.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/benchmark_qwen_reasoning.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/comprehensive_suite.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/graph_reasoning.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/halting_audit.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/initiative_benchmark.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/checkpoints/checkpoint_trained_dualloop.pt +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/controller.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/decoder.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/evidential.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/halting.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/matrix_helper.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/memory.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/open_concept.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/plasticity.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/verification.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop_controller.egg-info/SOURCES.txt +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop_controller.egg-info/dependency_links.txt +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop_controller.egg-info/requires.txt +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop_controller.egg-info/top_level.txt +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/setup.cfg +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_adapter_integration.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_dual_loop.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_episodic_self_correction.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_hypothesis_verification.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_matrix_helper.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_metacognitive_loop.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_plasticity_and_evidential.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_qwen_adapter.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_security_and_runtime.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_smart_brain_architecture.py +0 -0
- {dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_surprise_and_ddm.py +0 -0
{dual_loop_controller-2.2.1/dual_loop_controller.egg-info → dual_loop_controller-2.2.2}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dual-loop-controller
|
|
3
|
-
Version: 2.2.
|
|
3
|
+
Version: 2.2.2
|
|
4
4
|
Summary: A hardware-aligned, manifold-preserving latent deliberation framework for Transformers
|
|
5
5
|
Author: Ch3nOff
|
|
6
6
|
License-Expression: MIT
|
|
@@ -66,15 +66,15 @@ Standard autoregressive Transformers perform uniform $O(1)$ computation per toke
|
|
|
66
66
|
|
|
67
67
|
`dual-loop-controller` attaches seamlessly via non-invasive PyTorch forward hooks to any standard causal language model. No modifications to your underlying model weights are required:
|
|
68
68
|
|
|
69
|
-
| Model Family | Supported Architectures | Example Checkpoints |
|
|
69
|
+
| Model Family | Supported Architectures & Parameter Scales | Example Checkpoints |
|
|
70
70
|
| :--- | :--- | :--- |
|
|
71
|
-
| **Meta LLaMA** | LLaMA-2, LLaMA-3, LLaMA-3.1, LLaMA-3.2,
|
|
72
|
-
| **Mistral AI** | Mistral-7B, Mixtral-8x7B,
|
|
73
|
-
| **Qwen** | Qwen-1.5, Qwen-2, Qwen-2.5, Qwen-3.5 | `Qwen/Qwen2.5-
|
|
74
|
-
| **Google Gemma** | Gemma, Gemma-2 | `google/gemma-2-
|
|
75
|
-
| **DeepSeek** | DeepSeek-V2, DeepSeek-V3, DeepSeek-R1-Distill | `deepseek-ai/DeepSeek-R1-Distill-
|
|
71
|
+
| **Meta LLaMA** | LLaMA-2, LLaMA-3, LLaMA-3.1, LLaMA-3.2 (1B, 3B, 8B, 70B+) | `meta-llama/Meta-Llama-3-70B-Instruct`, `Llama-3.2-3B` |
|
|
72
|
+
| **Mistral AI** | Mistral-7B, Mixtral-8x7B, Mixtral-8x22B, Mistral Large | `mistralai/Mistral-7B-Instruct-v0.3`, `Mixtral-8x7B` |
|
|
73
|
+
| **Qwen** | Qwen-1.5, Qwen-2, Qwen-2.5, Qwen-3.5 (0.5B, 7B, 27B, 72B) | `Qwen/Qwen2.5-27B-Instruct`, `Qwen/Qwen2.5-72B`, `Qwen3.5-2B` |
|
|
74
|
+
| **Google Gemma** | Gemma, Gemma-2 (2B, 9B, 27B) | `google/gemma-2-27b-it`, `google/gemma-2-9b-it` |
|
|
75
|
+
| **DeepSeek** | DeepSeek-V2, DeepSeek-V3, DeepSeek-R1-Distill (1.5B to 70B) | `deepseek-ai/DeepSeek-R1-Distill-Llama-70B` |
|
|
76
76
|
| **Microsoft Phi**| Phi-2, Phi-3, Phi-3.5 | `microsoft/Phi-3-mini-4k-instruct` |
|
|
77
|
-
| **
|
|
77
|
+
| **Large Open-Weight (100B+)** | GPT-OSS-120B, Falcon-180B, DBRX, Bloom-176B | Any 70B–120B+ Transformer with multi-GPU sharding |
|
|
78
78
|
|
|
79
79
|
---
|
|
80
80
|
|
|
@@ -206,6 +206,48 @@ if match:
|
|
|
206
206
|
|
|
207
207
|
---
|
|
208
208
|
|
|
209
|
+
### 4. Scaling to Large Models (27B, 70B, 120B+) with Multi-GPU & 4-bit Quantization
|
|
210
|
+
|
|
211
|
+
On large models (such as **Qwen 27B**, **LLaMA-3 70B**, or **120B+ open-weight models**), standard Chain-of-Thought (CoT) prompting generates 1,000–3,000 discrete tokens, incurring 30–60 seconds of latency and massive KV-cache VRAM consumption.
|
|
212
|
+
|
|
213
|
+
**Dual-Loop Controller** solves this by performing System 2 deliberation in continuous latent space ($D=5120\dots 10240$), finishing in milliseconds with zero output token bloat. It is fully compatible with **Multi-GPU Sharding (`device_map="auto"`)** and **BitsAndBytes 4-bit / 8-bit Quantization**:
|
|
214
|
+
|
|
215
|
+
```python
|
|
216
|
+
import torch
|
|
217
|
+
from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig
|
|
218
|
+
from dual_loop import attach_dual_loop
|
|
219
|
+
|
|
220
|
+
# Example: Running on Qwen-27B, LLaMA-70B, or 120B model with 4-bit quantization
|
|
221
|
+
model_id = "Qwen/Qwen2.5-27B-Instruct" # or "meta-llama/Meta-Llama-3-70B-Instruct"
|
|
222
|
+
|
|
223
|
+
# 1. Configure 4-bit NF4 quantization to fit on consumer/prosumer GPUs
|
|
224
|
+
bnb_config = BitsAndBytesConfig(
|
|
225
|
+
load_in_4bit=True,
|
|
226
|
+
bnb_4bit_quant_type="nf4",
|
|
227
|
+
bnb_4bit_compute_dtype=torch.bfloat16
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
tokenizer = AutoTokenizer.from_pretrained(model_id)
|
|
231
|
+
base_model = AutoModelForCausalLM.from_pretrained(
|
|
232
|
+
model_id,
|
|
233
|
+
quantization_config=bnb_config,
|
|
234
|
+
device_map="auto" # Automatically shards across GPU 0, 1, etc.
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
# 2. Attach Dual-Loop Controller (Automatically matches layer device and precision)
|
|
238
|
+
model = attach_dual_loop(base_model, k_steps=2)
|
|
239
|
+
|
|
240
|
+
# 3. High-efficiency inference without CoT token overhead
|
|
241
|
+
prompt = "Question: Analyze the fault tolerance of this distributed Byzantine consensus protocol:\nAnswer:"
|
|
242
|
+
inputs = tokenizer(prompt, return_tensors="pt").to(base_model.device)
|
|
243
|
+
|
|
244
|
+
# Model deliberates in latent vectors (SRAM cache) rather than emitting 1000s of CoT tokens
|
|
245
|
+
output = model.generate(**inputs, max_new_tokens=128)
|
|
246
|
+
print(tokenizer.decode(output[0], skip_special_tokens=True))
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
---
|
|
250
|
+
|
|
209
251
|
## Why Should I Use Dual-Loop Controller?
|
|
210
252
|
|
|
211
253
|
* **Universal Compatibility**: Works with Llama, Mistral, Qwen, Gemma, DeepSeek, and any causal LM.
|
|
@@ -34,15 +34,15 @@ Standard autoregressive Transformers perform uniform $O(1)$ computation per toke
|
|
|
34
34
|
|
|
35
35
|
`dual-loop-controller` attaches seamlessly via non-invasive PyTorch forward hooks to any standard causal language model. No modifications to your underlying model weights are required:
|
|
36
36
|
|
|
37
|
-
| Model Family | Supported Architectures | Example Checkpoints |
|
|
37
|
+
| Model Family | Supported Architectures & Parameter Scales | Example Checkpoints |
|
|
38
38
|
| :--- | :--- | :--- |
|
|
39
|
-
| **Meta LLaMA** | LLaMA-2, LLaMA-3, LLaMA-3.1, LLaMA-3.2,
|
|
40
|
-
| **Mistral AI** | Mistral-7B, Mixtral-8x7B,
|
|
41
|
-
| **Qwen** | Qwen-1.5, Qwen-2, Qwen-2.5, Qwen-3.5 | `Qwen/Qwen2.5-
|
|
42
|
-
| **Google Gemma** | Gemma, Gemma-2 | `google/gemma-2-
|
|
43
|
-
| **DeepSeek** | DeepSeek-V2, DeepSeek-V3, DeepSeek-R1-Distill | `deepseek-ai/DeepSeek-R1-Distill-
|
|
39
|
+
| **Meta LLaMA** | LLaMA-2, LLaMA-3, LLaMA-3.1, LLaMA-3.2 (1B, 3B, 8B, 70B+) | `meta-llama/Meta-Llama-3-70B-Instruct`, `Llama-3.2-3B` |
|
|
40
|
+
| **Mistral AI** | Mistral-7B, Mixtral-8x7B, Mixtral-8x22B, Mistral Large | `mistralai/Mistral-7B-Instruct-v0.3`, `Mixtral-8x7B` |
|
|
41
|
+
| **Qwen** | Qwen-1.5, Qwen-2, Qwen-2.5, Qwen-3.5 (0.5B, 7B, 27B, 72B) | `Qwen/Qwen2.5-27B-Instruct`, `Qwen/Qwen2.5-72B`, `Qwen3.5-2B` |
|
|
42
|
+
| **Google Gemma** | Gemma, Gemma-2 (2B, 9B, 27B) | `google/gemma-2-27b-it`, `google/gemma-2-9b-it` |
|
|
43
|
+
| **DeepSeek** | DeepSeek-V2, DeepSeek-V3, DeepSeek-R1-Distill (1.5B to 70B) | `deepseek-ai/DeepSeek-R1-Distill-Llama-70B` |
|
|
44
44
|
| **Microsoft Phi**| Phi-2, Phi-3, Phi-3.5 | `microsoft/Phi-3-mini-4k-instruct` |
|
|
45
|
-
| **
|
|
45
|
+
| **Large Open-Weight (100B+)** | GPT-OSS-120B, Falcon-180B, DBRX, Bloom-176B | Any 70B–120B+ Transformer with multi-GPU sharding |
|
|
46
46
|
|
|
47
47
|
---
|
|
48
48
|
|
|
@@ -174,6 +174,48 @@ if match:
|
|
|
174
174
|
|
|
175
175
|
---
|
|
176
176
|
|
|
177
|
+
### 4. Scaling to Large Models (27B, 70B, 120B+) with Multi-GPU & 4-bit Quantization
|
|
178
|
+
|
|
179
|
+
On large models (such as **Qwen 27B**, **LLaMA-3 70B**, or **120B+ open-weight models**), standard Chain-of-Thought (CoT) prompting generates 1,000–3,000 discrete tokens, incurring 30–60 seconds of latency and massive KV-cache VRAM consumption.
|
|
180
|
+
|
|
181
|
+
**Dual-Loop Controller** solves this by performing System 2 deliberation in continuous latent space ($D=5120\dots 10240$), finishing in milliseconds with zero output token bloat. It is fully compatible with **Multi-GPU Sharding (`device_map="auto"`)** and **BitsAndBytes 4-bit / 8-bit Quantization**:
|
|
182
|
+
|
|
183
|
+
```python
|
|
184
|
+
import torch
|
|
185
|
+
from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig
|
|
186
|
+
from dual_loop import attach_dual_loop
|
|
187
|
+
|
|
188
|
+
# Example: Running on Qwen-27B, LLaMA-70B, or 120B model with 4-bit quantization
|
|
189
|
+
model_id = "Qwen/Qwen2.5-27B-Instruct" # or "meta-llama/Meta-Llama-3-70B-Instruct"
|
|
190
|
+
|
|
191
|
+
# 1. Configure 4-bit NF4 quantization to fit on consumer/prosumer GPUs
|
|
192
|
+
bnb_config = BitsAndBytesConfig(
|
|
193
|
+
load_in_4bit=True,
|
|
194
|
+
bnb_4bit_quant_type="nf4",
|
|
195
|
+
bnb_4bit_compute_dtype=torch.bfloat16
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
tokenizer = AutoTokenizer.from_pretrained(model_id)
|
|
199
|
+
base_model = AutoModelForCausalLM.from_pretrained(
|
|
200
|
+
model_id,
|
|
201
|
+
quantization_config=bnb_config,
|
|
202
|
+
device_map="auto" # Automatically shards across GPU 0, 1, etc.
|
|
203
|
+
)
|
|
204
|
+
|
|
205
|
+
# 2. Attach Dual-Loop Controller (Automatically matches layer device and precision)
|
|
206
|
+
model = attach_dual_loop(base_model, k_steps=2)
|
|
207
|
+
|
|
208
|
+
# 3. High-efficiency inference without CoT token overhead
|
|
209
|
+
prompt = "Question: Analyze the fault tolerance of this distributed Byzantine consensus protocol:\nAnswer:"
|
|
210
|
+
inputs = tokenizer(prompt, return_tensors="pt").to(base_model.device)
|
|
211
|
+
|
|
212
|
+
# Model deliberates in latent vectors (SRAM cache) rather than emitting 1000s of CoT tokens
|
|
213
|
+
output = model.generate(**inputs, max_new_tokens=128)
|
|
214
|
+
print(tokenizer.decode(output[0], skip_special_tokens=True))
|
|
215
|
+
```
|
|
216
|
+
|
|
217
|
+
---
|
|
218
|
+
|
|
177
219
|
## Why Should I Use Dual-Loop Controller?
|
|
178
220
|
|
|
179
221
|
* **Universal Compatibility**: Works with Llama, Mistral, Qwen, Gemma, DeepSeek, and any causal LM.
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2/dual_loop_controller.egg-info}/PKG-INFO
RENAMED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: dual-loop-controller
|
|
3
|
-
Version: 2.2.
|
|
3
|
+
Version: 2.2.2
|
|
4
4
|
Summary: A hardware-aligned, manifold-preserving latent deliberation framework for Transformers
|
|
5
5
|
Author: Ch3nOff
|
|
6
6
|
License-Expression: MIT
|
|
@@ -66,15 +66,15 @@ Standard autoregressive Transformers perform uniform $O(1)$ computation per toke
|
|
|
66
66
|
|
|
67
67
|
`dual-loop-controller` attaches seamlessly via non-invasive PyTorch forward hooks to any standard causal language model. No modifications to your underlying model weights are required:
|
|
68
68
|
|
|
69
|
-
| Model Family | Supported Architectures | Example Checkpoints |
|
|
69
|
+
| Model Family | Supported Architectures & Parameter Scales | Example Checkpoints |
|
|
70
70
|
| :--- | :--- | :--- |
|
|
71
|
-
| **Meta LLaMA** | LLaMA-2, LLaMA-3, LLaMA-3.1, LLaMA-3.2,
|
|
72
|
-
| **Mistral AI** | Mistral-7B, Mixtral-8x7B,
|
|
73
|
-
| **Qwen** | Qwen-1.5, Qwen-2, Qwen-2.5, Qwen-3.5 | `Qwen/Qwen2.5-
|
|
74
|
-
| **Google Gemma** | Gemma, Gemma-2 | `google/gemma-2-
|
|
75
|
-
| **DeepSeek** | DeepSeek-V2, DeepSeek-V3, DeepSeek-R1-Distill | `deepseek-ai/DeepSeek-R1-Distill-
|
|
71
|
+
| **Meta LLaMA** | LLaMA-2, LLaMA-3, LLaMA-3.1, LLaMA-3.2 (1B, 3B, 8B, 70B+) | `meta-llama/Meta-Llama-3-70B-Instruct`, `Llama-3.2-3B` |
|
|
72
|
+
| **Mistral AI** | Mistral-7B, Mixtral-8x7B, Mixtral-8x22B, Mistral Large | `mistralai/Mistral-7B-Instruct-v0.3`, `Mixtral-8x7B` |
|
|
73
|
+
| **Qwen** | Qwen-1.5, Qwen-2, Qwen-2.5, Qwen-3.5 (0.5B, 7B, 27B, 72B) | `Qwen/Qwen2.5-27B-Instruct`, `Qwen/Qwen2.5-72B`, `Qwen3.5-2B` |
|
|
74
|
+
| **Google Gemma** | Gemma, Gemma-2 (2B, 9B, 27B) | `google/gemma-2-27b-it`, `google/gemma-2-9b-it` |
|
|
75
|
+
| **DeepSeek** | DeepSeek-V2, DeepSeek-V3, DeepSeek-R1-Distill (1.5B to 70B) | `deepseek-ai/DeepSeek-R1-Distill-Llama-70B` |
|
|
76
76
|
| **Microsoft Phi**| Phi-2, Phi-3, Phi-3.5 | `microsoft/Phi-3-mini-4k-instruct` |
|
|
77
|
-
| **
|
|
77
|
+
| **Large Open-Weight (100B+)** | GPT-OSS-120B, Falcon-180B, DBRX, Bloom-176B | Any 70B–120B+ Transformer with multi-GPU sharding |
|
|
78
78
|
|
|
79
79
|
---
|
|
80
80
|
|
|
@@ -206,6 +206,48 @@ if match:
|
|
|
206
206
|
|
|
207
207
|
---
|
|
208
208
|
|
|
209
|
+
### 4. Scaling to Large Models (27B, 70B, 120B+) with Multi-GPU & 4-bit Quantization
|
|
210
|
+
|
|
211
|
+
On large models (such as **Qwen 27B**, **LLaMA-3 70B**, or **120B+ open-weight models**), standard Chain-of-Thought (CoT) prompting generates 1,000–3,000 discrete tokens, incurring 30–60 seconds of latency and massive KV-cache VRAM consumption.
|
|
212
|
+
|
|
213
|
+
**Dual-Loop Controller** solves this by performing System 2 deliberation in continuous latent space ($D=5120\dots 10240$), finishing in milliseconds with zero output token bloat. It is fully compatible with **Multi-GPU Sharding (`device_map="auto"`)** and **BitsAndBytes 4-bit / 8-bit Quantization**:
|
|
214
|
+
|
|
215
|
+
```python
|
|
216
|
+
import torch
|
|
217
|
+
from transformers import AutoModelForCausalLM, AutoTokenizer, BitsAndBytesConfig
|
|
218
|
+
from dual_loop import attach_dual_loop
|
|
219
|
+
|
|
220
|
+
# Example: Running on Qwen-27B, LLaMA-70B, or 120B model with 4-bit quantization
|
|
221
|
+
model_id = "Qwen/Qwen2.5-27B-Instruct" # or "meta-llama/Meta-Llama-3-70B-Instruct"
|
|
222
|
+
|
|
223
|
+
# 1. Configure 4-bit NF4 quantization to fit on consumer/prosumer GPUs
|
|
224
|
+
bnb_config = BitsAndBytesConfig(
|
|
225
|
+
load_in_4bit=True,
|
|
226
|
+
bnb_4bit_quant_type="nf4",
|
|
227
|
+
bnb_4bit_compute_dtype=torch.bfloat16
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
tokenizer = AutoTokenizer.from_pretrained(model_id)
|
|
231
|
+
base_model = AutoModelForCausalLM.from_pretrained(
|
|
232
|
+
model_id,
|
|
233
|
+
quantization_config=bnb_config,
|
|
234
|
+
device_map="auto" # Automatically shards across GPU 0, 1, etc.
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
# 2. Attach Dual-Loop Controller (Automatically matches layer device and precision)
|
|
238
|
+
model = attach_dual_loop(base_model, k_steps=2)
|
|
239
|
+
|
|
240
|
+
# 3. High-efficiency inference without CoT token overhead
|
|
241
|
+
prompt = "Question: Analyze the fault tolerance of this distributed Byzantine consensus protocol:\nAnswer:"
|
|
242
|
+
inputs = tokenizer(prompt, return_tensors="pt").to(base_model.device)
|
|
243
|
+
|
|
244
|
+
# Model deliberates in latent vectors (SRAM cache) rather than emitting 1000s of CoT tokens
|
|
245
|
+
output = model.generate(**inputs, max_new_tokens=128)
|
|
246
|
+
print(tokenizer.decode(output[0], skip_special_tokens=True))
|
|
247
|
+
```
|
|
248
|
+
|
|
249
|
+
---
|
|
250
|
+
|
|
209
251
|
## Why Should I Use Dual-Loop Controller?
|
|
210
252
|
|
|
211
253
|
* **Universal Compatibility**: Works with Llama, Mistral, Qwen, Gemma, DeepSeek, and any causal LM.
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "dual-loop-controller"
|
|
7
|
-
version = "2.2.
|
|
7
|
+
version = "2.2.2"
|
|
8
8
|
description = "A hardware-aligned, manifold-preserving latent deliberation framework for Transformers"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.9"
|
|
File without changes
|
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/adapters/latent_adapter.py
RENAMED
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/adapters/qwen_adapter.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/graph_reasoning.py
RENAMED
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop/benchmarks/halting_audit.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop_controller.egg-info/SOURCES.txt
RENAMED
|
File without changes
|
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/dual_loop_controller.egg-info/requires.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_episodic_self_correction.py
RENAMED
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_hypothesis_verification.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_plasticity_and_evidential.py
RENAMED
|
File without changes
|
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_security_and_runtime.py
RENAMED
|
File without changes
|
{dual_loop_controller-2.2.1 → dual_loop_controller-2.2.2}/tests/test_smart_brain_architecture.py
RENAMED
|
File without changes
|
|
File without changes
|