simthinkd 0.1.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {simthinkd-0.1.0/src/simthinkd.egg-info → simthinkd-0.2.1}/PKG-INFO +94 -14
- {simthinkd-0.1.0 → simthinkd-0.2.1}/README.md +93 -13
- {simthinkd-0.1.0 → simthinkd-0.2.1}/pyproject.toml +1 -1
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/__init__.py +9 -2
- simthinkd-0.2.1/src/simthinkd/score.py +137 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/server.py +16 -1
- {simthinkd-0.1.0 → simthinkd-0.2.1/src/simthinkd.egg-info}/PKG-INFO +94 -14
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/SOURCES.txt +4 -0
- simthinkd-0.2.1/tests/test_factory_twin.py +65 -0
- simthinkd-0.2.1/tests/test_score.py +67 -0
- simthinkd-0.2.1/tests/test_server_latency.py +30 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_web_page.py +2 -2
- {simthinkd-0.1.0 → simthinkd-0.2.1}/LICENSE +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/NOTICE +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/setup.cfg +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/bench.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/cli.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/core.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/data/doom_defend_states.json +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/integrations/__init__.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/integrations/langchain_tool.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/integrations/mcp_server.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/policy.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/toy.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/train.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/weights/SHA256SUMS +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/weights/doom-corridor.npz +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/weights/doom-defend.npz +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/dependency_links.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/entry_points.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/requires.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/top_level.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_notebook.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_package.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_space.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_web_parity.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: simthinkd
|
|
3
|
-
Version: 0.1
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
|
|
5
5
|
Author: Myeongseongsimjae AX Institute
|
|
6
6
|
License: Apache-2.0
|
|
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
|
|
|
28
28
|
Requires-Dist: onnxruntime>=1.17; extra == "onnx"
|
|
29
29
|
Dynamic: license-file
|
|
30
30
|
|
|
31
|
-
<p align="center"><
|
|
31
|
+
<p align="center"><img src="assets/banner_v2.png" alt="SimThink D: a local backup for cloud decisions" width="100%"></p>
|
|
32
32
|
|
|
33
33
|
<p align="center">
|
|
34
|
+
<a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
|
|
34
35
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
|
|
35
36
|
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
36
37
|
<a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
|
|
@@ -38,21 +39,23 @@ Dynamic: license-file
|
|
|
38
39
|
<a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
|
|
39
40
|
</p>
|
|
40
41
|
|
|
41
|
-
|
|
42
|
+
<p align="center"><b>Network down. Decisions stay local.</b></p>
|
|
42
43
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
</p>
|
|
44
|
+
SimThink D is a small CPU model for offline backup decisions.
|
|
45
|
+
It has 265,665 parameters and takes about 2 ms per decision on one CPU core.
|
|
46
46
|
|
|
47
|
-
|
|
47
|
+
In the factory simulation, the internet was cut for 15 seconds.
|
|
48
|
+
The local backup got 59 of 63 parts right, with no late decisions.
|
|
48
49
|
|
|
49
|
-
<
|
|
50
|
+
<p align="center"><a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Watch the factory simulation: local backup during a network outage" width="100%"></a></p>
|
|
50
51
|
|
|
51
|
-
|
|
52
|
+
<p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/#demo">Try in your browser</a> · <a href="assets/factory_fallback.mp4">Watch the video (72 s)</a> · <a href="examples/factory_twin/">Factory code</a> · <a href="docs/PAPER.md">Paper</a></p>
|
|
52
53
|
|
|
53
|
-
|
|
54
|
+
```bash
|
|
55
|
+
pip install simthinkd
|
|
56
|
+
```
|
|
54
57
|
|
|
55
|
-
|
|
58
|
+
What makes the next decision when your network goes down?
|
|
56
59
|
|
|
57
60
|
## Words used here
|
|
58
61
|
|
|
@@ -64,13 +67,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
|
|
|
64
67
|
## Install
|
|
65
68
|
|
|
66
69
|
```bash
|
|
67
|
-
pip install
|
|
70
|
+
pip install simthinkd
|
|
68
71
|
```
|
|
69
72
|
|
|
70
73
|
That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
|
|
71
74
|
|
|
72
75
|
```bash
|
|
73
|
-
pip install "simthinkd[train]
|
|
76
|
+
pip install "simthinkd[train]"
|
|
74
77
|
```
|
|
75
78
|
|
|
76
79
|
## Quickstart
|
|
@@ -102,6 +105,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
|
|
|
102
105
|
|
|
103
106
|
On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
|
|
104
107
|
|
|
108
|
+
## Score instead of choose
|
|
109
|
+
|
|
110
|
+
Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
import random
|
|
114
|
+
import simthinkd
|
|
115
|
+
from simthinkd import toy
|
|
116
|
+
|
|
117
|
+
SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
|
|
118
|
+
|
|
119
|
+
def risk(s): # your own rule or records give the number
|
|
120
|
+
value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
|
|
121
|
+
value += 10 if s["image"] == "blurry" else 0
|
|
122
|
+
value += 8 if s["belt"] == "fast" else 0
|
|
123
|
+
value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
|
|
124
|
+
return float(min(100, value))
|
|
125
|
+
|
|
126
|
+
rng = random.Random(7)
|
|
127
|
+
examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
|
|
128
|
+
s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
|
|
129
|
+
print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
|
|
130
|
+
# 87.99 (0.22 ms)
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
|
|
134
|
+
|
|
135
|
+
## Where it fits
|
|
136
|
+
|
|
137
|
+
SimThink D is a good fit when all three are true:
|
|
138
|
+
|
|
139
|
+
1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
|
|
140
|
+
2. **The answer is small.** One of up to 8 actions, or one number.
|
|
141
|
+
3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
|
|
142
|
+
|
|
143
|
+
| Use | Why it fits | Try it here |
|
|
144
|
+
|---|---|---|
|
|
145
|
+
| Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
|
|
146
|
+
| Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
|
|
147
|
+
| Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
|
|
148
|
+
| A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
|
|
149
|
+
|
|
150
|
+
### Use it together with a large model
|
|
151
|
+
|
|
152
|
+
Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
|
|
153
|
+
|
|
154
|
+
| Pattern | How it works | Try it here |
|
|
155
|
+
|---|---|---|
|
|
156
|
+
| **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
|
|
157
|
+
| **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
|
|
158
|
+
|
|
159
|
+
A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
|
|
160
|
+
|
|
161
|
+
It is not a good fit for:
|
|
162
|
+
|
|
163
|
+
- **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
|
|
164
|
+
- **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
|
|
165
|
+
- **Open-ended answers.** It picks from a list or returns one number. It does not write.
|
|
166
|
+
|
|
105
167
|
## Measure your own model
|
|
106
168
|
|
|
107
169
|
Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
|
|
@@ -115,6 +177,12 @@ The second line measures any server that accepts the [decision request](docs/PRO
|
|
|
115
177
|
|
|
116
178
|
## Same CPU, same states
|
|
117
179
|
|
|
180
|
+
<p align="center">
|
|
181
|
+
<img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
|
|
182
|
+
</p>
|
|
183
|
+
|
|
184
|
+
<p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
|
|
185
|
+
|
|
118
186
|
**Read this first.** Laya is a general model and was not trained on this game. Its published speed, about 33 ms per question, is on a GPU. We only had a CPU. So this table compares time inside a real-time loop. It does not compare overall quality.
|
|
119
187
|
|
|
120
188
|
We used one workstation CPU (6 threads) and 1,050 Doom states, then 10 live games on the same seeds. "Missed ticks" are ticks that passed before the decider answered.
|
|
@@ -133,7 +201,9 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
133
201
|
| Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
|
|
134
202
|
| Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
|
|
135
203
|
| Browser | [web/](web/): the same model in plain JavaScript, no server |
|
|
136
|
-
|
|
|
204
|
+
| A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
|
|
205
|
+
| Gradio | [space/](space/): a small web demo you can run locally or on Hugging Face Spaces |
|
|
206
|
+
| MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
|
|
137
207
|
| LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
|
|
138
208
|
|
|
139
209
|
## How it works
|
|
@@ -149,6 +219,14 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
149
219
|
- Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
|
|
150
220
|
- A decider copies its teacher. It is only as good as the teacher's rules.
|
|
151
221
|
- Probabilities are calibrated for the decider's own task only.
|
|
222
|
+
- A score model returns one number. It gives no probability or error bar with it.
|
|
223
|
+
|
|
224
|
+
## More
|
|
225
|
+
|
|
226
|
+
- [Figures from the paper](docs/FIGURES.md)
|
|
227
|
+
- [What you can reproduce](docs/REPRODUCE.md)
|
|
228
|
+
- [Decision request format](docs/PROTOCOL.md)
|
|
229
|
+
- [The factory video on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/)
|
|
152
230
|
|
|
153
231
|
## Citation
|
|
154
232
|
|
|
@@ -157,6 +235,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
|
|
|
157
235
|
- Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
|
|
158
236
|
- Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
|
|
159
237
|
|
|
238
|
+
The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
|
|
239
|
+
|
|
160
240
|
## Contributing
|
|
161
241
|
|
|
162
242
|
Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
<p align="center"><
|
|
1
|
+
<p align="center"><img src="assets/banner_v2.png" alt="SimThink D: a local backup for cloud decisions" width="100%"></p>
|
|
2
2
|
|
|
3
3
|
<p align="center">
|
|
4
|
+
<a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
|
|
4
5
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
|
|
5
6
|
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
6
7
|
<a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
|
|
@@ -8,21 +9,23 @@
|
|
|
8
9
|
<a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
|
|
9
10
|
</p>
|
|
10
11
|
|
|
11
|
-
|
|
12
|
+
<p align="center"><b>Network down. Decisions stay local.</b></p>
|
|
12
13
|
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
</p>
|
|
14
|
+
SimThink D is a small CPU model for offline backup decisions.
|
|
15
|
+
It has 265,665 parameters and takes about 2 ms per decision on one CPU core.
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
In the factory simulation, the internet was cut for 15 seconds.
|
|
18
|
+
The local backup got 59 of 63 parts right, with no late decisions.
|
|
18
19
|
|
|
19
|
-
<
|
|
20
|
+
<p align="center"><a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Watch the factory simulation: local backup during a network outage" width="100%"></a></p>
|
|
20
21
|
|
|
21
|
-
|
|
22
|
+
<p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/#demo">Try in your browser</a> · <a href="assets/factory_fallback.mp4">Watch the video (72 s)</a> · <a href="examples/factory_twin/">Factory code</a> · <a href="docs/PAPER.md">Paper</a></p>
|
|
22
23
|
|
|
23
|
-
|
|
24
|
+
```bash
|
|
25
|
+
pip install simthinkd
|
|
26
|
+
```
|
|
24
27
|
|
|
25
|
-
|
|
28
|
+
What makes the next decision when your network goes down?
|
|
26
29
|
|
|
27
30
|
## Words used here
|
|
28
31
|
|
|
@@ -34,13 +37,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
|
|
|
34
37
|
## Install
|
|
35
38
|
|
|
36
39
|
```bash
|
|
37
|
-
pip install
|
|
40
|
+
pip install simthinkd
|
|
38
41
|
```
|
|
39
42
|
|
|
40
43
|
That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
|
|
41
44
|
|
|
42
45
|
```bash
|
|
43
|
-
pip install "simthinkd[train]
|
|
46
|
+
pip install "simthinkd[train]"
|
|
44
47
|
```
|
|
45
48
|
|
|
46
49
|
## Quickstart
|
|
@@ -72,6 +75,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
|
|
|
72
75
|
|
|
73
76
|
On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
|
|
74
77
|
|
|
78
|
+
## Score instead of choose
|
|
79
|
+
|
|
80
|
+
Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
import random
|
|
84
|
+
import simthinkd
|
|
85
|
+
from simthinkd import toy
|
|
86
|
+
|
|
87
|
+
SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
|
|
88
|
+
|
|
89
|
+
def risk(s): # your own rule or records give the number
|
|
90
|
+
value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
|
|
91
|
+
value += 10 if s["image"] == "blurry" else 0
|
|
92
|
+
value += 8 if s["belt"] == "fast" else 0
|
|
93
|
+
value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
|
|
94
|
+
return float(min(100, value))
|
|
95
|
+
|
|
96
|
+
rng = random.Random(7)
|
|
97
|
+
examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
|
|
98
|
+
s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
|
|
99
|
+
print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
|
|
100
|
+
# 87.99 (0.22 ms)
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
|
|
104
|
+
|
|
105
|
+
## Where it fits
|
|
106
|
+
|
|
107
|
+
SimThink D is a good fit when all three are true:
|
|
108
|
+
|
|
109
|
+
1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
|
|
110
|
+
2. **The answer is small.** One of up to 8 actions, or one number.
|
|
111
|
+
3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
|
|
112
|
+
|
|
113
|
+
| Use | Why it fits | Try it here |
|
|
114
|
+
|---|---|---|
|
|
115
|
+
| Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
|
|
116
|
+
| Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
|
|
117
|
+
| Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
|
|
118
|
+
| A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
|
|
119
|
+
|
|
120
|
+
### Use it together with a large model
|
|
121
|
+
|
|
122
|
+
Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
|
|
123
|
+
|
|
124
|
+
| Pattern | How it works | Try it here |
|
|
125
|
+
|---|---|---|
|
|
126
|
+
| **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
|
|
127
|
+
| **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
|
|
128
|
+
|
|
129
|
+
A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
|
|
130
|
+
|
|
131
|
+
It is not a good fit for:
|
|
132
|
+
|
|
133
|
+
- **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
|
|
134
|
+
- **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
|
|
135
|
+
- **Open-ended answers.** It picks from a list or returns one number. It does not write.
|
|
136
|
+
|
|
75
137
|
## Measure your own model
|
|
76
138
|
|
|
77
139
|
Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
|
|
@@ -85,6 +147,12 @@ The second line measures any server that accepts the [decision request](docs/PRO
|
|
|
85
147
|
|
|
86
148
|
## Same CPU, same states
|
|
87
149
|
|
|
150
|
+
<p align="center">
|
|
151
|
+
<img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
|
|
152
|
+
</p>
|
|
153
|
+
|
|
154
|
+
<p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
|
|
155
|
+
|
|
88
156
|
**Read this first.** Laya is a general model and was not trained on this game. Its published speed, about 33 ms per question, is on a GPU. We only had a CPU. So this table compares time inside a real-time loop. It does not compare overall quality.
|
|
89
157
|
|
|
90
158
|
We used one workstation CPU (6 threads) and 1,050 Doom states, then 10 live games on the same seeds. "Missed ticks" are ticks that passed before the decider answered.
|
|
@@ -103,7 +171,9 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
103
171
|
| Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
|
|
104
172
|
| Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
|
|
105
173
|
| Browser | [web/](web/): the same model in plain JavaScript, no server |
|
|
106
|
-
|
|
|
174
|
+
| A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
|
|
175
|
+
| Gradio | [space/](space/): a small web demo you can run locally or on Hugging Face Spaces |
|
|
176
|
+
| MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
|
|
107
177
|
| LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
|
|
108
178
|
|
|
109
179
|
## How it works
|
|
@@ -119,6 +189,14 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
119
189
|
- Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
|
|
120
190
|
- A decider copies its teacher. It is only as good as the teacher's rules.
|
|
121
191
|
- Probabilities are calibrated for the decider's own task only.
|
|
192
|
+
- A score model returns one number. It gives no probability or error bar with it.
|
|
193
|
+
|
|
194
|
+
## More
|
|
195
|
+
|
|
196
|
+
- [Figures from the paper](docs/FIGURES.md)
|
|
197
|
+
- [What you can reproduce](docs/REPRODUCE.md)
|
|
198
|
+
- [Decision request format](docs/PROTOCOL.md)
|
|
199
|
+
- [The factory video on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/)
|
|
122
200
|
|
|
123
201
|
## Citation
|
|
124
202
|
|
|
@@ -127,6 +205,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
|
|
|
127
205
|
- Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
|
|
128
206
|
- Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
|
|
129
207
|
|
|
208
|
+
The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
|
|
209
|
+
|
|
130
210
|
## Contributing
|
|
131
211
|
|
|
132
212
|
Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "simthinkd"
|
|
7
|
-
version = "0.1
|
|
7
|
+
version = "0.2.1"
|
|
8
8
|
description = "A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -4,12 +4,19 @@
|
|
|
4
4
|
print(Decider("doom-defend").decide("seen: Demon left a30 d5 | enemies 1 | sway left | gun ready | ammo25"))
|
|
5
5
|
"""
|
|
6
6
|
from .core import Decider, Decision, available, build_request
|
|
7
|
+
from .score import Score, Scorer
|
|
7
8
|
|
|
8
|
-
__version__ = '0.1
|
|
9
|
-
__all__ = ['Decider', 'Decision', 'available', 'build_request', 'fit', '__version__']
|
|
9
|
+
__version__ = '0.2.1'
|
|
10
|
+
__all__ = ['Decider', 'Decision', 'Score', 'Scorer', 'available', 'build_request', 'fit', 'fit_score', '__version__']
|
|
10
11
|
|
|
11
12
|
|
|
12
13
|
def fit(*args, **kwargs):
|
|
13
14
|
"""Train a decider from (situation, action) pairs. See simthinkd.train.fit. Needs simthinkd[train]."""
|
|
14
15
|
from .train import fit as _fit
|
|
15
16
|
return _fit(*args, **kwargs)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def fit_score(*args, **kwargs):
|
|
20
|
+
"""Train a score model from (situation, number) pairs. See simthinkd.score.fit_score. Needs simthinkd[train]."""
|
|
21
|
+
from .score import fit_score as _fit_score
|
|
22
|
+
return _fit_score(*args, **kwargs)
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Score models: the same small network as a decider, but it returns one number instead of picking an action.
|
|
2
|
+
|
|
3
|
+
import simthinkd
|
|
4
|
+
s = simthinkd.fit_score(examples, goal="Rate the risk of this part.", out="risk", low=0, high=100)
|
|
5
|
+
s.score("part: defect dent | severity severe | image clear | belt normal | rework queue long")
|
|
6
|
+
# Score(value=87.3, ms=1.1)
|
|
7
|
+
|
|
8
|
+
`examples` is a list of (situation sentence, number) pairs. Training needs PyTorch; scoring needs only NumPy.
|
|
9
|
+
Targets are standardised, the network is trained with a squared-error loss, and the checkpoint with the lowest
|
|
10
|
+
validation mean absolute error is kept. Splits are made by a hash of each example, as in `fit`.
|
|
11
|
+
"""
|
|
12
|
+
import copy
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import time
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
|
|
23
|
+
from .core import build_request
|
|
24
|
+
from .policy import Policy, encode
|
|
25
|
+
from .train import _encode_rows, _ranker, _torch, sha, split_for
|
|
26
|
+
|
|
27
|
+
ACTION = {'SCORE': 'Return one number for this situation.'}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class Score:
|
|
32
|
+
value: float
|
|
33
|
+
ms: float = 0.0
|
|
34
|
+
|
|
35
|
+
def __str__(self):
|
|
36
|
+
return f'{self.value:.2f} ({self.ms:.2f} ms)'
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Scorer:
|
|
40
|
+
"""A trained score model. `Scorer("path/to/folder")`."""
|
|
41
|
+
|
|
42
|
+
def __init__(self, path):
|
|
43
|
+
path = Path(path)
|
|
44
|
+
self.card = json.loads((path / 'decider.json').read_text(encoding='utf8'))
|
|
45
|
+
if self.card.get('kind') != 'score':
|
|
46
|
+
raise ValueError(f'{path} is not a score model (use Decider for action models)')
|
|
47
|
+
self.policy = Policy(path / self.card['weights'])
|
|
48
|
+
self.sha256 = self.policy.digest
|
|
49
|
+
|
|
50
|
+
def score(self, state):
|
|
51
|
+
body = build_request(state, ACTION, self.card.get('goal', ''))
|
|
52
|
+
start = time.perf_counter()
|
|
53
|
+
x, _, _, _ = encode(body)
|
|
54
|
+
raw = float(self.policy.scores(x)[0])
|
|
55
|
+
value = raw * self.card['std'] + self.card['mean']
|
|
56
|
+
low, high = self.card.get('low'), self.card.get('high')
|
|
57
|
+
if low is not None:
|
|
58
|
+
value = max(low, value)
|
|
59
|
+
if high is not None:
|
|
60
|
+
value = min(high, value)
|
|
61
|
+
return Score(value, (time.perf_counter() - start) * 1000)
|
|
62
|
+
|
|
63
|
+
def __repr__(self):
|
|
64
|
+
return f'Scorer({self.card.get("name")!r}, sha256={self.sha256[:12]})'
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _mae(torch, model, part, mean, std):
|
|
68
|
+
x, _, m, y = part
|
|
69
|
+
with torch.inference_mode():
|
|
70
|
+
pred = torch.cat([model(x[i:i + 256], m[i:i + 256])[:, 0] for i in range(0, len(x), 256)])
|
|
71
|
+
return float((pred * std + mean - y).abs().mean())
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def fit_score(examples, goal='', out='my_scorer', steps=1500, seed=31, low=None, high=None, name=None, quiet=False):
|
|
75
|
+
"""Train a score model from (situation sentence, number) pairs. Returns a ready Scorer."""
|
|
76
|
+
torch = _torch()
|
|
77
|
+
out = Path(out)
|
|
78
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
log = (lambda record: None) if quiet else (lambda record: print(json.dumps(record), flush=True))
|
|
80
|
+
torch.set_num_threads(min(4, os.cpu_count() or 1))
|
|
81
|
+
torch.use_deterministic_algorithms(True)
|
|
82
|
+
device = 'cpu'
|
|
83
|
+
splits = {'train': [], 'validation': [], 'calibration': []}
|
|
84
|
+
for i, item in enumerate(examples):
|
|
85
|
+
state, value = (item['state'], item['value']) if isinstance(item, dict) else item
|
|
86
|
+
row_id = f'sc{i:07d}:' + hashlib.sha256(state.encode()).hexdigest()[:8]
|
|
87
|
+
splits[split_for(row_id)].append({'id': row_id, 'request': build_request(state, ACTION, goal),
|
|
88
|
+
'expected': {'operation': 'SCORE'}, 'value': float(value)})
|
|
89
|
+
splits['validation'] += splits.pop('calibration')
|
|
90
|
+
if min(len(v) for v in splits.values()) == 0:
|
|
91
|
+
raise ValueError('need enough examples for train/validation splits (about 100 or more)')
|
|
92
|
+
values = np.array([r['value'] for r in splits['train']], np.float32)
|
|
93
|
+
mean, std = float(values.mean()), float(values.std() or 1.)
|
|
94
|
+
parts = {}
|
|
95
|
+
for name_, items in splits.items():
|
|
96
|
+
(x, g, m, _), width = _encode_rows(torch, items, device)
|
|
97
|
+
parts[name_] = (x, g, m, torch.tensor([r['value'] for r in items], dtype=torch.float32))
|
|
98
|
+
torch.manual_seed(seed)
|
|
99
|
+
model = _ranker(torch, width).to(device)
|
|
100
|
+
optimizer = torch.optim.AdamW(model.parameters(), lr=.002, weight_decay=.0001)
|
|
101
|
+
rng = torch.Generator().manual_seed(seed + 2121)
|
|
102
|
+
x, _, m, y = parts['train']
|
|
103
|
+
target = (y - mean) / std
|
|
104
|
+
best, weights, best_step, events = float('inf'), copy.deepcopy(model.state_dict()), 0, []
|
|
105
|
+
started = time.perf_counter()
|
|
106
|
+
for step in range(steps + 1):
|
|
107
|
+
if step % 100 == 0 or step == steps:
|
|
108
|
+
model.eval()
|
|
109
|
+
val = _mae(torch, model, parts['validation'], mean, std)
|
|
110
|
+
model.train()
|
|
111
|
+
events.append({'step': step, 'validation_mae': val})
|
|
112
|
+
if val < best:
|
|
113
|
+
best, weights, best_step = val, copy.deepcopy(model.state_dict()), step
|
|
114
|
+
log({'seed': seed, 'step': step, 'validation_mae': round(val, 6)})
|
|
115
|
+
if step == steps:
|
|
116
|
+
break
|
|
117
|
+
batch = torch.randint(len(x), (min(96, len(x)),), generator=rng)
|
|
118
|
+
pred = model(x[batch], m[batch])[:, 0]
|
|
119
|
+
loss = ((pred - target[batch]) ** 2).mean()
|
|
120
|
+
optimizer.zero_grad(set_to_none=True)
|
|
121
|
+
loss.backward()
|
|
122
|
+
optimizer.step()
|
|
123
|
+
model.load_state_dict(weights)
|
|
124
|
+
model.eval()
|
|
125
|
+
arrays = {k: v.detach().cpu().numpy() for k, v in model.state_dict().items()}
|
|
126
|
+
path = out / 'weights.npz'
|
|
127
|
+
np.savez_compressed(path, **arrays, temperature=np.array(1.0))
|
|
128
|
+
card = {'kind': 'score', 'name': name or out.name, 'goal': goal, 'weights': path.name, 'sha256': sha(path),
|
|
129
|
+
'mean': mean, 'std': std, 'low': low, 'high': high,
|
|
130
|
+
'parameters': sum(p.numel() for p in model.parameters()),
|
|
131
|
+
'examples': {k: len(v) for k, v in splits.items()},
|
|
132
|
+
'training': {'seed': seed, 'steps_run': steps, 'selected_step': best_step, 'validation_mae': best,
|
|
133
|
+
'seconds': time.perf_counter() - started, 'events': events},
|
|
134
|
+
'torch': torch.__version__, 'created_at': datetime.now(timezone.utc).isoformat(),
|
|
135
|
+
'initialization': 'random (no pretrained weights)'}
|
|
136
|
+
(out / 'decider.json').write_text(json.dumps(card, ensure_ascii=False, indent=1), encoding='utf8')
|
|
137
|
+
return Scorer(out)
|
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
Binds to 127.0.0.1 by default. `--delay-ms` adds a fixed wait after inference (latency-injection experiments).
|
|
6
6
|
"""
|
|
7
7
|
import json
|
|
8
|
+
import socket
|
|
8
9
|
import time
|
|
9
10
|
from datetime import datetime, timezone
|
|
10
11
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
|
@@ -52,9 +53,23 @@ def make_handler(decider, name, delay_ms):
|
|
|
52
53
|
return Handler
|
|
53
54
|
|
|
54
55
|
|
|
56
|
+
class NoDelayHTTPServer(ThreadingHTTPServer):
|
|
57
|
+
"""Turns off Nagle's algorithm on every connection.
|
|
58
|
+
|
|
59
|
+
The handler writes the headers and the body separately. On a kept-alive connection the second small write waits
|
|
60
|
+
for the client's delayed ACK, so every answer took about 40 ms (measured 2026-10-03: median 42 ms kept-alive vs
|
|
61
|
+
1.4 ms with TCP_NODELAY). Clients that reuse connections (Java HttpURLConnection does) missed 30 ms deadlines.
|
|
62
|
+
"""
|
|
63
|
+
|
|
64
|
+
def get_request(self):
|
|
65
|
+
sock, addr = super().get_request()
|
|
66
|
+
sock.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)
|
|
67
|
+
return sock, addr
|
|
68
|
+
|
|
69
|
+
|
|
55
70
|
def serve(decider='doom-defend', host='127.0.0.1', port=11890, delay_ms=0.0, name=None):
|
|
56
71
|
decider = decider if isinstance(decider, Decider) else Decider(decider)
|
|
57
72
|
name = name or f'simthink-d:{decider.preset["name"]}'
|
|
58
|
-
server =
|
|
73
|
+
server = NoDelayHTTPServer((host, port), make_handler(decider, name, delay_ms))
|
|
59
74
|
print(json.dumps({'listening': f'{host}:{port}', 'weights_sha256': decider.sha256, 'model': name}), flush=True)
|
|
60
75
|
server.serve_forever()
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: simthinkd
|
|
3
|
-
Version: 0.1
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
|
|
5
5
|
Author: Myeongseongsimjae AX Institute
|
|
6
6
|
License: Apache-2.0
|
|
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
|
|
|
28
28
|
Requires-Dist: onnxruntime>=1.17; extra == "onnx"
|
|
29
29
|
Dynamic: license-file
|
|
30
30
|
|
|
31
|
-
<p align="center"><
|
|
31
|
+
<p align="center"><img src="assets/banner_v2.png" alt="SimThink D: a local backup for cloud decisions" width="100%"></p>
|
|
32
32
|
|
|
33
33
|
<p align="center">
|
|
34
|
+
<a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
|
|
34
35
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
|
|
35
36
|
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
36
37
|
<a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
|
|
@@ -38,21 +39,23 @@ Dynamic: license-file
|
|
|
38
39
|
<a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
|
|
39
40
|
</p>
|
|
40
41
|
|
|
41
|
-
|
|
42
|
+
<p align="center"><b>Network down. Decisions stay local.</b></p>
|
|
42
43
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
</p>
|
|
44
|
+
SimThink D is a small CPU model for offline backup decisions.
|
|
45
|
+
It has 265,665 parameters and takes about 2 ms per decision on one CPU core.
|
|
46
46
|
|
|
47
|
-
|
|
47
|
+
In the factory simulation, the internet was cut for 15 seconds.
|
|
48
|
+
The local backup got 59 of 63 parts right, with no late decisions.
|
|
48
49
|
|
|
49
|
-
<
|
|
50
|
+
<p align="center"><a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Watch the factory simulation: local backup during a network outage" width="100%"></a></p>
|
|
50
51
|
|
|
51
|
-
|
|
52
|
+
<p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/#demo">Try in your browser</a> · <a href="assets/factory_fallback.mp4">Watch the video (72 s)</a> · <a href="examples/factory_twin/">Factory code</a> · <a href="docs/PAPER.md">Paper</a></p>
|
|
52
53
|
|
|
53
|
-
|
|
54
|
+
```bash
|
|
55
|
+
pip install simthinkd
|
|
56
|
+
```
|
|
54
57
|
|
|
55
|
-
|
|
58
|
+
What makes the next decision when your network goes down?
|
|
56
59
|
|
|
57
60
|
## Words used here
|
|
58
61
|
|
|
@@ -64,13 +67,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
|
|
|
64
67
|
## Install
|
|
65
68
|
|
|
66
69
|
```bash
|
|
67
|
-
pip install
|
|
70
|
+
pip install simthinkd
|
|
68
71
|
```
|
|
69
72
|
|
|
70
73
|
That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
|
|
71
74
|
|
|
72
75
|
```bash
|
|
73
|
-
pip install "simthinkd[train]
|
|
76
|
+
pip install "simthinkd[train]"
|
|
74
77
|
```
|
|
75
78
|
|
|
76
79
|
## Quickstart
|
|
@@ -102,6 +105,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
|
|
|
102
105
|
|
|
103
106
|
On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
|
|
104
107
|
|
|
108
|
+
## Score instead of choose
|
|
109
|
+
|
|
110
|
+
Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
import random
|
|
114
|
+
import simthinkd
|
|
115
|
+
from simthinkd import toy
|
|
116
|
+
|
|
117
|
+
SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
|
|
118
|
+
|
|
119
|
+
def risk(s): # your own rule or records give the number
|
|
120
|
+
value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
|
|
121
|
+
value += 10 if s["image"] == "blurry" else 0
|
|
122
|
+
value += 8 if s["belt"] == "fast" else 0
|
|
123
|
+
value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
|
|
124
|
+
return float(min(100, value))
|
|
125
|
+
|
|
126
|
+
rng = random.Random(7)
|
|
127
|
+
examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
|
|
128
|
+
s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
|
|
129
|
+
print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
|
|
130
|
+
# 87.99 (0.22 ms)
|
|
131
|
+
```
|
|
132
|
+
|
|
133
|
+
The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
|
|
134
|
+
|
|
135
|
+
## Where it fits
|
|
136
|
+
|
|
137
|
+
SimThink D is a good fit when all three are true:
|
|
138
|
+
|
|
139
|
+
1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
|
|
140
|
+
2. **The answer is small.** One of up to 8 actions, or one number.
|
|
141
|
+
3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
|
|
142
|
+
|
|
143
|
+
| Use | Why it fits | Try it here |
|
|
144
|
+
|---|---|---|
|
|
145
|
+
| Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
|
|
146
|
+
| Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
|
|
147
|
+
| Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
|
|
148
|
+
| A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
|
|
149
|
+
|
|
150
|
+
### Use it together with a large model
|
|
151
|
+
|
|
152
|
+
Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
|
|
153
|
+
|
|
154
|
+
| Pattern | How it works | Try it here |
|
|
155
|
+
|---|---|---|
|
|
156
|
+
| **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
|
|
157
|
+
| **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
|
|
158
|
+
|
|
159
|
+
A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
|
|
160
|
+
|
|
161
|
+
It is not a good fit for:
|
|
162
|
+
|
|
163
|
+
- **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
|
|
164
|
+
- **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
|
|
165
|
+
- **Open-ended answers.** It picks from a list or returns one number. It does not write.
|
|
166
|
+
|
|
105
167
|
## Measure your own model
|
|
106
168
|
|
|
107
169
|
Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
|
|
@@ -115,6 +177,12 @@ The second line measures any server that accepts the [decision request](docs/PRO
|
|
|
115
177
|
|
|
116
178
|
## Same CPU, same states
|
|
117
179
|
|
|
180
|
+
<p align="center">
|
|
181
|
+
<img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
|
|
182
|
+
</p>
|
|
183
|
+
|
|
184
|
+
<p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
|
|
185
|
+
|
|
118
186
|
**Read this first.** Laya is a general model and was not trained on this game. Its published speed, about 33 ms per question, is on a GPU. We only had a CPU. So this table compares time inside a real-time loop. It does not compare overall quality.
|
|
119
187
|
|
|
120
188
|
We used one workstation CPU (6 threads) and 1,050 Doom states, then 10 live games on the same seeds. "Missed ticks" are ticks that passed before the decider answered.
|
|
@@ -133,7 +201,9 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
133
201
|
| Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
|
|
134
202
|
| Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
|
|
135
203
|
| Browser | [web/](web/): the same model in plain JavaScript, no server |
|
|
136
|
-
|
|
|
204
|
+
| A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
|
|
205
|
+
| Gradio | [space/](space/): a small web demo you can run locally or on Hugging Face Spaces |
|
|
206
|
+
| MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
|
|
137
207
|
| LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
|
|
138
208
|
|
|
139
209
|
## How it works
|
|
@@ -149,6 +219,14 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
149
219
|
- Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
|
|
150
220
|
- A decider copies its teacher. It is only as good as the teacher's rules.
|
|
151
221
|
- Probabilities are calibrated for the decider's own task only.
|
|
222
|
+
- A score model returns one number. It gives no probability or error bar with it.
|
|
223
|
+
|
|
224
|
+
## More
|
|
225
|
+
|
|
226
|
+
- [Figures from the paper](docs/FIGURES.md)
|
|
227
|
+
- [What you can reproduce](docs/REPRODUCE.md)
|
|
228
|
+
- [Decision request format](docs/PROTOCOL.md)
|
|
229
|
+
- [The factory video on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/)
|
|
152
230
|
|
|
153
231
|
## Citation
|
|
154
232
|
|
|
@@ -157,6 +235,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
|
|
|
157
235
|
- Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
|
|
158
236
|
- Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
|
|
159
237
|
|
|
238
|
+
The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
|
|
239
|
+
|
|
160
240
|
## Contributing
|
|
161
241
|
|
|
162
242
|
Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
@@ -7,6 +7,7 @@ src/simthinkd/bench.py
|
|
|
7
7
|
src/simthinkd/cli.py
|
|
8
8
|
src/simthinkd/core.py
|
|
9
9
|
src/simthinkd/policy.py
|
|
10
|
+
src/simthinkd/score.py
|
|
10
11
|
src/simthinkd/server.py
|
|
11
12
|
src/simthinkd/toy.py
|
|
12
13
|
src/simthinkd/train.py
|
|
@@ -23,8 +24,11 @@ src/simthinkd/integrations/mcp_server.py
|
|
|
23
24
|
src/simthinkd/weights/SHA256SUMS
|
|
24
25
|
src/simthinkd/weights/doom-corridor.npz
|
|
25
26
|
src/simthinkd/weights/doom-defend.npz
|
|
27
|
+
tests/test_factory_twin.py
|
|
26
28
|
tests/test_notebook.py
|
|
27
29
|
tests/test_package.py
|
|
30
|
+
tests/test_score.py
|
|
31
|
+
tests/test_server_latency.py
|
|
28
32
|
tests/test_space.py
|
|
29
33
|
tests/test_web_page.py
|
|
30
34
|
tests/test_web_parity.py
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Factory twin end-to-end test: train a decider, serve it, run the twin against it, check the results.
|
|
2
|
+
|
|
3
|
+
python -X utf8 tests/test_factory_twin.py (needs simthinkd[train])
|
|
4
|
+
|
|
5
|
+
Exit 0 = all pass. This runs the same three commands as examples/factory_twin/README.md.
|
|
6
|
+
"""
|
|
7
|
+
import json
|
|
8
|
+
import socket
|
|
9
|
+
import subprocess
|
|
10
|
+
import sys
|
|
11
|
+
import tempfile
|
|
12
|
+
import time
|
|
13
|
+
import urllib.request
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
TWIN = Path(__file__).resolve().parents[1] / "examples" / "factory_twin"
|
|
17
|
+
PY = sys.executable
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def free_port():
|
|
21
|
+
with socket.socket() as s:
|
|
22
|
+
s.bind(("127.0.0.1", 0))
|
|
23
|
+
return s.getsockname()[1]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def main():
|
|
27
|
+
results = []
|
|
28
|
+
with tempfile.TemporaryDirectory() as t:
|
|
29
|
+
t = Path(t)
|
|
30
|
+
r = subprocess.run([PY, "-X", "utf8", str(TWIN / "train_inspection_decider.py"), "--out", str(t / "decider")],
|
|
31
|
+
capture_output=True, text=True, cwd=TWIN)
|
|
32
|
+
results.append(("train", r.returncode == 0 and (t / "decider" / "weights.npz").exists(), r.stderr[-300:]))
|
|
33
|
+
port = free_port()
|
|
34
|
+
server = subprocess.Popen([PY, "-m", "simthinkd.cli", "serve", str(t / "decider"), "--port", str(port)],
|
|
35
|
+
stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True)
|
|
36
|
+
try:
|
|
37
|
+
health = None
|
|
38
|
+
for _ in range(60):
|
|
39
|
+
try:
|
|
40
|
+
with urllib.request.urlopen(f"http://127.0.0.1:{port}/health", timeout=1) as h:
|
|
41
|
+
health = json.loads(h.read())
|
|
42
|
+
break
|
|
43
|
+
except OSError:
|
|
44
|
+
time.sleep(0.5)
|
|
45
|
+
results.append(("serve", bool(health and health.get("weights_sha256")), str(health)))
|
|
46
|
+
out = t / "result.json"
|
|
47
|
+
r = subprocess.run([PY, "-X", "utf8", str(TWIN / "run.py"), "--arm", "D", "--parts", "50", "--timing", "live",
|
|
48
|
+
"--url", f"http://127.0.0.1:{port}", "--out", str(out)], capture_output=True, text=True, cwd=TWIN)
|
|
49
|
+
ok = r.returncode == 0 and out.exists()
|
|
50
|
+
results.append(("run arm D", ok, (r.stdout + r.stderr)[-400:]))
|
|
51
|
+
if ok:
|
|
52
|
+
m = json.loads(out.read_text(encoding="utf-8"))["metrics"]
|
|
53
|
+
results.append(("50 parts done", m["parts"] == 50, m))
|
|
54
|
+
results.append(("decisions on time (late < 10%)", m["late_rate"] < 0.10, m["late_rate"]))
|
|
55
|
+
results.append(("decider mostly right (>= 70%)", m["correct_rate"] >= 0.70, m["correct_rate"]))
|
|
56
|
+
finally:
|
|
57
|
+
server.terminate()
|
|
58
|
+
server.wait(timeout=10)
|
|
59
|
+
for name, ok, info in results:
|
|
60
|
+
print(("PASS " if ok else "FAIL ") + name + ("" if ok else f" | {info}"))
|
|
61
|
+
return 0 if results and all(ok for _, ok, _ in results) else 1
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
if __name__ == "__main__":
|
|
65
|
+
sys.exit(main())
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Score models end to end: train a risk score (0-100) from the toy inspection task on a CPU, then check it.
|
|
2
|
+
|
|
3
|
+
python -X utf8 tests/test_score.py (needs: pip install "simthinkd[train]")
|
|
4
|
+
The teacher is a fixed formula over the toy part, so the right answer is known for every state. Checks: training finishes,
|
|
5
|
+
held-out mean absolute error is small and far better than predicting the training mean, the ranking of parts is
|
|
6
|
+
preserved, values stay inside [low, high], the saved model reloads with the same hash, scoring needs no PyTorch path.
|
|
7
|
+
"""
|
|
8
|
+
import random
|
|
9
|
+
import sys
|
|
10
|
+
import tempfile
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
ROOT = Path(__file__).resolve().parents[1]
|
|
14
|
+
sys.path.insert(0, str(ROOT / 'src'))
|
|
15
|
+
|
|
16
|
+
import simthinkd # noqa: E402
|
|
17
|
+
from simthinkd import toy # noqa: E402
|
|
18
|
+
|
|
19
|
+
SEVERITY = {'minor': 20, 'moderate': 45, 'severe': 75}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def risk(s):
|
|
23
|
+
base = 0 if s['defect'] == 'none' else SEVERITY[s['severity']]
|
|
24
|
+
base += 10 if s['image'] == 'blurry' else 0
|
|
25
|
+
base += 8 if s['belt'] == 'fast' else 0
|
|
26
|
+
base += 5 if s['defect'] != 'none' and s['queue'] == 'long' else 0
|
|
27
|
+
return float(min(100, base))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def data(n, seed):
|
|
31
|
+
rng = random.Random(seed)
|
|
32
|
+
out = []
|
|
33
|
+
for _ in range(n):
|
|
34
|
+
s, text = toy.observe(rng)
|
|
35
|
+
out.append((text, risk(s)))
|
|
36
|
+
return out
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def main():
|
|
40
|
+
train, test = data(3000, 7), data(400, 99)
|
|
41
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
42
|
+
s = simthinkd.fit_score(train, goal='Rate how risky this part is, 0 to 100.', out=str(Path(tmp) / 'risk'),
|
|
43
|
+
low=0, high=100, quiet=True)
|
|
44
|
+
preds = [s.score(t).value for t, _ in test]
|
|
45
|
+
gold = [v for _, v in test]
|
|
46
|
+
mae = sum(abs(p - g) for p, g in zip(preds, gold)) / len(gold)
|
|
47
|
+
mean = sum(v for _, v in train) / len(train)
|
|
48
|
+
base = sum(abs(mean - g) for g in gold) / len(gold)
|
|
49
|
+
pairs = [(i, j) for i in range(0, len(test), 7) for j in range(3, len(test), 11) if gold[i] != gold[j]]
|
|
50
|
+
concord = sum((preds[i] - preds[j]) * (gold[i] - gold[j]) > 0 for i, j in pairs) / len(pairs)
|
|
51
|
+
again = simthinkd.Scorer(str(Path(tmp) / 'risk'))
|
|
52
|
+
checks = [
|
|
53
|
+
(f'held-out MAE {mae:.2f} below 3 points', mae < 3),
|
|
54
|
+
(f'much better than predicting the mean (MAE {base:.2f})', mae < base / 4),
|
|
55
|
+
(f'ranking preserved ({concord:.0%} of pairs in the right order)', concord > 0.9),
|
|
56
|
+
('values stay in [0, 100]', all(0 <= p <= 100 for p in preds)),
|
|
57
|
+
('reloaded model has the same hash', again.sha256 == s.sha256),
|
|
58
|
+
('same value after reload', abs(again.score(test[0][0]).value - preds[0]) < 1e-6),
|
|
59
|
+
(f'scoring time {s.score(test[1][0]).ms:.2f} ms under 10 ms', s.score(test[1][0]).ms < 10),
|
|
60
|
+
]
|
|
61
|
+
for name, ok in checks:
|
|
62
|
+
print(('PASS ' if ok else 'FAIL ') + name)
|
|
63
|
+
sys.exit(0 if all(ok for _, ok in checks) else 1)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
if __name__ == '__main__':
|
|
67
|
+
main()
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""The HTTP server must answer fast on a kept-alive connection (regression: ~40 ms Nagle/delayed-ACK stall in 0.2.0)."""
|
|
2
|
+
import http.client
|
|
3
|
+
import statistics
|
|
4
|
+
import threading
|
|
5
|
+
import time
|
|
6
|
+
|
|
7
|
+
from simthinkd.core import Decider
|
|
8
|
+
from simthinkd.server import NoDelayHTTPServer, make_handler
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def test_keep_alive_answers_fast():
|
|
12
|
+
decider = Decider('doom-defend')
|
|
13
|
+
server = NoDelayHTTPServer(('127.0.0.1', 0), make_handler(decider, 'test', 0.0))
|
|
14
|
+
threading.Thread(target=server.serve_forever, daemon=True).start()
|
|
15
|
+
try:
|
|
16
|
+
conn = http.client.HTTPConnection('127.0.0.1', server.server_address[1], timeout=2)
|
|
17
|
+
times = []
|
|
18
|
+
for _ in range(30):
|
|
19
|
+
start = time.perf_counter()
|
|
20
|
+
conn.request('GET', '/health')
|
|
21
|
+
conn.getresponse().read()
|
|
22
|
+
times.append((time.perf_counter() - start) * 1000)
|
|
23
|
+
assert statistics.median(times) < 20, times
|
|
24
|
+
finally:
|
|
25
|
+
server.shutdown()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
if __name__ == "__main__":
|
|
29
|
+
test_keep_alive_answers_fast()
|
|
30
|
+
print("server latency test passed")
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""Browser demo end to end: serve web/, open index.html in headless Chromium, make a decision, read the time.
|
|
1
|
+
"""Browser demo end to end: serve web/, open demo/index.html in headless Chromium, make a decision, read the time.
|
|
2
2
|
|
|
3
3
|
python -X utf8 tests/test_web_page.py (needs: pip install playwright; playwright install chromium)
|
|
4
4
|
Checks: page loads without console errors, the example decision is TURN_LEFT, a time and the one-tick verdict
|
|
@@ -20,7 +20,7 @@ def main():
|
|
|
20
20
|
handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(ROOT / 'web'))
|
|
21
21
|
server = http.server.ThreadingHTTPServer(('127.0.0.1', 0), handler)
|
|
22
22
|
threading.Thread(target=server.serve_forever, daemon=True).start()
|
|
23
|
-
url = f'http://127.0.0.1:{server.server_address[1]}/index.html'
|
|
23
|
+
url = f'http://127.0.0.1:{server.server_address[1]}/demo/index.html'
|
|
24
24
|
checks, errors = [], []
|
|
25
25
|
with sync_playwright() as p:
|
|
26
26
|
browser = p.chromium.launch()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|