simthinkd 0.1.0__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {simthinkd-0.1.0/src/simthinkd.egg-info → simthinkd-0.2.0}/PKG-INFO +77 -7
- {simthinkd-0.1.0 → simthinkd-0.2.0}/README.md +76 -6
- {simthinkd-0.1.0 → simthinkd-0.2.0}/pyproject.toml +1 -1
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/__init__.py +9 -2
- simthinkd-0.2.0/src/simthinkd/score.py +137 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0/src/simthinkd.egg-info}/PKG-INFO +77 -7
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/SOURCES.txt +3 -0
- simthinkd-0.2.0/tests/test_factory_twin.py +65 -0
- simthinkd-0.2.0/tests/test_score.py +67 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_web_page.py +2 -2
- {simthinkd-0.1.0 → simthinkd-0.2.0}/LICENSE +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/NOTICE +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/setup.cfg +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/bench.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/cli.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/core.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/data/doom_defend_states.json +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/integrations/__init__.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/integrations/langchain_tool.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/integrations/mcp_server.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/policy.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/server.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/toy.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/train.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/weights/SHA256SUMS +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/weights/doom-corridor.npz +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/weights/doom-defend.npz +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/dependency_links.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/entry_points.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/requires.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/top_level.txt +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_notebook.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_package.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_space.py +0 -0
- {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_web_parity.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: simthinkd
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
|
|
5
5
|
Author: Myeongseongsimjae AX Institute
|
|
6
6
|
License: Apache-2.0
|
|
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
|
|
|
28
28
|
Requires-Dist: onnxruntime>=1.17; extra == "onnx"
|
|
29
29
|
Dynamic: license-file
|
|
30
30
|
|
|
31
|
-
<p align="center"><
|
|
31
|
+
<p align="center"><img src="assets/banner.png" alt="SimThink D: a tiny decision model that runs on one CPU core" width="100%"></p>
|
|
32
32
|
|
|
33
33
|
<p align="center">
|
|
34
|
+
<a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
|
|
34
35
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
|
|
35
36
|
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
36
37
|
<a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
|
|
@@ -38,6 +39,8 @@ Dynamic: license-file
|
|
|
38
39
|
<a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
|
|
39
40
|
</p>
|
|
40
41
|
|
|
42
|
+
<p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/"><b>Project page, live demo and video →</b></a></p>
|
|
43
|
+
|
|
41
44
|
**A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
|
|
42
45
|
|
|
43
46
|
<p align="center">
|
|
@@ -48,11 +51,15 @@ Dynamic: license-file
|
|
|
48
51
|
|
|
49
52
|
<div align="center">
|
|
50
53
|
|
|
51
|
-
[Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](
|
|
54
|
+
[Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](https://mssj-ai-org.github.io/simthinkd/#demo) · [Gradio demo](space/) · [Factory twin](examples/factory_twin/) · [Paper](docs/PAPER.md) · [Reproduce the paper](docs/REPRODUCE.md) · [Figures](docs/FIGURES.md) · [Protocol](docs/PROTOCOL.md)
|
|
52
55
|
|
|
53
56
|
</div>
|
|
54
57
|
|
|
55
|
-
Video:
|
|
58
|
+
### Video: the internet goes down, the line keeps going
|
|
59
|
+
|
|
60
|
+
<a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Internet down. The line kept going." width="100%"></a>
|
|
61
|
+
|
|
62
|
+
A cloud decision service runs a simulated inspection line, with SimThink D on the factory PC as its backup. When an answer does not come back in time, or the network is cut, SimThink D makes the decision. [Watch the video](assets/factory_fallback.mp4) (72 s) or see it [on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/). The simulator is in [examples/factory_twin](examples/factory_twin/).
|
|
56
63
|
|
|
57
64
|
## Words used here
|
|
58
65
|
|
|
@@ -64,13 +71,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
|
|
|
64
71
|
## Install
|
|
65
72
|
|
|
66
73
|
```bash
|
|
67
|
-
pip install
|
|
74
|
+
pip install simthinkd
|
|
68
75
|
```
|
|
69
76
|
|
|
70
77
|
That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
|
|
71
78
|
|
|
72
79
|
```bash
|
|
73
|
-
pip install "simthinkd[train]
|
|
80
|
+
pip install "simthinkd[train]"
|
|
74
81
|
```
|
|
75
82
|
|
|
76
83
|
## Quickstart
|
|
@@ -102,6 +109,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
|
|
|
102
109
|
|
|
103
110
|
On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
|
|
104
111
|
|
|
112
|
+
## Score instead of choose
|
|
113
|
+
|
|
114
|
+
Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
import random
|
|
118
|
+
import simthinkd
|
|
119
|
+
from simthinkd import toy
|
|
120
|
+
|
|
121
|
+
SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
|
|
122
|
+
|
|
123
|
+
def risk(s): # your own rule or records give the number
|
|
124
|
+
value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
|
|
125
|
+
value += 10 if s["image"] == "blurry" else 0
|
|
126
|
+
value += 8 if s["belt"] == "fast" else 0
|
|
127
|
+
value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
|
|
128
|
+
return float(min(100, value))
|
|
129
|
+
|
|
130
|
+
rng = random.Random(7)
|
|
131
|
+
examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
|
|
132
|
+
s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
|
|
133
|
+
print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
|
|
134
|
+
# 87.99 (0.22 ms)
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
|
|
138
|
+
|
|
139
|
+
## Where it fits
|
|
140
|
+
|
|
141
|
+
SimThink D is a good fit when all three are true:
|
|
142
|
+
|
|
143
|
+
1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
|
|
144
|
+
2. **The answer is small.** One of up to 8 actions, or one number.
|
|
145
|
+
3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
|
|
146
|
+
|
|
147
|
+
| Use | Why it fits | Try it here |
|
|
148
|
+
|---|---|---|
|
|
149
|
+
| Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
|
|
150
|
+
| Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
|
|
151
|
+
| Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
|
|
152
|
+
| A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
|
|
153
|
+
|
|
154
|
+
### Use it together with a large model
|
|
155
|
+
|
|
156
|
+
Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
|
|
157
|
+
|
|
158
|
+
| Pattern | How it works | Try it here |
|
|
159
|
+
|---|---|---|
|
|
160
|
+
| **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
|
|
161
|
+
| **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
|
|
162
|
+
|
|
163
|
+
A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
|
|
164
|
+
|
|
165
|
+
It is not a good fit for:
|
|
166
|
+
|
|
167
|
+
- **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
|
|
168
|
+
- **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
|
|
169
|
+
- **Open-ended answers.** It picks from a list or returns one number. It does not write.
|
|
170
|
+
|
|
105
171
|
## Measure your own model
|
|
106
172
|
|
|
107
173
|
Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
|
|
@@ -133,7 +199,8 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
133
199
|
| Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
|
|
134
200
|
| Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
|
|
135
201
|
| Browser | [web/](web/): the same model in plain JavaScript, no server |
|
|
136
|
-
|
|
|
202
|
+
| A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
|
|
203
|
+
| MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
|
|
137
204
|
| LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
|
|
138
205
|
|
|
139
206
|
## How it works
|
|
@@ -149,6 +216,7 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
149
216
|
- Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
|
|
150
217
|
- A decider copies its teacher. It is only as good as the teacher's rules.
|
|
151
218
|
- Probabilities are calibrated for the decider's own task only.
|
|
219
|
+
- A score model returns one number. It gives no probability or error bar with it.
|
|
152
220
|
|
|
153
221
|
## Citation
|
|
154
222
|
|
|
@@ -157,6 +225,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
|
|
|
157
225
|
- Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
|
|
158
226
|
- Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
|
|
159
227
|
|
|
228
|
+
The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
|
|
229
|
+
|
|
160
230
|
## Contributing
|
|
161
231
|
|
|
162
232
|
Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
@@ -1,6 +1,7 @@
|
|
|
1
|
-
<p align="center"><
|
|
1
|
+
<p align="center"><img src="assets/banner.png" alt="SimThink D: a tiny decision model that runs on one CPU core" width="100%"></p>
|
|
2
2
|
|
|
3
3
|
<p align="center">
|
|
4
|
+
<a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
|
|
4
5
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
|
|
5
6
|
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
6
7
|
<a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
|
|
@@ -8,6 +9,8 @@
|
|
|
8
9
|
<a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
|
|
9
10
|
</p>
|
|
10
11
|
|
|
12
|
+
<p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/"><b>Project page, live demo and video →</b></a></p>
|
|
13
|
+
|
|
11
14
|
**A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
|
|
12
15
|
|
|
13
16
|
<p align="center">
|
|
@@ -18,11 +21,15 @@
|
|
|
18
21
|
|
|
19
22
|
<div align="center">
|
|
20
23
|
|
|
21
|
-
[Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](
|
|
24
|
+
[Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](https://mssj-ai-org.github.io/simthinkd/#demo) · [Gradio demo](space/) · [Factory twin](examples/factory_twin/) · [Paper](docs/PAPER.md) · [Reproduce the paper](docs/REPRODUCE.md) · [Figures](docs/FIGURES.md) · [Protocol](docs/PROTOCOL.md)
|
|
22
25
|
|
|
23
26
|
</div>
|
|
24
27
|
|
|
25
|
-
Video:
|
|
28
|
+
### Video: the internet goes down, the line keeps going
|
|
29
|
+
|
|
30
|
+
<a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Internet down. The line kept going." width="100%"></a>
|
|
31
|
+
|
|
32
|
+
A cloud decision service runs a simulated inspection line, with SimThink D on the factory PC as its backup. When an answer does not come back in time, or the network is cut, SimThink D makes the decision. [Watch the video](assets/factory_fallback.mp4) (72 s) or see it [on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/). The simulator is in [examples/factory_twin](examples/factory_twin/).
|
|
26
33
|
|
|
27
34
|
## Words used here
|
|
28
35
|
|
|
@@ -34,13 +41,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
|
|
|
34
41
|
## Install
|
|
35
42
|
|
|
36
43
|
```bash
|
|
37
|
-
pip install
|
|
44
|
+
pip install simthinkd
|
|
38
45
|
```
|
|
39
46
|
|
|
40
47
|
That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
|
|
41
48
|
|
|
42
49
|
```bash
|
|
43
|
-
pip install "simthinkd[train]
|
|
50
|
+
pip install "simthinkd[train]"
|
|
44
51
|
```
|
|
45
52
|
|
|
46
53
|
## Quickstart
|
|
@@ -72,6 +79,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
|
|
|
72
79
|
|
|
73
80
|
On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
|
|
74
81
|
|
|
82
|
+
## Score instead of choose
|
|
83
|
+
|
|
84
|
+
Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
|
|
85
|
+
|
|
86
|
+
```python
|
|
87
|
+
import random
|
|
88
|
+
import simthinkd
|
|
89
|
+
from simthinkd import toy
|
|
90
|
+
|
|
91
|
+
SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
|
|
92
|
+
|
|
93
|
+
def risk(s): # your own rule or records give the number
|
|
94
|
+
value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
|
|
95
|
+
value += 10 if s["image"] == "blurry" else 0
|
|
96
|
+
value += 8 if s["belt"] == "fast" else 0
|
|
97
|
+
value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
|
|
98
|
+
return float(min(100, value))
|
|
99
|
+
|
|
100
|
+
rng = random.Random(7)
|
|
101
|
+
examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
|
|
102
|
+
s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
|
|
103
|
+
print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
|
|
104
|
+
# 87.99 (0.22 ms)
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
|
|
108
|
+
|
|
109
|
+
## Where it fits
|
|
110
|
+
|
|
111
|
+
SimThink D is a good fit when all three are true:
|
|
112
|
+
|
|
113
|
+
1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
|
|
114
|
+
2. **The answer is small.** One of up to 8 actions, or one number.
|
|
115
|
+
3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
|
|
116
|
+
|
|
117
|
+
| Use | Why it fits | Try it here |
|
|
118
|
+
|---|---|---|
|
|
119
|
+
| Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
|
|
120
|
+
| Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
|
|
121
|
+
| Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
|
|
122
|
+
| A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
|
|
123
|
+
|
|
124
|
+
### Use it together with a large model
|
|
125
|
+
|
|
126
|
+
Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
|
|
127
|
+
|
|
128
|
+
| Pattern | How it works | Try it here |
|
|
129
|
+
|---|---|---|
|
|
130
|
+
| **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
|
|
131
|
+
| **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
|
|
132
|
+
|
|
133
|
+
A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
|
|
134
|
+
|
|
135
|
+
It is not a good fit for:
|
|
136
|
+
|
|
137
|
+
- **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
|
|
138
|
+
- **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
|
|
139
|
+
- **Open-ended answers.** It picks from a list or returns one number. It does not write.
|
|
140
|
+
|
|
75
141
|
## Measure your own model
|
|
76
142
|
|
|
77
143
|
Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
|
|
@@ -103,7 +169,8 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
103
169
|
| Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
|
|
104
170
|
| Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
|
|
105
171
|
| Browser | [web/](web/): the same model in plain JavaScript, no server |
|
|
106
|
-
|
|
|
172
|
+
| A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
|
|
173
|
+
| MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
|
|
107
174
|
| LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
|
|
108
175
|
|
|
109
176
|
## How it works
|
|
@@ -119,6 +186,7 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
119
186
|
- Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
|
|
120
187
|
- A decider copies its teacher. It is only as good as the teacher's rules.
|
|
121
188
|
- Probabilities are calibrated for the decider's own task only.
|
|
189
|
+
- A score model returns one number. It gives no probability or error bar with it.
|
|
122
190
|
|
|
123
191
|
## Citation
|
|
124
192
|
|
|
@@ -127,6 +195,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
|
|
|
127
195
|
- Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
|
|
128
196
|
- Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
|
|
129
197
|
|
|
198
|
+
The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
|
|
199
|
+
|
|
130
200
|
## Contributing
|
|
131
201
|
|
|
132
202
|
Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "simthinkd"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.2.0"
|
|
8
8
|
description = "A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
requires-python = ">=3.10"
|
|
@@ -4,12 +4,19 @@
|
|
|
4
4
|
print(Decider("doom-defend").decide("seen: Demon left a30 d5 | enemies 1 | sway left | gun ready | ammo25"))
|
|
5
5
|
"""
|
|
6
6
|
from .core import Decider, Decision, available, build_request
|
|
7
|
+
from .score import Score, Scorer
|
|
7
8
|
|
|
8
|
-
__version__ = '0.
|
|
9
|
-
__all__ = ['Decider', 'Decision', 'available', 'build_request', 'fit', '__version__']
|
|
9
|
+
__version__ = '0.2.0'
|
|
10
|
+
__all__ = ['Decider', 'Decision', 'Score', 'Scorer', 'available', 'build_request', 'fit', 'fit_score', '__version__']
|
|
10
11
|
|
|
11
12
|
|
|
12
13
|
def fit(*args, **kwargs):
|
|
13
14
|
"""Train a decider from (situation, action) pairs. See simthinkd.train.fit. Needs simthinkd[train]."""
|
|
14
15
|
from .train import fit as _fit
|
|
15
16
|
return _fit(*args, **kwargs)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def fit_score(*args, **kwargs):
|
|
20
|
+
"""Train a score model from (situation, number) pairs. See simthinkd.score.fit_score. Needs simthinkd[train]."""
|
|
21
|
+
from .score import fit_score as _fit_score
|
|
22
|
+
return _fit_score(*args, **kwargs)
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Score models: the same small network as a decider, but it returns one number instead of picking an action.
|
|
2
|
+
|
|
3
|
+
import simthinkd
|
|
4
|
+
s = simthinkd.fit_score(examples, goal="Rate the risk of this part.", out="risk", low=0, high=100)
|
|
5
|
+
s.score("part: defect dent | severity severe | image clear | belt normal | rework queue long")
|
|
6
|
+
# Score(value=87.3, ms=1.1)
|
|
7
|
+
|
|
8
|
+
`examples` is a list of (situation sentence, number) pairs. Training needs PyTorch; scoring needs only NumPy.
|
|
9
|
+
Targets are standardised, the network is trained with a squared-error loss, and the checkpoint with the lowest
|
|
10
|
+
validation mean absolute error is kept. Splits are made by a hash of each example, as in `fit`.
|
|
11
|
+
"""
|
|
12
|
+
import copy
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import time
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from datetime import datetime, timezone
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
|
|
23
|
+
from .core import build_request
|
|
24
|
+
from .policy import Policy, encode
|
|
25
|
+
from .train import _encode_rows, _ranker, _torch, sha, split_for
|
|
26
|
+
|
|
27
|
+
ACTION = {'SCORE': 'Return one number for this situation.'}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class Score:
|
|
32
|
+
value: float
|
|
33
|
+
ms: float = 0.0
|
|
34
|
+
|
|
35
|
+
def __str__(self):
|
|
36
|
+
return f'{self.value:.2f} ({self.ms:.2f} ms)'
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class Scorer:
|
|
40
|
+
"""A trained score model. `Scorer("path/to/folder")`."""
|
|
41
|
+
|
|
42
|
+
def __init__(self, path):
|
|
43
|
+
path = Path(path)
|
|
44
|
+
self.card = json.loads((path / 'decider.json').read_text(encoding='utf8'))
|
|
45
|
+
if self.card.get('kind') != 'score':
|
|
46
|
+
raise ValueError(f'{path} is not a score model (use Decider for action models)')
|
|
47
|
+
self.policy = Policy(path / self.card['weights'])
|
|
48
|
+
self.sha256 = self.policy.digest
|
|
49
|
+
|
|
50
|
+
def score(self, state):
|
|
51
|
+
body = build_request(state, ACTION, self.card.get('goal', ''))
|
|
52
|
+
start = time.perf_counter()
|
|
53
|
+
x, _, _, _ = encode(body)
|
|
54
|
+
raw = float(self.policy.scores(x)[0])
|
|
55
|
+
value = raw * self.card['std'] + self.card['mean']
|
|
56
|
+
low, high = self.card.get('low'), self.card.get('high')
|
|
57
|
+
if low is not None:
|
|
58
|
+
value = max(low, value)
|
|
59
|
+
if high is not None:
|
|
60
|
+
value = min(high, value)
|
|
61
|
+
return Score(value, (time.perf_counter() - start) * 1000)
|
|
62
|
+
|
|
63
|
+
def __repr__(self):
|
|
64
|
+
return f'Scorer({self.card.get("name")!r}, sha256={self.sha256[:12]})'
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _mae(torch, model, part, mean, std):
|
|
68
|
+
x, _, m, y = part
|
|
69
|
+
with torch.inference_mode():
|
|
70
|
+
pred = torch.cat([model(x[i:i + 256], m[i:i + 256])[:, 0] for i in range(0, len(x), 256)])
|
|
71
|
+
return float((pred * std + mean - y).abs().mean())
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def fit_score(examples, goal='', out='my_scorer', steps=1500, seed=31, low=None, high=None, name=None, quiet=False):
|
|
75
|
+
"""Train a score model from (situation sentence, number) pairs. Returns a ready Scorer."""
|
|
76
|
+
torch = _torch()
|
|
77
|
+
out = Path(out)
|
|
78
|
+
out.mkdir(parents=True, exist_ok=True)
|
|
79
|
+
log = (lambda record: None) if quiet else (lambda record: print(json.dumps(record), flush=True))
|
|
80
|
+
torch.set_num_threads(min(4, os.cpu_count() or 1))
|
|
81
|
+
torch.use_deterministic_algorithms(True)
|
|
82
|
+
device = 'cpu'
|
|
83
|
+
splits = {'train': [], 'validation': [], 'calibration': []}
|
|
84
|
+
for i, item in enumerate(examples):
|
|
85
|
+
state, value = (item['state'], item['value']) if isinstance(item, dict) else item
|
|
86
|
+
row_id = f'sc{i:07d}:' + hashlib.sha256(state.encode()).hexdigest()[:8]
|
|
87
|
+
splits[split_for(row_id)].append({'id': row_id, 'request': build_request(state, ACTION, goal),
|
|
88
|
+
'expected': {'operation': 'SCORE'}, 'value': float(value)})
|
|
89
|
+
splits['validation'] += splits.pop('calibration')
|
|
90
|
+
if min(len(v) for v in splits.values()) == 0:
|
|
91
|
+
raise ValueError('need enough examples for train/validation splits (about 100 or more)')
|
|
92
|
+
values = np.array([r['value'] for r in splits['train']], np.float32)
|
|
93
|
+
mean, std = float(values.mean()), float(values.std() or 1.)
|
|
94
|
+
parts = {}
|
|
95
|
+
for name_, items in splits.items():
|
|
96
|
+
(x, g, m, _), width = _encode_rows(torch, items, device)
|
|
97
|
+
parts[name_] = (x, g, m, torch.tensor([r['value'] for r in items], dtype=torch.float32))
|
|
98
|
+
torch.manual_seed(seed)
|
|
99
|
+
model = _ranker(torch, width).to(device)
|
|
100
|
+
optimizer = torch.optim.AdamW(model.parameters(), lr=.002, weight_decay=.0001)
|
|
101
|
+
rng = torch.Generator().manual_seed(seed + 2121)
|
|
102
|
+
x, _, m, y = parts['train']
|
|
103
|
+
target = (y - mean) / std
|
|
104
|
+
best, weights, best_step, events = float('inf'), copy.deepcopy(model.state_dict()), 0, []
|
|
105
|
+
started = time.perf_counter()
|
|
106
|
+
for step in range(steps + 1):
|
|
107
|
+
if step % 100 == 0 or step == steps:
|
|
108
|
+
model.eval()
|
|
109
|
+
val = _mae(torch, model, parts['validation'], mean, std)
|
|
110
|
+
model.train()
|
|
111
|
+
events.append({'step': step, 'validation_mae': val})
|
|
112
|
+
if val < best:
|
|
113
|
+
best, weights, best_step = val, copy.deepcopy(model.state_dict()), step
|
|
114
|
+
log({'seed': seed, 'step': step, 'validation_mae': round(val, 6)})
|
|
115
|
+
if step == steps:
|
|
116
|
+
break
|
|
117
|
+
batch = torch.randint(len(x), (min(96, len(x)),), generator=rng)
|
|
118
|
+
pred = model(x[batch], m[batch])[:, 0]
|
|
119
|
+
loss = ((pred - target[batch]) ** 2).mean()
|
|
120
|
+
optimizer.zero_grad(set_to_none=True)
|
|
121
|
+
loss.backward()
|
|
122
|
+
optimizer.step()
|
|
123
|
+
model.load_state_dict(weights)
|
|
124
|
+
model.eval()
|
|
125
|
+
arrays = {k: v.detach().cpu().numpy() for k, v in model.state_dict().items()}
|
|
126
|
+
path = out / 'weights.npz'
|
|
127
|
+
np.savez_compressed(path, **arrays, temperature=np.array(1.0))
|
|
128
|
+
card = {'kind': 'score', 'name': name or out.name, 'goal': goal, 'weights': path.name, 'sha256': sha(path),
|
|
129
|
+
'mean': mean, 'std': std, 'low': low, 'high': high,
|
|
130
|
+
'parameters': sum(p.numel() for p in model.parameters()),
|
|
131
|
+
'examples': {k: len(v) for k, v in splits.items()},
|
|
132
|
+
'training': {'seed': seed, 'steps_run': steps, 'selected_step': best_step, 'validation_mae': best,
|
|
133
|
+
'seconds': time.perf_counter() - started, 'events': events},
|
|
134
|
+
'torch': torch.__version__, 'created_at': datetime.now(timezone.utc).isoformat(),
|
|
135
|
+
'initialization': 'random (no pretrained weights)'}
|
|
136
|
+
(out / 'decider.json').write_text(json.dumps(card, ensure_ascii=False, indent=1), encoding='utf8')
|
|
137
|
+
return Scorer(out)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: simthinkd
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
|
|
5
5
|
Author: Myeongseongsimjae AX Institute
|
|
6
6
|
License: Apache-2.0
|
|
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
|
|
|
28
28
|
Requires-Dist: onnxruntime>=1.17; extra == "onnx"
|
|
29
29
|
Dynamic: license-file
|
|
30
30
|
|
|
31
|
-
<p align="center"><
|
|
31
|
+
<p align="center"><img src="assets/banner.png" alt="SimThink D: a tiny decision model that runs on one CPU core" width="100%"></p>
|
|
32
32
|
|
|
33
33
|
<p align="center">
|
|
34
|
+
<a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
|
|
34
35
|
<a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
|
|
35
36
|
<img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
|
|
36
37
|
<a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
|
|
@@ -38,6 +39,8 @@ Dynamic: license-file
|
|
|
38
39
|
<a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
|
|
39
40
|
</p>
|
|
40
41
|
|
|
42
|
+
<p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/"><b>Project page, live demo and video →</b></a></p>
|
|
43
|
+
|
|
41
44
|
**A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
|
|
42
45
|
|
|
43
46
|
<p align="center">
|
|
@@ -48,11 +51,15 @@ Dynamic: license-file
|
|
|
48
51
|
|
|
49
52
|
<div align="center">
|
|
50
53
|
|
|
51
|
-
[Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](
|
|
54
|
+
[Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](https://mssj-ai-org.github.io/simthinkd/#demo) · [Gradio demo](space/) · [Factory twin](examples/factory_twin/) · [Paper](docs/PAPER.md) · [Reproduce the paper](docs/REPRODUCE.md) · [Figures](docs/FIGURES.md) · [Protocol](docs/PROTOCOL.md)
|
|
52
55
|
|
|
53
56
|
</div>
|
|
54
57
|
|
|
55
|
-
Video:
|
|
58
|
+
### Video: the internet goes down, the line keeps going
|
|
59
|
+
|
|
60
|
+
<a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Internet down. The line kept going." width="100%"></a>
|
|
61
|
+
|
|
62
|
+
A cloud decision service runs a simulated inspection line, with SimThink D on the factory PC as its backup. When an answer does not come back in time, or the network is cut, SimThink D makes the decision. [Watch the video](assets/factory_fallback.mp4) (72 s) or see it [on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/). The simulator is in [examples/factory_twin](examples/factory_twin/).
|
|
56
63
|
|
|
57
64
|
## Words used here
|
|
58
65
|
|
|
@@ -64,13 +71,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
|
|
|
64
71
|
## Install
|
|
65
72
|
|
|
66
73
|
```bash
|
|
67
|
-
pip install
|
|
74
|
+
pip install simthinkd
|
|
68
75
|
```
|
|
69
76
|
|
|
70
77
|
That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
|
|
71
78
|
|
|
72
79
|
```bash
|
|
73
|
-
pip install "simthinkd[train]
|
|
80
|
+
pip install "simthinkd[train]"
|
|
74
81
|
```
|
|
75
82
|
|
|
76
83
|
## Quickstart
|
|
@@ -102,6 +109,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
|
|
|
102
109
|
|
|
103
110
|
On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
|
|
104
111
|
|
|
112
|
+
## Score instead of choose
|
|
113
|
+
|
|
114
|
+
Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
import random
|
|
118
|
+
import simthinkd
|
|
119
|
+
from simthinkd import toy
|
|
120
|
+
|
|
121
|
+
SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
|
|
122
|
+
|
|
123
|
+
def risk(s): # your own rule or records give the number
|
|
124
|
+
value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
|
|
125
|
+
value += 10 if s["image"] == "blurry" else 0
|
|
126
|
+
value += 8 if s["belt"] == "fast" else 0
|
|
127
|
+
value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
|
|
128
|
+
return float(min(100, value))
|
|
129
|
+
|
|
130
|
+
rng = random.Random(7)
|
|
131
|
+
examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
|
|
132
|
+
s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
|
|
133
|
+
print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
|
|
134
|
+
# 87.99 (0.22 ms)
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
|
|
138
|
+
|
|
139
|
+
## Where it fits
|
|
140
|
+
|
|
141
|
+
SimThink D is a good fit when all three are true:
|
|
142
|
+
|
|
143
|
+
1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
|
|
144
|
+
2. **The answer is small.** One of up to 8 actions, or one number.
|
|
145
|
+
3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
|
|
146
|
+
|
|
147
|
+
| Use | Why it fits | Try it here |
|
|
148
|
+
|---|---|---|
|
|
149
|
+
| Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
|
|
150
|
+
| Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
|
|
151
|
+
| Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
|
|
152
|
+
| A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
|
|
153
|
+
|
|
154
|
+
### Use it together with a large model
|
|
155
|
+
|
|
156
|
+
Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
|
|
157
|
+
|
|
158
|
+
| Pattern | How it works | Try it here |
|
|
159
|
+
|---|---|---|
|
|
160
|
+
| **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
|
|
161
|
+
| **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
|
|
162
|
+
|
|
163
|
+
A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
|
|
164
|
+
|
|
165
|
+
It is not a good fit for:
|
|
166
|
+
|
|
167
|
+
- **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
|
|
168
|
+
- **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
|
|
169
|
+
- **Open-ended answers.** It picks from a list or returns one number. It does not write.
|
|
170
|
+
|
|
105
171
|
## Measure your own model
|
|
106
172
|
|
|
107
173
|
Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
|
|
@@ -133,7 +199,8 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
133
199
|
| Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
|
|
134
200
|
| Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
|
|
135
201
|
| Browser | [web/](web/): the same model in plain JavaScript, no server |
|
|
136
|
-
|
|
|
202
|
+
| A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
|
|
203
|
+
| MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
|
|
137
204
|
| LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
|
|
138
205
|
|
|
139
206
|
## How it works
|
|
@@ -149,6 +216,7 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
|
|
|
149
216
|
- Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
|
|
150
217
|
- A decider copies its teacher. It is only as good as the teacher's rules.
|
|
151
218
|
- Probabilities are calibrated for the decider's own task only.
|
|
219
|
+
- A score model returns one number. It gives no probability or error bar with it.
|
|
152
220
|
|
|
153
221
|
## Citation
|
|
154
222
|
|
|
@@ -157,6 +225,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
|
|
|
157
225
|
- Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
|
|
158
226
|
- Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
|
|
159
227
|
|
|
228
|
+
The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
|
|
229
|
+
|
|
160
230
|
## Contributing
|
|
161
231
|
|
|
162
232
|
Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
|
|
@@ -7,6 +7,7 @@ src/simthinkd/bench.py
|
|
|
7
7
|
src/simthinkd/cli.py
|
|
8
8
|
src/simthinkd/core.py
|
|
9
9
|
src/simthinkd/policy.py
|
|
10
|
+
src/simthinkd/score.py
|
|
10
11
|
src/simthinkd/server.py
|
|
11
12
|
src/simthinkd/toy.py
|
|
12
13
|
src/simthinkd/train.py
|
|
@@ -23,8 +24,10 @@ src/simthinkd/integrations/mcp_server.py
|
|
|
23
24
|
src/simthinkd/weights/SHA256SUMS
|
|
24
25
|
src/simthinkd/weights/doom-corridor.npz
|
|
25
26
|
src/simthinkd/weights/doom-defend.npz
|
|
27
|
+
tests/test_factory_twin.py
|
|
26
28
|
tests/test_notebook.py
|
|
27
29
|
tests/test_package.py
|
|
30
|
+
tests/test_score.py
|
|
28
31
|
tests/test_space.py
|
|
29
32
|
tests/test_web_page.py
|
|
30
33
|
tests/test_web_parity.py
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Factory twin end-to-end test: train a decider, serve it, run the twin against it, check the results.
|
|
2
|
+
|
|
3
|
+
python -X utf8 tests/test_factory_twin.py (needs simthinkd[train])
|
|
4
|
+
|
|
5
|
+
Exit 0 = all pass. This runs the same three commands as examples/factory_twin/README.md.
|
|
6
|
+
"""
|
|
7
|
+
import json
|
|
8
|
+
import socket
|
|
9
|
+
import subprocess
|
|
10
|
+
import sys
|
|
11
|
+
import tempfile
|
|
12
|
+
import time
|
|
13
|
+
import urllib.request
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
|
|
16
|
+
TWIN = Path(__file__).resolve().parents[1] / "examples" / "factory_twin"
|
|
17
|
+
PY = sys.executable
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def free_port():
|
|
21
|
+
with socket.socket() as s:
|
|
22
|
+
s.bind(("127.0.0.1", 0))
|
|
23
|
+
return s.getsockname()[1]
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def main():
|
|
27
|
+
results = []
|
|
28
|
+
with tempfile.TemporaryDirectory() as t:
|
|
29
|
+
t = Path(t)
|
|
30
|
+
r = subprocess.run([PY, "-X", "utf8", str(TWIN / "train_inspection_decider.py"), "--out", str(t / "decider")],
|
|
31
|
+
capture_output=True, text=True, cwd=TWIN)
|
|
32
|
+
results.append(("train", r.returncode == 0 and (t / "decider" / "weights.npz").exists(), r.stderr[-300:]))
|
|
33
|
+
port = free_port()
|
|
34
|
+
server = subprocess.Popen([PY, "-m", "simthinkd.cli", "serve", str(t / "decider"), "--port", str(port)],
|
|
35
|
+
stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True)
|
|
36
|
+
try:
|
|
37
|
+
health = None
|
|
38
|
+
for _ in range(60):
|
|
39
|
+
try:
|
|
40
|
+
with urllib.request.urlopen(f"http://127.0.0.1:{port}/health", timeout=1) as h:
|
|
41
|
+
health = json.loads(h.read())
|
|
42
|
+
break
|
|
43
|
+
except OSError:
|
|
44
|
+
time.sleep(0.5)
|
|
45
|
+
results.append(("serve", bool(health and health.get("weights_sha256")), str(health)))
|
|
46
|
+
out = t / "result.json"
|
|
47
|
+
r = subprocess.run([PY, "-X", "utf8", str(TWIN / "run.py"), "--arm", "D", "--parts", "50", "--timing", "live",
|
|
48
|
+
"--url", f"http://127.0.0.1:{port}", "--out", str(out)], capture_output=True, text=True, cwd=TWIN)
|
|
49
|
+
ok = r.returncode == 0 and out.exists()
|
|
50
|
+
results.append(("run arm D", ok, (r.stdout + r.stderr)[-400:]))
|
|
51
|
+
if ok:
|
|
52
|
+
m = json.loads(out.read_text(encoding="utf-8"))["metrics"]
|
|
53
|
+
results.append(("50 parts done", m["parts"] == 50, m))
|
|
54
|
+
results.append(("decisions on time (late < 10%)", m["late_rate"] < 0.10, m["late_rate"]))
|
|
55
|
+
results.append(("decider mostly right (>= 70%)", m["correct_rate"] >= 0.70, m["correct_rate"]))
|
|
56
|
+
finally:
|
|
57
|
+
server.terminate()
|
|
58
|
+
server.wait(timeout=10)
|
|
59
|
+
for name, ok, info in results:
|
|
60
|
+
print(("PASS " if ok else "FAIL ") + name + ("" if ok else f" | {info}"))
|
|
61
|
+
return 0 if results and all(ok for _, ok, _ in results) else 1
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
if __name__ == "__main__":
|
|
65
|
+
sys.exit(main())
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Score models end to end: train a risk score (0-100) from the toy inspection task on a CPU, then check it.
|
|
2
|
+
|
|
3
|
+
python -X utf8 tests/test_score.py (needs: pip install "simthinkd[train]")
|
|
4
|
+
The teacher is a fixed formula over the toy part, so the right answer is known for every state. Checks: training finishes,
|
|
5
|
+
held-out mean absolute error is small and far better than predicting the training mean, the ranking of parts is
|
|
6
|
+
preserved, values stay inside [low, high], the saved model reloads with the same hash, scoring needs no PyTorch path.
|
|
7
|
+
"""
|
|
8
|
+
import random
|
|
9
|
+
import sys
|
|
10
|
+
import tempfile
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
ROOT = Path(__file__).resolve().parents[1]
|
|
14
|
+
sys.path.insert(0, str(ROOT / 'src'))
|
|
15
|
+
|
|
16
|
+
import simthinkd # noqa: E402
|
|
17
|
+
from simthinkd import toy # noqa: E402
|
|
18
|
+
|
|
19
|
+
SEVERITY = {'minor': 20, 'moderate': 45, 'severe': 75}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def risk(s):
|
|
23
|
+
base = 0 if s['defect'] == 'none' else SEVERITY[s['severity']]
|
|
24
|
+
base += 10 if s['image'] == 'blurry' else 0
|
|
25
|
+
base += 8 if s['belt'] == 'fast' else 0
|
|
26
|
+
base += 5 if s['defect'] != 'none' and s['queue'] == 'long' else 0
|
|
27
|
+
return float(min(100, base))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def data(n, seed):
|
|
31
|
+
rng = random.Random(seed)
|
|
32
|
+
out = []
|
|
33
|
+
for _ in range(n):
|
|
34
|
+
s, text = toy.observe(rng)
|
|
35
|
+
out.append((text, risk(s)))
|
|
36
|
+
return out
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def main():
|
|
40
|
+
train, test = data(3000, 7), data(400, 99)
|
|
41
|
+
with tempfile.TemporaryDirectory() as tmp:
|
|
42
|
+
s = simthinkd.fit_score(train, goal='Rate how risky this part is, 0 to 100.', out=str(Path(tmp) / 'risk'),
|
|
43
|
+
low=0, high=100, quiet=True)
|
|
44
|
+
preds = [s.score(t).value for t, _ in test]
|
|
45
|
+
gold = [v for _, v in test]
|
|
46
|
+
mae = sum(abs(p - g) for p, g in zip(preds, gold)) / len(gold)
|
|
47
|
+
mean = sum(v for _, v in train) / len(train)
|
|
48
|
+
base = sum(abs(mean - g) for g in gold) / len(gold)
|
|
49
|
+
pairs = [(i, j) for i in range(0, len(test), 7) for j in range(3, len(test), 11) if gold[i] != gold[j]]
|
|
50
|
+
concord = sum((preds[i] - preds[j]) * (gold[i] - gold[j]) > 0 for i, j in pairs) / len(pairs)
|
|
51
|
+
again = simthinkd.Scorer(str(Path(tmp) / 'risk'))
|
|
52
|
+
checks = [
|
|
53
|
+
(f'held-out MAE {mae:.2f} below 3 points', mae < 3),
|
|
54
|
+
(f'much better than predicting the mean (MAE {base:.2f})', mae < base / 4),
|
|
55
|
+
(f'ranking preserved ({concord:.0%} of pairs in the right order)', concord > 0.9),
|
|
56
|
+
('values stay in [0, 100]', all(0 <= p <= 100 for p in preds)),
|
|
57
|
+
('reloaded model has the same hash', again.sha256 == s.sha256),
|
|
58
|
+
('same value after reload', abs(again.score(test[0][0]).value - preds[0]) < 1e-6),
|
|
59
|
+
(f'scoring time {s.score(test[1][0]).ms:.2f} ms under 10 ms', s.score(test[1][0]).ms < 10),
|
|
60
|
+
]
|
|
61
|
+
for name, ok in checks:
|
|
62
|
+
print(('PASS ' if ok else 'FAIL ') + name)
|
|
63
|
+
sys.exit(0 if all(ok for _, ok in checks) else 1)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
if __name__ == '__main__':
|
|
67
|
+
main()
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
"""Browser demo end to end: serve web/, open index.html in headless Chromium, make a decision, read the time.
|
|
1
|
+
"""Browser demo end to end: serve web/, open demo/index.html in headless Chromium, make a decision, read the time.
|
|
2
2
|
|
|
3
3
|
python -X utf8 tests/test_web_page.py (needs: pip install playwright; playwright install chromium)
|
|
4
4
|
Checks: page loads without console errors, the example decision is TURN_LEFT, a time and the one-tick verdict
|
|
@@ -20,7 +20,7 @@ def main():
|
|
|
20
20
|
handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(ROOT / 'web'))
|
|
21
21
|
server = http.server.ThreadingHTTPServer(('127.0.0.1', 0), handler)
|
|
22
22
|
threading.Thread(target=server.serve_forever, daemon=True).start()
|
|
23
|
-
url = f'http://127.0.0.1:{server.server_address[1]}/index.html'
|
|
23
|
+
url = f'http://127.0.0.1:{server.server_address[1]}/demo/index.html'
|
|
24
24
|
checks, errors = [], []
|
|
25
25
|
with sync_playwright() as p:
|
|
26
26
|
browser = p.chromium.launch()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|