simthinkd 0.1.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. {simthinkd-0.1.0/src/simthinkd.egg-info → simthinkd-0.2.1}/PKG-INFO +94 -14
  2. {simthinkd-0.1.0 → simthinkd-0.2.1}/README.md +93 -13
  3. {simthinkd-0.1.0 → simthinkd-0.2.1}/pyproject.toml +1 -1
  4. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/__init__.py +9 -2
  5. simthinkd-0.2.1/src/simthinkd/score.py +137 -0
  6. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/server.py +16 -1
  7. {simthinkd-0.1.0 → simthinkd-0.2.1/src/simthinkd.egg-info}/PKG-INFO +94 -14
  8. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/SOURCES.txt +4 -0
  9. simthinkd-0.2.1/tests/test_factory_twin.py +65 -0
  10. simthinkd-0.2.1/tests/test_score.py +67 -0
  11. simthinkd-0.2.1/tests/test_server_latency.py +30 -0
  12. {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_web_page.py +2 -2
  13. {simthinkd-0.1.0 → simthinkd-0.2.1}/LICENSE +0 -0
  14. {simthinkd-0.1.0 → simthinkd-0.2.1}/NOTICE +0 -0
  15. {simthinkd-0.1.0 → simthinkd-0.2.1}/setup.cfg +0 -0
  16. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/bench.py +0 -0
  17. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/cli.py +0 -0
  18. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/core.py +0 -0
  19. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/data/doom_defend_states.json +0 -0
  20. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/integrations/__init__.py +0 -0
  21. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/integrations/langchain_tool.py +0 -0
  22. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/integrations/mcp_server.py +0 -0
  23. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/policy.py +0 -0
  24. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/toy.py +0 -0
  25. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/train.py +0 -0
  26. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/weights/SHA256SUMS +0 -0
  27. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/weights/doom-corridor.npz +0 -0
  28. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd/weights/doom-defend.npz +0 -0
  29. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/dependency_links.txt +0 -0
  30. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/entry_points.txt +0 -0
  31. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/requires.txt +0 -0
  32. {simthinkd-0.1.0 → simthinkd-0.2.1}/src/simthinkd.egg-info/top_level.txt +0 -0
  33. {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_notebook.py +0 -0
  34. {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_package.py +0 -0
  35. {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_space.py +0 -0
  36. {simthinkd-0.1.0 → simthinkd-0.2.1}/tests/test_web_parity.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: simthinkd
3
- Version: 0.1.0
3
+ Version: 0.2.1
4
4
  Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
5
5
  Author: Myeongseongsimjae AX Institute
6
6
  License: Apache-2.0
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
28
28
  Requires-Dist: onnxruntime>=1.17; extra == "onnx"
29
29
  Dynamic: license-file
30
30
 
31
- <p align="center"><b>SimThink D</b></p>
31
+ <p align="center"><img src="assets/banner_v2.png" alt="SimThink D: a local backup for cloud decisions" width="100%"></p>
32
32
 
33
33
  <p align="center">
34
+ <a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
34
35
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
35
36
  <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
36
37
  <a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
@@ -38,21 +39,23 @@ Dynamic: license-file
38
39
  <a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
39
40
  </p>
40
41
 
41
- **A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
42
+ <p align="center"><b>Network down. Decisions stay local.</b></p>
42
43
 
43
- <p align="center">
44
- <img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
45
- </p>
44
+ SimThink D is a small CPU model for offline backup decisions.
45
+ It has 265,665 parameters and takes about 2 ms per decision on one CPU core.
46
46
 
47
- <p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
47
+ In the factory simulation, the internet was cut for 15 seconds.
48
+ The local backup got 59 of 63 parts right, with no late decisions.
48
49
 
49
- <div align="center">
50
+ <p align="center"><a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Watch the factory simulation: local backup during a network outage" width="100%"></a></p>
50
51
 
51
- [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](web/index.html) · [Gradio demo](space/) · [Paper](docs/PAPER.md) · [Protocol](docs/PROTOCOL.md)
52
+ <p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/#demo">Try in your browser</a> · <a href="assets/factory_fallback.mp4">Watch the video (72 s)</a> · <a href="examples/factory_twin/">Factory code</a> · <a href="docs/PAPER.md">Paper</a></p>
52
53
 
53
- </div>
54
+ ```bash
55
+ pip install simthinkd
56
+ ```
54
57
 
55
- Video: [a simulated factory line keeps running when the internet drops, with SimThink D deciding on the factory PC](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/) (LinkedIn).
58
+ What makes the next decision when your network goes down?
56
59
 
57
60
  ## Words used here
58
61
 
@@ -64,13 +67,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
64
67
  ## Install
65
68
 
66
69
  ```bash
67
- pip install "simthinkd @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
70
+ pip install simthinkd
68
71
  ```
69
72
 
70
73
  That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
71
74
 
72
75
  ```bash
73
- pip install "simthinkd[train] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
76
+ pip install "simthinkd[train]"
74
77
  ```
75
78
 
76
79
  ## Quickstart
@@ -102,6 +105,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
102
105
 
103
106
  On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
104
107
 
108
+ ## Score instead of choose
109
+
110
+ Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
111
+
112
+ ```python
113
+ import random
114
+ import simthinkd
115
+ from simthinkd import toy
116
+
117
+ SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
118
+
119
+ def risk(s): # your own rule or records give the number
120
+ value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
121
+ value += 10 if s["image"] == "blurry" else 0
122
+ value += 8 if s["belt"] == "fast" else 0
123
+ value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
124
+ return float(min(100, value))
125
+
126
+ rng = random.Random(7)
127
+ examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
128
+ s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
129
+ print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
130
+ # 87.99 (0.22 ms)
131
+ ```
132
+
133
+ The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
134
+
135
+ ## Where it fits
136
+
137
+ SimThink D is a good fit when all three are true:
138
+
139
+ 1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
140
+ 2. **The answer is small.** One of up to 8 actions, or one number.
141
+ 3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
142
+
143
+ | Use | Why it fits | Try it here |
144
+ |---|---|---|
145
+ | Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
146
+ | Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
147
+ | Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
148
+ | A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
149
+
150
+ ### Use it together with a large model
151
+
152
+ Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
153
+
154
+ | Pattern | How it works | Try it here |
155
+ |---|---|---|
156
+ | **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
157
+ | **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
158
+
159
+ A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
160
+
161
+ It is not a good fit for:
162
+
163
+ - **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
164
+ - **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
165
+ - **Open-ended answers.** It picks from a list or returns one number. It does not write.
166
+
105
167
  ## Measure your own model
106
168
 
107
169
  Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
@@ -115,6 +177,12 @@ The second line measures any server that accepts the [decision request](docs/PRO
115
177
 
116
178
  ## Same CPU, same states
117
179
 
180
+ <p align="center">
181
+ <img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
182
+ </p>
183
+
184
+ <p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
185
+
118
186
  **Read this first.** Laya is a general model and was not trained on this game. Its published speed, about 33 ms per question, is on a GPU. We only had a CPU. So this table compares time inside a real-time loop. It does not compare overall quality.
119
187
 
120
188
  We used one workstation CPU (6 threads) and 1,050 Doom states, then 10 live games on the same seeds. "Missed ticks" are ticks that passed before the decider answered.
@@ -133,7 +201,9 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
133
201
  | Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
134
202
  | Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
135
203
  | Browser | [web/](web/): the same model in plain JavaScript, no server |
136
- | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"`, then `python -m simthinkd.integrations.mcp_server` |
204
+ | A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
205
+ | Gradio | [space/](space/): a small web demo you can run locally or on Hugging Face Spaces |
206
+ | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
137
207
  | LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
138
208
 
139
209
  ## How it works
@@ -149,6 +219,14 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
149
219
  - Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
150
220
  - A decider copies its teacher. It is only as good as the teacher's rules.
151
221
  - Probabilities are calibrated for the decider's own task only.
222
+ - A score model returns one number. It gives no probability or error bar with it.
223
+
224
+ ## More
225
+
226
+ - [Figures from the paper](docs/FIGURES.md)
227
+ - [What you can reproduce](docs/REPRODUCE.md)
228
+ - [Decision request format](docs/PROTOCOL.md)
229
+ - [The factory video on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/)
152
230
 
153
231
  ## Citation
154
232
 
@@ -157,6 +235,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
157
235
  - Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
158
236
  - Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
159
237
 
238
+ The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
239
+
160
240
  ## Contributing
161
241
 
162
242
  Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
@@ -1,6 +1,7 @@
1
- <p align="center"><b>SimThink D</b></p>
1
+ <p align="center"><img src="assets/banner_v2.png" alt="SimThink D: a local backup for cloud decisions" width="100%"></p>
2
2
 
3
3
  <p align="center">
4
+ <a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
4
5
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
5
6
  <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
6
7
  <a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
@@ -8,21 +9,23 @@
8
9
  <a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
9
10
  </p>
10
11
 
11
- **A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
12
+ <p align="center"><b>Network down. Decisions stay local.</b></p>
12
13
 
13
- <p align="center">
14
- <img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
15
- </p>
14
+ SimThink D is a small CPU model for offline backup decisions.
15
+ It has 265,665 parameters and takes about 2 ms per decision on one CPU core.
16
16
 
17
- <p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
17
+ In the factory simulation, the internet was cut for 15 seconds.
18
+ The local backup got 59 of 63 parts right, with no late decisions.
18
19
 
19
- <div align="center">
20
+ <p align="center"><a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Watch the factory simulation: local backup during a network outage" width="100%"></a></p>
20
21
 
21
- [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](web/index.html) · [Gradio demo](space/) · [Paper](docs/PAPER.md) · [Protocol](docs/PROTOCOL.md)
22
+ <p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/#demo">Try in your browser</a> · <a href="assets/factory_fallback.mp4">Watch the video (72 s)</a> · <a href="examples/factory_twin/">Factory code</a> · <a href="docs/PAPER.md">Paper</a></p>
22
23
 
23
- </div>
24
+ ```bash
25
+ pip install simthinkd
26
+ ```
24
27
 
25
- Video: [a simulated factory line keeps running when the internet drops, with SimThink D deciding on the factory PC](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/) (LinkedIn).
28
+ What makes the next decision when your network goes down?
26
29
 
27
30
  ## Words used here
28
31
 
@@ -34,13 +37,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
34
37
  ## Install
35
38
 
36
39
  ```bash
37
- pip install "simthinkd @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
40
+ pip install simthinkd
38
41
  ```
39
42
 
40
43
  That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
41
44
 
42
45
  ```bash
43
- pip install "simthinkd[train] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
46
+ pip install "simthinkd[train]"
44
47
  ```
45
48
 
46
49
  ## Quickstart
@@ -72,6 +75,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
72
75
 
73
76
  On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
74
77
 
78
+ ## Score instead of choose
79
+
80
+ Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
81
+
82
+ ```python
83
+ import random
84
+ import simthinkd
85
+ from simthinkd import toy
86
+
87
+ SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
88
+
89
+ def risk(s): # your own rule or records give the number
90
+ value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
91
+ value += 10 if s["image"] == "blurry" else 0
92
+ value += 8 if s["belt"] == "fast" else 0
93
+ value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
94
+ return float(min(100, value))
95
+
96
+ rng = random.Random(7)
97
+ examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
98
+ s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
99
+ print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
100
+ # 87.99 (0.22 ms)
101
+ ```
102
+
103
+ The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
104
+
105
+ ## Where it fits
106
+
107
+ SimThink D is a good fit when all three are true:
108
+
109
+ 1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
110
+ 2. **The answer is small.** One of up to 8 actions, or one number.
111
+ 3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
112
+
113
+ | Use | Why it fits | Try it here |
114
+ |---|---|---|
115
+ | Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
116
+ | Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
117
+ | Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
118
+ | A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
119
+
120
+ ### Use it together with a large model
121
+
122
+ Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
123
+
124
+ | Pattern | How it works | Try it here |
125
+ |---|---|---|
126
+ | **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
127
+ | **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
128
+
129
+ A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
130
+
131
+ It is not a good fit for:
132
+
133
+ - **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
134
+ - **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
135
+ - **Open-ended answers.** It picks from a list or returns one number. It does not write.
136
+
75
137
  ## Measure your own model
76
138
 
77
139
  Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
@@ -85,6 +147,12 @@ The second line measures any server that accepts the [decision request](docs/PRO
85
147
 
86
148
  ## Same CPU, same states
87
149
 
150
+ <p align="center">
151
+ <img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
152
+ </p>
153
+
154
+ <p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
155
+
88
156
  **Read this first.** Laya is a general model and was not trained on this game. Its published speed, about 33 ms per question, is on a GPU. We only had a CPU. So this table compares time inside a real-time loop. It does not compare overall quality.
89
157
 
90
158
  We used one workstation CPU (6 threads) and 1,050 Doom states, then 10 live games on the same seeds. "Missed ticks" are ticks that passed before the decider answered.
@@ -103,7 +171,9 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
103
171
  | Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
104
172
  | Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
105
173
  | Browser | [web/](web/): the same model in plain JavaScript, no server |
106
- | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"`, then `python -m simthinkd.integrations.mcp_server` |
174
+ | A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
175
+ | Gradio | [space/](space/): a small web demo you can run locally or on Hugging Face Spaces |
176
+ | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
107
177
  | LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
108
178
 
109
179
  ## How it works
@@ -119,6 +189,14 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
119
189
  - Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
120
190
  - A decider copies its teacher. It is only as good as the teacher's rules.
121
191
  - Probabilities are calibrated for the decider's own task only.
192
+ - A score model returns one number. It gives no probability or error bar with it.
193
+
194
+ ## More
195
+
196
+ - [Figures from the paper](docs/FIGURES.md)
197
+ - [What you can reproduce](docs/REPRODUCE.md)
198
+ - [Decision request format](docs/PROTOCOL.md)
199
+ - [The factory video on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/)
122
200
 
123
201
  ## Citation
124
202
 
@@ -127,6 +205,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
127
205
  - Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
128
206
  - Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
129
207
 
208
+ The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
209
+
130
210
  ## Contributing
131
211
 
132
212
  Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "simthinkd"
7
- version = "0.1.0"
7
+ version = "0.2.1"
8
8
  description = "A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -4,12 +4,19 @@
4
4
  print(Decider("doom-defend").decide("seen: Demon left a30 d5 | enemies 1 | sway left | gun ready | ammo25"))
5
5
  """
6
6
  from .core import Decider, Decision, available, build_request
7
+ from .score import Score, Scorer
7
8
 
8
- __version__ = '0.1.0'
9
- __all__ = ['Decider', 'Decision', 'available', 'build_request', 'fit', '__version__']
9
+ __version__ = '0.2.1'
10
+ __all__ = ['Decider', 'Decision', 'Score', 'Scorer', 'available', 'build_request', 'fit', 'fit_score', '__version__']
10
11
 
11
12
 
12
13
  def fit(*args, **kwargs):
13
14
  """Train a decider from (situation, action) pairs. See simthinkd.train.fit. Needs simthinkd[train]."""
14
15
  from .train import fit as _fit
15
16
  return _fit(*args, **kwargs)
17
+
18
+
19
+ def fit_score(*args, **kwargs):
20
+ """Train a score model from (situation, number) pairs. See simthinkd.score.fit_score. Needs simthinkd[train]."""
21
+ from .score import fit_score as _fit_score
22
+ return _fit_score(*args, **kwargs)
@@ -0,0 +1,137 @@
1
+ """Score models: the same small network as a decider, but it returns one number instead of picking an action.
2
+
3
+ import simthinkd
4
+ s = simthinkd.fit_score(examples, goal="Rate the risk of this part.", out="risk", low=0, high=100)
5
+ s.score("part: defect dent | severity severe | image clear | belt normal | rework queue long")
6
+ # Score(value=87.3, ms=1.1)
7
+
8
+ `examples` is a list of (situation sentence, number) pairs. Training needs PyTorch; scoring needs only NumPy.
9
+ Targets are standardised, the network is trained with a squared-error loss, and the checkpoint with the lowest
10
+ validation mean absolute error is kept. Splits are made by a hash of each example, as in `fit`.
11
+ """
12
+ import copy
13
+ import hashlib
14
+ import json
15
+ import os
16
+ import time
17
+ from dataclasses import dataclass
18
+ from datetime import datetime, timezone
19
+ from pathlib import Path
20
+
21
+ import numpy as np
22
+
23
+ from .core import build_request
24
+ from .policy import Policy, encode
25
+ from .train import _encode_rows, _ranker, _torch, sha, split_for
26
+
27
+ ACTION = {'SCORE': 'Return one number for this situation.'}
28
+
29
+
30
+ @dataclass
31
+ class Score:
32
+ value: float
33
+ ms: float = 0.0
34
+
35
+ def __str__(self):
36
+ return f'{self.value:.2f} ({self.ms:.2f} ms)'
37
+
38
+
39
+ class Scorer:
40
+ """A trained score model. `Scorer("path/to/folder")`."""
41
+
42
+ def __init__(self, path):
43
+ path = Path(path)
44
+ self.card = json.loads((path / 'decider.json').read_text(encoding='utf8'))
45
+ if self.card.get('kind') != 'score':
46
+ raise ValueError(f'{path} is not a score model (use Decider for action models)')
47
+ self.policy = Policy(path / self.card['weights'])
48
+ self.sha256 = self.policy.digest
49
+
50
+ def score(self, state):
51
+ body = build_request(state, ACTION, self.card.get('goal', ''))
52
+ start = time.perf_counter()
53
+ x, _, _, _ = encode(body)
54
+ raw = float(self.policy.scores(x)[0])
55
+ value = raw * self.card['std'] + self.card['mean']
56
+ low, high = self.card.get('low'), self.card.get('high')
57
+ if low is not None:
58
+ value = max(low, value)
59
+ if high is not None:
60
+ value = min(high, value)
61
+ return Score(value, (time.perf_counter() - start) * 1000)
62
+
63
+ def __repr__(self):
64
+ return f'Scorer({self.card.get("name")!r}, sha256={self.sha256[:12]})'
65
+
66
+
67
+ def _mae(torch, model, part, mean, std):
68
+ x, _, m, y = part
69
+ with torch.inference_mode():
70
+ pred = torch.cat([model(x[i:i + 256], m[i:i + 256])[:, 0] for i in range(0, len(x), 256)])
71
+ return float((pred * std + mean - y).abs().mean())
72
+
73
+
74
+ def fit_score(examples, goal='', out='my_scorer', steps=1500, seed=31, low=None, high=None, name=None, quiet=False):
75
+ """Train a score model from (situation sentence, number) pairs. Returns a ready Scorer."""
76
+ torch = _torch()
77
+ out = Path(out)
78
+ out.mkdir(parents=True, exist_ok=True)
79
+ log = (lambda record: None) if quiet else (lambda record: print(json.dumps(record), flush=True))
80
+ torch.set_num_threads(min(4, os.cpu_count() or 1))
81
+ torch.use_deterministic_algorithms(True)
82
+ device = 'cpu'
83
+ splits = {'train': [], 'validation': [], 'calibration': []}
84
+ for i, item in enumerate(examples):
85
+ state, value = (item['state'], item['value']) if isinstance(item, dict) else item
86
+ row_id = f'sc{i:07d}:' + hashlib.sha256(state.encode()).hexdigest()[:8]
87
+ splits[split_for(row_id)].append({'id': row_id, 'request': build_request(state, ACTION, goal),
88
+ 'expected': {'operation': 'SCORE'}, 'value': float(value)})
89
+ splits['validation'] += splits.pop('calibration')
90
+ if min(len(v) for v in splits.values()) == 0:
91
+ raise ValueError('need enough examples for train/validation splits (about 100 or more)')
92
+ values = np.array([r['value'] for r in splits['train']], np.float32)
93
+ mean, std = float(values.mean()), float(values.std() or 1.)
94
+ parts = {}
95
+ for name_, items in splits.items():
96
+ (x, g, m, _), width = _encode_rows(torch, items, device)
97
+ parts[name_] = (x, g, m, torch.tensor([r['value'] for r in items], dtype=torch.float32))
98
+ torch.manual_seed(seed)
99
+ model = _ranker(torch, width).to(device)
100
+ optimizer = torch.optim.AdamW(model.parameters(), lr=.002, weight_decay=.0001)
101
+ rng = torch.Generator().manual_seed(seed + 2121)
102
+ x, _, m, y = parts['train']
103
+ target = (y - mean) / std
104
+ best, weights, best_step, events = float('inf'), copy.deepcopy(model.state_dict()), 0, []
105
+ started = time.perf_counter()
106
+ for step in range(steps + 1):
107
+ if step % 100 == 0 or step == steps:
108
+ model.eval()
109
+ val = _mae(torch, model, parts['validation'], mean, std)
110
+ model.train()
111
+ events.append({'step': step, 'validation_mae': val})
112
+ if val < best:
113
+ best, weights, best_step = val, copy.deepcopy(model.state_dict()), step
114
+ log({'seed': seed, 'step': step, 'validation_mae': round(val, 6)})
115
+ if step == steps:
116
+ break
117
+ batch = torch.randint(len(x), (min(96, len(x)),), generator=rng)
118
+ pred = model(x[batch], m[batch])[:, 0]
119
+ loss = ((pred - target[batch]) ** 2).mean()
120
+ optimizer.zero_grad(set_to_none=True)
121
+ loss.backward()
122
+ optimizer.step()
123
+ model.load_state_dict(weights)
124
+ model.eval()
125
+ arrays = {k: v.detach().cpu().numpy() for k, v in model.state_dict().items()}
126
+ path = out / 'weights.npz'
127
+ np.savez_compressed(path, **arrays, temperature=np.array(1.0))
128
+ card = {'kind': 'score', 'name': name or out.name, 'goal': goal, 'weights': path.name, 'sha256': sha(path),
129
+ 'mean': mean, 'std': std, 'low': low, 'high': high,
130
+ 'parameters': sum(p.numel() for p in model.parameters()),
131
+ 'examples': {k: len(v) for k, v in splits.items()},
132
+ 'training': {'seed': seed, 'steps_run': steps, 'selected_step': best_step, 'validation_mae': best,
133
+ 'seconds': time.perf_counter() - started, 'events': events},
134
+ 'torch': torch.__version__, 'created_at': datetime.now(timezone.utc).isoformat(),
135
+ 'initialization': 'random (no pretrained weights)'}
136
+ (out / 'decider.json').write_text(json.dumps(card, ensure_ascii=False, indent=1), encoding='utf8')
137
+ return Scorer(out)
@@ -5,6 +5,7 @@
5
5
  Binds to 127.0.0.1 by default. `--delay-ms` adds a fixed wait after inference (latency-injection experiments).
6
6
  """
7
7
  import json
8
+ import socket
8
9
  import time
9
10
  from datetime import datetime, timezone
10
11
  from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
@@ -52,9 +53,23 @@ def make_handler(decider, name, delay_ms):
52
53
  return Handler
53
54
 
54
55
 
56
+ class NoDelayHTTPServer(ThreadingHTTPServer):
57
+ """Turns off Nagle's algorithm on every connection.
58
+
59
+ The handler writes the headers and the body separately. On a kept-alive connection the second small write waits
60
+ for the client's delayed ACK, so every answer took about 40 ms (measured 2026-10-03: median 42 ms kept-alive vs
61
+ 1.4 ms with TCP_NODELAY). Clients that reuse connections (Java HttpURLConnection does) missed 30 ms deadlines.
62
+ """
63
+
64
+ def get_request(self):
65
+ sock, addr = super().get_request()
66
+ sock.setsockopt(socket.IPPROTO_TCP, socket.TCP_NODELAY, 1)
67
+ return sock, addr
68
+
69
+
55
70
  def serve(decider='doom-defend', host='127.0.0.1', port=11890, delay_ms=0.0, name=None):
56
71
  decider = decider if isinstance(decider, Decider) else Decider(decider)
57
72
  name = name or f'simthink-d:{decider.preset["name"]}'
58
- server = ThreadingHTTPServer((host, port), make_handler(decider, name, delay_ms))
73
+ server = NoDelayHTTPServer((host, port), make_handler(decider, name, delay_ms))
59
74
  print(json.dumps({'listening': f'{host}:{port}', 'weights_sha256': decider.sha256, 'model': name}), flush=True)
60
75
  server.serve_forever()
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: simthinkd
3
- Version: 0.1.0
3
+ Version: 0.2.1
4
4
  Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
5
5
  Author: Myeongseongsimjae AX Institute
6
6
  License: Apache-2.0
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
28
28
  Requires-Dist: onnxruntime>=1.17; extra == "onnx"
29
29
  Dynamic: license-file
30
30
 
31
- <p align="center"><b>SimThink D</b></p>
31
+ <p align="center"><img src="assets/banner_v2.png" alt="SimThink D: a local backup for cloud decisions" width="100%"></p>
32
32
 
33
33
  <p align="center">
34
+ <a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
34
35
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
35
36
  <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
36
37
  <a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
@@ -38,21 +39,23 @@ Dynamic: license-file
38
39
  <a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
39
40
  </p>
40
41
 
41
- **A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
42
+ <p align="center"><b>Network down. Decisions stay local.</b></p>
42
43
 
43
- <p align="center">
44
- <img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
45
- </p>
44
+ SimThink D is a small CPU model for offline backup decisions.
45
+ It has 265,665 parameters and takes about 2 ms per decision on one CPU core.
46
46
 
47
- <p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
47
+ In the factory simulation, the internet was cut for 15 seconds.
48
+ The local backup got 59 of 63 parts right, with no late decisions.
48
49
 
49
- <div align="center">
50
+ <p align="center"><a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Watch the factory simulation: local backup during a network outage" width="100%"></a></p>
50
51
 
51
- [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](web/index.html) · [Gradio demo](space/) · [Paper](docs/PAPER.md) · [Protocol](docs/PROTOCOL.md)
52
+ <p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/#demo">Try in your browser</a> · <a href="assets/factory_fallback.mp4">Watch the video (72 s)</a> · <a href="examples/factory_twin/">Factory code</a> · <a href="docs/PAPER.md">Paper</a></p>
52
53
 
53
- </div>
54
+ ```bash
55
+ pip install simthinkd
56
+ ```
54
57
 
55
- Video: [a simulated factory line keeps running when the internet drops, with SimThink D deciding on the factory PC](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/) (LinkedIn).
58
+ What makes the next decision when your network goes down?
56
59
 
57
60
  ## Words used here
58
61
 
@@ -64,13 +67,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
64
67
  ## Install
65
68
 
66
69
  ```bash
67
- pip install "simthinkd @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
70
+ pip install simthinkd
68
71
  ```
69
72
 
70
73
  That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
71
74
 
72
75
  ```bash
73
- pip install "simthinkd[train] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
76
+ pip install "simthinkd[train]"
74
77
  ```
75
78
 
76
79
  ## Quickstart
@@ -102,6 +105,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
102
105
 
103
106
  On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
104
107
 
108
+ ## Score instead of choose
109
+
110
+ Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
111
+
112
+ ```python
113
+ import random
114
+ import simthinkd
115
+ from simthinkd import toy
116
+
117
+ SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
118
+
119
+ def risk(s): # your own rule or records give the number
120
+ value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
121
+ value += 10 if s["image"] == "blurry" else 0
122
+ value += 8 if s["belt"] == "fast" else 0
123
+ value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
124
+ return float(min(100, value))
125
+
126
+ rng = random.Random(7)
127
+ examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
128
+ s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
129
+ print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
130
+ # 87.99 (0.22 ms)
131
+ ```
132
+
133
+ The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
134
+
135
+ ## Where it fits
136
+
137
+ SimThink D is a good fit when all three are true:
138
+
139
+ 1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
140
+ 2. **The answer is small.** One of up to 8 actions, or one number.
141
+ 3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
142
+
143
+ | Use | Why it fits | Try it here |
144
+ |---|---|---|
145
+ | Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
146
+ | Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
147
+ | Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
148
+ | A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
149
+
150
+ ### Use it together with a large model
151
+
152
+ Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
153
+
154
+ | Pattern | How it works | Try it here |
155
+ |---|---|---|
156
+ | **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
157
+ | **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
158
+
159
+ A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
160
+
161
+ It is not a good fit for:
162
+
163
+ - **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
164
+ - **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
165
+ - **Open-ended answers.** It picks from a list or returns one number. It does not write.
166
+
105
167
  ## Measure your own model
106
168
 
107
169
  Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
@@ -115,6 +177,12 @@ The second line measures any server that accepts the [decision request](docs/PRO
115
177
 
116
178
  ## Same CPU, same states
117
179
 
180
+ <p align="center">
181
+ <img src="assets/side_by_side.gif" alt="Same Doom game, same seed, same CPU. Left: SimThink D answers every tick. Right: a 421M-parameter general decision model, used as published, misses most ticks while it thinks." width="100%" />
182
+ </p>
183
+
184
+ <p align="center"><sub>Same game, same seed, same 6-core CPU. Left: SimThink D, 1.9 ms per decision, 1 of 420 ticks missed. Right: Laya, a 421M-parameter open decision model, used as published without training on this game. It takes about 360 ms per decision and misses 390 of 420 ticks. A dark frame means the game moved on before the decider answered.</sub></p>
185
+
118
186
  **Read this first.** Laya is a general model and was not trained on this game. Its published speed, about 33 ms per question, is on a GPU. We only had a CPU. So this table compares time inside a real-time loop. It does not compare overall quality.
119
187
 
120
188
  We used one workstation CPU (6 threads) and 1,050 Doom states, then 10 live games on the same seeds. "Missed ticks" are ticks that passed before the decider answered.
@@ -133,7 +201,9 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
133
201
  | Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
134
202
  | Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
135
203
  | Browser | [web/](web/): the same model in plain JavaScript, no server |
136
- | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"`, then `python -m simthinkd.integrations.mcp_server` |
204
+ | A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
205
+ | Gradio | [space/](space/): a small web demo you can run locally or on Hugging Face Spaces |
206
+ | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
137
207
  | LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
138
208
 
139
209
  ## How it works
@@ -149,6 +219,14 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
149
219
  - Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
150
220
  - A decider copies its teacher. It is only as good as the teacher's rules.
151
221
  - Probabilities are calibrated for the decider's own task only.
222
+ - A score model returns one number. It gives no probability or error bar with it.
223
+
224
+ ## More
225
+
226
+ - [Figures from the paper](docs/FIGURES.md)
227
+ - [What you can reproduce](docs/REPRODUCE.md)
228
+ - [Decision request format](docs/PROTOCOL.md)
229
+ - [The factory video on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/)
152
230
 
153
231
  ## Citation
154
232
 
@@ -157,6 +235,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
157
235
  - Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
158
236
  - Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
159
237
 
238
+ The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
239
+
160
240
  ## Contributing
161
241
 
162
242
  Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
@@ -7,6 +7,7 @@ src/simthinkd/bench.py
7
7
  src/simthinkd/cli.py
8
8
  src/simthinkd/core.py
9
9
  src/simthinkd/policy.py
10
+ src/simthinkd/score.py
10
11
  src/simthinkd/server.py
11
12
  src/simthinkd/toy.py
12
13
  src/simthinkd/train.py
@@ -23,8 +24,11 @@ src/simthinkd/integrations/mcp_server.py
23
24
  src/simthinkd/weights/SHA256SUMS
24
25
  src/simthinkd/weights/doom-corridor.npz
25
26
  src/simthinkd/weights/doom-defend.npz
27
+ tests/test_factory_twin.py
26
28
  tests/test_notebook.py
27
29
  tests/test_package.py
30
+ tests/test_score.py
31
+ tests/test_server_latency.py
28
32
  tests/test_space.py
29
33
  tests/test_web_page.py
30
34
  tests/test_web_parity.py
@@ -0,0 +1,65 @@
1
+ """Factory twin end-to-end test: train a decider, serve it, run the twin against it, check the results.
2
+
3
+ python -X utf8 tests/test_factory_twin.py (needs simthinkd[train])
4
+
5
+ Exit 0 = all pass. This runs the same three commands as examples/factory_twin/README.md.
6
+ """
7
+ import json
8
+ import socket
9
+ import subprocess
10
+ import sys
11
+ import tempfile
12
+ import time
13
+ import urllib.request
14
+ from pathlib import Path
15
+
16
+ TWIN = Path(__file__).resolve().parents[1] / "examples" / "factory_twin"
17
+ PY = sys.executable
18
+
19
+
20
+ def free_port():
21
+ with socket.socket() as s:
22
+ s.bind(("127.0.0.1", 0))
23
+ return s.getsockname()[1]
24
+
25
+
26
+ def main():
27
+ results = []
28
+ with tempfile.TemporaryDirectory() as t:
29
+ t = Path(t)
30
+ r = subprocess.run([PY, "-X", "utf8", str(TWIN / "train_inspection_decider.py"), "--out", str(t / "decider")],
31
+ capture_output=True, text=True, cwd=TWIN)
32
+ results.append(("train", r.returncode == 0 and (t / "decider" / "weights.npz").exists(), r.stderr[-300:]))
33
+ port = free_port()
34
+ server = subprocess.Popen([PY, "-m", "simthinkd.cli", "serve", str(t / "decider"), "--port", str(port)],
35
+ stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True)
36
+ try:
37
+ health = None
38
+ for _ in range(60):
39
+ try:
40
+ with urllib.request.urlopen(f"http://127.0.0.1:{port}/health", timeout=1) as h:
41
+ health = json.loads(h.read())
42
+ break
43
+ except OSError:
44
+ time.sleep(0.5)
45
+ results.append(("serve", bool(health and health.get("weights_sha256")), str(health)))
46
+ out = t / "result.json"
47
+ r = subprocess.run([PY, "-X", "utf8", str(TWIN / "run.py"), "--arm", "D", "--parts", "50", "--timing", "live",
48
+ "--url", f"http://127.0.0.1:{port}", "--out", str(out)], capture_output=True, text=True, cwd=TWIN)
49
+ ok = r.returncode == 0 and out.exists()
50
+ results.append(("run arm D", ok, (r.stdout + r.stderr)[-400:]))
51
+ if ok:
52
+ m = json.loads(out.read_text(encoding="utf-8"))["metrics"]
53
+ results.append(("50 parts done", m["parts"] == 50, m))
54
+ results.append(("decisions on time (late < 10%)", m["late_rate"] < 0.10, m["late_rate"]))
55
+ results.append(("decider mostly right (>= 70%)", m["correct_rate"] >= 0.70, m["correct_rate"]))
56
+ finally:
57
+ server.terminate()
58
+ server.wait(timeout=10)
59
+ for name, ok, info in results:
60
+ print(("PASS " if ok else "FAIL ") + name + ("" if ok else f" | {info}"))
61
+ return 0 if results and all(ok for _, ok, _ in results) else 1
62
+
63
+
64
+ if __name__ == "__main__":
65
+ sys.exit(main())
@@ -0,0 +1,67 @@
1
+ """Score models end to end: train a risk score (0-100) from the toy inspection task on a CPU, then check it.
2
+
3
+ python -X utf8 tests/test_score.py (needs: pip install "simthinkd[train]")
4
+ The teacher is a fixed formula over the toy part, so the right answer is known for every state. Checks: training finishes,
5
+ held-out mean absolute error is small and far better than predicting the training mean, the ranking of parts is
6
+ preserved, values stay inside [low, high], the saved model reloads with the same hash, scoring needs no PyTorch path.
7
+ """
8
+ import random
9
+ import sys
10
+ import tempfile
11
+ from pathlib import Path
12
+
13
+ ROOT = Path(__file__).resolve().parents[1]
14
+ sys.path.insert(0, str(ROOT / 'src'))
15
+
16
+ import simthinkd # noqa: E402
17
+ from simthinkd import toy # noqa: E402
18
+
19
+ SEVERITY = {'minor': 20, 'moderate': 45, 'severe': 75}
20
+
21
+
22
+ def risk(s):
23
+ base = 0 if s['defect'] == 'none' else SEVERITY[s['severity']]
24
+ base += 10 if s['image'] == 'blurry' else 0
25
+ base += 8 if s['belt'] == 'fast' else 0
26
+ base += 5 if s['defect'] != 'none' and s['queue'] == 'long' else 0
27
+ return float(min(100, base))
28
+
29
+
30
+ def data(n, seed):
31
+ rng = random.Random(seed)
32
+ out = []
33
+ for _ in range(n):
34
+ s, text = toy.observe(rng)
35
+ out.append((text, risk(s)))
36
+ return out
37
+
38
+
39
+ def main():
40
+ train, test = data(3000, 7), data(400, 99)
41
+ with tempfile.TemporaryDirectory() as tmp:
42
+ s = simthinkd.fit_score(train, goal='Rate how risky this part is, 0 to 100.', out=str(Path(tmp) / 'risk'),
43
+ low=0, high=100, quiet=True)
44
+ preds = [s.score(t).value for t, _ in test]
45
+ gold = [v for _, v in test]
46
+ mae = sum(abs(p - g) for p, g in zip(preds, gold)) / len(gold)
47
+ mean = sum(v for _, v in train) / len(train)
48
+ base = sum(abs(mean - g) for g in gold) / len(gold)
49
+ pairs = [(i, j) for i in range(0, len(test), 7) for j in range(3, len(test), 11) if gold[i] != gold[j]]
50
+ concord = sum((preds[i] - preds[j]) * (gold[i] - gold[j]) > 0 for i, j in pairs) / len(pairs)
51
+ again = simthinkd.Scorer(str(Path(tmp) / 'risk'))
52
+ checks = [
53
+ (f'held-out MAE {mae:.2f} below 3 points', mae < 3),
54
+ (f'much better than predicting the mean (MAE {base:.2f})', mae < base / 4),
55
+ (f'ranking preserved ({concord:.0%} of pairs in the right order)', concord > 0.9),
56
+ ('values stay in [0, 100]', all(0 <= p <= 100 for p in preds)),
57
+ ('reloaded model has the same hash', again.sha256 == s.sha256),
58
+ ('same value after reload', abs(again.score(test[0][0]).value - preds[0]) < 1e-6),
59
+ (f'scoring time {s.score(test[1][0]).ms:.2f} ms under 10 ms', s.score(test[1][0]).ms < 10),
60
+ ]
61
+ for name, ok in checks:
62
+ print(('PASS ' if ok else 'FAIL ') + name)
63
+ sys.exit(0 if all(ok for _, ok in checks) else 1)
64
+
65
+
66
+ if __name__ == '__main__':
67
+ main()
@@ -0,0 +1,30 @@
1
+ """The HTTP server must answer fast on a kept-alive connection (regression: ~40 ms Nagle/delayed-ACK stall in 0.2.0)."""
2
+ import http.client
3
+ import statistics
4
+ import threading
5
+ import time
6
+
7
+ from simthinkd.core import Decider
8
+ from simthinkd.server import NoDelayHTTPServer, make_handler
9
+
10
+
11
+ def test_keep_alive_answers_fast():
12
+ decider = Decider('doom-defend')
13
+ server = NoDelayHTTPServer(('127.0.0.1', 0), make_handler(decider, 'test', 0.0))
14
+ threading.Thread(target=server.serve_forever, daemon=True).start()
15
+ try:
16
+ conn = http.client.HTTPConnection('127.0.0.1', server.server_address[1], timeout=2)
17
+ times = []
18
+ for _ in range(30):
19
+ start = time.perf_counter()
20
+ conn.request('GET', '/health')
21
+ conn.getresponse().read()
22
+ times.append((time.perf_counter() - start) * 1000)
23
+ assert statistics.median(times) < 20, times
24
+ finally:
25
+ server.shutdown()
26
+
27
+
28
+ if __name__ == "__main__":
29
+ test_keep_alive_answers_fast()
30
+ print("server latency test passed")
@@ -1,4 +1,4 @@
1
- """Browser demo end to end: serve web/, open index.html in headless Chromium, make a decision, read the time.
1
+ """Browser demo end to end: serve web/, open demo/index.html in headless Chromium, make a decision, read the time.
2
2
 
3
3
  python -X utf8 tests/test_web_page.py (needs: pip install playwright; playwright install chromium)
4
4
  Checks: page loads without console errors, the example decision is TURN_LEFT, a time and the one-tick verdict
@@ -20,7 +20,7 @@ def main():
20
20
  handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(ROOT / 'web'))
21
21
  server = http.server.ThreadingHTTPServer(('127.0.0.1', 0), handler)
22
22
  threading.Thread(target=server.serve_forever, daemon=True).start()
23
- url = f'http://127.0.0.1:{server.server_address[1]}/index.html'
23
+ url = f'http://127.0.0.1:{server.server_address[1]}/demo/index.html'
24
24
  checks, errors = [], []
25
25
  with sync_playwright() as p:
26
26
  browser = p.chromium.launch()
File without changes
File without changes
File without changes
File without changes