simthinkd 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. {simthinkd-0.1.0/src/simthinkd.egg-info → simthinkd-0.2.0}/PKG-INFO +77 -7
  2. {simthinkd-0.1.0 → simthinkd-0.2.0}/README.md +76 -6
  3. {simthinkd-0.1.0 → simthinkd-0.2.0}/pyproject.toml +1 -1
  4. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/__init__.py +9 -2
  5. simthinkd-0.2.0/src/simthinkd/score.py +137 -0
  6. {simthinkd-0.1.0 → simthinkd-0.2.0/src/simthinkd.egg-info}/PKG-INFO +77 -7
  7. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/SOURCES.txt +3 -0
  8. simthinkd-0.2.0/tests/test_factory_twin.py +65 -0
  9. simthinkd-0.2.0/tests/test_score.py +67 -0
  10. {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_web_page.py +2 -2
  11. {simthinkd-0.1.0 → simthinkd-0.2.0}/LICENSE +0 -0
  12. {simthinkd-0.1.0 → simthinkd-0.2.0}/NOTICE +0 -0
  13. {simthinkd-0.1.0 → simthinkd-0.2.0}/setup.cfg +0 -0
  14. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/bench.py +0 -0
  15. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/cli.py +0 -0
  16. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/core.py +0 -0
  17. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/data/doom_defend_states.json +0 -0
  18. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/integrations/__init__.py +0 -0
  19. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/integrations/langchain_tool.py +0 -0
  20. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/integrations/mcp_server.py +0 -0
  21. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/policy.py +0 -0
  22. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/server.py +0 -0
  23. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/toy.py +0 -0
  24. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/train.py +0 -0
  25. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/weights/SHA256SUMS +0 -0
  26. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/weights/doom-corridor.npz +0 -0
  27. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd/weights/doom-defend.npz +0 -0
  28. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/dependency_links.txt +0 -0
  29. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/entry_points.txt +0 -0
  30. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/requires.txt +0 -0
  31. {simthinkd-0.1.0 → simthinkd-0.2.0}/src/simthinkd.egg-info/top_level.txt +0 -0
  32. {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_notebook.py +0 -0
  33. {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_package.py +0 -0
  34. {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_space.py +0 -0
  35. {simthinkd-0.1.0 → simthinkd-0.2.0}/tests/test_web_parity.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: simthinkd
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
5
5
  Author: Myeongseongsimjae AX Institute
6
6
  License: Apache-2.0
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
28
28
  Requires-Dist: onnxruntime>=1.17; extra == "onnx"
29
29
  Dynamic: license-file
30
30
 
31
- <p align="center"><b>SimThink D</b></p>
31
+ <p align="center"><img src="assets/banner.png" alt="SimThink D: a tiny decision model that runs on one CPU core" width="100%"></p>
32
32
 
33
33
  <p align="center">
34
+ <a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
34
35
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
35
36
  <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
36
37
  <a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
@@ -38,6 +39,8 @@ Dynamic: license-file
38
39
  <a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
39
40
  </p>
40
41
 
42
+ <p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/"><b>Project page, live demo and video →</b></a></p>
43
+
41
44
  **A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
42
45
 
43
46
  <p align="center">
@@ -48,11 +51,15 @@ Dynamic: license-file
48
51
 
49
52
  <div align="center">
50
53
 
51
- [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](web/index.html) · [Gradio demo](space/) · [Paper](docs/PAPER.md) · [Protocol](docs/PROTOCOL.md)
54
+ [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](https://mssj-ai-org.github.io/simthinkd/#demo) · [Gradio demo](space/) · [Factory twin](examples/factory_twin/) · [Paper](docs/PAPER.md) · [Reproduce the paper](docs/REPRODUCE.md) · [Figures](docs/FIGURES.md) · [Protocol](docs/PROTOCOL.md)
52
55
 
53
56
  </div>
54
57
 
55
- Video: [a simulated factory line keeps running when the internet drops, with SimThink D deciding on the factory PC](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/) (LinkedIn).
58
+ ### Video: the internet goes down, the line keeps going
59
+
60
+ <a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Internet down. The line kept going." width="100%"></a>
61
+
62
+ A cloud decision service runs a simulated inspection line, with SimThink D on the factory PC as its backup. When an answer does not come back in time, or the network is cut, SimThink D makes the decision. [Watch the video](assets/factory_fallback.mp4) (72 s) or see it [on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/). The simulator is in [examples/factory_twin](examples/factory_twin/).
56
63
 
57
64
  ## Words used here
58
65
 
@@ -64,13 +71,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
64
71
  ## Install
65
72
 
66
73
  ```bash
67
- pip install "simthinkd @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
74
+ pip install simthinkd
68
75
  ```
69
76
 
70
77
  That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
71
78
 
72
79
  ```bash
73
- pip install "simthinkd[train] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
80
+ pip install "simthinkd[train]"
74
81
  ```
75
82
 
76
83
  ## Quickstart
@@ -102,6 +109,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
102
109
 
103
110
  On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
104
111
 
112
+ ## Score instead of choose
113
+
114
+ Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
115
+
116
+ ```python
117
+ import random
118
+ import simthinkd
119
+ from simthinkd import toy
120
+
121
+ SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
122
+
123
+ def risk(s): # your own rule or records give the number
124
+ value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
125
+ value += 10 if s["image"] == "blurry" else 0
126
+ value += 8 if s["belt"] == "fast" else 0
127
+ value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
128
+ return float(min(100, value))
129
+
130
+ rng = random.Random(7)
131
+ examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
132
+ s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
133
+ print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
134
+ # 87.99 (0.22 ms)
135
+ ```
136
+
137
+ The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
138
+
139
+ ## Where it fits
140
+
141
+ SimThink D is a good fit when all three are true:
142
+
143
+ 1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
144
+ 2. **The answer is small.** One of up to 8 actions, or one number.
145
+ 3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
146
+
147
+ | Use | Why it fits | Try it here |
148
+ |---|---|---|
149
+ | Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
150
+ | Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
151
+ | Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
152
+ | A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
153
+
154
+ ### Use it together with a large model
155
+
156
+ Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
157
+
158
+ | Pattern | How it works | Try it here |
159
+ |---|---|---|
160
+ | **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
161
+ | **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
162
+
163
+ A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
164
+
165
+ It is not a good fit for:
166
+
167
+ - **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
168
+ - **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
169
+ - **Open-ended answers.** It picks from a list or returns one number. It does not write.
170
+
105
171
  ## Measure your own model
106
172
 
107
173
  Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
@@ -133,7 +199,8 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
133
199
  | Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
134
200
  | Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
135
201
  | Browser | [web/](web/): the same model in plain JavaScript, no server |
136
- | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"`, then `python -m simthinkd.integrations.mcp_server` |
202
+ | A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
203
+ | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
137
204
  | LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
138
205
 
139
206
  ## How it works
@@ -149,6 +216,7 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
149
216
  - Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
150
217
  - A decider copies its teacher. It is only as good as the teacher's rules.
151
218
  - Probabilities are calibrated for the decider's own task only.
219
+ - A score model returns one number. It gives no probability or error bar with it.
152
220
 
153
221
  ## Citation
154
222
 
@@ -157,6 +225,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
157
225
  - Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
158
226
  - Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
159
227
 
228
+ The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
229
+
160
230
  ## Contributing
161
231
 
162
232
  Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
@@ -1,6 +1,7 @@
1
- <p align="center"><b>SimThink D</b></p>
1
+ <p align="center"><img src="assets/banner.png" alt="SimThink D: a tiny decision model that runs on one CPU core" width="100%"></p>
2
2
 
3
3
  <p align="center">
4
+ <a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
4
5
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
5
6
  <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
6
7
  <a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
@@ -8,6 +9,8 @@
8
9
  <a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
9
10
  </p>
10
11
 
12
+ <p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/"><b>Project page, live demo and video →</b></a></p>
13
+
11
14
  **A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
12
15
 
13
16
  <p align="center">
@@ -18,11 +21,15 @@
18
21
 
19
22
  <div align="center">
20
23
 
21
- [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](web/index.html) · [Gradio demo](space/) · [Paper](docs/PAPER.md) · [Protocol](docs/PROTOCOL.md)
24
+ [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](https://mssj-ai-org.github.io/simthinkd/#demo) · [Gradio demo](space/) · [Factory twin](examples/factory_twin/) · [Paper](docs/PAPER.md) · [Reproduce the paper](docs/REPRODUCE.md) · [Figures](docs/FIGURES.md) · [Protocol](docs/PROTOCOL.md)
22
25
 
23
26
  </div>
24
27
 
25
- Video: [a simulated factory line keeps running when the internet drops, with SimThink D deciding on the factory PC](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/) (LinkedIn).
28
+ ### Video: the internet goes down, the line keeps going
29
+
30
+ <a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Internet down. The line kept going." width="100%"></a>
31
+
32
+ A cloud decision service runs a simulated inspection line, with SimThink D on the factory PC as its backup. When an answer does not come back in time, or the network is cut, SimThink D makes the decision. [Watch the video](assets/factory_fallback.mp4) (72 s) or see it [on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/). The simulator is in [examples/factory_twin](examples/factory_twin/).
26
33
 
27
34
  ## Words used here
28
35
 
@@ -34,13 +41,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
34
41
  ## Install
35
42
 
36
43
  ```bash
37
- pip install "simthinkd @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
44
+ pip install simthinkd
38
45
  ```
39
46
 
40
47
  That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
41
48
 
42
49
  ```bash
43
- pip install "simthinkd[train] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
50
+ pip install "simthinkd[train]"
44
51
  ```
45
52
 
46
53
  ## Quickstart
@@ -72,6 +79,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
72
79
 
73
80
  On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
74
81
 
82
+ ## Score instead of choose
83
+
84
+ Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
85
+
86
+ ```python
87
+ import random
88
+ import simthinkd
89
+ from simthinkd import toy
90
+
91
+ SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
92
+
93
+ def risk(s): # your own rule or records give the number
94
+ value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
95
+ value += 10 if s["image"] == "blurry" else 0
96
+ value += 8 if s["belt"] == "fast" else 0
97
+ value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
98
+ return float(min(100, value))
99
+
100
+ rng = random.Random(7)
101
+ examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
102
+ s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
103
+ print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
104
+ # 87.99 (0.22 ms)
105
+ ```
106
+
107
+ The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
108
+
109
+ ## Where it fits
110
+
111
+ SimThink D is a good fit when all three are true:
112
+
113
+ 1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
114
+ 2. **The answer is small.** One of up to 8 actions, or one number.
115
+ 3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
116
+
117
+ | Use | Why it fits | Try it here |
118
+ |---|---|---|
119
+ | Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
120
+ | Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
121
+ | Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
122
+ | A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
123
+
124
+ ### Use it together with a large model
125
+
126
+ Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
127
+
128
+ | Pattern | How it works | Try it here |
129
+ |---|---|---|
130
+ | **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
131
+ | **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
132
+
133
+ A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
134
+
135
+ It is not a good fit for:
136
+
137
+ - **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
138
+ - **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
139
+ - **Open-ended answers.** It picks from a list or returns one number. It does not write.
140
+
75
141
  ## Measure your own model
76
142
 
77
143
  Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
@@ -103,7 +169,8 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
103
169
  | Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
104
170
  | Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
105
171
  | Browser | [web/](web/): the same model in plain JavaScript, no server |
106
- | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"`, then `python -m simthinkd.integrations.mcp_server` |
172
+ | A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
173
+ | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
107
174
  | LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
108
175
 
109
176
  ## How it works
@@ -119,6 +186,7 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
119
186
  - Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
120
187
  - A decider copies its teacher. It is only as good as the teacher's rules.
121
188
  - Probabilities are calibrated for the decider's own task only.
189
+ - A score model returns one number. It gives no probability or error bar with it.
122
190
 
123
191
  ## Citation
124
192
 
@@ -127,6 +195,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
127
195
  - Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
128
196
  - Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
129
197
 
198
+ The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
199
+
130
200
  ## Contributing
131
201
 
132
202
  Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "simthinkd"
7
- version = "0.1.0"
7
+ version = "0.2.0"
8
8
  description = "A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"
@@ -4,12 +4,19 @@
4
4
  print(Decider("doom-defend").decide("seen: Demon left a30 d5 | enemies 1 | sway left | gun ready | ammo25"))
5
5
  """
6
6
  from .core import Decider, Decision, available, build_request
7
+ from .score import Score, Scorer
7
8
 
8
- __version__ = '0.1.0'
9
- __all__ = ['Decider', 'Decision', 'available', 'build_request', 'fit', '__version__']
9
+ __version__ = '0.2.0'
10
+ __all__ = ['Decider', 'Decision', 'Score', 'Scorer', 'available', 'build_request', 'fit', 'fit_score', '__version__']
10
11
 
11
12
 
12
13
  def fit(*args, **kwargs):
13
14
  """Train a decider from (situation, action) pairs. See simthinkd.train.fit. Needs simthinkd[train]."""
14
15
  from .train import fit as _fit
15
16
  return _fit(*args, **kwargs)
17
+
18
+
19
+ def fit_score(*args, **kwargs):
20
+ """Train a score model from (situation, number) pairs. See simthinkd.score.fit_score. Needs simthinkd[train]."""
21
+ from .score import fit_score as _fit_score
22
+ return _fit_score(*args, **kwargs)
@@ -0,0 +1,137 @@
1
+ """Score models: the same small network as a decider, but it returns one number instead of picking an action.
2
+
3
+ import simthinkd
4
+ s = simthinkd.fit_score(examples, goal="Rate the risk of this part.", out="risk", low=0, high=100)
5
+ s.score("part: defect dent | severity severe | image clear | belt normal | rework queue long")
6
+ # Score(value=87.3, ms=1.1)
7
+
8
+ `examples` is a list of (situation sentence, number) pairs. Training needs PyTorch; scoring needs only NumPy.
9
+ Targets are standardised, the network is trained with a squared-error loss, and the checkpoint with the lowest
10
+ validation mean absolute error is kept. Splits are made by a hash of each example, as in `fit`.
11
+ """
12
+ import copy
13
+ import hashlib
14
+ import json
15
+ import os
16
+ import time
17
+ from dataclasses import dataclass
18
+ from datetime import datetime, timezone
19
+ from pathlib import Path
20
+
21
+ import numpy as np
22
+
23
+ from .core import build_request
24
+ from .policy import Policy, encode
25
+ from .train import _encode_rows, _ranker, _torch, sha, split_for
26
+
27
+ ACTION = {'SCORE': 'Return one number for this situation.'}
28
+
29
+
30
+ @dataclass
31
+ class Score:
32
+ value: float
33
+ ms: float = 0.0
34
+
35
+ def __str__(self):
36
+ return f'{self.value:.2f} ({self.ms:.2f} ms)'
37
+
38
+
39
+ class Scorer:
40
+ """A trained score model. `Scorer("path/to/folder")`."""
41
+
42
+ def __init__(self, path):
43
+ path = Path(path)
44
+ self.card = json.loads((path / 'decider.json').read_text(encoding='utf8'))
45
+ if self.card.get('kind') != 'score':
46
+ raise ValueError(f'{path} is not a score model (use Decider for action models)')
47
+ self.policy = Policy(path / self.card['weights'])
48
+ self.sha256 = self.policy.digest
49
+
50
+ def score(self, state):
51
+ body = build_request(state, ACTION, self.card.get('goal', ''))
52
+ start = time.perf_counter()
53
+ x, _, _, _ = encode(body)
54
+ raw = float(self.policy.scores(x)[0])
55
+ value = raw * self.card['std'] + self.card['mean']
56
+ low, high = self.card.get('low'), self.card.get('high')
57
+ if low is not None:
58
+ value = max(low, value)
59
+ if high is not None:
60
+ value = min(high, value)
61
+ return Score(value, (time.perf_counter() - start) * 1000)
62
+
63
+ def __repr__(self):
64
+ return f'Scorer({self.card.get("name")!r}, sha256={self.sha256[:12]})'
65
+
66
+
67
+ def _mae(torch, model, part, mean, std):
68
+ x, _, m, y = part
69
+ with torch.inference_mode():
70
+ pred = torch.cat([model(x[i:i + 256], m[i:i + 256])[:, 0] for i in range(0, len(x), 256)])
71
+ return float((pred * std + mean - y).abs().mean())
72
+
73
+
74
+ def fit_score(examples, goal='', out='my_scorer', steps=1500, seed=31, low=None, high=None, name=None, quiet=False):
75
+ """Train a score model from (situation sentence, number) pairs. Returns a ready Scorer."""
76
+ torch = _torch()
77
+ out = Path(out)
78
+ out.mkdir(parents=True, exist_ok=True)
79
+ log = (lambda record: None) if quiet else (lambda record: print(json.dumps(record), flush=True))
80
+ torch.set_num_threads(min(4, os.cpu_count() or 1))
81
+ torch.use_deterministic_algorithms(True)
82
+ device = 'cpu'
83
+ splits = {'train': [], 'validation': [], 'calibration': []}
84
+ for i, item in enumerate(examples):
85
+ state, value = (item['state'], item['value']) if isinstance(item, dict) else item
86
+ row_id = f'sc{i:07d}:' + hashlib.sha256(state.encode()).hexdigest()[:8]
87
+ splits[split_for(row_id)].append({'id': row_id, 'request': build_request(state, ACTION, goal),
88
+ 'expected': {'operation': 'SCORE'}, 'value': float(value)})
89
+ splits['validation'] += splits.pop('calibration')
90
+ if min(len(v) for v in splits.values()) == 0:
91
+ raise ValueError('need enough examples for train/validation splits (about 100 or more)')
92
+ values = np.array([r['value'] for r in splits['train']], np.float32)
93
+ mean, std = float(values.mean()), float(values.std() or 1.)
94
+ parts = {}
95
+ for name_, items in splits.items():
96
+ (x, g, m, _), width = _encode_rows(torch, items, device)
97
+ parts[name_] = (x, g, m, torch.tensor([r['value'] for r in items], dtype=torch.float32))
98
+ torch.manual_seed(seed)
99
+ model = _ranker(torch, width).to(device)
100
+ optimizer = torch.optim.AdamW(model.parameters(), lr=.002, weight_decay=.0001)
101
+ rng = torch.Generator().manual_seed(seed + 2121)
102
+ x, _, m, y = parts['train']
103
+ target = (y - mean) / std
104
+ best, weights, best_step, events = float('inf'), copy.deepcopy(model.state_dict()), 0, []
105
+ started = time.perf_counter()
106
+ for step in range(steps + 1):
107
+ if step % 100 == 0 or step == steps:
108
+ model.eval()
109
+ val = _mae(torch, model, parts['validation'], mean, std)
110
+ model.train()
111
+ events.append({'step': step, 'validation_mae': val})
112
+ if val < best:
113
+ best, weights, best_step = val, copy.deepcopy(model.state_dict()), step
114
+ log({'seed': seed, 'step': step, 'validation_mae': round(val, 6)})
115
+ if step == steps:
116
+ break
117
+ batch = torch.randint(len(x), (min(96, len(x)),), generator=rng)
118
+ pred = model(x[batch], m[batch])[:, 0]
119
+ loss = ((pred - target[batch]) ** 2).mean()
120
+ optimizer.zero_grad(set_to_none=True)
121
+ loss.backward()
122
+ optimizer.step()
123
+ model.load_state_dict(weights)
124
+ model.eval()
125
+ arrays = {k: v.detach().cpu().numpy() for k, v in model.state_dict().items()}
126
+ path = out / 'weights.npz'
127
+ np.savez_compressed(path, **arrays, temperature=np.array(1.0))
128
+ card = {'kind': 'score', 'name': name or out.name, 'goal': goal, 'weights': path.name, 'sha256': sha(path),
129
+ 'mean': mean, 'std': std, 'low': low, 'high': high,
130
+ 'parameters': sum(p.numel() for p in model.parameters()),
131
+ 'examples': {k: len(v) for k, v in splits.items()},
132
+ 'training': {'seed': seed, 'steps_run': steps, 'selected_step': best_step, 'validation_mae': best,
133
+ 'seconds': time.perf_counter() - started, 'events': events},
134
+ 'torch': torch.__version__, 'created_at': datetime.now(timezone.utc).isoformat(),
135
+ 'initialization': 'random (no pretrained weights)'}
136
+ (out / 'decider.json').write_text(json.dumps(card, ensure_ascii=False, indent=1), encoding='utf8')
137
+ return Scorer(out)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: simthinkd
3
- Version: 0.1.0
3
+ Version: 0.2.0
4
4
  Summary: A 265k-parameter decision model that picks one action in about 2 ms on one CPU core, inside real-time loops.
5
5
  Author: Myeongseongsimjae AX Institute
6
6
  License: Apache-2.0
@@ -28,9 +28,10 @@ Requires-Dist: onnx>=1.15; extra == "onnx"
28
28
  Requires-Dist: onnxruntime>=1.17; extra == "onnx"
29
29
  Dynamic: license-file
30
30
 
31
- <p align="center"><b>SimThink D</b></p>
31
+ <p align="center"><img src="assets/banner.png" alt="SimThink D: a tiny decision model that runs on one CPU core" width="100%"></p>
32
32
 
33
33
  <p align="center">
34
+ <a href="https://pypi.org/project/simthinkd/"><img src="https://img.shields.io/pypi/v/simthinkd" alt="PyPI"></a>
34
35
  <a href="LICENSE"><img src="https://img.shields.io/badge/license-Apache%202.0-blue" alt="License: Apache 2.0"></a>
35
36
  <img src="https://img.shields.io/badge/python-3.10%2B-blue" alt="Python 3.10+">
36
37
  <a href="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml"><img src="https://github.com/MSSJ-AI-ORG/simthinkd/actions/workflows/test.yml/badge.svg" alt="Tests"></a>
@@ -38,6 +39,8 @@ Dynamic: license-file
38
39
  <a href="https://doi.org/10.5281/zenodo.23111615"><img src="https://zenodo.org/badge/DOI/10.5281/zenodo.23111615.svg" alt="DOI"></a>
39
40
  </p>
40
41
 
42
+ <p align="center"><a href="https://mssj-ai-org.github.io/simthinkd/"><b>Project page, live demo and video →</b></a></p>
43
+
41
44
  **A tiny decision model that runs on one CPU core.** It has 265,665 parameters. It picks one action in about 2 ms. That is fast enough to decide inside every tick of a game or a control loop. No GPU is needed, not even for training.
42
45
 
43
46
  <p align="center">
@@ -48,11 +51,15 @@ Dynamic: license-file
48
51
 
49
52
  <div align="center">
50
53
 
51
- [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](web/index.html) · [Gradio demo](space/) · [Paper](docs/PAPER.md) · [Protocol](docs/PROTOCOL.md)
54
+ [Open in Colab](https://colab.research.google.com/github/MSSJ-AI-ORG/simthinkd/blob/main/notebooks/quickstart.ipynb) · [Try it in your browser](https://mssj-ai-org.github.io/simthinkd/#demo) · [Gradio demo](space/) · [Factory twin](examples/factory_twin/) · [Paper](docs/PAPER.md) · [Reproduce the paper](docs/REPRODUCE.md) · [Figures](docs/FIGURES.md) · [Protocol](docs/PROTOCOL.md)
52
55
 
53
56
  </div>
54
57
 
55
- Video: [a simulated factory line keeps running when the internet drops, with SimThink D deciding on the factory PC](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/) (LinkedIn).
58
+ ### Video: the internet goes down, the line keeps going
59
+
60
+ <a href="assets/factory_fallback.mp4"><img src="assets/factory_fallback_poster.jpg" alt="Internet down. The line kept going." width="100%"></a>
61
+
62
+ A cloud decision service runs a simulated inspection line, with SimThink D on the factory PC as its backup. When an answer does not come back in time, or the network is cut, SimThink D makes the decision. [Watch the video](assets/factory_fallback.mp4) (72 s) or see it [on LinkedIn](https://www.linkedin.com/feed/update/urn:li:activity:7510142949981057024/). The simulator is in [examples/factory_twin](examples/factory_twin/).
56
63
 
57
64
  ## Words used here
58
65
 
@@ -64,13 +71,13 @@ Video: [a simulated factory line keeps running when the internet drops, with Sim
64
71
  ## Install
65
72
 
66
73
  ```bash
67
- pip install "simthinkd @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
74
+ pip install simthinkd
68
75
  ```
69
76
 
70
77
  That is all you need to make decisions. It only needs NumPy. To train your own decider, add the `train` extra (it adds PyTorch, CPU build is fine):
71
78
 
72
79
  ```bash
73
- pip install "simthinkd[train] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"
80
+ pip install "simthinkd[train]"
74
81
  ```
75
82
 
76
83
  ## Quickstart
@@ -102,6 +109,65 @@ print(d.decide("part: defect dent | severity severe | image clear | belt normal
102
109
 
103
110
  On our test PC this trains on the CPU in about 8 seconds. The new decider then matched the teacher on 500 of 500 states it had not seen.
104
111
 
112
+ ## Score instead of choose
113
+
114
+ Sometimes you need a number, not an action: a risk level, a priority, an expected wait. `fit_score` trains the same small network to return one number. Give it (situation sentence, number) pairs.
115
+
116
+ ```python
117
+ import random
118
+ import simthinkd
119
+ from simthinkd import toy
120
+
121
+ SEVERITY = {"minor": 20, "moderate": 45, "severe": 75}
122
+
123
+ def risk(s): # your own rule or records give the number
124
+ value = 0 if s["defect"] == "none" else SEVERITY[s["severity"]]
125
+ value += 10 if s["image"] == "blurry" else 0
126
+ value += 8 if s["belt"] == "fast" else 0
127
+ value += 5 if s["defect"] != "none" and s["queue"] == "long" else 0
128
+ return float(min(100, value))
129
+
130
+ rng = random.Random(7)
131
+ examples = [(text, risk(state)) for state, text in (toy.observe(rng) for _ in range(3000))]
132
+ s = simthinkd.fit_score(examples, goal="Rate how risky this part is, 0 to 100.", out="risk", low=0, high=100, quiet=True)
133
+ print(s.score("part: defect dent | severity severe | image clear | belt fast | rework queue long"))
134
+ # 87.99 (0.22 ms)
135
+ ```
136
+
137
+ The full script is [examples/risk_score.py](examples/risk_score.py). On our test PC, a risk score trained this way was off by 0.13 points on average (on a 0 to 100 scale) for 400 parts it had not seen. It put every pair of parts in the right order.
138
+
139
+ ## Where it fits
140
+
141
+ SimThink D is a good fit when all three are true:
142
+
143
+ 1. **The situation fits in one short line.** A few fields, like "seen: Demon left a30 d5 | enemies 1". Not a long text.
144
+ 2. **The answer is small.** One of up to 8 actions, or one number.
145
+ 3. **The answer must be fast, cheap or offline.** Every game tick, every part on a line, every control step, or when the network is down.
146
+
147
+ | Use | Why it fits | Try it here |
148
+ |---|---|---|
149
+ | Game characters | One decision every tick, 35 ticks a second | The two Doom deciders, `simthinkd bench` |
150
+ | Backup decider on a factory PC | Decides when a cloud service is late or offline | [examples/factory_twin](examples/factory_twin/) |
151
+ | Risk or priority score | One number per item, in well under a millisecond | [examples/risk_score.py](examples/risk_score.py) |
152
+ | A rule you already have | Learns your rule from examples, then runs it in about 2 ms | [Train your own](#train-your-own-in-seconds) |
153
+
154
+ ### Use it together with a large model
155
+
156
+ Most of the time SimThink D works best as one part of a bigger system, not alone. Two patterns:
157
+
158
+ | Pattern | How it works | Try it here |
159
+ |---|---|---|
160
+ | **First filter** | SimThink D answers every case first. When it is not sure (its confidence is below a threshold you set), the case goes to a large model or a person. Easy cases stay cheap and fast; hard cases still get the big model. | `python examples/factory_twin/run.py --arm CASCADE --tau 0.9` |
161
+ | **Local backup** | A large model or cloud service answers first. When its answer is late or the network fails, SimThink D on the local machine answers instead. | The factory video above and [examples/factory_twin](examples/factory_twin/) |
162
+
163
+ A game character (NPC) is a natural first-filter case: SimThink D handles the moment-to-moment moves on every tick, and a large model is asked only for rare, slower choices such as planning or dialogue.
164
+
165
+ It is not a good fit for:
166
+
167
+ - **Judging long text** such as essays or reports. In our own test on essay sections, a simple word-count model did better.
168
+ - **Answers that are not in the line you give it.** If the answer depends on history the line does not contain, add that history to the line or use a different tool.
169
+ - **Open-ended answers.** It picks from a list or returns one number. It does not write.
170
+
105
171
  ## Measure your own model
106
172
 
107
173
  Does your model fit inside one tick? `simthinkd bench` replays 1,050 recorded Doom states that ship with the package. It times every decision, one request at a time.
@@ -133,7 +199,8 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
133
199
  | Any language, any engine | `simthinkd serve doom-defend --port 11890`, then POST the [decision request](docs/PROTOCOL.md) to `/v1/systemone` |
134
200
  | Unity / C# | [docs/INTEGRATION_UNITY.md](docs/INTEGRATION_UNITY.md): a client loop that keeps the game running while it waits |
135
201
  | Browser | [web/](web/): the same model in plain JavaScript, no server |
136
- | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp] @ git+https://github.com/MSSJ-AI-ORG/simthinkd"`, then `python -m simthinkd.integrations.mcp_server` |
202
+ | A factory line (simulator) | [examples/factory_twin/](examples/factory_twin/): an inspection conveyor with a 400 ms deadline per part |
203
+ | MCP (Claude Desktop, Cursor and others) | `pip install "simthinkd[mcp]"`, then `python -m simthinkd.integrations.mcp_server` |
137
204
  | LangChain / LangGraph | `from simthinkd.integrations.langchain_tool import simthinkd_tool` |
138
205
 
139
206
  ## How it works
@@ -149,6 +216,7 @@ SimThink D only knows what its teacher knows. It does not reason, read long text
149
216
  - Text input only. Turn numbers into short words or bins, like "d5" or "ammo25".
150
217
  - A decider copies its teacher. It is only as good as the teacher's rules.
151
218
  - Probabilities are calibrated for the decider's own task only.
219
+ - A score model returns one number. It gives no probability or error bar with it.
152
220
 
153
221
  ## Citation
154
222
 
@@ -157,6 +225,8 @@ If you use SimThink D, please cite it with [CITATION.cff](CITATION.cff). GitHub
157
225
  - Software: [doi:10.5281/zenodo.23111615](https://doi.org/10.5281/zenodo.23111615)
158
226
  - Paper (preprint): Shin, Lee, Jeong and Kwon, "Separating Decision Time from Decision Quality in the Real-Time Gap of Distilled Deciders: Evidence from a Game and a Conveyor Simulator", [doi:10.5281/zenodo.23111659](https://doi.org/10.5281/zenodo.23111659)
159
227
 
228
+ The two bundled deciders are the exact deciders evaluated in the paper (same SHA-256). [docs/REPRODUCE.md](docs/REPRODUCE.md) lists what you can rerun from this repository and what is not released yet.
229
+
160
230
  ## Contributing
161
231
 
162
232
  Bug reports and small pull requests are welcome. See [CONTRIBUTING.md](CONTRIBUTING.md).
@@ -7,6 +7,7 @@ src/simthinkd/bench.py
7
7
  src/simthinkd/cli.py
8
8
  src/simthinkd/core.py
9
9
  src/simthinkd/policy.py
10
+ src/simthinkd/score.py
10
11
  src/simthinkd/server.py
11
12
  src/simthinkd/toy.py
12
13
  src/simthinkd/train.py
@@ -23,8 +24,10 @@ src/simthinkd/integrations/mcp_server.py
23
24
  src/simthinkd/weights/SHA256SUMS
24
25
  src/simthinkd/weights/doom-corridor.npz
25
26
  src/simthinkd/weights/doom-defend.npz
27
+ tests/test_factory_twin.py
26
28
  tests/test_notebook.py
27
29
  tests/test_package.py
30
+ tests/test_score.py
28
31
  tests/test_space.py
29
32
  tests/test_web_page.py
30
33
  tests/test_web_parity.py
@@ -0,0 +1,65 @@
1
+ """Factory twin end-to-end test: train a decider, serve it, run the twin against it, check the results.
2
+
3
+ python -X utf8 tests/test_factory_twin.py (needs simthinkd[train])
4
+
5
+ Exit 0 = all pass. This runs the same three commands as examples/factory_twin/README.md.
6
+ """
7
+ import json
8
+ import socket
9
+ import subprocess
10
+ import sys
11
+ import tempfile
12
+ import time
13
+ import urllib.request
14
+ from pathlib import Path
15
+
16
+ TWIN = Path(__file__).resolve().parents[1] / "examples" / "factory_twin"
17
+ PY = sys.executable
18
+
19
+
20
+ def free_port():
21
+ with socket.socket() as s:
22
+ s.bind(("127.0.0.1", 0))
23
+ return s.getsockname()[1]
24
+
25
+
26
+ def main():
27
+ results = []
28
+ with tempfile.TemporaryDirectory() as t:
29
+ t = Path(t)
30
+ r = subprocess.run([PY, "-X", "utf8", str(TWIN / "train_inspection_decider.py"), "--out", str(t / "decider")],
31
+ capture_output=True, text=True, cwd=TWIN)
32
+ results.append(("train", r.returncode == 0 and (t / "decider" / "weights.npz").exists(), r.stderr[-300:]))
33
+ port = free_port()
34
+ server = subprocess.Popen([PY, "-m", "simthinkd.cli", "serve", str(t / "decider"), "--port", str(port)],
35
+ stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True)
36
+ try:
37
+ health = None
38
+ for _ in range(60):
39
+ try:
40
+ with urllib.request.urlopen(f"http://127.0.0.1:{port}/health", timeout=1) as h:
41
+ health = json.loads(h.read())
42
+ break
43
+ except OSError:
44
+ time.sleep(0.5)
45
+ results.append(("serve", bool(health and health.get("weights_sha256")), str(health)))
46
+ out = t / "result.json"
47
+ r = subprocess.run([PY, "-X", "utf8", str(TWIN / "run.py"), "--arm", "D", "--parts", "50", "--timing", "live",
48
+ "--url", f"http://127.0.0.1:{port}", "--out", str(out)], capture_output=True, text=True, cwd=TWIN)
49
+ ok = r.returncode == 0 and out.exists()
50
+ results.append(("run arm D", ok, (r.stdout + r.stderr)[-400:]))
51
+ if ok:
52
+ m = json.loads(out.read_text(encoding="utf-8"))["metrics"]
53
+ results.append(("50 parts done", m["parts"] == 50, m))
54
+ results.append(("decisions on time (late < 10%)", m["late_rate"] < 0.10, m["late_rate"]))
55
+ results.append(("decider mostly right (>= 70%)", m["correct_rate"] >= 0.70, m["correct_rate"]))
56
+ finally:
57
+ server.terminate()
58
+ server.wait(timeout=10)
59
+ for name, ok, info in results:
60
+ print(("PASS " if ok else "FAIL ") + name + ("" if ok else f" | {info}"))
61
+ return 0 if results and all(ok for _, ok, _ in results) else 1
62
+
63
+
64
+ if __name__ == "__main__":
65
+ sys.exit(main())
@@ -0,0 +1,67 @@
1
+ """Score models end to end: train a risk score (0-100) from the toy inspection task on a CPU, then check it.
2
+
3
+ python -X utf8 tests/test_score.py (needs: pip install "simthinkd[train]")
4
+ The teacher is a fixed formula over the toy part, so the right answer is known for every state. Checks: training finishes,
5
+ held-out mean absolute error is small and far better than predicting the training mean, the ranking of parts is
6
+ preserved, values stay inside [low, high], the saved model reloads with the same hash, scoring needs no PyTorch path.
7
+ """
8
+ import random
9
+ import sys
10
+ import tempfile
11
+ from pathlib import Path
12
+
13
+ ROOT = Path(__file__).resolve().parents[1]
14
+ sys.path.insert(0, str(ROOT / 'src'))
15
+
16
+ import simthinkd # noqa: E402
17
+ from simthinkd import toy # noqa: E402
18
+
19
+ SEVERITY = {'minor': 20, 'moderate': 45, 'severe': 75}
20
+
21
+
22
+ def risk(s):
23
+ base = 0 if s['defect'] == 'none' else SEVERITY[s['severity']]
24
+ base += 10 if s['image'] == 'blurry' else 0
25
+ base += 8 if s['belt'] == 'fast' else 0
26
+ base += 5 if s['defect'] != 'none' and s['queue'] == 'long' else 0
27
+ return float(min(100, base))
28
+
29
+
30
+ def data(n, seed):
31
+ rng = random.Random(seed)
32
+ out = []
33
+ for _ in range(n):
34
+ s, text = toy.observe(rng)
35
+ out.append((text, risk(s)))
36
+ return out
37
+
38
+
39
+ def main():
40
+ train, test = data(3000, 7), data(400, 99)
41
+ with tempfile.TemporaryDirectory() as tmp:
42
+ s = simthinkd.fit_score(train, goal='Rate how risky this part is, 0 to 100.', out=str(Path(tmp) / 'risk'),
43
+ low=0, high=100, quiet=True)
44
+ preds = [s.score(t).value for t, _ in test]
45
+ gold = [v for _, v in test]
46
+ mae = sum(abs(p - g) for p, g in zip(preds, gold)) / len(gold)
47
+ mean = sum(v for _, v in train) / len(train)
48
+ base = sum(abs(mean - g) for g in gold) / len(gold)
49
+ pairs = [(i, j) for i in range(0, len(test), 7) for j in range(3, len(test), 11) if gold[i] != gold[j]]
50
+ concord = sum((preds[i] - preds[j]) * (gold[i] - gold[j]) > 0 for i, j in pairs) / len(pairs)
51
+ again = simthinkd.Scorer(str(Path(tmp) / 'risk'))
52
+ checks = [
53
+ (f'held-out MAE {mae:.2f} below 3 points', mae < 3),
54
+ (f'much better than predicting the mean (MAE {base:.2f})', mae < base / 4),
55
+ (f'ranking preserved ({concord:.0%} of pairs in the right order)', concord > 0.9),
56
+ ('values stay in [0, 100]', all(0 <= p <= 100 for p in preds)),
57
+ ('reloaded model has the same hash', again.sha256 == s.sha256),
58
+ ('same value after reload', abs(again.score(test[0][0]).value - preds[0]) < 1e-6),
59
+ (f'scoring time {s.score(test[1][0]).ms:.2f} ms under 10 ms', s.score(test[1][0]).ms < 10),
60
+ ]
61
+ for name, ok in checks:
62
+ print(('PASS ' if ok else 'FAIL ') + name)
63
+ sys.exit(0 if all(ok for _, ok in checks) else 1)
64
+
65
+
66
+ if __name__ == '__main__':
67
+ main()
@@ -1,4 +1,4 @@
1
- """Browser demo end to end: serve web/, open index.html in headless Chromium, make a decision, read the time.
1
+ """Browser demo end to end: serve web/, open demo/index.html in headless Chromium, make a decision, read the time.
2
2
 
3
3
  python -X utf8 tests/test_web_page.py (needs: pip install playwright; playwright install chromium)
4
4
  Checks: page loads without console errors, the example decision is TURN_LEFT, a time and the one-tick verdict
@@ -20,7 +20,7 @@ def main():
20
20
  handler = functools.partial(http.server.SimpleHTTPRequestHandler, directory=str(ROOT / 'web'))
21
21
  server = http.server.ThreadingHTTPServer(('127.0.0.1', 0), handler)
22
22
  threading.Thread(target=server.serve_forever, daemon=True).start()
23
- url = f'http://127.0.0.1:{server.server_address[1]}/index.html'
23
+ url = f'http://127.0.0.1:{server.server_address[1]}/demo/index.html'
24
24
  checks, errors = [], []
25
25
  with sync_playwright() as p:
26
26
  browser = p.chromium.launch()
File without changes
File without changes
File without changes
File without changes