rawintent 4.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rawintent-4.0.0/LICENSE +21 -0
- rawintent-4.0.0/PKG-INFO +480 -0
- rawintent-4.0.0/README.md +451 -0
- rawintent-4.0.0/pyproject.toml +37 -0
- rawintent-4.0.0/rawintent/__init__.py +42 -0
- rawintent-4.0.0/rawintent/__main__.py +35 -0
- rawintent-4.0.0/rawintent/compiler/__init__.py +15 -0
- rawintent-4.0.0/rawintent/compiler/cli.py +179 -0
- rawintent-4.0.0/rawintent/compiler/compiler.py +92 -0
- rawintent-4.0.0/rawintent/compiler/dataset.py +336 -0
- rawintent-4.0.0/rawintent/compiler/deterministic.py +29 -0
- rawintent-4.0.0/rawintent/compiler/evaluation.py +169 -0
- rawintent-4.0.0/rawintent/compiler/executor.py +130 -0
- rawintent-4.0.0/rawintent/compiler/model.py +136 -0
- rawintent-4.0.0/rawintent/compiler/rawlang.py +284 -0
- rawintent-4.0.0/rawintent/compiler/tokenizer.py +58 -0
- rawintent-4.0.0/rawintent/compiler/training.py +189 -0
- rawintent-4.0.0/rawintent/conversation.py +52 -0
- rawintent-4.0.0/rawintent/core.py +338 -0
- rawintent-4.0.0/rawintent/english_python.py +271 -0
- rawintent-4.0.0/rawintent/entities.py +344 -0
- rawintent-4.0.0/rawintent/errors.py +119 -0
- rawintent-4.0.0/rawintent/models.py +101 -0
- rawintent-4.0.0/rawintent/packs.py +59 -0
- rawintent-4.0.0/rawintent/stdlib_packs.py +774 -0
- rawintent-4.0.0/rawintent/text.py +194 -0
- rawintent-4.0.0/rawintent.egg-info/PKG-INFO +480 -0
- rawintent-4.0.0/rawintent.egg-info/SOURCES.txt +34 -0
- rawintent-4.0.0/rawintent.egg-info/dependency_links.txt +1 -0
- rawintent-4.0.0/rawintent.egg-info/entry_points.txt +3 -0
- rawintent-4.0.0/rawintent.egg-info/requires.txt +10 -0
- rawintent-4.0.0/rawintent.egg-info/top_level.txt +1 -0
- rawintent-4.0.0/setup.cfg +4 -0
- rawintent-4.0.0/tests/test_basic.py +101 -0
- rawintent-4.0.0/tests/test_compiler.py +183 -0
- rawintent-4.0.0/tests/test_english_python.py +266 -0
rawintent-4.0.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 RawIntent contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
rawintent-4.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,480 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: rawintent
|
|
3
|
+
Version: 4.0.0
|
|
4
|
+
Summary: English-first Python runtime plus a train-from-scratch English-to-RawLang neural compiler
|
|
5
|
+
Author: RawIntent contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: nlp,intent,natural-language,compiler,transformer,rawlang,english,python
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Software Development :: Compilers
|
|
17
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Provides-Extra: http
|
|
22
|
+
Requires-Dist: requests>=2.31; extra == "http"
|
|
23
|
+
Provides-Extra: ml
|
|
24
|
+
Requires-Dist: torch>=2.0; extra == "ml"
|
|
25
|
+
Provides-Extra: all
|
|
26
|
+
Requires-Dist: requests>=2.31; extra == "all"
|
|
27
|
+
Requires-Dist: torch>=2.0; extra == "all"
|
|
28
|
+
Dynamic: license-file
|
|
29
|
+
|
|
30
|
+
# RawIntent 4.0 — Generalization-First English → RawLang → Python
|
|
31
|
+
|
|
32
|
+
RawIntent turns ordinary English into a small validated language called **RawLang**, then executes only registered Python capabilities.
|
|
33
|
+
|
|
34
|
+
Version 4.0 changes the training system so the neural compiler is explicitly trained and measured on **phrases it did not see during training**.
|
|
35
|
+
|
|
36
|
+
```text
|
|
37
|
+
messy / unseen English
|
|
38
|
+
↓
|
|
39
|
+
RawIntent compiler model
|
|
40
|
+
↓
|
|
41
|
+
RawLang
|
|
42
|
+
↓
|
|
43
|
+
parser + schema validation + permission gates
|
|
44
|
+
↓
|
|
45
|
+
registered Python functions
|
|
46
|
+
↓
|
|
47
|
+
result or plain-English error
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
The neural model does not execute arbitrary Python. It only emits RawLang.
|
|
51
|
+
|
|
52
|
+
## What changed in 4.0
|
|
53
|
+
|
|
54
|
+
RawIntent 4.0 adds:
|
|
55
|
+
|
|
56
|
+
- **phrase-family train/validation/test splits** performed before augmentation
|
|
57
|
+
- exact leakage checks so one wording family cannot appear in multiple splits
|
|
58
|
+
- held-out wording benchmarks for every action that has multiple example phrases
|
|
59
|
+
- separate **compositional generalization** families
|
|
60
|
+
- typo, slang, filler, synonym, casing, spacing, punctuation, and informal-English augmentation
|
|
61
|
+
- explicit **`UNKNOWN_CAPABILITY`** output for requests the runtime does not support
|
|
62
|
+
- unknown-capability precision / recall / F1 evaluation
|
|
63
|
+
- syntax-validity, exact-match, semantic-match, and action-accuracy metrics
|
|
64
|
+
- per-category metrics for single commands, compositions, and unsupported requests
|
|
65
|
+
- optional external validation data during training, preventing random near-duplicate leakage
|
|
66
|
+
|
|
67
|
+
This is the key idea: the model should learn the *meaning* of known capabilities rather than memorize exact sentences.
|
|
68
|
+
|
|
69
|
+
## Install
|
|
70
|
+
|
|
71
|
+
Runtime only:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
pip install .
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
With the from-scratch model trainer:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pip install ".[ml]"
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
With HTTP + ML extras:
|
|
84
|
+
|
|
85
|
+
```bash
|
|
86
|
+
pip install ".[all]"
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
## RawLang
|
|
90
|
+
|
|
91
|
+
Single command:
|
|
92
|
+
|
|
93
|
+
```text
|
|
94
|
+
random.integer low=1 high=100
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
Composition:
|
|
98
|
+
|
|
99
|
+
```text
|
|
100
|
+
LET n = random.integer low=1 high=100
|
|
101
|
+
LET root = math.sqrt value=$n
|
|
102
|
+
RETURN $root
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Unsupported capability:
|
|
106
|
+
|
|
107
|
+
```text
|
|
108
|
+
UNKNOWN_CAPABILITY
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
That last form is valid RawLang. It means the compiler understood that the request is outside the current capability schema instead of inventing a command.
|
|
112
|
+
|
|
113
|
+
## Use immediately without a trained model
|
|
114
|
+
|
|
115
|
+
```python
|
|
116
|
+
from rawintent import RawIntentLanguage
|
|
117
|
+
|
|
118
|
+
lang = RawIntentLanguage()
|
|
119
|
+
|
|
120
|
+
print(lang.compile("pick a random number between 1 and 10"))
|
|
121
|
+
# random.integer low=1 high=10
|
|
122
|
+
|
|
123
|
+
result = lang.run('sha256 hash "hello"')
|
|
124
|
+
print(result.value)
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Without a checkpoint, RawIntent uses the deterministic parser. This remains useful as a runtime, fallback, and data-bootstrap system.
|
|
128
|
+
|
|
129
|
+
## Generate a true generalization dataset
|
|
130
|
+
|
|
131
|
+
Recommended:
|
|
132
|
+
|
|
133
|
+
```bash
|
|
134
|
+
rawintent-lm generate-generalization-data \
|
|
135
|
+
--out-dir rawintent-generalization \
|
|
136
|
+
--train-variants 8 \
|
|
137
|
+
--eval-variants 6 \
|
|
138
|
+
--samples 3 \
|
|
139
|
+
--seed 7
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
It writes:
|
|
143
|
+
|
|
144
|
+
```text
|
|
145
|
+
rawintent-generalization/
|
|
146
|
+
├── train.jsonl
|
|
147
|
+
├── validation.jsonl
|
|
148
|
+
├── test.jsonl
|
|
149
|
+
└── manifest.json
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
The split occurs **before** augmentation.
|
|
153
|
+
|
|
154
|
+
If an action has phrases such as:
|
|
155
|
+
|
|
156
|
+
```text
|
|
157
|
+
TRAIN FAMILY:
|
|
158
|
+
pick a random number
|
|
159
|
+
|
|
160
|
+
VALIDATION FAMILY:
|
|
161
|
+
choose an integer in a range
|
|
162
|
+
|
|
163
|
+
TEST FAMILY:
|
|
164
|
+
grab a number between two values
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
then every noisy variation of the test family remains outside training.
|
|
168
|
+
|
|
169
|
+
This prevents a misleading setup where:
|
|
170
|
+
|
|
171
|
+
```text
|
|
172
|
+
TRAIN: pick a random number
|
|
173
|
+
TEST: please pick a random number
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
would be counted as generalization.
|
|
177
|
+
|
|
178
|
+
## Dataset records
|
|
179
|
+
|
|
180
|
+
Records contain family and category metadata:
|
|
181
|
+
|
|
182
|
+
```json
|
|
183
|
+
{
|
|
184
|
+
"source": "yo can u grab an integer minimum 10 maximum 100",
|
|
185
|
+
"target": "random.integer low=10 high=100",
|
|
186
|
+
"action": "random.integer",
|
|
187
|
+
"family": "random.integer:phrase:2:...",
|
|
188
|
+
"kind": "single",
|
|
189
|
+
"split": "test"
|
|
190
|
+
}
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
Unsupported requests look like:
|
|
194
|
+
|
|
195
|
+
```json
|
|
196
|
+
{
|
|
197
|
+
"source": "train a tensorflow image classifier",
|
|
198
|
+
"target": "UNKNOWN_CAPABILITY",
|
|
199
|
+
"action": "unknown",
|
|
200
|
+
"family": "unknown:tensorflow",
|
|
201
|
+
"kind": "unknown",
|
|
202
|
+
"split": "test"
|
|
203
|
+
}
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
## Train from scratch with held-out validation phrases
|
|
207
|
+
|
|
208
|
+
```bash
|
|
209
|
+
rawintent-lm train \
|
|
210
|
+
--data rawintent-generalization/train.jsonl \
|
|
211
|
+
--validation-data rawintent-generalization/validation.jsonl \
|
|
212
|
+
--out rawintent-compiler.pt \
|
|
213
|
+
--epochs 20 \
|
|
214
|
+
--d-model 192 \
|
|
215
|
+
--heads 6 \
|
|
216
|
+
--layers 4 \
|
|
217
|
+
--ff 768
|
|
218
|
+
```
|
|
219
|
+
|
|
220
|
+
When `--validation-data` is supplied, RawIntent checks that no phrase family appears in both training and validation.
|
|
221
|
+
|
|
222
|
+
The model initializes with new random weights. No pretrained model or external tokenizer is downloaded.
|
|
223
|
+
|
|
224
|
+
## Evaluate phrases the model never saw
|
|
225
|
+
|
|
226
|
+
```bash
|
|
227
|
+
rawintent-lm evaluate \
|
|
228
|
+
--checkpoint rawintent-compiler.pt \
|
|
229
|
+
--data rawintent-generalization/test.jsonl \
|
|
230
|
+
--out generalization-results.json
|
|
231
|
+
```
|
|
232
|
+
|
|
233
|
+
Metrics include:
|
|
234
|
+
|
|
235
|
+
```json
|
|
236
|
+
{
|
|
237
|
+
"syntax_validity": 0.0,
|
|
238
|
+
"exact_match": 0.0,
|
|
239
|
+
"semantic_match": 0.0,
|
|
240
|
+
"action_accuracy": 0.0,
|
|
241
|
+
"unknown": {
|
|
242
|
+
"precision": 0.0,
|
|
243
|
+
"recall": 0.0,
|
|
244
|
+
"f1": 0.0
|
|
245
|
+
},
|
|
246
|
+
"by_kind": {
|
|
247
|
+
"single": {},
|
|
248
|
+
"composition": {},
|
|
249
|
+
"unknown": {}
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
```
|
|
253
|
+
|
|
254
|
+
The zeros above are only an example of the JSON shape, not expected trained-model performance.
|
|
255
|
+
|
|
256
|
+
### What the metrics mean
|
|
257
|
+
|
|
258
|
+
- **syntax validity** — did the model output valid RawLang?
|
|
259
|
+
- **exact match** — exact target text match
|
|
260
|
+
- **semantic match** — same parsed program even if argument order differs
|
|
261
|
+
- **action accuracy** — did it choose the correct RawLang action sequence?
|
|
262
|
+
- **unknown precision/recall/F1** — can it reject truly unsupported capabilities without over-rejecting supported ones?
|
|
263
|
+
|
|
264
|
+
Failures are included in the report with the original English, target RawLang, generated RawLang, phrase family, and syntax error if applicable.
|
|
265
|
+
|
|
266
|
+
## Benchmark the deterministic parser too
|
|
267
|
+
|
|
268
|
+
```bash
|
|
269
|
+
rawintent-lm evaluate \
|
|
270
|
+
--deterministic \
|
|
271
|
+
--data rawintent-generalization/test.jsonl
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
This gives you a baseline against which to compare the neural compiler.
|
|
275
|
+
|
|
276
|
+
## Why unseen phrases can work
|
|
277
|
+
|
|
278
|
+
The model is not trained to memorize one sentence per command. It sees many actions expressed through multiple wording families and noisy forms.
|
|
279
|
+
|
|
280
|
+
For example it may train on:
|
|
281
|
+
|
|
282
|
+
```text
|
|
283
|
+
choose an integer from 1 to 20
|
|
284
|
+
pick a random value between 1 and 20
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
while the test set contains only:
|
|
288
|
+
|
|
289
|
+
```text
|
|
290
|
+
grab me some number anywhere in the one-to-twenty range
|
|
291
|
+
```
|
|
292
|
+
|
|
293
|
+
All map to:
|
|
294
|
+
|
|
295
|
+
```text
|
|
296
|
+
random.integer low=1 high=20
|
|
297
|
+
```
|
|
298
|
+
|
|
299
|
+
The held-out test split is designed to tell you whether that generalization is actually happening.
|
|
300
|
+
|
|
301
|
+
## Compositional generalization
|
|
302
|
+
|
|
303
|
+
Training also includes multi-step programs:
|
|
304
|
+
|
|
305
|
+
```text
|
|
306
|
+
LET password = secrets.password length=24
|
|
307
|
+
LET lower = text.lower text=$password
|
|
308
|
+
RETURN $lower
|
|
309
|
+
```
|
|
310
|
+
|
|
311
|
+
Entire composition families are held out. This tests whether the model can learn concepts such as:
|
|
312
|
+
|
|
313
|
+
```text
|
|
314
|
+
produce value A
|
|
315
|
+
feed A into operation B
|
|
316
|
+
return B
|
|
317
|
+
```
|
|
318
|
+
|
|
319
|
+
rather than memorizing one fixed multi-step example.
|
|
320
|
+
|
|
321
|
+
## Unknown concepts instead of hallucinated commands
|
|
322
|
+
|
|
323
|
+
The training data contains unsupported concepts such as external account operations, image/video generation, databases, deployment tools, live external data, and other capabilities not present in the registered runtime.
|
|
324
|
+
|
|
325
|
+
Expected output:
|
|
326
|
+
|
|
327
|
+
```text
|
|
328
|
+
UNKNOWN_CAPABILITY
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
Executing that RawLang returns a structured English error:
|
|
332
|
+
|
|
333
|
+
```python
|
|
334
|
+
from rawintent import RawLangExecutor
|
|
335
|
+
|
|
336
|
+
result = RawLangExecutor().execute("UNKNOWN_CAPABILITY")
|
|
337
|
+
print(result.error.code)
|
|
338
|
+
# UNKNOWN_CAPABILITY
|
|
339
|
+
|
|
340
|
+
print(result.error.message)
|
|
341
|
+
# I understand the request, but this RawIntent runtime does not have a registered capability for it yet.
|
|
342
|
+
```
|
|
343
|
+
|
|
344
|
+
## The custom model
|
|
345
|
+
|
|
346
|
+
RawIntent includes a small encoder-decoder Transformer implemented in PyTorch:
|
|
347
|
+
|
|
348
|
+
```python
|
|
349
|
+
from rawintent.compiler.model import ModelConfig, RawCompilerModel
|
|
350
|
+
|
|
351
|
+
config = ModelConfig(
|
|
352
|
+
d_model=192,
|
|
353
|
+
nhead=6,
|
|
354
|
+
num_encoder_layers=4,
|
|
355
|
+
num_decoder_layers=4,
|
|
356
|
+
dim_feedforward=768,
|
|
357
|
+
)
|
|
358
|
+
|
|
359
|
+
model = RawCompilerModel(config)
|
|
360
|
+
```
|
|
361
|
+
|
|
362
|
+
The tokenizer is a fixed UTF-8 byte tokenizer with 260 token IDs, so arbitrary Unicode text does not require a downloaded vocabulary.
|
|
363
|
+
|
|
364
|
+
## Compile with a trained checkpoint
|
|
365
|
+
|
|
366
|
+
```bash
|
|
367
|
+
rawintent-lm compile \
|
|
368
|
+
--checkpoint rawintent-compiler.pt \
|
|
369
|
+
"yo grab me some num anywhere from 20 up to 80"
|
|
370
|
+
```
|
|
371
|
+
|
|
372
|
+
Python:
|
|
373
|
+
|
|
374
|
+
```python
|
|
375
|
+
from rawintent import RawIntentLanguage
|
|
376
|
+
|
|
377
|
+
lang = RawIntentLanguage("rawintent-compiler.pt")
|
|
378
|
+
print(lang.compile("make me a strong 24 character password"))
|
|
379
|
+
```
|
|
380
|
+
|
|
381
|
+
Normal application mode can fall back to deterministic parsing if neural output is invalid. Evaluation bypasses that fallback so the benchmark measures the neural model itself.
|
|
382
|
+
|
|
383
|
+
## Execute RawLang directly
|
|
384
|
+
|
|
385
|
+
```python
|
|
386
|
+
from rawintent import RawLangExecutor
|
|
387
|
+
|
|
388
|
+
program = """
|
|
389
|
+
LET n = math.power base=3 exponent=2
|
|
390
|
+
LET root = math.sqrt value=$n
|
|
391
|
+
RETURN $root
|
|
392
|
+
"""
|
|
393
|
+
|
|
394
|
+
result = RawLangExecutor().execute(program)
|
|
395
|
+
print(result.value)
|
|
396
|
+
```
|
|
397
|
+
|
|
398
|
+
## Safety boundary
|
|
399
|
+
|
|
400
|
+
RawLang does not provide arbitrary `eval()`, `exec()`, `os.system()`, or unrestricted subprocess execution.
|
|
401
|
+
|
|
402
|
+
Unknown actions are rejected against the registered capability schema.
|
|
403
|
+
|
|
404
|
+
File changes are opt-in:
|
|
405
|
+
|
|
406
|
+
```python
|
|
407
|
+
from rawintent import EnglishPython
|
|
408
|
+
|
|
409
|
+
engine = EnglishPython(allow_file_changes=True)
|
|
410
|
+
```
|
|
411
|
+
|
|
412
|
+
Network writes are separately opt-in:
|
|
413
|
+
|
|
414
|
+
```python
|
|
415
|
+
engine = EnglishPython(
|
|
416
|
+
allow_network=True,
|
|
417
|
+
allow_network_writes=True,
|
|
418
|
+
)
|
|
419
|
+
```
|
|
420
|
+
|
|
421
|
+
## Python capability coverage
|
|
422
|
+
|
|
423
|
+
The runtime retains the common capability packs from RawIntent 2/3, including operations backed by:
|
|
424
|
+
|
|
425
|
+
- `random`, `secrets`
|
|
426
|
+
- `math`, `statistics`, `decimal`, `fractions`
|
|
427
|
+
- `datetime`, `time`, `calendar`, `uuid`
|
|
428
|
+
- `hashlib`, `hmac`, Base64, hex, gzip, zlib
|
|
429
|
+
- `json`, `csv`, `configparser`
|
|
430
|
+
- `re`, text, HTML, URL/query helpers
|
|
431
|
+
- `pathlib`, selected `os` / `shutil`
|
|
432
|
+
- `collections`, `itertools`, `heapq`, `bisect`, `fnmatch`, `difflib`
|
|
433
|
+
- HTTP with `requests` when installed, otherwise `urllib`
|
|
434
|
+
|
|
435
|
+
Inspect the exact language with:
|
|
436
|
+
|
|
437
|
+
```bash
|
|
438
|
+
rawintent-lm schema
|
|
439
|
+
```
|
|
440
|
+
|
|
441
|
+
## Third-party packages
|
|
442
|
+
|
|
443
|
+
Explicit module exposure remains available:
|
|
444
|
+
|
|
445
|
+
```python
|
|
446
|
+
from rawintent import EnglishPython
|
|
447
|
+
|
|
448
|
+
engine = EnglishPython()
|
|
449
|
+
engine.expose_module("numpy", ["mean", "median", "std"])
|
|
450
|
+
```
|
|
451
|
+
|
|
452
|
+
For neural understanding, add training families for those new RawLang actions, regenerate the split, and retrain.
|
|
453
|
+
|
|
454
|
+
## CLI summary
|
|
455
|
+
|
|
456
|
+
```bash
|
|
457
|
+
rawintent-lm schema
|
|
458
|
+
rawintent-lm generate-data --out bootstrap.jsonl
|
|
459
|
+
rawintent-lm generate-generalization-data --out-dir rawintent-generalization
|
|
460
|
+
rawintent-lm train --data rawintent-generalization/train.jsonl --validation-data rawintent-generalization/validation.jsonl --out model.pt
|
|
461
|
+
rawintent-lm evaluate --checkpoint model.pt --data rawintent-generalization/test.jsonl
|
|
462
|
+
rawintent-lm evaluate --deterministic --data rawintent-generalization/test.jsonl
|
|
463
|
+
rawintent-lm compile --checkpoint model.pt "pick any number one to ten"
|
|
464
|
+
rawintent-lm run --checkpoint model.pt 'sha256 hash "hello"'
|
|
465
|
+
rawintent-lm run-rawlang 'math.sqrt value=81'
|
|
466
|
+
```
|
|
467
|
+
|
|
468
|
+
## Important reality check
|
|
469
|
+
|
|
470
|
+
This architecture makes unseen phrasing *possible and measurable*; it does not make a small from-scratch model magically understand every sentence. Quality still depends on training diversity, model size, optimization, and the amount of real human language in the dataset.
|
|
471
|
+
|
|
472
|
+
Version 4.0's purpose is to make that distinction measurable: a model only scores well if it succeeds on wording and composition families that were genuinely held out from training.
|
|
473
|
+
|
|
474
|
+
## Development
|
|
475
|
+
|
|
476
|
+
```bash
|
|
477
|
+
python -m pytest -q
|
|
478
|
+
```
|
|
479
|
+
|
|
480
|
+
MIT licensed.
|