rawintent 4.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rawintent/__init__.py +42 -0
- rawintent/__main__.py +35 -0
- rawintent/compiler/__init__.py +15 -0
- rawintent/compiler/cli.py +179 -0
- rawintent/compiler/compiler.py +92 -0
- rawintent/compiler/dataset.py +336 -0
- rawintent/compiler/deterministic.py +29 -0
- rawintent/compiler/evaluation.py +169 -0
- rawintent/compiler/executor.py +130 -0
- rawintent/compiler/model.py +136 -0
- rawintent/compiler/rawlang.py +284 -0
- rawintent/compiler/tokenizer.py +58 -0
- rawintent/compiler/training.py +189 -0
- rawintent/conversation.py +52 -0
- rawintent/core.py +338 -0
- rawintent/english_python.py +271 -0
- rawintent/entities.py +344 -0
- rawintent/errors.py +119 -0
- rawintent/models.py +101 -0
- rawintent/packs.py +59 -0
- rawintent/stdlib_packs.py +774 -0
- rawintent/text.py +194 -0
- rawintent-4.0.0.dist-info/METADATA +480 -0
- rawintent-4.0.0.dist-info/RECORD +28 -0
- rawintent-4.0.0.dist-info/WHEEL +5 -0
- rawintent-4.0.0.dist-info/entry_points.txt +3 -0
- rawintent-4.0.0.dist-info/licenses/LICENSE +21 -0
- rawintent-4.0.0.dist-info/top_level.txt +1 -0
rawintent/__init__.py
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
from .core import EnglishRouter, RawIntent
|
|
2
|
+
from .conversation import Conversation
|
|
3
|
+
from .entities import slot
|
|
4
|
+
from .english_python import EnglishPython, DEFAULT_PACKS, PACK_ALIASES
|
|
5
|
+
from .errors import (
|
|
6
|
+
AmbiguousRequestError,
|
|
7
|
+
CapabilityDisabledError,
|
|
8
|
+
ConfigurationError,
|
|
9
|
+
HandlerError,
|
|
10
|
+
MissingArgumentError,
|
|
11
|
+
MissingDependencyError,
|
|
12
|
+
NegatedRequestError,
|
|
13
|
+
RawIntentError,
|
|
14
|
+
UnknownRequestError,
|
|
15
|
+
ValidationError,
|
|
16
|
+
english_exception,
|
|
17
|
+
)
|
|
18
|
+
from .models import EnglishIssue, Intent, Match, RunResult, Slot
|
|
19
|
+
from .packs import PackInfo
|
|
20
|
+
from .text import normalize, normalized_text
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"RawIntent", "EnglishRouter", "EnglishPython", "DEFAULT_PACKS", "PACK_ALIASES", "PackInfo",
|
|
24
|
+
"Conversation", "slot", "Slot", "Intent", "Match", "RunResult", "EnglishIssue",
|
|
25
|
+
"normalize", "normalized_text", "english_exception", "RawIntentError", "ConfigurationError",
|
|
26
|
+
"UnknownRequestError", "AmbiguousRequestError", "MissingArgumentError", "ValidationError",
|
|
27
|
+
"NegatedRequestError", "HandlerError", "CapabilityDisabledError", "MissingDependencyError",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
__version__ = "4.0.0"
|
|
31
|
+
|
|
32
|
+
# RawLang/compiler exports are dependency-free unless a neural checkpoint/model is requested.
|
|
33
|
+
from .compiler import (
|
|
34
|
+
ByteTokenizer, DeterministicCompiler, ExecutionResult, Instruction, NeuralCompiler,
|
|
35
|
+
Program, RawIntentLanguage, RawLangExecutor, RawLangSyntaxError, Reference,
|
|
36
|
+
ReturnStatement, UnknownCapabilityStatement, parse_program,
|
|
37
|
+
)
|
|
38
|
+
__all__ += [
|
|
39
|
+
"RawIntentLanguage", "NeuralCompiler", "DeterministicCompiler", "RawLangExecutor",
|
|
40
|
+
"ExecutionResult", "Program", "Instruction", "ReturnStatement", "UnknownCapabilityStatement", "Reference",
|
|
41
|
+
"RawLangSyntaxError", "parse_program", "ByteTokenizer",
|
|
42
|
+
]
|
rawintent/__main__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from .english_python import EnglishPython
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def main():
|
|
7
|
+
engine = EnglishPython()
|
|
8
|
+
print("RawIntent EnglishPython demo. Type an English command, 'packs', 'actions', or 'quit'.")
|
|
9
|
+
print("Examples: random number between 1 and 10 | sha256 hash \"hello\" | calculate 2 + 3 * 4")
|
|
10
|
+
while True:
|
|
11
|
+
try:
|
|
12
|
+
text = input("> ").strip()
|
|
13
|
+
except (EOFError, KeyboardInterrupt):
|
|
14
|
+
print()
|
|
15
|
+
break
|
|
16
|
+
if text.lower() in {"quit", "exit"}:
|
|
17
|
+
break
|
|
18
|
+
if text.lower() == "packs":
|
|
19
|
+
for item in engine.pack_info():
|
|
20
|
+
marker = "loaded" if item["loaded"] else "available"
|
|
21
|
+
print(f"{item['name']}: {marker} — {item['description']}")
|
|
22
|
+
continue
|
|
23
|
+
if text.lower() == "actions":
|
|
24
|
+
for name in engine.capabilities():
|
|
25
|
+
print(name)
|
|
26
|
+
continue
|
|
27
|
+
result = engine.run_safe(text)
|
|
28
|
+
if result.ok:
|
|
29
|
+
print(result.value)
|
|
30
|
+
else:
|
|
31
|
+
print(f"{result.error.code}: {result.error.message}")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
if __name__ == "__main__":
|
|
35
|
+
main()
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
from .compiler import NeuralCompiler, RawIntentLanguage
|
|
2
|
+
from .dataset import TrainingExample, generate_examples, generate_generalization_splits, read_jsonl, write_generalization_dataset, write_jsonl
|
|
3
|
+
from .deterministic import DeterministicCompiler
|
|
4
|
+
from .executor import ExecutionResult, RawLangExecutor
|
|
5
|
+
from .evaluation import evaluate_dataset, evaluate_examples, program_signature
|
|
6
|
+
from .rawlang import Instruction, Program, RawLangSyntaxError, Reference, ReturnStatement, UnknownCapabilityStatement, parse_program
|
|
7
|
+
from .tokenizer import ByteTokenizer
|
|
8
|
+
|
|
9
|
+
__all__ = [
|
|
10
|
+
"RawIntentLanguage", "NeuralCompiler", "DeterministicCompiler",
|
|
11
|
+
"RawLangExecutor", "ExecutionResult", "Program", "Instruction", "ReturnStatement",
|
|
12
|
+
"Reference", "RawLangSyntaxError", "UnknownCapabilityStatement", "parse_program", "ByteTokenizer",
|
|
13
|
+
"TrainingExample", "generate_examples", "generate_generalization_splits", "write_generalization_dataset", "read_jsonl", "write_jsonl",
|
|
14
|
+
"evaluate_dataset", "evaluate_examples", "program_signature",
|
|
15
|
+
]
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from ..english_python import EnglishPython
|
|
8
|
+
from .compiler import RawIntentLanguage
|
|
9
|
+
from .dataset import generate_examples, write_generalization_dataset, write_jsonl
|
|
10
|
+
from .executor import RawLangExecutor
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _parser() -> argparse.ArgumentParser:
|
|
14
|
+
p = argparse.ArgumentParser(prog="rawintent-lm", description="Train and use RawIntent's English -> RawLang compiler model.")
|
|
15
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
16
|
+
|
|
17
|
+
g = sub.add_parser("generate-data", help="Generate a bootstrap English/RawLang JSONL dataset.")
|
|
18
|
+
g.add_argument("--out", default="rawintent-train.jsonl")
|
|
19
|
+
g.add_argument("--variants", type=int, default=6)
|
|
20
|
+
g.add_argument("--samples", type=int, default=2, help="Different synthetic argument sets per registered phrase.")
|
|
21
|
+
g.add_argument("--seed", type=int, default=7)
|
|
22
|
+
|
|
23
|
+
gg = sub.add_parser("generate-generalization-data", help="Generate leakage-resistant train/validation/test data with held-out phrasings.")
|
|
24
|
+
gg.add_argument("--out-dir", default="rawintent-generalization")
|
|
25
|
+
gg.add_argument("--train-variants", type=int, default=8)
|
|
26
|
+
gg.add_argument("--eval-variants", type=int, default=6)
|
|
27
|
+
gg.add_argument("--samples", type=int, default=3)
|
|
28
|
+
gg.add_argument("--seed", type=int, default=7)
|
|
29
|
+
gg.add_argument("--no-unknown", action="store_true")
|
|
30
|
+
gg.add_argument("--no-compositions", action="store_true")
|
|
31
|
+
|
|
32
|
+
t = sub.add_parser("train", help="Train the compiler Transformer from scratch.")
|
|
33
|
+
t.add_argument("--data", required=True)
|
|
34
|
+
t.add_argument("--out", default="rawintent-compiler.pt")
|
|
35
|
+
t.add_argument("--validation-data", help="Held-out validation JSONL. Strongly recommended for generalization training.")
|
|
36
|
+
t.add_argument("--epochs", type=int, default=12)
|
|
37
|
+
t.add_argument("--batch-size", type=int, default=32)
|
|
38
|
+
t.add_argument("--lr", type=float, default=3e-4)
|
|
39
|
+
t.add_argument("--device", default="auto")
|
|
40
|
+
t.add_argument("--d-model", type=int, default=192)
|
|
41
|
+
t.add_argument("--heads", type=int, default=6)
|
|
42
|
+
t.add_argument("--layers", type=int, default=4)
|
|
43
|
+
t.add_argument("--ff", type=int, default=768)
|
|
44
|
+
|
|
45
|
+
e = sub.add_parser("evaluate", help="Measure unseen-phrasing and compositional generalization.")
|
|
46
|
+
e.add_argument("--data", required=True)
|
|
47
|
+
e.add_argument("--checkpoint")
|
|
48
|
+
e.add_argument("--deterministic", action="store_true", help="Benchmark the rule compiler instead of a neural checkpoint.")
|
|
49
|
+
e.add_argument("--device", default="cpu")
|
|
50
|
+
e.add_argument("--max-examples", type=int)
|
|
51
|
+
e.add_argument("--failure-limit", type=int, default=25)
|
|
52
|
+
e.add_argument("--out", help="Optional JSON metrics file.")
|
|
53
|
+
|
|
54
|
+
c = sub.add_parser("compile", help="Compile English to RawLang.")
|
|
55
|
+
c.add_argument("text")
|
|
56
|
+
c.add_argument("--checkpoint")
|
|
57
|
+
c.add_argument("--device", default="cpu")
|
|
58
|
+
c.add_argument("--no-fallback", action="store_true")
|
|
59
|
+
|
|
60
|
+
r = sub.add_parser("run", help="Compile and execute an English request.")
|
|
61
|
+
r.add_argument("text")
|
|
62
|
+
r.add_argument("--checkpoint")
|
|
63
|
+
r.add_argument("--device", default="cpu")
|
|
64
|
+
r.add_argument("--allow-network", action=argparse.BooleanOptionalAction, default=True)
|
|
65
|
+
r.add_argument("--allow-network-writes", action="store_true")
|
|
66
|
+
r.add_argument("--allow-file-changes", action="store_true")
|
|
67
|
+
|
|
68
|
+
x = sub.add_parser("run-rawlang", help="Execute RawLang directly.")
|
|
69
|
+
x.add_argument("source")
|
|
70
|
+
x.add_argument("--allow-network", action=argparse.BooleanOptionalAction, default=True)
|
|
71
|
+
x.add_argument("--allow-network-writes", action="store_true")
|
|
72
|
+
x.add_argument("--allow-file-changes", action="store_true")
|
|
73
|
+
|
|
74
|
+
s = sub.add_parser("schema", help="Print executable RawLang actions and parameters.")
|
|
75
|
+
s.add_argument("--json", action="store_true")
|
|
76
|
+
return p
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def main(argv: list[str] | None = None) -> int:
|
|
80
|
+
args = _parser().parse_args(argv)
|
|
81
|
+
if args.command == "generate-data":
|
|
82
|
+
count = write_jsonl(args.out, generate_examples(variants_per_phrase=args.variants, samples_per_phrase=args.samples, seed=args.seed))
|
|
83
|
+
print(f"Wrote {count} examples to {args.out}")
|
|
84
|
+
return 0
|
|
85
|
+
|
|
86
|
+
if args.command == "generate-generalization-data":
|
|
87
|
+
manifest = write_generalization_dataset(
|
|
88
|
+
args.out_dir,
|
|
89
|
+
train_variants=args.train_variants,
|
|
90
|
+
eval_variants=args.eval_variants,
|
|
91
|
+
samples_per_phrase=args.samples,
|
|
92
|
+
seed=args.seed,
|
|
93
|
+
include_unknown=not args.no_unknown,
|
|
94
|
+
include_compositions=not args.no_compositions,
|
|
95
|
+
)
|
|
96
|
+
print(json.dumps(manifest, indent=2))
|
|
97
|
+
return 0
|
|
98
|
+
|
|
99
|
+
if args.command == "train":
|
|
100
|
+
from .model import ModelConfig
|
|
101
|
+
from .training import TrainConfig, train
|
|
102
|
+
model_config = ModelConfig(
|
|
103
|
+
d_model=args.d_model,
|
|
104
|
+
nhead=args.heads,
|
|
105
|
+
num_encoder_layers=args.layers,
|
|
106
|
+
num_decoder_layers=args.layers,
|
|
107
|
+
dim_feedforward=args.ff,
|
|
108
|
+
)
|
|
109
|
+
metrics = train(
|
|
110
|
+
args.data,
|
|
111
|
+
args.out,
|
|
112
|
+
validation_path=args.validation_data,
|
|
113
|
+
model_config=model_config,
|
|
114
|
+
train_config=TrainConfig(
|
|
115
|
+
epochs=args.epochs,
|
|
116
|
+
batch_size=args.batch_size,
|
|
117
|
+
learning_rate=args.lr,
|
|
118
|
+
device=args.device,
|
|
119
|
+
),
|
|
120
|
+
)
|
|
121
|
+
print(json.dumps(metrics, indent=2))
|
|
122
|
+
return 0
|
|
123
|
+
|
|
124
|
+
if args.command == "evaluate":
|
|
125
|
+
from .evaluation import evaluate_dataset
|
|
126
|
+
metrics = evaluate_dataset(
|
|
127
|
+
args.data,
|
|
128
|
+
checkpoint=args.checkpoint,
|
|
129
|
+
device=args.device,
|
|
130
|
+
deterministic=args.deterministic,
|
|
131
|
+
max_examples=args.max_examples,
|
|
132
|
+
failure_limit=args.failure_limit,
|
|
133
|
+
)
|
|
134
|
+
rendered = json.dumps(metrics, indent=2)
|
|
135
|
+
print(rendered)
|
|
136
|
+
if args.out:
|
|
137
|
+
Path(args.out).write_text(rendered + "\n", encoding="utf-8")
|
|
138
|
+
return 0
|
|
139
|
+
|
|
140
|
+
if args.command == "compile":
|
|
141
|
+
lang = RawIntentLanguage(args.checkpoint, device=args.device, fallback=not args.no_fallback)
|
|
142
|
+
print(lang.compile(args.text))
|
|
143
|
+
return 0
|
|
144
|
+
|
|
145
|
+
if args.command in {"run", "run-rawlang"}:
|
|
146
|
+
engine = EnglishPython(
|
|
147
|
+
allow_network=args.allow_network,
|
|
148
|
+
allow_network_writes=args.allow_network_writes,
|
|
149
|
+
allow_file_changes=args.allow_file_changes,
|
|
150
|
+
)
|
|
151
|
+
if args.command == "run":
|
|
152
|
+
result = RawIntentLanguage(args.checkpoint, device=args.device, engine=engine).run(args.text)
|
|
153
|
+
else:
|
|
154
|
+
result = RawLangExecutor(engine).execute(args.source)
|
|
155
|
+
if result.ok:
|
|
156
|
+
print(result.value)
|
|
157
|
+
return 0
|
|
158
|
+
print(f"{result.error.code}: {result.error.message}")
|
|
159
|
+
if result.failed_line:
|
|
160
|
+
print(f"RawLang line: {result.failed_line}")
|
|
161
|
+
return 2
|
|
162
|
+
|
|
163
|
+
if args.command == "schema":
|
|
164
|
+
schema = RawLangExecutor().schema()
|
|
165
|
+
if args.json:
|
|
166
|
+
print(json.dumps(schema, indent=2, default=str))
|
|
167
|
+
else:
|
|
168
|
+
for action, info in schema.items():
|
|
169
|
+
params = ", ".join(
|
|
170
|
+
name + ("" if spec["required"] else "?")
|
|
171
|
+
for name, spec in info["parameters"].items()
|
|
172
|
+
)
|
|
173
|
+
print(f"{action}({params})")
|
|
174
|
+
return 0
|
|
175
|
+
return 1
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
if __name__ == "__main__":
|
|
179
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
from ..english_python import EnglishPython
|
|
7
|
+
from .deterministic import DeterministicCompiler
|
|
8
|
+
from .executor import ExecutionResult, RawLangExecutor
|
|
9
|
+
from .rawlang import Instruction, Program, UnknownCapabilityStatement, parse_program
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class NeuralCompiler:
|
|
13
|
+
"""Loads a trained RawCompilerModel and translates arbitrary English into RawLang."""
|
|
14
|
+
|
|
15
|
+
def __init__(self, checkpoint: str | Path, *, device: str = "cpu") -> None:
|
|
16
|
+
from .model import RawCompilerModel, compile_with_model
|
|
17
|
+
from .tokenizer import ByteTokenizer
|
|
18
|
+
|
|
19
|
+
self.model, self.metadata = RawCompilerModel.load_checkpoint(checkpoint, device=device)
|
|
20
|
+
self._compile_with_model = compile_with_model
|
|
21
|
+
self.tokenizer = ByteTokenizer()
|
|
22
|
+
self.device = device
|
|
23
|
+
|
|
24
|
+
def compile(self, text: str, *, validate_syntax: bool = True) -> str:
|
|
25
|
+
source = self._compile_with_model(self.model, text, self.tokenizer, device=self.device)
|
|
26
|
+
if validate_syntax:
|
|
27
|
+
parse_program(source)
|
|
28
|
+
return source
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class RawIntentLanguage:
|
|
32
|
+
"""High-level English -> RawLang -> validated Python capabilities interface.
|
|
33
|
+
|
|
34
|
+
Without a checkpoint it uses the deterministic compiler immediately. With a checkpoint it uses
|
|
35
|
+
the custom neural compiler and can optionally fall back to deterministic parsing when generation
|
|
36
|
+
is syntactically invalid.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(
|
|
40
|
+
self,
|
|
41
|
+
checkpoint: str | Path | None = None,
|
|
42
|
+
*,
|
|
43
|
+
device: str = "cpu",
|
|
44
|
+
fallback: bool = True,
|
|
45
|
+
engine: EnglishPython | None = None,
|
|
46
|
+
**engine_options: Any,
|
|
47
|
+
) -> None:
|
|
48
|
+
self.engine = engine or EnglishPython(**engine_options)
|
|
49
|
+
self.executor = RawLangExecutor(self.engine)
|
|
50
|
+
self.deterministic = DeterministicCompiler(self.engine)
|
|
51
|
+
self.neural = NeuralCompiler(checkpoint, device=device) if checkpoint else None
|
|
52
|
+
self.fallback = fallback
|
|
53
|
+
|
|
54
|
+
def _validate_generated_program(self, source: str) -> str:
|
|
55
|
+
import inspect
|
|
56
|
+
|
|
57
|
+
program = parse_program(source)
|
|
58
|
+
for stmt in program.statements:
|
|
59
|
+
if isinstance(stmt, UnknownCapabilityStatement):
|
|
60
|
+
continue
|
|
61
|
+
if not isinstance(stmt, Instruction):
|
|
62
|
+
continue
|
|
63
|
+
if stmt.action not in self.engine.intents:
|
|
64
|
+
raise ValueError(f"The model generated unknown RawLang action {stmt.action!r}.")
|
|
65
|
+
handler = self.engine.intents[stmt.action].handler
|
|
66
|
+
if handler is None:
|
|
67
|
+
raise ValueError(f"The model generated non-executable action {stmt.action!r}.")
|
|
68
|
+
# References are normal Python objects here, so signature binding can validate argument
|
|
69
|
+
# names and required fields without executing anything.
|
|
70
|
+
inspect.signature(handler).bind(**stmt.arguments)
|
|
71
|
+
return program.to_source()
|
|
72
|
+
|
|
73
|
+
def compile(self, text: str) -> str:
|
|
74
|
+
if self.neural is None:
|
|
75
|
+
return self.deterministic.compile(text)
|
|
76
|
+
try:
|
|
77
|
+
return self._validate_generated_program(self.neural.compile(text))
|
|
78
|
+
except Exception:
|
|
79
|
+
if not self.fallback:
|
|
80
|
+
raise
|
|
81
|
+
return self.deterministic.compile(text)
|
|
82
|
+
|
|
83
|
+
def parse(self, text: str) -> Program:
|
|
84
|
+
return parse_program(self.compile(text))
|
|
85
|
+
|
|
86
|
+
def run(self, text: str) -> ExecutionResult:
|
|
87
|
+
try:
|
|
88
|
+
rawlang = self.compile(text)
|
|
89
|
+
except Exception as exc:
|
|
90
|
+
from ..errors import english_exception
|
|
91
|
+
return ExecutionResult(False, error=english_exception(exc))
|
|
92
|
+
return self.executor.execute(rawlang)
|