leio-decision 0.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- leio_decision-0.3.2.dist-info/METADATA +438 -0
- leio_decision-0.3.2.dist-info/RECORD +43 -0
- leio_decision-0.3.2.dist-info/WHEEL +4 -0
- leio_decision-0.3.2.dist-info/entry_points.txt +6 -0
- olaya/__init__.py +65 -0
- olaya/axioms.py +393 -0
- olaya/benchmarks/__init__.py +51 -0
- olaya/benchmarks/analysis.py +383 -0
- olaya/benchmarks/features.py +517 -0
- olaya/benchmarks/learned.py +790 -0
- olaya/benchmarks/llm_judge.py +855 -0
- olaya/benchmarks/oaei.py +1645 -0
- olaya/benchmarks/reranker.py +633 -0
- olaya/benchmarks/structural.py +507 -0
- olaya/calibration.py +209 -0
- olaya/candidate_router.py +116 -0
- olaya/category_theory.py +341 -0
- olaya/data/bfo_ontology.json +527 -0
- olaya/data/doid_ontology.json +10507 -0
- olaya/data/gs1_ontology.json +632 -0
- olaya/data/incident_ontology.json +179 -0
- olaya/data/leio_ontology.json +553 -0
- olaya/data/schemaorg_ontology.json +9947 -0
- olaya/data/skos_ontology.json +61 -0
- olaya/data_gen.py +373 -0
- olaya/distilled_head.py +532 -0
- olaya/embeddings.py +327 -0
- olaya/engine.py +552 -0
- olaya/evaluation.py +118 -0
- olaya/hard_negative_sampler.py +106 -0
- olaya/model.py +151 -0
- olaya/ontology.py +651 -0
- olaya/owl2_rl.py +583 -0
- olaya/pii.py +420 -0
- olaya/real_abox.py +237 -0
- olaya/real_ontology_train.py +476 -0
- olaya/sequence.py +128 -0
- olaya/tokenizer.py +28 -0
- olaya/train.py +609 -0
- olaya/webmcp.py +1072 -0
- olaya/workflow.py +465 -0
- olaya/workflow_jsonld.py +263 -0
- olaya/workflow_replay.py +222 -0
|
@@ -0,0 +1,438 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: leio-decision
|
|
3
|
+
Version: 0.3.2
|
|
4
|
+
Summary: Ontology-Grounded System 1 Decision Model
|
|
5
|
+
Author: Dionisio & Team
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Requires-Dist: numpy>=1.26.0
|
|
8
|
+
Requires-Dist: onnxruntime>=1.17.0
|
|
9
|
+
Requires-Dist: pydantic>=2.5.0
|
|
10
|
+
Requires-Dist: torch>=2.2.0
|
|
11
|
+
Requires-Dist: transformers>=4.40.0
|
|
12
|
+
Provides-Extra: dev
|
|
13
|
+
Requires-Dist: pytest>=8.0.0; extra == 'dev'
|
|
14
|
+
Requires-Dist: rdflib>=7.0.0; extra == 'dev'
|
|
15
|
+
Requires-Dist: ruff>=0.3.0; extra == 'dev'
|
|
16
|
+
Provides-Extra: llm
|
|
17
|
+
Requires-Dist: anthropic>=0.40; extra == 'llm'
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# O-Laya: Ontology-Grounded System 1 Decisions
|
|
21
|
+
|
|
22
|
+
O-Laya turns a fast, non-autoregressive scorer into a decision engine whose output
|
|
23
|
+
respects a formal ontology. Candidates come from an OWL / RDF class hierarchy and
|
|
24
|
+
every output distribution is projected so that subsumption (`P(child) ≤ P(parent)`)
|
|
25
|
+
and disjointness (`A ⊓ B = ⊥`) hold. Each answer carries a CURIE, its lineage, the
|
|
26
|
+
per-concept probabilities, an entropy-based confidence, and a computed
|
|
27
|
+
`axioms_verified` flag.
|
|
28
|
+
|
|
29
|
+
It ships as three runtimes that share one ontology model, one axiom
|
|
30
|
+
specification ([`spec/AXIOMS.md`](spec/AXIOMS.md)) and one set of golden vectors
|
|
31
|
+
checked in CI, plus a TBox-bound workflow layer that records and replays
|
|
32
|
+
decisions as JSON-LD provenance.
|
|
33
|
+
|
|
34
|
+
## What is and is not here
|
|
35
|
+
|
|
36
|
+
Read this before anything else.
|
|
37
|
+
|
|
38
|
+
| Component | State |
|
|
39
|
+
|---|---|
|
|
40
|
+
| Ontology DAG, OWL 2 RL schema closure, axiom projection kernel | Implemented, deterministic, tested in Python, Rust and Node against shared golden vectors |
|
|
41
|
+
| OWL / RDF / SKOS / OBO ingestion to Arrow IPC (`owl-fast-core`, `olaya-core`) | Implemented; 13 real ontologies tracked as fixtures and exercised in tests |
|
|
42
|
+
| Formal Concept Analysis engine (`owl-fast-core::fca`) | Implemented and tested |
|
|
43
|
+
| Lean 4 proof core (`lean/`) | Steps 3–4 of the axiom kernel and the verifier are formalised over integer scores; `project_verified` and 15 supporting theorems, no `sorry`, axioms audited in CI. Softmax, confidence, and the `K`-sweep convergence bound are **not** proven. See [`spec/AXIOMS.md`](spec/AXIOMS.md). |
|
|
44
|
+
| TBox-bound workflow compiler, interpreter, JSON-LD traces, inference-free replay | Implemented and tested |
|
|
45
|
+
| Lexical scoring (label / synonym / description overlap with IDF) | Implemented in every runtime. **This is what every shipped runtime uses to score.** |
|
|
46
|
+
| Neural scoring head (Python, PyTorch) | Implemented and trainable. **No trained weights are committed.** `OLaya()` without a checkpoint is a randomly initialised encoder; its scores are not meaningful. |
|
|
47
|
+
| ONNX path in Rust (`onnx` feature) | Compiles. No model to load, no end-to-end test. |
|
|
48
|
+
| WebAssembly build | Builds in CI for `wasm32-unknown-unknown`; kernel unit-tested. |
|
|
49
|
+
| Calibration | Temperature fitting (`calibration.py`) is implemented and used at inference when a checkpoint carries a fitted temperature. Without one, confidence is normalised entropy with a fixed cardinality temperature and is **not** an empirically calibrated probability. |
|
|
50
|
+
| Published packages | `leio-decision` on PyPI and npm (part of the leio suite with `leio-code`), published by `release.yml` from the tagged commit; `olaya` 0.3.1 on PyPI is the same code under its previous name. Until the first `leio-decision` publish, install from a locally built wheel (see Installation). |
|
|
51
|
+
| HTTP edge demo (`/api/systemone`) | Live, lexical only, no ontology DAG. See [the edge contract](spec/systemone-reliability.md). |
|
|
52
|
+
|
|
53
|
+
No accuracy, ECE, or latency number in this repository applies to a trained
|
|
54
|
+
neural model, because none is committed. The only external benchmark we report
|
|
55
|
+
is the OAEI Conference track below, run with the lexical matcher on vendored
|
|
56
|
+
upstream data, reproducible with one command.
|
|
57
|
+
|
|
58
|
+
## OAEI Conference track (the one external benchmark)
|
|
59
|
+
|
|
60
|
+
O-Laya's matchers are evaluated with the standard OAEI alignment protocol on
|
|
61
|
+
the Conference track (OntoFarm, reference alignment ra1, 21 ontology pairs, 259
|
|
62
|
+
class and 46 property correspondences): every cross pair is scored, mutual-best
|
|
63
|
+
1:1 correspondences above a threshold fixed a priori (0.50) form the alignment,
|
|
64
|
+
and precision / recall / F1 are computed against ra1 for classes (M1),
|
|
65
|
+
properties (M2) and both (M3), micro-averaged over the 21 pairs as OAEI does.
|
|
66
|
+
Three matchers are reported: a dependency-free lexical matcher, a hybrid that
|
|
67
|
+
runs the lexical matcher first and lets a small local sentence encoder
|
|
68
|
+
(`intfloat/e5-small-v2`, CSLS-calibrated cosine over the label, the description
|
|
69
|
+
or the entity's path from the root) decide only the leftover
|
|
70
|
+
entities, and the hybrid plus training-free structural propagation over the
|
|
71
|
+
ontology graphs (Similarity Flooding, neighbour aggregation or embedding
|
|
72
|
+
smoothing, `olaya.benchmarks.structural`). Reproduce with
|
|
73
|
+
`.venv/bin/python scripts/benchmark_oaei.py` (about 10 min with the encoder, or
|
|
74
|
+
`--no-embeddings` for the lexical and lexical + structure rows in about 20 s;
|
|
75
|
+
vendored upstream data in `tests/fixtures/oaei/`).
|
|
76
|
+
|
|
77
|
+
| System (OAEI 2023, ra1, micro) | M1 P / R / F1 | M3 P / R / F1 |
|
|
78
|
+
| :--- | :---: | :---: |
|
|
79
|
+
| GraphMatcher | 0.78 / 0.89 / **0.83** | 0.76 / 0.80 / **0.78** |
|
|
80
|
+
| SORBETMtch | 0.78 / 0.75 / 0.76 | 0.78 / 0.64 / 0.70 |
|
|
81
|
+
| LogMap | 0.84 / 0.63 / 0.72 | 0.81 / 0.58 / 0.68 |
|
|
82
|
+
| Matcha | 0.76 / 0.68 / 0.72 | 0.70 / 0.64 / 0.67 |
|
|
83
|
+
| **olaya-hybrid (this repo, leave-one-pair-out CV)** | 0.81 / 0.69 / 0.75 | 0.69 / 0.66 / 0.68 |
|
|
84
|
+
| **olaya-hybrid + structure (this repo, leave-one-pair-out CV, union grid)** | 0.81 / 0.69 / 0.75 | 0.69 / 0.66 / 0.68 |
|
|
85
|
+
| **olaya-lexical (this repo, threshold 0.50)** | 0.83 / 0.64 / 0.73 | 0.73 / 0.60 / 0.66 |
|
|
86
|
+
| edna (baseline) | 0.88 / 0.54 / 0.67 | 0.79 / 0.47 / 0.59 |
|
|
87
|
+
| StringEquiv (baseline) | 0.88 / 0.50 / 0.64 | 0.80 / 0.43 / 0.56 |
|
|
88
|
+
|
|
89
|
+
Where we land: the hybrid's headline is cross-validated (threshold and fusion
|
|
90
|
+
weight chosen on the other 20 pairs for each held-out pair), which makes it the
|
|
91
|
+
honest number; the best configuration picked on the full set scores 0.76 / 0.69.
|
|
92
|
+
It ranks third on both tables: below GraphMatcher by 0.08 (M1) and 0.10 (M3),
|
|
93
|
+
below SORBETMtch by 0.01 and 0.02, tied with LogMap on M3 and above LogMap and
|
|
94
|
+
Matcha on M1. The lexical matcher alone is 0.02 behind the hybrid on both. It is
|
|
95
|
+
not state of the art. Structural propagation did not help: seeded by the hybrid, the
|
|
96
|
+
best structural grid point scores 0.75 / 0.69 on ra1 (optimistic), but when the
|
|
97
|
+
hybrid and structural grids are cross-validated together the selection never picks
|
|
98
|
+
a structural config once the hybrid grid contains the path text, so the headline
|
|
99
|
+
stays the hybrid. Structure does raise recall
|
|
100
|
+
(Similarity Flooding reaches M1 recall 0.84, close to GraphMatcher's 0.89) but
|
|
101
|
+
precision falls faster, because a training-free propagation cannot tell a matching
|
|
102
|
+
sibling pair from a merely parallel one; GraphMatcher's recall comes from attention
|
|
103
|
+
trained on these ontologies. The design choices (cascade fusion, CSLS, encoder,
|
|
104
|
+
propagation rules) were made after seeing ra1; only the numeric knobs are protected
|
|
105
|
+
by the CV. An error analysis of every missed and spurious correspondence
|
|
106
|
+
([`benchmarks/results/oaei-error-analysis.md`](benchmarks/results/oaei-error-analysis.md))
|
|
107
|
+
shows that recall is lost mostly at the threshold on compound labels and at the
|
|
108
|
+
scoring of synonyms; of everything it motivated (richer texts, head nouns, fused
|
|
109
|
+
scores, larger encoders and ensembles, Hungarian and margin decoders, adaptive
|
|
110
|
+
thresholds) only encoding the entity's path from the root helps under
|
|
111
|
+
cross-validation, and only on its own threshold sub-grid (+0.04 M3); on the
|
|
112
|
+
full grid the headline is unchanged within noise. Applying the OWL 2
|
|
113
|
+
RL coherence repair to the alignment *lowers* F1 on ra1 by about 0.02 (ra1 is the
|
|
114
|
+
unrepaired reference; the repaired rar2 is not vendored), so it is opt-in and not
|
|
115
|
+
part of the headline number. Full per-pair numbers, the threshold sweep, dataset
|
|
116
|
+
hashes and the published sources are in
|
|
117
|
+
[`benchmarks/results/oaei-conference-ra1.md`](benchmarks/results/oaei-conference-ra1.md)
|
|
118
|
+
and [`oaei-conference-ra1.json`](benchmarks/results/oaei-conference-ra1.json).
|
|
119
|
+
An LLM-judge stage (candidates from the hybrid, a language model asked with
|
|
120
|
+
ontology context whether each pair is equivalent; `scripts/benchmark_oaei_llm.py`,
|
|
121
|
+
category LLM-based) is built and validated offline with a reference-backed stub in
|
|
122
|
+
[`benchmarks/results/oaei-llm.md`](benchmarks/results/oaei-llm.md); its real-judge
|
|
123
|
+
numbers are pending an API key and are not part of the table above.
|
|
124
|
+
Earlier versions of this README reported hits@k with the gold target guaranteed to be
|
|
125
|
+
in the candidate set and a "100% violation elimination" figure computed on synthetic
|
|
126
|
+
distributions; both were retracted because they were not benchmark results.
|
|
127
|
+
|
|
128
|
+
## How it works
|
|
129
|
+
|
|
130
|
+
```mermaid
|
|
131
|
+
flowchart TD
|
|
132
|
+
subgraph "1. Candidate routing"
|
|
133
|
+
State["Input state (text / JSON)"] --> Router["Inverted index over labels, synonyms, descriptions<br/>(CamelCase-aware tokens, IDF weighting)"]
|
|
134
|
+
Onto[("Ontology<br/>classes, subClassOf, disjointWith")] --> Router
|
|
135
|
+
Router --> Sub["Active subgraph:<br/>top-K matches + full ancestor chain"]
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
subgraph "2. Scoring"
|
|
139
|
+
Sub --> Lex["Lexical overlap (all runtimes)"]
|
|
140
|
+
Sub --> Neural["Marker-pooling transformer head<br/>(Python; needs a trained checkpoint)"]
|
|
141
|
+
end
|
|
142
|
+
|
|
143
|
+
subgraph "3. Axiom projection (spec/AXIOMS.md)"
|
|
144
|
+
Lex & Neural --> Proj["Temperature softmax<br/>→ disjointness exclusion<br/>→ subsumption monotonicity<br/>→ normalised-entropy confidence"]
|
|
145
|
+
end
|
|
146
|
+
|
|
147
|
+
Proj --> Out["Grounded decision:<br/>CURIE + lineage, per-concept probabilities, confidence, axioms_verified"]
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
For `concept` questions the Python engine scores only the most specific routed
|
|
151
|
+
concepts and derives each ancestor's probability as the sum of its descendants',
|
|
152
|
+
so the hierarchy constraint holds by construction and the answer is never an
|
|
153
|
+
ancestor of another candidate.
|
|
154
|
+
|
|
155
|
+
## Runtimes
|
|
156
|
+
|
|
157
|
+
| Runtime | Package | Scoring | Scope |
|
|
158
|
+
|---|---|---|---|
|
|
159
|
+
| **Rust** | `olaya-core` (`crates/olaya-core`) | Lexical; optional ONNX cross-encoder behind the `onnx` feature (no bundled model) | Arrow-native ontology batches, OWL/Turtle ingestion via `owl-fast-core`, constant-time routing per query |
|
|
160
|
+
| **Python** | `olaya` (`src/olaya`) | Transformer marker-pooling head (untrained unless you supply a checkpoint) | Full `system_one` API: `concept`, `subsumes`, `choice`, `score`, `noul`; training, calibration, workflows, MCP server |
|
|
161
|
+
| **Node / TypeScript** | `leio` (`packages/olaya-node`) | Lexical | Zero-dependency ESM + CJS; `systemOne`, `choice`, `score`, `noul`; MCP server |
|
|
162
|
+
|
|
163
|
+
The `noul` primitive differs by design: Python returns `P(true)` for a yes/no
|
|
164
|
+
statement, Node returns an out-of-domain rejection decision, and the edge demo
|
|
165
|
+
returns a bounded negation-aware lexical score with explicit abstention.
|
|
166
|
+
|
|
167
|
+
## Installation
|
|
168
|
+
|
|
169
|
+
Until the first registry publish lands, install from a wheel built in this checkout. The
|
|
170
|
+
Central M&A toolkit consumes O-Laya the same way (a hash-pinned vendored wheel).
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
uv build --wheel --out-dir dist # dist/leio_decision-0.3.2-py3-none-any.whl
|
|
174
|
+
uvx --from dist/leio_decision-0.3.2-py3-none-any.whl leio-decision # MCP server straight from the wheel
|
|
175
|
+
uv pip install dist/leio_decision-0.3.2-py3-none-any.whl # into a venv: provides `leio-decision`, `leio`, `leio-mcp`, `olaya-mcp`
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
The wheel's base dependencies include torch (about 2 GB); `olaya.pii` alone is standard library.
|
|
179
|
+
|
|
180
|
+
Registry installs, live once the release workflow has published `leio-decision` to PyPI and
|
|
181
|
+
`leio-decision` to npm (wired in `release.yml`; first publish pending):
|
|
182
|
+
|
|
183
|
+
```bash
|
|
184
|
+
uvx leio-decision # Python MCP server, no install step
|
|
185
|
+
npx leio-decision # Node MCP server, zero dependencies
|
|
186
|
+
pip install leio-decision # Python SDK
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
From source (no pre-built artefacts are committed):
|
|
190
|
+
|
|
191
|
+
### Python
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
uv venv
|
|
195
|
+
uv pip install --index-url https://download.pytorch.org/whl/cpu torch # or a CUDA/MPS build
|
|
196
|
+
uv pip install -e ".[dev]"
|
|
197
|
+
.venv/bin/python -m pytest -q
|
|
198
|
+
```
|
|
199
|
+
|
|
200
|
+
```python
|
|
201
|
+
from olaya import OLaya
|
|
202
|
+
|
|
203
|
+
olaya = OLaya() # random-init encoder unless weights_path= points at a checkpoint
|
|
204
|
+
|
|
205
|
+
result = olaya.system_one(
|
|
206
|
+
state={"service": "billing-worker", "error": "Deadlock detected in PostgreSQL cluster."},
|
|
207
|
+
questions={
|
|
208
|
+
"triage": {"type": "concept", "root": "INCIDENT:ROOT",
|
|
209
|
+
"instructions": "Classify the operational incident."},
|
|
210
|
+
"is_data_issue": {"type": "subsumes", "concept": "INCIDENT:DATA"},
|
|
211
|
+
"urgency": {"type": "score", "criteria": ["low", "medium", "urgent", "critical"]},
|
|
212
|
+
"escalate": {"type": "noul", "instructions": "Should on-call be paged?"},
|
|
213
|
+
},
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
triage = result.answers["triage"]
|
|
217
|
+
triage.curie # a leaf under INCIDENT:ROOT, never the root itself
|
|
218
|
+
triage.lineage # ["INCIDENT:ROOT", "INCIDENT:DATA", "INCIDENT:DB_DEADLOCK"]-style path
|
|
219
|
+
triage.candidate_probabilities # distribution over the competing leaves, sums to 1
|
|
220
|
+
triage.probabilities # marginal P(C) for every concept in the active subgraph
|
|
221
|
+
triage.axioms_verified # True iff monotonicity and exclusion hold on the output
|
|
222
|
+
result.model # provenance: checkpoint hash, or "untrained-random-init"
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
Unknown `root` / `subsumes` concepts and malformed `choice` criteria raise `ValueError`. `choice`
|
|
226
|
+
accepts either a list of option names or a `{name: description}` dict.
|
|
227
|
+
|
|
228
|
+
To train a head on your own ontology and get a checkpoint the engine will load:
|
|
229
|
+
|
|
230
|
+
```bash
|
|
231
|
+
.venv/bin/python scripts/train_real_ontologies.py --domain schemaorg \
|
|
232
|
+
--ontology-path docs/public/ontologies/schemaorg_ontology.json --epochs 5 --device cpu
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
Training splits source observations into disjoint train / validation /
|
|
236
|
+
calibration / test groups before task expansion, selects the checkpoint on
|
|
237
|
+
validation, fits temperature on calibration, and reports accuracy, NLL, ECE and
|
|
238
|
+
Brier on test. See [evaluation integrity](spec/evaluation-integrity.md) for
|
|
239
|
+
what those numbers can and cannot establish.
|
|
240
|
+
|
|
241
|
+
### Rust
|
|
242
|
+
|
|
243
|
+
```toml
|
|
244
|
+
[dependencies]
|
|
245
|
+
olaya-core = { path = "crates/olaya-core" } # add features = ["onnx"] for the neural scorer
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
```rust
|
|
249
|
+
use olaya_core::{OntologyBatchBuilder, OlayaEngine};
|
|
250
|
+
|
|
251
|
+
let ttl = br#"
|
|
252
|
+
@prefix ex: <http://example.org/> .
|
|
253
|
+
@prefix owl: <http://www.w3.org/2002/07/owl#> .
|
|
254
|
+
@prefix rdfs: <http://www.w3.org/2000/01/rdf-schema#> .
|
|
255
|
+
ex:DBDeadlock a owl:Class ; rdfs:label "Database Deadlock" .
|
|
256
|
+
"#;
|
|
257
|
+
|
|
258
|
+
let batch = OntologyBatchBuilder::from_owl_bytes(ttl, owl_fast_core::RdfFormat::Turtle)?;
|
|
259
|
+
let engine = OlayaEngine::new(batch)?; // validates the schema, builds the index once
|
|
260
|
+
let result = engine.system_one("Fatal deadlock detected on PostgreSQL", 8)?;
|
|
261
|
+
println!("{} ({:.3})", result.selected_curie, result.confidence);
|
|
262
|
+
```
|
|
263
|
+
|
|
264
|
+
`cargo test --workspace` runs the suite. `cargo run --release -p olaya-core --bin bench_arrow`
|
|
265
|
+
measures lexical routing latency on your machine; we do not publish numbers from
|
|
266
|
+
other machines here.
|
|
267
|
+
|
|
268
|
+
`owl-fast-core::fca` builds a concept lattice from an object/attribute context
|
|
269
|
+
(Close-by-One over dense bitsets) and converts it into ontology entities the
|
|
270
|
+
engine can load:
|
|
271
|
+
|
|
272
|
+
```rust
|
|
273
|
+
use owl_fast_core::fca::{ConceptLattice, FormalContext};
|
|
274
|
+
|
|
275
|
+
let ctx = FormalContext::from_named_pairs(&[
|
|
276
|
+
("item_1", "perishable"),
|
|
277
|
+
("item_1", "cold_chain"),
|
|
278
|
+
("item_2", "gtin_tagged"),
|
|
279
|
+
]);
|
|
280
|
+
let lattice = ConceptLattice::build(ctx);
|
|
281
|
+
let entities = lattice.to_ontology_entities("FCA_PRODUCT");
|
|
282
|
+
let batch = olaya_core::OntologyBatchBuilder::from_entities(&entities)?;
|
|
283
|
+
```
|
|
284
|
+
|
|
285
|
+
### Node / TypeScript
|
|
286
|
+
|
|
287
|
+
```bash
|
|
288
|
+
cd packages/olaya-node && bun install && bun test && bun run build
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
```typescript
|
|
292
|
+
import { OntologyDAG, OLaya } from "leio-decision";
|
|
293
|
+
|
|
294
|
+
const dag = new OntologyDAG([
|
|
295
|
+
{ id: 0, curie: "INCIDENT:DATA", label: "Data & Storage Incident", parents: [], disjointWith: [1], synonyms: ["database"] },
|
|
296
|
+
{ id: 1, curie: "INCIDENT:BILLING", label: "Billing Incident", parents: [], disjointWith: [0], synonyms: ["payment"] },
|
|
297
|
+
]);
|
|
298
|
+
|
|
299
|
+
const decision = await new OLaya(dag).systemOne("Payment webhook timeout on credit card charge");
|
|
300
|
+
decision.selectedConcept.label; // "Billing Incident"
|
|
301
|
+
decision.axiomsVerified; // true
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
### MCP servers
|
|
305
|
+
|
|
306
|
+
Both the Python and Node packages expose a Model Context Protocol server over
|
|
307
|
+
stdio. Neither is published to a registry yet; run them from a checkout:
|
|
308
|
+
|
|
309
|
+
```bash
|
|
310
|
+
.venv/bin/olaya-mcp # Python runtime
|
|
311
|
+
cd packages/olaya-node && bun run src/cli.ts # Node runtime
|
|
312
|
+
```
|
|
313
|
+
|
|
314
|
+
Tools: `olaya_ground` (concept grounding with lineage and `axioms_verified`),
|
|
315
|
+
`olaya_subsumes`, `olaya_choice`, `olaya_score`; the Python server also offers
|
|
316
|
+
`olaya_extract_and_ground` (JSON-LD / GTIN extraction then grounding). Bundled ontologies are the JSON files under
|
|
317
|
+
`src/olaya/data/` and `docs/public/ontologies/`; the resource descriptions state
|
|
318
|
+
the actual class counts. Confidence is entropy-based, not calibrated, unless the
|
|
319
|
+
Python server loads a checkpoint with a fitted temperature.
|
|
320
|
+
|
|
321
|
+
## TBox-bound workflows
|
|
322
|
+
|
|
323
|
+
The Python JSON workflow compiler and interpreter bind workflow and decision
|
|
324
|
+
types, option classes, terminal intent properties, and declared
|
|
325
|
+
infrastructure / application resources to a supplied ontology. It emits JSON-LD
|
|
326
|
+
execution records with TBox identity and hash plus decision provenance, and
|
|
327
|
+
`olaya.workflow_replay.replay_workflow` re-checks a recorded trace against the
|
|
328
|
+
compiled policy without model inference. See the
|
|
329
|
+
[version 1 contract](spec/workflow-dsl-proposal.md) and the
|
|
330
|
+
[storage example](examples/workflows/storage.json):
|
|
331
|
+
|
|
332
|
+
```bash
|
|
333
|
+
python examples/workflows/run_storage.py
|
|
334
|
+
```
|
|
335
|
+
|
|
336
|
+
Terminal intents are records, not actions: the workflow layer never dispatches
|
|
337
|
+
goods or calls infrastructure. Replay checks policy consistency; content hashes
|
|
338
|
+
are not signatures.
|
|
339
|
+
|
|
340
|
+
## HTTP edge demo
|
|
341
|
+
|
|
342
|
+
`POST /api/systemone` at <https://public-nine-jade-58.vercel.app/api/systemone>
|
|
343
|
+
runs [docs/public/api/systemone.js](docs/public/api/systemone.js): bounded
|
|
344
|
+
lexical matching over `choice`, `score` and `noul` with explicit abstention.
|
|
345
|
+
It does not load an ontology DAG, trained weights, or the projection kernel.
|
|
346
|
+
The contract and its 34 regression cases are in
|
|
347
|
+
[spec/systemone-reliability.md](spec/systemone-reliability.md) and
|
|
348
|
+
`tests/systemone-edge.test.mjs`.
|
|
349
|
+
|
|
350
|
+
## Claude Desktop extension
|
|
351
|
+
|
|
352
|
+
`extensions/claude-desktop` packages the Python runtime, the workflow tools and
|
|
353
|
+
three skills as an `.mcpb` for macOS Apple Silicon. The build script expects a
|
|
354
|
+
checkpoint at `models/olaya_decision_model.pt`; none is committed, so you must
|
|
355
|
+
train or supply one first. See
|
|
356
|
+
[extensions/claude-desktop/README.md](extensions/claude-desktop/README.md).
|
|
357
|
+
|
|
358
|
+
## OWL 2 RL coverage
|
|
359
|
+
|
|
360
|
+
`src/olaya/owl2_rl.py` forward-chains the W3C OWL 2 RL rules over the TBox and an
|
|
361
|
+
optional ABox of named individuals, to a fixpoint, counting every firing:
|
|
362
|
+
|
|
363
|
+
- Schema: `scm-cls`, `scm-sco`, `scm-eqc1/2`, `scm-dco`, `scm-int`, `scm-uni`,
|
|
364
|
+
`owl:disjointUnionOf`, `scm-spo`, `scm-dom1/2`, `scm-rng1/2`, unsatisfiability.
|
|
365
|
+
- Instance: `cax-sco`, `cax-eqc1/2`, `cax-dw`, `cls-int1`, `cls-nothing2`,
|
|
366
|
+
`prp-dom`, `prp-rng`, `prp-spo1`, `prp-spo2` (property chains), `prp-inv1/2`,
|
|
367
|
+
`prp-symp`, `prp-trp`, `prp-fp`, `prp-ifp`, `prp-pdw`, `prp-asyp`, `prp-irp`,
|
|
368
|
+
and the equality rules `eq-sym`, `eq-trans`, `eq-rep-s/o`, `eq-diff1`
|
|
369
|
+
(`owl:sameAs` is kept as an equivalence class of individuals).
|
|
370
|
+
- Not supported: restrictions (`someValuesFrom`, `allValuesFrom`, `hasValue`,
|
|
371
|
+
cardinality), `complementOf`, `oneOf`, `hasKey`, n-ary `AllDisjoint*` /
|
|
372
|
+
`AllDifferent` lists, negative property assertions, datatypes and literals.
|
|
373
|
+
|
|
374
|
+
Inconsistencies are returned as a list (`abox_consistency_report`), never raised.
|
|
375
|
+
The per-rule status table, with reasons, is in [spec/OWL2RL.md](spec/OWL2RL.md).
|
|
376
|
+
The closure feeds the projection kernel so that inferred subsumption and
|
|
377
|
+
disjointness are enforced, not only asserted ones.
|
|
378
|
+
|
|
379
|
+
## Repository layout
|
|
380
|
+
|
|
381
|
+
```
|
|
382
|
+
olaya/
|
|
383
|
+
├── Cargo.toml # Rust workspace
|
|
384
|
+
├── pyproject.toml # Python package + ruff/pytest config
|
|
385
|
+
├── spec/
|
|
386
|
+
│ ├── AXIOMS.md # The shared axiom-projection kernel, normative
|
|
387
|
+
│ ├── axioms_golden.json # Test vectors every runtime must reproduce
|
|
388
|
+
│ ├── workflow-dsl-proposal.md, workflow-vocabulary.jsonld
|
|
389
|
+
│ └── systemone-reliability.md, evaluation-integrity.md
|
|
390
|
+
├── crates/
|
|
391
|
+
│ ├── olaya-core/ # Arrow-native engine, router, axioms, optional ONNX scorer
|
|
392
|
+
│ ├── olaya-wasm/ # Browser build of router + kernel
|
|
393
|
+
│ ├── owl-fast-core/ # OWL / RDF / SKOS extraction on oxrdf, FCA engine
|
|
394
|
+
│ └── mycelia-ort-utils/ # ONNX Runtime session helpers
|
|
395
|
+
├── src/olaya/ # Python engine, model, training, calibration, workflows, MCP
|
|
396
|
+
├── packages/olaya-node/ # leio
|
|
397
|
+
├── benchmarks/ # OAEI runner, latency scripts, results/
|
|
398
|
+
├── tests/fixtures/oaei/ # Vendored OAEI Conference ontologies + reference alignments
|
|
399
|
+
├── models/*.arrow # Compiled ontology fixtures (no model weights)
|
|
400
|
+
└── extensions/claude-desktop/ # MCPB bundle and skills
|
|
401
|
+
```
|
|
402
|
+
|
|
403
|
+
## Development
|
|
404
|
+
|
|
405
|
+
CI (`.github/workflows/ci.yml`) runs on every pull request: `cargo fmt --check`,
|
|
406
|
+
`cargo clippy -D warnings` (with and without `onnx`), `cargo test`, a wasm32
|
|
407
|
+
build, `ruff check` + `ruff format --check`, `pytest`, a check that the golden
|
|
408
|
+
vectors are regenerated, `tsc --noEmit`, `bun test`, `bun run build`, and the
|
|
409
|
+
edge regression suite.
|
|
410
|
+
|
|
411
|
+
To change the axiom kernel, edit the Python reference, run
|
|
412
|
+
`python scripts/gen_axiom_golden.py`, and bring the other runtimes to green
|
|
413
|
+
against the regenerated file.
|
|
414
|
+
|
|
415
|
+
## Roadmap
|
|
416
|
+
|
|
417
|
+
1. **A labelled corpus.** The next accuracy gate is a frozen, human-labelled set
|
|
418
|
+
with full-ontology retrieval, no-answer cases, and source-document separation.
|
|
419
|
+
Until it exists, every quality claim here says "lexical".
|
|
420
|
+
2. **A committed checkpoint.** Train the Python head on that corpus, commit the
|
|
421
|
+
checkpoint via a release asset, and report lexical versus neural on the same
|
|
422
|
+
split with the same latency budget.
|
|
423
|
+
3. **OAEI beyond lexical.** Add structural and definition-based matching to the
|
|
424
|
+
Conference matcher and re-run the same harness.
|
|
425
|
+
4. **Runtime convergence.** One scorer implementation generated for Rust, WASM
|
|
426
|
+
and TypeScript instead of parallel copies.
|
|
427
|
+
|
|
428
|
+
## License
|
|
429
|
+
|
|
430
|
+
MIT. Vendored OAEI data retains its upstream terms; see
|
|
431
|
+
`tests/fixtures/oaei/README.md`.
|
|
432
|
+
|
|
433
|
+
## Release builds
|
|
434
|
+
|
|
435
|
+
Release builds are opt-in: include `[release]` in a commit pushed to `main`, or
|
|
436
|
+
dispatch the Release workflow with `build_release=true`. See
|
|
437
|
+
[release instructions](docs/release.md) for Python wheels, npm packages,
|
|
438
|
+
complete Claude Desktop bundles, and Keychain signing.
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
olaya/__init__.py,sha256=2ikwy60EVhkvR0K2vfZfZDGWvNZCsusRKkQDJjWlVZo,1958
|
|
2
|
+
olaya/axioms.py,sha256=50MQW_f88RlL2e0Zhn8tqpp0qu7iwGNTCqpitVvgBb8,15632
|
|
3
|
+
olaya/calibration.py,sha256=_r5qDXKfnn4kU7LkrIotlT-mHBwZfsH9T7-LjaaO6ok,6820
|
|
4
|
+
olaya/candidate_router.py,sha256=K46F1hsNAm3UTrXDctRV3wizD7TcQlQXCURE05FOW3o,5064
|
|
5
|
+
olaya/category_theory.py,sha256=7J4QWLYT7FSSrXBYIveFt2iMbDN3T5O4YH2df1dGviM,13275
|
|
6
|
+
olaya/data_gen.py,sha256=XT0Fmm70pN3lZsFn3ydQ4p9BwcrpU-BAh1SB-dQHPI4,13564
|
|
7
|
+
olaya/distilled_head.py,sha256=OjKSPQsgyQGU9WkQ3WKujuo9Id-ilYFzQdn01YKFQAs,19893
|
|
8
|
+
olaya/embeddings.py,sha256=f1qe9ijX59nIOl38S-64LmV03cF4FyaDEdedd-YPDqU,12138
|
|
9
|
+
olaya/engine.py,sha256=22eb2y2PNAKO5BatC542EnuJvn6-PTFnBJ1SBtoOUVo,24790
|
|
10
|
+
olaya/evaluation.py,sha256=dRrU2XfXE4a4Ch0jE1-4Byze0-GyLdPM2VTTMPqRWeM,4719
|
|
11
|
+
olaya/hard_negative_sampler.py,sha256=JrImFQhRcJuXnPw-DrrmkgwhMnNG0p-rCKU6bbXLBjk,4309
|
|
12
|
+
olaya/model.py,sha256=c8Tu7fKZU7nSfADUQXnqUbAc27PPSjjH5zXsTyrmQbM,6172
|
|
13
|
+
olaya/ontology.py,sha256=MO-OA6j6h0gnoe3DXDjxI96n4BpeDdP6431W7ZGCcV8,27953
|
|
14
|
+
olaya/owl2_rl.py,sha256=GLjisHoYj-h16RUu9cRJPk8SsDmaRWGQuWOqdbBZiGw,27443
|
|
15
|
+
olaya/pii.py,sha256=Pk6UKEPQQ534KeP530fmp47qQqXK--iKEcG8yHRma5A,15641
|
|
16
|
+
olaya/real_abox.py,sha256=q2687bb1peW-5WGorlDAEDicdbHMbEQAW-RoJpOUL5s,8692
|
|
17
|
+
olaya/real_ontology_train.py,sha256=GMHj8mgB5l8g1o2yj3O_Or35ByrALUnd6jD-d_uywBU,23411
|
|
18
|
+
olaya/sequence.py,sha256=-aa_h1xDzH6KtqL-7ZkovoHA5a1PCE5imcnZo9vHajE,4546
|
|
19
|
+
olaya/tokenizer.py,sha256=866G92zQ0UsfJCftAMQfP_LWVFHbcClk019wehrrPuo,976
|
|
20
|
+
olaya/train.py,sha256=0Rk1g5gBqXduAyBsIzq567yXED2HS4wStdi_Nr6N1mo,24239
|
|
21
|
+
olaya/webmcp.py,sha256=TUDug72wFXtU3GNWgLHYGGTLFKy-YBMOMlfa96Y0mKo,42152
|
|
22
|
+
olaya/workflow.py,sha256=huchByjsWPLHRlyp2VcYr92L2URxROj-iOrT5I7b3tA,23016
|
|
23
|
+
olaya/workflow_jsonld.py,sha256=1p0tuJHpjLu79Cn8CQrJUDZf3R0z-kSQtxF_NNr7EO0,11611
|
|
24
|
+
olaya/workflow_replay.py,sha256=Xqe3Yzjzvwa6PL0FM8fxBklwb5EjJYBnvp5IagJ9ZXU,10407
|
|
25
|
+
olaya/benchmarks/__init__.py,sha256=7EPMhnDJwnBgEut8TOH7X7hjqXWYzb4WyWioQkAAozs,1134
|
|
26
|
+
olaya/benchmarks/analysis.py,sha256=595iCxenN28DGX8yDfSu354kMmhiGMlfb3Nks5RHxnY,14576
|
|
27
|
+
olaya/benchmarks/features.py,sha256=BQR-cSwIMHDJZPRTOtLQBLLQbZtk7jxPtIkab-n6lxs,21218
|
|
28
|
+
olaya/benchmarks/learned.py,sha256=jxU_KAsXWZHbrfvbMdjo8AtvLBEfU4kWd960PbRBv18,30543
|
|
29
|
+
olaya/benchmarks/llm_judge.py,sha256=77fyGPbF3kC9p9-W86TMZyKWgPB2y5fJYb4oFp44tJI,34536
|
|
30
|
+
olaya/benchmarks/oaei.py,sha256=8i6Ck1oCzA0xqftpKV4YJ7TqDNtdQxfqfcmoRxPYhzU,67264
|
|
31
|
+
olaya/benchmarks/reranker.py,sha256=UUdhfqzd0G1t4yycAUpcthOVE22FIGNoU78IdmPG5uU,25443
|
|
32
|
+
olaya/benchmarks/structural.py,sha256=ztr4eDovOY5OP7VgWzheJtR1BpH2nVdoxUxZOa9DgiI,21037
|
|
33
|
+
olaya/data/bfo_ontology.json,sha256=zpiW8nTwMgBfhxzXPrfHxyj-RRGBK3UVRKP7eJBsKEs,12482
|
|
34
|
+
olaya/data/doid_ontology.json,sha256=DSjvi96yOdSG_dGqa3FION_8KoTFlU62IMO8hbInfXY,314595
|
|
35
|
+
olaya/data/gs1_ontology.json,sha256=3nbbOeuyMyRl1UQIh7X5Fn2sQ7f7ud26Pw9QRtKyQ_g,21596
|
|
36
|
+
olaya/data/incident_ontology.json,sha256=wvwX8j0tThDDrL4saAH6IDagER4BR-MSK58BI853bpM,4505
|
|
37
|
+
olaya/data/leio_ontology.json,sha256=RP1wqvBqPm68b_DdDJovD24tIyH9q2qDlk3xCrBxTsk,16401
|
|
38
|
+
olaya/data/schemaorg_ontology.json,sha256=q8URLTErftg2I0V4m3EcCvh2p6BATEyEqztJIGClLxQ,313223
|
|
39
|
+
olaya/data/skos_ontology.json,sha256=qN9-WSntrUjKAggjH_NsZhdu3eNwG91ZE6YhvE45FF0,1585
|
|
40
|
+
leio_decision-0.3.2.dist-info/METADATA,sha256=A0SUZeB1aJ2NI44Er76Lmpf-3Z8IVlZ7bjf_0ujHbus,22884
|
|
41
|
+
leio_decision-0.3.2.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
42
|
+
leio_decision-0.3.2.dist-info/entry_points.txt,sha256=afdmdCullFPdRhtl1FUJ8UTOaUEz_xN1nwm_Ta65DZo,162
|
|
43
|
+
leio_decision-0.3.2.dist-info/RECORD,,
|
olaya/__init__.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""O-Laya: Ontology-Grounded System 1 Decision Model."""
|
|
2
|
+
|
|
3
|
+
from olaya.axioms import calibrate_and_project_distribution, enforce_subsumption_monotonicity
|
|
4
|
+
from olaya.candidate_router import CandidateRouter
|
|
5
|
+
from olaya.ontology import Concept, Individual, ObjectProperty, Ontology, create_incident_ontology
|
|
6
|
+
from olaya.owl2_rl import OWL2RLReasoner, abox_consistency_report
|
|
7
|
+
from olaya.pii import ObfuscationMode, ObfuscationResult, PIIEntity, PIIObfuscator
|
|
8
|
+
from olaya.real_abox import RealABoxFact, load_atlas_real_abox
|
|
9
|
+
from olaya.sequence import SequenceBundle, build_ontology_sequence
|
|
10
|
+
from olaya.workflow import CompiledWorkflow, compile_workflow, run_workflow
|
|
11
|
+
from olaya.workflow_replay import replay_workflow
|
|
12
|
+
|
|
13
|
+
try:
|
|
14
|
+
from olaya.calibration import CalibrationMetrics, compute_ece, fit_temperature
|
|
15
|
+
from olaya.engine import (
|
|
16
|
+
ChoiceAnswer,
|
|
17
|
+
ConceptAnswer,
|
|
18
|
+
NoulAnswer,
|
|
19
|
+
OLaya,
|
|
20
|
+
ScoreAnswer,
|
|
21
|
+
SubsumesAnswer,
|
|
22
|
+
SystemOneResult,
|
|
23
|
+
)
|
|
24
|
+
from olaya.model import OntologyDecisionHead, OntologyDecisionModel
|
|
25
|
+
from olaya.train import train_olaya
|
|
26
|
+
except ImportError:
|
|
27
|
+
pass
|
|
28
|
+
|
|
29
|
+
__all__ = [
|
|
30
|
+
"ObfuscationMode",
|
|
31
|
+
"ObfuscationResult",
|
|
32
|
+
"PIIEntity",
|
|
33
|
+
"PIIObfuscator",
|
|
34
|
+
"CompiledWorkflow",
|
|
35
|
+
"compile_workflow",
|
|
36
|
+
"run_workflow",
|
|
37
|
+
"replay_workflow",
|
|
38
|
+
"Ontology",
|
|
39
|
+
"Concept",
|
|
40
|
+
"Individual",
|
|
41
|
+
"ObjectProperty",
|
|
42
|
+
"OWL2RLReasoner",
|
|
43
|
+
"abox_consistency_report",
|
|
44
|
+
"create_incident_ontology",
|
|
45
|
+
"CandidateRouter",
|
|
46
|
+
"OntologyDecisionModel",
|
|
47
|
+
"OntologyDecisionHead",
|
|
48
|
+
"OLaya",
|
|
49
|
+
"SystemOneResult",
|
|
50
|
+
"ConceptAnswer",
|
|
51
|
+
"SubsumesAnswer",
|
|
52
|
+
"ChoiceAnswer",
|
|
53
|
+
"ScoreAnswer",
|
|
54
|
+
"NoulAnswer",
|
|
55
|
+
"calibrate_and_project_distribution",
|
|
56
|
+
"enforce_subsumption_monotonicity",
|
|
57
|
+
"build_ontology_sequence",
|
|
58
|
+
"SequenceBundle",
|
|
59
|
+
"train_olaya",
|
|
60
|
+
"compute_ece",
|
|
61
|
+
"fit_temperature",
|
|
62
|
+
"CalibrationMetrics",
|
|
63
|
+
"RealABoxFact",
|
|
64
|
+
"load_atlas_real_abox",
|
|
65
|
+
]
|