nomox-semantics 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nomox_semantics-0.1.0/.github/workflows/publish.yml +43 -0
- nomox_semantics-0.1.0/.gitignore +80 -0
- nomox_semantics-0.1.0/PKG-INFO +78 -0
- nomox_semantics-0.1.0/README.md +58 -0
- nomox_semantics-0.1.0/docs/README.md +71 -0
- nomox_semantics-0.1.0/docs/catalog.md +83 -0
- nomox_semantics-0.1.0/docs/gaps.md +122 -0
- nomox_semantics-0.1.0/docs/glossary.md +77 -0
- nomox_semantics-0.1.0/docs/linkage.md +170 -0
- nomox_semantics-0.1.0/docs/metrics.md +154 -0
- nomox_semantics-0.1.0/docs/ontology.md +297 -0
- nomox_semantics-0.1.0/docs/overview.md +77 -0
- nomox_semantics-0.1.0/docs/physical-layer.md +69 -0
- nomox_semantics-0.1.0/docs/queries.md +115 -0
- nomox_semantics-0.1.0/docs/usage.md +187 -0
- nomox_semantics-0.1.0/examples/01_quickstart.py +174 -0
- nomox_semantics-0.1.0/examples/02_advanced_ontology.py +185 -0
- nomox_semantics-0.1.0/examples/03_metrics_and_queries.py +204 -0
- nomox_semantics-0.1.0/examples/README.md +19 -0
- nomox_semantics-0.1.0/pyproject.toml +51 -0
- nomox_semantics-0.1.0/src/nomox_semantics/__init__.py +89 -0
- nomox_semantics-0.1.0/src/nomox_semantics/catalog.py +126 -0
- nomox_semantics-0.1.0/src/nomox_semantics/glossary.py +53 -0
- nomox_semantics-0.1.0/src/nomox_semantics/metric.py +176 -0
- nomox_semantics-0.1.0/src/nomox_semantics/physical.py +175 -0
- nomox_semantics-0.1.0/src/nomox_semantics/query.py +188 -0
- nomox_semantics-0.1.0/src/nomox_semantics/semantic.py +338 -0
- nomox_semantics-0.1.0/tests/conftest.py +336 -0
- nomox_semantics-0.1.0/tests/test_catalog.py +163 -0
- nomox_semantics-0.1.0/tests/test_glossary.py +46 -0
- nomox_semantics-0.1.0/tests/test_metric.py +108 -0
- nomox_semantics-0.1.0/tests/test_physical.py +115 -0
- nomox_semantics-0.1.0/tests/test_query.py +112 -0
- nomox_semantics-0.1.0/tests/test_semantic.py +207 -0
- nomox_semantics-0.1.0/tests/test_serialization.py +96 -0
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
build:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
steps:
|
|
11
|
+
- uses: actions/checkout@v4
|
|
12
|
+
|
|
13
|
+
- uses: actions/setup-python@v5
|
|
14
|
+
with:
|
|
15
|
+
python-version: "3.12"
|
|
16
|
+
|
|
17
|
+
- name: Install build
|
|
18
|
+
run: pip install build
|
|
19
|
+
|
|
20
|
+
- name: Build package
|
|
21
|
+
run: python -m build
|
|
22
|
+
|
|
23
|
+
- name: Upload dist as artifact
|
|
24
|
+
uses: actions/upload-artifact@v4
|
|
25
|
+
with:
|
|
26
|
+
name: dist
|
|
27
|
+
path: dist/
|
|
28
|
+
|
|
29
|
+
publish:
|
|
30
|
+
needs: build
|
|
31
|
+
runs-on: ubuntu-latest
|
|
32
|
+
environment: pypi
|
|
33
|
+
permissions:
|
|
34
|
+
id-token: write # Required for Trusted Publishing (OIDC)
|
|
35
|
+
steps:
|
|
36
|
+
- name: Download dist artifact
|
|
37
|
+
uses: actions/download-artifact@v4
|
|
38
|
+
with:
|
|
39
|
+
name: dist
|
|
40
|
+
path: dist/
|
|
41
|
+
|
|
42
|
+
- name: Publish to PyPI
|
|
43
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# Byte-compiled / optimised / DLL files
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
|
|
6
|
+
# C extensions
|
|
7
|
+
*.so
|
|
8
|
+
|
|
9
|
+
# Distribution / packaging
|
|
10
|
+
.Python
|
|
11
|
+
build/
|
|
12
|
+
dist/
|
|
13
|
+
wheels/
|
|
14
|
+
sdist/
|
|
15
|
+
develop-eggs/
|
|
16
|
+
eggs/
|
|
17
|
+
parts/
|
|
18
|
+
var/
|
|
19
|
+
*.egg
|
|
20
|
+
*.egg-info/
|
|
21
|
+
.installed.cfg
|
|
22
|
+
*.whl
|
|
23
|
+
*.tar.gz
|
|
24
|
+
|
|
25
|
+
# Installer logs
|
|
26
|
+
pip-log.txt
|
|
27
|
+
pip-delete-this-directory.txt
|
|
28
|
+
|
|
29
|
+
# Virtual environments
|
|
30
|
+
.venv/
|
|
31
|
+
venv/
|
|
32
|
+
env/
|
|
33
|
+
ENV/
|
|
34
|
+
|
|
35
|
+
# Testing / coverage
|
|
36
|
+
.pytest_cache/
|
|
37
|
+
.tox/
|
|
38
|
+
.nox/
|
|
39
|
+
.coverage
|
|
40
|
+
.coverage.*
|
|
41
|
+
coverage.xml
|
|
42
|
+
htmlcov/
|
|
43
|
+
nosetests.xml
|
|
44
|
+
.cache
|
|
45
|
+
|
|
46
|
+
# Type checking / linting caches
|
|
47
|
+
.mypy_cache/
|
|
48
|
+
.pyright_cache/
|
|
49
|
+
.pytype/
|
|
50
|
+
.ruff_cache/
|
|
51
|
+
.dmypy.json
|
|
52
|
+
|
|
53
|
+
# Documentation
|
|
54
|
+
docs/_build/
|
|
55
|
+
|
|
56
|
+
# Jupyter
|
|
57
|
+
.ipynb_checkpoints/
|
|
58
|
+
|
|
59
|
+
# IDE / editors
|
|
60
|
+
.vscode/
|
|
61
|
+
.idea/
|
|
62
|
+
*.swp
|
|
63
|
+
*.swo
|
|
64
|
+
*~
|
|
65
|
+
.project
|
|
66
|
+
.pydevproject
|
|
67
|
+
.ropeproject
|
|
68
|
+
|
|
69
|
+
# OS
|
|
70
|
+
.DS_Store
|
|
71
|
+
Thumbs.db
|
|
72
|
+
desktop.ini
|
|
73
|
+
|
|
74
|
+
# Logs
|
|
75
|
+
*.log
|
|
76
|
+
|
|
77
|
+
# Secrets / local config
|
|
78
|
+
.env
|
|
79
|
+
.env.*
|
|
80
|
+
!.env.example
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: nomox-semantics
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Semantic data model for LLM-consumable data catalog - shared contract across Nomox services.
|
|
5
|
+
Author-email: nomox <admin@get-nomox.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Classifier: Development Status :: 3 - Alpha
|
|
8
|
+
Classifier: Intended Audience :: Developers
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Requires-Python: >=3.11
|
|
13
|
+
Requires-Dist: pydantic<3,>=2.7
|
|
14
|
+
Provides-Extra: dev
|
|
15
|
+
Requires-Dist: mypy>=1; extra == 'dev'
|
|
16
|
+
Requires-Dist: pytest-cov>=4; extra == 'dev'
|
|
17
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
18
|
+
Requires-Dist: ruff>=0.1; extra == 'dev'
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
# Nomox Semantic Model
|
|
22
|
+
|
|
23
|
+
The shared knowledge contract between every Nomox service. Describes what data exists (physical layer) and what it means (ontology), and ties the two together so an agent can move from a natural-language concept all the way down to a specific column in a specific table.
|
|
24
|
+
|
|
25
|
+
## Installation
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install git+https://${GITHUB_TOKEN}@github.com/Nomox-ai/semantics-model.git
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Pin to a tag for reproducible installs:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install git+https://${GITHUB_TOKEN}@github.com/Nomox-ai/semantics-model.git@v0.1.0
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Quickstart
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from uuid import uuid4
|
|
41
|
+
from nomox_semantics import (
|
|
42
|
+
SemanticCatalog, DataSource, SourceType, Table, Column, ColumnDataType,
|
|
43
|
+
Entity, EntityAttribute, EntityTableMapping, AttributeRole, SemanticType,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
# 1. Physical layer — what exists in the database
|
|
47
|
+
src = DataSource(slug="prod", name="Prod", source_type=SourceType.POSTGRES)
|
|
48
|
+
users = Table(source_id=src.id, schema_name="public", table_name="users")
|
|
49
|
+
ltv = Column(table_id=users.id, source_id=src.id, column_name="ltv_cents",
|
|
50
|
+
ordinal_position=1, data_type=ColumnDataType.BIGINT, raw_data_type="bigint")
|
|
51
|
+
|
|
52
|
+
# 2. Semantic layer — what the data means in business terms
|
|
53
|
+
customer = Entity(
|
|
54
|
+
slug="customer", name="Customer", plural_name="Customers",
|
|
55
|
+
aliases=["user", "account holder"], # for natural-language matching
|
|
56
|
+
table_mappings=[EntityTableMapping(table_id=users.id, source_id=src.id)], # entity -> table
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# An EntityAttribute binds a business field to a physical column
|
|
60
|
+
ltv_attr = EntityAttribute(
|
|
61
|
+
entity_id=customer.id, column_id=ltv.id, source_id=src.id,
|
|
62
|
+
name="lifetime_value", display_name="Lifetime Value",
|
|
63
|
+
role=AttributeRole.MEASURE, default_aggregation="SUM", # tells the query engine it's aggregatable
|
|
64
|
+
semantic_type=SemanticType.CURRENCY, unit="EUR", # values are EUR, not raw cents
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# 3. Assemble the catalog — the root object every Nomox service consumes
|
|
68
|
+
catalog = SemanticCatalog(organisation_id=uuid4(), name="Acme")
|
|
69
|
+
catalog.sources[src.id] = src
|
|
70
|
+
catalog.tables[users.id] = users
|
|
71
|
+
catalog.columns[ltv.id] = ltv
|
|
72
|
+
catalog.entities[customer.id] = customer
|
|
73
|
+
catalog.attributes[ltv_attr.id] = ltv_attr
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Documentation
|
|
77
|
+
|
|
78
|
+
See [docs/](docs/) for the full design documentation.
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# Nomox Semantic Model
|
|
2
|
+
|
|
3
|
+
The shared knowledge contract between every Nomox service. Describes what data exists (physical layer) and what it means (ontology), and ties the two together so an agent can move from a natural-language concept all the way down to a specific column in a specific table.
|
|
4
|
+
|
|
5
|
+
## Installation
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
pip install git+https://${GITHUB_TOKEN}@github.com/Nomox-ai/semantics-model.git
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
Pin to a tag for reproducible installs:
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
pip install git+https://${GITHUB_TOKEN}@github.com/Nomox-ai/semantics-model.git@v0.1.0
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
## Quickstart
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
from uuid import uuid4
|
|
21
|
+
from nomox_semantics import (
|
|
22
|
+
SemanticCatalog, DataSource, SourceType, Table, Column, ColumnDataType,
|
|
23
|
+
Entity, EntityAttribute, EntityTableMapping, AttributeRole, SemanticType,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
# 1. Physical layer — what exists in the database
|
|
27
|
+
src = DataSource(slug="prod", name="Prod", source_type=SourceType.POSTGRES)
|
|
28
|
+
users = Table(source_id=src.id, schema_name="public", table_name="users")
|
|
29
|
+
ltv = Column(table_id=users.id, source_id=src.id, column_name="ltv_cents",
|
|
30
|
+
ordinal_position=1, data_type=ColumnDataType.BIGINT, raw_data_type="bigint")
|
|
31
|
+
|
|
32
|
+
# 2. Semantic layer — what the data means in business terms
|
|
33
|
+
customer = Entity(
|
|
34
|
+
slug="customer", name="Customer", plural_name="Customers",
|
|
35
|
+
aliases=["user", "account holder"], # for natural-language matching
|
|
36
|
+
table_mappings=[EntityTableMapping(table_id=users.id, source_id=src.id)], # entity -> table
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
# An EntityAttribute binds a business field to a physical column
|
|
40
|
+
ltv_attr = EntityAttribute(
|
|
41
|
+
entity_id=customer.id, column_id=ltv.id, source_id=src.id,
|
|
42
|
+
name="lifetime_value", display_name="Lifetime Value",
|
|
43
|
+
role=AttributeRole.MEASURE, default_aggregation="SUM", # tells the query engine it's aggregatable
|
|
44
|
+
semantic_type=SemanticType.CURRENCY, unit="EUR", # values are EUR, not raw cents
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
# 3. Assemble the catalog — the root object every Nomox service consumes
|
|
48
|
+
catalog = SemanticCatalog(organisation_id=uuid4(), name="Acme")
|
|
49
|
+
catalog.sources[src.id] = src
|
|
50
|
+
catalog.tables[users.id] = users
|
|
51
|
+
catalog.columns[ltv.id] = ltv
|
|
52
|
+
catalog.entities[customer.id] = customer
|
|
53
|
+
catalog.attributes[ltv_attr.id] = ltv_attr
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Documentation
|
|
57
|
+
|
|
58
|
+
See [docs/](docs/) for the full design documentation.
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
# Nomox Semantic Model — Documentation
|
|
2
|
+
|
|
3
|
+
The Nomox semantic model is the shared knowledge contract between every Nomox service. It describes **what data exists** (the physical layer) and **what it means** (the ontology), and ties the two together so that an agent answering a business question can move from a natural-language concept all the way down to a specific column in a specific table.
|
|
4
|
+
|
|
5
|
+
This documentation explains the model, the design decisions behind it, and where it currently has gaps.
|
|
6
|
+
|
|
7
|
+
## Reading order
|
|
8
|
+
|
|
9
|
+
1. [usage.md](usage.md) — runnable walkthrough; start here to see the model in action
|
|
10
|
+
2. [overview.md](overview.md) — the two-layer architecture and why it exists
|
|
11
|
+
3. [physical-layer.md](physical-layer.md) — `DataSource`, `Table`, `Column`, foreign keys
|
|
12
|
+
4. [ontology.md](ontology.md) — `BusinessDomain`, `Entity`, `EntityAttribute`, `Relationship`
|
|
13
|
+
5. [linkage.md](linkage.md) — how ontology concepts bind to physical objects
|
|
14
|
+
6. [metrics.md](metrics.md) — first-class business numbers
|
|
15
|
+
7. [queries.md](queries.md) — saved queries and reasoning traces
|
|
16
|
+
8. [glossary.md](glossary.md) — business vocabulary that doesn't fit elsewhere
|
|
17
|
+
9. [catalog.md](catalog.md) — the root `SemanticCatalog` object
|
|
18
|
+
10. [gaps.md](gaps.md) — known limitations and recommended next steps
|
|
19
|
+
|
|
20
|
+
Runnable examples live in [examples/](../examples/); tests in [tests/](../tests/).
|
|
21
|
+
|
|
22
|
+
## At a glance
|
|
23
|
+
|
|
24
|
+
```mermaid
|
|
25
|
+
flowchart TB
|
|
26
|
+
classDef root fill:#1e293b,stroke:#0f172a,color:#f8fafc;
|
|
27
|
+
classDef phys fill:#dbeafe,stroke:#1e40af,color:#1e3a8a;
|
|
28
|
+
classDef sem fill:#fef3c7,stroke:#92400e,color:#78350f;
|
|
29
|
+
classDef met fill:#dcfce7,stroke:#166534,color:#14532d;
|
|
30
|
+
classDef qry fill:#ede9fe,stroke:#6d28d9,color:#4c1d95;
|
|
31
|
+
|
|
32
|
+
SC[SemanticCatalog<br/>one per organisation]:::root
|
|
33
|
+
|
|
34
|
+
subgraph Physical["Physical layer"]
|
|
35
|
+
direction TB
|
|
36
|
+
DS[DataSource]:::phys --> T[Table]:::phys --> C[Column]:::phys
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
subgraph Semantic["Semantic layer (ontology)"]
|
|
40
|
+
direction TB
|
|
41
|
+
BD[BusinessDomain]:::sem --> E[Entity]:::sem
|
|
42
|
+
E --> EA[EntityAttribute]:::sem
|
|
43
|
+
E -. via Relationship .-> E
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
subgraph MetGlos["Metrics & glossary"]
|
|
47
|
+
direction TB
|
|
48
|
+
M[Metric]:::met
|
|
49
|
+
G[GlossaryTerm]:::met
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
subgraph Q["Queries"]
|
|
53
|
+
direction TB
|
|
54
|
+
SQ[SavedQuery]:::qry --> QT[QueryTrace]:::qry
|
|
55
|
+
D[Dashboard]:::qry
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
SC --> Physical
|
|
59
|
+
SC --> Semantic
|
|
60
|
+
SC --> MetGlos
|
|
61
|
+
SC --> Q
|
|
62
|
+
|
|
63
|
+
EA -- column_id --> C
|
|
64
|
+
E -- table_mappings --> T
|
|
65
|
+
M -- column_refs --> C
|
|
66
|
+
M -- entities_referenced --> E
|
|
67
|
+
G -. related_* .-> E
|
|
68
|
+
G -. related_* .-> M
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Everything an agent sees — entities, attributes, metrics, glossary terms — ultimately resolves to a column in a real source. The ontology is the **interpretation**; the physical layer is the **substrate**.
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# `SemanticCatalog` — the root object
|
|
2
|
+
|
|
3
|
+
The `SemanticCatalog` is the root of everything. One catalog per organisation; it holds the complete knowledge graph: physical schema, semantic layer, metrics, glossary, and saved queries.
|
|
4
|
+
|
|
5
|
+
Source: [src/nomox_semantics/catalog.py](../src/nomox_semantics/catalog.py)
|
|
6
|
+
|
|
7
|
+
## Shape
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
class SemanticCatalog(BaseModel):
|
|
11
|
+
id: UUID
|
|
12
|
+
organisation_id: UUID
|
|
13
|
+
name: str
|
|
14
|
+
|
|
15
|
+
# Physical layer
|
|
16
|
+
sources: dict[UUID, DataSource]
|
|
17
|
+
tables: dict[UUID, Table]
|
|
18
|
+
columns: dict[UUID, Column]
|
|
19
|
+
|
|
20
|
+
# Semantic layer
|
|
21
|
+
domains: dict[UUID, BusinessDomain]
|
|
22
|
+
entities: dict[UUID, Entity]
|
|
23
|
+
attributes: dict[UUID, EntityAttribute]
|
|
24
|
+
relationships: dict[UUID, Relationship]
|
|
25
|
+
metrics: dict[UUID, Metric]
|
|
26
|
+
glossary_terms: dict[UUID, GlossaryTerm]
|
|
27
|
+
|
|
28
|
+
# Queries
|
|
29
|
+
saved_queries: dict[UUID, SavedQuery]
|
|
30
|
+
dashboards: dict[UUID, Dashboard]
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Everything is keyed by UUID in a flat dict for O(1) lookups. References between objects are always by UUID, never by nested object — this keeps serialisation deterministic and avoids cycles.
|
|
34
|
+
|
|
35
|
+
## Persistence boundary
|
|
36
|
+
|
|
37
|
+
The catalog is an **in-memory representation**. Persistence is the Catalog service's responsibility — it stores the catalog in whatever shape its database prefers (typically normalised tables, one per dict). Other services hydrate a `SemanticCatalog` from the Catalog service's API and treat it as read-only or do diffs and submit updates.
|
|
38
|
+
|
|
39
|
+
## Lookup helpers
|
|
40
|
+
|
|
41
|
+
The catalog exposes common access patterns as methods so services don't have to scan the dicts directly.
|
|
42
|
+
|
|
43
|
+
### Physical layer
|
|
44
|
+
|
|
45
|
+
- `tables_for_source(source_id)` — every table in one source
|
|
46
|
+
- `columns_for_table(table_id)` — columns of one table, sorted by `ordinal_position`
|
|
47
|
+
|
|
48
|
+
### Semantic layer
|
|
49
|
+
|
|
50
|
+
- `entity_by_slug(slug)` — `Entity | None`
|
|
51
|
+
- `metric_by_slug(slug)` — `Metric | None`
|
|
52
|
+
- `golden_metrics()` — every `is_golden and not is_deprecated` metric
|
|
53
|
+
- `attributes_for_entity(entity_id)`
|
|
54
|
+
- `relationships_for_entity(entity_id)` — both directions (source or target)
|
|
55
|
+
- `entities_in_domain(domain_id)`
|
|
56
|
+
- `metrics_in_domain(domain_id)`
|
|
57
|
+
|
|
58
|
+
### Entity hierarchy
|
|
59
|
+
|
|
60
|
+
- `child_entities(entity_id)` — direct subtypes (entities whose `parent_entity_id` matches)
|
|
61
|
+
- `descendant_entities(entity_id)` — all transitive subtypes, walking the hierarchy
|
|
62
|
+
|
|
63
|
+
### Glossary
|
|
64
|
+
|
|
65
|
+
- `glossary_term_by_slug(slug)` — `GlossaryTerm | None`
|
|
66
|
+
- `glossary_terms_in_domain(domain_id)`
|
|
67
|
+
|
|
68
|
+
### Queries
|
|
69
|
+
|
|
70
|
+
- `example_queries(limit=20)` — high-quality few-shot examples (`is_example=True`, `SUCCESS`)
|
|
71
|
+
|
|
72
|
+
## What the catalog does NOT do
|
|
73
|
+
|
|
74
|
+
- It does not enforce **referential integrity** across objects. Validation that every `EntityAttribute.column_id` points at a real column, that every `EntityTableMapping.table_id` exists, that every `Relationship.via_entity_id` resolves, etc., is the responsibility of the writer (the Semantic Indexer or the BI platform). See [gaps.md](gaps.md) for a `validate()` design.
|
|
75
|
+
- It does not enforce **alias uniqueness**. Two entities can share an alias; consumer code decides how to resolve ambiguity.
|
|
76
|
+
- It does not version itself. There's `updated_at` on individual objects, but no history of changes.
|
|
77
|
+
- It does not persist itself. That's the Catalog service.
|
|
78
|
+
|
|
79
|
+
`EntityAttribute` does enforce its own internal invariant — the column-XOR-expression rule — via a Pydantic `@model_validator`. That is the only structural validation done in-model.
|
|
80
|
+
|
|
81
|
+
## Missing helpers worth adding
|
|
82
|
+
|
|
83
|
+
The lookups above cover the common cases. Reverse lookups (column → entity, column → metrics, metric → saved queries) and alias-aware lookups (`entity_by_alias(name)` etc.) are open work — see [gaps.md](gaps.md) under "Catalog ergonomics."
|
|
@@ -0,0 +1,122 @@
|
|
|
1
|
+
# Gap analysis — what's missing
|
|
2
|
+
|
|
3
|
+
This document tracks what the model currently does not handle. Each item is tagged with severity:
|
|
4
|
+
|
|
5
|
+
- **🟠Gap** — meaningful capability missing for the stated goals
|
|
6
|
+
- **🟡 Nice-to-have** — useful but not blocking
|
|
7
|
+
- **✅ Closed** — previously identified gap, now addressed
|
|
8
|
+
|
|
9
|
+
The bugs section is at the bottom — they're all closed but kept for the record.
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
## ✅ Closed gaps
|
|
14
|
+
|
|
15
|
+
These were identified in earlier passes and are now implemented.
|
|
16
|
+
|
|
17
|
+
| Gap | How it's addressed |
|
|
18
|
+
| --- | --- |
|
|
19
|
+
| Synonyms / aliases on entities and attributes | `aliases: list[str]` on `Entity`, `EntityAttribute`, and `Metric` (alongside `Metric.abbreviation` for the single-canonical form). |
|
|
20
|
+
| Entity hierarchy / inheritance | `Entity.parent_entity_id: UUID \| None` for single inheritance, plus `SemanticCatalog.child_entities()` / `descendant_entities()` traversals. |
|
|
21
|
+
| Inverse relationship names | `Relationship.inverse_name: str \| None`. |
|
|
22
|
+
| M:N via associative entity | `Relationship.via_entity_id: UUID \| None` records the join table for M:N relationships. |
|
|
23
|
+
| Computed / derived attributes | `EntityAttribute.expression: str \| None`, with `column_id` and `source_id` now optional. A model validator enforces exactly one of (column-backed, expression). `is_computed` convenience property. |
|
|
24
|
+
| Semantic data type on attributes | New `SemanticType` enum (`CURRENCY`, `PERCENTAGE`, `DURATION`, `COUNT`, `RATIO`, `EMAIL`, `URL`, `PHONE`, `COUNTRY_CODE`, `LATITUDE`, `LONGITUDE`, `IP_ADDRESS`, `OTHER`) on `EntityAttribute`, plus `unit: str \| None`. |
|
|
25
|
+
| Glossary terms link to operationalising concepts | `GlossaryTerm.related_entity_ids` / `related_attribute_ids` / `related_metric_ids`, plus `aliases`. Wired into `SemanticCatalog.glossary_terms` with `glossary_term_by_slug` / `glossary_terms_in_domain` helpers. |
|
|
26
|
+
|
|
27
|
+
---
|
|
28
|
+
|
|
29
|
+
## Still open
|
|
30
|
+
|
|
31
|
+
### 🟡 No quality / freshness metadata on entities and attributes
|
|
32
|
+
|
|
33
|
+
`DataSource` has `last_synced_at` and `sync_status`, but there's no way to say "the `signup_date` attribute is reliable from 2022 onwards but messy before that" or "this entity is 80% populated." For a system that explicitly tries to avoid hallucination, surfacing data-quality caveats is high-value.
|
|
34
|
+
|
|
35
|
+
Recommendation: a lightweight `quality_caveat: str` on `Entity` and `EntityAttribute` would do, or a richer `DataQuality` sub-object if there's appetite.
|
|
36
|
+
|
|
37
|
+
### 🟡 Sample-value cache is capped at 50 known values
|
|
38
|
+
|
|
39
|
+
`EntityAttribute.known_values` caps at 50. That's fine for low-cardinality dimensions. For medium-cardinality ones (a few hundred to a few thousand — country codes, product SKUs, customer tiers), the cap forces the catalog to either truncate or not record values at all. The query engine then can't resolve user inputs to canonical values.
|
|
40
|
+
|
|
41
|
+
Recommendation: either raise the cap for confirmed attributes, or store a separate `ValueIndex` per-attribute keyed by source.
|
|
42
|
+
|
|
43
|
+
### 🟡 No access policies / row-level security model
|
|
44
|
+
|
|
45
|
+
`Column.is_sensitive: bool` is the only access primitive. There's no notion of "this entity is visible to Sales but not to Engineering," no row-level scoping, no audit hook beyond the saved query trail.
|
|
46
|
+
|
|
47
|
+
Recommendation: out of scope for v1, but worth a placeholder field (`access_policy_id: UUID | None`) on `Entity`, `EntityAttribute`, and `Metric` so a future policy system can attach without schema churn.
|
|
48
|
+
|
|
49
|
+
### 🟡 No versioning of definitions
|
|
50
|
+
|
|
51
|
+
Metric definitions change (the SQL gets refined, filters get added). The current model overwrites in place — `updated_at` records *when* but not *what*. A historical answer in `SavedQuery.result` might no longer match the current metric definition, and there's no way to know.
|
|
52
|
+
|
|
53
|
+
Recommendation: at minimum, snapshot the metric's `sql_expression` and `base_filters` into the `QueryTrace` at execution time. Long-term, consider a `metric_version` field that increments on definition change.
|
|
54
|
+
|
|
55
|
+
---
|
|
56
|
+
|
|
57
|
+
## Catalog ergonomics
|
|
58
|
+
|
|
59
|
+
Helpers that obvious access patterns will need.
|
|
60
|
+
|
|
61
|
+
### 🟡 Reverse lookups not provided
|
|
62
|
+
|
|
63
|
+
- "What entity (if any) owns this column?" — currently requires scanning every `EntityAttribute`.
|
|
64
|
+
- "What metrics reference this column?" — currently requires scanning every `Metric.column_refs`.
|
|
65
|
+
- "What saved queries reference this metric?" — currently requires scanning every `SavedQuery`.
|
|
66
|
+
- "What glossary terms link to this entity?" — same, scan `GlossaryTerm.related_entity_ids`.
|
|
67
|
+
|
|
68
|
+
These are the natural questions when assessing the blast radius of a schema change. Recommendation: add `entity_for_column(column_id)`, `metrics_for_column(column_id)`, `saved_queries_for_metric(metric_id)`, `glossary_terms_for_entity(entity_id)`.
|
|
69
|
+
|
|
70
|
+
### 🟡 No alias-aware lookup helpers
|
|
71
|
+
|
|
72
|
+
Now that `aliases` exists on `Entity`, `EntityAttribute`, and `Metric`, the obvious next step is lookup methods that consult them — `entity_by_alias(name)`, `metric_by_alias(name)`, case-insensitive, matching across slug / name / plural_name / abbreviation / aliases. Without these, every NL-matching consumer rebuilds the same search logic.
|
|
73
|
+
|
|
74
|
+
### 🟡 No bulk validation pass
|
|
75
|
+
|
|
76
|
+
The catalog doesn't expose a "validate me" method to check that:
|
|
77
|
+
|
|
78
|
+
- Every column-backed `EntityAttribute.column_id` resolves and `source_id` matches
|
|
79
|
+
- Every computed `EntityAttribute.expression` parses (best-effort)
|
|
80
|
+
- Every `EntityTableMapping.table_id` resolves
|
|
81
|
+
- Every `Relationship.source/target_attribute_id` resolves and the roles are sensible (FK on source, IDENTIFIER on target)
|
|
82
|
+
- Every `Relationship.via_entity_id` (if set) refers to a real entity, and the relationship is M:N
|
|
83
|
+
- Every `Entity.parent_entity_id` (if set) doesn't introduce a cycle
|
|
84
|
+
- Every `Metric.column_refs[*].column_id` resolves
|
|
85
|
+
- Every `Metric.date_attribute_id` resolves and has `role == TEMPORAL`
|
|
86
|
+
- Every `GlossaryTerm.related_*` id resolves
|
|
87
|
+
|
|
88
|
+
This would be a single pass returning a list of validation errors and is exactly the kind of safety net a shared contract needs.
|
|
89
|
+
|
|
90
|
+
---
|
|
91
|
+
|
|
92
|
+
## Documentation & repository hygiene
|
|
93
|
+
|
|
94
|
+
All three previously-open hygiene gaps are now closed — kept here as a record of what landed:
|
|
95
|
+
|
|
96
|
+
- ✅ **`docs/usage.md`** — entry-point quickstart at [docs/usage.md](usage.md).
|
|
97
|
+
- ✅ **`examples/` populated** — three runnable scripts: quickstart, advanced ontology, metrics + queries. See [examples/](../examples/).
|
|
98
|
+
- ✅ **Test suite** — six test files under [tests/](../tests/), 93 tests at ~99% line coverage. Includes round-trip serialisation tests for every model and validator-rejection tests for the column-XOR-expression rule on `EntityAttribute`.
|
|
99
|
+
|
|
100
|
+
---
|
|
101
|
+
|
|
102
|
+
## Suggested next steps
|
|
103
|
+
|
|
104
|
+
1. **Alias-aware lookups** on the catalog — they're now the cheapest way to convert the new `aliases` data into agent value.
|
|
105
|
+
2. **Reverse-lookup helpers** — small surface, large convenience.
|
|
106
|
+
3. **`validate()` pass** — single biggest safety win, especially before the catalog is shared across services.
|
|
107
|
+
4. **Round-trip tests** for every model — protects against future schema drift.
|
|
108
|
+
5. Tackle quality metadata, value-index scaling, access policies, and metric versioning as usage demands them.
|
|
109
|
+
|
|
110
|
+
---
|
|
111
|
+
|
|
112
|
+
## ✅ Bugs (resolved, kept for record)
|
|
113
|
+
|
|
114
|
+
These were live bugs at the start of this work. All five are now closed.
|
|
115
|
+
|
|
116
|
+
| Bug | Resolution |
|
|
117
|
+
| --- | --- |
|
|
118
|
+
| `__init__.py` imported from non-existent `.models.*` submodules | Imports rewritten to `.physical`, `.semantic`, etc.; package moved to `src/nomox_semantics/`. |
|
|
119
|
+
| `catalog.py` used bare `from metric import …` etc. | Rewritten to relative imports `from .metric import …`. |
|
|
120
|
+
| `pyproject.toml` declared `packages = ["nomox_semantics"]` with no such directory | Package layout moved to `src/nomox_semantics/`; pyproject points at `src/nomox_semantics`. |
|
|
121
|
+
| `GlossaryTerm` defined but not wired up | Exported from `__init__.py`, added as `glossary_terms` field on `SemanticCatalog`, plus lookup helpers. |
|
|
122
|
+
| `datetime.utcnow()` used throughout — deprecated on 3.12+ | Replaced everywhere with `lambda: datetime.now(timezone.utc)`. |
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
# Business glossary
|
|
2
|
+
|
|
3
|
+
The glossary captures internal vocabulary that doesn't map cleanly onto a specific entity, attribute, or metric — just words the company uses that an outsider, a new hire, or an AI agent might not understand.
|
|
4
|
+
|
|
5
|
+
Source: [src/nomox_semantics/glossary.py](../src/nomox_semantics/glossary.py)
|
|
6
|
+
|
|
7
|
+
## When to use a glossary term
|
|
8
|
+
|
|
9
|
+
Use it for **language**, not for **structure**. If the concept can be expressed as an entity, attribute, or metric, model it that way and skip the glossary. Glossary terms are the fallback for things that are real in conversation but don't sit cleanly on the graph — but a term can still *point at* the structural concepts that realise it (see "Linking to the rest of the catalog" below).
|
|
10
|
+
|
|
11
|
+
Good candidates:
|
|
12
|
+
|
|
13
|
+
- **"Whale"** — a customer who spends more than €10k/month. Useful as a term, but it's a fuzzy threshold applied across multiple metrics, not an entity in its own right.
|
|
14
|
+
- **"Activation"** — the moment a user completes their first meaningful action. Defined informally; the exact moment may vary across product lines.
|
|
15
|
+
- **"T0"** — the date a contract goes live, used as the reference point for all cohort calculations.
|
|
16
|
+
- **"North Star"** — the team's primary KPI, currently `weekly_active_paid_users` but historically others.
|
|
17
|
+
|
|
18
|
+
Bad candidates (model these elsewhere):
|
|
19
|
+
|
|
20
|
+
- "Customer" → that's an `Entity`.
|
|
21
|
+
- "MRR" → that's a `Metric` with `abbreviation="MRR"`.
|
|
22
|
+
- "signup_date" → that's an `EntityAttribute`.
|
|
23
|
+
|
|
24
|
+
## The shape
|
|
25
|
+
|
|
26
|
+
```python
|
|
27
|
+
class GlossaryTerm(BaseModel):
|
|
28
|
+
id: UUID
|
|
29
|
+
slug: str
|
|
30
|
+
name: str
|
|
31
|
+
aliases: list[str]
|
|
32
|
+
domain_id: UUID | None
|
|
33
|
+
business_context: str # what this term means in the company's context
|
|
34
|
+
|
|
35
|
+
related_entity_ids: list[UUID]
|
|
36
|
+
related_attribute_ids: list[UUID]
|
|
37
|
+
related_metric_ids: list[UUID]
|
|
38
|
+
|
|
39
|
+
created_at: datetime
|
|
40
|
+
updated_at: datetime
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
A term has:
|
|
44
|
+
|
|
45
|
+
- A `slug` and human `name`, plus optional `aliases` for near-synonyms
|
|
46
|
+
- An optional `domain_id` to scope it to a `BusinessDomain`
|
|
47
|
+
- A `business_context` narrative — the only required content field
|
|
48
|
+
- Optional `related_*` lists pointing at the structural concepts that operationalise the term
|
|
49
|
+
|
|
50
|
+
The narrative is the heart of a glossary term. Everything else is navigation.
|
|
51
|
+
|
|
52
|
+
## Linking to the rest of the catalog
|
|
53
|
+
|
|
54
|
+
The three `related_*` lists are how a glossary term becomes a navigable connective in the graph rather than a dangling note.
|
|
55
|
+
|
|
56
|
+
A `"Whale"` term might link to the `Customer` entity and the `monthly_spend` metric:
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
GlossaryTerm(
|
|
60
|
+
slug="whale",
|
|
61
|
+
name="Whale",
|
|
62
|
+
business_context="A customer with monthly spend ≥ €10,000. Used in account-tier discussions.",
|
|
63
|
+
related_entity_ids=[customer.id],
|
|
64
|
+
related_metric_ids=[monthly_spend_metric.id],
|
|
65
|
+
)
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
When an agent reads the term, it can follow the links to the `Customer` entity and propose answering "how many whales do we have right now?" by filtering customers on `monthly_spend >= 10000`. Without the links, the term is just text — useful for humans, opaque to the query engine.
|
|
69
|
+
|
|
70
|
+
Keep the links **descriptive**, not prescriptive. A glossary term saying *"Whale relates to Customer"* doesn't bake a hard SQL rule into the catalog; it just helps agents find the right starting point.
|
|
71
|
+
|
|
72
|
+
## Wiring
|
|
73
|
+
|
|
74
|
+
`GlossaryTerm` is exported from the package and stored on `SemanticCatalog.glossary_terms`. Lookup helpers:
|
|
75
|
+
|
|
76
|
+
- `catalog.glossary_term_by_slug(slug)`
|
|
77
|
+
- `catalog.glossary_terms_in_domain(domain_id)`
|