nomox-semantics 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nomox_semantics/__init__.py +89 -0
- nomox_semantics/catalog.py +126 -0
- nomox_semantics/glossary.py +53 -0
- nomox_semantics/metric.py +176 -0
- nomox_semantics/physical.py +175 -0
- nomox_semantics/query.py +188 -0
- nomox_semantics/semantic.py +338 -0
- nomox_semantics-0.1.0.dist-info/METADATA +78 -0
- nomox_semantics-0.1.0.dist-info/RECORD +10 -0
- nomox_semantics-0.1.0.dist-info/WHEEL +4 -0
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""
|
|
2
|
+
nomox_semantics — the Nomox knowledge model.
|
|
3
|
+
|
|
4
|
+
Defines everything the system can express about a company's data.
|
|
5
|
+
This package is the shared contract between all Nomox microservices.
|
|
6
|
+
|
|
7
|
+
Physical layer (what exists): DataSource, Table, Column
|
|
8
|
+
Semantic layer (what it means): BusinessDomain, Entity, EntityAttribute, Relationship
|
|
9
|
+
Metrics (authoritative numbers): Metric
|
|
10
|
+
Glossary (business vocabulary): GlossaryTerm
|
|
11
|
+
Queries (stored answers): SavedQuery, QueryTrace, Dashboard
|
|
12
|
+
Root: SemanticCatalog
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from .catalog import SemanticCatalog
|
|
16
|
+
from .glossary import GlossaryTerm
|
|
17
|
+
from .metric import (
|
|
18
|
+
GranularityHint,
|
|
19
|
+
Metric,
|
|
20
|
+
MetricColumnRef,
|
|
21
|
+
MetricFilter,
|
|
22
|
+
MetricFormat,
|
|
23
|
+
MetricType,
|
|
24
|
+
)
|
|
25
|
+
from .physical import (
|
|
26
|
+
Column,
|
|
27
|
+
ColumnDataType,
|
|
28
|
+
DataSource,
|
|
29
|
+
ForeignKeyRef,
|
|
30
|
+
SourceType,
|
|
31
|
+
SyncStatus,
|
|
32
|
+
Table,
|
|
33
|
+
)
|
|
34
|
+
from .query import (
|
|
35
|
+
Dashboard,
|
|
36
|
+
QueryOrigin,
|
|
37
|
+
QueryResult,
|
|
38
|
+
QueryStatus,
|
|
39
|
+
QueryTrace,
|
|
40
|
+
QueryTraceStep,
|
|
41
|
+
SavedQuery,
|
|
42
|
+
)
|
|
43
|
+
from .semantic import (
|
|
44
|
+
AttributeRole,
|
|
45
|
+
BusinessDomain,
|
|
46
|
+
Entity,
|
|
47
|
+
EntityAttribute,
|
|
48
|
+
EntityTableMapping,
|
|
49
|
+
Relationship,
|
|
50
|
+
RelationshipCardinality,
|
|
51
|
+
RelationshipDiscovery,
|
|
52
|
+
SemanticType,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
__version__ = "0.1.0"
|
|
56
|
+
|
|
57
|
+
__all__ = [
|
|
58
|
+
"DataSource",
|
|
59
|
+
"SourceType",
|
|
60
|
+
"SyncStatus",
|
|
61
|
+
"Table",
|
|
62
|
+
"Column",
|
|
63
|
+
"ColumnDataType",
|
|
64
|
+
"ForeignKeyRef",
|
|
65
|
+
"BusinessDomain",
|
|
66
|
+
"Entity",
|
|
67
|
+
"EntityTableMapping",
|
|
68
|
+
"EntityAttribute",
|
|
69
|
+
"AttributeRole",
|
|
70
|
+
"SemanticType",
|
|
71
|
+
"Relationship",
|
|
72
|
+
"RelationshipCardinality",
|
|
73
|
+
"RelationshipDiscovery",
|
|
74
|
+
"Metric",
|
|
75
|
+
"MetricType",
|
|
76
|
+
"MetricFormat",
|
|
77
|
+
"GranularityHint",
|
|
78
|
+
"MetricFilter",
|
|
79
|
+
"MetricColumnRef",
|
|
80
|
+
"GlossaryTerm",
|
|
81
|
+
"SavedQuery",
|
|
82
|
+
"QueryStatus",
|
|
83
|
+
"QueryOrigin",
|
|
84
|
+
"QueryTrace",
|
|
85
|
+
"QueryTraceStep",
|
|
86
|
+
"QueryResult",
|
|
87
|
+
"Dashboard",
|
|
88
|
+
"SemanticCatalog",
|
|
89
|
+
]
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""
|
|
2
|
+
SemanticCatalog — the root object of the Nomox ontology.
|
|
3
|
+
|
|
4
|
+
One catalog per organisation. It holds the complete knowledge graph:
|
|
5
|
+
physical schema, semantic layer, metrics, and saved queries. This is
|
|
6
|
+
what the Catalog service persists, the Semantic Indexer writes to,
|
|
7
|
+
and every other service reads from.
|
|
8
|
+
|
|
9
|
+
In-memory representation: dict-keyed by UUID for O(1) lookups.
|
|
10
|
+
Persistence is the Catalog service's responsibility.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from datetime import datetime, timezone
|
|
16
|
+
from uuid import UUID, uuid4
|
|
17
|
+
|
|
18
|
+
from pydantic import BaseModel, Field
|
|
19
|
+
|
|
20
|
+
from .glossary import GlossaryTerm
|
|
21
|
+
from .metric import Metric
|
|
22
|
+
from .physical import Column, DataSource, Table
|
|
23
|
+
from .query import Dashboard, QueryStatus, SavedQuery
|
|
24
|
+
from .semantic import BusinessDomain, Entity, EntityAttribute, Relationship
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SemanticCatalog(BaseModel):
|
|
28
|
+
"""
|
|
29
|
+
The complete knowledge graph for one organisation.
|
|
30
|
+
|
|
31
|
+
Lookup helpers are provided for the most common access patterns.
|
|
32
|
+
Services should use these rather than scanning the dicts directly.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
id: UUID = Field(default_factory=uuid4)
|
|
36
|
+
organisation_id: UUID
|
|
37
|
+
name: str
|
|
38
|
+
|
|
39
|
+
# Physical layer
|
|
40
|
+
sources: dict[UUID, DataSource] = Field(default_factory=dict)
|
|
41
|
+
tables: dict[UUID, Table] = Field(default_factory=dict)
|
|
42
|
+
columns: dict[UUID, Column] = Field(default_factory=dict)
|
|
43
|
+
|
|
44
|
+
# Semantic layer
|
|
45
|
+
domains: dict[UUID, BusinessDomain] = Field(default_factory=dict)
|
|
46
|
+
entities: dict[UUID, Entity] = Field(default_factory=dict)
|
|
47
|
+
attributes: dict[UUID, EntityAttribute] = Field(default_factory=dict)
|
|
48
|
+
relationships: dict[UUID, Relationship] = Field(default_factory=dict)
|
|
49
|
+
metrics: dict[UUID, Metric] = Field(default_factory=dict)
|
|
50
|
+
glossary_terms: dict[UUID, GlossaryTerm] = Field(default_factory=dict)
|
|
51
|
+
|
|
52
|
+
# Queries
|
|
53
|
+
saved_queries: dict[UUID, SavedQuery] = Field(default_factory=dict)
|
|
54
|
+
dashboards: dict[UUID, Dashboard] = Field(default_factory=dict)
|
|
55
|
+
|
|
56
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
57
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
58
|
+
|
|
59
|
+
# ------------------------------------------------------------------ #
|
|
60
|
+
# Physical layer helpers
|
|
61
|
+
# ------------------------------------------------------------------ #
|
|
62
|
+
|
|
63
|
+
def tables_for_source(self, source_id: UUID) -> list[Table]:
|
|
64
|
+
return [t for t in self.tables.values() if t.source_id == source_id]
|
|
65
|
+
|
|
66
|
+
def columns_for_table(self, table_id: UUID) -> list[Column]:
|
|
67
|
+
return sorted(
|
|
68
|
+
[c for c in self.columns.values() if c.table_id == table_id],
|
|
69
|
+
key=lambda c: c.ordinal_position,
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
# ------------------------------------------------------------------ #
|
|
73
|
+
# Semantic layer helpers
|
|
74
|
+
# ------------------------------------------------------------------ #
|
|
75
|
+
|
|
76
|
+
def entity_by_slug(self, slug: str) -> Entity | None:
|
|
77
|
+
return next((e for e in self.entities.values() if e.slug == slug), None)
|
|
78
|
+
|
|
79
|
+
def metric_by_slug(self, slug: str) -> Metric | None:
|
|
80
|
+
return next((m for m in self.metrics.values() if m.slug == slug), None)
|
|
81
|
+
|
|
82
|
+
def golden_metrics(self) -> list[Metric]:
|
|
83
|
+
return [m for m in self.metrics.values() if m.is_golden and not m.is_deprecated]
|
|
84
|
+
|
|
85
|
+
def attributes_for_entity(self, entity_id: UUID) -> list[EntityAttribute]:
|
|
86
|
+
return [a for a in self.attributes.values() if a.entity_id == entity_id]
|
|
87
|
+
|
|
88
|
+
def relationships_for_entity(self, entity_id: UUID) -> list[Relationship]:
|
|
89
|
+
"""All relationships where this entity appears as source or target."""
|
|
90
|
+
return [
|
|
91
|
+
r for r in self.relationships.values() if r.source_entity_id == entity_id or r.target_entity_id == entity_id
|
|
92
|
+
]
|
|
93
|
+
|
|
94
|
+
def entities_in_domain(self, domain_id: UUID) -> list[Entity]:
|
|
95
|
+
return [e for e in self.entities.values() if e.domain_id == domain_id]
|
|
96
|
+
|
|
97
|
+
def child_entities(self, entity_id: UUID) -> list[Entity]:
|
|
98
|
+
"""Direct subtypes of an entity (entities with parent_entity_id == entity_id)."""
|
|
99
|
+
return [e for e in self.entities.values() if e.parent_entity_id == entity_id]
|
|
100
|
+
|
|
101
|
+
def descendant_entities(self, entity_id: UUID) -> list[Entity]:
|
|
102
|
+
"""All transitive subtypes of an entity."""
|
|
103
|
+
out: list[Entity] = []
|
|
104
|
+
stack = [entity_id]
|
|
105
|
+
while stack:
|
|
106
|
+
current = stack.pop()
|
|
107
|
+
for child in self.child_entities(current):
|
|
108
|
+
out.append(child)
|
|
109
|
+
stack.append(child.id)
|
|
110
|
+
return out
|
|
111
|
+
|
|
112
|
+
def metrics_in_domain(self, domain_id: UUID) -> list[Metric]:
|
|
113
|
+
return [m for m in self.metrics.values() if m.domain_id == domain_id]
|
|
114
|
+
|
|
115
|
+
def glossary_term_by_slug(self, slug: str) -> GlossaryTerm | None:
|
|
116
|
+
return next((t for t in self.glossary_terms.values() if t.slug == slug), None)
|
|
117
|
+
|
|
118
|
+
def glossary_terms_in_domain(self, domain_id: UUID) -> list[GlossaryTerm]:
|
|
119
|
+
return [t for t in self.glossary_terms.values() if t.domain_id == domain_id]
|
|
120
|
+
|
|
121
|
+
# ------------------------------------------------------------------ #
|
|
122
|
+
# Query helpers
|
|
123
|
+
# ------------------------------------------------------------------ #
|
|
124
|
+
|
|
125
|
+
def example_queries(self, limit: int = 20) -> list[SavedQuery]:
|
|
126
|
+
return [q for q in self.saved_queries.values() if q.is_example and q.status == QueryStatus.SUCCESS][:limit]
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Business glossary — internal terms and the context behind them.
|
|
3
|
+
|
|
4
|
+
A glossary term is for vocabulary that doesn't map to a specific entity,
|
|
5
|
+
attribute, or metric — just words the company uses that an outsider
|
|
6
|
+
(or a new hire, or an AI agent) might not understand.
|
|
7
|
+
|
|
8
|
+
Examples:
|
|
9
|
+
"Whale" — a customer who spends more than €10k/month
|
|
10
|
+
"Activation" — the moment a user completes their first meaningful action
|
|
11
|
+
"T0" — the date a contract goes live, used as the reference
|
|
12
|
+
point for all cohort calculations
|
|
13
|
+
|
|
14
|
+
Glossary terms can optionally point at the entities, attributes, and metrics
|
|
15
|
+
that operationalise them, so agents can follow the link from the conversational
|
|
16
|
+
term to the data that realises it.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from datetime import datetime, timezone
|
|
22
|
+
from uuid import UUID, uuid4
|
|
23
|
+
|
|
24
|
+
from pydantic import BaseModel, Field
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class GlossaryTerm(BaseModel):
|
|
28
|
+
id: UUID = Field(default_factory=uuid4)
|
|
29
|
+
slug: str
|
|
30
|
+
name: str
|
|
31
|
+
aliases: list[str] = Field(
|
|
32
|
+
default_factory=list,
|
|
33
|
+
description="Alternative spellings or near-synonyms used for this term in conversation.",
|
|
34
|
+
)
|
|
35
|
+
domain_id: UUID | None = None
|
|
36
|
+
business_context: str = Field(description="What this term means in the company's context.")
|
|
37
|
+
|
|
38
|
+
# Linkage to the rest of the ontology. Optional — glossary terms can stand alone.
|
|
39
|
+
related_entity_ids: list[UUID] = Field(
|
|
40
|
+
default_factory=list,
|
|
41
|
+
description="Entities that operationalise or are referenced by this term.",
|
|
42
|
+
)
|
|
43
|
+
related_attribute_ids: list[UUID] = Field(
|
|
44
|
+
default_factory=list,
|
|
45
|
+
description="EntityAttributes that operationalise or are referenced by this term.",
|
|
46
|
+
)
|
|
47
|
+
related_metric_ids: list[UUID] = Field(
|
|
48
|
+
default_factory=list,
|
|
49
|
+
description="Metrics that operationalise or are referenced by this term.",
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
53
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Metric — a named, reusable business computation.
|
|
3
|
+
|
|
4
|
+
Metrics are first-class citizens in the ontology, not just annotated columns.
|
|
5
|
+
They encode the authoritative definition of a business number: what it counts,
|
|
6
|
+
which filters always apply, how it should be displayed, and — crucially —
|
|
7
|
+
what it does NOT include. That last part lives in `business_context` and
|
|
8
|
+
is the primary defence against AI hallucination.
|
|
9
|
+
|
|
10
|
+
Golden metrics (is_golden=True) are the single company-wide definition for
|
|
11
|
+
a concept. The query engine and agents always prefer a golden metric over
|
|
12
|
+
an ad-hoc column aggregation.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from enum import Enum
|
|
19
|
+
from uuid import UUID, uuid4
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, Field
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class MetricType(str, Enum):
|
|
25
|
+
"""
|
|
26
|
+
Classification of how the metric is computed.
|
|
27
|
+
Drives SQL construction in the query engine.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
SIMPLE = "simple" # SUM/COUNT/AVG of a single column
|
|
31
|
+
RATIO = "ratio" # numerator / denominator
|
|
32
|
+
CUMULATIVE = "cumulative" # running total over time
|
|
33
|
+
PERIOD_OVER_PERIOD = "period_over_period" # WoW, MoM, YoY
|
|
34
|
+
DERIVED = "derived" # computed from other metrics
|
|
35
|
+
FUNNEL = "funnel" # multi-step conversion
|
|
36
|
+
CUSTOM = "custom" # free-form SQL expression
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class MetricFormat(str, Enum):
|
|
40
|
+
"""Display format hint for UI and agent responses."""
|
|
41
|
+
|
|
42
|
+
NUMBER = "number"
|
|
43
|
+
CURRENCY = "currency"
|
|
44
|
+
PERCENTAGE = "percentage"
|
|
45
|
+
DURATION = "duration"
|
|
46
|
+
RATIO = "ratio"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class GranularityHint(str, Enum):
|
|
50
|
+
"""Finest meaningful time granularity for this metric."""
|
|
51
|
+
|
|
52
|
+
HOUR = "hour"
|
|
53
|
+
DAY = "day"
|
|
54
|
+
WEEK = "week"
|
|
55
|
+
MONTH = "month"
|
|
56
|
+
QUARTER = "quarter"
|
|
57
|
+
YEAR = "year"
|
|
58
|
+
NONE = "none" # metric is not time-series meaningful
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
class MetricFilter(BaseModel):
|
|
62
|
+
"""
|
|
63
|
+
A named, always-applied filter baked into a metric definition.
|
|
64
|
+
|
|
65
|
+
Example: 'Active Customers' always has status = 'active'.
|
|
66
|
+
The query engine resolves `attribute_id` to a real column reference.
|
|
67
|
+
"""
|
|
68
|
+
|
|
69
|
+
attribute_id: UUID
|
|
70
|
+
operator: str = Field(description="SQL operator: '=', '!=', 'IN', '>', '<', 'LIKE', etc.")
|
|
71
|
+
value: str | list[str]
|
|
72
|
+
description: str = ""
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
class MetricColumnRef(BaseModel):
|
|
76
|
+
"""Granular lineage: which physical column feeds this metric."""
|
|
77
|
+
|
|
78
|
+
source_id: UUID
|
|
79
|
+
table_id: UUID
|
|
80
|
+
column_id: UUID
|
|
81
|
+
role: str = Field(description="How this column is used: 'numerator', 'denominator', 'dimension', 'filter', 'date'.")
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class Metric(BaseModel):
|
|
85
|
+
"""
|
|
86
|
+
A named, reusable business metric.
|
|
87
|
+
|
|
88
|
+
The sql_expression is parameterised using {source_slug.schema.table.column}
|
|
89
|
+
placeholders. The query engine resolves these to actual qualified names at
|
|
90
|
+
runtime, applying the correct dialect.
|
|
91
|
+
|
|
92
|
+
Example:
|
|
93
|
+
sql_expression = "SUM({prod-postgres.public.orders.amount_cents}) / 100.0"
|
|
94
|
+
|
|
95
|
+
For RATIO metrics, use numerator_expression + denominator_expression instead.
|
|
96
|
+
|
|
97
|
+
The business_context field is as important as the SQL:
|
|
98
|
+
"Use for top-line revenue reporting. Does NOT include pending or
|
|
99
|
+
cancelled orders. For refund-adjusted revenue, use net_revenue_eur."
|
|
100
|
+
|
|
101
|
+
This narrative is what agents read to decide WHICH metric to use — and
|
|
102
|
+
to avoid confidently computing the wrong thing.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
id: UUID = Field(default_factory=uuid4)
|
|
106
|
+
slug: str
|
|
107
|
+
name: str
|
|
108
|
+
abbreviation: str | None = Field(
|
|
109
|
+
default=None,
|
|
110
|
+
description="Single canonical short form, e.g. 'MRR'. Use `aliases` for additional synonyms.",
|
|
111
|
+
)
|
|
112
|
+
aliases: list[str] = Field(
|
|
113
|
+
default_factory=list,
|
|
114
|
+
description=(
|
|
115
|
+
"Alternative names used in conversation that should resolve to this metric, e.g. "
|
|
116
|
+
"['monthly recurring', 'subscription revenue'] for MRR."
|
|
117
|
+
),
|
|
118
|
+
)
|
|
119
|
+
domain_id: UUID | None = None
|
|
120
|
+
|
|
121
|
+
metric_type: MetricType = MetricType.SIMPLE
|
|
122
|
+
|
|
123
|
+
# SQL expression — parameterised, dialect-neutral
|
|
124
|
+
sql_expression: str
|
|
125
|
+
numerator_expression: str | None = None # RATIO metrics
|
|
126
|
+
denominator_expression: str | None = None # RATIO metrics
|
|
127
|
+
|
|
128
|
+
base_filters: list[MetricFilter] = Field(default_factory=list)
|
|
129
|
+
|
|
130
|
+
description_auto: str = ""
|
|
131
|
+
description_human: str = ""
|
|
132
|
+
business_context: str = Field(
|
|
133
|
+
default="",
|
|
134
|
+
description=(
|
|
135
|
+
"Narrative for agents: what this metric measures, what it does NOT include, "
|
|
136
|
+
"common misinterpretations to avoid. This is the hallucination guard."
|
|
137
|
+
),
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
format: MetricFormat = MetricFormat.NUMBER
|
|
141
|
+
unit: str | None = Field(default=None, description="E.g. 'EUR', '%', 'days'.")
|
|
142
|
+
decimal_places: int = Field(default=2, ge=0)
|
|
143
|
+
|
|
144
|
+
granularity_hint: GranularityHint = GranularityHint.DAY
|
|
145
|
+
date_attribute_id: UUID | None = Field(
|
|
146
|
+
default=None,
|
|
147
|
+
description="The EntityAttribute used for time-axis slicing (role must be TEMPORAL).",
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
# Lineage
|
|
151
|
+
entities_referenced: list[UUID] = Field(default_factory=list)
|
|
152
|
+
column_refs: list[MetricColumnRef] = Field(default_factory=list)
|
|
153
|
+
derived_from_metric_ids: list[UUID] = Field(
|
|
154
|
+
default_factory=list,
|
|
155
|
+
description="For DERIVED metrics: parent metric IDs.",
|
|
156
|
+
)
|
|
157
|
+
|
|
158
|
+
is_golden: bool = Field(
|
|
159
|
+
default=False,
|
|
160
|
+
description=(
|
|
161
|
+
"Golden metrics are the single authoritative definition for this concept. "
|
|
162
|
+
"Agents and the query engine prefer them over ad-hoc aggregations."
|
|
163
|
+
),
|
|
164
|
+
)
|
|
165
|
+
is_confirmed: bool = False
|
|
166
|
+
is_deprecated: bool = False
|
|
167
|
+
deprecation_note: str | None = None
|
|
168
|
+
superseded_by_metric_id: UUID | None = None
|
|
169
|
+
|
|
170
|
+
tags: list[str] = Field(default_factory=list)
|
|
171
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
172
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
173
|
+
|
|
174
|
+
@property
|
|
175
|
+
def description(self) -> str:
|
|
176
|
+
return self.description_human or self.description_auto
|
|
@@ -0,0 +1,175 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Physical layer — raw schema objects discovered from connected data sources.
|
|
3
|
+
|
|
4
|
+
These represent what physically exists: where data lives and what shape it has.
|
|
5
|
+
The semantic layer maps business meaning on top of these.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from datetime import datetime, timezone
|
|
11
|
+
from enum import Enum
|
|
12
|
+
from typing import Any
|
|
13
|
+
from uuid import UUID, uuid4
|
|
14
|
+
|
|
15
|
+
from pydantic import BaseModel, Field
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class SourceType(str, Enum):
|
|
19
|
+
POSTGRES = "postgres"
|
|
20
|
+
MYSQL = "mysql"
|
|
21
|
+
SQLITE = "sqlite"
|
|
22
|
+
BIGQUERY = "bigquery"
|
|
23
|
+
SNOWFLAKE = "snowflake"
|
|
24
|
+
REDSHIFT = "redshift"
|
|
25
|
+
TRINO = "trino"
|
|
26
|
+
EXCEL = "excel"
|
|
27
|
+
GOOGLE_SHEETS = "google_sheets"
|
|
28
|
+
CSV = "csv"
|
|
29
|
+
REST_API = "rest_api"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class ColumnDataType(str, Enum):
|
|
33
|
+
INTEGER = "integer"
|
|
34
|
+
BIGINT = "bigint"
|
|
35
|
+
FLOAT = "float"
|
|
36
|
+
DECIMAL = "decimal"
|
|
37
|
+
BOOLEAN = "boolean"
|
|
38
|
+
VARCHAR = "varchar"
|
|
39
|
+
TEXT = "text"
|
|
40
|
+
UUID = "uuid"
|
|
41
|
+
DATE = "date"
|
|
42
|
+
TIMESTAMP = "timestamp"
|
|
43
|
+
TIMESTAMP_TZ = "timestamp_tz"
|
|
44
|
+
JSON = "json"
|
|
45
|
+
ARRAY = "array"
|
|
46
|
+
OTHER = "other"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
class SyncStatus(str, Enum):
|
|
50
|
+
PENDING = "pending"
|
|
51
|
+
IN_PROGRESS = "in_progress"
|
|
52
|
+
SUCCESS = "success"
|
|
53
|
+
FAILED = "failed"
|
|
54
|
+
STALE = "stale"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class DataSource(BaseModel):
|
|
58
|
+
"""
|
|
59
|
+
A connected external data source — a database, spreadsheet, or warehouse.
|
|
60
|
+
|
|
61
|
+
Credentials are NOT stored here; they live in a secrets manager.
|
|
62
|
+
This model holds only structural and descriptive metadata.
|
|
63
|
+
"""
|
|
64
|
+
id: UUID = Field(default_factory=uuid4)
|
|
65
|
+
slug: str = Field(description="URL-safe unique identifier, e.g. 'prod-postgres'.")
|
|
66
|
+
name: str
|
|
67
|
+
source_type: SourceType
|
|
68
|
+
description: str = ""
|
|
69
|
+
|
|
70
|
+
# Non-sensitive connector parameters (host, port, database name, sheet ID, etc.)
|
|
71
|
+
connector_config: dict[str, Any] = Field(default_factory=dict)
|
|
72
|
+
|
|
73
|
+
last_synced_at: datetime | None = None
|
|
74
|
+
sync_status: SyncStatus = SyncStatus.PENDING
|
|
75
|
+
sync_error: str | None = None
|
|
76
|
+
|
|
77
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
78
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class Table(BaseModel):
|
|
82
|
+
"""
|
|
83
|
+
A table, view, or sheet within a DataSource.
|
|
84
|
+
|
|
85
|
+
`schema_name` is empty for flat sources (Excel, CSV).
|
|
86
|
+
"""
|
|
87
|
+
id: UUID = Field(default_factory=uuid4)
|
|
88
|
+
source_id: UUID
|
|
89
|
+
schema_name: str = ""
|
|
90
|
+
table_name: str
|
|
91
|
+
is_view: bool = False
|
|
92
|
+
|
|
93
|
+
row_count_estimate: int | None = None
|
|
94
|
+
size_bytes: int | None = None
|
|
95
|
+
|
|
96
|
+
# Dual-field description pattern used throughout:
|
|
97
|
+
# _auto is LLM-generated; _human is the authoritative human override.
|
|
98
|
+
description_auto: str = ""
|
|
99
|
+
description_human: str = ""
|
|
100
|
+
|
|
101
|
+
is_staging: bool = Field(
|
|
102
|
+
default=False,
|
|
103
|
+
description="Marks raw/staging tables not intended for direct querying.",
|
|
104
|
+
)
|
|
105
|
+
tags: list[str] = Field(default_factory=list)
|
|
106
|
+
|
|
107
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
108
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
109
|
+
|
|
110
|
+
@property
|
|
111
|
+
def fully_qualified_name(self) -> str:
|
|
112
|
+
return f"{self.schema_name}.{self.table_name}" if self.schema_name else self.table_name
|
|
113
|
+
|
|
114
|
+
@property
|
|
115
|
+
def description(self) -> str:
|
|
116
|
+
return self.description_human or self.description_auto
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class ForeignKeyRef(BaseModel):
|
|
120
|
+
"""A declared or inferred foreign key reference to another column."""
|
|
121
|
+
target_table_id: UUID
|
|
122
|
+
target_column_id: UUID
|
|
123
|
+
is_inferred: bool = Field(
|
|
124
|
+
default=False,
|
|
125
|
+
description="True when discovered by heuristics, not declared in the schema.",
|
|
126
|
+
)
|
|
127
|
+
confidence: float = Field(default=1.0, ge=0.0, le=1.0)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
class Column(BaseModel):
|
|
131
|
+
"""
|
|
132
|
+
A column or field within a Table.
|
|
133
|
+
|
|
134
|
+
Stores raw schema metadata and the enriched semantic metadata added
|
|
135
|
+
during catalog building (descriptions, sample values, sensitivity flags).
|
|
136
|
+
"""
|
|
137
|
+
id: UUID = Field(default_factory=uuid4)
|
|
138
|
+
table_id: UUID
|
|
139
|
+
source_id: UUID # denormalised for efficient filtering
|
|
140
|
+
|
|
141
|
+
column_name: str
|
|
142
|
+
ordinal_position: int
|
|
143
|
+
data_type: ColumnDataType
|
|
144
|
+
raw_data_type: str = Field(description="Original type string from the source, e.g. 'character varying(255)'.")
|
|
145
|
+
|
|
146
|
+
is_nullable: bool = True
|
|
147
|
+
is_primary_key: bool = False
|
|
148
|
+
is_unique: bool = False
|
|
149
|
+
|
|
150
|
+
foreign_key_refs: list[ForeignKeyRef] = Field(default_factory=list)
|
|
151
|
+
|
|
152
|
+
description_auto: str = ""
|
|
153
|
+
description_human: str = ""
|
|
154
|
+
|
|
155
|
+
sample_values: list[str] = Field(
|
|
156
|
+
default_factory=list,
|
|
157
|
+
max_length=10,
|
|
158
|
+
description="Representative sample values used during semantic enrichment.",
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
# Heuristic flags set by the DAL during discovery
|
|
162
|
+
is_temporal: bool = False
|
|
163
|
+
is_identifier: bool = False
|
|
164
|
+
is_sensitive: bool = Field(
|
|
165
|
+
default=False,
|
|
166
|
+
description="PII or otherwise sensitive — suppressed from agent context.",
|
|
167
|
+
)
|
|
168
|
+
tags: list[str] = Field(default_factory=list)
|
|
169
|
+
|
|
170
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
171
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
172
|
+
|
|
173
|
+
@property
|
|
174
|
+
def description(self) -> str:
|
|
175
|
+
return self.description_human or self.description_auto
|
nomox_semantics/query.py
ADDED
|
@@ -0,0 +1,188 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Query models — saved queries and their execution traces.
|
|
3
|
+
|
|
4
|
+
Every answer Nomox produces is a persisted object: the natural language
|
|
5
|
+
question, the SQL generated, the reasoning trace, and the result.
|
|
6
|
+
|
|
7
|
+
The trace is the trust mechanism. It lets non-technical users verify
|
|
8
|
+
answers and lets analysts build on them. It's also the few-shot corpus
|
|
9
|
+
the query engine learns from: SavedQuery.is_example=True marks a query
|
|
10
|
+
as a high-quality example for future SQL generation.
|
|
11
|
+
|
|
12
|
+
Dashboards are named collections of pinned SavedQueries.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from enum import Enum
|
|
19
|
+
from typing import Any
|
|
20
|
+
from uuid import UUID, uuid4
|
|
21
|
+
|
|
22
|
+
from pydantic import BaseModel, Field
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class QueryStatus(str, Enum):
|
|
26
|
+
PENDING = "pending"
|
|
27
|
+
RUNNING = "running"
|
|
28
|
+
SUCCESS = "success"
|
|
29
|
+
PARTIAL = "partial" # returned results but with caveats
|
|
30
|
+
FAILED = "failed"
|
|
31
|
+
UNSUPPORTED = "unsupported" # question outside the catalog's scope
|
|
32
|
+
AMBIGUOUS = "ambiguous" # question needs clarification before answering
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class QueryOrigin(str, Enum):
|
|
36
|
+
BI_PLATFORM = "bi_platform"
|
|
37
|
+
MCP_SERVER = "mcp_server"
|
|
38
|
+
API = "api"
|
|
39
|
+
DASHBOARD = "dashboard" # scheduled refresh
|
|
40
|
+
SYSTEM = "system" # internal catalog-building query
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class QueryTraceStep(BaseModel):
|
|
44
|
+
"""
|
|
45
|
+
One reasoning step in the chain from natural language to SQL.
|
|
46
|
+
|
|
47
|
+
Each step has a SQL fragment and a plain-English explanation of why
|
|
48
|
+
it was written. Together they form a transparent, inspectable audit
|
|
49
|
+
trail — "reading a clear explanation from a colleague, not a black box."
|
|
50
|
+
"""
|
|
51
|
+
step_number: int
|
|
52
|
+
title: str = Field(description="Short label, e.g. 'Filter by date range'.")
|
|
53
|
+
sql_fragment: str
|
|
54
|
+
explanation: str = Field(description="Plain English: what this step does and why.")
|
|
55
|
+
|
|
56
|
+
# Catalog objects referenced in this step (for lineage and analytics)
|
|
57
|
+
entities_used: list[UUID] = Field(default_factory=list)
|
|
58
|
+
attributes_used: list[UUID] = Field(default_factory=list)
|
|
59
|
+
metrics_used: list[UUID] = Field(default_factory=list)
|
|
60
|
+
relationships_used: list[UUID] = Field(default_factory=list)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class QueryTrace(BaseModel):
|
|
64
|
+
"""
|
|
65
|
+
Full reasoning trace for a query execution.
|
|
66
|
+
|
|
67
|
+
Contains ordered steps, the final assembled SQL, and aggregate lineage.
|
|
68
|
+
This is the artifact that agents expose and humans inspect.
|
|
69
|
+
"""
|
|
70
|
+
steps: list[QueryTraceStep] = Field(default_factory=list)
|
|
71
|
+
final_sql: str
|
|
72
|
+
sql_dialect: str = Field(
|
|
73
|
+
default="ansi",
|
|
74
|
+
description="SQL dialect the query was generated for.",
|
|
75
|
+
)
|
|
76
|
+
|
|
77
|
+
# Aggregate lineage across all steps
|
|
78
|
+
entities_used: list[UUID] = Field(default_factory=list)
|
|
79
|
+
attributes_used: list[UUID] = Field(default_factory=list)
|
|
80
|
+
metrics_used: list[UUID] = Field(default_factory=list)
|
|
81
|
+
sources_queried: list[UUID] = Field(default_factory=list)
|
|
82
|
+
|
|
83
|
+
confidence: float = Field(default=1.0, ge=0.0, le=1.0)
|
|
84
|
+
caveats: list[str] = Field(
|
|
85
|
+
default_factory=list,
|
|
86
|
+
description="Warnings or assumptions the user should be aware of.",
|
|
87
|
+
)
|
|
88
|
+
ambiguities_resolved: list[str] = Field(
|
|
89
|
+
default_factory=list,
|
|
90
|
+
description="Ambiguous terms encountered and how they were resolved.",
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class QueryResult(BaseModel):
|
|
95
|
+
"""
|
|
96
|
+
The output of executing a query.
|
|
97
|
+
|
|
98
|
+
Full result sets are streamed to the client and not persisted.
|
|
99
|
+
We store the schema and a row sample for re-display without re-execution.
|
|
100
|
+
"""
|
|
101
|
+
columns: list[str]
|
|
102
|
+
row_count: int
|
|
103
|
+
row_sample: list[list[Any]] = Field(
|
|
104
|
+
default_factory=list,
|
|
105
|
+
max_length=100,
|
|
106
|
+
description="Up to 100 rows stored for fast re-display.",
|
|
107
|
+
)
|
|
108
|
+
suggested_chart_type: str | None = Field(
|
|
109
|
+
default=None,
|
|
110
|
+
description="Suggested visualisation: 'bar', 'line', 'table', 'metric', 'pie'.",
|
|
111
|
+
)
|
|
112
|
+
x_axis_column: str | None = None
|
|
113
|
+
y_axis_columns: list[str] = Field(default_factory=list)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
class SavedQuery(BaseModel):
|
|
117
|
+
"""
|
|
118
|
+
A persisted query — the combination of question, SQL, trace, and result.
|
|
119
|
+
|
|
120
|
+
Saved queries serve multiple purposes:
|
|
121
|
+
- Re-execution without re-reasoning (fast dashboard refreshes)
|
|
122
|
+
- Few-shot SQL examples for the query engine (is_example=True)
|
|
123
|
+
- Audit trail
|
|
124
|
+
- Usage analytics (which entities and metrics are most queried)
|
|
125
|
+
"""
|
|
126
|
+
id: UUID = Field(default_factory=uuid4)
|
|
127
|
+
|
|
128
|
+
natural_language: str = Field(description="The original question, verbatim.")
|
|
129
|
+
canonical_question: str = Field(
|
|
130
|
+
default="",
|
|
131
|
+
description="Normalised form for similarity matching.",
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
status: QueryStatus = QueryStatus.PENDING
|
|
135
|
+
origin: QueryOrigin = QueryOrigin.BI_PLATFORM
|
|
136
|
+
|
|
137
|
+
trace: QueryTrace | None = None
|
|
138
|
+
result: QueryResult | None = None
|
|
139
|
+
error_message: str | None = None
|
|
140
|
+
|
|
141
|
+
# Catalog objects this query is about (derived from trace)
|
|
142
|
+
entities_referenced: list[UUID] = Field(default_factory=list)
|
|
143
|
+
metrics_referenced: list[UUID] = Field(default_factory=list)
|
|
144
|
+
domain_ids: list[UUID] = Field(default_factory=list)
|
|
145
|
+
|
|
146
|
+
created_by: str | None = None
|
|
147
|
+
session_id: str | None = None
|
|
148
|
+
|
|
149
|
+
is_pinned: bool = Field(default=False, description="Pinned queries appear in the catalog UI.")
|
|
150
|
+
is_example: bool = Field(
|
|
151
|
+
default=False,
|
|
152
|
+
description="High-quality example for few-shot SQL generation.",
|
|
153
|
+
)
|
|
154
|
+
title: str | None = None
|
|
155
|
+
description: str = ""
|
|
156
|
+
|
|
157
|
+
execution_time_ms: int | None = None
|
|
158
|
+
|
|
159
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
160
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
161
|
+
last_executed_at: datetime | None = None
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
class Dashboard(BaseModel):
|
|
165
|
+
"""
|
|
166
|
+
A named, shareable collection of pinned SavedQueries.
|
|
167
|
+
|
|
168
|
+
Each panel in the dashboard is a SavedQuery. Layout and per-panel
|
|
169
|
+
visualisation config are stored in panel_layout.
|
|
170
|
+
"""
|
|
171
|
+
id: UUID = Field(default_factory=uuid4)
|
|
172
|
+
slug: str
|
|
173
|
+
name: str
|
|
174
|
+
description: str = ""
|
|
175
|
+
domain_id: UUID | None = None
|
|
176
|
+
|
|
177
|
+
query_ids: list[UUID] = Field(default_factory=list)
|
|
178
|
+
panel_layout: list[dict[str, Any]] = Field(
|
|
179
|
+
default_factory=list,
|
|
180
|
+
description="UI layout per panel: {query_id, x, y, w, h, chart_type}.",
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
refresh_interval_minutes: int | None = None
|
|
184
|
+
created_by: str | None = None
|
|
185
|
+
is_public: bool = False
|
|
186
|
+
|
|
187
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
188
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
@@ -0,0 +1,338 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Semantic layer — business meaning mapped over the physical layer.
|
|
3
|
+
|
|
4
|
+
This is where raw tables and columns become Customers, Orders, Revenue,
|
|
5
|
+
and the relationships between them. The ontology lives here.
|
|
6
|
+
|
|
7
|
+
Three concepts form the core:
|
|
8
|
+
Entity — a named business concept (Customer, Order, Product)
|
|
9
|
+
EntityAttribute — a meaningful field on an entity (plan_type, created_at)
|
|
10
|
+
Relationship — a link between two entities (Order → Customer)
|
|
11
|
+
|
|
12
|
+
BusinessDomain groups them for navigation and scoping.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from enum import Enum
|
|
19
|
+
from uuid import UUID, uuid4
|
|
20
|
+
|
|
21
|
+
from pydantic import BaseModel, Field, model_validator
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class BusinessDomain(BaseModel):
|
|
25
|
+
"""
|
|
26
|
+
A top-level business domain used to classify entities and metrics.
|
|
27
|
+
|
|
28
|
+
Examples: Sales, Operations, Finance, Product, Marketing.
|
|
29
|
+
Domains are the primary navigation axis for human users and the
|
|
30
|
+
first scoping hint passed to agents.
|
|
31
|
+
"""
|
|
32
|
+
id: UUID = Field(default_factory=uuid4)
|
|
33
|
+
slug: str
|
|
34
|
+
name: str
|
|
35
|
+
description: str = ""
|
|
36
|
+
color: str = Field(default="#6366f1", description="Hex color for UI.")
|
|
37
|
+
icon: str = Field(default="database", description="Icon name for UI.")
|
|
38
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class EntityTableMapping(BaseModel):
|
|
42
|
+
"""
|
|
43
|
+
Maps a business Entity to a physical table.
|
|
44
|
+
|
|
45
|
+
An entity can span multiple tables (e.g. a Customer entity backed by
|
|
46
|
+
both `users` and `customer_profiles`). One table is primary; extensions
|
|
47
|
+
are joined on the identifier attribute.
|
|
48
|
+
"""
|
|
49
|
+
table_id: UUID
|
|
50
|
+
source_id: UUID # denormalised
|
|
51
|
+
is_primary: bool = True
|
|
52
|
+
join_condition: str | None = Field(
|
|
53
|
+
default=None,
|
|
54
|
+
description="SQL JOIN fragment for extension tables, e.g. 'users.id = customer_profiles.user_id'.",
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class Entity(BaseModel):
|
|
59
|
+
"""
|
|
60
|
+
A named business concept — the fundamental unit of the semantic layer.
|
|
61
|
+
|
|
62
|
+
Entities correspond to things the business talks about: Customer, Order,
|
|
63
|
+
Product, Invoice, Supplier. They are discovered automatically from schema
|
|
64
|
+
patterns by the Semantic Indexer and validated by humans.
|
|
65
|
+
|
|
66
|
+
An entity always maps to at least one physical table. The mapping tells
|
|
67
|
+
the query engine where to look; the entity itself is what agents reference.
|
|
68
|
+
|
|
69
|
+
Entities support single inheritance via `parent_entity_id` for cases like
|
|
70
|
+
IndividualCustomer / BusinessCustomer both being kinds of Party. Subtypes
|
|
71
|
+
inherit attributes conceptually; the inheritance is for ontology navigation
|
|
72
|
+
and abstract grouping, not SQL-level polymorphism.
|
|
73
|
+
"""
|
|
74
|
+
id: UUID = Field(default_factory=uuid4)
|
|
75
|
+
slug: str = Field(description="Stable machine-readable identifier, e.g. 'customer'.")
|
|
76
|
+
name: str
|
|
77
|
+
plural_name: str = Field(description="Used in UI and agent context, e.g. 'Customers'.")
|
|
78
|
+
aliases: list[str] = Field(
|
|
79
|
+
default_factory=list,
|
|
80
|
+
description=(
|
|
81
|
+
"Alternative names used in conversation that should resolve to this entity. "
|
|
82
|
+
"Drives natural-language matching, e.g. ['user', 'account holder'] for Customer."
|
|
83
|
+
),
|
|
84
|
+
)
|
|
85
|
+
domain_id: UUID | None = None
|
|
86
|
+
|
|
87
|
+
parent_entity_id: UUID | None = Field(
|
|
88
|
+
default=None,
|
|
89
|
+
description=(
|
|
90
|
+
"Optional parent entity for is-a hierarchies. "
|
|
91
|
+
"E.g. IndividualCustomer and BusinessCustomer both have parent_entity_id = Party."
|
|
92
|
+
),
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
description_auto: str = ""
|
|
96
|
+
description_human: str = ""
|
|
97
|
+
|
|
98
|
+
table_mappings: list[EntityTableMapping] = Field(default_factory=list)
|
|
99
|
+
|
|
100
|
+
# Discovery provenance
|
|
101
|
+
is_confirmed: bool = Field(
|
|
102
|
+
default=False,
|
|
103
|
+
description="False = auto-discovered by the Semantic Indexer, not yet human-validated.",
|
|
104
|
+
)
|
|
105
|
+
discovery_confidence: float = Field(
|
|
106
|
+
default=1.0, ge=0.0, le=1.0,
|
|
107
|
+
description="Indexer confidence score. 1.0 for human-defined entities.",
|
|
108
|
+
)
|
|
109
|
+
|
|
110
|
+
tags: list[str] = Field(default_factory=list)
|
|
111
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
112
|
+
updated_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
113
|
+
|
|
114
|
+
@property
|
|
115
|
+
def description(self) -> str:
|
|
116
|
+
return self.description_human or self.description_auto
|
|
117
|
+
|
|
118
|
+
@property
|
|
119
|
+
def primary_table_id(self) -> UUID | None:
|
|
120
|
+
for m in self.table_mappings:
|
|
121
|
+
if m.is_primary:
|
|
122
|
+
return m.table_id
|
|
123
|
+
return None
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
class AttributeRole(str, Enum):
|
|
127
|
+
"""
|
|
128
|
+
Semantic role of an attribute.
|
|
129
|
+
|
|
130
|
+
Drives how the query engine uses this attribute when translating
|
|
131
|
+
natural language to SQL. An agent reading a MEASURE knows it can
|
|
132
|
+
be aggregated; a TEMPORAL knows it can be used as a time axis.
|
|
133
|
+
"""
|
|
134
|
+
IDENTIFIER = "identifier" # primary key / surrogate id
|
|
135
|
+
DIMENSION = "dimension" # groupable categorical field
|
|
136
|
+
MEASURE = "measure" # numeric field, directly aggregatable
|
|
137
|
+
TEMPORAL = "temporal" # date/timestamp for time-series slicing
|
|
138
|
+
LABEL = "label" # human-readable name field (e.g. customer_name)
|
|
139
|
+
FLAG = "flag" # boolean status
|
|
140
|
+
FOREIGN_KEY = "foreign_key" # FK to another entity
|
|
141
|
+
OTHER = "other"
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
class SemanticType(str, Enum):
|
|
145
|
+
"""
|
|
146
|
+
Semantic interpretation of an attribute's values, beyond its SQL type.
|
|
147
|
+
|
|
148
|
+
Distinct from `AttributeRole` (which says how the attribute is used in SQL)
|
|
149
|
+
and from `Column.data_type` (which is the storage type). This describes what
|
|
150
|
+
the values *mean*: a `BIGINT` column might be a `CURRENCY` in cents or a
|
|
151
|
+
`DURATION` in milliseconds or just a `COUNT`. Agents and UIs use this to
|
|
152
|
+
format values and to refuse arithmetically nonsensical operations
|
|
153
|
+
(e.g. don't sum two LATITUDE values).
|
|
154
|
+
"""
|
|
155
|
+
CURRENCY = "currency" # monetary amount; pair with `unit` for ISO code
|
|
156
|
+
PERCENTAGE = "percentage" # 0–1 or 0–100, record which in description
|
|
157
|
+
DURATION = "duration" # time span; pair with `unit` (seconds, days, …)
|
|
158
|
+
COUNT = "count" # integer count of something
|
|
159
|
+
RATIO = "ratio" # dimensionless ratio
|
|
160
|
+
EMAIL = "email"
|
|
161
|
+
URL = "url"
|
|
162
|
+
PHONE = "phone"
|
|
163
|
+
COUNTRY_CODE = "country_code" # ISO 3166
|
|
164
|
+
LATITUDE = "latitude"
|
|
165
|
+
LONGITUDE = "longitude"
|
|
166
|
+
IP_ADDRESS = "ip_address"
|
|
167
|
+
OTHER = "other"
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
class EntityAttribute(BaseModel):
|
|
171
|
+
"""
|
|
172
|
+
A business-meaningful attribute on an Entity.
|
|
173
|
+
|
|
174
|
+
Maps one semantic attribute name either to a physical column (`column_id`)
|
|
175
|
+
or to a SQL expression (`expression`). The attribute name and description
|
|
176
|
+
are what agents and users see; the column or expression is how the query
|
|
177
|
+
engine resolves it.
|
|
178
|
+
|
|
179
|
+
Examples on a Customer entity:
|
|
180
|
+
customer_id → IDENTIFIER (backed by users.id)
|
|
181
|
+
name → LABEL (backed by users.full_name)
|
|
182
|
+
signup_date → TEMPORAL (backed by users.created_at)
|
|
183
|
+
plan_type → DIMENSION (known_values: free, starter, pro, enterprise)
|
|
184
|
+
lifetime_value → MEASURE (default_aggregation: SUM)
|
|
185
|
+
full_name → LABEL (computed: first_name || ' ' || last_name)
|
|
186
|
+
|
|
187
|
+
Exactly one of `column_id` or `expression` must be set. Computed attributes
|
|
188
|
+
(with `expression`) leave `column_id` and `source_id` as None — the expression
|
|
189
|
+
is the resolution path instead.
|
|
190
|
+
"""
|
|
191
|
+
id: UUID = Field(default_factory=uuid4)
|
|
192
|
+
entity_id: UUID
|
|
193
|
+
|
|
194
|
+
column_id: UUID | None = Field(
|
|
195
|
+
default=None,
|
|
196
|
+
description="Physical column backing this attribute. None for computed attributes.",
|
|
197
|
+
)
|
|
198
|
+
source_id: UUID | None = Field(
|
|
199
|
+
default=None,
|
|
200
|
+
description="Denormalised source id. None for computed attributes.",
|
|
201
|
+
)
|
|
202
|
+
expression: str | None = Field(
|
|
203
|
+
default=None,
|
|
204
|
+
description=(
|
|
205
|
+
"SQL expression for computed attributes, e.g. "
|
|
206
|
+
"\"first_name || ' ' || last_name\" or \"amount_cents / 100.0\". "
|
|
207
|
+
"Must be None when column_id is set."
|
|
208
|
+
),
|
|
209
|
+
)
|
|
210
|
+
|
|
211
|
+
name: str = Field(description="Business-facing attribute name, e.g. 'signup_date'.")
|
|
212
|
+
display_name: str = Field(description="UI/agent label, e.g. 'Signup Date'.")
|
|
213
|
+
aliases: list[str] = Field(
|
|
214
|
+
default_factory=list,
|
|
215
|
+
description=(
|
|
216
|
+
"Alternative names for natural-language matching, e.g. "
|
|
217
|
+
"['registration date', 'created'] for signup_date."
|
|
218
|
+
),
|
|
219
|
+
)
|
|
220
|
+
role: AttributeRole = AttributeRole.OTHER
|
|
221
|
+
|
|
222
|
+
description_auto: str = ""
|
|
223
|
+
description_human: str = ""
|
|
224
|
+
|
|
225
|
+
# Semantic typing (distinct from the column's SQL data_type)
|
|
226
|
+
semantic_type: SemanticType | None = Field(
|
|
227
|
+
default=None,
|
|
228
|
+
description="What the values mean semantically (currency, percentage, …).",
|
|
229
|
+
)
|
|
230
|
+
unit: str | None = Field(
|
|
231
|
+
default=None,
|
|
232
|
+
description="Unit for this attribute's values: 'EUR', '%', 'seconds', 'days', 'm'.",
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
# For MEASURE role
|
|
236
|
+
default_aggregation: str | None = Field(
|
|
237
|
+
default=None,
|
|
238
|
+
description="Default SQL aggregation: SUM, AVG, COUNT, MIN, MAX.",
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
# For DIMENSION role — populated when cardinality is low enough to enumerate
|
|
242
|
+
known_values: list[str] = Field(
|
|
243
|
+
default_factory=list,
|
|
244
|
+
max_length=50,
|
|
245
|
+
description="Known categorical values for low-cardinality dimensions.",
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
is_sensitive: bool = False # propagated from Column; suppressed in agent context
|
|
249
|
+
tags: list[str] = Field(default_factory=list)
|
|
250
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
251
|
+
|
|
252
|
+
@model_validator(mode="after")
|
|
253
|
+
def _check_column_xor_expression(self) -> EntityAttribute:
|
|
254
|
+
has_column = self.column_id is not None
|
|
255
|
+
has_expression = self.expression is not None and self.expression.strip() != ""
|
|
256
|
+
if has_column == has_expression:
|
|
257
|
+
raise ValueError(
|
|
258
|
+
"EntityAttribute must have exactly one of column_id or expression set "
|
|
259
|
+
"(column-backed attributes use column_id; computed attributes use expression)."
|
|
260
|
+
)
|
|
261
|
+
if has_column and self.source_id is None:
|
|
262
|
+
raise ValueError("Column-backed attributes must also set source_id.")
|
|
263
|
+
return self
|
|
264
|
+
|
|
265
|
+
@property
|
|
266
|
+
def description(self) -> str:
|
|
267
|
+
return self.description_human or self.description_auto
|
|
268
|
+
|
|
269
|
+
@property
|
|
270
|
+
def is_computed(self) -> bool:
|
|
271
|
+
return self.expression is not None and self.column_id is None
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
class RelationshipCardinality(str, Enum):
|
|
275
|
+
ONE_TO_ONE = "one_to_one"
|
|
276
|
+
ONE_TO_MANY = "one_to_many"
|
|
277
|
+
MANY_TO_ONE = "many_to_one"
|
|
278
|
+
MANY_TO_MANY = "many_to_many"
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
class RelationshipDiscovery(str, Enum):
|
|
282
|
+
"""How this relationship was discovered."""
|
|
283
|
+
DECLARED_FK = "declared_fk" # explicit foreign key in schema
|
|
284
|
+
INFERRED_NAME = "inferred_name" # column name heuristic
|
|
285
|
+
INFERRED_SAMPLE = "inferred_sample" # sample value overlap
|
|
286
|
+
HUMAN_DEFINED = "human_defined"
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
class Relationship(BaseModel):
|
|
290
|
+
"""
|
|
291
|
+
A directed relationship between two Entities.
|
|
292
|
+
|
|
293
|
+
Relationships are the edges of the ontology graph. They tell the query
|
|
294
|
+
engine how to JOIN entities, and tell agents which cross-entity questions
|
|
295
|
+
are answerable.
|
|
296
|
+
|
|
297
|
+
Direction convention:
|
|
298
|
+
source_entity (many side) → target_entity (one side)
|
|
299
|
+
e.g. Order → Customer (many orders per customer)
|
|
300
|
+
|
|
301
|
+
For MANY_TO_MANY relationships realised through an associative table
|
|
302
|
+
(e.g. Order ↔ Product via OrderItem), set `via_entity_id` to the
|
|
303
|
+
associative Entity. The query engine then knows to insert that table
|
|
304
|
+
into the JOIN path.
|
|
305
|
+
"""
|
|
306
|
+
id: UUID = Field(default_factory=uuid4)
|
|
307
|
+
slug: str
|
|
308
|
+
name: str = Field(description="Forward-direction label, e.g. 'Order belongs to Customer'.")
|
|
309
|
+
inverse_name: str | None = Field(
|
|
310
|
+
default=None,
|
|
311
|
+
description=(
|
|
312
|
+
"Reverse-direction label read from the target side, e.g. "
|
|
313
|
+
"'Customer has Orders'. When omitted, callers should fall back to `name`."
|
|
314
|
+
),
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
source_entity_id: UUID
|
|
318
|
+
target_entity_id: UUID
|
|
319
|
+
|
|
320
|
+
# Which attributes form the join key
|
|
321
|
+
source_attribute_id: UUID = Field(description="FK attribute on the source entity.")
|
|
322
|
+
target_attribute_id: UUID = Field(description="PK attribute on the target entity.")
|
|
323
|
+
|
|
324
|
+
cardinality: RelationshipCardinality = RelationshipCardinality.MANY_TO_ONE
|
|
325
|
+
via_entity_id: UUID | None = Field(
|
|
326
|
+
default=None,
|
|
327
|
+
description=(
|
|
328
|
+
"For MANY_TO_MANY relationships, the associative entity that joins source and target "
|
|
329
|
+
"(e.g. OrderItem between Order and Product). None for direct relationships."
|
|
330
|
+
),
|
|
331
|
+
)
|
|
332
|
+
discovery: RelationshipDiscovery = RelationshipDiscovery.INFERRED_NAME
|
|
333
|
+
confidence: float = Field(default=1.0, ge=0.0, le=1.0)
|
|
334
|
+
|
|
335
|
+
is_confirmed: bool = False
|
|
336
|
+
description: str = ""
|
|
337
|
+
|
|
338
|
+
created_at: datetime = Field(default_factory=lambda: datetime.now(timezone.utc))
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: nomox-semantics
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Semantic data model for LLM-consumable data catalog - shared contract across Nomox services.
|
|
5
|
+
Author-email: nomox <admin@get-nomox.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Classifier: Development Status :: 3 - Alpha
|
|
8
|
+
Classifier: Intended Audience :: Developers
|
|
9
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
10
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
12
|
+
Requires-Python: >=3.11
|
|
13
|
+
Requires-Dist: pydantic<3,>=2.7
|
|
14
|
+
Provides-Extra: dev
|
|
15
|
+
Requires-Dist: mypy>=1; extra == 'dev'
|
|
16
|
+
Requires-Dist: pytest-cov>=4; extra == 'dev'
|
|
17
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
18
|
+
Requires-Dist: ruff>=0.1; extra == 'dev'
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
# Nomox Semantic Model
|
|
22
|
+
|
|
23
|
+
The shared knowledge contract between every Nomox service. Describes what data exists (physical layer) and what it means (ontology), and ties the two together so an agent can move from a natural-language concept all the way down to a specific column in a specific table.
|
|
24
|
+
|
|
25
|
+
## Installation
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
pip install git+https://${GITHUB_TOKEN}@github.com/Nomox-ai/semantics-model.git
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Pin to a tag for reproducible installs:
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
pip install git+https://${GITHUB_TOKEN}@github.com/Nomox-ai/semantics-model.git@v0.1.0
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Quickstart
|
|
38
|
+
|
|
39
|
+
```python
|
|
40
|
+
from uuid import uuid4
|
|
41
|
+
from nomox_semantics import (
|
|
42
|
+
SemanticCatalog, DataSource, SourceType, Table, Column, ColumnDataType,
|
|
43
|
+
Entity, EntityAttribute, EntityTableMapping, AttributeRole, SemanticType,
|
|
44
|
+
)
|
|
45
|
+
|
|
46
|
+
# 1. Physical layer — what exists in the database
|
|
47
|
+
src = DataSource(slug="prod", name="Prod", source_type=SourceType.POSTGRES)
|
|
48
|
+
users = Table(source_id=src.id, schema_name="public", table_name="users")
|
|
49
|
+
ltv = Column(table_id=users.id, source_id=src.id, column_name="ltv_cents",
|
|
50
|
+
ordinal_position=1, data_type=ColumnDataType.BIGINT, raw_data_type="bigint")
|
|
51
|
+
|
|
52
|
+
# 2. Semantic layer — what the data means in business terms
|
|
53
|
+
customer = Entity(
|
|
54
|
+
slug="customer", name="Customer", plural_name="Customers",
|
|
55
|
+
aliases=["user", "account holder"], # for natural-language matching
|
|
56
|
+
table_mappings=[EntityTableMapping(table_id=users.id, source_id=src.id)], # entity -> table
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
# An EntityAttribute binds a business field to a physical column
|
|
60
|
+
ltv_attr = EntityAttribute(
|
|
61
|
+
entity_id=customer.id, column_id=ltv.id, source_id=src.id,
|
|
62
|
+
name="lifetime_value", display_name="Lifetime Value",
|
|
63
|
+
role=AttributeRole.MEASURE, default_aggregation="SUM", # tells the query engine it's aggregatable
|
|
64
|
+
semantic_type=SemanticType.CURRENCY, unit="EUR", # values are EUR, not raw cents
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
# 3. Assemble the catalog — the root object every Nomox service consumes
|
|
68
|
+
catalog = SemanticCatalog(organisation_id=uuid4(), name="Acme")
|
|
69
|
+
catalog.sources[src.id] = src
|
|
70
|
+
catalog.tables[users.id] = users
|
|
71
|
+
catalog.columns[ltv.id] = ltv
|
|
72
|
+
catalog.entities[customer.id] = customer
|
|
73
|
+
catalog.attributes[ltv_attr.id] = ltv_attr
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
## Documentation
|
|
77
|
+
|
|
78
|
+
See [docs/](docs/) for the full design documentation.
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
nomox_semantics/__init__.py,sha256=NQFTlXXgjyowB8ozD64G3_cUj9eQBrhLxR-THzalsNg,1901
|
|
2
|
+
nomox_semantics/catalog.py,sha256=uvvvZCJ-c_E-6x8BnWeK1GOy9IKCvS5W-F_4kocoPDo,5320
|
|
3
|
+
nomox_semantics/glossary.py,sha256=EAkGys6ui9YvkggZPNxWLwmP7PFuktx38jdj6vIt26k,2048
|
|
4
|
+
nomox_semantics/metric.py,sha256=Rmwfqe0-dTS4sPzmsPFgGEA3adwWsCae5SnXuH66n-A,5935
|
|
5
|
+
nomox_semantics/physical.py,sha256=lpi1uMhYhKbA0GN518rfF04fKK3qV_gwk3_akKdVx24,5361
|
|
6
|
+
nomox_semantics/query.py,sha256=xNh36QfM-8hVMDSt-HpyiRYqGi3NYE3jbb4oi7RFQFU,6581
|
|
7
|
+
nomox_semantics/semantic.py,sha256=d8eXVwcmxklPL6-0Lt8KKVkSQtbHhzNrvqz3PGVEejU,13048
|
|
8
|
+
nomox_semantics-0.1.0.dist-info/METADATA,sha256=p-Y8QNUAr_J9fVuMZcfwIRU17-Ex1Z5BXbDTttYLnRE,3051
|
|
9
|
+
nomox_semantics-0.1.0.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
|
|
10
|
+
nomox_semantics-0.1.0.dist-info/RECORD,,
|