groundstore 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ """Shared mapping-task contracts and persistence for Groundworkers workflows."""
2
+
3
+ from .context import MappingEvidencePacket, MappingReadContext, MappingReviewHandoff
4
+ from .contracts import (
5
+ DecisionStatus,
6
+ LifecycleStatus,
7
+ MappingCandidateSpec,
8
+ MappingDecisionSpec,
9
+ MappingEvidenceSpec,
10
+ MappingInputSpec,
11
+ MappingRunSpec,
12
+ )
13
+ from .engine import create_groundstore_engine, create_schema
14
+ from .store import MappingStore
15
+
16
+ __all__ = [
17
+ "DecisionStatus",
18
+ "LifecycleStatus",
19
+ "MappingCandidateSpec",
20
+ "MappingDecisionSpec",
21
+ "MappingEvidencePacket",
22
+ "MappingEvidenceSpec",
23
+ "MappingInputSpec",
24
+ "MappingReadContext",
25
+ "MappingReviewHandoff",
26
+ "MappingRunSpec",
27
+ "MappingStore",
28
+ "create_groundstore_engine",
29
+ "create_schema",
30
+ ]
groundstore/context.py ADDED
@@ -0,0 +1,245 @@
1
+ """Read-only views over persisted mapping work.
2
+
3
+ The persistence API remains useful to standalone writers. This module adds
4
+ the small, transport-neutral packet that a host or review client can consume
5
+ without depending on SQLAlchemy model instances or lazy relationships.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+ from typing import Any
12
+
13
+ from pydantic import BaseModel, ConfigDict, Field
14
+
15
+ from .models import (
16
+ MappingCandidate,
17
+ MappingDecision,
18
+ MappingDecisionEvent,
19
+ MappingEvidence,
20
+ MappingInput,
21
+ MappingRun,
22
+ )
23
+ from .store import MappingStore
24
+
25
+
26
+ class MappingEvidencePacket(BaseModel):
27
+ """Stable JSON contract for reviewing one persisted mapping input."""
28
+
29
+ model_config = ConfigDict(extra="forbid")
30
+
31
+ schema_version: str = "groundstore.mapping-evidence-packet.v1"
32
+ run: dict[str, Any]
33
+ input: dict[str, Any]
34
+ candidates: list[dict[str, Any]] = Field(default_factory=list)
35
+ evidence: list[dict[str, Any]] = Field(default_factory=list)
36
+ decision: dict[str, Any] | None = None
37
+ warnings: list[str] = Field(default_factory=list)
38
+
39
+
40
+ class MappingReviewHandoff(BaseModel):
41
+ """Review-task reference carrying a self-contained evidence packet."""
42
+
43
+ model_config = ConfigDict(extra="forbid")
44
+
45
+ schema_version: str = "groundstore.mapping-review-handoff.v1"
46
+ task_id: str = Field(min_length=1)
47
+ task_type: str = "mapping_review"
48
+ source_namespace: str = Field(min_length=1)
49
+ input_id: str = Field(min_length=1)
50
+ decision_status: str = "needs_review"
51
+ packet: MappingEvidencePacket
52
+ requested_by: str | None = None
53
+ metadata: dict[str, Any] = Field(default_factory=dict)
54
+
55
+
56
+ @dataclass(frozen=True, slots=True)
57
+ class MappingReadContext:
58
+ """Read-only Groundstore façade for host tools and review consumers."""
59
+
60
+ store: MappingStore
61
+
62
+ def status(
63
+ self,
64
+ source_namespace: str,
65
+ *,
66
+ target_system: str | None = None,
67
+ ) -> dict[str, Any]:
68
+ """Return the latest complete run and its common coverage summary."""
69
+
70
+ run = self.store.latest_successful_run(source_namespace, target_system=target_system)
71
+ if run is None:
72
+ return {
73
+ "source_namespace": source_namespace,
74
+ "target_system": target_system,
75
+ "latest_run": None,
76
+ "coverage": None,
77
+ }
78
+ return {
79
+ "source_namespace": source_namespace,
80
+ "target_system": target_system,
81
+ "latest_run": _run_payload(run),
82
+ "coverage": self.store.coverage(run.id),
83
+ }
84
+
85
+ def evidence_packet(self, input_id: str) -> MappingEvidencePacket:
86
+ """Build a detached packet with candidates, evidence, and decision history."""
87
+
88
+ input_record = self.store.get_input(input_id)
89
+ if input_record is None:
90
+ raise KeyError(f"unknown mapping input: {input_id}")
91
+ run = self.store.get_run(input_record.run_id)
92
+ if run is None: # pragma: no cover - protected by the database FK
93
+ raise KeyError(f"unknown mapping run: {input_record.run_id}")
94
+
95
+ candidates = self.store.get_candidates(input_id)
96
+ evidence = self.store.get_evidence(input_id)
97
+ decision = self.store.latest_decision(input_id)
98
+ selected_ids = (
99
+ self.store.get_decision_candidate_ids(decision.id) if decision is not None else []
100
+ )
101
+ events = self.store.get_decision_history(decision.id) if decision is not None else []
102
+ evidence_by_candidate: dict[str, list[dict[str, Any]]] = {}
103
+ unattached: list[dict[str, Any]] = []
104
+ for item in evidence:
105
+ payload = _evidence_payload(item)
106
+ if item.candidate_id is None:
107
+ unattached.append(payload)
108
+ else:
109
+ evidence_by_candidate.setdefault(item.candidate_id, []).append(payload)
110
+
111
+ candidate_payloads = []
112
+ for candidate in candidates:
113
+ payload = _candidate_payload(candidate)
114
+ payload["evidence"] = evidence_by_candidate.get(candidate.id, [])
115
+ candidate_payloads.append(payload)
116
+
117
+ decision_payload = None
118
+ if decision is not None:
119
+ decision_payload = _decision_payload(decision)
120
+ decision_payload["selected_candidate_ids"] = selected_ids
121
+ decision_payload["history"] = [_event_payload(event) for event in events]
122
+
123
+ return MappingEvidencePacket(
124
+ run=_run_payload(run),
125
+ input=_input_payload(input_record),
126
+ candidates=candidate_payloads,
127
+ evidence=unattached,
128
+ decision=decision_payload,
129
+ )
130
+
131
+ def review_handoff(
132
+ self,
133
+ input_id: str,
134
+ *,
135
+ requested_by: str | None = None,
136
+ metadata: dict[str, Any] | None = None,
137
+ ) -> MappingReviewHandoff:
138
+ """Return a deterministic review reference without mutating the store."""
139
+
140
+ packet = self.evidence_packet(input_id)
141
+ decision_status = (
142
+ packet.decision.get("decision_status", "needs_review")
143
+ if packet.decision
144
+ else "needs_review"
145
+ )
146
+ return MappingReviewHandoff(
147
+ task_id=f"mapping-review:{input_id}",
148
+ source_namespace=packet.input["source_namespace"],
149
+ input_id=input_id,
150
+ decision_status=str(decision_status),
151
+ packet=packet,
152
+ requested_by=requested_by,
153
+ metadata=metadata or {},
154
+ )
155
+
156
+
157
+ def _run_payload(run: MappingRun) -> dict[str, Any]:
158
+ return {
159
+ "id": run.id,
160
+ "source_namespace": run.source_namespace,
161
+ "source_fingerprint": run.source_fingerprint,
162
+ "source_snapshot": run.source_snapshot,
163
+ "target_system": run.target_system,
164
+ "target_release": run.target_release,
165
+ "algorithm_version": run.algorithm_version,
166
+ "policy_version": run.policy_version,
167
+ "lifecycle_status": run.lifecycle_status,
168
+ "last_error": run.last_error,
169
+ "created_at": run.created_at.isoformat(),
170
+ "updated_at": run.updated_at.isoformat(),
171
+ }
172
+
173
+
174
+ def _input_payload(input_record: MappingInput) -> dict[str, Any]:
175
+ return {
176
+ "id": input_record.id,
177
+ "run_id": input_record.run_id,
178
+ "source_namespace": input_record.source_namespace,
179
+ "source_kind": input_record.source_kind,
180
+ "source_key": input_record.source_key,
181
+ "source_fingerprint": input_record.source_fingerprint,
182
+ "normalized_projection": input_record.normalized_projection,
183
+ "lifecycle_status": input_record.lifecycle_status,
184
+ "retry_count": input_record.retry_count,
185
+ "last_error": input_record.last_error,
186
+ "created_at": input_record.created_at.isoformat(),
187
+ "updated_at": input_record.updated_at.isoformat(),
188
+ }
189
+
190
+
191
+ def _candidate_payload(candidate: MappingCandidate) -> dict[str, Any]:
192
+ return {
193
+ "id": candidate.id,
194
+ "target_namespace": candidate.target_namespace,
195
+ "target_vocabulary_id": candidate.target_vocabulary_id,
196
+ "target_concept_id": candidate.target_concept_id,
197
+ "target_code": candidate.target_code,
198
+ "target_grain": candidate.target_grain,
199
+ "target_role": candidate.target_role,
200
+ "method": candidate.method,
201
+ "rank": candidate.rank,
202
+ "score": candidate.score,
203
+ "confidence": candidate.confidence,
204
+ "rationale": candidate.rationale,
205
+ "metadata": candidate.metadata_,
206
+ "created_at": candidate.created_at.isoformat(),
207
+ }
208
+
209
+
210
+ def _evidence_payload(evidence: MappingEvidence) -> dict[str, Any]:
211
+ return {
212
+ "id": evidence.id,
213
+ "candidate_id": evidence.candidate_id,
214
+ "decision_id": evidence.decision_id,
215
+ "evidence_key": evidence.evidence_key,
216
+ "evidence_type": evidence.evidence_type,
217
+ "source_reference": evidence.source_reference,
218
+ "method": evidence.method,
219
+ "payload": evidence.payload,
220
+ "created_at": evidence.created_at.isoformat(),
221
+ }
222
+
223
+
224
+ def _decision_payload(decision: MappingDecision) -> dict[str, Any]:
225
+ return {
226
+ "id": decision.id,
227
+ "input_id": decision.input_id,
228
+ "decision_version": decision.decision_version,
229
+ "decision_status": decision.decision_status,
230
+ "outcome_code": decision.outcome_code,
231
+ "reason_codes": decision.reason_codes,
232
+ "decided_by": decision.decided_by,
233
+ "metadata": decision.metadata_,
234
+ "decided_at": decision.decided_at.isoformat(),
235
+ }
236
+
237
+
238
+ def _event_payload(event: MappingDecisionEvent) -> dict[str, Any]:
239
+ return {
240
+ "id": event.id,
241
+ "decision_id": event.decision_id,
242
+ "event_type": event.event_type,
243
+ "detail": event.detail,
244
+ "created_at": event.created_at.isoformat(),
245
+ }
@@ -0,0 +1,108 @@
1
+ """Source-independent contracts for mapping workflows."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from enum import StrEnum
6
+ from typing import Any
7
+
8
+ from pydantic import BaseModel, ConfigDict, Field, field_validator
9
+
10
+
11
+ class LifecycleStatus(StrEnum):
12
+ """Processing state for a run or input."""
13
+
14
+ PENDING = "pending"
15
+ IN_PROGRESS = "in_progress"
16
+ COMPLETE = "complete"
17
+ INCOMPLETE = "incomplete"
18
+ FAILED = "failed"
19
+
20
+
21
+ class DecisionStatus(StrEnum):
22
+ """Mapping outcome independent of the processing lifecycle."""
23
+
24
+ MAPPED = "mapped"
25
+ AMBIGUOUS = "ambiguous"
26
+ UNMAPPABLE = "unmappable"
27
+ NEEDS_REVIEW = "needs_review"
28
+
29
+
30
+ class MappingRunSpec(BaseModel):
31
+ """Stable identity and metadata for one mapping run."""
32
+
33
+ model_config = ConfigDict(extra="forbid")
34
+
35
+ source_namespace: str = Field(min_length=1)
36
+ source_fingerprint: str = Field(min_length=1, max_length=128)
37
+ source_snapshot: dict[str, Any] = Field(default_factory=dict)
38
+ target_system: str = Field(min_length=1)
39
+ target_release: str | None = None
40
+ algorithm_version: str = Field(min_length=1)
41
+ policy_version: str = Field(min_length=1)
42
+ lifecycle_status: LifecycleStatus = LifecycleStatus.PENDING
43
+
44
+
45
+ class MappingInputSpec(BaseModel):
46
+ """Source item presented to a mapping workflow."""
47
+
48
+ model_config = ConfigDict(extra="forbid")
49
+
50
+ source_namespace: str = Field(min_length=1)
51
+ source_kind: str = Field(min_length=1)
52
+ source_key: str = Field(min_length=1)
53
+ source_fingerprint: str = Field(min_length=1, max_length=128)
54
+ normalized_projection: dict[str, Any] = Field(default_factory=dict)
55
+ lifecycle_status: LifecycleStatus = LifecycleStatus.PENDING
56
+ retry_count: int = Field(default=0, ge=0)
57
+ last_error: str | None = None
58
+
59
+
60
+ class MappingCandidateSpec(BaseModel):
61
+ """One candidate target for a mapping input."""
62
+
63
+ model_config = ConfigDict(extra="forbid")
64
+
65
+ target_namespace: str = Field(min_length=1)
66
+ target_vocabulary_id: str | None = None
67
+ target_concept_id: str | None = None
68
+ target_code: str | None = None
69
+ target_grain: str | None = None
70
+ target_role: str | None = None
71
+ method: str = Field(min_length=1)
72
+ rank: int = Field(default=1, ge=1)
73
+ score: float | None = None
74
+ confidence: float | None = Field(default=None, ge=0, le=1)
75
+ rationale: str | None = None
76
+ metadata: dict[str, Any] = Field(default_factory=dict)
77
+
78
+ @field_validator("target_vocabulary_id", "target_concept_id", "target_code")
79
+ @classmethod
80
+ def reject_blank_identifiers(cls, value: str | None) -> str | None:
81
+ if value is not None and not value.strip():
82
+ raise ValueError("target identifiers cannot be blank")
83
+ return value
84
+
85
+
86
+ class MappingEvidenceSpec(BaseModel):
87
+ """Evidence attached to a candidate or decision."""
88
+
89
+ model_config = ConfigDict(extra="forbid")
90
+
91
+ evidence_key: str = Field(min_length=1, max_length=128)
92
+ evidence_type: str = Field(min_length=1)
93
+ source_reference: str | None = None
94
+ method: str | None = None
95
+ payload: dict[str, Any] = Field(default_factory=dict)
96
+
97
+
98
+ class MappingDecisionSpec(BaseModel):
99
+ """Versioned decision for one mapping input."""
100
+
101
+ model_config = ConfigDict(extra="forbid")
102
+
103
+ decision_status: DecisionStatus
104
+ selected_candidate_ids: list[str] = Field(default_factory=list)
105
+ outcome_code: str | None = None
106
+ reason_codes: list[str] = Field(default_factory=list)
107
+ decided_by: str | None = None
108
+ metadata: dict[str, Any] = Field(default_factory=dict)
groundstore/engine.py ADDED
@@ -0,0 +1,64 @@
1
+ """Engine and schema helpers, including oa-configurator resolution."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from collections.abc import Mapping
6
+ from typing import Any
7
+
8
+ import sqlalchemy as sa
9
+ from oa_configurator import ResolvedDatabase, Resolver
10
+ from sqlalchemy import Engine, event
11
+ from sqlalchemy.engine import URL
12
+
13
+ from .models import Base
14
+
15
+
16
+ def create_groundstore_engine(
17
+ url: str | URL | None = None,
18
+ *,
19
+ resolved_database: ResolvedDatabase | None = None,
20
+ resolver: Resolver | None = None,
21
+ database_name: str = "mapping_db",
22
+ execution_options: Mapping[str, Any] | None = None,
23
+ **engine_kwargs: Any,
24
+ ) -> Engine:
25
+ """Create a store engine from an explicit URL or OA database resource."""
26
+ if sum(value is not None for value in (url, resolved_database, resolver)) > 1:
27
+ raise ValueError("provide only one of url, resolved_database, or resolver")
28
+
29
+ if resolved_database is not None:
30
+ engine = resolved_database.create_engine(**engine_kwargs)
31
+ elif resolver is not None:
32
+ engine = resolver.resolve_database(database_name).create_engine(**engine_kwargs)
33
+ elif url is not None:
34
+ engine = sa.create_engine(url, **engine_kwargs)
35
+ else:
36
+ engine = (
37
+ Resolver.from_active_config()
38
+ .resolve_database(database_name)
39
+ .create_engine(**engine_kwargs)
40
+ )
41
+
42
+ if execution_options:
43
+ engine = engine.execution_options(**dict(execution_options))
44
+ if engine.dialect.name == "sqlite":
45
+ _enable_sqlite_foreign_keys(engine)
46
+ return engine
47
+
48
+
49
+ def create_schema(engine: Engine) -> None:
50
+ """Create all groundstore tables if they do not already exist."""
51
+ Base.metadata.create_all(engine)
52
+
53
+
54
+ def drop_schema(engine: Engine) -> None:
55
+ """Drop all groundstore tables; intended for isolated test databases."""
56
+ Base.metadata.drop_all(engine)
57
+
58
+
59
+ def _enable_sqlite_foreign_keys(engine: Engine) -> None:
60
+ @event.listens_for(engine, "connect")
61
+ def _set_foreign_keys(dbapi_connection: Any, _connection_record: Any) -> None:
62
+ cursor = dbapi_connection.cursor()
63
+ cursor.execute("PRAGMA foreign_keys=ON")
64
+ cursor.close()
groundstore/models.py ADDED
@@ -0,0 +1,219 @@
1
+ """SQLAlchemy persistence models for the shared mapping contract."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from datetime import datetime
6
+ from typing import Any
7
+
8
+ from sqlalchemy import (
9
+ JSON,
10
+ DateTime,
11
+ Float,
12
+ ForeignKey,
13
+ Index,
14
+ Integer,
15
+ String,
16
+ Text,
17
+ UniqueConstraint,
18
+ )
19
+ from sqlalchemy.orm import DeclarativeBase, Mapped, mapped_column, relationship
20
+
21
+
22
+ class Base(DeclarativeBase):
23
+ """Declarative base for groundstore tables."""
24
+
25
+
26
+ class MappingRun(Base):
27
+ __tablename__ = "mapping_runs"
28
+ __table_args__ = (
29
+ UniqueConstraint(
30
+ "source_namespace",
31
+ "source_fingerprint",
32
+ "target_system",
33
+ "target_release",
34
+ "algorithm_version",
35
+ "policy_version",
36
+ name="uq_mapping_run_identity",
37
+ ),
38
+ Index("ix_mapping_runs_source_status", "source_namespace", "lifecycle_status"),
39
+ )
40
+
41
+ id: Mapped[str] = mapped_column(String(36), primary_key=True)
42
+ source_namespace: Mapped[str] = mapped_column(String(100), nullable=False)
43
+ source_fingerprint: Mapped[str] = mapped_column(String(128), nullable=False)
44
+ source_snapshot: Mapped[dict[str, Any]] = mapped_column(JSON, nullable=False, default=dict)
45
+ target_system: Mapped[str] = mapped_column(String(100), nullable=False)
46
+ target_release: Mapped[str | None] = mapped_column(String(100))
47
+ algorithm_version: Mapped[str] = mapped_column(String(100), nullable=False)
48
+ policy_version: Mapped[str] = mapped_column(String(100), nullable=False)
49
+ lifecycle_status: Mapped[str] = mapped_column(String(20), nullable=False)
50
+ last_error: Mapped[str | None] = mapped_column(Text)
51
+ created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
52
+ updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
53
+
54
+ inputs: Mapped[list[MappingInput]] = relationship(
55
+ back_populates="run", cascade="all, delete-orphan"
56
+ )
57
+
58
+
59
+ class MappingInput(Base):
60
+ __tablename__ = "mapping_inputs"
61
+ __table_args__ = (
62
+ UniqueConstraint(
63
+ "run_id",
64
+ "source_namespace",
65
+ "source_kind",
66
+ "source_key",
67
+ "source_fingerprint",
68
+ name="uq_mapping_input_identity",
69
+ ),
70
+ Index("ix_mapping_inputs_run_status", "run_id", "lifecycle_status"),
71
+ )
72
+
73
+ id: Mapped[str] = mapped_column(String(36), primary_key=True)
74
+ run_id: Mapped[str] = mapped_column(
75
+ ForeignKey("mapping_runs.id", ondelete="CASCADE"), nullable=False
76
+ )
77
+ source_namespace: Mapped[str] = mapped_column(String(100), nullable=False)
78
+ source_kind: Mapped[str] = mapped_column(String(100), nullable=False)
79
+ source_key: Mapped[str] = mapped_column(String(255), nullable=False)
80
+ source_fingerprint: Mapped[str] = mapped_column(String(128), nullable=False)
81
+ normalized_projection: Mapped[dict[str, Any]] = mapped_column(
82
+ JSON, nullable=False, default=dict
83
+ )
84
+ lifecycle_status: Mapped[str] = mapped_column(String(20), nullable=False)
85
+ retry_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
86
+ last_error: Mapped[str | None] = mapped_column(Text)
87
+ created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
88
+ updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
89
+
90
+ run: Mapped[MappingRun] = relationship(back_populates="inputs")
91
+ candidates: Mapped[list[MappingCandidate]] = relationship(
92
+ back_populates="input", cascade="all, delete-orphan"
93
+ )
94
+ evidence: Mapped[list[MappingEvidence]] = relationship(
95
+ back_populates="input", cascade="all, delete-orphan"
96
+ )
97
+ decisions: Mapped[list[MappingDecision]] = relationship(
98
+ back_populates="input", cascade="all, delete-orphan"
99
+ )
100
+
101
+
102
+ class MappingCandidate(Base):
103
+ __tablename__ = "mapping_candidates"
104
+ __table_args__ = (
105
+ UniqueConstraint("input_id", "candidate_key", name="uq_mapping_candidate_key"),
106
+ )
107
+
108
+ id: Mapped[str] = mapped_column(String(36), primary_key=True)
109
+ input_id: Mapped[str] = mapped_column(
110
+ ForeignKey("mapping_inputs.id", ondelete="CASCADE"), nullable=False
111
+ )
112
+ candidate_key: Mapped[str] = mapped_column(String(128), nullable=False)
113
+ target_namespace: Mapped[str] = mapped_column(String(100), nullable=False)
114
+ target_vocabulary_id: Mapped[str | None] = mapped_column(String(100))
115
+ target_concept_id: Mapped[str | None] = mapped_column(String(100))
116
+ target_code: Mapped[str | None] = mapped_column(String(255))
117
+ target_grain: Mapped[str | None] = mapped_column(String(100))
118
+ target_role: Mapped[str | None] = mapped_column(String(100))
119
+ method: Mapped[str] = mapped_column(String(100), nullable=False)
120
+ rank: Mapped[int] = mapped_column(Integer, nullable=False)
121
+ score: Mapped[float | None] = mapped_column(Float)
122
+ confidence: Mapped[float | None] = mapped_column(Float)
123
+ rationale: Mapped[str | None] = mapped_column(Text)
124
+ metadata_: Mapped[dict[str, Any]] = mapped_column(
125
+ "metadata", JSON, nullable=False, default=dict
126
+ )
127
+ created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
128
+
129
+ input: Mapped[MappingInput] = relationship(back_populates="candidates")
130
+ evidence: Mapped[list[MappingEvidence]] = relationship(back_populates="candidate")
131
+ decisions: Mapped[list[MappingDecision]] = relationship(
132
+ secondary="mapping_decision_candidates", back_populates="selected_candidates"
133
+ )
134
+
135
+
136
+ class MappingEvidence(Base):
137
+ __tablename__ = "mapping_evidence"
138
+ __table_args__ = (
139
+ UniqueConstraint("input_id", "evidence_key", name="uq_mapping_evidence_key"),
140
+ Index("ix_mapping_evidence_candidate", "candidate_id"),
141
+ Index("ix_mapping_evidence_decision", "decision_id"),
142
+ )
143
+
144
+ id: Mapped[str] = mapped_column(String(36), primary_key=True)
145
+ input_id: Mapped[str] = mapped_column(
146
+ ForeignKey("mapping_inputs.id", ondelete="CASCADE"), nullable=False
147
+ )
148
+ candidate_id: Mapped[str | None] = mapped_column(
149
+ ForeignKey("mapping_candidates.id", ondelete="CASCADE")
150
+ )
151
+ decision_id: Mapped[str | None] = mapped_column(
152
+ ForeignKey("mapping_decisions.id", ondelete="CASCADE")
153
+ )
154
+ evidence_key: Mapped[str] = mapped_column(String(128), nullable=False)
155
+ evidence_type: Mapped[str] = mapped_column(String(100), nullable=False)
156
+ source_reference: Mapped[str | None] = mapped_column(Text)
157
+ method: Mapped[str | None] = mapped_column(String(100))
158
+ payload: Mapped[dict[str, Any]] = mapped_column(JSON, nullable=False, default=dict)
159
+ created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
160
+
161
+ input: Mapped[MappingInput] = relationship(back_populates="evidence")
162
+ candidate: Mapped[MappingCandidate | None] = relationship(back_populates="evidence")
163
+ decision: Mapped[MappingDecision | None] = relationship(back_populates="evidence")
164
+
165
+
166
+ class MappingDecision(Base):
167
+ __tablename__ = "mapping_decisions"
168
+ __table_args__ = (
169
+ UniqueConstraint("input_id", "decision_version", name="uq_mapping_decision_version"),
170
+ )
171
+
172
+ id: Mapped[str] = mapped_column(String(36), primary_key=True)
173
+ input_id: Mapped[str] = mapped_column(
174
+ ForeignKey("mapping_inputs.id", ondelete="CASCADE"), nullable=False
175
+ )
176
+ decision_version: Mapped[int] = mapped_column(Integer, nullable=False)
177
+ decision_status: Mapped[str] = mapped_column(String(20), nullable=False)
178
+ outcome_code: Mapped[str | None] = mapped_column(String(150))
179
+ reason_codes: Mapped[list[str]] = mapped_column(JSON, nullable=False, default=list)
180
+ decided_by: Mapped[str | None] = mapped_column(String(255))
181
+ metadata_: Mapped[dict[str, Any]] = mapped_column(
182
+ "metadata", JSON, nullable=False, default=dict
183
+ )
184
+ decided_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
185
+
186
+ input: Mapped[MappingInput] = relationship(back_populates="decisions")
187
+ selected_candidates: Mapped[list[MappingCandidate]] = relationship(
188
+ secondary="mapping_decision_candidates", back_populates="decisions"
189
+ )
190
+ evidence: Mapped[list[MappingEvidence]] = relationship(back_populates="decision")
191
+ history: Mapped[list[MappingDecisionEvent]] = relationship(
192
+ back_populates="decision", cascade="all, delete-orphan"
193
+ )
194
+
195
+
196
+ class MappingDecisionCandidate(Base):
197
+ __tablename__ = "mapping_decision_candidates"
198
+
199
+ decision_id: Mapped[str] = mapped_column(
200
+ ForeignKey("mapping_decisions.id", ondelete="CASCADE"), primary_key=True
201
+ )
202
+ candidate_id: Mapped[str] = mapped_column(
203
+ ForeignKey("mapping_candidates.id", ondelete="CASCADE"), primary_key=True
204
+ )
205
+
206
+
207
+ class MappingDecisionEvent(Base):
208
+ __tablename__ = "mapping_decision_events"
209
+ __table_args__ = (Index("ix_mapping_decision_events_decision", "decision_id", "created_at"),)
210
+
211
+ id: Mapped[str] = mapped_column(String(36), primary_key=True)
212
+ decision_id: Mapped[str] = mapped_column(
213
+ ForeignKey("mapping_decisions.id", ondelete="CASCADE"), nullable=False
214
+ )
215
+ event_type: Mapped[str] = mapped_column(String(100), nullable=False)
216
+ detail: Mapped[dict[str, Any]] = mapped_column(JSON, nullable=False, default=dict)
217
+ created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
218
+
219
+ decision: Mapped[MappingDecision] = relationship(back_populates="history")
groundstore/py.typed ADDED
File without changes
groundstore/store.py ADDED
@@ -0,0 +1,398 @@
1
+ """Repository-style persistence API for mapping workflows."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from collections import Counter
8
+ from datetime import UTC, datetime
9
+ from typing import Any
10
+ from uuid import uuid4
11
+
12
+ from oa_configurator import ResolvedDatabase, Resolver
13
+ from sqlalchemy import Engine, select
14
+ from sqlalchemy.orm import sessionmaker
15
+
16
+ from .contracts import (
17
+ MappingCandidateSpec,
18
+ MappingDecisionSpec,
19
+ MappingEvidenceSpec,
20
+ MappingInputSpec,
21
+ MappingRunSpec,
22
+ )
23
+ from .engine import create_groundstore_engine, create_schema
24
+ from .models import (
25
+ MappingCandidate,
26
+ MappingDecision,
27
+ MappingDecisionCandidate,
28
+ MappingDecisionEvent,
29
+ MappingEvidence,
30
+ MappingInput,
31
+ MappingRun,
32
+ )
33
+
34
+
35
+ class MappingStore:
36
+ """Persist and resume source-independent mapping workflows."""
37
+
38
+ def __init__(self, engine: Engine, *, initialize: bool = True) -> None:
39
+ self.engine = engine
40
+ self._session_factory = sessionmaker(bind=engine, expire_on_commit=False)
41
+ if initialize:
42
+ create_schema(engine)
43
+
44
+ @classmethod
45
+ def from_url(cls, url: str, **engine_kwargs: Any) -> MappingStore:
46
+ return cls(create_groundstore_engine(url, **engine_kwargs))
47
+
48
+ @classmethod
49
+ def from_database(cls, database: ResolvedDatabase, **engine_kwargs: Any) -> MappingStore:
50
+ """Create a store from an already-resolved OA database resource."""
51
+ return cls(create_groundstore_engine(resolved_database=database, **engine_kwargs))
52
+
53
+ @classmethod
54
+ def from_resolver(
55
+ cls, resolver: Resolver, database_name: str = "mapping_db", **engine_kwargs: Any
56
+ ) -> MappingStore:
57
+ engine = create_groundstore_engine(
58
+ resolver=resolver, database_name=database_name, **engine_kwargs
59
+ )
60
+ return cls(engine)
61
+
62
+ def get_or_create_run(self, spec: MappingRunSpec) -> MappingRun:
63
+ """Return the stable run for *spec*, preserving resumable state."""
64
+ with self._session_factory() as session:
65
+ run = session.scalar(
66
+ select(MappingRun).where(
67
+ MappingRun.source_namespace == spec.source_namespace,
68
+ MappingRun.source_fingerprint == spec.source_fingerprint,
69
+ MappingRun.target_system == spec.target_system,
70
+ MappingRun.target_release == spec.target_release,
71
+ MappingRun.algorithm_version == spec.algorithm_version,
72
+ MappingRun.policy_version == spec.policy_version,
73
+ )
74
+ )
75
+ if run is None:
76
+ now = _now()
77
+ run = MappingRun(
78
+ id=_id(),
79
+ **spec.model_dump(exclude={"lifecycle_status"}),
80
+ lifecycle_status=spec.lifecycle_status.value,
81
+ created_at=now,
82
+ updated_at=now,
83
+ )
84
+ session.add(run)
85
+ session.commit()
86
+ return run
87
+
88
+ def update_run(
89
+ self,
90
+ run_id: str,
91
+ *,
92
+ lifecycle_status: str | None = None,
93
+ last_error: str | None = None,
94
+ ) -> MappingRun:
95
+ with self._session_factory() as session:
96
+ run = session.get(MappingRun, run_id)
97
+ if run is None:
98
+ raise KeyError(f"unknown mapping run: {run_id}")
99
+ if lifecycle_status is not None:
100
+ run.lifecycle_status = lifecycle_status
101
+ run.last_error = last_error
102
+ run.updated_at = _now()
103
+ session.commit()
104
+ return run
105
+
106
+ def upsert_input(self, run_id: str, spec: MappingInputSpec) -> MappingInput:
107
+ """Insert or return an input, retaining candidates and decisions on retry."""
108
+ with self._session_factory() as session:
109
+ record = session.scalar(
110
+ select(MappingInput).where(
111
+ MappingInput.run_id == run_id,
112
+ MappingInput.source_namespace == spec.source_namespace,
113
+ MappingInput.source_kind == spec.source_kind,
114
+ MappingInput.source_key == spec.source_key,
115
+ MappingInput.source_fingerprint == spec.source_fingerprint,
116
+ )
117
+ )
118
+ if record is None:
119
+ now = _now()
120
+ record = MappingInput(
121
+ id=_id(),
122
+ run_id=run_id,
123
+ **spec.model_dump(exclude={"lifecycle_status"}),
124
+ lifecycle_status=spec.lifecycle_status.value,
125
+ created_at=now,
126
+ updated_at=now,
127
+ )
128
+ session.add(record)
129
+ session.commit()
130
+ return record
131
+
132
+ def update_input(
133
+ self,
134
+ input_id: str,
135
+ *,
136
+ lifecycle_status: str | None = None,
137
+ retry_count: int | None = None,
138
+ last_error: str | None = None,
139
+ ) -> MappingInput:
140
+ """Update processing state without replacing mapping evidence."""
141
+ with self._session_factory() as session:
142
+ record = session.get(MappingInput, input_id)
143
+ if record is None:
144
+ raise KeyError(f"unknown mapping input: {input_id}")
145
+ if lifecycle_status is not None:
146
+ record.lifecycle_status = lifecycle_status
147
+ if retry_count is not None:
148
+ if retry_count < 0:
149
+ raise ValueError("retry_count cannot be negative")
150
+ record.retry_count = retry_count
151
+ record.last_error = last_error
152
+ record.updated_at = _now()
153
+ session.commit()
154
+ return record
155
+
156
+ def get_candidates(self, input_id: str) -> list[MappingCandidate]:
157
+ """Return candidates for one input in deterministic rank order."""
158
+ with self._session_factory() as session:
159
+ return list(
160
+ session.scalars(
161
+ select(MappingCandidate)
162
+ .where(MappingCandidate.input_id == input_id)
163
+ .order_by(MappingCandidate.rank, MappingCandidate.created_at)
164
+ )
165
+ )
166
+
167
+ def upsert_candidate(self, input_id: str, spec: MappingCandidateSpec) -> MappingCandidate:
168
+ """Insert or return a candidate using a deterministic semantic key."""
169
+ key = _candidate_key(spec)
170
+ with self._session_factory() as session:
171
+ candidate = session.scalar(
172
+ select(MappingCandidate).where(
173
+ MappingCandidate.input_id == input_id,
174
+ MappingCandidate.candidate_key == key,
175
+ )
176
+ )
177
+ if candidate is None:
178
+ candidate = MappingCandidate(
179
+ id=_id(),
180
+ input_id=input_id,
181
+ candidate_key=key,
182
+ **spec.model_dump(exclude={"metadata"}),
183
+ metadata_=spec.metadata,
184
+ created_at=_now(),
185
+ )
186
+ session.add(candidate)
187
+ session.commit()
188
+ return candidate
189
+
190
+ def add_evidence(
191
+ self,
192
+ input_id: str,
193
+ spec: MappingEvidenceSpec,
194
+ *,
195
+ candidate_id: str | None = None,
196
+ decision_id: str | None = None,
197
+ ) -> MappingEvidence:
198
+ """Insert or return evidence, attached to exactly one candidate or decision."""
199
+ if (candidate_id is None) == (decision_id is None):
200
+ raise ValueError("evidence must reference exactly one candidate or decision")
201
+ with self._session_factory() as session:
202
+ if candidate_id is not None:
203
+ candidate = session.scalar(
204
+ select(MappingCandidate).where(
205
+ MappingCandidate.id == candidate_id,
206
+ MappingCandidate.input_id == input_id,
207
+ )
208
+ )
209
+ if candidate is None:
210
+ raise ValueError("candidate must belong to the input")
211
+ if decision_id is not None:
212
+ decision = session.scalar(
213
+ select(MappingDecision).where(
214
+ MappingDecision.id == decision_id,
215
+ MappingDecision.input_id == input_id,
216
+ )
217
+ )
218
+ if decision is None:
219
+ raise ValueError("decision must belong to the input")
220
+ evidence = session.scalar(
221
+ select(MappingEvidence).where(
222
+ MappingEvidence.input_id == input_id,
223
+ MappingEvidence.evidence_key == spec.evidence_key,
224
+ )
225
+ )
226
+ if evidence is None:
227
+ evidence = MappingEvidence(
228
+ id=_id(),
229
+ input_id=input_id,
230
+ candidate_id=candidate_id,
231
+ decision_id=decision_id,
232
+ **spec.model_dump(),
233
+ created_at=_now(),
234
+ )
235
+ session.add(evidence)
236
+ session.commit()
237
+ return evidence
238
+
239
+ def record_decision(self, input_id: str, spec: MappingDecisionSpec) -> MappingDecision:
240
+ """Append a decision version and its immutable history event."""
241
+ with self._session_factory() as session:
242
+ if len(spec.selected_candidate_ids) != len(set(spec.selected_candidate_ids)):
243
+ raise ValueError("selected candidates must be unique")
244
+ selected = list(
245
+ session.scalars(
246
+ select(MappingCandidate).where(
247
+ MappingCandidate.input_id == input_id,
248
+ MappingCandidate.id.in_(spec.selected_candidate_ids),
249
+ )
250
+ )
251
+ )
252
+ if len(selected) != len(set(spec.selected_candidate_ids)):
253
+ raise ValueError("all selected candidates must belong to the input")
254
+ latest = session.scalar(
255
+ select(MappingDecision)
256
+ .where(MappingDecision.input_id == input_id)
257
+ .order_by(MappingDecision.decision_version.desc())
258
+ )
259
+ version = 1 if latest is None else latest.decision_version + 1
260
+ decision = MappingDecision(
261
+ id=_id(),
262
+ input_id=input_id,
263
+ decision_version=version,
264
+ decision_status=spec.decision_status.value,
265
+ outcome_code=spec.outcome_code,
266
+ reason_codes=spec.reason_codes,
267
+ decided_by=spec.decided_by,
268
+ metadata_=spec.metadata,
269
+ decided_at=_now(),
270
+ )
271
+ decision.selected_candidates = selected
272
+ decision.history.append(
273
+ MappingDecisionEvent(
274
+ id=_id(),
275
+ event_type="decision_recorded",
276
+ detail={
277
+ "decision_status": spec.decision_status.value,
278
+ "decision_version": version,
279
+ },
280
+ created_at=_now(),
281
+ )
282
+ )
283
+ session.add(decision)
284
+ session.commit()
285
+ return decision
286
+
287
+ def latest_decision(self, input_id: str) -> MappingDecision | None:
288
+ with self._session_factory() as session:
289
+ return session.scalar(
290
+ select(MappingDecision)
291
+ .where(MappingDecision.input_id == input_id)
292
+ .order_by(MappingDecision.decision_version.desc())
293
+ )
294
+
295
+ def get_input(self, input_id: str) -> MappingInput | None:
296
+ """Return one input without relying on lazy relationships."""
297
+ with self._session_factory() as session:
298
+ return session.get(MappingInput, input_id)
299
+
300
+ def get_evidence(self, input_id: str) -> list[MappingEvidence]:
301
+ """Return all evidence for an input in insertion order."""
302
+ with self._session_factory() as session:
303
+ return list(
304
+ session.scalars(
305
+ select(MappingEvidence)
306
+ .where(MappingEvidence.input_id == input_id)
307
+ .order_by(MappingEvidence.created_at, MappingEvidence.id)
308
+ )
309
+ )
310
+
311
+ def get_decision_candidate_ids(self, decision_id: str) -> list[str]:
312
+ """Return selected candidate IDs in stable candidate order."""
313
+ with self._session_factory() as session:
314
+ rows = session.execute(
315
+ select(MappingDecisionCandidate.candidate_id)
316
+ .where(MappingDecisionCandidate.decision_id == decision_id)
317
+ .join(
318
+ MappingCandidate,
319
+ MappingCandidate.id == MappingDecisionCandidate.candidate_id,
320
+ )
321
+ .order_by(MappingCandidate.rank, MappingCandidate.id)
322
+ )
323
+ return [candidate_id for (candidate_id,) in rows]
324
+
325
+ def get_decision_history(self, decision_id: str) -> list[MappingDecisionEvent]:
326
+ """Return immutable history events for one decision."""
327
+ with self._session_factory() as session:
328
+ return list(
329
+ session.scalars(
330
+ select(MappingDecisionEvent)
331
+ .where(MappingDecisionEvent.decision_id == decision_id)
332
+ .order_by(MappingDecisionEvent.created_at, MappingDecisionEvent.id)
333
+ )
334
+ )
335
+
336
+ def coverage(self, run_id: str) -> dict[str, Any]:
337
+ """Return lifecycle and latest decision counts for a run."""
338
+ with self._session_factory() as session:
339
+ inputs = list(
340
+ session.scalars(select(MappingInput).where(MappingInput.run_id == run_id))
341
+ )
342
+ decisions = []
343
+ for input_record in inputs:
344
+ decision = session.scalar(
345
+ select(MappingDecision)
346
+ .where(MappingDecision.input_id == input_record.id)
347
+ .order_by(MappingDecision.decision_version.desc())
348
+ )
349
+ if decision is not None:
350
+ decisions.append(decision)
351
+ return {
352
+ "input_count": len(inputs),
353
+ "lifecycle_status": dict(Counter(item.lifecycle_status for item in inputs)),
354
+ "decision_status": dict(Counter(item.decision_status for item in decisions)),
355
+ }
356
+
357
+ def get_run(self, run_id: str) -> MappingRun | None:
358
+ with self._session_factory() as session:
359
+ return session.get(MappingRun, run_id)
360
+
361
+ def get_inputs(self, run_id: str) -> list[MappingInput]:
362
+ with self._session_factory() as session:
363
+ return list(session.scalars(select(MappingInput).where(MappingInput.run_id == run_id)))
364
+
365
+ def latest_successful_run(
366
+ self, source_namespace: str, *, target_system: str | None = None
367
+ ) -> MappingRun | None:
368
+ """Return the newest complete run for a source and target system."""
369
+ with self._session_factory() as session:
370
+ query = select(MappingRun).where(
371
+ MappingRun.source_namespace == source_namespace,
372
+ MappingRun.lifecycle_status == "complete",
373
+ )
374
+ if target_system is not None:
375
+ query = query.where(MappingRun.target_system == target_system)
376
+ return session.scalar(query.order_by(MappingRun.updated_at.desc()))
377
+
378
+
379
+ def _id() -> str:
380
+ return str(uuid4())
381
+
382
+
383
+ def _now() -> datetime:
384
+ return datetime.now(UTC)
385
+
386
+
387
+ def _candidate_key(spec: MappingCandidateSpec) -> str:
388
+ payload = {
389
+ "target_namespace": spec.target_namespace,
390
+ "target_vocabulary_id": spec.target_vocabulary_id,
391
+ "target_concept_id": spec.target_concept_id,
392
+ "target_code": spec.target_code,
393
+ "target_grain": spec.target_grain,
394
+ "target_role": spec.target_role,
395
+ "method": spec.method,
396
+ }
397
+ canonical = json.dumps(payload, sort_keys=True, separators=(",", ":"))
398
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
@@ -0,0 +1,14 @@
1
+ Metadata-Version: 2.5
2
+ Name: groundstore
3
+ Version: 0.1.0
4
+ Summary: Shared mapping-task contracts for Groundworkers workflows.
5
+ Requires-Python: >=3.12
6
+ Requires-Dist: oa-configurator<2,>=1.3.0
7
+ Requires-Dist: pydantic<3,>=2
8
+ Requires-Dist: sqlalchemy<3,>=2.0.45
9
+ Provides-Extra: dev
10
+ Requires-Dist: pytest<9,>=8; extra == 'dev'
11
+ Requires-Dist: ruff<1,>=0.4; extra == 'dev'
12
+ Requires-Dist: ty>=0.0.59; extra == 'dev'
13
+ Provides-Extra: postgres
14
+ Requires-Dist: psycopg[binary]<4,>=3.1; extra == 'postgres'
@@ -0,0 +1,10 @@
1
+ groundstore/__init__.py,sha256=wAvMCygyd7SqddIq8vzvRt71QC2RIIxs-F1Iry5qZn8,791
2
+ groundstore/context.py,sha256=d9RnKhFT4fk7Dx-1L9UqHwo2BQcqRauWNEWemEoThzI,8829
3
+ groundstore/contracts.py,sha256=WPn-5RAdmEZDy7JkW08949Wm04sbVYbFKXXFDrF_gaI,3524
4
+ groundstore/engine.py,sha256=plYQ0xhZv2RtI-kFFBiL8Djf625fT2iXUa7Lggp20U4,2171
5
+ groundstore/models.py,sha256=OrCfQiUN1yYSOGOgopZbruB2eWX_Rx7yJNO3miQ3QhA,9466
6
+ groundstore/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
7
+ groundstore/store.py,sha256=WCOffVJuhaePsPvcPrPGd3wub7Ce7qqZ_2oJcM38Ffw,16076
8
+ groundstore-0.1.0.dist-info/METADATA,sha256=oT-0XBcbcYb72egA4ERSzQQHKrNH5kfA6iakFkBPLrI,490
9
+ groundstore-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
10
+ groundstore-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any