groundstore 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- groundstore/__init__.py +30 -0
- groundstore/context.py +245 -0
- groundstore/contracts.py +108 -0
- groundstore/engine.py +64 -0
- groundstore/models.py +219 -0
- groundstore/py.typed +0 -0
- groundstore/store.py +398 -0
- groundstore-0.1.0.dist-info/METADATA +14 -0
- groundstore-0.1.0.dist-info/RECORD +10 -0
- groundstore-0.1.0.dist-info/WHEEL +4 -0
groundstore/__init__.py
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Shared mapping-task contracts and persistence for Groundworkers workflows."""
|
|
2
|
+
|
|
3
|
+
from .context import MappingEvidencePacket, MappingReadContext, MappingReviewHandoff
|
|
4
|
+
from .contracts import (
|
|
5
|
+
DecisionStatus,
|
|
6
|
+
LifecycleStatus,
|
|
7
|
+
MappingCandidateSpec,
|
|
8
|
+
MappingDecisionSpec,
|
|
9
|
+
MappingEvidenceSpec,
|
|
10
|
+
MappingInputSpec,
|
|
11
|
+
MappingRunSpec,
|
|
12
|
+
)
|
|
13
|
+
from .engine import create_groundstore_engine, create_schema
|
|
14
|
+
from .store import MappingStore
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"DecisionStatus",
|
|
18
|
+
"LifecycleStatus",
|
|
19
|
+
"MappingCandidateSpec",
|
|
20
|
+
"MappingDecisionSpec",
|
|
21
|
+
"MappingEvidencePacket",
|
|
22
|
+
"MappingEvidenceSpec",
|
|
23
|
+
"MappingInputSpec",
|
|
24
|
+
"MappingReadContext",
|
|
25
|
+
"MappingReviewHandoff",
|
|
26
|
+
"MappingRunSpec",
|
|
27
|
+
"MappingStore",
|
|
28
|
+
"create_groundstore_engine",
|
|
29
|
+
"create_schema",
|
|
30
|
+
]
|
groundstore/context.py
ADDED
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
"""Read-only views over persisted mapping work.
|
|
2
|
+
|
|
3
|
+
The persistence API remains useful to standalone writers. This module adds
|
|
4
|
+
the small, transport-neutral packet that a host or review client can consume
|
|
5
|
+
without depending on SQLAlchemy model instances or lazy relationships.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
14
|
+
|
|
15
|
+
from .models import (
|
|
16
|
+
MappingCandidate,
|
|
17
|
+
MappingDecision,
|
|
18
|
+
MappingDecisionEvent,
|
|
19
|
+
MappingEvidence,
|
|
20
|
+
MappingInput,
|
|
21
|
+
MappingRun,
|
|
22
|
+
)
|
|
23
|
+
from .store import MappingStore
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class MappingEvidencePacket(BaseModel):
|
|
27
|
+
"""Stable JSON contract for reviewing one persisted mapping input."""
|
|
28
|
+
|
|
29
|
+
model_config = ConfigDict(extra="forbid")
|
|
30
|
+
|
|
31
|
+
schema_version: str = "groundstore.mapping-evidence-packet.v1"
|
|
32
|
+
run: dict[str, Any]
|
|
33
|
+
input: dict[str, Any]
|
|
34
|
+
candidates: list[dict[str, Any]] = Field(default_factory=list)
|
|
35
|
+
evidence: list[dict[str, Any]] = Field(default_factory=list)
|
|
36
|
+
decision: dict[str, Any] | None = None
|
|
37
|
+
warnings: list[str] = Field(default_factory=list)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class MappingReviewHandoff(BaseModel):
|
|
41
|
+
"""Review-task reference carrying a self-contained evidence packet."""
|
|
42
|
+
|
|
43
|
+
model_config = ConfigDict(extra="forbid")
|
|
44
|
+
|
|
45
|
+
schema_version: str = "groundstore.mapping-review-handoff.v1"
|
|
46
|
+
task_id: str = Field(min_length=1)
|
|
47
|
+
task_type: str = "mapping_review"
|
|
48
|
+
source_namespace: str = Field(min_length=1)
|
|
49
|
+
input_id: str = Field(min_length=1)
|
|
50
|
+
decision_status: str = "needs_review"
|
|
51
|
+
packet: MappingEvidencePacket
|
|
52
|
+
requested_by: str | None = None
|
|
53
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True, slots=True)
|
|
57
|
+
class MappingReadContext:
|
|
58
|
+
"""Read-only Groundstore façade for host tools and review consumers."""
|
|
59
|
+
|
|
60
|
+
store: MappingStore
|
|
61
|
+
|
|
62
|
+
def status(
|
|
63
|
+
self,
|
|
64
|
+
source_namespace: str,
|
|
65
|
+
*,
|
|
66
|
+
target_system: str | None = None,
|
|
67
|
+
) -> dict[str, Any]:
|
|
68
|
+
"""Return the latest complete run and its common coverage summary."""
|
|
69
|
+
|
|
70
|
+
run = self.store.latest_successful_run(source_namespace, target_system=target_system)
|
|
71
|
+
if run is None:
|
|
72
|
+
return {
|
|
73
|
+
"source_namespace": source_namespace,
|
|
74
|
+
"target_system": target_system,
|
|
75
|
+
"latest_run": None,
|
|
76
|
+
"coverage": None,
|
|
77
|
+
}
|
|
78
|
+
return {
|
|
79
|
+
"source_namespace": source_namespace,
|
|
80
|
+
"target_system": target_system,
|
|
81
|
+
"latest_run": _run_payload(run),
|
|
82
|
+
"coverage": self.store.coverage(run.id),
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
def evidence_packet(self, input_id: str) -> MappingEvidencePacket:
|
|
86
|
+
"""Build a detached packet with candidates, evidence, and decision history."""
|
|
87
|
+
|
|
88
|
+
input_record = self.store.get_input(input_id)
|
|
89
|
+
if input_record is None:
|
|
90
|
+
raise KeyError(f"unknown mapping input: {input_id}")
|
|
91
|
+
run = self.store.get_run(input_record.run_id)
|
|
92
|
+
if run is None: # pragma: no cover - protected by the database FK
|
|
93
|
+
raise KeyError(f"unknown mapping run: {input_record.run_id}")
|
|
94
|
+
|
|
95
|
+
candidates = self.store.get_candidates(input_id)
|
|
96
|
+
evidence = self.store.get_evidence(input_id)
|
|
97
|
+
decision = self.store.latest_decision(input_id)
|
|
98
|
+
selected_ids = (
|
|
99
|
+
self.store.get_decision_candidate_ids(decision.id) if decision is not None else []
|
|
100
|
+
)
|
|
101
|
+
events = self.store.get_decision_history(decision.id) if decision is not None else []
|
|
102
|
+
evidence_by_candidate: dict[str, list[dict[str, Any]]] = {}
|
|
103
|
+
unattached: list[dict[str, Any]] = []
|
|
104
|
+
for item in evidence:
|
|
105
|
+
payload = _evidence_payload(item)
|
|
106
|
+
if item.candidate_id is None:
|
|
107
|
+
unattached.append(payload)
|
|
108
|
+
else:
|
|
109
|
+
evidence_by_candidate.setdefault(item.candidate_id, []).append(payload)
|
|
110
|
+
|
|
111
|
+
candidate_payloads = []
|
|
112
|
+
for candidate in candidates:
|
|
113
|
+
payload = _candidate_payload(candidate)
|
|
114
|
+
payload["evidence"] = evidence_by_candidate.get(candidate.id, [])
|
|
115
|
+
candidate_payloads.append(payload)
|
|
116
|
+
|
|
117
|
+
decision_payload = None
|
|
118
|
+
if decision is not None:
|
|
119
|
+
decision_payload = _decision_payload(decision)
|
|
120
|
+
decision_payload["selected_candidate_ids"] = selected_ids
|
|
121
|
+
decision_payload["history"] = [_event_payload(event) for event in events]
|
|
122
|
+
|
|
123
|
+
return MappingEvidencePacket(
|
|
124
|
+
run=_run_payload(run),
|
|
125
|
+
input=_input_payload(input_record),
|
|
126
|
+
candidates=candidate_payloads,
|
|
127
|
+
evidence=unattached,
|
|
128
|
+
decision=decision_payload,
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
def review_handoff(
|
|
132
|
+
self,
|
|
133
|
+
input_id: str,
|
|
134
|
+
*,
|
|
135
|
+
requested_by: str | None = None,
|
|
136
|
+
metadata: dict[str, Any] | None = None,
|
|
137
|
+
) -> MappingReviewHandoff:
|
|
138
|
+
"""Return a deterministic review reference without mutating the store."""
|
|
139
|
+
|
|
140
|
+
packet = self.evidence_packet(input_id)
|
|
141
|
+
decision_status = (
|
|
142
|
+
packet.decision.get("decision_status", "needs_review")
|
|
143
|
+
if packet.decision
|
|
144
|
+
else "needs_review"
|
|
145
|
+
)
|
|
146
|
+
return MappingReviewHandoff(
|
|
147
|
+
task_id=f"mapping-review:{input_id}",
|
|
148
|
+
source_namespace=packet.input["source_namespace"],
|
|
149
|
+
input_id=input_id,
|
|
150
|
+
decision_status=str(decision_status),
|
|
151
|
+
packet=packet,
|
|
152
|
+
requested_by=requested_by,
|
|
153
|
+
metadata=metadata or {},
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _run_payload(run: MappingRun) -> dict[str, Any]:
|
|
158
|
+
return {
|
|
159
|
+
"id": run.id,
|
|
160
|
+
"source_namespace": run.source_namespace,
|
|
161
|
+
"source_fingerprint": run.source_fingerprint,
|
|
162
|
+
"source_snapshot": run.source_snapshot,
|
|
163
|
+
"target_system": run.target_system,
|
|
164
|
+
"target_release": run.target_release,
|
|
165
|
+
"algorithm_version": run.algorithm_version,
|
|
166
|
+
"policy_version": run.policy_version,
|
|
167
|
+
"lifecycle_status": run.lifecycle_status,
|
|
168
|
+
"last_error": run.last_error,
|
|
169
|
+
"created_at": run.created_at.isoformat(),
|
|
170
|
+
"updated_at": run.updated_at.isoformat(),
|
|
171
|
+
}
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _input_payload(input_record: MappingInput) -> dict[str, Any]:
|
|
175
|
+
return {
|
|
176
|
+
"id": input_record.id,
|
|
177
|
+
"run_id": input_record.run_id,
|
|
178
|
+
"source_namespace": input_record.source_namespace,
|
|
179
|
+
"source_kind": input_record.source_kind,
|
|
180
|
+
"source_key": input_record.source_key,
|
|
181
|
+
"source_fingerprint": input_record.source_fingerprint,
|
|
182
|
+
"normalized_projection": input_record.normalized_projection,
|
|
183
|
+
"lifecycle_status": input_record.lifecycle_status,
|
|
184
|
+
"retry_count": input_record.retry_count,
|
|
185
|
+
"last_error": input_record.last_error,
|
|
186
|
+
"created_at": input_record.created_at.isoformat(),
|
|
187
|
+
"updated_at": input_record.updated_at.isoformat(),
|
|
188
|
+
}
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _candidate_payload(candidate: MappingCandidate) -> dict[str, Any]:
|
|
192
|
+
return {
|
|
193
|
+
"id": candidate.id,
|
|
194
|
+
"target_namespace": candidate.target_namespace,
|
|
195
|
+
"target_vocabulary_id": candidate.target_vocabulary_id,
|
|
196
|
+
"target_concept_id": candidate.target_concept_id,
|
|
197
|
+
"target_code": candidate.target_code,
|
|
198
|
+
"target_grain": candidate.target_grain,
|
|
199
|
+
"target_role": candidate.target_role,
|
|
200
|
+
"method": candidate.method,
|
|
201
|
+
"rank": candidate.rank,
|
|
202
|
+
"score": candidate.score,
|
|
203
|
+
"confidence": candidate.confidence,
|
|
204
|
+
"rationale": candidate.rationale,
|
|
205
|
+
"metadata": candidate.metadata_,
|
|
206
|
+
"created_at": candidate.created_at.isoformat(),
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _evidence_payload(evidence: MappingEvidence) -> dict[str, Any]:
|
|
211
|
+
return {
|
|
212
|
+
"id": evidence.id,
|
|
213
|
+
"candidate_id": evidence.candidate_id,
|
|
214
|
+
"decision_id": evidence.decision_id,
|
|
215
|
+
"evidence_key": evidence.evidence_key,
|
|
216
|
+
"evidence_type": evidence.evidence_type,
|
|
217
|
+
"source_reference": evidence.source_reference,
|
|
218
|
+
"method": evidence.method,
|
|
219
|
+
"payload": evidence.payload,
|
|
220
|
+
"created_at": evidence.created_at.isoformat(),
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _decision_payload(decision: MappingDecision) -> dict[str, Any]:
|
|
225
|
+
return {
|
|
226
|
+
"id": decision.id,
|
|
227
|
+
"input_id": decision.input_id,
|
|
228
|
+
"decision_version": decision.decision_version,
|
|
229
|
+
"decision_status": decision.decision_status,
|
|
230
|
+
"outcome_code": decision.outcome_code,
|
|
231
|
+
"reason_codes": decision.reason_codes,
|
|
232
|
+
"decided_by": decision.decided_by,
|
|
233
|
+
"metadata": decision.metadata_,
|
|
234
|
+
"decided_at": decision.decided_at.isoformat(),
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _event_payload(event: MappingDecisionEvent) -> dict[str, Any]:
|
|
239
|
+
return {
|
|
240
|
+
"id": event.id,
|
|
241
|
+
"decision_id": event.decision_id,
|
|
242
|
+
"event_type": event.event_type,
|
|
243
|
+
"detail": event.detail,
|
|
244
|
+
"created_at": event.created_at.isoformat(),
|
|
245
|
+
}
|
groundstore/contracts.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Source-independent contracts for mapping workflows."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from enum import StrEnum
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class LifecycleStatus(StrEnum):
|
|
12
|
+
"""Processing state for a run or input."""
|
|
13
|
+
|
|
14
|
+
PENDING = "pending"
|
|
15
|
+
IN_PROGRESS = "in_progress"
|
|
16
|
+
COMPLETE = "complete"
|
|
17
|
+
INCOMPLETE = "incomplete"
|
|
18
|
+
FAILED = "failed"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class DecisionStatus(StrEnum):
|
|
22
|
+
"""Mapping outcome independent of the processing lifecycle."""
|
|
23
|
+
|
|
24
|
+
MAPPED = "mapped"
|
|
25
|
+
AMBIGUOUS = "ambiguous"
|
|
26
|
+
UNMAPPABLE = "unmappable"
|
|
27
|
+
NEEDS_REVIEW = "needs_review"
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class MappingRunSpec(BaseModel):
|
|
31
|
+
"""Stable identity and metadata for one mapping run."""
|
|
32
|
+
|
|
33
|
+
model_config = ConfigDict(extra="forbid")
|
|
34
|
+
|
|
35
|
+
source_namespace: str = Field(min_length=1)
|
|
36
|
+
source_fingerprint: str = Field(min_length=1, max_length=128)
|
|
37
|
+
source_snapshot: dict[str, Any] = Field(default_factory=dict)
|
|
38
|
+
target_system: str = Field(min_length=1)
|
|
39
|
+
target_release: str | None = None
|
|
40
|
+
algorithm_version: str = Field(min_length=1)
|
|
41
|
+
policy_version: str = Field(min_length=1)
|
|
42
|
+
lifecycle_status: LifecycleStatus = LifecycleStatus.PENDING
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class MappingInputSpec(BaseModel):
|
|
46
|
+
"""Source item presented to a mapping workflow."""
|
|
47
|
+
|
|
48
|
+
model_config = ConfigDict(extra="forbid")
|
|
49
|
+
|
|
50
|
+
source_namespace: str = Field(min_length=1)
|
|
51
|
+
source_kind: str = Field(min_length=1)
|
|
52
|
+
source_key: str = Field(min_length=1)
|
|
53
|
+
source_fingerprint: str = Field(min_length=1, max_length=128)
|
|
54
|
+
normalized_projection: dict[str, Any] = Field(default_factory=dict)
|
|
55
|
+
lifecycle_status: LifecycleStatus = LifecycleStatus.PENDING
|
|
56
|
+
retry_count: int = Field(default=0, ge=0)
|
|
57
|
+
last_error: str | None = None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class MappingCandidateSpec(BaseModel):
|
|
61
|
+
"""One candidate target for a mapping input."""
|
|
62
|
+
|
|
63
|
+
model_config = ConfigDict(extra="forbid")
|
|
64
|
+
|
|
65
|
+
target_namespace: str = Field(min_length=1)
|
|
66
|
+
target_vocabulary_id: str | None = None
|
|
67
|
+
target_concept_id: str | None = None
|
|
68
|
+
target_code: str | None = None
|
|
69
|
+
target_grain: str | None = None
|
|
70
|
+
target_role: str | None = None
|
|
71
|
+
method: str = Field(min_length=1)
|
|
72
|
+
rank: int = Field(default=1, ge=1)
|
|
73
|
+
score: float | None = None
|
|
74
|
+
confidence: float | None = Field(default=None, ge=0, le=1)
|
|
75
|
+
rationale: str | None = None
|
|
76
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
|
77
|
+
|
|
78
|
+
@field_validator("target_vocabulary_id", "target_concept_id", "target_code")
|
|
79
|
+
@classmethod
|
|
80
|
+
def reject_blank_identifiers(cls, value: str | None) -> str | None:
|
|
81
|
+
if value is not None and not value.strip():
|
|
82
|
+
raise ValueError("target identifiers cannot be blank")
|
|
83
|
+
return value
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
class MappingEvidenceSpec(BaseModel):
|
|
87
|
+
"""Evidence attached to a candidate or decision."""
|
|
88
|
+
|
|
89
|
+
model_config = ConfigDict(extra="forbid")
|
|
90
|
+
|
|
91
|
+
evidence_key: str = Field(min_length=1, max_length=128)
|
|
92
|
+
evidence_type: str = Field(min_length=1)
|
|
93
|
+
source_reference: str | None = None
|
|
94
|
+
method: str | None = None
|
|
95
|
+
payload: dict[str, Any] = Field(default_factory=dict)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class MappingDecisionSpec(BaseModel):
|
|
99
|
+
"""Versioned decision for one mapping input."""
|
|
100
|
+
|
|
101
|
+
model_config = ConfigDict(extra="forbid")
|
|
102
|
+
|
|
103
|
+
decision_status: DecisionStatus
|
|
104
|
+
selected_candidate_ids: list[str] = Field(default_factory=list)
|
|
105
|
+
outcome_code: str | None = None
|
|
106
|
+
reason_codes: list[str] = Field(default_factory=list)
|
|
107
|
+
decided_by: str | None = None
|
|
108
|
+
metadata: dict[str, Any] = Field(default_factory=dict)
|
groundstore/engine.py
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
"""Engine and schema helpers, including oa-configurator resolution."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import sqlalchemy as sa
|
|
9
|
+
from oa_configurator import ResolvedDatabase, Resolver
|
|
10
|
+
from sqlalchemy import Engine, event
|
|
11
|
+
from sqlalchemy.engine import URL
|
|
12
|
+
|
|
13
|
+
from .models import Base
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def create_groundstore_engine(
|
|
17
|
+
url: str | URL | None = None,
|
|
18
|
+
*,
|
|
19
|
+
resolved_database: ResolvedDatabase | None = None,
|
|
20
|
+
resolver: Resolver | None = None,
|
|
21
|
+
database_name: str = "mapping_db",
|
|
22
|
+
execution_options: Mapping[str, Any] | None = None,
|
|
23
|
+
**engine_kwargs: Any,
|
|
24
|
+
) -> Engine:
|
|
25
|
+
"""Create a store engine from an explicit URL or OA database resource."""
|
|
26
|
+
if sum(value is not None for value in (url, resolved_database, resolver)) > 1:
|
|
27
|
+
raise ValueError("provide only one of url, resolved_database, or resolver")
|
|
28
|
+
|
|
29
|
+
if resolved_database is not None:
|
|
30
|
+
engine = resolved_database.create_engine(**engine_kwargs)
|
|
31
|
+
elif resolver is not None:
|
|
32
|
+
engine = resolver.resolve_database(database_name).create_engine(**engine_kwargs)
|
|
33
|
+
elif url is not None:
|
|
34
|
+
engine = sa.create_engine(url, **engine_kwargs)
|
|
35
|
+
else:
|
|
36
|
+
engine = (
|
|
37
|
+
Resolver.from_active_config()
|
|
38
|
+
.resolve_database(database_name)
|
|
39
|
+
.create_engine(**engine_kwargs)
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
if execution_options:
|
|
43
|
+
engine = engine.execution_options(**dict(execution_options))
|
|
44
|
+
if engine.dialect.name == "sqlite":
|
|
45
|
+
_enable_sqlite_foreign_keys(engine)
|
|
46
|
+
return engine
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def create_schema(engine: Engine) -> None:
|
|
50
|
+
"""Create all groundstore tables if they do not already exist."""
|
|
51
|
+
Base.metadata.create_all(engine)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def drop_schema(engine: Engine) -> None:
|
|
55
|
+
"""Drop all groundstore tables; intended for isolated test databases."""
|
|
56
|
+
Base.metadata.drop_all(engine)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _enable_sqlite_foreign_keys(engine: Engine) -> None:
|
|
60
|
+
@event.listens_for(engine, "connect")
|
|
61
|
+
def _set_foreign_keys(dbapi_connection: Any, _connection_record: Any) -> None:
|
|
62
|
+
cursor = dbapi_connection.cursor()
|
|
63
|
+
cursor.execute("PRAGMA foreign_keys=ON")
|
|
64
|
+
cursor.close()
|
groundstore/models.py
ADDED
|
@@ -0,0 +1,219 @@
|
|
|
1
|
+
"""SQLAlchemy persistence models for the shared mapping contract."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from sqlalchemy import (
|
|
9
|
+
JSON,
|
|
10
|
+
DateTime,
|
|
11
|
+
Float,
|
|
12
|
+
ForeignKey,
|
|
13
|
+
Index,
|
|
14
|
+
Integer,
|
|
15
|
+
String,
|
|
16
|
+
Text,
|
|
17
|
+
UniqueConstraint,
|
|
18
|
+
)
|
|
19
|
+
from sqlalchemy.orm import DeclarativeBase, Mapped, mapped_column, relationship
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class Base(DeclarativeBase):
|
|
23
|
+
"""Declarative base for groundstore tables."""
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
class MappingRun(Base):
|
|
27
|
+
__tablename__ = "mapping_runs"
|
|
28
|
+
__table_args__ = (
|
|
29
|
+
UniqueConstraint(
|
|
30
|
+
"source_namespace",
|
|
31
|
+
"source_fingerprint",
|
|
32
|
+
"target_system",
|
|
33
|
+
"target_release",
|
|
34
|
+
"algorithm_version",
|
|
35
|
+
"policy_version",
|
|
36
|
+
name="uq_mapping_run_identity",
|
|
37
|
+
),
|
|
38
|
+
Index("ix_mapping_runs_source_status", "source_namespace", "lifecycle_status"),
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
id: Mapped[str] = mapped_column(String(36), primary_key=True)
|
|
42
|
+
source_namespace: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
43
|
+
source_fingerprint: Mapped[str] = mapped_column(String(128), nullable=False)
|
|
44
|
+
source_snapshot: Mapped[dict[str, Any]] = mapped_column(JSON, nullable=False, default=dict)
|
|
45
|
+
target_system: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
46
|
+
target_release: Mapped[str | None] = mapped_column(String(100))
|
|
47
|
+
algorithm_version: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
48
|
+
policy_version: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
49
|
+
lifecycle_status: Mapped[str] = mapped_column(String(20), nullable=False)
|
|
50
|
+
last_error: Mapped[str | None] = mapped_column(Text)
|
|
51
|
+
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
52
|
+
updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
53
|
+
|
|
54
|
+
inputs: Mapped[list[MappingInput]] = relationship(
|
|
55
|
+
back_populates="run", cascade="all, delete-orphan"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class MappingInput(Base):
|
|
60
|
+
__tablename__ = "mapping_inputs"
|
|
61
|
+
__table_args__ = (
|
|
62
|
+
UniqueConstraint(
|
|
63
|
+
"run_id",
|
|
64
|
+
"source_namespace",
|
|
65
|
+
"source_kind",
|
|
66
|
+
"source_key",
|
|
67
|
+
"source_fingerprint",
|
|
68
|
+
name="uq_mapping_input_identity",
|
|
69
|
+
),
|
|
70
|
+
Index("ix_mapping_inputs_run_status", "run_id", "lifecycle_status"),
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
id: Mapped[str] = mapped_column(String(36), primary_key=True)
|
|
74
|
+
run_id: Mapped[str] = mapped_column(
|
|
75
|
+
ForeignKey("mapping_runs.id", ondelete="CASCADE"), nullable=False
|
|
76
|
+
)
|
|
77
|
+
source_namespace: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
78
|
+
source_kind: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
79
|
+
source_key: Mapped[str] = mapped_column(String(255), nullable=False)
|
|
80
|
+
source_fingerprint: Mapped[str] = mapped_column(String(128), nullable=False)
|
|
81
|
+
normalized_projection: Mapped[dict[str, Any]] = mapped_column(
|
|
82
|
+
JSON, nullable=False, default=dict
|
|
83
|
+
)
|
|
84
|
+
lifecycle_status: Mapped[str] = mapped_column(String(20), nullable=False)
|
|
85
|
+
retry_count: Mapped[int] = mapped_column(Integer, nullable=False, default=0)
|
|
86
|
+
last_error: Mapped[str | None] = mapped_column(Text)
|
|
87
|
+
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
88
|
+
updated_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
89
|
+
|
|
90
|
+
run: Mapped[MappingRun] = relationship(back_populates="inputs")
|
|
91
|
+
candidates: Mapped[list[MappingCandidate]] = relationship(
|
|
92
|
+
back_populates="input", cascade="all, delete-orphan"
|
|
93
|
+
)
|
|
94
|
+
evidence: Mapped[list[MappingEvidence]] = relationship(
|
|
95
|
+
back_populates="input", cascade="all, delete-orphan"
|
|
96
|
+
)
|
|
97
|
+
decisions: Mapped[list[MappingDecision]] = relationship(
|
|
98
|
+
back_populates="input", cascade="all, delete-orphan"
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class MappingCandidate(Base):
|
|
103
|
+
__tablename__ = "mapping_candidates"
|
|
104
|
+
__table_args__ = (
|
|
105
|
+
UniqueConstraint("input_id", "candidate_key", name="uq_mapping_candidate_key"),
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
id: Mapped[str] = mapped_column(String(36), primary_key=True)
|
|
109
|
+
input_id: Mapped[str] = mapped_column(
|
|
110
|
+
ForeignKey("mapping_inputs.id", ondelete="CASCADE"), nullable=False
|
|
111
|
+
)
|
|
112
|
+
candidate_key: Mapped[str] = mapped_column(String(128), nullable=False)
|
|
113
|
+
target_namespace: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
114
|
+
target_vocabulary_id: Mapped[str | None] = mapped_column(String(100))
|
|
115
|
+
target_concept_id: Mapped[str | None] = mapped_column(String(100))
|
|
116
|
+
target_code: Mapped[str | None] = mapped_column(String(255))
|
|
117
|
+
target_grain: Mapped[str | None] = mapped_column(String(100))
|
|
118
|
+
target_role: Mapped[str | None] = mapped_column(String(100))
|
|
119
|
+
method: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
120
|
+
rank: Mapped[int] = mapped_column(Integer, nullable=False)
|
|
121
|
+
score: Mapped[float | None] = mapped_column(Float)
|
|
122
|
+
confidence: Mapped[float | None] = mapped_column(Float)
|
|
123
|
+
rationale: Mapped[str | None] = mapped_column(Text)
|
|
124
|
+
metadata_: Mapped[dict[str, Any]] = mapped_column(
|
|
125
|
+
"metadata", JSON, nullable=False, default=dict
|
|
126
|
+
)
|
|
127
|
+
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
128
|
+
|
|
129
|
+
input: Mapped[MappingInput] = relationship(back_populates="candidates")
|
|
130
|
+
evidence: Mapped[list[MappingEvidence]] = relationship(back_populates="candidate")
|
|
131
|
+
decisions: Mapped[list[MappingDecision]] = relationship(
|
|
132
|
+
secondary="mapping_decision_candidates", back_populates="selected_candidates"
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
class MappingEvidence(Base):
|
|
137
|
+
__tablename__ = "mapping_evidence"
|
|
138
|
+
__table_args__ = (
|
|
139
|
+
UniqueConstraint("input_id", "evidence_key", name="uq_mapping_evidence_key"),
|
|
140
|
+
Index("ix_mapping_evidence_candidate", "candidate_id"),
|
|
141
|
+
Index("ix_mapping_evidence_decision", "decision_id"),
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
id: Mapped[str] = mapped_column(String(36), primary_key=True)
|
|
145
|
+
input_id: Mapped[str] = mapped_column(
|
|
146
|
+
ForeignKey("mapping_inputs.id", ondelete="CASCADE"), nullable=False
|
|
147
|
+
)
|
|
148
|
+
candidate_id: Mapped[str | None] = mapped_column(
|
|
149
|
+
ForeignKey("mapping_candidates.id", ondelete="CASCADE")
|
|
150
|
+
)
|
|
151
|
+
decision_id: Mapped[str | None] = mapped_column(
|
|
152
|
+
ForeignKey("mapping_decisions.id", ondelete="CASCADE")
|
|
153
|
+
)
|
|
154
|
+
evidence_key: Mapped[str] = mapped_column(String(128), nullable=False)
|
|
155
|
+
evidence_type: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
156
|
+
source_reference: Mapped[str | None] = mapped_column(Text)
|
|
157
|
+
method: Mapped[str | None] = mapped_column(String(100))
|
|
158
|
+
payload: Mapped[dict[str, Any]] = mapped_column(JSON, nullable=False, default=dict)
|
|
159
|
+
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
160
|
+
|
|
161
|
+
input: Mapped[MappingInput] = relationship(back_populates="evidence")
|
|
162
|
+
candidate: Mapped[MappingCandidate | None] = relationship(back_populates="evidence")
|
|
163
|
+
decision: Mapped[MappingDecision | None] = relationship(back_populates="evidence")
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
class MappingDecision(Base):
|
|
167
|
+
__tablename__ = "mapping_decisions"
|
|
168
|
+
__table_args__ = (
|
|
169
|
+
UniqueConstraint("input_id", "decision_version", name="uq_mapping_decision_version"),
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
id: Mapped[str] = mapped_column(String(36), primary_key=True)
|
|
173
|
+
input_id: Mapped[str] = mapped_column(
|
|
174
|
+
ForeignKey("mapping_inputs.id", ondelete="CASCADE"), nullable=False
|
|
175
|
+
)
|
|
176
|
+
decision_version: Mapped[int] = mapped_column(Integer, nullable=False)
|
|
177
|
+
decision_status: Mapped[str] = mapped_column(String(20), nullable=False)
|
|
178
|
+
outcome_code: Mapped[str | None] = mapped_column(String(150))
|
|
179
|
+
reason_codes: Mapped[list[str]] = mapped_column(JSON, nullable=False, default=list)
|
|
180
|
+
decided_by: Mapped[str | None] = mapped_column(String(255))
|
|
181
|
+
metadata_: Mapped[dict[str, Any]] = mapped_column(
|
|
182
|
+
"metadata", JSON, nullable=False, default=dict
|
|
183
|
+
)
|
|
184
|
+
decided_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
185
|
+
|
|
186
|
+
input: Mapped[MappingInput] = relationship(back_populates="decisions")
|
|
187
|
+
selected_candidates: Mapped[list[MappingCandidate]] = relationship(
|
|
188
|
+
secondary="mapping_decision_candidates", back_populates="decisions"
|
|
189
|
+
)
|
|
190
|
+
evidence: Mapped[list[MappingEvidence]] = relationship(back_populates="decision")
|
|
191
|
+
history: Mapped[list[MappingDecisionEvent]] = relationship(
|
|
192
|
+
back_populates="decision", cascade="all, delete-orphan"
|
|
193
|
+
)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
class MappingDecisionCandidate(Base):
|
|
197
|
+
__tablename__ = "mapping_decision_candidates"
|
|
198
|
+
|
|
199
|
+
decision_id: Mapped[str] = mapped_column(
|
|
200
|
+
ForeignKey("mapping_decisions.id", ondelete="CASCADE"), primary_key=True
|
|
201
|
+
)
|
|
202
|
+
candidate_id: Mapped[str] = mapped_column(
|
|
203
|
+
ForeignKey("mapping_candidates.id", ondelete="CASCADE"), primary_key=True
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
class MappingDecisionEvent(Base):
|
|
208
|
+
__tablename__ = "mapping_decision_events"
|
|
209
|
+
__table_args__ = (Index("ix_mapping_decision_events_decision", "decision_id", "created_at"),)
|
|
210
|
+
|
|
211
|
+
id: Mapped[str] = mapped_column(String(36), primary_key=True)
|
|
212
|
+
decision_id: Mapped[str] = mapped_column(
|
|
213
|
+
ForeignKey("mapping_decisions.id", ondelete="CASCADE"), nullable=False
|
|
214
|
+
)
|
|
215
|
+
event_type: Mapped[str] = mapped_column(String(100), nullable=False)
|
|
216
|
+
detail: Mapped[dict[str, Any]] = mapped_column(JSON, nullable=False, default=dict)
|
|
217
|
+
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), nullable=False)
|
|
218
|
+
|
|
219
|
+
decision: Mapped[MappingDecision] = relationship(back_populates="history")
|
groundstore/py.typed
ADDED
|
File without changes
|
groundstore/store.py
ADDED
|
@@ -0,0 +1,398 @@
|
|
|
1
|
+
"""Repository-style persistence API for mapping workflows."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from collections import Counter
|
|
8
|
+
from datetime import UTC, datetime
|
|
9
|
+
from typing import Any
|
|
10
|
+
from uuid import uuid4
|
|
11
|
+
|
|
12
|
+
from oa_configurator import ResolvedDatabase, Resolver
|
|
13
|
+
from sqlalchemy import Engine, select
|
|
14
|
+
from sqlalchemy.orm import sessionmaker
|
|
15
|
+
|
|
16
|
+
from .contracts import (
|
|
17
|
+
MappingCandidateSpec,
|
|
18
|
+
MappingDecisionSpec,
|
|
19
|
+
MappingEvidenceSpec,
|
|
20
|
+
MappingInputSpec,
|
|
21
|
+
MappingRunSpec,
|
|
22
|
+
)
|
|
23
|
+
from .engine import create_groundstore_engine, create_schema
|
|
24
|
+
from .models import (
|
|
25
|
+
MappingCandidate,
|
|
26
|
+
MappingDecision,
|
|
27
|
+
MappingDecisionCandidate,
|
|
28
|
+
MappingDecisionEvent,
|
|
29
|
+
MappingEvidence,
|
|
30
|
+
MappingInput,
|
|
31
|
+
MappingRun,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class MappingStore:
|
|
36
|
+
"""Persist and resume source-independent mapping workflows."""
|
|
37
|
+
|
|
38
|
+
def __init__(self, engine: Engine, *, initialize: bool = True) -> None:
|
|
39
|
+
self.engine = engine
|
|
40
|
+
self._session_factory = sessionmaker(bind=engine, expire_on_commit=False)
|
|
41
|
+
if initialize:
|
|
42
|
+
create_schema(engine)
|
|
43
|
+
|
|
44
|
+
@classmethod
|
|
45
|
+
def from_url(cls, url: str, **engine_kwargs: Any) -> MappingStore:
|
|
46
|
+
return cls(create_groundstore_engine(url, **engine_kwargs))
|
|
47
|
+
|
|
48
|
+
@classmethod
|
|
49
|
+
def from_database(cls, database: ResolvedDatabase, **engine_kwargs: Any) -> MappingStore:
|
|
50
|
+
"""Create a store from an already-resolved OA database resource."""
|
|
51
|
+
return cls(create_groundstore_engine(resolved_database=database, **engine_kwargs))
|
|
52
|
+
|
|
53
|
+
@classmethod
|
|
54
|
+
def from_resolver(
|
|
55
|
+
cls, resolver: Resolver, database_name: str = "mapping_db", **engine_kwargs: Any
|
|
56
|
+
) -> MappingStore:
|
|
57
|
+
engine = create_groundstore_engine(
|
|
58
|
+
resolver=resolver, database_name=database_name, **engine_kwargs
|
|
59
|
+
)
|
|
60
|
+
return cls(engine)
|
|
61
|
+
|
|
62
|
+
def get_or_create_run(self, spec: MappingRunSpec) -> MappingRun:
|
|
63
|
+
"""Return the stable run for *spec*, preserving resumable state."""
|
|
64
|
+
with self._session_factory() as session:
|
|
65
|
+
run = session.scalar(
|
|
66
|
+
select(MappingRun).where(
|
|
67
|
+
MappingRun.source_namespace == spec.source_namespace,
|
|
68
|
+
MappingRun.source_fingerprint == spec.source_fingerprint,
|
|
69
|
+
MappingRun.target_system == spec.target_system,
|
|
70
|
+
MappingRun.target_release == spec.target_release,
|
|
71
|
+
MappingRun.algorithm_version == spec.algorithm_version,
|
|
72
|
+
MappingRun.policy_version == spec.policy_version,
|
|
73
|
+
)
|
|
74
|
+
)
|
|
75
|
+
if run is None:
|
|
76
|
+
now = _now()
|
|
77
|
+
run = MappingRun(
|
|
78
|
+
id=_id(),
|
|
79
|
+
**spec.model_dump(exclude={"lifecycle_status"}),
|
|
80
|
+
lifecycle_status=spec.lifecycle_status.value,
|
|
81
|
+
created_at=now,
|
|
82
|
+
updated_at=now,
|
|
83
|
+
)
|
|
84
|
+
session.add(run)
|
|
85
|
+
session.commit()
|
|
86
|
+
return run
|
|
87
|
+
|
|
88
|
+
def update_run(
|
|
89
|
+
self,
|
|
90
|
+
run_id: str,
|
|
91
|
+
*,
|
|
92
|
+
lifecycle_status: str | None = None,
|
|
93
|
+
last_error: str | None = None,
|
|
94
|
+
) -> MappingRun:
|
|
95
|
+
with self._session_factory() as session:
|
|
96
|
+
run = session.get(MappingRun, run_id)
|
|
97
|
+
if run is None:
|
|
98
|
+
raise KeyError(f"unknown mapping run: {run_id}")
|
|
99
|
+
if lifecycle_status is not None:
|
|
100
|
+
run.lifecycle_status = lifecycle_status
|
|
101
|
+
run.last_error = last_error
|
|
102
|
+
run.updated_at = _now()
|
|
103
|
+
session.commit()
|
|
104
|
+
return run
|
|
105
|
+
|
|
106
|
+
def upsert_input(self, run_id: str, spec: MappingInputSpec) -> MappingInput:
|
|
107
|
+
"""Insert or return an input, retaining candidates and decisions on retry."""
|
|
108
|
+
with self._session_factory() as session:
|
|
109
|
+
record = session.scalar(
|
|
110
|
+
select(MappingInput).where(
|
|
111
|
+
MappingInput.run_id == run_id,
|
|
112
|
+
MappingInput.source_namespace == spec.source_namespace,
|
|
113
|
+
MappingInput.source_kind == spec.source_kind,
|
|
114
|
+
MappingInput.source_key == spec.source_key,
|
|
115
|
+
MappingInput.source_fingerprint == spec.source_fingerprint,
|
|
116
|
+
)
|
|
117
|
+
)
|
|
118
|
+
if record is None:
|
|
119
|
+
now = _now()
|
|
120
|
+
record = MappingInput(
|
|
121
|
+
id=_id(),
|
|
122
|
+
run_id=run_id,
|
|
123
|
+
**spec.model_dump(exclude={"lifecycle_status"}),
|
|
124
|
+
lifecycle_status=spec.lifecycle_status.value,
|
|
125
|
+
created_at=now,
|
|
126
|
+
updated_at=now,
|
|
127
|
+
)
|
|
128
|
+
session.add(record)
|
|
129
|
+
session.commit()
|
|
130
|
+
return record
|
|
131
|
+
|
|
132
|
+
def update_input(
|
|
133
|
+
self,
|
|
134
|
+
input_id: str,
|
|
135
|
+
*,
|
|
136
|
+
lifecycle_status: str | None = None,
|
|
137
|
+
retry_count: int | None = None,
|
|
138
|
+
last_error: str | None = None,
|
|
139
|
+
) -> MappingInput:
|
|
140
|
+
"""Update processing state without replacing mapping evidence."""
|
|
141
|
+
with self._session_factory() as session:
|
|
142
|
+
record = session.get(MappingInput, input_id)
|
|
143
|
+
if record is None:
|
|
144
|
+
raise KeyError(f"unknown mapping input: {input_id}")
|
|
145
|
+
if lifecycle_status is not None:
|
|
146
|
+
record.lifecycle_status = lifecycle_status
|
|
147
|
+
if retry_count is not None:
|
|
148
|
+
if retry_count < 0:
|
|
149
|
+
raise ValueError("retry_count cannot be negative")
|
|
150
|
+
record.retry_count = retry_count
|
|
151
|
+
record.last_error = last_error
|
|
152
|
+
record.updated_at = _now()
|
|
153
|
+
session.commit()
|
|
154
|
+
return record
|
|
155
|
+
|
|
156
|
+
def get_candidates(self, input_id: str) -> list[MappingCandidate]:
|
|
157
|
+
"""Return candidates for one input in deterministic rank order."""
|
|
158
|
+
with self._session_factory() as session:
|
|
159
|
+
return list(
|
|
160
|
+
session.scalars(
|
|
161
|
+
select(MappingCandidate)
|
|
162
|
+
.where(MappingCandidate.input_id == input_id)
|
|
163
|
+
.order_by(MappingCandidate.rank, MappingCandidate.created_at)
|
|
164
|
+
)
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
def upsert_candidate(self, input_id: str, spec: MappingCandidateSpec) -> MappingCandidate:
|
|
168
|
+
"""Insert or return a candidate using a deterministic semantic key."""
|
|
169
|
+
key = _candidate_key(spec)
|
|
170
|
+
with self._session_factory() as session:
|
|
171
|
+
candidate = session.scalar(
|
|
172
|
+
select(MappingCandidate).where(
|
|
173
|
+
MappingCandidate.input_id == input_id,
|
|
174
|
+
MappingCandidate.candidate_key == key,
|
|
175
|
+
)
|
|
176
|
+
)
|
|
177
|
+
if candidate is None:
|
|
178
|
+
candidate = MappingCandidate(
|
|
179
|
+
id=_id(),
|
|
180
|
+
input_id=input_id,
|
|
181
|
+
candidate_key=key,
|
|
182
|
+
**spec.model_dump(exclude={"metadata"}),
|
|
183
|
+
metadata_=spec.metadata,
|
|
184
|
+
created_at=_now(),
|
|
185
|
+
)
|
|
186
|
+
session.add(candidate)
|
|
187
|
+
session.commit()
|
|
188
|
+
return candidate
|
|
189
|
+
|
|
190
|
+
def add_evidence(
|
|
191
|
+
self,
|
|
192
|
+
input_id: str,
|
|
193
|
+
spec: MappingEvidenceSpec,
|
|
194
|
+
*,
|
|
195
|
+
candidate_id: str | None = None,
|
|
196
|
+
decision_id: str | None = None,
|
|
197
|
+
) -> MappingEvidence:
|
|
198
|
+
"""Insert or return evidence, attached to exactly one candidate or decision."""
|
|
199
|
+
if (candidate_id is None) == (decision_id is None):
|
|
200
|
+
raise ValueError("evidence must reference exactly one candidate or decision")
|
|
201
|
+
with self._session_factory() as session:
|
|
202
|
+
if candidate_id is not None:
|
|
203
|
+
candidate = session.scalar(
|
|
204
|
+
select(MappingCandidate).where(
|
|
205
|
+
MappingCandidate.id == candidate_id,
|
|
206
|
+
MappingCandidate.input_id == input_id,
|
|
207
|
+
)
|
|
208
|
+
)
|
|
209
|
+
if candidate is None:
|
|
210
|
+
raise ValueError("candidate must belong to the input")
|
|
211
|
+
if decision_id is not None:
|
|
212
|
+
decision = session.scalar(
|
|
213
|
+
select(MappingDecision).where(
|
|
214
|
+
MappingDecision.id == decision_id,
|
|
215
|
+
MappingDecision.input_id == input_id,
|
|
216
|
+
)
|
|
217
|
+
)
|
|
218
|
+
if decision is None:
|
|
219
|
+
raise ValueError("decision must belong to the input")
|
|
220
|
+
evidence = session.scalar(
|
|
221
|
+
select(MappingEvidence).where(
|
|
222
|
+
MappingEvidence.input_id == input_id,
|
|
223
|
+
MappingEvidence.evidence_key == spec.evidence_key,
|
|
224
|
+
)
|
|
225
|
+
)
|
|
226
|
+
if evidence is None:
|
|
227
|
+
evidence = MappingEvidence(
|
|
228
|
+
id=_id(),
|
|
229
|
+
input_id=input_id,
|
|
230
|
+
candidate_id=candidate_id,
|
|
231
|
+
decision_id=decision_id,
|
|
232
|
+
**spec.model_dump(),
|
|
233
|
+
created_at=_now(),
|
|
234
|
+
)
|
|
235
|
+
session.add(evidence)
|
|
236
|
+
session.commit()
|
|
237
|
+
return evidence
|
|
238
|
+
|
|
239
|
+
def record_decision(self, input_id: str, spec: MappingDecisionSpec) -> MappingDecision:
|
|
240
|
+
"""Append a decision version and its immutable history event."""
|
|
241
|
+
with self._session_factory() as session:
|
|
242
|
+
if len(spec.selected_candidate_ids) != len(set(spec.selected_candidate_ids)):
|
|
243
|
+
raise ValueError("selected candidates must be unique")
|
|
244
|
+
selected = list(
|
|
245
|
+
session.scalars(
|
|
246
|
+
select(MappingCandidate).where(
|
|
247
|
+
MappingCandidate.input_id == input_id,
|
|
248
|
+
MappingCandidate.id.in_(spec.selected_candidate_ids),
|
|
249
|
+
)
|
|
250
|
+
)
|
|
251
|
+
)
|
|
252
|
+
if len(selected) != len(set(spec.selected_candidate_ids)):
|
|
253
|
+
raise ValueError("all selected candidates must belong to the input")
|
|
254
|
+
latest = session.scalar(
|
|
255
|
+
select(MappingDecision)
|
|
256
|
+
.where(MappingDecision.input_id == input_id)
|
|
257
|
+
.order_by(MappingDecision.decision_version.desc())
|
|
258
|
+
)
|
|
259
|
+
version = 1 if latest is None else latest.decision_version + 1
|
|
260
|
+
decision = MappingDecision(
|
|
261
|
+
id=_id(),
|
|
262
|
+
input_id=input_id,
|
|
263
|
+
decision_version=version,
|
|
264
|
+
decision_status=spec.decision_status.value,
|
|
265
|
+
outcome_code=spec.outcome_code,
|
|
266
|
+
reason_codes=spec.reason_codes,
|
|
267
|
+
decided_by=spec.decided_by,
|
|
268
|
+
metadata_=spec.metadata,
|
|
269
|
+
decided_at=_now(),
|
|
270
|
+
)
|
|
271
|
+
decision.selected_candidates = selected
|
|
272
|
+
decision.history.append(
|
|
273
|
+
MappingDecisionEvent(
|
|
274
|
+
id=_id(),
|
|
275
|
+
event_type="decision_recorded",
|
|
276
|
+
detail={
|
|
277
|
+
"decision_status": spec.decision_status.value,
|
|
278
|
+
"decision_version": version,
|
|
279
|
+
},
|
|
280
|
+
created_at=_now(),
|
|
281
|
+
)
|
|
282
|
+
)
|
|
283
|
+
session.add(decision)
|
|
284
|
+
session.commit()
|
|
285
|
+
return decision
|
|
286
|
+
|
|
287
|
+
def latest_decision(self, input_id: str) -> MappingDecision | None:
|
|
288
|
+
with self._session_factory() as session:
|
|
289
|
+
return session.scalar(
|
|
290
|
+
select(MappingDecision)
|
|
291
|
+
.where(MappingDecision.input_id == input_id)
|
|
292
|
+
.order_by(MappingDecision.decision_version.desc())
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
def get_input(self, input_id: str) -> MappingInput | None:
|
|
296
|
+
"""Return one input without relying on lazy relationships."""
|
|
297
|
+
with self._session_factory() as session:
|
|
298
|
+
return session.get(MappingInput, input_id)
|
|
299
|
+
|
|
300
|
+
def get_evidence(self, input_id: str) -> list[MappingEvidence]:
|
|
301
|
+
"""Return all evidence for an input in insertion order."""
|
|
302
|
+
with self._session_factory() as session:
|
|
303
|
+
return list(
|
|
304
|
+
session.scalars(
|
|
305
|
+
select(MappingEvidence)
|
|
306
|
+
.where(MappingEvidence.input_id == input_id)
|
|
307
|
+
.order_by(MappingEvidence.created_at, MappingEvidence.id)
|
|
308
|
+
)
|
|
309
|
+
)
|
|
310
|
+
|
|
311
|
+
def get_decision_candidate_ids(self, decision_id: str) -> list[str]:
|
|
312
|
+
"""Return selected candidate IDs in stable candidate order."""
|
|
313
|
+
with self._session_factory() as session:
|
|
314
|
+
rows = session.execute(
|
|
315
|
+
select(MappingDecisionCandidate.candidate_id)
|
|
316
|
+
.where(MappingDecisionCandidate.decision_id == decision_id)
|
|
317
|
+
.join(
|
|
318
|
+
MappingCandidate,
|
|
319
|
+
MappingCandidate.id == MappingDecisionCandidate.candidate_id,
|
|
320
|
+
)
|
|
321
|
+
.order_by(MappingCandidate.rank, MappingCandidate.id)
|
|
322
|
+
)
|
|
323
|
+
return [candidate_id for (candidate_id,) in rows]
|
|
324
|
+
|
|
325
|
+
def get_decision_history(self, decision_id: str) -> list[MappingDecisionEvent]:
|
|
326
|
+
"""Return immutable history events for one decision."""
|
|
327
|
+
with self._session_factory() as session:
|
|
328
|
+
return list(
|
|
329
|
+
session.scalars(
|
|
330
|
+
select(MappingDecisionEvent)
|
|
331
|
+
.where(MappingDecisionEvent.decision_id == decision_id)
|
|
332
|
+
.order_by(MappingDecisionEvent.created_at, MappingDecisionEvent.id)
|
|
333
|
+
)
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
def coverage(self, run_id: str) -> dict[str, Any]:
|
|
337
|
+
"""Return lifecycle and latest decision counts for a run."""
|
|
338
|
+
with self._session_factory() as session:
|
|
339
|
+
inputs = list(
|
|
340
|
+
session.scalars(select(MappingInput).where(MappingInput.run_id == run_id))
|
|
341
|
+
)
|
|
342
|
+
decisions = []
|
|
343
|
+
for input_record in inputs:
|
|
344
|
+
decision = session.scalar(
|
|
345
|
+
select(MappingDecision)
|
|
346
|
+
.where(MappingDecision.input_id == input_record.id)
|
|
347
|
+
.order_by(MappingDecision.decision_version.desc())
|
|
348
|
+
)
|
|
349
|
+
if decision is not None:
|
|
350
|
+
decisions.append(decision)
|
|
351
|
+
return {
|
|
352
|
+
"input_count": len(inputs),
|
|
353
|
+
"lifecycle_status": dict(Counter(item.lifecycle_status for item in inputs)),
|
|
354
|
+
"decision_status": dict(Counter(item.decision_status for item in decisions)),
|
|
355
|
+
}
|
|
356
|
+
|
|
357
|
+
def get_run(self, run_id: str) -> MappingRun | None:
|
|
358
|
+
with self._session_factory() as session:
|
|
359
|
+
return session.get(MappingRun, run_id)
|
|
360
|
+
|
|
361
|
+
def get_inputs(self, run_id: str) -> list[MappingInput]:
|
|
362
|
+
with self._session_factory() as session:
|
|
363
|
+
return list(session.scalars(select(MappingInput).where(MappingInput.run_id == run_id)))
|
|
364
|
+
|
|
365
|
+
def latest_successful_run(
|
|
366
|
+
self, source_namespace: str, *, target_system: str | None = None
|
|
367
|
+
) -> MappingRun | None:
|
|
368
|
+
"""Return the newest complete run for a source and target system."""
|
|
369
|
+
with self._session_factory() as session:
|
|
370
|
+
query = select(MappingRun).where(
|
|
371
|
+
MappingRun.source_namespace == source_namespace,
|
|
372
|
+
MappingRun.lifecycle_status == "complete",
|
|
373
|
+
)
|
|
374
|
+
if target_system is not None:
|
|
375
|
+
query = query.where(MappingRun.target_system == target_system)
|
|
376
|
+
return session.scalar(query.order_by(MappingRun.updated_at.desc()))
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _id() -> str:
|
|
380
|
+
return str(uuid4())
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _now() -> datetime:
|
|
384
|
+
return datetime.now(UTC)
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def _candidate_key(spec: MappingCandidateSpec) -> str:
|
|
388
|
+
payload = {
|
|
389
|
+
"target_namespace": spec.target_namespace,
|
|
390
|
+
"target_vocabulary_id": spec.target_vocabulary_id,
|
|
391
|
+
"target_concept_id": spec.target_concept_id,
|
|
392
|
+
"target_code": spec.target_code,
|
|
393
|
+
"target_grain": spec.target_grain,
|
|
394
|
+
"target_role": spec.target_role,
|
|
395
|
+
"method": spec.method,
|
|
396
|
+
}
|
|
397
|
+
canonical = json.dumps(payload, sort_keys=True, separators=(",", ":"))
|
|
398
|
+
return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: groundstore
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Shared mapping-task contracts for Groundworkers workflows.
|
|
5
|
+
Requires-Python: >=3.12
|
|
6
|
+
Requires-Dist: oa-configurator<2,>=1.3.0
|
|
7
|
+
Requires-Dist: pydantic<3,>=2
|
|
8
|
+
Requires-Dist: sqlalchemy<3,>=2.0.45
|
|
9
|
+
Provides-Extra: dev
|
|
10
|
+
Requires-Dist: pytest<9,>=8; extra == 'dev'
|
|
11
|
+
Requires-Dist: ruff<1,>=0.4; extra == 'dev'
|
|
12
|
+
Requires-Dist: ty>=0.0.59; extra == 'dev'
|
|
13
|
+
Provides-Extra: postgres
|
|
14
|
+
Requires-Dist: psycopg[binary]<4,>=3.1; extra == 'postgres'
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
groundstore/__init__.py,sha256=wAvMCygyd7SqddIq8vzvRt71QC2RIIxs-F1Iry5qZn8,791
|
|
2
|
+
groundstore/context.py,sha256=d9RnKhFT4fk7Dx-1L9UqHwo2BQcqRauWNEWemEoThzI,8829
|
|
3
|
+
groundstore/contracts.py,sha256=WPn-5RAdmEZDy7JkW08949Wm04sbVYbFKXXFDrF_gaI,3524
|
|
4
|
+
groundstore/engine.py,sha256=plYQ0xhZv2RtI-kFFBiL8Djf625fT2iXUa7Lggp20U4,2171
|
|
5
|
+
groundstore/models.py,sha256=OrCfQiUN1yYSOGOgopZbruB2eWX_Rx7yJNO3miQ3QhA,9466
|
|
6
|
+
groundstore/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
7
|
+
groundstore/store.py,sha256=WCOffVJuhaePsPvcPrPGd3wub7Ce7qqZ_2oJcM38Ffw,16076
|
|
8
|
+
groundstore-0.1.0.dist-info/METADATA,sha256=oT-0XBcbcYb72egA4ERSzQQHKrNH5kfA6iakFkBPLrI,490
|
|
9
|
+
groundstore-0.1.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
|
|
10
|
+
groundstore-0.1.0.dist-info/RECORD,,
|