lp2graph 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. lp2graph/__init__.py +54 -0
  2. lp2graph/cli.py +238 -0
  3. lp2graph/codec/__init__.py +41 -0
  4. lp2graph/codec/latex.py +884 -0
  5. lp2graph/codec/normalize.py +82 -0
  6. lp2graph/core/__init__.py +35 -0
  7. lp2graph/core/graph.py +183 -0
  8. lp2graph/core/loader.py +63 -0
  9. lp2graph/core/model.py +437 -0
  10. lp2graph/core/validate.py +237 -0
  11. lp2graph/export/__init__.py +13 -0
  12. lp2graph/export/dgl.py +51 -0
  13. lp2graph/export/latex.py +126 -0
  14. lp2graph/export/networkx_adapter.py +50 -0
  15. lp2graph/export/pyg.py +79 -0
  16. lp2graph/export/pyomo_stub.py +81 -0
  17. lp2graph/metrics/__init__.py +58 -0
  18. lp2graph/metrics/classification.py +113 -0
  19. lp2graph/metrics/flags.py +122 -0
  20. lp2graph/metrics/result.py +26 -0
  21. lp2graph/metrics/structural.py +236 -0
  22. lp2graph/mining/__init__.py +47 -0
  23. lp2graph/mining/cluster/__init__.py +65 -0
  24. lp2graph/mining/cluster/agglomerative.py +82 -0
  25. lp2graph/mining/cluster/distance.py +65 -0
  26. lp2graph/mining/cluster/operator.py +218 -0
  27. lp2graph/mining/cluster/silhouette.py +88 -0
  28. lp2graph/mining/cluster/stability.py +178 -0
  29. lp2graph/mining/cluster/taxonomy.py +268 -0
  30. lp2graph/mining/corpusmgr/__init__.py +70 -0
  31. lp2graph/mining/corpusmgr/dedup.py +183 -0
  32. lp2graph/mining/corpusmgr/manager.py +79 -0
  33. lp2graph/mining/corpusmgr/manifest.py +82 -0
  34. lp2graph/mining/corpusmgr/record.py +101 -0
  35. lp2graph/mining/corpusmgr/select.py +128 -0
  36. lp2graph/mining/homologize/__init__.py +82 -0
  37. lp2graph/mining/homologize/concept.py +134 -0
  38. lp2graph/mining/homologize/entity.py +217 -0
  39. lp2graph/mining/homologize/lemmatize.py +80 -0
  40. lp2graph/mining/homologize/signature.py +166 -0
  41. lp2graph/mining/homologize/thesaurus.py +70 -0
  42. lp2graph/mining/homologize/tokenize.py +255 -0
  43. lp2graph/mining/homologize/vectorize.py +141 -0
  44. lp2graph/mining/ingest/__init__.py +59 -0
  45. lp2graph/mining/ingest/code_importers.py +104 -0
  46. lp2graph/mining/ingest/dispatch.py +148 -0
  47. lp2graph/mining/ingest/latex_normalizer.py +243 -0
  48. lp2graph/mining/ingest/pyomo_importer.py +297 -0
  49. lp2graph/mining/ingest/result.py +124 -0
  50. lp2graph/mining/isomorphism/__init__.py +26 -0
  51. lp2graph/mining/isomorphism/report.py +178 -0
  52. lp2graph/mining/label/__init__.py +70 -0
  53. lp2graph/mining/label/classifier.py +161 -0
  54. lp2graph/mining/label/features.py +35 -0
  55. lp2graph/mining/label/guardrails.py +176 -0
  56. lp2graph/mining/label/loop.py +314 -0
  57. lp2graph/mining/label/rules.py +92 -0
  58. lp2graph/mining/label/store.py +164 -0
  59. lp2graph/mining/label/vocab.py +64 -0
  60. lp2graph/mining/provenance.py +90 -0
  61. lp2graph/mining/versions.py +51 -0
  62. lp2graph/nl/__init__.py +15 -0
  63. lp2graph/nl/describe.py +301 -0
  64. lp2graph/render/__init__.py +11 -0
  65. lp2graph/render/palette.py +80 -0
  66. lp2graph/render/svg.py +220 -0
  67. lp2graph/solve/__init__.py +50 -0
  68. lp2graph/solve/grounder.py +405 -0
  69. lp2graph/solve/instance.py +76 -0
  70. lp2graph/transform/__init__.py +30 -0
  71. lp2graph/transform/bigm.py +173 -0
  72. lp2graph/views/__init__.py +17 -0
  73. lp2graph/views/ground.py +477 -0
  74. lp2graph/views/hybrid.py +202 -0
  75. lp2graph/views/schema.py +208 -0
  76. lp2graph-0.3.0.dist-info/METADATA +206 -0
  77. lp2graph-0.3.0.dist-info/RECORD +80 -0
  78. lp2graph-0.3.0.dist-info/WHEEL +4 -0
  79. lp2graph-0.3.0.dist-info/entry_points.txt +2 -0
  80. lp2graph-0.3.0.dist-info/licenses/LICENSE +205 -0
@@ -0,0 +1,268 @@
1
+ """Bottom-up multi-level taxonomy induction (M3).
2
+
3
+ Runs the ``CN`` operator at each level of the method's bottom-up taxonomy:
4
+
5
+ - **Level V** — decision variables and parameters, clustered on lexical
6
+ concepts + their type signatures.
7
+ - **Level C** — constraints and the objective, clustered on lexical concepts
8
+ + type signatures *conditioned on Level-V membership*: each constraint's
9
+ referenced variables contribute a ``vcluster:<name>`` feature, so two
10
+ constraints that bind structurally-similar variable families are pulled
11
+ together.
12
+ - **Level M** — whole models, clustered on family/type histograms, presence
13
+ flags, and bucketed structural metrics.
14
+
15
+ Plus two text-only one-dimensional clusterings — **domain** and **solution
16
+ approach** — over the model-level free text.
17
+
18
+ The whole induction is deterministic given the versioned
19
+ :class:`~lp2graph.mining.cluster.operator.ClusterConfig`.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from collections import Counter
25
+ from collections.abc import Sequence
26
+ from dataclasses import dataclass
27
+
28
+ from lp2graph.core.model import ConstraintTemplate, Formulation
29
+ from lp2graph.metrics.flags import presence_flags
30
+ from lp2graph.metrics.structural import model_completeness, structural_summary
31
+ from lp2graph.mining.cluster.operator import CN, ClusterConfig, NamedClustering
32
+ from lp2graph.mining.homologize.concept import concept_bag
33
+ from lp2graph.mining.homologize.entity import (
34
+ Entity,
35
+ corpus_entities,
36
+ signature_documents,
37
+ text_dimension,
38
+ )
39
+ from lp2graph.mining.homologize.vectorize import (
40
+ ConceptVectorizer,
41
+ Vocabulary,
42
+ build_vocabulary,
43
+ )
44
+ from lp2graph.views.schema import schema
45
+
46
+
47
+ @dataclass(frozen=True, slots=True)
48
+ class LevelResult:
49
+ """One level's entities, vocabulary, and named clustering."""
50
+
51
+ level: str
52
+ entities: tuple[Entity, ...]
53
+ vocabulary: Vocabulary
54
+ clustering: NamedClustering
55
+
56
+ def named_partition(self) -> dict[str, tuple[str, ...]]:
57
+ """Map cluster name → the entity ids it contains (for reporting)."""
58
+ out: dict[str, tuple[str, ...]] = {}
59
+ for cid, idxs in self.clustering.members.items():
60
+ name = self.clustering.names[cid]
61
+ out[name] = tuple(self.entities[i].id for i in idxs)
62
+ return out
63
+
64
+
65
+ @dataclass(frozen=True, slots=True)
66
+ class Taxonomy:
67
+ """The full multi-level taxonomy over a corpus."""
68
+
69
+ level_v: LevelResult
70
+ level_c: LevelResult
71
+ level_m: LevelResult
72
+ domain: LevelResult
73
+ solution_approach: LevelResult
74
+
75
+ def summary(self) -> dict[str, int]:
76
+ """Cluster counts per level (a compact, diffable digest)."""
77
+ return {
78
+ "V": self.level_v.clustering.n_clusters,
79
+ "C": self.level_c.clustering.n_clusters,
80
+ "M": self.level_m.clustering.n_clusters,
81
+ "domain": self.domain.clustering.n_clusters,
82
+ "solution_approach": self.solution_approach.clustering.n_clusters,
83
+ }
84
+
85
+
86
+ def _merge(*docs: dict[str, int]) -> dict[str, int]:
87
+ merged: Counter[str] = Counter()
88
+ for d in docs:
89
+ merged.update(d)
90
+ return dict(merged)
91
+
92
+
93
+ def _run_pass(
94
+ level: str,
95
+ entities: Sequence[Entity],
96
+ documents: Sequence[dict[str, int]],
97
+ config: ClusterConfig,
98
+ ) -> LevelResult:
99
+ vocab = build_vocabulary(documents)
100
+ _vectorizer, vectors = ConceptVectorizer.fit_transform(documents, vocabulary=vocab)
101
+ clustering = CN(entities, vectors, vocab, config)
102
+ return LevelResult(
103
+ level=level,
104
+ entities=tuple(entities),
105
+ vocabulary=vocab,
106
+ clustering=clustering,
107
+ )
108
+
109
+
110
+ def _referenced_variable_ids(f: Formulation) -> dict[str, list[str]]:
111
+ """Map each Level-C entity id of ``f`` to the variable entity ids it uses."""
112
+ out: dict[str, list[str]] = {}
113
+ for c in f.constraints:
114
+ cid = f"{f.id}::constraint::{c.name}"
115
+ refs = sorted({t.ref for t in (*c.lhs, *c.rhs) if t.ref_kind == "variable"})
116
+ out[cid] = [f"{f.id}::variable::{r}" for r in refs]
117
+ if f.objective is not None:
118
+ oid = f"{f.id}::objective::{f.objective.name}"
119
+ refs = sorted({t.ref for t in f.objective.terms if t.ref_kind == "variable"})
120
+ out[oid] = [f"{f.id}::variable::{r}" for r in refs]
121
+ return out
122
+
123
+
124
+ _DENSITY_EDGES = (0.05, 0.15, 0.30)
125
+ _CVR_EDGES = (0.75, 1.25, 2.0)
126
+
127
+
128
+ def _bucket(value: float, edges: Sequence[float]) -> int:
129
+ for i, e in enumerate(edges):
130
+ if value < e:
131
+ return i
132
+ return len(edges)
133
+
134
+
135
+ _SIZE_EDGES = (10.0, 50.0, 200.0, 1000.0)
136
+ _DIAMETER_CAP = 12
137
+
138
+
139
+ def model_feature_document(f: Formulation) -> dict[str, int]:
140
+ """Level-M structural feature document.
141
+
142
+ Covers the model-level channel of the paper's Level M: the presence-flag
143
+ vector ``φ`` and the structural metrics of Sec.~lp2graph — minimal size
144
+ ``S_min``, constraint/variable ratio ``R_C/V``, graph diameter ``D_G``,
145
+ edge density, and the two well-formedness indicators (coherence and
146
+ completeness) — each bucketed (or flagged) so it joins the
147
+ TF-IDF feature space. The raw declarative kind histograms are retained as
148
+ supplementary signal; the *induced* Level-V/Level-C cluster histograms are
149
+ added on top in :func:`induce` (they need the lower passes' output).
150
+ """
151
+ doc: Counter[str] = Counter()
152
+ doc[f"family:{f.family}"] += 1
153
+ for c in f.constraints:
154
+ doc[f"ckind:{c.kind}"] += 1
155
+ if c.domain_class:
156
+ doc[f"cdom:{c.domain_class}"] += 1
157
+ for v in f.variables:
158
+ doc[f"vdom:{v.domain}"] += 1
159
+ if v.domain_role:
160
+ doc[f"vrole:{v.domain_role}"] += 1
161
+ for name, result in presence_flags(f).items():
162
+ if bool(result.value):
163
+ doc[f"flag:{name}"] += 1
164
+ g = schema(f)
165
+ metrics = structural_summary(g)
166
+ density = float(metrics["edge_density"].value)
167
+ cvr = float(metrics["constraint_variable_ratio"].value)
168
+ size = float(metrics["minimal_size"].value)
169
+ diameter = int(metrics["graph_diameter"].value)
170
+ coherent = int(metrics["model_coherence"].value)
171
+ complete = int(model_completeness(f).value)
172
+ doc[f"density_bin:{_bucket(density, _DENSITY_EDGES)}"] += 1
173
+ doc[f"cvr_bin:{_bucket(cvr, _CVR_EDGES)}"] += 1
174
+ doc[f"size_bin:{_bucket(size, _SIZE_EDGES)}"] += 1
175
+ doc[f"diam:{min(diameter, _DIAMETER_CAP)}"] += 1
176
+ doc[f"coherent:{coherent}"] += 1
177
+ doc[f"complete:{complete}"] += 1
178
+ return dict(doc)
179
+
180
+
181
+ def induce(formulations: Sequence[Formulation], config: ClusterConfig | None = None) -> Taxonomy:
182
+ """Induce the full multi-level taxonomy over ``formulations``."""
183
+ cfg = config or ClusterConfig()
184
+
185
+ # Level V -------------------------------------------------------------
186
+ v_entities = corpus_entities(formulations, "V")
187
+ v_sig = signature_documents(v_entities)
188
+ v_docs = [_merge(concept_bag(e.text), v_sig[i]) for i, e in enumerate(v_entities)]
189
+ level_v = _run_pass("V", v_entities, v_docs, cfg)
190
+ v_label_by_id = {e.id: level_v.clustering.name_of(i) for i, e in enumerate(v_entities)}
191
+
192
+ # Level C (conditioned on Level-V membership) -------------------------
193
+ c_entities = corpus_entities(formulations, "C")
194
+ c_sig = signature_documents(c_entities)
195
+ ref_map: dict[str, list[str]] = {}
196
+ con_by_id: dict[str, ConstraintTemplate] = {}
197
+ for f in formulations:
198
+ ref_map.update(_referenced_variable_ids(f))
199
+ for c in f.constraints:
200
+ con_by_id[f"{f.id}::constraint::{c.name}"] = c
201
+ c_docs: list[dict[str, int]] = []
202
+ for i, e in enumerate(c_entities):
203
+ cond: Counter[str] = Counter()
204
+ # Histogram of Level-V cluster membership over the coupled variables.
205
+ for vid in ref_map.get(e.id, []):
206
+ name = v_label_by_id.get(vid)
207
+ if name is not None:
208
+ cond[f"vcluster:{name}"] += 1
209
+ # Relevant per-constraint presence flags (paper Level C: big-M,
210
+ # aggregation). Comparator + quantifier/restriction pattern + referent
211
+ # multiset already arrive via the signature document (c_sig).
212
+ con = con_by_id.get(e.id)
213
+ if con is not None:
214
+ if con.kind == "big_m" or con.indicator is not None:
215
+ cond["cflag:big_m"] += 1
216
+ if any(t.operator != "none" for t in (*con.lhs, *con.rhs)):
217
+ cond["cflag:aggregation"] += 1
218
+ c_docs.append(_merge(concept_bag(e.text), c_sig[i], dict(cond)))
219
+ level_c = _run_pass("C", c_entities, c_docs, cfg)
220
+ c_label_by_id = {e.id: level_c.clustering.name_of(i) for i, e in enumerate(c_entities)}
221
+
222
+ # Level M (conditioned on the induced Level-V/Level-C partitions) ------
223
+ m_entities = corpus_entities(formulations, "M")
224
+ m_docs: list[dict[str, int]] = []
225
+ for f in formulations:
226
+ induced: Counter[str] = Counter()
227
+ # Histogram of induced Level-C families present in the model.
228
+ for c in f.constraints:
229
+ fam = c_label_by_id.get(f"{f.id}::constraint::{c.name}")
230
+ if fam is not None:
231
+ induced[f"cfamily:{fam}"] += 1
232
+ if f.objective is not None:
233
+ fam = c_label_by_id.get(f"{f.id}::objective::{f.objective.name}")
234
+ if fam is not None:
235
+ induced[f"cfamily:{fam}"] += 1
236
+ # Histogram of induced Level-V types (variables and parameters).
237
+ for v in f.variables:
238
+ vtype = v_label_by_id.get(f"{f.id}::variable::{v.name}")
239
+ if vtype is not None:
240
+ induced[f"vtype:{vtype}"] += 1
241
+ for p in f.parameters:
242
+ vtype = v_label_by_id.get(f"{f.id}::parameter::{p.name}")
243
+ if vtype is not None:
244
+ induced[f"vtype:{vtype}"] += 1
245
+ m_docs.append(_merge(model_feature_document(f), dict(induced)))
246
+ level_m = _run_pass("M", m_entities, m_docs, cfg)
247
+
248
+ # Text-only one-dimensional clusterings -------------------------------
249
+ domain_docs = [concept_bag(text_dimension(e, "domain")) for e in m_entities]
250
+ domain = _run_pass("domain", m_entities, [dict(d) for d in domain_docs], cfg)
251
+ sol_docs = [concept_bag(text_dimension(e, "solution_approach")) for e in m_entities]
252
+ solution = _run_pass("solution_approach", m_entities, [dict(d) for d in sol_docs], cfg)
253
+
254
+ return Taxonomy(
255
+ level_v=level_v,
256
+ level_c=level_c,
257
+ level_m=level_m,
258
+ domain=domain,
259
+ solution_approach=solution,
260
+ )
261
+
262
+
263
+ __all__ = [
264
+ "LevelResult",
265
+ "Taxonomy",
266
+ "induce",
267
+ "model_feature_document",
268
+ ]
@@ -0,0 +1,70 @@
1
+ """M5 — corpus & provenance manager.
2
+
3
+ The corpus manager makes the extracted corpus *regenerable from queries +
4
+ freeze date* and makes representative selection a pure, reproducible
5
+ function. It provides:
6
+
7
+ - :class:`ProvenanceRecord` — bibliographic + categorization provenance for
8
+ one corpus entry (``citation_count`` is the count at the freeze date).
9
+ - :class:`CorpusManifest` — the reproducible record of what was searched
10
+ (queries + frozen search date), round-trippable to/from a plain dict.
11
+ - :func:`schema_graph_hash` / :func:`bibliographic_key` / :func:`deduplicate`
12
+ — deterministic dedup by schema-graph structure OR bibliographic key
13
+ (transitive), with a documented representative tie-break.
14
+ - :func:`select_representatives` — reproducible per-cluster representative
15
+ choice with documented citation/quality fallbacks.
16
+ - :class:`CorpusManager` — a thin facade tying the manifest and entries
17
+ together.
18
+
19
+ Everything here is deterministic: identical inputs yield identical output.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from lp2graph.mining.corpusmgr.dedup import (
25
+ DedupResult,
26
+ bibliographic_key,
27
+ deduplicate,
28
+ schema_graph_hash,
29
+ )
30
+ from lp2graph.mining.corpusmgr.manager import CorpusManager
31
+ from lp2graph.mining.corpusmgr.manifest import (
32
+ MANIFEST_SCHEMA_VERSION,
33
+ CorpusManifest,
34
+ manifest_from_dict,
35
+ manifest_to_dict,
36
+ )
37
+ from lp2graph.mining.corpusmgr.record import (
38
+ PRIORITY_CELLS,
39
+ QUALITY_TIERS,
40
+ PriorityCell,
41
+ ProvenanceRecord,
42
+ QualityTier,
43
+ quality_rank,
44
+ )
45
+ from lp2graph.mining.corpusmgr.select import (
46
+ RepresentativeChoice,
47
+ SelectionReason,
48
+ select_representatives,
49
+ )
50
+
51
+ __all__ = [
52
+ "MANIFEST_SCHEMA_VERSION",
53
+ "PRIORITY_CELLS",
54
+ "QUALITY_TIERS",
55
+ "CorpusManager",
56
+ "CorpusManifest",
57
+ "DedupResult",
58
+ "PriorityCell",
59
+ "ProvenanceRecord",
60
+ "QualityTier",
61
+ "RepresentativeChoice",
62
+ "SelectionReason",
63
+ "bibliographic_key",
64
+ "deduplicate",
65
+ "manifest_from_dict",
66
+ "manifest_to_dict",
67
+ "quality_rank",
68
+ "schema_graph_hash",
69
+ "select_representatives",
70
+ ]
@@ -0,0 +1,183 @@
1
+ """Deterministic deduplication of corpus formulations (M5).
2
+
3
+ Two formulations are *duplicates* if they share either of two signals:
4
+
5
+ - the **schema-graph hash** — a stable digest of the schema view's
6
+ *structure only* (node classes/subtypes/shapes + typed edges), ignoring
7
+ every cosmetic name, description, and id. Two structurally identical
8
+ formulations with different naming therefore collide.
9
+ - the **bibliographic key** — a normalized ``venue + year + source_id`` key,
10
+ catching the same publication ingested twice.
11
+
12
+ :func:`deduplicate` clusters items that share *either* signal, transitively
13
+ (union-find), and picks a representative per group by citation count with
14
+ documented, deterministic tie-breaks. Everything is sorted so repeated runs
15
+ over the same input produce byte-identical output.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import hashlib
21
+ import re
22
+ from collections.abc import Sequence
23
+ from dataclasses import dataclass
24
+
25
+ from lp2graph.core.graph import Graph
26
+ from lp2graph.core.model import Formulation
27
+ from lp2graph.mining.corpusmgr.record import ProvenanceRecord
28
+ from lp2graph.views.schema import schema
29
+
30
+ _PUNCT_RE = re.compile(r"[^\w]+")
31
+
32
+
33
+ def _canonicalize_graph(g: Graph) -> str:
34
+ """Render ``g`` as a canonical, name-free, sorted structural string.
35
+
36
+ Node *ids* and *labels* (which carry user-chosen names) are dropped; we
37
+ map each node id to a structural signature ``(cls, subtype, shape)`` and
38
+ describe edges by the structural signatures of their endpoints plus the
39
+ edge ``type`` and ``role``. Sorting every component makes the result
40
+ invariant to insertion order and to cosmetic renaming.
41
+ """
42
+ sig: dict[str, str] = {}
43
+ for n in g.nodes:
44
+ sig[n.id] = f"{n.cls}|{n.subtype}|{','.join(n.shape)}"
45
+
46
+ node_lines = sorted(sig.values())
47
+ edge_lines = sorted(f"{sig[e.src]}->{sig[e.dst]}|{e.type}|{e.role}" for e in g.edges)
48
+
49
+ return "N\n" + "\n".join(node_lines) + "\nE\n" + "\n".join(edge_lines)
50
+
51
+
52
+ def schema_graph_hash(f: Formulation) -> str:
53
+ """Return a stable hex digest of the schema-graph *structure* of ``f``.
54
+
55
+ The digest is derived from node classes/subtypes/shapes and typed edges
56
+ of the schema view only — never from names, descriptions, or ids — so
57
+ two structurally identical formulations hash equal regardless of
58
+ cosmetic naming. Uses SHA-256 and sorts all components for determinism.
59
+ """
60
+ canonical = _canonicalize_graph(schema(f))
61
+ return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
62
+
63
+
64
+ def _normalize(text: str) -> str:
65
+ """Casefold and collapse non-word runs to single spaces; strip ends."""
66
+ return _PUNCT_RE.sub(" ", text).casefold().strip()
67
+
68
+
69
+ def bibliographic_key(record: ProvenanceRecord) -> str:
70
+ """Return a normalized bibliographic match key for ``record``.
71
+
72
+ Built from ``venue + year + source_id``, casefolded with punctuation and
73
+ whitespace normalized, so trivially different spellings of the same
74
+ citation collapse to one key.
75
+ """
76
+ year = "" if record.year is None else str(record.year)
77
+ parts = [_normalize(record.venue), year, _normalize(record.source_id)]
78
+ return "|".join(parts)
79
+
80
+
81
+ @dataclass(frozen=True, slots=True)
82
+ class DedupResult:
83
+ """The outcome of :func:`deduplicate`.
84
+
85
+ ``groups`` are the duplicate clusters as tuples of indices into the
86
+ input sequence; each group is sorted ascending, and the groups
87
+ themselves are sorted. ``representatives`` gives, positionally aligned
88
+ with ``groups``, the chosen representative index for each group.
89
+ """
90
+
91
+ groups: tuple[tuple[int, ...], ...]
92
+ representatives: tuple[int, ...]
93
+
94
+ def representative_index(self, group_position: int) -> int:
95
+ """Representative input-index of the group at ``group_position``."""
96
+ return self.representatives[group_position]
97
+
98
+
99
+ class _UnionFind:
100
+ __slots__ = ("_parent",)
101
+
102
+ def __init__(self, n: int) -> None:
103
+ self._parent = list(range(n))
104
+
105
+ def find(self, x: int) -> int:
106
+ root = x
107
+ while self._parent[root] != root:
108
+ root = self._parent[root]
109
+ # path compression
110
+ while self._parent[x] != root:
111
+ self._parent[x], x = root, self._parent[x]
112
+ return root
113
+
114
+ def union(self, a: int, b: int) -> None:
115
+ ra, rb = self.find(a), self.find(b)
116
+ if ra != rb:
117
+ # attach larger root index under smaller for stable representatives
118
+ lo, hi = (ra, rb) if ra < rb else (rb, ra)
119
+ self._parent[hi] = lo
120
+
121
+
122
+ def _pick_representative(
123
+ members: Sequence[int],
124
+ records: Sequence[ProvenanceRecord],
125
+ ) -> int:
126
+ """Highest citation_count; tie -> best quality tier; tie -> lowest index."""
127
+ return min(
128
+ members,
129
+ key=lambda i: (
130
+ -records[i].citation_count,
131
+ records[i].quality_rank,
132
+ i,
133
+ ),
134
+ )
135
+
136
+
137
+ def deduplicate(
138
+ items: Sequence[tuple[Formulation, ProvenanceRecord]],
139
+ ) -> DedupResult:
140
+ """Group ``items`` that share a schema-graph hash OR a bibliographic key.
141
+
142
+ Grouping is transitive (union-find): if A matches B structurally and B
143
+ matches C bibliographically, all three land in one group. Within each
144
+ group the representative is the item with the highest
145
+ ``citation_count``, ties broken by best quality tier then lowest input
146
+ index. Fully deterministic.
147
+ """
148
+ n = len(items)
149
+ uf = _UnionFind(n)
150
+
151
+ by_hash: dict[str, int] = {}
152
+ by_bib: dict[str, int] = {}
153
+
154
+ for i, (formulation, record) in enumerate(items):
155
+ h = schema_graph_hash(formulation)
156
+ if h in by_hash:
157
+ uf.union(i, by_hash[h])
158
+ else:
159
+ by_hash[h] = i
160
+
161
+ b = bibliographic_key(record)
162
+ if b in by_bib:
163
+ uf.union(i, by_bib[b])
164
+ else:
165
+ by_bib[b] = i
166
+
167
+ clusters: dict[int, list[int]] = {}
168
+ for i in range(n):
169
+ clusters.setdefault(uf.find(i), []).append(i)
170
+
171
+ records = [r for _, r in items]
172
+ groups = sorted(tuple(sorted(members)) for members in clusters.values())
173
+ representatives = tuple(_pick_representative(g, records) for g in groups)
174
+
175
+ return DedupResult(groups=tuple(groups), representatives=representatives)
176
+
177
+
178
+ __all__ = [
179
+ "DedupResult",
180
+ "bibliographic_key",
181
+ "deduplicate",
182
+ "schema_graph_hash",
183
+ ]
@@ -0,0 +1,79 @@
1
+ """Thin corpus & provenance manager facade (M5).
2
+
3
+ :class:`CorpusManager` ties the pieces together: it holds the regeneration
4
+ :class:`~lp2graph.mining.corpusmgr.manifest.CorpusManifest` plus the list of
5
+ ``(Formulation, ProvenanceRecord)`` entries, and exposes the deterministic
6
+ operations (deduplicate, representative selection) and the manifest
7
+ (de)serialization that make the corpus *"regenerable from queries + freeze
8
+ date"*. It deliberately stays thin — all logic lives in the dedicated
9
+ modules.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from collections.abc import Iterable, Mapping, Sequence
15
+ from dataclasses import dataclass, field
16
+ from typing import Any
17
+
18
+ from lp2graph.core.model import Formulation
19
+ from lp2graph.mining.corpusmgr.dedup import DedupResult, deduplicate
20
+ from lp2graph.mining.corpusmgr.manifest import CorpusManifest
21
+ from lp2graph.mining.corpusmgr.record import ProvenanceRecord
22
+ from lp2graph.mining.corpusmgr.select import (
23
+ RepresentativeChoice,
24
+ select_representatives,
25
+ )
26
+
27
+
28
+ @dataclass(frozen=True, slots=True)
29
+ class CorpusManager:
30
+ """A corpus = a regeneration manifest + provenance-tagged formulations."""
31
+
32
+ manifest: CorpusManifest
33
+ entries: tuple[tuple[Formulation, ProvenanceRecord], ...] = field(default_factory=tuple)
34
+
35
+ @classmethod
36
+ def build(
37
+ cls,
38
+ manifest: CorpusManifest,
39
+ entries: Iterable[tuple[Formulation, ProvenanceRecord]],
40
+ ) -> CorpusManager:
41
+ """Construct a manager from a manifest and an iterable of entries."""
42
+ return cls(manifest=manifest, entries=tuple(entries))
43
+
44
+ @property
45
+ def records(self) -> tuple[ProvenanceRecord, ...]:
46
+ """The provenance records, positionally aligned with :attr:`entries`."""
47
+ return tuple(r for _, r in self.entries)
48
+
49
+ def deduplicate(self) -> DedupResult:
50
+ """Deterministically deduplicate the corpus entries."""
51
+ return deduplicate(self.entries)
52
+
53
+ def representatives(
54
+ self,
55
+ clusters: Mapping[str, Sequence[int]],
56
+ *,
57
+ benchmark_fallback: Mapping[str, int] | None = None,
58
+ ) -> dict[str, RepresentativeChoice]:
59
+ """Pick a reproducible representative per named cluster."""
60
+ return select_representatives(clusters, self.records, benchmark_fallback=benchmark_fallback)
61
+
62
+ def to_manifest_dict(self) -> dict[str, Any]:
63
+ """Serialize the regeneration manifest to a plain dict."""
64
+ return self.manifest.to_dict()
65
+
66
+ @classmethod
67
+ def from_manifest_dict(
68
+ cls,
69
+ data: dict[str, Any],
70
+ entries: Iterable[tuple[Formulation, ProvenanceRecord]] = (),
71
+ ) -> CorpusManager:
72
+ """Rebuild a manager from a serialized manifest (+ optional entries)."""
73
+ return cls(
74
+ manifest=CorpusManifest.from_dict(data),
75
+ entries=tuple(entries),
76
+ )
77
+
78
+
79
+ __all__ = ["CorpusManager"]
@@ -0,0 +1,82 @@
1
+ """The corpus regeneration manifest (M5).
2
+
3
+ The acceptance criterion for the corpus manager is *"corpus regenerable from
4
+ queries + freeze date"*. A :class:`CorpusManifest` is exactly that record:
5
+ the literal query strings that were issued and the frozen search date they
6
+ were issued against. Given the manifest, re-running the same queries at the
7
+ same freeze date reproduces the same candidate set — so the manifest is the
8
+ reproducible provenance of the whole corpus, not of any single entry.
9
+
10
+ It is a dependency-free frozen dataclass with a plain-``dict`` (JSON
11
+ round-trippable) serialization so it can be stored alongside the extracted
12
+ JSON and diffed.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from dataclasses import dataclass
18
+ from typing import Any
19
+
20
+ #: Bumped when the on-disk manifest dict layout changes.
21
+ MANIFEST_SCHEMA_VERSION = "1"
22
+
23
+
24
+ @dataclass(frozen=True, slots=True)
25
+ class CorpusManifest:
26
+ """The reproducible record of *what was searched* to build the corpus.
27
+
28
+ ``frozen_search_date`` is the ISO-8601 date the searches were frozen at
29
+ (an input string, never ``today``). ``queries`` is the ordered, literal
30
+ tuple of query strings issued. Together they make the candidate set
31
+ regenerable.
32
+ """
33
+
34
+ frozen_search_date: str
35
+ queries: tuple[str, ...] = ()
36
+ notes: str = ""
37
+
38
+ def __post_init__(self) -> None:
39
+ if not self.frozen_search_date:
40
+ raise ValueError("frozen_search_date must be a non-empty ISO date string")
41
+
42
+ def to_dict(self) -> dict[str, Any]:
43
+ """Serialize to a plain, JSON-round-trippable dict."""
44
+ return {
45
+ "manifest_schema_version": MANIFEST_SCHEMA_VERSION,
46
+ "frozen_search_date": self.frozen_search_date,
47
+ "queries": list(self.queries),
48
+ "notes": self.notes,
49
+ }
50
+
51
+ @classmethod
52
+ def from_dict(cls, data: dict[str, Any]) -> CorpusManifest:
53
+ """Reconstruct a manifest from its :meth:`to_dict` form."""
54
+ version = data.get("manifest_schema_version", MANIFEST_SCHEMA_VERSION)
55
+ if version != MANIFEST_SCHEMA_VERSION:
56
+ raise ValueError(
57
+ f"unsupported manifest_schema_version {version!r} "
58
+ f"(expected {MANIFEST_SCHEMA_VERSION!r})"
59
+ )
60
+ return cls(
61
+ frozen_search_date=str(data["frozen_search_date"]),
62
+ queries=tuple(str(q) for q in data.get("queries", ())),
63
+ notes=str(data.get("notes", "")),
64
+ )
65
+
66
+
67
+ def manifest_to_dict(manifest: CorpusManifest) -> dict[str, Any]:
68
+ """Free-function form of :meth:`CorpusManifest.to_dict`."""
69
+ return manifest.to_dict()
70
+
71
+
72
+ def manifest_from_dict(data: dict[str, Any]) -> CorpusManifest:
73
+ """Free-function form of :meth:`CorpusManifest.from_dict`."""
74
+ return CorpusManifest.from_dict(data)
75
+
76
+
77
+ __all__ = [
78
+ "MANIFEST_SCHEMA_VERSION",
79
+ "CorpusManifest",
80
+ "manifest_from_dict",
81
+ "manifest_to_dict",
82
+ ]