lp2graph 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lp2graph/__init__.py +54 -0
- lp2graph/cli.py +238 -0
- lp2graph/codec/__init__.py +41 -0
- lp2graph/codec/latex.py +884 -0
- lp2graph/codec/normalize.py +82 -0
- lp2graph/core/__init__.py +35 -0
- lp2graph/core/graph.py +183 -0
- lp2graph/core/loader.py +63 -0
- lp2graph/core/model.py +437 -0
- lp2graph/core/validate.py +237 -0
- lp2graph/export/__init__.py +13 -0
- lp2graph/export/dgl.py +51 -0
- lp2graph/export/latex.py +126 -0
- lp2graph/export/networkx_adapter.py +50 -0
- lp2graph/export/pyg.py +79 -0
- lp2graph/export/pyomo_stub.py +81 -0
- lp2graph/metrics/__init__.py +58 -0
- lp2graph/metrics/classification.py +113 -0
- lp2graph/metrics/flags.py +122 -0
- lp2graph/metrics/result.py +26 -0
- lp2graph/metrics/structural.py +236 -0
- lp2graph/mining/__init__.py +47 -0
- lp2graph/mining/cluster/__init__.py +65 -0
- lp2graph/mining/cluster/agglomerative.py +82 -0
- lp2graph/mining/cluster/distance.py +65 -0
- lp2graph/mining/cluster/operator.py +218 -0
- lp2graph/mining/cluster/silhouette.py +88 -0
- lp2graph/mining/cluster/stability.py +178 -0
- lp2graph/mining/cluster/taxonomy.py +268 -0
- lp2graph/mining/corpusmgr/__init__.py +70 -0
- lp2graph/mining/corpusmgr/dedup.py +183 -0
- lp2graph/mining/corpusmgr/manager.py +79 -0
- lp2graph/mining/corpusmgr/manifest.py +82 -0
- lp2graph/mining/corpusmgr/record.py +101 -0
- lp2graph/mining/corpusmgr/select.py +128 -0
- lp2graph/mining/homologize/__init__.py +82 -0
- lp2graph/mining/homologize/concept.py +134 -0
- lp2graph/mining/homologize/entity.py +217 -0
- lp2graph/mining/homologize/lemmatize.py +80 -0
- lp2graph/mining/homologize/signature.py +166 -0
- lp2graph/mining/homologize/thesaurus.py +70 -0
- lp2graph/mining/homologize/tokenize.py +255 -0
- lp2graph/mining/homologize/vectorize.py +141 -0
- lp2graph/mining/ingest/__init__.py +59 -0
- lp2graph/mining/ingest/code_importers.py +104 -0
- lp2graph/mining/ingest/dispatch.py +148 -0
- lp2graph/mining/ingest/latex_normalizer.py +243 -0
- lp2graph/mining/ingest/pyomo_importer.py +297 -0
- lp2graph/mining/ingest/result.py +124 -0
- lp2graph/mining/isomorphism/__init__.py +26 -0
- lp2graph/mining/isomorphism/report.py +178 -0
- lp2graph/mining/label/__init__.py +70 -0
- lp2graph/mining/label/classifier.py +161 -0
- lp2graph/mining/label/features.py +35 -0
- lp2graph/mining/label/guardrails.py +176 -0
- lp2graph/mining/label/loop.py +314 -0
- lp2graph/mining/label/rules.py +92 -0
- lp2graph/mining/label/store.py +164 -0
- lp2graph/mining/label/vocab.py +64 -0
- lp2graph/mining/provenance.py +90 -0
- lp2graph/mining/versions.py +51 -0
- lp2graph/nl/__init__.py +15 -0
- lp2graph/nl/describe.py +301 -0
- lp2graph/render/__init__.py +11 -0
- lp2graph/render/palette.py +80 -0
- lp2graph/render/svg.py +220 -0
- lp2graph/solve/__init__.py +50 -0
- lp2graph/solve/grounder.py +405 -0
- lp2graph/solve/instance.py +76 -0
- lp2graph/transform/__init__.py +30 -0
- lp2graph/transform/bigm.py +173 -0
- lp2graph/views/__init__.py +17 -0
- lp2graph/views/ground.py +477 -0
- lp2graph/views/hybrid.py +202 -0
- lp2graph/views/schema.py +208 -0
- lp2graph-0.3.0.dist-info/METADATA +206 -0
- lp2graph-0.3.0.dist-info/RECORD +80 -0
- lp2graph-0.3.0.dist-info/WHEEL +4 -0
- lp2graph-0.3.0.dist-info/entry_points.txt +2 -0
- lp2graph-0.3.0.dist-info/licenses/LICENSE +205 -0
|
@@ -0,0 +1,268 @@
|
|
|
1
|
+
"""Bottom-up multi-level taxonomy induction (M3).
|
|
2
|
+
|
|
3
|
+
Runs the ``CN`` operator at each level of the method's bottom-up taxonomy:
|
|
4
|
+
|
|
5
|
+
- **Level V** — decision variables and parameters, clustered on lexical
|
|
6
|
+
concepts + their type signatures.
|
|
7
|
+
- **Level C** — constraints and the objective, clustered on lexical concepts
|
|
8
|
+
+ type signatures *conditioned on Level-V membership*: each constraint's
|
|
9
|
+
referenced variables contribute a ``vcluster:<name>`` feature, so two
|
|
10
|
+
constraints that bind structurally-similar variable families are pulled
|
|
11
|
+
together.
|
|
12
|
+
- **Level M** — whole models, clustered on family/type histograms, presence
|
|
13
|
+
flags, and bucketed structural metrics.
|
|
14
|
+
|
|
15
|
+
Plus two text-only one-dimensional clusterings — **domain** and **solution
|
|
16
|
+
approach** — over the model-level free text.
|
|
17
|
+
|
|
18
|
+
The whole induction is deterministic given the versioned
|
|
19
|
+
:class:`~lp2graph.mining.cluster.operator.ClusterConfig`.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from collections import Counter
|
|
25
|
+
from collections.abc import Sequence
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
|
|
28
|
+
from lp2graph.core.model import ConstraintTemplate, Formulation
|
|
29
|
+
from lp2graph.metrics.flags import presence_flags
|
|
30
|
+
from lp2graph.metrics.structural import model_completeness, structural_summary
|
|
31
|
+
from lp2graph.mining.cluster.operator import CN, ClusterConfig, NamedClustering
|
|
32
|
+
from lp2graph.mining.homologize.concept import concept_bag
|
|
33
|
+
from lp2graph.mining.homologize.entity import (
|
|
34
|
+
Entity,
|
|
35
|
+
corpus_entities,
|
|
36
|
+
signature_documents,
|
|
37
|
+
text_dimension,
|
|
38
|
+
)
|
|
39
|
+
from lp2graph.mining.homologize.vectorize import (
|
|
40
|
+
ConceptVectorizer,
|
|
41
|
+
Vocabulary,
|
|
42
|
+
build_vocabulary,
|
|
43
|
+
)
|
|
44
|
+
from lp2graph.views.schema import schema
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True, slots=True)
|
|
48
|
+
class LevelResult:
|
|
49
|
+
"""One level's entities, vocabulary, and named clustering."""
|
|
50
|
+
|
|
51
|
+
level: str
|
|
52
|
+
entities: tuple[Entity, ...]
|
|
53
|
+
vocabulary: Vocabulary
|
|
54
|
+
clustering: NamedClustering
|
|
55
|
+
|
|
56
|
+
def named_partition(self) -> dict[str, tuple[str, ...]]:
|
|
57
|
+
"""Map cluster name → the entity ids it contains (for reporting)."""
|
|
58
|
+
out: dict[str, tuple[str, ...]] = {}
|
|
59
|
+
for cid, idxs in self.clustering.members.items():
|
|
60
|
+
name = self.clustering.names[cid]
|
|
61
|
+
out[name] = tuple(self.entities[i].id for i in idxs)
|
|
62
|
+
return out
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass(frozen=True, slots=True)
|
|
66
|
+
class Taxonomy:
|
|
67
|
+
"""The full multi-level taxonomy over a corpus."""
|
|
68
|
+
|
|
69
|
+
level_v: LevelResult
|
|
70
|
+
level_c: LevelResult
|
|
71
|
+
level_m: LevelResult
|
|
72
|
+
domain: LevelResult
|
|
73
|
+
solution_approach: LevelResult
|
|
74
|
+
|
|
75
|
+
def summary(self) -> dict[str, int]:
|
|
76
|
+
"""Cluster counts per level (a compact, diffable digest)."""
|
|
77
|
+
return {
|
|
78
|
+
"V": self.level_v.clustering.n_clusters,
|
|
79
|
+
"C": self.level_c.clustering.n_clusters,
|
|
80
|
+
"M": self.level_m.clustering.n_clusters,
|
|
81
|
+
"domain": self.domain.clustering.n_clusters,
|
|
82
|
+
"solution_approach": self.solution_approach.clustering.n_clusters,
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _merge(*docs: dict[str, int]) -> dict[str, int]:
|
|
87
|
+
merged: Counter[str] = Counter()
|
|
88
|
+
for d in docs:
|
|
89
|
+
merged.update(d)
|
|
90
|
+
return dict(merged)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _run_pass(
|
|
94
|
+
level: str,
|
|
95
|
+
entities: Sequence[Entity],
|
|
96
|
+
documents: Sequence[dict[str, int]],
|
|
97
|
+
config: ClusterConfig,
|
|
98
|
+
) -> LevelResult:
|
|
99
|
+
vocab = build_vocabulary(documents)
|
|
100
|
+
_vectorizer, vectors = ConceptVectorizer.fit_transform(documents, vocabulary=vocab)
|
|
101
|
+
clustering = CN(entities, vectors, vocab, config)
|
|
102
|
+
return LevelResult(
|
|
103
|
+
level=level,
|
|
104
|
+
entities=tuple(entities),
|
|
105
|
+
vocabulary=vocab,
|
|
106
|
+
clustering=clustering,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _referenced_variable_ids(f: Formulation) -> dict[str, list[str]]:
|
|
111
|
+
"""Map each Level-C entity id of ``f`` to the variable entity ids it uses."""
|
|
112
|
+
out: dict[str, list[str]] = {}
|
|
113
|
+
for c in f.constraints:
|
|
114
|
+
cid = f"{f.id}::constraint::{c.name}"
|
|
115
|
+
refs = sorted({t.ref for t in (*c.lhs, *c.rhs) if t.ref_kind == "variable"})
|
|
116
|
+
out[cid] = [f"{f.id}::variable::{r}" for r in refs]
|
|
117
|
+
if f.objective is not None:
|
|
118
|
+
oid = f"{f.id}::objective::{f.objective.name}"
|
|
119
|
+
refs = sorted({t.ref for t in f.objective.terms if t.ref_kind == "variable"})
|
|
120
|
+
out[oid] = [f"{f.id}::variable::{r}" for r in refs]
|
|
121
|
+
return out
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
_DENSITY_EDGES = (0.05, 0.15, 0.30)
|
|
125
|
+
_CVR_EDGES = (0.75, 1.25, 2.0)
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _bucket(value: float, edges: Sequence[float]) -> int:
|
|
129
|
+
for i, e in enumerate(edges):
|
|
130
|
+
if value < e:
|
|
131
|
+
return i
|
|
132
|
+
return len(edges)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
_SIZE_EDGES = (10.0, 50.0, 200.0, 1000.0)
|
|
136
|
+
_DIAMETER_CAP = 12
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def model_feature_document(f: Formulation) -> dict[str, int]:
|
|
140
|
+
"""Level-M structural feature document.
|
|
141
|
+
|
|
142
|
+
Covers the model-level channel of the paper's Level M: the presence-flag
|
|
143
|
+
vector ``φ`` and the structural metrics of Sec.~lp2graph — minimal size
|
|
144
|
+
``S_min``, constraint/variable ratio ``R_C/V``, graph diameter ``D_G``,
|
|
145
|
+
edge density, and the two well-formedness indicators (coherence and
|
|
146
|
+
completeness) — each bucketed (or flagged) so it joins the
|
|
147
|
+
TF-IDF feature space. The raw declarative kind histograms are retained as
|
|
148
|
+
supplementary signal; the *induced* Level-V/Level-C cluster histograms are
|
|
149
|
+
added on top in :func:`induce` (they need the lower passes' output).
|
|
150
|
+
"""
|
|
151
|
+
doc: Counter[str] = Counter()
|
|
152
|
+
doc[f"family:{f.family}"] += 1
|
|
153
|
+
for c in f.constraints:
|
|
154
|
+
doc[f"ckind:{c.kind}"] += 1
|
|
155
|
+
if c.domain_class:
|
|
156
|
+
doc[f"cdom:{c.domain_class}"] += 1
|
|
157
|
+
for v in f.variables:
|
|
158
|
+
doc[f"vdom:{v.domain}"] += 1
|
|
159
|
+
if v.domain_role:
|
|
160
|
+
doc[f"vrole:{v.domain_role}"] += 1
|
|
161
|
+
for name, result in presence_flags(f).items():
|
|
162
|
+
if bool(result.value):
|
|
163
|
+
doc[f"flag:{name}"] += 1
|
|
164
|
+
g = schema(f)
|
|
165
|
+
metrics = structural_summary(g)
|
|
166
|
+
density = float(metrics["edge_density"].value)
|
|
167
|
+
cvr = float(metrics["constraint_variable_ratio"].value)
|
|
168
|
+
size = float(metrics["minimal_size"].value)
|
|
169
|
+
diameter = int(metrics["graph_diameter"].value)
|
|
170
|
+
coherent = int(metrics["model_coherence"].value)
|
|
171
|
+
complete = int(model_completeness(f).value)
|
|
172
|
+
doc[f"density_bin:{_bucket(density, _DENSITY_EDGES)}"] += 1
|
|
173
|
+
doc[f"cvr_bin:{_bucket(cvr, _CVR_EDGES)}"] += 1
|
|
174
|
+
doc[f"size_bin:{_bucket(size, _SIZE_EDGES)}"] += 1
|
|
175
|
+
doc[f"diam:{min(diameter, _DIAMETER_CAP)}"] += 1
|
|
176
|
+
doc[f"coherent:{coherent}"] += 1
|
|
177
|
+
doc[f"complete:{complete}"] += 1
|
|
178
|
+
return dict(doc)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def induce(formulations: Sequence[Formulation], config: ClusterConfig | None = None) -> Taxonomy:
|
|
182
|
+
"""Induce the full multi-level taxonomy over ``formulations``."""
|
|
183
|
+
cfg = config or ClusterConfig()
|
|
184
|
+
|
|
185
|
+
# Level V -------------------------------------------------------------
|
|
186
|
+
v_entities = corpus_entities(formulations, "V")
|
|
187
|
+
v_sig = signature_documents(v_entities)
|
|
188
|
+
v_docs = [_merge(concept_bag(e.text), v_sig[i]) for i, e in enumerate(v_entities)]
|
|
189
|
+
level_v = _run_pass("V", v_entities, v_docs, cfg)
|
|
190
|
+
v_label_by_id = {e.id: level_v.clustering.name_of(i) for i, e in enumerate(v_entities)}
|
|
191
|
+
|
|
192
|
+
# Level C (conditioned on Level-V membership) -------------------------
|
|
193
|
+
c_entities = corpus_entities(formulations, "C")
|
|
194
|
+
c_sig = signature_documents(c_entities)
|
|
195
|
+
ref_map: dict[str, list[str]] = {}
|
|
196
|
+
con_by_id: dict[str, ConstraintTemplate] = {}
|
|
197
|
+
for f in formulations:
|
|
198
|
+
ref_map.update(_referenced_variable_ids(f))
|
|
199
|
+
for c in f.constraints:
|
|
200
|
+
con_by_id[f"{f.id}::constraint::{c.name}"] = c
|
|
201
|
+
c_docs: list[dict[str, int]] = []
|
|
202
|
+
for i, e in enumerate(c_entities):
|
|
203
|
+
cond: Counter[str] = Counter()
|
|
204
|
+
# Histogram of Level-V cluster membership over the coupled variables.
|
|
205
|
+
for vid in ref_map.get(e.id, []):
|
|
206
|
+
name = v_label_by_id.get(vid)
|
|
207
|
+
if name is not None:
|
|
208
|
+
cond[f"vcluster:{name}"] += 1
|
|
209
|
+
# Relevant per-constraint presence flags (paper Level C: big-M,
|
|
210
|
+
# aggregation). Comparator + quantifier/restriction pattern + referent
|
|
211
|
+
# multiset already arrive via the signature document (c_sig).
|
|
212
|
+
con = con_by_id.get(e.id)
|
|
213
|
+
if con is not None:
|
|
214
|
+
if con.kind == "big_m" or con.indicator is not None:
|
|
215
|
+
cond["cflag:big_m"] += 1
|
|
216
|
+
if any(t.operator != "none" for t in (*con.lhs, *con.rhs)):
|
|
217
|
+
cond["cflag:aggregation"] += 1
|
|
218
|
+
c_docs.append(_merge(concept_bag(e.text), c_sig[i], dict(cond)))
|
|
219
|
+
level_c = _run_pass("C", c_entities, c_docs, cfg)
|
|
220
|
+
c_label_by_id = {e.id: level_c.clustering.name_of(i) for i, e in enumerate(c_entities)}
|
|
221
|
+
|
|
222
|
+
# Level M (conditioned on the induced Level-V/Level-C partitions) ------
|
|
223
|
+
m_entities = corpus_entities(formulations, "M")
|
|
224
|
+
m_docs: list[dict[str, int]] = []
|
|
225
|
+
for f in formulations:
|
|
226
|
+
induced: Counter[str] = Counter()
|
|
227
|
+
# Histogram of induced Level-C families present in the model.
|
|
228
|
+
for c in f.constraints:
|
|
229
|
+
fam = c_label_by_id.get(f"{f.id}::constraint::{c.name}")
|
|
230
|
+
if fam is not None:
|
|
231
|
+
induced[f"cfamily:{fam}"] += 1
|
|
232
|
+
if f.objective is not None:
|
|
233
|
+
fam = c_label_by_id.get(f"{f.id}::objective::{f.objective.name}")
|
|
234
|
+
if fam is not None:
|
|
235
|
+
induced[f"cfamily:{fam}"] += 1
|
|
236
|
+
# Histogram of induced Level-V types (variables and parameters).
|
|
237
|
+
for v in f.variables:
|
|
238
|
+
vtype = v_label_by_id.get(f"{f.id}::variable::{v.name}")
|
|
239
|
+
if vtype is not None:
|
|
240
|
+
induced[f"vtype:{vtype}"] += 1
|
|
241
|
+
for p in f.parameters:
|
|
242
|
+
vtype = v_label_by_id.get(f"{f.id}::parameter::{p.name}")
|
|
243
|
+
if vtype is not None:
|
|
244
|
+
induced[f"vtype:{vtype}"] += 1
|
|
245
|
+
m_docs.append(_merge(model_feature_document(f), dict(induced)))
|
|
246
|
+
level_m = _run_pass("M", m_entities, m_docs, cfg)
|
|
247
|
+
|
|
248
|
+
# Text-only one-dimensional clusterings -------------------------------
|
|
249
|
+
domain_docs = [concept_bag(text_dimension(e, "domain")) for e in m_entities]
|
|
250
|
+
domain = _run_pass("domain", m_entities, [dict(d) for d in domain_docs], cfg)
|
|
251
|
+
sol_docs = [concept_bag(text_dimension(e, "solution_approach")) for e in m_entities]
|
|
252
|
+
solution = _run_pass("solution_approach", m_entities, [dict(d) for d in sol_docs], cfg)
|
|
253
|
+
|
|
254
|
+
return Taxonomy(
|
|
255
|
+
level_v=level_v,
|
|
256
|
+
level_c=level_c,
|
|
257
|
+
level_m=level_m,
|
|
258
|
+
domain=domain,
|
|
259
|
+
solution_approach=solution,
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
__all__ = [
|
|
264
|
+
"LevelResult",
|
|
265
|
+
"Taxonomy",
|
|
266
|
+
"induce",
|
|
267
|
+
"model_feature_document",
|
|
268
|
+
]
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""M5 — corpus & provenance manager.
|
|
2
|
+
|
|
3
|
+
The corpus manager makes the extracted corpus *regenerable from queries +
|
|
4
|
+
freeze date* and makes representative selection a pure, reproducible
|
|
5
|
+
function. It provides:
|
|
6
|
+
|
|
7
|
+
- :class:`ProvenanceRecord` — bibliographic + categorization provenance for
|
|
8
|
+
one corpus entry (``citation_count`` is the count at the freeze date).
|
|
9
|
+
- :class:`CorpusManifest` — the reproducible record of what was searched
|
|
10
|
+
(queries + frozen search date), round-trippable to/from a plain dict.
|
|
11
|
+
- :func:`schema_graph_hash` / :func:`bibliographic_key` / :func:`deduplicate`
|
|
12
|
+
— deterministic dedup by schema-graph structure OR bibliographic key
|
|
13
|
+
(transitive), with a documented representative tie-break.
|
|
14
|
+
- :func:`select_representatives` — reproducible per-cluster representative
|
|
15
|
+
choice with documented citation/quality fallbacks.
|
|
16
|
+
- :class:`CorpusManager` — a thin facade tying the manifest and entries
|
|
17
|
+
together.
|
|
18
|
+
|
|
19
|
+
Everything here is deterministic: identical inputs yield identical output.
|
|
20
|
+
"""
|
|
21
|
+
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
from lp2graph.mining.corpusmgr.dedup import (
|
|
25
|
+
DedupResult,
|
|
26
|
+
bibliographic_key,
|
|
27
|
+
deduplicate,
|
|
28
|
+
schema_graph_hash,
|
|
29
|
+
)
|
|
30
|
+
from lp2graph.mining.corpusmgr.manager import CorpusManager
|
|
31
|
+
from lp2graph.mining.corpusmgr.manifest import (
|
|
32
|
+
MANIFEST_SCHEMA_VERSION,
|
|
33
|
+
CorpusManifest,
|
|
34
|
+
manifest_from_dict,
|
|
35
|
+
manifest_to_dict,
|
|
36
|
+
)
|
|
37
|
+
from lp2graph.mining.corpusmgr.record import (
|
|
38
|
+
PRIORITY_CELLS,
|
|
39
|
+
QUALITY_TIERS,
|
|
40
|
+
PriorityCell,
|
|
41
|
+
ProvenanceRecord,
|
|
42
|
+
QualityTier,
|
|
43
|
+
quality_rank,
|
|
44
|
+
)
|
|
45
|
+
from lp2graph.mining.corpusmgr.select import (
|
|
46
|
+
RepresentativeChoice,
|
|
47
|
+
SelectionReason,
|
|
48
|
+
select_representatives,
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
__all__ = [
|
|
52
|
+
"MANIFEST_SCHEMA_VERSION",
|
|
53
|
+
"PRIORITY_CELLS",
|
|
54
|
+
"QUALITY_TIERS",
|
|
55
|
+
"CorpusManager",
|
|
56
|
+
"CorpusManifest",
|
|
57
|
+
"DedupResult",
|
|
58
|
+
"PriorityCell",
|
|
59
|
+
"ProvenanceRecord",
|
|
60
|
+
"QualityTier",
|
|
61
|
+
"RepresentativeChoice",
|
|
62
|
+
"SelectionReason",
|
|
63
|
+
"bibliographic_key",
|
|
64
|
+
"deduplicate",
|
|
65
|
+
"manifest_from_dict",
|
|
66
|
+
"manifest_to_dict",
|
|
67
|
+
"quality_rank",
|
|
68
|
+
"schema_graph_hash",
|
|
69
|
+
"select_representatives",
|
|
70
|
+
]
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
"""Deterministic deduplication of corpus formulations (M5).
|
|
2
|
+
|
|
3
|
+
Two formulations are *duplicates* if they share either of two signals:
|
|
4
|
+
|
|
5
|
+
- the **schema-graph hash** — a stable digest of the schema view's
|
|
6
|
+
*structure only* (node classes/subtypes/shapes + typed edges), ignoring
|
|
7
|
+
every cosmetic name, description, and id. Two structurally identical
|
|
8
|
+
formulations with different naming therefore collide.
|
|
9
|
+
- the **bibliographic key** — a normalized ``venue + year + source_id`` key,
|
|
10
|
+
catching the same publication ingested twice.
|
|
11
|
+
|
|
12
|
+
:func:`deduplicate` clusters items that share *either* signal, transitively
|
|
13
|
+
(union-find), and picks a representative per group by citation count with
|
|
14
|
+
documented, deterministic tie-breaks. Everything is sorted so repeated runs
|
|
15
|
+
over the same input produce byte-identical output.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
import re
|
|
22
|
+
from collections.abc import Sequence
|
|
23
|
+
from dataclasses import dataclass
|
|
24
|
+
|
|
25
|
+
from lp2graph.core.graph import Graph
|
|
26
|
+
from lp2graph.core.model import Formulation
|
|
27
|
+
from lp2graph.mining.corpusmgr.record import ProvenanceRecord
|
|
28
|
+
from lp2graph.views.schema import schema
|
|
29
|
+
|
|
30
|
+
_PUNCT_RE = re.compile(r"[^\w]+")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _canonicalize_graph(g: Graph) -> str:
|
|
34
|
+
"""Render ``g`` as a canonical, name-free, sorted structural string.
|
|
35
|
+
|
|
36
|
+
Node *ids* and *labels* (which carry user-chosen names) are dropped; we
|
|
37
|
+
map each node id to a structural signature ``(cls, subtype, shape)`` and
|
|
38
|
+
describe edges by the structural signatures of their endpoints plus the
|
|
39
|
+
edge ``type`` and ``role``. Sorting every component makes the result
|
|
40
|
+
invariant to insertion order and to cosmetic renaming.
|
|
41
|
+
"""
|
|
42
|
+
sig: dict[str, str] = {}
|
|
43
|
+
for n in g.nodes:
|
|
44
|
+
sig[n.id] = f"{n.cls}|{n.subtype}|{','.join(n.shape)}"
|
|
45
|
+
|
|
46
|
+
node_lines = sorted(sig.values())
|
|
47
|
+
edge_lines = sorted(f"{sig[e.src]}->{sig[e.dst]}|{e.type}|{e.role}" for e in g.edges)
|
|
48
|
+
|
|
49
|
+
return "N\n" + "\n".join(node_lines) + "\nE\n" + "\n".join(edge_lines)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def schema_graph_hash(f: Formulation) -> str:
|
|
53
|
+
"""Return a stable hex digest of the schema-graph *structure* of ``f``.
|
|
54
|
+
|
|
55
|
+
The digest is derived from node classes/subtypes/shapes and typed edges
|
|
56
|
+
of the schema view only — never from names, descriptions, or ids — so
|
|
57
|
+
two structurally identical formulations hash equal regardless of
|
|
58
|
+
cosmetic naming. Uses SHA-256 and sorts all components for determinism.
|
|
59
|
+
"""
|
|
60
|
+
canonical = _canonicalize_graph(schema(f))
|
|
61
|
+
return hashlib.sha256(canonical.encode("utf-8")).hexdigest()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _normalize(text: str) -> str:
|
|
65
|
+
"""Casefold and collapse non-word runs to single spaces; strip ends."""
|
|
66
|
+
return _PUNCT_RE.sub(" ", text).casefold().strip()
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def bibliographic_key(record: ProvenanceRecord) -> str:
|
|
70
|
+
"""Return a normalized bibliographic match key for ``record``.
|
|
71
|
+
|
|
72
|
+
Built from ``venue + year + source_id``, casefolded with punctuation and
|
|
73
|
+
whitespace normalized, so trivially different spellings of the same
|
|
74
|
+
citation collapse to one key.
|
|
75
|
+
"""
|
|
76
|
+
year = "" if record.year is None else str(record.year)
|
|
77
|
+
parts = [_normalize(record.venue), year, _normalize(record.source_id)]
|
|
78
|
+
return "|".join(parts)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
@dataclass(frozen=True, slots=True)
|
|
82
|
+
class DedupResult:
|
|
83
|
+
"""The outcome of :func:`deduplicate`.
|
|
84
|
+
|
|
85
|
+
``groups`` are the duplicate clusters as tuples of indices into the
|
|
86
|
+
input sequence; each group is sorted ascending, and the groups
|
|
87
|
+
themselves are sorted. ``representatives`` gives, positionally aligned
|
|
88
|
+
with ``groups``, the chosen representative index for each group.
|
|
89
|
+
"""
|
|
90
|
+
|
|
91
|
+
groups: tuple[tuple[int, ...], ...]
|
|
92
|
+
representatives: tuple[int, ...]
|
|
93
|
+
|
|
94
|
+
def representative_index(self, group_position: int) -> int:
|
|
95
|
+
"""Representative input-index of the group at ``group_position``."""
|
|
96
|
+
return self.representatives[group_position]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
class _UnionFind:
|
|
100
|
+
__slots__ = ("_parent",)
|
|
101
|
+
|
|
102
|
+
def __init__(self, n: int) -> None:
|
|
103
|
+
self._parent = list(range(n))
|
|
104
|
+
|
|
105
|
+
def find(self, x: int) -> int:
|
|
106
|
+
root = x
|
|
107
|
+
while self._parent[root] != root:
|
|
108
|
+
root = self._parent[root]
|
|
109
|
+
# path compression
|
|
110
|
+
while self._parent[x] != root:
|
|
111
|
+
self._parent[x], x = root, self._parent[x]
|
|
112
|
+
return root
|
|
113
|
+
|
|
114
|
+
def union(self, a: int, b: int) -> None:
|
|
115
|
+
ra, rb = self.find(a), self.find(b)
|
|
116
|
+
if ra != rb:
|
|
117
|
+
# attach larger root index under smaller for stable representatives
|
|
118
|
+
lo, hi = (ra, rb) if ra < rb else (rb, ra)
|
|
119
|
+
self._parent[hi] = lo
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _pick_representative(
|
|
123
|
+
members: Sequence[int],
|
|
124
|
+
records: Sequence[ProvenanceRecord],
|
|
125
|
+
) -> int:
|
|
126
|
+
"""Highest citation_count; tie -> best quality tier; tie -> lowest index."""
|
|
127
|
+
return min(
|
|
128
|
+
members,
|
|
129
|
+
key=lambda i: (
|
|
130
|
+
-records[i].citation_count,
|
|
131
|
+
records[i].quality_rank,
|
|
132
|
+
i,
|
|
133
|
+
),
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def deduplicate(
|
|
138
|
+
items: Sequence[tuple[Formulation, ProvenanceRecord]],
|
|
139
|
+
) -> DedupResult:
|
|
140
|
+
"""Group ``items`` that share a schema-graph hash OR a bibliographic key.
|
|
141
|
+
|
|
142
|
+
Grouping is transitive (union-find): if A matches B structurally and B
|
|
143
|
+
matches C bibliographically, all three land in one group. Within each
|
|
144
|
+
group the representative is the item with the highest
|
|
145
|
+
``citation_count``, ties broken by best quality tier then lowest input
|
|
146
|
+
index. Fully deterministic.
|
|
147
|
+
"""
|
|
148
|
+
n = len(items)
|
|
149
|
+
uf = _UnionFind(n)
|
|
150
|
+
|
|
151
|
+
by_hash: dict[str, int] = {}
|
|
152
|
+
by_bib: dict[str, int] = {}
|
|
153
|
+
|
|
154
|
+
for i, (formulation, record) in enumerate(items):
|
|
155
|
+
h = schema_graph_hash(formulation)
|
|
156
|
+
if h in by_hash:
|
|
157
|
+
uf.union(i, by_hash[h])
|
|
158
|
+
else:
|
|
159
|
+
by_hash[h] = i
|
|
160
|
+
|
|
161
|
+
b = bibliographic_key(record)
|
|
162
|
+
if b in by_bib:
|
|
163
|
+
uf.union(i, by_bib[b])
|
|
164
|
+
else:
|
|
165
|
+
by_bib[b] = i
|
|
166
|
+
|
|
167
|
+
clusters: dict[int, list[int]] = {}
|
|
168
|
+
for i in range(n):
|
|
169
|
+
clusters.setdefault(uf.find(i), []).append(i)
|
|
170
|
+
|
|
171
|
+
records = [r for _, r in items]
|
|
172
|
+
groups = sorted(tuple(sorted(members)) for members in clusters.values())
|
|
173
|
+
representatives = tuple(_pick_representative(g, records) for g in groups)
|
|
174
|
+
|
|
175
|
+
return DedupResult(groups=tuple(groups), representatives=representatives)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
__all__ = [
|
|
179
|
+
"DedupResult",
|
|
180
|
+
"bibliographic_key",
|
|
181
|
+
"deduplicate",
|
|
182
|
+
"schema_graph_hash",
|
|
183
|
+
]
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""Thin corpus & provenance manager facade (M5).
|
|
2
|
+
|
|
3
|
+
:class:`CorpusManager` ties the pieces together: it holds the regeneration
|
|
4
|
+
:class:`~lp2graph.mining.corpusmgr.manifest.CorpusManifest` plus the list of
|
|
5
|
+
``(Formulation, ProvenanceRecord)`` entries, and exposes the deterministic
|
|
6
|
+
operations (deduplicate, representative selection) and the manifest
|
|
7
|
+
(de)serialization that make the corpus *"regenerable from queries + freeze
|
|
8
|
+
date"*. It deliberately stays thin — all logic lives in the dedicated
|
|
9
|
+
modules.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
from typing import Any
|
|
17
|
+
|
|
18
|
+
from lp2graph.core.model import Formulation
|
|
19
|
+
from lp2graph.mining.corpusmgr.dedup import DedupResult, deduplicate
|
|
20
|
+
from lp2graph.mining.corpusmgr.manifest import CorpusManifest
|
|
21
|
+
from lp2graph.mining.corpusmgr.record import ProvenanceRecord
|
|
22
|
+
from lp2graph.mining.corpusmgr.select import (
|
|
23
|
+
RepresentativeChoice,
|
|
24
|
+
select_representatives,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True, slots=True)
|
|
29
|
+
class CorpusManager:
|
|
30
|
+
"""A corpus = a regeneration manifest + provenance-tagged formulations."""
|
|
31
|
+
|
|
32
|
+
manifest: CorpusManifest
|
|
33
|
+
entries: tuple[tuple[Formulation, ProvenanceRecord], ...] = field(default_factory=tuple)
|
|
34
|
+
|
|
35
|
+
@classmethod
|
|
36
|
+
def build(
|
|
37
|
+
cls,
|
|
38
|
+
manifest: CorpusManifest,
|
|
39
|
+
entries: Iterable[tuple[Formulation, ProvenanceRecord]],
|
|
40
|
+
) -> CorpusManager:
|
|
41
|
+
"""Construct a manager from a manifest and an iterable of entries."""
|
|
42
|
+
return cls(manifest=manifest, entries=tuple(entries))
|
|
43
|
+
|
|
44
|
+
@property
|
|
45
|
+
def records(self) -> tuple[ProvenanceRecord, ...]:
|
|
46
|
+
"""The provenance records, positionally aligned with :attr:`entries`."""
|
|
47
|
+
return tuple(r for _, r in self.entries)
|
|
48
|
+
|
|
49
|
+
def deduplicate(self) -> DedupResult:
|
|
50
|
+
"""Deterministically deduplicate the corpus entries."""
|
|
51
|
+
return deduplicate(self.entries)
|
|
52
|
+
|
|
53
|
+
def representatives(
|
|
54
|
+
self,
|
|
55
|
+
clusters: Mapping[str, Sequence[int]],
|
|
56
|
+
*,
|
|
57
|
+
benchmark_fallback: Mapping[str, int] | None = None,
|
|
58
|
+
) -> dict[str, RepresentativeChoice]:
|
|
59
|
+
"""Pick a reproducible representative per named cluster."""
|
|
60
|
+
return select_representatives(clusters, self.records, benchmark_fallback=benchmark_fallback)
|
|
61
|
+
|
|
62
|
+
def to_manifest_dict(self) -> dict[str, Any]:
|
|
63
|
+
"""Serialize the regeneration manifest to a plain dict."""
|
|
64
|
+
return self.manifest.to_dict()
|
|
65
|
+
|
|
66
|
+
@classmethod
|
|
67
|
+
def from_manifest_dict(
|
|
68
|
+
cls,
|
|
69
|
+
data: dict[str, Any],
|
|
70
|
+
entries: Iterable[tuple[Formulation, ProvenanceRecord]] = (),
|
|
71
|
+
) -> CorpusManager:
|
|
72
|
+
"""Rebuild a manager from a serialized manifest (+ optional entries)."""
|
|
73
|
+
return cls(
|
|
74
|
+
manifest=CorpusManifest.from_dict(data),
|
|
75
|
+
entries=tuple(entries),
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
__all__ = ["CorpusManager"]
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
"""The corpus regeneration manifest (M5).
|
|
2
|
+
|
|
3
|
+
The acceptance criterion for the corpus manager is *"corpus regenerable from
|
|
4
|
+
queries + freeze date"*. A :class:`CorpusManifest` is exactly that record:
|
|
5
|
+
the literal query strings that were issued and the frozen search date they
|
|
6
|
+
were issued against. Given the manifest, re-running the same queries at the
|
|
7
|
+
same freeze date reproduces the same candidate set — so the manifest is the
|
|
8
|
+
reproducible provenance of the whole corpus, not of any single entry.
|
|
9
|
+
|
|
10
|
+
It is a dependency-free frozen dataclass with a plain-``dict`` (JSON
|
|
11
|
+
round-trippable) serialization so it can be stored alongside the extracted
|
|
12
|
+
JSON and diffed.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
#: Bumped when the on-disk manifest dict layout changes.
|
|
21
|
+
MANIFEST_SCHEMA_VERSION = "1"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
@dataclass(frozen=True, slots=True)
|
|
25
|
+
class CorpusManifest:
|
|
26
|
+
"""The reproducible record of *what was searched* to build the corpus.
|
|
27
|
+
|
|
28
|
+
``frozen_search_date`` is the ISO-8601 date the searches were frozen at
|
|
29
|
+
(an input string, never ``today``). ``queries`` is the ordered, literal
|
|
30
|
+
tuple of query strings issued. Together they make the candidate set
|
|
31
|
+
regenerable.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
frozen_search_date: str
|
|
35
|
+
queries: tuple[str, ...] = ()
|
|
36
|
+
notes: str = ""
|
|
37
|
+
|
|
38
|
+
def __post_init__(self) -> None:
|
|
39
|
+
if not self.frozen_search_date:
|
|
40
|
+
raise ValueError("frozen_search_date must be a non-empty ISO date string")
|
|
41
|
+
|
|
42
|
+
def to_dict(self) -> dict[str, Any]:
|
|
43
|
+
"""Serialize to a plain, JSON-round-trippable dict."""
|
|
44
|
+
return {
|
|
45
|
+
"manifest_schema_version": MANIFEST_SCHEMA_VERSION,
|
|
46
|
+
"frozen_search_date": self.frozen_search_date,
|
|
47
|
+
"queries": list(self.queries),
|
|
48
|
+
"notes": self.notes,
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
@classmethod
|
|
52
|
+
def from_dict(cls, data: dict[str, Any]) -> CorpusManifest:
|
|
53
|
+
"""Reconstruct a manifest from its :meth:`to_dict` form."""
|
|
54
|
+
version = data.get("manifest_schema_version", MANIFEST_SCHEMA_VERSION)
|
|
55
|
+
if version != MANIFEST_SCHEMA_VERSION:
|
|
56
|
+
raise ValueError(
|
|
57
|
+
f"unsupported manifest_schema_version {version!r} "
|
|
58
|
+
f"(expected {MANIFEST_SCHEMA_VERSION!r})"
|
|
59
|
+
)
|
|
60
|
+
return cls(
|
|
61
|
+
frozen_search_date=str(data["frozen_search_date"]),
|
|
62
|
+
queries=tuple(str(q) for q in data.get("queries", ())),
|
|
63
|
+
notes=str(data.get("notes", "")),
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def manifest_to_dict(manifest: CorpusManifest) -> dict[str, Any]:
|
|
68
|
+
"""Free-function form of :meth:`CorpusManifest.to_dict`."""
|
|
69
|
+
return manifest.to_dict()
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def manifest_from_dict(data: dict[str, Any]) -> CorpusManifest:
|
|
73
|
+
"""Free-function form of :meth:`CorpusManifest.from_dict`."""
|
|
74
|
+
return CorpusManifest.from_dict(data)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
__all__ = [
|
|
78
|
+
"MANIFEST_SCHEMA_VERSION",
|
|
79
|
+
"CorpusManifest",
|
|
80
|
+
"manifest_from_dict",
|
|
81
|
+
"manifest_to_dict",
|
|
82
|
+
]
|