dense-evolution 8.3.0__py3-none-win_amd64.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dashboard_core/__init__.py +115 -0
- dashboard_core/_gate_tables.py +30 -0
- dashboard_core/band_structure.py +71 -0
- dashboard_core/circuit_builder_component.py +232 -0
- dashboard_core/circuit_diagram.py +216 -0
- dashboard_core/crypto_protocols.py +77 -0
- dashboard_core/engine.py +326 -0
- dashboard_core/graphical_builder.py +114 -0
- dashboard_core/hamiltonians.py +593 -0
- dashboard_core/mass_decomposition_tool.py +47 -0
- dashboard_core/mitigation.py +343 -0
- dashboard_core/native_hf_diagnostics.py +62 -0
- dashboard_core/noise_tools.py +125 -0
- dashboard_core/qasm_library.py +233 -0
- dashboard_core/qmmm.py +16 -0
- dashboard_core/rag_tool.py +45 -0
- dashboard_core/state_visuals.py +288 -0
- dashboard_core/system_limits.py +60 -0
- dashboard_core/vector_healing.py +102 -0
- dashboard_core/visuals.py +158 -0
- dashboard_core/vqe.py +533 -0
- dashboard_core/wormhole.py +580 -0
- dense_evolution/__init__.py +114 -0
- dense_evolution/autodiff.py +10 -0
- dense_evolution/backends/__init__.py +5 -0
- dense_evolution/backends/chunk/__init__.py +37 -0
- dense_evolution/backends/chunk/_engine_imports.py +57 -0
- dense_evolution/backends/chunk/circuit_chunker.py +55 -0
- dense_evolution/backends/chunk/core.py +432 -0
- dense_evolution/backends/chunk/disk_overflow.py +232 -0
- dense_evolution/backends/chunk/geometry.py +95 -0
- dense_evolution/backends/chunk/guard.py +190 -0
- dense_evolution/backends/chunk/kernels.py +531 -0
- dense_evolution/backends/mps.py +1569 -0
- dense_evolution/backends/statevector.py +616 -0
- dense_evolution/chunk.py +25 -0
- dense_evolution/circuits/__init__.py +20 -0
- dense_evolution/circuits/compiler.py +488 -0
- dense_evolution/circuits/diagram.py +94 -0
- dense_evolution/circuits/gates.py +91 -0
- dense_evolution/circuits/parser.py +632 -0
- dense_evolution/circuits/qft.py +66 -0
- dense_evolution/circuits/random_circuit.py +85 -0
- dense_evolution/circuits/registry.py +74 -0
- dense_evolution/circuits/topology.py +79 -0
- dense_evolution/circuits/trotter.py +265 -0
- dense_evolution/circuits/uccsd.py +275 -0
- dense_evolution/cli.py +199 -0
- dense_evolution/compiler.py +9 -0
- dense_evolution/config.py +49 -0
- dense_evolution/drawing.py +10 -0
- dense_evolution/entropy.py +9 -0
- dense_evolution/fermions.py +9 -0
- dense_evolution/gates.py +9 -0
- dense_evolution/harrison_tb.py +16 -0
- dense_evolution/healing.py +18 -0
- dense_evolution/interop/__init__.py +18 -0
- dense_evolution/interop/qiskit_pennylane.py +406 -0
- dense_evolution/measurement.py +10 -0
- dense_evolution/mitigation/__init__.py +54 -0
- dense_evolution/mitigation/healing.py +215 -0
- dense_evolution/mitigation/kl_divergence.py +93 -0
- dense_evolution/mitigation/magic_entropy.py +163 -0
- dense_evolution/mitigation/magic_entropy_shadows.py +262 -0
- dense_evolution/mitigation/renyi.py +168 -0
- dense_evolution/mitigation/stabilizer_renyi_entropy.py +103 -0
- dense_evolution/mitigation/zne.py +990 -0
- dense_evolution/mps.py +9 -0
- dense_evolution/native_hf/__init__.py +26 -0
- dense_evolution/native_hf/_libcint/LICENSE-libcint +10 -0
- dense_evolution/native_hf/_libcint/libdecint.dll +0 -0
- dense_evolution/native_hf/assembly.py +304 -0
- dense_evolution/native_hf/basis.py +117 -0
- dense_evolution/native_hf/boys.py +35 -0
- dense_evolution/native_hf/bridge.py +112 -0
- dense_evolution/native_hf/cartesian.py +64 -0
- dense_evolution/native_hf/coulomb.py +196 -0
- dense_evolution/native_hf/differentiable.py +53 -0
- dense_evolution/native_hf/gaussians.py +79 -0
- dense_evolution/native_hf/kinetic.py +52 -0
- dense_evolution/native_hf/libcint_bridge.py +167 -0
- dense_evolution/native_hf/overlap.py +91 -0
- dense_evolution/native_hf/scf.py +404 -0
- dense_evolution/noise/__init__.py +79 -0
- dense_evolution/noise/coherent_attack.py +264 -0
- dense_evolution/noise/cosmic_ray.py +61 -0
- dense_evolution/noise/density_matrix_channels.py +78 -0
- dense_evolution/noise/differentiable.py +66 -0
- dense_evolution/noise/kraus/__init__.py +6 -0
- dense_evolution/noise/kraus/amplitude_damping.py +47 -0
- dense_evolution/noise/kraus/bitflip.py +22 -0
- dense_evolution/noise/kraus/combined.py +16 -0
- dense_evolution/noise/kraus/depolarizing.py +47 -0
- dense_evolution/noise/kraus/ideal.py +10 -0
- dense_evolution/noise/kraus/phaseflip.py +21 -0
- dense_evolution/noise/kraus_channels.py +285 -0
- dense_evolution/noise/oscillating.py +32 -0
- dense_evolution/noise/pink.py +80 -0
- dense_evolution/observables.py +11 -0
- dense_evolution/parser.py +9 -0
- dense_evolution/physics/__init__.py +27 -0
- dense_evolution/physics/entropy.py +161 -0
- dense_evolution/physics/fermions.py +322 -0
- dense_evolution/physics/observables.py +523 -0
- dense_evolution/physics/qec.py +1113 -0
- dense_evolution/physics/spectral.py +143 -0
- dense_evolution/physics/states.py +43 -0
- dense_evolution/protocols/__init__.py +27 -0
- dense_evolution/protocols/bb84.py +133 -0
- dense_evolution/protocols/di_qkd_ghz.py +199 -0
- dense_evolution/protocols/dicka_protocol2.py +124 -0
- dense_evolution/qec.py +20 -0
- dense_evolution/qft.py +9 -0
- dense_evolution/qmmm/__init__.py +13 -0
- dense_evolution/qmmm/ase_bridge.py +97 -0
- dense_evolution/qmmm/forces.py +388 -0
- dense_evolution/qmmm/propagation.py +80 -0
- dense_evolution/qmmm/region.py +137 -0
- dense_evolution/random_circuit.py +15 -0
- dense_evolution/registry.py +9 -0
- dense_evolution/simulator.py +10 -0
- dense_evolution/solvers/__init__.py +19 -0
- dense_evolution/solvers/autodiff.py +169 -0
- dense_evolution/solvers/harrison_tb.py +189 -0
- dense_evolution/solvers/vhd_tb.py +187 -0
- dense_evolution/states.py +9 -0
- dense_evolution/topology.py +9 -0
- dense_evolution/trotter.py +9 -0
- dense_evolution/utils/__init__.py +13 -0
- dense_evolution/utils/drawing.py +101 -0
- dense_evolution/utils/mass_decomposition.py +246 -0
- dense_evolution/utils/measurement.py +94 -0
- dense_evolution/vhd_tb.py +16 -0
- dense_evolution-8.3.0.dist-info/METADATA +366 -0
- dense_evolution-8.3.0.dist-info/RECORD +165 -0
- dense_evolution-8.3.0.dist-info/WHEEL +5 -0
- dense_evolution-8.3.0.dist-info/entry_points.txt +2 -0
- dense_evolution-8.3.0.dist-info/licenses/license.md +58 -0
- dense_evolution-8.3.0.dist-info/top_level.txt +5 -0
- ia_utils/__init__.py +0 -0
- ia_utils/adversarial_vector_attack.py +196 -0
- ia_utils/rag.py +288 -0
- ia_utils/vector_healing.py +399 -0
- local_site/__init__.py +0 -0
- local_site/app/__init__.py +0 -0
- local_site/app/server.py +1009 -0
- mcp_server/__init__.py +0 -0
- mcp_server/client.py +324 -0
- mcp_server/config.py +32 -0
- mcp_server/models.py +347 -0
- mcp_server/molecules.py +71 -0
- mcp_server/server.py +119 -0
- mcp_server/tools/__init__.py +0 -0
- mcp_server/tools/chemistry_tools.py +225 -0
- mcp_server/tools/circuit_tools.py +83 -0
- mcp_server/tools/crypto_tools.py +66 -0
- mcp_server/tools/mitigation_tools.py +81 -0
- mcp_server/tools/noise_tools.py +60 -0
- mcp_server/tools/retrieval_tools.py +44 -0
- mcp_server/tools/system_tools.py +149 -0
- mcp_server/tools/wormhole_tools.py +142 -0
- mcp_server/utils/__init__.py +0 -0
- mcp_server/utils/cache.py +55 -0
- mcp_server/utils/images.py +67 -0
- mcp_server/utils/truncation.py +38 -0
|
@@ -0,0 +1,196 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Gradient-based adversarial stress-testing for enhanced_dense_healing_hybrid's
|
|
3
|
+
Phi-Trigger decision -- a targeted, crafted perturbation instead of the
|
|
4
|
+
random NaN/Inf corruption this healing pipeline is normally tested
|
|
5
|
+
against, adapted from IGME's chained-differentiable-attack idea
|
|
6
|
+
(arXiv:2607.27465, "IGME: Efficient Chained Method Ensemble for
|
|
7
|
+
Transferable Semantic Segmentation Attacks") applied to vector sequences
|
|
8
|
+
instead of image segmentation.
|
|
9
|
+
|
|
10
|
+
evaluate_phi_trigger (dense_evolution.healing) makes its keep-vs-median-
|
|
11
|
+
replace decision by thresholding |v_dinamic| against NON_STATIC_THRESHOLD_A
|
|
12
|
+
-- a hard step, not differentiable at the boundary. But v_dinamic itself
|
|
13
|
+
(via calculate_phi_ab -> calculate_vettore_dinamico) is built entirely
|
|
14
|
+
from norms, dot products, log, and clip -- all JAX-differentiable. This
|
|
15
|
+
crafts a minimal perturbation to a single vector in the sequence, via
|
|
16
|
+
gradient ascent/descent on |v_dinamic| (projected back into an epsilon
|
|
17
|
+
L2-ball each step, the standard PGD pattern), that flips the trigger's
|
|
18
|
+
decision at that point:
|
|
19
|
+
|
|
20
|
+
- "flip_to_dynamic": push an originally-static point (would be replaced
|
|
21
|
+
by the local median) across the threshold so the trigger keeps it
|
|
22
|
+
as-is instead -- the more security-relevant direction, since it
|
|
23
|
+
represents a worst-case corruption crafted to *evade* the healer by
|
|
24
|
+
looking like genuine motion, rather than obvious noise.
|
|
25
|
+
- "flip_to_static": push an originally-dynamic point (would be kept)
|
|
26
|
+
across the threshold so the trigger discards it as noise instead --
|
|
27
|
+
the failure mode of genuine signal getting wrongly suppressed.
|
|
28
|
+
|
|
29
|
+
This does not attack the median-filter fallback itself, only the
|
|
30
|
+
Phi-Trigger's keep-vs-replace decision -- see craft_adversarial_healing_
|
|
31
|
+
perturbation's own docstring for what "success" means precisely.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from typing import Optional
|
|
35
|
+
|
|
36
|
+
import numpy as np
|
|
37
|
+
import jax
|
|
38
|
+
import jax.numpy as jnp
|
|
39
|
+
|
|
40
|
+
from dense_evolution.mitigation.healing import calculate_phi_ab, calculate_vettore_dinamico, GLOBAL_CONSTANTS
|
|
41
|
+
|
|
42
|
+
__all__ = ['craft_adversarial_healing_perturbation']
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _trigger_magnitude(state_B, state_A, ipg_vector):
|
|
46
|
+
"""|v_dinamic| -- the continuous quantity evaluate_phi_trigger
|
|
47
|
+
thresholds. Differentiable w.r.t. state_B (the vector being attacked);
|
|
48
|
+
state_A/ipg_vector are treated as fixed context, not attacked."""
|
|
49
|
+
phi_ab = calculate_phi_ab(state_A, state_B, ipg_vector)
|
|
50
|
+
E_A = jnp.linalg.norm(state_A)
|
|
51
|
+
E_B = jnp.linalg.norm(state_B)
|
|
52
|
+
v_dinamic = calculate_vettore_dinamico(E_A, E_B, phi_ab)
|
|
53
|
+
return jnp.abs(v_dinamic)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
_trigger_magnitude_grad = jax.grad(_trigger_magnitude, argnums=0)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def craft_adversarial_healing_perturbation(
|
|
60
|
+
vettori: np.ndarray,
|
|
61
|
+
target_idx: int,
|
|
62
|
+
radius_baseline: Optional[int] = None,
|
|
63
|
+
epsilon: float = 0.1,
|
|
64
|
+
n_steps: int = 50,
|
|
65
|
+
step_size: Optional[float] = None,
|
|
66
|
+
direction: str = "flip_to_dynamic",
|
|
67
|
+
) -> dict:
|
|
68
|
+
"""Crafts a minimal adversarial perturbation to vettori[target_idx],
|
|
69
|
+
within an L2 epsilon-ball, that flips enhanced_dense_healing_hybrid's
|
|
70
|
+
Phi-Trigger decision at that index -- a targeted stress test, not
|
|
71
|
+
random noise.
|
|
72
|
+
|
|
73
|
+
Reproduces enhanced_dense_healing_hybrid's own per-step computation
|
|
74
|
+
at target_idx exactly (same baseline_mean window, same adaptive
|
|
75
|
+
radius default, same inter-point-gradient vector) so the crafted
|
|
76
|
+
perturbation is faithful to what the real healing pipeline would
|
|
77
|
+
actually see, not a simplified stand-in.
|
|
78
|
+
|
|
79
|
+
BUG FIX: step_size used to default to 2*epsilon/n_steps -- tying the
|
|
80
|
+
per-step move size to the epsilon budget. Verified directly this
|
|
81
|
+
makes LARGER epsilon give WORSE (higher final |v_dinamic|) results,
|
|
82
|
+
not better: a bigger budget means a bigger step, which overshoots
|
|
83
|
+
and oscillates around the minimum instead of converging to it,
|
|
84
|
+
which is the opposite of what a bigger budget should ever do for a
|
|
85
|
+
correct optimizer. step_size is now independent of epsilon (a small
|
|
86
|
+
fixed default, tuned to v_dinamic's typical local scale) -- epsilon
|
|
87
|
+
only bounds where the iterate is allowed to end up (via projection
|
|
88
|
+
after each step), not how big each step is.
|
|
89
|
+
|
|
90
|
+
Args:
|
|
91
|
+
vettori: array-like, shape (n_steps, dim), the (unperturbed)
|
|
92
|
+
vector sequence. Not modified in place.
|
|
93
|
+
target_idx: index to attack; must be >= 2 (the healing loop's
|
|
94
|
+
own starting point) and < len(vettori).
|
|
95
|
+
radius_baseline: same meaning as enhanced_dense_healing_hybrid's
|
|
96
|
+
own parameter; None uses the same adaptive default.
|
|
97
|
+
epsilon: L2-norm budget for the perturbation (the attack is
|
|
98
|
+
projected back into this ball after every gradient step).
|
|
99
|
+
n_steps: number of PGD-style gradient steps.
|
|
100
|
+
step_size: per-step move size along the normalized gradient;
|
|
101
|
+
None uses a small fixed default (0.02) independent of
|
|
102
|
+
epsilon -- see the bug-fix note above for why.
|
|
103
|
+
direction: "flip_to_dynamic" (evade -- make static-looking
|
|
104
|
+
input pass through unhealed) or "flip_to_static" (suppress
|
|
105
|
+
-- make dynamic-looking input get median-replaced instead).
|
|
106
|
+
|
|
107
|
+
Returns:
|
|
108
|
+
dict with:
|
|
109
|
+
perturbed_vettori: copy of vettori with vettori[target_idx]
|
|
110
|
+
replaced by the crafted perturbation.
|
|
111
|
+
success: bool, True only if the trigger decision actually
|
|
112
|
+
flipped in the requested direction (a small epsilon or
|
|
113
|
+
too few steps can fail to cross the threshold).
|
|
114
|
+
original_trigger_active / final_trigger_active: bool, the
|
|
115
|
+
Phi-Trigger's decision (True = dynamic/kept) before and
|
|
116
|
+
after the perturbation.
|
|
117
|
+
original_magnitude / final_magnitude: float, |v_dinamic|
|
|
118
|
+
before and after.
|
|
119
|
+
perturbation_norm: float, actual L2 norm of the applied
|
|
120
|
+
perturbation (<= epsilon).
|
|
121
|
+
"""
|
|
122
|
+
vettori = np.asarray(vettori, dtype=np.float64)
|
|
123
|
+
n, hidden_dim = vettori.shape
|
|
124
|
+
if target_idx < 2 or target_idx >= n:
|
|
125
|
+
raise ValueError(f"target_idx must be in [2, {n - 1}], got {target_idx} (n={n})")
|
|
126
|
+
if direction not in ("flip_to_dynamic", "flip_to_static"):
|
|
127
|
+
raise ValueError(f"direction must be 'flip_to_dynamic' or 'flip_to_static', got {direction!r}")
|
|
128
|
+
if epsilon <= 0:
|
|
129
|
+
raise ValueError(f"epsilon must be positive, got {epsilon}")
|
|
130
|
+
|
|
131
|
+
if radius_baseline is None:
|
|
132
|
+
radius_baseline = min(20, max(3, n // 3))
|
|
133
|
+
lo = max(0, target_idx - radius_baseline)
|
|
134
|
+
|
|
135
|
+
state_A = jnp.array(np.mean(vettori[lo:target_idx], axis=0))
|
|
136
|
+
ipg_raw = vettori[target_idx - 1] - vettori[target_idx - 2]
|
|
137
|
+
norm_ipg_raw = np.linalg.norm(ipg_raw)
|
|
138
|
+
ipg_vector = jnp.array(ipg_raw / norm_ipg_raw) if norm_ipg_raw > 1e-9 else jnp.array(ipg_raw)
|
|
139
|
+
|
|
140
|
+
original_state_B = jnp.array(vettori[target_idx])
|
|
141
|
+
threshold = GLOBAL_CONSTANTS['NON_STATIC_THRESHOLD_A']
|
|
142
|
+
original_magnitude = float(_trigger_magnitude(original_state_B, state_A, ipg_vector))
|
|
143
|
+
original_trigger_active = original_magnitude > threshold
|
|
144
|
+
|
|
145
|
+
sign = 1.0 if direction == "flip_to_dynamic" else -1.0
|
|
146
|
+
if step_size is None:
|
|
147
|
+
step_size = 0.02
|
|
148
|
+
|
|
149
|
+
# Track the best iterate seen (most favorable to `direction`), not
|
|
150
|
+
# just the last one -- with a fixed small step size the trajectory
|
|
151
|
+
# can overshoot past the optimum and drift back worse on later
|
|
152
|
+
# steps, especially once it's pinned against the epsilon-ball
|
|
153
|
+
# boundary; keeping the best-so-far makes the result robust to
|
|
154
|
+
# exactly how many steps were requested.
|
|
155
|
+
state_B = original_state_B
|
|
156
|
+
best_state_B = original_state_B
|
|
157
|
+
best_magnitude = original_magnitude
|
|
158
|
+
for _ in range(n_steps):
|
|
159
|
+
grad = _trigger_magnitude_grad(state_B, state_A, ipg_vector)
|
|
160
|
+
grad_norm = jnp.linalg.norm(grad)
|
|
161
|
+
step = jnp.where(grad_norm > 1e-12, grad / grad_norm, jnp.zeros_like(grad))
|
|
162
|
+
state_B = state_B + sign * step_size * step
|
|
163
|
+
# Project back into the epsilon L2-ball around the original vector.
|
|
164
|
+
delta = state_B - original_state_B
|
|
165
|
+
delta_norm = jnp.linalg.norm(delta)
|
|
166
|
+
state_B = jnp.where(delta_norm > epsilon, original_state_B + delta / delta_norm * epsilon, state_B)
|
|
167
|
+
|
|
168
|
+
current_magnitude = float(_trigger_magnitude(state_B, state_A, ipg_vector))
|
|
169
|
+
is_better = (current_magnitude > best_magnitude) if direction == "flip_to_dynamic" else (current_magnitude < best_magnitude)
|
|
170
|
+
if is_better:
|
|
171
|
+
best_magnitude = current_magnitude
|
|
172
|
+
best_state_B = state_B
|
|
173
|
+
|
|
174
|
+
state_B = best_state_B
|
|
175
|
+
final_magnitude = float(_trigger_magnitude(state_B, state_A, ipg_vector))
|
|
176
|
+
final_trigger_active = final_magnitude > threshold
|
|
177
|
+
perturbation_norm = float(jnp.linalg.norm(state_B - original_state_B))
|
|
178
|
+
|
|
179
|
+
perturbed_vettori = vettori.copy()
|
|
180
|
+
perturbed_vettori[target_idx] = np.asarray(state_B)
|
|
181
|
+
|
|
182
|
+
flipped = final_trigger_active != original_trigger_active
|
|
183
|
+
success = flipped and (
|
|
184
|
+
(direction == "flip_to_dynamic" and final_trigger_active)
|
|
185
|
+
or (direction == "flip_to_static" and not final_trigger_active)
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
return {
|
|
189
|
+
"perturbed_vettori": perturbed_vettori,
|
|
190
|
+
"success": bool(success),
|
|
191
|
+
"original_trigger_active": bool(original_trigger_active),
|
|
192
|
+
"final_trigger_active": bool(final_trigger_active),
|
|
193
|
+
"original_magnitude": original_magnitude,
|
|
194
|
+
"final_magnitude": final_magnitude,
|
|
195
|
+
"perturbation_norm": perturbation_norm,
|
|
196
|
+
}
|
ia_utils/rag.py
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Hybrid two-stage retrieval (sparse TF-IDF UNION dense bi-encoder, pretrained
|
|
3
|
+
cross-encoder rerank) for grounding an agent's output in a local document
|
|
4
|
+
collection, instead of trusting an unverified citation.
|
|
5
|
+
|
|
6
|
+
Promoted from quantumrag (Desktop/Fullwork/quantumrag), a local RAG tool used
|
|
7
|
+
through Dense-Evolution-Discovery to check physics/chemistry claims against
|
|
8
|
+
real, independently-verified arXiv papers before citing them -- validated
|
|
9
|
+
there across ~30 topic collections (~6300 chunks) before this promotion. Only
|
|
10
|
+
the retrieval mechanism moves here, not quantumrag's own paper corpus or
|
|
11
|
+
built indexes -- those stay local, out of the package.
|
|
12
|
+
|
|
13
|
+
Stage 1 pools candidates from TF-IDF cosine (exact-term overlap, cheap, whole
|
|
14
|
+
collection) UNION a dense bi-encoder (catches paraphrases/synonyms TF-IDF
|
|
15
|
+
misses on its own -- Karpukhin et al. 2020, "Dense Passage Retrieval for
|
|
16
|
+
Open-Domain Question Answering", arXiv:2004.04906). Stage 2 reranks that pool
|
|
17
|
+
with a pretrained cross-encoder (Nogueira & Cho 2019, "Passage Re-ranking
|
|
18
|
+
with BERT", arXiv:1901.04085) -- a slower model that reads query+chunk
|
|
19
|
+
together instead of comparing two separate vectors, applied only to the
|
|
20
|
+
(small) pooled candidates, never the whole collection.
|
|
21
|
+
|
|
22
|
+
Needs the optional `rag` extra (scikit-learn + sentence-transformers) --
|
|
23
|
+
same pattern as native_hf.libcint_bridge (pyscf) and qmmm.region (rdkit):
|
|
24
|
+
the whole module is off-limits without it, since even the TF-IDF-only path
|
|
25
|
+
(build_index(compute_embeddings=False), search(rerank=False)) is built on
|
|
26
|
+
scikit-learn's TfidfVectorizer/cosine_similarity. chunk_text (pure Python,
|
|
27
|
+
no vectorizer/model involved) is the one function usable with no extra at
|
|
28
|
+
all -- it doesn't import this module's sklearn/sentence-transformers names.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import json
|
|
34
|
+
import pickle
|
|
35
|
+
import re
|
|
36
|
+
from dataclasses import dataclass
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
from typing import Optional
|
|
39
|
+
|
|
40
|
+
import numpy as np
|
|
41
|
+
|
|
42
|
+
CHUNK_SIZE_CHARS = 1200
|
|
43
|
+
CHUNK_OVERLAP_CHARS = 200
|
|
44
|
+
DEFAULT_EMBEDDING_MODEL = "sentence-transformers/all-MiniLM-L6-v2"
|
|
45
|
+
DEFAULT_CROSS_ENCODER_MODEL = "cross-encoder/ms-marco-MiniLM-L-6-v2"
|
|
46
|
+
DEFAULT_POOL = 20
|
|
47
|
+
|
|
48
|
+
# Keyed by model name so build_index/search calls with the same default
|
|
49
|
+
# model (the normal case) only ever load it once per process, regardless of
|
|
50
|
+
# how many collections/queries are processed.
|
|
51
|
+
_embedder_cache: dict = {}
|
|
52
|
+
_reranker_cache: dict = {}
|
|
53
|
+
|
|
54
|
+
_MISSING_RAG_EXTRA = (
|
|
55
|
+
"needs the optional 'rag' extra (pip install dense-evolution[rag]) -- "
|
|
56
|
+
"failed to import: {}"
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _get_embedder(model_name: str):
|
|
61
|
+
if model_name not in _embedder_cache:
|
|
62
|
+
try:
|
|
63
|
+
from sentence_transformers import SentenceTransformer
|
|
64
|
+
except ImportError as _import_error: # pragma: no cover -- only reachable without sentence-transformers installed, which CI here always has
|
|
65
|
+
raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
|
|
66
|
+
_embedder_cache[model_name] = SentenceTransformer(model_name)
|
|
67
|
+
return _embedder_cache[model_name]
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _get_reranker(model_name: str):
|
|
71
|
+
if model_name not in _reranker_cache:
|
|
72
|
+
try:
|
|
73
|
+
from sentence_transformers import CrossEncoder
|
|
74
|
+
except ImportError as _import_error: # pragma: no cover -- only reachable without sentence-transformers installed, which CI here always has
|
|
75
|
+
raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
|
|
76
|
+
_reranker_cache[model_name] = CrossEncoder(model_name)
|
|
77
|
+
return _reranker_cache[model_name]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def chunk_text(
|
|
81
|
+
text: str,
|
|
82
|
+
source: str,
|
|
83
|
+
chunk_size: int = CHUNK_SIZE_CHARS,
|
|
84
|
+
overlap: int = CHUNK_OVERLAP_CHARS,
|
|
85
|
+
) -> list:
|
|
86
|
+
chunks = []
|
|
87
|
+
start = 0
|
|
88
|
+
n = len(text)
|
|
89
|
+
while start < n:
|
|
90
|
+
end = min(start + chunk_size, n)
|
|
91
|
+
piece = text[start:end].strip()
|
|
92
|
+
if piece:
|
|
93
|
+
chunks.append({"source": source, "text": piece})
|
|
94
|
+
if end == n:
|
|
95
|
+
break
|
|
96
|
+
start = end - overlap
|
|
97
|
+
return chunks
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@dataclass
|
|
101
|
+
class RagIndex:
|
|
102
|
+
chunks: list
|
|
103
|
+
vectorizer: object # sklearn.feature_extraction.text.TfidfVectorizer, fitted
|
|
104
|
+
matrix: object # scipy.sparse matrix, vectorizer's transform of every chunk's text
|
|
105
|
+
embeddings: Optional[np.ndarray] = None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def build_index(
|
|
109
|
+
documents,
|
|
110
|
+
embedding_model: str = DEFAULT_EMBEDDING_MODEL,
|
|
111
|
+
compute_embeddings: bool = True,
|
|
112
|
+
chunk_size: int = CHUNK_SIZE_CHARS,
|
|
113
|
+
overlap: int = CHUNK_OVERLAP_CHARS,
|
|
114
|
+
) -> RagIndex:
|
|
115
|
+
"""
|
|
116
|
+
documents: iterable of (text, source) pairs, one per already-extracted
|
|
117
|
+
document (PDF/markdown/plain text parsing happens before this call --
|
|
118
|
+
this module only chunks and indexes text it's handed). compute_embeddings
|
|
119
|
+
can be set False to skip the (slower) bi-encoder pass and build a
|
|
120
|
+
TF-IDF-only index -- still needs the 'rag' extra for scikit-learn, just
|
|
121
|
+
not sentence-transformers' model download; search() then falls back to
|
|
122
|
+
plain cosine regardless of `rerank`.
|
|
123
|
+
"""
|
|
124
|
+
try:
|
|
125
|
+
from sklearn.feature_extraction.text import TfidfVectorizer
|
|
126
|
+
except ImportError as _import_error: # pragma: no cover -- only reachable without scikit-learn installed, which CI here always has
|
|
127
|
+
raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
|
|
128
|
+
|
|
129
|
+
all_chunks = []
|
|
130
|
+
for text, source in documents:
|
|
131
|
+
all_chunks.extend(chunk_text(text, source, chunk_size, overlap))
|
|
132
|
+
if not all_chunks:
|
|
133
|
+
raise ValueError("no chunks produced -- 'documents' was empty or every text was blank")
|
|
134
|
+
|
|
135
|
+
texts = [c["text"] for c in all_chunks]
|
|
136
|
+
vectorizer = TfidfVectorizer(stop_words="english", max_features=20000, ngram_range=(1, 2))
|
|
137
|
+
matrix = vectorizer.fit_transform(texts)
|
|
138
|
+
|
|
139
|
+
embeddings = None
|
|
140
|
+
if compute_embeddings:
|
|
141
|
+
embeddings = _get_embedder(embedding_model).encode(
|
|
142
|
+
texts, normalize_embeddings=True, show_progress_bar=False
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
return RagIndex(chunks=all_chunks, vectorizer=vectorizer, matrix=matrix, embeddings=embeddings)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def save_index(index: RagIndex, index_dir) -> None:
|
|
149
|
+
index_dir = Path(index_dir)
|
|
150
|
+
index_dir.mkdir(parents=True, exist_ok=True)
|
|
151
|
+
with open(index_dir / "chunks.json", "w", encoding="utf-8") as f:
|
|
152
|
+
json.dump(index.chunks, f, ensure_ascii=False, indent=2)
|
|
153
|
+
with open(index_dir / "vectorizer.pkl", "wb") as f:
|
|
154
|
+
pickle.dump(index.vectorizer, f)
|
|
155
|
+
with open(index_dir / "matrix.pkl", "wb") as f:
|
|
156
|
+
pickle.dump(index.matrix, f)
|
|
157
|
+
if index.embeddings is not None:
|
|
158
|
+
np.save(index_dir / "embeddings.npy", index.embeddings)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def load_index(index_dir) -> RagIndex:
|
|
162
|
+
index_dir = Path(index_dir)
|
|
163
|
+
with open(index_dir / "chunks.json", encoding="utf-8") as f:
|
|
164
|
+
chunks = json.load(f)
|
|
165
|
+
with open(index_dir / "vectorizer.pkl", "rb") as f:
|
|
166
|
+
vectorizer = pickle.load(f)
|
|
167
|
+
with open(index_dir / "matrix.pkl", "rb") as f:
|
|
168
|
+
matrix = pickle.load(f)
|
|
169
|
+
emb_path = index_dir / "embeddings.npy"
|
|
170
|
+
embeddings = np.load(emb_path) if emb_path.exists() else None
|
|
171
|
+
return RagIndex(chunks=chunks, vectorizer=vectorizer, matrix=matrix, embeddings=embeddings)
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def search(
|
|
175
|
+
query: str,
|
|
176
|
+
index: RagIndex,
|
|
177
|
+
top: int = 3,
|
|
178
|
+
rerank: bool = True,
|
|
179
|
+
pool: int = DEFAULT_POOL,
|
|
180
|
+
embedding_model: str = DEFAULT_EMBEDDING_MODEL,
|
|
181
|
+
cross_encoder_model: str = DEFAULT_CROSS_ENCODER_MODEL,
|
|
182
|
+
) -> list:
|
|
183
|
+
"""
|
|
184
|
+
Returns the top `top` chunks as a list of dicts (rank, source, text,
|
|
185
|
+
score, score_type, cosine), highest-scoring first.
|
|
186
|
+
|
|
187
|
+
rerank=True (default): stage 1 pools `pool` candidates from TF-IDF
|
|
188
|
+
cosine UNION dense-embedding similarity (if the index has embeddings --
|
|
189
|
+
plain TF-IDF pool otherwise), stage 2 reorders that pool with the
|
|
190
|
+
cross-encoder and returns its top `top` ("score_type": "rerank").
|
|
191
|
+
rerank=False: plain TF-IDF cosine ranking, no sentence-transformers model
|
|
192
|
+
load needed (still needs the 'rag' extra's scikit-learn for cosine_similarity
|
|
193
|
+
itself) -- useful as a baseline, or when only embeddings are unavailable
|
|
194
|
+
for this index but a scored ranking is still wanted.
|
|
195
|
+
"""
|
|
196
|
+
try:
|
|
197
|
+
from sklearn.metrics.pairwise import cosine_similarity
|
|
198
|
+
except ImportError as _import_error: # pragma: no cover -- only reachable without scikit-learn installed, which CI here always has
|
|
199
|
+
raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
|
|
200
|
+
|
|
201
|
+
query_vec = index.vectorizer.transform([query])
|
|
202
|
+
cosine_scores = cosine_similarity(query_vec, index.matrix)[0]
|
|
203
|
+
tfidf_order = cosine_scores.argsort()[::-1]
|
|
204
|
+
|
|
205
|
+
if rerank and index.embeddings is not None:
|
|
206
|
+
query_emb = _get_embedder(embedding_model).encode([query], normalize_embeddings=True)[0]
|
|
207
|
+
dense_scores = index.embeddings @ query_emb
|
|
208
|
+
dense_pool = dense_scores.argsort()[::-1][:pool]
|
|
209
|
+
candidate_idx = sorted(set(tfidf_order[:pool].tolist()) | set(dense_pool.tolist()))
|
|
210
|
+
elif rerank:
|
|
211
|
+
candidate_idx = tfidf_order[:pool].tolist()
|
|
212
|
+
else:
|
|
213
|
+
candidate_idx = []
|
|
214
|
+
|
|
215
|
+
if rerank and candidate_idx:
|
|
216
|
+
pairs = [(query, index.chunks[i]["text"][:1024]) for i in candidate_idx]
|
|
217
|
+
rerank_scores = _get_reranker(cross_encoder_model).predict(pairs)
|
|
218
|
+
order = rerank_scores.argsort()[::-1][:top]
|
|
219
|
+
top_idx = [candidate_idx[j] for j in order]
|
|
220
|
+
scores = {candidate_idx[j]: float(rerank_scores[j]) for j in order}
|
|
221
|
+
score_type = "rerank"
|
|
222
|
+
else:
|
|
223
|
+
top_idx = tfidf_order[:top].tolist()
|
|
224
|
+
scores = {i: float(cosine_scores[i]) for i in top_idx}
|
|
225
|
+
score_type = "cosine"
|
|
226
|
+
|
|
227
|
+
return [
|
|
228
|
+
{
|
|
229
|
+
"rank": rank,
|
|
230
|
+
"source": index.chunks[i]["source"],
|
|
231
|
+
"text": index.chunks[i]["text"],
|
|
232
|
+
"score": scores[i],
|
|
233
|
+
"score_type": score_type,
|
|
234
|
+
"cosine": float(cosine_scores[i]),
|
|
235
|
+
}
|
|
236
|
+
for rank, i in enumerate(top_idx, 1)
|
|
237
|
+
]
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def search_exact(pattern: str, index: RagIndex, regex: bool = False, max_hits: int = 10, context: int = 300) -> list:
|
|
241
|
+
"""Substring/regex search over an index's raw chunk text -- no
|
|
242
|
+
embedding, no reranker, no model download, so this needs neither the
|
|
243
|
+
'rag' extra nor a built index's embeddings (works even on an index
|
|
244
|
+
built with compute_embeddings=False).
|
|
245
|
+
|
|
246
|
+
Semantic search (search(), above) ranks by topical similarity, which
|
|
247
|
+
can bury a short, specific, load-bearing phrase (an exact clause, a
|
|
248
|
+
fixed parameter value, a named condition) under chunks that are
|
|
249
|
+
merely more topically central. Use this instead when you already know
|
|
250
|
+
roughly what wording you're looking for and semantic search isn't
|
|
251
|
+
surfacing it high enough.
|
|
252
|
+
|
|
253
|
+
regex=False (default) treats `pattern` as a literal, case-insensitive
|
|
254
|
+
substring (re.escape'd internally); regex=True treats it as a real
|
|
255
|
+
regular expression, case-sensitive, exactly as written. Stops after
|
|
256
|
+
`max_hits` matches (a `truncated` field on the last result marks this,
|
|
257
|
+
rather than silently returning a partial, unmarked list); `context`
|
|
258
|
+
characters of surrounding text are kept on each side of the match.
|
|
259
|
+
|
|
260
|
+
Returns a list of {source, chunk_index, match, snippet, start, end}
|
|
261
|
+
dicts, at most one per chunk -- the chunk's first match only, same as
|
|
262
|
+
the original quantumrag CLI this was ported from, not an exhaustive
|
|
263
|
+
enumeration of every occurrence within a chunk. Empty list (not an
|
|
264
|
+
error) if nothing matches.
|
|
265
|
+
"""
|
|
266
|
+
flags = 0 if regex else re.IGNORECASE
|
|
267
|
+
compiled = re.compile(pattern if regex else re.escape(pattern), flags)
|
|
268
|
+
hits = []
|
|
269
|
+
for i, chunk in enumerate(index.chunks):
|
|
270
|
+
text = chunk["text"]
|
|
271
|
+
m = compiled.search(text)
|
|
272
|
+
if not m:
|
|
273
|
+
continue
|
|
274
|
+
lo = max(0, m.start() - context)
|
|
275
|
+
hi = min(len(text), m.end() + context)
|
|
276
|
+
snippet = text[lo:hi]
|
|
277
|
+
hits.append({
|
|
278
|
+
"source": chunk["source"],
|
|
279
|
+
"chunk_index": i,
|
|
280
|
+
"match": m.group(0),
|
|
281
|
+
"snippet": ("..." if lo > 0 else "") + snippet + ("..." if hi < len(text) else ""),
|
|
282
|
+
"start": m.start(),
|
|
283
|
+
"end": m.end(),
|
|
284
|
+
})
|
|
285
|
+
if len(hits) >= max_hits:
|
|
286
|
+
hits[-1]["truncated"] = True
|
|
287
|
+
break
|
|
288
|
+
return hits
|