dense-evolution 8.3.0__py3-none-win_amd64.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (165) hide show
  1. dashboard_core/__init__.py +115 -0
  2. dashboard_core/_gate_tables.py +30 -0
  3. dashboard_core/band_structure.py +71 -0
  4. dashboard_core/circuit_builder_component.py +232 -0
  5. dashboard_core/circuit_diagram.py +216 -0
  6. dashboard_core/crypto_protocols.py +77 -0
  7. dashboard_core/engine.py +326 -0
  8. dashboard_core/graphical_builder.py +114 -0
  9. dashboard_core/hamiltonians.py +593 -0
  10. dashboard_core/mass_decomposition_tool.py +47 -0
  11. dashboard_core/mitigation.py +343 -0
  12. dashboard_core/native_hf_diagnostics.py +62 -0
  13. dashboard_core/noise_tools.py +125 -0
  14. dashboard_core/qasm_library.py +233 -0
  15. dashboard_core/qmmm.py +16 -0
  16. dashboard_core/rag_tool.py +45 -0
  17. dashboard_core/state_visuals.py +288 -0
  18. dashboard_core/system_limits.py +60 -0
  19. dashboard_core/vector_healing.py +102 -0
  20. dashboard_core/visuals.py +158 -0
  21. dashboard_core/vqe.py +533 -0
  22. dashboard_core/wormhole.py +580 -0
  23. dense_evolution/__init__.py +114 -0
  24. dense_evolution/autodiff.py +10 -0
  25. dense_evolution/backends/__init__.py +5 -0
  26. dense_evolution/backends/chunk/__init__.py +37 -0
  27. dense_evolution/backends/chunk/_engine_imports.py +57 -0
  28. dense_evolution/backends/chunk/circuit_chunker.py +55 -0
  29. dense_evolution/backends/chunk/core.py +432 -0
  30. dense_evolution/backends/chunk/disk_overflow.py +232 -0
  31. dense_evolution/backends/chunk/geometry.py +95 -0
  32. dense_evolution/backends/chunk/guard.py +190 -0
  33. dense_evolution/backends/chunk/kernels.py +531 -0
  34. dense_evolution/backends/mps.py +1569 -0
  35. dense_evolution/backends/statevector.py +616 -0
  36. dense_evolution/chunk.py +25 -0
  37. dense_evolution/circuits/__init__.py +20 -0
  38. dense_evolution/circuits/compiler.py +488 -0
  39. dense_evolution/circuits/diagram.py +94 -0
  40. dense_evolution/circuits/gates.py +91 -0
  41. dense_evolution/circuits/parser.py +632 -0
  42. dense_evolution/circuits/qft.py +66 -0
  43. dense_evolution/circuits/random_circuit.py +85 -0
  44. dense_evolution/circuits/registry.py +74 -0
  45. dense_evolution/circuits/topology.py +79 -0
  46. dense_evolution/circuits/trotter.py +265 -0
  47. dense_evolution/circuits/uccsd.py +275 -0
  48. dense_evolution/cli.py +199 -0
  49. dense_evolution/compiler.py +9 -0
  50. dense_evolution/config.py +49 -0
  51. dense_evolution/drawing.py +10 -0
  52. dense_evolution/entropy.py +9 -0
  53. dense_evolution/fermions.py +9 -0
  54. dense_evolution/gates.py +9 -0
  55. dense_evolution/harrison_tb.py +16 -0
  56. dense_evolution/healing.py +18 -0
  57. dense_evolution/interop/__init__.py +18 -0
  58. dense_evolution/interop/qiskit_pennylane.py +406 -0
  59. dense_evolution/measurement.py +10 -0
  60. dense_evolution/mitigation/__init__.py +54 -0
  61. dense_evolution/mitigation/healing.py +215 -0
  62. dense_evolution/mitigation/kl_divergence.py +93 -0
  63. dense_evolution/mitigation/magic_entropy.py +163 -0
  64. dense_evolution/mitigation/magic_entropy_shadows.py +262 -0
  65. dense_evolution/mitigation/renyi.py +168 -0
  66. dense_evolution/mitigation/stabilizer_renyi_entropy.py +103 -0
  67. dense_evolution/mitigation/zne.py +990 -0
  68. dense_evolution/mps.py +9 -0
  69. dense_evolution/native_hf/__init__.py +26 -0
  70. dense_evolution/native_hf/_libcint/LICENSE-libcint +10 -0
  71. dense_evolution/native_hf/_libcint/libdecint.dll +0 -0
  72. dense_evolution/native_hf/assembly.py +304 -0
  73. dense_evolution/native_hf/basis.py +117 -0
  74. dense_evolution/native_hf/boys.py +35 -0
  75. dense_evolution/native_hf/bridge.py +112 -0
  76. dense_evolution/native_hf/cartesian.py +64 -0
  77. dense_evolution/native_hf/coulomb.py +196 -0
  78. dense_evolution/native_hf/differentiable.py +53 -0
  79. dense_evolution/native_hf/gaussians.py +79 -0
  80. dense_evolution/native_hf/kinetic.py +52 -0
  81. dense_evolution/native_hf/libcint_bridge.py +167 -0
  82. dense_evolution/native_hf/overlap.py +91 -0
  83. dense_evolution/native_hf/scf.py +404 -0
  84. dense_evolution/noise/__init__.py +79 -0
  85. dense_evolution/noise/coherent_attack.py +264 -0
  86. dense_evolution/noise/cosmic_ray.py +61 -0
  87. dense_evolution/noise/density_matrix_channels.py +78 -0
  88. dense_evolution/noise/differentiable.py +66 -0
  89. dense_evolution/noise/kraus/__init__.py +6 -0
  90. dense_evolution/noise/kraus/amplitude_damping.py +47 -0
  91. dense_evolution/noise/kraus/bitflip.py +22 -0
  92. dense_evolution/noise/kraus/combined.py +16 -0
  93. dense_evolution/noise/kraus/depolarizing.py +47 -0
  94. dense_evolution/noise/kraus/ideal.py +10 -0
  95. dense_evolution/noise/kraus/phaseflip.py +21 -0
  96. dense_evolution/noise/kraus_channels.py +285 -0
  97. dense_evolution/noise/oscillating.py +32 -0
  98. dense_evolution/noise/pink.py +80 -0
  99. dense_evolution/observables.py +11 -0
  100. dense_evolution/parser.py +9 -0
  101. dense_evolution/physics/__init__.py +27 -0
  102. dense_evolution/physics/entropy.py +161 -0
  103. dense_evolution/physics/fermions.py +322 -0
  104. dense_evolution/physics/observables.py +523 -0
  105. dense_evolution/physics/qec.py +1113 -0
  106. dense_evolution/physics/spectral.py +143 -0
  107. dense_evolution/physics/states.py +43 -0
  108. dense_evolution/protocols/__init__.py +27 -0
  109. dense_evolution/protocols/bb84.py +133 -0
  110. dense_evolution/protocols/di_qkd_ghz.py +199 -0
  111. dense_evolution/protocols/dicka_protocol2.py +124 -0
  112. dense_evolution/qec.py +20 -0
  113. dense_evolution/qft.py +9 -0
  114. dense_evolution/qmmm/__init__.py +13 -0
  115. dense_evolution/qmmm/ase_bridge.py +97 -0
  116. dense_evolution/qmmm/forces.py +388 -0
  117. dense_evolution/qmmm/propagation.py +80 -0
  118. dense_evolution/qmmm/region.py +137 -0
  119. dense_evolution/random_circuit.py +15 -0
  120. dense_evolution/registry.py +9 -0
  121. dense_evolution/simulator.py +10 -0
  122. dense_evolution/solvers/__init__.py +19 -0
  123. dense_evolution/solvers/autodiff.py +169 -0
  124. dense_evolution/solvers/harrison_tb.py +189 -0
  125. dense_evolution/solvers/vhd_tb.py +187 -0
  126. dense_evolution/states.py +9 -0
  127. dense_evolution/topology.py +9 -0
  128. dense_evolution/trotter.py +9 -0
  129. dense_evolution/utils/__init__.py +13 -0
  130. dense_evolution/utils/drawing.py +101 -0
  131. dense_evolution/utils/mass_decomposition.py +246 -0
  132. dense_evolution/utils/measurement.py +94 -0
  133. dense_evolution/vhd_tb.py +16 -0
  134. dense_evolution-8.3.0.dist-info/METADATA +366 -0
  135. dense_evolution-8.3.0.dist-info/RECORD +165 -0
  136. dense_evolution-8.3.0.dist-info/WHEEL +5 -0
  137. dense_evolution-8.3.0.dist-info/entry_points.txt +2 -0
  138. dense_evolution-8.3.0.dist-info/licenses/license.md +58 -0
  139. dense_evolution-8.3.0.dist-info/top_level.txt +5 -0
  140. ia_utils/__init__.py +0 -0
  141. ia_utils/adversarial_vector_attack.py +196 -0
  142. ia_utils/rag.py +288 -0
  143. ia_utils/vector_healing.py +399 -0
  144. local_site/__init__.py +0 -0
  145. local_site/app/__init__.py +0 -0
  146. local_site/app/server.py +1009 -0
  147. mcp_server/__init__.py +0 -0
  148. mcp_server/client.py +324 -0
  149. mcp_server/config.py +32 -0
  150. mcp_server/models.py +347 -0
  151. mcp_server/molecules.py +71 -0
  152. mcp_server/server.py +119 -0
  153. mcp_server/tools/__init__.py +0 -0
  154. mcp_server/tools/chemistry_tools.py +225 -0
  155. mcp_server/tools/circuit_tools.py +83 -0
  156. mcp_server/tools/crypto_tools.py +66 -0
  157. mcp_server/tools/mitigation_tools.py +81 -0
  158. mcp_server/tools/noise_tools.py +60 -0
  159. mcp_server/tools/retrieval_tools.py +44 -0
  160. mcp_server/tools/system_tools.py +149 -0
  161. mcp_server/tools/wormhole_tools.py +142 -0
  162. mcp_server/utils/__init__.py +0 -0
  163. mcp_server/utils/cache.py +55 -0
  164. mcp_server/utils/images.py +67 -0
  165. mcp_server/utils/truncation.py +38 -0
@@ -0,0 +1,196 @@
1
+ """
2
+ Gradient-based adversarial stress-testing for enhanced_dense_healing_hybrid's
3
+ Phi-Trigger decision -- a targeted, crafted perturbation instead of the
4
+ random NaN/Inf corruption this healing pipeline is normally tested
5
+ against, adapted from IGME's chained-differentiable-attack idea
6
+ (arXiv:2607.27465, "IGME: Efficient Chained Method Ensemble for
7
+ Transferable Semantic Segmentation Attacks") applied to vector sequences
8
+ instead of image segmentation.
9
+
10
+ evaluate_phi_trigger (dense_evolution.healing) makes its keep-vs-median-
11
+ replace decision by thresholding |v_dinamic| against NON_STATIC_THRESHOLD_A
12
+ -- a hard step, not differentiable at the boundary. But v_dinamic itself
13
+ (via calculate_phi_ab -> calculate_vettore_dinamico) is built entirely
14
+ from norms, dot products, log, and clip -- all JAX-differentiable. This
15
+ crafts a minimal perturbation to a single vector in the sequence, via
16
+ gradient ascent/descent on |v_dinamic| (projected back into an epsilon
17
+ L2-ball each step, the standard PGD pattern), that flips the trigger's
18
+ decision at that point:
19
+
20
+ - "flip_to_dynamic": push an originally-static point (would be replaced
21
+ by the local median) across the threshold so the trigger keeps it
22
+ as-is instead -- the more security-relevant direction, since it
23
+ represents a worst-case corruption crafted to *evade* the healer by
24
+ looking like genuine motion, rather than obvious noise.
25
+ - "flip_to_static": push an originally-dynamic point (would be kept)
26
+ across the threshold so the trigger discards it as noise instead --
27
+ the failure mode of genuine signal getting wrongly suppressed.
28
+
29
+ This does not attack the median-filter fallback itself, only the
30
+ Phi-Trigger's keep-vs-replace decision -- see craft_adversarial_healing_
31
+ perturbation's own docstring for what "success" means precisely.
32
+ """
33
+
34
+ from typing import Optional
35
+
36
+ import numpy as np
37
+ import jax
38
+ import jax.numpy as jnp
39
+
40
+ from dense_evolution.mitigation.healing import calculate_phi_ab, calculate_vettore_dinamico, GLOBAL_CONSTANTS
41
+
42
+ __all__ = ['craft_adversarial_healing_perturbation']
43
+
44
+
45
+ def _trigger_magnitude(state_B, state_A, ipg_vector):
46
+ """|v_dinamic| -- the continuous quantity evaluate_phi_trigger
47
+ thresholds. Differentiable w.r.t. state_B (the vector being attacked);
48
+ state_A/ipg_vector are treated as fixed context, not attacked."""
49
+ phi_ab = calculate_phi_ab(state_A, state_B, ipg_vector)
50
+ E_A = jnp.linalg.norm(state_A)
51
+ E_B = jnp.linalg.norm(state_B)
52
+ v_dinamic = calculate_vettore_dinamico(E_A, E_B, phi_ab)
53
+ return jnp.abs(v_dinamic)
54
+
55
+
56
+ _trigger_magnitude_grad = jax.grad(_trigger_magnitude, argnums=0)
57
+
58
+
59
+ def craft_adversarial_healing_perturbation(
60
+ vettori: np.ndarray,
61
+ target_idx: int,
62
+ radius_baseline: Optional[int] = None,
63
+ epsilon: float = 0.1,
64
+ n_steps: int = 50,
65
+ step_size: Optional[float] = None,
66
+ direction: str = "flip_to_dynamic",
67
+ ) -> dict:
68
+ """Crafts a minimal adversarial perturbation to vettori[target_idx],
69
+ within an L2 epsilon-ball, that flips enhanced_dense_healing_hybrid's
70
+ Phi-Trigger decision at that index -- a targeted stress test, not
71
+ random noise.
72
+
73
+ Reproduces enhanced_dense_healing_hybrid's own per-step computation
74
+ at target_idx exactly (same baseline_mean window, same adaptive
75
+ radius default, same inter-point-gradient vector) so the crafted
76
+ perturbation is faithful to what the real healing pipeline would
77
+ actually see, not a simplified stand-in.
78
+
79
+ BUG FIX: step_size used to default to 2*epsilon/n_steps -- tying the
80
+ per-step move size to the epsilon budget. Verified directly this
81
+ makes LARGER epsilon give WORSE (higher final |v_dinamic|) results,
82
+ not better: a bigger budget means a bigger step, which overshoots
83
+ and oscillates around the minimum instead of converging to it,
84
+ which is the opposite of what a bigger budget should ever do for a
85
+ correct optimizer. step_size is now independent of epsilon (a small
86
+ fixed default, tuned to v_dinamic's typical local scale) -- epsilon
87
+ only bounds where the iterate is allowed to end up (via projection
88
+ after each step), not how big each step is.
89
+
90
+ Args:
91
+ vettori: array-like, shape (n_steps, dim), the (unperturbed)
92
+ vector sequence. Not modified in place.
93
+ target_idx: index to attack; must be >= 2 (the healing loop's
94
+ own starting point) and < len(vettori).
95
+ radius_baseline: same meaning as enhanced_dense_healing_hybrid's
96
+ own parameter; None uses the same adaptive default.
97
+ epsilon: L2-norm budget for the perturbation (the attack is
98
+ projected back into this ball after every gradient step).
99
+ n_steps: number of PGD-style gradient steps.
100
+ step_size: per-step move size along the normalized gradient;
101
+ None uses a small fixed default (0.02) independent of
102
+ epsilon -- see the bug-fix note above for why.
103
+ direction: "flip_to_dynamic" (evade -- make static-looking
104
+ input pass through unhealed) or "flip_to_static" (suppress
105
+ -- make dynamic-looking input get median-replaced instead).
106
+
107
+ Returns:
108
+ dict with:
109
+ perturbed_vettori: copy of vettori with vettori[target_idx]
110
+ replaced by the crafted perturbation.
111
+ success: bool, True only if the trigger decision actually
112
+ flipped in the requested direction (a small epsilon or
113
+ too few steps can fail to cross the threshold).
114
+ original_trigger_active / final_trigger_active: bool, the
115
+ Phi-Trigger's decision (True = dynamic/kept) before and
116
+ after the perturbation.
117
+ original_magnitude / final_magnitude: float, |v_dinamic|
118
+ before and after.
119
+ perturbation_norm: float, actual L2 norm of the applied
120
+ perturbation (<= epsilon).
121
+ """
122
+ vettori = np.asarray(vettori, dtype=np.float64)
123
+ n, hidden_dim = vettori.shape
124
+ if target_idx < 2 or target_idx >= n:
125
+ raise ValueError(f"target_idx must be in [2, {n - 1}], got {target_idx} (n={n})")
126
+ if direction not in ("flip_to_dynamic", "flip_to_static"):
127
+ raise ValueError(f"direction must be 'flip_to_dynamic' or 'flip_to_static', got {direction!r}")
128
+ if epsilon <= 0:
129
+ raise ValueError(f"epsilon must be positive, got {epsilon}")
130
+
131
+ if radius_baseline is None:
132
+ radius_baseline = min(20, max(3, n // 3))
133
+ lo = max(0, target_idx - radius_baseline)
134
+
135
+ state_A = jnp.array(np.mean(vettori[lo:target_idx], axis=0))
136
+ ipg_raw = vettori[target_idx - 1] - vettori[target_idx - 2]
137
+ norm_ipg_raw = np.linalg.norm(ipg_raw)
138
+ ipg_vector = jnp.array(ipg_raw / norm_ipg_raw) if norm_ipg_raw > 1e-9 else jnp.array(ipg_raw)
139
+
140
+ original_state_B = jnp.array(vettori[target_idx])
141
+ threshold = GLOBAL_CONSTANTS['NON_STATIC_THRESHOLD_A']
142
+ original_magnitude = float(_trigger_magnitude(original_state_B, state_A, ipg_vector))
143
+ original_trigger_active = original_magnitude > threshold
144
+
145
+ sign = 1.0 if direction == "flip_to_dynamic" else -1.0
146
+ if step_size is None:
147
+ step_size = 0.02
148
+
149
+ # Track the best iterate seen (most favorable to `direction`), not
150
+ # just the last one -- with a fixed small step size the trajectory
151
+ # can overshoot past the optimum and drift back worse on later
152
+ # steps, especially once it's pinned against the epsilon-ball
153
+ # boundary; keeping the best-so-far makes the result robust to
154
+ # exactly how many steps were requested.
155
+ state_B = original_state_B
156
+ best_state_B = original_state_B
157
+ best_magnitude = original_magnitude
158
+ for _ in range(n_steps):
159
+ grad = _trigger_magnitude_grad(state_B, state_A, ipg_vector)
160
+ grad_norm = jnp.linalg.norm(grad)
161
+ step = jnp.where(grad_norm > 1e-12, grad / grad_norm, jnp.zeros_like(grad))
162
+ state_B = state_B + sign * step_size * step
163
+ # Project back into the epsilon L2-ball around the original vector.
164
+ delta = state_B - original_state_B
165
+ delta_norm = jnp.linalg.norm(delta)
166
+ state_B = jnp.where(delta_norm > epsilon, original_state_B + delta / delta_norm * epsilon, state_B)
167
+
168
+ current_magnitude = float(_trigger_magnitude(state_B, state_A, ipg_vector))
169
+ is_better = (current_magnitude > best_magnitude) if direction == "flip_to_dynamic" else (current_magnitude < best_magnitude)
170
+ if is_better:
171
+ best_magnitude = current_magnitude
172
+ best_state_B = state_B
173
+
174
+ state_B = best_state_B
175
+ final_magnitude = float(_trigger_magnitude(state_B, state_A, ipg_vector))
176
+ final_trigger_active = final_magnitude > threshold
177
+ perturbation_norm = float(jnp.linalg.norm(state_B - original_state_B))
178
+
179
+ perturbed_vettori = vettori.copy()
180
+ perturbed_vettori[target_idx] = np.asarray(state_B)
181
+
182
+ flipped = final_trigger_active != original_trigger_active
183
+ success = flipped and (
184
+ (direction == "flip_to_dynamic" and final_trigger_active)
185
+ or (direction == "flip_to_static" and not final_trigger_active)
186
+ )
187
+
188
+ return {
189
+ "perturbed_vettori": perturbed_vettori,
190
+ "success": bool(success),
191
+ "original_trigger_active": bool(original_trigger_active),
192
+ "final_trigger_active": bool(final_trigger_active),
193
+ "original_magnitude": original_magnitude,
194
+ "final_magnitude": final_magnitude,
195
+ "perturbation_norm": perturbation_norm,
196
+ }
ia_utils/rag.py ADDED
@@ -0,0 +1,288 @@
1
+ """
2
+ Hybrid two-stage retrieval (sparse TF-IDF UNION dense bi-encoder, pretrained
3
+ cross-encoder rerank) for grounding an agent's output in a local document
4
+ collection, instead of trusting an unverified citation.
5
+
6
+ Promoted from quantumrag (Desktop/Fullwork/quantumrag), a local RAG tool used
7
+ through Dense-Evolution-Discovery to check physics/chemistry claims against
8
+ real, independently-verified arXiv papers before citing them -- validated
9
+ there across ~30 topic collections (~6300 chunks) before this promotion. Only
10
+ the retrieval mechanism moves here, not quantumrag's own paper corpus or
11
+ built indexes -- those stay local, out of the package.
12
+
13
+ Stage 1 pools candidates from TF-IDF cosine (exact-term overlap, cheap, whole
14
+ collection) UNION a dense bi-encoder (catches paraphrases/synonyms TF-IDF
15
+ misses on its own -- Karpukhin et al. 2020, "Dense Passage Retrieval for
16
+ Open-Domain Question Answering", arXiv:2004.04906). Stage 2 reranks that pool
17
+ with a pretrained cross-encoder (Nogueira & Cho 2019, "Passage Re-ranking
18
+ with BERT", arXiv:1901.04085) -- a slower model that reads query+chunk
19
+ together instead of comparing two separate vectors, applied only to the
20
+ (small) pooled candidates, never the whole collection.
21
+
22
+ Needs the optional `rag` extra (scikit-learn + sentence-transformers) --
23
+ same pattern as native_hf.libcint_bridge (pyscf) and qmmm.region (rdkit):
24
+ the whole module is off-limits without it, since even the TF-IDF-only path
25
+ (build_index(compute_embeddings=False), search(rerank=False)) is built on
26
+ scikit-learn's TfidfVectorizer/cosine_similarity. chunk_text (pure Python,
27
+ no vectorizer/model involved) is the one function usable with no extra at
28
+ all -- it doesn't import this module's sklearn/sentence-transformers names.
29
+ """
30
+
31
+ from __future__ import annotations
32
+
33
+ import json
34
+ import pickle
35
+ import re
36
+ from dataclasses import dataclass
37
+ from pathlib import Path
38
+ from typing import Optional
39
+
40
+ import numpy as np
41
+
42
+ CHUNK_SIZE_CHARS = 1200
43
+ CHUNK_OVERLAP_CHARS = 200
44
+ DEFAULT_EMBEDDING_MODEL = "sentence-transformers/all-MiniLM-L6-v2"
45
+ DEFAULT_CROSS_ENCODER_MODEL = "cross-encoder/ms-marco-MiniLM-L-6-v2"
46
+ DEFAULT_POOL = 20
47
+
48
+ # Keyed by model name so build_index/search calls with the same default
49
+ # model (the normal case) only ever load it once per process, regardless of
50
+ # how many collections/queries are processed.
51
+ _embedder_cache: dict = {}
52
+ _reranker_cache: dict = {}
53
+
54
+ _MISSING_RAG_EXTRA = (
55
+ "needs the optional 'rag' extra (pip install dense-evolution[rag]) -- "
56
+ "failed to import: {}"
57
+ )
58
+
59
+
60
+ def _get_embedder(model_name: str):
61
+ if model_name not in _embedder_cache:
62
+ try:
63
+ from sentence_transformers import SentenceTransformer
64
+ except ImportError as _import_error: # pragma: no cover -- only reachable without sentence-transformers installed, which CI here always has
65
+ raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
66
+ _embedder_cache[model_name] = SentenceTransformer(model_name)
67
+ return _embedder_cache[model_name]
68
+
69
+
70
+ def _get_reranker(model_name: str):
71
+ if model_name not in _reranker_cache:
72
+ try:
73
+ from sentence_transformers import CrossEncoder
74
+ except ImportError as _import_error: # pragma: no cover -- only reachable without sentence-transformers installed, which CI here always has
75
+ raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
76
+ _reranker_cache[model_name] = CrossEncoder(model_name)
77
+ return _reranker_cache[model_name]
78
+
79
+
80
+ def chunk_text(
81
+ text: str,
82
+ source: str,
83
+ chunk_size: int = CHUNK_SIZE_CHARS,
84
+ overlap: int = CHUNK_OVERLAP_CHARS,
85
+ ) -> list:
86
+ chunks = []
87
+ start = 0
88
+ n = len(text)
89
+ while start < n:
90
+ end = min(start + chunk_size, n)
91
+ piece = text[start:end].strip()
92
+ if piece:
93
+ chunks.append({"source": source, "text": piece})
94
+ if end == n:
95
+ break
96
+ start = end - overlap
97
+ return chunks
98
+
99
+
100
+ @dataclass
101
+ class RagIndex:
102
+ chunks: list
103
+ vectorizer: object # sklearn.feature_extraction.text.TfidfVectorizer, fitted
104
+ matrix: object # scipy.sparse matrix, vectorizer's transform of every chunk's text
105
+ embeddings: Optional[np.ndarray] = None
106
+
107
+
108
+ def build_index(
109
+ documents,
110
+ embedding_model: str = DEFAULT_EMBEDDING_MODEL,
111
+ compute_embeddings: bool = True,
112
+ chunk_size: int = CHUNK_SIZE_CHARS,
113
+ overlap: int = CHUNK_OVERLAP_CHARS,
114
+ ) -> RagIndex:
115
+ """
116
+ documents: iterable of (text, source) pairs, one per already-extracted
117
+ document (PDF/markdown/plain text parsing happens before this call --
118
+ this module only chunks and indexes text it's handed). compute_embeddings
119
+ can be set False to skip the (slower) bi-encoder pass and build a
120
+ TF-IDF-only index -- still needs the 'rag' extra for scikit-learn, just
121
+ not sentence-transformers' model download; search() then falls back to
122
+ plain cosine regardless of `rerank`.
123
+ """
124
+ try:
125
+ from sklearn.feature_extraction.text import TfidfVectorizer
126
+ except ImportError as _import_error: # pragma: no cover -- only reachable without scikit-learn installed, which CI here always has
127
+ raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
128
+
129
+ all_chunks = []
130
+ for text, source in documents:
131
+ all_chunks.extend(chunk_text(text, source, chunk_size, overlap))
132
+ if not all_chunks:
133
+ raise ValueError("no chunks produced -- 'documents' was empty or every text was blank")
134
+
135
+ texts = [c["text"] for c in all_chunks]
136
+ vectorizer = TfidfVectorizer(stop_words="english", max_features=20000, ngram_range=(1, 2))
137
+ matrix = vectorizer.fit_transform(texts)
138
+
139
+ embeddings = None
140
+ if compute_embeddings:
141
+ embeddings = _get_embedder(embedding_model).encode(
142
+ texts, normalize_embeddings=True, show_progress_bar=False
143
+ )
144
+
145
+ return RagIndex(chunks=all_chunks, vectorizer=vectorizer, matrix=matrix, embeddings=embeddings)
146
+
147
+
148
+ def save_index(index: RagIndex, index_dir) -> None:
149
+ index_dir = Path(index_dir)
150
+ index_dir.mkdir(parents=True, exist_ok=True)
151
+ with open(index_dir / "chunks.json", "w", encoding="utf-8") as f:
152
+ json.dump(index.chunks, f, ensure_ascii=False, indent=2)
153
+ with open(index_dir / "vectorizer.pkl", "wb") as f:
154
+ pickle.dump(index.vectorizer, f)
155
+ with open(index_dir / "matrix.pkl", "wb") as f:
156
+ pickle.dump(index.matrix, f)
157
+ if index.embeddings is not None:
158
+ np.save(index_dir / "embeddings.npy", index.embeddings)
159
+
160
+
161
+ def load_index(index_dir) -> RagIndex:
162
+ index_dir = Path(index_dir)
163
+ with open(index_dir / "chunks.json", encoding="utf-8") as f:
164
+ chunks = json.load(f)
165
+ with open(index_dir / "vectorizer.pkl", "rb") as f:
166
+ vectorizer = pickle.load(f)
167
+ with open(index_dir / "matrix.pkl", "rb") as f:
168
+ matrix = pickle.load(f)
169
+ emb_path = index_dir / "embeddings.npy"
170
+ embeddings = np.load(emb_path) if emb_path.exists() else None
171
+ return RagIndex(chunks=chunks, vectorizer=vectorizer, matrix=matrix, embeddings=embeddings)
172
+
173
+
174
+ def search(
175
+ query: str,
176
+ index: RagIndex,
177
+ top: int = 3,
178
+ rerank: bool = True,
179
+ pool: int = DEFAULT_POOL,
180
+ embedding_model: str = DEFAULT_EMBEDDING_MODEL,
181
+ cross_encoder_model: str = DEFAULT_CROSS_ENCODER_MODEL,
182
+ ) -> list:
183
+ """
184
+ Returns the top `top` chunks as a list of dicts (rank, source, text,
185
+ score, score_type, cosine), highest-scoring first.
186
+
187
+ rerank=True (default): stage 1 pools `pool` candidates from TF-IDF
188
+ cosine UNION dense-embedding similarity (if the index has embeddings --
189
+ plain TF-IDF pool otherwise), stage 2 reorders that pool with the
190
+ cross-encoder and returns its top `top` ("score_type": "rerank").
191
+ rerank=False: plain TF-IDF cosine ranking, no sentence-transformers model
192
+ load needed (still needs the 'rag' extra's scikit-learn for cosine_similarity
193
+ itself) -- useful as a baseline, or when only embeddings are unavailable
194
+ for this index but a scored ranking is still wanted.
195
+ """
196
+ try:
197
+ from sklearn.metrics.pairwise import cosine_similarity
198
+ except ImportError as _import_error: # pragma: no cover -- only reachable without scikit-learn installed, which CI here always has
199
+ raise ImportError(_MISSING_RAG_EXTRA.format(_import_error)) from _import_error
200
+
201
+ query_vec = index.vectorizer.transform([query])
202
+ cosine_scores = cosine_similarity(query_vec, index.matrix)[0]
203
+ tfidf_order = cosine_scores.argsort()[::-1]
204
+
205
+ if rerank and index.embeddings is not None:
206
+ query_emb = _get_embedder(embedding_model).encode([query], normalize_embeddings=True)[0]
207
+ dense_scores = index.embeddings @ query_emb
208
+ dense_pool = dense_scores.argsort()[::-1][:pool]
209
+ candidate_idx = sorted(set(tfidf_order[:pool].tolist()) | set(dense_pool.tolist()))
210
+ elif rerank:
211
+ candidate_idx = tfidf_order[:pool].tolist()
212
+ else:
213
+ candidate_idx = []
214
+
215
+ if rerank and candidate_idx:
216
+ pairs = [(query, index.chunks[i]["text"][:1024]) for i in candidate_idx]
217
+ rerank_scores = _get_reranker(cross_encoder_model).predict(pairs)
218
+ order = rerank_scores.argsort()[::-1][:top]
219
+ top_idx = [candidate_idx[j] for j in order]
220
+ scores = {candidate_idx[j]: float(rerank_scores[j]) for j in order}
221
+ score_type = "rerank"
222
+ else:
223
+ top_idx = tfidf_order[:top].tolist()
224
+ scores = {i: float(cosine_scores[i]) for i in top_idx}
225
+ score_type = "cosine"
226
+
227
+ return [
228
+ {
229
+ "rank": rank,
230
+ "source": index.chunks[i]["source"],
231
+ "text": index.chunks[i]["text"],
232
+ "score": scores[i],
233
+ "score_type": score_type,
234
+ "cosine": float(cosine_scores[i]),
235
+ }
236
+ for rank, i in enumerate(top_idx, 1)
237
+ ]
238
+
239
+
240
+ def search_exact(pattern: str, index: RagIndex, regex: bool = False, max_hits: int = 10, context: int = 300) -> list:
241
+ """Substring/regex search over an index's raw chunk text -- no
242
+ embedding, no reranker, no model download, so this needs neither the
243
+ 'rag' extra nor a built index's embeddings (works even on an index
244
+ built with compute_embeddings=False).
245
+
246
+ Semantic search (search(), above) ranks by topical similarity, which
247
+ can bury a short, specific, load-bearing phrase (an exact clause, a
248
+ fixed parameter value, a named condition) under chunks that are
249
+ merely more topically central. Use this instead when you already know
250
+ roughly what wording you're looking for and semantic search isn't
251
+ surfacing it high enough.
252
+
253
+ regex=False (default) treats `pattern` as a literal, case-insensitive
254
+ substring (re.escape'd internally); regex=True treats it as a real
255
+ regular expression, case-sensitive, exactly as written. Stops after
256
+ `max_hits` matches (a `truncated` field on the last result marks this,
257
+ rather than silently returning a partial, unmarked list); `context`
258
+ characters of surrounding text are kept on each side of the match.
259
+
260
+ Returns a list of {source, chunk_index, match, snippet, start, end}
261
+ dicts, at most one per chunk -- the chunk's first match only, same as
262
+ the original quantumrag CLI this was ported from, not an exhaustive
263
+ enumeration of every occurrence within a chunk. Empty list (not an
264
+ error) if nothing matches.
265
+ """
266
+ flags = 0 if regex else re.IGNORECASE
267
+ compiled = re.compile(pattern if regex else re.escape(pattern), flags)
268
+ hits = []
269
+ for i, chunk in enumerate(index.chunks):
270
+ text = chunk["text"]
271
+ m = compiled.search(text)
272
+ if not m:
273
+ continue
274
+ lo = max(0, m.start() - context)
275
+ hi = min(len(text), m.end() + context)
276
+ snippet = text[lo:hi]
277
+ hits.append({
278
+ "source": chunk["source"],
279
+ "chunk_index": i,
280
+ "match": m.group(0),
281
+ "snippet": ("..." if lo > 0 else "") + snippet + ("..." if hi < len(text) else ""),
282
+ "start": m.start(),
283
+ "end": m.end(),
284
+ })
285
+ if len(hits) >= max_hits:
286
+ hits[-1]["truncated"] = True
287
+ break
288
+ return hits