rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,123 @@
1
+ """Reranking (FR-C.4, FR-Q.2): a cross-encoder precision gate over the retrieved candidates.
2
+
3
+ A BGE-reranker cross-encoder re-scores each retrieved candidate against the query and cuts the list to
4
+ a top-k set, the precision gate before any expensive downstream work (synthesis). Unlike the bi-encoder
5
+ retrieval legs (T19/T21), a cross-encoder reads the query and the passage together, so it is more
6
+ precise but too costly to run over the whole index — it runs only over the already-fused candidate list
7
+ (T21).
8
+
9
+ This capability is a pure function of `(query, passage)` pairs: it is given the candidates *with their
10
+ passage text* and returns them reranked and cut. It deliberately does not fetch the text itself — the
11
+ passage a candidate is scored on (the chunk summary, or the full chunk text from the parse manifest,
12
+ FR-I.1) is a wiring choice the compiled query graph makes, not this capability's, so reranking stays
13
+ decoupled from the store and the manifest. Grounded on `FlagEmbedding.FlagAutoReranker` (the public
14
+ auto-reranker; `.compute_score` over sentence pairs).
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ from typing import Protocol, runtime_checkable
20
+
21
+ from pydantic import BaseModel
22
+
23
+
24
+ DEFAULT_TOP_K = 5 # the precision-gate cut before synthesis; the eval and caller can override
25
+ DEFAULT_RERANKER_MODEL = "BAAI/bge-reranker-v2-m3" # the BGE-M3 companion cross-encoder
26
+
27
+
28
+ @runtime_checkable
29
+ class Reranker(Protocol):
30
+ """The cross-encoder seam: score each passage against the query (higher is more relevant)."""
31
+
32
+ def score(self, query: str, passages: list[str]) -> list[float]: ...
33
+
34
+
35
+ class Passage(BaseModel):
36
+ """A retrieved candidate with the passage text to score (the reranker's input unit)."""
37
+
38
+ model_config = {"frozen": True}
39
+
40
+ chunk_id: str
41
+ source_doc_id: str
42
+ text: str
43
+
44
+
45
+ class ScoredCandidate(BaseModel):
46
+ """One reranked candidate: the relevance score the cross-encoder gave it."""
47
+
48
+ model_config = {"frozen": True}
49
+
50
+ chunk_id: str
51
+ source_doc_id: str
52
+ score: float
53
+
54
+
55
+ class RerankResult(BaseModel):
56
+ """The reranking capability's output: the top-k candidates by cross-encoder score (FR-C.4)."""
57
+
58
+ model_config = {"frozen": True}
59
+
60
+ query: str
61
+ candidates: list[ScoredCandidate] # cross-encoder order, best first, cut to top_k
62
+
63
+
64
+ class BGEReranker:
65
+ """The real reranker: `FlagEmbedding.FlagAutoReranker` (model loaded lazily).
66
+
67
+ THREAD-SAFE (engine issue 0016): FlagEmbedding mutates the model in place on every call (the same in-place
68
+ `.to()`/`.eval()` conversions that segfault the shared BGE-M3 embedder under thread concurrency), so a shared
69
+ reranker is guarded by an instance lock too -- defense-in-depth for any concurrent `score` caller."""
70
+
71
+ def __init__(self, model_name: str = DEFAULT_RERANKER_MODEL, *, use_fp16: bool = False,
72
+ model: object | None = None) -> None:
73
+ import threading
74
+
75
+ self._lock = threading.Lock() # issue 0016: serialize the in-place-mutating compute_score across threads
76
+ if model is not None: # injected (hermetic tests)
77
+ self._model = model
78
+ else:
79
+ from FlagEmbedding import FlagAutoReranker
80
+
81
+ self._model = FlagAutoReranker.from_finetuned(model_name, use_fp16=use_fp16)
82
+
83
+ def score(self, query: str, passages: list[str]) -> list[float]:
84
+ if not passages:
85
+ return []
86
+ with self._lock: # issue 0016
87
+ scores = self._model.compute_score([(query, passage) for passage in passages])
88
+ # compute_score returns a scalar for a single pair; normalize to a list of floats.
89
+ if not isinstance(scores, (list, tuple)):
90
+ scores = [scores]
91
+ return [float(s) for s in scores]
92
+
93
+
94
+ def rerank(
95
+ query: str,
96
+ passages: list[Passage],
97
+ *,
98
+ reranker: Reranker,
99
+ top_k: int = DEFAULT_TOP_K,
100
+ ) -> RerankResult:
101
+ """Rerank the candidate passages by cross-encoder relevance and cut to the top-`k` set.
102
+
103
+ The cross-encoder scores each `(query, passage.text)` pair; the passages are sorted by score
104
+ descending (stable, so equal scores keep their incoming fused order) and cut to `top_k`. An empty
105
+ candidate list reranks to an empty result.
106
+ """
107
+ if not passages:
108
+ return RerankResult(query=query, candidates=[])
109
+
110
+ scores = reranker.score(query, [p.text for p in passages])
111
+ if len(scores) != len(passages):
112
+ raise ValueError(
113
+ f"reranker returned {len(scores)} scores for {len(passages)} passages (must be 1:1)"
114
+ )
115
+
116
+ scored = [
117
+ ScoredCandidate(chunk_id=p.chunk_id, source_doc_id=p.source_doc_id, score=score)
118
+ for p, score in zip(passages, scores)
119
+ ]
120
+ scored.sort(key=lambda c: c.score, reverse=True) # stable: ties keep the incoming fused order
121
+ return RerankResult(query=query, candidates=scored[:top_k])
122
+
123
+
@@ -0,0 +1,126 @@
1
+ """CAP-REG-3 (FR-Q, ADR-0033): the KG-primary retrieval core, packaged out of `eval/kg_primary.py`.
2
+
3
+ The eval script ranked ACORD candidates through inline closures (`pool_of`, `_match`, `_tiebreak`) wrapped in
4
+ MODE/VARIANT/MATCH ablation scaffolding. This module lifts the ADOPTED operating point out of that scaffolding
5
+ as three pure, deterministic `function` capabilities, each registered under its FR-C slug so the query graph
6
+ can bind them (`typed_constraint_match_rank` / `dense_rank_tiebreak` back the adopted Leg B via
7
+ `property_boosted_retrieval`; `candidate_routing` is a general union-combiner utility, unused since the redundant
8
+ `cross_corpus_retrieval` subgraph was retired):
9
+
10
+ - **candidate_routing** -- the union combiner: several routing signals (LLM / LegalBERT classifier /
11
+ dimension-prior) each propose ranked functions; union them (first-wins, recall-safe) and fetch the
12
+ candidate clause pool for that function set (KG-5e). The store pool lookup is an injected seam.
13
+ - **typed_constraint_match_rank** -- grade each candidate by how many query (dimension, value) constraints
14
+ its grounded typed props satisfy, under KG-5a canonicalization + subsumption (`constraint_match_count`).
15
+ Recall-safe: a zero-match candidate keeps its place (stable sort), never dropped.
16
+ - **dense_rank_tiebreak** -- order candidates by descending cosine to the query vector: the embedding signal
17
+ that breaks constraint-match ties meaningfully (KG-6 / V4). Pure -- vectors come from `embedding`.
18
+
19
+ The three carry NO LLM call (the LLM front door is `query_constraint_extraction` + query function
20
+ classification, upstream); they are exact, deterministic compute -> `function`, not `subgraph`.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from typing import Callable, Sequence
26
+
27
+ from pydantic import BaseModel
28
+
29
+ # The store seam for candidate_routing: given the routed functions, return the candidate clause ids. Bound in
30
+ # production (LG-3c) to the store's function->pool query; injected as a fake in tests.
31
+ PoolFn = Callable[[list[str]], list[str]]
32
+ # EP-CORE-1b: the DOMAIN (dimension, value) match-counter, INJECTED so this retrieval mechanism imports no domain
33
+ # vocab (the contract subsumption/canonicalization lives in `contracts.value_match.constraint_match_count`, provided
34
+ # by the domain caller). `(query_constraints, clause_props) -> int`.
35
+ MatchCountFn = Callable[[set, set], int]
36
+
37
+ _NONE_FUNCTION = "NONE" # the classifier's no-function sentinel; never routes a pool
38
+
39
+
40
+ class CandidatePool(BaseModel):
41
+ """candidate_routing output: the unioned routing functions and the candidate clause ids they select."""
42
+
43
+ functions: list[str]
44
+ candidate_ids: list[str]
45
+
46
+
47
+ class RankedClause(BaseModel):
48
+ """One graded candidate: its clause id and how many query constraints its props satisfy (KG-5a)."""
49
+
50
+ clause_id: str
51
+ match_score: float
52
+
53
+
54
+ class MatchRanking(BaseModel):
55
+ """typed_constraint_match_rank output: candidates in descending graded (constraint-match) order."""
56
+
57
+ ranked: list[RankedClause]
58
+
59
+
60
+ class DenseScoredClause(BaseModel):
61
+ """One candidate scored by cosine similarity to the query vector."""
62
+
63
+ clause_id: str
64
+ cosine: float
65
+
66
+
67
+ class DenseRanking(BaseModel):
68
+ """dense_rank_tiebreak output: candidates in descending cosine order."""
69
+
70
+ ranked: list[DenseScoredClause]
71
+
72
+
73
+ def _cosine(a: Sequence[float], b: Sequence[float]) -> float:
74
+ """Cosine similarity; 0.0 when either vector has zero norm (mirrors eval/kg_primary._cosine)."""
75
+ dot = sum(x * y for x, y in zip(a, b))
76
+ na = sum(x * x for x in a) ** 0.5
77
+ nb = sum(y * y for y in b) ** 0.5
78
+ return dot / (na * nb) if na and nb else 0.0
79
+
80
+
81
+ def candidate_routing(
82
+ function_predictions: Sequence[Sequence[str]], *, pool_fn: PoolFn
83
+ ) -> CandidatePool:
84
+ """Union the ranked function predictions from each router (first-wins, order-preserving; empty and the
85
+ `NONE` sentinel dropped), then fetch the candidate clause pool for that function set. No functions -> an
86
+ empty pool with no store lookup. The recall-safe union combiner (KG-5e)."""
87
+ functions: list[str] = []
88
+ for predictions in function_predictions:
89
+ for function in predictions:
90
+ if function and function != _NONE_FUNCTION and function not in functions:
91
+ functions.append(function)
92
+ candidate_ids = pool_fn(functions) if functions else []
93
+ return CandidatePool(functions=functions, candidate_ids=candidate_ids)
94
+
95
+
96
+ def typed_constraint_match_rank(
97
+ query_constraints: set, candidate_props: Sequence[tuple[str, set]], *, match_count_fn: MatchCountFn
98
+ ) -> MatchRanking:
99
+ """Grade each candidate by how many of the query's (dimension, value) constraints its grounded typed props
100
+ satisfy -- the match semantics are the INJECTED `match_count_fn` (the contract pack supplies KG-5a
101
+ canonicalization + subsumption via `constraint_match_count`; this mechanism stays domain-free). Returns
102
+ descending graded order. Recall-safe: the sort is stable, so a zero-match candidate keeps its input position."""
103
+ scored = [
104
+ RankedClause(clause_id=clause_id, match_score=float(match_count_fn(query_constraints, props)))
105
+ for clause_id, props in candidate_props
106
+ ]
107
+ ranked = sorted(scored, key=lambda r: -r.match_score) # stable -> ties keep input order
108
+ return MatchRanking(ranked=ranked)
109
+
110
+
111
+ def dense_rank_tiebreak(
112
+ query_vector: Sequence[float], candidate_vectors: Sequence[tuple[str, Sequence[float]]]
113
+ ) -> DenseRanking:
114
+ """Order candidates by descending cosine similarity to the query vector -- the embedding tiebreak signal
115
+ (KG-6 / V4). Pure: the vectors come from the `embedding` capability, not from a store or a model here."""
116
+ scored = [
117
+ DenseScoredClause(clause_id=clause_id, cosine=_cosine(query_vector, vector))
118
+ for clause_id, vector in candidate_vectors
119
+ ]
120
+ ranked = sorted(scored, key=lambda r: -r.cosine)
121
+ return DenseRanking(ranked=ranked)
122
+
123
+
124
+ # (EP-CORE-1b/ADR-0118: candidate_routing / typed_constraint_match_rank / dense_rank_tiebreak are de-registered
125
+ # from ARD -- generic retrieval primitives now, composed by direct import (the contract leg injects the domain
126
+ # match-counter). Their register_* functions were removed.)