rag-wright 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- rag_wright/__init__.py +13 -0
- rag_wright/api/__init__.py +33 -0
- rag_wright/api/config.py +59 -0
- rag_wright/api/discover.py +70 -0
- rag_wright/api/documents.py +39 -0
- rag_wright/api/ids.py +31 -0
- rag_wright/api/invoke.py +99 -0
- rag_wright/api/kg.py +61 -0
- rag_wright/api/mcp.py +94 -0
- rag_wright/api/usage.py +30 -0
- rag_wright/api/workspace.py +85 -0
- rag_wright/capabilities/__init__.py +8 -0
- rag_wright/capabilities/answer_generator.py +427 -0
- rag_wright/capabilities/ard.py +286 -0
- rag_wright/capabilities/assertion_extraction.py +79 -0
- rag_wright/capabilities/chunk_read.py +58 -0
- rag_wright/capabilities/chunk_write.py +163 -0
- rag_wright/capabilities/claim_extraction.py +153 -0
- rag_wright/capabilities/clause_exception_linking.py +117 -0
- rag_wright/capabilities/compliance_judgment.py +322 -0
- rag_wright/capabilities/compliance_store.py +87 -0
- rag_wright/capabilities/contract_kg_serve.py +156 -0
- rag_wright/capabilities/contract_kg_store.py +251 -0
- rag_wright/capabilities/dg_extraction.py +585 -0
- rag_wright/capabilities/disambiguation.py +163 -0
- rag_wright/capabilities/document_parse.py +87 -0
- rag_wright/capabilities/document_scope.py +49 -0
- rag_wright/capabilities/embedding.py +164 -0
- rag_wright/capabilities/embedding_profiles.py +43 -0
- rag_wright/capabilities/entity_resolution.py +154 -0
- rag_wright/capabilities/fusion.py +64 -0
- rag_wright/capabilities/graph_extraction.py +243 -0
- rag_wright/capabilities/graph_query.py +73 -0
- rag_wright/capabilities/graph_storage.py +111 -0
- rag_wright/capabilities/highlight_serve.py +142 -0
- rag_wright/capabilities/hybrid_search.py +65 -0
- rag_wright/capabilities/invoke.py +31 -0
- rag_wright/capabilities/jev_decision.py +38 -0
- rag_wright/capabilities/manifests.py +872 -0
- rag_wright/capabilities/okf_navigate.py +456 -0
- rag_wright/capabilities/parsing.py +286 -0
- rag_wright/capabilities/property_boosted_retrieval.py +125 -0
- rag_wright/capabilities/query_function_classifier.py +94 -0
- rag_wright/capabilities/query_understanding.py +109 -0
- rag_wright/capabilities/registry.py +262 -0
- rag_wright/capabilities/remote_encoders.py +94 -0
- rag_wright/capabilities/requirement_extraction.py +247 -0
- rag_wright/capabilities/reranking.py +123 -0
- rag_wright/capabilities/retrieval_core.py +126 -0
- rag_wright/capabilities/rlm_chunking.py +808 -0
- rag_wright/capabilities/rlm_synthesis.py +316 -0
- rag_wright/capabilities/scan_quality.py +136 -0
- rag_wright/capabilities/span_relevance_judgment.py +191 -0
- rag_wright/capabilities/vision_to_text.py +85 -0
- rag_wright/capabilities/vlm_ocr.py +85 -0
- rag_wright/contracts/__init__.py +6 -0
- rag_wright/contracts/chunk.py +79 -0
- rag_wright/contracts/compliance.py +303 -0
- rag_wright/contracts/contract_meta.py +27 -0
- rag_wright/contracts/extraction.py +130 -0
- rag_wright/contracts/function.py +167 -0
- rag_wright/contracts/function_routing.py +91 -0
- rag_wright/contracts/highlight.py +74 -0
- rag_wright/contracts/identifiers.py +153 -0
- rag_wright/contracts/jurisdiction.py +96 -0
- rag_wright/contracts/ontology.py +142 -0
- rag_wright/contracts/property.py +201 -0
- rag_wright/contracts/provenance.py +78 -0
- rag_wright/contracts/query_intent.py +53 -0
- rag_wright/contracts/span.py +76 -0
- rag_wright/contracts/value_match.py +84 -0
- rag_wright/corpus/__init__.py +0 -0
- rag_wright/corpus/canonicalize.py +116 -0
- rag_wright/corpus/cuad.py +153 -0
- rag_wright/corpus/cuad_ingestion.py +72 -0
- rag_wright/corpus/document_parser.py +299 -0
- rag_wright/corpus/edgar.py +231 -0
- rag_wright/corpus/gcs_ingestion.py +120 -0
- rag_wright/corpus/http.py +110 -0
- rag_wright/corpus/selection.py +152 -0
- rag_wright/mcp/__init__.py +11 -0
- rag_wright/mcp/compliance_server.py +299 -0
- rag_wright/mcp/intra_document_qa_server.py +170 -0
- rag_wright/mcp/relational_qa_server.py +171 -0
- rag_wright/mcp/session_store.py +64 -0
- rag_wright/mcp/typed_property_retrieval_server.py +191 -0
- rag_wright/models/__init__.py +8 -0
- rag_wright/models/profiles.py +331 -0
- rag_wright/models/seam.py +497 -0
- rag_wright/models/tag_structured.py +285 -0
- rag_wright/models/tracing.py +179 -0
- rag_wright/models/usage.py +102 -0
- rag_wright/okf/__init__.py +11 -0
- rag_wright/okf/compile.py +292 -0
- rag_wright/okf/document.py +47 -0
- rag_wright/okf/enrich.py +176 -0
- rag_wright/okf/links.py +190 -0
- rag_wright/okf/lint.py +105 -0
- rag_wright/ontology/__init__.py +6 -0
- rag_wright/ontology/_generated_template_meta.py +60 -0
- rag_wright/ontology/_generated_vocab.py +52 -0
- rag_wright/ontology/clause_template.py +964 -0
- rag_wright/ontology/codegen.py +84 -0
- rag_wright/ontology/compliance_bridge.ttl +186 -0
- rag_wright/ontology/contract_bridge.ttl +2685 -0
- rag_wright/ontology/contract_taxonomy.py +24 -0
- rag_wright/ontology/derive.py +58 -0
- rag_wright/ontology/loader.py +435 -0
- rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
- rag_wright/ontology/registry.py +87 -0
- rag_wright/ontology/template_introspect.py +100 -0
- rag_wright/py.typed +0 -0
- rag_wright/reference/__init__.py +2 -0
- rag_wright/reference/compliance.py +41 -0
- rag_wright/reference/contract_seam.py +123 -0
- rag_wright/skills/__init__.py +7 -0
- rag_wright/skills/claim_extraction/SKILL.md +47 -0
- rag_wright/skills/claim_extraction/__init__.py +1 -0
- rag_wright/skills/claim_extraction/template.py +50 -0
- rag_wright/skills/compliance_judgment/SKILL.md +59 -0
- rag_wright/skills/corpus_ingest/SKILL.md +106 -0
- rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
- rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
- rag_wright/skills/generation/SKILL.md +64 -0
- rag_wright/skills/generation/__init__.py +1 -0
- rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
- rag_wright/skills/okf_navigate/SKILL.md +137 -0
- rag_wright/skills/requirement_extraction/SKILL.md +47 -0
- rag_wright/skills/requirement_extraction/__init__.py +1 -0
- rag_wright/skills/requirement_extraction/template.py +50 -0
- rag_wright/skills/rlm/SKILL.md +186 -0
- rag_wright/skills/rlm/__init__.py +31 -0
- rag_wright/skills/rlm/agent.py +292 -0
- rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
- rag_wright/skills/vision_to_text/SKILL.md +36 -0
- rag_wright/skills/vision_to_text/__init__.py +1 -0
- rag_wright/spans/__init__.py +1 -0
- rag_wright/spans/boundary.py +78 -0
- rag_wright/spans/clause_function_classifier.py +490 -0
- rag_wright/spans/clause_kg_extractor.py +337 -0
- rag_wright/spans/cuad_labels.py +81 -0
- rag_wright/spans/dim_classifier.py +158 -0
- rag_wright/spans/dim_fleet.json +411 -0
- rag_wright/spans/function_classifier.py +77 -0
- rag_wright/spans/function_families.py +62 -0
- rag_wright/spans/hybrid_classifier.py +103 -0
- rag_wright/spans/legalbert_classifier.py +83 -0
- rag_wright/spans/model_capabilities.py +107 -0
- rag_wright/spans/new_function_labels.py +111 -0
- rag_wright/spans/page_map.py +68 -0
- rag_wright/spans/property_extractor.py +365 -0
- rag_wright/spans/property_grounding.py +182 -0
- rag_wright/spans/reclassify.py +77 -0
- rag_wright/spans/scarce_function_labels.py +105 -0
- rag_wright/spans/segment.py +341 -0
- rag_wright/spans/semantic_judge.py +197 -0
- rag_wright/spans/symbolic_validation.py +131 -0
- rag_wright/spans/tag_clause_extractor.py +182 -0
- rag_wright/store/__init__.py +6 -0
- rag_wright/store/arcadedb.py +1135 -0
- rag_wright/store/chunk_text.py +66 -0
- rag_wright/store/seam.py +213 -0
- rag_wright/subgraphs/__init__.py +0 -0
- rag_wright/subgraphs/async_ingestion.py +204 -0
- rag_wright/subgraphs/compliance_check.py +1042 -0
- rag_wright/subgraphs/compliance_ingestion.py +306 -0
- rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
- rag_wright/subgraphs/graph_extraction.py +102 -0
- rag_wright/subgraphs/intra_document_qa.py +328 -0
- rag_wright/subgraphs/observability.py +140 -0
- rag_wright/subgraphs/query_constraint_extraction.py +73 -0
- rag_wright/subgraphs/relational_qa.py +165 -0
- rag_wright/subgraphs/requirement_extraction.py +137 -0
- rag_wright/subgraphs/scaffold.py +65 -0
- rag_wright/subgraphs/semantic_chunking.py +183 -0
- rag_wright/subgraphs/typed_clause_extraction.py +172 -0
- rag_wright/subgraphs/typed_property_retrieval.py +278 -0
- rag_wright/util/__init__.py +1 -0
- rag_wright/util/concurrent.py +153 -0
- rag_wright/util/spacy_model.py +45 -0
- rag_wright-0.1.0.dist-info/METADATA +168 -0
- rag_wright-0.1.0.dist-info/RECORD +184 -0
- rag_wright-0.1.0.dist-info/WHEEL +4 -0
- rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""LG-2: `graph_extraction` as a granular, PARALLEL LangGraph subgraph.
|
|
2
|
+
|
|
3
|
+
The extractor stack (FR-C.6), behind the `Extractor` seam, expressed as a fan-out/fan-in graph rather than a
|
|
4
|
+
sequential `run_extractors` loop. The default stack is now the single GP-1B docling-graph party extractor
|
|
5
|
+
(ADR-0035; the retired T23-27 hybrid was spaCy NER + contract-LLM + escalation), but the graph is generic over
|
|
6
|
+
any list of extractors:
|
|
7
|
+
|
|
8
|
+
START
|
|
9
|
+
/ | \\ (one node per extractor, run in parallel)
|
|
10
|
+
ex0 ex1 ...
|
|
11
|
+
\\ | /
|
|
12
|
+
merge (ExtractionResult.merge over all results)
|
|
13
|
+
|
|
|
14
|
+
END
|
|
15
|
+
|
|
16
|
+
- **parallel fan-out**: the extractors are independent (each reads the same chunk text), so they run in one
|
|
17
|
+
superstep; each appends its `ExtractionResult` to a reducer-accumulated `results` list (no write conflict).
|
|
18
|
+
- **graceful degradation**: an extractor that fails after retries contributes an EMPTY `ExtractionResult`
|
|
19
|
+
(via runtime.execution_info.node_attempt) rather than dropping the whole chunk -- partial facts beat none.
|
|
20
|
+
- **observability**: the GP-1B extractor calls docling-graph (a raw-SDK call that bypasses LangChain), so it
|
|
21
|
+
is INVISIBLE to tracing unless wrapped -- each extractor node is wrapped in `raw_llm_span` per the
|
|
22
|
+
GraphWright observability contract.
|
|
23
|
+
- **merge**: `ExtractionResult.merge` combines mentions + clause/relationship facts, deterministically.
|
|
24
|
+
|
|
25
|
+
`extractors` is injected (defaulting to `default_extractors()`), so the graph is hermetically testable with
|
|
26
|
+
stub extractors -- no live docling-graph / no network.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import operator
|
|
32
|
+
from typing import Annotated, Any, Optional, TypedDict
|
|
33
|
+
|
|
34
|
+
from langgraph.graph import END, START, StateGraph
|
|
35
|
+
from langgraph.runtime import Runtime
|
|
36
|
+
|
|
37
|
+
from rag_wright.capabilities.graph_extraction import Extractor, default_extractors
|
|
38
|
+
from rag_wright.contracts.extraction import ExtractionResult
|
|
39
|
+
from rag_wright.contracts.identifiers import ChunkId
|
|
40
|
+
from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, raw_llm_span
|
|
41
|
+
from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction # retryable-blip signal
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class GraphExtractionState(TypedDict, total=False):
|
|
45
|
+
chunk_id: ChunkId
|
|
46
|
+
text: str
|
|
47
|
+
results: Annotated[list, operator.add] # each extractor node appends its ExtractionResult (reducer)
|
|
48
|
+
result: ExtractionResult # the merged output
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _extractor_node(extractor: Extractor, max_attempts: int):
|
|
52
|
+
name = getattr(extractor, "name", extractor.__class__.__name__)
|
|
53
|
+
|
|
54
|
+
def node(state: GraphExtractionState, runtime: Runtime) -> GraphExtractionState:
|
|
55
|
+
attempt = runtime.execution_info.node_attempt
|
|
56
|
+
# GP-1B docling-graph is a raw-SDK call (ADR-0035); the model id is carried by the extractor.
|
|
57
|
+
with raw_llm_span(f"graph_extraction.{name}", model=getattr(extractor, "model_id", "docling-graph")):
|
|
58
|
+
try:
|
|
59
|
+
result = extractor.extract(state["chunk_id"], state["text"])
|
|
60
|
+
except Exception as exc: # noqa: BLE001 - retry (as a transient), or degrade to empty on exhaustion
|
|
61
|
+
if attempt >= max_attempts:
|
|
62
|
+
return {"results": [ExtractionResult(chunk_id=state["chunk_id"])]} # graceful: no facts
|
|
63
|
+
raise TransientExtraction(str(exc)) from exc # normalize so the RetryPolicy retries it
|
|
64
|
+
return {"results": [result]}
|
|
65
|
+
|
|
66
|
+
return node
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def build_graph_extraction(
|
|
70
|
+
extractors: Optional[list[Extractor]] = None,
|
|
71
|
+
*,
|
|
72
|
+
retry_policy: Any = DEFAULT_RETRY,
|
|
73
|
+
):
|
|
74
|
+
"""Compile the `graph_extraction` subgraph (per chunk). `extractors` is injected (defaulting to the live
|
|
75
|
+
hybrid stack). `retry_policy` is each extractor node's policy (overridable for fast tests)."""
|
|
76
|
+
|
|
77
|
+
extractors = extractors if extractors is not None else default_extractors()
|
|
78
|
+
max_attempts = int(getattr(retry_policy, "max_attempts", 3))
|
|
79
|
+
|
|
80
|
+
def merge(state: GraphExtractionState) -> GraphExtractionState:
|
|
81
|
+
return {"result": ExtractionResult.merge(state["chunk_id"], state.get("results", []))}
|
|
82
|
+
|
|
83
|
+
g = StateGraph(GraphExtractionState)
|
|
84
|
+
g.add_node("merge", merge)
|
|
85
|
+
for i, extractor in enumerate(extractors):
|
|
86
|
+
name = f"extract_{i}_{getattr(extractor, 'name', 'x')}"
|
|
87
|
+
g.add_node(name, _extractor_node(extractor, max_attempts), retry_policy=retry_policy)
|
|
88
|
+
g.add_edge(START, name) # fan-out: all extractors run in parallel
|
|
89
|
+
g.add_edge(name, "merge") # fan-in: merge runs after all complete
|
|
90
|
+
g.add_edge("merge", END)
|
|
91
|
+
return g.compile()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def register_graph_extraction_subgraph(registry) -> None:
|
|
95
|
+
"""LG-2: register `graph_extraction` (subgraph). The manifest/slug already exist (reclassified in
|
|
96
|
+
CAP-REG-1); this binds the LangGraph runnable's contract (ExtractionResult)."""
|
|
97
|
+
registry.register(
|
|
98
|
+
"graph_extraction",
|
|
99
|
+
contract=ExtractionResult,
|
|
100
|
+
kind="subgraph",
|
|
101
|
+
display_name="Graph extraction (GP-1B docling-graph party/relational)",
|
|
102
|
+
)
|
|
@@ -0,0 +1,328 @@
|
|
|
1
|
+
"""LG-3b: `intra_document_qa` as a composite LangGraph subgraph.
|
|
2
|
+
|
|
3
|
+
Answer a question scoped to ONE contract with a grounded, cited answer, composing the intra-contract scoped
|
|
4
|
+
KG query (KG-4, `contract_kg_serve`) with grounded answer generation (FR-Q.6):
|
|
5
|
+
|
|
6
|
+
START --> serve [RetryPolicy] (scoped KG query: contract_id + question -> cited CitedClauses)
|
|
7
|
+
|
|
|
8
|
+
v
|
|
9
|
+
assemble [RetryPolicy] (rehydrate each clause's REAL span text -> cited EvidenceItem)
|
|
10
|
+
| --orphan span--> dead_letter --> END
|
|
11
|
+
v
|
|
12
|
+
generate (generate_answer: grounded/cited/abstaining answer) --> END
|
|
13
|
+
|
|
14
|
+
Query-side posture (matches `query_constraint_extraction` / `relational_qa`): judicious hardening.
|
|
15
|
+
- **serve** retries a transient store blip; on retry exhaustion it DEGRADES to no clauses -> empty evidence
|
|
16
|
+
-> the generator abstains, so the query is never dropped.
|
|
17
|
+
- **assemble** rehydrates each served clause to its OPERATIVE SPAN TEXT (the clause node stores only
|
|
18
|
+
id/function/folio; the text lives on the SPAN records, reached via each clause's `span_id` provenance).
|
|
19
|
+
The evidence the generator reads is the real clause language, cited by `clause_id`, with the typed
|
|
20
|
+
`(dimension, value)` facts appended and the worst-case property confidence surfaced (FR-S.4). A span_id
|
|
21
|
+
the store cannot resolve (an orphan) dead-letters rather than fabricating; a property-less clause (no
|
|
22
|
+
span) cites its function label. A transient rehydration blip retries, then dead-letters on exhaustion.
|
|
23
|
+
- **generate** enforces no-claim-without-a-citation in code (FR-Q.6): empty evidence abstains with no model
|
|
24
|
+
call; fabricated citations are dropped.
|
|
25
|
+
|
|
26
|
+
`serve_fn` / `clause_text_fn` / `generate_fn` are dependency-injected so the graph is hermetically testable
|
|
27
|
+
with stubs -- no live LLM or store. `production_intra_document_qa` wires the real scoped query
|
|
28
|
+
(function-classified + serve), the span-text rehydration (`spans_by_contract`), and `generate_answer`. None is
|
|
29
|
+
a raw-SDK call (serve/rehydrate are store reads; generate_answer uses the LangChain seam, auto-captured), so
|
|
30
|
+
per-node visibility is a `business_span`.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
from __future__ import annotations
|
|
34
|
+
|
|
35
|
+
from typing import Any, Awaitable, Callable, Optional, TypedDict
|
|
36
|
+
|
|
37
|
+
from langgraph.graph import END, START, StateGraph
|
|
38
|
+
from langgraph.runtime import Runtime
|
|
39
|
+
|
|
40
|
+
from rag_wright.capabilities.answer_generator import EvidenceItem, GeneratedAnswer
|
|
41
|
+
from rag_wright.capabilities.clause_exception_linking import CAP_FUNCTION
|
|
42
|
+
from rag_wright.capabilities.contract_kg_serve import CitedClause, CitedProperty
|
|
43
|
+
from rag_wright.contracts.provenance import ConfidenceTag
|
|
44
|
+
from rag_wright.models import tracing # 0048: emit retrieval as a Langfuse span, split from the generation
|
|
45
|
+
from rag_wright.subgraphs.scaffold import DEFAULT_RETRY, business_span, dead_letter
|
|
46
|
+
from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction # shared retryable-blip signal
|
|
47
|
+
|
|
48
|
+
# serve_fn: (contract_id, question) -> the scoped clauses (must raise on a transient store blip).
|
|
49
|
+
ServeFn = Callable[[str, str], list[CitedClause]]
|
|
50
|
+
# clause_text_fn: (contract_id, clauses) -> {clause_id: operative-span text}; a clause with a span_id that
|
|
51
|
+
# cannot be resolved raises KeyError (no silent drop); a property-less clause is simply absent from the map.
|
|
52
|
+
ClauseTextFn = Callable[[str, list[CitedClause]], dict[str, str]]
|
|
53
|
+
# ASYNC-C1 (ADR-0057): generate_fn is async (the model call gets a true wall-clock deadline via the seam).
|
|
54
|
+
GenerateFn = Callable[[str, list[EvidenceItem]], Awaitable[GeneratedAnswer]]
|
|
55
|
+
|
|
56
|
+
# Worst-case provenance surfaced to the generator: the least-trusted tag among a clause's properties wins.
|
|
57
|
+
_CONFIDENCE_ORDER = ("AMBIGUOUS", "INFERRED", "EXTRACTED")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class IntraDocumentQAState(TypedDict, total=False):
|
|
61
|
+
contract_id: str
|
|
62
|
+
question: str
|
|
63
|
+
clauses: list[CitedClause]
|
|
64
|
+
evidence: list[EvidenceItem]
|
|
65
|
+
answer: GeneratedAnswer
|
|
66
|
+
dead_letter: Optional[dict]
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _clause_confidence(properties: list[CitedProperty]) -> Optional[str]:
|
|
70
|
+
tags = {p.confidence for p in properties if p.confidence}
|
|
71
|
+
for tag in _CONFIDENCE_ORDER: # worst-case first
|
|
72
|
+
if tag in tags:
|
|
73
|
+
return tag
|
|
74
|
+
return None
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _humanize_properties(clause: CitedClause) -> str:
|
|
78
|
+
"""Body-less path (issue 0011): render typed facts as reader-safe natural text -- NOT the `[dim=value]`
|
|
79
|
+
schema syntax the model latched onto and paraphrased. Domain-agnostic prettify (snake_case -> spaces); the
|
|
80
|
+
structured form still rides out-of-band on `EvidenceItem.properties`. No `[`, no `=`, no backticks."""
|
|
81
|
+
def pretty(s: str) -> str:
|
|
82
|
+
return str(s).replace("_", " ")
|
|
83
|
+
return "; ".join(f"{pretty(p.dimension)}: {pretty(p.value)}" for p in clause.properties)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _clause_to_evidence(clause: CitedClause, body: Optional[str]) -> Optional[EvidenceItem]:
|
|
87
|
+
"""One cited evidence item: the clause's REAL span text (when rehydrated), cited by `clause_id`, confidence
|
|
88
|
+
surfaced. Typed properties ride OUT-OF-BAND on `EvidenceItem.properties`, never in the evidence text.
|
|
89
|
+
|
|
90
|
+
Engine issue 0002 / ADR-0054: the KG-assigned function label is NOT put in the evidence text (it was the
|
|
91
|
+
source of the auto-tag paraphrase leak -- a sometimes-wrong classification narrated in the engine's voice).
|
|
92
|
+
|
|
93
|
+
Engine issue 0011 / ADR-0064: the SAME move for typed properties. They used to be concatenated into the text
|
|
94
|
+
as `[dimension=value; ...]`; the model paraphrased that schema string into prose ("as indicated by the typed
|
|
95
|
+
property cap_quantum=..."), around the bracket scrub. They now travel out-of-band on `EvidenceItem.properties`
|
|
96
|
+
(code-generated `{dimension, value}`), so the generator never sees the schema tokens and narration is
|
|
97
|
+
structurally impossible. The body span already states in natural language what the properties encode, so
|
|
98
|
+
dropping them from the text costs the generator nothing; they stay available for the product's UI chips.
|
|
99
|
+
|
|
100
|
+
Body-less path: where a clause has NO span text, the properties are the only content, so they must stay
|
|
101
|
+
citable -- rendered as reader-safe natural text (`_humanize_properties`), never the `[dim=value]` syntax.
|
|
102
|
+
|
|
103
|
+
A clause with no span text AND no typed facts is contentless -- nothing to ground a citation on -- so it is
|
|
104
|
+
DROPPED (returns None), rather than cited by a bare function label (superseding PREC-1a's fallback)."""
|
|
105
|
+
props = [{"dimension": p.dimension, "value": p.value} for p in clause.properties] or None
|
|
106
|
+
if body:
|
|
107
|
+
text = body # 0011: properties NOT appended -> the generator cannot quote the schema tokens
|
|
108
|
+
elif props:
|
|
109
|
+
text = _humanize_properties(clause) # body-less: reader-safe natural text, still citable
|
|
110
|
+
else:
|
|
111
|
+
return None # contentless: no span text, no facts -> not citable -> drop
|
|
112
|
+
if clause.exception_of:
|
|
113
|
+
# ADR-0044: an INFERRED carve-out/exception to a cap clause -> frame it as such and surface INFERRED
|
|
114
|
+
# confidence, so the generator answers "capped, EXCEPT ..." and treats it as inferred, never a hard claim.
|
|
115
|
+
return EvidenceItem(
|
|
116
|
+
chunk_id=clause.clause_id,
|
|
117
|
+
text=f"[Exception to the liability cap (inferred)] {text}",
|
|
118
|
+
confidence=ConfidenceTag.INFERRED.value, properties=props)
|
|
119
|
+
return EvidenceItem(chunk_id=clause.clause_id, text=text,
|
|
120
|
+
confidence=_clause_confidence(clause.properties), properties=props)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def attach_exception_links(
|
|
124
|
+
clauses: list[CitedClause], exceptions_fn: Callable[[str], list[dict]], *, contract_id: str
|
|
125
|
+
) -> list[CitedClause]:
|
|
126
|
+
"""ADR-0044 query consumption: for each served Cap clause, pull its `IsExceptionTo` carve-outs
|
|
127
|
+
(`exceptions_fn(cap_clause_id) -> [{clause_id, function, span_id}]`) and include them as INFERRED exceptions
|
|
128
|
+
(`exception_of` set), deduped. Pulls the cap's conditions into the evidence EVEN IF the classifier did not
|
|
129
|
+
return the Uncapped function -- that is the point. A clause already served (via classification) is marked
|
|
130
|
+
as this cap's exception; a not-yet-served one is added (property-less, rehydrated from its own span_id)."""
|
|
131
|
+
by_id = {c.clause_id: c for c in clauses}
|
|
132
|
+
for cap in list(clauses):
|
|
133
|
+
if cap.function != CAP_FUNCTION or cap.exception_of:
|
|
134
|
+
continue
|
|
135
|
+
for exc in exceptions_fn(cap.clause_id):
|
|
136
|
+
eid = exc.get("clause_id")
|
|
137
|
+
if not eid or eid == cap.clause_id:
|
|
138
|
+
continue
|
|
139
|
+
if eid in by_id:
|
|
140
|
+
if not by_id[eid].exception_of:
|
|
141
|
+
by_id[eid].exception_of = cap.clause_id
|
|
142
|
+
else:
|
|
143
|
+
added = CitedClause(
|
|
144
|
+
contract_id=contract_id, clause_id=eid, function=exc.get("function") or "",
|
|
145
|
+
span_id=exc.get("span_id") or "", exception_of=cap.clause_id, properties=[])
|
|
146
|
+
by_id[eid] = added
|
|
147
|
+
clauses.append(added)
|
|
148
|
+
return clauses
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def rehydrate_clause_texts(store: Any, contract_id: str, clauses: list[CitedClause]) -> dict[str, str]:
|
|
152
|
+
"""Map each clause to its operative-span TEXT for citation. A clause WITH typed properties uses its property
|
|
153
|
+
span_ids (grounding invariant: they MUST resolve, else KeyError). A PROPERTY-LESS clause uses its OWN
|
|
154
|
+
`span_id` (1:1, ADR-0025) -- real span text, not a bare function label, and never a function-label guess
|
|
155
|
+
(which is one-to-many). A clause with no span link at all (legacy pre-backfill) is omitted, and the evidence
|
|
156
|
+
builder falls back to the function label. This is why the clause-level span_id is persisted."""
|
|
157
|
+
functions = sorted({c.function for c in clauses if c.function})
|
|
158
|
+
text_by_span = {row["span_id"]: row["text"] for row in store.spans_by_contract(contract_id, functions)}
|
|
159
|
+
out: dict[str, str] = {}
|
|
160
|
+
for clause in clauses:
|
|
161
|
+
span_ids = list(dict.fromkeys(p.span_id for p in clause.properties if p.span_id))
|
|
162
|
+
if span_ids:
|
|
163
|
+
bodies = []
|
|
164
|
+
for span_id in span_ids:
|
|
165
|
+
if span_id not in text_by_span: # property provenance MUST resolve (grounding invariant)
|
|
166
|
+
raise KeyError(
|
|
167
|
+
f"clause {clause.clause_id}: span {span_id!r} has no text in contract {contract_id}")
|
|
168
|
+
bodies.append(text_by_span[span_id])
|
|
169
|
+
out[clause.clause_id] = " ".join(bodies)
|
|
170
|
+
elif clause.span_id and clause.span_id in text_by_span:
|
|
171
|
+
out[clause.clause_id] = text_by_span[clause.span_id] # property-less: its OWN span (1:1), real text
|
|
172
|
+
return out
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def build_intra_document_qa(
|
|
176
|
+
serve_fn: ServeFn,
|
|
177
|
+
clause_text_fn: ClauseTextFn,
|
|
178
|
+
generate_fn: GenerateFn,
|
|
179
|
+
*,
|
|
180
|
+
retry_policy: Any = DEFAULT_RETRY,
|
|
181
|
+
):
|
|
182
|
+
"""Compile the `intra_document_qa` subgraph. `serve_fn` / `clause_text_fn` / `generate_fn` are injected for
|
|
183
|
+
hermetic testing; `retry_policy` is the serve and assemble nodes' policy (overridable for fast tests)."""
|
|
184
|
+
max_attempts = int(getattr(retry_policy, "max_attempts", 3))
|
|
185
|
+
|
|
186
|
+
def serve(state: IntraDocumentQAState, runtime: Runtime) -> IntraDocumentQAState:
|
|
187
|
+
# node_attempt is 1-indexed; a transient blip re-raises so the RetryPolicy retries, EXCEPT on the
|
|
188
|
+
# final attempt where it degrades to NO clauses (the generator abstains -- the query is never lost).
|
|
189
|
+
attempt = runtime.execution_info.node_attempt
|
|
190
|
+
with business_span("intra_document_qa.serve", contract_id=state["contract_id"]), \
|
|
191
|
+
tracing.traced_step("intra_document_qa.serve"): # 0048: retrieval span, separable from generate
|
|
192
|
+
try:
|
|
193
|
+
clauses = serve_fn(state["contract_id"], state["question"])
|
|
194
|
+
except Exception as exc: # noqa: BLE001 - transient -> retry, or degrade to empty on exhaustion
|
|
195
|
+
if attempt >= max_attempts:
|
|
196
|
+
return {"clauses": []}
|
|
197
|
+
raise TransientExtraction(str(exc)) from exc
|
|
198
|
+
return {"clauses": clauses}
|
|
199
|
+
|
|
200
|
+
def assemble(state: IntraDocumentQAState, runtime: Runtime) -> IntraDocumentQAState:
|
|
201
|
+
clauses = state.get("clauses", [])
|
|
202
|
+
if not clauses:
|
|
203
|
+
return {"evidence": []}
|
|
204
|
+
attempt = runtime.execution_info.node_attempt
|
|
205
|
+
contract_id = state["contract_id"]
|
|
206
|
+
with business_span("intra_document_qa.assemble", clause_count=len(clauses)), \
|
|
207
|
+
tracing.traced_step("intra_document_qa.assemble"): # 0048: rehydration/retrieval span
|
|
208
|
+
try:
|
|
209
|
+
texts = clause_text_fn(contract_id, clauses)
|
|
210
|
+
except KeyError as exc: # an orphan span_id is a pipeline inconsistency: surface, never fabricate
|
|
211
|
+
return {"dead_letter": dead_letter(
|
|
212
|
+
"clause_text_orphan_span", contract_id=contract_id, error=str(exc))}
|
|
213
|
+
except Exception as exc: # noqa: BLE001 - transient store blip -> retry, or dead-letter on exhaust
|
|
214
|
+
if attempt >= max_attempts:
|
|
215
|
+
return {"dead_letter": dead_letter(
|
|
216
|
+
"clause_text_rehydration_failed", contract_id=contract_id, error=str(exc))}
|
|
217
|
+
raise TransientExtraction(str(exc)) from exc
|
|
218
|
+
# drop contentless clauses (no span text, no facts) -> None (engine issue 0002 / ADR-0054)
|
|
219
|
+
evidence = [ev for c in clauses if (ev := _clause_to_evidence(c, texts.get(c.clause_id))) is not None]
|
|
220
|
+
return {"evidence": evidence}
|
|
221
|
+
|
|
222
|
+
async def generate(state: IntraDocumentQAState) -> IntraDocumentQAState:
|
|
223
|
+
with business_span("intra_document_qa.generate"):
|
|
224
|
+
answer = await generate_fn(state["question"], state.get("evidence", []))
|
|
225
|
+
return {"answer": answer}
|
|
226
|
+
|
|
227
|
+
g = StateGraph(IntraDocumentQAState)
|
|
228
|
+
g.add_node("serve", serve, retry_policy=retry_policy)
|
|
229
|
+
g.add_node("assemble", assemble, retry_policy=retry_policy)
|
|
230
|
+
g.add_node("generate", generate)
|
|
231
|
+
g.add_edge(START, "serve")
|
|
232
|
+
g.add_edge("serve", "assemble")
|
|
233
|
+
g.add_conditional_edges("assemble", lambda s: "end" if s.get("dead_letter") else "generate",
|
|
234
|
+
{"generate": "generate", "end": END})
|
|
235
|
+
g.add_edge("generate", END)
|
|
236
|
+
return g.compile()
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _answer_model_for_impl(model_id: str | None = None, **kwargs: Any) -> Any:
|
|
240
|
+
"""Indirection over `answer_model_for` so `production_intra_document_qa` can default the answer model (and
|
|
241
|
+
tests can monkeypatch this hook). Lazy import keeps the subgraph module import-light."""
|
|
242
|
+
from rag_wright.capabilities.answer_generator import answer_model_for
|
|
243
|
+
return answer_model_for(model_id, **kwargs)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def production_intra_document_qa(
|
|
247
|
+
*, store: Any, reranker: Any = None, answer_model: Any = None, answer_model_id: str | None = None,
|
|
248
|
+
top_k: int = 12,
|
|
249
|
+
):
|
|
250
|
+
"""Wire the real scoped query + span-text rehydration + `generate_answer` into the composite. `serve` takes
|
|
251
|
+
the WHOLE per-contract clause KG (`contract_clause_index`) and ranks it by BGE cross-encoder relevance to the
|
|
252
|
+
question, serving only the top-`k` (bounded evidence). Rehydration maps each clause's `span_id` provenance to
|
|
253
|
+
its operative-span text (`spans_by_contract`, contract-scoped and light -- no dense vectors).
|
|
254
|
+
|
|
255
|
+
ADR-0047: the function-classifier narrowing (`clauses_of_function`) was REPLACED by this SEMANTIC narrowing --
|
|
256
|
+
a clause mislabel can no longer hide the real clause (BGE ranks by MEANING, not the LegalBERT label), while
|
|
257
|
+
the top-`k` cap keeps generation bounded (a contract can hold 100+ clauses ~= 13k tokens, too much to dump
|
|
258
|
+
whole). The `reranker` defaults to `BGEReranker` (injectable for tests / the A100 path).
|
|
259
|
+
|
|
260
|
+
The answer model defaults to `answer_model_for(answer_model_id)` (GENERAL role when None), so the configured
|
|
261
|
+
generation model automatically takes the RIGHT path -- the client-side free-text tag-parse for a
|
|
262
|
+
`client_side_structured` model (self-hosted Gemma), the structured-output seam otherwise. A caller may still
|
|
263
|
+
inject a specific `answer_model` (tests, or to force a strategy). Imports are lazy so the subgraph module
|
|
264
|
+
stays import-light and hermetic (tests inject stubs)."""
|
|
265
|
+
from rag_wright.capabilities.answer_generator import agenerate_answer
|
|
266
|
+
|
|
267
|
+
if answer_model is None:
|
|
268
|
+
answer_model = _answer_model_for_impl(answer_model_id)
|
|
269
|
+
if reranker is None:
|
|
270
|
+
from rag_wright.capabilities.reranking import BGEReranker
|
|
271
|
+
reranker = BGEReranker()
|
|
272
|
+
from rag_wright.capabilities.contract_kg_serve import contract_clause_index
|
|
273
|
+
from rag_wright.capabilities.contract_kg_store import ContractKGStore
|
|
274
|
+
|
|
275
|
+
# EP-REF-1a-ii: the typed-KG edge reads moved to the domain store extension; the raw `store` still serves the
|
|
276
|
+
# generic retrieval (span search, rehydrate). The serving reads go through a ContractKGStore over it.
|
|
277
|
+
ckg = ContractKGStore(store)
|
|
278
|
+
|
|
279
|
+
def serve(contract_id: str, question: str) -> list[CitedClause]:
|
|
280
|
+
# ADR-0044: pull each cap clause's INFERRED carve-outs (IsExceptionTo) so "how is liability capped, and
|
|
281
|
+
# under what conditions?" sees "capped, except uncapped for ...".
|
|
282
|
+
# 0006-D: include_untyped so a clause the classifier left NONE is still a candidate (recall must not
|
|
283
|
+
# depend on classification -- else a classifier miss is a silent recall hole, engine issue 0006).
|
|
284
|
+
base = attach_exception_links(
|
|
285
|
+
contract_clause_index(ckg, contract_id, include_untyped=True), ckg.exceptions_of_clause,
|
|
286
|
+
contract_id=contract_id)
|
|
287
|
+
if len(base) <= top_k:
|
|
288
|
+
return base # small contract -> no narrowing needed
|
|
289
|
+
# SEMANTIC top-K (ADR-0047): BGE-rerank the contract's clauses by relevance to the question.
|
|
290
|
+
texts = rehydrate_clause_texts(store, contract_id, base)
|
|
291
|
+
scores = reranker.score(question, [texts.get(c.clause_id, "") for c in base])
|
|
292
|
+
ranked = [c for c, _ in sorted(zip(base, scores), key=lambda cs: -cs[1])]
|
|
293
|
+
top = ranked[:top_k]
|
|
294
|
+
kept = {c.clause_id for c in top}
|
|
295
|
+
for c in base: # ADR-0044: keep a kept cap clause's carve-outs even if they scored outside top-K
|
|
296
|
+
if c.exception_of and c.exception_of in kept and c.clause_id not in kept:
|
|
297
|
+
top.append(c)
|
|
298
|
+
kept.add(c.clause_id)
|
|
299
|
+
return top
|
|
300
|
+
|
|
301
|
+
def clause_text(contract_id: str, clauses: list[CitedClause]) -> dict[str, str]:
|
|
302
|
+
return rehydrate_clause_texts(store, contract_id, clauses)
|
|
303
|
+
|
|
304
|
+
async def generate(question: str, evidence: list[EvidenceItem]) -> GeneratedAnswer:
|
|
305
|
+
return await agenerate_answer(question, evidence, model=answer_model)
|
|
306
|
+
|
|
307
|
+
return build_intra_document_qa(serve, clause_text, generate)
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def register_intra_document_qa(registry) -> None:
|
|
311
|
+
"""LG-3b: register `intra_document_qa` (composite subgraph; scoped KG query -> rehydrate -> generate)."""
|
|
312
|
+
registry.register(
|
|
313
|
+
"intra_document_qa",
|
|
314
|
+
contract=GeneratedAnswer,
|
|
315
|
+
kind="subgraph",
|
|
316
|
+
display_name="Intra-document QA (cited answer scoped to one contract)",
|
|
317
|
+
)
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
async def ainvoke(resources, inputs: dict):
|
|
321
|
+
"""EP-CORE-2 (ADR-0118): the capability invoke factory (impl_ref target)."""
|
|
322
|
+
from rag_wright.capabilities.answer_generator import answer_model_for
|
|
323
|
+
from rag_wright.models.profiles import ModelRole
|
|
324
|
+
|
|
325
|
+
graph = production_intra_document_qa(store=resources._store,
|
|
326
|
+
answer_model=answer_model_for(resources.model_id(ModelRole.GENERAL)),
|
|
327
|
+
top_k=inputs.get("top_k", 12))
|
|
328
|
+
return await graph.ainvoke({"contract_id": inputs["contract_id"], "question": inputs["question"]})
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
"""LG-0 observability: the vendor-neutral seam over the AMBIENT OpenTelemetry tracer.
|
|
2
|
+
|
|
3
|
+
Per GraphWright's observability contract (`temp/observability-contract.md`): we DO NOT create a tracer
|
|
4
|
+
provider, exporter, or Langfuse client, and we DO NOT initialize global tracing. GraphWright installs global
|
|
5
|
+
OTel instrumentation and exports to Langfuse (or ANY OTLP backend -- Phoenix/Jaeger/collector; the backend is
|
|
6
|
+
swappable with **zero change here**, which is the whole point). Our only jobs:
|
|
7
|
+
|
|
8
|
+
1. Build every model through the LangChain seam (`models/seam.py` -> `ChatOpenAI`), so token counts + latency
|
|
9
|
+
are captured automatically -- no code here for those.
|
|
10
|
+
2. Instrument the ONE gap: **raw-SDK** calls that bypass LangChain (docling-graph / LiteLLM, sandboxed or
|
|
11
|
+
out-of-band model calls) -- give them a span on the AMBIENT tracer with a model name + token usage.
|
|
12
|
+
3. Optionally add domain **business spans** (retrieval stats, a custom segment) on the ambient tracer.
|
|
13
|
+
4. Propagate the OTel context across any boundary WE introduce (separate process / worker / queue / custom
|
|
14
|
+
async loop). Std-lib threads are carried by GraphWright's threading instrumentation.
|
|
15
|
+
|
|
16
|
+
This module is a safe NO-OP unless a REAL tracer provider is installed (GraphWright's runtime): standalone /
|
|
17
|
+
hermetic tests, every helper no-ops; under GraphWright it attaches to the installed provider and nests
|
|
18
|
+
correctly. Activation hinges on a real provider being set, NOT on `opentelemetry` merely being importable --
|
|
19
|
+
the API can arrive as a transitive dependency (e.g. via FastMCP) with no SDK/provider configured, and that must
|
|
20
|
+
NOT flip instrumentation on. Never import Langfuse; never construct a provider. Ground OTel via the framework
|
|
21
|
+
index / docs when it is installed under GraphWright.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
from __future__ import annotations
|
|
25
|
+
|
|
26
|
+
from contextlib import contextmanager
|
|
27
|
+
from typing import Any, Iterator, Optional
|
|
28
|
+
|
|
29
|
+
try: # the API may be importable standalone (transitive dep) -> gate ACTIVATION on a real provider, not this
|
|
30
|
+
from opentelemetry import context as _context
|
|
31
|
+
from opentelemetry import trace as _trace
|
|
32
|
+
from opentelemetry.propagate import extract as _extract
|
|
33
|
+
from opentelemetry.propagate import inject as _inject
|
|
34
|
+
|
|
35
|
+
_OTEL_IMPORTABLE = True
|
|
36
|
+
except Exception: # noqa: BLE001 - opentelemetry not installed -> graceful no-op seam
|
|
37
|
+
_OTEL_IMPORTABLE = False
|
|
38
|
+
|
|
39
|
+
# The default (unconfigured) providers the OTel API returns before GraphWright sets a real SDK provider. When
|
|
40
|
+
# the current provider is one of these, spans are non-recording, so every helper must no-op.
|
|
41
|
+
_NOOP_PROVIDER_TYPES = frozenset({"ProxyTracerProvider", "NoOpTracerProvider", "DefaultTracerProvider"})
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _provider_installed() -> bool:
|
|
45
|
+
"""True only when a REAL tracer provider is installed (GraphWright's instrumented runtime), not the default
|
|
46
|
+
proxy/no-op the API ships with. This, not mere importability, is what activates the seam."""
|
|
47
|
+
if not _OTEL_IMPORTABLE:
|
|
48
|
+
return False
|
|
49
|
+
return type(_trace.get_tracer_provider()).__name__ not in _NOOP_PROVIDER_TYPES
|
|
50
|
+
|
|
51
|
+
# OpenTelemetry GenAI semantic-convention attribute names (what OTLP backends read for token/cost views).
|
|
52
|
+
_GENAI_MODEL = "gen_ai.request.model"
|
|
53
|
+
_GENAI_SYSTEM = "gen_ai.system"
|
|
54
|
+
_GENAI_IN = "gen_ai.usage.input_tokens"
|
|
55
|
+
_GENAI_OUT = "gen_ai.usage.output_tokens"
|
|
56
|
+
_GENAI_TOTAL = "gen_ai.usage.total_tokens"
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def otel_active() -> bool:
|
|
60
|
+
"""True when a real tracer provider is installed (i.e. running under GraphWright's instrumented runtime)."""
|
|
61
|
+
return _provider_installed()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _tracer():
|
|
65
|
+
return _trace.get_tracer("rag_wright.subgraphs") if _provider_installed() else None
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@contextmanager
|
|
69
|
+
def business_span(name: str, **attributes: Any) -> Iterator[Any]:
|
|
70
|
+
"""A domain/business span on the AMBIENT tracer (never a new provider). No-op when OTel is absent.
|
|
71
|
+
|
|
72
|
+
For domain steps / retrieval stats -- NOT for standard LangChain LLM/tool calls (already captured; a manual
|
|
73
|
+
span would duplicate them).
|
|
74
|
+
"""
|
|
75
|
+
tr = _tracer()
|
|
76
|
+
if tr is None:
|
|
77
|
+
yield None
|
|
78
|
+
return
|
|
79
|
+
with tr.start_as_current_span(name) as span:
|
|
80
|
+
for key, value in attributes.items():
|
|
81
|
+
span.set_attribute(key, value)
|
|
82
|
+
yield span
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@contextmanager
|
|
86
|
+
def raw_llm_span(name: str, *, model: str, system: Optional[str] = None) -> Iterator[Any]:
|
|
87
|
+
"""Instrument a RAW-SDK model call (the one gap: docling-graph/LiteLLM, sandboxed calls) on the ambient
|
|
88
|
+
tracer, so it shows up in traces with a model name. Call `record_tokens(span, ...)` after the call for
|
|
89
|
+
usage. Duration is the span's own. No-op when OTel is absent."""
|
|
90
|
+
tr = _tracer()
|
|
91
|
+
if tr is None:
|
|
92
|
+
yield None
|
|
93
|
+
return
|
|
94
|
+
with tr.start_as_current_span(name) as span:
|
|
95
|
+
span.set_attribute(_GENAI_MODEL, model)
|
|
96
|
+
if system:
|
|
97
|
+
span.set_attribute(_GENAI_SYSTEM, system)
|
|
98
|
+
yield span
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def record_tokens(
|
|
102
|
+
span: Any,
|
|
103
|
+
*,
|
|
104
|
+
input_tokens: Optional[int] = None,
|
|
105
|
+
output_tokens: Optional[int] = None,
|
|
106
|
+
total_tokens: Optional[int] = None,
|
|
107
|
+
) -> None:
|
|
108
|
+
"""Record token usage on a raw-SDK span (the only counting you own; LangChain calls set these themselves).
|
|
109
|
+
No-op on a None span."""
|
|
110
|
+
if span is None:
|
|
111
|
+
return
|
|
112
|
+
if input_tokens is not None:
|
|
113
|
+
span.set_attribute(_GENAI_IN, input_tokens)
|
|
114
|
+
if output_tokens is not None:
|
|
115
|
+
span.set_attribute(_GENAI_OUT, output_tokens)
|
|
116
|
+
if total_tokens is not None:
|
|
117
|
+
span.set_attribute(_GENAI_TOTAL, total_tokens)
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def inject_context(carrier: dict) -> dict:
|
|
121
|
+
"""Producer side: serialize the current OTel context into `carrier` before crossing a boundary you
|
|
122
|
+
introduce (separate process / external worker / queue / custom async). Returns the carrier. No-op without
|
|
123
|
+
OTel. Std-lib threads do NOT need this (GraphWright instruments them)."""
|
|
124
|
+
if _provider_installed():
|
|
125
|
+
_inject(carrier)
|
|
126
|
+
return carrier
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
@contextmanager
|
|
130
|
+
def attach_context(carrier: dict) -> Iterator[None]:
|
|
131
|
+
"""Consumer side: re-attach a context carried across a boundary so the work nests under the original run.
|
|
132
|
+
No-op without a real provider."""
|
|
133
|
+
if not _provider_installed():
|
|
134
|
+
yield
|
|
135
|
+
return
|
|
136
|
+
token = _context.attach(_extract(carrier))
|
|
137
|
+
try:
|
|
138
|
+
yield
|
|
139
|
+
finally:
|
|
140
|
+
_context.detach(token)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""LG-2: `query_constraint_extraction` as a hardened LangGraph subgraph.
|
|
2
|
+
|
|
3
|
+
The query-side counterpart of `typed_clause_extraction` (LG-1): run the SAME granite + `clause_template`
|
|
4
|
+
extractor on the QUERY, producing its typed (dimension, value) constraints (KG-5b). Two differences from the
|
|
5
|
+
clause subgraph, both deliberate:
|
|
6
|
+
|
|
7
|
+
- **no reground / escalation** -- the query IS the source text, and KG-5d found reground-on-query FALSE-FLAGS
|
|
8
|
+
real constraints (e.g. "control the defense" != the clause cue "control of the defense"), so we do not
|
|
9
|
+
ground query constraints; a single extract call is the pattern (KG-5e, granite json_schema);
|
|
10
|
+
- **no dead-letter-drops-the-item** -- a query that yields nothing degrades to an EMPTY constraint set
|
|
11
|
+
(the retrieval falls back to embedding-only ranking), never a dropped item.
|
|
12
|
+
|
|
13
|
+
Hardening applied JUDICIOUSLY (not every subgraph needs every primitive): a single extract with **graceful
|
|
14
|
+
degradation** -- a transient OR genuine failure yields an EMPTY constraint set (the query survives), so there
|
|
15
|
+
is no RetryPolicy or dead-letter here. The observability wrap on the raw-SDK (docling-graph/LiteLLM) call is
|
|
16
|
+
kept. `record_fn` is injected for hermetic testing.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
from typing import Callable, Optional, TypedDict
|
|
22
|
+
|
|
23
|
+
from langgraph.graph import END, START, StateGraph
|
|
24
|
+
|
|
25
|
+
from rag_wright.contracts.property import ClausePropertyRecord
|
|
26
|
+
from rag_wright.subgraphs.scaffold import raw_llm_span
|
|
27
|
+
from rag_wright.subgraphs.typed_clause_extraction import TransientExtraction # shared retryable-blip signal
|
|
28
|
+
|
|
29
|
+
RecordFn = Callable[[str, str], Optional[ClausePropertyRecord]]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class QueryConstraintState(TypedDict, total=False):
|
|
33
|
+
query_text: str
|
|
34
|
+
model_id: str
|
|
35
|
+
record: Optional[ClausePropertyRecord]
|
|
36
|
+
constraints: list # [(dimension, value), ...] -- empty when extraction yields nothing
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _constraints_of(record: Optional[ClausePropertyRecord]) -> list:
|
|
40
|
+
if record is None:
|
|
41
|
+
return []
|
|
42
|
+
return [(a.dimension.value, a.value) for a in record.assertions]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def build_query_constraint_extraction(record_fn: RecordFn, *, model_id: str):
|
|
46
|
+
"""Compile the `query_constraint_extraction` subgraph. `record_fn` is injected for hermetic testing.
|
|
47
|
+
A genuine extraction failure (record_fn -> None) or a transient failure that exhausts retries degrades to
|
|
48
|
+
an empty constraint set -- the query is never dropped."""
|
|
49
|
+
|
|
50
|
+
def extract(state: QueryConstraintState) -> QueryConstraintState:
|
|
51
|
+
model = state.get("model_id", model_id)
|
|
52
|
+
try:
|
|
53
|
+
with raw_llm_span("query_constraint_extraction.extract", model=model):
|
|
54
|
+
record = record_fn(state["query_text"], model)
|
|
55
|
+
except TransientExtraction:
|
|
56
|
+
record = None # any failure -> empty constraints; the query survives (embedding-only fallback)
|
|
57
|
+
return {"record": record, "constraints": _constraints_of(record)}
|
|
58
|
+
|
|
59
|
+
g = StateGraph(QueryConstraintState)
|
|
60
|
+
g.add_node("extract", extract)
|
|
61
|
+
g.add_edge(START, "extract")
|
|
62
|
+
g.add_edge("extract", END)
|
|
63
|
+
return g.compile()
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def register_query_constraint_extraction(registry) -> None:
|
|
67
|
+
"""LG-2: register `query_constraint_extraction` (subgraph; query-side typed constraint extraction)."""
|
|
68
|
+
registry.register(
|
|
69
|
+
"query_constraint_extraction",
|
|
70
|
+
contract=ClausePropertyRecord,
|
|
71
|
+
kind="subgraph",
|
|
72
|
+
display_name="Query constraint extraction",
|
|
73
|
+
)
|