rag-wright 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (184) hide show
  1. rag_wright/__init__.py +13 -0
  2. rag_wright/api/__init__.py +33 -0
  3. rag_wright/api/config.py +59 -0
  4. rag_wright/api/discover.py +70 -0
  5. rag_wright/api/documents.py +39 -0
  6. rag_wright/api/ids.py +31 -0
  7. rag_wright/api/invoke.py +99 -0
  8. rag_wright/api/kg.py +61 -0
  9. rag_wright/api/mcp.py +94 -0
  10. rag_wright/api/usage.py +30 -0
  11. rag_wright/api/workspace.py +85 -0
  12. rag_wright/capabilities/__init__.py +8 -0
  13. rag_wright/capabilities/answer_generator.py +427 -0
  14. rag_wright/capabilities/ard.py +286 -0
  15. rag_wright/capabilities/assertion_extraction.py +79 -0
  16. rag_wright/capabilities/chunk_read.py +58 -0
  17. rag_wright/capabilities/chunk_write.py +163 -0
  18. rag_wright/capabilities/claim_extraction.py +153 -0
  19. rag_wright/capabilities/clause_exception_linking.py +117 -0
  20. rag_wright/capabilities/compliance_judgment.py +322 -0
  21. rag_wright/capabilities/compliance_store.py +87 -0
  22. rag_wright/capabilities/contract_kg_serve.py +156 -0
  23. rag_wright/capabilities/contract_kg_store.py +251 -0
  24. rag_wright/capabilities/dg_extraction.py +585 -0
  25. rag_wright/capabilities/disambiguation.py +163 -0
  26. rag_wright/capabilities/document_parse.py +87 -0
  27. rag_wright/capabilities/document_scope.py +49 -0
  28. rag_wright/capabilities/embedding.py +164 -0
  29. rag_wright/capabilities/embedding_profiles.py +43 -0
  30. rag_wright/capabilities/entity_resolution.py +154 -0
  31. rag_wright/capabilities/fusion.py +64 -0
  32. rag_wright/capabilities/graph_extraction.py +243 -0
  33. rag_wright/capabilities/graph_query.py +73 -0
  34. rag_wright/capabilities/graph_storage.py +111 -0
  35. rag_wright/capabilities/highlight_serve.py +142 -0
  36. rag_wright/capabilities/hybrid_search.py +65 -0
  37. rag_wright/capabilities/invoke.py +31 -0
  38. rag_wright/capabilities/jev_decision.py +38 -0
  39. rag_wright/capabilities/manifests.py +872 -0
  40. rag_wright/capabilities/okf_navigate.py +456 -0
  41. rag_wright/capabilities/parsing.py +286 -0
  42. rag_wright/capabilities/property_boosted_retrieval.py +125 -0
  43. rag_wright/capabilities/query_function_classifier.py +94 -0
  44. rag_wright/capabilities/query_understanding.py +109 -0
  45. rag_wright/capabilities/registry.py +262 -0
  46. rag_wright/capabilities/remote_encoders.py +94 -0
  47. rag_wright/capabilities/requirement_extraction.py +247 -0
  48. rag_wright/capabilities/reranking.py +123 -0
  49. rag_wright/capabilities/retrieval_core.py +126 -0
  50. rag_wright/capabilities/rlm_chunking.py +808 -0
  51. rag_wright/capabilities/rlm_synthesis.py +316 -0
  52. rag_wright/capabilities/scan_quality.py +136 -0
  53. rag_wright/capabilities/span_relevance_judgment.py +191 -0
  54. rag_wright/capabilities/vision_to_text.py +85 -0
  55. rag_wright/capabilities/vlm_ocr.py +85 -0
  56. rag_wright/contracts/__init__.py +6 -0
  57. rag_wright/contracts/chunk.py +79 -0
  58. rag_wright/contracts/compliance.py +303 -0
  59. rag_wright/contracts/contract_meta.py +27 -0
  60. rag_wright/contracts/extraction.py +130 -0
  61. rag_wright/contracts/function.py +167 -0
  62. rag_wright/contracts/function_routing.py +91 -0
  63. rag_wright/contracts/highlight.py +74 -0
  64. rag_wright/contracts/identifiers.py +153 -0
  65. rag_wright/contracts/jurisdiction.py +96 -0
  66. rag_wright/contracts/ontology.py +142 -0
  67. rag_wright/contracts/property.py +201 -0
  68. rag_wright/contracts/provenance.py +78 -0
  69. rag_wright/contracts/query_intent.py +53 -0
  70. rag_wright/contracts/span.py +76 -0
  71. rag_wright/contracts/value_match.py +84 -0
  72. rag_wright/corpus/__init__.py +0 -0
  73. rag_wright/corpus/canonicalize.py +116 -0
  74. rag_wright/corpus/cuad.py +153 -0
  75. rag_wright/corpus/cuad_ingestion.py +72 -0
  76. rag_wright/corpus/document_parser.py +299 -0
  77. rag_wright/corpus/edgar.py +231 -0
  78. rag_wright/corpus/gcs_ingestion.py +120 -0
  79. rag_wright/corpus/http.py +110 -0
  80. rag_wright/corpus/selection.py +152 -0
  81. rag_wright/mcp/__init__.py +11 -0
  82. rag_wright/mcp/compliance_server.py +299 -0
  83. rag_wright/mcp/intra_document_qa_server.py +170 -0
  84. rag_wright/mcp/relational_qa_server.py +171 -0
  85. rag_wright/mcp/session_store.py +64 -0
  86. rag_wright/mcp/typed_property_retrieval_server.py +191 -0
  87. rag_wright/models/__init__.py +8 -0
  88. rag_wright/models/profiles.py +331 -0
  89. rag_wright/models/seam.py +497 -0
  90. rag_wright/models/tag_structured.py +285 -0
  91. rag_wright/models/tracing.py +179 -0
  92. rag_wright/models/usage.py +102 -0
  93. rag_wright/okf/__init__.py +11 -0
  94. rag_wright/okf/compile.py +292 -0
  95. rag_wright/okf/document.py +47 -0
  96. rag_wright/okf/enrich.py +176 -0
  97. rag_wright/okf/links.py +190 -0
  98. rag_wright/okf/lint.py +105 -0
  99. rag_wright/ontology/__init__.py +6 -0
  100. rag_wright/ontology/_generated_template_meta.py +60 -0
  101. rag_wright/ontology/_generated_vocab.py +52 -0
  102. rag_wright/ontology/clause_template.py +964 -0
  103. rag_wright/ontology/codegen.py +84 -0
  104. rag_wright/ontology/compliance_bridge.ttl +186 -0
  105. rag_wright/ontology/contract_bridge.ttl +2685 -0
  106. rag_wright/ontology/contract_taxonomy.py +24 -0
  107. rag_wright/ontology/derive.py +58 -0
  108. rag_wright/ontology/loader.py +435 -0
  109. rag_wright/ontology/packs/ftc_16cfr255.ttl +29 -0
  110. rag_wright/ontology/registry.py +87 -0
  111. rag_wright/ontology/template_introspect.py +100 -0
  112. rag_wright/py.typed +0 -0
  113. rag_wright/reference/__init__.py +2 -0
  114. rag_wright/reference/compliance.py +41 -0
  115. rag_wright/reference/contract_seam.py +123 -0
  116. rag_wright/skills/__init__.py +7 -0
  117. rag_wright/skills/claim_extraction/SKILL.md +47 -0
  118. rag_wright/skills/claim_extraction/__init__.py +1 -0
  119. rag_wright/skills/claim_extraction/template.py +50 -0
  120. rag_wright/skills/compliance_judgment/SKILL.md +59 -0
  121. rag_wright/skills/corpus_ingest/SKILL.md +106 -0
  122. rag_wright/skills/extraction_semantic_judge/SKILL.md +51 -0
  123. rag_wright/skills/extraction_semantic_judge/__init__.py +1 -0
  124. rag_wright/skills/generation/SKILL.md +64 -0
  125. rag_wright/skills/generation/__init__.py +1 -0
  126. rag_wright/skills/generic_compliance_judgment/SKILL.md +58 -0
  127. rag_wright/skills/okf_navigate/SKILL.md +137 -0
  128. rag_wright/skills/requirement_extraction/SKILL.md +47 -0
  129. rag_wright/skills/requirement_extraction/__init__.py +1 -0
  130. rag_wright/skills/requirement_extraction/template.py +50 -0
  131. rag_wright/skills/rlm/SKILL.md +186 -0
  132. rag_wright/skills/rlm/__init__.py +31 -0
  133. rag_wright/skills/rlm/agent.py +292 -0
  134. rag_wright/skills/span_relevance_judgment/SKILL.md +67 -0
  135. rag_wright/skills/vision_to_text/SKILL.md +36 -0
  136. rag_wright/skills/vision_to_text/__init__.py +1 -0
  137. rag_wright/spans/__init__.py +1 -0
  138. rag_wright/spans/boundary.py +78 -0
  139. rag_wright/spans/clause_function_classifier.py +490 -0
  140. rag_wright/spans/clause_kg_extractor.py +337 -0
  141. rag_wright/spans/cuad_labels.py +81 -0
  142. rag_wright/spans/dim_classifier.py +158 -0
  143. rag_wright/spans/dim_fleet.json +411 -0
  144. rag_wright/spans/function_classifier.py +77 -0
  145. rag_wright/spans/function_families.py +62 -0
  146. rag_wright/spans/hybrid_classifier.py +103 -0
  147. rag_wright/spans/legalbert_classifier.py +83 -0
  148. rag_wright/spans/model_capabilities.py +107 -0
  149. rag_wright/spans/new_function_labels.py +111 -0
  150. rag_wright/spans/page_map.py +68 -0
  151. rag_wright/spans/property_extractor.py +365 -0
  152. rag_wright/spans/property_grounding.py +182 -0
  153. rag_wright/spans/reclassify.py +77 -0
  154. rag_wright/spans/scarce_function_labels.py +105 -0
  155. rag_wright/spans/segment.py +341 -0
  156. rag_wright/spans/semantic_judge.py +197 -0
  157. rag_wright/spans/symbolic_validation.py +131 -0
  158. rag_wright/spans/tag_clause_extractor.py +182 -0
  159. rag_wright/store/__init__.py +6 -0
  160. rag_wright/store/arcadedb.py +1135 -0
  161. rag_wright/store/chunk_text.py +66 -0
  162. rag_wright/store/seam.py +213 -0
  163. rag_wright/subgraphs/__init__.py +0 -0
  164. rag_wright/subgraphs/async_ingestion.py +204 -0
  165. rag_wright/subgraphs/compliance_check.py +1042 -0
  166. rag_wright/subgraphs/compliance_ingestion.py +306 -0
  167. rag_wright/subgraphs/contract_ingestion_pipeline.py +999 -0
  168. rag_wright/subgraphs/graph_extraction.py +102 -0
  169. rag_wright/subgraphs/intra_document_qa.py +328 -0
  170. rag_wright/subgraphs/observability.py +140 -0
  171. rag_wright/subgraphs/query_constraint_extraction.py +73 -0
  172. rag_wright/subgraphs/relational_qa.py +165 -0
  173. rag_wright/subgraphs/requirement_extraction.py +137 -0
  174. rag_wright/subgraphs/scaffold.py +65 -0
  175. rag_wright/subgraphs/semantic_chunking.py +183 -0
  176. rag_wright/subgraphs/typed_clause_extraction.py +172 -0
  177. rag_wright/subgraphs/typed_property_retrieval.py +278 -0
  178. rag_wright/util/__init__.py +1 -0
  179. rag_wright/util/concurrent.py +153 -0
  180. rag_wright/util/spacy_model.py +45 -0
  181. rag_wright-0.1.0.dist-info/METADATA +168 -0
  182. rag_wright-0.1.0.dist-info/RECORD +184 -0
  183. rag_wright-0.1.0.dist-info/WHEEL +4 -0
  184. rag_wright-0.1.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,105 @@
1
+ """Step-4 (classifier rare-class improvement): source SILVER training labels for the SCARCE CUAD function
2
+ classes (0.00-recall / <35 real train spans). Same T60 pattern as `new_function_labels.py`: the operative
3
+ spans CUAD leaves unlabeled (NONE) are candidates -- these classes are under-annotated, so their real
4
+ instances hide in the NONE pool -- narrowed by a cheap high-recall KEYWORD pre-filter, then confirmed by an
5
+ LLM. Unlike T60 these are EXISTING `ClauseCategory` types (they have CUAD gold too); the silver only augments
6
+ TRAINING (mined from train contracts; the SEED=0 holdout stays pure gold, no leakage).
7
+
8
+ Confirmer defaults to the GENERAL model (Gemma) per the benchmarked model decisions -- its structured
9
+ thinking-disable (ADR-0032) makes this single-label forced call reliable; A/B against DeepSeek if it
10
+ underperforms. Keyword pre-filter is unit-testable + free; the confirmer is stub-injectable.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from typing import Optional, Protocol, runtime_checkable
16
+
17
+ from pydantic import BaseModel
18
+
19
+ from rag_wright.contracts.function import canonical_function
20
+ from rag_wright.models.profiles import ModelRole, model_for
21
+ from rag_wright.models.seam import build_structured
22
+
23
+ NONE_LABEL = "NONE"
24
+
25
+ # High-recall keyword pre-filter (lowercased substring). Broad on purpose -- precision is the LLM's job; this
26
+ # only must avoid dropping true positives. Keys MUST be canonical FUNCTION_LABELS.
27
+ SCARCE_KEYWORDS: dict[str, tuple[str, ...]] = {
28
+ "Unlimited/All-You-Can-Eat-License": (
29
+ "unlimited", "all-you-can-eat", "all you can eat", "without limit", "no limit",
30
+ "unrestricted", "no restriction on the number",
31
+ ),
32
+ "Irrevocable Or Perpetual License": (
33
+ "irrevocable", "perpetual", "in perpetuity",
34
+ ),
35
+ "Notice Period To Terminate Renewal": (
36
+ "renew", "auto-renew", "automatically renew", "written notice", "days notice",
37
+ "days' notice", "prior written notice", "non-renew", "not to renew",
38
+ ),
39
+ "Most Favored Nation": (
40
+ "most favored", "most-favored", "no less favorable", "as favorable as", "mfn",
41
+ ),
42
+ "Non-Disparagement": (
43
+ "disparage", "denigrate", "defame",
44
+ ),
45
+ "Third Party Beneficiary": (
46
+ "third party beneficiar", "third-party beneficiar", "no third party",
47
+ ),
48
+ }
49
+
50
+ _VALID = {canonical_function(k) for k in SCARCE_KEYWORDS} # canonical-guard the keys at import
51
+ assert None not in _VALID, "SCARCE_KEYWORDS keys must be canonical FUNCTION_LABELS"
52
+
53
+
54
+ class ScarceLabel(BaseModel):
55
+ """Structured confirm output: the chosen scarce label string (validated to the candidates by the caller)."""
56
+
57
+ label: str
58
+
59
+
60
+ def scarce_candidates(text: str) -> frozenset[str]:
61
+ """The scarce labels a span MIGHT be, by keyword. Empty => skip the LLM entirely."""
62
+ low = text.lower()
63
+ return frozenset(
64
+ label for label, kws in SCARCE_KEYWORDS.items() if any(kw in low for kw in kws)
65
+ )
66
+
67
+
68
+ def scarce_prompt(text: str, candidates: frozenset[str]) -> str:
69
+ options = "\n- ".join(sorted(candidates))
70
+ return (
71
+ "You label a contract clause span for a function classifier. A keyword filter flagged it as possibly "
72
+ "one of these clause types:\n- " + options + "\n\n"
73
+ "Decide which ONE it ACTUALLY is by its operative meaning (not a passing mention), or answer NONE if "
74
+ "it is none of them. Respond with EXACTLY one of the type strings above, or NONE.\n\n"
75
+ "Span:\n" + text[:2000]
76
+ )
77
+
78
+
79
+ @runtime_checkable
80
+ class ScarceConfirmer(Protocol):
81
+ def __call__(self, text: str, candidates: frozenset[str]) -> str: ...
82
+
83
+
84
+ class SeamScarceConfirmer:
85
+ """The real confirmer: structured output through the model-profile seam. Defaults to GENERAL (Gemma);
86
+ pass `model_id` to A/B another model (e.g. DeepSeek). Returns a validated label in `candidates` or NONE
87
+ (a hallucinated / off-list label is treated as NONE, conservative). Retries a transient bare `None`."""
88
+
89
+ def __init__(self, model_id: Optional[str] = None, *, retries: int = 3) -> None:
90
+ self._runnable = build_structured(model_id or model_for(ModelRole.GENERAL), ScarceLabel)
91
+ self._retries = retries
92
+
93
+ def __call__(self, text: str, candidates: frozenset[str]) -> str:
94
+ prompt = scarce_prompt(text, candidates)
95
+ last_error: Exception | None = None
96
+ for _ in range(self._retries):
97
+ try:
98
+ v = self._runnable.invoke(prompt)
99
+ except Exception as e: # noqa: BLE001 - transient provider/parse error; retry
100
+ last_error = e
101
+ continue
102
+ if v is not None:
103
+ canon = canonical_function(v.label)
104
+ return canon if canon in candidates else NONE_LABEL
105
+ raise last_error or ValueError("no structured output after retries")
@@ -0,0 +1,341 @@
1
+ """T55 (FR-R, ADR-0025): operative-span segmenter — re-chunk a clause into operative spans.
2
+
3
+ The recall investigation showed the function/property signal is a LOCAL operative provision, which a
4
+ whole-clause embedding drowns. So we index at the span level (small-to-big): split a clause body on legal
5
+ structure and point each span back to its parent clause. The span is the classified/retrieved unit; the parent
6
+ clause is what we return and rerank.
7
+
8
+ Deterministic and byte-faithful. Split signals (no LLM, no new dependency — the spaCy escalation is added only
9
+ if the downstream metric demands it):
10
+ - enumeration markers at a provision start: `(a)`, `(i)`, `1.`, `12.1`, `12.1.1`, `§`,
11
+ - semicolon / colon list separators,
12
+ - sentence-ending `.` that is a genuine terminator (not a known abbreviation, not a decimal / section ref).
13
+ Sub-floor fragments (a lone heading like "12.1 Limitation of Liability.") fold into the following provision.
14
+
15
+ Invariant: the spans TILE the body -- `"".join(s.text) == body` -- so no character is lost, duplicated, or
16
+ reordered. The `function` tag and embeddings are filled by later tasks (T56); this task produces the spans and
17
+ their parent pointers only.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import re
23
+
24
+ from pydantic import BaseModel
25
+
26
+ from rag_wright.contracts.span import SpanRecord
27
+
28
+ DEFAULT_MIN_CHARS = 25 # a span whose stripped text is shorter folds into its neighbour (a bare heading/marker)
29
+
30
+ # Abbreviations whose trailing '.' does not end a provision (lower-cased, no trailing dot).
31
+ _ABBREV = {
32
+ "inc", "corp", "co", "ltd", "llc", "llp", "plc", "no", "nos", "art", "sec", "secs", "para", "paras",
33
+ "cf", "vs", "v", "mr", "mrs", "ms", "dr", "st", "ave", "etc", "al", "viz", "e.g", "i.e", "u.s", "u.s.c",
34
+ }
35
+
36
+ # An enumeration marker opening a provision: (a) (iv) (12) a) 12) 1. 12.1 12.1.1 § -- captured as group 1.
37
+ _ENUM = re.compile(
38
+ r"(?:(?<=\s)|(?<=[.;:])|\A)\s*"
39
+ r"(\(\s*(?:[a-zA-Z]|[ivxlcdm]{1,4}|\d{1,3})\s*\)" # (a) (iv) (12)
40
+ r"|(?:[a-zA-Z]|[ivxlcdm]{1,4}|\d{1,3})\)" # a) iv) 12)
41
+ r"|\d+(?:\.\d+){1,3}\.?" # 12.1 12.1.1.
42
+ r"|§+)\s+(?=[A-Z\"'(])" # followed by space + capital / quote / paren
43
+ )
44
+
45
+ # A sentence/list terminator followed by whitespace + start of a new provision.
46
+ _TERM = re.compile(r"[.;:]\s+(?=[A-Z\"'(])")
47
+
48
+ # 0006-B: a leading enumeration marker to strip before deciding if a span is a bare HEADING ('9.', '(a)', '12.1').
49
+ _LEADING_ENUM = re.compile(r"^\s*(?:\(?[\dA-Za-z]{1,4}\s*[.)]|\d+(?:\.\d+){0,3}\.?|§+)\s+")
50
+
51
+ # issue 0014: a markdown TABLE row (a line whose first non-space character is a pipe). A run of >=2 such lines is
52
+ # a table block, kept as ONE atomic operative span -- the rows are meaningless without the header row, so a fee
53
+ # schedule / payment table must retrieve as a unit (header + all rows), never split by the paragraph/sentence cutter.
54
+ _TABLE_LINE = re.compile(r"^[ \t]*\|")
55
+ _MIN_TABLE_LINES = 2
56
+
57
+
58
+ def _table_block_ranges(body: str) -> list[tuple[int, int]]:
59
+ """issue 0014: the char ranges of contiguous markdown table blocks (>= `_MIN_TABLE_LINES` consecutive
60
+ pipe-led lines). Each range is atomic: `segment_clause` cuts at its edges and never inside it, so the whole
61
+ table is one operative span. Byte offsets tile the body (a range ends at the start of the first non-table
62
+ line, i.e. after the last row's newline)."""
63
+ ranges: list[tuple[int, int]] = []
64
+ run_start: int | None = None
65
+ run_lines = 0
66
+ pos = 0
67
+ for line in body.splitlines(keepends=True):
68
+ if _TABLE_LINE.match(line):
69
+ if run_start is None:
70
+ run_start, run_lines = pos, 0
71
+ run_lines += 1
72
+ else:
73
+ if run_start is not None and run_lines >= _MIN_TABLE_LINES:
74
+ ranges.append((run_start, pos))
75
+ run_start, run_lines = None, 0
76
+ pos += len(line)
77
+ if run_start is not None and run_lines >= _MIN_TABLE_LINES:
78
+ ranges.append((run_start, pos))
79
+ return ranges
80
+
81
+
82
+ _MIN_CLAUSE_ALPHA = 6 # fewer alphabetic chars than this = a page number / "By:" / "9" -- not a clause
83
+ _MIN_ALLCAPS_WORDS = 6 # a short ALL-CAPS span is a label/heading ("EXHIBIT C"); a long one may be a real clause
84
+ _WORD_RE = re.compile(r"[A-Za-z]{2,}")
85
+ # A signature / execution-block or notice-block label line -- universal, domain-neutral contract furniture
86
+ # ("By: /s/ ...", "Name:", "Title:", "Attest:", "Its:", "Date:"; and the notice-block contacts "Attention:",
87
+ # "Fax:", "Email:", "Telephone:"). The trailing ':' anchored right after the leading word makes it a LABEL, not a
88
+ # provision that merely mentions the word (e.g. "By signing below, the parties agree ..." and "All notices shall
89
+ # be sent to the following address:" both start with other words, so neither matches). issue 0036 adds the notice
90
+ # contacts; a bare street/city address line has no such label and stays recall-first (kept -> a thin clause).
91
+ _FURNITURE_LINE = re.compile(
92
+ r"^\s*(?:by|name|title|attest|witness|its|date|signature"
93
+ r"|attention|attn|facsimile|fax|e-?mail|telephone|tel|phone)\s*:", re.IGNORECASE)
94
+ # A table-of-contents entry: a dotted leader (>=3 dots, optionally spaced) running to a trailing page number.
95
+ # High-precision furniture; a decimal like "Section 3.1" or "(see Section 3.1)." has no long leader-to-page run.
96
+ _TOC_LEADER = re.compile(r"(?:\.\s*){3,}\d+\s*$")
97
+
98
+
99
+ def is_extractable_span(text: str) -> bool:
100
+ """EXTRACT-GUARD-1: whether a span is a CLAUSE worth attempting typed-property extraction on. RECALL-FIRST:
101
+ returns True for anything carrying lowercase prose (a real provision has function words -- 'the', 'shall',
102
+ 'of'); only DECLINES clear document FURNITURE that bears no clause properties and only burns a docling-graph
103
+ call + retries -- a page number ('9'), a docket reference, a short bare ALL-CAPS heading ('EXHIBIT C',
104
+ 'FORM OF SUBLICENSE', 'AMENDMENT OF DEFINITIONS.'), or a signature/execution-block label line ('By: /s/ ...',
105
+ 'Name:', 'Title:').
106
+
107
+ A skipped span is NOT a content loss: it stays in the SPAN INDEX for retrieval (indexing is independent of
108
+ clause extraction); the guard only declines to mint a clause-KG node from furniture. A long ALL-CAPS provision
109
+ (a capitalised disclaimer, `>= _MIN_ALLCAPS_WORDS` words) is still extracted."""
110
+ t = text.strip()
111
+ if sum(c.isalpha() for c in t) < _MIN_CLAUSE_ALPHA: # near-empty: page numbers, "By:", pure digits/punct
112
+ return False
113
+ if _FURNITURE_LINE.match(t): # a signature/execution-block or notice-block contact-label line, not a provision
114
+ return False
115
+ if _TOC_LEADER.search(t): # a table-of-contents dotted-leader-to-page-number line, not a provision
116
+ return False
117
+ if not any(c.islower() for c in t): # ALL-CAPS: a label/heading unless it is a long (capitalised) clause
118
+ return len(_WORD_RE.findall(t)) >= _MIN_ALLCAPS_WORDS
119
+ return True # has lowercase prose -> treat as a real clause (recall-first)
120
+
121
+
122
+ # issue 0038/0039: a span that STARTS a new numbered contract section -- the provision unit. DEPTH-CAPPED at
123
+ # TWO levels ('2.' or '2.1.'): a top-level or one-level-nested number opens a provision, but a DEEPER number
124
+ # ('10.5.1.', '10.5.1.1.') is a LIST ITEM within its parent provision and must FOLD IN, per 0038's own grouping
125
+ # rule (issue 0039: a naive marker/number regex over-splits at depth 3-4, trading one granularity bug for a
126
+ # smaller one). A parenthesised letter/roman item ('(a)', '(i)') is likewise not a section (no leading digit).
127
+ _SECTION_START = re.compile(r"^\(?\d{1,2}(?:\.\d{1,2})?\)?\.\s")
128
+ # A section-WORD prefix ('Section 8', 'Article 2', 'Clause 12', 'Sec.'/'Art.', '§3') before a number -- the dominant
129
+ # contract heading style. Stripped so the depth-capped number rule above decides exactly as for a bare '8.'. The
130
+ # digit lookahead means a prose line like 'Section hereof shall mean ...' is left untouched (not a start).
131
+ _SECTION_WORD = re.compile(r"^(?:§\s*|(?:section|article|clause|sec|art)\.?\s+)(?=\(?\d)", re.IGNORECASE)
132
+ # The number FOLLOWING a section word (depth-capped to two levels; trailing '.'/')' optional, since the section
133
+ # word already signals a heading): 'Section 8.' and 'Clause 12 Governing Law' both start a provision.
134
+ _SECTION_NUM = re.compile(r"^\(?\d{1,2}(?:\.\d{1,2})?\)?[.)]?(?:\s|$)")
135
+
136
+
137
+ def starts_new_provision(text: str) -> bool:
138
+ """issue 0038/0039: whether a span BEGINS a new provision, used to group contiguous spans into a provision for
139
+ clause extraction (retrieval stays per span). Tiered, deterministic:
140
+
141
+ - a span with a LEADING NUMBER is decided SOLELY by `_SECTION_START` (depth-capped to two levels): '2.' or
142
+ '2.1.' starts a provision; a deeper '10.5.1.'/'10.5.1.1.' is a list item and FOLDS IN (issue 0039 -- a
143
+ depth-blind rule over-splits nested list items into their own provisions);
144
+ - a span with NO leading number starts a provision if it is a bare Title-case heading or a short ALL-CAPS
145
+ heading (an un-numbered but headed contract).
146
+
147
+ When a document has none of these, no intra-chunk boundary fires and grouping falls back to the chunk (the
148
+ caller also breaks on a chunk change), so a heading-less contract degrades to chunk-level -- never one clause
149
+ per sentence, never per document."""
150
+ t = text.strip()
151
+ m = _SECTION_WORD.match(t)
152
+ if m: # an explicit 'Section/Article/Clause N' heading -> a start (depth-capped; trailing period optional)
153
+ return bool(_SECTION_NUM.match(t[m.end():]))
154
+ if re.match(r"^\(?\d", t): # a numbered item: ONLY the depth-capped section rule decides (no heading override)
155
+ return bool(_SECTION_START.match(t))
156
+ if _is_bare_heading(t): # un-numbered but titled ('Governing Law', a short Title-case line)
157
+ return True
158
+ if t and not any(c.islower() for c in t) and len(_WORD_RE.findall(t)) < _MIN_ALLCAPS_WORDS:
159
+ return True # a short ALL-CAPS heading ('CONFIDENTIALITY')
160
+ return False
161
+
162
+
163
+ def provision_boundary_verdict(text: str) -> str:
164
+ """Three-way boundary classification used to group spans into provisions: `"start"` (a confident, deterministic
165
+ provision start), `"continue"` (clearly provision body), or `"uncertain"` (a short, plausibly-heading line in a
166
+ style the deterministic rules do not confidently classify -- e.g. a roman-numeral or colon heading). Only the
167
+ `"uncertain"` residue is sent to a decision model (Jev) by `spans.boundary`; a deterministic `"start"`/
168
+ `"continue"` never pays for a model call, so cost stays bounded to the ambiguous SHORT lines -- and new heading
169
+ styles get a model's judgment instead of another regex (the flexibility regex alone cannot give)."""
170
+ t = text.strip()
171
+ if starts_new_provision(t):
172
+ return "start"
173
+ if _heading_candidate(t):
174
+ return "uncertain"
175
+ return "continue"
176
+
177
+
178
+ def _heading_candidate(text: str) -> bool:
179
+ """A short line that plausibly BEGINS a provision but was not a confident deterministic start: an enumerated
180
+ lead-in (number / roman numeral / letter) we did not confidently start, or a short capitalized title-like line
181
+ that is NOT a full sentence (no sentence punctuation). Deliberately broad -- the model decides -- but bounded to
182
+ SHORT lines, so ordinary body prose (long, lowercase-led, or a terminated sentence) is never a candidate."""
183
+ t = text.strip()
184
+ if not t or len(t) > 90 or t[0].islower():
185
+ return False # empty, long, or lowercase-led -> provision body, never a candidate
186
+ if _LEADING_ENUM.match(t):
187
+ return True # an enumerated lead-in the deterministic rules did not confidently start
188
+ return not (re.search(r"\.\s", t) or t.endswith(".")) # short + capitalized + NOT a full sentence -> a heading candidate
189
+
190
+
191
+ def _is_bare_heading(text: str) -> bool:
192
+ """A bare SECTION HEADING (e.g. '9. Limitation of Liability') -- a short enumerated/Title-case line with NO
193
+ sentence terminator. It must fold INTO its body, never stand alone: a standalone heading gets classified as
194
+ a clause pointing at a bare heading, which pollutes evidence and can hide the real clause (issue 0006). A
195
+ genuine short provision carries an operative sentence (terminal '.'/';'/':'), so it is NOT a heading."""
196
+ t = text.strip()
197
+ if not t or len(t) > 60:
198
+ return False
199
+ rest = _LEADING_ENUM.sub("", t, count=1) # drop a leading '9.' / '(a)' / '12.1' enumeration marker
200
+ if not rest or not rest[0].isupper(): # a heading's title starts capitalised
201
+ return False
202
+ return not re.search(r"[.;:]", rest) # a bare title has no sentence punctuation; a provision does
203
+
204
+
205
+ class OperativeSpan(BaseModel):
206
+ """One operative span of a clause, pointing back to its parent clause (FR-R small-to-big unit)."""
207
+
208
+ span_id: str # "{parent_chunk_id}#{span_index}" -- embeds the parent (identifier rule, ADR-0025)
209
+ parent_chunk_id: str
210
+ parent_okf_path: str # where the parent clause lives in the clause OKF bundle (locate/fetch for rerank)
211
+ span_index: int
212
+ start: int # char offset into the parent clause body
213
+ end: int # exclusive; spans tile the body: body[start:end] concatenated == body
214
+ text: str # body[start:end] (raw slice; strip at use time)
215
+ pages: list[int] = [] # issue 0032: the source page(s) this span's canonical range overlaps (set at ingest)
216
+ bbox: tuple[float, float, float, float] | None = None # best-effort single-item box (l, t, r, b)
217
+
218
+
219
+ def _boundaries(body: str) -> list[int]:
220
+ """Deterministic cut offsets that partition `body` into operative spans (includes 0 and len(body)).
221
+
222
+ A markdown table block (issue 0014) is atomic: cuts are forced at its edges and every candidate cut INSIDE
223
+ it is suppressed, so the whole table stays one span (header + rows)."""
224
+ blocks = _table_block_ranges(body)
225
+
226
+ def _inside(c: int) -> bool: # strictly inside a table block -> not a valid cut
227
+ return any(bs < c < be for bs, be in blocks)
228
+
229
+ cuts: set[int] = {0, len(body)}
230
+ for bs, be in blocks: # a table block's edges are always boundaries
231
+ cuts.add(bs)
232
+ cuts.add(be)
233
+ for m in re.finditer(r"\n+", body): # paragraph breaks
234
+ if not _inside(m.end()):
235
+ cuts.add(m.end())
236
+ for m in _ENUM.finditer(body): # split BEFORE an enumeration marker opening a provision
237
+ if not _inside(m.start(1)):
238
+ cuts.add(m.start(1))
239
+ for m in _TERM.finditer(body): # split AFTER a genuine sentence/list terminator
240
+ p = m.start() # index of the '.' ';' or ':'
241
+ if _inside(m.end()):
242
+ continue
243
+ if body[p] == ".":
244
+ prev = body[p - 1] if p > 0 else ""
245
+ if prev.isdigit(): # decimal / section reference (12.1) -- not a terminator
246
+ continue
247
+ word = re.search(r"([A-Za-z.]+)$", body[max(0, p - 8):p])
248
+ if word and word.group(1).lower().strip(".") in _ABBREV: # known abbreviation
249
+ continue
250
+ cuts.add(m.end())
251
+ return sorted(cuts)
252
+
253
+
254
+ def _merge_subfloor(ranges: list[tuple[int, int]], body: str, min_chars: int) -> list[tuple[int, int]]:
255
+ """Fold a sub-floor fragment (bare heading/marker) into the FOLLOWING span; a trailing one into the previous.
256
+ Merges only extend adjacent ranges, so the tiling invariant (contiguous, gap-free) is preserved."""
257
+ merged: list[tuple[int, int]] = []
258
+ carry: int | None = None
259
+ for i, (s, e) in enumerate(ranges):
260
+ start = carry if carry is not None else s
261
+ is_last = i == len(ranges) - 1
262
+ # fold FORWARD a sub-floor fragment OR a bare heading (0006-B) -- so a heading never stands alone
263
+ if (len(body[start:e].strip()) < min_chars or _is_bare_heading(body[start:e])) and not is_last:
264
+ carry = start # carry its start into the next span (its body)
265
+ continue
266
+ merged.append((start, e))
267
+ carry = None
268
+ if carry is not None: # trailing sub-floor: extend the previous span to the end
269
+ last_end = ranges[-1][1]
270
+ if merged:
271
+ merged[-1] = (merged[-1][0], last_end)
272
+ else:
273
+ merged.append((carry, last_end))
274
+ return merged
275
+
276
+
277
+ def segment_clause(
278
+ parent_chunk_id: str,
279
+ body: str,
280
+ *,
281
+ parent_okf_path: str = "",
282
+ min_chars: int = DEFAULT_MIN_CHARS,
283
+ ) -> list[OperativeSpan]:
284
+ """Segment a clause body into operative spans. Deterministic; spans tile the body byte-faithfully."""
285
+ if not body:
286
+ return []
287
+ cuts = _boundaries(body)
288
+ ranges = [(cuts[i], cuts[i + 1]) for i in range(len(cuts) - 1)]
289
+ ranges = _merge_subfloor(ranges, body, min_chars)
290
+ return [
291
+ OperativeSpan(
292
+ span_id=f"{parent_chunk_id}#{i}",
293
+ parent_chunk_id=parent_chunk_id,
294
+ parent_okf_path=parent_okf_path,
295
+ span_index=i,
296
+ start=s,
297
+ end=e,
298
+ text=body[s:e],
299
+ )
300
+ for i, (s, e) in enumerate(ranges)
301
+ ]
302
+
303
+
304
+ def to_span_record(
305
+ op: OperativeSpan,
306
+ *,
307
+ contract_id: str,
308
+ chunk_doc_start: int,
309
+ dense_vector: list[float],
310
+ sparse_vector: dict[int, float],
311
+ function: str = "",
312
+ functions: list[str] | None = None,
313
+ parent_okf_path: str | None = None,
314
+ ) -> SpanRecord:
315
+ """CU-B2 (ADR-0029): OperativeSpan -> SpanRecord with DOCUMENT-ABSOLUTE offsets.
316
+
317
+ Composes `doc_start = chunk_doc_start + op.start`, `doc_end = chunk_doc_start + op.end` (the span's
318
+ clause-relative offsets shifted by the parent chunk's offset in the canonical document text, CU-B1). The
319
+ RAW span text (`op.text = body[start:end]`) is stored -- NOT stripped -- so the citation invariant
320
+ `canonical_document_text[doc_start:doc_end] == span.text` holds byte-faithfully. The caller may embed over
321
+ `op.text.strip()`; the stored text stays raw for the highlight.
322
+ """
323
+ return SpanRecord(
324
+ span_id=op.span_id,
325
+ parent_chunk_id=op.parent_chunk_id,
326
+ parent_okf_path=op.parent_okf_path if parent_okf_path is None else parent_okf_path,
327
+ span_index=op.span_index,
328
+ text=op.text,
329
+ function=function,
330
+ functions=list(functions) if functions else ([function] if function else []),
331
+ dense_vector=dense_vector,
332
+ sparse_vector=sparse_vector,
333
+ contract_id=contract_id,
334
+ doc_start=chunk_doc_start + op.start,
335
+ doc_end=chunk_doc_start + op.end,
336
+ page=(op.pages[0] if op.pages else None), # issue 0032: FIRST page for the singular highlight field
337
+ pages=list(op.pages), # ALL pages the span overlaps (cross-page clause -> a list)
338
+ bbox=op.bbox, # best-effort single-item box
339
+ )
340
+
341
+
@@ -0,0 +1,197 @@
1
+ """JUDGE-SEMANTIC (ADR-0040), SKILL-SPLIT: Layer 3 of the neuro-symbolic extraction-fidelity cascade, the ONLY
2
+ place an LLM is spent on judging -- split into a SKILL + a deterministic FUNCTION.
3
+
4
+ Per the capability-architecture principle (a `function` is deterministic and takes no model; a single LLM act is
5
+ an authored `agent_skill`; a workflow is a `subgraph`), the semantic judge is two capabilities:
6
+
7
+ - **`extraction_semantic_judge` (agent_skill)** -- the verify-or-refute reading METHOD, authored as
8
+ `skills/extraction_semantic_judge/SKILL.md` and applied through the model seam (product = Granite, ADR-0039).
9
+ Given one property (`dimension = value`, with its meaning) and the clause text, it returns a raw
10
+ `SemanticVerdict` (supported / reason). `build_semantic_judge_fn` is its runtime; `structured_factory` is
11
+ injected for hermetic tests.
12
+ - **`extraction_semantic_gate` (function)** -- `semantic_judge`: DETERMINISTIC, no model. Selects the surviving
13
+ (non-AMBIGUOUS) assertions on a SEMANTIC dimension (`SEMANTIC_DIMENSIONS`), runs the skill over each
14
+ concurrently, and downgrades a refuted one to AMBIGUOUS (kept but flagged), exactly like `reground` /
15
+ `symbolic_validate`. A judge that fails/returns None leaves the assertion untouched (never downgrade on a
16
+ judge error). Ingestion-side only; NOT on queries (KG-5d: judging the query text false-flags real constraints).
17
+
18
+ Layers 1-2 (lexical grounding + symbolic SHACL) already cleared the type / cardinality / deontic / textual-anchor
19
+ errors deterministically, so this LLM call class is small and focused: the closed SEMANTIC dimensions whose value
20
+ is a READING with no surface form (`mutuality=mutual` vs `unilateral`, `favorability`, `cap_basis`, ...).
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import asyncio
26
+ from pathlib import Path
27
+ from typing import Any, Callable, Optional
28
+
29
+ from pydantic import BaseModel
30
+
31
+ from rag_wright.contracts.property import CLOSED_VOCAB, ClausePropertyRecord, PropertyDimension
32
+ from rag_wright.contracts.provenance import ConfidenceTag
33
+ from rag_wright.models.tag_structured import build_tag_structured
34
+ from rag_wright.spans.property_grounding import GROUNDING_CUES
35
+ from rag_wright.util.concurrent import map_concurrent
36
+
37
+ _SKILL_PATH = Path(__file__).parents[1] / "skills" / "extraction_semantic_judge" / "SKILL.md"
38
+
39
+ # The closed SEMANTIC dimensions: a closed vocabulary (so not open-valued) with NO lexical cue (so not
40
+ # checkable by the grounding judge) -- their value is a reading, not a surface token. This is exactly the
41
+ # residual Layers 1-2 cannot reach.
42
+ SEMANTIC_DIMENSIONS: frozenset[PropertyDimension] = frozenset(
43
+ d for d in CLOSED_VOCAB if d not in GROUNDING_CUES
44
+ )
45
+
46
+ # A short plain-language gloss so the judge understands what each (dimension, value) claims about the clause.
47
+ _DIMENSION_GLOSS: dict[PropertyDimension, str] = {
48
+ PropertyDimension.MUTUALITY: "whether the obligation runs BOTH ways (mutual) or only one party's (unilateral)",
49
+ PropertyDimension.FAVORABILITY: "which side the term favors (buyer_favorable / seller_favorable)",
50
+ PropertyDimension.PARTY_ASYMMETRY: "whether the terms are the same for both parties (symmetric) or differ per party",
51
+ PropertyDimension.CAP_BASIS: "the shape of the liability cap (a fixed_fee amount vs a multiple_of_fees)",
52
+ PropertyDimension.LAW_MULTIPLICITY: "whether ONE governing law applies (single) or more than one (multiple)",
53
+ PropertyDimension.IP_OWNERSHIP: "who owns the IP (assigned to one party / joint / retained by the originator)",
54
+ PropertyDimension.NONSOLICIT_TARGET: "who may not be solicited (employees / customers)",
55
+ PropertyDimension.RENEWAL_MECHANISM: "how the term renews (auto-renews / requires_notice to renew)",
56
+ PropertyDimension.COC_CONSENT: "how a change of control is treated (consent_required / notice_only / unrestricted)",
57
+ PropertyDimension.ASSIGNMENT_CONSENT: "how assignment is treated (consent_required / notice_only / free)",
58
+ PropertyDimension.MFN_SCOPE: "what a most-favored-nation term covers (price / terms / price_and_terms)",
59
+ PropertyDimension.TERMINATION_RIGHT: "who may terminate for convenience (either_party / one_party)",
60
+ }
61
+
62
+ # The per-call appendix bound onto the SKILL method (the static method teaches the reading; the specific
63
+ # property + clause are appended at call time, the compliance_judgment `_PROMPT_TAIL` pattern).
64
+ _PROMPT_TAIL = "\n\nProperty: {dimension} = {value}\nMeaning: {gloss}\n\nClause:\n{clause}"
65
+
66
+
67
+ class SemanticVerdict(BaseModel):
68
+ """The judge's ruling on one semantic assertion: is the reading supported by the clause text?"""
69
+
70
+ supported: bool
71
+ reason: str = ""
72
+
73
+
74
+ # judge_fn: (dimension, value, clause_text) -> verdict, or None if the judge could not rule (left untouched).
75
+ JudgeFn = Callable[[PropertyDimension, str, str], Optional[SemanticVerdict]]
76
+
77
+
78
+ def judgment_method() -> str:
79
+ """The semantic-judge method (the `extraction_semantic_judge` SKILL body, YAML frontmatter stripped) used as
80
+ the judge's system/method prompt. Authored knowledge (skills/extraction_semantic_judge/SKILL.md)."""
81
+ text = _SKILL_PATH.read_text(encoding="utf-8")
82
+ if text.startswith("---"):
83
+ marker = text.find("\n---", 3)
84
+ if marker != -1:
85
+ text = text[marker + 4 :]
86
+ return text.strip()
87
+
88
+
89
+ def _judge_prompt(method: str, dimension: PropertyDimension, value: str, text: str) -> str:
90
+ return method + _PROMPT_TAIL.format(
91
+ dimension=dimension.value, value=value,
92
+ gloss=_DIMENSION_GLOSS.get(dimension, dimension.value), clause=text)
93
+
94
+
95
+ def build_semantic_judge_fn(model_id: str, *, structured_factory=build_tag_structured) -> JudgeFn:
96
+ """The `extraction_semantic_judge` SKILL's runtime: a Granite-backed verify-or-refute `JudgeFn` through the
97
+ model seam (model-neutral; the product points the seam at self-hosted vLLM-Granite). The SKILL.md method is
98
+ the system prompt; the specific property + clause are appended. `structured_factory` is injected for tests."""
99
+ method = judgment_method()
100
+
101
+ def judge(dimension: PropertyDimension, value: str, text: str) -> Optional[SemanticVerdict]:
102
+ return structured_factory(model_id, SemanticVerdict, label="semantic_judge.judge").invoke(
103
+ _judge_prompt(method, dimension, value, text))
104
+
105
+ return judge
106
+
107
+
108
+ def build_asemantic_judge_fn(model_id: str, *, structured_factory=build_tag_structured):
109
+ """ASYNC-B2b (ADR-0057): the async twin of `build_semantic_judge_fn` -- the judge call on the async seam
110
+ (`.ainvoke`, true wall-clock deadline)."""
111
+ method = judgment_method()
112
+
113
+ async def ajudge(dimension: PropertyDimension, value: str, text: str) -> Optional[SemanticVerdict]:
114
+ return await structured_factory(model_id, SemanticVerdict, label="semantic_judge.judge").ainvoke(
115
+ _judge_prompt(method, dimension, value, text))
116
+
117
+ return ajudge
118
+
119
+
120
+ def semantic_judge(
121
+ record: ClausePropertyRecord, text: str, judge_fn: JudgeFn, *, max_concurrency: int = 8
122
+ ) -> ClausePropertyRecord:
123
+ """`extraction_semantic_gate` (FUNCTION -- deterministic, no model): the ADR-0040 Layer-3 quality gate. Apply
124
+ the `extraction_semantic_judge` SKILL (`judge_fn`) to every surviving (non-AMBIGUOUS) assertion on a SEMANTIC
125
+ dimension; downgrade a refuted one to AMBIGUOUS. A no-op when there is nothing semantic to judge. The judge
126
+ calls run concurrently (async + semaphore, per the parallel-LLM rule). A None verdict is a judge failure ->
127
+ the assertion is left untouched (never downgrade on a judge error). The model lives in the SKILL, not here."""
128
+ targets = [
129
+ a for a in record.assertions
130
+ if a.dimension in SEMANTIC_DIMENSIONS and a.confidence != ConfidenceTag.AMBIGUOUS
131
+ ]
132
+ if not targets:
133
+ return record
134
+ verdicts = map_concurrent(
135
+ targets, lambda a: judge_fn(a.dimension, a.value, text), max_concurrency=max_concurrency
136
+ )
137
+ return _apply_verdicts(record, targets, verdicts)
138
+
139
+
140
+ def _apply_verdicts(record: ClausePropertyRecord, targets: list, verdicts: list) -> ClausePropertyRecord:
141
+ """Downgrade to AMBIGUOUS every target assertion whose verdict refuted it; a None verdict (judge failure)
142
+ leaves the assertion untouched. Shared by the sync and async gates."""
143
+ refuted = {id(a) for a, v in zip(targets, verdicts) if v is not None and not v.supported}
144
+ if not refuted:
145
+ return record
146
+ new = [
147
+ a.model_copy(update={"confidence": ConfidenceTag.AMBIGUOUS}) if id(a) in refuted else a
148
+ for a in record.assertions
149
+ ]
150
+ return record.model_copy(update={"assertions": new})
151
+
152
+
153
+ def _semantic_targets(record: ClausePropertyRecord) -> list:
154
+ return [a for a in record.assertions
155
+ if a.dimension in SEMANTIC_DIMENSIONS and a.confidence != ConfidenceTag.AMBIGUOUS]
156
+
157
+
158
+ async def asemantic_judge(
159
+ record: ClausePropertyRecord, text: str, ajudge_fn: Any, *, max_concurrency: int = 8
160
+ ) -> ClausePropertyRecord:
161
+ """ASYNC-B2b (ADR-0057): the async twin of `semantic_judge`. Judges each surviving semantic assertion via the
162
+ async judge, concurrently, bounded by a semaphore (native form of the parallel-LLM rule); each call carries
163
+ the true wall-clock deadline. A None verdict leaves the assertion untouched (never downgrade on a failure)."""
164
+ targets = _semantic_targets(record)
165
+ if not targets:
166
+ return record
167
+ sem = asyncio.Semaphore(max_concurrency)
168
+
169
+ async def _one(a: Any) -> Any:
170
+ async with sem:
171
+ return await ajudge_fn(a.dimension, a.value, text)
172
+
173
+ verdicts = list(await asyncio.gather(*(_one(a) for a in targets)))
174
+ return _apply_verdicts(record, targets, verdicts)
175
+
176
+
177
+ def register_extraction_semantic_judge(registry) -> None:
178
+ """Register `extraction_semantic_judge` as an AGENT_SKILL (ADR-0040 Layer 3): a single grounded LLM
179
+ verify-or-refute reading, authored as `skills/extraction_semantic_judge/SKILL.md` and applied via the seam.
180
+ Typed output = `SemanticVerdict`."""
181
+ registry.register(
182
+ "extraction_semantic_judge",
183
+ contract=SemanticVerdict,
184
+ kind="agent_skill",
185
+ display_name="Extraction semantic judge (clause property -> supported?; authored skill)",
186
+ )
187
+
188
+
189
+ def register_extraction_semantic_gate(registry) -> None:
190
+ """Register `extraction_semantic_gate` (FUNCTION -- deterministic): apply the `extraction_semantic_judge`
191
+ SKILL over each surviving semantic assertion and downgrade a refuted one to AMBIGUOUS. No model."""
192
+ registry.register(
193
+ "extraction_semantic_gate",
194
+ contract=ClausePropertyRecord,
195
+ kind="function",
196
+ display_name="Extraction semantic gate (semantic-dimension AMBIGUOUS downgrade)",
197
+ )