ragpreflight 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,52 @@
1
+ """ragpreflight — Pre-ingestion RAG document quality audit.
2
+
3
+ Grounded in Garani 2026 (doi:10.18653/v1/2026.trustnlp-main.27).
4
+ Zero API keys required. Everything runs locally.
5
+
6
+ Quick start:
7
+ from ragpreflight import scan_document, audit_corpus, analyze_chunks
8
+
9
+ report = scan_document("document.pdf")
10
+ print(report.score) # 0-100
11
+ print(report.issues) # list[Issue]
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ __version__ = "0.1.0"
17
+ __author__ = "Anupama Garani"
18
+ __license__ = "MIT"
19
+
20
+ from ragpreflight.models import (
21
+ CorpusReport,
22
+ ChunkReport,
23
+ DocumentReport,
24
+ Issue,
25
+ IssueCategory,
26
+ Severity,
27
+ )
28
+ from ragpreflight.scanner import scan_document
29
+ from ragpreflight.chunker import analyze_chunks
30
+ from ragpreflight.corpus import audit_corpus
31
+ from ragpreflight.profiles import get_profile, PROFILES
32
+ from ragpreflight.config import load_config
33
+
34
+ __all__ = [
35
+ "__version__",
36
+ # Data models
37
+ "DocumentReport",
38
+ "ChunkReport",
39
+ "CorpusReport",
40
+ "Issue",
41
+ "IssueCategory",
42
+ "Severity",
43
+ # Core functions
44
+ "scan_document",
45
+ "analyze_chunks",
46
+ "audit_corpus",
47
+ # Profiles
48
+ "get_profile",
49
+ "PROFILES",
50
+ # Config
51
+ "load_config",
52
+ ]
@@ -0,0 +1,284 @@
1
+ """Constants, patterns, and default thresholds for ragpreflight."""
2
+
3
+ from __future__ import annotations
4
+
5
+ # ---------------------------------------------------------------------------
6
+ # OCR error detection patterns
7
+ # ---------------------------------------------------------------------------
8
+
9
+ OCR_SUBSTITUTION_PATTERNS: dict[str, str | None] = {
10
+ # Character confusions
11
+ "l_I_1": r"(?<=[a-z])[I1](?=[a-z])", # "cIinical" → "clinical"
12
+ "O_0": r"(?<=[a-z])0(?=[a-z])", # "pr0tocol" → "protocol"
13
+ "rn_m": r"(?<=[a-z])rn(?=[a-z])", # "inforrnation" → "information"
14
+ "fi_fl_ligature": r"[fifl]", # Ligature artifacts
15
+ "broken_ligatures": r"(?<=\w)[ffi](?=\w)", # Split ligatures
16
+ # Spacing artifacts
17
+ "mid_word_spaces": r"(?<=[a-z]) (?=[a-z]{2,})", # "pati ent" → "patient"
18
+ # Encoding artifacts
19
+ "mojibake": r"[ÃÂâéé]", # UTF-8 decoded as Latin-1
20
+ # Common OCR numeral/letter confusions in context
21
+ "zero_as_o": r"\b0[a-z]+\b", # "0ne" instead of "one"
22
+ "merged_words": None, # Detected via dictionary lookup, not regex
23
+ }
24
+
25
+ # Human-readable descriptions for each OCR pattern, shown in issue output.
26
+ # Each entry: (label, what_it_is, rag_impact)
27
+ OCR_PATTERN_DESCRIPTIONS: dict[str, tuple[str, str, str]] = {
28
+ "l_I_1": (
29
+ "l / I / 1 confusion",
30
+ "The scanner misread lowercase l, capital I, and digit 1 as each other "
31
+ "(e.g. 'cIinical' instead of 'clinical', '1iver' instead of 'liver').",
32
+ "Keyword and semantic search both fail — a user querying 'clinical' won't "
33
+ "match 'cIinical' in the vector index.",
34
+ ),
35
+ "O_0": (
36
+ "O / 0 confusion",
37
+ "The letter O was substituted with the digit 0 "
38
+ "(e.g. 'pr0tocol' instead of 'protocol').",
39
+ "Breaks exact-match retrieval and degrades embedding quality for technical terms.",
40
+ ),
41
+ "rn_m": (
42
+ "rn / m split",
43
+ "The letter m was split into rn by the scanner "
44
+ "(e.g. 'inforrnation' instead of 'information').",
45
+ "Common in low-DPI scans. Misspelled tokens create retrieval gaps — "
46
+ "the LLM may still understand the text but retrieval ranking degrades.",
47
+ ),
48
+ "fi_fl_ligature": (
49
+ "fi / fl ligature characters",
50
+ "Unicode ligature characters fi (fi) and fl (fl) survived extraction. "
51
+ "These are single characters, not two letters, so 'flight' ≠ 'flight'.",
52
+ "Users searching 'flight', 'financial', or 'flexible' won't match "
53
+ "chunks containing the ligature forms. Affects millions of typeset PDFs.",
54
+ ),
55
+ "broken_ligatures": (
56
+ "broken ligatures",
57
+ "Ligature characters were split mid-word during extraction, "
58
+ "inserting phantom letter combinations into words.",
59
+ "Creates misspelled tokens that fragment embedding representation "
60
+ "and fail keyword search.",
61
+ ),
62
+ "mid_word_spaces": (
63
+ "mid-word spaces",
64
+ "Spaces were inserted inside words during scanning "
65
+ "(e.g. 'pati ent' instead of 'patient', 'treat ment' instead of 'treatment').",
66
+ "Each broken word becomes two meaningless tokens. Semantic embeddings "
67
+ "for entire sentences degrade, and exact-match search fails completely.",
68
+ ),
69
+ "mojibake": (
70
+ "mojibake (encoding corruption)",
71
+ "UTF-8 text was decoded as Latin-1 or Windows-1252, producing garbled "
72
+ "characters like é, £, ’ instead of é, £, '.",
73
+ "Garbled characters corrupt every sentence they appear in. "
74
+ "The LLM may misinterpret or hallucinate around them.",
75
+ ),
76
+ "zero_as_o": (
77
+ "digit-as-letter substitution",
78
+ "The digit 0 was used where the letter o was intended at word boundaries "
79
+ "(e.g. '0ne' instead of 'one', '0ver' instead of 'over').",
80
+ "Common in scanned documents. Breaks numeric-aware retrieval and "
81
+ "creates unknown tokens in the embedding model's vocabulary.",
82
+ ),
83
+ }
84
+
85
+ # F-code mapping: IssueCategory value → list of (mode_id, relationship, explanation)
86
+ # Applied post-scan to populate taxonomy_refs on all issues automatically.
87
+ ISSUE_CATEGORY_TO_FCODE: dict[str, list[tuple[str, str, str]]] = {
88
+ "ocr": [
89
+ ("F3", "direct",
90
+ "OCR artifacts directly cause Document Quality failures (F3): garbled tokens "
91
+ "degrade embedding quality and break keyword retrieval (Garani 2026, §Ingestion)."),
92
+ ],
93
+ "encoding": [
94
+ ("F3", "direct",
95
+ "Encoding corruption produces garbled text that directly causes Document Quality "
96
+ "failures (F3) — embeddings for corrupted sentences are unreliable."),
97
+ ],
98
+ "content": [
99
+ ("F3", "proxy",
100
+ "Low content density or empty pages are a proxy for Layout Parsing Errors (F3). "
101
+ "Pages that yield no extractable text indicate failed parsing — embeddings for "
102
+ "blank or garbled chunks degrade corpus-wide retrieval recall."),
103
+ ],
104
+ "structure": [
105
+ ("F3", "proxy",
106
+ "Detected tables are a proxy for Layout Parsing Errors (F3) — heterogeneous "
107
+ "layouts resist uniform text extraction."),
108
+ ("F7", "risk_signal",
109
+ "Tables and structured elements are a risk signal for Chunking Boundary Errors (F7). "
110
+ "A chunker that does not understand table structure will split rows mid-cell, "
111
+ "producing incoherent chunks that hurt retrieval precision."),
112
+ ],
113
+ "metadata": [
114
+ ("F11", "risk_signal",
115
+ "Sparse or missing metadata is a risk signal for Low Recall / Ranking Failures (F11). "
116
+ "Without title, author, or date, retrieval systems cannot filter or re-rank by source "
117
+ "quality — relevant documents are harder to surface above the top-k cutoff."),
118
+ ],
119
+ "chunking": [
120
+ ("F7", "direct",
121
+ "Chunking boundary issues directly cause Structure-Unaware Chunking failures (F7): "
122
+ "mid-sentence cuts and table splits degrade chunk coherence and retrieval precision."),
123
+ ],
124
+ "duplication": [
125
+ ("F11", "proxy",
126
+ "Near-duplicate documents are a proxy for Redundant/Duplicate Context (F11). "
127
+ "Duplicate chunks inflate context windows and dilute the relevant signal."),
128
+ ],
129
+ "staleness": [
130
+ ("F1", "proxy",
131
+ "File-age detection is a proxy for Outdated/Stale Data (F1). "
132
+ "This does NOT confirm content is outdated — only that the file is old. "
133
+ "Human review required."),
134
+ ],
135
+ }
136
+
137
+ # Minimum word length to be considered a "real" word for density calculations
138
+ MIN_WORD_LENGTH = 2
139
+
140
+ # Text considered "empty" if fewer than this many words
141
+ EMPTY_PAGE_WORD_THRESHOLD = 10
142
+
143
+ # File size limits
144
+ DEFAULT_MAX_FILE_SIZE_MB = 100
145
+ HARD_MAX_FILE_SIZE_MB = 500
146
+
147
+ # Bytes per MB
148
+ BYTES_PER_MB = 1024 * 1024
149
+
150
+ # ---------------------------------------------------------------------------
151
+ # Score weights for DocumentReport computation
152
+ # ---------------------------------------------------------------------------
153
+
154
+ SCORE_WEIGHTS = {
155
+ "text_extractability": 0.30,
156
+ "ocr_cleanliness": 0.25,
157
+ "structural_integrity": 0.20,
158
+ "metadata_completeness": 0.10,
159
+ "content_density": 0.15,
160
+ }
161
+
162
+ # ---------------------------------------------------------------------------
163
+ # Encoding detection
164
+ # ---------------------------------------------------------------------------
165
+
166
+ MIN_ENCODING_CONFIDENCE = 0.8 # charset-normalizer confidence below this → flag
167
+
168
+ # ---------------------------------------------------------------------------
169
+ # Common control characters (exclude normal ones like \n, \r, \t)
170
+ # ---------------------------------------------------------------------------
171
+
172
+ CONTROL_CHAR_PATTERN = r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f-\x9f]"
173
+
174
+ # ---------------------------------------------------------------------------
175
+ # Boilerplate detection for HTML
176
+ # ---------------------------------------------------------------------------
177
+
178
+ HTML_BOILERPLATE_TAGS = {
179
+ "nav", "footer", "header", "aside", "script", "style", "noscript",
180
+ "advertisement", "cookie-banner",
181
+ }
182
+
183
+ # ---------------------------------------------------------------------------
184
+ # Metadata fields we look for
185
+ # ---------------------------------------------------------------------------
186
+
187
+ PDF_METADATA_FIELDS = ["title", "author", "subject", "keywords", "creationdate", "creator"]
188
+ DOCX_METADATA_FIELDS = ["title", "author", "subject", "keywords", "created", "modified"]
189
+
190
+ # ---------------------------------------------------------------------------
191
+ # Supported file extensions
192
+ # ---------------------------------------------------------------------------
193
+
194
+ SUPPORTED_EXTENSIONS = {
195
+ ".pdf", ".docx", ".txt", ".csv", ".tsv", ".html", ".htm", ".md", ".markdown",
196
+ ".pptx", ".xlsx", ".xls", ".ipynb", ".srt", ".vtt",
197
+ }
198
+
199
+ # ---------------------------------------------------------------------------
200
+ # PII detection patterns
201
+ # ---------------------------------------------------------------------------
202
+
203
+ PII_PATTERNS: dict[str, str] = {
204
+ "email": r"\b[A-Za-z0-9._%+\-]+@[A-Za-z0-9.\-]+\.[A-Za-z]{2,}\b",
205
+ "us_ssn": r"\b\d{3}-\d{2}-\d{4}\b",
206
+ "us_phone": r"\b(?:\+1[\s.-]?)?\(?\d{3}\)?[\s.-]?\d{3}[\s.-]?\d{4}\b",
207
+ "credit_card": r"\b(?:\d{4}[\s\-]?){3}\d{4}\b",
208
+ "ip_address": r"\b\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}\b",
209
+ }
210
+
211
+ # Minimum PII hits before raising a WARNING (avoids false-positive noise)
212
+ PII_MIN_HITS = 3
213
+
214
+ # ---------------------------------------------------------------------------
215
+ # Formula / equation detection patterns (PDF text artifacts)
216
+ # ---------------------------------------------------------------------------
217
+
218
+ FORMULA_PATTERNS = [
219
+ r"[αβγδεζηθικλμνξπρστυφχψωΑΒΓΔΕΖΗΘΙΚΛΜΝΞΠΡΣΤΥΦΧΨΩ]", # Greek letters
220
+ r"\b(?:sin|cos|tan|cot|log|ln|exp|sqrt|lim|sum|prod|int)\s*[\(\[]", # Math functions
221
+ r"[∑∏∫∂∇∞≤≥≠≈±×÷√∝∈∉⊂⊃∪∩]", # Unicode math symbols
222
+ r"\$[^$\n]{2,80}\$", # LaTeX inline math $...$
223
+ r"\\\[.{2,200}\\\]", # LaTeX display math \[...\]
224
+ r"\b[a-zA-Z]\s*=\s*[-+]?\d*\.?\d+\s*[+\-*/^]", # Simple assignments: x = 3 +
225
+ ]
226
+
227
+ # Minimum formula pattern hits to flag a document as math-heavy
228
+ FORMULA_MIN_HITS = 5
229
+
230
+ # ---------------------------------------------------------------------------
231
+ # Language detection
232
+ # ---------------------------------------------------------------------------
233
+
234
+ LANGUAGE_DETECTION_MIN_CHARS = 200 # Skip language detection on very short texts
235
+ LANGUAGE_DETECTION_SAMPLE_CHARS = 5000 # Use first N chars for language detection
236
+
237
+ # ---------------------------------------------------------------------------
238
+ # Chunking defaults
239
+ # ---------------------------------------------------------------------------
240
+
241
+ DEFAULT_CHUNK_SIZE = 512
242
+ DEFAULT_CHUNK_OVERLAP = 50
243
+
244
+ # Structural break markers inside chunks
245
+ STRUCTURAL_BREAK_PATTERNS = [
246
+ r"\|\s*[-:]+\s*\|", # Table row boundary
247
+ r"^\s*[-*+]\s", # List item start
248
+ r"^#{1,6}\s", # Markdown header
249
+ r"```", # Code block fence
250
+ ]
251
+
252
+ # ---------------------------------------------------------------------------
253
+ # Temporal staleness default (months)
254
+ # ---------------------------------------------------------------------------
255
+
256
+ DEFAULT_STALE_MONTHS = 12
257
+
258
+ # ---------------------------------------------------------------------------
259
+ # Retrieval simulation
260
+ # ---------------------------------------------------------------------------
261
+
262
+ DEFAULT_RETRIEVAL_TOP_K = 5
263
+ DEFAULT_QUERY_FAILURE_THRESHOLD = 0.5 # cosine similarity below this → failed query
264
+ DEFAULT_QUERIES_PER_DOC = 7
265
+
266
+ QUERY_TEMPLATES = [
267
+ "What is {entity}?",
268
+ "Explain {concept}.",
269
+ "Summarize {section_title}.",
270
+ "What are the details of {key_phrase}?",
271
+ "How does {entity} work?",
272
+ "What is the purpose of {key_phrase}?",
273
+ "Describe {concept} in detail.",
274
+ ]
275
+
276
+ # ---------------------------------------------------------------------------
277
+ # Report colours (rich markup)
278
+ # ---------------------------------------------------------------------------
279
+
280
+ SEVERITY_COLOURS = {
281
+ "critical": "bold red",
282
+ "warning": "yellow",
283
+ "info": "cyan",
284
+ }