ragpreflight 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. ragpreflight-0.1.0/.github/workflows/ci.yml +64 -0
  2. ragpreflight-0.1.0/.gitignore +44 -0
  3. ragpreflight-0.1.0/.pre-commit-hooks.yaml +19 -0
  4. ragpreflight-0.1.0/CHANGELOG.md +70 -0
  5. ragpreflight-0.1.0/CITATION.cff +34 -0
  6. ragpreflight-0.1.0/CLAUDE.md +354 -0
  7. ragpreflight-0.1.0/CONTRIBUTING.md +79 -0
  8. ragpreflight-0.1.0/LICENSE +21 -0
  9. ragpreflight-0.1.0/PKG-INFO +390 -0
  10. ragpreflight-0.1.0/README.md +340 -0
  11. ragpreflight-0.1.0/docs/olmocr-bench-report.html +180 -0
  12. ragpreflight-0.1.0/docs/profiles.md +183 -0
  13. ragpreflight-0.1.0/docs/taxonomy-coverage.md +126 -0
  14. ragpreflight-0.1.0/pyproject.toml +92 -0
  15. ragpreflight-0.1.0/src/ragpreflight/__init__.py +52 -0
  16. ragpreflight-0.1.0/src/ragpreflight/_constants.py +284 -0
  17. ragpreflight-0.1.0/src/ragpreflight/chunker.py +648 -0
  18. ragpreflight-0.1.0/src/ragpreflight/cli.py +625 -0
  19. ragpreflight-0.1.0/src/ragpreflight/config.py +163 -0
  20. ragpreflight-0.1.0/src/ragpreflight/corpus.py +635 -0
  21. ragpreflight-0.1.0/src/ragpreflight/integrations/__init__.py +1 -0
  22. ragpreflight-0.1.0/src/ragpreflight/integrations/langchain.py +131 -0
  23. ragpreflight-0.1.0/src/ragpreflight/integrations/llamaindex.py +137 -0
  24. ragpreflight-0.1.0/src/ragpreflight/models.py +230 -0
  25. ragpreflight-0.1.0/src/ragpreflight/profiles.py +66 -0
  26. ragpreflight-0.1.0/src/ragpreflight/report.py +751 -0
  27. ragpreflight-0.1.0/src/ragpreflight/retrieval.py +405 -0
  28. ragpreflight-0.1.0/src/ragpreflight/scanner.py +1965 -0
  29. ragpreflight-0.1.0/src/ragpreflight/taxonomy/__init__.py +93 -0
  30. ragpreflight-0.1.0/src/ragpreflight/taxonomy/coverage.py +42 -0
  31. ragpreflight-0.1.0/src/ragpreflight/taxonomy/garani_2026.yaml +651 -0
  32. ragpreflight-0.1.0/src/ragpreflight/taxonomy/loader.py +73 -0
  33. ragpreflight-0.1.0/src/ragpreflight/taxonomy/models.py +47 -0
  34. ragpreflight-0.1.0/src/ragpreflight/utils.py +233 -0
  35. ragpreflight-0.1.0/tests/conftest.py +117 -0
  36. ragpreflight-0.1.0/tests/fixtures/clean.pdf +0 -0
  37. ragpreflight-0.1.0/tests/fixtures/empty.pdf +33 -0
  38. ragpreflight-0.1.0/tests/fixtures/empty.txt +1 -0
  39. ragpreflight-0.1.0/tests/fixtures/messy.csv +6 -0
  40. ragpreflight-0.1.0/tests/fixtures/messy_table.pdf +0 -0
  41. ragpreflight-0.1.0/tests/fixtures/ocr_errors.pdf +0 -0
  42. ragpreflight-0.1.0/tests/fixtures/ocr_errors.txt +23 -0
  43. ragpreflight-0.1.0/tests/fixtures/sample.csv +11 -0
  44. ragpreflight-0.1.0/tests/fixtures/sample.docx +0 -0
  45. ragpreflight-0.1.0/tests/fixtures/sample.html +55 -0
  46. ragpreflight-0.1.0/tests/fixtures/sample.ipynb +59 -0
  47. ragpreflight-0.1.0/tests/fixtures/sample.md +34 -0
  48. ragpreflight-0.1.0/tests/fixtures/sample.pptx +0 -0
  49. ragpreflight-0.1.0/tests/fixtures/sample.srt +15 -0
  50. ragpreflight-0.1.0/tests/fixtures/sample.txt +23 -0
  51. ragpreflight-0.1.0/tests/fixtures/sample.xlsx +0 -0
  52. ragpreflight-0.1.0/tests/fixtures/scanned.pdf +0 -0
  53. ragpreflight-0.1.0/tests/test_chunker.py +127 -0
  54. ragpreflight-0.1.0/tests/test_cli.py +148 -0
  55. ragpreflight-0.1.0/tests/test_corpus.py +99 -0
  56. ragpreflight-0.1.0/tests/test_retrieval.py +94 -0
  57. ragpreflight-0.1.0/tests/test_scanner.py +202 -0
  58. ragpreflight-0.1.0/tests/test_taxonomy.py +196 -0
  59. ragpreflight-0.1.0/tests/test_utils.py +174 -0
@@ -0,0 +1,64 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ branches: [main, develop]
6
+ pull_request:
7
+ branches: [main]
8
+
9
+ jobs:
10
+ test:
11
+ name: Test Python ${{ matrix.python-version }}
12
+ runs-on: ubuntu-latest
13
+ strategy:
14
+ fail-fast: false
15
+ matrix:
16
+ python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
17
+
18
+ steps:
19
+ - uses: actions/checkout@v4
20
+
21
+ - name: Set up Python ${{ matrix.python-version }}
22
+ uses: actions/setup-python@v5
23
+ with:
24
+ python-version: ${{ matrix.python-version }}
25
+
26
+ - name: Install dependencies
27
+ run: |
28
+ python -m pip install --upgrade pip
29
+ pip install -e ".[dev]"
30
+
31
+ - name: Run tests
32
+ run: pytest --tb=short -q -m "not slow"
33
+
34
+ - name: Upload coverage
35
+ if: matrix.python-version == '3.11'
36
+ run: pytest --cov=ragcheck --cov-report=xml -q -m "not slow"
37
+ continue-on-error: true
38
+
39
+ lint:
40
+ name: Lint & Type Check
41
+ runs-on: ubuntu-latest
42
+
43
+ steps:
44
+ - uses: actions/checkout@v4
45
+
46
+ - name: Set up Python
47
+ uses: actions/setup-python@v5
48
+ with:
49
+ python-version: "3.11"
50
+
51
+ - name: Install dependencies
52
+ run: |
53
+ python -m pip install --upgrade pip
54
+ pip install -e ".[dev]"
55
+
56
+ - name: Ruff (lint)
57
+ run: ruff check src/ tests/
58
+
59
+ - name: Ruff (format check)
60
+ run: ruff format --check src/ tests/
61
+
62
+ - name: mypy
63
+ run: mypy src/ragcheck/
64
+ continue-on-error: true # Type checking warnings don't fail the build initially
@@ -0,0 +1,44 @@
1
+ # Build / distribution
2
+ dist/
3
+ build/
4
+ *.egg-info/
5
+ *.egg
6
+
7
+ # Python
8
+ __pycache__/
9
+ *.py[cod]
10
+ *.pyo
11
+ .Python
12
+
13
+ # Virtual environments
14
+ .venv/
15
+ venv/
16
+ env/
17
+
18
+ # Testing
19
+ .pytest_cache/
20
+ .coverage
21
+ htmlcov/
22
+ .mypy_cache/
23
+ .ruff_cache/
24
+
25
+ # macOS
26
+ .DS_Store
27
+ .AppleDouble
28
+
29
+ # IDE
30
+ .idea/
31
+ .vscode/
32
+ *.swp
33
+
34
+ # Claude Code
35
+ .claude/
36
+
37
+ # Private / internal session files
38
+ claude-session-history.md
39
+ *SESSION*.md
40
+ *CAPABILITY_REPORT*.md
41
+ *REFACTOR_SESSION*.md
42
+ PLAN_*.md
43
+ TECHNICAL_REFERENCE.md
44
+ research/
@@ -0,0 +1,19 @@
1
+ - id: ragpreflight-scan
2
+ name: ragpreflight — document readiness scan
3
+ description: Scan staged documents for RAG quality issues before they enter the corpus.
4
+ language: python
5
+ entry: ragpreflight score
6
+ types_or: [pdf, file]
7
+ args: []
8
+ pass_filenames: true
9
+ stages: [pre-commit]
10
+
11
+ - id: ragpreflight-scan-strict
12
+ name: ragpreflight — strict document readiness scan
13
+ description: Fail the commit if any document scores below the strict profile threshold.
14
+ language: python
15
+ entry: ragpreflight scan
16
+ types_or: [pdf, file]
17
+ args: [--profile, strict, --quiet]
18
+ pass_filenames: true
19
+ stages: [pre-commit]
@@ -0,0 +1,70 @@
1
+ # Changelog
2
+
3
+ All notable changes to ragpreflight will be documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [Unreleased]
9
+
10
+ ## [0.1.0] — 2026-09-29
11
+
12
+ First public release. Grounded in Garani 2026 (doi:10.18653/v1/2026.trustnlp-main.27).
13
+
14
+ ### Added
15
+
16
+ - **Document Scanner** (`ragpreflight scan`) — score individual documents 0–100 across 5 quality dimensions:
17
+ - Text extractability (30%)
18
+ - OCR cleanliness (25%)
19
+ - Structural integrity (20%)
20
+ - Metadata completeness (10%)
21
+ - Content density (15%)
22
+
23
+ - **Supported formats**: PDF (text + scanned), DOCX, TXT, CSV/TSV, HTML, Markdown, XLSX, PPTX, IPYNB, SRT
24
+
25
+ - **Corpus Auditor** (`ragpreflight audit`) — scan entire directories with:
26
+ - Near-duplicate detection via MinHash LSH (datasketch) or shingle Jaccard fallback
27
+ - File-age freshness proxy (not a content-staleness guarantee — see F1)
28
+ - Conflict candidate detection for human review (high embedding similarity pairs)
29
+ - Format distribution statistics and worst-offender list
30
+
31
+ - **Chunk Diagnostics** (`ragpreflight chunks`) — analyse chunked text for:
32
+ - Semantic coherence via sentence-transformers (optional; returns `None` not fake `1.0` when absent)
33
+ - Structural break detection (tables, lists, code blocks)
34
+ - Boundary quality (mid-sentence cuts)
35
+ - Information density
36
+
37
+ - **Retrieval Simulation** (`ragpreflight simulate`) — synthetic query generation + similarity hit rate:
38
+ - Named `synthetic_retrieval_hit_rate` (not `precision@k` — no labeled relevance)
39
+ - Dead chunk rate, query failure rate
40
+ - Optional LLM-powered query generation (`--use-llm`)
41
+
42
+ - **Garani 2026 Taxonomy** (`ragpreflight taxonomy`, `ragpreflight coverage`):
43
+ - All 33 failure modes across 7 pipeline stages, loaded from YAML
44
+ - Evidence levels: 9 Strong, 12 Moderate, 12 Limited
45
+ - `detector_status` per mode: direct / proxy / risk_signal / runtime_required / unsupported
46
+ - `TaxonomyReference` on every `Issue` linking findings to F1–F33
47
+
48
+ - **Quality Profiles**: `permissive`, `standard`, `strict`
49
+
50
+ - **Output Formats**: rich terminal tables, JSON (`--json`), self-contained HTML (`--format html`), SARIF (`--format sarif`)
51
+
52
+ - **CI/CD friendly**: `ragpreflight score <file>` prints only the score; exit code 1 on critical issues
53
+
54
+ - **Config file**: `.ragpreflight.toml` in project root or home directory
55
+
56
+ - **Zero telemetry**: no tracking, no API keys for core functionality, fully offline-capable
57
+
58
+ ### Fixed (vs original ragcheck codebase)
59
+
60
+ - `coherence_score` fallback was returning fake `1.0` when sentence-transformers absent — now returns `None`
61
+ - `precision_at_k` renamed to `synthetic_retrieval_hit_rate` (no labeled relevance existed)
62
+ - Staleness message now says "file-age proxy" — filesystem mtime does not prove content outdated
63
+ - `_find_contradiction_candidates` renamed to `_find_conflict_candidates` — these are not confirmed contradictions
64
+ - `_scan_xlsx` docstring falsely claimed `.xls` support — openpyxl only handles `.xlsx`
65
+ - `[cloud]` extra removed — no connector module existed
66
+ - `scikit-learn` removed from `[full]` — was unused
67
+ - Python minimum bumped to `>=3.10` (3.9 EOL)
68
+
69
+ [Unreleased]: https://github.com/anusky95/ragpreflight/compare/v0.1.0...HEAD
70
+ [0.1.0]: https://github.com/anusky95/ragpreflight/releases/tag/v0.1.0
@@ -0,0 +1,34 @@
1
+ cff-version: 1.2.0
2
+ message: "If you use this software, please cite both the software and the paper below."
3
+ type: software
4
+ title: "ragpreflight"
5
+ abstract: "Pre-ingestion RAG document quality audit tool grounded in the Garani 2026 systematic taxonomy of 33 RAG failure modes across 7 pipeline stages."
6
+ authors:
7
+ - family-names: Garani
8
+ given-names: Anupama
9
+ email: anupamagarani95@gmail.com
10
+ orcid: "https://orcid.org/0000-0000-0000-0000"
11
+ repository-code: "https://github.com/anusky95/ragpreflight"
12
+ url: "https://pypi.org/project/ragpreflight/"
13
+ license: MIT
14
+ version: "0.1.0"
15
+ date-released: "2026-09-29"
16
+ keywords:
17
+ - RAG
18
+ - retrieval-augmented-generation
19
+ - document-quality
20
+ - pre-ingestion
21
+ - failure-modes
22
+ - taxonomy
23
+ - NLP
24
+ preferred-citation:
25
+ type: conference-paper
26
+ title: "A Systematic Taxonomy of Failure Modes in Retrieval-Augmented Generation Systems"
27
+ authors:
28
+ - family-names: Garani
29
+ given-names: Anupama
30
+ collection-title: "Proceedings of the 6th Workshop on Trustworthy Natural Language Processing (TrustNLP 2026)"
31
+ publisher: "Association for Computational Linguistics"
32
+ year: 2026
33
+ doi: "10.18653/v1/2026.trustnlp-main.27"
34
+ url: "https://aclanthology.org/2026.trustnlp-main.27/"
@@ -0,0 +1,354 @@
1
+ # CLAUDE.md — RAGCheck Project Brief
2
+
3
+ ## What Is This
4
+
5
+ RAGCheck is an open-source Python package that scores document readiness for RAG pipelines. It tells you what's broken in your documents BEFORE you embed and retrieve them — not after your chatbot starts hallucinating.
6
+
7
+ **One-liner:** `pip install ragcheck` → `ragcheck scan my_document.pdf` → readiness score 0-100 with actionable issues.
8
+
9
+ ## Design Principles (Follow These Always)
10
+
11
+ 1. **Zero API keys for core functionality.** Everything runs locally by default. LLM-powered features are optional extras behind `--use-llm` flags. Never make an external API call without the user explicitly opting in.
12
+ 2. **Lightweight base install.** Core dependencies: `pymupdf`, `pdfplumber`, `chardet`, `click`, `rich`. Heavy deps like `sentence-transformers` and `datasketch` are optional extras (`pip install ragcheck[full]`).
13
+ 3. **Never crash on bad input.** Malformed PDFs, 500MB files, zip bombs disguised as docs, empty files, binary garbage — handle all of it gracefully with clear error messages. Wrap all file I/O in try/except and return meaningful errors, not tracebacks.
14
+ 4. **Typed dataclasses for all outputs.** Every public function returns typed dataclasses, not raw dicts. Users get autocomplete and type checking for free.
15
+ 5. **Tests alongside code.** Every module gets a corresponding test file. Write tests as you implement, not as a separate phase. Include fixture files in `tests/fixtures/` for reproducibility.
16
+ 6. **Src layout.** Use `src/ragcheck/` package structure with `pyproject.toml`. No `setup.py`.
17
+ 7. **Fail informatively.** When something goes wrong, the error message should tell the user what happened, what file caused it, and what to do about it.
18
+
19
+ ## Target Users
20
+
21
+ - ML engineers building RAG pipelines who want to debug document quality issues
22
+ - Data engineers responsible for ingestion pipelines
23
+ - Teams evaluating vendor document processing tools
24
+ - Anyone who has ever asked "why is my RAG system hallucinating?"
25
+
26
+ ## Package Structure
27
+
28
+ ```
29
+ ragcheck/
30
+ ├── CLAUDE.md # This file
31
+ ├── LICENSE # MIT
32
+ ├── README.md # Badges, install, quickstart, examples
33
+ ├── pyproject.toml # Build config, dependencies, optional extras
34
+ ├── .github/
35
+ │ └── workflows/
36
+ │ └── ci.yml # pytest + ruff + mypy
37
+ ├── src/
38
+ │ └── ragcheck/
39
+ │ ├── __init__.py # Public API exports + __version__
40
+ │ ├── cli.py # Click CLI entry point
41
+ │ ├── models.py # All dataclasses (DocumentReport, ChunkReport, CorpusReport, Issue)
42
+ │ ├── scanner.py # Document scanning module
43
+ │ ├── chunker.py # Chunk analysis module
44
+ │ ├── corpus.py # Corpus-level analysis module
45
+ │ ├── retrieval.py # Retrieval simulation module
46
+ │ ├── profiles.py # Threshold profiles (strict, standard, permissive)
47
+ │ ├── report.py # Output formatting (terminal, JSON, HTML)
48
+ │ ├── utils.py # File detection, encoding helpers, size guards
49
+ │ └── _constants.py # OCR error patterns, default thresholds, magic numbers
50
+ ├── tests/
51
+ │ ├── conftest.py # Shared fixtures
52
+ │ ├── fixtures/ # Test PDFs, DOCX, CSVs, etc.
53
+ │ │ ├── clean.pdf
54
+ │ │ ├── ocr_errors.pdf
55
+ │ │ ├── scanned.pdf
56
+ │ │ ├── messy_table.pdf
57
+ │ │ ├── empty.pdf
58
+ │ │ ├── sample.docx
59
+ │ │ ├── sample.csv
60
+ │ │ └── sample.md
61
+ │ ├── test_scanner.py
62
+ │ ├── test_chunker.py
63
+ │ ├── test_corpus.py
64
+ │ ├── test_retrieval.py
65
+ │ ├── test_cli.py
66
+ │ └── test_utils.py
67
+ └── docs/
68
+ └── profiles.md # Explain strict/standard/permissive thresholds
69
+ ```
70
+
71
+ ## Data Models (Core Contract — Do Not Deviate)
72
+
73
+ ```python
74
+ from dataclasses import dataclass, field
75
+ from enum import Enum
76
+ from typing import Optional
77
+
78
+ class Severity(Enum):
79
+ CRITICAL = "critical" # Will definitely cause RAG failures
80
+ WARNING = "warning" # Likely to cause issues
81
+ INFO = "info" # Worth knowing, may not cause problems
82
+
83
+ class IssueCategory(Enum):
84
+ OCR = "ocr"
85
+ ENCODING = "encoding"
86
+ STRUCTURE = "structure"
87
+ CONTENT = "content"
88
+ METADATA = "metadata"
89
+ CHUNKING = "chunking"
90
+ DUPLICATION = "duplication"
91
+ STALENESS = "staleness"
92
+
93
+ @dataclass
94
+ class Issue:
95
+ category: IssueCategory
96
+ severity: Severity
97
+ message: str # Human-readable description
98
+ location: Optional[str] = None # e.g., "page 3", "rows 12-15", "chunk 7"
99
+ suggestion: Optional[str] = None # Actionable fix
100
+
101
+ @dataclass
102
+ class DocumentReport:
103
+ filepath: str
104
+ score: int # 0-100, overall readiness
105
+ issues: list[Issue] = field(default_factory=list)
106
+ page_count: int = 0
107
+ text_extractable_ratio: float = 0.0 # 0.0-1.0, how much text is extractable
108
+ file_format: str = ""
109
+ file_size_bytes: int = 0
110
+
111
+ @dataclass
112
+ class ChunkReport:
113
+ chunk_index: int
114
+ text_preview: str # First 100 chars
115
+ coherence_score: float # 0.0-1.0
116
+ issues: list[Issue] = field(default_factory=list)
117
+
118
+ @dataclass
119
+ class CorpusReport:
120
+ directory: str
121
+ total_documents: int
122
+ average_score: float
123
+ documents: list[DocumentReport] = field(default_factory=list)
124
+ corpus_issues: list[Issue] = field(default_factory=list) # Cross-document issues
125
+ duplicate_groups: list[list[str]] = field(default_factory=list) # Groups of near-dupes
126
+ ```
127
+
128
+ ## CLI Interface
129
+
130
+ ```bash
131
+ # Single document scan
132
+ ragcheck scan document.pdf
133
+ ragcheck scan document.pdf --json
134
+ ragcheck scan document.pdf --profile strict
135
+
136
+ # Full corpus audit
137
+ ragcheck audit ./knowledge_base/
138
+ ragcheck audit ./knowledge_base/ --format html --output report.html
139
+ ragcheck audit ./knowledge_base/ --profile medical
140
+
141
+ # Chunk analysis (requires optional deps)
142
+ ragcheck chunks document.pdf --strategy recursive --size 512 --overlap 50
143
+
144
+ # Retrieval simulation (requires optional deps)
145
+ ragcheck simulate ./knowledge_base/ --queries 50
146
+ ragcheck simulate ./knowledge_base/ --use-llm --llm-endpoint http://localhost:11434
147
+
148
+ # Quick check — just the score, nothing else
149
+ ragcheck score document.pdf
150
+ ```
151
+
152
+ ## Profiles (Threshold Presets)
153
+
154
+ ```python
155
+ PROFILES = {
156
+ "permissive": {
157
+ "min_document_score": 40,
158
+ "min_chunk_coherence": 0.5,
159
+ "max_ocr_error_rate": 0.10,
160
+ "max_duplicate_similarity": 0.95,
161
+ "description": "For chatbots, FAQ systems, internal tools"
162
+ },
163
+ "standard": {
164
+ "min_document_score": 60,
165
+ "min_chunk_coherence": 0.65,
166
+ "max_ocr_error_rate": 0.05,
167
+ "max_duplicate_similarity": 0.90,
168
+ "description": "For enterprise search, customer support RAG"
169
+ },
170
+ "strict": {
171
+ "min_document_score": 80,
172
+ "min_chunk_coherence": 0.8,
173
+ "max_ocr_error_rate": 0.02,
174
+ "max_duplicate_similarity": 0.85,
175
+ "description": "For medical, legal, financial RAG systems"
176
+ }
177
+ }
178
+ ```
179
+
180
+ ## Supported File Formats
181
+
182
+ | Format | Base Install | Detection |
183
+ |--------|-------------|-----------|
184
+ | PDF (text-based) | ✅ pymupdf + pdfplumber | OCR errors, structure, tables |
185
+ | PDF (scanned) | ✅ pymupdf | Flags as low-extractability, suggests OCR |
186
+ | DOCX | ✅ python-docx | Structure, metadata, encoding |
187
+ | TXT | ✅ built-in | Encoding, content density |
188
+ | CSV/TSV | ✅ built-in | Encoding, structure, consistency |
189
+ | HTML | ✅ beautifulsoup4 | Boilerplate ratio, structure |
190
+ | Markdown | ✅ built-in | Structure, header hierarchy |
191
+
192
+ ## OCR Error Detection Patterns
193
+
194
+ These are the common substitution patterns to check. Store in `_constants.py`:
195
+
196
+ ```python
197
+ OCR_SUBSTITUTION_PATTERNS = {
198
+ # Character confusions
199
+ "l/I/1": r"(?<=[a-z])[I1](?=[a-z])", # "cIinical" → "clinical"
200
+ "O/0": r"(?<=[a-z])0(?=[a-z])", # "pr0tocol" → "protocol"
201
+ "rn/m": r"(?<=[a-z])rn(?=[a-z])", # "inforrnation" → "information"
202
+ "fi/fl_ligature": r"[fifl]", # Ligature artifacts
203
+ "broken_ligatures": r"(?<=\w)[ffi](?=\w)", # Split ligatures
204
+ # Spacing artifacts
205
+ "mid_word_spaces": r"(?<=[a-z]) (?=[a-z]{2})", # "pati ent" → "patient"
206
+ "merged_words": None, # Detected via dictionary lookup, not regex
207
+ # Encoding artifacts
208
+ "mojibake": r"[â’©é]", # UTF-8 decoded as Latin-1
209
+ }
210
+ ```
211
+
212
+ ## What NOT to Build
213
+
214
+ - **No web UI.** This is a CLI/library tool. Web dashboards can come from the community.
215
+ - **No database.** Reports are generated on-the-fly. Users can pipe JSON to wherever they want.
216
+ - **No user accounts or telemetry.** Zero tracking, zero phoning home.
217
+ - **No custom ML models.** Use existing sentence-transformers and standard NLP tools. Don't train anything.
218
+ - **No vendor lock-in.** The `--use-llm` flag works with any OpenAI-compatible endpoint (OpenAI, Ollama, vLLM, Azure, etc.), not just one provider.
219
+
220
+ ## Implementation Phases
221
+
222
+ ### Phase 1: Scaffold
223
+ Set up the package structure exactly as described above. `pyproject.toml` with:
224
+ - Build system: hatchling
225
+ - Python: >=3.9
226
+ - Core deps: pymupdf, pdfplumber, chardet, click, rich, python-docx, beautifulsoup4
227
+ - Optional extras: `[full]` = sentence-transformers, datasketch; `[llm]` = openai
228
+ - CLI entry point: `ragcheck = ragcheck.cli:main`
229
+ - Linting: ruff
230
+ - Type checking: mypy
231
+ - Testing: pytest
232
+
233
+ Create a working CLI that responds to `ragcheck --version` and `ragcheck scan --help`.
234
+ Write README.md with badges (PyPI, Python version, License, CI), one-liner install, quickstart example, and feature comparison table vs RAGAS/Unstructured/LlamaIndex.
235
+
236
+ ### Phase 2: Document Scanner (`scanner.py`)
237
+ Build the core scanning engine. For each supported file format, detect:
238
+ - **Encoding issues** via chardet (flag confidence < 0.8)
239
+ - **Text extractable ratio** — what percentage of pages/content yield actual text vs images/scans
240
+ - **OCR error patterns** using the regex patterns in `_constants.py`
241
+ - **Character-level anomalies** — unusual Unicode, control characters, excessive whitespace
242
+ - **Empty or near-empty pages** (< 10 words)
243
+ - **Embedded tables or structured data** that need special chunking (detect via cell/row patterns)
244
+ - **Metadata presence** — does the doc have title, author, date, or is it bare?
245
+ - **File size sanity** — flag files > 100MB with a warning, reject > 500MB by default (configurable)
246
+
247
+ The scanner returns a `DocumentReport` with a score 0-100 computed as weighted sum:
248
+ - Text extractability: 30%
249
+ - OCR cleanliness: 25%
250
+ - Structural integrity: 20%
251
+ - Metadata completeness: 10%
252
+ - Content density: 15%
253
+
254
+ Write tests for each file format with both clean and problematic fixtures.
255
+
256
+ ### Phase 3: Chunk Diagnostics (`chunker.py`)
257
+ Build the chunk analysis module. Given a document and a chunking strategy:
258
+ - Default: recursive character split at 512 tokens, 50 token overlap
259
+ - Accept a pluggable splitter function: `Callable[[str], list[str]]`
260
+
261
+ For each chunk, score:
262
+ - **Semantic coherence** using sentence-transformers `all-MiniLM-L6-v2` (optional dep — if not installed, skip this metric and warn)
263
+ - **Structural breaks** — does the chunk split a table, list, or code block mid-element?
264
+ - **Metadata preservation** — does the chunk retain source page number, section header context?
265
+ - **Information density** — ratio of meaningful tokens vs boilerplate/stopwords
266
+ - **Boundary quality** — does the chunk start/end mid-sentence?
267
+
268
+ Return a list of `ChunkReport` objects. Flag chunks below profile thresholds.
269
+ Make chunking strategy pluggable so users can bring their own splitter.
270
+
271
+ ### Phase 4: Corpus Analysis (`corpus.py`)
272
+ Build the corpus-level analysis module. Given a directory of documents:
273
+ - Run `scanner.py` on each file
274
+ - **Near-duplicate detection** using MinHash/LSH via datasketch (optional dep)
275
+ - **Contradiction candidates** — find doc pairs covering the same topic (embedding similarity > 0.85) with different factual claims. Surface pairs for human review, don't auto-judge.
276
+ - **Temporal coverage** — if docs have detectable dates (file metadata, content regex), flag docs older than a configurable threshold (default: 12 months)
277
+ - **Format distribution** — breakdown by file type
278
+ - **Worst offenders** — bottom 10% by score
279
+ - **Corpus statistics** — total docs, average score, score distribution
280
+
281
+ Return a `CorpusReport`. Show a progress bar via `rich` for directories with 5+ files.
282
+
283
+ ### Phase 5: Retrieval Simulation (`retrieval.py`)
284
+ Build the retrieval simulation module. Generate synthetic test queries WITHOUT requiring any LLM API:
285
+ - **Extractive methods**: Pull key phrases (TF-IDF top terms), named entities (simple regex-based NER for common types), and section headers → form natural question templates
286
+ - Template patterns: "What is {entity}?", "Explain {concept}", "Summarize {section_title}", "What are the details of {key_phrase}?"
287
+ - Generate 5-10 queries per document by default
288
+
289
+ Test retrieval:
290
+ - Embed queries and chunks using sentence-transformers
291
+ - Compute cosine similarity, retrieve top-k (default k=5)
292
+ - Score: precision@k, percentage of dead chunks (never retrieved by any query), query failure rate (best match similarity < 0.5)
293
+
294
+ Optional `--use-llm` flag:
295
+ - Accepts any OpenAI-compatible endpoint via `--llm-endpoint` (default: http://localhost:11434 for Ollama)
296
+ - Uses the LLM to generate more diverse, natural queries
297
+ - Falls back to extractive if LLM endpoint is unreachable
298
+
299
+ ### Phase 6: CLI & Reporting (`cli.py`, `report.py`)
300
+ Polish the CLI commands:
301
+ - `ragcheck scan <file>` — single document, terminal table output by default
302
+ - `ragcheck audit <directory>` — full corpus audit with progress bar
303
+ - `ragcheck chunks <file>` — chunk analysis (warns if optional deps missing)
304
+ - `ragcheck simulate <directory>` — retrieval simulation (warns if optional deps missing)
305
+ - `ragcheck score <file>` — just the number, for scripting (`echo $(ragcheck score doc.pdf)`)
306
+
307
+ Output formats:
308
+ - Default: `rich` terminal table with color-coded severity
309
+ - `--json` flag: structured JSON to stdout
310
+ - `--format html --output report.html`: self-contained HTML report with expandable sections
311
+ - `--quiet` flag: suppress all output except the score (for CI/CD pipelines)
312
+
313
+ Config file support:
314
+ - `.ragcheckrc` (TOML format) in project root or home directory
315
+ - Override profile, thresholds, file size limits, output preferences
316
+ - CLI flags override config file values
317
+
318
+ ### Phase 7: Harden for Release
319
+ Review entire codebase for:
320
+ - **Type hints** on every function signature and return type
321
+ - **Docstrings** on every public method (Google style)
322
+ - **Input validation** — clear ValueError/TypeError with actionable messages
323
+ - **Edge cases** — empty directories, permission errors, symlinks, binary files with wrong extensions
324
+ - **Security** — no path traversal, no arbitrary code execution, no eval/exec, no pickling untrusted data, size limits on all file reads
325
+ - **CI pipeline** — GitHub Actions running pytest, ruff, mypy on Python 3.9/3.10/3.11/3.12
326
+ - **`--verbose` flag** for debug logging (use stdlib `logging`, not print)
327
+ - **CONTRIBUTING.md** with dev setup instructions, PR guidelines, and code style expectations
328
+ - **CHANGELOG.md** with initial release notes
329
+
330
+ ### Phase 8: Integration & Dogfooding
331
+ Wire up the full pipeline so `ragcheck audit mydir/` runs scanning → chunk analysis → corpus analysis in sequence with a unified report. Each module stays independently importable.
332
+
333
+ Then run ragcheck against:
334
+ 1. The test fixtures in `tests/fixtures/`
335
+ 2. The ragcheck source code itself (`.py` files as TXT)
336
+ 3. The README.md
337
+ 4. A deliberately corrupted PDF (create one in tests)
338
+
339
+ Fix any crashes, confusing output, or scores that seem wrong. Verify that:
340
+ - A clean, well-structured PDF scores > 85
341
+ - A scanned PDF with OCR errors scores < 50
342
+ - An empty file scores 0 with a clear CRITICAL issue
343
+ - The CLI never shows a traceback to the user (all errors are caught and formatted)
344
+
345
+ ## Code Style Rules
346
+
347
+ - Use ruff for formatting and linting (line length 99)
348
+ - Use Google-style docstrings
349
+ - Prefer early returns over deep nesting
350
+ - No global mutable state
351
+ - No `print()` — use `click.echo()` in CLI, `logging` everywhere else
352
+ - Constants in SCREAMING_SNAKE_CASE in `_constants.py`
353
+ - Private functions prefixed with underscore
354
+ - No `# type: ignore` without a comment explaining why
@@ -0,0 +1,79 @@
1
+ # Contributing to RAGCheck
2
+
3
+ Thank you for your interest in contributing! This guide will help you set up your development environment and understand our conventions.
4
+
5
+ ## Dev Setup
6
+
7
+ ```bash
8
+ # 1. Fork and clone
9
+ git clone https://github.com/your-fork/ragcheck.git
10
+ cd ragcheck
11
+
12
+ # 2. Create a virtual environment
13
+ python -m venv .venv
14
+ source .venv/bin/activate # Windows: .venv\Scripts\activate
15
+
16
+ # 3. Install in editable mode with dev dependencies
17
+ pip install -e ".[dev]"
18
+
19
+ # 4. (Optional) Full extras for all features
20
+ pip install -e ".[dev,full,llm]"
21
+
22
+ # 5. Verify setup
23
+ ragcheck --version
24
+ pytest -q
25
+ ```
26
+
27
+ ## Running Tests
28
+
29
+ ```bash
30
+ pytest # all tests
31
+ pytest tests/test_scanner.py # one module
32
+ pytest -k "test_scan" # by name pattern
33
+ pytest --cov=ragcheck # with coverage
34
+ ```
35
+
36
+ ## Code Style
37
+
38
+ We use **ruff** for formatting and linting. Line length is 99 characters.
39
+
40
+ ```bash
41
+ ruff check src/ tests/ # lint
42
+ ruff format src/ tests/ # auto-format
43
+ mypy src/ragcheck/ # type check
44
+ ```
45
+
46
+ All of the above run automatically in CI. Your PR must pass lint + tests before it can be merged.
47
+
48
+ ## Design Principles (Read Before Contributing)
49
+
50
+ 1. **Zero API keys for core functionality.** LLM-powered features must be behind `--use-llm` flags.
51
+ 2. **Lightweight base install.** Heavy deps go in `[full]` or `[llm]` extras — never in core.
52
+ 3. **Never crash on bad input.** Wrap all file I/O in try/except. Return errors, not tracebacks.
53
+ 4. **Typed dataclasses for all outputs.** Public functions return typed dataclasses, not dicts.
54
+ 5. **Tests alongside code.** Every module has a corresponding test file.
55
+ 6. **No `print()` in library code.** Use `logging` everywhere; `click.echo()` in CLI only.
56
+
57
+ ## PR Guidelines
58
+
59
+ - Keep PRs focused: one feature or fix per PR
60
+ - Update tests alongside your changes
61
+ - Add your change to `CHANGELOG.md` under `[Unreleased]`
62
+ - Make sure `pytest` and `ruff check` pass locally before opening a PR
63
+
64
+ ## Adding a New File Format
65
+
66
+ 1. Add the extension to `SUPPORTED_EXTENSIONS` in `_constants.py`
67
+ 2. Write a `_scan_<format>` function in `scanner.py` following the existing pattern
68
+ 3. Register it in `_FORMAT_SCANNERS` in `scanner.py`
69
+ 4. Add at least one test fixture in `tests/fixtures/`
70
+ 5. Write tests in `tests/test_scanner.py`
71
+
72
+ ## Reporting Bugs
73
+
74
+ Open an issue at [github.com/ragcheck/ragcheck/issues](https://github.com/ragcheck/ragcheck/issues) with:
75
+ - RAGCheck version (`ragcheck --version`)
76
+ - Python version
77
+ - OS
78
+ - Command or code that triggered the bug
79
+ - Full error output (with `--verbose` if applicable)