ragpreflight 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ragpreflight-0.1.0/.github/workflows/ci.yml +64 -0
- ragpreflight-0.1.0/.gitignore +44 -0
- ragpreflight-0.1.0/.pre-commit-hooks.yaml +19 -0
- ragpreflight-0.1.0/CHANGELOG.md +70 -0
- ragpreflight-0.1.0/CITATION.cff +34 -0
- ragpreflight-0.1.0/CLAUDE.md +354 -0
- ragpreflight-0.1.0/CONTRIBUTING.md +79 -0
- ragpreflight-0.1.0/LICENSE +21 -0
- ragpreflight-0.1.0/PKG-INFO +390 -0
- ragpreflight-0.1.0/README.md +340 -0
- ragpreflight-0.1.0/docs/olmocr-bench-report.html +180 -0
- ragpreflight-0.1.0/docs/profiles.md +183 -0
- ragpreflight-0.1.0/docs/taxonomy-coverage.md +126 -0
- ragpreflight-0.1.0/pyproject.toml +92 -0
- ragpreflight-0.1.0/src/ragpreflight/__init__.py +52 -0
- ragpreflight-0.1.0/src/ragpreflight/_constants.py +284 -0
- ragpreflight-0.1.0/src/ragpreflight/chunker.py +648 -0
- ragpreflight-0.1.0/src/ragpreflight/cli.py +625 -0
- ragpreflight-0.1.0/src/ragpreflight/config.py +163 -0
- ragpreflight-0.1.0/src/ragpreflight/corpus.py +635 -0
- ragpreflight-0.1.0/src/ragpreflight/integrations/__init__.py +1 -0
- ragpreflight-0.1.0/src/ragpreflight/integrations/langchain.py +131 -0
- ragpreflight-0.1.0/src/ragpreflight/integrations/llamaindex.py +137 -0
- ragpreflight-0.1.0/src/ragpreflight/models.py +230 -0
- ragpreflight-0.1.0/src/ragpreflight/profiles.py +66 -0
- ragpreflight-0.1.0/src/ragpreflight/report.py +751 -0
- ragpreflight-0.1.0/src/ragpreflight/retrieval.py +405 -0
- ragpreflight-0.1.0/src/ragpreflight/scanner.py +1965 -0
- ragpreflight-0.1.0/src/ragpreflight/taxonomy/__init__.py +93 -0
- ragpreflight-0.1.0/src/ragpreflight/taxonomy/coverage.py +42 -0
- ragpreflight-0.1.0/src/ragpreflight/taxonomy/garani_2026.yaml +651 -0
- ragpreflight-0.1.0/src/ragpreflight/taxonomy/loader.py +73 -0
- ragpreflight-0.1.0/src/ragpreflight/taxonomy/models.py +47 -0
- ragpreflight-0.1.0/src/ragpreflight/utils.py +233 -0
- ragpreflight-0.1.0/tests/conftest.py +117 -0
- ragpreflight-0.1.0/tests/fixtures/clean.pdf +0 -0
- ragpreflight-0.1.0/tests/fixtures/empty.pdf +33 -0
- ragpreflight-0.1.0/tests/fixtures/empty.txt +1 -0
- ragpreflight-0.1.0/tests/fixtures/messy.csv +6 -0
- ragpreflight-0.1.0/tests/fixtures/messy_table.pdf +0 -0
- ragpreflight-0.1.0/tests/fixtures/ocr_errors.pdf +0 -0
- ragpreflight-0.1.0/tests/fixtures/ocr_errors.txt +23 -0
- ragpreflight-0.1.0/tests/fixtures/sample.csv +11 -0
- ragpreflight-0.1.0/tests/fixtures/sample.docx +0 -0
- ragpreflight-0.1.0/tests/fixtures/sample.html +55 -0
- ragpreflight-0.1.0/tests/fixtures/sample.ipynb +59 -0
- ragpreflight-0.1.0/tests/fixtures/sample.md +34 -0
- ragpreflight-0.1.0/tests/fixtures/sample.pptx +0 -0
- ragpreflight-0.1.0/tests/fixtures/sample.srt +15 -0
- ragpreflight-0.1.0/tests/fixtures/sample.txt +23 -0
- ragpreflight-0.1.0/tests/fixtures/sample.xlsx +0 -0
- ragpreflight-0.1.0/tests/fixtures/scanned.pdf +0 -0
- ragpreflight-0.1.0/tests/test_chunker.py +127 -0
- ragpreflight-0.1.0/tests/test_cli.py +148 -0
- ragpreflight-0.1.0/tests/test_corpus.py +99 -0
- ragpreflight-0.1.0/tests/test_retrieval.py +94 -0
- ragpreflight-0.1.0/tests/test_scanner.py +202 -0
- ragpreflight-0.1.0/tests/test_taxonomy.py +196 -0
- ragpreflight-0.1.0/tests/test_utils.py +174 -0
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main, develop]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [main]
|
|
8
|
+
|
|
9
|
+
jobs:
|
|
10
|
+
test:
|
|
11
|
+
name: Test Python ${{ matrix.python-version }}
|
|
12
|
+
runs-on: ubuntu-latest
|
|
13
|
+
strategy:
|
|
14
|
+
fail-fast: false
|
|
15
|
+
matrix:
|
|
16
|
+
python-version: ["3.9", "3.10", "3.11", "3.12", "3.13"]
|
|
17
|
+
|
|
18
|
+
steps:
|
|
19
|
+
- uses: actions/checkout@v4
|
|
20
|
+
|
|
21
|
+
- name: Set up Python ${{ matrix.python-version }}
|
|
22
|
+
uses: actions/setup-python@v5
|
|
23
|
+
with:
|
|
24
|
+
python-version: ${{ matrix.python-version }}
|
|
25
|
+
|
|
26
|
+
- name: Install dependencies
|
|
27
|
+
run: |
|
|
28
|
+
python -m pip install --upgrade pip
|
|
29
|
+
pip install -e ".[dev]"
|
|
30
|
+
|
|
31
|
+
- name: Run tests
|
|
32
|
+
run: pytest --tb=short -q -m "not slow"
|
|
33
|
+
|
|
34
|
+
- name: Upload coverage
|
|
35
|
+
if: matrix.python-version == '3.11'
|
|
36
|
+
run: pytest --cov=ragcheck --cov-report=xml -q -m "not slow"
|
|
37
|
+
continue-on-error: true
|
|
38
|
+
|
|
39
|
+
lint:
|
|
40
|
+
name: Lint & Type Check
|
|
41
|
+
runs-on: ubuntu-latest
|
|
42
|
+
|
|
43
|
+
steps:
|
|
44
|
+
- uses: actions/checkout@v4
|
|
45
|
+
|
|
46
|
+
- name: Set up Python
|
|
47
|
+
uses: actions/setup-python@v5
|
|
48
|
+
with:
|
|
49
|
+
python-version: "3.11"
|
|
50
|
+
|
|
51
|
+
- name: Install dependencies
|
|
52
|
+
run: |
|
|
53
|
+
python -m pip install --upgrade pip
|
|
54
|
+
pip install -e ".[dev]"
|
|
55
|
+
|
|
56
|
+
- name: Ruff (lint)
|
|
57
|
+
run: ruff check src/ tests/
|
|
58
|
+
|
|
59
|
+
- name: Ruff (format check)
|
|
60
|
+
run: ruff format --check src/ tests/
|
|
61
|
+
|
|
62
|
+
- name: mypy
|
|
63
|
+
run: mypy src/ragcheck/
|
|
64
|
+
continue-on-error: true # Type checking warnings don't fail the build initially
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Build / distribution
|
|
2
|
+
dist/
|
|
3
|
+
build/
|
|
4
|
+
*.egg-info/
|
|
5
|
+
*.egg
|
|
6
|
+
|
|
7
|
+
# Python
|
|
8
|
+
__pycache__/
|
|
9
|
+
*.py[cod]
|
|
10
|
+
*.pyo
|
|
11
|
+
.Python
|
|
12
|
+
|
|
13
|
+
# Virtual environments
|
|
14
|
+
.venv/
|
|
15
|
+
venv/
|
|
16
|
+
env/
|
|
17
|
+
|
|
18
|
+
# Testing
|
|
19
|
+
.pytest_cache/
|
|
20
|
+
.coverage
|
|
21
|
+
htmlcov/
|
|
22
|
+
.mypy_cache/
|
|
23
|
+
.ruff_cache/
|
|
24
|
+
|
|
25
|
+
# macOS
|
|
26
|
+
.DS_Store
|
|
27
|
+
.AppleDouble
|
|
28
|
+
|
|
29
|
+
# IDE
|
|
30
|
+
.idea/
|
|
31
|
+
.vscode/
|
|
32
|
+
*.swp
|
|
33
|
+
|
|
34
|
+
# Claude Code
|
|
35
|
+
.claude/
|
|
36
|
+
|
|
37
|
+
# Private / internal session files
|
|
38
|
+
claude-session-history.md
|
|
39
|
+
*SESSION*.md
|
|
40
|
+
*CAPABILITY_REPORT*.md
|
|
41
|
+
*REFACTOR_SESSION*.md
|
|
42
|
+
PLAN_*.md
|
|
43
|
+
TECHNICAL_REFERENCE.md
|
|
44
|
+
research/
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
- id: ragpreflight-scan
|
|
2
|
+
name: ragpreflight — document readiness scan
|
|
3
|
+
description: Scan staged documents for RAG quality issues before they enter the corpus.
|
|
4
|
+
language: python
|
|
5
|
+
entry: ragpreflight score
|
|
6
|
+
types_or: [pdf, file]
|
|
7
|
+
args: []
|
|
8
|
+
pass_filenames: true
|
|
9
|
+
stages: [pre-commit]
|
|
10
|
+
|
|
11
|
+
- id: ragpreflight-scan-strict
|
|
12
|
+
name: ragpreflight — strict document readiness scan
|
|
13
|
+
description: Fail the commit if any document scores below the strict profile threshold.
|
|
14
|
+
language: python
|
|
15
|
+
entry: ragpreflight scan
|
|
16
|
+
types_or: [pdf, file]
|
|
17
|
+
args: [--profile, strict, --quiet]
|
|
18
|
+
pass_filenames: true
|
|
19
|
+
stages: [pre-commit]
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to ragpreflight will be documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
## [0.1.0] — 2026-09-29
|
|
11
|
+
|
|
12
|
+
First public release. Grounded in Garani 2026 (doi:10.18653/v1/2026.trustnlp-main.27).
|
|
13
|
+
|
|
14
|
+
### Added
|
|
15
|
+
|
|
16
|
+
- **Document Scanner** (`ragpreflight scan`) — score individual documents 0–100 across 5 quality dimensions:
|
|
17
|
+
- Text extractability (30%)
|
|
18
|
+
- OCR cleanliness (25%)
|
|
19
|
+
- Structural integrity (20%)
|
|
20
|
+
- Metadata completeness (10%)
|
|
21
|
+
- Content density (15%)
|
|
22
|
+
|
|
23
|
+
- **Supported formats**: PDF (text + scanned), DOCX, TXT, CSV/TSV, HTML, Markdown, XLSX, PPTX, IPYNB, SRT
|
|
24
|
+
|
|
25
|
+
- **Corpus Auditor** (`ragpreflight audit`) — scan entire directories with:
|
|
26
|
+
- Near-duplicate detection via MinHash LSH (datasketch) or shingle Jaccard fallback
|
|
27
|
+
- File-age freshness proxy (not a content-staleness guarantee — see F1)
|
|
28
|
+
- Conflict candidate detection for human review (high embedding similarity pairs)
|
|
29
|
+
- Format distribution statistics and worst-offender list
|
|
30
|
+
|
|
31
|
+
- **Chunk Diagnostics** (`ragpreflight chunks`) — analyse chunked text for:
|
|
32
|
+
- Semantic coherence via sentence-transformers (optional; returns `None` not fake `1.0` when absent)
|
|
33
|
+
- Structural break detection (tables, lists, code blocks)
|
|
34
|
+
- Boundary quality (mid-sentence cuts)
|
|
35
|
+
- Information density
|
|
36
|
+
|
|
37
|
+
- **Retrieval Simulation** (`ragpreflight simulate`) — synthetic query generation + similarity hit rate:
|
|
38
|
+
- Named `synthetic_retrieval_hit_rate` (not `precision@k` — no labeled relevance)
|
|
39
|
+
- Dead chunk rate, query failure rate
|
|
40
|
+
- Optional LLM-powered query generation (`--use-llm`)
|
|
41
|
+
|
|
42
|
+
- **Garani 2026 Taxonomy** (`ragpreflight taxonomy`, `ragpreflight coverage`):
|
|
43
|
+
- All 33 failure modes across 7 pipeline stages, loaded from YAML
|
|
44
|
+
- Evidence levels: 9 Strong, 12 Moderate, 12 Limited
|
|
45
|
+
- `detector_status` per mode: direct / proxy / risk_signal / runtime_required / unsupported
|
|
46
|
+
- `TaxonomyReference` on every `Issue` linking findings to F1–F33
|
|
47
|
+
|
|
48
|
+
- **Quality Profiles**: `permissive`, `standard`, `strict`
|
|
49
|
+
|
|
50
|
+
- **Output Formats**: rich terminal tables, JSON (`--json`), self-contained HTML (`--format html`), SARIF (`--format sarif`)
|
|
51
|
+
|
|
52
|
+
- **CI/CD friendly**: `ragpreflight score <file>` prints only the score; exit code 1 on critical issues
|
|
53
|
+
|
|
54
|
+
- **Config file**: `.ragpreflight.toml` in project root or home directory
|
|
55
|
+
|
|
56
|
+
- **Zero telemetry**: no tracking, no API keys for core functionality, fully offline-capable
|
|
57
|
+
|
|
58
|
+
### Fixed (vs original ragcheck codebase)
|
|
59
|
+
|
|
60
|
+
- `coherence_score` fallback was returning fake `1.0` when sentence-transformers absent — now returns `None`
|
|
61
|
+
- `precision_at_k` renamed to `synthetic_retrieval_hit_rate` (no labeled relevance existed)
|
|
62
|
+
- Staleness message now says "file-age proxy" — filesystem mtime does not prove content outdated
|
|
63
|
+
- `_find_contradiction_candidates` renamed to `_find_conflict_candidates` — these are not confirmed contradictions
|
|
64
|
+
- `_scan_xlsx` docstring falsely claimed `.xls` support — openpyxl only handles `.xlsx`
|
|
65
|
+
- `[cloud]` extra removed — no connector module existed
|
|
66
|
+
- `scikit-learn` removed from `[full]` — was unused
|
|
67
|
+
- Python minimum bumped to `>=3.10` (3.9 EOL)
|
|
68
|
+
|
|
69
|
+
[Unreleased]: https://github.com/anusky95/ragpreflight/compare/v0.1.0...HEAD
|
|
70
|
+
[0.1.0]: https://github.com/anusky95/ragpreflight/releases/tag/v0.1.0
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
cff-version: 1.2.0
|
|
2
|
+
message: "If you use this software, please cite both the software and the paper below."
|
|
3
|
+
type: software
|
|
4
|
+
title: "ragpreflight"
|
|
5
|
+
abstract: "Pre-ingestion RAG document quality audit tool grounded in the Garani 2026 systematic taxonomy of 33 RAG failure modes across 7 pipeline stages."
|
|
6
|
+
authors:
|
|
7
|
+
- family-names: Garani
|
|
8
|
+
given-names: Anupama
|
|
9
|
+
email: anupamagarani95@gmail.com
|
|
10
|
+
orcid: "https://orcid.org/0000-0000-0000-0000"
|
|
11
|
+
repository-code: "https://github.com/anusky95/ragpreflight"
|
|
12
|
+
url: "https://pypi.org/project/ragpreflight/"
|
|
13
|
+
license: MIT
|
|
14
|
+
version: "0.1.0"
|
|
15
|
+
date-released: "2026-09-29"
|
|
16
|
+
keywords:
|
|
17
|
+
- RAG
|
|
18
|
+
- retrieval-augmented-generation
|
|
19
|
+
- document-quality
|
|
20
|
+
- pre-ingestion
|
|
21
|
+
- failure-modes
|
|
22
|
+
- taxonomy
|
|
23
|
+
- NLP
|
|
24
|
+
preferred-citation:
|
|
25
|
+
type: conference-paper
|
|
26
|
+
title: "A Systematic Taxonomy of Failure Modes in Retrieval-Augmented Generation Systems"
|
|
27
|
+
authors:
|
|
28
|
+
- family-names: Garani
|
|
29
|
+
given-names: Anupama
|
|
30
|
+
collection-title: "Proceedings of the 6th Workshop on Trustworthy Natural Language Processing (TrustNLP 2026)"
|
|
31
|
+
publisher: "Association for Computational Linguistics"
|
|
32
|
+
year: 2026
|
|
33
|
+
doi: "10.18653/v1/2026.trustnlp-main.27"
|
|
34
|
+
url: "https://aclanthology.org/2026.trustnlp-main.27/"
|
|
@@ -0,0 +1,354 @@
|
|
|
1
|
+
# CLAUDE.md — RAGCheck Project Brief
|
|
2
|
+
|
|
3
|
+
## What Is This
|
|
4
|
+
|
|
5
|
+
RAGCheck is an open-source Python package that scores document readiness for RAG pipelines. It tells you what's broken in your documents BEFORE you embed and retrieve them — not after your chatbot starts hallucinating.
|
|
6
|
+
|
|
7
|
+
**One-liner:** `pip install ragcheck` → `ragcheck scan my_document.pdf` → readiness score 0-100 with actionable issues.
|
|
8
|
+
|
|
9
|
+
## Design Principles (Follow These Always)
|
|
10
|
+
|
|
11
|
+
1. **Zero API keys for core functionality.** Everything runs locally by default. LLM-powered features are optional extras behind `--use-llm` flags. Never make an external API call without the user explicitly opting in.
|
|
12
|
+
2. **Lightweight base install.** Core dependencies: `pymupdf`, `pdfplumber`, `chardet`, `click`, `rich`. Heavy deps like `sentence-transformers` and `datasketch` are optional extras (`pip install ragcheck[full]`).
|
|
13
|
+
3. **Never crash on bad input.** Malformed PDFs, 500MB files, zip bombs disguised as docs, empty files, binary garbage — handle all of it gracefully with clear error messages. Wrap all file I/O in try/except and return meaningful errors, not tracebacks.
|
|
14
|
+
4. **Typed dataclasses for all outputs.** Every public function returns typed dataclasses, not raw dicts. Users get autocomplete and type checking for free.
|
|
15
|
+
5. **Tests alongside code.** Every module gets a corresponding test file. Write tests as you implement, not as a separate phase. Include fixture files in `tests/fixtures/` for reproducibility.
|
|
16
|
+
6. **Src layout.** Use `src/ragcheck/` package structure with `pyproject.toml`. No `setup.py`.
|
|
17
|
+
7. **Fail informatively.** When something goes wrong, the error message should tell the user what happened, what file caused it, and what to do about it.
|
|
18
|
+
|
|
19
|
+
## Target Users
|
|
20
|
+
|
|
21
|
+
- ML engineers building RAG pipelines who want to debug document quality issues
|
|
22
|
+
- Data engineers responsible for ingestion pipelines
|
|
23
|
+
- Teams evaluating vendor document processing tools
|
|
24
|
+
- Anyone who has ever asked "why is my RAG system hallucinating?"
|
|
25
|
+
|
|
26
|
+
## Package Structure
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
ragcheck/
|
|
30
|
+
├── CLAUDE.md # This file
|
|
31
|
+
├── LICENSE # MIT
|
|
32
|
+
├── README.md # Badges, install, quickstart, examples
|
|
33
|
+
├── pyproject.toml # Build config, dependencies, optional extras
|
|
34
|
+
├── .github/
|
|
35
|
+
│ └── workflows/
|
|
36
|
+
│ └── ci.yml # pytest + ruff + mypy
|
|
37
|
+
├── src/
|
|
38
|
+
│ └── ragcheck/
|
|
39
|
+
│ ├── __init__.py # Public API exports + __version__
|
|
40
|
+
│ ├── cli.py # Click CLI entry point
|
|
41
|
+
│ ├── models.py # All dataclasses (DocumentReport, ChunkReport, CorpusReport, Issue)
|
|
42
|
+
│ ├── scanner.py # Document scanning module
|
|
43
|
+
│ ├── chunker.py # Chunk analysis module
|
|
44
|
+
│ ├── corpus.py # Corpus-level analysis module
|
|
45
|
+
│ ├── retrieval.py # Retrieval simulation module
|
|
46
|
+
│ ├── profiles.py # Threshold profiles (strict, standard, permissive)
|
|
47
|
+
│ ├── report.py # Output formatting (terminal, JSON, HTML)
|
|
48
|
+
│ ├── utils.py # File detection, encoding helpers, size guards
|
|
49
|
+
│ └── _constants.py # OCR error patterns, default thresholds, magic numbers
|
|
50
|
+
├── tests/
|
|
51
|
+
│ ├── conftest.py # Shared fixtures
|
|
52
|
+
│ ├── fixtures/ # Test PDFs, DOCX, CSVs, etc.
|
|
53
|
+
│ │ ├── clean.pdf
|
|
54
|
+
│ │ ├── ocr_errors.pdf
|
|
55
|
+
│ │ ├── scanned.pdf
|
|
56
|
+
│ │ ├── messy_table.pdf
|
|
57
|
+
│ │ ├── empty.pdf
|
|
58
|
+
│ │ ├── sample.docx
|
|
59
|
+
│ │ ├── sample.csv
|
|
60
|
+
│ │ └── sample.md
|
|
61
|
+
│ ├── test_scanner.py
|
|
62
|
+
│ ├── test_chunker.py
|
|
63
|
+
│ ├── test_corpus.py
|
|
64
|
+
│ ├── test_retrieval.py
|
|
65
|
+
│ ├── test_cli.py
|
|
66
|
+
│ └── test_utils.py
|
|
67
|
+
└── docs/
|
|
68
|
+
└── profiles.md # Explain strict/standard/permissive thresholds
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Data Models (Core Contract — Do Not Deviate)
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from dataclasses import dataclass, field
|
|
75
|
+
from enum import Enum
|
|
76
|
+
from typing import Optional
|
|
77
|
+
|
|
78
|
+
class Severity(Enum):
|
|
79
|
+
CRITICAL = "critical" # Will definitely cause RAG failures
|
|
80
|
+
WARNING = "warning" # Likely to cause issues
|
|
81
|
+
INFO = "info" # Worth knowing, may not cause problems
|
|
82
|
+
|
|
83
|
+
class IssueCategory(Enum):
|
|
84
|
+
OCR = "ocr"
|
|
85
|
+
ENCODING = "encoding"
|
|
86
|
+
STRUCTURE = "structure"
|
|
87
|
+
CONTENT = "content"
|
|
88
|
+
METADATA = "metadata"
|
|
89
|
+
CHUNKING = "chunking"
|
|
90
|
+
DUPLICATION = "duplication"
|
|
91
|
+
STALENESS = "staleness"
|
|
92
|
+
|
|
93
|
+
@dataclass
|
|
94
|
+
class Issue:
|
|
95
|
+
category: IssueCategory
|
|
96
|
+
severity: Severity
|
|
97
|
+
message: str # Human-readable description
|
|
98
|
+
location: Optional[str] = None # e.g., "page 3", "rows 12-15", "chunk 7"
|
|
99
|
+
suggestion: Optional[str] = None # Actionable fix
|
|
100
|
+
|
|
101
|
+
@dataclass
|
|
102
|
+
class DocumentReport:
|
|
103
|
+
filepath: str
|
|
104
|
+
score: int # 0-100, overall readiness
|
|
105
|
+
issues: list[Issue] = field(default_factory=list)
|
|
106
|
+
page_count: int = 0
|
|
107
|
+
text_extractable_ratio: float = 0.0 # 0.0-1.0, how much text is extractable
|
|
108
|
+
file_format: str = ""
|
|
109
|
+
file_size_bytes: int = 0
|
|
110
|
+
|
|
111
|
+
@dataclass
|
|
112
|
+
class ChunkReport:
|
|
113
|
+
chunk_index: int
|
|
114
|
+
text_preview: str # First 100 chars
|
|
115
|
+
coherence_score: float # 0.0-1.0
|
|
116
|
+
issues: list[Issue] = field(default_factory=list)
|
|
117
|
+
|
|
118
|
+
@dataclass
|
|
119
|
+
class CorpusReport:
|
|
120
|
+
directory: str
|
|
121
|
+
total_documents: int
|
|
122
|
+
average_score: float
|
|
123
|
+
documents: list[DocumentReport] = field(default_factory=list)
|
|
124
|
+
corpus_issues: list[Issue] = field(default_factory=list) # Cross-document issues
|
|
125
|
+
duplicate_groups: list[list[str]] = field(default_factory=list) # Groups of near-dupes
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
## CLI Interface
|
|
129
|
+
|
|
130
|
+
```bash
|
|
131
|
+
# Single document scan
|
|
132
|
+
ragcheck scan document.pdf
|
|
133
|
+
ragcheck scan document.pdf --json
|
|
134
|
+
ragcheck scan document.pdf --profile strict
|
|
135
|
+
|
|
136
|
+
# Full corpus audit
|
|
137
|
+
ragcheck audit ./knowledge_base/
|
|
138
|
+
ragcheck audit ./knowledge_base/ --format html --output report.html
|
|
139
|
+
ragcheck audit ./knowledge_base/ --profile medical
|
|
140
|
+
|
|
141
|
+
# Chunk analysis (requires optional deps)
|
|
142
|
+
ragcheck chunks document.pdf --strategy recursive --size 512 --overlap 50
|
|
143
|
+
|
|
144
|
+
# Retrieval simulation (requires optional deps)
|
|
145
|
+
ragcheck simulate ./knowledge_base/ --queries 50
|
|
146
|
+
ragcheck simulate ./knowledge_base/ --use-llm --llm-endpoint http://localhost:11434
|
|
147
|
+
|
|
148
|
+
# Quick check — just the score, nothing else
|
|
149
|
+
ragcheck score document.pdf
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## Profiles (Threshold Presets)
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
PROFILES = {
|
|
156
|
+
"permissive": {
|
|
157
|
+
"min_document_score": 40,
|
|
158
|
+
"min_chunk_coherence": 0.5,
|
|
159
|
+
"max_ocr_error_rate": 0.10,
|
|
160
|
+
"max_duplicate_similarity": 0.95,
|
|
161
|
+
"description": "For chatbots, FAQ systems, internal tools"
|
|
162
|
+
},
|
|
163
|
+
"standard": {
|
|
164
|
+
"min_document_score": 60,
|
|
165
|
+
"min_chunk_coherence": 0.65,
|
|
166
|
+
"max_ocr_error_rate": 0.05,
|
|
167
|
+
"max_duplicate_similarity": 0.90,
|
|
168
|
+
"description": "For enterprise search, customer support RAG"
|
|
169
|
+
},
|
|
170
|
+
"strict": {
|
|
171
|
+
"min_document_score": 80,
|
|
172
|
+
"min_chunk_coherence": 0.8,
|
|
173
|
+
"max_ocr_error_rate": 0.02,
|
|
174
|
+
"max_duplicate_similarity": 0.85,
|
|
175
|
+
"description": "For medical, legal, financial RAG systems"
|
|
176
|
+
}
|
|
177
|
+
}
|
|
178
|
+
```
|
|
179
|
+
|
|
180
|
+
## Supported File Formats
|
|
181
|
+
|
|
182
|
+
| Format | Base Install | Detection |
|
|
183
|
+
|--------|-------------|-----------|
|
|
184
|
+
| PDF (text-based) | ✅ pymupdf + pdfplumber | OCR errors, structure, tables |
|
|
185
|
+
| PDF (scanned) | ✅ pymupdf | Flags as low-extractability, suggests OCR |
|
|
186
|
+
| DOCX | ✅ python-docx | Structure, metadata, encoding |
|
|
187
|
+
| TXT | ✅ built-in | Encoding, content density |
|
|
188
|
+
| CSV/TSV | ✅ built-in | Encoding, structure, consistency |
|
|
189
|
+
| HTML | ✅ beautifulsoup4 | Boilerplate ratio, structure |
|
|
190
|
+
| Markdown | ✅ built-in | Structure, header hierarchy |
|
|
191
|
+
|
|
192
|
+
## OCR Error Detection Patterns
|
|
193
|
+
|
|
194
|
+
These are the common substitution patterns to check. Store in `_constants.py`:
|
|
195
|
+
|
|
196
|
+
```python
|
|
197
|
+
OCR_SUBSTITUTION_PATTERNS = {
|
|
198
|
+
# Character confusions
|
|
199
|
+
"l/I/1": r"(?<=[a-z])[I1](?=[a-z])", # "cIinical" → "clinical"
|
|
200
|
+
"O/0": r"(?<=[a-z])0(?=[a-z])", # "pr0tocol" → "protocol"
|
|
201
|
+
"rn/m": r"(?<=[a-z])rn(?=[a-z])", # "inforrnation" → "information"
|
|
202
|
+
"fi/fl_ligature": r"[fifl]", # Ligature artifacts
|
|
203
|
+
"broken_ligatures": r"(?<=\w)[ffi](?=\w)", # Split ligatures
|
|
204
|
+
# Spacing artifacts
|
|
205
|
+
"mid_word_spaces": r"(?<=[a-z]) (?=[a-z]{2})", # "pati ent" → "patient"
|
|
206
|
+
"merged_words": None, # Detected via dictionary lookup, not regex
|
|
207
|
+
# Encoding artifacts
|
|
208
|
+
"mojibake": r"[â’©é]", # UTF-8 decoded as Latin-1
|
|
209
|
+
}
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
## What NOT to Build
|
|
213
|
+
|
|
214
|
+
- **No web UI.** This is a CLI/library tool. Web dashboards can come from the community.
|
|
215
|
+
- **No database.** Reports are generated on-the-fly. Users can pipe JSON to wherever they want.
|
|
216
|
+
- **No user accounts or telemetry.** Zero tracking, zero phoning home.
|
|
217
|
+
- **No custom ML models.** Use existing sentence-transformers and standard NLP tools. Don't train anything.
|
|
218
|
+
- **No vendor lock-in.** The `--use-llm` flag works with any OpenAI-compatible endpoint (OpenAI, Ollama, vLLM, Azure, etc.), not just one provider.
|
|
219
|
+
|
|
220
|
+
## Implementation Phases
|
|
221
|
+
|
|
222
|
+
### Phase 1: Scaffold
|
|
223
|
+
Set up the package structure exactly as described above. `pyproject.toml` with:
|
|
224
|
+
- Build system: hatchling
|
|
225
|
+
- Python: >=3.9
|
|
226
|
+
- Core deps: pymupdf, pdfplumber, chardet, click, rich, python-docx, beautifulsoup4
|
|
227
|
+
- Optional extras: `[full]` = sentence-transformers, datasketch; `[llm]` = openai
|
|
228
|
+
- CLI entry point: `ragcheck = ragcheck.cli:main`
|
|
229
|
+
- Linting: ruff
|
|
230
|
+
- Type checking: mypy
|
|
231
|
+
- Testing: pytest
|
|
232
|
+
|
|
233
|
+
Create a working CLI that responds to `ragcheck --version` and `ragcheck scan --help`.
|
|
234
|
+
Write README.md with badges (PyPI, Python version, License, CI), one-liner install, quickstart example, and feature comparison table vs RAGAS/Unstructured/LlamaIndex.
|
|
235
|
+
|
|
236
|
+
### Phase 2: Document Scanner (`scanner.py`)
|
|
237
|
+
Build the core scanning engine. For each supported file format, detect:
|
|
238
|
+
- **Encoding issues** via chardet (flag confidence < 0.8)
|
|
239
|
+
- **Text extractable ratio** — what percentage of pages/content yield actual text vs images/scans
|
|
240
|
+
- **OCR error patterns** using the regex patterns in `_constants.py`
|
|
241
|
+
- **Character-level anomalies** — unusual Unicode, control characters, excessive whitespace
|
|
242
|
+
- **Empty or near-empty pages** (< 10 words)
|
|
243
|
+
- **Embedded tables or structured data** that need special chunking (detect via cell/row patterns)
|
|
244
|
+
- **Metadata presence** — does the doc have title, author, date, or is it bare?
|
|
245
|
+
- **File size sanity** — flag files > 100MB with a warning, reject > 500MB by default (configurable)
|
|
246
|
+
|
|
247
|
+
The scanner returns a `DocumentReport` with a score 0-100 computed as weighted sum:
|
|
248
|
+
- Text extractability: 30%
|
|
249
|
+
- OCR cleanliness: 25%
|
|
250
|
+
- Structural integrity: 20%
|
|
251
|
+
- Metadata completeness: 10%
|
|
252
|
+
- Content density: 15%
|
|
253
|
+
|
|
254
|
+
Write tests for each file format with both clean and problematic fixtures.
|
|
255
|
+
|
|
256
|
+
### Phase 3: Chunk Diagnostics (`chunker.py`)
|
|
257
|
+
Build the chunk analysis module. Given a document and a chunking strategy:
|
|
258
|
+
- Default: recursive character split at 512 tokens, 50 token overlap
|
|
259
|
+
- Accept a pluggable splitter function: `Callable[[str], list[str]]`
|
|
260
|
+
|
|
261
|
+
For each chunk, score:
|
|
262
|
+
- **Semantic coherence** using sentence-transformers `all-MiniLM-L6-v2` (optional dep — if not installed, skip this metric and warn)
|
|
263
|
+
- **Structural breaks** — does the chunk split a table, list, or code block mid-element?
|
|
264
|
+
- **Metadata preservation** — does the chunk retain source page number, section header context?
|
|
265
|
+
- **Information density** — ratio of meaningful tokens vs boilerplate/stopwords
|
|
266
|
+
- **Boundary quality** — does the chunk start/end mid-sentence?
|
|
267
|
+
|
|
268
|
+
Return a list of `ChunkReport` objects. Flag chunks below profile thresholds.
|
|
269
|
+
Make chunking strategy pluggable so users can bring their own splitter.
|
|
270
|
+
|
|
271
|
+
### Phase 4: Corpus Analysis (`corpus.py`)
|
|
272
|
+
Build the corpus-level analysis module. Given a directory of documents:
|
|
273
|
+
- Run `scanner.py` on each file
|
|
274
|
+
- **Near-duplicate detection** using MinHash/LSH via datasketch (optional dep)
|
|
275
|
+
- **Contradiction candidates** — find doc pairs covering the same topic (embedding similarity > 0.85) with different factual claims. Surface pairs for human review, don't auto-judge.
|
|
276
|
+
- **Temporal coverage** — if docs have detectable dates (file metadata, content regex), flag docs older than a configurable threshold (default: 12 months)
|
|
277
|
+
- **Format distribution** — breakdown by file type
|
|
278
|
+
- **Worst offenders** — bottom 10% by score
|
|
279
|
+
- **Corpus statistics** — total docs, average score, score distribution
|
|
280
|
+
|
|
281
|
+
Return a `CorpusReport`. Show a progress bar via `rich` for directories with 5+ files.
|
|
282
|
+
|
|
283
|
+
### Phase 5: Retrieval Simulation (`retrieval.py`)
|
|
284
|
+
Build the retrieval simulation module. Generate synthetic test queries WITHOUT requiring any LLM API:
|
|
285
|
+
- **Extractive methods**: Pull key phrases (TF-IDF top terms), named entities (simple regex-based NER for common types), and section headers → form natural question templates
|
|
286
|
+
- Template patterns: "What is {entity}?", "Explain {concept}", "Summarize {section_title}", "What are the details of {key_phrase}?"
|
|
287
|
+
- Generate 5-10 queries per document by default
|
|
288
|
+
|
|
289
|
+
Test retrieval:
|
|
290
|
+
- Embed queries and chunks using sentence-transformers
|
|
291
|
+
- Compute cosine similarity, retrieve top-k (default k=5)
|
|
292
|
+
- Score: precision@k, percentage of dead chunks (never retrieved by any query), query failure rate (best match similarity < 0.5)
|
|
293
|
+
|
|
294
|
+
Optional `--use-llm` flag:
|
|
295
|
+
- Accepts any OpenAI-compatible endpoint via `--llm-endpoint` (default: http://localhost:11434 for Ollama)
|
|
296
|
+
- Uses the LLM to generate more diverse, natural queries
|
|
297
|
+
- Falls back to extractive if LLM endpoint is unreachable
|
|
298
|
+
|
|
299
|
+
### Phase 6: CLI & Reporting (`cli.py`, `report.py`)
|
|
300
|
+
Polish the CLI commands:
|
|
301
|
+
- `ragcheck scan <file>` — single document, terminal table output by default
|
|
302
|
+
- `ragcheck audit <directory>` — full corpus audit with progress bar
|
|
303
|
+
- `ragcheck chunks <file>` — chunk analysis (warns if optional deps missing)
|
|
304
|
+
- `ragcheck simulate <directory>` — retrieval simulation (warns if optional deps missing)
|
|
305
|
+
- `ragcheck score <file>` — just the number, for scripting (`echo $(ragcheck score doc.pdf)`)
|
|
306
|
+
|
|
307
|
+
Output formats:
|
|
308
|
+
- Default: `rich` terminal table with color-coded severity
|
|
309
|
+
- `--json` flag: structured JSON to stdout
|
|
310
|
+
- `--format html --output report.html`: self-contained HTML report with expandable sections
|
|
311
|
+
- `--quiet` flag: suppress all output except the score (for CI/CD pipelines)
|
|
312
|
+
|
|
313
|
+
Config file support:
|
|
314
|
+
- `.ragcheckrc` (TOML format) in project root or home directory
|
|
315
|
+
- Override profile, thresholds, file size limits, output preferences
|
|
316
|
+
- CLI flags override config file values
|
|
317
|
+
|
|
318
|
+
### Phase 7: Harden for Release
|
|
319
|
+
Review entire codebase for:
|
|
320
|
+
- **Type hints** on every function signature and return type
|
|
321
|
+
- **Docstrings** on every public method (Google style)
|
|
322
|
+
- **Input validation** — clear ValueError/TypeError with actionable messages
|
|
323
|
+
- **Edge cases** — empty directories, permission errors, symlinks, binary files with wrong extensions
|
|
324
|
+
- **Security** — no path traversal, no arbitrary code execution, no eval/exec, no pickling untrusted data, size limits on all file reads
|
|
325
|
+
- **CI pipeline** — GitHub Actions running pytest, ruff, mypy on Python 3.9/3.10/3.11/3.12
|
|
326
|
+
- **`--verbose` flag** for debug logging (use stdlib `logging`, not print)
|
|
327
|
+
- **CONTRIBUTING.md** with dev setup instructions, PR guidelines, and code style expectations
|
|
328
|
+
- **CHANGELOG.md** with initial release notes
|
|
329
|
+
|
|
330
|
+
### Phase 8: Integration & Dogfooding
|
|
331
|
+
Wire up the full pipeline so `ragcheck audit mydir/` runs scanning → chunk analysis → corpus analysis in sequence with a unified report. Each module stays independently importable.
|
|
332
|
+
|
|
333
|
+
Then run ragcheck against:
|
|
334
|
+
1. The test fixtures in `tests/fixtures/`
|
|
335
|
+
2. The ragcheck source code itself (`.py` files as TXT)
|
|
336
|
+
3. The README.md
|
|
337
|
+
4. A deliberately corrupted PDF (create one in tests)
|
|
338
|
+
|
|
339
|
+
Fix any crashes, confusing output, or scores that seem wrong. Verify that:
|
|
340
|
+
- A clean, well-structured PDF scores > 85
|
|
341
|
+
- A scanned PDF with OCR errors scores < 50
|
|
342
|
+
- An empty file scores 0 with a clear CRITICAL issue
|
|
343
|
+
- The CLI never shows a traceback to the user (all errors are caught and formatted)
|
|
344
|
+
|
|
345
|
+
## Code Style Rules
|
|
346
|
+
|
|
347
|
+
- Use ruff for formatting and linting (line length 99)
|
|
348
|
+
- Use Google-style docstrings
|
|
349
|
+
- Prefer early returns over deep nesting
|
|
350
|
+
- No global mutable state
|
|
351
|
+
- No `print()` — use `click.echo()` in CLI, `logging` everywhere else
|
|
352
|
+
- Constants in SCREAMING_SNAKE_CASE in `_constants.py`
|
|
353
|
+
- Private functions prefixed with underscore
|
|
354
|
+
- No `# type: ignore` without a comment explaining why
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# Contributing to RAGCheck
|
|
2
|
+
|
|
3
|
+
Thank you for your interest in contributing! This guide will help you set up your development environment and understand our conventions.
|
|
4
|
+
|
|
5
|
+
## Dev Setup
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
# 1. Fork and clone
|
|
9
|
+
git clone https://github.com/your-fork/ragcheck.git
|
|
10
|
+
cd ragcheck
|
|
11
|
+
|
|
12
|
+
# 2. Create a virtual environment
|
|
13
|
+
python -m venv .venv
|
|
14
|
+
source .venv/bin/activate # Windows: .venv\Scripts\activate
|
|
15
|
+
|
|
16
|
+
# 3. Install in editable mode with dev dependencies
|
|
17
|
+
pip install -e ".[dev]"
|
|
18
|
+
|
|
19
|
+
# 4. (Optional) Full extras for all features
|
|
20
|
+
pip install -e ".[dev,full,llm]"
|
|
21
|
+
|
|
22
|
+
# 5. Verify setup
|
|
23
|
+
ragcheck --version
|
|
24
|
+
pytest -q
|
|
25
|
+
```
|
|
26
|
+
|
|
27
|
+
## Running Tests
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
pytest # all tests
|
|
31
|
+
pytest tests/test_scanner.py # one module
|
|
32
|
+
pytest -k "test_scan" # by name pattern
|
|
33
|
+
pytest --cov=ragcheck # with coverage
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
## Code Style
|
|
37
|
+
|
|
38
|
+
We use **ruff** for formatting and linting. Line length is 99 characters.
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
ruff check src/ tests/ # lint
|
|
42
|
+
ruff format src/ tests/ # auto-format
|
|
43
|
+
mypy src/ragcheck/ # type check
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
All of the above run automatically in CI. Your PR must pass lint + tests before it can be merged.
|
|
47
|
+
|
|
48
|
+
## Design Principles (Read Before Contributing)
|
|
49
|
+
|
|
50
|
+
1. **Zero API keys for core functionality.** LLM-powered features must be behind `--use-llm` flags.
|
|
51
|
+
2. **Lightweight base install.** Heavy deps go in `[full]` or `[llm]` extras — never in core.
|
|
52
|
+
3. **Never crash on bad input.** Wrap all file I/O in try/except. Return errors, not tracebacks.
|
|
53
|
+
4. **Typed dataclasses for all outputs.** Public functions return typed dataclasses, not dicts.
|
|
54
|
+
5. **Tests alongside code.** Every module has a corresponding test file.
|
|
55
|
+
6. **No `print()` in library code.** Use `logging` everywhere; `click.echo()` in CLI only.
|
|
56
|
+
|
|
57
|
+
## PR Guidelines
|
|
58
|
+
|
|
59
|
+
- Keep PRs focused: one feature or fix per PR
|
|
60
|
+
- Update tests alongside your changes
|
|
61
|
+
- Add your change to `CHANGELOG.md` under `[Unreleased]`
|
|
62
|
+
- Make sure `pytest` and `ruff check` pass locally before opening a PR
|
|
63
|
+
|
|
64
|
+
## Adding a New File Format
|
|
65
|
+
|
|
66
|
+
1. Add the extension to `SUPPORTED_EXTENSIONS` in `_constants.py`
|
|
67
|
+
2. Write a `_scan_<format>` function in `scanner.py` following the existing pattern
|
|
68
|
+
3. Register it in `_FORMAT_SCANNERS` in `scanner.py`
|
|
69
|
+
4. Add at least one test fixture in `tests/fixtures/`
|
|
70
|
+
5. Write tests in `tests/test_scanner.py`
|
|
71
|
+
|
|
72
|
+
## Reporting Bugs
|
|
73
|
+
|
|
74
|
+
Open an issue at [github.com/ragcheck/ragcheck/issues](https://github.com/ragcheck/ragcheck/issues) with:
|
|
75
|
+
- RAGCheck version (`ragcheck --version`)
|
|
76
|
+
- Python version
|
|
77
|
+
- OS
|
|
78
|
+
- Command or code that triggered the bug
|
|
79
|
+
- Full error output (with `--verbose` if applicable)
|