docsmind 0.1.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docsmind-0.3.0/.gitignore +63 -0
- docsmind-0.3.0/CHANGELOG.md +110 -0
- docsmind-0.3.0/PKG-INFO +321 -0
- docsmind-0.3.0/README.md +264 -0
- docsmind-0.3.0/pyproject.toml +92 -0
- docsmind-0.3.0/src/docsmind/__init__.py +29 -0
- docsmind-0.3.0/src/docsmind/adapters/blobs/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/adapters/blobs/filesystem.py +39 -0
- docsmind-0.3.0/src/docsmind/adapters/embeddings/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/adapters/llm/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/adapters/storage/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/adapters/storage/postgres/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/adapters/storage/postgres/engine.py +24 -0
- docsmind-0.3.0/src/docsmind/adapters/storage/postgres/repository.py +393 -0
- docsmind-0.3.0/src/docsmind/adapters/storage/postgres/tables.py +162 -0
- docsmind-0.3.0/src/docsmind/adapters/tracing/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/adapters/vision/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/agent/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/agent/subagents/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/agent/subagents/change_analysis/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/agent/tools/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/cli/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/cli/app.py +43 -0
- docsmind-0.3.0/src/docsmind/cli/chat.py +26 -0
- docsmind-0.3.0/src/docsmind/cli/db.py +44 -0
- docsmind-0.3.0/src/docsmind/cli/diff.py +32 -0
- docsmind-0.3.0/src/docsmind/cli/doctor.py +86 -0
- docsmind-0.3.0/src/docsmind/cli/ingest.py +27 -0
- docsmind-0.3.0/src/docsmind/cli/inspect.py +209 -0
- docsmind-0.3.0/src/docsmind/cli/serve.py +29 -0
- docsmind-0.3.0/src/docsmind/client.py +94 -0
- docsmind-0.3.0/src/docsmind/config.py +51 -0
- docsmind-0.3.0/src/docsmind/domain/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/domain/changes.py +66 -0
- docsmind-0.3.0/src/docsmind/domain/citations.py +26 -0
- docsmind-0.3.0/src/docsmind/domain/documents.py +156 -0
- docsmind-0.3.0/src/docsmind/domain/errors.py +33 -0
- docsmind-0.3.0/src/docsmind/domain/ports.py +81 -0
- docsmind-0.3.0/src/docsmind/domain/tables.py +53 -0
- docsmind-0.3.0/src/docsmind/ingestion/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/ingestion/chunking/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/ingestion/chunking/page.py +129 -0
- docsmind-0.3.0/src/docsmind/ingestion/chunking/structural.py +61 -0
- docsmind-0.3.0/src/docsmind/ingestion/enrichment/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/ingestion/parsers/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/ingestion/parsers/pdf.py +590 -0
- docsmind-0.3.0/src/docsmind/ingestion/pipeline.py +115 -0
- docsmind-0.3.0/src/docsmind/py.typed +0 -0
- docsmind-0.3.0/src/docsmind/retrieval/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/server/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/server/app.py +88 -0
- docsmind-0.3.0/src/docsmind/server/routes/__init__.py +0 -0
- docsmind-0.3.0/src/docsmind/server/routes/documents.py +196 -0
- docsmind-0.3.0/src/docsmind/server/ui/assets/index-DSQOFo_H.css +1 -0
- docsmind-0.3.0/src/docsmind/server/ui/assets/index-QBP12Bcv.js +5385 -0
- docsmind-0.3.0/src/docsmind/server/ui/favicon.svg +1 -0
- docsmind-0.3.0/src/docsmind/server/ui/icons.svg +24 -0
- docsmind-0.3.0/src/docsmind/server/ui/index.html +14 -0
- docsmind-0.1.0/CHANGELOG.md +0 -9
- docsmind-0.1.0/PKG-INFO +0 -22
- docsmind-0.1.0/README.md +0 -3
- docsmind-0.1.0/pyproject.toml +0 -35
- docsmind-0.1.0/uv.lock +0 -733
- {docsmind-0.1.0/src/docsmind → docsmind-0.3.0/src/docsmind/adapters}/__init__.py +0 -0
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# --- Python ---
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
*.so
|
|
6
|
+
.Python
|
|
7
|
+
build/
|
|
8
|
+
dist/
|
|
9
|
+
*.egg-info/
|
|
10
|
+
.eggs/
|
|
11
|
+
*.egg
|
|
12
|
+
|
|
13
|
+
# --- Virtual environments ---
|
|
14
|
+
.venv/
|
|
15
|
+
venv/
|
|
16
|
+
env/
|
|
17
|
+
ENV/
|
|
18
|
+
|
|
19
|
+
# --- Testing / linting / type-checking ---
|
|
20
|
+
.pytest_cache/
|
|
21
|
+
.mypy_cache/
|
|
22
|
+
.ruff_cache/
|
|
23
|
+
.coverage
|
|
24
|
+
.coverage.*
|
|
25
|
+
htmlcov/
|
|
26
|
+
coverage.xml
|
|
27
|
+
|
|
28
|
+
# --- uv / pip ---
|
|
29
|
+
uv.lock.bak
|
|
30
|
+
|
|
31
|
+
# --- docsmind local runtime artifacts ---
|
|
32
|
+
# Local filesystem BlobStore (extracted images from ingestion) — never
|
|
33
|
+
# something to commit; it's per-machine, per-ingest state.
|
|
34
|
+
.docsmind/
|
|
35
|
+
# Generated by scripts/make_test_pdf.py for local testing — not a committed
|
|
36
|
+
# fixture (tests/unit/test_pdf_parser.py generates its own PDF in-memory).
|
|
37
|
+
/sample.pdf
|
|
38
|
+
|
|
39
|
+
# --- Environment / secrets ---
|
|
40
|
+
.env
|
|
41
|
+
.env.*
|
|
42
|
+
!.env.example
|
|
43
|
+
|
|
44
|
+
# --- Postgres (local Docker volume mounts, if you ever add bind mounts) ---
|
|
45
|
+
*.sql.bak
|
|
46
|
+
|
|
47
|
+
# --- Editors / OS ---
|
|
48
|
+
.vscode/
|
|
49
|
+
.idea/
|
|
50
|
+
*.swp
|
|
51
|
+
.DS_Store
|
|
52
|
+
Thumbs.db
|
|
53
|
+
|
|
54
|
+
# --- Jupyter, if you use it for exploration ---
|
|
55
|
+
.ipynb_checkpoints/
|
|
56
|
+
|
|
57
|
+
# --- Alembic ---
|
|
58
|
+
# (versions/ IS tracked — only local scratch files some editors drop here)
|
|
59
|
+
alembic/versions/*.pyc
|
|
60
|
+
|
|
61
|
+
# --- ui/ (Vite/React build output, once ui/ has a real app in it) ---
|
|
62
|
+
ui/node_modules/
|
|
63
|
+
ui/dist/
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project will be documented in this file.
|
|
4
|
+
|
|
5
|
+
## [0.3.0] - 2026-10-03
|
|
6
|
+
|
|
7
|
+
### Added
|
|
8
|
+
|
|
9
|
+
- Bundled frontend UI in the Python package: release UI assets are copied into
|
|
10
|
+
`src/docsmind/server/ui` and served by `docsmind serve` at `/`.
|
|
11
|
+
- `scripts/bundle_frontend.py` helper to copy `../frontend/dist` into the package
|
|
12
|
+
before publishing.
|
|
13
|
+
- `docsmind delete <version-id> [-y]` command to remove one ingested version and
|
|
14
|
+
its derived sections/chunks/images/tables.
|
|
15
|
+
- HTTP `DELETE /api/documents/{version_id}` endpoint.
|
|
16
|
+
|
|
17
|
+
### Changed
|
|
18
|
+
|
|
19
|
+
- `docsmind serve` now serves the bundled SPA (when present), in addition to
|
|
20
|
+
`/api/*` and `/blobs/*`.
|
|
21
|
+
- Ingestion now checks `content_hash` before parsing and raises a friendly
|
|
22
|
+
"already ingested" error instead of surfacing a raw Postgres unique-constraint
|
|
23
|
+
failure.
|
|
24
|
+
|
|
25
|
+
### Fixed
|
|
26
|
+
|
|
27
|
+
- Image extraction de-duplicates repeated embedded images (e.g. page logo reused
|
|
28
|
+
across many pages) using perceptual hash, preventing inflated image counts.
|
|
29
|
+
- API duplicate-ingest responses now return HTTP 409 with structured details.
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
## [0.2.0] - 2026-09-27
|
|
34
|
+
|
|
35
|
+
First functional release: documents can now be ingested from PDF into
|
|
36
|
+
Postgres/pgvector and inspected from the command line. Chat, embeddings and
|
|
37
|
+
change analysis are scaffolded but not implemented yet (see "Not yet
|
|
38
|
+
implemented" below).
|
|
39
|
+
|
|
40
|
+
### Added
|
|
41
|
+
|
|
42
|
+
- **PDF ingestion pipeline** (`docsmind ingest <file> --collection <name>`),
|
|
43
|
+
built on PyMuPDF:
|
|
44
|
+
- Full document text is extracted and saved per version, together with
|
|
45
|
+
stats (character, word and estimated token counts).
|
|
46
|
+
- Outline is taken from the PDF's own table of contents/bookmarks when one
|
|
47
|
+
exists. Documents without a TOC get an empty outline; structure is
|
|
48
|
+
deliberately not guessed from fonts or numbering.
|
|
49
|
+
- Text is chunked by page for vector search: one chunk per page, with
|
|
50
|
+
unusually dense pages split into ~3,500-character pieces with a
|
|
51
|
+
100-character overlap.
|
|
52
|
+
- Images are extracted with their page and bounding box, skipping tiny
|
|
53
|
+
icons, and stored with a perceptual hash for future de-duplication. They
|
|
54
|
+
are saved under a readable path
|
|
55
|
+
(`images/<collection>/<title>/<version>-<id>/`) instead of a bare UUID.
|
|
56
|
+
- **Postgres storage** with pgvector: Alembic migrations for documents,
|
|
57
|
+
versions, sections, images, chunks (with a generated `tsvector` column for
|
|
58
|
+
keyword search) and editable agent tables with revision history. The vector
|
|
59
|
+
index uses `halfvec` so 3072-dimension embeddings
|
|
60
|
+
(`text-embedding-3-large`) fit within pgvector's HNSW limit.
|
|
61
|
+
- **Inspection commands**: `docsmind versions`, `docsmind fulltext`,
|
|
62
|
+
`docsmind outline` and `docsmind chunks` read back what was saved, so
|
|
63
|
+
results can be checked without writing SQL.
|
|
64
|
+
- **Database commands**: `docsmind db up` (local Docker Postgres with
|
|
65
|
+
pgvector), `docsmind db upgrade` (apply migrations) and `docsmind db psql`.
|
|
66
|
+
- `docsmind doctor` checks the database URL, blob storage and API keys, and
|
|
67
|
+
detects Neon/Supabase connection strings.
|
|
68
|
+
- `Docsmind` / `AsyncDocsmind` Python facade for ingesting documents.
|
|
69
|
+
- Domain models for documents, versions, sections, chunks, images, citations,
|
|
70
|
+
change sets and editable tables, with `Protocol` ports for LLMs, embedders,
|
|
71
|
+
vision captioners, vector stores, blob stores and parsers.
|
|
72
|
+
- Environment-driven configuration (`DOCSMIND_*` variables, `.env` support)
|
|
73
|
+
and a documented `.env.example`.
|
|
74
|
+
- Filesystem blob storage for extracted images.
|
|
75
|
+
- GitHub Actions: a test workflow across Python 3.10, 3.11 and 3.12 with a
|
|
76
|
+
wheel install check, and a publish workflow using PyPI trusted publishing.
|
|
77
|
+
- Unit tests for the PDF parser and page chunker, a runnable parser demo
|
|
78
|
+
(`examples/parse_pdf_demo.py`) and a synthetic test-PDF generator.
|
|
79
|
+
- `.gitignore`, `docker-compose.yml` and `.python-version`.
|
|
80
|
+
|
|
81
|
+
### Changed
|
|
82
|
+
|
|
83
|
+
- `--version` on `docsmind ingest` is now optional and defaults to today's
|
|
84
|
+
date.
|
|
85
|
+
- The local Docker Postgres listens on port 5433 instead of 5432 to avoid
|
|
86
|
+
clashing with a locally installed Postgres.
|
|
87
|
+
- Package metadata: author email, explicit Python 3.10-3.12 classifiers and
|
|
88
|
+
the issue tracker URL.
|
|
89
|
+
|
|
90
|
+
### Removed
|
|
91
|
+
|
|
92
|
+
- Heading detection from font size, boldness and numbering patterns, and
|
|
93
|
+
section-based chunking. Neither generalised across real-world PDFs, and
|
|
94
|
+
chunking no longer depends on document structure.
|
|
95
|
+
|
|
96
|
+
### Not yet implemented
|
|
97
|
+
|
|
98
|
+
- Embeddings and hybrid vector/keyword search. Chunks are stored without
|
|
99
|
+
embeddings, and `hybrid_search` raises `NotImplementedError`.
|
|
100
|
+
- Image captioning with a vision model.
|
|
101
|
+
- The chat agent behind `docsmind chat`.
|
|
102
|
+
- DOCX and plain-text parsers, S3/Supabase blob storage, and the server/UI.
|
|
103
|
+
- Multi-column reading order: PDFs with genuine multi-column layouts may have
|
|
104
|
+
their text interleaved in the saved full text.
|
|
105
|
+
|
|
106
|
+
## [0.1.0] - 2026-09-12
|
|
107
|
+
|
|
108
|
+
### Added
|
|
109
|
+
|
|
110
|
+
- Initial release to reserve the package name on PyPI.
|
docsmind-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,321 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: docsmind
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Chat with your documents. Version-aware retrieval, change analysis, and an editable table workspace, backed by Postgres/pgvector.
|
|
5
|
+
Project-URL: Homepage, https://github.com/yauheniya-ai/docsmind
|
|
6
|
+
Project-URL: Repository, https://github.com/yauheniya-ai/docsmind
|
|
7
|
+
Project-URL: Issues, https://github.com/yauheniya-ai/docsmind/issues
|
|
8
|
+
Author-email: Yauheniya Varabyova <yauheniya.ai@gmail.com>
|
|
9
|
+
License: MIT
|
|
10
|
+
Keywords: agent,compliance,documents,pgvector,rag
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Text Processing :: Indexing
|
|
20
|
+
Requires-Python: >=3.10
|
|
21
|
+
Requires-Dist: alembic>=1.13
|
|
22
|
+
Requires-Dist: asyncpg>=0.29
|
|
23
|
+
Requires-Dist: httpx>=0.27
|
|
24
|
+
Requires-Dist: imagehash>=4.3
|
|
25
|
+
Requires-Dist: pgvector>=0.2.5
|
|
26
|
+
Requires-Dist: pillow>=10.0
|
|
27
|
+
Requires-Dist: pydantic-settings>=2.2
|
|
28
|
+
Requires-Dist: pydantic>=2.6
|
|
29
|
+
Requires-Dist: pymupdf>=1.24
|
|
30
|
+
Requires-Dist: rich>=13.7
|
|
31
|
+
Requires-Dist: sqlalchemy[asyncio]>=2.0
|
|
32
|
+
Requires-Dist: tiktoken>=0.7
|
|
33
|
+
Requires-Dist: typer>=0.12
|
|
34
|
+
Provides-Extra: all
|
|
35
|
+
Requires-Dist: anthropic>=0.34; extra == 'all'
|
|
36
|
+
Requires-Dist: fastapi>=0.111; extra == 'all'
|
|
37
|
+
Requires-Dist: openai>=1.30; extra == 'all'
|
|
38
|
+
Requires-Dist: python-multipart>=0.0.9; extra == 'all'
|
|
39
|
+
Requires-Dist: uvicorn[standard]>=0.30; extra == 'all'
|
|
40
|
+
Provides-Extra: anthropic
|
|
41
|
+
Requires-Dist: anthropic>=0.34; extra == 'anthropic'
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
44
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
45
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
46
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
47
|
+
Requires-Dist: testcontainers[postgres]>=4.4; extra == 'dev'
|
|
48
|
+
Provides-Extra: mlflow
|
|
49
|
+
Requires-Dist: mlflow>=2.14; extra == 'mlflow'
|
|
50
|
+
Provides-Extra: openai
|
|
51
|
+
Requires-Dist: openai>=1.30; extra == 'openai'
|
|
52
|
+
Provides-Extra: server
|
|
53
|
+
Requires-Dist: fastapi>=0.111; extra == 'server'
|
|
54
|
+
Requires-Dist: python-multipart>=0.0.9; extra == 'server'
|
|
55
|
+
Requires-Dist: uvicorn[standard]>=0.30; extra == 'server'
|
|
56
|
+
Description-Content-Type: text/markdown
|
|
57
|
+
|
|
58
|
+
# docsmind
|
|
59
|
+
|
|
60
|
+
[](https://pypi.org/project/docsmind/)
|
|
61
|
+
[](https://pypi.org/project/docsmind/)
|
|
62
|
+
[](https://pepy.tech/project/docsmind)
|
|
63
|
+
[](https://pepy.tech/project/docsmind)
|
|
64
|
+
[](https://github.com/yauheniya-ai/docsmind/blob/main/LICENSE)
|
|
65
|
+
[](https://github.com/astral-sh/ruff)
|
|
66
|
+
[](#status)
|
|
67
|
+
|
|
68
|
+
Chat with your documents. Track what changed between versions, and why it matters.
|
|
69
|
+
|
|
70
|
+
`docsmind` ingests PDFs (and soon DOCX/text) into a version-aware Postgres/pgvector
|
|
71
|
+
store, and gives you an agent, **the Mind**, to chat with them. A built-in subagent,
|
|
72
|
+
**diffra**, will do change analysis between document versions: what changed, where, and
|
|
73
|
+
its compliance implications, with citations back to exact pages.
|
|
74
|
+
|
|
75
|
+
> **Alpha.** Ingestion and inspection work today. Embeddings, chat and change analysis
|
|
76
|
+
> are designed and scaffolded but not implemented yet; see [What works today](#what-works-today).
|
|
77
|
+
|
|
78
|
+
## What works today
|
|
79
|
+
|
|
80
|
+
| | |
|
|
81
|
+
|---|---|
|
|
82
|
+
| ✅ PDF ingestion | Full text, stats, outline (from the PDF's own table of contents), page-based chunks, images |
|
|
83
|
+
| ✅ Postgres + pgvector storage | Local Docker, or any Postgres with pgvector (Neon, Supabase) |
|
|
84
|
+
| ✅ Inspection CLI | `versions`, `fulltext`, `outline`, `chunks` read back exactly what was saved |
|
|
85
|
+
| ✅ `docsmind doctor` | Checks database, blob storage and API keys, and tells you what to fix |
|
|
86
|
+
| ✅ Table extraction | Extract and store tables skipping repeating ones (e.g. headers or footers) |
|
|
87
|
+
| ✅ server + UI | FastAPI backend (`docsmind serve`) + React UI (`frontend/`): add documents, browse indexed content, image and table galleries |
|
|
88
|
+
| 🚧 Chat agent (the Mind) | `docsmind chat` exists as a command, without an agent behind it yet |
|
|
89
|
+
| 🚧 Image captioning | Images are extracted and stored; vision captioning is next |
|
|
90
|
+
| 🚧 Embeddings + hybrid search | Chunks are stored without embeddings for now |
|
|
91
|
+
| 🚧 Change analysis (diffra) | `docsmind diff` exists as a command, without an implementation yet |
|
|
92
|
+
| 🚧 DOCX/text parsers, S3/Supabase blob storage | Planned |
|
|
93
|
+
|
|
94
|
+
## Install
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
pip install docsmind
|
|
98
|
+
# or, with optional extras:
|
|
99
|
+
pip install "docsmind[all]"
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Extras: `docsmind[openai]`, `docsmind[anthropic]`, `docsmind[server]` (FastAPI, serves the bundled UI),
|
|
103
|
+
`docsmind[mlflow]` (tracing/eval export). Requires Python 3.10+.
|
|
104
|
+
|
|
105
|
+
## Quickstart
|
|
106
|
+
|
|
107
|
+
### 1. Get a Postgres with pgvector
|
|
108
|
+
|
|
109
|
+
Pick one:
|
|
110
|
+
|
|
111
|
+
**A. Local Docker (recommended for trying it out)**
|
|
112
|
+
```bash
|
|
113
|
+
docsmind db up # docker compose up -d, pgvector/pgvector image on port 5433
|
|
114
|
+
docsmind db upgrade # applies migrations, creates the vector extension
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
**B. Neon or Supabase** (no Docker needed; both support pgvector)
|
|
118
|
+
```bash
|
|
119
|
+
# in .env:
|
|
120
|
+
# DOCSMIND_DATABASE_URL=postgresql+asyncpg://<user>:<password>@<host>/<db>
|
|
121
|
+
docsmind db upgrade
|
|
122
|
+
```
|
|
123
|
+
Extracted images are saved to local disk (`DOCSMIND_BLOB_ROOT`) for now. S3 and
|
|
124
|
+
Supabase Storage backends are planned.
|
|
125
|
+
|
|
126
|
+
Then check that everything is wired correctly:
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
docsmind doctor
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
Configuration is environment-driven (`DOCSMIND_*`, or a `.env` file). Copy
|
|
133
|
+
`.env.example` to `.env` to see every option. Model API keys aren't needed yet;
|
|
134
|
+
they'll be required once embeddings and chat land.
|
|
135
|
+
|
|
136
|
+
### 2. Ingest a PDF
|
|
137
|
+
|
|
138
|
+
```bash
|
|
139
|
+
docsmind ingest report.pdf --collection policies
|
|
140
|
+
docsmind ingest report_v2.pdf --collection policies --version 2025 # version label is optional
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
or from Python:
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
from docsmind import Docsmind
|
|
147
|
+
|
|
148
|
+
dm = Docsmind()
|
|
149
|
+
version = dm.ingest("report.pdf", collection="policies") # version defaults to today's date
|
|
150
|
+
print(version.id)
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
### 3. Look at what was saved
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
docsmind versions # every ingested version, with its id, word count and chunk count
|
|
157
|
+
docsmind fulltext <version-id> # saved full text (--full for everything)
|
|
158
|
+
docsmind outline <version-id> # outline from the PDF's table of contents
|
|
159
|
+
docsmind chunks <version-id> --page 12 # the chunk(s) saved for one page
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
You can also inspect the tables directly with `docsmind db psql`.
|
|
163
|
+
|
|
164
|
+
### 4. Browse it in the UI
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
pip install "docsmind[server]"
|
|
168
|
+
docsmind serve # FastAPI backend on http://127.0.0.1:8000
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
Open `http://127.0.0.1:8000`.
|
|
172
|
+
|
|
173
|
+
The React UI is bundled into the Python package and served directly by `docsmind serve`
|
|
174
|
+
from `docsmind/src/docsmind/server/ui`.
|
|
175
|
+
|
|
176
|
+
Clicking **+ Add Document** in the UI runs the same `docsmind ingest` pipeline on the
|
|
177
|
+
uploaded PDF. Selecting a document shows its saved full text/outline (**Content**), and
|
|
178
|
+
galleries of everything extracted from it (**Images**, **Tables**).
|
|
179
|
+
|
|
180
|
+
For contributors refreshing bundled UI assets before a release:
|
|
181
|
+
|
|
182
|
+
```bash
|
|
183
|
+
cd ../frontend
|
|
184
|
+
npm install
|
|
185
|
+
npm run build
|
|
186
|
+
cd ../docsmind
|
|
187
|
+
python scripts/bundle_frontend.py
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
### 5. Coming next
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
print(dm.chat("What are our data retention obligations?")) # planned
|
|
194
|
+
change_set = dm.compare(v1.id, v2.id) # planned (diffra)
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
## How ingestion works
|
|
198
|
+
|
|
199
|
+
Ingestion is deliberately simple and predictable. There's no layout guessing, because
|
|
200
|
+
heuristics that work on one PDF break on the next.
|
|
201
|
+
|
|
202
|
+
- **Full text** is extracted for the whole document and saved as-is, with character,
|
|
203
|
+
word and estimated token counts.
|
|
204
|
+
- **Outline** comes only from the PDF's own table of contents/bookmarks. If the PDF
|
|
205
|
+
has none, the outline is empty. Structure is never guessed from fonts or numbering.
|
|
206
|
+
- **Chunks** are made per page for vector search: most pages (~3,000 characters) become
|
|
207
|
+
one chunk, and unusually dense pages are split into ~3,500-character pieces with a
|
|
208
|
+
100-character overlap. Chunks exist to find relevant passages by meaning, not to
|
|
209
|
+
reconstruct structure.
|
|
210
|
+
- **Images** are extracted with their page and position, skipping tiny icons, and saved
|
|
211
|
+
under a readable path: `images/<collection>/<title>/<version>-<id>/`.
|
|
212
|
+
Visually repeated images (e.g. logo on every page) are de-duplicated by perceptual hash.
|
|
213
|
+
- **Tables** are detected with PyMuPDF and rendered as PNG snippets, saved under
|
|
214
|
+
`tables/<collection>/<title>/<version>-<id>/`.
|
|
215
|
+
|
|
216
|
+
## CLI reference
|
|
217
|
+
|
|
218
|
+
| Command | What it does |
|
|
219
|
+
|---|---|
|
|
220
|
+
| `docsmind ingest <file> -c <collection> [-v <label>]` | Parse a PDF and save it |
|
|
221
|
+
| `docsmind versions` | List ingested versions with ids and stats |
|
|
222
|
+
| `docsmind fulltext <id> [--full]` | Print the saved full text |
|
|
223
|
+
| `docsmind outline <id>` | Print the saved outline |
|
|
224
|
+
| `docsmind chunks <id> [--page N] [--full]` | Print saved chunks |
|
|
225
|
+
| `docsmind delete <id> [-y]` | Delete one version and its derived chunks/images/tables |
|
|
226
|
+
| `docsmind db up` / `upgrade` / `psql` | Start local Postgres, apply migrations, open psql |
|
|
227
|
+
| `docsmind doctor` | Check your setup |
|
|
228
|
+
| `docsmind serve [--host] [--port] [--reload]` | Run the FastAPI backend for the `frontend/` UI |
|
|
229
|
+
| `docsmind chat`, `docsmind diff` | Planned |
|
|
230
|
+
|
|
231
|
+
## Architecture
|
|
232
|
+
|
|
233
|
+
```
|
|
234
|
+
src/docsmind/
|
|
235
|
+
├── __init__.py # exports Docsmind + key models only
|
|
236
|
+
├── client.py # Docsmind facade (sync + async)
|
|
237
|
+
├── config.py # pydantic-settings, env-driven
|
|
238
|
+
│
|
|
239
|
+
├── domain/ # pure data + interfaces, no I/O
|
|
240
|
+
│ ├── documents.py # Document, DocumentVersion, Section, Chunk, Image, ParsedDocument
|
|
241
|
+
│ ├── changes.py # Change, ChangeSet, ImpactAssessment (diffra)
|
|
242
|
+
│ ├── citations.py # Citation
|
|
243
|
+
│ ├── tables.py # AgentTable, TableRevision
|
|
244
|
+
│ └── ports.py # LLM, Embedder, VisionCaptioner, VectorStore, BlobStore, Parser
|
|
245
|
+
│
|
|
246
|
+
├── ingestion/
|
|
247
|
+
│ ├── pipeline.py # parse -> save (embed step planned)
|
|
248
|
+
│ ├── parsers/ # pdf.py (PyMuPDF); docx, text planned
|
|
249
|
+
│ ├── chunking/ # page.py: page-based chunks with overlap
|
|
250
|
+
│ └── enrichment/ # caption.py (planned)
|
|
251
|
+
│
|
|
252
|
+
├── adapters/ # every concrete implementation, imported lazily
|
|
253
|
+
│ ├── storage/postgres/ # tables, engine, repository
|
|
254
|
+
│ ├── blobs/ # filesystem.py; s3, supabase planned
|
|
255
|
+
│ ├── llm/ embeddings/ vision/ tracing/ # planned
|
|
256
|
+
│
|
|
257
|
+
├── cli/ # Typer app: ingest, versions, fulltext, outline, chunks, db, doctor
|
|
258
|
+
├── retrieval/ # hybrid vector + keyword search (planned)
|
|
259
|
+
├── agent/ # the Mind + diffra subagent (planned)
|
|
260
|
+
└── server/ # FastAPI app + routes backing frontend/ (documents: ingest, read back content/images/tables)
|
|
261
|
+
|
|
262
|
+
alembic/ tests/ examples/ docs/
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
The UI source lives one level up as a sibling project: `frontend/` (React + TypeScript +
|
|
266
|
+
TailwindCSS, see the repo-root `CLAUDE.md`). Release builds are bundled into
|
|
267
|
+
`src/docsmind/server/ui` and served by `server/app.py` at `/`. The UI talks to `server/`
|
|
268
|
+
over `/api/*`, and loads extracted images/tables straight from disk via `/blobs/*`.
|
|
269
|
+
|
|
270
|
+
## Artwork attribution
|
|
271
|
+
|
|
272
|
+
- Workspace 3D graph artwork inspiration: Angela Galliat,
|
|
273
|
+
"Neural nervous system of a brain" (CodePen):
|
|
274
|
+
https://codepen.io/agalliat/pen/vYGXJxQ
|
|
275
|
+
|
|
276
|
+
### Design notes
|
|
277
|
+
|
|
278
|
+
- **Ports and adapters, once.** `domain/ports.py` defines every external dependency as a
|
|
279
|
+
`Protocol`. The core never imports `openai`, `anthropic` or `mlflow` directly. Only
|
|
280
|
+
adapters do, and lazily, so a bare `pip install docsmind` doesn't pull in SDKs you're
|
|
281
|
+
not using.
|
|
282
|
+
- **Text and images share one retrieval path.** An image becomes a chunk
|
|
283
|
+
(`chunk_type = 'image'`) whose content will be its caption, embedded with the same
|
|
284
|
+
embedder and stored in the same `chunks` table. One index, one citation type.
|
|
285
|
+
- **Version-aware schema.** Every chunk, image and outline entry belongs to a
|
|
286
|
+
`DocumentVersion`, so results can be filtered and cited "as of" a specific revision.
|
|
287
|
+
Outline paths (e.g. `03.13.11`) give version comparison something stable to align on
|
|
288
|
+
where a document has a table of contents.
|
|
289
|
+
- **Tables are shared state, not a UI feature.** `AgentTable` is a domain object that
|
|
290
|
+
both the agent and the user write to (append-only revisions). This is distinct from
|
|
291
|
+
the `tables` storage table (extracted PDF table *regions*, rendered as images
|
|
292
|
+
alongside `images` — browsable read-only in the UI's **Tables** gallery).
|
|
293
|
+
|
|
294
|
+
## Development
|
|
295
|
+
|
|
296
|
+
```bash
|
|
297
|
+
git clone https://github.com/yauheniya-ai/docsmind
|
|
298
|
+
cd docsmind
|
|
299
|
+
uv sync --all-groups --all-extras # or: pip install -e ".[dev,all]"
|
|
300
|
+
docsmind db up && docsmind db upgrade
|
|
301
|
+
uv run pytest
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
`requires-python = ">=3.10"` is a floor, not a target. `uv sync` will use 3.12 (see
|
|
305
|
+
`.python-version`), and CI (`.github/workflows/test.yml`) runs the tests across
|
|
306
|
+
3.10, 3.11 and 3.12 so the floor is actually verified.
|
|
307
|
+
|
|
308
|
+
To try the parser on a PDF without a database:
|
|
309
|
+
|
|
310
|
+
```bash
|
|
311
|
+
python examples/parse_pdf_demo.py path/to/file.pdf
|
|
312
|
+
```
|
|
313
|
+
|
|
314
|
+
## Status
|
|
315
|
+
|
|
316
|
+
`0.x`: the public API (`Docsmind`, the domain models) may still change. Semantic
|
|
317
|
+
versioning applies from `1.0`. See the [changelog](CHANGELOG.md) for what's in each release.
|
|
318
|
+
|
|
319
|
+
## License
|
|
320
|
+
|
|
321
|
+
MIT
|