docsmind 0.1.0__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. docsmind-0.2.0/.env.example +29 -0
  2. docsmind-0.2.0/.github/workflows/publish.yml +49 -0
  3. docsmind-0.2.0/.github/workflows/test.yml +59 -0
  4. docsmind-0.2.0/.gitignore +63 -0
  5. docsmind-0.2.0/.python-version +1 -0
  6. docsmind-0.2.0/CHANGELOG.md +84 -0
  7. docsmind-0.2.0/PKG-INFO +274 -0
  8. docsmind-0.2.0/README.md +220 -0
  9. docsmind-0.2.0/alembic/env.py +83 -0
  10. docsmind-0.2.0/alembic/script.py.mako +26 -0
  11. docsmind-0.2.0/alembic/versions/0001_initial_schema.py +173 -0
  12. docsmind-0.2.0/alembic/versions/0002_full_text_and_stats.py +36 -0
  13. docsmind-0.2.0/alembic.ini +43 -0
  14. docsmind-0.2.0/docker-compose.yml +20 -0
  15. docsmind-0.2.0/examples/NIST.SP.800-171r2.pdf +0 -0
  16. docsmind-0.2.0/examples/NIST.SP.800-171r3.pdf +0 -0
  17. docsmind-0.2.0/examples/parse_pdf_demo.py +77 -0
  18. docsmind-0.2.0/examples/sample.pdf +0 -0
  19. docsmind-0.2.0/pyproject.toml +82 -0
  20. docsmind-0.2.0/scripts/make_test_pdf.py +118 -0
  21. docsmind-0.2.0/src/docsmind/__init__.py +29 -0
  22. docsmind-0.2.0/src/docsmind/adapters/blobs/__init__.py +0 -0
  23. docsmind-0.2.0/src/docsmind/adapters/blobs/filesystem.py +39 -0
  24. docsmind-0.2.0/src/docsmind/adapters/embeddings/__init__.py +0 -0
  25. docsmind-0.2.0/src/docsmind/adapters/llm/__init__.py +0 -0
  26. docsmind-0.2.0/src/docsmind/adapters/storage/__init__.py +0 -0
  27. docsmind-0.2.0/src/docsmind/adapters/storage/postgres/__init__.py +0 -0
  28. docsmind-0.2.0/src/docsmind/adapters/storage/postgres/engine.py +24 -0
  29. docsmind-0.2.0/src/docsmind/adapters/storage/postgres/repository.py +269 -0
  30. docsmind-0.2.0/src/docsmind/adapters/storage/postgres/tables.py +145 -0
  31. docsmind-0.2.0/src/docsmind/adapters/tracing/__init__.py +0 -0
  32. docsmind-0.2.0/src/docsmind/adapters/vision/__init__.py +0 -0
  33. docsmind-0.2.0/src/docsmind/agent/__init__.py +0 -0
  34. docsmind-0.2.0/src/docsmind/agent/subagents/__init__.py +0 -0
  35. docsmind-0.2.0/src/docsmind/agent/subagents/change_analysis/__init__.py +0 -0
  36. docsmind-0.2.0/src/docsmind/agent/tools/__init__.py +0 -0
  37. docsmind-0.2.0/src/docsmind/cli/__init__.py +0 -0
  38. docsmind-0.2.0/src/docsmind/cli/app.py +40 -0
  39. docsmind-0.2.0/src/docsmind/cli/chat.py +26 -0
  40. docsmind-0.2.0/src/docsmind/cli/db.py +44 -0
  41. docsmind-0.2.0/src/docsmind/cli/diff.py +32 -0
  42. docsmind-0.2.0/src/docsmind/cli/doctor.py +86 -0
  43. docsmind-0.2.0/src/docsmind/cli/ingest.py +22 -0
  44. docsmind-0.2.0/src/docsmind/cli/inspect.py +172 -0
  45. docsmind-0.2.0/src/docsmind/client.py +94 -0
  46. docsmind-0.2.0/src/docsmind/config.py +51 -0
  47. docsmind-0.2.0/src/docsmind/domain/__init__.py +0 -0
  48. docsmind-0.2.0/src/docsmind/domain/changes.py +66 -0
  49. docsmind-0.2.0/src/docsmind/domain/citations.py +26 -0
  50. docsmind-0.2.0/src/docsmind/domain/documents.py +127 -0
  51. docsmind-0.2.0/src/docsmind/domain/ports.py +81 -0
  52. docsmind-0.2.0/src/docsmind/domain/tables.py +53 -0
  53. docsmind-0.2.0/src/docsmind/ingestion/__init__.py +0 -0
  54. docsmind-0.2.0/src/docsmind/ingestion/chunking/__init__.py +0 -0
  55. docsmind-0.2.0/src/docsmind/ingestion/chunking/page.py +76 -0
  56. docsmind-0.2.0/src/docsmind/ingestion/chunking/structural.py +61 -0
  57. docsmind-0.2.0/src/docsmind/ingestion/enrichment/__init__.py +0 -0
  58. docsmind-0.2.0/src/docsmind/ingestion/parsers/__init__.py +0 -0
  59. docsmind-0.2.0/src/docsmind/ingestion/parsers/pdf.py +255 -0
  60. docsmind-0.2.0/src/docsmind/ingestion/pipeline.py +85 -0
  61. docsmind-0.2.0/src/docsmind/py.typed +0 -0
  62. docsmind-0.2.0/src/docsmind/retrieval/__init__.py +0 -0
  63. docsmind-0.2.0/src/docsmind/server/__init__.py +0 -0
  64. docsmind-0.2.0/src/docsmind/server/routes/__init__.py +0 -0
  65. docsmind-0.2.0/tests/unit/test_pdf_parser.py +292 -0
  66. docsmind-0.2.0/uv.lock +5722 -0
  67. docsmind-0.1.0/CHANGELOG.md +0 -9
  68. docsmind-0.1.0/PKG-INFO +0 -22
  69. docsmind-0.1.0/README.md +0 -3
  70. docsmind-0.1.0/pyproject.toml +0 -35
  71. docsmind-0.1.0/uv.lock +0 -733
  72. {docsmind-0.1.0/src/docsmind → docsmind-0.2.0/src/docsmind/adapters}/__init__.py +0 -0
@@ -0,0 +1,29 @@
1
+ # Copy to .env and fill in. All vars are prefixed DOCSMIND_ (see config.py).
2
+
3
+ # --- storage: pick ONE path below ---
4
+
5
+ # Path A: local Docker (docsmind db up)
6
+ DOCSMIND_DATABASE_URL=postgresql+asyncpg://docsmind:docsmind@localhost:5432/docsmind
7
+
8
+ # Path B: Supabase (Storage bucket doubles as blob store)
9
+ # DOCSMIND_DATABASE_URL=postgresql+asyncpg://postgres:<password>@db.<project>.supabase.co:5432/postgres
10
+ # DOCSMIND_BLOB_BACKEND=supabase
11
+ # DOCSMIND_SUPABASE_URL=https://<project>.supabase.co
12
+ # DOCSMIND_SUPABASE_KEY=<service_role_key>
13
+
14
+ # Path C: Neon (pair with an S3-compatible bucket, e.g. Cloudflare R2)
15
+ # DOCSMIND_DATABASE_URL=postgresql+asyncpg://<user>:<password>@<project>.neon.tech/docsmind
16
+ # DOCSMIND_BLOB_BACKEND=s3
17
+ # DOCSMIND_S3_BUCKET=docsmind-images
18
+ # DOCSMIND_S3_ENDPOINT_URL=https://<account_id>.r2.cloudflarestorage.com
19
+
20
+ # --- models ---
21
+ DOCSMIND_LLM_PROVIDER=anthropic
22
+ DOCSMIND_LLM_MODEL=claude-sonnet-4-6
23
+ DOCSMIND_EMBEDDING_PROVIDER=openai
24
+ DOCSMIND_EMBEDDING_MODEL=text-embedding-3-large
25
+ # DOCSMIND_EMBEDDING_DIMENSIONS=3072 # native size; shorten via API's `dimensions` param if needed
26
+ DOCSMIND_VISION_PROVIDER=anthropic
27
+
28
+ DOCSMIND_ANTHROPIC_API_KEY=
29
+ DOCSMIND_OPENAI_API_KEY=
@@ -0,0 +1,49 @@
1
+ name: publish
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+ workflow_dispatch:
7
+ inputs:
8
+ test_pypi:
9
+ description: "Publish to TestPyPI instead of PyPI"
10
+ type: boolean
11
+ default: true
12
+
13
+ jobs:
14
+ build:
15
+ runs-on: ubuntu-latest
16
+ steps:
17
+ - uses: actions/checkout@v4
18
+ - uses: astral-sh/setup-uv@v3
19
+ - run: uv build
20
+ - uses: actions/upload-artifact@v4
21
+ with:
22
+ name: dist
23
+ path: dist/
24
+
25
+ publish-testpypi:
26
+ if: github.event_name == 'workflow_dispatch' && inputs.test_pypi
27
+ needs: build
28
+ runs-on: ubuntu-latest
29
+ environment: testpypi
30
+ permissions:
31
+ id-token: write # trusted publishing — no API token stored in secrets
32
+ steps:
33
+ - uses: actions/download-artifact@v4
34
+ with: { name: dist, path: dist }
35
+ - uses: pypa/gh-action-pypi-publish@release/v1
36
+ with:
37
+ repository-url: https://test.pypi.org/legacy/
38
+
39
+ publish-pypi:
40
+ if: github.event_name == 'release'
41
+ needs: build
42
+ runs-on: ubuntu-latest
43
+ environment: pypi
44
+ permissions:
45
+ id-token: write
46
+ steps:
47
+ - uses: actions/download-artifact@v4
48
+ with: { name: dist, path: dist }
49
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,59 @@
1
+ name: test
2
+
3
+ on:
4
+ push:
5
+ branches: [main]
6
+ pull_request:
7
+
8
+ jobs:
9
+ test:
10
+ runs-on: ubuntu-latest
11
+ strategy:
12
+ fail-fast: false
13
+ matrix:
14
+ # 3.10 is our declared floor (see pyproject.toml `requires-python`);
15
+ # 3.12 is the default dev version (see .python-version). Test both
16
+ # ends plus the one in between so a dependency silently requiring a
17
+ # newer interpreter shows up here instead of in someone's install.
18
+ python-version: ["3.10", "3.11", "3.12"]
19
+
20
+ steps:
21
+ - uses: actions/checkout@v4
22
+
23
+ - name: Install uv
24
+ uses: astral-sh/setup-uv@v3
25
+ with:
26
+ version: "latest"
27
+
28
+ - name: Set up Python ${{ matrix.python-version }}
29
+ run: uv python install ${{ matrix.python-version }}
30
+
31
+ - name: Install dependencies
32
+ run: uv sync --all-groups --all-extras --python ${{ matrix.python-version }}
33
+
34
+ - name: Lint (ruff)
35
+ run: uv run ruff check .
36
+
37
+ - name: Type check (mypy)
38
+ run: uv run mypy src/docsmind
39
+ continue-on-error: true # tighten once the core modules are fleshed out
40
+
41
+ - name: Test
42
+ # testcontainers[postgres] pulls pgvector/pgvector and manages its own
43
+ # container via the Docker socket, which GitHub's ubuntu runners expose
44
+ # by default — no separate `services:` postgres block needed.
45
+ run: uv run pytest -v
46
+
47
+ build:
48
+ runs-on: ubuntu-latest
49
+ needs: test
50
+ steps:
51
+ - uses: actions/checkout@v4
52
+ - uses: astral-sh/setup-uv@v3
53
+ - name: Build wheel and sdist
54
+ run: uv build
55
+ - name: Check wheel installs cleanly (core deps only, no extras)
56
+ run: |
57
+ uv venv /tmp/check-venv
58
+ uv pip install --python /tmp/check-venv/bin/python dist/*.whl
59
+ /tmp/check-venv/bin/python -c "import docsmind; print(docsmind.__version__)"
@@ -0,0 +1,63 @@
1
+ # --- Python ---
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+ *.so
6
+ .Python
7
+ build/
8
+ dist/
9
+ *.egg-info/
10
+ .eggs/
11
+ *.egg
12
+
13
+ # --- Virtual environments ---
14
+ .venv/
15
+ venv/
16
+ env/
17
+ ENV/
18
+
19
+ # --- Testing / linting / type-checking ---
20
+ .pytest_cache/
21
+ .mypy_cache/
22
+ .ruff_cache/
23
+ .coverage
24
+ .coverage.*
25
+ htmlcov/
26
+ coverage.xml
27
+
28
+ # --- uv / pip ---
29
+ uv.lock.bak
30
+
31
+ # --- docsmind local runtime artifacts ---
32
+ # Local filesystem BlobStore (extracted images from ingestion) — never
33
+ # something to commit; it's per-machine, per-ingest state.
34
+ .docsmind/
35
+ # Generated by scripts/make_test_pdf.py for local testing — not a committed
36
+ # fixture (tests/unit/test_pdf_parser.py generates its own PDF in-memory).
37
+ /sample.pdf
38
+
39
+ # --- Environment / secrets ---
40
+ .env
41
+ .env.*
42
+ !.env.example
43
+
44
+ # --- Postgres (local Docker volume mounts, if you ever add bind mounts) ---
45
+ *.sql.bak
46
+
47
+ # --- Editors / OS ---
48
+ .vscode/
49
+ .idea/
50
+ *.swp
51
+ .DS_Store
52
+ Thumbs.db
53
+
54
+ # --- Jupyter, if you use it for exploration ---
55
+ .ipynb_checkpoints/
56
+
57
+ # --- Alembic ---
58
+ # (versions/ IS tracked — only local scratch files some editors drop here)
59
+ alembic/versions/*.pyc
60
+
61
+ # --- ui/ (Vite/React build output, once ui/ has a real app in it) ---
62
+ ui/node_modules/
63
+ ui/dist/
@@ -0,0 +1 @@
1
+ 3.12
@@ -0,0 +1,84 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project will be documented in this file.
4
+
5
+ ## [0.2.0] - 2026-09-27
6
+
7
+ First functional release: documents can now be ingested from PDF into
8
+ Postgres/pgvector and inspected from the command line. Chat, embeddings and
9
+ change analysis are scaffolded but not implemented yet (see "Not yet
10
+ implemented" below).
11
+
12
+ ### Added
13
+
14
+ - **PDF ingestion pipeline** (`docsmind ingest <file> --collection <name>`),
15
+ built on PyMuPDF:
16
+ - Full document text is extracted and saved per version, together with
17
+ stats (character, word and estimated token counts).
18
+ - Outline is taken from the PDF's own table of contents/bookmarks when one
19
+ exists. Documents without a TOC get an empty outline; structure is
20
+ deliberately not guessed from fonts or numbering.
21
+ - Text is chunked by page for vector search: one chunk per page, with
22
+ unusually dense pages split into ~3,500-character pieces with a
23
+ 100-character overlap.
24
+ - Images are extracted with their page and bounding box, skipping tiny
25
+ icons, and stored with a perceptual hash for future de-duplication. They
26
+ are saved under a readable path
27
+ (`images/<collection>/<title>/<version>-<id>/`) instead of a bare UUID.
28
+ - **Postgres storage** with pgvector: Alembic migrations for documents,
29
+ versions, sections, images, chunks (with a generated `tsvector` column for
30
+ keyword search) and editable agent tables with revision history. The vector
31
+ index uses `halfvec` so 3072-dimension embeddings
32
+ (`text-embedding-3-large`) fit within pgvector's HNSW limit.
33
+ - **Inspection commands**: `docsmind versions`, `docsmind fulltext`,
34
+ `docsmind outline` and `docsmind chunks` read back what was saved, so
35
+ results can be checked without writing SQL.
36
+ - **Database commands**: `docsmind db up` (local Docker Postgres with
37
+ pgvector), `docsmind db upgrade` (apply migrations) and `docsmind db psql`.
38
+ - `docsmind doctor` checks the database URL, blob storage and API keys, and
39
+ detects Neon/Supabase connection strings.
40
+ - `Docsmind` / `AsyncDocsmind` Python facade for ingesting documents.
41
+ - Domain models for documents, versions, sections, chunks, images, citations,
42
+ change sets and editable tables, with `Protocol` ports for LLMs, embedders,
43
+ vision captioners, vector stores, blob stores and parsers.
44
+ - Environment-driven configuration (`DOCSMIND_*` variables, `.env` support)
45
+ and a documented `.env.example`.
46
+ - Filesystem blob storage for extracted images.
47
+ - GitHub Actions: a test workflow across Python 3.10, 3.11 and 3.12 with a
48
+ wheel install check, and a publish workflow using PyPI trusted publishing.
49
+ - Unit tests for the PDF parser and page chunker, a runnable parser demo
50
+ (`examples/parse_pdf_demo.py`) and a synthetic test-PDF generator.
51
+ - `.gitignore`, `docker-compose.yml` and `.python-version`.
52
+
53
+ ### Changed
54
+
55
+ - `--version` on `docsmind ingest` is now optional and defaults to today's
56
+ date.
57
+ - The local Docker Postgres listens on port 5433 instead of 5432 to avoid
58
+ clashing with a locally installed Postgres.
59
+ - Package metadata: author email, explicit Python 3.10-3.12 classifiers and
60
+ the issue tracker URL.
61
+
62
+ ### Removed
63
+
64
+ - Heading detection from font size, boldness and numbering patterns, and
65
+ section-based chunking. Neither generalised across real-world PDFs, and
66
+ chunking no longer depends on document structure.
67
+
68
+ ### Not yet implemented
69
+
70
+ - Embeddings and hybrid vector/keyword search. Chunks are stored without
71
+ embeddings, and `hybrid_search` raises `NotImplementedError`.
72
+ - Image captioning with a vision model.
73
+ - The chat agent behind `docsmind chat` and the change-analysis subagent
74
+ behind `docsmind diff`. Both commands exist but have no implementation
75
+ behind them yet.
76
+ - DOCX and plain-text parsers, S3/Supabase blob storage, and the server/UI.
77
+ - Multi-column reading order: PDFs with genuine multi-column layouts may have
78
+ their text interleaved in the saved full text.
79
+
80
+ ## [0.1.0] - 2026-09-12
81
+
82
+ ### Added
83
+
84
+ - Initial release to reserve the package name on PyPI.
@@ -0,0 +1,274 @@
1
+ Metadata-Version: 2.5
2
+ Name: docsmind
3
+ Version: 0.2.0
4
+ Summary: Chat with your documents. Version-aware retrieval, change analysis, and an editable table workspace, backed by Postgres/pgvector.
5
+ Project-URL: Homepage, https://github.com/yauheniya-ai/docsmind
6
+ Project-URL: Repository, https://github.com/yauheniya-ai/docsmind
7
+ Project-URL: Issues, https://github.com/yauheniya-ai/docsmind/issues
8
+ Author-email: Yauheniya Varabyova <yauheniya.ai@gmail.com>
9
+ License: MIT
10
+ Keywords: agent,compliance,documents,pgvector,rag
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Text Processing :: Indexing
20
+ Requires-Python: >=3.10
21
+ Requires-Dist: alembic>=1.13
22
+ Requires-Dist: asyncpg>=0.29
23
+ Requires-Dist: httpx>=0.27
24
+ Requires-Dist: imagehash>=4.3
25
+ Requires-Dist: pgvector>=0.2.5
26
+ Requires-Dist: pillow>=10.0
27
+ Requires-Dist: pydantic-settings>=2.2
28
+ Requires-Dist: pydantic>=2.6
29
+ Requires-Dist: pymupdf>=1.24
30
+ Requires-Dist: rich>=13.7
31
+ Requires-Dist: sqlalchemy[asyncio]>=2.0
32
+ Requires-Dist: typer>=0.12
33
+ Provides-Extra: all
34
+ Requires-Dist: anthropic>=0.34; extra == 'all'
35
+ Requires-Dist: fastapi>=0.111; extra == 'all'
36
+ Requires-Dist: openai>=1.30; extra == 'all'
37
+ Requires-Dist: uvicorn[standard]>=0.30; extra == 'all'
38
+ Provides-Extra: anthropic
39
+ Requires-Dist: anthropic>=0.34; extra == 'anthropic'
40
+ Provides-Extra: dev
41
+ Requires-Dist: mypy>=1.10; extra == 'dev'
42
+ Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
43
+ Requires-Dist: pytest>=8.0; extra == 'dev'
44
+ Requires-Dist: ruff>=0.5; extra == 'dev'
45
+ Requires-Dist: testcontainers[postgres]>=4.4; extra == 'dev'
46
+ Provides-Extra: mlflow
47
+ Requires-Dist: mlflow>=2.14; extra == 'mlflow'
48
+ Provides-Extra: openai
49
+ Requires-Dist: openai>=1.30; extra == 'openai'
50
+ Provides-Extra: server
51
+ Requires-Dist: fastapi>=0.111; extra == 'server'
52
+ Requires-Dist: uvicorn[standard]>=0.30; extra == 'server'
53
+ Description-Content-Type: text/markdown
54
+
55
+ # docsmind
56
+
57
+ [![PyPI version](https://img.shields.io/pypi/v/docsmind.svg)](https://pypi.org/project/docsmind/)
58
+ [![Python versions](https://img.shields.io/pypi/pyversions/docsmind.svg)](https://pypi.org/project/docsmind/)
59
+ [![Downloads](https://static.pepy.tech/badge/docsmind)](https://pepy.tech/project/docsmind)
60
+ [![Downloads / month](https://static.pepy.tech/badge/docsmind/month)](https://pepy.tech/project/docsmind)
61
+ [![Tests](https://github.com/yauheniya-ai/docsmind/actions/workflows/test.yml/badge.svg)](https://github.com/yauheniya-ai/docsmind/actions/workflows/test.yml)
62
+ [![Coverage](https://img.shields.io/endpoint?url=https://gist.githubusercontent.com/yauheniya-ai/<GIST_ID>/raw/docsmind-coverage.json)](https://github.com/yauheniya-ai/docsmind/actions/workflows/test.yml)
63
+ [![License: MIT](https://img.shields.io/pypi/l/docsmind.svg)](https://github.com/yauheniya-ai/docsmind/blob/main/LICENSE)
64
+ [![Code style: ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
65
+ [![Status: alpha](https://img.shields.io/badge/status-alpha-orange.svg)](#status)
66
+
67
+ Chat with your documents. Track what changed between versions, and why it matters.
68
+
69
+ `docsmind` ingests PDFs (and soon DOCX/text) into a version-aware Postgres/pgvector
70
+ store, and gives you an agent, **the Mind**, to chat with them. A built-in subagent,
71
+ **diffra**, will do change analysis between document versions: what changed, where, and
72
+ its compliance implications, with citations back to exact pages.
73
+
74
+ > **Alpha.** Ingestion and inspection work today. Embeddings, chat and change analysis
75
+ > are designed and scaffolded but not implemented yet; see [What works today](#what-works-today).
76
+
77
+ ## What works today
78
+
79
+ | | |
80
+ |---|---|
81
+ | ✅ PDF ingestion | Full text, stats, outline (from the PDF's own table of contents), page-based chunks, images |
82
+ | ✅ Postgres + pgvector storage | Local Docker, or any Postgres with pgvector (Neon, Supabase) |
83
+ | ✅ Inspection CLI | `versions`, `fulltext`, `outline`, `chunks` read back exactly what was saved |
84
+ | ✅ `docsmind doctor` | Checks database, blob storage and API keys, and tells you what to fix |
85
+ | 🚧 Embeddings + hybrid search | Chunks are stored without embeddings for now |
86
+ | 🚧 Image captioning | Images are extracted and stored; vision captioning is next |
87
+ | 🚧 Chat agent (the Mind) | `docsmind chat` exists as a command, without an agent behind it yet |
88
+ | 🚧 Change analysis (diffra) | `docsmind diff` exists as a command, without an implementation yet |
89
+ | 🚧 DOCX/text parsers, S3/Supabase blob storage, server + UI | Planned |
90
+
91
+ ## Install
92
+
93
+ ```bash
94
+ pip install docsmind
95
+ # or, with optional extras:
96
+ pip install "docsmind[all]"
97
+ ```
98
+
99
+ Extras: `docsmind[openai]`, `docsmind[anthropic]`, `docsmind[server]` (FastAPI, planned UI),
100
+ `docsmind[mlflow]` (tracing/eval export). Requires Python 3.10+.
101
+
102
+ ## Quickstart
103
+
104
+ ### 1. Get a Postgres with pgvector
105
+
106
+ Pick one:
107
+
108
+ **A. Local Docker (recommended for trying it out)**
109
+ ```bash
110
+ docsmind db up # docker compose up -d, pgvector/pgvector image on port 5433
111
+ docsmind db upgrade # applies migrations, creates the vector extension
112
+ ```
113
+
114
+ **B. Neon or Supabase** (no Docker needed; both support pgvector)
115
+ ```bash
116
+ # in .env:
117
+ # DOCSMIND_DATABASE_URL=postgresql+asyncpg://<user>:<password>@<host>/<db>
118
+ docsmind db upgrade
119
+ ```
120
+ Extracted images are saved to local disk (`DOCSMIND_BLOB_ROOT`) for now. S3 and
121
+ Supabase Storage backends are planned.
122
+
123
+ Then check that everything is wired correctly:
124
+
125
+ ```bash
126
+ docsmind doctor
127
+ ```
128
+
129
+ Configuration is environment-driven (`DOCSMIND_*`, or a `.env` file). Copy
130
+ `.env.example` to `.env` to see every option. Model API keys aren't needed yet;
131
+ they'll be required once embeddings and chat land.
132
+
133
+ ### 2. Ingest a PDF
134
+
135
+ ```bash
136
+ docsmind ingest report.pdf --collection policies
137
+ docsmind ingest report_v2.pdf --collection policies --version 2025 # version label is optional
138
+ ```
139
+
140
+ or from Python:
141
+
142
+ ```python
143
+ from docsmind import Docsmind
144
+
145
+ dm = Docsmind()
146
+ version = dm.ingest("report.pdf", collection="policies") # version defaults to today's date
147
+ print(version.id)
148
+ ```
149
+
150
+ ### 3. Look at what was saved
151
+
152
+ ```bash
153
+ docsmind versions # every ingested version, with its id, word count and chunk count
154
+ docsmind fulltext <version-id> # saved full text (--full for everything)
155
+ docsmind outline <version-id> # outline from the PDF's table of contents
156
+ docsmind chunks <version-id> --page 12 # the chunk(s) saved for one page
157
+ ```
158
+
159
+ You can also inspect the tables directly with `docsmind db psql`.
160
+
161
+ ### 4. Coming next
162
+
163
+ ```python
164
+ print(dm.chat("What are our data retention obligations?")) # planned
165
+ change_set = dm.compare(v1.id, v2.id) # planned (diffra)
166
+ ```
167
+
168
+ ## How ingestion works
169
+
170
+ Ingestion is deliberately simple and predictable. There's no layout guessing, because
171
+ heuristics that work on one PDF break on the next.
172
+
173
+ - **Full text** is extracted for the whole document and saved as-is, with character,
174
+ word and estimated token counts.
175
+ - **Outline** comes only from the PDF's own table of contents/bookmarks. If the PDF
176
+ has none, the outline is empty. Structure is never guessed from fonts or numbering.
177
+ - **Chunks** are made per page for vector search: most pages (~3,000 characters) become
178
+ one chunk, and unusually dense pages are split into ~3,500-character pieces with a
179
+ 100-character overlap. Chunks exist to find relevant passages by meaning, not to
180
+ reconstruct structure.
181
+ - **Images** are extracted with their page and position, skipping tiny icons, and saved
182
+ under a readable path: `images/<collection>/<title>/<version>-<id>/`.
183
+
184
+ ## CLI reference
185
+
186
+ | Command | What it does |
187
+ |---|---|
188
+ | `docsmind ingest <file> -c <collection> [-v <label>]` | Parse a PDF and save it |
189
+ | `docsmind versions` | List ingested versions with ids and stats |
190
+ | `docsmind fulltext <id> [--full]` | Print the saved full text |
191
+ | `docsmind outline <id>` | Print the saved outline |
192
+ | `docsmind chunks <id> [--page N] [--full]` | Print saved chunks |
193
+ | `docsmind db up` / `upgrade` / `psql` | Start local Postgres, apply migrations, open psql |
194
+ | `docsmind doctor` | Check your setup |
195
+ | `docsmind chat`, `docsmind diff` | Planned |
196
+
197
+ ## Architecture
198
+
199
+ ```
200
+ src/docsmind/
201
+ ├── __init__.py # exports Docsmind + key models only
202
+ ├── client.py # Docsmind facade (sync + async)
203
+ ├── config.py # pydantic-settings, env-driven
204
+ │
205
+ ├── domain/ # pure data + interfaces, no I/O
206
+ │ ├── documents.py # Document, DocumentVersion, Section, Chunk, Image, ParsedDocument
207
+ │ ├── changes.py # Change, ChangeSet, ImpactAssessment (diffra)
208
+ │ ├── citations.py # Citation
209
+ │ ├── tables.py # AgentTable, TableRevision
210
+ │ └── ports.py # LLM, Embedder, VisionCaptioner, VectorStore, BlobStore, Parser
211
+ │
212
+ ├── ingestion/
213
+ │ ├── pipeline.py # parse -> save (embed step planned)
214
+ │ ├── parsers/ # pdf.py (PyMuPDF); docx, text planned
215
+ │ ├── chunking/ # page.py: page-based chunks with overlap
216
+ │ └── enrichment/ # caption.py (planned)
217
+ │
218
+ ├── adapters/ # every concrete implementation, imported lazily
219
+ │ ├── storage/postgres/ # tables, engine, repository
220
+ │ ├── blobs/ # filesystem.py; s3, supabase planned
221
+ │ ├── llm/ embeddings/ vision/ tracing/ # planned
222
+ │
223
+ ├── cli/ # Typer app: ingest, versions, fulltext, outline, chunks, db, doctor
224
+ ├── retrieval/ # hybrid vector + keyword search (planned)
225
+ ├── agent/ # the Mind + diffra subagent (planned)
226
+ └── server/ # FastAPI + UI (planned)
227
+
228
+ alembic/ tests/ examples/ docs/ ui/
229
+ ```
230
+
231
+ ### Design notes
232
+
233
+ - **Ports and adapters, once.** `domain/ports.py` defines every external dependency as a
234
+ `Protocol`. The core never imports `openai`, `anthropic` or `mlflow` directly. Only
235
+ adapters do, and lazily, so a bare `pip install docsmind` doesn't pull in SDKs you're
236
+ not using.
237
+ - **Text and images share one retrieval path.** An image becomes a chunk
238
+ (`chunk_type = 'image'`) whose content will be its caption, embedded with the same
239
+ embedder and stored in the same `chunks` table. One index, one citation type.
240
+ - **Version-aware schema.** Every chunk, image and outline entry belongs to a
241
+ `DocumentVersion`, so results can be filtered and cited "as of" a specific revision.
242
+ Outline paths (e.g. `03.13.11`) give version comparison something stable to align on
243
+ where a document has a table of contents.
244
+ - **Tables are shared state, not a UI feature.** `AgentTable` is a domain object that
245
+ both the agent and the user write to (append-only revisions).
246
+
247
+ ## Development
248
+
249
+ ```bash
250
+ git clone https://github.com/yauheniya-ai/docsmind
251
+ cd docsmind
252
+ uv sync --all-groups --all-extras # or: pip install -e ".[dev,all]"
253
+ docsmind db up && docsmind db upgrade
254
+ uv run pytest
255
+ ```
256
+
257
+ `requires-python = ">=3.10"` is a floor, not a target. `uv sync` will use 3.12 (see
258
+ `.python-version`), and CI (`.github/workflows/test.yml`) runs the tests across
259
+ 3.10, 3.11 and 3.12 so the floor is actually verified.
260
+
261
+ To try the parser on a PDF without a database:
262
+
263
+ ```bash
264
+ python examples/parse_pdf_demo.py path/to/file.pdf
265
+ ```
266
+
267
+ ## Status
268
+
269
+ `0.x`: the public API (`Docsmind`, the domain models) may still change. Semantic
270
+ versioning applies from `1.0`. See the [changelog](CHANGELOG.md) for what's in each release.
271
+
272
+ ## License
273
+
274
+ MIT