vespera 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vespera/__init__.py +3 -0
- vespera/cli.py +133 -0
- vespera/config.py +18 -0
- vespera/documents/__init__.py +0 -0
- vespera/documents/docx_.py +14 -0
- vespera/documents/loader.py +52 -0
- vespera/documents/pdf.py +15 -0
- vespera/llm/__init__.py +0 -0
- vespera/llm/base.py +9 -0
- vespera/llm/ollama.py +59 -0
- vespera/review/__init__.py +0 -0
- vespera/review/aggregator.py +45 -0
- vespera/review/analyzer.py +145 -0
- vespera/review/models.py +91 -0
- vespera/review/prompts.py +83 -0
- vespera/review/report.py +108 -0
- vespera-0.1.0.dist-info/METADATA +158 -0
- vespera-0.1.0.dist-info/RECORD +21 -0
- vespera-0.1.0.dist-info/WHEEL +4 -0
- vespera-0.1.0.dist-info/entry_points.txt +2 -0
- vespera-0.1.0.dist-info/licenses/LICENSE +202 -0
vespera/__init__.py
ADDED
vespera/cli.py
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""Vespera command-line interface."""
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
import typer
|
|
7
|
+
from rich.console import Console
|
|
8
|
+
from rich.progress import BarColumn, Progress, TaskProgressColumn, TextColumn, TimeElapsedColumn
|
|
9
|
+
|
|
10
|
+
import vespera
|
|
11
|
+
from vespera.config import DEFAULT_MODEL, DEFAULT_OLLAMA_HOST, ReviewConfig
|
|
12
|
+
from vespera.documents.loader import discover_documents, load_document
|
|
13
|
+
from vespera.llm.ollama import OllamaError, OllamaProvider
|
|
14
|
+
from vespera.review.aggregator import aggregate_findings
|
|
15
|
+
from vespera.review.analyzer import analyze_document, cross_document_findings
|
|
16
|
+
from vespera.review.models import DocumentSummary, Finding
|
|
17
|
+
from vespera.review.report import write_outputs
|
|
18
|
+
|
|
19
|
+
app = typer.Typer(add_completion=False, no_args_is_help=True)
|
|
20
|
+
console = Console()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _version_callback(value: bool):
|
|
24
|
+
if value:
|
|
25
|
+
console.print(f"vespera {vespera.__version__}")
|
|
26
|
+
raise typer.Exit()
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@app.callback()
|
|
30
|
+
def main(
|
|
31
|
+
version: bool = typer.Option(
|
|
32
|
+
False, "--version", callback=_version_callback, is_eager=True, help="Show version."
|
|
33
|
+
),
|
|
34
|
+
):
|
|
35
|
+
"""Vespera — local-first AI due diligence."""
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@app.command()
|
|
39
|
+
def review(
|
|
40
|
+
path: Path = typer.Argument(..., exists=True, file_okay=False, help="Dataroom directory."),
|
|
41
|
+
model: str = typer.Option(DEFAULT_MODEL, "--model", "-m", help="Ollama model to use."),
|
|
42
|
+
output: Path = typer.Option(Path("vespera-output"), "--output", "-o", help="Output directory."),
|
|
43
|
+
host: str = typer.Option(DEFAULT_OLLAMA_HOST, "--host", help="Ollama server URL."),
|
|
44
|
+
):
|
|
45
|
+
"""Review a local dataroom and produce a due diligence report."""
|
|
46
|
+
config = ReviewConfig(model=model, ollama_host=host, output_dir=output)
|
|
47
|
+
provider = OllamaProvider(model=config.model, host=config.ollama_host)
|
|
48
|
+
|
|
49
|
+
console.print("\n[bold]Vespera[/bold]\n")
|
|
50
|
+
console.print(f"Reviewing [cyan]{path}[/cyan]\n")
|
|
51
|
+
|
|
52
|
+
paths = discover_documents(path)
|
|
53
|
+
console.print(f"Documents found: {len(paths)}")
|
|
54
|
+
if not paths:
|
|
55
|
+
console.print("[yellow]No supported documents (.pdf, .docx, .txt, .md) found.[/yellow]")
|
|
56
|
+
raise typer.Exit(code=1)
|
|
57
|
+
|
|
58
|
+
findings: list[Finding] = []
|
|
59
|
+
summaries: dict[str, DocumentSummary] = {}
|
|
60
|
+
reviewed: list[str] = []
|
|
61
|
+
empty: list[str] = []
|
|
62
|
+
processed = 0
|
|
63
|
+
|
|
64
|
+
try:
|
|
65
|
+
with Progress(
|
|
66
|
+
TextColumn("[progress.description]{task.description}"),
|
|
67
|
+
BarColumn(),
|
|
68
|
+
TaskProgressColumn(),
|
|
69
|
+
TimeElapsedColumn(),
|
|
70
|
+
console=console,
|
|
71
|
+
) as progress:
|
|
72
|
+
task = progress.add_task("Analysing documents", total=len(paths))
|
|
73
|
+
for doc_path in paths:
|
|
74
|
+
relative_name = str(doc_path.relative_to(path))
|
|
75
|
+
progress.update(task, description=f"Analysing {relative_name}")
|
|
76
|
+
document = load_document(doc_path)
|
|
77
|
+
if document.is_empty:
|
|
78
|
+
empty.append(relative_name)
|
|
79
|
+
else:
|
|
80
|
+
doc_findings, summary = analyze_document(
|
|
81
|
+
document, provider, config, relative_name
|
|
82
|
+
)
|
|
83
|
+
findings.extend(doc_findings)
|
|
84
|
+
if summary is not None:
|
|
85
|
+
summaries[relative_name] = summary
|
|
86
|
+
reviewed.append(relative_name)
|
|
87
|
+
processed += 1
|
|
88
|
+
progress.advance(task)
|
|
89
|
+
|
|
90
|
+
cross_task = progress.add_task("Cross-referencing documents", total=1)
|
|
91
|
+
findings.extend(cross_document_findings(summaries, provider, config))
|
|
92
|
+
progress.advance(cross_task)
|
|
93
|
+
except OllamaError as error:
|
|
94
|
+
console.print(f"\n[red]Error:[/red] {error}")
|
|
95
|
+
raise typer.Exit(code=1)
|
|
96
|
+
|
|
97
|
+
findings = aggregate_findings(findings)
|
|
98
|
+
report_path, findings_path = write_outputs(findings, reviewed, config.output_dir, empty)
|
|
99
|
+
|
|
100
|
+
console.print(f"Documents processed: {processed}\n")
|
|
101
|
+
console.print("[bold]Findings:[/bold]")
|
|
102
|
+
category_counts = Counter(f.category for f in findings)
|
|
103
|
+
if category_counts:
|
|
104
|
+
for category, count in category_counts.most_common():
|
|
105
|
+
console.print(f"- {category[0].upper() + category[1:]}: {count}")
|
|
106
|
+
else:
|
|
107
|
+
console.print("- No findings recorded")
|
|
108
|
+
console.print(f"\nReport: [green]{report_path}[/green]")
|
|
109
|
+
console.print(f"Evidence: [green]{findings_path}[/green]")
|
|
110
|
+
console.print("\n[dim]All document analysis was performed locally.[/dim]\n")
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@app.command()
|
|
114
|
+
def models(
|
|
115
|
+
host: str = typer.Option(DEFAULT_OLLAMA_HOST, "--host", help="Ollama server URL."),
|
|
116
|
+
):
|
|
117
|
+
"""Show the default model and locally installed Ollama models."""
|
|
118
|
+
console.print(f"Default model: [cyan]{DEFAULT_MODEL}[/cyan]")
|
|
119
|
+
provider = OllamaProvider(model=DEFAULT_MODEL, host=host)
|
|
120
|
+
local = provider.list_local_models()
|
|
121
|
+
if local:
|
|
122
|
+
console.print("Locally installed Ollama models:")
|
|
123
|
+
for name in local:
|
|
124
|
+
marker = " [green](default)[/green]" if name == DEFAULT_MODEL else ""
|
|
125
|
+
console.print(f"- {name}{marker}")
|
|
126
|
+
else:
|
|
127
|
+
console.print(
|
|
128
|
+
f"[yellow]Could not list local models — is Ollama running at {host}?[/yellow]"
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
if __name__ == "__main__":
|
|
133
|
+
app()
|
vespera/config.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""Configuration defaults for a review run."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
DEFAULT_MODEL = "qwen3:8b"
|
|
8
|
+
DEFAULT_OLLAMA_HOST = os.environ.get("VESPERA_OLLAMA_HOST", "http://localhost:11434")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class ReviewConfig:
|
|
13
|
+
model: str = DEFAULT_MODEL
|
|
14
|
+
ollama_host: str = DEFAULT_OLLAMA_HOST
|
|
15
|
+
chunk_chars: int = 7000
|
|
16
|
+
chunk_overlap_chars: int = 400
|
|
17
|
+
max_evidence_chars: int = 300
|
|
18
|
+
output_dir: Path = field(default_factory=lambda: Path("vespera-output"))
|
|
File without changes
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""DOCX text extraction via python-docx."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
import docx
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def extract_text(path: Path) -> str:
|
|
9
|
+
document = docx.Document(str(path))
|
|
10
|
+
parts = [paragraph.text for paragraph in document.paragraphs]
|
|
11
|
+
for table in document.tables:
|
|
12
|
+
for row in table.rows:
|
|
13
|
+
parts.append(" | ".join(cell.text for cell in row.cells))
|
|
14
|
+
return "\n".join(parts)
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Document discovery and text extraction."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from vespera.documents import docx_, pdf
|
|
7
|
+
|
|
8
|
+
SUPPORTED_EXTENSIONS = {".pdf", ".docx", ".txt", ".md"}
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class Page:
|
|
13
|
+
number: int | None # 1-based page number, None when the format has no pages
|
|
14
|
+
text: str
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class Document:
|
|
19
|
+
path: Path
|
|
20
|
+
pages: list[Page]
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def text(self) -> str:
|
|
24
|
+
return "\n".join(page.text for page in self.pages)
|
|
25
|
+
|
|
26
|
+
@property
|
|
27
|
+
def is_empty(self) -> bool:
|
|
28
|
+
return not self.text.strip()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def discover_documents(root: Path) -> list[Path]:
|
|
32
|
+
"""Recursively find supported documents under root, skipping hidden files/dirs."""
|
|
33
|
+
found = []
|
|
34
|
+
for path in sorted(root.rglob("*")):
|
|
35
|
+
if not path.is_file():
|
|
36
|
+
continue
|
|
37
|
+
if any(part.startswith(".") for part in path.relative_to(root).parts):
|
|
38
|
+
continue
|
|
39
|
+
if path.suffix.lower() in SUPPORTED_EXTENSIONS:
|
|
40
|
+
found.append(path)
|
|
41
|
+
return found
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def load_document(path: Path) -> Document:
|
|
45
|
+
suffix = path.suffix.lower()
|
|
46
|
+
if suffix == ".pdf":
|
|
47
|
+
pages = pdf.extract_pages(path)
|
|
48
|
+
elif suffix == ".docx":
|
|
49
|
+
pages = [Page(number=None, text=docx_.extract_text(path))]
|
|
50
|
+
else: # .txt, .md
|
|
51
|
+
pages = [Page(number=None, text=path.read_text(encoding="utf-8", errors="replace"))]
|
|
52
|
+
return Document(path=path, pages=pages)
|
vespera/documents/pdf.py
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""PDF text extraction via PyMuPDF."""
|
|
2
|
+
|
|
3
|
+
from pathlib import Path
|
|
4
|
+
|
|
5
|
+
import pymupdf
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def extract_pages(path: Path) -> list:
|
|
9
|
+
from vespera.documents.loader import Page
|
|
10
|
+
|
|
11
|
+
pages = []
|
|
12
|
+
with pymupdf.open(path) as doc:
|
|
13
|
+
for index, page in enumerate(doc, start=1):
|
|
14
|
+
pages.append(Page(number=index, text=page.get_text()))
|
|
15
|
+
return pages
|
vespera/llm/__init__.py
ADDED
|
File without changes
|
vespera/llm/base.py
ADDED
vespera/llm/ollama.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Ollama-backed LLM provider using the local HTTP API."""
|
|
2
|
+
|
|
3
|
+
import httpx
|
|
4
|
+
from pydantic import BaseModel, ValidationError
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class OllamaError(RuntimeError):
|
|
8
|
+
pass
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class OllamaProvider:
|
|
12
|
+
def __init__(self, model: str, host: str = "http://localhost:11434", timeout: float = 300.0):
|
|
13
|
+
self.model = model
|
|
14
|
+
self.host = host.rstrip("/")
|
|
15
|
+
self._client = httpx.Client(base_url=self.host, timeout=timeout)
|
|
16
|
+
|
|
17
|
+
def generate_structured(self, prompt: str, schema: type[BaseModel]) -> BaseModel:
|
|
18
|
+
last_error: Exception | None = None
|
|
19
|
+
for _ in range(2): # one retry on invalid output
|
|
20
|
+
content = self._chat(prompt, schema)
|
|
21
|
+
try:
|
|
22
|
+
return schema.model_validate_json(content)
|
|
23
|
+
except ValidationError as error:
|
|
24
|
+
last_error = error
|
|
25
|
+
raise OllamaError(f"Model returned output that failed validation twice: {last_error}")
|
|
26
|
+
|
|
27
|
+
def _chat(self, prompt: str, schema: type[BaseModel]) -> str:
|
|
28
|
+
payload = {
|
|
29
|
+
"model": self.model,
|
|
30
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
31
|
+
"stream": False,
|
|
32
|
+
"format": schema.model_json_schema(),
|
|
33
|
+
"think": False,
|
|
34
|
+
"options": {"temperature": 0.1},
|
|
35
|
+
}
|
|
36
|
+
try:
|
|
37
|
+
response = self._client.post("/api/chat", json=payload)
|
|
38
|
+
response.raise_for_status()
|
|
39
|
+
except httpx.ConnectError as error:
|
|
40
|
+
raise OllamaError(
|
|
41
|
+
f"Cannot reach Ollama at {self.host}. Is it running? "
|
|
42
|
+
"Install from https://ollama.com and run: ollama serve"
|
|
43
|
+
) from error
|
|
44
|
+
except httpx.HTTPStatusError as error:
|
|
45
|
+
detail = error.response.text[:300]
|
|
46
|
+
if error.response.status_code == 404:
|
|
47
|
+
raise OllamaError(
|
|
48
|
+
f"Model '{self.model}' not found. Pull it first: ollama pull {self.model}"
|
|
49
|
+
) from error
|
|
50
|
+
raise OllamaError(f"Ollama request failed: {detail}") from error
|
|
51
|
+
return response.json()["message"]["content"]
|
|
52
|
+
|
|
53
|
+
def list_local_models(self) -> list[str]:
|
|
54
|
+
try:
|
|
55
|
+
response = self._client.get("/api/tags")
|
|
56
|
+
response.raise_for_status()
|
|
57
|
+
except httpx.HTTPError:
|
|
58
|
+
return []
|
|
59
|
+
return [item["name"] for item in response.json().get("models", [])]
|
|
File without changes
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Aggregate and deduplicate findings across chunks and documents."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from vespera.review.models import SEVERITY_ORDER, Finding
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def _normalize(title: str) -> str:
|
|
9
|
+
return re.sub(r"[^a-z0-9 ]", "", title.lower()).strip()
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _similar(a: str, b: str) -> bool:
|
|
13
|
+
"""Cheap fuzzy match: identical normalized titles, or high word overlap."""
|
|
14
|
+
if a == b:
|
|
15
|
+
return True
|
|
16
|
+
words_a, words_b = set(a.split()), set(b.split())
|
|
17
|
+
if not words_a or not words_b:
|
|
18
|
+
return False
|
|
19
|
+
overlap = len(words_a & words_b) / min(len(words_a), len(words_b))
|
|
20
|
+
return overlap >= 0.8
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def aggregate_findings(findings: list[Finding]) -> list[Finding]:
|
|
24
|
+
"""Collapse near-duplicate findings (same category + file + similar title).
|
|
25
|
+
|
|
26
|
+
Keeps the highest-confidence instance. Result is sorted by severity then confidence.
|
|
27
|
+
"""
|
|
28
|
+
kept: list[Finding] = []
|
|
29
|
+
for finding in findings:
|
|
30
|
+
duplicate_of = None
|
|
31
|
+
for existing in kept:
|
|
32
|
+
if (
|
|
33
|
+
existing.category == finding.category
|
|
34
|
+
and existing.source_file == finding.source_file
|
|
35
|
+
and _similar(_normalize(existing.title), _normalize(finding.title))
|
|
36
|
+
):
|
|
37
|
+
duplicate_of = existing
|
|
38
|
+
break
|
|
39
|
+
if duplicate_of is None:
|
|
40
|
+
kept.append(finding)
|
|
41
|
+
elif finding.confidence > duplicate_of.confidence:
|
|
42
|
+
kept[kept.index(duplicate_of)] = finding
|
|
43
|
+
|
|
44
|
+
kept.sort(key=lambda f: (SEVERITY_ORDER[f.severity], -f.confidence, f.source_file))
|
|
45
|
+
return kept
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""Chunking and per-document / cross-document analysis."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
|
|
7
|
+
from vespera.config import ReviewConfig
|
|
8
|
+
from vespera.documents.loader import Document
|
|
9
|
+
from vespera.llm.base import LLMProvider
|
|
10
|
+
from vespera.review import prompts
|
|
11
|
+
from vespera.review.models import (
|
|
12
|
+
ChunkFindings,
|
|
13
|
+
CrossDocumentFindings,
|
|
14
|
+
DocumentSummary,
|
|
15
|
+
Finding,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class Chunk:
|
|
21
|
+
text: str
|
|
22
|
+
start_page: int | None
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
# small local models like to report "signatures present" as a "missing signatures"
|
|
26
|
+
# finding despite instructions; a signature finding that affirms completeness is noise
|
|
27
|
+
_POSITIVE_SIGNATURE_NOTE = re.compile(
|
|
28
|
+
r"signatures? (are |is )?present"
|
|
29
|
+
r"|(is|are|been|was|were|parties)( duly| fully)? signed"
|
|
30
|
+
r"|fully executed|duly executed"
|
|
31
|
+
r"|signature (blocks?|areas?) (is|are) complete"
|
|
32
|
+
r"|(both|all) parties (have |has )?signed",
|
|
33
|
+
re.IGNORECASE,
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _is_noise(finding) -> bool:
|
|
38
|
+
if finding.category != "missing signatures":
|
|
39
|
+
return False
|
|
40
|
+
if finding.severity == "info":
|
|
41
|
+
return True
|
|
42
|
+
return bool(_POSITIVE_SIGNATURE_NOTE.search(f"{finding.title} {finding.summary}"))
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def chunk_document(document: Document, chunk_chars: int, overlap_chars: int = 0) -> list[Chunk]:
|
|
46
|
+
"""Split a document into chunks, keeping track of the page each chunk starts on."""
|
|
47
|
+
chunks: list[Chunk] = []
|
|
48
|
+
current_text = ""
|
|
49
|
+
current_page: int | None = None
|
|
50
|
+
|
|
51
|
+
for page in document.pages:
|
|
52
|
+
remaining = page.text
|
|
53
|
+
while remaining:
|
|
54
|
+
if not current_text:
|
|
55
|
+
current_page = page.number
|
|
56
|
+
space = chunk_chars - len(current_text)
|
|
57
|
+
current_text += remaining[:space]
|
|
58
|
+
remaining = remaining[space:]
|
|
59
|
+
if len(current_text) >= chunk_chars:
|
|
60
|
+
chunks.append(Chunk(text=current_text, start_page=current_page))
|
|
61
|
+
# carry a small overlap so clauses split across a boundary aren't lost
|
|
62
|
+
current_text = current_text[-overlap_chars:] if overlap_chars else ""
|
|
63
|
+
if current_text.strip():
|
|
64
|
+
chunks.append(Chunk(text=current_text, start_page=current_page))
|
|
65
|
+
return chunks
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def analyze_document(
|
|
69
|
+
document: Document,
|
|
70
|
+
provider: LLMProvider,
|
|
71
|
+
config: ReviewConfig,
|
|
72
|
+
relative_name: str,
|
|
73
|
+
) -> tuple[list[Finding], DocumentSummary | None]:
|
|
74
|
+
"""Extract findings from one document, plus a summary for the cross-document pass."""
|
|
75
|
+
findings: list[Finding] = []
|
|
76
|
+
for chunk in chunk_document(document, config.chunk_chars, config.chunk_overlap_chars):
|
|
77
|
+
page_info = f"Page: {chunk.start_page}" if chunk.start_page else "Page: not applicable"
|
|
78
|
+
prompt = prompts.CHUNK_PROMPT.format(
|
|
79
|
+
role=prompts.ANALYST_ROLE,
|
|
80
|
+
source_file=relative_name,
|
|
81
|
+
page_info=page_info,
|
|
82
|
+
chunk_text=chunk.text,
|
|
83
|
+
)
|
|
84
|
+
result = provider.generate_structured(prompt, ChunkFindings)
|
|
85
|
+
for extracted in result.findings:
|
|
86
|
+
if _is_noise(extracted):
|
|
87
|
+
continue
|
|
88
|
+
findings.append(
|
|
89
|
+
Finding(
|
|
90
|
+
category=extracted.category,
|
|
91
|
+
title=extracted.title,
|
|
92
|
+
summary=extracted.summary,
|
|
93
|
+
severity=extracted.severity,
|
|
94
|
+
source_file=relative_name,
|
|
95
|
+
source_page=chunk.start_page,
|
|
96
|
+
evidence=extracted.evidence[: config.max_evidence_chars],
|
|
97
|
+
confidence=extracted.confidence,
|
|
98
|
+
)
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
summary: DocumentSummary | None = None
|
|
102
|
+
summary_prompt = prompts.SUMMARY_PROMPT.format(
|
|
103
|
+
role=prompts.ANALYST_ROLE,
|
|
104
|
+
source_file=relative_name,
|
|
105
|
+
document_text=document.text[: config.chunk_chars],
|
|
106
|
+
)
|
|
107
|
+
try:
|
|
108
|
+
summary = provider.generate_structured(summary_prompt, DocumentSummary)
|
|
109
|
+
except Exception:
|
|
110
|
+
summary = None # the cross-document pass is best-effort
|
|
111
|
+
return findings, summary
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def cross_document_findings(
|
|
115
|
+
summaries: dict[str, DocumentSummary],
|
|
116
|
+
provider: LLMProvider,
|
|
117
|
+
config: ReviewConfig,
|
|
118
|
+
) -> list[Finding]:
|
|
119
|
+
"""One pass over all document summaries to spot missing references and conflicts."""
|
|
120
|
+
if len(summaries) < 2:
|
|
121
|
+
return []
|
|
122
|
+
file_list = "\n".join(f"- {name}" for name in summaries)
|
|
123
|
+
rendered = "\n".join(
|
|
124
|
+
f"### {name}\n{json.dumps(summary.model_dump(), indent=2)}"
|
|
125
|
+
for name, summary in summaries.items()
|
|
126
|
+
)
|
|
127
|
+
prompt = prompts.CROSS_DOCUMENT_PROMPT.format(
|
|
128
|
+
role=prompts.ANALYST_ROLE, file_list=file_list, summaries=rendered
|
|
129
|
+
)
|
|
130
|
+
result = provider.generate_structured(prompt, CrossDocumentFindings)
|
|
131
|
+
allowed = {"missing documents explicitly referenced elsewhere", "inconsistencies between documents"}
|
|
132
|
+
return [
|
|
133
|
+
Finding(
|
|
134
|
+
category=extracted.category,
|
|
135
|
+
title=extracted.title,
|
|
136
|
+
summary=extracted.summary,
|
|
137
|
+
severity=extracted.severity,
|
|
138
|
+
source_file="(multiple documents)",
|
|
139
|
+
source_page=None,
|
|
140
|
+
evidence=extracted.evidence[: config.max_evidence_chars],
|
|
141
|
+
confidence=extracted.confidence,
|
|
142
|
+
)
|
|
143
|
+
for extracted in result.findings
|
|
144
|
+
if extracted.category in allowed
|
|
145
|
+
]
|
vespera/review/models.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
"""Pydantic models for structured findings."""
|
|
2
|
+
|
|
3
|
+
from typing import Literal, get_args
|
|
4
|
+
|
|
5
|
+
from pydantic import BaseModel, Field, field_validator
|
|
6
|
+
|
|
7
|
+
Category = Literal[
|
|
8
|
+
"parties",
|
|
9
|
+
"important dates",
|
|
10
|
+
"contract type",
|
|
11
|
+
"governing law",
|
|
12
|
+
"change-of-control clauses",
|
|
13
|
+
"termination rights",
|
|
14
|
+
"assignment restrictions",
|
|
15
|
+
"exclusivity",
|
|
16
|
+
"unusual obligations",
|
|
17
|
+
"material liabilities",
|
|
18
|
+
"IP ownership / assignment",
|
|
19
|
+
"confidentiality obligations",
|
|
20
|
+
"missing signatures",
|
|
21
|
+
"missing documents explicitly referenced elsewhere",
|
|
22
|
+
"inconsistencies between documents",
|
|
23
|
+
"potential red flags",
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
CATEGORY_ORDER: list[str] = list(get_args(Category))
|
|
27
|
+
|
|
28
|
+
Severity = Literal["info", "low", "medium", "high"]
|
|
29
|
+
|
|
30
|
+
SEVERITY_ORDER = {"high": 0, "medium": 1, "low": 2, "info": 3}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class Finding(BaseModel):
|
|
34
|
+
category: str
|
|
35
|
+
title: str
|
|
36
|
+
summary: str
|
|
37
|
+
severity: Severity
|
|
38
|
+
source_file: str
|
|
39
|
+
source_page: int | None
|
|
40
|
+
evidence: str
|
|
41
|
+
confidence: float = Field(ge=0.0, le=1.0)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class ExtractedFinding(BaseModel):
|
|
45
|
+
"""What the LLM returns for a chunk; source location is stamped by code."""
|
|
46
|
+
|
|
47
|
+
category: Category
|
|
48
|
+
title: str
|
|
49
|
+
summary: str
|
|
50
|
+
severity: Severity
|
|
51
|
+
evidence: str = Field(description="Short verbatim excerpt from the text, max ~40 words")
|
|
52
|
+
confidence: float = Field(ge=0.0, le=1.0, description="Between 0.0 and 1.0")
|
|
53
|
+
|
|
54
|
+
@field_validator("confidence", mode="before")
|
|
55
|
+
@classmethod
|
|
56
|
+
def _coerce_percentage(cls, value):
|
|
57
|
+
# local models often answer 90 when they mean 0.9
|
|
58
|
+
if isinstance(value, (int, float)) and 1.0 < value <= 100.0:
|
|
59
|
+
return value / 100.0
|
|
60
|
+
return value
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
class ChunkFindings(BaseModel):
|
|
64
|
+
findings: list[ExtractedFinding] = Field(
|
|
65
|
+
default_factory=list,
|
|
66
|
+
description="Findings supported by the supplied text. Empty if none.",
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class DocumentSummary(BaseModel):
|
|
71
|
+
"""One-line facts per document, used for the cross-document pass."""
|
|
72
|
+
|
|
73
|
+
contract_type: str = Field(description="e.g. 'Master Services Agreement', or 'unknown'")
|
|
74
|
+
parties: list[str] = Field(default_factory=list)
|
|
75
|
+
key_dates: list[str] = Field(default_factory=list)
|
|
76
|
+
referenced_documents: list[str] = Field(
|
|
77
|
+
default_factory=list,
|
|
78
|
+
description="Other documents, schedules, or exhibits this document explicitly refers to",
|
|
79
|
+
)
|
|
80
|
+
key_facts: list[str] = Field(
|
|
81
|
+
default_factory=list,
|
|
82
|
+
description=(
|
|
83
|
+
"Material factual assertions worth cross-checking: monetary amounts and "
|
|
84
|
+
"commitments, ownership claims (e.g. who owns which IP), and key terms"
|
|
85
|
+
),
|
|
86
|
+
)
|
|
87
|
+
signed: bool = Field(description="Whether the document appears to be executed/signed")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class CrossDocumentFindings(BaseModel):
|
|
91
|
+
findings: list[ExtractedFinding] = Field(default_factory=list)
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Prompts for the due diligence analysis passes."""
|
|
2
|
+
|
|
3
|
+
ANALYST_ROLE = """\
|
|
4
|
+
You are a cautious due diligence analyst performing a first-pass review of documents \
|
|
5
|
+
in a dataroom for an M&A / investment transaction. You are not a lawyer and you never \
|
|
6
|
+
give legal advice or definitive legal conclusions. You only report findings that are \
|
|
7
|
+
directly supported by the supplied text. You distinguish facts from possible concerns. \
|
|
8
|
+
If the text does not support any finding, return an empty findings list — never invent \
|
|
9
|
+
or speculate beyond the text.\
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
CHUNK_PROMPT = """\
|
|
13
|
+
{role}
|
|
14
|
+
|
|
15
|
+
Review the following excerpt from a document and extract due diligence findings.
|
|
16
|
+
|
|
17
|
+
Document: {source_file}
|
|
18
|
+
{page_info}
|
|
19
|
+
|
|
20
|
+
Rules:
|
|
21
|
+
- Only report what this text supports. No speculation. Empty list is a valid answer.
|
|
22
|
+
- "evidence" must be a short verbatim quote from the text (under 40 words).
|
|
23
|
+
- "confidence" is a number between 0.0 and 1.0 reflecting how clearly the text
|
|
24
|
+
supports the finding.
|
|
25
|
+
- Severity guide: "info" for neutral facts (parties, dates, contract type, governing law);
|
|
26
|
+
"low"/"medium" for terms a buyer should review (termination rights, assignment
|
|
27
|
+
restrictions, confidentiality); "medium"/"high" for terms that can materially affect
|
|
28
|
+
a transaction (change-of-control, exclusivity, uncapped liabilities, missing IP
|
|
29
|
+
assignment). A "missing signatures" finding is always at least "medium".
|
|
30
|
+
- Report a "missing signatures" finding ONLY when a signature area is visibly blank or
|
|
31
|
+
incomplete. Never report a finding to say that signatures ARE present, and never
|
|
32
|
+
report "missing signatures" merely because this excerpt lacks a signature block.
|
|
33
|
+
- Never report a finding whose only content is that something is in order or
|
|
34
|
+
unremarkable.
|
|
35
|
+
- Use the "contract type" category for at most one finding that identifies what kind of
|
|
36
|
+
document this is. Commercial terms such as fees, minimum purchase commitments, or
|
|
37
|
+
price adjustment rights belong under "material liabilities", "unusual obligations",
|
|
38
|
+
or "potential red flags".
|
|
39
|
+
|
|
40
|
+
--- DOCUMENT TEXT START ---
|
|
41
|
+
{chunk_text}
|
|
42
|
+
--- DOCUMENT TEXT END ---
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
SUMMARY_PROMPT = """\
|
|
46
|
+
{role}
|
|
47
|
+
|
|
48
|
+
Summarise the key facts of this document for cross-referencing against other documents
|
|
49
|
+
in the same dataroom. Only state what the text supports; use "unknown" when unclear.
|
|
50
|
+
For "signed": true only if the text shows completed signatures (names/dates filled in).
|
|
51
|
+
|
|
52
|
+
Document: {source_file}
|
|
53
|
+
|
|
54
|
+
--- DOCUMENT TEXT START ---
|
|
55
|
+
{document_text}
|
|
56
|
+
--- DOCUMENT TEXT END ---
|
|
57
|
+
"""
|
|
58
|
+
|
|
59
|
+
CROSS_DOCUMENT_PROMPT = """\
|
|
60
|
+
{role}
|
|
61
|
+
|
|
62
|
+
Below are structured summaries of every document found in a dataroom. Compare them and
|
|
63
|
+
report only two kinds of findings:
|
|
64
|
+
|
|
65
|
+
1. category "missing documents explicitly referenced elsewhere": check every entry in
|
|
66
|
+
each document's "referenced_documents" against the list of documents present; report
|
|
67
|
+
any referenced schedule, exhibit, or agreement that is not in the dataroom.
|
|
68
|
+
2. category "inconsistencies between documents": compare the "key_facts" across
|
|
69
|
+
documents and report conflicting facts — for example different monetary amounts for
|
|
70
|
+
the same commitment, contradictory claims about who owns an asset or intellectual
|
|
71
|
+
property, different names for the same party, or conflicting dates.
|
|
72
|
+
|
|
73
|
+
Rules:
|
|
74
|
+
- Only report what these summaries support. Empty list is a valid answer.
|
|
75
|
+
- "evidence" must name the documents involved and the conflicting or missing item.
|
|
76
|
+
- Do not repeat single-document findings (signatures, clauses); those are handled elsewhere.
|
|
77
|
+
|
|
78
|
+
Documents present in the dataroom:
|
|
79
|
+
{file_list}
|
|
80
|
+
|
|
81
|
+
Document summaries:
|
|
82
|
+
{summaries}
|
|
83
|
+
"""
|
vespera/review/report.py
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""Render the Markdown report and findings JSON."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
from datetime import date
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
from vespera.review.models import CATEGORY_ORDER, SEVERITY_ORDER, Finding
|
|
8
|
+
|
|
9
|
+
DISCLAIMER = (
|
|
10
|
+
"> **Important:** This report was produced by automated document triage. It is not "
|
|
11
|
+
"legal, financial, or investment advice, and it does not replace review by qualified "
|
|
12
|
+
"legal, financial, or other professional advisers. Findings may be incomplete or "
|
|
13
|
+
"incorrect and must be independently verified against the source documents."
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
LIMITATIONS = """\
|
|
17
|
+
- Analysis was performed by a local language model and may miss issues or misread context.
|
|
18
|
+
- Scanned or image-only documents are not analysed (no OCR in this version).
|
|
19
|
+
- Page references are approximate for findings that span page boundaries.
|
|
20
|
+
- Only documents in supported formats (PDF, DOCX, TXT, MD) were reviewed.
|
|
21
|
+
- Cross-document checks are based on extracted summaries, not full-text comparison.\
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _finding_line(finding: Finding) -> str:
|
|
26
|
+
location = finding.source_file
|
|
27
|
+
if finding.source_page:
|
|
28
|
+
location += f", p. {finding.source_page}"
|
|
29
|
+
lines = [
|
|
30
|
+
f"- **{finding.title}** — severity: {finding.severity}, "
|
|
31
|
+
f"confidence: {finding.confidence:.0%}",
|
|
32
|
+
f" - {finding.summary}",
|
|
33
|
+
f" - Source: `{location}`",
|
|
34
|
+
]
|
|
35
|
+
if finding.evidence.strip():
|
|
36
|
+
lines.append(f' - Evidence: "{finding.evidence.strip()}"')
|
|
37
|
+
return "\n".join(lines)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def render_markdown(
|
|
41
|
+
findings: list[Finding],
|
|
42
|
+
documents: list[str],
|
|
43
|
+
empty_documents: list[str] | None = None,
|
|
44
|
+
) -> str:
|
|
45
|
+
high_priority = [f for f in findings if f.severity in ("high", "medium")]
|
|
46
|
+
counts = {s: sum(1 for f in findings if f.severity == s) for s in SEVERITY_ORDER}
|
|
47
|
+
|
|
48
|
+
parts = [
|
|
49
|
+
"# Vespera Due Diligence Review",
|
|
50
|
+
"",
|
|
51
|
+
f"*Generated on {date.today().isoformat()} · All analysis performed locally*",
|
|
52
|
+
"",
|
|
53
|
+
DISCLAIMER,
|
|
54
|
+
"",
|
|
55
|
+
"## Executive Summary",
|
|
56
|
+
"",
|
|
57
|
+
f"Vespera reviewed **{len(documents)} documents** and recorded "
|
|
58
|
+
f"**{len(findings)} findings**: {counts['high']} high, {counts['medium']} medium, "
|
|
59
|
+
f"{counts['low']} low severity, and {counts['info']} informational.",
|
|
60
|
+
"",
|
|
61
|
+
]
|
|
62
|
+
if high_priority:
|
|
63
|
+
parts.append(
|
|
64
|
+
"Items flagged for priority attention: "
|
|
65
|
+
+ "; ".join(f.title for f in high_priority[:6])
|
|
66
|
+
+ "."
|
|
67
|
+
)
|
|
68
|
+
parts.append("")
|
|
69
|
+
|
|
70
|
+
parts += ["## High Priority Findings", ""]
|
|
71
|
+
if high_priority:
|
|
72
|
+
parts += [_finding_line(f) for f in high_priority]
|
|
73
|
+
else:
|
|
74
|
+
parts.append("No high or medium severity findings were recorded.")
|
|
75
|
+
parts.append("")
|
|
76
|
+
|
|
77
|
+
parts += ["## Findings by Category", ""]
|
|
78
|
+
for category in CATEGORY_ORDER:
|
|
79
|
+
in_category = [f for f in findings if f.category == category]
|
|
80
|
+
if not in_category:
|
|
81
|
+
continue
|
|
82
|
+
parts += [f"### {category[0].upper() + category[1:]} ({len(in_category)})", ""]
|
|
83
|
+
parts += [_finding_line(f) for f in in_category]
|
|
84
|
+
parts.append("")
|
|
85
|
+
|
|
86
|
+
parts += ["## Documents Reviewed", ""]
|
|
87
|
+
parts += [f"- `{name}`" for name in documents]
|
|
88
|
+
if empty_documents:
|
|
89
|
+
parts += ["", "Documents with no extractable text (possibly scanned images):"]
|
|
90
|
+
parts += [f"- `{name}`" for name in empty_documents]
|
|
91
|
+
parts += ["", "## Limitations", "", LIMITATIONS, ""]
|
|
92
|
+
return "\n".join(parts)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def write_outputs(
|
|
96
|
+
findings: list[Finding],
|
|
97
|
+
documents: list[str],
|
|
98
|
+
output_dir: Path,
|
|
99
|
+
empty_documents: list[str] | None = None,
|
|
100
|
+
) -> tuple[Path, Path]:
|
|
101
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
102
|
+
report_path = output_dir / "report.md"
|
|
103
|
+
findings_path = output_dir / "findings.json"
|
|
104
|
+
report_path.write_text(render_markdown(findings, documents, empty_documents), encoding="utf-8")
|
|
105
|
+
findings_path.write_text(
|
|
106
|
+
json.dumps([f.model_dump() for f in findings], indent=2), encoding="utf-8"
|
|
107
|
+
)
|
|
108
|
+
return report_path, findings_path
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: vespera
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Local-first AI due diligence. Review a dataroom without sending confidential documents to a third-party AI provider.
|
|
5
|
+
Project-URL: Homepage, https://github.com/VesperaSystems/vespera
|
|
6
|
+
Project-URL: Repository, https://github.com/VesperaSystems/vespera
|
|
7
|
+
Project-URL: Issues, https://github.com/VesperaSystems/vespera/issues
|
|
8
|
+
Author: Vespera Systems
|
|
9
|
+
License-Expression: Apache-2.0
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: contracts,dataroom,due-diligence,llm,local-first,m&a,ollama
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Environment :: Console
|
|
14
|
+
Classifier: Intended Audience :: Financial and Insurance Industry
|
|
15
|
+
Classifier: Intended Audience :: Legal Industry
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Office/Business :: Financial
|
|
21
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
22
|
+
Requires-Python: >=3.12
|
|
23
|
+
Requires-Dist: httpx>=0.27
|
|
24
|
+
Requires-Dist: pydantic>=2.7
|
|
25
|
+
Requires-Dist: pymupdf>=1.24
|
|
26
|
+
Requires-Dist: python-docx>=1.1
|
|
27
|
+
Requires-Dist: rich>=13.7
|
|
28
|
+
Requires-Dist: typer>=0.12
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: build; extra == 'dev'
|
|
31
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
32
|
+
Requires-Dist: twine; extra == 'dev'
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+
# Vespera
|
|
36
|
+
|
|
37
|
+
**Local-first AI due diligence.**
|
|
38
|
+
|
|
39
|
+
Review a dataroom without sending confidential documents to a third-party AI provider.
|
|
40
|
+
|
|
41
|
+
```bash
|
|
42
|
+
pip install vespera
|
|
43
|
+
vespera review ./dataroom
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
Vespera scans the documents in a local folder — contracts, board minutes, NDAs — and produces a structured due diligence report with evidence-backed findings, each linked to its source document and page. All analysis runs on your machine via [Ollama](https://ollama.com). Document contents never leave your network.
|
|
47
|
+
|
|
48
|
+
## What it does
|
|
49
|
+
|
|
50
|
+
Vespera performs the *first-pass* review of a document collection for M&A, VC, and PE due diligence:
|
|
51
|
+
|
|
52
|
+
- Recursively discovers PDF, DOCX, TXT, and Markdown documents
|
|
53
|
+
- Extracts text and analyses it with a local LLM
|
|
54
|
+
- Produces structured findings across categories including change-of-control clauses, termination rights, assignment restrictions, exclusivity, IP ownership, material liabilities, missing signatures, and cross-document inconsistencies
|
|
55
|
+
- Writes a Markdown report and a machine-readable JSON evidence file
|
|
56
|
+
|
|
57
|
+
```text
|
|
58
|
+
Vespera
|
|
59
|
+
|
|
60
|
+
Reviewing ./dataroom
|
|
61
|
+
|
|
62
|
+
Documents found: 6
|
|
63
|
+
Documents processed: 6
|
|
64
|
+
|
|
65
|
+
Findings:
|
|
66
|
+
- Change-of-control clauses: 1
|
|
67
|
+
- Termination rights: 3
|
|
68
|
+
- Missing signatures: 1
|
|
69
|
+
- IP ownership / assignment: 2
|
|
70
|
+
|
|
71
|
+
Report: vespera-output/report.md
|
|
72
|
+
Evidence: vespera-output/findings.json
|
|
73
|
+
|
|
74
|
+
All document analysis was performed locally.
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Every finding carries its category, severity, a short verbatim evidence excerpt, a confidence score, and the source file and page.
|
|
78
|
+
|
|
79
|
+
## Privacy model
|
|
80
|
+
|
|
81
|
+
- **No cloud calls for analysis.** Inference runs on a local Ollama server (`localhost` by default).
|
|
82
|
+
- **No telemetry, no accounts, no database.** Vespera reads your documents and writes two output files. That's it.
|
|
83
|
+
- The only network activity you'll ever need is `ollama pull` to download a model once.
|
|
84
|
+
|
|
85
|
+
## Quick start
|
|
86
|
+
|
|
87
|
+
1. Install [Ollama](https://ollama.com) and pull a model:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
ollama pull qwen3:8b
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
2. Install Vespera (Python 3.12+):
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
pip install vespera
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
3. Review a dataroom:
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
vespera review ./dataroom
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Try it on the included synthetic example:
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
git clone https://github.com/VesperaSystems/vespera
|
|
109
|
+
cd vespera
|
|
110
|
+
vespera review ./examples/sample-dataroom
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
### Commands
|
|
114
|
+
|
|
115
|
+
```bash
|
|
116
|
+
vespera review PATH [--model qwen3:8b] [--output vespera-output] [--host http://localhost:11434]
|
|
117
|
+
vespera models # show default + locally installed Ollama models
|
|
118
|
+
vespera --version
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
## Supported document types
|
|
122
|
+
|
|
123
|
+
| Format | Notes |
|
|
124
|
+
| --- | --- |
|
|
125
|
+
| PDF | Priority format; per-page source references |
|
|
126
|
+
| DOCX | Paragraphs and tables; no page numbers |
|
|
127
|
+
| TXT / MD | Plain text |
|
|
128
|
+
|
|
129
|
+
Scanned image-only documents are not analysed in this version (no OCR).
|
|
130
|
+
|
|
131
|
+
## Limitations
|
|
132
|
+
|
|
133
|
+
Vespera is automated document triage. It is **not** legal, financial, or investment advice, and it does not replace review by qualified professionals. Local language models can miss issues and misread context; findings must be verified against the source documents. Vespera is designed to tell a human professional *where to look first* — not to make decisions.
|
|
134
|
+
|
|
135
|
+
## Roadmap
|
|
136
|
+
|
|
137
|
+
- OCR for scanned documents
|
|
138
|
+
- More document formats (XLSX, EML, PPTX)
|
|
139
|
+
- Additional local model providers (llama.cpp, MLX)
|
|
140
|
+
- Configurable finding categories and custom review checklists
|
|
141
|
+
- Multi-language document support
|
|
142
|
+
|
|
143
|
+
## Development
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
git clone https://github.com/VesperaSystems/vespera
|
|
147
|
+
cd vespera
|
|
148
|
+
python -m venv .venv
|
|
149
|
+
source .venv/bin/activate
|
|
150
|
+
pip install -e ".[dev]"
|
|
151
|
+
pytest
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
The LLM is behind a tiny provider interface (`vespera/llm/base.py`), so tests inject a fake provider and never require Ollama.
|
|
155
|
+
|
|
156
|
+
## License
|
|
157
|
+
|
|
158
|
+
[Apache-2.0](LICENSE)
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
vespera/__init__.py,sha256=PFOxmfya9CVMO39ruE7LFTdXt7rK2w4pc21kbKjmgPk,68
|
|
2
|
+
vespera/cli.py,sha256=dfGw_5St3zifWry-lXU1ku5y7R00KqfJJ7mtOj_tiDY,5162
|
|
3
|
+
vespera/config.py,sha256=aIxkDvumITbWUs0Qdgb12lQGap19UW4-SDfrBf0T7ps,519
|
|
4
|
+
vespera/documents/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
5
|
+
vespera/documents/docx_.py,sha256=-nqZxp8HS25wvLJXqQMzR3PBGxfwOWhJcgv-2h_rK0o,391
|
|
6
|
+
vespera/documents/loader.py,sha256=_GH-iw-mOR8MWsoOthqpFTnGoFrNwf6dtP-ydPn8BpA,1425
|
|
7
|
+
vespera/documents/pdf.py,sha256=u_NeBpSVBwvvGXzivDgQQ51UKrp5UAxA4ME1NSu_e88,356
|
|
8
|
+
vespera/llm/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
9
|
+
vespera/llm/base.py,sha256=OrEjqbz9ibinT_bxHhzR9aKFrroING224f-2aDsNYI4,253
|
|
10
|
+
vespera/llm/ollama.py,sha256=0MfhFNFvLX90yaGNc9qBzUVyBNLYjZdWPTFNHizGqkU,2338
|
|
11
|
+
vespera/review/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
12
|
+
vespera/review/aggregator.py,sha256=bXbGCpmtLEA5jgNux86ghDY1a6aw4qTQcaTALuTZ7hg,1540
|
|
13
|
+
vespera/review/analyzer.py,sha256=9s0rkS-eTOwnjUmAp3jMeGUTBOIf736HRKlequ2amug,5255
|
|
14
|
+
vespera/review/models.py,sha256=7UjpsQm294eITJ8dkQKEPPkLGvLrJwaG7019rEelQ9E,2829
|
|
15
|
+
vespera/review/prompts.py,sha256=NSzDdENwcrBk7R28CkRz-O0421wWCdmEpdh0ZOBoqcw,3563
|
|
16
|
+
vespera/review/report.py,sha256=nDFS_0BMEg0yJA1x4Jm4BgIKvBid3S56xcuZf7T-SwY,4017
|
|
17
|
+
vespera-0.1.0.dist-info/METADATA,sha256=mOYX6sBnGg1ut5ED7MbFyYlDUnAYTiMZ5JKfO763e6E,5070
|
|
18
|
+
vespera-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
19
|
+
vespera-0.1.0.dist-info/entry_points.txt,sha256=8yE1aNac0w7fVKyC6esee5bQkwlVJACz3vxK4X1-H04,44
|
|
20
|
+
vespera-0.1.0.dist-info/licenses/LICENSE,sha256=z8d0m5b2O9McPEK1xHG_dWgUBT6EfBDz6wA0F7xSPTA,11358
|
|
21
|
+
vespera-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
|
|
2
|
+
Apache License
|
|
3
|
+
Version 2.0, January 2004
|
|
4
|
+
http://www.apache.org/licenses/
|
|
5
|
+
|
|
6
|
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
|
7
|
+
|
|
8
|
+
1. Definitions.
|
|
9
|
+
|
|
10
|
+
"License" shall mean the terms and conditions for use, reproduction,
|
|
11
|
+
and distribution as defined by Sections 1 through 9 of this document.
|
|
12
|
+
|
|
13
|
+
"Licensor" shall mean the copyright owner or entity authorized by
|
|
14
|
+
the copyright owner that is granting the License.
|
|
15
|
+
|
|
16
|
+
"Legal Entity" shall mean the union of the acting entity and all
|
|
17
|
+
other entities that control, are controlled by, or are under common
|
|
18
|
+
control with that entity. For the purposes of this definition,
|
|
19
|
+
"control" means (i) the power, direct or indirect, to cause the
|
|
20
|
+
direction or management of such entity, whether by contract or
|
|
21
|
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
|
22
|
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
|
23
|
+
|
|
24
|
+
"You" (or "Your") shall mean an individual or Legal Entity
|
|
25
|
+
exercising permissions granted by this License.
|
|
26
|
+
|
|
27
|
+
"Source" form shall mean the preferred form for making modifications,
|
|
28
|
+
including but not limited to software source code, documentation
|
|
29
|
+
source, and configuration files.
|
|
30
|
+
|
|
31
|
+
"Object" form shall mean any form resulting from mechanical
|
|
32
|
+
transformation or translation of a Source form, including but
|
|
33
|
+
not limited to compiled object code, generated documentation,
|
|
34
|
+
and conversions to other media types.
|
|
35
|
+
|
|
36
|
+
"Work" shall mean the work of authorship, whether in Source or
|
|
37
|
+
Object form, made available under the License, as indicated by a
|
|
38
|
+
copyright notice that is included in or attached to the work
|
|
39
|
+
(an example is provided in the Appendix below).
|
|
40
|
+
|
|
41
|
+
"Derivative Works" shall mean any work, whether in Source or Object
|
|
42
|
+
form, that is based on (or derived from) the Work and for which the
|
|
43
|
+
editorial revisions, annotations, elaborations, or other modifications
|
|
44
|
+
represent, as a whole, an original work of authorship. For the purposes
|
|
45
|
+
of this License, Derivative Works shall not include works that remain
|
|
46
|
+
separable from, or merely link (or bind by name) to the interfaces of,
|
|
47
|
+
the Work and Derivative Works thereof.
|
|
48
|
+
|
|
49
|
+
"Contribution" shall mean any work of authorship, including
|
|
50
|
+
the original version of the Work and any modifications or additions
|
|
51
|
+
to that Work or Derivative Works thereof, that is intentionally
|
|
52
|
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
|
53
|
+
or by an individual or Legal Entity authorized to submit on behalf of
|
|
54
|
+
the copyright owner. For the purposes of this definition, "submitted"
|
|
55
|
+
means any form of electronic, verbal, or written communication sent
|
|
56
|
+
to the Licensor or its representatives, including but not limited to
|
|
57
|
+
communication on electronic mailing lists, source code control systems,
|
|
58
|
+
and issue tracking systems that are managed by, or on behalf of, the
|
|
59
|
+
Licensor for the purpose of discussing and improving the Work, but
|
|
60
|
+
excluding communication that is conspicuously marked or otherwise
|
|
61
|
+
designated in writing by the copyright owner as "Not a Contribution."
|
|
62
|
+
|
|
63
|
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
|
64
|
+
on behalf of whom a Contribution has been received by Licensor and
|
|
65
|
+
subsequently incorporated within the Work.
|
|
66
|
+
|
|
67
|
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
|
68
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
69
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
70
|
+
copyright license to reproduce, prepare Derivative Works of,
|
|
71
|
+
publicly display, publicly perform, sublicense, and distribute the
|
|
72
|
+
Work and such Derivative Works in Source or Object form.
|
|
73
|
+
|
|
74
|
+
3. Grant of Patent License. Subject to the terms and conditions of
|
|
75
|
+
this License, each Contributor hereby grants to You a perpetual,
|
|
76
|
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
|
77
|
+
(except as stated in this section) patent license to make, have made,
|
|
78
|
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
|
79
|
+
where such license applies only to those patent claims licensable
|
|
80
|
+
by such Contributor that are necessarily infringed by their
|
|
81
|
+
Contribution(s) alone or by combination of their Contribution(s)
|
|
82
|
+
with the Work to which such Contribution(s) was submitted. If You
|
|
83
|
+
institute patent litigation against any entity (including a
|
|
84
|
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
|
85
|
+
or a Contribution incorporated within the Work constitutes direct
|
|
86
|
+
or contributory patent infringement, then any patent licenses
|
|
87
|
+
granted to You under this License for that Work shall terminate
|
|
88
|
+
as of the date such litigation is filed.
|
|
89
|
+
|
|
90
|
+
4. Redistribution. You may reproduce and distribute copies of the
|
|
91
|
+
Work or Derivative Works thereof in any medium, with or without
|
|
92
|
+
modifications, and in Source or Object form, provided that You
|
|
93
|
+
meet the following conditions:
|
|
94
|
+
|
|
95
|
+
(a) You must give any other recipients of the Work or
|
|
96
|
+
Derivative Works a copy of this License; and
|
|
97
|
+
|
|
98
|
+
(b) You must cause any modified files to carry prominent notices
|
|
99
|
+
stating that You changed the files; and
|
|
100
|
+
|
|
101
|
+
(c) You must retain, in the Source form of any Derivative Works
|
|
102
|
+
that You distribute, all copyright, patent, trademark, and
|
|
103
|
+
attribution notices from the Source form of the Work,
|
|
104
|
+
excluding those notices that do not pertain to any part of
|
|
105
|
+
the Derivative Works; and
|
|
106
|
+
|
|
107
|
+
(d) If the Work includes a "NOTICE" text file as part of its
|
|
108
|
+
distribution, then any Derivative Works that You distribute must
|
|
109
|
+
include a readable copy of the attribution notices contained
|
|
110
|
+
within such NOTICE file, excluding those notices that do not
|
|
111
|
+
pertain to any part of the Derivative Works, in at least one
|
|
112
|
+
of the following places: within a NOTICE text file distributed
|
|
113
|
+
as part of the Derivative Works; within the Source form or
|
|
114
|
+
documentation, if provided along with the Derivative Works; or,
|
|
115
|
+
within a display generated by the Derivative Works, if and
|
|
116
|
+
wherever such third-party notices normally appear. The contents
|
|
117
|
+
of the NOTICE file are for informational purposes only and
|
|
118
|
+
do not modify the License. You may add Your own attribution
|
|
119
|
+
notices within Derivative Works that You distribute, alongside
|
|
120
|
+
or as an addendum to the NOTICE text from the Work, provided
|
|
121
|
+
that such additional attribution notices cannot be construed
|
|
122
|
+
as modifying the License.
|
|
123
|
+
|
|
124
|
+
You may add Your own copyright statement to Your modifications and
|
|
125
|
+
may provide additional or different license terms and conditions
|
|
126
|
+
for use, reproduction, or distribution of Your modifications, or
|
|
127
|
+
for any such Derivative Works as a whole, provided Your use,
|
|
128
|
+
reproduction, and distribution of the Work otherwise complies with
|
|
129
|
+
the conditions stated in this License.
|
|
130
|
+
|
|
131
|
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
|
132
|
+
any Contribution intentionally submitted for inclusion in the Work
|
|
133
|
+
by You to the Licensor shall be under the terms and conditions of
|
|
134
|
+
this License, without any additional terms or conditions.
|
|
135
|
+
Notwithstanding the above, nothing herein shall supersede or modify
|
|
136
|
+
the terms of any separate license agreement you may have executed
|
|
137
|
+
with Licensor regarding such Contributions.
|
|
138
|
+
|
|
139
|
+
6. Trademarks. This License does not grant permission to use the trade
|
|
140
|
+
names, trademarks, service marks, or product names of the Licensor,
|
|
141
|
+
except as required for reasonable and customary use in describing the
|
|
142
|
+
origin of the Work and reproducing the content of the NOTICE file.
|
|
143
|
+
|
|
144
|
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
|
145
|
+
agreed to in writing, Licensor provides the Work (and each
|
|
146
|
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
|
147
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
|
148
|
+
implied, including, without limitation, any warranties or conditions
|
|
149
|
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
|
150
|
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
|
151
|
+
appropriateness of using or redistributing the Work and assume any
|
|
152
|
+
risks associated with Your exercise of permissions under this License.
|
|
153
|
+
|
|
154
|
+
8. Limitation of Liability. In no event and under no legal theory,
|
|
155
|
+
whether in tort (including negligence), contract, or otherwise,
|
|
156
|
+
unless required by applicable law (such as deliberate and grossly
|
|
157
|
+
negligent acts) or agreed to in writing, shall any Contributor be
|
|
158
|
+
liable to You for damages, including any direct, indirect, special,
|
|
159
|
+
incidental, or consequential damages of any character arising as a
|
|
160
|
+
result of this License or out of the use or inability to use the
|
|
161
|
+
Work (including but not limited to damages for loss of goodwill,
|
|
162
|
+
work stoppage, computer failure or malfunction, or any and all
|
|
163
|
+
other commercial damages or losses), even if such Contributor
|
|
164
|
+
has been advised of the possibility of such damages.
|
|
165
|
+
|
|
166
|
+
9. Accepting Warranty or Additional Liability. While redistributing
|
|
167
|
+
the Work or Derivative Works thereof, You may choose to offer,
|
|
168
|
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
|
169
|
+
or other liability obligations and/or rights consistent with this
|
|
170
|
+
License. However, in accepting such obligations, You may act only
|
|
171
|
+
on Your own behalf and on Your sole responsibility, not on behalf
|
|
172
|
+
of any other Contributor, and only if You agree to indemnify,
|
|
173
|
+
defend, and hold each Contributor harmless for any liability
|
|
174
|
+
incurred by, or claims asserted against, such Contributor by reason
|
|
175
|
+
of your accepting any such warranty or additional liability.
|
|
176
|
+
|
|
177
|
+
END OF TERMS AND CONDITIONS
|
|
178
|
+
|
|
179
|
+
APPENDIX: How to apply the Apache License to your work.
|
|
180
|
+
|
|
181
|
+
To apply the Apache License to your work, attach the following
|
|
182
|
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
|
183
|
+
replaced with your own identifying information. (Don't include
|
|
184
|
+
the brackets!) The text should be enclosed in the appropriate
|
|
185
|
+
comment syntax for the file format. We also recommend that a
|
|
186
|
+
file or class name and description of purpose be included on the
|
|
187
|
+
same "printed page" as the copyright notice for easier
|
|
188
|
+
identification within third-party archives.
|
|
189
|
+
|
|
190
|
+
Copyright [yyyy] [name of copyright owner]
|
|
191
|
+
|
|
192
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
193
|
+
you may not use this file except in compliance with the License.
|
|
194
|
+
You may obtain a copy of the License at
|
|
195
|
+
|
|
196
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
197
|
+
|
|
198
|
+
Unless required by applicable law or agreed to in writing, software
|
|
199
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
200
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
201
|
+
See the License for the specific language governing permissions and
|
|
202
|
+
limitations under the License.
|