sirdas 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sirdas-0.3.0/.gitignore +3 -0
- sirdas-0.3.0/PKG-INFO +53 -0
- sirdas-0.3.0/README.md +33 -0
- sirdas-0.3.0/pyproject.toml +30 -0
- sirdas-0.3.0/src/sirdas/__init__.py +42 -0
- sirdas-0.3.0/src/sirdas/client.py +223 -0
- sirdas-0.3.0/src/sirdas/integrations.py +57 -0
- sirdas-0.3.0/tests/test_client.py +73 -0
sirdas-0.3.0/.gitignore
ADDED
sirdas-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: sirdas
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: Local document parsing for AI agents and ML: PDF, Word, Excel, PowerPoint, HTML, email and scans to Markdown, cited fields, chunks and datasets. Nothing is uploaded.
|
|
5
|
+
Project-URL: Homepage, https://sirdas.app
|
|
6
|
+
Project-URL: Documentation, https://sirdas.app/docs/agents/python.md
|
|
7
|
+
Author: Sırdaş
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
Keywords: ai-agents,anonymization,dataset,document-parsing,fine-tuning,langchain,llamaindex,llm,local-first,ocr,pdf,rag
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
13
|
+
Classifier: Topic :: Text Processing
|
|
14
|
+
Requires-Python: >=3.9
|
|
15
|
+
Provides-Extra: langchain
|
|
16
|
+
Requires-Dist: langchain-core>=0.2; extra == 'langchain'
|
|
17
|
+
Provides-Extra: llamaindex
|
|
18
|
+
Requires-Dist: llama-index-core>=0.10; extra == 'llamaindex'
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
|
|
21
|
+
# sirdas (Python)
|
|
22
|
+
|
|
23
|
+
Local document parsing for AI agents and machine learning. PDF (including scans), Word, Excel, PowerPoint, OpenDocument, EPUB, HTML, CSV, email and images become Markdown, **fields cited with page and position**, tables, form fields, cited chunks and training datasets. **Nothing is uploaded**: it runs on your machine.
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install sirdas # needs Node.js 20+ (the engine is shared with the CLI and MCP server)
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
```python
|
|
30
|
+
import sirdas
|
|
31
|
+
|
|
32
|
+
doc = sirdas.read("factura.pdf") # one call does it all
|
|
33
|
+
doc["document_type"] # "factura"
|
|
34
|
+
{f["name"]: f["value"] for f in doc["fields"]} # número, CUFE, NIT, IVA, total… each with page, evidence and bbox
|
|
35
|
+
doc["next_steps"] # what to do next
|
|
36
|
+
|
|
37
|
+
sirdas.convert("informe.docx")["markdown"]
|
|
38
|
+
sirdas.split_documents("lote-escaneado.pdf", output_dir="separados/")
|
|
39
|
+
sirdas.forms("formulario.pdf") # filled fields and checkboxes
|
|
40
|
+
sirdas.dataset(["a.pdf", "b.docx"], "ds/", format="alpaca") # + Hugging Face dataset card
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
LangChain and LlamaIndex:
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from sirdas.integrations import SirdasLoader, SirdasReader
|
|
47
|
+
docs = SirdasLoader("contrato.pdf").load() # langchain Documents, stable ids
|
|
48
|
+
nodes = SirdasReader().load_data("historia.pdf") # llama_index Documents
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Personal data is anonymized by default (`anonymize=False` to keep it). Password-protected PDFs raise `PasswordRequired`; pass `password=`.
|
|
52
|
+
|
|
53
|
+
Docs: https://sirdas.app/docs/agents/python.md
|
sirdas-0.3.0/README.md
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# sirdas (Python)
|
|
2
|
+
|
|
3
|
+
Local document parsing for AI agents and machine learning. PDF (including scans), Word, Excel, PowerPoint, OpenDocument, EPUB, HTML, CSV, email and images become Markdown, **fields cited with page and position**, tables, form fields, cited chunks and training datasets. **Nothing is uploaded**: it runs on your machine.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pip install sirdas # needs Node.js 20+ (the engine is shared with the CLI and MCP server)
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
```python
|
|
10
|
+
import sirdas
|
|
11
|
+
|
|
12
|
+
doc = sirdas.read("factura.pdf") # one call does it all
|
|
13
|
+
doc["document_type"] # "factura"
|
|
14
|
+
{f["name"]: f["value"] for f in doc["fields"]} # número, CUFE, NIT, IVA, total… each with page, evidence and bbox
|
|
15
|
+
doc["next_steps"] # what to do next
|
|
16
|
+
|
|
17
|
+
sirdas.convert("informe.docx")["markdown"]
|
|
18
|
+
sirdas.split_documents("lote-escaneado.pdf", output_dir="separados/")
|
|
19
|
+
sirdas.forms("formulario.pdf") # filled fields and checkboxes
|
|
20
|
+
sirdas.dataset(["a.pdf", "b.docx"], "ds/", format="alpaca") # + Hugging Face dataset card
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
LangChain and LlamaIndex:
|
|
24
|
+
|
|
25
|
+
```python
|
|
26
|
+
from sirdas.integrations import SirdasLoader, SirdasReader
|
|
27
|
+
docs = SirdasLoader("contrato.pdf").load() # langchain Documents, stable ids
|
|
28
|
+
nodes = SirdasReader().load_data("historia.pdf") # llama_index Documents
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
Personal data is anonymized by default (`anonymize=False` to keep it). Password-protected PDFs raise `PasswordRequired`; pass `password=`.
|
|
32
|
+
|
|
33
|
+
Docs: https://sirdas.app/docs/agents/python.md
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling>=1.25"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "sirdas"
|
|
7
|
+
version = "0.3.0"
|
|
8
|
+
description = "Local document parsing for AI agents and ML: PDF, Word, Excel, PowerPoint, HTML, email and scans to Markdown, cited fields, chunks and datasets. Nothing is uploaded."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
authors = [{ name = "Sırdaş" }]
|
|
13
|
+
keywords = ["pdf", "document-parsing", "rag", "llm", "ai-agents", "ocr", "anonymization", "dataset", "fine-tuning", "langchain", "llamaindex", "local-first"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Programming Language :: Python :: 3",
|
|
16
|
+
"License :: OSI Approved :: MIT License",
|
|
17
|
+
"Topic :: Text Processing",
|
|
18
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
[project.optional-dependencies]
|
|
22
|
+
langchain = ["langchain-core>=0.2"]
|
|
23
|
+
llamaindex = ["llama-index-core>=0.10"]
|
|
24
|
+
|
|
25
|
+
[project.urls]
|
|
26
|
+
Homepage = "https://sirdas.app"
|
|
27
|
+
Documentation = "https://sirdas.app/docs/agents/python.md"
|
|
28
|
+
|
|
29
|
+
[tool.hatch.build.targets.wheel]
|
|
30
|
+
packages = ["src/sirdas"]
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Sırdaş for Python: local document parsing for agents and ML.
|
|
2
|
+
|
|
3
|
+
Everything runs on this machine through the ``sirdas`` CLI (Node.js >= 20);
|
|
4
|
+
no file, text or field is sent anywhere.
|
|
5
|
+
|
|
6
|
+
>>> import sirdas
|
|
7
|
+
>>> doc = sirdas.read("factura.pdf")
|
|
8
|
+
>>> doc["document_type"], [f["value"] for f in doc["fields"] if f["name"] == "total"]
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from .client import (
|
|
12
|
+
SirdasError,
|
|
13
|
+
PasswordRequired,
|
|
14
|
+
Sirdas,
|
|
15
|
+
read,
|
|
16
|
+
convert,
|
|
17
|
+
split_documents,
|
|
18
|
+
forms,
|
|
19
|
+
classify,
|
|
20
|
+
chunks,
|
|
21
|
+
tables,
|
|
22
|
+
dataset,
|
|
23
|
+
protect,
|
|
24
|
+
unlock,
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
__all__ = [
|
|
28
|
+
"SirdasError",
|
|
29
|
+
"PasswordRequired",
|
|
30
|
+
"Sirdas",
|
|
31
|
+
"read",
|
|
32
|
+
"convert",
|
|
33
|
+
"split_documents",
|
|
34
|
+
"forms",
|
|
35
|
+
"classify",
|
|
36
|
+
"chunks",
|
|
37
|
+
"tables",
|
|
38
|
+
"dataset",
|
|
39
|
+
"protect",
|
|
40
|
+
"unlock",
|
|
41
|
+
]
|
|
42
|
+
__version__ = "0.3.0"
|
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
"""Thin, typed-by-convention wrapper over ``sirdas <command> --json``.
|
|
2
|
+
|
|
3
|
+
Why a CLI wrapper and not a port: the engine (layout, OCR, anonymization,
|
|
4
|
+
field rules) lives in one TypeScript codebase shared by the web app, the MCP
|
|
5
|
+
server and the CLI. Wrapping it keeps Python results identical to what agents
|
|
6
|
+
get, instead of a second implementation that drifts.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import json
|
|
12
|
+
import os
|
|
13
|
+
import shutil
|
|
14
|
+
import subprocess
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any, Dict, List, Optional, Sequence, Union
|
|
17
|
+
|
|
18
|
+
PathLike = Union[str, "os.PathLike[str]"]
|
|
19
|
+
CLI_VERSION = "0.3.0"
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class SirdasError(RuntimeError):
|
|
23
|
+
"""The CLI could not process the file. ``code`` is 1 for usage, 2 for the file."""
|
|
24
|
+
|
|
25
|
+
def __init__(self, message: str, code: int):
|
|
26
|
+
super().__init__(message)
|
|
27
|
+
self.code = code
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class PasswordRequired(SirdasError):
|
|
31
|
+
"""The PDF needs a password to open (or the one given is wrong)."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _default_command() -> List[str]:
|
|
35
|
+
# 1) An explicit binary or script, 2) sirdas on PATH, 3) npx with a pinned version.
|
|
36
|
+
explicit = os.environ.get("SIRDAS_BIN")
|
|
37
|
+
if explicit:
|
|
38
|
+
return ["node", explicit] if explicit.endswith(".js") else [explicit]
|
|
39
|
+
found = shutil.which("sirdas")
|
|
40
|
+
if found:
|
|
41
|
+
return [found]
|
|
42
|
+
npx = shutil.which("npx")
|
|
43
|
+
if npx:
|
|
44
|
+
return [npx, "-y", f"sirdas@{CLI_VERSION}"]
|
|
45
|
+
raise SirdasError(
|
|
46
|
+
"Sırdaş needs Node.js 20+. Install it from https://nodejs.org, or set SIRDAS_BIN to the sirdas executable.",
|
|
47
|
+
1,
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class Sirdas:
|
|
52
|
+
"""Client. Reuse one instance; it only resolves the executable once."""
|
|
53
|
+
|
|
54
|
+
def __init__(self, command: Optional[Sequence[str]] = None, timeout: Optional[float] = 600):
|
|
55
|
+
self.command = list(command) if command else _default_command()
|
|
56
|
+
self.timeout = timeout
|
|
57
|
+
|
|
58
|
+
def _run(self, args: Sequence[str], password: Optional[str] = None, parse: bool = True) -> Any:
|
|
59
|
+
env = dict(os.environ)
|
|
60
|
+
if password is not None:
|
|
61
|
+
# Por entorno y no por argumento: los argumentos se ven en `ps`.
|
|
62
|
+
env["SIRDAS_PASSWORD"] = password
|
|
63
|
+
proc = subprocess.run(
|
|
64
|
+
[*self.command, *[str(a) for a in args]],
|
|
65
|
+
capture_output=True,
|
|
66
|
+
text=True,
|
|
67
|
+
encoding="utf-8",
|
|
68
|
+
env=env,
|
|
69
|
+
stdin=subprocess.DEVNULL,
|
|
70
|
+
timeout=self.timeout,
|
|
71
|
+
)
|
|
72
|
+
if proc.returncode != 0:
|
|
73
|
+
message = (proc.stderr or proc.stdout).strip() or f"sirdas exited with {proc.returncode}"
|
|
74
|
+
if "contraseña" in message.lower() or "password" in message.lower():
|
|
75
|
+
raise PasswordRequired(message, proc.returncode)
|
|
76
|
+
raise SirdasError(message, proc.returncode)
|
|
77
|
+
return json.loads(proc.stdout) if parse else proc.stdout
|
|
78
|
+
|
|
79
|
+
# ---- reading ----
|
|
80
|
+
|
|
81
|
+
def read(
|
|
82
|
+
self,
|
|
83
|
+
path: PathLike,
|
|
84
|
+
*,
|
|
85
|
+
password: Optional[str] = None,
|
|
86
|
+
anonymize: bool = True,
|
|
87
|
+
ocr: bool = True,
|
|
88
|
+
max_tokens: int = 1500,
|
|
89
|
+
) -> Dict[str, Any]:
|
|
90
|
+
"""Everything in one call: type, cited fields (page, evidence, bbox), tables,
|
|
91
|
+
form fields, checkboxes, batch segments, anonymized Markdown, chunks, next steps."""
|
|
92
|
+
args = ["leer", path, "--json", "--max-tokens", max_tokens]
|
|
93
|
+
if not anonymize:
|
|
94
|
+
args.append("--no-anonymize")
|
|
95
|
+
if not ocr:
|
|
96
|
+
args.append("--no-ocr")
|
|
97
|
+
result = self._run(args, password=password)
|
|
98
|
+
if result.get("status") in ("needs_password", "wrong_password"):
|
|
99
|
+
raise PasswordRequired(result["status"], 2)
|
|
100
|
+
# Mismo nombre que devuelve el MCP, para que un ejemplo sirva en los dos.
|
|
101
|
+
result.setdefault("document_type", (result.get("document") or {}).get("type"))
|
|
102
|
+
return result
|
|
103
|
+
|
|
104
|
+
def convert(self, path: PathLike) -> Dict[str, Any]:
|
|
105
|
+
"""DOCX, XLSX, PPTX, ODT, EPUB, HTML, CSV, EML, RTF… → {format, markdown, tables, warnings}."""
|
|
106
|
+
return self._run(["convertir", path, "--json"])
|
|
107
|
+
|
|
108
|
+
def split_documents(self, path: PathLike, output_dir: Optional[PathLike] = None) -> Dict[str, Any]:
|
|
109
|
+
"""Where each document starts inside a batch PDF; writes one PDF each if ``output_dir``."""
|
|
110
|
+
args: List[Any] = ["separar", path, "--json"]
|
|
111
|
+
if output_dir:
|
|
112
|
+
args += ["-d", output_dir, "--force"]
|
|
113
|
+
return self._run(args)
|
|
114
|
+
|
|
115
|
+
def forms(self, path: PathLike) -> Dict[str, Any]:
|
|
116
|
+
"""Fillable form fields and checkboxes."""
|
|
117
|
+
return self._run(["formulario", path, "--json"])
|
|
118
|
+
|
|
119
|
+
def classify(self, path: PathLike) -> Dict[str, Any]:
|
|
120
|
+
return self._run(["classify", path, "--json"])
|
|
121
|
+
|
|
122
|
+
def chunks(self, path: PathLike, *, max_tokens: int = 1500, anonymize: bool = True) -> List[Dict[str, Any]]:
|
|
123
|
+
"""Cited chunks: stable ``id``, ``pages``, ``section``, ``text``, ``tokens``."""
|
|
124
|
+
args: List[Any] = ["chunks", path, "--json", "--max-tokens", max_tokens]
|
|
125
|
+
if not anonymize:
|
|
126
|
+
args.append("--no-anonymize")
|
|
127
|
+
return self._run(args)["chunks"]
|
|
128
|
+
|
|
129
|
+
def tables(self, path: PathLike) -> List[Dict[str, Any]]:
|
|
130
|
+
return self._run(["tables", path, "--json"])["tables"]
|
|
131
|
+
|
|
132
|
+
# ---- ML ----
|
|
133
|
+
|
|
134
|
+
def dataset(
|
|
135
|
+
self,
|
|
136
|
+
paths: Sequence[PathLike],
|
|
137
|
+
output_dir: PathLike,
|
|
138
|
+
*,
|
|
139
|
+
format: str = "chat",
|
|
140
|
+
split: bool = True,
|
|
141
|
+
anonymize: bool = True,
|
|
142
|
+
dedupe: bool = True,
|
|
143
|
+
max_tokens: int = 2000,
|
|
144
|
+
hugging_face: bool = True,
|
|
145
|
+
) -> List[str]:
|
|
146
|
+
"""Write a training/RAG dataset. Formats: text, chat, prompt-completion, alpaca,
|
|
147
|
+
sharegpt, langchain, llamaindex, embeddings. Returns the written files."""
|
|
148
|
+
args: List[Any] = ["dataset", *paths, "--format", format, "--max-tokens", max_tokens, "-d", output_dir, "--force"]
|
|
149
|
+
if split:
|
|
150
|
+
args.append("--split")
|
|
151
|
+
if not anonymize:
|
|
152
|
+
args.append("--no-anonymize")
|
|
153
|
+
if not dedupe:
|
|
154
|
+
args.append("--no-dedupe")
|
|
155
|
+
if hugging_face:
|
|
156
|
+
args.append("--hf")
|
|
157
|
+
out = self._run(args, parse=False)
|
|
158
|
+
return [line for line in out.splitlines() if line.strip() and Path(line.strip()).exists()]
|
|
159
|
+
|
|
160
|
+
# ---- security ----
|
|
161
|
+
|
|
162
|
+
def protect(self, path: PathLike, password: str, output: PathLike, *, allow_copy: bool = True, allow_print: bool = True) -> Path:
|
|
163
|
+
args: List[Any] = ["protect", path, "-o", output, "--force"]
|
|
164
|
+
if not allow_copy:
|
|
165
|
+
args.append("--no-copy")
|
|
166
|
+
if not allow_print:
|
|
167
|
+
args.append("--no-print")
|
|
168
|
+
self._run(args, password=password, parse=False)
|
|
169
|
+
return Path(output)
|
|
170
|
+
|
|
171
|
+
def unlock(self, path: PathLike, output: PathLike, password: str = "") -> Path:
|
|
172
|
+
self._run(["unlock", path, "-o", output, "--force"], password=password, parse=False)
|
|
173
|
+
return Path(output)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
_default: Optional[Sirdas] = None
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _client() -> Sirdas:
|
|
180
|
+
global _default
|
|
181
|
+
if _default is None:
|
|
182
|
+
_default = Sirdas()
|
|
183
|
+
return _default
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def read(path: PathLike, **kw: Any) -> Dict[str, Any]:
|
|
187
|
+
return _client().read(path, **kw)
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def convert(path: PathLike) -> Dict[str, Any]:
|
|
191
|
+
return _client().convert(path)
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def split_documents(path: PathLike, output_dir: Optional[PathLike] = None) -> Dict[str, Any]:
|
|
195
|
+
return _client().split_documents(path, output_dir)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def forms(path: PathLike) -> Dict[str, Any]:
|
|
199
|
+
return _client().forms(path)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def classify(path: PathLike) -> Dict[str, Any]:
|
|
203
|
+
return _client().classify(path)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def chunks(path: PathLike, **kw: Any) -> List[Dict[str, Any]]:
|
|
207
|
+
return _client().chunks(path, **kw)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def tables(path: PathLike) -> List[Dict[str, Any]]:
|
|
211
|
+
return _client().tables(path)
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def dataset(paths: Sequence[PathLike], output_dir: PathLike, **kw: Any) -> List[str]:
|
|
215
|
+
return _client().dataset(paths, output_dir, **kw)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def protect(path: PathLike, password: str, output: PathLike, **kw: Any) -> Path:
|
|
219
|
+
return _client().protect(path, password, output, **kw)
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def unlock(path: PathLike, output: PathLike, password: str = "") -> Path:
|
|
223
|
+
return _client().unlock(path, output, password)
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""LangChain and LlamaIndex loaders. Imported lazily so neither is a hard dependency."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Iterator, List, Optional
|
|
6
|
+
|
|
7
|
+
from .client import PathLike, Sirdas
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def _metadata(doc: dict, chunk: dict, source: str) -> dict:
|
|
11
|
+
return {
|
|
12
|
+
"source": source,
|
|
13
|
+
"id": chunk["id"],
|
|
14
|
+
"pages": chunk.get("pages"),
|
|
15
|
+
"section": " › ".join(chunk.get("section") or []),
|
|
16
|
+
"document_type": doc.get("document_type"),
|
|
17
|
+
"tokens": chunk.get("tokens"),
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class SirdasLoader:
|
|
22
|
+
"""LangChain document loader: one Document per cited chunk.
|
|
23
|
+
|
|
24
|
+
from sirdas.integrations import SirdasLoader
|
|
25
|
+
docs = SirdasLoader("contrato.pdf").load()
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
def __init__(self, path: PathLike, *, anonymize: bool = True, max_tokens: int = 1000, client: Optional[Sirdas] = None):
|
|
29
|
+
self.path, self.anonymize, self.max_tokens = str(path), anonymize, max_tokens
|
|
30
|
+
self.client = client or Sirdas()
|
|
31
|
+
|
|
32
|
+
def lazy_load(self) -> Iterator[Any]:
|
|
33
|
+
from langchain_core.documents import Document
|
|
34
|
+
|
|
35
|
+
result = self.client.read(self.path, anonymize=self.anonymize, max_tokens=self.max_tokens)
|
|
36
|
+
for chunk in result.get("chunks", []):
|
|
37
|
+
yield Document(page_content=chunk["text"], metadata=_metadata(result, chunk, self.path), id=chunk["id"])
|
|
38
|
+
|
|
39
|
+
def load(self) -> List[Any]:
|
|
40
|
+
return list(self.lazy_load())
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class SirdasReader:
|
|
44
|
+
"""LlamaIndex reader: one TextNode-ready Document per cited chunk, ids stable across runs."""
|
|
45
|
+
|
|
46
|
+
def __init__(self, *, anonymize: bool = True, max_tokens: int = 1000, client: Optional[Sirdas] = None):
|
|
47
|
+
self.anonymize, self.max_tokens = anonymize, max_tokens
|
|
48
|
+
self.client = client or Sirdas()
|
|
49
|
+
|
|
50
|
+
def load_data(self, file: PathLike, extra_info: Optional[dict] = None) -> List[Any]:
|
|
51
|
+
from llama_index.core import Document
|
|
52
|
+
|
|
53
|
+
result = self.client.read(str(file), anonymize=self.anonymize, max_tokens=self.max_tokens)
|
|
54
|
+
return [
|
|
55
|
+
Document(text=c["text"], id_=c["id"], metadata={**_metadata(result, c, str(file)), **(extra_info or {})})
|
|
56
|
+
for c in result.get("chunks", [])
|
|
57
|
+
]
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Integration tests against the real CLI. Set SIRDAS_BIN to packages/cli/dist/main.js."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import subprocess
|
|
6
|
+
import sys
|
|
7
|
+
import tempfile
|
|
8
|
+
import unittest
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
|
|
12
|
+
|
|
13
|
+
import sirdas # noqa: E402
|
|
14
|
+
|
|
15
|
+
ROOT = Path(__file__).resolve().parents[3]
|
|
16
|
+
FIXTURES = ROOT / "packages" / "core" / "test" / "fixtures"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def make_invoice(path: Path) -> None:
|
|
20
|
+
script = """
|
|
21
|
+
import { PDFDocument, StandardFonts } from "pdf-lib"; import { writeFileSync } from "node:fs";
|
|
22
|
+
const d = await PDFDocument.create(); const f = await d.embedFont(StandardFonts.Helvetica); const p = d.addPage([595, 842]);
|
|
23
|
+
["FACTURA ELECTRONICA DE VENTA No. FE-77", "CUFE: 1a2b3c4d5e6f", "Cliente: Luis Rojas, C.C. 80.123.456", "Total a pagar: $ 50.000"]
|
|
24
|
+
.forEach((l, i) => p.drawText(l, { x: 50, y: 780 - i * 22, size: 12, font: f }));
|
|
25
|
+
writeFileSync(process.argv[1], await d.save());
|
|
26
|
+
"""
|
|
27
|
+
subprocess.run(["node", "--input-type=module", "-e", script, str(path)], check=True, cwd=ROOT)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class ClientTest(unittest.TestCase):
|
|
31
|
+
@classmethod
|
|
32
|
+
def setUpClass(cls):
|
|
33
|
+
os.environ.setdefault("SIRDAS_BIN", str(ROOT / "packages" / "cli" / "dist" / "main.js"))
|
|
34
|
+
cls.tmp = Path(tempfile.mkdtemp())
|
|
35
|
+
cls.invoice = cls.tmp / "factura.pdf"
|
|
36
|
+
make_invoice(cls.invoice)
|
|
37
|
+
|
|
38
|
+
def test_read_extracts_cited_fields_and_hides_personal_data(self):
|
|
39
|
+
doc = sirdas.read(self.invoice)
|
|
40
|
+
self.assertEqual(doc["status"], "ok")
|
|
41
|
+
self.assertEqual(doc["document_type"], "factura")
|
|
42
|
+
fields = {f["name"]: f for f in doc["fields"]}
|
|
43
|
+
self.assertEqual(fields["total"]["value"], "50.000")
|
|
44
|
+
self.assertEqual(fields["total"]["page"], 1)
|
|
45
|
+
self.assertNotIn("80.123.456", json.dumps(doc))
|
|
46
|
+
|
|
47
|
+
def test_convert_office(self):
|
|
48
|
+
r = sirdas.convert(FIXTURES / "libro.xlsx")
|
|
49
|
+
self.assertEqual(r["format"], "xlsx")
|
|
50
|
+
self.assertIn("FE-1", r["markdown"])
|
|
51
|
+
|
|
52
|
+
def test_chunks_have_stable_ids(self):
|
|
53
|
+
a = sirdas.chunks(self.invoice)
|
|
54
|
+
b = sirdas.chunks(self.invoice)
|
|
55
|
+
self.assertTrue(a and a[0]["id"].startswith("fnv1a:"))
|
|
56
|
+
self.assertEqual([c["id"] for c in a], [c["id"] for c in b])
|
|
57
|
+
|
|
58
|
+
def test_dataset_writes_hugging_face_layout(self):
|
|
59
|
+
out = self.tmp / "ds"
|
|
60
|
+
files = sirdas.dataset([self.invoice, FIXTURES / "deck.pptx"], out, format="sharegpt", split=False, dedupe=False)
|
|
61
|
+
names = sorted(Path(f).name for f in files)
|
|
62
|
+
self.assertIn("README.md", names)
|
|
63
|
+
self.assertIn("train.jsonl", names)
|
|
64
|
+
|
|
65
|
+
def test_password_roundtrip(self):
|
|
66
|
+
locked = sirdas.protect(self.invoice, "clave", self.tmp / "cerrado.pdf")
|
|
67
|
+
with self.assertRaises(sirdas.PasswordRequired):
|
|
68
|
+
sirdas.read(locked)
|
|
69
|
+
self.assertEqual(sirdas.read(locked, password="clave")["status"], "ok")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
if __name__ == "__main__":
|
|
73
|
+
unittest.main()
|