sirdas 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,3 @@
1
+ dist/
2
+ __pycache__/
3
+ *.egg-info/
sirdas-0.3.0/PKG-INFO ADDED
@@ -0,0 +1,53 @@
1
+ Metadata-Version: 2.5
2
+ Name: sirdas
3
+ Version: 0.3.0
4
+ Summary: Local document parsing for AI agents and ML: PDF, Word, Excel, PowerPoint, HTML, email and scans to Markdown, cited fields, chunks and datasets. Nothing is uploaded.
5
+ Project-URL: Homepage, https://sirdas.app
6
+ Project-URL: Documentation, https://sirdas.app/docs/agents/python.md
7
+ Author: Sırdaş
8
+ License-Expression: MIT
9
+ Keywords: ai-agents,anonymization,dataset,document-parsing,fine-tuning,langchain,llamaindex,llm,local-first,ocr,pdf,rag
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
13
+ Classifier: Topic :: Text Processing
14
+ Requires-Python: >=3.9
15
+ Provides-Extra: langchain
16
+ Requires-Dist: langchain-core>=0.2; extra == 'langchain'
17
+ Provides-Extra: llamaindex
18
+ Requires-Dist: llama-index-core>=0.10; extra == 'llamaindex'
19
+ Description-Content-Type: text/markdown
20
+
21
+ # sirdas (Python)
22
+
23
+ Local document parsing for AI agents and machine learning. PDF (including scans), Word, Excel, PowerPoint, OpenDocument, EPUB, HTML, CSV, email and images become Markdown, **fields cited with page and position**, tables, form fields, cited chunks and training datasets. **Nothing is uploaded**: it runs on your machine.
24
+
25
+ ```bash
26
+ pip install sirdas # needs Node.js 20+ (the engine is shared with the CLI and MCP server)
27
+ ```
28
+
29
+ ```python
30
+ import sirdas
31
+
32
+ doc = sirdas.read("factura.pdf") # one call does it all
33
+ doc["document_type"] # "factura"
34
+ {f["name"]: f["value"] for f in doc["fields"]} # número, CUFE, NIT, IVA, total… each with page, evidence and bbox
35
+ doc["next_steps"] # what to do next
36
+
37
+ sirdas.convert("informe.docx")["markdown"]
38
+ sirdas.split_documents("lote-escaneado.pdf", output_dir="separados/")
39
+ sirdas.forms("formulario.pdf") # filled fields and checkboxes
40
+ sirdas.dataset(["a.pdf", "b.docx"], "ds/", format="alpaca") # + Hugging Face dataset card
41
+ ```
42
+
43
+ LangChain and LlamaIndex:
44
+
45
+ ```python
46
+ from sirdas.integrations import SirdasLoader, SirdasReader
47
+ docs = SirdasLoader("contrato.pdf").load() # langchain Documents, stable ids
48
+ nodes = SirdasReader().load_data("historia.pdf") # llama_index Documents
49
+ ```
50
+
51
+ Personal data is anonymized by default (`anonymize=False` to keep it). Password-protected PDFs raise `PasswordRequired`; pass `password=`.
52
+
53
+ Docs: https://sirdas.app/docs/agents/python.md
sirdas-0.3.0/README.md ADDED
@@ -0,0 +1,33 @@
1
+ # sirdas (Python)
2
+
3
+ Local document parsing for AI agents and machine learning. PDF (including scans), Word, Excel, PowerPoint, OpenDocument, EPUB, HTML, CSV, email and images become Markdown, **fields cited with page and position**, tables, form fields, cited chunks and training datasets. **Nothing is uploaded**: it runs on your machine.
4
+
5
+ ```bash
6
+ pip install sirdas # needs Node.js 20+ (the engine is shared with the CLI and MCP server)
7
+ ```
8
+
9
+ ```python
10
+ import sirdas
11
+
12
+ doc = sirdas.read("factura.pdf") # one call does it all
13
+ doc["document_type"] # "factura"
14
+ {f["name"]: f["value"] for f in doc["fields"]} # número, CUFE, NIT, IVA, total… each with page, evidence and bbox
15
+ doc["next_steps"] # what to do next
16
+
17
+ sirdas.convert("informe.docx")["markdown"]
18
+ sirdas.split_documents("lote-escaneado.pdf", output_dir="separados/")
19
+ sirdas.forms("formulario.pdf") # filled fields and checkboxes
20
+ sirdas.dataset(["a.pdf", "b.docx"], "ds/", format="alpaca") # + Hugging Face dataset card
21
+ ```
22
+
23
+ LangChain and LlamaIndex:
24
+
25
+ ```python
26
+ from sirdas.integrations import SirdasLoader, SirdasReader
27
+ docs = SirdasLoader("contrato.pdf").load() # langchain Documents, stable ids
28
+ nodes = SirdasReader().load_data("historia.pdf") # llama_index Documents
29
+ ```
30
+
31
+ Personal data is anonymized by default (`anonymize=False` to keep it). Password-protected PDFs raise `PasswordRequired`; pass `password=`.
32
+
33
+ Docs: https://sirdas.app/docs/agents/python.md
@@ -0,0 +1,30 @@
1
+ [build-system]
2
+ requires = ["hatchling>=1.25"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "sirdas"
7
+ version = "0.3.0"
8
+ description = "Local document parsing for AI agents and ML: PDF, Word, Excel, PowerPoint, HTML, email and scans to Markdown, cited fields, chunks and datasets. Nothing is uploaded."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ requires-python = ">=3.9"
12
+ authors = [{ name = "Sırdaş" }]
13
+ keywords = ["pdf", "document-parsing", "rag", "llm", "ai-agents", "ocr", "anonymization", "dataset", "fine-tuning", "langchain", "llamaindex", "local-first"]
14
+ classifiers = [
15
+ "Programming Language :: Python :: 3",
16
+ "License :: OSI Approved :: MIT License",
17
+ "Topic :: Text Processing",
18
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
19
+ ]
20
+
21
+ [project.optional-dependencies]
22
+ langchain = ["langchain-core>=0.2"]
23
+ llamaindex = ["llama-index-core>=0.10"]
24
+
25
+ [project.urls]
26
+ Homepage = "https://sirdas.app"
27
+ Documentation = "https://sirdas.app/docs/agents/python.md"
28
+
29
+ [tool.hatch.build.targets.wheel]
30
+ packages = ["src/sirdas"]
@@ -0,0 +1,42 @@
1
+ """Sırdaş for Python: local document parsing for agents and ML.
2
+
3
+ Everything runs on this machine through the ``sirdas`` CLI (Node.js >= 20);
4
+ no file, text or field is sent anywhere.
5
+
6
+ >>> import sirdas
7
+ >>> doc = sirdas.read("factura.pdf")
8
+ >>> doc["document_type"], [f["value"] for f in doc["fields"] if f["name"] == "total"]
9
+ """
10
+
11
+ from .client import (
12
+ SirdasError,
13
+ PasswordRequired,
14
+ Sirdas,
15
+ read,
16
+ convert,
17
+ split_documents,
18
+ forms,
19
+ classify,
20
+ chunks,
21
+ tables,
22
+ dataset,
23
+ protect,
24
+ unlock,
25
+ )
26
+
27
+ __all__ = [
28
+ "SirdasError",
29
+ "PasswordRequired",
30
+ "Sirdas",
31
+ "read",
32
+ "convert",
33
+ "split_documents",
34
+ "forms",
35
+ "classify",
36
+ "chunks",
37
+ "tables",
38
+ "dataset",
39
+ "protect",
40
+ "unlock",
41
+ ]
42
+ __version__ = "0.3.0"
@@ -0,0 +1,223 @@
1
+ """Thin, typed-by-convention wrapper over ``sirdas <command> --json``.
2
+
3
+ Why a CLI wrapper and not a port: the engine (layout, OCR, anonymization,
4
+ field rules) lives in one TypeScript codebase shared by the web app, the MCP
5
+ server and the CLI. Wrapping it keeps Python results identical to what agents
6
+ get, instead of a second implementation that drifts.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import os
13
+ import shutil
14
+ import subprocess
15
+ from pathlib import Path
16
+ from typing import Any, Dict, List, Optional, Sequence, Union
17
+
18
+ PathLike = Union[str, "os.PathLike[str]"]
19
+ CLI_VERSION = "0.3.0"
20
+
21
+
22
+ class SirdasError(RuntimeError):
23
+ """The CLI could not process the file. ``code`` is 1 for usage, 2 for the file."""
24
+
25
+ def __init__(self, message: str, code: int):
26
+ super().__init__(message)
27
+ self.code = code
28
+
29
+
30
+ class PasswordRequired(SirdasError):
31
+ """The PDF needs a password to open (or the one given is wrong)."""
32
+
33
+
34
+ def _default_command() -> List[str]:
35
+ # 1) An explicit binary or script, 2) sirdas on PATH, 3) npx with a pinned version.
36
+ explicit = os.environ.get("SIRDAS_BIN")
37
+ if explicit:
38
+ return ["node", explicit] if explicit.endswith(".js") else [explicit]
39
+ found = shutil.which("sirdas")
40
+ if found:
41
+ return [found]
42
+ npx = shutil.which("npx")
43
+ if npx:
44
+ return [npx, "-y", f"sirdas@{CLI_VERSION}"]
45
+ raise SirdasError(
46
+ "Sırdaş needs Node.js 20+. Install it from https://nodejs.org, or set SIRDAS_BIN to the sirdas executable.",
47
+ 1,
48
+ )
49
+
50
+
51
+ class Sirdas:
52
+ """Client. Reuse one instance; it only resolves the executable once."""
53
+
54
+ def __init__(self, command: Optional[Sequence[str]] = None, timeout: Optional[float] = 600):
55
+ self.command = list(command) if command else _default_command()
56
+ self.timeout = timeout
57
+
58
+ def _run(self, args: Sequence[str], password: Optional[str] = None, parse: bool = True) -> Any:
59
+ env = dict(os.environ)
60
+ if password is not None:
61
+ # Por entorno y no por argumento: los argumentos se ven en `ps`.
62
+ env["SIRDAS_PASSWORD"] = password
63
+ proc = subprocess.run(
64
+ [*self.command, *[str(a) for a in args]],
65
+ capture_output=True,
66
+ text=True,
67
+ encoding="utf-8",
68
+ env=env,
69
+ stdin=subprocess.DEVNULL,
70
+ timeout=self.timeout,
71
+ )
72
+ if proc.returncode != 0:
73
+ message = (proc.stderr or proc.stdout).strip() or f"sirdas exited with {proc.returncode}"
74
+ if "contraseña" in message.lower() or "password" in message.lower():
75
+ raise PasswordRequired(message, proc.returncode)
76
+ raise SirdasError(message, proc.returncode)
77
+ return json.loads(proc.stdout) if parse else proc.stdout
78
+
79
+ # ---- reading ----
80
+
81
+ def read(
82
+ self,
83
+ path: PathLike,
84
+ *,
85
+ password: Optional[str] = None,
86
+ anonymize: bool = True,
87
+ ocr: bool = True,
88
+ max_tokens: int = 1500,
89
+ ) -> Dict[str, Any]:
90
+ """Everything in one call: type, cited fields (page, evidence, bbox), tables,
91
+ form fields, checkboxes, batch segments, anonymized Markdown, chunks, next steps."""
92
+ args = ["leer", path, "--json", "--max-tokens", max_tokens]
93
+ if not anonymize:
94
+ args.append("--no-anonymize")
95
+ if not ocr:
96
+ args.append("--no-ocr")
97
+ result = self._run(args, password=password)
98
+ if result.get("status") in ("needs_password", "wrong_password"):
99
+ raise PasswordRequired(result["status"], 2)
100
+ # Mismo nombre que devuelve el MCP, para que un ejemplo sirva en los dos.
101
+ result.setdefault("document_type", (result.get("document") or {}).get("type"))
102
+ return result
103
+
104
+ def convert(self, path: PathLike) -> Dict[str, Any]:
105
+ """DOCX, XLSX, PPTX, ODT, EPUB, HTML, CSV, EML, RTF… → {format, markdown, tables, warnings}."""
106
+ return self._run(["convertir", path, "--json"])
107
+
108
+ def split_documents(self, path: PathLike, output_dir: Optional[PathLike] = None) -> Dict[str, Any]:
109
+ """Where each document starts inside a batch PDF; writes one PDF each if ``output_dir``."""
110
+ args: List[Any] = ["separar", path, "--json"]
111
+ if output_dir:
112
+ args += ["-d", output_dir, "--force"]
113
+ return self._run(args)
114
+
115
+ def forms(self, path: PathLike) -> Dict[str, Any]:
116
+ """Fillable form fields and checkboxes."""
117
+ return self._run(["formulario", path, "--json"])
118
+
119
+ def classify(self, path: PathLike) -> Dict[str, Any]:
120
+ return self._run(["classify", path, "--json"])
121
+
122
+ def chunks(self, path: PathLike, *, max_tokens: int = 1500, anonymize: bool = True) -> List[Dict[str, Any]]:
123
+ """Cited chunks: stable ``id``, ``pages``, ``section``, ``text``, ``tokens``."""
124
+ args: List[Any] = ["chunks", path, "--json", "--max-tokens", max_tokens]
125
+ if not anonymize:
126
+ args.append("--no-anonymize")
127
+ return self._run(args)["chunks"]
128
+
129
+ def tables(self, path: PathLike) -> List[Dict[str, Any]]:
130
+ return self._run(["tables", path, "--json"])["tables"]
131
+
132
+ # ---- ML ----
133
+
134
+ def dataset(
135
+ self,
136
+ paths: Sequence[PathLike],
137
+ output_dir: PathLike,
138
+ *,
139
+ format: str = "chat",
140
+ split: bool = True,
141
+ anonymize: bool = True,
142
+ dedupe: bool = True,
143
+ max_tokens: int = 2000,
144
+ hugging_face: bool = True,
145
+ ) -> List[str]:
146
+ """Write a training/RAG dataset. Formats: text, chat, prompt-completion, alpaca,
147
+ sharegpt, langchain, llamaindex, embeddings. Returns the written files."""
148
+ args: List[Any] = ["dataset", *paths, "--format", format, "--max-tokens", max_tokens, "-d", output_dir, "--force"]
149
+ if split:
150
+ args.append("--split")
151
+ if not anonymize:
152
+ args.append("--no-anonymize")
153
+ if not dedupe:
154
+ args.append("--no-dedupe")
155
+ if hugging_face:
156
+ args.append("--hf")
157
+ out = self._run(args, parse=False)
158
+ return [line for line in out.splitlines() if line.strip() and Path(line.strip()).exists()]
159
+
160
+ # ---- security ----
161
+
162
+ def protect(self, path: PathLike, password: str, output: PathLike, *, allow_copy: bool = True, allow_print: bool = True) -> Path:
163
+ args: List[Any] = ["protect", path, "-o", output, "--force"]
164
+ if not allow_copy:
165
+ args.append("--no-copy")
166
+ if not allow_print:
167
+ args.append("--no-print")
168
+ self._run(args, password=password, parse=False)
169
+ return Path(output)
170
+
171
+ def unlock(self, path: PathLike, output: PathLike, password: str = "") -> Path:
172
+ self._run(["unlock", path, "-o", output, "--force"], password=password, parse=False)
173
+ return Path(output)
174
+
175
+
176
+ _default: Optional[Sirdas] = None
177
+
178
+
179
+ def _client() -> Sirdas:
180
+ global _default
181
+ if _default is None:
182
+ _default = Sirdas()
183
+ return _default
184
+
185
+
186
+ def read(path: PathLike, **kw: Any) -> Dict[str, Any]:
187
+ return _client().read(path, **kw)
188
+
189
+
190
+ def convert(path: PathLike) -> Dict[str, Any]:
191
+ return _client().convert(path)
192
+
193
+
194
+ def split_documents(path: PathLike, output_dir: Optional[PathLike] = None) -> Dict[str, Any]:
195
+ return _client().split_documents(path, output_dir)
196
+
197
+
198
+ def forms(path: PathLike) -> Dict[str, Any]:
199
+ return _client().forms(path)
200
+
201
+
202
+ def classify(path: PathLike) -> Dict[str, Any]:
203
+ return _client().classify(path)
204
+
205
+
206
+ def chunks(path: PathLike, **kw: Any) -> List[Dict[str, Any]]:
207
+ return _client().chunks(path, **kw)
208
+
209
+
210
+ def tables(path: PathLike) -> List[Dict[str, Any]]:
211
+ return _client().tables(path)
212
+
213
+
214
+ def dataset(paths: Sequence[PathLike], output_dir: PathLike, **kw: Any) -> List[str]:
215
+ return _client().dataset(paths, output_dir, **kw)
216
+
217
+
218
+ def protect(path: PathLike, password: str, output: PathLike, **kw: Any) -> Path:
219
+ return _client().protect(path, password, output, **kw)
220
+
221
+
222
+ def unlock(path: PathLike, output: PathLike, password: str = "") -> Path:
223
+ return _client().unlock(path, output, password)
@@ -0,0 +1,57 @@
1
+ """LangChain and LlamaIndex loaders. Imported lazily so neither is a hard dependency."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any, Iterator, List, Optional
6
+
7
+ from .client import PathLike, Sirdas
8
+
9
+
10
+ def _metadata(doc: dict, chunk: dict, source: str) -> dict:
11
+ return {
12
+ "source": source,
13
+ "id": chunk["id"],
14
+ "pages": chunk.get("pages"),
15
+ "section": " › ".join(chunk.get("section") or []),
16
+ "document_type": doc.get("document_type"),
17
+ "tokens": chunk.get("tokens"),
18
+ }
19
+
20
+
21
+ class SirdasLoader:
22
+ """LangChain document loader: one Document per cited chunk.
23
+
24
+ from sirdas.integrations import SirdasLoader
25
+ docs = SirdasLoader("contrato.pdf").load()
26
+ """
27
+
28
+ def __init__(self, path: PathLike, *, anonymize: bool = True, max_tokens: int = 1000, client: Optional[Sirdas] = None):
29
+ self.path, self.anonymize, self.max_tokens = str(path), anonymize, max_tokens
30
+ self.client = client or Sirdas()
31
+
32
+ def lazy_load(self) -> Iterator[Any]:
33
+ from langchain_core.documents import Document
34
+
35
+ result = self.client.read(self.path, anonymize=self.anonymize, max_tokens=self.max_tokens)
36
+ for chunk in result.get("chunks", []):
37
+ yield Document(page_content=chunk["text"], metadata=_metadata(result, chunk, self.path), id=chunk["id"])
38
+
39
+ def load(self) -> List[Any]:
40
+ return list(self.lazy_load())
41
+
42
+
43
+ class SirdasReader:
44
+ """LlamaIndex reader: one TextNode-ready Document per cited chunk, ids stable across runs."""
45
+
46
+ def __init__(self, *, anonymize: bool = True, max_tokens: int = 1000, client: Optional[Sirdas] = None):
47
+ self.anonymize, self.max_tokens = anonymize, max_tokens
48
+ self.client = client or Sirdas()
49
+
50
+ def load_data(self, file: PathLike, extra_info: Optional[dict] = None) -> List[Any]:
51
+ from llama_index.core import Document
52
+
53
+ result = self.client.read(str(file), anonymize=self.anonymize, max_tokens=self.max_tokens)
54
+ return [
55
+ Document(text=c["text"], id_=c["id"], metadata={**_metadata(result, c, str(file)), **(extra_info or {})})
56
+ for c in result.get("chunks", [])
57
+ ]
@@ -0,0 +1,73 @@
1
+ """Integration tests against the real CLI. Set SIRDAS_BIN to packages/cli/dist/main.js."""
2
+
3
+ import json
4
+ import os
5
+ import subprocess
6
+ import sys
7
+ import tempfile
8
+ import unittest
9
+ from pathlib import Path
10
+
11
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "src"))
12
+
13
+ import sirdas # noqa: E402
14
+
15
+ ROOT = Path(__file__).resolve().parents[3]
16
+ FIXTURES = ROOT / "packages" / "core" / "test" / "fixtures"
17
+
18
+
19
+ def make_invoice(path: Path) -> None:
20
+ script = """
21
+ import { PDFDocument, StandardFonts } from "pdf-lib"; import { writeFileSync } from "node:fs";
22
+ const d = await PDFDocument.create(); const f = await d.embedFont(StandardFonts.Helvetica); const p = d.addPage([595, 842]);
23
+ ["FACTURA ELECTRONICA DE VENTA No. FE-77", "CUFE: 1a2b3c4d5e6f", "Cliente: Luis Rojas, C.C. 80.123.456", "Total a pagar: $ 50.000"]
24
+ .forEach((l, i) => p.drawText(l, { x: 50, y: 780 - i * 22, size: 12, font: f }));
25
+ writeFileSync(process.argv[1], await d.save());
26
+ """
27
+ subprocess.run(["node", "--input-type=module", "-e", script, str(path)], check=True, cwd=ROOT)
28
+
29
+
30
+ class ClientTest(unittest.TestCase):
31
+ @classmethod
32
+ def setUpClass(cls):
33
+ os.environ.setdefault("SIRDAS_BIN", str(ROOT / "packages" / "cli" / "dist" / "main.js"))
34
+ cls.tmp = Path(tempfile.mkdtemp())
35
+ cls.invoice = cls.tmp / "factura.pdf"
36
+ make_invoice(cls.invoice)
37
+
38
+ def test_read_extracts_cited_fields_and_hides_personal_data(self):
39
+ doc = sirdas.read(self.invoice)
40
+ self.assertEqual(doc["status"], "ok")
41
+ self.assertEqual(doc["document_type"], "factura")
42
+ fields = {f["name"]: f for f in doc["fields"]}
43
+ self.assertEqual(fields["total"]["value"], "50.000")
44
+ self.assertEqual(fields["total"]["page"], 1)
45
+ self.assertNotIn("80.123.456", json.dumps(doc))
46
+
47
+ def test_convert_office(self):
48
+ r = sirdas.convert(FIXTURES / "libro.xlsx")
49
+ self.assertEqual(r["format"], "xlsx")
50
+ self.assertIn("FE-1", r["markdown"])
51
+
52
+ def test_chunks_have_stable_ids(self):
53
+ a = sirdas.chunks(self.invoice)
54
+ b = sirdas.chunks(self.invoice)
55
+ self.assertTrue(a and a[0]["id"].startswith("fnv1a:"))
56
+ self.assertEqual([c["id"] for c in a], [c["id"] for c in b])
57
+
58
+ def test_dataset_writes_hugging_face_layout(self):
59
+ out = self.tmp / "ds"
60
+ files = sirdas.dataset([self.invoice, FIXTURES / "deck.pptx"], out, format="sharegpt", split=False, dedupe=False)
61
+ names = sorted(Path(f).name for f in files)
62
+ self.assertIn("README.md", names)
63
+ self.assertIn("train.jsonl", names)
64
+
65
+ def test_password_roundtrip(self):
66
+ locked = sirdas.protect(self.invoice, "clave", self.tmp / "cerrado.pdf")
67
+ with self.assertRaises(sirdas.PasswordRequired):
68
+ sirdas.read(locked)
69
+ self.assertEqual(sirdas.read(locked, password="clave")["status"], "ok")
70
+
71
+
72
+ if __name__ == "__main__":
73
+ unittest.main()