sirdas 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sirdas/__init__.py ADDED
@@ -0,0 +1,42 @@
1
+ """Sırdaş for Python: local document parsing for agents and ML.
2
+
3
+ Everything runs on this machine through the ``sirdas`` CLI (Node.js >= 20);
4
+ no file, text or field is sent anywhere.
5
+
6
+ >>> import sirdas
7
+ >>> doc = sirdas.read("factura.pdf")
8
+ >>> doc["document_type"], [f["value"] for f in doc["fields"] if f["name"] == "total"]
9
+ """
10
+
11
+ from .client import (
12
+ SirdasError,
13
+ PasswordRequired,
14
+ Sirdas,
15
+ read,
16
+ convert,
17
+ split_documents,
18
+ forms,
19
+ classify,
20
+ chunks,
21
+ tables,
22
+ dataset,
23
+ protect,
24
+ unlock,
25
+ )
26
+
27
+ __all__ = [
28
+ "SirdasError",
29
+ "PasswordRequired",
30
+ "Sirdas",
31
+ "read",
32
+ "convert",
33
+ "split_documents",
34
+ "forms",
35
+ "classify",
36
+ "chunks",
37
+ "tables",
38
+ "dataset",
39
+ "protect",
40
+ "unlock",
41
+ ]
42
+ __version__ = "0.3.0"
sirdas/client.py ADDED
@@ -0,0 +1,223 @@
1
+ """Thin, typed-by-convention wrapper over ``sirdas <command> --json``.
2
+
3
+ Why a CLI wrapper and not a port: the engine (layout, OCR, anonymization,
4
+ field rules) lives in one TypeScript codebase shared by the web app, the MCP
5
+ server and the CLI. Wrapping it keeps Python results identical to what agents
6
+ get, instead of a second implementation that drifts.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ import os
13
+ import shutil
14
+ import subprocess
15
+ from pathlib import Path
16
+ from typing import Any, Dict, List, Optional, Sequence, Union
17
+
18
+ PathLike = Union[str, "os.PathLike[str]"]
19
+ CLI_VERSION = "0.3.0"
20
+
21
+
22
+ class SirdasError(RuntimeError):
23
+ """The CLI could not process the file. ``code`` is 1 for usage, 2 for the file."""
24
+
25
+ def __init__(self, message: str, code: int):
26
+ super().__init__(message)
27
+ self.code = code
28
+
29
+
30
+ class PasswordRequired(SirdasError):
31
+ """The PDF needs a password to open (or the one given is wrong)."""
32
+
33
+
34
+ def _default_command() -> List[str]:
35
+ # 1) An explicit binary or script, 2) sirdas on PATH, 3) npx with a pinned version.
36
+ explicit = os.environ.get("SIRDAS_BIN")
37
+ if explicit:
38
+ return ["node", explicit] if explicit.endswith(".js") else [explicit]
39
+ found = shutil.which("sirdas")
40
+ if found:
41
+ return [found]
42
+ npx = shutil.which("npx")
43
+ if npx:
44
+ return [npx, "-y", f"sirdas@{CLI_VERSION}"]
45
+ raise SirdasError(
46
+ "Sırdaş needs Node.js 20+. Install it from https://nodejs.org, or set SIRDAS_BIN to the sirdas executable.",
47
+ 1,
48
+ )
49
+
50
+
51
+ class Sirdas:
52
+ """Client. Reuse one instance; it only resolves the executable once."""
53
+
54
+ def __init__(self, command: Optional[Sequence[str]] = None, timeout: Optional[float] = 600):
55
+ self.command = list(command) if command else _default_command()
56
+ self.timeout = timeout
57
+
58
+ def _run(self, args: Sequence[str], password: Optional[str] = None, parse: bool = True) -> Any:
59
+ env = dict(os.environ)
60
+ if password is not None:
61
+ # Por entorno y no por argumento: los argumentos se ven en `ps`.
62
+ env["SIRDAS_PASSWORD"] = password
63
+ proc = subprocess.run(
64
+ [*self.command, *[str(a) for a in args]],
65
+ capture_output=True,
66
+ text=True,
67
+ encoding="utf-8",
68
+ env=env,
69
+ stdin=subprocess.DEVNULL,
70
+ timeout=self.timeout,
71
+ )
72
+ if proc.returncode != 0:
73
+ message = (proc.stderr or proc.stdout).strip() or f"sirdas exited with {proc.returncode}"
74
+ if "contraseña" in message.lower() or "password" in message.lower():
75
+ raise PasswordRequired(message, proc.returncode)
76
+ raise SirdasError(message, proc.returncode)
77
+ return json.loads(proc.stdout) if parse else proc.stdout
78
+
79
+ # ---- reading ----
80
+
81
+ def read(
82
+ self,
83
+ path: PathLike,
84
+ *,
85
+ password: Optional[str] = None,
86
+ anonymize: bool = True,
87
+ ocr: bool = True,
88
+ max_tokens: int = 1500,
89
+ ) -> Dict[str, Any]:
90
+ """Everything in one call: type, cited fields (page, evidence, bbox), tables,
91
+ form fields, checkboxes, batch segments, anonymized Markdown, chunks, next steps."""
92
+ args = ["leer", path, "--json", "--max-tokens", max_tokens]
93
+ if not anonymize:
94
+ args.append("--no-anonymize")
95
+ if not ocr:
96
+ args.append("--no-ocr")
97
+ result = self._run(args, password=password)
98
+ if result.get("status") in ("needs_password", "wrong_password"):
99
+ raise PasswordRequired(result["status"], 2)
100
+ # Mismo nombre que devuelve el MCP, para que un ejemplo sirva en los dos.
101
+ result.setdefault("document_type", (result.get("document") or {}).get("type"))
102
+ return result
103
+
104
+ def convert(self, path: PathLike) -> Dict[str, Any]:
105
+ """DOCX, XLSX, PPTX, ODT, EPUB, HTML, CSV, EML, RTF… → {format, markdown, tables, warnings}."""
106
+ return self._run(["convertir", path, "--json"])
107
+
108
+ def split_documents(self, path: PathLike, output_dir: Optional[PathLike] = None) -> Dict[str, Any]:
109
+ """Where each document starts inside a batch PDF; writes one PDF each if ``output_dir``."""
110
+ args: List[Any] = ["separar", path, "--json"]
111
+ if output_dir:
112
+ args += ["-d", output_dir, "--force"]
113
+ return self._run(args)
114
+
115
+ def forms(self, path: PathLike) -> Dict[str, Any]:
116
+ """Fillable form fields and checkboxes."""
117
+ return self._run(["formulario", path, "--json"])
118
+
119
+ def classify(self, path: PathLike) -> Dict[str, Any]:
120
+ return self._run(["classify", path, "--json"])
121
+
122
+ def chunks(self, path: PathLike, *, max_tokens: int = 1500, anonymize: bool = True) -> List[Dict[str, Any]]:
123
+ """Cited chunks: stable ``id``, ``pages``, ``section``, ``text``, ``tokens``."""
124
+ args: List[Any] = ["chunks", path, "--json", "--max-tokens", max_tokens]
125
+ if not anonymize:
126
+ args.append("--no-anonymize")
127
+ return self._run(args)["chunks"]
128
+
129
+ def tables(self, path: PathLike) -> List[Dict[str, Any]]:
130
+ return self._run(["tables", path, "--json"])["tables"]
131
+
132
+ # ---- ML ----
133
+
134
+ def dataset(
135
+ self,
136
+ paths: Sequence[PathLike],
137
+ output_dir: PathLike,
138
+ *,
139
+ format: str = "chat",
140
+ split: bool = True,
141
+ anonymize: bool = True,
142
+ dedupe: bool = True,
143
+ max_tokens: int = 2000,
144
+ hugging_face: bool = True,
145
+ ) -> List[str]:
146
+ """Write a training/RAG dataset. Formats: text, chat, prompt-completion, alpaca,
147
+ sharegpt, langchain, llamaindex, embeddings. Returns the written files."""
148
+ args: List[Any] = ["dataset", *paths, "--format", format, "--max-tokens", max_tokens, "-d", output_dir, "--force"]
149
+ if split:
150
+ args.append("--split")
151
+ if not anonymize:
152
+ args.append("--no-anonymize")
153
+ if not dedupe:
154
+ args.append("--no-dedupe")
155
+ if hugging_face:
156
+ args.append("--hf")
157
+ out = self._run(args, parse=False)
158
+ return [line for line in out.splitlines() if line.strip() and Path(line.strip()).exists()]
159
+
160
+ # ---- security ----
161
+
162
+ def protect(self, path: PathLike, password: str, output: PathLike, *, allow_copy: bool = True, allow_print: bool = True) -> Path:
163
+ args: List[Any] = ["protect", path, "-o", output, "--force"]
164
+ if not allow_copy:
165
+ args.append("--no-copy")
166
+ if not allow_print:
167
+ args.append("--no-print")
168
+ self._run(args, password=password, parse=False)
169
+ return Path(output)
170
+
171
+ def unlock(self, path: PathLike, output: PathLike, password: str = "") -> Path:
172
+ self._run(["unlock", path, "-o", output, "--force"], password=password, parse=False)
173
+ return Path(output)
174
+
175
+
176
+ _default: Optional[Sirdas] = None
177
+
178
+
179
+ def _client() -> Sirdas:
180
+ global _default
181
+ if _default is None:
182
+ _default = Sirdas()
183
+ return _default
184
+
185
+
186
+ def read(path: PathLike, **kw: Any) -> Dict[str, Any]:
187
+ return _client().read(path, **kw)
188
+
189
+
190
+ def convert(path: PathLike) -> Dict[str, Any]:
191
+ return _client().convert(path)
192
+
193
+
194
+ def split_documents(path: PathLike, output_dir: Optional[PathLike] = None) -> Dict[str, Any]:
195
+ return _client().split_documents(path, output_dir)
196
+
197
+
198
+ def forms(path: PathLike) -> Dict[str, Any]:
199
+ return _client().forms(path)
200
+
201
+
202
+ def classify(path: PathLike) -> Dict[str, Any]:
203
+ return _client().classify(path)
204
+
205
+
206
+ def chunks(path: PathLike, **kw: Any) -> List[Dict[str, Any]]:
207
+ return _client().chunks(path, **kw)
208
+
209
+
210
+ def tables(path: PathLike) -> List[Dict[str, Any]]:
211
+ return _client().tables(path)
212
+
213
+
214
+ def dataset(paths: Sequence[PathLike], output_dir: PathLike, **kw: Any) -> List[str]:
215
+ return _client().dataset(paths, output_dir, **kw)
216
+
217
+
218
+ def protect(path: PathLike, password: str, output: PathLike, **kw: Any) -> Path:
219
+ return _client().protect(path, password, output, **kw)
220
+
221
+
222
+ def unlock(path: PathLike, output: PathLike, password: str = "") -> Path:
223
+ return _client().unlock(path, output, password)
sirdas/integrations.py ADDED
@@ -0,0 +1,57 @@
1
+ """LangChain and LlamaIndex loaders. Imported lazily so neither is a hard dependency."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any, Iterator, List, Optional
6
+
7
+ from .client import PathLike, Sirdas
8
+
9
+
10
+ def _metadata(doc: dict, chunk: dict, source: str) -> dict:
11
+ return {
12
+ "source": source,
13
+ "id": chunk["id"],
14
+ "pages": chunk.get("pages"),
15
+ "section": " › ".join(chunk.get("section") or []),
16
+ "document_type": doc.get("document_type"),
17
+ "tokens": chunk.get("tokens"),
18
+ }
19
+
20
+
21
+ class SirdasLoader:
22
+ """LangChain document loader: one Document per cited chunk.
23
+
24
+ from sirdas.integrations import SirdasLoader
25
+ docs = SirdasLoader("contrato.pdf").load()
26
+ """
27
+
28
+ def __init__(self, path: PathLike, *, anonymize: bool = True, max_tokens: int = 1000, client: Optional[Sirdas] = None):
29
+ self.path, self.anonymize, self.max_tokens = str(path), anonymize, max_tokens
30
+ self.client = client or Sirdas()
31
+
32
+ def lazy_load(self) -> Iterator[Any]:
33
+ from langchain_core.documents import Document
34
+
35
+ result = self.client.read(self.path, anonymize=self.anonymize, max_tokens=self.max_tokens)
36
+ for chunk in result.get("chunks", []):
37
+ yield Document(page_content=chunk["text"], metadata=_metadata(result, chunk, self.path), id=chunk["id"])
38
+
39
+ def load(self) -> List[Any]:
40
+ return list(self.lazy_load())
41
+
42
+
43
+ class SirdasReader:
44
+ """LlamaIndex reader: one TextNode-ready Document per cited chunk, ids stable across runs."""
45
+
46
+ def __init__(self, *, anonymize: bool = True, max_tokens: int = 1000, client: Optional[Sirdas] = None):
47
+ self.anonymize, self.max_tokens = anonymize, max_tokens
48
+ self.client = client or Sirdas()
49
+
50
+ def load_data(self, file: PathLike, extra_info: Optional[dict] = None) -> List[Any]:
51
+ from llama_index.core import Document
52
+
53
+ result = self.client.read(str(file), anonymize=self.anonymize, max_tokens=self.max_tokens)
54
+ return [
55
+ Document(text=c["text"], id_=c["id"], metadata={**_metadata(result, c, str(file)), **(extra_info or {})})
56
+ for c in result.get("chunks", [])
57
+ ]
@@ -0,0 +1,53 @@
1
+ Metadata-Version: 2.5
2
+ Name: sirdas
3
+ Version: 0.3.0
4
+ Summary: Local document parsing for AI agents and ML: PDF, Word, Excel, PowerPoint, HTML, email and scans to Markdown, cited fields, chunks and datasets. Nothing is uploaded.
5
+ Project-URL: Homepage, https://sirdas.app
6
+ Project-URL: Documentation, https://sirdas.app/docs/agents/python.md
7
+ Author: Sırdaş
8
+ License-Expression: MIT
9
+ Keywords: ai-agents,anonymization,dataset,document-parsing,fine-tuning,langchain,llamaindex,llm,local-first,ocr,pdf,rag
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
13
+ Classifier: Topic :: Text Processing
14
+ Requires-Python: >=3.9
15
+ Provides-Extra: langchain
16
+ Requires-Dist: langchain-core>=0.2; extra == 'langchain'
17
+ Provides-Extra: llamaindex
18
+ Requires-Dist: llama-index-core>=0.10; extra == 'llamaindex'
19
+ Description-Content-Type: text/markdown
20
+
21
+ # sirdas (Python)
22
+
23
+ Local document parsing for AI agents and machine learning. PDF (including scans), Word, Excel, PowerPoint, OpenDocument, EPUB, HTML, CSV, email and images become Markdown, **fields cited with page and position**, tables, form fields, cited chunks and training datasets. **Nothing is uploaded**: it runs on your machine.
24
+
25
+ ```bash
26
+ pip install sirdas # needs Node.js 20+ (the engine is shared with the CLI and MCP server)
27
+ ```
28
+
29
+ ```python
30
+ import sirdas
31
+
32
+ doc = sirdas.read("factura.pdf") # one call does it all
33
+ doc["document_type"] # "factura"
34
+ {f["name"]: f["value"] for f in doc["fields"]} # número, CUFE, NIT, IVA, total… each with page, evidence and bbox
35
+ doc["next_steps"] # what to do next
36
+
37
+ sirdas.convert("informe.docx")["markdown"]
38
+ sirdas.split_documents("lote-escaneado.pdf", output_dir="separados/")
39
+ sirdas.forms("formulario.pdf") # filled fields and checkboxes
40
+ sirdas.dataset(["a.pdf", "b.docx"], "ds/", format="alpaca") # + Hugging Face dataset card
41
+ ```
42
+
43
+ LangChain and LlamaIndex:
44
+
45
+ ```python
46
+ from sirdas.integrations import SirdasLoader, SirdasReader
47
+ docs = SirdasLoader("contrato.pdf").load() # langchain Documents, stable ids
48
+ nodes = SirdasReader().load_data("historia.pdf") # llama_index Documents
49
+ ```
50
+
51
+ Personal data is anonymized by default (`anonymize=False` to keep it). Password-protected PDFs raise `PasswordRequired`; pass `password=`.
52
+
53
+ Docs: https://sirdas.app/docs/agents/python.md
@@ -0,0 +1,6 @@
1
+ sirdas/__init__.py,sha256=ZbQ8eLwRc5QcAbHdd_2e1umEGOyNLH9Z3wHSRMBgodY,794
2
+ sirdas/client.py,sha256=LN-IZ_beh3nS4Nym03a1hSAq8Lvhpg351Wrr5Q19PJ8,8079
3
+ sirdas/integrations.py,sha256=Y6yMDV3mvbnP3SPqXRhu3t_zXiFAIBOBowlzGc_-An0,2180
4
+ sirdas-0.3.0.dist-info/METADATA,sha256=bycUY_9mIRFSToznNKzbWPiNPI7vHsZHP4E-FdXAaEQ,2484
5
+ sirdas-0.3.0.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
6
+ sirdas-0.3.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any