langchain-flatmark 1.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,5 @@
1
+ """LangChain integration for flatmark."""
2
+
3
+ from langchain_flatmark.document_loaders import FlatmarkLoader, FlatmarkError
4
+
5
+ __all__ = ["FlatmarkLoader", "FlatmarkError"]
@@ -0,0 +1,163 @@
1
+ """flatmark document loader for LangChain.
2
+
3
+ Generated by listing-sync from the flatmark OpenAPI document; changes here are
4
+ overwritten.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import mimetypes
10
+ import os
11
+ import time
12
+ from collections.abc import Iterable, Iterator
13
+ from pathlib import Path, PurePosixPath
14
+ from urllib.parse import quote, urlsplit
15
+
16
+ import httpx
17
+ from langchain_core.document_loaders import BaseLoader
18
+ from langchain_core.documents import Document
19
+
20
+ BASE_URL = "https://api.flatmark.dev"
21
+ API_KEY_ENV = "FLATMARK_API_KEY"
22
+ KEY_URL = "https://flatmark.dev/go/langchain?to=/app/api-keys"
23
+ AUTH_HEADER = "X-API-Key"
24
+ USER_AGENT = "langchain-flatmark"
25
+ CONVERT_PATH = "/v1/convert"
26
+ CONVERT_FILE = "file"
27
+ TEXT_FIELD = "markdown"
28
+ META_FIELD = "meta"
29
+ SUBMIT_PATH = "/v1/convert/jobs"
30
+ SUBMIT_FILE = "file"
31
+ POLL_PATH = "/v1/jobs/{job_id}"
32
+ RESULT_PATH = "/v1/convert/jobs/{job_id}/result"
33
+ JOB_ID = "job_id"
34
+ DONE = ('succeeded', 'failed')
35
+ # Standard types older Pythons' mimetypes table lacks (added in 3.13).
36
+ OFFICE_TYPES = {
37
+ ".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
38
+ ".pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
39
+ ".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
40
+ }
41
+
42
+
43
+ class FlatmarkError(RuntimeError):
44
+ """The flatmark API answered with an error, or a queued job failed."""
45
+
46
+
47
+ def _media_type(name: str) -> str:
48
+ guess = mimetypes.guess_type(name)[0]
49
+ return guess or OFFICE_TYPES.get(Path(name).suffix.lower(), "application/octet-stream")
50
+
51
+
52
+ class FlatmarkLoader(BaseLoader):
53
+ """Load files and URLs as Markdown `Document`s through the flatmark API.
54
+
55
+ Each source becomes one `Document` whose `page_content` is the Markdown and
56
+ whose metadata carries `source` (the path or URL as given).
57
+
58
+ Args:
59
+ file_path: A local path or an http(s) URL, or an iterable of them.
60
+ api_key: The flatmark API key; defaults to the `FLATMARK_API_KEY`
61
+ environment variable. Optional for direct conversion (anonymous
62
+ calls run at a lower rate limit), required with `use_queue=True`.
63
+ use_queue: Convert through the queue (`POST /v1/convert/jobs`) instead of
64
+ the direct call (`POST /v1/convert`): the loader submits a job, polls
65
+ it every `poll_interval` seconds until it finishes or `timeout`
66
+ seconds pass, and downloads its Markdown.
67
+ base_url: The API base URL.
68
+ poll_interval: Seconds between job polls.
69
+ timeout: Seconds a queued job may take before the loader gives up.
70
+ client: An `httpx.Client` to send every request with (for proxies,
71
+ retries or tests); the loader makes and closes its own otherwise.
72
+ """
73
+
74
+ def __init__(
75
+ self,
76
+ file_path: str | os.PathLike | Iterable[str | os.PathLike],
77
+ *,
78
+ api_key: str | None = None,
79
+ use_queue: bool = False,
80
+ base_url: str = BASE_URL,
81
+ poll_interval: float = 2.0,
82
+ timeout: float = 600.0,
83
+ client: httpx.Client | None = None,
84
+ ) -> None:
85
+ if isinstance(file_path, (str, os.PathLike)):
86
+ file_path = [file_path]
87
+ self.file_paths = [os.fspath(p) for p in file_path]
88
+ self.api_key = api_key or os.environ.get(API_KEY_ENV) or None
89
+ if use_queue and not self.api_key:
90
+ raise ValueError(
91
+ f"use_queue=True needs an API key: pass api_key= or set "
92
+ f"{API_KEY_ENV}. Get one at {KEY_URL}"
93
+ )
94
+ self.use_queue = use_queue
95
+ self.base_url = base_url.rstrip("/")
96
+ self.poll_interval = poll_interval
97
+ self.timeout = timeout
98
+ self.client = client
99
+
100
+ def lazy_load(self) -> Iterator[Document]:
101
+ """Convert each source in turn, yielding one `Document` per source."""
102
+ client = self.client or httpx.Client(timeout=120.0, follow_redirects=True)
103
+ try:
104
+ for source in self.file_paths:
105
+ name, data, media = self._read(client, source)
106
+ if self.use_queue:
107
+ yield self._queued(client, source, name, data, media)
108
+ else:
109
+ yield self._direct(client, source, name, data, media)
110
+ finally:
111
+ if self.client is None:
112
+ client.close()
113
+
114
+ def _read(self, client: httpx.Client, source: str) -> tuple[str, bytes, str]:
115
+ """(file name, bytes, media type) of a path or URL. A URL is fetched
116
+ without the API key: it goes to someone else's server."""
117
+ if source.startswith(("http://", "https://")):
118
+ response = client.get(source, headers={"User-Agent": USER_AGENT})
119
+ response.raise_for_status()
120
+ name = PurePosixPath(urlsplit(source).path).name or "document"
121
+ media = response.headers.get("content-type", "").split(";")[0].strip()
122
+ if not media or media == "application/octet-stream":
123
+ media = _media_type(name)
124
+ return name, response.content, media
125
+ path = Path(source)
126
+ return path.name, path.read_bytes(), _media_type(path.name)
127
+
128
+ def _call(self, client: httpx.Client, method: str, path: str, **kwargs) -> httpx.Response:
129
+ headers = {"User-Agent": USER_AGENT}
130
+ if self.api_key:
131
+ headers[AUTH_HEADER] = self.api_key
132
+ response = client.request(method, self.base_url + path, headers=headers, **kwargs)
133
+ if response.is_error:
134
+ message = f"{method} {path} answered {response.status_code}: {response.text[:500]}"
135
+ if response.status_code == 429 and not self.api_key:
136
+ message += f" (an API key lifts the limit: {KEY_URL})"
137
+ raise FlatmarkError(message)
138
+ return response
139
+
140
+ def _direct(self, client, source, name, data, media) -> Document:
141
+ answer = self._call(
142
+ client, "POST", CONVERT_PATH, files={CONVERT_FILE: (name, data, media)}
143
+ ).json()
144
+ meta = (answer.get(META_FIELD) or {}) if META_FIELD else {}
145
+ return Document(page_content=answer[TEXT_FIELD], metadata={**meta, "source": source})
146
+
147
+ def _queued(self, client, source, name, data, media) -> Document:
148
+ job_id = self._call(
149
+ client, "POST", SUBMIT_PATH, files={SUBMIT_FILE: (name, data, media)}
150
+ ).json()[JOB_ID]
151
+ at = "{" + JOB_ID + "}"
152
+ deadline = time.monotonic() + self.timeout
153
+ while True:
154
+ job = self._call(client, "GET", POLL_PATH.replace(at, quote(job_id, safe=""))).json()
155
+ if job.get("status") in DONE:
156
+ break
157
+ if time.monotonic() >= deadline:
158
+ raise FlatmarkError(f"job {job_id} not finished after {self.timeout:g} s")
159
+ time.sleep(self.poll_interval)
160
+ if job["status"] != DONE[0]:
161
+ raise FlatmarkError(f"job {job_id} {job['status']}: {job.get('error') or 'no reason given'}")
162
+ text = self._call(client, "GET", RESULT_PATH.replace(at, quote(job_id, safe=""))).text
163
+ return Document(page_content=text, metadata={"source": source, JOB_ID: job_id})
File without changes
@@ -0,0 +1,55 @@
1
+ Metadata-Version: 2.5
2
+ Name: langchain-flatmark
3
+ Version: 1.0.1
4
+ Summary: Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
5
+ Project-URL: Homepage, https://flatmark.dev
6
+ Project-URL: Documentation, https://github.com/flatmark-dev/flatmark-integrations/blob/main/langchain/README.md
7
+ Project-URL: Repository, https://github.com/flatmark-dev/flatmark-integrations
8
+ Project-URL: Issues, https://github.com/flatmark-dev/flatmark-integrations/issues
9
+ Author-email: podshalocef <contact@podshalocef.com>
10
+ License-Expression: MIT
11
+ Keywords: document-loader,flatmark,langchain,markdown
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
+ Requires-Python: >=3.10
16
+ Requires-Dist: httpx<1,>=0.28
17
+ Requires-Dist: langchain-core<2,>=1.6
18
+ Description-Content-Type: text/markdown
19
+
20
+ # langchain-flatmark
21
+
22
+ Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
23
+
24
+ A LangChain document loader for the [flatmark API](https://flatmark.dev): `FlatmarkLoader` turns local files and URLs into Markdown `Document`s.
25
+
26
+ ```sh
27
+ pip install langchain-flatmark
28
+ ```
29
+
30
+ ```python
31
+ from langchain_flatmark import FlatmarkLoader
32
+
33
+ docs = FlatmarkLoader("report.pdf").load()
34
+ print(docs[0].page_content)
35
+ ```
36
+
37
+ Direct conversion (`POST /v1/convert`) works without a key at a lower rate limit. [Get an API key](https://flatmark.dev/go/langchain?to=/app/api-keys) and pass it as `api_key=` or set `FLATMARK_API_KEY`.
38
+
39
+ For large or scanned files, convert through the queue (an API key is required):
40
+
41
+ ```python
42
+ loader = FlatmarkLoader(["scan.pdf", "https://example.com/deck.pptx"], use_queue=True)
43
+ for doc in loader.lazy_load():
44
+ print(doc.metadata, len(doc.page_content))
45
+ ```
46
+
47
+ The loader submits `POST /v1/convert/jobs`, polls `GET /v1/jobs/{job_id}` until the job's status is `succeeded` or `failed`, then downloads `GET /v1/convert/jobs/{job_id}/result`.
48
+
49
+ ## Documents
50
+
51
+ One `Document` per source; `page_content` is the Markdown. `metadata` carries `source` (the path or URL as given) plus the `meta` fields of the answer — and `job_id` for a queued conversion.
52
+
53
+ A URL is downloaded by the loader without your key, then uploaded. Other options: `base_url`, `poll_interval`, `timeout`, and `client` (an `httpx.Client` for proxies or retries).
54
+
55
+ API reference: https://flatmark.dev/docs · Support: https://flatmark.dev/support · Generated from [`openapi.json`](../openapi.json).
@@ -0,0 +1,6 @@
1
+ langchain_flatmark/__init__.py,sha256=fC_5WRmQA5m9rskV-i-uWnptvaZdb1N4hPCv9xCc12c,168
2
+ langchain_flatmark/document_loaders.py,sha256=bhqBYdWdZlB_TRedcvSZQWKK-1VY-UkmsAwSmx1EuyA,7114
3
+ langchain_flatmark/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ langchain_flatmark-1.0.1.dist-info/METADATA,sha256=4WX6bFl6Jg4jFM3iCsWhs53gMNuefh_bStD7ktbUrjE,2586
5
+ langchain_flatmark-1.0.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
6
+ langchain_flatmark-1.0.1.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.32.4
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any