langchain-flatmark 1.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""flatmark document loader for LangChain.
|
|
2
|
+
|
|
3
|
+
Generated by listing-sync from the flatmark OpenAPI document; changes here are
|
|
4
|
+
overwritten.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import mimetypes
|
|
10
|
+
import os
|
|
11
|
+
import time
|
|
12
|
+
from collections.abc import Iterable, Iterator
|
|
13
|
+
from pathlib import Path, PurePosixPath
|
|
14
|
+
from urllib.parse import quote, urlsplit
|
|
15
|
+
|
|
16
|
+
import httpx
|
|
17
|
+
from langchain_core.document_loaders import BaseLoader
|
|
18
|
+
from langchain_core.documents import Document
|
|
19
|
+
|
|
20
|
+
BASE_URL = "https://api.flatmark.dev"
|
|
21
|
+
API_KEY_ENV = "FLATMARK_API_KEY"
|
|
22
|
+
KEY_URL = "https://flatmark.dev/go/langchain?to=/app/api-keys"
|
|
23
|
+
AUTH_HEADER = "X-API-Key"
|
|
24
|
+
USER_AGENT = "langchain-flatmark"
|
|
25
|
+
CONVERT_PATH = "/v1/convert"
|
|
26
|
+
CONVERT_FILE = "file"
|
|
27
|
+
TEXT_FIELD = "markdown"
|
|
28
|
+
META_FIELD = "meta"
|
|
29
|
+
SUBMIT_PATH = "/v1/convert/jobs"
|
|
30
|
+
SUBMIT_FILE = "file"
|
|
31
|
+
POLL_PATH = "/v1/jobs/{job_id}"
|
|
32
|
+
RESULT_PATH = "/v1/convert/jobs/{job_id}/result"
|
|
33
|
+
JOB_ID = "job_id"
|
|
34
|
+
DONE = ('succeeded', 'failed')
|
|
35
|
+
# Standard types older Pythons' mimetypes table lacks (added in 3.13).
|
|
36
|
+
OFFICE_TYPES = {
|
|
37
|
+
".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
38
|
+
".pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
|
39
|
+
".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class FlatmarkError(RuntimeError):
|
|
44
|
+
"""The flatmark API answered with an error, or a queued job failed."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _media_type(name: str) -> str:
|
|
48
|
+
guess = mimetypes.guess_type(name)[0]
|
|
49
|
+
return guess or OFFICE_TYPES.get(Path(name).suffix.lower(), "application/octet-stream")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class FlatmarkLoader(BaseLoader):
|
|
53
|
+
"""Load files and URLs as Markdown `Document`s through the flatmark API.
|
|
54
|
+
|
|
55
|
+
Each source becomes one `Document` whose `page_content` is the Markdown and
|
|
56
|
+
whose metadata carries `source` (the path or URL as given).
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
file_path: A local path or an http(s) URL, or an iterable of them.
|
|
60
|
+
api_key: The flatmark API key; defaults to the `FLATMARK_API_KEY`
|
|
61
|
+
environment variable. Optional for direct conversion (anonymous
|
|
62
|
+
calls run at a lower rate limit), required with `use_queue=True`.
|
|
63
|
+
use_queue: Convert through the queue (`POST /v1/convert/jobs`) instead of
|
|
64
|
+
the direct call (`POST /v1/convert`): the loader submits a job, polls
|
|
65
|
+
it every `poll_interval` seconds until it finishes or `timeout`
|
|
66
|
+
seconds pass, and downloads its Markdown.
|
|
67
|
+
base_url: The API base URL.
|
|
68
|
+
poll_interval: Seconds between job polls.
|
|
69
|
+
timeout: Seconds a queued job may take before the loader gives up.
|
|
70
|
+
client: An `httpx.Client` to send every request with (for proxies,
|
|
71
|
+
retries or tests); the loader makes and closes its own otherwise.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
def __init__(
|
|
75
|
+
self,
|
|
76
|
+
file_path: str | os.PathLike | Iterable[str | os.PathLike],
|
|
77
|
+
*,
|
|
78
|
+
api_key: str | None = None,
|
|
79
|
+
use_queue: bool = False,
|
|
80
|
+
base_url: str = BASE_URL,
|
|
81
|
+
poll_interval: float = 2.0,
|
|
82
|
+
timeout: float = 600.0,
|
|
83
|
+
client: httpx.Client | None = None,
|
|
84
|
+
) -> None:
|
|
85
|
+
if isinstance(file_path, (str, os.PathLike)):
|
|
86
|
+
file_path = [file_path]
|
|
87
|
+
self.file_paths = [os.fspath(p) for p in file_path]
|
|
88
|
+
self.api_key = api_key or os.environ.get(API_KEY_ENV) or None
|
|
89
|
+
if use_queue and not self.api_key:
|
|
90
|
+
raise ValueError(
|
|
91
|
+
f"use_queue=True needs an API key: pass api_key= or set "
|
|
92
|
+
f"{API_KEY_ENV}. Get one at {KEY_URL}"
|
|
93
|
+
)
|
|
94
|
+
self.use_queue = use_queue
|
|
95
|
+
self.base_url = base_url.rstrip("/")
|
|
96
|
+
self.poll_interval = poll_interval
|
|
97
|
+
self.timeout = timeout
|
|
98
|
+
self.client = client
|
|
99
|
+
|
|
100
|
+
def lazy_load(self) -> Iterator[Document]:
|
|
101
|
+
"""Convert each source in turn, yielding one `Document` per source."""
|
|
102
|
+
client = self.client or httpx.Client(timeout=120.0, follow_redirects=True)
|
|
103
|
+
try:
|
|
104
|
+
for source in self.file_paths:
|
|
105
|
+
name, data, media = self._read(client, source)
|
|
106
|
+
if self.use_queue:
|
|
107
|
+
yield self._queued(client, source, name, data, media)
|
|
108
|
+
else:
|
|
109
|
+
yield self._direct(client, source, name, data, media)
|
|
110
|
+
finally:
|
|
111
|
+
if self.client is None:
|
|
112
|
+
client.close()
|
|
113
|
+
|
|
114
|
+
def _read(self, client: httpx.Client, source: str) -> tuple[str, bytes, str]:
|
|
115
|
+
"""(file name, bytes, media type) of a path or URL. A URL is fetched
|
|
116
|
+
without the API key: it goes to someone else's server."""
|
|
117
|
+
if source.startswith(("http://", "https://")):
|
|
118
|
+
response = client.get(source, headers={"User-Agent": USER_AGENT})
|
|
119
|
+
response.raise_for_status()
|
|
120
|
+
name = PurePosixPath(urlsplit(source).path).name or "document"
|
|
121
|
+
media = response.headers.get("content-type", "").split(";")[0].strip()
|
|
122
|
+
if not media or media == "application/octet-stream":
|
|
123
|
+
media = _media_type(name)
|
|
124
|
+
return name, response.content, media
|
|
125
|
+
path = Path(source)
|
|
126
|
+
return path.name, path.read_bytes(), _media_type(path.name)
|
|
127
|
+
|
|
128
|
+
def _call(self, client: httpx.Client, method: str, path: str, **kwargs) -> httpx.Response:
|
|
129
|
+
headers = {"User-Agent": USER_AGENT}
|
|
130
|
+
if self.api_key:
|
|
131
|
+
headers[AUTH_HEADER] = self.api_key
|
|
132
|
+
response = client.request(method, self.base_url + path, headers=headers, **kwargs)
|
|
133
|
+
if response.is_error:
|
|
134
|
+
message = f"{method} {path} answered {response.status_code}: {response.text[:500]}"
|
|
135
|
+
if response.status_code == 429 and not self.api_key:
|
|
136
|
+
message += f" (an API key lifts the limit: {KEY_URL})"
|
|
137
|
+
raise FlatmarkError(message)
|
|
138
|
+
return response
|
|
139
|
+
|
|
140
|
+
def _direct(self, client, source, name, data, media) -> Document:
|
|
141
|
+
answer = self._call(
|
|
142
|
+
client, "POST", CONVERT_PATH, files={CONVERT_FILE: (name, data, media)}
|
|
143
|
+
).json()
|
|
144
|
+
meta = (answer.get(META_FIELD) or {}) if META_FIELD else {}
|
|
145
|
+
return Document(page_content=answer[TEXT_FIELD], metadata={**meta, "source": source})
|
|
146
|
+
|
|
147
|
+
def _queued(self, client, source, name, data, media) -> Document:
|
|
148
|
+
job_id = self._call(
|
|
149
|
+
client, "POST", SUBMIT_PATH, files={SUBMIT_FILE: (name, data, media)}
|
|
150
|
+
).json()[JOB_ID]
|
|
151
|
+
at = "{" + JOB_ID + "}"
|
|
152
|
+
deadline = time.monotonic() + self.timeout
|
|
153
|
+
while True:
|
|
154
|
+
job = self._call(client, "GET", POLL_PATH.replace(at, quote(job_id, safe=""))).json()
|
|
155
|
+
if job.get("status") in DONE:
|
|
156
|
+
break
|
|
157
|
+
if time.monotonic() >= deadline:
|
|
158
|
+
raise FlatmarkError(f"job {job_id} not finished after {self.timeout:g} s")
|
|
159
|
+
time.sleep(self.poll_interval)
|
|
160
|
+
if job["status"] != DONE[0]:
|
|
161
|
+
raise FlatmarkError(f"job {job_id} {job['status']}: {job.get('error') or 'no reason given'}")
|
|
162
|
+
text = self._call(client, "GET", RESULT_PATH.replace(at, quote(job_id, safe=""))).text
|
|
163
|
+
return Document(page_content=text, metadata={"source": source, JOB_ID: job_id})
|
|
File without changes
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: langchain-flatmark
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
|
|
5
|
+
Project-URL: Homepage, https://flatmark.dev
|
|
6
|
+
Project-URL: Documentation, https://github.com/flatmark-dev/flatmark-integrations/blob/main/langchain/README.md
|
|
7
|
+
Project-URL: Repository, https://github.com/flatmark-dev/flatmark-integrations
|
|
8
|
+
Project-URL: Issues, https://github.com/flatmark-dev/flatmark-integrations/issues
|
|
9
|
+
Author-email: podshalocef <contact@podshalocef.com>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
Keywords: document-loader,flatmark,langchain,markdown
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Requires-Dist: httpx<1,>=0.28
|
|
17
|
+
Requires-Dist: langchain-core<2,>=1.6
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# langchain-flatmark
|
|
21
|
+
|
|
22
|
+
Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
|
|
23
|
+
|
|
24
|
+
A LangChain document loader for the [flatmark API](https://flatmark.dev): `FlatmarkLoader` turns local files and URLs into Markdown `Document`s.
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
pip install langchain-flatmark
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from langchain_flatmark import FlatmarkLoader
|
|
32
|
+
|
|
33
|
+
docs = FlatmarkLoader("report.pdf").load()
|
|
34
|
+
print(docs[0].page_content)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Direct conversion (`POST /v1/convert`) works without a key at a lower rate limit. [Get an API key](https://flatmark.dev/go/langchain?to=/app/api-keys) and pass it as `api_key=` or set `FLATMARK_API_KEY`.
|
|
38
|
+
|
|
39
|
+
For large or scanned files, convert through the queue (an API key is required):
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
loader = FlatmarkLoader(["scan.pdf", "https://example.com/deck.pptx"], use_queue=True)
|
|
43
|
+
for doc in loader.lazy_load():
|
|
44
|
+
print(doc.metadata, len(doc.page_content))
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
The loader submits `POST /v1/convert/jobs`, polls `GET /v1/jobs/{job_id}` until the job's status is `succeeded` or `failed`, then downloads `GET /v1/convert/jobs/{job_id}/result`.
|
|
48
|
+
|
|
49
|
+
## Documents
|
|
50
|
+
|
|
51
|
+
One `Document` per source; `page_content` is the Markdown. `metadata` carries `source` (the path or URL as given) plus the `meta` fields of the answer — and `job_id` for a queued conversion.
|
|
52
|
+
|
|
53
|
+
A URL is downloaded by the loader without your key, then uploaded. Other options: `base_url`, `poll_interval`, `timeout`, and `client` (an `httpx.Client` for proxies or retries).
|
|
54
|
+
|
|
55
|
+
API reference: https://flatmark.dev/docs · Support: https://flatmark.dev/support · Generated from [`openapi.json`](../openapi.json).
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
langchain_flatmark/__init__.py,sha256=fC_5WRmQA5m9rskV-i-uWnptvaZdb1N4hPCv9xCc12c,168
|
|
2
|
+
langchain_flatmark/document_loaders.py,sha256=bhqBYdWdZlB_TRedcvSZQWKK-1VY-UkmsAwSmx1EuyA,7114
|
|
3
|
+
langchain_flatmark/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
4
|
+
langchain_flatmark-1.0.1.dist-info/METADATA,sha256=4WX6bFl6Jg4jFM3iCsWhs53gMNuefh_bStD7ktbUrjE,2586
|
|
5
|
+
langchain_flatmark-1.0.1.dist-info/WHEEL,sha256=W3fkpkm7-wf9vBI5Z-7s0eWkeM-spu78I8Neb98DeEg,87
|
|
6
|
+
langchain_flatmark-1.0.1.dist-info/RECORD,,
|