langchain-flatmark 1.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,55 @@
1
+ Metadata-Version: 2.5
2
+ Name: langchain-flatmark
3
+ Version: 1.0.1
4
+ Summary: Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
5
+ Project-URL: Homepage, https://flatmark.dev
6
+ Project-URL: Documentation, https://github.com/flatmark-dev/flatmark-integrations/blob/main/langchain/README.md
7
+ Project-URL: Repository, https://github.com/flatmark-dev/flatmark-integrations
8
+ Project-URL: Issues, https://github.com/flatmark-dev/flatmark-integrations/issues
9
+ Author-email: podshalocef <contact@podshalocef.com>
10
+ License-Expression: MIT
11
+ Keywords: document-loader,flatmark,langchain,markdown
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
+ Requires-Python: >=3.10
16
+ Requires-Dist: httpx<1,>=0.28
17
+ Requires-Dist: langchain-core<2,>=1.6
18
+ Description-Content-Type: text/markdown
19
+
20
+ # langchain-flatmark
21
+
22
+ Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
23
+
24
+ A LangChain document loader for the [flatmark API](https://flatmark.dev): `FlatmarkLoader` turns local files and URLs into Markdown `Document`s.
25
+
26
+ ```sh
27
+ pip install langchain-flatmark
28
+ ```
29
+
30
+ ```python
31
+ from langchain_flatmark import FlatmarkLoader
32
+
33
+ docs = FlatmarkLoader("report.pdf").load()
34
+ print(docs[0].page_content)
35
+ ```
36
+
37
+ Direct conversion (`POST /v1/convert`) works without a key at a lower rate limit. [Get an API key](https://flatmark.dev/go/langchain?to=/app/api-keys) and pass it as `api_key=` or set `FLATMARK_API_KEY`.
38
+
39
+ For large or scanned files, convert through the queue (an API key is required):
40
+
41
+ ```python
42
+ loader = FlatmarkLoader(["scan.pdf", "https://example.com/deck.pptx"], use_queue=True)
43
+ for doc in loader.lazy_load():
44
+ print(doc.metadata, len(doc.page_content))
45
+ ```
46
+
47
+ The loader submits `POST /v1/convert/jobs`, polls `GET /v1/jobs/{job_id}` until the job's status is `succeeded` or `failed`, then downloads `GET /v1/convert/jobs/{job_id}/result`.
48
+
49
+ ## Documents
50
+
51
+ One `Document` per source; `page_content` is the Markdown. `metadata` carries `source` (the path or URL as given) plus the `meta` fields of the answer — and `job_id` for a queued conversion.
52
+
53
+ A URL is downloaded by the loader without your key, then uploaded. Other options: `base_url`, `poll_interval`, `timeout`, and `client` (an `httpx.Client` for proxies or retries).
54
+
55
+ API reference: https://flatmark.dev/docs · Support: https://flatmark.dev/support · Generated from [`openapi.json`](../openapi.json).
@@ -0,0 +1,36 @@
1
+ # langchain-flatmark
2
+
3
+ Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
4
+
5
+ A LangChain document loader for the [flatmark API](https://flatmark.dev): `FlatmarkLoader` turns local files and URLs into Markdown `Document`s.
6
+
7
+ ```sh
8
+ pip install langchain-flatmark
9
+ ```
10
+
11
+ ```python
12
+ from langchain_flatmark import FlatmarkLoader
13
+
14
+ docs = FlatmarkLoader("report.pdf").load()
15
+ print(docs[0].page_content)
16
+ ```
17
+
18
+ Direct conversion (`POST /v1/convert`) works without a key at a lower rate limit. [Get an API key](https://flatmark.dev/go/langchain?to=/app/api-keys) and pass it as `api_key=` or set `FLATMARK_API_KEY`.
19
+
20
+ For large or scanned files, convert through the queue (an API key is required):
21
+
22
+ ```python
23
+ loader = FlatmarkLoader(["scan.pdf", "https://example.com/deck.pptx"], use_queue=True)
24
+ for doc in loader.lazy_load():
25
+ print(doc.metadata, len(doc.page_content))
26
+ ```
27
+
28
+ The loader submits `POST /v1/convert/jobs`, polls `GET /v1/jobs/{job_id}` until the job's status is `succeeded` or `failed`, then downloads `GET /v1/convert/jobs/{job_id}/result`.
29
+
30
+ ## Documents
31
+
32
+ One `Document` per source; `page_content` is the Markdown. `metadata` carries `source` (the path or URL as given) plus the `meta` fields of the answer — and `job_id` for a queued conversion.
33
+
34
+ A URL is downloaded by the loader without your key, then uploaded. Other options: `base_url`, `poll_interval`, `timeout`, and `client` (an `httpx.Client` for proxies or retries).
35
+
36
+ API reference: https://flatmark.dev/docs · Support: https://flatmark.dev/support · Generated from [`openapi.json`](../openapi.json).
@@ -0,0 +1,5 @@
1
+ """LangChain integration for flatmark."""
2
+
3
+ from langchain_flatmark.document_loaders import FlatmarkLoader, FlatmarkError
4
+
5
+ __all__ = ["FlatmarkLoader", "FlatmarkError"]
@@ -0,0 +1,163 @@
1
+ """flatmark document loader for LangChain.
2
+
3
+ Generated by listing-sync from the flatmark OpenAPI document; changes here are
4
+ overwritten.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import mimetypes
10
+ import os
11
+ import time
12
+ from collections.abc import Iterable, Iterator
13
+ from pathlib import Path, PurePosixPath
14
+ from urllib.parse import quote, urlsplit
15
+
16
+ import httpx
17
+ from langchain_core.document_loaders import BaseLoader
18
+ from langchain_core.documents import Document
19
+
20
+ BASE_URL = "https://api.flatmark.dev"
21
+ API_KEY_ENV = "FLATMARK_API_KEY"
22
+ KEY_URL = "https://flatmark.dev/go/langchain?to=/app/api-keys"
23
+ AUTH_HEADER = "X-API-Key"
24
+ USER_AGENT = "langchain-flatmark"
25
+ CONVERT_PATH = "/v1/convert"
26
+ CONVERT_FILE = "file"
27
+ TEXT_FIELD = "markdown"
28
+ META_FIELD = "meta"
29
+ SUBMIT_PATH = "/v1/convert/jobs"
30
+ SUBMIT_FILE = "file"
31
+ POLL_PATH = "/v1/jobs/{job_id}"
32
+ RESULT_PATH = "/v1/convert/jobs/{job_id}/result"
33
+ JOB_ID = "job_id"
34
+ DONE = ('succeeded', 'failed')
35
+ # Standard types older Pythons' mimetypes table lacks (added in 3.13).
36
+ OFFICE_TYPES = {
37
+ ".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
38
+ ".pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
39
+ ".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
40
+ }
41
+
42
+
43
+ class FlatmarkError(RuntimeError):
44
+ """The flatmark API answered with an error, or a queued job failed."""
45
+
46
+
47
+ def _media_type(name: str) -> str:
48
+ guess = mimetypes.guess_type(name)[0]
49
+ return guess or OFFICE_TYPES.get(Path(name).suffix.lower(), "application/octet-stream")
50
+
51
+
52
+ class FlatmarkLoader(BaseLoader):
53
+ """Load files and URLs as Markdown `Document`s through the flatmark API.
54
+
55
+ Each source becomes one `Document` whose `page_content` is the Markdown and
56
+ whose metadata carries `source` (the path or URL as given).
57
+
58
+ Args:
59
+ file_path: A local path or an http(s) URL, or an iterable of them.
60
+ api_key: The flatmark API key; defaults to the `FLATMARK_API_KEY`
61
+ environment variable. Optional for direct conversion (anonymous
62
+ calls run at a lower rate limit), required with `use_queue=True`.
63
+ use_queue: Convert through the queue (`POST /v1/convert/jobs`) instead of
64
+ the direct call (`POST /v1/convert`): the loader submits a job, polls
65
+ it every `poll_interval` seconds until it finishes or `timeout`
66
+ seconds pass, and downloads its Markdown.
67
+ base_url: The API base URL.
68
+ poll_interval: Seconds between job polls.
69
+ timeout: Seconds a queued job may take before the loader gives up.
70
+ client: An `httpx.Client` to send every request with (for proxies,
71
+ retries or tests); the loader makes and closes its own otherwise.
72
+ """
73
+
74
+ def __init__(
75
+ self,
76
+ file_path: str | os.PathLike | Iterable[str | os.PathLike],
77
+ *,
78
+ api_key: str | None = None,
79
+ use_queue: bool = False,
80
+ base_url: str = BASE_URL,
81
+ poll_interval: float = 2.0,
82
+ timeout: float = 600.0,
83
+ client: httpx.Client | None = None,
84
+ ) -> None:
85
+ if isinstance(file_path, (str, os.PathLike)):
86
+ file_path = [file_path]
87
+ self.file_paths = [os.fspath(p) for p in file_path]
88
+ self.api_key = api_key or os.environ.get(API_KEY_ENV) or None
89
+ if use_queue and not self.api_key:
90
+ raise ValueError(
91
+ f"use_queue=True needs an API key: pass api_key= or set "
92
+ f"{API_KEY_ENV}. Get one at {KEY_URL}"
93
+ )
94
+ self.use_queue = use_queue
95
+ self.base_url = base_url.rstrip("/")
96
+ self.poll_interval = poll_interval
97
+ self.timeout = timeout
98
+ self.client = client
99
+
100
+ def lazy_load(self) -> Iterator[Document]:
101
+ """Convert each source in turn, yielding one `Document` per source."""
102
+ client = self.client or httpx.Client(timeout=120.0, follow_redirects=True)
103
+ try:
104
+ for source in self.file_paths:
105
+ name, data, media = self._read(client, source)
106
+ if self.use_queue:
107
+ yield self._queued(client, source, name, data, media)
108
+ else:
109
+ yield self._direct(client, source, name, data, media)
110
+ finally:
111
+ if self.client is None:
112
+ client.close()
113
+
114
+ def _read(self, client: httpx.Client, source: str) -> tuple[str, bytes, str]:
115
+ """(file name, bytes, media type) of a path or URL. A URL is fetched
116
+ without the API key: it goes to someone else's server."""
117
+ if source.startswith(("http://", "https://")):
118
+ response = client.get(source, headers={"User-Agent": USER_AGENT})
119
+ response.raise_for_status()
120
+ name = PurePosixPath(urlsplit(source).path).name or "document"
121
+ media = response.headers.get("content-type", "").split(";")[0].strip()
122
+ if not media or media == "application/octet-stream":
123
+ media = _media_type(name)
124
+ return name, response.content, media
125
+ path = Path(source)
126
+ return path.name, path.read_bytes(), _media_type(path.name)
127
+
128
+ def _call(self, client: httpx.Client, method: str, path: str, **kwargs) -> httpx.Response:
129
+ headers = {"User-Agent": USER_AGENT}
130
+ if self.api_key:
131
+ headers[AUTH_HEADER] = self.api_key
132
+ response = client.request(method, self.base_url + path, headers=headers, **kwargs)
133
+ if response.is_error:
134
+ message = f"{method} {path} answered {response.status_code}: {response.text[:500]}"
135
+ if response.status_code == 429 and not self.api_key:
136
+ message += f" (an API key lifts the limit: {KEY_URL})"
137
+ raise FlatmarkError(message)
138
+ return response
139
+
140
+ def _direct(self, client, source, name, data, media) -> Document:
141
+ answer = self._call(
142
+ client, "POST", CONVERT_PATH, files={CONVERT_FILE: (name, data, media)}
143
+ ).json()
144
+ meta = (answer.get(META_FIELD) or {}) if META_FIELD else {}
145
+ return Document(page_content=answer[TEXT_FIELD], metadata={**meta, "source": source})
146
+
147
+ def _queued(self, client, source, name, data, media) -> Document:
148
+ job_id = self._call(
149
+ client, "POST", SUBMIT_PATH, files={SUBMIT_FILE: (name, data, media)}
150
+ ).json()[JOB_ID]
151
+ at = "{" + JOB_ID + "}"
152
+ deadline = time.monotonic() + self.timeout
153
+ while True:
154
+ job = self._call(client, "GET", POLL_PATH.replace(at, quote(job_id, safe=""))).json()
155
+ if job.get("status") in DONE:
156
+ break
157
+ if time.monotonic() >= deadline:
158
+ raise FlatmarkError(f"job {job_id} not finished after {self.timeout:g} s")
159
+ time.sleep(self.poll_interval)
160
+ if job["status"] != DONE[0]:
161
+ raise FlatmarkError(f"job {job_id} {job['status']}: {job.get('error') or 'no reason given'}")
162
+ text = self._call(client, "GET", RESULT_PATH.replace(at, quote(job_id, safe=""))).text
163
+ return Document(page_content=text, metadata={"source": source, JOB_ID: job_id})
File without changes
@@ -0,0 +1,28 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "langchain-flatmark"
7
+ version = "1.0.1"
8
+ description = "Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ requires-python = ">=3.10"
12
+ authors = [{ name = "podshalocef", email = "contact@podshalocef.com" }]
13
+ keywords = ["langchain", "document-loader", "markdown", "flatmark"]
14
+ classifiers = [
15
+ "Intended Audience :: Developers",
16
+ "Programming Language :: Python :: 3",
17
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
18
+ ]
19
+ dependencies = ["langchain-core>=1.6,<2", "httpx>=0.28,<1"]
20
+
21
+ [project.urls]
22
+ Homepage = "https://flatmark.dev"
23
+ Documentation = "https://github.com/flatmark-dev/flatmark-integrations/blob/main/langchain/README.md"
24
+ Repository = "https://github.com/flatmark-dev/flatmark-integrations"
25
+ Issues = "https://github.com/flatmark-dev/flatmark-integrations/issues"
26
+
27
+ [tool.hatch.build.targets.wheel]
28
+ packages = ["langchain_flatmark"]
@@ -0,0 +1,134 @@
1
+ """Tests for langchain_flatmark, against an httpx.MockTransport (no network)."""
2
+
3
+ import asyncio
4
+ import json
5
+ import os
6
+ import tempfile
7
+ import unittest
8
+ from pathlib import Path
9
+ from unittest import mock
10
+
11
+ import httpx
12
+
13
+ from langchain_flatmark import FlatmarkLoader, FlatmarkError
14
+ from langchain_flatmark.document_loaders import (
15
+ AUTH_HEADER,
16
+ BASE_URL,
17
+ CONVERT_PATH,
18
+ JOB_ID,
19
+ META_FIELD,
20
+ POLL_PATH,
21
+ RESULT_PATH,
22
+ SUBMIT_PATH,
23
+ TEXT_FIELD,
24
+ )
25
+
26
+ AT = "{" + JOB_ID + "}"
27
+
28
+
29
+ class Api:
30
+ """A fake flatmark API: records requests, answers by path."""
31
+
32
+ def __init__(self, statuses=("running", "succeeded"), error=None):
33
+ self.requests, self.statuses, self.error = [], list(statuses), error
34
+
35
+ def __call__(self, request):
36
+ request.read()
37
+ self.requests.append(request)
38
+ path = request.url.path
39
+ if request.url.host == "files.example.com":
40
+ return httpx.Response(200, content=b"%PDF-1.7 remote", headers={"content-type": "application/pdf"})
41
+ if self.error:
42
+ return httpx.Response(self.error, json={"detail": "nope"})
43
+ if path.endswith(CONVERT_PATH):
44
+ answer = {TEXT_FIELD: "# Hello"}
45
+ if META_FIELD:
46
+ answer[META_FIELD] = {"filename": "a.pdf", "title": "Hello"}
47
+ return httpx.Response(200, json=answer)
48
+ if path.endswith(SUBMIT_PATH):
49
+ return httpx.Response(202, json={JOB_ID: "j1", "status": "queued"})
50
+ if path.endswith(POLL_PATH.replace(AT, "j1")):
51
+ status = self.statuses.pop(0)
52
+ return httpx.Response(200, json={"id": "j1", "status": status, "error": "too big" if status == "failed" else None})
53
+ if path.endswith(RESULT_PATH.replace(AT, "j1")):
54
+ return httpx.Response(200, text="# Queued", headers={"content-type": "text/markdown"})
55
+ return httpx.Response(404, json={"detail": path})
56
+
57
+
58
+ class LoaderTest(unittest.TestCase):
59
+ def setUp(self):
60
+ tmp = tempfile.TemporaryDirectory()
61
+ self.addCleanup(tmp.cleanup)
62
+ self.pdf = Path(tmp.name) / "a.pdf"
63
+ self.pdf.write_bytes(b"%PDF-1.7 local")
64
+ env = mock.patch.dict(os.environ, {}, clear=True)
65
+ env.start()
66
+ self.addCleanup(env.stop)
67
+
68
+ def loader(self, api, source=None, **kwargs):
69
+ client = httpx.Client(transport=httpx.MockTransport(api))
70
+ self.addCleanup(client.close)
71
+ return FlatmarkLoader(source or str(self.pdf), client=client, poll_interval=0, **kwargs)
72
+
73
+ def test_direct_convert_is_anonymous_without_a_key(self):
74
+ api = Api()
75
+ [doc] = self.loader(api).load()
76
+ self.assertEqual(doc.page_content, "# Hello")
77
+ self.assertEqual(doc.metadata["source"], str(self.pdf))
78
+ [request] = api.requests
79
+ self.assertEqual(str(request.url), BASE_URL + CONVERT_PATH)
80
+ self.assertNotIn(AUTH_HEADER.lower(), request.headers)
81
+ self.assertIn(b'filename="a.pdf"', request.content)
82
+ self.assertIn(b"Content-Type: application/pdf", request.content)
83
+
84
+ def test_key_from_argument_or_environment(self):
85
+ api = Api()
86
+ self.loader(api, api_key="k1").load()
87
+ os.environ["FLATMARK_API_KEY"] = "k2"
88
+ self.loader(api).load()
89
+ self.assertEqual([r.headers[AUTH_HEADER] for r in api.requests], ["k1", "k2"])
90
+
91
+ def test_url_is_fetched_without_the_key(self):
92
+ api = Api()
93
+ url = "https://files.example.com/papers/b.pdf"
94
+ [doc] = self.loader(api, url, api_key="k1").load()
95
+ fetch, convert = api.requests
96
+ self.assertNotIn(AUTH_HEADER.lower(), fetch.headers)
97
+ self.assertEqual(convert.headers[AUTH_HEADER], "k1")
98
+ self.assertIn(b'filename="b.pdf"', convert.content)
99
+ self.assertEqual(doc.metadata["source"], url)
100
+
101
+ def test_several_sources_one_document_each(self):
102
+ docs = self.loader(Api(), [str(self.pdf), self.pdf]).load()
103
+ self.assertEqual(len(docs), 2)
104
+
105
+ def test_queue_submits_polls_and_downloads(self):
106
+ api = Api()
107
+ [doc] = self.loader(api, api_key="k1", use_queue=True).load()
108
+ self.assertEqual(doc.page_content, "# Queued")
109
+ self.assertEqual(doc.metadata, {"source": str(self.pdf), JOB_ID: "j1"})
110
+ paths = [r.url.path for r in api.requests]
111
+ self.assertEqual(
112
+ paths,
113
+ [SUBMIT_PATH, POLL_PATH.replace(AT, "j1"), POLL_PATH.replace(AT, "j1"), RESULT_PATH.replace(AT, "j1")],
114
+ )
115
+
116
+ def test_failed_job_raises(self):
117
+ with self.assertRaisesRegex(FlatmarkError, "too big"):
118
+ self.loader(Api(statuses=["failed"]), api_key="k1", use_queue=True).load()
119
+
120
+ def test_queue_needs_a_key(self):
121
+ with self.assertRaisesRegex(ValueError, "API key"):
122
+ FlatmarkLoader(str(self.pdf), use_queue=True)
123
+
124
+ def test_api_error_raises_with_status(self):
125
+ with self.assertRaisesRegex(FlatmarkError, "429"):
126
+ self.loader(Api(error=429)).load()
127
+
128
+ def test_async_load(self):
129
+ docs = asyncio.run(self.loader(Api()).aload())
130
+ self.assertEqual(docs[0].page_content, "# Hello")
131
+
132
+
133
+ if __name__ == "__main__":
134
+ unittest.main()