langchain-flatmark 1.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langchain_flatmark-1.0.1/PKG-INFO +55 -0
- langchain_flatmark-1.0.1/README.md +36 -0
- langchain_flatmark-1.0.1/langchain_flatmark/__init__.py +5 -0
- langchain_flatmark-1.0.1/langchain_flatmark/document_loaders.py +163 -0
- langchain_flatmark-1.0.1/langchain_flatmark/py.typed +0 -0
- langchain_flatmark-1.0.1/pyproject.toml +28 -0
- langchain_flatmark-1.0.1/tests/test_document_loaders.py +134 -0
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: langchain-flatmark
|
|
3
|
+
Version: 1.0.1
|
|
4
|
+
Summary: Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
|
|
5
|
+
Project-URL: Homepage, https://flatmark.dev
|
|
6
|
+
Project-URL: Documentation, https://github.com/flatmark-dev/flatmark-integrations/blob/main/langchain/README.md
|
|
7
|
+
Project-URL: Repository, https://github.com/flatmark-dev/flatmark-integrations
|
|
8
|
+
Project-URL: Issues, https://github.com/flatmark-dev/flatmark-integrations/issues
|
|
9
|
+
Author-email: podshalocef <contact@podshalocef.com>
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
Keywords: document-loader,flatmark,langchain,markdown
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.10
|
|
16
|
+
Requires-Dist: httpx<1,>=0.28
|
|
17
|
+
Requires-Dist: langchain-core<2,>=1.6
|
|
18
|
+
Description-Content-Type: text/markdown
|
|
19
|
+
|
|
20
|
+
# langchain-flatmark
|
|
21
|
+
|
|
22
|
+
Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
|
|
23
|
+
|
|
24
|
+
A LangChain document loader for the [flatmark API](https://flatmark.dev): `FlatmarkLoader` turns local files and URLs into Markdown `Document`s.
|
|
25
|
+
|
|
26
|
+
```sh
|
|
27
|
+
pip install langchain-flatmark
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from langchain_flatmark import FlatmarkLoader
|
|
32
|
+
|
|
33
|
+
docs = FlatmarkLoader("report.pdf").load()
|
|
34
|
+
print(docs[0].page_content)
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
Direct conversion (`POST /v1/convert`) works without a key at a lower rate limit. [Get an API key](https://flatmark.dev/go/langchain?to=/app/api-keys) and pass it as `api_key=` or set `FLATMARK_API_KEY`.
|
|
38
|
+
|
|
39
|
+
For large or scanned files, convert through the queue (an API key is required):
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
loader = FlatmarkLoader(["scan.pdf", "https://example.com/deck.pptx"], use_queue=True)
|
|
43
|
+
for doc in loader.lazy_load():
|
|
44
|
+
print(doc.metadata, len(doc.page_content))
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
The loader submits `POST /v1/convert/jobs`, polls `GET /v1/jobs/{job_id}` until the job's status is `succeeded` or `failed`, then downloads `GET /v1/convert/jobs/{job_id}/result`.
|
|
48
|
+
|
|
49
|
+
## Documents
|
|
50
|
+
|
|
51
|
+
One `Document` per source; `page_content` is the Markdown. `metadata` carries `source` (the path or URL as given) plus the `meta` fields of the answer — and `job_id` for a queued conversion.
|
|
52
|
+
|
|
53
|
+
A URL is downloaded by the loader without your key, then uploaded. Other options: `base_url`, `poll_interval`, `timeout`, and `client` (an `httpx.Client` for proxies or retries).
|
|
54
|
+
|
|
55
|
+
API reference: https://flatmark.dev/docs · Support: https://flatmark.dev/support · Generated from [`openapi.json`](../openapi.json).
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# langchain-flatmark
|
|
2
|
+
|
|
3
|
+
Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany.
|
|
4
|
+
|
|
5
|
+
A LangChain document loader for the [flatmark API](https://flatmark.dev): `FlatmarkLoader` turns local files and URLs into Markdown `Document`s.
|
|
6
|
+
|
|
7
|
+
```sh
|
|
8
|
+
pip install langchain-flatmark
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
```python
|
|
12
|
+
from langchain_flatmark import FlatmarkLoader
|
|
13
|
+
|
|
14
|
+
docs = FlatmarkLoader("report.pdf").load()
|
|
15
|
+
print(docs[0].page_content)
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
Direct conversion (`POST /v1/convert`) works without a key at a lower rate limit. [Get an API key](https://flatmark.dev/go/langchain?to=/app/api-keys) and pass it as `api_key=` or set `FLATMARK_API_KEY`.
|
|
19
|
+
|
|
20
|
+
For large or scanned files, convert through the queue (an API key is required):
|
|
21
|
+
|
|
22
|
+
```python
|
|
23
|
+
loader = FlatmarkLoader(["scan.pdf", "https://example.com/deck.pptx"], use_queue=True)
|
|
24
|
+
for doc in loader.lazy_load():
|
|
25
|
+
print(doc.metadata, len(doc.page_content))
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
The loader submits `POST /v1/convert/jobs`, polls `GET /v1/jobs/{job_id}` until the job's status is `succeeded` or `failed`, then downloads `GET /v1/convert/jobs/{job_id}/result`.
|
|
29
|
+
|
|
30
|
+
## Documents
|
|
31
|
+
|
|
32
|
+
One `Document` per source; `page_content` is the Markdown. `metadata` carries `source` (the path or URL as given) plus the `meta` fields of the answer — and `job_id` for a queued conversion.
|
|
33
|
+
|
|
34
|
+
A URL is downloaded by the loader without your key, then uploaded. Other options: `base_url`, `poll_interval`, `timeout`, and `client` (an `httpx.Client` for proxies or retries).
|
|
35
|
+
|
|
36
|
+
API reference: https://flatmark.dev/docs · Support: https://flatmark.dev/support · Generated from [`openapi.json`](../openapi.json).
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
"""flatmark document loader for LangChain.
|
|
2
|
+
|
|
3
|
+
Generated by listing-sync from the flatmark OpenAPI document; changes here are
|
|
4
|
+
overwritten.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import mimetypes
|
|
10
|
+
import os
|
|
11
|
+
import time
|
|
12
|
+
from collections.abc import Iterable, Iterator
|
|
13
|
+
from pathlib import Path, PurePosixPath
|
|
14
|
+
from urllib.parse import quote, urlsplit
|
|
15
|
+
|
|
16
|
+
import httpx
|
|
17
|
+
from langchain_core.document_loaders import BaseLoader
|
|
18
|
+
from langchain_core.documents import Document
|
|
19
|
+
|
|
20
|
+
BASE_URL = "https://api.flatmark.dev"
|
|
21
|
+
API_KEY_ENV = "FLATMARK_API_KEY"
|
|
22
|
+
KEY_URL = "https://flatmark.dev/go/langchain?to=/app/api-keys"
|
|
23
|
+
AUTH_HEADER = "X-API-Key"
|
|
24
|
+
USER_AGENT = "langchain-flatmark"
|
|
25
|
+
CONVERT_PATH = "/v1/convert"
|
|
26
|
+
CONVERT_FILE = "file"
|
|
27
|
+
TEXT_FIELD = "markdown"
|
|
28
|
+
META_FIELD = "meta"
|
|
29
|
+
SUBMIT_PATH = "/v1/convert/jobs"
|
|
30
|
+
SUBMIT_FILE = "file"
|
|
31
|
+
POLL_PATH = "/v1/jobs/{job_id}"
|
|
32
|
+
RESULT_PATH = "/v1/convert/jobs/{job_id}/result"
|
|
33
|
+
JOB_ID = "job_id"
|
|
34
|
+
DONE = ('succeeded', 'failed')
|
|
35
|
+
# Standard types older Pythons' mimetypes table lacks (added in 3.13).
|
|
36
|
+
OFFICE_TYPES = {
|
|
37
|
+
".docx": "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
|
38
|
+
".pptx": "application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
|
39
|
+
".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
class FlatmarkError(RuntimeError):
|
|
44
|
+
"""The flatmark API answered with an error, or a queued job failed."""
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _media_type(name: str) -> str:
|
|
48
|
+
guess = mimetypes.guess_type(name)[0]
|
|
49
|
+
return guess or OFFICE_TYPES.get(Path(name).suffix.lower(), "application/octet-stream")
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class FlatmarkLoader(BaseLoader):
|
|
53
|
+
"""Load files and URLs as Markdown `Document`s through the flatmark API.
|
|
54
|
+
|
|
55
|
+
Each source becomes one `Document` whose `page_content` is the Markdown and
|
|
56
|
+
whose metadata carries `source` (the path or URL as given).
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
file_path: A local path or an http(s) URL, or an iterable of them.
|
|
60
|
+
api_key: The flatmark API key; defaults to the `FLATMARK_API_KEY`
|
|
61
|
+
environment variable. Optional for direct conversion (anonymous
|
|
62
|
+
calls run at a lower rate limit), required with `use_queue=True`.
|
|
63
|
+
use_queue: Convert through the queue (`POST /v1/convert/jobs`) instead of
|
|
64
|
+
the direct call (`POST /v1/convert`): the loader submits a job, polls
|
|
65
|
+
it every `poll_interval` seconds until it finishes or `timeout`
|
|
66
|
+
seconds pass, and downloads its Markdown.
|
|
67
|
+
base_url: The API base URL.
|
|
68
|
+
poll_interval: Seconds between job polls.
|
|
69
|
+
timeout: Seconds a queued job may take before the loader gives up.
|
|
70
|
+
client: An `httpx.Client` to send every request with (for proxies,
|
|
71
|
+
retries or tests); the loader makes and closes its own otherwise.
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
def __init__(
|
|
75
|
+
self,
|
|
76
|
+
file_path: str | os.PathLike | Iterable[str | os.PathLike],
|
|
77
|
+
*,
|
|
78
|
+
api_key: str | None = None,
|
|
79
|
+
use_queue: bool = False,
|
|
80
|
+
base_url: str = BASE_URL,
|
|
81
|
+
poll_interval: float = 2.0,
|
|
82
|
+
timeout: float = 600.0,
|
|
83
|
+
client: httpx.Client | None = None,
|
|
84
|
+
) -> None:
|
|
85
|
+
if isinstance(file_path, (str, os.PathLike)):
|
|
86
|
+
file_path = [file_path]
|
|
87
|
+
self.file_paths = [os.fspath(p) for p in file_path]
|
|
88
|
+
self.api_key = api_key or os.environ.get(API_KEY_ENV) or None
|
|
89
|
+
if use_queue and not self.api_key:
|
|
90
|
+
raise ValueError(
|
|
91
|
+
f"use_queue=True needs an API key: pass api_key= or set "
|
|
92
|
+
f"{API_KEY_ENV}. Get one at {KEY_URL}"
|
|
93
|
+
)
|
|
94
|
+
self.use_queue = use_queue
|
|
95
|
+
self.base_url = base_url.rstrip("/")
|
|
96
|
+
self.poll_interval = poll_interval
|
|
97
|
+
self.timeout = timeout
|
|
98
|
+
self.client = client
|
|
99
|
+
|
|
100
|
+
def lazy_load(self) -> Iterator[Document]:
|
|
101
|
+
"""Convert each source in turn, yielding one `Document` per source."""
|
|
102
|
+
client = self.client or httpx.Client(timeout=120.0, follow_redirects=True)
|
|
103
|
+
try:
|
|
104
|
+
for source in self.file_paths:
|
|
105
|
+
name, data, media = self._read(client, source)
|
|
106
|
+
if self.use_queue:
|
|
107
|
+
yield self._queued(client, source, name, data, media)
|
|
108
|
+
else:
|
|
109
|
+
yield self._direct(client, source, name, data, media)
|
|
110
|
+
finally:
|
|
111
|
+
if self.client is None:
|
|
112
|
+
client.close()
|
|
113
|
+
|
|
114
|
+
def _read(self, client: httpx.Client, source: str) -> tuple[str, bytes, str]:
|
|
115
|
+
"""(file name, bytes, media type) of a path or URL. A URL is fetched
|
|
116
|
+
without the API key: it goes to someone else's server."""
|
|
117
|
+
if source.startswith(("http://", "https://")):
|
|
118
|
+
response = client.get(source, headers={"User-Agent": USER_AGENT})
|
|
119
|
+
response.raise_for_status()
|
|
120
|
+
name = PurePosixPath(urlsplit(source).path).name or "document"
|
|
121
|
+
media = response.headers.get("content-type", "").split(";")[0].strip()
|
|
122
|
+
if not media or media == "application/octet-stream":
|
|
123
|
+
media = _media_type(name)
|
|
124
|
+
return name, response.content, media
|
|
125
|
+
path = Path(source)
|
|
126
|
+
return path.name, path.read_bytes(), _media_type(path.name)
|
|
127
|
+
|
|
128
|
+
def _call(self, client: httpx.Client, method: str, path: str, **kwargs) -> httpx.Response:
|
|
129
|
+
headers = {"User-Agent": USER_AGENT}
|
|
130
|
+
if self.api_key:
|
|
131
|
+
headers[AUTH_HEADER] = self.api_key
|
|
132
|
+
response = client.request(method, self.base_url + path, headers=headers, **kwargs)
|
|
133
|
+
if response.is_error:
|
|
134
|
+
message = f"{method} {path} answered {response.status_code}: {response.text[:500]}"
|
|
135
|
+
if response.status_code == 429 and not self.api_key:
|
|
136
|
+
message += f" (an API key lifts the limit: {KEY_URL})"
|
|
137
|
+
raise FlatmarkError(message)
|
|
138
|
+
return response
|
|
139
|
+
|
|
140
|
+
def _direct(self, client, source, name, data, media) -> Document:
|
|
141
|
+
answer = self._call(
|
|
142
|
+
client, "POST", CONVERT_PATH, files={CONVERT_FILE: (name, data, media)}
|
|
143
|
+
).json()
|
|
144
|
+
meta = (answer.get(META_FIELD) or {}) if META_FIELD else {}
|
|
145
|
+
return Document(page_content=answer[TEXT_FIELD], metadata={**meta, "source": source})
|
|
146
|
+
|
|
147
|
+
def _queued(self, client, source, name, data, media) -> Document:
|
|
148
|
+
job_id = self._call(
|
|
149
|
+
client, "POST", SUBMIT_PATH, files={SUBMIT_FILE: (name, data, media)}
|
|
150
|
+
).json()[JOB_ID]
|
|
151
|
+
at = "{" + JOB_ID + "}"
|
|
152
|
+
deadline = time.monotonic() + self.timeout
|
|
153
|
+
while True:
|
|
154
|
+
job = self._call(client, "GET", POLL_PATH.replace(at, quote(job_id, safe=""))).json()
|
|
155
|
+
if job.get("status") in DONE:
|
|
156
|
+
break
|
|
157
|
+
if time.monotonic() >= deadline:
|
|
158
|
+
raise FlatmarkError(f"job {job_id} not finished after {self.timeout:g} s")
|
|
159
|
+
time.sleep(self.poll_interval)
|
|
160
|
+
if job["status"] != DONE[0]:
|
|
161
|
+
raise FlatmarkError(f"job {job_id} {job['status']}: {job.get('error') or 'no reason given'}")
|
|
162
|
+
text = self._call(client, "GET", RESULT_PATH.replace(at, quote(job_id, safe=""))).text
|
|
163
|
+
return Document(page_content=text, metadata={"source": source, JOB_ID: job_id})
|
|
File without changes
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "langchain-flatmark"
|
|
7
|
+
version = "1.0.1"
|
|
8
|
+
description = "Document to Markdown API and MCP server for PDF, Word, PowerPoint, Excel and HTML. OCR queue for large files. Hosted in Germany."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.10"
|
|
12
|
+
authors = [{ name = "podshalocef", email = "contact@podshalocef.com" }]
|
|
13
|
+
keywords = ["langchain", "document-loader", "markdown", "flatmark"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Intended Audience :: Developers",
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
18
|
+
]
|
|
19
|
+
dependencies = ["langchain-core>=1.6,<2", "httpx>=0.28,<1"]
|
|
20
|
+
|
|
21
|
+
[project.urls]
|
|
22
|
+
Homepage = "https://flatmark.dev"
|
|
23
|
+
Documentation = "https://github.com/flatmark-dev/flatmark-integrations/blob/main/langchain/README.md"
|
|
24
|
+
Repository = "https://github.com/flatmark-dev/flatmark-integrations"
|
|
25
|
+
Issues = "https://github.com/flatmark-dev/flatmark-integrations/issues"
|
|
26
|
+
|
|
27
|
+
[tool.hatch.build.targets.wheel]
|
|
28
|
+
packages = ["langchain_flatmark"]
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
"""Tests for langchain_flatmark, against an httpx.MockTransport (no network)."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import tempfile
|
|
7
|
+
import unittest
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from unittest import mock
|
|
10
|
+
|
|
11
|
+
import httpx
|
|
12
|
+
|
|
13
|
+
from langchain_flatmark import FlatmarkLoader, FlatmarkError
|
|
14
|
+
from langchain_flatmark.document_loaders import (
|
|
15
|
+
AUTH_HEADER,
|
|
16
|
+
BASE_URL,
|
|
17
|
+
CONVERT_PATH,
|
|
18
|
+
JOB_ID,
|
|
19
|
+
META_FIELD,
|
|
20
|
+
POLL_PATH,
|
|
21
|
+
RESULT_PATH,
|
|
22
|
+
SUBMIT_PATH,
|
|
23
|
+
TEXT_FIELD,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
AT = "{" + JOB_ID + "}"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class Api:
|
|
30
|
+
"""A fake flatmark API: records requests, answers by path."""
|
|
31
|
+
|
|
32
|
+
def __init__(self, statuses=("running", "succeeded"), error=None):
|
|
33
|
+
self.requests, self.statuses, self.error = [], list(statuses), error
|
|
34
|
+
|
|
35
|
+
def __call__(self, request):
|
|
36
|
+
request.read()
|
|
37
|
+
self.requests.append(request)
|
|
38
|
+
path = request.url.path
|
|
39
|
+
if request.url.host == "files.example.com":
|
|
40
|
+
return httpx.Response(200, content=b"%PDF-1.7 remote", headers={"content-type": "application/pdf"})
|
|
41
|
+
if self.error:
|
|
42
|
+
return httpx.Response(self.error, json={"detail": "nope"})
|
|
43
|
+
if path.endswith(CONVERT_PATH):
|
|
44
|
+
answer = {TEXT_FIELD: "# Hello"}
|
|
45
|
+
if META_FIELD:
|
|
46
|
+
answer[META_FIELD] = {"filename": "a.pdf", "title": "Hello"}
|
|
47
|
+
return httpx.Response(200, json=answer)
|
|
48
|
+
if path.endswith(SUBMIT_PATH):
|
|
49
|
+
return httpx.Response(202, json={JOB_ID: "j1", "status": "queued"})
|
|
50
|
+
if path.endswith(POLL_PATH.replace(AT, "j1")):
|
|
51
|
+
status = self.statuses.pop(0)
|
|
52
|
+
return httpx.Response(200, json={"id": "j1", "status": status, "error": "too big" if status == "failed" else None})
|
|
53
|
+
if path.endswith(RESULT_PATH.replace(AT, "j1")):
|
|
54
|
+
return httpx.Response(200, text="# Queued", headers={"content-type": "text/markdown"})
|
|
55
|
+
return httpx.Response(404, json={"detail": path})
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class LoaderTest(unittest.TestCase):
|
|
59
|
+
def setUp(self):
|
|
60
|
+
tmp = tempfile.TemporaryDirectory()
|
|
61
|
+
self.addCleanup(tmp.cleanup)
|
|
62
|
+
self.pdf = Path(tmp.name) / "a.pdf"
|
|
63
|
+
self.pdf.write_bytes(b"%PDF-1.7 local")
|
|
64
|
+
env = mock.patch.dict(os.environ, {}, clear=True)
|
|
65
|
+
env.start()
|
|
66
|
+
self.addCleanup(env.stop)
|
|
67
|
+
|
|
68
|
+
def loader(self, api, source=None, **kwargs):
|
|
69
|
+
client = httpx.Client(transport=httpx.MockTransport(api))
|
|
70
|
+
self.addCleanup(client.close)
|
|
71
|
+
return FlatmarkLoader(source or str(self.pdf), client=client, poll_interval=0, **kwargs)
|
|
72
|
+
|
|
73
|
+
def test_direct_convert_is_anonymous_without_a_key(self):
|
|
74
|
+
api = Api()
|
|
75
|
+
[doc] = self.loader(api).load()
|
|
76
|
+
self.assertEqual(doc.page_content, "# Hello")
|
|
77
|
+
self.assertEqual(doc.metadata["source"], str(self.pdf))
|
|
78
|
+
[request] = api.requests
|
|
79
|
+
self.assertEqual(str(request.url), BASE_URL + CONVERT_PATH)
|
|
80
|
+
self.assertNotIn(AUTH_HEADER.lower(), request.headers)
|
|
81
|
+
self.assertIn(b'filename="a.pdf"', request.content)
|
|
82
|
+
self.assertIn(b"Content-Type: application/pdf", request.content)
|
|
83
|
+
|
|
84
|
+
def test_key_from_argument_or_environment(self):
|
|
85
|
+
api = Api()
|
|
86
|
+
self.loader(api, api_key="k1").load()
|
|
87
|
+
os.environ["FLATMARK_API_KEY"] = "k2"
|
|
88
|
+
self.loader(api).load()
|
|
89
|
+
self.assertEqual([r.headers[AUTH_HEADER] for r in api.requests], ["k1", "k2"])
|
|
90
|
+
|
|
91
|
+
def test_url_is_fetched_without_the_key(self):
|
|
92
|
+
api = Api()
|
|
93
|
+
url = "https://files.example.com/papers/b.pdf"
|
|
94
|
+
[doc] = self.loader(api, url, api_key="k1").load()
|
|
95
|
+
fetch, convert = api.requests
|
|
96
|
+
self.assertNotIn(AUTH_HEADER.lower(), fetch.headers)
|
|
97
|
+
self.assertEqual(convert.headers[AUTH_HEADER], "k1")
|
|
98
|
+
self.assertIn(b'filename="b.pdf"', convert.content)
|
|
99
|
+
self.assertEqual(doc.metadata["source"], url)
|
|
100
|
+
|
|
101
|
+
def test_several_sources_one_document_each(self):
|
|
102
|
+
docs = self.loader(Api(), [str(self.pdf), self.pdf]).load()
|
|
103
|
+
self.assertEqual(len(docs), 2)
|
|
104
|
+
|
|
105
|
+
def test_queue_submits_polls_and_downloads(self):
|
|
106
|
+
api = Api()
|
|
107
|
+
[doc] = self.loader(api, api_key="k1", use_queue=True).load()
|
|
108
|
+
self.assertEqual(doc.page_content, "# Queued")
|
|
109
|
+
self.assertEqual(doc.metadata, {"source": str(self.pdf), JOB_ID: "j1"})
|
|
110
|
+
paths = [r.url.path for r in api.requests]
|
|
111
|
+
self.assertEqual(
|
|
112
|
+
paths,
|
|
113
|
+
[SUBMIT_PATH, POLL_PATH.replace(AT, "j1"), POLL_PATH.replace(AT, "j1"), RESULT_PATH.replace(AT, "j1")],
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
def test_failed_job_raises(self):
|
|
117
|
+
with self.assertRaisesRegex(FlatmarkError, "too big"):
|
|
118
|
+
self.loader(Api(statuses=["failed"]), api_key="k1", use_queue=True).load()
|
|
119
|
+
|
|
120
|
+
def test_queue_needs_a_key(self):
|
|
121
|
+
with self.assertRaisesRegex(ValueError, "API key"):
|
|
122
|
+
FlatmarkLoader(str(self.pdf), use_queue=True)
|
|
123
|
+
|
|
124
|
+
def test_api_error_raises_with_status(self):
|
|
125
|
+
with self.assertRaisesRegex(FlatmarkError, "429"):
|
|
126
|
+
self.loader(Api(error=429)).load()
|
|
127
|
+
|
|
128
|
+
def test_async_load(self):
|
|
129
|
+
docs = asyncio.run(self.loader(Api()).aload())
|
|
130
|
+
self.assertEqual(docs[0].page_content, "# Hello")
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
if __name__ == "__main__":
|
|
134
|
+
unittest.main()
|