PyPI - docling - Versions diffs - 1.0.2__tar.gz → 1.1.1__tar.gz - Mend

docling 1.0.2tar.gz → 1.1.1tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (25) hide show

{docling-1.0.2 → docling-1.1.1}/PKG-INFO RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.1
 Name: docling
-Version: 1.0.2
+Version: 1.1.1
 Summary: Docling PDF conversion package
 Home-page: https://github.com/DS4SD/docling
 License: MIT
@@ -30,6 +30,7 @@ Requires-Dist: huggingface_hub (>=0.23,<1)
 Requires-Dist: pydantic (>=2.0.0,<3.0.0)
 Requires-Dist: pydantic-settings (>=2.3.0,<3.0.0)
 Requires-Dist: pypdfium2 (>=4.30.0,<5.0.0)
+Requires-Dist: requests (>=2.32.3,<3.0.0)
 Project-URL: Repository, https://github.com/DS4SD/docling
 Description-Content-Type: text/markdown
@@ -65,19 +66,35 @@ To use Docling, simply install `docling` from your package manager, e.g. pip:
 pip install docling
 ```
-> [!NOTE]
+> [!NOTE]
 > Works on macOS and Linux environments. Windows platforms are currently not tested.
 ### Development setup
 To develop for Docling, you need Python 3.10 / 3.11 / 3.12 and Poetry. You can then install from your local clone's root dir:
 ```bash
-poetry install
+poetry install --all-extras
 ```
 ## Usage
-For basic usage, see the [convert.py](https://github.com/DS4SD/docling/blob/main/examples/convert.py) example module. Run with:
+### Convert a single document
+To convert invidual PDF documents, use `convert_single()`, for example:
+```python
+from docling.document_converter import DocumentConverter
+source = "https://arxiv.org/pdf/2206.01062"  # PDF path or URL
+converter = DocumentConverter()
+doc = converter.convert_single(source)
+print(doc.export_to_markdown())  # output: "## DocLayNet: A Large Human-Annotated Dataset for Document-Layout Analysis [...]"
+```
+### Convert a batch of documents
+For an example of converting multiple documents, see [convert.py](https://github.com/DS4SD/docling/blob/main/examples/convert.py).
+From a local repo clone, you can run it with:
 ```
 python examples/convert.py
@@ -93,7 +110,7 @@ You can control if table structure recognition or OCR should be performed by arg
 doc_converter = DocumentConverter(
     artifacts_path=artifacts_path,
     pipeline_options=PipelineOptions(
-        do_table_structure=False,  # controls if table structure is recovered
+        do_table_structure=False,  # controls if table structure is recovered
         do_ocr=True,  # controls if OCR is applied (ignores programmatic content)
     ),
 )
@@ -125,7 +142,7 @@ conv_input = DocumentConversionInput.from_paths(
 )
 ```
-### Convert from binary PDF streams
+### Convert from binary PDF streams
 You can convert PDFs from a binary stream instead of from the filesystem as follows:
 ```python

{docling-1.0.2 → docling-1.1.1}/README.md RENAMED Viewed

@@ -30,19 +30,35 @@ To use Docling, simply install `docling` from your package manager, e.g. pip:
 pip install docling
 ```
-> [!NOTE]
+> [!NOTE]
 > Works on macOS and Linux environments. Windows platforms are currently not tested.
 ### Development setup
 To develop for Docling, you need Python 3.10 / 3.11 / 3.12 and Poetry. You can then install from your local clone's root dir:
 ```bash
-poetry install
+poetry install --all-extras
 ```
 ## Usage
-For basic usage, see the [convert.py](https://github.com/DS4SD/docling/blob/main/examples/convert.py) example module. Run with:
+### Convert a single document
+To convert invidual PDF documents, use `convert_single()`, for example:
+```python
+from docling.document_converter import DocumentConverter
+source = "https://arxiv.org/pdf/2206.01062"  # PDF path or URL
+converter = DocumentConverter()
+doc = converter.convert_single(source)
+print(doc.export_to_markdown())  # output: "## DocLayNet: A Large Human-Annotated Dataset for Document-Layout Analysis [...]"
+```
+### Convert a batch of documents
+For an example of converting multiple documents, see [convert.py](https://github.com/DS4SD/docling/blob/main/examples/convert.py).
+From a local repo clone, you can run it with:
 ```
 python examples/convert.py
@@ -58,7 +74,7 @@ You can control if table structure recognition or OCR should be performed by arg
 doc_converter = DocumentConverter(
     artifacts_path=artifacts_path,
     pipeline_options=PipelineOptions(
-        do_table_structure=False,  # controls if table structure is recovered
+        do_table_structure=False,  # controls if table structure is recovered
         do_ocr=True,  # controls if OCR is applied (ignores programmatic content)
     ),
 )
@@ -90,7 +106,7 @@ conv_input = DocumentConversionInput.from_paths(
 )
 ```
-### Convert from binary PDF streams
+### Convert from binary PDF streams
 You can convert PDFs from a binary stream instead of from the filesystem as follows:
 ```python

{docling-1.0.2 → docling-1.1.1}/docling/backend/pypdfium2_backend.py RENAMED Viewed

@@ -201,13 +201,7 @@ class PyPdfiumPageBackend(PdfPageBackend):
 class PyPdfiumDocumentBackend(PdfDocumentBackend):
     def __init__(self, path_or_stream: Iterable[Union[BytesIO, Path]]):
         super().__init__(path_or_stream)
-        if isinstance(path_or_stream, Path):
-            self._pdoc = pdfium.PdfDocument(path_or_stream)
-        elif isinstance(path_or_stream, BytesIO):
-            self._pdoc = pdfium.PdfDocument(
-                path_or_stream
-            )  # TODO Fix me, won't accept bytes.
+        self._pdoc = pdfium.PdfDocument(path_or_stream)
     def page_count(self) -> int:
         return len(self._pdoc)

{docling-1.0.2 → docling-1.1.1}/docling/document_converter.py RENAMED Viewed

@@ -1,11 +1,15 @@
 import functools
 import logging
+import tempfile
 import time
 import traceback
 from pathlib import Path
 from typing import Iterable, Optional, Type, Union
+import requests
+from docling_core.types import Document
 from PIL import ImageDraw
+from pydantic import AnyHttpUrl, TypeAdapter, ValidationError
 from docling.backend.abstract_backend import PdfDocumentBackend
 from docling.datamodel.base_models import (
@@ -32,6 +36,7 @@ _log = logging.getLogger(__name__)
 class DocumentConverter:
     _layout_model_path = "model_artifacts/layout/beehive_v0.0.5"
     _table_model_path = "model_artifacts/tableformer"
+    _default_download_filename = "file.pdf"
     def __init__(
         self,
@@ -80,6 +85,57 @@ class DocumentConverter:
             # Note: Pdfium backend is not thread-safe, thread pool usage was disabled.
             yield from map(self.process_document, input_batch)
+    def convert_single(self, source: Path | AnyHttpUrl | str) -> Document:
+        """Convert a single document.
+        Args:
+            source (Path | AnyHttpUrl | str): The PDF input source. Can be a path or URL.
+        Raises:
+            ValueError: If source is of unexpected type.
+            RuntimeError: If conversion fails.
+        Returns:
+            Document: The converted document object.
+        """
+        with tempfile.TemporaryDirectory() as temp_dir:
+            try:
+                http_url: AnyHttpUrl = TypeAdapter(AnyHttpUrl).validate_python(source)
+                res = requests.get(http_url, stream=True)
+                res.raise_for_status()
+                fname = None
+                # try to get filename from response header
+                if cont_disp := res.headers.get("Content-Disposition"):
+                    for par in cont_disp.strip().split(";"):
+                        # currently only handling directive "filename" (not "*filename")
+                        if (split := par.split("=")) and split[0].strip() == "filename":
+                            fname = "=".join(split[1:]).strip().strip("'\"") or None
+                            break
+                # otherwise, use name from URL:
+                if fname is None:
+                    fname = Path(http_url.path).name or self._default_download_filename
+                local_path = Path(temp_dir) / fname
+                with open(local_path, "wb") as f:
+                    for chunk in res.iter_content(chunk_size=1024):  # using 1-KB chunks
+                        f.write(chunk)
+            except ValidationError:
+                try:
+                    local_path = TypeAdapter(Path).validate_python(source)
+                except ValidationError:
+                    raise ValueError(
+                        f"Unexpected file path type encountered: {type(source)}"
+                    )
+            conv_inp = DocumentConversionInput.from_paths(paths=[local_path])
+            converted_docs_iter = self.convert(conv_inp)
+            converted_doc: ConvertedDocument = next(converted_docs_iter)
+        if converted_doc.status not in {
+            ConversionStatus.SUCCESS,
+            ConversionStatus.SUCCESS_WITH_ERRORS,
+        }:
+            raise RuntimeError(f"Conversion failed with status: {converted_doc.status}")
+        doc = converted_doc.to_ds_document()
+        return doc
     def process_document(self, in_doc: InputDocument) -> ConvertedDocument:
         start_doc_time = time.time()
         converted_doc = ConvertedDocument(input=in_doc)

{docling-1.0.2 → docling-1.1.1}/docling/models/table_structure_model.py RENAMED Viewed

@@ -114,12 +114,15 @@ class TableStructureModel:
                     for element in table_out["tf_responses"]:
                         if not self.do_cell_matching:
-                            the_bbox = BoundingBox.model_validate(element["bbox"])
+                            the_bbox = BoundingBox.model_validate(
+                                element["bbox"]
+                            ).scaled(1 / self.scale)
                             text_piece = page._backend.get_text_in_rect(the_bbox)
                             element["bbox"]["token"] = text_piece
                         tc = TableCell.model_validate(element)
-                        tc.bbox = tc.bbox.scaled(1 / self.scale)
+                        if self.do_cell_matching:
+                            tc.bbox = tc.bbox.scaled(1 / self.scale)
                         table_cells.append(tc)
                     # Retrieving cols/rows, after post processing:

{docling-1.0.2 → docling-1.1.1}/pyproject.toml RENAMED Viewed

@@ -1,6 +1,6 @@
 [tool.poetry]
 name = "docling"
-version = "1.0.2"  # DO NOT EDIT, updated automatically
+version = "1.1.1"  # DO NOT EDIT, updated automatically
 description = "Docling PDF conversion package"
 authors = ["Christoph Auer <cau@zurich.ibm.com>", "Michele Dolfi <dol@zurich.ibm.com>", "Maxim Lysak <mly@zurich.ibm.com>", "Nikos Livathinos <nli@zurich.ibm.com>", "Ahmed Nassar <ahn@zurich.ibm.com>", "Peter Staar <taa@zurich.ibm.com>"]
 license = "MIT"
@@ -30,6 +30,7 @@ filetype = "^1.2.0"
 pypdfium2 = "^4.30.0"
 pydantic-settings = "^2.3.0"
 huggingface_hub = ">=0.23,<1"
+requests = "^2.32.3"
 easyocr = { version = "^1.7", optional = true }
 [tool.poetry.group.dev.dependencies]