table-stitcher 0.4.2__tar.gz → 0.4.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/CHANGELOG.md +33 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/PKG-INFO +42 -32
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/README.md +40 -30
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/pyproject.toml +1 -1
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/README.md +3 -3
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/docling.py +51 -5
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/merger.py +23 -1
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/conftest.py +9 -1
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/15-page-druglist.corp.expected.yaml +3 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/covid-misc-labs-4pg.pt2.expected.yaml +3 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/retirement-portfolio.corp.expected.yaml +3 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_docling_adapter.py +71 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_merger.py +44 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.claude/settings.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/ISSUE_TEMPLATE/config.yml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/dependabot.yml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/pull_request_template.md +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/workflows/ci.yml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/workflows/release.yml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/workflows/upstream-smoke.yml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.gitignore +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.pre-commit-config.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/CONTRIBUTING.md +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/LICENSE +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/SECURITY.md +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/examples/basic_pipeline.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/examples/system_controller.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/scripts/regenerate_docling_snapshots.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/scripts/release_gate.sh +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/__init__.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/__init__.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/base.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/models.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/py.typed +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/README.md +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/__init__.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/fixtures/tablemeta/headerless-width-drift.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/__init__.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/_tools/__init__.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/_tools/regenerate_expected.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/_synth/__init__.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/_synth/generate.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/.gitkeep +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/kaop-study-mixed-3pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/kaop-study-mixed-3pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/kaop-study-mixed-3pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/lab-panels-3pg.corp.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/lab-panels-3pg.corp.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/lab-panels-3pg.corp.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/false-merge/category-rows-thematic-2pg.corp.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/false-merge/category-rows-thematic-2pg.corp.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/false-merge/category-rows-thematic-2pg.corp.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/15-page-druglist.corp.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/15-page-druglist.corp.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/gene-symbols-6pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/gene-symbols-6pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/gene-symbols-6pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rct-study-table-5pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rct-study-table-5pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rct-study-table-5pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rrt-outcomes-2pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rrt-outcomes-2pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rrt-outcomes-2pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/uveitis-case-series-5pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/uveitis-case-series-5pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/uveitis-case-series-5pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/covid-misc-labs-4pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/covid-misc-labs-4pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/lit-review-3pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/lit-review-3pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/lit-review-3pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/retirement-portfolio.corp.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/retirement-portfolio.corp.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/.gitkeep +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/biological-process-2pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/biological-process-2pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/biological-process-2pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/symptoms-mediators-4pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/symptoms-mediators-4pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/symptoms-mediators-4pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/corporate-history-2pg.edinet.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/corporate-history-2pg.edinet.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/corporate-history-2pg.edinet.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/subsidiaries-4pg.edinet.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/subsidiaries-4pg.edinet.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/subsidiaries-4pg.edinet.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/.gitkeep +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/varicose-veins-new-table-header-7pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/varicose-veins-new-table-header-7pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/varicose-veins-new-table-header-7pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/page-gap-too-large/unrelated-tables-gap4.synth.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/page-gap-too-large/unrelated-tables-gap4.synth.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/page-gap-too-large/unrelated-tables-gap4.synth.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/4-page-substance-list.corp.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/4-page-substance-list.corp.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/4-page-substance-list.corp.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/cell-markers-4pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/cell-markers-4pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/cell-markers-4pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/challenge-categories-2pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/challenge-categories-2pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/challenge-categories-2pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/fungal-taxonomy-4pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/fungal-taxonomy-4pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/fungal-taxonomy-4pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/rowspan-insurance-payout.corp.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/rowspan-insurance-payout.corp.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/rowspan-insurance-payout.corp.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/search-strategies-2pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/search-strategies-2pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/search-strategies-2pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/study-sample-7pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/study-sample-7pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/study-sample-7pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/themes-exemplars-3pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/themes-exemplars-3pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/themes-exemplars-3pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/simple-continuation/sample-table.corp.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/simple-continuation/sample-table.corp.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/simple-continuation/sample-table.corp.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/spillover/note-overflow.synth.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/spillover/note-overflow.synth.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/spillover/note-overflow.synth.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/.gitkeep +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/abx-literature-review-7pg.pt2.docling.json +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/abx-literature-review-7pg.pt2.expected.yaml +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/abx-literature-review-7pg.pt2.pdf +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/test_fixtures.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_intervening_content_guard.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_public_api.py +0 -0
- {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_tablemeta_fixtures.py +0 -0
|
@@ -7,6 +7,39 @@ the project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [0.4.4] — 2026-08-13
|
|
11
|
+
|
|
12
|
+
### Fixed
|
|
13
|
+
|
|
14
|
+
- **Duplicate header labels crashed the multipage merge** (`merger.py`).
|
|
15
|
+
Extracted headers that repeat a label — e.g. SEC 13F voting-authority
|
|
16
|
+
triplets where TableFormer emits `COLUMN 8` three times — made
|
|
17
|
+
`_build_generic_merged_table` raise `ValueError: Reindexing only valid
|
|
18
|
+
with uniquely valued Index objects`, because `pd.concat` cannot align
|
|
19
|
+
fragments on a non-unique column Index. Downstream consumers that
|
|
20
|
+
fail-soft on the error silently kept a degraded table set for the whole
|
|
21
|
+
document. Fragments are now merged under positionally deduped labels
|
|
22
|
+
(collision-safe against pre-existing `X.1`-style names) and the original
|
|
23
|
+
duplicated labels are restored on the merged output, matching how
|
|
24
|
+
single-fragment tables pass through untouched.
|
|
25
|
+
|
|
26
|
+
## [0.4.3] — 2026-06-11
|
|
27
|
+
|
|
28
|
+
### Fixed
|
|
29
|
+
|
|
30
|
+
- **Reprinted continuation-page headers appended as data rows on multi-page
|
|
31
|
+
merge** (`adapters/docling.py`). When a table's column header is reprinted at
|
|
32
|
+
the top of each page — especially a multi-row (hierarchical) header — the
|
|
33
|
+
repeated header rows survived the merge as bogus data rows, misaligning the
|
|
34
|
+
stitched table. Injection now drops a body row when it is *both* flagged
|
|
35
|
+
`column_header` by Docling *and* a tokenized match (Jaccard ≥ 0.6) for the
|
|
36
|
+
reconstructed header block. Both signals are required: the flag alone is
|
|
37
|
+
unreliable (Docling over-flags rowspan/continuation *data* rows as headers),
|
|
38
|
+
and the tokenized comparison is punctuation-agnostic, so per-cell OCR drift
|
|
39
|
+
such as `(S$)` vs `($$)` is tolerated without any threshold tuning. The merged
|
|
40
|
+
DataFrame (`lt.df`) is unchanged; only the injected document is de-duplicated.
|
|
41
|
+
A `debug` log reports each dropped row.
|
|
42
|
+
|
|
10
43
|
## [0.4.2] — 2026-06-08
|
|
11
44
|
|
|
12
45
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
2
|
Name: table-stitcher
|
|
3
|
-
Version: 0.4.
|
|
3
|
+
Version: 0.4.4
|
|
4
4
|
Summary: Reassemble tables split across page boundaries in PDF extraction
|
|
5
5
|
Project-URL: Homepage, https://github.com/pebbleroad/table-stitcher
|
|
6
6
|
Project-URL: Repository, https://github.com/pebbleroad/table-stitcher
|
|
@@ -103,8 +103,8 @@ from table_stitcher import stitch_tables
|
|
|
103
103
|
|
|
104
104
|
converter = DocumentConverter()
|
|
105
105
|
doc = converter.convert("report.pdf").document
|
|
106
|
-
doc = stitch_tables(doc)
|
|
107
|
-
|
|
106
|
+
doc = stitch_tables(doc) # merged tables; ready for
|
|
107
|
+
# export_to_markdown() / HTML / LLM
|
|
108
108
|
```
|
|
109
109
|
|
|
110
110
|
`stitch_tables()` mutates `doc` in place and returns the same object. If you
|
|
@@ -132,10 +132,10 @@ Runnable end-to-end scripts live in [`examples/`](examples/):
|
|
|
132
132
|
from table_stitcher import stitch_tables, MultiPageConfig
|
|
133
133
|
|
|
134
134
|
config = MultiPageConfig(
|
|
135
|
-
max_page_gap=1,
|
|
136
|
-
max_width_difference=2,
|
|
137
|
-
header_sim_strict=0.6,
|
|
138
|
-
stitch_separator="\n",
|
|
135
|
+
max_page_gap=1, # Only merge tables on consecutive pages
|
|
136
|
+
max_width_difference=2, # Column count tolerance
|
|
137
|
+
header_sim_strict=0.6, # Threshold for repeated header detection
|
|
138
|
+
stitch_separator="\n", # Join character for split content
|
|
139
139
|
)
|
|
140
140
|
|
|
141
141
|
doc = stitch_tables(doc, config=config)
|
|
@@ -148,6 +148,7 @@ from typing import Any, List
|
|
|
148
148
|
from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
|
|
149
149
|
from table_stitcher.adapters.base import TableStitcherAdapter
|
|
150
150
|
|
|
151
|
+
|
|
151
152
|
class MyParserAdapter:
|
|
152
153
|
def extract(self, doc, cfg: MultiPageConfig) -> List[TableMeta]:
|
|
153
154
|
"""Read tables from your document format into TableMeta objects."""
|
|
@@ -157,6 +158,7 @@ class MyParserAdapter:
|
|
|
157
158
|
"""Write merged results back into your document format."""
|
|
158
159
|
...
|
|
159
160
|
|
|
161
|
+
|
|
160
162
|
stitcher = TableStitcher(adapter=MyParserAdapter())
|
|
161
163
|
doc = stitcher.stitch(doc)
|
|
162
164
|
```
|
|
@@ -258,7 +260,13 @@ from typing import Any, List
|
|
|
258
260
|
import pandas as pd
|
|
259
261
|
from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
|
|
260
262
|
from table_stitcher.adapters.base import TableStitcherAdapter
|
|
261
|
-
from table_stitcher.merger import
|
|
263
|
+
from table_stitcher.merger import (
|
|
264
|
+
tokenize,
|
|
265
|
+
normalize_col_name,
|
|
266
|
+
is_numeric_like_colnames,
|
|
267
|
+
first_row_has_number,
|
|
268
|
+
)
|
|
269
|
+
|
|
262
270
|
|
|
263
271
|
class MyParserAdapter:
|
|
264
272
|
def extract(self, doc: Any, cfg: MultiPageConfig) -> List[TableMeta]:
|
|
@@ -281,32 +289,32 @@ class MyParserAdapter:
|
|
|
281
289
|
# 4. Tokenize first row (fallback similarity signal)
|
|
282
290
|
first_row_tokens = set()
|
|
283
291
|
if df.shape[0] > 0:
|
|
284
|
-
first_row_tokens = tokenize(
|
|
285
|
-
" ".join(str(x) for x in df.iloc[0].tolist())
|
|
286
|
-
)
|
|
292
|
+
first_row_tokens = tokenize(" ".join(str(x) for x in df.iloc[0].tolist()))
|
|
287
293
|
|
|
288
294
|
# 5. Classify: is_headerless, is_header_orphan, is_data_orphan
|
|
289
295
|
raw_columns = [str(c) for c in df.columns]
|
|
290
|
-
is_headerless = df.attrs.get(
|
|
291
|
-
|
|
292
|
-
tables_meta.append(
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
|
|
309
|
-
|
|
296
|
+
is_headerless = df.attrs.get("is_headerless", False)
|
|
297
|
+
|
|
298
|
+
tables_meta.append(
|
|
299
|
+
TableMeta(
|
|
300
|
+
idx=idx,
|
|
301
|
+
df=df,
|
|
302
|
+
start_page=start_page,
|
|
303
|
+
pages=pages,
|
|
304
|
+
width=df.shape[1],
|
|
305
|
+
header_tokens=header_tokens,
|
|
306
|
+
first_row_tokens=first_row_tokens,
|
|
307
|
+
raw_columns=raw_columns,
|
|
308
|
+
vert_center=None, # Set if bbox available
|
|
309
|
+
vert_top=None, # Normalized 0-1, 0=top of page
|
|
310
|
+
vert_bottom=None, # Normalized 0-1, 1=bottom of page
|
|
311
|
+
is_header_orphan=False, # True if headers-only, no/few data rows
|
|
312
|
+
is_data_orphan=False, # True if data-only, no real headers
|
|
313
|
+
numeric_like_cols=is_numeric_like_colnames(raw_columns),
|
|
314
|
+
row_count=df.shape[0],
|
|
315
|
+
is_headerless=is_headerless,
|
|
316
|
+
)
|
|
317
|
+
)
|
|
310
318
|
return tables_meta
|
|
311
319
|
|
|
312
320
|
def inject(self, doc: Any, logical_tables: List[LogicalTable]) -> Any:
|
|
@@ -325,6 +333,7 @@ class MyParserAdapter:
|
|
|
325
333
|
|
|
326
334
|
return doc
|
|
327
335
|
|
|
336
|
+
|
|
328
337
|
# Use it:
|
|
329
338
|
stitcher = TableStitcher(adapter=MyParserAdapter())
|
|
330
339
|
doc = stitcher.stitch(doc)
|
|
@@ -374,6 +383,7 @@ except StitchingError as e:
|
|
|
374
383
|
|
|
375
384
|
```python
|
|
376
385
|
import logging
|
|
386
|
+
|
|
377
387
|
logging.getLogger("table_stitcher").setLevel(logging.INFO)
|
|
378
388
|
```
|
|
379
389
|
|
|
@@ -65,8 +65,8 @@ from table_stitcher import stitch_tables
|
|
|
65
65
|
|
|
66
66
|
converter = DocumentConverter()
|
|
67
67
|
doc = converter.convert("report.pdf").document
|
|
68
|
-
doc = stitch_tables(doc)
|
|
69
|
-
|
|
68
|
+
doc = stitch_tables(doc) # merged tables; ready for
|
|
69
|
+
# export_to_markdown() / HTML / LLM
|
|
70
70
|
```
|
|
71
71
|
|
|
72
72
|
`stitch_tables()` mutates `doc` in place and returns the same object. If you
|
|
@@ -94,10 +94,10 @@ Runnable end-to-end scripts live in [`examples/`](examples/):
|
|
|
94
94
|
from table_stitcher import stitch_tables, MultiPageConfig
|
|
95
95
|
|
|
96
96
|
config = MultiPageConfig(
|
|
97
|
-
max_page_gap=1,
|
|
98
|
-
max_width_difference=2,
|
|
99
|
-
header_sim_strict=0.6,
|
|
100
|
-
stitch_separator="\n",
|
|
97
|
+
max_page_gap=1, # Only merge tables on consecutive pages
|
|
98
|
+
max_width_difference=2, # Column count tolerance
|
|
99
|
+
header_sim_strict=0.6, # Threshold for repeated header detection
|
|
100
|
+
stitch_separator="\n", # Join character for split content
|
|
101
101
|
)
|
|
102
102
|
|
|
103
103
|
doc = stitch_tables(doc, config=config)
|
|
@@ -110,6 +110,7 @@ from typing import Any, List
|
|
|
110
110
|
from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
|
|
111
111
|
from table_stitcher.adapters.base import TableStitcherAdapter
|
|
112
112
|
|
|
113
|
+
|
|
113
114
|
class MyParserAdapter:
|
|
114
115
|
def extract(self, doc, cfg: MultiPageConfig) -> List[TableMeta]:
|
|
115
116
|
"""Read tables from your document format into TableMeta objects."""
|
|
@@ -119,6 +120,7 @@ class MyParserAdapter:
|
|
|
119
120
|
"""Write merged results back into your document format."""
|
|
120
121
|
...
|
|
121
122
|
|
|
123
|
+
|
|
122
124
|
stitcher = TableStitcher(adapter=MyParserAdapter())
|
|
123
125
|
doc = stitcher.stitch(doc)
|
|
124
126
|
```
|
|
@@ -220,7 +222,13 @@ from typing import Any, List
|
|
|
220
222
|
import pandas as pd
|
|
221
223
|
from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
|
|
222
224
|
from table_stitcher.adapters.base import TableStitcherAdapter
|
|
223
|
-
from table_stitcher.merger import
|
|
225
|
+
from table_stitcher.merger import (
|
|
226
|
+
tokenize,
|
|
227
|
+
normalize_col_name,
|
|
228
|
+
is_numeric_like_colnames,
|
|
229
|
+
first_row_has_number,
|
|
230
|
+
)
|
|
231
|
+
|
|
224
232
|
|
|
225
233
|
class MyParserAdapter:
|
|
226
234
|
def extract(self, doc: Any, cfg: MultiPageConfig) -> List[TableMeta]:
|
|
@@ -243,32 +251,32 @@ class MyParserAdapter:
|
|
|
243
251
|
# 4. Tokenize first row (fallback similarity signal)
|
|
244
252
|
first_row_tokens = set()
|
|
245
253
|
if df.shape[0] > 0:
|
|
246
|
-
first_row_tokens = tokenize(
|
|
247
|
-
" ".join(str(x) for x in df.iloc[0].tolist())
|
|
248
|
-
)
|
|
254
|
+
first_row_tokens = tokenize(" ".join(str(x) for x in df.iloc[0].tolist()))
|
|
249
255
|
|
|
250
256
|
# 5. Classify: is_headerless, is_header_orphan, is_data_orphan
|
|
251
257
|
raw_columns = [str(c) for c in df.columns]
|
|
252
|
-
is_headerless = df.attrs.get(
|
|
253
|
-
|
|
254
|
-
tables_meta.append(
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
258
|
+
is_headerless = df.attrs.get("is_headerless", False)
|
|
259
|
+
|
|
260
|
+
tables_meta.append(
|
|
261
|
+
TableMeta(
|
|
262
|
+
idx=idx,
|
|
263
|
+
df=df,
|
|
264
|
+
start_page=start_page,
|
|
265
|
+
pages=pages,
|
|
266
|
+
width=df.shape[1],
|
|
267
|
+
header_tokens=header_tokens,
|
|
268
|
+
first_row_tokens=first_row_tokens,
|
|
269
|
+
raw_columns=raw_columns,
|
|
270
|
+
vert_center=None, # Set if bbox available
|
|
271
|
+
vert_top=None, # Normalized 0-1, 0=top of page
|
|
272
|
+
vert_bottom=None, # Normalized 0-1, 1=bottom of page
|
|
273
|
+
is_header_orphan=False, # True if headers-only, no/few data rows
|
|
274
|
+
is_data_orphan=False, # True if data-only, no real headers
|
|
275
|
+
numeric_like_cols=is_numeric_like_colnames(raw_columns),
|
|
276
|
+
row_count=df.shape[0],
|
|
277
|
+
is_headerless=is_headerless,
|
|
278
|
+
)
|
|
279
|
+
)
|
|
272
280
|
return tables_meta
|
|
273
281
|
|
|
274
282
|
def inject(self, doc: Any, logical_tables: List[LogicalTable]) -> Any:
|
|
@@ -287,6 +295,7 @@ class MyParserAdapter:
|
|
|
287
295
|
|
|
288
296
|
return doc
|
|
289
297
|
|
|
298
|
+
|
|
290
299
|
# Use it:
|
|
291
300
|
stitcher = TableStitcher(adapter=MyParserAdapter())
|
|
292
301
|
doc = stitcher.stitch(doc)
|
|
@@ -336,6 +345,7 @@ except StitchingError as e:
|
|
|
336
345
|
|
|
337
346
|
```python
|
|
338
347
|
import logging
|
|
348
|
+
|
|
339
349
|
logging.getLogger("table_stitcher").setLevel(logging.INFO)
|
|
340
350
|
```
|
|
341
351
|
|
|
@@ -104,9 +104,9 @@ A few structural constants live at the top of `docling.py` rather than
|
|
|
104
104
|
in `MultiPageConfig`:
|
|
105
105
|
|
|
106
106
|
```python
|
|
107
|
-
_MAX_HEADER_CELL_LEN = 30
|
|
108
|
-
_DATA_PATTERNS
|
|
109
|
-
_AUTO_COLNAME_RE
|
|
107
|
+
_MAX_HEADER_CELL_LEN = 30 # header cells typically short; data cells longer
|
|
108
|
+
_DATA_PATTERNS # regex list for "this cell is data, not header"
|
|
109
|
+
_AUTO_COLNAME_RE # "Column_N" / "Unnamed: N" parser placeholders
|
|
110
110
|
```
|
|
111
111
|
|
|
112
112
|
These are **adapter-intrinsic** — tuning them changes how the adapter
|
|
@@ -499,6 +499,42 @@ def _reemit_body_row(
|
|
|
499
499
|
return grid_row, distinct
|
|
500
500
|
|
|
501
501
|
|
|
502
|
+
# Jaccard threshold for recognizing a body row as a reprinted continuation
|
|
503
|
+
# header. Combined with Docling's column_header flag — both must hold — so it
|
|
504
|
+
# stays conservative. Matches the merger's header_sim_strict default; the
|
|
505
|
+
# punctuation-agnostic tokenizer makes it tolerant of per-cell OCR drift such
|
|
506
|
+
# as "(S$)" vs "($$)".
|
|
507
|
+
_REPEATED_HEADER_SIM = 0.6
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def _row_token_set(cells: list[TableCell]) -> set:
|
|
511
|
+
"""Union of tokenized cell text for a grid row (duplicate span cells fold in)."""
|
|
512
|
+
toks: set = set()
|
|
513
|
+
for c in cells:
|
|
514
|
+
if c:
|
|
515
|
+
toks |= tokenize(getattr(c, "text", "") or "")
|
|
516
|
+
return toks
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def _is_reprinted_header(orig_row: list[TableCell], header_sigs: list[set]) -> bool:
|
|
520
|
+
"""True if ``orig_row`` is a reprinted continuation header to drop from the body.
|
|
521
|
+
|
|
522
|
+
Docling reprints the column header at the top of each continuation page; on
|
|
523
|
+
a multi-row header those rows survive the merge as bogus data rows. They are
|
|
524
|
+
dropped only when BOTH signals agree: Docling flagged the row
|
|
525
|
+
``column_header`` AND it is a tokenized match for one of the reconstructed
|
|
526
|
+
header-block rows. The flag alone is unreliable — Docling over-flags
|
|
527
|
+
rowspan/continuation *data* rows as headers — so the content match guards
|
|
528
|
+
against deleting real data.
|
|
529
|
+
"""
|
|
530
|
+
if not any(getattr(c, "column_header", False) for c in orig_row if c):
|
|
531
|
+
return False
|
|
532
|
+
toks = _row_token_set(orig_row)
|
|
533
|
+
if not toks:
|
|
534
|
+
return False
|
|
535
|
+
return any(jaccard(toks, sig) >= _REPEATED_HEADER_SIM for sig in header_sigs)
|
|
536
|
+
|
|
537
|
+
|
|
502
538
|
def _dataframe_to_docling_data(
|
|
503
539
|
df: pd.DataFrame,
|
|
504
540
|
original_data: Optional[TableData] = None,
|
|
@@ -587,12 +623,15 @@ def _dataframe_to_docling_data(
|
|
|
587
623
|
|
|
588
624
|
# --- Build data rows from merged DataFrame ---
|
|
589
625
|
# Index member fragments' original body rows so spanning cells survive the
|
|
590
|
-
# round-trip (see
|
|
626
|
+
# round-trip (see _index_member_rows).
|
|
591
627
|
body_index = _index_member_rows(member_data) if member_data else {}
|
|
628
|
+
# Tokenized signatures of the reconstructed header block, for dropping
|
|
629
|
+
# reprinted continuation headers (see _is_reprinted_header).
|
|
630
|
+
header_sigs = [s for s in (_row_token_set(h) for h in orig_header_rows) if s]
|
|
592
631
|
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
632
|
+
emitted = 0
|
|
633
|
+
for _, row in df.iterrows():
|
|
634
|
+
table_row_idx = num_header_rows + emitted
|
|
596
635
|
row_vals = ["" if (pd.isna(v) or v is None) else str(v) for v in row]
|
|
597
636
|
|
|
598
637
|
# Re-emit untouched rows from their original grid cells (preserves
|
|
@@ -601,9 +640,15 @@ def _dataframe_to_docling_data(
|
|
|
601
640
|
bucket = body_index.get(tuple(row_vals))
|
|
602
641
|
if bucket:
|
|
603
642
|
orig_row = bucket.pop(0)
|
|
643
|
+
if header_sigs and _is_reprinted_header(orig_row, header_sigs):
|
|
644
|
+
# Reprinted header from a continuation page — already present as
|
|
645
|
+
# the header block; drop it instead of duplicating into the body.
|
|
646
|
+
log.debug("Dropped reprinted continuation header row from merged body.")
|
|
647
|
+
continue
|
|
604
648
|
grid_row, distinct = _reemit_body_row(orig_row, table_row_idx, has_row_headers)
|
|
605
649
|
grid.append(grid_row)
|
|
606
650
|
table_cells.extend(distinct)
|
|
651
|
+
emitted += 1
|
|
607
652
|
continue
|
|
608
653
|
|
|
609
654
|
grid_row: list[TableCell] = []
|
|
@@ -625,8 +670,9 @@ def _dataframe_to_docling_data(
|
|
|
625
670
|
table_cells.append(cell)
|
|
626
671
|
|
|
627
672
|
grid.append(grid_row)
|
|
673
|
+
emitted += 1
|
|
628
674
|
|
|
629
|
-
num_total_rows = num_header_rows +
|
|
675
|
+
num_total_rows = num_header_rows + emitted
|
|
630
676
|
|
|
631
677
|
return TableData(num_rows=num_total_rows, num_cols=num_cols, table_cells=table_cells, grid=grid)
|
|
632
678
|
|
|
@@ -619,13 +619,34 @@ def _build_orphan_merged_table(
|
|
|
619
619
|
)
|
|
620
620
|
|
|
621
621
|
|
|
622
|
+
def _dedupe_labels(labels: list[str]) -> list[str]:
|
|
623
|
+
"""Make labels unique by suffixing repeats (``B, B`` -> ``B, B.1``)."""
|
|
624
|
+
used: set[str] = set()
|
|
625
|
+
counts: dict[str, int] = {}
|
|
626
|
+
out: list[str] = []
|
|
627
|
+
for label in labels:
|
|
628
|
+
candidate = label
|
|
629
|
+
while candidate in used:
|
|
630
|
+
counts[label] = counts.get(label, 0) + 1
|
|
631
|
+
candidate = f"{label}.{counts[label]}"
|
|
632
|
+
used.add(candidate)
|
|
633
|
+
out.append(candidate)
|
|
634
|
+
return out
|
|
635
|
+
|
|
636
|
+
|
|
622
637
|
def _build_generic_merged_table(
|
|
623
638
|
members: list[int], meta_by_idx: dict[int, TableMeta], cfg: MultiPageConfig
|
|
624
639
|
) -> tuple[pd.DataFrame, set[int], list[str]]:
|
|
625
640
|
"""Build merged table for the general case."""
|
|
626
641
|
base = meta_by_idx[members[0]]
|
|
627
642
|
merged_df = base.df.copy()
|
|
628
|
-
|
|
643
|
+
# Duplicate header labels are normal in the wild (rowspan/colspan
|
|
644
|
+
# parsers, 13F voting-authority triplets), but pd.concat cannot align
|
|
645
|
+
# frames on a non-unique column Index. Merge under deduped labels and
|
|
646
|
+
# restore the originals on the way out.
|
|
647
|
+
original_cols = [str(c) for c in base.df.columns]
|
|
648
|
+
canonical_cols = _dedupe_labels(original_cols)
|
|
649
|
+
merged_df.columns = canonical_cols
|
|
629
650
|
merged_pages = set(base.pages)
|
|
630
651
|
warnings: list[str] = []
|
|
631
652
|
prev = base
|
|
@@ -648,6 +669,7 @@ def _build_generic_merged_table(
|
|
|
648
669
|
merged_pages.update(m.pages)
|
|
649
670
|
prev = m
|
|
650
671
|
|
|
672
|
+
merged_df.columns = original_cols + [str(c) for c in merged_df.columns[len(original_cols) :]]
|
|
651
673
|
return merged_df, merged_pages, warnings
|
|
652
674
|
|
|
653
675
|
|
|
@@ -333,7 +333,15 @@ def assert_public_stitch_injects_docling_doc(
|
|
|
333
333
|
ctx = f"public stitch for members={members}, pages={exp['pages']}"
|
|
334
334
|
|
|
335
335
|
assert getattr(anchor.data, "num_rows", 0) > 0, f"{ctx}: anchor has no data"
|
|
336
|
-
if "
|
|
336
|
+
if "injected_rows" in exp:
|
|
337
|
+
# Exact injected row count. Used when injection legitimately differs
|
|
338
|
+
# from the parser-neutral shape — e.g. reprinted continuation-page
|
|
339
|
+
# headers are dropped from the body (they remain in the merged
|
|
340
|
+
# DataFrame but are not duplicated into the stitched document).
|
|
341
|
+
assert anchor.data.num_rows == exp["injected_rows"], (
|
|
342
|
+
f"{ctx}: injected rows {anchor.data.num_rows} != expected {exp['injected_rows']}"
|
|
343
|
+
)
|
|
344
|
+
elif "shape" in exp:
|
|
337
345
|
# +1 or more for header rows; this guards that merged data was injected.
|
|
338
346
|
assert anchor.data.num_rows >= exp["shape"][0] + 1, (
|
|
339
347
|
f"{ctx}: anchor rows {anchor.data.num_rows} do not contain merged body"
|
|
@@ -37,6 +37,9 @@ logical_tables:
|
|
|
37
37
|
shape:
|
|
38
38
|
- 500
|
|
39
39
|
- 3
|
|
40
|
+
# Injection drops the reprinted header block (title + column names) that the
|
|
41
|
+
# merged DataFrame still carries as its first two rows: 500 - 2 = header(2) + body(498).
|
|
42
|
+
injected_rows: 500
|
|
40
43
|
columns:
|
|
41
44
|
- Column_0
|
|
42
45
|
- Column_1
|
|
@@ -388,6 +388,77 @@ class TestBodySpanPreservation:
|
|
|
388
388
|
assert all(cell.col_span == 1 for cell in td.grid[1])
|
|
389
389
|
|
|
390
390
|
|
|
391
|
+
class TestReprintedHeaderDedup:
|
|
392
|
+
"""Reprinted continuation-page headers are dropped from the injected body,
|
|
393
|
+
but column_header-flagged rows that don't match the header (Docling
|
|
394
|
+
over-flagging rowspan/continuation data) are kept.
|
|
395
|
+
"""
|
|
396
|
+
|
|
397
|
+
@staticmethod
|
|
398
|
+
def _cell(text, r, c, *, header):
|
|
399
|
+
return TableCell(
|
|
400
|
+
text=text,
|
|
401
|
+
row_span=1,
|
|
402
|
+
col_span=1,
|
|
403
|
+
column_header=header,
|
|
404
|
+
row_header=False,
|
|
405
|
+
start_row_offset_idx=r,
|
|
406
|
+
end_row_offset_idx=r + 1,
|
|
407
|
+
start_col_offset_idx=c,
|
|
408
|
+
end_col_offset_idx=c + 1,
|
|
409
|
+
)
|
|
410
|
+
|
|
411
|
+
def _table(self, rows: list) -> TableData:
|
|
412
|
+
grid = []
|
|
413
|
+
flat = []
|
|
414
|
+
for r, (cells_text, is_header) in enumerate(rows):
|
|
415
|
+
grid_row = [self._cell(t, r, c, header=is_header) for c, t in enumerate(cells_text)]
|
|
416
|
+
grid.append(grid_row)
|
|
417
|
+
flat.extend(grid_row)
|
|
418
|
+
return TableData(num_rows=len(rows), num_cols=len(rows[0][0]), table_cells=flat, grid=grid)
|
|
419
|
+
|
|
420
|
+
def test_drops_reprinted_header_keeps_misflagged_data(self):
|
|
421
|
+
# Anchor: 1-row header "(S$)" + one data row.
|
|
422
|
+
anchor = self._table([(["SECTION", "LIMIT (S$)"], True), (["Death", "100"], False)])
|
|
423
|
+
# Continuation: header reprinted with OCR drift "($$)", a data row, and
|
|
424
|
+
# a row Docling wrongly flagged column_header (real data, distinct text).
|
|
425
|
+
satellite = self._table(
|
|
426
|
+
[
|
|
427
|
+
(["SECTION", "LIMIT ($$)"], True),
|
|
428
|
+
(["Injury", "50"], False),
|
|
429
|
+
(["Sub-limit per accident", "999"], True),
|
|
430
|
+
]
|
|
431
|
+
)
|
|
432
|
+
# Merged DataFrame: the merger concatenates everything, including the
|
|
433
|
+
# reprinted header and the mis-flagged row.
|
|
434
|
+
merged_df = pd.DataFrame(
|
|
435
|
+
[
|
|
436
|
+
["Death", "100"],
|
|
437
|
+
["SECTION", "LIMIT ($$)"],
|
|
438
|
+
["Injury", "50"],
|
|
439
|
+
["Sub-limit per accident", "999"],
|
|
440
|
+
],
|
|
441
|
+
columns=["Column_0", "Column_1"],
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
td = _dataframe_to_docling_data(
|
|
445
|
+
merged_df, original_data=anchor, member_data=[anchor, satellite]
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
body = [r for r in td.grid if not any(getattr(c, "column_header", False) for c in r if c)]
|
|
449
|
+
body_text = " ".join(str(c.text) for r in body for c in r if c)
|
|
450
|
+
|
|
451
|
+
# Reprinted header (drifted) dropped from the body...
|
|
452
|
+
assert "($$)" not in body_text
|
|
453
|
+
# ...but the header block still carries the anchor's "(S$)".
|
|
454
|
+
assert any("(S$)" in (c.text or "") for r in td.grid for c in r if c)
|
|
455
|
+
# Real data preserved, including the column_header-mis-flagged row.
|
|
456
|
+
assert "Death" in body_text and "Injury" in body_text
|
|
457
|
+
assert "Sub-limit per accident" in body_text and "999" in body_text
|
|
458
|
+
# 1 header row + 3 body rows (one reprinted header removed from 4).
|
|
459
|
+
assert td.num_rows == 4
|
|
460
|
+
|
|
461
|
+
|
|
391
462
|
class TestAdapterProtocol:
|
|
392
463
|
"""Verify DoclingAdapter satisfies the protocol."""
|
|
393
464
|
|
|
@@ -228,6 +228,50 @@ class TestWidthOverflowPolicy:
|
|
|
228
228
|
align_dataframe_to_header(df, ["A"], meta, cfg)
|
|
229
229
|
|
|
230
230
|
|
|
231
|
+
class TestDuplicateHeaderLabels:
|
|
232
|
+
def test_merge_survives_duplicate_header_labels(self):
|
|
233
|
+
"""
|
|
234
|
+
Extracted headers repeating a label (e.g. 13F voting-authority
|
|
235
|
+
triplets emitting 'COLUMN 8' x3) must not crash the multipage
|
|
236
|
+
merge: pd.concat cannot align frames on a non-unique column Index.
|
|
237
|
+
"""
|
|
238
|
+
cols = ["A", "B", "B"]
|
|
239
|
+
df1 = pd.DataFrame([["1", "2", "3"]], columns=cols)
|
|
240
|
+
df2 = pd.DataFrame([["4", "5", "6"]], columns=cols)
|
|
241
|
+
metas = [
|
|
242
|
+
_make_meta(idx=0, df=df1, start_page=1),
|
|
243
|
+
_make_meta(idx=1, df=df2, start_page=2),
|
|
244
|
+
]
|
|
245
|
+
results = merge_multipage_tables(metas, MultiPageConfig())
|
|
246
|
+
assert len(results) == 1
|
|
247
|
+
assert results[0].df.shape == (2, 3)
|
|
248
|
+
assert results[0].df.iloc[1].tolist() == ["4", "5", "6"]
|
|
249
|
+
# Original (duplicated) labels are preserved in the output, matching
|
|
250
|
+
# how single-fragment tables pass through untouched.
|
|
251
|
+
assert list(results[0].df.columns) == cols
|
|
252
|
+
|
|
253
|
+
def test_duplicate_labels_with_wider_continuation(self):
|
|
254
|
+
cols = ["A", "B", "B"]
|
|
255
|
+
df1 = pd.DataFrame([["1", "2", "3"]], columns=cols)
|
|
256
|
+
df2 = pd.DataFrame([["4", "5", "6", "7"]], columns=cols + ["C"])
|
|
257
|
+
metas = [
|
|
258
|
+
_make_meta(idx=0, df=df1, start_page=1),
|
|
259
|
+
_make_meta(idx=1, df=df2, start_page=2),
|
|
260
|
+
]
|
|
261
|
+
results = merge_multipage_tables(metas, MultiPageConfig())
|
|
262
|
+
assert len(results) == 1
|
|
263
|
+
assert results[0].df.shape == (2, 4)
|
|
264
|
+
assert results[0].df.iloc[1, 3] == "7"
|
|
265
|
+
assert list(results[0].df.columns)[:3] == cols
|
|
266
|
+
|
|
267
|
+
def test_dedupe_labels_avoids_existing_suffix_collision(self):
|
|
268
|
+
from table_stitcher.merger import _dedupe_labels
|
|
269
|
+
|
|
270
|
+
out = _dedupe_labels(["X", "X", "X.1"])
|
|
271
|
+
assert len(set(out)) == 3
|
|
272
|
+
assert out[0] == "X"
|
|
273
|
+
|
|
274
|
+
|
|
231
275
|
class TestMergeTrace:
|
|
232
276
|
def test_logical_table_explains_merge_reason_and_signals(self):
|
|
233
277
|
df = pd.DataFrame({"Name": ["Alice"], "Age": ["30"]})
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/fixtures/tablemeta/headerless-width-drift.yaml
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/_tools/regenerate_expected.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/.gitkeep
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/.gitkeep
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|