table-stitcher 0.4.2__tar.gz → 0.4.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (133) hide show
  1. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/CHANGELOG.md +33 -0
  2. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/PKG-INFO +42 -32
  3. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/README.md +40 -30
  4. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/pyproject.toml +1 -1
  5. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/README.md +3 -3
  6. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/docling.py +51 -5
  7. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/merger.py +23 -1
  8. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/conftest.py +9 -1
  9. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/15-page-druglist.corp.expected.yaml +3 -0
  10. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/covid-misc-labs-4pg.pt2.expected.yaml +3 -0
  11. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/retirement-portfolio.corp.expected.yaml +3 -0
  12. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_docling_adapter.py +71 -0
  13. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_merger.py +44 -0
  14. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.claude/settings.json +0 -0
  15. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/ISSUE_TEMPLATE/bug_report.yml +0 -0
  16. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/ISSUE_TEMPLATE/config.yml +0 -0
  17. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/ISSUE_TEMPLATE/feature_request.yml +0 -0
  18. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/dependabot.yml +0 -0
  19. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/pull_request_template.md +0 -0
  20. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/workflows/ci.yml +0 -0
  21. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/workflows/release.yml +0 -0
  22. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.github/workflows/upstream-smoke.yml +0 -0
  23. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.gitignore +0 -0
  24. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/.pre-commit-config.yaml +0 -0
  25. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/CONTRIBUTING.md +0 -0
  26. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/LICENSE +0 -0
  27. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/SECURITY.md +0 -0
  28. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/examples/basic_pipeline.py +0 -0
  29. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/examples/system_controller.py +0 -0
  30. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/scripts/regenerate_docling_snapshots.py +0 -0
  31. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/scripts/release_gate.sh +0 -0
  32. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/__init__.py +0 -0
  33. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/__init__.py +0 -0
  34. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/adapters/base.py +0 -0
  35. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/models.py +0 -0
  36. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/src/table_stitcher/py.typed +0 -0
  37. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/README.md +0 -0
  38. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/__init__.py +0 -0
  39. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/fixtures/tablemeta/headerless-width-drift.yaml +0 -0
  40. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/__init__.py +0 -0
  41. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/_tools/__init__.py +0 -0
  42. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/_tools/regenerate_expected.py +0 -0
  43. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/_synth/__init__.py +0 -0
  44. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/_synth/generate.py +0 -0
  45. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/.gitkeep +0 -0
  46. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/kaop-study-mixed-3pg.pt2.docling.json +0 -0
  47. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/kaop-study-mixed-3pg.pt2.expected.yaml +0 -0
  48. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/kaop-study-mixed-3pg.pt2.pdf +0 -0
  49. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/lab-panels-3pg.corp.docling.json +0 -0
  50. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/lab-panels-3pg.corp.expected.yaml +0 -0
  51. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/distinct-tables-no-merge/lab-panels-3pg.corp.pdf +0 -0
  52. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/false-merge/category-rows-thematic-2pg.corp.docling.json +0 -0
  53. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/false-merge/category-rows-thematic-2pg.corp.expected.yaml +0 -0
  54. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/false-merge/category-rows-thematic-2pg.corp.pdf +0 -0
  55. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/15-page-druglist.corp.docling.json +0 -0
  56. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/15-page-druglist.corp.pdf +0 -0
  57. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/gene-symbols-6pg.pt2.docling.json +0 -0
  58. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/gene-symbols-6pg.pt2.expected.yaml +0 -0
  59. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/gene-symbols-6pg.pt2.pdf +0 -0
  60. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rct-study-table-5pg.pt2.docling.json +0 -0
  61. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rct-study-table-5pg.pt2.expected.yaml +0 -0
  62. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rct-study-table-5pg.pt2.pdf +0 -0
  63. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rrt-outcomes-2pg.pt2.docling.json +0 -0
  64. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rrt-outcomes-2pg.pt2.expected.yaml +0 -0
  65. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/rrt-outcomes-2pg.pt2.pdf +0 -0
  66. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/uveitis-case-series-5pg.pt2.docling.json +0 -0
  67. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/uveitis-case-series-5pg.pt2.expected.yaml +0 -0
  68. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/headerless-continuation/uveitis-case-series-5pg.pt2.pdf +0 -0
  69. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/covid-misc-labs-4pg.pt2.docling.json +0 -0
  70. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/covid-misc-labs-4pg.pt2.pdf +0 -0
  71. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/lit-review-3pg.pt2.docling.json +0 -0
  72. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/lit-review-3pg.pt2.expected.yaml +0 -0
  73. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/lit-review-3pg.pt2.pdf +0 -0
  74. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/retirement-portfolio.corp.docling.json +0 -0
  75. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/inconsistent-header-detection/retirement-portfolio.corp.pdf +0 -0
  76. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/.gitkeep +0 -0
  77. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/biological-process-2pg.pt2.docling.json +0 -0
  78. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/biological-process-2pg.pt2.expected.yaml +0 -0
  79. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/biological-process-2pg.pt2.pdf +0 -0
  80. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/symptoms-mediators-4pg.pt2.docling.json +0 -0
  81. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/symptoms-mediators-4pg.pt2.expected.yaml +0 -0
  82. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/loose-header-layout/symptoms-mediators-4pg.pt2.pdf +0 -0
  83. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/corporate-history-2pg.edinet.docling.json +0 -0
  84. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/corporate-history-2pg.edinet.expected.yaml +0 -0
  85. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/corporate-history-2pg.edinet.pdf +0 -0
  86. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/subsidiaries-4pg.edinet.docling.json +0 -0
  87. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/subsidiaries-4pg.edinet.expected.yaml +0 -0
  88. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/multilingual/subsidiaries-4pg.edinet.pdf +0 -0
  89. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/.gitkeep +0 -0
  90. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/varicose-veins-new-table-header-7pg.pt2.docling.json +0 -0
  91. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/varicose-veins-new-table-header-7pg.pt2.expected.yaml +0 -0
  92. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/orphan-pair/varicose-veins-new-table-header-7pg.pt2.pdf +0 -0
  93. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/page-gap-too-large/unrelated-tables-gap4.synth.docling.json +0 -0
  94. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/page-gap-too-large/unrelated-tables-gap4.synth.expected.yaml +0 -0
  95. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/page-gap-too-large/unrelated-tables-gap4.synth.pdf +0 -0
  96. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/4-page-substance-list.corp.docling.json +0 -0
  97. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/4-page-substance-list.corp.expected.yaml +0 -0
  98. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/4-page-substance-list.corp.pdf +0 -0
  99. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/cell-markers-4pg.pt2.docling.json +0 -0
  100. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/cell-markers-4pg.pt2.expected.yaml +0 -0
  101. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/cell-markers-4pg.pt2.pdf +0 -0
  102. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/challenge-categories-2pg.pt2.docling.json +0 -0
  103. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/challenge-categories-2pg.pt2.expected.yaml +0 -0
  104. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/challenge-categories-2pg.pt2.pdf +0 -0
  105. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/fungal-taxonomy-4pg.pt2.docling.json +0 -0
  106. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/fungal-taxonomy-4pg.pt2.expected.yaml +0 -0
  107. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/fungal-taxonomy-4pg.pt2.pdf +0 -0
  108. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/rowspan-insurance-payout.corp.docling.json +0 -0
  109. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/rowspan-insurance-payout.corp.expected.yaml +0 -0
  110. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/rowspan-insurance-payout.corp.pdf +0 -0
  111. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/search-strategies-2pg.pt2.docling.json +0 -0
  112. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/search-strategies-2pg.pt2.expected.yaml +0 -0
  113. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/search-strategies-2pg.pt2.pdf +0 -0
  114. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/study-sample-7pg.pt2.docling.json +0 -0
  115. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/study-sample-7pg.pt2.expected.yaml +0 -0
  116. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/study-sample-7pg.pt2.pdf +0 -0
  117. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/themes-exemplars-3pg.pt2.docling.json +0 -0
  118. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/themes-exemplars-3pg.pt2.expected.yaml +0 -0
  119. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/repeated-header/themes-exemplars-3pg.pt2.pdf +0 -0
  120. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/simple-continuation/sample-table.corp.docling.json +0 -0
  121. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/simple-continuation/sample-table.corp.expected.yaml +0 -0
  122. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/simple-continuation/sample-table.corp.pdf +0 -0
  123. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/spillover/note-overflow.synth.docling.json +0 -0
  124. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/spillover/note-overflow.synth.expected.yaml +0 -0
  125. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/spillover/note-overflow.synth.pdf +0 -0
  126. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/.gitkeep +0 -0
  127. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/abx-literature-review-7pg.pt2.docling.json +0 -0
  128. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/abx-literature-review-7pg.pt2.expected.yaml +0 -0
  129. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/fixtures/width-drift/abx-literature-review-7pg.pt2.pdf +0 -0
  130. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/integration/test_fixtures.py +0 -0
  131. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_intervening_content_guard.py +0 -0
  132. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_public_api.py +0 -0
  133. {table_stitcher-0.4.2 → table_stitcher-0.4.4}/tests/test_tablemeta_fixtures.py +0 -0
@@ -7,6 +7,39 @@ the project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html
7
7
 
8
8
  ## [Unreleased]
9
9
 
10
+ ## [0.4.4] — 2026-08-13
11
+
12
+ ### Fixed
13
+
14
+ - **Duplicate header labels crashed the multipage merge** (`merger.py`).
15
+ Extracted headers that repeat a label — e.g. SEC 13F voting-authority
16
+ triplets where TableFormer emits `COLUMN 8` three times — made
17
+ `_build_generic_merged_table` raise `ValueError: Reindexing only valid
18
+ with uniquely valued Index objects`, because `pd.concat` cannot align
19
+ fragments on a non-unique column Index. Downstream consumers that
20
+ fail-soft on the error silently kept a degraded table set for the whole
21
+ document. Fragments are now merged under positionally deduped labels
22
+ (collision-safe against pre-existing `X.1`-style names) and the original
23
+ duplicated labels are restored on the merged output, matching how
24
+ single-fragment tables pass through untouched.
25
+
26
+ ## [0.4.3] — 2026-06-11
27
+
28
+ ### Fixed
29
+
30
+ - **Reprinted continuation-page headers appended as data rows on multi-page
31
+ merge** (`adapters/docling.py`). When a table's column header is reprinted at
32
+ the top of each page — especially a multi-row (hierarchical) header — the
33
+ repeated header rows survived the merge as bogus data rows, misaligning the
34
+ stitched table. Injection now drops a body row when it is *both* flagged
35
+ `column_header` by Docling *and* a tokenized match (Jaccard ≥ 0.6) for the
36
+ reconstructed header block. Both signals are required: the flag alone is
37
+ unreliable (Docling over-flags rowspan/continuation *data* rows as headers),
38
+ and the tokenized comparison is punctuation-agnostic, so per-cell OCR drift
39
+ such as `(S$)` vs `($$)` is tolerated without any threshold tuning. The merged
40
+ DataFrame (`lt.df`) is unchanged; only the injected document is de-duplicated.
41
+ A `debug` log reports each dropped row.
42
+
10
43
  ## [0.4.2] — 2026-06-08
11
44
 
12
45
  ### Fixed
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: table-stitcher
3
- Version: 0.4.2
3
+ Version: 0.4.4
4
4
  Summary: Reassemble tables split across page boundaries in PDF extraction
5
5
  Project-URL: Homepage, https://github.com/pebbleroad/table-stitcher
6
6
  Project-URL: Repository, https://github.com/pebbleroad/table-stitcher
@@ -103,8 +103,8 @@ from table_stitcher import stitch_tables
103
103
 
104
104
  converter = DocumentConverter()
105
105
  doc = converter.convert("report.pdf").document
106
- doc = stitch_tables(doc) # merged tables; ready for
107
- # export_to_markdown() / HTML / LLM
106
+ doc = stitch_tables(doc) # merged tables; ready for
107
+ # export_to_markdown() / HTML / LLM
108
108
  ```
109
109
 
110
110
  `stitch_tables()` mutates `doc` in place and returns the same object. If you
@@ -132,10 +132,10 @@ Runnable end-to-end scripts live in [`examples/`](examples/):
132
132
  from table_stitcher import stitch_tables, MultiPageConfig
133
133
 
134
134
  config = MultiPageConfig(
135
- max_page_gap=1, # Only merge tables on consecutive pages
136
- max_width_difference=2, # Column count tolerance
137
- header_sim_strict=0.6, # Threshold for repeated header detection
138
- stitch_separator="\n", # Join character for split content
135
+ max_page_gap=1, # Only merge tables on consecutive pages
136
+ max_width_difference=2, # Column count tolerance
137
+ header_sim_strict=0.6, # Threshold for repeated header detection
138
+ stitch_separator="\n", # Join character for split content
139
139
  )
140
140
 
141
141
  doc = stitch_tables(doc, config=config)
@@ -148,6 +148,7 @@ from typing import Any, List
148
148
  from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
149
149
  from table_stitcher.adapters.base import TableStitcherAdapter
150
150
 
151
+
151
152
  class MyParserAdapter:
152
153
  def extract(self, doc, cfg: MultiPageConfig) -> List[TableMeta]:
153
154
  """Read tables from your document format into TableMeta objects."""
@@ -157,6 +158,7 @@ class MyParserAdapter:
157
158
  """Write merged results back into your document format."""
158
159
  ...
159
160
 
161
+
160
162
  stitcher = TableStitcher(adapter=MyParserAdapter())
161
163
  doc = stitcher.stitch(doc)
162
164
  ```
@@ -258,7 +260,13 @@ from typing import Any, List
258
260
  import pandas as pd
259
261
  from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
260
262
  from table_stitcher.adapters.base import TableStitcherAdapter
261
- from table_stitcher.merger import tokenize, normalize_col_name, is_numeric_like_colnames, first_row_has_number
263
+ from table_stitcher.merger import (
264
+ tokenize,
265
+ normalize_col_name,
266
+ is_numeric_like_colnames,
267
+ first_row_has_number,
268
+ )
269
+
262
270
 
263
271
  class MyParserAdapter:
264
272
  def extract(self, doc: Any, cfg: MultiPageConfig) -> List[TableMeta]:
@@ -281,32 +289,32 @@ class MyParserAdapter:
281
289
  # 4. Tokenize first row (fallback similarity signal)
282
290
  first_row_tokens = set()
283
291
  if df.shape[0] > 0:
284
- first_row_tokens = tokenize(
285
- " ".join(str(x) for x in df.iloc[0].tolist())
286
- )
292
+ first_row_tokens = tokenize(" ".join(str(x) for x in df.iloc[0].tolist()))
287
293
 
288
294
  # 5. Classify: is_headerless, is_header_orphan, is_data_orphan
289
295
  raw_columns = [str(c) for c in df.columns]
290
- is_headerless = df.attrs.get('is_headerless', False)
291
-
292
- tables_meta.append(TableMeta(
293
- idx=idx,
294
- df=df,
295
- start_page=start_page,
296
- pages=pages,
297
- width=df.shape[1],
298
- header_tokens=header_tokens,
299
- first_row_tokens=first_row_tokens,
300
- raw_columns=raw_columns,
301
- vert_center=None, # Set if bbox available
302
- vert_top=None, # Normalized 0-1, 0=top of page
303
- vert_bottom=None, # Normalized 0-1, 1=bottom of page
304
- is_header_orphan=False, # True if headers-only, no/few data rows
305
- is_data_orphan=False, # True if data-only, no real headers
306
- numeric_like_cols=is_numeric_like_colnames(raw_columns),
307
- row_count=df.shape[0],
308
- is_headerless=is_headerless,
309
- ))
296
+ is_headerless = df.attrs.get("is_headerless", False)
297
+
298
+ tables_meta.append(
299
+ TableMeta(
300
+ idx=idx,
301
+ df=df,
302
+ start_page=start_page,
303
+ pages=pages,
304
+ width=df.shape[1],
305
+ header_tokens=header_tokens,
306
+ first_row_tokens=first_row_tokens,
307
+ raw_columns=raw_columns,
308
+ vert_center=None, # Set if bbox available
309
+ vert_top=None, # Normalized 0-1, 0=top of page
310
+ vert_bottom=None, # Normalized 0-1, 1=bottom of page
311
+ is_header_orphan=False, # True if headers-only, no/few data rows
312
+ is_data_orphan=False, # True if data-only, no real headers
313
+ numeric_like_cols=is_numeric_like_colnames(raw_columns),
314
+ row_count=df.shape[0],
315
+ is_headerless=is_headerless,
316
+ )
317
+ )
310
318
  return tables_meta
311
319
 
312
320
  def inject(self, doc: Any, logical_tables: List[LogicalTable]) -> Any:
@@ -325,6 +333,7 @@ class MyParserAdapter:
325
333
 
326
334
  return doc
327
335
 
336
+
328
337
  # Use it:
329
338
  stitcher = TableStitcher(adapter=MyParserAdapter())
330
339
  doc = stitcher.stitch(doc)
@@ -374,6 +383,7 @@ except StitchingError as e:
374
383
 
375
384
  ```python
376
385
  import logging
386
+
377
387
  logging.getLogger("table_stitcher").setLevel(logging.INFO)
378
388
  ```
379
389
 
@@ -65,8 +65,8 @@ from table_stitcher import stitch_tables
65
65
 
66
66
  converter = DocumentConverter()
67
67
  doc = converter.convert("report.pdf").document
68
- doc = stitch_tables(doc) # merged tables; ready for
69
- # export_to_markdown() / HTML / LLM
68
+ doc = stitch_tables(doc) # merged tables; ready for
69
+ # export_to_markdown() / HTML / LLM
70
70
  ```
71
71
 
72
72
  `stitch_tables()` mutates `doc` in place and returns the same object. If you
@@ -94,10 +94,10 @@ Runnable end-to-end scripts live in [`examples/`](examples/):
94
94
  from table_stitcher import stitch_tables, MultiPageConfig
95
95
 
96
96
  config = MultiPageConfig(
97
- max_page_gap=1, # Only merge tables on consecutive pages
98
- max_width_difference=2, # Column count tolerance
99
- header_sim_strict=0.6, # Threshold for repeated header detection
100
- stitch_separator="\n", # Join character for split content
97
+ max_page_gap=1, # Only merge tables on consecutive pages
98
+ max_width_difference=2, # Column count tolerance
99
+ header_sim_strict=0.6, # Threshold for repeated header detection
100
+ stitch_separator="\n", # Join character for split content
101
101
  )
102
102
 
103
103
  doc = stitch_tables(doc, config=config)
@@ -110,6 +110,7 @@ from typing import Any, List
110
110
  from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
111
111
  from table_stitcher.adapters.base import TableStitcherAdapter
112
112
 
113
+
113
114
  class MyParserAdapter:
114
115
  def extract(self, doc, cfg: MultiPageConfig) -> List[TableMeta]:
115
116
  """Read tables from your document format into TableMeta objects."""
@@ -119,6 +120,7 @@ class MyParserAdapter:
119
120
  """Write merged results back into your document format."""
120
121
  ...
121
122
 
123
+
122
124
  stitcher = TableStitcher(adapter=MyParserAdapter())
123
125
  doc = stitcher.stitch(doc)
124
126
  ```
@@ -220,7 +222,13 @@ from typing import Any, List
220
222
  import pandas as pd
221
223
  from table_stitcher import TableStitcher, MultiPageConfig, TableMeta, LogicalTable
222
224
  from table_stitcher.adapters.base import TableStitcherAdapter
223
- from table_stitcher.merger import tokenize, normalize_col_name, is_numeric_like_colnames, first_row_has_number
225
+ from table_stitcher.merger import (
226
+ tokenize,
227
+ normalize_col_name,
228
+ is_numeric_like_colnames,
229
+ first_row_has_number,
230
+ )
231
+
224
232
 
225
233
  class MyParserAdapter:
226
234
  def extract(self, doc: Any, cfg: MultiPageConfig) -> List[TableMeta]:
@@ -243,32 +251,32 @@ class MyParserAdapter:
243
251
  # 4. Tokenize first row (fallback similarity signal)
244
252
  first_row_tokens = set()
245
253
  if df.shape[0] > 0:
246
- first_row_tokens = tokenize(
247
- " ".join(str(x) for x in df.iloc[0].tolist())
248
- )
254
+ first_row_tokens = tokenize(" ".join(str(x) for x in df.iloc[0].tolist()))
249
255
 
250
256
  # 5. Classify: is_headerless, is_header_orphan, is_data_orphan
251
257
  raw_columns = [str(c) for c in df.columns]
252
- is_headerless = df.attrs.get('is_headerless', False)
253
-
254
- tables_meta.append(TableMeta(
255
- idx=idx,
256
- df=df,
257
- start_page=start_page,
258
- pages=pages,
259
- width=df.shape[1],
260
- header_tokens=header_tokens,
261
- first_row_tokens=first_row_tokens,
262
- raw_columns=raw_columns,
263
- vert_center=None, # Set if bbox available
264
- vert_top=None, # Normalized 0-1, 0=top of page
265
- vert_bottom=None, # Normalized 0-1, 1=bottom of page
266
- is_header_orphan=False, # True if headers-only, no/few data rows
267
- is_data_orphan=False, # True if data-only, no real headers
268
- numeric_like_cols=is_numeric_like_colnames(raw_columns),
269
- row_count=df.shape[0],
270
- is_headerless=is_headerless,
271
- ))
258
+ is_headerless = df.attrs.get("is_headerless", False)
259
+
260
+ tables_meta.append(
261
+ TableMeta(
262
+ idx=idx,
263
+ df=df,
264
+ start_page=start_page,
265
+ pages=pages,
266
+ width=df.shape[1],
267
+ header_tokens=header_tokens,
268
+ first_row_tokens=first_row_tokens,
269
+ raw_columns=raw_columns,
270
+ vert_center=None, # Set if bbox available
271
+ vert_top=None, # Normalized 0-1, 0=top of page
272
+ vert_bottom=None, # Normalized 0-1, 1=bottom of page
273
+ is_header_orphan=False, # True if headers-only, no/few data rows
274
+ is_data_orphan=False, # True if data-only, no real headers
275
+ numeric_like_cols=is_numeric_like_colnames(raw_columns),
276
+ row_count=df.shape[0],
277
+ is_headerless=is_headerless,
278
+ )
279
+ )
272
280
  return tables_meta
273
281
 
274
282
  def inject(self, doc: Any, logical_tables: List[LogicalTable]) -> Any:
@@ -287,6 +295,7 @@ class MyParserAdapter:
287
295
 
288
296
  return doc
289
297
 
298
+
290
299
  # Use it:
291
300
  stitcher = TableStitcher(adapter=MyParserAdapter())
292
301
  doc = stitcher.stitch(doc)
@@ -336,6 +345,7 @@ except StitchingError as e:
336
345
 
337
346
  ```python
338
347
  import logging
348
+
339
349
  logging.getLogger("table_stitcher").setLevel(logging.INFO)
340
350
  ```
341
351
 
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "table-stitcher"
7
- version = "0.4.2"
7
+ version = "0.4.4"
8
8
  description = "Reassemble tables split across page boundaries in PDF extraction"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -104,9 +104,9 @@ A few structural constants live at the top of `docling.py` rather than
104
104
  in `MultiPageConfig`:
105
105
 
106
106
  ```python
107
- _MAX_HEADER_CELL_LEN = 30 # header cells typically short; data cells longer
108
- _DATA_PATTERNS # regex list for "this cell is data, not header"
109
- _AUTO_COLNAME_RE # "Column_N" / "Unnamed: N" parser placeholders
107
+ _MAX_HEADER_CELL_LEN = 30 # header cells typically short; data cells longer
108
+ _DATA_PATTERNS # regex list for "this cell is data, not header"
109
+ _AUTO_COLNAME_RE # "Column_N" / "Unnamed: N" parser placeholders
110
110
  ```
111
111
 
112
112
  These are **adapter-intrinsic** — tuning them changes how the adapter
@@ -499,6 +499,42 @@ def _reemit_body_row(
499
499
  return grid_row, distinct
500
500
 
501
501
 
502
+ # Jaccard threshold for recognizing a body row as a reprinted continuation
503
+ # header. Combined with Docling's column_header flag — both must hold — so it
504
+ # stays conservative. Matches the merger's header_sim_strict default; the
505
+ # punctuation-agnostic tokenizer makes it tolerant of per-cell OCR drift such
506
+ # as "(S$)" vs "($$)".
507
+ _REPEATED_HEADER_SIM = 0.6
508
+
509
+
510
+ def _row_token_set(cells: list[TableCell]) -> set:
511
+ """Union of tokenized cell text for a grid row (duplicate span cells fold in)."""
512
+ toks: set = set()
513
+ for c in cells:
514
+ if c:
515
+ toks |= tokenize(getattr(c, "text", "") or "")
516
+ return toks
517
+
518
+
519
+ def _is_reprinted_header(orig_row: list[TableCell], header_sigs: list[set]) -> bool:
520
+ """True if ``orig_row`` is a reprinted continuation header to drop from the body.
521
+
522
+ Docling reprints the column header at the top of each continuation page; on
523
+ a multi-row header those rows survive the merge as bogus data rows. They are
524
+ dropped only when BOTH signals agree: Docling flagged the row
525
+ ``column_header`` AND it is a tokenized match for one of the reconstructed
526
+ header-block rows. The flag alone is unreliable — Docling over-flags
527
+ rowspan/continuation *data* rows as headers — so the content match guards
528
+ against deleting real data.
529
+ """
530
+ if not any(getattr(c, "column_header", False) for c in orig_row if c):
531
+ return False
532
+ toks = _row_token_set(orig_row)
533
+ if not toks:
534
+ return False
535
+ return any(jaccard(toks, sig) >= _REPEATED_HEADER_SIM for sig in header_sigs)
536
+
537
+
502
538
  def _dataframe_to_docling_data(
503
539
  df: pd.DataFrame,
504
540
  original_data: Optional[TableData] = None,
@@ -587,12 +623,15 @@ def _dataframe_to_docling_data(
587
623
 
588
624
  # --- Build data rows from merged DataFrame ---
589
625
  # Index member fragments' original body rows so spanning cells survive the
590
- # round-trip (see _index_member_body_rows).
626
+ # round-trip (see _index_member_rows).
591
627
  body_index = _index_member_rows(member_data) if member_data else {}
628
+ # Tokenized signatures of the reconstructed header block, for dropping
629
+ # reprinted continuation headers (see _is_reprinted_header).
630
+ header_sigs = [s for s in (_row_token_set(h) for h in orig_header_rows) if s]
592
631
 
593
- for i, (_, row) in enumerate(df.iterrows()):
594
- table_row_idx = num_header_rows + i
595
-
632
+ emitted = 0
633
+ for _, row in df.iterrows():
634
+ table_row_idx = num_header_rows + emitted
596
635
  row_vals = ["" if (pd.isna(v) or v is None) else str(v) for v in row]
597
636
 
598
637
  # Re-emit untouched rows from their original grid cells (preserves
@@ -601,9 +640,15 @@ def _dataframe_to_docling_data(
601
640
  bucket = body_index.get(tuple(row_vals))
602
641
  if bucket:
603
642
  orig_row = bucket.pop(0)
643
+ if header_sigs and _is_reprinted_header(orig_row, header_sigs):
644
+ # Reprinted header from a continuation page — already present as
645
+ # the header block; drop it instead of duplicating into the body.
646
+ log.debug("Dropped reprinted continuation header row from merged body.")
647
+ continue
604
648
  grid_row, distinct = _reemit_body_row(orig_row, table_row_idx, has_row_headers)
605
649
  grid.append(grid_row)
606
650
  table_cells.extend(distinct)
651
+ emitted += 1
607
652
  continue
608
653
 
609
654
  grid_row: list[TableCell] = []
@@ -625,8 +670,9 @@ def _dataframe_to_docling_data(
625
670
  table_cells.append(cell)
626
671
 
627
672
  grid.append(grid_row)
673
+ emitted += 1
628
674
 
629
- num_total_rows = num_header_rows + len(df)
675
+ num_total_rows = num_header_rows + emitted
630
676
 
631
677
  return TableData(num_rows=num_total_rows, num_cols=num_cols, table_cells=table_cells, grid=grid)
632
678
 
@@ -619,13 +619,34 @@ def _build_orphan_merged_table(
619
619
  )
620
620
 
621
621
 
622
+ def _dedupe_labels(labels: list[str]) -> list[str]:
623
+ """Make labels unique by suffixing repeats (``B, B`` -> ``B, B.1``)."""
624
+ used: set[str] = set()
625
+ counts: dict[str, int] = {}
626
+ out: list[str] = []
627
+ for label in labels:
628
+ candidate = label
629
+ while candidate in used:
630
+ counts[label] = counts.get(label, 0) + 1
631
+ candidate = f"{label}.{counts[label]}"
632
+ used.add(candidate)
633
+ out.append(candidate)
634
+ return out
635
+
636
+
622
637
  def _build_generic_merged_table(
623
638
  members: list[int], meta_by_idx: dict[int, TableMeta], cfg: MultiPageConfig
624
639
  ) -> tuple[pd.DataFrame, set[int], list[str]]:
625
640
  """Build merged table for the general case."""
626
641
  base = meta_by_idx[members[0]]
627
642
  merged_df = base.df.copy()
628
- canonical_cols = [str(c) for c in base.df.columns]
643
+ # Duplicate header labels are normal in the wild (rowspan/colspan
644
+ # parsers, 13F voting-authority triplets), but pd.concat cannot align
645
+ # frames on a non-unique column Index. Merge under deduped labels and
646
+ # restore the originals on the way out.
647
+ original_cols = [str(c) for c in base.df.columns]
648
+ canonical_cols = _dedupe_labels(original_cols)
649
+ merged_df.columns = canonical_cols
629
650
  merged_pages = set(base.pages)
630
651
  warnings: list[str] = []
631
652
  prev = base
@@ -648,6 +669,7 @@ def _build_generic_merged_table(
648
669
  merged_pages.update(m.pages)
649
670
  prev = m
650
671
 
672
+ merged_df.columns = original_cols + [str(c) for c in merged_df.columns[len(original_cols) :]]
651
673
  return merged_df, merged_pages, warnings
652
674
 
653
675
 
@@ -333,7 +333,15 @@ def assert_public_stitch_injects_docling_doc(
333
333
  ctx = f"public stitch for members={members}, pages={exp['pages']}"
334
334
 
335
335
  assert getattr(anchor.data, "num_rows", 0) > 0, f"{ctx}: anchor has no data"
336
- if "shape" in exp:
336
+ if "injected_rows" in exp:
337
+ # Exact injected row count. Used when injection legitimately differs
338
+ # from the parser-neutral shape — e.g. reprinted continuation-page
339
+ # headers are dropped from the body (they remain in the merged
340
+ # DataFrame but are not duplicated into the stitched document).
341
+ assert anchor.data.num_rows == exp["injected_rows"], (
342
+ f"{ctx}: injected rows {anchor.data.num_rows} != expected {exp['injected_rows']}"
343
+ )
344
+ elif "shape" in exp:
337
345
  # +1 or more for header rows; this guards that merged data was injected.
338
346
  assert anchor.data.num_rows >= exp["shape"][0] + 1, (
339
347
  f"{ctx}: anchor rows {anchor.data.num_rows} do not contain merged body"
@@ -37,6 +37,9 @@ logical_tables:
37
37
  shape:
38
38
  - 500
39
39
  - 3
40
+ # Injection drops the reprinted header block (title + column names) that the
41
+ # merged DataFrame still carries as its first two rows: 500 - 2 = header(2) + body(498).
42
+ injected_rows: 500
40
43
  columns:
41
44
  - Column_0
42
45
  - Column_1
@@ -73,6 +73,9 @@ logical_tables:
73
73
  shape:
74
74
  - 46
75
75
  - 6
76
+ # Injection drops the reprinted column header from the continuation page:
77
+ # 46 - 1 = header(1) + body(45).
78
+ injected_rows: 46
76
79
  columns:
77
80
  - Column_0
78
81
  - Column_1
@@ -23,6 +23,9 @@ logical_tables:
23
23
  shape:
24
24
  - 145
25
25
  - 3
26
+ # Injection drops the reprinted column header from the continuation pages:
27
+ # 145 - 1 = header(1) + body(144).
28
+ injected_rows: 145
26
29
  columns:
27
30
  - Contribution Type
28
31
  - Investment Name
@@ -388,6 +388,77 @@ class TestBodySpanPreservation:
388
388
  assert all(cell.col_span == 1 for cell in td.grid[1])
389
389
 
390
390
 
391
+ class TestReprintedHeaderDedup:
392
+ """Reprinted continuation-page headers are dropped from the injected body,
393
+ but column_header-flagged rows that don't match the header (Docling
394
+ over-flagging rowspan/continuation data) are kept.
395
+ """
396
+
397
+ @staticmethod
398
+ def _cell(text, r, c, *, header):
399
+ return TableCell(
400
+ text=text,
401
+ row_span=1,
402
+ col_span=1,
403
+ column_header=header,
404
+ row_header=False,
405
+ start_row_offset_idx=r,
406
+ end_row_offset_idx=r + 1,
407
+ start_col_offset_idx=c,
408
+ end_col_offset_idx=c + 1,
409
+ )
410
+
411
+ def _table(self, rows: list) -> TableData:
412
+ grid = []
413
+ flat = []
414
+ for r, (cells_text, is_header) in enumerate(rows):
415
+ grid_row = [self._cell(t, r, c, header=is_header) for c, t in enumerate(cells_text)]
416
+ grid.append(grid_row)
417
+ flat.extend(grid_row)
418
+ return TableData(num_rows=len(rows), num_cols=len(rows[0][0]), table_cells=flat, grid=grid)
419
+
420
+ def test_drops_reprinted_header_keeps_misflagged_data(self):
421
+ # Anchor: 1-row header "(S$)" + one data row.
422
+ anchor = self._table([(["SECTION", "LIMIT (S$)"], True), (["Death", "100"], False)])
423
+ # Continuation: header reprinted with OCR drift "($$)", a data row, and
424
+ # a row Docling wrongly flagged column_header (real data, distinct text).
425
+ satellite = self._table(
426
+ [
427
+ (["SECTION", "LIMIT ($$)"], True),
428
+ (["Injury", "50"], False),
429
+ (["Sub-limit per accident", "999"], True),
430
+ ]
431
+ )
432
+ # Merged DataFrame: the merger concatenates everything, including the
433
+ # reprinted header and the mis-flagged row.
434
+ merged_df = pd.DataFrame(
435
+ [
436
+ ["Death", "100"],
437
+ ["SECTION", "LIMIT ($$)"],
438
+ ["Injury", "50"],
439
+ ["Sub-limit per accident", "999"],
440
+ ],
441
+ columns=["Column_0", "Column_1"],
442
+ )
443
+
444
+ td = _dataframe_to_docling_data(
445
+ merged_df, original_data=anchor, member_data=[anchor, satellite]
446
+ )
447
+
448
+ body = [r for r in td.grid if not any(getattr(c, "column_header", False) for c in r if c)]
449
+ body_text = " ".join(str(c.text) for r in body for c in r if c)
450
+
451
+ # Reprinted header (drifted) dropped from the body...
452
+ assert "($$)" not in body_text
453
+ # ...but the header block still carries the anchor's "(S$)".
454
+ assert any("(S$)" in (c.text or "") for r in td.grid for c in r if c)
455
+ # Real data preserved, including the column_header-mis-flagged row.
456
+ assert "Death" in body_text and "Injury" in body_text
457
+ assert "Sub-limit per accident" in body_text and "999" in body_text
458
+ # 1 header row + 3 body rows (one reprinted header removed from 4).
459
+ assert td.num_rows == 4
460
+
461
+
391
462
  class TestAdapterProtocol:
392
463
  """Verify DoclingAdapter satisfies the protocol."""
393
464
 
@@ -228,6 +228,50 @@ class TestWidthOverflowPolicy:
228
228
  align_dataframe_to_header(df, ["A"], meta, cfg)
229
229
 
230
230
 
231
+ class TestDuplicateHeaderLabels:
232
+ def test_merge_survives_duplicate_header_labels(self):
233
+ """
234
+ Extracted headers repeating a label (e.g. 13F voting-authority
235
+ triplets emitting 'COLUMN 8' x3) must not crash the multipage
236
+ merge: pd.concat cannot align frames on a non-unique column Index.
237
+ """
238
+ cols = ["A", "B", "B"]
239
+ df1 = pd.DataFrame([["1", "2", "3"]], columns=cols)
240
+ df2 = pd.DataFrame([["4", "5", "6"]], columns=cols)
241
+ metas = [
242
+ _make_meta(idx=0, df=df1, start_page=1),
243
+ _make_meta(idx=1, df=df2, start_page=2),
244
+ ]
245
+ results = merge_multipage_tables(metas, MultiPageConfig())
246
+ assert len(results) == 1
247
+ assert results[0].df.shape == (2, 3)
248
+ assert results[0].df.iloc[1].tolist() == ["4", "5", "6"]
249
+ # Original (duplicated) labels are preserved in the output, matching
250
+ # how single-fragment tables pass through untouched.
251
+ assert list(results[0].df.columns) == cols
252
+
253
+ def test_duplicate_labels_with_wider_continuation(self):
254
+ cols = ["A", "B", "B"]
255
+ df1 = pd.DataFrame([["1", "2", "3"]], columns=cols)
256
+ df2 = pd.DataFrame([["4", "5", "6", "7"]], columns=cols + ["C"])
257
+ metas = [
258
+ _make_meta(idx=0, df=df1, start_page=1),
259
+ _make_meta(idx=1, df=df2, start_page=2),
260
+ ]
261
+ results = merge_multipage_tables(metas, MultiPageConfig())
262
+ assert len(results) == 1
263
+ assert results[0].df.shape == (2, 4)
264
+ assert results[0].df.iloc[1, 3] == "7"
265
+ assert list(results[0].df.columns)[:3] == cols
266
+
267
+ def test_dedupe_labels_avoids_existing_suffix_collision(self):
268
+ from table_stitcher.merger import _dedupe_labels
269
+
270
+ out = _dedupe_labels(["X", "X", "X.1"])
271
+ assert len(set(out)) == 3
272
+ assert out[0] == "X"
273
+
274
+
231
275
  class TestMergeTrace:
232
276
  def test_logical_table_explains_merge_reason_and_signals(self):
233
277
  df = pd.DataFrame({"Name": ["Alice"], "Age": ["30"]})
File without changes