graph-knowledge-doc-parser 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
- graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
- kg_doc_parser/__init__.py +9 -0
- kg_doc_parser/cast_hinting.py +19 -0
- kg_doc_parser/document_ingester_logger.py +766 -0
- kg_doc_parser/models.py +277 -0
- kg_doc_parser/ocr.py +752 -0
- kg_doc_parser/pdf2png.py +286 -0
- kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
- kg_doc_parser/text_processing_utils.py +30 -0
- kg_doc_parser/utils/__init__.py +0 -0
- kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
- kg_doc_parser/utils/file_loaders.py +405 -0
- kg_doc_parser/utils/langchain.py +220 -0
- kg_doc_parser/utils/log.py +135 -0
- kg_doc_parser/utils/version_chaining.py +1278 -0
- kg_doc_parser/workflow_ingest/__init__.py +187 -0
- kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
- kg_doc_parser/workflow_ingest/adapters.py +212 -0
- kg_doc_parser/workflow_ingest/cache.py +63 -0
- kg_doc_parser/workflow_ingest/cli.py +324 -0
- kg_doc_parser/workflow_ingest/clients.py +444 -0
- kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
- kg_doc_parser/workflow_ingest/design.py +208 -0
- kg_doc_parser/workflow_ingest/handlers.py +617 -0
- kg_doc_parser/workflow_ingest/models.py +575 -0
- kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
- kg_doc_parser/workflow_ingest/page_index.py +473 -0
- kg_doc_parser/workflow_ingest/parser_core.py +862 -0
- kg_doc_parser/workflow_ingest/parsing.py +249 -0
- kg_doc_parser/workflow_ingest/probe.py +164 -0
- kg_doc_parser/workflow_ingest/providers.py +412 -0
- kg_doc_parser/workflow_ingest/runners.py +546 -0
- kg_doc_parser/workflow_ingest/semantics.py +231 -0
- kg_doc_parser/workflow_ingest/service.py +112 -0
- kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
|
@@ -0,0 +1,1278 @@
|
|
|
1
|
+
import pathlib
|
|
2
|
+
import logging
|
|
3
|
+
logger = logging.getLogger(__name__)
|
|
4
|
+
logger.setLevel(logging.DEBUG)
|
|
5
|
+
|
|
6
|
+
import os
|
|
7
|
+
|
|
8
|
+
#import logging.handlers
|
|
9
|
+
#logger.addHandler(logging.handlers.RotatingFileHandler(os.path.join('.', 'logs', __name__)))
|
|
10
|
+
from .log import SQLiteHandler
|
|
11
|
+
sqlite_handler = SQLiteHandler(os.path.join('.','logs', 'application_logs.db'))
|
|
12
|
+
sqlite_handler.setLevel(logging.DEBUG)
|
|
13
|
+
logger.addHandler(sqlite_handler)
|
|
14
|
+
|
|
15
|
+
from pydantic import BaseModel, Field, ValidationError, model_validator
|
|
16
|
+
import uuid
|
|
17
|
+
|
|
18
|
+
import datetime
|
|
19
|
+
import hashlib
|
|
20
|
+
import dotenv
|
|
21
|
+
from typing import Literal, Optional, List, Dict, Any
|
|
22
|
+
from joblib import Memory
|
|
23
|
+
memory = Memory(location = "./.version_chain")
|
|
24
|
+
|
|
25
|
+
dotenv.load_dotenv()
|
|
26
|
+
|
|
27
|
+
import sqlite3
|
|
28
|
+
|
|
29
|
+
# ====== optional PDF -> PNG renderers ======
|
|
30
|
+
# we try pdf2image first, but fall back to PyMuPDF if needed
|
|
31
|
+
try:
|
|
32
|
+
from pdf2image import convert_from_path
|
|
33
|
+
_HAS_PDF2IMAGE = True
|
|
34
|
+
except Exception:
|
|
35
|
+
_HAS_PDF2IMAGE = False
|
|
36
|
+
try:
|
|
37
|
+
import fitz # PyMuPDF
|
|
38
|
+
_HAS_PYMUPDF = True
|
|
39
|
+
except Exception:
|
|
40
|
+
_HAS_PYMUPDF = False
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# ===========================================
|
|
44
|
+
# PDF → PNG → hash (no temp file, in-memory)
|
|
45
|
+
# ===========================================
|
|
46
|
+
|
|
47
|
+
def pdf_page_hashes_as_png(
|
|
48
|
+
file_path: str,
|
|
49
|
+
dpi: int = 300,
|
|
50
|
+
algo: str = "sha256",
|
|
51
|
+
img_format: str = "PNG",
|
|
52
|
+
) -> list[str]:
|
|
53
|
+
"""
|
|
54
|
+
Render each page to PNG in-memory and hash the bytes.
|
|
55
|
+
Returns list of hex digests in page order.
|
|
56
|
+
Includes no temp-file writes.
|
|
57
|
+
|
|
58
|
+
NOTE: this requires either pdf2image+poppler OR PyMuPDF.
|
|
59
|
+
"""
|
|
60
|
+
try:
|
|
61
|
+
if _HAS_PDF2IMAGE:
|
|
62
|
+
print(f"using pdf2image to convert file {file_path}")
|
|
63
|
+
images = convert_from_path(file_path, dpi=dpi)
|
|
64
|
+
out: list[str] = []
|
|
65
|
+
for i, img in enumerate(images):
|
|
66
|
+
print(f'page-{i}', end = ' ')
|
|
67
|
+
import io
|
|
68
|
+
buf = io.BytesIO()
|
|
69
|
+
img.save(buf, format=img_format)
|
|
70
|
+
data = buf.getvalue()
|
|
71
|
+
h = hashlib.new(algo)
|
|
72
|
+
h.update(data)
|
|
73
|
+
out.append(h.hexdigest())
|
|
74
|
+
return out
|
|
75
|
+
except Exception as e:
|
|
76
|
+
|
|
77
|
+
if _HAS_PYMUPDF:
|
|
78
|
+
print(f"using fitz/pymupdf to convert file {file_path}")
|
|
79
|
+
import fitz
|
|
80
|
+
doc = fitz.open(file_path)
|
|
81
|
+
out: list[str] = []
|
|
82
|
+
for i, page in enumerate(doc):
|
|
83
|
+
print(f'page-{i}', end = ' ')
|
|
84
|
+
# dpi → matrix
|
|
85
|
+
zoom = dpi / 72.0
|
|
86
|
+
mat = fitz.Matrix(zoom, zoom)
|
|
87
|
+
pix = page.get_pixmap(matrix=mat)
|
|
88
|
+
data = pix.tobytes("png")
|
|
89
|
+
h = hashlib.new(algo)
|
|
90
|
+
h.update(data)
|
|
91
|
+
out.append(h.hexdigest())
|
|
92
|
+
return out
|
|
93
|
+
raise e
|
|
94
|
+
|
|
95
|
+
raise RuntimeError(
|
|
96
|
+
"No PDF renderer available. Install `pdf2image` (plus poppler) or `PyMuPDF`."
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
# ============================================================
|
|
101
|
+
# smarter subsequence check: longer -> dict(hash -> [pages])
|
|
102
|
+
# ============================================================
|
|
103
|
+
|
|
104
|
+
from collections import defaultdict
|
|
105
|
+
from bisect import bisect_right
|
|
106
|
+
|
|
107
|
+
def build_pos_index(seq: list[str]) -> dict[str, list[int]]:
|
|
108
|
+
"""
|
|
109
|
+
seq[i] = hash_at_page_i -> index[h] = sorted list of page numbers
|
|
110
|
+
"""
|
|
111
|
+
idx: dict[str, list[int]] = defaultdict(list)
|
|
112
|
+
for i, h in enumerate(seq):
|
|
113
|
+
idx[h].append(i)
|
|
114
|
+
return idx
|
|
115
|
+
|
|
116
|
+
def is_ordered_subsequence(shorter: list[str], longer_idx: dict[str, list[int]]) -> bool:
|
|
117
|
+
"""
|
|
118
|
+
Check that every hash in `shorter` can be found in `longer_idx`
|
|
119
|
+
in strictly increasing page order. Gaps allowed.
|
|
120
|
+
"""
|
|
121
|
+
prev_pos = -1
|
|
122
|
+
for h in shorter:
|
|
123
|
+
positions = longer_idx.get(h)
|
|
124
|
+
if not positions:
|
|
125
|
+
return False
|
|
126
|
+
j = bisect_right(positions, prev_pos)
|
|
127
|
+
if j == len(positions):
|
|
128
|
+
return False
|
|
129
|
+
prev_pos = positions[j]
|
|
130
|
+
return True
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
# =========================
|
|
134
|
+
# SQLite-backed VersionChainDB class for persistent version chain management
|
|
135
|
+
# =========================
|
|
136
|
+
|
|
137
|
+
class VersionChainDB:
|
|
138
|
+
"""
|
|
139
|
+
SQLite-backed class for managing multiple version chains of PDF files.
|
|
140
|
+
Each chain is a linked list of nodes (PDF files) with metadata.
|
|
141
|
+
Supports CRUD, append, prepend, and insert-between operations.
|
|
142
|
+
"""
|
|
143
|
+
|
|
144
|
+
def __init__(self, db_path: str = "version_chains.db"):
|
|
145
|
+
self.db_path = db_path
|
|
146
|
+
self.conn = sqlite3.connect(self.db_path)
|
|
147
|
+
self._create_tables()
|
|
148
|
+
|
|
149
|
+
def _create_tables(self):
|
|
150
|
+
cur = self.conn.cursor()
|
|
151
|
+
cur.execute("""
|
|
152
|
+
CREATE TABLE IF NOT EXISTS chains (
|
|
153
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
154
|
+
name TEXT
|
|
155
|
+
);
|
|
156
|
+
""")
|
|
157
|
+
cur.execute("""
|
|
158
|
+
CREATE TABLE IF NOT EXISTS nodes (
|
|
159
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
160
|
+
chain_id INTEGER,
|
|
161
|
+
file_path TEXT,
|
|
162
|
+
file_size INTEGER,
|
|
163
|
+
file_hash TEXT,
|
|
164
|
+
prev_id INTEGER,
|
|
165
|
+
next_id INTEGER,
|
|
166
|
+
created_at TEXT,
|
|
167
|
+
metadata_json TEXT,
|
|
168
|
+
FOREIGN KEY(chain_id) REFERENCES chains(id),
|
|
169
|
+
FOREIGN KEY(prev_id) REFERENCES nodes(id),
|
|
170
|
+
FOREIGN KEY(next_id) REFERENCES nodes(id)
|
|
171
|
+
);
|
|
172
|
+
""")
|
|
173
|
+
cur.execute("""
|
|
174
|
+
CREATE TABLE IF NOT EXISTS duplicates (
|
|
175
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
176
|
+
file_name TEXT,
|
|
177
|
+
file_hash TEXT,
|
|
178
|
+
duplicate_of_file_name TEXT,
|
|
179
|
+
duplicate_of_file_hash TEXT,
|
|
180
|
+
chain_id INTEGER,
|
|
181
|
+
node_id INTEGER,
|
|
182
|
+
created_at TEXT
|
|
183
|
+
);
|
|
184
|
+
""")
|
|
185
|
+
|
|
186
|
+
# NEW: per-page PNG-hashes
|
|
187
|
+
# we store BOTH node_id (canonical) and file_path (so you can query by name)
|
|
188
|
+
cur.execute("""
|
|
189
|
+
CREATE TABLE IF NOT EXISTS page_hashes (
|
|
190
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
191
|
+
node_id INTEGER NOT NULL,
|
|
192
|
+
file_path TEXT NOT NULL,
|
|
193
|
+
page_num INTEGER NOT NULL,
|
|
194
|
+
page_hash TEXT NOT NULL,
|
|
195
|
+
render_dpi INTEGER NOT NULL DEFAULT 150,
|
|
196
|
+
render_format TEXT NOT NULL DEFAULT 'PNG',
|
|
197
|
+
render_algo TEXT NOT NULL DEFAULT 'sha256',
|
|
198
|
+
UNIQUE (node_id, page_num),
|
|
199
|
+
FOREIGN KEY(node_id) REFERENCES nodes(id)
|
|
200
|
+
);
|
|
201
|
+
""")
|
|
202
|
+
cur.execute("CREATE INDEX IF NOT EXISTS idx_page_hashes_node ON page_hashes(node_id);")
|
|
203
|
+
cur.execute("CREATE INDEX IF NOT EXISTS idx_page_hashes_hash ON page_hashes(page_hash);")
|
|
204
|
+
|
|
205
|
+
# optional: convenience view to see filename alongside page hashes
|
|
206
|
+
cur.execute("""
|
|
207
|
+
CREATE VIEW IF NOT EXISTS vw_page_hashes AS
|
|
208
|
+
SELECT
|
|
209
|
+
ph.id,
|
|
210
|
+
ph.node_id,
|
|
211
|
+
ph.file_path,
|
|
212
|
+
ph.page_num,
|
|
213
|
+
ph.page_hash,
|
|
214
|
+
ph.render_dpi,
|
|
215
|
+
ph.render_format,
|
|
216
|
+
ph.render_algo
|
|
217
|
+
FROM page_hashes ph;
|
|
218
|
+
""")
|
|
219
|
+
|
|
220
|
+
if not self.conn.in_transaction:
|
|
221
|
+
self.conn.commit()
|
|
222
|
+
def find_smaller_files_contained_in_this_sequence(self, node_id: int) -> list[dict]:
|
|
223
|
+
"""
|
|
224
|
+
We (node_id) are the *bigger* document (or at least, we might be).
|
|
225
|
+
For every other document:
|
|
226
|
+
- take its FIRST page hash
|
|
227
|
+
- if that hash appears in ANY page of *us*, try to match the whole smaller doc
|
|
228
|
+
"""
|
|
229
|
+
big_seq = self.get_page_hashes(node_id)
|
|
230
|
+
if not big_seq:
|
|
231
|
+
return []
|
|
232
|
+
|
|
233
|
+
big_len = len(big_seq)
|
|
234
|
+
|
|
235
|
+
# 1) index our own pages: hash -> [positions]
|
|
236
|
+
big_idx = build_pos_index(big_seq)
|
|
237
|
+
|
|
238
|
+
# 2) to reduce rows, we only want smaller docs whose first-page hash is
|
|
239
|
+
# one of OUR hashes.
|
|
240
|
+
big_hash_set = set(big_seq)
|
|
241
|
+
placeholders = ",".join("?" * len(big_hash_set))
|
|
242
|
+
|
|
243
|
+
cur = self.conn.cursor()
|
|
244
|
+
|
|
245
|
+
if placeholders:
|
|
246
|
+
cur.execute(
|
|
247
|
+
f"""
|
|
248
|
+
SELECT node_id, page_hash
|
|
249
|
+
FROM page_hashes
|
|
250
|
+
WHERE page_num = 0
|
|
251
|
+
AND node_id <> ?
|
|
252
|
+
AND page_hash IN ({placeholders})
|
|
253
|
+
""",
|
|
254
|
+
(node_id, *big_hash_set),
|
|
255
|
+
)
|
|
256
|
+
else:
|
|
257
|
+
# big doc somehow has no hashes? weird, just return
|
|
258
|
+
return []
|
|
259
|
+
|
|
260
|
+
first_pages = cur.fetchall()
|
|
261
|
+
|
|
262
|
+
# our own info
|
|
263
|
+
cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (node_id,))
|
|
264
|
+
this_row = cur.fetchone()
|
|
265
|
+
this_file_path = this_row[0] if this_row else None
|
|
266
|
+
this_file_hash = this_row[1] if this_row else None
|
|
267
|
+
this_chain_id = this_row[2] if this_row else None
|
|
268
|
+
|
|
269
|
+
results: list[dict] = []
|
|
270
|
+
|
|
271
|
+
for small_id, small_first_hash in first_pages:
|
|
272
|
+
# we already know small_first_hash is in our set,
|
|
273
|
+
# but it can appear multiple times (same page repeated), so:
|
|
274
|
+
positions = big_idx.get(small_first_hash, [])
|
|
275
|
+
if not positions:
|
|
276
|
+
continue
|
|
277
|
+
|
|
278
|
+
# fetch full smaller sequence
|
|
279
|
+
small_seq = self.get_page_hashes(small_id)
|
|
280
|
+
small_len = len(small_seq)
|
|
281
|
+
if not small_seq:
|
|
282
|
+
continue
|
|
283
|
+
|
|
284
|
+
matched = False
|
|
285
|
+
for pos in positions:
|
|
286
|
+
# can we fit the smaller starting at this pos?
|
|
287
|
+
if pos + small_len > big_len:
|
|
288
|
+
continue
|
|
289
|
+
if big_seq[pos:pos + small_len] == small_seq:
|
|
290
|
+
matched = True
|
|
291
|
+
break
|
|
292
|
+
|
|
293
|
+
if matched:
|
|
294
|
+
# hydrate smaller
|
|
295
|
+
cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (small_id,))
|
|
296
|
+
small_row = cur.fetchone()
|
|
297
|
+
results.append({
|
|
298
|
+
"small_node_id": small_id,
|
|
299
|
+
"small_file_path": small_row[0] if small_row else None,
|
|
300
|
+
"small_file_hash": small_row[1] if small_row else None,
|
|
301
|
+
"small_chain_id": small_row[2] if small_row else None,
|
|
302
|
+
"big_node_id": node_id,
|
|
303
|
+
"big_file_path": this_file_path,
|
|
304
|
+
"big_file_hash": this_file_hash,
|
|
305
|
+
"big_chain_id": this_chain_id,
|
|
306
|
+
})
|
|
307
|
+
|
|
308
|
+
return results
|
|
309
|
+
def get_nodes_with_pagecount_at_most(self, n_pages: int, exclude_node_id: int) -> list[int]:
|
|
310
|
+
"""
|
|
311
|
+
Return node_ids whose page_count <= n_pages, excluding the given node.
|
|
312
|
+
Useful when *this* node is longer and we want to see if we contain shorter ones.
|
|
313
|
+
"""
|
|
314
|
+
cur = self.conn.cursor()
|
|
315
|
+
cur.execute("""
|
|
316
|
+
SELECT ph.node_id, COUNT(*) AS c
|
|
317
|
+
FROM page_hashes ph
|
|
318
|
+
GROUP BY ph.node_id
|
|
319
|
+
HAVING c <= ?
|
|
320
|
+
""", (n_pages,))
|
|
321
|
+
out: list[int] = []
|
|
322
|
+
for node_id, c in cur.fetchall():
|
|
323
|
+
if node_id != exclude_node_id:
|
|
324
|
+
out.append(node_id)
|
|
325
|
+
return out
|
|
326
|
+
def get_canonical_files(self) -> list[dict]:
|
|
327
|
+
cur = self.conn.cursor()
|
|
328
|
+
cur.execute("""
|
|
329
|
+
SELECT n.id, n.file_path, n.file_hash, n.file_size, n.chain_id, n.created_at
|
|
330
|
+
FROM nodes AS n
|
|
331
|
+
WHERE n.file_path NOT IN (
|
|
332
|
+
SELECT d.file_name
|
|
333
|
+
FROM duplicates AS d
|
|
334
|
+
WHERE d.file_name IS NOT NULL
|
|
335
|
+
)
|
|
336
|
+
ORDER BY n.id
|
|
337
|
+
""")
|
|
338
|
+
rows = cur.fetchall()
|
|
339
|
+
return [
|
|
340
|
+
{
|
|
341
|
+
"id": r[0],
|
|
342
|
+
"file_path": r[1],
|
|
343
|
+
"file_hash": r[2],
|
|
344
|
+
"file_size": r[3],
|
|
345
|
+
"chain_id": r[4],
|
|
346
|
+
"created_at": r[5],
|
|
347
|
+
}
|
|
348
|
+
for r in rows
|
|
349
|
+
]
|
|
350
|
+
# --------------------------
|
|
351
|
+
# page-hash helpers
|
|
352
|
+
# --------------------------
|
|
353
|
+
|
|
354
|
+
def insert_page_hashes(
|
|
355
|
+
self,
|
|
356
|
+
node_id: int,
|
|
357
|
+
file_path: str,
|
|
358
|
+
page_hashes: list[str],
|
|
359
|
+
dpi: int = 300,
|
|
360
|
+
render_format: str = "PNG",
|
|
361
|
+
algo: str = "sha256",
|
|
362
|
+
):
|
|
363
|
+
"""
|
|
364
|
+
Store the ordered page-hash sequence for a node, replacing old ones if any.
|
|
365
|
+
"""
|
|
366
|
+
cur = self.conn.cursor()
|
|
367
|
+
cur.execute("DELETE FROM page_hashes WHERE node_id = ?", (node_id,))
|
|
368
|
+
cur.executemany(
|
|
369
|
+
"""
|
|
370
|
+
INSERT INTO page_hashes
|
|
371
|
+
(node_id, file_path, page_num, page_hash, render_dpi, render_format, render_algo)
|
|
372
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
373
|
+
""",
|
|
374
|
+
[
|
|
375
|
+
(node_id, file_path, i, h, dpi, render_format, algo)
|
|
376
|
+
for i, h in enumerate(page_hashes)
|
|
377
|
+
],
|
|
378
|
+
)
|
|
379
|
+
if not self.conn.in_transaction:
|
|
380
|
+
self.conn.commit()
|
|
381
|
+
|
|
382
|
+
def get_page_hashes(self, node_id: int) -> list[str]:
|
|
383
|
+
cur = self.conn.cursor()
|
|
384
|
+
cur.execute("""
|
|
385
|
+
SELECT page_hash
|
|
386
|
+
FROM page_hashes
|
|
387
|
+
WHERE node_id = ?
|
|
388
|
+
ORDER BY page_num ASC
|
|
389
|
+
""", (node_id,))
|
|
390
|
+
return [r[0] for r in cur.fetchall()]
|
|
391
|
+
|
|
392
|
+
def get_nodes_with_pagecount_at_least(self, n_pages: int, exclude_node_id: int) -> list[int]:
|
|
393
|
+
"""
|
|
394
|
+
Return node_ids whose page_count >= n_pages, excluding the given node.
|
|
395
|
+
"""
|
|
396
|
+
cur = self.conn.cursor()
|
|
397
|
+
cur.execute("""
|
|
398
|
+
SELECT ph.node_id, COUNT(*) AS c
|
|
399
|
+
FROM page_hashes ph
|
|
400
|
+
GROUP BY ph.node_id
|
|
401
|
+
HAVING c >= ?
|
|
402
|
+
""", (n_pages,))
|
|
403
|
+
out: list[int] = []
|
|
404
|
+
for node_id, c in cur.fetchall():
|
|
405
|
+
if node_id != exclude_node_id:
|
|
406
|
+
out.append(node_id)
|
|
407
|
+
return out
|
|
408
|
+
|
|
409
|
+
def find_bigger_files_containing_this_sequence(self, node_id: int) -> list[dict]:
|
|
410
|
+
"""
|
|
411
|
+
We (node_id) are the *smaller* candidate.
|
|
412
|
+
Look for ANY other file that has our FIRST page-hash somewhere (any page),
|
|
413
|
+
and also has enough pages after that to hold our whole sequence.
|
|
414
|
+
Then confirm with a Python slice compare.
|
|
415
|
+
"""
|
|
416
|
+
small_seq = self.get_page_hashes(node_id)
|
|
417
|
+
if not small_seq:
|
|
418
|
+
return []
|
|
419
|
+
|
|
420
|
+
first_hash = small_seq[0]
|
|
421
|
+
small_len = len(small_seq)
|
|
422
|
+
|
|
423
|
+
cur = self.conn.cursor()
|
|
424
|
+
# 1) SQL prune:
|
|
425
|
+
# - find ANY page (not just page 0) whose hash == our first page
|
|
426
|
+
# - make sure from that page to the end there are >= small_len pages
|
|
427
|
+
cur.execute(
|
|
428
|
+
"""
|
|
429
|
+
WITH maxp AS (
|
|
430
|
+
SELECT node_id, MAX(page_num) AS max_page
|
|
431
|
+
FROM page_hashes
|
|
432
|
+
GROUP BY node_id
|
|
433
|
+
)
|
|
434
|
+
SELECT ph.node_id, ph.page_num, maxp.max_page
|
|
435
|
+
FROM page_hashes AS ph
|
|
436
|
+
JOIN maxp ON ph.node_id = maxp.node_id
|
|
437
|
+
WHERE ph.page_hash = ?
|
|
438
|
+
AND ph.node_id <> ?
|
|
439
|
+
AND (ph.page_num + ?) <= (maxp.max_page + 1)
|
|
440
|
+
""",
|
|
441
|
+
(first_hash, node_id, small_len),
|
|
442
|
+
)
|
|
443
|
+
candidates = cur.fetchall()
|
|
444
|
+
|
|
445
|
+
# our own info
|
|
446
|
+
cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (node_id,))
|
|
447
|
+
this_row = cur.fetchone()
|
|
448
|
+
this_file_path = this_row[0] if this_row else None
|
|
449
|
+
this_file_hash = this_row[1] if this_row else None
|
|
450
|
+
this_chain_id = this_row[2] if this_row else None
|
|
451
|
+
|
|
452
|
+
results: list[dict] = []
|
|
453
|
+
|
|
454
|
+
for cand_id, start_page, max_page in candidates:
|
|
455
|
+
big_seq = self.get_page_hashes(cand_id)
|
|
456
|
+
# we already know indexing won't go out of range because of SQL condition
|
|
457
|
+
if big_seq[start_page:start_page + small_len] == small_seq:
|
|
458
|
+
# hydrate bigger file
|
|
459
|
+
cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (cand_id,))
|
|
460
|
+
big_row = cur.fetchone()
|
|
461
|
+
results.append({
|
|
462
|
+
"small_node_id": node_id,
|
|
463
|
+
"small_file_path": this_file_path,
|
|
464
|
+
"small_file_hash": this_file_hash,
|
|
465
|
+
"small_chain_id": this_chain_id,
|
|
466
|
+
"big_node_id": cand_id,
|
|
467
|
+
"big_file_path": big_row[0] if big_row else None,
|
|
468
|
+
"big_file_hash": big_row[1] if big_row else None,
|
|
469
|
+
"big_chain_id": big_row[2] if big_row else None,
|
|
470
|
+
})
|
|
471
|
+
|
|
472
|
+
return results
|
|
473
|
+
|
|
474
|
+
def mark_subdocument_duplicate(self, info: dict):
|
|
475
|
+
"""
|
|
476
|
+
info must contain:
|
|
477
|
+
small_file_path, small_file_hash, big_file_path, big_file_hash,
|
|
478
|
+
small_chain_id, small_node_id
|
|
479
|
+
We'll just reuse your existing duplicates table.
|
|
480
|
+
"""
|
|
481
|
+
self.insert_duplicate(
|
|
482
|
+
file_name=info["small_file_path"],
|
|
483
|
+
file_hash=info["small_file_hash"],
|
|
484
|
+
duplicate_of_file_name=info["big_file_path"],
|
|
485
|
+
duplicate_of_file_hash=info["big_file_hash"],
|
|
486
|
+
chain_id=info["small_chain_id"],
|
|
487
|
+
node_id=info["small_node_id"],
|
|
488
|
+
)
|
|
489
|
+
|
|
490
|
+
# --------------------------
|
|
491
|
+
|
|
492
|
+
def create_chain(self, name: Optional[str] = None) -> int:
|
|
493
|
+
cur = self.conn.cursor()
|
|
494
|
+
cur.execute("INSERT INTO chains (name) VALUES (?)", (name,))
|
|
495
|
+
if not self.conn.in_transaction:
|
|
496
|
+
self.conn.commit()
|
|
497
|
+
return cur.lastrowid
|
|
498
|
+
|
|
499
|
+
def delete_chain(self, chain_id: int):
|
|
500
|
+
cur = self.conn.cursor()
|
|
501
|
+
cur.execute("DELETE FROM nodes WHERE chain_id = ?", (chain_id,))
|
|
502
|
+
cur.execute("DELETE FROM chains WHERE id = ?", (chain_id,))
|
|
503
|
+
if not self.conn.in_transaction:
|
|
504
|
+
self.conn.commit()
|
|
505
|
+
|
|
506
|
+
def add_node(self, chain_id: int, file_root: str, file_path: str, file_size: int, file_hash: str,
|
|
507
|
+
position: str = "append", ref_node_id: Optional[int] = None,
|
|
508
|
+
metadata_json: Optional[str] = None) -> int:
|
|
509
|
+
"""
|
|
510
|
+
ref_node_id : between and append is the node before the new addition, preprend is the node id prepended to
|
|
511
|
+
"""
|
|
512
|
+
if not file_path.lower().endswith('.pdf'):
|
|
513
|
+
raise ValueError("Only PDF files (.pdf) are allowed.")
|
|
514
|
+
cur = self.conn.cursor()
|
|
515
|
+
created_at = datetime.datetime.now().isoformat()
|
|
516
|
+
# Find head/tail for prepend/append
|
|
517
|
+
if position == "prepend":
|
|
518
|
+
cur.execute("SELECT id FROM nodes WHERE chain_id = ? AND prev_id IS NULL", (chain_id,))
|
|
519
|
+
head = cur.fetchone()
|
|
520
|
+
prev_id = None
|
|
521
|
+
next_id = head[0] if head else None
|
|
522
|
+
# Update old head's prev_id
|
|
523
|
+
if head:
|
|
524
|
+
cur.execute("UPDATE nodes SET prev_id = NULL WHERE id = ?", (head[0],))
|
|
525
|
+
elif position == "append":
|
|
526
|
+
cur.execute("SELECT id FROM nodes WHERE chain_id = ? AND next_id IS NULL", (chain_id,))
|
|
527
|
+
tail = cur.fetchone()
|
|
528
|
+
prev_id = tail[0] if tail else None
|
|
529
|
+
next_id = None
|
|
530
|
+
# Update old tail's next_id
|
|
531
|
+
if tail:
|
|
532
|
+
cur.execute("UPDATE nodes SET next_id = NULL WHERE id = ?", (tail[0],))
|
|
533
|
+
elif position == "between":
|
|
534
|
+
if ref_node_id is None:
|
|
535
|
+
raise ValueError("ref_node_id must be provided for 'between' insertion.")
|
|
536
|
+
# Insert after ref_node_id
|
|
537
|
+
cur.execute("SELECT next_id FROM nodes WHERE id = ?", (ref_node_id,))
|
|
538
|
+
next_id = cur.fetchone()
|
|
539
|
+
next_id = next_id[0] if next_id else None
|
|
540
|
+
prev_id = ref_node_id
|
|
541
|
+
# Update links
|
|
542
|
+
cur.execute("UPDATE nodes SET next_id = NULL WHERE id = ?", (ref_node_id,))
|
|
543
|
+
if next_id:
|
|
544
|
+
cur.execute("UPDATE nodes SET prev_id = NULL WHERE id = ?", (next_id,))
|
|
545
|
+
else:
|
|
546
|
+
raise ValueError("position must be 'append', 'prepend', or 'between'.")
|
|
547
|
+
cur.execute("""
|
|
548
|
+
INSERT INTO nodes (chain_id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json)
|
|
549
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
550
|
+
""", (chain_id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json))
|
|
551
|
+
node_id = cur.lastrowid
|
|
552
|
+
# Update neighbors
|
|
553
|
+
if position == "prepend" and next_id:
|
|
554
|
+
cur.execute("UPDATE nodes SET prev_id = ? WHERE id = ?", (node_id, next_id))
|
|
555
|
+
if position == "append" and prev_id:
|
|
556
|
+
cur.execute("UPDATE nodes SET next_id = ? WHERE id = ?", (node_id, prev_id))
|
|
557
|
+
if position == "between":
|
|
558
|
+
cur.execute("UPDATE nodes SET next_id = ? WHERE id = ?", (node_id, prev_id))
|
|
559
|
+
if next_id:
|
|
560
|
+
cur.execute("UPDATE nodes SET prev_id = ? WHERE id = ?", (node_id, next_id))
|
|
561
|
+
if not self.conn.in_transaction:
|
|
562
|
+
self.conn.commit()
|
|
563
|
+
|
|
564
|
+
try:
|
|
565
|
+
page_hashes = pdf_page_hashes_as_png(os.path.join(file_root, file_path))
|
|
566
|
+
self.insert_page_hashes(
|
|
567
|
+
node_id=node_id,
|
|
568
|
+
file_path=file_path,
|
|
569
|
+
page_hashes=page_hashes,
|
|
570
|
+
dpi=150,
|
|
571
|
+
render_format="PNG",
|
|
572
|
+
algo="sha256",
|
|
573
|
+
)
|
|
574
|
+
|
|
575
|
+
# 1) I am short -> check longer ones
|
|
576
|
+
longer_matches = self.find_bigger_files_containing_this_sequence(node_id)
|
|
577
|
+
for info in longer_matches:
|
|
578
|
+
self.mark_subdocument_duplicate(info)
|
|
579
|
+
|
|
580
|
+
# 2) I am long -> check shorter ones
|
|
581
|
+
shorter_matches = self.find_smaller_files_contained_in_this_sequence(node_id)
|
|
582
|
+
for info in shorter_matches:
|
|
583
|
+
self.mark_subdocument_duplicate(info)
|
|
584
|
+
|
|
585
|
+
except Exception as e:
|
|
586
|
+
logger.exception(f"Error computing/storing page hashes for {file_path}: {e}")
|
|
587
|
+
self.conn.rollback()
|
|
588
|
+
raise e
|
|
589
|
+
|
|
590
|
+
return node_id
|
|
591
|
+
|
|
592
|
+
def list_all_chains(self):
|
|
593
|
+
return [self.get_chain(chain['id']) for chain in self.find_chains()]
|
|
594
|
+
|
|
595
|
+
def get_chain(self, chain_id: int) -> List[Dict[str, Any]]:
|
|
596
|
+
cur = self.conn.cursor()
|
|
597
|
+
# Find head node
|
|
598
|
+
cur.execute("SELECT id FROM nodes WHERE chain_id = ? AND prev_id IS NULL", (chain_id,))
|
|
599
|
+
head = cur.fetchone()
|
|
600
|
+
if not head:
|
|
601
|
+
return []
|
|
602
|
+
node_id = head[0]
|
|
603
|
+
chain = []
|
|
604
|
+
while node_id:
|
|
605
|
+
cur.execute("SELECT id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json FROM nodes WHERE id = ?", (node_id,))
|
|
606
|
+
row = cur.fetchone()
|
|
607
|
+
if not row:
|
|
608
|
+
break
|
|
609
|
+
node = {
|
|
610
|
+
"id": row[0],
|
|
611
|
+
"file_path": row[1],
|
|
612
|
+
"file_size": row[2],
|
|
613
|
+
"file_hash": row[3],
|
|
614
|
+
"prev_id": row[4],
|
|
615
|
+
"next_id": row[5],
|
|
616
|
+
"created_at": row[6],
|
|
617
|
+
"metadata_json": row[7]
|
|
618
|
+
}
|
|
619
|
+
chain.append(node)
|
|
620
|
+
node_id = row[5] # next_id
|
|
621
|
+
return chain
|
|
622
|
+
|
|
623
|
+
def update_node(self, node_id: int, **fields):
|
|
624
|
+
cur = self.conn.cursor()
|
|
625
|
+
allowed = {"file_path", "file_size", "file_hash", "metadata_json"}
|
|
626
|
+
updates = []
|
|
627
|
+
values = []
|
|
628
|
+
for k, v in fields.items():
|
|
629
|
+
if k in allowed:
|
|
630
|
+
updates.append(f"{k} = ?")
|
|
631
|
+
values.append(v)
|
|
632
|
+
if not updates:
|
|
633
|
+
return
|
|
634
|
+
values.append(node_id)
|
|
635
|
+
cur.execute(f"UPDATE nodes SET {', '.join(updates)} WHERE id = ?", values)
|
|
636
|
+
if not self.conn.in_transaction:
|
|
637
|
+
self.conn.commit()
|
|
638
|
+
|
|
639
|
+
def delete_node(self, node_id: int):
|
|
640
|
+
cur = self.conn.cursor()
|
|
641
|
+
# Relink neighbors
|
|
642
|
+
cur.execute("SELECT prev_id, next_id FROM nodes WHERE id = ?", (node_id,))
|
|
643
|
+
row = cur.fetchone()
|
|
644
|
+
if row:
|
|
645
|
+
prev_id, next_id = row
|
|
646
|
+
if prev_id:
|
|
647
|
+
cur.execute("UPDATE nodes SET next_id = ? WHERE id = ?", (next_id, prev_id))
|
|
648
|
+
if next_id:
|
|
649
|
+
cur.execute("UPDATE nodes SET prev_id = ? WHERE id = ?", (prev_id, next_id))
|
|
650
|
+
cur.execute("DELETE FROM nodes WHERE id = ?", (node_id,))
|
|
651
|
+
# also delete page-hashes for this node
|
|
652
|
+
cur.execute("DELETE FROM page_hashes WHERE node_id = ?", (node_id,))
|
|
653
|
+
if not self.conn.in_transaction:
|
|
654
|
+
self.conn.commit()
|
|
655
|
+
|
|
656
|
+
def find_chains(self) -> List[Dict[str, Any]]:
|
|
657
|
+
cur = self.conn.cursor()
|
|
658
|
+
cur.execute("SELECT id, name FROM chains")
|
|
659
|
+
return [{"id": row[0], "name": row[1]} for row in cur.fetchall()]
|
|
660
|
+
|
|
661
|
+
def find_nodes(self, chain_id: int) -> List[Dict[str, Any]]:
|
|
662
|
+
cur = self.conn.cursor()
|
|
663
|
+
cur.execute("SELECT id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json FROM nodes WHERE chain_id = ?", (chain_id,))
|
|
664
|
+
return [
|
|
665
|
+
{
|
|
666
|
+
"id": row[0],
|
|
667
|
+
"file_path": row[1],
|
|
668
|
+
"file_size": row[2],
|
|
669
|
+
"file_hash": row[3],
|
|
670
|
+
"prev_id": row[4],
|
|
671
|
+
"next_id": row[5],
|
|
672
|
+
"created_at": row[6],
|
|
673
|
+
"metadata_json": row[7]
|
|
674
|
+
}
|
|
675
|
+
for row in cur.fetchall()
|
|
676
|
+
]
|
|
677
|
+
def get_canonical_for_hash(self, file_hash: str):
|
|
678
|
+
"""
|
|
679
|
+
Return (id, file_path, chain_id) of the earliest node we have for this hash.
|
|
680
|
+
"""
|
|
681
|
+
cur = self.conn.cursor()
|
|
682
|
+
cur.execute(
|
|
683
|
+
"""
|
|
684
|
+
SELECT id, file_path, chain_id
|
|
685
|
+
FROM nodes
|
|
686
|
+
WHERE file_hash = ?
|
|
687
|
+
ORDER BY id ASC
|
|
688
|
+
LIMIT 1
|
|
689
|
+
""",
|
|
690
|
+
(file_hash,),
|
|
691
|
+
)
|
|
692
|
+
return cur.fetchone()
|
|
693
|
+
def insert_duplicate(self, file_name, file_hash, duplicate_of_file_name, duplicate_of_file_hash, chain_id=None, node_id=None):
|
|
694
|
+
cur = self.conn.cursor()
|
|
695
|
+
created_at = datetime.datetime.now().isoformat()
|
|
696
|
+
cur.execute("""
|
|
697
|
+
INSERT INTO duplicates (file_name, file_hash, duplicate_of_file_name, duplicate_of_file_hash, chain_id, node_id, created_at)
|
|
698
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
699
|
+
""", (file_name, file_hash, duplicate_of_file_name, duplicate_of_file_hash, chain_id, node_id, created_at))
|
|
700
|
+
if not self.conn.in_transaction:
|
|
701
|
+
self.conn.commit()
|
|
702
|
+
|
|
703
|
+
def find_duplicate_by_name(self, file_name):
|
|
704
|
+
cur = self.conn.cursor()
|
|
705
|
+
cur.execute("SELECT * FROM duplicates WHERE file_name = ?", (file_name,))
|
|
706
|
+
return cur.fetchall()
|
|
707
|
+
|
|
708
|
+
def find_duplicate_by_hash(self, file_hash):
|
|
709
|
+
cur = self.conn.cursor()
|
|
710
|
+
cur.execute("SELECT * FROM duplicates WHERE file_hash = ?", (file_hash,))
|
|
711
|
+
return cur.fetchall()
|
|
712
|
+
def get_canonical_page_statistics(self) -> dict:
|
|
713
|
+
"""
|
|
714
|
+
Compute statistics on canonical (non-duplicate) documents.
|
|
715
|
+
|
|
716
|
+
Returns:
|
|
717
|
+
dict with:
|
|
718
|
+
- total_canonical_docs (int)
|
|
719
|
+
- total_pages (int)
|
|
720
|
+
- details (list of dict) → each with {file_path, page_count}
|
|
721
|
+
|
|
722
|
+
Canonical = nodes whose file_path NOT in duplicates.file_name.
|
|
723
|
+
"""
|
|
724
|
+
cur = self.conn.cursor()
|
|
725
|
+
cur.execute("""
|
|
726
|
+
SELECT
|
|
727
|
+
n.file_path,
|
|
728
|
+
COUNT(ph.page_num) AS page_count
|
|
729
|
+
FROM nodes AS n
|
|
730
|
+
JOIN page_hashes AS ph
|
|
731
|
+
ON ph.node_id = n.id
|
|
732
|
+
WHERE n.file_path NOT IN (
|
|
733
|
+
SELECT d.file_name FROM duplicates AS d
|
|
734
|
+
WHERE d.file_name IS NOT NULL
|
|
735
|
+
)
|
|
736
|
+
GROUP BY n.id
|
|
737
|
+
ORDER BY page_count DESC;
|
|
738
|
+
""")
|
|
739
|
+
rows = cur.fetchall()
|
|
740
|
+
|
|
741
|
+
stats = {
|
|
742
|
+
"total_canonical_docs": len(rows),
|
|
743
|
+
"total_pages": sum(r[1] for r in rows),
|
|
744
|
+
"details": [{"file_path": r[0], "page_count": r[1]} for r in rows]
|
|
745
|
+
}
|
|
746
|
+
return stats
|
|
747
|
+
def name_exists(self, file_name):
|
|
748
|
+
cur = self.conn.cursor()
|
|
749
|
+
cur.execute("SELECT 1 FROM nodes WHERE file_path = ? LIMIT 1", (file_name,))
|
|
750
|
+
if cur.fetchone():
|
|
751
|
+
print('match from nodes.file_path')
|
|
752
|
+
return True
|
|
753
|
+
cur.execute("SELECT 1 FROM duplicates WHERE file_name = ? LIMIT 1", (file_name,))
|
|
754
|
+
if cur.fetchone():
|
|
755
|
+
print('match from duplicates.file_name')
|
|
756
|
+
return True
|
|
757
|
+
return False
|
|
758
|
+
|
|
759
|
+
def hash_exists(self, file_hash):
|
|
760
|
+
cur = self.conn.cursor()
|
|
761
|
+
cur.execute("SELECT 1 FROM nodes WHERE file_hash = ? LIMIT 1", (file_hash,))
|
|
762
|
+
if cur.fetchone():
|
|
763
|
+
return True
|
|
764
|
+
cur.execute("SELECT 1 FROM duplicates WHERE file_hash = ? LIMIT 1", (file_hash,))
|
|
765
|
+
if cur.fetchone():
|
|
766
|
+
return True
|
|
767
|
+
return False
|
|
768
|
+
|
|
769
|
+
def is_duplicate_name_or_hash(self, file_name, file_hash):
|
|
770
|
+
return self.name_exists(file_name) or self.hash_exists(file_hash)
|
|
771
|
+
|
|
772
|
+
def close(self):
|
|
773
|
+
self.conn.close()
|
|
774
|
+
def is_canonical_by_name(self, file_name: str) -> bool:
|
|
775
|
+
"""
|
|
776
|
+
Decide if this file_path is the canonical/kept one.
|
|
777
|
+
|
|
778
|
+
Rules (based on your current usage):
|
|
779
|
+
- if this file_name appears in duplicates.file_name -> NOT canonical
|
|
780
|
+
(because you always put the thing-to-hide on the left)
|
|
781
|
+
- else -> canonical
|
|
782
|
+
- if we also want to be careful with old rows, we can fallback to hash
|
|
783
|
+
"""
|
|
784
|
+
cur = self.conn.cursor()
|
|
785
|
+
|
|
786
|
+
# 1) if it's explicitly marked as a duplicate, it's not canonical
|
|
787
|
+
cur.execute("SELECT 1 FROM duplicates WHERE file_name = ? LIMIT 1", (file_name,))
|
|
788
|
+
if cur.fetchone():
|
|
789
|
+
return False
|
|
790
|
+
|
|
791
|
+
# 2) try to get its hash from nodes
|
|
792
|
+
cur.execute("""
|
|
793
|
+
SELECT file_hash
|
|
794
|
+
FROM nodes
|
|
795
|
+
WHERE file_path = ?
|
|
796
|
+
LIMIT 1
|
|
797
|
+
""", (file_name,))
|
|
798
|
+
row = cur.fetchone()
|
|
799
|
+
if row:
|
|
800
|
+
file_hash = row[0]
|
|
801
|
+
else:
|
|
802
|
+
# maybe it only lives in duplicates table (rare, but let's check)
|
|
803
|
+
cur.execute("""
|
|
804
|
+
SELECT file_hash
|
|
805
|
+
FROM duplicates
|
|
806
|
+
WHERE file_name = ?
|
|
807
|
+
LIMIT 1
|
|
808
|
+
""", (file_name,))
|
|
809
|
+
row = cur.fetchone()
|
|
810
|
+
if row:
|
|
811
|
+
file_hash = row[0]
|
|
812
|
+
else:
|
|
813
|
+
# not in nodes, not in duplicates: we don't know it -> treat as canonical/new
|
|
814
|
+
return True
|
|
815
|
+
|
|
816
|
+
# 3) (optional) if you want to be extra safe:
|
|
817
|
+
# check if there is an entry "some other file" -> this hash
|
|
818
|
+
# but because your direction is always "duplicate file_name -> canonical duplicate_of_file_name",
|
|
819
|
+
# step (1) is usually enough.
|
|
820
|
+
return True
|
|
821
|
+
def get_non_duplicate_files(self) -> list[dict]:
|
|
822
|
+
"""
|
|
823
|
+
Return all nodes that are not listed as duplicates.
|
|
824
|
+
"""
|
|
825
|
+
cur = self.conn.cursor()
|
|
826
|
+
cur.execute("""
|
|
827
|
+
SELECT n.id, n.file_path, n.file_hash, n.chain_id, n.created_at
|
|
828
|
+
FROM nodes AS n
|
|
829
|
+
WHERE n.file_hash NOT IN (
|
|
830
|
+
SELECT d.file_hash FROM duplicates AS d
|
|
831
|
+
)
|
|
832
|
+
ORDER BY n.created_at ASC;
|
|
833
|
+
""")
|
|
834
|
+
rows = cur.fetchall()
|
|
835
|
+
return [
|
|
836
|
+
{"id": r[0], "file_path": r[1], "file_hash": r[2],
|
|
837
|
+
"chain_id": r[3], "created_at": r[4]}
|
|
838
|
+
for r in rows
|
|
839
|
+
]
|
|
840
|
+
|
|
841
|
+
# --------------------------------------------------------------------------------
|
|
842
|
+
# The rest is your original LLM + ingestion logic, mostly unchanged
|
|
843
|
+
# --------------------------------------------------------------------------------
|
|
844
|
+
|
|
845
|
+
class FileMetadata(BaseModel):
|
|
846
|
+
"metadata about a file"
|
|
847
|
+
document_name: str = Field(..., description="document id")
|
|
848
|
+
file_size: int = Field(..., description = "file size in bytes")
|
|
849
|
+
date_modified : str = Field(..., description = "date modified / copied to the analysis system")
|
|
850
|
+
date_created : str = Field(..., description = "date created")
|
|
851
|
+
file_hash : Optional[str] = Field(default = None, description = "file hash")
|
|
852
|
+
def __hash__(self):
|
|
853
|
+
if self.file_hash is None:
|
|
854
|
+
raise ValueError("file_hash must be provided for the model to be hashable")
|
|
855
|
+
return int(self.file_hash, base=16)
|
|
856
|
+
|
|
857
|
+
@memory.cache
|
|
858
|
+
def get_file_hash(file_path, last_modified, size_bytes, algorithm="sha256", block_size=65536, ):
|
|
859
|
+
"""Compute a hash for the given file using the specified algorithm."""
|
|
860
|
+
h = hashlib.new(algorithm)
|
|
861
|
+
with open(file_path, "rb") as f:
|
|
862
|
+
for chunk in iter(lambda: f.read(block_size), b""):
|
|
863
|
+
h.update(chunk)
|
|
864
|
+
return h.hexdigest()
|
|
865
|
+
|
|
866
|
+
def get_folder_metadata(folder_path, hash_algorithm="sha256"):
|
|
867
|
+
metadata_list = []
|
|
868
|
+
for root, dirs, files in os.walk(folder_path):
|
|
869
|
+
for name in files:
|
|
870
|
+
file_path = os.path.join(root, name)
|
|
871
|
+
stats = os.stat(file_path)
|
|
872
|
+
file_hash = get_file_hash(file_path, last_modified = stats.st_mtime,
|
|
873
|
+
size_bytes= stats.st_size,
|
|
874
|
+
algorithm=hash_algorithm)
|
|
875
|
+
metadata_list.append(
|
|
876
|
+
FileMetadata(
|
|
877
|
+
document_name = name,
|
|
878
|
+
file_size = stats.st_size,
|
|
879
|
+
file_hash = file_hash,
|
|
880
|
+
date_modified= str(datetime.datetime.fromtimestamp(stats.st_mtime)),
|
|
881
|
+
date_created =str(datetime.datetime.fromtimestamp(stats.st_birthtime ))
|
|
882
|
+
)
|
|
883
|
+
)
|
|
884
|
+
return metadata_list
|
|
885
|
+
|
|
886
|
+
class FileVersion(BaseModel):
|
|
887
|
+
"representing a file version"
|
|
888
|
+
filename: str = Field(..., description = "the filename of the current version")
|
|
889
|
+
prev: Optional[str] = Field(..., description = "The file name of the previous version. If it is brandnew not superceding/ overwriting any other, set None/Null. ")
|
|
890
|
+
supercede_reason : str = Field(..., description = "Why this version fully supercede the previous.")
|
|
891
|
+
supercede_mode : Literal["Duplicate", "FileEdit", "ContractUpdate", "N/A"] = Field(..., description = """
|
|
892
|
+
the mode of superceding, Duplicate means the file is just a duplicated copy.
|
|
893
|
+
FileEdit is small edit that does not change any activated contract terms. FileEdit is applicable to any contract file change across drafts.
|
|
894
|
+
Contract update is the update that truely reflect the signed contract with intention to update.
|
|
895
|
+
"N/A" when there is nothing to supercede when it is the first version. """)
|
|
896
|
+
|
|
897
|
+
class VersionChain(BaseModel):
|
|
898
|
+
"A linked list that the represent the evolution of a file. "
|
|
899
|
+
chain: list[FileVersion] = Field(..., description = "A single file mutation chain, first element is the root, subsequent files is the mutated version of the preceding. ")
|
|
900
|
+
pass
|
|
901
|
+
|
|
902
|
+
class FileVersionChainingResponse(BaseModel):
|
|
903
|
+
"Answer response format of file version chaining"
|
|
904
|
+
reasoning: str = Field(..., description = "reasoning at overall response level of thinking")
|
|
905
|
+
chains: list[VersionChain] = Field(..., description = "a array/ list of linked list with first element the raw verion before any changes. The next element is the file version that overwrites/ supercede the previous version. ")
|
|
906
|
+
root_agreement : str = Field(..., description = "The file name of the origin/master agreement that covers everything before any term variations are applied to. ")
|
|
907
|
+
pass
|
|
908
|
+
|
|
909
|
+
@memory.cache
|
|
910
|
+
def version_chain(metadata_list: list[FileMetadata], model = 'gemini-2.5-pro', attempt = 0):
|
|
911
|
+
from langchain_google_genai import ChatGoogleGenerativeAI
|
|
912
|
+
from langchain_core.messages import HumanMessage, SystemMessage
|
|
913
|
+
|
|
914
|
+
messages = [SystemMessage(
|
|
915
|
+
"You need to provide version chain to a list of given file data. You need to sort out which one is superceded by which. "
|
|
916
|
+
"You are given a list of file metadata and output file versioning chains. You need to reason through why one is before or after another file. "
|
|
917
|
+
"You need to cover all files, if it has no previous version. Give result as one or more file version chains. "
|
|
918
|
+
"There are some subtle file hint they are the same. Sometimes the same file chain do not share same name but some part of the file is the same. "
|
|
919
|
+
"The same file version chain may have the file renamed but preserve similar meaning, partially renamed or truncated. "
|
|
920
|
+
"For example a file maybe called in its original version `schedule 1.pdf` but the next version available is not will be schedule 1v4_final_final.pdf with potentially v2 (version 2) and v3 missing and a strange final suffix appended. \n"
|
|
921
|
+
"Of course there can be some simpler case maybe just as simple as `MSA Company name v1.pdf` and the next is simply MSA Company name v2.pdf"
|
|
922
|
+
"Your job is to track these chains as much as possible. "
|
|
923
|
+
"Example, the evolution / edit / terms changes of scheule A should be distinct from the changes of schedule B. "
|
|
924
|
+
"Remember, we are dealing with version, do not add schedule 3 after schedule 2, but only add schedule 3 version 2 after schedule 3. The do not use reading order (e.g. section 4 after section 3). "
|
|
925
|
+
"We are only concerned with estimating changes made to the same section. If it is a separate section/ contract. Or the varaition is dealing with different terms, they should follow its own new chain. "
|
|
926
|
+
"In case the file is purely copy and pasted with a different number at the end, try to choose latest, with largest number as the newer version. "
|
|
927
|
+
"The given file list comes from os files. Each file must show up exactly once in the answer chain. "
|
|
928
|
+
),
|
|
929
|
+
HumanMessage(f"{[i.model_dump(exclude = ['file_hash']) for i in metadata_list]}")]
|
|
930
|
+
|
|
931
|
+
llm = ChatGoogleGenerativeAI(model = model)
|
|
932
|
+
cnt = 0
|
|
933
|
+
max_cnt = 4
|
|
934
|
+
while cnt <= max_cnt:
|
|
935
|
+
chaining_result = llm.with_structured_output(FileVersionChainingResponse, include_raw = True).invoke(messages)
|
|
936
|
+
chains = chaining_result['parsed'].model_dump()['chains']
|
|
937
|
+
files = [cc['filename'] for c in chains for cc in c['chain']]
|
|
938
|
+
from collections import Counter
|
|
939
|
+
counter = Counter(files)
|
|
940
|
+
duplicated = []
|
|
941
|
+
for i in counter.most_common():
|
|
942
|
+
if i[1] > 1:
|
|
943
|
+
duplicated.append(i)
|
|
944
|
+
if duplicated:
|
|
945
|
+
messages.append(f"Trial answer : the field chains= {chains}")
|
|
946
|
+
messages.append(f"files duplicated: {str(duplicated)}")
|
|
947
|
+
else:
|
|
948
|
+
break
|
|
949
|
+
cnt += 1
|
|
950
|
+
if cnt == max_cnt:
|
|
951
|
+
raise(Exception(f"Max retry {max_cnt} reached" ))
|
|
952
|
+
return chaining_result['parsed'].model_dump()['chains']
|
|
953
|
+
|
|
954
|
+
@memory.cache
|
|
955
|
+
def dedup_llm_pick_newest(meta_list_dumped, model = 'gemini-2.5-flash'):
|
|
956
|
+
from langchain_google_genai import ChatGoogleGenerativeAI
|
|
957
|
+
from langchain_core.messages import HumanMessage, SystemMessage
|
|
958
|
+
|
|
959
|
+
representative_file_name = None
|
|
960
|
+
name_list = [i['document_name'] for i in meta_list_dumped]
|
|
961
|
+
|
|
962
|
+
class DedupResponse(BaseModel):
|
|
963
|
+
reasoning: str = Field(..., description = "The reasoning steps to the final answer. ")
|
|
964
|
+
representative_file_name: str = Field(..., description = "The file name of the most representative file. ")
|
|
965
|
+
|
|
966
|
+
cnt = 0
|
|
967
|
+
while representative_file_name not in name_list:
|
|
968
|
+
messages = [SystemMessage("You are given a list of files sharing the same file hash and you need to choose one that is the most representative. "),
|
|
969
|
+
HumanMessage(f"{meta_list_dumped}")]
|
|
970
|
+
llm = ChatGoogleGenerativeAI(model = model)
|
|
971
|
+
res = llm.with_structured_output(DedupResponse, include_raw = True).invoke(messages)
|
|
972
|
+
if not res.get('parsing_error'):
|
|
973
|
+
representative_file_name = res['parsed'].representative_file_name
|
|
974
|
+
if representative_file_name not in name_list:
|
|
975
|
+
messages.append(SystemMessage("Error, the answer mentioned file name does not exist in any of the file at all. Only choose the exact file name in the list. "))
|
|
976
|
+
else:
|
|
977
|
+
return representative_file_name
|
|
978
|
+
else:
|
|
979
|
+
pass
|
|
980
|
+
cnt +=1
|
|
981
|
+
if cnt > 5:
|
|
982
|
+
raise Exception("LLM error, retried max reached and cannot choose dedup filename.")
|
|
983
|
+
pass
|
|
984
|
+
|
|
985
|
+
def dedup(list_meta: list[FileMetadata]):
|
|
986
|
+
d: dict[str, set[FileMetadata]] = {}
|
|
987
|
+
for meta in list_meta:
|
|
988
|
+
m = meta.model_dump(exclude = ['file_hash'])
|
|
989
|
+
if meta.file_hash not in d:
|
|
990
|
+
d[meta.file_hash] = [m]
|
|
991
|
+
else:
|
|
992
|
+
d[meta.file_hash].append(m)
|
|
993
|
+
res = {}
|
|
994
|
+
dup_res = []
|
|
995
|
+
for hs, meta_set in d.items():
|
|
996
|
+
if len(meta_set) > 1:
|
|
997
|
+
dup = {i['document_name']: i for i in meta_set}
|
|
998
|
+
logger.info(f"deduping set {';'.join(list(dup))}")
|
|
999
|
+
src = dedup_llm_pick_newest(meta_set)
|
|
1000
|
+
res[hs] = dup.pop(src)
|
|
1001
|
+
dup_res.extend([FileVersion.model_validate({"filename": d['document_name'], "prev": src, "supercede_reason": f"duplicate of file {src}", "supercede_mode" : "Duplicate"}) for d in dup.values()])
|
|
1002
|
+
else:
|
|
1003
|
+
res[hs] = list(meta_set)[0]
|
|
1004
|
+
return res, dup_res
|
|
1005
|
+
|
|
1006
|
+
def check_missing_db(db: VersionChainDB, d_hash_to_meta):
|
|
1007
|
+
# Get all filenames in DB
|
|
1008
|
+
all_db_files = set()
|
|
1009
|
+
for chain in db.find_chains():
|
|
1010
|
+
for node in db.get_chain(chain['id']):
|
|
1011
|
+
all_db_files.add(node['file_path'])
|
|
1012
|
+
available_files = set(meta.document_name for meta in d_hash_to_meta.values())
|
|
1013
|
+
missing = available_files - all_db_files
|
|
1014
|
+
return missing
|
|
1015
|
+
|
|
1016
|
+
class FileAddition(FileVersion):
|
|
1017
|
+
"file addition"
|
|
1018
|
+
filename: str = Field(..., description = 'the file name of the new file')
|
|
1019
|
+
prev: Optional[str] = Field(None, description = 'the file name its previous version, set Null or None if it belong to new chain')
|
|
1020
|
+
next: Optional[str] = Field(None, description = 'the file name its next version, use only when this file is inserted to the beginning of an existing file version chain. ')
|
|
1021
|
+
supercede_reason : str = Field(..., description = "Why this version fully supercede the previous.")
|
|
1022
|
+
supercede_mode : Literal["Duplicate", "FileEdit", "ContractUpdate", "N/A"] = Field(..., description = """
|
|
1023
|
+
the mode of superceding, Duplicate means the file is just a duplicated copy.
|
|
1024
|
+
FileEdit is small edit that does not change any activated contract terms. FileEdit is applicable to any contract file change across drafts.
|
|
1025
|
+
Contract update is the update that truely reflect the signed contract with intention to update.
|
|
1026
|
+
"N/A" when there is nothing to supercede when it is the first version. """)
|
|
1027
|
+
@model_validator(mode = 'after')
|
|
1028
|
+
def only_one_prev_next_none(self):
|
|
1029
|
+
if (self.prev is None) or (self.next is None):
|
|
1030
|
+
return self
|
|
1031
|
+
else:
|
|
1032
|
+
raise Exception('at most one of prev or next is None')
|
|
1033
|
+
|
|
1034
|
+
class AddFilesResponse(BaseModel):
|
|
1035
|
+
"the result of the file addition"
|
|
1036
|
+
reasoning: str = Field(..., description = 'reasoning in top level perspective')
|
|
1037
|
+
additions : list[FileAddition] = Field(..., description = 'list of file additions')
|
|
1038
|
+
|
|
1039
|
+
@memory.cache
|
|
1040
|
+
def add_new_file_to_existing_chains(chains, all_d_hash_to_meta: dict[str, FileMetadata], new_file_name, model = 'gemini-2.5-pro', attempt = 0 ) -> AddFilesResponse:
|
|
1041
|
+
from langchain_core.messages import HumanMessage, SystemMessage
|
|
1042
|
+
messages = [SystemMessage("Given existing file versioning chain and a new file with metadata of all files, you need to decide the position of the new file in the version chain. "
|
|
1043
|
+
"All files have distinct file hash. "
|
|
1044
|
+
"if existing chain is file_v1, file_v1.1, file2 and a new file file_v_1.5 and you think it should be inserted between file_v1.1 and file2. "
|
|
1045
|
+
"Remember, we are dealing with version, do not add schedule 3 after schedule 2, but only add schedule 3 version 2 after schedule 3. The do not use reading order (e.g. section 4 after section 3). "
|
|
1046
|
+
"We are only concerned with estimating changes made to the same section. If it is a separate section/ contract. Or the varaition is dealing with different terms, they should follow its own new chain. "
|
|
1047
|
+
"then set the prev of file_v_1.5 to be file_v1.1, remember if it is newer than one but looks older than another, you need to reason about the possibility of insertion instead of appending directly at the end. "),
|
|
1048
|
+
HumanMessage(f"Existing version chain: {chains}" + "\n\n" +
|
|
1049
|
+
f"Metadata file-hash to file-meta lookup for all files: {[i.model_dump(exclude = set(['file_hash'])) for i in all_d_hash_to_meta.values()]}" + "\n\n" +
|
|
1050
|
+
f"New file to add: {new_file_name}"
|
|
1051
|
+
),
|
|
1052
|
+
|
|
1053
|
+
]
|
|
1054
|
+
from langchain_google_genai import ChatGoogleGenerativeAI
|
|
1055
|
+
max_retry = 3
|
|
1056
|
+
n = 0
|
|
1057
|
+
while True:
|
|
1058
|
+
llm = ChatGoogleGenerativeAI(model = model)
|
|
1059
|
+
res = llm.with_structured_output(AddFilesResponse, include_raw = True).invoke(messages)
|
|
1060
|
+
if res.get("parsing_error"):
|
|
1061
|
+
n += 1
|
|
1062
|
+
if n >= max_retry:
|
|
1063
|
+
raise Exception(f"Max Retry reached {max_retry}")
|
|
1064
|
+
else:
|
|
1065
|
+
return res.get('parsed')
|
|
1066
|
+
|
|
1067
|
+
def apply_chain_updates_db(db: VersionChainDB, updates, all_meta: dict[str, FileMetadata]):
|
|
1068
|
+
additions: list[FileAddition] = updates.additions
|
|
1069
|
+
changes_made = {"added_nodes": [], "added_chains": []}
|
|
1070
|
+
for addition in additions:
|
|
1071
|
+
meta = all_meta.get(addition.filename)
|
|
1072
|
+
if addition.prev is None and addition.next is None:
|
|
1073
|
+
# New chain
|
|
1074
|
+
chain_id = db.create_chain(name=f"Chain_for_{addition.filename}")
|
|
1075
|
+
changes_made["added_chains"].append(chain_id)
|
|
1076
|
+
changes_made["added_nodes"].append(db.add_node(
|
|
1077
|
+
chain_id=chain_id,
|
|
1078
|
+
file_path=addition.filename,
|
|
1079
|
+
file_size=meta.file_size if meta else 0,
|
|
1080
|
+
file_hash=(meta.file_hash if (meta and meta.file_hash is not None) else ""),
|
|
1081
|
+
position="append",
|
|
1082
|
+
metadata_json=str(addition.model_dump())
|
|
1083
|
+
))
|
|
1084
|
+
else:
|
|
1085
|
+
# Find the chain and reference node for insertion
|
|
1086
|
+
target_chain_id = None
|
|
1087
|
+
ref_node_id = None
|
|
1088
|
+
position = None
|
|
1089
|
+
for chain in db.find_chains():
|
|
1090
|
+
nodes = db.get_chain(chain['id'])
|
|
1091
|
+
# Insert at beginning (prepend)
|
|
1092
|
+
if addition.prev is None and addition.next is not None:
|
|
1093
|
+
for node in nodes:
|
|
1094
|
+
if node['file_path'] == addition.next:
|
|
1095
|
+
target_chain_id = chain['id']
|
|
1096
|
+
ref_node_id = node['id']
|
|
1097
|
+
position = "prepend"
|
|
1098
|
+
if node['prev_id']: # next existed, so it's actually "between"
|
|
1099
|
+
position = "between"
|
|
1100
|
+
else:
|
|
1101
|
+
position = "prepend"
|
|
1102
|
+
break
|
|
1103
|
+
# Insert at end (append)
|
|
1104
|
+
elif addition.prev is not None and addition.next is None:
|
|
1105
|
+
for node in nodes:
|
|
1106
|
+
if node['file_path'] == addition.prev:
|
|
1107
|
+
target_chain_id = chain['id']
|
|
1108
|
+
ref_node_id = node['id']
|
|
1109
|
+
if node['next_id']:
|
|
1110
|
+
position = "between"
|
|
1111
|
+
else:
|
|
1112
|
+
position = "append"
|
|
1113
|
+
break
|
|
1114
|
+
# Insert between
|
|
1115
|
+
elif addition.prev is not None and addition.next is not None:
|
|
1116
|
+
for node in nodes:
|
|
1117
|
+
if node['file_path'] == addition.prev:
|
|
1118
|
+
target_chain_id = chain['id']
|
|
1119
|
+
ref_node_id = node['id']
|
|
1120
|
+
position = "between"
|
|
1121
|
+
break
|
|
1122
|
+
if target_chain_id is not None:
|
|
1123
|
+
break
|
|
1124
|
+
if target_chain_id is not None and ref_node_id is not None and position is not None:
|
|
1125
|
+
changes_made["added_nodes"].append(db.add_node(
|
|
1126
|
+
chain_id=target_chain_id,
|
|
1127
|
+
file_path=addition.filename,
|
|
1128
|
+
file_size=meta.file_size if meta else 0,
|
|
1129
|
+
file_hash=(meta.file_hash if (meta and meta.file_hash is not None) else ""),
|
|
1130
|
+
position=position,
|
|
1131
|
+
ref_node_id=ref_node_id,
|
|
1132
|
+
metadata_json=str(addition.model_dump())
|
|
1133
|
+
))
|
|
1134
|
+
# if position is None, skip for now
|
|
1135
|
+
return changes_made
|
|
1136
|
+
|
|
1137
|
+
|
|
1138
|
+
# ---------------------------
|
|
1139
|
+
# optional: backfill
|
|
1140
|
+
# ---------------------------
|
|
1141
|
+
def backfill_page_hashes(db: VersionChainDB, file_root, dpi: int = 300):
|
|
1142
|
+
"""
|
|
1143
|
+
Run once on an existing DB to populate page_hashes and page-based duplicates.
|
|
1144
|
+
"""
|
|
1145
|
+
cur = db.conn.cursor()
|
|
1146
|
+
cur.execute("SELECT id, file_path FROM nodes")
|
|
1147
|
+
rows = cur.fetchall()
|
|
1148
|
+
for node_id, file_path in rows:
|
|
1149
|
+
full_path = os.path.join(file_root, file_path)
|
|
1150
|
+
if not file_path.lower().endswith(".pdf"):
|
|
1151
|
+
continue
|
|
1152
|
+
if not os.path.exists(full_path):
|
|
1153
|
+
continue
|
|
1154
|
+
try:
|
|
1155
|
+
ph = pdf_page_hashes_as_png(full_path, dpi=dpi)
|
|
1156
|
+
db.insert_page_hashes(node_id, file_path, ph, dpi=dpi)
|
|
1157
|
+
contained_in = db.find_bigger_files_containing_this_sequence(node_id)
|
|
1158
|
+
for info in contained_in:
|
|
1159
|
+
db.mark_subdocument_duplicate(info)
|
|
1160
|
+
except Exception as e:
|
|
1161
|
+
logger.exception(f"Backfill page-hash failed for {file_path}: {e}")
|
|
1162
|
+
raise e
|
|
1163
|
+
|
|
1164
|
+
|
|
1165
|
+
if __name__ == "__main__":
|
|
1166
|
+
from ..pdf2png import RawFileLoader
|
|
1167
|
+
import pandas as pd
|
|
1168
|
+
in_compare_root = os.path.join('..', 'doc_data', 'raw_documents')
|
|
1169
|
+
loader = RawFileLoader(
|
|
1170
|
+
env_flist_path=None,
|
|
1171
|
+
walk_root=os.path.join('..', 'doc_data', 'raw_documents', 'Samples - 9 Oct 2025'),
|
|
1172
|
+
compare_root=in_compare_root,
|
|
1173
|
+
include=['dirs']
|
|
1174
|
+
)
|
|
1175
|
+
filechain_folder_root = os.path.join('..', 'doc_data', 'file_version_chains')
|
|
1176
|
+
|
|
1177
|
+
for rel_in_path in loader:
|
|
1178
|
+
foldername = os.path.join(in_compare_root, rel_in_path)
|
|
1179
|
+
out_folder = os.path.join(filechain_folder_root, rel_in_path)
|
|
1180
|
+
if not os.path.exists(out_folder):
|
|
1181
|
+
os.makedirs(out_folder)
|
|
1182
|
+
folder_meta = get_folder_metadata(foldername)
|
|
1183
|
+
print(folder_meta)
|
|
1184
|
+
|
|
1185
|
+
db = VersionChainDB(db_path=os.path.join(out_folder, "file_index.sqlite"))
|
|
1186
|
+
|
|
1187
|
+
# 1) first-level name/hash dedup against existing DB rows
|
|
1188
|
+
filtered_meta = []
|
|
1189
|
+
for m in folder_meta:
|
|
1190
|
+
if db.name_exists(m.document_name):
|
|
1191
|
+
# If name exists, skip entirely
|
|
1192
|
+
continue
|
|
1193
|
+
elif db.hash_exists(m.file_hash):
|
|
1194
|
+
# If name does NOT exist, but hash exists, insert duplicate
|
|
1195
|
+
canon = db.get_canonical_for_hash(m.file_hash)
|
|
1196
|
+
if canon:
|
|
1197
|
+
canon_id, canon_path, canon_chain_id = canon
|
|
1198
|
+
db.insert_duplicate(
|
|
1199
|
+
file_name=m.document_name, # this new one is the duplicate
|
|
1200
|
+
file_hash=m.file_hash,
|
|
1201
|
+
duplicate_of_file_name=canon_path, # point to the canonical existing one
|
|
1202
|
+
duplicate_of_file_hash=m.file_hash,
|
|
1203
|
+
chain_id=canon_chain_id,
|
|
1204
|
+
node_id=None, # we don't have a node_id for the new one yet
|
|
1205
|
+
)
|
|
1206
|
+
db.conn.commit()
|
|
1207
|
+
continue
|
|
1208
|
+
else:
|
|
1209
|
+
# If neither exists, process as new file
|
|
1210
|
+
filtered_meta.append(m)
|
|
1211
|
+
|
|
1212
|
+
# 2) LLM-driven dedup for the new batch
|
|
1213
|
+
d_hash_to_meta, dup = dedup(filtered_meta)
|
|
1214
|
+
df_dup = pd.DataFrame([d.model_dump(exclude=['file_hash']) for d in dup])
|
|
1215
|
+
d_hash_to_meta = {k: FileMetadata.model_validate(v) for k, v in d_hash_to_meta.items()}
|
|
1216
|
+
df_to_concat = [df_dup]
|
|
1217
|
+
attempt = 0
|
|
1218
|
+
chains_response = version_chain(list(d_hash_to_meta.values()), attempt=attempt)
|
|
1219
|
+
chains = chains_response
|
|
1220
|
+
if attempt == 0:
|
|
1221
|
+
# emulate hallucination case when a file meta missing
|
|
1222
|
+
popped_chain = chains.pop(-1)
|
|
1223
|
+
for chain in chains:
|
|
1224
|
+
if len(chain['chain']) > 3:
|
|
1225
|
+
popped_doc = chain['chain'].pop(1)
|
|
1226
|
+
break
|
|
1227
|
+
|
|
1228
|
+
filename_to_meta = {i.document_name: i for i in filtered_meta}
|
|
1229
|
+
try:
|
|
1230
|
+
db.conn.execute("BEGIN")
|
|
1231
|
+
|
|
1232
|
+
# persist each duplicate
|
|
1233
|
+
for file_version in dup:
|
|
1234
|
+
dup_meta = next((m for m in filtered_meta if m.document_name == file_version.filename), None)
|
|
1235
|
+
rep_meta = next((m for m in filtered_meta if m.document_name == file_version.prev), None)
|
|
1236
|
+
db.insert_duplicate(
|
|
1237
|
+
file_name=file_version.filename,
|
|
1238
|
+
file_hash=dup_meta.file_hash if dup_meta else None,
|
|
1239
|
+
duplicate_of_file_name=file_version.prev,
|
|
1240
|
+
duplicate_of_file_hash=rep_meta.file_hash if rep_meta else None,
|
|
1241
|
+
chain_id=None,
|
|
1242
|
+
node_id=None
|
|
1243
|
+
)
|
|
1244
|
+
|
|
1245
|
+
# persist each chain
|
|
1246
|
+
for chain_dict in chains_response:
|
|
1247
|
+
chain_id = db.create_chain(name=str(uuid.uuid1()))
|
|
1248
|
+
for file_version in chain_dict['chain']:
|
|
1249
|
+
meta = filename_to_meta.get(file_version['filename'])
|
|
1250
|
+
db.add_node(
|
|
1251
|
+
chain_id=chain_id,
|
|
1252
|
+
file_path=file_version['filename'],
|
|
1253
|
+
file_size=meta.file_size if meta else 0,
|
|
1254
|
+
file_hash=(meta.file_hash if (meta and meta.file_hash is not None) else ""),
|
|
1255
|
+
position="append",
|
|
1256
|
+
metadata_json=str(file_version)
|
|
1257
|
+
)
|
|
1258
|
+
|
|
1259
|
+
db.conn.commit()
|
|
1260
|
+
except Exception as e:
|
|
1261
|
+
db.conn.rollback()
|
|
1262
|
+
print(f"Error inserting chain: {e}")
|
|
1263
|
+
raise e
|
|
1264
|
+
|
|
1265
|
+
df = pd.DataFrame([j for i in chains for j in i['chain']])
|
|
1266
|
+
df_to_concat.append(df)
|
|
1267
|
+
missing = check_missing_db(db, d_hash_to_meta)
|
|
1268
|
+
max_retry = 6
|
|
1269
|
+
while missing:
|
|
1270
|
+
updates = add_new_file_to_existing_chains(chains, d_hash_to_meta, missing, attempt=attempt)
|
|
1271
|
+
apply_chain_updates_db(db, updates, filename_to_meta)
|
|
1272
|
+
missing = check_missing_db(db, d_hash_to_meta)
|
|
1273
|
+
attempt += 1
|
|
1274
|
+
if attempt > max_retry:
|
|
1275
|
+
raise Exception(f"Max retry {max_retry} reached without fixing version chain")
|
|
1276
|
+
import pathlib
|
|
1277
|
+
df_out = pd.concat(df_to_concat)
|
|
1278
|
+
df_out.to_csv(pathlib.Path(foldername).parts[-1] + '.csv')
|