graph-knowledge-doc-parser 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. graph_knowledge_doc_parser-0.1.0.dist-info/METADATA +326 -0
  2. graph_knowledge_doc_parser-0.1.0.dist-info/RECORD +38 -0
  3. graph_knowledge_doc_parser-0.1.0.dist-info/WHEEL +4 -0
  4. graph_knowledge_doc_parser-0.1.0.dist-info/entry_points.txt +3 -0
  5. kg_doc_parser/__init__.py +9 -0
  6. kg_doc_parser/cast_hinting.py +19 -0
  7. kg_doc_parser/document_ingester_logger.py +766 -0
  8. kg_doc_parser/models.py +277 -0
  9. kg_doc_parser/ocr.py +752 -0
  10. kg_doc_parser/pdf2png.py +286 -0
  11. kg_doc_parser/semantic_document_splitting_layerwise_edits.py +3302 -0
  12. kg_doc_parser/text_processing_utils.py +30 -0
  13. kg_doc_parser/utils/__init__.py +0 -0
  14. kg_doc_parser/utils/bounded_threadpool_executor.py +37 -0
  15. kg_doc_parser/utils/file_loaders.py +405 -0
  16. kg_doc_parser/utils/langchain.py +220 -0
  17. kg_doc_parser/utils/log.py +135 -0
  18. kg_doc_parser/utils/version_chaining.py +1278 -0
  19. kg_doc_parser/workflow_ingest/__init__.py +187 -0
  20. kg_doc_parser/workflow_ingest/_kogwistar.py +13 -0
  21. kg_doc_parser/workflow_ingest/adapters.py +212 -0
  22. kg_doc_parser/workflow_ingest/cache.py +63 -0
  23. kg_doc_parser/workflow_ingest/cli.py +324 -0
  24. kg_doc_parser/workflow_ingest/clients.py +444 -0
  25. kg_doc_parser/workflow_ingest/demo_harness.py +427 -0
  26. kg_doc_parser/workflow_ingest/design.py +208 -0
  27. kg_doc_parser/workflow_ingest/handlers.py +617 -0
  28. kg_doc_parser/workflow_ingest/models.py +575 -0
  29. kg_doc_parser/workflow_ingest/ocr_pipeline.py +1581 -0
  30. kg_doc_parser/workflow_ingest/page_index.py +473 -0
  31. kg_doc_parser/workflow_ingest/parser_core.py +862 -0
  32. kg_doc_parser/workflow_ingest/parsing.py +249 -0
  33. kg_doc_parser/workflow_ingest/probe.py +164 -0
  34. kg_doc_parser/workflow_ingest/providers.py +412 -0
  35. kg_doc_parser/workflow_ingest/runners.py +546 -0
  36. kg_doc_parser/workflow_ingest/semantics.py +231 -0
  37. kg_doc_parser/workflow_ingest/service.py +112 -0
  38. kg_doc_parser/workflow_ingest/smoke_assets.py +62 -0
@@ -0,0 +1,1278 @@
1
+ import pathlib
2
+ import logging
3
+ logger = logging.getLogger(__name__)
4
+ logger.setLevel(logging.DEBUG)
5
+
6
+ import os
7
+
8
+ #import logging.handlers
9
+ #logger.addHandler(logging.handlers.RotatingFileHandler(os.path.join('.', 'logs', __name__)))
10
+ from .log import SQLiteHandler
11
+ sqlite_handler = SQLiteHandler(os.path.join('.','logs', 'application_logs.db'))
12
+ sqlite_handler.setLevel(logging.DEBUG)
13
+ logger.addHandler(sqlite_handler)
14
+
15
+ from pydantic import BaseModel, Field, ValidationError, model_validator
16
+ import uuid
17
+
18
+ import datetime
19
+ import hashlib
20
+ import dotenv
21
+ from typing import Literal, Optional, List, Dict, Any
22
+ from joblib import Memory
23
+ memory = Memory(location = "./.version_chain")
24
+
25
+ dotenv.load_dotenv()
26
+
27
+ import sqlite3
28
+
29
+ # ====== optional PDF -> PNG renderers ======
30
+ # we try pdf2image first, but fall back to PyMuPDF if needed
31
+ try:
32
+ from pdf2image import convert_from_path
33
+ _HAS_PDF2IMAGE = True
34
+ except Exception:
35
+ _HAS_PDF2IMAGE = False
36
+ try:
37
+ import fitz # PyMuPDF
38
+ _HAS_PYMUPDF = True
39
+ except Exception:
40
+ _HAS_PYMUPDF = False
41
+
42
+
43
+ # ===========================================
44
+ # PDF → PNG → hash (no temp file, in-memory)
45
+ # ===========================================
46
+
47
+ def pdf_page_hashes_as_png(
48
+ file_path: str,
49
+ dpi: int = 300,
50
+ algo: str = "sha256",
51
+ img_format: str = "PNG",
52
+ ) -> list[str]:
53
+ """
54
+ Render each page to PNG in-memory and hash the bytes.
55
+ Returns list of hex digests in page order.
56
+ Includes no temp-file writes.
57
+
58
+ NOTE: this requires either pdf2image+poppler OR PyMuPDF.
59
+ """
60
+ try:
61
+ if _HAS_PDF2IMAGE:
62
+ print(f"using pdf2image to convert file {file_path}")
63
+ images = convert_from_path(file_path, dpi=dpi)
64
+ out: list[str] = []
65
+ for i, img in enumerate(images):
66
+ print(f'page-{i}', end = ' ')
67
+ import io
68
+ buf = io.BytesIO()
69
+ img.save(buf, format=img_format)
70
+ data = buf.getvalue()
71
+ h = hashlib.new(algo)
72
+ h.update(data)
73
+ out.append(h.hexdigest())
74
+ return out
75
+ except Exception as e:
76
+
77
+ if _HAS_PYMUPDF:
78
+ print(f"using fitz/pymupdf to convert file {file_path}")
79
+ import fitz
80
+ doc = fitz.open(file_path)
81
+ out: list[str] = []
82
+ for i, page in enumerate(doc):
83
+ print(f'page-{i}', end = ' ')
84
+ # dpi → matrix
85
+ zoom = dpi / 72.0
86
+ mat = fitz.Matrix(zoom, zoom)
87
+ pix = page.get_pixmap(matrix=mat)
88
+ data = pix.tobytes("png")
89
+ h = hashlib.new(algo)
90
+ h.update(data)
91
+ out.append(h.hexdigest())
92
+ return out
93
+ raise e
94
+
95
+ raise RuntimeError(
96
+ "No PDF renderer available. Install `pdf2image` (plus poppler) or `PyMuPDF`."
97
+ )
98
+
99
+
100
+ # ============================================================
101
+ # smarter subsequence check: longer -> dict(hash -> [pages])
102
+ # ============================================================
103
+
104
+ from collections import defaultdict
105
+ from bisect import bisect_right
106
+
107
+ def build_pos_index(seq: list[str]) -> dict[str, list[int]]:
108
+ """
109
+ seq[i] = hash_at_page_i -> index[h] = sorted list of page numbers
110
+ """
111
+ idx: dict[str, list[int]] = defaultdict(list)
112
+ for i, h in enumerate(seq):
113
+ idx[h].append(i)
114
+ return idx
115
+
116
+ def is_ordered_subsequence(shorter: list[str], longer_idx: dict[str, list[int]]) -> bool:
117
+ """
118
+ Check that every hash in `shorter` can be found in `longer_idx`
119
+ in strictly increasing page order. Gaps allowed.
120
+ """
121
+ prev_pos = -1
122
+ for h in shorter:
123
+ positions = longer_idx.get(h)
124
+ if not positions:
125
+ return False
126
+ j = bisect_right(positions, prev_pos)
127
+ if j == len(positions):
128
+ return False
129
+ prev_pos = positions[j]
130
+ return True
131
+
132
+
133
+ # =========================
134
+ # SQLite-backed VersionChainDB class for persistent version chain management
135
+ # =========================
136
+
137
+ class VersionChainDB:
138
+ """
139
+ SQLite-backed class for managing multiple version chains of PDF files.
140
+ Each chain is a linked list of nodes (PDF files) with metadata.
141
+ Supports CRUD, append, prepend, and insert-between operations.
142
+ """
143
+
144
+ def __init__(self, db_path: str = "version_chains.db"):
145
+ self.db_path = db_path
146
+ self.conn = sqlite3.connect(self.db_path)
147
+ self._create_tables()
148
+
149
+ def _create_tables(self):
150
+ cur = self.conn.cursor()
151
+ cur.execute("""
152
+ CREATE TABLE IF NOT EXISTS chains (
153
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
154
+ name TEXT
155
+ );
156
+ """)
157
+ cur.execute("""
158
+ CREATE TABLE IF NOT EXISTS nodes (
159
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
160
+ chain_id INTEGER,
161
+ file_path TEXT,
162
+ file_size INTEGER,
163
+ file_hash TEXT,
164
+ prev_id INTEGER,
165
+ next_id INTEGER,
166
+ created_at TEXT,
167
+ metadata_json TEXT,
168
+ FOREIGN KEY(chain_id) REFERENCES chains(id),
169
+ FOREIGN KEY(prev_id) REFERENCES nodes(id),
170
+ FOREIGN KEY(next_id) REFERENCES nodes(id)
171
+ );
172
+ """)
173
+ cur.execute("""
174
+ CREATE TABLE IF NOT EXISTS duplicates (
175
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
176
+ file_name TEXT,
177
+ file_hash TEXT,
178
+ duplicate_of_file_name TEXT,
179
+ duplicate_of_file_hash TEXT,
180
+ chain_id INTEGER,
181
+ node_id INTEGER,
182
+ created_at TEXT
183
+ );
184
+ """)
185
+
186
+ # NEW: per-page PNG-hashes
187
+ # we store BOTH node_id (canonical) and file_path (so you can query by name)
188
+ cur.execute("""
189
+ CREATE TABLE IF NOT EXISTS page_hashes (
190
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
191
+ node_id INTEGER NOT NULL,
192
+ file_path TEXT NOT NULL,
193
+ page_num INTEGER NOT NULL,
194
+ page_hash TEXT NOT NULL,
195
+ render_dpi INTEGER NOT NULL DEFAULT 150,
196
+ render_format TEXT NOT NULL DEFAULT 'PNG',
197
+ render_algo TEXT NOT NULL DEFAULT 'sha256',
198
+ UNIQUE (node_id, page_num),
199
+ FOREIGN KEY(node_id) REFERENCES nodes(id)
200
+ );
201
+ """)
202
+ cur.execute("CREATE INDEX IF NOT EXISTS idx_page_hashes_node ON page_hashes(node_id);")
203
+ cur.execute("CREATE INDEX IF NOT EXISTS idx_page_hashes_hash ON page_hashes(page_hash);")
204
+
205
+ # optional: convenience view to see filename alongside page hashes
206
+ cur.execute("""
207
+ CREATE VIEW IF NOT EXISTS vw_page_hashes AS
208
+ SELECT
209
+ ph.id,
210
+ ph.node_id,
211
+ ph.file_path,
212
+ ph.page_num,
213
+ ph.page_hash,
214
+ ph.render_dpi,
215
+ ph.render_format,
216
+ ph.render_algo
217
+ FROM page_hashes ph;
218
+ """)
219
+
220
+ if not self.conn.in_transaction:
221
+ self.conn.commit()
222
+ def find_smaller_files_contained_in_this_sequence(self, node_id: int) -> list[dict]:
223
+ """
224
+ We (node_id) are the *bigger* document (or at least, we might be).
225
+ For every other document:
226
+ - take its FIRST page hash
227
+ - if that hash appears in ANY page of *us*, try to match the whole smaller doc
228
+ """
229
+ big_seq = self.get_page_hashes(node_id)
230
+ if not big_seq:
231
+ return []
232
+
233
+ big_len = len(big_seq)
234
+
235
+ # 1) index our own pages: hash -> [positions]
236
+ big_idx = build_pos_index(big_seq)
237
+
238
+ # 2) to reduce rows, we only want smaller docs whose first-page hash is
239
+ # one of OUR hashes.
240
+ big_hash_set = set(big_seq)
241
+ placeholders = ",".join("?" * len(big_hash_set))
242
+
243
+ cur = self.conn.cursor()
244
+
245
+ if placeholders:
246
+ cur.execute(
247
+ f"""
248
+ SELECT node_id, page_hash
249
+ FROM page_hashes
250
+ WHERE page_num = 0
251
+ AND node_id <> ?
252
+ AND page_hash IN ({placeholders})
253
+ """,
254
+ (node_id, *big_hash_set),
255
+ )
256
+ else:
257
+ # big doc somehow has no hashes? weird, just return
258
+ return []
259
+
260
+ first_pages = cur.fetchall()
261
+
262
+ # our own info
263
+ cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (node_id,))
264
+ this_row = cur.fetchone()
265
+ this_file_path = this_row[0] if this_row else None
266
+ this_file_hash = this_row[1] if this_row else None
267
+ this_chain_id = this_row[2] if this_row else None
268
+
269
+ results: list[dict] = []
270
+
271
+ for small_id, small_first_hash in first_pages:
272
+ # we already know small_first_hash is in our set,
273
+ # but it can appear multiple times (same page repeated), so:
274
+ positions = big_idx.get(small_first_hash, [])
275
+ if not positions:
276
+ continue
277
+
278
+ # fetch full smaller sequence
279
+ small_seq = self.get_page_hashes(small_id)
280
+ small_len = len(small_seq)
281
+ if not small_seq:
282
+ continue
283
+
284
+ matched = False
285
+ for pos in positions:
286
+ # can we fit the smaller starting at this pos?
287
+ if pos + small_len > big_len:
288
+ continue
289
+ if big_seq[pos:pos + small_len] == small_seq:
290
+ matched = True
291
+ break
292
+
293
+ if matched:
294
+ # hydrate smaller
295
+ cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (small_id,))
296
+ small_row = cur.fetchone()
297
+ results.append({
298
+ "small_node_id": small_id,
299
+ "small_file_path": small_row[0] if small_row else None,
300
+ "small_file_hash": small_row[1] if small_row else None,
301
+ "small_chain_id": small_row[2] if small_row else None,
302
+ "big_node_id": node_id,
303
+ "big_file_path": this_file_path,
304
+ "big_file_hash": this_file_hash,
305
+ "big_chain_id": this_chain_id,
306
+ })
307
+
308
+ return results
309
+ def get_nodes_with_pagecount_at_most(self, n_pages: int, exclude_node_id: int) -> list[int]:
310
+ """
311
+ Return node_ids whose page_count <= n_pages, excluding the given node.
312
+ Useful when *this* node is longer and we want to see if we contain shorter ones.
313
+ """
314
+ cur = self.conn.cursor()
315
+ cur.execute("""
316
+ SELECT ph.node_id, COUNT(*) AS c
317
+ FROM page_hashes ph
318
+ GROUP BY ph.node_id
319
+ HAVING c <= ?
320
+ """, (n_pages,))
321
+ out: list[int] = []
322
+ for node_id, c in cur.fetchall():
323
+ if node_id != exclude_node_id:
324
+ out.append(node_id)
325
+ return out
326
+ def get_canonical_files(self) -> list[dict]:
327
+ cur = self.conn.cursor()
328
+ cur.execute("""
329
+ SELECT n.id, n.file_path, n.file_hash, n.file_size, n.chain_id, n.created_at
330
+ FROM nodes AS n
331
+ WHERE n.file_path NOT IN (
332
+ SELECT d.file_name
333
+ FROM duplicates AS d
334
+ WHERE d.file_name IS NOT NULL
335
+ )
336
+ ORDER BY n.id
337
+ """)
338
+ rows = cur.fetchall()
339
+ return [
340
+ {
341
+ "id": r[0],
342
+ "file_path": r[1],
343
+ "file_hash": r[2],
344
+ "file_size": r[3],
345
+ "chain_id": r[4],
346
+ "created_at": r[5],
347
+ }
348
+ for r in rows
349
+ ]
350
+ # --------------------------
351
+ # page-hash helpers
352
+ # --------------------------
353
+
354
+ def insert_page_hashes(
355
+ self,
356
+ node_id: int,
357
+ file_path: str,
358
+ page_hashes: list[str],
359
+ dpi: int = 300,
360
+ render_format: str = "PNG",
361
+ algo: str = "sha256",
362
+ ):
363
+ """
364
+ Store the ordered page-hash sequence for a node, replacing old ones if any.
365
+ """
366
+ cur = self.conn.cursor()
367
+ cur.execute("DELETE FROM page_hashes WHERE node_id = ?", (node_id,))
368
+ cur.executemany(
369
+ """
370
+ INSERT INTO page_hashes
371
+ (node_id, file_path, page_num, page_hash, render_dpi, render_format, render_algo)
372
+ VALUES (?, ?, ?, ?, ?, ?, ?)
373
+ """,
374
+ [
375
+ (node_id, file_path, i, h, dpi, render_format, algo)
376
+ for i, h in enumerate(page_hashes)
377
+ ],
378
+ )
379
+ if not self.conn.in_transaction:
380
+ self.conn.commit()
381
+
382
+ def get_page_hashes(self, node_id: int) -> list[str]:
383
+ cur = self.conn.cursor()
384
+ cur.execute("""
385
+ SELECT page_hash
386
+ FROM page_hashes
387
+ WHERE node_id = ?
388
+ ORDER BY page_num ASC
389
+ """, (node_id,))
390
+ return [r[0] for r in cur.fetchall()]
391
+
392
+ def get_nodes_with_pagecount_at_least(self, n_pages: int, exclude_node_id: int) -> list[int]:
393
+ """
394
+ Return node_ids whose page_count >= n_pages, excluding the given node.
395
+ """
396
+ cur = self.conn.cursor()
397
+ cur.execute("""
398
+ SELECT ph.node_id, COUNT(*) AS c
399
+ FROM page_hashes ph
400
+ GROUP BY ph.node_id
401
+ HAVING c >= ?
402
+ """, (n_pages,))
403
+ out: list[int] = []
404
+ for node_id, c in cur.fetchall():
405
+ if node_id != exclude_node_id:
406
+ out.append(node_id)
407
+ return out
408
+
409
+ def find_bigger_files_containing_this_sequence(self, node_id: int) -> list[dict]:
410
+ """
411
+ We (node_id) are the *smaller* candidate.
412
+ Look for ANY other file that has our FIRST page-hash somewhere (any page),
413
+ and also has enough pages after that to hold our whole sequence.
414
+ Then confirm with a Python slice compare.
415
+ """
416
+ small_seq = self.get_page_hashes(node_id)
417
+ if not small_seq:
418
+ return []
419
+
420
+ first_hash = small_seq[0]
421
+ small_len = len(small_seq)
422
+
423
+ cur = self.conn.cursor()
424
+ # 1) SQL prune:
425
+ # - find ANY page (not just page 0) whose hash == our first page
426
+ # - make sure from that page to the end there are >= small_len pages
427
+ cur.execute(
428
+ """
429
+ WITH maxp AS (
430
+ SELECT node_id, MAX(page_num) AS max_page
431
+ FROM page_hashes
432
+ GROUP BY node_id
433
+ )
434
+ SELECT ph.node_id, ph.page_num, maxp.max_page
435
+ FROM page_hashes AS ph
436
+ JOIN maxp ON ph.node_id = maxp.node_id
437
+ WHERE ph.page_hash = ?
438
+ AND ph.node_id <> ?
439
+ AND (ph.page_num + ?) <= (maxp.max_page + 1)
440
+ """,
441
+ (first_hash, node_id, small_len),
442
+ )
443
+ candidates = cur.fetchall()
444
+
445
+ # our own info
446
+ cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (node_id,))
447
+ this_row = cur.fetchone()
448
+ this_file_path = this_row[0] if this_row else None
449
+ this_file_hash = this_row[1] if this_row else None
450
+ this_chain_id = this_row[2] if this_row else None
451
+
452
+ results: list[dict] = []
453
+
454
+ for cand_id, start_page, max_page in candidates:
455
+ big_seq = self.get_page_hashes(cand_id)
456
+ # we already know indexing won't go out of range because of SQL condition
457
+ if big_seq[start_page:start_page + small_len] == small_seq:
458
+ # hydrate bigger file
459
+ cur.execute("SELECT file_path, file_hash, chain_id FROM nodes WHERE id = ?", (cand_id,))
460
+ big_row = cur.fetchone()
461
+ results.append({
462
+ "small_node_id": node_id,
463
+ "small_file_path": this_file_path,
464
+ "small_file_hash": this_file_hash,
465
+ "small_chain_id": this_chain_id,
466
+ "big_node_id": cand_id,
467
+ "big_file_path": big_row[0] if big_row else None,
468
+ "big_file_hash": big_row[1] if big_row else None,
469
+ "big_chain_id": big_row[2] if big_row else None,
470
+ })
471
+
472
+ return results
473
+
474
+ def mark_subdocument_duplicate(self, info: dict):
475
+ """
476
+ info must contain:
477
+ small_file_path, small_file_hash, big_file_path, big_file_hash,
478
+ small_chain_id, small_node_id
479
+ We'll just reuse your existing duplicates table.
480
+ """
481
+ self.insert_duplicate(
482
+ file_name=info["small_file_path"],
483
+ file_hash=info["small_file_hash"],
484
+ duplicate_of_file_name=info["big_file_path"],
485
+ duplicate_of_file_hash=info["big_file_hash"],
486
+ chain_id=info["small_chain_id"],
487
+ node_id=info["small_node_id"],
488
+ )
489
+
490
+ # --------------------------
491
+
492
+ def create_chain(self, name: Optional[str] = None) -> int:
493
+ cur = self.conn.cursor()
494
+ cur.execute("INSERT INTO chains (name) VALUES (?)", (name,))
495
+ if not self.conn.in_transaction:
496
+ self.conn.commit()
497
+ return cur.lastrowid
498
+
499
+ def delete_chain(self, chain_id: int):
500
+ cur = self.conn.cursor()
501
+ cur.execute("DELETE FROM nodes WHERE chain_id = ?", (chain_id,))
502
+ cur.execute("DELETE FROM chains WHERE id = ?", (chain_id,))
503
+ if not self.conn.in_transaction:
504
+ self.conn.commit()
505
+
506
+ def add_node(self, chain_id: int, file_root: str, file_path: str, file_size: int, file_hash: str,
507
+ position: str = "append", ref_node_id: Optional[int] = None,
508
+ metadata_json: Optional[str] = None) -> int:
509
+ """
510
+ ref_node_id : between and append is the node before the new addition, preprend is the node id prepended to
511
+ """
512
+ if not file_path.lower().endswith('.pdf'):
513
+ raise ValueError("Only PDF files (.pdf) are allowed.")
514
+ cur = self.conn.cursor()
515
+ created_at = datetime.datetime.now().isoformat()
516
+ # Find head/tail for prepend/append
517
+ if position == "prepend":
518
+ cur.execute("SELECT id FROM nodes WHERE chain_id = ? AND prev_id IS NULL", (chain_id,))
519
+ head = cur.fetchone()
520
+ prev_id = None
521
+ next_id = head[0] if head else None
522
+ # Update old head's prev_id
523
+ if head:
524
+ cur.execute("UPDATE nodes SET prev_id = NULL WHERE id = ?", (head[0],))
525
+ elif position == "append":
526
+ cur.execute("SELECT id FROM nodes WHERE chain_id = ? AND next_id IS NULL", (chain_id,))
527
+ tail = cur.fetchone()
528
+ prev_id = tail[0] if tail else None
529
+ next_id = None
530
+ # Update old tail's next_id
531
+ if tail:
532
+ cur.execute("UPDATE nodes SET next_id = NULL WHERE id = ?", (tail[0],))
533
+ elif position == "between":
534
+ if ref_node_id is None:
535
+ raise ValueError("ref_node_id must be provided for 'between' insertion.")
536
+ # Insert after ref_node_id
537
+ cur.execute("SELECT next_id FROM nodes WHERE id = ?", (ref_node_id,))
538
+ next_id = cur.fetchone()
539
+ next_id = next_id[0] if next_id else None
540
+ prev_id = ref_node_id
541
+ # Update links
542
+ cur.execute("UPDATE nodes SET next_id = NULL WHERE id = ?", (ref_node_id,))
543
+ if next_id:
544
+ cur.execute("UPDATE nodes SET prev_id = NULL WHERE id = ?", (next_id,))
545
+ else:
546
+ raise ValueError("position must be 'append', 'prepend', or 'between'.")
547
+ cur.execute("""
548
+ INSERT INTO nodes (chain_id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json)
549
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)
550
+ """, (chain_id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json))
551
+ node_id = cur.lastrowid
552
+ # Update neighbors
553
+ if position == "prepend" and next_id:
554
+ cur.execute("UPDATE nodes SET prev_id = ? WHERE id = ?", (node_id, next_id))
555
+ if position == "append" and prev_id:
556
+ cur.execute("UPDATE nodes SET next_id = ? WHERE id = ?", (node_id, prev_id))
557
+ if position == "between":
558
+ cur.execute("UPDATE nodes SET next_id = ? WHERE id = ?", (node_id, prev_id))
559
+ if next_id:
560
+ cur.execute("UPDATE nodes SET prev_id = ? WHERE id = ?", (node_id, next_id))
561
+ if not self.conn.in_transaction:
562
+ self.conn.commit()
563
+
564
+ try:
565
+ page_hashes = pdf_page_hashes_as_png(os.path.join(file_root, file_path))
566
+ self.insert_page_hashes(
567
+ node_id=node_id,
568
+ file_path=file_path,
569
+ page_hashes=page_hashes,
570
+ dpi=150,
571
+ render_format="PNG",
572
+ algo="sha256",
573
+ )
574
+
575
+ # 1) I am short -> check longer ones
576
+ longer_matches = self.find_bigger_files_containing_this_sequence(node_id)
577
+ for info in longer_matches:
578
+ self.mark_subdocument_duplicate(info)
579
+
580
+ # 2) I am long -> check shorter ones
581
+ shorter_matches = self.find_smaller_files_contained_in_this_sequence(node_id)
582
+ for info in shorter_matches:
583
+ self.mark_subdocument_duplicate(info)
584
+
585
+ except Exception as e:
586
+ logger.exception(f"Error computing/storing page hashes for {file_path}: {e}")
587
+ self.conn.rollback()
588
+ raise e
589
+
590
+ return node_id
591
+
592
+ def list_all_chains(self):
593
+ return [self.get_chain(chain['id']) for chain in self.find_chains()]
594
+
595
+ def get_chain(self, chain_id: int) -> List[Dict[str, Any]]:
596
+ cur = self.conn.cursor()
597
+ # Find head node
598
+ cur.execute("SELECT id FROM nodes WHERE chain_id = ? AND prev_id IS NULL", (chain_id,))
599
+ head = cur.fetchone()
600
+ if not head:
601
+ return []
602
+ node_id = head[0]
603
+ chain = []
604
+ while node_id:
605
+ cur.execute("SELECT id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json FROM nodes WHERE id = ?", (node_id,))
606
+ row = cur.fetchone()
607
+ if not row:
608
+ break
609
+ node = {
610
+ "id": row[0],
611
+ "file_path": row[1],
612
+ "file_size": row[2],
613
+ "file_hash": row[3],
614
+ "prev_id": row[4],
615
+ "next_id": row[5],
616
+ "created_at": row[6],
617
+ "metadata_json": row[7]
618
+ }
619
+ chain.append(node)
620
+ node_id = row[5] # next_id
621
+ return chain
622
+
623
+ def update_node(self, node_id: int, **fields):
624
+ cur = self.conn.cursor()
625
+ allowed = {"file_path", "file_size", "file_hash", "metadata_json"}
626
+ updates = []
627
+ values = []
628
+ for k, v in fields.items():
629
+ if k in allowed:
630
+ updates.append(f"{k} = ?")
631
+ values.append(v)
632
+ if not updates:
633
+ return
634
+ values.append(node_id)
635
+ cur.execute(f"UPDATE nodes SET {', '.join(updates)} WHERE id = ?", values)
636
+ if not self.conn.in_transaction:
637
+ self.conn.commit()
638
+
639
+ def delete_node(self, node_id: int):
640
+ cur = self.conn.cursor()
641
+ # Relink neighbors
642
+ cur.execute("SELECT prev_id, next_id FROM nodes WHERE id = ?", (node_id,))
643
+ row = cur.fetchone()
644
+ if row:
645
+ prev_id, next_id = row
646
+ if prev_id:
647
+ cur.execute("UPDATE nodes SET next_id = ? WHERE id = ?", (next_id, prev_id))
648
+ if next_id:
649
+ cur.execute("UPDATE nodes SET prev_id = ? WHERE id = ?", (prev_id, next_id))
650
+ cur.execute("DELETE FROM nodes WHERE id = ?", (node_id,))
651
+ # also delete page-hashes for this node
652
+ cur.execute("DELETE FROM page_hashes WHERE node_id = ?", (node_id,))
653
+ if not self.conn.in_transaction:
654
+ self.conn.commit()
655
+
656
+ def find_chains(self) -> List[Dict[str, Any]]:
657
+ cur = self.conn.cursor()
658
+ cur.execute("SELECT id, name FROM chains")
659
+ return [{"id": row[0], "name": row[1]} for row in cur.fetchall()]
660
+
661
+ def find_nodes(self, chain_id: int) -> List[Dict[str, Any]]:
662
+ cur = self.conn.cursor()
663
+ cur.execute("SELECT id, file_path, file_size, file_hash, prev_id, next_id, created_at, metadata_json FROM nodes WHERE chain_id = ?", (chain_id,))
664
+ return [
665
+ {
666
+ "id": row[0],
667
+ "file_path": row[1],
668
+ "file_size": row[2],
669
+ "file_hash": row[3],
670
+ "prev_id": row[4],
671
+ "next_id": row[5],
672
+ "created_at": row[6],
673
+ "metadata_json": row[7]
674
+ }
675
+ for row in cur.fetchall()
676
+ ]
677
+ def get_canonical_for_hash(self, file_hash: str):
678
+ """
679
+ Return (id, file_path, chain_id) of the earliest node we have for this hash.
680
+ """
681
+ cur = self.conn.cursor()
682
+ cur.execute(
683
+ """
684
+ SELECT id, file_path, chain_id
685
+ FROM nodes
686
+ WHERE file_hash = ?
687
+ ORDER BY id ASC
688
+ LIMIT 1
689
+ """,
690
+ (file_hash,),
691
+ )
692
+ return cur.fetchone()
693
+ def insert_duplicate(self, file_name, file_hash, duplicate_of_file_name, duplicate_of_file_hash, chain_id=None, node_id=None):
694
+ cur = self.conn.cursor()
695
+ created_at = datetime.datetime.now().isoformat()
696
+ cur.execute("""
697
+ INSERT INTO duplicates (file_name, file_hash, duplicate_of_file_name, duplicate_of_file_hash, chain_id, node_id, created_at)
698
+ VALUES (?, ?, ?, ?, ?, ?, ?)
699
+ """, (file_name, file_hash, duplicate_of_file_name, duplicate_of_file_hash, chain_id, node_id, created_at))
700
+ if not self.conn.in_transaction:
701
+ self.conn.commit()
702
+
703
+ def find_duplicate_by_name(self, file_name):
704
+ cur = self.conn.cursor()
705
+ cur.execute("SELECT * FROM duplicates WHERE file_name = ?", (file_name,))
706
+ return cur.fetchall()
707
+
708
+ def find_duplicate_by_hash(self, file_hash):
709
+ cur = self.conn.cursor()
710
+ cur.execute("SELECT * FROM duplicates WHERE file_hash = ?", (file_hash,))
711
+ return cur.fetchall()
712
+ def get_canonical_page_statistics(self) -> dict:
713
+ """
714
+ Compute statistics on canonical (non-duplicate) documents.
715
+
716
+ Returns:
717
+ dict with:
718
+ - total_canonical_docs (int)
719
+ - total_pages (int)
720
+ - details (list of dict) → each with {file_path, page_count}
721
+
722
+ Canonical = nodes whose file_path NOT in duplicates.file_name.
723
+ """
724
+ cur = self.conn.cursor()
725
+ cur.execute("""
726
+ SELECT
727
+ n.file_path,
728
+ COUNT(ph.page_num) AS page_count
729
+ FROM nodes AS n
730
+ JOIN page_hashes AS ph
731
+ ON ph.node_id = n.id
732
+ WHERE n.file_path NOT IN (
733
+ SELECT d.file_name FROM duplicates AS d
734
+ WHERE d.file_name IS NOT NULL
735
+ )
736
+ GROUP BY n.id
737
+ ORDER BY page_count DESC;
738
+ """)
739
+ rows = cur.fetchall()
740
+
741
+ stats = {
742
+ "total_canonical_docs": len(rows),
743
+ "total_pages": sum(r[1] for r in rows),
744
+ "details": [{"file_path": r[0], "page_count": r[1]} for r in rows]
745
+ }
746
+ return stats
747
+ def name_exists(self, file_name):
748
+ cur = self.conn.cursor()
749
+ cur.execute("SELECT 1 FROM nodes WHERE file_path = ? LIMIT 1", (file_name,))
750
+ if cur.fetchone():
751
+ print('match from nodes.file_path')
752
+ return True
753
+ cur.execute("SELECT 1 FROM duplicates WHERE file_name = ? LIMIT 1", (file_name,))
754
+ if cur.fetchone():
755
+ print('match from duplicates.file_name')
756
+ return True
757
+ return False
758
+
759
+ def hash_exists(self, file_hash):
760
+ cur = self.conn.cursor()
761
+ cur.execute("SELECT 1 FROM nodes WHERE file_hash = ? LIMIT 1", (file_hash,))
762
+ if cur.fetchone():
763
+ return True
764
+ cur.execute("SELECT 1 FROM duplicates WHERE file_hash = ? LIMIT 1", (file_hash,))
765
+ if cur.fetchone():
766
+ return True
767
+ return False
768
+
769
+ def is_duplicate_name_or_hash(self, file_name, file_hash):
770
+ return self.name_exists(file_name) or self.hash_exists(file_hash)
771
+
772
+ def close(self):
773
+ self.conn.close()
774
+ def is_canonical_by_name(self, file_name: str) -> bool:
775
+ """
776
+ Decide if this file_path is the canonical/kept one.
777
+
778
+ Rules (based on your current usage):
779
+ - if this file_name appears in duplicates.file_name -> NOT canonical
780
+ (because you always put the thing-to-hide on the left)
781
+ - else -> canonical
782
+ - if we also want to be careful with old rows, we can fallback to hash
783
+ """
784
+ cur = self.conn.cursor()
785
+
786
+ # 1) if it's explicitly marked as a duplicate, it's not canonical
787
+ cur.execute("SELECT 1 FROM duplicates WHERE file_name = ? LIMIT 1", (file_name,))
788
+ if cur.fetchone():
789
+ return False
790
+
791
+ # 2) try to get its hash from nodes
792
+ cur.execute("""
793
+ SELECT file_hash
794
+ FROM nodes
795
+ WHERE file_path = ?
796
+ LIMIT 1
797
+ """, (file_name,))
798
+ row = cur.fetchone()
799
+ if row:
800
+ file_hash = row[0]
801
+ else:
802
+ # maybe it only lives in duplicates table (rare, but let's check)
803
+ cur.execute("""
804
+ SELECT file_hash
805
+ FROM duplicates
806
+ WHERE file_name = ?
807
+ LIMIT 1
808
+ """, (file_name,))
809
+ row = cur.fetchone()
810
+ if row:
811
+ file_hash = row[0]
812
+ else:
813
+ # not in nodes, not in duplicates: we don't know it -> treat as canonical/new
814
+ return True
815
+
816
+ # 3) (optional) if you want to be extra safe:
817
+ # check if there is an entry "some other file" -> this hash
818
+ # but because your direction is always "duplicate file_name -> canonical duplicate_of_file_name",
819
+ # step (1) is usually enough.
820
+ return True
821
+ def get_non_duplicate_files(self) -> list[dict]:
822
+ """
823
+ Return all nodes that are not listed as duplicates.
824
+ """
825
+ cur = self.conn.cursor()
826
+ cur.execute("""
827
+ SELECT n.id, n.file_path, n.file_hash, n.chain_id, n.created_at
828
+ FROM nodes AS n
829
+ WHERE n.file_hash NOT IN (
830
+ SELECT d.file_hash FROM duplicates AS d
831
+ )
832
+ ORDER BY n.created_at ASC;
833
+ """)
834
+ rows = cur.fetchall()
835
+ return [
836
+ {"id": r[0], "file_path": r[1], "file_hash": r[2],
837
+ "chain_id": r[3], "created_at": r[4]}
838
+ for r in rows
839
+ ]
840
+
841
+ # --------------------------------------------------------------------------------
842
+ # The rest is your original LLM + ingestion logic, mostly unchanged
843
+ # --------------------------------------------------------------------------------
844
+
845
+ class FileMetadata(BaseModel):
846
+ "metadata about a file"
847
+ document_name: str = Field(..., description="document id")
848
+ file_size: int = Field(..., description = "file size in bytes")
849
+ date_modified : str = Field(..., description = "date modified / copied to the analysis system")
850
+ date_created : str = Field(..., description = "date created")
851
+ file_hash : Optional[str] = Field(default = None, description = "file hash")
852
+ def __hash__(self):
853
+ if self.file_hash is None:
854
+ raise ValueError("file_hash must be provided for the model to be hashable")
855
+ return int(self.file_hash, base=16)
856
+
857
+ @memory.cache
858
+ def get_file_hash(file_path, last_modified, size_bytes, algorithm="sha256", block_size=65536, ):
859
+ """Compute a hash for the given file using the specified algorithm."""
860
+ h = hashlib.new(algorithm)
861
+ with open(file_path, "rb") as f:
862
+ for chunk in iter(lambda: f.read(block_size), b""):
863
+ h.update(chunk)
864
+ return h.hexdigest()
865
+
866
+ def get_folder_metadata(folder_path, hash_algorithm="sha256"):
867
+ metadata_list = []
868
+ for root, dirs, files in os.walk(folder_path):
869
+ for name in files:
870
+ file_path = os.path.join(root, name)
871
+ stats = os.stat(file_path)
872
+ file_hash = get_file_hash(file_path, last_modified = stats.st_mtime,
873
+ size_bytes= stats.st_size,
874
+ algorithm=hash_algorithm)
875
+ metadata_list.append(
876
+ FileMetadata(
877
+ document_name = name,
878
+ file_size = stats.st_size,
879
+ file_hash = file_hash,
880
+ date_modified= str(datetime.datetime.fromtimestamp(stats.st_mtime)),
881
+ date_created =str(datetime.datetime.fromtimestamp(stats.st_birthtime ))
882
+ )
883
+ )
884
+ return metadata_list
885
+
886
+ class FileVersion(BaseModel):
887
+ "representing a file version"
888
+ filename: str = Field(..., description = "the filename of the current version")
889
+ prev: Optional[str] = Field(..., description = "The file name of the previous version. If it is brandnew not superceding/ overwriting any other, set None/Null. ")
890
+ supercede_reason : str = Field(..., description = "Why this version fully supercede the previous.")
891
+ supercede_mode : Literal["Duplicate", "FileEdit", "ContractUpdate", "N/A"] = Field(..., description = """
892
+ the mode of superceding, Duplicate means the file is just a duplicated copy.
893
+ FileEdit is small edit that does not change any activated contract terms. FileEdit is applicable to any contract file change across drafts.
894
+ Contract update is the update that truely reflect the signed contract with intention to update.
895
+ "N/A" when there is nothing to supercede when it is the first version. """)
896
+
897
+ class VersionChain(BaseModel):
898
+ "A linked list that the represent the evolution of a file. "
899
+ chain: list[FileVersion] = Field(..., description = "A single file mutation chain, first element is the root, subsequent files is the mutated version of the preceding. ")
900
+ pass
901
+
902
+ class FileVersionChainingResponse(BaseModel):
903
+ "Answer response format of file version chaining"
904
+ reasoning: str = Field(..., description = "reasoning at overall response level of thinking")
905
+ chains: list[VersionChain] = Field(..., description = "a array/ list of linked list with first element the raw verion before any changes. The next element is the file version that overwrites/ supercede the previous version. ")
906
+ root_agreement : str = Field(..., description = "The file name of the origin/master agreement that covers everything before any term variations are applied to. ")
907
+ pass
908
+
909
+ @memory.cache
910
+ def version_chain(metadata_list: list[FileMetadata], model = 'gemini-2.5-pro', attempt = 0):
911
+ from langchain_google_genai import ChatGoogleGenerativeAI
912
+ from langchain_core.messages import HumanMessage, SystemMessage
913
+
914
+ messages = [SystemMessage(
915
+ "You need to provide version chain to a list of given file data. You need to sort out which one is superceded by which. "
916
+ "You are given a list of file metadata and output file versioning chains. You need to reason through why one is before or after another file. "
917
+ "You need to cover all files, if it has no previous version. Give result as one or more file version chains. "
918
+ "There are some subtle file hint they are the same. Sometimes the same file chain do not share same name but some part of the file is the same. "
919
+ "The same file version chain may have the file renamed but preserve similar meaning, partially renamed or truncated. "
920
+ "For example a file maybe called in its original version `schedule 1.pdf` but the next version available is not will be schedule 1v4_final_final.pdf with potentially v2 (version 2) and v3 missing and a strange final suffix appended. \n"
921
+ "Of course there can be some simpler case maybe just as simple as `MSA Company name v1.pdf` and the next is simply MSA Company name v2.pdf"
922
+ "Your job is to track these chains as much as possible. "
923
+ "Example, the evolution / edit / terms changes of scheule A should be distinct from the changes of schedule B. "
924
+ "Remember, we are dealing with version, do not add schedule 3 after schedule 2, but only add schedule 3 version 2 after schedule 3. The do not use reading order (e.g. section 4 after section 3). "
925
+ "We are only concerned with estimating changes made to the same section. If it is a separate section/ contract. Or the varaition is dealing with different terms, they should follow its own new chain. "
926
+ "In case the file is purely copy and pasted with a different number at the end, try to choose latest, with largest number as the newer version. "
927
+ "The given file list comes from os files. Each file must show up exactly once in the answer chain. "
928
+ ),
929
+ HumanMessage(f"{[i.model_dump(exclude = ['file_hash']) for i in metadata_list]}")]
930
+
931
+ llm = ChatGoogleGenerativeAI(model = model)
932
+ cnt = 0
933
+ max_cnt = 4
934
+ while cnt <= max_cnt:
935
+ chaining_result = llm.with_structured_output(FileVersionChainingResponse, include_raw = True).invoke(messages)
936
+ chains = chaining_result['parsed'].model_dump()['chains']
937
+ files = [cc['filename'] for c in chains for cc in c['chain']]
938
+ from collections import Counter
939
+ counter = Counter(files)
940
+ duplicated = []
941
+ for i in counter.most_common():
942
+ if i[1] > 1:
943
+ duplicated.append(i)
944
+ if duplicated:
945
+ messages.append(f"Trial answer : the field chains= {chains}")
946
+ messages.append(f"files duplicated: {str(duplicated)}")
947
+ else:
948
+ break
949
+ cnt += 1
950
+ if cnt == max_cnt:
951
+ raise(Exception(f"Max retry {max_cnt} reached" ))
952
+ return chaining_result['parsed'].model_dump()['chains']
953
+
954
+ @memory.cache
955
+ def dedup_llm_pick_newest(meta_list_dumped, model = 'gemini-2.5-flash'):
956
+ from langchain_google_genai import ChatGoogleGenerativeAI
957
+ from langchain_core.messages import HumanMessage, SystemMessage
958
+
959
+ representative_file_name = None
960
+ name_list = [i['document_name'] for i in meta_list_dumped]
961
+
962
+ class DedupResponse(BaseModel):
963
+ reasoning: str = Field(..., description = "The reasoning steps to the final answer. ")
964
+ representative_file_name: str = Field(..., description = "The file name of the most representative file. ")
965
+
966
+ cnt = 0
967
+ while representative_file_name not in name_list:
968
+ messages = [SystemMessage("You are given a list of files sharing the same file hash and you need to choose one that is the most representative. "),
969
+ HumanMessage(f"{meta_list_dumped}")]
970
+ llm = ChatGoogleGenerativeAI(model = model)
971
+ res = llm.with_structured_output(DedupResponse, include_raw = True).invoke(messages)
972
+ if not res.get('parsing_error'):
973
+ representative_file_name = res['parsed'].representative_file_name
974
+ if representative_file_name not in name_list:
975
+ messages.append(SystemMessage("Error, the answer mentioned file name does not exist in any of the file at all. Only choose the exact file name in the list. "))
976
+ else:
977
+ return representative_file_name
978
+ else:
979
+ pass
980
+ cnt +=1
981
+ if cnt > 5:
982
+ raise Exception("LLM error, retried max reached and cannot choose dedup filename.")
983
+ pass
984
+
985
+ def dedup(list_meta: list[FileMetadata]):
986
+ d: dict[str, set[FileMetadata]] = {}
987
+ for meta in list_meta:
988
+ m = meta.model_dump(exclude = ['file_hash'])
989
+ if meta.file_hash not in d:
990
+ d[meta.file_hash] = [m]
991
+ else:
992
+ d[meta.file_hash].append(m)
993
+ res = {}
994
+ dup_res = []
995
+ for hs, meta_set in d.items():
996
+ if len(meta_set) > 1:
997
+ dup = {i['document_name']: i for i in meta_set}
998
+ logger.info(f"deduping set {';'.join(list(dup))}")
999
+ src = dedup_llm_pick_newest(meta_set)
1000
+ res[hs] = dup.pop(src)
1001
+ dup_res.extend([FileVersion.model_validate({"filename": d['document_name'], "prev": src, "supercede_reason": f"duplicate of file {src}", "supercede_mode" : "Duplicate"}) for d in dup.values()])
1002
+ else:
1003
+ res[hs] = list(meta_set)[0]
1004
+ return res, dup_res
1005
+
1006
+ def check_missing_db(db: VersionChainDB, d_hash_to_meta):
1007
+ # Get all filenames in DB
1008
+ all_db_files = set()
1009
+ for chain in db.find_chains():
1010
+ for node in db.get_chain(chain['id']):
1011
+ all_db_files.add(node['file_path'])
1012
+ available_files = set(meta.document_name for meta in d_hash_to_meta.values())
1013
+ missing = available_files - all_db_files
1014
+ return missing
1015
+
1016
+ class FileAddition(FileVersion):
1017
+ "file addition"
1018
+ filename: str = Field(..., description = 'the file name of the new file')
1019
+ prev: Optional[str] = Field(None, description = 'the file name its previous version, set Null or None if it belong to new chain')
1020
+ next: Optional[str] = Field(None, description = 'the file name its next version, use only when this file is inserted to the beginning of an existing file version chain. ')
1021
+ supercede_reason : str = Field(..., description = "Why this version fully supercede the previous.")
1022
+ supercede_mode : Literal["Duplicate", "FileEdit", "ContractUpdate", "N/A"] = Field(..., description = """
1023
+ the mode of superceding, Duplicate means the file is just a duplicated copy.
1024
+ FileEdit is small edit that does not change any activated contract terms. FileEdit is applicable to any contract file change across drafts.
1025
+ Contract update is the update that truely reflect the signed contract with intention to update.
1026
+ "N/A" when there is nothing to supercede when it is the first version. """)
1027
+ @model_validator(mode = 'after')
1028
+ def only_one_prev_next_none(self):
1029
+ if (self.prev is None) or (self.next is None):
1030
+ return self
1031
+ else:
1032
+ raise Exception('at most one of prev or next is None')
1033
+
1034
+ class AddFilesResponse(BaseModel):
1035
+ "the result of the file addition"
1036
+ reasoning: str = Field(..., description = 'reasoning in top level perspective')
1037
+ additions : list[FileAddition] = Field(..., description = 'list of file additions')
1038
+
1039
+ @memory.cache
1040
+ def add_new_file_to_existing_chains(chains, all_d_hash_to_meta: dict[str, FileMetadata], new_file_name, model = 'gemini-2.5-pro', attempt = 0 ) -> AddFilesResponse:
1041
+ from langchain_core.messages import HumanMessage, SystemMessage
1042
+ messages = [SystemMessage("Given existing file versioning chain and a new file with metadata of all files, you need to decide the position of the new file in the version chain. "
1043
+ "All files have distinct file hash. "
1044
+ "if existing chain is file_v1, file_v1.1, file2 and a new file file_v_1.5 and you think it should be inserted between file_v1.1 and file2. "
1045
+ "Remember, we are dealing with version, do not add schedule 3 after schedule 2, but only add schedule 3 version 2 after schedule 3. The do not use reading order (e.g. section 4 after section 3). "
1046
+ "We are only concerned with estimating changes made to the same section. If it is a separate section/ contract. Or the varaition is dealing with different terms, they should follow its own new chain. "
1047
+ "then set the prev of file_v_1.5 to be file_v1.1, remember if it is newer than one but looks older than another, you need to reason about the possibility of insertion instead of appending directly at the end. "),
1048
+ HumanMessage(f"Existing version chain: {chains}" + "\n\n" +
1049
+ f"Metadata file-hash to file-meta lookup for all files: {[i.model_dump(exclude = set(['file_hash'])) for i in all_d_hash_to_meta.values()]}" + "\n\n" +
1050
+ f"New file to add: {new_file_name}"
1051
+ ),
1052
+
1053
+ ]
1054
+ from langchain_google_genai import ChatGoogleGenerativeAI
1055
+ max_retry = 3
1056
+ n = 0
1057
+ while True:
1058
+ llm = ChatGoogleGenerativeAI(model = model)
1059
+ res = llm.with_structured_output(AddFilesResponse, include_raw = True).invoke(messages)
1060
+ if res.get("parsing_error"):
1061
+ n += 1
1062
+ if n >= max_retry:
1063
+ raise Exception(f"Max Retry reached {max_retry}")
1064
+ else:
1065
+ return res.get('parsed')
1066
+
1067
+ def apply_chain_updates_db(db: VersionChainDB, updates, all_meta: dict[str, FileMetadata]):
1068
+ additions: list[FileAddition] = updates.additions
1069
+ changes_made = {"added_nodes": [], "added_chains": []}
1070
+ for addition in additions:
1071
+ meta = all_meta.get(addition.filename)
1072
+ if addition.prev is None and addition.next is None:
1073
+ # New chain
1074
+ chain_id = db.create_chain(name=f"Chain_for_{addition.filename}")
1075
+ changes_made["added_chains"].append(chain_id)
1076
+ changes_made["added_nodes"].append(db.add_node(
1077
+ chain_id=chain_id,
1078
+ file_path=addition.filename,
1079
+ file_size=meta.file_size if meta else 0,
1080
+ file_hash=(meta.file_hash if (meta and meta.file_hash is not None) else ""),
1081
+ position="append",
1082
+ metadata_json=str(addition.model_dump())
1083
+ ))
1084
+ else:
1085
+ # Find the chain and reference node for insertion
1086
+ target_chain_id = None
1087
+ ref_node_id = None
1088
+ position = None
1089
+ for chain in db.find_chains():
1090
+ nodes = db.get_chain(chain['id'])
1091
+ # Insert at beginning (prepend)
1092
+ if addition.prev is None and addition.next is not None:
1093
+ for node in nodes:
1094
+ if node['file_path'] == addition.next:
1095
+ target_chain_id = chain['id']
1096
+ ref_node_id = node['id']
1097
+ position = "prepend"
1098
+ if node['prev_id']: # next existed, so it's actually "between"
1099
+ position = "between"
1100
+ else:
1101
+ position = "prepend"
1102
+ break
1103
+ # Insert at end (append)
1104
+ elif addition.prev is not None and addition.next is None:
1105
+ for node in nodes:
1106
+ if node['file_path'] == addition.prev:
1107
+ target_chain_id = chain['id']
1108
+ ref_node_id = node['id']
1109
+ if node['next_id']:
1110
+ position = "between"
1111
+ else:
1112
+ position = "append"
1113
+ break
1114
+ # Insert between
1115
+ elif addition.prev is not None and addition.next is not None:
1116
+ for node in nodes:
1117
+ if node['file_path'] == addition.prev:
1118
+ target_chain_id = chain['id']
1119
+ ref_node_id = node['id']
1120
+ position = "between"
1121
+ break
1122
+ if target_chain_id is not None:
1123
+ break
1124
+ if target_chain_id is not None and ref_node_id is not None and position is not None:
1125
+ changes_made["added_nodes"].append(db.add_node(
1126
+ chain_id=target_chain_id,
1127
+ file_path=addition.filename,
1128
+ file_size=meta.file_size if meta else 0,
1129
+ file_hash=(meta.file_hash if (meta and meta.file_hash is not None) else ""),
1130
+ position=position,
1131
+ ref_node_id=ref_node_id,
1132
+ metadata_json=str(addition.model_dump())
1133
+ ))
1134
+ # if position is None, skip for now
1135
+ return changes_made
1136
+
1137
+
1138
+ # ---------------------------
1139
+ # optional: backfill
1140
+ # ---------------------------
1141
+ def backfill_page_hashes(db: VersionChainDB, file_root, dpi: int = 300):
1142
+ """
1143
+ Run once on an existing DB to populate page_hashes and page-based duplicates.
1144
+ """
1145
+ cur = db.conn.cursor()
1146
+ cur.execute("SELECT id, file_path FROM nodes")
1147
+ rows = cur.fetchall()
1148
+ for node_id, file_path in rows:
1149
+ full_path = os.path.join(file_root, file_path)
1150
+ if not file_path.lower().endswith(".pdf"):
1151
+ continue
1152
+ if not os.path.exists(full_path):
1153
+ continue
1154
+ try:
1155
+ ph = pdf_page_hashes_as_png(full_path, dpi=dpi)
1156
+ db.insert_page_hashes(node_id, file_path, ph, dpi=dpi)
1157
+ contained_in = db.find_bigger_files_containing_this_sequence(node_id)
1158
+ for info in contained_in:
1159
+ db.mark_subdocument_duplicate(info)
1160
+ except Exception as e:
1161
+ logger.exception(f"Backfill page-hash failed for {file_path}: {e}")
1162
+ raise e
1163
+
1164
+
1165
+ if __name__ == "__main__":
1166
+ from ..pdf2png import RawFileLoader
1167
+ import pandas as pd
1168
+ in_compare_root = os.path.join('..', 'doc_data', 'raw_documents')
1169
+ loader = RawFileLoader(
1170
+ env_flist_path=None,
1171
+ walk_root=os.path.join('..', 'doc_data', 'raw_documents', 'Samples - 9 Oct 2025'),
1172
+ compare_root=in_compare_root,
1173
+ include=['dirs']
1174
+ )
1175
+ filechain_folder_root = os.path.join('..', 'doc_data', 'file_version_chains')
1176
+
1177
+ for rel_in_path in loader:
1178
+ foldername = os.path.join(in_compare_root, rel_in_path)
1179
+ out_folder = os.path.join(filechain_folder_root, rel_in_path)
1180
+ if not os.path.exists(out_folder):
1181
+ os.makedirs(out_folder)
1182
+ folder_meta = get_folder_metadata(foldername)
1183
+ print(folder_meta)
1184
+
1185
+ db = VersionChainDB(db_path=os.path.join(out_folder, "file_index.sqlite"))
1186
+
1187
+ # 1) first-level name/hash dedup against existing DB rows
1188
+ filtered_meta = []
1189
+ for m in folder_meta:
1190
+ if db.name_exists(m.document_name):
1191
+ # If name exists, skip entirely
1192
+ continue
1193
+ elif db.hash_exists(m.file_hash):
1194
+ # If name does NOT exist, but hash exists, insert duplicate
1195
+ canon = db.get_canonical_for_hash(m.file_hash)
1196
+ if canon:
1197
+ canon_id, canon_path, canon_chain_id = canon
1198
+ db.insert_duplicate(
1199
+ file_name=m.document_name, # this new one is the duplicate
1200
+ file_hash=m.file_hash,
1201
+ duplicate_of_file_name=canon_path, # point to the canonical existing one
1202
+ duplicate_of_file_hash=m.file_hash,
1203
+ chain_id=canon_chain_id,
1204
+ node_id=None, # we don't have a node_id for the new one yet
1205
+ )
1206
+ db.conn.commit()
1207
+ continue
1208
+ else:
1209
+ # If neither exists, process as new file
1210
+ filtered_meta.append(m)
1211
+
1212
+ # 2) LLM-driven dedup for the new batch
1213
+ d_hash_to_meta, dup = dedup(filtered_meta)
1214
+ df_dup = pd.DataFrame([d.model_dump(exclude=['file_hash']) for d in dup])
1215
+ d_hash_to_meta = {k: FileMetadata.model_validate(v) for k, v in d_hash_to_meta.items()}
1216
+ df_to_concat = [df_dup]
1217
+ attempt = 0
1218
+ chains_response = version_chain(list(d_hash_to_meta.values()), attempt=attempt)
1219
+ chains = chains_response
1220
+ if attempt == 0:
1221
+ # emulate hallucination case when a file meta missing
1222
+ popped_chain = chains.pop(-1)
1223
+ for chain in chains:
1224
+ if len(chain['chain']) > 3:
1225
+ popped_doc = chain['chain'].pop(1)
1226
+ break
1227
+
1228
+ filename_to_meta = {i.document_name: i for i in filtered_meta}
1229
+ try:
1230
+ db.conn.execute("BEGIN")
1231
+
1232
+ # persist each duplicate
1233
+ for file_version in dup:
1234
+ dup_meta = next((m for m in filtered_meta if m.document_name == file_version.filename), None)
1235
+ rep_meta = next((m for m in filtered_meta if m.document_name == file_version.prev), None)
1236
+ db.insert_duplicate(
1237
+ file_name=file_version.filename,
1238
+ file_hash=dup_meta.file_hash if dup_meta else None,
1239
+ duplicate_of_file_name=file_version.prev,
1240
+ duplicate_of_file_hash=rep_meta.file_hash if rep_meta else None,
1241
+ chain_id=None,
1242
+ node_id=None
1243
+ )
1244
+
1245
+ # persist each chain
1246
+ for chain_dict in chains_response:
1247
+ chain_id = db.create_chain(name=str(uuid.uuid1()))
1248
+ for file_version in chain_dict['chain']:
1249
+ meta = filename_to_meta.get(file_version['filename'])
1250
+ db.add_node(
1251
+ chain_id=chain_id,
1252
+ file_path=file_version['filename'],
1253
+ file_size=meta.file_size if meta else 0,
1254
+ file_hash=(meta.file_hash if (meta and meta.file_hash is not None) else ""),
1255
+ position="append",
1256
+ metadata_json=str(file_version)
1257
+ )
1258
+
1259
+ db.conn.commit()
1260
+ except Exception as e:
1261
+ db.conn.rollback()
1262
+ print(f"Error inserting chain: {e}")
1263
+ raise e
1264
+
1265
+ df = pd.DataFrame([j for i in chains for j in i['chain']])
1266
+ df_to_concat.append(df)
1267
+ missing = check_missing_db(db, d_hash_to_meta)
1268
+ max_retry = 6
1269
+ while missing:
1270
+ updates = add_new_file_to_existing_chains(chains, d_hash_to_meta, missing, attempt=attempt)
1271
+ apply_chain_updates_db(db, updates, filename_to_meta)
1272
+ missing = check_missing_db(db, d_hash_to_meta)
1273
+ attempt += 1
1274
+ if attempt > max_retry:
1275
+ raise Exception(f"Max retry {max_retry} reached without fixing version chain")
1276
+ import pathlib
1277
+ df_out = pd.concat(df_to_concat)
1278
+ df_out.to_csv(pathlib.Path(foldername).parts[-1] + '.csv')