@adeu/core 1.22.0 → 1.23.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.cjs CHANGED
@@ -45,6 +45,7 @@ __export(index_exports, {
45
45
  extract_outline: () => extract_outline,
46
46
  finalize_document: () => finalize_document,
47
47
  generate_edits_from_text: () => generate_edits_from_text,
48
+ generate_structured_edits: () => generate_structured_edits,
48
49
  identifyEngine: () => identifyEngine,
49
50
  paginate: () => paginate,
50
51
  split_structural_appendix: () => split_structural_appendix,
@@ -310,6 +311,13 @@ var DocumentObject = class _DocumentObject {
310
311
  async save() {
311
312
  for (const part of this.pkg.parts) {
312
313
  let xmlStr = serializeXml(part._element.ownerDocument || part._element);
314
+ if (xmlStr.includes("w16du:") && !xmlStr.includes("xmlns:w16du=")) {
315
+ part._element.setAttribute(
316
+ "xmlns:w16du",
317
+ "http://schemas.microsoft.com/office/word/2023/wordml/word16du"
318
+ );
319
+ xmlStr = serializeXml(part._element.ownerDocument || part._element);
320
+ }
313
321
  if (!xmlStr.startsWith("<?xml")) {
314
322
  xmlStr = '<?xml version="1.0" encoding="UTF-8" standalone="yes"?>\n' + xmlStr;
315
323
  }
@@ -897,7 +905,27 @@ var QN_W_OUTLINELVL = "w:outlineLvl";
897
905
  var QN_W_NUMPR = "w:numPr";
898
906
  var QN_W_NUMID = "w:numId";
899
907
  var QN_W_ILVL = "w:ilvl";
908
+ var QN_W_DRAWING = "w:drawing";
909
+ var QN_W_OBJECT = "w:object";
910
+ var QN_W_PICT = "w:pict";
911
+ var QN_WP_DOCPR = "wp:docPr";
912
+ var QN_V_IMAGEDATA = "v:imagedata";
913
+ var QN_O_TITLE = "o:title";
900
914
  var _CUSTOM_HEADING_NAME_RE = /Heading[ ]?([1-6])(?![0-9])/;
915
+ function _resolve_package(obj) {
916
+ let cur = obj;
917
+ const seen = /* @__PURE__ */ new Set();
918
+ while (cur && !seen.has(cur)) {
919
+ seen.add(cur);
920
+ if (cur.package) return cur.package;
921
+ if (cur.pkg) return cur.pkg;
922
+ if (cur.part && (cur.part.package || cur.part.pkg)) {
923
+ return cur.part.package || cur.part.pkg;
924
+ }
925
+ cur = cur._parent || cur.part || null;
926
+ }
927
+ return null;
928
+ }
901
929
  function _get_style_cache(part) {
902
930
  const pkg = part.package || part.pkg || (part.part ? part.part.pkg : null);
903
931
  if (pkg && pkg._adeu_style_cache) {
@@ -931,6 +959,8 @@ function _get_style_cache(part) {
931
959
  const based_on_el = findChild(s, "w:basedOn");
932
960
  const based_on = based_on_el ? based_on_el.getAttribute("w:val") : null;
933
961
  let outline_lvl = null;
962
+ let num_id = null;
963
+ let num_ilvl = null;
934
964
  const pPr = findChild(s, "w:pPr");
935
965
  if (pPr) {
936
966
  const oLvl = findChild(pPr, "w:outlineLvl");
@@ -938,6 +968,19 @@ function _get_style_cache(part) {
938
968
  const val = oLvl.getAttribute("w:val");
939
969
  if (val && /^\d+$/.test(val)) outline_lvl = parseInt(val, 10);
940
970
  }
971
+ const numPr = findChild(pPr, "w:numPr");
972
+ if (numPr) {
973
+ const numId_el = findChild(numPr, "w:numId");
974
+ if (numId_el) {
975
+ const n_val = numId_el.getAttribute("w:val");
976
+ if (n_val && n_val !== "0") num_id = n_val;
977
+ }
978
+ const ilvl_el = findChild(numPr, "w:ilvl");
979
+ if (ilvl_el) {
980
+ const i_val = ilvl_el.getAttribute("w:val");
981
+ if (i_val && /^\d+$/.test(i_val)) num_ilvl = parseInt(i_val, 10);
982
+ }
983
+ }
941
984
  }
942
985
  let bold = null;
943
986
  const rPr = findChild(s, "w:rPr");
@@ -948,23 +991,46 @@ function _get_style_cache(part) {
948
991
  bold = val !== "0" && val !== "false" && val !== "off";
949
992
  }
950
993
  }
951
- raw_styles[s_id] = { name, based_on, outline_level: outline_lvl, bold };
994
+ raw_styles[s_id] = {
995
+ name,
996
+ based_on,
997
+ outline_level: outline_lvl,
998
+ bold,
999
+ num_id,
1000
+ num_ilvl
1001
+ };
952
1002
  }
953
1003
  const resolve_style = (s_id, visited) => {
954
1004
  if (cache[s_id]) return cache[s_id];
955
1005
  if (visited.has(s_id) || !raw_styles[s_id])
956
- return { name: s_id, outline_level: null, bold: false };
1006
+ return {
1007
+ name: s_id,
1008
+ outline_level: null,
1009
+ bold: false,
1010
+ num_id: null,
1011
+ num_ilvl: null
1012
+ };
957
1013
  visited.add(s_id);
958
1014
  const raw = raw_styles[s_id];
959
1015
  const based_on_id = raw.based_on;
960
1016
  let o_lvl = raw.outline_level;
961
1017
  let bold_val = raw.bold !== null ? raw.bold : false;
1018
+ let n_id = raw.num_id;
1019
+ let n_ilvl = raw.num_ilvl;
962
1020
  if (based_on_id) {
963
1021
  const parent = resolve_style(based_on_id, visited);
964
1022
  if (o_lvl === null) o_lvl = parent.outline_level;
965
1023
  if (raw.bold === null) bold_val = parent.bold;
966
- }
967
- const resolved = { name: raw.name, outline_level: o_lvl, bold: bold_val };
1024
+ if (n_id === null) n_id = parent.num_id ?? null;
1025
+ if (n_ilvl === null) n_ilvl = parent.num_ilvl ?? null;
1026
+ }
1027
+ const resolved = {
1028
+ name: raw.name,
1029
+ outline_level: o_lvl,
1030
+ bold: bold_val,
1031
+ num_id: n_id,
1032
+ num_ilvl: n_ilvl
1033
+ };
968
1034
  cache[s_id] = resolved;
969
1035
  return resolved;
970
1036
  };
@@ -973,6 +1039,68 @@ function _get_style_cache(part) {
973
1039
  if (pkg) pkg._adeu_style_cache = result;
974
1040
  return result;
975
1041
  }
1042
+ function _get_numbering_cache(part) {
1043
+ const pkg = _resolve_package(part);
1044
+ if (!pkg) return {};
1045
+ if (pkg._adeu_numbering_cache) return pkg._adeu_numbering_cache;
1046
+ const cache = {};
1047
+ let numbering_root = null;
1048
+ try {
1049
+ const numberingPart = (pkg.parts || []).find(
1050
+ (p) => String(p.partname).endsWith("/numbering.xml")
1051
+ );
1052
+ if (numberingPart) numbering_root = numberingPart._element;
1053
+ } catch {
1054
+ numbering_root = null;
1055
+ }
1056
+ if (numbering_root) {
1057
+ const abstract_fmts = {};
1058
+ for (const abstract of findAllDescendants(numbering_root, "w:abstractNum")) {
1059
+ const a_id = abstract.getAttribute("w:abstractNumId");
1060
+ if (a_id === null) continue;
1061
+ const lvl_map = {};
1062
+ for (const lvl of findAllDescendants(abstract, "w:lvl")) {
1063
+ const ilvl_val = lvl.getAttribute("w:ilvl");
1064
+ const fmt_el = findChild(lvl, "w:numFmt");
1065
+ if (ilvl_val !== null && /^-?\d+$/.test(ilvl_val) && fmt_el) {
1066
+ const fmt = fmt_el.getAttribute("w:val");
1067
+ if (fmt) lvl_map[parseInt(ilvl_val, 10)] = fmt;
1068
+ }
1069
+ }
1070
+ abstract_fmts[a_id] = lvl_map;
1071
+ }
1072
+ for (const num of findAllDescendants(numbering_root, "w:num")) {
1073
+ const n_id = num.getAttribute("w:numId");
1074
+ const a_ref = findChild(num, "w:abstractNumId");
1075
+ if (n_id === null || !a_ref) continue;
1076
+ const a_id = a_ref.getAttribute("w:val");
1077
+ if (a_id !== null && abstract_fmts[a_id] !== void 0) {
1078
+ cache[n_id] = abstract_fmts[a_id];
1079
+ }
1080
+ }
1081
+ }
1082
+ pkg._adeu_numbering_cache = cache;
1083
+ return cache;
1084
+ }
1085
+ function get_list_marker(paragraph_part, num_id, ilvl) {
1086
+ let fmt = null;
1087
+ if (num_id !== null && num_id !== void 0) {
1088
+ const lvl_map = _get_numbering_cache(paragraph_part)[num_id];
1089
+ if (lvl_map && Object.keys(lvl_map).length > 0) {
1090
+ fmt = lvl_map[ilvl] !== void 0 ? lvl_map[ilvl] : null;
1091
+ if (fmt === null) {
1092
+ for (let lookup = ilvl; lookup >= 0; lookup--) {
1093
+ if (lvl_map[lookup] !== void 0) {
1094
+ fmt = lvl_map[lookup];
1095
+ break;
1096
+ }
1097
+ }
1098
+ }
1099
+ }
1100
+ }
1101
+ if (fmt !== null && fmt !== "bullet") return "1. ";
1102
+ return "* ";
1103
+ }
976
1104
  function _detect_heading_level_from_name(name) {
977
1105
  if (!name) return null;
978
1106
  const match = name.match(_CUSTOM_HEADING_NAME_RE);
@@ -1050,21 +1178,48 @@ function get_paragraph_prefix(paragraph, style_cache, default_pstyle) {
1050
1178
  if (/^\d+$/.test(match)) return "#".repeat(parseInt(match, 10)) + " ";
1051
1179
  }
1052
1180
  if (style_name === "Title") return "# ";
1181
+ let list_num_id = null;
1182
+ let list_ilvl = null;
1183
+ let numbering_disabled = false;
1053
1184
  if (pPr) {
1054
1185
  const numPr = findChild(pPr, QN_W_NUMPR);
1055
1186
  if (numPr) {
1056
1187
  const numId = findChild(numPr, QN_W_NUMID);
1057
- if (numId && numId.getAttribute(QN_W_VAL) !== "0") {
1058
- let level = 0;
1059
- const ilvl = findChild(numPr, QN_W_ILVL);
1060
- if (ilvl) {
1061
- const valAttr = ilvl.getAttribute(QN_W_VAL);
1062
- if (valAttr) level = parseInt(valAttr, 10) || 0;
1188
+ if (numId) {
1189
+ const val = numId.getAttribute(QN_W_VAL);
1190
+ if (val === "0") {
1191
+ numbering_disabled = true;
1192
+ } else if (val) {
1193
+ list_num_id = val;
1194
+ const ilvl = findChild(numPr, QN_W_ILVL);
1195
+ if (ilvl) {
1196
+ const valAttr = ilvl.getAttribute(QN_W_VAL);
1197
+ if (valAttr !== null && /^\d+$/.test(valAttr)) {
1198
+ list_ilvl = parseInt(valAttr, 10);
1199
+ }
1200
+ }
1063
1201
  }
1064
- return " ".repeat(level) + "* ";
1065
1202
  }
1066
1203
  }
1067
1204
  }
1205
+ if (list_num_id === null && !numbering_disabled && style_info) {
1206
+ const style_num_id = style_info.num_id;
1207
+ if (style_num_id) {
1208
+ list_num_id = style_num_id;
1209
+ if (list_ilvl === null) {
1210
+ list_ilvl = style_info.num_ilvl !== void 0 ? style_info.num_ilvl : null;
1211
+ }
1212
+ }
1213
+ }
1214
+ if (list_num_id !== null) {
1215
+ const level = list_ilvl !== null ? list_ilvl : 0;
1216
+ const marker = get_list_marker(
1217
+ paragraph._parent ? paragraph._parent.part || paragraph._parent : null,
1218
+ list_num_id,
1219
+ level
1220
+ );
1221
+ return " ".repeat(level) + marker;
1222
+ }
1068
1223
  if (style_name && style_name !== "Normal") {
1069
1224
  const custom_level = _detect_heading_level_from_name(style_name);
1070
1225
  if (custom_level !== null) return "#".repeat(custom_level) + " ";
@@ -1162,6 +1317,9 @@ function* iter_block_items(parent) {
1162
1317
  for (const child of notes) {
1163
1318
  if (child.getAttribute("w:type") === "separator" || child.getAttribute("w:type") === "continuationSeparator")
1164
1319
  continue;
1320
+ const note_id = child.getAttribute("w:id");
1321
+ if (note_id !== null && /^\s*[-+]?\d+\s*$/.test(note_id) && parseInt(note_id, 10) <= 0)
1322
+ continue;
1165
1323
  yield new FootnoteItem(child, parent, parent.note_type);
1166
1324
  }
1167
1325
  return;
@@ -1177,19 +1335,24 @@ function* iter_block_items(parent) {
1177
1335
  }
1178
1336
  }
1179
1337
  function* iter_document_parts(doc) {
1338
+ for (const [container] of iter_document_parts_with_kind(doc)) {
1339
+ yield container;
1340
+ }
1341
+ }
1342
+ function* iter_document_parts_with_kind(doc) {
1180
1343
  const headers = doc.pkg.parts.filter(
1181
1344
  (p) => p.contentType === "application/vnd.openxmlformats-officedocument.wordprocessingml.header+xml"
1182
1345
  );
1183
- for (const h of headers) yield h;
1184
- yield doc;
1346
+ for (const h of headers) yield [h, "header"];
1347
+ yield [doc, "body"];
1185
1348
  const footers = doc.pkg.parts.filter(
1186
1349
  (p) => p.contentType === "application/vnd.openxmlformats-officedocument.wordprocessingml.footer+xml"
1187
1350
  );
1188
- for (const f of footers) yield f;
1351
+ for (const f of footers) yield [f, "footer"];
1189
1352
  const fnPart = doc.pkg.getPartByPath("word/footnotes.xml");
1190
1353
  const enPart = doc.pkg.getPartByPath("word/endnotes.xml");
1191
- if (fnPart) yield new NotesPart(fnPart, "fn");
1192
- if (enPart) yield new NotesPart(enPart, "en");
1354
+ if (fnPart) yield [new NotesPart(fnPart, "fn"), "footnotes"];
1355
+ if (enPart) yield [new NotesPart(enPart, "en"), "endnotes"];
1193
1356
  }
1194
1357
  function _is_page_instr(instr) {
1195
1358
  if (!instr) return false;
@@ -1227,7 +1390,22 @@ function* iter_paragraph_content(paragraph) {
1227
1390
  const child = r_element.childNodes[i];
1228
1391
  if (child.nodeType !== 1) continue;
1229
1392
  const tag = child.tagName;
1230
- if (tag === QN_W_COMMENTREFERENCE) {
1393
+ if (tag === QN_W_DRAWING || tag === QN_W_OBJECT) {
1394
+ const doc_pr = findAllDescendants(child, QN_WP_DOCPR)[0] || null;
1395
+ let alt = "";
1396
+ let img_id = "0";
1397
+ if (doc_pr) {
1398
+ alt = doc_pr.getAttribute("descr") || doc_pr.getAttribute("title") || "";
1399
+ img_id = doc_pr.getAttribute("id") || "0";
1400
+ }
1401
+ yield { type: "image", id: img_id, date: alt };
1402
+ } else if (tag === QN_W_PICT) {
1403
+ const imagedata = findAllDescendants(child, QN_V_IMAGEDATA)[0] || null;
1404
+ if (imagedata) {
1405
+ const alt = imagedata.getAttribute(QN_O_TITLE) || "";
1406
+ yield { type: "image", id: "vml", date: alt };
1407
+ }
1408
+ } else if (tag === QN_W_COMMENTREFERENCE) {
1231
1409
  const ref_id = child.getAttribute(QN_W_ID);
1232
1410
  if (ref_id) yield { type: "ref", id: ref_id };
1233
1411
  } else if (tag === QN_W_FOOTNOTEREFERENCE) {
@@ -1335,6 +1513,11 @@ var DocumentMapper = class {
1335
1513
  full_text = "";
1336
1514
  spans = [];
1337
1515
  appendix_start_index = -1;
1516
+ // [start, end, kind] per projected part, in projection order. Spans carry
1517
+ // the matching index in .part_index. Together these let the engine refuse
1518
+ // or re-anchor edits at OPC part boundaries (QA 2026-07-18 C1).
1519
+ part_ranges = [];
1520
+ _current_part_index = 0;
1338
1521
  _text_chunks = [];
1339
1522
  _plain_projection = null;
1340
1523
  constructor(doc, clean_view = false, original_view = false) {
@@ -1350,12 +1533,19 @@ var DocumentMapper = class {
1350
1533
  this._text_chunks = [];
1351
1534
  this.full_text = "";
1352
1535
  this._plain_projection = null;
1353
- for (const part of iter_document_parts(this.doc)) {
1536
+ this.part_ranges = [];
1537
+ this._current_part_index = 0;
1538
+ let part_idx = 0;
1539
+ for (const [part, part_kind] of iter_document_parts_with_kind(this.doc)) {
1540
+ this._current_part_index = part_idx;
1541
+ const part_start = current_offset;
1354
1542
  current_offset = this._map_blocks(part, current_offset);
1543
+ this.part_ranges.push([part_start, current_offset, part_kind]);
1355
1544
  if (this.spans.length > 0 && this.spans[this.spans.length - 1].text !== "\n\n") {
1356
1545
  this._add_virtual_text("\n\n", current_offset, null);
1357
1546
  current_offset += 2;
1358
1547
  }
1548
+ part_idx++;
1359
1549
  }
1360
1550
  while (this.spans.length > 0 && this.spans[this.spans.length - 1].text === "\n\n") {
1361
1551
  this.spans.pop();
@@ -1364,6 +1554,47 @@ var DocumentMapper = class {
1364
1554
  this.full_text = this._text_chunks.join("");
1365
1555
  this.appendix_start_index = -1;
1366
1556
  }
1557
+ /** [part_index, start, end, kind] for parts that projected any text. */
1558
+ _nonempty_part_ranges() {
1559
+ const out = [];
1560
+ for (let i = 0; i < this.part_ranges.length; i++) {
1561
+ const [s, e, k] = this.part_ranges[i];
1562
+ if (e > s) out.push([i, s, e, k]);
1563
+ }
1564
+ return out;
1565
+ }
1566
+ part_kind_of(part_index) {
1567
+ if (part_index >= 0 && part_index < this.part_ranges.length) {
1568
+ return this.part_ranges[part_index][2];
1569
+ }
1570
+ return null;
1571
+ }
1572
+ /** Kind of the part whose projected range contains `index`, or null. */
1573
+ part_kind_at(index) {
1574
+ for (const [, start, end, kind] of this._nonempty_part_ranges()) {
1575
+ if (start <= index && index <= end) return kind;
1576
+ }
1577
+ return null;
1578
+ }
1579
+ /**
1580
+ * When `index` falls strictly AFTER one part's text and at-or-before the
1581
+ * start of the next part's text (i.e. inside the "\n\n" separator or
1582
+ * exactly at the next part's first character), returns
1583
+ * [previous_part_index, next_part_index]. Returns null everywhere else —
1584
+ * including index == previous part's end, which is an ordinary
1585
+ * end-of-part text position, not a boundary gap.
1586
+ */
1587
+ part_boundary_at(index) {
1588
+ const ranges = this._nonempty_part_ranges();
1589
+ for (let j = 1; j < ranges.length; j++) {
1590
+ const [prev_i, , prev_end] = ranges[j - 1];
1591
+ const [next_i, next_start] = ranges[j];
1592
+ if (prev_end < index && index <= next_start) {
1593
+ return [prev_i, next_i];
1594
+ }
1595
+ }
1596
+ return null;
1597
+ }
1367
1598
  _map_blocks(container, offset) {
1368
1599
  let current = offset;
1369
1600
  const c_type = container.constructor.name;
@@ -1493,7 +1724,8 @@ ${header}`;
1493
1724
  end: current,
1494
1725
  text: "",
1495
1726
  run: null,
1496
- paragraph
1727
+ paragraph,
1728
+ part_index: this._current_part_index
1497
1729
  };
1498
1730
  this.spans.push(span);
1499
1731
  const active_ids = /* @__PURE__ */ new Set();
@@ -1525,7 +1757,8 @@ ${header}`;
1525
1757
  ins_id: i_id || void 0,
1526
1758
  del_id: d_id || void 0,
1527
1759
  hyperlink_id: active_hyperlink_id || void 0,
1528
- comment_ids: c_ids.length > 0 ? c_ids : void 0
1760
+ comment_ids: c_ids.length > 0 ? c_ids : void 0,
1761
+ part_index: this._current_part_index
1529
1762
  };
1530
1763
  this.spans.push(s);
1531
1764
  this._text_chunks.push(txt);
@@ -1704,7 +1937,15 @@ ${header}`;
1704
1937
  else if (ev.type === "del_end") delete active_del[ev.id];
1705
1938
  else if (ev.type === "fmt_start") active_fmt[ev.id] = ev;
1706
1939
  else if (ev.type === "fmt_end") delete active_fmt[ev.id];
1707
- else if (ev.type === "footnote" || ev.type === "endnote") {
1940
+ else if (ev.type === "image") {
1941
+ const hidden = this.clean_view && Object.keys(active_del).length > 0 || this.original_view && Object.keys(active_ins).length > 0;
1942
+ if (!hidden) {
1943
+ const alt = (ev.date || "image").replace(/\]/g, ")").replace(/\n/g, " ");
1944
+ const txt = `![${alt}](docx-image:${ev.id})`;
1945
+ this._add_virtual_text(txt, current, paragraph, null, true);
1946
+ current += txt.length;
1947
+ }
1948
+ } else if (ev.type === "footnote" || ev.type === "endnote") {
1708
1949
  flush_pending_runs();
1709
1950
  current_wrappers = ["", ""];
1710
1951
  current_style = ["", ""];
@@ -1819,14 +2060,16 @@ ${header}`;
1819
2060
  }
1820
2061
  return [...change_lines, ...comment_lines].join("\n");
1821
2062
  }
1822
- _add_virtual_text(text, offset, context_paragraph, hyperlink_id = null) {
2063
+ _add_virtual_text(text, offset, context_paragraph, hyperlink_id = null, is_image_marker = false) {
1823
2064
  const span = {
1824
2065
  start: offset,
1825
2066
  end: offset + text.length,
1826
2067
  text,
1827
2068
  run: null,
1828
2069
  paragraph: context_paragraph,
1829
- hyperlink_id: hyperlink_id || void 0
2070
+ hyperlink_id: hyperlink_id || void 0,
2071
+ part_index: this._current_part_index,
2072
+ is_image_marker: is_image_marker || void 0
1830
2073
  };
1831
2074
  this.spans.push(span);
1832
2075
  this._text_chunks.push(text);
@@ -2426,6 +2669,327 @@ function generate_edits_from_text(original_text, modified_text) {
2426
2669
  }
2427
2670
  return edits;
2428
2671
  }
2672
+ function _sequence_opcodes(a, b) {
2673
+ const n = a.length;
2674
+ const m = b.length;
2675
+ const dp = [];
2676
+ for (let i2 = 0; i2 <= n; i2++) dp.push(new Int32Array(m + 1));
2677
+ for (let i2 = n - 1; i2 >= 0; i2--) {
2678
+ for (let j2 = m - 1; j2 >= 0; j2--) {
2679
+ dp[i2][j2] = a[i2] === b[j2] ? dp[i2 + 1][j2 + 1] + 1 : Math.max(dp[i2 + 1][j2], dp[i2][j2 + 1]);
2680
+ }
2681
+ }
2682
+ const ops = [];
2683
+ let i = 0;
2684
+ let j = 0;
2685
+ let pend_i1 = 0;
2686
+ let pend_j1 = 0;
2687
+ let pend_del = 0;
2688
+ let pend_ins = 0;
2689
+ const flushPending = () => {
2690
+ if (pend_del === 0 && pend_ins === 0) return;
2691
+ const tag = pend_del > 0 && pend_ins > 0 ? "replace" : pend_del > 0 ? "delete" : "insert";
2692
+ ops.push([tag, pend_i1, pend_i1 + pend_del, pend_j1, pend_j1 + pend_ins]);
2693
+ pend_del = 0;
2694
+ pend_ins = 0;
2695
+ };
2696
+ while (i < n || j < m) {
2697
+ if (i < n && j < m && a[i] === b[j]) {
2698
+ flushPending();
2699
+ const i1 = i;
2700
+ const j1 = j;
2701
+ while (i < n && j < m && a[i] === b[j]) {
2702
+ i++;
2703
+ j++;
2704
+ }
2705
+ ops.push(["equal", i1, i, j1, j]);
2706
+ } else {
2707
+ if (pend_del === 0 && pend_ins === 0) {
2708
+ pend_i1 = i;
2709
+ pend_j1 = j;
2710
+ }
2711
+ if (j < m && (i === n || dp[i][j + 1] >= dp[i + 1][j])) {
2712
+ pend_ins++;
2713
+ j++;
2714
+ } else {
2715
+ pend_del++;
2716
+ i++;
2717
+ }
2718
+ }
2719
+ }
2720
+ flushPending();
2721
+ return ops;
2722
+ }
2723
+ var _IMAGE_MARKER_RE = /!\[[^\]]*\]\(docx-image:[^)]*\)/g;
2724
+ function _drop_image_marker_hunks(edits, warnings) {
2725
+ const _normalized = (s) => s.replace(_IMAGE_MARKER_RE, "").replace(/\s+/g, " ").trim();
2726
+ const kept = [];
2727
+ for (const e of edits) {
2728
+ const t_imgs = ((e.target_text || "").match(_IMAGE_MARKER_RE) || []).sort();
2729
+ const n_imgs = ((e.new_text || "").match(_IMAGE_MARKER_RE) || []).sort();
2730
+ if (JSON.stringify(t_imgs) === JSON.stringify(n_imgs)) {
2731
+ kept.push(e);
2732
+ continue;
2733
+ }
2734
+ if (_normalized(e.target_text || "") === _normalized(e.new_text || "")) {
2735
+ warnings.push(
2736
+ "An inline image was added or removed between the documents. Images cannot be transferred by text edits \u2014 the image-only difference was skipped; apply it manually in Word."
2737
+ );
2738
+ } else {
2739
+ warnings.push(
2740
+ "A text change overlapping an added/removed inline image was skipped \u2014 images cannot be transferred by text edits. Apply that section manually in Word."
2741
+ );
2742
+ }
2743
+ }
2744
+ return kept;
2745
+ }
2746
+ function _rows_are_plain(text, table) {
2747
+ for (const row of table.rows) {
2748
+ if (text.substring(row.start, row.end) !== row.cells.join(" | ")) {
2749
+ return false;
2750
+ }
2751
+ }
2752
+ return true;
2753
+ }
2754
+ var _ROW_KEY_SEP = "";
2755
+ function _table_row_opcodes(rows_o, rows_m) {
2756
+ const keys_o = rows_o.map((r) => r.cells.join(_ROW_KEY_SEP));
2757
+ const keys_m = rows_m.map((r) => r.cells.join(_ROW_KEY_SEP));
2758
+ const opcodes = _sequence_opcodes(keys_o, keys_m);
2759
+ for (const [tag, i1, i2, j1, j2] of opcodes) {
2760
+ if (tag === "insert" || tag === "delete" || tag === "replace" && i2 - i1 !== j2 - j1) {
2761
+ return opcodes;
2762
+ }
2763
+ }
2764
+ return null;
2765
+ }
2766
+ function _row_ops_for_table(table_o, table_m, opcodes, warnings) {
2767
+ const rows_o = table_o.rows;
2768
+ const rows_m = table_m.rows;
2769
+ const keys_o = rows_o.map((r) => r.cells.join(_ROW_KEY_SEP));
2770
+ const row_text = (r) => r.cells.join(" | ");
2771
+ if (new Set(keys_o).size !== keys_o.length) {
2772
+ warnings.push(
2773
+ "A table contains rows with identical text; the generated row operations anchor by row text and may be rejected as ambiguous at apply time. If that happens, apply the row changes with explicit insert_row/delete_row edits."
2774
+ );
2775
+ }
2776
+ const removed = /* @__PURE__ */ new Set();
2777
+ const replaced_new_text = {};
2778
+ for (const [tag, i1, i2, j1, j2] of opcodes) {
2779
+ if (tag === "delete") {
2780
+ for (let k = i1; k < i2; k++) removed.add(k);
2781
+ } else if (tag === "replace") {
2782
+ const pairs = Math.min(i2 - i1, j2 - j1);
2783
+ for (let k = i1 + pairs; k < i2; k++) removed.add(k);
2784
+ for (let k = 0; k < pairs; k++) {
2785
+ replaced_new_text[i1 + k] = row_text(rows_m[j1 + k]);
2786
+ }
2787
+ }
2788
+ }
2789
+ const surviving = [];
2790
+ for (let k = 0; k < rows_o.length; k++) {
2791
+ if (!removed.has(k)) surviving.push(k);
2792
+ }
2793
+ const anchor_text = (orig_idx) => {
2794
+ return replaced_new_text[orig_idx] !== void 0 ? replaced_new_text[orig_idx] : row_text(rows_o[orig_idx]);
2795
+ };
2796
+ const insert_ops = (new_rows, at_orig_index) => {
2797
+ const ops2 = [];
2798
+ const before = surviving.filter((k) => k < at_orig_index);
2799
+ const after = surviving.filter((k) => k >= at_orig_index);
2800
+ if (before.length > 0) {
2801
+ const anchor_idx = before[before.length - 1];
2802
+ const anchor = anchor_text(anchor_idx);
2803
+ for (const r of [...new_rows].reverse()) {
2804
+ ops2.push({
2805
+ type: "insert_row",
2806
+ target_text: anchor,
2807
+ position: "below",
2808
+ cells: [...r.cells],
2809
+ // Pin to the anchor row's offset: text anchors alone are
2810
+ // ambiguous when tables share identical rows. Pins do not
2811
+ // survive JSON round-trips (the strict text match applies then,
2812
+ // failing closed with the duplicate-row warning above).
2813
+ _match_start_index: rows_o[anchor_idx].start
2814
+ });
2815
+ }
2816
+ } else if (after.length > 0) {
2817
+ const anchor_idx = after[0];
2818
+ const anchor = anchor_text(anchor_idx);
2819
+ for (const r of new_rows) {
2820
+ ops2.push({
2821
+ type: "insert_row",
2822
+ target_text: anchor,
2823
+ position: "above",
2824
+ cells: [...r.cells],
2825
+ _match_start_index: rows_o[anchor_idx].start
2826
+ });
2827
+ }
2828
+ } else {
2829
+ warnings.push(
2830
+ "A table gained rows but no original row survives to anchor them; these row insertions were skipped \u2014 add them with explicit insert_row operations."
2831
+ );
2832
+ }
2833
+ return ops2;
2834
+ };
2835
+ const ops = [];
2836
+ for (const [tag, i1, i2, j1, j2] of opcodes) {
2837
+ if (tag === "equal") continue;
2838
+ if (tag === "replace") {
2839
+ const pairs = Math.min(i2 - i1, j2 - j1);
2840
+ for (let k = 0; k < pairs; k++) {
2841
+ const o_txt = row_text(rows_o[i1 + k]);
2842
+ const m_txt = row_text(rows_m[j1 + k]);
2843
+ if (o_txt !== m_txt) {
2844
+ ops.push({
2845
+ type: "modify",
2846
+ target_text: o_txt,
2847
+ new_text: m_txt,
2848
+ comment: "Diff: Table row modified"
2849
+ });
2850
+ }
2851
+ }
2852
+ for (let k = i1 + pairs; k < i2; k++) {
2853
+ ops.push({
2854
+ type: "delete_row",
2855
+ target_text: row_text(rows_o[k]),
2856
+ _match_start_index: rows_o[k].start
2857
+ });
2858
+ }
2859
+ const surplus_new = rows_m.slice(j1 + pairs, j2);
2860
+ if (surplus_new.length > 0) {
2861
+ ops.push(...insert_ops(surplus_new, i2));
2862
+ }
2863
+ } else if (tag === "delete") {
2864
+ for (let k = i1; k < i2; k++) {
2865
+ ops.push({
2866
+ type: "delete_row",
2867
+ target_text: row_text(rows_o[k]),
2868
+ _match_start_index: rows_o[k].start
2869
+ });
2870
+ }
2871
+ } else if (tag === "insert") {
2872
+ ops.push(...insert_ops(rows_m.slice(j1, j2), i1));
2873
+ }
2874
+ }
2875
+ return ops;
2876
+ }
2877
+ function generate_structured_edits(text_orig, struct_orig, text_mod, struct_mod) {
2878
+ const warnings = [];
2879
+ const edits = [];
2880
+ const kinds_o = struct_orig.part_ranges.filter(([s, e]) => e > s).map(([, , k]) => k);
2881
+ const kinds_m = struct_mod.part_ranges.filter(([s, e]) => e > s).map(([, , k]) => k);
2882
+ if (JSON.stringify(kinds_o) !== JSON.stringify(kinds_m)) {
2883
+ warnings.push(
2884
+ `The documents have different part layouts (${kinds_o.join(" + ") || "none"} vs ${kinds_m.join(" + ") || "none"}); comparing flattened text instead. Header/footer additions or removals cannot be expressed as text edits.`
2885
+ );
2886
+ const flat = _drop_image_marker_hunks(
2887
+ generate_edits_from_text(text_orig, text_mod),
2888
+ warnings
2889
+ );
2890
+ return { edits: [...flat], warnings };
2891
+ }
2892
+ const ranges_o = struct_orig.part_ranges.filter(([s, e]) => e > s).map(([s, e]) => [s, e]);
2893
+ const ranges_m = struct_mod.part_ranges.filter(([s, e]) => e > s).map(([s, e]) => [s, e]);
2894
+ for (let p = 0; p < ranges_o.length; p++) {
2895
+ const [po_start, po_end] = ranges_o[p];
2896
+ const [pm_start, pm_end] = ranges_m[p];
2897
+ const tables_o = struct_orig.tables.filter(
2898
+ (t) => po_start <= t.start && t.end <= po_end
2899
+ );
2900
+ const tables_m = struct_mod.tables.filter(
2901
+ (t) => pm_start <= t.start && t.end <= pm_end
2902
+ );
2903
+ const tables_alignable = tables_o.length === tables_m.length && tables_o.every((t) => _rows_are_plain(text_orig, t)) && tables_m.every((t) => _rows_are_plain(text_mod, t));
2904
+ if (tables_o.length !== tables_m.length) {
2905
+ warnings.push(
2906
+ `A ${kinds_o[p]} part has ${tables_o.length} table(s) in the original but ${tables_m.length} in the modified document; its tables were compared as plain text. Adding or removing whole tables is not supported via diff/apply.`
2907
+ );
2908
+ }
2909
+ if (!tables_alignable) {
2910
+ const part_edits = generate_edits_from_text(
2911
+ text_orig.substring(po_start, po_end),
2912
+ text_mod.substring(pm_start, pm_end)
2913
+ );
2914
+ for (const e of part_edits) {
2915
+ e._match_start_index = (e._match_start_index || 0) + po_start;
2916
+ }
2917
+ edits.push(...part_edits);
2918
+ continue;
2919
+ }
2920
+ const boundaries_o = [
2921
+ [po_start, po_start],
2922
+ ...tables_o.map((t) => [t.start, t.end]),
2923
+ [po_end, po_end]
2924
+ ];
2925
+ const boundaries_m = [
2926
+ [pm_start, pm_start],
2927
+ ...tables_m.map((t) => [t.start, t.end]),
2928
+ [pm_end, pm_end]
2929
+ ];
2930
+ for (let seg_idx = 0; seg_idx < boundaries_o.length - 1; seg_idx++) {
2931
+ const seg_o_start = boundaries_o[seg_idx][1];
2932
+ const seg_o_end = boundaries_o[seg_idx + 1][0];
2933
+ const seg_m_start = boundaries_m[seg_idx][1];
2934
+ const seg_m_end = boundaries_m[seg_idx + 1][0];
2935
+ const seg_edits = generate_edits_from_text(
2936
+ text_orig.substring(seg_o_start, seg_o_end),
2937
+ text_mod.substring(seg_m_start, seg_m_end)
2938
+ );
2939
+ for (const e of seg_edits) {
2940
+ e._match_start_index = (e._match_start_index || 0) + seg_o_start;
2941
+ }
2942
+ edits.push(...seg_edits);
2943
+ if (seg_idx < tables_o.length) {
2944
+ const t_o = tables_o[seg_idx];
2945
+ const t_m = tables_m[seg_idx];
2946
+ const row_opcodes = _table_row_opcodes(t_o.rows, t_m.rows);
2947
+ if (row_opcodes !== null) {
2948
+ edits.push(..._row_ops_for_table(t_o, t_m, row_opcodes, warnings));
2949
+ } else {
2950
+ const tbl_edits = generate_edits_from_text(
2951
+ text_orig.substring(t_o.start, t_o.end),
2952
+ text_mod.substring(t_m.start, t_m.end)
2953
+ );
2954
+ for (const e of tbl_edits) {
2955
+ e._match_start_index = (e._match_start_index || 0) + t_o.start;
2956
+ }
2957
+ edits.push(...tbl_edits);
2958
+ }
2959
+ }
2960
+ }
2961
+ }
2962
+ const countOccurrences = (haystack, needle) => {
2963
+ let count = 0;
2964
+ let from = 0;
2965
+ while (true) {
2966
+ const idx = haystack.indexOf(needle, from);
2967
+ if (idx === -1) break;
2968
+ count++;
2969
+ from = idx + needle.length;
2970
+ }
2971
+ return count;
2972
+ };
2973
+ let ambiguous_anchor_warned = false;
2974
+ for (const e of edits) {
2975
+ if ((e.type === "insert_row" || e.type === "delete_row") && !ambiguous_anchor_warned) {
2976
+ if (e.target_text && countOccurrences(text_orig, e.target_text) > 1) {
2977
+ warnings.push(
2978
+ `The row anchor "${e.target_text.substring(0, 60)}" appears more than once in the document. Applying this diff from its JSON output may be rejected as ambiguous \u2014 make the anchor rows unique, or apply the row changes with explicit insert_row/delete_row edits.`
2979
+ );
2980
+ ambiguous_anchor_warned = true;
2981
+ }
2982
+ }
2983
+ }
2984
+ const modify_edits = edits.filter(
2985
+ (e) => e.type === "modify"
2986
+ );
2987
+ const kept_modifies = new Set(_drop_image_marker_hunks(modify_edits, warnings));
2988
+ const final_edits = edits.filter(
2989
+ (e) => e.type !== "modify" || kept_modifies.has(e)
2990
+ );
2991
+ return { edits: final_edits, warnings };
2992
+ }
2429
2993
  function create_unified_diff(original_text, modified_text, context_lines = 3) {
2430
2994
  const dmp = new import_diff_match_patch.default.diff_match_patch();
2431
2995
  dmp.Diff_Timeout = 2;
@@ -2733,6 +3297,30 @@ function _strip_balanced_markers(text) {
2733
3297
  function _replace_smart_quotes(text) {
2734
3298
  return text.replace(/“/g, '"').replace(/”/g, '"').replace(/‘/g, "'").replace(/’/g, "'");
2735
3299
  }
3300
+ function _strip_markdown_for_matching(text) {
3301
+ const result = [];
3302
+ const position_map = [];
3303
+ let i = 0;
3304
+ while (i < text.length) {
3305
+ const pair = text.substring(i, i + 2);
3306
+ if (i < text.length - 1 && (pair === "**" || pair === "__")) {
3307
+ i += 2;
3308
+ continue;
3309
+ }
3310
+ if (text[i] === "*" || text[i] === "_") {
3311
+ const prev_char = i > 0 ? text[i - 1] : " ";
3312
+ const next_char = i < text.length - 1 ? text[i + 1] : " ";
3313
+ if ([" ", "\n", " "].includes(prev_char) || [" ", "\n", " "].includes(next_char)) {
3314
+ i += 1;
3315
+ continue;
3316
+ }
3317
+ }
3318
+ position_map.push(i);
3319
+ result.push(text[i]);
3320
+ i += 1;
3321
+ }
3322
+ return [result.join(""), position_map];
3323
+ }
2736
3324
  function _find_safe_boundaries(text, start, end) {
2737
3325
  let new_start = start;
2738
3326
  let new_end = end;
@@ -2836,31 +3424,63 @@ function _make_fuzzy_regex(target_text) {
2836
3424
  if (remaining) parts.push(escapeRegExp(remaining));
2837
3425
  return parts.join("");
2838
3426
  }
2839
- function _find_match_in_text(text, target) {
2840
- if (!target) return [-1, -1];
2841
- let idx = text.indexOf(target);
2842
- if (idx !== -1) return _find_safe_boundaries(text, idx, idx + target.length);
3427
+ function _find_all_matches_in_text(text, target, is_regex = false) {
3428
+ if (!target) return [];
3429
+ if (is_regex) {
3430
+ try {
3431
+ return userFindAllMatches(target, text).map((m) => [m.start, m.end]);
3432
+ } catch (e) {
3433
+ if (e instanceof RegexTimeoutError) throw e;
3434
+ return [];
3435
+ }
3436
+ }
3437
+ const findAllLiteral = (haystack, needle) => {
3438
+ const out = [];
3439
+ let from = 0;
3440
+ while (true) {
3441
+ const idx = haystack.indexOf(needle, from);
3442
+ if (idx === -1) break;
3443
+ out.push([idx, idx + needle.length]);
3444
+ from = idx + needle.length;
3445
+ }
3446
+ return out;
3447
+ };
3448
+ let spans = findAllLiteral(text, target);
3449
+ if (spans.length > 0) {
3450
+ return spans.map(([s, e]) => _find_safe_boundaries(text, s, e));
3451
+ }
2843
3452
  const norm_text = _replace_smart_quotes(text);
2844
3453
  const norm_target = _replace_smart_quotes(target);
2845
- idx = norm_text.indexOf(norm_target);
2846
- if (idx !== -1)
2847
- return _find_safe_boundaries(text, idx, idx + norm_target.length);
3454
+ spans = findAllLiteral(norm_text, norm_target);
3455
+ if (spans.length > 0) {
3456
+ return spans.map(([s, e]) => _find_safe_boundaries(text, s, e));
3457
+ }
3458
+ const [stripped_text, pos_map] = _strip_markdown_for_matching(norm_text);
3459
+ const [stripped_target] = _strip_markdown_for_matching(norm_target);
3460
+ if (stripped_target && (stripped_text !== norm_text || stripped_target !== norm_target)) {
3461
+ const results = [];
3462
+ for (const [p_start, p_end] of findAllLiteral(stripped_text, stripped_target)) {
3463
+ const raw_start = pos_map[p_start];
3464
+ const raw_end = pos_map[p_end - 1] + 1;
3465
+ results.push(_find_safe_boundaries(text, raw_start, raw_end));
3466
+ }
3467
+ if (results.length > 0) return results;
3468
+ }
2848
3469
  try {
2849
- const pattern = new RegExp(_make_fuzzy_regex(target));
2850
- const match = pattern.exec(text);
2851
- if (match) {
2852
- const raw_start = match.index;
2853
- const raw_end = match.index + match[0].length;
3470
+ const pattern = new RegExp(_make_fuzzy_regex(target), "g");
3471
+ const results = [];
3472
+ for (const match of text.matchAll(pattern)) {
2854
3473
  const [refined_start, refined_end] = _refine_match_boundaries(
2855
3474
  text,
2856
- raw_start,
2857
- raw_end
3475
+ match.index,
3476
+ match.index + match[0].length
2858
3477
  );
2859
- return _find_safe_boundaries(text, refined_start, refined_end);
3478
+ results.push(_find_safe_boundaries(text, refined_start, refined_end));
2860
3479
  }
3480
+ if (results.length > 0) return results;
2861
3481
  } catch (e) {
2862
3482
  }
2863
- return [-1, -1];
3483
+ return [];
2864
3484
  }
2865
3485
  function _build_critic_markup(target_text, new_text, comment, edit_index, include_index, highlight_only) {
2866
3486
  const parts = [];
@@ -2892,19 +3512,55 @@ function _build_critic_markup(target_text, new_text, comment, edit_index, includ
2892
3512
  }
2893
3513
  return parts.join("");
2894
3514
  }
2895
- function apply_edits_to_markdown(markdown_text, edits, include_index = false, highlight_only = false) {
3515
+ function apply_edits_to_markdown(markdown_text, edits, include_index = false, highlight_only = false, edit_reports) {
2896
3516
  if (!edits || edits.length === 0) return markdown_text;
3517
+ const _report = (idx, status, error = null, occurrences = 0) => {
3518
+ if (edit_reports) edit_reports.push({ index: idx, status, error, occurrences });
3519
+ };
2897
3520
  const matched_edits = [];
2898
3521
  for (let idx = 0; idx < edits.length; idx++) {
2899
3522
  const edit = edits[idx];
2900
3523
  const target = edit.target_text || "";
3524
+ const match_mode = edit.match_mode || "strict";
3525
+ const is_regex = Boolean(edit.regex);
2901
3526
  if (!target) {
3527
+ _report(
3528
+ idx,
3529
+ "failed",
3530
+ `- Edit ${idx + 1} Failed: target_text is empty. Pure insertions are expressed as a replacement: put the text immediately around the insertion point in target_text and repeat it (plus the new text) in new_text.`
3531
+ );
3532
+ continue;
3533
+ }
3534
+ let spans;
3535
+ try {
3536
+ spans = _find_all_matches_in_text(markdown_text, target, is_regex);
3537
+ } catch (e) {
3538
+ if (!(e instanceof RegexTimeoutError)) throw e;
3539
+ _report(idx, "failed", `- Edit ${idx + 1} Failed: ${e.message}`);
2902
3540
  continue;
2903
3541
  }
2904
- const [start, end] = _find_match_in_text(markdown_text, target);
2905
- if (start === -1) continue;
2906
- const actual_matched_text = markdown_text.substring(start, end);
2907
- matched_edits.push([start, end, actual_matched_text, edit, idx]);
3542
+ if (spans.length === 0) {
3543
+ _report(
3544
+ idx,
3545
+ "failed",
3546
+ `- Edit ${idx + 1} Failed: Target text not found in document:
3547
+ "${target.substring(0, 80)}"`
3548
+ );
3549
+ continue;
3550
+ }
3551
+ if (spans.length > 1 && match_mode === "strict") {
3552
+ _report(
3553
+ idx,
3554
+ "failed",
3555
+ format_ambiguity_error(idx + 1, target, markdown_text, spans)
3556
+ );
3557
+ continue;
3558
+ }
3559
+ const selected = match_mode === "strict" || match_mode === "first" ? spans.slice(0, 1) : spans;
3560
+ for (const [start, end] of selected) {
3561
+ matched_edits.push([start, end, markdown_text.substring(start, end), edit, idx]);
3562
+ }
3563
+ _report(idx, "applied", null, selected.length);
2908
3564
  }
2909
3565
  const matched_edits_filtered = [];
2910
3566
  const occupied_ranges = [];
@@ -2914,6 +3570,16 @@ function apply_edits_to_markdown(markdown_text, edits, include_index = false, hi
2914
3570
  for (const [occ_start, occ_end] of occupied_ranges) {
2915
3571
  if (start < occ_end && end > occ_start) {
2916
3572
  overlaps = true;
3573
+ if (edit_reports) {
3574
+ const msg = `- Edit ${orig_idx + 1} Failed: overlaps with a previously matched edit.`;
3575
+ for (const r of edit_reports) {
3576
+ if (r.index === orig_idx) {
3577
+ r.status = "failed";
3578
+ r.error = msg;
3579
+ r.occurrences = 0;
3580
+ }
3581
+ }
3582
+ }
2917
3583
  break;
2918
3584
  }
2919
3585
  }
@@ -3182,6 +3848,15 @@ function validate_edit_strings(edits, index_offset = 0) {
3182
3848
  }
3183
3849
  }
3184
3850
  }
3851
+ if (t_text.includes("docx-image:") || n_text.includes("docx-image:")) {
3852
+ const t_imgs = (t_text.match(/!\[[^\]]*\]\(docx-image:[^)]*\)/g) || []).sort();
3853
+ const n_imgs = (n_text.match(/!\[[^\]]*\]\(docx-image:[^)]*\)/g) || []).sort();
3854
+ if (JSON.stringify(t_imgs) !== JSON.stringify(n_imgs)) {
3855
+ errors.push(
3856
+ `- Edit ${i + 1 + index_offset} Failed: image markers (![alt](docx-image:N)) are read-only projections of embedded images. They cannot be inserted, altered, or removed via text replacement \u2014 edit the text around the image instead.`
3857
+ );
3858
+ }
3859
+ }
3185
3860
  if (t_text.includes("{#") || n_text.includes("{#")) {
3186
3861
  const t_anchors = t_text.match(/\{#[^\}]+\}/g) || [];
3187
3862
  const n_anchors = n_text.match(/\{#[^\}]+\}/g) || [];
@@ -3905,6 +4580,43 @@ var RedlineEngine = class _RedlineEngine {
3905
4580
  element.setAttribute("xml:space", "preserve");
3906
4581
  }
3907
4582
  }
4583
+ /**
4584
+ * Walks `element` to its XML root element. Word (and LibreOffice, which
4585
+ * refuses to LOAD such files) only supports comment ranges in the main
4586
+ * document story ("w:document") — never in headers, footers, footnotes or
4587
+ * endnotes (QA 2026-07-18 H4/C1).
4588
+ */
4589
+ _comment_anchor_in_main_story(element) {
4590
+ let root = element;
4591
+ while (root.parentNode && root.parentNode.nodeType === 1) {
4592
+ root = root.parentNode;
4593
+ }
4594
+ return root.tagName === "w:document";
4595
+ }
4596
+ /**
4597
+ * When the anchor lives outside the main document story, records a
4598
+ * user-visible warning and returns true (caller must skip the comment).
4599
+ * The tracked change itself still applies — only the bubble is dropped.
4600
+ */
4601
+ _skip_comment_outside_main_story(element, text) {
4602
+ if (this._comment_anchor_in_main_story(element)) return false;
4603
+ let root = element;
4604
+ while (root.parentNode && root.parentNode.nodeType === 1) {
4605
+ root = root.parentNode;
4606
+ }
4607
+ const story = {
4608
+ "w:ftr": "footer",
4609
+ "w:hdr": "header",
4610
+ "w:footnotes": "footnote",
4611
+ "w:endnotes": "endnote"
4612
+ }[root.tagName] || "non-body";
4613
+ const msg = `- Warning: the comment "${text.substring(0, 60)}" was NOT attached: Word does not support comments inside a ${story} part, and writing one produces a document other applications cannot open. The tracked change itself was applied.`;
4614
+ this.skipped_details.push(msg);
4615
+ console.error(
4616
+ `Comment anchor outside main story; comment dropped (story=${story})`
4617
+ );
4618
+ return true;
4619
+ }
3908
4620
  /**
3909
4621
  * Attaches a comment that wraps a contiguous range within a single paragraph.
3910
4622
  * start_element and end_element must both be direct children of parent_element
@@ -3913,6 +4625,8 @@ var RedlineEngine = class _RedlineEngine {
3913
4625
  */
3914
4626
  _attach_comment(parent_element, start_element, end_element, text) {
3915
4627
  if (!text) return;
4628
+ if (!parent_element || !start_element || !end_element) return;
4629
+ if (this._skip_comment_outside_main_story(parent_element, text)) return;
3916
4630
  const comment_id = this.comments_manager.addComment(this.author, text);
3917
4631
  const xmlDoc = parent_element.ownerDocument;
3918
4632
  const range_start = xmlDoc.createElement("w:commentRangeStart");
@@ -3946,6 +4660,9 @@ var RedlineEngine = class _RedlineEngine {
3946
4660
  */
3947
4661
  _attach_comment_spanning(start_p, start_el, end_p, end_el, text) {
3948
4662
  if (!text) return;
4663
+ if (!start_p || !end_p) return;
4664
+ if (this._skip_comment_outside_main_story(start_p, text) || this._skip_comment_outside_main_story(end_p, text))
4665
+ return;
3949
4666
  const comment_id = this.comments_manager.addComment(this.author, text);
3950
4667
  const xmlDocStart = start_p.ownerDocument;
3951
4668
  const xmlDocEnd = end_p.ownerDocument;
@@ -4557,6 +5274,51 @@ var RedlineEngine = class _RedlineEngine {
4557
5274
  pfx,
4558
5275
  (edit.new_text || "").length - sfx
4559
5276
  );
5277
+ const multi_part_doc = target_mapper.part_ranges.filter((r) => r[1] > r[0]).length > 1;
5278
+ const raw_span_parts = multi_part_doc ? Array.from(
5279
+ new Set(
5280
+ target_mapper.spans.filter(
5281
+ (s) => s.run !== null && s.end > m_start && s.start < m_start + m_len
5282
+ ).map((s) => s.part_index)
5283
+ )
5284
+ ).sort((a, b) => a - b) : [];
5285
+ if (raw_span_parts.length > 1) {
5286
+ const kinds = raw_span_parts.map((pi) => target_mapper.part_kind_of(pi) || "?").join(" \u2192 ");
5287
+ errors.push(
5288
+ `- Edit ${i + 1 + index_offset} Failed: target_text spans a structural document-part boundary (${kinds}). Headers, body, footers and footnotes are separate Word parts \u2014 an edit cannot cross between them. Anchor the edit on text within a single part (split it into one edit per part if both sides must change).`
5289
+ );
5290
+ }
5291
+ const eff_start = m_start + pfx;
5292
+ const eff_end = m_start + m_len - sfx;
5293
+ if (eff_end > eff_start) {
5294
+ const overlapping = target_mapper.spans.filter(
5295
+ (s) => s.end > eff_start && s.start < eff_end && (s.run !== null || s.text.trim() !== "")
5296
+ );
5297
+ if (overlapping.some((s) => s.is_image_marker)) {
5298
+ errors.push(
5299
+ `- Edit ${i + 1 + index_offset} Failed: the target overlaps a read-only image marker (![alt](docx-image:N)). Images cannot be edited or removed via text replacement \u2014 target the text around the image instead.`
5300
+ );
5301
+ }
5302
+ }
5303
+ if (edit.comment && (edit.new_text || "") === (edit.target_text || "")) {
5304
+ const kind_here = target_mapper.part_kind_at(m_start);
5305
+ if (kind_here !== null && kind_here !== "body") {
5306
+ errors.push(
5307
+ `- Edit ${i + 1 + index_offset} Failed: comments cannot be anchored inside a ${kind_here} part \u2014 Word only supports comments in the main document body. Comment on the related body text instead.`
5308
+ );
5309
+ }
5310
+ }
5311
+ if (_RedlineEngine._introduces_table_row_text(
5312
+ target_mapper,
5313
+ m_start,
5314
+ m_len,
5315
+ final_target,
5316
+ final_new
5317
+ )) {
5318
+ errors.push(
5319
+ `- Edit ${i + 1 + index_offset} Failed: new_text introduces a pipe-delimited row line inside a table. Text replacement cannot create table rows \u2014 use the structured 'insert_row' operation instead (e.g. {"type": "insert_row", "target_text": "<anchor row text>", "cells": ["...", "..."]}).`
5320
+ );
5321
+ }
4560
5322
  if (final_target.includes("\n\n")) {
4561
5323
  const balanced = matched.split("\n\n").length === (edit.new_text || "").split("\n\n").length;
4562
5324
  if (!balanced) {
@@ -4653,6 +5415,19 @@ var RedlineEngine = class _RedlineEngine {
4653
5415
  * [start, start+length) in `mapper`, or null if that text is not inside a
4654
5416
  * table row.
4655
5417
  */
5418
+ /**
5419
+ * True when a replacement anchored in a table would ADD line-separated
5420
+ * pipe-delimited content — the text shape of a table row. Writing that
5421
+ * into a cell renders a fake row inside one cell while the real grid
5422
+ * stays unchanged (QA 2026-07-18 C2); such edits must use insert_row.
5423
+ */
5424
+ static _introduces_table_row_text(mapper, start, length, final_target, final_new) {
5425
+ if (!final_new.includes("\n") || !final_new.includes(" | ")) return false;
5426
+ const new_pipe_lines = final_new.split("\n").filter((line) => line.includes(" | ")).length;
5427
+ const old_pipe_lines = final_target.split("\n").filter((line) => line.includes(" | ")).length;
5428
+ if (new_pipe_lines <= old_pipe_lines) return false;
5429
+ return _RedlineEngine._column_count_at(mapper, start, Math.max(length, 1)) !== null;
5430
+ }
4656
5431
  static _column_count_at(mapper, start, length) {
4657
5432
  for (const s of mapper.spans) {
4658
5433
  if (s.run === null || s.end <= start || s.start >= start + length) {
@@ -5081,7 +5856,7 @@ var RedlineEngine = class _RedlineEngine {
5081
5856
  const counted_split_groups = /* @__PURE__ */ new Set();
5082
5857
  for (const [edit, orig_new] of resolved_edits) {
5083
5858
  const start = edit._resolved_start_idx || 0;
5084
- const end = start + (edit.target_text ? edit.target_text.length : 0);
5859
+ const end = edit.type === "insert_row" ? start : start + (edit.target_text ? edit.target_text.length : 0);
5085
5860
  const overlaps = occupied_ranges.some(
5086
5861
  ([occ_start, occ_end]) => start < occ_end && end > occ_start
5087
5862
  );
@@ -5771,13 +6546,49 @@ var RedlineEngine = class _RedlineEngine {
5771
6546
  return true;
5772
6547
  }
5773
6548
  if (op === "INSERTION") {
5774
- const [anchor_run, anchor_para] = active_mapper.get_insertion_anchor(
5775
- start_idx,
5776
- rebuild_map
5777
- );
6549
+ let final_new_text = edit.new_text || "";
6550
+ let boundary_anchor = null;
6551
+ const boundary = typeof active_mapper.part_boundary_at === "function" ? active_mapper.part_boundary_at(start_idx) : null;
6552
+ const is_machine_pure_insertion = !edit.target_text && (edit._parent_edit_ref === void 0 || edit._parent_edit_ref === null);
6553
+ if (boundary !== null && is_machine_pure_insertion) {
6554
+ const [prev_i, next_i] = boundary;
6555
+ const prev_kind = active_mapper.part_kind_of(prev_i);
6556
+ const next_kind = active_mapper.part_kind_of(next_i);
6557
+ if (prev_kind === "body" && next_kind !== "body") {
6558
+ const real_before = active_mapper.spans.filter(
6559
+ (s) => s.run !== null && s.part_index === prev_i
6560
+ );
6561
+ if (real_before.length > 0) {
6562
+ boundary_anchor = real_before[real_before.length - 1];
6563
+ }
6564
+ }
6565
+ }
6566
+ let anchor_run;
6567
+ let anchor_para;
6568
+ if (boundary_anchor !== null) {
6569
+ anchor_run = boundary_anchor.run;
6570
+ anchor_para = boundary_anchor.paragraph;
6571
+ if (!final_new_text.startsWith("\n")) {
6572
+ final_new_text = "\n\n" + final_new_text;
6573
+ }
6574
+ } else {
6575
+ [anchor_run, anchor_para] = active_mapper.get_insertion_anchor(
6576
+ start_idx,
6577
+ rebuild_map
6578
+ );
6579
+ }
5778
6580
  if (!anchor_run && !anchor_para) return false;
5779
- const _bug233_new = edit.new_text || "";
5780
- const _bug233_trailing_break = /\n\s*$/.test(_bug233_new);
6581
+ if (_RedlineEngine._introduces_table_row_text(
6582
+ active_mapper,
6583
+ start_idx,
6584
+ 1,
6585
+ "",
6586
+ final_new_text
6587
+ )) {
6588
+ return false;
6589
+ }
6590
+ const _bug233_new = final_new_text;
6591
+ const _bug233_trailing_break = boundary_anchor === null && /\n\s*$/.test(_bug233_new);
5781
6592
  let _bug233_target_para = null;
5782
6593
  {
5783
6594
  const startingSpans = active_mapper.spans.filter(
@@ -5847,7 +6658,7 @@ var RedlineEngine = class _RedlineEngine {
5847
6658
  }
5848
6659
  }
5849
6660
  const result = this._track_insert_multiline(
5850
- edit.new_text || "",
6661
+ final_new_text,
5851
6662
  anchor_run,
5852
6663
  anchor_para,
5853
6664
  ins_id
@@ -5936,6 +6747,20 @@ var RedlineEngine = class _RedlineEngine {
5936
6747
  }
5937
6748
  return true;
5938
6749
  }
6750
+ if ((op === "DELETION" || op === "MODIFICATION") && length && active_mapper.part_ranges.filter((r) => r[1] > r[0]).length > 1) {
6751
+ const crossed_parts = /* @__PURE__ */ new Set();
6752
+ for (const s of active_mapper.spans) {
6753
+ if (s.run !== null && s.end > start_idx && s.start < start_idx + length) {
6754
+ crossed_parts.add(s.part_index);
6755
+ }
6756
+ }
6757
+ if (crossed_parts.size > 1) {
6758
+ console.error(
6759
+ `Refusing edit that spans OPC part boundary (start=${start_idx}, parts=${Array.from(crossed_parts).sort().join(",")})`
6760
+ );
6761
+ return false;
6762
+ }
6763
+ }
5939
6764
  const target_runs = active_mapper.find_target_runs_by_index(
5940
6765
  start_idx,
5941
6766
  length,
@@ -6626,6 +7451,35 @@ function strip_image_alt_text(doc) {
6626
7451
  }
6627
7452
  return count ? [`Image alt text: ${count} auto-generated descriptions removed`] : [];
6628
7453
  }
7454
+ function detect_watermarks(doc) {
7455
+ const warnings = [];
7456
+ const seen = /* @__PURE__ */ new Set();
7457
+ for (const part of doc.pkg.parts) {
7458
+ const name = String(part.partname);
7459
+ let location;
7460
+ if (name.startsWith("/word/header")) location = "header";
7461
+ else if (name.startsWith("/word/footer")) location = "footer";
7462
+ else if (name === "/word/document.xml") location = "body";
7463
+ else continue;
7464
+ let textpaths;
7465
+ try {
7466
+ textpaths = findDescendantsByLocalName(part._element, "textpath");
7467
+ } catch {
7468
+ continue;
7469
+ }
7470
+ for (const tp of textpaths) {
7471
+ const text = (tp.getAttribute("string") || "").trim();
7472
+ const key = `${location}${text}`;
7473
+ if (seen.has(key)) continue;
7474
+ seen.add(key);
7475
+ const label = text ? `"${_truncate(text, 60)}"` : "(no text)";
7476
+ warnings.push(
7477
+ `Watermark-like text object in ${location}: ${label} \u2014 NOT removed. It remains in the document package and may render in other applications; review and remove it in Word if it must not reach the counterparty.`
7478
+ );
7479
+ }
7480
+ }
7481
+ return warnings;
7482
+ }
6629
7483
  function audit_hyperlinks(doc) {
6630
7484
  const internal = ["sharepoint.com", "onedrive.com", ".internal", "intranet", "localhost", "10.", "192.168.", "172.16."];
6631
7485
  const warnings = [];
@@ -6678,25 +7532,53 @@ function _get_paragraph_text(p) {
6678
7532
  }
6679
7533
  return text;
6680
7534
  }
7535
+ var _TERM_BODY = "[A-Z][A-Za-z0-9\\s\\-&'\u2019]{1,60}";
7536
+ var _LEADING_TERM_RE = new RegExp(
7537
+ `^(?:[\\d.\\-()a-zA-Z]+\\s*)?["\u201C](${_TERM_BODY})["\u201D]`,
7538
+ "d"
7539
+ );
7540
+ var _SENTENCE_TERM_RE = new RegExp(
7541
+ `(?<=[.;:!?])\\s+(?:[\\d.\\-()a-zA-Z]+\\s+)?["\u201C](${_TERM_BODY})["\u201D]`,
7542
+ "dg"
7543
+ );
7544
+ var _INLINE_TERM_RE = new RegExp(
7545
+ `\\([^)]*?["\u201C](${_TERM_BODY})["\u201D][^)]*?\\)`,
7546
+ "dg"
7547
+ );
7548
+ function extract_terms_from_paragraph(text) {
7549
+ const found = [];
7550
+ const leading = _LEADING_TERM_RE.exec(text);
7551
+ if (leading && leading.indices && leading.indices[1]) {
7552
+ found.push([leading.indices[1][0], leading[1].trim()]);
7553
+ }
7554
+ for (const m of text.matchAll(_SENTENCE_TERM_RE)) {
7555
+ const indices = m.indices;
7556
+ if (indices && indices[1]) found.push([indices[1][0], m[1].trim()]);
7557
+ }
7558
+ for (const m of text.matchAll(_INLINE_TERM_RE)) {
7559
+ const indices = m.indices;
7560
+ if (indices && indices[1]) found.push([indices[1][0], m[1].trim()]);
7561
+ }
7562
+ const terms = [];
7563
+ const seen_positions = /* @__PURE__ */ new Set();
7564
+ found.sort((a, b) => a[0] - b[0]);
7565
+ for (const [pos, term] of found) {
7566
+ if (seen_positions.has(pos)) continue;
7567
+ seen_positions.add(pos);
7568
+ terms.push(term);
7569
+ }
7570
+ return terms;
7571
+ }
6681
7572
  function extract_all_domain_metadata(doc, base_text) {
6682
7573
  const definitions = {};
6683
7574
  const duplicates = /* @__PURE__ */ new Set();
6684
7575
  const raw_anchors = {};
6685
7576
  const raw_references = [];
6686
- const leading_re = /^(?:[\d.\-()a-zA-Z]+\s*)?["“]([A-Z][A-Za-z0-9\s\-&'’]{1,60})["”]/;
6687
- const inline_re = /\([^)]*?["“]([A-Z][A-Za-z0-9\s\-&'’]{1,60})["”][^)]*?\)/g;
6688
7577
  for (const item of iter_block_items(doc)) {
6689
7578
  if (!(item instanceof Paragraph)) continue;
6690
7579
  const text = _get_paragraph_text(item).trim();
6691
7580
  if (!text) continue;
6692
- const extracted_terms = [];
6693
- const leading_match = text.match(leading_re);
6694
- if (leading_match) extracted_terms.push(leading_match[1].trim());
6695
- const inline_matches = text.matchAll(inline_re);
6696
- for (const m of inline_matches) {
6697
- extracted_terms.push(m[1].trim());
6698
- }
6699
- for (const term of extracted_terms) {
7581
+ for (const term of extract_terms_from_paragraph(text)) {
6700
7582
  if (definitions[term]) duplicates.add(term);
6701
7583
  else definitions[term] = { count: 0 };
6702
7584
  }
@@ -6924,27 +7806,32 @@ function build_structural_appendix(doc, base_text) {
6924
7806
  }
6925
7807
 
6926
7808
  // src/ingest.ts
6927
- async function extractTextFromBuffer(buffer, cleanView = false) {
7809
+ async function extractTextFromBuffer(buffer, cleanView = false, includeAppendix = true) {
6928
7810
  const doc = await DocumentObject.load(buffer);
6929
- return _extractTextFromDoc(doc, cleanView);
7811
+ return _extractTextFromDoc(doc, cleanView, includeAppendix);
6930
7812
  }
6931
- function _extractTextFromDoc(doc, cleanView = false, includeAppendix = true, return_paragraph_offsets = false) {
7813
+ function _extractTextFromDoc(doc, cleanView = false, includeAppendix = true, return_paragraph_offsets = false, return_structure = false) {
6932
7814
  const comments_map = extract_comments_data(doc.pkg);
6933
7815
  const full_text = [];
6934
7816
  const paragraph_offsets = /* @__PURE__ */ new Map();
7817
+ const structure = return_structure ? { part_ranges: [], tables: [] } : null;
6935
7818
  let cursor = 0;
6936
- for (const part of iter_document_parts(doc)) {
7819
+ for (const [part, part_kind] of iter_document_parts_with_kind(doc)) {
6937
7820
  const part_cursor = full_text.length > 0 ? cursor + 2 : cursor;
6938
7821
  const part_text = _extract_blocks(
6939
7822
  part,
6940
7823
  comments_map,
6941
7824
  cleanView,
6942
7825
  part_cursor,
6943
- return_paragraph_offsets ? paragraph_offsets : void 0
7826
+ return_paragraph_offsets ? paragraph_offsets : void 0,
7827
+ structure ? structure.tables : void 0
6944
7828
  );
6945
7829
  if (part_text) {
6946
7830
  if (full_text.length > 0) cursor += 2;
6947
7831
  full_text.push(part_text);
7832
+ if (structure) {
7833
+ structure.part_ranges.push([cursor, cursor + part_text.length, part_kind]);
7834
+ }
6948
7835
  cursor += part_text.length;
6949
7836
  }
6950
7837
  }
@@ -6956,9 +7843,12 @@ function _extractTextFromDoc(doc, cleanView = false, includeAppendix = true, ret
6956
7843
  if (return_paragraph_offsets) {
6957
7844
  return { text: base_text, paragraph_offsets };
6958
7845
  }
7846
+ if (structure) {
7847
+ return { text: base_text, structure };
7848
+ }
6959
7849
  return base_text;
6960
7850
  }
6961
- function _extract_blocks(container, comments_map, cleanView, cursor, paragraph_offsets) {
7851
+ function _extract_blocks(container, comments_map, cleanView, cursor, paragraph_offsets, table_acc) {
6962
7852
  const part = container.part || container;
6963
7853
  const [style_cache, default_pstyle] = _get_style_cache(part);
6964
7854
  const blocks = [];
@@ -7012,17 +7902,23 @@ ${header}`;
7012
7902
  is_first_para = false;
7013
7903
  is_first_block = false;
7014
7904
  } else if (item instanceof Table) {
7905
+ const geometry = table_acc ? { start: block_start, end: block_start, rows: [] } : null;
7015
7906
  const table_text = extract_table(
7016
7907
  item,
7017
7908
  comments_map,
7018
7909
  cleanView,
7019
7910
  block_start,
7020
- paragraph_offsets
7911
+ paragraph_offsets,
7912
+ geometry
7021
7913
  );
7022
7914
  if (table_text) {
7023
7915
  blocks.push(table_text);
7024
7916
  local_cursor = block_start + table_text.length;
7025
7917
  is_first_block = false;
7918
+ if (geometry && table_acc) {
7919
+ geometry.end = block_start + table_text.length;
7920
+ table_acc.push(geometry);
7921
+ }
7026
7922
  } else if (!is_first_block) {
7027
7923
  local_cursor -= 2;
7028
7924
  }
@@ -7031,7 +7927,7 @@ ${header}`;
7031
7927
  }
7032
7928
  return blocks.join("\n\n");
7033
7929
  }
7034
- function extract_table(table, comments_map, cleanView, cursor, paragraph_offsets) {
7930
+ function extract_table(table, comments_map, cleanView, cursor, paragraph_offsets, geometry) {
7035
7931
  const rows_text = [];
7036
7932
  let rows_processed = 0;
7037
7933
  let local_cursor = cursor;
@@ -7077,6 +7973,13 @@ function extract_table(table, comments_map, cleanView, cursor, paragraph_offsets
7077
7973
  }
7078
7974
  rows_text.push(row_str);
7079
7975
  local_cursor = row_start + row_str.length;
7976
+ if (geometry) {
7977
+ geometry.rows.push({
7978
+ start: row_start,
7979
+ end: local_cursor,
7980
+ cells: [...cell_texts]
7981
+ });
7982
+ }
7080
7983
  rows_processed++;
7081
7984
  }
7082
7985
  return rows_text.join("\n");
@@ -7228,7 +8131,12 @@ function build_paragraph_text(paragraph, comments_map, cleanView, style_cache, d
7228
8131
  else if (ev.type === "del_end") delete active_del[ev.id];
7229
8132
  else if (ev.type === "fmt_start") active_fmt[ev.id] = ev;
7230
8133
  else if (ev.type === "fmt_end") delete active_fmt[ev.id];
7231
- else if (ev.type === "footnote" || ev.type === "endnote") {
8134
+ else if (ev.type === "image") {
8135
+ if (!(cleanView && Object.keys(active_del).length > 0)) {
8136
+ const alt = (ev.date || "image").replace(/\]/g, ")").replace(/\n/g, " ");
8137
+ parts.push(`![${alt}](docx-image:${ev.id})`);
8138
+ }
8139
+ } else if (ev.type === "footnote" || ev.type === "endnote") {
7232
8140
  if (pending_text) {
7233
8141
  parts.push(
7234
8142
  `${current_wrappers[0]}${pending_text}${current_wrappers[1]}`
@@ -8030,6 +8938,7 @@ async function finalize_document(doc, options) {
8030
8938
  report.add_transform_lines(strip_image_alt_text(doc));
8031
8939
  const warnings = audit_hyperlinks(doc);
8032
8940
  for (const w of warnings) report.warnings.push(w);
8941
+ for (const w of detect_watermarks(doc)) report.warnings.push(w);
8033
8942
  report.add_transform_lines(normalize_change_dates(doc));
8034
8943
  if (options.protection_mode === "read_only" || options.protection_mode === "encrypt") {
8035
8944
  if (options.protection_mode === "encrypt") {
@@ -8097,6 +9006,7 @@ function identifyEngine() {
8097
9006
  extract_outline,
8098
9007
  finalize_document,
8099
9008
  generate_edits_from_text,
9009
+ generate_structured_edits,
8100
9010
  identifyEngine,
8101
9011
  paginate,
8102
9012
  split_structural_appendix,