docuhand 0.1.0.dev1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- docuhand/__init__.py +3 -0
- docuhand/__main__.py +4 -0
- docuhand/_compat.py +28 -0
- docuhand/engine/__init__.py +38 -0
- docuhand/engine/com_thread.py +88 -0
- docuhand/engine/com_utils.py +56 -0
- docuhand/engine/container_guard.py +166 -0
- docuhand/engine/convert_plan.py +43 -0
- docuhand/engine/edit_plan.py +71 -0
- docuhand/engine/extraction.py +175 -0
- docuhand/engine/live_edit.py +202 -0
- docuhand/engine/merge_plan.py +58 -0
- docuhand/engine/office_app.py +828 -0
- docuhand/engine/pdf_plan.py +49 -0
- docuhand/engine/template_plan.py +107 -0
- docuhand/engine/templating.py +316 -0
- docuhand/engine/wd_constants.py +23 -0
- docuhand/errors.py +202 -0
- docuhand/safety/__init__.py +10 -0
- docuhand/safety/allowlist.py +41 -0
- docuhand/safety/audit.py +52 -0
- docuhand/safety/policy.py +29 -0
- docuhand/server.py +221 -0
- docuhand/tools/__init__.py +7 -0
- docuhand/tools/convert.py +228 -0
- docuhand/tools/edit_open.py +92 -0
- docuhand/tools/export_pdf.py +92 -0
- docuhand/tools/extract.py +63 -0
- docuhand/tools/fill_template.py +115 -0
- docuhand/tools/inspect.py +57 -0
- docuhand/tools/merge.py +143 -0
- docuhand-0.1.0.dev1.dist-info/METADATA +14 -0
- docuhand-0.1.0.dev1.dist-info/RECORD +36 -0
- docuhand-0.1.0.dev1.dist-info/WHEEL +4 -0
- docuhand-0.1.0.dev1.dist-info/entry_points.txt +2 -0
- docuhand-0.1.0.dev1.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Pure planning for export_pdf — page-range grammar, zero COM.
|
|
2
|
+
|
|
3
|
+
v0.1 grammar: ``"all"`` / ``None`` (whole document) or a single contiguous
|
|
4
|
+
1-based range ``"7"`` / ``"3-9"``. Word's ExportAsFixedFormat only accepts
|
|
5
|
+
one From/To pair anyway; comma lists would silently lie about fidelity.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
WD_EXPORT_FORMAT_PDF = 17
|
|
13
|
+
WD_EXPORT_ALL_DOCUMENT = 0
|
|
14
|
+
WD_EXPORT_FROM_TO = 3
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def parse_page_range(page_range: Any) -> tuple[int, int] | None:
|
|
18
|
+
"""``"3-9"`` → (3, 9); ``"7"`` → (7, 7); ``None``/``"all"`` → None.
|
|
19
|
+
|
|
20
|
+
Raises ValueError with an llm-oriented message on anything else.
|
|
21
|
+
"""
|
|
22
|
+
if page_range is None:
|
|
23
|
+
return None
|
|
24
|
+
if not isinstance(page_range, str):
|
|
25
|
+
raise ValueError(f"page_range must be a string like 'all', '7' or '3-9' (got {type(page_range).__name__})")
|
|
26
|
+
text = page_range.strip().lower()
|
|
27
|
+
if text in ("", "all", "*"):
|
|
28
|
+
return None
|
|
29
|
+
parts = text.replace("–", "-").replace("—", "-").split("-")
|
|
30
|
+
if len(parts) == 1:
|
|
31
|
+
try:
|
|
32
|
+
page = int(parts[0])
|
|
33
|
+
except ValueError:
|
|
34
|
+
raise ValueError(f"page_range must be 'all', a page number, or 'from-to' (got {page_range!r})") from None
|
|
35
|
+
if page < 1:
|
|
36
|
+
raise ValueError(f"page numbers are 1-based (got {page_range!r})")
|
|
37
|
+
return (page, page)
|
|
38
|
+
if len(parts) == 2:
|
|
39
|
+
try:
|
|
40
|
+
lo, hi = int(parts[0]), int(parts[1])
|
|
41
|
+
except ValueError:
|
|
42
|
+
raise ValueError(f"page_range must be 'all', a page number, or 'from-to' (got {page_range!r})") from None
|
|
43
|
+
if lo < 1 or hi < lo:
|
|
44
|
+
raise ValueError(f"page_range must satisfy 1 <= from <= to (got {page_range!r})")
|
|
45
|
+
return (lo, hi)
|
|
46
|
+
raise ValueError(
|
|
47
|
+
f"only a single contiguous page range is supported in v0.1 (got {page_range!r}); "
|
|
48
|
+
"call export_pdf once per range"
|
|
49
|
+
)
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Pure planning for fill_template — zero COM, fully unit-testable.
|
|
2
|
+
|
|
3
|
+
Template fill modes (v0.1):
|
|
4
|
+
- ``content_controls`` map a control's Tag/Title to a data key
|
|
5
|
+
- ``bookmarks`` replace the bookmark's range with the value
|
|
6
|
+
- ``placeholders`` find/replace literal ``{{key}}`` tokens
|
|
7
|
+
|
|
8
|
+
``auto`` runs all three in that order, so one template can mix mechanisms.
|
|
9
|
+
The engine only ever receives validated (template, output, data, mode).
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
# Wildcard Find pattern for leftover {{...}} scan (\{ and \} are escaped
|
|
18
|
+
# braces in Word wildcard syntax; * = any run of characters).
|
|
19
|
+
LEFTOVER_WILDCARD = r"\{\{*\}\}"
|
|
20
|
+
|
|
21
|
+
MAX_KEY_LEN = 100
|
|
22
|
+
MAX_VALUE_LEN = 200_000
|
|
23
|
+
OUTPUT_FORMATS = {".doc": 0, ".docx": 16} # SaveFormat numbers (pitfall #1)
|
|
24
|
+
|
|
25
|
+
_MODES = ("auto", "content_controls", "bookmarks", "placeholders")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def normalize_mode(mode: str) -> str:
|
|
29
|
+
m = (mode or "auto").strip().lower()
|
|
30
|
+
if m not in _MODES:
|
|
31
|
+
raise ValueError(
|
|
32
|
+
f"Unsupported mode: {mode!r} (use 'auto', 'placeholders', 'bookmarks' or 'content_controls')"
|
|
33
|
+
)
|
|
34
|
+
return m
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def validate_data(data: Any) -> dict[str, str]:
|
|
38
|
+
"""Coerce to {str: str} and enforce key/value sanity rules.
|
|
39
|
+
|
|
40
|
+
Keys must be 1..MAX_KEY_LEN chars without braces or newlines (they are
|
|
41
|
+
spliced into a literal ``{{key}}`` search token and a Word Find argument,
|
|
42
|
+
both of which break on those). Values are stringified and length-capped.
|
|
43
|
+
"""
|
|
44
|
+
if not isinstance(data, dict):
|
|
45
|
+
raise ValueError("data must be a JSON object mapping keys to replacement text")
|
|
46
|
+
clean: dict[str, str] = {}
|
|
47
|
+
for key, value in data.items():
|
|
48
|
+
k = str(key)
|
|
49
|
+
if not (0 < len(k) <= MAX_KEY_LEN):
|
|
50
|
+
raise ValueError(f"data key length must be 1..{MAX_KEY_LEN}: {k[:40]!r}")
|
|
51
|
+
if any(ch in k for ch in "{}\r\n\x0b"):
|
|
52
|
+
raise ValueError(
|
|
53
|
+
f"data key must not contain braces or newlines: {k[:40]!r} "
|
|
54
|
+
"(the key is used inside a literal {{key}} token)"
|
|
55
|
+
)
|
|
56
|
+
v = value if isinstance(value, str) else str(value)
|
|
57
|
+
if len(v) > MAX_VALUE_LEN:
|
|
58
|
+
raise ValueError(
|
|
59
|
+
f"value for {k[:40]!r} exceeds {MAX_VALUE_LEN} chars; "
|
|
60
|
+
"split the document work into multiple calls"
|
|
61
|
+
)
|
|
62
|
+
clean[k] = v
|
|
63
|
+
return clean
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def resolve_output(template: Path, output: str | None) -> Path:
|
|
67
|
+
"""Default output = '<stem>_filled' next to the template, same format.
|
|
68
|
+
|
|
69
|
+
The template file itself is sacred — same-path output is rejected and
|
|
70
|
+
only .doc/.docx targets are supported in v0.1.
|
|
71
|
+
"""
|
|
72
|
+
t = template.expanduser()
|
|
73
|
+
if output:
|
|
74
|
+
out = Path(output).expanduser()
|
|
75
|
+
else:
|
|
76
|
+
out = t.with_name(t.stem + "_filled" + t.suffix.lower())
|
|
77
|
+
ext = out.suffix.lower()
|
|
78
|
+
if ext not in OUTPUT_FORMATS:
|
|
79
|
+
raise ValueError(
|
|
80
|
+
f"output_path extension must be .doc or .docx (got {ext or 'none'}); "
|
|
81
|
+
"use export_pdf for PDF output"
|
|
82
|
+
)
|
|
83
|
+
try:
|
|
84
|
+
same = out.exists() and out.samefile(t)
|
|
85
|
+
except OSError:
|
|
86
|
+
same = str(out.resolve()).lower() == str(t.resolve()).lower()
|
|
87
|
+
if same:
|
|
88
|
+
raise ValueError(
|
|
89
|
+
f"output_path must differ from the template ({t}); "
|
|
90
|
+
"fill_template never rewrites its own template"
|
|
91
|
+
)
|
|
92
|
+
return out
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def parse_bookmark_names(bookmark_args: Any) -> list[str]:
|
|
96
|
+
"""Optional explicit bookmark list — None/omitted means 'all bookmarks'."""
|
|
97
|
+
if bookmark_args is None:
|
|
98
|
+
return []
|
|
99
|
+
if not isinstance(bookmark_args, (list, tuple)):
|
|
100
|
+
raise ValueError("bookmarks must be a list of bookmark names")
|
|
101
|
+
names: list[str] = []
|
|
102
|
+
for b in bookmark_args:
|
|
103
|
+
s = str(b).strip()
|
|
104
|
+
if not s:
|
|
105
|
+
raise ValueError("bookmark names must be non-empty")
|
|
106
|
+
names.append(s)
|
|
107
|
+
return names
|
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
"""Templating engine: fill templates via content controls / bookmarks /
|
|
2
|
+
``{{placeholder}}`` tokens — STA-only, same failover contract as the rest
|
|
3
|
+
of the engine (docs/pitfalls.md battle scars applied throughout):
|
|
4
|
+
|
|
5
|
+
- **NameFarEast restore** (production scar #1): replacing ``Range.Text``
|
|
6
|
+
resets East-Asian font names to the theme default. Any value spliced
|
|
7
|
+
into a range whose text contains CJK gets its NameFarEast re-set from a
|
|
8
|
+
pre-read font snapshot, and the whole document gets a final sweep before
|
|
9
|
+
SaveAs2 — because SaveAs2 itself can reset it again.
|
|
10
|
+
- **Literal Find, never wildcards**, for ``{{key}}`` tokens: data values
|
|
11
|
+
are user text; a wildcard pass would let ``[a-z]*`` in a value eat the
|
|
12
|
+
document. Keys are validated to exclude braces/newlines anyway.
|
|
13
|
+
- Old-engine nets catch AttributeError (pitfall #13): ContentControls and
|
|
14
|
+
Find.Wildcards may not exist on Word 2007 / WPS oddities — degrade, don't die.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from typing import Any
|
|
21
|
+
|
|
22
|
+
from ..errors import (
|
|
23
|
+
DocuhandError,
|
|
24
|
+
EngineUnavailableError,
|
|
25
|
+
FileNotFoundError,
|
|
26
|
+
OpenFailedError,
|
|
27
|
+
PasswordProtectedError,
|
|
28
|
+
TemplateFillError,
|
|
29
|
+
)
|
|
30
|
+
from .container_guard import detect_encryption, validate_container
|
|
31
|
+
from .com_utils import (
|
|
32
|
+
_g,
|
|
33
|
+
_looks_encrypted,
|
|
34
|
+
_looks_like_arg_mismatch,
|
|
35
|
+
_short_exc,
|
|
36
|
+
file_claims_encrypted,
|
|
37
|
+
)
|
|
38
|
+
from .wd_constants import WD_DO_NOT_SAVE_CHANGES
|
|
39
|
+
from pathlib import Path
|
|
40
|
+
|
|
41
|
+
# SaveFormat numbers (pitfall #1 — drive Word by number, never by name)
|
|
42
|
+
_CJK_RE = re.compile(r"[\u3000-\u9fff\uf900-\ufaff\uff00-\uffef]")
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _has_cjk(text: str) -> bool:
|
|
46
|
+
return bool(_CJK_RE.search(text or ""))
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _snapshot_font(font: Any) -> dict[str, Any]:
|
|
50
|
+
"""Capture the fonts of a range BEFORE any text replacement."""
|
|
51
|
+
return {
|
|
52
|
+
"name": _g(lambda: font.Name),
|
|
53
|
+
"far_east": _g(lambda: font.NameFarEast),
|
|
54
|
+
"size": _g(lambda: font.Size),
|
|
55
|
+
"bold": _g(lambda: font.Bold),
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _restore_font(font: Any, snap: dict[str, Any], value: str) -> None:
|
|
60
|
+
"""Re-apply what the replacement may have clobbered.
|
|
61
|
+
|
|
62
|
+
Only touch NameFarEast when the spliced value actually contains CJK
|
|
63
|
+
(a pure-Latin value has no far-east glyphs, and re-asserting a CJK font
|
|
64
|
+
name over a Latin-only run is harmless but pointless).
|
|
65
|
+
"""
|
|
66
|
+
if snap.get("far_east") and _has_cjk(value):
|
|
67
|
+
try:
|
|
68
|
+
font.NameFarEast = snap["far_east"]
|
|
69
|
+
except Exception:
|
|
70
|
+
pass
|
|
71
|
+
if snap.get("name"):
|
|
72
|
+
try:
|
|
73
|
+
font.Name = snap["name"]
|
|
74
|
+
except Exception:
|
|
75
|
+
pass
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _replace_bookmarks(doc: Any, data: dict[str, str], wanted: list[str]) -> tuple[list[str], list[str]]:
|
|
79
|
+
"""Fill bookmarks; returns (filled, missing) key lists."""
|
|
80
|
+
filled: list[str] = []
|
|
81
|
+
missing: list[str] = []
|
|
82
|
+
try:
|
|
83
|
+
count = int(doc.Bookmarks.Count)
|
|
84
|
+
except Exception:
|
|
85
|
+
count = 0
|
|
86
|
+
present: dict[str, Any] = {}
|
|
87
|
+
for i in range(count):
|
|
88
|
+
bm = _g(lambda i=i: doc.Bookmarks.Item(i + 1))
|
|
89
|
+
name = _g(lambda bm=bm: bm.Name)
|
|
90
|
+
if name:
|
|
91
|
+
present[name] = bm
|
|
92
|
+
targets = wanted if wanted else list(present.keys())
|
|
93
|
+
for key in targets:
|
|
94
|
+
if key not in data:
|
|
95
|
+
continue # a bookmark without data is not an error — leave it
|
|
96
|
+
bm = present.get(key)
|
|
97
|
+
if bm is None:
|
|
98
|
+
missing.append(key)
|
|
99
|
+
continue
|
|
100
|
+
rng = _g(lambda bm=bm: bm.Range)
|
|
101
|
+
if rng is None:
|
|
102
|
+
missing.append(key)
|
|
103
|
+
continue
|
|
104
|
+
value = data[key]
|
|
105
|
+
snap = _snapshot_font(_g(lambda rng=rng: rng.Font, {}) or {})
|
|
106
|
+
try:
|
|
107
|
+
rng.Text = value
|
|
108
|
+
except Exception:
|
|
109
|
+
missing.append(key)
|
|
110
|
+
continue
|
|
111
|
+
_restore_font(_g(lambda rng=rng: rng.Font, {}) or {}, snap, value)
|
|
112
|
+
filled.append(key)
|
|
113
|
+
return filled, missing
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _replace_placeholders(doc: Any, data: dict[str, str]) -> tuple[list[str], list[str]]:
|
|
117
|
+
"""Literal {{key}} → value. Returns (filled, unmatched).
|
|
118
|
+
|
|
119
|
+
PORTABLE RECIPE (probe-verified on Word-hijack 12.0 and real KWPS):
|
|
120
|
+
``Find.Execute(Replace:=wdReplaceAll)`` reports True but silently
|
|
121
|
+
replaces NOTHING on this engine family — the word-processor accepts
|
|
122
|
+
the call and drops the Replacement. The two-step works everywhere:
|
|
123
|
+
per iteration take a FRESH full-story Range + Find (ClearFormatting,
|
|
124
|
+
Forward, Wrap=wdFindContinue), Execute(token) with no Replace arg,
|
|
125
|
+
then set ``f.Parent.Text = value`` (Parent IS the matched range).
|
|
126
|
+
Loop until Execute returns False. Never let the data value leak into
|
|
127
|
+
FindText (injection would loop forever on self-matching values).
|
|
128
|
+
"""
|
|
129
|
+
filled: list[str] = []
|
|
130
|
+
unmatched: list[str] = []
|
|
131
|
+
for key, value in data.items():
|
|
132
|
+
token = "{{" + key + "}}"
|
|
133
|
+
n = 0
|
|
134
|
+
while n < 500: # hard cap: a self-replicating value cannot hang us
|
|
135
|
+
find = _g(lambda: doc.Content.Find, None)
|
|
136
|
+
if find is None:
|
|
137
|
+
break
|
|
138
|
+
try:
|
|
139
|
+
find.ClearFormatting()
|
|
140
|
+
find.Forward = True
|
|
141
|
+
find.Wrap = 1 # wdFindContinue — whole story from the top
|
|
142
|
+
except Exception:
|
|
143
|
+
pass
|
|
144
|
+
try:
|
|
145
|
+
hit = find.Execute(token) # find only — NEVER Replace here
|
|
146
|
+
except Exception:
|
|
147
|
+
break
|
|
148
|
+
if not hit:
|
|
149
|
+
break
|
|
150
|
+
try:
|
|
151
|
+
find.Parent.Text = value # replace the matched range
|
|
152
|
+
except Exception:
|
|
153
|
+
unmatched.append(key)
|
|
154
|
+
break
|
|
155
|
+
n += 1
|
|
156
|
+
if n:
|
|
157
|
+
filled.append(key)
|
|
158
|
+
else:
|
|
159
|
+
unmatched.append(key)
|
|
160
|
+
return filled, unmatched
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _count_leftover_placeholders(doc: Any) -> list[str]:
|
|
164
|
+
"""Scan for {{...}} remnants so the agent learns what did not fill.
|
|
165
|
+
|
|
166
|
+
Two-step find (same portable recipe as replacement): fresh Content
|
|
167
|
+
Range per iteration; Parent.Text reads the matched token.
|
|
168
|
+
"""
|
|
169
|
+
leftovers: list[str] = []
|
|
170
|
+
for _ in range(50): # report at most 50 — enough to act on
|
|
171
|
+
find = _g(lambda: doc.Content.Find, None)
|
|
172
|
+
if find is None:
|
|
173
|
+
break
|
|
174
|
+
try:
|
|
175
|
+
find.ClearFormatting()
|
|
176
|
+
find.Forward = True
|
|
177
|
+
find.Wrap = 1
|
|
178
|
+
find.MatchWildcards = True
|
|
179
|
+
hit = find.Execute(r"\{\{*\}\}")
|
|
180
|
+
except Exception:
|
|
181
|
+
# wildcard mode unsupported on this engine — fall back to
|
|
182
|
+
# literal scan of the two most common remnants
|
|
183
|
+
try:
|
|
184
|
+
find2 = doc.Content.Find
|
|
185
|
+
find2.ClearFormatting()
|
|
186
|
+
find2.MatchWildcards = False
|
|
187
|
+
find2.Wrap = 1
|
|
188
|
+
hit = find2.Execute("{{")
|
|
189
|
+
except Exception:
|
|
190
|
+
break
|
|
191
|
+
if not hit:
|
|
192
|
+
break
|
|
193
|
+
token = _g(lambda f2=find2: f2.Parent.Text, "")
|
|
194
|
+
if token:
|
|
195
|
+
leftovers.append(str(token)[:120])
|
|
196
|
+
try:
|
|
197
|
+
find2.Parent.Text = "" # strip so the loop advances
|
|
198
|
+
except Exception:
|
|
199
|
+
break
|
|
200
|
+
continue
|
|
201
|
+
if not hit:
|
|
202
|
+
break
|
|
203
|
+
token = _g(lambda: find.Parent.Text, "")
|
|
204
|
+
if token:
|
|
205
|
+
leftovers.append(str(token)[:120])
|
|
206
|
+
try:
|
|
207
|
+
find.Parent.Text = "" # strip so the loop advances
|
|
208
|
+
except Exception:
|
|
209
|
+
break
|
|
210
|
+
return leftovers
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def _replace_content_controls(doc: Any, data: dict[str, str]) -> tuple[list[str], list[str]]:
|
|
214
|
+
"""Map ContentControl Tag/Title → data key. Optional path (2007 lacks it)."""
|
|
215
|
+
filled: list[str] = []
|
|
216
|
+
unmatched: list[str] = []
|
|
217
|
+
try:
|
|
218
|
+
controls = doc.ContentControls
|
|
219
|
+
count = int(controls.Count)
|
|
220
|
+
except AttributeError:
|
|
221
|
+
return filled, unmatched # old engine — degrade silently
|
|
222
|
+
except Exception:
|
|
223
|
+
return filled, unmatched
|
|
224
|
+
for i in range(count):
|
|
225
|
+
cc = _g(lambda i=i: controls.Item(i + 1))
|
|
226
|
+
if cc is None:
|
|
227
|
+
continue
|
|
228
|
+
tag = _g(lambda cc=cc: cc.Tag, "") or ""
|
|
229
|
+
title = _g(lambda cc=cc: cc.Title, "") or ""
|
|
230
|
+
key = tag if tag in data else (title if title in data else None)
|
|
231
|
+
if key is None:
|
|
232
|
+
continue
|
|
233
|
+
try:
|
|
234
|
+
# Setting Range.Text keeps the control (and its styling) alive;
|
|
235
|
+
# plain text mapping only in v0.1.
|
|
236
|
+
cc.LockContents = False
|
|
237
|
+
cc.Range.Text = data[key]
|
|
238
|
+
filled.append(key)
|
|
239
|
+
except Exception:
|
|
240
|
+
unmatched.append(key)
|
|
241
|
+
return filled, unmatched
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _final_far_east_sweep(doc: Any) -> None:
|
|
245
|
+
"""Whole-document NameFarEast re-assertion before/after SaveAs2.
|
|
246
|
+
|
|
247
|
+
The production scar (200 government personnel files): replacing
|
|
248
|
+
``Range.Text`` resets East-Asian font names document-wide, and SaveAs2
|
|
249
|
+
can reset them AGAIN. This sweep re-pins NameFarEast on the story
|
|
250
|
+
ranges that carry text — idempotent, and cheap compared to a
|
|
251
|
+
re-typeset.
|
|
252
|
+
"""
|
|
253
|
+
for story_getter in (
|
|
254
|
+
lambda: doc.Content,
|
|
255
|
+
lambda: doc.StoryRanges.Item(1), # wdMainTextStory
|
|
256
|
+
):
|
|
257
|
+
rng = _g(story_getter, None)
|
|
258
|
+
if rng is None:
|
|
259
|
+
continue
|
|
260
|
+
font = _g(lambda rng=rng: rng.Font, None)
|
|
261
|
+
if font is None:
|
|
262
|
+
continue
|
|
263
|
+
current = _g(lambda font=font: font.NameFarEast, None)
|
|
264
|
+
if not current:
|
|
265
|
+
# engine lost the name entirely — pin the CJK workhorse default
|
|
266
|
+
for candidate in ("宋体", "SimSun"):
|
|
267
|
+
try:
|
|
268
|
+
font.NameFarEast = candidate
|
|
269
|
+
break
|
|
270
|
+
except Exception:
|
|
271
|
+
continue
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
class _FillOutcome:
|
|
275
|
+
__slots__ = ("filled", "missing", "controls_ok", "controls_bad", "leftovers", "bookmarks_seen")
|
|
276
|
+
|
|
277
|
+
def __init__(self) -> None:
|
|
278
|
+
self.filled: list[str] = []
|
|
279
|
+
self.missing: list[str] = []
|
|
280
|
+
self.controls_ok: list[str] = []
|
|
281
|
+
self.controls_bad: list[str] = []
|
|
282
|
+
self.leftovers: list[str] = []
|
|
283
|
+
self.bookmarks_seen = 0
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _fill_open_document(
|
|
287
|
+
doc: Any,
|
|
288
|
+
data: dict[str, str],
|
|
289
|
+
mode: str,
|
|
290
|
+
bookmark_names: list[str],
|
|
291
|
+
) -> _FillOutcome:
|
|
292
|
+
out = _FillOutcome()
|
|
293
|
+
try:
|
|
294
|
+
out.bookmarks_seen = int(_g(lambda: doc.Bookmarks.Count, 0) or 0)
|
|
295
|
+
except Exception:
|
|
296
|
+
pass
|
|
297
|
+
|
|
298
|
+
if mode in ("auto", "content_controls"):
|
|
299
|
+
ok, bad = _replace_content_controls(doc, data)
|
|
300
|
+
out.controls_ok, out.controls_bad = ok, bad
|
|
301
|
+
|
|
302
|
+
if mode in ("auto", "bookmarks"):
|
|
303
|
+
# only bookmark keys actually present in data participate
|
|
304
|
+
wanted = [n for n in bookmark_names if n in data] if bookmark_names else []
|
|
305
|
+
filled, missing = _replace_bookmarks(doc, data, wanted)
|
|
306
|
+
out.filled.extend(filled)
|
|
307
|
+
out.missing.extend(missing)
|
|
308
|
+
|
|
309
|
+
if mode in ("auto", "placeholders"):
|
|
310
|
+
remaining = {k: v for k, v in data.items() if k not in set(out.filled)}
|
|
311
|
+
filled, unmatched = _replace_placeholders(doc, remaining)
|
|
312
|
+
out.filled.extend(filled)
|
|
313
|
+
out.missing.extend(unmatched)
|
|
314
|
+
|
|
315
|
+
out.leftovers = _count_leftover_placeholders(doc)
|
|
316
|
+
return out
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Shared Word constant numbers — locale-independent (docs/pitfalls.md #1).
|
|
2
|
+
|
|
3
|
+
Kept in a leaf module so engine submodules can share them without
|
|
4
|
+
import cycles. Drive Word by number, never by display string.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
# document-close / quit disposition
|
|
8
|
+
WD_DO_NOT_SAVE_CHANGES = 0
|
|
9
|
+
WD_ALERTS_NONE = 0
|
|
10
|
+
|
|
11
|
+
# ComputeStatistics selectors
|
|
12
|
+
WD_STAT_WORDS = 0
|
|
13
|
+
WD_STAT_PAGES = 2
|
|
14
|
+
WD_STAT_PARAGRAPHS = 4
|
|
15
|
+
|
|
16
|
+
# ExportAsFixedFormat / SaveAs PDF
|
|
17
|
+
WD_EXPORT_FORMAT_PDF = 17
|
|
18
|
+
WD_EXPORT_ALL_DOCUMENT = 0
|
|
19
|
+
WD_EXPORT_FROM_TO = 3
|
|
20
|
+
|
|
21
|
+
# SaveFormat
|
|
22
|
+
WD_FORMAT_DOC = 0
|
|
23
|
+
WD_FORMAT_DOCX = 16
|
docuhand/errors.py
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
"""Structured error model — architecture decision #3.
|
|
2
|
+
|
|
3
|
+
Errors are read by *LLMs*, not just humans. Every DocuHand error carries:
|
|
4
|
+
- ``error_code``: stable machine token the agent can branch on
|
|
5
|
+
- ``human_message``: what a person would want to see
|
|
6
|
+
- ``llm_hint``: what the agent should *do next* to self-correct
|
|
7
|
+
|
|
8
|
+
Tool implementations raise :class:`DocuhandError` subclasses; the server
|
|
9
|
+
layer converts them into structured result dicts so the agent can read and
|
|
10
|
+
recover from them instead of staring at a raw COM traceback.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class DocuhandError(Exception):
|
|
19
|
+
error_code = "DOCUHAND_ERROR" # never used directly; subclasses override
|
|
20
|
+
|
|
21
|
+
def __init__(
|
|
22
|
+
self,
|
|
23
|
+
human_message: str,
|
|
24
|
+
*,
|
|
25
|
+
llm_hint: str = "",
|
|
26
|
+
details: dict[str, Any] | None = None,
|
|
27
|
+
) -> None:
|
|
28
|
+
super().__init__(human_message)
|
|
29
|
+
self.human_message = human_message
|
|
30
|
+
self.llm_hint = llm_hint
|
|
31
|
+
self.details = details or {}
|
|
32
|
+
|
|
33
|
+
def to_dict(self) -> dict[str, Any]:
|
|
34
|
+
return {
|
|
35
|
+
"ok": False,
|
|
36
|
+
"error": {
|
|
37
|
+
"error_code": self.error_code,
|
|
38
|
+
"human_message": self.human_message,
|
|
39
|
+
"llm_hint": self.llm_hint,
|
|
40
|
+
"details": self.details,
|
|
41
|
+
},
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class PathNotAllowedError(DocuhandError):
|
|
46
|
+
error_code = "PATH_NOT_ALLOWED"
|
|
47
|
+
|
|
48
|
+
def __init__(self, path: str, allowlist: list[str]) -> None:
|
|
49
|
+
super().__init__(
|
|
50
|
+
f"Path is outside the configured allowlist: {path}",
|
|
51
|
+
llm_hint=(
|
|
52
|
+
"Ask the user to add the directory to the DOCUHAND_ALLOWLIST "
|
|
53
|
+
"environment variable (os.pathsep-separated absolute dirs), "
|
|
54
|
+
"or to pass a path inside an already-allowed directory."
|
|
55
|
+
),
|
|
56
|
+
details={"path": path, "allowlist": allowlist},
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
class FileNotFoundError(DocuhandError):
|
|
61
|
+
error_code = "FILE_NOT_FOUND"
|
|
62
|
+
|
|
63
|
+
def __init__(self, path: str) -> None:
|
|
64
|
+
super().__init__(
|
|
65
|
+
f"File not found: {path}",
|
|
66
|
+
llm_hint=(
|
|
67
|
+
"Check the spelling of the path and that the drive letter is "
|
|
68
|
+
"correct. Use an absolute Windows path (e.g. D:\\Reports\\a.doc)."
|
|
69
|
+
),
|
|
70
|
+
details={"path": path},
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
class FileLockedError(DocuhandError):
|
|
75
|
+
error_code = "FILE_LOCKED"
|
|
76
|
+
|
|
77
|
+
def __init__(self, path: str, detail: str = "") -> None:
|
|
78
|
+
super().__init__(
|
|
79
|
+
f"File is locked by another process: {path}" + (f" ({detail})" if detail else ""),
|
|
80
|
+
llm_hint=(
|
|
81
|
+
"The file is probably open in Word/WPS right now. For a "
|
|
82
|
+
"read-only inspection retry once; to *modify* it, ask the "
|
|
83
|
+
"user to close it first (edit_open_document arrives in v0.1)."
|
|
84
|
+
),
|
|
85
|
+
details={"path": path},
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
class InvalidParamError(DocuhandError):
|
|
90
|
+
error_code = "INVALID_PARAM"
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
class EngineUnavailableError(DocuhandError):
|
|
94
|
+
error_code = "ENGINE_UNAVAILABLE"
|
|
95
|
+
|
|
96
|
+
def __init__(self, attempts: list[dict[str, str]]) -> None:
|
|
97
|
+
super().__init__(
|
|
98
|
+
"No Office engine available (tried Word.Application and KWPS.Application)",
|
|
99
|
+
llm_hint=(
|
|
100
|
+
"Neither Microsoft Word nor WPS Office could be launched via "
|
|
101
|
+
"COM. Tell the user to install either one, or check that the "
|
|
102
|
+
"app is not stuck in a first-run dialog."
|
|
103
|
+
),
|
|
104
|
+
details={"attempts": attempts},
|
|
105
|
+
)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
class OpenFailedError(DocuhandError):
|
|
109
|
+
error_code = "OPEN_FAILED"
|
|
110
|
+
|
|
111
|
+
def __init__(self, path: str, engine: str, com_error: str) -> None:
|
|
112
|
+
super().__init__(
|
|
113
|
+
f"{engine} failed to open: {path}",
|
|
114
|
+
llm_hint=(
|
|
115
|
+
"The file may be corrupted, password-protected, or in a "
|
|
116
|
+
"format this engine cannot open. Report is_format_recognized "
|
|
117
|
+
"from the inspection payload to the user and suggest trying "
|
|
118
|
+
"the other engine (DOCUHAND_ENGINE=wps or =word)."
|
|
119
|
+
),
|
|
120
|
+
details={"path": path, "engine": engine, "com_error": com_error},
|
|
121
|
+
)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class PasswordProtectedError(DocuhandError):
|
|
125
|
+
error_code = "FILE_PASSWORD_PROTECTED"
|
|
126
|
+
|
|
127
|
+
def __init__(self, path: str, engine: str) -> None:
|
|
128
|
+
super().__init__(
|
|
129
|
+
f"Document is password-protected, cannot be opened read-only: {path}",
|
|
130
|
+
llm_hint=(
|
|
131
|
+
"Opening it would require the password. Ask the user for the "
|
|
132
|
+
"password — never guess one. No other engine will help; do "
|
|
133
|
+
"not retry."
|
|
134
|
+
),
|
|
135
|
+
details={"path": path, "engine": engine},
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class ComCallTimeoutError(DocuhandError):
|
|
140
|
+
error_code = "COM_TIMEOUT"
|
|
141
|
+
|
|
142
|
+
def __init__(self, timeout_s: float) -> None:
|
|
143
|
+
super().__init__(
|
|
144
|
+
f"COM operation timed out after {timeout_s:.0f}s",
|
|
145
|
+
llm_hint=(
|
|
146
|
+
"Word/WPS may be showing a modal dialog (first-run wizard, "
|
|
147
|
+
"activation, recovery prompt). Ask the user to check the "
|
|
148
|
+
"screen, then retry once."
|
|
149
|
+
),
|
|
150
|
+
details={"timeout_s": timeout_s},
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class ReadOnlyError(DocuhandError):
|
|
155
|
+
error_code = "SERVER_READ_ONLY"
|
|
156
|
+
|
|
157
|
+
def __init__(self, tool: str) -> None:
|
|
158
|
+
super().__init__(
|
|
159
|
+
f"Server runs in read-only mode (DOCUHAND_READONLY); '{tool}' was blocked",
|
|
160
|
+
llm_hint=(
|
|
161
|
+
"The user locked this server to read-only operation. Only "
|
|
162
|
+
"inspect_document and extract_content are available. To allow "
|
|
163
|
+
"writes the user must remove DOCUHAND_READONLY from the server "
|
|
164
|
+
"config and restart it. Do not retry the same tool."
|
|
165
|
+
),
|
|
166
|
+
details={"tool": tool},
|
|
167
|
+
)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
class TemplateFillError(DocuhandError):
|
|
171
|
+
error_code = "TEMPLATE_FILL_INCOMPLETE"
|
|
172
|
+
|
|
173
|
+
def __init__(self, path: str, detail: str, missing: list[str] | None = None) -> None:
|
|
174
|
+
super().__init__(
|
|
175
|
+
f"Template fill finished with problems: {path}",
|
|
176
|
+
llm_hint=(
|
|
177
|
+
"Check 'details.missing_keys' / 'details.problems'. Missing "
|
|
178
|
+
"bookmark keys mean the template lacks that bookmark — remove "
|
|
179
|
+
"the key from data or fix the template. Leftover {{placeholders}} "
|
|
180
|
+
"usually mean the data key spelling differs from the template."
|
|
181
|
+
),
|
|
182
|
+
details={"path": path, "detail": detail, "missing_keys": missing or []},
|
|
183
|
+
)
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
class EngineDeadError(DocuhandError):
|
|
187
|
+
error_code = "ENGINE_DIED"
|
|
188
|
+
|
|
189
|
+
def __init__(self, engine: str, com_error: str) -> None:
|
|
190
|
+
super().__init__(
|
|
191
|
+
f"The {engine} engine died mid-operation (RPC crash)",
|
|
192
|
+
llm_hint=(
|
|
193
|
+
"This engine instance crashed — the next call automatically "
|
|
194
|
+
"relaunches it (or fails over to the other engine). Retry once "
|
|
195
|
+
"before reporting a problem."
|
|
196
|
+
),
|
|
197
|
+
details={"engine": engine, "com_error": com_error},
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
class InternalError(DocuhandError):
|
|
202
|
+
error_code = "INTERNAL_ERROR"
|