exegete 0.14.1a0.dev1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- exegete/__init__.py +16 -0
- exegete/coder_comparison.py +208 -0
- exegete/cursors.py +218 -0
- exegete/database.py +12621 -0
- exegete/env_settings.py +174 -0
- exegete/memo_privacy.py +215 -0
- exegete/names.py +68 -0
- exegete/new_project.py +1027 -0
- exegete/preview_tokens.py +613 -0
- exegete/project_settings.py +1082 -0
- exegete/pseudonymise.py +2796 -0
- exegete/refi_export.py +645 -0
- exegete/server.py +18899 -0
- exegete/sessions.py +1153 -0
- exegete/state_folder.py +360 -0
- exegete/transition.py +1049 -0
- exegete-0.14.1a0.dev1.dist-info/METADATA +515 -0
- exegete-0.14.1a0.dev1.dist-info/RECORD +24 -0
- exegete-0.14.1a0.dev1.dist-info/WHEEL +5 -0
- exegete-0.14.1a0.dev1.dist-info/entry_points.txt +2 -0
- exegete-0.14.1a0.dev1.dist-info/licenses/COPYING.LESSER +165 -0
- exegete-0.14.1a0.dev1.dist-info/licenses/NOTICE +773 -0
- exegete-0.14.1a0.dev1.dist-info/licenses/legal/GPL-3.0.txt +674 -0
- exegete-0.14.1a0.dev1.dist-info/top_level.txt +1 -0
exegete/refi_export.py
ADDED
|
@@ -0,0 +1,645 @@
|
|
|
1
|
+
# SPDX-License-Identifier: LGPL-3.0-or-later
|
|
2
|
+
"""REFI-QDA XML export functionality for AI coding suggestions.
|
|
3
|
+
|
|
4
|
+
Generates REFI-QDA compliant XML files (.qdpx) that can be imported into
|
|
5
|
+
Qualcoder and other QDA software supporting the REFI-QDA standard
|
|
6
|
+
(QDA-XML 1.0, specification document v1.5).
|
|
7
|
+
|
|
8
|
+
Position convention (QualCoder's, shared by this database): character
|
|
9
|
+
offsets into the plain text as stored, 0-based, end-exclusive, newlines
|
|
10
|
+
are single \n characters. The exported .txt payloads are written verbatim
|
|
11
|
+
(UTF-8, no BOM) so positions remain valid on re-import.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
import re
|
|
15
|
+
import uuid
|
|
16
|
+
import logging
|
|
17
|
+
import xml.etree.ElementTree as ET
|
|
18
|
+
from xml.dom import minidom
|
|
19
|
+
import zipfile
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from typing import List, Dict, Optional, Set
|
|
22
|
+
from datetime import datetime, timezone
|
|
23
|
+
|
|
24
|
+
from . import names
|
|
25
|
+
from .database import QualcoderDatabase, error_label, error_text
|
|
26
|
+
from .sessions import CodingSuggestion, memo_with_reading
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger(__name__)
|
|
29
|
+
|
|
30
|
+
# REFI-QDA XML namespace: "1.5" is the revision of the SPEC DOCUMENT; the
|
|
31
|
+
# wire format is QDA-XML 1.0 and this is its one and only namespace.
|
|
32
|
+
NAMESPACE = "urn:QDA-XML:project:1.0"
|
|
33
|
+
SCHEMA_LOCATION = (
|
|
34
|
+
"urn:QDA-XML:project:1.0 "
|
|
35
|
+
"http://schema.qdasoftware.org/versions/Project/v1.0/Project.xsd"
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
# GUIDType pattern from the spec (optionally brace-wrapped is also legal,
|
|
39
|
+
# but we always emit the bare lowercase form)
|
|
40
|
+
_GUID_RE = re.compile(
|
|
41
|
+
r"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-"
|
|
42
|
+
r"[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$"
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
# RGBType pattern from the spec
|
|
46
|
+
_COLOR_RE = re.compile(r"^#([A-Fa-f0-9]{6}|[A-Fa-f0-9]{3})$")
|
|
47
|
+
|
|
48
|
+
# Spec §8.5: max internal file size
|
|
49
|
+
_MAX_INTERNAL_FILE_BYTES = 2_147_483_647
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _xml_safe(text) -> str:
|
|
53
|
+
"""Strip characters that are invalid in XML 1.0 from a string.
|
|
54
|
+
|
|
55
|
+
ElementTree will happily serialise C0 control characters, but the
|
|
56
|
+
result is not well-formed XML: minidom (and any conformant parser,
|
|
57
|
+
including the QDA tools importing the .qdpx) rejects it. Database
|
|
58
|
+
content (code names, memos, file names) is user-authored and may
|
|
59
|
+
contain such characters, so every DB-derived string is filtered
|
|
60
|
+
before it enters the XML tree.
|
|
61
|
+
|
|
62
|
+
Valid XML 1.0 chars: #x9 #xA #xD, #x20-#xD7FF, #xE000-#xFFFD,
|
|
63
|
+
#x10000-#x10FFFF (surrogates excluded by the D7FF/E000 bounds).
|
|
64
|
+
"""
|
|
65
|
+
if text is None:
|
|
66
|
+
return ""
|
|
67
|
+
return "".join(
|
|
68
|
+
ch for ch in str(text)
|
|
69
|
+
if ch in ("\t", "\n", "\r")
|
|
70
|
+
or 0x20 <= ord(ch) <= 0xD7FF
|
|
71
|
+
or 0xE000 <= ord(ch) <= 0xFFFD
|
|
72
|
+
or 0x10000 <= ord(ch) <= 0x10FFFF
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _utc_now() -> str:
|
|
77
|
+
"""xsd:dateTime in actual UTC (not local time mislabelled as Z)."""
|
|
78
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class RefiQdaExporter:
|
|
82
|
+
"""Generate REFI-QDA XML files from coding suggestions."""
|
|
83
|
+
|
|
84
|
+
def __init__(self, db: QualcoderDatabase,
|
|
85
|
+
ai_user_name: str = "AI Coding Assistant"):
|
|
86
|
+
"""Initialise exporter with database connection.
|
|
87
|
+
|
|
88
|
+
Args:
|
|
89
|
+
db: QualcoderDatabase instance
|
|
90
|
+
ai_user_name: Display name for the AI coder User element
|
|
91
|
+
(P1-2: the server passes the configured coder name).
|
|
92
|
+
The deterministic GUID key stays "ai_coder" so the
|
|
93
|
+
same project always exports the same User guid.
|
|
94
|
+
"""
|
|
95
|
+
self.db = db
|
|
96
|
+
self.ai_user_name = ai_user_name
|
|
97
|
+
|
|
98
|
+
def create_project_xml(
|
|
99
|
+
self,
|
|
100
|
+
suggestions: List[CodingSuggestion],
|
|
101
|
+
project_name: str = "AI Coding Suggestions",
|
|
102
|
+
origin: str = names.SERVER_NAME
|
|
103
|
+
) -> ET.Element:
|
|
104
|
+
"""Create the main REFI-QDA project XML structure.
|
|
105
|
+
|
|
106
|
+
Args:
|
|
107
|
+
suggestions: List of coding suggestions to export
|
|
108
|
+
project_name: Name for the REFI-QDA project
|
|
109
|
+
origin: Software origin identifier
|
|
110
|
+
|
|
111
|
+
Returns:
|
|
112
|
+
XML Element representing the project
|
|
113
|
+
"""
|
|
114
|
+
# Register namespace
|
|
115
|
+
ET.register_namespace('', NAMESPACE)
|
|
116
|
+
|
|
117
|
+
# Create root element. NOTE: attributes are UNQUALIFIED in the
|
|
118
|
+
# REFI-QDA schema (attributeFormDefault="unqualified") — a
|
|
119
|
+
# namespaced name attribute would be schema-invalid.
|
|
120
|
+
root = ET.Element(
|
|
121
|
+
f"{{{NAMESPACE}}}Project",
|
|
122
|
+
attrib={
|
|
123
|
+
"name": _xml_safe(project_name),
|
|
124
|
+
"origin": _xml_safe(origin),
|
|
125
|
+
"creatingUserGUID": self.db.get_or_create_user_guid("ai_coder"),
|
|
126
|
+
"creationDateTime": _utc_now()
|
|
127
|
+
}
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
# Add schema location
|
|
131
|
+
root.set(
|
|
132
|
+
"{http://www.w3.org/2001/XMLSchema-instance}schemaLocation",
|
|
133
|
+
SCHEMA_LOCATION
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
# Add sections (Users -> CodeBook -> Sources: the XSD sequence with
|
|
137
|
+
# the optional elements we don't emit skipped)
|
|
138
|
+
self._add_users_section(root)
|
|
139
|
+
self._add_codebook_section(root, suggestions)
|
|
140
|
+
self._add_sources_section(root, suggestions)
|
|
141
|
+
|
|
142
|
+
return root
|
|
143
|
+
|
|
144
|
+
def _add_users_section(self, root: ET.Element) -> None:
|
|
145
|
+
"""Add Users section to XML.
|
|
146
|
+
|
|
147
|
+
Args:
|
|
148
|
+
root: Root XML element
|
|
149
|
+
"""
|
|
150
|
+
users_elem = ET.SubElement(root, f"{{{NAMESPACE}}}Users")
|
|
151
|
+
|
|
152
|
+
# Add AI coder user (name from the P1-2 attribution config)
|
|
153
|
+
ET.SubElement(
|
|
154
|
+
users_elem,
|
|
155
|
+
f"{{{NAMESPACE}}}User",
|
|
156
|
+
attrib={
|
|
157
|
+
"guid": self.db.get_or_create_user_guid("ai_coder"),
|
|
158
|
+
"name": _xml_safe(self.ai_user_name)
|
|
159
|
+
}
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
def _add_codebook_section(
|
|
163
|
+
self,
|
|
164
|
+
root: ET.Element,
|
|
165
|
+
suggestions: List[CodingSuggestion]
|
|
166
|
+
) -> None:
|
|
167
|
+
"""Add CodeBook section with codes referenced in suggestions.
|
|
168
|
+
|
|
169
|
+
Category hierarchy is preserved: REFI-QDA expresses categories as
|
|
170
|
+
nested Code elements with isCodable="false" (QualCoder's own
|
|
171
|
+
convention on both export and import), so referenced codes are
|
|
172
|
+
emitted inside their full category chain.
|
|
173
|
+
|
|
174
|
+
Sub-codes (S9, v16+): a referenced sub-code is emitted NESTED
|
|
175
|
+
inside its parent code's codable Code element, recursively, with
|
|
176
|
+
the whole parent-code chain included, matching master's export
|
|
177
|
+
(refi.py:3228-3248; top-level condition :3279-3281). Master's
|
|
178
|
+
importer reconstructs supercid from codable children of codable
|
|
179
|
+
codes (refi.py:242-254), so a correct export round-trips instead
|
|
180
|
+
of silently flattening the hierarchy.
|
|
181
|
+
|
|
182
|
+
Args:
|
|
183
|
+
root: Root XML element
|
|
184
|
+
suggestions: List of coding suggestions
|
|
185
|
+
"""
|
|
186
|
+
codebook_elem = ET.SubElement(root, f"{{{NAMESPACE}}}CodeBook")
|
|
187
|
+
codes_elem = ET.SubElement(codebook_elem, f"{{{NAMESPACE}}}Codes")
|
|
188
|
+
|
|
189
|
+
code_ids = set(s.code_id for s in suggestions)
|
|
190
|
+
code_guids = self.db.get_code_guids()
|
|
191
|
+
|
|
192
|
+
all_codes = {c["id"]: c for c in self.db.list_codes()}
|
|
193
|
+
all_categories = {c["id"]: c for c in self.db.list_categories()}
|
|
194
|
+
|
|
195
|
+
# Sub-codes: pull each referenced code's parent-code chain into the
|
|
196
|
+
# export too (the sub-code must nest inside its parent element)
|
|
197
|
+
needed_codes: Set[int] = set()
|
|
198
|
+
for code_id in code_ids:
|
|
199
|
+
current = code_id
|
|
200
|
+
seen_codes: Set[int] = set()
|
|
201
|
+
while current in all_codes and current not in seen_codes:
|
|
202
|
+
seen_codes.add(current)
|
|
203
|
+
needed_codes.add(current)
|
|
204
|
+
parent = all_codes[current].get("parent_code_id")
|
|
205
|
+
if parent is None:
|
|
206
|
+
break
|
|
207
|
+
current = parent
|
|
208
|
+
|
|
209
|
+
# Which categories are needed? Walk each needed code's chain up
|
|
210
|
+
needed_categories: Set[int] = set()
|
|
211
|
+
for code_id in needed_codes:
|
|
212
|
+
code = all_codes.get(code_id)
|
|
213
|
+
if not code:
|
|
214
|
+
continue
|
|
215
|
+
cat_id = code.get("category_id")
|
|
216
|
+
seen: Set[int] = set()
|
|
217
|
+
while cat_id is not None and cat_id in all_categories and cat_id not in seen:
|
|
218
|
+
seen.add(cat_id)
|
|
219
|
+
needed_categories.add(cat_id)
|
|
220
|
+
cat_id = all_categories[cat_id].get("parent_id")
|
|
221
|
+
|
|
222
|
+
category_elems: Dict[int, ET.Element] = {}
|
|
223
|
+
|
|
224
|
+
def _category_elem(cat_id: int) -> ET.Element:
|
|
225
|
+
"""Get or create the (nested) element for a category."""
|
|
226
|
+
if cat_id in category_elems:
|
|
227
|
+
return category_elems[cat_id]
|
|
228
|
+
cat = all_categories[cat_id]
|
|
229
|
+
parent_id = cat.get("parent_id")
|
|
230
|
+
if parent_id is not None and parent_id in needed_categories:
|
|
231
|
+
parent_elem = _category_elem(parent_id)
|
|
232
|
+
else:
|
|
233
|
+
parent_elem = codes_elem
|
|
234
|
+
elem = ET.SubElement(
|
|
235
|
+
parent_elem,
|
|
236
|
+
f"{{{NAMESPACE}}}Code",
|
|
237
|
+
attrib={
|
|
238
|
+
"guid": self.db.generate_deterministic_guid("category", cat_id),
|
|
239
|
+
"name": _xml_safe(cat["name"]),
|
|
240
|
+
"isCodable": "false"
|
|
241
|
+
}
|
|
242
|
+
)
|
|
243
|
+
if cat.get("memo"):
|
|
244
|
+
desc = ET.SubElement(elem, f"{{{NAMESPACE}}}Description")
|
|
245
|
+
desc.text = _xml_safe(cat["memo"])
|
|
246
|
+
category_elems[cat_id] = elem
|
|
247
|
+
return elem
|
|
248
|
+
|
|
249
|
+
# Emit categories root-down (dict order does not matter — recursion
|
|
250
|
+
# in _category_elem builds parents first)
|
|
251
|
+
for cat_id in needed_categories:
|
|
252
|
+
_category_elem(cat_id)
|
|
253
|
+
|
|
254
|
+
# Emit the needed codes: a top-level code goes inside its category
|
|
255
|
+
# chain (or Codes root); a sub-code nests inside its parent code's
|
|
256
|
+
# element, recursively (S9)
|
|
257
|
+
code_elems: Dict[int, ET.Element] = {}
|
|
258
|
+
|
|
259
|
+
def _code_elem(code_id: int):
|
|
260
|
+
if code_id in code_elems:
|
|
261
|
+
return code_elems[code_id]
|
|
262
|
+
code = all_codes.get(code_id)
|
|
263
|
+
if not code:
|
|
264
|
+
logger.warning(f"Could not export code {code_id}: not found")
|
|
265
|
+
return None
|
|
266
|
+
parent_code = code.get("parent_code_id")
|
|
267
|
+
if parent_code is not None and parent_code in needed_codes:
|
|
268
|
+
parent_elem = _code_elem(parent_code)
|
|
269
|
+
if parent_elem is None:
|
|
270
|
+
parent_elem = codes_elem
|
|
271
|
+
else:
|
|
272
|
+
cat_id = code.get("category_id")
|
|
273
|
+
parent_elem = (category_elems.get(cat_id, codes_elem)
|
|
274
|
+
if cat_id is not None else codes_elem)
|
|
275
|
+
|
|
276
|
+
elem = ET.SubElement(
|
|
277
|
+
parent_elem,
|
|
278
|
+
f"{{{NAMESPACE}}}Code",
|
|
279
|
+
attrib={
|
|
280
|
+
"guid": code_guids[code_id],
|
|
281
|
+
"name": _xml_safe(code["name"]),
|
|
282
|
+
"isCodable": "true"
|
|
283
|
+
}
|
|
284
|
+
)
|
|
285
|
+
# Add color only when it is a valid RGBType value
|
|
286
|
+
color = code.get("color")
|
|
287
|
+
if color and _COLOR_RE.match(str(color)):
|
|
288
|
+
elem.set("color", str(color))
|
|
289
|
+
# Add description (memo) only when non-empty
|
|
290
|
+
if code.get("memo"):
|
|
291
|
+
desc_elem = ET.SubElement(elem, f"{{{NAMESPACE}}}Description")
|
|
292
|
+
desc_elem.text = _xml_safe(code["memo"])
|
|
293
|
+
code_elems[code_id] = elem
|
|
294
|
+
return elem
|
|
295
|
+
|
|
296
|
+
for code_id in sorted(needed_codes):
|
|
297
|
+
_code_elem(code_id)
|
|
298
|
+
|
|
299
|
+
def _add_sources_section(
|
|
300
|
+
self,
|
|
301
|
+
root: ET.Element,
|
|
302
|
+
suggestions: List[CodingSuggestion]
|
|
303
|
+
) -> None:
|
|
304
|
+
"""Add Sources section with files and coded segments.
|
|
305
|
+
|
|
306
|
+
Sources are referenced with the internal:// URL scheme and GUID
|
|
307
|
+
filenames per spec §8.3/8.4; QualCoder's importer hard-depends on
|
|
308
|
+
the internal:/ prefix (refi.py:880).
|
|
309
|
+
|
|
310
|
+
Args:
|
|
311
|
+
root: Root XML element
|
|
312
|
+
suggestions: List of coding suggestions
|
|
313
|
+
"""
|
|
314
|
+
sources_elem = ET.SubElement(root, f"{{{NAMESPACE}}}Sources")
|
|
315
|
+
|
|
316
|
+
# Group suggestions by file
|
|
317
|
+
file_suggestions: Dict[int, List[CodingSuggestion]] = {}
|
|
318
|
+
for suggestion in suggestions:
|
|
319
|
+
file_suggestions.setdefault(suggestion.file_id, []).append(suggestion)
|
|
320
|
+
|
|
321
|
+
file_guids = self.db.get_file_guids()
|
|
322
|
+
code_guids = self.db.get_code_guids()
|
|
323
|
+
user_guid = self.db.get_or_create_user_guid("ai_coder")
|
|
324
|
+
|
|
325
|
+
# GUID uniqueness within one document is mandatory (spec §3)
|
|
326
|
+
used_guids: Set[str] = set()
|
|
327
|
+
|
|
328
|
+
# Create a TextSource for each file
|
|
329
|
+
for file_id, file_sug_list in file_suggestions.items():
|
|
330
|
+
file_content = self.db.get_file_content(file_id)
|
|
331
|
+
if not file_content or not (file_content.get("content") or ""):
|
|
332
|
+
# Never emit a TextSource whose payload will not be written:
|
|
333
|
+
# a dangling plainTextPath crashes QualCoder's importer.
|
|
334
|
+
# export_to_qdpx validates this up front, so this is only a
|
|
335
|
+
# defensive skip for direct callers.
|
|
336
|
+
logger.warning(
|
|
337
|
+
f"Skipping file {file_id}: no text content to export"
|
|
338
|
+
)
|
|
339
|
+
continue
|
|
340
|
+
|
|
341
|
+
source_elem = ET.SubElement(
|
|
342
|
+
sources_elem,
|
|
343
|
+
f"{{{NAMESPACE}}}TextSource",
|
|
344
|
+
attrib={
|
|
345
|
+
"guid": file_guids[file_id],
|
|
346
|
+
"name": _xml_safe(file_content["name"]),
|
|
347
|
+
"plainTextPath": f"internal://{file_guids[file_id]}.txt",
|
|
348
|
+
"creatingUser": user_guid,
|
|
349
|
+
"creationDateTime": _utc_now()
|
|
350
|
+
}
|
|
351
|
+
)
|
|
352
|
+
|
|
353
|
+
# Add description (memo) only when non-empty
|
|
354
|
+
if file_content.get("memo"):
|
|
355
|
+
desc_elem = ET.SubElement(source_elem, f"{{{NAMESPACE}}}Description")
|
|
356
|
+
desc_elem.text = _xml_safe(file_content["memo"])
|
|
357
|
+
|
|
358
|
+
# Add PlainTextSelection elements for each suggestion
|
|
359
|
+
for suggestion in file_sug_list:
|
|
360
|
+
self._add_plain_text_selection(
|
|
361
|
+
source_elem,
|
|
362
|
+
suggestion,
|
|
363
|
+
code_guids,
|
|
364
|
+
user_guid,
|
|
365
|
+
used_guids
|
|
366
|
+
)
|
|
367
|
+
|
|
368
|
+
def _add_plain_text_selection(
|
|
369
|
+
self,
|
|
370
|
+
source_elem: ET.Element,
|
|
371
|
+
suggestion: CodingSuggestion,
|
|
372
|
+
code_guids: Dict[int, str],
|
|
373
|
+
user_guid: str,
|
|
374
|
+
used_guids: Set[str]
|
|
375
|
+
) -> None:
|
|
376
|
+
"""Create PlainTextSelection element for a coded segment.
|
|
377
|
+
|
|
378
|
+
Args:
|
|
379
|
+
source_elem: Parent source element
|
|
380
|
+
suggestion: CodingSuggestion to export
|
|
381
|
+
code_guids: Mapping of code IDs to GUIDs
|
|
382
|
+
user_guid: GUID of the creating user
|
|
383
|
+
used_guids: GUIDs already used in this document (uniqueness is
|
|
384
|
+
mandatory within one REFI-QDA document)
|
|
385
|
+
"""
|
|
386
|
+
# Selection GUID: session-supplied values are untrusted — validate
|
|
387
|
+
# the format and document-uniqueness, minting a fresh one otherwise
|
|
388
|
+
selection_guid = suggestion.guid
|
|
389
|
+
if (not isinstance(selection_guid, str)
|
|
390
|
+
or not _GUID_RE.match(selection_guid)
|
|
391
|
+
or selection_guid in used_guids):
|
|
392
|
+
selection_guid = str(uuid.uuid4())
|
|
393
|
+
used_guids.add(selection_guid)
|
|
394
|
+
|
|
395
|
+
selection_elem = ET.SubElement(
|
|
396
|
+
source_elem,
|
|
397
|
+
f"{{{NAMESPACE}}}PlainTextSelection",
|
|
398
|
+
attrib={
|
|
399
|
+
"guid": selection_guid,
|
|
400
|
+
"startPosition": str(suggestion.start_pos),
|
|
401
|
+
"endPosition": str(suggestion.end_pos),
|
|
402
|
+
"creatingUser": user_guid,
|
|
403
|
+
"creationDateTime": _utc_now()
|
|
404
|
+
}
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
# Description: the reading in words, then the reasoning, as
|
|
408
|
+
# apply_codings writes the memo (owner ruling 21: never a number).
|
|
409
|
+
# A project export's rows carry no label of their own (reading is
|
|
410
|
+
# None): an applied AI coding's memo already says it in words.
|
|
411
|
+
memo_text = memo_with_reading(suggestion.reasoning or "",
|
|
412
|
+
suggestion.reading)
|
|
413
|
+
if memo_text:
|
|
414
|
+
desc_elem = ET.SubElement(selection_elem, f"{{{NAMESPACE}}}Description")
|
|
415
|
+
desc_elem.text = _xml_safe(memo_text)
|
|
416
|
+
|
|
417
|
+
# Coding GUID: deterministic over file/code/BOTH positions (start
|
|
418
|
+
# alone collided for same-start selections), deduped per document
|
|
419
|
+
coding_guid = self.db.generate_deterministic_guid(
|
|
420
|
+
"coding",
|
|
421
|
+
f"{suggestion.file_id}_{suggestion.code_id}_"
|
|
422
|
+
f"{suggestion.start_pos}_{suggestion.end_pos}"
|
|
423
|
+
)
|
|
424
|
+
if coding_guid in used_guids:
|
|
425
|
+
coding_guid = str(uuid.uuid4())
|
|
426
|
+
used_guids.add(coding_guid)
|
|
427
|
+
|
|
428
|
+
coding_elem = ET.SubElement(
|
|
429
|
+
selection_elem,
|
|
430
|
+
f"{{{NAMESPACE}}}Coding",
|
|
431
|
+
attrib={
|
|
432
|
+
"guid": coding_guid,
|
|
433
|
+
"creatingUser": user_guid,
|
|
434
|
+
"creationDateTime": _utc_now()
|
|
435
|
+
}
|
|
436
|
+
)
|
|
437
|
+
|
|
438
|
+
# Add CodeRef pointing to the code
|
|
439
|
+
ET.SubElement(
|
|
440
|
+
coding_elem,
|
|
441
|
+
f"{{{NAMESPACE}}}CodeRef",
|
|
442
|
+
attrib={
|
|
443
|
+
"targetGUID": code_guids[suggestion.code_id]
|
|
444
|
+
}
|
|
445
|
+
)
|
|
446
|
+
|
|
447
|
+
def prettify_xml(self, elem: ET.Element) -> str:
|
|
448
|
+
"""Convert XML element to pretty-printed string.
|
|
449
|
+
|
|
450
|
+
Args:
|
|
451
|
+
elem: XML element to prettify
|
|
452
|
+
|
|
453
|
+
Returns:
|
|
454
|
+
Pretty-printed XML string
|
|
455
|
+
"""
|
|
456
|
+
# Convert to string
|
|
457
|
+
rough_string = ET.tostring(elem, encoding='utf-8')
|
|
458
|
+
|
|
459
|
+
# Parse and prettify. Security note: this parses ONLY the string we
|
|
460
|
+
# just serialized ourselves (never external/untrusted XML), so the
|
|
461
|
+
# stdlib parser's XXE/entity-expansion caveats do not apply here.
|
|
462
|
+
reparsed = minidom.parseString(rough_string)
|
|
463
|
+
return reparsed.toprettyxml(indent=" ", encoding='utf-8').decode('utf-8')
|
|
464
|
+
|
|
465
|
+
def export_to_qdpx(
|
|
466
|
+
self,
|
|
467
|
+
suggestions: List[CodingSuggestion],
|
|
468
|
+
output_path: str,
|
|
469
|
+
project_name: str = "AI Coding Suggestions"
|
|
470
|
+
) -> str:
|
|
471
|
+
"""Export suggestions as .qdpx file (ZIP with XML).
|
|
472
|
+
|
|
473
|
+
Container layout per spec §8: project.qde at the archive root plus
|
|
474
|
+
a flat sources/ folder whose members are GUID-named .txt files
|
|
475
|
+
(UTF-8, no BOM), referenced from the XML as internal://<guid>.txt.
|
|
476
|
+
|
|
477
|
+
Suggestions are validated first; stale code/file references,
|
|
478
|
+
out-of-bounds positions, or files without text content fail the
|
|
479
|
+
export loudly instead of producing a .qdpx that crashes importers.
|
|
480
|
+
|
|
481
|
+
Args:
|
|
482
|
+
suggestions: List of coding suggestions to export
|
|
483
|
+
output_path: Path where to save the .qdpx file
|
|
484
|
+
project_name: Name for the REFI-QDA project
|
|
485
|
+
|
|
486
|
+
Returns:
|
|
487
|
+
Path to created .qdpx file
|
|
488
|
+
|
|
489
|
+
Raises:
|
|
490
|
+
ValueError: If any suggestion fails validation
|
|
491
|
+
RuntimeError: If the export itself fails
|
|
492
|
+
"""
|
|
493
|
+
# Validate BEFORE building anything (Gap 4: stale IDs used to
|
|
494
|
+
# KeyError mid-export; empty-content files left dangling members)
|
|
495
|
+
problems = self.validate_suggestions(suggestions)
|
|
496
|
+
if problems:
|
|
497
|
+
shown = "; ".join(problems[:10])
|
|
498
|
+
more = f" (+{len(problems) - 10} more)" if len(problems) > 10 else ""
|
|
499
|
+
raise ValueError(f"Export validation failed: {shown}{more}")
|
|
500
|
+
|
|
501
|
+
try:
|
|
502
|
+
# Create output directory if needed
|
|
503
|
+
output_file = Path(output_path).expanduser()
|
|
504
|
+
output_file.parent.mkdir(parents=True, exist_ok=True)
|
|
505
|
+
|
|
506
|
+
# Generate XML
|
|
507
|
+
logger.info("Generating REFI-QDA XML...")
|
|
508
|
+
xml_root = self.create_project_xml(suggestions, project_name)
|
|
509
|
+
|
|
510
|
+
# Serialize ONCE. The previous minidom pretty-print re-parse
|
|
511
|
+
# roughly doubled peak memory (~147 MB for a 4 MB project) and
|
|
512
|
+
# dominated wall-clock at scale (track6); pretty-printing a
|
|
513
|
+
# machine-interchange file buys nothing. prettify_xml remains
|
|
514
|
+
# available for callers that want readable output.
|
|
515
|
+
xml_string = ET.tostring(
|
|
516
|
+
xml_root, encoding="unicode", xml_declaration=True
|
|
517
|
+
)
|
|
518
|
+
|
|
519
|
+
file_guids = self.db.get_file_guids()
|
|
520
|
+
|
|
521
|
+
# Create .qdpx file (ZIP archive)
|
|
522
|
+
logger.info("Creating a .qdpx archive")
|
|
523
|
+
with zipfile.ZipFile(output_file, 'w', zipfile.ZIP_DEFLATED) as zipf:
|
|
524
|
+
# project.qde at the root, lowercase (QualCoder's importer
|
|
525
|
+
# hard-codes this name). Python str -> UTF-8, never a BOM:
|
|
526
|
+
# a BOM would silently shift every position by one on
|
|
527
|
+
# re-import.
|
|
528
|
+
zipf.writestr('project.qde', xml_string)
|
|
529
|
+
|
|
530
|
+
# Add source payloads under sources/<guid>.txt, verbatim
|
|
531
|
+
# (positions are only meaningful against the exact text)
|
|
532
|
+
file_ids = set(s.file_id for s in suggestions)
|
|
533
|
+
logger.info(f"Adding {len(file_ids)} source files to archive...")
|
|
534
|
+
|
|
535
|
+
for file_id in file_ids:
|
|
536
|
+
file_content = self.db.get_file_content(file_id)
|
|
537
|
+
content = (file_content or {}).get("content") or ""
|
|
538
|
+
if not content:
|
|
539
|
+
continue # validated above; defensive only
|
|
540
|
+
member = f"sources/{file_guids[file_id]}.txt"
|
|
541
|
+
zipf.writestr(member, content)
|
|
542
|
+
logger.debug("Added a source file to the archive")
|
|
543
|
+
|
|
544
|
+
logger.info("Exported %s suggestion(s) to a .qdpx archive",
|
|
545
|
+
len(suggestions))
|
|
546
|
+
return str(output_file)
|
|
547
|
+
|
|
548
|
+
except ValueError:
|
|
549
|
+
raise
|
|
550
|
+
except Exception as e:
|
|
551
|
+
logger.error("Failed to export to REFI-QDA: %s", error_label(e))
|
|
552
|
+
raise RuntimeError(
|
|
553
|
+
f"REFI-QDA export failed: {error_text(e)}") from None
|
|
554
|
+
|
|
555
|
+
def validate_suggestions(self, suggestions: List[CodingSuggestion]) -> List[str]:
|
|
556
|
+
"""Validate suggestions before export.
|
|
557
|
+
|
|
558
|
+
Args:
|
|
559
|
+
suggestions: List of suggestions to validate
|
|
560
|
+
|
|
561
|
+
Returns:
|
|
562
|
+
List of validation warnings/errors (empty if all valid)
|
|
563
|
+
"""
|
|
564
|
+
warnings = []
|
|
565
|
+
|
|
566
|
+
# Check if we have any suggestions
|
|
567
|
+
if not suggestions:
|
|
568
|
+
warnings.append("No suggestions to export")
|
|
569
|
+
return warnings
|
|
570
|
+
|
|
571
|
+
# Get all codes and files from database for validation
|
|
572
|
+
try:
|
|
573
|
+
codes = self.db.list_codes()
|
|
574
|
+
code_ids = {c["id"] for c in codes}
|
|
575
|
+
|
|
576
|
+
files = self.db.list_files()
|
|
577
|
+
file_ids = {f["id"] for f in files}
|
|
578
|
+
|
|
579
|
+
except Exception as e:
|
|
580
|
+
warnings.append(f"Could not load project data for validation: "
|
|
581
|
+
f"{error_text(e)}")
|
|
582
|
+
return warnings
|
|
583
|
+
|
|
584
|
+
# File text lengths (also identifies files with no exportable text)
|
|
585
|
+
content_lengths: Dict[int, int] = {}
|
|
586
|
+
|
|
587
|
+
# Validate each suggestion
|
|
588
|
+
for i, suggestion in enumerate(suggestions):
|
|
589
|
+
# Check code exists
|
|
590
|
+
if suggestion.code_id not in code_ids:
|
|
591
|
+
warnings.append(
|
|
592
|
+
f"Suggestion {i}: Code ID {suggestion.code_id} not found in project"
|
|
593
|
+
)
|
|
594
|
+
|
|
595
|
+
# Check file exists
|
|
596
|
+
if suggestion.file_id not in file_ids:
|
|
597
|
+
warnings.append(
|
|
598
|
+
f"Suggestion {i}: File ID {suggestion.file_id} not found in project"
|
|
599
|
+
)
|
|
600
|
+
else:
|
|
601
|
+
if suggestion.file_id not in content_lengths:
|
|
602
|
+
fc = self.db.get_file_content(suggestion.file_id)
|
|
603
|
+
content = (fc or {}).get("content") or ""
|
|
604
|
+
content_lengths[suggestion.file_id] = len(content)
|
|
605
|
+
if content and len(content.encode("utf-8")) > _MAX_INTERNAL_FILE_BYTES:
|
|
606
|
+
warnings.append(
|
|
607
|
+
f"File {suggestion.file_id} exceeds the REFI-QDA "
|
|
608
|
+
f"2 GiB internal file limit"
|
|
609
|
+
)
|
|
610
|
+
length = content_lengths[suggestion.file_id]
|
|
611
|
+
if length == 0:
|
|
612
|
+
warnings.append(
|
|
613
|
+
f"Suggestion {i}: File ID {suggestion.file_id} has no "
|
|
614
|
+
f"text content; selections cannot be exported for it"
|
|
615
|
+
)
|
|
616
|
+
elif isinstance(suggestion.end_pos, int) and suggestion.end_pos > length:
|
|
617
|
+
warnings.append(
|
|
618
|
+
f"Suggestion {i}: End position {suggestion.end_pos} is "
|
|
619
|
+
f"beyond the file text (length {length})"
|
|
620
|
+
)
|
|
621
|
+
|
|
622
|
+
# Check positions are valid. Damaged projects can carry rows with
|
|
623
|
+
# NULL positions (QA2-6) — report them, never TypeError on them.
|
|
624
|
+
if (not isinstance(suggestion.start_pos, int)
|
|
625
|
+
or isinstance(suggestion.start_pos, bool)
|
|
626
|
+
or not isinstance(suggestion.end_pos, int)
|
|
627
|
+
or isinstance(suggestion.end_pos, bool)):
|
|
628
|
+
warnings.append(
|
|
629
|
+
f"Suggestion {i}: missing or non-integer positions "
|
|
630
|
+
f"(start={suggestion.start_pos!r}, end={suggestion.end_pos!r}) "
|
|
631
|
+
f"; the coding row may be damaged"
|
|
632
|
+
)
|
|
633
|
+
continue
|
|
634
|
+
|
|
635
|
+
if suggestion.start_pos < 0:
|
|
636
|
+
warnings.append(
|
|
637
|
+
f"Suggestion {i}: Invalid start position {suggestion.start_pos}"
|
|
638
|
+
)
|
|
639
|
+
|
|
640
|
+
if suggestion.end_pos <= suggestion.start_pos:
|
|
641
|
+
warnings.append(
|
|
642
|
+
f"Suggestion {i}: End position {suggestion.end_pos} must be greater than start position {suggestion.start_pos}"
|
|
643
|
+
)
|
|
644
|
+
|
|
645
|
+
return warnings
|