openansho 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
openansho/__init__.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"
openansho/__main__.py ADDED
@@ -0,0 +1,39 @@
1
+ import sys
2
+ from pathlib import Path
3
+
4
+ from PySide6.QtGui import QIcon
5
+ from PySide6.QtWidgets import QApplication
6
+
7
+ from openansho.ui.main_window import MainWindow
8
+
9
+
10
+ ICON_FILENAME = "kanji_shou_app_icon.png"
11
+
12
+
13
+ def _icon_path() -> Path:
14
+ # Three layouts to cover: PyInstaller extracts bundled data files (see the
15
+ # Makefile's --add-data) under sys._MEIPASS; an installed wheel carries the
16
+ # icon inside the package (see pyproject's force-include); a source checkout
17
+ # has it at the repo root, which is where the Makefile reads it from.
18
+ meipass = getattr(sys, "_MEIPASS", None)
19
+ if meipass:
20
+ return Path(meipass) / "images" / ICON_FILENAME
21
+ packaged = Path(__file__).resolve().parent / ICON_FILENAME
22
+ if packaged.exists():
23
+ return packaged
24
+ return Path(__file__).resolve().parent.parent.parent / "images" / ICON_FILENAME
25
+
26
+
27
+ def main() -> None:
28
+ app = QApplication(sys.argv)
29
+ app.setWindowIcon(QIcon(str(_icon_path())))
30
+ window = MainWindow()
31
+ # Start in the built-in tutorial so the app opens onto something codeable;
32
+ # it's an in-memory project, so it costs the user nothing to abandon.
33
+ window.open_tutorial_project()
34
+ window.show()
35
+ sys.exit(app.exec())
36
+
37
+
38
+ if __name__ == "__main__":
39
+ main()
openansho/db.py ADDED
@@ -0,0 +1,363 @@
1
+ """SQLite-backed data access layer for OpenAnsho projects.
2
+
3
+ A project is a single .sqlite file containing documents, a codebook
4
+ (codes, possibly nested), and the coded segments linking the two.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import sqlite3
10
+ from dataclasses import dataclass
11
+ from datetime import datetime, timezone
12
+ from pathlib import Path
13
+
14
+ SCHEMA = """
15
+ CREATE TABLE IF NOT EXISTS documents (
16
+ id INTEGER PRIMARY KEY,
17
+ name TEXT NOT NULL,
18
+ content TEXT NOT NULL,
19
+ created_at TEXT NOT NULL
20
+ );
21
+
22
+ CREATE TABLE IF NOT EXISTS codes (
23
+ id INTEGER PRIMARY KEY,
24
+ name TEXT NOT NULL,
25
+ parent_id INTEGER REFERENCES codes(id) ON DELETE CASCADE,
26
+ color TEXT,
27
+ color_class TEXT,
28
+ description TEXT,
29
+ created_at TEXT NOT NULL
30
+ );
31
+
32
+ CREATE TABLE IF NOT EXISTS segments (
33
+ id INTEGER PRIMARY KEY,
34
+ document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
35
+ code_id INTEGER NOT NULL REFERENCES codes(id) ON DELETE CASCADE,
36
+ start_offset INTEGER NOT NULL,
37
+ end_offset INTEGER NOT NULL,
38
+ memo TEXT,
39
+ created_by TEXT,
40
+ created_at TEXT NOT NULL
41
+ );
42
+ """
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class Document:
47
+ id: int
48
+ name: str
49
+ content: str
50
+ created_at: str
51
+
52
+
53
+ @dataclass(frozen=True)
54
+ class Code:
55
+ id: int
56
+ name: str
57
+ parent_id: int | None
58
+ color: str | None
59
+ color_class: str | None
60
+ description: str | None
61
+ created_at: str
62
+
63
+
64
+ @dataclass(frozen=True)
65
+ class Segment:
66
+ id: int
67
+ document_id: int
68
+ code_id: int
69
+ start_offset: int
70
+ end_offset: int
71
+ memo: str | None
72
+ created_by: str | None
73
+ created_at: str
74
+
75
+
76
+ def _now() -> str:
77
+ return datetime.now(timezone.utc).isoformat()
78
+
79
+
80
+ def connect(path: str | Path) -> sqlite3.Connection:
81
+ conn = sqlite3.connect(path)
82
+ conn.row_factory = sqlite3.Row
83
+ conn.execute("PRAGMA foreign_keys = ON")
84
+ init_db(conn)
85
+ return conn
86
+
87
+
88
+ def init_db(conn: sqlite3.Connection) -> None:
89
+ conn.executescript(SCHEMA)
90
+ _migrate(conn)
91
+ conn.commit()
92
+
93
+
94
+ def _migrate(conn: sqlite3.Connection) -> None:
95
+ """Add columns introduced after a project's initial creation."""
96
+ columns = {row["name"] for row in conn.execute("PRAGMA table_info(segments)")}
97
+ if "created_by" not in columns:
98
+ conn.execute("ALTER TABLE segments ADD COLUMN created_by TEXT")
99
+
100
+ code_columns = {row["name"] for row in conn.execute("PRAGMA table_info(codes)")}
101
+ if "color_class" not in code_columns:
102
+ conn.execute("ALTER TABLE codes ADD COLUMN color_class TEXT")
103
+ if "description" not in code_columns:
104
+ conn.execute("ALTER TABLE codes ADD COLUMN description TEXT")
105
+
106
+
107
+ def create_document(conn: sqlite3.Connection, name: str, content: str) -> Document:
108
+ cur = conn.execute(
109
+ "INSERT INTO documents (name, content, created_at) VALUES (?, ?, ?)",
110
+ (name, content, _now()),
111
+ )
112
+ conn.commit()
113
+ return get_document(conn, cur.lastrowid)
114
+
115
+
116
+ def get_document(conn: sqlite3.Connection, document_id: int) -> Document | None:
117
+ row = conn.execute(
118
+ "SELECT * FROM documents WHERE id = ?", (document_id,)
119
+ ).fetchone()
120
+ return Document(**row) if row else None
121
+
122
+
123
+ def list_documents(conn: sqlite3.Connection) -> list[Document]:
124
+ rows = conn.execute("SELECT * FROM documents ORDER BY id").fetchall()
125
+ return [Document(**row) for row in rows]
126
+
127
+
128
+ def get_document_by_name(conn: sqlite3.Connection, name: str) -> Document | None:
129
+ row = conn.execute(
130
+ "SELECT * FROM documents WHERE name = ?", (name,)
131
+ ).fetchone()
132
+ return Document(**row) if row else None
133
+
134
+
135
+ def delete_document(conn: sqlite3.Connection, document_id: int) -> None:
136
+ conn.execute("DELETE FROM documents WHERE id = ?", (document_id,))
137
+ conn.commit()
138
+
139
+
140
+ def update_document_content(
141
+ conn: sqlite3.Connection, document_id: int, content: str
142
+ ) -> Document:
143
+ conn.execute(
144
+ "UPDATE documents SET content = ? WHERE id = ?", (content, document_id)
145
+ )
146
+ conn.commit()
147
+ return get_document(conn, document_id)
148
+
149
+
150
+ def create_code(
151
+ conn: sqlite3.Connection,
152
+ name: str,
153
+ parent_id: int | None = None,
154
+ color: str | None = None,
155
+ color_class: str | None = None,
156
+ description: str | None = None,
157
+ ) -> Code:
158
+ cur = conn.execute(
159
+ "INSERT INTO codes (name, parent_id, color, color_class, description, created_at) "
160
+ "VALUES (?, ?, ?, ?, ?, ?)",
161
+ (name, parent_id, color, color_class, description, _now()),
162
+ )
163
+ conn.commit()
164
+ return get_code(conn, cur.lastrowid)
165
+
166
+
167
+ def get_code(conn: sqlite3.Connection, code_id: int) -> Code | None:
168
+ row = conn.execute("SELECT * FROM codes WHERE id = ?", (code_id,)).fetchone()
169
+ return Code(**row) if row else None
170
+
171
+
172
+ def list_codes(conn: sqlite3.Connection) -> list[Code]:
173
+ rows = conn.execute("SELECT * FROM codes ORDER BY id").fetchall()
174
+ return [Code(**row) for row in rows]
175
+
176
+
177
+ def rename_code(conn: sqlite3.Connection, code_id: int, name: str) -> Code:
178
+ conn.execute("UPDATE codes SET name = ? WHERE id = ?", (name, code_id))
179
+ conn.commit()
180
+ return get_code(conn, code_id)
181
+
182
+
183
+ def set_code_description(conn: sqlite3.Connection, code_id: int, description: str | None) -> Code:
184
+ conn.execute("UPDATE codes SET description = ? WHERE id = ?", (description, code_id))
185
+ conn.commit()
186
+ return get_code(conn, code_id)
187
+
188
+
189
+ def set_code_parent(conn: sqlite3.Connection, code_id: int, parent_id: int | None) -> Code:
190
+ conn.execute("UPDATE codes SET parent_id = ? WHERE id = ?", (parent_id, code_id))
191
+ conn.commit()
192
+ return get_code(conn, code_id)
193
+
194
+
195
+ def set_code_color(
196
+ conn: sqlite3.Connection, code_id: int, color: str, color_class: str
197
+ ) -> Code:
198
+ conn.execute(
199
+ "UPDATE codes SET color = ?, color_class = ? WHERE id = ?",
200
+ (color, color_class, code_id),
201
+ )
202
+ conn.commit()
203
+ return get_code(conn, code_id)
204
+
205
+
206
+ def count_codes_by_color_class(conn: sqlite3.Connection) -> dict[str, int]:
207
+ """Map color_class -> number of codes (root or child) assigned to it."""
208
+ rows = conn.execute(
209
+ "SELECT color_class, COUNT(*) AS count FROM codes "
210
+ "WHERE color_class IS NOT NULL GROUP BY color_class"
211
+ ).fetchall()
212
+ return {row["color_class"]: row["count"] for row in rows}
213
+
214
+
215
+ def delete_code(conn: sqlite3.Connection, code_id: int) -> None:
216
+ """Delete a code, re-parenting its children to its own parent (or to root).
217
+
218
+ `codes.parent_id` cascades on delete, so children would otherwise be
219
+ deleted along with their parent; re-pointing them first avoids that.
220
+ """
221
+ code = get_code(conn, code_id)
222
+ if code is None:
223
+ return
224
+ conn.execute(
225
+ "UPDATE codes SET parent_id = ? WHERE parent_id = ?", (code.parent_id, code_id)
226
+ )
227
+ conn.execute("DELETE FROM codes WHERE id = ?", (code_id,))
228
+ conn.commit()
229
+
230
+
231
+ def merge_codes(conn: sqlite3.Connection, keep_id: int, merge_id: int) -> None:
232
+ """Merge `merge_id` into `keep_id`.
233
+
234
+ Segments coded with `merge_id` are re-coded to `keep_id`, dropping any
235
+ that would exactly duplicate a segment `keep_id` already has. Children of
236
+ `merge_id` are re-parented to `keep_id` before it's deleted, the same
237
+ cascade-dodging move `delete_code` makes for its own children.
238
+ """
239
+ keep_segments = {
240
+ (s.document_id, s.start_offset, s.end_offset, s.created_by)
241
+ for s in list_segments_for_code(conn, keep_id)
242
+ }
243
+ for segment in list_segments_for_code(conn, merge_id):
244
+ key = (segment.document_id, segment.start_offset, segment.end_offset, segment.created_by)
245
+ if key in keep_segments:
246
+ conn.execute("DELETE FROM segments WHERE id = ?", (segment.id,))
247
+ else:
248
+ conn.execute(
249
+ "UPDATE segments SET code_id = ? WHERE id = ?", (keep_id, segment.id)
250
+ )
251
+ keep_segments.add(key)
252
+ conn.execute(
253
+ "UPDATE codes SET parent_id = ? WHERE parent_id = ?", (keep_id, merge_id)
254
+ )
255
+ conn.execute("DELETE FROM codes WHERE id = ?", (merge_id,))
256
+ conn.commit()
257
+
258
+
259
+ def create_segment(
260
+ conn: sqlite3.Connection,
261
+ document_id: int,
262
+ code_id: int,
263
+ start_offset: int,
264
+ end_offset: int,
265
+ memo: str | None = None,
266
+ created_by: str | None = None,
267
+ ) -> Segment:
268
+ if end_offset <= start_offset:
269
+ raise ValueError("end_offset must be greater than start_offset")
270
+ existing = conn.execute(
271
+ """
272
+ SELECT * FROM segments
273
+ WHERE document_id = ? AND code_id = ? AND start_offset = ? AND end_offset = ?
274
+ AND created_by IS ?
275
+ """,
276
+ (document_id, code_id, start_offset, end_offset, created_by),
277
+ ).fetchone()
278
+ if existing is not None:
279
+ return Segment(**existing)
280
+ cur = conn.execute(
281
+ """
282
+ INSERT INTO segments
283
+ (document_id, code_id, start_offset, end_offset, memo, created_by, created_at)
284
+ VALUES (?, ?, ?, ?, ?, ?, ?)
285
+ """,
286
+ (document_id, code_id, start_offset, end_offset, memo, created_by, _now()),
287
+ )
288
+ conn.commit()
289
+ return get_segment(conn, cur.lastrowid)
290
+
291
+
292
+ def get_segment(conn: sqlite3.Connection, segment_id: int) -> Segment | None:
293
+ row = conn.execute(
294
+ "SELECT * FROM segments WHERE id = ?", (segment_id,)
295
+ ).fetchone()
296
+ return Segment(**row) if row else None
297
+
298
+
299
+ def list_segments_for_document(
300
+ conn: sqlite3.Connection, document_id: int
301
+ ) -> list[Segment]:
302
+ rows = conn.execute(
303
+ "SELECT * FROM segments WHERE document_id = ? ORDER BY start_offset",
304
+ (document_id,),
305
+ ).fetchall()
306
+ return [Segment(**row) for row in rows]
307
+
308
+
309
+ def list_segments_for_code(conn: sqlite3.Connection, code_id: int) -> list[Segment]:
310
+ rows = conn.execute(
311
+ "SELECT * FROM segments WHERE code_id = ? ORDER BY id", (code_id,)
312
+ ).fetchall()
313
+ return [Segment(**row) for row in rows]
314
+
315
+
316
+ def list_all_segments(conn: sqlite3.Connection) -> list[Segment]:
317
+ rows = conn.execute("SELECT * FROM segments ORDER BY id").fetchall()
318
+ return [Segment(**row) for row in rows]
319
+
320
+
321
+ def list_distinct_usernames(conn: sqlite3.Connection) -> list[str | None]:
322
+ """Distinct `segments.created_by` values with at least one segment.
323
+
324
+ `None` is included if any segment has no recorded creator.
325
+ """
326
+ rows = conn.execute("SELECT DISTINCT created_by FROM segments").fetchall()
327
+ return [row["created_by"] for row in rows]
328
+
329
+
330
+ def count_segments_by_code(
331
+ conn: sqlite3.Connection, document_id: int | None = None
332
+ ) -> dict[int, int]:
333
+ """Map code_id -> number of segments coded with it.
334
+
335
+ Codes with no segments are omitted, so callers should default to 0.
336
+ """
337
+ if document_id is None:
338
+ rows = conn.execute(
339
+ "SELECT code_id, COUNT(*) AS count FROM segments GROUP BY code_id"
340
+ ).fetchall()
341
+ else:
342
+ rows = conn.execute(
343
+ "SELECT code_id, COUNT(*) AS count FROM segments WHERE document_id = ? "
344
+ "GROUP BY code_id",
345
+ (document_id,),
346
+ ).fetchall()
347
+ return {row["code_id"]: row["count"] for row in rows}
348
+
349
+
350
+ def delete_segment(conn: sqlite3.Connection, segment_id: int) -> None:
351
+ conn.execute("DELETE FROM segments WHERE id = ?", (segment_id,))
352
+ conn.commit()
353
+
354
+
355
+ def update_segment_offsets(
356
+ conn: sqlite3.Connection, segment_id: int, start_offset: int, end_offset: int
357
+ ) -> Segment:
358
+ conn.execute(
359
+ "UPDATE segments SET start_offset = ?, end_offset = ? WHERE id = ?",
360
+ (start_offset, end_offset, segment_id),
361
+ )
362
+ conn.commit()
363
+ return get_segment(conn, segment_id)
Binary file
openansho/reporting.py ADDED
@@ -0,0 +1,153 @@
1
+ """Export and aggregate reporting over a project's coded segments.
2
+
3
+ These functions operate directly on a sqlite3.Connection so they can
4
+ be used from the UI, a future CLI, or tests without any Qt dependency.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import csv
10
+ import json
11
+ import sqlite3
12
+ from pathlib import Path
13
+
14
+ from openansho import db
15
+ from openansho.db import Code
16
+
17
+ CSV_FIELDNAMES = [
18
+ "document",
19
+ "code",
20
+ "start_offset",
21
+ "end_offset",
22
+ "text",
23
+ "memo",
24
+ "username",
25
+ "created_at",
26
+ ]
27
+
28
+
29
+ def code_path(codes_by_id: dict[int, Code], code_id: int) -> str:
30
+ parts = []
31
+ current = codes_by_id.get(code_id)
32
+ while current is not None:
33
+ parts.append(current.name)
34
+ current = codes_by_id.get(current.parent_id) if current.parent_id else None
35
+ return " > ".join(reversed(parts))
36
+
37
+
38
+ def _segment_rows(conn: sqlite3.Connection) -> list[dict]:
39
+ codes_by_id = {code.id: code for code in db.list_codes(conn)}
40
+ documents = db.list_documents(conn)
41
+
42
+ rows = []
43
+ for document in documents:
44
+ for segment in db.list_segments_for_document(conn, document.id):
45
+ rows.append(
46
+ {
47
+ "document": document.name,
48
+ "code": code_path(codes_by_id, segment.code_id),
49
+ "start_offset": segment.start_offset,
50
+ "end_offset": segment.end_offset,
51
+ "text": document.content[segment.start_offset : segment.end_offset],
52
+ "memo": segment.memo or "",
53
+ "username": segment.created_by or "",
54
+ "created_at": segment.created_at,
55
+ }
56
+ )
57
+ rows.sort(key=lambda row: (row["document"], row["start_offset"]))
58
+ return rows
59
+
60
+
61
+ def export_segments_csv(conn: sqlite3.Connection, path: str | Path) -> int:
62
+ rows = _segment_rows(conn)
63
+ with open(path, "w", newline="", encoding="utf-8") as f:
64
+ writer = csv.DictWriter(f, fieldnames=CSV_FIELDNAMES)
65
+ writer.writeheader()
66
+ writer.writerows(rows)
67
+ return len(rows)
68
+
69
+
70
+ def export_segments_json(conn: sqlite3.Connection, path: str | Path) -> int:
71
+ rows = _segment_rows(conn)
72
+ with open(path, "w", encoding="utf-8") as f:
73
+ json.dump(rows, f, indent=2)
74
+ return len(rows)
75
+
76
+
77
+ def code_frequency(conn: sqlite3.Connection) -> list[dict]:
78
+ codes_by_id = {code.id: code for code in db.list_codes(conn)}
79
+ rows = [
80
+ {
81
+ "code_id": code_id,
82
+ "path": code_path(codes_by_id, code_id),
83
+ "count": len(db.list_segments_for_code(conn, code_id)),
84
+ }
85
+ for code_id in codes_by_id
86
+ ]
87
+ rows.sort(key=lambda row: (-row["count"], row["path"]))
88
+ return rows
89
+
90
+
91
+ def code_user_frequency(conn: sqlite3.Connection) -> list[dict]:
92
+ """One row per (code, user) pair: the code, its parent, the username, and
93
+ how many segments that user has coded with that code across the dataset."""
94
+ codes_by_id = {code.id: code for code in db.list_codes(conn)}
95
+
96
+ counts: dict[tuple[int, str], int] = {}
97
+ for code_id in codes_by_id:
98
+ for segment in db.list_segments_for_code(conn, code_id):
99
+ key = (code_id, segment.created_by or "")
100
+ counts[key] = counts.get(key, 0) + 1
101
+
102
+ rows = []
103
+ for (code_id, username), count in counts.items():
104
+ code = codes_by_id[code_id]
105
+ parent = codes_by_id.get(code.parent_id) if code.parent_id else None
106
+ rows.append(
107
+ {
108
+ "code": code.name,
109
+ "parent": parent.name if parent else "",
110
+ "username": username,
111
+ "count": count,
112
+ }
113
+ )
114
+ rows.sort(key=lambda row: (row["code"], row["username"]))
115
+ return rows
116
+
117
+
118
+ CODE_FREQUENCY_CSV_FIELDNAMES = ["code", "count"]
119
+ CODE_USER_FREQUENCY_CSV_FIELDNAMES = ["code", "parent", "username", "count"]
120
+
121
+
122
+ def export_code_frequency_csv(conn: sqlite3.Connection, path: str | Path) -> int:
123
+ rows = code_frequency(conn)
124
+ with open(path, "w", newline="", encoding="utf-8") as f:
125
+ writer = csv.DictWriter(f, fieldnames=CODE_FREQUENCY_CSV_FIELDNAMES)
126
+ writer.writeheader()
127
+ for row in rows:
128
+ writer.writerow({"code": row["path"], "count": row["count"]})
129
+ return len(rows)
130
+
131
+
132
+ def export_code_frequency_json(conn: sqlite3.Connection, path: str | Path) -> int:
133
+ rows = code_frequency(conn)
134
+ export_rows = [{"code": row["path"], "count": row["count"]} for row in rows]
135
+ with open(path, "w", encoding="utf-8") as f:
136
+ json.dump(export_rows, f, indent=2)
137
+ return len(rows)
138
+
139
+
140
+ def export_code_user_frequency_csv(conn: sqlite3.Connection, path: str | Path) -> int:
141
+ rows = code_user_frequency(conn)
142
+ with open(path, "w", newline="", encoding="utf-8") as f:
143
+ writer = csv.DictWriter(f, fieldnames=CODE_USER_FREQUENCY_CSV_FIELDNAMES)
144
+ writer.writeheader()
145
+ writer.writerows(rows)
146
+ return len(rows)
147
+
148
+
149
+ def export_code_user_frequency_json(conn: sqlite3.Connection, path: str | Path) -> int:
150
+ rows = code_user_frequency(conn)
151
+ with open(path, "w", encoding="utf-8") as f:
152
+ json.dump(rows, f, indent=2)
153
+ return len(rows)
@@ -0,0 +1,97 @@
1
+ """Turn an on-disk document into the plain text OpenAnsho codes against.
2
+
3
+ Like `db.py` and `reporting.py`, this module has no Qt imports: the UI hands it
4
+ a path and gets back a string (or a `DocumentReadError` carrying a message fit
5
+ for a dialog).
6
+
7
+ The `.docx` reader is hand-rolled on top of `zipfile`/`xml.etree` rather than
8
+ `python-docx` so the packaged app keeps its single third-party dependency.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import xml.etree.ElementTree as ET
14
+ import zipfile
15
+ from pathlib import Path
16
+
17
+ W_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
18
+ DOCX_BODY_PART = "word/document.xml"
19
+
20
+
21
+ class DocumentReadError(Exception):
22
+ """A document couldn't be read; the message is shown to the user as-is."""
23
+
24
+
25
+ def read_document_text(path: Path) -> str:
26
+ """Read `path` as plain text, dispatching on its extension."""
27
+ suffix = path.suffix.lower()
28
+ if suffix == ".docx":
29
+ return read_docx_text(path)
30
+ if suffix == ".doc":
31
+ raise DocumentReadError(
32
+ f"{path.name} is a legacy Word document. Open it in Word and save it "
33
+ "as .docx (or plain text), then import that."
34
+ )
35
+ try:
36
+ return path.read_text(encoding="utf-8")
37
+ except UnicodeDecodeError:
38
+ raise DocumentReadError(f"Could not read {path.name} as UTF-8 text.") from None
39
+ except OSError as exc:
40
+ raise DocumentReadError(f"Could not read {path.name}: {exc}") from None
41
+
42
+
43
+ def read_docx_text(path: Path) -> str:
44
+ """Extract the body text of a `.docx` file, one line per paragraph.
45
+
46
+ Paragraphs are visited in document order, including those nested inside
47
+ tables (each table cell's paragraphs become their own lines).
48
+ """
49
+ try:
50
+ with zipfile.ZipFile(path) as archive:
51
+ try:
52
+ body_xml = archive.read(DOCX_BODY_PART)
53
+ except KeyError:
54
+ raise DocumentReadError(
55
+ f"{path.name} is not a valid Word document "
56
+ f"(it has no {DOCX_BODY_PART})."
57
+ ) from None
58
+ except zipfile.BadZipFile:
59
+ raise DocumentReadError(
60
+ f"{path.name} is not a valid Word document (it is not a .docx archive)."
61
+ ) from None
62
+ except OSError as exc:
63
+ raise DocumentReadError(f"Could not read {path.name}: {exc}") from None
64
+
65
+ try:
66
+ root = ET.fromstring(body_xml)
67
+ except ET.ParseError as exc:
68
+ raise DocumentReadError(f"Could not parse {path.name}: {exc}") from None
69
+
70
+ body = root.find(_w("body"))
71
+ if body is None:
72
+ body = root
73
+ paragraphs = [_paragraph_text(p) for p in body.iter(_w("p"))]
74
+ return "\n".join(paragraphs)
75
+
76
+
77
+ def _w(tag: str) -> str:
78
+ return f"{{{W_NAMESPACE}}}{tag}"
79
+
80
+
81
+ def _paragraph_text(paragraph: ET.Element) -> str:
82
+ """Flatten one `w:p` into a line of text.
83
+
84
+ Only `w:t` runs contribute characters, which conveniently skips field
85
+ instructions (`w:instrText`) and tracked deletions (`w:delText`); tabs and
86
+ in-paragraph breaks are preserved as whitespace.
87
+ """
88
+ pieces: list[str] = []
89
+ for node in paragraph.iter():
90
+ tag = node.tag
91
+ if tag == _w("t"):
92
+ pieces.append(node.text or "")
93
+ elif tag == _w("tab"):
94
+ pieces.append("\t")
95
+ elif tag in (_w("br"), _w("cr")):
96
+ pieces.append("\n")
97
+ return "".join(pieces)
openansho/tutorial.py ADDED
@@ -0,0 +1,41 @@
1
+ """The built-in tutorial project.
2
+
3
+ The app opens this project at startup so a first-time user has something to
4
+ code without importing anything first. Its database lives in memory rather
5
+ than in a .sqlite file, so whatever codebook the user builds while following
6
+ along is discarded on quit and the tutorial is fresh again on the next launch.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import sqlite3
12
+ import sys
13
+ from pathlib import Path
14
+
15
+ from openansho import db
16
+
17
+ DOCUMENT_NAME = "tutorial.txt"
18
+
19
+
20
+ def tutorial_text_path() -> Path:
21
+ """Locate the tutorial text, bundled or running from source.
22
+
23
+ PyInstaller extracts the bundled copy (see the Makefile's --add-data and
24
+ OpenAnsho.spec) under sys._MEIPASS at runtime; fall back to the file
25
+ sitting next to this module when running from a source checkout.
26
+ """
27
+ meipass = getattr(sys, "_MEIPASS", None)
28
+ if meipass:
29
+ return Path(meipass) / "openansho" / DOCUMENT_NAME
30
+ return Path(__file__).resolve().with_name(DOCUMENT_NAME)
31
+
32
+
33
+ def read_tutorial_text() -> str:
34
+ return tutorial_text_path().read_text(encoding="utf-8")
35
+
36
+
37
+ def open_tutorial_project() -> sqlite3.Connection:
38
+ """Open a throwaway in-memory project holding just the tutorial document."""
39
+ conn = db.connect(":memory:")
40
+ db.create_document(conn, DOCUMENT_NAME, read_tutorial_text())
41
+ return conn