openansho 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- openansho/__init__.py +1 -0
- openansho/__main__.py +39 -0
- openansho/db.py +363 -0
- openansho/kanji_shou_app_icon.png +0 -0
- openansho/reporting.py +153 -0
- openansho/text_extract.py +97 -0
- openansho/tutorial.py +41 -0
- openansho/tutorial.txt +31 -0
- openansho/ui/__init__.py +0 -0
- openansho/ui/checkable_combo_box.py +121 -0
- openansho/ui/code_filter_input.py +19 -0
- openansho/ui/code_tree.py +30 -0
- openansho/ui/font_scale.py +75 -0
- openansho/ui/main_window.py +2082 -0
- openansho/ui/merge_codes_dialog.py +69 -0
- openansho/ui/os_theme.py +59 -0
- openansho/ui/preferences_dialog.py +57 -0
- openansho/ui/report_dialog.py +45 -0
- openansho/ui/shortcuts_dialog.py +97 -0
- openansho/ui/vim_viewer.py +858 -0
- openansho/user.py +86 -0
- openansho-0.1.0.dist-info/METADATA +140 -0
- openansho-0.1.0.dist-info/RECORD +26 -0
- openansho-0.1.0.dist-info/WHEEL +4 -0
- openansho-0.1.0.dist-info/entry_points.txt +2 -0
- openansho-0.1.0.dist-info/licenses/LICENSE +21 -0
openansho/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
openansho/__main__.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
import sys
|
|
2
|
+
from pathlib import Path
|
|
3
|
+
|
|
4
|
+
from PySide6.QtGui import QIcon
|
|
5
|
+
from PySide6.QtWidgets import QApplication
|
|
6
|
+
|
|
7
|
+
from openansho.ui.main_window import MainWindow
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
ICON_FILENAME = "kanji_shou_app_icon.png"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _icon_path() -> Path:
|
|
14
|
+
# Three layouts to cover: PyInstaller extracts bundled data files (see the
|
|
15
|
+
# Makefile's --add-data) under sys._MEIPASS; an installed wheel carries the
|
|
16
|
+
# icon inside the package (see pyproject's force-include); a source checkout
|
|
17
|
+
# has it at the repo root, which is where the Makefile reads it from.
|
|
18
|
+
meipass = getattr(sys, "_MEIPASS", None)
|
|
19
|
+
if meipass:
|
|
20
|
+
return Path(meipass) / "images" / ICON_FILENAME
|
|
21
|
+
packaged = Path(__file__).resolve().parent / ICON_FILENAME
|
|
22
|
+
if packaged.exists():
|
|
23
|
+
return packaged
|
|
24
|
+
return Path(__file__).resolve().parent.parent.parent / "images" / ICON_FILENAME
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def main() -> None:
|
|
28
|
+
app = QApplication(sys.argv)
|
|
29
|
+
app.setWindowIcon(QIcon(str(_icon_path())))
|
|
30
|
+
window = MainWindow()
|
|
31
|
+
# Start in the built-in tutorial so the app opens onto something codeable;
|
|
32
|
+
# it's an in-memory project, so it costs the user nothing to abandon.
|
|
33
|
+
window.open_tutorial_project()
|
|
34
|
+
window.show()
|
|
35
|
+
sys.exit(app.exec())
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
if __name__ == "__main__":
|
|
39
|
+
main()
|
openansho/db.py
ADDED
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
"""SQLite-backed data access layer for OpenAnsho projects.
|
|
2
|
+
|
|
3
|
+
A project is a single .sqlite file containing documents, a codebook
|
|
4
|
+
(codes, possibly nested), and the coded segments linking the two.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import sqlite3
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from datetime import datetime, timezone
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
SCHEMA = """
|
|
15
|
+
CREATE TABLE IF NOT EXISTS documents (
|
|
16
|
+
id INTEGER PRIMARY KEY,
|
|
17
|
+
name TEXT NOT NULL,
|
|
18
|
+
content TEXT NOT NULL,
|
|
19
|
+
created_at TEXT NOT NULL
|
|
20
|
+
);
|
|
21
|
+
|
|
22
|
+
CREATE TABLE IF NOT EXISTS codes (
|
|
23
|
+
id INTEGER PRIMARY KEY,
|
|
24
|
+
name TEXT NOT NULL,
|
|
25
|
+
parent_id INTEGER REFERENCES codes(id) ON DELETE CASCADE,
|
|
26
|
+
color TEXT,
|
|
27
|
+
color_class TEXT,
|
|
28
|
+
description TEXT,
|
|
29
|
+
created_at TEXT NOT NULL
|
|
30
|
+
);
|
|
31
|
+
|
|
32
|
+
CREATE TABLE IF NOT EXISTS segments (
|
|
33
|
+
id INTEGER PRIMARY KEY,
|
|
34
|
+
document_id INTEGER NOT NULL REFERENCES documents(id) ON DELETE CASCADE,
|
|
35
|
+
code_id INTEGER NOT NULL REFERENCES codes(id) ON DELETE CASCADE,
|
|
36
|
+
start_offset INTEGER NOT NULL,
|
|
37
|
+
end_offset INTEGER NOT NULL,
|
|
38
|
+
memo TEXT,
|
|
39
|
+
created_by TEXT,
|
|
40
|
+
created_at TEXT NOT NULL
|
|
41
|
+
);
|
|
42
|
+
"""
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class Document:
|
|
47
|
+
id: int
|
|
48
|
+
name: str
|
|
49
|
+
content: str
|
|
50
|
+
created_at: str
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
@dataclass(frozen=True)
|
|
54
|
+
class Code:
|
|
55
|
+
id: int
|
|
56
|
+
name: str
|
|
57
|
+
parent_id: int | None
|
|
58
|
+
color: str | None
|
|
59
|
+
color_class: str | None
|
|
60
|
+
description: str | None
|
|
61
|
+
created_at: str
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True)
|
|
65
|
+
class Segment:
|
|
66
|
+
id: int
|
|
67
|
+
document_id: int
|
|
68
|
+
code_id: int
|
|
69
|
+
start_offset: int
|
|
70
|
+
end_offset: int
|
|
71
|
+
memo: str | None
|
|
72
|
+
created_by: str | None
|
|
73
|
+
created_at: str
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _now() -> str:
|
|
77
|
+
return datetime.now(timezone.utc).isoformat()
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def connect(path: str | Path) -> sqlite3.Connection:
|
|
81
|
+
conn = sqlite3.connect(path)
|
|
82
|
+
conn.row_factory = sqlite3.Row
|
|
83
|
+
conn.execute("PRAGMA foreign_keys = ON")
|
|
84
|
+
init_db(conn)
|
|
85
|
+
return conn
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def init_db(conn: sqlite3.Connection) -> None:
|
|
89
|
+
conn.executescript(SCHEMA)
|
|
90
|
+
_migrate(conn)
|
|
91
|
+
conn.commit()
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _migrate(conn: sqlite3.Connection) -> None:
|
|
95
|
+
"""Add columns introduced after a project's initial creation."""
|
|
96
|
+
columns = {row["name"] for row in conn.execute("PRAGMA table_info(segments)")}
|
|
97
|
+
if "created_by" not in columns:
|
|
98
|
+
conn.execute("ALTER TABLE segments ADD COLUMN created_by TEXT")
|
|
99
|
+
|
|
100
|
+
code_columns = {row["name"] for row in conn.execute("PRAGMA table_info(codes)")}
|
|
101
|
+
if "color_class" not in code_columns:
|
|
102
|
+
conn.execute("ALTER TABLE codes ADD COLUMN color_class TEXT")
|
|
103
|
+
if "description" not in code_columns:
|
|
104
|
+
conn.execute("ALTER TABLE codes ADD COLUMN description TEXT")
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def create_document(conn: sqlite3.Connection, name: str, content: str) -> Document:
|
|
108
|
+
cur = conn.execute(
|
|
109
|
+
"INSERT INTO documents (name, content, created_at) VALUES (?, ?, ?)",
|
|
110
|
+
(name, content, _now()),
|
|
111
|
+
)
|
|
112
|
+
conn.commit()
|
|
113
|
+
return get_document(conn, cur.lastrowid)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def get_document(conn: sqlite3.Connection, document_id: int) -> Document | None:
|
|
117
|
+
row = conn.execute(
|
|
118
|
+
"SELECT * FROM documents WHERE id = ?", (document_id,)
|
|
119
|
+
).fetchone()
|
|
120
|
+
return Document(**row) if row else None
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def list_documents(conn: sqlite3.Connection) -> list[Document]:
|
|
124
|
+
rows = conn.execute("SELECT * FROM documents ORDER BY id").fetchall()
|
|
125
|
+
return [Document(**row) for row in rows]
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def get_document_by_name(conn: sqlite3.Connection, name: str) -> Document | None:
|
|
129
|
+
row = conn.execute(
|
|
130
|
+
"SELECT * FROM documents WHERE name = ?", (name,)
|
|
131
|
+
).fetchone()
|
|
132
|
+
return Document(**row) if row else None
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def delete_document(conn: sqlite3.Connection, document_id: int) -> None:
|
|
136
|
+
conn.execute("DELETE FROM documents WHERE id = ?", (document_id,))
|
|
137
|
+
conn.commit()
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def update_document_content(
|
|
141
|
+
conn: sqlite3.Connection, document_id: int, content: str
|
|
142
|
+
) -> Document:
|
|
143
|
+
conn.execute(
|
|
144
|
+
"UPDATE documents SET content = ? WHERE id = ?", (content, document_id)
|
|
145
|
+
)
|
|
146
|
+
conn.commit()
|
|
147
|
+
return get_document(conn, document_id)
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def create_code(
|
|
151
|
+
conn: sqlite3.Connection,
|
|
152
|
+
name: str,
|
|
153
|
+
parent_id: int | None = None,
|
|
154
|
+
color: str | None = None,
|
|
155
|
+
color_class: str | None = None,
|
|
156
|
+
description: str | None = None,
|
|
157
|
+
) -> Code:
|
|
158
|
+
cur = conn.execute(
|
|
159
|
+
"INSERT INTO codes (name, parent_id, color, color_class, description, created_at) "
|
|
160
|
+
"VALUES (?, ?, ?, ?, ?, ?)",
|
|
161
|
+
(name, parent_id, color, color_class, description, _now()),
|
|
162
|
+
)
|
|
163
|
+
conn.commit()
|
|
164
|
+
return get_code(conn, cur.lastrowid)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def get_code(conn: sqlite3.Connection, code_id: int) -> Code | None:
|
|
168
|
+
row = conn.execute("SELECT * FROM codes WHERE id = ?", (code_id,)).fetchone()
|
|
169
|
+
return Code(**row) if row else None
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def list_codes(conn: sqlite3.Connection) -> list[Code]:
|
|
173
|
+
rows = conn.execute("SELECT * FROM codes ORDER BY id").fetchall()
|
|
174
|
+
return [Code(**row) for row in rows]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def rename_code(conn: sqlite3.Connection, code_id: int, name: str) -> Code:
|
|
178
|
+
conn.execute("UPDATE codes SET name = ? WHERE id = ?", (name, code_id))
|
|
179
|
+
conn.commit()
|
|
180
|
+
return get_code(conn, code_id)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def set_code_description(conn: sqlite3.Connection, code_id: int, description: str | None) -> Code:
|
|
184
|
+
conn.execute("UPDATE codes SET description = ? WHERE id = ?", (description, code_id))
|
|
185
|
+
conn.commit()
|
|
186
|
+
return get_code(conn, code_id)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def set_code_parent(conn: sqlite3.Connection, code_id: int, parent_id: int | None) -> Code:
|
|
190
|
+
conn.execute("UPDATE codes SET parent_id = ? WHERE id = ?", (parent_id, code_id))
|
|
191
|
+
conn.commit()
|
|
192
|
+
return get_code(conn, code_id)
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def set_code_color(
|
|
196
|
+
conn: sqlite3.Connection, code_id: int, color: str, color_class: str
|
|
197
|
+
) -> Code:
|
|
198
|
+
conn.execute(
|
|
199
|
+
"UPDATE codes SET color = ?, color_class = ? WHERE id = ?",
|
|
200
|
+
(color, color_class, code_id),
|
|
201
|
+
)
|
|
202
|
+
conn.commit()
|
|
203
|
+
return get_code(conn, code_id)
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def count_codes_by_color_class(conn: sqlite3.Connection) -> dict[str, int]:
|
|
207
|
+
"""Map color_class -> number of codes (root or child) assigned to it."""
|
|
208
|
+
rows = conn.execute(
|
|
209
|
+
"SELECT color_class, COUNT(*) AS count FROM codes "
|
|
210
|
+
"WHERE color_class IS NOT NULL GROUP BY color_class"
|
|
211
|
+
).fetchall()
|
|
212
|
+
return {row["color_class"]: row["count"] for row in rows}
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def delete_code(conn: sqlite3.Connection, code_id: int) -> None:
|
|
216
|
+
"""Delete a code, re-parenting its children to its own parent (or to root).
|
|
217
|
+
|
|
218
|
+
`codes.parent_id` cascades on delete, so children would otherwise be
|
|
219
|
+
deleted along with their parent; re-pointing them first avoids that.
|
|
220
|
+
"""
|
|
221
|
+
code = get_code(conn, code_id)
|
|
222
|
+
if code is None:
|
|
223
|
+
return
|
|
224
|
+
conn.execute(
|
|
225
|
+
"UPDATE codes SET parent_id = ? WHERE parent_id = ?", (code.parent_id, code_id)
|
|
226
|
+
)
|
|
227
|
+
conn.execute("DELETE FROM codes WHERE id = ?", (code_id,))
|
|
228
|
+
conn.commit()
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def merge_codes(conn: sqlite3.Connection, keep_id: int, merge_id: int) -> None:
|
|
232
|
+
"""Merge `merge_id` into `keep_id`.
|
|
233
|
+
|
|
234
|
+
Segments coded with `merge_id` are re-coded to `keep_id`, dropping any
|
|
235
|
+
that would exactly duplicate a segment `keep_id` already has. Children of
|
|
236
|
+
`merge_id` are re-parented to `keep_id` before it's deleted, the same
|
|
237
|
+
cascade-dodging move `delete_code` makes for its own children.
|
|
238
|
+
"""
|
|
239
|
+
keep_segments = {
|
|
240
|
+
(s.document_id, s.start_offset, s.end_offset, s.created_by)
|
|
241
|
+
for s in list_segments_for_code(conn, keep_id)
|
|
242
|
+
}
|
|
243
|
+
for segment in list_segments_for_code(conn, merge_id):
|
|
244
|
+
key = (segment.document_id, segment.start_offset, segment.end_offset, segment.created_by)
|
|
245
|
+
if key in keep_segments:
|
|
246
|
+
conn.execute("DELETE FROM segments WHERE id = ?", (segment.id,))
|
|
247
|
+
else:
|
|
248
|
+
conn.execute(
|
|
249
|
+
"UPDATE segments SET code_id = ? WHERE id = ?", (keep_id, segment.id)
|
|
250
|
+
)
|
|
251
|
+
keep_segments.add(key)
|
|
252
|
+
conn.execute(
|
|
253
|
+
"UPDATE codes SET parent_id = ? WHERE parent_id = ?", (keep_id, merge_id)
|
|
254
|
+
)
|
|
255
|
+
conn.execute("DELETE FROM codes WHERE id = ?", (merge_id,))
|
|
256
|
+
conn.commit()
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def create_segment(
|
|
260
|
+
conn: sqlite3.Connection,
|
|
261
|
+
document_id: int,
|
|
262
|
+
code_id: int,
|
|
263
|
+
start_offset: int,
|
|
264
|
+
end_offset: int,
|
|
265
|
+
memo: str | None = None,
|
|
266
|
+
created_by: str | None = None,
|
|
267
|
+
) -> Segment:
|
|
268
|
+
if end_offset <= start_offset:
|
|
269
|
+
raise ValueError("end_offset must be greater than start_offset")
|
|
270
|
+
existing = conn.execute(
|
|
271
|
+
"""
|
|
272
|
+
SELECT * FROM segments
|
|
273
|
+
WHERE document_id = ? AND code_id = ? AND start_offset = ? AND end_offset = ?
|
|
274
|
+
AND created_by IS ?
|
|
275
|
+
""",
|
|
276
|
+
(document_id, code_id, start_offset, end_offset, created_by),
|
|
277
|
+
).fetchone()
|
|
278
|
+
if existing is not None:
|
|
279
|
+
return Segment(**existing)
|
|
280
|
+
cur = conn.execute(
|
|
281
|
+
"""
|
|
282
|
+
INSERT INTO segments
|
|
283
|
+
(document_id, code_id, start_offset, end_offset, memo, created_by, created_at)
|
|
284
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
285
|
+
""",
|
|
286
|
+
(document_id, code_id, start_offset, end_offset, memo, created_by, _now()),
|
|
287
|
+
)
|
|
288
|
+
conn.commit()
|
|
289
|
+
return get_segment(conn, cur.lastrowid)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def get_segment(conn: sqlite3.Connection, segment_id: int) -> Segment | None:
|
|
293
|
+
row = conn.execute(
|
|
294
|
+
"SELECT * FROM segments WHERE id = ?", (segment_id,)
|
|
295
|
+
).fetchone()
|
|
296
|
+
return Segment(**row) if row else None
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def list_segments_for_document(
|
|
300
|
+
conn: sqlite3.Connection, document_id: int
|
|
301
|
+
) -> list[Segment]:
|
|
302
|
+
rows = conn.execute(
|
|
303
|
+
"SELECT * FROM segments WHERE document_id = ? ORDER BY start_offset",
|
|
304
|
+
(document_id,),
|
|
305
|
+
).fetchall()
|
|
306
|
+
return [Segment(**row) for row in rows]
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def list_segments_for_code(conn: sqlite3.Connection, code_id: int) -> list[Segment]:
|
|
310
|
+
rows = conn.execute(
|
|
311
|
+
"SELECT * FROM segments WHERE code_id = ? ORDER BY id", (code_id,)
|
|
312
|
+
).fetchall()
|
|
313
|
+
return [Segment(**row) for row in rows]
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def list_all_segments(conn: sqlite3.Connection) -> list[Segment]:
|
|
317
|
+
rows = conn.execute("SELECT * FROM segments ORDER BY id").fetchall()
|
|
318
|
+
return [Segment(**row) for row in rows]
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def list_distinct_usernames(conn: sqlite3.Connection) -> list[str | None]:
|
|
322
|
+
"""Distinct `segments.created_by` values with at least one segment.
|
|
323
|
+
|
|
324
|
+
`None` is included if any segment has no recorded creator.
|
|
325
|
+
"""
|
|
326
|
+
rows = conn.execute("SELECT DISTINCT created_by FROM segments").fetchall()
|
|
327
|
+
return [row["created_by"] for row in rows]
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def count_segments_by_code(
|
|
331
|
+
conn: sqlite3.Connection, document_id: int | None = None
|
|
332
|
+
) -> dict[int, int]:
|
|
333
|
+
"""Map code_id -> number of segments coded with it.
|
|
334
|
+
|
|
335
|
+
Codes with no segments are omitted, so callers should default to 0.
|
|
336
|
+
"""
|
|
337
|
+
if document_id is None:
|
|
338
|
+
rows = conn.execute(
|
|
339
|
+
"SELECT code_id, COUNT(*) AS count FROM segments GROUP BY code_id"
|
|
340
|
+
).fetchall()
|
|
341
|
+
else:
|
|
342
|
+
rows = conn.execute(
|
|
343
|
+
"SELECT code_id, COUNT(*) AS count FROM segments WHERE document_id = ? "
|
|
344
|
+
"GROUP BY code_id",
|
|
345
|
+
(document_id,),
|
|
346
|
+
).fetchall()
|
|
347
|
+
return {row["code_id"]: row["count"] for row in rows}
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def delete_segment(conn: sqlite3.Connection, segment_id: int) -> None:
|
|
351
|
+
conn.execute("DELETE FROM segments WHERE id = ?", (segment_id,))
|
|
352
|
+
conn.commit()
|
|
353
|
+
|
|
354
|
+
|
|
355
|
+
def update_segment_offsets(
|
|
356
|
+
conn: sqlite3.Connection, segment_id: int, start_offset: int, end_offset: int
|
|
357
|
+
) -> Segment:
|
|
358
|
+
conn.execute(
|
|
359
|
+
"UPDATE segments SET start_offset = ?, end_offset = ? WHERE id = ?",
|
|
360
|
+
(start_offset, end_offset, segment_id),
|
|
361
|
+
)
|
|
362
|
+
conn.commit()
|
|
363
|
+
return get_segment(conn, segment_id)
|
|
Binary file
|
openansho/reporting.py
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
"""Export and aggregate reporting over a project's coded segments.
|
|
2
|
+
|
|
3
|
+
These functions operate directly on a sqlite3.Connection so they can
|
|
4
|
+
be used from the UI, a future CLI, or tests without any Qt dependency.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import csv
|
|
10
|
+
import json
|
|
11
|
+
import sqlite3
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from openansho import db
|
|
15
|
+
from openansho.db import Code
|
|
16
|
+
|
|
17
|
+
CSV_FIELDNAMES = [
|
|
18
|
+
"document",
|
|
19
|
+
"code",
|
|
20
|
+
"start_offset",
|
|
21
|
+
"end_offset",
|
|
22
|
+
"text",
|
|
23
|
+
"memo",
|
|
24
|
+
"username",
|
|
25
|
+
"created_at",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def code_path(codes_by_id: dict[int, Code], code_id: int) -> str:
|
|
30
|
+
parts = []
|
|
31
|
+
current = codes_by_id.get(code_id)
|
|
32
|
+
while current is not None:
|
|
33
|
+
parts.append(current.name)
|
|
34
|
+
current = codes_by_id.get(current.parent_id) if current.parent_id else None
|
|
35
|
+
return " > ".join(reversed(parts))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _segment_rows(conn: sqlite3.Connection) -> list[dict]:
|
|
39
|
+
codes_by_id = {code.id: code for code in db.list_codes(conn)}
|
|
40
|
+
documents = db.list_documents(conn)
|
|
41
|
+
|
|
42
|
+
rows = []
|
|
43
|
+
for document in documents:
|
|
44
|
+
for segment in db.list_segments_for_document(conn, document.id):
|
|
45
|
+
rows.append(
|
|
46
|
+
{
|
|
47
|
+
"document": document.name,
|
|
48
|
+
"code": code_path(codes_by_id, segment.code_id),
|
|
49
|
+
"start_offset": segment.start_offset,
|
|
50
|
+
"end_offset": segment.end_offset,
|
|
51
|
+
"text": document.content[segment.start_offset : segment.end_offset],
|
|
52
|
+
"memo": segment.memo or "",
|
|
53
|
+
"username": segment.created_by or "",
|
|
54
|
+
"created_at": segment.created_at,
|
|
55
|
+
}
|
|
56
|
+
)
|
|
57
|
+
rows.sort(key=lambda row: (row["document"], row["start_offset"]))
|
|
58
|
+
return rows
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def export_segments_csv(conn: sqlite3.Connection, path: str | Path) -> int:
|
|
62
|
+
rows = _segment_rows(conn)
|
|
63
|
+
with open(path, "w", newline="", encoding="utf-8") as f:
|
|
64
|
+
writer = csv.DictWriter(f, fieldnames=CSV_FIELDNAMES)
|
|
65
|
+
writer.writeheader()
|
|
66
|
+
writer.writerows(rows)
|
|
67
|
+
return len(rows)
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def export_segments_json(conn: sqlite3.Connection, path: str | Path) -> int:
|
|
71
|
+
rows = _segment_rows(conn)
|
|
72
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
73
|
+
json.dump(rows, f, indent=2)
|
|
74
|
+
return len(rows)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def code_frequency(conn: sqlite3.Connection) -> list[dict]:
|
|
78
|
+
codes_by_id = {code.id: code for code in db.list_codes(conn)}
|
|
79
|
+
rows = [
|
|
80
|
+
{
|
|
81
|
+
"code_id": code_id,
|
|
82
|
+
"path": code_path(codes_by_id, code_id),
|
|
83
|
+
"count": len(db.list_segments_for_code(conn, code_id)),
|
|
84
|
+
}
|
|
85
|
+
for code_id in codes_by_id
|
|
86
|
+
]
|
|
87
|
+
rows.sort(key=lambda row: (-row["count"], row["path"]))
|
|
88
|
+
return rows
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def code_user_frequency(conn: sqlite3.Connection) -> list[dict]:
|
|
92
|
+
"""One row per (code, user) pair: the code, its parent, the username, and
|
|
93
|
+
how many segments that user has coded with that code across the dataset."""
|
|
94
|
+
codes_by_id = {code.id: code for code in db.list_codes(conn)}
|
|
95
|
+
|
|
96
|
+
counts: dict[tuple[int, str], int] = {}
|
|
97
|
+
for code_id in codes_by_id:
|
|
98
|
+
for segment in db.list_segments_for_code(conn, code_id):
|
|
99
|
+
key = (code_id, segment.created_by or "")
|
|
100
|
+
counts[key] = counts.get(key, 0) + 1
|
|
101
|
+
|
|
102
|
+
rows = []
|
|
103
|
+
for (code_id, username), count in counts.items():
|
|
104
|
+
code = codes_by_id[code_id]
|
|
105
|
+
parent = codes_by_id.get(code.parent_id) if code.parent_id else None
|
|
106
|
+
rows.append(
|
|
107
|
+
{
|
|
108
|
+
"code": code.name,
|
|
109
|
+
"parent": parent.name if parent else "",
|
|
110
|
+
"username": username,
|
|
111
|
+
"count": count,
|
|
112
|
+
}
|
|
113
|
+
)
|
|
114
|
+
rows.sort(key=lambda row: (row["code"], row["username"]))
|
|
115
|
+
return rows
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
CODE_FREQUENCY_CSV_FIELDNAMES = ["code", "count"]
|
|
119
|
+
CODE_USER_FREQUENCY_CSV_FIELDNAMES = ["code", "parent", "username", "count"]
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def export_code_frequency_csv(conn: sqlite3.Connection, path: str | Path) -> int:
|
|
123
|
+
rows = code_frequency(conn)
|
|
124
|
+
with open(path, "w", newline="", encoding="utf-8") as f:
|
|
125
|
+
writer = csv.DictWriter(f, fieldnames=CODE_FREQUENCY_CSV_FIELDNAMES)
|
|
126
|
+
writer.writeheader()
|
|
127
|
+
for row in rows:
|
|
128
|
+
writer.writerow({"code": row["path"], "count": row["count"]})
|
|
129
|
+
return len(rows)
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def export_code_frequency_json(conn: sqlite3.Connection, path: str | Path) -> int:
|
|
133
|
+
rows = code_frequency(conn)
|
|
134
|
+
export_rows = [{"code": row["path"], "count": row["count"]} for row in rows]
|
|
135
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
136
|
+
json.dump(export_rows, f, indent=2)
|
|
137
|
+
return len(rows)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def export_code_user_frequency_csv(conn: sqlite3.Connection, path: str | Path) -> int:
|
|
141
|
+
rows = code_user_frequency(conn)
|
|
142
|
+
with open(path, "w", newline="", encoding="utf-8") as f:
|
|
143
|
+
writer = csv.DictWriter(f, fieldnames=CODE_USER_FREQUENCY_CSV_FIELDNAMES)
|
|
144
|
+
writer.writeheader()
|
|
145
|
+
writer.writerows(rows)
|
|
146
|
+
return len(rows)
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def export_code_user_frequency_json(conn: sqlite3.Connection, path: str | Path) -> int:
|
|
150
|
+
rows = code_user_frequency(conn)
|
|
151
|
+
with open(path, "w", encoding="utf-8") as f:
|
|
152
|
+
json.dump(rows, f, indent=2)
|
|
153
|
+
return len(rows)
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Turn an on-disk document into the plain text OpenAnsho codes against.
|
|
2
|
+
|
|
3
|
+
Like `db.py` and `reporting.py`, this module has no Qt imports: the UI hands it
|
|
4
|
+
a path and gets back a string (or a `DocumentReadError` carrying a message fit
|
|
5
|
+
for a dialog).
|
|
6
|
+
|
|
7
|
+
The `.docx` reader is hand-rolled on top of `zipfile`/`xml.etree` rather than
|
|
8
|
+
`python-docx` so the packaged app keeps its single third-party dependency.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import xml.etree.ElementTree as ET
|
|
14
|
+
import zipfile
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
W_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
|
18
|
+
DOCX_BODY_PART = "word/document.xml"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class DocumentReadError(Exception):
|
|
22
|
+
"""A document couldn't be read; the message is shown to the user as-is."""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def read_document_text(path: Path) -> str:
|
|
26
|
+
"""Read `path` as plain text, dispatching on its extension."""
|
|
27
|
+
suffix = path.suffix.lower()
|
|
28
|
+
if suffix == ".docx":
|
|
29
|
+
return read_docx_text(path)
|
|
30
|
+
if suffix == ".doc":
|
|
31
|
+
raise DocumentReadError(
|
|
32
|
+
f"{path.name} is a legacy Word document. Open it in Word and save it "
|
|
33
|
+
"as .docx (or plain text), then import that."
|
|
34
|
+
)
|
|
35
|
+
try:
|
|
36
|
+
return path.read_text(encoding="utf-8")
|
|
37
|
+
except UnicodeDecodeError:
|
|
38
|
+
raise DocumentReadError(f"Could not read {path.name} as UTF-8 text.") from None
|
|
39
|
+
except OSError as exc:
|
|
40
|
+
raise DocumentReadError(f"Could not read {path.name}: {exc}") from None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def read_docx_text(path: Path) -> str:
|
|
44
|
+
"""Extract the body text of a `.docx` file, one line per paragraph.
|
|
45
|
+
|
|
46
|
+
Paragraphs are visited in document order, including those nested inside
|
|
47
|
+
tables (each table cell's paragraphs become their own lines).
|
|
48
|
+
"""
|
|
49
|
+
try:
|
|
50
|
+
with zipfile.ZipFile(path) as archive:
|
|
51
|
+
try:
|
|
52
|
+
body_xml = archive.read(DOCX_BODY_PART)
|
|
53
|
+
except KeyError:
|
|
54
|
+
raise DocumentReadError(
|
|
55
|
+
f"{path.name} is not a valid Word document "
|
|
56
|
+
f"(it has no {DOCX_BODY_PART})."
|
|
57
|
+
) from None
|
|
58
|
+
except zipfile.BadZipFile:
|
|
59
|
+
raise DocumentReadError(
|
|
60
|
+
f"{path.name} is not a valid Word document (it is not a .docx archive)."
|
|
61
|
+
) from None
|
|
62
|
+
except OSError as exc:
|
|
63
|
+
raise DocumentReadError(f"Could not read {path.name}: {exc}") from None
|
|
64
|
+
|
|
65
|
+
try:
|
|
66
|
+
root = ET.fromstring(body_xml)
|
|
67
|
+
except ET.ParseError as exc:
|
|
68
|
+
raise DocumentReadError(f"Could not parse {path.name}: {exc}") from None
|
|
69
|
+
|
|
70
|
+
body = root.find(_w("body"))
|
|
71
|
+
if body is None:
|
|
72
|
+
body = root
|
|
73
|
+
paragraphs = [_paragraph_text(p) for p in body.iter(_w("p"))]
|
|
74
|
+
return "\n".join(paragraphs)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _w(tag: str) -> str:
|
|
78
|
+
return f"{{{W_NAMESPACE}}}{tag}"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _paragraph_text(paragraph: ET.Element) -> str:
|
|
82
|
+
"""Flatten one `w:p` into a line of text.
|
|
83
|
+
|
|
84
|
+
Only `w:t` runs contribute characters, which conveniently skips field
|
|
85
|
+
instructions (`w:instrText`) and tracked deletions (`w:delText`); tabs and
|
|
86
|
+
in-paragraph breaks are preserved as whitespace.
|
|
87
|
+
"""
|
|
88
|
+
pieces: list[str] = []
|
|
89
|
+
for node in paragraph.iter():
|
|
90
|
+
tag = node.tag
|
|
91
|
+
if tag == _w("t"):
|
|
92
|
+
pieces.append(node.text or "")
|
|
93
|
+
elif tag == _w("tab"):
|
|
94
|
+
pieces.append("\t")
|
|
95
|
+
elif tag in (_w("br"), _w("cr")):
|
|
96
|
+
pieces.append("\n")
|
|
97
|
+
return "".join(pieces)
|
openansho/tutorial.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
"""The built-in tutorial project.
|
|
2
|
+
|
|
3
|
+
The app opens this project at startup so a first-time user has something to
|
|
4
|
+
code without importing anything first. Its database lives in memory rather
|
|
5
|
+
than in a .sqlite file, so whatever codebook the user builds while following
|
|
6
|
+
along is discarded on quit and the tutorial is fresh again on the next launch.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import sqlite3
|
|
12
|
+
import sys
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from openansho import db
|
|
16
|
+
|
|
17
|
+
DOCUMENT_NAME = "tutorial.txt"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def tutorial_text_path() -> Path:
|
|
21
|
+
"""Locate the tutorial text, bundled or running from source.
|
|
22
|
+
|
|
23
|
+
PyInstaller extracts the bundled copy (see the Makefile's --add-data and
|
|
24
|
+
OpenAnsho.spec) under sys._MEIPASS at runtime; fall back to the file
|
|
25
|
+
sitting next to this module when running from a source checkout.
|
|
26
|
+
"""
|
|
27
|
+
meipass = getattr(sys, "_MEIPASS", None)
|
|
28
|
+
if meipass:
|
|
29
|
+
return Path(meipass) / "openansho" / DOCUMENT_NAME
|
|
30
|
+
return Path(__file__).resolve().with_name(DOCUMENT_NAME)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def read_tutorial_text() -> str:
|
|
34
|
+
return tutorial_text_path().read_text(encoding="utf-8")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def open_tutorial_project() -> sqlite3.Connection:
|
|
38
|
+
"""Open a throwaway in-memory project holding just the tutorial document."""
|
|
39
|
+
conn = db.connect(":memory:")
|
|
40
|
+
db.create_document(conn, DOCUMENT_NAME, read_tutorial_text())
|
|
41
|
+
return conn
|