exegete 0.14.1a0.dev1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
exegete/__init__.py ADDED
@@ -0,0 +1,16 @@
1
+ # SPDX-License-Identifier: LGPL-3.0-or-later
2
+ """Exegete: a qualitative analysis application you use in conversation
3
+ with an AI assistant, compatible with QualCoder. An MCP server, formerly
4
+ qualcoder-mcp."""
5
+
6
+ from importlib.metadata import PackageNotFoundError, version
7
+
8
+ from .names import DISTRIBUTION
9
+
10
+ try:
11
+ # Single source of truth: the installed package metadata (pyproject.toml).
12
+ # Not the old name: with the old name's package installed beside this
13
+ # one, version("qualcoder-mcp") would report that package's version.
14
+ __version__ = version(DISTRIBUTION)
15
+ except PackageNotFoundError: # running from a source tree without install
16
+ __version__ = "0.0.0+unknown"
@@ -0,0 +1,208 @@
1
+ # SPDX-License-Identifier: LGPL-3.0-or-later
2
+ """Coder-comparison statistics (v0.12 B3, D2).
3
+
4
+ QualCoder shows these numbers in two dialogs (`reports.py:820-1395` and
5
+ `report_compare_coder_file.py:52-1143` at the pinned master 9bddf17).
6
+ This module computes them from character sets rather than from segment
7
+ counts, and reproduces QualCoder's own expressions exactly, in its own
8
+ order, so that where the counts agree the values are bit-identical.
9
+
10
+ Two names, deliberately. `kappa_qualcoder` is QualCoder's "Kappa" column
11
+ reproduced from the same counts; it is NOT Cohen's kappa (its chance term
12
+ is a product of four proportions over the coded characters only, and its
13
+ own docstring at `reports.py:1124` describes a different formula from the
14
+ one the code computes at `:1147`). `kappa_cohen` is the textbook
15
+ statistic over every character in scope. Both are always present, so no
16
+ reader has to guess which one they are looking at, and neither is ever
17
+ called plain `kappa` for our own numbers.
18
+ """
19
+
20
+ from typing import Any, Dict, List, Optional, Sequence
21
+
22
+ # The undefined-value reasons. Never the string "zerodiv", which is what
23
+ # QualCoder's dialog displays in the Kappa column (reports.py:1140,
24
+ # :1006): a reason belongs in a field a reader can act on, not in a
25
+ # number's place.
26
+ NOTE_NO_CHARACTERS = "no characters in scope"
27
+ NOTE_NO_VARIANCE = (
28
+ "undefined: neither coder applied this code in the selected scope "
29
+ "(no variance)")
30
+ NOTE_BOTH_CODED_ALL = (
31
+ "kappa_cohen undefined: both coders coded every character in scope "
32
+ "(no variance); kappa_qualcoder is 1.0 by QualCoder's formula")
33
+ MEAN_COHEN_FEWER_CODES_NOTE = (
34
+ "codes_included counts the codes behind kappa_qualcoder. kappa_cohen "
35
+ "is undefined for codes where both coders coded every character in "
36
+ "scope, so its mean is over codes_included_kappa_cohen codes "
37
+ "instead. The per-code rows say which.")
38
+
39
+
40
+ def kappa_qualcoder(coded_a: int, coded_b: int, both: int) -> Optional[float]:
41
+ """QualCoder's "Kappa" column, expression for expression.
42
+
43
+ A verbatim transcription of `reports.py:1140-1151` (identical at
44
+ `report_compare_coder_file.py:818-830` and at
45
+ `3.8.2:reports.py:812-820`), including the order of the operations,
46
+ so the floating-point result is the same one the dialog shows. The
47
+ `ZeroDivisionError` branch fires only when neither coder applied the
48
+ code, which is the only undefined case here: the chance term is at
49
+ most 1/16, so `1 - Pe` is never zero.
50
+ """
51
+ try:
52
+ unique_codings = coded_a + coded_b - both
53
+ Po = both / unique_codings
54
+ Pyes = coded_a / unique_codings * coded_b / unique_codings
55
+ Pno = ((unique_codings - coded_a) / unique_codings
56
+ * (unique_codings - coded_b) / unique_codings)
57
+ Pe = Pyes * Pno
58
+ return round((Po - Pe) / (1 - Pe), 4)
59
+ except ZeroDivisionError:
60
+ return None
61
+
62
+
63
+ def kappa_cohen(characters: int, coded_a: int, coded_b: int,
64
+ both: int) -> Optional[float]:
65
+ """Cohen's kappa on the 2x2 table over every character in scope.
66
+
67
+ Undefined exactly when there is no variance to correct for: both
68
+ coders coded nothing, or both coded everything. Sensitive to the
69
+ amount of uncoded text, which is the prevalence effect QualCoder's
70
+ author was reacting to when they wrote their own formula.
71
+ """
72
+ n = characters
73
+ if n <= 0:
74
+ return None
75
+ neither = n - coded_a - coded_b + both
76
+ Po = (both + neither) / n
77
+ Pe = (coded_a / n) * (coded_b / n) + ((n - coded_a) / n) * ((n - coded_b) / n)
78
+ if Pe == 1:
79
+ return None
80
+ return round((Po - Pe) / (1 - Pe), 4)
81
+
82
+
83
+ def statistics(characters: int, coded_a: int, coded_b: int,
84
+ both: int) -> Dict[str, Any]:
85
+ """Every field of one comparison row, from four counts.
86
+
87
+ Percentages use QualCoder's expressions and its 2-decimal rounding;
88
+ the kappas use 4 decimals, as its own code does.
89
+ """
90
+ a_only = coded_a - both
91
+ b_only = coded_b - both
92
+ union = coded_a + coded_b - both
93
+ neither = characters - union
94
+ row: Dict[str, Any] = {
95
+ "characters": characters,
96
+ "coded_a": coded_a,
97
+ "coded_b": coded_b,
98
+ "both": both,
99
+ "a_only": a_only,
100
+ "b_only": b_only,
101
+ "neither": neither,
102
+ }
103
+ if characters <= 0:
104
+ row.update({
105
+ "agreement_pct": None, "dual_coded_pct": None,
106
+ "uncoded_pct": None, "disagreement_pct": None,
107
+ "agree_coded_only_pct": None,
108
+ "kappa_qualcoder": None, "kappa_cohen": None,
109
+ "kappa_note": NOTE_NO_CHARACTERS,
110
+ })
111
+ return row
112
+
113
+ agreement = round(100 * (both + neither) / characters, 2)
114
+ row["agreement_pct"] = agreement
115
+ row["dual_coded_pct"] = round(100 * both / characters, 2)
116
+ row["uncoded_pct"] = round(100 * neither / characters, 2)
117
+ row["disagreement_pct"] = round(100 - agreement, 2)
118
+ row["agree_coded_only_pct"] = (round(100 * both / union, 2)
119
+ if union else None)
120
+ row["kappa_qualcoder"] = kappa_qualcoder(coded_a, coded_b, both)
121
+ row["kappa_cohen"] = kappa_cohen(characters, coded_a, coded_b, both)
122
+ if union == 0:
123
+ row["kappa_note"] = NOTE_NO_VARIANCE
124
+ elif coded_a == coded_b == characters:
125
+ row["kappa_note"] = NOTE_BOTH_CODED_ALL
126
+ return row
127
+
128
+
129
+ # ---------------------------------------------------------------------------
130
+ # The GUI-divergence port (D2 3.8): QualCoder's own multiplicity counting
131
+ # ---------------------------------------------------------------------------
132
+ # This exists ONLY so a result can say what QualCoder's dialog would show
133
+ # for the same data when a coder's own segments of one code overlap, and
134
+ # so the parity tests can compare the two. It is never the headline
135
+ # number. A verbatim port of the counting loop and statistics of
136
+ # `reports.py:1061-1152` (cited also to `3.8.2:reports.py:705-822`).
137
+
138
+ def qualcoder_report_values(text_length: int,
139
+ spans_a: Sequence[Sequence[int]],
140
+ spans_b: Sequence[Sequence[int]]) -> Dict[str, Any]:
141
+ """What QualCoder's Coder comparison report would show.
142
+
143
+ A verbatim port of `reports.py:1061-1108` at 9bddf17 (the same code at
144
+ `3.8.2:reports.py:705-765`), including the details that make it differ
145
+ from ours: ONE shared array is incremented once per coded character
146
+ PER SEGMENT for both coders, so two overlapping segments of the same
147
+ coder push a character to 2; "dual coded" is then literally `count ==
148
+ 2`, so such a character is counted as agreement although only one
149
+ coder coded it, and a character covered by three segments is counted
150
+ as neither uncoded, single nor dual and vanishes from the
151
+ percentages. Characters beyond the end of the text are dropped by its
152
+ `IndexError` catch (`:1067-1073`).
153
+ """
154
+ coded0 = coded1 = 0
155
+ char_list = [0] * max(0, int(text_length))
156
+ for pos0, pos1 in spans_a:
157
+ for char in range(int(pos0), int(pos1)):
158
+ if 0 <= char < text_length:
159
+ char_list[char] += 1
160
+ coded0 += 1
161
+ for pos0, pos1 in spans_b:
162
+ for char in range(int(pos0), int(pos1)):
163
+ if 0 <= char < text_length:
164
+ char_list[char] += 1
165
+ coded1 += 1
166
+ uncoded = single_coded = dual_coded = 0
167
+ for char in char_list:
168
+ if char == 0:
169
+ uncoded += 1
170
+ if char == 1:
171
+ single_coded += 1
172
+ if char == 2:
173
+ dual_coded += 1
174
+ values: Dict[str, Any] = {"coded0": coded0, "coded1": coded1,
175
+ "dual_coded": dual_coded,
176
+ "single_coded": single_coded,
177
+ "uncoded": uncoded,
178
+ "characters": int(text_length)}
179
+ if text_length:
180
+ agreement = round(100 * (dual_coded + uncoded) / text_length, 2)
181
+ values["agreement_pct"] = agreement
182
+ values["dual_coded_pct"] = round(100 * dual_coded / text_length, 2)
183
+ values["uncoded_pct"] = round(100 * uncoded / text_length, 2)
184
+ values["disagreement_pct"] = round(100 - agreement, 2)
185
+ try:
186
+ values["agree_coded_only_pct"] = round(
187
+ 100 * dual_coded / (dual_coded + single_coded), 2)
188
+ except ZeroDivisionError:
189
+ # Upstream shows the string "zero div" here; a reason belongs
190
+ # in a note, so ours is null and the note says why (D2 3.7).
191
+ values["agree_coded_only_pct"] = None
192
+ else:
193
+ values["agreement_pct"] = None
194
+ values["dual_coded_pct"] = None
195
+ values["uncoded_pct"] = None
196
+ values["disagreement_pct"] = None
197
+ values["agree_coded_only_pct"] = None
198
+ # The GUI's own column name, reproduced under the GUI's own name
199
+ # inside this disclosure block only (X3).
200
+ values["kappa"] = kappa_qualcoder(coded0, coded1, dual_coded)
201
+ return values
202
+
203
+
204
+ SAME_CODER_OVERLAP_NOTE = (
205
+ "QualCoder's Coder comparison report counts a character once per "
206
+ "segment, so overlapping segments of the same code by the same coder "
207
+ "change its numbers; the values above are what QualCoder would show. "
208
+ "compare_coders counts each character once per coder.")
exegete/cursors.py ADDED
@@ -0,0 +1,218 @@
1
+ # SPDX-License-Identifier: LGPL-3.0-or-later
2
+ """Deterministic keyset cursors for paged reads (v0.12, D4 3.2).
3
+
4
+ A cursor is a POSITION, not a permission and not a stored result set. It
5
+ carries the sort key of the last item a page returned, and the next page
6
+ is computed by re-querying for the first item that sorts strictly after
7
+ it. Nothing is stored anywhere: no server-side result set, no session
8
+ file, no memory. A host that recycles the process between turns can hand
9
+ the same cursor back and get the same continuation, and two hosts walking
10
+ one project do not interfere.
11
+
12
+ These tokens are deliberately NOT the authorisation tokens of
13
+ `preview_tokens.py` (D4 3.11, H1). Those prove that a preview of a
14
+ destructive operation was computed and are signed; these say "carry on
15
+ from here" and are signed by nothing, because a caller who forges a
16
+ position gets a page it could have asked for anyway (D4 5.4). The two
17
+ prefixes, `c1.` and `qcp1.`, stay distinct so neither is ever mistaken
18
+ for the other.
19
+
20
+ The token is bound to the tool and to the call's other arguments through
21
+ a fingerprint, so a cursor cannot be replayed against a different query,
22
+ a different coder filter, or a different tool.
23
+ """
24
+
25
+ import base64
26
+ import hashlib
27
+ import json
28
+ import os
29
+ from pathlib import Path
30
+ from typing import Any, Dict, List, Optional, Sequence, Tuple
31
+
32
+ CURSOR_PREFIX = "c1."
33
+ # A key of five short values plus the fingerprint fits in well under 200
34
+ # characters; the cap is a hostile-input bound, not a design limit.
35
+ CURSOR_MAX_LENGTH = 1024
36
+
37
+ # Tool tags. Short because the token travels in every paged result.
38
+ TAG_SEARCH_FILES = "sf"
39
+ # "sct2" since v0.14 (reads and exports, fix round 4): the key's file name
40
+ # is its stored bytes as hex, where "sct" carried the name as text. Under
41
+ # one tag a name made only of hex digits ("01", "2024", "beef") in a
42
+ # cursor minted before the change was read as bytes, and the walk
43
+ # repeated or skipped rows; with its own tag every earlier cursor gets
44
+ # the one cursor refusal, and the search starts again.
45
+ TAG_SEARCH_CODED_TEXT = "sct2"
46
+ TAG_CODED_SEGMENTS = "gcs"
47
+
48
+ CURSOR_TOO_LONG = "cursor is too long (limit 1024 characters)."
49
+ # The running total a cursor carries is the caller's claim about the
50
+ # pages BEFORE this one; nothing here can verify it, and it is reported
51
+ # in the result as `returned_so_far`. Bound it so a tampered cursor
52
+ # cannot put an arbitrary integer in front of a researcher as though
53
+ # this server had counted it. The bound is far above any real walk: a
54
+ # project with this many coded segments would exhaust the character
55
+ # budget thousands of pages earlier (fix round 4).
56
+ CURSOR_MAX_RETURNED_SO_FAR = 10_000_000
57
+
58
+ DATABASE_CHANGED_NOTE = (
59
+ "The project database changed after this cursor was issued; positions "
60
+ "are recomputed on every page, but counts and ordering may differ from "
61
+ "the earlier pages.")
62
+
63
+
64
+ class CursorError(ValueError):
65
+ """An unusable cursor. The message never echoes the token."""
66
+
67
+
68
+ def cursor_invalid_message(tool_name: str) -> str:
69
+ """The one text every unusable cursor gets (D4 3.7).
70
+
71
+ Garbage, a token minted for another tool, a token minted for other
72
+ arguments and a token whose key has the wrong shape are one failure
73
+ from the caller's point of view, and the answer is the same: start
74
+ again without the cursor. The token is never echoed back, so a
75
+ tampered value cannot smuggle text into the conversation.
76
+ """
77
+ return (f"cursor is not valid for {tool_name} with these arguments. "
78
+ f"Call the tool again without cursor to start from the "
79
+ f"beginning.")
80
+
81
+
82
+ def _canonical(obj: Any) -> str:
83
+ """The canonical JSON this module hashes and encodes."""
84
+ return json.dumps(obj, sort_keys=True, separators=(",", ":"),
85
+ ensure_ascii=True)
86
+
87
+
88
+ def fingerprint_arguments(tool: str, arguments: Dict[str, Any]) -> str:
89
+ """A 16-hex digest binding a cursor to this call's other arguments.
90
+
91
+ The caller passes the arguments already canonicalised (defaults
92
+ filled in, order-irrelevant lists sorted and de-duplicated, `coder`
93
+ normalised), so a call that means the same thing produces the same
94
+ fingerprint whatever order the host serialised it in. It carries no
95
+ authority: it decides only whether a cursor belongs to this query.
96
+ """
97
+ payload = {"t": tool, "a": arguments}
98
+ return hashlib.sha256(_canonical(payload).encode("utf-8")).hexdigest()[:16]
99
+
100
+
101
+ def database_stamp(qda_path: Any) -> Optional[List[int]]:
102
+ """`(st_mtime_ns, st_size)` of data.qda, or None when unavailable.
103
+
104
+ Both values travel INSIDE the cursor and therefore into the
105
+ transcript; PRIVACY.md enumerates them, and `list_available_projects`
106
+ already reports the same two in plain form (fix round 4).
107
+
108
+ A HEURISTIC (D4 3.2.5) and labelled as one wherever it is reported:
109
+ mtime granularity, journal and WAL side files, and copy tools that
110
+ preserve timestamps all mean a changed database can look unchanged
111
+ and an unchanged one can look changed. Correctness never depends on
112
+ it; it only lets a page say that the ground may have moved.
113
+ """
114
+ try:
115
+ st = os.stat(str(qda_path))
116
+ except OSError:
117
+ return None
118
+ return [st.st_mtime_ns, st.st_size]
119
+
120
+
121
+ def encode_cursor(tool_tag: str, fingerprint: str, key: Sequence[Any],
122
+ returned_so_far: int,
123
+ stamp: Optional[List[int]]) -> str:
124
+ """Mint a cursor for the item last returned."""
125
+ payload = {
126
+ "t": tool_tag,
127
+ "f": fingerprint,
128
+ "k": list(key),
129
+ "n": int(returned_so_far),
130
+ "d": list(stamp) if stamp else [],
131
+ }
132
+ raw = _canonical(payload).encode("utf-8")
133
+ return CURSOR_PREFIX + base64.urlsafe_b64encode(raw).decode(
134
+ "ascii").rstrip("=")
135
+
136
+
137
+ def decode_cursor(token: Any, tool_tag: str, fingerprint: str,
138
+ key_shape: Sequence[type]) -> Tuple[List[Any], int,
139
+ Optional[List[int]]]:
140
+ """Read a cursor, or raise CursorError.
141
+
142
+ Every check is a refusal, never a repair: the prefix, the length, the
143
+ base64, the JSON shape, the exact key set, the tool tag, the
144
+ fingerprint of the current arguments, the shape of the sort key and a
145
+ non-negative count. A cursor that fails any of them is not a cursor
146
+ for this call.
147
+
148
+ Returns:
149
+ (key, returned_so_far, stamp)
150
+ """
151
+ if not isinstance(token, str):
152
+ raise CursorError("cursor must be a string")
153
+ token = token.strip()
154
+ if len(token) > CURSOR_MAX_LENGTH:
155
+ raise CursorError(CURSOR_TOO_LONG)
156
+ if not token.startswith(CURSOR_PREFIX):
157
+ raise CursorError("cursor has an unknown format")
158
+ body = token[len(CURSOR_PREFIX):]
159
+ padding = "=" * (-len(body) % 4)
160
+ try:
161
+ raw = base64.urlsafe_b64decode(body + padding)
162
+ data = json.loads(raw.decode("utf-8"))
163
+ except Exception as e: # noqa: BLE001 - any failure
164
+ raise CursorError("cursor could not be decoded") from e
165
+ if not isinstance(data, dict) or set(data) != {"t", "f", "k", "n", "d"}:
166
+ raise CursorError("cursor has the wrong shape")
167
+ if data["t"] != tool_tag:
168
+ raise CursorError("cursor belongs to another tool")
169
+ if not isinstance(data["f"], str) or data["f"] != fingerprint:
170
+ raise CursorError("cursor belongs to another query")
171
+ key = data["k"]
172
+ if not isinstance(key, list) or len(key) != len(key_shape):
173
+ raise CursorError("cursor key has the wrong shape")
174
+ for value, expected in zip(key, key_shape):
175
+ if expected is str:
176
+ if not isinstance(value, str):
177
+ raise CursorError("cursor key has the wrong shape")
178
+ elif expected is int:
179
+ if not isinstance(value, int) or isinstance(value, bool):
180
+ raise CursorError("cursor key has the wrong shape")
181
+ elif expected is object:
182
+ # A nullable text column: a string or None
183
+ if value is not None and not isinstance(value, str):
184
+ raise CursorError("cursor key has the wrong shape")
185
+ n = data["n"]
186
+ if (not isinstance(n, int) or isinstance(n, bool) or n < 0
187
+ or n > CURSOR_MAX_RETURNED_SO_FAR):
188
+ raise CursorError("cursor count is not a count")
189
+ stamp = data["d"]
190
+ if not isinstance(stamp, list) or not all(
191
+ isinstance(v, int) and not isinstance(v, bool) for v in stamp):
192
+ raise CursorError("cursor stamp has the wrong shape")
193
+ return key, n, (stamp or None)
194
+
195
+
196
+ def page_block(limit: int, returned: int, returned_so_far: int,
197
+ has_more: bool, next_cursor: Optional[str],
198
+ exhaustive: bool) -> Dict[str, Any]:
199
+ """The `page` block every paged result carries (D4 3.2.7).
200
+
201
+ `returned`, `has_more` and `exhaustive` are computed on this page.
202
+ `returned_so_far` is this page's `returned` added to the count the
203
+ CURSOR carried, so on any page but the first it rests on a value
204
+ the caller supplied and nothing here can check. It is bounded at
205
+ decode (CURSOR_MAX_RETURNED_SO_FAR) so a tampered cursor cannot put
206
+ an arbitrary integer in a result, and PRIVACY.md says plainly where
207
+ the number comes from; a stronger guarantee would mean signing
208
+ cursors, which D4 3.11 deliberately does not do, because a caller
209
+ who forges a POSITION only gets a page it could have asked for.
210
+ """
211
+ return {
212
+ "limit": limit,
213
+ "returned": returned,
214
+ "returned_so_far": returned_so_far,
215
+ "has_more": has_more,
216
+ "next_cursor": next_cursor,
217
+ "exhaustive": exhaustive,
218
+ }