exegete 0.14.1a0.dev1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- exegete/__init__.py +16 -0
- exegete/coder_comparison.py +208 -0
- exegete/cursors.py +218 -0
- exegete/database.py +12621 -0
- exegete/env_settings.py +174 -0
- exegete/memo_privacy.py +215 -0
- exegete/names.py +68 -0
- exegete/new_project.py +1027 -0
- exegete/preview_tokens.py +613 -0
- exegete/project_settings.py +1082 -0
- exegete/pseudonymise.py +2796 -0
- exegete/refi_export.py +645 -0
- exegete/server.py +18899 -0
- exegete/sessions.py +1153 -0
- exegete/state_folder.py +360 -0
- exegete/transition.py +1049 -0
- exegete-0.14.1a0.dev1.dist-info/METADATA +515 -0
- exegete-0.14.1a0.dev1.dist-info/RECORD +24 -0
- exegete-0.14.1a0.dev1.dist-info/WHEEL +5 -0
- exegete-0.14.1a0.dev1.dist-info/entry_points.txt +2 -0
- exegete-0.14.1a0.dev1.dist-info/licenses/COPYING.LESSER +165 -0
- exegete-0.14.1a0.dev1.dist-info/licenses/NOTICE +773 -0
- exegete-0.14.1a0.dev1.dist-info/licenses/legal/GPL-3.0.txt +674 -0
- exegete-0.14.1a0.dev1.dist-info/top_level.txt +1 -0
exegete/__init__.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# SPDX-License-Identifier: LGPL-3.0-or-later
|
|
2
|
+
"""Exegete: a qualitative analysis application you use in conversation
|
|
3
|
+
with an AI assistant, compatible with QualCoder. An MCP server, formerly
|
|
4
|
+
qualcoder-mcp."""
|
|
5
|
+
|
|
6
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
7
|
+
|
|
8
|
+
from .names import DISTRIBUTION
|
|
9
|
+
|
|
10
|
+
try:
|
|
11
|
+
# Single source of truth: the installed package metadata (pyproject.toml).
|
|
12
|
+
# Not the old name: with the old name's package installed beside this
|
|
13
|
+
# one, version("qualcoder-mcp") would report that package's version.
|
|
14
|
+
__version__ = version(DISTRIBUTION)
|
|
15
|
+
except PackageNotFoundError: # running from a source tree without install
|
|
16
|
+
__version__ = "0.0.0+unknown"
|
|
@@ -0,0 +1,208 @@
|
|
|
1
|
+
# SPDX-License-Identifier: LGPL-3.0-or-later
|
|
2
|
+
"""Coder-comparison statistics (v0.12 B3, D2).
|
|
3
|
+
|
|
4
|
+
QualCoder shows these numbers in two dialogs (`reports.py:820-1395` and
|
|
5
|
+
`report_compare_coder_file.py:52-1143` at the pinned master 9bddf17).
|
|
6
|
+
This module computes them from character sets rather than from segment
|
|
7
|
+
counts, and reproduces QualCoder's own expressions exactly, in its own
|
|
8
|
+
order, so that where the counts agree the values are bit-identical.
|
|
9
|
+
|
|
10
|
+
Two names, deliberately. `kappa_qualcoder` is QualCoder's "Kappa" column
|
|
11
|
+
reproduced from the same counts; it is NOT Cohen's kappa (its chance term
|
|
12
|
+
is a product of four proportions over the coded characters only, and its
|
|
13
|
+
own docstring at `reports.py:1124` describes a different formula from the
|
|
14
|
+
one the code computes at `:1147`). `kappa_cohen` is the textbook
|
|
15
|
+
statistic over every character in scope. Both are always present, so no
|
|
16
|
+
reader has to guess which one they are looking at, and neither is ever
|
|
17
|
+
called plain `kappa` for our own numbers.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from typing import Any, Dict, List, Optional, Sequence
|
|
21
|
+
|
|
22
|
+
# The undefined-value reasons. Never the string "zerodiv", which is what
|
|
23
|
+
# QualCoder's dialog displays in the Kappa column (reports.py:1140,
|
|
24
|
+
# :1006): a reason belongs in a field a reader can act on, not in a
|
|
25
|
+
# number's place.
|
|
26
|
+
NOTE_NO_CHARACTERS = "no characters in scope"
|
|
27
|
+
NOTE_NO_VARIANCE = (
|
|
28
|
+
"undefined: neither coder applied this code in the selected scope "
|
|
29
|
+
"(no variance)")
|
|
30
|
+
NOTE_BOTH_CODED_ALL = (
|
|
31
|
+
"kappa_cohen undefined: both coders coded every character in scope "
|
|
32
|
+
"(no variance); kappa_qualcoder is 1.0 by QualCoder's formula")
|
|
33
|
+
MEAN_COHEN_FEWER_CODES_NOTE = (
|
|
34
|
+
"codes_included counts the codes behind kappa_qualcoder. kappa_cohen "
|
|
35
|
+
"is undefined for codes where both coders coded every character in "
|
|
36
|
+
"scope, so its mean is over codes_included_kappa_cohen codes "
|
|
37
|
+
"instead. The per-code rows say which.")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def kappa_qualcoder(coded_a: int, coded_b: int, both: int) -> Optional[float]:
|
|
41
|
+
"""QualCoder's "Kappa" column, expression for expression.
|
|
42
|
+
|
|
43
|
+
A verbatim transcription of `reports.py:1140-1151` (identical at
|
|
44
|
+
`report_compare_coder_file.py:818-830` and at
|
|
45
|
+
`3.8.2:reports.py:812-820`), including the order of the operations,
|
|
46
|
+
so the floating-point result is the same one the dialog shows. The
|
|
47
|
+
`ZeroDivisionError` branch fires only when neither coder applied the
|
|
48
|
+
code, which is the only undefined case here: the chance term is at
|
|
49
|
+
most 1/16, so `1 - Pe` is never zero.
|
|
50
|
+
"""
|
|
51
|
+
try:
|
|
52
|
+
unique_codings = coded_a + coded_b - both
|
|
53
|
+
Po = both / unique_codings
|
|
54
|
+
Pyes = coded_a / unique_codings * coded_b / unique_codings
|
|
55
|
+
Pno = ((unique_codings - coded_a) / unique_codings
|
|
56
|
+
* (unique_codings - coded_b) / unique_codings)
|
|
57
|
+
Pe = Pyes * Pno
|
|
58
|
+
return round((Po - Pe) / (1 - Pe), 4)
|
|
59
|
+
except ZeroDivisionError:
|
|
60
|
+
return None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def kappa_cohen(characters: int, coded_a: int, coded_b: int,
|
|
64
|
+
both: int) -> Optional[float]:
|
|
65
|
+
"""Cohen's kappa on the 2x2 table over every character in scope.
|
|
66
|
+
|
|
67
|
+
Undefined exactly when there is no variance to correct for: both
|
|
68
|
+
coders coded nothing, or both coded everything. Sensitive to the
|
|
69
|
+
amount of uncoded text, which is the prevalence effect QualCoder's
|
|
70
|
+
author was reacting to when they wrote their own formula.
|
|
71
|
+
"""
|
|
72
|
+
n = characters
|
|
73
|
+
if n <= 0:
|
|
74
|
+
return None
|
|
75
|
+
neither = n - coded_a - coded_b + both
|
|
76
|
+
Po = (both + neither) / n
|
|
77
|
+
Pe = (coded_a / n) * (coded_b / n) + ((n - coded_a) / n) * ((n - coded_b) / n)
|
|
78
|
+
if Pe == 1:
|
|
79
|
+
return None
|
|
80
|
+
return round((Po - Pe) / (1 - Pe), 4)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def statistics(characters: int, coded_a: int, coded_b: int,
|
|
84
|
+
both: int) -> Dict[str, Any]:
|
|
85
|
+
"""Every field of one comparison row, from four counts.
|
|
86
|
+
|
|
87
|
+
Percentages use QualCoder's expressions and its 2-decimal rounding;
|
|
88
|
+
the kappas use 4 decimals, as its own code does.
|
|
89
|
+
"""
|
|
90
|
+
a_only = coded_a - both
|
|
91
|
+
b_only = coded_b - both
|
|
92
|
+
union = coded_a + coded_b - both
|
|
93
|
+
neither = characters - union
|
|
94
|
+
row: Dict[str, Any] = {
|
|
95
|
+
"characters": characters,
|
|
96
|
+
"coded_a": coded_a,
|
|
97
|
+
"coded_b": coded_b,
|
|
98
|
+
"both": both,
|
|
99
|
+
"a_only": a_only,
|
|
100
|
+
"b_only": b_only,
|
|
101
|
+
"neither": neither,
|
|
102
|
+
}
|
|
103
|
+
if characters <= 0:
|
|
104
|
+
row.update({
|
|
105
|
+
"agreement_pct": None, "dual_coded_pct": None,
|
|
106
|
+
"uncoded_pct": None, "disagreement_pct": None,
|
|
107
|
+
"agree_coded_only_pct": None,
|
|
108
|
+
"kappa_qualcoder": None, "kappa_cohen": None,
|
|
109
|
+
"kappa_note": NOTE_NO_CHARACTERS,
|
|
110
|
+
})
|
|
111
|
+
return row
|
|
112
|
+
|
|
113
|
+
agreement = round(100 * (both + neither) / characters, 2)
|
|
114
|
+
row["agreement_pct"] = agreement
|
|
115
|
+
row["dual_coded_pct"] = round(100 * both / characters, 2)
|
|
116
|
+
row["uncoded_pct"] = round(100 * neither / characters, 2)
|
|
117
|
+
row["disagreement_pct"] = round(100 - agreement, 2)
|
|
118
|
+
row["agree_coded_only_pct"] = (round(100 * both / union, 2)
|
|
119
|
+
if union else None)
|
|
120
|
+
row["kappa_qualcoder"] = kappa_qualcoder(coded_a, coded_b, both)
|
|
121
|
+
row["kappa_cohen"] = kappa_cohen(characters, coded_a, coded_b, both)
|
|
122
|
+
if union == 0:
|
|
123
|
+
row["kappa_note"] = NOTE_NO_VARIANCE
|
|
124
|
+
elif coded_a == coded_b == characters:
|
|
125
|
+
row["kappa_note"] = NOTE_BOTH_CODED_ALL
|
|
126
|
+
return row
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# ---------------------------------------------------------------------------
|
|
130
|
+
# The GUI-divergence port (D2 3.8): QualCoder's own multiplicity counting
|
|
131
|
+
# ---------------------------------------------------------------------------
|
|
132
|
+
# This exists ONLY so a result can say what QualCoder's dialog would show
|
|
133
|
+
# for the same data when a coder's own segments of one code overlap, and
|
|
134
|
+
# so the parity tests can compare the two. It is never the headline
|
|
135
|
+
# number. A verbatim port of the counting loop and statistics of
|
|
136
|
+
# `reports.py:1061-1152` (cited also to `3.8.2:reports.py:705-822`).
|
|
137
|
+
|
|
138
|
+
def qualcoder_report_values(text_length: int,
|
|
139
|
+
spans_a: Sequence[Sequence[int]],
|
|
140
|
+
spans_b: Sequence[Sequence[int]]) -> Dict[str, Any]:
|
|
141
|
+
"""What QualCoder's Coder comparison report would show.
|
|
142
|
+
|
|
143
|
+
A verbatim port of `reports.py:1061-1108` at 9bddf17 (the same code at
|
|
144
|
+
`3.8.2:reports.py:705-765`), including the details that make it differ
|
|
145
|
+
from ours: ONE shared array is incremented once per coded character
|
|
146
|
+
PER SEGMENT for both coders, so two overlapping segments of the same
|
|
147
|
+
coder push a character to 2; "dual coded" is then literally `count ==
|
|
148
|
+
2`, so such a character is counted as agreement although only one
|
|
149
|
+
coder coded it, and a character covered by three segments is counted
|
|
150
|
+
as neither uncoded, single nor dual and vanishes from the
|
|
151
|
+
percentages. Characters beyond the end of the text are dropped by its
|
|
152
|
+
`IndexError` catch (`:1067-1073`).
|
|
153
|
+
"""
|
|
154
|
+
coded0 = coded1 = 0
|
|
155
|
+
char_list = [0] * max(0, int(text_length))
|
|
156
|
+
for pos0, pos1 in spans_a:
|
|
157
|
+
for char in range(int(pos0), int(pos1)):
|
|
158
|
+
if 0 <= char < text_length:
|
|
159
|
+
char_list[char] += 1
|
|
160
|
+
coded0 += 1
|
|
161
|
+
for pos0, pos1 in spans_b:
|
|
162
|
+
for char in range(int(pos0), int(pos1)):
|
|
163
|
+
if 0 <= char < text_length:
|
|
164
|
+
char_list[char] += 1
|
|
165
|
+
coded1 += 1
|
|
166
|
+
uncoded = single_coded = dual_coded = 0
|
|
167
|
+
for char in char_list:
|
|
168
|
+
if char == 0:
|
|
169
|
+
uncoded += 1
|
|
170
|
+
if char == 1:
|
|
171
|
+
single_coded += 1
|
|
172
|
+
if char == 2:
|
|
173
|
+
dual_coded += 1
|
|
174
|
+
values: Dict[str, Any] = {"coded0": coded0, "coded1": coded1,
|
|
175
|
+
"dual_coded": dual_coded,
|
|
176
|
+
"single_coded": single_coded,
|
|
177
|
+
"uncoded": uncoded,
|
|
178
|
+
"characters": int(text_length)}
|
|
179
|
+
if text_length:
|
|
180
|
+
agreement = round(100 * (dual_coded + uncoded) / text_length, 2)
|
|
181
|
+
values["agreement_pct"] = agreement
|
|
182
|
+
values["dual_coded_pct"] = round(100 * dual_coded / text_length, 2)
|
|
183
|
+
values["uncoded_pct"] = round(100 * uncoded / text_length, 2)
|
|
184
|
+
values["disagreement_pct"] = round(100 - agreement, 2)
|
|
185
|
+
try:
|
|
186
|
+
values["agree_coded_only_pct"] = round(
|
|
187
|
+
100 * dual_coded / (dual_coded + single_coded), 2)
|
|
188
|
+
except ZeroDivisionError:
|
|
189
|
+
# Upstream shows the string "zero div" here; a reason belongs
|
|
190
|
+
# in a note, so ours is null and the note says why (D2 3.7).
|
|
191
|
+
values["agree_coded_only_pct"] = None
|
|
192
|
+
else:
|
|
193
|
+
values["agreement_pct"] = None
|
|
194
|
+
values["dual_coded_pct"] = None
|
|
195
|
+
values["uncoded_pct"] = None
|
|
196
|
+
values["disagreement_pct"] = None
|
|
197
|
+
values["agree_coded_only_pct"] = None
|
|
198
|
+
# The GUI's own column name, reproduced under the GUI's own name
|
|
199
|
+
# inside this disclosure block only (X3).
|
|
200
|
+
values["kappa"] = kappa_qualcoder(coded0, coded1, dual_coded)
|
|
201
|
+
return values
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
SAME_CODER_OVERLAP_NOTE = (
|
|
205
|
+
"QualCoder's Coder comparison report counts a character once per "
|
|
206
|
+
"segment, so overlapping segments of the same code by the same coder "
|
|
207
|
+
"change its numbers; the values above are what QualCoder would show. "
|
|
208
|
+
"compare_coders counts each character once per coder.")
|
exegete/cursors.py
ADDED
|
@@ -0,0 +1,218 @@
|
|
|
1
|
+
# SPDX-License-Identifier: LGPL-3.0-or-later
|
|
2
|
+
"""Deterministic keyset cursors for paged reads (v0.12, D4 3.2).
|
|
3
|
+
|
|
4
|
+
A cursor is a POSITION, not a permission and not a stored result set. It
|
|
5
|
+
carries the sort key of the last item a page returned, and the next page
|
|
6
|
+
is computed by re-querying for the first item that sorts strictly after
|
|
7
|
+
it. Nothing is stored anywhere: no server-side result set, no session
|
|
8
|
+
file, no memory. A host that recycles the process between turns can hand
|
|
9
|
+
the same cursor back and get the same continuation, and two hosts walking
|
|
10
|
+
one project do not interfere.
|
|
11
|
+
|
|
12
|
+
These tokens are deliberately NOT the authorisation tokens of
|
|
13
|
+
`preview_tokens.py` (D4 3.11, H1). Those prove that a preview of a
|
|
14
|
+
destructive operation was computed and are signed; these say "carry on
|
|
15
|
+
from here" and are signed by nothing, because a caller who forges a
|
|
16
|
+
position gets a page it could have asked for anyway (D4 5.4). The two
|
|
17
|
+
prefixes, `c1.` and `qcp1.`, stay distinct so neither is ever mistaken
|
|
18
|
+
for the other.
|
|
19
|
+
|
|
20
|
+
The token is bound to the tool and to the call's other arguments through
|
|
21
|
+
a fingerprint, so a cursor cannot be replayed against a different query,
|
|
22
|
+
a different coder filter, or a different tool.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
import base64
|
|
26
|
+
import hashlib
|
|
27
|
+
import json
|
|
28
|
+
import os
|
|
29
|
+
from pathlib import Path
|
|
30
|
+
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
|
31
|
+
|
|
32
|
+
CURSOR_PREFIX = "c1."
|
|
33
|
+
# A key of five short values plus the fingerprint fits in well under 200
|
|
34
|
+
# characters; the cap is a hostile-input bound, not a design limit.
|
|
35
|
+
CURSOR_MAX_LENGTH = 1024
|
|
36
|
+
|
|
37
|
+
# Tool tags. Short because the token travels in every paged result.
|
|
38
|
+
TAG_SEARCH_FILES = "sf"
|
|
39
|
+
# "sct2" since v0.14 (reads and exports, fix round 4): the key's file name
|
|
40
|
+
# is its stored bytes as hex, where "sct" carried the name as text. Under
|
|
41
|
+
# one tag a name made only of hex digits ("01", "2024", "beef") in a
|
|
42
|
+
# cursor minted before the change was read as bytes, and the walk
|
|
43
|
+
# repeated or skipped rows; with its own tag every earlier cursor gets
|
|
44
|
+
# the one cursor refusal, and the search starts again.
|
|
45
|
+
TAG_SEARCH_CODED_TEXT = "sct2"
|
|
46
|
+
TAG_CODED_SEGMENTS = "gcs"
|
|
47
|
+
|
|
48
|
+
CURSOR_TOO_LONG = "cursor is too long (limit 1024 characters)."
|
|
49
|
+
# The running total a cursor carries is the caller's claim about the
|
|
50
|
+
# pages BEFORE this one; nothing here can verify it, and it is reported
|
|
51
|
+
# in the result as `returned_so_far`. Bound it so a tampered cursor
|
|
52
|
+
# cannot put an arbitrary integer in front of a researcher as though
|
|
53
|
+
# this server had counted it. The bound is far above any real walk: a
|
|
54
|
+
# project with this many coded segments would exhaust the character
|
|
55
|
+
# budget thousands of pages earlier (fix round 4).
|
|
56
|
+
CURSOR_MAX_RETURNED_SO_FAR = 10_000_000
|
|
57
|
+
|
|
58
|
+
DATABASE_CHANGED_NOTE = (
|
|
59
|
+
"The project database changed after this cursor was issued; positions "
|
|
60
|
+
"are recomputed on every page, but counts and ordering may differ from "
|
|
61
|
+
"the earlier pages.")
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class CursorError(ValueError):
|
|
65
|
+
"""An unusable cursor. The message never echoes the token."""
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def cursor_invalid_message(tool_name: str) -> str:
|
|
69
|
+
"""The one text every unusable cursor gets (D4 3.7).
|
|
70
|
+
|
|
71
|
+
Garbage, a token minted for another tool, a token minted for other
|
|
72
|
+
arguments and a token whose key has the wrong shape are one failure
|
|
73
|
+
from the caller's point of view, and the answer is the same: start
|
|
74
|
+
again without the cursor. The token is never echoed back, so a
|
|
75
|
+
tampered value cannot smuggle text into the conversation.
|
|
76
|
+
"""
|
|
77
|
+
return (f"cursor is not valid for {tool_name} with these arguments. "
|
|
78
|
+
f"Call the tool again without cursor to start from the "
|
|
79
|
+
f"beginning.")
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _canonical(obj: Any) -> str:
|
|
83
|
+
"""The canonical JSON this module hashes and encodes."""
|
|
84
|
+
return json.dumps(obj, sort_keys=True, separators=(",", ":"),
|
|
85
|
+
ensure_ascii=True)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def fingerprint_arguments(tool: str, arguments: Dict[str, Any]) -> str:
|
|
89
|
+
"""A 16-hex digest binding a cursor to this call's other arguments.
|
|
90
|
+
|
|
91
|
+
The caller passes the arguments already canonicalised (defaults
|
|
92
|
+
filled in, order-irrelevant lists sorted and de-duplicated, `coder`
|
|
93
|
+
normalised), so a call that means the same thing produces the same
|
|
94
|
+
fingerprint whatever order the host serialised it in. It carries no
|
|
95
|
+
authority: it decides only whether a cursor belongs to this query.
|
|
96
|
+
"""
|
|
97
|
+
payload = {"t": tool, "a": arguments}
|
|
98
|
+
return hashlib.sha256(_canonical(payload).encode("utf-8")).hexdigest()[:16]
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def database_stamp(qda_path: Any) -> Optional[List[int]]:
|
|
102
|
+
"""`(st_mtime_ns, st_size)` of data.qda, or None when unavailable.
|
|
103
|
+
|
|
104
|
+
Both values travel INSIDE the cursor and therefore into the
|
|
105
|
+
transcript; PRIVACY.md enumerates them, and `list_available_projects`
|
|
106
|
+
already reports the same two in plain form (fix round 4).
|
|
107
|
+
|
|
108
|
+
A HEURISTIC (D4 3.2.5) and labelled as one wherever it is reported:
|
|
109
|
+
mtime granularity, journal and WAL side files, and copy tools that
|
|
110
|
+
preserve timestamps all mean a changed database can look unchanged
|
|
111
|
+
and an unchanged one can look changed. Correctness never depends on
|
|
112
|
+
it; it only lets a page say that the ground may have moved.
|
|
113
|
+
"""
|
|
114
|
+
try:
|
|
115
|
+
st = os.stat(str(qda_path))
|
|
116
|
+
except OSError:
|
|
117
|
+
return None
|
|
118
|
+
return [st.st_mtime_ns, st.st_size]
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def encode_cursor(tool_tag: str, fingerprint: str, key: Sequence[Any],
|
|
122
|
+
returned_so_far: int,
|
|
123
|
+
stamp: Optional[List[int]]) -> str:
|
|
124
|
+
"""Mint a cursor for the item last returned."""
|
|
125
|
+
payload = {
|
|
126
|
+
"t": tool_tag,
|
|
127
|
+
"f": fingerprint,
|
|
128
|
+
"k": list(key),
|
|
129
|
+
"n": int(returned_so_far),
|
|
130
|
+
"d": list(stamp) if stamp else [],
|
|
131
|
+
}
|
|
132
|
+
raw = _canonical(payload).encode("utf-8")
|
|
133
|
+
return CURSOR_PREFIX + base64.urlsafe_b64encode(raw).decode(
|
|
134
|
+
"ascii").rstrip("=")
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def decode_cursor(token: Any, tool_tag: str, fingerprint: str,
|
|
138
|
+
key_shape: Sequence[type]) -> Tuple[List[Any], int,
|
|
139
|
+
Optional[List[int]]]:
|
|
140
|
+
"""Read a cursor, or raise CursorError.
|
|
141
|
+
|
|
142
|
+
Every check is a refusal, never a repair: the prefix, the length, the
|
|
143
|
+
base64, the JSON shape, the exact key set, the tool tag, the
|
|
144
|
+
fingerprint of the current arguments, the shape of the sort key and a
|
|
145
|
+
non-negative count. A cursor that fails any of them is not a cursor
|
|
146
|
+
for this call.
|
|
147
|
+
|
|
148
|
+
Returns:
|
|
149
|
+
(key, returned_so_far, stamp)
|
|
150
|
+
"""
|
|
151
|
+
if not isinstance(token, str):
|
|
152
|
+
raise CursorError("cursor must be a string")
|
|
153
|
+
token = token.strip()
|
|
154
|
+
if len(token) > CURSOR_MAX_LENGTH:
|
|
155
|
+
raise CursorError(CURSOR_TOO_LONG)
|
|
156
|
+
if not token.startswith(CURSOR_PREFIX):
|
|
157
|
+
raise CursorError("cursor has an unknown format")
|
|
158
|
+
body = token[len(CURSOR_PREFIX):]
|
|
159
|
+
padding = "=" * (-len(body) % 4)
|
|
160
|
+
try:
|
|
161
|
+
raw = base64.urlsafe_b64decode(body + padding)
|
|
162
|
+
data = json.loads(raw.decode("utf-8"))
|
|
163
|
+
except Exception as e: # noqa: BLE001 - any failure
|
|
164
|
+
raise CursorError("cursor could not be decoded") from e
|
|
165
|
+
if not isinstance(data, dict) or set(data) != {"t", "f", "k", "n", "d"}:
|
|
166
|
+
raise CursorError("cursor has the wrong shape")
|
|
167
|
+
if data["t"] != tool_tag:
|
|
168
|
+
raise CursorError("cursor belongs to another tool")
|
|
169
|
+
if not isinstance(data["f"], str) or data["f"] != fingerprint:
|
|
170
|
+
raise CursorError("cursor belongs to another query")
|
|
171
|
+
key = data["k"]
|
|
172
|
+
if not isinstance(key, list) or len(key) != len(key_shape):
|
|
173
|
+
raise CursorError("cursor key has the wrong shape")
|
|
174
|
+
for value, expected in zip(key, key_shape):
|
|
175
|
+
if expected is str:
|
|
176
|
+
if not isinstance(value, str):
|
|
177
|
+
raise CursorError("cursor key has the wrong shape")
|
|
178
|
+
elif expected is int:
|
|
179
|
+
if not isinstance(value, int) or isinstance(value, bool):
|
|
180
|
+
raise CursorError("cursor key has the wrong shape")
|
|
181
|
+
elif expected is object:
|
|
182
|
+
# A nullable text column: a string or None
|
|
183
|
+
if value is not None and not isinstance(value, str):
|
|
184
|
+
raise CursorError("cursor key has the wrong shape")
|
|
185
|
+
n = data["n"]
|
|
186
|
+
if (not isinstance(n, int) or isinstance(n, bool) or n < 0
|
|
187
|
+
or n > CURSOR_MAX_RETURNED_SO_FAR):
|
|
188
|
+
raise CursorError("cursor count is not a count")
|
|
189
|
+
stamp = data["d"]
|
|
190
|
+
if not isinstance(stamp, list) or not all(
|
|
191
|
+
isinstance(v, int) and not isinstance(v, bool) for v in stamp):
|
|
192
|
+
raise CursorError("cursor stamp has the wrong shape")
|
|
193
|
+
return key, n, (stamp or None)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def page_block(limit: int, returned: int, returned_so_far: int,
|
|
197
|
+
has_more: bool, next_cursor: Optional[str],
|
|
198
|
+
exhaustive: bool) -> Dict[str, Any]:
|
|
199
|
+
"""The `page` block every paged result carries (D4 3.2.7).
|
|
200
|
+
|
|
201
|
+
`returned`, `has_more` and `exhaustive` are computed on this page.
|
|
202
|
+
`returned_so_far` is this page's `returned` added to the count the
|
|
203
|
+
CURSOR carried, so on any page but the first it rests on a value
|
|
204
|
+
the caller supplied and nothing here can check. It is bounded at
|
|
205
|
+
decode (CURSOR_MAX_RETURNED_SO_FAR) so a tampered cursor cannot put
|
|
206
|
+
an arbitrary integer in a result, and PRIVACY.md says plainly where
|
|
207
|
+
the number comes from; a stronger guarantee would mean signing
|
|
208
|
+
cursors, which D4 3.11 deliberately does not do, because a caller
|
|
209
|
+
who forges a POSITION only gets a page it could have asked for.
|
|
210
|
+
"""
|
|
211
|
+
return {
|
|
212
|
+
"limit": limit,
|
|
213
|
+
"returned": returned,
|
|
214
|
+
"returned_so_far": returned_so_far,
|
|
215
|
+
"has_more": has_more,
|
|
216
|
+
"next_cursor": next_cursor,
|
|
217
|
+
"exhaustive": exhaustive,
|
|
218
|
+
}
|