mdcx 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mdcx/__init__.py +41 -0
- mdcx/archive.py +574 -0
- mdcx/cli.py +422 -0
- mdcx/convert/__init__.py +15 -0
- mdcx/convert/chapters.py +143 -0
- mdcx/convert/compact.py +152 -0
- mdcx/convert/convert.py +358 -0
- mdcx/convert/engines.py +305 -0
- mdcx/convert/extract.py +137 -0
- mdcx/convert/index.py +426 -0
- mdcx/convert/paths.py +202 -0
- mdcx/convert/pdf.py +186 -0
- mdcx/convert/verify.py +103 -0
- mdcx/mcp_server.py +191 -0
- mdcx/search.py +404 -0
- mdcx-1.0.0.dist-info/METADATA +266 -0
- mdcx-1.0.0.dist-info/RECORD +21 -0
- mdcx-1.0.0.dist-info/WHEEL +4 -0
- mdcx-1.0.0.dist-info/entry_points.txt +4 -0
- mdcx-1.0.0.dist-info/licenses/LICENSE +202 -0
- mdcx-1.0.0.dist-info/licenses/NOTICE +33 -0
mdcx/__init__.py
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# Copyright 2026 Jorge Ellena G.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""Convert document collections to verified Markdown and make them queryable.
|
|
16
|
+
|
|
17
|
+
The package covers three stages:
|
|
18
|
+
|
|
19
|
+
Conversion
|
|
20
|
+
Each document is converted to Markdown and checked against the text the
|
|
21
|
+
original actually exposes, read with a library independent from the engine
|
|
22
|
+
that performed the conversion. Content the structured engine omits is
|
|
23
|
+
appended verbatim rather than reported as lost.
|
|
24
|
+
|
|
25
|
+
Packaging
|
|
26
|
+
The resulting corpus, its search index and the provenance of every passage
|
|
27
|
+
fit into a single encrypted ``.mdcx`` file. Its header can be read without
|
|
28
|
+
the key, so the issuer and the integrity of a file can be verified before
|
|
29
|
+
deciding to open it.
|
|
30
|
+
|
|
31
|
+
Retrieval
|
|
32
|
+
A query returns the passages that answer it, each with its exact source, in
|
|
33
|
+
milliseconds and without sending the whole collection through a model's
|
|
34
|
+
context window.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
__version__ = "1.0.0"
|
|
38
|
+
|
|
39
|
+
from . import archive, search # noqa: F401
|
|
40
|
+
|
|
41
|
+
__all__ = ["archive", "search", "__version__"]
|
mdcx/archive.py
ADDED
|
@@ -0,0 +1,574 @@
|
|
|
1
|
+
# Copyright 2026 Jorge Ellena G.
|
|
2
|
+
#
|
|
3
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
# you may not use this file except in compliance with the License.
|
|
5
|
+
# You may obtain a copy of the License at
|
|
6
|
+
#
|
|
7
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
#
|
|
9
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
# See the License for the specific language governing permissions and
|
|
13
|
+
# limitations under the License.
|
|
14
|
+
|
|
15
|
+
"""The .mdcx container format.
|
|
16
|
+
|
|
17
|
+
A converted corpus, its search index and the provenance of every passage
|
|
18
|
+
held in a single encrypted file.
|
|
19
|
+
|
|
20
|
+
The header is stored in clear text so that the issuer, the version and the
|
|
21
|
+
integrity of a file can be checked without the key, which is what is needed
|
|
22
|
+
to decide whether to open it. The body is encrypted with AES-256-GCM, which
|
|
23
|
+
authenticates as well as conceals: altering one byte makes decryption fail
|
|
24
|
+
rather than return corrupted data. The key is derived from a passphrase with
|
|
25
|
+
scrypt.
|
|
26
|
+
|
|
27
|
+
Inside the body, a SQLite database with an FTS5 index provides retrieval
|
|
28
|
+
without any external service.
|
|
29
|
+
|
|
30
|
+
The format encrypts at rest and decrypts in memory when opened. This is not
|
|
31
|
+
searchable encryption, where data is queried without ever being decrypted;
|
|
32
|
+
that is a separate field with documented leakage attacks and per-query costs
|
|
33
|
+
measured in seconds.
|
|
34
|
+
|
|
35
|
+
python -m mdcx.archive pack --output ./corpus_md --target corpus.mdcx --key "..."
|
|
36
|
+
python -m mdcx.archive info corpus.mdcx
|
|
37
|
+
python -m mdcx.archive search corpus.mdcx "a question" --key "..."
|
|
38
|
+
python -m mdcx.archive export corpus.mdcx --target ./restored --key "..."
|
|
39
|
+
"""
|
|
40
|
+
from __future__ import annotations
|
|
41
|
+
|
|
42
|
+
import argparse
|
|
43
|
+
import hashlib
|
|
44
|
+
import json
|
|
45
|
+
import math
|
|
46
|
+
import os
|
|
47
|
+
import sqlite3
|
|
48
|
+
import struct
|
|
49
|
+
import sys
|
|
50
|
+
import time
|
|
51
|
+
from pathlib import Path
|
|
52
|
+
|
|
53
|
+
MAGIC = b"MDCX"
|
|
54
|
+
VERSION = 1
|
|
55
|
+
|
|
56
|
+
SCRYPT_N = 2 ** 15
|
|
57
|
+
SCRYPT_R = 8
|
|
58
|
+
SCRYPT_P = 1
|
|
59
|
+
KEY_BYTES = 32
|
|
60
|
+
|
|
61
|
+
def _derive_key(key: str, salt: bytes) -> bytes:
|
|
62
|
+
memory = 128 * SCRYPT_N * SCRYPT_R
|
|
63
|
+
return hashlib.scrypt(key.encode("utf-8"), salt=salt,
|
|
64
|
+
n=SCRYPT_N, r=SCRYPT_R, p=SCRYPT_P, dklen=KEY_BYTES,
|
|
65
|
+
maxmem=memory * 2)
|
|
66
|
+
|
|
67
|
+
def _encrypt(datos: bytes, clave_derivada: bytes) -> tuple[bytes, bytes]:
|
|
68
|
+
from cryptography.hazmat.primitives.ciphers.aead import AESGCM
|
|
69
|
+
nonce = os.urandom(12)
|
|
70
|
+
return nonce, AESGCM(clave_derivada).encrypt(nonce, datos, None)
|
|
71
|
+
|
|
72
|
+
def _decrypt(body: bytes, clave_derivada: bytes, nonce: bytes) -> bytes:
|
|
73
|
+
from cryptography.hazmat.primitives.ciphers.aead import AESGCM
|
|
74
|
+
return AESGCM(clave_derivada).decrypt(nonce, body, None)
|
|
75
|
+
|
|
76
|
+
def _build_database(folder: Path) -> tuple[bytes, dict]:
|
|
77
|
+
"""Build the in-memory database with documents, index and provenance."""
|
|
78
|
+
from . import search as B
|
|
79
|
+
|
|
80
|
+
docs = B.load_documents(folder)
|
|
81
|
+
connection = sqlite3.connect(":memory:")
|
|
82
|
+
connection.executescript("""
|
|
83
|
+
PRAGMA journal_mode = OFF;
|
|
84
|
+
CREATE TABLE document (
|
|
85
|
+
id INTEGER PRIMARY KEY,
|
|
86
|
+
name TEXT NOT NULL,
|
|
87
|
+
pseudopath TEXT NOT NULL,
|
|
88
|
+
source TEXT NOT NULL,
|
|
89
|
+
folder TEXT,
|
|
90
|
+
archive TEXT,
|
|
91
|
+
verification_status TEXT,
|
|
92
|
+
-- Normalised text of the whole document. Literal matching runs here rather
|
|
93
|
+
-- than over passages, because a quoted phrase often crosses the boundary
|
|
94
|
+
-- between paragraphs.
|
|
95
|
+
normalized_text TEXT
|
|
96
|
+
);
|
|
97
|
+
CREATE TABLE passage (
|
|
98
|
+
id INTEGER PRIMARY KEY,
|
|
99
|
+
documento_id INTEGER NOT NULL REFERENCES document(id),
|
|
100
|
+
position INTEGER NOT NULL,
|
|
101
|
+
text TEXT NOT NULL
|
|
102
|
+
);
|
|
103
|
+
-- The index is declared external to the content so the text is not stored
|
|
104
|
+
-- twice: FTS5 indexes what lives in the passage table.
|
|
105
|
+
CREATE VIRTUAL TABLE passage_fts USING fts5(
|
|
106
|
+
text, content='passage', content_rowid='id', tokenize='unicode61'
|
|
107
|
+
);
|
|
108
|
+
CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT);
|
|
109
|
+
-- Document frequency per term. FTS5 holds this internally but does not
|
|
110
|
+
-- expose it usably, and without it the package cannot rank by the same
|
|
111
|
+
-- criterion as the folder-based search.
|
|
112
|
+
CREATE TABLE df (term TEXT PRIMARY KEY, passages INTEGER NOT NULL);
|
|
113
|
+
""")
|
|
114
|
+
|
|
115
|
+
n_passages = 0
|
|
116
|
+
for i, d in enumerate(docs, 1):
|
|
117
|
+
text = d["text"]
|
|
118
|
+
archive = ""
|
|
119
|
+
status = ""
|
|
120
|
+
for line in text.splitlines()[:12]:
|
|
121
|
+
if line.startswith("source_format:"):
|
|
122
|
+
archive = line.split(":", 1)[1].strip()
|
|
123
|
+
elif line.startswith("verification_status:"):
|
|
124
|
+
status = line.split(":", 1)[1].strip()
|
|
125
|
+
connection.execute(
|
|
126
|
+
"INSERT INTO document VALUES (?,?,?,?,?,?,?,?)",
|
|
127
|
+
(i, d["name"], d["pseudopath"], d["source"], d["folder"], archive, status,
|
|
128
|
+
d["norm"]))
|
|
129
|
+
for j, bloque in enumerate(d["blocks"] if "blocks" in d else _split_blocks(text)):
|
|
130
|
+
if not bloque.strip():
|
|
131
|
+
continue
|
|
132
|
+
n_passages += 1
|
|
133
|
+
connection.execute("INSERT INTO passage VALUES (?,?,?,?)",
|
|
134
|
+
(n_passages, i, j, bloque))
|
|
135
|
+
|
|
136
|
+
connection.execute("INSERT INTO passage_fts(passage_fts) VALUES('rebuild')")
|
|
137
|
+
|
|
138
|
+
from . import search as _B
|
|
139
|
+
from collections import Counter as _Counter
|
|
140
|
+
|
|
141
|
+
df_count: _Counter = _Counter()
|
|
142
|
+
lengths: list[int] = []
|
|
143
|
+
for (text,) in connection.execute("SELECT text FROM passage"):
|
|
144
|
+
tk = _B._TOKEN_RE.findall(_B._normalize(text))
|
|
145
|
+
lengths.append(len(tk))
|
|
146
|
+
for t in set(tk):
|
|
147
|
+
if len(t) > 2:
|
|
148
|
+
df_count[t] += 1
|
|
149
|
+
connection.executemany("INSERT INTO df VALUES (?,?)", df_count.items())
|
|
150
|
+
avg_length = sum(lengths) / len(lengths) if lengths else 60.0
|
|
151
|
+
|
|
152
|
+
summary = {
|
|
153
|
+
"documents": len(docs),
|
|
154
|
+
"passages": n_passages,
|
|
155
|
+
"created_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
|
156
|
+
"source_folder": folder.name,
|
|
157
|
+
"largo_medio_pasaje": round(avg_length, 2),
|
|
158
|
+
"terminos_indexados": len(df_count),
|
|
159
|
+
}
|
|
160
|
+
manifiesto = folder / "_manifest.json"
|
|
161
|
+
if manifiesto.exists():
|
|
162
|
+
try:
|
|
163
|
+
m = json.loads(manifiesto.read_text(encoding="utf-8"))
|
|
164
|
+
summary["conversion"] = m.get("summary", {})
|
|
165
|
+
except Exception: # noqa: BLE001
|
|
166
|
+
pass
|
|
167
|
+
for k, v in summary.items():
|
|
168
|
+
connection.execute("INSERT INTO meta VALUES (?,?)",
|
|
169
|
+
(k, json.dumps(v) if not isinstance(v, str) else v))
|
|
170
|
+
connection.commit()
|
|
171
|
+
|
|
172
|
+
datos = connection.serialize()
|
|
173
|
+
connection.close()
|
|
174
|
+
return bytes(datos), summary
|
|
175
|
+
|
|
176
|
+
def _split_blocks(text: str) -> list[str]:
|
|
177
|
+
return [b for b in text.split("\n\n") if b.strip()]
|
|
178
|
+
|
|
179
|
+
def generate_signing_key() -> tuple[str, str]:
|
|
180
|
+
"""Create an Ed25519 key pair and return it as (private, public) hex strings.
|
|
181
|
+
|
|
182
|
+
The private key signs packages; the public key lets anyone verify who issued
|
|
183
|
+
one. Only the public half is meant to be distributed.
|
|
184
|
+
"""
|
|
185
|
+
from cryptography.hazmat.primitives.asymmetric import ed25519
|
|
186
|
+
|
|
187
|
+
private = ed25519.Ed25519PrivateKey.generate()
|
|
188
|
+
return (private.private_bytes_raw().hex(),
|
|
189
|
+
private.public_key().public_bytes_raw().hex())
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def _sign(digest: str, signing_key: str) -> str:
|
|
193
|
+
from cryptography.hazmat.primitives.asymmetric import ed25519
|
|
194
|
+
|
|
195
|
+
key = ed25519.Ed25519PrivateKey.from_private_bytes(bytes.fromhex(signing_key))
|
|
196
|
+
return key.sign(digest.encode("ascii")).hex()
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def verify_signature(path: Path, public_key: str) -> bool:
|
|
200
|
+
"""Report whether a package was signed by the holder of this public key.
|
|
201
|
+
|
|
202
|
+
The signature covers the digest of the encrypted body, so it attests both the
|
|
203
|
+
issuer and the content: altering either invalidates it. Verification needs
|
|
204
|
+
neither the encryption key nor the contents of the package.
|
|
205
|
+
"""
|
|
206
|
+
from cryptography.exceptions import InvalidSignature
|
|
207
|
+
from cryptography.hazmat.primitives.asymmetric import ed25519
|
|
208
|
+
|
|
209
|
+
header = read_header(path)
|
|
210
|
+
signature = header.get("signature")
|
|
211
|
+
if not signature:
|
|
212
|
+
return False
|
|
213
|
+
# The signature covers the digest recorded in the header, so it must be checked
|
|
214
|
+
# together with the integrity of the body. Verifying the signature alone would
|
|
215
|
+
# accept a package whose body had been replaced while the header was left intact:
|
|
216
|
+
# the stored digest would still match the signature, and the content would not.
|
|
217
|
+
if not header.get("_intact"):
|
|
218
|
+
return False
|
|
219
|
+
try:
|
|
220
|
+
key = ed25519.Ed25519PublicKey.from_public_bytes(bytes.fromhex(public_key))
|
|
221
|
+
key.verify(bytes.fromhex(signature), header["body_digest"].encode("ascii"))
|
|
222
|
+
except (InvalidSignature, ValueError):
|
|
223
|
+
return False
|
|
224
|
+
return True
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def pack(folder: Path, target: Path, key: str, issuer: str = "",
|
|
228
|
+
signing_key: str = "") -> dict:
|
|
229
|
+
"""Write the .mdcx file and return its figures."""
|
|
230
|
+
import lzma
|
|
231
|
+
|
|
232
|
+
t0 = time.perf_counter()
|
|
233
|
+
base_score, summary = _build_database(folder)
|
|
234
|
+
t_base = time.perf_counter() - t0
|
|
235
|
+
|
|
236
|
+
t0 = time.perf_counter()
|
|
237
|
+
compressed = lzma.compress(base_score, preset=6)
|
|
238
|
+
t_comp = time.perf_counter() - t0
|
|
239
|
+
|
|
240
|
+
salt = os.urandom(16)
|
|
241
|
+
t0 = time.perf_counter()
|
|
242
|
+
derived_key = _derive_key(key, salt)
|
|
243
|
+
nonce, body = _encrypt(compressed, derived_key)
|
|
244
|
+
t_cifrado = time.perf_counter() - t0
|
|
245
|
+
|
|
246
|
+
header = {
|
|
247
|
+
"file_format": "mdcx",
|
|
248
|
+
"version": VERSION,
|
|
249
|
+
"issuer": issuer,
|
|
250
|
+
"created_utc": summary["created_utc"],
|
|
251
|
+
"documents": summary["documents"],
|
|
252
|
+
"passages": summary["passages"],
|
|
253
|
+
"encryption": "AES-256-GCM",
|
|
254
|
+
"key_derivation": {"algorithm": "scrypt", "n": SCRYPT_N, "r": SCRYPT_R, "p": SCRYPT_P},
|
|
255
|
+
"compression": "lzma",
|
|
256
|
+
"salt": salt.hex(),
|
|
257
|
+
"nonce": nonce.hex(),
|
|
258
|
+
"body_digest": hashlib.sha256(body).hexdigest(),
|
|
259
|
+
"signature": "",
|
|
260
|
+
"public_key": "",
|
|
261
|
+
"conversion": summary.get("conversion", {}),
|
|
262
|
+
}
|
|
263
|
+
if signing_key:
|
|
264
|
+
from cryptography.hazmat.primitives.asymmetric import ed25519
|
|
265
|
+
|
|
266
|
+
private = ed25519.Ed25519PrivateKey.from_private_bytes(bytes.fromhex(signing_key))
|
|
267
|
+
header["signature"] = _sign(header["body_digest"], signing_key)
|
|
268
|
+
header["public_key"] = private.public_key().public_bytes_raw().hex()
|
|
269
|
+
|
|
270
|
+
encoded_header = json.dumps(header, ensure_ascii=False).encode("utf-8")
|
|
271
|
+
|
|
272
|
+
with open(target, "wb") as f:
|
|
273
|
+
f.write(MAGIC)
|
|
274
|
+
f.write(struct.pack("<I", len(encoded_header)))
|
|
275
|
+
f.write(encoded_header)
|
|
276
|
+
f.write(body)
|
|
277
|
+
|
|
278
|
+
return {
|
|
279
|
+
"bytes_database": len(base_score),
|
|
280
|
+
"bytes_compressed": len(compressed),
|
|
281
|
+
"bytes_file": target.stat().st_size,
|
|
282
|
+
"seconds_index": round(t_base, 2),
|
|
283
|
+
"seconds_compress": round(t_comp, 2),
|
|
284
|
+
"seconds_encrypt": round(t_cifrado, 2),
|
|
285
|
+
**summary,
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
def read_header(path: Path) -> dict:
|
|
289
|
+
"""Header and integrity status, without requiring the key."""
|
|
290
|
+
with open(path, "rb") as f:
|
|
291
|
+
if f.read(4) != MAGIC:
|
|
292
|
+
raise ValueError("not an .mdcx file")
|
|
293
|
+
(n,) = struct.unpack("<I", f.read(4))
|
|
294
|
+
header = json.loads(f.read(n).decode("utf-8"))
|
|
295
|
+
body = f.read()
|
|
296
|
+
header["_intact"] = hashlib.sha256(body).hexdigest() == header.get("body_digest")
|
|
297
|
+
header["_signed"] = bool(header.get("signature"))
|
|
298
|
+
header["_body_bytes"] = len(body)
|
|
299
|
+
return header
|
|
300
|
+
|
|
301
|
+
def open_package(path: Path, key: str) -> tuple[sqlite3.Connection, dict]:
|
|
302
|
+
"""Decrypt in memory and return a connection ready to query.
|
|
303
|
+
|
|
304
|
+
Nothing is written to disk, so an open package leaves no plaintext copy."""
|
|
305
|
+
import lzma
|
|
306
|
+
|
|
307
|
+
with open(path, "rb") as f:
|
|
308
|
+
if f.read(4) != MAGIC:
|
|
309
|
+
raise ValueError("not an .mdcx file")
|
|
310
|
+
(n,) = struct.unpack("<I", f.read(4))
|
|
311
|
+
header = json.loads(f.read(n).decode("utf-8"))
|
|
312
|
+
body = f.read()
|
|
313
|
+
|
|
314
|
+
if hashlib.sha256(body).hexdigest() != header.get("body_digest"):
|
|
315
|
+
raise ValueError("file has been altered: body digest does not match")
|
|
316
|
+
|
|
317
|
+
derived_key = _derive_key(key, bytes.fromhex(header["salt"]))
|
|
318
|
+
try:
|
|
319
|
+
compressed = _decrypt(body, derived_key, bytes.fromhex(header["nonce"]))
|
|
320
|
+
except Exception as exc: # noqa: BLE001
|
|
321
|
+
raise ValueError("incorrect key or corrupted file") from exc
|
|
322
|
+
|
|
323
|
+
header["_intact"] = True
|
|
324
|
+
header["_body_bytes"] = len(body)
|
|
325
|
+
|
|
326
|
+
connection = sqlite3.connect(":memory:")
|
|
327
|
+
connection.deserialize(lzma.decompress(compressed))
|
|
328
|
+
return connection, header
|
|
329
|
+
|
|
330
|
+
_SQL_BASE = """
|
|
331
|
+
SELECT d.name, d.pseudopath, d.source, p.text, bm25(passage_fts) AS score
|
|
332
|
+
FROM passage_fts
|
|
333
|
+
JOIN passage p ON p.id = passage_fts.rowid
|
|
334
|
+
JOIN document d ON d.id = p.documento_id
|
|
335
|
+
WHERE passage_fts MATCH ?
|
|
336
|
+
"""
|
|
337
|
+
|
|
338
|
+
def _run_match(connection: sqlite3.Connection, expr: str, limit: int,
|
|
339
|
+
only: str | None) -> list[dict]:
|
|
340
|
+
sql = _SQL_BASE
|
|
341
|
+
params: list = [expr]
|
|
342
|
+
if only:
|
|
343
|
+
sql += " AND d.source = ?"
|
|
344
|
+
params.append(only.upper())
|
|
345
|
+
sql += " ORDER BY score LIMIT ?"
|
|
346
|
+
params.append(limit)
|
|
347
|
+
try:
|
|
348
|
+
rows = connection.execute(sql, params).fetchall()
|
|
349
|
+
except sqlite3.OperationalError:
|
|
350
|
+
return []
|
|
351
|
+
return [{"document": r[0], "pseudopath": r[1], "source": r[2],
|
|
352
|
+
"passage": r[3], "score": round(-r[4], 3)}
|
|
353
|
+
for r in rows]
|
|
354
|
+
|
|
355
|
+
K1 = 1.5
|
|
356
|
+
B_LENGTH = 0.45
|
|
357
|
+
DOC_TOP_PASSAGES = 8
|
|
358
|
+
|
|
359
|
+
CANDIDATES = 1200
|
|
360
|
+
|
|
361
|
+
def query(connection: sqlite3.Connection, query_text: str, limit: int = 8,
|
|
362
|
+
only: str | None = None) -> list[dict]:
|
|
363
|
+
"""Resolve a query, ranking by document rather than by isolated passage."""
|
|
364
|
+
from . import search as B
|
|
365
|
+
|
|
366
|
+
phrase = query_text.strip().split(".")[0][:160].strip()
|
|
367
|
+
effective = phrase if len(phrase.split()) >= 5 else query_text
|
|
368
|
+
|
|
369
|
+
terms = [t for t in B._TOKEN_RE.findall(B._normalize(effective)) if len(t) > 2]
|
|
370
|
+
terms = B.expand_terms(terms)
|
|
371
|
+
if not terms:
|
|
372
|
+
return []
|
|
373
|
+
distinct_terms = set(terms)
|
|
374
|
+
|
|
375
|
+
expr = " OR ".join(f'"{t}"' for t in distinct_terms)
|
|
376
|
+
candidates = _run_match(connection, expr, CANDIDATES, only)
|
|
377
|
+
if not candidates:
|
|
378
|
+
return []
|
|
379
|
+
|
|
380
|
+
df, n_passages, avg_length = _corpus_statistics(connection)
|
|
381
|
+
|
|
382
|
+
by_document: dict[str, list[dict]] = {}
|
|
383
|
+
for r in candidates:
|
|
384
|
+
frec = _term_frequencies(r["passage"], distinct_terms)
|
|
385
|
+
if not frec:
|
|
386
|
+
continue
|
|
387
|
+
largo = max(len(B._TOKEN_RE.findall(B._normalize(r["passage"]))), 1)
|
|
388
|
+
score = 0.0
|
|
389
|
+
for t, f in frec.items():
|
|
390
|
+
d_t = df.get(t, 1)
|
|
391
|
+
idf = math.log(1 + (n_passages - d_t + 0.5) / (d_t + 0.5))
|
|
392
|
+
score += idf * (f * (K1 + 1)) / (
|
|
393
|
+
f + K1 * (1 - B_LENGTH + B_LENGTH * largo / avg_length))
|
|
394
|
+
r = dict(r)
|
|
395
|
+
r["score"] = round(score, 3)
|
|
396
|
+
r["terms"] = sorted(frec)
|
|
397
|
+
by_document.setdefault(r["document"], []).append(r)
|
|
398
|
+
|
|
399
|
+
ranking = []
|
|
400
|
+
for name, passages in by_document.items():
|
|
401
|
+
passages.sort(key=lambda x: -x["score"])
|
|
402
|
+
base_score = sum(x["score"] for x in passages[:DOC_TOP_PASSAGES])
|
|
403
|
+
coverage = max(len(x["terms"]) for x in passages) / max(len(distinct_terms), 1)
|
|
404
|
+
ranking.append((base_score * coverage, passages))
|
|
405
|
+
ranking.sort(key=lambda par: -par[0])
|
|
406
|
+
|
|
407
|
+
if len(phrase.split()) >= 5:
|
|
408
|
+
needle = B._normalize(phrase)
|
|
409
|
+
preferred = [n for (n, t) in connection.execute(
|
|
410
|
+
"SELECT name, normalized_text FROM document") if t and needle in t]
|
|
411
|
+
if preferred:
|
|
412
|
+
position = {d: i for i, d in enumerate(preferred)}
|
|
413
|
+
ranking.sort(key=lambda par: (position.get(par[1][0]["document"], len(position)),
|
|
414
|
+
-par[0]))
|
|
415
|
+
|
|
416
|
+
out: list[dict] = []
|
|
417
|
+
for round_index in range(DOC_TOP_PASSAGES):
|
|
418
|
+
for score, passages in ranking:
|
|
419
|
+
if round_index < len(passages):
|
|
420
|
+
r = dict(passages[round_index])
|
|
421
|
+
r["score_documento"] = round(score, 3)
|
|
422
|
+
out.append(r)
|
|
423
|
+
if len(out) >= limit:
|
|
424
|
+
return out
|
|
425
|
+
return out[:limit]
|
|
426
|
+
|
|
427
|
+
def _term_frequencies(text: str, terms: set[str]) -> dict[str, int]:
|
|
428
|
+
from . import search as B
|
|
429
|
+
|
|
430
|
+
cuenta: dict[str, int] = {}
|
|
431
|
+
for t in B._TOKEN_RE.findall(B._normalize(text)):
|
|
432
|
+
if t in terms:
|
|
433
|
+
cuenta[t] = cuenta.get(t, 0) + 1
|
|
434
|
+
return cuenta
|
|
435
|
+
|
|
436
|
+
_STATS_CACHE: dict[int, tuple] = {}
|
|
437
|
+
|
|
438
|
+
def _corpus_statistics(connection: sqlite3.Connection) -> tuple[dict, int, float]:
|
|
439
|
+
"""Document frequency per term and mean passage length, as packed."""
|
|
440
|
+
key = id(connection)
|
|
441
|
+
if key not in _STATS_CACHE:
|
|
442
|
+
df = {t: n for t, n in connection.execute("SELECT term, passages FROM df")}
|
|
443
|
+
row = connection.execute("SELECT value FROM meta WHERE key='passages'").fetchone()
|
|
444
|
+
n = int(json.loads(row[0])) if row else max(len(df), 1)
|
|
445
|
+
row = connection.execute(
|
|
446
|
+
"SELECT value FROM meta WHERE key='largo_medio_pasaje'").fetchone()
|
|
447
|
+
lm = float(json.loads(row[0])) if row else 60.0
|
|
448
|
+
_STATS_CACHE[key] = (df, n, lm)
|
|
449
|
+
return _STATS_CACHE[key]
|
|
450
|
+
|
|
451
|
+
def export(path: Path, key: str, target: Path) -> dict:
|
|
452
|
+
"""Rebuild the Markdown folder from the package.
|
|
453
|
+
|
|
454
|
+
Directory structure is restored from the pseudopath stored with each document."""
|
|
455
|
+
connection, header = open_package(path, key)
|
|
456
|
+
try:
|
|
457
|
+
rows = connection.execute(
|
|
458
|
+
"SELECT d.pseudopath, d.name, group_concat(p.text, char(10) || char(10)) "
|
|
459
|
+
"FROM document d JOIN passage p ON p.documento_id = d.id "
|
|
460
|
+
"GROUP BY d.id ORDER BY p.position").fetchall()
|
|
461
|
+
written = 0
|
|
462
|
+
for pseudopath, name, text in rows:
|
|
463
|
+
relative = pseudopath[2:] if pseudopath.startswith("@/") else pseudopath
|
|
464
|
+
path = target / relative
|
|
465
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
466
|
+
path.write_text((text or "") + "\n", encoding="utf-8")
|
|
467
|
+
written += 1
|
|
468
|
+
finally:
|
|
469
|
+
connection.close()
|
|
470
|
+
return {"documents": written, "target": str(target),
|
|
471
|
+
"created_utc": header.get("created_utc")}
|
|
472
|
+
|
|
473
|
+
def _direction(value: str) -> str:
|
|
474
|
+
"""Human-readable direction label."""
|
|
475
|
+
return {"SENT": "SENT", "RECEIVED": "RECEIVED",
|
|
476
|
+
"EMITIDO": "SENT", "RECIBIDO": "RECEIVED"}.get(value, "OTHER")
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def main() -> int:
|
|
480
|
+
try:
|
|
481
|
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
482
|
+
except Exception:
|
|
483
|
+
pass
|
|
484
|
+
|
|
485
|
+
ap = argparse.ArgumentParser(description="The .mdcx format: an indexed, encrypted, portable corpus")
|
|
486
|
+
sub = ap.add_subparsers(dest="action", required=True)
|
|
487
|
+
|
|
488
|
+
e = sub.add_parser("pack")
|
|
489
|
+
e.add_argument("--output", default="Output")
|
|
490
|
+
e.add_argument("--target", default="corpus.mdcx")
|
|
491
|
+
e.add_argument("--key", required=True)
|
|
492
|
+
e.add_argument("--issuer", default="")
|
|
493
|
+
e.add_argument("--signing-key", default="",
|
|
494
|
+
help="hex private key to sign the package with")
|
|
495
|
+
|
|
496
|
+
k = sub.add_parser("keygen")
|
|
497
|
+
|
|
498
|
+
v = sub.add_parser("verify")
|
|
499
|
+
v.add_argument("path")
|
|
500
|
+
v.add_argument("--public-key", required=True)
|
|
501
|
+
|
|
502
|
+
i = sub.add_parser("info")
|
|
503
|
+
i.add_argument("path")
|
|
504
|
+
|
|
505
|
+
x = sub.add_parser("export")
|
|
506
|
+
x.add_argument("path")
|
|
507
|
+
x.add_argument("--target", required=True)
|
|
508
|
+
x.add_argument("--key", required=True)
|
|
509
|
+
|
|
510
|
+
b = sub.add_parser("search")
|
|
511
|
+
b.add_argument("path")
|
|
512
|
+
b.add_argument("query_text")
|
|
513
|
+
b.add_argument("--key", required=True)
|
|
514
|
+
b.add_argument("--limit", type=int, default=5)
|
|
515
|
+
b.add_argument("--only", choices=["received", "sent"])
|
|
516
|
+
|
|
517
|
+
args = ap.parse_args()
|
|
518
|
+
|
|
519
|
+
if args.action == "pack":
|
|
520
|
+
r = pack(Path(args.output), Path(args.target), args.key, args.issuer,
|
|
521
|
+
args.signing_key)
|
|
522
|
+
print(f"Packed: {args.target}")
|
|
523
|
+
print(f" documents {r['documents']} passages {r['passages']}")
|
|
524
|
+
print(f" database {r['bytes_database']:,} -> compressed {r['bytes_compressed']:,} "
|
|
525
|
+
f"-> file {r['bytes_file']:,} bytes".replace(",", "."))
|
|
526
|
+
print(f" index {r['seconds_index']}s compress {r['seconds_compress']}s "
|
|
527
|
+
f"encrypt {r['seconds_encrypt']}s")
|
|
528
|
+
return 0
|
|
529
|
+
|
|
530
|
+
if args.action == "keygen":
|
|
531
|
+
private, public = generate_signing_key()
|
|
532
|
+
print("Keep the private key secret; distribute only the public key.")
|
|
533
|
+
print(f" private: {private}")
|
|
534
|
+
print(f" public : {public}")
|
|
535
|
+
return 0
|
|
536
|
+
|
|
537
|
+
if args.action == "verify":
|
|
538
|
+
valid = verify_signature(Path(args.path), args.public_key)
|
|
539
|
+
print("Signature valid: the package was issued by the holder of this key "
|
|
540
|
+
"and has not been altered." if valid else
|
|
541
|
+
"Signature not valid: unsigned package, different key, or altered content.")
|
|
542
|
+
return 0 if valid else 1
|
|
543
|
+
|
|
544
|
+
if args.action == "info":
|
|
545
|
+
header = read_header(Path(args.path))
|
|
546
|
+
print(f"Format : {header['file_format']} v{header['version']}")
|
|
547
|
+
print(f"Issuer : {header.get('issuer') or '(not declared)'}")
|
|
548
|
+
print(f"Created : {header['created_utc']}")
|
|
549
|
+
print(f"Content : {header['documents']} documents, {header['passages']} passages")
|
|
550
|
+
print(f"Encryption: {header['encryption']} with {header['key_derivation']['algorithm']}")
|
|
551
|
+
print(f"Integrity : {'intact' if header['_intact'] else 'ALTERED'}")
|
|
552
|
+
print(f"Signature : {'present' if header.get('_signed') else 'none'}"
|
|
553
|
+
+ (f" (public key {header['public_key'][:16]}...)" if header.get('public_key') else ""))
|
|
554
|
+
if header.get("conversion"):
|
|
555
|
+
print(f"Conversion: {json.dumps(header['conversion'], ensure_ascii=False)[:200]}")
|
|
556
|
+
return 0
|
|
557
|
+
|
|
558
|
+
if args.action == "export":
|
|
559
|
+
r = export(Path(args.path), args.key, Path(args.target))
|
|
560
|
+
print(f"Exported {r['documents']} documents to {r['target']}")
|
|
561
|
+
return 0
|
|
562
|
+
|
|
563
|
+
connection, header = open_package(Path(args.path), args.key)
|
|
564
|
+
results = query(connection, args.query_text, args.limit, args.only)
|
|
565
|
+
print(f"{len(results)} passage(s)\n")
|
|
566
|
+
for r in results:
|
|
567
|
+
print("-" * 96)
|
|
568
|
+
print(f"[{r['source']}] {r['document']} [score {r['score']}]")
|
|
569
|
+
print(f"{r['pseudopath']}")
|
|
570
|
+
print(r["passage"][:1200])
|
|
571
|
+
return 0
|
|
572
|
+
|
|
573
|
+
if __name__ == "__main__":
|
|
574
|
+
raise SystemExit(main())
|