mdcx 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mdcx/__init__.py ADDED
@@ -0,0 +1,41 @@
1
+ # Copyright 2026 Jorge Ellena G.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """Convert document collections to verified Markdown and make them queryable.
16
+
17
+ The package covers three stages:
18
+
19
+ Conversion
20
+ Each document is converted to Markdown and checked against the text the
21
+ original actually exposes, read with a library independent from the engine
22
+ that performed the conversion. Content the structured engine omits is
23
+ appended verbatim rather than reported as lost.
24
+
25
+ Packaging
26
+ The resulting corpus, its search index and the provenance of every passage
27
+ fit into a single encrypted ``.mdcx`` file. Its header can be read without
28
+ the key, so the issuer and the integrity of a file can be verified before
29
+ deciding to open it.
30
+
31
+ Retrieval
32
+ A query returns the passages that answer it, each with its exact source, in
33
+ milliseconds and without sending the whole collection through a model's
34
+ context window.
35
+ """
36
+
37
+ __version__ = "1.0.0"
38
+
39
+ from . import archive, search # noqa: F401
40
+
41
+ __all__ = ["archive", "search", "__version__"]
mdcx/archive.py ADDED
@@ -0,0 +1,574 @@
1
+ # Copyright 2026 Jorge Ellena G.
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """The .mdcx container format.
16
+
17
+ A converted corpus, its search index and the provenance of every passage
18
+ held in a single encrypted file.
19
+
20
+ The header is stored in clear text so that the issuer, the version and the
21
+ integrity of a file can be checked without the key, which is what is needed
22
+ to decide whether to open it. The body is encrypted with AES-256-GCM, which
23
+ authenticates as well as conceals: altering one byte makes decryption fail
24
+ rather than return corrupted data. The key is derived from a passphrase with
25
+ scrypt.
26
+
27
+ Inside the body, a SQLite database with an FTS5 index provides retrieval
28
+ without any external service.
29
+
30
+ The format encrypts at rest and decrypts in memory when opened. This is not
31
+ searchable encryption, where data is queried without ever being decrypted;
32
+ that is a separate field with documented leakage attacks and per-query costs
33
+ measured in seconds.
34
+
35
+ python -m mdcx.archive pack --output ./corpus_md --target corpus.mdcx --key "..."
36
+ python -m mdcx.archive info corpus.mdcx
37
+ python -m mdcx.archive search corpus.mdcx "a question" --key "..."
38
+ python -m mdcx.archive export corpus.mdcx --target ./restored --key "..."
39
+ """
40
+ from __future__ import annotations
41
+
42
+ import argparse
43
+ import hashlib
44
+ import json
45
+ import math
46
+ import os
47
+ import sqlite3
48
+ import struct
49
+ import sys
50
+ import time
51
+ from pathlib import Path
52
+
53
+ MAGIC = b"MDCX"
54
+ VERSION = 1
55
+
56
+ SCRYPT_N = 2 ** 15
57
+ SCRYPT_R = 8
58
+ SCRYPT_P = 1
59
+ KEY_BYTES = 32
60
+
61
+ def _derive_key(key: str, salt: bytes) -> bytes:
62
+ memory = 128 * SCRYPT_N * SCRYPT_R
63
+ return hashlib.scrypt(key.encode("utf-8"), salt=salt,
64
+ n=SCRYPT_N, r=SCRYPT_R, p=SCRYPT_P, dklen=KEY_BYTES,
65
+ maxmem=memory * 2)
66
+
67
+ def _encrypt(datos: bytes, clave_derivada: bytes) -> tuple[bytes, bytes]:
68
+ from cryptography.hazmat.primitives.ciphers.aead import AESGCM
69
+ nonce = os.urandom(12)
70
+ return nonce, AESGCM(clave_derivada).encrypt(nonce, datos, None)
71
+
72
+ def _decrypt(body: bytes, clave_derivada: bytes, nonce: bytes) -> bytes:
73
+ from cryptography.hazmat.primitives.ciphers.aead import AESGCM
74
+ return AESGCM(clave_derivada).decrypt(nonce, body, None)
75
+
76
+ def _build_database(folder: Path) -> tuple[bytes, dict]:
77
+ """Build the in-memory database with documents, index and provenance."""
78
+ from . import search as B
79
+
80
+ docs = B.load_documents(folder)
81
+ connection = sqlite3.connect(":memory:")
82
+ connection.executescript("""
83
+ PRAGMA journal_mode = OFF;
84
+ CREATE TABLE document (
85
+ id INTEGER PRIMARY KEY,
86
+ name TEXT NOT NULL,
87
+ pseudopath TEXT NOT NULL,
88
+ source TEXT NOT NULL,
89
+ folder TEXT,
90
+ archive TEXT,
91
+ verification_status TEXT,
92
+ -- Normalised text of the whole document. Literal matching runs here rather
93
+ -- than over passages, because a quoted phrase often crosses the boundary
94
+ -- between paragraphs.
95
+ normalized_text TEXT
96
+ );
97
+ CREATE TABLE passage (
98
+ id INTEGER PRIMARY KEY,
99
+ documento_id INTEGER NOT NULL REFERENCES document(id),
100
+ position INTEGER NOT NULL,
101
+ text TEXT NOT NULL
102
+ );
103
+ -- The index is declared external to the content so the text is not stored
104
+ -- twice: FTS5 indexes what lives in the passage table.
105
+ CREATE VIRTUAL TABLE passage_fts USING fts5(
106
+ text, content='passage', content_rowid='id', tokenize='unicode61'
107
+ );
108
+ CREATE TABLE meta (key TEXT PRIMARY KEY, value TEXT);
109
+ -- Document frequency per term. FTS5 holds this internally but does not
110
+ -- expose it usably, and without it the package cannot rank by the same
111
+ -- criterion as the folder-based search.
112
+ CREATE TABLE df (term TEXT PRIMARY KEY, passages INTEGER NOT NULL);
113
+ """)
114
+
115
+ n_passages = 0
116
+ for i, d in enumerate(docs, 1):
117
+ text = d["text"]
118
+ archive = ""
119
+ status = ""
120
+ for line in text.splitlines()[:12]:
121
+ if line.startswith("source_format:"):
122
+ archive = line.split(":", 1)[1].strip()
123
+ elif line.startswith("verification_status:"):
124
+ status = line.split(":", 1)[1].strip()
125
+ connection.execute(
126
+ "INSERT INTO document VALUES (?,?,?,?,?,?,?,?)",
127
+ (i, d["name"], d["pseudopath"], d["source"], d["folder"], archive, status,
128
+ d["norm"]))
129
+ for j, bloque in enumerate(d["blocks"] if "blocks" in d else _split_blocks(text)):
130
+ if not bloque.strip():
131
+ continue
132
+ n_passages += 1
133
+ connection.execute("INSERT INTO passage VALUES (?,?,?,?)",
134
+ (n_passages, i, j, bloque))
135
+
136
+ connection.execute("INSERT INTO passage_fts(passage_fts) VALUES('rebuild')")
137
+
138
+ from . import search as _B
139
+ from collections import Counter as _Counter
140
+
141
+ df_count: _Counter = _Counter()
142
+ lengths: list[int] = []
143
+ for (text,) in connection.execute("SELECT text FROM passage"):
144
+ tk = _B._TOKEN_RE.findall(_B._normalize(text))
145
+ lengths.append(len(tk))
146
+ for t in set(tk):
147
+ if len(t) > 2:
148
+ df_count[t] += 1
149
+ connection.executemany("INSERT INTO df VALUES (?,?)", df_count.items())
150
+ avg_length = sum(lengths) / len(lengths) if lengths else 60.0
151
+
152
+ summary = {
153
+ "documents": len(docs),
154
+ "passages": n_passages,
155
+ "created_utc": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
156
+ "source_folder": folder.name,
157
+ "largo_medio_pasaje": round(avg_length, 2),
158
+ "terminos_indexados": len(df_count),
159
+ }
160
+ manifiesto = folder / "_manifest.json"
161
+ if manifiesto.exists():
162
+ try:
163
+ m = json.loads(manifiesto.read_text(encoding="utf-8"))
164
+ summary["conversion"] = m.get("summary", {})
165
+ except Exception: # noqa: BLE001
166
+ pass
167
+ for k, v in summary.items():
168
+ connection.execute("INSERT INTO meta VALUES (?,?)",
169
+ (k, json.dumps(v) if not isinstance(v, str) else v))
170
+ connection.commit()
171
+
172
+ datos = connection.serialize()
173
+ connection.close()
174
+ return bytes(datos), summary
175
+
176
+ def _split_blocks(text: str) -> list[str]:
177
+ return [b for b in text.split("\n\n") if b.strip()]
178
+
179
+ def generate_signing_key() -> tuple[str, str]:
180
+ """Create an Ed25519 key pair and return it as (private, public) hex strings.
181
+
182
+ The private key signs packages; the public key lets anyone verify who issued
183
+ one. Only the public half is meant to be distributed.
184
+ """
185
+ from cryptography.hazmat.primitives.asymmetric import ed25519
186
+
187
+ private = ed25519.Ed25519PrivateKey.generate()
188
+ return (private.private_bytes_raw().hex(),
189
+ private.public_key().public_bytes_raw().hex())
190
+
191
+
192
+ def _sign(digest: str, signing_key: str) -> str:
193
+ from cryptography.hazmat.primitives.asymmetric import ed25519
194
+
195
+ key = ed25519.Ed25519PrivateKey.from_private_bytes(bytes.fromhex(signing_key))
196
+ return key.sign(digest.encode("ascii")).hex()
197
+
198
+
199
+ def verify_signature(path: Path, public_key: str) -> bool:
200
+ """Report whether a package was signed by the holder of this public key.
201
+
202
+ The signature covers the digest of the encrypted body, so it attests both the
203
+ issuer and the content: altering either invalidates it. Verification needs
204
+ neither the encryption key nor the contents of the package.
205
+ """
206
+ from cryptography.exceptions import InvalidSignature
207
+ from cryptography.hazmat.primitives.asymmetric import ed25519
208
+
209
+ header = read_header(path)
210
+ signature = header.get("signature")
211
+ if not signature:
212
+ return False
213
+ # The signature covers the digest recorded in the header, so it must be checked
214
+ # together with the integrity of the body. Verifying the signature alone would
215
+ # accept a package whose body had been replaced while the header was left intact:
216
+ # the stored digest would still match the signature, and the content would not.
217
+ if not header.get("_intact"):
218
+ return False
219
+ try:
220
+ key = ed25519.Ed25519PublicKey.from_public_bytes(bytes.fromhex(public_key))
221
+ key.verify(bytes.fromhex(signature), header["body_digest"].encode("ascii"))
222
+ except (InvalidSignature, ValueError):
223
+ return False
224
+ return True
225
+
226
+
227
+ def pack(folder: Path, target: Path, key: str, issuer: str = "",
228
+ signing_key: str = "") -> dict:
229
+ """Write the .mdcx file and return its figures."""
230
+ import lzma
231
+
232
+ t0 = time.perf_counter()
233
+ base_score, summary = _build_database(folder)
234
+ t_base = time.perf_counter() - t0
235
+
236
+ t0 = time.perf_counter()
237
+ compressed = lzma.compress(base_score, preset=6)
238
+ t_comp = time.perf_counter() - t0
239
+
240
+ salt = os.urandom(16)
241
+ t0 = time.perf_counter()
242
+ derived_key = _derive_key(key, salt)
243
+ nonce, body = _encrypt(compressed, derived_key)
244
+ t_cifrado = time.perf_counter() - t0
245
+
246
+ header = {
247
+ "file_format": "mdcx",
248
+ "version": VERSION,
249
+ "issuer": issuer,
250
+ "created_utc": summary["created_utc"],
251
+ "documents": summary["documents"],
252
+ "passages": summary["passages"],
253
+ "encryption": "AES-256-GCM",
254
+ "key_derivation": {"algorithm": "scrypt", "n": SCRYPT_N, "r": SCRYPT_R, "p": SCRYPT_P},
255
+ "compression": "lzma",
256
+ "salt": salt.hex(),
257
+ "nonce": nonce.hex(),
258
+ "body_digest": hashlib.sha256(body).hexdigest(),
259
+ "signature": "",
260
+ "public_key": "",
261
+ "conversion": summary.get("conversion", {}),
262
+ }
263
+ if signing_key:
264
+ from cryptography.hazmat.primitives.asymmetric import ed25519
265
+
266
+ private = ed25519.Ed25519PrivateKey.from_private_bytes(bytes.fromhex(signing_key))
267
+ header["signature"] = _sign(header["body_digest"], signing_key)
268
+ header["public_key"] = private.public_key().public_bytes_raw().hex()
269
+
270
+ encoded_header = json.dumps(header, ensure_ascii=False).encode("utf-8")
271
+
272
+ with open(target, "wb") as f:
273
+ f.write(MAGIC)
274
+ f.write(struct.pack("<I", len(encoded_header)))
275
+ f.write(encoded_header)
276
+ f.write(body)
277
+
278
+ return {
279
+ "bytes_database": len(base_score),
280
+ "bytes_compressed": len(compressed),
281
+ "bytes_file": target.stat().st_size,
282
+ "seconds_index": round(t_base, 2),
283
+ "seconds_compress": round(t_comp, 2),
284
+ "seconds_encrypt": round(t_cifrado, 2),
285
+ **summary,
286
+ }
287
+
288
+ def read_header(path: Path) -> dict:
289
+ """Header and integrity status, without requiring the key."""
290
+ with open(path, "rb") as f:
291
+ if f.read(4) != MAGIC:
292
+ raise ValueError("not an .mdcx file")
293
+ (n,) = struct.unpack("<I", f.read(4))
294
+ header = json.loads(f.read(n).decode("utf-8"))
295
+ body = f.read()
296
+ header["_intact"] = hashlib.sha256(body).hexdigest() == header.get("body_digest")
297
+ header["_signed"] = bool(header.get("signature"))
298
+ header["_body_bytes"] = len(body)
299
+ return header
300
+
301
+ def open_package(path: Path, key: str) -> tuple[sqlite3.Connection, dict]:
302
+ """Decrypt in memory and return a connection ready to query.
303
+
304
+ Nothing is written to disk, so an open package leaves no plaintext copy."""
305
+ import lzma
306
+
307
+ with open(path, "rb") as f:
308
+ if f.read(4) != MAGIC:
309
+ raise ValueError("not an .mdcx file")
310
+ (n,) = struct.unpack("<I", f.read(4))
311
+ header = json.loads(f.read(n).decode("utf-8"))
312
+ body = f.read()
313
+
314
+ if hashlib.sha256(body).hexdigest() != header.get("body_digest"):
315
+ raise ValueError("file has been altered: body digest does not match")
316
+
317
+ derived_key = _derive_key(key, bytes.fromhex(header["salt"]))
318
+ try:
319
+ compressed = _decrypt(body, derived_key, bytes.fromhex(header["nonce"]))
320
+ except Exception as exc: # noqa: BLE001
321
+ raise ValueError("incorrect key or corrupted file") from exc
322
+
323
+ header["_intact"] = True
324
+ header["_body_bytes"] = len(body)
325
+
326
+ connection = sqlite3.connect(":memory:")
327
+ connection.deserialize(lzma.decompress(compressed))
328
+ return connection, header
329
+
330
+ _SQL_BASE = """
331
+ SELECT d.name, d.pseudopath, d.source, p.text, bm25(passage_fts) AS score
332
+ FROM passage_fts
333
+ JOIN passage p ON p.id = passage_fts.rowid
334
+ JOIN document d ON d.id = p.documento_id
335
+ WHERE passage_fts MATCH ?
336
+ """
337
+
338
+ def _run_match(connection: sqlite3.Connection, expr: str, limit: int,
339
+ only: str | None) -> list[dict]:
340
+ sql = _SQL_BASE
341
+ params: list = [expr]
342
+ if only:
343
+ sql += " AND d.source = ?"
344
+ params.append(only.upper())
345
+ sql += " ORDER BY score LIMIT ?"
346
+ params.append(limit)
347
+ try:
348
+ rows = connection.execute(sql, params).fetchall()
349
+ except sqlite3.OperationalError:
350
+ return []
351
+ return [{"document": r[0], "pseudopath": r[1], "source": r[2],
352
+ "passage": r[3], "score": round(-r[4], 3)}
353
+ for r in rows]
354
+
355
+ K1 = 1.5
356
+ B_LENGTH = 0.45
357
+ DOC_TOP_PASSAGES = 8
358
+
359
+ CANDIDATES = 1200
360
+
361
+ def query(connection: sqlite3.Connection, query_text: str, limit: int = 8,
362
+ only: str | None = None) -> list[dict]:
363
+ """Resolve a query, ranking by document rather than by isolated passage."""
364
+ from . import search as B
365
+
366
+ phrase = query_text.strip().split(".")[0][:160].strip()
367
+ effective = phrase if len(phrase.split()) >= 5 else query_text
368
+
369
+ terms = [t for t in B._TOKEN_RE.findall(B._normalize(effective)) if len(t) > 2]
370
+ terms = B.expand_terms(terms)
371
+ if not terms:
372
+ return []
373
+ distinct_terms = set(terms)
374
+
375
+ expr = " OR ".join(f'"{t}"' for t in distinct_terms)
376
+ candidates = _run_match(connection, expr, CANDIDATES, only)
377
+ if not candidates:
378
+ return []
379
+
380
+ df, n_passages, avg_length = _corpus_statistics(connection)
381
+
382
+ by_document: dict[str, list[dict]] = {}
383
+ for r in candidates:
384
+ frec = _term_frequencies(r["passage"], distinct_terms)
385
+ if not frec:
386
+ continue
387
+ largo = max(len(B._TOKEN_RE.findall(B._normalize(r["passage"]))), 1)
388
+ score = 0.0
389
+ for t, f in frec.items():
390
+ d_t = df.get(t, 1)
391
+ idf = math.log(1 + (n_passages - d_t + 0.5) / (d_t + 0.5))
392
+ score += idf * (f * (K1 + 1)) / (
393
+ f + K1 * (1 - B_LENGTH + B_LENGTH * largo / avg_length))
394
+ r = dict(r)
395
+ r["score"] = round(score, 3)
396
+ r["terms"] = sorted(frec)
397
+ by_document.setdefault(r["document"], []).append(r)
398
+
399
+ ranking = []
400
+ for name, passages in by_document.items():
401
+ passages.sort(key=lambda x: -x["score"])
402
+ base_score = sum(x["score"] for x in passages[:DOC_TOP_PASSAGES])
403
+ coverage = max(len(x["terms"]) for x in passages) / max(len(distinct_terms), 1)
404
+ ranking.append((base_score * coverage, passages))
405
+ ranking.sort(key=lambda par: -par[0])
406
+
407
+ if len(phrase.split()) >= 5:
408
+ needle = B._normalize(phrase)
409
+ preferred = [n for (n, t) in connection.execute(
410
+ "SELECT name, normalized_text FROM document") if t and needle in t]
411
+ if preferred:
412
+ position = {d: i for i, d in enumerate(preferred)}
413
+ ranking.sort(key=lambda par: (position.get(par[1][0]["document"], len(position)),
414
+ -par[0]))
415
+
416
+ out: list[dict] = []
417
+ for round_index in range(DOC_TOP_PASSAGES):
418
+ for score, passages in ranking:
419
+ if round_index < len(passages):
420
+ r = dict(passages[round_index])
421
+ r["score_documento"] = round(score, 3)
422
+ out.append(r)
423
+ if len(out) >= limit:
424
+ return out
425
+ return out[:limit]
426
+
427
+ def _term_frequencies(text: str, terms: set[str]) -> dict[str, int]:
428
+ from . import search as B
429
+
430
+ cuenta: dict[str, int] = {}
431
+ for t in B._TOKEN_RE.findall(B._normalize(text)):
432
+ if t in terms:
433
+ cuenta[t] = cuenta.get(t, 0) + 1
434
+ return cuenta
435
+
436
+ _STATS_CACHE: dict[int, tuple] = {}
437
+
438
+ def _corpus_statistics(connection: sqlite3.Connection) -> tuple[dict, int, float]:
439
+ """Document frequency per term and mean passage length, as packed."""
440
+ key = id(connection)
441
+ if key not in _STATS_CACHE:
442
+ df = {t: n for t, n in connection.execute("SELECT term, passages FROM df")}
443
+ row = connection.execute("SELECT value FROM meta WHERE key='passages'").fetchone()
444
+ n = int(json.loads(row[0])) if row else max(len(df), 1)
445
+ row = connection.execute(
446
+ "SELECT value FROM meta WHERE key='largo_medio_pasaje'").fetchone()
447
+ lm = float(json.loads(row[0])) if row else 60.0
448
+ _STATS_CACHE[key] = (df, n, lm)
449
+ return _STATS_CACHE[key]
450
+
451
+ def export(path: Path, key: str, target: Path) -> dict:
452
+ """Rebuild the Markdown folder from the package.
453
+
454
+ Directory structure is restored from the pseudopath stored with each document."""
455
+ connection, header = open_package(path, key)
456
+ try:
457
+ rows = connection.execute(
458
+ "SELECT d.pseudopath, d.name, group_concat(p.text, char(10) || char(10)) "
459
+ "FROM document d JOIN passage p ON p.documento_id = d.id "
460
+ "GROUP BY d.id ORDER BY p.position").fetchall()
461
+ written = 0
462
+ for pseudopath, name, text in rows:
463
+ relative = pseudopath[2:] if pseudopath.startswith("@/") else pseudopath
464
+ path = target / relative
465
+ path.parent.mkdir(parents=True, exist_ok=True)
466
+ path.write_text((text or "") + "\n", encoding="utf-8")
467
+ written += 1
468
+ finally:
469
+ connection.close()
470
+ return {"documents": written, "target": str(target),
471
+ "created_utc": header.get("created_utc")}
472
+
473
+ def _direction(value: str) -> str:
474
+ """Human-readable direction label."""
475
+ return {"SENT": "SENT", "RECEIVED": "RECEIVED",
476
+ "EMITIDO": "SENT", "RECIBIDO": "RECEIVED"}.get(value, "OTHER")
477
+
478
+
479
+ def main() -> int:
480
+ try:
481
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace")
482
+ except Exception:
483
+ pass
484
+
485
+ ap = argparse.ArgumentParser(description="The .mdcx format: an indexed, encrypted, portable corpus")
486
+ sub = ap.add_subparsers(dest="action", required=True)
487
+
488
+ e = sub.add_parser("pack")
489
+ e.add_argument("--output", default="Output")
490
+ e.add_argument("--target", default="corpus.mdcx")
491
+ e.add_argument("--key", required=True)
492
+ e.add_argument("--issuer", default="")
493
+ e.add_argument("--signing-key", default="",
494
+ help="hex private key to sign the package with")
495
+
496
+ k = sub.add_parser("keygen")
497
+
498
+ v = sub.add_parser("verify")
499
+ v.add_argument("path")
500
+ v.add_argument("--public-key", required=True)
501
+
502
+ i = sub.add_parser("info")
503
+ i.add_argument("path")
504
+
505
+ x = sub.add_parser("export")
506
+ x.add_argument("path")
507
+ x.add_argument("--target", required=True)
508
+ x.add_argument("--key", required=True)
509
+
510
+ b = sub.add_parser("search")
511
+ b.add_argument("path")
512
+ b.add_argument("query_text")
513
+ b.add_argument("--key", required=True)
514
+ b.add_argument("--limit", type=int, default=5)
515
+ b.add_argument("--only", choices=["received", "sent"])
516
+
517
+ args = ap.parse_args()
518
+
519
+ if args.action == "pack":
520
+ r = pack(Path(args.output), Path(args.target), args.key, args.issuer,
521
+ args.signing_key)
522
+ print(f"Packed: {args.target}")
523
+ print(f" documents {r['documents']} passages {r['passages']}")
524
+ print(f" database {r['bytes_database']:,} -> compressed {r['bytes_compressed']:,} "
525
+ f"-> file {r['bytes_file']:,} bytes".replace(",", "."))
526
+ print(f" index {r['seconds_index']}s compress {r['seconds_compress']}s "
527
+ f"encrypt {r['seconds_encrypt']}s")
528
+ return 0
529
+
530
+ if args.action == "keygen":
531
+ private, public = generate_signing_key()
532
+ print("Keep the private key secret; distribute only the public key.")
533
+ print(f" private: {private}")
534
+ print(f" public : {public}")
535
+ return 0
536
+
537
+ if args.action == "verify":
538
+ valid = verify_signature(Path(args.path), args.public_key)
539
+ print("Signature valid: the package was issued by the holder of this key "
540
+ "and has not been altered." if valid else
541
+ "Signature not valid: unsigned package, different key, or altered content.")
542
+ return 0 if valid else 1
543
+
544
+ if args.action == "info":
545
+ header = read_header(Path(args.path))
546
+ print(f"Format : {header['file_format']} v{header['version']}")
547
+ print(f"Issuer : {header.get('issuer') or '(not declared)'}")
548
+ print(f"Created : {header['created_utc']}")
549
+ print(f"Content : {header['documents']} documents, {header['passages']} passages")
550
+ print(f"Encryption: {header['encryption']} with {header['key_derivation']['algorithm']}")
551
+ print(f"Integrity : {'intact' if header['_intact'] else 'ALTERED'}")
552
+ print(f"Signature : {'present' if header.get('_signed') else 'none'}"
553
+ + (f" (public key {header['public_key'][:16]}...)" if header.get('public_key') else ""))
554
+ if header.get("conversion"):
555
+ print(f"Conversion: {json.dumps(header['conversion'], ensure_ascii=False)[:200]}")
556
+ return 0
557
+
558
+ if args.action == "export":
559
+ r = export(Path(args.path), args.key, Path(args.target))
560
+ print(f"Exported {r['documents']} documents to {r['target']}")
561
+ return 0
562
+
563
+ connection, header = open_package(Path(args.path), args.key)
564
+ results = query(connection, args.query_text, args.limit, args.only)
565
+ print(f"{len(results)} passage(s)\n")
566
+ for r in results:
567
+ print("-" * 96)
568
+ print(f"[{r['source']}] {r['document']} [score {r['score']}]")
569
+ print(f"{r['pseudopath']}")
570
+ print(r["passage"][:1200])
571
+ return 0
572
+
573
+ if __name__ == "__main__":
574
+ raise SystemExit(main())