esbi-cli 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (72) hide show
  1. esbi_cli/__init__.py +8 -0
  2. esbi_cli/ask/__init__.py +0 -0
  3. esbi_cli/ask/answer.py +256 -0
  4. esbi_cli/bench/__init__.py +0 -0
  5. esbi_cli/bench/cases.py +57 -0
  6. esbi_cli/bench/metrics.py +23 -0
  7. esbi_cli/bench/report.py +117 -0
  8. esbi_cli/bench/runner.py +114 -0
  9. esbi_cli/capture/__init__.py +0 -0
  10. esbi_cli/capture/inbox.py +63 -0
  11. esbi_cli/capture/legacy.py +49 -0
  12. esbi_cli/cli.py +1387 -0
  13. esbi_cli/config.py +344 -0
  14. esbi_cli/doctor.py +391 -0
  15. esbi_cli/evaluate.py +91 -0
  16. esbi_cli/export.py +137 -0
  17. esbi_cli/extract/__init__.py +107 -0
  18. esbi_cli/extract/clip.py +30 -0
  19. esbi_cli/extract/html.py +60 -0
  20. esbi_cli/extract/image.py +58 -0
  21. esbi_cli/extract/pdf.py +109 -0
  22. esbi_cli/gitops.py +101 -0
  23. esbi_cli/index.py +303 -0
  24. esbi_cli/ingest/__init__.py +0 -0
  25. esbi_cli/ingest/apply.py +480 -0
  26. esbi_cli/ingest/chunks.py +49 -0
  27. esbi_cli/ingest/connect.py +87 -0
  28. esbi_cli/ingest/digest.py +91 -0
  29. esbi_cli/ingest/pipeline.py +176 -0
  30. esbi_cli/ingest/plan.py +231 -0
  31. esbi_cli/ingest/read.py +105 -0
  32. esbi_cli/ingest/retrieve.py +59 -0
  33. esbi_cli/init.py +176 -0
  34. esbi_cli/interrupts.py +90 -0
  35. esbi_cli/lang.py +341 -0
  36. esbi_cli/links.py +10 -0
  37. esbi_cli/lint/__init__.py +0 -0
  38. esbi_cli/lint/checks.py +178 -0
  39. esbi_cli/lint/report.py +60 -0
  40. esbi_cli/llm/__init__.py +0 -0
  41. esbi_cli/llm/adapter.py +393 -0
  42. esbi_cli/llm/schemas.py +146 -0
  43. esbi_cli/mail/__init__.py +0 -0
  44. esbi_cli/mail/convert.py +194 -0
  45. esbi_cli/mail/credentials.py +65 -0
  46. esbi_cli/mail/fetch.py +154 -0
  47. esbi_cli/mail/imap.py +92 -0
  48. esbi_cli/netguard.py +127 -0
  49. esbi_cli/privacy.py +81 -0
  50. esbi_cli/queue.py +179 -0
  51. esbi_cli/reingest.py +165 -0
  52. esbi_cli/report/__init__.py +0 -0
  53. esbi_cli/report/daily_index.py +235 -0
  54. esbi_cli/report/index_md.py +21 -0
  55. esbi_cli/report/readstate.py +26 -0
  56. esbi_cli/run.py +100 -0
  57. esbi_cli/runlock.py +31 -0
  58. esbi_cli/runlog.py +80 -0
  59. esbi_cli/schedule.py +106 -0
  60. esbi_cli/templates/SCHEMA.md +52 -0
  61. esbi_cli/templates/clipper-template.json +17 -0
  62. esbi_cli/templates/clipper-youtube-template.json +18 -0
  63. esbi_cli/templates/config.example.toml +108 -0
  64. esbi_cli/update.py +247 -0
  65. esbi_cli/vault.py +188 -0
  66. esbi_cli/wizards/clipper.sh +271 -0
  67. esbi_cli/wizards/email.sh +265 -0
  68. esbi_cli-0.2.1.dist-info/METADATA +167 -0
  69. esbi_cli-0.2.1.dist-info/RECORD +72 -0
  70. esbi_cli-0.2.1.dist-info/WHEEL +4 -0
  71. esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
  72. esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
esbi_cli/init.py ADDED
@@ -0,0 +1,176 @@
1
+ """`sb init`: a new vault and a config file, without overwriting anything that already exists."""
2
+
3
+ import json
4
+ import re
5
+ import subprocess
6
+ from pathlib import Path
7
+
8
+ from esbi_cli import lang
9
+ from esbi_cli.gitops import STATE_IGNORE, has_git
10
+
11
+ TEMPLATES = Path(__file__).parent / "templates"
12
+ EXAMPLE_CONFIG = TEMPLATES / "config.example.toml" # inside the package: an installed copy has it
13
+ FOLDERS = (
14
+ "inbox",
15
+ "raw",
16
+ "attachments",
17
+ *(
18
+ f"wiki/{kind}"
19
+ for kind in ("sources", "concepts", "entities", "syntheses", "review", "daily")
20
+ ),
21
+ )
22
+
23
+
24
+ def _files(language: str) -> dict[str, str]:
25
+ schema = (TEMPLATES / "SCHEMA.md").read_text(encoding="utf-8")
26
+ return {
27
+ "SCHEMA.md": schema.replace("{language}", lang.name(language)),
28
+ "index.md": f"# {lang.t(language, 'index_title')}\n",
29
+ "log.md": "# Log\n",
30
+ ".gitignore": ".obsidian/workspace*.json\n.obsidian/cache\n.trash/\n.DS_Store\n"
31
+ + "\n".join(STATE_IGNORE)
32
+ + "\n",
33
+ }
34
+
35
+
36
+ def init_vault(root: Path, language: str = lang.DEFAULT) -> list[str]:
37
+ """Create what is missing; return what was created."""
38
+ created = []
39
+ for folder in FOLDERS:
40
+ if not (root / folder).is_dir():
41
+ (root / folder).mkdir(parents=True)
42
+ created.append(f"{folder}/")
43
+ for name, text in _files(language).items():
44
+ if not (root / name).exists():
45
+ (root / name).write_text(text, encoding="utf-8")
46
+ created.append(name)
47
+ if not (root / ".git").exists() and has_git():
48
+ subprocess.run(["git", "init", "-q", "-b", "main"], cwd=root, check=True)
49
+ created.append("git repository")
50
+ return created
51
+
52
+
53
+ MODELS = {
54
+ "local": "Everything stays on this Mac: the notes are written by a local model.",
55
+ "subscription": (
56
+ "The text of each source goes to Anthropic through your Claude subscription "
57
+ "(the official `claude` tool, `claude auth login` once). Email is read only by the local model."
58
+ ),
59
+ "api": (
60
+ "The text of each source goes to Anthropic through its API, billed per token: set "
61
+ "ANTHROPIC_API_KEY in the environment. Email is read only by the local model."
62
+ ),
63
+ }
64
+ RUNTIMES = ("ollama", "lmstudio")
65
+
66
+
67
+ def local_model(runtime: str, name: str | None) -> str:
68
+ """`<provider>/<name>` of the local model: Ollama's default, or the LM Studio identifier."""
69
+ return f"lmstudio/{name}" if runtime == "lmstudio" else f"ollama/{name or 'llama3.2:latest'}"
70
+
71
+
72
+ def llm_sections(model: str, runtime: str, name: str | None, base_url: str | None = None) -> str:
73
+ """The [llm.*] sections for the chosen kind of model."""
74
+ local = local_model(runtime, name)
75
+ if model == "local":
76
+ return (
77
+ f'[llm.summarize]\nmodel = "{local}"\ntimeout_seconds = 300\n'
78
+ + (f"base_url = {json.dumps(base_url)}\n" if base_url else "")
79
+ + ("num_ctx = 8192\nmax_tokens = 1200\n" if runtime == "ollama" else "")
80
+ )
81
+ main = "claude-cli/default" if model == "subscription" else "anthropic/claude-sonnet-5-5"
82
+ sections = [
83
+ f'[llm.summarize]\nmodel = "{main}"\nfallback = "{local}"\ntimeout_seconds = 300\n',
84
+ f'[llm.ask]\nmodel = "{main}"\nfallback = "{local}"\ntimeout_seconds = 600\n',
85
+ f'[llm.private]\nmodel = "{local}"\n', # email is read only by the local model
86
+ ]
87
+ if model == "subscription":
88
+ sections.insert(
89
+ 1, f'[llm.synthesize]\nmodel = "{main}"\nfallback = "{local}"\ntimeout_seconds = 600\n'
90
+ )
91
+ return "\n".join(sections)
92
+
93
+
94
+ OCR_MODEL = "qwen3-vl:2b-instruct"
95
+ OCR_SECTION = f'[llm.ocr]\nmodel = "ollama/{OCR_MODEL}"\ntimeout_seconds = 600\n'
96
+
97
+
98
+ def write_config(
99
+ path: Path,
100
+ vault: Path,
101
+ model: str = "local",
102
+ *,
103
+ viewer: str = "obsidian",
104
+ nightly: str | None = None,
105
+ runtime: str = "ollama",
106
+ local_name: str | None = None,
107
+ ocr: bool = False,
108
+ base_url: str | None = None,
109
+ language: str = lang.DEFAULT,
110
+ ) -> bool:
111
+ """Write the example config with this vault and choices; False if a config is already there."""
112
+ if path.exists():
113
+ return False
114
+ text = EXAMPLE_CONFIG.read_text(encoding="utf-8")
115
+ text = re.sub(r'^vault = ".*"', f'vault = "{vault}"', text, count=1, flags=re.M)
116
+ text = re.sub(
117
+ r'^language = ".*"$',
118
+ f'language = "{language}"\nviewer = "{viewer}"',
119
+ text,
120
+ count=1,
121
+ flags=re.M,
122
+ )
123
+ if nightly:
124
+ text = re.sub(
125
+ r'^nightly_time = "[^"]*"', f'nightly_time = "{nightly}"', text, count=1, flags=re.M
126
+ )
127
+ llm = llm_sections(model, runtime, local_name, base_url)
128
+ if ocr: # a local vision model reads images and scanned PDFs: nothing leaves this machine
129
+ llm += "\n" + OCR_SECTION
130
+ text = re.sub(
131
+ r"^\[llm\.summarize\].*?(?=^# Optional|^\[bench\])",
132
+ llm + "\n",
133
+ text,
134
+ count=1,
135
+ flags=re.M | re.S,
136
+ )
137
+ path.parent.mkdir(parents=True, exist_ok=True)
138
+ path.write_text(text, encoding="utf-8")
139
+ return True
140
+
141
+
142
+ def connect_remote(root: Path, url: str) -> bool:
143
+ """Add `origin` as the backup remote; False if the vault already has one (left alone)."""
144
+ if (
145
+ subprocess.run(
146
+ ["git", "remote", "get-url", "origin"], cwd=root, capture_output=True
147
+ ).returncode
148
+ == 0
149
+ ):
150
+ return False
151
+ subprocess.run(["git", "remote", "add", "origin", url], cwd=root, check=True)
152
+ return True
153
+
154
+
155
+ def set_email_block(path: Path, user: str, label: str = "esbi-cli") -> None:
156
+ """Turn on the [email] section of a config file for this Gmail address and label, replacing the
157
+ block if there is one and touching nothing else. The password is never written here."""
158
+ block = (
159
+ "[email]\n"
160
+ "enabled = true\n"
161
+ 'imap_host = "imap.gmail.com"\n'
162
+ f'mailbox = "{label}"\n'
163
+ f'user = "{user}"\n'
164
+ "# The app password lives in the macOS Keychain (service esbi-cli-imap), never in this file.\n"
165
+ )
166
+ text = path.read_text(encoding="utf-8")
167
+ match = re.search(r"^\[email\]\n.*?(?=^\[|\Z)", text, re.S | re.M)
168
+ if match:
169
+ kept = [ # choices made by hand survive a re-run of the setup
170
+ line for line in match.group().splitlines() if line.startswith("follow_links")
171
+ ]
172
+ block += "".join(line + "\n" for line in kept)
173
+ text = text[: match.start()] + block + "\n" + text[match.end() :]
174
+ else:
175
+ text = text.rstrip("\n") + "\n\n" + block
176
+ path.write_text(text, encoding="utf-8")
esbi_cli/interrupts.py ADDED
@@ -0,0 +1,90 @@
1
+ """SIGINT and SIGTERM as an exception in the main thread, so the code that owns a resource can put
2
+ it back. Handlers exist only inside `handling()`. Within it a signal raises `Interrupted` only
3
+ inside `interruptible()` and outside `deferred()`; anywhere else it waits for the next safe point,
4
+ so a note is written whole or not at all.
5
+ """
6
+
7
+ import signal
8
+ import sys
9
+ import threading
10
+ from contextlib import contextmanager
11
+
12
+ SIGNALS = (signal.SIGINT, signal.SIGTERM)
13
+
14
+ # ponytail: module state, one run per process; a per-run object if two runs ever share a process
15
+ _depth = 0 # a signal raises only at 0
16
+ _pending: int | None = None # the first signal received; later ones are ignored while it is handled
17
+
18
+
19
+ class Interrupted(BaseException):
20
+ """Not an Exception, like KeyboardInterrupt: `except Exception` must not turn it into a failure."""
21
+
22
+ def __init__(self, signum: int):
23
+ super().__init__(signum)
24
+ self.signum = signum
25
+
26
+
27
+ def _handle(signum, frame) -> None:
28
+ global _pending
29
+ if _pending is None:
30
+ _pending = signum
31
+ if _depth == 0:
32
+ raise Interrupted(signum)
33
+
34
+
35
+ def _raise_if_pending() -> None:
36
+ if _depth == 0 and _pending is not None:
37
+ raise Interrupted(_pending)
38
+
39
+
40
+ @contextmanager
41
+ def handling():
42
+ """Install the handlers for the block, then restore the previous ones. The block is protected
43
+ until `interruptible()`. Outside the main thread nothing is installed."""
44
+ global _depth, _pending
45
+ if threading.current_thread() is not threading.main_thread(): # signal.signal needs it
46
+ yield
47
+ return
48
+ previous = {s: signal.signal(s, _handle) for s in SIGNALS}
49
+ _depth, _pending = 1, None
50
+ try:
51
+ yield
52
+ finally:
53
+ _depth, _pending = 0, None
54
+ for s, handler in previous.items():
55
+ signal.signal(s, handler)
56
+
57
+
58
+ @contextmanager
59
+ def interruptible():
60
+ """A signal raises `Interrupted` inside this block (one that came earlier raises on entry)."""
61
+ global _depth
62
+ _depth -= 1
63
+ try:
64
+ _raise_if_pending()
65
+ yield
66
+ finally:
67
+ _depth += 1
68
+
69
+
70
+ @contextmanager
71
+ def deferred():
72
+ """A signal inside this block waits until it ends. Costs nothing when no handler is installed."""
73
+ global _depth
74
+ _depth += 1
75
+ try:
76
+ yield
77
+ finally:
78
+ _depth -= 1
79
+ _raise_if_pending() # normal exit only: a real error is not replaced
80
+
81
+
82
+ @contextmanager
83
+ def exit_when_interrupted():
84
+ """For commands with no queue (`sb ingest`, `sb reingest`): stop with the conventional code."""
85
+ try:
86
+ with handling(), interruptible():
87
+ yield
88
+ except Interrupted as exc:
89
+ print("Interrupted.", file=sys.stderr)
90
+ raise SystemExit(128 + exc.signum) from None
esbi_cli/lang.py ADDED
@@ -0,0 +1,341 @@
1
+ """The languages the wiki can be written in: one catalogue, nothing else knows a language.
2
+
3
+ Code asks for text by language-independent key (`t("es", "summary")`) and recognises headings of
4
+ every catalogued language (`key_of`), so a vault written in one language and later switched to
5
+ another still reads. Adding a language is adding one entry to `LANGUAGES`: see
6
+ https://rubenamaury.github.io/esbi-cli/docs/how-to/add-a-language/.
7
+
8
+ Each entry has:
9
+ - `name`: how the prompts call the language ("Write all text in Spanish").
10
+ - `hint`: an optional extra line for the prompts, for a language a small model needs more help with.
11
+ - `stopwords`: common words used to notice an answer in the wrong language; empty skips the check.
12
+ - `generic_terms`: words too general to be a glossary entry ("data", "system"); they are dropped.
13
+ - `disclaimers`: a regex for a definition that says it has none ("not defined in the text").
14
+ - `relation_examples`: short labels for how two ideas relate, as the model should write them.
15
+ - `placeholders`: how a model that copied the prompt's wording starts a "summary" ("Executive summary of ...").
16
+ - `ask_example` / `rewrite_example`: worked outputs shown to the model, in this language.
17
+ - `labels`: every piece of text the program writes into the wiki.
18
+ """
19
+
20
+ import re
21
+ import unicodedata
22
+ from collections.abc import Iterable
23
+
24
+ DEFAULT = "en"
25
+
26
+ LANGUAGES: dict[str, dict] = {
27
+ "en": {
28
+ "name": "English",
29
+ "hint": "",
30
+ "relation_examples": '"extends", "complements", "improves", "uses", "is an example of"',
31
+ "stopwords": "the of and to in is that for with are this on as by from be an",
32
+ "generic_terms": "data information system process technology example method approach result problem",
33
+ "disclaimers": r"not (defined|mentioned|specified|provided|explained)|(does|do) not (define|mention|specify|explain|provide)|no definition",
34
+ "placeholders": ("executive summary", "summary of"),
35
+ "ask_example": (
36
+ '{"title": "What is a graph", "one_liner": "A graph is a set of nodes joined by edges.", '
37
+ '"answer": "A graph is a set of nodes connected by edges [[Graph]]. '
38
+ 'It is used to model networks [[Networks]].", "cited_pages": ["Graph", "Networks"]}'
39
+ ),
40
+ "rewrite_example": (
41
+ 'question "When should a judge accept and when escalate?" -> {"terms": ["judge", '
42
+ '"accept", "escalate", "confidence", "confident", "threshold", "review"]}'
43
+ ),
44
+ "labels": {
45
+ # sections of a source note
46
+ "summary": "Executive summary",
47
+ "abstract": "Detailed summary",
48
+ "insights": "Key ideas",
49
+ "key_points": "Key points",
50
+ "terms": "Key terms",
51
+ "quotes": "Key quotes",
52
+ "diagram": "Diagram",
53
+ "figures": "Figures",
54
+ "connections": "Connections to your wiki",
55
+ "open_questions": "Open questions",
56
+ "concepts": "Concepts",
57
+ "entities": "Entities",
58
+ "related": "Related",
59
+ "contradictions": "Possible contradictions",
60
+ # elsewhere in notes
61
+ "from_source": "From",
62
+ "original_source": "Original source",
63
+ "untitled": "Untitled",
64
+ "source_suffix": "(source)",
65
+ "figure": "Figure",
66
+ "image": "Image",
67
+ "no_subject": "No subject",
68
+ "page_abbr": "p.",
69
+ "contradiction_callout": "Possible contradiction ({date}) with {link}: {note}",
70
+ "contradiction_review_name": "{date} contradiction - {source}",
71
+ "contradiction_review_title": "Possible contradictions from {link}",
72
+ "contradiction_review_footer": "Review and resolve; then delete this note.",
73
+ # ask
74
+ "no_answer": "I find nothing about this in the wiki.",
75
+ "question_label": "Question",
76
+ "sources": "Sources",
77
+ "synthesis_suffix": "(synthesis)",
78
+ "answer_title": "Answer",
79
+ # index.md and log.md
80
+ "index_title": "Index",
81
+ "index_blurb": "Catalogue of the wiki ({total} pages). Kept by the worker.",
82
+ "syntheses": "Syntheses",
83
+ "log_created": "created: {names}",
84
+ "log_updated": "updated: {names}",
85
+ "log_review": "to review: {n}",
86
+ # the daily index
87
+ "daily_title": "Index of {date}",
88
+ "daily_read": "Read",
89
+ "daily_processed": "Processed today",
90
+ "daily_queue": "Tomorrow's queue",
91
+ "daily_review": "To review",
92
+ "daily_revisit": "Revisit",
93
+ "daily_runs": "Runs",
94
+ "daily_stats": "Statistics",
95
+ "nothing_read": "_Nothing new marked as read._",
96
+ "nothing_processed": "_Nothing was processed today._",
97
+ "queue_empty": "_The queue is empty._",
98
+ "queue_one": "1 source in the queue",
99
+ "queue_many": "{n} sources in the queue",
100
+ "queue_cap": " (showing the {n} oldest)",
101
+ "nothing_pending": "_Nothing pending._",
102
+ "could_not_process": "- Could not process {name} ({error})",
103
+ "retrying": "- Retrying {name} (attempt {n} of {max}): {error}",
104
+ "no_old_notes": "_No older notes to revisit yet._",
105
+ "stat_pages": "- Pages: {total} (sources: {sources}, concepts: {concepts}, entities: {entities})",
106
+ "stat_today": "- Today: {processed} sources processed, {touched} concepts/entities updated",
107
+ "stat_read": "- Read: {read} of {total} sources",
108
+ "stat_orphans": "- Orphans (no incoming links): {n}",
109
+ "no_runs": "_No runs yet today._",
110
+ "run_manual": " (manual)",
111
+ "run_line": "- {at}{manual} — processed: {ingested}, skipped: {skipped}, failed: {failed} · {tokens} tokens · {duration}",
112
+ "run_minutes": "{n} min",
113
+ "run_under_minute": "<1 min",
114
+ "run_stopped": " · stopped: {reason}",
115
+ "stop_max_sources": "source limit",
116
+ "stop_token_budget": "token budget",
117
+ "stop_llm_unavailable": "model unavailable",
118
+ "home_today": "- Today's index: {link}",
119
+ "home_unread": "- Unread: {n} sources",
120
+ "home_queue": "- Queued: {n} sources",
121
+ # the lint report
122
+ "lint_title": "Lint report ({date})",
123
+ "lint_orphan": "Orphan pages",
124
+ "lint_broken_link": "Broken links",
125
+ "lint_missing_field": "Missing fields",
126
+ "lint_unlinked_mention": "Unlinked mentions",
127
+ "lint_near_duplicate": "Possible duplicates",
128
+ "lint_missing_concept": "Concepts without a page",
129
+ "lint_missing_field_line": "- [[{page}]]: missing `{field}`",
130
+ "lint_unlinked_line": "- [[{page}]] mentions [[{target}]] without linking it",
131
+ "lint_missing_concept_line": "- «{name}» appears in {sources} and has no page of its own",
132
+ "lint_more": "… and {n} more",
133
+ "lint_footer": "The worker only reports; nothing is fixed by itself. This report is regenerated on every run.",
134
+ # the benchmark report
135
+ "bench_title": "Model benchmark ({when})",
136
+ "bench_header": "| model | success | no retry | median (s) | tokens/case | est. cost (USD) | {extra} |",
137
+ "bench_extra_ingest": "right language / concepts",
138
+ "bench_extra_ask": "cites the expected page",
139
+ "bench_routing": "Routing recommendation",
140
+ "bench_none": "no recommendation (no reliable model)",
141
+ "bench_footer": "Only a suggestion: change `[llm.*]` in config.toml by hand if you agree.",
142
+ "bench_question": "What is {title}?",
143
+ },
144
+ },
145
+ "es": {
146
+ "name": "Spanish",
147
+ "hint": "",
148
+ "relation_examples": '"amplía", "complementa", "mejora", "usa", "es un ejemplo de"',
149
+ "stopwords": "de la el que en los las y un una para con por del se es al como más pero sus",
150
+ "generic_terms": "datos información sistema proceso tecnología ejemplo método enfoque resultado problema",
151
+ "disclaimers": r"no (se )?(define|menciona|especifica|explica|proporciona|detalla)|no (est[aá]|aparece) (definid|especificad|en el texto)|sin definici[oó]n",
152
+ "placeholders": ("resumen ejecutivo", "resumen de"),
153
+ "ask_example": (
154
+ '{"title": "Qué es un grafo", "one_liner": "Un grafo es un conjunto de nodos unidos por aristas.", '
155
+ '"answer": "Un grafo es un conjunto de nodos conectados por aristas [[Grafo]]. '
156
+ 'Se usa para modelar redes [[Redes]].", "cited_pages": ["Grafo", "Redes"]}'
157
+ ),
158
+ "rewrite_example": (
159
+ 'question "¿Cuándo debe un juez aceptar y cuándo escalar?" -> {"terms": ["juez", "judge", '
160
+ '"aceptar", "accept", "escalar", "escalate", "confianza", "confident"]}'
161
+ ),
162
+ "labels": {
163
+ "summary": "Resumen ejecutivo",
164
+ "abstract": "Resumen detallado",
165
+ "insights": "Ideas clave",
166
+ "key_points": "Puntos clave",
167
+ "terms": "Términos clave",
168
+ "quotes": "Frases clave",
169
+ "diagram": "Diagrama",
170
+ "figures": "Figuras",
171
+ "connections": "Conexiones con tu wiki",
172
+ "open_questions": "Preguntas abiertas",
173
+ "concepts": "Conceptos",
174
+ "entities": "Entidades",
175
+ "related": "Relacionado",
176
+ "contradictions": "Posibles contradicciones",
177
+ "from_source": "Desde",
178
+ "original_source": "Fuente original",
179
+ "untitled": "Sin título",
180
+ "source_suffix": "(fuente)",
181
+ "figure": "Figura",
182
+ "image": "Imagen",
183
+ "no_subject": "Sin asunto",
184
+ "page_abbr": "p.",
185
+ "contradiction_callout": "Posible contradicción ({date}) con {link}: {note}",
186
+ "contradiction_review_name": "{date} contradicción - {source}",
187
+ "contradiction_review_title": "Posibles contradicciones desde {link}",
188
+ "contradiction_review_footer": "Revisa y resuelve; luego borra esta nota.",
189
+ "no_answer": "No encuentro nada sobre esto en la wiki.",
190
+ "question_label": "Pregunta",
191
+ "sources": "Fuentes",
192
+ "synthesis_suffix": "(síntesis)",
193
+ "answer_title": "Respuesta",
194
+ "index_title": "Índice",
195
+ "index_blurb": "Catálogo de la wiki ({total} páginas). Lo mantiene el worker.",
196
+ "syntheses": "Síntesis",
197
+ "log_created": "creadas: {names}",
198
+ "log_updated": "actualizadas: {names}",
199
+ "log_review": "por revisar: {n}",
200
+ "daily_title": "Índice del {date}",
201
+ "daily_read": "Leído",
202
+ "daily_processed": "Procesado hoy",
203
+ "daily_queue": "Cola de mañana",
204
+ "daily_review": "Por revisar",
205
+ "daily_revisit": "Repasar",
206
+ "daily_runs": "Ejecuciones",
207
+ "daily_stats": "Estadísticas",
208
+ "nothing_read": "_Nada nuevo marcado como leído._",
209
+ "nothing_processed": "_Hoy no se procesó nada._",
210
+ "queue_empty": "_La cola está vacía._",
211
+ "queue_one": "1 fuente en cola",
212
+ "queue_many": "{n} fuentes en cola",
213
+ "queue_cap": " (se muestran las {n} más antiguas)",
214
+ "nothing_pending": "_Nada pendiente._",
215
+ "could_not_process": "- No se pudo procesar {name} ({error})",
216
+ "retrying": "- Reintentando {name} (intento {n} de {max}): {error}",
217
+ "no_old_notes": "_Todavía no hay notas antiguas para repasar._",
218
+ "stat_pages": "- Páginas: {total} (fuentes: {sources}, conceptos: {concepts}, entidades: {entities})",
219
+ "stat_today": "- Hoy: {processed} fuentes procesadas, {touched} conceptos/entidades actualizados",
220
+ "stat_read": "- Leídas: {read} de {total} fuentes",
221
+ "stat_orphans": "- Huérfanas (sin enlaces entrantes): {n}",
222
+ "no_runs": "_Todavía no hubo ejecuciones hoy._",
223
+ "run_manual": " (manual)",
224
+ "run_line": "- {at}{manual} — procesadas: {ingested}, omitidas: {skipped}, fallidas: {failed} · {tokens} tokens · {duration}",
225
+ "run_minutes": "{n} min",
226
+ "run_under_minute": "<1 min",
227
+ "run_stopped": " · detenida: {reason}",
228
+ "stop_max_sources": "límite de fuentes",
229
+ "stop_token_budget": "presupuesto de tokens",
230
+ "stop_llm_unavailable": "LLM no disponible",
231
+ "home_today": "- Índice de hoy: {link}",
232
+ "home_unread": "- Sin leer: {n} fuentes",
233
+ "home_queue": "- En cola: {n} fuentes",
234
+ "lint_title": "Informe de lint ({date})",
235
+ "lint_orphan": "Páginas huérfanas",
236
+ "lint_broken_link": "Enlaces rotos",
237
+ "lint_missing_field": "Campos que faltan",
238
+ "lint_unlinked_mention": "Menciones sin enlazar",
239
+ "lint_near_duplicate": "Posibles duplicados",
240
+ "lint_missing_concept": "Conceptos sin página",
241
+ "lint_missing_field_line": "- [[{page}]]: falta `{field}`",
242
+ "lint_unlinked_line": "- [[{page}]] menciona [[{target}]] sin enlazarlo",
243
+ "lint_missing_concept_line": "- «{name}» aparece en {sources} y no tiene página propia",
244
+ "lint_more": "… y {n} más",
245
+ "lint_footer": "El worker solo informa; nada se corrige solo. Este informe se regenera en cada ejecución.",
246
+ "bench_title": "Benchmark de modelos ({when})",
247
+ "bench_header": "| modelo | éxito | sin reintento | mediana (s) | tokens/caso | coste est. (USD) | {extra} |",
248
+ "bench_extra_ingest": "idioma correcto / conceptos",
249
+ "bench_extra_ask": "cita la página esperada",
250
+ "bench_routing": "Recomendación de enrutado",
251
+ "bench_none": "sin recomendación (ningún modelo fiable)",
252
+ "bench_footer": "Solo es una sugerencia: cambia `[llm.*]` en config.toml a mano si te convence.",
253
+ "bench_question": "¿Qué es {title}?",
254
+ },
255
+ },
256
+ }
257
+
258
+
259
+ def supported() -> str:
260
+ return ", ".join(LANGUAGES)
261
+
262
+
263
+ def get(code: str) -> dict:
264
+ """The catalogue entry for `code`; a clear error naming the supported languages otherwise."""
265
+ try:
266
+ return LANGUAGES[code]
267
+ except (KeyError, TypeError):
268
+ raise ValueError(f"[notes].language must be one of: {supported()}, got {code!r}") from None
269
+
270
+
271
+ def name(code: str) -> str:
272
+ return get(code)["name"]
273
+
274
+
275
+ def t(code: str, key: str, **values) -> str:
276
+ """The text for `key` in language `code`, with `{placeholders}` filled in."""
277
+ return get(code)["labels"][key].format(**values)
278
+
279
+
280
+ def _fold(text: str) -> str:
281
+ plain = "".join(c for c in unicodedata.normalize("NFKD", text) if not unicodedata.combining(c))
282
+ return " ".join(plain.casefold().split())
283
+
284
+
285
+ def every(key: str) -> list[str]:
286
+ """The label for `key` in every language: what a reader must accept."""
287
+ return list(dict.fromkeys(entry["labels"][key] for entry in LANGUAGES.values()))
288
+
289
+
290
+ def key_of(heading: str) -> str | None:
291
+ """The key a heading belongs to, in whatever language it was written; None for any other."""
292
+ wanted = _fold(heading)
293
+ for entry in LANGUAGES.values():
294
+ for key, label in entry["labels"].items():
295
+ if _fold(label) == wanted:
296
+ return key
297
+ return None
298
+
299
+
300
+ def instruction(code: str) -> str:
301
+ """The sentence every prompt carries to say which language to write in."""
302
+ entry = get(code)
303
+ return (
304
+ f"Write ALL text in {entry['name']}, whatever language the source or the SCHEMA is in; "
305
+ f"only verbatim quotes and term names keep the source's language. {entry['hint']}"
306
+ ).strip()
307
+
308
+
309
+ def _hits(text: str, code: str) -> int:
310
+ words = set(LANGUAGES[code]["stopwords"].split())
311
+ return sum(w in words for w in re.findall(r"[^\W\d_]+", text.lower()))
312
+
313
+
314
+ def wrong_language(text: str, code: str, min_hits: int = 4) -> bool:
315
+ """Is `text` clearly in another catalogued language than `code`? A cheap stopword count: small
316
+ models often answer in the source's language. A language with no stopwords is never judged.
317
+ `min_hits` is how many stopwords it takes; a one-sentence definition needs fewer."""
318
+ if not get(code)["stopwords"]:
319
+ return False
320
+ own = _hits(text, code)
321
+ return any(
322
+ _hits(text, other) >= min_hits and _hits(text, other) > 1.5 * own
323
+ for other, entry in LANGUAGES.items()
324
+ if other != code and entry["stopwords"]
325
+ )
326
+
327
+
328
+ SHORT_FIELD_CHARS = 300 # under this, a field is a sentence or two: fewer stopwords are enough
329
+
330
+
331
+ def leaking(fields: Iterable[tuple[str, str]], code: str) -> list[str]:
332
+ """The names of the (name, text) fields written in another catalogued language than `code`,
333
+ each field judged on its own: joined with a long field in the right language, a short one in
334
+ the wrong language goes unnoticed. A language without a word list is never judged."""
335
+ return list(
336
+ dict.fromkeys(
337
+ name
338
+ for name, text in fields
339
+ if wrong_language(text, code, 4 if len(text) >= SHORT_FIELD_CHARS else 2)
340
+ )
341
+ )
esbi_cli/links.py ADDED
@@ -0,0 +1,10 @@
1
+ """Wikilink parsing shared by the daily index and lint."""
2
+
3
+ import re
4
+
5
+ WIKILINK = re.compile(r"\[\[([^\]|#]+)")
6
+
7
+
8
+ def link_targets(text: str) -> list[str]:
9
+ """Page names linked from text: `[[Title]]`, `[[Title|alias]]` and `[[Title#Heading]]` all give Title."""
10
+ return [t.strip() for t in WIKILINK.findall(text)]
File without changes