esbi-cli 0.2.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- esbi_cli/__init__.py +8 -0
- esbi_cli/ask/__init__.py +0 -0
- esbi_cli/ask/answer.py +256 -0
- esbi_cli/bench/__init__.py +0 -0
- esbi_cli/bench/cases.py +57 -0
- esbi_cli/bench/metrics.py +23 -0
- esbi_cli/bench/report.py +117 -0
- esbi_cli/bench/runner.py +114 -0
- esbi_cli/capture/__init__.py +0 -0
- esbi_cli/capture/inbox.py +63 -0
- esbi_cli/capture/legacy.py +49 -0
- esbi_cli/cli.py +1387 -0
- esbi_cli/config.py +344 -0
- esbi_cli/doctor.py +391 -0
- esbi_cli/evaluate.py +91 -0
- esbi_cli/export.py +137 -0
- esbi_cli/extract/__init__.py +107 -0
- esbi_cli/extract/clip.py +30 -0
- esbi_cli/extract/html.py +60 -0
- esbi_cli/extract/image.py +58 -0
- esbi_cli/extract/pdf.py +109 -0
- esbi_cli/gitops.py +101 -0
- esbi_cli/index.py +303 -0
- esbi_cli/ingest/__init__.py +0 -0
- esbi_cli/ingest/apply.py +480 -0
- esbi_cli/ingest/chunks.py +49 -0
- esbi_cli/ingest/connect.py +87 -0
- esbi_cli/ingest/digest.py +91 -0
- esbi_cli/ingest/pipeline.py +176 -0
- esbi_cli/ingest/plan.py +231 -0
- esbi_cli/ingest/read.py +105 -0
- esbi_cli/ingest/retrieve.py +59 -0
- esbi_cli/init.py +176 -0
- esbi_cli/interrupts.py +90 -0
- esbi_cli/lang.py +341 -0
- esbi_cli/links.py +10 -0
- esbi_cli/lint/__init__.py +0 -0
- esbi_cli/lint/checks.py +178 -0
- esbi_cli/lint/report.py +60 -0
- esbi_cli/llm/__init__.py +0 -0
- esbi_cli/llm/adapter.py +393 -0
- esbi_cli/llm/schemas.py +146 -0
- esbi_cli/mail/__init__.py +0 -0
- esbi_cli/mail/convert.py +194 -0
- esbi_cli/mail/credentials.py +65 -0
- esbi_cli/mail/fetch.py +154 -0
- esbi_cli/mail/imap.py +92 -0
- esbi_cli/netguard.py +127 -0
- esbi_cli/privacy.py +81 -0
- esbi_cli/queue.py +179 -0
- esbi_cli/reingest.py +165 -0
- esbi_cli/report/__init__.py +0 -0
- esbi_cli/report/daily_index.py +235 -0
- esbi_cli/report/index_md.py +21 -0
- esbi_cli/report/readstate.py +26 -0
- esbi_cli/run.py +100 -0
- esbi_cli/runlock.py +31 -0
- esbi_cli/runlog.py +80 -0
- esbi_cli/schedule.py +106 -0
- esbi_cli/templates/SCHEMA.md +52 -0
- esbi_cli/templates/clipper-template.json +17 -0
- esbi_cli/templates/clipper-youtube-template.json +18 -0
- esbi_cli/templates/config.example.toml +108 -0
- esbi_cli/update.py +247 -0
- esbi_cli/vault.py +188 -0
- esbi_cli/wizards/clipper.sh +271 -0
- esbi_cli/wizards/email.sh +265 -0
- esbi_cli-0.2.1.dist-info/METADATA +167 -0
- esbi_cli-0.2.1.dist-info/RECORD +72 -0
- esbi_cli-0.2.1.dist-info/WHEEL +4 -0
- esbi_cli-0.2.1.dist-info/entry_points.txt +3 -0
- esbi_cli-0.2.1.dist-info/licenses/LICENSE +21 -0
esbi_cli/init.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""`sb init`: a new vault and a config file, without overwriting anything that already exists."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
import subprocess
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from esbi_cli import lang
|
|
9
|
+
from esbi_cli.gitops import STATE_IGNORE, has_git
|
|
10
|
+
|
|
11
|
+
TEMPLATES = Path(__file__).parent / "templates"
|
|
12
|
+
EXAMPLE_CONFIG = TEMPLATES / "config.example.toml" # inside the package: an installed copy has it
|
|
13
|
+
FOLDERS = (
|
|
14
|
+
"inbox",
|
|
15
|
+
"raw",
|
|
16
|
+
"attachments",
|
|
17
|
+
*(
|
|
18
|
+
f"wiki/{kind}"
|
|
19
|
+
for kind in ("sources", "concepts", "entities", "syntheses", "review", "daily")
|
|
20
|
+
),
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _files(language: str) -> dict[str, str]:
|
|
25
|
+
schema = (TEMPLATES / "SCHEMA.md").read_text(encoding="utf-8")
|
|
26
|
+
return {
|
|
27
|
+
"SCHEMA.md": schema.replace("{language}", lang.name(language)),
|
|
28
|
+
"index.md": f"# {lang.t(language, 'index_title')}\n",
|
|
29
|
+
"log.md": "# Log\n",
|
|
30
|
+
".gitignore": ".obsidian/workspace*.json\n.obsidian/cache\n.trash/\n.DS_Store\n"
|
|
31
|
+
+ "\n".join(STATE_IGNORE)
|
|
32
|
+
+ "\n",
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def init_vault(root: Path, language: str = lang.DEFAULT) -> list[str]:
|
|
37
|
+
"""Create what is missing; return what was created."""
|
|
38
|
+
created = []
|
|
39
|
+
for folder in FOLDERS:
|
|
40
|
+
if not (root / folder).is_dir():
|
|
41
|
+
(root / folder).mkdir(parents=True)
|
|
42
|
+
created.append(f"{folder}/")
|
|
43
|
+
for name, text in _files(language).items():
|
|
44
|
+
if not (root / name).exists():
|
|
45
|
+
(root / name).write_text(text, encoding="utf-8")
|
|
46
|
+
created.append(name)
|
|
47
|
+
if not (root / ".git").exists() and has_git():
|
|
48
|
+
subprocess.run(["git", "init", "-q", "-b", "main"], cwd=root, check=True)
|
|
49
|
+
created.append("git repository")
|
|
50
|
+
return created
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
MODELS = {
|
|
54
|
+
"local": "Everything stays on this Mac: the notes are written by a local model.",
|
|
55
|
+
"subscription": (
|
|
56
|
+
"The text of each source goes to Anthropic through your Claude subscription "
|
|
57
|
+
"(the official `claude` tool, `claude auth login` once). Email is read only by the local model."
|
|
58
|
+
),
|
|
59
|
+
"api": (
|
|
60
|
+
"The text of each source goes to Anthropic through its API, billed per token: set "
|
|
61
|
+
"ANTHROPIC_API_KEY in the environment. Email is read only by the local model."
|
|
62
|
+
),
|
|
63
|
+
}
|
|
64
|
+
RUNTIMES = ("ollama", "lmstudio")
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def local_model(runtime: str, name: str | None) -> str:
|
|
68
|
+
"""`<provider>/<name>` of the local model: Ollama's default, or the LM Studio identifier."""
|
|
69
|
+
return f"lmstudio/{name}" if runtime == "lmstudio" else f"ollama/{name or 'llama3.2:latest'}"
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def llm_sections(model: str, runtime: str, name: str | None, base_url: str | None = None) -> str:
|
|
73
|
+
"""The [llm.*] sections for the chosen kind of model."""
|
|
74
|
+
local = local_model(runtime, name)
|
|
75
|
+
if model == "local":
|
|
76
|
+
return (
|
|
77
|
+
f'[llm.summarize]\nmodel = "{local}"\ntimeout_seconds = 300\n'
|
|
78
|
+
+ (f"base_url = {json.dumps(base_url)}\n" if base_url else "")
|
|
79
|
+
+ ("num_ctx = 8192\nmax_tokens = 1200\n" if runtime == "ollama" else "")
|
|
80
|
+
)
|
|
81
|
+
main = "claude-cli/default" if model == "subscription" else "anthropic/claude-sonnet-5-5"
|
|
82
|
+
sections = [
|
|
83
|
+
f'[llm.summarize]\nmodel = "{main}"\nfallback = "{local}"\ntimeout_seconds = 300\n',
|
|
84
|
+
f'[llm.ask]\nmodel = "{main}"\nfallback = "{local}"\ntimeout_seconds = 600\n',
|
|
85
|
+
f'[llm.private]\nmodel = "{local}"\n', # email is read only by the local model
|
|
86
|
+
]
|
|
87
|
+
if model == "subscription":
|
|
88
|
+
sections.insert(
|
|
89
|
+
1, f'[llm.synthesize]\nmodel = "{main}"\nfallback = "{local}"\ntimeout_seconds = 600\n'
|
|
90
|
+
)
|
|
91
|
+
return "\n".join(sections)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
OCR_MODEL = "qwen3-vl:2b-instruct"
|
|
95
|
+
OCR_SECTION = f'[llm.ocr]\nmodel = "ollama/{OCR_MODEL}"\ntimeout_seconds = 600\n'
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def write_config(
|
|
99
|
+
path: Path,
|
|
100
|
+
vault: Path,
|
|
101
|
+
model: str = "local",
|
|
102
|
+
*,
|
|
103
|
+
viewer: str = "obsidian",
|
|
104
|
+
nightly: str | None = None,
|
|
105
|
+
runtime: str = "ollama",
|
|
106
|
+
local_name: str | None = None,
|
|
107
|
+
ocr: bool = False,
|
|
108
|
+
base_url: str | None = None,
|
|
109
|
+
language: str = lang.DEFAULT,
|
|
110
|
+
) -> bool:
|
|
111
|
+
"""Write the example config with this vault and choices; False if a config is already there."""
|
|
112
|
+
if path.exists():
|
|
113
|
+
return False
|
|
114
|
+
text = EXAMPLE_CONFIG.read_text(encoding="utf-8")
|
|
115
|
+
text = re.sub(r'^vault = ".*"', f'vault = "{vault}"', text, count=1, flags=re.M)
|
|
116
|
+
text = re.sub(
|
|
117
|
+
r'^language = ".*"$',
|
|
118
|
+
f'language = "{language}"\nviewer = "{viewer}"',
|
|
119
|
+
text,
|
|
120
|
+
count=1,
|
|
121
|
+
flags=re.M,
|
|
122
|
+
)
|
|
123
|
+
if nightly:
|
|
124
|
+
text = re.sub(
|
|
125
|
+
r'^nightly_time = "[^"]*"', f'nightly_time = "{nightly}"', text, count=1, flags=re.M
|
|
126
|
+
)
|
|
127
|
+
llm = llm_sections(model, runtime, local_name, base_url)
|
|
128
|
+
if ocr: # a local vision model reads images and scanned PDFs: nothing leaves this machine
|
|
129
|
+
llm += "\n" + OCR_SECTION
|
|
130
|
+
text = re.sub(
|
|
131
|
+
r"^\[llm\.summarize\].*?(?=^# Optional|^\[bench\])",
|
|
132
|
+
llm + "\n",
|
|
133
|
+
text,
|
|
134
|
+
count=1,
|
|
135
|
+
flags=re.M | re.S,
|
|
136
|
+
)
|
|
137
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
138
|
+
path.write_text(text, encoding="utf-8")
|
|
139
|
+
return True
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def connect_remote(root: Path, url: str) -> bool:
|
|
143
|
+
"""Add `origin` as the backup remote; False if the vault already has one (left alone)."""
|
|
144
|
+
if (
|
|
145
|
+
subprocess.run(
|
|
146
|
+
["git", "remote", "get-url", "origin"], cwd=root, capture_output=True
|
|
147
|
+
).returncode
|
|
148
|
+
== 0
|
|
149
|
+
):
|
|
150
|
+
return False
|
|
151
|
+
subprocess.run(["git", "remote", "add", "origin", url], cwd=root, check=True)
|
|
152
|
+
return True
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def set_email_block(path: Path, user: str, label: str = "esbi-cli") -> None:
|
|
156
|
+
"""Turn on the [email] section of a config file for this Gmail address and label, replacing the
|
|
157
|
+
block if there is one and touching nothing else. The password is never written here."""
|
|
158
|
+
block = (
|
|
159
|
+
"[email]\n"
|
|
160
|
+
"enabled = true\n"
|
|
161
|
+
'imap_host = "imap.gmail.com"\n'
|
|
162
|
+
f'mailbox = "{label}"\n'
|
|
163
|
+
f'user = "{user}"\n'
|
|
164
|
+
"# The app password lives in the macOS Keychain (service esbi-cli-imap), never in this file.\n"
|
|
165
|
+
)
|
|
166
|
+
text = path.read_text(encoding="utf-8")
|
|
167
|
+
match = re.search(r"^\[email\]\n.*?(?=^\[|\Z)", text, re.S | re.M)
|
|
168
|
+
if match:
|
|
169
|
+
kept = [ # choices made by hand survive a re-run of the setup
|
|
170
|
+
line for line in match.group().splitlines() if line.startswith("follow_links")
|
|
171
|
+
]
|
|
172
|
+
block += "".join(line + "\n" for line in kept)
|
|
173
|
+
text = text[: match.start()] + block + "\n" + text[match.end() :]
|
|
174
|
+
else:
|
|
175
|
+
text = text.rstrip("\n") + "\n\n" + block
|
|
176
|
+
path.write_text(text, encoding="utf-8")
|
esbi_cli/interrupts.py
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""SIGINT and SIGTERM as an exception in the main thread, so the code that owns a resource can put
|
|
2
|
+
it back. Handlers exist only inside `handling()`. Within it a signal raises `Interrupted` only
|
|
3
|
+
inside `interruptible()` and outside `deferred()`; anywhere else it waits for the next safe point,
|
|
4
|
+
so a note is written whole or not at all.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import signal
|
|
8
|
+
import sys
|
|
9
|
+
import threading
|
|
10
|
+
from contextlib import contextmanager
|
|
11
|
+
|
|
12
|
+
SIGNALS = (signal.SIGINT, signal.SIGTERM)
|
|
13
|
+
|
|
14
|
+
# ponytail: module state, one run per process; a per-run object if two runs ever share a process
|
|
15
|
+
_depth = 0 # a signal raises only at 0
|
|
16
|
+
_pending: int | None = None # the first signal received; later ones are ignored while it is handled
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class Interrupted(BaseException):
|
|
20
|
+
"""Not an Exception, like KeyboardInterrupt: `except Exception` must not turn it into a failure."""
|
|
21
|
+
|
|
22
|
+
def __init__(self, signum: int):
|
|
23
|
+
super().__init__(signum)
|
|
24
|
+
self.signum = signum
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _handle(signum, frame) -> None:
|
|
28
|
+
global _pending
|
|
29
|
+
if _pending is None:
|
|
30
|
+
_pending = signum
|
|
31
|
+
if _depth == 0:
|
|
32
|
+
raise Interrupted(signum)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _raise_if_pending() -> None:
|
|
36
|
+
if _depth == 0 and _pending is not None:
|
|
37
|
+
raise Interrupted(_pending)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@contextmanager
|
|
41
|
+
def handling():
|
|
42
|
+
"""Install the handlers for the block, then restore the previous ones. The block is protected
|
|
43
|
+
until `interruptible()`. Outside the main thread nothing is installed."""
|
|
44
|
+
global _depth, _pending
|
|
45
|
+
if threading.current_thread() is not threading.main_thread(): # signal.signal needs it
|
|
46
|
+
yield
|
|
47
|
+
return
|
|
48
|
+
previous = {s: signal.signal(s, _handle) for s in SIGNALS}
|
|
49
|
+
_depth, _pending = 1, None
|
|
50
|
+
try:
|
|
51
|
+
yield
|
|
52
|
+
finally:
|
|
53
|
+
_depth, _pending = 0, None
|
|
54
|
+
for s, handler in previous.items():
|
|
55
|
+
signal.signal(s, handler)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@contextmanager
|
|
59
|
+
def interruptible():
|
|
60
|
+
"""A signal raises `Interrupted` inside this block (one that came earlier raises on entry)."""
|
|
61
|
+
global _depth
|
|
62
|
+
_depth -= 1
|
|
63
|
+
try:
|
|
64
|
+
_raise_if_pending()
|
|
65
|
+
yield
|
|
66
|
+
finally:
|
|
67
|
+
_depth += 1
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@contextmanager
|
|
71
|
+
def deferred():
|
|
72
|
+
"""A signal inside this block waits until it ends. Costs nothing when no handler is installed."""
|
|
73
|
+
global _depth
|
|
74
|
+
_depth += 1
|
|
75
|
+
try:
|
|
76
|
+
yield
|
|
77
|
+
finally:
|
|
78
|
+
_depth -= 1
|
|
79
|
+
_raise_if_pending() # normal exit only: a real error is not replaced
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@contextmanager
|
|
83
|
+
def exit_when_interrupted():
|
|
84
|
+
"""For commands with no queue (`sb ingest`, `sb reingest`): stop with the conventional code."""
|
|
85
|
+
try:
|
|
86
|
+
with handling(), interruptible():
|
|
87
|
+
yield
|
|
88
|
+
except Interrupted as exc:
|
|
89
|
+
print("Interrupted.", file=sys.stderr)
|
|
90
|
+
raise SystemExit(128 + exc.signum) from None
|
esbi_cli/lang.py
ADDED
|
@@ -0,0 +1,341 @@
|
|
|
1
|
+
"""The languages the wiki can be written in: one catalogue, nothing else knows a language.
|
|
2
|
+
|
|
3
|
+
Code asks for text by language-independent key (`t("es", "summary")`) and recognises headings of
|
|
4
|
+
every catalogued language (`key_of`), so a vault written in one language and later switched to
|
|
5
|
+
another still reads. Adding a language is adding one entry to `LANGUAGES`: see
|
|
6
|
+
https://rubenamaury.github.io/esbi-cli/docs/how-to/add-a-language/.
|
|
7
|
+
|
|
8
|
+
Each entry has:
|
|
9
|
+
- `name`: how the prompts call the language ("Write all text in Spanish").
|
|
10
|
+
- `hint`: an optional extra line for the prompts, for a language a small model needs more help with.
|
|
11
|
+
- `stopwords`: common words used to notice an answer in the wrong language; empty skips the check.
|
|
12
|
+
- `generic_terms`: words too general to be a glossary entry ("data", "system"); they are dropped.
|
|
13
|
+
- `disclaimers`: a regex for a definition that says it has none ("not defined in the text").
|
|
14
|
+
- `relation_examples`: short labels for how two ideas relate, as the model should write them.
|
|
15
|
+
- `placeholders`: how a model that copied the prompt's wording starts a "summary" ("Executive summary of ...").
|
|
16
|
+
- `ask_example` / `rewrite_example`: worked outputs shown to the model, in this language.
|
|
17
|
+
- `labels`: every piece of text the program writes into the wiki.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
import re
|
|
21
|
+
import unicodedata
|
|
22
|
+
from collections.abc import Iterable
|
|
23
|
+
|
|
24
|
+
DEFAULT = "en"
|
|
25
|
+
|
|
26
|
+
LANGUAGES: dict[str, dict] = {
|
|
27
|
+
"en": {
|
|
28
|
+
"name": "English",
|
|
29
|
+
"hint": "",
|
|
30
|
+
"relation_examples": '"extends", "complements", "improves", "uses", "is an example of"',
|
|
31
|
+
"stopwords": "the of and to in is that for with are this on as by from be an",
|
|
32
|
+
"generic_terms": "data information system process technology example method approach result problem",
|
|
33
|
+
"disclaimers": r"not (defined|mentioned|specified|provided|explained)|(does|do) not (define|mention|specify|explain|provide)|no definition",
|
|
34
|
+
"placeholders": ("executive summary", "summary of"),
|
|
35
|
+
"ask_example": (
|
|
36
|
+
'{"title": "What is a graph", "one_liner": "A graph is a set of nodes joined by edges.", '
|
|
37
|
+
'"answer": "A graph is a set of nodes connected by edges [[Graph]]. '
|
|
38
|
+
'It is used to model networks [[Networks]].", "cited_pages": ["Graph", "Networks"]}'
|
|
39
|
+
),
|
|
40
|
+
"rewrite_example": (
|
|
41
|
+
'question "When should a judge accept and when escalate?" -> {"terms": ["judge", '
|
|
42
|
+
'"accept", "escalate", "confidence", "confident", "threshold", "review"]}'
|
|
43
|
+
),
|
|
44
|
+
"labels": {
|
|
45
|
+
# sections of a source note
|
|
46
|
+
"summary": "Executive summary",
|
|
47
|
+
"abstract": "Detailed summary",
|
|
48
|
+
"insights": "Key ideas",
|
|
49
|
+
"key_points": "Key points",
|
|
50
|
+
"terms": "Key terms",
|
|
51
|
+
"quotes": "Key quotes",
|
|
52
|
+
"diagram": "Diagram",
|
|
53
|
+
"figures": "Figures",
|
|
54
|
+
"connections": "Connections to your wiki",
|
|
55
|
+
"open_questions": "Open questions",
|
|
56
|
+
"concepts": "Concepts",
|
|
57
|
+
"entities": "Entities",
|
|
58
|
+
"related": "Related",
|
|
59
|
+
"contradictions": "Possible contradictions",
|
|
60
|
+
# elsewhere in notes
|
|
61
|
+
"from_source": "From",
|
|
62
|
+
"original_source": "Original source",
|
|
63
|
+
"untitled": "Untitled",
|
|
64
|
+
"source_suffix": "(source)",
|
|
65
|
+
"figure": "Figure",
|
|
66
|
+
"image": "Image",
|
|
67
|
+
"no_subject": "No subject",
|
|
68
|
+
"page_abbr": "p.",
|
|
69
|
+
"contradiction_callout": "Possible contradiction ({date}) with {link}: {note}",
|
|
70
|
+
"contradiction_review_name": "{date} contradiction - {source}",
|
|
71
|
+
"contradiction_review_title": "Possible contradictions from {link}",
|
|
72
|
+
"contradiction_review_footer": "Review and resolve; then delete this note.",
|
|
73
|
+
# ask
|
|
74
|
+
"no_answer": "I find nothing about this in the wiki.",
|
|
75
|
+
"question_label": "Question",
|
|
76
|
+
"sources": "Sources",
|
|
77
|
+
"synthesis_suffix": "(synthesis)",
|
|
78
|
+
"answer_title": "Answer",
|
|
79
|
+
# index.md and log.md
|
|
80
|
+
"index_title": "Index",
|
|
81
|
+
"index_blurb": "Catalogue of the wiki ({total} pages). Kept by the worker.",
|
|
82
|
+
"syntheses": "Syntheses",
|
|
83
|
+
"log_created": "created: {names}",
|
|
84
|
+
"log_updated": "updated: {names}",
|
|
85
|
+
"log_review": "to review: {n}",
|
|
86
|
+
# the daily index
|
|
87
|
+
"daily_title": "Index of {date}",
|
|
88
|
+
"daily_read": "Read",
|
|
89
|
+
"daily_processed": "Processed today",
|
|
90
|
+
"daily_queue": "Tomorrow's queue",
|
|
91
|
+
"daily_review": "To review",
|
|
92
|
+
"daily_revisit": "Revisit",
|
|
93
|
+
"daily_runs": "Runs",
|
|
94
|
+
"daily_stats": "Statistics",
|
|
95
|
+
"nothing_read": "_Nothing new marked as read._",
|
|
96
|
+
"nothing_processed": "_Nothing was processed today._",
|
|
97
|
+
"queue_empty": "_The queue is empty._",
|
|
98
|
+
"queue_one": "1 source in the queue",
|
|
99
|
+
"queue_many": "{n} sources in the queue",
|
|
100
|
+
"queue_cap": " (showing the {n} oldest)",
|
|
101
|
+
"nothing_pending": "_Nothing pending._",
|
|
102
|
+
"could_not_process": "- Could not process {name} ({error})",
|
|
103
|
+
"retrying": "- Retrying {name} (attempt {n} of {max}): {error}",
|
|
104
|
+
"no_old_notes": "_No older notes to revisit yet._",
|
|
105
|
+
"stat_pages": "- Pages: {total} (sources: {sources}, concepts: {concepts}, entities: {entities})",
|
|
106
|
+
"stat_today": "- Today: {processed} sources processed, {touched} concepts/entities updated",
|
|
107
|
+
"stat_read": "- Read: {read} of {total} sources",
|
|
108
|
+
"stat_orphans": "- Orphans (no incoming links): {n}",
|
|
109
|
+
"no_runs": "_No runs yet today._",
|
|
110
|
+
"run_manual": " (manual)",
|
|
111
|
+
"run_line": "- {at}{manual} — processed: {ingested}, skipped: {skipped}, failed: {failed} · {tokens} tokens · {duration}",
|
|
112
|
+
"run_minutes": "{n} min",
|
|
113
|
+
"run_under_minute": "<1 min",
|
|
114
|
+
"run_stopped": " · stopped: {reason}",
|
|
115
|
+
"stop_max_sources": "source limit",
|
|
116
|
+
"stop_token_budget": "token budget",
|
|
117
|
+
"stop_llm_unavailable": "model unavailable",
|
|
118
|
+
"home_today": "- Today's index: {link}",
|
|
119
|
+
"home_unread": "- Unread: {n} sources",
|
|
120
|
+
"home_queue": "- Queued: {n} sources",
|
|
121
|
+
# the lint report
|
|
122
|
+
"lint_title": "Lint report ({date})",
|
|
123
|
+
"lint_orphan": "Orphan pages",
|
|
124
|
+
"lint_broken_link": "Broken links",
|
|
125
|
+
"lint_missing_field": "Missing fields",
|
|
126
|
+
"lint_unlinked_mention": "Unlinked mentions",
|
|
127
|
+
"lint_near_duplicate": "Possible duplicates",
|
|
128
|
+
"lint_missing_concept": "Concepts without a page",
|
|
129
|
+
"lint_missing_field_line": "- [[{page}]]: missing `{field}`",
|
|
130
|
+
"lint_unlinked_line": "- [[{page}]] mentions [[{target}]] without linking it",
|
|
131
|
+
"lint_missing_concept_line": "- «{name}» appears in {sources} and has no page of its own",
|
|
132
|
+
"lint_more": "… and {n} more",
|
|
133
|
+
"lint_footer": "The worker only reports; nothing is fixed by itself. This report is regenerated on every run.",
|
|
134
|
+
# the benchmark report
|
|
135
|
+
"bench_title": "Model benchmark ({when})",
|
|
136
|
+
"bench_header": "| model | success | no retry | median (s) | tokens/case | est. cost (USD) | {extra} |",
|
|
137
|
+
"bench_extra_ingest": "right language / concepts",
|
|
138
|
+
"bench_extra_ask": "cites the expected page",
|
|
139
|
+
"bench_routing": "Routing recommendation",
|
|
140
|
+
"bench_none": "no recommendation (no reliable model)",
|
|
141
|
+
"bench_footer": "Only a suggestion: change `[llm.*]` in config.toml by hand if you agree.",
|
|
142
|
+
"bench_question": "What is {title}?",
|
|
143
|
+
},
|
|
144
|
+
},
|
|
145
|
+
"es": {
|
|
146
|
+
"name": "Spanish",
|
|
147
|
+
"hint": "",
|
|
148
|
+
"relation_examples": '"amplía", "complementa", "mejora", "usa", "es un ejemplo de"',
|
|
149
|
+
"stopwords": "de la el que en los las y un una para con por del se es al como más pero sus",
|
|
150
|
+
"generic_terms": "datos información sistema proceso tecnología ejemplo método enfoque resultado problema",
|
|
151
|
+
"disclaimers": r"no (se )?(define|menciona|especifica|explica|proporciona|detalla)|no (est[aá]|aparece) (definid|especificad|en el texto)|sin definici[oó]n",
|
|
152
|
+
"placeholders": ("resumen ejecutivo", "resumen de"),
|
|
153
|
+
"ask_example": (
|
|
154
|
+
'{"title": "Qué es un grafo", "one_liner": "Un grafo es un conjunto de nodos unidos por aristas.", '
|
|
155
|
+
'"answer": "Un grafo es un conjunto de nodos conectados por aristas [[Grafo]]. '
|
|
156
|
+
'Se usa para modelar redes [[Redes]].", "cited_pages": ["Grafo", "Redes"]}'
|
|
157
|
+
),
|
|
158
|
+
"rewrite_example": (
|
|
159
|
+
'question "¿Cuándo debe un juez aceptar y cuándo escalar?" -> {"terms": ["juez", "judge", '
|
|
160
|
+
'"aceptar", "accept", "escalar", "escalate", "confianza", "confident"]}'
|
|
161
|
+
),
|
|
162
|
+
"labels": {
|
|
163
|
+
"summary": "Resumen ejecutivo",
|
|
164
|
+
"abstract": "Resumen detallado",
|
|
165
|
+
"insights": "Ideas clave",
|
|
166
|
+
"key_points": "Puntos clave",
|
|
167
|
+
"terms": "Términos clave",
|
|
168
|
+
"quotes": "Frases clave",
|
|
169
|
+
"diagram": "Diagrama",
|
|
170
|
+
"figures": "Figuras",
|
|
171
|
+
"connections": "Conexiones con tu wiki",
|
|
172
|
+
"open_questions": "Preguntas abiertas",
|
|
173
|
+
"concepts": "Conceptos",
|
|
174
|
+
"entities": "Entidades",
|
|
175
|
+
"related": "Relacionado",
|
|
176
|
+
"contradictions": "Posibles contradicciones",
|
|
177
|
+
"from_source": "Desde",
|
|
178
|
+
"original_source": "Fuente original",
|
|
179
|
+
"untitled": "Sin título",
|
|
180
|
+
"source_suffix": "(fuente)",
|
|
181
|
+
"figure": "Figura",
|
|
182
|
+
"image": "Imagen",
|
|
183
|
+
"no_subject": "Sin asunto",
|
|
184
|
+
"page_abbr": "p.",
|
|
185
|
+
"contradiction_callout": "Posible contradicción ({date}) con {link}: {note}",
|
|
186
|
+
"contradiction_review_name": "{date} contradicción - {source}",
|
|
187
|
+
"contradiction_review_title": "Posibles contradicciones desde {link}",
|
|
188
|
+
"contradiction_review_footer": "Revisa y resuelve; luego borra esta nota.",
|
|
189
|
+
"no_answer": "No encuentro nada sobre esto en la wiki.",
|
|
190
|
+
"question_label": "Pregunta",
|
|
191
|
+
"sources": "Fuentes",
|
|
192
|
+
"synthesis_suffix": "(síntesis)",
|
|
193
|
+
"answer_title": "Respuesta",
|
|
194
|
+
"index_title": "Índice",
|
|
195
|
+
"index_blurb": "Catálogo de la wiki ({total} páginas). Lo mantiene el worker.",
|
|
196
|
+
"syntheses": "Síntesis",
|
|
197
|
+
"log_created": "creadas: {names}",
|
|
198
|
+
"log_updated": "actualizadas: {names}",
|
|
199
|
+
"log_review": "por revisar: {n}",
|
|
200
|
+
"daily_title": "Índice del {date}",
|
|
201
|
+
"daily_read": "Leído",
|
|
202
|
+
"daily_processed": "Procesado hoy",
|
|
203
|
+
"daily_queue": "Cola de mañana",
|
|
204
|
+
"daily_review": "Por revisar",
|
|
205
|
+
"daily_revisit": "Repasar",
|
|
206
|
+
"daily_runs": "Ejecuciones",
|
|
207
|
+
"daily_stats": "Estadísticas",
|
|
208
|
+
"nothing_read": "_Nada nuevo marcado como leído._",
|
|
209
|
+
"nothing_processed": "_Hoy no se procesó nada._",
|
|
210
|
+
"queue_empty": "_La cola está vacía._",
|
|
211
|
+
"queue_one": "1 fuente en cola",
|
|
212
|
+
"queue_many": "{n} fuentes en cola",
|
|
213
|
+
"queue_cap": " (se muestran las {n} más antiguas)",
|
|
214
|
+
"nothing_pending": "_Nada pendiente._",
|
|
215
|
+
"could_not_process": "- No se pudo procesar {name} ({error})",
|
|
216
|
+
"retrying": "- Reintentando {name} (intento {n} de {max}): {error}",
|
|
217
|
+
"no_old_notes": "_Todavía no hay notas antiguas para repasar._",
|
|
218
|
+
"stat_pages": "- Páginas: {total} (fuentes: {sources}, conceptos: {concepts}, entidades: {entities})",
|
|
219
|
+
"stat_today": "- Hoy: {processed} fuentes procesadas, {touched} conceptos/entidades actualizados",
|
|
220
|
+
"stat_read": "- Leídas: {read} de {total} fuentes",
|
|
221
|
+
"stat_orphans": "- Huérfanas (sin enlaces entrantes): {n}",
|
|
222
|
+
"no_runs": "_Todavía no hubo ejecuciones hoy._",
|
|
223
|
+
"run_manual": " (manual)",
|
|
224
|
+
"run_line": "- {at}{manual} — procesadas: {ingested}, omitidas: {skipped}, fallidas: {failed} · {tokens} tokens · {duration}",
|
|
225
|
+
"run_minutes": "{n} min",
|
|
226
|
+
"run_under_minute": "<1 min",
|
|
227
|
+
"run_stopped": " · detenida: {reason}",
|
|
228
|
+
"stop_max_sources": "límite de fuentes",
|
|
229
|
+
"stop_token_budget": "presupuesto de tokens",
|
|
230
|
+
"stop_llm_unavailable": "LLM no disponible",
|
|
231
|
+
"home_today": "- Índice de hoy: {link}",
|
|
232
|
+
"home_unread": "- Sin leer: {n} fuentes",
|
|
233
|
+
"home_queue": "- En cola: {n} fuentes",
|
|
234
|
+
"lint_title": "Informe de lint ({date})",
|
|
235
|
+
"lint_orphan": "Páginas huérfanas",
|
|
236
|
+
"lint_broken_link": "Enlaces rotos",
|
|
237
|
+
"lint_missing_field": "Campos que faltan",
|
|
238
|
+
"lint_unlinked_mention": "Menciones sin enlazar",
|
|
239
|
+
"lint_near_duplicate": "Posibles duplicados",
|
|
240
|
+
"lint_missing_concept": "Conceptos sin página",
|
|
241
|
+
"lint_missing_field_line": "- [[{page}]]: falta `{field}`",
|
|
242
|
+
"lint_unlinked_line": "- [[{page}]] menciona [[{target}]] sin enlazarlo",
|
|
243
|
+
"lint_missing_concept_line": "- «{name}» aparece en {sources} y no tiene página propia",
|
|
244
|
+
"lint_more": "… y {n} más",
|
|
245
|
+
"lint_footer": "El worker solo informa; nada se corrige solo. Este informe se regenera en cada ejecución.",
|
|
246
|
+
"bench_title": "Benchmark de modelos ({when})",
|
|
247
|
+
"bench_header": "| modelo | éxito | sin reintento | mediana (s) | tokens/caso | coste est. (USD) | {extra} |",
|
|
248
|
+
"bench_extra_ingest": "idioma correcto / conceptos",
|
|
249
|
+
"bench_extra_ask": "cita la página esperada",
|
|
250
|
+
"bench_routing": "Recomendación de enrutado",
|
|
251
|
+
"bench_none": "sin recomendación (ningún modelo fiable)",
|
|
252
|
+
"bench_footer": "Solo es una sugerencia: cambia `[llm.*]` en config.toml a mano si te convence.",
|
|
253
|
+
"bench_question": "¿Qué es {title}?",
|
|
254
|
+
},
|
|
255
|
+
},
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def supported() -> str:
|
|
260
|
+
return ", ".join(LANGUAGES)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def get(code: str) -> dict:
|
|
264
|
+
"""The catalogue entry for `code`; a clear error naming the supported languages otherwise."""
|
|
265
|
+
try:
|
|
266
|
+
return LANGUAGES[code]
|
|
267
|
+
except (KeyError, TypeError):
|
|
268
|
+
raise ValueError(f"[notes].language must be one of: {supported()}, got {code!r}") from None
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def name(code: str) -> str:
|
|
272
|
+
return get(code)["name"]
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def t(code: str, key: str, **values) -> str:
|
|
276
|
+
"""The text for `key` in language `code`, with `{placeholders}` filled in."""
|
|
277
|
+
return get(code)["labels"][key].format(**values)
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
def _fold(text: str) -> str:
|
|
281
|
+
plain = "".join(c for c in unicodedata.normalize("NFKD", text) if not unicodedata.combining(c))
|
|
282
|
+
return " ".join(plain.casefold().split())
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def every(key: str) -> list[str]:
|
|
286
|
+
"""The label for `key` in every language: what a reader must accept."""
|
|
287
|
+
return list(dict.fromkeys(entry["labels"][key] for entry in LANGUAGES.values()))
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def key_of(heading: str) -> str | None:
|
|
291
|
+
"""The key a heading belongs to, in whatever language it was written; None for any other."""
|
|
292
|
+
wanted = _fold(heading)
|
|
293
|
+
for entry in LANGUAGES.values():
|
|
294
|
+
for key, label in entry["labels"].items():
|
|
295
|
+
if _fold(label) == wanted:
|
|
296
|
+
return key
|
|
297
|
+
return None
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def instruction(code: str) -> str:
|
|
301
|
+
"""The sentence every prompt carries to say which language to write in."""
|
|
302
|
+
entry = get(code)
|
|
303
|
+
return (
|
|
304
|
+
f"Write ALL text in {entry['name']}, whatever language the source or the SCHEMA is in; "
|
|
305
|
+
f"only verbatim quotes and term names keep the source's language. {entry['hint']}"
|
|
306
|
+
).strip()
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _hits(text: str, code: str) -> int:
|
|
310
|
+
words = set(LANGUAGES[code]["stopwords"].split())
|
|
311
|
+
return sum(w in words for w in re.findall(r"[^\W\d_]+", text.lower()))
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def wrong_language(text: str, code: str, min_hits: int = 4) -> bool:
|
|
315
|
+
"""Is `text` clearly in another catalogued language than `code`? A cheap stopword count: small
|
|
316
|
+
models often answer in the source's language. A language with no stopwords is never judged.
|
|
317
|
+
`min_hits` is how many stopwords it takes; a one-sentence definition needs fewer."""
|
|
318
|
+
if not get(code)["stopwords"]:
|
|
319
|
+
return False
|
|
320
|
+
own = _hits(text, code)
|
|
321
|
+
return any(
|
|
322
|
+
_hits(text, other) >= min_hits and _hits(text, other) > 1.5 * own
|
|
323
|
+
for other, entry in LANGUAGES.items()
|
|
324
|
+
if other != code and entry["stopwords"]
|
|
325
|
+
)
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
SHORT_FIELD_CHARS = 300 # under this, a field is a sentence or two: fewer stopwords are enough
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def leaking(fields: Iterable[tuple[str, str]], code: str) -> list[str]:
|
|
332
|
+
"""The names of the (name, text) fields written in another catalogued language than `code`,
|
|
333
|
+
each field judged on its own: joined with a long field in the right language, a short one in
|
|
334
|
+
the wrong language goes unnoticed. A language without a word list is never judged."""
|
|
335
|
+
return list(
|
|
336
|
+
dict.fromkeys(
|
|
337
|
+
name
|
|
338
|
+
for name, text in fields
|
|
339
|
+
if wrong_language(text, code, 4 if len(text) >= SHORT_FIELD_CHARS else 2)
|
|
340
|
+
)
|
|
341
|
+
)
|
esbi_cli/links.py
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Wikilink parsing shared by the daily index and lint."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
WIKILINK = re.compile(r"\[\[([^\]|#]+)")
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def link_targets(text: str) -> list[str]:
|
|
9
|
+
"""Page names linked from text: `[[Title]]`, `[[Title|alias]]` and `[[Title#Heading]]` all give Title."""
|
|
10
|
+
return [t.strip() for t in WIKILINK.findall(text)]
|
|
File without changes
|