timbro 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. timbro/__init__.py +11 -0
  2. timbro/analyze.py +449 -0
  3. timbro/cleanup/__init__.py +15 -0
  4. timbro/cleanup/latex.py +118 -0
  5. timbro/cleanup/papers.py +261 -0
  6. timbro/cli.py +251 -0
  7. timbro/concreteness.py +94 -0
  8. timbro/config.py +142 -0
  9. timbro/flow.py +128 -0
  10. timbro/fw.py +88 -0
  11. timbro/hedge.py +104 -0
  12. timbro/lexicons/boosters.txt +66 -0
  13. timbro/lexicons/connectives_conditional.txt +17 -0
  14. timbro/lexicons/hedges.txt +90 -0
  15. timbro/lexicons/hype.txt +61 -0
  16. timbro/lexicons/negations.txt +22 -0
  17. timbro/lexicons/plain_wording.txt +223 -0
  18. timbro/mcp_server.py +80 -0
  19. timbro/metric.py +120 -0
  20. timbro/model.py +510 -0
  21. timbro/norms/NOTICE.md +9 -0
  22. timbro/norms/concreteness_brysbaert2014.csv.gz +0 -0
  23. timbro/profiles.py +253 -0
  24. timbro/report.py +224 -0
  25. timbro/rewrite.py +63 -0
  26. timbro/rubrics/__init__.py +23 -0
  27. timbro/rubrics/base.py +50 -0
  28. timbro/rubrics/density/__init__.py +5 -0
  29. timbro/rubrics/density/checks.py +127 -0
  30. timbro/rubrics/density/rubric.py +24 -0
  31. timbro/rubrics/features.py +804 -0
  32. timbro/rubrics/registry.py +15 -0
  33. timbro/rubrics/report.py +51 -0
  34. timbro/rubrics/rules.py +496 -0
  35. timbro/rubrics/schimel/__init__.py +5 -0
  36. timbro/rubrics/schimel/rubric.py +21 -0
  37. timbro/rubrics/sections.py +32 -0
  38. timbro/rubrics/slop/__init__.py +5 -0
  39. timbro/rubrics/slop/checks.py +91 -0
  40. timbro/rubrics/slop/rubric.py +26 -0
  41. timbro/sample/contrast/01-synergy.md +9 -0
  42. timbro/sample/contrast/02-revolutionize.md +11 -0
  43. timbro/sample/contrast/03-paradigm.md +9 -0
  44. timbro/sample/exemplars/01-shipping.md +9 -0
  45. timbro/sample/exemplars/02-debugging.md +9 -0
  46. timbro/sample/exemplars/03-reviews.md +9 -0
  47. timbro/sample/exemplars/04-estimates.md +9 -0
  48. timbro/spacy_model.py +30 -0
  49. timbro/tells.py +324 -0
  50. timbro/text.py +169 -0
  51. timbro-0.6.0.dist-info/METADATA +309 -0
  52. timbro-0.6.0.dist-info/RECORD +55 -0
  53. timbro-0.6.0.dist-info/WHEEL +4 -0
  54. timbro-0.6.0.dist-info/entry_points.txt +3 -0
  55. timbro-0.6.0.dist-info/licenses/LICENSE +21 -0
timbro/__init__.py ADDED
@@ -0,0 +1,11 @@
1
+ from timbro.model import VoiceModel, ScoreResult, FeatureMove, MarkdownAxis, features, read_corpus
2
+ from timbro.flow import FlowReport, flow_report
3
+ from timbro.profiles import Profile, add_file, add_text, get_profile, init_profile, list_profiles
4
+ from timbro.rubrics import check_text
5
+
6
+ __all__ = [
7
+ "VoiceModel", "ScoreResult", "FeatureMove", "MarkdownAxis", "features", "read_corpus",
8
+ "FlowReport", "flow_report",
9
+ "Profile", "get_profile", "list_profiles", "init_profile", "add_text", "add_file",
10
+ "check_text",
11
+ ]
timbro/analyze.py ADDED
@@ -0,0 +1,449 @@
1
+ """``timbro analyze`` (#17): one deterministic linguistic feature vector per document.
2
+
3
+ No LLM, no network at analyze time. Profiles/rubrics/scoring/verdicts are untouched --
4
+ this is a standalone slice for the SKILL.md linguistic-topology paper (paper/README.md
5
+ WS2 Issue A). `struct_*` runs on the raw markdown; everything else runs on prose
6
+ isolated via the #16 markup stripper (`timbro.text.strip_markup`).
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import csv
12
+ import json
13
+ import re
14
+ import sys
15
+ from collections import Counter
16
+ from functools import lru_cache
17
+ from pathlib import Path
18
+
19
+ import textdescriptives as td
20
+ import yaml
21
+ from lexical_diversity import lex_div
22
+ from wordfreq import zipf_frequency
23
+
24
+ from timbro.model import POS_TAGS
25
+ from timbro.text import strip_markup
26
+
27
+ _FRONTMATTER = re.compile(r"\A---[ \t]*\n(.*?)\n---[ \t]*\n?", re.S)
28
+ _FENCE = re.compile(r"(```|~~~).*?\1", re.S)
29
+ _HEADING = re.compile(r"(?m)^[ \t]*(#{1,6})[ \t]+.*$")
30
+ _TABLE_SEPARATOR = re.compile(r"(?m)^[ \t]*:?-{2,}:?(?:[ \t]*\|[ \t]*:?-{2,}:?)+[ \t]*$")
31
+ _BULLET_LIST = re.compile(r"^[ \t]*[-*+][ \t]+")
32
+ _ORDERED_LIST = re.compile(r"^[ \t]*\d+\.[ \t]+")
33
+ _BLANK_LINE = re.compile(r"\n[ \t]*\n")
34
+ _SENTENCE_END = re.compile(r"[.!?]+") # naive sentence count; _struct runs without spaCy
35
+
36
+ # Folk-advice exploratory features (#21).
37
+ _INLINE_CODE_SPAN = re.compile(r"`([^`\n]+)`")
38
+ _MD_LINK = re.compile(r"\[(.*?)\]\((.*?)\)")
39
+ _EXTERNAL_REF = re.compile(r"(scripts/|references/|assets/)[^\s)\]]*")
40
+ _NAMED_SECTIONS = (
41
+ "examples", "guidelines", "when to use", "procedure", "pitfalls", "usage", "instructions",
42
+ )
43
+ _NAME_FORMAT = re.compile(r"[a-z0-9-]{1,64}")
44
+ _HEADING_MARK = re.compile(r"^[ \t]*#{1,6}[ \t]+")
45
+ _FM_WHEN_CLAUSE = re.compile(r"\b(when|use (this|it) (when|for|to)|whenever|if you)\b", re.I)
46
+ _FM_OR_WORD = re.compile(r"\bor\b", re.I)
47
+ _FM_WILDCARD = re.compile(r"\b(any|all|every|always|whenever|anything|everything)\b", re.I)
48
+ _ALLCAPS = re.compile(r"\b[A-Z][A-Z']+\b")
49
+ _ALLCAPS_DIRECTIVES = {"ALWAYS", "NEVER", "MUST", "NOT", "DON'T", "DO"}
50
+ _CONTRASTIVE = re.compile(r"^(do|don'?t|correct|incorrect|good|bad)\s*:", re.I)
51
+ _CONTRASTIVE_SYMBOLS = ("✅", "❌", "✓", "✗")
52
+
53
+ _CONTENT_POS = {"NOUN", "PROPN", "VERB", "ADJ", "ADV"}
54
+ _COH_CONTENT_POS = {"NOUN", "PROPN", "VERB", "ADJ"}
55
+ _CLAUSAL_DEPS = {"ccomp", "xcomp", "advcl", "acl", "relcl"}
56
+ _CONDITIONAL_MARKS = {"if", "unless", "when"}
57
+ _SECOND_PERSON = {"you", "your", "yours", "you're", "yourself"}
58
+ _CROSS_REFERENCE = re.compile(
59
+ r"\b(?:see also|see below|see above|refer to|as described in|cf\.)", re.I
60
+ )
61
+ _LEXICON_DIR = Path(__file__).parent / "lexicons"
62
+
63
+
64
+ @lru_cache(maxsize=None)
65
+ def _lexicon(name: str) -> tuple[tuple[str, ...], ...]:
66
+ lines = (_LEXICON_DIR / name).read_text(encoding="utf-8").splitlines()
67
+ entries = (ln.strip() for ln in lines if ln.strip() and not ln.startswith("#"))
68
+ return tuple(tuple(entry.lower().split()) for entry in entries)
69
+
70
+
71
+ @lru_cache(maxsize=None)
72
+ def _plain_pairs() -> tuple[tuple[tuple[str, ...], str], ...]:
73
+ lines = (_LEXICON_DIR / "plain_wording.txt").read_text(encoding="utf-8").splitlines()
74
+ pairs = []
75
+ for ln in lines:
76
+ if not ln.strip() or ln.startswith("#"):
77
+ continue
78
+ complex_side, _, plain = ln.partition("\t")
79
+ pairs.append((tuple(complex_side.lower().split()), plain.strip()))
80
+ return tuple(pairs)
81
+
82
+
83
+ @lru_cache(maxsize=None)
84
+ def _hype_entries() -> tuple[tuple[str, ...], ...]:
85
+ # Hype is matched on surface FORM, not lemma: en_core_web_sm mangles participial adjectives
86
+ # ("groundbreaking" -> "groundbreake") and splits hyphenated compounds into three tokens
87
+ # ("world-class" -> world - class), so the lemma-matching used by the frozen boosters/hedges
88
+ # lexicons silently drops most of this list. Entries are tokenized the same way spaCy splits
89
+ # (hyphen kept as its own token) so multiword and hyphenated terms line up against the forms.
90
+ lines = (_LEXICON_DIR / "hype.txt").read_text(encoding="utf-8").splitlines()
91
+ entries = []
92
+ for ln in lines:
93
+ stripped = ln.strip()
94
+ if not stripped or stripped.startswith("#"):
95
+ continue
96
+ entries.append(tuple(stripped.lower().replace("-", " - ").split()))
97
+ return tuple(entries)
98
+
99
+
100
+ def _lexicon_matches(lemmas: list[str], entries: tuple[tuple[str, ...], ...]) -> int:
101
+ count = 0
102
+ for entry in entries:
103
+ width = len(entry)
104
+ for i in range(len(lemmas) - width + 1):
105
+ if tuple(lemmas[i : i + width]) == entry:
106
+ count += 1
107
+ return count
108
+
109
+ # textdescriptives' extract_df column -> our snake_case name (exact renames are decisions
110
+ # from the #17 spec: syn_mean_dependency_distance, not syn_dependency_distance_mean).
111
+ _TD_RENAME = {
112
+ "n_tokens": "desc_tokens",
113
+ "n_unique_tokens": "desc_unique_tokens",
114
+ "proportion_unique_tokens": "desc_proportion_unique_tokens",
115
+ "n_characters": "desc_characters",
116
+ "n_sentences": "desc_sentences",
117
+ "token_length_mean": "desc_token_length_mean",
118
+ "token_length_median": "desc_token_length_median",
119
+ "token_length_std": "desc_token_length_std",
120
+ "sentence_length_mean": "desc_sentence_length_mean",
121
+ "sentence_length_median": "desc_sentence_length_median",
122
+ "sentence_length_std": "desc_sentence_length_std",
123
+ "syllables_per_token_mean": "desc_syllables_per_token_mean",
124
+ "syllables_per_token_median": "desc_syllables_per_token_median",
125
+ "syllables_per_token_std": "desc_syllables_per_token_std",
126
+ "flesch_reading_ease": "read_flesch_reading_ease",
127
+ "flesch_kincaid_grade": "read_flesch_kincaid_grade",
128
+ "smog": "read_smog",
129
+ "gunning_fog": "read_gunning_fog",
130
+ "automated_readability_index": "read_automated_readability_index",
131
+ "coleman_liau_index": "read_coleman_liau_index",
132
+ "lix": "read_lix",
133
+ "rix": "read_rix",
134
+ "dependency_distance_mean": "syn_mean_dependency_distance",
135
+ "dependency_distance_std": "syn_dependency_distance_std",
136
+ "prop_adjacent_dependency_relation_mean": "syn_prop_adjacent_dependency_relation_mean",
137
+ "prop_adjacent_dependency_relation_std": "syn_prop_adjacent_dependency_relation_std",
138
+ "first_order_coherence": "coh_first_order_coherence",
139
+ "second_order_coherence": "coh_second_order_coherence",
140
+ }
141
+
142
+
143
+ @lru_cache(maxsize=1)
144
+ def _analyze_nlp():
145
+ from timbro.spacy_model import load_spacy
146
+
147
+ nlp = load_spacy(disable=["ner"])
148
+ for name in ("descriptive_stats", "readability", "dependency_distance", "coherence"):
149
+ nlp.add_pipe(f"textdescriptives/{name}")
150
+ return nlp
151
+
152
+
153
+ @lru_cache(maxsize=1)
154
+ def _dep_labels() -> tuple[str, ...]:
155
+ return tuple(sorted(_analyze_nlp().get_pipe("parser").labels))
156
+
157
+
158
+ def _clean(value):
159
+ """NaN -> None so JSON/CSV output is valid; everything else passes through."""
160
+ if isinstance(value, float) and value != value: # NaN != NaN
161
+ return None
162
+ return value
163
+
164
+
165
+ def _token_depth(tok) -> int:
166
+ # spaCy builds a fresh Token wrapper per `.head` access, so `is` identity never
167
+ # matches even at the root (whose head is itself by index) -- compare by `.i`.
168
+ depth = 0
169
+ while tok.head.i != tok.i:
170
+ tok = tok.head
171
+ depth += 1
172
+ return depth
173
+
174
+
175
+ def _imperative_ratio(sentences) -> float | None:
176
+ if not sentences:
177
+ return None
178
+ imperative = 0
179
+ for sent in sentences:
180
+ root = sent.root
181
+ if root.tag_ != "VB":
182
+ continue
183
+ if any(c.dep_ in {"nsubj", "nsubjpass"} for c in root.children):
184
+ continue
185
+ if sent.text.rstrip().endswith("?"):
186
+ continue
187
+ imperative += 1
188
+ return imperative / len(sentences)
189
+
190
+
191
+ def _first_person_ratio(sentences) -> float | None:
192
+ if not sentences:
193
+ return None
194
+ count = sum(
195
+ 1
196
+ for sent in sentences
197
+ if any(c.dep_ == "nsubj" and c.lemma_.lower() in {"i", "we"} for c in sent.root.children)
198
+ )
199
+ return count / len(sentences)
200
+
201
+
202
+ def _conditional_clauses_per_sentence(doc, n_sentences: int) -> float | None:
203
+ if not n_sentences:
204
+ return None
205
+ count = sum(
206
+ 1
207
+ for tok in doc
208
+ if tok.dep_ == "advcl"
209
+ and any(c.dep_ == "mark" and c.lemma_.lower() in _CONDITIONAL_MARKS for c in tok.children)
210
+ )
211
+ return count / n_sentences
212
+
213
+
214
+ def _struct_features(raw: str) -> tuple[dict, str]:
215
+ n = len(raw)
216
+ headings = list(_HEADING.finditer(raw))
217
+ code_chars = sum(len(m.group(0)) for m in _FENCE.finditer(raw))
218
+ lines = raw.split("\n")
219
+ non_blank = [ln for ln in lines if ln.strip()]
220
+ ordered_lines = sum(1 for ln in lines if _ORDERED_LIST.match(ln))
221
+ bullet_lines = sum(1 for ln in lines if _BULLET_LIST.match(ln))
222
+ list_lines = ordered_lines + bullet_lines
223
+ prose = strip_markup(raw)
224
+
225
+ inline_chars = sum(len(s) for s in _INLINE_CODE_SPAN.findall(_FENCE.sub("", raw)))
226
+
227
+ # External refs: bare scripts//references//assets/ tokens (with markdown links
228
+ # collapsed to their text) plus the same pattern inside every link target.
229
+ bare_refs = len(_EXTERNAL_REF.findall(_MD_LINK.sub(lambda m: m.group(1), raw)))
230
+ target_refs = sum(len(_EXTERNAL_REF.findall(m.group(2))) for m in _MD_LINK.finditer(raw))
231
+
232
+ named_section = 0
233
+ for m in headings:
234
+ htext = _HEADING_MARK.sub("", m.group(0)).strip().lower()
235
+ if any(s in htext for s in _NAMED_SECTIONS):
236
+ named_section = 1
237
+ break
238
+
239
+ # Long-paragraph ratio (#22): split blank-line-delimited paragraphs of the raw text with
240
+ # frontmatter and fenced code removed, so neither is double-counted as prose.
241
+ without_fences = _FENCE.sub("", _FRONTMATTER.sub("", raw))
242
+ paragraphs = [p for p in _BLANK_LINE.split(without_fences) if p.strip()]
243
+ long_paragraphs = sum(1 for p in paragraphs if len(_SENTENCE_END.findall(p)) > 6)
244
+
245
+ frontmatter = {}
246
+ fm_match = _FRONTMATTER.match(raw)
247
+ if fm_match:
248
+ try:
249
+ loaded = yaml.safe_load(fm_match.group(1))
250
+ except yaml.YAMLError:
251
+ loaded = None
252
+ if isinstance(loaded, dict):
253
+ frontmatter = loaded
254
+
255
+ name = frontmatter.get("name")
256
+ description = frontmatter.get("description")
257
+ if isinstance(description, str) and description:
258
+ fm_tokens = len(_analyze_nlp()(description))
259
+ wildcards = len(_FM_WILDCARD.findall(description))
260
+ fm_desc = {
261
+ "fm_desc_present": 1,
262
+ "fm_desc_tokens": fm_tokens,
263
+ "fm_desc_when_clause": 1 if _FM_WHEN_CLAUSE.search(description) else 0,
264
+ "fm_desc_or_count": len(_FM_OR_WORD.findall(description)),
265
+ "fm_desc_wildcard_per_token": wildcards / fm_tokens if fm_tokens else 0,
266
+ }
267
+ else:
268
+ fm_desc = {
269
+ "fm_desc_present": 0,
270
+ "fm_desc_tokens": 0,
271
+ "fm_desc_when_clause": 0,
272
+ "fm_desc_or_count": 0,
273
+ "fm_desc_wildcard_per_token": 0,
274
+ }
275
+
276
+ struct = {
277
+ "struct_heading_count": len(headings),
278
+ "struct_max_heading_depth": max((len(m.group(1)) for m in headings), default=0),
279
+ "struct_code_char_ratio": code_chars / n if n else None,
280
+ "struct_list_item_ratio": list_lines / len(non_blank) if non_blank else None,
281
+ "struct_table_count": len(_TABLE_SEPARATOR.findall(raw)),
282
+ "struct_prose_ratio": len(prose) / n if n else None,
283
+ "struct_frontmatter_field_count": len(frontmatter),
284
+ "struct_line_count": len(lines) if raw else 0,
285
+ "struct_inline_code_char_ratio": inline_chars / n if n else 0.0,
286
+ "struct_ordered_list_ratio": ordered_lines / len(non_blank) if non_blank else 0.0,
287
+ "struct_bullet_list_ratio": bullet_lines / len(non_blank) if non_blank else 0.0,
288
+ "struct_external_ref_count": bare_refs + target_refs,
289
+ "struct_named_section_present": named_section,
290
+ "struct_name_format_valid": 1 if isinstance(name, str) and _NAME_FORMAT.fullmatch(name) else 0,
291
+ "struct_long_paragraph_ratio": long_paragraphs / len(paragraphs) if paragraphs else 0.0,
292
+ **fm_desc,
293
+ "frontmatter_json": json.dumps(frontmatter, default=str),
294
+ }
295
+ return struct, prose
296
+
297
+
298
+ def _nlp_features(prose: str) -> dict:
299
+ doc = _analyze_nlp()(prose[:100000])
300
+ out = {}
301
+
302
+ row = td.extract_df(doc).iloc[0].to_dict()
303
+ for src, dst in _TD_RENAME.items():
304
+ out[dst] = _clean(row.get(src))
305
+
306
+ sentences = list(doc.sents)
307
+ if sentences:
308
+ depths = [max((_token_depth(t) for t in sent), default=0) for sent in sentences]
309
+ clausal = sum(1 for t in doc if t.dep_ in _CLAUSAL_DEPS)
310
+ out["syn_mean_tree_depth"] = sum(depths) / len(depths)
311
+ out["syn_clausal_per_sentence"] = clausal / len(sentences)
312
+ else:
313
+ out["syn_mean_tree_depth"] = None
314
+ out["syn_clausal_per_sentence"] = None
315
+
316
+ if sentences:
317
+ long_sents = sum(
318
+ 1 for sent in sentences if sum(1 for t in sent if not t.is_space) > 25
319
+ )
320
+ out["read_long_sentence_ratio"] = long_sents / len(sentences)
321
+ else:
322
+ out["read_long_sentence_ratio"] = 0.0
323
+
324
+ out["dict_imperative_ratio"] = _imperative_ratio(sentences)
325
+ out["dict_first_person_subject_ratio"] = _first_person_ratio(sentences)
326
+ out["dict_contrastive_example_count"] = sum(
327
+ 1
328
+ for ln in prose.split("\n")
329
+ if (s := ln.lstrip()) and (_CONTRASTIVE.match(s) or s[0] in _CONTRASTIVE_SYMBOLS)
330
+ )
331
+
332
+ all_tokens = [t for t in doc if not t.is_space]
333
+ n_all = len(all_tokens)
334
+ allcaps_hits = sum(1 for w in _ALLCAPS.findall(prose) if w in _ALLCAPS_DIRECTIVES)
335
+ out["dict_allcaps_directive_per_1k"] = allcaps_hits / n_all * 1000 if n_all else 0.0
336
+ pos_counts = Counter(t.pos_ for t in all_tokens)
337
+ for tag in POS_TAGS:
338
+ out[f"posdep_pos_{tag}"] = pos_counts.get(tag, 0) / n_all if n_all else 0.0
339
+ dep_counts = Counter(t.dep_ for t in all_tokens)
340
+ for label in _dep_labels():
341
+ out[f"posdep_dep_{label}"] = dep_counts.get(label, 0) / n_all if n_all else 0.0
342
+
343
+ content_lemmas = [t.lemma_.lower() for t in doc if t.is_alpha and t.pos_ in _CONTENT_POS]
344
+ out["lex_mtld"] = lex_div.mtld(content_lemmas)
345
+ out["lex_hdd"] = lex_div.hdd(content_lemmas)
346
+ zipfs = [zipf_frequency(w, "en") for w in content_lemmas]
347
+ out["lex_zipf_mean"] = sum(zipfs) / len(zipfs) if zipfs else None
348
+
349
+ all_lemmas = [t.lemma_.lower() for t in all_tokens]
350
+ out["dict_hedge_per_1k"] = (
351
+ _lexicon_matches(all_lemmas, _lexicon("hedges.txt")) / n_all * 1000 if n_all else 0.0
352
+ )
353
+ out["dict_booster_per_1k"] = (
354
+ _lexicon_matches(all_lemmas, _lexicon("boosters.txt")) / n_all * 1000 if n_all else 0.0
355
+ )
356
+ all_forms = [t.text.lower() for t in all_tokens]
357
+ out["dict_hype_per_1k"] = (
358
+ _lexicon_matches(all_forms, _hype_entries()) / n_all * 1000 if n_all else 0.0
359
+ )
360
+ out["dict_negation_per_1k"] = (
361
+ _lexicon_matches(all_lemmas, _lexicon("negations.txt")) / n_all * 1000 if n_all else 0.0
362
+ )
363
+ out["dict_conditional_per_1k"] = (
364
+ _lexicon_matches(all_lemmas, _lexicon("connectives_conditional.txt")) / n_all * 1000
365
+ if n_all
366
+ else 0.0
367
+ )
368
+ out["dict_conditional_clauses_per_sentence"] = _conditional_clauses_per_sentence(
369
+ doc, len(sentences)
370
+ )
371
+
372
+ second_person = sum(1 for t in all_tokens if t.text.lower() in _SECOND_PERSON)
373
+ out["dict_second_person_per_1k"] = second_person / n_all * 1000 if n_all else 0.0
374
+
375
+ cross_refs = len(_CROSS_REFERENCE.findall(prose))
376
+ out["dict_cross_reference_per_1k"] = cross_refs / n_all * 1000 if n_all else 0.0
377
+
378
+ plain_matches = 0
379
+ plain_replacements = []
380
+ for entry, plain in _plain_pairs():
381
+ hits = _lexicon_matches(all_lemmas, (entry,))
382
+ if hits:
383
+ plain_matches += hits
384
+ plain_replacements.append([" ".join(entry), plain])
385
+ out["dict_plain_replacement_per_1k"] = plain_matches / n_all * 1000 if n_all else 0.0
386
+ out["dict_plain_replacements_json"] = json.dumps(plain_replacements)
387
+
388
+ if len(sentences) < 2:
389
+ out["coh_lemma_overlap_adj"] = None
390
+ else:
391
+ sent_lemma_sets = [
392
+ {t.lemma_.lower() for t in sent if t.pos_ in _COH_CONTENT_POS} for sent in sentences
393
+ ]
394
+ overlaps = []
395
+ for a, b in zip(sent_lemma_sets, sent_lemma_sets[1:]):
396
+ union = a | b
397
+ overlaps.append(len(a & b) / len(union) if union else 0.0)
398
+ out["coh_lemma_overlap_adj"] = sum(overlaps) / len(overlaps)
399
+
400
+ return out
401
+
402
+
403
+ def analyze_text(raw: str) -> dict:
404
+ """One feature vector (no `path` key) for a raw document's text."""
405
+ struct, prose = _struct_features(raw)
406
+ return {**struct, **_nlp_features(prose)}
407
+
408
+
409
+ def analyze_file(path: Path) -> dict:
410
+ text = path.read_text(encoding="utf-8")
411
+ return {"path": str(path), **analyze_text(text)}
412
+
413
+
414
+ def _write_jsonl(rows: list[dict], out) -> None:
415
+ for row in rows:
416
+ out.write(json.dumps(row) + "\n")
417
+
418
+
419
+ def _write_csv(rows: list[dict], out) -> None:
420
+ fieldnames = [k for k in rows[0] if not k.endswith("_json")]
421
+ writer = csv.DictWriter(out, fieldnames=fieldnames, extrasaction="ignore")
422
+ writer.writeheader()
423
+ writer.writerows(rows)
424
+
425
+
426
+ def run_analyze(paths: list[str], fmt: str = "jsonl", out_path: str | None = None) -> int:
427
+ files = []
428
+ for p in paths:
429
+ path = Path(p)
430
+ if path.suffix not in {".md", ".txt"}:
431
+ print(f"skipping {p}: not a .md/.txt file", file=sys.stderr)
432
+ continue
433
+ if not path.is_file():
434
+ print(f"skipping {p}: no such file", file=sys.stderr)
435
+ continue
436
+ files.append(path)
437
+
438
+ if not files:
439
+ print("no .md/.txt files to analyze", file=sys.stderr)
440
+ return 1
441
+
442
+ rows = [analyze_file(path) for path in files]
443
+ write = _write_csv if fmt == "csv" else _write_jsonl
444
+ if out_path:
445
+ with open(out_path, "w", encoding="utf-8", newline="") as fh:
446
+ write(rows, fh)
447
+ else:
448
+ write(rows, sys.stdout)
449
+ return 0
@@ -0,0 +1,15 @@
1
+ """Cleanup helpers for preparing corpora before Timbro style analysis."""
2
+
3
+ from .latex import detex_file, preprocess_runtime_text, tex_to_markdown
4
+ from .papers import cleanup_markdown_file, cleanup_paper_markdown, clean_extracted_text, extract_prose_excerpt, split_frontmatter
5
+
6
+ __all__ = [
7
+ "detex_file",
8
+ "preprocess_runtime_text",
9
+ "tex_to_markdown",
10
+ "cleanup_markdown_file",
11
+ "cleanup_paper_markdown",
12
+ "clean_extracted_text",
13
+ "extract_prose_excerpt",
14
+ "split_frontmatter",
15
+ ]
@@ -0,0 +1,118 @@
1
+ """Utilities for converting LaTeX sources into Timbro-ready plain text."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import shutil
7
+ import subprocess
8
+ from pathlib import Path
9
+
10
+ from timbro.cleanup.papers import clean_extracted_text, cleanup_paper_markdown
11
+
12
+ _DROP_ENVS = (
13
+ "tikzpicture",
14
+ "figure",
15
+ "figure*",
16
+ "table",
17
+ "table*",
18
+ "tabular",
19
+ "tabularx",
20
+ "equation",
21
+ "equation*",
22
+ "align",
23
+ "align*",
24
+ "aligned",
25
+ "lstlisting",
26
+ "minted",
27
+ "verbatim",
28
+ "thebibliography",
29
+ )
30
+
31
+
32
+ def _normalize_detex_output(text: str) -> str:
33
+ text = text.replace("\r\n", "\n").replace("\r", "\n")
34
+ text = re.sub(r"(?i)\.abstract\b", ".", text)
35
+ text = re.sub(r"(?i)^abstract\s*", "Abstract\n\n", text)
36
+ text = re.sub(r"\b(?:sub)*section([A-Z])", r"\n\n\1", text)
37
+ text = re.sub(r"\b(?:cite|ref|label)[A-Za-z0-9:_-]*\b", " ", text)
38
+ text = re.sub(r"\n{3,}", "\n\n", text)
39
+ text = re.sub(r"[ \t]+", " ", text)
40
+ return text.strip()
41
+
42
+
43
+ def _strip_latex_source_noise(text: str) -> str:
44
+ text = re.sub(r"(?m)^\s*%.*$", "", text)
45
+ for env in _DROP_ENVS:
46
+ text = re.sub(rf"\\begin\{{{re.escape(env)}\}}.*?\\end\{{{re.escape(env)}\}}", "\n", text, flags=re.S)
47
+ text = re.sub(r"\\(?:label|ref|eqref|autoref|cref|Cref|pageref|cite|citet|citep|citealt|citealp|footnote)\*?(?:\[[^\]]*\])?\{[^}]*\}", " ", text)
48
+ text = re.sub(r"\\(?:begin|end)\{(?:itemize|enumerate|description)\}", "\n", text)
49
+ text = re.sub(r"\\item(?:\[[^\]]*\])?", "\n- ", text)
50
+ text = re.sub(r"\\(?:centering|smallskip|medskip|bigskip|noindent|vspace\*?|hspace\*?)\{[^}]*\}", " ", text)
51
+ text = re.sub(r"\\(?:centering|smallskip|medskip|bigskip|noindent)", " ", text)
52
+ return text
53
+
54
+
55
+ def has_detex() -> bool:
56
+ return shutil.which("detex") is not None
57
+
58
+
59
+ def looks_like_latex(text: str) -> bool:
60
+ markers = [
61
+ r"\\begin\{",
62
+ r"\\end\{",
63
+ r"\\section\*?\{",
64
+ r"\\subsection\*?\{",
65
+ r"\\title\{",
66
+ r"\\author\{",
67
+ r"\\cite\{",
68
+ r"\\ref\{",
69
+ r"\\label\{",
70
+ r"\\documentclass",
71
+ ]
72
+ hits = sum(bool(re.search(pat, text)) for pat in markers)
73
+ return hits >= 1 or "\\begin{document}" in text or "\\end{document}" in text
74
+
75
+
76
+ def detex_text(text: str, *, replace_math: bool = True) -> str:
77
+ """Strip LaTeX commands from raw `.tex` content using the external `detex` tool.
78
+
79
+ Raises `RuntimeError` if `detex` is not installed or returns a non-zero exit code.
80
+ """
81
+ exe = shutil.which("detex")
82
+ if exe is None:
83
+ raise RuntimeError("detex is not installed. Install opendetex to ingest .tex files.")
84
+
85
+ cmd = [exe, "-n"]
86
+ if replace_math:
87
+ cmd.append("-r")
88
+ proc = subprocess.run(
89
+ cmd,
90
+ input=_strip_latex_source_noise(text),
91
+ text=True,
92
+ capture_output=True,
93
+ check=False,
94
+ )
95
+ if proc.returncode != 0:
96
+ raise RuntimeError(proc.stderr.strip() or "detex failed")
97
+ return _normalize_detex_output(proc.stdout)
98
+
99
+
100
+ def detex_file(path: str | Path, *, replace_math: bool = True) -> str:
101
+ file_path = Path(path)
102
+ return detex_text(file_path.read_text(encoding="utf-8", errors="ignore"), replace_math=replace_math)
103
+
104
+
105
+ def tex_to_markdown(path: str | Path, *, replace_math: bool = True) -> str:
106
+ return cleanup_paper_markdown(detex_file(path, replace_math=replace_math))
107
+
108
+
109
+ def preprocess_runtime_text(text: str) -> str:
110
+ """Normalize text before scoring.
111
+
112
+ If the input looks like raw LaTeX and `detex` is available, strip the TeX markup
113
+ and run lightweight cleanup on the result while preserving the original order.
114
+ Otherwise, return the text with only lightweight whitespace normalization.
115
+ """
116
+ if looks_like_latex(text) and has_detex():
117
+ return clean_extracted_text(detex_text(text))
118
+ return clean_extracted_text(text)