timbro 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- timbro/__init__.py +11 -0
- timbro/analyze.py +449 -0
- timbro/cleanup/__init__.py +15 -0
- timbro/cleanup/latex.py +118 -0
- timbro/cleanup/papers.py +261 -0
- timbro/cli.py +251 -0
- timbro/concreteness.py +94 -0
- timbro/config.py +142 -0
- timbro/flow.py +128 -0
- timbro/fw.py +88 -0
- timbro/hedge.py +104 -0
- timbro/lexicons/boosters.txt +66 -0
- timbro/lexicons/connectives_conditional.txt +17 -0
- timbro/lexicons/hedges.txt +90 -0
- timbro/lexicons/hype.txt +61 -0
- timbro/lexicons/negations.txt +22 -0
- timbro/lexicons/plain_wording.txt +223 -0
- timbro/mcp_server.py +80 -0
- timbro/metric.py +120 -0
- timbro/model.py +510 -0
- timbro/norms/NOTICE.md +9 -0
- timbro/norms/concreteness_brysbaert2014.csv.gz +0 -0
- timbro/profiles.py +253 -0
- timbro/report.py +224 -0
- timbro/rewrite.py +63 -0
- timbro/rubrics/__init__.py +23 -0
- timbro/rubrics/base.py +50 -0
- timbro/rubrics/density/__init__.py +5 -0
- timbro/rubrics/density/checks.py +127 -0
- timbro/rubrics/density/rubric.py +24 -0
- timbro/rubrics/features.py +804 -0
- timbro/rubrics/registry.py +15 -0
- timbro/rubrics/report.py +51 -0
- timbro/rubrics/rules.py +496 -0
- timbro/rubrics/schimel/__init__.py +5 -0
- timbro/rubrics/schimel/rubric.py +21 -0
- timbro/rubrics/sections.py +32 -0
- timbro/rubrics/slop/__init__.py +5 -0
- timbro/rubrics/slop/checks.py +91 -0
- timbro/rubrics/slop/rubric.py +26 -0
- timbro/sample/contrast/01-synergy.md +9 -0
- timbro/sample/contrast/02-revolutionize.md +11 -0
- timbro/sample/contrast/03-paradigm.md +9 -0
- timbro/sample/exemplars/01-shipping.md +9 -0
- timbro/sample/exemplars/02-debugging.md +9 -0
- timbro/sample/exemplars/03-reviews.md +9 -0
- timbro/sample/exemplars/04-estimates.md +9 -0
- timbro/spacy_model.py +30 -0
- timbro/tells.py +324 -0
- timbro/text.py +169 -0
- timbro-0.6.0.dist-info/METADATA +309 -0
- timbro-0.6.0.dist-info/RECORD +55 -0
- timbro-0.6.0.dist-info/WHEEL +4 -0
- timbro-0.6.0.dist-info/entry_points.txt +3 -0
- timbro-0.6.0.dist-info/licenses/LICENSE +21 -0
timbro/__init__.py
ADDED
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
from timbro.model import VoiceModel, ScoreResult, FeatureMove, MarkdownAxis, features, read_corpus
|
|
2
|
+
from timbro.flow import FlowReport, flow_report
|
|
3
|
+
from timbro.profiles import Profile, add_file, add_text, get_profile, init_profile, list_profiles
|
|
4
|
+
from timbro.rubrics import check_text
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"VoiceModel", "ScoreResult", "FeatureMove", "MarkdownAxis", "features", "read_corpus",
|
|
8
|
+
"FlowReport", "flow_report",
|
|
9
|
+
"Profile", "get_profile", "list_profiles", "init_profile", "add_text", "add_file",
|
|
10
|
+
"check_text",
|
|
11
|
+
]
|
timbro/analyze.py
ADDED
|
@@ -0,0 +1,449 @@
|
|
|
1
|
+
"""``timbro analyze`` (#17): one deterministic linguistic feature vector per document.
|
|
2
|
+
|
|
3
|
+
No LLM, no network at analyze time. Profiles/rubrics/scoring/verdicts are untouched --
|
|
4
|
+
this is a standalone slice for the SKILL.md linguistic-topology paper (paper/README.md
|
|
5
|
+
WS2 Issue A). `struct_*` runs on the raw markdown; everything else runs on prose
|
|
6
|
+
isolated via the #16 markup stripper (`timbro.text.strip_markup`).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import csv
|
|
12
|
+
import json
|
|
13
|
+
import re
|
|
14
|
+
import sys
|
|
15
|
+
from collections import Counter
|
|
16
|
+
from functools import lru_cache
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
|
|
19
|
+
import textdescriptives as td
|
|
20
|
+
import yaml
|
|
21
|
+
from lexical_diversity import lex_div
|
|
22
|
+
from wordfreq import zipf_frequency
|
|
23
|
+
|
|
24
|
+
from timbro.model import POS_TAGS
|
|
25
|
+
from timbro.text import strip_markup
|
|
26
|
+
|
|
27
|
+
_FRONTMATTER = re.compile(r"\A---[ \t]*\n(.*?)\n---[ \t]*\n?", re.S)
|
|
28
|
+
_FENCE = re.compile(r"(```|~~~).*?\1", re.S)
|
|
29
|
+
_HEADING = re.compile(r"(?m)^[ \t]*(#{1,6})[ \t]+.*$")
|
|
30
|
+
_TABLE_SEPARATOR = re.compile(r"(?m)^[ \t]*:?-{2,}:?(?:[ \t]*\|[ \t]*:?-{2,}:?)+[ \t]*$")
|
|
31
|
+
_BULLET_LIST = re.compile(r"^[ \t]*[-*+][ \t]+")
|
|
32
|
+
_ORDERED_LIST = re.compile(r"^[ \t]*\d+\.[ \t]+")
|
|
33
|
+
_BLANK_LINE = re.compile(r"\n[ \t]*\n")
|
|
34
|
+
_SENTENCE_END = re.compile(r"[.!?]+") # naive sentence count; _struct runs without spaCy
|
|
35
|
+
|
|
36
|
+
# Folk-advice exploratory features (#21).
|
|
37
|
+
_INLINE_CODE_SPAN = re.compile(r"`([^`\n]+)`")
|
|
38
|
+
_MD_LINK = re.compile(r"\[(.*?)\]\((.*?)\)")
|
|
39
|
+
_EXTERNAL_REF = re.compile(r"(scripts/|references/|assets/)[^\s)\]]*")
|
|
40
|
+
_NAMED_SECTIONS = (
|
|
41
|
+
"examples", "guidelines", "when to use", "procedure", "pitfalls", "usage", "instructions",
|
|
42
|
+
)
|
|
43
|
+
_NAME_FORMAT = re.compile(r"[a-z0-9-]{1,64}")
|
|
44
|
+
_HEADING_MARK = re.compile(r"^[ \t]*#{1,6}[ \t]+")
|
|
45
|
+
_FM_WHEN_CLAUSE = re.compile(r"\b(when|use (this|it) (when|for|to)|whenever|if you)\b", re.I)
|
|
46
|
+
_FM_OR_WORD = re.compile(r"\bor\b", re.I)
|
|
47
|
+
_FM_WILDCARD = re.compile(r"\b(any|all|every|always|whenever|anything|everything)\b", re.I)
|
|
48
|
+
_ALLCAPS = re.compile(r"\b[A-Z][A-Z']+\b")
|
|
49
|
+
_ALLCAPS_DIRECTIVES = {"ALWAYS", "NEVER", "MUST", "NOT", "DON'T", "DO"}
|
|
50
|
+
_CONTRASTIVE = re.compile(r"^(do|don'?t|correct|incorrect|good|bad)\s*:", re.I)
|
|
51
|
+
_CONTRASTIVE_SYMBOLS = ("✅", "❌", "✓", "✗")
|
|
52
|
+
|
|
53
|
+
_CONTENT_POS = {"NOUN", "PROPN", "VERB", "ADJ", "ADV"}
|
|
54
|
+
_COH_CONTENT_POS = {"NOUN", "PROPN", "VERB", "ADJ"}
|
|
55
|
+
_CLAUSAL_DEPS = {"ccomp", "xcomp", "advcl", "acl", "relcl"}
|
|
56
|
+
_CONDITIONAL_MARKS = {"if", "unless", "when"}
|
|
57
|
+
_SECOND_PERSON = {"you", "your", "yours", "you're", "yourself"}
|
|
58
|
+
_CROSS_REFERENCE = re.compile(
|
|
59
|
+
r"\b(?:see also|see below|see above|refer to|as described in|cf\.)", re.I
|
|
60
|
+
)
|
|
61
|
+
_LEXICON_DIR = Path(__file__).parent / "lexicons"
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@lru_cache(maxsize=None)
|
|
65
|
+
def _lexicon(name: str) -> tuple[tuple[str, ...], ...]:
|
|
66
|
+
lines = (_LEXICON_DIR / name).read_text(encoding="utf-8").splitlines()
|
|
67
|
+
entries = (ln.strip() for ln in lines if ln.strip() and not ln.startswith("#"))
|
|
68
|
+
return tuple(tuple(entry.lower().split()) for entry in entries)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@lru_cache(maxsize=None)
|
|
72
|
+
def _plain_pairs() -> tuple[tuple[tuple[str, ...], str], ...]:
|
|
73
|
+
lines = (_LEXICON_DIR / "plain_wording.txt").read_text(encoding="utf-8").splitlines()
|
|
74
|
+
pairs = []
|
|
75
|
+
for ln in lines:
|
|
76
|
+
if not ln.strip() or ln.startswith("#"):
|
|
77
|
+
continue
|
|
78
|
+
complex_side, _, plain = ln.partition("\t")
|
|
79
|
+
pairs.append((tuple(complex_side.lower().split()), plain.strip()))
|
|
80
|
+
return tuple(pairs)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
@lru_cache(maxsize=None)
|
|
84
|
+
def _hype_entries() -> tuple[tuple[str, ...], ...]:
|
|
85
|
+
# Hype is matched on surface FORM, not lemma: en_core_web_sm mangles participial adjectives
|
|
86
|
+
# ("groundbreaking" -> "groundbreake") and splits hyphenated compounds into three tokens
|
|
87
|
+
# ("world-class" -> world - class), so the lemma-matching used by the frozen boosters/hedges
|
|
88
|
+
# lexicons silently drops most of this list. Entries are tokenized the same way spaCy splits
|
|
89
|
+
# (hyphen kept as its own token) so multiword and hyphenated terms line up against the forms.
|
|
90
|
+
lines = (_LEXICON_DIR / "hype.txt").read_text(encoding="utf-8").splitlines()
|
|
91
|
+
entries = []
|
|
92
|
+
for ln in lines:
|
|
93
|
+
stripped = ln.strip()
|
|
94
|
+
if not stripped or stripped.startswith("#"):
|
|
95
|
+
continue
|
|
96
|
+
entries.append(tuple(stripped.lower().replace("-", " - ").split()))
|
|
97
|
+
return tuple(entries)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _lexicon_matches(lemmas: list[str], entries: tuple[tuple[str, ...], ...]) -> int:
|
|
101
|
+
count = 0
|
|
102
|
+
for entry in entries:
|
|
103
|
+
width = len(entry)
|
|
104
|
+
for i in range(len(lemmas) - width + 1):
|
|
105
|
+
if tuple(lemmas[i : i + width]) == entry:
|
|
106
|
+
count += 1
|
|
107
|
+
return count
|
|
108
|
+
|
|
109
|
+
# textdescriptives' extract_df column -> our snake_case name (exact renames are decisions
|
|
110
|
+
# from the #17 spec: syn_mean_dependency_distance, not syn_dependency_distance_mean).
|
|
111
|
+
_TD_RENAME = {
|
|
112
|
+
"n_tokens": "desc_tokens",
|
|
113
|
+
"n_unique_tokens": "desc_unique_tokens",
|
|
114
|
+
"proportion_unique_tokens": "desc_proportion_unique_tokens",
|
|
115
|
+
"n_characters": "desc_characters",
|
|
116
|
+
"n_sentences": "desc_sentences",
|
|
117
|
+
"token_length_mean": "desc_token_length_mean",
|
|
118
|
+
"token_length_median": "desc_token_length_median",
|
|
119
|
+
"token_length_std": "desc_token_length_std",
|
|
120
|
+
"sentence_length_mean": "desc_sentence_length_mean",
|
|
121
|
+
"sentence_length_median": "desc_sentence_length_median",
|
|
122
|
+
"sentence_length_std": "desc_sentence_length_std",
|
|
123
|
+
"syllables_per_token_mean": "desc_syllables_per_token_mean",
|
|
124
|
+
"syllables_per_token_median": "desc_syllables_per_token_median",
|
|
125
|
+
"syllables_per_token_std": "desc_syllables_per_token_std",
|
|
126
|
+
"flesch_reading_ease": "read_flesch_reading_ease",
|
|
127
|
+
"flesch_kincaid_grade": "read_flesch_kincaid_grade",
|
|
128
|
+
"smog": "read_smog",
|
|
129
|
+
"gunning_fog": "read_gunning_fog",
|
|
130
|
+
"automated_readability_index": "read_automated_readability_index",
|
|
131
|
+
"coleman_liau_index": "read_coleman_liau_index",
|
|
132
|
+
"lix": "read_lix",
|
|
133
|
+
"rix": "read_rix",
|
|
134
|
+
"dependency_distance_mean": "syn_mean_dependency_distance",
|
|
135
|
+
"dependency_distance_std": "syn_dependency_distance_std",
|
|
136
|
+
"prop_adjacent_dependency_relation_mean": "syn_prop_adjacent_dependency_relation_mean",
|
|
137
|
+
"prop_adjacent_dependency_relation_std": "syn_prop_adjacent_dependency_relation_std",
|
|
138
|
+
"first_order_coherence": "coh_first_order_coherence",
|
|
139
|
+
"second_order_coherence": "coh_second_order_coherence",
|
|
140
|
+
}
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
@lru_cache(maxsize=1)
|
|
144
|
+
def _analyze_nlp():
|
|
145
|
+
from timbro.spacy_model import load_spacy
|
|
146
|
+
|
|
147
|
+
nlp = load_spacy(disable=["ner"])
|
|
148
|
+
for name in ("descriptive_stats", "readability", "dependency_distance", "coherence"):
|
|
149
|
+
nlp.add_pipe(f"textdescriptives/{name}")
|
|
150
|
+
return nlp
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
@lru_cache(maxsize=1)
|
|
154
|
+
def _dep_labels() -> tuple[str, ...]:
|
|
155
|
+
return tuple(sorted(_analyze_nlp().get_pipe("parser").labels))
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def _clean(value):
|
|
159
|
+
"""NaN -> None so JSON/CSV output is valid; everything else passes through."""
|
|
160
|
+
if isinstance(value, float) and value != value: # NaN != NaN
|
|
161
|
+
return None
|
|
162
|
+
return value
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _token_depth(tok) -> int:
|
|
166
|
+
# spaCy builds a fresh Token wrapper per `.head` access, so `is` identity never
|
|
167
|
+
# matches even at the root (whose head is itself by index) -- compare by `.i`.
|
|
168
|
+
depth = 0
|
|
169
|
+
while tok.head.i != tok.i:
|
|
170
|
+
tok = tok.head
|
|
171
|
+
depth += 1
|
|
172
|
+
return depth
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _imperative_ratio(sentences) -> float | None:
|
|
176
|
+
if not sentences:
|
|
177
|
+
return None
|
|
178
|
+
imperative = 0
|
|
179
|
+
for sent in sentences:
|
|
180
|
+
root = sent.root
|
|
181
|
+
if root.tag_ != "VB":
|
|
182
|
+
continue
|
|
183
|
+
if any(c.dep_ in {"nsubj", "nsubjpass"} for c in root.children):
|
|
184
|
+
continue
|
|
185
|
+
if sent.text.rstrip().endswith("?"):
|
|
186
|
+
continue
|
|
187
|
+
imperative += 1
|
|
188
|
+
return imperative / len(sentences)
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
def _first_person_ratio(sentences) -> float | None:
|
|
192
|
+
if not sentences:
|
|
193
|
+
return None
|
|
194
|
+
count = sum(
|
|
195
|
+
1
|
|
196
|
+
for sent in sentences
|
|
197
|
+
if any(c.dep_ == "nsubj" and c.lemma_.lower() in {"i", "we"} for c in sent.root.children)
|
|
198
|
+
)
|
|
199
|
+
return count / len(sentences)
|
|
200
|
+
|
|
201
|
+
|
|
202
|
+
def _conditional_clauses_per_sentence(doc, n_sentences: int) -> float | None:
|
|
203
|
+
if not n_sentences:
|
|
204
|
+
return None
|
|
205
|
+
count = sum(
|
|
206
|
+
1
|
|
207
|
+
for tok in doc
|
|
208
|
+
if tok.dep_ == "advcl"
|
|
209
|
+
and any(c.dep_ == "mark" and c.lemma_.lower() in _CONDITIONAL_MARKS for c in tok.children)
|
|
210
|
+
)
|
|
211
|
+
return count / n_sentences
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _struct_features(raw: str) -> tuple[dict, str]:
|
|
215
|
+
n = len(raw)
|
|
216
|
+
headings = list(_HEADING.finditer(raw))
|
|
217
|
+
code_chars = sum(len(m.group(0)) for m in _FENCE.finditer(raw))
|
|
218
|
+
lines = raw.split("\n")
|
|
219
|
+
non_blank = [ln for ln in lines if ln.strip()]
|
|
220
|
+
ordered_lines = sum(1 for ln in lines if _ORDERED_LIST.match(ln))
|
|
221
|
+
bullet_lines = sum(1 for ln in lines if _BULLET_LIST.match(ln))
|
|
222
|
+
list_lines = ordered_lines + bullet_lines
|
|
223
|
+
prose = strip_markup(raw)
|
|
224
|
+
|
|
225
|
+
inline_chars = sum(len(s) for s in _INLINE_CODE_SPAN.findall(_FENCE.sub("", raw)))
|
|
226
|
+
|
|
227
|
+
# External refs: bare scripts//references//assets/ tokens (with markdown links
|
|
228
|
+
# collapsed to their text) plus the same pattern inside every link target.
|
|
229
|
+
bare_refs = len(_EXTERNAL_REF.findall(_MD_LINK.sub(lambda m: m.group(1), raw)))
|
|
230
|
+
target_refs = sum(len(_EXTERNAL_REF.findall(m.group(2))) for m in _MD_LINK.finditer(raw))
|
|
231
|
+
|
|
232
|
+
named_section = 0
|
|
233
|
+
for m in headings:
|
|
234
|
+
htext = _HEADING_MARK.sub("", m.group(0)).strip().lower()
|
|
235
|
+
if any(s in htext for s in _NAMED_SECTIONS):
|
|
236
|
+
named_section = 1
|
|
237
|
+
break
|
|
238
|
+
|
|
239
|
+
# Long-paragraph ratio (#22): split blank-line-delimited paragraphs of the raw text with
|
|
240
|
+
# frontmatter and fenced code removed, so neither is double-counted as prose.
|
|
241
|
+
without_fences = _FENCE.sub("", _FRONTMATTER.sub("", raw))
|
|
242
|
+
paragraphs = [p for p in _BLANK_LINE.split(without_fences) if p.strip()]
|
|
243
|
+
long_paragraphs = sum(1 for p in paragraphs if len(_SENTENCE_END.findall(p)) > 6)
|
|
244
|
+
|
|
245
|
+
frontmatter = {}
|
|
246
|
+
fm_match = _FRONTMATTER.match(raw)
|
|
247
|
+
if fm_match:
|
|
248
|
+
try:
|
|
249
|
+
loaded = yaml.safe_load(fm_match.group(1))
|
|
250
|
+
except yaml.YAMLError:
|
|
251
|
+
loaded = None
|
|
252
|
+
if isinstance(loaded, dict):
|
|
253
|
+
frontmatter = loaded
|
|
254
|
+
|
|
255
|
+
name = frontmatter.get("name")
|
|
256
|
+
description = frontmatter.get("description")
|
|
257
|
+
if isinstance(description, str) and description:
|
|
258
|
+
fm_tokens = len(_analyze_nlp()(description))
|
|
259
|
+
wildcards = len(_FM_WILDCARD.findall(description))
|
|
260
|
+
fm_desc = {
|
|
261
|
+
"fm_desc_present": 1,
|
|
262
|
+
"fm_desc_tokens": fm_tokens,
|
|
263
|
+
"fm_desc_when_clause": 1 if _FM_WHEN_CLAUSE.search(description) else 0,
|
|
264
|
+
"fm_desc_or_count": len(_FM_OR_WORD.findall(description)),
|
|
265
|
+
"fm_desc_wildcard_per_token": wildcards / fm_tokens if fm_tokens else 0,
|
|
266
|
+
}
|
|
267
|
+
else:
|
|
268
|
+
fm_desc = {
|
|
269
|
+
"fm_desc_present": 0,
|
|
270
|
+
"fm_desc_tokens": 0,
|
|
271
|
+
"fm_desc_when_clause": 0,
|
|
272
|
+
"fm_desc_or_count": 0,
|
|
273
|
+
"fm_desc_wildcard_per_token": 0,
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
struct = {
|
|
277
|
+
"struct_heading_count": len(headings),
|
|
278
|
+
"struct_max_heading_depth": max((len(m.group(1)) for m in headings), default=0),
|
|
279
|
+
"struct_code_char_ratio": code_chars / n if n else None,
|
|
280
|
+
"struct_list_item_ratio": list_lines / len(non_blank) if non_blank else None,
|
|
281
|
+
"struct_table_count": len(_TABLE_SEPARATOR.findall(raw)),
|
|
282
|
+
"struct_prose_ratio": len(prose) / n if n else None,
|
|
283
|
+
"struct_frontmatter_field_count": len(frontmatter),
|
|
284
|
+
"struct_line_count": len(lines) if raw else 0,
|
|
285
|
+
"struct_inline_code_char_ratio": inline_chars / n if n else 0.0,
|
|
286
|
+
"struct_ordered_list_ratio": ordered_lines / len(non_blank) if non_blank else 0.0,
|
|
287
|
+
"struct_bullet_list_ratio": bullet_lines / len(non_blank) if non_blank else 0.0,
|
|
288
|
+
"struct_external_ref_count": bare_refs + target_refs,
|
|
289
|
+
"struct_named_section_present": named_section,
|
|
290
|
+
"struct_name_format_valid": 1 if isinstance(name, str) and _NAME_FORMAT.fullmatch(name) else 0,
|
|
291
|
+
"struct_long_paragraph_ratio": long_paragraphs / len(paragraphs) if paragraphs else 0.0,
|
|
292
|
+
**fm_desc,
|
|
293
|
+
"frontmatter_json": json.dumps(frontmatter, default=str),
|
|
294
|
+
}
|
|
295
|
+
return struct, prose
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _nlp_features(prose: str) -> dict:
|
|
299
|
+
doc = _analyze_nlp()(prose[:100000])
|
|
300
|
+
out = {}
|
|
301
|
+
|
|
302
|
+
row = td.extract_df(doc).iloc[0].to_dict()
|
|
303
|
+
for src, dst in _TD_RENAME.items():
|
|
304
|
+
out[dst] = _clean(row.get(src))
|
|
305
|
+
|
|
306
|
+
sentences = list(doc.sents)
|
|
307
|
+
if sentences:
|
|
308
|
+
depths = [max((_token_depth(t) for t in sent), default=0) for sent in sentences]
|
|
309
|
+
clausal = sum(1 for t in doc if t.dep_ in _CLAUSAL_DEPS)
|
|
310
|
+
out["syn_mean_tree_depth"] = sum(depths) / len(depths)
|
|
311
|
+
out["syn_clausal_per_sentence"] = clausal / len(sentences)
|
|
312
|
+
else:
|
|
313
|
+
out["syn_mean_tree_depth"] = None
|
|
314
|
+
out["syn_clausal_per_sentence"] = None
|
|
315
|
+
|
|
316
|
+
if sentences:
|
|
317
|
+
long_sents = sum(
|
|
318
|
+
1 for sent in sentences if sum(1 for t in sent if not t.is_space) > 25
|
|
319
|
+
)
|
|
320
|
+
out["read_long_sentence_ratio"] = long_sents / len(sentences)
|
|
321
|
+
else:
|
|
322
|
+
out["read_long_sentence_ratio"] = 0.0
|
|
323
|
+
|
|
324
|
+
out["dict_imperative_ratio"] = _imperative_ratio(sentences)
|
|
325
|
+
out["dict_first_person_subject_ratio"] = _first_person_ratio(sentences)
|
|
326
|
+
out["dict_contrastive_example_count"] = sum(
|
|
327
|
+
1
|
|
328
|
+
for ln in prose.split("\n")
|
|
329
|
+
if (s := ln.lstrip()) and (_CONTRASTIVE.match(s) or s[0] in _CONTRASTIVE_SYMBOLS)
|
|
330
|
+
)
|
|
331
|
+
|
|
332
|
+
all_tokens = [t for t in doc if not t.is_space]
|
|
333
|
+
n_all = len(all_tokens)
|
|
334
|
+
allcaps_hits = sum(1 for w in _ALLCAPS.findall(prose) if w in _ALLCAPS_DIRECTIVES)
|
|
335
|
+
out["dict_allcaps_directive_per_1k"] = allcaps_hits / n_all * 1000 if n_all else 0.0
|
|
336
|
+
pos_counts = Counter(t.pos_ for t in all_tokens)
|
|
337
|
+
for tag in POS_TAGS:
|
|
338
|
+
out[f"posdep_pos_{tag}"] = pos_counts.get(tag, 0) / n_all if n_all else 0.0
|
|
339
|
+
dep_counts = Counter(t.dep_ for t in all_tokens)
|
|
340
|
+
for label in _dep_labels():
|
|
341
|
+
out[f"posdep_dep_{label}"] = dep_counts.get(label, 0) / n_all if n_all else 0.0
|
|
342
|
+
|
|
343
|
+
content_lemmas = [t.lemma_.lower() for t in doc if t.is_alpha and t.pos_ in _CONTENT_POS]
|
|
344
|
+
out["lex_mtld"] = lex_div.mtld(content_lemmas)
|
|
345
|
+
out["lex_hdd"] = lex_div.hdd(content_lemmas)
|
|
346
|
+
zipfs = [zipf_frequency(w, "en") for w in content_lemmas]
|
|
347
|
+
out["lex_zipf_mean"] = sum(zipfs) / len(zipfs) if zipfs else None
|
|
348
|
+
|
|
349
|
+
all_lemmas = [t.lemma_.lower() for t in all_tokens]
|
|
350
|
+
out["dict_hedge_per_1k"] = (
|
|
351
|
+
_lexicon_matches(all_lemmas, _lexicon("hedges.txt")) / n_all * 1000 if n_all else 0.0
|
|
352
|
+
)
|
|
353
|
+
out["dict_booster_per_1k"] = (
|
|
354
|
+
_lexicon_matches(all_lemmas, _lexicon("boosters.txt")) / n_all * 1000 if n_all else 0.0
|
|
355
|
+
)
|
|
356
|
+
all_forms = [t.text.lower() for t in all_tokens]
|
|
357
|
+
out["dict_hype_per_1k"] = (
|
|
358
|
+
_lexicon_matches(all_forms, _hype_entries()) / n_all * 1000 if n_all else 0.0
|
|
359
|
+
)
|
|
360
|
+
out["dict_negation_per_1k"] = (
|
|
361
|
+
_lexicon_matches(all_lemmas, _lexicon("negations.txt")) / n_all * 1000 if n_all else 0.0
|
|
362
|
+
)
|
|
363
|
+
out["dict_conditional_per_1k"] = (
|
|
364
|
+
_lexicon_matches(all_lemmas, _lexicon("connectives_conditional.txt")) / n_all * 1000
|
|
365
|
+
if n_all
|
|
366
|
+
else 0.0
|
|
367
|
+
)
|
|
368
|
+
out["dict_conditional_clauses_per_sentence"] = _conditional_clauses_per_sentence(
|
|
369
|
+
doc, len(sentences)
|
|
370
|
+
)
|
|
371
|
+
|
|
372
|
+
second_person = sum(1 for t in all_tokens if t.text.lower() in _SECOND_PERSON)
|
|
373
|
+
out["dict_second_person_per_1k"] = second_person / n_all * 1000 if n_all else 0.0
|
|
374
|
+
|
|
375
|
+
cross_refs = len(_CROSS_REFERENCE.findall(prose))
|
|
376
|
+
out["dict_cross_reference_per_1k"] = cross_refs / n_all * 1000 if n_all else 0.0
|
|
377
|
+
|
|
378
|
+
plain_matches = 0
|
|
379
|
+
plain_replacements = []
|
|
380
|
+
for entry, plain in _plain_pairs():
|
|
381
|
+
hits = _lexicon_matches(all_lemmas, (entry,))
|
|
382
|
+
if hits:
|
|
383
|
+
plain_matches += hits
|
|
384
|
+
plain_replacements.append([" ".join(entry), plain])
|
|
385
|
+
out["dict_plain_replacement_per_1k"] = plain_matches / n_all * 1000 if n_all else 0.0
|
|
386
|
+
out["dict_plain_replacements_json"] = json.dumps(plain_replacements)
|
|
387
|
+
|
|
388
|
+
if len(sentences) < 2:
|
|
389
|
+
out["coh_lemma_overlap_adj"] = None
|
|
390
|
+
else:
|
|
391
|
+
sent_lemma_sets = [
|
|
392
|
+
{t.lemma_.lower() for t in sent if t.pos_ in _COH_CONTENT_POS} for sent in sentences
|
|
393
|
+
]
|
|
394
|
+
overlaps = []
|
|
395
|
+
for a, b in zip(sent_lemma_sets, sent_lemma_sets[1:]):
|
|
396
|
+
union = a | b
|
|
397
|
+
overlaps.append(len(a & b) / len(union) if union else 0.0)
|
|
398
|
+
out["coh_lemma_overlap_adj"] = sum(overlaps) / len(overlaps)
|
|
399
|
+
|
|
400
|
+
return out
|
|
401
|
+
|
|
402
|
+
|
|
403
|
+
def analyze_text(raw: str) -> dict:
|
|
404
|
+
"""One feature vector (no `path` key) for a raw document's text."""
|
|
405
|
+
struct, prose = _struct_features(raw)
|
|
406
|
+
return {**struct, **_nlp_features(prose)}
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def analyze_file(path: Path) -> dict:
|
|
410
|
+
text = path.read_text(encoding="utf-8")
|
|
411
|
+
return {"path": str(path), **analyze_text(text)}
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _write_jsonl(rows: list[dict], out) -> None:
|
|
415
|
+
for row in rows:
|
|
416
|
+
out.write(json.dumps(row) + "\n")
|
|
417
|
+
|
|
418
|
+
|
|
419
|
+
def _write_csv(rows: list[dict], out) -> None:
|
|
420
|
+
fieldnames = [k for k in rows[0] if not k.endswith("_json")]
|
|
421
|
+
writer = csv.DictWriter(out, fieldnames=fieldnames, extrasaction="ignore")
|
|
422
|
+
writer.writeheader()
|
|
423
|
+
writer.writerows(rows)
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def run_analyze(paths: list[str], fmt: str = "jsonl", out_path: str | None = None) -> int:
|
|
427
|
+
files = []
|
|
428
|
+
for p in paths:
|
|
429
|
+
path = Path(p)
|
|
430
|
+
if path.suffix not in {".md", ".txt"}:
|
|
431
|
+
print(f"skipping {p}: not a .md/.txt file", file=sys.stderr)
|
|
432
|
+
continue
|
|
433
|
+
if not path.is_file():
|
|
434
|
+
print(f"skipping {p}: no such file", file=sys.stderr)
|
|
435
|
+
continue
|
|
436
|
+
files.append(path)
|
|
437
|
+
|
|
438
|
+
if not files:
|
|
439
|
+
print("no .md/.txt files to analyze", file=sys.stderr)
|
|
440
|
+
return 1
|
|
441
|
+
|
|
442
|
+
rows = [analyze_file(path) for path in files]
|
|
443
|
+
write = _write_csv if fmt == "csv" else _write_jsonl
|
|
444
|
+
if out_path:
|
|
445
|
+
with open(out_path, "w", encoding="utf-8", newline="") as fh:
|
|
446
|
+
write(rows, fh)
|
|
447
|
+
else:
|
|
448
|
+
write(rows, sys.stdout)
|
|
449
|
+
return 0
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Cleanup helpers for preparing corpora before Timbro style analysis."""
|
|
2
|
+
|
|
3
|
+
from .latex import detex_file, preprocess_runtime_text, tex_to_markdown
|
|
4
|
+
from .papers import cleanup_markdown_file, cleanup_paper_markdown, clean_extracted_text, extract_prose_excerpt, split_frontmatter
|
|
5
|
+
|
|
6
|
+
__all__ = [
|
|
7
|
+
"detex_file",
|
|
8
|
+
"preprocess_runtime_text",
|
|
9
|
+
"tex_to_markdown",
|
|
10
|
+
"cleanup_markdown_file",
|
|
11
|
+
"cleanup_paper_markdown",
|
|
12
|
+
"clean_extracted_text",
|
|
13
|
+
"extract_prose_excerpt",
|
|
14
|
+
"split_frontmatter",
|
|
15
|
+
]
|
timbro/cleanup/latex.py
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""Utilities for converting LaTeX sources into Timbro-ready plain text."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import shutil
|
|
7
|
+
import subprocess
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from timbro.cleanup.papers import clean_extracted_text, cleanup_paper_markdown
|
|
11
|
+
|
|
12
|
+
_DROP_ENVS = (
|
|
13
|
+
"tikzpicture",
|
|
14
|
+
"figure",
|
|
15
|
+
"figure*",
|
|
16
|
+
"table",
|
|
17
|
+
"table*",
|
|
18
|
+
"tabular",
|
|
19
|
+
"tabularx",
|
|
20
|
+
"equation",
|
|
21
|
+
"equation*",
|
|
22
|
+
"align",
|
|
23
|
+
"align*",
|
|
24
|
+
"aligned",
|
|
25
|
+
"lstlisting",
|
|
26
|
+
"minted",
|
|
27
|
+
"verbatim",
|
|
28
|
+
"thebibliography",
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _normalize_detex_output(text: str) -> str:
|
|
33
|
+
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
|
34
|
+
text = re.sub(r"(?i)\.abstract\b", ".", text)
|
|
35
|
+
text = re.sub(r"(?i)^abstract\s*", "Abstract\n\n", text)
|
|
36
|
+
text = re.sub(r"\b(?:sub)*section([A-Z])", r"\n\n\1", text)
|
|
37
|
+
text = re.sub(r"\b(?:cite|ref|label)[A-Za-z0-9:_-]*\b", " ", text)
|
|
38
|
+
text = re.sub(r"\n{3,}", "\n\n", text)
|
|
39
|
+
text = re.sub(r"[ \t]+", " ", text)
|
|
40
|
+
return text.strip()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _strip_latex_source_noise(text: str) -> str:
|
|
44
|
+
text = re.sub(r"(?m)^\s*%.*$", "", text)
|
|
45
|
+
for env in _DROP_ENVS:
|
|
46
|
+
text = re.sub(rf"\\begin\{{{re.escape(env)}\}}.*?\\end\{{{re.escape(env)}\}}", "\n", text, flags=re.S)
|
|
47
|
+
text = re.sub(r"\\(?:label|ref|eqref|autoref|cref|Cref|pageref|cite|citet|citep|citealt|citealp|footnote)\*?(?:\[[^\]]*\])?\{[^}]*\}", " ", text)
|
|
48
|
+
text = re.sub(r"\\(?:begin|end)\{(?:itemize|enumerate|description)\}", "\n", text)
|
|
49
|
+
text = re.sub(r"\\item(?:\[[^\]]*\])?", "\n- ", text)
|
|
50
|
+
text = re.sub(r"\\(?:centering|smallskip|medskip|bigskip|noindent|vspace\*?|hspace\*?)\{[^}]*\}", " ", text)
|
|
51
|
+
text = re.sub(r"\\(?:centering|smallskip|medskip|bigskip|noindent)", " ", text)
|
|
52
|
+
return text
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def has_detex() -> bool:
|
|
56
|
+
return shutil.which("detex") is not None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def looks_like_latex(text: str) -> bool:
|
|
60
|
+
markers = [
|
|
61
|
+
r"\\begin\{",
|
|
62
|
+
r"\\end\{",
|
|
63
|
+
r"\\section\*?\{",
|
|
64
|
+
r"\\subsection\*?\{",
|
|
65
|
+
r"\\title\{",
|
|
66
|
+
r"\\author\{",
|
|
67
|
+
r"\\cite\{",
|
|
68
|
+
r"\\ref\{",
|
|
69
|
+
r"\\label\{",
|
|
70
|
+
r"\\documentclass",
|
|
71
|
+
]
|
|
72
|
+
hits = sum(bool(re.search(pat, text)) for pat in markers)
|
|
73
|
+
return hits >= 1 or "\\begin{document}" in text or "\\end{document}" in text
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def detex_text(text: str, *, replace_math: bool = True) -> str:
|
|
77
|
+
"""Strip LaTeX commands from raw `.tex` content using the external `detex` tool.
|
|
78
|
+
|
|
79
|
+
Raises `RuntimeError` if `detex` is not installed or returns a non-zero exit code.
|
|
80
|
+
"""
|
|
81
|
+
exe = shutil.which("detex")
|
|
82
|
+
if exe is None:
|
|
83
|
+
raise RuntimeError("detex is not installed. Install opendetex to ingest .tex files.")
|
|
84
|
+
|
|
85
|
+
cmd = [exe, "-n"]
|
|
86
|
+
if replace_math:
|
|
87
|
+
cmd.append("-r")
|
|
88
|
+
proc = subprocess.run(
|
|
89
|
+
cmd,
|
|
90
|
+
input=_strip_latex_source_noise(text),
|
|
91
|
+
text=True,
|
|
92
|
+
capture_output=True,
|
|
93
|
+
check=False,
|
|
94
|
+
)
|
|
95
|
+
if proc.returncode != 0:
|
|
96
|
+
raise RuntimeError(proc.stderr.strip() or "detex failed")
|
|
97
|
+
return _normalize_detex_output(proc.stdout)
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def detex_file(path: str | Path, *, replace_math: bool = True) -> str:
|
|
101
|
+
file_path = Path(path)
|
|
102
|
+
return detex_text(file_path.read_text(encoding="utf-8", errors="ignore"), replace_math=replace_math)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def tex_to_markdown(path: str | Path, *, replace_math: bool = True) -> str:
|
|
106
|
+
return cleanup_paper_markdown(detex_file(path, replace_math=replace_math))
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def preprocess_runtime_text(text: str) -> str:
|
|
110
|
+
"""Normalize text before scoring.
|
|
111
|
+
|
|
112
|
+
If the input looks like raw LaTeX and `detex` is available, strip the TeX markup
|
|
113
|
+
and run lightweight cleanup on the result while preserving the original order.
|
|
114
|
+
Otherwise, return the text with only lightweight whitespace normalization.
|
|
115
|
+
"""
|
|
116
|
+
if looks_like_latex(text) and has_detex():
|
|
117
|
+
return clean_extracted_text(detex_text(text))
|
|
118
|
+
return clean_extracted_text(text)
|