utterplan 0.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
utterplan/format.py ADDED
@@ -0,0 +1,34 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ from pathlib import Path
5
+ from typing import Any
6
+
7
+ from .model import FORMAT, SCHEMA_VERSION, UtterancePlan
8
+
9
+ SCHEMA_PATH = Path(__file__).resolve().parent.parent / "spec" / "utterplan.schema.json"
10
+ PACKAGE_SCHEMA_PATH = Path(__file__).resolve().parent / "utterplan.schema.json"
11
+
12
+
13
+ def validate_json(value: Any) -> UtterancePlan:
14
+ return UtterancePlan.from_dict(value)
15
+
16
+
17
+ def validate_file(path: str | Path) -> UtterancePlan:
18
+ return UtterancePlan.load(path)
19
+
20
+
21
+ def schema() -> dict[str, Any]:
22
+ path = SCHEMA_PATH if SCHEMA_PATH.exists() else PACKAGE_SCHEMA_PATH
23
+ return json.loads(path.read_text(encoding="utf-8"))
24
+
25
+
26
+ __all__ = [
27
+ "FORMAT",
28
+ "SCHEMA_VERSION",
29
+ "SCHEMA_PATH",
30
+ "PACKAGE_SCHEMA_PATH",
31
+ "validate_json",
32
+ "validate_file",
33
+ "schema",
34
+ ]
utterplan/hashing.py ADDED
@@ -0,0 +1,43 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ from typing import Any
6
+
7
+
8
+ def canonical_json(value: Any) -> str:
9
+ return json.dumps(
10
+ value, ensure_ascii=False, sort_keys=True, separators=(",", ":"), allow_nan=False
11
+ )
12
+
13
+
14
+ def semantic_hash(value: Any) -> str:
15
+ return "sha256:" + hashlib.sha256(canonical_json(value).encode("utf-8")).hexdigest()
16
+
17
+
18
+ def unit_hash_payload(unit: Any) -> dict[str, Any]:
19
+ marker_values: Any = getattr(unit, "marker_values", None)
20
+ if marker_values is None:
21
+ marker_values = getattr(unit, "marker_ids", ())
22
+ markers = []
23
+ for marker in marker_values:
24
+ if hasattr(marker, "to_dict"):
25
+ value = marker.to_dict()
26
+ value.pop("id", None)
27
+ markers.append(value)
28
+ else:
29
+ markers.append(marker)
30
+ return {
31
+ "hash_schema": "utterplan-unit-v1",
32
+ "segments": [
33
+ {
34
+ "text": segment.text,
35
+ "language": segment.language,
36
+ "directives": segment.directives.to_dict(),
37
+ "pause_before": segment.pause_before.to_dict(),
38
+ "pause_after": segment.pause_after.to_dict(),
39
+ }
40
+ for segment in unit.segments
41
+ ],
42
+ "markers": markers,
43
+ }
utterplan/language.py ADDED
@@ -0,0 +1,93 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Iterable, Sequence
4
+ from dataclasses import dataclass
5
+ from typing import Any
6
+
7
+ from .exceptions import LanguagePlanError
8
+
9
+
10
+ @dataclass(frozen=True, slots=True)
11
+ class LanguageRun:
12
+ id: str
13
+ spoken_start: int
14
+ spoken_end: int
15
+ language: str
16
+ source: str = "document-default"
17
+
18
+ def to_dict(self) -> dict[str, object]:
19
+ return {
20
+ "id": self.id,
21
+ "spoken_start": self.spoken_start,
22
+ "spoken_end": self.spoken_end,
23
+ "language": self.language,
24
+ "source": self.source,
25
+ }
26
+
27
+
28
+ def normalize_language(language: str, aliases: dict[str, str] | None = None) -> str:
29
+ if not isinstance(language, str):
30
+ raise LanguagePlanError("language must be a string")
31
+ value = language.strip().lower().replace("_", "-")
32
+ if not value:
33
+ raise LanguagePlanError("language must not be empty")
34
+ alias_map = {key.lower().replace("_", "-"): val for key, val in (aliases or {}).items()}
35
+ return alias_map.get(value, value).strip().lower().replace("_", "-")
36
+
37
+
38
+ def build_language_runs(
39
+ text: str,
40
+ spans: Iterable[tuple[int, int, str, str]],
41
+ default_language: str,
42
+ aliases: dict[str, str] | None = None,
43
+ ) -> tuple[LanguageRun, ...]:
44
+ default = normalize_language(default_language, aliases)
45
+ explicit = [
46
+ (start, end, normalize_language(lang, aliases), source)
47
+ for start, end, lang, source in spans
48
+ ]
49
+ for start, end, _lang, _source in explicit:
50
+ if not (0 <= start < end <= len(text)):
51
+ raise LanguagePlanError(f"language span {start}:{end} is outside spoken text")
52
+ for index, left in enumerate(explicit):
53
+ for right in explicit[index + 1 :]:
54
+ if left[2] == right[2] or not (left[0] < right[1] and right[0] < left[1]):
55
+ continue
56
+ nested = (left[0] <= right[0] and right[1] <= left[1]) or (
57
+ right[0] <= left[0] and left[1] <= right[1]
58
+ )
59
+ if not nested:
60
+ raise LanguagePlanError(
61
+ f"Conflicting crossing language spans: {left[0]}:{left[1]} and {right[0]}:{right[1]}"
62
+ )
63
+ positions = sorted({0, len(text), *(point for span in explicit for point in span[:2])})
64
+ runs: list[LanguageRun] = []
65
+ for start, end in zip(positions, positions[1:], strict=False):
66
+ if start == end:
67
+ continue
68
+ covering = [span for span in explicit if span[0] <= start and end <= span[1]]
69
+ chosen = min(covering, key=lambda span: (span[1] - span[0], -span[0])) if covering else None
70
+ language = chosen[2] if chosen else default
71
+ source = chosen[3] if chosen else "document-default"
72
+ if runs and runs[-1].language == language and runs[-1].spoken_end == start:
73
+ runs[-1] = LanguageRun(runs[-1].id, runs[-1].spoken_start, end, language, source)
74
+ else:
75
+ runs.append(LanguageRun(f"lang-{len(runs)}", start, end, language, source))
76
+ return tuple(runs)
77
+
78
+
79
+ def spans_from_annotations(annotations: Sequence[Any]) -> list[tuple[int, int, str, str]]:
80
+ result: list[tuple[int, int, str, str]] = []
81
+ for item in annotations:
82
+ attrs = getattr(item, "attrs", {})
83
+ language = attrs.get("lang") or attrs.get("language")
84
+ if not language or str(attrs.get("scope", "")).lower() in {"pronunciation", "phoneme"}:
85
+ continue
86
+ start = getattr(item, "spoken_start", None)
87
+ end = getattr(item, "spoken_end", None)
88
+ if start is None or end is None:
89
+ start = getattr(item, "structural_start", None)
90
+ end = getattr(item, "structural_end", None)
91
+ if start is not None and end is not None:
92
+ result.append((int(start), int(end), str(language), "explicit-span"))
93
+ return result
@@ -0,0 +1,179 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from dataclasses import dataclass
5
+ from typing import Any
6
+
7
+ from .config import LinguisticsConfig
8
+ from .language import LanguageRun
9
+ from .model import TokenAnnotation
10
+ from .spacy_models import resolve_spacy_model
11
+
12
+
13
+ @dataclass(frozen=True, slots=True)
14
+ class RunAnalysis:
15
+ """Request-local linguistic state. Never serialize this object."""
16
+
17
+ language: str
18
+ start: int
19
+ end: int
20
+ tokens: tuple[TokenAnnotation, ...]
21
+ provider_doc: object | None = None
22
+ model_name: str | None = None
23
+
24
+
25
+ @dataclass(frozen=True, slots=True)
26
+ class LinguisticAnalysis:
27
+ language: str
28
+ text: str
29
+ tokens: tuple[TokenAnnotation, ...]
30
+ model_name: str | None = None
31
+ provider_doc: object | None = None
32
+
33
+
34
+ class LinguisticResourcePool:
35
+ """Cache local spaCy pipelines without downloading a model."""
36
+
37
+ def __init__(self) -> None:
38
+ self._pipelines: dict[str, Any] = {}
39
+
40
+ def pipeline(self, model: str | None, *, require: bool = False) -> Any | None:
41
+ if not model:
42
+ return None
43
+ if model in self._pipelines:
44
+ return self._pipelines[model]
45
+ try:
46
+ import spacy
47
+
48
+ pipeline = spacy.load(model)
49
+ except (ImportError, OSError, ValueError) as exc:
50
+ if require:
51
+ raise RuntimeError(f"Requested spaCy model {model!r} is unavailable") from exc
52
+ return None
53
+ self._pipelines[model] = pipeline
54
+ return pipeline
55
+
56
+ def clear(self) -> None:
57
+ self._pipelines.clear()
58
+
59
+ def _select_model(self, language: str, config: LinguisticsConfig) -> str | None:
60
+ if config.spacy_model:
61
+ return config.spacy_model
62
+ if config.spacy_model_size:
63
+ return resolve_spacy_model(language, config.spacy_model_size)
64
+ if config.use_spacy is False:
65
+ return None
66
+ if config.use_spacy is not True:
67
+ return None
68
+ try:
69
+ import spacy
70
+
71
+ installed = set(spacy.util.get_installed_models())
72
+ except ImportError:
73
+ return None
74
+ base = language.split("-", 1)[0].lower()
75
+ families = (f"{base}_core_web_", f"{base}_core_news_")
76
+ for suffix in ("trf", "lg", "md", "sm"):
77
+ for family in families:
78
+ candidate = family + suffix
79
+ if candidate in installed:
80
+ return candidate
81
+ return None
82
+
83
+ def analyze(self, text: str, run: LanguageRun, config: LinguisticsConfig) -> LinguisticAnalysis:
84
+ model = self._select_model(run.language, config)
85
+ use_spacy = config.use_spacy if config.use_spacy is not None else model is not None
86
+ if use_spacy:
87
+ pipeline = self.pipeline(model, require=config.require_spacy)
88
+ if pipeline is not None:
89
+ doc = pipeline(text)
90
+ return LinguisticAnalysis(
91
+ run.language,
92
+ text,
93
+ tuple(
94
+ TokenAnnotation(
95
+ int(token.idx),
96
+ int(token.idx + len(token.text)),
97
+ str(token.text),
98
+ getattr(token, "pos_", None) or None,
99
+ getattr(token, "tag_", None) or None,
100
+ getattr(token, "lemma_", None) or None,
101
+ run.language,
102
+ f"token-{i}",
103
+ )
104
+ for i, token in enumerate(doc)
105
+ if token.text
106
+ ),
107
+ model,
108
+ doc,
109
+ )
110
+ if config.require_spacy:
111
+ raise RuntimeError("spaCy is required but no local model is available")
112
+ return LinguisticAnalysis(
113
+ run.language, text, _fallback_tokens(text, run.language), None, None
114
+ )
115
+
116
+
117
+ def _fallback_tokens(text: str, language: str) -> tuple[TokenAnnotation, ...]:
118
+ return tuple(
119
+ TokenAnnotation(
120
+ match.start(),
121
+ match.end(),
122
+ match.group(0),
123
+ None,
124
+ None,
125
+ match.group(0).lower(),
126
+ language,
127
+ f"token-{i}",
128
+ )
129
+ for i, match in enumerate(re.finditer(r"\S+", text))
130
+ )
131
+
132
+
133
+ def analyze_run_analyses(
134
+ text: str,
135
+ runs: tuple[LanguageRun, ...],
136
+ config: LinguisticsConfig,
137
+ pool: LinguisticResourcePool,
138
+ ) -> tuple[RunAnalysis, ...]:
139
+ analyses: list[RunAnalysis] = []
140
+ for run in runs:
141
+ local = text[run.spoken_start : run.spoken_end]
142
+ analysis = pool.analyze(local, run, config)
143
+ tokens = tuple(
144
+ TokenAnnotation(
145
+ token.spoken_start + run.spoken_start,
146
+ token.spoken_end + run.spoken_start,
147
+ token.text,
148
+ token.pos,
149
+ token.tag,
150
+ token.lemma,
151
+ token.language,
152
+ f"token-{sum(len(item.tokens) for item in analyses) + i}",
153
+ )
154
+ for i, token in enumerate(analysis.tokens)
155
+ )
156
+ analyses.append(
157
+ RunAnalysis(
158
+ run.language,
159
+ run.spoken_start,
160
+ run.spoken_end,
161
+ tokens,
162
+ analysis.provider_doc,
163
+ analysis.model_name,
164
+ )
165
+ )
166
+ return tuple(analyses)
167
+
168
+
169
+ def analyze_runs(
170
+ text: str,
171
+ runs: tuple[LanguageRun, ...],
172
+ config: LinguisticsConfig,
173
+ pool: LinguisticResourcePool,
174
+ ) -> tuple[TokenAnnotation, ...]:
175
+ return tuple(
176
+ token
177
+ for analysis in analyze_run_analyses(text, runs, config, pool)
178
+ for token in analysis.tokens
179
+ )