limatus 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
limatus/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """Limatus: a diagnose-and-steer quality loop for AI-generated content."""
2
+
3
+ __version__ = "0.2.0"
limatus/_util.py ADDED
@@ -0,0 +1,21 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ from typing import Any
6
+
7
+
8
+ def hash_stable(value: Any) -> str:
9
+ if isinstance(value, str):
10
+ payload = value
11
+ else:
12
+ payload = json.dumps(value, sort_keys=True, separators=(",", ":"), default=str)
13
+ return hashlib.sha256(payload.encode("utf-8")).hexdigest()
14
+
15
+
16
+ def hash_short(value: Any) -> str:
17
+ return hash_stable(value)[:16]
18
+
19
+
20
+ # Default model for constrained rewrite-option generation.
21
+ DEFAULT_EDITORIAL_REWRITE_MODEL = "gpt-5.4-mini"
limatus/cli.py ADDED
@@ -0,0 +1,43 @@
1
+ """Limatus CLI entry point.
2
+
3
+ limatus diagnose --draft <file> --profile <style-profile.yml> [...]
4
+ limatus options --draft <file> --profile <style-profile.yml> \\
5
+ --diagnosis <diagnosis.json> --decisions <decisions.json> \\
6
+ --skill <editorial-rewrite-skill.yml> [...]
7
+
8
+ See editorial_commands.py for each subcommand's full flag set.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import sys
13
+
14
+ from . import __version__
15
+ from .editorial_commands import editorial_diagnose, editorial_options
16
+
17
+ COMMANDS = {
18
+ "diagnose": editorial_diagnose,
19
+ "options": editorial_options,
20
+ }
21
+
22
+
23
+ def main(argv: list[str] | None = None) -> int:
24
+ args = list(sys.argv[1:] if argv is None else argv)
25
+ if not args or args[0] in {"-h", "--help"}:
26
+ print(__doc__)
27
+ return 0
28
+ if args[0] in {"-V", "--version"}:
29
+ print(f"limatus {__version__}")
30
+ return 0
31
+
32
+ command, flags = args[0], args[1:]
33
+ handler = COMMANDS.get(command)
34
+ if handler is None:
35
+ print(f"limatus: unknown command '{command}'. Try one of: {', '.join(sorted(COMMANDS))}")
36
+ return 1
37
+
38
+ handler(flags)
39
+ return 0
40
+
41
+
42
+ if __name__ == "__main__":
43
+ sys.exit(main())
@@ -0,0 +1,111 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from pathlib import Path
7
+
8
+ from .editorial_diagnosis import diagnose_draft
9
+ from .editorial_diagnosis_schema import validate_diagnosis
10
+ from .editorial_markup import render_annotated_markus, render_annotated_xml
11
+ from .editorial_options_schema import validate_decisions
12
+ from .editorial_rewrite_options import generate_rewrite_options
13
+ from .editorial_style import load_style_profile
14
+ from ._util import DEFAULT_EDITORIAL_REWRITE_MODEL
15
+
16
+
17
+ def editorial_diagnose(flags: list[str]) -> None:
18
+ parser = argparse.ArgumentParser(prog="limatus diagnose")
19
+ input_group = parser.add_mutually_exclusive_group(required=True)
20
+ input_group.add_argument("--draft", help="Path to the draft file (read-only).")
21
+ input_group.add_argument("--text", help="Draft text to diagnose without reading a file.")
22
+ parser.add_argument("--profile", required=True, help="Path to the style profile YAML.")
23
+ parser.add_argument("--output", default="", help="Optional path to write diagnostic JSON.")
24
+ parser.add_argument(
25
+ "--markup-out",
26
+ default="",
27
+ help="Optional path to write Markus-annotated Markdown (editorial-finding directives).",
28
+ )
29
+ parser.add_argument(
30
+ "--xml-out",
31
+ default="",
32
+ help="Optional path to write editorial annotation XML.",
33
+ )
34
+ args = parser.parse_args(flags)
35
+
36
+ profile_path = Path(args.profile).resolve()
37
+ if args.draft:
38
+ draft_path = Path(args.draft).resolve()
39
+ if not draft_path.is_file():
40
+ raise ValueError(f"Draft file not found: {draft_path}")
41
+ draft_text = draft_path.read_text(encoding="utf-8")
42
+ else:
43
+ draft_text = args.text
44
+
45
+ style_profile = load_style_profile(profile_path)
46
+ diagnosis = diagnose_draft(draft_text, style_profile=style_profile)
47
+ rendered = json.dumps(diagnosis, indent=2) + "\n"
48
+
49
+ if args.markup_out:
50
+ markup_path = Path(args.markup_out).resolve()
51
+ markup_path.write_text(render_annotated_markus(draft_text, diagnosis), encoding="utf-8")
52
+
53
+ if args.xml_out:
54
+ xml_path = Path(args.xml_out).resolve()
55
+ xml_path.write_text(render_annotated_xml(draft_text, diagnosis), encoding="utf-8")
56
+
57
+ if args.output:
58
+ output_path = Path(args.output).resolve()
59
+ output_path.write_text(rendered, encoding="utf-8")
60
+ return
61
+
62
+ sys.stdout.write(rendered)
63
+
64
+
65
+ def editorial_options(flags: list[str]) -> None:
66
+ parser = argparse.ArgumentParser(prog="limatus options")
67
+ parser.add_argument("--draft", required=True, help="Path to the draft file (read-only).")
68
+ parser.add_argument("--profile", required=True, help="Path to the style profile YAML.")
69
+ parser.add_argument("--diagnosis", required=True, help="Path to validated diagnostic JSON.")
70
+ parser.add_argument("--decisions", required=True, help="Path to steering decisions JSON.")
71
+ parser.add_argument("--output", default="", help="Optional path to write options JSON.")
72
+ parser.add_argument("--model", default=DEFAULT_EDITORIAL_REWRITE_MODEL, help="OpenAI model id.")
73
+ parser.add_argument(
74
+ "--skill",
75
+ required=True,
76
+ help="Path to editorial rewrite skill YAML.",
77
+ )
78
+ args = parser.parse_args(flags)
79
+
80
+ draft_path = Path(args.draft).resolve()
81
+ if not draft_path.is_file():
82
+ raise ValueError(f"Draft file not found: {draft_path}")
83
+ draft_text = draft_path.read_text(encoding="utf-8")
84
+
85
+ diagnosis_path = Path(args.diagnosis).resolve()
86
+ decisions_path = Path(args.decisions).resolve()
87
+ diagnosis_payload = json.loads(diagnosis_path.read_text(encoding="utf-8"))
88
+ decisions_payload = json.loads(decisions_path.read_text(encoding="utf-8"))
89
+ if not isinstance(decisions_payload, list):
90
+ raise ValueError("Decisions JSON must be a list.")
91
+
92
+ style_profile = load_style_profile(Path(args.profile).resolve())
93
+ validate_diagnosis(diagnosis_payload)
94
+ validate_decisions(decisions_payload)
95
+
96
+ options = generate_rewrite_options(
97
+ draft_text,
98
+ style_profile=style_profile,
99
+ diagnosis=diagnosis_payload,
100
+ decisions=decisions_payload,
101
+ model=args.model,
102
+ skill_path=args.skill,
103
+ )
104
+ rendered = json.dumps(options, indent=2) + "\n"
105
+
106
+ if args.output:
107
+ output_path = Path(args.output).resolve()
108
+ output_path.write_text(rendered, encoding="utf-8")
109
+ return
110
+
111
+ sys.stdout.write(rendered)
@@ -0,0 +1,51 @@
1
+ from __future__ import annotations
2
+
3
+ from pathlib import Path
4
+ from typing import Any
5
+
6
+ import yaml
7
+
8
+ from .editorial_style import StyleProfileValidationError
9
+
10
+
11
+ def load_editorial_corpus_manifest(path: str | Path) -> dict[str, list[dict[str, Any]]]:
12
+ manifest_path = Path(path).resolve()
13
+ raw = yaml.safe_load(manifest_path.read_text(encoding="utf-8"))
14
+ if not isinstance(raw, dict):
15
+ raise StyleProfileValidationError(f"Editorial corpus manifest must be a mapping: {manifest_path}")
16
+
17
+ manifest: dict[str, list[dict[str, Any]]] = {"mustFail": [], "mustPass": []}
18
+ for section in ("mustFail", "mustPass"):
19
+ entries = raw.get(section)
20
+ if entries is None:
21
+ continue
22
+ if not isinstance(entries, list):
23
+ raise StyleProfileValidationError(f"{section} must be a list in {manifest_path}")
24
+ normalized: list[dict[str, Any]] = []
25
+ for index, entry in enumerate(entries):
26
+ if not isinstance(entry, dict):
27
+ raise StyleProfileValidationError(f"{section}[{index}] must be a mapping in {manifest_path}")
28
+ entry_id = str(entry.get("id", "")).strip()
29
+ rel_path = str(entry.get("path", "")).strip()
30
+ if not entry_id or not rel_path:
31
+ raise StyleProfileValidationError(
32
+ f"{section}[{index}] requires id and path in {manifest_path}"
33
+ )
34
+ draft_path = (manifest_path.parent / rel_path).resolve()
35
+ if not draft_path.is_file():
36
+ raise StyleProfileValidationError(f"Corpus draft not found: {draft_path}")
37
+ normalized_entry = {"id": entry_id, "path": rel_path, "draftPath": str(draft_path)}
38
+ if section == "mustFail":
39
+ expect_terms = entry.get("expectTerms")
40
+ if not isinstance(expect_terms, list) or not expect_terms:
41
+ raise StyleProfileValidationError(
42
+ f"{section}[{index}].expectTerms must be a non-empty list in {manifest_path}"
43
+ )
44
+ normalized_entry["expectTerms"] = [str(term).strip() for term in expect_terms if str(term).strip()]
45
+ else:
46
+ source_sample = entry.get("sourceSample")
47
+ if isinstance(source_sample, str) and source_sample.strip():
48
+ normalized_entry["sourceSample"] = source_sample.strip()
49
+ normalized.append(normalized_entry)
50
+ manifest[section] = normalized
51
+ return manifest
@@ -0,0 +1,242 @@
1
+ from __future__ import annotations
2
+
3
+ import gzip
4
+ from dataclasses import dataclass
5
+ from typing import Any
6
+
7
+ from .editorial_diagnosis_schema import stable_finding_id
8
+ from .editorial_text import sentence_spans, tokenize, word_count
9
+
10
+ # Function-word inventory aligned with Ure (1971) / Halliday lexical-density practice.
11
+ FUNCTION_WORDS: frozenset[str] = frozenset(
12
+ {
13
+ "a",
14
+ "about",
15
+ "above",
16
+ "after",
17
+ "again",
18
+ "against",
19
+ "all",
20
+ "am",
21
+ "an",
22
+ "and",
23
+ "any",
24
+ "are",
25
+ "as",
26
+ "at",
27
+ "be",
28
+ "because",
29
+ "been",
30
+ "before",
31
+ "being",
32
+ "below",
33
+ "between",
34
+ "both",
35
+ "but",
36
+ "by",
37
+ "can",
38
+ "could",
39
+ "did",
40
+ "do",
41
+ "does",
42
+ "doing",
43
+ "down",
44
+ "during",
45
+ "each",
46
+ "few",
47
+ "for",
48
+ "from",
49
+ "further",
50
+ "had",
51
+ "has",
52
+ "have",
53
+ "having",
54
+ "he",
55
+ "her",
56
+ "here",
57
+ "hers",
58
+ "herself",
59
+ "him",
60
+ "himself",
61
+ "his",
62
+ "how",
63
+ "i",
64
+ "if",
65
+ "in",
66
+ "into",
67
+ "is",
68
+ "it",
69
+ "its",
70
+ "itself",
71
+ "just",
72
+ "me",
73
+ "more",
74
+ "most",
75
+ "my",
76
+ "myself",
77
+ "no",
78
+ "nor",
79
+ "not",
80
+ "now",
81
+ "of",
82
+ "off",
83
+ "on",
84
+ "once",
85
+ "only",
86
+ "or",
87
+ "other",
88
+ "our",
89
+ "ours",
90
+ "ourselves",
91
+ "out",
92
+ "over",
93
+ "own",
94
+ "same",
95
+ "she",
96
+ "should",
97
+ "so",
98
+ "some",
99
+ "such",
100
+ "than",
101
+ "that",
102
+ "the",
103
+ "their",
104
+ "theirs",
105
+ "them",
106
+ "themselves",
107
+ "then",
108
+ "there",
109
+ "these",
110
+ "they",
111
+ "this",
112
+ "those",
113
+ "through",
114
+ "to",
115
+ "too",
116
+ "under",
117
+ "until",
118
+ "up",
119
+ "very",
120
+ "was",
121
+ "we",
122
+ "were",
123
+ "what",
124
+ "when",
125
+ "where",
126
+ "which",
127
+ "while",
128
+ "who",
129
+ "whom",
130
+ "why",
131
+ "will",
132
+ "with",
133
+ "would",
134
+ "you",
135
+ "your",
136
+ "yours",
137
+ "yourself",
138
+ "yourselves",
139
+ }
140
+ )
141
+
142
+
143
+ @dataclass(frozen=True)
144
+ class DensityThresholds:
145
+ min_words: int
146
+ min_lexical_density: float
147
+ max_gzip_ratio: float
148
+
149
+
150
+ @dataclass(frozen=True)
151
+ class DensitySummary:
152
+ word_count: int
153
+ sentence_count: int
154
+ lexical_density: float
155
+ gzip_ratio: float
156
+
157
+
158
+ @dataclass(frozen=True)
159
+ class DensityAnalysis:
160
+ summary: DensitySummary
161
+ findings: tuple[dict[str, Any], ...]
162
+
163
+
164
+ def lexical_density(tokens: list[str]) -> float:
165
+ if not tokens:
166
+ return 0.0
167
+ content_words = sum(1 for token in tokens if token not in FUNCTION_WORDS)
168
+ return content_words / len(tokens)
169
+
170
+
171
+ def gzip_ratio(text: str) -> float:
172
+ encoded = text.encode("utf-8")
173
+ if not encoded:
174
+ return 1.0
175
+ compressed = gzip.compress(encoded)
176
+ return len(compressed) / len(encoded)
177
+
178
+
179
+ def analyze_density(text: str, thresholds: DensityThresholds) -> DensityAnalysis:
180
+ normalized = text.replace("\r\n", "\n")
181
+ tokens = tokenize(normalized)
182
+ total_words = len(tokens)
183
+ total_sentences = len(sentence_spans(normalized))
184
+ density_value = lexical_density(tokens)
185
+ compression_value = gzip_ratio(normalized)
186
+
187
+ summary = DensitySummary(
188
+ word_count=total_words,
189
+ sentence_count=total_sentences,
190
+ lexical_density=round(density_value, 4),
191
+ gzip_ratio=round(compression_value, 4),
192
+ )
193
+
194
+ if total_words < thresholds.min_words:
195
+ return DensityAnalysis(summary=summary, findings=())
196
+
197
+ findings: list[dict[str, Any]] = []
198
+ if density_value < thresholds.min_lexical_density:
199
+ findings.append(
200
+ _document_finding(
201
+ "low_lexical_density",
202
+ normalized,
203
+ (
204
+ "Document lexical density is below the profile threshold "
205
+ f"({density_value:.3f} < {thresholds.min_lexical_density:.3f})."
206
+ ),
207
+ )
208
+ )
209
+ if compression_value <= thresholds.max_gzip_ratio:
210
+ findings.append(
211
+ _document_finding(
212
+ "high_compressibility",
213
+ normalized,
214
+ (
215
+ "Document compresses unusually well for its length "
216
+ f"(gzip ratio {compression_value:.3f} <= {thresholds.max_gzip_ratio:.3f})."
217
+ ),
218
+ )
219
+ )
220
+
221
+ return DensityAnalysis(summary=summary, findings=tuple(findings))
222
+
223
+
224
+ def density_summary_as_dict(summary: DensitySummary) -> dict[str, Any]:
225
+ return {
226
+ "wordCount": summary.word_count,
227
+ "sentenceCount": summary.sentence_count,
228
+ "lexicalDensity": summary.lexical_density,
229
+ "gzipRatio": summary.gzip_ratio,
230
+ }
231
+
232
+
233
+ def _document_finding(kind: str, text: str, rationale: str) -> dict[str, Any]:
234
+ start = 0
235
+ end = len(text)
236
+ return {
237
+ "id": stable_finding_id(kind, text, start, end),
238
+ "kind": kind,
239
+ "excerpt": text[start:end],
240
+ "span": {"start": start, "end": end},
241
+ "rationale": rationale,
242
+ }