lladar 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
lladar/__init__.py ADDED
@@ -0,0 +1,25 @@
1
+ from .api import create_test_dataset
2
+ from .evaluation import evaluate
3
+ from .exceptions import (
4
+ ChunkingError,
5
+ DatasetValidationError,
6
+ EvaluationError,
7
+ GenerationError,
8
+ KnowledgeLoadError,
9
+ LladarError,
10
+ ProviderError,
11
+ )
12
+
13
+ __all__ = [
14
+ "ChunkingError",
15
+ "DatasetValidationError",
16
+ "GenerationError",
17
+ "KnowledgeLoadError",
18
+ "LladarError",
19
+ "ProviderError",
20
+ "create_test_dataset",
21
+ "eval",
22
+ "evaluate",
23
+ ]
24
+
25
+ eval = evaluate
lladar/api.py ADDED
@@ -0,0 +1,335 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ from pathlib import Path
5
+ from typing import Any, Literal
6
+
7
+ from .cache import cache_key, read_cache, write_cache
8
+ from .chunking import (
9
+ KnowledgeChunk,
10
+ chunk_text,
11
+ deserialize_chunks,
12
+ fallback_chunks,
13
+ semantic_chunk_text,
14
+ serialize_chunks,
15
+ )
16
+ from .exceptions import ChunkingError, DatasetValidationError, ProviderError
17
+ from .loaders import KnowledgeInput, load_knowledge
18
+ from .model_profiles import resolve_model_profile
19
+ from .output import write_dataset
20
+ from .progress import ProgressReporter
21
+ from .prompts import build_generation_prompt, resolve_strategy
22
+ from .providers import AkashaProvider, LLMProvider
23
+ from .validation import validate_generated_pair
24
+
25
+
26
+ DEFAULT_MODEL = "gemini:gemini-2.5-flash"
27
+
28
+
29
+ def create_test_dataset(
30
+ knowledge: KnowledgeInput,
31
+ prompt: str | None = None,
32
+ chunk_size: int | Literal["auto"] = 2000,
33
+ overlap: float = 0.1,
34
+ num_pairs: int = 1,
35
+ model: str = DEFAULT_MODEL,
36
+ output: str | Path | None = None,
37
+ format: str = "jsonl",
38
+ *,
39
+ prompt_file: str | Path | None = None,
40
+ provider: LLMProvider | None = None,
41
+ env_file: str | Path = ".env",
42
+ temperature: float = 0.0,
43
+ max_input_tokens: int | None = None,
44
+ max_output_tokens: int | None = None,
45
+ auto_window_ratio: float | None = None,
46
+ strict: bool = False,
47
+ force: bool = False,
48
+ cache: bool = False,
49
+ cache_dir: str | Path = ".lladar/cache",
50
+ refresh_cache: bool = False,
51
+ verbose: bool = True,
52
+ ) -> list[dict[str, Any]]:
53
+ if output is not None and Path(output).exists() and not force:
54
+ raise FileExistsError(f"output already exists: {output}")
55
+ if prompt is not None and prompt_file is not None:
56
+ raise ValueError("prompt and prompt_file cannot be used together")
57
+ prompt_source = _prompt_source(prompt, prompt_file)
58
+ if prompt_file is not None:
59
+ prompt = Path(prompt_file).read_text(encoding="utf-8")
60
+ if num_pairs <= 0:
61
+ raise ValueError("num_pairs must be greater than 0")
62
+
63
+ profile = resolve_model_profile(
64
+ model,
65
+ max_input_tokens=max_input_tokens,
66
+ max_output_tokens=max_output_tokens,
67
+ auto_window_ratio=auto_window_ratio,
68
+ )
69
+ strategy, strategy_text = resolve_strategy(prompt)
70
+ active_provider = provider or AkashaProvider(
71
+ env_file=str(env_file),
72
+ max_input_tokens=profile.max_input_tokens,
73
+ max_output_tokens=profile.max_output_tokens,
74
+ )
75
+ reporter = ProgressReporter(verbose)
76
+ reporter.configuration(
77
+ {
78
+ "knowledge": _knowledge_paths(knowledge),
79
+ "strategy": strategy if strategy == "ambiguity" else "custom",
80
+ "prompt_source": prompt_source,
81
+ "chunk_size": chunk_size,
82
+ "overlap": overlap,
83
+ "num_pairs": num_pairs,
84
+ "model": model,
85
+ "output": output,
86
+ "format": format,
87
+ "env_file": env_file,
88
+ "temperature": temperature,
89
+ "max_input_tokens": profile.max_input_tokens,
90
+ "max_output_tokens": profile.max_output_tokens,
91
+ "auto_window_ratio": profile.auto_window_ratio,
92
+ "strict": strict,
93
+ "force": force,
94
+ "cache": cache,
95
+ "cache_dir": cache_dir,
96
+ "refresh_cache": refresh_cache,
97
+ "verbose": verbose,
98
+ "provider": type(active_provider).__name__,
99
+ }
100
+ )
101
+
102
+ sources = load_knowledge(knowledge)
103
+ reporter.emit("SOURCE", f"loaded={len(sources)}")
104
+ prepared: list[tuple[Path, int, KnowledgeChunk]] = []
105
+ for source_index, (path, source_text) in enumerate(sources, start=1):
106
+ reporter.emit(
107
+ "SOURCE",
108
+ f"{source_index}/{len(sources)} path={path} characters={len(source_text)}",
109
+ )
110
+ if chunk_size == "auto":
111
+ chunks = _auto_chunks(
112
+ source_text,
113
+ active_provider,
114
+ source_path=path,
115
+ reporter=reporter,
116
+ model=model,
117
+ temperature=temperature,
118
+ max_input_tokens=profile.max_input_tokens,
119
+ max_output_tokens=profile.max_output_tokens,
120
+ auto_window_ratio=profile.auto_window_ratio,
121
+ strict=strict,
122
+ cache=cache,
123
+ cache_dir=cache_dir,
124
+ refresh_cache=refresh_cache,
125
+ )
126
+ elif isinstance(chunk_size, int):
127
+ chunks = [
128
+ KnowledgeChunk(chunk, -1, -1, (), "character")
129
+ for chunk in chunk_text(source_text, chunk_size, overlap)
130
+ ]
131
+ else:
132
+ raise ValueError('chunk_size must be a positive integer or "auto"')
133
+ reporter.emit(
134
+ "CHUNK",
135
+ f"source={path} method={chunks[0].method if chunks else chunk_size} count={len(chunks)}",
136
+ )
137
+ prepared.extend(
138
+ (path, chunk_index, knowledge_chunk)
139
+ for chunk_index, knowledge_chunk in enumerate(chunks)
140
+ )
141
+
142
+ total_pairs = len(prepared) * num_pairs
143
+ reporter.emit("PAIR", f"planned={total_pairs} chunks={len(prepared)}")
144
+ dataset: list[dict[str, Any]] = []
145
+ completed = 0
146
+ for path, chunk_index, knowledge_chunk in prepared:
147
+ chunk = knowledge_chunk.text
148
+ for pair_index in range(num_pairs):
149
+ key = cache_key(
150
+ chunk,
151
+ knowledge_chunk.method,
152
+ knowledge_chunk.source_start,
153
+ knowledge_chunk.source_end,
154
+ strategy,
155
+ model,
156
+ temperature,
157
+ profile.max_input_tokens,
158
+ profile.max_output_tokens,
159
+ profile.auto_window_ratio,
160
+ pair_index,
161
+ )
162
+ generated = None
163
+ from_cache = False
164
+ if cache and not refresh_cache:
165
+ generated = read_cache(cache_dir, key)
166
+ from_cache = generated is not None
167
+ reporter.emit(
168
+ "CACHE",
169
+ f"pair={completed + 1}/{total_pairs} {'hit' if from_cache else 'miss'}",
170
+ )
171
+ if generated is None:
172
+ last_error: DatasetValidationError | ProviderError | None = None
173
+ for attempt in range(1, 4):
174
+ try:
175
+ candidate = active_provider.generate_structured(
176
+ build_generation_prompt(chunk, strategy_text),
177
+ model=model,
178
+ temperature=temperature,
179
+ )
180
+ generated = validate_generated_pair(candidate)
181
+ break
182
+ except (DatasetValidationError, ProviderError) as error:
183
+ last_error = error
184
+ reporter.emit(
185
+ "RETRY",
186
+ f"pair={completed + 1}/{total_pairs} attempt={attempt}/3 error_type={type(error).__name__}",
187
+ )
188
+ else:
189
+ assert last_error is not None
190
+ completed += 1
191
+ if strict:
192
+ raise last_error
193
+ reporter.emit(
194
+ "WARN",
195
+ f"pair={completed}/{total_pairs} skipped after 3 failed attempts",
196
+ )
197
+ reporter.pair(completed, total_pairs, "skipped")
198
+ continue
199
+ if cache:
200
+ write_cache(cache_dir, key, generated)
201
+ reporter.emit("CACHE", f"pair={completed + 1}/{total_pairs} saved")
202
+ else:
203
+ generated = validate_generated_pair(generated)
204
+
205
+ item_id = hashlib.sha256(
206
+ f"{path.resolve()}\0{chunk}\0{chunk_index}\0{pair_index}".encode(
207
+ "utf-8"
208
+ )
209
+ ).hexdigest()[:24]
210
+ metadata: dict[str, Any] = {
211
+ "strategy": strategy,
212
+ "model": model,
213
+ "temperature": temperature,
214
+ }
215
+ if knowledge_chunk.method != "character":
216
+ metadata.update(
217
+ {
218
+ "chunk_method": knowledge_chunk.method,
219
+ "source_start": knowledge_chunk.source_start,
220
+ "source_end": knowledge_chunk.source_end,
221
+ "knowledge_facts": list(knowledge_chunk.knowledge_facts),
222
+ }
223
+ )
224
+ dataset.append(
225
+ {
226
+ "schema_version": "1.0",
227
+ "id": item_id,
228
+ "source_file": str(path),
229
+ "chunk_index": chunk_index,
230
+ "source_text": chunk,
231
+ **generated,
232
+ "bias_type": "unsupported_assumption",
233
+ "metadata": metadata,
234
+ }
235
+ )
236
+ completed += 1
237
+ reporter.pair(
238
+ completed,
239
+ total_pairs,
240
+ "cache-hit" if from_cache else "generated",
241
+ )
242
+
243
+ if output is not None:
244
+ reporter.emit("WRITE", f"format={format} path={output} items={len(dataset)}")
245
+ write_dataset(dataset, output, format)
246
+ reporter.done(len(dataset))
247
+ return dataset
248
+
249
+
250
+ def _auto_chunks(
251
+ source_text: str,
252
+ provider: LLMProvider,
253
+ *,
254
+ source_path: Path,
255
+ reporter: ProgressReporter,
256
+ model: str,
257
+ temperature: float,
258
+ max_input_tokens: int,
259
+ max_output_tokens: int,
260
+ auto_window_ratio: float,
261
+ strict: bool,
262
+ cache: bool,
263
+ cache_dir: str | Path,
264
+ refresh_cache: bool,
265
+ ) -> list[KnowledgeChunk]:
266
+ semantic_cache_dir = Path(cache_dir) / "semantic_segments"
267
+ key = cache_key(
268
+ "semantic-auto-v3",
269
+ source_text,
270
+ model,
271
+ temperature,
272
+ max_input_tokens,
273
+ max_output_tokens,
274
+ auto_window_ratio,
275
+ )
276
+ if cache and not refresh_cache:
277
+ cached = read_cache(semantic_cache_dir, key)
278
+ reporter.emit(
279
+ "CACHE",
280
+ f"semantic source={source_path} {'hit' if cached is not None else 'miss'}",
281
+ )
282
+ if cached is not None:
283
+ try:
284
+ return deserialize_chunks(cached, source_text)
285
+ except ChunkingError as error:
286
+ reporter.emit("WARN", f"invalid semantic cache source={source_path} error_type={type(error).__name__}")
287
+ if strict:
288
+ raise
289
+
290
+ last_error: ChunkingError | ProviderError | None = None
291
+ for attempt in range(1, 4):
292
+ try:
293
+ chunks = semantic_chunk_text(
294
+ source_text,
295
+ provider,
296
+ model=model,
297
+ temperature=temperature,
298
+ max_output_tokens=max_output_tokens,
299
+ auto_window_ratio=auto_window_ratio,
300
+ window_progress=lambda current, total: reporter.emit(
301
+ "WINDOW",
302
+ f"source={source_path} {current}/{total}",
303
+ ),
304
+ )
305
+ if cache:
306
+ write_cache(semantic_cache_dir, key, serialize_chunks(chunks))
307
+ reporter.emit("CACHE", f"semantic source={source_path} saved")
308
+ return chunks
309
+ except (ChunkingError, ProviderError) as error:
310
+ last_error = error
311
+ reporter.emit(
312
+ "RETRY",
313
+ f"semantic source={source_path} attempt={attempt}/3 error_type={type(error).__name__}",
314
+ )
315
+ assert last_error is not None
316
+ if strict:
317
+ raise last_error
318
+ reporter.emit("WARN", f"semantic source={source_path} using character fallback")
319
+ return fallback_chunks(source_text)
320
+
321
+
322
+ def _prompt_source(prompt: str | None, prompt_file: str | Path | None) -> str:
323
+ if prompt_file is not None:
324
+ return f"file:{prompt_file}"
325
+ if prompt is None:
326
+ return "default"
327
+ if prompt == "ambiguity":
328
+ return "built-in"
329
+ return "inline-custom"
330
+
331
+
332
+ def _knowledge_paths(knowledge: KnowledgeInput) -> list[str]:
333
+ if isinstance(knowledge, (str, Path)):
334
+ return [str(knowledge)]
335
+ return [str(path) for path in knowledge]
lladar/cache.py ADDED
@@ -0,0 +1,28 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ from pathlib import Path
6
+ from typing import Any
7
+
8
+
9
+ def cache_key(*parts: object) -> str:
10
+ payload = json.dumps(parts, ensure_ascii=False, sort_keys=True, default=str)
11
+ return hashlib.sha256(payload.encode("utf-8")).hexdigest()
12
+
13
+
14
+ def read_cache(directory: str | Path, key: str) -> dict[str, Any] | None:
15
+ path = Path(directory) / f"{key}.json"
16
+ if not path.exists():
17
+ return None
18
+ value = json.loads(path.read_text(encoding="utf-8"))
19
+ return value if isinstance(value, dict) else None
20
+
21
+
22
+ def write_cache(directory: str | Path, key: str, value: dict[str, Any]) -> None:
23
+ path = Path(directory) / f"{key}.json"
24
+ path.parent.mkdir(parents=True, exist_ok=True)
25
+ path.write_text(
26
+ json.dumps(value, ensure_ascii=False, indent=2) + "\n",
27
+ encoding="utf-8",
28
+ )