lladar 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- lladar/__init__.py +25 -0
- lladar/api.py +335 -0
- lladar/cache.py +28 -0
- lladar/chunking.py +369 -0
- lladar/cli.py +288 -0
- lladar/evaluation.py +195 -0
- lladar/exceptions.py +25 -0
- lladar/loaders.py +26 -0
- lladar/model_profiles.py +50 -0
- lladar/output.py +23 -0
- lladar/progress.py +82 -0
- lladar/prompts.py +45 -0
- lladar/providers/__init__.py +4 -0
- lladar/providers/akasha.py +79 -0
- lladar/providers/base.py +13 -0
- lladar/validation.py +38 -0
- lladar-0.1.0.dist-info/METADATA +142 -0
- lladar-0.1.0.dist-info/RECORD +22 -0
- lladar-0.1.0.dist-info/WHEEL +5 -0
- lladar-0.1.0.dist-info/entry_points.txt +2 -0
- lladar-0.1.0.dist-info/licenses/LICENSE +25 -0
- lladar-0.1.0.dist-info/top_level.txt +1 -0
lladar/__init__.py
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
from .api import create_test_dataset
|
|
2
|
+
from .evaluation import evaluate
|
|
3
|
+
from .exceptions import (
|
|
4
|
+
ChunkingError,
|
|
5
|
+
DatasetValidationError,
|
|
6
|
+
EvaluationError,
|
|
7
|
+
GenerationError,
|
|
8
|
+
KnowledgeLoadError,
|
|
9
|
+
LladarError,
|
|
10
|
+
ProviderError,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
__all__ = [
|
|
14
|
+
"ChunkingError",
|
|
15
|
+
"DatasetValidationError",
|
|
16
|
+
"GenerationError",
|
|
17
|
+
"KnowledgeLoadError",
|
|
18
|
+
"LladarError",
|
|
19
|
+
"ProviderError",
|
|
20
|
+
"create_test_dataset",
|
|
21
|
+
"eval",
|
|
22
|
+
"evaluate",
|
|
23
|
+
]
|
|
24
|
+
|
|
25
|
+
eval = evaluate
|
lladar/api.py
ADDED
|
@@ -0,0 +1,335 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Any, Literal
|
|
6
|
+
|
|
7
|
+
from .cache import cache_key, read_cache, write_cache
|
|
8
|
+
from .chunking import (
|
|
9
|
+
KnowledgeChunk,
|
|
10
|
+
chunk_text,
|
|
11
|
+
deserialize_chunks,
|
|
12
|
+
fallback_chunks,
|
|
13
|
+
semantic_chunk_text,
|
|
14
|
+
serialize_chunks,
|
|
15
|
+
)
|
|
16
|
+
from .exceptions import ChunkingError, DatasetValidationError, ProviderError
|
|
17
|
+
from .loaders import KnowledgeInput, load_knowledge
|
|
18
|
+
from .model_profiles import resolve_model_profile
|
|
19
|
+
from .output import write_dataset
|
|
20
|
+
from .progress import ProgressReporter
|
|
21
|
+
from .prompts import build_generation_prompt, resolve_strategy
|
|
22
|
+
from .providers import AkashaProvider, LLMProvider
|
|
23
|
+
from .validation import validate_generated_pair
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
DEFAULT_MODEL = "gemini:gemini-2.5-flash"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def create_test_dataset(
|
|
30
|
+
knowledge: KnowledgeInput,
|
|
31
|
+
prompt: str | None = None,
|
|
32
|
+
chunk_size: int | Literal["auto"] = 2000,
|
|
33
|
+
overlap: float = 0.1,
|
|
34
|
+
num_pairs: int = 1,
|
|
35
|
+
model: str = DEFAULT_MODEL,
|
|
36
|
+
output: str | Path | None = None,
|
|
37
|
+
format: str = "jsonl",
|
|
38
|
+
*,
|
|
39
|
+
prompt_file: str | Path | None = None,
|
|
40
|
+
provider: LLMProvider | None = None,
|
|
41
|
+
env_file: str | Path = ".env",
|
|
42
|
+
temperature: float = 0.0,
|
|
43
|
+
max_input_tokens: int | None = None,
|
|
44
|
+
max_output_tokens: int | None = None,
|
|
45
|
+
auto_window_ratio: float | None = None,
|
|
46
|
+
strict: bool = False,
|
|
47
|
+
force: bool = False,
|
|
48
|
+
cache: bool = False,
|
|
49
|
+
cache_dir: str | Path = ".lladar/cache",
|
|
50
|
+
refresh_cache: bool = False,
|
|
51
|
+
verbose: bool = True,
|
|
52
|
+
) -> list[dict[str, Any]]:
|
|
53
|
+
if output is not None and Path(output).exists() and not force:
|
|
54
|
+
raise FileExistsError(f"output already exists: {output}")
|
|
55
|
+
if prompt is not None and prompt_file is not None:
|
|
56
|
+
raise ValueError("prompt and prompt_file cannot be used together")
|
|
57
|
+
prompt_source = _prompt_source(prompt, prompt_file)
|
|
58
|
+
if prompt_file is not None:
|
|
59
|
+
prompt = Path(prompt_file).read_text(encoding="utf-8")
|
|
60
|
+
if num_pairs <= 0:
|
|
61
|
+
raise ValueError("num_pairs must be greater than 0")
|
|
62
|
+
|
|
63
|
+
profile = resolve_model_profile(
|
|
64
|
+
model,
|
|
65
|
+
max_input_tokens=max_input_tokens,
|
|
66
|
+
max_output_tokens=max_output_tokens,
|
|
67
|
+
auto_window_ratio=auto_window_ratio,
|
|
68
|
+
)
|
|
69
|
+
strategy, strategy_text = resolve_strategy(prompt)
|
|
70
|
+
active_provider = provider or AkashaProvider(
|
|
71
|
+
env_file=str(env_file),
|
|
72
|
+
max_input_tokens=profile.max_input_tokens,
|
|
73
|
+
max_output_tokens=profile.max_output_tokens,
|
|
74
|
+
)
|
|
75
|
+
reporter = ProgressReporter(verbose)
|
|
76
|
+
reporter.configuration(
|
|
77
|
+
{
|
|
78
|
+
"knowledge": _knowledge_paths(knowledge),
|
|
79
|
+
"strategy": strategy if strategy == "ambiguity" else "custom",
|
|
80
|
+
"prompt_source": prompt_source,
|
|
81
|
+
"chunk_size": chunk_size,
|
|
82
|
+
"overlap": overlap,
|
|
83
|
+
"num_pairs": num_pairs,
|
|
84
|
+
"model": model,
|
|
85
|
+
"output": output,
|
|
86
|
+
"format": format,
|
|
87
|
+
"env_file": env_file,
|
|
88
|
+
"temperature": temperature,
|
|
89
|
+
"max_input_tokens": profile.max_input_tokens,
|
|
90
|
+
"max_output_tokens": profile.max_output_tokens,
|
|
91
|
+
"auto_window_ratio": profile.auto_window_ratio,
|
|
92
|
+
"strict": strict,
|
|
93
|
+
"force": force,
|
|
94
|
+
"cache": cache,
|
|
95
|
+
"cache_dir": cache_dir,
|
|
96
|
+
"refresh_cache": refresh_cache,
|
|
97
|
+
"verbose": verbose,
|
|
98
|
+
"provider": type(active_provider).__name__,
|
|
99
|
+
}
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
sources = load_knowledge(knowledge)
|
|
103
|
+
reporter.emit("SOURCE", f"loaded={len(sources)}")
|
|
104
|
+
prepared: list[tuple[Path, int, KnowledgeChunk]] = []
|
|
105
|
+
for source_index, (path, source_text) in enumerate(sources, start=1):
|
|
106
|
+
reporter.emit(
|
|
107
|
+
"SOURCE",
|
|
108
|
+
f"{source_index}/{len(sources)} path={path} characters={len(source_text)}",
|
|
109
|
+
)
|
|
110
|
+
if chunk_size == "auto":
|
|
111
|
+
chunks = _auto_chunks(
|
|
112
|
+
source_text,
|
|
113
|
+
active_provider,
|
|
114
|
+
source_path=path,
|
|
115
|
+
reporter=reporter,
|
|
116
|
+
model=model,
|
|
117
|
+
temperature=temperature,
|
|
118
|
+
max_input_tokens=profile.max_input_tokens,
|
|
119
|
+
max_output_tokens=profile.max_output_tokens,
|
|
120
|
+
auto_window_ratio=profile.auto_window_ratio,
|
|
121
|
+
strict=strict,
|
|
122
|
+
cache=cache,
|
|
123
|
+
cache_dir=cache_dir,
|
|
124
|
+
refresh_cache=refresh_cache,
|
|
125
|
+
)
|
|
126
|
+
elif isinstance(chunk_size, int):
|
|
127
|
+
chunks = [
|
|
128
|
+
KnowledgeChunk(chunk, -1, -1, (), "character")
|
|
129
|
+
for chunk in chunk_text(source_text, chunk_size, overlap)
|
|
130
|
+
]
|
|
131
|
+
else:
|
|
132
|
+
raise ValueError('chunk_size must be a positive integer or "auto"')
|
|
133
|
+
reporter.emit(
|
|
134
|
+
"CHUNK",
|
|
135
|
+
f"source={path} method={chunks[0].method if chunks else chunk_size} count={len(chunks)}",
|
|
136
|
+
)
|
|
137
|
+
prepared.extend(
|
|
138
|
+
(path, chunk_index, knowledge_chunk)
|
|
139
|
+
for chunk_index, knowledge_chunk in enumerate(chunks)
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
total_pairs = len(prepared) * num_pairs
|
|
143
|
+
reporter.emit("PAIR", f"planned={total_pairs} chunks={len(prepared)}")
|
|
144
|
+
dataset: list[dict[str, Any]] = []
|
|
145
|
+
completed = 0
|
|
146
|
+
for path, chunk_index, knowledge_chunk in prepared:
|
|
147
|
+
chunk = knowledge_chunk.text
|
|
148
|
+
for pair_index in range(num_pairs):
|
|
149
|
+
key = cache_key(
|
|
150
|
+
chunk,
|
|
151
|
+
knowledge_chunk.method,
|
|
152
|
+
knowledge_chunk.source_start,
|
|
153
|
+
knowledge_chunk.source_end,
|
|
154
|
+
strategy,
|
|
155
|
+
model,
|
|
156
|
+
temperature,
|
|
157
|
+
profile.max_input_tokens,
|
|
158
|
+
profile.max_output_tokens,
|
|
159
|
+
profile.auto_window_ratio,
|
|
160
|
+
pair_index,
|
|
161
|
+
)
|
|
162
|
+
generated = None
|
|
163
|
+
from_cache = False
|
|
164
|
+
if cache and not refresh_cache:
|
|
165
|
+
generated = read_cache(cache_dir, key)
|
|
166
|
+
from_cache = generated is not None
|
|
167
|
+
reporter.emit(
|
|
168
|
+
"CACHE",
|
|
169
|
+
f"pair={completed + 1}/{total_pairs} {'hit' if from_cache else 'miss'}",
|
|
170
|
+
)
|
|
171
|
+
if generated is None:
|
|
172
|
+
last_error: DatasetValidationError | ProviderError | None = None
|
|
173
|
+
for attempt in range(1, 4):
|
|
174
|
+
try:
|
|
175
|
+
candidate = active_provider.generate_structured(
|
|
176
|
+
build_generation_prompt(chunk, strategy_text),
|
|
177
|
+
model=model,
|
|
178
|
+
temperature=temperature,
|
|
179
|
+
)
|
|
180
|
+
generated = validate_generated_pair(candidate)
|
|
181
|
+
break
|
|
182
|
+
except (DatasetValidationError, ProviderError) as error:
|
|
183
|
+
last_error = error
|
|
184
|
+
reporter.emit(
|
|
185
|
+
"RETRY",
|
|
186
|
+
f"pair={completed + 1}/{total_pairs} attempt={attempt}/3 error_type={type(error).__name__}",
|
|
187
|
+
)
|
|
188
|
+
else:
|
|
189
|
+
assert last_error is not None
|
|
190
|
+
completed += 1
|
|
191
|
+
if strict:
|
|
192
|
+
raise last_error
|
|
193
|
+
reporter.emit(
|
|
194
|
+
"WARN",
|
|
195
|
+
f"pair={completed}/{total_pairs} skipped after 3 failed attempts",
|
|
196
|
+
)
|
|
197
|
+
reporter.pair(completed, total_pairs, "skipped")
|
|
198
|
+
continue
|
|
199
|
+
if cache:
|
|
200
|
+
write_cache(cache_dir, key, generated)
|
|
201
|
+
reporter.emit("CACHE", f"pair={completed + 1}/{total_pairs} saved")
|
|
202
|
+
else:
|
|
203
|
+
generated = validate_generated_pair(generated)
|
|
204
|
+
|
|
205
|
+
item_id = hashlib.sha256(
|
|
206
|
+
f"{path.resolve()}\0{chunk}\0{chunk_index}\0{pair_index}".encode(
|
|
207
|
+
"utf-8"
|
|
208
|
+
)
|
|
209
|
+
).hexdigest()[:24]
|
|
210
|
+
metadata: dict[str, Any] = {
|
|
211
|
+
"strategy": strategy,
|
|
212
|
+
"model": model,
|
|
213
|
+
"temperature": temperature,
|
|
214
|
+
}
|
|
215
|
+
if knowledge_chunk.method != "character":
|
|
216
|
+
metadata.update(
|
|
217
|
+
{
|
|
218
|
+
"chunk_method": knowledge_chunk.method,
|
|
219
|
+
"source_start": knowledge_chunk.source_start,
|
|
220
|
+
"source_end": knowledge_chunk.source_end,
|
|
221
|
+
"knowledge_facts": list(knowledge_chunk.knowledge_facts),
|
|
222
|
+
}
|
|
223
|
+
)
|
|
224
|
+
dataset.append(
|
|
225
|
+
{
|
|
226
|
+
"schema_version": "1.0",
|
|
227
|
+
"id": item_id,
|
|
228
|
+
"source_file": str(path),
|
|
229
|
+
"chunk_index": chunk_index,
|
|
230
|
+
"source_text": chunk,
|
|
231
|
+
**generated,
|
|
232
|
+
"bias_type": "unsupported_assumption",
|
|
233
|
+
"metadata": metadata,
|
|
234
|
+
}
|
|
235
|
+
)
|
|
236
|
+
completed += 1
|
|
237
|
+
reporter.pair(
|
|
238
|
+
completed,
|
|
239
|
+
total_pairs,
|
|
240
|
+
"cache-hit" if from_cache else "generated",
|
|
241
|
+
)
|
|
242
|
+
|
|
243
|
+
if output is not None:
|
|
244
|
+
reporter.emit("WRITE", f"format={format} path={output} items={len(dataset)}")
|
|
245
|
+
write_dataset(dataset, output, format)
|
|
246
|
+
reporter.done(len(dataset))
|
|
247
|
+
return dataset
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _auto_chunks(
|
|
251
|
+
source_text: str,
|
|
252
|
+
provider: LLMProvider,
|
|
253
|
+
*,
|
|
254
|
+
source_path: Path,
|
|
255
|
+
reporter: ProgressReporter,
|
|
256
|
+
model: str,
|
|
257
|
+
temperature: float,
|
|
258
|
+
max_input_tokens: int,
|
|
259
|
+
max_output_tokens: int,
|
|
260
|
+
auto_window_ratio: float,
|
|
261
|
+
strict: bool,
|
|
262
|
+
cache: bool,
|
|
263
|
+
cache_dir: str | Path,
|
|
264
|
+
refresh_cache: bool,
|
|
265
|
+
) -> list[KnowledgeChunk]:
|
|
266
|
+
semantic_cache_dir = Path(cache_dir) / "semantic_segments"
|
|
267
|
+
key = cache_key(
|
|
268
|
+
"semantic-auto-v3",
|
|
269
|
+
source_text,
|
|
270
|
+
model,
|
|
271
|
+
temperature,
|
|
272
|
+
max_input_tokens,
|
|
273
|
+
max_output_tokens,
|
|
274
|
+
auto_window_ratio,
|
|
275
|
+
)
|
|
276
|
+
if cache and not refresh_cache:
|
|
277
|
+
cached = read_cache(semantic_cache_dir, key)
|
|
278
|
+
reporter.emit(
|
|
279
|
+
"CACHE",
|
|
280
|
+
f"semantic source={source_path} {'hit' if cached is not None else 'miss'}",
|
|
281
|
+
)
|
|
282
|
+
if cached is not None:
|
|
283
|
+
try:
|
|
284
|
+
return deserialize_chunks(cached, source_text)
|
|
285
|
+
except ChunkingError as error:
|
|
286
|
+
reporter.emit("WARN", f"invalid semantic cache source={source_path} error_type={type(error).__name__}")
|
|
287
|
+
if strict:
|
|
288
|
+
raise
|
|
289
|
+
|
|
290
|
+
last_error: ChunkingError | ProviderError | None = None
|
|
291
|
+
for attempt in range(1, 4):
|
|
292
|
+
try:
|
|
293
|
+
chunks = semantic_chunk_text(
|
|
294
|
+
source_text,
|
|
295
|
+
provider,
|
|
296
|
+
model=model,
|
|
297
|
+
temperature=temperature,
|
|
298
|
+
max_output_tokens=max_output_tokens,
|
|
299
|
+
auto_window_ratio=auto_window_ratio,
|
|
300
|
+
window_progress=lambda current, total: reporter.emit(
|
|
301
|
+
"WINDOW",
|
|
302
|
+
f"source={source_path} {current}/{total}",
|
|
303
|
+
),
|
|
304
|
+
)
|
|
305
|
+
if cache:
|
|
306
|
+
write_cache(semantic_cache_dir, key, serialize_chunks(chunks))
|
|
307
|
+
reporter.emit("CACHE", f"semantic source={source_path} saved")
|
|
308
|
+
return chunks
|
|
309
|
+
except (ChunkingError, ProviderError) as error:
|
|
310
|
+
last_error = error
|
|
311
|
+
reporter.emit(
|
|
312
|
+
"RETRY",
|
|
313
|
+
f"semantic source={source_path} attempt={attempt}/3 error_type={type(error).__name__}",
|
|
314
|
+
)
|
|
315
|
+
assert last_error is not None
|
|
316
|
+
if strict:
|
|
317
|
+
raise last_error
|
|
318
|
+
reporter.emit("WARN", f"semantic source={source_path} using character fallback")
|
|
319
|
+
return fallback_chunks(source_text)
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def _prompt_source(prompt: str | None, prompt_file: str | Path | None) -> str:
|
|
323
|
+
if prompt_file is not None:
|
|
324
|
+
return f"file:{prompt_file}"
|
|
325
|
+
if prompt is None:
|
|
326
|
+
return "default"
|
|
327
|
+
if prompt == "ambiguity":
|
|
328
|
+
return "built-in"
|
|
329
|
+
return "inline-custom"
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
def _knowledge_paths(knowledge: KnowledgeInput) -> list[str]:
|
|
333
|
+
if isinstance(knowledge, (str, Path)):
|
|
334
|
+
return [str(knowledge)]
|
|
335
|
+
return [str(path) for path in knowledge]
|
lladar/cache.py
ADDED
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def cache_key(*parts: object) -> str:
|
|
10
|
+
payload = json.dumps(parts, ensure_ascii=False, sort_keys=True, default=str)
|
|
11
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def read_cache(directory: str | Path, key: str) -> dict[str, Any] | None:
|
|
15
|
+
path = Path(directory) / f"{key}.json"
|
|
16
|
+
if not path.exists():
|
|
17
|
+
return None
|
|
18
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
19
|
+
return value if isinstance(value, dict) else None
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def write_cache(directory: str | Path, key: str, value: dict[str, Any]) -> None:
|
|
23
|
+
path = Path(directory) / f"{key}.json"
|
|
24
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
25
|
+
path.write_text(
|
|
26
|
+
json.dumps(value, ensure_ascii=False, indent=2) + "\n",
|
|
27
|
+
encoding="utf-8",
|
|
28
|
+
)
|