xtremeparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- xtremeparse/__init__.py +34 -0
- xtremeparse/chunking.py +83 -0
- xtremeparse/contracts.py +117 -0
- xtremeparse/corrections.py +166 -0
- xtremeparse/evalkit.py +43 -0
- xtremeparse/evals.py +97 -0
- xtremeparse/executor.py +208 -0
- xtremeparse/extractor.py +114 -0
- xtremeparse/judging.py +58 -0
- xtremeparse/merge.py +28 -0
- xtremeparse/paths.py +38 -0
- xtremeparse/prompting.py +60 -0
- xtremeparse/router.py +710 -0
- xtremeparse/scheduling.py +57 -0
- xtremeparse/testing.py +34 -0
- xtremeparse/units.py +90 -0
- xtremeparse-0.1.0.dist-info/METADATA +119 -0
- xtremeparse-0.1.0.dist-info/RECORD +20 -0
- xtremeparse-0.1.0.dist-info/WHEEL +4 -0
- xtremeparse-0.1.0.dist-info/licenses/LICENSE +21 -0
xtremeparse/__init__.py
ADDED
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""XtremeParse: extreme-concurrency structured extraction — chunk, route, fan out, self-correct."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
from xtremeparse.contracts import (
|
|
6
|
+
AgentResult,
|
|
7
|
+
AgentRunner,
|
|
8
|
+
ExtractionResult,
|
|
9
|
+
Issue,
|
|
10
|
+
JSONSchema,
|
|
11
|
+
Trace,
|
|
12
|
+
Validator,
|
|
13
|
+
)
|
|
14
|
+
from xtremeparse.extractor import Extractor
|
|
15
|
+
from xtremeparse.scheduling import TaskScheduler, report_tokens, scheduler_from_spec
|
|
16
|
+
|
|
17
|
+
try:
|
|
18
|
+
__version__ = version('xtremeparse')
|
|
19
|
+
except PackageNotFoundError:
|
|
20
|
+
__version__ = '0.1.0'
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
'Extractor',
|
|
24
|
+
'AgentResult',
|
|
25
|
+
'AgentRunner',
|
|
26
|
+
'ExtractionResult',
|
|
27
|
+
'Issue',
|
|
28
|
+
'JSONSchema',
|
|
29
|
+
'Trace',
|
|
30
|
+
'Validator',
|
|
31
|
+
'TaskScheduler',
|
|
32
|
+
'report_tokens',
|
|
33
|
+
'scheduler_from_spec',
|
|
34
|
+
]
|
xtremeparse/chunking.py
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Deterministic three-tier chunker: markdown structure, sentence
|
|
2
|
+
punctuation, character windows.
|
|
3
|
+
|
|
4
|
+
The same input always yields the same chunks, every chunk is an exact
|
|
5
|
+
substring of the (newline-normalized) input in reading order, and any
|
|
6
|
+
input — including a single line without punctuation — yields chunks.
|
|
7
|
+
Chunk ids are list positions.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
|
|
14
|
+
_HEADING = re.compile(r'^#{1,6}\s')
|
|
15
|
+
_LIST_ITEM = re.compile(r'^\s*(?:[-*+]+\s+|•\s*|\d+[、)]\s*|\d+\.\s+)')
|
|
16
|
+
_TABLE_ROW = re.compile(r'^\s*\|')
|
|
17
|
+
# Sentence enders, CJK and latin; '.' only when not followed by a digit so
|
|
18
|
+
# dates like 2015.3 never split.
|
|
19
|
+
_SENTENCE_END = re.compile(r'(?<=[。.!?;!?;])|(?<=\.)(?!\d)')
|
|
20
|
+
|
|
21
|
+
MAX_CHARS = 250
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def normalize_newlines(text: str) -> str:
|
|
25
|
+
"""Owned at extraction entry: chunks and the shared prompt prefix
|
|
26
|
+
must agree on newlines."""
|
|
27
|
+
return text.replace('\r\n', '\n').replace('\r', '\n')
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def chunk_text(text: str, *, max_chars: int = MAX_CHARS) -> list[str]:
|
|
31
|
+
"""Split text into chunks, numbered by position.
|
|
32
|
+
|
|
33
|
+
Tier 1 splits on structure: headings open blocks that absorb the plain
|
|
34
|
+
lines after them, list items and table rows stand alone, blank lines
|
|
35
|
+
end paragraphs. Tier 2 sentence-splits oversized blocks and greedily
|
|
36
|
+
packs fragments up to ``max_chars``. Tier 3 hard-slices whatever still
|
|
37
|
+
exceeds the ceiling (dense text without punctuation). Every chunk is
|
|
38
|
+
at most ``max_chars`` long.
|
|
39
|
+
"""
|
|
40
|
+
if not (text := normalize_newlines(text)).strip():
|
|
41
|
+
return []
|
|
42
|
+
return [c for block in _blocks(text) for c in _pack(_SENTENCE_END.split(block), max_chars)]
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _blocks(text: str) -> list[str]:
|
|
46
|
+
blocks, para = [], []
|
|
47
|
+
|
|
48
|
+
def flush():
|
|
49
|
+
if para:
|
|
50
|
+
blocks.append('\n'.join(para))
|
|
51
|
+
para.clear()
|
|
52
|
+
|
|
53
|
+
for line in text.split('\n'):
|
|
54
|
+
if _HEADING.match(line):
|
|
55
|
+
flush()
|
|
56
|
+
para.append(line)
|
|
57
|
+
elif not line.strip():
|
|
58
|
+
flush()
|
|
59
|
+
elif _LIST_ITEM.match(line) or _TABLE_ROW.match(line):
|
|
60
|
+
flush()
|
|
61
|
+
blocks.append(line)
|
|
62
|
+
else:
|
|
63
|
+
para.append(line)
|
|
64
|
+
flush()
|
|
65
|
+
return blocks
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _pack(pieces: list[str], max_chars: int) -> list[str]:
|
|
69
|
+
chunks, current = [], ''
|
|
70
|
+
for piece in pieces:
|
|
71
|
+
if len(piece) > max_chars:
|
|
72
|
+
if current:
|
|
73
|
+
chunks.append(current)
|
|
74
|
+
current = ''
|
|
75
|
+
chunks.extend(piece[i:i + max_chars] for i in range(0, len(piece), max_chars))
|
|
76
|
+
elif len(current) + len(piece) > max_chars:
|
|
77
|
+
chunks.append(current)
|
|
78
|
+
current = piece
|
|
79
|
+
else:
|
|
80
|
+
current += piece
|
|
81
|
+
if current:
|
|
82
|
+
chunks.append(current)
|
|
83
|
+
return chunks
|
xtremeparse/contracts.py
ADDED
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""Neutral contracts between xtremeparse and the host application.
|
|
2
|
+
|
|
3
|
+
The library never imports a schema vendor, an agent framework, or a
|
|
4
|
+
validator. Everything crosses the border as plain data: JSON Schema in,
|
|
5
|
+
JSON-shaped dict out, issues and agent calls behind the protocols below.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Callable, Optional, Protocol, Sequence, runtime_checkable
|
|
12
|
+
|
|
13
|
+
# A plain JSON Schema mapping (``schema.to_json_schema()`` output, or any
|
|
14
|
+
# hand-written schema with the same shape).
|
|
15
|
+
JSONSchema = dict[str, Any]
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@runtime_checkable
|
|
19
|
+
class Issue(Protocol):
|
|
20
|
+
"""A validation problem found in extracted data.
|
|
21
|
+
|
|
22
|
+
Structural protocol: any object with these attributes satisfies it,
|
|
23
|
+
zero adapter code. ``expected``/``got`` are optional free-form context
|
|
24
|
+
(hosts pass lists for enums, arbitrary values, sentinels for absent).
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
path: str
|
|
28
|
+
message: str
|
|
29
|
+
code: str
|
|
30
|
+
expected: object
|
|
31
|
+
got: object
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# Validates complete extracted data, returns the issues the host considers
|
|
35
|
+
# error-level. Feeding only errors is the host's severity policy: the
|
|
36
|
+
# correction loop retries whatever it receives.
|
|
37
|
+
Validator = Callable[[dict], Sequence[Issue]]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass
|
|
41
|
+
class AgentResult:
|
|
42
|
+
"""One agent turn's outcome.
|
|
43
|
+
|
|
44
|
+
``data`` is already coerced to the requested ``result_schema`` by the
|
|
45
|
+
adapter (e.g. PydanticAI structured output). ``history`` is the opaque
|
|
46
|
+
message history for multi-turn continuation; the library passes it back
|
|
47
|
+
verbatim on correction rounds. Usage/cost telemetry is deliberately not
|
|
48
|
+
here — the adapter layer instruments its own agent calls.
|
|
49
|
+
"""
|
|
50
|
+
|
|
51
|
+
data: Any
|
|
52
|
+
history: list
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
@runtime_checkable
|
|
56
|
+
class AgentRunner(Protocol):
|
|
57
|
+
"""The single agent-call abstraction the library owns.
|
|
58
|
+
|
|
59
|
+
The host adapts this to its agent framework (PydanticAI today, anything
|
|
60
|
+
else tomorrow). Prompt layout is frozen for cache reasons:
|
|
61
|
+
``instructions`` carries the unit's semantic card and ``content`` the
|
|
62
|
+
shared, byte-identical payload (full text + full schema) that makes the
|
|
63
|
+
provider's KV cache hit across every call of one extraction;
|
|
64
|
+
``scope`` carries this call's private slice — the routed chunks of one
|
|
65
|
+
group (None for the router). On ``history`` rounds the transcript
|
|
66
|
+
already carries the card and scope, so the caller passes ``''`` for
|
|
67
|
+
both — re-sending them would re-bill the same text, and only the
|
|
68
|
+
per-round ``feedback`` is fresh. ``result_schema`` is the JSON Schema the
|
|
69
|
+
returned ``data`` must satisfy; ``tools``/``history``/``feedback``
|
|
70
|
+
serve the multi-turn correction loop.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
async def run(
|
|
74
|
+
self,
|
|
75
|
+
*,
|
|
76
|
+
instructions: str,
|
|
77
|
+
result_schema: JSONSchema,
|
|
78
|
+
content: str,
|
|
79
|
+
scope: Optional[str] = None,
|
|
80
|
+
tools: Optional[list] = None,
|
|
81
|
+
history: Optional[list] = None,
|
|
82
|
+
feedback: Optional[Sequence[Issue]] = None,
|
|
83
|
+
) -> AgentResult:
|
|
84
|
+
...
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
@dataclass
|
|
88
|
+
class Trace:
|
|
89
|
+
"""Orchestration record: what was chunked, routed, run and retried.
|
|
90
|
+
|
|
91
|
+
Populated by the pipeline stages; read by eval suites and telemetry.
|
|
92
|
+
Token/latency accounting of agent calls stays with the adapter — this
|
|
93
|
+
trace records orchestration events only. ``prompts`` marks each
|
|
94
|
+
prompt slot 'default' or with a short hash of the host's override
|
|
95
|
+
template (see prompting.provenance) — regressions stay attributable
|
|
96
|
+
to whose prompt produced them.
|
|
97
|
+
"""
|
|
98
|
+
|
|
99
|
+
chunks: list = field(default_factory=list)
|
|
100
|
+
router: Optional[dict] = None
|
|
101
|
+
groups: list = field(default_factory=list)
|
|
102
|
+
corrections: list = field(default_factory=list)
|
|
103
|
+
prompts: dict = field(default_factory=dict)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@dataclass
|
|
107
|
+
class ExtractionResult:
|
|
108
|
+
"""Extraction never raises on bad data — it reports.
|
|
109
|
+
|
|
110
|
+
``data`` is the best-effort merged result (missing pieces stay absent),
|
|
111
|
+
``issues`` the unresolved error-level issues from the last validation,
|
|
112
|
+
both lenient by design. Strictness is the caller's policy.
|
|
113
|
+
"""
|
|
114
|
+
|
|
115
|
+
data: dict
|
|
116
|
+
issues: list
|
|
117
|
+
trace: Trace
|
|
@@ -0,0 +1,166 @@
|
|
|
1
|
+
"""Correction loop: whole-data validation, issue routing, bounded re-runs.
|
|
2
|
+
|
|
3
|
+
The injected validator defines severity — whatever it feeds is retried.
|
|
4
|
+
Issues route back to the specialist call that owns their path (per-item
|
|
5
|
+
calls only when the item index is derivable), and re-runs carry the call's
|
|
6
|
+
conversation history plus the routed feedback. Stops on clean, budget
|
|
7
|
+
(default 2 rounds), or no progress (issue paths identical to the previous
|
|
8
|
+
round). Never raises on bad data; unresolved issues return with the data.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import asyncio
|
|
14
|
+
import re
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from typing import Optional
|
|
17
|
+
|
|
18
|
+
from xtremeparse.contracts import AgentRunner, Validator
|
|
19
|
+
from xtremeparse.paths import resolve, resolve_list
|
|
20
|
+
from xtremeparse.executor import Call, Execution, dispatch_specialist, values_from_calls
|
|
21
|
+
from xtremeparse.merge import merge
|
|
22
|
+
from xtremeparse.prompting import json_len
|
|
23
|
+
from xtremeparse.scheduling import TaskScheduler
|
|
24
|
+
from xtremeparse.units import MISC
|
|
25
|
+
|
|
26
|
+
MAX_ROUNDS = 2
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Round:
|
|
31
|
+
"""One correction round's trace material."""
|
|
32
|
+
|
|
33
|
+
unit_path: str
|
|
34
|
+
item: Optional[int]
|
|
35
|
+
issue_paths: list
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
async def correct(runner: AgentRunner, execution: Execution, *, validator: Validator,
|
|
39
|
+
payload: str, scheduler: TaskScheduler,
|
|
40
|
+
max_rounds: int = MAX_ROUNDS,
|
|
41
|
+
specialist_instructions: str = None) -> tuple:
|
|
42
|
+
"""Re-run failing calls with feedback until clean, budgeted, or
|
|
43
|
+
stuck. Returns ``(data, issues, rounds)`` — always lenient. Call
|
|
44
|
+
results mutate in place; re-merge from ``execution.calls`` rather
|
|
45
|
+
than the now-stale ``execution.values``. ``specialist_instructions``
|
|
46
|
+
must be the same override the first round ran with — a correction
|
|
47
|
+
round continues that call's conversation history."""
|
|
48
|
+
calls = execution.calls
|
|
49
|
+
data = merge(values_from_calls(calls))
|
|
50
|
+
issues = list(validator(data) or [])
|
|
51
|
+
rounds, seen = [], None
|
|
52
|
+
for _ in range(max_rounds):
|
|
53
|
+
paths = {i.path for i in issues}
|
|
54
|
+
if not issues or paths == seen:
|
|
55
|
+
break
|
|
56
|
+
seen = paths
|
|
57
|
+
routed = _route(calls, issues)
|
|
58
|
+
if not routed:
|
|
59
|
+
break
|
|
60
|
+
rounds.extend(Round(call.unit.path, call.item, [i.path for i in feedback])
|
|
61
|
+
for call, feedback in routed)
|
|
62
|
+
tasks = [await dispatch_specialist(
|
|
63
|
+
runner, call.unit, payload=payload, scope=call.scope,
|
|
64
|
+
whole=call.strategy == 'whole', scheduler=scheduler,
|
|
65
|
+
specialist_instructions=specialist_instructions,
|
|
66
|
+
history=call.result.history if call.result else None,
|
|
67
|
+
feedback=feedback)
|
|
68
|
+
for call, feedback in routed]
|
|
69
|
+
for (call, _), result in zip(routed, await asyncio.gather(*tasks)):
|
|
70
|
+
call.result = result
|
|
71
|
+
data = merge(values_from_calls(calls))
|
|
72
|
+
issues = list(validator(data) or [])
|
|
73
|
+
return data, issues, rounds
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _route(calls: list, issues: list) -> list:
|
|
77
|
+
"""Map issues to owning calls. Longest unit path wins; per-item calls
|
|
78
|
+
match only their item; $misc is the fallback for unmatched paths."""
|
|
79
|
+
grouped = {}
|
|
80
|
+
for issue in issues:
|
|
81
|
+
if call := _owner(calls, issue.path):
|
|
82
|
+
grouped.setdefault(id(call), (call, []))[1].append(issue)
|
|
83
|
+
return list(grouped.values())
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _owner(calls: list, path: str):
|
|
87
|
+
covering = [c for c in calls if c.unit.path != MISC and _under(path, c.unit.path)]
|
|
88
|
+
exact = [c for c in covering if c.item is None and not c.batch
|
|
89
|
+
or c.item is not None and _under(path, f'{c.unit.path}[{c.item}]')
|
|
90
|
+
or c.batch and (_item_index(path, c.unit.path) or -1) in c.batch]
|
|
91
|
+
pool = exact or [c for c in calls if c.unit.path == MISC]
|
|
92
|
+
return max(pool, key=lambda c: len(c.unit.path), default=None)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _item_index(path: str, unit_path: str):
|
|
96
|
+
m = re.fullmatch(rf'{re.escape(unit_path)}\[(\d+)\](\..*)?', path)
|
|
97
|
+
return int(m.group(1)) if m else None
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _under(path: str, root: str) -> bool:
|
|
101
|
+
"""``path`` is ``root`` itself or a proper child of it."""
|
|
102
|
+
return path == root or path.startswith(f'{root}.') or path.startswith(f'{root}[')
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def item_chars(budgets: dict, data: dict) -> dict:
|
|
106
|
+
"""Per budgeted path: ``[(key, arranged, chars)]`` — one entry per
|
|
107
|
+
item for lists, one for the whole value otherwise, each carrying
|
|
108
|
+
its own arranged budget (a lone number covers every item; a short
|
|
109
|
+
list's last value covers items beyond it). The counting surface for
|
|
110
|
+
budget reads (trace, eval audits); the unit is the compact
|
|
111
|
+
serialized item JSON (keys and punctuation included) — exactly what
|
|
112
|
+
a specialist types and pays decode for. Overruns are ACCEPTED,
|
|
113
|
+
never retried: a retry costs a full extra decode, the very thing
|
|
114
|
+
budgets exist to save — budgets shape batch scheduling only."""
|
|
115
|
+
out = {}
|
|
116
|
+
for path, arranged in budgets.items():
|
|
117
|
+
if (value := resolve(data, path)) is None:
|
|
118
|
+
continue
|
|
119
|
+
values = arranged if isinstance(arranged, list) else [arranged]
|
|
120
|
+
entries = []
|
|
121
|
+
for i, item in (enumerate(value) if isinstance(value, list)
|
|
122
|
+
else [(None, value)]):
|
|
123
|
+
budget = values[i] if i is not None and i < len(values) else values[-1]
|
|
124
|
+
entries.append((f'{path}[{i}]' if i is not None else path,
|
|
125
|
+
budget, json_len(item)))
|
|
126
|
+
out[path] = entries
|
|
127
|
+
return out
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
@dataclass
|
|
131
|
+
class _CountIssue:
|
|
132
|
+
"""A routed item count the merged data does not honour — typically a
|
|
133
|
+
whole-array call that collapsed instances into fewer entries."""
|
|
134
|
+
|
|
135
|
+
path: str
|
|
136
|
+
code: str = 'item_count'
|
|
137
|
+
message: str = ''
|
|
138
|
+
expected: int = None
|
|
139
|
+
got: int = None
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def count_issues(counts: dict, data: dict) -> list:
|
|
143
|
+
"""The router's declared counts (map-validated ground truth) against
|
|
144
|
+
the merged arrays. A short array means instances were collapsed or
|
|
145
|
+
dropped — silent to schema validation (nothing declares minItems).
|
|
146
|
+
Items concatenate in item-index order, so a short array is missing
|
|
147
|
+
its tail: each missing index becomes its own issue and routes to the
|
|
148
|
+
call that owns it (a batch member or a single). One that survives a
|
|
149
|
+
retry stops via the no-progress rule."""
|
|
150
|
+
issues = []
|
|
151
|
+
for path, declared in counts.items():
|
|
152
|
+
actual = len(resolve_list(data, path))
|
|
153
|
+
if actual < declared:
|
|
154
|
+
issues += [_CountIssue(
|
|
155
|
+
f'{path}[{i}]', expected=declared, got=actual,
|
|
156
|
+
message=f'{path} is missing item {i} of {declared} — the '
|
|
157
|
+
f'array holds {actual}; return every instance as '
|
|
158
|
+
'its own entry, without splitting or duplicating')
|
|
159
|
+
for i in range(actual, declared)]
|
|
160
|
+
elif actual > declared:
|
|
161
|
+
issues.append(_CountIssue(
|
|
162
|
+
path, expected=declared, got=actual,
|
|
163
|
+
message=f'declared {declared} items but the array holds '
|
|
164
|
+
f'{actual} — merge the duplicates'))
|
|
165
|
+
return issues
|
|
166
|
+
|
xtremeparse/evalkit.py
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Eval support: plain-data reads over the orchestration trace.
|
|
2
|
+
|
|
3
|
+
The lib keeps the reads its shipped evaluators (the ``evals`` extra)
|
|
4
|
+
need: the executor's calling conventions (per_item_budgets) and the
|
|
5
|
+
router's contract (router_overlap). Hosts own their general eval
|
|
6
|
+
tooling — digest projections, anchor checks, assertion DSLs are each
|
|
7
|
+
host's selection.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def per_item_budgets(groups: list) -> dict:
|
|
14
|
+
"""Trace groups → the per-path budget map ``item_chars`` reads: one
|
|
15
|
+
number or the item-index-ordered list."""
|
|
16
|
+
whole, by_item = {}, {}
|
|
17
|
+
for g in groups:
|
|
18
|
+
if not g.get('budget'):
|
|
19
|
+
continue
|
|
20
|
+
items = g.get('batch') or ([g['item']] if g['item'] is not None else None)
|
|
21
|
+
if items is None: # whole-array call: its list is document-ordered
|
|
22
|
+
whole[g['unit']] = g['budget']
|
|
23
|
+
continue
|
|
24
|
+
values = g['budget'] if isinstance(g['budget'], list) else None
|
|
25
|
+
for k, i in enumerate(items):
|
|
26
|
+
by_item.setdefault(g['unit'], {})[i] = values[k] if values else g['budget']
|
|
27
|
+
return whole | {unit: [items[i] for i in sorted(items)]
|
|
28
|
+
for unit, items in by_item.items()}
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def router_overlap(router, groups: list) -> tuple:
|
|
32
|
+
"""(overlap, items_lost) between the router's assignments and the
|
|
33
|
+
executed calls; only per-item executions are held to item presence
|
|
34
|
+
(whole-strategy coalescing is legitimate)."""
|
|
35
|
+
assignments = router['assignments'] if router else []
|
|
36
|
+
raw = sum(len(a['chunks']) for a in assignments)
|
|
37
|
+
present = {(g['unit'], i) for g in groups
|
|
38
|
+
for i in (g.get('batch') or [g['item']])}
|
|
39
|
+
per_item_units = {g['unit'] for g in groups if g['strategy'] == 'per-item'}
|
|
40
|
+
declared = {(a['unit'], a.get('item')) for a in assignments
|
|
41
|
+
if a['unit'] in per_item_units}
|
|
42
|
+
return (raw - sum(len(g['chunk_ids']) for g in groups),
|
|
43
|
+
len(declared - present))
|
xtremeparse/evals.py
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""pydantic-evals adapters over the trace reads — install the ``evals``
|
|
2
|
+
extra (``pip install xtremeparse[evals]``). The evaluators accept any
|
|
3
|
+
output carrying ``.trace`` and ``.data`` the way ``ExtractionResult``
|
|
4
|
+
does; hosts keep their domain evaluators beside these.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
|
|
11
|
+
from pydantic_evals.evaluators import Evaluator, EvaluatorContext, EvaluationReason
|
|
12
|
+
|
|
13
|
+
from xtremeparse.corrections import item_chars
|
|
14
|
+
from xtremeparse.evalkit import per_item_budgets, router_overlap
|
|
15
|
+
from xtremeparse.judging import JUDGE_PLACEHOLDERS, judge
|
|
16
|
+
from xtremeparse.prompting import check_placeholders
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class RouterOverlap(Evaluator):
|
|
20
|
+
"""Invariant tripwire: router validation makes overlap and phantom
|
|
21
|
+
items structurally impossible — a red here means the router contract
|
|
22
|
+
regressed, not a flaky model."""
|
|
23
|
+
|
|
24
|
+
def evaluate(self, ctx: EvaluatorContext):
|
|
25
|
+
overlap, lost = router_overlap(ctx.output.trace.router,
|
|
26
|
+
ctx.output.trace.groups)
|
|
27
|
+
return EvaluationReason(lost == 0, f'overlap={overlap}, items_lost={lost}')
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class BudgetFit(Evaluator):
|
|
31
|
+
"""Allocation audit: the model ARRANGES each budget — its estimate
|
|
32
|
+
of the characters an item's output JSON will run to, resolved by
|
|
33
|
+
code against the routed material — so actuals should land near
|
|
34
|
+
the arranged number. The lower edge is loose on purpose: an idle
|
|
35
|
+
estimate on a naturally short item is harmless (output runs at
|
|
36
|
+
natural size either way). The upper edge flags arrangements so
|
|
37
|
+
far off the router clearly wasn't counting. Per item: a mean
|
|
38
|
+
would hide spread."""
|
|
39
|
+
|
|
40
|
+
BAND = (0.4, 2.0)
|
|
41
|
+
|
|
42
|
+
def evaluate(self, ctx: EvaluatorContext):
|
|
43
|
+
ratios, wild = {}, []
|
|
44
|
+
budgets = per_item_budgets(ctx.output.trace.groups)
|
|
45
|
+
for path, entries in item_chars(budgets, ctx.output.data).items():
|
|
46
|
+
unit = path.rsplit('.', 1)[-1]
|
|
47
|
+
judged = [e for e in entries if e[1] and e[2]] # budgeted, non-empty
|
|
48
|
+
if rs := [round(chars / budget, 2) for _, budget, chars in judged]:
|
|
49
|
+
ratios[unit] = f'{min(rs)}' if len(rs) == 1 else f'{min(rs)}-{max(rs)}'
|
|
50
|
+
wild += [key.rsplit('.', 1)[-1] for key, budget, chars in judged
|
|
51
|
+
if not self.BAND[0] <= chars / budget <= self.BAND[1]]
|
|
52
|
+
return EvaluationReason(
|
|
53
|
+
not wild, f'act/budget {ratios or "no budgets declared"}'
|
|
54
|
+
+ (f', wild: {wild}' if wild else ''))
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
class RubricJudge(Evaluator):
|
|
58
|
+
"""One rubric call through an injected AgentRunner (see
|
|
59
|
+
xtremeparse.judging) — the judge runs on whatever framework the
|
|
60
|
+
host adapted, where pydantic-evals' own LLMJudge ties the harness
|
|
61
|
+
to a pydantic-ai model. ``source`` is the judged material's
|
|
62
|
+
document, held at construction; ``output_of`` projects the case
|
|
63
|
+
output into what the rubric asks about (e.g. a digest projection
|
|
64
|
+
for containment questions) — see docs/prompting.md for the
|
|
65
|
+
measured judge guidance. The judge's own model settings are
|
|
66
|
+
recommendations carried at the adapter."""
|
|
67
|
+
|
|
68
|
+
def __init__(self, runner, rubric: str, source: str = '', *,
|
|
69
|
+
output_of=None, instructions: str = None):
|
|
70
|
+
if instructions is not None:
|
|
71
|
+
check_placeholders(instructions, JUDGE_PLACEHOLDERS, 'instructions')
|
|
72
|
+
self.runner = runner
|
|
73
|
+
self.rubric = rubric
|
|
74
|
+
self.source = source
|
|
75
|
+
self.output_of = output_of
|
|
76
|
+
self.instructions = instructions
|
|
77
|
+
|
|
78
|
+
async def evaluate(self, ctx: EvaluatorContext):
|
|
79
|
+
output = _text(self.output_of(ctx.output) if self.output_of
|
|
80
|
+
else ctx.output)
|
|
81
|
+
verdict = await judge(self.runner, rubric=self.rubric,
|
|
82
|
+
source=self.source, output=output,
|
|
83
|
+
instructions=self.instructions)
|
|
84
|
+
return EvaluationReason(verdict.ok, verdict.reason)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _text(value) -> str:
|
|
88
|
+
"""Stable judge-facing text: an ExtractionResult-shaped value is
|
|
89
|
+
judged by its data; a plain string stays verbatim (JSON-escaping a
|
|
90
|
+
document-sized string would bury its structure); the rest
|
|
91
|
+
serializes compactly (``default=str`` absorbs harness objects like
|
|
92
|
+
Path)."""
|
|
93
|
+
if hasattr(value, 'data'):
|
|
94
|
+
value = value.data
|
|
95
|
+
if isinstance(value, str):
|
|
96
|
+
return value
|
|
97
|
+
return json.dumps(value, ensure_ascii=False, sort_keys=True, default=str)
|