xtremeparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,34 @@
1
+ """XtremeParse: extreme-concurrency structured extraction — chunk, route, fan out, self-correct."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ from xtremeparse.contracts import (
6
+ AgentResult,
7
+ AgentRunner,
8
+ ExtractionResult,
9
+ Issue,
10
+ JSONSchema,
11
+ Trace,
12
+ Validator,
13
+ )
14
+ from xtremeparse.extractor import Extractor
15
+ from xtremeparse.scheduling import TaskScheduler, report_tokens, scheduler_from_spec
16
+
17
+ try:
18
+ __version__ = version('xtremeparse')
19
+ except PackageNotFoundError:
20
+ __version__ = '0.1.0'
21
+
22
+ __all__ = [
23
+ 'Extractor',
24
+ 'AgentResult',
25
+ 'AgentRunner',
26
+ 'ExtractionResult',
27
+ 'Issue',
28
+ 'JSONSchema',
29
+ 'Trace',
30
+ 'Validator',
31
+ 'TaskScheduler',
32
+ 'report_tokens',
33
+ 'scheduler_from_spec',
34
+ ]
@@ -0,0 +1,83 @@
1
+ """Deterministic three-tier chunker: markdown structure, sentence
2
+ punctuation, character windows.
3
+
4
+ The same input always yields the same chunks, every chunk is an exact
5
+ substring of the (newline-normalized) input in reading order, and any
6
+ input — including a single line without punctuation — yields chunks.
7
+ Chunk ids are list positions.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import re
13
+
14
+ _HEADING = re.compile(r'^#{1,6}\s')
15
+ _LIST_ITEM = re.compile(r'^\s*(?:[-*+]+\s+|•\s*|\d+[、)]\s*|\d+\.\s+)')
16
+ _TABLE_ROW = re.compile(r'^\s*\|')
17
+ # Sentence enders, CJK and latin; '.' only when not followed by a digit so
18
+ # dates like 2015.3 never split.
19
+ _SENTENCE_END = re.compile(r'(?<=[。.!?;!?;])|(?<=\.)(?!\d)')
20
+
21
+ MAX_CHARS = 250
22
+
23
+
24
+ def normalize_newlines(text: str) -> str:
25
+ """Owned at extraction entry: chunks and the shared prompt prefix
26
+ must agree on newlines."""
27
+ return text.replace('\r\n', '\n').replace('\r', '\n')
28
+
29
+
30
+ def chunk_text(text: str, *, max_chars: int = MAX_CHARS) -> list[str]:
31
+ """Split text into chunks, numbered by position.
32
+
33
+ Tier 1 splits on structure: headings open blocks that absorb the plain
34
+ lines after them, list items and table rows stand alone, blank lines
35
+ end paragraphs. Tier 2 sentence-splits oversized blocks and greedily
36
+ packs fragments up to ``max_chars``. Tier 3 hard-slices whatever still
37
+ exceeds the ceiling (dense text without punctuation). Every chunk is
38
+ at most ``max_chars`` long.
39
+ """
40
+ if not (text := normalize_newlines(text)).strip():
41
+ return []
42
+ return [c for block in _blocks(text) for c in _pack(_SENTENCE_END.split(block), max_chars)]
43
+
44
+
45
+ def _blocks(text: str) -> list[str]:
46
+ blocks, para = [], []
47
+
48
+ def flush():
49
+ if para:
50
+ blocks.append('\n'.join(para))
51
+ para.clear()
52
+
53
+ for line in text.split('\n'):
54
+ if _HEADING.match(line):
55
+ flush()
56
+ para.append(line)
57
+ elif not line.strip():
58
+ flush()
59
+ elif _LIST_ITEM.match(line) or _TABLE_ROW.match(line):
60
+ flush()
61
+ blocks.append(line)
62
+ else:
63
+ para.append(line)
64
+ flush()
65
+ return blocks
66
+
67
+
68
+ def _pack(pieces: list[str], max_chars: int) -> list[str]:
69
+ chunks, current = [], ''
70
+ for piece in pieces:
71
+ if len(piece) > max_chars:
72
+ if current:
73
+ chunks.append(current)
74
+ current = ''
75
+ chunks.extend(piece[i:i + max_chars] for i in range(0, len(piece), max_chars))
76
+ elif len(current) + len(piece) > max_chars:
77
+ chunks.append(current)
78
+ current = piece
79
+ else:
80
+ current += piece
81
+ if current:
82
+ chunks.append(current)
83
+ return chunks
@@ -0,0 +1,117 @@
1
+ """Neutral contracts between xtremeparse and the host application.
2
+
3
+ The library never imports a schema vendor, an agent framework, or a
4
+ validator. Everything crosses the border as plain data: JSON Schema in,
5
+ JSON-shaped dict out, issues and agent calls behind the protocols below.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field
11
+ from typing import Any, Callable, Optional, Protocol, Sequence, runtime_checkable
12
+
13
+ # A plain JSON Schema mapping (``schema.to_json_schema()`` output, or any
14
+ # hand-written schema with the same shape).
15
+ JSONSchema = dict[str, Any]
16
+
17
+
18
+ @runtime_checkable
19
+ class Issue(Protocol):
20
+ """A validation problem found in extracted data.
21
+
22
+ Structural protocol: any object with these attributes satisfies it,
23
+ zero adapter code. ``expected``/``got`` are optional free-form context
24
+ (hosts pass lists for enums, arbitrary values, sentinels for absent).
25
+ """
26
+
27
+ path: str
28
+ message: str
29
+ code: str
30
+ expected: object
31
+ got: object
32
+
33
+
34
+ # Validates complete extracted data, returns the issues the host considers
35
+ # error-level. Feeding only errors is the host's severity policy: the
36
+ # correction loop retries whatever it receives.
37
+ Validator = Callable[[dict], Sequence[Issue]]
38
+
39
+
40
+ @dataclass
41
+ class AgentResult:
42
+ """One agent turn's outcome.
43
+
44
+ ``data`` is already coerced to the requested ``result_schema`` by the
45
+ adapter (e.g. PydanticAI structured output). ``history`` is the opaque
46
+ message history for multi-turn continuation; the library passes it back
47
+ verbatim on correction rounds. Usage/cost telemetry is deliberately not
48
+ here — the adapter layer instruments its own agent calls.
49
+ """
50
+
51
+ data: Any
52
+ history: list
53
+
54
+
55
+ @runtime_checkable
56
+ class AgentRunner(Protocol):
57
+ """The single agent-call abstraction the library owns.
58
+
59
+ The host adapts this to its agent framework (PydanticAI today, anything
60
+ else tomorrow). Prompt layout is frozen for cache reasons:
61
+ ``instructions`` carries the unit's semantic card and ``content`` the
62
+ shared, byte-identical payload (full text + full schema) that makes the
63
+ provider's KV cache hit across every call of one extraction;
64
+ ``scope`` carries this call's private slice — the routed chunks of one
65
+ group (None for the router). On ``history`` rounds the transcript
66
+ already carries the card and scope, so the caller passes ``''`` for
67
+ both — re-sending them would re-bill the same text, and only the
68
+ per-round ``feedback`` is fresh. ``result_schema`` is the JSON Schema the
69
+ returned ``data`` must satisfy; ``tools``/``history``/``feedback``
70
+ serve the multi-turn correction loop.
71
+ """
72
+
73
+ async def run(
74
+ self,
75
+ *,
76
+ instructions: str,
77
+ result_schema: JSONSchema,
78
+ content: str,
79
+ scope: Optional[str] = None,
80
+ tools: Optional[list] = None,
81
+ history: Optional[list] = None,
82
+ feedback: Optional[Sequence[Issue]] = None,
83
+ ) -> AgentResult:
84
+ ...
85
+
86
+
87
+ @dataclass
88
+ class Trace:
89
+ """Orchestration record: what was chunked, routed, run and retried.
90
+
91
+ Populated by the pipeline stages; read by eval suites and telemetry.
92
+ Token/latency accounting of agent calls stays with the adapter — this
93
+ trace records orchestration events only. ``prompts`` marks each
94
+ prompt slot 'default' or with a short hash of the host's override
95
+ template (see prompting.provenance) — regressions stay attributable
96
+ to whose prompt produced them.
97
+ """
98
+
99
+ chunks: list = field(default_factory=list)
100
+ router: Optional[dict] = None
101
+ groups: list = field(default_factory=list)
102
+ corrections: list = field(default_factory=list)
103
+ prompts: dict = field(default_factory=dict)
104
+
105
+
106
+ @dataclass
107
+ class ExtractionResult:
108
+ """Extraction never raises on bad data — it reports.
109
+
110
+ ``data`` is the best-effort merged result (missing pieces stay absent),
111
+ ``issues`` the unresolved error-level issues from the last validation,
112
+ both lenient by design. Strictness is the caller's policy.
113
+ """
114
+
115
+ data: dict
116
+ issues: list
117
+ trace: Trace
@@ -0,0 +1,166 @@
1
+ """Correction loop: whole-data validation, issue routing, bounded re-runs.
2
+
3
+ The injected validator defines severity — whatever it feeds is retried.
4
+ Issues route back to the specialist call that owns their path (per-item
5
+ calls only when the item index is derivable), and re-runs carry the call's
6
+ conversation history plus the routed feedback. Stops on clean, budget
7
+ (default 2 rounds), or no progress (issue paths identical to the previous
8
+ round). Never raises on bad data; unresolved issues return with the data.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import asyncio
14
+ import re
15
+ from dataclasses import dataclass
16
+ from typing import Optional
17
+
18
+ from xtremeparse.contracts import AgentRunner, Validator
19
+ from xtremeparse.paths import resolve, resolve_list
20
+ from xtremeparse.executor import Call, Execution, dispatch_specialist, values_from_calls
21
+ from xtremeparse.merge import merge
22
+ from xtremeparse.prompting import json_len
23
+ from xtremeparse.scheduling import TaskScheduler
24
+ from xtremeparse.units import MISC
25
+
26
+ MAX_ROUNDS = 2
27
+
28
+
29
+ @dataclass
30
+ class Round:
31
+ """One correction round's trace material."""
32
+
33
+ unit_path: str
34
+ item: Optional[int]
35
+ issue_paths: list
36
+
37
+
38
+ async def correct(runner: AgentRunner, execution: Execution, *, validator: Validator,
39
+ payload: str, scheduler: TaskScheduler,
40
+ max_rounds: int = MAX_ROUNDS,
41
+ specialist_instructions: str = None) -> tuple:
42
+ """Re-run failing calls with feedback until clean, budgeted, or
43
+ stuck. Returns ``(data, issues, rounds)`` — always lenient. Call
44
+ results mutate in place; re-merge from ``execution.calls`` rather
45
+ than the now-stale ``execution.values``. ``specialist_instructions``
46
+ must be the same override the first round ran with — a correction
47
+ round continues that call's conversation history."""
48
+ calls = execution.calls
49
+ data = merge(values_from_calls(calls))
50
+ issues = list(validator(data) or [])
51
+ rounds, seen = [], None
52
+ for _ in range(max_rounds):
53
+ paths = {i.path for i in issues}
54
+ if not issues or paths == seen:
55
+ break
56
+ seen = paths
57
+ routed = _route(calls, issues)
58
+ if not routed:
59
+ break
60
+ rounds.extend(Round(call.unit.path, call.item, [i.path for i in feedback])
61
+ for call, feedback in routed)
62
+ tasks = [await dispatch_specialist(
63
+ runner, call.unit, payload=payload, scope=call.scope,
64
+ whole=call.strategy == 'whole', scheduler=scheduler,
65
+ specialist_instructions=specialist_instructions,
66
+ history=call.result.history if call.result else None,
67
+ feedback=feedback)
68
+ for call, feedback in routed]
69
+ for (call, _), result in zip(routed, await asyncio.gather(*tasks)):
70
+ call.result = result
71
+ data = merge(values_from_calls(calls))
72
+ issues = list(validator(data) or [])
73
+ return data, issues, rounds
74
+
75
+
76
+ def _route(calls: list, issues: list) -> list:
77
+ """Map issues to owning calls. Longest unit path wins; per-item calls
78
+ match only their item; $misc is the fallback for unmatched paths."""
79
+ grouped = {}
80
+ for issue in issues:
81
+ if call := _owner(calls, issue.path):
82
+ grouped.setdefault(id(call), (call, []))[1].append(issue)
83
+ return list(grouped.values())
84
+
85
+
86
+ def _owner(calls: list, path: str):
87
+ covering = [c for c in calls if c.unit.path != MISC and _under(path, c.unit.path)]
88
+ exact = [c for c in covering if c.item is None and not c.batch
89
+ or c.item is not None and _under(path, f'{c.unit.path}[{c.item}]')
90
+ or c.batch and (_item_index(path, c.unit.path) or -1) in c.batch]
91
+ pool = exact or [c for c in calls if c.unit.path == MISC]
92
+ return max(pool, key=lambda c: len(c.unit.path), default=None)
93
+
94
+
95
+ def _item_index(path: str, unit_path: str):
96
+ m = re.fullmatch(rf'{re.escape(unit_path)}\[(\d+)\](\..*)?', path)
97
+ return int(m.group(1)) if m else None
98
+
99
+
100
+ def _under(path: str, root: str) -> bool:
101
+ """``path`` is ``root`` itself or a proper child of it."""
102
+ return path == root or path.startswith(f'{root}.') or path.startswith(f'{root}[')
103
+
104
+
105
+ def item_chars(budgets: dict, data: dict) -> dict:
106
+ """Per budgeted path: ``[(key, arranged, chars)]`` — one entry per
107
+ item for lists, one for the whole value otherwise, each carrying
108
+ its own arranged budget (a lone number covers every item; a short
109
+ list's last value covers items beyond it). The counting surface for
110
+ budget reads (trace, eval audits); the unit is the compact
111
+ serialized item JSON (keys and punctuation included) — exactly what
112
+ a specialist types and pays decode for. Overruns are ACCEPTED,
113
+ never retried: a retry costs a full extra decode, the very thing
114
+ budgets exist to save — budgets shape batch scheduling only."""
115
+ out = {}
116
+ for path, arranged in budgets.items():
117
+ if (value := resolve(data, path)) is None:
118
+ continue
119
+ values = arranged if isinstance(arranged, list) else [arranged]
120
+ entries = []
121
+ for i, item in (enumerate(value) if isinstance(value, list)
122
+ else [(None, value)]):
123
+ budget = values[i] if i is not None and i < len(values) else values[-1]
124
+ entries.append((f'{path}[{i}]' if i is not None else path,
125
+ budget, json_len(item)))
126
+ out[path] = entries
127
+ return out
128
+
129
+
130
+ @dataclass
131
+ class _CountIssue:
132
+ """A routed item count the merged data does not honour — typically a
133
+ whole-array call that collapsed instances into fewer entries."""
134
+
135
+ path: str
136
+ code: str = 'item_count'
137
+ message: str = ''
138
+ expected: int = None
139
+ got: int = None
140
+
141
+
142
+ def count_issues(counts: dict, data: dict) -> list:
143
+ """The router's declared counts (map-validated ground truth) against
144
+ the merged arrays. A short array means instances were collapsed or
145
+ dropped — silent to schema validation (nothing declares minItems).
146
+ Items concatenate in item-index order, so a short array is missing
147
+ its tail: each missing index becomes its own issue and routes to the
148
+ call that owns it (a batch member or a single). One that survives a
149
+ retry stops via the no-progress rule."""
150
+ issues = []
151
+ for path, declared in counts.items():
152
+ actual = len(resolve_list(data, path))
153
+ if actual < declared:
154
+ issues += [_CountIssue(
155
+ f'{path}[{i}]', expected=declared, got=actual,
156
+ message=f'{path} is missing item {i} of {declared} — the '
157
+ f'array holds {actual}; return every instance as '
158
+ 'its own entry, without splitting or duplicating')
159
+ for i in range(actual, declared)]
160
+ elif actual > declared:
161
+ issues.append(_CountIssue(
162
+ path, expected=declared, got=actual,
163
+ message=f'declared {declared} items but the array holds '
164
+ f'{actual} — merge the duplicates'))
165
+ return issues
166
+
xtremeparse/evalkit.py ADDED
@@ -0,0 +1,43 @@
1
+ """Eval support: plain-data reads over the orchestration trace.
2
+
3
+ The lib keeps the reads its shipped evaluators (the ``evals`` extra)
4
+ need: the executor's calling conventions (per_item_budgets) and the
5
+ router's contract (router_overlap). Hosts own their general eval
6
+ tooling — digest projections, anchor checks, assertion DSLs are each
7
+ host's selection.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+
13
+ def per_item_budgets(groups: list) -> dict:
14
+ """Trace groups → the per-path budget map ``item_chars`` reads: one
15
+ number or the item-index-ordered list."""
16
+ whole, by_item = {}, {}
17
+ for g in groups:
18
+ if not g.get('budget'):
19
+ continue
20
+ items = g.get('batch') or ([g['item']] if g['item'] is not None else None)
21
+ if items is None: # whole-array call: its list is document-ordered
22
+ whole[g['unit']] = g['budget']
23
+ continue
24
+ values = g['budget'] if isinstance(g['budget'], list) else None
25
+ for k, i in enumerate(items):
26
+ by_item.setdefault(g['unit'], {})[i] = values[k] if values else g['budget']
27
+ return whole | {unit: [items[i] for i in sorted(items)]
28
+ for unit, items in by_item.items()}
29
+
30
+
31
+ def router_overlap(router, groups: list) -> tuple:
32
+ """(overlap, items_lost) between the router's assignments and the
33
+ executed calls; only per-item executions are held to item presence
34
+ (whole-strategy coalescing is legitimate)."""
35
+ assignments = router['assignments'] if router else []
36
+ raw = sum(len(a['chunks']) for a in assignments)
37
+ present = {(g['unit'], i) for g in groups
38
+ for i in (g.get('batch') or [g['item']])}
39
+ per_item_units = {g['unit'] for g in groups if g['strategy'] == 'per-item'}
40
+ declared = {(a['unit'], a.get('item')) for a in assignments
41
+ if a['unit'] in per_item_units}
42
+ return (raw - sum(len(g['chunk_ids']) for g in groups),
43
+ len(declared - present))
xtremeparse/evals.py ADDED
@@ -0,0 +1,97 @@
1
+ """pydantic-evals adapters over the trace reads — install the ``evals``
2
+ extra (``pip install xtremeparse[evals]``). The evaluators accept any
3
+ output carrying ``.trace`` and ``.data`` the way ``ExtractionResult``
4
+ does; hosts keep their domain evaluators beside these.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+
11
+ from pydantic_evals.evaluators import Evaluator, EvaluatorContext, EvaluationReason
12
+
13
+ from xtremeparse.corrections import item_chars
14
+ from xtremeparse.evalkit import per_item_budgets, router_overlap
15
+ from xtremeparse.judging import JUDGE_PLACEHOLDERS, judge
16
+ from xtremeparse.prompting import check_placeholders
17
+
18
+
19
+ class RouterOverlap(Evaluator):
20
+ """Invariant tripwire: router validation makes overlap and phantom
21
+ items structurally impossible — a red here means the router contract
22
+ regressed, not a flaky model."""
23
+
24
+ def evaluate(self, ctx: EvaluatorContext):
25
+ overlap, lost = router_overlap(ctx.output.trace.router,
26
+ ctx.output.trace.groups)
27
+ return EvaluationReason(lost == 0, f'overlap={overlap}, items_lost={lost}')
28
+
29
+
30
+ class BudgetFit(Evaluator):
31
+ """Allocation audit: the model ARRANGES each budget — its estimate
32
+ of the characters an item's output JSON will run to, resolved by
33
+ code against the routed material — so actuals should land near
34
+ the arranged number. The lower edge is loose on purpose: an idle
35
+ estimate on a naturally short item is harmless (output runs at
36
+ natural size either way). The upper edge flags arrangements so
37
+ far off the router clearly wasn't counting. Per item: a mean
38
+ would hide spread."""
39
+
40
+ BAND = (0.4, 2.0)
41
+
42
+ def evaluate(self, ctx: EvaluatorContext):
43
+ ratios, wild = {}, []
44
+ budgets = per_item_budgets(ctx.output.trace.groups)
45
+ for path, entries in item_chars(budgets, ctx.output.data).items():
46
+ unit = path.rsplit('.', 1)[-1]
47
+ judged = [e for e in entries if e[1] and e[2]] # budgeted, non-empty
48
+ if rs := [round(chars / budget, 2) for _, budget, chars in judged]:
49
+ ratios[unit] = f'{min(rs)}' if len(rs) == 1 else f'{min(rs)}-{max(rs)}'
50
+ wild += [key.rsplit('.', 1)[-1] for key, budget, chars in judged
51
+ if not self.BAND[0] <= chars / budget <= self.BAND[1]]
52
+ return EvaluationReason(
53
+ not wild, f'act/budget {ratios or "no budgets declared"}'
54
+ + (f', wild: {wild}' if wild else ''))
55
+
56
+
57
+ class RubricJudge(Evaluator):
58
+ """One rubric call through an injected AgentRunner (see
59
+ xtremeparse.judging) — the judge runs on whatever framework the
60
+ host adapted, where pydantic-evals' own LLMJudge ties the harness
61
+ to a pydantic-ai model. ``source`` is the judged material's
62
+ document, held at construction; ``output_of`` projects the case
63
+ output into what the rubric asks about (e.g. a digest projection
64
+ for containment questions) — see docs/prompting.md for the
65
+ measured judge guidance. The judge's own model settings are
66
+ recommendations carried at the adapter."""
67
+
68
+ def __init__(self, runner, rubric: str, source: str = '', *,
69
+ output_of=None, instructions: str = None):
70
+ if instructions is not None:
71
+ check_placeholders(instructions, JUDGE_PLACEHOLDERS, 'instructions')
72
+ self.runner = runner
73
+ self.rubric = rubric
74
+ self.source = source
75
+ self.output_of = output_of
76
+ self.instructions = instructions
77
+
78
+ async def evaluate(self, ctx: EvaluatorContext):
79
+ output = _text(self.output_of(ctx.output) if self.output_of
80
+ else ctx.output)
81
+ verdict = await judge(self.runner, rubric=self.rubric,
82
+ source=self.source, output=output,
83
+ instructions=self.instructions)
84
+ return EvaluationReason(verdict.ok, verdict.reason)
85
+
86
+
87
+ def _text(value) -> str:
88
+ """Stable judge-facing text: an ExtractionResult-shaped value is
89
+ judged by its data; a plain string stays verbatim (JSON-escaping a
90
+ document-sized string would bury its structure); the rest
91
+ serializes compactly (``default=str`` absorbs harness objects like
92
+ Path)."""
93
+ if hasattr(value, 'data'):
94
+ value = value.data
95
+ if isinstance(value, str):
96
+ return value
97
+ return json.dumps(value, ensure_ascii=False, sort_keys=True, default=str)