openodke 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- openodke/__init__.py +118 -0
- openodke/_text.py +37 -0
- openodke/chunking.py +164 -0
- openodke/cli/__init__.py +0 -0
- openodke/cli/main.py +399 -0
- openodke/corroborate/__init__.py +54 -0
- openodke/corroborate/merge.py +400 -0
- openodke/corroborate/normalize.py +346 -0
- openodke/corroborate/provenance.py +63 -0
- openodke/corroborate/resolve.py +410 -0
- openodke/corroborate/score.py +129 -0
- openodke/eval/__init__.py +102 -0
- openodke/eval/ablation.py +270 -0
- openodke/eval/calibration.py +142 -0
- openodke/eval/cost.py +258 -0
- openodke/eval/extraction.py +258 -0
- openodke/eval/formats.py +329 -0
- openodke/eval/grounding.py +159 -0
- openodke/eval/report.py +185 -0
- openodke/eval/resolution.py +202 -0
- openodke/eval/routing.py +99 -0
- openodke/eval/runner.py +145 -0
- openodke/eval/sinks.py +94 -0
- openodke/eval/validation.py +80 -0
- openodke/extract/__init__.py +24 -0
- openodke/extract/_common.py +202 -0
- openodke/extract/hybrid.py +145 -0
- openodke/extract/llm.py +412 -0
- openodke/extract/pattern.py +366 -0
- openodke/ground/__init__.py +32 -0
- openodke/ground/llm.py +304 -0
- openodke/ground/retry.py +151 -0
- openodke/ground/span.py +161 -0
- openodke/infer/__init__.py +82 -0
- openodke/infer/build.py +417 -0
- openodke/infer/candidates.py +107 -0
- openodke/infer/llm.py +483 -0
- openodke/infer/merge.py +438 -0
- openodke/infer/names.py +125 -0
- openodke/infer/propose.py +730 -0
- openodke/infer/review.py +220 -0
- openodke/infer/sample.py +257 -0
- openodke/llm/__init__.py +60 -0
- openodke/llm/base.py +111 -0
- openodke/llm/litellm_client.py +86 -0
- openodke/llm/openai_compat.py +143 -0
- openodke/llm/registry.py +66 -0
- openodke/llm/roles.py +59 -0
- openodke/llm/testing.py +351 -0
- openodke/loaders/__init__.py +61 -0
- openodke/loaders/base.py +75 -0
- openodke/loaders/directory.py +109 -0
- openodke/loaders/docx.py +182 -0
- openodke/loaders/html.py +495 -0
- openodke/loaders/pdf.py +137 -0
- openodke/loaders/records.py +196 -0
- openodke/loaders/sourcemap.py +247 -0
- openodke/loaders/structured.py +209 -0
- openodke/loaders/text.py +130 -0
- openodke/ontology/__init__.py +582 -0
- openodke/ontology/diff.py +224 -0
- openodke/ontology/from_models.py +204 -0
- openodke/ontology/from_neo4j.py +403 -0
- openodke/ontology/from_owl.py +599 -0
- openodke/ontology/load.py +216 -0
- openodke/ontology/validate.py +231 -0
- openodke/pipeline.py +253 -0
- openodke/py.typed +0 -0
- openodke/run/__init__.py +46 -0
- openodke/run/build.py +945 -0
- openodke/run/config.py +293 -0
- openodke/run/execute.py +330 -0
- openodke/sinks/__init__.py +12 -0
- openodke/sinks/bulk.py +526 -0
- openodke/sinks/jsonl.py +48 -0
- openodke/sinks/neo4j.py +727 -0
- openodke/sinks/networkx.py +177 -0
- openodke/sinks/rdf.py +385 -0
- openodke/stages.py +445 -0
- openodke/types.py +357 -0
- openodke/validators.py +65 -0
- openodke-0.1.0.dist-info/METADATA +368 -0
- openodke-0.1.0.dist-info/RECORD +87 -0
- openodke-0.1.0.dist-info/WHEEL +4 -0
- openodke-0.1.0.dist-info/entry_points.txt +3 -0
- openodke-0.1.0.dist-info/licenses/LICENSE +202 -0
- openodke-0.1.0.dist-info/licenses/NOTICE +26 -0
openodke/__init__.py
ADDED
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""openodke — ontology-guided knowledge extraction.
|
|
2
|
+
|
|
3
|
+
Text in, a grounded knowledge graph out, ready for Neo4j or any graph store.
|
|
4
|
+
|
|
5
|
+
An independent implementation of the architecture in ODKE+ (arXiv:2509.04696),
|
|
6
|
+
generalised from one production knowledge graph to a general-purpose SDK. See
|
|
7
|
+
NOTICE for the relationship to that paper.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from openodke.chunking import SentenceChunker
|
|
11
|
+
from openodke.corroborate import (
|
|
12
|
+
EvidenceScorer,
|
|
13
|
+
NativeResolver,
|
|
14
|
+
SignatureCorroborator,
|
|
15
|
+
ValueNormalizer,
|
|
16
|
+
)
|
|
17
|
+
from openodke.extract import HybridExtractor, LLMExtractor, PatternExtractor
|
|
18
|
+
from openodke.ontology import (
|
|
19
|
+
Diagnostic,
|
|
20
|
+
EntityType,
|
|
21
|
+
Ontology,
|
|
22
|
+
OntologyImportWarning,
|
|
23
|
+
OntologyLoadError,
|
|
24
|
+
OntologySnippet,
|
|
25
|
+
Predicate,
|
|
26
|
+
Qualifier,
|
|
27
|
+
)
|
|
28
|
+
from openodke.pipeline import DoubleStageWarning, Pipeline
|
|
29
|
+
from openodke.stages import (
|
|
30
|
+
Chunker,
|
|
31
|
+
Constrainer,
|
|
32
|
+
Corroborator,
|
|
33
|
+
Delegated,
|
|
34
|
+
Extractor,
|
|
35
|
+
Grounder,
|
|
36
|
+
Inferrer,
|
|
37
|
+
Loader,
|
|
38
|
+
Normalizer,
|
|
39
|
+
PlatformProfile,
|
|
40
|
+
Resolver,
|
|
41
|
+
Router,
|
|
42
|
+
Scorer,
|
|
43
|
+
Sink,
|
|
44
|
+
Validator,
|
|
45
|
+
)
|
|
46
|
+
from openodke.types import (
|
|
47
|
+
Chunk,
|
|
48
|
+
Document,
|
|
49
|
+
Entity,
|
|
50
|
+
EntityLink,
|
|
51
|
+
Evidence,
|
|
52
|
+
Fact,
|
|
53
|
+
GroundingVerdict,
|
|
54
|
+
KnowledgeGraph,
|
|
55
|
+
LinkKind,
|
|
56
|
+
Polarity,
|
|
57
|
+
Resolution,
|
|
58
|
+
RouteVerdict,
|
|
59
|
+
SourceTier,
|
|
60
|
+
Span,
|
|
61
|
+
ValidationVerdict,
|
|
62
|
+
)
|
|
63
|
+
from openodke.validators import VerdictValidator
|
|
64
|
+
|
|
65
|
+
__version__ = "0.1.0"
|
|
66
|
+
|
|
67
|
+
__all__ = [
|
|
68
|
+
"Chunk",
|
|
69
|
+
"Chunker",
|
|
70
|
+
"Constrainer",
|
|
71
|
+
"Corroborator",
|
|
72
|
+
"Delegated",
|
|
73
|
+
"Diagnostic",
|
|
74
|
+
"Document",
|
|
75
|
+
"DoubleStageWarning",
|
|
76
|
+
"Entity",
|
|
77
|
+
"EntityLink",
|
|
78
|
+
"EntityType",
|
|
79
|
+
"Evidence",
|
|
80
|
+
"EvidenceScorer",
|
|
81
|
+
"Extractor",
|
|
82
|
+
"Fact",
|
|
83
|
+
"Grounder",
|
|
84
|
+
"GroundingVerdict",
|
|
85
|
+
"HybridExtractor",
|
|
86
|
+
"Inferrer",
|
|
87
|
+
"KnowledgeGraph",
|
|
88
|
+
"LLMExtractor",
|
|
89
|
+
"LinkKind",
|
|
90
|
+
"Loader",
|
|
91
|
+
"NativeResolver",
|
|
92
|
+
"Normalizer",
|
|
93
|
+
"Ontology",
|
|
94
|
+
"OntologyImportWarning",
|
|
95
|
+
"OntologyLoadError",
|
|
96
|
+
"OntologySnippet",
|
|
97
|
+
"PatternExtractor",
|
|
98
|
+
"Pipeline",
|
|
99
|
+
"PlatformProfile",
|
|
100
|
+
"Polarity",
|
|
101
|
+
"Predicate",
|
|
102
|
+
"Qualifier",
|
|
103
|
+
"Resolution",
|
|
104
|
+
"Resolver",
|
|
105
|
+
"RouteVerdict",
|
|
106
|
+
"Router",
|
|
107
|
+
"Scorer",
|
|
108
|
+
"SentenceChunker",
|
|
109
|
+
"SignatureCorroborator",
|
|
110
|
+
"Sink",
|
|
111
|
+
"SourceTier",
|
|
112
|
+
"Span",
|
|
113
|
+
"ValidationVerdict",
|
|
114
|
+
"Validator",
|
|
115
|
+
"ValueNormalizer",
|
|
116
|
+
"VerdictValidator",
|
|
117
|
+
"__version__",
|
|
118
|
+
]
|
openodke/_text.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Offset-keeping text helpers shared by the loaders and the pattern extractor."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections.abc import Iterator
|
|
7
|
+
|
|
8
|
+
_LINE = re.compile(r"[^\r\n]*(?:\r\n|\r|\n)?")
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def iter_lines(text: str, start: int = 0, end: int | None = None) -> Iterator[tuple[int, int]]:
|
|
12
|
+
"""`(start, end)` of every line in `text[start:end]`, line break excluded.
|
|
13
|
+
|
|
14
|
+
Not `str.splitlines`: that also breaks on form feeds and Unicode separators,
|
|
15
|
+
which a Markdown table or a `Key: value` block never does, and it returns
|
|
16
|
+
strings rather than the offsets a span needs.
|
|
17
|
+
"""
|
|
18
|
+
stop = len(text) if end is None else end
|
|
19
|
+
pos = start
|
|
20
|
+
while pos < stop:
|
|
21
|
+
match = _LINE.match(text, pos, stop)
|
|
22
|
+
if match is None: # pragma: no cover - the pattern matches the empty string
|
|
23
|
+
break
|
|
24
|
+
line_end = match.end()
|
|
25
|
+
while line_end > pos and text[line_end - 1] in "\r\n":
|
|
26
|
+
line_end -= 1
|
|
27
|
+
yield pos, line_end
|
|
28
|
+
pos = match.end()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def trim(text: str, start: int, end: int) -> tuple[int, int]:
|
|
32
|
+
"""Narrow `[start, end)` past whitespace on both sides, without copying."""
|
|
33
|
+
while start < end and text[start].isspace():
|
|
34
|
+
start += 1
|
|
35
|
+
while end > start and text[end - 1].isspace():
|
|
36
|
+
end -= 1
|
|
37
|
+
return start, end
|
openodke/chunking.py
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Splitting documents into chunks without losing a single offset.
|
|
2
|
+
|
|
3
|
+
DECISIONS #3 made spans character offsets and #19 made the chunk the unit of the
|
|
4
|
+
pipeline. Together they put the weight of provenance on this module: a chunk's
|
|
5
|
+
`text` must be exactly `doc.text[start:end]`, or a span an extractor finds inside
|
|
6
|
+
a chunk points at characters that were never there and the grounder checks a
|
|
7
|
+
quote against nothing.
|
|
8
|
+
|
|
9
|
+
So nothing here normalises. Paragraphs and sentences are found by scanning the
|
|
10
|
+
original string, a chunk is a pair of indices into it, and its text is sliced
|
|
11
|
+
rather than rebuilt. CRLF, tabs, runs of spaces and non-ASCII text survive
|
|
12
|
+
because they are never touched. Offsets are Python string indices — code points
|
|
13
|
+
— which is what `Span` and `Chunk` already mean.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
from collections.abc import Iterator
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
|
|
22
|
+
from openodke.types import Chunk, Document
|
|
23
|
+
|
|
24
|
+
# `\r(?!\n)`: without the lookahead one CRLF backtracks into two line breaks and
|
|
25
|
+
# every Windows line becomes a paragraph.
|
|
26
|
+
_LINE_BREAK = r"(?:\r\n|\r(?!\n)|\n)"
|
|
27
|
+
_PARAGRAPH_BREAK = re.compile(
|
|
28
|
+
rf"{_LINE_BREAK}(?:[^\S\r\n]*{_LINE_BREAK})+" "|\N{PARAGRAPH SEPARATOR}"
|
|
29
|
+
)
|
|
30
|
+
_CLOSERS = "\"'”’)\\]」』"
|
|
31
|
+
# Western terminators need whitespace after them ("3.14", "example.com" are not
|
|
32
|
+
# ends); CJK full stops do not, because CJK text has no spaces to wait for.
|
|
33
|
+
_SENTENCE_END = re.compile(rf"[.!?…]+[{_CLOSERS}]*(?=\s|\Z)|[。!?]+[{_CLOSERS}]*")
|
|
34
|
+
_WORD = re.compile(r"\S+")
|
|
35
|
+
# Only the ones that are nearly always followed by a capitalised name. A miss
|
|
36
|
+
# here merges two sentences, which is the safe direction: it never splits one.
|
|
37
|
+
_ABBREVIATIONS = frozenset(
|
|
38
|
+
{"mr", "mrs", "ms", "dr", "prof", "sr", "jr", "st", "vs", "e.g", "i.e", "cf", "fig", "approx"}
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass(frozen=True, slots=True)
|
|
43
|
+
class _Sentence:
|
|
44
|
+
start: int
|
|
45
|
+
end: int
|
|
46
|
+
words: int
|
|
47
|
+
opens_paragraph: bool
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
class SentenceChunker:
|
|
51
|
+
"""Packs whole sentences into chunks of at most `max_words` words.
|
|
52
|
+
|
|
53
|
+
Boundaries are chosen in order of preference: a paragraph break, then a
|
|
54
|
+
sentence break, and never anything smaller. A router asked "fact or
|
|
55
|
+
narrative?" about half a sentence is asked nothing (DECISIONS #19), so a
|
|
56
|
+
single sentence longer than the cap becomes a chunk of its own rather than
|
|
57
|
+
being cut — and a document with no punctuation at all is one sentence per
|
|
58
|
+
paragraph.
|
|
59
|
+
|
|
60
|
+
A chunk ends at the last paragraph break inside it, unless that would leave
|
|
61
|
+
it less than half full; a heading stranded as its own chunk is as useless to
|
|
62
|
+
a router as half a sentence. Chunks start and end on non-whitespace, so the
|
|
63
|
+
whitespace between two chunks belongs to neither.
|
|
64
|
+
|
|
65
|
+
`overlap` repeats that many trailing sentences at the start of the next
|
|
66
|
+
chunk, so a fact stated across a boundary is seen whole at least once. It is
|
|
67
|
+
shed first whenever it would crowd out new text. Overlapping chunks emit the
|
|
68
|
+
same fact twice; the corroborator merges them by signature.
|
|
69
|
+
"""
|
|
70
|
+
|
|
71
|
+
def __init__(self, max_words: int = 200, *, overlap: int = 0) -> None:
|
|
72
|
+
if max_words < 1:
|
|
73
|
+
raise ValueError("max_words must be at least 1")
|
|
74
|
+
if overlap < 0:
|
|
75
|
+
raise ValueError("overlap cannot be negative")
|
|
76
|
+
self.max_words = max_words
|
|
77
|
+
self.overlap = overlap
|
|
78
|
+
|
|
79
|
+
def chunk(self, doc: Document) -> Iterator[Chunk]:
|
|
80
|
+
sentences = list(_segment(doc.text))
|
|
81
|
+
n = len(sentences)
|
|
82
|
+
prefix = [0]
|
|
83
|
+
for s in sentences:
|
|
84
|
+
prefix.append(prefix[-1] + s.words)
|
|
85
|
+
|
|
86
|
+
first = 0 # first sentence of the chunk being built
|
|
87
|
+
covered = 0 # sentences [0, covered) already emitted
|
|
88
|
+
index = 0
|
|
89
|
+
while covered < n:
|
|
90
|
+
end = first
|
|
91
|
+
while end < n and prefix[end + 1] - prefix[first] <= self.max_words:
|
|
92
|
+
end += 1
|
|
93
|
+
if end <= covered:
|
|
94
|
+
if first < covered:
|
|
95
|
+
# The carried-over overlap leaves no room for anything new.
|
|
96
|
+
first += 1
|
|
97
|
+
continue
|
|
98
|
+
# One sentence over the cap: it stands alone, uncut.
|
|
99
|
+
end = covered + 1
|
|
100
|
+
|
|
101
|
+
if end < n and not sentences[end].opens_paragraph:
|
|
102
|
+
for k in range(end - 1, covered, -1):
|
|
103
|
+
if sentences[k].opens_paragraph:
|
|
104
|
+
if 2 * (prefix[k] - prefix[first]) >= self.max_words:
|
|
105
|
+
end = k
|
|
106
|
+
break
|
|
107
|
+
|
|
108
|
+
start, stop = sentences[first].start, sentences[end - 1].end
|
|
109
|
+
yield Chunk(
|
|
110
|
+
doc_id=doc.id, start=start, end=stop, text=doc.text[start:stop], index=index
|
|
111
|
+
)
|
|
112
|
+
index += 1
|
|
113
|
+
covered = end
|
|
114
|
+
first = max(end - self.overlap, first + 1)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _segment(text: str) -> Iterator[_Sentence]:
|
|
118
|
+
"""Every sentence in `text`, in order, with whitespace trimmed from both ends."""
|
|
119
|
+
cursor = 0
|
|
120
|
+
for brk in [*_PARAGRAPH_BREAK.finditer(text), None]:
|
|
121
|
+
stop = len(text) if brk is None else brk.start()
|
|
122
|
+
start, end = _trim(text, cursor, stop)
|
|
123
|
+
opens = True
|
|
124
|
+
for s, e in _sentences_in(text, start, end):
|
|
125
|
+
yield _Sentence(s, e, len(_WORD.findall(text, s, e)), opens)
|
|
126
|
+
opens = False
|
|
127
|
+
if brk is not None:
|
|
128
|
+
cursor = brk.end()
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _sentences_in(text: str, start: int, end: int) -> Iterator[tuple[int, int]]:
|
|
132
|
+
cursor = start
|
|
133
|
+
for match in _SENTENCE_END.finditer(text, start, end):
|
|
134
|
+
if _is_abbreviation(text, cursor, match, end):
|
|
135
|
+
continue
|
|
136
|
+
yield cursor, match.end()
|
|
137
|
+
cursor, _ = _trim(text, match.end(), end)
|
|
138
|
+
if cursor < end:
|
|
139
|
+
yield cursor, end
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _is_abbreviation(text: str, sentence_start: int, match: re.Match[str], end: int) -> bool:
|
|
143
|
+
if match.group().rstrip(_CLOSERS) != ".":
|
|
144
|
+
return False
|
|
145
|
+
following, _ = _trim(text, match.end(), end)
|
|
146
|
+
if following < end and text[following].islower():
|
|
147
|
+
return True
|
|
148
|
+
token_start = match.start()
|
|
149
|
+
while token_start > sentence_start and not text[token_start - 1].isspace():
|
|
150
|
+
token_start -= 1
|
|
151
|
+
token = text[token_start : match.start()].lstrip(_CLOSERS + "(“‘")
|
|
152
|
+
# A single letter is an initial: "J. R. R. Tolkien".
|
|
153
|
+
return token.casefold() in _ABBREVIATIONS or (len(token) == 1 and token.isalpha())
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _trim(text: str, start: int, end: int) -> tuple[int, int]:
|
|
157
|
+
while start < end and text[start].isspace():
|
|
158
|
+
start += 1
|
|
159
|
+
while end > start and text[end - 1].isspace():
|
|
160
|
+
end -= 1
|
|
161
|
+
return start, end
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
__all__ = ["SentenceChunker"]
|
openodke/cli/__init__.py
ADDED
|
File without changes
|