openodke 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. openodke/__init__.py +118 -0
  2. openodke/_text.py +37 -0
  3. openodke/chunking.py +164 -0
  4. openodke/cli/__init__.py +0 -0
  5. openodke/cli/main.py +399 -0
  6. openodke/corroborate/__init__.py +54 -0
  7. openodke/corroborate/merge.py +400 -0
  8. openodke/corroborate/normalize.py +346 -0
  9. openodke/corroborate/provenance.py +63 -0
  10. openodke/corroborate/resolve.py +410 -0
  11. openodke/corroborate/score.py +129 -0
  12. openodke/eval/__init__.py +102 -0
  13. openodke/eval/ablation.py +270 -0
  14. openodke/eval/calibration.py +142 -0
  15. openodke/eval/cost.py +258 -0
  16. openodke/eval/extraction.py +258 -0
  17. openodke/eval/formats.py +329 -0
  18. openodke/eval/grounding.py +159 -0
  19. openodke/eval/report.py +185 -0
  20. openodke/eval/resolution.py +202 -0
  21. openodke/eval/routing.py +99 -0
  22. openodke/eval/runner.py +145 -0
  23. openodke/eval/sinks.py +94 -0
  24. openodke/eval/validation.py +80 -0
  25. openodke/extract/__init__.py +24 -0
  26. openodke/extract/_common.py +202 -0
  27. openodke/extract/hybrid.py +145 -0
  28. openodke/extract/llm.py +412 -0
  29. openodke/extract/pattern.py +366 -0
  30. openodke/ground/__init__.py +32 -0
  31. openodke/ground/llm.py +304 -0
  32. openodke/ground/retry.py +151 -0
  33. openodke/ground/span.py +161 -0
  34. openodke/infer/__init__.py +82 -0
  35. openodke/infer/build.py +417 -0
  36. openodke/infer/candidates.py +107 -0
  37. openodke/infer/llm.py +483 -0
  38. openodke/infer/merge.py +438 -0
  39. openodke/infer/names.py +125 -0
  40. openodke/infer/propose.py +730 -0
  41. openodke/infer/review.py +220 -0
  42. openodke/infer/sample.py +257 -0
  43. openodke/llm/__init__.py +60 -0
  44. openodke/llm/base.py +111 -0
  45. openodke/llm/litellm_client.py +86 -0
  46. openodke/llm/openai_compat.py +143 -0
  47. openodke/llm/registry.py +66 -0
  48. openodke/llm/roles.py +59 -0
  49. openodke/llm/testing.py +351 -0
  50. openodke/loaders/__init__.py +61 -0
  51. openodke/loaders/base.py +75 -0
  52. openodke/loaders/directory.py +109 -0
  53. openodke/loaders/docx.py +182 -0
  54. openodke/loaders/html.py +495 -0
  55. openodke/loaders/pdf.py +137 -0
  56. openodke/loaders/records.py +196 -0
  57. openodke/loaders/sourcemap.py +247 -0
  58. openodke/loaders/structured.py +209 -0
  59. openodke/loaders/text.py +130 -0
  60. openodke/ontology/__init__.py +582 -0
  61. openodke/ontology/diff.py +224 -0
  62. openodke/ontology/from_models.py +204 -0
  63. openodke/ontology/from_neo4j.py +403 -0
  64. openodke/ontology/from_owl.py +599 -0
  65. openodke/ontology/load.py +216 -0
  66. openodke/ontology/validate.py +231 -0
  67. openodke/pipeline.py +253 -0
  68. openodke/py.typed +0 -0
  69. openodke/run/__init__.py +46 -0
  70. openodke/run/build.py +945 -0
  71. openodke/run/config.py +293 -0
  72. openodke/run/execute.py +330 -0
  73. openodke/sinks/__init__.py +12 -0
  74. openodke/sinks/bulk.py +526 -0
  75. openodke/sinks/jsonl.py +48 -0
  76. openodke/sinks/neo4j.py +727 -0
  77. openodke/sinks/networkx.py +177 -0
  78. openodke/sinks/rdf.py +385 -0
  79. openodke/stages.py +445 -0
  80. openodke/types.py +357 -0
  81. openodke/validators.py +65 -0
  82. openodke-0.1.0.dist-info/METADATA +368 -0
  83. openodke-0.1.0.dist-info/RECORD +87 -0
  84. openodke-0.1.0.dist-info/WHEEL +4 -0
  85. openodke-0.1.0.dist-info/entry_points.txt +3 -0
  86. openodke-0.1.0.dist-info/licenses/LICENSE +202 -0
  87. openodke-0.1.0.dist-info/licenses/NOTICE +26 -0
openodke/__init__.py ADDED
@@ -0,0 +1,118 @@
1
+ """openodke — ontology-guided knowledge extraction.
2
+
3
+ Text in, a grounded knowledge graph out, ready for Neo4j or any graph store.
4
+
5
+ An independent implementation of the architecture in ODKE+ (arXiv:2509.04696),
6
+ generalised from one production knowledge graph to a general-purpose SDK. See
7
+ NOTICE for the relationship to that paper.
8
+ """
9
+
10
+ from openodke.chunking import SentenceChunker
11
+ from openodke.corroborate import (
12
+ EvidenceScorer,
13
+ NativeResolver,
14
+ SignatureCorroborator,
15
+ ValueNormalizer,
16
+ )
17
+ from openodke.extract import HybridExtractor, LLMExtractor, PatternExtractor
18
+ from openodke.ontology import (
19
+ Diagnostic,
20
+ EntityType,
21
+ Ontology,
22
+ OntologyImportWarning,
23
+ OntologyLoadError,
24
+ OntologySnippet,
25
+ Predicate,
26
+ Qualifier,
27
+ )
28
+ from openodke.pipeline import DoubleStageWarning, Pipeline
29
+ from openodke.stages import (
30
+ Chunker,
31
+ Constrainer,
32
+ Corroborator,
33
+ Delegated,
34
+ Extractor,
35
+ Grounder,
36
+ Inferrer,
37
+ Loader,
38
+ Normalizer,
39
+ PlatformProfile,
40
+ Resolver,
41
+ Router,
42
+ Scorer,
43
+ Sink,
44
+ Validator,
45
+ )
46
+ from openodke.types import (
47
+ Chunk,
48
+ Document,
49
+ Entity,
50
+ EntityLink,
51
+ Evidence,
52
+ Fact,
53
+ GroundingVerdict,
54
+ KnowledgeGraph,
55
+ LinkKind,
56
+ Polarity,
57
+ Resolution,
58
+ RouteVerdict,
59
+ SourceTier,
60
+ Span,
61
+ ValidationVerdict,
62
+ )
63
+ from openodke.validators import VerdictValidator
64
+
65
+ __version__ = "0.1.0"
66
+
67
+ __all__ = [
68
+ "Chunk",
69
+ "Chunker",
70
+ "Constrainer",
71
+ "Corroborator",
72
+ "Delegated",
73
+ "Diagnostic",
74
+ "Document",
75
+ "DoubleStageWarning",
76
+ "Entity",
77
+ "EntityLink",
78
+ "EntityType",
79
+ "Evidence",
80
+ "EvidenceScorer",
81
+ "Extractor",
82
+ "Fact",
83
+ "Grounder",
84
+ "GroundingVerdict",
85
+ "HybridExtractor",
86
+ "Inferrer",
87
+ "KnowledgeGraph",
88
+ "LLMExtractor",
89
+ "LinkKind",
90
+ "Loader",
91
+ "NativeResolver",
92
+ "Normalizer",
93
+ "Ontology",
94
+ "OntologyImportWarning",
95
+ "OntologyLoadError",
96
+ "OntologySnippet",
97
+ "PatternExtractor",
98
+ "Pipeline",
99
+ "PlatformProfile",
100
+ "Polarity",
101
+ "Predicate",
102
+ "Qualifier",
103
+ "Resolution",
104
+ "Resolver",
105
+ "RouteVerdict",
106
+ "Router",
107
+ "Scorer",
108
+ "SentenceChunker",
109
+ "SignatureCorroborator",
110
+ "Sink",
111
+ "SourceTier",
112
+ "Span",
113
+ "ValidationVerdict",
114
+ "Validator",
115
+ "ValueNormalizer",
116
+ "VerdictValidator",
117
+ "__version__",
118
+ ]
openodke/_text.py ADDED
@@ -0,0 +1,37 @@
1
+ """Offset-keeping text helpers shared by the loaders and the pattern extractor."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Iterator
7
+
8
+ _LINE = re.compile(r"[^\r\n]*(?:\r\n|\r|\n)?")
9
+
10
+
11
+ def iter_lines(text: str, start: int = 0, end: int | None = None) -> Iterator[tuple[int, int]]:
12
+ """`(start, end)` of every line in `text[start:end]`, line break excluded.
13
+
14
+ Not `str.splitlines`: that also breaks on form feeds and Unicode separators,
15
+ which a Markdown table or a `Key: value` block never does, and it returns
16
+ strings rather than the offsets a span needs.
17
+ """
18
+ stop = len(text) if end is None else end
19
+ pos = start
20
+ while pos < stop:
21
+ match = _LINE.match(text, pos, stop)
22
+ if match is None: # pragma: no cover - the pattern matches the empty string
23
+ break
24
+ line_end = match.end()
25
+ while line_end > pos and text[line_end - 1] in "\r\n":
26
+ line_end -= 1
27
+ yield pos, line_end
28
+ pos = match.end()
29
+
30
+
31
+ def trim(text: str, start: int, end: int) -> tuple[int, int]:
32
+ """Narrow `[start, end)` past whitespace on both sides, without copying."""
33
+ while start < end and text[start].isspace():
34
+ start += 1
35
+ while end > start and text[end - 1].isspace():
36
+ end -= 1
37
+ return start, end
openodke/chunking.py ADDED
@@ -0,0 +1,164 @@
1
+ """Splitting documents into chunks without losing a single offset.
2
+
3
+ DECISIONS #3 made spans character offsets and #19 made the chunk the unit of the
4
+ pipeline. Together they put the weight of provenance on this module: a chunk's
5
+ `text` must be exactly `doc.text[start:end]`, or a span an extractor finds inside
6
+ a chunk points at characters that were never there and the grounder checks a
7
+ quote against nothing.
8
+
9
+ So nothing here normalises. Paragraphs and sentences are found by scanning the
10
+ original string, a chunk is a pair of indices into it, and its text is sliced
11
+ rather than rebuilt. CRLF, tabs, runs of spaces and non-ASCII text survive
12
+ because they are never touched. Offsets are Python string indices — code points
13
+ — which is what `Span` and `Chunk` already mean.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import re
19
+ from collections.abc import Iterator
20
+ from dataclasses import dataclass
21
+
22
+ from openodke.types import Chunk, Document
23
+
24
+ # `\r(?!\n)`: without the lookahead one CRLF backtracks into two line breaks and
25
+ # every Windows line becomes a paragraph.
26
+ _LINE_BREAK = r"(?:\r\n|\r(?!\n)|\n)"
27
+ _PARAGRAPH_BREAK = re.compile(
28
+ rf"{_LINE_BREAK}(?:[^\S\r\n]*{_LINE_BREAK})+" "|\N{PARAGRAPH SEPARATOR}"
29
+ )
30
+ _CLOSERS = "\"'”’)\\]」』"
31
+ # Western terminators need whitespace after them ("3.14", "example.com" are not
32
+ # ends); CJK full stops do not, because CJK text has no spaces to wait for.
33
+ _SENTENCE_END = re.compile(rf"[.!?…]+[{_CLOSERS}]*(?=\s|\Z)|[。!?]+[{_CLOSERS}]*")
34
+ _WORD = re.compile(r"\S+")
35
+ # Only the ones that are nearly always followed by a capitalised name. A miss
36
+ # here merges two sentences, which is the safe direction: it never splits one.
37
+ _ABBREVIATIONS = frozenset(
38
+ {"mr", "mrs", "ms", "dr", "prof", "sr", "jr", "st", "vs", "e.g", "i.e", "cf", "fig", "approx"}
39
+ )
40
+
41
+
42
+ @dataclass(frozen=True, slots=True)
43
+ class _Sentence:
44
+ start: int
45
+ end: int
46
+ words: int
47
+ opens_paragraph: bool
48
+
49
+
50
+ class SentenceChunker:
51
+ """Packs whole sentences into chunks of at most `max_words` words.
52
+
53
+ Boundaries are chosen in order of preference: a paragraph break, then a
54
+ sentence break, and never anything smaller. A router asked "fact or
55
+ narrative?" about half a sentence is asked nothing (DECISIONS #19), so a
56
+ single sentence longer than the cap becomes a chunk of its own rather than
57
+ being cut — and a document with no punctuation at all is one sentence per
58
+ paragraph.
59
+
60
+ A chunk ends at the last paragraph break inside it, unless that would leave
61
+ it less than half full; a heading stranded as its own chunk is as useless to
62
+ a router as half a sentence. Chunks start and end on non-whitespace, so the
63
+ whitespace between two chunks belongs to neither.
64
+
65
+ `overlap` repeats that many trailing sentences at the start of the next
66
+ chunk, so a fact stated across a boundary is seen whole at least once. It is
67
+ shed first whenever it would crowd out new text. Overlapping chunks emit the
68
+ same fact twice; the corroborator merges them by signature.
69
+ """
70
+
71
+ def __init__(self, max_words: int = 200, *, overlap: int = 0) -> None:
72
+ if max_words < 1:
73
+ raise ValueError("max_words must be at least 1")
74
+ if overlap < 0:
75
+ raise ValueError("overlap cannot be negative")
76
+ self.max_words = max_words
77
+ self.overlap = overlap
78
+
79
+ def chunk(self, doc: Document) -> Iterator[Chunk]:
80
+ sentences = list(_segment(doc.text))
81
+ n = len(sentences)
82
+ prefix = [0]
83
+ for s in sentences:
84
+ prefix.append(prefix[-1] + s.words)
85
+
86
+ first = 0 # first sentence of the chunk being built
87
+ covered = 0 # sentences [0, covered) already emitted
88
+ index = 0
89
+ while covered < n:
90
+ end = first
91
+ while end < n and prefix[end + 1] - prefix[first] <= self.max_words:
92
+ end += 1
93
+ if end <= covered:
94
+ if first < covered:
95
+ # The carried-over overlap leaves no room for anything new.
96
+ first += 1
97
+ continue
98
+ # One sentence over the cap: it stands alone, uncut.
99
+ end = covered + 1
100
+
101
+ if end < n and not sentences[end].opens_paragraph:
102
+ for k in range(end - 1, covered, -1):
103
+ if sentences[k].opens_paragraph:
104
+ if 2 * (prefix[k] - prefix[first]) >= self.max_words:
105
+ end = k
106
+ break
107
+
108
+ start, stop = sentences[first].start, sentences[end - 1].end
109
+ yield Chunk(
110
+ doc_id=doc.id, start=start, end=stop, text=doc.text[start:stop], index=index
111
+ )
112
+ index += 1
113
+ covered = end
114
+ first = max(end - self.overlap, first + 1)
115
+
116
+
117
+ def _segment(text: str) -> Iterator[_Sentence]:
118
+ """Every sentence in `text`, in order, with whitespace trimmed from both ends."""
119
+ cursor = 0
120
+ for brk in [*_PARAGRAPH_BREAK.finditer(text), None]:
121
+ stop = len(text) if brk is None else brk.start()
122
+ start, end = _trim(text, cursor, stop)
123
+ opens = True
124
+ for s, e in _sentences_in(text, start, end):
125
+ yield _Sentence(s, e, len(_WORD.findall(text, s, e)), opens)
126
+ opens = False
127
+ if brk is not None:
128
+ cursor = brk.end()
129
+
130
+
131
+ def _sentences_in(text: str, start: int, end: int) -> Iterator[tuple[int, int]]:
132
+ cursor = start
133
+ for match in _SENTENCE_END.finditer(text, start, end):
134
+ if _is_abbreviation(text, cursor, match, end):
135
+ continue
136
+ yield cursor, match.end()
137
+ cursor, _ = _trim(text, match.end(), end)
138
+ if cursor < end:
139
+ yield cursor, end
140
+
141
+
142
+ def _is_abbreviation(text: str, sentence_start: int, match: re.Match[str], end: int) -> bool:
143
+ if match.group().rstrip(_CLOSERS) != ".":
144
+ return False
145
+ following, _ = _trim(text, match.end(), end)
146
+ if following < end and text[following].islower():
147
+ return True
148
+ token_start = match.start()
149
+ while token_start > sentence_start and not text[token_start - 1].isspace():
150
+ token_start -= 1
151
+ token = text[token_start : match.start()].lstrip(_CLOSERS + "(“‘")
152
+ # A single letter is an initial: "J. R. R. Tolkien".
153
+ return token.casefold() in _ABBREVIATIONS or (len(token) == 1 and token.isalpha())
154
+
155
+
156
+ def _trim(text: str, start: int, end: int) -> tuple[int, int]:
157
+ while start < end and text[start].isspace():
158
+ start += 1
159
+ while end > start and text[end - 1].isspace():
160
+ end -= 1
161
+ return start, end
162
+
163
+
164
+ __all__ = ["SentenceChunker"]
File without changes