python-corekit 0.3.0__py3-none-any.whl → 0.4.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- corekit/exceptionator/__init__.py +68 -0
- corekit/exceptionator/actions.py +46 -0
- corekit/exceptionator/asgi.py +115 -0
- corekit/exceptionator/constants.py +7 -0
- corekit/exceptionator/exceptionator.py +62 -0
- corekit/exceptionator/fingerprint.py +86 -0
- corekit/exceptionator/guard.py +53 -0
- corekit/exceptionator/jobs.py +52 -0
- corekit/exceptionator/models.py +98 -0
- corekit/http/__init__.py +5 -0
- corekit/http/client.py +34 -3
- corekit/http/stream.py +110 -0
- corekit/jobs/runner.py +22 -2
- corekit/llm/__init__.py +134 -0
- corekit/llm/client.py +179 -0
- corekit/llm/enum.py +123 -0
- corekit/llm/events.py +96 -0
- corekit/llm/messages.py +173 -0
- corekit/llm/prompts/__init__.py +19 -0
- corekit/llm/prompts/enum.py +54 -0
- corekit/llm/prompts/exceptions.py +22 -0
- corekit/llm/prompts/loader.py +139 -0
- corekit/llm/prompts/template.py +53 -0
- corekit/llm/protocols.py +65 -0
- corekit/llm/streaming.py +149 -0
- corekit/llm/tools/__init__.py +19 -0
- corekit/llm/tools/base.py +118 -0
- corekit/llm/tools/detection.py +99 -0
- corekit/llm/tools/loop.py +255 -0
- corekit/llm/tools/registry.py +103 -0
- corekit/llm/wire.py +199 -0
- corekit/observability/__init__.py +3 -3
- corekit/observability/benchmarkable.py +16 -2
- corekit/observability/timing/split.py +14 -0
- corekit/observability/timing/timer.py +31 -9
- corekit/schemas/__init__.py +2 -1
- corekit/schemas/version.py +58 -0
- corekit/utils/__init__.py +4 -2
- corekit/utils/collections.py +16 -1
- corekit/utils/text.py +21 -2
- {python_corekit-0.3.0.dist-info → python_corekit-0.4.1.dist-info}/METADATA +56 -4
- {python_corekit-0.3.0.dist-info → python_corekit-0.4.1.dist-info}/RECORD +45 -16
- {python_corekit-0.3.0.dist-info → python_corekit-0.4.1.dist-info}/WHEEL +0 -0
- {python_corekit-0.3.0.dist-info → python_corekit-0.4.1.dist-info}/licenses/LICENSE +0 -0
- {python_corekit-0.3.0.dist-info → python_corekit-0.4.1.dist-info}/top_level.txt +0 -0
corekit/llm/wire.py
ADDED
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Structured OpenAI completion-chunk parsing and tool-call accumulation.
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import logging
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from pydantic import BaseModel, Field
|
|
12
|
+
|
|
13
|
+
from corekit.llm.enum import WireField
|
|
14
|
+
from corekit.llm.events import UsageEvent
|
|
15
|
+
from corekit.llm.tools.base import ToolCall
|
|
16
|
+
from corekit.utils import safe_dict, safe_string
|
|
17
|
+
from corekit.utils.collections import attr_or_key
|
|
18
|
+
|
|
19
|
+
__all__ = [
|
|
20
|
+
"ChoiceDelta",
|
|
21
|
+
"CompletionTurn",
|
|
22
|
+
"FunctionCallDelta",
|
|
23
|
+
"ToolCallAccumulator",
|
|
24
|
+
"ToolCallDelta",
|
|
25
|
+
"UsageInfo",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger(__name__)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class FunctionCallDelta(BaseModel):
|
|
32
|
+
"""
|
|
33
|
+
Incremental ``function`` fields on a streamed tool-call delta.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
name: str = ""
|
|
37
|
+
arguments: str = ""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ToolCallDelta(BaseModel):
|
|
41
|
+
"""
|
|
42
|
+
One streamed tool-call fragment (may arrive across many chunks).
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
index: int = 0
|
|
46
|
+
id: str = ""
|
|
47
|
+
function: FunctionCallDelta = Field(default_factory=FunctionCallDelta)
|
|
48
|
+
|
|
49
|
+
@classmethod
|
|
50
|
+
def from_raw(cls, raw: Any) -> ToolCallDelta | None:
|
|
51
|
+
"""
|
|
52
|
+
Parse a dict or SDK object into a delta; return None if index is missing.
|
|
53
|
+
"""
|
|
54
|
+
index = attr_or_key(raw, WireField.INDEX)
|
|
55
|
+
if index is None:
|
|
56
|
+
return None
|
|
57
|
+
function_raw = safe_dict(attr_or_key(raw, WireField.FUNCTION))
|
|
58
|
+
return cls(
|
|
59
|
+
index=int(index),
|
|
60
|
+
id=safe_string(attr_or_key(raw, WireField.ID)),
|
|
61
|
+
function=FunctionCallDelta(
|
|
62
|
+
name=safe_string(attr_or_key(function_raw, WireField.NAME)),
|
|
63
|
+
arguments=safe_string(attr_or_key(function_raw, WireField.ARGUMENTS)),
|
|
64
|
+
),
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
class ToolCallAccumulator(BaseModel):
|
|
69
|
+
"""
|
|
70
|
+
Merges streamed tool-call deltas for a single index into one call.
|
|
71
|
+
"""
|
|
72
|
+
|
|
73
|
+
id: str = ""
|
|
74
|
+
name: str = ""
|
|
75
|
+
arguments: str = ""
|
|
76
|
+
|
|
77
|
+
def merge(self, delta: ToolCallDelta) -> None:
|
|
78
|
+
"""
|
|
79
|
+
Append name/argument fragments from one delta.
|
|
80
|
+
"""
|
|
81
|
+
if delta.id:
|
|
82
|
+
self.id = delta.id
|
|
83
|
+
if delta.function.name:
|
|
84
|
+
self.name += delta.function.name
|
|
85
|
+
if delta.function.arguments:
|
|
86
|
+
self.arguments += delta.function.arguments
|
|
87
|
+
|
|
88
|
+
def to_tool_call(self) -> ToolCall:
|
|
89
|
+
"""
|
|
90
|
+
Parse accumulated argument JSON into a ``ToolCall``.
|
|
91
|
+
"""
|
|
92
|
+
parsed: dict[str, Any] = {}
|
|
93
|
+
try:
|
|
94
|
+
parsed = json.loads(self.arguments)
|
|
95
|
+
except json.JSONDecodeError:
|
|
96
|
+
logger.warning("Bad tool args for %r: %r", self.name, self.arguments)
|
|
97
|
+
return ToolCall(id=self.id, name=self.name, arguments=safe_dict(parsed))
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
class ChoiceDelta(BaseModel):
|
|
101
|
+
"""
|
|
102
|
+
``choices[0].delta`` fields we care about from one chunk.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
content: str | None = None
|
|
106
|
+
reasoning_content: str | None = None
|
|
107
|
+
tool_calls: list[ToolCallDelta] = Field(default_factory=list)
|
|
108
|
+
finish_reason: str | None = None
|
|
109
|
+
|
|
110
|
+
@classmethod
|
|
111
|
+
def from_chunk(cls, chunk: Any) -> ChoiceDelta | None:
|
|
112
|
+
"""
|
|
113
|
+
Extract the first choice's delta (and finish_reason) from a chunk.
|
|
114
|
+
"""
|
|
115
|
+
choices = attr_or_key(chunk, WireField.CHOICES)
|
|
116
|
+
if not choices:
|
|
117
|
+
return None
|
|
118
|
+
choice = choices[0]
|
|
119
|
+
delta_raw = attr_or_key(choice, WireField.DELTA)
|
|
120
|
+
if delta_raw is None:
|
|
121
|
+
return cls(finish_reason=attr_or_key(choice, WireField.FINISH_REASON))
|
|
122
|
+
|
|
123
|
+
tool_calls: list[ToolCallDelta] = []
|
|
124
|
+
for raw_tc in attr_or_key(delta_raw, WireField.TOOL_CALLS) or []:
|
|
125
|
+
parsed = ToolCallDelta.from_raw(raw_tc)
|
|
126
|
+
if parsed is not None:
|
|
127
|
+
tool_calls.append(parsed)
|
|
128
|
+
|
|
129
|
+
return cls(
|
|
130
|
+
content=attr_or_key(delta_raw, WireField.CONTENT),
|
|
131
|
+
reasoning_content=attr_or_key(delta_raw, WireField.REASONING_CONTENT),
|
|
132
|
+
tool_calls=tool_calls,
|
|
133
|
+
finish_reason=attr_or_key(choice, WireField.FINISH_REASON),
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
class UsageInfo(BaseModel):
|
|
138
|
+
"""
|
|
139
|
+
Token usage from a completion chunk or final response.
|
|
140
|
+
"""
|
|
141
|
+
|
|
142
|
+
prompt_tokens: int | None = None
|
|
143
|
+
completion_tokens: int | None = None
|
|
144
|
+
total_tokens: int | None = None
|
|
145
|
+
|
|
146
|
+
@classmethod
|
|
147
|
+
def from_chunk(cls, chunk: Any) -> UsageInfo | None:
|
|
148
|
+
"""
|
|
149
|
+
Read ``usage`` from a chunk when present.
|
|
150
|
+
"""
|
|
151
|
+
usage = attr_or_key(chunk, WireField.USAGE)
|
|
152
|
+
if usage is None:
|
|
153
|
+
return None
|
|
154
|
+
return cls(
|
|
155
|
+
prompt_tokens=attr_or_key(usage, WireField.PROMPT_TOKENS),
|
|
156
|
+
completion_tokens=attr_or_key(usage, WireField.COMPLETION_TOKENS),
|
|
157
|
+
total_tokens=attr_or_key(usage, WireField.TOTAL_TOKENS),
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
def to_event(self) -> UsageEvent:
|
|
161
|
+
"""
|
|
162
|
+
Convert to a stream ``UsageEvent``.
|
|
163
|
+
"""
|
|
164
|
+
return UsageEvent(
|
|
165
|
+
total_tokens=self.total_tokens,
|
|
166
|
+
prompt_tokens=self.prompt_tokens,
|
|
167
|
+
completion_tokens=self.completion_tokens,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
class CompletionTurn(BaseModel):
|
|
172
|
+
"""
|
|
173
|
+
Mutable state accumulated while streaming one ``complete()`` call.
|
|
174
|
+
"""
|
|
175
|
+
|
|
176
|
+
tool_calls: dict[int, ToolCallAccumulator] = Field(default_factory=dict)
|
|
177
|
+
usage: UsageInfo | None = None
|
|
178
|
+
finish_reason: str | None = None
|
|
179
|
+
first_token_ms: float | None = None
|
|
180
|
+
|
|
181
|
+
def absorb_delta(self, delta: ChoiceDelta) -> None:
|
|
182
|
+
"""
|
|
183
|
+
Merge one choice delta into this turn.
|
|
184
|
+
"""
|
|
185
|
+
if delta.finish_reason:
|
|
186
|
+
self.finish_reason = delta.finish_reason
|
|
187
|
+
for tool_delta in delta.tool_calls:
|
|
188
|
+
bucket = self.tool_calls.setdefault(tool_delta.index, ToolCallAccumulator())
|
|
189
|
+
bucket.merge(tool_delta)
|
|
190
|
+
|
|
191
|
+
def absorb_usage(self, usage: UsageInfo | None) -> None:
|
|
192
|
+
if usage is not None:
|
|
193
|
+
self.usage = usage
|
|
194
|
+
|
|
195
|
+
def parsed_tool_calls(self) -> list[ToolCall]:
|
|
196
|
+
"""
|
|
197
|
+
Finalize accumulated tool-call indices into ``ToolCall`` values.
|
|
198
|
+
"""
|
|
199
|
+
return [self.tool_calls[idx].to_tool_call() for idx in sorted(self.tool_calls)]
|
|
@@ -8,10 +8,10 @@ are the same concern -- knowing what a running system is doing.
|
|
|
8
8
|
|
|
9
9
|
class Importer(Benchmarkable):
|
|
10
10
|
def run(self) -> None:
|
|
11
|
-
self.
|
|
12
|
-
self.
|
|
11
|
+
self.reset_timing()
|
|
12
|
+
self.timing("started")
|
|
13
13
|
...
|
|
14
|
-
self.timing("finished")
|
|
14
|
+
self.timing("finished") # total since reset; latest since started
|
|
15
15
|
"""
|
|
16
16
|
|
|
17
17
|
from corekit.observability.benchmarkable import Benchmarkable
|
|
@@ -18,12 +18,26 @@ class Benchmarkable(Loggable):
|
|
|
18
18
|
super().__init__(*args, **kwargs)
|
|
19
19
|
self._timing = Timer()
|
|
20
20
|
|
|
21
|
+
def reset_timing(self) -> None:
|
|
22
|
+
"""
|
|
23
|
+
Restart the underlying timer for a new unit of work.
|
|
24
|
+
"""
|
|
25
|
+
self._timing.reset()
|
|
26
|
+
|
|
27
|
+
def elapsed_ms(self) -> float:
|
|
28
|
+
"""
|
|
29
|
+
Milliseconds since start (or last ``reset_timing``), without a split.
|
|
30
|
+
"""
|
|
31
|
+
return self._timing.elapsed() * 1000
|
|
32
|
+
|
|
21
33
|
def timing(self, split_name: str | None = None) -> Split:
|
|
22
34
|
"""
|
|
23
35
|
Log the time since the last split and return it.
|
|
24
36
|
|
|
25
|
-
|
|
26
|
-
|
|
37
|
+
``Split.total`` / ``Split.total_ms`` are since start (or last reset).
|
|
38
|
+
``Split.latest`` / ``Split.latest_ms`` are since the previous split.
|
|
39
|
+
The log line is unchanged. The return value is for a caller that wants
|
|
40
|
+
the numbers without parsing that line.
|
|
27
41
|
"""
|
|
28
42
|
split = self._timing.split(split_name=split_name)
|
|
29
43
|
self.info(str(split))
|
|
@@ -8,6 +8,20 @@ class Split(BaseModel):
|
|
|
8
8
|
name: str | None = None
|
|
9
9
|
split_name: str | None = None
|
|
10
10
|
|
|
11
|
+
@property
|
|
12
|
+
def total_ms(self) -> float:
|
|
13
|
+
"""
|
|
14
|
+
``total`` in milliseconds.
|
|
15
|
+
"""
|
|
16
|
+
return self.total * 1000
|
|
17
|
+
|
|
18
|
+
@property
|
|
19
|
+
def latest_ms(self) -> float:
|
|
20
|
+
"""
|
|
21
|
+
``latest`` in milliseconds.
|
|
22
|
+
"""
|
|
23
|
+
return self.latest * 1000
|
|
24
|
+
|
|
11
25
|
def __str__(self) -> str:
|
|
12
26
|
identifier = ""
|
|
13
27
|
if self.name:
|
|
@@ -5,28 +5,50 @@ from corekit.observability.timing.split import Split
|
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
class Timer:
|
|
8
|
+
start: float
|
|
9
|
+
latest: float
|
|
10
|
+
num: int
|
|
11
|
+
precision: int
|
|
12
|
+
name: str
|
|
13
|
+
|
|
8
14
|
def __init__(self, precision: int = DEFAULT_PRECISION) -> None:
|
|
9
15
|
# perf_counter is monotonic. time.time() can step backwards, which
|
|
10
16
|
# makes a split look negative for no reason the caller can act on.
|
|
11
|
-
_time = time.perf_counter()
|
|
12
|
-
self.start = _time
|
|
13
|
-
self.latest = _time
|
|
14
|
-
self.num = 0
|
|
15
|
-
self.name = type(self).__name__
|
|
16
17
|
self.precision = precision if precision > 0 else DEFAULT_PRECISION
|
|
18
|
+
self.name = type(self).__name__
|
|
19
|
+
self.reset()
|
|
20
|
+
|
|
21
|
+
def reset(self) -> None:
|
|
22
|
+
"""
|
|
23
|
+
Restart the clock and clear split numbering.
|
|
24
|
+
|
|
25
|
+
Use at the start of a new timed unit of work (a request, a model turn,
|
|
26
|
+
a pipeline stage) so ``total`` measures that unit rather than the
|
|
27
|
+
object's whole lifetime.
|
|
28
|
+
"""
|
|
29
|
+
now = time.perf_counter()
|
|
30
|
+
self.start = now
|
|
31
|
+
self.latest = now
|
|
32
|
+
self.num = 0
|
|
17
33
|
|
|
18
34
|
def _round(self, value: float) -> float:
|
|
19
35
|
return round(value, self.precision)
|
|
20
36
|
|
|
37
|
+
def elapsed(self) -> float:
|
|
38
|
+
"""
|
|
39
|
+
Seconds since start (or last ``reset``), without recording a split.
|
|
40
|
+
"""
|
|
41
|
+
return self._round(time.perf_counter() - self.start)
|
|
42
|
+
|
|
21
43
|
def split(self, split_name: str | None = None) -> Split:
|
|
22
|
-
|
|
44
|
+
now = time.perf_counter()
|
|
23
45
|
self.num += 1
|
|
24
46
|
split = Split(
|
|
25
47
|
num=self.num,
|
|
26
|
-
total=self._round(
|
|
27
|
-
latest=self._round(
|
|
48
|
+
total=self._round(now - self.start),
|
|
49
|
+
latest=self._round(now - self.latest),
|
|
28
50
|
name=self.name,
|
|
29
51
|
split_name=split_name,
|
|
30
52
|
)
|
|
31
|
-
self.latest =
|
|
53
|
+
self.latest = now
|
|
32
54
|
return split
|
corekit/schemas/__init__.py
CHANGED
|
@@ -6,5 +6,6 @@ raises where it is introduced rather than somewhere further along.
|
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
8
|
from corekit.schemas.enum import IntegerEnum, StringEnum, ValidatingEnum
|
|
9
|
+
from corekit.schemas.version import SemanticVersion
|
|
9
10
|
|
|
10
|
-
__all__ = ["IntegerEnum", "StringEnum", "ValidatingEnum"]
|
|
11
|
+
__all__ = ["IntegerEnum", "SemanticVersion", "StringEnum", "ValidatingEnum"]
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Semantic version (``major.minor.patch`` with optional ``-extra``).
|
|
3
|
+
"""
|
|
4
|
+
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
from typing import Self
|
|
9
|
+
|
|
10
|
+
from pydantic import BaseModel, Field
|
|
11
|
+
|
|
12
|
+
from corekit.utils import safe_int, safe_string
|
|
13
|
+
|
|
14
|
+
__all__ = ["SemanticVersion"]
|
|
15
|
+
|
|
16
|
+
_VERSION_RE = re.compile(r"^(?P<major>\d+)(?:\.(?P<minor>\d+)(?:\.(?P<patch>\d+)?)?)?(?:-(?P<extra>[A-Za-z0-9._-]+))?$")
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class SemanticVersion(BaseModel):
|
|
20
|
+
"""
|
|
21
|
+
Structured ``major.minor.patch`` with an optional ``-extra`` suffix.
|
|
22
|
+
|
|
23
|
+
Suitable for filenames, config pins, and other places that need a parsed
|
|
24
|
+
SemVer-like value without tying to packaging metadata.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
major: int = Field(ge=0)
|
|
28
|
+
minor: int = Field(default=0, ge=0)
|
|
29
|
+
patch: int = Field(default=0, ge=0)
|
|
30
|
+
extra: str | None = None
|
|
31
|
+
|
|
32
|
+
def __str__(self) -> str:
|
|
33
|
+
version = f"{self.major}.{self.minor}.{self.patch}"
|
|
34
|
+
if self.extra:
|
|
35
|
+
return f"{version}-{self.extra}"
|
|
36
|
+
return version
|
|
37
|
+
|
|
38
|
+
def __repr__(self) -> str:
|
|
39
|
+
return f"SemanticVersion({self})"
|
|
40
|
+
|
|
41
|
+
@classmethod
|
|
42
|
+
def parse(cls, value: SemanticVersion | str | int) -> Self:
|
|
43
|
+
"""
|
|
44
|
+
Accept a version object, ``\"1.0.0\"``, or a bare major ``1`` (→ ``1.0.0``).
|
|
45
|
+
"""
|
|
46
|
+
if isinstance(value, SemanticVersion):
|
|
47
|
+
return value
|
|
48
|
+
if isinstance(value, int):
|
|
49
|
+
return cls(major=value)
|
|
50
|
+
match = _VERSION_RE.fullmatch(safe_string(value))
|
|
51
|
+
if match is None:
|
|
52
|
+
raise ValueError(f"Invalid semantic version: {value!r}")
|
|
53
|
+
return cls(
|
|
54
|
+
major=safe_int(match.group("major")),
|
|
55
|
+
minor=safe_int(match.group("minor")),
|
|
56
|
+
patch=safe_int(match.group("patch")),
|
|
57
|
+
extra=match.group("extra"),
|
|
58
|
+
)
|
corekit/utils/__init__.py
CHANGED
|
@@ -9,7 +9,7 @@ configuration need.
|
|
|
9
9
|
"""
|
|
10
10
|
|
|
11
11
|
from corekit.utils.coercion import safe_dict, safe_float, safe_int, safe_list, safe_string, safe_tuple
|
|
12
|
-
from corekit.utils.collections import UNSET, MultiMatch, keygetter, repeated_get, split_list
|
|
12
|
+
from corekit.utils.collections import UNSET, MultiMatch, attr_or_key, keygetter, repeated_get, split_list
|
|
13
13
|
from corekit.utils.ids import (
|
|
14
14
|
SHORTCODE_CHARS,
|
|
15
15
|
generate_session_token,
|
|
@@ -19,7 +19,7 @@ from corekit.utils.ids import (
|
|
|
19
19
|
)
|
|
20
20
|
from corekit.utils.payload import Payload, decode_payload, encode_payload
|
|
21
21
|
from corekit.utils.raise_exc import raise_exc
|
|
22
|
-
from corekit.utils.text import join_lines, list_to_english, sanitize_filename
|
|
22
|
+
from corekit.utils.text import join_lines, list_to_english, sanitize_filename, truncate
|
|
23
23
|
from corekit.utils.time import isoformat_now, parse_docker_timestamp, time_now, timedelta_now, timestamp_now
|
|
24
24
|
from corekit.utils.validators import false_validator, true_validator
|
|
25
25
|
from corekit.utils.void import void
|
|
@@ -29,6 +29,7 @@ __all__ = [
|
|
|
29
29
|
"UNSET",
|
|
30
30
|
"MultiMatch",
|
|
31
31
|
"Payload",
|
|
32
|
+
"attr_or_key",
|
|
32
33
|
"decode_payload",
|
|
33
34
|
"encode_payload",
|
|
34
35
|
"false_validator",
|
|
@@ -55,5 +56,6 @@ __all__ = [
|
|
|
55
56
|
"timedelta_now",
|
|
56
57
|
"timestamp_now",
|
|
57
58
|
"true_validator",
|
|
59
|
+
"truncate",
|
|
58
60
|
"void",
|
|
59
61
|
]
|
corekit/utils/collections.py
CHANGED
|
@@ -10,7 +10,7 @@ missing or the wrong type.
|
|
|
10
10
|
from copy import deepcopy
|
|
11
11
|
from typing import Any, Callable
|
|
12
12
|
|
|
13
|
-
__all__ = ["MultiMatch", "keygetter", "repeated_get", "split_list"]
|
|
13
|
+
__all__ = ["MultiMatch", "attr_or_key", "keygetter", "repeated_get", "split_list"]
|
|
14
14
|
|
|
15
15
|
KEY_SEPARATOR = "."
|
|
16
16
|
|
|
@@ -25,6 +25,21 @@ def split_list(items: list[Any], index: int) -> tuple[list[Any], list[Any]]:
|
|
|
25
25
|
return items[:index], items[index:]
|
|
26
26
|
|
|
27
27
|
|
|
28
|
+
def attr_or_key(obj: Any, name: str) -> Any:
|
|
29
|
+
"""
|
|
30
|
+
Read ``name`` from a mapping or an object attribute.
|
|
31
|
+
|
|
32
|
+
Returns ``None`` when ``obj`` is ``None``, the key is missing, or the
|
|
33
|
+
attribute is absent. Useful for SDK payloads that arrive as either dicts
|
|
34
|
+
or attribute-bearing chunk objects.
|
|
35
|
+
"""
|
|
36
|
+
if obj is None:
|
|
37
|
+
return None
|
|
38
|
+
if isinstance(obj, dict):
|
|
39
|
+
return obj.get(name)
|
|
40
|
+
return getattr(obj, name, None)
|
|
41
|
+
|
|
42
|
+
|
|
28
43
|
class MultiMatch(dict[str, Any]):
|
|
29
44
|
"""
|
|
30
45
|
The result of a path step that matched more than one key.
|
corekit/utils/text.py
CHANGED
|
@@ -2,13 +2,14 @@
|
|
|
2
2
|
String assembly helpers.
|
|
3
3
|
|
|
4
4
|
Small formatters for turning collections into text a human will read -- log
|
|
5
|
-
lines, notification bodies, error messages -- plus filename sanitization
|
|
5
|
+
lines, notification bodies, error messages -- plus filename sanitization and
|
|
6
|
+
length limits.
|
|
6
7
|
"""
|
|
7
8
|
|
|
8
9
|
import re
|
|
9
10
|
from typing import Any, Literal
|
|
10
11
|
|
|
11
|
-
__all__ = ["join_lines", "list_to_english", "sanitize_filename"]
|
|
12
|
+
__all__ = ["join_lines", "list_to_english", "sanitize_filename", "truncate"]
|
|
12
13
|
|
|
13
14
|
FALLBACK_FILENAME = "unnamed"
|
|
14
15
|
_UNSAFE_CHARS = re.compile(r"[^a-z0-9\-]")
|
|
@@ -43,6 +44,24 @@ def list_to_english(items: list[str], ending: Literal["and", "or"] | None = None
|
|
|
43
44
|
return f"{buffer}, {ending} {last}"
|
|
44
45
|
|
|
45
46
|
|
|
47
|
+
def truncate(text: str, limit: int, *, suffix: str = "") -> str:
|
|
48
|
+
"""
|
|
49
|
+
Cut ``text`` to at most ``limit`` characters, appending ``suffix`` when cut.
|
|
50
|
+
|
|
51
|
+
``suffix`` counts toward ``limit``, so the returned string is never longer
|
|
52
|
+
than ``limit`` (unless ``suffix`` itself is longer than ``limit``, in which
|
|
53
|
+
case ``suffix`` is returned truncated to ``limit``).
|
|
54
|
+
"""
|
|
55
|
+
if limit <= 0:
|
|
56
|
+
return ""
|
|
57
|
+
if len(text) <= limit:
|
|
58
|
+
return text
|
|
59
|
+
if len(suffix) >= limit:
|
|
60
|
+
return suffix[:limit]
|
|
61
|
+
keep = limit - len(suffix)
|
|
62
|
+
return text[:keep] + suffix
|
|
63
|
+
|
|
64
|
+
|
|
46
65
|
def sanitize_filename(raw: str, fallback: str = FALLBACK_FILENAME) -> str:
|
|
47
66
|
"""
|
|
48
67
|
Reduce a string to lowercase kebab-case safe for use as a filename stem.
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: python-corekit
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: Shared foundations for Python projects: logging, benchmarking, registries, FastAPI application and routers, SQL statements and migrations, background tasks, and
|
|
3
|
+
Version: 0.4.1
|
|
4
|
+
Summary: Shared foundations for Python projects: logging, benchmarking, registries, FastAPI application and routers, SQL statements and migrations, background tasks, ETL, and LLM chat/tool primitives
|
|
5
5
|
Author: Steven Jacobsen
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/stevejaker/corekit
|
|
@@ -18,6 +18,7 @@ License-File: LICENSE
|
|
|
18
18
|
Requires-Dist: pydantic<3,>=2.10
|
|
19
19
|
Requires-Dist: pydantic-settings<3,>=2.0
|
|
20
20
|
Requires-Dist: fastapi<1,>=0.115
|
|
21
|
+
Requires-Dist: starlette<0.47,>=0.40
|
|
21
22
|
Requires-Dist: sqlmodel<0.1,>=0.0.16
|
|
22
23
|
Requires-Dist: SQLAlchemy<3,>=2.0
|
|
23
24
|
Requires-Dist: redis<7,>=5.0
|
|
@@ -37,7 +38,8 @@ Dynamic: license-file
|
|
|
37
38
|
|
|
38
39
|
Shared foundations for Python projects: structured logging, benchmarking,
|
|
39
40
|
registries, a FastAPI application with routers and handlers, SQL statements and
|
|
40
|
-
migrations, background tasks, an in-memory record store,
|
|
41
|
+
migrations, background tasks, an in-memory record store, ETL scaffolding, and
|
|
42
|
+
OpenAI-shaped chat / tool-loop primitives.
|
|
41
43
|
|
|
42
44
|
Requires Python 3.11+.
|
|
43
45
|
|
|
@@ -55,7 +57,7 @@ extras to remember, and no import that fails because something was left out.
|
|
|
55
57
|
Pin a compatible release rather than tracking whatever is newest:
|
|
56
58
|
|
|
57
59
|
```
|
|
58
|
-
python-corekit~=0.
|
|
60
|
+
python-corekit~=0.4.0
|
|
59
61
|
```
|
|
60
62
|
|
|
61
63
|
Before 1.0, the minor version carries breaking changes.
|
|
@@ -88,6 +90,29 @@ class Report(Benchmarkable):
|
|
|
88
90
|
self.timing("queried") # logs the time since the previous split
|
|
89
91
|
```
|
|
90
92
|
|
|
93
|
+
## Exceptionator
|
|
94
|
+
|
|
95
|
+
Catch an exception, run one action, then rethrow so the existing failure path
|
|
96
|
+
still happens. Built-in actions are no-op and log-traceback; apps supply their
|
|
97
|
+
own (persist, notify, …).
|
|
98
|
+
|
|
99
|
+
```python
|
|
100
|
+
from corekit.exceptionator import (
|
|
101
|
+
Exceptionator,
|
|
102
|
+
ExceptionatorMiddleware,
|
|
103
|
+
LogTracebackAction,
|
|
104
|
+
install_run_task_hook,
|
|
105
|
+
set_process_exceptionator,
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
exceptionator = Exceptionator(LogTracebackAction(), service="api")
|
|
109
|
+
set_process_exceptionator(exceptionator)
|
|
110
|
+
install_run_task_hook(exceptionator) # optional; run_task captures with task name
|
|
111
|
+
```
|
|
112
|
+
|
|
113
|
+
`ExceptionatorMiddleware` is raw ASGI (not `BaseHTTPMiddleware`). With no
|
|
114
|
+
`run_task` hook installed, jobs behave as in 0.4.0.
|
|
115
|
+
|
|
91
116
|
## Assembling an application
|
|
92
117
|
|
|
93
118
|
`Application` is a plain `FastAPI` subclass — every constructor argument,
|
|
@@ -281,6 +306,32 @@ ceiling, not 9,999 threads.
|
|
|
281
306
|
Failures propagate by default. Pass `raise_on_error=False` to log and skip them
|
|
282
307
|
instead, which loses results silently and so is opt-in.
|
|
283
308
|
|
|
309
|
+
## LLM chat and tools
|
|
310
|
+
|
|
311
|
+
OpenAI-shaped messages, a tool registry, an agentic loop, and a thin
|
|
312
|
+
``ChatCompletionsClient`` (a ``BaseApiClient``) for any OpenAI-compatible
|
|
313
|
+
server — llama.cpp, vLLM, OpenAI, and so on. Not a multi-provider generation
|
|
314
|
+
facade.
|
|
315
|
+
|
|
316
|
+
```python
|
|
317
|
+
from corekit.llm import ChatCompletionsClient, PromptLoader, ToolLoop, ToolRegistry
|
|
318
|
+
|
|
319
|
+
client = ChatCompletionsClient("http://localhost:8080/v1", model="local")
|
|
320
|
+
prompts = PromptLoader("/path/to/prompts") # mount any compatible tree
|
|
321
|
+
tmpl = prompts.load("summarization", "general", system="1.0.0", user="1.0.0")
|
|
322
|
+
|
|
323
|
+
registry = ToolRegistry()
|
|
324
|
+
registry.register(MyTool())
|
|
325
|
+
loop = ToolLoop(client, registry)
|
|
326
|
+
async for event in loop.stream(tmpl.messages(text="…", style="brief", length="short")):
|
|
327
|
+
...
|
|
328
|
+
```
|
|
329
|
+
|
|
330
|
+
``PromptLoader`` reads versioned ``.txt`` files from a directory you point at
|
|
331
|
+
(``{kind}/{name}/{role}-v{version}.txt``, plus ``common/`` fragments). Prompt
|
|
332
|
+
*text* stays with the consumer — corekit only ships the loader. See
|
|
333
|
+
`docs/LLM_EXTRACTION.md`.
|
|
334
|
+
|
|
284
335
|
## HTTP clients
|
|
285
336
|
|
|
286
337
|
```python
|
|
@@ -394,6 +445,7 @@ corekit/
|
|
|
394
445
|
|
|
395
446
|
api/ Application, lifespan, middleware, routers
|
|
396
447
|
docker/ notifications/ etl/
|
|
448
|
+
llm/ messages, tools, agentic loop (no GenerationClient)
|
|
397
449
|
|
|
398
450
|
events/ log_monitor/ built on the capabilities above
|
|
399
451
|
```
|