crystil 0.8.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
crystil/__init__.py ADDED
@@ -0,0 +1,102 @@
1
+ r"""
2
+ ___ _ _ _
3
+ / __|_ _ _ _ __| |_(_) |
4
+ | (__| '_| || (_-< _| | |
5
+ \___|_| \_, /__/\__|_|_|AI
6
+ |__/ 07312025 / optimus codex
7
+ """
8
+
9
+ import os
10
+ from uuid import uuid4
11
+
12
+ from crystil._config import Config
13
+ from crystil._errors import CrystilRequestInterceptedError
14
+ from crystil._providers import Anthropic as LlmProviderAnthropic
15
+ from crystil._providers import Google as LlmProviderGoogle
16
+ from crystil._providers import Groq as LlmProviderGroq
17
+ from crystil._providers import LangChain as LlmProviderLangChain
18
+ from crystil._providers import OpenAi as LlmProviderOpenAi
19
+ from crystil._providers import PydanticAi as LlmProviderPydanticAi
20
+ from crystil._sentinel import Sentinel
21
+ from crystil.api._workflow import Workflow, Workflows
22
+
23
+ __all__ = ["Crystil", "CrystilRequestInterceptedError"]
24
+
25
+
26
+ class Crystil:
27
+ def __init__(self, api_key=None):
28
+ if api_key is None:
29
+ api_key = os.environ.get("CRYSTIL_API_KEY", None)
30
+
31
+ if api_key is None:
32
+ raise RuntimeError(
33
+ "API key is missing. Either set the CRYSTIL_API_KEY environment "
34
+ + "variable or set the api_key parameter when instantiating Crystil."
35
+ )
36
+
37
+ self.config = Config()
38
+ self.config.api_key = api_key
39
+ self.config.tx_uuid = uuid4()
40
+ self.sentinel = Sentinel(self.config)
41
+
42
+ self.anthropic = LlmProviderAnthropic(self)
43
+ self.google = LlmProviderGoogle(self)
44
+ self.groq = LlmProviderGroq(self)
45
+ self.langchain = LlmProviderLangChain(self)
46
+ self.openai = LlmProviderOpenAi(self)
47
+ self.pydantic_ai = LlmProviderPydanticAi(self)
48
+
49
+ self.workflow = Workflow(self.config)
50
+ self.workflows = Workflows(self.config)
51
+
52
+ def attribution(
53
+ self,
54
+ parent_id=None,
55
+ parent_name=None,
56
+ subsidiary_id=None,
57
+ subsidiary_name=None,
58
+ # -- Deprecated parameters! They are here for backwards compatibility only.
59
+ parent_uuid=None,
60
+ subsidiary_uuid=None,
61
+ ):
62
+ if parent_id is None:
63
+ raise RuntimeError("a string parent_id is required")
64
+
65
+ parent_id = str(parent_id)
66
+
67
+ if len(parent_id) > 100:
68
+ raise RuntimeError("parent_id cannot be greater than 100 characters")
69
+
70
+ if parent_name is not None and len(parent_name) > 100:
71
+ raise RuntimeError("parent_name cannot be greater than 100 characters")
72
+
73
+ if subsidiary_name is not None and subsidiary_id is None:
74
+ raise RuntimeError(
75
+ "a string subsidiary_id is required if a subsidiary_name is provided"
76
+ )
77
+
78
+ if subsidiary_id is not None:
79
+ subsidiary_id = str(subsidiary_id)
80
+
81
+ if len(subsidiary_id) > 100:
82
+ raise RuntimeError(
83
+ "subsidiary_id cannot be greater than 100 characters"
84
+ )
85
+
86
+ if subsidiary_name is not None and len(subsidiary_name) > 100:
87
+ raise RuntimeError("subsidiary_name cannot be greater than 100 characters")
88
+
89
+ subsidiary = None
90
+ if subsidiary_id is not None:
91
+ subsidiary = {"id": subsidiary_id, "name": subsidiary_name}
92
+
93
+ self.config.attribution = {
94
+ "parent": {"id": parent_id, "name": parent_name},
95
+ "subsidiary": subsidiary,
96
+ }
97
+
98
+ return self
99
+
100
+ def new_transaction(self):
101
+ self.config.tx_uuid = uuid4()
102
+ return self
crystil/_base.py ADDED
@@ -0,0 +1,443 @@
1
+ r"""
2
+ ___ _ _ _
3
+ / __|_ _ _ _ __| |_(_) |
4
+ | (__| '_| || (_-< _| | |
5
+ \___|_| \_, /__/\__|_|_|AI
6
+ |__/ 07312025 / optimus codex
7
+ """
8
+
9
+ import base64
10
+ import copy
11
+ import json
12
+ from typing import Any, Callable, Optional
13
+
14
+ from google.protobuf import json_format
15
+
16
+ from crystil._config import Config
17
+ from crystil._constants import (
18
+ LANGCHAIN_CHATBEDROCK_CLIENT_TITLE,
19
+ LANGCHAIN_CLIENT_PROVIDER,
20
+ )
21
+ from crystil._errors import CrystilRequestInterceptedError
22
+ from crystil._network import Api
23
+ from crystil._utils import merge_chunk
24
+
25
+ # These are attributes that are in ProtoBuf, but since they
26
+ # are recursive they cause issues making into JSON. Also, they
27
+ # are not part of the data we care about.
28
+ PROTOBUF_SKIP_ATTRS = frozenset(
29
+ [
30
+ "_client",
31
+ "_http",
32
+ "_session",
33
+ "_transport",
34
+ "__objclass__", # Descriptor metadata
35
+ "__doc__", # Documentation
36
+ "_member_map_", # Enum metadata
37
+ "_value2member_map_", # Enum metadata
38
+ "_member_names_", # Enum metadata
39
+ ]
40
+ )
41
+
42
+
43
+ class BaseClient:
44
+ def __init__(self, config: Config):
45
+ self.config = config
46
+ self.stream = False
47
+
48
+
49
+ class BaseInvoke:
50
+ def __init__(self, config: Config, method):
51
+ self.config = config
52
+ self._method = method
53
+ self._client_provider = None
54
+ self._client_title = None
55
+ self._client_version = None
56
+ self._client_title_resolver: Optional[Callable[[dict[str, Any]], str]] = None
57
+ self._uses_protobuf = False
58
+ self._include_stream_options = False
59
+ self._stream_options_via_extra_body: bool = False
60
+ self._stream_response_extractor = None
61
+
62
+ def client_is_bedrock(self):
63
+ return (
64
+ self._client_provider == LANGCHAIN_CLIENT_PROVIDER
65
+ and self._client_title == LANGCHAIN_CHATBEDROCK_CLIENT_TITLE
66
+ )
67
+
68
+ def configure_for_streaming_usage(self, kwargs):
69
+ if self._include_stream_options:
70
+ if kwargs.get("stream", None) == True:
71
+ if self._stream_options_via_extra_body:
72
+ # Groq's Python SDK doesn't expose `stream_options` as a
73
+ # top-level kwarg (the function signature rejects it with
74
+ # TypeError), but its underlying HTTP API supports it.
75
+ # `extra_body` is the SDK's documented escape hatch for
76
+ # arbitrary additional request-body fields.
77
+ extra_body = kwargs.get("extra_body", None)
78
+ if extra_body is None or not isinstance(extra_body, dict):
79
+ kwargs["extra_body"] = {}
80
+ stream_options = kwargs["extra_body"].get("stream_options", None)
81
+ if stream_options is None or not isinstance(stream_options, dict):
82
+ kwargs["extra_body"]["stream_options"] = {}
83
+ kwargs["extra_body"]["stream_options"]["include_usage"] = True
84
+ else:
85
+ stream_options = kwargs.get("stream_options", None)
86
+ if stream_options is None or not isinstance(
87
+ kwargs["stream_options"], dict
88
+ ):
89
+ kwargs["stream_options"] = {}
90
+
91
+ kwargs["stream_options"]["include_usage"] = True
92
+
93
+ return kwargs
94
+
95
+ def dict_to_json(self, dict_):
96
+ result = {}
97
+ for key, value in dict_.items():
98
+ if key not in PROTOBUF_SKIP_ATTRS:
99
+ if isinstance(value, list):
100
+ result[key] = self.list_to_json(value)
101
+ elif isinstance(value, dict):
102
+ result[key] = self.dict_to_json(value)
103
+ else:
104
+ if hasattr(value, "__dict__"):
105
+ result[key] = self.dict_to_json(value.__dict__)
106
+ elif isinstance(value, bytes):
107
+ # If left as bytes, the payload sent to Collector won't work.
108
+ # Gemini 3.0 passes a though_signature field is binary.
109
+ result[key] = base64.b64encode(value).decode("utf-8")
110
+ else:
111
+ result[key] = value
112
+
113
+ return result
114
+
115
+ def _format_kwargs(self, kwargs):
116
+ formatted_kwargs = None
117
+
118
+ if self._uses_protobuf:
119
+ request = kwargs.get("request")
120
+ if request:
121
+ if kwargs["request"].__dict__ and kwargs["request"].__dict__.get("_pb"):
122
+ # The "_pb" is for handling the protobuf structure. It is used
123
+ # in the LangChain Google code paths.
124
+ formatted_kwargs = json.loads(
125
+ json_format.MessageToJson(kwargs["request"].__dict__["_pb"])
126
+ )
127
+ else:
128
+ formatted_kwargs = self.dict_to_json(copy.deepcopy(kwargs))
129
+ else:
130
+ formatted_kwargs = copy.deepcopy(kwargs)
131
+ if self.provider_is_langchain():
132
+ if "response_format" in formatted_kwargs and isinstance(
133
+ formatted_kwargs["response_format"], object
134
+ ):
135
+ """
136
+ We are likely processing the result of LangChain's structured
137
+ output runnable. The object defined in "response_format" is
138
+ recursive (it refers to itself) so formatting it into a dictionary
139
+ will result in an RecursionError. We also do not need the data in
140
+ this object so we are going to discard it here.
141
+ """
142
+
143
+ del formatted_kwargs["response_format"]
144
+
145
+ formatted_kwargs = self.dict_to_json(formatted_kwargs)
146
+
147
+ return formatted_kwargs
148
+
149
+ def _format_payload(
150
+ self,
151
+ client_provider,
152
+ client_title,
153
+ client_version,
154
+ start_time,
155
+ end_time,
156
+ query,
157
+ response,
158
+ ):
159
+ response_json = self.response_to_json(response)
160
+
161
+ payload = {
162
+ "attribution": self.config.attribution,
163
+ "conversation": {
164
+ "client": {
165
+ "provider": client_provider,
166
+ "title": client_title,
167
+ "version": client_version,
168
+ },
169
+ "query": query,
170
+ "response": response_json,
171
+ },
172
+ "meta": {
173
+ "api": {"key": self.config.api_key},
174
+ "fnfg": {
175
+ "exc": None,
176
+ "status": "succeeded",
177
+ },
178
+ "sdk": {"client": "python", "version": self.config.version},
179
+ },
180
+ "time": {"end": end_time, "start": start_time},
181
+ "tx": {"uuid": str(self.config.tx_uuid)},
182
+ }
183
+
184
+ return payload
185
+
186
+ def _format_response(self, raw_response):
187
+ formatted_response = copy.deepcopy(raw_response)
188
+ if self._uses_protobuf:
189
+ # If _uses_protobuf is true, it means the response needs special formatting.
190
+ # Below the _pb case is for protobuf, and the model_dump is for Pydantic which
191
+ # is used in the v4 version of google-genai.
192
+ if not isinstance(formatted_response, (list, dict)):
193
+ # The "_pb" is for handling the protobuf structure. It is used
194
+ # in the LangChain Google code paths.
195
+ if formatted_response.__dict__ and formatted_response.__dict__.get(
196
+ "_pb"
197
+ ):
198
+ formatted_response = json.loads(
199
+ json_format.MessageToJson(formatted_response.__dict__["_pb"])
200
+ )
201
+ elif hasattr(formatted_response, "model_dump"):
202
+ # New path: Pydantic-based response (google-genai >= v4)
203
+ formatted_response = formatted_response.model_dump(mode="json")
204
+
205
+ return formatted_response
206
+
207
+ def _format_intercept_payload(
208
+ self,
209
+ client_provider: Optional[str],
210
+ client_title: str,
211
+ client_version: str,
212
+ kwargs: dict,
213
+ ) -> dict:
214
+ return {
215
+ "attribution": self.config.attribution,
216
+ "conversation": {
217
+ "client": {
218
+ "provider": client_provider,
219
+ "title": client_title,
220
+ "version": client_version,
221
+ },
222
+ "request": self._format_kwargs(kwargs),
223
+ },
224
+ "meta": {
225
+ "sdk": {"client": "python", "version": self.config.version},
226
+ },
227
+ }
228
+
229
+ def get_response_content(self, raw_response):
230
+ if (
231
+ raw_response.__class__.__name__ == "LegacyAPIResponse"
232
+ and raw_response.__class__.__module__ == "openai._legacy_response"
233
+ ):
234
+ """
235
+ Library: langchain-openai
236
+ Version: > 0.3.31
237
+
238
+ Calling the chat / invoke method of the client no longer returns the JSON
239
+ response but instead an object that looks like an API response. This
240
+ object does not inherit from a base class we can reliably identify and
241
+ we do not want to force the OpenAI library as a dependency.
242
+ """
243
+
244
+ return json.loads(raw_response.text)
245
+
246
+ return raw_response
247
+
248
+ def list_to_json(self, list_):
249
+ result = []
250
+ for entry in list_:
251
+ if isinstance(entry, list):
252
+ result.append(self.list_to_json(entry))
253
+ elif isinstance(entry, dict):
254
+ result.append(self.dict_to_json(entry))
255
+ else:
256
+ if hasattr(entry, "__dict__"):
257
+ # This is used in the uses_protobuf cases with LangChain Google.
258
+ result.append(self.dict_to_json(entry.__dict__))
259
+ elif isinstance(entry, bytes):
260
+ # If left as bytes, the payload sent to Collector won't work.
261
+ result.append(base64.b64encode(entry).decode("utf-8"))
262
+ else:
263
+ result.append(entry)
264
+
265
+ return result
266
+
267
+ def _make_relevance_intercept_request(self, **kwargs) -> tuple[bool, Optional[str]]:
268
+ response = Api(self.config).post(
269
+ "sentinel/relevance/intercept",
270
+ self._format_intercept_payload(
271
+ self._client_provider,
272
+ self._resolve_client_title(kwargs),
273
+ self._client_version,
274
+ kwargs,
275
+ ),
276
+ retryable=False,
277
+ timeout=self.config.secs_irrelevant_request_timeout,
278
+ )
279
+
280
+ # Default to relevant if the response is missing expected boolean.
281
+ relevant: bool = response.get("relevant", True)
282
+ reason: Optional[str] = None
283
+
284
+ if not relevant:
285
+ reason = response.get("reason", "Irrelevant request blocked.")
286
+
287
+ return relevant, reason
288
+
289
+ def provider_is_langchain(self):
290
+ return self._client_provider == LANGCHAIN_CLIENT_PROVIDER
291
+
292
+ def _raise_if_irrelevant(self, **kwargs):
293
+ try:
294
+ relevant, reason = self._make_relevance_intercept_request(**kwargs)
295
+ except Exception as e:
296
+ # The caller should be unaware of any unexpected
297
+ # errors that may occur.
298
+ return
299
+
300
+ if not relevant:
301
+ raise CrystilRequestInterceptedError(reason)
302
+
303
+ def response_to_json(self, response):
304
+ data = response
305
+ if isinstance(data, list):
306
+ result = self.list_to_json(data)
307
+ else:
308
+ if not isinstance(data, dict):
309
+ data = response.__dict__
310
+
311
+ result = {}
312
+
313
+ for key, value in data.items():
314
+ if key not in PROTOBUF_SKIP_ATTRS:
315
+ if isinstance(value, list):
316
+ result[key] = self.list_to_json(value)
317
+ elif isinstance(value, dict):
318
+ result[key] = self.dict_to_json(value)
319
+ else:
320
+ if hasattr(value, "__dict__"):
321
+ result[key] = self.dict_to_json(value.__dict__)
322
+ elif isinstance(value, bytes):
323
+ # If left as bytes, the payload sent to Collector won't work
324
+ result[key] = base64.b64encode(value).decode("utf-8")
325
+ else:
326
+ result[key] = value
327
+
328
+ return result
329
+
330
+ def set_client(self, provider, title, version):
331
+ self._client_provider = provider
332
+ self._client_title = title
333
+ self._client_version = version
334
+ return self
335
+
336
+ # Override the static client title with a per-call value derived from the
337
+ # request kwargs. Used by Groq, where the framework hosts many third-party
338
+ # models and the model-ID prefix carries the LLM family
339
+ # (`meta-llama/...` → `meta-llama`).
340
+ #
341
+ # Implementations are responsible for picking a sensible fallback
342
+ # (typically the framework name) when the kwargs don't yield a useful
343
+ # prefix. See `_clients.py::groq_title_from_model` for the canonical
344
+ # example.
345
+ def title_from_kwargs(self, fn: Callable[[dict[str, Any]], str]):
346
+ self._client_title_resolver = fn
347
+ return self
348
+
349
+ def _resolve_client_title(self, kwargs: dict[str, Any]):
350
+ if self._client_title_resolver is None:
351
+ return self._client_title
352
+ return self._client_title_resolver(kwargs)
353
+
354
+ # Opt in to injecting stream_options.include_usage = True for streaming calls.
355
+ # Used by chat-completions wrappers (OpenAI, LangChain-OpenAI, Groq).
356
+ #
357
+ # `via_extra_body=True` routes the injection through the upstream SDK's
358
+ # `extra_body` escape hatch instead of the top-level `stream_options` kwarg.
359
+ # Required for Groq's Python SDK, whose `create()` signature does not
360
+ # accept `stream_options` directly (TypeError) even though the underlying
361
+ # HTTP API supports it.
362
+ def include_stream_options(self, via_extra_body: bool = False):
363
+ self._include_stream_options = True
364
+ self._stream_options_via_extra_body = via_extra_body
365
+ return self
366
+
367
+ def stream_response_extractor(self, fn):
368
+ # Override how the final telemetry response is extracted from the merged
369
+ # stream accumulator. The callable receives the merged chunk dict and
370
+ # returns the object that should be serialised into the payload.
371
+ #
372
+ # Default (None): use get_response_content(raw_response) as-is, which
373
+ # is correct for chat completions chunks that share a single schema.
374
+ #
375
+ # Responses API streaming: events have varying types and the actual
376
+ # Response is nested inside the terminal "response.completed" event:
377
+ # {"type": "response.completed", "response": <Response>, ...}
378
+ # The extractor pulls out raw_response["response"] so the telemetry
379
+ # payload matches the non-streaming responses.create() shape.
380
+ self._stream_response_extractor = fn
381
+ return self
382
+
383
+ def _extract_stream_response(self, raw_response):
384
+ if self._stream_response_extractor is not None:
385
+ return self._stream_response_extractor(raw_response)
386
+ return self.get_response_content(raw_response)
387
+
388
+ def uses_protobuf(self):
389
+ self._uses_protobuf = True
390
+ return self
391
+
392
+
393
+ class BaseIterator:
394
+ def __init__(self, config: Config, source_iterator):
395
+ self.config = config
396
+ self.source_iterator = source_iterator
397
+ self.iterator = None
398
+ self.raw_response = None
399
+
400
+ def configure_invoke(self, invoke: BaseInvoke):
401
+ self.invoke = invoke
402
+ return self
403
+
404
+ def configure_request(self, kwargs, time_start):
405
+ self._kwargs = kwargs
406
+ self._time_start = time_start
407
+ return self
408
+
409
+ def process_chunk(self, chunk):
410
+ if self.invoke._uses_protobuf is True:
411
+ formatted_chunk = copy.deepcopy(chunk)
412
+ if formatted_chunk.__dict__ and formatted_chunk.__dict__.get("_pb"):
413
+ # The "_pb" is for handling the protobuf structure. It is used
414
+ # in the LangChain Google code paths.
415
+ self.raw_response.append(
416
+ json.loads(
417
+ json_format.MessageToJson(formatted_chunk.__dict__["_pb"])
418
+ )
419
+ )
420
+ elif hasattr(formatted_chunk, "model_dump"):
421
+ # New path: Pydantic-based response (google-genai >= v4)
422
+ self.raw_response.append(formatted_chunk.model_dump(mode="json"))
423
+ else:
424
+ self.raw_response = merge_chunk(self.raw_response, chunk.__dict__)
425
+
426
+ return self
427
+
428
+ def set_raw_response(self):
429
+ if self.raw_response is not None:
430
+ return self
431
+
432
+ self.raw_response = {}
433
+ if self.invoke._uses_protobuf:
434
+ self.raw_response = []
435
+
436
+ return self
437
+
438
+
439
+ class BaseProvider:
440
+ def __init__(self, parent):
441
+ self.client = None
442
+ self.parent = parent
443
+ self.config = parent.config