@ashx-j/lunr-ai 0.2.22 → 0.2.24

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/dist/api/anthropic-claude-code-bridge.d.ts +6 -0
  2. package/dist/api/anthropic-claude-code-bridge.d.ts.map +1 -0
  3. package/dist/api/anthropic-claude-code-bridge.js +582 -0
  4. package/dist/api/anthropic-claude-code-bridge.js.map +1 -0
  5. package/dist/api/anthropic-messages.d.ts.map +1 -1
  6. package/dist/api/anthropic-messages.js +7 -19
  7. package/dist/api/anthropic-messages.js.map +1 -1
  8. package/dist/auth/anthropic-route.d.ts +7 -0
  9. package/dist/auth/anthropic-route.d.ts.map +1 -0
  10. package/dist/auth/anthropic-route.js +25 -0
  11. package/dist/auth/anthropic-route.js.map +1 -0
  12. package/dist/auth/oauth/anthropic.d.ts +1 -6
  13. package/dist/auth/oauth/anthropic.d.ts.map +1 -1
  14. package/dist/auth/oauth/anthropic.js +214 -277
  15. package/dist/auth/oauth/anthropic.js.map +1 -1
  16. package/dist/auth/resolve.d.ts.map +1 -1
  17. package/dist/auth/resolve.js +27 -0
  18. package/dist/auth/resolve.js.map +1 -1
  19. package/dist/auth/types.d.ts +15 -2
  20. package/dist/auth/types.d.ts.map +1 -1
  21. package/dist/auth/types.js.map +1 -1
  22. package/dist/cli.d.ts.map +1 -1
  23. package/dist/cli.js +2 -0
  24. package/dist/cli.js.map +1 -1
  25. package/dist/env-api-keys.d.ts.map +1 -1
  26. package/dist/env-api-keys.js +1 -2
  27. package/dist/env-api-keys.js.map +1 -1
  28. package/dist/index.d.ts +1 -0
  29. package/dist/index.d.ts.map +1 -1
  30. package/dist/index.js +1 -0
  31. package/dist/index.js.map +1 -1
  32. package/dist/models.d.ts.map +1 -1
  33. package/dist/models.js +19 -3
  34. package/dist/models.js.map +1 -1
  35. package/dist/providers/anthropic.d.ts.map +1 -1
  36. package/dist/providers/anthropic.js +32 -3
  37. package/dist/providers/anthropic.js.map +1 -1
  38. package/dist/providers/radius.d.ts.map +1 -1
  39. package/dist/providers/radius.js +5 -1
  40. package/dist/providers/radius.js.map +1 -1
  41. package/dist/types.d.ts +13 -0
  42. package/dist/types.d.ts.map +1 -1
  43. package/dist/types.js.map +1 -1
  44. package/package.json +15 -2
  45. package/vendor/hermes-claude-subscription-directsdk/ATTRIBUTION.md +9 -0
  46. package/vendor/hermes-claude-subscription-directsdk/LICENSE +21 -0
  47. package/vendor/hermes-claude-subscription-directsdk/README.md +179 -0
  48. package/vendor/hermes-claude-subscription-directsdk/admission.py +201 -0
  49. package/vendor/hermes-claude-subscription-directsdk/agent/reasoning_effort.py +17 -0
  50. package/vendor/hermes-claude-subscription-directsdk/directsdk.py +629 -0
  51. package/vendor/hermes-claude-subscription-directsdk/directsdk_setup.py +116 -0
  52. package/vendor/hermes-claude-subscription-directsdk/inert_mcp.py +24 -0
  53. package/vendor/hermes-claude-subscription-directsdk/lunr_bridge.py +89 -0
  54. package/vendor/hermes-claude-subscription-directsdk/lunr_setup_bridge.py +49 -0
  55. package/vendor/hermes-claude-subscription-directsdk/manifest.json +32 -0
  56. package/vendor/hermes-claude-subscription-directsdk/model_catalog.py +44 -0
  57. package/vendor/hermes-claude-subscription-directsdk/patches/0001-lunr-lifecycle-hooks.patch +14 -0
  58. package/vendor/hermes-claude-subscription-directsdk/tools/schema_sanitizer.py +23 -0
@@ -0,0 +1,629 @@
1
+ """Request-scoped Claude Code transport with host-owned HTTP admission."""
2
+ from __future__ import annotations
3
+
4
+ import asyncio
5
+ import copy
6
+ import json
7
+ import math
8
+ import os
9
+ from pathlib import Path
10
+ import queue
11
+ import re
12
+ import signal
13
+ import subprocess
14
+ import sys
15
+ import tempfile
16
+ import threading
17
+ import time
18
+ import weakref
19
+ from types import SimpleNamespace
20
+
21
+ try:
22
+ from .admission import Admission
23
+ from .model_catalog import native_model, supports_adaptive_thinking
24
+ from .directsdk_setup import INSTALL_HINT, _resolve as resolve_claude
25
+ except ImportError:
26
+ from admission import Admission
27
+ from model_catalog import native_model, supports_adaptive_thinking
28
+ from directsdk_setup import INSTALL_HINT, _resolve as resolve_claude
29
+
30
+
31
+ class ClaudeCodeMissing(RuntimeError):
32
+ """The official Claude Code CLI this transport drives is not installed (or not on PATH)."""
33
+
34
+
35
+ CARRIER = 'claude-subscription-directsdk-experimental.native_assistant'
36
+ PREFIX = 'mcp__hermes__'
37
+
38
+
39
+ class Object(SimpleNamespace):
40
+ def model_dump(self, **_):
41
+ def unpack(v):
42
+ if isinstance(v, Object):
43
+ return {k: unpack(x) for k, x in vars(v).items()}
44
+ if isinstance(v, list):
45
+ return [unpack(x) for x in v]
46
+ return copy.deepcopy(v)
47
+ return unpack(self)
48
+
49
+
50
+ def obj(value):
51
+ if isinstance(value, dict):
52
+ return Object(**{k: copy.deepcopy(v) if k in ('reasoning_details', 'native_usage') else obj(v) for k, v in value.items()})
53
+ if isinstance(value, list):
54
+ return [obj(v) for v in value]
55
+ return value
56
+
57
+
58
+ def projection(message):
59
+ calls = []
60
+ for tc in message.get('tool_calls') or []:
61
+ f = tc['function']
62
+ args = f['arguments']
63
+ calls.append({'id': tc['id'], 'name': f['name'],
64
+ 'input': json.loads(args) if isinstance(args, str) else args})
65
+ return {'content': (message.get('content') or '').strip(), 'tool_calls': calls}
66
+
67
+
68
+ def content_blocks(content):
69
+ if content is None:
70
+ return []
71
+ if isinstance(content, str):
72
+ return [{'type': 'text', 'text': content}] if content else []
73
+ result = []
74
+ for block in content:
75
+ kind = block.get('type')
76
+ if kind == 'text':
77
+ result.append(copy.deepcopy(block))
78
+ elif kind == 'image_url':
79
+ url = block['image_url']['url']
80
+ if not url.startswith('data:') or ';base64,' not in url:
81
+ raise ValueError('Only base64 data image_url inputs are supported')
82
+ media, data = url[5:].split(';base64,', 1)
83
+ result.append({'type': 'image', 'source': {'type': 'base64', 'media_type': media, 'data': data}})
84
+ elif kind in ('image', 'document', 'tool_result'):
85
+ result.append(copy.deepcopy(block))
86
+ else:
87
+ raise ValueError(f'Unsupported content block: {kind}')
88
+ return result
89
+
90
+
91
+ def prepare_history(messages):
92
+ system, frames = [], []
93
+ for message in messages:
94
+ role = message.get('role')
95
+ if role in ('system', 'developer'):
96
+ if frames:
97
+ raise ValueError('System/developer messages must precede conversation history')
98
+ if not isinstance(message.get('content'), str):
99
+ raise ValueError('System content must be text')
100
+ system.append(message['content'])
101
+ continue
102
+ if role == 'assistant':
103
+ details = message.get('reasoning_details') or []
104
+ carriers = [d for d in details if isinstance(d, dict) and d.get('type') == CARRIER]
105
+ if carriers:
106
+ if len(carriers) != 1 or carriers[0].get('version') != 1:
107
+ raise ValueError('Unsupported native assistant carrier version')
108
+ carrier = carriers[0]
109
+ expected = {**carrier['projection'], 'content': carrier['projection']['content'].strip()}
110
+ if projection(message) == expected:
111
+ for native in carrier['messages']:
112
+ frames.append({'type': 'assistant', 'message': copy.deepcopy(native)})
113
+ continue
114
+ # Host compaction/hooks own visible history. Never restore stale
115
+ # pre-edit blocks or attach their signatures to rewritten content.
116
+ blocks = content_blocks(message.get('content'))
117
+ for call in projection(message)['tool_calls']:
118
+ blocks.append({'type': 'tool_use', 'id': call['id'], 'name': PREFIX + call['name'], 'input': call['input']})
119
+ elif role == 'tool':
120
+ role = 'user'
121
+ blocks = [{'type': 'tool_result', 'tool_use_id': message['tool_call_id'],
122
+ 'content': message.get('content') if isinstance(message.get('content'), str) else content_blocks(message.get('content'))}]
123
+ if message.get('is_error') is not None:
124
+ blocks[0]['is_error'] = bool(message['is_error'])
125
+ elif role == 'user':
126
+ blocks = content_blocks(message.get('content'))
127
+ else:
128
+ raise ValueError(f'Unsupported message role: {role}')
129
+ if frames and frames[-1]['type'] == role and role == 'user':
130
+ frames[-1]['message']['content'].extend(blocks)
131
+ else:
132
+ frames.append({'type': role, 'message': {'role': role, 'content': blocks}})
133
+ if not frames or frames[-1]['type'] != 'user' or not frames[-1]['message']['content']:
134
+ raise ValueError('History must end in a nonempty user/tool-result message; assistant prefill is unsupported')
135
+ return '\n\n'.join(system), frames
136
+
137
+
138
+ _BANNED_TOP_LEVEL = ('oneOf', 'allOf', 'anyOf')
139
+
140
+
141
+ def normalize_input_schema(schema):
142
+ """Anthropic's validator hard-400s on top-level oneOf/allOf/anyOf and on the null branch of
143
+ nullable unions. The host normalizes both in ``agent.anthropic_message_convert``, but only for
144
+ ``api_mode='messages'``; this transport is ``chat_completions``, so mirror it here. The
145
+ combinators are advisory (handlers re-validate their arguments); nested unions stay untouched."""
146
+ from tools.schema_sanitizer import strip_nullable_unions
147
+ normalized = strip_nullable_unions(schema, keep_nullable_hint=False)
148
+ if any(key in normalized for key in _BANNED_TOP_LEVEL):
149
+ normalized = {k: v for k, v in normalized.items() if k not in _BANNED_TOP_LEVEL}
150
+ normalized.setdefault('type', 'object')
151
+ if normalized.get('type') == 'object' and not isinstance(normalized.get('properties'), dict):
152
+ normalized = {**normalized, 'properties': {}}
153
+ return normalized
154
+
155
+
156
+ def request_body(kwargs):
157
+ allowed = {'model', 'messages', 'tools', 'stream', 'stream_options', 'max_tokens', 'max_completion_tokens',
158
+ 'temperature', 'top_p', 'stop', 'extra_body', 'timeout', 'tool_choice', 'parallel_tool_calls', 'n', 'response_format'}
159
+ unknown = set(kwargs) - allowed
160
+ if unknown:
161
+ raise ValueError('Unsupported request parameters: ' + ', '.join(sorted(unknown)))
162
+ if kwargs.get('n', 1) != 1 or kwargs.get('tool_choice', 'auto') not in ('auto', None):
163
+ raise ValueError('Only n=1 and tool_choice=auto are supported')
164
+ if kwargs.get('parallel_tool_calls') is False:
165
+ raise ValueError('parallel_tool_calls=False is unsupported')
166
+ if kwargs.get('stream_options') not in (None, {}, {'include_usage': True}, {'include_usage': False}):
167
+ raise ValueError('Unsupported stream_options')
168
+ extra = kwargs.get('extra_body')
169
+ if extra is None:
170
+ extra = {}
171
+ if not isinstance(extra, dict):
172
+ raise ValueError('extra_body must be an object')
173
+ unknown_extra = set(extra) - {'max_tokens', 'temperature', 'top_p', 'stop_sequences', 'reasoning', 'response_format'}
174
+ if unknown_extra:
175
+ raise ValueError('Unsupported extra_body fields: ' + ', '.join(sorted(unknown_extra)))
176
+ body = copy.deepcopy(extra)
177
+ reasoning = body.pop('reasoning', None)
178
+ if reasoning is not None:
179
+ if not isinstance(reasoning, dict) or set(reasoning) - {'enabled', 'effort'}:
180
+ raise ValueError('reasoning supports enabled and effort only')
181
+ if 'enabled' in reasoning and type(reasoning['enabled']) is not bool:
182
+ raise ValueError('reasoning.enabled must be boolean')
183
+ from agent.reasoning_effort import clamp_effort
184
+ effort = clamp_effort(reasoning.get('effort'), ('none', 'low', 'medium', 'high', 'xhigh', 'max'))
185
+ if effort not in (None, 'none', 'low', 'medium', 'high', 'xhigh', 'max'):
186
+ raise ValueError('Unsupported native reasoning effort')
187
+ if reasoning.get('enabled') is False or effort == 'none':
188
+ body['thinking'] = {'type': 'disabled'}
189
+ # Native clear-thinking context edits are invalid when thinking is disabled.
190
+ body['context_management'] = {'edits': []}
191
+ else:
192
+ # Routes without adaptive thinking (Haiku 4.5) 400 on the block; their own
193
+ # default thinking plus the effort signal below stand in for it.
194
+ if reasoning.get('enabled') is True and supports_adaptive_thinking(kwargs.get('model')):
195
+ body['thinking'] = {'type': 'adaptive'}
196
+ if effort:
197
+ body['output_config'] = {'effort': effort}
198
+ response_format = kwargs.get('response_format', body.pop('response_format', None))
199
+ if response_format and response_format.get('type') != 'text':
200
+ if response_format.get('type') != 'json_schema':
201
+ raise ValueError('Only json_schema structured output is supported')
202
+ schema = response_format.get('json_schema', {}).get('schema')
203
+ if not isinstance(schema, dict):
204
+ raise ValueError('response_format requires a JSON Schema object')
205
+ body.setdefault('output_config', {})['format'] = {'type': 'json_schema', 'schema': schema}
206
+ for key in ('max_tokens', 'temperature', 'top_p'):
207
+ if kwargs.get(key) is not None:
208
+ body[key] = kwargs[key]
209
+ if kwargs.get('max_completion_tokens') is not None:
210
+ if 'max_tokens' in body:
211
+ raise ValueError('Specify only one output-token limit')
212
+ body['max_tokens'] = kwargs['max_completion_tokens']
213
+ if kwargs.get('stop') is not None:
214
+ stop = kwargs['stop']
215
+ body['stop_sequences'] = [stop] if isinstance(stop, str) else stop
216
+ for key in ('temperature', 'top_p'):
217
+ if key in body and (isinstance(body[key], bool) or not isinstance(body[key], (int, float)) or not math.isfinite(body[key]) or not 0 <= body[key] <= 1):
218
+ raise ValueError(f'{key} must be finite and between zero and one')
219
+ # Subscription models reject sampling controls, including Hermes' title
220
+ # generator default. Match the host's sampling-forbidden model behavior.
221
+ body.pop(key, None)
222
+ if 'max_tokens' in body and (type(body['max_tokens']) is not int or body['max_tokens'] < 1):
223
+ raise ValueError('max_tokens must be a positive integer')
224
+ if 'stop_sequences' in body and (not isinstance(body['stop_sequences'], list) or not all(isinstance(x, str) and x for x in body['stop_sequences'])):
225
+ raise ValueError('stop_sequences must be a list of nonempty strings')
226
+ manifest, tools, names = [], [], set()
227
+ for tool in kwargs.get('tools') or []:
228
+ if tool.get('type') != 'function':
229
+ raise ValueError('Only function tools are supported')
230
+ f = tool['function']
231
+ name = f['name']
232
+ if not isinstance(name, str) or not re.fullmatch(r'[A-Za-z0-9_-]{1,50}', name) or name in names:
233
+ raise ValueError('Tool names must be unique ASCII identifiers of at most 50 characters')
234
+ if f.get('strict'):
235
+ raise ValueError('Strict function schemas are unsupported')
236
+ names.add(name)
237
+ schema, description = f.get('parameters', {'type': 'object'}), f.get('description', '')
238
+ if not isinstance(schema, dict) or not isinstance(description, str):
239
+ raise ValueError('Tool schema must be an object and description a string')
240
+ # The manifest (inert MCP server) and the request body must advertise the same shape.
241
+ schema = normalize_input_schema(schema)
242
+ manifest.append({'name': name, 'description': description, 'inputSchema': schema})
243
+ tools.append({'name': PREFIX + name, 'description': description, 'input_schema': schema})
244
+ body['tools'] = tools
245
+ encoded = json.dumps(body, separators=(',', ':'), allow_nan=False)
246
+ return encoded, manifest, names
247
+
248
+
249
+ class Request:
250
+ def __init__(self, client):
251
+ self.client, self.process = client, None
252
+ self.stream = None
253
+ self.cancelled = threading.Event()
254
+ self.admission = None
255
+ self.lock = threading.Lock()
256
+
257
+ def cancel(self):
258
+ self.cancelled.set()
259
+ if self.admission is not None:
260
+ self.admission.abort()
261
+ with self.lock:
262
+ if self.process is not None:
263
+ kill_process_tree(self.process)
264
+
265
+ def spawn(self, command, *, stdin=subprocess.DEVNULL, **kwargs):
266
+ with self.lock:
267
+ if self.cancelled.is_set():
268
+ raise RuntimeError('Claude request cancelled')
269
+ self.process = subprocess.Popen(command, stdin=stdin, **kwargs, **_own_process_group())
270
+ if self.client.on_spawn is not None:
271
+ self.client.on_spawn(self.process.pid)
272
+ return self.process
273
+
274
+
275
+ def _own_process_group():
276
+ """Popen kwargs that put native (and the node/cmd children it spawns) in a group we can kill as one."""
277
+ if os.name == 'nt':
278
+ return {'creationflags': subprocess.CREATE_NEW_PROCESS_GROUP}
279
+ return {'start_new_session': True}
280
+
281
+
282
+ def kill_process_tree(process):
283
+ """Kill native and every descendant: the npm shim is cmd.exe -> node on Windows, and a plain
284
+ Popen.kill() would orphan the node child that holds the real request open."""
285
+ if process.poll() is not None:
286
+ return
287
+ if os.name == 'nt':
288
+ subprocess.run(['taskkill', '/F', '/T', '/PID', str(process.pid)], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, check=False)
289
+ return
290
+ try:
291
+ os.killpg(process.pid, signal.SIGKILL) # windows-footgun: ok — the nt branch above never reaches this line
292
+ except (ProcessLookupError, PermissionError):
293
+ # ESRCH: already gone. EPERM: macOS answers killpg with EPERM once the group leader is a zombie.
294
+ pass
295
+
296
+
297
+ # Node, cmd.exe and Claude Code's own config lookup need these even when the caller hands us a
298
+ # deliberately minimal environment; without SystemRoot a Windows child cannot even open a socket.
299
+ _WINDOWS_ESSENTIALS = ('SYSTEMROOT', 'SYSTEMDRIVE', 'COMSPEC', 'PATHEXT', 'TEMP', 'TMP', 'USERPROFILE', 'APPDATA', 'LOCALAPPDATA', 'PROGRAMDATA')
300
+
301
+
302
+ def _with_windows_essentials(env):
303
+ if os.name != 'nt':
304
+ return env
305
+ present = {key.upper() for key in env}
306
+ for key, value in os.environ.items():
307
+ if key.upper() in _WINDOWS_ESSENTIALS and key.upper() not in present:
308
+ env[key] = value
309
+ return env
310
+
311
+
312
+ class Stream:
313
+ def __init__(self, iterator, request):
314
+ self.iterator, self.request = iterator, request
315
+ self._advancing = threading.Lock()
316
+ request.stream = weakref.ref(self)
317
+ def __iter__(self):
318
+ return self
319
+ def __next__(self):
320
+ with self._advancing:
321
+ return next(self.iterator)
322
+ def close(self):
323
+ self.request.cancel()
324
+ # An active consumer unwinds itself after cancellation. A paused/unstarted
325
+ # generator has no active owner and can be finalized here.
326
+ if self._advancing.acquire(blocking=False):
327
+ try:
328
+ self.iterator.close()
329
+ with self.request.client._lock:
330
+ self.request.client._requests.discard(self.request)
331
+ finally:
332
+ self._advancing.release()
333
+ def __enter__(self):
334
+ return self
335
+ def __exit__(self, *_):
336
+ self.close()
337
+
338
+
339
+ class AsyncStream:
340
+ def __init__(self, stream):
341
+ self.stream = stream
342
+ def __aiter__(self):
343
+ return self
344
+ async def __anext__(self):
345
+ def advance():
346
+ try:
347
+ return True, next(self.stream)
348
+ except StopIteration:
349
+ return False, None
350
+ try:
351
+ present, item = await asyncio.to_thread(advance)
352
+ except asyncio.CancelledError:
353
+ self.stream.close()
354
+ raise
355
+ if not present:
356
+ raise StopAsyncIteration
357
+ return item
358
+ async def aclose(self):
359
+ self.stream.close()
360
+ async def __aenter__(self):
361
+ return self
362
+ async def __aexit__(self, *_):
363
+ await self.aclose()
364
+
365
+
366
+ class Client:
367
+ HERMES_SKIP_TRANSPORT_WRAP = True
368
+ HERMES_SKIP_ASYNC_WRAP = True
369
+
370
+ def __init__(self, command=None, args=None, env=None, timeout=180, on_spawn=None, on_replay=None, **_):
371
+ # Hermes snapshots routing metadata from client-shaped objects; this is not a credential.
372
+ self.api_key = 'external-process'
373
+ self.base_url = 'process://claude-subscription-directsdk-experimental'
374
+ self.env = dict(env) if env is not None else None
375
+ source_env = self.env if self.env is not None else os.environ
376
+ command = command or source_env.get('CLAUDE_SUBSCRIPTION_DIRECTSDK_COMMAND') or 'claude'
377
+ self.command = ([command] if isinstance(command, str) else list(command)) + list(args or [])
378
+ self.timeout = timeout if isinstance(timeout, (int, float)) else 180
379
+ self.on_spawn = on_spawn
380
+ self.on_replay = on_replay
381
+ self._lock, self._requests, self._closed = threading.Lock(), set(), False
382
+ self.chat = SimpleNamespace(completions=SimpleNamespace(create=self.create))
383
+
384
+ def cancel(self):
385
+ """Fast cross-thread cancellation: signal owned groups; never close caller-thread FDs."""
386
+ with self._lock:
387
+ requests = tuple(self._requests)
388
+ for request in requests:
389
+ request.cancel()
390
+
391
+ def close(self):
392
+ with self._lock:
393
+ self._closed = True
394
+ requests = tuple(self._requests)
395
+ for request in requests:
396
+ stream = request.stream() if request.stream else None
397
+ if stream is not None:
398
+ stream.close()
399
+ else:
400
+ request.cancel()
401
+
402
+ def create(self, **kwargs):
403
+ # Hermes' auxiliary seam returns this same object and awaits create.
404
+ try:
405
+ asyncio.get_running_loop()
406
+ except RuntimeError:
407
+ pass
408
+ else:
409
+ return self._acreate(**kwargs)
410
+ return self._create(**kwargs)
411
+
412
+ async def _acreate(self, **kwargs):
413
+ task = asyncio.create_task(asyncio.to_thread(self._create, **kwargs))
414
+ try:
415
+ result = await task
416
+ return AsyncStream(result) if kwargs.get('stream') else result
417
+ except asyncio.CancelledError:
418
+ self.cancel()
419
+ raise
420
+
421
+ def _create(self, **kwargs):
422
+ body, manifest, names = request_body(kwargs)
423
+ system, frames = prepare_history(kwargs.get('messages', []))
424
+ if not isinstance(kwargs.get('model'), str) or not kwargs['model']:
425
+ raise ValueError('model is required')
426
+ request = Request(self)
427
+ with self._lock:
428
+ if self._closed:
429
+ raise RuntimeError('Claude client is closed')
430
+ self._requests.add(request)
431
+ stream = Stream(self._run(request, kwargs, body, manifest, names, system, frames), request)
432
+ if kwargs.get('stream'):
433
+ return stream
434
+ try:
435
+ for chunk in stream:
436
+ if hasattr(chunk, '_response'):
437
+ return chunk._response
438
+ raise RuntimeError('Native response missing')
439
+ finally:
440
+ stream.close()
441
+
442
+ def _run(self, request, kwargs, body, manifest, names, system, frames):
443
+ p = None
444
+ reader = None
445
+ try:
446
+ timeout = kwargs.get('timeout', self.timeout)
447
+ timeout = getattr(timeout, 'read', timeout)
448
+ if not isinstance(timeout, (int, float)) or timeout <= 0:
449
+ raise ValueError('timeout must be positive seconds')
450
+ # Windows refuses to delete a directory a dying child still holds as cwd; the owner thread's
451
+ # p.wait() below reaps before we leave the block, and stragglers must not fail the request.
452
+ with tempfile.TemporaryDirectory(prefix='claude-directsdk-', ignore_cleanup_errors=True) as tmp:
453
+ root = Path(tmp)
454
+ (root / 'tools.json').write_text(json.dumps(manifest), encoding='utf-8')
455
+ mcp = {'mcpServers': {'hermes': {'command': sys.executable, 'args': [str(Path(__file__).with_name('inert_mcp.py')), str(root / 'tools.json')]}}}
456
+ env = _with_windows_essentials(dict(self.env if self.env is not None else os.environ))
457
+ if self.env is None:
458
+ conflicts = [key for key in ('ANTHROPIC_API_KEY', 'ANTHROPIC_AUTH_TOKEN', 'ANTHROPIC_BASE_URL', 'ANTHROPIC_FOUNDRY_API_KEY') if env.get(key)]
459
+ conflicts += [key for key in ('CLAUDE_CODE_USE_BEDROCK', 'CLAUDE_CODE_USE_VERTEX', 'CLAUDE_CODE_USE_FOUNDRY') if env.get(key, '').lower() not in ('', '0', 'false', 'no', 'off')]
460
+ if conflicts:
461
+ raise ValueError('OAuth provider refuses conflicting native auth/backend overrides: ' + ', '.join(conflicts))
462
+ # Fail with the install hint, not a Popen FileNotFoundError, when Claude Code is absent.
463
+ resolved = resolve_claude(self.command, env)
464
+ if resolved is None:
465
+ raise ClaudeCodeMissing(INSTALL_HINT)
466
+ config = env.pop('CLAUDE_SUBSCRIPTION_DIRECTSDK_CONFIG_DIR', None)
467
+ if config:
468
+ env['CLAUDE_CONFIG_DIR'] = config
469
+ env.pop('CLAUDE_CODE_EXTRA_BODY', None)
470
+ env.update(ENABLE_TOOL_SEARCH='false', CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC='1', CLAUDE_CODE_MAX_RETRIES='0', DISABLE_AUTO_COMPACT='1', DISABLE_COMPACT='1')
471
+ # Hermes owns budgets; native's replayed reminder invalidates cached history.
472
+ env['CLAUDE_CODE_TOTAL_TOKENS_REMINDER'] = 'off'
473
+ request.admission = Admission(env.get('ANTHROPIC_BASE_URL', 'https://api.anthropic.com'), timeout)
474
+ env['ANTHROPIC_BASE_URL'] = request.admission.url
475
+ # Native settings apply env inside the process, avoiding execve's
476
+ # per-argument/environment-string limit for full Hermes schemas.
477
+ (root / 'settings.json').write_text(json.dumps({'env': {'CLAUDE_CODE_EXTRA_BODY': body}}), encoding='utf-8')
478
+ (root / 'system.md').write_text(system, encoding='utf-8')
479
+ if 'max_tokens' in json.loads(body):
480
+ env['CLAUDE_CODE_MAX_OUTPUT_TOKENS'] = str(json.loads(body)['max_tokens'])
481
+ # The resolved path matters on Windows: CreateProcess finds claude.exe on PATH but not the npm claude.cmd shim.
482
+ command = resolved + ['-p', '--model', native_model(kwargs['model']), '--input-format', 'stream-json', '--output-format', 'stream-json', '--verbose', '--include-partial-messages', '--tools', '', '--system-prompt-file', str(root / 'system.md'), '--settings', str(root / 'settings.json'), '--setting-sources', '', '--strict-mcp-config', '--disable-slash-commands', '--max-turns', '1', '--permission-mode', 'dontAsk', '--no-session-persistence', '--mcp-config', json.dumps(mcp)]
483
+ p = request.spawn(command, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, text=True, encoding='utf-8', cwd=tmp, env=env)
484
+ events = queue.Queue()
485
+ def read():
486
+ try:
487
+ for line in p.stdout:
488
+ events.put(json.loads(line))
489
+ except Exception as error:
490
+ events.put(error)
491
+ finally:
492
+ # The consumer may close while paused at a yielded chunk.
493
+ # Reaping belongs to this owner thread, never cancel().
494
+ p.wait()
495
+ events.put(None)
496
+ reader = threading.Thread(target=read, daemon=True)
497
+ reader.start()
498
+ deadline = time.monotonic() + timeout
499
+ def receive():
500
+ nonlocal deadline
501
+ while True:
502
+ if request.cancelled.is_set():
503
+ raise RuntimeError('Claude request cancelled')
504
+ remaining = deadline - time.monotonic()
505
+ if remaining <= 0:
506
+ raise TimeoutError('Claude request timed out')
507
+ try:
508
+ event = events.get(timeout=min(remaining, .2))
509
+ except queue.Empty:
510
+ continue
511
+ if isinstance(event, Exception):
512
+ # The offending stdout line is the whole diagnosis (a shim banner, a stray print); keep it.
513
+ raise RuntimeError('Invalid native stream-json output: ' + repr((getattr(event, 'doc', None) or str(event))[:300])) from event
514
+ deadline = time.monotonic() + timeout
515
+ return event
516
+ for index, frame in enumerate(frames):
517
+ frame = copy.deepcopy(frame)
518
+ if frame['type'] == 'user' and index < len(frames) - 1:
519
+ frame['shouldQuery'] = False
520
+ p.stdin.write(json.dumps(frame, allow_nan=False) + '\n')
521
+ p.stdin.flush()
522
+ if frame.get('shouldQuery') is False:
523
+ while True:
524
+ ack = receive()
525
+ if ack is None:
526
+ raise RuntimeError('Native exited before replay acknowledgment')
527
+ if ack.get('type') == 'result':
528
+ if ack.get('num_turns') != 0 or ack.get('is_error'):
529
+ raise RuntimeError('Native history replay not supported: expected zero-turn acknowledgment')
530
+ if self.on_replay is not None:
531
+ self.on_replay()
532
+ break
533
+ p.stdin.close()
534
+ assistants, results, stopped, emitted = [], [], False, ''
535
+ native_error = None
536
+ while True:
537
+ event = receive()
538
+ if event is None:
539
+ break
540
+ kind = event.get('type')
541
+ if kind == 'assistant':
542
+ if event.get('error') or event.get('message', {}).get('error'):
543
+ detail = '\n'.join(b.get('text', '') for b in event.get('message', {}).get('content', []) if b.get('type') == 'text')
544
+ native_error = detail
545
+ else:
546
+ assistants.append(event['message'])
547
+ elif kind == 'result':
548
+ results.append(event)
549
+ elif kind == 'stream_event':
550
+ native = event['event']
551
+ if native['type'] == 'message_stop':
552
+ stopped = True
553
+ delta = native.get('delta', {})
554
+ if delta.get('type') == 'text_delta':
555
+ emitted += delta['text']
556
+ yield self._chunk(kwargs['model'], {'content': delta['text']})
557
+ elif delta.get('type') == 'thinking_delta':
558
+ yield self._chunk(kwargs['model'], {'reasoning_content': delta['thinking']})
559
+ p.wait(timeout=max(.1, deadline-time.monotonic()))
560
+ reader.join(timeout=1)
561
+ if request.cancelled.is_set():
562
+ raise RuntimeError('Claude request cancelled')
563
+ admission = request.admission
564
+ if admission.used:
565
+ if admission.status != 200 or not admission.capture.complete:
566
+ # Native's last error is the admission denial; name the first attempt's outcome so reports are diagnosable.
567
+ first = f'first upstream attempt: status {admission.status}, capture ' + ('complete' if admission.capture.complete else 'incomplete') + (f', relay failure {admission.failure}' if admission.failure else '') + f', native retries denied: {admission.denied}'
568
+ if admission.error_text():
569
+ first += ', upstream said: ' + admission.error_text()[:500]
570
+ raise RuntimeError(f'Incomplete upstream response ({first})' + (': ' + native_error if native_error else ''))
571
+ assistants = [admission.capture.message]
572
+ stopped = True
573
+ native_failure_handled = admission.denied or (admission.used and assistants[0].get('stop_reason') == 'refusal')
574
+ if native_error and not native_failure_handled:
575
+ raise RuntimeError('Native API error: ' + native_error)
576
+ if len(results) != 1 or not assistants or not stopped:
577
+ raise RuntimeError('Incomplete native response: assistant, message_stop and one result required')
578
+ final = results[0]
579
+ blocks = [b for a in assistants for b in a['content']]
580
+ calls = []
581
+ for block in blocks:
582
+ if block.get('type') == 'tool_use':
583
+ name = block['name']
584
+ if not name.startswith(PREFIX) or name[len(PREFIX):] not in names:
585
+ raise RuntimeError('Native returned a tool outside the current host inventory')
586
+ calls.append({'id': block['id'], 'type': 'function', 'function': {'name': name[len(PREFIX):], 'arguments': json.dumps(block['input'], separators=(',', ':'), allow_nan=False)}})
587
+ boundary = bool(calls) and final.get('subtype') == 'error_max_turns' and p.returncode == 1
588
+ if not boundary and not native_failure_handled and (p.returncode != 0 or final.get('is_error') or final.get('subtype') != 'success'):
589
+ raise RuntimeError('Native request failed: ' + str(final.get('subtype')))
590
+ usage = assistants[0]['usage'] if admission.used else final.get('usage')
591
+ if not isinstance(usage, dict) or not all(isinstance(usage.get(k), (int, float)) for k in ('input_tokens', 'output_tokens')):
592
+ raise RuntimeError('Native result missing complete token usage')
593
+ text = ''.join(b.get('text', '') for b in blocks if b.get('type') == 'text')
594
+ if emitted != text:
595
+ if text.startswith(emitted):
596
+ yield self._chunk(kwargs['model'], {'content': text[len(emitted):]})
597
+ else:
598
+ raise RuntimeError('Native final text differs from incremental stream')
599
+ message = {'role': 'assistant', 'content': text or None, 'tool_calls': calls or None,
600
+ 'reasoning_content': ''.join(b.get('thinking', '') for b in blocks if b.get('type') == 'thinking') or None}
601
+ carrier = {'type': CARRIER, 'version': 1, 'messages': assistants, 'projection': projection(message)}
602
+ message['reasoning_details'] = [carrier]
603
+ inp = usage['input_tokens'] + usage.get('cache_read_input_tokens', 0) + usage.get('cache_creation_input_tokens', 0)
604
+ normalized_usage = {'prompt_tokens': inp, 'completion_tokens': usage['output_tokens'], 'total_tokens': inp + usage['output_tokens'], 'prompt_tokens_details': {'cached_tokens': usage.get('cache_read_input_tokens', 0)}, 'cache_creation_input_tokens': usage.get('cache_creation_input_tokens', 0), 'native_usage': usage,
605
+ 'completion_tokens_details': {'reasoning_tokens': usage.get('output_tokens_details', {}).get('thinking_tokens', 0)},
606
+ 'native_cost': {'total_cost_usd': final.get('total_cost_usd'), 'modelUsage': final.get('modelUsage')}}
607
+ normalized_usage['native_admission'] = {'upstream_requests': int(admission.used), 'blocked_requests': admission.denied, 'request_id': admission.request_id}
608
+ finish = 'tool_calls' if calls else ('length' if any(a.get('stop_reason') in ('max_tokens', 'model_context_window_exceeded') for a in assistants) else 'stop')
609
+ response = obj({'id': assistants[-1].get('id', 'claude-native'), 'model': kwargs['model'], 'object': 'chat.completion', 'choices': [{'index': 0, 'finish_reason': finish, 'message': message}], 'usage': normalized_usage})
610
+ chunk = self._chunk(kwargs['model'], {'content': None, 'tool_calls': [dict(tc, index=i) for i, tc in enumerate(calls)] or None, 'reasoning_details': [carrier]}, finish, normalized_usage)
611
+ chunk._response = response
612
+ yield chunk
613
+ finally:
614
+ request.cancel()
615
+ if request.admission is not None:
616
+ request.admission.close()
617
+ if p is not None:
618
+ p.wait(timeout=5)
619
+ if reader is not None:
620
+ reader.join(timeout=5)
621
+ for pipe in (p.stdin, p.stdout):
622
+ if pipe and not pipe.closed:
623
+ pipe.close()
624
+ with self._lock:
625
+ self._requests.discard(request)
626
+
627
+ @staticmethod
628
+ def _chunk(model, delta, finish=None, usage=None):
629
+ return obj({'id': 'claude-native', 'model': model, 'object': 'chat.completion.chunk', 'choices': [{'index': 0, 'delta': {'content': None, 'tool_calls': None, 'reasoning_details': None, **delta}, 'finish_reason': finish}], 'usage': usage})