venice-cli 0.83.2__tar.gz → 0.83.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {venice_cli-0.83.2 → venice_cli-0.83.4}/PKG-INFO +7 -1
- {venice_cli-0.83.2 → venice_cli-0.83.4}/README.md +6 -0
- venice_cli-0.83.4/src/venice/__init__.py +1 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_agent.py +50 -5
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_compact.py +6 -1
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_openai.py +48 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_repl.py +6 -4
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_review.py +5 -2
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/chat.py +31 -2
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/code.py +25 -12
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_agent.py +94 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_chat.py +15 -8
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_code_command.py +109 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_compact.py +17 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_repl.py +26 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_review.py +26 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_sessions.py +16 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_shared_openai.py +35 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_spawn.py +22 -0
- venice_cli-0.83.2/src/venice/__init__.py +0 -1
- {venice_cli-0.83.2 → venice_cli-0.83.4}/.gitignore +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/CONTRIBUTING.md +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/LICENSE +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/Makefile +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/SECURITY.md +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/bin/venice +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/install.sh +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/pyproject.toml +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/__main__.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/audio_player.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/audio_post.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/auth.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/billing.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/cli.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/client.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/__init__.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_audio.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_browser.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_code.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_exec.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_index.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_mailbox.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_mcp.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_mcp_client.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_memory.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_models.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_persona.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_queue.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_session.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_shared.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/_steer.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/balance.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/bg_remove.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/completion.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/config.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/contact_sheet.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/embed.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/image.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/image_edit.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/index.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/login.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/master.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/mcp_serve.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/memory.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/models.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/music.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/review.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/search.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/secret.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/sessions.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/sfx.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/tts.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/upscale.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/commands/video.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/config.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/image_montage.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/mcp_server.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/src/venice/userconfig.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/__init__.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/_drive.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/_hygiene.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/_mcp_fake_server.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/_venice_fake_server.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_auth.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_balance.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_bg_remove.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_browser.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_cli.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_client.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_code.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_completion.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_config.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_contact_sheet.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_drive_cli.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_embed.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_exec.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_fake_server.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_image.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_image_edit.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_index.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_login.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_mailbox.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_master.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_mcp_client.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_mcp_serve.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_mcp_tools.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_memory.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_models.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_music.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_persona.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_planner.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_profiles.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_search.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_secret.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_sfx.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_shared.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_source_hygiene.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_steer.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_tts.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_upscale.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_video.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/tests/test_web_search.py +0 -0
- {venice_cli-0.83.2 → venice_cli-0.83.4}/uninstall.sh +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: venice-cli
|
|
3
|
-
Version: 0.83.
|
|
3
|
+
Version: 0.83.4
|
|
4
4
|
Summary: A stdlib-only Python CLI for the Venice.ai API: audio, video, images, chat, and embeddings.
|
|
5
5
|
Project-URL: Homepage, https://github.com/gobha-me/venice-cli
|
|
6
6
|
Project-URL: Repository, https://github.com/gobha-me/venice-cli
|
|
@@ -662,6 +662,12 @@ parameters, `max-tool-calls`, the `venice code` sandbox root, and the running
|
|
|
662
662
|
token/cost usage — so resuming restores the whole context, not just the messages.
|
|
663
663
|
The API key is never written to a session.
|
|
664
664
|
|
|
665
|
+
`venice code` also assigns each session an opaque `prompt_cache_key`, sent as an
|
|
666
|
+
OpenAI-compatible routing hint so successive plan, tool-loop, and verification calls
|
|
667
|
+
stay on cache-affine backend infrastructure when the provider supports it. The key is
|
|
668
|
+
not a credential. It survives `--resume` and `/reset`; disposable subagents receive
|
|
669
|
+
their own keys, while compaction's deliberately fresh summary request receives none.
|
|
670
|
+
|
|
665
671
|
One-shot `venice code "task"` runs are sessions too (they persist unless
|
|
666
672
|
`--ephemeral`), so an unattended `--auto` run is resumable, inspectable, and —
|
|
667
673
|
new in this release — **steerable while it runs** (see below).
|
|
@@ -622,6 +622,12 @@ parameters, `max-tool-calls`, the `venice code` sandbox root, and the running
|
|
|
622
622
|
token/cost usage — so resuming restores the whole context, not just the messages.
|
|
623
623
|
The API key is never written to a session.
|
|
624
624
|
|
|
625
|
+
`venice code` also assigns each session an opaque `prompt_cache_key`, sent as an
|
|
626
|
+
OpenAI-compatible routing hint so successive plan, tool-loop, and verification calls
|
|
627
|
+
stay on cache-affine backend infrastructure when the provider supports it. The key is
|
|
628
|
+
not a credential. It survives `--resume` and `/reset`; disposable subagents receive
|
|
629
|
+
their own keys, while compaction's deliberately fresh summary request receives none.
|
|
630
|
+
|
|
625
631
|
One-shot `venice code "task"` runs are sessions too (they persist unless
|
|
626
632
|
`--ephemeral`), so an unattended `--auto` run is resumable, inspectable, and —
|
|
627
633
|
new in this release — **steerable while it runs** (see below).
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.83.4"
|
|
@@ -46,6 +46,7 @@ from . import _mcp
|
|
|
46
46
|
from . import _memory
|
|
47
47
|
from . import _models
|
|
48
48
|
from . import _compact
|
|
49
|
+
from . import _openai
|
|
49
50
|
from .models import MODEL_TYPES
|
|
50
51
|
|
|
51
52
|
|
|
@@ -2707,11 +2708,51 @@ def supports_function_calling(models, model_id) -> Optional[bool]:
|
|
|
2707
2708
|
# --------------------------------------------------------------------------- #
|
|
2708
2709
|
# The loop
|
|
2709
2710
|
# --------------------------------------------------------------------------- #
|
|
2711
|
+
_REASONING_FIELDS = ("reasoning_content", "reasoning_details", "reasoning")
|
|
2712
|
+
|
|
2713
|
+
|
|
2714
|
+
def _message_field(msg, name):
|
|
2715
|
+
"""Read one response-message field from an SDK model or a plain mapping."""
|
|
2716
|
+
if isinstance(msg, dict):
|
|
2717
|
+
return msg.get(name)
|
|
2718
|
+
return getattr(msg, name, None) if msg is not None else None
|
|
2719
|
+
|
|
2720
|
+
|
|
2721
|
+
def _request_value(value):
|
|
2722
|
+
"""Turn nested SDK response values into request-safe Python values.
|
|
2723
|
+
|
|
2724
|
+
``reasoning_content`` is normally a string, but the compatible
|
|
2725
|
+
``reasoning_details`` extension may contain SDK model objects. Serialize only
|
|
2726
|
+
the selected value recursively instead of dumping the complete response message;
|
|
2727
|
+
the latter also carries response-only metadata rejected by request schemas.
|
|
2728
|
+
"""
|
|
2729
|
+
if hasattr(value, "model_dump"):
|
|
2730
|
+
return value.model_dump(exclude_none=True)
|
|
2731
|
+
if isinstance(value, list):
|
|
2732
|
+
return [_request_value(item) for item in value]
|
|
2733
|
+
if isinstance(value, tuple):
|
|
2734
|
+
return [_request_value(item) for item in value]
|
|
2735
|
+
if isinstance(value, dict):
|
|
2736
|
+
return {key: _request_value(item) for key, item in value.items()}
|
|
2737
|
+
return value
|
|
2738
|
+
|
|
2739
|
+
|
|
2710
2740
|
def _assistant_dict(msg) -> dict:
|
|
2711
|
-
"""Reconstruct
|
|
2712
|
-
|
|
2713
|
-
|
|
2714
|
-
|
|
2741
|
+
"""Reconstruct the replayable subset of an assistant response.
|
|
2742
|
+
|
|
2743
|
+
Keep content and exact tool-call ids plus one deliberately allowlisted reasoning
|
|
2744
|
+
extension. Compatible providers use three aliases; when a response unexpectedly
|
|
2745
|
+
carries more than one, preserve the first in ``_REASONING_FIELDS`` order rather
|
|
2746
|
+
than sending conflicting thinking payloads back on the next request. Do not use a
|
|
2747
|
+
whole-message ``model_dump()``: response-only metadata can be rejected on replay.
|
|
2748
|
+
"""
|
|
2749
|
+
d = {"role": "assistant", "content": (_message_field(msg, "content") or "")}
|
|
2750
|
+
for field in _REASONING_FIELDS:
|
|
2751
|
+
value = _message_field(msg, field)
|
|
2752
|
+
if value is not None:
|
|
2753
|
+
d[field] = _request_value(value)
|
|
2754
|
+
break
|
|
2755
|
+
tcs = _message_field(msg, "tool_calls")
|
|
2715
2756
|
if tcs:
|
|
2716
2757
|
d["tool_calls"] = [
|
|
2717
2758
|
{
|
|
@@ -3400,9 +3441,13 @@ def _run_disposable(
|
|
|
3400
3441
|
{"role": "system", "content": sys_prompt},
|
|
3401
3442
|
{"role": "user", "content": task},
|
|
3402
3443
|
]
|
|
3444
|
+
# #128: this is a fresh conversation, not a continuation of the parent. Replace
|
|
3445
|
+
# (rather than inherit) its routing key; every call inside this disposable loop
|
|
3446
|
+
# then shares affinity, while two workers cannot collide on one backend identity.
|
|
3447
|
+
child_kwargs = _openai.with_prompt_cache_key(base_kwargs)
|
|
3403
3448
|
with _capture_stdout() as buf:
|
|
3404
3449
|
run_loop(
|
|
3405
|
-
oai, model, messages,
|
|
3450
|
+
oai, model, messages, child_kwargs, tools,
|
|
3406
3451
|
max_tool_calls=max_tool_calls, yes=True, json_out=False,
|
|
3407
3452
|
budget=budget, ledger=ledger,
|
|
3408
3453
|
)
|
|
@@ -50,6 +50,8 @@ import time
|
|
|
50
50
|
from dataclasses import dataclass
|
|
51
51
|
from typing import List, Optional, Tuple
|
|
52
52
|
|
|
53
|
+
from . import _openai
|
|
54
|
+
|
|
53
55
|
# Rough chars-per-token for English/code text. Deliberately conservative
|
|
54
56
|
# (overestimate tokens) so the fallback triggers compaction a little early
|
|
55
57
|
# rather than a little late; real counts from `usage` override it anyway.
|
|
@@ -315,7 +317,10 @@ def compact_messages(
|
|
|
315
317
|
prefix, tail = split
|
|
316
318
|
sys_msgs = messages[: len(messages) - len(prefix) - len(tail)]
|
|
317
319
|
|
|
318
|
-
|
|
320
|
+
# #128: compaction is a deliberately fresh summarization conversation. Reusing
|
|
321
|
+
# the parent session's routing identity would mix unrelated prefixes on one cache
|
|
322
|
+
# affinity key; preserve all other generation/Venice parameters while stripping it.
|
|
323
|
+
kwargs = _openai.without_prompt_cache_key(base_kwargs)
|
|
319
324
|
kwargs.pop("stream", None)
|
|
320
325
|
kwargs.pop("stream_options", None)
|
|
321
326
|
kwargs.pop("tools", None)
|
|
@@ -13,7 +13,55 @@ catalog, so the missing-SDK path never touches the network.
|
|
|
13
13
|
"""
|
|
14
14
|
from __future__ import annotations
|
|
15
15
|
|
|
16
|
+
import os
|
|
16
17
|
import sys
|
|
18
|
+
from typing import Optional
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
_PROMPT_CACHE_KEY = "prompt_cache_key"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def prompt_cache_key(kwargs: Optional[dict]) -> Optional[str]:
|
|
25
|
+
"""The OpenAI-compatible cache-affinity key carried by ``extra_body``.
|
|
26
|
+
|
|
27
|
+
The project supports ``openai>=1.40``, predating the SDK's typed
|
|
28
|
+
``prompt_cache_key`` argument. Keeping the wire field in ``extra_body`` makes it
|
|
29
|
+
available on every supported SDK version; the SDK merges it into the request's
|
|
30
|
+
top-level JSON object.
|
|
31
|
+
"""
|
|
32
|
+
extra = (kwargs or {}).get("extra_body")
|
|
33
|
+
value = extra.get(_PROMPT_CACHE_KEY) if isinstance(extra, dict) else None
|
|
34
|
+
return value if isinstance(value, str) and value else None
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def with_prompt_cache_key(kwargs: Optional[dict], key: Optional[str] = None) -> dict:
|
|
38
|
+
"""Copy generation kwargs and set one opaque prompt-cache routing key.
|
|
39
|
+
|
|
40
|
+
A missing ``key`` mints a new conversation identity. Nested ``extra_body`` is
|
|
41
|
+
copied before mutation so a disposable subagent can replace its parent's key
|
|
42
|
+
without changing the parent session or its Venice extension parameters.
|
|
43
|
+
"""
|
|
44
|
+
out = dict(kwargs or {})
|
|
45
|
+
extra = out.get("extra_body")
|
|
46
|
+
extra = dict(extra) if isinstance(extra, dict) else {}
|
|
47
|
+
extra[_PROMPT_CACHE_KEY] = key or f"venice-{os.urandom(16).hex()}"
|
|
48
|
+
out["extra_body"] = extra
|
|
49
|
+
return out
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def without_prompt_cache_key(kwargs: Optional[dict]) -> dict:
|
|
53
|
+
"""Copy generation kwargs without a parent conversation's affinity key."""
|
|
54
|
+
out = dict(kwargs or {})
|
|
55
|
+
extra = out.get("extra_body")
|
|
56
|
+
if not isinstance(extra, dict) or _PROMPT_CACHE_KEY not in extra:
|
|
57
|
+
return out
|
|
58
|
+
extra = dict(extra)
|
|
59
|
+
extra.pop(_PROMPT_CACHE_KEY, None)
|
|
60
|
+
if extra:
|
|
61
|
+
out["extra_body"] = extra
|
|
62
|
+
else:
|
|
63
|
+
out.pop("extra_body", None)
|
|
64
|
+
return out
|
|
17
65
|
|
|
18
66
|
|
|
19
67
|
def import_openai(label: str):
|
|
@@ -309,13 +309,13 @@ def _compose_in_editor(initial: str = "") -> Optional[str]:
|
|
|
309
309
|
# One turn
|
|
310
310
|
# --------------------------------------------------------------------------- #
|
|
311
311
|
def _stream_turn(oai, chat, model: str, messages: List[dict], gen_kwargs: dict):
|
|
312
|
-
"""One streamed turn; returns (
|
|
312
|
+
"""One streamed turn; returns (assistant_message, usage) for replay + budget."""
|
|
313
313
|
kwargs = dict(gen_kwargs)
|
|
314
314
|
kwargs["model"] = model
|
|
315
315
|
kwargs["messages"] = messages
|
|
316
316
|
kwargs["stream"] = True
|
|
317
317
|
kwargs["stream_options"] = {"include_usage": True}
|
|
318
|
-
return chat.
|
|
318
|
+
return chat._consume_stream_message_full(oai.chat.completions.create(**kwargs))
|
|
319
319
|
|
|
320
320
|
|
|
321
321
|
def _do_turn(oai, openai, chat, text, messages, gen_kwargs, state, args) -> None:
|
|
@@ -413,7 +413,9 @@ def _turn(oai, openai, chat, text, messages, gen_kwargs, state, args) -> bool:
|
|
|
413
413
|
)
|
|
414
414
|
else:
|
|
415
415
|
_t0 = time.monotonic()
|
|
416
|
-
|
|
416
|
+
reply_message, usage = _stream_turn(
|
|
417
|
+
oai, chat, state["model"], messages, gen_kwargs,
|
|
418
|
+
)
|
|
417
419
|
if budget is not None:
|
|
418
420
|
budget.observe(usage)
|
|
419
421
|
if ledger is not None:
|
|
@@ -422,7 +424,7 @@ def _turn(oai, openai, chat, text, messages, gen_kwargs, state, args) -> bool:
|
|
|
422
424
|
# difference is real -- do not compare a streamed row against a buffered
|
|
423
425
|
# one and conclude the provider got slower.
|
|
424
426
|
ledger.record(usage, seconds=time.monotonic() - _t0)
|
|
425
|
-
messages.append(
|
|
427
|
+
messages.append(reply_message)
|
|
426
428
|
except KeyboardInterrupt:
|
|
427
429
|
# Ctrl-C aborts just this turn -- roll it back and keep the session. With #79's
|
|
428
430
|
# attached steering this is the *second* Ctrl+C (or Ctrl+C at the steer prompt);
|
|
@@ -56,7 +56,7 @@ import threading
|
|
|
56
56
|
import time
|
|
57
57
|
from typing import Dict, List, Optional, Tuple
|
|
58
58
|
|
|
59
|
-
from . import _agent, _code, _exec
|
|
59
|
+
from . import _agent, _code, _exec, _openai
|
|
60
60
|
from ._exec import ( # shared exec rails (#33): one gate for every git shell-out
|
|
61
61
|
DEFAULT_EXEC_TIMEOUT,
|
|
62
62
|
MAX_OUTPUT_CHARS,
|
|
@@ -584,6 +584,9 @@ def _retry_for_verdict(oai, model: str, report: str, base_kwargs: dict,
|
|
|
584
584
|
it is supposed to be bounded by.
|
|
585
585
|
"""
|
|
586
586
|
_t0 = time.monotonic()
|
|
587
|
+
# #128: this retry is intentionally a fresh one-shot, not a continuation of the
|
|
588
|
+
# author or reviewer conversation, so it must not borrow either affinity identity.
|
|
589
|
+
retry_kwargs = _openai.without_prompt_cache_key(base_kwargs)
|
|
587
590
|
resp = oai.chat.completions.create(
|
|
588
591
|
model=model,
|
|
589
592
|
messages=[
|
|
@@ -591,7 +594,7 @@ def _retry_for_verdict(oai, model: str, report: str, base_kwargs: dict,
|
|
|
591
594
|
{"role": "assistant", "content": report or ""},
|
|
592
595
|
{"role": "user", "content": RETRY_MSG},
|
|
593
596
|
],
|
|
594
|
-
**
|
|
597
|
+
**retry_kwargs,
|
|
595
598
|
)
|
|
596
599
|
if ledger is not None:
|
|
597
600
|
ledger.record(getattr(resp, "usage", None), seconds=time.monotonic() - _t0)
|
|
@@ -706,9 +706,21 @@ def _consume_stream_full(stream):
|
|
|
706
706
|
The REPL's auto-compact budget (#48) observes the server-reported prompt
|
|
707
707
|
token count; printing behavior is identical to `_consume_stream`.
|
|
708
708
|
"""
|
|
709
|
+
message, usage = _consume_stream_message_full(stream)
|
|
710
|
+
return message["content"], usage
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
def _consume_stream_message_full(stream):
|
|
714
|
+
"""Consume a stream and retain its replayable assistant-message extensions.
|
|
715
|
+
|
|
716
|
+
The ordinary one-shot surface still returns/prints only visible content. The
|
|
717
|
+
multi-turn REPL uses this richer result so Kimi-style reasoning deltas survive in
|
|
718
|
+
history and the next request can replay the complete assistant turn.
|
|
719
|
+
"""
|
|
709
720
|
citations = None
|
|
710
721
|
usage = None
|
|
711
722
|
parts: list = []
|
|
723
|
+
reasoning_parts = {field: [] for field in _agent._REASONING_FIELDS}
|
|
712
724
|
for chunk in stream:
|
|
713
725
|
vp = getattr(chunk, "venice_parameters", None)
|
|
714
726
|
if vp is not None and citations is None:
|
|
@@ -716,16 +728,33 @@ def _consume_stream_full(stream):
|
|
|
716
728
|
if getattr(chunk, "usage", None):
|
|
717
729
|
usage = chunk.usage
|
|
718
730
|
if chunk.choices:
|
|
719
|
-
|
|
731
|
+
delta = chunk.choices[0].delta
|
|
732
|
+
piece = getattr(delta, "content", None)
|
|
720
733
|
if piece:
|
|
721
734
|
sys.stdout.write(piece)
|
|
722
735
|
sys.stdout.flush()
|
|
723
736
|
parts.append(piece)
|
|
737
|
+
for field in _agent._REASONING_FIELDS:
|
|
738
|
+
value = getattr(delta, field, None)
|
|
739
|
+
if value is not None:
|
|
740
|
+
reasoning_parts[field].append(_agent._request_value(value))
|
|
724
741
|
if parts:
|
|
725
742
|
sys.stdout.write("\n")
|
|
726
743
|
_print_citations(citations)
|
|
727
744
|
_print_usage(usage)
|
|
728
|
-
|
|
745
|
+
message = {"role": "assistant", "content": "".join(parts)}
|
|
746
|
+
for field in _agent._REASONING_FIELDS:
|
|
747
|
+
values = reasoning_parts[field]
|
|
748
|
+
if not values:
|
|
749
|
+
continue
|
|
750
|
+
if all(isinstance(value, str) for value in values):
|
|
751
|
+
message[field] = "".join(values)
|
|
752
|
+
elif all(isinstance(value, list) for value in values):
|
|
753
|
+
message[field] = [item for value in values for item in value]
|
|
754
|
+
else:
|
|
755
|
+
message[field] = values[0] if len(values) == 1 else values
|
|
756
|
+
break
|
|
757
|
+
return message, usage
|
|
729
758
|
|
|
730
759
|
|
|
731
760
|
def _run_stream(oai, kwargs: dict) -> int:
|
|
@@ -532,7 +532,7 @@ def _human_pause(acc):
|
|
|
532
532
|
acc[0] += time.monotonic() - t
|
|
533
533
|
|
|
534
534
|
|
|
535
|
-
def _no_tool_turn(oai, model, messages, gen_kwargs, oai_tools, *, ledger=None) ->
|
|
535
|
+
def _no_tool_turn(oai, model, messages, gen_kwargs, oai_tools, *, ledger=None) -> dict:
|
|
536
536
|
"""One completion with tools advertised but ``tool_choice="none"`` (no side
|
|
537
537
|
effects) -- used for the plan turn and the acceptance-check turn.
|
|
538
538
|
|
|
@@ -550,9 +550,8 @@ def _no_tool_turn(oai, model, messages, gen_kwargs, oai_tools, *, ledger=None) -
|
|
|
550
550
|
# transcript (see above), so they are the largest rows in the trace -- a trace
|
|
551
551
|
# whose biggest calls read `n/a` would be worse than no trace at all.
|
|
552
552
|
ledger.record(getattr(resp, "usage", None), seconds=time.monotonic() - _t0)
|
|
553
|
-
if getattr(resp, "choices", None)
|
|
554
|
-
|
|
555
|
-
return ""
|
|
553
|
+
msg = resp.choices[0].message if getattr(resp, "choices", None) else None
|
|
554
|
+
return _agent._assistant_dict(msg)
|
|
556
555
|
|
|
557
556
|
|
|
558
557
|
# Promoted to `_agent` (#52): the scout subagent firewalls its stdout the same way,
|
|
@@ -715,6 +714,16 @@ def _run(args) -> int:
|
|
|
715
714
|
# venice_scout). `_gen_kwargs` reads only args.temperature/max_tokens -- no
|
|
716
715
|
# dependency on `tools`, so the reorder is safe.
|
|
717
716
|
gen_kwargs = PROFILE.build_gen_kwargs(args)
|
|
717
|
+
# #128: Venice can route requests carrying one stable prompt_cache_key to the
|
|
718
|
+
# same cache-bearing backend. Restore the saved key when one exists; old sessions
|
|
719
|
+
# receive a key on first resume, and fresh/ephemeral runs mint one before the plan
|
|
720
|
+
# turn so plan -> execute -> verify all share affinity. It rides in extra_body for
|
|
721
|
+
# compatibility with the project's openai>=1.40 floor and is persisted automatically
|
|
722
|
+
# with the session's generation kwargs.
|
|
723
|
+
saved_cache_key = (
|
|
724
|
+
_openai.prompt_cache_key(session.gen_kwargs) if session is not None else None
|
|
725
|
+
)
|
|
726
|
+
gen_kwargs = _openai.with_prompt_cache_key(gen_kwargs, saved_cache_key)
|
|
718
727
|
# #52 planner slice: the session's shared dispatch record list. scout/spawn append
|
|
719
728
|
# every launched dispatch to it; venice_merge (and the --json envelope) roll it up.
|
|
720
729
|
dispatches = [] if planner else None
|
|
@@ -856,12 +865,13 @@ def _run_oneshot(args, oai, openai, model, tools, system, gen_kwargs, root, task
|
|
|
856
865
|
try:
|
|
857
866
|
# Inside the re-plan loop: each `edit` revision buys another plan turn,
|
|
858
867
|
# so recording per call (not once) is what makes the total honest.
|
|
859
|
-
|
|
860
|
-
|
|
868
|
+
plan_message = _no_tool_turn(oai, model, plan_messages, gen_kwargs,
|
|
869
|
+
oai_tools, ledger=ledger)
|
|
870
|
+
plan_text = plan_message.get("content") or ""
|
|
861
871
|
except openai.OpenAIError as e:
|
|
862
872
|
_finish(ledger, t0, human, json_out=args.json)
|
|
863
873
|
return _openai.status_to_exit(openai, e, "code")
|
|
864
|
-
messages.append(
|
|
874
|
+
messages.append(plan_message)
|
|
865
875
|
|
|
866
876
|
if args.plan_only:
|
|
867
877
|
_finish(ledger, t0, human, json_out=args.json)
|
|
@@ -990,14 +1000,17 @@ def _run_oneshot(args, oai, openai, model, tools, system, gen_kwargs, root, task
|
|
|
990
1000
|
if not args.no_verify and not args.no_plan:
|
|
991
1001
|
messages.append({"role": "user", "content": _VERIFY_MSG})
|
|
992
1002
|
try:
|
|
993
|
-
|
|
994
|
-
|
|
1003
|
+
report_message = _no_tool_turn(oai, model, messages, gen_kwargs, oai_tools,
|
|
1004
|
+
ledger=ledger)
|
|
1005
|
+
report = report_message.get("content") or ""
|
|
995
1006
|
parsed = _parse_verdict(report)
|
|
996
1007
|
if parsed is None: # re-prompt ONCE for the exact verdict line
|
|
997
|
-
messages.append(
|
|
1008
|
+
messages.append(report_message)
|
|
998
1009
|
messages.append({"role": "user", "content": _VERIFY_RETRY_MSG})
|
|
999
|
-
|
|
1000
|
-
|
|
1010
|
+
retry_message = _no_tool_turn(
|
|
1011
|
+
oai, model, messages, gen_kwargs, oai_tools, ledger=ledger,
|
|
1012
|
+
)
|
|
1013
|
+
retry = retry_message.get("content") or ""
|
|
1001
1014
|
report = f"{report}\n{retry}" if report else retry
|
|
1002
1015
|
parsed = _parse_verdict(retry)
|
|
1003
1016
|
except openai.OpenAIError as e:
|
|
@@ -13,6 +13,7 @@ import tempfile
|
|
|
13
13
|
import threading
|
|
14
14
|
import time
|
|
15
15
|
import unittest
|
|
16
|
+
from types import SimpleNamespace
|
|
16
17
|
from unittest import mock
|
|
17
18
|
|
|
18
19
|
from venice.commands import _agent
|
|
@@ -46,6 +47,99 @@ def _free_tool():
|
|
|
46
47
|
return _tool("t", lambda a, *, confirm=False: {"status": "ok"})
|
|
47
48
|
|
|
48
49
|
|
|
50
|
+
class TestAssistantReplay(unittest.TestCase):
|
|
51
|
+
"""#129: compatible reasoning fields survive every buffered history seam."""
|
|
52
|
+
|
|
53
|
+
def test_synthetic_sdk_message_replays_reasoning_and_exact_tool_call(self):
|
|
54
|
+
from openai.types.chat import ChatCompletion
|
|
55
|
+
|
|
56
|
+
response = ChatCompletion.model_validate({
|
|
57
|
+
"id": "chatcmpl-test",
|
|
58
|
+
"object": "chat.completion",
|
|
59
|
+
"created": 0,
|
|
60
|
+
"model": "kimi-k3",
|
|
61
|
+
"choices": [{
|
|
62
|
+
"index": 0,
|
|
63
|
+
"finish_reason": "tool_calls",
|
|
64
|
+
"message": {
|
|
65
|
+
"role": "assistant",
|
|
66
|
+
"content": "I will inspect it.",
|
|
67
|
+
"reasoning_content": "reasoning bytes: \u2603",
|
|
68
|
+
"tool_calls": [{
|
|
69
|
+
"id": "call-123",
|
|
70
|
+
"type": "function",
|
|
71
|
+
"function": {"name": "t", "arguments": '{"x":1}'},
|
|
72
|
+
}],
|
|
73
|
+
},
|
|
74
|
+
}],
|
|
75
|
+
})
|
|
76
|
+
|
|
77
|
+
self.assertEqual(_agent._assistant_dict(response.choices[0].message), {
|
|
78
|
+
"role": "assistant",
|
|
79
|
+
"content": "I will inspect it.",
|
|
80
|
+
"reasoning_content": "reasoning bytes: \u2603",
|
|
81
|
+
"tool_calls": [{
|
|
82
|
+
"id": "call-123",
|
|
83
|
+
"type": "function",
|
|
84
|
+
"function": {"name": "t", "arguments": '{"x":1}'},
|
|
85
|
+
}],
|
|
86
|
+
})
|
|
87
|
+
|
|
88
|
+
def test_alias_precedence_is_explicit_and_response_metadata_is_dropped(self):
|
|
89
|
+
msg = SimpleNamespace(
|
|
90
|
+
content="visible",
|
|
91
|
+
tool_calls=None,
|
|
92
|
+
reasoning_content="preferred",
|
|
93
|
+
reasoning_details=[{"type": "summary", "text": "other"}],
|
|
94
|
+
reasoning="last",
|
|
95
|
+
refusal="response-only",
|
|
96
|
+
)
|
|
97
|
+
self.assertEqual(_agent._assistant_dict(msg), {
|
|
98
|
+
"role": "assistant",
|
|
99
|
+
"content": "visible",
|
|
100
|
+
"reasoning_content": "preferred",
|
|
101
|
+
})
|
|
102
|
+
|
|
103
|
+
def test_non_reasoning_message_shape_is_unchanged(self):
|
|
104
|
+
msg = SimpleNamespace(content=None, tool_calls=None)
|
|
105
|
+
self.assertEqual(_agent._assistant_dict(msg), {
|
|
106
|
+
"role": "assistant", "content": "",
|
|
107
|
+
})
|
|
108
|
+
|
|
109
|
+
def test_next_tool_request_includes_reasoning_byte_for_byte(self):
|
|
110
|
+
first = FakeToolCompletion(
|
|
111
|
+
tool_calls=[_FnCall("c1", "t", "{}")],
|
|
112
|
+
reasoning_content="keep this exactly \u2603",
|
|
113
|
+
)
|
|
114
|
+
fake, calls = _fake_oai([first, FakeToolCompletion("done")])
|
|
115
|
+
messages = [{"role": "user", "content": "go"}]
|
|
116
|
+
with mock.patch.object(sys, "stdout", io.StringIO()), \
|
|
117
|
+
mock.patch.object(sys, "stderr", io.StringIO()):
|
|
118
|
+
_agent.run_loop(
|
|
119
|
+
fake, "kimi-k3", messages, {}, [_free_tool()],
|
|
120
|
+
max_tool_calls=0, yes=True, json_out=False,
|
|
121
|
+
)
|
|
122
|
+
self.assertEqual(
|
|
123
|
+
calls[1]["messages"][1]["reasoning_content"],
|
|
124
|
+
"keep this exactly \u2603",
|
|
125
|
+
)
|
|
126
|
+
|
|
127
|
+
def test_forced_final_response_keeps_reasoning_in_history(self):
|
|
128
|
+
seq = [
|
|
129
|
+
FakeToolCompletion(tool_calls=[_FnCall("c1", "t", "{}")]),
|
|
130
|
+
FakeToolCompletion("wrapped", reasoning_content="final thought"),
|
|
131
|
+
]
|
|
132
|
+
fake, _calls = _fake_oai(seq)
|
|
133
|
+
messages = [{"role": "user", "content": "go"}]
|
|
134
|
+
with mock.patch.object(sys, "stdout", io.StringIO()), \
|
|
135
|
+
mock.patch.object(sys, "stderr", io.StringIO()):
|
|
136
|
+
_agent.run_loop(
|
|
137
|
+
fake, "kimi-k3", messages, {}, [_free_tool()],
|
|
138
|
+
max_tool_calls=1, yes=True, json_out=False,
|
|
139
|
+
)
|
|
140
|
+
self.assertEqual(messages[-1]["reasoning_content"], "final thought")
|
|
141
|
+
|
|
142
|
+
|
|
49
143
|
def _tty(value=True):
|
|
50
144
|
m = mock.MagicMock()
|
|
51
145
|
m.isatty.return_value = value
|
|
@@ -95,18 +95,23 @@ class FakeCompletion:
|
|
|
95
95
|
|
|
96
96
|
|
|
97
97
|
class _Delta:
|
|
98
|
-
def __init__(self, content):
|
|
98
|
+
def __init__(self, content, **fields):
|
|
99
99
|
self.content = content
|
|
100
|
+
for name, value in fields.items():
|
|
101
|
+
setattr(self, name, value)
|
|
100
102
|
|
|
101
103
|
|
|
102
104
|
class _StreamChoice:
|
|
103
|
-
def __init__(self, content):
|
|
104
|
-
self.delta = _Delta(content)
|
|
105
|
+
def __init__(self, content, **fields):
|
|
106
|
+
self.delta = _Delta(content, **fields)
|
|
105
107
|
|
|
106
108
|
|
|
107
109
|
class FakeChunk:
|
|
108
|
-
def __init__(self, content=None, usage=None, venice_parameters=None):
|
|
109
|
-
self.choices =
|
|
110
|
+
def __init__(self, content=None, usage=None, venice_parameters=None, **delta_fields):
|
|
111
|
+
self.choices = (
|
|
112
|
+
[_StreamChoice(content, **delta_fields)]
|
|
113
|
+
if content is not None or delta_fields else []
|
|
114
|
+
)
|
|
110
115
|
self.usage = usage
|
|
111
116
|
self.venice_parameters = venice_parameters
|
|
112
117
|
|
|
@@ -166,9 +171,11 @@ class _FnCall:
|
|
|
166
171
|
|
|
167
172
|
|
|
168
173
|
class _ToolMsg:
|
|
169
|
-
def __init__(self, content=None, tool_calls=None):
|
|
174
|
+
def __init__(self, content=None, tool_calls=None, **fields):
|
|
170
175
|
self.content = content
|
|
171
176
|
self.tool_calls = tool_calls
|
|
177
|
+
for name, value in fields.items():
|
|
178
|
+
setattr(self, name, value)
|
|
172
179
|
|
|
173
180
|
|
|
174
181
|
class _ToolChoice:
|
|
@@ -180,8 +187,8 @@ class FakeToolCompletion:
|
|
|
180
187
|
"""A completion whose message may carry tool_calls (None => a final answer)."""
|
|
181
188
|
|
|
182
189
|
def __init__(self, content=None, tool_calls=None, venice_parameters=None,
|
|
183
|
-
usage=None):
|
|
184
|
-
self.choices = [_ToolChoice(_ToolMsg(content, tool_calls))]
|
|
190
|
+
usage=None, **message_fields):
|
|
191
|
+
self.choices = [_ToolChoice(_ToolMsg(content, tool_calls, **message_fields))]
|
|
185
192
|
self.venice_parameters = venice_parameters
|
|
186
193
|
self.usage = usage
|
|
187
194
|
|