patchpal 0.22.8__tar.gz → 0.23.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {patchpal-0.22.8/patchpal.egg-info → patchpal-0.23.1}/PKG-INFO +6 -2
- {patchpal-0.22.8 → patchpal-0.23.1}/README.md +5 -1
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/__init__.py +1 -1
- patchpal-0.23.1/patchpal/agent/bedrock_profile_utils.py +226 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/agent/function_calling.py +183 -43
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/agent/react.py +17 -20
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/prompts/system_prompt.md +1 -1
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/common.py +98 -1
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/definitions.py +8 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/find_tool.py +48 -15
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/repo_map.py +11 -2
- {patchpal-0.22.8 → patchpal-0.23.1/patchpal.egg-info}/PKG-INFO +6 -2
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/SOURCES.txt +2 -0
- patchpal-0.23.1/tests/test_memory_locations.py +117 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/LICENSE +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/MANIFEST.in +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/agent/__init__.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/__init__.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/autopilot.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/interactive.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/mcp.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/sandbox.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/streaming.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/config.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/context.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/permissions.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/prompts/react_prompt.md +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/skills.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/__init__.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/audit.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/code_analysis.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/file_reading.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/file_writing.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/grep_tool.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/image_handler.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/mcp.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/shell_tools.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/todo_tools.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/tool_schema.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/user_interaction.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/web_tools.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/dependency_links.txt +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/entry_points.txt +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/requires.txt +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/top_level.txt +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/pyproject.toml +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/setup.cfg +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_agent.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_cli.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_config_dynamic.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_context.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_custom_tools.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_enabled_tools.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_find_tool.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_guardrails.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_image_blocking.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_maximum_security.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_mcp_config.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_memory.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_operational_safety.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_optional_tools.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_permissions.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_react.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_reasoning_content.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_repo_map.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_simplified_prompt.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_skills.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_streaming.py +0 -0
- {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_tools.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: patchpal
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.23.1
|
|
4
4
|
Summary: An agentic coding and automation assistant, supporting both local and cloud LLMs
|
|
5
5
|
Author: PatchPal Contributors
|
|
6
6
|
License-Expression: Apache-2.0
|
|
@@ -64,7 +64,7 @@ Most agent frameworks are [built in TypeScript](https://news.ycombinator.com/ite
|
|
|
64
64
|
- [Built-In](https://amaiya.github.io/patchpal/features/tools/) and [Custom Tools](https://amaiya.github.io/patchpal/features/custom-tools/)
|
|
65
65
|
- [Skills System](https://amaiya.github.io/patchpal/features/skills/) and [MCP Integration](https://amaiya.github.io/patchpal/features/mcp/)
|
|
66
66
|
- [Autopilot Mode](https://amaiya.github.io/patchpal/usage/autopilot/) using [Ralph Wiggum loops](https://github.com/amaiya/patchpal/tree/main/examples/ralph/)
|
|
67
|
-
- [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `~/.patchpal/repos/<repo-name>/MEMORY.md` at startup.
|
|
67
|
+
- [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `MEMORY.md` (repository root or `~/.patchpal/repos/<repo-name>/MEMORY.md`) at startup.
|
|
68
68
|
|
|
69
69
|
PatchPal prioritizes customizability: custom tools, custom skills, a flexible Python API, and support for any tool-calling LLM.
|
|
70
70
|
|
|
@@ -177,6 +177,10 @@ While originally designed for software development, PatchPal is also a general-p
|
|
|
177
177
|
2. PatchPal includes a [unique guardrails system](https://amaiya.github.io/patchpal/safety/) that is better suited to privacy-conscious use cases involving sensitive data.
|
|
178
178
|
3. We needed an agent harness that seamlessly works with [both local and cloud models](https://amaiya.github.io/patchpal/models/overview/#supported-models), including AWS GovCloud Bedrock models.
|
|
179
179
|
|
|
180
|
+
> I noticed there's another package called PatchPal on GitLab. Are they related?
|
|
181
|
+
|
|
182
|
+
No, they're separate projects. The [project with the same name on GitLab](https://gitlab.com/patchpal-ai) is unrelated to this one.
|
|
183
|
+
|
|
180
184
|
> On Windows Subsystem for Linux (WSL), why is it stalling intermittently at "Thinking..."?
|
|
181
185
|
|
|
182
186
|
This is a [known issue](https://github.com/microsoft/WSL/issues/6264#issuecomment-762154193) with WSL2.
|
|
@@ -16,7 +16,7 @@ Most agent frameworks are [built in TypeScript](https://news.ycombinator.com/ite
|
|
|
16
16
|
- [Built-In](https://amaiya.github.io/patchpal/features/tools/) and [Custom Tools](https://amaiya.github.io/patchpal/features/custom-tools/)
|
|
17
17
|
- [Skills System](https://amaiya.github.io/patchpal/features/skills/) and [MCP Integration](https://amaiya.github.io/patchpal/features/mcp/)
|
|
18
18
|
- [Autopilot Mode](https://amaiya.github.io/patchpal/usage/autopilot/) using [Ralph Wiggum loops](https://github.com/amaiya/patchpal/tree/main/examples/ralph/)
|
|
19
|
-
- [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `~/.patchpal/repos/<repo-name>/MEMORY.md` at startup.
|
|
19
|
+
- [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `MEMORY.md` (repository root or `~/.patchpal/repos/<repo-name>/MEMORY.md`) at startup.
|
|
20
20
|
|
|
21
21
|
PatchPal prioritizes customizability: custom tools, custom skills, a flexible Python API, and support for any tool-calling LLM.
|
|
22
22
|
|
|
@@ -129,6 +129,10 @@ While originally designed for software development, PatchPal is also a general-p
|
|
|
129
129
|
2. PatchPal includes a [unique guardrails system](https://amaiya.github.io/patchpal/safety/) that is better suited to privacy-conscious use cases involving sensitive data.
|
|
130
130
|
3. We needed an agent harness that seamlessly works with [both local and cloud models](https://amaiya.github.io/patchpal/models/overview/#supported-models), including AWS GovCloud Bedrock models.
|
|
131
131
|
|
|
132
|
+
> I noticed there's another package called PatchPal on GitLab. Are they related?
|
|
133
|
+
|
|
134
|
+
No, they're separate projects. The [project with the same name on GitLab](https://gitlab.com/patchpal-ai) is unrelated to this one.
|
|
135
|
+
|
|
132
136
|
> On Windows Subsystem for Linux (WSL), why is it stalling intermittently at "Thinking..."?
|
|
133
137
|
|
|
134
138
|
This is a [known issue](https://github.com/microsoft/WSL/issues/6264#issuecomment-762154193) with WSL2.
|
|
@@ -0,0 +1,226 @@
|
|
|
1
|
+
"""Utilities for AWS Bedrock application inference profiles.
|
|
2
|
+
|
|
3
|
+
Application inference profiles (tagged profiles) don't include model names in their ARNs,
|
|
4
|
+
making it impossible to statically determine model capabilities or pricing. This module
|
|
5
|
+
provides functions to detect the underlying model and its capabilities at runtime.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import litellm
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _extract_model_from_arn(arn: str) -> str | None:
|
|
12
|
+
"""Try to extract underlying model info from an inference profile ARN using AWS API.
|
|
13
|
+
|
|
14
|
+
Args:
|
|
15
|
+
arn: The inference profile ARN
|
|
16
|
+
|
|
17
|
+
Returns:
|
|
18
|
+
Model name if found, None otherwise
|
|
19
|
+
"""
|
|
20
|
+
try:
|
|
21
|
+
import os
|
|
22
|
+
|
|
23
|
+
import boto3
|
|
24
|
+
|
|
25
|
+
# Extract region from ARN (arn:aws-us-gov:bedrock:us-gov-east-1:...)
|
|
26
|
+
parts = arn.split(":")
|
|
27
|
+
if len(parts) >= 4:
|
|
28
|
+
region = parts[3]
|
|
29
|
+
else:
|
|
30
|
+
region = os.getenv("AWS_REGION_NAME") or os.getenv("AWS_REGION") or "us-east-1"
|
|
31
|
+
|
|
32
|
+
# Create bedrock client
|
|
33
|
+
bedrock = boto3.client("bedrock", region_name=region)
|
|
34
|
+
|
|
35
|
+
# Get inference profile details
|
|
36
|
+
response = bedrock.get_inference_profile(inferenceProfileIdentifier=arn)
|
|
37
|
+
|
|
38
|
+
# Extract model info from response
|
|
39
|
+
if "models" in response and response["models"]:
|
|
40
|
+
# Get first model from the profile
|
|
41
|
+
first_model = response["models"][0]
|
|
42
|
+
if "modelArn" in first_model:
|
|
43
|
+
# Extract model ID from ARN
|
|
44
|
+
# e.g., arn:aws-us-gov:bedrock:us-gov-west-1::foundation-model/anthropic.claude-sonnet-4-5-20250929-v1:0
|
|
45
|
+
model_arn = first_model["modelArn"]
|
|
46
|
+
|
|
47
|
+
# Try multiple extraction methods
|
|
48
|
+
if "foundation-model" in model_arn:
|
|
49
|
+
# Split on the foundation-model part
|
|
50
|
+
parts = model_arn.split("foundation-model/")
|
|
51
|
+
if len(parts) > 1:
|
|
52
|
+
return parts[1]
|
|
53
|
+
|
|
54
|
+
return None
|
|
55
|
+
except Exception:
|
|
56
|
+
# API call failed or boto3 not available
|
|
57
|
+
return None
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def detect_model_capabilities(
|
|
61
|
+
model_id: str, litellm_kwargs: dict = None
|
|
62
|
+
) -> tuple[bool, str | None]:
|
|
63
|
+
"""Detect prompt caching support and underlying model name for application inference profiles.
|
|
64
|
+
|
|
65
|
+
This is useful for application inference profiles (tagged profiles) where the
|
|
66
|
+
underlying model name is not in the ARN, making it impossible to statically
|
|
67
|
+
determine model capabilities or pricing.
|
|
68
|
+
|
|
69
|
+
Args:
|
|
70
|
+
model_id: Full LiteLLM model identifier (e.g., bedrock/converse/arn:...)
|
|
71
|
+
litellm_kwargs: Optional kwargs to pass to litellm.completion
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
Tuple of (caching_supported: bool, model_name: str | None)
|
|
75
|
+
- caching_supported: True if prompt caching works with tools
|
|
76
|
+
- model_name: Detected model name from response metadata (e.g., "claude-3-5-sonnet-20241022")
|
|
77
|
+
"""
|
|
78
|
+
if litellm_kwargs is None:
|
|
79
|
+
litellm_kwargs = {}
|
|
80
|
+
|
|
81
|
+
# Create a minimal test message with cache markers
|
|
82
|
+
test_messages = [
|
|
83
|
+
{
|
|
84
|
+
"role": "user",
|
|
85
|
+
"content": [
|
|
86
|
+
{
|
|
87
|
+
"type": "text",
|
|
88
|
+
"text": "What is the capital of France?",
|
|
89
|
+
"cache_control": {"type": "ephemeral"},
|
|
90
|
+
}
|
|
91
|
+
],
|
|
92
|
+
}
|
|
93
|
+
]
|
|
94
|
+
|
|
95
|
+
# Include a minimal tool to match real usage (some models only support caching without tools)
|
|
96
|
+
test_tools = [
|
|
97
|
+
{
|
|
98
|
+
"type": "function",
|
|
99
|
+
"function": {
|
|
100
|
+
"name": "get_info",
|
|
101
|
+
"description": "Get information",
|
|
102
|
+
"parameters": {
|
|
103
|
+
"type": "object",
|
|
104
|
+
"properties": {"query": {"type": "string", "description": "The query"}},
|
|
105
|
+
"required": ["query"],
|
|
106
|
+
},
|
|
107
|
+
},
|
|
108
|
+
}
|
|
109
|
+
]
|
|
110
|
+
|
|
111
|
+
caching_supported = False
|
|
112
|
+
detected_model = None
|
|
113
|
+
|
|
114
|
+
# For ARNs, try to get model info from AWS API first
|
|
115
|
+
if "arn:aws" in model_id and "inference-profile" in model_id:
|
|
116
|
+
# Extract just the ARN (remove bedrock/converse/ prefix if present)
|
|
117
|
+
arn = model_id.replace("bedrock/converse/", "").replace("bedrock/", "")
|
|
118
|
+
detected_model = _extract_model_from_arn(arn)
|
|
119
|
+
|
|
120
|
+
try:
|
|
121
|
+
# Try with Anthropic-style cache_control AND tools (matches real usage)
|
|
122
|
+
response = litellm.completion(
|
|
123
|
+
model=model_id,
|
|
124
|
+
messages=test_messages,
|
|
125
|
+
tools=test_tools,
|
|
126
|
+
tool_choice="auto", # Include tool_choice like the real agent
|
|
127
|
+
max_tokens=10,
|
|
128
|
+
**litellm_kwargs,
|
|
129
|
+
)
|
|
130
|
+
# If we got here without error, caching is supported
|
|
131
|
+
caching_supported = True
|
|
132
|
+
|
|
133
|
+
# Only try to extract model from response if we didn't get it from AWS API
|
|
134
|
+
if not detected_model:
|
|
135
|
+
# Try to extract model name from response metadata
|
|
136
|
+
# Bedrock responses include model info in various places
|
|
137
|
+
if hasattr(response, "_hidden_params") and response._hidden_params:
|
|
138
|
+
# LiteLLM stores raw response data here
|
|
139
|
+
hidden = response._hidden_params
|
|
140
|
+
|
|
141
|
+
# Check optional_params which may contain raw boto3 response
|
|
142
|
+
if "optional_params" in hidden and isinstance(hidden["optional_params"], dict):
|
|
143
|
+
optional = hidden["optional_params"]
|
|
144
|
+
# Bedrock converse API may include model info in the response
|
|
145
|
+
if "model" in optional:
|
|
146
|
+
detected_model = optional["model"]
|
|
147
|
+
elif "modelId" in optional:
|
|
148
|
+
detected_model = optional["modelId"]
|
|
149
|
+
|
|
150
|
+
# Check standard fields
|
|
151
|
+
if not detected_model and "model_id" in hidden and hidden["model_id"]:
|
|
152
|
+
detected_model = hidden["model_id"]
|
|
153
|
+
elif not detected_model and "model" in hidden and hidden["model"]:
|
|
154
|
+
detected_model = hidden["model"]
|
|
155
|
+
|
|
156
|
+
# Check response metadata
|
|
157
|
+
if not detected_model and hasattr(response, "model"):
|
|
158
|
+
model_val = response.model
|
|
159
|
+
# Skip if it's just the ARN we passed in
|
|
160
|
+
if model_val and "application-inference-profile" not in model_val:
|
|
161
|
+
detected_model = model_val
|
|
162
|
+
|
|
163
|
+
# Try to extract from response choices/usage if available
|
|
164
|
+
if not detected_model and hasattr(response, "usage"):
|
|
165
|
+
usage = response.usage
|
|
166
|
+
if hasattr(usage, "model") and usage.model:
|
|
167
|
+
detected_model = usage.model
|
|
168
|
+
|
|
169
|
+
except Exception as e:
|
|
170
|
+
error_msg = str(e).lower()
|
|
171
|
+
# Check for caching-specific errors
|
|
172
|
+
if any(
|
|
173
|
+
phrase in error_msg
|
|
174
|
+
for phrase in [
|
|
175
|
+
"prompt caching",
|
|
176
|
+
"cache_control",
|
|
177
|
+
"cachepoint",
|
|
178
|
+
"unsupported model",
|
|
179
|
+
"did not allow prompt caching",
|
|
180
|
+
]
|
|
181
|
+
):
|
|
182
|
+
# Caching not supported, but still try to detect model without caching
|
|
183
|
+
caching_supported = False
|
|
184
|
+
|
|
185
|
+
# Only retry if we don't already have model from AWS API
|
|
186
|
+
if not detected_model:
|
|
187
|
+
try:
|
|
188
|
+
# Retry without cache markers to detect model
|
|
189
|
+
simple_messages = [{"role": "user", "content": "Hi"}]
|
|
190
|
+
response = litellm.completion(
|
|
191
|
+
model=model_id,
|
|
192
|
+
messages=simple_messages,
|
|
193
|
+
tools=test_tools,
|
|
194
|
+
max_tokens=5,
|
|
195
|
+
**litellm_kwargs,
|
|
196
|
+
)
|
|
197
|
+
# Try to extract model from response
|
|
198
|
+
if hasattr(response, "_hidden_params") and response._hidden_params:
|
|
199
|
+
hidden = response._hidden_params
|
|
200
|
+
if "model_id" in hidden:
|
|
201
|
+
detected_model = hidden["model_id"]
|
|
202
|
+
elif "model" in hidden:
|
|
203
|
+
detected_model = hidden["model"]
|
|
204
|
+
if not detected_model and hasattr(response, "model"):
|
|
205
|
+
detected_model = response.model
|
|
206
|
+
except Exception:
|
|
207
|
+
pass # Could not detect model
|
|
208
|
+
else:
|
|
209
|
+
# Different error (auth, network, etc.)
|
|
210
|
+
caching_supported = False
|
|
211
|
+
|
|
212
|
+
return caching_supported, detected_model
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def test_prompt_caching_support(model_id: str, litellm_kwargs: dict = None) -> bool:
|
|
216
|
+
"""Test if a model supports prompt caching (backward compatibility wrapper).
|
|
217
|
+
|
|
218
|
+
Args:
|
|
219
|
+
model_id: Full LiteLLM model identifier (e.g., bedrock/converse/arn:...)
|
|
220
|
+
litellm_kwargs: Optional kwargs to pass to litellm.completion
|
|
221
|
+
|
|
222
|
+
Returns:
|
|
223
|
+
True if prompt caching is supported, False otherwise
|
|
224
|
+
"""
|
|
225
|
+
caching_supported, _ = detect_model_capabilities(model_id, litellm_kwargs)
|
|
226
|
+
return caching_supported
|
|
@@ -28,14 +28,36 @@ LLM_TIMEOUT = config.LLM_TIMEOUT
|
|
|
28
28
|
|
|
29
29
|
|
|
30
30
|
def _is_bedrock_arn(model_id: str) -> bool:
|
|
31
|
-
"""Check if a model ID is a Bedrock ARN.
|
|
31
|
+
"""Check if a model ID is a Bedrock ARN.
|
|
32
|
+
|
|
33
|
+
Supports all Bedrock inference profile ARN formats:
|
|
34
|
+
- arn:aws:bedrock:region:account:inference-profile/profile-id
|
|
35
|
+
- arn:aws-us-gov:bedrock:region:account:inference-profile/profile-id
|
|
36
|
+
- arn:aws:bedrock:region:account:application-inference-profile/app-id
|
|
37
|
+
- arn:aws-us-gov:bedrock:region:account:application-inference-profile/app-id
|
|
38
|
+
"""
|
|
32
39
|
return (
|
|
33
40
|
model_id.startswith("arn:aws")
|
|
34
41
|
and ":bedrock:" in model_id
|
|
35
|
-
and "
|
|
42
|
+
and "inference-profile/" in model_id
|
|
36
43
|
)
|
|
37
44
|
|
|
38
45
|
|
|
46
|
+
def _is_application_inference_profile(model_id: str) -> bool:
|
|
47
|
+
"""Check if a model ID is a Bedrock application inference profile (tagged profile).
|
|
48
|
+
|
|
49
|
+
Application inference profiles are tagged profiles that don't include the underlying
|
|
50
|
+
model name in the ARN, making it impossible to statically determine model capabilities.
|
|
51
|
+
|
|
52
|
+
Args:
|
|
53
|
+
model_id: Model identifier (may or may not have bedrock/ prefix)
|
|
54
|
+
|
|
55
|
+
Returns:
|
|
56
|
+
True if this is an application inference profile ARN
|
|
57
|
+
"""
|
|
58
|
+
return ":application-inference-profile/" in model_id
|
|
59
|
+
|
|
60
|
+
|
|
39
61
|
def _normalize_bedrock_model_id(model_id: str) -> str:
|
|
40
62
|
"""Normalize Bedrock model ID to ensure it has the bedrock/ prefix.
|
|
41
63
|
|
|
@@ -51,7 +73,11 @@ def _normalize_bedrock_model_id(model_id: str) -> str:
|
|
|
51
73
|
|
|
52
74
|
# If it looks like a Bedrock ARN, add the prefix
|
|
53
75
|
if _is_bedrock_arn(model_id):
|
|
54
|
-
|
|
76
|
+
# Application inference profiles require the converse API
|
|
77
|
+
if ":application-inference-profile/" in model_id:
|
|
78
|
+
return f"bedrock/converse/{model_id}"
|
|
79
|
+
else:
|
|
80
|
+
return f"bedrock/{model_id}"
|
|
55
81
|
|
|
56
82
|
# If it's a standard Bedrock model ID (e.g., anthropic.claude-v2)
|
|
57
83
|
# Check if it looks like a Bedrock model format
|
|
@@ -274,6 +300,10 @@ def _supports_prompt_caching(model_id: str) -> bool:
|
|
|
274
300
|
# Bedrock Nova models support caching
|
|
275
301
|
if model_id.startswith("bedrock/") and "amazon.nova" in model_id.lower():
|
|
276
302
|
return True
|
|
303
|
+
# Bedrock ARNs (all types): enable caching and let Bedrock handle it
|
|
304
|
+
# If the underlying model doesn't support caching, Bedrock will ignore the markers
|
|
305
|
+
if model_id.startswith("bedrock/") and "inference-profile/" in model_id:
|
|
306
|
+
return True
|
|
277
307
|
return False
|
|
278
308
|
|
|
279
309
|
|
|
@@ -301,12 +331,14 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
|
|
|
301
331
|
|
|
302
332
|
# Determine cache marker format based on provider
|
|
303
333
|
# Anthropic models (direct or via Bedrock) use cache_control
|
|
304
|
-
#
|
|
305
|
-
|
|
306
|
-
|
|
334
|
+
# Nova models use cachePoint
|
|
335
|
+
# For Bedrock ARNs without model name, default to cache_control (most common)
|
|
336
|
+
if model_id.startswith("bedrock/") and "amazon.nova" in model_id.lower():
|
|
337
|
+
# Nova models explicitly use cachePoint
|
|
307
338
|
cache_marker = {"cachePoint": {"type": "default"}}
|
|
308
339
|
else:
|
|
309
340
|
# Anthropic models (direct or via Bedrock) use cache_control
|
|
341
|
+
# Also default for Bedrock ARNs (most use Anthropic/Claude)
|
|
310
342
|
cache_marker = {"cache_control": {"type": "ephemeral"}}
|
|
311
343
|
|
|
312
344
|
# Count existing cache markers across all messages
|
|
@@ -328,8 +360,30 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
|
|
|
328
360
|
max_cache_markers = 4
|
|
329
361
|
available_slots = max_cache_markers - existing_cache_count
|
|
330
362
|
|
|
363
|
+
# Debug logging
|
|
364
|
+
import os
|
|
365
|
+
|
|
366
|
+
if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
|
|
367
|
+
print("\n[CACHE DEBUG] _apply_prompt_caching called")
|
|
368
|
+
print(f" Total messages: {len(messages)}")
|
|
369
|
+
print(f" Existing cache markers: {existing_cache_count}")
|
|
370
|
+
print(f" Available slots: {available_slots}")
|
|
371
|
+
# Show which messages have markers
|
|
372
|
+
marked_indices = []
|
|
373
|
+
for i, msg in enumerate(messages):
|
|
374
|
+
if isinstance(msg.get("content"), list):
|
|
375
|
+
for block in msg["content"]:
|
|
376
|
+
if isinstance(block, dict) and (
|
|
377
|
+
"cache_control" in block or "cachePoint" in block
|
|
378
|
+
):
|
|
379
|
+
marked_indices.append(i)
|
|
380
|
+
break
|
|
381
|
+
print(f" Messages with existing markers: {marked_indices}")
|
|
382
|
+
|
|
331
383
|
if available_slots <= 0:
|
|
332
384
|
# Already at or over limit, don't add any more cache markers
|
|
385
|
+
if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
|
|
386
|
+
print(" Result: No slots available, returning without changes")
|
|
333
387
|
return messages
|
|
334
388
|
|
|
335
389
|
# Find system messages (usually at the start)
|
|
@@ -341,6 +395,9 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
|
|
|
341
395
|
non_system_messages[-2:] if len(non_system_messages) >= 2 else non_system_messages
|
|
342
396
|
)
|
|
343
397
|
|
|
398
|
+
if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
|
|
399
|
+
print(f" Last 2 non-system indices: {last_two_indices}")
|
|
400
|
+
|
|
344
401
|
# Build list of candidate indices to cache (prioritize system messages)
|
|
345
402
|
candidates = []
|
|
346
403
|
|
|
@@ -373,6 +430,12 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
|
|
|
373
430
|
if not has_cache:
|
|
374
431
|
candidates.append(idx)
|
|
375
432
|
|
|
433
|
+
import os
|
|
434
|
+
|
|
435
|
+
if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
|
|
436
|
+
print(f" Candidates for marking: {candidates}")
|
|
437
|
+
print(f" Will mark first {available_slots} candidates")
|
|
438
|
+
|
|
376
439
|
# Apply cache markers to candidates, respecting the available slots
|
|
377
440
|
# This ensures we never exceed the 4 marker limit
|
|
378
441
|
for idx in candidates[:available_slots]:
|
|
@@ -393,6 +456,20 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
|
|
|
393
456
|
"content": [{"type": "text", "text": content_text, **cache_marker}],
|
|
394
457
|
}
|
|
395
458
|
|
|
459
|
+
if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
|
|
460
|
+
# Show final state
|
|
461
|
+
final_marked = []
|
|
462
|
+
for i, msg in enumerate(messages):
|
|
463
|
+
if isinstance(msg.get("content"), list):
|
|
464
|
+
for block in msg["content"]:
|
|
465
|
+
if isinstance(block, dict) and (
|
|
466
|
+
"cache_control" in block or "cachePoint" in block
|
|
467
|
+
):
|
|
468
|
+
final_marked.append(i)
|
|
469
|
+
break
|
|
470
|
+
print(f" Final messages with markers: {final_marked}")
|
|
471
|
+
print(f" Markers applied to indices: {[c for c in candidates[:available_slots]]}")
|
|
472
|
+
|
|
396
473
|
return messages
|
|
397
474
|
|
|
398
475
|
|
|
@@ -509,6 +586,49 @@ class PatchPalAgent:
|
|
|
509
586
|
if litellm_kwargs:
|
|
510
587
|
self.litellm_kwargs.update(litellm_kwargs)
|
|
511
588
|
|
|
589
|
+
# Detect capabilities for application inference profiles
|
|
590
|
+
# These ARNs don't include model names, so we test with a minimal request
|
|
591
|
+
self.prompt_caching_supported = None # None = untested, True/False after test
|
|
592
|
+
self.detected_model_name = None # Store detected model for cost tracking
|
|
593
|
+
if _is_application_inference_profile(self.model_id):
|
|
594
|
+
# Test capabilities with a minimal request
|
|
595
|
+
try:
|
|
596
|
+
from patchpal.agent.bedrock_profile_utils import detect_model_capabilities
|
|
597
|
+
|
|
598
|
+
print("\033[2mℹ️ Detecting model capabilities...\033[0m", flush=True)
|
|
599
|
+
self.prompt_caching_supported, self.detected_model_name = detect_model_capabilities(
|
|
600
|
+
self.model_id, self.litellm_kwargs
|
|
601
|
+
)
|
|
602
|
+
if self.prompt_caching_supported:
|
|
603
|
+
print("\033[2m✓ Prompt caching is supported\033[0m", flush=True)
|
|
604
|
+
else:
|
|
605
|
+
print("\033[2m✗ Prompt caching is not supported\033[0m", flush=True)
|
|
606
|
+
|
|
607
|
+
if self.detected_model_name:
|
|
608
|
+
print(f"\033[2m✓ Detected model: {self.detected_model_name}\033[0m", flush=True)
|
|
609
|
+
|
|
610
|
+
# Update context limit based on detected model
|
|
611
|
+
try:
|
|
612
|
+
model_info = litellm.get_model_info(f"bedrock/{self.detected_model_name}")
|
|
613
|
+
max_input = model_info.get("max_input_tokens")
|
|
614
|
+
if max_input and isinstance(max_input, (int, float)) and max_input > 0:
|
|
615
|
+
self.context_manager.context_limit = int(max_input)
|
|
616
|
+
except Exception:
|
|
617
|
+
pass # Keep default limit
|
|
618
|
+
else:
|
|
619
|
+
print(
|
|
620
|
+
"\033[2m⚠ Could not detect underlying model name (cost tracking may be inaccurate)\033[0m",
|
|
621
|
+
flush=True,
|
|
622
|
+
)
|
|
623
|
+
except Exception:
|
|
624
|
+
# If test fails, assume caching not supported and model unknown
|
|
625
|
+
self.prompt_caching_supported = False
|
|
626
|
+
self.detected_model_name = None
|
|
627
|
+
elif _supports_prompt_caching(self.model_id):
|
|
628
|
+
self.prompt_caching_supported = True
|
|
629
|
+
else:
|
|
630
|
+
self.prompt_caching_supported = False
|
|
631
|
+
|
|
512
632
|
# Load MEMORY.md if it exists and has non-template content
|
|
513
633
|
self._load_project_memory()
|
|
514
634
|
|
|
@@ -523,38 +643,33 @@ class PatchPalAgent:
|
|
|
523
643
|
def _load_project_memory(self):
|
|
524
644
|
"""Load MEMORY.md file at session start if it has non-template content."""
|
|
525
645
|
try:
|
|
526
|
-
from patchpal.tools.common import
|
|
646
|
+
from patchpal.tools.common import get_memory_info
|
|
527
647
|
|
|
528
|
-
|
|
529
|
-
if not MEMORY_FILE.exists():
|
|
530
|
-
return
|
|
531
|
-
|
|
532
|
-
memory_content = MEMORY_FILE.read_text(encoding="utf-8")
|
|
648
|
+
info = get_memory_info()
|
|
533
649
|
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
if "---" in memory_content:
|
|
537
|
-
parts = memory_content.split("---", 1)
|
|
538
|
-
if len(parts) > 1:
|
|
539
|
-
user_content = parts[1].strip()
|
|
540
|
-
if user_content and len(user_content) > 10:
|
|
541
|
-
has_user_content = True
|
|
650
|
+
if not info["exists"]:
|
|
651
|
+
return
|
|
542
652
|
|
|
543
653
|
# Build the message - include full content if user added info, otherwise just location
|
|
544
|
-
if
|
|
654
|
+
if info["has_content"]:
|
|
545
655
|
memory_msg = f"""# Project Memory (from MEMORY.md)
|
|
546
656
|
|
|
547
|
-
{
|
|
657
|
+
{info["content"]}
|
|
548
658
|
|
|
549
|
-
The information above is from {
|
|
550
|
-
To update it, use edit_file("{
|
|
659
|
+
The information above is from MEMORY.md ({info["location_note"]}) and persists across sessions.
|
|
660
|
+
To update it, use edit_file("{info["path"]}", ...) or write_file("{info["path"]}", ...)."""
|
|
551
661
|
else:
|
|
552
662
|
# Empty template - just inform agent
|
|
553
663
|
memory_msg = f"""# Project Memory (MEMORY.md)
|
|
554
664
|
|
|
555
|
-
Your project memory file is located at: {
|
|
665
|
+
Your project memory file is located at: {info["path"]}
|
|
556
666
|
|
|
557
|
-
It's currently empty (just the template). The file is automatically loaded at session start.
|
|
667
|
+
It's currently empty (just the template). The file is automatically loaded at session start.
|
|
668
|
+
|
|
669
|
+
Note: You can use MEMORY.md in either location:
|
|
670
|
+
- Repository root: MEMORY.md (can be version controlled)
|
|
671
|
+
- Home directory: ~/.patchpal/repos/<repo-name>/MEMORY.md (default)
|
|
672
|
+
Repository root takes priority if both exist."""
|
|
558
673
|
|
|
559
674
|
# Add as a system message at the start
|
|
560
675
|
self.messages.insert(
|
|
@@ -847,7 +962,17 @@ It's currently empty (just the template). The file is automatically loaded at se
|
|
|
847
962
|
float: The calculated cost in dollars
|
|
848
963
|
"""
|
|
849
964
|
try:
|
|
850
|
-
|
|
965
|
+
# For application inference profiles, use detected model name for pricing
|
|
966
|
+
model_for_pricing = self.model_id
|
|
967
|
+
if _is_application_inference_profile(self.model_id) and self.detected_model_name:
|
|
968
|
+
# Map detected model name to a pricing model
|
|
969
|
+
# e.g., "anthropic.claude-3-5-sonnet-20241022-v2:0" -> "bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0"
|
|
970
|
+
if not self.detected_model_name.startswith("bedrock/"):
|
|
971
|
+
model_for_pricing = f"bedrock/{self.detected_model_name}"
|
|
972
|
+
else:
|
|
973
|
+
model_for_pricing = self.detected_model_name
|
|
974
|
+
|
|
975
|
+
model_info = litellm.get_model_info(model_for_pricing)
|
|
851
976
|
input_cost_per_token = model_info.get("input_cost_per_token", 0)
|
|
852
977
|
output_cost_per_token = model_info.get("output_cost_per_token", 0)
|
|
853
978
|
|
|
@@ -881,19 +1006,26 @@ It's currently empty (just the template). The file is automatically loaded at se
|
|
|
881
1006
|
cost += cache_read_tokens * input_cost_per_token * 0.1
|
|
882
1007
|
|
|
883
1008
|
# Handle OpenAI cache pricing (prompt_tokens_details.cached_tokens)
|
|
1009
|
+
# IMPORTANT: For Bedrock, LiteLLM populates prompt_tokens_details.cached_tokens
|
|
1010
|
+
# with cache_read_input_tokens for compatibility, but we already handled those above.
|
|
1011
|
+
# Only process this field if we're NOT using Bedrock-style cache fields.
|
|
884
1012
|
openai_cached_tokens = 0
|
|
885
|
-
if
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
|
|
892
|
-
|
|
893
|
-
|
|
894
|
-
|
|
895
|
-
#
|
|
896
|
-
|
|
1013
|
+
if not (cache_creation_tokens or cache_read_tokens): # Only for non-Bedrock models
|
|
1014
|
+
if (
|
|
1015
|
+
hasattr(usage, "prompt_tokens_details")
|
|
1016
|
+
and usage.prompt_tokens_details is not None
|
|
1017
|
+
):
|
|
1018
|
+
prompt_details = usage.prompt_tokens_details
|
|
1019
|
+
if hasattr(prompt_details, "cached_tokens") and prompt_details.cached_tokens:
|
|
1020
|
+
# Ensure cached_tokens is a number, not a mock or None
|
|
1021
|
+
if isinstance(prompt_details.cached_tokens, (int, float)):
|
|
1022
|
+
openai_cached_tokens = prompt_details.cached_tokens
|
|
1023
|
+
# Use cached_input_cost_per_token if available, otherwise fallback to 0.5x multiplier
|
|
1024
|
+
if cached_input_cost_per_token > 0:
|
|
1025
|
+
cost += openai_cached_tokens * cached_input_cost_per_token
|
|
1026
|
+
else:
|
|
1027
|
+
# Fallback: OpenAI cached tokens typically cost 50% of regular input
|
|
1028
|
+
cost += openai_cached_tokens * input_cost_per_token * 0.5
|
|
897
1029
|
|
|
898
1030
|
# Regular input tokens (excluding all cache tokens)
|
|
899
1031
|
regular_input = (
|
|
@@ -1048,8 +1180,10 @@ It's currently empty (just the template). The file is automatically loaded at se
|
|
|
1048
1180
|
# Filter images if BLOCK_IMAGES is enabled (for non-vision models or user preference)
|
|
1049
1181
|
messages = self.image_handler.filter_images_if_blocked(messages)
|
|
1050
1182
|
|
|
1051
|
-
# Apply prompt caching for supported models
|
|
1052
|
-
|
|
1183
|
+
# Apply prompt caching for supported models
|
|
1184
|
+
# Check instance variable for dynamically-tested models (application inference profiles)
|
|
1185
|
+
if self.prompt_caching_supported:
|
|
1186
|
+
messages = _apply_prompt_caching(messages, self.model_id)
|
|
1053
1187
|
|
|
1054
1188
|
# Use LiteLLM for all providers
|
|
1055
1189
|
try:
|
|
@@ -1260,6 +1394,7 @@ It's currently empty (just the template). The file is automatically loaded at se
|
|
|
1260
1394
|
)
|
|
1261
1395
|
elif tool_name == "get_repo_map":
|
|
1262
1396
|
max_files = tool_args.get("max_files", 100)
|
|
1397
|
+
max_depth = tool_args.get("max_depth")
|
|
1263
1398
|
patterns = ""
|
|
1264
1399
|
if tool_args.get("include_patterns"):
|
|
1265
1400
|
patterns = (
|
|
@@ -1269,8 +1404,9 @@ It's currently empty (just the template). The file is automatically loaded at se
|
|
|
1269
1404
|
patterns = (
|
|
1270
1405
|
f" (exclude: {', '.join(tool_args['exclude_patterns'])})"
|
|
1271
1406
|
)
|
|
1407
|
+
depth_info = f", depth≤{max_depth}" if max_depth is not None else ""
|
|
1272
1408
|
print(
|
|
1273
|
-
f"\033[2m🗺️ Generating repository map (max {max_files} files{patterns})...\033[0m",
|
|
1409
|
+
f"\033[2m🗺️ Generating repository map (max {max_files} files{depth_info}{patterns})...\033[0m",
|
|
1274
1410
|
flush=True,
|
|
1275
1411
|
)
|
|
1276
1412
|
elif tool_name == "get_file_info":
|
|
@@ -1295,8 +1431,12 @@ It's currently empty (just the template). The file is automatically loaded at se
|
|
|
1295
1431
|
)
|
|
1296
1432
|
elif tool_name == "find":
|
|
1297
1433
|
pattern_desc = tool_args.get("pattern", "*")
|
|
1434
|
+
max_depth = tool_args.get("max_depth")
|
|
1435
|
+
depth_info = (
|
|
1436
|
+
f" (depth≤{max_depth})" if max_depth is not None else ""
|
|
1437
|
+
)
|
|
1298
1438
|
print(
|
|
1299
|
-
f"\033[2m📂 Finding files: {pattern_desc}\033[0m",
|
|
1439
|
+
f"\033[2m📂 Finding files: {pattern_desc}{depth_info}\033[0m",
|
|
1300
1440
|
flush=True,
|
|
1301
1441
|
)
|
|
1302
1442
|
elif tool_name == "list_skills":
|