patchpal 0.22.8__tar.gz → 0.23.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. {patchpal-0.22.8/patchpal.egg-info → patchpal-0.23.1}/PKG-INFO +6 -2
  2. {patchpal-0.22.8 → patchpal-0.23.1}/README.md +5 -1
  3. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/__init__.py +1 -1
  4. patchpal-0.23.1/patchpal/agent/bedrock_profile_utils.py +226 -0
  5. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/agent/function_calling.py +183 -43
  6. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/agent/react.py +17 -20
  7. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/prompts/system_prompt.md +1 -1
  8. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/common.py +98 -1
  9. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/definitions.py +8 -0
  10. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/find_tool.py +48 -15
  11. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/repo_map.py +11 -2
  12. {patchpal-0.22.8 → patchpal-0.23.1/patchpal.egg-info}/PKG-INFO +6 -2
  13. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/SOURCES.txt +2 -0
  14. patchpal-0.23.1/tests/test_memory_locations.py +117 -0
  15. {patchpal-0.22.8 → patchpal-0.23.1}/LICENSE +0 -0
  16. {patchpal-0.22.8 → patchpal-0.23.1}/MANIFEST.in +0 -0
  17. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/agent/__init__.py +0 -0
  18. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/__init__.py +0 -0
  19. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/autopilot.py +0 -0
  20. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/interactive.py +0 -0
  21. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/mcp.py +0 -0
  22. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/sandbox.py +0 -0
  23. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/cli/streaming.py +0 -0
  24. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/config.py +0 -0
  25. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/context.py +0 -0
  26. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/permissions.py +0 -0
  27. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/prompts/react_prompt.md +0 -0
  28. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/skills.py +0 -0
  29. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/__init__.py +0 -0
  30. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/audit.py +0 -0
  31. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/code_analysis.py +0 -0
  32. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/file_reading.py +0 -0
  33. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/file_writing.py +0 -0
  34. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/grep_tool.py +0 -0
  35. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/image_handler.py +0 -0
  36. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/mcp.py +0 -0
  37. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/shell_tools.py +0 -0
  38. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/todo_tools.py +0 -0
  39. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/tool_schema.py +0 -0
  40. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/user_interaction.py +0 -0
  41. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal/tools/web_tools.py +0 -0
  42. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/dependency_links.txt +0 -0
  43. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/entry_points.txt +0 -0
  44. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/requires.txt +0 -0
  45. {patchpal-0.22.8 → patchpal-0.23.1}/patchpal.egg-info/top_level.txt +0 -0
  46. {patchpal-0.22.8 → patchpal-0.23.1}/pyproject.toml +0 -0
  47. {patchpal-0.22.8 → patchpal-0.23.1}/setup.cfg +0 -0
  48. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_agent.py +0 -0
  49. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_cli.py +0 -0
  50. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_config_dynamic.py +0 -0
  51. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_context.py +0 -0
  52. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_custom_tools.py +0 -0
  53. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_enabled_tools.py +0 -0
  54. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_find_tool.py +0 -0
  55. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_guardrails.py +0 -0
  56. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_image_blocking.py +0 -0
  57. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_maximum_security.py +0 -0
  58. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_mcp_config.py +0 -0
  59. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_memory.py +0 -0
  60. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_operational_safety.py +0 -0
  61. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_optional_tools.py +0 -0
  62. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_permissions.py +0 -0
  63. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_react.py +0 -0
  64. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_reasoning_content.py +0 -0
  65. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_repo_map.py +0 -0
  66. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_simplified_prompt.py +0 -0
  67. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_skills.py +0 -0
  68. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_streaming.py +0 -0
  69. {patchpal-0.22.8 → patchpal-0.23.1}/tests/test_tools.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: patchpal
3
- Version: 0.22.8
3
+ Version: 0.23.1
4
4
  Summary: An agentic coding and automation assistant, supporting both local and cloud LLMs
5
5
  Author: PatchPal Contributors
6
6
  License-Expression: Apache-2.0
@@ -64,7 +64,7 @@ Most agent frameworks are [built in TypeScript](https://news.ycombinator.com/ite
64
64
  - [Built-In](https://amaiya.github.io/patchpal/features/tools/) and [Custom Tools](https://amaiya.github.io/patchpal/features/custom-tools/)
65
65
  - [Skills System](https://amaiya.github.io/patchpal/features/skills/) and [MCP Integration](https://amaiya.github.io/patchpal/features/mcp/)
66
66
  - [Autopilot Mode](https://amaiya.github.io/patchpal/usage/autopilot/) using [Ralph Wiggum loops](https://github.com/amaiya/patchpal/tree/main/examples/ralph/)
67
- - [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `~/.patchpal/repos/<repo-name>/MEMORY.md` at startup.
67
+ - [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `MEMORY.md` (repository root or `~/.patchpal/repos/<repo-name>/MEMORY.md`) at startup.
68
68
 
69
69
  PatchPal prioritizes customizability: custom tools, custom skills, a flexible Python API, and support for any tool-calling LLM.
70
70
 
@@ -177,6 +177,10 @@ While originally designed for software development, PatchPal is also a general-p
177
177
  2. PatchPal includes a [unique guardrails system](https://amaiya.github.io/patchpal/safety/) that is better suited to privacy-conscious use cases involving sensitive data.
178
178
  3. We needed an agent harness that seamlessly works with [both local and cloud models](https://amaiya.github.io/patchpal/models/overview/#supported-models), including AWS GovCloud Bedrock models.
179
179
 
180
+ > I noticed there's another package called PatchPal on GitLab. Are they related?
181
+
182
+ No, they're separate projects. The [project with the same name on GitLab](https://gitlab.com/patchpal-ai) is unrelated to this one.
183
+
180
184
  > On Windows Subsystem for Linux (WSL), why is it stalling intermittently at "Thinking..."?
181
185
 
182
186
  This is a [known issue](https://github.com/microsoft/WSL/issues/6264#issuecomment-762154193) with WSL2.
@@ -16,7 +16,7 @@ Most agent frameworks are [built in TypeScript](https://news.ycombinator.com/ite
16
16
  - [Built-In](https://amaiya.github.io/patchpal/features/tools/) and [Custom Tools](https://amaiya.github.io/patchpal/features/custom-tools/)
17
17
  - [Skills System](https://amaiya.github.io/patchpal/features/skills/) and [MCP Integration](https://amaiya.github.io/patchpal/features/mcp/)
18
18
  - [Autopilot Mode](https://amaiya.github.io/patchpal/usage/autopilot/) using [Ralph Wiggum loops](https://github.com/amaiya/patchpal/tree/main/examples/ralph/)
19
- - [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `~/.patchpal/repos/<repo-name>/MEMORY.md` at startup.
19
+ - [Project Memory](https://amaiya.github.io/patchpal/features/memory/) automatically loads project context from `MEMORY.md` (repository root or `~/.patchpal/repos/<repo-name>/MEMORY.md`) at startup.
20
20
 
21
21
  PatchPal prioritizes customizability: custom tools, custom skills, a flexible Python API, and support for any tool-calling LLM.
22
22
 
@@ -129,6 +129,10 @@ While originally designed for software development, PatchPal is also a general-p
129
129
  2. PatchPal includes a [unique guardrails system](https://amaiya.github.io/patchpal/safety/) that is better suited to privacy-conscious use cases involving sensitive data.
130
130
  3. We needed an agent harness that seamlessly works with [both local and cloud models](https://amaiya.github.io/patchpal/models/overview/#supported-models), including AWS GovCloud Bedrock models.
131
131
 
132
+ > I noticed there's another package called PatchPal on GitLab. Are they related?
133
+
134
+ No, they're separate projects. The [project with the same name on GitLab](https://gitlab.com/patchpal-ai) is unrelated to this one.
135
+
132
136
  > On Windows Subsystem for Linux (WSL), why is it stalling intermittently at "Thinking..."?
133
137
 
134
138
  This is a [known issue](https://github.com/microsoft/WSL/issues/6264#issuecomment-762154193) with WSL2.
@@ -1,6 +1,6 @@
1
1
  """PatchPal - An open-source Claude Code clone implemented purely in Python."""
2
2
 
3
- __version__ = "0.22.8"
3
+ __version__ = "0.23.1"
4
4
 
5
5
  from patchpal.agent import create_agent, create_react_agent
6
6
  from patchpal.cli.autopilot import autopilot_loop
@@ -0,0 +1,226 @@
1
+ """Utilities for AWS Bedrock application inference profiles.
2
+
3
+ Application inference profiles (tagged profiles) don't include model names in their ARNs,
4
+ making it impossible to statically determine model capabilities or pricing. This module
5
+ provides functions to detect the underlying model and its capabilities at runtime.
6
+ """
7
+
8
+ import litellm
9
+
10
+
11
+ def _extract_model_from_arn(arn: str) -> str | None:
12
+ """Try to extract underlying model info from an inference profile ARN using AWS API.
13
+
14
+ Args:
15
+ arn: The inference profile ARN
16
+
17
+ Returns:
18
+ Model name if found, None otherwise
19
+ """
20
+ try:
21
+ import os
22
+
23
+ import boto3
24
+
25
+ # Extract region from ARN (arn:aws-us-gov:bedrock:us-gov-east-1:...)
26
+ parts = arn.split(":")
27
+ if len(parts) >= 4:
28
+ region = parts[3]
29
+ else:
30
+ region = os.getenv("AWS_REGION_NAME") or os.getenv("AWS_REGION") or "us-east-1"
31
+
32
+ # Create bedrock client
33
+ bedrock = boto3.client("bedrock", region_name=region)
34
+
35
+ # Get inference profile details
36
+ response = bedrock.get_inference_profile(inferenceProfileIdentifier=arn)
37
+
38
+ # Extract model info from response
39
+ if "models" in response and response["models"]:
40
+ # Get first model from the profile
41
+ first_model = response["models"][0]
42
+ if "modelArn" in first_model:
43
+ # Extract model ID from ARN
44
+ # e.g., arn:aws-us-gov:bedrock:us-gov-west-1::foundation-model/anthropic.claude-sonnet-4-5-20250929-v1:0
45
+ model_arn = first_model["modelArn"]
46
+
47
+ # Try multiple extraction methods
48
+ if "foundation-model" in model_arn:
49
+ # Split on the foundation-model part
50
+ parts = model_arn.split("foundation-model/")
51
+ if len(parts) > 1:
52
+ return parts[1]
53
+
54
+ return None
55
+ except Exception:
56
+ # API call failed or boto3 not available
57
+ return None
58
+
59
+
60
+ def detect_model_capabilities(
61
+ model_id: str, litellm_kwargs: dict = None
62
+ ) -> tuple[bool, str | None]:
63
+ """Detect prompt caching support and underlying model name for application inference profiles.
64
+
65
+ This is useful for application inference profiles (tagged profiles) where the
66
+ underlying model name is not in the ARN, making it impossible to statically
67
+ determine model capabilities or pricing.
68
+
69
+ Args:
70
+ model_id: Full LiteLLM model identifier (e.g., bedrock/converse/arn:...)
71
+ litellm_kwargs: Optional kwargs to pass to litellm.completion
72
+
73
+ Returns:
74
+ Tuple of (caching_supported: bool, model_name: str | None)
75
+ - caching_supported: True if prompt caching works with tools
76
+ - model_name: Detected model name from response metadata (e.g., "claude-3-5-sonnet-20241022")
77
+ """
78
+ if litellm_kwargs is None:
79
+ litellm_kwargs = {}
80
+
81
+ # Create a minimal test message with cache markers
82
+ test_messages = [
83
+ {
84
+ "role": "user",
85
+ "content": [
86
+ {
87
+ "type": "text",
88
+ "text": "What is the capital of France?",
89
+ "cache_control": {"type": "ephemeral"},
90
+ }
91
+ ],
92
+ }
93
+ ]
94
+
95
+ # Include a minimal tool to match real usage (some models only support caching without tools)
96
+ test_tools = [
97
+ {
98
+ "type": "function",
99
+ "function": {
100
+ "name": "get_info",
101
+ "description": "Get information",
102
+ "parameters": {
103
+ "type": "object",
104
+ "properties": {"query": {"type": "string", "description": "The query"}},
105
+ "required": ["query"],
106
+ },
107
+ },
108
+ }
109
+ ]
110
+
111
+ caching_supported = False
112
+ detected_model = None
113
+
114
+ # For ARNs, try to get model info from AWS API first
115
+ if "arn:aws" in model_id and "inference-profile" in model_id:
116
+ # Extract just the ARN (remove bedrock/converse/ prefix if present)
117
+ arn = model_id.replace("bedrock/converse/", "").replace("bedrock/", "")
118
+ detected_model = _extract_model_from_arn(arn)
119
+
120
+ try:
121
+ # Try with Anthropic-style cache_control AND tools (matches real usage)
122
+ response = litellm.completion(
123
+ model=model_id,
124
+ messages=test_messages,
125
+ tools=test_tools,
126
+ tool_choice="auto", # Include tool_choice like the real agent
127
+ max_tokens=10,
128
+ **litellm_kwargs,
129
+ )
130
+ # If we got here without error, caching is supported
131
+ caching_supported = True
132
+
133
+ # Only try to extract model from response if we didn't get it from AWS API
134
+ if not detected_model:
135
+ # Try to extract model name from response metadata
136
+ # Bedrock responses include model info in various places
137
+ if hasattr(response, "_hidden_params") and response._hidden_params:
138
+ # LiteLLM stores raw response data here
139
+ hidden = response._hidden_params
140
+
141
+ # Check optional_params which may contain raw boto3 response
142
+ if "optional_params" in hidden and isinstance(hidden["optional_params"], dict):
143
+ optional = hidden["optional_params"]
144
+ # Bedrock converse API may include model info in the response
145
+ if "model" in optional:
146
+ detected_model = optional["model"]
147
+ elif "modelId" in optional:
148
+ detected_model = optional["modelId"]
149
+
150
+ # Check standard fields
151
+ if not detected_model and "model_id" in hidden and hidden["model_id"]:
152
+ detected_model = hidden["model_id"]
153
+ elif not detected_model and "model" in hidden and hidden["model"]:
154
+ detected_model = hidden["model"]
155
+
156
+ # Check response metadata
157
+ if not detected_model and hasattr(response, "model"):
158
+ model_val = response.model
159
+ # Skip if it's just the ARN we passed in
160
+ if model_val and "application-inference-profile" not in model_val:
161
+ detected_model = model_val
162
+
163
+ # Try to extract from response choices/usage if available
164
+ if not detected_model and hasattr(response, "usage"):
165
+ usage = response.usage
166
+ if hasattr(usage, "model") and usage.model:
167
+ detected_model = usage.model
168
+
169
+ except Exception as e:
170
+ error_msg = str(e).lower()
171
+ # Check for caching-specific errors
172
+ if any(
173
+ phrase in error_msg
174
+ for phrase in [
175
+ "prompt caching",
176
+ "cache_control",
177
+ "cachepoint",
178
+ "unsupported model",
179
+ "did not allow prompt caching",
180
+ ]
181
+ ):
182
+ # Caching not supported, but still try to detect model without caching
183
+ caching_supported = False
184
+
185
+ # Only retry if we don't already have model from AWS API
186
+ if not detected_model:
187
+ try:
188
+ # Retry without cache markers to detect model
189
+ simple_messages = [{"role": "user", "content": "Hi"}]
190
+ response = litellm.completion(
191
+ model=model_id,
192
+ messages=simple_messages,
193
+ tools=test_tools,
194
+ max_tokens=5,
195
+ **litellm_kwargs,
196
+ )
197
+ # Try to extract model from response
198
+ if hasattr(response, "_hidden_params") and response._hidden_params:
199
+ hidden = response._hidden_params
200
+ if "model_id" in hidden:
201
+ detected_model = hidden["model_id"]
202
+ elif "model" in hidden:
203
+ detected_model = hidden["model"]
204
+ if not detected_model and hasattr(response, "model"):
205
+ detected_model = response.model
206
+ except Exception:
207
+ pass # Could not detect model
208
+ else:
209
+ # Different error (auth, network, etc.)
210
+ caching_supported = False
211
+
212
+ return caching_supported, detected_model
213
+
214
+
215
+ def test_prompt_caching_support(model_id: str, litellm_kwargs: dict = None) -> bool:
216
+ """Test if a model supports prompt caching (backward compatibility wrapper).
217
+
218
+ Args:
219
+ model_id: Full LiteLLM model identifier (e.g., bedrock/converse/arn:...)
220
+ litellm_kwargs: Optional kwargs to pass to litellm.completion
221
+
222
+ Returns:
223
+ True if prompt caching is supported, False otherwise
224
+ """
225
+ caching_supported, _ = detect_model_capabilities(model_id, litellm_kwargs)
226
+ return caching_supported
@@ -28,14 +28,36 @@ LLM_TIMEOUT = config.LLM_TIMEOUT
28
28
 
29
29
 
30
30
  def _is_bedrock_arn(model_id: str) -> bool:
31
- """Check if a model ID is a Bedrock ARN."""
31
+ """Check if a model ID is a Bedrock ARN.
32
+
33
+ Supports all Bedrock inference profile ARN formats:
34
+ - arn:aws:bedrock:region:account:inference-profile/profile-id
35
+ - arn:aws-us-gov:bedrock:region:account:inference-profile/profile-id
36
+ - arn:aws:bedrock:region:account:application-inference-profile/app-id
37
+ - arn:aws-us-gov:bedrock:region:account:application-inference-profile/app-id
38
+ """
32
39
  return (
33
40
  model_id.startswith("arn:aws")
34
41
  and ":bedrock:" in model_id
35
- and ":inference-profile/" in model_id
42
+ and "inference-profile/" in model_id
36
43
  )
37
44
 
38
45
 
46
+ def _is_application_inference_profile(model_id: str) -> bool:
47
+ """Check if a model ID is a Bedrock application inference profile (tagged profile).
48
+
49
+ Application inference profiles are tagged profiles that don't include the underlying
50
+ model name in the ARN, making it impossible to statically determine model capabilities.
51
+
52
+ Args:
53
+ model_id: Model identifier (may or may not have bedrock/ prefix)
54
+
55
+ Returns:
56
+ True if this is an application inference profile ARN
57
+ """
58
+ return ":application-inference-profile/" in model_id
59
+
60
+
39
61
  def _normalize_bedrock_model_id(model_id: str) -> str:
40
62
  """Normalize Bedrock model ID to ensure it has the bedrock/ prefix.
41
63
 
@@ -51,7 +73,11 @@ def _normalize_bedrock_model_id(model_id: str) -> str:
51
73
 
52
74
  # If it looks like a Bedrock ARN, add the prefix
53
75
  if _is_bedrock_arn(model_id):
54
- return f"bedrock/{model_id}"
76
+ # Application inference profiles require the converse API
77
+ if ":application-inference-profile/" in model_id:
78
+ return f"bedrock/converse/{model_id}"
79
+ else:
80
+ return f"bedrock/{model_id}"
55
81
 
56
82
  # If it's a standard Bedrock model ID (e.g., anthropic.claude-v2)
57
83
  # Check if it looks like a Bedrock model format
@@ -274,6 +300,10 @@ def _supports_prompt_caching(model_id: str) -> bool:
274
300
  # Bedrock Nova models support caching
275
301
  if model_id.startswith("bedrock/") and "amazon.nova" in model_id.lower():
276
302
  return True
303
+ # Bedrock ARNs (all types): enable caching and let Bedrock handle it
304
+ # If the underlying model doesn't support caching, Bedrock will ignore the markers
305
+ if model_id.startswith("bedrock/") and "inference-profile/" in model_id:
306
+ return True
277
307
  return False
278
308
 
279
309
 
@@ -301,12 +331,14 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
301
331
 
302
332
  # Determine cache marker format based on provider
303
333
  # Anthropic models (direct or via Bedrock) use cache_control
304
- # Other Bedrock models (Nova, etc.) use cachePoint
305
- if model_id.startswith("bedrock/") and "anthropic" not in model_id.lower():
306
- # Non-Anthropic Bedrock models (Nova, etc.) use cachePoint
334
+ # Nova models use cachePoint
335
+ # For Bedrock ARNs without model name, default to cache_control (most common)
336
+ if model_id.startswith("bedrock/") and "amazon.nova" in model_id.lower():
337
+ # Nova models explicitly use cachePoint
307
338
  cache_marker = {"cachePoint": {"type": "default"}}
308
339
  else:
309
340
  # Anthropic models (direct or via Bedrock) use cache_control
341
+ # Also default for Bedrock ARNs (most use Anthropic/Claude)
310
342
  cache_marker = {"cache_control": {"type": "ephemeral"}}
311
343
 
312
344
  # Count existing cache markers across all messages
@@ -328,8 +360,30 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
328
360
  max_cache_markers = 4
329
361
  available_slots = max_cache_markers - existing_cache_count
330
362
 
363
+ # Debug logging
364
+ import os
365
+
366
+ if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
367
+ print("\n[CACHE DEBUG] _apply_prompt_caching called")
368
+ print(f" Total messages: {len(messages)}")
369
+ print(f" Existing cache markers: {existing_cache_count}")
370
+ print(f" Available slots: {available_slots}")
371
+ # Show which messages have markers
372
+ marked_indices = []
373
+ for i, msg in enumerate(messages):
374
+ if isinstance(msg.get("content"), list):
375
+ for block in msg["content"]:
376
+ if isinstance(block, dict) and (
377
+ "cache_control" in block or "cachePoint" in block
378
+ ):
379
+ marked_indices.append(i)
380
+ break
381
+ print(f" Messages with existing markers: {marked_indices}")
382
+
331
383
  if available_slots <= 0:
332
384
  # Already at or over limit, don't add any more cache markers
385
+ if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
386
+ print(" Result: No slots available, returning without changes")
333
387
  return messages
334
388
 
335
389
  # Find system messages (usually at the start)
@@ -341,6 +395,9 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
341
395
  non_system_messages[-2:] if len(non_system_messages) >= 2 else non_system_messages
342
396
  )
343
397
 
398
+ if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
399
+ print(f" Last 2 non-system indices: {last_two_indices}")
400
+
344
401
  # Build list of candidate indices to cache (prioritize system messages)
345
402
  candidates = []
346
403
 
@@ -373,6 +430,12 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
373
430
  if not has_cache:
374
431
  candidates.append(idx)
375
432
 
433
+ import os
434
+
435
+ if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
436
+ print(f" Candidates for marking: {candidates}")
437
+ print(f" Will mark first {available_slots} candidates")
438
+
376
439
  # Apply cache markers to candidates, respecting the available slots
377
440
  # This ensures we never exceed the 4 marker limit
378
441
  for idx in candidates[:available_slots]:
@@ -393,6 +456,20 @@ def _apply_prompt_caching(messages: List[Dict[str, Any]], model_id: str) -> List
393
456
  "content": [{"type": "text", "text": content_text, **cache_marker}],
394
457
  }
395
458
 
459
+ if os.environ.get("PATCHPAL_DEBUG_CACHE") == "true":
460
+ # Show final state
461
+ final_marked = []
462
+ for i, msg in enumerate(messages):
463
+ if isinstance(msg.get("content"), list):
464
+ for block in msg["content"]:
465
+ if isinstance(block, dict) and (
466
+ "cache_control" in block or "cachePoint" in block
467
+ ):
468
+ final_marked.append(i)
469
+ break
470
+ print(f" Final messages with markers: {final_marked}")
471
+ print(f" Markers applied to indices: {[c for c in candidates[:available_slots]]}")
472
+
396
473
  return messages
397
474
 
398
475
 
@@ -509,6 +586,49 @@ class PatchPalAgent:
509
586
  if litellm_kwargs:
510
587
  self.litellm_kwargs.update(litellm_kwargs)
511
588
 
589
+ # Detect capabilities for application inference profiles
590
+ # These ARNs don't include model names, so we test with a minimal request
591
+ self.prompt_caching_supported = None # None = untested, True/False after test
592
+ self.detected_model_name = None # Store detected model for cost tracking
593
+ if _is_application_inference_profile(self.model_id):
594
+ # Test capabilities with a minimal request
595
+ try:
596
+ from patchpal.agent.bedrock_profile_utils import detect_model_capabilities
597
+
598
+ print("\033[2mℹ️ Detecting model capabilities...\033[0m", flush=True)
599
+ self.prompt_caching_supported, self.detected_model_name = detect_model_capabilities(
600
+ self.model_id, self.litellm_kwargs
601
+ )
602
+ if self.prompt_caching_supported:
603
+ print("\033[2m✓ Prompt caching is supported\033[0m", flush=True)
604
+ else:
605
+ print("\033[2m✗ Prompt caching is not supported\033[0m", flush=True)
606
+
607
+ if self.detected_model_name:
608
+ print(f"\033[2m✓ Detected model: {self.detected_model_name}\033[0m", flush=True)
609
+
610
+ # Update context limit based on detected model
611
+ try:
612
+ model_info = litellm.get_model_info(f"bedrock/{self.detected_model_name}")
613
+ max_input = model_info.get("max_input_tokens")
614
+ if max_input and isinstance(max_input, (int, float)) and max_input > 0:
615
+ self.context_manager.context_limit = int(max_input)
616
+ except Exception:
617
+ pass # Keep default limit
618
+ else:
619
+ print(
620
+ "\033[2m⚠ Could not detect underlying model name (cost tracking may be inaccurate)\033[0m",
621
+ flush=True,
622
+ )
623
+ except Exception:
624
+ # If test fails, assume caching not supported and model unknown
625
+ self.prompt_caching_supported = False
626
+ self.detected_model_name = None
627
+ elif _supports_prompt_caching(self.model_id):
628
+ self.prompt_caching_supported = True
629
+ else:
630
+ self.prompt_caching_supported = False
631
+
512
632
  # Load MEMORY.md if it exists and has non-template content
513
633
  self._load_project_memory()
514
634
 
@@ -523,38 +643,33 @@ class PatchPalAgent:
523
643
  def _load_project_memory(self):
524
644
  """Load MEMORY.md file at session start if it has non-template content."""
525
645
  try:
526
- from patchpal.tools.common import MEMORY_FILE
646
+ from patchpal.tools.common import get_memory_info
527
647
 
528
- # Always tell the agent where MEMORY.md is located
529
- if not MEMORY_FILE.exists():
530
- return
531
-
532
- memory_content = MEMORY_FILE.read_text(encoding="utf-8")
648
+ info = get_memory_info()
533
649
 
534
- # Check if user has added content after the "---" separator
535
- has_user_content = False
536
- if "---" in memory_content:
537
- parts = memory_content.split("---", 1)
538
- if len(parts) > 1:
539
- user_content = parts[1].strip()
540
- if user_content and len(user_content) > 10:
541
- has_user_content = True
650
+ if not info["exists"]:
651
+ return
542
652
 
543
653
  # Build the message - include full content if user added info, otherwise just location
544
- if has_user_content:
654
+ if info["has_content"]:
545
655
  memory_msg = f"""# Project Memory (from MEMORY.md)
546
656
 
547
- {memory_content}
657
+ {info["content"]}
548
658
 
549
- The information above is from {MEMORY_FILE} and persists across sessions.
550
- To update it, use edit_file("{MEMORY_FILE}", ...) or write_file("{MEMORY_FILE}", ...)."""
659
+ The information above is from MEMORY.md ({info["location_note"]}) and persists across sessions.
660
+ To update it, use edit_file("{info["path"]}", ...) or write_file("{info["path"]}", ...)."""
551
661
  else:
552
662
  # Empty template - just inform agent
553
663
  memory_msg = f"""# Project Memory (MEMORY.md)
554
664
 
555
- Your project memory file is located at: {MEMORY_FILE}
665
+ Your project memory file is located at: {info["path"]}
556
666
 
557
- It's currently empty (just the template). The file is automatically loaded at session start."""
667
+ It's currently empty (just the template). The file is automatically loaded at session start.
668
+
669
+ Note: You can use MEMORY.md in either location:
670
+ - Repository root: MEMORY.md (can be version controlled)
671
+ - Home directory: ~/.patchpal/repos/<repo-name>/MEMORY.md (default)
672
+ Repository root takes priority if both exist."""
558
673
 
559
674
  # Add as a system message at the start
560
675
  self.messages.insert(
@@ -847,7 +962,17 @@ It's currently empty (just the template). The file is automatically loaded at se
847
962
  float: The calculated cost in dollars
848
963
  """
849
964
  try:
850
- model_info = litellm.get_model_info(self.model_id)
965
+ # For application inference profiles, use detected model name for pricing
966
+ model_for_pricing = self.model_id
967
+ if _is_application_inference_profile(self.model_id) and self.detected_model_name:
968
+ # Map detected model name to a pricing model
969
+ # e.g., "anthropic.claude-3-5-sonnet-20241022-v2:0" -> "bedrock/anthropic.claude-3-5-sonnet-20241022-v2:0"
970
+ if not self.detected_model_name.startswith("bedrock/"):
971
+ model_for_pricing = f"bedrock/{self.detected_model_name}"
972
+ else:
973
+ model_for_pricing = self.detected_model_name
974
+
975
+ model_info = litellm.get_model_info(model_for_pricing)
851
976
  input_cost_per_token = model_info.get("input_cost_per_token", 0)
852
977
  output_cost_per_token = model_info.get("output_cost_per_token", 0)
853
978
 
@@ -881,19 +1006,26 @@ It's currently empty (just the template). The file is automatically loaded at se
881
1006
  cost += cache_read_tokens * input_cost_per_token * 0.1
882
1007
 
883
1008
  # Handle OpenAI cache pricing (prompt_tokens_details.cached_tokens)
1009
+ # IMPORTANT: For Bedrock, LiteLLM populates prompt_tokens_details.cached_tokens
1010
+ # with cache_read_input_tokens for compatibility, but we already handled those above.
1011
+ # Only process this field if we're NOT using Bedrock-style cache fields.
884
1012
  openai_cached_tokens = 0
885
- if hasattr(usage, "prompt_tokens_details") and usage.prompt_tokens_details is not None:
886
- prompt_details = usage.prompt_tokens_details
887
- if hasattr(prompt_details, "cached_tokens") and prompt_details.cached_tokens:
888
- # Ensure cached_tokens is a number, not a mock or None
889
- if isinstance(prompt_details.cached_tokens, (int, float)):
890
- openai_cached_tokens = prompt_details.cached_tokens
891
- # Use cached_input_cost_per_token if available, otherwise fallback to 0.5x multiplier
892
- if cached_input_cost_per_token > 0:
893
- cost += openai_cached_tokens * cached_input_cost_per_token
894
- else:
895
- # Fallback: OpenAI cached tokens typically cost 50% of regular input
896
- cost += openai_cached_tokens * input_cost_per_token * 0.5
1013
+ if not (cache_creation_tokens or cache_read_tokens): # Only for non-Bedrock models
1014
+ if (
1015
+ hasattr(usage, "prompt_tokens_details")
1016
+ and usage.prompt_tokens_details is not None
1017
+ ):
1018
+ prompt_details = usage.prompt_tokens_details
1019
+ if hasattr(prompt_details, "cached_tokens") and prompt_details.cached_tokens:
1020
+ # Ensure cached_tokens is a number, not a mock or None
1021
+ if isinstance(prompt_details.cached_tokens, (int, float)):
1022
+ openai_cached_tokens = prompt_details.cached_tokens
1023
+ # Use cached_input_cost_per_token if available, otherwise fallback to 0.5x multiplier
1024
+ if cached_input_cost_per_token > 0:
1025
+ cost += openai_cached_tokens * cached_input_cost_per_token
1026
+ else:
1027
+ # Fallback: OpenAI cached tokens typically cost 50% of regular input
1028
+ cost += openai_cached_tokens * input_cost_per_token * 0.5
897
1029
 
898
1030
  # Regular input tokens (excluding all cache tokens)
899
1031
  regular_input = (
@@ -1048,8 +1180,10 @@ It's currently empty (just the template). The file is automatically loaded at se
1048
1180
  # Filter images if BLOCK_IMAGES is enabled (for non-vision models or user preference)
1049
1181
  messages = self.image_handler.filter_images_if_blocked(messages)
1050
1182
 
1051
- # Apply prompt caching for supported models (Anthropic/Claude)
1052
- messages = _apply_prompt_caching(messages, self.model_id)
1183
+ # Apply prompt caching for supported models
1184
+ # Check instance variable for dynamically-tested models (application inference profiles)
1185
+ if self.prompt_caching_supported:
1186
+ messages = _apply_prompt_caching(messages, self.model_id)
1053
1187
 
1054
1188
  # Use LiteLLM for all providers
1055
1189
  try:
@@ -1260,6 +1394,7 @@ It's currently empty (just the template). The file is automatically loaded at se
1260
1394
  )
1261
1395
  elif tool_name == "get_repo_map":
1262
1396
  max_files = tool_args.get("max_files", 100)
1397
+ max_depth = tool_args.get("max_depth")
1263
1398
  patterns = ""
1264
1399
  if tool_args.get("include_patterns"):
1265
1400
  patterns = (
@@ -1269,8 +1404,9 @@ It's currently empty (just the template). The file is automatically loaded at se
1269
1404
  patterns = (
1270
1405
  f" (exclude: {', '.join(tool_args['exclude_patterns'])})"
1271
1406
  )
1407
+ depth_info = f", depth≤{max_depth}" if max_depth is not None else ""
1272
1408
  print(
1273
- f"\033[2m🗺️ Generating repository map (max {max_files} files{patterns})...\033[0m",
1409
+ f"\033[2m🗺️ Generating repository map (max {max_files} files{depth_info}{patterns})...\033[0m",
1274
1410
  flush=True,
1275
1411
  )
1276
1412
  elif tool_name == "get_file_info":
@@ -1295,8 +1431,12 @@ It's currently empty (just the template). The file is automatically loaded at se
1295
1431
  )
1296
1432
  elif tool_name == "find":
1297
1433
  pattern_desc = tool_args.get("pattern", "*")
1434
+ max_depth = tool_args.get("max_depth")
1435
+ depth_info = (
1436
+ f" (depth≤{max_depth})" if max_depth is not None else ""
1437
+ )
1298
1438
  print(
1299
- f"\033[2m📂 Finding files: {pattern_desc}\033[0m",
1439
+ f"\033[2m📂 Finding files: {pattern_desc}{depth_info}\033[0m",
1300
1440
  flush=True,
1301
1441
  )
1302
1442
  elif tool_name == "list_skills":