agentnova 0.3.7__tar.gz → 0.3.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (89) hide show
  1. {agentnova-0.3.7 → agentnova-0.3.8}/PKG-INFO +4 -3
  2. {agentnova-0.3.7 → agentnova-0.3.8}/README.md +4 -3
  3. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/__init__.py +1 -1
  4. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/acp_plugin.py +7 -3
  5. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/agent.py +234 -19
  6. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/backends/ollama.py +46 -36
  7. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/cli.py +107 -84
  8. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/colors.py +203 -203
  9. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/helpers.py +38 -20
  10. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/tool_cache.py +223 -188
  11. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/loader.py +5 -2
  12. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/SKILL.md +1 -0
  13. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/soul/loader.py +12 -4
  14. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/souls/nova-helper/SOUL.md +29 -1
  15. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/tools/sandboxed_repl.py +4 -5
  16. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova.egg-info/PKG-INFO +4 -3
  17. {agentnova-0.3.7 → agentnova-0.3.8}/pyproject.toml +1 -1
  18. {agentnova-0.3.7 → agentnova-0.3.8}/LICENSE +0 -0
  19. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/__main__.py +0 -0
  20. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/agent_mode.py +0 -0
  21. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/backends/__init__.py +0 -0
  22. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/backends/base.py +0 -0
  23. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/backends/bitnet.py +0 -0
  24. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/config.py +0 -0
  25. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/__init__.py +0 -0
  26. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/args_normal.py +0 -0
  27. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/error_recovery.py +0 -0
  28. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/math_prompts.py +0 -0
  29. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/memory.py +0 -0
  30. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/model_config.py +0 -0
  31. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/model_family_config.py +0 -0
  32. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/models.py +0 -0
  33. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/openresponses.py +0 -0
  34. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/prompts.py +0 -0
  35. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/tool_parse.py +0 -0
  36. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/core/types.py +0 -0
  37. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/00_basic_agent.py +0 -0
  38. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/01_quick_diagnostic.py +0 -0
  39. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/02_tool_test.py +0 -0
  40. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/03_reasoning_test.py +0 -0
  41. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/04_gsm8k_benchmark.py +0 -0
  42. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/05_common_sense.py +0 -0
  43. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/06_causal_reasoning.py +0 -0
  44. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/07_logical_deduction.py +0 -0
  45. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/08_reading_comprehension.py +0 -0
  46. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/09_general_knowledge.py +0 -0
  47. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/10_implicit_reasoning.py +0 -0
  48. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/11_analogical_reasoning.py +0 -0
  49. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/examples/__init__.py +0 -0
  50. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/model_discovery.py +0 -0
  51. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/orchestrator.py +0 -0
  52. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/py.typed +0 -0
  53. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/shared_args.py +0 -0
  54. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/__init__.py +0 -0
  55. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/acp/SKILL.md +0 -0
  56. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/datetime/SKILL.md +0 -0
  57. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/__init__.py +0 -0
  58. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/aggregate_benchmark.py +0 -0
  59. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/generate_report.py +0 -0
  60. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/improve_description.py +0 -0
  61. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/init_skill.py +0 -0
  62. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/package_skill.py +0 -0
  63. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/quick_validate.py +0 -0
  64. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/run_eval.py +0 -0
  65. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/run_loop.py +0 -0
  66. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/security_scan.py +0 -0
  67. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/test_package_skill.py +0 -0
  68. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/test_quick_validate.py +0 -0
  69. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/utils.py +0 -0
  70. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/skill-creator/scripts/validate.py +0 -0
  71. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/skills/web-search/SKILL.md +0 -0
  72. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/soul/__init__.py +0 -0
  73. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/soul/types.py +0 -0
  74. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/souls/nova-helper/IDENTITY.md +0 -0
  75. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/souls/nova-helper/STYLE.md +0 -0
  76. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/souls/nova-helper/soul.json +0 -0
  77. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/tools/__init__.py +0 -0
  78. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/tools/builtins.py +0 -0
  79. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova/tools/registry.py +0 -0
  80. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova.egg-info/SOURCES.txt +0 -0
  81. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova.egg-info/dependency_links.txt +0 -0
  82. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova.egg-info/entry_points.txt +0 -0
  83. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova.egg-info/requires.txt +0 -0
  84. {agentnova-0.3.7 → agentnova-0.3.8}/agentnova.egg-info/top_level.txt +0 -0
  85. {agentnova-0.3.7 → agentnova-0.3.8}/localclaw/__init__.py +0 -0
  86. {agentnova-0.3.7 → agentnova-0.3.8}/localclaw-redirect/localclaw/__init__.py +0 -0
  87. {agentnova-0.3.7 → agentnova-0.3.8}/setup.cfg +0 -0
  88. {agentnova-0.3.7 → agentnova-0.3.8}/tests/test_agent.py +0 -0
  89. {agentnova-0.3.7 → agentnova-0.3.8}/tests/test_spec_compliance.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: agentnova
3
- Version: 0.3.7
3
+ Version: 0.3.8
4
4
  Summary: ⚛️ AgentNova - A minimal, hackable agentic framework for local LLM inference
5
5
  Author-email: VTSTech <veritas@vts-tech.org>
6
6
  Maintainer-email: VTSTech <veritas@vts-tech.org>
@@ -32,7 +32,7 @@ Requires-Dist: black>=23.0; extra == "dev"
32
32
  Requires-Dist: ruff>=0.1.0; extra == "dev"
33
33
  Dynamic: license-file
34
34
 
35
- # ⚛️ AgentNova R03.7
35
+ # ⚛️ AgentNova R03.8
36
36
 
37
37
  **Status: Alpha**
38
38
 
@@ -49,7 +49,8 @@ Inspired by the architecture of OpenClaw, rebuilt from scratch for local-first o
49
49
 
50
50
  [![License](https://img.shields.io/badge/License-MIT-blue)](#license) [![Go to Python website](https://img.shields.io/badge/dynamic/toml?url=https%3A%2F%2Fraw.githubusercontent.com%2FVTSTech%2FAgentNova%2Frefs%2Fheads%2Fmain%2Fpyproject.toml&query=project.requires-python&label=python&logo=python&logoColor=white)](https://python.org)
51
51
 
52
- <img width="1378" height="996" alt="image" src="https://github.com/user-attachments/assets/0fe69695-73d9-4ce1-999f-08443f879971" />
52
+ <img width="1314" height="993" alt="image" src="https://github.com/user-attachments/assets/586b1a44-441e-41ae-85c2-b65b6a0816b9" />
53
+ <img width="1063" height="574" alt="image" src="https://github.com/user-attachments/assets/eab6f4ad-810c-4741-b637-e120f9ccb974" />
53
54
 
54
55
  ## 📚 Documentation
55
56
 
@@ -1,4 +1,4 @@
1
- # ⚛️ AgentNova R03.7
1
+ # ⚛️ AgentNova R03.8
2
2
 
3
3
  **Status: Alpha**
4
4
 
@@ -15,7 +15,8 @@ Inspired by the architecture of OpenClaw, rebuilt from scratch for local-first o
15
15
 
16
16
  [![License](https://img.shields.io/badge/License-MIT-blue)](#license) [![Go to Python website](https://img.shields.io/badge/dynamic/toml?url=https%3A%2F%2Fraw.githubusercontent.com%2FVTSTech%2FAgentNova%2Frefs%2Fheads%2Fmain%2Fpyproject.toml&query=project.requires-python&label=python&logo=python&logoColor=white)](https://python.org)
17
17
 
18
- <img width="1378" height="996" alt="image" src="https://github.com/user-attachments/assets/0fe69695-73d9-4ce1-999f-08443f879971" />
18
+ <img width="1314" height="993" alt="image" src="https://github.com/user-attachments/assets/586b1a44-441e-41ae-85c2-b65b6a0816b9" />
19
+ <img width="1063" height="574" alt="image" src="https://github.com/user-attachments/assets/eab6f4ad-810c-4741-b637-e120f9ccb974" />
19
20
 
20
21
  ## 📚 Documentation
21
22
 
@@ -349,4 +350,4 @@ Contributions welcome! Please read the contributing guidelines first.
349
350
 
350
351
  - Built for local inference with [Ollama](https://ollama.ai)
351
352
  - Optimized for small, efficient models
352
- - Inspired by ReAct and other agentic frameworks
353
+ - Inspired by ReAct and other agentic frameworks
@@ -28,7 +28,7 @@ Example Usage:
28
28
  agent = Agent(model="qwen2.5:0.5b", soul="/path/to/soul/package")
29
29
  """
30
30
 
31
- __version__ = "0.3.7"
31
+ __version__ = "0.3.8"
32
32
  __author__ = "VTSTech"
33
33
  __status__ = "Alpha"
34
34
 
@@ -281,6 +281,10 @@ class ACPPlugin:
281
281
  "get_note": "READ",
282
282
  "list_notes": "READ",
283
283
  "list_directory": "READ",
284
+ # ACP v1.0.4: A2A agent-to-agent communication
285
+ "a2a_send": "A2A",
286
+ "a2a_request": "A2A",
287
+ "a2a_response": "A2A",
284
288
  }
285
289
 
286
290
  # v1.0.4: Auto-derive capabilities from tool action map if not explicitly provided
@@ -1089,11 +1093,11 @@ class ACPPlugin:
1089
1093
  """Format tool arguments as a target string for ACP."""
1090
1094
  # Tool-specific formatting
1091
1095
  if tool_name == "read_file":
1092
- return args.get("path", "unknown")
1096
+ return args.get("file_path", args.get("path", "unknown"))
1093
1097
  elif tool_name == "write_file":
1094
- return args.get("path", "unknown")
1098
+ return args.get("file_path", args.get("path", "unknown"))
1095
1099
  elif tool_name == "edit_file":
1096
- return args.get("path", "unknown")
1100
+ return args.get("file_path", args.get("path", "unknown"))
1097
1101
  elif tool_name == "shell":
1098
1102
  return args.get("command", "")[:100]
1099
1103
  elif tool_name == "calculator":
@@ -40,6 +40,7 @@ from .core.openresponses import (
40
40
  MessageItem, FunctionCallItem, FunctionCallOutputItem, ReasoningItem,
41
41
  OutputText, InputText,
42
42
  RequestConfig, Error,
43
+ EventType, ResponseEvent, OutputItemEvent,
43
44
  create_message_item, create_function_call_item, create_function_call_output,
44
45
  create_function_call_output_item,
45
46
  stream_response_events,
@@ -132,7 +133,7 @@ class Agent:
132
133
  system_prompt: Custom system prompt (overrides soul)
133
134
  soul: Path to Soul Spec package (default: "nova-helper")
134
135
  soul_level: Progressive disclosure level for soul (1-3)
135
- num_ctx: Context window size in tokens (default: 4096)
136
+ num_ctx: Context window size in tokens (default: 8192)
136
137
  temperature: Sampling temperature (default: model-specific)
137
138
  top_p: Nucleus sampling probability (default: model-specific)
138
139
  num_predict: Maximum tokens to generate (default: model-specific)
@@ -143,12 +144,12 @@ class Agent:
143
144
  self.model = model
144
145
  self.max_steps = max_steps
145
146
  self.debug = debug
146
- # Get num_ctx from: explicit param > config/env > default 4096
147
+ # Get num_ctx from: explicit param > config/env > default 8192
147
148
  if num_ctx is not None:
148
149
  self.num_ctx = num_ctx
149
150
  else:
150
151
  config = get_config()
151
- self.num_ctx = config.num_ctx if config.num_ctx else 4096
152
+ self.num_ctx = config.num_ctx if config.num_ctx else 8192
152
153
 
153
154
  # Generation parameters (use model defaults if not specified)
154
155
  self._temperature = temperature
@@ -441,6 +442,31 @@ Final Answer: <the answer>
441
442
  tokens = gen_response.get("usage", {}).get("total_tokens", 0)
442
443
  total_tokens += tokens
443
444
 
445
+ # OpenResponses: Handle finish_reason from backend
446
+ finish_reason = gen_response.get("_finish_reason", "stop")
447
+ if finish_reason == "length":
448
+ # Token budget exhausted — response is incomplete
449
+ if self.debug:
450
+ print(f" [OpenResponses] finish_reason='length' — marking incomplete")
451
+ steps.append(StepResult(
452
+ type=StepResultType.MAX_STEPS,
453
+ content="Response truncated: token limit reached",
454
+ tokens_used=tokens,
455
+ ))
456
+ response.mark_incomplete()
457
+ break
458
+ elif finish_reason == "content_filter":
459
+ # Content was filtered — response failed
460
+ if self.debug:
461
+ print(f" [OpenResponses] finish_reason='content_filter' — marking failed")
462
+ steps.append(StepResult(
463
+ type=StepResultType.ERROR,
464
+ error="Response blocked by content filter",
465
+ tokens_used=tokens,
466
+ ))
467
+ response.mark_failed({"message": "Content filtered by provider", "type": "content_filter"})
468
+ break
469
+
444
470
  if self.debug:
445
471
  print(f" Content: {content[:200] if content else '(empty)'}...")
446
472
  print(f" Native tool calls: {native_tool_calls}")
@@ -873,6 +899,10 @@ Final Answer: <the answer>
873
899
  OpenResponses specification. It yields Server-Sent Events (SSE) that
874
900
  describe the response lifecycle and content deltas.
875
901
 
902
+ IMPORTANT: The agentic loop is fully supported during streaming.
903
+ When the model produces a tool call, it is executed and the loop
904
+ continues, streaming the next model response.
905
+
876
906
  SSE Event Sequence (per OpenResponses spec):
877
907
  1. response.queued - Response is queued
878
908
  2. response.in_progress - Response started
@@ -895,6 +925,8 @@ Final Answer: <the answer>
895
925
  for sse_event in agent.run_stream("Hello!"):
896
926
  print(sse_event) # SSE formatted event
897
927
  """
928
+ start_time = time.time()
929
+
898
930
  # Create OpenResponses Response object
899
931
  response = Response(
900
932
  model=self.model,
@@ -903,9 +935,8 @@ Final Answer: <the answer>
903
935
  allowed_tools=self._allowed_tools or [],
904
936
  )
905
937
 
906
- if self.debug and not self._is_comp_mode:
907
- print(f"\n[OpenResponses] Response created: id={response.id}")
908
- print(f"[OpenResponses] Response status: {response.status.value}")
938
+ if self.debug:
939
+ print(f"\n[OpenResponses stream] Response created: id={response.id}")
909
940
 
910
941
  # Add user prompt to memory
911
942
  self.memory.add("user", prompt)
@@ -914,22 +945,190 @@ Final Answer: <the answer>
914
945
  user_item = create_message_item("user", prompt)
915
946
  response.input.append(user_item)
916
947
 
917
- if self.debug and not self._is_comp_mode:
918
- print(f"[OpenResponses] Input item added: id={user_item.id}, type={user_item.type}, role={user_item.role}")
919
-
920
948
  if self.debug:
921
- print(f"\n[AgentNova] Model: {self.model}")
922
- print(f"[AgentNova] Backend: {self.backend.base_url}")
923
- print(f"[AgentNova] tool_choice: {self.tool_choice.type.value}")
924
- print(f"[AgentNova] Tools: {self.tools.names()}")
925
- print(f"[AgentNova] Prompt (streaming): {prompt}\n")
949
+ print(f"\n[AgentNova stream] Model: {self.model}")
950
+ print(f"[AgentNova stream] Backend: {self.backend.base_url}")
951
+ print(f"[AgentNova stream] tool_choice: {self.tool_choice.type.value}")
952
+ print(f"[AgentNova stream] Tools: {self.tools.names()}")
953
+ print(f"[AgentNova stream] Prompt: {prompt}\n")
954
+
955
+ # OpenResponses: Agentic Loop (streaming variant)
956
+ # Stream model output, check for tool calls, execute them, repeat.
957
+ _expecting_final_answer = False
958
+ _last_successful_result = None
959
+ tool_call_count = 0
960
+
961
+ for step_num in range(self.max_steps):
962
+ if self.debug:
963
+ print(f"[Stream Step {step_num + 1}]")
964
+
965
+ # Collect the full streamed response
966
+ full_content = ""
967
+
968
+ # Stream model response, collecting content for tool-call detection
969
+ try:
970
+ for chunk in self._generate_stream_chunks(prompt):
971
+ full_content += chunk
972
+ except Exception as e:
973
+ if self.debug:
974
+ print(f" [Stream] ERROR: {e}")
975
+ # Emit failure event
976
+ response.mark_failed({"message": str(e), "type": "stream_error"})
977
+ fail_event = ResponseEvent(
978
+ type=EventType.RESPONSE_FAILED,
979
+ response=response,
980
+ )
981
+ yield fail_event.to_sse()
982
+ return
983
+
984
+ # Parse for tool calls (ReAct format)
985
+ tool_calls_found = []
986
+
987
+ if full_content:
988
+ parsed_calls = self._parser.parse(full_content)
989
+ for call in parsed_calls:
990
+ if hasattr(call, 'thought') and call.thought:
991
+ reasoning_item = ReasoningItem(
992
+ content=[OutputText(text=call.thought)]
993
+ )
994
+ reasoning_item.status = ItemStatus.COMPLETED
995
+ response.add_output_item(reasoning_item)
926
996
 
927
- # Get text chunks from backend streaming
928
- text_chunks = self._generate_stream_chunks(prompt)
997
+ tool_calls_found.append({
998
+ "name": call.name,
999
+ "arguments": call.arguments,
1000
+ "id": "",
1001
+ "final_answer": getattr(call, 'final_answer', None),
1002
+ })
929
1003
 
930
- # Wrap with OpenResponses SSE events using stream_response_events()
931
- for sse_event in stream_response_events(response, text_chunks, debug=self.debug):
932
- yield sse_event
1004
+ # ---- Execute tool calls if found ----
1005
+ if tool_calls_found:
1006
+ # Final Answer enforcement (same logic as run())
1007
+ if _expecting_final_answer and _last_successful_result is not None:
1008
+ text_chunks_gen = iter([_last_successful_result])
1009
+ for sse_event in stream_response_events(
1010
+ Response(model=self.model, status=ResponseStatus.IN_PROGRESS,
1011
+ tool_choice=self.tool_choice, allowed_tools=self._allowed_tools or []),
1012
+ text_chunks_gen, debug=self.debug,
1013
+ ):
1014
+ yield sse_event
1015
+ return
1016
+
1017
+ pending_final_answer = None
1018
+ self.memory.add("assistant", full_content)
1019
+
1020
+ for tc in tool_calls_found:
1021
+ tool_name = tc["name"]
1022
+ tool_args = tc["arguments"]
1023
+
1024
+ if tc.get("final_answer"):
1025
+ pending_final_answer = tc["final_answer"]
1026
+
1027
+ # Check allowed_tools
1028
+ if self._allowed_tools and tool_name not in self._allowed_tools:
1029
+ error_msg = f"Tool '{tool_name}' not in allowed_tools: {self._allowed_tools}"
1030
+ self.memory.add("user", f"Observation: Error: {error_msg}")
1031
+ continue
1032
+
1033
+ # Fuzzy match
1034
+ tool_name = self._parser._fuzzy_match_tool(tool_name)
1035
+
1036
+ # Create FunctionCallItem and emit SSE events
1037
+ fc_item = create_function_call_item(tool_name, tool_args)
1038
+ fc_item.status = ItemStatus.IN_PROGRESS
1039
+ response.add_output_item(fc_item)
1040
+ output_index = len(response.output) - 1
1041
+
1042
+ fc_added = OutputItemEvent(
1043
+ type=EventType.OUTPUT_ITEM_ADDED,
1044
+ item=fc_item,
1045
+ output_index=output_index,
1046
+ )
1047
+ yield fc_added.to_sse()
1048
+
1049
+ # Execute the tool
1050
+ result = self._execute_tool(tool_name, tool_args, prompt)
1051
+ tool_call_count += 1
1052
+
1053
+ fc_item.status = ItemStatus.COMPLETED
1054
+
1055
+ fc_done = OutputItemEvent(
1056
+ type=EventType.OUTPUT_ITEM_DONE,
1057
+ item=fc_item,
1058
+ output_index=output_index,
1059
+ )
1060
+ yield fc_done.to_sse()
1061
+
1062
+ # Create function_call_output
1063
+ fco_item = create_function_call_output(fc_item.call_id, str(result))
1064
+ response.add_output_item(fco_item)
1065
+
1066
+ # Build observation and add to memory
1067
+ is_error = is_error_result(str(result))
1068
+ observation_msg = build_enhanced_observation(
1069
+ tool_name=tool_name,
1070
+ result=str(result),
1071
+ tracker=self._error_tracker,
1072
+ available_tools=self.tools.names(),
1073
+ is_error=is_error,
1074
+ )
1075
+
1076
+ if is_error:
1077
+ _expecting_final_answer = False
1078
+ else:
1079
+ _expecting_final_answer = True
1080
+ _last_successful_result = str(result)
1081
+
1082
+ self.memory.add("user", observation_msg)
1083
+
1084
+ # Check for pending final answer
1085
+ if pending_final_answer:
1086
+ text_chunks_gen = iter([pending_final_answer])
1087
+ for sse_event in stream_response_events(
1088
+ Response(model=self.model, status=ResponseStatus.IN_PROGRESS,
1089
+ tool_choice=self.tool_choice, allowed_tools=self._allowed_tools or []),
1090
+ text_chunks_gen, debug=self.debug,
1091
+ ):
1092
+ yield sse_event
1093
+ return
1094
+
1095
+ # Continue the agentic loop (next streaming iteration)
1096
+ continue
1097
+
1098
+ # ---- No tool calls — stream final response ----
1099
+ # Check for Final Answer format
1100
+ if self._parser.is_final_answer(full_content):
1101
+ answer = self._parser.extract_final_answer(full_content)
1102
+ text_chunks_gen = iter([answer])
1103
+ else:
1104
+ text_chunks_gen = iter([full_content])
1105
+
1106
+ # Stream the final response with proper OpenResponses events
1107
+ final_response = Response(
1108
+ model=self.model,
1109
+ status=ResponseStatus.IN_PROGRESS,
1110
+ tool_choice=self.tool_choice,
1111
+ allowed_tools=self._allowed_tools or [],
1112
+ )
1113
+ # Carry over any items from previous loop iterations
1114
+ final_response.output = response.output
1115
+ final_response.input = response.input
1116
+ final_response.usage = response.usage
1117
+
1118
+ for sse_event in stream_response_events(final_response, text_chunks_gen, debug=self.debug):
1119
+ yield sse_event
1120
+
1121
+ # Only one pass needed when there are no tool calls
1122
+ return
1123
+
1124
+ else:
1125
+ # Max steps reached
1126
+ response.mark_incomplete()
1127
+ incomplete_event = ResponseEvent(
1128
+ type=EventType.RESPONSE_INCOMPLETE,
1129
+ response=response,
1130
+ )
1131
+ yield incomplete_event.to_sse()
933
1132
 
934
1133
  def _generate_stream_chunks(self, prompt: str) -> Generator[str, None, None]:
935
1134
  """
@@ -1076,6 +1275,11 @@ Final Answer: <the answer>
1076
1275
  if self._num_predict is not None:
1077
1276
  backend_kwargs["num_predict"] = self._num_predict
1078
1277
 
1278
+ # OpenResponses: Forward tool_choice to backend API
1279
+ # This allows the backend to enforce tool invocation constraints natively
1280
+ if self.tool_choice and self.tool_choice.type != ToolChoiceType.AUTO:
1281
+ backend_kwargs["tool_choice"] = self.tool_choice.to_dict()
1282
+
1079
1283
  # Pass tools for native tool calling (OpenResponses/ChatCompletions compliant)
1080
1284
  # ReAct parsing remains as fallback for models without native support
1081
1285
  tools_for_backend = self.tools.all() if self.tools and len(self.tools) > 0 else None
@@ -1101,6 +1305,17 @@ Final Answer: <the answer>
1101
1305
  **backend_kwargs,
1102
1306
  )
1103
1307
 
1308
+ # OpenResponses / Chat Completions: Handle finish_reason
1309
+ # Per spec, finish_reason affects response status:
1310
+ # "stop" → normal completion (default)
1311
+ # "length" → incomplete — token budget exhausted
1312
+ # "content_filter" → failed — content was filtered
1313
+ finish_reason = response.get("finish_reason", "stop")
1314
+ if self.debug:
1315
+ print(f" [DEBUG] finish_reason: {finish_reason}")
1316
+ # Store for caller to consume
1317
+ response["_finish_reason"] = finish_reason
1318
+
1104
1319
  if self.debug:
1105
1320
  print(f" [DEBUG] Response keys: {list(response.keys())}")
1106
1321
  print(f" [DEBUG] Content: {response.get('content', '')[:100]}...")
@@ -761,14 +761,22 @@ class OllamaBackend(BaseBackend):
761
761
 
762
762
  def get_model_info(self, model: str) -> dict | None:
763
763
  """
764
- Get detailed model information from Ollama.
764
+ Get detailed model information from Ollama (with per-instance caching).
765
765
 
766
766
  Uses /api/show endpoint which returns:
767
767
  - modelfile
768
768
  - parameters (including num_ctx)
769
769
  - template
770
770
  - details (family, parameter count, etc.)
771
+
772
+ Results are cached for the lifetime of the backend instance.
771
773
  """
774
+ # Check instance cache first
775
+ if not hasattr(self, '_model_info_cache'):
776
+ self._model_info_cache = {}
777
+ if model in self._model_info_cache:
778
+ return self._model_info_cache[model]
779
+
772
780
  import urllib.request
773
781
  import urllib.error
774
782
 
@@ -785,7 +793,9 @@ class OllamaBackend(BaseBackend):
785
793
  )
786
794
 
787
795
  with urllib.request.urlopen(req, timeout=10) as response:
788
- return json.loads(response.read().decode("utf-8"))
796
+ data = json.loads(response.read().decode("utf-8"))
797
+ self._model_info_cache[model] = data
798
+ return data
789
799
 
790
800
  except (urllib.error.HTTPError, urllib.error.URLError):
791
801
  return None
@@ -861,54 +871,54 @@ class OllamaBackend(BaseBackend):
861
871
  """
862
872
  Get the model's maximum trained context window size.
863
873
 
864
- This is the context_length from model_info, representing the model's
865
- capability regardless of runtime settings.
874
+ Resolution order (most authoritative first):
875
+ 1. API /api/show → model_info.<family>.context_length (exact model data)
876
+ 2. API /api/show → details.family → FAMILY_CONTEXT_DEFAULTS lookup
877
+ 3. Caller-provided family → FAMILY_CONTEXT_DEFAULTS lookup
878
+ 4. Hardcoded fallback 4096
866
879
 
867
880
  Args:
868
881
  model: Model name
869
- family: Optional family name (uses default if provided)
882
+ family: Optional family name (fallback if API doesn't report it)
870
883
 
871
884
  Returns:
872
885
  Maximum context window size in tokens
873
886
  """
874
- # Fast path: use family default
875
- if family:
876
- ctx = self.get_context_by_family(family)
877
- if ctx:
878
- return ctx
879
-
880
- # Slow path: get from API
887
+ # Primary: ask the API for the model's actual context_length
881
888
  info = self.get_model_info(model)
882
889
 
883
- if not info:
884
- return 4096 # Fallback
885
-
886
- # Check model_info for context_length
887
- # Key format: "<family>.context_length" (e.g., "gemma3.context_length", "qwen2.context_length")
888
- model_info = info.get("model_info", {})
889
- for key, value in model_info.items():
890
- if key.endswith(".context_length"):
890
+ if info:
891
+ model_info = info.get("model_info", {})
892
+ # Key format: "<family>.context_length" (e.g., "gemma3.context_length")
893
+ for key, value in model_info.items():
894
+ if key.endswith(".context_length"):
895
+ try:
896
+ return int(value)
897
+ except (ValueError, TypeError):
898
+ pass
899
+
900
+ # Fallback: bare "context_length" key (some Ollama versions)
901
+ if "context_length" in model_info:
891
902
  try:
892
- return int(value)
903
+ return int(model_info["context_length"])
893
904
  except (ValueError, TypeError):
894
905
  pass
906
+
907
+ # Try family from API details (more reliable than caller-provided)
908
+ details = info.get("details", {})
909
+ api_family = details.get("family", "")
910
+ if api_family:
911
+ ctx = self.get_context_by_family(api_family)
912
+ if ctx:
913
+ return ctx
895
914
 
896
- # Fallback: check for bare "context_length" key (some versions)
897
- if "context_length" in model_info:
898
- try:
899
- return int(model_info["context_length"])
900
- except (ValueError, TypeError):
901
- pass
902
-
903
- # Check details for family
904
- details = info.get("details", {})
905
- api_family = details.get("family", "").lower()
906
-
907
- ctx = self.get_context_by_family(api_family)
908
- if ctx:
909
- return ctx
915
+ # Fallback: use caller-provided family or hardcoded table
916
+ if family:
917
+ ctx = self.get_context_by_family(family)
918
+ if ctx:
919
+ return ctx
910
920
 
911
- return 4096 # Fallback
921
+ return 4096 # Ultimate fallback
912
922
 
913
923
  def get_model_context_size(self, model: str, family: str | None = None) -> int:
914
924
  """