sigit-code 1.5.7__tar.gz → 1.5.8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. {sigit_code-1.5.7 → sigit_code-1.5.8}/CHANGELOG.md +76 -0
  2. {sigit_code-1.5.7 → sigit_code-1.5.8}/Cargo.lock +1 -1
  3. {sigit_code-1.5.7 → sigit_code-1.5.8}/Cargo.toml +1 -1
  4. {sigit_code-1.5.7 → sigit_code-1.5.8}/PKG-INFO +1 -1
  5. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/backend.rs +228 -22
  6. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/chat.rs +127 -0
  7. sigit_code-1.5.8/src/inline_tool_calls.rs +420 -0
  8. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/main.rs +777 -76
  9. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/models.rs +50 -0
  10. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/permissions.rs +132 -7
  11. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/settings.rs +12 -25
  12. {sigit_code-1.5.7 → sigit_code-1.5.8}/tests/acp_endpoint_errors.rs +94 -0
  13. {sigit_code-1.5.7 → sigit_code-1.5.8}/tests/acp_permissions.rs +546 -0
  14. sigit_code-1.5.8/tests/acp_session_load.rs +320 -0
  15. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/AGENTS.md +0 -0
  16. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/agent-client-protocol/SKILL.md +0 -0
  17. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/ai-assisted-coding/SKILL.md +0 -0
  18. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/branding/SKILL.md +0 -0
  19. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/run-sigit/SKILL.md +0 -0
  20. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/run-sigit/driver.mjs +0 -0
  21. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/run-sigit/tui-smoke.sh +0 -0
  22. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/sigit-code-release/SKILL.md +0 -0
  23. {sigit_code-1.5.7 → sigit_code-1.5.8}/.agents/skills/tool-calling/SKILL.md +0 -0
  24. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/agent-client-protocol/SKILL.md +0 -0
  25. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/ai-assisted-coding/SKILL.md +0 -0
  26. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/branding/SKILL.md +0 -0
  27. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/run-sigit/SKILL.md +0 -0
  28. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/run-sigit/driver.mjs +0 -0
  29. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/run-sigit/tui-smoke.sh +0 -0
  30. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/sigit-code-release/SKILL.md +0 -0
  31. {sigit_code-1.5.7 → sigit_code-1.5.8}/.claude/skills/tool-calling/SKILL.md +0 -0
  32. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/ci.yml +0 -0
  33. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-aur.yml +0 -0
  34. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-crates.yml +0 -0
  35. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-github.yml +0 -0
  36. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-homebrew.yml +0 -0
  37. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-npm.yml +0 -0
  38. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-nuget.yml +0 -0
  39. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-pypi.yml +0 -0
  40. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-scoop.yml +0 -0
  41. {sigit_code-1.5.7 → sigit_code-1.5.8}/.github/workflows/release-winget.yml +0 -0
  42. {sigit_code-1.5.7 → sigit_code-1.5.8}/.gitignore +0 -0
  43. {sigit_code-1.5.7 → sigit_code-1.5.8}/.nvmrc +0 -0
  44. {sigit_code-1.5.7 → sigit_code-1.5.8}/AGENTS.md +0 -0
  45. {sigit_code-1.5.7 → sigit_code-1.5.8}/CLAUDE.md +0 -0
  46. {sigit_code-1.5.7 → sigit_code-1.5.8}/LICENSE +0 -0
  47. {sigit_code-1.5.7 → sigit_code-1.5.8}/README.md +0 -0
  48. {sigit_code-1.5.7 → sigit_code-1.5.8}/docs/hooks.md +0 -0
  49. {sigit_code-1.5.7 → sigit_code-1.5.8}/docs/mcp.md +0 -0
  50. {sigit_code-1.5.7 → sigit_code-1.5.8}/examples/settings-with-hooks.toml +0 -0
  51. {sigit_code-1.5.7 → sigit_code-1.5.8}/examples/skills/README.md +0 -0
  52. {sigit_code-1.5.7 → sigit_code-1.5.8}/examples/skills/commit-message/SKILL.md +0 -0
  53. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/README.md.tmpl +0 -0
  54. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/package-compat.json.tmpl +0 -0
  55. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/package-main.json.tmpl +0 -0
  56. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/package.json.tmpl +0 -0
  57. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/scripts/render-main-package.cjs +0 -0
  58. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/scripts/render-platform-package.cjs +0 -0
  59. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/sigit/.gitignore +0 -0
  60. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/sigit/README.md +0 -0
  61. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/sigit/package.json +0 -0
  62. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/sigit/src/index.ts +0 -0
  63. {sigit_code-1.5.7 → sigit_code-1.5.8}/npm/sigit/tsconfig.json +0 -0
  64. {sigit_code-1.5.7 → sigit_code-1.5.8}/nuget/.gitignore +0 -0
  65. {sigit_code-1.5.7 → sigit_code-1.5.8}/nuget/sigit/Program.cs +0 -0
  66. {sigit_code-1.5.7 → sigit_code-1.5.8}/nuget/sigit/README.md +0 -0
  67. {sigit_code-1.5.7 → sigit_code-1.5.8}/nuget/sigit/SiGit.Code.csproj +0 -0
  68. {sigit_code-1.5.7 → sigit_code-1.5.8}/packaging/aur/PKGBUILD.in +0 -0
  69. {sigit_code-1.5.7 → sigit_code-1.5.8}/packaging/nfpm.yaml +0 -0
  70. {sigit_code-1.5.7 → sigit_code-1.5.8}/packaging/winget/getSigit.siGitCode.installer.yaml.in +0 -0
  71. {sigit_code-1.5.7 → sigit_code-1.5.8}/packaging/winget/getSigit.siGitCode.locale.en-US.yaml.in +0 -0
  72. {sigit_code-1.5.7 → sigit_code-1.5.8}/packaging/winget/getSigit.siGitCode.yaml.in +0 -0
  73. {sigit_code-1.5.7 → sigit_code-1.5.8}/pypi/README.md +0 -0
  74. {sigit_code-1.5.7 → sigit_code-1.5.8}/pypi/pyproject.toml +0 -0
  75. {sigit_code-1.5.7 → sigit_code-1.5.8}/pyproject.toml +0 -0
  76. {sigit_code-1.5.7 → sigit_code-1.5.8}/rust-toolchain.toml +0 -0
  77. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/account.rs +0 -0
  78. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/browser_auth.rs +0 -0
  79. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/commands.rs +0 -0
  80. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/credentials.rs +0 -0
  81. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/frontmatter.rs +0 -0
  82. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/headless.rs +0 -0
  83. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/hooks.rs +0 -0
  84. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/instructions.rs +0 -0
  85. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/mcp.rs +0 -0
  86. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/provider.rs +0 -0
  87. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/session_store.rs +0 -0
  88. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/setup.rs +0 -0
  89. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/skills.rs +0 -0
  90. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/subagents.rs +0 -0
  91. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/tools.rs +0 -0
  92. {sigit_code-1.5.7 → sigit_code-1.5.8}/src/workspace.rs +0 -0
  93. {sigit_code-1.5.7 → sigit_code-1.5.8}/tests/acp_multi_root.rs +0 -0
  94. {sigit_code-1.5.7 → sigit_code-1.5.8}/tests/acp_tool_stdin.rs +0 -0
  95. {sigit_code-1.5.7 → sigit_code-1.5.8}/tests/headless_mode.rs +0 -0
@@ -1,5 +1,81 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.5.8
4
+
5
+ ### What changed
6
+
7
+ - **Editor panels can switch permission mode without a slash command.** ACP
8
+ clients now get a Permissions selector next to the model controls with
9
+ Manual, Auto, and Plan choices for the current session. Manual keeps the
10
+ existing approval prompts, Auto lets mutating tools run unattended while
11
+ still respecting explicit deny rules in `settings.toml`, and Plan keeps the
12
+ agent in research-only mode. The selector follows `/plan` and `/clear`, and
13
+ it is deliberately session-scoped so a risky Auto choice does not persist
14
+ into the next task
15
+ - **Tool calls in the editor are worth opening now.** Zed only drew the
16
+ disclosure arrow on the handful of cards siGit set `content` on — the model
17
+ download and switch spinners — because everything else carried nothing but
18
+ `rawInput` and `rawOutput`, fields ACP gives clients no display guidance for.
19
+ Every tool call now gets a fenced content block: the pretty-printed arguments
20
+ while it runs, then the tool's output once it finishes, with a `(no output)`
21
+ placeholder for a silent command like `git add`. Cards are titled
22
+ `<tool> · <arg>` rather than the bare tool name, path-bearing tools set
23
+ `locations` so the editor can follow along, and a replayed session gets the
24
+ same treatment as a live one
25
+ - **The context window is visible before compaction fires.** Both model pickers
26
+ show each model's window, and the TUI title bar has a gauge for how much of
27
+ it the conversation is using. Cloud tiers report their own window instead of
28
+ the compaction budget — the budget is when siGit Code summarizes history, not
29
+ how much the model can hold, and labelling one with the other understated the
30
+ window by an order of magnitude
31
+ - **`write_todos` renders as a plan, not a tool card.** Zed and other clients
32
+ have real progress UI for `session/update` plans, so the model's todo list
33
+ goes there
34
+ - **Picker changes read as status rather than chat.** Switching model or
35
+ inference backend is UI state, so it renders as a completed think-kind tool
36
+ call. The sign-in prompt stays an assistant message, since it needs the user
37
+ to act on it
38
+
39
+ ### Fixed
40
+
41
+ - **A reopened thread comes back with its history.** Clicking a saved thread in
42
+ Zed sends `session/load`, and the client draws the thread purely from the
43
+ `session/update` notifications the agent streams while that request is in
44
+ flight. siGit restored the saved history into the backend, which is what
45
+ makes the model remember, but sent the client nothing — so the thread opened
46
+ empty and looked like a brand new conversation. The snapshot is now turned
47
+ into updates before it is restored: user and assistant text as message chunks
48
+ with reasoning stripped, each tool call completed with its result folded in,
49
+ and `write_todos` as a plan the way it renders live. System messages stay
50
+ out, since they seeded the model and were never on screen
51
+ - **Compaction no longer fails in every session that ran a tool.**
52
+ `compact_history` asked for the summary through `complete(None, None)`, which
53
+ sends no tools array but left the live history in place. That history is
54
+ thick with assistant `tool_calls` and `role: "tool"` messages, and an
55
+ endpoint handed tool shapes with no schema to check them against rejects the
56
+ request — Anthropic answers 400. So compaction failed on every attempt in any
57
+ session that had run a single tool, however small the history was, and then
58
+ retried on every prompt and tool round while that history kept growing. The
59
+ conversation now goes as a flattened transcript in one user message, which
60
+ keeps what the summary needs and drops the shapes that only mean anything
61
+ next to a tool schema
62
+ - **A tool call emitted as text is no longer dropped.** Qwen 3, GLM and
63
+ DeepSeek write a call as `<tool_call>NAME<arg_key>…` in their chat template
64
+ and rely on the serving stack to parse it back into `tool_calls`. When that
65
+ doesn't happen the tag arrives as ordinary content, so it was rendered
66
+ verbatim in the editor and the turn ended as though the model had chosen to
67
+ answer in prose. Both the streaming and non-streaming paths now scan content
68
+ for those blocks and turn well-formed ones back into real calls, typing
69
+ argument values from the turn's own tool schemas. The streaming scanner holds
70
+ back only enough text to catch a tag straddling a chunk boundary, so ordinary
71
+ answers still stream token by token. A block that doesn't match the expected
72
+ shape is left in the text untouched: reissuing a `run_command` is cheap, but
73
+ guessing wrong at a half-parsed `edit_file` would write the wrong change to a
74
+ file
75
+ - **Tool-call content survives awkward output.** Whitespace-only output is
76
+ preserved rather than collapsed, malformed arguments no longer get a
77
+ misleading JSON fence, and fence language identifiers are sanitized
78
+
3
79
  ## 1.5.7
4
80
 
5
81
  ### What changed
@@ -5778,7 +5778,7 @@ checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
5778
5778
 
5779
5779
  [[package]]
5780
5780
  name = "sigit"
5781
- version = "1.5.7"
5781
+ version = "1.5.8"
5782
5782
  dependencies = [
5783
5783
  "agent-client-protocol",
5784
5784
  "anyhow",
@@ -1,6 +1,6 @@
1
1
  [package]
2
2
  name = "sigit"
3
- version = "1.5.7"
3
+ version = "1.5.8"
4
4
  edition = "2024"
5
5
  description = "siGit Code — ACP-compatible AI coding agent. Sí, git."
6
6
  documentation = "https://github.com/getsigit/sigit"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: sigit-code
3
- Version: 1.5.7
3
+ Version: 1.5.8
4
4
  Classifier: Development Status :: 4 - Beta
5
5
  Classifier: Environment :: Console
6
6
  Classifier: Intended Audience :: Developers
@@ -28,7 +28,8 @@ use tokio::sync::Mutex;
28
28
  // ── Neutral types ───────────────────────────────────────────────────────────────
29
29
 
30
30
  /// A tool the model may call, in a provider-neutral form. `parameters_schema` is
31
- /// a JSON Schema encoded as a string (matching how siGit already declares tools).
31
+ /// a JSON Schema encoded as a string (matching how siGit Code already declares
32
+ /// tools).
32
33
  #[derive(Debug, Clone)]
33
34
  pub struct ToolSpec {
34
35
  pub name: String,
@@ -75,6 +76,51 @@ pub const COMPACT_KEEP_LAST: usize = 6;
75
76
  const SUMMARIZE_PROMPT: &str = "Summarize this coding session so far: decisions made, \
76
77
  files touched, current state, open items. Be concise and factual.";
77
78
 
79
+ /// Render a history snapshot as a plain-text transcript, with tool calls and
80
+ /// tool results spelled out as prose rather than left in their wire shapes.
81
+ ///
82
+ /// Compaction summarizes a conversation that is, by definition, thick with
83
+ /// `tool_calls` and `role: "tool"` messages — but the summarization round asks
84
+ /// for a plain answer and so sends no `tools` array. Forwarding the raw shapes
85
+ /// in that request produces tool blocks with no schema to validate against,
86
+ /// which strict endpoints reject outright: Anthropic answers 400, so every
87
+ /// compaction of a session that had ever run a tool failed, permanently, no
88
+ /// matter how small the history was. Flattening to text keeps everything the
89
+ /// summary actually needs and drops the shapes that only make sense alongside
90
+ /// a tool schema. It also sidesteps orphaned `tool_call_id`s and role-
91
+ /// alternation rules, neither of which a transcript can violate.
92
+ fn transcript_for_summary(history: &[serde_json::Value]) -> String {
93
+ let mut lines: Vec<String> = Vec::new();
94
+ for message in history {
95
+ let role = message["role"].as_str().unwrap_or("user");
96
+ // The system prompt is carried over verbatim, so it needn't be summarized.
97
+ if role == "system" {
98
+ continue;
99
+ }
100
+
101
+ let mut parts: Vec<String> = Vec::new();
102
+ if let Some(text) = message["content"].as_str()
103
+ && !text.trim().is_empty()
104
+ {
105
+ parts.push(text.to_string());
106
+ }
107
+ for call in message["tool_calls"].as_array().into_iter().flatten() {
108
+ parts.push(format!(
109
+ "called {}({})",
110
+ call["function"]["name"].as_str().unwrap_or("tool"),
111
+ call["function"]["arguments"].as_str().unwrap_or_default(),
112
+ ));
113
+ }
114
+ if parts.is_empty() {
115
+ continue;
116
+ }
117
+
118
+ let label = if role == "tool" { "tool result" } else { role };
119
+ lines.push(format!("{label}: {}", parts.join("\n")));
120
+ }
121
+ lines.join("\n\n")
122
+ }
123
+
78
124
  /// Crude token estimate for a history snapshot: serialized characters / 4.
79
125
  /// Deliberately model-agnostic — it only needs to be in the right ballpark to
80
126
  /// decide when compaction is worth an extra inference round.
@@ -482,15 +528,22 @@ impl OpenAiBackend {
482
528
  return Err(describe_api_error(status, &body));
483
529
  }
484
530
 
531
+ // The specs are needed downstream to type the arguments of any tool
532
+ // call the model emitted as text rather than as a structured call.
533
+ let tools = tools.unwrap_or(&[]);
485
534
  if let Some(sink) = sink {
486
- self.consume_stream(response, sink).await
535
+ self.consume_stream(response, sink, tools).await
487
536
  } else {
488
- self.consume_json(response).await
537
+ self.consume_json(response, tools).await
489
538
  }
490
539
  }
491
540
 
492
541
  /// Parse a single non-streaming chat-completion response.
493
- async fn consume_json(&self, response: reqwest::Response) -> Result<TurnResult, BackendError> {
542
+ async fn consume_json(
543
+ &self,
544
+ response: reqwest::Response,
545
+ tools: &[ToolSpec],
546
+ ) -> Result<TurnResult, BackendError> {
494
547
  let parsed: ChatCompletion = response
495
548
  .json()
496
549
  .await
@@ -515,6 +568,41 @@ impl OpenAiBackend {
515
568
  })
516
569
  .collect();
517
570
 
571
+ // Some models write a tool call out as literal `<tool_call>` text
572
+ // instead of using the structured field (see `inline_tool_calls`).
573
+ // Recover it, or the turn ends with the tag rendered as prose and
574
+ // whatever the model meant to do is dropped.
575
+ if tool_calls.is_empty() {
576
+ let (cleaned, recovered) = crate::inline_tool_calls::extract(&text, tools);
577
+ if !recovered.is_empty() {
578
+ log::warn!(
579
+ "recovered {} tool call(s) the model emitted as text instead of a structured call",
580
+ recovered.len()
581
+ );
582
+ let tool_calls: Vec<ToolCall> = recovered
583
+ .into_iter()
584
+ .enumerate()
585
+ .map(|(index, call)| ToolCall {
586
+ id: format!("call_recovered_{index}"),
587
+ name: call.name,
588
+ arguments: call.arguments,
589
+ })
590
+ .collect();
591
+ // Record the recovered shape, not the raw tag: the tool
592
+ // results that follow have to answer an assistant message
593
+ // that actually carries these calls, or the next request is
594
+ // rejected for orphaned tool results.
595
+ self.history
596
+ .lock()
597
+ .await
598
+ .push(streamed_assistant_history(&cleaned, &tool_calls));
599
+ return Ok(TurnResult {
600
+ text: cleaned,
601
+ tool_calls,
602
+ });
603
+ }
604
+ }
605
+
518
606
  // Record the assistant turn so later tool results have context.
519
607
  self.history.lock().await.push(message.into_history_value());
520
608
 
@@ -528,6 +616,7 @@ impl OpenAiBackend {
528
616
  &self,
529
617
  response: reqwest::Response,
530
618
  sink: &TokenSink,
619
+ tools: &[ToolSpec],
531
620
  ) -> Result<TurnResult, BackendError> {
532
621
  use futures::StreamExt;
533
622
 
@@ -538,6 +627,13 @@ impl OpenAiBackend {
538
627
  let mut text = String::new();
539
628
  let mut tool_accum: Vec<StreamingToolCall> = Vec::new();
540
629
  let mut done = false;
630
+ // Recovers a tool call the model wrote as literal `<tool_call>` text
631
+ // instead of a structured delta. Scanning here (rather than after the
632
+ // stream) keeps the tag off the UI: content goes straight to `sink` as
633
+ // it arrives, so by the time a whole turn is assembled the tag has
634
+ // already been rendered. See `inline_tool_calls`.
635
+ let mut scanner = crate::inline_tool_calls::StreamScanner::new(tools);
636
+ let mut recovered: Vec<ToolCall> = Vec::new();
541
637
 
542
638
  while let Some(item) = stream.next().await {
543
639
  let bytes = item.map_err(|error| format!("stream read error: {error}"))?;
@@ -583,9 +679,31 @@ impl OpenAiBackend {
583
679
  if let Some(content) = choice.delta.content
584
680
  && !content.is_empty()
585
681
  {
586
- text.push_str(&content);
587
- if sink.send(content).is_err() {
588
- // Consumer dropped (turn cancelled) — stop reading.
682
+ let mut cancelled = false;
683
+ for event in scanner.push(&content) {
684
+ match event {
685
+ crate::inline_tool_calls::ScanEvent::Text(chunk) => {
686
+ text.push_str(&chunk);
687
+ if sink.send(chunk).is_err() {
688
+ // Consumer dropped (turn cancelled).
689
+ cancelled = true;
690
+ break;
691
+ }
692
+ }
693
+ crate::inline_tool_calls::ScanEvent::ToolCall(call) => {
694
+ log::warn!(
695
+ "recovered tool call '{}' the model emitted as text instead of a structured call",
696
+ call.name
697
+ );
698
+ recovered.push(ToolCall {
699
+ id: format!("call_recovered_{}", recovered.len()),
700
+ name: call.name,
701
+ arguments: call.arguments,
702
+ });
703
+ }
704
+ }
705
+ }
706
+ if cancelled {
589
707
  done = true;
590
708
  break;
591
709
  }
@@ -615,7 +733,13 @@ impl OpenAiBackend {
615
733
  }
616
734
  }
617
735
 
618
- let tool_calls: Vec<ToolCall> = tool_accum
736
+ // Text held back waiting on a tag that never closed is just text.
737
+ if let Some(leftover) = scanner.take_pending() {
738
+ text.push_str(&leftover);
739
+ let _ = sink.send(leftover);
740
+ }
741
+
742
+ let mut tool_calls: Vec<ToolCall> = tool_accum
619
743
  .iter()
620
744
  .filter(|call| !call.name.is_empty())
621
745
  .enumerate()
@@ -629,6 +753,7 @@ impl OpenAiBackend {
629
753
  arguments: call.arguments.clone(),
630
754
  })
631
755
  .collect();
756
+ tool_calls.extend(recovered);
632
757
 
633
758
  // Record the assistant turn so later tool results have context.
634
759
  self.history
@@ -738,12 +863,30 @@ impl InferenceBackend for OpenAiBackend {
738
863
  async fn compact_history(&self, keep_last: usize) -> Result<(), BackendError> {
739
864
  let snapshot: Vec<serde_json::Value> = self.history.lock().await.clone();
740
865
 
741
- // Ask the endpoint for a summary of the conversation so far, through
742
- // the ordinary completion machinery (non-streaming).
743
- self.history
744
- .lock()
745
- .await
746
- .push(serde_json::json!({ "role": "user", "content": SUMMARIZE_PROMPT }));
866
+ let system = snapshot
867
+ .first()
868
+ .filter(|message| message["role"] == "system")
869
+ .cloned();
870
+
871
+ // Ask the endpoint for a summary of the conversation so far, through the
872
+ // ordinary completion machinery (non-streaming). The request carries the
873
+ // conversation as a flattened transcript in a single user message rather
874
+ // than the live history: this round offers no tools, and a tool-shaped
875
+ // history sent without a tool schema is rejected upstream (see
876
+ // `transcript_for_summary`).
877
+ let mut request = Vec::new();
878
+ if let Some(system) = system.clone() {
879
+ request.push(system);
880
+ }
881
+ request.push(serde_json::json!({
882
+ "role": "user",
883
+ "content": format!(
884
+ "{}\n\n{SUMMARIZE_PROMPT}",
885
+ transcript_for_summary(&snapshot),
886
+ ),
887
+ }));
888
+ *self.history.lock().await = request;
889
+
747
890
  let summary = match self.complete(None, None).await {
748
891
  Ok(result) => result.text,
749
892
  Err(error) => {
@@ -753,10 +896,6 @@ impl InferenceBackend for OpenAiBackend {
753
896
  }
754
897
  };
755
898
 
756
- let system = snapshot
757
- .first()
758
- .filter(|message| message["role"] == "system")
759
- .cloned();
760
899
  let non_system: Vec<serde_json::Value> = snapshot
761
900
  .iter()
762
901
  .filter(|message| message["role"] != "system")
@@ -1342,12 +1481,16 @@ mod tests {
1342
1481
  }
1343
1482
 
1344
1483
  /// Minimal scripted OpenAI-compatible endpoint: accepts one HTTP request on
1345
- /// a std listener and answers with a fixed non-streaming completion.
1346
- fn spawn_completion_stub(summary: &str) -> std::net::SocketAddr {
1484
+ /// a std listener and answers with a fixed non-streaming completion. The
1485
+ /// receiver yields the request body the backend actually put on the wire.
1486
+ fn spawn_completion_stub(
1487
+ summary: &str,
1488
+ ) -> (std::net::SocketAddr, std::sync::mpsc::Receiver<String>) {
1347
1489
  use std::io::{Read, Write};
1348
1490
 
1349
1491
  let listener = std::net::TcpListener::bind("127.0.0.1:0").unwrap();
1350
1492
  let addr = listener.local_addr().unwrap();
1493
+ let (sender, receiver) = std::sync::mpsc::channel();
1351
1494
  let body = serde_json::json!({
1352
1495
  "choices": [{ "message": { "role": "assistant", "content": summary } }]
1353
1496
  })
@@ -1376,6 +1519,9 @@ mod tests {
1376
1519
  })
1377
1520
  .unwrap_or(0);
1378
1521
  if request.len() >= headers_end + 4 + content_length {
1522
+ let _ = sender.send(
1523
+ String::from_utf8_lossy(&request[headers_end + 4..]).into_owned(),
1524
+ );
1379
1525
  break;
1380
1526
  }
1381
1527
  }
@@ -1388,12 +1534,12 @@ mod tests {
1388
1534
  );
1389
1535
  let _ = stream.write_all(response.as_bytes());
1390
1536
  });
1391
- addr
1537
+ (addr, receiver)
1392
1538
  }
1393
1539
 
1394
1540
  #[tokio::test]
1395
1541
  async fn compact_history_rebuilds_system_summary_and_tail() {
1396
- let addr = spawn_completion_stub("We refactored backend.rs; tests pass.");
1542
+ let (addr, _requests) = spawn_completion_stub("We refactored backend.rs; tests pass.");
1397
1543
  let backend = OpenAiBackend::new(
1398
1544
  format!("http://{addr}/v1"),
1399
1545
  "test-key",
@@ -1430,6 +1576,66 @@ mod tests {
1430
1576
  );
1431
1577
  }
1432
1578
 
1579
+ /// Compacting a tool-heavy session must not put tool shapes on the wire.
1580
+ /// The summarization round offers no `tools`, and endpoints reject tool
1581
+ /// calls and tool results that arrive without a schema — which used to make
1582
+ /// compaction fail forever in any session that had run a single tool.
1583
+ #[tokio::test]
1584
+ async fn compact_history_sends_no_tool_artifacts() {
1585
+ let (addr, requests) = spawn_completion_stub("Ran git status on main.");
1586
+ let backend = OpenAiBackend::new(
1587
+ format!("http://{addr}/v1"),
1588
+ "test-key",
1589
+ "test-model",
1590
+ Some("be helpful".into()),
1591
+ );
1592
+ {
1593
+ let mut history = backend.history.lock().await;
1594
+ history.push(serde_json::json!({ "role": "user", "content": "check the repo" }));
1595
+ history.push(serde_json::json!({
1596
+ "role": "assistant",
1597
+ "content": null,
1598
+ "tool_calls": [{
1599
+ "id": "call_1",
1600
+ "type": "function",
1601
+ "function": { "name": "run_command", "arguments": "{\"command\":\"git status\"}" },
1602
+ }],
1603
+ }));
1604
+ history.push(serde_json::json!({
1605
+ "role": "tool",
1606
+ "tool_call_id": "call_1",
1607
+ "content": "on branch main",
1608
+ }));
1609
+ }
1610
+
1611
+ backend.compact_history(2).await.unwrap();
1612
+
1613
+ let body: serde_json::Value =
1614
+ serde_json::from_str(&requests.recv().unwrap()).expect("request body is JSON");
1615
+ assert!(
1616
+ body.get("tools").is_none(),
1617
+ "summarization offers no tools: {body}"
1618
+ );
1619
+ for message in body["messages"].as_array().unwrap() {
1620
+ assert!(
1621
+ message.get("tool_calls").is_none(),
1622
+ "no tool_calls may be sent without a schema: {message}"
1623
+ );
1624
+ assert_ne!(
1625
+ message["role"], "tool",
1626
+ "no tool results may be sent without a schema: {message}"
1627
+ );
1628
+ }
1629
+
1630
+ // The tool round still has to survive into the summary request as prose,
1631
+ // or the summary loses the work the session actually did.
1632
+ let transcript = body["messages"].as_array().unwrap().last().unwrap()["content"]
1633
+ .as_str()
1634
+ .unwrap();
1635
+ assert!(transcript.contains("called run_command({\"command\":\"git status\"})"));
1636
+ assert!(transcript.contains("tool result: on branch main"));
1637
+ }
1638
+
1433
1639
  #[tokio::test]
1434
1640
  async fn compact_history_failure_leaves_history_intact() {
1435
1641
  // No listener at this address: the summarization request fails, and