millforge 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. millforge/__init__.py +1174 -0
  2. millforge/_forge/LICENSE +21 -0
  3. millforge/_forge/PROVENANCE.json +295 -0
  4. millforge/_forge/UPDATE_POLICY.md +24 -0
  5. millforge/_forge/__init__.py +14 -0
  6. millforge/_forge/adapter.py +2232 -0
  7. millforge/_forge/base_runner.py +121 -0
  8. millforge/_forge/clients/__init__.py +10 -0
  9. millforge/_forge/clients/base.py +200 -0
  10. millforge/_forge/context/__init__.py +23 -0
  11. millforge/_forge/context/manager.py +178 -0
  12. millforge/_forge/context/strategies.py +335 -0
  13. millforge/_forge/core/__init__.py +16 -0
  14. millforge/_forge/core/inference.py +433 -0
  15. millforge/_forge/core/messages.py +119 -0
  16. millforge/_forge/core/runner.py +479 -0
  17. millforge/_forge/core/steps.py +108 -0
  18. millforge/_forge/core/workflow.py +400 -0
  19. millforge/_forge/errors.py +222 -0
  20. millforge/_forge/guardrails/__init__.py +21 -0
  21. millforge/_forge/guardrails/error_tracker.py +71 -0
  22. millforge/_forge/guardrails/guardrails.py +194 -0
  23. millforge/_forge/guardrails/nudge.py +47 -0
  24. millforge/_forge/guardrails/response_validator.py +119 -0
  25. millforge/_forge/guardrails/step_enforcer.py +183 -0
  26. millforge/_forge/prompts/__init__.py +16 -0
  27. millforge/_forge/prompts/nudges.py +95 -0
  28. millforge/_forge/prompts/templates.py +285 -0
  29. millforge/_version.py +3 -0
  30. millforge/artifacts.py +570 -0
  31. millforge/base/__init__.py +97 -0
  32. millforge/base/composition.py +402 -0
  33. millforge/base/context.py +285 -0
  34. millforge/base/harness.py +138 -0
  35. millforge/base/identity.py +465 -0
  36. millforge/base/options.py +34 -0
  37. millforge/base/platform.py +17 -0
  38. millforge/base/prompt.py +317 -0
  39. millforge/base/runner.py +546 -0
  40. millforge/compiled_plan.py +970 -0
  41. millforge/compiler/__init__.py +231 -0
  42. millforge/compiler/artifact_validation.py +257 -0
  43. millforge/compiler/canonicalization.py +169 -0
  44. millforge/compiler/capabilities.py +66 -0
  45. millforge/compiler/catalogs.py +500 -0
  46. millforge/compiler/diagnostics.py +491 -0
  47. millforge/compiler/graph.py +678 -0
  48. millforge/compiler/lowering.py +198 -0
  49. millforge/compiler/output.py +692 -0
  50. millforge/compiler/parsing.py +1424 -0
  51. millforge/compiler/requests.py +1180 -0
  52. millforge/compiler/schema_validation.py +272 -0
  53. millforge/compiler/semantic.py +490 -0
  54. millforge/compiler/service.py +448 -0
  55. millforge/compiler/source.py +375 -0
  56. millforge/compiler/validators.py +184 -0
  57. millforge/connectors/__init__.py +95 -0
  58. millforge/connectors/admission.py +801 -0
  59. millforge/connectors/broker.py +202 -0
  60. millforge/connectors/contracts.py +1159 -0
  61. millforge/connectors/diagnostics.py +189 -0
  62. millforge/connectors/fake.py +66 -0
  63. millforge/connectors/runtime.py +236 -0
  64. millforge/contracts.py +2860 -0
  65. millforge/custom_tools/__init__.py +67 -0
  66. millforge/custom_tools/compiler.py +724 -0
  67. millforge/custom_tools/contracts.py +1093 -0
  68. millforge/custom_tools/diagnostics.py +205 -0
  69. millforge/eval_artifacts.py +952 -0
  70. millforge/eval_boundary.py +2435 -0
  71. millforge/eval_fixtures/__init__.py +1 -0
  72. millforge/eval_fixtures/default_pack/__init__.py +1 -0
  73. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.bug_diagnosis.traceback.v1.json +52 -0
  74. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.direct_edit.import_sort.v1.json +52 -0
  75. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.evidence_discipline.no_source_change.v1.json +51 -0
  76. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.false_closure.visible_green.v1.json +52 -0
  77. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.multi_file.api_contract.v1.json +54 -0
  78. millforge/eval_fixtures/default_pack/fixtures/fixture.08a.recovery.malformed_artifact.v1.json +54 -0
  79. millforge/eval_fixtures/default_pack/manifest.json +12 -0
  80. millforge/eval_modes.py +1282 -0
  81. millforge/eval_presets.py +1398 -0
  82. millforge/eval_reports.py +2517 -0
  83. millforge/eval_suite.py +2429 -0
  84. millforge/eval_trials.py +2632 -0
  85. millforge/eval_workflow.py +794 -0
  86. millforge/exceptions.py +122 -0
  87. millforge/model_backend.py +2098 -0
  88. millforge/protocols.py +340 -0
  89. millforge/py.typed +0 -0
  90. millforge/runtime.py +1791 -0
  91. millforge/testing/__init__.py +1089 -0
  92. millforge/tools/__init__.py +83 -0
  93. millforge/tools/builtin_runtime.py +1339 -0
  94. millforge/tools/builtins.py +773 -0
  95. millforge/tools/execution.py +1545 -0
  96. millforge/tools/path_policy.py +155 -0
  97. millforge/tools/pi_compat/PI_LICENSE +21 -0
  98. millforge/tools/pi_compat/PROVENANCE.json +55 -0
  99. millforge/tools/pi_compat/UPDATE_POLICY.md +36 -0
  100. millforge/tools/pi_compat/__init__.py +34 -0
  101. millforge/tools/pi_compat/contracts.py +49 -0
  102. millforge/tools/pi_compat/editing.py +390 -0
  103. millforge/tools/pi_compat/mutations.py +57 -0
  104. millforge/tools/pi_compat/operations.py +401 -0
  105. millforge/tools/pi_compat/paths.py +155 -0
  106. millforge/tools/pi_compat/process.py +1375 -0
  107. millforge/tools/pi_compat/search.py +738 -0
  108. millforge/tools/pi_compat/truncation.py +267 -0
  109. millforge/tools/pi_compat_catalog.py +396 -0
  110. millforge/tools/pi_compat_runtime.py +460 -0
  111. millforge/tools/registry.py +553 -0
  112. millforge/tools/results.py +533 -0
  113. millforge-0.1.0.dist-info/METADATA +844 -0
  114. millforge-0.1.0.dist-info/RECORD +116 -0
  115. millforge-0.1.0.dist-info/WHEEL +4 -0
  116. millforge-0.1.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,335 @@
1
+ """Compaction strategies for context window management."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from abc import ABC, abstractmethod
6
+ from dataclasses import replace
7
+
8
+ from millforge._forge.core.messages import Message, MessageRole, MessageType
9
+
10
+
11
+ def _estimate_tokens(messages: list[Message]) -> int:
12
+ return (
13
+ sum(
14
+ len(message.content) + len(message.reasoning_content or "")
15
+ for message in messages
16
+ )
17
+ // 4
18
+ )
19
+
20
+
21
+ def _replay_steps(
22
+ messages: list[Message],
23
+ eligible_end: int,
24
+ ) -> tuple[set[int], set[int]]:
25
+ replay_steps: set[int] = set()
26
+ complete_steps: set[int] = set()
27
+ for index, message in enumerate(messages[:eligible_end]):
28
+ step = message.metadata.step_index
29
+ if (
30
+ index < 2
31
+ or step is None
32
+ or not message.tool_calls
33
+ or message.reasoning_content is None
34
+ ):
35
+ continue
36
+ replay_steps.add(step)
37
+ step_messages = [
38
+ item
39
+ for item in messages[2:eligible_end]
40
+ if item.metadata.step_index == step
41
+ ]
42
+ call_messages = [item for item in step_messages if item.tool_calls]
43
+ if len(call_messages) != 1:
44
+ continue
45
+ call_message = call_messages[0]
46
+ call_index = step_messages.index(call_message)
47
+ results = [
48
+ item
49
+ for item in step_messages[call_index + 1 :]
50
+ if item.role is MessageRole.TOOL
51
+ ]
52
+ calls = call_message.tool_calls or []
53
+ if len(results) != len(calls):
54
+ continue
55
+ if all(
56
+ result.tool_call_id == call.call_id and result.tool_name == call.name
57
+ for call, result in zip(calls, results, strict=True)
58
+ ):
59
+ complete_steps.add(step)
60
+ return replay_steps, complete_steps
61
+
62
+
63
+ class CompactStrategy(ABC):
64
+ """Interface for context compaction strategies.
65
+
66
+ Recommended compaction priority (cut first -> preserve longest):
67
+ 1. step_nudge, retry_nudge — ephemeral corrections, no long-term value
68
+ 2. tool_result — truncate to first line; raw data is expendable once processed
69
+ 3. tool_call — collapse to one-liner (tool name + args)
70
+ 4. reasoning — preserve as long as possible; this is the model's interpretive context
71
+ 5. Recent iterations (within keep_recent window) — fully intact
72
+ """
73
+
74
+ @abstractmethod
75
+ def compact(
76
+ self,
77
+ messages: list[Message],
78
+ budget_tokens: int,
79
+ *,
80
+ step_hint: str = "",
81
+ ) -> tuple[list[Message], int]:
82
+ """Return a compacted copy of the message history and the phase reached.
83
+
84
+ Returns a tuple of (compacted_messages, phase_reached). The phase int
85
+ indicates how aggressively the strategy compacted: 0 means no
86
+ compaction was applied, 1+ is implementation-defined. Strategies
87
+ without internal phases should return 1.
88
+
89
+ The strategy owns its own threshold logic. It receives the full
90
+ budget_tokens and decides whether to compact and how aggressively.
91
+ Return phase 0 if no compaction was needed.
92
+
93
+ Must preserve (never cut):
94
+ - The system prompt (messages[0])
95
+ - The original user input (messages[1])
96
+ """
97
+ ...
98
+
99
+
100
+ class NoCompact(CompactStrategy):
101
+ """Passthrough strategy. Returns messages unchanged.
102
+
103
+ Use when VRAM is abundant (32GB+) or workflows are short.
104
+ """
105
+
106
+ def compact(
107
+ self,
108
+ messages: list[Message],
109
+ budget_tokens: int,
110
+ *,
111
+ step_hint: str = "",
112
+ ) -> tuple[list[Message], int]:
113
+ return list(messages), 0
114
+
115
+
116
+ class SlidingWindowCompact(CompactStrategy):
117
+ """Keeps the system prompt, original user input, and the last N iterations.
118
+
119
+ Simple and predictable. Good baseline for testing. Uses step_index to
120
+ identify iteration boundaries (handles variable-size parallel tool batches).
121
+ """
122
+
123
+ def __init__(self, keep_recent: int, compact_threshold: float = 0.75) -> None:
124
+ self.keep_recent = keep_recent
125
+ self.compact_threshold = compact_threshold
126
+
127
+ def compact(
128
+ self,
129
+ messages: list[Message],
130
+ budget_tokens: int,
131
+ *,
132
+ step_hint: str = "",
133
+ ) -> tuple[list[Message], int]:
134
+ trigger = int(budget_tokens * self.compact_threshold)
135
+ if _estimate_tokens(messages) < trigger:
136
+ return list(messages), 0
137
+ eligible_end = TieredCompact._find_eligible_end(messages, self.keep_recent)
138
+ if eligible_end <= 2:
139
+ return list(messages), 1
140
+ return [messages[0], messages[1]] + messages[eligible_end:], 1
141
+
142
+
143
+ class TieredCompact(CompactStrategy):
144
+ """Three-phase compaction with explicit priority order.
145
+
146
+ Each phase fires only if the previous phase didn't reduce tokens below
147
+ the trigger threshold. keep_recent controls how many recent loop iterations
148
+ (each iteration = one assistant message + N tool result messages) are
149
+ fully preserved before older content is eligible for compaction.
150
+
151
+ Phase priority (cut first -> preserve longest):
152
+ 1. Nudges/retries dropped, tool_results truncated to first ~200 chars
153
+ 2. Tool_results dropped entirely — reasoning and text_response preserved
154
+ 3. Reasoning and text_response dropped — only tool_call skeleton remains
155
+ """
156
+
157
+ TRUNCATE_CHARS = 200
158
+
159
+ def __init__(
160
+ self,
161
+ keep_recent: int = 2,
162
+ compact_threshold: float = 0.75,
163
+ phase_thresholds: tuple[float, float, float] | None = None,
164
+ ) -> None:
165
+ """
166
+ Args:
167
+ keep_recent: Number of recent loop iterations to keep fully intact.
168
+ Tune based on workflow depth — shallow workflows (3-5 steps)
169
+ can use 2-3, deep workflows (8-10+) may need 4-6.
170
+ compact_threshold: Fraction of budget that triggers compaction.
171
+ Used as the threshold for all three phases when
172
+ phase_thresholds is not set.
173
+ phase_thresholds: Per-phase compaction thresholds as fractions
174
+ of the context budget. A tuple of (phase1, phase2, phase3).
175
+ Example: ``(0.60, 0.75, 0.90)`` means Phase 1 fires at 60%,
176
+ Phase 2 at 75%, Phase 3 at 90%. Overrides compact_threshold.
177
+ """
178
+ self.keep_recent = keep_recent
179
+ if phase_thresholds is not None:
180
+ self._phase_triggers = phase_thresholds
181
+ else:
182
+ self._phase_triggers = (
183
+ compact_threshold,
184
+ compact_threshold,
185
+ compact_threshold,
186
+ )
187
+
188
+ @staticmethod
189
+ def _find_eligible_end(messages: list[Message], keep_recent: int) -> int:
190
+ """Find the boundary index: messages before this are eligible for compaction.
191
+
192
+ Uses step_index from message metadata to identify iteration boundaries.
193
+ With parallel tool calls, one iteration may produce variable numbers of
194
+ messages (1 TOOL_CALL + N TOOL_RESULTs), so counting by step_index is
195
+ more accurate than a flat message count.
196
+ """
197
+ # Collect distinct step_index values from messages after the protected
198
+ # header (messages[0] and [1] have step_index=None).
199
+ seen_steps: list[int] = []
200
+ for m in messages[2:]:
201
+ si = m.metadata.step_index
202
+ if si is not None and (not seen_steps or seen_steps[-1] != si):
203
+ seen_steps.append(si)
204
+
205
+ if len(seen_steps) <= keep_recent:
206
+ # Not enough iterations to compact anything
207
+ return 2
208
+
209
+ # Protect the last keep_recent iterations
210
+ cutoff_step = seen_steps[-keep_recent]
211
+ # Find the first message index with step_index >= cutoff_step
212
+ for i in range(2, len(messages)):
213
+ si = messages[i].metadata.step_index
214
+ if si is not None and si >= cutoff_step:
215
+ return i
216
+ return len(messages)
217
+
218
+ def compact(
219
+ self,
220
+ messages: list[Message],
221
+ budget_tokens: int,
222
+ *,
223
+ step_hint: str = "",
224
+ ) -> tuple[list[Message], int]:
225
+ """Apply tiered compaction: Phase 1 -> Phase 2 -> Phase 3.
226
+
227
+ Each phase has its own threshold (fraction of budget_tokens).
228
+ A phase only runs if estimated tokens exceed its threshold.
229
+ """
230
+ tokens = _estimate_tokens(messages)
231
+ t1 = int(budget_tokens * self._phase_triggers[0])
232
+ t2 = int(budget_tokens * self._phase_triggers[1])
233
+ t3 = int(budget_tokens * self._phase_triggers[2])
234
+
235
+ # Nothing to do if below the lowest threshold
236
+ if tokens < t1:
237
+ return list(messages), 0
238
+
239
+ # Determine the boundary: everything before this index is eligible
240
+ # messages[0] and messages[1] are always protected
241
+ eligible_end = self._find_eligible_end(messages, self.keep_recent)
242
+
243
+ # Phase 1: Drop nudges/retries, truncate tool_results to first line
244
+ result = self._phase1(messages, eligible_end)
245
+ if _estimate_tokens(result) < t2:
246
+ return result, 1
247
+
248
+ # Phase 2: Phase 1 + drop tool_results entirely
249
+ result = self._phase2(messages, eligible_end)
250
+ if _estimate_tokens(result) < t3:
251
+ return result, 2
252
+
253
+ # Phase 3: Phase 2 + drop reasoning and text_response (tool_call skeleton only)
254
+ result = self._phase3(messages, eligible_end)
255
+ return result, 3
256
+
257
+ def _phase1(self, messages: list[Message], eligible_end: int) -> list[Message]:
258
+ """Drop nudges/retries and truncate tool_results outside keep_recent."""
259
+ result: list[Message] = []
260
+ replay_steps, _ = _replay_steps(messages, eligible_end)
261
+ for i, msg in enumerate(messages):
262
+ if 2 <= i < eligible_end:
263
+ replay_result = (
264
+ msg.metadata.step_index in replay_steps
265
+ and msg.role is MessageRole.TOOL
266
+ )
267
+ if (
268
+ msg.metadata.type
269
+ in (
270
+ MessageType.STEP_NUDGE,
271
+ MessageType.PREREQUISITE_NUDGE,
272
+ MessageType.RETRY_NUDGE,
273
+ )
274
+ and not replay_result
275
+ ):
276
+ continue
277
+ if msg.metadata.type == MessageType.TOOL_RESULT or replay_result:
278
+ if len(msg.content) > self.TRUNCATE_CHARS:
279
+ kept = msg.content[: self.TRUNCATE_CHARS]
280
+ removed = len(msg.content) - self.TRUNCATE_CHARS
281
+ result.append(
282
+ replace(
283
+ msg,
284
+ content=f"{kept}\n[Truncated — {removed} chars removed]",
285
+ )
286
+ )
287
+ continue
288
+ result.append(msg)
289
+ return result
290
+
291
+ def _phase2(self, messages: list[Message], eligible_end: int) -> list[Message]:
292
+ """Phase 1 + drop tool_results entirely. Reasoning and text preserved."""
293
+ result: list[Message] = []
294
+ replay_steps, complete_steps = _replay_steps(messages, eligible_end)
295
+ for i, msg in enumerate(messages):
296
+ if 2 <= i < eligible_end:
297
+ step = msg.metadata.step_index
298
+ if step in complete_steps:
299
+ continue
300
+ if step in replay_steps:
301
+ result.append(msg)
302
+ continue
303
+ if msg.metadata.type in (
304
+ MessageType.STEP_NUDGE,
305
+ MessageType.PREREQUISITE_NUDGE,
306
+ MessageType.RETRY_NUDGE,
307
+ MessageType.TOOL_RESULT,
308
+ ):
309
+ continue
310
+ result.append(msg)
311
+ return result
312
+
313
+ def _phase3(self, messages: list[Message], eligible_end: int) -> list[Message]:
314
+ """Phase 2 + drop reasoning and text_response. Tool_call skeleton only."""
315
+ result: list[Message] = []
316
+ replay_steps, complete_steps = _replay_steps(messages, eligible_end)
317
+ for i, msg in enumerate(messages):
318
+ if 2 <= i < eligible_end:
319
+ step = msg.metadata.step_index
320
+ if step in complete_steps:
321
+ continue
322
+ if step in replay_steps:
323
+ result.append(msg)
324
+ continue
325
+ if msg.metadata.type in (
326
+ MessageType.STEP_NUDGE,
327
+ MessageType.PREREQUISITE_NUDGE,
328
+ MessageType.RETRY_NUDGE,
329
+ MessageType.TOOL_RESULT,
330
+ MessageType.REASONING,
331
+ MessageType.TEXT_RESPONSE,
332
+ ):
333
+ continue
334
+ result.append(msg)
335
+ return result
@@ -0,0 +1,16 @@
1
+ """Private Forge core guarded-loop subset."""
2
+
3
+ from typing import TYPE_CHECKING
4
+
5
+ if TYPE_CHECKING:
6
+ from millforge._forge.core.runner import WorkflowRunner
7
+
8
+ __all__ = ["WorkflowRunner"]
9
+
10
+
11
+ def __getattr__(name: str) -> object:
12
+ if name == "WorkflowRunner":
13
+ from millforge._forge.core.runner import WorkflowRunner
14
+
15
+ return WorkflowRunner
16
+ raise AttributeError(name)