algo-cli-runtime 0.14.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (237) hide show
  1. algo_cli/__init__.py +3 -0
  2. algo_cli/__main__.py +7 -0
  3. algo_cli/_internal/__init__.py +12 -0
  4. algo_cli/_internal/policy_chain.py +259 -0
  5. algo_cli/action_registry.py +1047 -0
  6. algo_cli/agent_blocks.py +550 -0
  7. algo_cli/agent_pipeline.py +1457 -0
  8. algo_cli/agent_threads.py +308 -0
  9. algo_cli/animations.py +316 -0
  10. algo_cli/cache_admission.py +209 -0
  11. algo_cli/capability_mask.py +66 -0
  12. algo_cli/chat_protocol.py +116 -0
  13. algo_cli/chatgpt_auth.py +510 -0
  14. algo_cli/chatgpt_client.py +657 -0
  15. algo_cli/code_rag.py +479 -0
  16. algo_cli/config.py +651 -0
  17. algo_cli/context_budget.py +679 -0
  18. algo_cli/credential_helpers.py +315 -0
  19. algo_cli/deliberation.py +29 -0
  20. algo_cli/display.py +1470 -0
  21. algo_cli/evals/__init__.py +21 -0
  22. algo_cli/evals/algorithm_effectiveness.py +560 -0
  23. algo_cli/evals/competitive_harness_rating.py +702 -0
  24. algo_cli/evals/cot_quality.py +220 -0
  25. algo_cli/evals/harness_retrieval_benchmark.py +401 -0
  26. algo_cli/evals/performance_regression.py +136 -0
  27. algo_cli/evals/scorecard_grading.py +308 -0
  28. algo_cli/evals/session_distribution.py +84 -0
  29. algo_cli/execution_guardrails.py +806 -0
  30. algo_cli/extensions_manifest.py +84 -0
  31. algo_cli/git_evidence.py +227 -0
  32. algo_cli/google_workspace.py +407 -0
  33. algo_cli/google_workspace_auth.py +523 -0
  34. algo_cli/harness.py +2587 -0
  35. algo_cli/identity.py +557 -0
  36. algo_cli/index_compute_lab.py +228 -0
  37. algo_cli/inference_harness.py +70 -0
  38. algo_cli/intelligence/__init__.py +1103 -0
  39. algo_cli/intelligence/acrobat_config.py +307 -0
  40. algo_cli/intelligence/acrobat_manifests.py +338 -0
  41. algo_cli/intelligence/acrobat_models.py +195 -0
  42. algo_cli/intelligence/acrobat_pipeline.py +295 -0
  43. algo_cli/intelligence/acrobat_runtime.py +302 -0
  44. algo_cli/intelligence/acrobat_security.py +261 -0
  45. algo_cli/intelligence/acrobat_workflows.py +226 -0
  46. algo_cli/intelligence/actionability.py +165 -0
  47. algo_cli/intelligence/adversarial_audit.py +136 -0
  48. algo_cli/intelligence/agent_arena.py +92 -0
  49. algo_cli/intelligence/agent_benchmark.py +236 -0
  50. algo_cli/intelligence/agent_runtime.py +171 -0
  51. algo_cli/intelligence/agents_as_tools.py +70 -0
  52. algo_cli/intelligence/artifact_binding.py +80 -0
  53. algo_cli/intelligence/autonomous_engineer.py +1976 -0
  54. algo_cli/intelligence/backpressure.py +99 -0
  55. algo_cli/intelligence/bloom_filter.py +186 -0
  56. algo_cli/intelligence/bonferroni.py +66 -0
  57. algo_cli/intelligence/boundary_compaction.py +98 -0
  58. algo_cli/intelligence/catalog_verifier.py +172 -0
  59. algo_cli/intelligence/cavecrew.py +118 -0
  60. algo_cli/intelligence/changelog.py +176 -0
  61. algo_cli/intelligence/checkpoint_resume.py +92 -0
  62. algo_cli/intelligence/circuit_breaker.py +88 -0
  63. algo_cli/intelligence/clarification_gate.py +101 -0
  64. algo_cli/intelligence/code_graph.py +180 -0
  65. algo_cli/intelligence/coderank.py +97 -0
  66. algo_cli/intelligence/consistent_hash.py +150 -0
  67. algo_cli/intelligence/consortium_synthesis.py +139 -0
  68. algo_cli/intelligence/construction/__init__.py +241 -0
  69. algo_cli/intelligence/construction/common.py +273 -0
  70. algo_cli/intelligence/construction/documents.py +496 -0
  71. algo_cli/intelligence/construction/labor_units.py +1395 -0
  72. algo_cli/intelligence/construction/payments.py +470 -0
  73. algo_cli/intelligence/construction/risk.py +784 -0
  74. algo_cli/intelligence/content_extractor.py +132 -0
  75. algo_cli/intelligence/context_adaptive.py +102 -0
  76. algo_cli/intelligence/context_ops.py +95 -0
  77. algo_cli/intelligence/count_min.py +145 -0
  78. algo_cli/intelligence/cow_state.py +103 -0
  79. algo_cli/intelligence/critic_loop.py +119 -0
  80. algo_cli/intelligence/cross_source.py +113 -0
  81. algo_cli/intelligence/daemon_mode.py +99 -0
  82. algo_cli/intelligence/dag_orchestration.py +151 -0
  83. algo_cli/intelligence/deep_research.py +155 -0
  84. algo_cli/intelligence/degenerate_detector.py +78 -0
  85. algo_cli/intelligence/delta_report.py +92 -0
  86. algo_cli/intelligence/discovery_event_log.py +92 -0
  87. algo_cli/intelligence/document_ingest.py +298 -0
  88. algo_cli/intelligence/dual_layer_validate.py +151 -0
  89. algo_cli/intelligence/echo_fidelity.py +73 -0
  90. algo_cli/intelligence/ema_tuning.py +104 -0
  91. algo_cli/intelligence/event_log.py +92 -0
  92. algo_cli/intelligence/evidence_graph.py +114 -0
  93. algo_cli/intelligence/extension_host.py +162 -0
  94. algo_cli/intelligence/extension_manifest.py +115 -0
  95. algo_cli/intelligence/falsification_suite.py +178 -0
  96. algo_cli/intelligence/finance/__init__.py +169 -0
  97. algo_cli/intelligence/finance/anomalies.py +135 -0
  98. algo_cli/intelligence/finance/ap_ar.py +351 -0
  99. algo_cli/intelligence/finance/cash.py +162 -0
  100. algo_cli/intelligence/finance/close.py +332 -0
  101. algo_cli/intelligence/finance/common.py +244 -0
  102. algo_cli/intelligence/finance/construction.py +135 -0
  103. algo_cli/intelligence/finance/controls.py +172 -0
  104. algo_cli/intelligence/finance/evidence.py +119 -0
  105. algo_cli/intelligence/finance/exceptions.py +157 -0
  106. algo_cli/intelligence/finance/reconciliations.py +254 -0
  107. algo_cli/intelligence/finance/revenue.py +109 -0
  108. algo_cli/intelligence/finance/tax.py +74 -0
  109. algo_cli/intelligence/finance/workpapers.py +111 -0
  110. algo_cli/intelligence/finding_record.py +120 -0
  111. algo_cli/intelligence/flow_dag.py +267 -0
  112. algo_cli/intelligence/gatherer.py +223 -0
  113. algo_cli/intelligence/golden_master.py +98 -0
  114. algo_cli/intelligence/graph_rag.py +195 -0
  115. algo_cli/intelligence/group_chat.py +143 -0
  116. algo_cli/intelligence/hash_dedup.py +145 -0
  117. algo_cli/intelligence/hyperloglog.py +128 -0
  118. algo_cli/intelligence/incremental_index.py +316 -0
  119. algo_cli/intelligence/index_store.py +16 -0
  120. algo_cli/intelligence/iteration_plan.py +133 -0
  121. algo_cli/intelligence/kernel_plugins.py +167 -0
  122. algo_cli/intelligence/lesson_catalog.py +135 -0
  123. algo_cli/intelligence/llm_fallback.py +169 -0
  124. algo_cli/intelligence/log2_histogram.py +267 -0
  125. algo_cli/intelligence/lsp_integration.py +147 -0
  126. algo_cli/intelligence/memory_evolution.py +117 -0
  127. algo_cli/intelligence/minhash_lsh.py +182 -0
  128. algo_cli/intelligence/multi_model_score.py +174 -0
  129. algo_cli/intelligence/multi_tier_grade.py +211 -0
  130. algo_cli/intelligence/negative_controls.py +113 -0
  131. algo_cli/intelligence/numeric_clamp.py +63 -0
  132. algo_cli/intelligence/occ_editor.py +66 -0
  133. algo_cli/intelligence/output_normalize.py +112 -0
  134. algo_cli/intelligence/parallel_delegation.py +98 -0
  135. algo_cli/intelligence/parallel_fanout.py +104 -0
  136. algo_cli/intelligence/permission_modes.py +105 -0
  137. algo_cli/intelligence/pre_push_gate.py +68 -0
  138. algo_cli/intelligence/prefetch.py +171 -0
  139. algo_cli/intelligence/process_framework.py +217 -0
  140. algo_cli/intelligence/project_graph.py +387 -0
  141. algo_cli/intelligence/query_expansion.py +146 -0
  142. algo_cli/intelligence/ralph_loop.py +117 -0
  143. algo_cli/intelligence/rate_limiter.py +153 -0
  144. algo_cli/intelligence/refactor_transaction.py +94 -0
  145. algo_cli/intelligence/research_workspace.py +108 -0
  146. algo_cli/intelligence/retraction_ledger.py +72 -0
  147. algo_cli/intelligence/saga_pattern.py +88 -0
  148. algo_cli/intelligence/session_fork.py +100 -0
  149. algo_cli/intelligence/shadow_editor.py +67 -0
  150. algo_cli/intelligence/shell_session.py +213 -0
  151. algo_cli/intelligence/source_registry.py +143 -0
  152. algo_cli/intelligence/spawn_scales.py +99 -0
  153. algo_cli/intelligence/stat_stability.py +104 -0
  154. algo_cli/intelligence/structural_validator.py +148 -0
  155. algo_cli/intelligence/subagent_spawner.py +111 -0
  156. algo_cli/intelligence/symmetric_verify.py +70 -0
  157. algo_cli/intelligence/task_classifier.py +129 -0
  158. algo_cli/intelligence/team_execution.py +122 -0
  159. algo_cli/intelligence/tiered_access.py +121 -0
  160. algo_cli/intelligence/utility_registry.py +159 -0
  161. algo_cli/intuition_engine.py +560 -0
  162. algo_cli/intuition_injector.py +82 -0
  163. algo_cli/kernels/__init__.py +5 -0
  164. algo_cli/kernels/manifest.py +763 -0
  165. algo_cli/main.py +3903 -0
  166. algo_cli/memory_candidates.py +541 -0
  167. algo_cli/memory_echo_veil.py +394 -0
  168. algo_cli/memory_runtime.py +112 -0
  169. algo_cli/model_info.py +548 -0
  170. algo_cli/model_profile.py +160 -0
  171. algo_cli/model_routing.py +74 -0
  172. algo_cli/oneshot.py +331 -0
  173. algo_cli/perf_telemetry.py +389 -0
  174. algo_cli/plugins.py +245 -0
  175. algo_cli/private_event_store.py +654 -0
  176. algo_cli/quantization/__init__.py +24 -0
  177. algo_cli/quantization/lloyd_max.py +98 -0
  178. algo_cli/quantization/turbo_quant.py +308 -0
  179. algo_cli/reasoning/__init__.py +46 -0
  180. algo_cli/reasoning/combinatorial.py +356 -0
  181. algo_cli/reasoning/graph_of_thought.py +297 -0
  182. algo_cli/reasoning/mcts.py +220 -0
  183. algo_cli/reasoning/neuro_symbolic.py +250 -0
  184. algo_cli/reasoning/react.py +246 -0
  185. algo_cli/reasoning/reflexion.py +225 -0
  186. algo_cli/reasoning/tree_of_thought.py +241 -0
  187. algo_cli/reasoning_bridge.py +150 -0
  188. algo_cli/reconciliation.py +284 -0
  189. algo_cli/reflex.py +385 -0
  190. algo_cli/resources/docs/ALGO.md +13958 -0
  191. algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
  192. algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
  193. algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
  194. algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
  195. algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
  196. algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
  197. algo_cli/resources/docs/main-split-map.md +35 -0
  198. algo_cli/resources/docs/privacy-and-context.md +48 -0
  199. algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
  200. algo_cli/resources/skills/README.md +26 -0
  201. algo_cli/resources/skills/algo-cli.md +59 -0
  202. algo_cli/resources/skills/edit-file-precision.md +49 -0
  203. algo_cli/resources/skills/harness-search-first.md +47 -0
  204. algo_cli/resources/skills/memory-recall-ritual.md +51 -0
  205. algo_cli/resources/skills/qol-algorithms.md +224 -0
  206. algo_cli/resources/skills/smart-error-recovery.md +56 -0
  207. algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
  208. algo_cli/retrieval_algorithms.py +127 -0
  209. algo_cli/runtime_qos.py +236 -0
  210. algo_cli/runtime_services.py +320 -0
  211. algo_cli/session_commands.py +95 -0
  212. algo_cli/session_mode.py +113 -0
  213. algo_cli/skills.py +430 -0
  214. algo_cli/slash_dispatch.py +1265 -0
  215. algo_cli/small_context.py +206 -0
  216. algo_cli/spawn_budget.py +89 -0
  217. algo_cli/task_ledger.py +84 -0
  218. algo_cli/task_router.py +197 -0
  219. algo_cli/tool_context.py +94 -0
  220. algo_cli/tool_contract.py +99 -0
  221. algo_cli/tool_policy.py +357 -0
  222. algo_cli/tool_runtime.py +647 -0
  223. algo_cli/tools.py +3056 -0
  224. algo_cli/url_scheme.py +174 -0
  225. algo_cli/verify.py +154 -0
  226. algo_cli/version_manifest.py +178 -0
  227. algo_cli/vision_screenshot_verify.py +76 -0
  228. algo_cli/workspace_resolver.py +68 -0
  229. algo_cli/x_account.py +209 -0
  230. algo_cli/xai_auth.py +374 -0
  231. algo_cli/xai_client.py +600 -0
  232. algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
  233. algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
  234. algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
  235. algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
  236. algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
  237. ollama_cli/__init__.py +67 -0
@@ -0,0 +1,541 @@
1
+ """Deterministic, privacy-gated durable-memory candidate processing.
2
+
3
+ Only the original user-authored text is accepted as input. The module does no
4
+ model calls, embeddings, retrieval, or inspection of assistant/tool output.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import hashlib
10
+ import json
11
+ import math
12
+ import re
13
+ import unicodedata
14
+ from collections import Counter
15
+ from collections.abc import Callable, Mapping, Sequence
16
+ from dataclasses import dataclass
17
+ from datetime import datetime, timezone
18
+ from pathlib import Path
19
+ from typing import Any
20
+
21
+ from .config import _atomic_write_text, _exclusive_state_lock
22
+
23
+ STATE_VERSION = 1
24
+ MAX_SOURCE_CHARS = 12_000
25
+ MAX_CANDIDATES_PER_TURN = 3
26
+ MAX_STORED_PER_TURN = 1
27
+ MAX_DAILY_WRITES = 5
28
+ MAX_AUTO_FINGERPRINTS = 64
29
+ MAX_MEMORY_CHARS = 12_000
30
+ MIN_WORDS = 3
31
+ MAX_WORDS = 40
32
+ MAX_CANDIDATE_CHARS = 240
33
+ NEAR_DUPLICATE_JACCARD = 0.90
34
+ NEAR_DUPLICATE_LENGTH_RATIO = 0.80
35
+
36
+ PersistenceFn = Callable[[str], bool]
37
+ TelemetryFn = Callable[[dict[str, Any]], None]
38
+
39
+ _FENCE_RE = re.compile(r"^\s*(```|~~~)")
40
+ _FORWARDED_RE = re.compile(r"^\s*-{2,}\s*(?:original|forwarded) message\s*-{2,}\s*$", re.I)
41
+ _SENTENCE_SPLIT_RE = re.compile(r"(?<=[.!?])\s+|(?<=[.!?][\"'”’])\s+|[\r\n]+")
42
+ _REMEMBER_RE = re.compile(
43
+ r"^(?:(?:also|and)\s+)?(?:please\s+)?remember(?:\s+that)?\s*[:,-]?\s+(.+)$",
44
+ re.I,
45
+ )
46
+ _GLOBAL_PREFIXES: tuple[tuple[str, re.Pattern[str]], ...] = (
47
+ ("from_now_on", re.compile(r"^from now on\s*[,;:-]?\s+(.+)$", re.I)),
48
+ ("going_forward", re.compile(r"^going forward\s*[,;:-]?\s+(.+)$", re.I)),
49
+ ("by_default", re.compile(r"^by default\s*[,;:-]?\s+(.+)$", re.I)),
50
+ )
51
+ _STANDING_RE = re.compile(
52
+ r"^(?:i|we|you)\s+(?:should\s+)?(?:always|never)\b.+$|^(?:always|never)\b.+$",
53
+ re.I,
54
+ )
55
+ _WORD_RE = re.compile(r"[\w./~+:-]+", re.UNICODE)
56
+ _INLINE_CODE_RE = re.compile(
57
+ r"`|\{\{|\}\}|=>|\(\)\s*[;{]|\b[A-Z][A-Z0-9_]{2,}\s*=|<[/!]?[A-Za-z][^>]*>"
58
+ )
59
+ _TRANSIENT_RE = re.compile(
60
+ r"\b(?:now|today|tomorrow|yesterday|tonight|this (?:week|month|year)|"
61
+ r"next (?:week|month)|right now|for now|currently|"
62
+ r"at the moment|in this (?:task|turn|session|request)|this (?:task|turn|session|request)|"
63
+ r"the current (?:task|turn|session|request|branch|commit)|temporary|temporarily|"
64
+ r"pending|in progress|next step|just failed|just finished)\b|"
65
+ r"\buntil\s+(?:today|tomorrow|tonight|next\b)|\b\d{4}-\d{2}-\d{2}\b",
66
+ re.I,
67
+ )
68
+ _TASK_RE = re.compile(
69
+ r"^to\s+\w+\b|^(?:run|fix|update|check|review|create|delete|commit|push|merge|"
70
+ r"build|test|open|read|write|install|send|call|buy|schedule|deploy|publish)\b|"
71
+ r"\b(?:todo|to-do|(?:i|we|you)\s+(?:still\s+)?need to|need to finish|"
72
+ r"must finish|finish this|complete this|remind me to)\b",
73
+ re.I,
74
+ )
75
+ _SECRET_ASSIGNMENT_RE = re.compile(
76
+ r"\b(?:password|passwd|passphrase|api[ _-]?key|client[ _-]?secret|"
77
+ r"access[ _-]?token|refresh[ _-]?token|id[ _-]?token|private[ _-]?key)\b"
78
+ r"\s*(?:=|:|\bis\b)\s*[\"']?\S+",
79
+ re.I,
80
+ )
81
+ _SECRET_TOKEN_RE = re.compile(
82
+ r"\b(?:sk-[A-Za-z0-9_-]{12,}|github_pat_[A-Za-z0-9_]{12,}|"
83
+ r"gh[pousr]_[A-Za-z0-9]{12,}|xox[baprs]-[A-Za-z0-9-]{12,}|"
84
+ r"AIza[A-Za-z0-9_-]{20,}|AKIA[A-Z0-9]{16}|ya29\.[A-Za-z0-9_-]{12,})\b"
85
+ )
86
+ _BEARER_RE = re.compile(r"\bBearer\s+[A-Za-z0-9._~+/=-]{8,}", re.I)
87
+ _JWT_RE = re.compile(r"\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b")
88
+ _PEM_RE = re.compile(r"-----BEGIN(?: [A-Z0-9]+)? PRIVATE KEY-----", re.I)
89
+ _CREDENTIALED_URL_RE = re.compile(r"[a-z][a-z0-9+.-]*://[^\s/:@]+:[^\s/@]+@", re.I)
90
+ _EMAIL_RE = re.compile(r"(?<![\w.+-])[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}(?![\w-])", re.I)
91
+ _PHONE_RE = re.compile(
92
+ r"(?<!\d)(?:\+?\d{1,3}[-.\s]?)?(?:\(?\d{3}\)?[-.\s]?)\d{3}[-.\s]?\d{4}(?!\d)"
93
+ )
94
+ _SSN_RE = re.compile(
95
+ r"(?<!\d)\d{3}-\d{2}-\d{4}(?!\d)|"
96
+ r"\b(?:ssn|social security(?: number)?)\D{0,12}\d{9}\b",
97
+ re.I,
98
+ )
99
+ _CARD_CANDIDATE_RE = re.compile(r"(?<!\d)(?:\d[ -]?){13,19}(?!\d)")
100
+ _ENTROPY_TOKEN_RE = re.compile(r"[A-Za-z0-9+/=_-]{24,}")
101
+ _DURABILITY_BOILERPLATE = frozenset(
102
+ {"always", "default", "going", "forward", "please", "prefer", "remember", "that"}
103
+ )
104
+ _NEGATIONS = frozenset({"never", "no", "not", "without"})
105
+
106
+
107
+ @dataclass(frozen=True)
108
+ class MemoryCandidate:
109
+ text: str
110
+ marker: str
111
+
112
+
113
+ @dataclass(frozen=True)
114
+ class EligibilityDecision:
115
+ eligible: bool
116
+ reason: str
117
+ fingerprint: str
118
+
119
+
120
+ def _bounded_source(text: str) -> str:
121
+ raw = str(text or "")
122
+ if len(raw) <= MAX_SOURCE_CHARS:
123
+ return raw
124
+ # Slicing and joining the head/tail could cross a removed quote/fence
125
+ # boundary and turn pasted content into an apparent user-authored marker.
126
+ # Oversized turns therefore fail closed instead of being reassembled.
127
+ return ""
128
+
129
+
130
+ def _strip_untrusted_blocks(text: str) -> str:
131
+ lines: list[str] = []
132
+ in_fence = False
133
+ fence = ""
134
+ for raw_line in _bounded_source(text).splitlines():
135
+ fence_match = _FENCE_RE.match(raw_line)
136
+ if fence_match:
137
+ marker = fence_match.group(1)
138
+ if not in_fence:
139
+ in_fence = True
140
+ fence = marker
141
+ elif marker == fence:
142
+ in_fence = False
143
+ fence = ""
144
+ continue
145
+ if in_fence:
146
+ continue
147
+ if raw_line.lstrip().startswith(">"):
148
+ continue
149
+ if _FORWARDED_RE.match(raw_line):
150
+ break
151
+ lines.append(raw_line)
152
+ return "\n".join(lines)
153
+
154
+
155
+ def _clean_candidate_text(text: str) -> str:
156
+ return " ".join(str(text or "").strip().strip("\"'").split())
157
+
158
+
159
+ def _is_wrapped_quote(text: str) -> bool:
160
+ stripped = str(text or "").strip()
161
+ return len(stripped) >= 2 and (stripped[0], stripped[-1]) in {
162
+ ('"', '"'),
163
+ ("'", "'"),
164
+ ("“", "”"),
165
+ ("‘", "’"),
166
+ }
167
+
168
+
169
+ def _extract_candidates_with_overflow(text: str) -> tuple[list[MemoryCandidate], int]:
170
+ extracted: list[MemoryCandidate] = []
171
+ seen: set[tuple[str, str]] = set()
172
+ total = 0
173
+ for segment in _SENTENCE_SPLIT_RE.split(_strip_untrusted_blocks(text)):
174
+ if _is_wrapped_quote(segment):
175
+ continue
176
+ segment = _clean_candidate_text(segment)
177
+ if not segment:
178
+ continue
179
+ marker = ""
180
+ body = ""
181
+ remember_match = _REMEMBER_RE.match(segment)
182
+ if remember_match:
183
+ marker = "remember"
184
+ body = remember_match.group(1)
185
+ else:
186
+ for candidate_marker, pattern in _GLOBAL_PREFIXES:
187
+ match = pattern.match(segment)
188
+ if match:
189
+ marker = candidate_marker
190
+ body = match.group(1)
191
+ break
192
+ if not marker and _STANDING_RE.match(segment):
193
+ marker = "standing_rule"
194
+ body = segment
195
+ if not marker:
196
+ continue
197
+ body = _clean_candidate_text(body)
198
+ if not body:
199
+ continue
200
+ key = (marker, normalize_memory_text(body))
201
+ if key in seen:
202
+ continue
203
+ seen.add(key)
204
+ total += 1
205
+ if len(extracted) < MAX_CANDIDATES_PER_TURN:
206
+ extracted.append(MemoryCandidate(text=body, marker=marker))
207
+ return extracted, max(0, total - len(extracted))
208
+
209
+
210
+ def extract_candidates(text: str) -> list[MemoryCandidate]:
211
+ """Extract at most three candidates from explicit durable-marker sentences."""
212
+
213
+ candidates, _overflow = _extract_candidates_with_overflow(text)
214
+ return candidates
215
+
216
+
217
+ def normalize_memory_text(text: str) -> str:
218
+ normalized = unicodedata.normalize("NFKC", str(text or "")).casefold()
219
+ normalized = re.sub(r"\b(?:from now on|going forward|by default|please remember(?: that)?)\b", " ", normalized)
220
+ normalized = re.sub(r"(?<=\w)[.!?,;:]+(?=\s|$)", " ", normalized)
221
+ normalized = re.sub(r"[^\w./~+:-]+", " ", normalized)
222
+ return " ".join(normalized.split())
223
+
224
+
225
+ def memory_fingerprint(text: str) -> str:
226
+ return hashlib.sha256(normalize_memory_text(text).encode("utf-8")).hexdigest()
227
+
228
+
229
+ def _dedupe_tokens(text: str) -> set[str]:
230
+ return {
231
+ token
232
+ for token in normalize_memory_text(text).split()
233
+ if token not in _DURABILITY_BOILERPLATE
234
+ }
235
+
236
+
237
+ def _near_duplicate(left: str, right: str) -> bool:
238
+ left_tokens = _dedupe_tokens(left)
239
+ right_tokens = _dedupe_tokens(right)
240
+ if not left_tokens or not right_tokens:
241
+ return False
242
+ if (left_tokens & _NEGATIONS) != (right_tokens & _NEGATIONS):
243
+ return False
244
+ length_ratio = min(len(left_tokens), len(right_tokens)) / max(len(left_tokens), len(right_tokens))
245
+ if length_ratio < NEAR_DUPLICATE_LENGTH_RATIO:
246
+ return False
247
+ union = left_tokens | right_tokens
248
+ return len(left_tokens & right_tokens) / len(union) >= NEAR_DUPLICATE_JACCARD
249
+
250
+
251
+ def _luhn_valid(number: str) -> bool:
252
+ digits = [int(char) for char in number if char.isdigit()]
253
+ if not 13 <= len(digits) <= 19 or len(set(digits)) == 1:
254
+ return False
255
+ checksum = 0
256
+ parity = len(digits) % 2
257
+ for index, digit in enumerate(digits):
258
+ value = digit
259
+ if index % 2 == parity:
260
+ value *= 2
261
+ if value > 9:
262
+ value -= 9
263
+ checksum += value
264
+ return checksum % 10 == 0
265
+
266
+
267
+ def _entropy(token: str) -> float:
268
+ counts = Counter(token)
269
+ length = len(token)
270
+ return -sum((count / length) * math.log2(count / length) for count in counts.values())
271
+
272
+
273
+ def _has_high_entropy_token(text: str) -> bool:
274
+ for token in _ENTROPY_TOKEN_RE.findall(text):
275
+ categories = sum(
276
+ (
277
+ any(char.islower() for char in token),
278
+ any(char.isupper() for char in token),
279
+ any(char.isdigit() for char in token),
280
+ any(not char.isalnum() for char in token),
281
+ )
282
+ )
283
+ if categories >= 3 and _entropy(token) >= 3.5:
284
+ return True
285
+ return False
286
+
287
+
288
+ def _privacy_reason(text: str) -> str | None:
289
+ if (
290
+ _SECRET_ASSIGNMENT_RE.search(text)
291
+ or _SECRET_TOKEN_RE.search(text)
292
+ or _BEARER_RE.search(text)
293
+ or _JWT_RE.search(text)
294
+ or _PEM_RE.search(text)
295
+ or _CREDENTIALED_URL_RE.search(text)
296
+ or _has_high_entropy_token(text)
297
+ ):
298
+ return "secret"
299
+ if _EMAIL_RE.search(text):
300
+ return "email"
301
+ if _PHONE_RE.search(text):
302
+ return "phone"
303
+ if _SSN_RE.search(text):
304
+ return "ssn"
305
+ if any(_luhn_valid(match.group(0)) for match in _CARD_CANDIDATE_RE.finditer(text)):
306
+ return "payment_card"
307
+ return None
308
+
309
+
310
+ def evaluate_candidate(
311
+ candidate: MemoryCandidate,
312
+ existing_memories: Sequence[str] = (),
313
+ accepted_fingerprints: Sequence[str] = (),
314
+ ) -> EligibilityDecision:
315
+ """Apply deterministic durability, privacy, length, and duplicate gates."""
316
+
317
+ text = _clean_candidate_text(candidate.text)
318
+ fingerprint = memory_fingerprint(text)
319
+ privacy_reason = _privacy_reason(text)
320
+ if privacy_reason:
321
+ return EligibilityDecision(False, privacy_reason, fingerprint)
322
+ if len(text) > MAX_CANDIDATE_CHARS:
323
+ return EligibilityDecision(False, "too_long", fingerprint)
324
+ word_count = len(_WORD_RE.findall(text))
325
+ if word_count < MIN_WORDS:
326
+ return EligibilityDecision(False, "too_short", fingerprint)
327
+ if word_count > MAX_WORDS:
328
+ return EligibilityDecision(False, "too_many_words", fingerprint)
329
+ if _INLINE_CODE_RE.search(text):
330
+ return EligibilityDecision(False, "code", fingerprint)
331
+ if candidate.marker == "remember" and _TASK_RE.search(text):
332
+ return EligibilityDecision(False, "task_or_imperative", fingerprint)
333
+ if _TRANSIENT_RE.search(text):
334
+ return EligibilityDecision(False, "transient", fingerprint)
335
+ if fingerprint in set(accepted_fingerprints):
336
+ return EligibilityDecision(False, "duplicate_fingerprint", fingerprint)
337
+ normalized = normalize_memory_text(text)
338
+ for existing in existing_memories:
339
+ if normalized == normalize_memory_text(existing):
340
+ return EligibilityDecision(False, "duplicate_exact", fingerprint)
341
+ if _near_duplicate(text, str(existing)):
342
+ return EligibilityDecision(False, "duplicate_near", fingerprint)
343
+ return EligibilityDecision(True, "eligible", fingerprint)
344
+
345
+
346
+ def _empty_state() -> dict[str, Any]:
347
+ return {"version": STATE_VERSION, "accepted": [], "stored_total": 0}
348
+
349
+
350
+ def _effective_limit(value: int | None, default: int) -> int:
351
+ if value is None:
352
+ return default
353
+ try:
354
+ # Config may lower a safety limit, but cannot expand bounded state.
355
+ return min(default, max(0, int(value)))
356
+ except (TypeError, ValueError, OverflowError):
357
+ return default
358
+
359
+
360
+ def _load_state(path: Path, *, entry_limit: int) -> dict[str, Any]:
361
+ if not path.exists():
362
+ return _empty_state()
363
+ payload = json.loads(path.read_text(encoding="utf-8"))
364
+ if not isinstance(payload, dict) or payload.get("version") != STATE_VERSION:
365
+ raise ValueError("unsupported memory candidate state")
366
+ accepted = payload.get("accepted")
367
+ if not isinstance(accepted, list):
368
+ raise ValueError("memory candidate accepted list is malformed")
369
+ cleaned: list[dict[str, str]] = []
370
+ bounded_entries = accepted[-entry_limit:] if entry_limit else []
371
+ for entry in bounded_entries:
372
+ if not isinstance(entry, Mapping):
373
+ continue
374
+ fingerprint = str(entry.get("fingerprint") or "")
375
+ day = str(entry.get("day") or "")
376
+ if re.fullmatch(r"[0-9a-f]{64}", fingerprint) and re.fullmatch(r"\d{4}-\d{2}-\d{2}", day):
377
+ cleaned.append({"fingerprint": fingerprint, "day": day})
378
+ return {
379
+ "version": STATE_VERSION,
380
+ "accepted": cleaned,
381
+ "stored_total": max(0, int(payload.get("stored_total") or 0)),
382
+ }
383
+
384
+
385
+ def _emit_telemetry(callback: TelemetryFn | None, result: dict[str, Any]) -> None:
386
+ if callback is None:
387
+ return
388
+ try:
389
+ callback(dict(result))
390
+ except Exception:
391
+ return
392
+
393
+
394
+ def process_memory_candidates(
395
+ original_user_text: str,
396
+ existing_memories: Sequence[str],
397
+ state_path: Path | str,
398
+ enabled: bool,
399
+ persist: PersistenceFn,
400
+ *,
401
+ now: datetime | None = None,
402
+ telemetry: TelemetryFn | None = None,
403
+ daily_limit: int | None = None,
404
+ entry_limit: int | None = None,
405
+ char_limit: int | None = None,
406
+ ) -> dict[str, Any]:
407
+ """Evaluate and persist at most one privacy-safe durable memory.
408
+
409
+ The returned payload and optional telemetry contain aggregate reason counts
410
+ only. Candidate text and rejected fingerprints are deliberately omitted.
411
+ """
412
+
413
+ effective_daily_limit = _effective_limit(daily_limit, MAX_DAILY_WRITES)
414
+ effective_entry_limit = _effective_limit(entry_limit, MAX_AUTO_FINGERPRINTS)
415
+ effective_char_limit = _effective_limit(char_limit, MAX_MEMORY_CHARS)
416
+ reason_counts: Counter[str] = Counter()
417
+ base_result: dict[str, Any] = {
418
+ "version": STATE_VERSION,
419
+ "status": "disabled" if not enabled else "no_candidates",
420
+ "reason": "automatic memory candidates are disabled" if not enabled else "no durable marker found",
421
+ "counts": {"extracted": 0, "evaluated": 0, "eligible": 0, "stored": 0, "rejected": 0},
422
+ "reason_counts": {},
423
+ "limits": {
424
+ "candidates_per_turn": MAX_CANDIDATES_PER_TURN,
425
+ "stored_per_turn": MAX_STORED_PER_TURN,
426
+ "daily_writes": effective_daily_limit,
427
+ "auto_fingerprints": effective_entry_limit,
428
+ "memory_chars": effective_char_limit,
429
+ },
430
+ "state": {"auto_fingerprints": 0, "daily_writes": 0},
431
+ }
432
+ if not enabled:
433
+ _emit_telemetry(telemetry, base_result)
434
+ return base_result
435
+
436
+ candidates, overflow = _extract_candidates_with_overflow(original_user_text)
437
+ base_result["counts"]["extracted"] = len(candidates)
438
+ if overflow:
439
+ reason_counts["candidate_limit"] += overflow
440
+ if not candidates:
441
+ reason_counts["no_durable_marker"] += 1
442
+ base_result["reason_counts"] = dict(sorted(reason_counts.items()))
443
+ _emit_telemetry(telemetry, base_result)
444
+ return base_result
445
+
446
+ path = Path(state_path)
447
+ utc_now = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
448
+ today = utc_now.date().isoformat()
449
+ current_memories = [str(memory) for memory in existing_memories]
450
+ memory_chars = sum(len(memory) for memory in current_memories)
451
+ stored_count = 0
452
+ eligible_count = 0
453
+ try:
454
+ with _exclusive_state_lock(path):
455
+ state = _load_state(path, entry_limit=effective_entry_limit)
456
+ accepted = list(state["accepted"])
457
+ accepted_fingerprints = [entry["fingerprint"] for entry in accepted]
458
+ daily_writes = sum(1 for entry in accepted if entry["day"] == today)
459
+ for candidate in candidates:
460
+ base_result["counts"]["evaluated"] += 1
461
+ decision = evaluate_candidate(
462
+ candidate,
463
+ current_memories,
464
+ accepted_fingerprints,
465
+ )
466
+ if not decision.eligible:
467
+ reason_counts[decision.reason] += 1
468
+ continue
469
+ eligible_count += 1
470
+ if stored_count >= MAX_STORED_PER_TURN:
471
+ reason_counts["turn_write_limit"] += 1
472
+ continue
473
+ if daily_writes >= effective_daily_limit:
474
+ reason_counts["daily_write_limit"] += 1
475
+ continue
476
+ if len(accepted) >= effective_entry_limit:
477
+ reason_counts["auto_fingerprint_capacity"] += 1
478
+ continue
479
+ if memory_chars + len(candidate.text) > effective_char_limit:
480
+ reason_counts["memory_char_capacity"] += 1
481
+ continue
482
+ try:
483
+ persisted = bool(persist(candidate.text))
484
+ except Exception:
485
+ reason_counts["persistence_error"] += 1
486
+ continue
487
+ if not persisted:
488
+ reason_counts["persistence_rejected"] += 1
489
+ current_memories.append(candidate.text)
490
+ continue
491
+ accepted.append({"fingerprint": decision.fingerprint, "day": today})
492
+ accepted_fingerprints.append(decision.fingerprint)
493
+ current_memories.append(candidate.text)
494
+ memory_chars += len(candidate.text)
495
+ daily_writes += 1
496
+ stored_count += 1
497
+ reason_counts["stored"] += 1
498
+ state = {
499
+ "version": STATE_VERSION,
500
+ "accepted": accepted[-effective_entry_limit:] if effective_entry_limit else [],
501
+ "stored_total": int(state.get("stored_total") or 0) + stored_count,
502
+ }
503
+ _atomic_write_text(path, json.dumps(state, indent=2, sort_keys=True))
504
+ except (OSError, TimeoutError, ValueError, json.JSONDecodeError):
505
+ reason_counts["state_error"] += 1
506
+ base_result["status"] = "error"
507
+ base_result["reason"] = "memory candidate state was unavailable"
508
+ base_result["counts"]["eligible"] = eligible_count
509
+ base_result["counts"]["stored"] = stored_count
510
+ base_result["counts"]["rejected"] = len(candidates) - stored_count
511
+ base_result["reason_counts"] = dict(sorted(reason_counts.items()))
512
+ _emit_telemetry(telemetry, base_result)
513
+ return base_result
514
+
515
+ base_result["counts"]["eligible"] = eligible_count
516
+ base_result["counts"]["stored"] = stored_count
517
+ base_result["counts"]["rejected"] = len(candidates) - stored_count
518
+ base_result["reason_counts"] = dict(sorted(reason_counts.items()))
519
+ base_result["state"] = {
520
+ "auto_fingerprints": len(accepted),
521
+ "daily_writes": daily_writes,
522
+ }
523
+ if stored_count:
524
+ base_result["status"] = "stored"
525
+ base_result["reason"] = "stored one durable memory candidate"
526
+ else:
527
+ base_result["status"] = "rejected"
528
+ base_result["reason"] = "no candidate passed every eligibility and capacity gate"
529
+ _emit_telemetry(telemetry, base_result)
530
+ return base_result
531
+
532
+
533
+ __all__ = [
534
+ "EligibilityDecision",
535
+ "MemoryCandidate",
536
+ "evaluate_candidate",
537
+ "extract_candidates",
538
+ "memory_fingerprint",
539
+ "normalize_memory_text",
540
+ "process_memory_candidates",
541
+ ]