algo-cli-runtime 0.14.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- algo_cli/__init__.py +3 -0
- algo_cli/__main__.py +7 -0
- algo_cli/_internal/__init__.py +12 -0
- algo_cli/_internal/policy_chain.py +259 -0
- algo_cli/action_registry.py +1047 -0
- algo_cli/agent_blocks.py +550 -0
- algo_cli/agent_pipeline.py +1457 -0
- algo_cli/agent_threads.py +308 -0
- algo_cli/animations.py +316 -0
- algo_cli/cache_admission.py +209 -0
- algo_cli/capability_mask.py +66 -0
- algo_cli/chat_protocol.py +116 -0
- algo_cli/chatgpt_auth.py +510 -0
- algo_cli/chatgpt_client.py +657 -0
- algo_cli/code_rag.py +479 -0
- algo_cli/config.py +651 -0
- algo_cli/context_budget.py +679 -0
- algo_cli/credential_helpers.py +315 -0
- algo_cli/deliberation.py +29 -0
- algo_cli/display.py +1470 -0
- algo_cli/evals/__init__.py +21 -0
- algo_cli/evals/algorithm_effectiveness.py +560 -0
- algo_cli/evals/competitive_harness_rating.py +702 -0
- algo_cli/evals/cot_quality.py +220 -0
- algo_cli/evals/harness_retrieval_benchmark.py +401 -0
- algo_cli/evals/performance_regression.py +136 -0
- algo_cli/evals/scorecard_grading.py +308 -0
- algo_cli/evals/session_distribution.py +84 -0
- algo_cli/execution_guardrails.py +806 -0
- algo_cli/extensions_manifest.py +84 -0
- algo_cli/git_evidence.py +227 -0
- algo_cli/google_workspace.py +407 -0
- algo_cli/google_workspace_auth.py +523 -0
- algo_cli/harness.py +2587 -0
- algo_cli/identity.py +557 -0
- algo_cli/index_compute_lab.py +228 -0
- algo_cli/inference_harness.py +70 -0
- algo_cli/intelligence/__init__.py +1103 -0
- algo_cli/intelligence/acrobat_config.py +307 -0
- algo_cli/intelligence/acrobat_manifests.py +338 -0
- algo_cli/intelligence/acrobat_models.py +195 -0
- algo_cli/intelligence/acrobat_pipeline.py +295 -0
- algo_cli/intelligence/acrobat_runtime.py +302 -0
- algo_cli/intelligence/acrobat_security.py +261 -0
- algo_cli/intelligence/acrobat_workflows.py +226 -0
- algo_cli/intelligence/actionability.py +165 -0
- algo_cli/intelligence/adversarial_audit.py +136 -0
- algo_cli/intelligence/agent_arena.py +92 -0
- algo_cli/intelligence/agent_benchmark.py +236 -0
- algo_cli/intelligence/agent_runtime.py +171 -0
- algo_cli/intelligence/agents_as_tools.py +70 -0
- algo_cli/intelligence/artifact_binding.py +80 -0
- algo_cli/intelligence/autonomous_engineer.py +1976 -0
- algo_cli/intelligence/backpressure.py +99 -0
- algo_cli/intelligence/bloom_filter.py +186 -0
- algo_cli/intelligence/bonferroni.py +66 -0
- algo_cli/intelligence/boundary_compaction.py +98 -0
- algo_cli/intelligence/catalog_verifier.py +172 -0
- algo_cli/intelligence/cavecrew.py +118 -0
- algo_cli/intelligence/changelog.py +176 -0
- algo_cli/intelligence/checkpoint_resume.py +92 -0
- algo_cli/intelligence/circuit_breaker.py +88 -0
- algo_cli/intelligence/clarification_gate.py +101 -0
- algo_cli/intelligence/code_graph.py +180 -0
- algo_cli/intelligence/coderank.py +97 -0
- algo_cli/intelligence/consistent_hash.py +150 -0
- algo_cli/intelligence/consortium_synthesis.py +139 -0
- algo_cli/intelligence/construction/__init__.py +241 -0
- algo_cli/intelligence/construction/common.py +273 -0
- algo_cli/intelligence/construction/documents.py +496 -0
- algo_cli/intelligence/construction/labor_units.py +1395 -0
- algo_cli/intelligence/construction/payments.py +470 -0
- algo_cli/intelligence/construction/risk.py +784 -0
- algo_cli/intelligence/content_extractor.py +132 -0
- algo_cli/intelligence/context_adaptive.py +102 -0
- algo_cli/intelligence/context_ops.py +95 -0
- algo_cli/intelligence/count_min.py +145 -0
- algo_cli/intelligence/cow_state.py +103 -0
- algo_cli/intelligence/critic_loop.py +119 -0
- algo_cli/intelligence/cross_source.py +113 -0
- algo_cli/intelligence/daemon_mode.py +99 -0
- algo_cli/intelligence/dag_orchestration.py +151 -0
- algo_cli/intelligence/deep_research.py +155 -0
- algo_cli/intelligence/degenerate_detector.py +78 -0
- algo_cli/intelligence/delta_report.py +92 -0
- algo_cli/intelligence/discovery_event_log.py +92 -0
- algo_cli/intelligence/document_ingest.py +298 -0
- algo_cli/intelligence/dual_layer_validate.py +151 -0
- algo_cli/intelligence/echo_fidelity.py +73 -0
- algo_cli/intelligence/ema_tuning.py +104 -0
- algo_cli/intelligence/event_log.py +92 -0
- algo_cli/intelligence/evidence_graph.py +114 -0
- algo_cli/intelligence/extension_host.py +162 -0
- algo_cli/intelligence/extension_manifest.py +115 -0
- algo_cli/intelligence/falsification_suite.py +178 -0
- algo_cli/intelligence/finance/__init__.py +169 -0
- algo_cli/intelligence/finance/anomalies.py +135 -0
- algo_cli/intelligence/finance/ap_ar.py +351 -0
- algo_cli/intelligence/finance/cash.py +162 -0
- algo_cli/intelligence/finance/close.py +332 -0
- algo_cli/intelligence/finance/common.py +244 -0
- algo_cli/intelligence/finance/construction.py +135 -0
- algo_cli/intelligence/finance/controls.py +172 -0
- algo_cli/intelligence/finance/evidence.py +119 -0
- algo_cli/intelligence/finance/exceptions.py +157 -0
- algo_cli/intelligence/finance/reconciliations.py +254 -0
- algo_cli/intelligence/finance/revenue.py +109 -0
- algo_cli/intelligence/finance/tax.py +74 -0
- algo_cli/intelligence/finance/workpapers.py +111 -0
- algo_cli/intelligence/finding_record.py +120 -0
- algo_cli/intelligence/flow_dag.py +267 -0
- algo_cli/intelligence/gatherer.py +223 -0
- algo_cli/intelligence/golden_master.py +98 -0
- algo_cli/intelligence/graph_rag.py +195 -0
- algo_cli/intelligence/group_chat.py +143 -0
- algo_cli/intelligence/hash_dedup.py +145 -0
- algo_cli/intelligence/hyperloglog.py +128 -0
- algo_cli/intelligence/incremental_index.py +316 -0
- algo_cli/intelligence/index_store.py +16 -0
- algo_cli/intelligence/iteration_plan.py +133 -0
- algo_cli/intelligence/kernel_plugins.py +167 -0
- algo_cli/intelligence/lesson_catalog.py +135 -0
- algo_cli/intelligence/llm_fallback.py +169 -0
- algo_cli/intelligence/log2_histogram.py +267 -0
- algo_cli/intelligence/lsp_integration.py +147 -0
- algo_cli/intelligence/memory_evolution.py +117 -0
- algo_cli/intelligence/minhash_lsh.py +182 -0
- algo_cli/intelligence/multi_model_score.py +174 -0
- algo_cli/intelligence/multi_tier_grade.py +211 -0
- algo_cli/intelligence/negative_controls.py +113 -0
- algo_cli/intelligence/numeric_clamp.py +63 -0
- algo_cli/intelligence/occ_editor.py +66 -0
- algo_cli/intelligence/output_normalize.py +112 -0
- algo_cli/intelligence/parallel_delegation.py +98 -0
- algo_cli/intelligence/parallel_fanout.py +104 -0
- algo_cli/intelligence/permission_modes.py +105 -0
- algo_cli/intelligence/pre_push_gate.py +68 -0
- algo_cli/intelligence/prefetch.py +171 -0
- algo_cli/intelligence/process_framework.py +217 -0
- algo_cli/intelligence/project_graph.py +387 -0
- algo_cli/intelligence/query_expansion.py +146 -0
- algo_cli/intelligence/ralph_loop.py +117 -0
- algo_cli/intelligence/rate_limiter.py +153 -0
- algo_cli/intelligence/refactor_transaction.py +94 -0
- algo_cli/intelligence/research_workspace.py +108 -0
- algo_cli/intelligence/retraction_ledger.py +72 -0
- algo_cli/intelligence/saga_pattern.py +88 -0
- algo_cli/intelligence/session_fork.py +100 -0
- algo_cli/intelligence/shadow_editor.py +67 -0
- algo_cli/intelligence/shell_session.py +213 -0
- algo_cli/intelligence/source_registry.py +143 -0
- algo_cli/intelligence/spawn_scales.py +99 -0
- algo_cli/intelligence/stat_stability.py +104 -0
- algo_cli/intelligence/structural_validator.py +148 -0
- algo_cli/intelligence/subagent_spawner.py +111 -0
- algo_cli/intelligence/symmetric_verify.py +70 -0
- algo_cli/intelligence/task_classifier.py +129 -0
- algo_cli/intelligence/team_execution.py +122 -0
- algo_cli/intelligence/tiered_access.py +121 -0
- algo_cli/intelligence/utility_registry.py +159 -0
- algo_cli/intuition_engine.py +560 -0
- algo_cli/intuition_injector.py +82 -0
- algo_cli/kernels/__init__.py +5 -0
- algo_cli/kernels/manifest.py +763 -0
- algo_cli/main.py +3903 -0
- algo_cli/memory_candidates.py +541 -0
- algo_cli/memory_echo_veil.py +394 -0
- algo_cli/memory_runtime.py +112 -0
- algo_cli/model_info.py +548 -0
- algo_cli/model_profile.py +160 -0
- algo_cli/model_routing.py +74 -0
- algo_cli/oneshot.py +331 -0
- algo_cli/perf_telemetry.py +389 -0
- algo_cli/plugins.py +245 -0
- algo_cli/private_event_store.py +654 -0
- algo_cli/quantization/__init__.py +24 -0
- algo_cli/quantization/lloyd_max.py +98 -0
- algo_cli/quantization/turbo_quant.py +308 -0
- algo_cli/reasoning/__init__.py +46 -0
- algo_cli/reasoning/combinatorial.py +356 -0
- algo_cli/reasoning/graph_of_thought.py +297 -0
- algo_cli/reasoning/mcts.py +220 -0
- algo_cli/reasoning/neuro_symbolic.py +250 -0
- algo_cli/reasoning/react.py +246 -0
- algo_cli/reasoning/reflexion.py +225 -0
- algo_cli/reasoning/tree_of_thought.py +241 -0
- algo_cli/reasoning_bridge.py +150 -0
- algo_cli/reconciliation.py +284 -0
- algo_cli/reflex.py +385 -0
- algo_cli/resources/docs/ALGO.md +13958 -0
- algo_cli/resources/docs/algo-cli-algorithm-evidence-contract.md +60 -0
- algo_cli/resources/docs/algo-cli-execution-verification-contract.md +59 -0
- algo_cli/resources/docs/algo-cli-memory-lifecycle-contract.md +72 -0
- algo_cli/resources/docs/harness-extension-cleanup-recommendation.md +41 -0
- algo_cli/resources/docs/index-compute-lab-integration.md +32 -0
- algo_cli/resources/docs/inference-harness-loop-blueprint-2026-06.md +55 -0
- algo_cli/resources/docs/main-split-map.md +35 -0
- algo_cli/resources/docs/privacy-and-context.md +48 -0
- algo_cli/resources/docs/reflex-loop-v0.2.md +354 -0
- algo_cli/resources/skills/README.md +26 -0
- algo_cli/resources/skills/algo-cli.md +59 -0
- algo_cli/resources/skills/edit-file-precision.md +49 -0
- algo_cli/resources/skills/harness-search-first.md +47 -0
- algo_cli/resources/skills/memory-recall-ritual.md +51 -0
- algo_cli/resources/skills/qol-algorithms.md +224 -0
- algo_cli/resources/skills/smart-error-recovery.md +56 -0
- algo_cli/resources/skills/tool-selection-cheatsheet.md +65 -0
- algo_cli/retrieval_algorithms.py +127 -0
- algo_cli/runtime_qos.py +236 -0
- algo_cli/runtime_services.py +320 -0
- algo_cli/session_commands.py +95 -0
- algo_cli/session_mode.py +113 -0
- algo_cli/skills.py +430 -0
- algo_cli/slash_dispatch.py +1265 -0
- algo_cli/small_context.py +206 -0
- algo_cli/spawn_budget.py +89 -0
- algo_cli/task_ledger.py +84 -0
- algo_cli/task_router.py +197 -0
- algo_cli/tool_context.py +94 -0
- algo_cli/tool_contract.py +99 -0
- algo_cli/tool_policy.py +357 -0
- algo_cli/tool_runtime.py +647 -0
- algo_cli/tools.py +3056 -0
- algo_cli/url_scheme.py +174 -0
- algo_cli/verify.py +154 -0
- algo_cli/version_manifest.py +178 -0
- algo_cli/vision_screenshot_verify.py +76 -0
- algo_cli/workspace_resolver.py +68 -0
- algo_cli/x_account.py +209 -0
- algo_cli/xai_auth.py +374 -0
- algo_cli/xai_client.py +600 -0
- algo_cli_runtime-0.14.0.dist-info/METADATA +369 -0
- algo_cli_runtime-0.14.0.dist-info/RECORD +237 -0
- algo_cli_runtime-0.14.0.dist-info/WHEEL +4 -0
- algo_cli_runtime-0.14.0.dist-info/entry_points.txt +3 -0
- algo_cli_runtime-0.14.0.dist-info/licenses/LICENSE +21 -0
- ollama_cli/__init__.py +67 -0
|
@@ -0,0 +1,541 @@
|
|
|
1
|
+
"""Deterministic, privacy-gated durable-memory candidate processing.
|
|
2
|
+
|
|
3
|
+
Only the original user-authored text is accepted as input. The module does no
|
|
4
|
+
model calls, embeddings, retrieval, or inspection of assistant/tool output.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import hashlib
|
|
10
|
+
import json
|
|
11
|
+
import math
|
|
12
|
+
import re
|
|
13
|
+
import unicodedata
|
|
14
|
+
from collections import Counter
|
|
15
|
+
from collections.abc import Callable, Mapping, Sequence
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
from datetime import datetime, timezone
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
from .config import _atomic_write_text, _exclusive_state_lock
|
|
22
|
+
|
|
23
|
+
STATE_VERSION = 1
|
|
24
|
+
MAX_SOURCE_CHARS = 12_000
|
|
25
|
+
MAX_CANDIDATES_PER_TURN = 3
|
|
26
|
+
MAX_STORED_PER_TURN = 1
|
|
27
|
+
MAX_DAILY_WRITES = 5
|
|
28
|
+
MAX_AUTO_FINGERPRINTS = 64
|
|
29
|
+
MAX_MEMORY_CHARS = 12_000
|
|
30
|
+
MIN_WORDS = 3
|
|
31
|
+
MAX_WORDS = 40
|
|
32
|
+
MAX_CANDIDATE_CHARS = 240
|
|
33
|
+
NEAR_DUPLICATE_JACCARD = 0.90
|
|
34
|
+
NEAR_DUPLICATE_LENGTH_RATIO = 0.80
|
|
35
|
+
|
|
36
|
+
PersistenceFn = Callable[[str], bool]
|
|
37
|
+
TelemetryFn = Callable[[dict[str, Any]], None]
|
|
38
|
+
|
|
39
|
+
_FENCE_RE = re.compile(r"^\s*(```|~~~)")
|
|
40
|
+
_FORWARDED_RE = re.compile(r"^\s*-{2,}\s*(?:original|forwarded) message\s*-{2,}\s*$", re.I)
|
|
41
|
+
_SENTENCE_SPLIT_RE = re.compile(r"(?<=[.!?])\s+|(?<=[.!?][\"'”’])\s+|[\r\n]+")
|
|
42
|
+
_REMEMBER_RE = re.compile(
|
|
43
|
+
r"^(?:(?:also|and)\s+)?(?:please\s+)?remember(?:\s+that)?\s*[:,-]?\s+(.+)$",
|
|
44
|
+
re.I,
|
|
45
|
+
)
|
|
46
|
+
_GLOBAL_PREFIXES: tuple[tuple[str, re.Pattern[str]], ...] = (
|
|
47
|
+
("from_now_on", re.compile(r"^from now on\s*[,;:-]?\s+(.+)$", re.I)),
|
|
48
|
+
("going_forward", re.compile(r"^going forward\s*[,;:-]?\s+(.+)$", re.I)),
|
|
49
|
+
("by_default", re.compile(r"^by default\s*[,;:-]?\s+(.+)$", re.I)),
|
|
50
|
+
)
|
|
51
|
+
_STANDING_RE = re.compile(
|
|
52
|
+
r"^(?:i|we|you)\s+(?:should\s+)?(?:always|never)\b.+$|^(?:always|never)\b.+$",
|
|
53
|
+
re.I,
|
|
54
|
+
)
|
|
55
|
+
_WORD_RE = re.compile(r"[\w./~+:-]+", re.UNICODE)
|
|
56
|
+
_INLINE_CODE_RE = re.compile(
|
|
57
|
+
r"`|\{\{|\}\}|=>|\(\)\s*[;{]|\b[A-Z][A-Z0-9_]{2,}\s*=|<[/!]?[A-Za-z][^>]*>"
|
|
58
|
+
)
|
|
59
|
+
_TRANSIENT_RE = re.compile(
|
|
60
|
+
r"\b(?:now|today|tomorrow|yesterday|tonight|this (?:week|month|year)|"
|
|
61
|
+
r"next (?:week|month)|right now|for now|currently|"
|
|
62
|
+
r"at the moment|in this (?:task|turn|session|request)|this (?:task|turn|session|request)|"
|
|
63
|
+
r"the current (?:task|turn|session|request|branch|commit)|temporary|temporarily|"
|
|
64
|
+
r"pending|in progress|next step|just failed|just finished)\b|"
|
|
65
|
+
r"\buntil\s+(?:today|tomorrow|tonight|next\b)|\b\d{4}-\d{2}-\d{2}\b",
|
|
66
|
+
re.I,
|
|
67
|
+
)
|
|
68
|
+
_TASK_RE = re.compile(
|
|
69
|
+
r"^to\s+\w+\b|^(?:run|fix|update|check|review|create|delete|commit|push|merge|"
|
|
70
|
+
r"build|test|open|read|write|install|send|call|buy|schedule|deploy|publish)\b|"
|
|
71
|
+
r"\b(?:todo|to-do|(?:i|we|you)\s+(?:still\s+)?need to|need to finish|"
|
|
72
|
+
r"must finish|finish this|complete this|remind me to)\b",
|
|
73
|
+
re.I,
|
|
74
|
+
)
|
|
75
|
+
_SECRET_ASSIGNMENT_RE = re.compile(
|
|
76
|
+
r"\b(?:password|passwd|passphrase|api[ _-]?key|client[ _-]?secret|"
|
|
77
|
+
r"access[ _-]?token|refresh[ _-]?token|id[ _-]?token|private[ _-]?key)\b"
|
|
78
|
+
r"\s*(?:=|:|\bis\b)\s*[\"']?\S+",
|
|
79
|
+
re.I,
|
|
80
|
+
)
|
|
81
|
+
_SECRET_TOKEN_RE = re.compile(
|
|
82
|
+
r"\b(?:sk-[A-Za-z0-9_-]{12,}|github_pat_[A-Za-z0-9_]{12,}|"
|
|
83
|
+
r"gh[pousr]_[A-Za-z0-9]{12,}|xox[baprs]-[A-Za-z0-9-]{12,}|"
|
|
84
|
+
r"AIza[A-Za-z0-9_-]{20,}|AKIA[A-Z0-9]{16}|ya29\.[A-Za-z0-9_-]{12,})\b"
|
|
85
|
+
)
|
|
86
|
+
_BEARER_RE = re.compile(r"\bBearer\s+[A-Za-z0-9._~+/=-]{8,}", re.I)
|
|
87
|
+
_JWT_RE = re.compile(r"\beyJ[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\.[A-Za-z0-9_-]{8,}\b")
|
|
88
|
+
_PEM_RE = re.compile(r"-----BEGIN(?: [A-Z0-9]+)? PRIVATE KEY-----", re.I)
|
|
89
|
+
_CREDENTIALED_URL_RE = re.compile(r"[a-z][a-z0-9+.-]*://[^\s/:@]+:[^\s/@]+@", re.I)
|
|
90
|
+
_EMAIL_RE = re.compile(r"(?<![\w.+-])[A-Z0-9._%+-]+@[A-Z0-9.-]+\.[A-Z]{2,}(?![\w-])", re.I)
|
|
91
|
+
_PHONE_RE = re.compile(
|
|
92
|
+
r"(?<!\d)(?:\+?\d{1,3}[-.\s]?)?(?:\(?\d{3}\)?[-.\s]?)\d{3}[-.\s]?\d{4}(?!\d)"
|
|
93
|
+
)
|
|
94
|
+
_SSN_RE = re.compile(
|
|
95
|
+
r"(?<!\d)\d{3}-\d{2}-\d{4}(?!\d)|"
|
|
96
|
+
r"\b(?:ssn|social security(?: number)?)\D{0,12}\d{9}\b",
|
|
97
|
+
re.I,
|
|
98
|
+
)
|
|
99
|
+
_CARD_CANDIDATE_RE = re.compile(r"(?<!\d)(?:\d[ -]?){13,19}(?!\d)")
|
|
100
|
+
_ENTROPY_TOKEN_RE = re.compile(r"[A-Za-z0-9+/=_-]{24,}")
|
|
101
|
+
_DURABILITY_BOILERPLATE = frozenset(
|
|
102
|
+
{"always", "default", "going", "forward", "please", "prefer", "remember", "that"}
|
|
103
|
+
)
|
|
104
|
+
_NEGATIONS = frozenset({"never", "no", "not", "without"})
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
@dataclass(frozen=True)
|
|
108
|
+
class MemoryCandidate:
|
|
109
|
+
text: str
|
|
110
|
+
marker: str
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
@dataclass(frozen=True)
|
|
114
|
+
class EligibilityDecision:
|
|
115
|
+
eligible: bool
|
|
116
|
+
reason: str
|
|
117
|
+
fingerprint: str
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _bounded_source(text: str) -> str:
|
|
121
|
+
raw = str(text or "")
|
|
122
|
+
if len(raw) <= MAX_SOURCE_CHARS:
|
|
123
|
+
return raw
|
|
124
|
+
# Slicing and joining the head/tail could cross a removed quote/fence
|
|
125
|
+
# boundary and turn pasted content into an apparent user-authored marker.
|
|
126
|
+
# Oversized turns therefore fail closed instead of being reassembled.
|
|
127
|
+
return ""
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _strip_untrusted_blocks(text: str) -> str:
|
|
131
|
+
lines: list[str] = []
|
|
132
|
+
in_fence = False
|
|
133
|
+
fence = ""
|
|
134
|
+
for raw_line in _bounded_source(text).splitlines():
|
|
135
|
+
fence_match = _FENCE_RE.match(raw_line)
|
|
136
|
+
if fence_match:
|
|
137
|
+
marker = fence_match.group(1)
|
|
138
|
+
if not in_fence:
|
|
139
|
+
in_fence = True
|
|
140
|
+
fence = marker
|
|
141
|
+
elif marker == fence:
|
|
142
|
+
in_fence = False
|
|
143
|
+
fence = ""
|
|
144
|
+
continue
|
|
145
|
+
if in_fence:
|
|
146
|
+
continue
|
|
147
|
+
if raw_line.lstrip().startswith(">"):
|
|
148
|
+
continue
|
|
149
|
+
if _FORWARDED_RE.match(raw_line):
|
|
150
|
+
break
|
|
151
|
+
lines.append(raw_line)
|
|
152
|
+
return "\n".join(lines)
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _clean_candidate_text(text: str) -> str:
|
|
156
|
+
return " ".join(str(text or "").strip().strip("\"'").split())
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _is_wrapped_quote(text: str) -> bool:
|
|
160
|
+
stripped = str(text or "").strip()
|
|
161
|
+
return len(stripped) >= 2 and (stripped[0], stripped[-1]) in {
|
|
162
|
+
('"', '"'),
|
|
163
|
+
("'", "'"),
|
|
164
|
+
("“", "”"),
|
|
165
|
+
("‘", "’"),
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _extract_candidates_with_overflow(text: str) -> tuple[list[MemoryCandidate], int]:
|
|
170
|
+
extracted: list[MemoryCandidate] = []
|
|
171
|
+
seen: set[tuple[str, str]] = set()
|
|
172
|
+
total = 0
|
|
173
|
+
for segment in _SENTENCE_SPLIT_RE.split(_strip_untrusted_blocks(text)):
|
|
174
|
+
if _is_wrapped_quote(segment):
|
|
175
|
+
continue
|
|
176
|
+
segment = _clean_candidate_text(segment)
|
|
177
|
+
if not segment:
|
|
178
|
+
continue
|
|
179
|
+
marker = ""
|
|
180
|
+
body = ""
|
|
181
|
+
remember_match = _REMEMBER_RE.match(segment)
|
|
182
|
+
if remember_match:
|
|
183
|
+
marker = "remember"
|
|
184
|
+
body = remember_match.group(1)
|
|
185
|
+
else:
|
|
186
|
+
for candidate_marker, pattern in _GLOBAL_PREFIXES:
|
|
187
|
+
match = pattern.match(segment)
|
|
188
|
+
if match:
|
|
189
|
+
marker = candidate_marker
|
|
190
|
+
body = match.group(1)
|
|
191
|
+
break
|
|
192
|
+
if not marker and _STANDING_RE.match(segment):
|
|
193
|
+
marker = "standing_rule"
|
|
194
|
+
body = segment
|
|
195
|
+
if not marker:
|
|
196
|
+
continue
|
|
197
|
+
body = _clean_candidate_text(body)
|
|
198
|
+
if not body:
|
|
199
|
+
continue
|
|
200
|
+
key = (marker, normalize_memory_text(body))
|
|
201
|
+
if key in seen:
|
|
202
|
+
continue
|
|
203
|
+
seen.add(key)
|
|
204
|
+
total += 1
|
|
205
|
+
if len(extracted) < MAX_CANDIDATES_PER_TURN:
|
|
206
|
+
extracted.append(MemoryCandidate(text=body, marker=marker))
|
|
207
|
+
return extracted, max(0, total - len(extracted))
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def extract_candidates(text: str) -> list[MemoryCandidate]:
|
|
211
|
+
"""Extract at most three candidates from explicit durable-marker sentences."""
|
|
212
|
+
|
|
213
|
+
candidates, _overflow = _extract_candidates_with_overflow(text)
|
|
214
|
+
return candidates
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def normalize_memory_text(text: str) -> str:
|
|
218
|
+
normalized = unicodedata.normalize("NFKC", str(text or "")).casefold()
|
|
219
|
+
normalized = re.sub(r"\b(?:from now on|going forward|by default|please remember(?: that)?)\b", " ", normalized)
|
|
220
|
+
normalized = re.sub(r"(?<=\w)[.!?,;:]+(?=\s|$)", " ", normalized)
|
|
221
|
+
normalized = re.sub(r"[^\w./~+:-]+", " ", normalized)
|
|
222
|
+
return " ".join(normalized.split())
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def memory_fingerprint(text: str) -> str:
|
|
226
|
+
return hashlib.sha256(normalize_memory_text(text).encode("utf-8")).hexdigest()
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _dedupe_tokens(text: str) -> set[str]:
|
|
230
|
+
return {
|
|
231
|
+
token
|
|
232
|
+
for token in normalize_memory_text(text).split()
|
|
233
|
+
if token not in _DURABILITY_BOILERPLATE
|
|
234
|
+
}
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _near_duplicate(left: str, right: str) -> bool:
|
|
238
|
+
left_tokens = _dedupe_tokens(left)
|
|
239
|
+
right_tokens = _dedupe_tokens(right)
|
|
240
|
+
if not left_tokens or not right_tokens:
|
|
241
|
+
return False
|
|
242
|
+
if (left_tokens & _NEGATIONS) != (right_tokens & _NEGATIONS):
|
|
243
|
+
return False
|
|
244
|
+
length_ratio = min(len(left_tokens), len(right_tokens)) / max(len(left_tokens), len(right_tokens))
|
|
245
|
+
if length_ratio < NEAR_DUPLICATE_LENGTH_RATIO:
|
|
246
|
+
return False
|
|
247
|
+
union = left_tokens | right_tokens
|
|
248
|
+
return len(left_tokens & right_tokens) / len(union) >= NEAR_DUPLICATE_JACCARD
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _luhn_valid(number: str) -> bool:
|
|
252
|
+
digits = [int(char) for char in number if char.isdigit()]
|
|
253
|
+
if not 13 <= len(digits) <= 19 or len(set(digits)) == 1:
|
|
254
|
+
return False
|
|
255
|
+
checksum = 0
|
|
256
|
+
parity = len(digits) % 2
|
|
257
|
+
for index, digit in enumerate(digits):
|
|
258
|
+
value = digit
|
|
259
|
+
if index % 2 == parity:
|
|
260
|
+
value *= 2
|
|
261
|
+
if value > 9:
|
|
262
|
+
value -= 9
|
|
263
|
+
checksum += value
|
|
264
|
+
return checksum % 10 == 0
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
def _entropy(token: str) -> float:
|
|
268
|
+
counts = Counter(token)
|
|
269
|
+
length = len(token)
|
|
270
|
+
return -sum((count / length) * math.log2(count / length) for count in counts.values())
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _has_high_entropy_token(text: str) -> bool:
|
|
274
|
+
for token in _ENTROPY_TOKEN_RE.findall(text):
|
|
275
|
+
categories = sum(
|
|
276
|
+
(
|
|
277
|
+
any(char.islower() for char in token),
|
|
278
|
+
any(char.isupper() for char in token),
|
|
279
|
+
any(char.isdigit() for char in token),
|
|
280
|
+
any(not char.isalnum() for char in token),
|
|
281
|
+
)
|
|
282
|
+
)
|
|
283
|
+
if categories >= 3 and _entropy(token) >= 3.5:
|
|
284
|
+
return True
|
|
285
|
+
return False
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
def _privacy_reason(text: str) -> str | None:
|
|
289
|
+
if (
|
|
290
|
+
_SECRET_ASSIGNMENT_RE.search(text)
|
|
291
|
+
or _SECRET_TOKEN_RE.search(text)
|
|
292
|
+
or _BEARER_RE.search(text)
|
|
293
|
+
or _JWT_RE.search(text)
|
|
294
|
+
or _PEM_RE.search(text)
|
|
295
|
+
or _CREDENTIALED_URL_RE.search(text)
|
|
296
|
+
or _has_high_entropy_token(text)
|
|
297
|
+
):
|
|
298
|
+
return "secret"
|
|
299
|
+
if _EMAIL_RE.search(text):
|
|
300
|
+
return "email"
|
|
301
|
+
if _PHONE_RE.search(text):
|
|
302
|
+
return "phone"
|
|
303
|
+
if _SSN_RE.search(text):
|
|
304
|
+
return "ssn"
|
|
305
|
+
if any(_luhn_valid(match.group(0)) for match in _CARD_CANDIDATE_RE.finditer(text)):
|
|
306
|
+
return "payment_card"
|
|
307
|
+
return None
|
|
308
|
+
|
|
309
|
+
|
|
310
|
+
def evaluate_candidate(
|
|
311
|
+
candidate: MemoryCandidate,
|
|
312
|
+
existing_memories: Sequence[str] = (),
|
|
313
|
+
accepted_fingerprints: Sequence[str] = (),
|
|
314
|
+
) -> EligibilityDecision:
|
|
315
|
+
"""Apply deterministic durability, privacy, length, and duplicate gates."""
|
|
316
|
+
|
|
317
|
+
text = _clean_candidate_text(candidate.text)
|
|
318
|
+
fingerprint = memory_fingerprint(text)
|
|
319
|
+
privacy_reason = _privacy_reason(text)
|
|
320
|
+
if privacy_reason:
|
|
321
|
+
return EligibilityDecision(False, privacy_reason, fingerprint)
|
|
322
|
+
if len(text) > MAX_CANDIDATE_CHARS:
|
|
323
|
+
return EligibilityDecision(False, "too_long", fingerprint)
|
|
324
|
+
word_count = len(_WORD_RE.findall(text))
|
|
325
|
+
if word_count < MIN_WORDS:
|
|
326
|
+
return EligibilityDecision(False, "too_short", fingerprint)
|
|
327
|
+
if word_count > MAX_WORDS:
|
|
328
|
+
return EligibilityDecision(False, "too_many_words", fingerprint)
|
|
329
|
+
if _INLINE_CODE_RE.search(text):
|
|
330
|
+
return EligibilityDecision(False, "code", fingerprint)
|
|
331
|
+
if candidate.marker == "remember" and _TASK_RE.search(text):
|
|
332
|
+
return EligibilityDecision(False, "task_or_imperative", fingerprint)
|
|
333
|
+
if _TRANSIENT_RE.search(text):
|
|
334
|
+
return EligibilityDecision(False, "transient", fingerprint)
|
|
335
|
+
if fingerprint in set(accepted_fingerprints):
|
|
336
|
+
return EligibilityDecision(False, "duplicate_fingerprint", fingerprint)
|
|
337
|
+
normalized = normalize_memory_text(text)
|
|
338
|
+
for existing in existing_memories:
|
|
339
|
+
if normalized == normalize_memory_text(existing):
|
|
340
|
+
return EligibilityDecision(False, "duplicate_exact", fingerprint)
|
|
341
|
+
if _near_duplicate(text, str(existing)):
|
|
342
|
+
return EligibilityDecision(False, "duplicate_near", fingerprint)
|
|
343
|
+
return EligibilityDecision(True, "eligible", fingerprint)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _empty_state() -> dict[str, Any]:
|
|
347
|
+
return {"version": STATE_VERSION, "accepted": [], "stored_total": 0}
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _effective_limit(value: int | None, default: int) -> int:
|
|
351
|
+
if value is None:
|
|
352
|
+
return default
|
|
353
|
+
try:
|
|
354
|
+
# Config may lower a safety limit, but cannot expand bounded state.
|
|
355
|
+
return min(default, max(0, int(value)))
|
|
356
|
+
except (TypeError, ValueError, OverflowError):
|
|
357
|
+
return default
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
def _load_state(path: Path, *, entry_limit: int) -> dict[str, Any]:
|
|
361
|
+
if not path.exists():
|
|
362
|
+
return _empty_state()
|
|
363
|
+
payload = json.loads(path.read_text(encoding="utf-8"))
|
|
364
|
+
if not isinstance(payload, dict) or payload.get("version") != STATE_VERSION:
|
|
365
|
+
raise ValueError("unsupported memory candidate state")
|
|
366
|
+
accepted = payload.get("accepted")
|
|
367
|
+
if not isinstance(accepted, list):
|
|
368
|
+
raise ValueError("memory candidate accepted list is malformed")
|
|
369
|
+
cleaned: list[dict[str, str]] = []
|
|
370
|
+
bounded_entries = accepted[-entry_limit:] if entry_limit else []
|
|
371
|
+
for entry in bounded_entries:
|
|
372
|
+
if not isinstance(entry, Mapping):
|
|
373
|
+
continue
|
|
374
|
+
fingerprint = str(entry.get("fingerprint") or "")
|
|
375
|
+
day = str(entry.get("day") or "")
|
|
376
|
+
if re.fullmatch(r"[0-9a-f]{64}", fingerprint) and re.fullmatch(r"\d{4}-\d{2}-\d{2}", day):
|
|
377
|
+
cleaned.append({"fingerprint": fingerprint, "day": day})
|
|
378
|
+
return {
|
|
379
|
+
"version": STATE_VERSION,
|
|
380
|
+
"accepted": cleaned,
|
|
381
|
+
"stored_total": max(0, int(payload.get("stored_total") or 0)),
|
|
382
|
+
}
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _emit_telemetry(callback: TelemetryFn | None, result: dict[str, Any]) -> None:
|
|
386
|
+
if callback is None:
|
|
387
|
+
return
|
|
388
|
+
try:
|
|
389
|
+
callback(dict(result))
|
|
390
|
+
except Exception:
|
|
391
|
+
return
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def process_memory_candidates(
|
|
395
|
+
original_user_text: str,
|
|
396
|
+
existing_memories: Sequence[str],
|
|
397
|
+
state_path: Path | str,
|
|
398
|
+
enabled: bool,
|
|
399
|
+
persist: PersistenceFn,
|
|
400
|
+
*,
|
|
401
|
+
now: datetime | None = None,
|
|
402
|
+
telemetry: TelemetryFn | None = None,
|
|
403
|
+
daily_limit: int | None = None,
|
|
404
|
+
entry_limit: int | None = None,
|
|
405
|
+
char_limit: int | None = None,
|
|
406
|
+
) -> dict[str, Any]:
|
|
407
|
+
"""Evaluate and persist at most one privacy-safe durable memory.
|
|
408
|
+
|
|
409
|
+
The returned payload and optional telemetry contain aggregate reason counts
|
|
410
|
+
only. Candidate text and rejected fingerprints are deliberately omitted.
|
|
411
|
+
"""
|
|
412
|
+
|
|
413
|
+
effective_daily_limit = _effective_limit(daily_limit, MAX_DAILY_WRITES)
|
|
414
|
+
effective_entry_limit = _effective_limit(entry_limit, MAX_AUTO_FINGERPRINTS)
|
|
415
|
+
effective_char_limit = _effective_limit(char_limit, MAX_MEMORY_CHARS)
|
|
416
|
+
reason_counts: Counter[str] = Counter()
|
|
417
|
+
base_result: dict[str, Any] = {
|
|
418
|
+
"version": STATE_VERSION,
|
|
419
|
+
"status": "disabled" if not enabled else "no_candidates",
|
|
420
|
+
"reason": "automatic memory candidates are disabled" if not enabled else "no durable marker found",
|
|
421
|
+
"counts": {"extracted": 0, "evaluated": 0, "eligible": 0, "stored": 0, "rejected": 0},
|
|
422
|
+
"reason_counts": {},
|
|
423
|
+
"limits": {
|
|
424
|
+
"candidates_per_turn": MAX_CANDIDATES_PER_TURN,
|
|
425
|
+
"stored_per_turn": MAX_STORED_PER_TURN,
|
|
426
|
+
"daily_writes": effective_daily_limit,
|
|
427
|
+
"auto_fingerprints": effective_entry_limit,
|
|
428
|
+
"memory_chars": effective_char_limit,
|
|
429
|
+
},
|
|
430
|
+
"state": {"auto_fingerprints": 0, "daily_writes": 0},
|
|
431
|
+
}
|
|
432
|
+
if not enabled:
|
|
433
|
+
_emit_telemetry(telemetry, base_result)
|
|
434
|
+
return base_result
|
|
435
|
+
|
|
436
|
+
candidates, overflow = _extract_candidates_with_overflow(original_user_text)
|
|
437
|
+
base_result["counts"]["extracted"] = len(candidates)
|
|
438
|
+
if overflow:
|
|
439
|
+
reason_counts["candidate_limit"] += overflow
|
|
440
|
+
if not candidates:
|
|
441
|
+
reason_counts["no_durable_marker"] += 1
|
|
442
|
+
base_result["reason_counts"] = dict(sorted(reason_counts.items()))
|
|
443
|
+
_emit_telemetry(telemetry, base_result)
|
|
444
|
+
return base_result
|
|
445
|
+
|
|
446
|
+
path = Path(state_path)
|
|
447
|
+
utc_now = (now or datetime.now(timezone.utc)).astimezone(timezone.utc)
|
|
448
|
+
today = utc_now.date().isoformat()
|
|
449
|
+
current_memories = [str(memory) for memory in existing_memories]
|
|
450
|
+
memory_chars = sum(len(memory) for memory in current_memories)
|
|
451
|
+
stored_count = 0
|
|
452
|
+
eligible_count = 0
|
|
453
|
+
try:
|
|
454
|
+
with _exclusive_state_lock(path):
|
|
455
|
+
state = _load_state(path, entry_limit=effective_entry_limit)
|
|
456
|
+
accepted = list(state["accepted"])
|
|
457
|
+
accepted_fingerprints = [entry["fingerprint"] for entry in accepted]
|
|
458
|
+
daily_writes = sum(1 for entry in accepted if entry["day"] == today)
|
|
459
|
+
for candidate in candidates:
|
|
460
|
+
base_result["counts"]["evaluated"] += 1
|
|
461
|
+
decision = evaluate_candidate(
|
|
462
|
+
candidate,
|
|
463
|
+
current_memories,
|
|
464
|
+
accepted_fingerprints,
|
|
465
|
+
)
|
|
466
|
+
if not decision.eligible:
|
|
467
|
+
reason_counts[decision.reason] += 1
|
|
468
|
+
continue
|
|
469
|
+
eligible_count += 1
|
|
470
|
+
if stored_count >= MAX_STORED_PER_TURN:
|
|
471
|
+
reason_counts["turn_write_limit"] += 1
|
|
472
|
+
continue
|
|
473
|
+
if daily_writes >= effective_daily_limit:
|
|
474
|
+
reason_counts["daily_write_limit"] += 1
|
|
475
|
+
continue
|
|
476
|
+
if len(accepted) >= effective_entry_limit:
|
|
477
|
+
reason_counts["auto_fingerprint_capacity"] += 1
|
|
478
|
+
continue
|
|
479
|
+
if memory_chars + len(candidate.text) > effective_char_limit:
|
|
480
|
+
reason_counts["memory_char_capacity"] += 1
|
|
481
|
+
continue
|
|
482
|
+
try:
|
|
483
|
+
persisted = bool(persist(candidate.text))
|
|
484
|
+
except Exception:
|
|
485
|
+
reason_counts["persistence_error"] += 1
|
|
486
|
+
continue
|
|
487
|
+
if not persisted:
|
|
488
|
+
reason_counts["persistence_rejected"] += 1
|
|
489
|
+
current_memories.append(candidate.text)
|
|
490
|
+
continue
|
|
491
|
+
accepted.append({"fingerprint": decision.fingerprint, "day": today})
|
|
492
|
+
accepted_fingerprints.append(decision.fingerprint)
|
|
493
|
+
current_memories.append(candidate.text)
|
|
494
|
+
memory_chars += len(candidate.text)
|
|
495
|
+
daily_writes += 1
|
|
496
|
+
stored_count += 1
|
|
497
|
+
reason_counts["stored"] += 1
|
|
498
|
+
state = {
|
|
499
|
+
"version": STATE_VERSION,
|
|
500
|
+
"accepted": accepted[-effective_entry_limit:] if effective_entry_limit else [],
|
|
501
|
+
"stored_total": int(state.get("stored_total") or 0) + stored_count,
|
|
502
|
+
}
|
|
503
|
+
_atomic_write_text(path, json.dumps(state, indent=2, sort_keys=True))
|
|
504
|
+
except (OSError, TimeoutError, ValueError, json.JSONDecodeError):
|
|
505
|
+
reason_counts["state_error"] += 1
|
|
506
|
+
base_result["status"] = "error"
|
|
507
|
+
base_result["reason"] = "memory candidate state was unavailable"
|
|
508
|
+
base_result["counts"]["eligible"] = eligible_count
|
|
509
|
+
base_result["counts"]["stored"] = stored_count
|
|
510
|
+
base_result["counts"]["rejected"] = len(candidates) - stored_count
|
|
511
|
+
base_result["reason_counts"] = dict(sorted(reason_counts.items()))
|
|
512
|
+
_emit_telemetry(telemetry, base_result)
|
|
513
|
+
return base_result
|
|
514
|
+
|
|
515
|
+
base_result["counts"]["eligible"] = eligible_count
|
|
516
|
+
base_result["counts"]["stored"] = stored_count
|
|
517
|
+
base_result["counts"]["rejected"] = len(candidates) - stored_count
|
|
518
|
+
base_result["reason_counts"] = dict(sorted(reason_counts.items()))
|
|
519
|
+
base_result["state"] = {
|
|
520
|
+
"auto_fingerprints": len(accepted),
|
|
521
|
+
"daily_writes": daily_writes,
|
|
522
|
+
}
|
|
523
|
+
if stored_count:
|
|
524
|
+
base_result["status"] = "stored"
|
|
525
|
+
base_result["reason"] = "stored one durable memory candidate"
|
|
526
|
+
else:
|
|
527
|
+
base_result["status"] = "rejected"
|
|
528
|
+
base_result["reason"] = "no candidate passed every eligibility and capacity gate"
|
|
529
|
+
_emit_telemetry(telemetry, base_result)
|
|
530
|
+
return base_result
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
__all__ = [
|
|
534
|
+
"EligibilityDecision",
|
|
535
|
+
"MemoryCandidate",
|
|
536
|
+
"evaluate_candidate",
|
|
537
|
+
"extract_candidates",
|
|
538
|
+
"memory_fingerprint",
|
|
539
|
+
"normalize_memory_text",
|
|
540
|
+
"process_memory_candidates",
|
|
541
|
+
]
|