memleaf 0.2.55__tar.gz → 0.2.56__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memleaf-0.2.55 → memleaf-0.2.56}/CHANGELOG.md +8 -0
- {memleaf-0.2.55/src/memleaf.egg-info → memleaf-0.2.56}/PKG-INFO +3 -3
- {memleaf-0.2.55 → memleaf-0.2.56}/README.en.md +2 -2
- {memleaf-0.2.55 → memleaf-0.2.56}/README.md +2 -2
- {memleaf-0.2.55 → memleaf-0.2.56}/pyproject.toml +1 -1
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/__init__.py +1 -1
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/admission.py +35 -6
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/evidence_structure.py +149 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/evidence_syntax.py +43 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_provider/plugin.yaml +1 -1
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/model_execution.py +3 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/process_common.py +41 -137
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/process_jobs.py +11 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/single_pass_memory_planner.py +59 -41
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/single_pass_plan.py +47 -5
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/turn_audit.py +40 -12
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/validation.py +93 -0
- {memleaf-0.2.55 → memleaf-0.2.56/src/memleaf.egg-info}/PKG-INFO +3 -3
- {memleaf-0.2.55 → memleaf-0.2.56}/LICENSE +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/MANIFEST.in +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/capture-budget-design.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/config-migrations.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/core-refactor.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/evidence-retention.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/extraction-latency.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/gate-evidence-boundary.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/general-processing.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/hermes-mcp-runtime.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/processing-quality-acceptance.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/docs/v0.2.26-processing-status.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/examples/README.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/examples/basic_usage.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/examples/mcp_stdio.ndjson +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/install.ps1 +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/install.sh +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/setup.cfg +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/__main__.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/adapters/__init__.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/adapters/antigravity.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/adapters/base.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/adapters/codex.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/adapters/hermes.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/batch_review.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/budget.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/capture.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/cli.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/compaction.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/config.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/create_coordinator.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/credentials.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/evidence_budget.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/evidence_policy.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/extraction_budget.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/extraction_capability.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/extraction_work_state.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/frontmatter.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_provider/README.md +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_provider/__init__.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_provider/_mcp_client.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_provider/_provider.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_provider/_shared.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_provider/evidence_budget.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/hermes_runtime.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/host_events.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/host_runtime.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/inbox.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/index.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/inspection.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/installer.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/llm/__init__.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/llm/base.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/llm/claude_compatible.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/llm/gemini.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/llm/openai_compatible.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/llm/router.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/llm/thinking.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/locking.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/mcp_server.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/memory_commit.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/memory_planner.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/memory_writer.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/model_capabilities.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/model_discovery.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/models.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/native_index.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/native_registration.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/parallel_model.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/planning_context.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/process_journal.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/process_owner.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/processing.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/prompts.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/provenance.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/recording_policy.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/redaction.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/retention.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/retrieval.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/retrieval_gate.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/scope_maintenance.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/scope_state.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/service.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/source_policy.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/state_layout.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/subprocess_flags.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/summary_batch.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/target_reconciliation.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/turn_plan.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/update_coordinator.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/update_review.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf/vault.py +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf.egg-info/SOURCES.txt +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf.egg-info/dependency_links.txt +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf.egg-info/entry_points.txt +0 -0
- {memleaf-0.2.55 → memleaf-0.2.56}/src/memleaf.egg-info/top_level.txt +0 -0
|
@@ -2,6 +2,14 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to memleaf are documented here.
|
|
4
4
|
|
|
5
|
+
## 0.2.56 — 2026-09-14
|
|
6
|
+
|
|
7
|
+
- Keep assistant reports source-safe at candidate granularity. Markdown headings, paragraphs and list items become exact immutable evidence units with section context; broad whole-report `whole_unit` claims are deferred, while short unstructured replies remain intact. An assistant-only offer, question or forward commitment cannot establish user intent, but a factual report can still be admitted when its evidence and future value are independently grounded.
|
|
8
|
+
- Ground project Scope from the candidate's own admitted claims, including their exact quote and section context. An explicit project label makes a `global` answer unsafe, and an ungrounded project Scope defers only that candidate instead of silently dropping the Scope or losing the source-backed memory. Candidate-level validation details are persisted through the bounded audit path.
|
|
9
|
+
- Share one boundary-safe calendar-token grammar across summary grounding and todo deadline checks. Standalone ISO, Chinese and numeric dates reject identifiers, decimals and invalid values; scheduling cues such as planned, target and scheduled now authorize deadline dates, while an ambiguous candidate-local deadline is reported and never guessed.
|
|
10
|
+
|
|
11
|
+
Verification for this release used 18 local synthetic regression tests with no model call and no production Vault write. Python 3.11 compilation/imports and `git diff --check` pass, and the release metadata is synchronized across the package, Hermes provider, READMEs and changelog. Real-model adherence and a newly installed Hermes replay remain post-release runtime acceptance.
|
|
12
|
+
|
|
5
13
|
## 0.2.55 — 2026-09-14
|
|
6
14
|
|
|
7
15
|
- Classify memory candidates by future reuse with a shorter B3 instruction: point-in-time counts and snapshots without a trend, threshold, obligation, decision or later comparison are `no_future_value`, while future-use facts remain eligible. Project ownership now has a conservative Core guard: an explicit single-project label cannot silently persist as `global`, but ordinary platform, product and vendor mentions still do not establish ownership. Self-contained memories refer to the conversation person as the user rather than as an owner. The B3 system prompt shrinks from 6395 to 5686 UTF-8 bytes.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memleaf
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.56
|
|
4
4
|
Summary: A local-first Markdown memory core for AI agents
|
|
5
5
|
Author: memleaf contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -23,8 +23,8 @@ Dynamic: license-file
|
|
|
23
23
|
|
|
24
24
|
[English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
25
25
|
|
|
26
|
-
> **版本:0.2.
|
|
27
|
-
>
|
|
26
|
+
> **版本:0.2.56。**
|
|
27
|
+
> 自动提取按助手回复中的 Markdown 结构建立候选级来源,过宽的整段引用和未被用户接受的助手意图不会形成记忆;项目 Scope 只由候选自身证据确认,无法 grounding 的候选会延后而不会被改写为 `global`。日期和截止日期使用统一的边界安全解析,todo 只在候选自己的证据给出唯一明确截止日时填充。Markdown 仍是唯一事实源。
|
|
28
28
|
> **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
|
|
29
29
|
|
|
30
30
|
## 项目定位
|
|
@@ -4,8 +4,8 @@
|
|
|
4
4
|
|
|
5
5
|
[中文](README.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
6
6
|
|
|
7
|
-
> **Version: 0.2.
|
|
8
|
-
> Automatic extraction now
|
|
7
|
+
> **Version: 0.2.56.**
|
|
8
|
+
> Automatic extraction now creates candidate-local source units from Markdown structure in assistant replies; broad whole-report citations and assistant-only intent are not allowed to create memories. Project Scope is grounded only from each candidate's own evidence, and an ungrounded candidate is deferred instead of being rewritten as `global`. Dates and deadlines use one boundary-safe parser, and a todo receives a due date only when its own evidence supplies one unambiguous deadline. Markdown remains the sole source of truth with no SQLite runtime dependency. Acceptance covers deterministic regression suites and synthetic inputs; it does not claim real-mail or customer-business acceptance.
|
|
9
9
|
> **The current release supports Hermes and Codex.** Antigravity is not detected, installed, or configured.
|
|
10
10
|
|
|
11
11
|
## Project scope
|
|
@@ -4,8 +4,8 @@
|
|
|
4
4
|
|
|
5
5
|
[English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
6
6
|
|
|
7
|
-
> **版本:0.2.
|
|
8
|
-
>
|
|
7
|
+
> **版本:0.2.56。**
|
|
8
|
+
> 自动提取按助手回复中的 Markdown 结构建立候选级来源,过宽的整段引用和未被用户接受的助手意图不会形成记忆;项目 Scope 只由候选自身证据确认,无法 grounding 的候选会延后而不会被改写为 `global`。日期和截止日期使用统一的边界安全解析,todo 只在候选自己的证据给出唯一明确截止日时填充。Markdown 仍是唯一事实源。
|
|
9
9
|
> **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
|
|
10
10
|
|
|
11
11
|
## 项目定位
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"""Local-first Markdown memory core for AI agents."""
|
|
2
2
|
|
|
3
|
-
__version__ = "0.2.
|
|
3
|
+
__version__ = "0.2.56"
|
|
4
4
|
|
|
5
5
|
from .config import DEFAULT_CONFIG, default_config, load_config, save_config
|
|
6
6
|
from .frontmatter import FrontmatterError, dump_frontmatter, dump_yaml, load_yaml, parse_frontmatter
|
|
@@ -16,11 +16,13 @@ from typing import Any, Iterable, Mapping
|
|
|
16
16
|
from .validation import ModelOutputError, parse_strict_json
|
|
17
17
|
from .evidence_syntax import (
|
|
18
18
|
_BULLET, _CLOSED_TASK, _EXAMPLE, _EXTERNAL_OWNER, _HEADING, _NEGATIVE_TASK,
|
|
19
|
-
_POLITE, _QUERY_START, _QUERY_WORD, _READ_ONLY_CONTROL,
|
|
19
|
+
_POLITE, _QUERY_START, _QUERY_WORD, _READ_ONLY_CONTROL, _assistant_intent_only,
|
|
20
|
+
_clauses, _query,
|
|
20
21
|
)
|
|
21
22
|
from .evidence_structure import (
|
|
22
23
|
MAX_EXTERNAL_UNIT_BYTES, _EXTERNAL_MARKER, _external_blocks,
|
|
23
|
-
_has_external_structure,
|
|
24
|
+
_has_external_structure, _has_markdown_structure, _markdown_blocks,
|
|
25
|
+
_structured_external_blocks, _whole_unit_is_safe,
|
|
24
26
|
)
|
|
25
27
|
|
|
26
28
|
|
|
@@ -173,9 +175,14 @@ def analyze_turn_evidence(events: Iterable[Mapping[str, Any]]) -> tuple[Evidence
|
|
|
173
175
|
def inventory(key: str, role: str, text: str, meta: Mapping[str, Any] | None = None) -> None:
|
|
174
176
|
meta = meta or {}
|
|
175
177
|
if role == "assistant":
|
|
176
|
-
#
|
|
177
|
-
#
|
|
178
|
-
|
|
178
|
+
# Keep an unstructured short reply intact, but account for explicit
|
|
179
|
+
# Markdown blocks independently so one long report cannot authorize
|
|
180
|
+
# unrelated sibling facts through a whole-unit claim.
|
|
181
|
+
fragments = (
|
|
182
|
+
_markdown_blocks(text)
|
|
183
|
+
if text.strip() and _has_markdown_structure(text)
|
|
184
|
+
else ([(0, len(text), text, "plain", ())] if text.strip() else [])
|
|
185
|
+
)
|
|
179
186
|
elif role == "external":
|
|
180
187
|
# A tool result is one physical source record. Splitting it on
|
|
181
188
|
# punctuation made JSON/document bodies look like thousands of
|
|
@@ -251,7 +258,9 @@ def _canonical_text(text: str) -> str:
|
|
|
251
258
|
|
|
252
259
|
|
|
253
260
|
def validate_bindings(value: Any, units: Iterable[EvidenceUnit],
|
|
254
|
-
candidates: Iterable[Mapping[str, Any]]
|
|
261
|
+
candidates: Iterable[Mapping[str, Any]], *,
|
|
262
|
+
allow_broad_whole_unit_candidates: Iterable[str] = (),
|
|
263
|
+
) -> dict[str, list[dict[str, Any]]]:
|
|
255
264
|
"""Validate model judgments against exact immutable source fragments.
|
|
256
265
|
|
|
257
266
|
Matching a quotation proves provenance, not the truth of a proposition.
|
|
@@ -260,6 +269,11 @@ def validate_bindings(value: Any, units: Iterable[EvidenceUnit],
|
|
|
260
269
|
"""
|
|
261
270
|
by_unit = {u.unit_id: u for u in units}
|
|
262
271
|
by_candidate = {c["candidate_id"]: c for c in candidates}
|
|
272
|
+
allowed_broad_candidates = {
|
|
273
|
+
value.casefold()
|
|
274
|
+
for value in allow_broad_whole_unit_candidates
|
|
275
|
+
if isinstance(value, str) and value
|
|
276
|
+
}
|
|
263
277
|
if not isinstance(value, list):
|
|
264
278
|
raise ModelOutputError("evidence_bindings must be a list", validation_detail="invalid_evidence",
|
|
265
279
|
evidence_check="binding_shape")
|
|
@@ -299,6 +313,16 @@ def validate_bindings(value: Any, units: Iterable[EvidenceUnit],
|
|
|
299
313
|
if claim["whole_unit"] is not True:
|
|
300
314
|
raise ModelOutputError("whole_unit must be true", validation_detail="invalid_evidence",
|
|
301
315
|
evidence_check="binding_shape")
|
|
316
|
+
if (
|
|
317
|
+
cid.casefold() not in allowed_broad_candidates
|
|
318
|
+
and unit.source_role == "assistant"
|
|
319
|
+
and not _whole_unit_is_safe(unit.text)
|
|
320
|
+
):
|
|
321
|
+
raise ModelOutputError(
|
|
322
|
+
"whole_unit is too broad for a structured assistant report",
|
|
323
|
+
validation_detail="whole_unit_too_broad",
|
|
324
|
+
evidence_check="invalid_span",
|
|
325
|
+
)
|
|
302
326
|
# Explicitly selecting one supplied immutable source unit is
|
|
303
327
|
# equivalent to quoting that whole unit. Never repair a bad
|
|
304
328
|
# quote or resolve an ID outside this invocation's inventory.
|
|
@@ -422,6 +446,11 @@ def admission_reason(candidate: Mapping[str, Any], units: Iterable[EvidenceUnit]
|
|
|
422
446
|
support = supporting_units(candidate, units)
|
|
423
447
|
if not support:
|
|
424
448
|
return "evidence_not_supported", ()
|
|
449
|
+
if candidate.get("_evidence_bindings") and all(
|
|
450
|
+
unit.source_role == "assistant" and _assistant_intent_only(unit.text)
|
|
451
|
+
for unit in support
|
|
452
|
+
):
|
|
453
|
+
return "assistant_restatement", support
|
|
425
454
|
if candidate.get("type") == "todo":
|
|
426
455
|
# Negative or third-party facts may still be retained as facts or used
|
|
427
456
|
# for a verified state update. They must not become a new active task.
|
|
@@ -6,6 +6,9 @@ import re
|
|
|
6
6
|
from typing import Iterable
|
|
7
7
|
|
|
8
8
|
MAX_EXTERNAL_UNIT_BYTES = 32 * 1024
|
|
9
|
+
# Keep whole-unit admission bounded even when a source has no explicit Markdown
|
|
10
|
+
# marker. The threshold is a structural safety limit, not a topic heuristic.
|
|
11
|
+
MAX_WHOLE_UNIT_CHARS = 512
|
|
9
12
|
|
|
10
13
|
def _external_blocks(text: str) -> Iterable[tuple[int, int, str, str, tuple[str, ...]]]:
|
|
11
14
|
"""Yield deterministic, exact source blocks for one external record.
|
|
@@ -88,6 +91,152 @@ def _external_blocks(text: str) -> Iterable[tuple[int, int, str, str, tuple[str,
|
|
|
88
91
|
start = end
|
|
89
92
|
|
|
90
93
|
_EXTERNAL_MARKER = re.compile(r"^\s*(?:#{1,6}\s+|[-*+•]\s+|\d+[.)、]\s+)")
|
|
94
|
+
_MARKDOWN_HEADING = re.compile(r"^\s{0,3}(?P<marks>#{1,6})\s+(?P<label>.+?)\s*$")
|
|
95
|
+
_MARKDOWN_ITEM = re.compile(r"^\s*(?:[-*+•]|\d+[.)、])\s+")
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _has_markdown_structure(text: str) -> bool:
|
|
99
|
+
"""Return whether a conversation reply has explicit block structure."""
|
|
100
|
+
|
|
101
|
+
if "\n\n" in text or "\r\n\r\n" in text:
|
|
102
|
+
return True
|
|
103
|
+
return any(
|
|
104
|
+
_MARKDOWN_HEADING.match(line) is not None
|
|
105
|
+
or _MARKDOWN_ITEM.match(line) is not None
|
|
106
|
+
for line in text.splitlines()
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _whole_unit_is_safe(text: str) -> bool:
|
|
111
|
+
"""Return whether selecting one assistant unit is structurally narrow."""
|
|
112
|
+
|
|
113
|
+
if not isinstance(text, str) or not text.strip():
|
|
114
|
+
return False
|
|
115
|
+
if len(text) > MAX_WHOLE_UNIT_CHARS:
|
|
116
|
+
return False
|
|
117
|
+
if not _has_markdown_structure(text):
|
|
118
|
+
return True
|
|
119
|
+
return sum(1 for _ in _markdown_blocks(text)) <= 1
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _markdown_blocks(
|
|
123
|
+
text: str,
|
|
124
|
+
) -> Iterable[tuple[int, int, str, str, tuple[str, ...]]]:
|
|
125
|
+
"""Yield exact Markdown headings, list items and paragraphs.
|
|
126
|
+
|
|
127
|
+
The splitter is layout based only. Every emitted span is a contiguous
|
|
128
|
+
slice of ``text``; heading labels are carried as section context for later
|
|
129
|
+
blocks so a candidate can cite independent facts without inheriting a
|
|
130
|
+
sibling's body or date.
|
|
131
|
+
"""
|
|
132
|
+
|
|
133
|
+
lines: list[tuple[int, int, str]] = []
|
|
134
|
+
cursor = 0
|
|
135
|
+
for raw in text.splitlines(True):
|
|
136
|
+
line_end = cursor + len(raw)
|
|
137
|
+
body = raw[:-1] if raw.endswith("\n") else raw
|
|
138
|
+
if body.endswith("\r"):
|
|
139
|
+
body = body[:-1]
|
|
140
|
+
lines.append((cursor, line_end, body))
|
|
141
|
+
cursor = line_end
|
|
142
|
+
if cursor < len(text):
|
|
143
|
+
lines.append((cursor, len(text), text[cursor:]))
|
|
144
|
+
if not lines:
|
|
145
|
+
return
|
|
146
|
+
|
|
147
|
+
headings: list[tuple[int, str]] = []
|
|
148
|
+
item_context: list[tuple[int, str]] = []
|
|
149
|
+
current_start: int | None = None
|
|
150
|
+
current_end: int | None = None
|
|
151
|
+
current_section: tuple[str, ...] = ()
|
|
152
|
+
current_syntax = "markdown_paragraph"
|
|
153
|
+
current_kind = ""
|
|
154
|
+
|
|
155
|
+
def emit() -> tuple[int, int, str, str, tuple[str, ...]] | None:
|
|
156
|
+
if current_start is None or current_end is None or current_start >= current_end:
|
|
157
|
+
return None
|
|
158
|
+
return current_start, current_end, text[current_start:current_end], current_syntax, current_section
|
|
159
|
+
|
|
160
|
+
def flush() -> tuple[int, int, str, str, tuple[str, ...]] | None:
|
|
161
|
+
nonlocal current_start, current_end, current_section, current_syntax, current_kind
|
|
162
|
+
value = emit()
|
|
163
|
+
current_start = current_end = None
|
|
164
|
+
current_section = ()
|
|
165
|
+
current_syntax = "markdown_paragraph"
|
|
166
|
+
current_kind = ""
|
|
167
|
+
return value
|
|
168
|
+
|
|
169
|
+
for line_start, line_end, body in lines:
|
|
170
|
+
left = len(body) - len(body.lstrip())
|
|
171
|
+
right = len(body.rstrip())
|
|
172
|
+
value = body.strip()
|
|
173
|
+
if not value:
|
|
174
|
+
value = flush()
|
|
175
|
+
if value is not None:
|
|
176
|
+
yield value
|
|
177
|
+
continue
|
|
178
|
+
|
|
179
|
+
heading = _MARKDOWN_HEADING.match(body)
|
|
180
|
+
item = _MARKDOWN_ITEM.match(body)
|
|
181
|
+
if heading:
|
|
182
|
+
value_before = flush()
|
|
183
|
+
if value_before is not None:
|
|
184
|
+
yield value_before
|
|
185
|
+
item_context.clear()
|
|
186
|
+
level = len(heading.group("marks"))
|
|
187
|
+
headings = [(depth, label) for depth, label in headings if depth < level]
|
|
188
|
+
section = tuple(label for _, label in headings)
|
|
189
|
+
current_start = line_start + left
|
|
190
|
+
current_end = line_start + right
|
|
191
|
+
current_section = section
|
|
192
|
+
current_syntax = "markdown_heading"
|
|
193
|
+
current_kind = "heading"
|
|
194
|
+
# Store the heading itself as a logical structural unit, then use
|
|
195
|
+
# it as context for following siblings.
|
|
196
|
+
emitted = flush()
|
|
197
|
+
if emitted is not None:
|
|
198
|
+
yield emitted
|
|
199
|
+
headings.append((level, heading.group("label").strip()))
|
|
200
|
+
continue
|
|
201
|
+
|
|
202
|
+
if item:
|
|
203
|
+
item_indent = left
|
|
204
|
+
while item_context and item_indent <= item_context[-1][0]:
|
|
205
|
+
item_context.pop()
|
|
206
|
+
if current_kind == "item":
|
|
207
|
+
value_before = flush()
|
|
208
|
+
if value_before is not None:
|
|
209
|
+
yield value_before
|
|
210
|
+
elif current_kind in {"heading", "paragraph"}:
|
|
211
|
+
value_before = flush()
|
|
212
|
+
if value_before is not None:
|
|
213
|
+
yield value_before
|
|
214
|
+
current_start = line_start + left
|
|
215
|
+
current_end = line_start + right
|
|
216
|
+
current_section = tuple(label for _, label in headings) + tuple(
|
|
217
|
+
label for _, label in item_context
|
|
218
|
+
)
|
|
219
|
+
current_syntax = "markdown_item"
|
|
220
|
+
current_kind = "item"
|
|
221
|
+
item_context.append((item_indent, body[item.end():].strip()))
|
|
222
|
+
continue
|
|
223
|
+
|
|
224
|
+
# A heading is always a complete block. A following ordinary line is
|
|
225
|
+
# therefore a paragraph under that heading, even without a blank line.
|
|
226
|
+
if current_kind == "heading":
|
|
227
|
+
value_before = flush()
|
|
228
|
+
if value_before is not None:
|
|
229
|
+
yield value_before
|
|
230
|
+
if current_start is None:
|
|
231
|
+
current_start = line_start + left
|
|
232
|
+
current_section = tuple(label for _, label in headings)
|
|
233
|
+
current_syntax = "markdown_paragraph"
|
|
234
|
+
current_kind = "paragraph"
|
|
235
|
+
current_end = line_start + right
|
|
236
|
+
|
|
237
|
+
value = flush()
|
|
238
|
+
if value is not None:
|
|
239
|
+
yield value
|
|
91
240
|
|
|
92
241
|
def _has_external_structure(text: str) -> bool:
|
|
93
242
|
"""Recognize structural boundaries without treating every line as one."""
|
|
@@ -38,6 +38,49 @@ _CLOSED_TASK = re.compile(r"(?:已|已经).{0,4}(?:全部|均)?(?:完成|取消|
|
|
|
38
38
|
_EXTERNAL_OWNER = re.compile(r"(?:客户|供应商|第三方)(?:自行|自己)?(?:需要|需|负责|必须|应当|要(?!求))|"
|
|
39
39
|
r"\b(?:customer|vendor|supplier|third party)\s+(?:must|needs? to|is responsible)\b", re.I)
|
|
40
40
|
|
|
41
|
+
# Speech-act shape only. This recognizes an assistant-only offer, question or
|
|
42
|
+
# forward commitment without naming a domain, tool or business workflow. A
|
|
43
|
+
# factual lead such as "I can confirm ..." remains eligible as an external
|
|
44
|
+
# report; the caller still validates its evidence and future value.
|
|
45
|
+
_ASSISTANT_INTENT = re.compile(
|
|
46
|
+
r"^(?:would\s+you\s+like\s+me\s+to|do\s+you\s+want\s+me\s+to|shall\s+i|let\s+me|"
|
|
47
|
+
r"i\s+(?:can|could|will|would|shall|may)|"
|
|
48
|
+
r"要我|是否需要我|需要我|我(?:可以|能|会|将|来))\s*(?P<body>.+)$",
|
|
49
|
+
re.IGNORECASE,
|
|
50
|
+
)
|
|
51
|
+
_ASSISTANT_ASSERTIVE = re.compile(
|
|
52
|
+
r"^(?:confirm|verify|report|explain|answer|summari[sz]e|describe|state|note|"
|
|
53
|
+
r"observe|see|found|find|确认|核对|报告|说明|解释|回答|总结|描述|观察|发现|指出)",
|
|
54
|
+
re.IGNORECASE,
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _assistant_intent_only(text: str) -> bool:
|
|
59
|
+
"""Recognize a standalone assistant offer/question/commitment.
|
|
60
|
+
|
|
61
|
+
The result is deliberately conservative: mixed report text and factual
|
|
62
|
+
assertions stay eligible for semantic future-value review. Only a whole
|
|
63
|
+
candidate quote whose speech act is an unaccepted offer is blocked.
|
|
64
|
+
"""
|
|
65
|
+
|
|
66
|
+
if not isinstance(text, str):
|
|
67
|
+
return False
|
|
68
|
+
value = text.strip().strip("` ")
|
|
69
|
+
value = re.sub(r"^(?:[-*+•]|\d+[.)、])\s+", "", value)
|
|
70
|
+
if not value or "\n" in value:
|
|
71
|
+
return False
|
|
72
|
+
match = _ASSISTANT_INTENT.fullmatch(value.rstrip("。.!!??"))
|
|
73
|
+
if match is None:
|
|
74
|
+
return False
|
|
75
|
+
body = match.group("body").lstrip()
|
|
76
|
+
if _ASSISTANT_ASSERTIVE.match(body):
|
|
77
|
+
# A future-tense commitment remains an intent even when its verb is
|
|
78
|
+
# epistemic (for example, "I will verify ..."). Present capability
|
|
79
|
+
# statements such as "I can confirm ..." remain report evidence.
|
|
80
|
+
lead = value[:match.start("body")]
|
|
81
|
+
return bool(re.search(r"\b(?:will|would|shall)\b|我(?:会|将|来)", lead, re.I))
|
|
82
|
+
return True
|
|
83
|
+
|
|
41
84
|
def _query(text: str) -> bool:
|
|
42
85
|
text = _POLITE.sub("", text.strip())
|
|
43
86
|
control = text.rstrip("。!?!?;;.! ")
|
|
@@ -516,6 +516,9 @@ class ModelExecutor:
|
|
|
516
516
|
"repair_rejected_semantic_drift_count": int(bucket.get("repair_rejected_semantic_drift_count", 0)),
|
|
517
517
|
"parse_accepted_count": int(bucket.get("parse_accepted_count", 0)),
|
|
518
518
|
"decision_case_normalization_count": int(bucket.get("decision_case_normalization_count", 0)),
|
|
519
|
+
"b3_normalization_count": int(bucket.get("b3_normalization_count", 0)),
|
|
520
|
+
"b3_candidate_deferred_count": int(bucket.get("b3_candidate_deferred_count", 0)),
|
|
521
|
+
"b3_ungrounded_scope_dropped_count": int(bucket.get("b3_ungrounded_scope_dropped_count", 0)),
|
|
519
522
|
"b3_due_date_ambiguous_count": int(bucket.get("b3_due_date_ambiguous_count", 0)),
|
|
520
523
|
}
|
|
521
524
|
for field in _PROVIDER_METRIC_FIELDS:
|