memleaf 0.2.37__tar.gz → 0.2.39__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memleaf-0.2.37 → memleaf-0.2.39}/CHANGELOG.md +15 -0
- {memleaf-0.2.37/src/memleaf.egg-info → memleaf-0.2.39}/PKG-INFO +2 -2
- {memleaf-0.2.37 → memleaf-0.2.39}/README.en.md +1 -1
- {memleaf-0.2.37 → memleaf-0.2.39}/README.md +1 -1
- {memleaf-0.2.37 → memleaf-0.2.39}/pyproject.toml +1 -1
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/__init__.py +1 -1
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/config.py +21 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/hermes_provider/plugin.yaml +1 -1
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/llm/base.py +10 -0
- memleaf-0.2.39/src/memleaf/llm/claude_compatible.py +67 -0
- memleaf-0.2.39/src/memleaf/llm/gemini.py +73 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/llm/openai_compatible.py +42 -11
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/llm/router.py +17 -0
- memleaf-0.2.39/src/memleaf/llm/thinking.py +259 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/memory_planner.py +55 -77
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/model_execution.py +148 -12
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/planning_context.py +1 -31
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/process_jobs.py +95 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/prompts.py +84 -57
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/update_review.py +8 -1
- {memleaf-0.2.37 → memleaf-0.2.39/src/memleaf.egg-info}/PKG-INFO +2 -2
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf.egg-info/SOURCES.txt +3 -0
- memleaf-0.2.39/tests/test_gate_scope_latency_v038.py +201 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_maintenance_v2.py +2 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_new_scope_source_grounding.py +22 -6
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_processing_observability_concurrency.py +6 -1
- memleaf-0.2.39/tests/test_provider_neutral_thinking_v039.py +246 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_session_lineage.py +8 -4
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b1.py +12 -5
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b3d_scope_maintenance.py +3 -3
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_v023_scope_correction.py +1 -1
- memleaf-0.2.37/src/memleaf/llm/claude_compatible.py +0 -31
- memleaf-0.2.37/src/memleaf/llm/gemini.py +0 -35
- {memleaf-0.2.37 → memleaf-0.2.39}/IMPLEMENTATION_PLAN.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/LICENSE +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/MANIFEST.in +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/RELEASE_CHECKLIST.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/capture-budget-design.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/config-migrations.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/core-refactor.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/evidence-retention.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/gate-evidence-boundary.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/general-processing.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/hermes-mcp-runtime.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/performance.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/processing-quality-acceptance.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/docs/v0.2.26-processing-status.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/examples/README.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/examples/basic_usage.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/examples/live_core_lifecycle_acceptance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/examples/live_processing_acceptance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/examples/mcp_stdio.ndjson +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/install.ps1 +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/install.sh +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/setup.cfg +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/__main__.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/adapters/__init__.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/adapters/antigravity.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/adapters/base.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/adapters/codex.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/adapters/hermes.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/admission.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/budget.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/capture.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/cli.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/compaction.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/create_coordinator.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/credentials.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/evidence_budget.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/evidence_policy.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/frontmatter.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/hermes_provider/README.md +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/hermes_provider/__init__.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/hermes_provider/evidence_budget.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/hermes_runtime.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/host_events.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/host_runtime.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/inbox.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/index.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/inspection.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/installer.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/llm/__init__.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/locking.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/mcp_server.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/memory_commit.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/memory_writer.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/model_discovery.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/models.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/native_index.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/native_registration.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/parallel_model.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/process_common.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/process_journal.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/process_owner.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/processing.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/provenance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/recording_policy.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/redaction.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/retention.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/retrieval.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/retrieval_gate.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/scope_maintenance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/scope_state.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/service.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/source_policy.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/state_layout.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/target_reconciliation.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/turn_audit.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/turn_plan.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/update_coordinator.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/validation.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf/vault.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf.egg-info/dependency_links.txt +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf.egg-info/entry_points.txt +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/src/memleaf.egg-info/top_level.txt +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/__init__.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/semantic_fixtures.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_admission_noise.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_automatic_duplicate_noop_collision.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_candidate_polarity.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_codex_install.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_codex_native_cli.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_config_migrations_v028.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_context_budget.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_conversation_only.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_credential_safety.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_cross_host_acceptance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_cross_turn_dedupe_regressions.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_due_date_grounding.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_due_date_grounding_retry.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_email_actionable_coverage.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_evidence_budget.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_evidence_retention_policy.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_external_source_dates.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_extraction_quality_regressions.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_gate_capacity.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_gate_schema_repair.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_general_evidence_admission.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_general_tool_provenance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_global_todo_acceptance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_global_todo_query_no_write.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_global_todo_retrieval.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_hermes_native_registration.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_hermes_provider.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_hermes_runtime_install.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_hermes_stdio_transport.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_hermes_transport_evidence.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_hermes_windows_subprocess.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_host_events.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_host_runtime_contract.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_inspection_state_v028.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_install.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_long_run_hygiene.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_model_discovery.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_model_owned_fields.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_partial_retry_idempotency.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_phase2_model_decisions.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_process_jobs.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_process_owner_locking.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_processing_contract_v026.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_pypi_install.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_read_only_deferred_isolation.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_retrieval_gate.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_retrieval_v2.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_review_source_context.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_revision_digest.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_shared_memory_refactor.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_source_neutral_todos_v028.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_a.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b2a.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b2b.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b3a_commit.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b3a_contract.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b3b_native_context.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b3b_native_index.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b3b_scope.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_b3c_retrieval.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_c1_mcp.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_c2_init.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_stage_c3_packaging.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_state_layout_v028.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_summary_date_grounding_integration.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_target_reconciliation.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_target_reconciliation_integration.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_todo_state_recovery.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_update_review.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_update_target_recovery.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_upgrade_preserves_vault.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_v2_gate_limits.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_v2_host_flow.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_v2_mcp_flow.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_v2_nomatch_semantics.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_v2_search_gate_acceptance.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_whole_unit_bindings.py +0 -0
- {memleaf-0.2.37 → memleaf-0.2.39}/tests/test_windows_public_mcp_launcher.py +0 -0
|
@@ -2,6 +2,21 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to memleaf are documented here.
|
|
4
4
|
|
|
5
|
+
## 0.2.39 — 2026-09-09
|
|
6
|
+
|
|
7
|
+
- Make `llm.thinking` a provider-neutral model policy instead of a DeepSeek-only request feature. Gate, summarize and compact continue to request `low` by default for every configured API model stage.
|
|
8
|
+
- Translate that policy through each supported protocol: OpenAI reasoning-capable Chat Completions use `reasoning_effort=low`; DeepSeek keeps its explicit thinking switch plus low effort; current Claude effort-capable Messages models use `output_config.effort=low` with adaptive thinking where the model generation requires it; Gemini 3+ uses the lowest supported thinking level (normally `low`, with documented `minimal` fallbacks where `low` is unavailable), while Gemini 2.5 maps low to the native 1,024-token thinking budget.
|
|
9
|
+
- Keep compatibility fail-safe for older or unknown models: memleaf does not send speculative reasoning fields that the model cannot accept. Per-call telemetry now distinguishes requested thinking mode from effective mode and the fixed provider control used, so unsupported/provider-default execution is visible instead of being mislabeled as low.
|
|
10
|
+
- Omit sampling temperature when an OpenAI reasoning request or current Claude effort request does not safely accept that parameter. Existing Markdown/Vault, extraction, review, retrieval and write semantics are unchanged.
|
|
11
|
+
|
|
12
|
+
## 0.2.38 — 2026-09-09
|
|
13
|
+
|
|
14
|
+
- Unify automatic project-Scope grounding: registered and newly named model-selected projects now use the same exact candidate-bound source check. Remove the later registered-name occurrence conflict scan that could misclassify an implementation platform/product mention as ownership and reject the correct new project.
|
|
15
|
+
- Strengthen final CREATE/UPDATE semantic review so a `project:<name>` Scope with `scope_source=model` is itself treated as an affiliation claim; product/platform/system/notification/implementation mentions cannot authorize project ownership, and an explicit contradictory owner defers instead of silently changing Scope.
|
|
16
|
+
- Add safe per-model-call telemetry with fixed operation classes (`gate_primary`, `gate_coverage_repair`, format repair, summarize/review/coordination variants), request duration, input/output size, provider token usage, DeepSeek cache-hit/miss tokens and reasoning-token counts when supplied. Prompt/response text and credentials are never persisted.
|
|
17
|
+
- Add explicit `llm.thinking` configuration for Gate/summarize/compact. The default is `low`, retaining reasoning at the lowest supported effort; users may select `disabled`, `default`, `high`, or `max` explicitly. DeepSeek OpenAI-format calls send the corresponding thinking controls.
|
|
18
|
+
- Reduce Gate input cost by removing a duplicated system-policy tail and replace coverage re-checks with a narrow unresolved-evidence protocol instead of rerunning the full Gate prompt. Deterministic validation does not claim a specific real-provider latency reduction.
|
|
19
|
+
|
|
5
20
|
## 0.2.37 — 2026-09-09
|
|
6
21
|
|
|
7
22
|
- Add a dedicated `memleaf-mcpw` GUI entry point for the Hermes public MCP on Windows. The GUI-subsystem launcher does not allocate a console window even when an older Hermes/MCP SDK starts it without `CREATE_NO_WINDOW`; it enters the same `memleaf.mcp_server:main` implementation and keeps the same stdio JSON-RPC protocol.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memleaf
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.39
|
|
4
4
|
Summary: A local-first Markdown memory core for AI agents
|
|
5
5
|
Author: memleaf contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -23,7 +23,7 @@ Dynamic: license-file
|
|
|
23
23
|
|
|
24
24
|
[English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
25
25
|
|
|
26
|
-
> **版本:0.2.
|
|
26
|
+
> **版本:0.2.39。**
|
|
27
27
|
> 记忆只从用户与 Agent 的可见对话提炼。Agent 已在回复中整理的事实、项目进展和明确待办可作为来源;邮件、附件、网页和终端等工具原文不进入记忆提炼。模型负责保留原话中的不确定性,现有记忆仅用于比较、去重和更新。Markdown 仍是唯一事实源。
|
|
28
28
|
> **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
|
|
29
29
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
[中文](README.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
6
6
|
|
|
7
|
-
> **Version: 0.2.
|
|
7
|
+
> **Version: 0.2.39.**
|
|
8
8
|
> Automatic extraction now uses only the current turn's visible user input and final assistant reply. Raw tool output, attachments, web/file/terminal payloads and legacy tool-evidence bodies are not new source evidence; existing or retrieved memory remains comparison context rather than source authority. Background processing is persisted as a local job and can be checked through the read-only `process_status` MCP tool; failed work remains retryable and fail closed. The release also tightens source/date grounding, target reconciliation, duplicate/no-op handling and semantic review before writes. Markdown remains the sole source of truth with no SQLite runtime dependency. Acceptance covers deterministic regression suites and synthetic inputs; it does not claim real-mail or customer-business acceptance.
|
|
9
9
|
> **The current release supports Hermes and Codex.** Antigravity is not detected, installed, or configured.
|
|
10
10
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
[English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
6
6
|
|
|
7
|
-
> **版本:0.2.
|
|
7
|
+
> **版本:0.2.39。**
|
|
8
8
|
> 记忆只从用户与 Agent 的可见对话提炼。Agent 已在回复中整理的事实、项目进展和明确待办可作为来源;邮件、附件、网页和终端等工具原文不进入记忆提炼。模型负责保留原话中的不确定性,现有记忆仅用于比较、去重和更新。Markdown 仍是唯一事实源。
|
|
9
9
|
> **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
|
|
10
10
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"""Local-first Markdown memory core for AI agents."""
|
|
2
2
|
|
|
3
|
-
__version__ = "0.2.
|
|
3
|
+
__version__ = "0.2.39"
|
|
4
4
|
|
|
5
5
|
from .config import DEFAULT_CONFIG, default_config, load_config, save_config
|
|
6
6
|
from .frontmatter import FrontmatterError, dump_frontmatter, dump_yaml, load_yaml, parse_frontmatter
|
|
@@ -19,6 +19,9 @@ MAX_REQUEST_TIMEOUT = 240
|
|
|
19
19
|
DEFAULT_MODEL_CONCURRENCY = 3
|
|
20
20
|
MIN_MODEL_CONCURRENCY = 1
|
|
21
21
|
MAX_MODEL_CONCURRENCY = 8
|
|
22
|
+
THINKING_PURPOSES = ("gate", "summarize", "compact")
|
|
23
|
+
THINKING_MODES = frozenset({"default", "disabled", "low", "high", "max"})
|
|
24
|
+
DEFAULT_THINKING = {purpose: "low" for purpose in THINKING_PURPOSES}
|
|
22
25
|
|
|
23
26
|
|
|
24
27
|
def _normalize_request_timeout(value: Any) -> int | float:
|
|
@@ -41,6 +44,21 @@ def _normalize_model_concurrency(value: Any) -> int:
|
|
|
41
44
|
return value
|
|
42
45
|
|
|
43
46
|
|
|
47
|
+
def _normalize_thinking_settings(value: Any) -> dict[str, str]:
|
|
48
|
+
if value is None:
|
|
49
|
+
value = {}
|
|
50
|
+
if not isinstance(value, Mapping):
|
|
51
|
+
raise ValueError("invalid memleaf llm.thinking settings")
|
|
52
|
+
if set(value) - set(THINKING_PURPOSES):
|
|
53
|
+
raise ValueError("invalid memleaf llm.thinking settings")
|
|
54
|
+
result = dict(DEFAULT_THINKING)
|
|
55
|
+
for purpose, mode in value.items():
|
|
56
|
+
if not isinstance(mode, str) or mode not in THINKING_MODES:
|
|
57
|
+
raise ValueError("invalid memleaf llm.thinking settings")
|
|
58
|
+
result[purpose] = mode
|
|
59
|
+
return result
|
|
60
|
+
|
|
61
|
+
|
|
44
62
|
DEFAULT_CONFIG: dict[str, Any] = {
|
|
45
63
|
"vault": "~/.memleaf",
|
|
46
64
|
"agents": {"codex": True, "hermes": True, "antigravity": False},
|
|
@@ -75,6 +93,7 @@ DEFAULT_CONFIG: dict[str, Any] = {
|
|
|
75
93
|
"context_window": 200000,
|
|
76
94
|
"request_timeout": DEFAULT_REQUEST_TIMEOUT,
|
|
77
95
|
"diagnostic_logging": False,
|
|
96
|
+
"thinking": dict(DEFAULT_THINKING),
|
|
78
97
|
},
|
|
79
98
|
}
|
|
80
99
|
|
|
@@ -188,6 +207,7 @@ def load_config(path: Path | str, *, vault: Path | str | None = None) -> dict[st
|
|
|
188
207
|
raise ValueError("invalid memleaf llm settings")
|
|
189
208
|
llm = dict(llm)
|
|
190
209
|
llm["request_timeout"] = _normalize_request_timeout(llm.get("request_timeout", DEFAULT_REQUEST_TIMEOUT))
|
|
210
|
+
llm["thinking"] = _normalize_thinking_settings(llm.get("thinking"))
|
|
191
211
|
if type(llm.get("diagnostic_logging", False)) is not bool:
|
|
192
212
|
raise ValueError("invalid memleaf llm.diagnostic_logging")
|
|
193
213
|
merged["llm"] = llm
|
|
@@ -231,6 +251,7 @@ def save_config(path: Path | str, config: Mapping[str, Any]) -> None:
|
|
|
231
251
|
normalized_llm["request_timeout"] = _normalize_request_timeout(
|
|
232
252
|
normalized_llm.get("request_timeout", DEFAULT_REQUEST_TIMEOUT)
|
|
233
253
|
)
|
|
254
|
+
normalized_llm["thinking"] = _normalize_thinking_settings(normalized_llm.get("thinking"))
|
|
234
255
|
diagnostic_logging = normalized_llm.get("diagnostic_logging", False)
|
|
235
256
|
if type(diagnostic_logging) is not bool:
|
|
236
257
|
raise ValueError("invalid memleaf llm.diagnostic_logging")
|
|
@@ -6,6 +6,7 @@ import json
|
|
|
6
6
|
import inspect
|
|
7
7
|
import math
|
|
8
8
|
import socket
|
|
9
|
+
import threading
|
|
9
10
|
import urllib.error
|
|
10
11
|
import urllib.request
|
|
11
12
|
from typing import Any, Callable, Mapping, Optional, Protocol
|
|
@@ -246,10 +247,19 @@ class HTTPModelBackend:
|
|
|
246
247
|
self.model = model
|
|
247
248
|
self.timeout = normalize_request_timeout(timeout)
|
|
248
249
|
self._opener = opener or urllib.request.urlopen
|
|
250
|
+
self._call_metrics_local = threading.local()
|
|
249
251
|
# The built-in stateless urllib transport can be used concurrently.
|
|
250
252
|
# An injected opener is caller-owned and therefore defaults to serial.
|
|
251
253
|
self.parallel_safe = opener is None
|
|
252
254
|
|
|
255
|
+
def _set_call_metrics(self, value: Mapping[str, Any] | None) -> None:
|
|
256
|
+
self._call_metrics_local.value = dict(value) if isinstance(value, Mapping) else {}
|
|
257
|
+
|
|
258
|
+
def consume_call_metrics(self) -> dict[str, Any]:
|
|
259
|
+
value = getattr(self._call_metrics_local, "value", {})
|
|
260
|
+
self._call_metrics_local.value = {}
|
|
261
|
+
return dict(value) if isinstance(value, Mapping) else {}
|
|
262
|
+
|
|
253
263
|
@staticmethod
|
|
254
264
|
def _is_timeout_reason(value: Any) -> bool:
|
|
255
265
|
if isinstance(value, (TimeoutError, socket.timeout)):
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Claude-compatible messages adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Callable, Mapping, Optional
|
|
6
|
+
|
|
7
|
+
from .base import DEFAULT_REQUEST_TIMEOUT, HTTPModelBackend
|
|
8
|
+
from .thinking import claude_messages_controls, requested_thinking_mode
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ClaudeCompatibleBackend(HTTPModelBackend):
|
|
12
|
+
provider = "claude"
|
|
13
|
+
|
|
14
|
+
def __init__(
|
|
15
|
+
self,
|
|
16
|
+
*,
|
|
17
|
+
base_url: str,
|
|
18
|
+
api_key: str,
|
|
19
|
+
model: str,
|
|
20
|
+
timeout: float = DEFAULT_REQUEST_TIMEOUT,
|
|
21
|
+
opener: Optional[Callable[..., Any]] = None,
|
|
22
|
+
thinking: Mapping[str, Any] | None = None,
|
|
23
|
+
):
|
|
24
|
+
super().__init__(base_url=base_url, api_key=api_key, model=model, timeout=timeout, opener=opener)
|
|
25
|
+
self.thinking = dict(thinking) if isinstance(thinking, Mapping) else {}
|
|
26
|
+
|
|
27
|
+
@staticmethod
|
|
28
|
+
def _usage_metrics(value: Mapping[str, Any], thinking_metrics: Mapping[str, Any]) -> dict[str, Any]:
|
|
29
|
+
result: dict[str, Any] = dict(thinking_metrics)
|
|
30
|
+
usage = value.get("usage")
|
|
31
|
+
if not isinstance(usage, Mapping):
|
|
32
|
+
return result
|
|
33
|
+
input_tokens = usage.get("input_tokens")
|
|
34
|
+
output_tokens = usage.get("output_tokens")
|
|
35
|
+
if isinstance(input_tokens, int) and not isinstance(input_tokens, bool) and 0 <= input_tokens <= 10_000_000:
|
|
36
|
+
result["prompt_tokens"] = input_tokens
|
|
37
|
+
if isinstance(output_tokens, int) and not isinstance(output_tokens, bool) and 0 <= output_tokens <= 10_000_000:
|
|
38
|
+
result["completion_tokens"] = output_tokens
|
|
39
|
+
if "prompt_tokens" in result and "completion_tokens" in result:
|
|
40
|
+
result["total_tokens"] = result["prompt_tokens"] + result["completion_tokens"]
|
|
41
|
+
cache_read = usage.get("cache_read_input_tokens")
|
|
42
|
+
if isinstance(cache_read, int) and not isinstance(cache_read, bool) and 0 <= cache_read <= 10_000_000:
|
|
43
|
+
result["prompt_cache_hit_tokens"] = cache_read
|
|
44
|
+
return result
|
|
45
|
+
|
|
46
|
+
def complete(self, prompt: str, *, system: str = "", purpose: str = "", temperature: float = 0.0) -> str:
|
|
47
|
+
self._set_call_metrics({})
|
|
48
|
+
requested = requested_thinking_mode(self.thinking, purpose)
|
|
49
|
+
controls, thinking_metrics, omit_temperature = claude_messages_controls(self.model, requested)
|
|
50
|
+
payload: dict[str, Any] = {
|
|
51
|
+
"model": self.model,
|
|
52
|
+
"max_tokens": 4096,
|
|
53
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
54
|
+
}
|
|
55
|
+
if not omit_temperature:
|
|
56
|
+
payload["temperature"] = temperature
|
|
57
|
+
if system:
|
|
58
|
+
payload["system"] = system
|
|
59
|
+
payload.update(controls)
|
|
60
|
+
value = self._post_json(
|
|
61
|
+
self.base_url + "/v1/messages",
|
|
62
|
+
payload,
|
|
63
|
+
{"x-api-key": self.api_key, "anthropic-version": "2023-06-01"},
|
|
64
|
+
stage=purpose,
|
|
65
|
+
)
|
|
66
|
+
self._set_call_metrics(self._usage_metrics(value, thinking_metrics))
|
|
67
|
+
return self._text(value.get("content"), stage=purpose)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Gemini generateContent adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import urllib.parse
|
|
6
|
+
from typing import Any, Callable, Mapping, Optional
|
|
7
|
+
|
|
8
|
+
from .base import DEFAULT_REQUEST_TIMEOUT, HTTPModelBackend, ModelError
|
|
9
|
+
from .thinking import gemini_generate_controls, requested_thinking_mode
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class GeminiBackend(HTTPModelBackend):
|
|
13
|
+
provider = "gemini"
|
|
14
|
+
|
|
15
|
+
def __init__(
|
|
16
|
+
self,
|
|
17
|
+
*,
|
|
18
|
+
base_url: str,
|
|
19
|
+
api_key: str,
|
|
20
|
+
model: str,
|
|
21
|
+
timeout: float = DEFAULT_REQUEST_TIMEOUT,
|
|
22
|
+
opener: Optional[Callable[..., Any]] = None,
|
|
23
|
+
thinking: Mapping[str, Any] | None = None,
|
|
24
|
+
):
|
|
25
|
+
super().__init__(base_url=base_url, api_key=api_key, model=model, timeout=timeout, opener=opener)
|
|
26
|
+
self.thinking = dict(thinking) if isinstance(thinking, Mapping) else {}
|
|
27
|
+
|
|
28
|
+
@staticmethod
|
|
29
|
+
def _usage_metrics(value: Mapping[str, Any], thinking_metrics: Mapping[str, Any]) -> dict[str, Any]:
|
|
30
|
+
result: dict[str, Any] = dict(thinking_metrics)
|
|
31
|
+
usage = value.get("usageMetadata")
|
|
32
|
+
if not isinstance(usage, Mapping):
|
|
33
|
+
return result
|
|
34
|
+
fields = {
|
|
35
|
+
"promptTokenCount": "prompt_tokens",
|
|
36
|
+
"candidatesTokenCount": "completion_tokens",
|
|
37
|
+
"totalTokenCount": "total_tokens",
|
|
38
|
+
"cachedContentTokenCount": "prompt_cache_hit_tokens",
|
|
39
|
+
"thoughtsTokenCount": "reasoning_tokens",
|
|
40
|
+
}
|
|
41
|
+
for source, target in fields.items():
|
|
42
|
+
item = usage.get(source)
|
|
43
|
+
if isinstance(item, int) and not isinstance(item, bool) and 0 <= item <= 10_000_000:
|
|
44
|
+
result[target] = item
|
|
45
|
+
return result
|
|
46
|
+
|
|
47
|
+
def complete(self, prompt: str, *, system: str = "", purpose: str = "", temperature: float = 0.0) -> str:
|
|
48
|
+
self._set_call_metrics({})
|
|
49
|
+
text = f"{system}\n\n{prompt}" if system else prompt
|
|
50
|
+
requested = requested_thinking_mode(self.thinking, purpose)
|
|
51
|
+
thinking_config, thinking_metrics, omit_temperature = gemini_generate_controls(self.model, requested)
|
|
52
|
+
generation_config: dict[str, Any] = {}
|
|
53
|
+
if not omit_temperature:
|
|
54
|
+
generation_config["temperature"] = temperature
|
|
55
|
+
generation_config.update(thinking_config)
|
|
56
|
+
payload = {
|
|
57
|
+
"contents": [{"role": "user", "parts": [{"text": text}]}],
|
|
58
|
+
"generationConfig": generation_config,
|
|
59
|
+
}
|
|
60
|
+
endpoint = "/v1beta/models/" + urllib.parse.quote(self.model, safe="") + ":generateContent"
|
|
61
|
+
url = self.base_url + endpoint + "?key=" + urllib.parse.quote(self.api_key, safe="")
|
|
62
|
+
value = self._post_json(url, payload, {}, stage=purpose)
|
|
63
|
+
self._set_call_metrics(self._usage_metrics(value, thinking_metrics))
|
|
64
|
+
candidates = value.get("candidates")
|
|
65
|
+
if not isinstance(candidates, list) or not candidates or not isinstance(candidates[0], Mapping):
|
|
66
|
+
raise ModelError("model response has no candidates", code="model_invalid_response", stage=purpose)
|
|
67
|
+
content = candidates[0].get("content")
|
|
68
|
+
if not isinstance(content, Mapping):
|
|
69
|
+
raise ModelError("model response has no content", code="model_invalid_response", stage=purpose)
|
|
70
|
+
parts = content.get("parts")
|
|
71
|
+
if not isinstance(parts, list):
|
|
72
|
+
raise ModelError("model response has no parts", code="model_invalid_response", stage=purpose)
|
|
73
|
+
return self._text(parts, stage=purpose)
|
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
from typing import Any, Callable, Mapping, Optional
|
|
6
6
|
|
|
7
7
|
from .base import DEFAULT_REQUEST_TIMEOUT, HTTPModelBackend, ModelError
|
|
8
|
+
from .thinking import openai_chat_controls, requested_thinking_mode
|
|
8
9
|
|
|
9
10
|
|
|
10
11
|
class OpenAICompatibleBackend(HTTPModelBackend):
|
|
@@ -20,10 +21,12 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
20
21
|
opener: Optional[Callable[..., Any]] = None,
|
|
21
22
|
json_mode: bool = False,
|
|
22
23
|
provider_name: str = "openai",
|
|
24
|
+
thinking: Mapping[str, Any] | None = None,
|
|
23
25
|
):
|
|
24
26
|
super().__init__(base_url=base_url, api_key=api_key, model=model, timeout=timeout, opener=opener)
|
|
25
27
|
self.json_mode = bool(json_mode)
|
|
26
28
|
self.provider_name = provider_name.casefold() if isinstance(provider_name, str) else "openai"
|
|
29
|
+
self.thinking = dict(thinking) if isinstance(thinking, Mapping) else {}
|
|
27
30
|
|
|
28
31
|
@staticmethod
|
|
29
32
|
def _response_text_chars(value: Any) -> int:
|
|
@@ -52,12 +55,8 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
52
55
|
else:
|
|
53
56
|
finish_reason = finish_reason.casefold()
|
|
54
57
|
if finish_reason not in {
|
|
55
|
-
"stop",
|
|
56
|
-
"
|
|
57
|
-
"tool_calls",
|
|
58
|
-
"function_call",
|
|
59
|
-
"content_filter",
|
|
60
|
-
"insufficient_system_resource",
|
|
58
|
+
"stop", "length", "tool_calls", "function_call",
|
|
59
|
+
"content_filter", "insufficient_system_resource",
|
|
61
60
|
}:
|
|
62
61
|
finish_reason = "unknown"
|
|
63
62
|
usage = value.get("usage")
|
|
@@ -83,16 +82,47 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
83
82
|
"reasoning_chars": reasoning_chars,
|
|
84
83
|
}
|
|
85
84
|
|
|
85
|
+
@staticmethod
|
|
86
|
+
def _usage_metrics(
|
|
87
|
+
value: Mapping[str, Any],
|
|
88
|
+
*,
|
|
89
|
+
thinking_metrics: Mapping[str, Any],
|
|
90
|
+
) -> dict[str, Any]:
|
|
91
|
+
usage = value.get("usage")
|
|
92
|
+
result: dict[str, Any] = dict(thinking_metrics)
|
|
93
|
+
if not isinstance(usage, Mapping):
|
|
94
|
+
return result
|
|
95
|
+
for key in (
|
|
96
|
+
"prompt_tokens", "completion_tokens", "total_tokens",
|
|
97
|
+
"prompt_cache_hit_tokens", "prompt_cache_miss_tokens",
|
|
98
|
+
):
|
|
99
|
+
item = usage.get(key)
|
|
100
|
+
if isinstance(item, int) and not isinstance(item, bool) and 0 <= item <= 10_000_000:
|
|
101
|
+
result[key] = item
|
|
102
|
+
prompt_details = usage.get("prompt_tokens_details")
|
|
103
|
+
cached = prompt_details.get("cached_tokens") if isinstance(prompt_details, Mapping) else None
|
|
104
|
+
if "prompt_cache_hit_tokens" not in result and isinstance(cached, int) and not isinstance(cached, bool) and 0 <= cached <= 10_000_000:
|
|
105
|
+
result["prompt_cache_hit_tokens"] = cached
|
|
106
|
+
details = usage.get("completion_tokens_details")
|
|
107
|
+
reasoning = details.get("reasoning_tokens") if isinstance(details, Mapping) else None
|
|
108
|
+
if isinstance(reasoning, int) and not isinstance(reasoning, bool) and 0 <= reasoning <= 10_000_000:
|
|
109
|
+
result["reasoning_tokens"] = reasoning
|
|
110
|
+
return result
|
|
111
|
+
|
|
86
112
|
def complete(self, prompt: str, *, system: str = "", purpose: str = "", temperature: float = 0.0) -> str:
|
|
113
|
+
self._set_call_metrics({})
|
|
87
114
|
messages = []
|
|
88
115
|
if system:
|
|
89
116
|
messages.append({"role": "system", "content": system})
|
|
90
117
|
messages.append({"role": "user", "content": prompt})
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
}
|
|
118
|
+
requested = requested_thinking_mode(self.thinking, purpose)
|
|
119
|
+
controls, thinking_metrics, omit_temperature = openai_chat_controls(
|
|
120
|
+
self.provider_name, self.model, requested
|
|
121
|
+
)
|
|
122
|
+
payload: dict[str, Any] = {"model": self.model, "messages": messages}
|
|
123
|
+
if not omit_temperature:
|
|
124
|
+
payload["temperature"] = temperature
|
|
125
|
+
payload.update(controls)
|
|
96
126
|
if self.json_mode and purpose in {"gate", "summarize", "compact"}:
|
|
97
127
|
payload["response_format"] = {"type": "json_object"}
|
|
98
128
|
value = self._post_json(
|
|
@@ -101,6 +131,7 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
101
131
|
{"Authorization": f"Bearer {self.api_key}"},
|
|
102
132
|
stage=purpose,
|
|
103
133
|
)
|
|
134
|
+
self._set_call_metrics(self._usage_metrics(value, thinking_metrics=thinking_metrics))
|
|
104
135
|
choices = value.get("choices")
|
|
105
136
|
if not isinstance(choices, list) or not choices or not isinstance(choices[0], Mapping):
|
|
106
137
|
raise ModelError(
|
|
@@ -4,6 +4,7 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import logging
|
|
6
6
|
import os
|
|
7
|
+
import threading
|
|
7
8
|
from typing import Any, Callable, Mapping, Optional
|
|
8
9
|
|
|
9
10
|
from .base import (
|
|
@@ -44,6 +45,7 @@ class ModelRouter:
|
|
|
44
45
|
self.host = self._coerce_host(host)
|
|
45
46
|
self.api = self._coerce_api(api) if api is not None else self._build_api()
|
|
46
47
|
self.diagnostics: list[dict[str, str]] = []
|
|
48
|
+
self._call_metrics_local = threading.local()
|
|
47
49
|
|
|
48
50
|
@classmethod
|
|
49
51
|
def from_config(cls, config: Mapping[str, Any], **kwargs: Any) -> "ModelRouter":
|
|
@@ -111,6 +113,7 @@ class ModelRouter:
|
|
|
111
113
|
"api_key": api_key,
|
|
112
114
|
"model": model,
|
|
113
115
|
"timeout": request_timeout,
|
|
116
|
+
"thinking": config.get("thinking") if isinstance(config.get("thinking"), Mapping) else None,
|
|
114
117
|
}
|
|
115
118
|
try:
|
|
116
119
|
if protocol in ("claude", "anthropic") or "claude" in provider or "anthropic" in provider:
|
|
@@ -142,6 +145,7 @@ class ModelRouter:
|
|
|
142
145
|
return str(getattr(backend, "provider", "unknown")), str(getattr(backend, "model", "unknown"))
|
|
143
146
|
|
|
144
147
|
def _call(self, backend: ModelBackend, prompt: str, *, system: str, purpose: str, temperature: float) -> str:
|
|
148
|
+
self._call_metrics_local.value = {}
|
|
145
149
|
try:
|
|
146
150
|
value = backend.complete(prompt, system=system, purpose=purpose, temperature=temperature)
|
|
147
151
|
except ModelError as error:
|
|
@@ -149,10 +153,23 @@ class ModelRouter:
|
|
|
149
153
|
raise
|
|
150
154
|
except Exception as error:
|
|
151
155
|
raise ModelError("model backend failed", stage=purpose) from error
|
|
156
|
+
finally:
|
|
157
|
+
consume = getattr(backend, "consume_call_metrics", None)
|
|
158
|
+
if callable(consume):
|
|
159
|
+
try:
|
|
160
|
+
metrics = consume()
|
|
161
|
+
except Exception:
|
|
162
|
+
metrics = {}
|
|
163
|
+
self._call_metrics_local.value = dict(metrics) if isinstance(metrics, Mapping) else {}
|
|
152
164
|
if not isinstance(value, str):
|
|
153
165
|
raise ModelError("model backend returned non-text output", code="model_invalid_response", stage=purpose)
|
|
154
166
|
return value
|
|
155
167
|
|
|
168
|
+
def consume_call_metrics(self) -> dict[str, Any]:
|
|
169
|
+
value = getattr(self._call_metrics_local, "value", {})
|
|
170
|
+
self._call_metrics_local.value = {}
|
|
171
|
+
return dict(value) if isinstance(value, Mapping) else {}
|
|
172
|
+
|
|
156
173
|
def complete(self, prompt: str, *, system: str = "", purpose: str = "", temperature: float = 0.0) -> str:
|
|
157
174
|
if not isinstance(prompt, str):
|
|
158
175
|
raise TypeError("model prompt must be text")
|