memleaf 0.2.38__tar.gz → 0.2.39__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memleaf-0.2.38 → memleaf-0.2.39}/CHANGELOG.md +7 -0
- {memleaf-0.2.38/src/memleaf.egg-info → memleaf-0.2.39}/PKG-INFO +2 -2
- {memleaf-0.2.38 → memleaf-0.2.39}/README.en.md +1 -1
- {memleaf-0.2.38 → memleaf-0.2.39}/README.md +1 -1
- {memleaf-0.2.38 → memleaf-0.2.39}/pyproject.toml +1 -1
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/__init__.py +1 -1
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/hermes_provider/plugin.yaml +1 -1
- memleaf-0.2.39/src/memleaf/llm/claude_compatible.py +67 -0
- memleaf-0.2.39/src/memleaf/llm/gemini.py +73 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/llm/openai_compatible.py +22 -27
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/llm/router.py +1 -1
- memleaf-0.2.39/src/memleaf/llm/thinking.py +259 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/model_execution.py +12 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/process_jobs.py +12 -0
- {memleaf-0.2.38 → memleaf-0.2.39/src/memleaf.egg-info}/PKG-INFO +2 -2
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf.egg-info/SOURCES.txt +2 -0
- memleaf-0.2.39/tests/test_provider_neutral_thinking_v039.py +246 -0
- memleaf-0.2.38/src/memleaf/llm/claude_compatible.py +0 -31
- memleaf-0.2.38/src/memleaf/llm/gemini.py +0 -35
- {memleaf-0.2.38 → memleaf-0.2.39}/IMPLEMENTATION_PLAN.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/LICENSE +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/MANIFEST.in +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/RELEASE_CHECKLIST.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/capture-budget-design.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/config-migrations.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/core-refactor.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/evidence-retention.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/gate-evidence-boundary.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/general-processing.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/hermes-mcp-runtime.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/performance.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/processing-quality-acceptance.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/docs/v0.2.26-processing-status.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/examples/README.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/examples/basic_usage.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/examples/live_core_lifecycle_acceptance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/examples/live_processing_acceptance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/examples/mcp_stdio.ndjson +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/install.ps1 +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/install.sh +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/setup.cfg +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/__main__.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/adapters/__init__.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/adapters/antigravity.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/adapters/base.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/adapters/codex.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/adapters/hermes.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/admission.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/budget.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/capture.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/cli.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/compaction.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/config.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/create_coordinator.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/credentials.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/evidence_budget.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/evidence_policy.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/frontmatter.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/hermes_provider/README.md +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/hermes_provider/__init__.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/hermes_provider/evidence_budget.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/hermes_runtime.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/host_events.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/host_runtime.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/inbox.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/index.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/inspection.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/installer.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/llm/__init__.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/llm/base.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/locking.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/mcp_server.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/memory_commit.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/memory_planner.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/memory_writer.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/model_discovery.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/models.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/native_index.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/native_registration.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/parallel_model.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/planning_context.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/process_common.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/process_journal.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/process_owner.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/processing.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/prompts.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/provenance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/recording_policy.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/redaction.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/retention.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/retrieval.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/retrieval_gate.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/scope_maintenance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/scope_state.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/service.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/source_policy.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/state_layout.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/target_reconciliation.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/turn_audit.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/turn_plan.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/update_coordinator.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/update_review.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/validation.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf/vault.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf.egg-info/dependency_links.txt +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf.egg-info/entry_points.txt +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/src/memleaf.egg-info/top_level.txt +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/__init__.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/semantic_fixtures.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_admission_noise.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_automatic_duplicate_noop_collision.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_candidate_polarity.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_codex_install.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_codex_native_cli.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_config_migrations_v028.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_context_budget.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_conversation_only.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_credential_safety.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_cross_host_acceptance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_cross_turn_dedupe_regressions.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_due_date_grounding.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_due_date_grounding_retry.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_email_actionable_coverage.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_evidence_budget.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_evidence_retention_policy.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_external_source_dates.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_extraction_quality_regressions.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_gate_capacity.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_gate_schema_repair.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_gate_scope_latency_v038.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_general_evidence_admission.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_general_tool_provenance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_global_todo_acceptance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_global_todo_query_no_write.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_global_todo_retrieval.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_hermes_native_registration.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_hermes_provider.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_hermes_runtime_install.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_hermes_stdio_transport.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_hermes_transport_evidence.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_hermes_windows_subprocess.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_host_events.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_host_runtime_contract.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_inspection_state_v028.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_install.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_long_run_hygiene.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_maintenance_v2.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_model_discovery.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_model_owned_fields.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_new_scope_source_grounding.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_partial_retry_idempotency.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_phase2_model_decisions.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_process_jobs.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_process_owner_locking.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_processing_contract_v026.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_processing_observability_concurrency.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_pypi_install.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_read_only_deferred_isolation.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_retrieval_gate.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_retrieval_v2.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_review_source_context.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_revision_digest.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_session_lineage.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_shared_memory_refactor.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_source_neutral_todos_v028.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_a.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b1.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b2a.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b2b.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b3a_commit.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b3a_contract.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b3b_native_context.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b3b_native_index.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b3b_scope.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b3c_retrieval.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_b3d_scope_maintenance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_c1_mcp.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_c2_init.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_stage_c3_packaging.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_state_layout_v028.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_summary_date_grounding_integration.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_target_reconciliation.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_target_reconciliation_integration.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_todo_state_recovery.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_update_review.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_update_target_recovery.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_upgrade_preserves_vault.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_v023_scope_correction.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_v2_gate_limits.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_v2_host_flow.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_v2_mcp_flow.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_v2_nomatch_semantics.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_v2_search_gate_acceptance.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_whole_unit_bindings.py +0 -0
- {memleaf-0.2.38 → memleaf-0.2.39}/tests/test_windows_public_mcp_launcher.py +0 -0
|
@@ -2,6 +2,13 @@
|
|
|
2
2
|
|
|
3
3
|
All notable changes to memleaf are documented here.
|
|
4
4
|
|
|
5
|
+
## 0.2.39 — 2026-09-09
|
|
6
|
+
|
|
7
|
+
- Make `llm.thinking` a provider-neutral model policy instead of a DeepSeek-only request feature. Gate, summarize and compact continue to request `low` by default for every configured API model stage.
|
|
8
|
+
- Translate that policy through each supported protocol: OpenAI reasoning-capable Chat Completions use `reasoning_effort=low`; DeepSeek keeps its explicit thinking switch plus low effort; current Claude effort-capable Messages models use `output_config.effort=low` with adaptive thinking where the model generation requires it; Gemini 3+ uses the lowest supported thinking level (normally `low`, with documented `minimal` fallbacks where `low` is unavailable), while Gemini 2.5 maps low to the native 1,024-token thinking budget.
|
|
9
|
+
- Keep compatibility fail-safe for older or unknown models: memleaf does not send speculative reasoning fields that the model cannot accept. Per-call telemetry now distinguishes requested thinking mode from effective mode and the fixed provider control used, so unsupported/provider-default execution is visible instead of being mislabeled as low.
|
|
10
|
+
- Omit sampling temperature when an OpenAI reasoning request or current Claude effort request does not safely accept that parameter. Existing Markdown/Vault, extraction, review, retrieval and write semantics are unchanged.
|
|
11
|
+
|
|
5
12
|
## 0.2.38 — 2026-09-09
|
|
6
13
|
|
|
7
14
|
- Unify automatic project-Scope grounding: registered and newly named model-selected projects now use the same exact candidate-bound source check. Remove the later registered-name occurrence conflict scan that could misclassify an implementation platform/product mention as ownership and reject the correct new project.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memleaf
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.39
|
|
4
4
|
Summary: A local-first Markdown memory core for AI agents
|
|
5
5
|
Author: memleaf contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -23,7 +23,7 @@ Dynamic: license-file
|
|
|
23
23
|
|
|
24
24
|
[English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
25
25
|
|
|
26
|
-
> **版本:0.2.
|
|
26
|
+
> **版本:0.2.39。**
|
|
27
27
|
> 记忆只从用户与 Agent 的可见对话提炼。Agent 已在回复中整理的事实、项目进展和明确待办可作为来源;邮件、附件、网页和终端等工具原文不进入记忆提炼。模型负责保留原话中的不确定性,现有记忆仅用于比较、去重和更新。Markdown 仍是唯一事实源。
|
|
28
28
|
> **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
|
|
29
29
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
[中文](README.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
6
6
|
|
|
7
|
-
> **Version: 0.2.
|
|
7
|
+
> **Version: 0.2.39.**
|
|
8
8
|
> Automatic extraction now uses only the current turn's visible user input and final assistant reply. Raw tool output, attachments, web/file/terminal payloads and legacy tool-evidence bodies are not new source evidence; existing or retrieved memory remains comparison context rather than source authority. Background processing is persisted as a local job and can be checked through the read-only `process_status` MCP tool; failed work remains retryable and fail closed. The release also tightens source/date grounding, target reconciliation, duplicate/no-op handling and semantic review before writes. Markdown remains the sole source of truth with no SQLite runtime dependency. Acceptance covers deterministic regression suites and synthetic inputs; it does not claim real-mail or customer-business acceptance.
|
|
9
9
|
> **The current release supports Hermes and Codex.** Antigravity is not detected, installed, or configured.
|
|
10
10
|
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
|
|
5
5
|
[English](README.en.md) · [PyPI](https://pypi.org/project/memleaf/) · [GitHub](https://github.com/miffyblueboo/memleaf)
|
|
6
6
|
|
|
7
|
-
> **版本:0.2.
|
|
7
|
+
> **版本:0.2.39。**
|
|
8
8
|
> 记忆只从用户与 Agent 的可见对话提炼。Agent 已在回复中整理的事实、项目进展和明确待办可作为来源;邮件、附件、网页和终端等工具原文不进入记忆提炼。模型负责保留原话中的不确定性,现有记忆仅用于比较、去重和更新。Markdown 仍是唯一事实源。
|
|
9
9
|
> **当前版本支持 Hermes 和 Codex。** Antigravity(反重力)不检测、不安装、不配置。
|
|
10
10
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
"""Local-first Markdown memory core for AI agents."""
|
|
2
2
|
|
|
3
|
-
__version__ = "0.2.
|
|
3
|
+
__version__ = "0.2.39"
|
|
4
4
|
|
|
5
5
|
from .config import DEFAULT_CONFIG, default_config, load_config, save_config
|
|
6
6
|
from .frontmatter import FrontmatterError, dump_frontmatter, dump_yaml, load_yaml, parse_frontmatter
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Claude-compatible messages adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any, Callable, Mapping, Optional
|
|
6
|
+
|
|
7
|
+
from .base import DEFAULT_REQUEST_TIMEOUT, HTTPModelBackend
|
|
8
|
+
from .thinking import claude_messages_controls, requested_thinking_mode
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class ClaudeCompatibleBackend(HTTPModelBackend):
|
|
12
|
+
provider = "claude"
|
|
13
|
+
|
|
14
|
+
def __init__(
|
|
15
|
+
self,
|
|
16
|
+
*,
|
|
17
|
+
base_url: str,
|
|
18
|
+
api_key: str,
|
|
19
|
+
model: str,
|
|
20
|
+
timeout: float = DEFAULT_REQUEST_TIMEOUT,
|
|
21
|
+
opener: Optional[Callable[..., Any]] = None,
|
|
22
|
+
thinking: Mapping[str, Any] | None = None,
|
|
23
|
+
):
|
|
24
|
+
super().__init__(base_url=base_url, api_key=api_key, model=model, timeout=timeout, opener=opener)
|
|
25
|
+
self.thinking = dict(thinking) if isinstance(thinking, Mapping) else {}
|
|
26
|
+
|
|
27
|
+
@staticmethod
|
|
28
|
+
def _usage_metrics(value: Mapping[str, Any], thinking_metrics: Mapping[str, Any]) -> dict[str, Any]:
|
|
29
|
+
result: dict[str, Any] = dict(thinking_metrics)
|
|
30
|
+
usage = value.get("usage")
|
|
31
|
+
if not isinstance(usage, Mapping):
|
|
32
|
+
return result
|
|
33
|
+
input_tokens = usage.get("input_tokens")
|
|
34
|
+
output_tokens = usage.get("output_tokens")
|
|
35
|
+
if isinstance(input_tokens, int) and not isinstance(input_tokens, bool) and 0 <= input_tokens <= 10_000_000:
|
|
36
|
+
result["prompt_tokens"] = input_tokens
|
|
37
|
+
if isinstance(output_tokens, int) and not isinstance(output_tokens, bool) and 0 <= output_tokens <= 10_000_000:
|
|
38
|
+
result["completion_tokens"] = output_tokens
|
|
39
|
+
if "prompt_tokens" in result and "completion_tokens" in result:
|
|
40
|
+
result["total_tokens"] = result["prompt_tokens"] + result["completion_tokens"]
|
|
41
|
+
cache_read = usage.get("cache_read_input_tokens")
|
|
42
|
+
if isinstance(cache_read, int) and not isinstance(cache_read, bool) and 0 <= cache_read <= 10_000_000:
|
|
43
|
+
result["prompt_cache_hit_tokens"] = cache_read
|
|
44
|
+
return result
|
|
45
|
+
|
|
46
|
+
def complete(self, prompt: str, *, system: str = "", purpose: str = "", temperature: float = 0.0) -> str:
|
|
47
|
+
self._set_call_metrics({})
|
|
48
|
+
requested = requested_thinking_mode(self.thinking, purpose)
|
|
49
|
+
controls, thinking_metrics, omit_temperature = claude_messages_controls(self.model, requested)
|
|
50
|
+
payload: dict[str, Any] = {
|
|
51
|
+
"model": self.model,
|
|
52
|
+
"max_tokens": 4096,
|
|
53
|
+
"messages": [{"role": "user", "content": prompt}],
|
|
54
|
+
}
|
|
55
|
+
if not omit_temperature:
|
|
56
|
+
payload["temperature"] = temperature
|
|
57
|
+
if system:
|
|
58
|
+
payload["system"] = system
|
|
59
|
+
payload.update(controls)
|
|
60
|
+
value = self._post_json(
|
|
61
|
+
self.base_url + "/v1/messages",
|
|
62
|
+
payload,
|
|
63
|
+
{"x-api-key": self.api_key, "anthropic-version": "2023-06-01"},
|
|
64
|
+
stage=purpose,
|
|
65
|
+
)
|
|
66
|
+
self._set_call_metrics(self._usage_metrics(value, thinking_metrics))
|
|
67
|
+
return self._text(value.get("content"), stage=purpose)
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Gemini generateContent adapter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import urllib.parse
|
|
6
|
+
from typing import Any, Callable, Mapping, Optional
|
|
7
|
+
|
|
8
|
+
from .base import DEFAULT_REQUEST_TIMEOUT, HTTPModelBackend, ModelError
|
|
9
|
+
from .thinking import gemini_generate_controls, requested_thinking_mode
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class GeminiBackend(HTTPModelBackend):
|
|
13
|
+
provider = "gemini"
|
|
14
|
+
|
|
15
|
+
def __init__(
|
|
16
|
+
self,
|
|
17
|
+
*,
|
|
18
|
+
base_url: str,
|
|
19
|
+
api_key: str,
|
|
20
|
+
model: str,
|
|
21
|
+
timeout: float = DEFAULT_REQUEST_TIMEOUT,
|
|
22
|
+
opener: Optional[Callable[..., Any]] = None,
|
|
23
|
+
thinking: Mapping[str, Any] | None = None,
|
|
24
|
+
):
|
|
25
|
+
super().__init__(base_url=base_url, api_key=api_key, model=model, timeout=timeout, opener=opener)
|
|
26
|
+
self.thinking = dict(thinking) if isinstance(thinking, Mapping) else {}
|
|
27
|
+
|
|
28
|
+
@staticmethod
|
|
29
|
+
def _usage_metrics(value: Mapping[str, Any], thinking_metrics: Mapping[str, Any]) -> dict[str, Any]:
|
|
30
|
+
result: dict[str, Any] = dict(thinking_metrics)
|
|
31
|
+
usage = value.get("usageMetadata")
|
|
32
|
+
if not isinstance(usage, Mapping):
|
|
33
|
+
return result
|
|
34
|
+
fields = {
|
|
35
|
+
"promptTokenCount": "prompt_tokens",
|
|
36
|
+
"candidatesTokenCount": "completion_tokens",
|
|
37
|
+
"totalTokenCount": "total_tokens",
|
|
38
|
+
"cachedContentTokenCount": "prompt_cache_hit_tokens",
|
|
39
|
+
"thoughtsTokenCount": "reasoning_tokens",
|
|
40
|
+
}
|
|
41
|
+
for source, target in fields.items():
|
|
42
|
+
item = usage.get(source)
|
|
43
|
+
if isinstance(item, int) and not isinstance(item, bool) and 0 <= item <= 10_000_000:
|
|
44
|
+
result[target] = item
|
|
45
|
+
return result
|
|
46
|
+
|
|
47
|
+
def complete(self, prompt: str, *, system: str = "", purpose: str = "", temperature: float = 0.0) -> str:
|
|
48
|
+
self._set_call_metrics({})
|
|
49
|
+
text = f"{system}\n\n{prompt}" if system else prompt
|
|
50
|
+
requested = requested_thinking_mode(self.thinking, purpose)
|
|
51
|
+
thinking_config, thinking_metrics, omit_temperature = gemini_generate_controls(self.model, requested)
|
|
52
|
+
generation_config: dict[str, Any] = {}
|
|
53
|
+
if not omit_temperature:
|
|
54
|
+
generation_config["temperature"] = temperature
|
|
55
|
+
generation_config.update(thinking_config)
|
|
56
|
+
payload = {
|
|
57
|
+
"contents": [{"role": "user", "parts": [{"text": text}]}],
|
|
58
|
+
"generationConfig": generation_config,
|
|
59
|
+
}
|
|
60
|
+
endpoint = "/v1beta/models/" + urllib.parse.quote(self.model, safe="") + ":generateContent"
|
|
61
|
+
url = self.base_url + endpoint + "?key=" + urllib.parse.quote(self.api_key, safe="")
|
|
62
|
+
value = self._post_json(url, payload, {}, stage=purpose)
|
|
63
|
+
self._set_call_metrics(self._usage_metrics(value, thinking_metrics))
|
|
64
|
+
candidates = value.get("candidates")
|
|
65
|
+
if not isinstance(candidates, list) or not candidates or not isinstance(candidates[0], Mapping):
|
|
66
|
+
raise ModelError("model response has no candidates", code="model_invalid_response", stage=purpose)
|
|
67
|
+
content = candidates[0].get("content")
|
|
68
|
+
if not isinstance(content, Mapping):
|
|
69
|
+
raise ModelError("model response has no content", code="model_invalid_response", stage=purpose)
|
|
70
|
+
parts = content.get("parts")
|
|
71
|
+
if not isinstance(parts, list):
|
|
72
|
+
raise ModelError("model response has no parts", code="model_invalid_response", stage=purpose)
|
|
73
|
+
return self._text(parts, stage=purpose)
|
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
from typing import Any, Callable, Mapping, Optional
|
|
6
6
|
|
|
7
7
|
from .base import DEFAULT_REQUEST_TIMEOUT, HTTPModelBackend, ModelError
|
|
8
|
+
from .thinking import openai_chat_controls, requested_thinking_mode
|
|
8
9
|
|
|
9
10
|
|
|
10
11
|
class OpenAICompatibleBackend(HTTPModelBackend):
|
|
@@ -54,12 +55,8 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
54
55
|
else:
|
|
55
56
|
finish_reason = finish_reason.casefold()
|
|
56
57
|
if finish_reason not in {
|
|
57
|
-
"stop",
|
|
58
|
-
"
|
|
59
|
-
"tool_calls",
|
|
60
|
-
"function_call",
|
|
61
|
-
"content_filter",
|
|
62
|
-
"insufficient_system_resource",
|
|
58
|
+
"stop", "length", "tool_calls", "function_call",
|
|
59
|
+
"content_filter", "insufficient_system_resource",
|
|
63
60
|
}:
|
|
64
61
|
finish_reason = "unknown"
|
|
65
62
|
usage = value.get("usage")
|
|
@@ -85,16 +82,14 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
85
82
|
"reasoning_chars": reasoning_chars,
|
|
86
83
|
}
|
|
87
84
|
|
|
88
|
-
def _thinking_mode(self, purpose: str) -> str:
|
|
89
|
-
if purpose not in {"gate", "summarize", "compact"}:
|
|
90
|
-
return "default"
|
|
91
|
-
value = self.thinking.get(purpose, "low")
|
|
92
|
-
return value if value in {"default", "disabled", "low", "high", "max"} else "low"
|
|
93
|
-
|
|
94
85
|
@staticmethod
|
|
95
|
-
def _usage_metrics(
|
|
86
|
+
def _usage_metrics(
|
|
87
|
+
value: Mapping[str, Any],
|
|
88
|
+
*,
|
|
89
|
+
thinking_metrics: Mapping[str, Any],
|
|
90
|
+
) -> dict[str, Any]:
|
|
96
91
|
usage = value.get("usage")
|
|
97
|
-
result: dict[str, Any] =
|
|
92
|
+
result: dict[str, Any] = dict(thinking_metrics)
|
|
98
93
|
if not isinstance(usage, Mapping):
|
|
99
94
|
return result
|
|
100
95
|
for key in (
|
|
@@ -104,6 +99,10 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
104
99
|
item = usage.get(key)
|
|
105
100
|
if isinstance(item, int) and not isinstance(item, bool) and 0 <= item <= 10_000_000:
|
|
106
101
|
result[key] = item
|
|
102
|
+
prompt_details = usage.get("prompt_tokens_details")
|
|
103
|
+
cached = prompt_details.get("cached_tokens") if isinstance(prompt_details, Mapping) else None
|
|
104
|
+
if "prompt_cache_hit_tokens" not in result and isinstance(cached, int) and not isinstance(cached, bool) and 0 <= cached <= 10_000_000:
|
|
105
|
+
result["prompt_cache_hit_tokens"] = cached
|
|
107
106
|
details = usage.get("completion_tokens_details")
|
|
108
107
|
reasoning = details.get("reasoning_tokens") if isinstance(details, Mapping) else None
|
|
109
108
|
if isinstance(reasoning, int) and not isinstance(reasoning, bool) and 0 <= reasoning <= 10_000_000:
|
|
@@ -116,18 +115,14 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
116
115
|
if system:
|
|
117
116
|
messages.append({"role": "system", "content": system})
|
|
118
117
|
messages.append({"role": "user", "content": prompt})
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
}
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
payload["thinking"] = {"type": "disabled"}
|
|
128
|
-
else:
|
|
129
|
-
payload["thinking"] = {"type": "enabled"}
|
|
130
|
-
payload["reasoning_effort"] = thinking_mode
|
|
118
|
+
requested = requested_thinking_mode(self.thinking, purpose)
|
|
119
|
+
controls, thinking_metrics, omit_temperature = openai_chat_controls(
|
|
120
|
+
self.provider_name, self.model, requested
|
|
121
|
+
)
|
|
122
|
+
payload: dict[str, Any] = {"model": self.model, "messages": messages}
|
|
123
|
+
if not omit_temperature:
|
|
124
|
+
payload["temperature"] = temperature
|
|
125
|
+
payload.update(controls)
|
|
131
126
|
if self.json_mode and purpose in {"gate", "summarize", "compact"}:
|
|
132
127
|
payload["response_format"] = {"type": "json_object"}
|
|
133
128
|
value = self._post_json(
|
|
@@ -136,6 +131,7 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
136
131
|
{"Authorization": f"Bearer {self.api_key}"},
|
|
137
132
|
stage=purpose,
|
|
138
133
|
)
|
|
134
|
+
self._set_call_metrics(self._usage_metrics(value, thinking_metrics=thinking_metrics))
|
|
139
135
|
choices = value.get("choices")
|
|
140
136
|
if not isinstance(choices, list) or not choices or not isinstance(choices[0], Mapping):
|
|
141
137
|
raise ModelError(
|
|
@@ -152,7 +148,6 @@ class OpenAICompatibleBackend(HTTPModelBackend):
|
|
|
152
148
|
stage=purpose,
|
|
153
149
|
validation_reason="response_shape",
|
|
154
150
|
)
|
|
155
|
-
self._set_call_metrics(self._usage_metrics(value, thinking_mode=thinking_mode))
|
|
156
151
|
try:
|
|
157
152
|
return self._text(message.get("content"), stage=purpose)
|
|
158
153
|
except ModelError as error:
|
|
@@ -113,6 +113,7 @@ class ModelRouter:
|
|
|
113
113
|
"api_key": api_key,
|
|
114
114
|
"model": model,
|
|
115
115
|
"timeout": request_timeout,
|
|
116
|
+
"thinking": config.get("thinking") if isinstance(config.get("thinking"), Mapping) else None,
|
|
116
117
|
}
|
|
117
118
|
try:
|
|
118
119
|
if protocol in ("claude", "anthropic") or "claude" in provider or "anthropic" in provider:
|
|
@@ -124,7 +125,6 @@ class ModelRouter:
|
|
|
124
125
|
**kwargs,
|
|
125
126
|
json_mode=provider in _JSON_MODE_PROVIDERS,
|
|
126
127
|
provider_name=provider or "openai",
|
|
127
|
-
thinking=config.get("thinking") if isinstance(config.get("thinking"), Mapping) else None,
|
|
128
128
|
)
|
|
129
129
|
except (ModelError, ValueError, TypeError):
|
|
130
130
|
return None
|
|
@@ -0,0 +1,259 @@
|
|
|
1
|
+
"""Provider-neutral thinking policy and protocol capability mapping."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any, Mapping
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
THINKING_PURPOSES = frozenset({"gate", "summarize", "compact"})
|
|
10
|
+
THINKING_MODES = frozenset({"default", "disabled", "low", "high", "max"})
|
|
11
|
+
THINKING_EFFECTIVE_MODES = frozenset(
|
|
12
|
+
{"provider_default", "unsupported", "disabled", "minimal", "low", "high", "max"}
|
|
13
|
+
)
|
|
14
|
+
THINKING_CONTROLS = frozenset(
|
|
15
|
+
{
|
|
16
|
+
"provider_default",
|
|
17
|
+
"unsupported",
|
|
18
|
+
"openai_reasoning_effort",
|
|
19
|
+
"deepseek_thinking_effort",
|
|
20
|
+
"anthropic_effort",
|
|
21
|
+
"anthropic_adaptive_effort",
|
|
22
|
+
"gemini_thinking_level",
|
|
23
|
+
"gemini_thinking_budget",
|
|
24
|
+
}
|
|
25
|
+
)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def requested_thinking_mode(settings: Any, purpose: str) -> str:
|
|
29
|
+
"""Return memleaf's requested stage policy, defaulting every model stage to low."""
|
|
30
|
+
|
|
31
|
+
if purpose not in THINKING_PURPOSES:
|
|
32
|
+
return "default"
|
|
33
|
+
value = settings.get(purpose, "low") if isinstance(settings, Mapping) else "low"
|
|
34
|
+
return value if isinstance(value, str) and value in THINKING_MODES else "low"
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def thinking_metrics(requested: str, effective: str, control: str) -> dict[str, str]:
|
|
38
|
+
requested = requested if requested in THINKING_MODES else "low"
|
|
39
|
+
effective = effective if effective in THINKING_EFFECTIVE_MODES else "unsupported"
|
|
40
|
+
control = control if control in THINKING_CONTROLS else "unsupported"
|
|
41
|
+
return {
|
|
42
|
+
"thinking_mode": requested,
|
|
43
|
+
"thinking_effective": effective,
|
|
44
|
+
"thinking_control": control,
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _model_leaf(model: Any) -> str:
|
|
49
|
+
if not isinstance(model, str):
|
|
50
|
+
return ""
|
|
51
|
+
return model.casefold().strip().replace("_", "-").rsplit("/", 1)[-1]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _openai_reasoning_model(model: Any) -> bool:
|
|
55
|
+
leaf = _model_leaf(model)
|
|
56
|
+
return leaf.startswith(("gpt-5", "gpt-6", "o1", "o3", "o4", "gpt-oss"))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _openai_none_supported(model: Any) -> bool:
|
|
60
|
+
leaf = _model_leaf(model)
|
|
61
|
+
return leaf.startswith(("gpt-5.1", "gpt-5.2", "gpt-5.3", "gpt-5.4", "gpt-5.5", "gpt-5.6"))
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _openai_max_supported(model: Any) -> bool:
|
|
65
|
+
leaf = _model_leaf(model)
|
|
66
|
+
return leaf.startswith(("gpt-6", "gpt-5.6"))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def openai_chat_controls(
|
|
70
|
+
provider_name: Any,
|
|
71
|
+
model: Any,
|
|
72
|
+
requested: str,
|
|
73
|
+
) -> tuple[dict[str, Any], dict[str, str], bool]:
|
|
74
|
+
"""Map one policy request to an OpenAI-format Chat Completions payload.
|
|
75
|
+
|
|
76
|
+
The final bool says that sampling temperature must be omitted because the
|
|
77
|
+
selected reasoning mode/model does not accept or use it safely.
|
|
78
|
+
"""
|
|
79
|
+
|
|
80
|
+
provider = provider_name.casefold().strip() if isinstance(provider_name, str) else ""
|
|
81
|
+
if requested == "default":
|
|
82
|
+
return {}, thinking_metrics(requested, "provider_default", "provider_default"), False
|
|
83
|
+
|
|
84
|
+
if provider == "deepseek":
|
|
85
|
+
if requested == "disabled":
|
|
86
|
+
return (
|
|
87
|
+
{"thinking": {"type": "disabled"}},
|
|
88
|
+
thinking_metrics(requested, "disabled", "deepseek_thinking_effort"),
|
|
89
|
+
False,
|
|
90
|
+
)
|
|
91
|
+
effort = requested if requested in {"low", "high", "max"} else "low"
|
|
92
|
+
return (
|
|
93
|
+
{"thinking": {"type": "enabled"}, "reasoning_effort": effort},
|
|
94
|
+
thinking_metrics(requested, effort, "deepseek_thinking_effort"),
|
|
95
|
+
True,
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
# Unknown OpenAI-compatible services are deliberately not assumed to
|
|
99
|
+
# accept reasoning_effort merely because they implement Chat Completions.
|
|
100
|
+
if "openai" not in provider or not _openai_reasoning_model(model):
|
|
101
|
+
return {}, thinking_metrics(requested, "unsupported", "unsupported"), False
|
|
102
|
+
|
|
103
|
+
if requested == "disabled":
|
|
104
|
+
if not _openai_none_supported(model):
|
|
105
|
+
return {}, thinking_metrics(requested, "unsupported", "unsupported"), True
|
|
106
|
+
return (
|
|
107
|
+
{"reasoning_effort": "none"},
|
|
108
|
+
thinking_metrics(requested, "disabled", "openai_reasoning_effort"),
|
|
109
|
+
True,
|
|
110
|
+
)
|
|
111
|
+
if requested == "max" and not _openai_max_supported(model):
|
|
112
|
+
return (
|
|
113
|
+
{"reasoning_effort": "high"},
|
|
114
|
+
thinking_metrics(requested, "high", "openai_reasoning_effort"),
|
|
115
|
+
True,
|
|
116
|
+
)
|
|
117
|
+
effort = requested if requested in {"low", "high", "max"} else "low"
|
|
118
|
+
return (
|
|
119
|
+
{"reasoning_effort": effort},
|
|
120
|
+
thinking_metrics(requested, effort, "openai_reasoning_effort"),
|
|
121
|
+
True,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _claude_profile(model: Any) -> str:
|
|
126
|
+
"""Return only capabilities documented for a recognizable Claude family."""
|
|
127
|
+
|
|
128
|
+
leaf = _model_leaf(model)
|
|
129
|
+
if not leaf.startswith("claude-"):
|
|
130
|
+
return "unsupported"
|
|
131
|
+
if any(name in leaf for name in ("fable-5", "mythos-5", "mythos-preview")):
|
|
132
|
+
return "always_adaptive"
|
|
133
|
+
if re.match(r"claude-(?:opus|sonnet)-5(?:-|$)", leaf):
|
|
134
|
+
return "default_adaptive"
|
|
135
|
+
if re.match(r"claude-(?:opus|sonnet)-4-(?:6|7|8|9)(?:-|$)", leaf):
|
|
136
|
+
return "explicit_adaptive"
|
|
137
|
+
if re.match(r"claude-opus-4-5(?:-|$)", leaf):
|
|
138
|
+
return "effort_only"
|
|
139
|
+
return "unsupported"
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def claude_messages_controls(
|
|
143
|
+
model: Any,
|
|
144
|
+
requested: str,
|
|
145
|
+
) -> tuple[dict[str, Any], dict[str, str], bool]:
|
|
146
|
+
"""Map policy effort to current Claude Messages capabilities."""
|
|
147
|
+
|
|
148
|
+
profile = _claude_profile(model)
|
|
149
|
+
if profile == "unsupported":
|
|
150
|
+
return {}, thinking_metrics(requested, "unsupported", "unsupported"), False
|
|
151
|
+
|
|
152
|
+
# Claude 4.7+ and adaptive-thinking requests require the default sampling
|
|
153
|
+
# temperature. Opus 4.5 can use effort without enabling extended thinking.
|
|
154
|
+
omit_temperature = profile != "effort_only"
|
|
155
|
+
if requested == "default":
|
|
156
|
+
return {}, thinking_metrics(requested, "provider_default", "provider_default"), omit_temperature
|
|
157
|
+
|
|
158
|
+
if requested == "disabled":
|
|
159
|
+
if profile == "always_adaptive":
|
|
160
|
+
return (
|
|
161
|
+
{"output_config": {"effort": "low"}},
|
|
162
|
+
thinking_metrics(requested, "low", "anthropic_effort"),
|
|
163
|
+
omit_temperature,
|
|
164
|
+
)
|
|
165
|
+
if profile == "effort_only":
|
|
166
|
+
return (
|
|
167
|
+
{"output_config": {"effort": "low"}},
|
|
168
|
+
thinking_metrics(requested, "disabled", "anthropic_effort"),
|
|
169
|
+
omit_temperature,
|
|
170
|
+
)
|
|
171
|
+
return (
|
|
172
|
+
{"thinking": {"type": "disabled"}, "output_config": {"effort": "low"}},
|
|
173
|
+
thinking_metrics(requested, "disabled", "anthropic_effort"),
|
|
174
|
+
omit_temperature,
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
effort = requested if requested in {"low", "high", "max"} else "low"
|
|
178
|
+
if profile in {"explicit_adaptive", "effort_only"} and effort == "max":
|
|
179
|
+
# Low is the default requested path. For legacy effort-capable families,
|
|
180
|
+
# avoid sending a higher enum unless the current family documents it.
|
|
181
|
+
effort = "high"
|
|
182
|
+
payload: dict[str, Any] = {"output_config": {"effort": effort}}
|
|
183
|
+
control = "anthropic_effort"
|
|
184
|
+
if profile == "explicit_adaptive":
|
|
185
|
+
payload["thinking"] = {"type": "adaptive"}
|
|
186
|
+
control = "anthropic_adaptive_effort"
|
|
187
|
+
return payload, thinking_metrics(requested, effort, control), omit_temperature
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def _gemini_version(model: Any) -> tuple[int, int] | None:
|
|
191
|
+
leaf = _model_leaf(model)
|
|
192
|
+
match = re.match(r"gemini-(\d+)(?:\.(\d+))?", leaf)
|
|
193
|
+
if match is None:
|
|
194
|
+
return None
|
|
195
|
+
return int(match.group(1)), int(match.group(2) or 0)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _gemini3_low_level(model: Any) -> tuple[str, str]:
|
|
199
|
+
"""Return the lowest safe Gemini 3.x level for a recognized exception."""
|
|
200
|
+
|
|
201
|
+
leaf = _model_leaf(model)
|
|
202
|
+
# Gemini 3.1 Flash-Lite Image documents minimal/high but not low. Minimal
|
|
203
|
+
# is lower than the requested low policy and therefore the safe latency
|
|
204
|
+
# preserving fallback instead of an invalid low value.
|
|
205
|
+
if leaf.startswith("gemini-3.1-flash-lite-image"):
|
|
206
|
+
return "minimal", "minimal"
|
|
207
|
+
return "low", "low"
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def gemini_generate_controls(
|
|
211
|
+
model: Any,
|
|
212
|
+
requested: str,
|
|
213
|
+
) -> tuple[dict[str, Any], dict[str, str], bool]:
|
|
214
|
+
"""Map policy effort to native Gemini generateContent thinkingConfig.
|
|
215
|
+
|
|
216
|
+
The final bool asks the adapter to omit explicit temperature for Gemini 3.x,
|
|
217
|
+
following the current API guidance for thinking models.
|
|
218
|
+
"""
|
|
219
|
+
|
|
220
|
+
version = _gemini_version(model)
|
|
221
|
+
if requested == "default":
|
|
222
|
+
omit_temperature = bool(version and version[0] >= 3)
|
|
223
|
+
return {}, thinking_metrics(requested, "provider_default", "provider_default"), omit_temperature
|
|
224
|
+
if version is None:
|
|
225
|
+
return {}, thinking_metrics(requested, "unsupported", "unsupported"), False
|
|
226
|
+
|
|
227
|
+
major, minor = version
|
|
228
|
+
leaf = _model_leaf(model)
|
|
229
|
+
if major >= 3:
|
|
230
|
+
if requested in {"disabled", "low"}:
|
|
231
|
+
level, effective = _gemini3_low_level(model)
|
|
232
|
+
else:
|
|
233
|
+
level, effective = "high", "high"
|
|
234
|
+
return (
|
|
235
|
+
{"thinkingConfig": {"thinkingLevel": level}},
|
|
236
|
+
thinking_metrics(requested, effective, "gemini_thinking_level"),
|
|
237
|
+
True,
|
|
238
|
+
)
|
|
239
|
+
|
|
240
|
+
if major == 2 and minor == 5:
|
|
241
|
+
if requested == "disabled":
|
|
242
|
+
if "pro" in leaf:
|
|
243
|
+
budget = 1024
|
|
244
|
+
effective = "low"
|
|
245
|
+
else:
|
|
246
|
+
budget = 0
|
|
247
|
+
effective = "disabled"
|
|
248
|
+
elif requested == "low":
|
|
249
|
+
budget = 1024
|
|
250
|
+
effective = "low"
|
|
251
|
+
else:
|
|
252
|
+
budget = 24576
|
|
253
|
+
effective = "high"
|
|
254
|
+
return (
|
|
255
|
+
{"thinkingConfig": {"thinkingBudget": budget}},
|
|
256
|
+
thinking_metrics(requested, effective, "gemini_thinking_budget"),
|
|
257
|
+
False,
|
|
258
|
+
)
|
|
259
|
+
return {}, thinking_metrics(requested, "unsupported", "unsupported"), False
|
|
@@ -30,6 +30,12 @@ _PROVIDER_METRIC_FIELDS = (
|
|
|
30
30
|
)
|
|
31
31
|
_METRIC_OPERATION_SUFFIXES = ("primary", "format_repair")
|
|
32
32
|
_MAX_METRIC_CALLS = 256
|
|
33
|
+
_THINKING_EFFECTIVE_MODES = frozenset({"provider_default", "unsupported", "disabled", "minimal", "low", "high", "max"})
|
|
34
|
+
_THINKING_CONTROLS = frozenset({
|
|
35
|
+
"provider_default", "unsupported", "openai_reasoning_effort",
|
|
36
|
+
"deepseek_thinking_effort", "anthropic_effort", "anthropic_adaptive_effort",
|
|
37
|
+
"gemini_thinking_level", "gemini_thinking_budget",
|
|
38
|
+
})
|
|
33
39
|
|
|
34
40
|
|
|
35
41
|
def _metric_bucket() -> dict[str, Any]:
|
|
@@ -124,6 +130,12 @@ class ModelExecutor:
|
|
|
124
130
|
mode = value.get("thinking_mode")
|
|
125
131
|
if mode in {"default", "disabled", "low", "high", "max"}:
|
|
126
132
|
result["thinking_mode"] = mode
|
|
133
|
+
effective = value.get("thinking_effective")
|
|
134
|
+
if effective in _THINKING_EFFECTIVE_MODES:
|
|
135
|
+
result["thinking_effective"] = effective
|
|
136
|
+
control = value.get("thinking_control")
|
|
137
|
+
if control in _THINKING_CONTROLS:
|
|
138
|
+
result["thinking_control"] = control
|
|
127
139
|
return result
|
|
128
140
|
|
|
129
141
|
@staticmethod
|
|
@@ -69,6 +69,12 @@ _MODEL_CALL_INT_FIELDS = frozenset({
|
|
|
69
69
|
"reasoning_tokens",
|
|
70
70
|
})
|
|
71
71
|
_MAX_MODEL_CALL_ROWS = 256
|
|
72
|
+
_THINKING_EFFECTIVE_MODES = frozenset({"provider_default", "unsupported", "disabled", "minimal", "low", "high", "max"})
|
|
73
|
+
_THINKING_CONTROLS = frozenset({
|
|
74
|
+
"provider_default", "unsupported", "openai_reasoning_effort",
|
|
75
|
+
"deepseek_thinking_effort", "anthropic_effort", "anthropic_adaptive_effort",
|
|
76
|
+
"gemini_thinking_level", "gemini_thinking_budget",
|
|
77
|
+
})
|
|
72
78
|
|
|
73
79
|
|
|
74
80
|
def _now() -> str:
|
|
@@ -232,6 +238,12 @@ def _safe_model_metrics(value: Any) -> dict[str, Any]:
|
|
|
232
238
|
mode = raw.get("thinking_mode")
|
|
233
239
|
if mode in {"default", "disabled", "low", "high", "max"}:
|
|
234
240
|
row["thinking_mode"] = mode
|
|
241
|
+
effective = raw.get("thinking_effective")
|
|
242
|
+
if effective in _THINKING_EFFECTIVE_MODES:
|
|
243
|
+
row["thinking_effective"] = effective
|
|
244
|
+
control = raw.get("thinking_control")
|
|
245
|
+
if control in _THINKING_CONTROLS:
|
|
246
|
+
row["thinking_control"] = control
|
|
235
247
|
bounded_calls.append(row)
|
|
236
248
|
if bounded_calls:
|
|
237
249
|
result["calls"] = bounded_calls
|