dflash-console 0.3.232__tar.gz → 0.3.240__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {dflash_console-0.3.232 → dflash_console-0.3.240}/.gitignore +1 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/PKG-INFO +5 -3
- {dflash_console-0.3.232 → dflash_console-0.3.240}/README.md +4 -2
- {dflash_console-0.3.232 → dflash_console-0.3.240}/api/app.py +111 -9
- {dflash_console-0.3.232 → dflash_console-0.3.240}/api/gateway.py +79 -5
- dflash_console-0.3.240/core/chat_queue.py +55 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/config.py +5 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/embedding_server.py +4 -2
- dflash_console-0.3.240/core/gateway_access_log.py +39 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gateway_routing.py +16 -1
- dflash_console-0.3.240/core/hf_catalog_draft.py +100 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_catalog_index.py +20 -2
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_catalog_recommend.py +10 -1
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_local_match.py +10 -2
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_model_fit.py +16 -2
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/huggingface.py +182 -27
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/local_models.py +12 -10
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_presets.py +13 -2
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/server_boot.py +159 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/version.py +1 -1
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/vision_setup.py +91 -12
- {dflash_console-0.3.232 → dflash_console-0.3.240}/pyproject.toml +1 -1
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-shell.css +9753 -9718
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/index.html +3 -3
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/chat-live.js +8 -3
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/dflash-console-ui.js +1 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/download-queue.js +988 -945
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/downloads-live.js +100 -14
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/model-search-live.js +130 -19
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/models-live.js +6 -2
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/server-live.js +5659 -5585
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/settings-live.js +11 -2
- {dflash_console-0.3.232 → dflash_console-0.3.240}/LICENSE +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/NOTICE.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/api/__init__.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo.png +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo_only.png +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo_only2.png +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo_only_clear.png +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_txt_logo.png +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/config.example.json +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/__init__.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_access_log.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_catalog.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_introspection.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_providers.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/auto_register.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/catalog_load.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/chat_proxy.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/chat_ready.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/chat_vision.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/client_identity.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/components_hub.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/dflash_generation.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/diagnostics_bundle.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/display_names.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/engine_state.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/freetoken_runtime_install.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/fs_browse.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/fs_reveal.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gguf_meta.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_devices.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_policy.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_process_memory_windows.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_processes.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_relief.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hardware_apply.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hardware_info.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_catalog_cache.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_engines.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_install.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/inference_stats.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/library_import.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/load_progress.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/log_utils.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/memory_guardrails.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_discovery.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_paths.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_runtime_policy.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_stack.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/net_listeners.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/node_connect.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/ocr_setup.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/remote_nodes.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtime.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtime_install_job.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtime_recommendations.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/__init__.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/base.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/contention.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/faster_whisper.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/freetoken.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/noop.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/ollama.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/piper.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/registry.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/stt.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/transformers_hf.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/vibevoice.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/vllm.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/setup.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/stack_match.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/status_report.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/support_journal.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/system_stats.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/transformers_runtime_install.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/core/vllm_runtime_install.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/__init__.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/__main__.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/cli.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/commands.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/http.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/render.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/resolve.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/server_takeover.py +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/AGENT-PROMPT-CLIENT-IDENTITY.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/AGENT-PROMPT-DFLASH-TRANSLATION-PERF-AI-TOOLS.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/AGENT-PROMPT-TRANSLATEGEMMA-AI-TOOLS.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ARCHITECTURE.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/BUG-REPORTS.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/CLI.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/CLIENT-IDENTITY.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/CURSOR.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/GOING-PUBLIC.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/LICENSING.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/LINUX-CLI.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/MULTI-MODAL-AGENT-PROMPT.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/MULTI-MODAL-RUNTIME-PLAN.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/PRODUCTION.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/RELEASING.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/STT-ENGINE-DECISION.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/USER-GUIDE.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/announcements/public-preview-v0.3.103.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/announcements/public-preview-v0.3.122.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/announcements/public-preview.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/automatic-updates.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/plans/shared-speak-stt-service.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.103.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.106.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.117.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.118.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.119.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.120.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.121.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.122.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.123.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.128.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.129.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.132.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.133.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.208.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/README.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/devices-view.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/model-search.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/models-view.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/nodes-v1-plan.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-appearance.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-chat.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-developer.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-general.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-hardware.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-integrations.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-lm-link.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-model-defaults.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-runtime.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-runtimes.md +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/app-functional.html +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-console.css +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-console.css.bak-density +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-shell.css.bak-density +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/studio.css +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/about-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/app-settings-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/app.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/bug-report-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/component-install-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/console-api.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/dashboard.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/desktop-shell.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/docs-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/library-browse-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/library-scan-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/library-type-options.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/log-format.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/model-card.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/model-catalog-groups.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/nodes-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/runtime-recommendations.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/runtime-steppers.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/select-theme.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/server-reload-watch.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/settings-live.js.api-providers-bak +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/settings-live.js.pre-api-merge-bak +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/setup-wizard-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/speak-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/stack-wizard.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/status-feed.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/sysbar-live.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/table-column-resize.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/ui-layout-prefs.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/vendor/marked.min.js +0 -0
- {dflash_console-0.3.232 → dflash_console-0.3.240}/static/vendor/purify.min.js +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: dflash-console
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.240
|
|
4
4
|
Summary: Local control panel and terminal CLI for DFlash stacks and local model runtimes
|
|
5
5
|
Project-URL: Homepage, https://github.com/ilan4ever/Dflash-Console
|
|
6
6
|
Project-URL: Repository, https://github.com/ilan4ever/Dflash-Console
|
|
@@ -674,7 +674,7 @@ from one UI, then talk to them through a single OpenAI-compatible port.
|
|
|
674
674
|
> supported platform**; Linux and macOS are not supported or tested (see
|
|
675
675
|
> [Platform support](#platform-support) below).
|
|
676
676
|
|
|
677
|
-
**Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.
|
|
677
|
+
**Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.240
|
|
678
678
|
|
|
679
679
|
## Download (Windows)
|
|
680
680
|
|
|
@@ -753,7 +753,7 @@ Typical first session:
|
|
|
753
753
|
3. For a DFlash GGUF, right-click and **Find and attach draft** if you want speculative decoding.
|
|
754
754
|
4. Chat in the **Playground**, or point any OpenAI client at `http://127.0.0.1:8001/v1` with optional `X-DFlash-Client: YourApp` so **Engines** shows who is using each model.
|
|
755
755
|
|
|
756
|
-
### Recent improvements (v0.3.
|
|
756
|
+
### Recent improvements (v0.3.240)
|
|
757
757
|
|
|
758
758
|
- **Engine standby** — Running toggle gates load/chat until you arm the pipeline
|
|
759
759
|
- **External GPU cards** — OneVoice, LM Studio, and other apps on the GPU; compact mobile layout; loading state expires when models are ready
|
|
@@ -797,6 +797,8 @@ http://127.0.0.1:8001/v1
|
|
|
797
797
|
```
|
|
798
798
|
|
|
799
799
|
The gateway routes chat, embeddings, TTS, and STT to the loaded engine.
|
|
800
|
+
|
|
801
|
+
**Concurrent chat:** multiple OpenAI-compatible clients may call `chat/completions` on the same loaded model at once (Harness agent turn + title, or two UI actions). The Console serializes ready/JIT per engine, then lets llama-server use parallel slots. You should not see `upstream HTTP 409` for that steady-state overlap. HTTP 409 is still used for strict model mismatch (`X-DFlash-Strict-Model`), a checkpoint already loaded on a *different* engine, and stack/vision repair.
|
|
800
802
|
Model names are tolerant (engine id, file name, or an alias such as `gpt-4o`).
|
|
801
803
|
|
|
802
804
|
**Client identity:** send `X-DFlash-Client: YourApp` on load and chat requests so the
|
|
@@ -8,7 +8,7 @@ from one UI, then talk to them through a single OpenAI-compatible port.
|
|
|
8
8
|
> supported platform**; Linux and macOS are not supported or tested (see
|
|
9
9
|
> [Platform support](#platform-support) below).
|
|
10
10
|
|
|
11
|
-
**Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.
|
|
11
|
+
**Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.240
|
|
12
12
|
|
|
13
13
|
## Download (Windows)
|
|
14
14
|
|
|
@@ -87,7 +87,7 @@ Typical first session:
|
|
|
87
87
|
3. For a DFlash GGUF, right-click and **Find and attach draft** if you want speculative decoding.
|
|
88
88
|
4. Chat in the **Playground**, or point any OpenAI client at `http://127.0.0.1:8001/v1` with optional `X-DFlash-Client: YourApp` so **Engines** shows who is using each model.
|
|
89
89
|
|
|
90
|
-
### Recent improvements (v0.3.
|
|
90
|
+
### Recent improvements (v0.3.240)
|
|
91
91
|
|
|
92
92
|
- **Engine standby** — Running toggle gates load/chat until you arm the pipeline
|
|
93
93
|
- **External GPU cards** — OneVoice, LM Studio, and other apps on the GPU; compact mobile layout; loading state expires when models are ready
|
|
@@ -131,6 +131,8 @@ http://127.0.0.1:8001/v1
|
|
|
131
131
|
```
|
|
132
132
|
|
|
133
133
|
The gateway routes chat, embeddings, TTS, and STT to the loaded engine.
|
|
134
|
+
|
|
135
|
+
**Concurrent chat:** multiple OpenAI-compatible clients may call `chat/completions` on the same loaded model at once (Harness agent turn + title, or two UI actions). The Console serializes ready/JIT per engine, then lets llama-server use parallel slots. You should not see `upstream HTTP 409` for that steady-state overlap. HTTP 409 is still used for strict model mismatch (`X-DFlash-Strict-Model`), a checkpoint already loaded on a *different* engine, and stack/vision repair.
|
|
134
136
|
Model names are tolerant (engine id, file name, or an alias such as `gpt-4o`).
|
|
135
137
|
|
|
136
138
|
**Client identity:** send `X-DFlash-Client: YourApp` on load and chat requests so the
|
|
@@ -3067,10 +3067,57 @@ def _ensure_server_ready_for_chat(
|
|
|
3067
3067
|
},
|
|
3068
3068
|
)
|
|
3069
3069
|
|
|
3070
|
+
from core.inference_stats import is_proxy_generating
|
|
3071
|
+
|
|
3072
|
+
def _status_as_loaded(status: dict[str, Any]) -> dict[str, Any]:
|
|
3073
|
+
if status.get('status') != 'loaded':
|
|
3074
|
+
status = {**status, 'status': 'loaded'}
|
|
3075
|
+
return status
|
|
3076
|
+
|
|
3077
|
+
def _synthetic_loaded_from_config() -> dict[str, Any]:
|
|
3078
|
+
"""When a probe fails under concurrent load, trust the configured id."""
|
|
3079
|
+
model_id = str(server.get('model_id') or '').strip()
|
|
3080
|
+
loaded = [model_id] if model_id else []
|
|
3081
|
+
return {
|
|
3082
|
+
**server,
|
|
3083
|
+
'running': True,
|
|
3084
|
+
'status': 'loaded' if loaded else 'running',
|
|
3085
|
+
'booting': False,
|
|
3086
|
+
'loaded_models': loaded,
|
|
3087
|
+
'active_model_id': loaded[0] if loaded else '',
|
|
3088
|
+
'ready_for_chat': bool(loaded),
|
|
3089
|
+
}
|
|
3090
|
+
|
|
3070
3091
|
live = build_server_status(server, cfg=cfg)
|
|
3092
|
+
|
|
3093
|
+
# Another chat is already mid-flight on this engine: never JIT-load (that
|
|
3094
|
+
# path raises model_already_loaded_elsewhere / stack-repair 409s and can
|
|
3095
|
+
# stop_server under DFlash draft profiles). Share the live or configured id.
|
|
3096
|
+
if is_proxy_generating(server_id):
|
|
3097
|
+
if not live.get('loaded_models'):
|
|
3098
|
+
live = _synthetic_loaded_from_config()
|
|
3099
|
+
live = _status_as_loaded(live)
|
|
3100
|
+
if required_context and cfg.get('context_auto_grow') is not False:
|
|
3101
|
+
loaded_ctx = _loaded_per_slot_context(server)
|
|
3102
|
+
if loaded_ctx and required_context > loaded_ctx:
|
|
3103
|
+
# Growing reloads the engine — wait until the active turn ends.
|
|
3104
|
+
deadline = time.time() + 180.0
|
|
3105
|
+
while time.time() < deadline and is_proxy_generating(server_id):
|
|
3106
|
+
time.sleep(0.05)
|
|
3107
|
+
if not is_proxy_generating(server_id):
|
|
3108
|
+
return _ensure_server_ready_for_chat(
|
|
3109
|
+
server_id,
|
|
3110
|
+
server,
|
|
3111
|
+
cfg,
|
|
3112
|
+
client_label=client_label,
|
|
3113
|
+
required_context=required_context,
|
|
3114
|
+
)
|
|
3115
|
+
# Still busy after wait: serve with current context rather than 409.
|
|
3116
|
+
note_engine_active_client(server_id, client_label=client_label)
|
|
3117
|
+
return live
|
|
3118
|
+
|
|
3071
3119
|
if live.get('loaded_models'):
|
|
3072
|
-
|
|
3073
|
-
live = {**live, 'status': 'loaded'}
|
|
3120
|
+
live = _status_as_loaded(live)
|
|
3074
3121
|
# Already loaded. Auto-grow if this request needs more context than
|
|
3075
3122
|
# the loaded model provides; otherwise share the model as-is.
|
|
3076
3123
|
if required_context and cfg.get('context_auto_grow') is not False:
|
|
@@ -3080,6 +3127,23 @@ def _ensure_server_ready_for_chat(
|
|
|
3080
3127
|
note_engine_active_client(server_id, client_label=client_label)
|
|
3081
3128
|
return live
|
|
3082
3129
|
|
|
3130
|
+
# Probe can return empty loaded_models while llama is busy serving another
|
|
3131
|
+
# client. Retry briefly before treating the engine as idle for JIT load.
|
|
3132
|
+
host = str(server.get('host') or '127.0.0.1').strip() or '127.0.0.1'
|
|
3133
|
+
port = int(server.get('port') or 0)
|
|
3134
|
+
if port > 0 and tcp_port_open(host, port):
|
|
3135
|
+
for _ in range(6):
|
|
3136
|
+
time.sleep(0.05)
|
|
3137
|
+
live = build_server_status(server, cfg=cfg)
|
|
3138
|
+
if live.get('loaded_models') or is_proxy_generating(server_id):
|
|
3139
|
+
break
|
|
3140
|
+
if is_proxy_generating(server_id) and not live.get('loaded_models'):
|
|
3141
|
+
live = _synthetic_loaded_from_config()
|
|
3142
|
+
if live.get('loaded_models'):
|
|
3143
|
+
live = _status_as_loaded(live)
|
|
3144
|
+
note_engine_active_client(server_id, client_label=client_label)
|
|
3145
|
+
return live
|
|
3146
|
+
|
|
3083
3147
|
if live.get('status') == 'booting':
|
|
3084
3148
|
return _wait_until_loaded()
|
|
3085
3149
|
|
|
@@ -3421,6 +3485,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
|
|
|
3421
3485
|
import json
|
|
3422
3486
|
import urllib.error
|
|
3423
3487
|
|
|
3488
|
+
_chat_gate = None
|
|
3489
|
+
_chat_gate_held = False
|
|
3490
|
+
|
|
3424
3491
|
from core.chat_proxy import (
|
|
3425
3492
|
apply_reasoning_policy,
|
|
3426
3493
|
chat_upstream_read_timeout,
|
|
@@ -3481,6 +3548,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
|
|
|
3481
3548
|
body_json = None
|
|
3482
3549
|
required_context = 0
|
|
3483
3550
|
else:
|
|
3551
|
+
server = _require_server(cfg, server_id)
|
|
3552
|
+
_arm_server_engine_for_api(server_id, cfg)
|
|
3553
|
+
cfg = load_config()
|
|
3484
3554
|
server = _require_server(cfg, server_id)
|
|
3485
3555
|
raw = await request.body()
|
|
3486
3556
|
try:
|
|
@@ -3528,17 +3598,31 @@ async def proxy_chat_completions(server_id: str, request: Request):
|
|
|
3528
3598
|
header_context = request_load_context_size(request) or 0
|
|
3529
3599
|
body_context = chat_body_load_context_size(body_json) or 0
|
|
3530
3600
|
required_context = max(estimated_context or 0, header_context or 0, body_context or 0)
|
|
3531
|
-
|
|
3532
|
-
|
|
3533
|
-
|
|
3534
|
-
|
|
3535
|
-
|
|
3536
|
-
|
|
3537
|
-
|
|
3601
|
+
from core.chat_queue import chat_server_gate
|
|
3602
|
+
|
|
3603
|
+
_chat_gate = chat_server_gate(server_id)
|
|
3604
|
+
await _chat_gate.acquire()
|
|
3605
|
+
_chat_gate_held = True
|
|
3606
|
+
try:
|
|
3607
|
+
live = _ensure_server_ready_for_chat(
|
|
3608
|
+
server_id,
|
|
3609
|
+
server,
|
|
3610
|
+
cfg,
|
|
3611
|
+
client_label=_request_client_label(request),
|
|
3612
|
+
required_context=required_context or None,
|
|
3613
|
+
)
|
|
3614
|
+
except Exception:
|
|
3615
|
+
if _chat_gate_held:
|
|
3616
|
+
_chat_gate.release()
|
|
3617
|
+
_chat_gate_held = False
|
|
3618
|
+
raise
|
|
3538
3619
|
|
|
3539
3620
|
api_url = str(server.get('api_url') or '')
|
|
3540
3621
|
base = api_base_url(api_url)
|
|
3541
3622
|
if not base:
|
|
3623
|
+
if _chat_gate_held and _chat_gate is not None:
|
|
3624
|
+
_chat_gate.release()
|
|
3625
|
+
_chat_gate_held = False
|
|
3542
3626
|
raise HTTPException(status_code=400, detail='engine api_url not configured')
|
|
3543
3627
|
# Non-reasoning models never negotiate reasoning: strip reasoning_effort and
|
|
3544
3628
|
# thinking toggles so the API returns the regular chat behaviour.
|
|
@@ -3551,6 +3635,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
|
|
|
3551
3635
|
attributed_client=attributed,
|
|
3552
3636
|
)
|
|
3553
3637
|
if reasoning_error:
|
|
3638
|
+
if _chat_gate_held and _chat_gate is not None:
|
|
3639
|
+
_chat_gate.release()
|
|
3640
|
+
_chat_gate_held = False
|
|
3554
3641
|
raise HTTPException(
|
|
3555
3642
|
status_code=400,
|
|
3556
3643
|
detail={
|
|
@@ -3595,6 +3682,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
|
|
|
3595
3682
|
and requested_model
|
|
3596
3683
|
and not model_ids_compatible(requested_model, upstream_model_id)
|
|
3597
3684
|
):
|
|
3685
|
+
if _chat_gate_held and _chat_gate is not None:
|
|
3686
|
+
_chat_gate.release()
|
|
3687
|
+
_chat_gate_held = False
|
|
3598
3688
|
raise HTTPException(
|
|
3599
3689
|
status_code=409,
|
|
3600
3690
|
detail={
|
|
@@ -3640,11 +3730,20 @@ async def proxy_chat_completions(server_id: str, request: Request):
|
|
|
3640
3730
|
)
|
|
3641
3731
|
except urllib.error.HTTPError as exc:
|
|
3642
3732
|
mark_inference_end(server_id, client_label=client_label)
|
|
3733
|
+
if _chat_gate_held and _chat_gate is not None:
|
|
3734
|
+
_chat_gate.release()
|
|
3735
|
+
_chat_gate_held = False
|
|
3643
3736
|
detail = exc.read().decode('utf-8', errors='replace')
|
|
3644
3737
|
raise HTTPException(status_code=exc.code, detail=detail) from exc
|
|
3645
3738
|
except Exception as exc:
|
|
3646
3739
|
mark_inference_end(server_id, client_label=client_label)
|
|
3740
|
+
if _chat_gate_held and _chat_gate is not None:
|
|
3741
|
+
_chat_gate.release()
|
|
3742
|
+
_chat_gate_held = False
|
|
3647
3743
|
raise HTTPException(status_code=502, detail=str(exc)) from exc
|
|
3744
|
+
if _chat_gate_held and _chat_gate is not None:
|
|
3745
|
+
_chat_gate.release()
|
|
3746
|
+
_chat_gate_held = False
|
|
3648
3747
|
|
|
3649
3748
|
keepalive_interval = 15.0
|
|
3650
3749
|
|
|
@@ -3717,6 +3816,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
|
|
|
3717
3816
|
model_id=str(live.get('active_model_id') or server.get('model_id') or ''),
|
|
3718
3817
|
client_label=client_label,
|
|
3719
3818
|
)
|
|
3819
|
+
if _chat_gate_held and _chat_gate is not None:
|
|
3820
|
+
_chat_gate.release()
|
|
3821
|
+
_chat_gate_held = False
|
|
3720
3822
|
active_model = str(live.get('active_model_id') or server.get('model_id') or '')
|
|
3721
3823
|
disconnect_task = asyncio.create_task(
|
|
3722
3824
|
_abort_upstream_when_client_disconnects(
|
|
@@ -15,6 +15,13 @@ Routes:
|
|
|
15
15
|
GET /health gateway + console health
|
|
16
16
|
GET / info
|
|
17
17
|
|
|
18
|
+
Concurrent OpenAI clients on the same loaded engine (for example DeepSeek Harness
|
|
19
|
+
main turn + session title) are accepted: the Console queues the ready/start
|
|
20
|
+
section per ``server_id``, then llama-server multiplexes within ``parallel_slots``.
|
|
21
|
+
Steady-state overlaps must not return HTTP 409. Intentional 409s remain for
|
|
22
|
+
``X-DFlash-Strict-Model`` mismatches, duplicate checkpoint on another engine,
|
|
23
|
+
DFlash stack / vision repair, and load/unload conflicts.
|
|
24
|
+
|
|
18
25
|
The default chat engine is ``config.json -> gateway_server_id`` (falls back to
|
|
19
26
|
the first enabled non-embedding engine); embeddings route to the first enabled
|
|
20
27
|
embedding engine. The gateway is started on the UI process lifespan, so it is
|
|
@@ -41,6 +48,7 @@ from core.gateway_routing import (
|
|
|
41
48
|
enabled_chat_servers,
|
|
42
49
|
gateway_model_aliases,
|
|
43
50
|
gateway_public_model_ids,
|
|
51
|
+
is_cursor_compat_model_id,
|
|
44
52
|
resolve_chat_server,
|
|
45
53
|
_advertised_engine_server,
|
|
46
54
|
)
|
|
@@ -51,6 +59,11 @@ logger = logging.getLogger(__name__)
|
|
|
51
59
|
|
|
52
60
|
gateway_app = FastAPI(title='DFlash Console OpenAI Gateway', version='0.1.0')
|
|
53
61
|
|
|
62
|
+
|
|
63
|
+
def gateway_cloud_chat_enabled(cfg: dict[str, Any]) -> bool:
|
|
64
|
+
"""Allow active Settings cloud models unless explicitly disabled."""
|
|
65
|
+
return cfg.get('gateway_cloud_chat_enabled') is not False
|
|
66
|
+
|
|
54
67
|
_FORWARD_HEADERS = {
|
|
55
68
|
'content-type',
|
|
56
69
|
'accept',
|
|
@@ -383,6 +396,9 @@ async def list_models() -> dict[str, Any]:
|
|
|
383
396
|
pass
|
|
384
397
|
from core.api_providers import cloud_model_entries, with_api_label
|
|
385
398
|
|
|
399
|
+
if not gateway_cloud_chat_enabled(cfg):
|
|
400
|
+
return {'object': 'list', 'data': data}
|
|
401
|
+
|
|
386
402
|
for entry in cloud_model_entries(cfg):
|
|
387
403
|
mid = str(entry.get('id') or '').strip()
|
|
388
404
|
if not mid or mid.lower() in listed_ids:
|
|
@@ -432,14 +448,29 @@ async def _forward_chat(
|
|
|
432
448
|
try:
|
|
433
449
|
async with httpx.AsyncClient(timeout=None) as client:
|
|
434
450
|
async with client.stream('POST', url, content=body, headers=headers) as upstream:
|
|
435
|
-
upstream.
|
|
451
|
+
if upstream.status_code >= 400:
|
|
452
|
+
raw_err = await upstream.aread()
|
|
453
|
+
# Re-raise as HTTPStatusError with body already buffered.
|
|
454
|
+
response = httpx.Response(
|
|
455
|
+
upstream.status_code,
|
|
456
|
+
content=raw_err,
|
|
457
|
+
request=upstream.request,
|
|
458
|
+
headers=upstream.headers,
|
|
459
|
+
)
|
|
460
|
+
raise httpx.HTTPStatusError(
|
|
461
|
+
f'Client error {upstream.status_code}',
|
|
462
|
+
request=upstream.request,
|
|
463
|
+
response=response,
|
|
464
|
+
)
|
|
436
465
|
async for chunk in upstream.aiter_bytes():
|
|
437
466
|
yield chunk
|
|
438
467
|
except httpx.HTTPStatusError as exc:
|
|
439
468
|
status = int(exc.response.status_code or 500)
|
|
440
469
|
logger.warning('gateway chat upstream HTTP %s for %s', status, url)
|
|
441
470
|
try:
|
|
442
|
-
raw =
|
|
471
|
+
raw = exc.response.content or b''
|
|
472
|
+
if not raw:
|
|
473
|
+
raw = await exc.response.aread()
|
|
443
474
|
except Exception:
|
|
444
475
|
raw = b''
|
|
445
476
|
try: # DFLASH_LOG_400_BODIES (stream path)
|
|
@@ -569,20 +600,58 @@ async def chat_completions(request: Request) -> Response:
|
|
|
569
600
|
if not isinstance(model, str):
|
|
570
601
|
model = ''
|
|
571
602
|
from core.api_providers import resolve_cloud_provider_for_model
|
|
572
|
-
|
|
573
|
-
|
|
603
|
+
from core.client_identity import resolve_client_label
|
|
604
|
+
from core.gateway_access_log import record_gateway_route
|
|
605
|
+
|
|
606
|
+
# A cloud model selected as the gateway default must also handle clients
|
|
607
|
+
# that omit ``model`` or send a Cursor-compatible placeholder.
|
|
608
|
+
configured_model = str(cfg.get('gateway_server_id') or '').strip()
|
|
609
|
+
if (
|
|
610
|
+
gateway_cloud_chat_enabled(cfg)
|
|
611
|
+
and configured_model
|
|
612
|
+
and (not model or is_cursor_compat_model_id(model))
|
|
613
|
+
and resolve_cloud_provider_for_model(cfg, configured_model) is not None
|
|
614
|
+
):
|
|
615
|
+
model = configured_model
|
|
616
|
+
|
|
617
|
+
client_label = resolve_client_label(request)
|
|
618
|
+
cloud_provider = (
|
|
619
|
+
resolve_cloud_provider_for_model(cfg, model)
|
|
620
|
+
if gateway_cloud_chat_enabled(cfg)
|
|
621
|
+
else None
|
|
622
|
+
)
|
|
574
623
|
if cloud_provider is not None:
|
|
624
|
+
provider_id = str(cloud_provider.get('id') or 'cloud')
|
|
625
|
+
record_gateway_route(
|
|
626
|
+
model=model,
|
|
627
|
+
route='cloud',
|
|
628
|
+
target=provider_id,
|
|
629
|
+
client=client_label,
|
|
630
|
+
note=str(cloud_provider.get('base_url') or ''),
|
|
631
|
+
)
|
|
575
632
|
body: bytes | None = None
|
|
576
633
|
if isinstance(payload, dict):
|
|
634
|
+
payload['model'] = model
|
|
577
635
|
try:
|
|
578
636
|
body = json.dumps(payload).encode('utf-8')
|
|
579
637
|
except Exception:
|
|
580
638
|
body = None
|
|
581
639
|
if body is None:
|
|
582
640
|
body = await request.body()
|
|
583
|
-
|
|
641
|
+
response = await _forward_cloud_chat(request, provider=cloud_provider, body=body)
|
|
642
|
+
if isinstance(response, Response):
|
|
643
|
+
response.headers['X-DFlash-Route'] = 'cloud'
|
|
644
|
+
response.headers['X-DFlash-Provider-Id'] = provider_id
|
|
645
|
+
return response
|
|
584
646
|
server, upstream = await _resolve_chat_target(cfg, model)
|
|
585
647
|
sid = str(server.get('id') or '')
|
|
648
|
+
record_gateway_route(
|
|
649
|
+
model=model,
|
|
650
|
+
route='local',
|
|
651
|
+
target=sid or str(server.get('label') or ''),
|
|
652
|
+
client=client_label,
|
|
653
|
+
note=str(upstream or ''),
|
|
654
|
+
)
|
|
586
655
|
if sid:
|
|
587
656
|
from core.engine_state import note_engine_on
|
|
588
657
|
|
|
@@ -603,6 +672,10 @@ async def chat_completions(request: Request) -> Response:
|
|
|
603
672
|
reasoning_model = model_has_reasoning(server)
|
|
604
673
|
client_label = resolve_client_label(request)
|
|
605
674
|
attributed = bool(client_label) and client_label != LABEL_UNKNOWN_API
|
|
675
|
+
# External OpenAI clients (Cursor, SDKs) use the gateway without X-DFlash-Client.
|
|
676
|
+
# Treat them like attributed clients so low max_tokens disables reasoning instead of 400.
|
|
677
|
+
if not attributed:
|
|
678
|
+
attributed = True
|
|
606
679
|
disable_reasoning, reasoning_error = resolve_disable_reasoning_for_chat(
|
|
607
680
|
body,
|
|
608
681
|
reasoning=reasoning_model,
|
|
@@ -641,6 +714,7 @@ async def chat_completions(request: Request) -> Response:
|
|
|
641
714
|
filter_reasoning = disable_reasoning
|
|
642
715
|
response = await _forward_chat(request, url, body, filter_reasoning=filter_reasoning)
|
|
643
716
|
if isinstance(response, Response):
|
|
717
|
+
response.headers['X-DFlash-Route'] = 'local'
|
|
644
718
|
response.headers['X-DFlash-Server-Id'] = sid
|
|
645
719
|
return response
|
|
646
720
|
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Per-engine chat gates so concurrent OpenAI clients share one loaded model safely.
|
|
2
|
+
|
|
3
|
+
Steady-state overlapping ``/v1/chat/completions`` (e.g. DeepSeek Harness main turn +
|
|
4
|
+
session title) must not re-enter JIT load and raise HTTP 409. Callers hold the
|
|
5
|
+
gate through ready-check + ``mark_inference_start`` (and opening an upstream
|
|
6
|
+
stream); llama-server then multiplexes within ``parallel_slots``.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import asyncio
|
|
12
|
+
import threading
|
|
13
|
+
from collections.abc import AsyncIterator
|
|
14
|
+
from contextlib import asynccontextmanager
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
_GATE_GUARD = threading.Lock()
|
|
18
|
+
_GATES: dict[str, asyncio.Lock] = {}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def chat_server_gate(server_id: str) -> asyncio.Lock:
|
|
22
|
+
"""Return the asyncio lock for this engine id (created on first use)."""
|
|
23
|
+
sid = str(server_id or '').strip() or '_'
|
|
24
|
+
with _GATE_GUARD:
|
|
25
|
+
lock = _GATES.get(sid)
|
|
26
|
+
if lock is None:
|
|
27
|
+
lock = asyncio.Lock()
|
|
28
|
+
_GATES[sid] = lock
|
|
29
|
+
return lock
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
@asynccontextmanager
|
|
33
|
+
async def hold_chat_server_gate(server_id: str) -> AsyncIterator[None]:
|
|
34
|
+
"""Acquire the per-engine chat gate for the critical ready/start section."""
|
|
35
|
+
lock = chat_server_gate(server_id)
|
|
36
|
+
await lock.acquire()
|
|
37
|
+
try:
|
|
38
|
+
yield
|
|
39
|
+
finally:
|
|
40
|
+
lock.release()
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def reset_chat_server_gates_for_tests() -> None:
|
|
44
|
+
"""Drop all gates (unit tests only)."""
|
|
45
|
+
with _GATE_GUARD:
|
|
46
|
+
_GATES.clear()
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def chat_gate_snapshot() -> dict[str, Any]:
|
|
50
|
+
"""Debug helper: which engines currently hold a gate."""
|
|
51
|
+
with _GATE_GUARD:
|
|
52
|
+
return {
|
|
53
|
+
sid: {'locked': lock.locked()}
|
|
54
|
+
for sid, lock in _GATES.items()
|
|
55
|
+
}
|
|
@@ -905,6 +905,11 @@ def normalize_server(entry: dict[str, Any]) -> dict[str, Any]:
|
|
|
905
905
|
result['target_path'] = target_path
|
|
906
906
|
if draft_path:
|
|
907
907
|
result['draft_path'] = draft_path
|
|
908
|
+
if entry.get('dflash_draft_disabled') is True:
|
|
909
|
+
result['dflash_draft_disabled'] = True
|
|
910
|
+
reason = str(entry.get('dflash_draft_disabled_reason') or '').strip()
|
|
911
|
+
if reason:
|
|
912
|
+
result['dflash_draft_disabled_reason'] = reason
|
|
908
913
|
mmproj_path = str(entry.get('mmproj_path') or '').strip()
|
|
909
914
|
if mmproj_path:
|
|
910
915
|
result['mmproj_path'] = mmproj_path
|
|
@@ -43,8 +43,10 @@ def llama_server_binary(*, cfg: dict[str, Any] | None = None) -> Path:
|
|
|
43
43
|
|
|
44
44
|
|
|
45
45
|
def resolve_embedding_model_path(server: dict[str, Any], *, cfg: dict[str, Any] | None = None) -> Path:
|
|
46
|
-
|
|
47
|
-
|
|
46
|
+
# Read the field directly: this function is reachable from
|
|
47
|
+
# resolve_model_stack -> normalize_server, and re-normalizing here
|
|
48
|
+
# would recurse back into the same clamp_server_context_fields path.
|
|
49
|
+
target = str(server.get('target_path') or '').strip()
|
|
48
50
|
if target:
|
|
49
51
|
path = Path(target)
|
|
50
52
|
if path.is_file():
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Append-only log of OpenAI gateway chat routing (local engine vs cloud provider)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import threading
|
|
6
|
+
import time
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from core.log_utils import rotate_log
|
|
10
|
+
|
|
11
|
+
ROOT = Path(__file__).resolve().parent.parent
|
|
12
|
+
LOG_PATH = ROOT / 'logs' / 'gateway-access.log'
|
|
13
|
+
_lock = threading.Lock()
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def record_gateway_route(
|
|
17
|
+
*,
|
|
18
|
+
model: str,
|
|
19
|
+
route: str,
|
|
20
|
+
target: str,
|
|
21
|
+
client: str = '',
|
|
22
|
+
status: int = 0,
|
|
23
|
+
duration_ms: float = 0.0,
|
|
24
|
+
note: str = '',
|
|
25
|
+
) -> None:
|
|
26
|
+
stamp = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime())
|
|
27
|
+
line = (
|
|
28
|
+
f'[{stamp}] model={model!r} route={route} target={target!r} '
|
|
29
|
+
f'status={int(status or 0)} ms={round(float(duration_ms or 0.0), 2)}'
|
|
30
|
+
)
|
|
31
|
+
if client:
|
|
32
|
+
line += f' client={client!r}'
|
|
33
|
+
if note:
|
|
34
|
+
line += f' note={note!r}'
|
|
35
|
+
with _lock:
|
|
36
|
+
LOG_PATH.parent.mkdir(parents=True, exist_ok=True)
|
|
37
|
+
rotate_log(LOG_PATH)
|
|
38
|
+
with LOG_PATH.open('a', encoding='utf-8') as fh:
|
|
39
|
+
fh.write(line + '\n')
|
|
@@ -25,12 +25,27 @@ def is_cursor_compat_model_id(model: str) -> bool:
|
|
|
25
25
|
|
|
26
26
|
def default_gateway_chat_server(cfg: dict[str, Any]) -> dict[str, Any]:
|
|
27
27
|
"""Default chat engine for empty model or Cursor-compat aliases."""
|
|
28
|
+
wanted = str(cfg.get('gateway_server_id') or '').strip()
|
|
29
|
+
if wanted:
|
|
30
|
+
from core.api_providers import cloud_model_entries, with_api_label
|
|
31
|
+
|
|
32
|
+
for entry in cloud_model_entries(cfg):
|
|
33
|
+
if str(entry.get('id') or '').strip().lower() != wanted.lower():
|
|
34
|
+
continue
|
|
35
|
+
label = str(entry.get('label') or wanted).strip() or wanted
|
|
36
|
+
return {
|
|
37
|
+
'id': wanted,
|
|
38
|
+
'model_id': wanted,
|
|
39
|
+
'label': with_api_label(label),
|
|
40
|
+
'cloud': True,
|
|
41
|
+
'enabled': True,
|
|
42
|
+
'context_size': 8192,
|
|
43
|
+
}
|
|
28
44
|
servers = enabled_chat_servers(cfg)
|
|
29
45
|
if not servers:
|
|
30
46
|
from fastapi import HTTPException
|
|
31
47
|
|
|
32
48
|
raise HTTPException(status_code=503, detail='no enabled chat engine available')
|
|
33
|
-
wanted = str(cfg.get('gateway_server_id') or '').strip()
|
|
34
49
|
if wanted:
|
|
35
50
|
for server in servers:
|
|
36
51
|
if str(server.get('id') or '') == wanted:
|
|
@@ -0,0 +1,100 @@
|
|
|
1
|
+
"""Detect MTP / llama.cpp draft sidecars in Hugging Face catalog rows."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
_DRAFT_NAME_RE = re.compile(
|
|
9
|
+
r'(?:^|[._-])(?:draft|fastmtp|mtp-head|mtp-draft)(?:[._-]|\.gguf$)',
|
|
10
|
+
re.I,
|
|
11
|
+
)
|
|
12
|
+
_PARAM_B_RE = re.compile(r'(\d+(?:\.\d+)?)\s*b\b', re.I)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _repo_slug(row: dict[str, Any]) -> str:
|
|
16
|
+
repo_id = str(row.get('id') or row.get('label') or '').strip()
|
|
17
|
+
return repo_id.split('/')[-1].lower() if '/' in repo_id else repo_id.lower()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _param_billions(slug: str) -> float | None:
|
|
21
|
+
matches = [float(value) for value in _PARAM_B_RE.findall(slug or '')]
|
|
22
|
+
if not matches:
|
|
23
|
+
return None
|
|
24
|
+
return max(matches)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _display_size_gb(row: dict[str, Any]) -> float | None:
|
|
28
|
+
size_gb = row.get('size_gb')
|
|
29
|
+
if isinstance(size_gb, (int, float)) and float(size_gb) > 0:
|
|
30
|
+
return float(size_gb)
|
|
31
|
+
return None
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _bytes_implied_gb(row: dict[str, Any]) -> float | None:
|
|
35
|
+
try:
|
|
36
|
+
size_bytes = int(row.get('size_bytes') or 0)
|
|
37
|
+
except (TypeError, ValueError):
|
|
38
|
+
return None
|
|
39
|
+
if size_bytes <= 0:
|
|
40
|
+
return None
|
|
41
|
+
return size_bytes / (1024 ** 3)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def catalog_draft_primary_row(
|
|
45
|
+
row: dict[str, Any],
|
|
46
|
+
*,
|
|
47
|
+
gguf_files: list[dict[str, Any]] | None = None,
|
|
48
|
+
) -> bool:
|
|
49
|
+
"""True when the catalog row is dominated by an MTP/draft sidecar, not a full target GGUF."""
|
|
50
|
+
if not isinstance(row, dict):
|
|
51
|
+
return False
|
|
52
|
+
if row.get('accelerator_only'):
|
|
53
|
+
return False
|
|
54
|
+
if not (row.get('has_gguf') or gguf_files):
|
|
55
|
+
return False
|
|
56
|
+
|
|
57
|
+
slug = _repo_slug(row)
|
|
58
|
+
params_b = _param_billions(slug)
|
|
59
|
+
display_gb = _display_size_gb(row)
|
|
60
|
+
bytes_gb = _bytes_implied_gb(row)
|
|
61
|
+
|
|
62
|
+
from core.hf_local_match import is_auxiliary_gguf_filename
|
|
63
|
+
from core.hf_model_fit import quant_sizes_gb
|
|
64
|
+
|
|
65
|
+
files = gguf_files if isinstance(gguf_files, list) else row.get('gguf_files')
|
|
66
|
+
non_aux = [
|
|
67
|
+
item for item in (files or [])
|
|
68
|
+
if isinstance(item, dict)
|
|
69
|
+
and not is_auxiliary_gguf_filename(str(item.get('filename') or ''))
|
|
70
|
+
]
|
|
71
|
+
target_sizes = quant_sizes_gb(non_aux) if non_aux else []
|
|
72
|
+
max_target_gb = max(target_sizes) if target_sizes else None
|
|
73
|
+
|
|
74
|
+
if params_b and params_b >= 7:
|
|
75
|
+
if display_gb is not None and display_gb < 6:
|
|
76
|
+
return True
|
|
77
|
+
if max_target_gb is not None and max_target_gb < 6:
|
|
78
|
+
return True
|
|
79
|
+
if (
|
|
80
|
+
display_gb is not None
|
|
81
|
+
and display_gb < 6
|
|
82
|
+
and bytes_gb is not None
|
|
83
|
+
and bytes_gb >= 8
|
|
84
|
+
):
|
|
85
|
+
return True
|
|
86
|
+
|
|
87
|
+
if non_aux and all(
|
|
88
|
+
_DRAFT_NAME_RE.search(str(item.get('filename') or ''))
|
|
89
|
+
for item in non_aux
|
|
90
|
+
):
|
|
91
|
+
return True
|
|
92
|
+
|
|
93
|
+
if len(non_aux) == 1:
|
|
94
|
+
name = str(non_aux[0].get('filename') or '').lower()
|
|
95
|
+
if _DRAFT_NAME_RE.search(name):
|
|
96
|
+
return True
|
|
97
|
+
if 'fastmtp' in name and (max_target_gb or display_gb or 0) < 6:
|
|
98
|
+
return True
|
|
99
|
+
|
|
100
|
+
return False
|