dflash-console 0.3.232__tar.gz → 0.3.240__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (202) hide show
  1. {dflash_console-0.3.232 → dflash_console-0.3.240}/.gitignore +1 -0
  2. {dflash_console-0.3.232 → dflash_console-0.3.240}/PKG-INFO +5 -3
  3. {dflash_console-0.3.232 → dflash_console-0.3.240}/README.md +4 -2
  4. {dflash_console-0.3.232 → dflash_console-0.3.240}/api/app.py +111 -9
  5. {dflash_console-0.3.232 → dflash_console-0.3.240}/api/gateway.py +79 -5
  6. dflash_console-0.3.240/core/chat_queue.py +55 -0
  7. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/config.py +5 -0
  8. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/embedding_server.py +4 -2
  9. dflash_console-0.3.240/core/gateway_access_log.py +39 -0
  10. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gateway_routing.py +16 -1
  11. dflash_console-0.3.240/core/hf_catalog_draft.py +100 -0
  12. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_catalog_index.py +20 -2
  13. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_catalog_recommend.py +10 -1
  14. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_local_match.py +10 -2
  15. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_model_fit.py +16 -2
  16. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/huggingface.py +182 -27
  17. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/local_models.py +12 -10
  18. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_presets.py +13 -2
  19. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/server_boot.py +159 -0
  20. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/version.py +1 -1
  21. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/vision_setup.py +91 -12
  22. {dflash_console-0.3.232 → dflash_console-0.3.240}/pyproject.toml +1 -1
  23. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-shell.css +9753 -9718
  24. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/index.html +3 -3
  25. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/chat-live.js +8 -3
  26. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/dflash-console-ui.js +1 -0
  27. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/download-queue.js +988 -945
  28. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/downloads-live.js +100 -14
  29. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/model-search-live.js +130 -19
  30. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/models-live.js +6 -2
  31. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/server-live.js +5659 -5585
  32. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/settings-live.js +11 -2
  33. {dflash_console-0.3.232 → dflash_console-0.3.240}/LICENSE +0 -0
  34. {dflash_console-0.3.232 → dflash_console-0.3.240}/NOTICE.md +0 -0
  35. {dflash_console-0.3.232 → dflash_console-0.3.240}/api/__init__.py +0 -0
  36. {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo.png +0 -0
  37. {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo_only.png +0 -0
  38. {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo_only2.png +0 -0
  39. {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_console_logo_only_clear.png +0 -0
  40. {dflash_console-0.3.232 → dflash_console-0.3.240}/assets/dflash_txt_logo.png +0 -0
  41. {dflash_console-0.3.232 → dflash_console-0.3.240}/config.example.json +0 -0
  42. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/__init__.py +0 -0
  43. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_access_log.py +0 -0
  44. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_catalog.py +0 -0
  45. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_introspection.py +0 -0
  46. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/api_providers.py +0 -0
  47. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/auto_register.py +0 -0
  48. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/catalog_load.py +0 -0
  49. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/chat_proxy.py +0 -0
  50. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/chat_ready.py +0 -0
  51. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/chat_vision.py +0 -0
  52. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/client_identity.py +0 -0
  53. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/components_hub.py +0 -0
  54. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/dflash_generation.py +0 -0
  55. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/diagnostics_bundle.py +0 -0
  56. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/display_names.py +0 -0
  57. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/engine_state.py +0 -0
  58. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/freetoken_runtime_install.py +0 -0
  59. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/fs_browse.py +0 -0
  60. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/fs_reveal.py +0 -0
  61. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gguf_meta.py +0 -0
  62. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_devices.py +0 -0
  63. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_policy.py +0 -0
  64. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_process_memory_windows.py +0 -0
  65. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_processes.py +0 -0
  66. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/gpu_relief.py +0 -0
  67. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hardware_apply.py +0 -0
  68. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hardware_info.py +0 -0
  69. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_catalog_cache.py +0 -0
  70. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_engines.py +0 -0
  71. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/hf_install.py +0 -0
  72. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/inference_stats.py +0 -0
  73. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/library_import.py +0 -0
  74. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/load_progress.py +0 -0
  75. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/log_utils.py +0 -0
  76. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/memory_guardrails.py +0 -0
  77. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_discovery.py +0 -0
  78. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_paths.py +0 -0
  79. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_runtime_policy.py +0 -0
  80. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/model_stack.py +0 -0
  81. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/net_listeners.py +0 -0
  82. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/node_connect.py +0 -0
  83. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/ocr_setup.py +0 -0
  84. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/remote_nodes.py +0 -0
  85. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtime.py +0 -0
  86. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtime_install_job.py +0 -0
  87. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtime_recommendations.py +0 -0
  88. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/__init__.py +0 -0
  89. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/base.py +0 -0
  90. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/contention.py +0 -0
  91. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/faster_whisper.py +0 -0
  92. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/freetoken.py +0 -0
  93. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/noop.py +0 -0
  94. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/ollama.py +0 -0
  95. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/piper.py +0 -0
  96. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/registry.py +0 -0
  97. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/stt.py +0 -0
  98. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/transformers_hf.py +0 -0
  99. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/vibevoice.py +0 -0
  100. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/runtimes/vllm.py +0 -0
  101. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/setup.py +0 -0
  102. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/stack_match.py +0 -0
  103. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/status_report.py +0 -0
  104. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/support_journal.py +0 -0
  105. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/system_stats.py +0 -0
  106. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/transformers_runtime_install.py +0 -0
  107. {dflash_console-0.3.232 → dflash_console-0.3.240}/core/vllm_runtime_install.py +0 -0
  108. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/__init__.py +0 -0
  109. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/__main__.py +0 -0
  110. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/cli.py +0 -0
  111. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/commands.py +0 -0
  112. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/http.py +0 -0
  113. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/render.py +0 -0
  114. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/resolve.py +0 -0
  115. {dflash_console-0.3.232 → dflash_console-0.3.240}/dflash_cli/server_takeover.py +0 -0
  116. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/AGENT-PROMPT-CLIENT-IDENTITY.md +0 -0
  117. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/AGENT-PROMPT-DFLASH-TRANSLATION-PERF-AI-TOOLS.md +0 -0
  118. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/AGENT-PROMPT-TRANSLATEGEMMA-AI-TOOLS.md +0 -0
  119. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ARCHITECTURE.md +0 -0
  120. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/BUG-REPORTS.md +0 -0
  121. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/CLI.md +0 -0
  122. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/CLIENT-IDENTITY.md +0 -0
  123. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/CURSOR.md +0 -0
  124. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/GOING-PUBLIC.md +0 -0
  125. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/LICENSING.md +0 -0
  126. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/LINUX-CLI.md +0 -0
  127. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/MULTI-MODAL-AGENT-PROMPT.md +0 -0
  128. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/MULTI-MODAL-RUNTIME-PLAN.md +0 -0
  129. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/PRODUCTION.md +0 -0
  130. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/RELEASING.md +0 -0
  131. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/STT-ENGINE-DECISION.md +0 -0
  132. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/USER-GUIDE.md +0 -0
  133. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/announcements/public-preview-v0.3.103.md +0 -0
  134. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/announcements/public-preview-v0.3.122.md +0 -0
  135. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/announcements/public-preview.md +0 -0
  136. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/automatic-updates.md +0 -0
  137. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/plans/shared-speak-stt-service.md +0 -0
  138. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.103.md +0 -0
  139. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.106.md +0 -0
  140. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.117.md +0 -0
  141. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.118.md +0 -0
  142. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.119.md +0 -0
  143. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.120.md +0 -0
  144. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.121.md +0 -0
  145. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.122.md +0 -0
  146. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.123.md +0 -0
  147. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.128.md +0 -0
  148. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.129.md +0 -0
  149. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.132.md +0 -0
  150. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.133.md +0 -0
  151. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/release-notes-0.3.208.md +0 -0
  152. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/README.md +0 -0
  153. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/devices-view.md +0 -0
  154. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/model-search.md +0 -0
  155. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/models-view.md +0 -0
  156. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/nodes-v1-plan.md +0 -0
  157. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-appearance.md +0 -0
  158. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-chat.md +0 -0
  159. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-developer.md +0 -0
  160. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-general.md +0 -0
  161. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-hardware.md +0 -0
  162. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-integrations.md +0 -0
  163. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-lm-link.md +0 -0
  164. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-model-defaults.md +0 -0
  165. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-runtime.md +0 -0
  166. {dflash_console-0.3.232 → dflash_console-0.3.240}/docs/ui/settings-runtimes.md +0 -0
  167. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/app-functional.html +0 -0
  168. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-console.css +0 -0
  169. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-console.css.bak-density +0 -0
  170. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/dflash-shell.css.bak-density +0 -0
  171. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/css/studio.css +0 -0
  172. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/about-live.js +0 -0
  173. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/app-settings-live.js +0 -0
  174. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/app.js +0 -0
  175. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/bug-report-live.js +0 -0
  176. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/component-install-live.js +0 -0
  177. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/console-api.js +0 -0
  178. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/dashboard.js +0 -0
  179. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/desktop-shell.js +0 -0
  180. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/docs-live.js +0 -0
  181. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/library-browse-live.js +0 -0
  182. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/library-scan-live.js +0 -0
  183. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/library-type-options.js +0 -0
  184. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/log-format.js +0 -0
  185. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/model-card.js +0 -0
  186. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/model-catalog-groups.js +0 -0
  187. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/nodes-live.js +0 -0
  188. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/runtime-recommendations.js +0 -0
  189. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/runtime-steppers.js +0 -0
  190. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/select-theme.js +0 -0
  191. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/server-reload-watch.js +0 -0
  192. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/settings-live.js.api-providers-bak +0 -0
  193. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/settings-live.js.pre-api-merge-bak +0 -0
  194. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/setup-wizard-live.js +0 -0
  195. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/speak-live.js +0 -0
  196. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/stack-wizard.js +0 -0
  197. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/status-feed.js +0 -0
  198. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/sysbar-live.js +0 -0
  199. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/table-column-resize.js +0 -0
  200. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/js/ui-layout-prefs.js +0 -0
  201. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/vendor/marked.min.js +0 -0
  202. {dflash_console-0.3.232 → dflash_console-0.3.240}/static/vendor/purify.min.js +0 -0
@@ -46,6 +46,7 @@ runtimes/*
46
46
  # Generated by tools/dflash-setup-ui/build.ps1 from package.json
47
47
  tools/dflash-setup-ui/SetupVersion.cs
48
48
  tools/dflash-setup-ui/bin/
49
+ tools/dflash-setup-ui/bin_payload/
49
50
 
50
51
  # Python package build
51
52
  dist-pypi/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: dflash-console
3
- Version: 0.3.232
3
+ Version: 0.3.240
4
4
  Summary: Local control panel and terminal CLI for DFlash stacks and local model runtimes
5
5
  Project-URL: Homepage, https://github.com/ilan4ever/Dflash-Console
6
6
  Project-URL: Repository, https://github.com/ilan4ever/Dflash-Console
@@ -674,7 +674,7 @@ from one UI, then talk to them through a single OpenAI-compatible port.
674
674
  > supported platform**; Linux and macOS are not supported or tested (see
675
675
  > [Platform support](#platform-support) below).
676
676
 
677
- **Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.232
677
+ **Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.240
678
678
 
679
679
  ## Download (Windows)
680
680
 
@@ -753,7 +753,7 @@ Typical first session:
753
753
  3. For a DFlash GGUF, right-click and **Find and attach draft** if you want speculative decoding.
754
754
  4. Chat in the **Playground**, or point any OpenAI client at `http://127.0.0.1:8001/v1` with optional `X-DFlash-Client: YourApp` so **Engines** shows who is using each model.
755
755
 
756
- ### Recent improvements (v0.3.232)
756
+ ### Recent improvements (v0.3.240)
757
757
 
758
758
  - **Engine standby** — Running toggle gates load/chat until you arm the pipeline
759
759
  - **External GPU cards** — OneVoice, LM Studio, and other apps on the GPU; compact mobile layout; loading state expires when models are ready
@@ -797,6 +797,8 @@ http://127.0.0.1:8001/v1
797
797
  ```
798
798
 
799
799
  The gateway routes chat, embeddings, TTS, and STT to the loaded engine.
800
+
801
+ **Concurrent chat:** multiple OpenAI-compatible clients may call `chat/completions` on the same loaded model at once (Harness agent turn + title, or two UI actions). The Console serializes ready/JIT per engine, then lets llama-server use parallel slots. You should not see `upstream HTTP 409` for that steady-state overlap. HTTP 409 is still used for strict model mismatch (`X-DFlash-Strict-Model`), a checkpoint already loaded on a *different* engine, and stack/vision repair.
800
802
  Model names are tolerant (engine id, file name, or an alias such as `gpt-4o`).
801
803
 
802
804
  **Client identity:** send `X-DFlash-Client: YourApp` on load and chat requests so the
@@ -8,7 +8,7 @@ from one UI, then talk to them through a single OpenAI-compatible port.
8
8
  > supported platform**; Linux and macOS are not supported or tested (see
9
9
  > [Platform support](#platform-support) below).
10
10
 
11
- **Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.232
11
+ **Developer:** ILAN AVIV · **UI:** [http://127.0.0.1:8900/](http://127.0.0.1:8900/) · **Version:** v0.3.240
12
12
 
13
13
  ## Download (Windows)
14
14
 
@@ -87,7 +87,7 @@ Typical first session:
87
87
  3. For a DFlash GGUF, right-click and **Find and attach draft** if you want speculative decoding.
88
88
  4. Chat in the **Playground**, or point any OpenAI client at `http://127.0.0.1:8001/v1` with optional `X-DFlash-Client: YourApp` so **Engines** shows who is using each model.
89
89
 
90
- ### Recent improvements (v0.3.232)
90
+ ### Recent improvements (v0.3.240)
91
91
 
92
92
  - **Engine standby** — Running toggle gates load/chat until you arm the pipeline
93
93
  - **External GPU cards** — OneVoice, LM Studio, and other apps on the GPU; compact mobile layout; loading state expires when models are ready
@@ -131,6 +131,8 @@ http://127.0.0.1:8001/v1
131
131
  ```
132
132
 
133
133
  The gateway routes chat, embeddings, TTS, and STT to the loaded engine.
134
+
135
+ **Concurrent chat:** multiple OpenAI-compatible clients may call `chat/completions` on the same loaded model at once (Harness agent turn + title, or two UI actions). The Console serializes ready/JIT per engine, then lets llama-server use parallel slots. You should not see `upstream HTTP 409` for that steady-state overlap. HTTP 409 is still used for strict model mismatch (`X-DFlash-Strict-Model`), a checkpoint already loaded on a *different* engine, and stack/vision repair.
134
136
  Model names are tolerant (engine id, file name, or an alias such as `gpt-4o`).
135
137
 
136
138
  **Client identity:** send `X-DFlash-Client: YourApp` on load and chat requests so the
@@ -3067,10 +3067,57 @@ def _ensure_server_ready_for_chat(
3067
3067
  },
3068
3068
  )
3069
3069
 
3070
+ from core.inference_stats import is_proxy_generating
3071
+
3072
+ def _status_as_loaded(status: dict[str, Any]) -> dict[str, Any]:
3073
+ if status.get('status') != 'loaded':
3074
+ status = {**status, 'status': 'loaded'}
3075
+ return status
3076
+
3077
+ def _synthetic_loaded_from_config() -> dict[str, Any]:
3078
+ """When a probe fails under concurrent load, trust the configured id."""
3079
+ model_id = str(server.get('model_id') or '').strip()
3080
+ loaded = [model_id] if model_id else []
3081
+ return {
3082
+ **server,
3083
+ 'running': True,
3084
+ 'status': 'loaded' if loaded else 'running',
3085
+ 'booting': False,
3086
+ 'loaded_models': loaded,
3087
+ 'active_model_id': loaded[0] if loaded else '',
3088
+ 'ready_for_chat': bool(loaded),
3089
+ }
3090
+
3070
3091
  live = build_server_status(server, cfg=cfg)
3092
+
3093
+ # Another chat is already mid-flight on this engine: never JIT-load (that
3094
+ # path raises model_already_loaded_elsewhere / stack-repair 409s and can
3095
+ # stop_server under DFlash draft profiles). Share the live or configured id.
3096
+ if is_proxy_generating(server_id):
3097
+ if not live.get('loaded_models'):
3098
+ live = _synthetic_loaded_from_config()
3099
+ live = _status_as_loaded(live)
3100
+ if required_context and cfg.get('context_auto_grow') is not False:
3101
+ loaded_ctx = _loaded_per_slot_context(server)
3102
+ if loaded_ctx and required_context > loaded_ctx:
3103
+ # Growing reloads the engine — wait until the active turn ends.
3104
+ deadline = time.time() + 180.0
3105
+ while time.time() < deadline and is_proxy_generating(server_id):
3106
+ time.sleep(0.05)
3107
+ if not is_proxy_generating(server_id):
3108
+ return _ensure_server_ready_for_chat(
3109
+ server_id,
3110
+ server,
3111
+ cfg,
3112
+ client_label=client_label,
3113
+ required_context=required_context,
3114
+ )
3115
+ # Still busy after wait: serve with current context rather than 409.
3116
+ note_engine_active_client(server_id, client_label=client_label)
3117
+ return live
3118
+
3071
3119
  if live.get('loaded_models'):
3072
- if live.get('status') != 'loaded':
3073
- live = {**live, 'status': 'loaded'}
3120
+ live = _status_as_loaded(live)
3074
3121
  # Already loaded. Auto-grow if this request needs more context than
3075
3122
  # the loaded model provides; otherwise share the model as-is.
3076
3123
  if required_context and cfg.get('context_auto_grow') is not False:
@@ -3080,6 +3127,23 @@ def _ensure_server_ready_for_chat(
3080
3127
  note_engine_active_client(server_id, client_label=client_label)
3081
3128
  return live
3082
3129
 
3130
+ # Probe can return empty loaded_models while llama is busy serving another
3131
+ # client. Retry briefly before treating the engine as idle for JIT load.
3132
+ host = str(server.get('host') or '127.0.0.1').strip() or '127.0.0.1'
3133
+ port = int(server.get('port') or 0)
3134
+ if port > 0 and tcp_port_open(host, port):
3135
+ for _ in range(6):
3136
+ time.sleep(0.05)
3137
+ live = build_server_status(server, cfg=cfg)
3138
+ if live.get('loaded_models') or is_proxy_generating(server_id):
3139
+ break
3140
+ if is_proxy_generating(server_id) and not live.get('loaded_models'):
3141
+ live = _synthetic_loaded_from_config()
3142
+ if live.get('loaded_models'):
3143
+ live = _status_as_loaded(live)
3144
+ note_engine_active_client(server_id, client_label=client_label)
3145
+ return live
3146
+
3083
3147
  if live.get('status') == 'booting':
3084
3148
  return _wait_until_loaded()
3085
3149
 
@@ -3421,6 +3485,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
3421
3485
  import json
3422
3486
  import urllib.error
3423
3487
 
3488
+ _chat_gate = None
3489
+ _chat_gate_held = False
3490
+
3424
3491
  from core.chat_proxy import (
3425
3492
  apply_reasoning_policy,
3426
3493
  chat_upstream_read_timeout,
@@ -3481,6 +3548,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
3481
3548
  body_json = None
3482
3549
  required_context = 0
3483
3550
  else:
3551
+ server = _require_server(cfg, server_id)
3552
+ _arm_server_engine_for_api(server_id, cfg)
3553
+ cfg = load_config()
3484
3554
  server = _require_server(cfg, server_id)
3485
3555
  raw = await request.body()
3486
3556
  try:
@@ -3528,17 +3598,31 @@ async def proxy_chat_completions(server_id: str, request: Request):
3528
3598
  header_context = request_load_context_size(request) or 0
3529
3599
  body_context = chat_body_load_context_size(body_json) or 0
3530
3600
  required_context = max(estimated_context or 0, header_context or 0, body_context or 0)
3531
- live = _ensure_server_ready_for_chat(
3532
- server_id,
3533
- server,
3534
- cfg,
3535
- client_label=_request_client_label(request),
3536
- required_context=required_context or None,
3537
- )
3601
+ from core.chat_queue import chat_server_gate
3602
+
3603
+ _chat_gate = chat_server_gate(server_id)
3604
+ await _chat_gate.acquire()
3605
+ _chat_gate_held = True
3606
+ try:
3607
+ live = _ensure_server_ready_for_chat(
3608
+ server_id,
3609
+ server,
3610
+ cfg,
3611
+ client_label=_request_client_label(request),
3612
+ required_context=required_context or None,
3613
+ )
3614
+ except Exception:
3615
+ if _chat_gate_held:
3616
+ _chat_gate.release()
3617
+ _chat_gate_held = False
3618
+ raise
3538
3619
 
3539
3620
  api_url = str(server.get('api_url') or '')
3540
3621
  base = api_base_url(api_url)
3541
3622
  if not base:
3623
+ if _chat_gate_held and _chat_gate is not None:
3624
+ _chat_gate.release()
3625
+ _chat_gate_held = False
3542
3626
  raise HTTPException(status_code=400, detail='engine api_url not configured')
3543
3627
  # Non-reasoning models never negotiate reasoning: strip reasoning_effort and
3544
3628
  # thinking toggles so the API returns the regular chat behaviour.
@@ -3551,6 +3635,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
3551
3635
  attributed_client=attributed,
3552
3636
  )
3553
3637
  if reasoning_error:
3638
+ if _chat_gate_held and _chat_gate is not None:
3639
+ _chat_gate.release()
3640
+ _chat_gate_held = False
3554
3641
  raise HTTPException(
3555
3642
  status_code=400,
3556
3643
  detail={
@@ -3595,6 +3682,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
3595
3682
  and requested_model
3596
3683
  and not model_ids_compatible(requested_model, upstream_model_id)
3597
3684
  ):
3685
+ if _chat_gate_held and _chat_gate is not None:
3686
+ _chat_gate.release()
3687
+ _chat_gate_held = False
3598
3688
  raise HTTPException(
3599
3689
  status_code=409,
3600
3690
  detail={
@@ -3640,11 +3730,20 @@ async def proxy_chat_completions(server_id: str, request: Request):
3640
3730
  )
3641
3731
  except urllib.error.HTTPError as exc:
3642
3732
  mark_inference_end(server_id, client_label=client_label)
3733
+ if _chat_gate_held and _chat_gate is not None:
3734
+ _chat_gate.release()
3735
+ _chat_gate_held = False
3643
3736
  detail = exc.read().decode('utf-8', errors='replace')
3644
3737
  raise HTTPException(status_code=exc.code, detail=detail) from exc
3645
3738
  except Exception as exc:
3646
3739
  mark_inference_end(server_id, client_label=client_label)
3740
+ if _chat_gate_held and _chat_gate is not None:
3741
+ _chat_gate.release()
3742
+ _chat_gate_held = False
3647
3743
  raise HTTPException(status_code=502, detail=str(exc)) from exc
3744
+ if _chat_gate_held and _chat_gate is not None:
3745
+ _chat_gate.release()
3746
+ _chat_gate_held = False
3648
3747
 
3649
3748
  keepalive_interval = 15.0
3650
3749
 
@@ -3717,6 +3816,9 @@ async def proxy_chat_completions(server_id: str, request: Request):
3717
3816
  model_id=str(live.get('active_model_id') or server.get('model_id') or ''),
3718
3817
  client_label=client_label,
3719
3818
  )
3819
+ if _chat_gate_held and _chat_gate is not None:
3820
+ _chat_gate.release()
3821
+ _chat_gate_held = False
3720
3822
  active_model = str(live.get('active_model_id') or server.get('model_id') or '')
3721
3823
  disconnect_task = asyncio.create_task(
3722
3824
  _abort_upstream_when_client_disconnects(
@@ -15,6 +15,13 @@ Routes:
15
15
  GET /health gateway + console health
16
16
  GET / info
17
17
 
18
+ Concurrent OpenAI clients on the same loaded engine (for example DeepSeek Harness
19
+ main turn + session title) are accepted: the Console queues the ready/start
20
+ section per ``server_id``, then llama-server multiplexes within ``parallel_slots``.
21
+ Steady-state overlaps must not return HTTP 409. Intentional 409s remain for
22
+ ``X-DFlash-Strict-Model`` mismatches, duplicate checkpoint on another engine,
23
+ DFlash stack / vision repair, and load/unload conflicts.
24
+
18
25
  The default chat engine is ``config.json -> gateway_server_id`` (falls back to
19
26
  the first enabled non-embedding engine); embeddings route to the first enabled
20
27
  embedding engine. The gateway is started on the UI process lifespan, so it is
@@ -41,6 +48,7 @@ from core.gateway_routing import (
41
48
  enabled_chat_servers,
42
49
  gateway_model_aliases,
43
50
  gateway_public_model_ids,
51
+ is_cursor_compat_model_id,
44
52
  resolve_chat_server,
45
53
  _advertised_engine_server,
46
54
  )
@@ -51,6 +59,11 @@ logger = logging.getLogger(__name__)
51
59
 
52
60
  gateway_app = FastAPI(title='DFlash Console OpenAI Gateway', version='0.1.0')
53
61
 
62
+
63
+ def gateway_cloud_chat_enabled(cfg: dict[str, Any]) -> bool:
64
+ """Allow active Settings cloud models unless explicitly disabled."""
65
+ return cfg.get('gateway_cloud_chat_enabled') is not False
66
+
54
67
  _FORWARD_HEADERS = {
55
68
  'content-type',
56
69
  'accept',
@@ -383,6 +396,9 @@ async def list_models() -> dict[str, Any]:
383
396
  pass
384
397
  from core.api_providers import cloud_model_entries, with_api_label
385
398
 
399
+ if not gateway_cloud_chat_enabled(cfg):
400
+ return {'object': 'list', 'data': data}
401
+
386
402
  for entry in cloud_model_entries(cfg):
387
403
  mid = str(entry.get('id') or '').strip()
388
404
  if not mid or mid.lower() in listed_ids:
@@ -432,14 +448,29 @@ async def _forward_chat(
432
448
  try:
433
449
  async with httpx.AsyncClient(timeout=None) as client:
434
450
  async with client.stream('POST', url, content=body, headers=headers) as upstream:
435
- upstream.raise_for_status()
451
+ if upstream.status_code >= 400:
452
+ raw_err = await upstream.aread()
453
+ # Re-raise as HTTPStatusError with body already buffered.
454
+ response = httpx.Response(
455
+ upstream.status_code,
456
+ content=raw_err,
457
+ request=upstream.request,
458
+ headers=upstream.headers,
459
+ )
460
+ raise httpx.HTTPStatusError(
461
+ f'Client error {upstream.status_code}',
462
+ request=upstream.request,
463
+ response=response,
464
+ )
436
465
  async for chunk in upstream.aiter_bytes():
437
466
  yield chunk
438
467
  except httpx.HTTPStatusError as exc:
439
468
  status = int(exc.response.status_code or 500)
440
469
  logger.warning('gateway chat upstream HTTP %s for %s', status, url)
441
470
  try:
442
- raw = await exc.response.aread()
471
+ raw = exc.response.content or b''
472
+ if not raw:
473
+ raw = await exc.response.aread()
443
474
  except Exception:
444
475
  raw = b''
445
476
  try: # DFLASH_LOG_400_BODIES (stream path)
@@ -569,20 +600,58 @@ async def chat_completions(request: Request) -> Response:
569
600
  if not isinstance(model, str):
570
601
  model = ''
571
602
  from core.api_providers import resolve_cloud_provider_for_model
572
-
573
- cloud_provider = resolve_cloud_provider_for_model(cfg, model)
603
+ from core.client_identity import resolve_client_label
604
+ from core.gateway_access_log import record_gateway_route
605
+
606
+ # A cloud model selected as the gateway default must also handle clients
607
+ # that omit ``model`` or send a Cursor-compatible placeholder.
608
+ configured_model = str(cfg.get('gateway_server_id') or '').strip()
609
+ if (
610
+ gateway_cloud_chat_enabled(cfg)
611
+ and configured_model
612
+ and (not model or is_cursor_compat_model_id(model))
613
+ and resolve_cloud_provider_for_model(cfg, configured_model) is not None
614
+ ):
615
+ model = configured_model
616
+
617
+ client_label = resolve_client_label(request)
618
+ cloud_provider = (
619
+ resolve_cloud_provider_for_model(cfg, model)
620
+ if gateway_cloud_chat_enabled(cfg)
621
+ else None
622
+ )
574
623
  if cloud_provider is not None:
624
+ provider_id = str(cloud_provider.get('id') or 'cloud')
625
+ record_gateway_route(
626
+ model=model,
627
+ route='cloud',
628
+ target=provider_id,
629
+ client=client_label,
630
+ note=str(cloud_provider.get('base_url') or ''),
631
+ )
575
632
  body: bytes | None = None
576
633
  if isinstance(payload, dict):
634
+ payload['model'] = model
577
635
  try:
578
636
  body = json.dumps(payload).encode('utf-8')
579
637
  except Exception:
580
638
  body = None
581
639
  if body is None:
582
640
  body = await request.body()
583
- return await _forward_cloud_chat(request, provider=cloud_provider, body=body)
641
+ response = await _forward_cloud_chat(request, provider=cloud_provider, body=body)
642
+ if isinstance(response, Response):
643
+ response.headers['X-DFlash-Route'] = 'cloud'
644
+ response.headers['X-DFlash-Provider-Id'] = provider_id
645
+ return response
584
646
  server, upstream = await _resolve_chat_target(cfg, model)
585
647
  sid = str(server.get('id') or '')
648
+ record_gateway_route(
649
+ model=model,
650
+ route='local',
651
+ target=sid or str(server.get('label') or ''),
652
+ client=client_label,
653
+ note=str(upstream or ''),
654
+ )
586
655
  if sid:
587
656
  from core.engine_state import note_engine_on
588
657
 
@@ -603,6 +672,10 @@ async def chat_completions(request: Request) -> Response:
603
672
  reasoning_model = model_has_reasoning(server)
604
673
  client_label = resolve_client_label(request)
605
674
  attributed = bool(client_label) and client_label != LABEL_UNKNOWN_API
675
+ # External OpenAI clients (Cursor, SDKs) use the gateway without X-DFlash-Client.
676
+ # Treat them like attributed clients so low max_tokens disables reasoning instead of 400.
677
+ if not attributed:
678
+ attributed = True
606
679
  disable_reasoning, reasoning_error = resolve_disable_reasoning_for_chat(
607
680
  body,
608
681
  reasoning=reasoning_model,
@@ -641,6 +714,7 @@ async def chat_completions(request: Request) -> Response:
641
714
  filter_reasoning = disable_reasoning
642
715
  response = await _forward_chat(request, url, body, filter_reasoning=filter_reasoning)
643
716
  if isinstance(response, Response):
717
+ response.headers['X-DFlash-Route'] = 'local'
644
718
  response.headers['X-DFlash-Server-Id'] = sid
645
719
  return response
646
720
 
@@ -0,0 +1,55 @@
1
+ """Per-engine chat gates so concurrent OpenAI clients share one loaded model safely.
2
+
3
+ Steady-state overlapping ``/v1/chat/completions`` (e.g. DeepSeek Harness main turn +
4
+ session title) must not re-enter JIT load and raise HTTP 409. Callers hold the
5
+ gate through ready-check + ``mark_inference_start`` (and opening an upstream
6
+ stream); llama-server then multiplexes within ``parallel_slots``.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import asyncio
12
+ import threading
13
+ from collections.abc import AsyncIterator
14
+ from contextlib import asynccontextmanager
15
+ from typing import Any
16
+
17
+ _GATE_GUARD = threading.Lock()
18
+ _GATES: dict[str, asyncio.Lock] = {}
19
+
20
+
21
+ def chat_server_gate(server_id: str) -> asyncio.Lock:
22
+ """Return the asyncio lock for this engine id (created on first use)."""
23
+ sid = str(server_id or '').strip() or '_'
24
+ with _GATE_GUARD:
25
+ lock = _GATES.get(sid)
26
+ if lock is None:
27
+ lock = asyncio.Lock()
28
+ _GATES[sid] = lock
29
+ return lock
30
+
31
+
32
+ @asynccontextmanager
33
+ async def hold_chat_server_gate(server_id: str) -> AsyncIterator[None]:
34
+ """Acquire the per-engine chat gate for the critical ready/start section."""
35
+ lock = chat_server_gate(server_id)
36
+ await lock.acquire()
37
+ try:
38
+ yield
39
+ finally:
40
+ lock.release()
41
+
42
+
43
+ def reset_chat_server_gates_for_tests() -> None:
44
+ """Drop all gates (unit tests only)."""
45
+ with _GATE_GUARD:
46
+ _GATES.clear()
47
+
48
+
49
+ def chat_gate_snapshot() -> dict[str, Any]:
50
+ """Debug helper: which engines currently hold a gate."""
51
+ with _GATE_GUARD:
52
+ return {
53
+ sid: {'locked': lock.locked()}
54
+ for sid, lock in _GATES.items()
55
+ }
@@ -905,6 +905,11 @@ def normalize_server(entry: dict[str, Any]) -> dict[str, Any]:
905
905
  result['target_path'] = target_path
906
906
  if draft_path:
907
907
  result['draft_path'] = draft_path
908
+ if entry.get('dflash_draft_disabled') is True:
909
+ result['dflash_draft_disabled'] = True
910
+ reason = str(entry.get('dflash_draft_disabled_reason') or '').strip()
911
+ if reason:
912
+ result['dflash_draft_disabled_reason'] = reason
908
913
  mmproj_path = str(entry.get('mmproj_path') or '').strip()
909
914
  if mmproj_path:
910
915
  result['mmproj_path'] = mmproj_path
@@ -43,8 +43,10 @@ def llama_server_binary(*, cfg: dict[str, Any] | None = None) -> Path:
43
43
 
44
44
 
45
45
  def resolve_embedding_model_path(server: dict[str, Any], *, cfg: dict[str, Any] | None = None) -> Path:
46
- entry = normalize_server(server)
47
- target = str(entry.get('target_path') or '').strip()
46
+ # Read the field directly: this function is reachable from
47
+ # resolve_model_stack -> normalize_server, and re-normalizing here
48
+ # would recurse back into the same clamp_server_context_fields path.
49
+ target = str(server.get('target_path') or '').strip()
48
50
  if target:
49
51
  path = Path(target)
50
52
  if path.is_file():
@@ -0,0 +1,39 @@
1
+ """Append-only log of OpenAI gateway chat routing (local engine vs cloud provider)."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import threading
6
+ import time
7
+ from pathlib import Path
8
+
9
+ from core.log_utils import rotate_log
10
+
11
+ ROOT = Path(__file__).resolve().parent.parent
12
+ LOG_PATH = ROOT / 'logs' / 'gateway-access.log'
13
+ _lock = threading.Lock()
14
+
15
+
16
+ def record_gateway_route(
17
+ *,
18
+ model: str,
19
+ route: str,
20
+ target: str,
21
+ client: str = '',
22
+ status: int = 0,
23
+ duration_ms: float = 0.0,
24
+ note: str = '',
25
+ ) -> None:
26
+ stamp = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime())
27
+ line = (
28
+ f'[{stamp}] model={model!r} route={route} target={target!r} '
29
+ f'status={int(status or 0)} ms={round(float(duration_ms or 0.0), 2)}'
30
+ )
31
+ if client:
32
+ line += f' client={client!r}'
33
+ if note:
34
+ line += f' note={note!r}'
35
+ with _lock:
36
+ LOG_PATH.parent.mkdir(parents=True, exist_ok=True)
37
+ rotate_log(LOG_PATH)
38
+ with LOG_PATH.open('a', encoding='utf-8') as fh:
39
+ fh.write(line + '\n')
@@ -25,12 +25,27 @@ def is_cursor_compat_model_id(model: str) -> bool:
25
25
 
26
26
  def default_gateway_chat_server(cfg: dict[str, Any]) -> dict[str, Any]:
27
27
  """Default chat engine for empty model or Cursor-compat aliases."""
28
+ wanted = str(cfg.get('gateway_server_id') or '').strip()
29
+ if wanted:
30
+ from core.api_providers import cloud_model_entries, with_api_label
31
+
32
+ for entry in cloud_model_entries(cfg):
33
+ if str(entry.get('id') or '').strip().lower() != wanted.lower():
34
+ continue
35
+ label = str(entry.get('label') or wanted).strip() or wanted
36
+ return {
37
+ 'id': wanted,
38
+ 'model_id': wanted,
39
+ 'label': with_api_label(label),
40
+ 'cloud': True,
41
+ 'enabled': True,
42
+ 'context_size': 8192,
43
+ }
28
44
  servers = enabled_chat_servers(cfg)
29
45
  if not servers:
30
46
  from fastapi import HTTPException
31
47
 
32
48
  raise HTTPException(status_code=503, detail='no enabled chat engine available')
33
- wanted = str(cfg.get('gateway_server_id') or '').strip()
34
49
  if wanted:
35
50
  for server in servers:
36
51
  if str(server.get('id') or '') == wanted:
@@ -0,0 +1,100 @@
1
+ """Detect MTP / llama.cpp draft sidecars in Hugging Face catalog rows."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from typing import Any
7
+
8
+ _DRAFT_NAME_RE = re.compile(
9
+ r'(?:^|[._-])(?:draft|fastmtp|mtp-head|mtp-draft)(?:[._-]|\.gguf$)',
10
+ re.I,
11
+ )
12
+ _PARAM_B_RE = re.compile(r'(\d+(?:\.\d+)?)\s*b\b', re.I)
13
+
14
+
15
+ def _repo_slug(row: dict[str, Any]) -> str:
16
+ repo_id = str(row.get('id') or row.get('label') or '').strip()
17
+ return repo_id.split('/')[-1].lower() if '/' in repo_id else repo_id.lower()
18
+
19
+
20
+ def _param_billions(slug: str) -> float | None:
21
+ matches = [float(value) for value in _PARAM_B_RE.findall(slug or '')]
22
+ if not matches:
23
+ return None
24
+ return max(matches)
25
+
26
+
27
+ def _display_size_gb(row: dict[str, Any]) -> float | None:
28
+ size_gb = row.get('size_gb')
29
+ if isinstance(size_gb, (int, float)) and float(size_gb) > 0:
30
+ return float(size_gb)
31
+ return None
32
+
33
+
34
+ def _bytes_implied_gb(row: dict[str, Any]) -> float | None:
35
+ try:
36
+ size_bytes = int(row.get('size_bytes') or 0)
37
+ except (TypeError, ValueError):
38
+ return None
39
+ if size_bytes <= 0:
40
+ return None
41
+ return size_bytes / (1024 ** 3)
42
+
43
+
44
+ def catalog_draft_primary_row(
45
+ row: dict[str, Any],
46
+ *,
47
+ gguf_files: list[dict[str, Any]] | None = None,
48
+ ) -> bool:
49
+ """True when the catalog row is dominated by an MTP/draft sidecar, not a full target GGUF."""
50
+ if not isinstance(row, dict):
51
+ return False
52
+ if row.get('accelerator_only'):
53
+ return False
54
+ if not (row.get('has_gguf') or gguf_files):
55
+ return False
56
+
57
+ slug = _repo_slug(row)
58
+ params_b = _param_billions(slug)
59
+ display_gb = _display_size_gb(row)
60
+ bytes_gb = _bytes_implied_gb(row)
61
+
62
+ from core.hf_local_match import is_auxiliary_gguf_filename
63
+ from core.hf_model_fit import quant_sizes_gb
64
+
65
+ files = gguf_files if isinstance(gguf_files, list) else row.get('gguf_files')
66
+ non_aux = [
67
+ item for item in (files or [])
68
+ if isinstance(item, dict)
69
+ and not is_auxiliary_gguf_filename(str(item.get('filename') or ''))
70
+ ]
71
+ target_sizes = quant_sizes_gb(non_aux) if non_aux else []
72
+ max_target_gb = max(target_sizes) if target_sizes else None
73
+
74
+ if params_b and params_b >= 7:
75
+ if display_gb is not None and display_gb < 6:
76
+ return True
77
+ if max_target_gb is not None and max_target_gb < 6:
78
+ return True
79
+ if (
80
+ display_gb is not None
81
+ and display_gb < 6
82
+ and bytes_gb is not None
83
+ and bytes_gb >= 8
84
+ ):
85
+ return True
86
+
87
+ if non_aux and all(
88
+ _DRAFT_NAME_RE.search(str(item.get('filename') or ''))
89
+ for item in non_aux
90
+ ):
91
+ return True
92
+
93
+ if len(non_aux) == 1:
94
+ name = str(non_aux[0].get('filename') or '').lower()
95
+ if _DRAFT_NAME_RE.search(name):
96
+ return True
97
+ if 'fastmtp' in name and (max_target_gb or display_gb or 0) < 6:
98
+ return True
99
+
100
+ return False