kiapi 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kiapi/__init__.py +3 -0
- kiapi/__main__.py +4 -0
- kiapi/api/__init__.py +26 -0
- kiapi/api/_constants/require_auth.py +14 -0
- kiapi/api/_helpers/build_job_responses.py +73 -0
- kiapi/api/_helpers/build_openapi.py +55 -0
- kiapi/api/_helpers/build_train_responses.py +21 -0
- kiapi/api/_helpers/get_accept.py +20 -0
- kiapi/api/_helpers/get_ctx.py +7 -0
- kiapi/api/_helpers/get_worker.py +7 -0
- kiapi/api/_helpers/register_capability_endpoints.py +70 -0
- kiapi/api/_helpers/submit_and_maybe_wait.py +113 -0
- kiapi/api/_settings.py +50 -0
- kiapi/api/_views/async_job_response.py +21 -0
- kiapi/api/_views/capability_model_spec.py +49 -0
- kiapi/api/app.py +150 -0
- kiapi/api/audio/acestep/_views/extract_response.py +57 -0
- kiapi/api/audio/acestep/_views/track_response.py +57 -0
- kiapi/api/audio/acestep/router.py +190 -0
- kiapi/api/audio/audiogen/_views/audio_response.py +46 -0
- kiapi/api/audio/audiogen/router.py +77 -0
- kiapi/api/chat/_views/chat_completion_response.py +71 -0
- kiapi/api/chat/router.py +253 -0
- kiapi/api/embedding/_views/embedding_response.py +35 -0
- kiapi/api/embedding/router.py +82 -0
- kiapi/api/files/_operations/get_file.py +14 -0
- kiapi/api/files/_operations/get_file_id.py +18 -0
- kiapi/api/files/_views/file_delete_response.py +14 -0
- kiapi/api/files/_views/file_list_response.py +15 -0
- kiapi/api/files/router.py +105 -0
- kiapi/api/health/_views/health_response.py +26 -0
- kiapi/api/health/router.py +28 -0
- kiapi/api/image/depthpro/_views/estimate_response.py +70 -0
- kiapi/api/image/depthpro/router.py +81 -0
- kiapi/api/image/ernie/_views/image_response.py +50 -0
- kiapi/api/image/ernie/router.py +181 -0
- kiapi/api/image/flux2/_views/image_response.py +53 -0
- kiapi/api/image/flux2/router.py +197 -0
- kiapi/api/image/ideogram4/_views/image_response.py +55 -0
- kiapi/api/image/ideogram4/router.py +87 -0
- kiapi/api/image/qwen/_views/image_response.py +47 -0
- kiapi/api/image/qwen/router.py +141 -0
- kiapi/api/image/seedvr2/_views/image_response.py +50 -0
- kiapi/api/image/seedvr2/router.py +83 -0
- kiapi/api/image/zimage/_views/image_response.py +49 -0
- kiapi/api/image/zimage/router.py +131 -0
- kiapi/api/jobs/_operations/get_job_id.py +18 -0
- kiapi/api/jobs/_views/job_delete_response.py +19 -0
- kiapi/api/jobs/_views/job_list_response.py +15 -0
- kiapi/api/jobs/router.py +62 -0
- kiapi/api/models/_schemas/openai_model_spec.py +26 -0
- kiapi/api/models/_views/model_list_response.py +15 -0
- kiapi/api/models/router.py +29 -0
- kiapi/api/video/ltx2/_views/__init__.py +1 -0
- kiapi/api/video/ltx2/_views/video_response.py +59 -0
- kiapi/api/video/ltx2/router.py +89 -0
- kiapi/api/web/_views/__init__.py +7 -0
- kiapi/api/web/_views/fetch_api_request.py +26 -0
- kiapi/api/web/_views/fetch_error_response.py +25 -0
- kiapi/api/web/router.py +302 -0
- kiapi/capabilities/__init__.py +24 -0
- kiapi/capabilities/_exceptions/CapabilityError.py +2 -0
- kiapi/capabilities/_exceptions/ValidationError.py +2 -0
- kiapi/capabilities/_helpers/attach_mflux_progress.py +50 -0
- kiapi/capabilities/_helpers/get_file_path.py +20 -0
- kiapi/capabilities/_helpers/resolve_file_ref.py +16 -0
- kiapi/capabilities/acestep/README.ja.md +174 -0
- kiapi/capabilities/acestep/README.md +167 -0
- kiapi/capabilities/acestep/__init__.py +37 -0
- kiapi/capabilities/acestep/_constants/description.py +122 -0
- kiapi/capabilities/acestep/_exceptions/ace_step_engine_error.py +2 -0
- kiapi/capabilities/acestep/_helpers/register.py +96 -0
- kiapi/capabilities/acestep/_models/acestep.py +105 -0
- kiapi/capabilities/acestep/_operations/handle_cover.py +23 -0
- kiapi/capabilities/acestep/_operations/handle_extract.py +38 -0
- kiapi/capabilities/acestep/_operations/handle_generate.py +18 -0
- kiapi/capabilities/acestep/_operations/handle_repaint.py +25 -0
- kiapi/capabilities/acestep/_operations/resolve_ace_step_paths.py +30 -0
- kiapi/capabilities/acestep/_operations/resolve_cover_params.py +21 -0
- kiapi/capabilities/acestep/_operations/resolve_engine.py +10 -0
- kiapi/capabilities/acestep/_operations/resolve_extract_params.py +24 -0
- kiapi/capabilities/acestep/_operations/resolve_generate_params.py +18 -0
- kiapi/capabilities/acestep/_operations/resolve_repaint_params.py +22 -0
- kiapi/capabilities/acestep/_schemas/ace_step_paths.py +29 -0
- kiapi/capabilities/acestep/_services/ace_step_engine.py +143 -0
- kiapi/capabilities/acestep/_settings.py +82 -0
- kiapi/capabilities/acestep/_views/ace_step_base.py +61 -0
- kiapi/capabilities/acestep/_views/cover_params.py +26 -0
- kiapi/capabilities/acestep/_views/cover_request.py +49 -0
- kiapi/capabilities/acestep/_views/extract_params.py +24 -0
- kiapi/capabilities/acestep/_views/extract_request.py +33 -0
- kiapi/capabilities/acestep/_views/generate_params.py +25 -0
- kiapi/capabilities/acestep/_views/generate_request.py +54 -0
- kiapi/capabilities/acestep/_views/repaint_params.py +27 -0
- kiapi/capabilities/acestep/_views/repaint_request.py +51 -0
- kiapi/capabilities/acestep/_views/track_params.py +7 -0
- kiapi/capabilities/acestep/worker_subprocess.py +247 -0
- kiapi/capabilities/audiogen/README.ja.md +49 -0
- kiapi/capabilities/audiogen/README.md +48 -0
- kiapi/capabilities/audiogen/__init__.py +25 -0
- kiapi/capabilities/audiogen/_constants/description.py +60 -0
- kiapi/capabilities/audiogen/_helpers/attach_audiogen_progress.py +27 -0
- kiapi/capabilities/audiogen/_helpers/register.py +42 -0
- kiapi/capabilities/audiogen/_helpers/validate_generate.py +14 -0
- kiapi/capabilities/audiogen/_models/audiogen.py +98 -0
- kiapi/capabilities/audiogen/_operations/handle_generate.py +27 -0
- kiapi/capabilities/audiogen/_operations/resolve_generate_params.py +25 -0
- kiapi/capabilities/audiogen/_settings.py +34 -0
- kiapi/capabilities/audiogen/_views/generate_params.py +14 -0
- kiapi/capabilities/audiogen/_views/generate_request.py +96 -0
- kiapi/capabilities/chat/README.ja.md +822 -0
- kiapi/capabilities/chat/README.md +806 -0
- kiapi/capabilities/chat/__init__.py +25 -0
- kiapi/capabilities/chat/_constants/description.py +163 -0
- kiapi/capabilities/chat/_helpers/register.py +67 -0
- kiapi/capabilities/chat/_models/qwen3_5.py +183 -0
- kiapi/capabilities/chat/_models/qwen3_omni.py +242 -0
- kiapi/capabilities/chat/_operations/apply_template.py +74 -0
- kiapi/capabilities/chat/_operations/completed_hermes_tool_call_text.py +9 -0
- kiapi/capabilities/chat/_operations/emit_streaming_response.py +297 -0
- kiapi/capabilities/chat/_operations/ensure_streaming_detokenizer_compat.py +49 -0
- kiapi/capabilities/chat/_operations/format_response.py +72 -0
- kiapi/capabilities/chat/_operations/handle_chat.py +25 -0
- kiapi/capabilities/chat/_operations/parse_hermes_tool_calls.py +63 -0
- kiapi/capabilities/chat/_operations/parse_json_tool_calls.py +66 -0
- kiapi/capabilities/chat/_operations/parse_messages.py +326 -0
- kiapi/capabilities/chat/_operations/resolve_chat_params.py +50 -0
- kiapi/capabilities/chat/_operations/stream_text_from_tokens.py +32 -0
- kiapi/capabilities/chat/_operations/strip_tool_calls.py +15 -0
- kiapi/capabilities/chat/_settings.py +74 -0
- kiapi/capabilities/chat/_utils/apply_parallel_tool_call_policy.py +17 -0
- kiapi/capabilities/chat/_utils/apply_seed.py +8 -0
- kiapi/capabilities/chat/_utils/load_audio_mono.py +31 -0
- kiapi/capabilities/chat/_utils/load_mlx_vlm.py +12 -0
- kiapi/capabilities/chat/_utils/warmup_params.py +26 -0
- kiapi/capabilities/chat/_views/chat_params.py +34 -0
- kiapi/capabilities/chat/_views/chat_request.py +153 -0
- kiapi/capabilities/depthpro/README.ja.md +131 -0
- kiapi/capabilities/depthpro/README.md +124 -0
- kiapi/capabilities/depthpro/__init__.py +23 -0
- kiapi/capabilities/depthpro/_constants/description.py +39 -0
- kiapi/capabilities/depthpro/_helpers/register.py +46 -0
- kiapi/capabilities/depthpro/_helpers/validate_estimate.py +32 -0
- kiapi/capabilities/depthpro/_models/depthpro.py +144 -0
- kiapi/capabilities/depthpro/_operations/handle_estimate.py +50 -0
- kiapi/capabilities/depthpro/_operations/is_quantize_override.py +12 -0
- kiapi/capabilities/depthpro/_operations/resolve_estimate_params.py +24 -0
- kiapi/capabilities/depthpro/_settings.py +46 -0
- kiapi/capabilities/depthpro/_views/estimate_params.py +21 -0
- kiapi/capabilities/depthpro/_views/estimate_request.py +63 -0
- kiapi/capabilities/embedding/README.ja.md +137 -0
- kiapi/capabilities/embedding/README.md +128 -0
- kiapi/capabilities/embedding/__init__.py +24 -0
- kiapi/capabilities/embedding/_constants/description.py +47 -0
- kiapi/capabilities/embedding/_helpers/register.py +66 -0
- kiapi/capabilities/embedding/_models/qwen3_text.py +53 -0
- kiapi/capabilities/embedding/_models/qwen3_vl.py +97 -0
- kiapi/capabilities/embedding/_operations/format_response.py +24 -0
- kiapi/capabilities/embedding/_operations/handle_embed.py +24 -0
- kiapi/capabilities/embedding/_operations/parse_inputs.py +108 -0
- kiapi/capabilities/embedding/_operations/resolve_embed_params.py +18 -0
- kiapi/capabilities/embedding/_settings.py +27 -0
- kiapi/capabilities/embedding/_utils/to_vector.py +20 -0
- kiapi/capabilities/embedding/_views/embed_params.py +17 -0
- kiapi/capabilities/embedding/_views/embed_request.py +39 -0
- kiapi/capabilities/ernie/README.ja.md +242 -0
- kiapi/capabilities/ernie/README.md +223 -0
- kiapi/capabilities/ernie/__init__.py +34 -0
- kiapi/capabilities/ernie/_constants/description.py +71 -0
- kiapi/capabilities/ernie/_helpers/register.py +60 -0
- kiapi/capabilities/ernie/_helpers/validate_edit.py +19 -0
- kiapi/capabilities/ernie/_helpers/validate_generate.py +8 -0
- kiapi/capabilities/ernie/_models/ernie.py +356 -0
- kiapi/capabilities/ernie/_operations/handle_edit.py +39 -0
- kiapi/capabilities/ernie/_operations/handle_generate.py +38 -0
- kiapi/capabilities/ernie/_operations/handle_train.py +77 -0
- kiapi/capabilities/ernie/_operations/is_quantize_override.py +16 -0
- kiapi/capabilities/ernie/_operations/resolve_edit_params.py +44 -0
- kiapi/capabilities/ernie/_operations/resolve_generate_params.py +40 -0
- kiapi/capabilities/ernie/_operations/resolve_lora_params.py +23 -0
- kiapi/capabilities/ernie/_operations/resolve_train_params.py +43 -0
- kiapi/capabilities/ernie/_operations/validate_common.py +33 -0
- kiapi/capabilities/ernie/_schemas/lora_ref.py +27 -0
- kiapi/capabilities/ernie/_settings.py +151 -0
- kiapi/capabilities/ernie/_views/edit_params.py +36 -0
- kiapi/capabilities/ernie/_views/edit_request.py +142 -0
- kiapi/capabilities/ernie/_views/generate_params.py +32 -0
- kiapi/capabilities/ernie/_views/generate_request.py +127 -0
- kiapi/capabilities/ernie/_views/train_params.py +28 -0
- kiapi/capabilities/ernie/_views/train_request.py +96 -0
- kiapi/capabilities/flux2/README.ja.md +260 -0
- kiapi/capabilities/flux2/README.md +241 -0
- kiapi/capabilities/flux2/__init__.py +36 -0
- kiapi/capabilities/flux2/_constants/description.py +81 -0
- kiapi/capabilities/flux2/_helpers/register.py +98 -0
- kiapi/capabilities/flux2/_helpers/validate_edit.py +12 -0
- kiapi/capabilities/flux2/_helpers/validate_generate.py +8 -0
- kiapi/capabilities/flux2/_models/flux2.py +396 -0
- kiapi/capabilities/flux2/_operations/handle_edit.py +44 -0
- kiapi/capabilities/flux2/_operations/handle_generate.py +46 -0
- kiapi/capabilities/flux2/_operations/handle_train.py +105 -0
- kiapi/capabilities/flux2/_operations/is_quantize_override.py +16 -0
- kiapi/capabilities/flux2/_operations/resolve_edit_params.py +43 -0
- kiapi/capabilities/flux2/_operations/resolve_generate_params.py +44 -0
- kiapi/capabilities/flux2/_operations/resolve_lora_params.py +23 -0
- kiapi/capabilities/flux2/_operations/resolve_train_params.py +42 -0
- kiapi/capabilities/flux2/_operations/role_spec.py +13 -0
- kiapi/capabilities/flux2/_operations/split_repo.py +11 -0
- kiapi/capabilities/flux2/_operations/validate_common.py +35 -0
- kiapi/capabilities/flux2/_schemas/lora_ref.py +27 -0
- kiapi/capabilities/flux2/_settings.py +181 -0
- kiapi/capabilities/flux2/_views/edit_params.py +36 -0
- kiapi/capabilities/flux2/_views/edit_request.py +145 -0
- kiapi/capabilities/flux2/_views/generate_params.py +36 -0
- kiapi/capabilities/flux2/_views/generate_request.py +150 -0
- kiapi/capabilities/flux2/_views/train_params.py +32 -0
- kiapi/capabilities/flux2/_views/train_request.py +102 -0
- kiapi/capabilities/ideogram4/README.ja.md +230 -0
- kiapi/capabilities/ideogram4/README.md +219 -0
- kiapi/capabilities/ideogram4/__init__.py +24 -0
- kiapi/capabilities/ideogram4/_constants/description.py +65 -0
- kiapi/capabilities/ideogram4/_helpers/register.py +44 -0
- kiapi/capabilities/ideogram4/_helpers/validate_generate.py +31 -0
- kiapi/capabilities/ideogram4/_models/ideogram4.py +120 -0
- kiapi/capabilities/ideogram4/_operations/handle_generate.py +30 -0
- kiapi/capabilities/ideogram4/_operations/is_quantize_override.py +12 -0
- kiapi/capabilities/ideogram4/_operations/resolve_generate_params.py +30 -0
- kiapi/capabilities/ideogram4/_settings.py +89 -0
- kiapi/capabilities/ideogram4/_views/generate_params.py +27 -0
- kiapi/capabilities/ideogram4/_views/generate_request.py +118 -0
- kiapi/capabilities/ltx2/README.ja.md +287 -0
- kiapi/capabilities/ltx2/README.md +274 -0
- kiapi/capabilities/ltx2/__init__.py +26 -0
- kiapi/capabilities/ltx2/_constants/description.py +81 -0
- kiapi/capabilities/ltx2/_helpers/register.py +58 -0
- kiapi/capabilities/ltx2/_helpers/validate_generate.py +42 -0
- kiapi/capabilities/ltx2/_models/ltx2.py +106 -0
- kiapi/capabilities/ltx2/_operations/detect_mode.py +19 -0
- kiapi/capabilities/ltx2/_operations/handle_generate.py +52 -0
- kiapi/capabilities/ltx2/_operations/resolve_generate_params.py +31 -0
- kiapi/capabilities/ltx2/_settings.py +93 -0
- kiapi/capabilities/ltx2/_views/generate_params.py +34 -0
- kiapi/capabilities/ltx2/_views/generate_request.py +151 -0
- kiapi/capabilities/qwen/README.ja.md +251 -0
- kiapi/capabilities/qwen/README.md +235 -0
- kiapi/capabilities/qwen/__init__.py +30 -0
- kiapi/capabilities/qwen/_constants/description.py +53 -0
- kiapi/capabilities/qwen/_helpers/register.py +60 -0
- kiapi/capabilities/qwen/_helpers/validate_edit.py +12 -0
- kiapi/capabilities/qwen/_helpers/validate_generate.py +12 -0
- kiapi/capabilities/qwen/_models/qwen.py +224 -0
- kiapi/capabilities/qwen/_operations/handle_edit.py +44 -0
- kiapi/capabilities/qwen/_operations/handle_generate.py +46 -0
- kiapi/capabilities/qwen/_operations/is_quantize_override.py +15 -0
- kiapi/capabilities/qwen/_operations/resolve_edit_params.py +35 -0
- kiapi/capabilities/qwen/_operations/resolve_generate_params.py +39 -0
- kiapi/capabilities/qwen/_operations/resolve_lora_params.py +23 -0
- kiapi/capabilities/qwen/_operations/validate_common.py +32 -0
- kiapi/capabilities/qwen/_schemas/lora_ref.py +27 -0
- kiapi/capabilities/qwen/_settings.py +107 -0
- kiapi/capabilities/qwen/_views/edit_params.py +36 -0
- kiapi/capabilities/qwen/_views/edit_request.py +138 -0
- kiapi/capabilities/qwen/_views/generate_params.py +37 -0
- kiapi/capabilities/qwen/_views/generate_request.py +149 -0
- kiapi/capabilities/seedvr2/README.ja.md +234 -0
- kiapi/capabilities/seedvr2/README.md +219 -0
- kiapi/capabilities/seedvr2/__init__.py +24 -0
- kiapi/capabilities/seedvr2/_constants/description.py +46 -0
- kiapi/capabilities/seedvr2/_helpers/register.py +60 -0
- kiapi/capabilities/seedvr2/_helpers/validate_upscale.py +52 -0
- kiapi/capabilities/seedvr2/_models/seedvr2.py +140 -0
- kiapi/capabilities/seedvr2/_operations/handle_upscale.py +42 -0
- kiapi/capabilities/seedvr2/_operations/is_quantize_override.py +12 -0
- kiapi/capabilities/seedvr2/_operations/resolve_upscale_params.py +30 -0
- kiapi/capabilities/seedvr2/_settings.py +74 -0
- kiapi/capabilities/seedvr2/_views/upscale_params.py +27 -0
- kiapi/capabilities/seedvr2/_views/upscale_request.py +94 -0
- kiapi/capabilities/web/README.ja.md +187 -0
- kiapi/capabilities/web/README.md +171 -0
- kiapi/capabilities/web/__init__.py +30 -0
- kiapi/capabilities/web/_constants/description.py +38 -0
- kiapi/capabilities/web/_exceptions/FetchError.py +24 -0
- kiapi/capabilities/web/_exceptions/SearchBackendError.py +11 -0
- kiapi/capabilities/web/_helpers/register.py +59 -0
- kiapi/capabilities/web/_models/crawl4ai.py +32 -0
- kiapi/capabilities/web/_models/searxng.py +44 -0
- kiapi/capabilities/web/_operations/handle_fetch.py +145 -0
- kiapi/capabilities/web/_operations/handle_search.py +56 -0
- kiapi/capabilities/web/_operations/resolve_fetch_params.py +19 -0
- kiapi/capabilities/web/_operations/resolve_search_params.py +34 -0
- kiapi/capabilities/web/_operations/resolve_searxng_config_dir.py +21 -0
- kiapi/capabilities/web/_services/__init__.py +1 -0
- kiapi/capabilities/web/_services/web_backend_process.py +144 -0
- kiapi/capabilities/web/_settings.py +138 -0
- kiapi/capabilities/web/_views/fetch_params.py +27 -0
- kiapi/capabilities/web/_views/fetch_request.py +29 -0
- kiapi/capabilities/web/_views/fetch_result.py +20 -0
- kiapi/capabilities/web/_views/search_params.py +43 -0
- kiapi/capabilities/web/_views/search_request.py +93 -0
- kiapi/capabilities/web/_views/search_response.py +49 -0
- kiapi/capabilities/web/resources/__init__.py +0 -0
- kiapi/capabilities/web/resources/searxng/__init__.py +0 -0
- kiapi/capabilities/web/resources/searxng/settings.yml +11 -0
- kiapi/capabilities/zimage/README.ja.md +249 -0
- kiapi/capabilities/zimage/README.md +231 -0
- kiapi/capabilities/zimage/__init__.py +34 -0
- kiapi/capabilities/zimage/_constants/description.py +73 -0
- kiapi/capabilities/zimage/_helpers/register.py +62 -0
- kiapi/capabilities/zimage/_helpers/validate_generate.py +37 -0
- kiapi/capabilities/zimage/_models/zimage.py +278 -0
- kiapi/capabilities/zimage/_operations/handle_generate.py +38 -0
- kiapi/capabilities/zimage/_operations/handle_train.py +78 -0
- kiapi/capabilities/zimage/_operations/is_quantize_override.py +11 -0
- kiapi/capabilities/zimage/_operations/resolve_generate_params.py +39 -0
- kiapi/capabilities/zimage/_operations/resolve_lora_params.py +23 -0
- kiapi/capabilities/zimage/_operations/resolve_train_params.py +38 -0
- kiapi/capabilities/zimage/_schemas/lora_ref.py +27 -0
- kiapi/capabilities/zimage/_settings.py +166 -0
- kiapi/capabilities/zimage/_views/generate_params.py +23 -0
- kiapi/capabilities/zimage/_views/generate_request.py +127 -0
- kiapi/capabilities/zimage/_views/train_params.py +22 -0
- kiapi/capabilities/zimage/_views/train_request.py +101 -0
- kiapi/cli/__init__.py +11 -0
- kiapi/cli/_helpers/dedupe_resources.py +12 -0
- kiapi/cli/_helpers/register_all_capabilities.py +36 -0
- kiapi/cli/_helpers/run_resources.py +24 -0
- kiapi/cli/_helpers/select_specs.py +67 -0
- kiapi/cli/activate/_operations/preflight_hf_access.py +129 -0
- kiapi/cli/activate/cli.py +37 -0
- kiapi/cli/check/_helpers/check_audio_acestep_turbo.py +22 -0
- kiapi/cli/check/_helpers/check_audio_acestep_xl_base.py +22 -0
- kiapi/cli/check/_helpers/check_audio_audiogen_medium.py +24 -0
- kiapi/cli/check/_helpers/check_chat_chat_qwen3_6_27b.py +19 -0
- kiapi/cli/check/_helpers/check_chat_chat_qwen3_omni.py +19 -0
- kiapi/cli/check/_helpers/check_embedding_embedding_qwen3_embedding_8b.py +14 -0
- kiapi/cli/check/_helpers/check_embedding_embedding_qwen3_vl_embedding_2b.py +19 -0
- kiapi/cli/check/_helpers/check_image_depthpro_base.py +21 -0
- kiapi/cli/check/_helpers/check_image_ernie_base.py +23 -0
- kiapi/cli/check/_helpers/check_image_ernie_turbo.py +23 -0
- kiapi/cli/check/_helpers/check_image_flux2_klein_9b.py +23 -0
- kiapi/cli/check/_helpers/check_image_flux2_klein_base_4b.py +23 -0
- kiapi/cli/check/_helpers/check_image_flux2_klein_base_9b.py +23 -0
- kiapi/cli/check/_helpers/check_image_ideogram4_fp8.py +23 -0
- kiapi/cli/check/_helpers/check_image_qwen_edit_2509.py +27 -0
- kiapi/cli/check/_helpers/check_image_qwen_image.py +23 -0
- kiapi/cli/check/_helpers/check_image_seedvr2_3b.py +23 -0
- kiapi/cli/check/_helpers/check_image_seedvr2_7b.py +23 -0
- kiapi/cli/check/_helpers/check_image_zimage_base.py +23 -0
- kiapi/cli/check/_helpers/check_image_zimage_turbo.py +23 -0
- kiapi/cli/check/_helpers/check_video_ltx2_distilled.py +26 -0
- kiapi/cli/check/_helpers/check_web_web_fetch.py +14 -0
- kiapi/cli/check/_helpers/check_web_web_search.py +14 -0
- kiapi/cli/check/_operations/build_check_result.py +20 -0
- kiapi/cli/check/_operations/check_context.py +10 -0
- kiapi/cli/check/_operations/create_sample_image.py +46 -0
- kiapi/cli/check/_schemas/check_result.py +12 -0
- kiapi/cli/check/cli.py +89 -0
- kiapi/cli/cli.py +29 -0
- kiapi/cli/config/cli.py +17 -0
- kiapi/cli/config/edit/cli.py +32 -0
- kiapi/cli/config/init/cli.py +33 -0
- kiapi/cli/config/show/cli.py +13 -0
- kiapi/cli/config/template/cli.py +36 -0
- kiapi/cli/deactivate/cli.py +49 -0
- kiapi/cli/run/asgi.py +16 -0
- kiapi/cli/run/cli.py +59 -0
- kiapi/cli/service/__init__.py +5 -0
- kiapi/cli/service/_services/service_manager.py +107 -0
- kiapi/cli/service/cli.py +19 -0
- kiapi/cli/service/install/cli.py +17 -0
- kiapi/cli/service/start/cli.py +22 -0
- kiapi/cli/service/status/cli.py +81 -0
- kiapi/cli/service/stop/cli.py +19 -0
- kiapi/cli/service/uninstall/cli.py +16 -0
- kiapi/cli/status/cli.py +226 -0
- kiapi/core/app/__init__.py +16 -0
- kiapi/core/app/_schemas/app_context.py +28 -0
- kiapi/core/app/_services/user_directory.py +59 -0
- kiapi/core/app/_settings.py +46 -0
- kiapi/core/capability/__init__.py +11 -0
- kiapi/core/capability/_schemas/capability_spec.py +17 -0
- kiapi/core/capability/_services/capability_spec_registry.py +21 -0
- kiapi/core/config/__init__.py +11 -0
- kiapi/core/config/_exceptions/user_config_error.py +2 -0
- kiapi/core/config/_helpers/get_user_settings_path.py +7 -0
- kiapi/core/config/_helpers/load_user_settings.py +54 -0
- kiapi/core/file/__init__.py +37 -0
- kiapi/core/file/_helpers/create_file_store.py +7 -0
- kiapi/core/file/_helpers/resolve_file_ref.py +90 -0
- kiapi/core/file/_schemas/file_data_url_ref.py +19 -0
- kiapi/core/file/_schemas/file_id_ref.py +16 -0
- kiapi/core/file/_schemas/file_record.py +46 -0
- kiapi/core/file/_schemas/file_url_ref.py +16 -0
- kiapi/core/file/_services/file_store.py +138 -0
- kiapi/core/file/_settings.py +22 -0
- kiapi/core/file/_types/file_id.py +2 -0
- kiapi/core/file/_types/file_ref.py +12 -0
- kiapi/core/file/_views/resolved_file_ref.py +16 -0
- kiapi/core/job/__init__.py +24 -0
- kiapi/core/job/_enums/job_status.py +9 -0
- kiapi/core/job/_helpers/creep_progress.py +123 -0
- kiapi/core/job/_models/job.py +121 -0
- kiapi/core/job/_services/job_store.py +33 -0
- kiapi/core/job/_services/progress_reporter.py +79 -0
- kiapi/core/job/_types/job_id.py +3 -0
- kiapi/core/job/_types/job_result.py +5 -0
- kiapi/core/job/_types/job_type.py +4 -0
- kiapi/core/logging/__init__.py +14 -0
- kiapi/core/logging/_helpers/get_log_level.py +8 -0
- kiapi/core/logging/_helpers/setup_logger.py +14 -0
- kiapi/core/logging/_settings.py +34 -0
- kiapi/core/logging/_types/log_level.py +3 -0
- kiapi/core/memory/__init__.py +21 -0
- kiapi/core/memory/_exceptions/memory_budget_error.py +2 -0
- kiapi/core/memory/_helpers/create_memory_manager.py +17 -0
- kiapi/core/memory/_operations/log_memory_limit.py +29 -0
- kiapi/core/memory/_operations/resolve_effective_memory_limit_gb.py +13 -0
- kiapi/core/memory/_operations/resolve_framework_strategy.py +42 -0
- kiapi/core/memory/_schemas/framework_strategy.py +14 -0
- kiapi/core/memory/_schemas/memory_stats.py +24 -0
- kiapi/core/memory/_schemas/resident_model.py +14 -0
- kiapi/core/memory/_schemas/resident_model_stats.py +49 -0
- kiapi/core/memory/_services/memory_manager.py +257 -0
- kiapi/core/memory/_settings.py +36 -0
- kiapi/core/memory/_types/framework_name.py +4 -0
- kiapi/core/memory/_types/resident_key.py +4 -0
- kiapi/core/model/__init__.py +36 -0
- kiapi/core/model/_exceptions/unknown_model_error.py +2 -0
- kiapi/core/model/_schemas/model_spec.py +145 -0
- kiapi/core/model/_services/model_registry.py +85 -0
- kiapi/core/model/_types/model_alias.py +2 -0
- kiapi/core/model/_types/model_domain.py +6 -0
- kiapi/core/model/_types/model_family.py +5 -0
- kiapi/core/model/_types/model_key.py +8 -0
- kiapi/core/model/_types/model_name.py +4 -0
- kiapi/core/model/_types/model_repo.py +3 -0
- kiapi/core/net/__init__.py +16 -0
- kiapi/core/net/_exceptions/unsafe_url_error.py +7 -0
- kiapi/core/net/_helpers/verify_public_url.py +68 -0
- kiapi/core/net/_settings.py +28 -0
- kiapi/core/setup/__init__.py +25 -0
- kiapi/core/setup/_exceptions/setup_required_error.py +2 -0
- kiapi/core/setup/_schemas/docker_image_resource.py +37 -0
- kiapi/core/setup/_schemas/hf_snapshot_resource.py +50 -0
- kiapi/core/setup/_schemas/local_path_resource.py +43 -0
- kiapi/core/setup/_schemas/python_package_resource.py +58 -0
- kiapi/core/setup/_schemas/python_venv_resource.py +63 -0
- kiapi/core/setup/_schemas/setup_status.py +7 -0
- kiapi/core/setup/_schemas/url_file_resource.py +43 -0
- kiapi/core/setup/_services/setup_manager.py +331 -0
- kiapi/core/setup/_types/setup_resource.py +21 -0
- kiapi/core/setup/_types/setup_target.py +11 -0
- kiapi/core/workdir/__init__.py +8 -0
- kiapi/core/workdir/_helpers/create_work_dir.py +11 -0
- kiapi/core/workdir/_settings.py +21 -0
- kiapi/core/worker/__init__.py +23 -0
- kiapi/core/worker/_helpers/create_worker.py +9 -0
- kiapi/core/worker/_services/worker.py +165 -0
- kiapi/core/worker/_settings.py +36 -0
- kiapi/core/worker/_types/job_thunk.py +6 -0
- kiapi-0.1.0.dist-info/METADATA +337 -0
- kiapi-0.1.0.dist-info/RECORD +465 -0
- kiapi-0.1.0.dist-info/WHEEL +4 -0
- kiapi-0.1.0.dist-info/entry_points.txt +2 -0
- kiapi-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
"""Global, budget-bounded resident model cache shared by every capability.
|
|
2
|
+
|
|
3
|
+
This generalizes mlx-vlm-server's per-server manager to the whole of kiapi.
|
|
4
|
+
Because the worker is single-flight (one job at a time, see worker.py), the model
|
|
5
|
+
acquired for the running job is the only one that will generate next, so it is
|
|
6
|
+
correct to budget only *its* peak headroom. Before a model is used we ensure:
|
|
7
|
+
|
|
8
|
+
Σ(resident weights, excluding the target)
|
|
9
|
+
+ target.weight + target.peak_headroom ≤ memory_limit_gb
|
|
10
|
+
|
|
11
|
+
evicting residents until it holds. Eviction order is **(priority asc, last_used
|
|
12
|
+
asc)**: lower-priority models go first, and among equal priority the
|
|
13
|
+
least-recently-used. So a small model you want resident gets a high ``priority``
|
|
14
|
+
and survives even when a big model churns through.
|
|
15
|
+
|
|
16
|
+
Two concepts are kept separate:
|
|
17
|
+
- **resident weights** — long-lived, reusable, evictable (tracked here),
|
|
18
|
+
- **peak headroom** — transient memory the *running* job needs on top of
|
|
19
|
+
weights; only one job runs at a time, so only the target's headroom matters.
|
|
20
|
+
|
|
21
|
+
Framework specifics (so eviction actually frees memory) live in
|
|
22
|
+
:func:`resolve_framework_strategy`:
|
|
23
|
+
- MLX: dropping Python refs is not enough — the Metal allocator keeps a buffer
|
|
24
|
+
cache. After deleting we ``gc.collect()`` then ``mx.clear_cache()``. Memory is
|
|
25
|
+
measured via ``mx.get_active_memory`` deltas.
|
|
26
|
+
- Other frameworks plug in via a new FrameworkStrategy and an optional
|
|
27
|
+
per-module ``release(payload)`` hook.
|
|
28
|
+
|
|
29
|
+
All mutating methods run on the single worker thread; only ``stats()`` is read
|
|
30
|
+
from the event-loop thread, so a small lock guards the dict.
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
import gc
|
|
34
|
+
import logging
|
|
35
|
+
import threading
|
|
36
|
+
import time
|
|
37
|
+
import traceback
|
|
38
|
+
|
|
39
|
+
from kiapi.core.model import ModelSpec
|
|
40
|
+
|
|
41
|
+
from .._exceptions.memory_budget_error import MemoryBudgetError
|
|
42
|
+
from .._operations.resolve_framework_strategy import resolve_framework_strategy
|
|
43
|
+
from .._schemas.memory_stats import MemoryStats
|
|
44
|
+
from .._schemas.resident_model import ResidentModel
|
|
45
|
+
from .._schemas.resident_model_stats import ResidentModelStats
|
|
46
|
+
from .._settings import MemorySettings
|
|
47
|
+
from .._types.resident_key import ResidentKey
|
|
48
|
+
|
|
49
|
+
_GB = 1024**3
|
|
50
|
+
logger = logging.getLogger(__name__)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
class MemoryManager:
|
|
54
|
+
def __init__(self, settings: MemorySettings) -> None:
|
|
55
|
+
self._settings: MemorySettings = settings
|
|
56
|
+
self._loaded: dict[ResidentKey, ResidentModel] = {}
|
|
57
|
+
self._lock = threading.Lock()
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def memory_limit_gb(self) -> float:
|
|
61
|
+
limit = self._settings.memory_limit_gb
|
|
62
|
+
|
|
63
|
+
if limit is None:
|
|
64
|
+
raise RuntimeError("memory_limit_gb must be resolved before use")
|
|
65
|
+
|
|
66
|
+
return limit
|
|
67
|
+
|
|
68
|
+
@property
|
|
69
|
+
def default_ttl_s(self) -> float:
|
|
70
|
+
return self._settings.default_ttl_s
|
|
71
|
+
|
|
72
|
+
# -- TTL -------------------------------------------------------------------
|
|
73
|
+
|
|
74
|
+
def _effective_ttl(self, spec: ModelSpec) -> float | None:
|
|
75
|
+
ttl = spec.ttl_seconds
|
|
76
|
+
|
|
77
|
+
if ttl is None:
|
|
78
|
+
ttl = self.default_ttl_s
|
|
79
|
+
if ttl <= 0:
|
|
80
|
+
return None
|
|
81
|
+
|
|
82
|
+
return ttl
|
|
83
|
+
|
|
84
|
+
def sweep_expired(
|
|
85
|
+
self, *, exclude_key: ResidentKey | None = None, now: float | None = None
|
|
86
|
+
) -> list[str]:
|
|
87
|
+
"""Runs on the worker thread — it frees framework memory, so it must not
|
|
88
|
+
be called from the event loop. Returns the freed models' names."""
|
|
89
|
+
now = now if now is not None else time.monotonic()
|
|
90
|
+
expired: list[ResidentKey] = []
|
|
91
|
+
for key, ld in list(self._loaded.items()):
|
|
92
|
+
if key == exclude_key:
|
|
93
|
+
continue
|
|
94
|
+
ttl = self._effective_ttl(ld.spec)
|
|
95
|
+
if ttl is not None and (now - ld.last_used) > ttl:
|
|
96
|
+
expired.append(key)
|
|
97
|
+
freed = []
|
|
98
|
+
for key in expired:
|
|
99
|
+
name = self._loaded[key].spec.name if key in self._loaded else key
|
|
100
|
+
self._evict(key, reason="ttl")
|
|
101
|
+
freed.append(name)
|
|
102
|
+
return freed
|
|
103
|
+
|
|
104
|
+
# -- public (worker thread) ------------------------------------------------
|
|
105
|
+
|
|
106
|
+
def acquire(self, spec: ModelSpec) -> object:
|
|
107
|
+
# Free any idle-expired models first (don't touch the one we're about to
|
|
108
|
+
# use) so their memory is reclaimed before we budget/load.
|
|
109
|
+
self.sweep_expired(exclude_key=spec.key)
|
|
110
|
+
existing = self._loaded.get(spec.key)
|
|
111
|
+
if existing is not None:
|
|
112
|
+
# Re-check budget: another model may have loaded since, and this
|
|
113
|
+
# model's generation headroom must still fit.
|
|
114
|
+
self._ensure_budget(spec, existing.weight_gb)
|
|
115
|
+
existing.last_used = time.monotonic()
|
|
116
|
+
return existing.payload
|
|
117
|
+
|
|
118
|
+
self._ensure_budget(spec, spec.weight_gb) # estimate (not loaded yet)
|
|
119
|
+
loaded = self._load(spec)
|
|
120
|
+
with self._lock:
|
|
121
|
+
self._loaded[spec.key] = loaded
|
|
122
|
+
return loaded.payload
|
|
123
|
+
|
|
124
|
+
def reserve(
|
|
125
|
+
self, need_gb: float, *, exclude_key: ResidentKey | None = None
|
|
126
|
+
) -> None:
|
|
127
|
+
"""Make budget room for a transient job that loads/frees its own weights.
|
|
128
|
+
|
|
129
|
+
For providers that don't keep a resident model (e.g. mlx-video's LTX-2,
|
|
130
|
+
which loads and frees per call), there's nothing to ``acquire``. Instead
|
|
131
|
+
the handler reserves the peak it will transiently need: we evict residents
|
|
132
|
+
by (priority asc, last_used asc) until ``resident_total + need ≤ limit``.
|
|
133
|
+
Single-flight guarantees nothing else runs while the job uses that room,
|
|
134
|
+
and the provider frees it when done."""
|
|
135
|
+
limit = self.memory_limit_gb
|
|
136
|
+
if need_gb > limit:
|
|
137
|
+
raise MemoryBudgetError(
|
|
138
|
+
f"transient job needs ~{need_gb:.1f} GB but the budget is only {limit:.1f} GB"
|
|
139
|
+
)
|
|
140
|
+
while self._resident_weight_gb(exclude_key=exclude_key) + need_gb > limit:
|
|
141
|
+
victim = self._victim(exclude_key=exclude_key)
|
|
142
|
+
if victim is None:
|
|
143
|
+
raise MemoryBudgetError(
|
|
144
|
+
f"transient job needs ~{need_gb:.1f} GB but the budget is "
|
|
145
|
+
f"{limit:.1f} GB with nothing left to evict"
|
|
146
|
+
)
|
|
147
|
+
self._evict(victim)
|
|
148
|
+
|
|
149
|
+
def stats(self) -> MemoryStats:
|
|
150
|
+
"""Snapshot for /health (safe from the event-loop thread)."""
|
|
151
|
+
now = time.monotonic()
|
|
152
|
+
with self._lock:
|
|
153
|
+
loaded: list[ResidentModelStats] = []
|
|
154
|
+
resident_total = 0.0
|
|
155
|
+
for ld in self._loaded.values():
|
|
156
|
+
ttl = self._effective_ttl(ld.spec)
|
|
157
|
+
idle = round(now - ld.last_used, 1)
|
|
158
|
+
resident_total += ld.weight_gb
|
|
159
|
+
loaded.append(
|
|
160
|
+
ResidentModelStats(
|
|
161
|
+
name=ld.spec.name,
|
|
162
|
+
family=ld.spec.family,
|
|
163
|
+
domain=ld.spec.domain,
|
|
164
|
+
repo=ld.spec.repo,
|
|
165
|
+
weight_gb=round(ld.weight_gb, 1),
|
|
166
|
+
priority=ld.spec.priority,
|
|
167
|
+
idle_s=idle,
|
|
168
|
+
ttl_s=ttl,
|
|
169
|
+
expires_in_s=round(ttl - idle, 1) if ttl is not None else None,
|
|
170
|
+
)
|
|
171
|
+
)
|
|
172
|
+
return MemoryStats(
|
|
173
|
+
loaded=loaded,
|
|
174
|
+
resident_gb=round(resident_total, 1),
|
|
175
|
+
budget_gb=self.memory_limit_gb,
|
|
176
|
+
)
|
|
177
|
+
|
|
178
|
+
def shutdown(self) -> None:
|
|
179
|
+
for key in list(self._loaded):
|
|
180
|
+
self._evict(key)
|
|
181
|
+
|
|
182
|
+
# -- budget / eviction -----------------------------------------------------
|
|
183
|
+
|
|
184
|
+
def _resident_weight_gb(self, exclude_key: ResidentKey | None = None) -> float:
|
|
185
|
+
return sum(
|
|
186
|
+
ld.weight_gb for key, ld in self._loaded.items() if key != exclude_key
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
def _ensure_budget(self, spec: ModelSpec, target_weight_gb: float) -> None:
|
|
190
|
+
limit = self.memory_limit_gb
|
|
191
|
+
need = target_weight_gb + spec.peak_headroom_gb
|
|
192
|
+
while self._resident_weight_gb(exclude_key=spec.key) + need > limit:
|
|
193
|
+
victim = self._victim(exclude_key=spec.key)
|
|
194
|
+
if victim is None:
|
|
195
|
+
raise MemoryBudgetError(
|
|
196
|
+
f"model {spec.name!r} needs ~{need:.1f} GB but the budget is "
|
|
197
|
+
f"{limit:.1f} GB with nothing left to evict"
|
|
198
|
+
)
|
|
199
|
+
self._evict(victim)
|
|
200
|
+
|
|
201
|
+
def _victim(self, *, exclude_key: ResidentKey | None) -> ResidentKey | None:
|
|
202
|
+
cands = [
|
|
203
|
+
(ld.spec.priority, ld.last_used, key)
|
|
204
|
+
for key, ld in self._loaded.items()
|
|
205
|
+
if key != exclude_key
|
|
206
|
+
]
|
|
207
|
+
if not cands:
|
|
208
|
+
return None
|
|
209
|
+
cands.sort() # priority asc, then last_used asc
|
|
210
|
+
return cands[0][2]
|
|
211
|
+
|
|
212
|
+
def _evict(self, key: ResidentKey, *, reason: str = "budget") -> None:
|
|
213
|
+
with self._lock:
|
|
214
|
+
loaded = self._loaded.pop(key, None)
|
|
215
|
+
if loaded is None:
|
|
216
|
+
return
|
|
217
|
+
spec = loaded.spec
|
|
218
|
+
strategy = resolve_framework_strategy(spec.framework)
|
|
219
|
+
before = strategy.active_bytes()
|
|
220
|
+
# Optional per-module cleanup (framework-specific extras) before dropping.
|
|
221
|
+
release = getattr(spec.module, "release", None)
|
|
222
|
+
if release is not None:
|
|
223
|
+
try:
|
|
224
|
+
release(loaded.payload)
|
|
225
|
+
except Exception:
|
|
226
|
+
traceback.print_exc()
|
|
227
|
+
del loaded
|
|
228
|
+
gc.collect()
|
|
229
|
+
strategy.free_caches()
|
|
230
|
+
freed = max(before - strategy.active_bytes(), 0) / _GB
|
|
231
|
+
logger.info(
|
|
232
|
+
"evicted %s (%s) [%s] - freed ~%.1f GB",
|
|
233
|
+
spec.name,
|
|
234
|
+
spec.key,
|
|
235
|
+
reason,
|
|
236
|
+
freed,
|
|
237
|
+
)
|
|
238
|
+
|
|
239
|
+
# -- loading ---------------------------------------------------------------
|
|
240
|
+
|
|
241
|
+
def _load(self, spec: ModelSpec) -> ResidentModel:
|
|
242
|
+
strategy = resolve_framework_strategy(spec.framework)
|
|
243
|
+
logger.info("loading %s (%s) ...", spec.name, spec.key)
|
|
244
|
+
before = strategy.active_bytes()
|
|
245
|
+
payload = spec.module.load(spec)
|
|
246
|
+
measured = max(strategy.active_bytes() - before, 0) / _GB
|
|
247
|
+
# Reconcile estimate with reality; fall back to the estimate if the delta
|
|
248
|
+
# looks bogus (e.g. weights still lazy / not yet materialized).
|
|
249
|
+
weight_gb = measured if measured > 0.5 else spec.weight_gb
|
|
250
|
+
logger.info(
|
|
251
|
+
"loaded %s - weight ~%.1f GB (estimate %.1f, measured %.1f)",
|
|
252
|
+
spec.name,
|
|
253
|
+
weight_gb,
|
|
254
|
+
spec.weight_gb,
|
|
255
|
+
measured,
|
|
256
|
+
)
|
|
257
|
+
return ResidentModel(spec, payload, weight_gb, time.monotonic())
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
from pydantic import Field
|
|
2
|
+
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
3
|
+
from pydantic_settings_manager import SettingsManager
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class MemorySettings(BaseSettings):
|
|
7
|
+
"""Settings for the model residency memory budget and idle TTL shared by all capabilities."""
|
|
8
|
+
|
|
9
|
+
model_config = SettingsConfigDict(
|
|
10
|
+
env_prefix="KIAPI_",
|
|
11
|
+
extra="ignore",
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
memory_limit_gb: float | None = Field(
|
|
15
|
+
default=None,
|
|
16
|
+
title="Memory budget GB",
|
|
17
|
+
description=(
|
|
18
|
+
"Memory budget in GB shared by resident models and runtime peaks across all capabilities.\n"
|
|
19
|
+
"If unset, kiapi uses 80% of the machine's total memory.\n"
|
|
20
|
+
"Before a job starts, lower-priority and older resident models are "
|
|
21
|
+
"released until the job fits within this value."
|
|
22
|
+
),
|
|
23
|
+
)
|
|
24
|
+
|
|
25
|
+
default_ttl_s: float = Field(
|
|
26
|
+
default=1800.0,
|
|
27
|
+
title="Default idle TTL seconds",
|
|
28
|
+
description=(
|
|
29
|
+
"Default number of seconds before an unused resident model is auto-released.\n"
|
|
30
|
+
"Set to 0.0 or less to disable TTL release and evict only under "
|
|
31
|
+
"memory pressure."
|
|
32
|
+
),
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
settings_manager = SettingsManager(MemorySettings)
|
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""The model registry: the catalog of servable models across all families.
|
|
2
|
+
|
|
3
|
+
Public surface (unchanged from the former ``core/model.py``):
|
|
4
|
+
|
|
5
|
+
- :class:`ModelSpec` — a registry entry (variant + family/domain + repo +
|
|
6
|
+
handler module + memory/budget metadata).
|
|
7
|
+
- :class:`ModelRegistry` / the ``model_registry`` singleton — register,
|
|
8
|
+
list, group, and resolve specs.
|
|
9
|
+
- :class:`UnknownModelError` — raised by ``resolve`` on an unknown ``model``.
|
|
10
|
+
|
|
11
|
+
The semantic axes are spelled out as types in :mod:`._types`
|
|
12
|
+
(``ModelName`` / ``ModelFamily`` / ``ModelDomain`` / ``ModelRepo``).
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from ._exceptions.unknown_model_error import UnknownModelError
|
|
16
|
+
from ._schemas.model_spec import ModelSpec
|
|
17
|
+
from ._services.model_registry import ModelRegistry, model_registry
|
|
18
|
+
from ._types.model_alias import ModelAlias
|
|
19
|
+
from ._types.model_domain import ModelDomain
|
|
20
|
+
from ._types.model_family import ModelFamily
|
|
21
|
+
from ._types.model_key import ModelKey
|
|
22
|
+
from ._types.model_name import ModelName
|
|
23
|
+
from ._types.model_repo import ModelRepo
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"ModelAlias",
|
|
27
|
+
"ModelDomain",
|
|
28
|
+
"ModelFamily",
|
|
29
|
+
"ModelKey",
|
|
30
|
+
"ModelName",
|
|
31
|
+
"ModelRegistry",
|
|
32
|
+
"ModelRepo",
|
|
33
|
+
"ModelSpec",
|
|
34
|
+
"UnknownModelError",
|
|
35
|
+
"model_registry",
|
|
36
|
+
]
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
"""A registry entry: everything needed to discover, resolve, load, and budget a
|
|
2
|
+
servable model variant.
|
|
3
|
+
|
|
4
|
+
Each :class:`ModelSpec` ties together:
|
|
5
|
+
|
|
6
|
+
- the public ``name`` (the variant) used in a request's ``model`` field,
|
|
7
|
+
- its ``family`` (resolution namespace + cache-key prefix + OpenAPI topic) and
|
|
8
|
+
``domain`` (grouping),
|
|
9
|
+
- the Hugging Face ``repo`` to load,
|
|
10
|
+
- the per-model **handler module** that owns its own flow,
|
|
11
|
+
- memory estimates (``weight_gb`` / ``peak_headroom_gb``), ``priority``,
|
|
12
|
+
``framework``, ``resident``, ``ttl_seconds`` (see core/memory.py).
|
|
13
|
+
|
|
14
|
+
A handler module must expose:
|
|
15
|
+
|
|
16
|
+
FEATURES : set[str] # modalities/features, family-specific
|
|
17
|
+
load(spec) -> payload # load weights, return an opaque payload
|
|
18
|
+
run(payload, req, settings) -> result # one job's worth of work
|
|
19
|
+
# optional:
|
|
20
|
+
warmup(payload) -> None # prime kernels (else: load only)
|
|
21
|
+
release(payload) -> None # extra cleanup beyond framework default
|
|
22
|
+
|
|
23
|
+
Estimates are intentionally rough; core/memory.py reconciles ``weight_gb`` with
|
|
24
|
+
the real measured footprint after the first load.
|
|
25
|
+
|
|
26
|
+
This is a Pydantic model so the family model-list endpoints
|
|
27
|
+
(``/v1/{domain}/{family}/models``) can serialize specs directly and have their
|
|
28
|
+
fields documented in OpenAPI. The handler ``module`` is excluded from output.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from types import ModuleType
|
|
32
|
+
|
|
33
|
+
from pydantic import BaseModel, ConfigDict, Field, computed_field
|
|
34
|
+
from pydantic.json_schema import SkipJsonSchema
|
|
35
|
+
|
|
36
|
+
from kiapi.core.setup import SetupResource
|
|
37
|
+
|
|
38
|
+
from .._types.model_alias import ModelAlias
|
|
39
|
+
from .._types.model_domain import ModelDomain
|
|
40
|
+
from .._types.model_family import ModelFamily
|
|
41
|
+
from .._types.model_name import ModelName
|
|
42
|
+
from .._types.model_repo import ModelRepo
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class ModelSpec(BaseModel):
|
|
46
|
+
model_config = ConfigDict(arbitrary_types_allowed=True, frozen=True)
|
|
47
|
+
|
|
48
|
+
name: ModelName = Field(
|
|
49
|
+
...,
|
|
50
|
+
description="Public model variant name passed in a request's model field; "
|
|
51
|
+
"named by the variant alone, never repeating the family.",
|
|
52
|
+
examples=["turbo"],
|
|
53
|
+
)
|
|
54
|
+
family: ModelFamily = Field(
|
|
55
|
+
...,
|
|
56
|
+
description="Family that owns the operation vocabulary and resolves this "
|
|
57
|
+
"variant; also used as the API identifier and OpenAPI topic.",
|
|
58
|
+
examples=["zimage"],
|
|
59
|
+
)
|
|
60
|
+
domain: ModelDomain = Field(
|
|
61
|
+
...,
|
|
62
|
+
description="Modality bucket used for discovery and grouping.",
|
|
63
|
+
examples=["image"],
|
|
64
|
+
)
|
|
65
|
+
repo: ModelRepo = Field(
|
|
66
|
+
...,
|
|
67
|
+
description="Hugging Face repo id (or local path) the weights load from.",
|
|
68
|
+
examples=["mflux/z-image-turbo"],
|
|
69
|
+
)
|
|
70
|
+
module: SkipJsonSchema[ModuleType] = Field(
|
|
71
|
+
...,
|
|
72
|
+
exclude=True,
|
|
73
|
+
description="Per-model handler module that owns this model's load/run flow "
|
|
74
|
+
"(internal; not serialized).",
|
|
75
|
+
)
|
|
76
|
+
weight_gb: float = Field(
|
|
77
|
+
...,
|
|
78
|
+
ge=0.0,
|
|
79
|
+
description="Rough estimate of resident weight footprint in GB. "
|
|
80
|
+
"core/memory reconciles this with the measured footprint after first load.",
|
|
81
|
+
examples=[6.0],
|
|
82
|
+
)
|
|
83
|
+
peak_headroom_gb: float = Field(
|
|
84
|
+
...,
|
|
85
|
+
ge=0.0,
|
|
86
|
+
description="Rough estimate of peak memory above the weights during a job, "
|
|
87
|
+
"in GB — the extra budget headroom kept free while running.",
|
|
88
|
+
examples=[4.0],
|
|
89
|
+
)
|
|
90
|
+
framework: str = Field(
|
|
91
|
+
default="mlx",
|
|
92
|
+
description='Backend that owns this model\'s memory (e.g. "mlx", or "rss" '
|
|
93
|
+
"for a subprocess whose footprint kiapi can only see as process RSS).",
|
|
94
|
+
examples=["mlx"],
|
|
95
|
+
)
|
|
96
|
+
priority: int = Field(
|
|
97
|
+
default=0,
|
|
98
|
+
description="Eviction priority. Residents are evicted by (priority asc, "
|
|
99
|
+
"last_used asc); a higher value keeps a model resident under churn.",
|
|
100
|
+
examples=[0],
|
|
101
|
+
)
|
|
102
|
+
aliases: tuple[ModelAlias, ...] = Field(
|
|
103
|
+
default_factory=tuple,
|
|
104
|
+
description="Extra names that also resolve to this spec.",
|
|
105
|
+
examples=[["omni", "qwen3-omni-30b"]],
|
|
106
|
+
)
|
|
107
|
+
default: bool = Field(
|
|
108
|
+
default=False,
|
|
109
|
+
exclude=True,
|
|
110
|
+
description="Whether this is the family's default variant when a request "
|
|
111
|
+
"omits model. If no spec sets it, the first-registered spec is the default "
|
|
112
|
+
"(internal; not serialized).",
|
|
113
|
+
)
|
|
114
|
+
resident: bool = Field(
|
|
115
|
+
default=True,
|
|
116
|
+
description="Whether the model is held resident after loading. Transient "
|
|
117
|
+
"specs (resident=False) load+free per call and reserve budget instead of "
|
|
118
|
+
"acquiring it.",
|
|
119
|
+
examples=[True],
|
|
120
|
+
)
|
|
121
|
+
ttl_seconds: float | None = Field(
|
|
122
|
+
default=None,
|
|
123
|
+
description="Optional idle TTL (seconds). None inherits the global default; "
|
|
124
|
+
"a value <= 0 means never expire (pin resident).",
|
|
125
|
+
examples=[1800.0],
|
|
126
|
+
)
|
|
127
|
+
setup_resources: tuple[SetupResource, ...] = Field(
|
|
128
|
+
default_factory=tuple,
|
|
129
|
+
description="Resources that must be activated before this model can run: "
|
|
130
|
+
"Hugging Face snapshots, Docker images, local paths, URL files, or Python "
|
|
131
|
+
"virtual environments.",
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
@property
|
|
135
|
+
def key(self) -> str:
|
|
136
|
+
"""Stable identity in the resident cache (unique per loaded weights)."""
|
|
137
|
+
return f"{self.family}:{self.repo}"
|
|
138
|
+
|
|
139
|
+
@computed_field( # type: ignore[prop-decorator]
|
|
140
|
+
description="Handler-declared modalities/features (NOT the family).",
|
|
141
|
+
examples=[["text", "image"]],
|
|
142
|
+
)
|
|
143
|
+
@property
|
|
144
|
+
def features(self) -> set[str]:
|
|
145
|
+
return set(getattr(self.module, "FEATURES", set()))
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
"""The model registry: the catalog of servable models across all families.
|
|
2
|
+
|
|
3
|
+
kiapi organizes models on two axes:
|
|
4
|
+
|
|
5
|
+
- ``domain`` — the modality bucket for discovery/grouping: "chat", "embedding",
|
|
6
|
+
"audio", "video", "image".
|
|
7
|
+
- ``family`` — the provider/model family that owns an *operation vocabulary*,
|
|
8
|
+
spelled as a single lowercase token from the upstream's package/model name
|
|
9
|
+
("acestep", "audiogen", "ltx2", "chat", "embedding"). The family is the
|
|
10
|
+
resolution namespace and the capability OpenAPI document.
|
|
11
|
+
|
|
12
|
+
Generation endpoints are organized as ``/v1/<domain>/<family>/<op>`` — each
|
|
13
|
+
family exposes exactly its own operations and request shape (no forced common
|
|
14
|
+
schema). ``model`` selects a *variant within the family* (e.g. acestep
|
|
15
|
+
``xl-base`` / ``turbo``), named by the variant alone (it must not repeat the
|
|
16
|
+
family). chat/embedding stay standardized modality APIs (family == domain).
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from .._exceptions.unknown_model_error import UnknownModelError
|
|
20
|
+
from .._schemas.model_spec import ModelSpec
|
|
21
|
+
from .._types.model_domain import ModelDomain
|
|
22
|
+
from .._types.model_family import ModelFamily
|
|
23
|
+
from .._types.model_key import ModelKey
|
|
24
|
+
from .._types.model_name import ModelName
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class ModelRegistry:
|
|
28
|
+
def __init__(self) -> None:
|
|
29
|
+
self._specs: list[ModelSpec] = []
|
|
30
|
+
self._index: dict[tuple[ModelFamily, ModelKey], ModelSpec] = {}
|
|
31
|
+
|
|
32
|
+
def register(self, spec: ModelSpec) -> None:
|
|
33
|
+
for k in (spec.name, spec.repo, *spec.aliases):
|
|
34
|
+
self._index[(spec.family, k.lower())] = spec
|
|
35
|
+
self._specs.append(spec)
|
|
36
|
+
|
|
37
|
+
def list_specs(self, family: ModelFamily | None = None) -> list[ModelSpec]:
|
|
38
|
+
if family is None:
|
|
39
|
+
return list(self._specs)
|
|
40
|
+
return [s for s in self._specs if s.family == family]
|
|
41
|
+
|
|
42
|
+
def families(self, domain: ModelDomain | None = None) -> list[ModelFamily]:
|
|
43
|
+
seen: list[ModelFamily] = []
|
|
44
|
+
for s in self._specs:
|
|
45
|
+
if domain is not None and s.domain != domain:
|
|
46
|
+
continue
|
|
47
|
+
if s.family not in seen:
|
|
48
|
+
seen.append(s.family)
|
|
49
|
+
return seen
|
|
50
|
+
|
|
51
|
+
def domains(self) -> list[ModelDomain]:
|
|
52
|
+
seen: list[ModelDomain] = []
|
|
53
|
+
for s in self._specs:
|
|
54
|
+
if s.domain not in seen:
|
|
55
|
+
seen.append(s.domain)
|
|
56
|
+
return seen
|
|
57
|
+
|
|
58
|
+
def domain_of(self, family: ModelFamily) -> ModelDomain | None:
|
|
59
|
+
for s in self._specs:
|
|
60
|
+
if s.family == family:
|
|
61
|
+
return s.domain
|
|
62
|
+
return None
|
|
63
|
+
|
|
64
|
+
def default_name(self, family: ModelFamily) -> ModelName | None:
|
|
65
|
+
specs = self.list_specs(family)
|
|
66
|
+
if not specs:
|
|
67
|
+
return None
|
|
68
|
+
for s in specs:
|
|
69
|
+
if s.default:
|
|
70
|
+
return s.name
|
|
71
|
+
return specs[0].name
|
|
72
|
+
|
|
73
|
+
def resolve(self, family: ModelFamily, name: ModelName | None) -> ModelSpec:
|
|
74
|
+
default = self.default_name(family)
|
|
75
|
+
key = (name or default or "").strip().lower()
|
|
76
|
+
spec = self._index.get((family, key))
|
|
77
|
+
if spec is None:
|
|
78
|
+
known = sorted({s.name for s in self.list_specs(family)})
|
|
79
|
+
raise UnknownModelError(
|
|
80
|
+
f"unknown {family} model {name!r}; available: {known}"
|
|
81
|
+
)
|
|
82
|
+
return spec
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
model_registry = ModelRegistry()
|
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
from typing import Literal
|
|
2
|
+
|
|
3
|
+
ModelDomain = Literal["chat", "embedding", "audio", "video", "image", "web"]
|
|
4
|
+
"""The modality bucket used for discovery/grouping. A fixed, closed set — every
|
|
5
|
+
family belongs to exactly one domain. Generation endpoints are organized as
|
|
6
|
+
``/v1/<domain>/<family>/<op>``."""
|
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
type ModelFamily = str
|
|
2
|
+
"""The provider/model family that owns an *operation vocabulary*, spelled as a
|
|
3
|
+
single lowercase token from the upstream's package/model name ("acestep",
|
|
4
|
+
"audiogen", "ltx2", "chat", "embedding"). The family is the resolution
|
|
5
|
+
namespace, the cache-key prefix, and the capability OpenAPI topic."""
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
from .model_alias import ModelAlias
|
|
2
|
+
from .model_name import ModelName
|
|
3
|
+
from .model_repo import ModelRepo
|
|
4
|
+
|
|
5
|
+
type ModelKey = ModelName | ModelRepo | ModelAlias
|
|
6
|
+
"""A lookup token a request may use to address a model within its family — any of
|
|
7
|
+
the spec's ``name``, ``repo``, or ``aliases``. The registry indexes specs by
|
|
8
|
+
``(family, key.lower())``, so resolution is case-insensitive."""
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Networking guards shared across capabilities.
|
|
2
|
+
|
|
3
|
+
Currently a single SSRF guard for user-supplied URLs; see
|
|
4
|
+
:func:`verify_public_url`.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from ._exceptions.unsafe_url_error import UnsafeURLError
|
|
8
|
+
from ._helpers.verify_public_url import verify_public_url
|
|
9
|
+
from ._settings import NetSettings, settings_manager
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"NetSettings",
|
|
13
|
+
"UnsafeURLError",
|
|
14
|
+
"settings_manager",
|
|
15
|
+
"verify_public_url",
|
|
16
|
+
]
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
class UnsafeURLError(ValueError):
|
|
2
|
+
"""A user-supplied URL was rejected before fetching (SSRF guard).
|
|
3
|
+
|
|
4
|
+
Raised for non-http(s) schemes, unresolvable hosts, or hosts that resolve to
|
|
5
|
+
a non-public address (loopback, private, link-local, etc.). Callers map this
|
|
6
|
+
to a 400-class error.
|
|
7
|
+
"""
|