rpr-cli 0.3.3__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/PKG-INFO +2 -2
- rpr_cli-0.3.4/src/rpr/__init__.py +1 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/init.py +6 -9
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/settings.md +1 -63
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/BACKEND_FLOW.md +39 -18
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/in_memory_history_store.py +2 -2
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/postgres_history_store.py +2 -2
- rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/models/dtos/__init__.py +3 -0
- rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/models/dtos/vllm_model_dtos.py +113 -0
- rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/VLLM_SETUP.md +31 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/__init__.py +5 -2
- rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/model_catalog.py +84 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/vllm_agent_factory.py +57 -11
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/vllm_agent_factory_protocol.py +4 -3
- rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/vllm_inference_coordinator.py +207 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/runtimes/agents_sdk.py +17 -16
- rpr_cli-0.3.3/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/vllm_message_mapping.py → rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/runtimes/agents_sdk_message_mapping.py +15 -3
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_agents_sdk_runtime.py +88 -1
- rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_vllm_model_catalog.py +768 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_vllm_provider.py +30 -5
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/domain_container.md +12 -7
- rpr_cli-0.3.4/src/rpr/scaffolds/special_files/domain_settings.md +74 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_add.py +50 -4
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_domain.py +3 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_init.py +11 -0
- rpr_cli-0.3.3/src/rpr/__init__.py +0 -1
- rpr_cli-0.3.3/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/model_catalog.py +0 -106
- rpr_cli-0.3.3/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_vllm_model_catalog.py +0 -168
- rpr_cli-0.3.3/src/rpr/scaffolds/special_files/domain_settings.md +0 -136
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/.github/agents/story.agent.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/.github/workflows/publish-pypi.yml +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/.gitignore +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/CLAUDE.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/GEMINI.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/LICENSE +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/README.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/changelogs/mui_v9.txt +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/code-map.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/nx_uv_setup.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/patch_output.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/plan-languageAwareTemplateAdd.prompt.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/pyproject.toml +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/Cargo.lock +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/Cargo.toml +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/pyproject.toml +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/rpr_parser.pyi +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/config/container.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/config/mod.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/config/settings.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/graph.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/lib.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/dtos/mod.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/entities/mod.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/entities/users.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/mod.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/parse.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/repos/base.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/repos/mod.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/repos/users.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/resolve.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/walker.rs +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/uv.lock +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/special_files.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/approval.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/bootstrap.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/client.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/control.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/mock.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/runtime.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/session.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/base.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/mutating.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/readonly.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/registry.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/catalog.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/chat_service.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/checks.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/cli_adapter.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/completer.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/conversation_service.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/prompt_service.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/selector.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/shell.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/base.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/instructions.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/packages.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/workspace.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/cli.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/add.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/chat.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/check.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/api.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/domain.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/engine.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/storybook.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/ui.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/map.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/refresh.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/rename.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/settings.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/sync.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/update.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/context.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/dependencies.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/dependency_manifest_data.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/base.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/claude.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/codex.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/copilot.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/cursor.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/gemini.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/github.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/manifest_refresh.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/architecture.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/chains.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/classifier.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/coverage.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/dependencies.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/extractor.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/graph.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/output.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/responsibility.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/topology.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/walker.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/package_updates.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/project_rename.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/base_entity.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/base_repo.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/container.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/all.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/api.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/domain.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/frontend.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/rust-engine.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/setup-guide.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/tooling-setup.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/tooling.instructions.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/README.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/content.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/httpChatStreamClient.spec.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/httpChatStreamClient.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/types.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/ChatComposer.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/ChatComposer.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/__screenshots__/ChatComposer.spec.tsx/ChatComposer-renders-custom-action-buttons-beside-the-attachment-control-1.png +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/fileDrop.spec.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/fileDrop.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/ChatComposerMentionTokens.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/ChatSuggestionsPopover.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/LocalFileSuggestionAction.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/localFileSuggestions.spec.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/localFileSuggestions.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/triggerDetection.spec.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/triggerDetection.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/types.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/useChatSuggestions.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/types.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/useChatComposer.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/useChatStream.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/useChatStream.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperience.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperience.styles.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperience.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceComposer.styles.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceComposer.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceHeader.styles.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceHeader.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ThreadHistory.styles.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ThreadHistory.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/chatStorage.spec.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/chatStorage.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/types.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/useChatExperience.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/useChatExperience.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/testing/createMockChatStreamClient.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/testing/createMockSuggestionProvider.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/testing/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessage.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessageActions.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessageContent.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessageContent.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatThread.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatThread.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/README.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/contentFormat.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/streaming.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/types.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponent.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponent.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponentList.spec.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponentList.tsx +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/index.ts +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/fetch.service.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/sticky-navigation.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/sticky-navigation.spec.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/vitest-browser-config.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/api/routes/chat_routes.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/api/tests/test_chat.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/models/dtos/chat_dtos.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/models/dtos/conversation_history_dtos.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/models/entities/conversation_history.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/repos/conversation_history.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/agent_session_codec.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/chat_service.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/conversation_history_service.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/conversation_operations.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/history_codec.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/protocols/conversation_history_store.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/protocols/conversation_runtime.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/protocols/llm_provider.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_chat_dtos.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_chat_service.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_history_codec.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_history_projection.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_persistence_config.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_postgres_history_store.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/runtimes/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/camel_case_model.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/dto_util.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/encrypted_column.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/logging.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/mapper_util.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/partial_update.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/test_encrypted_column.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/templates/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/templates/registry.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/console.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/markdown.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/prompt_session.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/renderers.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/theme.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/workspace.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/README.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/chat-attachment-upload-gap.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/00-bootstrap-cli.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/01-rpr-init.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/02-rpr-generate-domain.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/03-rpr-generate-api.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/04-rpr-generate-ui.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/05-rpr-generate-engine.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/06-rpr-generate-storybook.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/07-rpr-sync-rules.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/08-rpr-add-template.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/09-rpr-check.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/10-rpr-domain-config.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/18-rpr-terminal-ui-foundation.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/19-rpr-markdown-rendering.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/20-rpr-chat-repl.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/21-rpr-agent-tool-loop.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/22-rpr-chat-tui.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/23-rpr-chat-unified-command-shell.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/24-rpr-async-chat-runtime-and-services.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/25-rpr-prompt-service-and-conversation-lifecycle.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/26-rpr-shared-command-catalog.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/27-rpr-slash-command-selection-panel.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/28-rpr-chat-cancellable-and-multiline-input.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/29-rpr-file-context-selection.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/11-rpr-map-command.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/12-rpr-map-file-walker.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/13-rpr-map-rust-engine.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/13a-rpr-map-gitignore.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/14-rpr-map-graph-builder.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/15-rpr-map-analysis-bridge.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/16-rpr-map-layer-classifier.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17-rpr-map-output.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17a-rpr-map-directory-topology.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17b-rpr-map-architecture-assessment.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17c-rpr-map-representative-chains.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17d-rpr-map-multi-resolution-output.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/map/30-rpr-map-directory-view-and-cycle-detection.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/map/31-rpr-map-coupling-hotspots-and-verdict-reasoning.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/map/32-rpr-map-dependencies-responsibilities-and-test-coverage.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/package-maintenance/default-vite-8-oxc-toolchain.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/package-maintenance/rpr-update-packages.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/package-maintenance/typescript-7-frontend-scaffold-compatibility.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/__init__.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_architecture.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_architecture_unwrapped.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_catalog.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_chat.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_chat_service.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_check.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_completer.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_context.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_api.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_engine.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_storybook.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_ui.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_chains.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_classifier.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_coverage.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_dependencies.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_extractor.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_graph.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_output.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_responsibility.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_topology.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_walker.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_package_updates.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_packaging.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_prompt_session.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_refresh.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_rename.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_rpr_parser_stub.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_selector.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_settings.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_sync.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_ui.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_update_packages.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/untitled:plan-chatAttachmentUploadGap.prompt.md +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/update_output.py +0 -0
- {rpr_cli-0.3.3 → rpr_cli-0.3.4}/uv.lock +0 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.3.4"
|
|
@@ -727,8 +727,8 @@ uv.lock
|
|
|
727
727
|
ctx.write_file(ctx.project_dir / ".gitignore", content)
|
|
728
728
|
|
|
729
729
|
|
|
730
|
-
def _write_env_example(ctx: RunContext) -> None:
|
|
731
|
-
content = """# Database
|
|
730
|
+
def _write_env_example(ctx: RunContext, name_underscore: str) -> None:
|
|
731
|
+
content = f"""# Database
|
|
732
732
|
DATABASE_URL=
|
|
733
733
|
|
|
734
734
|
# Auth
|
|
@@ -742,13 +742,10 @@ FERNET_KEY=
|
|
|
742
742
|
API_BASE_URL=http://localhost:8000
|
|
743
743
|
|
|
744
744
|
# vLLM
|
|
745
|
-
VLLM_BASE_URL=http://localhost:8000/v1
|
|
746
745
|
VLLM_API_KEY=EMPTY
|
|
747
|
-
|
|
748
|
-
#
|
|
749
|
-
#
|
|
750
|
-
# VLLM_MODEL_CATALOG=[{"id":"gemma","label":"Gemma","modelId":"google/gemma","baseUrl":"http://localhost:8000/v1","reasoningParser":"gemma4","supportsThinking":true}]
|
|
751
|
-
# VLLM_DEFAULT_MODEL_PROFILE_ID=gemma
|
|
746
|
+
# Model profiles, the default profile, restoration policy, and awake-model
|
|
747
|
+
# capacity are maintained in
|
|
748
|
+
# packages/python/{name_underscore}_domain/src/{name_underscore}_domain/services/agents/model_catalog.py.
|
|
752
749
|
VLLM_THINKING_EFFORT=high
|
|
753
750
|
"""
|
|
754
751
|
ctx.write_file(ctx.project_dir / ".env.example", content)
|
|
@@ -893,7 +890,7 @@ def init(
|
|
|
893
890
|
_write_dependency_cruiser_config(ctx)
|
|
894
891
|
_write_pre_commit_config(ctx)
|
|
895
892
|
_write_gitignore(ctx)
|
|
896
|
-
_write_env_example(ctx)
|
|
893
|
+
_write_env_example(ctx, variables["name_underscore"])
|
|
897
894
|
|
|
898
895
|
# ------------------------------------------------------------------
|
|
899
896
|
# Step 5 – AI instruction files
|
|
@@ -9,54 +9,10 @@ The first Python block is the generated file used by `rpr generate domain` and `
|
|
|
9
9
|
```python
|
|
10
10
|
from __future__ import annotations
|
|
11
11
|
|
|
12
|
-
from pydantic import
|
|
12
|
+
from pydantic import Field
|
|
13
13
|
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
14
14
|
|
|
15
15
|
|
|
16
|
-
class VllmModelProfileSettings(BaseModel):
|
|
17
|
-
"""One vLLM deployment available to the chat application."""
|
|
18
|
-
|
|
19
|
-
id: str = Field(description="Stable public model-profile identifier.")
|
|
20
|
-
label: str = Field(description="User-facing model label.")
|
|
21
|
-
model_id: str = Field(
|
|
22
|
-
validation_alias=AliasChoices("modelId", "model_id"),
|
|
23
|
-
description="Model identifier sent to the vLLM Responses API.",
|
|
24
|
-
)
|
|
25
|
-
base_url: str = Field(
|
|
26
|
-
validation_alias=AliasChoices("baseUrl", "base_url"),
|
|
27
|
-
description="OpenAI-compatible vLLM endpoint for this profile.",
|
|
28
|
-
)
|
|
29
|
-
reasoning_parser: str | None = Field(
|
|
30
|
-
default=None,
|
|
31
|
-
validation_alias=AliasChoices("reasoningParser", "reasoning_parser"),
|
|
32
|
-
description="Documented vLLM parser required by this deployment.",
|
|
33
|
-
)
|
|
34
|
-
supports_thinking: bool = Field(
|
|
35
|
-
default=False,
|
|
36
|
-
validation_alias=AliasChoices("supportsThinking", "supports_thinking"),
|
|
37
|
-
description="Whether this deployment emits native reasoning output.",
|
|
38
|
-
)
|
|
39
|
-
description: str | None = Field(
|
|
40
|
-
default=None, description="Optional user-facing model description."
|
|
41
|
-
)
|
|
42
|
-
|
|
43
|
-
provider_properties: dict[str, JsonValue] = Field(
|
|
44
|
-
default_factory=dict,
|
|
45
|
-
validation_alias=AliasChoices("providerProperties", "provider_properties"),
|
|
46
|
-
description="vLLM-specific deployment properties not exposed by the chat API.",
|
|
47
|
-
)
|
|
48
|
-
|
|
49
|
-
@model_validator(mode="after")
|
|
50
|
-
def validate_execution_mode(self) -> VllmModelProfileSettings:
|
|
51
|
-
"""Validate the optional vLLM execution mode stored in provider properties."""
|
|
52
|
-
mode = self.provider_properties.get("mode", "online")
|
|
53
|
-
if not isinstance(mode, str) or mode not in {"online", "offline"}:
|
|
54
|
-
raise ValueError(
|
|
55
|
-
"VLLM providerProperties.mode must be either 'online' or 'offline'."
|
|
56
|
-
)
|
|
57
|
-
return self
|
|
58
|
-
|
|
59
|
-
|
|
60
16
|
class Settings(BaseSettings):
|
|
61
17
|
model_config = SettingsConfigDict(
|
|
62
18
|
env_file=".env",
|
|
@@ -96,28 +52,10 @@ class Settings(BaseSettings):
|
|
|
96
52
|
default=20,
|
|
97
53
|
description="Maximum keepalive HTTP connections",
|
|
98
54
|
)
|
|
99
|
-
vllm_base_url: str = Field(
|
|
100
|
-
default="http://localhost:8000/v1",
|
|
101
|
-
description="OpenAI-compatible vLLM server base URL",
|
|
102
|
-
)
|
|
103
55
|
vllm_api_key: str = Field(
|
|
104
56
|
default="EMPTY",
|
|
105
57
|
description="API key sent to vLLM model servers",
|
|
106
58
|
)
|
|
107
|
-
vllm_model_id: str = Field(
|
|
108
|
-
default="google/gemma-4-12B-it-qat-q4_0-gguf",
|
|
109
|
-
description="Legacy default vLLM model when no catalog is configured",
|
|
110
|
-
)
|
|
111
|
-
vllm_model_catalog: list[VllmModelProfileSettings] = Field(
|
|
112
|
-
default_factory=list,
|
|
113
|
-
validation_alias="VLLM_MODEL_CATALOG",
|
|
114
|
-
description="Configured public model profiles and their vLLM endpoints.",
|
|
115
|
-
)
|
|
116
|
-
vllm_default_model_profile_id: str | None = Field(
|
|
117
|
-
default=None,
|
|
118
|
-
validation_alias="VLLM_DEFAULT_MODEL_PROFILE_ID",
|
|
119
|
-
description="Public ID of the default catalog profile.",
|
|
120
|
-
)
|
|
121
59
|
vllm_thinking_effort: str = Field(
|
|
122
60
|
default="high",
|
|
123
61
|
description="vLLM thinking effort for reasoning-capable profiles",
|
|
@@ -99,21 +99,25 @@ flowchart LR
|
|
|
99
99
|
|
|
100
100
|
runtimePort("ConversationRuntimeProtocol\nprovider-neutral execution contract")
|
|
101
101
|
runtime("runtimes/agents_sdk.py\nAgentsSdkConversationRuntime")
|
|
102
|
-
factoryPort("VllmAgentFactoryProtocol\
|
|
102
|
+
factoryPort("VllmAgentFactoryProtocol\nacquire an Agents SDK Agent")
|
|
103
103
|
factory("agents/vllm_agent_factory.py\nVllmAgentFactory")
|
|
104
|
+
coordinator("agents/vllm_inference_coordinator.py\nVllmInferenceCoordinator")
|
|
104
105
|
catalog("agents/model_catalog.py\nVllmModelCatalog")
|
|
105
106
|
vllm("OpenAI-compatible vLLM\nResponses API")
|
|
106
107
|
offline("Offline executor\nnot registered yet")
|
|
107
108
|
|
|
108
109
|
runtimePort -. "implemented by" .-> runtime
|
|
109
|
-
runtime -->|"
|
|
110
|
+
runtime -->|"reserve selected model for stream"| factoryPort
|
|
110
111
|
factoryPort -. "implemented by" .-> factory
|
|
111
|
-
factory -->|"resolve
|
|
112
|
+
factory -->|"resolve and create cached agent"| catalog
|
|
113
|
+
factory -->|"reserve profile"| coordinator
|
|
114
|
+
coordinator -->|"capacity and priority policy"| catalog
|
|
115
|
+
coordinator -->|"sleep-state control"| vllm
|
|
112
116
|
factory -->|"online profile: Agent + AsyncOpenAI"| vllm
|
|
113
|
-
|
|
117
|
+
coordinator -. "offline profile: raises\nVllmOfflineModeNotImplementedError" .-> offline
|
|
114
118
|
|
|
115
119
|
class runtimePort,factoryPort protocol;
|
|
116
|
-
class runtime,factory,catalog,vllm concrete;
|
|
120
|
+
class runtime,factory,coordinator,catalog,vllm concrete;
|
|
117
121
|
class offline future;
|
|
118
122
|
```
|
|
119
123
|
|
|
@@ -122,7 +126,9 @@ flowchart LR
|
|
|
122
126
|
If it still uses the Agents SDK but needs a different model vendor, replace or
|
|
123
127
|
extend `VllmAgentFactory` behind `VllmAgentFactoryProtocol` and update
|
|
124
128
|
`Container.vllm_agent_factory`. The current offline branch stops in
|
|
125
|
-
`VllmAgentFactory.
|
|
129
|
+
`VllmAgentFactory.acquire()` before inference starts. Keep the acquisition
|
|
130
|
+
context open for the complete response stream so a deployment cannot be slept
|
|
131
|
+
while it is still generating.
|
|
126
132
|
|
|
127
133
|
### 4. History and agent-session storage
|
|
128
134
|
|
|
@@ -159,9 +165,9 @@ flowchart LR
|
|
|
159
165
|
runtime, also replace the session factory dependency used by
|
|
160
166
|
`AgentsSdkConversationRuntime`.
|
|
161
167
|
|
|
162
|
-
### 5. Model catalog:
|
|
168
|
+
### 5. Model catalog: Python profiles to picker
|
|
163
169
|
|
|
164
|
-
**Purpose:** the catalog is the allow-list between
|
|
170
|
+
**Purpose:** the catalog is the allow-list between Python-owned configuration, the
|
|
165
171
|
browser picker, and agent creation. It keeps base URLs, parser configuration, and
|
|
166
172
|
provider-only properties on the server.
|
|
167
173
|
|
|
@@ -173,30 +179,38 @@ flowchart LR
|
|
|
173
179
|
classDef api fill:#fef3c7,stroke:#d97706,color:#78350f,stroke-width:2px;
|
|
174
180
|
classDef ui fill:#dbeafe,stroke:#2563eb,color:#172554,stroke-width:2px;
|
|
175
181
|
|
|
176
|
-
|
|
177
|
-
|
|
182
|
+
profiles("agents/model_catalog.py\nPython catalog DTO + policy")
|
|
183
|
+
dto("models/dtos/vllm_model_dtos.py\nVllmModelCatalogDto / VllmModelProfileDto")
|
|
178
184
|
container("config/container.py\nContainer.vllm_model_catalog")
|
|
179
185
|
catalog("agents/model_catalog.py\nVllmModelCatalog")
|
|
180
186
|
route("chat_routes.py\nGET /api/chat/models")
|
|
181
187
|
loader("services/chatModels.ts\nloadChatModels()")
|
|
182
188
|
picker("MainPage.tsx / ChatExperience\nmodels + initialModelId")
|
|
183
189
|
|
|
184
|
-
|
|
185
|
-
|
|
190
|
+
profiles --> catalog
|
|
191
|
+
dto -->|"validates internal profiles"| catalog
|
|
186
192
|
container --> catalog
|
|
187
193
|
catalog -->|"public_profiles(): safe metadata"| route
|
|
188
194
|
route --> loader
|
|
189
195
|
loader --> picker
|
|
190
196
|
|
|
191
|
-
class
|
|
192
|
-
class container,catalog domain;
|
|
197
|
+
class profiles config;
|
|
198
|
+
class dto,container,catalog domain;
|
|
193
199
|
class route api;
|
|
194
200
|
class loader,picker ui;
|
|
195
201
|
```
|
|
196
202
|
|
|
197
|
-
To add a selectable model, update
|
|
198
|
-
|
|
199
|
-
|
|
203
|
+
To add a selectable model, update `VLLM_MODEL_CATALOG` in
|
|
204
|
+
`agents/model_catalog.py`. Set exactly one profile's `is_default=True` and give
|
|
205
|
+
each profile a one-based `priority` (`1` is highest). Set `max_awake_models` to
|
|
206
|
+
the number of deployments the host can retain in memory; then verify the catalog
|
|
207
|
+
profile resolves and `GET /api/chat/models` returns the safe metadata used by
|
|
208
|
+
`MainPage.tsx`.
|
|
209
|
+
|
|
210
|
+
Every online profile must point to a vLLM server started with
|
|
211
|
+
`VLLM_SERVER_DEV_MODE=1` and `--enable-sleep-mode`. The coordinator uses
|
|
212
|
+
`/is_sleeping`, `/sleep`, and `/wake_up`; keep these development-only endpoints
|
|
213
|
+
local or otherwise private.
|
|
200
214
|
|
|
201
215
|
### 6. Protocol-to-concrete registration map
|
|
202
216
|
|
|
@@ -253,6 +267,7 @@ sequenceDiagram
|
|
|
253
267
|
participant Store as PostgresConversationHistoryStore
|
|
254
268
|
participant Runtime as AgentsSdkConversationRuntime
|
|
255
269
|
participant Factory as VllmAgentFactory
|
|
270
|
+
participant Coordinator as VllmInferenceCoordinator
|
|
256
271
|
participant Session as PostgresAgentSessionFactory
|
|
257
272
|
participant VLLM as vLLM Responses API
|
|
258
273
|
|
|
@@ -269,9 +284,13 @@ sequenceDiagram
|
|
|
269
284
|
Store-->>Service: turn_id
|
|
270
285
|
Service->>Runtime: stream(ConversationExecution)
|
|
271
286
|
activate Runtime
|
|
272
|
-
Runtime->>Factory:
|
|
287
|
+
Runtime->>Factory: acquire(model_id, thinking_effort)
|
|
273
288
|
Factory->>Catalog: resolve(model_id)
|
|
274
289
|
alt profile mode is online
|
|
290
|
+
Factory->>Coordinator: reserve_profile(profile_id)
|
|
291
|
+
Coordinator->>Catalog: profiles + max_awake_models
|
|
292
|
+
Coordinator->>VLLM: inspect, sleep lower-priority deployment, wake target
|
|
293
|
+
Coordinator-->>Factory: reserved profile
|
|
275
294
|
Factory-->>Runtime: cached/new Agent and AsyncOpenAI client
|
|
276
295
|
Runtime->>Session: create(execution)
|
|
277
296
|
Session-->>Runtime: persisted Agents SDK session
|
|
@@ -283,6 +302,8 @@ sequenceDiagram
|
|
|
283
302
|
Route-->>UI: SSE thinking event or data frame
|
|
284
303
|
end
|
|
285
304
|
Runtime-->>Service: ConversationThinkingCompleted + ConversationCompleted
|
|
305
|
+
Runtime->>Factory: release acquisition
|
|
306
|
+
Factory->>Coordinator: restore default when configured
|
|
286
307
|
Service->>Store: finish(..., complete)
|
|
287
308
|
Store-->>Service: persisted transcript and session ledger
|
|
288
309
|
else profile mode is offline
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
3
|
import asyncio
|
|
4
|
-
from collections.abc import
|
|
4
|
+
from collections.abc import AsyncGenerator
|
|
5
5
|
from contextlib import asynccontextmanager
|
|
6
6
|
|
|
7
7
|
from {name_underscore}_domain.services.protocols.llm_provider import ChatConversationTurn
|
|
@@ -16,7 +16,7 @@ class InMemoryConversationHistoryStore:
|
|
|
16
16
|
self._locks: dict[str, asyncio.Lock] = {}
|
|
17
17
|
|
|
18
18
|
@asynccontextmanager
|
|
19
|
-
async def serialized(self, thread_id: str) ->
|
|
19
|
+
async def serialized(self, thread_id: str) -> AsyncGenerator[None, None]:
|
|
20
20
|
lock = self._locks.setdefault(thread_id, asyncio.Lock())
|
|
21
21
|
async with lock:
|
|
22
22
|
yield
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
3
|
import asyncio
|
|
4
|
-
from collections.abc import
|
|
4
|
+
from collections.abc import AsyncGenerator
|
|
5
5
|
from contextlib import asynccontextmanager
|
|
6
6
|
|
|
7
7
|
from sqlalchemy import text
|
|
@@ -34,7 +34,7 @@ class PostgresConversationHistoryStore:
|
|
|
34
34
|
configure_encrypted_columns(encryption_key)
|
|
35
35
|
|
|
36
36
|
@asynccontextmanager
|
|
37
|
-
async def serialized(self, thread_id: str) ->
|
|
37
|
+
async def serialized(self, thread_id: str) -> AsyncGenerator[None, None]:
|
|
38
38
|
connection = await self._engine.connect()
|
|
39
39
|
acquired = False
|
|
40
40
|
try:
|
rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/models/dtos/vllm_model_dtos.py
ADDED
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import Literal, Self, cast
|
|
4
|
+
|
|
5
|
+
from pydantic import Field, JsonValue, model_validator
|
|
6
|
+
|
|
7
|
+
from {name_underscore}_domain.models.dtos.chat_dtos import ChatModelProfileDto
|
|
8
|
+
from {name_underscore}_domain.utils.camel_case_model import CamelCaseModel
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class VllmModelProfileDto(CamelCaseModel):
|
|
12
|
+
"""Internal configuration for one vLLM model deployment."""
|
|
13
|
+
|
|
14
|
+
id: str = Field(description="Stable public model-profile identifier.")
|
|
15
|
+
label: str = Field(description="User-facing model label.")
|
|
16
|
+
model_id: str = Field(
|
|
17
|
+
description="Model identifier sent to the vLLM Responses API."
|
|
18
|
+
)
|
|
19
|
+
base_url: str = Field(
|
|
20
|
+
description="OpenAI-compatible vLLM endpoint for this profile."
|
|
21
|
+
)
|
|
22
|
+
reasoning_parser: str | None = Field(
|
|
23
|
+
default=None,
|
|
24
|
+
description="Documented vLLM parser required by this deployment.",
|
|
25
|
+
)
|
|
26
|
+
supports_thinking: bool = Field(
|
|
27
|
+
default=False,
|
|
28
|
+
description="Whether this deployment emits native reasoning output.",
|
|
29
|
+
)
|
|
30
|
+
supports_chat: bool = Field(
|
|
31
|
+
default=True,
|
|
32
|
+
description="Whether this deployment can be selected by the chat runtime.",
|
|
33
|
+
)
|
|
34
|
+
supports_ocr: bool = Field(
|
|
35
|
+
default=False,
|
|
36
|
+
description="Whether this deployment is available to an OCR service.",
|
|
37
|
+
)
|
|
38
|
+
is_default: bool = Field(
|
|
39
|
+
default=False,
|
|
40
|
+
description="Whether this is the default chat model profile.",
|
|
41
|
+
)
|
|
42
|
+
priority: int = Field(
|
|
43
|
+
default=1,
|
|
44
|
+
ge=1,
|
|
45
|
+
description="Awake-model retention priority; 1 is the highest priority.",
|
|
46
|
+
)
|
|
47
|
+
description: str | None = Field(
|
|
48
|
+
default=None,
|
|
49
|
+
description="Optional user-facing model description.",
|
|
50
|
+
)
|
|
51
|
+
provider_properties: dict[str, JsonValue] = Field(
|
|
52
|
+
default_factory=dict,
|
|
53
|
+
description="vLLM-specific deployment properties kept inside the domain.",
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
@model_validator(mode="after")
|
|
57
|
+
def validate_execution_mode(self) -> Self:
|
|
58
|
+
"""Validate the optional execution mode stored in provider properties."""
|
|
59
|
+
mode = self.provider_properties.get("mode", "online")
|
|
60
|
+
if not isinstance(mode, str) or mode not in {"online", "offline"}:
|
|
61
|
+
raise ValueError(
|
|
62
|
+
"VLLM providerProperties.mode must be either 'online' or 'offline'."
|
|
63
|
+
)
|
|
64
|
+
return self
|
|
65
|
+
|
|
66
|
+
@property
|
|
67
|
+
def execution_mode(self) -> Literal["online", "offline"]:
|
|
68
|
+
"""Return the configured execution mode, defaulting to online."""
|
|
69
|
+
return cast(
|
|
70
|
+
Literal["online", "offline"],
|
|
71
|
+
self.provider_properties.get("mode", "online"),
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
def to_chat_model_profile_dto(self, *, is_default: bool) -> ChatModelProfileDto:
|
|
75
|
+
"""Map internal deployment data to safe model-picker metadata."""
|
|
76
|
+
return ChatModelProfileDto(
|
|
77
|
+
id=self.id,
|
|
78
|
+
label=self.label,
|
|
79
|
+
description=self.description,
|
|
80
|
+
is_default=is_default,
|
|
81
|
+
supports_thinking=self.supports_thinking,
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
class VllmModelCatalogDto(CamelCaseModel):
|
|
86
|
+
"""Internal policy and profile data for all configured vLLM deployments."""
|
|
87
|
+
|
|
88
|
+
profiles: list[VllmModelProfileDto] = Field(
|
|
89
|
+
min_length=1,
|
|
90
|
+
description="Configured vLLM model profiles.",
|
|
91
|
+
)
|
|
92
|
+
max_awake_models: int = Field(
|
|
93
|
+
default=1,
|
|
94
|
+
ge=1,
|
|
95
|
+
description="Maximum number of distinct online vLLM deployments kept awake.",
|
|
96
|
+
)
|
|
97
|
+
restore_default_after_request: bool = Field(
|
|
98
|
+
default=True,
|
|
99
|
+
description="Whether to reactivate the default profile after another request.",
|
|
100
|
+
)
|
|
101
|
+
|
|
102
|
+
@model_validator(mode="after")
|
|
103
|
+
def validate_default_profile(self) -> Self:
|
|
104
|
+
"""Require one chat model profile to be designated as the default."""
|
|
105
|
+
if sum(profile.is_default for profile in self.profiles) != 1:
|
|
106
|
+
raise ValueError(
|
|
107
|
+
"The VLLM catalog must contain exactly one default profile."
|
|
108
|
+
)
|
|
109
|
+
if not next(
|
|
110
|
+
profile for profile in self.profiles if profile.is_default
|
|
111
|
+
).supports_chat:
|
|
112
|
+
raise ValueError("The default VLLM profile must support chat.")
|
|
113
|
+
return self
|
rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/VLLM_SETUP.md
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# vLLM model catalog and sleep coordination
|
|
2
|
+
|
|
3
|
+
Model profiles, the default profile, awake-model capacity, and restoration policy
|
|
4
|
+
are defined in `model_catalog.py`. Do not put model IDs or base URLs in `.env`.
|
|
5
|
+
|
|
6
|
+
Start every online catalog deployment with vLLM development mode and sleep mode
|
|
7
|
+
enabled. For the generated Gemma profile:
|
|
8
|
+
|
|
9
|
+
```sh
|
|
10
|
+
VLLM_WSL2_ENABLE_PIN_MEMORY=1 \
|
|
11
|
+
VLLM_SERVER_DEV_MODE=1 \
|
|
12
|
+
vllm serve google/gemma-4-12B-it-qat-w4a16-ct \
|
|
13
|
+
--max-model-len 8192 \
|
|
14
|
+
--gpu-memory-utilization 0.70 \
|
|
15
|
+
--max-num-seqs 1 \
|
|
16
|
+
--enable-auto-tool-choice \
|
|
17
|
+
--tool-call-parser gemma4 \
|
|
18
|
+
--reasoning-parser gemma4 \
|
|
19
|
+
--enable-sleep-mode \
|
|
20
|
+
--port 8000
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
The coordinator calls `/is_sleeping`, `/sleep`, and `/wake_up` on each online
|
|
24
|
+
deployment to enforce `max_awake_models`. Keep these development-only endpoints
|
|
25
|
+
local or otherwise private. Lower numeric `priority` values are retained first;
|
|
26
|
+
`1` is the highest priority. When `restore_default_after_request` is enabled, a
|
|
27
|
+
temporary profile is released and the default profile is restored after the
|
|
28
|
+
complete response stream.
|
|
29
|
+
|
|
30
|
+
Each reasoning-capable profile must match its server's `--reasoning-parser`
|
|
31
|
+
setting. The generated runtime sends `reasoning.effort` per request.
|
|
@@ -1,15 +1,18 @@
|
|
|
1
|
-
from .model_catalog import UnknownModelProfileError, VllmModelCatalog
|
|
1
|
+
from .model_catalog import UnknownModelProfileError, VllmModelCatalog
|
|
2
2
|
from .vllm_agent_factory import (
|
|
3
3
|
VllmAgentFactory,
|
|
4
|
+
VllmModelSwitchError,
|
|
4
5
|
VllmOfflineModeNotImplementedError,
|
|
5
6
|
)
|
|
6
7
|
from .vllm_agent_factory_protocol import VllmAgentFactoryProtocol
|
|
8
|
+
from .vllm_inference_coordinator import VllmInferenceCoordinator
|
|
7
9
|
|
|
8
10
|
__all__ = [
|
|
9
11
|
"UnknownModelProfileError",
|
|
10
12
|
"VllmAgentFactory",
|
|
11
13
|
"VllmAgentFactoryProtocol",
|
|
12
14
|
"VllmModelCatalog",
|
|
13
|
-
"
|
|
15
|
+
"VllmInferenceCoordinator",
|
|
16
|
+
"VllmModelSwitchError",
|
|
14
17
|
"VllmOfflineModeNotImplementedError",
|
|
15
18
|
]
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from {name_underscore}_domain.models.dtos.chat_dtos import ChatModelProfileDto
|
|
4
|
+
from {name_underscore}_domain.models.dtos.vllm_model_dtos import (
|
|
5
|
+
VllmModelCatalogDto,
|
|
6
|
+
VllmModelProfileDto,
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
VLLM_MODEL_CATALOG = VllmModelCatalogDto(
|
|
10
|
+
max_awake_models=1,
|
|
11
|
+
restore_default_after_request=True,
|
|
12
|
+
profiles=[
|
|
13
|
+
VllmModelProfileDto(
|
|
14
|
+
id="gemma",
|
|
15
|
+
label="Gemma 4 12B",
|
|
16
|
+
model_id="google/gemma-4-12B-it-qat-w4a16-ct",
|
|
17
|
+
base_url="http://localhost:8000/v1",
|
|
18
|
+
reasoning_parser="gemma4",
|
|
19
|
+
supports_thinking=True,
|
|
20
|
+
is_default=True,
|
|
21
|
+
priority=1,
|
|
22
|
+
),
|
|
23
|
+
],
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class UnknownModelProfileError(ValueError):
|
|
28
|
+
"""Raised when a request refers to a model profile unavailable to this app."""
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class VllmModelCatalog:
|
|
32
|
+
"""Returns and resolves the Python-owned vLLM deployment catalog."""
|
|
33
|
+
|
|
34
|
+
def __init__(
|
|
35
|
+
self,
|
|
36
|
+
catalog: VllmModelCatalogDto | None = None,
|
|
37
|
+
) -> None:
|
|
38
|
+
self._catalog = (catalog or VLLM_MODEL_CATALOG).model_copy(deep=True)
|
|
39
|
+
configured_profiles = tuple(self._catalog.profiles)
|
|
40
|
+
|
|
41
|
+
profile_ids = [profile.id for profile in configured_profiles]
|
|
42
|
+
if len(set(profile_ids)) != len(profile_ids):
|
|
43
|
+
raise ValueError("VLLM model profile IDs must be unique.")
|
|
44
|
+
|
|
45
|
+
self._profiles = {profile.id: profile for profile in configured_profiles}
|
|
46
|
+
self._default_profile_id = next(
|
|
47
|
+
profile.id for profile in configured_profiles if profile.is_default
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
@property
|
|
51
|
+
def default_profile_id(self) -> str:
|
|
52
|
+
return self._default_profile_id
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def catalog(self) -> VllmModelCatalogDto:
|
|
56
|
+
"""Return a copied catalog configuration and model profile list."""
|
|
57
|
+
return self._catalog.model_copy(deep=True)
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def max_awake_models(self) -> int:
|
|
61
|
+
return self._catalog.max_awake_models
|
|
62
|
+
|
|
63
|
+
@property
|
|
64
|
+
def restore_default_after_request(self) -> bool:
|
|
65
|
+
return self._catalog.restore_default_after_request
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def profiles(self) -> tuple[VllmModelProfileDto, ...]:
|
|
69
|
+
"""Return configured profiles for internal deployment coordination."""
|
|
70
|
+
return tuple(self._profiles.values())
|
|
71
|
+
|
|
72
|
+
def resolve(self, model_id: str | None) -> VllmModelProfileDto:
|
|
73
|
+
profile_id = model_id or self._default_profile_id
|
|
74
|
+
try:
|
|
75
|
+
return self._profiles[profile_id]
|
|
76
|
+
except KeyError as error:
|
|
77
|
+
raise UnknownModelProfileError(profile_id) from error
|
|
78
|
+
|
|
79
|
+
def public_profiles(self) -> list[ChatModelProfileDto]:
|
|
80
|
+
return [
|
|
81
|
+
profile.to_chat_model_profile_dto(is_default=profile.is_default)
|
|
82
|
+
for profile in self._profiles.values()
|
|
83
|
+
if profile.supports_chat
|
|
84
|
+
]
|
|
@@ -1,21 +1,32 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
from collections.abc import AsyncGenerator
|
|
4
|
+
from contextlib import asynccontextmanager
|
|
5
|
+
|
|
6
|
+
import httpx
|
|
3
7
|
from agents import Agent, ModelSettings, OpenAIResponsesModel, set_tracing_disabled
|
|
4
8
|
from openai import AsyncOpenAI
|
|
5
9
|
|
|
10
|
+
from {name_underscore}_domain.models.dtos.vllm_model_dtos import VllmModelProfileDto
|
|
6
11
|
from {name_underscore}_domain.services.agents.model_catalog import VllmModelCatalog
|
|
12
|
+
from {name_underscore}_domain.services.agents.vllm_inference_coordinator import (
|
|
13
|
+
VllmInferenceCoordinator,
|
|
14
|
+
VllmModelSwitchError,
|
|
15
|
+
VllmOfflineModeNotImplementedError,
|
|
16
|
+
)
|
|
7
17
|
from {name_underscore}_domain.utils.logging import get_logger
|
|
8
18
|
|
|
9
19
|
set_tracing_disabled(True)
|
|
10
20
|
|
|
11
|
-
|
|
12
21
|
_FINAL_OUTPUT_INSTRUCTIONS = """You are an AI assistant. Keep your responses concise and helpful. Keep your thoughts to a minimum try not to generate code when thinking."""
|
|
13
22
|
|
|
14
23
|
_logger = get_logger("VllmAgentFactory")
|
|
15
24
|
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
""
|
|
25
|
+
__all__ = [
|
|
26
|
+
"VllmAgentFactory",
|
|
27
|
+
"VllmModelSwitchError",
|
|
28
|
+
"VllmOfflineModeNotImplementedError",
|
|
29
|
+
]
|
|
19
30
|
|
|
20
31
|
|
|
21
32
|
class VllmAgentFactory:
|
|
@@ -27,25 +38,48 @@ class VllmAgentFactory:
|
|
|
27
38
|
api_key: str,
|
|
28
39
|
http_timeout: float,
|
|
29
40
|
thinking_effort: str | None,
|
|
41
|
+
coordinator: VllmInferenceCoordinator | None = None,
|
|
42
|
+
control_client: httpx.AsyncClient | None = None,
|
|
30
43
|
) -> None:
|
|
31
44
|
self._catalog = catalog
|
|
32
45
|
self._api_key = api_key
|
|
33
46
|
self._http_timeout = http_timeout
|
|
34
47
|
self._thinking_effort = thinking_effort
|
|
48
|
+
if coordinator is None:
|
|
49
|
+
if control_client is None:
|
|
50
|
+
raise TypeError("VllmAgentFactory requires a vLLM coordinator.")
|
|
51
|
+
coordinator = VllmInferenceCoordinator(
|
|
52
|
+
catalog=catalog,
|
|
53
|
+
api_key=api_key,
|
|
54
|
+
control_client=control_client,
|
|
55
|
+
)
|
|
56
|
+
self._coordinator = coordinator
|
|
35
57
|
self._clients: dict[str, AsyncOpenAI] = {}
|
|
36
58
|
self._agents: dict[tuple[str, str | None], Agent[None]] = {}
|
|
37
59
|
|
|
38
|
-
|
|
39
|
-
|
|
60
|
+
@property
|
|
61
|
+
def active_profile_id(self) -> str | None:
|
|
62
|
+
"""Return the profile most recently activated for an inference stream."""
|
|
63
|
+
return self._coordinator.active_profile_id
|
|
64
|
+
|
|
65
|
+
def resolve_profile(self, model_id: str | None) -> VllmModelProfileDto:
|
|
66
|
+
return self._coordinator.resolve_profile(model_id)
|
|
67
|
+
|
|
68
|
+
@asynccontextmanager
|
|
69
|
+
async def acquire(
|
|
70
|
+
self, model_id: str | None = None, thinking_effort: str | None = None
|
|
71
|
+
) -> AsyncGenerator[Agent[None], None]:
|
|
72
|
+
"""Exclusively activate a model for the lifetime of one inference stream."""
|
|
73
|
+
profile = self._require_chat_profile(model_id)
|
|
74
|
+
agent = self.create(profile.id, thinking_effort)
|
|
75
|
+
async with self._coordinator.reserve_profile(profile.id):
|
|
76
|
+
yield agent
|
|
40
77
|
|
|
41
78
|
def create(
|
|
42
79
|
self, model_id: str | None = None, thinking_effort: str | None = None
|
|
43
80
|
) -> Agent[None]:
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
raise VllmOfflineModeNotImplementedError(
|
|
47
|
-
f"Offline vLLM model profile '{profile.id}' is not implemented."
|
|
48
|
-
)
|
|
81
|
+
"""Create or reuse an agent without managing its deployment lifecycle."""
|
|
82
|
+
profile = self._require_chat_profile(model_id)
|
|
49
83
|
effective_effort = (
|
|
50
84
|
thinking_effort or self._thinking_effort
|
|
51
85
|
if profile.supports_thinking
|
|
@@ -81,3 +115,15 @@ class VllmAgentFactory:
|
|
|
81
115
|
)
|
|
82
116
|
self._agents[key] = agent
|
|
83
117
|
return agent
|
|
118
|
+
|
|
119
|
+
def _require_chat_profile(self, model_id: str | None) -> VllmModelProfileDto:
|
|
120
|
+
profile = self._coordinator.resolve_profile(model_id)
|
|
121
|
+
if not profile.supports_chat:
|
|
122
|
+
raise ValueError(
|
|
123
|
+
f"VLLM model profile '{profile.id}' does not support chat."
|
|
124
|
+
)
|
|
125
|
+
return profile
|
|
126
|
+
|
|
127
|
+
@staticmethod
|
|
128
|
+
def _build_url(base_url: str, replace_path: str = "") -> str:
|
|
129
|
+
return VllmInferenceCoordinator._build_url(base_url, replace_path)
|