rpr-cli 0.3.3__tar.gz → 0.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (328) hide show
  1. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/PKG-INFO +2 -2
  2. rpr_cli-0.3.4/src/rpr/__init__.py +1 -0
  3. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/init.py +6 -9
  4. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/settings.md +1 -63
  5. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/BACKEND_FLOW.md +39 -18
  6. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/in_memory_history_store.py +2 -2
  7. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/postgres_history_store.py +2 -2
  8. rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/models/dtos/__init__.py +3 -0
  9. rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/models/dtos/vllm_model_dtos.py +113 -0
  10. rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/VLLM_SETUP.md +31 -0
  11. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/__init__.py +5 -2
  12. rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/model_catalog.py +84 -0
  13. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/vllm_agent_factory.py +57 -11
  14. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/vllm_agent_factory_protocol.py +4 -3
  15. rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/vllm_inference_coordinator.py +207 -0
  16. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/runtimes/agents_sdk.py +17 -16
  17. rpr_cli-0.3.3/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/vllm_message_mapping.py → rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/runtimes/agents_sdk_message_mapping.py +15 -3
  18. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_agents_sdk_runtime.py +88 -1
  19. rpr_cli-0.3.4/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_vllm_model_catalog.py +768 -0
  20. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_vllm_provider.py +30 -5
  21. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/domain_container.md +12 -7
  22. rpr_cli-0.3.4/src/rpr/scaffolds/special_files/domain_settings.md +74 -0
  23. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_add.py +50 -4
  24. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_domain.py +3 -0
  25. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_init.py +11 -0
  26. rpr_cli-0.3.3/src/rpr/__init__.py +0 -1
  27. rpr_cli-0.3.3/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/agents/model_catalog.py +0 -106
  28. rpr_cli-0.3.3/src/rpr/scaffolds/py_special_files/vllm_provider/domain/tests/test_vllm_model_catalog.py +0 -168
  29. rpr_cli-0.3.3/src/rpr/scaffolds/special_files/domain_settings.md +0 -136
  30. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/.github/agents/story.agent.md +0 -0
  31. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/.github/workflows/publish-pypi.yml +0 -0
  32. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/.gitignore +0 -0
  33. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/CLAUDE.md +0 -0
  34. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/GEMINI.md +0 -0
  35. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/LICENSE +0 -0
  36. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/README.md +0 -0
  37. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/changelogs/mui_v9.txt +0 -0
  38. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/code-map.md +0 -0
  39. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/nx_uv_setup.md +0 -0
  40. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/patch_output.py +0 -0
  41. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/plan-languageAwareTemplateAdd.prompt.md +0 -0
  42. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/pyproject.toml +0 -0
  43. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/Cargo.lock +0 -0
  44. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/Cargo.toml +0 -0
  45. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/pyproject.toml +0 -0
  46. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/rpr_parser.pyi +0 -0
  47. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/config/container.rs +0 -0
  48. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/config/mod.rs +0 -0
  49. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/config/settings.rs +0 -0
  50. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/graph.rs +0 -0
  51. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/lib.rs +0 -0
  52. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/dtos/mod.rs +0 -0
  53. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/entities/mod.rs +0 -0
  54. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/entities/users.rs +0 -0
  55. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/models/mod.rs +0 -0
  56. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/parse.rs +0 -0
  57. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/repos/base.rs +0 -0
  58. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/repos/mod.rs +0 -0
  59. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/repos/users.rs +0 -0
  60. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/resolve.rs +0 -0
  61. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/src/walker.rs +0 -0
  62. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/rpr-parser/uv.lock +0 -0
  63. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/special_files.md +0 -0
  64. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/__init__.py +0 -0
  65. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/approval.py +0 -0
  66. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/bootstrap.py +0 -0
  67. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/client.py +0 -0
  68. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/control.py +0 -0
  69. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/mock.py +0 -0
  70. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/runtime.py +0 -0
  71. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/session.py +0 -0
  72. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/__init__.py +0 -0
  73. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/base.py +0 -0
  74. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/mutating.py +0 -0
  75. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/readonly.py +0 -0
  76. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/agent/tools/registry.py +0 -0
  77. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/__init__.py +0 -0
  78. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/catalog.py +0 -0
  79. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/chat_service.py +0 -0
  80. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/checks.py +0 -0
  81. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/cli_adapter.py +0 -0
  82. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/completer.py +0 -0
  83. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/conversation_service.py +0 -0
  84. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/prompt_service.py +0 -0
  85. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/selector.py +0 -0
  86. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/application/shell.py +0 -0
  87. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/__init__.py +0 -0
  88. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/base.py +0 -0
  89. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/instructions.py +0 -0
  90. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/packages.py +0 -0
  91. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/checks/workspace.py +0 -0
  92. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/cli.py +0 -0
  93. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/__init__.py +0 -0
  94. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/add.py +0 -0
  95. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/chat.py +0 -0
  96. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/check.py +0 -0
  97. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/__init__.py +0 -0
  98. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/api.py +0 -0
  99. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/domain.py +0 -0
  100. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/engine.py +0 -0
  101. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/storybook.py +0 -0
  102. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/generate/ui.py +0 -0
  103. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/map.py +0 -0
  104. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/refresh.py +0 -0
  105. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/rename.py +0 -0
  106. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/settings.py +0 -0
  107. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/sync.py +0 -0
  108. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/commands/update.py +0 -0
  109. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/context.py +0 -0
  110. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/dependencies.py +0 -0
  111. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/dependency_manifest_data.py +0 -0
  112. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/__init__.py +0 -0
  113. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/base.py +0 -0
  114. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/claude.py +0 -0
  115. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/codex.py +0 -0
  116. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/copilot.py +0 -0
  117. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/cursor.py +0 -0
  118. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/gemini.py +0 -0
  119. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/generators/github.py +0 -0
  120. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/manifest_refresh.py +0 -0
  121. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/__init__.py +0 -0
  122. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/architecture.py +0 -0
  123. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/chains.py +0 -0
  124. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/classifier.py +0 -0
  125. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/coverage.py +0 -0
  126. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/dependencies.py +0 -0
  127. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/extractor.py +0 -0
  128. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/graph.py +0 -0
  129. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/output.py +0 -0
  130. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/responsibility.py +0 -0
  131. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/topology.py +0 -0
  132. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/map/walker.py +0 -0
  133. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/package_updates.py +0 -0
  134. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/project_rename.py +0 -0
  135. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/base_entity.md +0 -0
  136. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/base_repo.md +0 -0
  137. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/domain/container.md +0 -0
  138. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/all.instructions.md +0 -0
  139. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/api.instructions.md +0 -0
  140. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/domain.instructions.md +0 -0
  141. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/frontend.instructions.md +0 -0
  142. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/rust-engine.instructions.md +0 -0
  143. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/setup-guide.instructions.md +0 -0
  144. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/tooling-setup.instructions.md +0 -0
  145. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/instructions/tooling.instructions.md +0 -0
  146. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/README.md +0 -0
  147. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/content.ts +0 -0
  148. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/httpChatStreamClient.spec.ts +0 -0
  149. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/httpChatStreamClient.ts +0 -0
  150. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/index.ts +0 -0
  151. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/client/types.ts +0 -0
  152. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/ChatComposer.spec.tsx +0 -0
  153. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/ChatComposer.tsx +0 -0
  154. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/__screenshots__/ChatComposer.spec.tsx/ChatComposer-renders-custom-action-buttons-beside-the-attachment-control-1.png +0 -0
  155. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/fileDrop.spec.ts +0 -0
  156. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/fileDrop.ts +0 -0
  157. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/index.ts +0 -0
  158. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/ChatComposerMentionTokens.tsx +0 -0
  159. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/ChatSuggestionsPopover.tsx +0 -0
  160. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/LocalFileSuggestionAction.tsx +0 -0
  161. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/index.ts +0 -0
  162. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/localFileSuggestions.spec.ts +0 -0
  163. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/localFileSuggestions.ts +0 -0
  164. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/triggerDetection.spec.ts +0 -0
  165. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/triggerDetection.ts +0 -0
  166. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/types.ts +0 -0
  167. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/suggestions/useChatSuggestions.ts +0 -0
  168. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/types.ts +0 -0
  169. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/useChatComposer.ts +0 -0
  170. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/useChatStream.spec.tsx +0 -0
  171. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/composer/useChatStream.ts +0 -0
  172. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperience.spec.tsx +0 -0
  173. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperience.styles.tsx +0 -0
  174. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperience.tsx +0 -0
  175. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceComposer.styles.tsx +0 -0
  176. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceComposer.tsx +0 -0
  177. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceHeader.styles.tsx +0 -0
  178. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ChatExperienceHeader.tsx +0 -0
  179. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ThreadHistory.styles.tsx +0 -0
  180. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/ThreadHistory.tsx +0 -0
  181. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/chatStorage.spec.ts +0 -0
  182. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/chatStorage.ts +0 -0
  183. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/index.ts +0 -0
  184. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/types.ts +0 -0
  185. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/useChatExperience.spec.tsx +0 -0
  186. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/experience/useChatExperience.ts +0 -0
  187. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/index.ts +0 -0
  188. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/testing/createMockChatStreamClient.ts +0 -0
  189. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/testing/createMockSuggestionProvider.ts +0 -0
  190. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/testing/index.ts +0 -0
  191. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessage.tsx +0 -0
  192. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessageActions.tsx +0 -0
  193. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessageContent.spec.tsx +0 -0
  194. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatMessageContent.tsx +0 -0
  195. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatThread.spec.tsx +0 -0
  196. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/ChatThread.tsx +0 -0
  197. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/README.md +0 -0
  198. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/contentFormat.ts +0 -0
  199. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/index.ts +0 -0
  200. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/streaming.ts +0 -0
  201. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/thread/types.ts +0 -0
  202. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponent.spec.tsx +0 -0
  203. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponent.tsx +0 -0
  204. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponentList.spec.tsx +0 -0
  205. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/VirtualComponentList.tsx +0 -0
  206. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/chat/virtual-list/index.ts +0 -0
  207. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/fetch.service.md +0 -0
  208. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/sticky-navigation.md +0 -0
  209. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/sticky-navigation.spec.md +0 -0
  210. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/js_special_files/vitest-browser-config.md +0 -0
  211. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/api/routes/chat_routes.py +0 -0
  212. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/api/tests/test_chat.py +0 -0
  213. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/models/dtos/chat_dtos.py +0 -0
  214. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/models/dtos/conversation_history_dtos.py +0 -0
  215. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/models/entities/conversation_history.py +0 -0
  216. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/repos/conversation_history.py +0 -0
  217. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/agent_session_codec.py +0 -0
  218. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/chat_service.py +0 -0
  219. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/conversation_history_service.py +0 -0
  220. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/conversation_operations.py +0 -0
  221. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/history_codec.py +0 -0
  222. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/protocols/conversation_history_store.py +0 -0
  223. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/protocols/conversation_runtime.py +0 -0
  224. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/services/protocols/llm_provider.py +0 -0
  225. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_chat_dtos.py +0 -0
  226. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_chat_service.py +0 -0
  227. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_history_codec.py +0 -0
  228. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_history_projection.py +0 -0
  229. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_persistence_config.py +0 -0
  230. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/chat/domain/tests/test_postgres_history_store.py +0 -0
  231. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/py_special_files/vllm_provider/domain/services/runtimes/__init__.py +0 -0
  232. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/camel_case_model.md +0 -0
  233. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/dto_util.md +0 -0
  234. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/encrypted_column.md +0 -0
  235. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/logging.md +0 -0
  236. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/mapper_util.md +0 -0
  237. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/partial_update.md +0 -0
  238. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/scaffolds/special_files/test_encrypted_column.md +0 -0
  239. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/templates/__init__.py +0 -0
  240. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/templates/registry.py +0 -0
  241. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/__init__.py +0 -0
  242. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/console.py +0 -0
  243. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/markdown.py +0 -0
  244. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/prompt_session.py +0 -0
  245. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/renderers.py +0 -0
  246. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/ui/theme.py +0 -0
  247. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/src/rpr/workspace.py +0 -0
  248. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/README.md +0 -0
  249. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/chat-attachment-upload-gap.md +0 -0
  250. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/00-bootstrap-cli.md +0 -0
  251. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/01-rpr-init.md +0 -0
  252. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/02-rpr-generate-domain.md +0 -0
  253. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/03-rpr-generate-api.md +0 -0
  254. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/04-rpr-generate-ui.md +0 -0
  255. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/05-rpr-generate-engine.md +0 -0
  256. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/06-rpr-generate-storybook.md +0 -0
  257. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/07-rpr-sync-rules.md +0 -0
  258. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/08-rpr-add-template.md +0 -0
  259. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/09-rpr-check.md +0 -0
  260. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/10-rpr-domain-config.md +0 -0
  261. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/18-rpr-terminal-ui-foundation.md +0 -0
  262. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/19-rpr-markdown-rendering.md +0 -0
  263. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/20-rpr-chat-repl.md +0 -0
  264. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/21-rpr-agent-tool-loop.md +0 -0
  265. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/22-rpr-chat-tui.md +0 -0
  266. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/23-rpr-chat-unified-command-shell.md +0 -0
  267. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/24-rpr-async-chat-runtime-and-services.md +0 -0
  268. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/25-rpr-prompt-service-and-conversation-lifecycle.md +0 -0
  269. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/26-rpr-shared-command-catalog.md +0 -0
  270. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/27-rpr-slash-command-selection-panel.md +0 -0
  271. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/28-rpr-chat-cancellable-and-multiline-input.md +0 -0
  272. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/chat/29-rpr-file-context-selection.md +0 -0
  273. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/11-rpr-map-command.md +0 -0
  274. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/12-rpr-map-file-walker.md +0 -0
  275. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/13-rpr-map-rust-engine.md +0 -0
  276. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/13a-rpr-map-gitignore.md +0 -0
  277. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/14-rpr-map-graph-builder.md +0 -0
  278. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/15-rpr-map-analysis-bridge.md +0 -0
  279. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/16-rpr-map-layer-classifier.md +0 -0
  280. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17-rpr-map-output.md +0 -0
  281. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17a-rpr-map-directory-topology.md +0 -0
  282. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17b-rpr-map-architecture-assessment.md +0 -0
  283. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17c-rpr-map-representative-chains.md +0 -0
  284. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/completed/map/17d-rpr-map-multi-resolution-output.md +0 -0
  285. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/map/30-rpr-map-directory-view-and-cycle-detection.md +0 -0
  286. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/map/31-rpr-map-coupling-hotspots-and-verdict-reasoning.md +0 -0
  287. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/map/32-rpr-map-dependencies-responsibilities-and-test-coverage.md +0 -0
  288. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/package-maintenance/default-vite-8-oxc-toolchain.md +0 -0
  289. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/package-maintenance/rpr-update-packages.md +0 -0
  290. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/stories/package-maintenance/typescript-7-frontend-scaffold-compatibility.md +0 -0
  291. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/__init__.py +0 -0
  292. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_architecture.py +0 -0
  293. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_architecture_unwrapped.py +0 -0
  294. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_catalog.py +0 -0
  295. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_chat.py +0 -0
  296. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_chat_service.py +0 -0
  297. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_check.py +0 -0
  298. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_completer.py +0 -0
  299. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_context.py +0 -0
  300. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_api.py +0 -0
  301. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_engine.py +0 -0
  302. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_storybook.py +0 -0
  303. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_generate_ui.py +0 -0
  304. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map.py +0 -0
  305. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_chains.py +0 -0
  306. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_classifier.py +0 -0
  307. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_coverage.py +0 -0
  308. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_dependencies.py +0 -0
  309. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_extractor.py +0 -0
  310. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_graph.py +0 -0
  311. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_output.py +0 -0
  312. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_responsibility.py +0 -0
  313. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_topology.py +0 -0
  314. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_map_walker.py +0 -0
  315. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_package_updates.py +0 -0
  316. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_packaging.py +0 -0
  317. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_prompt_session.py +0 -0
  318. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_refresh.py +0 -0
  319. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_rename.py +0 -0
  320. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_rpr_parser_stub.py +0 -0
  321. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_selector.py +0 -0
  322. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_settings.py +0 -0
  323. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_sync.py +0 -0
  324. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_ui.py +0 -0
  325. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/tests/test_update_packages.py +0 -0
  326. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/untitled:plan-chatAttachmentUploadGap.prompt.md +0 -0
  327. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/update_output.py +0 -0
  328. {rpr_cli-0.3.3 → rpr_cli-0.3.4}/uv.lock +0 -0
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.4
1
+ Metadata-Version: 2.5
2
2
  Name: rpr-cli
3
- Version: 0.3.3
3
+ Version: 0.3.4
4
4
  Summary: RPR CLI -- React Python Rust monorepo scaffolding orchestrator
5
5
  License-Expression: MIT
6
6
  License-File: LICENSE
@@ -0,0 +1 @@
1
+ __version__ = "0.3.4"
@@ -727,8 +727,8 @@ uv.lock
727
727
  ctx.write_file(ctx.project_dir / ".gitignore", content)
728
728
 
729
729
 
730
- def _write_env_example(ctx: RunContext) -> None:
731
- content = """# Database
730
+ def _write_env_example(ctx: RunContext, name_underscore: str) -> None:
731
+ content = f"""# Database
732
732
  DATABASE_URL=
733
733
 
734
734
  # Auth
@@ -742,13 +742,10 @@ FERNET_KEY=
742
742
  API_BASE_URL=http://localhost:8000
743
743
 
744
744
  # vLLM
745
- VLLM_BASE_URL=http://localhost:8000/v1
746
745
  VLLM_API_KEY=EMPTY
747
- VLLM_MODEL_ID=google/gemma-4-12B-it-qat-q4_0-gguf
748
- # Optional multi-server catalog. Each reasoning-capable profile must match its
749
- # server's reasoning parser. Omitted providerProperties.mode defaults to online.
750
- # VLLM_MODEL_CATALOG=[{"id":"gemma","label":"Gemma","modelId":"google/gemma","baseUrl":"http://localhost:8000/v1","reasoningParser":"gemma4","supportsThinking":true}]
751
- # VLLM_DEFAULT_MODEL_PROFILE_ID=gemma
746
+ # Model profiles, the default profile, restoration policy, and awake-model
747
+ # capacity are maintained in
748
+ # packages/python/{name_underscore}_domain/src/{name_underscore}_domain/services/agents/model_catalog.py.
752
749
  VLLM_THINKING_EFFORT=high
753
750
  """
754
751
  ctx.write_file(ctx.project_dir / ".env.example", content)
@@ -893,7 +890,7 @@ def init(
893
890
  _write_dependency_cruiser_config(ctx)
894
891
  _write_pre_commit_config(ctx)
895
892
  _write_gitignore(ctx)
896
- _write_env_example(ctx)
893
+ _write_env_example(ctx, variables["name_underscore"])
897
894
 
898
895
  # ------------------------------------------------------------------
899
896
  # Step 5 – AI instruction files
@@ -9,54 +9,10 @@ The first Python block is the generated file used by `rpr generate domain` and `
9
9
  ```python
10
10
  from __future__ import annotations
11
11
 
12
- from pydantic import AliasChoices, BaseModel, Field, JsonValue, model_validator
12
+ from pydantic import Field
13
13
  from pydantic_settings import BaseSettings, SettingsConfigDict
14
14
 
15
15
 
16
- class VllmModelProfileSettings(BaseModel):
17
- """One vLLM deployment available to the chat application."""
18
-
19
- id: str = Field(description="Stable public model-profile identifier.")
20
- label: str = Field(description="User-facing model label.")
21
- model_id: str = Field(
22
- validation_alias=AliasChoices("modelId", "model_id"),
23
- description="Model identifier sent to the vLLM Responses API.",
24
- )
25
- base_url: str = Field(
26
- validation_alias=AliasChoices("baseUrl", "base_url"),
27
- description="OpenAI-compatible vLLM endpoint for this profile.",
28
- )
29
- reasoning_parser: str | None = Field(
30
- default=None,
31
- validation_alias=AliasChoices("reasoningParser", "reasoning_parser"),
32
- description="Documented vLLM parser required by this deployment.",
33
- )
34
- supports_thinking: bool = Field(
35
- default=False,
36
- validation_alias=AliasChoices("supportsThinking", "supports_thinking"),
37
- description="Whether this deployment emits native reasoning output.",
38
- )
39
- description: str | None = Field(
40
- default=None, description="Optional user-facing model description."
41
- )
42
-
43
- provider_properties: dict[str, JsonValue] = Field(
44
- default_factory=dict,
45
- validation_alias=AliasChoices("providerProperties", "provider_properties"),
46
- description="vLLM-specific deployment properties not exposed by the chat API.",
47
- )
48
-
49
- @model_validator(mode="after")
50
- def validate_execution_mode(self) -> VllmModelProfileSettings:
51
- """Validate the optional vLLM execution mode stored in provider properties."""
52
- mode = self.provider_properties.get("mode", "online")
53
- if not isinstance(mode, str) or mode not in {"online", "offline"}:
54
- raise ValueError(
55
- "VLLM providerProperties.mode must be either 'online' or 'offline'."
56
- )
57
- return self
58
-
59
-
60
16
  class Settings(BaseSettings):
61
17
  model_config = SettingsConfigDict(
62
18
  env_file=".env",
@@ -96,28 +52,10 @@ class Settings(BaseSettings):
96
52
  default=20,
97
53
  description="Maximum keepalive HTTP connections",
98
54
  )
99
- vllm_base_url: str = Field(
100
- default="http://localhost:8000/v1",
101
- description="OpenAI-compatible vLLM server base URL",
102
- )
103
55
  vllm_api_key: str = Field(
104
56
  default="EMPTY",
105
57
  description="API key sent to vLLM model servers",
106
58
  )
107
- vllm_model_id: str = Field(
108
- default="google/gemma-4-12B-it-qat-q4_0-gguf",
109
- description="Legacy default vLLM model when no catalog is configured",
110
- )
111
- vllm_model_catalog: list[VllmModelProfileSettings] = Field(
112
- default_factory=list,
113
- validation_alias="VLLM_MODEL_CATALOG",
114
- description="Configured public model profiles and their vLLM endpoints.",
115
- )
116
- vllm_default_model_profile_id: str | None = Field(
117
- default=None,
118
- validation_alias="VLLM_DEFAULT_MODEL_PROFILE_ID",
119
- description="Public ID of the default catalog profile.",
120
- )
121
59
  vllm_thinking_effort: str = Field(
122
60
  default="high",
123
61
  description="vLLM thinking effort for reasoning-capable profiles",
@@ -99,21 +99,25 @@ flowchart LR
99
99
 
100
100
  runtimePort("ConversationRuntimeProtocol\nprovider-neutral execution contract")
101
101
  runtime("runtimes/agents_sdk.py\nAgentsSdkConversationRuntime")
102
- factoryPort("VllmAgentFactoryProtocol\ncreate an Agents SDK Agent")
102
+ factoryPort("VllmAgentFactoryProtocol\nacquire an Agents SDK Agent")
103
103
  factory("agents/vllm_agent_factory.py\nVllmAgentFactory")
104
+ coordinator("agents/vllm_inference_coordinator.py\nVllmInferenceCoordinator")
104
105
  catalog("agents/model_catalog.py\nVllmModelCatalog")
105
106
  vllm("OpenAI-compatible vLLM\nResponses API")
106
107
  offline("Offline executor\nnot registered yet")
107
108
 
108
109
  runtimePort -. "implemented by" .-> runtime
109
- runtime -->|"create selected agent"| factoryPort
110
+ runtime -->|"reserve selected model for stream"| factoryPort
110
111
  factoryPort -. "implemented by" .-> factory
111
- factory -->|"resolve public profile"| catalog
112
+ factory -->|"resolve and create cached agent"| catalog
113
+ factory -->|"reserve profile"| coordinator
114
+ coordinator -->|"capacity and priority policy"| catalog
115
+ coordinator -->|"sleep-state control"| vllm
112
116
  factory -->|"online profile: Agent + AsyncOpenAI"| vllm
113
- factory -. "offline profile: raises\nVllmOfflineModeNotImplementedError" .-> offline
117
+ coordinator -. "offline profile: raises\nVllmOfflineModeNotImplementedError" .-> offline
114
118
 
115
119
  class runtimePort,factoryPort protocol;
116
- class runtime,factory,catalog,vllm concrete;
120
+ class runtime,factory,coordinator,catalog,vllm concrete;
117
121
  class offline future;
118
122
  ```
119
123
 
@@ -122,7 +126,9 @@ flowchart LR
122
126
  If it still uses the Agents SDK but needs a different model vendor, replace or
123
127
  extend `VllmAgentFactory` behind `VllmAgentFactoryProtocol` and update
124
128
  `Container.vllm_agent_factory`. The current offline branch stops in
125
- `VllmAgentFactory.create()` before any agent is constructed.
129
+ `VllmAgentFactory.acquire()` before inference starts. Keep the acquisition
130
+ context open for the complete response stream so a deployment cannot be slept
131
+ while it is still generating.
126
132
 
127
133
  ### 4. History and agent-session storage
128
134
 
@@ -159,9 +165,9 @@ flowchart LR
159
165
  runtime, also replace the session factory dependency used by
160
166
  `AgentsSdkConversationRuntime`.
161
167
 
162
- ### 5. Model catalog: configuration to picker
168
+ ### 5. Model catalog: Python profiles to picker
163
169
 
164
- **Purpose:** the catalog is the allow-list between local configuration, the
170
+ **Purpose:** the catalog is the allow-list between Python-owned configuration, the
165
171
  browser picker, and agent creation. It keeps base URLs, parser configuration, and
166
172
  provider-only properties on the server.
167
173
 
@@ -173,30 +179,38 @@ flowchart LR
173
179
  classDef api fill:#fef3c7,stroke:#d97706,color:#78350f,stroke-width:2px;
174
180
  classDef ui fill:#dbeafe,stroke:#2563eb,color:#172554,stroke-width:2px;
175
181
 
176
- env(".env\nVLLM_MODEL_CATALOG\nVLLM_DEFAULT_MODEL_PROFILE_ID")
177
- settings("config/settings.py\nSettings / VllmModelProfileSettings")
182
+ profiles("agents/model_catalog.py\nPython catalog DTO + policy")
183
+ dto("models/dtos/vllm_model_dtos.py\nVllmModelCatalogDto / VllmModelProfileDto")
178
184
  container("config/container.py\nContainer.vllm_model_catalog")
179
185
  catalog("agents/model_catalog.py\nVllmModelCatalog")
180
186
  route("chat_routes.py\nGET /api/chat/models")
181
187
  loader("services/chatModels.ts\nloadChatModels()")
182
188
  picker("MainPage.tsx / ChatExperience\nmodels + initialModelId")
183
189
 
184
- env --> settings
185
- settings --> container
190
+ profiles --> catalog
191
+ dto -->|"validates internal profiles"| catalog
186
192
  container --> catalog
187
193
  catalog -->|"public_profiles(): safe metadata"| route
188
194
  route --> loader
189
195
  loader --> picker
190
196
 
191
- class env,settings config;
192
- class container,catalog domain;
197
+ class profiles config;
198
+ class dto,container,catalog domain;
193
199
  class route api;
194
200
  class loader,picker ui;
195
201
  ```
196
202
 
197
- To add a selectable model, update local `.env` and the documented non-secret
198
- shape in `.env.example`; then verify the catalog profile resolves and
199
- `GET /api/chat/models` returns the safe metadata used by `MainPage.tsx`.
203
+ To add a selectable model, update `VLLM_MODEL_CATALOG` in
204
+ `agents/model_catalog.py`. Set exactly one profile's `is_default=True` and give
205
+ each profile a one-based `priority` (`1` is highest). Set `max_awake_models` to
206
+ the number of deployments the host can retain in memory; then verify the catalog
207
+ profile resolves and `GET /api/chat/models` returns the safe metadata used by
208
+ `MainPage.tsx`.
209
+
210
+ Every online profile must point to a vLLM server started with
211
+ `VLLM_SERVER_DEV_MODE=1` and `--enable-sleep-mode`. The coordinator uses
212
+ `/is_sleeping`, `/sleep`, and `/wake_up`; keep these development-only endpoints
213
+ local or otherwise private.
200
214
 
201
215
  ### 6. Protocol-to-concrete registration map
202
216
 
@@ -253,6 +267,7 @@ sequenceDiagram
253
267
  participant Store as PostgresConversationHistoryStore
254
268
  participant Runtime as AgentsSdkConversationRuntime
255
269
  participant Factory as VllmAgentFactory
270
+ participant Coordinator as VllmInferenceCoordinator
256
271
  participant Session as PostgresAgentSessionFactory
257
272
  participant VLLM as vLLM Responses API
258
273
 
@@ -269,9 +284,13 @@ sequenceDiagram
269
284
  Store-->>Service: turn_id
270
285
  Service->>Runtime: stream(ConversationExecution)
271
286
  activate Runtime
272
- Runtime->>Factory: create(model_id, thinking_effort)
287
+ Runtime->>Factory: acquire(model_id, thinking_effort)
273
288
  Factory->>Catalog: resolve(model_id)
274
289
  alt profile mode is online
290
+ Factory->>Coordinator: reserve_profile(profile_id)
291
+ Coordinator->>Catalog: profiles + max_awake_models
292
+ Coordinator->>VLLM: inspect, sleep lower-priority deployment, wake target
293
+ Coordinator-->>Factory: reserved profile
275
294
  Factory-->>Runtime: cached/new Agent and AsyncOpenAI client
276
295
  Runtime->>Session: create(execution)
277
296
  Session-->>Runtime: persisted Agents SDK session
@@ -283,6 +302,8 @@ sequenceDiagram
283
302
  Route-->>UI: SSE thinking event or data frame
284
303
  end
285
304
  Runtime-->>Service: ConversationThinkingCompleted + ConversationCompleted
305
+ Runtime->>Factory: release acquisition
306
+ Factory->>Coordinator: restore default when configured
286
307
  Service->>Store: finish(..., complete)
287
308
  Store-->>Service: persisted transcript and session ledger
288
309
  else profile mode is offline
@@ -1,7 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
3
  import asyncio
4
- from collections.abc import AsyncIterator
4
+ from collections.abc import AsyncGenerator
5
5
  from contextlib import asynccontextmanager
6
6
 
7
7
  from {name_underscore}_domain.services.protocols.llm_provider import ChatConversationTurn
@@ -16,7 +16,7 @@ class InMemoryConversationHistoryStore:
16
16
  self._locks: dict[str, asyncio.Lock] = {}
17
17
 
18
18
  @asynccontextmanager
19
- async def serialized(self, thread_id: str) -> AsyncIterator[None]:
19
+ async def serialized(self, thread_id: str) -> AsyncGenerator[None, None]:
20
20
  lock = self._locks.setdefault(thread_id, asyncio.Lock())
21
21
  async with lock:
22
22
  yield
@@ -1,7 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
3
  import asyncio
4
- from collections.abc import AsyncIterator
4
+ from collections.abc import AsyncGenerator
5
5
  from contextlib import asynccontextmanager
6
6
 
7
7
  from sqlalchemy import text
@@ -34,7 +34,7 @@ class PostgresConversationHistoryStore:
34
34
  configure_encrypted_columns(encryption_key)
35
35
 
36
36
  @asynccontextmanager
37
- async def serialized(self, thread_id: str) -> AsyncIterator[None]:
37
+ async def serialized(self, thread_id: str) -> AsyncGenerator[None, None]:
38
38
  connection = await self._engine.connect()
39
39
  acquired = False
40
40
  try:
@@ -0,0 +1,3 @@
1
+ from .vllm_model_dtos import VllmModelCatalogDto, VllmModelProfileDto
2
+
3
+ __all__ = ["VllmModelCatalogDto", "VllmModelProfileDto"]
@@ -0,0 +1,113 @@
1
+ from __future__ import annotations
2
+
3
+ from typing import Literal, Self, cast
4
+
5
+ from pydantic import Field, JsonValue, model_validator
6
+
7
+ from {name_underscore}_domain.models.dtos.chat_dtos import ChatModelProfileDto
8
+ from {name_underscore}_domain.utils.camel_case_model import CamelCaseModel
9
+
10
+
11
+ class VllmModelProfileDto(CamelCaseModel):
12
+ """Internal configuration for one vLLM model deployment."""
13
+
14
+ id: str = Field(description="Stable public model-profile identifier.")
15
+ label: str = Field(description="User-facing model label.")
16
+ model_id: str = Field(
17
+ description="Model identifier sent to the vLLM Responses API."
18
+ )
19
+ base_url: str = Field(
20
+ description="OpenAI-compatible vLLM endpoint for this profile."
21
+ )
22
+ reasoning_parser: str | None = Field(
23
+ default=None,
24
+ description="Documented vLLM parser required by this deployment.",
25
+ )
26
+ supports_thinking: bool = Field(
27
+ default=False,
28
+ description="Whether this deployment emits native reasoning output.",
29
+ )
30
+ supports_chat: bool = Field(
31
+ default=True,
32
+ description="Whether this deployment can be selected by the chat runtime.",
33
+ )
34
+ supports_ocr: bool = Field(
35
+ default=False,
36
+ description="Whether this deployment is available to an OCR service.",
37
+ )
38
+ is_default: bool = Field(
39
+ default=False,
40
+ description="Whether this is the default chat model profile.",
41
+ )
42
+ priority: int = Field(
43
+ default=1,
44
+ ge=1,
45
+ description="Awake-model retention priority; 1 is the highest priority.",
46
+ )
47
+ description: str | None = Field(
48
+ default=None,
49
+ description="Optional user-facing model description.",
50
+ )
51
+ provider_properties: dict[str, JsonValue] = Field(
52
+ default_factory=dict,
53
+ description="vLLM-specific deployment properties kept inside the domain.",
54
+ )
55
+
56
+ @model_validator(mode="after")
57
+ def validate_execution_mode(self) -> Self:
58
+ """Validate the optional execution mode stored in provider properties."""
59
+ mode = self.provider_properties.get("mode", "online")
60
+ if not isinstance(mode, str) or mode not in {"online", "offline"}:
61
+ raise ValueError(
62
+ "VLLM providerProperties.mode must be either 'online' or 'offline'."
63
+ )
64
+ return self
65
+
66
+ @property
67
+ def execution_mode(self) -> Literal["online", "offline"]:
68
+ """Return the configured execution mode, defaulting to online."""
69
+ return cast(
70
+ Literal["online", "offline"],
71
+ self.provider_properties.get("mode", "online"),
72
+ )
73
+
74
+ def to_chat_model_profile_dto(self, *, is_default: bool) -> ChatModelProfileDto:
75
+ """Map internal deployment data to safe model-picker metadata."""
76
+ return ChatModelProfileDto(
77
+ id=self.id,
78
+ label=self.label,
79
+ description=self.description,
80
+ is_default=is_default,
81
+ supports_thinking=self.supports_thinking,
82
+ )
83
+
84
+
85
+ class VllmModelCatalogDto(CamelCaseModel):
86
+ """Internal policy and profile data for all configured vLLM deployments."""
87
+
88
+ profiles: list[VllmModelProfileDto] = Field(
89
+ min_length=1,
90
+ description="Configured vLLM model profiles.",
91
+ )
92
+ max_awake_models: int = Field(
93
+ default=1,
94
+ ge=1,
95
+ description="Maximum number of distinct online vLLM deployments kept awake.",
96
+ )
97
+ restore_default_after_request: bool = Field(
98
+ default=True,
99
+ description="Whether to reactivate the default profile after another request.",
100
+ )
101
+
102
+ @model_validator(mode="after")
103
+ def validate_default_profile(self) -> Self:
104
+ """Require one chat model profile to be designated as the default."""
105
+ if sum(profile.is_default for profile in self.profiles) != 1:
106
+ raise ValueError(
107
+ "The VLLM catalog must contain exactly one default profile."
108
+ )
109
+ if not next(
110
+ profile for profile in self.profiles if profile.is_default
111
+ ).supports_chat:
112
+ raise ValueError("The default VLLM profile must support chat.")
113
+ return self
@@ -0,0 +1,31 @@
1
+ # vLLM model catalog and sleep coordination
2
+
3
+ Model profiles, the default profile, awake-model capacity, and restoration policy
4
+ are defined in `model_catalog.py`. Do not put model IDs or base URLs in `.env`.
5
+
6
+ Start every online catalog deployment with vLLM development mode and sleep mode
7
+ enabled. For the generated Gemma profile:
8
+
9
+ ```sh
10
+ VLLM_WSL2_ENABLE_PIN_MEMORY=1 \
11
+ VLLM_SERVER_DEV_MODE=1 \
12
+ vllm serve google/gemma-4-12B-it-qat-w4a16-ct \
13
+ --max-model-len 8192 \
14
+ --gpu-memory-utilization 0.70 \
15
+ --max-num-seqs 1 \
16
+ --enable-auto-tool-choice \
17
+ --tool-call-parser gemma4 \
18
+ --reasoning-parser gemma4 \
19
+ --enable-sleep-mode \
20
+ --port 8000
21
+ ```
22
+
23
+ The coordinator calls `/is_sleeping`, `/sleep`, and `/wake_up` on each online
24
+ deployment to enforce `max_awake_models`. Keep these development-only endpoints
25
+ local or otherwise private. Lower numeric `priority` values are retained first;
26
+ `1` is the highest priority. When `restore_default_after_request` is enabled, a
27
+ temporary profile is released and the default profile is restored after the
28
+ complete response stream.
29
+
30
+ Each reasoning-capable profile must match its server's `--reasoning-parser`
31
+ setting. The generated runtime sends `reasoning.effort` per request.
@@ -1,15 +1,18 @@
1
- from .model_catalog import UnknownModelProfileError, VllmModelCatalog, VllmModelProfile
1
+ from .model_catalog import UnknownModelProfileError, VllmModelCatalog
2
2
  from .vllm_agent_factory import (
3
3
  VllmAgentFactory,
4
+ VllmModelSwitchError,
4
5
  VllmOfflineModeNotImplementedError,
5
6
  )
6
7
  from .vllm_agent_factory_protocol import VllmAgentFactoryProtocol
8
+ from .vllm_inference_coordinator import VllmInferenceCoordinator
7
9
 
8
10
  __all__ = [
9
11
  "UnknownModelProfileError",
10
12
  "VllmAgentFactory",
11
13
  "VllmAgentFactoryProtocol",
12
14
  "VllmModelCatalog",
13
- "VllmModelProfile",
15
+ "VllmInferenceCoordinator",
16
+ "VllmModelSwitchError",
14
17
  "VllmOfflineModeNotImplementedError",
15
18
  ]
@@ -0,0 +1,84 @@
1
+ from __future__ import annotations
2
+
3
+ from {name_underscore}_domain.models.dtos.chat_dtos import ChatModelProfileDto
4
+ from {name_underscore}_domain.models.dtos.vllm_model_dtos import (
5
+ VllmModelCatalogDto,
6
+ VllmModelProfileDto,
7
+ )
8
+
9
+ VLLM_MODEL_CATALOG = VllmModelCatalogDto(
10
+ max_awake_models=1,
11
+ restore_default_after_request=True,
12
+ profiles=[
13
+ VllmModelProfileDto(
14
+ id="gemma",
15
+ label="Gemma 4 12B",
16
+ model_id="google/gemma-4-12B-it-qat-w4a16-ct",
17
+ base_url="http://localhost:8000/v1",
18
+ reasoning_parser="gemma4",
19
+ supports_thinking=True,
20
+ is_default=True,
21
+ priority=1,
22
+ ),
23
+ ],
24
+ )
25
+
26
+
27
+ class UnknownModelProfileError(ValueError):
28
+ """Raised when a request refers to a model profile unavailable to this app."""
29
+
30
+
31
+ class VllmModelCatalog:
32
+ """Returns and resolves the Python-owned vLLM deployment catalog."""
33
+
34
+ def __init__(
35
+ self,
36
+ catalog: VllmModelCatalogDto | None = None,
37
+ ) -> None:
38
+ self._catalog = (catalog or VLLM_MODEL_CATALOG).model_copy(deep=True)
39
+ configured_profiles = tuple(self._catalog.profiles)
40
+
41
+ profile_ids = [profile.id for profile in configured_profiles]
42
+ if len(set(profile_ids)) != len(profile_ids):
43
+ raise ValueError("VLLM model profile IDs must be unique.")
44
+
45
+ self._profiles = {profile.id: profile for profile in configured_profiles}
46
+ self._default_profile_id = next(
47
+ profile.id for profile in configured_profiles if profile.is_default
48
+ )
49
+
50
+ @property
51
+ def default_profile_id(self) -> str:
52
+ return self._default_profile_id
53
+
54
+ @property
55
+ def catalog(self) -> VllmModelCatalogDto:
56
+ """Return a copied catalog configuration and model profile list."""
57
+ return self._catalog.model_copy(deep=True)
58
+
59
+ @property
60
+ def max_awake_models(self) -> int:
61
+ return self._catalog.max_awake_models
62
+
63
+ @property
64
+ def restore_default_after_request(self) -> bool:
65
+ return self._catalog.restore_default_after_request
66
+
67
+ @property
68
+ def profiles(self) -> tuple[VllmModelProfileDto, ...]:
69
+ """Return configured profiles for internal deployment coordination."""
70
+ return tuple(self._profiles.values())
71
+
72
+ def resolve(self, model_id: str | None) -> VllmModelProfileDto:
73
+ profile_id = model_id or self._default_profile_id
74
+ try:
75
+ return self._profiles[profile_id]
76
+ except KeyError as error:
77
+ raise UnknownModelProfileError(profile_id) from error
78
+
79
+ def public_profiles(self) -> list[ChatModelProfileDto]:
80
+ return [
81
+ profile.to_chat_model_profile_dto(is_default=profile.is_default)
82
+ for profile in self._profiles.values()
83
+ if profile.supports_chat
84
+ ]
@@ -1,21 +1,32 @@
1
1
  from __future__ import annotations
2
2
 
3
+ from collections.abc import AsyncGenerator
4
+ from contextlib import asynccontextmanager
5
+
6
+ import httpx
3
7
  from agents import Agent, ModelSettings, OpenAIResponsesModel, set_tracing_disabled
4
8
  from openai import AsyncOpenAI
5
9
 
10
+ from {name_underscore}_domain.models.dtos.vllm_model_dtos import VllmModelProfileDto
6
11
  from {name_underscore}_domain.services.agents.model_catalog import VllmModelCatalog
12
+ from {name_underscore}_domain.services.agents.vllm_inference_coordinator import (
13
+ VllmInferenceCoordinator,
14
+ VllmModelSwitchError,
15
+ VllmOfflineModeNotImplementedError,
16
+ )
7
17
  from {name_underscore}_domain.utils.logging import get_logger
8
18
 
9
19
  set_tracing_disabled(True)
10
20
 
11
-
12
21
  _FINAL_OUTPUT_INSTRUCTIONS = """You are an AI assistant. Keep your responses concise and helpful. Keep your thoughts to a minimum try not to generate code when thinking."""
13
22
 
14
23
  _logger = get_logger("VllmAgentFactory")
15
24
 
16
-
17
- class VllmOfflineModeNotImplementedError(NotImplementedError):
18
- """Raised when an offline vLLM profile has no registered executor yet."""
25
+ __all__ = [
26
+ "VllmAgentFactory",
27
+ "VllmModelSwitchError",
28
+ "VllmOfflineModeNotImplementedError",
29
+ ]
19
30
 
20
31
 
21
32
  class VllmAgentFactory:
@@ -27,25 +38,48 @@ class VllmAgentFactory:
27
38
  api_key: str,
28
39
  http_timeout: float,
29
40
  thinking_effort: str | None,
41
+ coordinator: VllmInferenceCoordinator | None = None,
42
+ control_client: httpx.AsyncClient | None = None,
30
43
  ) -> None:
31
44
  self._catalog = catalog
32
45
  self._api_key = api_key
33
46
  self._http_timeout = http_timeout
34
47
  self._thinking_effort = thinking_effort
48
+ if coordinator is None:
49
+ if control_client is None:
50
+ raise TypeError("VllmAgentFactory requires a vLLM coordinator.")
51
+ coordinator = VllmInferenceCoordinator(
52
+ catalog=catalog,
53
+ api_key=api_key,
54
+ control_client=control_client,
55
+ )
56
+ self._coordinator = coordinator
35
57
  self._clients: dict[str, AsyncOpenAI] = {}
36
58
  self._agents: dict[tuple[str, str | None], Agent[None]] = {}
37
59
 
38
- def resolve_profile(self, model_id: str | None):
39
- return self._catalog.resolve(model_id)
60
+ @property
61
+ def active_profile_id(self) -> str | None:
62
+ """Return the profile most recently activated for an inference stream."""
63
+ return self._coordinator.active_profile_id
64
+
65
+ def resolve_profile(self, model_id: str | None) -> VllmModelProfileDto:
66
+ return self._coordinator.resolve_profile(model_id)
67
+
68
+ @asynccontextmanager
69
+ async def acquire(
70
+ self, model_id: str | None = None, thinking_effort: str | None = None
71
+ ) -> AsyncGenerator[Agent[None], None]:
72
+ """Exclusively activate a model for the lifetime of one inference stream."""
73
+ profile = self._require_chat_profile(model_id)
74
+ agent = self.create(profile.id, thinking_effort)
75
+ async with self._coordinator.reserve_profile(profile.id):
76
+ yield agent
40
77
 
41
78
  def create(
42
79
  self, model_id: str | None = None, thinking_effort: str | None = None
43
80
  ) -> Agent[None]:
44
- profile = self._catalog.resolve(model_id)
45
- if profile.execution_mode == "offline":
46
- raise VllmOfflineModeNotImplementedError(
47
- f"Offline vLLM model profile '{profile.id}' is not implemented."
48
- )
81
+ """Create or reuse an agent without managing its deployment lifecycle."""
82
+ profile = self._require_chat_profile(model_id)
49
83
  effective_effort = (
50
84
  thinking_effort or self._thinking_effort
51
85
  if profile.supports_thinking
@@ -81,3 +115,15 @@ class VllmAgentFactory:
81
115
  )
82
116
  self._agents[key] = agent
83
117
  return agent
118
+
119
+ def _require_chat_profile(self, model_id: str | None) -> VllmModelProfileDto:
120
+ profile = self._coordinator.resolve_profile(model_id)
121
+ if not profile.supports_chat:
122
+ raise ValueError(
123
+ f"VLLM model profile '{profile.id}' does not support chat."
124
+ )
125
+ return profile
126
+
127
+ @staticmethod
128
+ def _build_url(base_url: str, replace_path: str = "") -> str:
129
+ return VllmInferenceCoordinator._build_url(base_url, replace_path)