claude-smart 0.2.43 → 0.2.45

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/.claude-plugin/marketplace.json +3 -3
  2. package/README.md +1 -1
  3. package/bin/claude-smart.js +2 -2
  4. package/package.json +2 -2
  5. package/plugin/.claude-plugin/plugin.json +2 -2
  6. package/plugin/.codex-plugin/plugin.json +1 -1
  7. package/plugin/README.md +21 -1
  8. package/plugin/pyproject.toml +3 -3
  9. package/plugin/scripts/_lib.sh +72 -28
  10. package/plugin/scripts/backend-service.sh +55 -7
  11. package/plugin/scripts/cli.sh +2 -1
  12. package/plugin/scripts/dashboard-service.sh +7 -5
  13. package/plugin/scripts/ensure-plugin-root.sh +1 -0
  14. package/plugin/scripts/hook_entry.sh +7 -5
  15. package/plugin/scripts/smart-install.sh +2 -2
  16. package/plugin/src/README.md +57 -0
  17. package/plugin/uv.lock +126 -5
  18. package/plugin/vendor/reflexio/.env.example +9 -0
  19. package/plugin/vendor/reflexio/pyproject.toml +5 -2
  20. package/plugin/vendor/reflexio/reflexio/cli/bootstrap_config.py +0 -1
  21. package/plugin/vendor/reflexio/reflexio/cli/commands/setup_cmd.py +7 -4
  22. package/plugin/vendor/reflexio/reflexio/cli/env_loader.py +4 -4
  23. package/plugin/vendor/reflexio/reflexio/lib/_storage_labels.py +2 -2
  24. package/plugin/vendor/reflexio/reflexio/models/api_schema/domain/entities.py +13 -4
  25. package/plugin/vendor/reflexio/reflexio/models/api_schema/retriever_schema.py +3 -1
  26. package/plugin/vendor/reflexio/reflexio/models/api_schema/validators.py +59 -6
  27. package/plugin/vendor/reflexio/reflexio/models/config_schema.py +1 -1
  28. package/plugin/vendor/reflexio/reflexio/server/README.md +7 -1
  29. package/plugin/vendor/reflexio/reflexio/server/api.py +188 -34
  30. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/README.md +34 -0
  31. package/plugin/vendor/reflexio/reflexio/server/api_endpoints/publisher_api.py +33 -11
  32. package/plugin/vendor/reflexio/reflexio/server/llm/embedding_service.py +278 -29
  33. package/plugin/vendor/reflexio/reflexio/server/llm/litellm_client.py +457 -181
  34. package/plugin/vendor/reflexio/reflexio/server/llm/llm_utils.py +28 -0
  35. package/plugin/vendor/reflexio/reflexio/server/llm/model_defaults.py +22 -12
  36. package/plugin/vendor/reflexio/reflexio/server/llm/providers/embedding_service_provider.py +137 -9
  37. package/plugin/vendor/reflexio/reflexio/server/llm/providers/local_embedding_provider.py +1 -1
  38. package/plugin/vendor/reflexio/reflexio/server/llm/providers/nomic_embedding_provider.py +36 -3
  39. package/plugin/vendor/reflexio/reflexio/server/llm/rerank/cross_encoder_reranker.py +13 -3
  40. package/plugin/vendor/reflexio/reflexio/server/llm/tools.py +17 -0
  41. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.3.0.prompt.md +1 -1
  42. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.3.1.prompt.md +69 -0
  43. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_consolidation/v2.3.2.prompt.md +71 -0
  44. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.2.0.prompt.md +27 -23
  45. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.2.2.prompt.md +234 -0
  46. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context/v4.2.3.prompt.md +244 -0
  47. package/plugin/vendor/reflexio/reflexio/server/prompt/prompt_bank/playbook_extraction_context_expert/v3.3.0.prompt.md +19 -9
  48. package/plugin/vendor/reflexio/reflexio/server/services/README.md +58 -0
  49. package/plugin/vendor/reflexio/reflexio/server/services/agent_success_evaluation/group_evaluation_runner.py +1 -5
  50. package/plugin/vendor/reflexio/reflexio/server/services/base_generation_service.py +44 -2
  51. package/plugin/vendor/reflexio/reflexio/server/services/embedding_text.py +62 -0
  52. package/plugin/vendor/reflexio/reflexio/server/services/extraction/pending_tool_call_dispatch.py +8 -1
  53. package/plugin/vendor/reflexio/reflexio/server/services/extraction/resumable_agent.py +91 -24
  54. package/plugin/vendor/reflexio/reflexio/server/services/extraction/resume_worker.py +3 -1
  55. package/plugin/vendor/reflexio/reflexio/server/services/extractor_config_utils.py +6 -3
  56. package/plugin/vendor/reflexio/reflexio/server/services/extractor_interaction_utils.py +1 -1
  57. package/plugin/vendor/reflexio/reflexio/server/services/generation_service.py +18 -5
  58. package/plugin/vendor/reflexio/reflexio/server/services/operation_state_utils.py +2 -2
  59. package/plugin/vendor/reflexio/reflexio/server/services/playbook/playbook_consolidator.py +88 -3
  60. package/plugin/vendor/reflexio/reflexio/server/services/profile/profile_deduplicator.py +36 -5
  61. package/plugin/vendor/reflexio/reflexio/server/services/profile/profile_generation_service.py +6 -3
  62. package/plugin/vendor/reflexio/reflexio/server/services/reflection/reflection_service.py +6 -3
  63. package/plugin/vendor/reflexio/reflexio/server/services/retrieval/relevance_floor.py +32 -22
  64. package/plugin/vendor/reflexio/reflexio/server/services/service_utils.py +89 -4
  65. package/plugin/vendor/reflexio/reflexio/server/services/storage/sqlite_storage/_agent_run.py +46 -1
  66. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_agent_run.py +12 -0
  67. package/plugin/vendor/reflexio/reflexio/server/services/storage/storage_base/_operations.py +1 -1
  68. package/plugin/vendor/reflexio/reflexio/server/services/unified_search_service.py +8 -4
@@ -2,29 +2,113 @@
2
2
 
3
3
  from __future__ import annotations
4
4
 
5
+ import logging
5
6
  import math
7
+ import os
6
8
  import threading
9
+ import time
10
+ from collections.abc import AsyncIterator
11
+ from contextlib import asynccontextmanager
12
+ from dataclasses import dataclass, field
7
13
  from typing import Any
8
14
 
9
15
  from fastapi import FastAPI, HTTPException
10
16
  from pydantic import BaseModel, Field
11
17
 
18
+ from reflexio.server.llm.llm_utils import positive_int_env
12
19
  from reflexio.server.llm.providers.local_embedding_provider import LocalEmbedder
13
20
  from reflexio.server.llm.providers.nomic_embedding_provider import (
14
21
  NomicEmbedder,
15
22
  is_nomic_model,
16
23
  )
17
24
 
18
- _MINILM_MODEL = "local/minilm-l6-v2"
25
+ logger = logging.getLogger(__name__)
26
+
27
+ MINILM_MODEL = "local/minilm-l6-v2"
28
+ NOMIC_TEXT_MODEL = "local/nomic-embed-text-v1.5"
19
29
  _SUPPORTED_MODELS = {
20
- _MINILM_MODEL,
30
+ MINILM_MODEL,
21
31
  "local/nomic-embed-v1.5",
22
- "local/nomic-embed-text-v1.5",
32
+ NOMIC_TEXT_MODEL,
23
33
  }
24
34
  _ACTIVE_MODEL: str | None = None
25
35
  _ACTIVE_MODEL_LOCK = threading.Lock()
26
36
 
27
- app = FastAPI(title="Reflexio Embedding Service")
37
+ DEFAULT_OSS_EMBEDDING_MODEL = MINILM_MODEL
38
+
39
+ # Bound how many embed/encode calls run at once. The endpoint is a sync ``def``
40
+ # served from Starlette's threadpool, so without a guard a burst of requests
41
+ # would run that many model.encode() calls in parallel, stacking their
42
+ # activation memory and OOM-killing the daemon. The semaphore caps simultaneous
43
+ # encodes; excess requests block on acquire() and are picked up when a slot
44
+ # frees — they queue, they are never rejected. Pair with a small encode
45
+ # batch_size so each in-flight encode stays cheap.
46
+ _DEFAULT_MAX_CONCURRENCY = 4
47
+ _ENV_MAX_CONCURRENCY = "REFLEXIO_EMBED_MAX_CONCURRENCY"
48
+ _ENCODE_SEMAPHORE: threading.BoundedSemaphore | None = None
49
+ _ENCODE_SEMAPHORE_LOCK = threading.Lock()
50
+
51
+ # Opportunistically coalesce concurrent small requests before calling
52
+ # model.encode(). This keeps request concurrency thread-based while giving the
53
+ # GPU larger batches during bursts.
54
+ _DEFAULT_MICRO_BATCH_DELAY_MS = 5
55
+ _DEFAULT_MICRO_BATCH_MAX_TEXTS = 64
56
+ _ENV_MICRO_BATCH_DELAY_MS = "REFLEXIO_EMBED_MICRO_BATCH_DELAY_MS"
57
+ _ENV_MICRO_BATCH_MAX_TEXTS = "REFLEXIO_EMBED_MICRO_BATCH_MAX_TEXTS"
58
+ _MICRO_BATCH_CONDITION = threading.Condition()
59
+ _MICRO_BATCH_QUEUE: list[_EmbeddingJob] = []
60
+ _ACTIVE_BATCH_PROCESSORS = 0
61
+ # Failsafe bound for a submitter waiting on its job (see _embed_texts).
62
+ _JOB_WAIT_TIMEOUT_SECONDS = 600.0
63
+
64
+
65
+ @dataclass
66
+ class _EmbeddingJob:
67
+ model: str
68
+ texts: list[str]
69
+ done: threading.Event = field(default_factory=threading.Event)
70
+ result: list[list[float]] | None = None
71
+ error: BaseException | None = None
72
+
73
+
74
+ def _nonnegative_int_env(name: str, default: int) -> int:
75
+ raw = os.environ.get(name)
76
+ if not raw:
77
+ return default
78
+ try:
79
+ value = int(raw)
80
+ except ValueError:
81
+ logger.warning("Invalid %s=%r; falling back to default %d", name, raw, default)
82
+ return default
83
+ return value if value >= 0 else default
84
+
85
+
86
+ def _max_concurrency() -> int:
87
+ """Resolve the max simultaneous encodes from env, defaulting to 4."""
88
+ return positive_int_env(_ENV_MAX_CONCURRENCY, _DEFAULT_MAX_CONCURRENCY, logger)
89
+
90
+
91
+ def _micro_batch_delay_seconds() -> float:
92
+ return (
93
+ _nonnegative_int_env(_ENV_MICRO_BATCH_DELAY_MS, _DEFAULT_MICRO_BATCH_DELAY_MS)
94
+ / 1000
95
+ )
96
+
97
+
98
+ def _micro_batch_max_texts() -> int:
99
+ return positive_int_env(
100
+ _ENV_MICRO_BATCH_MAX_TEXTS, _DEFAULT_MICRO_BATCH_MAX_TEXTS, logger
101
+ )
102
+
103
+
104
+ def _encode_semaphore() -> threading.BoundedSemaphore:
105
+ """Return the process-wide encode semaphore, building it on first use."""
106
+ global _ENCODE_SEMAPHORE
107
+ if _ENCODE_SEMAPHORE is None:
108
+ with _ENCODE_SEMAPHORE_LOCK:
109
+ if _ENCODE_SEMAPHORE is None:
110
+ _ENCODE_SEMAPHORE = threading.BoundedSemaphore(_max_concurrency())
111
+ return _ENCODE_SEMAPHORE
28
112
 
29
113
 
30
114
  class EmbeddingRequest(BaseModel):
@@ -45,36 +129,198 @@ class EmbeddingResponse(BaseModel):
45
129
  model: str
46
130
 
47
131
 
48
- @app.get("/health")
49
- def health() -> dict[str, Any]:
50
- """Health check endpoint."""
51
- return {"status": "ok", "active_model": _ACTIVE_MODEL}
132
+ def create_embedding_app(default_model: str | None = None) -> FastAPI:
133
+ """Create the embedding daemon app and optionally warm a default model."""
52
134
 
135
+ @asynccontextmanager
136
+ async def lifespan(_app: FastAPI) -> AsyncIterator[None]:
137
+ if not default_model:
138
+ yield
139
+ return
140
+ try:
141
+ _embed_texts(default_model, ["reflexio embedding daemon warmup"])
142
+ except Exception:
143
+ logger.exception("Failed to warm embedding model %s", default_model)
144
+ raise
145
+ yield
53
146
 
54
- @app.post("/v1/embeddings")
55
- def create_embeddings(request: EmbeddingRequest) -> EmbeddingResponse:
56
- """Create embeddings using the daemon's single active local model."""
57
- _activate_model(request.model)
58
- texts = [request.input] if isinstance(request.input, str) else list(request.input)
59
- if is_nomic_model(request.model):
60
- embeddings = NomicEmbedder.get().embed(texts)
61
- elif request.model == _MINILM_MODEL:
62
- embeddings = LocalEmbedder.get().embed(texts)
63
- else:
64
- raise HTTPException(
65
- status_code=400, detail=f"Unsupported model: {request.model}"
147
+ embedding_app = FastAPI(title="Reflexio Embedding Service", lifespan=lifespan)
148
+
149
+ @embedding_app.get("/health")
150
+ def health() -> dict[str, Any]:
151
+ """Health check endpoint."""
152
+ return {"status": "ok", "active_model": _ACTIVE_MODEL}
153
+
154
+ @embedding_app.post("/v1/embeddings")
155
+ def create_embeddings(request: EmbeddingRequest) -> EmbeddingResponse:
156
+ """Create embeddings using the daemon's single active local model."""
157
+ texts = (
158
+ [request.input] if isinstance(request.input, str) else list(request.input)
66
159
  )
160
+ embeddings = _embed_texts(request.model, texts)
67
161
 
68
- if request.dimensions:
69
- embeddings = [_resize_embedding(vec, request.dimensions) for vec in embeddings]
162
+ if request.dimensions:
163
+ embeddings = [
164
+ _resize_embedding(vec, request.dimensions) for vec in embeddings
165
+ ]
70
166
 
71
- return EmbeddingResponse(
72
- data=[
73
- EmbeddingData(embedding=embedding, index=index)
74
- for index, embedding in enumerate(embeddings)
75
- ],
76
- model=request.model,
77
- )
167
+ return EmbeddingResponse(
168
+ data=[
169
+ EmbeddingData(embedding=embedding, index=index)
170
+ for index, embedding in enumerate(embeddings)
171
+ ],
172
+ model=request.model,
173
+ )
174
+
175
+ return embedding_app
176
+
177
+
178
+ def _embed_texts(model: str, texts: list[str]) -> list[list[float]]:
179
+ _activate_model(model)
180
+ if not texts:
181
+ return []
182
+
183
+ job = _EmbeddingJob(model=model, texts=list(texts))
184
+ should_process = False
185
+
186
+ global _ACTIVE_BATCH_PROCESSORS
187
+ with _MICRO_BATCH_CONDITION:
188
+ _MICRO_BATCH_QUEUE.append(job)
189
+ if _max_concurrency() > _ACTIVE_BATCH_PROCESSORS:
190
+ _ACTIVE_BATCH_PROCESSORS += 1
191
+ should_process = True
192
+ _MICRO_BATCH_CONDITION.notify()
193
+
194
+ if should_process:
195
+ threading.Thread(
196
+ target=_run_micro_batch_processor,
197
+ daemon=True,
198
+ name="embedding-micro-batch",
199
+ ).start()
200
+
201
+ # Last-resort failsafe: a processor thread that dies between taking and
202
+ # completing jobs would otherwise leave this request hanging forever. The
203
+ # bound is far above any legitimate CPU bulk encode.
204
+ if not job.done.wait(timeout=_JOB_WAIT_TIMEOUT_SECONDS):
205
+ raise RuntimeError(
206
+ f"Embedding micro-batch did not complete within "
207
+ f"{_JOB_WAIT_TIMEOUT_SECONDS:.0f}s"
208
+ )
209
+ if job.error is not None:
210
+ raise job.error
211
+ if job.result is None:
212
+ raise RuntimeError("Embedding micro-batch completed without a result")
213
+ return job.result
214
+
215
+
216
+ def _run_micro_batch_processor() -> None:
217
+ global _ACTIVE_BATCH_PROCESSORS
218
+ try:
219
+ while True:
220
+ jobs = _take_micro_batch()
221
+ if not jobs:
222
+ with _MICRO_BATCH_CONDITION:
223
+ if _MICRO_BATCH_QUEUE:
224
+ continue
225
+ _ACTIVE_BATCH_PROCESSORS -= 1
226
+ _MICRO_BATCH_CONDITION.notify_all()
227
+ return
228
+ _process_micro_batch(jobs)
229
+ except BaseException:
230
+ with _MICRO_BATCH_CONDITION:
231
+ _ACTIVE_BATCH_PROCESSORS -= 1
232
+ _MICRO_BATCH_CONDITION.notify_all()
233
+ raise
234
+
235
+
236
+ def _take_micro_batch() -> list[_EmbeddingJob]:
237
+ max_texts = _micro_batch_max_texts()
238
+ deadline = time.monotonic() + _micro_batch_delay_seconds()
239
+
240
+ with _MICRO_BATCH_CONDITION:
241
+ if not _MICRO_BATCH_QUEUE:
242
+ return []
243
+
244
+ first = _MICRO_BATCH_QUEUE.pop(0)
245
+ jobs = [first]
246
+ total_texts = len(first.texts)
247
+
248
+ while total_texts < max_texts:
249
+ compatible_index = next(
250
+ (
251
+ index
252
+ for index, candidate in enumerate(_MICRO_BATCH_QUEUE)
253
+ if candidate.model == first.model
254
+ and total_texts + len(candidate.texts) <= max_texts
255
+ ),
256
+ None,
257
+ )
258
+ if compatible_index is not None:
259
+ candidate = _MICRO_BATCH_QUEUE.pop(compatible_index)
260
+ jobs.append(candidate)
261
+ total_texts += len(candidate.texts)
262
+ continue
263
+
264
+ remaining = deadline - time.monotonic()
265
+ if remaining <= 0:
266
+ break
267
+ _MICRO_BATCH_CONDITION.wait(timeout=remaining)
268
+
269
+ return jobs
270
+
271
+
272
+ def _process_micro_batch(jobs: list[_EmbeddingJob]) -> None:
273
+ texts: list[str] = []
274
+ slices: list[tuple[_EmbeddingJob, int, int]] = []
275
+ for job in jobs:
276
+ start = len(texts)
277
+ texts.extend(job.texts)
278
+ slices.append((job, start, len(texts)))
279
+
280
+ try:
281
+ embeddings = _encode_texts_now(jobs[0].model, texts)
282
+ except BaseException as exc:
283
+ for job in jobs:
284
+ job.error = exc
285
+ job.done.set()
286
+ return
287
+
288
+ if len(embeddings) != len(texts):
289
+ # Slicing a short result would silently hand jobs truncated/empty
290
+ # vectors; fail every job loudly at the source instead.
291
+ mismatch = RuntimeError(
292
+ f"Embedding count mismatch: encoded {len(texts)} texts but got "
293
+ f"{len(embeddings)} embeddings"
294
+ )
295
+ for job in jobs:
296
+ job.error = mismatch
297
+ job.done.set()
298
+ return
299
+
300
+ for job, start, end in slices:
301
+ job.result = embeddings[start:end]
302
+ job.done.set()
303
+
304
+
305
+ def _encode_texts_now(model: str, texts: list[str]) -> list[list[float]]:
306
+ semaphore = _encode_semaphore()
307
+ # Non-blocking probe purely for observability: if no slot is free, this
308
+ # request will have to wait. The real acquire below blocks until a slot
309
+ # opens, so the request queues and is picked up later — never rejected.
310
+ if not semaphore.acquire(blocking=False):
311
+ logger.info(
312
+ "Embedding request queued; all %d encode slots busy",
313
+ _max_concurrency(),
314
+ )
315
+ semaphore.acquire()
316
+ try:
317
+ if is_nomic_model(model):
318
+ return NomicEmbedder.get().embed(texts)
319
+ if model == MINILM_MODEL:
320
+ return LocalEmbedder.get().embed(texts)
321
+ raise HTTPException(status_code=400, detail=f"Unsupported model: {model}")
322
+ finally:
323
+ semaphore.release()
78
324
 
79
325
 
80
326
  def _activate_model(model: str) -> None:
@@ -108,3 +354,6 @@ def _resize_embedding(vec: list[float], dimensions: int) -> list[float]:
108
354
  if norm <= 0:
109
355
  return sliced
110
356
  return [value / norm for value in sliced]
357
+
358
+
359
+ app = create_embedding_app(default_model=DEFAULT_OSS_EMBEDDING_MODEL)