ltcai 11.7.0 → 11.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. package/README.md +73 -70
  2. package/docs/BENCHMARKS.md +9 -2
  3. package/docs/CHANGELOG.md +157 -0
  4. package/docs/CI_AND_RELEASE_GATES.md +126 -41
  5. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  6. package/docs/DEVELOPMENT.md +34 -14
  7. package/docs/LEGACY_COMPATIBILITY.md +10 -6
  8. package/docs/ONBOARDING.md +5 -3
  9. package/docs/OPERATIONS.md +5 -1
  10. package/docs/PERMISSION_MODE.md +14 -9
  11. package/docs/TRUST_MODEL.md +13 -6
  12. package/docs/USABILITY_AUDIT.md +5 -0
  13. package/docs/WHY_LATTICE.md +6 -4
  14. package/docs/kg-schema.md +7 -3
  15. package/docs/mcp-tools.md +73 -79
  16. package/docs/security-model.md +6 -3
  17. package/lattice_brain/__init__.py +1 -1
  18. package/lattice_brain/graph/_kg_common/__init__.py +1 -54
  19. package/lattice_brain/graph/_kg_common/text.py +14 -450
  20. package/lattice_brain/ingestion/__init__.py +6 -3
  21. package/lattice_brain/multimodal/__init__.py +9 -3
  22. package/latticeai/__init__.py +1 -1
  23. package/latticeai/api/agent_worker_seam.py +12 -1
  24. package/latticeai/api/models.py +18 -110
  25. package/latticeai/api/search.py +7 -30
  26. package/latticeai/api/worker_compute.py +53 -107
  27. package/latticeai/api/worker_seams.py +17 -2
  28. package/latticeai/core/http_origin.py +3 -3
  29. package/latticeai/core/messages.py +0 -5
  30. package/latticeai/core/policy.py +1 -6
  31. package/latticeai/core/quiet.py +1 -20
  32. package/latticeai/core/security.py +29 -83
  33. package/latticeai/core/sessions.py +95 -4
  34. package/latticeai/core/users.py +0 -38
  35. package/latticeai/models/router/catalog.py +2 -2
  36. package/latticeai/models/router/generation.py +81 -14
  37. package/latticeai/models/router/loading.py +41 -5
  38. package/latticeai/runtime/access_runtime.py +7 -4
  39. package/latticeai/runtime/build_phases/features.py +8 -31
  40. package/latticeai/runtime/build_phases/foundation.py +7 -16
  41. package/latticeai/runtime/build_phases/web.py +3 -3
  42. package/latticeai/runtime/build_phases/worker_profile.py +21 -28
  43. package/latticeai/runtime/runtime_context.py +0 -2
  44. package/latticeai/services/architecture_readiness.py +18 -19
  45. package/latticeai/services/process_audit.py +1 -22
  46. package/latticeai/services/product_readiness.py +33 -11
  47. package/latticeai/services/voice_capture.py +8 -28
  48. package/latticeai/tools/__init__.py +6 -46
  49. package/latticeai/tools/commands.py +9 -15
  50. package/latticeai/tools/knowledge.py +0 -6
  51. package/package.json +3 -5
  52. package/requirements.txt +0 -1
  53. package/scripts/check_current_release_docs.mjs +1 -1
  54. package/scripts/check_openapi_drift.mjs +3 -2
  55. package/scripts/check_server_i18n.mjs +5 -4
  56. package/scripts/compose_openapi.py +2 -1
  57. package/scripts/export_openapi.py +5 -4
  58. package/scripts/gen_worker_allowlist_fixture.py +2 -2
  59. package/scripts/openapi_route_families.json +14 -73
  60. package/scripts/release_screen_claims.json +131 -27
  61. package/src-tauri/Cargo.lock +11 -10
  62. package/src-tauri/Cargo.toml +1 -1
  63. package/src-tauri/tauri.conf.json +1 -1
  64. package/static/app/asset-manifest.json +41 -41
  65. package/static/app/assets/{Act-BPcVAbOL.js → Act-B4WT81kh.js} +1 -1
  66. package/static/app/assets/AdminConsole--Wf71m-o.js +1 -0
  67. package/static/app/assets/{Brain-CT92Kos0.js → Brain-CtYaa26c.js} +2 -2
  68. package/static/app/assets/BrainHome-Be2VPJEc.js +2 -0
  69. package/static/app/assets/BrainSignals-C0__xgpG.js +1 -0
  70. package/static/app/assets/Capture-DdNi5Peb.js +1 -0
  71. package/static/app/assets/Chronicle-B3hNveeI.js +1 -0
  72. package/static/app/assets/CommandPalette-CIsnSsFL.js +1 -0
  73. package/static/app/assets/Library-Bl5XClFV.js +1 -0
  74. package/static/app/assets/LivingBrain-DjkTB_gh.js +1 -0
  75. package/static/app/assets/{ProductFlow-DXBC6brE.js → ProductFlow-DCUNRNHs.js} +1 -1
  76. package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CD3yWvUB.js} +2 -2
  77. package/static/app/assets/System-BJ6jQ_SL.js +1 -0
  78. package/static/app/assets/arrow-left-6_28Z0qH.js +1 -0
  79. package/static/app/assets/{bot-Cn8bWRuq.js → bot-Bc3Q27YR.js} +1 -1
  80. package/static/app/assets/brain-B9BDrMTe.js +1 -0
  81. package/static/app/assets/{button-Ct9f2_oT.js → button-D6JcpYcf.js} +1 -1
  82. package/static/app/assets/circle-check-BFu9lD-3.js +1 -0
  83. package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-BGMiV8UU.js} +1 -1
  84. package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-DoanLHnd.js} +1 -1
  85. package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-DwzNf82m.js} +1 -1
  86. package/static/app/assets/{download-bv1KEPGQ.js → download-Ddw49yCV.js} +1 -1
  87. package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-Brd6Kvto.js} +1 -1
  88. package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-Bu-DTJdB.js} +1 -1
  89. package/static/app/assets/index-CGdg_aq9.css +2 -0
  90. package/static/app/assets/{index-Do83hDzJ.js → index-CWKRRsLW.js} +4 -4
  91. package/static/app/assets/input-CEqsxtil.js +1 -0
  92. package/static/app/assets/{link-2-BPJOFlAy.js → link-2-DZ4OA5tJ.js} +1 -1
  93. package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D9TR0F8b.js} +1 -1
  94. package/static/app/assets/primitives-DORg7Z_7.js +1 -0
  95. package/static/app/assets/search-0NQ21wXe.js +1 -0
  96. package/static/app/assets/{share-2-YNX_NtMU.js → share-2-BC5FirFv.js} +1 -1
  97. package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-DUbR2W2s.js} +1 -1
  98. package/static/app/assets/textarea-jtQcRSXo.js +1 -0
  99. package/static/app/assets/{useFocusTrap-ZVI98jaW.js → useFocusTrap-HRemcWId.js} +1 -1
  100. package/static/app/assets/{useMutation-CVC4qv_D.js → useMutation-DqlFE-Bw.js} +1 -1
  101. package/static/app/assets/{useQuery-C7BeG4HU.js → useQuery-BizqBNGw.js} +1 -1
  102. package/static/app/assets/utils-WgW4V69R.js +4 -0
  103. package/static/app/assets/workspace-DSek3jCY.js +1 -0
  104. package/static/app/index.html +4 -4
  105. package/static/sw.js +1 -1
  106. package/lattice_brain/ingestion/pipeline.py +0 -108
  107. package/latticeai/api/local_files.py +0 -44
  108. package/latticeai/api/tools.py +0 -126
  109. package/latticeai/api/voice_capture.py +0 -32
  110. package/latticeai/core/agent_permission.py +0 -85
  111. package/scripts/agent_eval.py +0 -34
  112. package/scripts/brain_quality_eval.py +0 -37
  113. package/scripts/check_legacy_debt.mjs +0 -91
  114. package/scripts/check_python.py +0 -100
  115. package/scripts/chunking_parity_corpus.py +0 -449
  116. package/scripts/generate_agent_parity_fixtures.py +0 -771
  117. package/scripts/generate_chunking_parity_fixtures.py +0 -259
  118. package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
  119. package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
  120. package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
  121. package/static/app/assets/Capture-BsTokYkk.js +0 -1
  122. package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
  123. package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
  124. package/static/app/assets/Library-BGJbG9Hd.js +0 -1
  125. package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
  126. package/static/app/assets/System-CMHSO9qM.js +0 -1
  127. package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
  128. package/static/app/assets/brain-CQJberbE.js +0 -1
  129. package/static/app/assets/circle-check-DruOxB-4.js +0 -1
  130. package/static/app/assets/index-D9x-kSNy.css +0 -2
  131. package/static/app/assets/input-BLXVNmj1.js +0 -1
  132. package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
  133. package/static/app/assets/search-CT9aho2j.js +0 -1
  134. package/static/app/assets/textarea-DqwLnli4.js +0 -1
  135. package/static/app/assets/utils-CiFtIdZq.js +0 -4
  136. package/static/app/assets/workspace-DQz9vIId.js +0 -1
@@ -1,9 +1,27 @@
1
- """Password hashing, rate limiting, IP detection, file-magic validation."""
1
+ """Token hashing, secret redaction, and the worker's per-user rate limiter.
2
+
3
+ What is *not* here any more, and why: password hashing / verification, the
4
+ ``client_ip`` resolver with its per-IP login limiter, and the file-magic
5
+ extension check were all ported to ``lattice-auth`` / ``lattice-chat`` in
6
+ v11.6.0, when the front door and every write moved to Rust. They were left
7
+ behind as fully orphaned Python — no route, no service, no script called them
8
+ — and a second copy of an authentication rule that nothing enforces is worse
9
+ than no copy: it reads like a live guard.
10
+
11
+ What stays, stays because something still calls it:
12
+
13
+ * :func:`enforce_rate_limit` — charged by every ``/worker/*`` and ``/agent/*``
14
+ seam through ``_admit``.
15
+ * :func:`redact_secret_text` / :func:`redact_secrets` — and they additionally
16
+ generate the Rust redaction fixture via ``scripts/gen_redact_fixture.py``.
17
+ * the trusted-proxy allowlist — ``_peer_is_trusted_proxy`` is what
18
+ :mod:`latticeai.core.http_origin` (and through it the CSRF check) asks before
19
+ believing an ``X-Forwarded-Host``.
20
+ """
2
21
 
3
22
  import hashlib
4
23
  import ipaddress
5
24
  import re
6
- import secrets
7
25
  import threading
8
26
  import time
9
27
  from typing import Any, Dict, List
@@ -26,21 +44,6 @@ def sha256_hex(text: str) -> str:
26
44
  return hashlib.sha256(text.encode("utf-8")).hexdigest()
27
45
 
28
46
 
29
- def hash_password(password: str) -> str:
30
- salt = secrets.token_hex(16)
31
- key = hashlib.scrypt(password.encode(), salt=salt.encode(), n=16384, r=8, p=1)
32
- return f"{salt}:{key.hex()}"
33
-
34
-
35
- def verify_password(password: str, hashed: str) -> bool:
36
- try:
37
- salt, key_hex = hashed.split(":", 1)
38
- key = hashlib.scrypt(password.encode(), salt=salt.encode(), n=16384, r=8, p=1)
39
- return secrets.compare_digest(key.hex(), key_hex)
40
- except Exception:
41
- return False
42
-
43
-
44
47
  def host_is_loopback(host: str) -> bool:
45
48
  if host in {"localhost", "127.0.0.1", "::1"}:
46
49
  return True
@@ -50,15 +53,15 @@ def host_is_loopback(host: str) -> bool:
50
53
  return False
51
54
 
52
55
 
53
- # ── Trusted-proxy handling ────────────────────────────────────────────────────
54
- # ``client_ip`` is the key used for IP rate limiting (login / register) and for
55
- # audit logging. A forwarded header (``X-Forwarded-For`` / ``CF-Connecting-IP``)
56
- # is *client-controllable*, so honoring it unconditionally lets anyone spoof
57
- # their source IP and bypass per-IP rate limits. We therefore trust those headers
58
- # ONLY when the direct peer is a configured trusted proxy (e.g. the Cloudflare /
59
- # Vercel edge in front of the app). Default: no trusted proxies → use the peer
60
- # address, which is the safe, local-first behaviour.
61
- _FORWARDED_HEADERS = ("CF-Connecting-IP", "X-Forwarded-For")
56
+ # ── Trusted-proxy allowlist ──────────────────────────────────────────────────
57
+ # The per-IP login limiter this list was born for is ``lattice-auth``'s now, and
58
+ # ``client_ip`` went with it. The list itself is still load-bearing on this
59
+ # side: :func:`latticeai.core.http_origin.peer_may_forward` — and through it the
60
+ # CSRF front-door check — asks ``_peer_is_trusted_proxy`` whether an
61
+ # ``X-Forwarded-Host`` from this peer may be believed. Off loopback the answer
62
+ # is "only for an operator-listed proxy", which is what makes forging a front
63
+ # door from another machine impossible. ``configure_trusted_proxies`` is the one
64
+ # way that list is ever non-empty, so it stays with the guard it feeds.
62
65
  _trusted_proxies: List["ipaddress._BaseNetwork"] = []
63
66
 
64
67
 
@@ -96,46 +99,6 @@ def _peer_is_trusted_proxy(peer: str) -> bool:
96
99
  return any(addr in net for net in _trusted_proxies)
97
100
 
98
101
 
99
- def client_ip(request) -> str:
100
- peer = request.client.host if request.client else ""
101
- # Only a trusted proxy's forwarded headers are honoured; otherwise the
102
- # client-supplied header is ignored so per-IP rate limits cannot be spoofed.
103
- if _peer_is_trusted_proxy(peer):
104
- for header in _FORWARDED_HEADERS:
105
- val = request.headers.get(header)
106
- if val:
107
- candidate = val.split(",")[0].strip()
108
- try:
109
- ipaddress.ip_address(candidate)
110
- return candidate
111
- except ValueError:
112
- quiet()
113
- continue
114
- return peer or "unknown"
115
-
116
-
117
- _FILE_MAGIC: Dict[str, List[bytes]] = {
118
- ".pdf": [b"%PDF-"],
119
- ".docx": [b"PK\x03\x04"],
120
- ".xlsx": [b"PK\x03\x04"],
121
- ".pptx": [b"PK\x03\x04"],
122
- ".zip": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"],
123
- ".png": [b"\x89PNG\r\n\x1a\n"],
124
- ".jpg": [b"\xff\xd8\xff"],
125
- ".jpeg": [b"\xff\xd8\xff"],
126
- ".gif": [b"GIF87a", b"GIF89a"],
127
- }
128
-
129
-
130
- def bytes_match_extension(data: bytes, ext: str) -> bool:
131
- ext = (ext or "").lower()
132
- signatures = _FILE_MAGIC.get(ext)
133
- if not signatures:
134
- return True
135
- head = data[:16]
136
- return any(head.startswith(sig) for sig in signatures)
137
-
138
-
139
102
  SECRET_KEY_HINTS = (
140
103
  "api_key",
141
104
  "apikey",
@@ -209,23 +172,6 @@ def redact_secrets(value: Any) -> Any:
209
172
  return value
210
173
 
211
174
 
212
- # ── IP-based rate limiting (registration / login) ────────────────────────────
213
- _ip_rate_windows: dict = {}
214
- _ip_rate_lock = threading.Lock()
215
-
216
-
217
- def check_ip_rate_limit(ip: str, action: str, max_calls: int, window_secs: float) -> None:
218
- key = (ip, action)
219
- now = time.time()
220
- cutoff = now - window_secs
221
- with _ip_rate_lock:
222
- calls = [t for t in _ip_rate_windows.get(key, []) if t > cutoff]
223
- if len(calls) >= max_calls:
224
- raise HTTPException(status_code=429, detail="요청이 너무 많습니다. 잠시 후 다시 시도하세요.")
225
- calls.append(now)
226
- _ip_rate_windows[key] = calls
227
-
228
-
229
175
  # ── Per-user token-bucket rate limiting ──────────────────────────────────────
230
176
  _RATE_LIMITS = {
231
177
  "chat": (30, 0.5),
@@ -4,6 +4,21 @@ v4: bearer tokens are stored **hashed** (sha256) at rest — a process that can
4
4
  read ``sessions.json`` must not be able to hijack every session. Pre-v4 files
5
5
  holding raw tokens are migrated transparently on load (sessions survive the
6
6
  upgrade; the raw token never touches disk again).
7
+
8
+ 11.8.0: the in-memory map is a **cache of the file, not the file itself.**
9
+ Since v11.6.0 the writer is ``lattice-auth``: it creates the session and
10
+ appends it to ``sessions.json``, and this process only ever reads. Loading
11
+ once in ``__init__`` therefore made every login that happened after worker
12
+ boot invisible here — silently under ``trusted_local_owner`` (the anonymous
13
+ owner path answers first), and as a flat 401 under
14
+ ``LATTICEAI_REQUIRE_AUTH=true``, for a token that is sitting in the file.
15
+
16
+ A lookup that misses now re-reads the file before giving up. Two guards keep
17
+ that from turning a token-guessing burst into a disk-read burst: the re-read
18
+ is skipped when the file's ``mtime_ns``/``size`` are unchanged since the last
19
+ load, and — for a file that *is* changing — throttled to one parse per
20
+ ``SESSION_RELOAD_MIN_INTERVAL``. Both are checked under the same lock that
21
+ guards the map, so concurrent misses collapse into one read.
7
22
  """
8
23
 
9
24
  import json
@@ -19,6 +34,11 @@ from latticeai.core.security import sha256_hex
19
34
 
20
35
  SESSION_TTL = 60 * 60 * 24 # 24 hours
21
36
  SESSION_REFRESH_THRESHOLD = 60 * 15 # only persist if >15 min since last bump
37
+ #: Floor between two miss-triggered re-reads of ``sessions.json``. A wrong or
38
+ #: expired token is the common case for an unauthenticated probe, and each one
39
+ #: misses; without a floor a burst of them would be a burst of file reads.
40
+ #: Well under any human's retry, so a real login is still picked up at once.
41
+ SESSION_RELOAD_MIN_INTERVAL = 1.0
22
42
  _lock = threading.Lock()
23
43
 
24
44
  _HEX64 = frozenset("0123456789abcdef")
@@ -79,6 +99,19 @@ def persist_sessions(sessions: Dict[str, tuple], data_dir: Optional[Path] = None
79
99
  logging.warning("persist_sessions failed: %s", e)
80
100
 
81
101
 
102
+ def _sessions_stamp(data_dir: Optional[Path] = None) -> Optional[tuple]:
103
+ """``(mtime_ns, size)`` of ``sessions.json``, or ``None`` when unreadable.
104
+
105
+ ``None`` is *not* "unchanged": a file that has just appeared and one that
106
+ was just removed both need the map rebuilt, and both land here.
107
+ """
108
+ try:
109
+ stat = _sessions_file(data_dir).stat()
110
+ except OSError:
111
+ return None
112
+ return (stat.st_mtime_ns, stat.st_size)
113
+
114
+
82
115
  def _entry_subject(entry: tuple) -> Optional[str]:
83
116
  return entry[0] if entry else None
84
117
 
@@ -107,12 +140,63 @@ class SessionStore:
107
140
  self._ttl_seconds = int(ttl_seconds or SESSION_TTL)
108
141
  self._refresh_threshold_seconds = int(refresh_threshold_seconds or SESSION_REFRESH_THRESHOLD)
109
142
  self._sessions: Dict[str, tuple] = load_sessions(data_dir)
143
+ # Stamped *after* the load, so a pre-v4 file that ``load_sessions``
144
+ # rewrote on the way in is not mistaken for a foreign write.
145
+ self._loaded_stamp: Optional[tuple] = _sessions_stamp(data_dir)
146
+ # ``-inf`` and not "now": the first *changed* file must be free to
147
+ # load, or a login that lands in the same second as the worker's boot
148
+ # stays invisible for a second for no reason.
149
+ self._last_reload_at: float = float("-inf")
150
+
151
+ def _reload_if_stale(self, *, force: bool = False) -> bool:
152
+ """Re-read ``sessions.json`` when it has changed. Caller holds ``_lock``.
153
+
154
+ Returns whether the map was replaced. The two guards answer two
155
+ different costs, in the order that keeps both cheap:
156
+
157
+ * the **stamp** runs first and on every miss. It is one ``stat``, and
158
+ when the file has not moved — the overwhelming case, since a wrong
159
+ token is what an unauthenticated probe sends — that is the entire
160
+ cost. No parse, no allocation.
161
+ * the **interval** runs only once the file *has* moved, and bounds how
162
+ often a busy file is actually parsed. It never drops the change: the
163
+ stamp is recorded only when the load really happened, so the next
164
+ miss past the interval still sees the file as new.
165
+
166
+ ``force`` skips the interval only. It is for the two paths that are
167
+ about to *write* the file: a throttled read there would merge onto a
168
+ stale map and drop somebody else's session, and a login or a logout is
169
+ rare enough that the read costs nothing worth counting.
170
+ """
171
+ stamp = _sessions_stamp(self._data_dir)
172
+ if stamp == self._loaded_stamp:
173
+ return False
174
+ now = time.monotonic()
175
+ if not force and now - self._last_reload_at < SESSION_RELOAD_MIN_INTERVAL:
176
+ return False
177
+ self._last_reload_at = now
178
+ self._sessions = load_sessions(self._data_dir)
179
+ self._loaded_stamp = stamp
180
+ return True
181
+
182
+ def _persist(self) -> None:
183
+ """Write the map, then re-stamp. Caller holds ``_lock``.
184
+
185
+ Re-stamping matters: without it our own write looks like somebody
186
+ else's the next time a lookup misses, and the store re-reads the file
187
+ it just produced.
188
+ """
189
+ persist_sessions(self._sessions, self._data_dir)
190
+ self._loaded_stamp = _sessions_stamp(self._data_dir)
110
191
 
111
192
  def create(self, subject: str, *, email: Optional[str] = None) -> str:
112
193
  token = secrets.token_urlsafe(32)
113
194
  with _lock:
195
+ # Merge onto whatever is on disk rather than over it — this process
196
+ # is not the only writer.
197
+ self._reload_if_stale(force=True)
114
198
  self._sessions[_hash_token(token)] = (subject, time.time(), email or subject)
115
- persist_sessions(self._sessions, self._data_dir)
199
+ self._persist()
116
200
  return token
117
201
 
118
202
  def get_email(self, token: str) -> Optional[str]:
@@ -128,21 +212,28 @@ class SessionStore:
128
212
  key = _hash_token(token)
129
213
  with _lock:
130
214
  entry = self._sessions.get(key)
215
+ if entry is None and self._reload_if_stale():
216
+ # A miss is the only thing that can be wrong because the map is
217
+ # stale: an entry we *do* hold was really issued, and one we
218
+ # hold that the file no longer has is handled by the TTL and by
219
+ # ``active_session_email``'s account check on every request.
220
+ entry = self._sessions.get(key)
131
221
  if entry is None:
132
222
  return None
133
223
  created_at = _entry_created_at(entry)
134
224
  if now - created_at > self._ttl_seconds:
135
225
  self._sessions.pop(key, None)
136
- persist_sessions(self._sessions, self._data_dir)
226
+ self._persist()
137
227
  return None
138
228
  if now - created_at > self._refresh_threshold_seconds:
139
229
  refreshed = (_entry_subject(entry), now, _entry_email(entry))
140
230
  self._sessions[key] = refreshed
141
- persist_sessions(self._sessions, self._data_dir)
231
+ self._persist()
142
232
  return refreshed
143
233
  return entry
144
234
 
145
235
  def invalidate(self, token: str) -> None:
146
236
  with _lock:
237
+ self._reload_if_stale(force=True)
147
238
  self._sessions.pop(_hash_token(token), None)
148
- persist_sessions(self._sessions, self._data_dir)
239
+ self._persist()
@@ -4,9 +4,7 @@ from __future__ import annotations
4
4
 
5
5
  import json
6
6
  import shutil
7
- import sqlite3
8
7
  import uuid
9
- from contextlib import closing
10
8
  from pathlib import Path
11
9
  from typing import Any, Dict, Optional
12
10
 
@@ -83,11 +81,6 @@ def load_users_file(path: Path) -> Dict[str, Any]:
83
81
  return migrated
84
82
 
85
83
 
86
- def save_users_file(path: Path, users: Dict[str, Any]) -> None:
87
- migrated, _, _ = migrate_users(users)
88
- atomic_write_json(path, migrated)
89
-
90
-
91
84
  def user_id_for_email(users: Dict[str, Any], email: Optional[str]) -> Optional[str]:
92
85
  if not email:
93
86
  return None
@@ -98,34 +91,3 @@ def user_id_for_email(users: Dict[str, Any], email: Optional[str]) -> Optional[s
98
91
  if isinstance(user, dict):
99
92
  return user.get("id") or stable_user_id(normalized)
100
93
  return stable_user_id(normalized)
101
-
102
-
103
- def migrate_knowledge_graph_identity(db_path: Path, email_to_id: Dict[str, str]) -> int:
104
- """Rewrite KG owner/creator identity columns from email to stable UUIDs."""
105
- if not db_path.exists() or not email_to_id:
106
- return 0
107
- changed = 0
108
- with closing(sqlite3.connect(db_path)) as conn, conn:
109
- tables = {
110
- row[0] for row in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")
111
- }
112
- for email, user_id in email_to_id.items():
113
- normalized = normalize_email(email)
114
- if "nodes_v2" in tables:
115
- cur = conn.execute("UPDATE nodes_v2 SET owner_id=? WHERE LOWER(owner_id)=?", (user_id, normalized))
116
- changed += cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0
117
- if "edges_v2" in tables:
118
- cur = conn.execute("UPDATE edges_v2 SET created_by=? WHERE LOWER(created_by)=?", (user_id, normalized))
119
- changed += cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0
120
- if "ingestion_provenance" in tables:
121
- cur = conn.execute("UPDATE ingestion_provenance SET owner=? WHERE LOWER(owner)=?", (user_id, normalized))
122
- changed += cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0
123
- if changed:
124
- conn.execute(
125
- "CREATE TABLE IF NOT EXISTS kg_meta (key TEXT PRIMARY KEY, value TEXT NOT NULL)"
126
- )
127
- conn.execute(
128
- "INSERT OR REPLACE INTO kg_meta(key, value) VALUES('identity_uuid_migrated_at', ?)",
129
- (_now(),),
130
- )
131
- return changed
@@ -64,6 +64,6 @@ def parse_model_ref(model_id: str) -> tuple[str, str]:
64
64
  return provider, model
65
65
  if provider in {"local_mlx", "mlx"}:
66
66
  return "local_mlx", model
67
- if model_id.startswith("local_mlx:"):
68
- return "local_mlx", model_id.split(":", 1)[1] # pragma: no cover — dead: a "local_mlx:" ref always has a ":" and returned above
67
+ # No trailing ``local_mlx:`` case: such a ref contains a colon, so it was
68
+ # already answered by the branch above.
69
69
  return "local_mlx", model_id
@@ -10,7 +10,7 @@ mistaken for the model's answer.
10
10
  import asyncio
11
11
  import base64
12
12
  import io
13
- from typing import Any, AsyncIterator, Optional
13
+ from typing import Any, AsyncIterator, List, Optional
14
14
 
15
15
  from PIL import Image
16
16
 
@@ -20,7 +20,46 @@ from ._contract import RouterCore as _Core
20
20
  from .branding import SYSTEM_PROMPT, _compose_system, normalize_branding
21
21
  from .catalog import CloudModel
22
22
  from .errors import ModelStreamError, _stream_failure
23
- from .loading import _mlx_sampler, executor
23
+ from .loading import _mlx_sampler, apply_stop_strings, executor
24
+
25
+
26
+ def _stream_until_stop(
27
+ model,
28
+ tokenizer,
29
+ prompt: str,
30
+ image,
31
+ max_tokens: int,
32
+ sampler,
33
+ draft_model,
34
+ use_vlm: bool,
35
+ stops: List[str],
36
+ ) -> str:
37
+ """Generate on the worker thread, ending at the first stop string.
38
+
39
+ The buffered backends have no stop-string argument, so honouring one means
40
+ driving the *streaming* backend and breaking out of it. That is the whole
41
+ difference between a stop string and trimming the answer afterwards: the
42
+ tokens past the stop are never produced, which is time a 2B model would
43
+ otherwise spend explaining a tool call it has already emitted.
44
+
45
+ Runs inside :data:`executor`, like every other MLX call, so the GPU stream
46
+ stays on one thread.
47
+ """
48
+ if use_vlm:
49
+ from mlx_vlm import stream_generate as vlm_stream
50
+
51
+ chunks = vlm_stream(model, tokenizer, prompt=prompt, image=image, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model, draft_kind="mtp")
52
+ else:
53
+ from mlx_lm import stream_generate as lm_stream
54
+
55
+ chunks = lm_stream(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model)
56
+ text = ""
57
+ for chunk in chunks:
58
+ piece = chunk.text if hasattr(chunk, "text") else (chunk[0] if isinstance(chunk, tuple) else str(chunk))
59
+ text += piece
60
+ if any(marker in text for marker in stops):
61
+ break
62
+ return apply_stop_strings(text, stops)
24
63
 
25
64
 
26
65
  class _GenerationMixin(_Core):
@@ -65,12 +104,22 @@ class _GenerationMixin(_Core):
65
104
  max_tokens: int = 4096,
66
105
  temperature: float = 0.2,
67
106
  image_data: Optional[str] = None,
107
+ stop: Optional[List[str]] = None,
68
108
  ) -> str:
69
- """Generate with a request-scoped model without changing the default."""
109
+ """Generate with a request-scoped model without changing the default.
110
+
111
+ ``stop`` ends the reply at the first of the given strings. Local
112
+ generation switches to the streaming backend to do it, so the tokens
113
+ after the stop are never produced rather than merely trimmed; the cloud
114
+ path forwards the list to the provider, which stops server-side. Absent
115
+ or empty means what it always meant — generate to ``max_tokens``.
116
+ """
70
117
  _selected, cached = self._model_snapshot(model_id)
71
118
  if cached is None:
72
119
  return "No model."
73
- return await self._generate_cached(cached, message, context, max_tokens, temperature, image_data)
120
+ return await self._generate_cached(
121
+ cached, message, context, max_tokens, temperature, image_data, stop
122
+ )
74
123
 
75
124
  async def generate(
76
125
  self,
@@ -79,8 +128,11 @@ class _GenerationMixin(_Core):
79
128
  max_tokens: int = 4096,
80
129
  temperature: float = 0.2,
81
130
  image_data: Optional[str] = None,
131
+ stop: Optional[List[str]] = None,
82
132
  ) -> str:
83
- return await self.generate_as(None, message, context, max_tokens, temperature, image_data)
133
+ return await self.generate_as(
134
+ None, message, context, max_tokens, temperature, image_data, stop
135
+ )
84
136
 
85
137
  async def _generate_cached(
86
138
  self,
@@ -90,9 +142,12 @@ class _GenerationMixin(_Core):
90
142
  max_tokens: int,
91
143
  temperature: float,
92
144
  image_data: Optional[str],
145
+ stop: Optional[List[str]] = None,
93
146
  ) -> str:
94
147
  if isinstance(cached, CloudModel):
95
- return await self._cloud_generate(cached, message, context, max_tokens, temperature)
148
+ return await self._cloud_generate(
149
+ cached, message, context, max_tokens, temperature, stop
150
+ )
96
151
 
97
152
  model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
98
153
  use_vlm = loader_kind == "mlx_vlm"
@@ -101,27 +156,38 @@ class _GenerationMixin(_Core):
101
156
  if use_vlm
102
157
  else self._build_prompt(message, context, tokenizer)
103
158
  )
104
-
159
+ stops = [marker for marker in (stop or []) if marker]
160
+
105
161
  loop = asyncio.get_event_loop()
106
-
162
+
107
163
  def _gen():
108
164
  import mlx.core as mx # type: ignore[no-redef]
109
165
 
110
166
  mx.set_default_device(mx.gpu) # type: ignore[arg-type]
167
+ # Decoded here, not above: base64 + PIL is CPU work, and this
168
+ # function is the part that runs off the event loop.
169
+ image = self._prep_image(image_data) if image_data else None
170
+ sampler = _mlx_sampler(temperature)
171
+ if stops:
172
+ return _stream_until_stop(
173
+ model, tokenizer, prompt, image, max_tokens, sampler, draft_model,
174
+ use_vlm, stops,
175
+ )
111
176
  if use_vlm:
112
177
  from mlx_vlm import generate as vlm_gen
113
- return vlm_gen(model, tokenizer, prompt=prompt, image=self._prep_image(image_data) if image_data else None, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model, draft_kind="mtp")
178
+ return vlm_gen(model, tokenizer, prompt=prompt, image=image, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model, draft_kind="mtp")
114
179
  from mlx_lm import generate as lm_gen
115
- return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=_mlx_sampler(temperature), draft_model=draft_model)
180
+ return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model)
116
181
  result = await loop.run_in_executor(executor, _gen)
117
182
  # mlx-vlm might return a GenerationResult object; extract the text
118
- if hasattr(result, "text"):
119
- return normalize_branding(result.text)
120
- return normalize_branding(str(result))
183
+ text = result.text if hasattr(result, "text") else str(result)
184
+ return normalize_branding(apply_stop_strings(text, stops))
121
185
 
122
- async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float) -> str:
186
+ async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float, stop: Optional[List[str]] = None) -> str:
123
187
  context = normalize_branding(context)
124
188
  system = _compose_system(SYSTEM_PROMPT, context)
189
+ stops = [marker for marker in (stop or []) if marker]
190
+ extra = {"stop": stops} if stops else {}
125
191
  try:
126
192
  response = await cloud.client.chat.completions.create(
127
193
  model=cloud.model,
@@ -131,6 +197,7 @@ class _GenerationMixin(_Core):
131
197
  ],
132
198
  max_tokens=max_tokens,
133
199
  temperature=temperature,
200
+ **extra,
134
201
  )
135
202
  except Exception as e:
136
203
  raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
@@ -25,6 +25,7 @@ import gc
25
25
  from concurrent.futures import ThreadPoolExecutor
26
26
  from typing import Any, Dict, List, Optional
27
27
 
28
+ from latticeai.core.quiet import quiet
28
29
  from latticeai.models.model_providers import (
29
30
  OPENAI_COMPATIBLE_PROVIDERS,
30
31
  PROVIDER_MODEL_CATALOG,
@@ -120,12 +121,47 @@ def ensure_mlx_runtime() -> None:
120
121
  def _mlx_sampler(temperature: float):
121
122
  """Build an MLX sampler callable for the given temperature.
122
123
 
123
- Lattice v2.2 keeps local execution on MLX-VLM only. Returning ``None`` lets
124
- MLX-VLM use its bundled default sampler without pulling another generation
125
- package into the runtime contract.
124
+ Until v11.9.0 this discarded its argument and returned ``None``, which let
125
+ MLX pick its own default sampler. The cost was invisible and large: the
126
+ agent loop asks for 0.1 when it wants a tool call and 0.0 on the strict
127
+ verification re-ask, and **locally neither did anything**. A cloud model got
128
+ the deterministic re-ask it was designed around; the 2B model on this
129
+ machine got the same creative sampler it had on the first attempt, and the
130
+ "one strict retry" was one more roll of the same dice.
131
+
132
+ ``mlx_lm.sample_utils.make_sampler`` is the sampler both backends accept —
133
+ ``mlx_vlm`` takes the callable straight through — and it is already an
134
+ installed dependency of the text path. If it cannot be imported (an
135
+ MLX-VLM-only install, a version that moved it) the answer is the old one:
136
+ ``None``, the backend's default. A missing sampler must never be a failed
137
+ generation.
126
138
  """
127
- _ = temperature
128
- return
139
+ try:
140
+ from mlx_lm.sample_utils import make_sampler
141
+ except Exception: # noqa: BLE001 — an absent sampler is not a failure
142
+ quiet("MLX sampler unavailable; using the backend default")
143
+ return None
144
+ try:
145
+ return make_sampler(temp=float(temperature))
146
+ except Exception: # noqa: BLE001 — a signature change must not stop a run
147
+ quiet("MLX sampler construction failed; using the backend default")
148
+ return None
149
+
150
+
151
+ def apply_stop_strings(text: str, stop: Optional[List[str]]) -> str:
152
+ """Cut ``text`` at the earliest stop string, if any is present.
153
+
154
+ Shared by the buffered and streaming local paths so "where does this reply
155
+ end" has one answer. Empty stop strings are ignored: a caller that sends
156
+ ``""`` means "no stop", not "stop before the first character".
157
+ """
158
+ if not stop:
159
+ return text
160
+ cut = min(
161
+ (text.index(marker) for marker in stop if marker and marker in text),
162
+ default=-1,
163
+ )
164
+ return text if cut < 0 else text[:cut]
129
165
 
130
166
 
131
167
  class _LoadingMixin(_Core):
@@ -45,8 +45,10 @@ def build_access_runtime(
45
45
  # extend this trust to public or non-loopback bindings, even if an invalid
46
46
  # caller constructs this runtime with ``require_auth=False`` directly.
47
47
  externally_reachable = is_externally_reachable(config)
48
+ # One flag, not two: "authentication is required" is exactly
49
+ # ``not trusted_local_owner``, so deriving it separately only creates a
50
+ # pair that a later edit can put out of step.
48
51
  trusted_local_owner = not require_auth and not externally_reachable
49
- effective_require_auth = bool(require_auth or externally_reachable)
50
52
 
51
53
  def get_user_role(email: str, users: Optional[Dict] = None) -> str:
52
54
  users = users or load_users()
@@ -128,9 +130,10 @@ def build_access_runtime(
128
130
  # shared local vaults, and Local User profile behavior compatible;
129
131
  # get_user_role() supplies the explicit owner authorization role.
130
132
  return ""
131
- if effective_require_auth and not email:
132
- raise http_exception(status_code=401, detail="인증이 필요합니다.")
133
- return email or "" # pragma: no cover — unreachable: trusted_local_owner is the exact complement of effective_require_auth
133
+ # Reaching here means the caller is not the trusted local owner and no
134
+ # session produced an email — i.e. authentication is required and was
135
+ # not supplied. There is no third outcome to fall through to.
136
+ raise http_exception(status_code=401, detail="인증이 필요합니다.")
134
137
 
135
138
  def require_admin(request: request_type) -> tuple[str, Dict]:
136
139
  users = load_users()