ltcai 11.7.0 → 11.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +73 -70
- package/docs/BENCHMARKS.md +9 -2
- package/docs/CHANGELOG.md +157 -0
- package/docs/CI_AND_RELEASE_GATES.md +126 -41
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +34 -14
- package/docs/LEGACY_COMPATIBILITY.md +10 -6
- package/docs/ONBOARDING.md +5 -3
- package/docs/OPERATIONS.md +5 -1
- package/docs/PERMISSION_MODE.md +14 -9
- package/docs/TRUST_MODEL.md +13 -6
- package/docs/USABILITY_AUDIT.md +5 -0
- package/docs/WHY_LATTICE.md +6 -4
- package/docs/kg-schema.md +7 -3
- package/docs/mcp-tools.md +73 -79
- package/docs/security-model.md +6 -3
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +1 -54
- package/lattice_brain/graph/_kg_common/text.py +14 -450
- package/lattice_brain/ingestion/__init__.py +6 -3
- package/lattice_brain/multimodal/__init__.py +9 -3
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +12 -1
- package/latticeai/api/models.py +18 -110
- package/latticeai/api/search.py +7 -30
- package/latticeai/api/worker_compute.py +53 -107
- package/latticeai/api/worker_seams.py +17 -2
- package/latticeai/core/http_origin.py +3 -3
- package/latticeai/core/messages.py +0 -5
- package/latticeai/core/policy.py +1 -6
- package/latticeai/core/quiet.py +1 -20
- package/latticeai/core/security.py +29 -83
- package/latticeai/core/sessions.py +95 -4
- package/latticeai/core/users.py +0 -38
- package/latticeai/models/router/catalog.py +2 -2
- package/latticeai/models/router/generation.py +81 -14
- package/latticeai/models/router/loading.py +41 -5
- package/latticeai/runtime/access_runtime.py +7 -4
- package/latticeai/runtime/build_phases/features.py +8 -31
- package/latticeai/runtime/build_phases/foundation.py +7 -16
- package/latticeai/runtime/build_phases/web.py +3 -3
- package/latticeai/runtime/build_phases/worker_profile.py +21 -28
- package/latticeai/runtime/runtime_context.py +0 -2
- package/latticeai/services/architecture_readiness.py +18 -19
- package/latticeai/services/process_audit.py +1 -22
- package/latticeai/services/product_readiness.py +33 -11
- package/latticeai/services/voice_capture.py +8 -28
- package/latticeai/tools/__init__.py +6 -46
- package/latticeai/tools/commands.py +9 -15
- package/latticeai/tools/knowledge.py +0 -6
- package/package.json +3 -5
- package/requirements.txt +0 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_openapi_drift.mjs +3 -2
- package/scripts/check_server_i18n.mjs +5 -4
- package/scripts/compose_openapi.py +2 -1
- package/scripts/export_openapi.py +5 -4
- package/scripts/gen_worker_allowlist_fixture.py +2 -2
- package/scripts/openapi_route_families.json +14 -73
- package/scripts/release_screen_claims.json +131 -27
- package/src-tauri/Cargo.lock +11 -10
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -41
- package/static/app/assets/{Act-BPcVAbOL.js → Act-B4WT81kh.js} +1 -1
- package/static/app/assets/AdminConsole--Wf71m-o.js +1 -0
- package/static/app/assets/{Brain-CT92Kos0.js → Brain-CtYaa26c.js} +2 -2
- package/static/app/assets/BrainHome-Be2VPJEc.js +2 -0
- package/static/app/assets/BrainSignals-C0__xgpG.js +1 -0
- package/static/app/assets/Capture-DdNi5Peb.js +1 -0
- package/static/app/assets/Chronicle-B3hNveeI.js +1 -0
- package/static/app/assets/CommandPalette-CIsnSsFL.js +1 -0
- package/static/app/assets/Library-Bl5XClFV.js +1 -0
- package/static/app/assets/LivingBrain-DjkTB_gh.js +1 -0
- package/static/app/assets/{ProductFlow-DXBC6brE.js → ProductFlow-DCUNRNHs.js} +1 -1
- package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CD3yWvUB.js} +2 -2
- package/static/app/assets/System-BJ6jQ_SL.js +1 -0
- package/static/app/assets/arrow-left-6_28Z0qH.js +1 -0
- package/static/app/assets/{bot-Cn8bWRuq.js → bot-Bc3Q27YR.js} +1 -1
- package/static/app/assets/brain-B9BDrMTe.js +1 -0
- package/static/app/assets/{button-Ct9f2_oT.js → button-D6JcpYcf.js} +1 -1
- package/static/app/assets/circle-check-BFu9lD-3.js +1 -0
- package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-BGMiV8UU.js} +1 -1
- package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-DoanLHnd.js} +1 -1
- package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-DwzNf82m.js} +1 -1
- package/static/app/assets/{download-bv1KEPGQ.js → download-Ddw49yCV.js} +1 -1
- package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-Brd6Kvto.js} +1 -1
- package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-Bu-DTJdB.js} +1 -1
- package/static/app/assets/index-CGdg_aq9.css +2 -0
- package/static/app/assets/{index-Do83hDzJ.js → index-CWKRRsLW.js} +4 -4
- package/static/app/assets/input-CEqsxtil.js +1 -0
- package/static/app/assets/{link-2-BPJOFlAy.js → link-2-DZ4OA5tJ.js} +1 -1
- package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D9TR0F8b.js} +1 -1
- package/static/app/assets/primitives-DORg7Z_7.js +1 -0
- package/static/app/assets/search-0NQ21wXe.js +1 -0
- package/static/app/assets/{share-2-YNX_NtMU.js → share-2-BC5FirFv.js} +1 -1
- package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-DUbR2W2s.js} +1 -1
- package/static/app/assets/textarea-jtQcRSXo.js +1 -0
- package/static/app/assets/{useFocusTrap-ZVI98jaW.js → useFocusTrap-HRemcWId.js} +1 -1
- package/static/app/assets/{useMutation-CVC4qv_D.js → useMutation-DqlFE-Bw.js} +1 -1
- package/static/app/assets/{useQuery-C7BeG4HU.js → useQuery-BizqBNGw.js} +1 -1
- package/static/app/assets/utils-WgW4V69R.js +4 -0
- package/static/app/assets/workspace-DSek3jCY.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/ingestion/pipeline.py +0 -108
- package/latticeai/api/local_files.py +0 -44
- package/latticeai/api/tools.py +0 -126
- package/latticeai/api/voice_capture.py +0 -32
- package/latticeai/core/agent_permission.py +0 -85
- package/scripts/agent_eval.py +0 -34
- package/scripts/brain_quality_eval.py +0 -37
- package/scripts/check_legacy_debt.mjs +0 -91
- package/scripts/check_python.py +0 -100
- package/scripts/chunking_parity_corpus.py +0 -449
- package/scripts/generate_agent_parity_fixtures.py +0 -771
- package/scripts/generate_chunking_parity_fixtures.py +0 -259
- package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
- package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
- package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
- package/static/app/assets/Capture-BsTokYkk.js +0 -1
- package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
- package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
- package/static/app/assets/Library-BGJbG9Hd.js +0 -1
- package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
- package/static/app/assets/System-CMHSO9qM.js +0 -1
- package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
- package/static/app/assets/brain-CQJberbE.js +0 -1
- package/static/app/assets/circle-check-DruOxB-4.js +0 -1
- package/static/app/assets/index-D9x-kSNy.css +0 -2
- package/static/app/assets/input-BLXVNmj1.js +0 -1
- package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
- package/static/app/assets/search-CT9aho2j.js +0 -1
- package/static/app/assets/textarea-DqwLnli4.js +0 -1
- package/static/app/assets/utils-CiFtIdZq.js +0 -4
- package/static/app/assets/workspace-DQz9vIId.js +0 -1
|
@@ -1,9 +1,27 @@
|
|
|
1
|
-
"""
|
|
1
|
+
"""Token hashing, secret redaction, and the worker's per-user rate limiter.
|
|
2
|
+
|
|
3
|
+
What is *not* here any more, and why: password hashing / verification, the
|
|
4
|
+
``client_ip`` resolver with its per-IP login limiter, and the file-magic
|
|
5
|
+
extension check were all ported to ``lattice-auth`` / ``lattice-chat`` in
|
|
6
|
+
v11.6.0, when the front door and every write moved to Rust. They were left
|
|
7
|
+
behind as fully orphaned Python — no route, no service, no script called them
|
|
8
|
+
— and a second copy of an authentication rule that nothing enforces is worse
|
|
9
|
+
than no copy: it reads like a live guard.
|
|
10
|
+
|
|
11
|
+
What stays, stays because something still calls it:
|
|
12
|
+
|
|
13
|
+
* :func:`enforce_rate_limit` — charged by every ``/worker/*`` and ``/agent/*``
|
|
14
|
+
seam through ``_admit``.
|
|
15
|
+
* :func:`redact_secret_text` / :func:`redact_secrets` — and they additionally
|
|
16
|
+
generate the Rust redaction fixture via ``scripts/gen_redact_fixture.py``.
|
|
17
|
+
* the trusted-proxy allowlist — ``_peer_is_trusted_proxy`` is what
|
|
18
|
+
:mod:`latticeai.core.http_origin` (and through it the CSRF check) asks before
|
|
19
|
+
believing an ``X-Forwarded-Host``.
|
|
20
|
+
"""
|
|
2
21
|
|
|
3
22
|
import hashlib
|
|
4
23
|
import ipaddress
|
|
5
24
|
import re
|
|
6
|
-
import secrets
|
|
7
25
|
import threading
|
|
8
26
|
import time
|
|
9
27
|
from typing import Any, Dict, List
|
|
@@ -26,21 +44,6 @@ def sha256_hex(text: str) -> str:
|
|
|
26
44
|
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
27
45
|
|
|
28
46
|
|
|
29
|
-
def hash_password(password: str) -> str:
|
|
30
|
-
salt = secrets.token_hex(16)
|
|
31
|
-
key = hashlib.scrypt(password.encode(), salt=salt.encode(), n=16384, r=8, p=1)
|
|
32
|
-
return f"{salt}:{key.hex()}"
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
def verify_password(password: str, hashed: str) -> bool:
|
|
36
|
-
try:
|
|
37
|
-
salt, key_hex = hashed.split(":", 1)
|
|
38
|
-
key = hashlib.scrypt(password.encode(), salt=salt.encode(), n=16384, r=8, p=1)
|
|
39
|
-
return secrets.compare_digest(key.hex(), key_hex)
|
|
40
|
-
except Exception:
|
|
41
|
-
return False
|
|
42
|
-
|
|
43
|
-
|
|
44
47
|
def host_is_loopback(host: str) -> bool:
|
|
45
48
|
if host in {"localhost", "127.0.0.1", "::1"}:
|
|
46
49
|
return True
|
|
@@ -50,15 +53,15 @@ def host_is_loopback(host: str) -> bool:
|
|
|
50
53
|
return False
|
|
51
54
|
|
|
52
55
|
|
|
53
|
-
# ── Trusted-proxy
|
|
54
|
-
#
|
|
55
|
-
#
|
|
56
|
-
#
|
|
57
|
-
#
|
|
58
|
-
#
|
|
59
|
-
#
|
|
60
|
-
#
|
|
61
|
-
|
|
56
|
+
# ── Trusted-proxy allowlist ──────────────────────────────────────────────────
|
|
57
|
+
# The per-IP login limiter this list was born for is ``lattice-auth``'s now, and
|
|
58
|
+
# ``client_ip`` went with it. The list itself is still load-bearing on this
|
|
59
|
+
# side: :func:`latticeai.core.http_origin.peer_may_forward` — and through it the
|
|
60
|
+
# CSRF front-door check — asks ``_peer_is_trusted_proxy`` whether an
|
|
61
|
+
# ``X-Forwarded-Host`` from this peer may be believed. Off loopback the answer
|
|
62
|
+
# is "only for an operator-listed proxy", which is what makes forging a front
|
|
63
|
+
# door from another machine impossible. ``configure_trusted_proxies`` is the one
|
|
64
|
+
# way that list is ever non-empty, so it stays with the guard it feeds.
|
|
62
65
|
_trusted_proxies: List["ipaddress._BaseNetwork"] = []
|
|
63
66
|
|
|
64
67
|
|
|
@@ -96,46 +99,6 @@ def _peer_is_trusted_proxy(peer: str) -> bool:
|
|
|
96
99
|
return any(addr in net for net in _trusted_proxies)
|
|
97
100
|
|
|
98
101
|
|
|
99
|
-
def client_ip(request) -> str:
|
|
100
|
-
peer = request.client.host if request.client else ""
|
|
101
|
-
# Only a trusted proxy's forwarded headers are honoured; otherwise the
|
|
102
|
-
# client-supplied header is ignored so per-IP rate limits cannot be spoofed.
|
|
103
|
-
if _peer_is_trusted_proxy(peer):
|
|
104
|
-
for header in _FORWARDED_HEADERS:
|
|
105
|
-
val = request.headers.get(header)
|
|
106
|
-
if val:
|
|
107
|
-
candidate = val.split(",")[0].strip()
|
|
108
|
-
try:
|
|
109
|
-
ipaddress.ip_address(candidate)
|
|
110
|
-
return candidate
|
|
111
|
-
except ValueError:
|
|
112
|
-
quiet()
|
|
113
|
-
continue
|
|
114
|
-
return peer or "unknown"
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
_FILE_MAGIC: Dict[str, List[bytes]] = {
|
|
118
|
-
".pdf": [b"%PDF-"],
|
|
119
|
-
".docx": [b"PK\x03\x04"],
|
|
120
|
-
".xlsx": [b"PK\x03\x04"],
|
|
121
|
-
".pptx": [b"PK\x03\x04"],
|
|
122
|
-
".zip": [b"PK\x03\x04", b"PK\x05\x06", b"PK\x07\x08"],
|
|
123
|
-
".png": [b"\x89PNG\r\n\x1a\n"],
|
|
124
|
-
".jpg": [b"\xff\xd8\xff"],
|
|
125
|
-
".jpeg": [b"\xff\xd8\xff"],
|
|
126
|
-
".gif": [b"GIF87a", b"GIF89a"],
|
|
127
|
-
}
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
def bytes_match_extension(data: bytes, ext: str) -> bool:
|
|
131
|
-
ext = (ext or "").lower()
|
|
132
|
-
signatures = _FILE_MAGIC.get(ext)
|
|
133
|
-
if not signatures:
|
|
134
|
-
return True
|
|
135
|
-
head = data[:16]
|
|
136
|
-
return any(head.startswith(sig) for sig in signatures)
|
|
137
|
-
|
|
138
|
-
|
|
139
102
|
SECRET_KEY_HINTS = (
|
|
140
103
|
"api_key",
|
|
141
104
|
"apikey",
|
|
@@ -209,23 +172,6 @@ def redact_secrets(value: Any) -> Any:
|
|
|
209
172
|
return value
|
|
210
173
|
|
|
211
174
|
|
|
212
|
-
# ── IP-based rate limiting (registration / login) ────────────────────────────
|
|
213
|
-
_ip_rate_windows: dict = {}
|
|
214
|
-
_ip_rate_lock = threading.Lock()
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
def check_ip_rate_limit(ip: str, action: str, max_calls: int, window_secs: float) -> None:
|
|
218
|
-
key = (ip, action)
|
|
219
|
-
now = time.time()
|
|
220
|
-
cutoff = now - window_secs
|
|
221
|
-
with _ip_rate_lock:
|
|
222
|
-
calls = [t for t in _ip_rate_windows.get(key, []) if t > cutoff]
|
|
223
|
-
if len(calls) >= max_calls:
|
|
224
|
-
raise HTTPException(status_code=429, detail="요청이 너무 많습니다. 잠시 후 다시 시도하세요.")
|
|
225
|
-
calls.append(now)
|
|
226
|
-
_ip_rate_windows[key] = calls
|
|
227
|
-
|
|
228
|
-
|
|
229
175
|
# ── Per-user token-bucket rate limiting ──────────────────────────────────────
|
|
230
176
|
_RATE_LIMITS = {
|
|
231
177
|
"chat": (30, 0.5),
|
|
@@ -4,6 +4,21 @@ v4: bearer tokens are stored **hashed** (sha256) at rest — a process that can
|
|
|
4
4
|
read ``sessions.json`` must not be able to hijack every session. Pre-v4 files
|
|
5
5
|
holding raw tokens are migrated transparently on load (sessions survive the
|
|
6
6
|
upgrade; the raw token never touches disk again).
|
|
7
|
+
|
|
8
|
+
11.8.0: the in-memory map is a **cache of the file, not the file itself.**
|
|
9
|
+
Since v11.6.0 the writer is ``lattice-auth``: it creates the session and
|
|
10
|
+
appends it to ``sessions.json``, and this process only ever reads. Loading
|
|
11
|
+
once in ``__init__`` therefore made every login that happened after worker
|
|
12
|
+
boot invisible here — silently under ``trusted_local_owner`` (the anonymous
|
|
13
|
+
owner path answers first), and as a flat 401 under
|
|
14
|
+
``LATTICEAI_REQUIRE_AUTH=true``, for a token that is sitting in the file.
|
|
15
|
+
|
|
16
|
+
A lookup that misses now re-reads the file before giving up. Two guards keep
|
|
17
|
+
that from turning a token-guessing burst into a disk-read burst: the re-read
|
|
18
|
+
is skipped when the file's ``mtime_ns``/``size`` are unchanged since the last
|
|
19
|
+
load, and — for a file that *is* changing — throttled to one parse per
|
|
20
|
+
``SESSION_RELOAD_MIN_INTERVAL``. Both are checked under the same lock that
|
|
21
|
+
guards the map, so concurrent misses collapse into one read.
|
|
7
22
|
"""
|
|
8
23
|
|
|
9
24
|
import json
|
|
@@ -19,6 +34,11 @@ from latticeai.core.security import sha256_hex
|
|
|
19
34
|
|
|
20
35
|
SESSION_TTL = 60 * 60 * 24 # 24 hours
|
|
21
36
|
SESSION_REFRESH_THRESHOLD = 60 * 15 # only persist if >15 min since last bump
|
|
37
|
+
#: Floor between two miss-triggered re-reads of ``sessions.json``. A wrong or
|
|
38
|
+
#: expired token is the common case for an unauthenticated probe, and each one
|
|
39
|
+
#: misses; without a floor a burst of them would be a burst of file reads.
|
|
40
|
+
#: Well under any human's retry, so a real login is still picked up at once.
|
|
41
|
+
SESSION_RELOAD_MIN_INTERVAL = 1.0
|
|
22
42
|
_lock = threading.Lock()
|
|
23
43
|
|
|
24
44
|
_HEX64 = frozenset("0123456789abcdef")
|
|
@@ -79,6 +99,19 @@ def persist_sessions(sessions: Dict[str, tuple], data_dir: Optional[Path] = None
|
|
|
79
99
|
logging.warning("persist_sessions failed: %s", e)
|
|
80
100
|
|
|
81
101
|
|
|
102
|
+
def _sessions_stamp(data_dir: Optional[Path] = None) -> Optional[tuple]:
|
|
103
|
+
"""``(mtime_ns, size)`` of ``sessions.json``, or ``None`` when unreadable.
|
|
104
|
+
|
|
105
|
+
``None`` is *not* "unchanged": a file that has just appeared and one that
|
|
106
|
+
was just removed both need the map rebuilt, and both land here.
|
|
107
|
+
"""
|
|
108
|
+
try:
|
|
109
|
+
stat = _sessions_file(data_dir).stat()
|
|
110
|
+
except OSError:
|
|
111
|
+
return None
|
|
112
|
+
return (stat.st_mtime_ns, stat.st_size)
|
|
113
|
+
|
|
114
|
+
|
|
82
115
|
def _entry_subject(entry: tuple) -> Optional[str]:
|
|
83
116
|
return entry[0] if entry else None
|
|
84
117
|
|
|
@@ -107,12 +140,63 @@ class SessionStore:
|
|
|
107
140
|
self._ttl_seconds = int(ttl_seconds or SESSION_TTL)
|
|
108
141
|
self._refresh_threshold_seconds = int(refresh_threshold_seconds or SESSION_REFRESH_THRESHOLD)
|
|
109
142
|
self._sessions: Dict[str, tuple] = load_sessions(data_dir)
|
|
143
|
+
# Stamped *after* the load, so a pre-v4 file that ``load_sessions``
|
|
144
|
+
# rewrote on the way in is not mistaken for a foreign write.
|
|
145
|
+
self._loaded_stamp: Optional[tuple] = _sessions_stamp(data_dir)
|
|
146
|
+
# ``-inf`` and not "now": the first *changed* file must be free to
|
|
147
|
+
# load, or a login that lands in the same second as the worker's boot
|
|
148
|
+
# stays invisible for a second for no reason.
|
|
149
|
+
self._last_reload_at: float = float("-inf")
|
|
150
|
+
|
|
151
|
+
def _reload_if_stale(self, *, force: bool = False) -> bool:
|
|
152
|
+
"""Re-read ``sessions.json`` when it has changed. Caller holds ``_lock``.
|
|
153
|
+
|
|
154
|
+
Returns whether the map was replaced. The two guards answer two
|
|
155
|
+
different costs, in the order that keeps both cheap:
|
|
156
|
+
|
|
157
|
+
* the **stamp** runs first and on every miss. It is one ``stat``, and
|
|
158
|
+
when the file has not moved — the overwhelming case, since a wrong
|
|
159
|
+
token is what an unauthenticated probe sends — that is the entire
|
|
160
|
+
cost. No parse, no allocation.
|
|
161
|
+
* the **interval** runs only once the file *has* moved, and bounds how
|
|
162
|
+
often a busy file is actually parsed. It never drops the change: the
|
|
163
|
+
stamp is recorded only when the load really happened, so the next
|
|
164
|
+
miss past the interval still sees the file as new.
|
|
165
|
+
|
|
166
|
+
``force`` skips the interval only. It is for the two paths that are
|
|
167
|
+
about to *write* the file: a throttled read there would merge onto a
|
|
168
|
+
stale map and drop somebody else's session, and a login or a logout is
|
|
169
|
+
rare enough that the read costs nothing worth counting.
|
|
170
|
+
"""
|
|
171
|
+
stamp = _sessions_stamp(self._data_dir)
|
|
172
|
+
if stamp == self._loaded_stamp:
|
|
173
|
+
return False
|
|
174
|
+
now = time.monotonic()
|
|
175
|
+
if not force and now - self._last_reload_at < SESSION_RELOAD_MIN_INTERVAL:
|
|
176
|
+
return False
|
|
177
|
+
self._last_reload_at = now
|
|
178
|
+
self._sessions = load_sessions(self._data_dir)
|
|
179
|
+
self._loaded_stamp = stamp
|
|
180
|
+
return True
|
|
181
|
+
|
|
182
|
+
def _persist(self) -> None:
|
|
183
|
+
"""Write the map, then re-stamp. Caller holds ``_lock``.
|
|
184
|
+
|
|
185
|
+
Re-stamping matters: without it our own write looks like somebody
|
|
186
|
+
else's the next time a lookup misses, and the store re-reads the file
|
|
187
|
+
it just produced.
|
|
188
|
+
"""
|
|
189
|
+
persist_sessions(self._sessions, self._data_dir)
|
|
190
|
+
self._loaded_stamp = _sessions_stamp(self._data_dir)
|
|
110
191
|
|
|
111
192
|
def create(self, subject: str, *, email: Optional[str] = None) -> str:
|
|
112
193
|
token = secrets.token_urlsafe(32)
|
|
113
194
|
with _lock:
|
|
195
|
+
# Merge onto whatever is on disk rather than over it — this process
|
|
196
|
+
# is not the only writer.
|
|
197
|
+
self._reload_if_stale(force=True)
|
|
114
198
|
self._sessions[_hash_token(token)] = (subject, time.time(), email or subject)
|
|
115
|
-
|
|
199
|
+
self._persist()
|
|
116
200
|
return token
|
|
117
201
|
|
|
118
202
|
def get_email(self, token: str) -> Optional[str]:
|
|
@@ -128,21 +212,28 @@ class SessionStore:
|
|
|
128
212
|
key = _hash_token(token)
|
|
129
213
|
with _lock:
|
|
130
214
|
entry = self._sessions.get(key)
|
|
215
|
+
if entry is None and self._reload_if_stale():
|
|
216
|
+
# A miss is the only thing that can be wrong because the map is
|
|
217
|
+
# stale: an entry we *do* hold was really issued, and one we
|
|
218
|
+
# hold that the file no longer has is handled by the TTL and by
|
|
219
|
+
# ``active_session_email``'s account check on every request.
|
|
220
|
+
entry = self._sessions.get(key)
|
|
131
221
|
if entry is None:
|
|
132
222
|
return None
|
|
133
223
|
created_at = _entry_created_at(entry)
|
|
134
224
|
if now - created_at > self._ttl_seconds:
|
|
135
225
|
self._sessions.pop(key, None)
|
|
136
|
-
|
|
226
|
+
self._persist()
|
|
137
227
|
return None
|
|
138
228
|
if now - created_at > self._refresh_threshold_seconds:
|
|
139
229
|
refreshed = (_entry_subject(entry), now, _entry_email(entry))
|
|
140
230
|
self._sessions[key] = refreshed
|
|
141
|
-
|
|
231
|
+
self._persist()
|
|
142
232
|
return refreshed
|
|
143
233
|
return entry
|
|
144
234
|
|
|
145
235
|
def invalidate(self, token: str) -> None:
|
|
146
236
|
with _lock:
|
|
237
|
+
self._reload_if_stale(force=True)
|
|
147
238
|
self._sessions.pop(_hash_token(token), None)
|
|
148
|
-
|
|
239
|
+
self._persist()
|
package/latticeai/core/users.py
CHANGED
|
@@ -4,9 +4,7 @@ from __future__ import annotations
|
|
|
4
4
|
|
|
5
5
|
import json
|
|
6
6
|
import shutil
|
|
7
|
-
import sqlite3
|
|
8
7
|
import uuid
|
|
9
|
-
from contextlib import closing
|
|
10
8
|
from pathlib import Path
|
|
11
9
|
from typing import Any, Dict, Optional
|
|
12
10
|
|
|
@@ -83,11 +81,6 @@ def load_users_file(path: Path) -> Dict[str, Any]:
|
|
|
83
81
|
return migrated
|
|
84
82
|
|
|
85
83
|
|
|
86
|
-
def save_users_file(path: Path, users: Dict[str, Any]) -> None:
|
|
87
|
-
migrated, _, _ = migrate_users(users)
|
|
88
|
-
atomic_write_json(path, migrated)
|
|
89
|
-
|
|
90
|
-
|
|
91
84
|
def user_id_for_email(users: Dict[str, Any], email: Optional[str]) -> Optional[str]:
|
|
92
85
|
if not email:
|
|
93
86
|
return None
|
|
@@ -98,34 +91,3 @@ def user_id_for_email(users: Dict[str, Any], email: Optional[str]) -> Optional[s
|
|
|
98
91
|
if isinstance(user, dict):
|
|
99
92
|
return user.get("id") or stable_user_id(normalized)
|
|
100
93
|
return stable_user_id(normalized)
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
def migrate_knowledge_graph_identity(db_path: Path, email_to_id: Dict[str, str]) -> int:
|
|
104
|
-
"""Rewrite KG owner/creator identity columns from email to stable UUIDs."""
|
|
105
|
-
if not db_path.exists() or not email_to_id:
|
|
106
|
-
return 0
|
|
107
|
-
changed = 0
|
|
108
|
-
with closing(sqlite3.connect(db_path)) as conn, conn:
|
|
109
|
-
tables = {
|
|
110
|
-
row[0] for row in conn.execute("SELECT name FROM sqlite_master WHERE type='table'")
|
|
111
|
-
}
|
|
112
|
-
for email, user_id in email_to_id.items():
|
|
113
|
-
normalized = normalize_email(email)
|
|
114
|
-
if "nodes_v2" in tables:
|
|
115
|
-
cur = conn.execute("UPDATE nodes_v2 SET owner_id=? WHERE LOWER(owner_id)=?", (user_id, normalized))
|
|
116
|
-
changed += cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0
|
|
117
|
-
if "edges_v2" in tables:
|
|
118
|
-
cur = conn.execute("UPDATE edges_v2 SET created_by=? WHERE LOWER(created_by)=?", (user_id, normalized))
|
|
119
|
-
changed += cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0
|
|
120
|
-
if "ingestion_provenance" in tables:
|
|
121
|
-
cur = conn.execute("UPDATE ingestion_provenance SET owner=? WHERE LOWER(owner)=?", (user_id, normalized))
|
|
122
|
-
changed += cur.rowcount if cur.rowcount and cur.rowcount > 0 else 0
|
|
123
|
-
if changed:
|
|
124
|
-
conn.execute(
|
|
125
|
-
"CREATE TABLE IF NOT EXISTS kg_meta (key TEXT PRIMARY KEY, value TEXT NOT NULL)"
|
|
126
|
-
)
|
|
127
|
-
conn.execute(
|
|
128
|
-
"INSERT OR REPLACE INTO kg_meta(key, value) VALUES('identity_uuid_migrated_at', ?)",
|
|
129
|
-
(_now(),),
|
|
130
|
-
)
|
|
131
|
-
return changed
|
|
@@ -64,6 +64,6 @@ def parse_model_ref(model_id: str) -> tuple[str, str]:
|
|
|
64
64
|
return provider, model
|
|
65
65
|
if provider in {"local_mlx", "mlx"}:
|
|
66
66
|
return "local_mlx", model
|
|
67
|
-
|
|
68
|
-
|
|
67
|
+
# No trailing ``local_mlx:`` case: such a ref contains a colon, so it was
|
|
68
|
+
# already answered by the branch above.
|
|
69
69
|
return "local_mlx", model_id
|
|
@@ -10,7 +10,7 @@ mistaken for the model's answer.
|
|
|
10
10
|
import asyncio
|
|
11
11
|
import base64
|
|
12
12
|
import io
|
|
13
|
-
from typing import Any, AsyncIterator, Optional
|
|
13
|
+
from typing import Any, AsyncIterator, List, Optional
|
|
14
14
|
|
|
15
15
|
from PIL import Image
|
|
16
16
|
|
|
@@ -20,7 +20,46 @@ from ._contract import RouterCore as _Core
|
|
|
20
20
|
from .branding import SYSTEM_PROMPT, _compose_system, normalize_branding
|
|
21
21
|
from .catalog import CloudModel
|
|
22
22
|
from .errors import ModelStreamError, _stream_failure
|
|
23
|
-
from .loading import _mlx_sampler, executor
|
|
23
|
+
from .loading import _mlx_sampler, apply_stop_strings, executor
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _stream_until_stop(
|
|
27
|
+
model,
|
|
28
|
+
tokenizer,
|
|
29
|
+
prompt: str,
|
|
30
|
+
image,
|
|
31
|
+
max_tokens: int,
|
|
32
|
+
sampler,
|
|
33
|
+
draft_model,
|
|
34
|
+
use_vlm: bool,
|
|
35
|
+
stops: List[str],
|
|
36
|
+
) -> str:
|
|
37
|
+
"""Generate on the worker thread, ending at the first stop string.
|
|
38
|
+
|
|
39
|
+
The buffered backends have no stop-string argument, so honouring one means
|
|
40
|
+
driving the *streaming* backend and breaking out of it. That is the whole
|
|
41
|
+
difference between a stop string and trimming the answer afterwards: the
|
|
42
|
+
tokens past the stop are never produced, which is time a 2B model would
|
|
43
|
+
otherwise spend explaining a tool call it has already emitted.
|
|
44
|
+
|
|
45
|
+
Runs inside :data:`executor`, like every other MLX call, so the GPU stream
|
|
46
|
+
stays on one thread.
|
|
47
|
+
"""
|
|
48
|
+
if use_vlm:
|
|
49
|
+
from mlx_vlm import stream_generate as vlm_stream
|
|
50
|
+
|
|
51
|
+
chunks = vlm_stream(model, tokenizer, prompt=prompt, image=image, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model, draft_kind="mtp")
|
|
52
|
+
else:
|
|
53
|
+
from mlx_lm import stream_generate as lm_stream
|
|
54
|
+
|
|
55
|
+
chunks = lm_stream(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model)
|
|
56
|
+
text = ""
|
|
57
|
+
for chunk in chunks:
|
|
58
|
+
piece = chunk.text if hasattr(chunk, "text") else (chunk[0] if isinstance(chunk, tuple) else str(chunk))
|
|
59
|
+
text += piece
|
|
60
|
+
if any(marker in text for marker in stops):
|
|
61
|
+
break
|
|
62
|
+
return apply_stop_strings(text, stops)
|
|
24
63
|
|
|
25
64
|
|
|
26
65
|
class _GenerationMixin(_Core):
|
|
@@ -65,12 +104,22 @@ class _GenerationMixin(_Core):
|
|
|
65
104
|
max_tokens: int = 4096,
|
|
66
105
|
temperature: float = 0.2,
|
|
67
106
|
image_data: Optional[str] = None,
|
|
107
|
+
stop: Optional[List[str]] = None,
|
|
68
108
|
) -> str:
|
|
69
|
-
"""Generate with a request-scoped model without changing the default.
|
|
109
|
+
"""Generate with a request-scoped model without changing the default.
|
|
110
|
+
|
|
111
|
+
``stop`` ends the reply at the first of the given strings. Local
|
|
112
|
+
generation switches to the streaming backend to do it, so the tokens
|
|
113
|
+
after the stop are never produced rather than merely trimmed; the cloud
|
|
114
|
+
path forwards the list to the provider, which stops server-side. Absent
|
|
115
|
+
or empty means what it always meant — generate to ``max_tokens``.
|
|
116
|
+
"""
|
|
70
117
|
_selected, cached = self._model_snapshot(model_id)
|
|
71
118
|
if cached is None:
|
|
72
119
|
return "No model."
|
|
73
|
-
return await self._generate_cached(
|
|
120
|
+
return await self._generate_cached(
|
|
121
|
+
cached, message, context, max_tokens, temperature, image_data, stop
|
|
122
|
+
)
|
|
74
123
|
|
|
75
124
|
async def generate(
|
|
76
125
|
self,
|
|
@@ -79,8 +128,11 @@ class _GenerationMixin(_Core):
|
|
|
79
128
|
max_tokens: int = 4096,
|
|
80
129
|
temperature: float = 0.2,
|
|
81
130
|
image_data: Optional[str] = None,
|
|
131
|
+
stop: Optional[List[str]] = None,
|
|
82
132
|
) -> str:
|
|
83
|
-
return await self.generate_as(
|
|
133
|
+
return await self.generate_as(
|
|
134
|
+
None, message, context, max_tokens, temperature, image_data, stop
|
|
135
|
+
)
|
|
84
136
|
|
|
85
137
|
async def _generate_cached(
|
|
86
138
|
self,
|
|
@@ -90,9 +142,12 @@ class _GenerationMixin(_Core):
|
|
|
90
142
|
max_tokens: int,
|
|
91
143
|
temperature: float,
|
|
92
144
|
image_data: Optional[str],
|
|
145
|
+
stop: Optional[List[str]] = None,
|
|
93
146
|
) -> str:
|
|
94
147
|
if isinstance(cached, CloudModel):
|
|
95
|
-
return await self._cloud_generate(
|
|
148
|
+
return await self._cloud_generate(
|
|
149
|
+
cached, message, context, max_tokens, temperature, stop
|
|
150
|
+
)
|
|
96
151
|
|
|
97
152
|
model, tokenizer, draft_model, loader_kind = self._unpack_local_cache(cached)
|
|
98
153
|
use_vlm = loader_kind == "mlx_vlm"
|
|
@@ -101,27 +156,38 @@ class _GenerationMixin(_Core):
|
|
|
101
156
|
if use_vlm
|
|
102
157
|
else self._build_prompt(message, context, tokenizer)
|
|
103
158
|
)
|
|
104
|
-
|
|
159
|
+
stops = [marker for marker in (stop or []) if marker]
|
|
160
|
+
|
|
105
161
|
loop = asyncio.get_event_loop()
|
|
106
|
-
|
|
162
|
+
|
|
107
163
|
def _gen():
|
|
108
164
|
import mlx.core as mx # type: ignore[no-redef]
|
|
109
165
|
|
|
110
166
|
mx.set_default_device(mx.gpu) # type: ignore[arg-type]
|
|
167
|
+
# Decoded here, not above: base64 + PIL is CPU work, and this
|
|
168
|
+
# function is the part that runs off the event loop.
|
|
169
|
+
image = self._prep_image(image_data) if image_data else None
|
|
170
|
+
sampler = _mlx_sampler(temperature)
|
|
171
|
+
if stops:
|
|
172
|
+
return _stream_until_stop(
|
|
173
|
+
model, tokenizer, prompt, image, max_tokens, sampler, draft_model,
|
|
174
|
+
use_vlm, stops,
|
|
175
|
+
)
|
|
111
176
|
if use_vlm:
|
|
112
177
|
from mlx_vlm import generate as vlm_gen
|
|
113
|
-
return vlm_gen(model, tokenizer, prompt=prompt, image=
|
|
178
|
+
return vlm_gen(model, tokenizer, prompt=prompt, image=image, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model, draft_kind="mtp")
|
|
114
179
|
from mlx_lm import generate as lm_gen
|
|
115
|
-
return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=
|
|
180
|
+
return lm_gen(model, tokenizer, prompt=prompt, max_tokens=max_tokens, sampler=sampler, draft_model=draft_model)
|
|
116
181
|
result = await loop.run_in_executor(executor, _gen)
|
|
117
182
|
# mlx-vlm might return a GenerationResult object; extract the text
|
|
118
|
-
if hasattr(result, "text")
|
|
119
|
-
|
|
120
|
-
return normalize_branding(str(result))
|
|
183
|
+
text = result.text if hasattr(result, "text") else str(result)
|
|
184
|
+
return normalize_branding(apply_stop_strings(text, stops))
|
|
121
185
|
|
|
122
|
-
async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float) -> str:
|
|
186
|
+
async def _cloud_generate(self, cloud: CloudModel, message: str, context: Optional[str], max_tokens: int, temperature: float, stop: Optional[List[str]] = None) -> str:
|
|
123
187
|
context = normalize_branding(context)
|
|
124
188
|
system = _compose_system(SYSTEM_PROMPT, context)
|
|
189
|
+
stops = [marker for marker in (stop or []) if marker]
|
|
190
|
+
extra = {"stop": stops} if stops else {}
|
|
125
191
|
try:
|
|
126
192
|
response = await cloud.client.chat.completions.create(
|
|
127
193
|
model=cloud.model,
|
|
@@ -131,6 +197,7 @@ class _GenerationMixin(_Core):
|
|
|
131
197
|
],
|
|
132
198
|
max_tokens=max_tokens,
|
|
133
199
|
temperature=temperature,
|
|
200
|
+
**extra,
|
|
134
201
|
)
|
|
135
202
|
except Exception as e:
|
|
136
203
|
raise RuntimeError(self._local_server_error_hint(cloud, e)) from e
|
|
@@ -25,6 +25,7 @@ import gc
|
|
|
25
25
|
from concurrent.futures import ThreadPoolExecutor
|
|
26
26
|
from typing import Any, Dict, List, Optional
|
|
27
27
|
|
|
28
|
+
from latticeai.core.quiet import quiet
|
|
28
29
|
from latticeai.models.model_providers import (
|
|
29
30
|
OPENAI_COMPATIBLE_PROVIDERS,
|
|
30
31
|
PROVIDER_MODEL_CATALOG,
|
|
@@ -120,12 +121,47 @@ def ensure_mlx_runtime() -> None:
|
|
|
120
121
|
def _mlx_sampler(temperature: float):
|
|
121
122
|
"""Build an MLX sampler callable for the given temperature.
|
|
122
123
|
|
|
123
|
-
|
|
124
|
-
MLX
|
|
125
|
-
|
|
124
|
+
Until v11.9.0 this discarded its argument and returned ``None``, which let
|
|
125
|
+
MLX pick its own default sampler. The cost was invisible and large: the
|
|
126
|
+
agent loop asks for 0.1 when it wants a tool call and 0.0 on the strict
|
|
127
|
+
verification re-ask, and **locally neither did anything**. A cloud model got
|
|
128
|
+
the deterministic re-ask it was designed around; the 2B model on this
|
|
129
|
+
machine got the same creative sampler it had on the first attempt, and the
|
|
130
|
+
"one strict retry" was one more roll of the same dice.
|
|
131
|
+
|
|
132
|
+
``mlx_lm.sample_utils.make_sampler`` is the sampler both backends accept —
|
|
133
|
+
``mlx_vlm`` takes the callable straight through — and it is already an
|
|
134
|
+
installed dependency of the text path. If it cannot be imported (an
|
|
135
|
+
MLX-VLM-only install, a version that moved it) the answer is the old one:
|
|
136
|
+
``None``, the backend's default. A missing sampler must never be a failed
|
|
137
|
+
generation.
|
|
126
138
|
"""
|
|
127
|
-
|
|
128
|
-
|
|
139
|
+
try:
|
|
140
|
+
from mlx_lm.sample_utils import make_sampler
|
|
141
|
+
except Exception: # noqa: BLE001 — an absent sampler is not a failure
|
|
142
|
+
quiet("MLX sampler unavailable; using the backend default")
|
|
143
|
+
return None
|
|
144
|
+
try:
|
|
145
|
+
return make_sampler(temp=float(temperature))
|
|
146
|
+
except Exception: # noqa: BLE001 — a signature change must not stop a run
|
|
147
|
+
quiet("MLX sampler construction failed; using the backend default")
|
|
148
|
+
return None
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def apply_stop_strings(text: str, stop: Optional[List[str]]) -> str:
|
|
152
|
+
"""Cut ``text`` at the earliest stop string, if any is present.
|
|
153
|
+
|
|
154
|
+
Shared by the buffered and streaming local paths so "where does this reply
|
|
155
|
+
end" has one answer. Empty stop strings are ignored: a caller that sends
|
|
156
|
+
``""`` means "no stop", not "stop before the first character".
|
|
157
|
+
"""
|
|
158
|
+
if not stop:
|
|
159
|
+
return text
|
|
160
|
+
cut = min(
|
|
161
|
+
(text.index(marker) for marker in stop if marker and marker in text),
|
|
162
|
+
default=-1,
|
|
163
|
+
)
|
|
164
|
+
return text if cut < 0 else text[:cut]
|
|
129
165
|
|
|
130
166
|
|
|
131
167
|
class _LoadingMixin(_Core):
|
|
@@ -45,8 +45,10 @@ def build_access_runtime(
|
|
|
45
45
|
# extend this trust to public or non-loopback bindings, even if an invalid
|
|
46
46
|
# caller constructs this runtime with ``require_auth=False`` directly.
|
|
47
47
|
externally_reachable = is_externally_reachable(config)
|
|
48
|
+
# One flag, not two: "authentication is required" is exactly
|
|
49
|
+
# ``not trusted_local_owner``, so deriving it separately only creates a
|
|
50
|
+
# pair that a later edit can put out of step.
|
|
48
51
|
trusted_local_owner = not require_auth and not externally_reachable
|
|
49
|
-
effective_require_auth = bool(require_auth or externally_reachable)
|
|
50
52
|
|
|
51
53
|
def get_user_role(email: str, users: Optional[Dict] = None) -> str:
|
|
52
54
|
users = users or load_users()
|
|
@@ -128,9 +130,10 @@ def build_access_runtime(
|
|
|
128
130
|
# shared local vaults, and Local User profile behavior compatible;
|
|
129
131
|
# get_user_role() supplies the explicit owner authorization role.
|
|
130
132
|
return ""
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
133
|
+
# Reaching here means the caller is not the trusted local owner and no
|
|
134
|
+
# session produced an email — i.e. authentication is required and was
|
|
135
|
+
# not supplied. There is no third outcome to fall through to.
|
|
136
|
+
raise http_exception(status_code=401, detail="인증이 필요합니다.")
|
|
134
137
|
|
|
135
138
|
def require_admin(request: request_type) -> tuple[str, Dict]:
|
|
136
139
|
users = load_users()
|