ltcai 10.6.0 → 10.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +61 -52
- package/docs/CHANGELOG.md +126 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +1 -1
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_contract.py +15 -0
- package/lattice_brain/graph/provenance.py +61 -5
- package/lattice_brain/graph/retrieval.py +16 -4
- package/lattice_brain/graph/retrieval_reads.py +136 -23
- package/lattice_brain/graph/retrieval_vector.py +156 -8
- package/lattice_brain/ingestion.py +14 -1
- package/lattice_brain/ingestion_jobs.py +255 -5
- package/lattice_brain/portability.py +5 -2
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/browser.py +8 -14
- package/latticeai/api/knowledge_graph.py +46 -16
- package/latticeai/api/workspace.py +4 -11
- package/latticeai/api/workspace_scope.py +125 -0
- package/latticeai/core/agent.py +68 -3
- package/latticeai/core/config.py +6 -0
- package/latticeai/core/csrf.py +293 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/models/router.py +61 -16
- package/latticeai/runtime/build_phases.py +2 -0
- package/latticeai/runtime/config_runtime.py +2 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/runtime/web_runtime.py +31 -4
- package/latticeai/services/architecture_readiness.py +1 -1
- package/latticeai/services/product_readiness.py +1 -1
- package/latticeai/services/search_service.py +9 -1
- package/latticeai/services/tool_dispatch.py +13 -7
- package/latticeai/tools/__init__.py +1 -0
- package/latticeai/tools/documents.py +44 -13
- package/package.json +1 -1
- package/scripts/build_frontend_assets.mjs +21 -4
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_frontend_build_freshness.mjs +170 -0
- package/src-tauri/Cargo.lock +1 -1
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +39 -39
- package/static/app/assets/{Act-B3MSgNsJ.js → Act-aNud-lKL.js} +1 -1
- package/static/app/assets/{AdminConsole-Ds8u36lf.js → AdminConsole-QiCTH68K.js} +1 -1
- package/static/app/assets/{Brain-XdHCIB6a.js → Brain-CMFh5q6k.js} +1 -1
- package/static/app/assets/{BrainHome-Dp8gzQoF.js → BrainHome-B4Ar3atu.js} +2 -2
- package/static/app/assets/{BrainSignals-D4yZflVt.js → BrainSignals-RVMlmTGC.js} +1 -1
- package/static/app/assets/{Capture-BwSZmiZ8.js → Capture-DEBo-0vZ.js} +1 -1
- package/static/app/assets/{CommandPalette-Ds0DnRSC.js → CommandPalette-bW5WxWWZ.js} +1 -1
- package/static/app/assets/{Library-SqzHjyfx.js → Library-5gFexm83.js} +1 -1
- package/static/app/assets/{LivingBrain-DpKt-NKE.js → LivingBrain-FlHFfu9i.js} +1 -1
- package/static/app/assets/ProductFlow-CFUNDOHu.js +1 -0
- package/static/app/assets/{ReviewCard-DVPi1LPZ.js → ReviewCard-BHj86h2Z.js} +2 -2
- package/static/app/assets/{System-w-9miIG8.js → System-CeNHoZRu.js} +1 -1
- package/static/app/assets/{activity-C0QavXjd.js → activity-7_ZmZqN0.js} +1 -1
- package/static/app/assets/arrow-left-Dv2Tiwhe.js +1 -0
- package/static/app/assets/{bot-C1-MbzCF.js → bot-D1gX4xks.js} +1 -1
- package/static/app/assets/{brain-CeRqzJVc.js → brain-Dn_bDfl4.js} +1 -1
- package/static/app/assets/{button-Z-2N8PUb.js → button-DPFZ9lGw.js} +1 -1
- package/static/app/assets/{circle-pause-CgjCLdmO.js → circle-pause-B2P72YOO.js} +1 -1
- package/static/app/assets/{circle-play-Bj45uClm.js → circle-play-BiC8z2vg.js} +1 -1
- package/static/app/assets/{cpu-B-2Chwfy.js → cpu-Cx-wRR_V.js} +1 -1
- package/static/app/assets/{download-CKmiOG3C.js → download-CKxlzxsK.js} +1 -1
- package/static/app/assets/{folder-open-fE7lLvhZ.js → folder-open-Bbjju0tt.js} +1 -1
- package/static/app/assets/{hard-drive-rDl1xsJ1.js → hard-drive-Bon1VBvq.js} +1 -1
- package/static/app/assets/index-DYUs0cWy.css +2 -0
- package/static/app/assets/{index-AEIqmjwZ.js → index-mLP0-YNO.js} +3 -3
- package/static/app/assets/{input-BflJYT5-.js → input-DrMc0Xns.js} +1 -1
- package/static/app/assets/{permissionCopy-BNNvkXSX.js → permissionCopy-BGUsI7vw.js} +1 -1
- package/static/app/assets/{primitives-CP68OWk2.js → primitives-CEMTjBz1.js} +1 -1
- package/static/app/assets/search-HsIji1wY.js +1 -0
- package/static/app/assets/{share-2-CyjG2_yY.js → share-2-D7THHq5K.js} +1 -1
- package/static/app/assets/{shield-alert-C_JBPWYd.js → shield-alert-BoU4_8r9.js} +1 -1
- package/static/app/assets/{textarea-Ds1Exelb.js → textarea-rU1Lb7ia.js} +1 -1
- package/static/app/assets/{useFocusTrap-BgIZQ4if.js → useFocusTrap-BgvK4Nkx.js} +1 -1
- package/static/app/assets/{useQuery-6Bu27NQe.js → useQuery-CT2ChyuU.js} +1 -1
- package/static/app/assets/{users-31lxlciQ.js → users-DlbfHBQV.js} +1 -1
- package/static/app/assets/{utils-C0-C5mZc.js → utils-fEGWreKB.js} +1 -1
- package/static/app/assets/workspace-ClDBz_0f.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/static/app/assets/ProductFlow-BimeVtWH.js +0 -1
- package/static/app/assets/arrow-left-TsAKz-s_.js +0 -1
- package/static/app/assets/index-DcMODGjM.css +0 -2
- package/static/app/assets/search-B4O4iIgg.js +0 -1
- package/static/app/assets/workspace-DAB-urHL.js +0 -1
|
@@ -9,15 +9,29 @@ progress* — a genuinely separate concern with its own frozen wire schema
|
|
|
9
9
|
The seam is also where a real scheduler (thread pool, rq, celery) would plug
|
|
10
10
|
in without touching the pipeline.
|
|
11
11
|
|
|
12
|
+
Job state is **durable** (review 2026-08 P1 #3): ``done_indices`` used to live
|
|
13
|
+
only in this process's heap, so a restart mid-import silently lost the resume
|
|
14
|
+
point and re-ingesting meant replaying every item. :class:`IngestionJobStore`
|
|
15
|
+
persists the queue to SQLite — by default the same database file the knowledge
|
|
16
|
+
graph and :mod:`lattice_brain.conversations` use, so the existing
|
|
17
|
+
backup/restore covers it with no manifest change. A queue built without a
|
|
18
|
+
``db_path`` stays purely in-memory and says so through :meth:`describe`.
|
|
19
|
+
|
|
12
20
|
``IngestionItem`` is imported only for type checking: the pipeline module owns
|
|
13
21
|
that dataclass, and a runtime import here would be circular.
|
|
14
22
|
"""
|
|
15
23
|
|
|
16
24
|
from __future__ import annotations
|
|
17
25
|
|
|
26
|
+
import dataclasses
|
|
27
|
+
import json
|
|
28
|
+
import logging
|
|
29
|
+
import sqlite3
|
|
18
30
|
import threading
|
|
31
|
+
from contextlib import contextmanager
|
|
19
32
|
from dataclasses import dataclass, field
|
|
20
|
-
from
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import TYPE_CHECKING, Any, Dict, Iterator, List, Optional, Set
|
|
21
35
|
|
|
22
36
|
from .utils import utc_now_iso
|
|
23
37
|
|
|
@@ -27,6 +41,9 @@ if TYPE_CHECKING: # pragma: no cover - typing only
|
|
|
27
41
|
|
|
28
42
|
JOB_ERRORS_CAP = 50 # per-job error records kept (failed count keeps counting)
|
|
29
43
|
|
|
44
|
+
#: Statuses a job may hold on disk. Frozen wire schema of ``/api/ingestion/jobs*``.
|
|
45
|
+
JOB_STATUSES = ("queued", "running", "completed", "failed", "partial")
|
|
46
|
+
|
|
30
47
|
|
|
31
48
|
@dataclass
|
|
32
49
|
class BackgroundIngestionJob:
|
|
@@ -79,17 +96,231 @@ class BackgroundIngestionJob:
|
|
|
79
96
|
}
|
|
80
97
|
|
|
81
98
|
|
|
99
|
+
def _item_payload(item: IngestionItem) -> Dict[str, Any]:
|
|
100
|
+
"""One ingestion item as JSON-safe data (dataclass fields only)."""
|
|
101
|
+
return dataclasses.asdict(item)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _item_from_payload(payload: Dict[str, Any]) -> IngestionItem:
|
|
105
|
+
"""Rebuild an ``IngestionItem`` from persisted data.
|
|
106
|
+
|
|
107
|
+
Unknown keys are dropped rather than raising: a job written by an older
|
|
108
|
+
build must still resume on a newer one. Imported here (not at module
|
|
109
|
+
scope) because ``lattice_brain.ingestion`` imports *this* module.
|
|
110
|
+
"""
|
|
111
|
+
from .ingestion import IngestionItem as _Item
|
|
112
|
+
|
|
113
|
+
known = {f.name for f in dataclasses.fields(_Item)}
|
|
114
|
+
kwargs = {key: value for key, value in payload.items() if key in known}
|
|
115
|
+
metadata = kwargs.get("metadata")
|
|
116
|
+
if not isinstance(metadata, dict):
|
|
117
|
+
kwargs["metadata"] = {}
|
|
118
|
+
return _Item(**kwargs)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class IngestionJobStore:
|
|
122
|
+
"""SQLite persistence for :class:`BackgroundIngestionJob`.
|
|
123
|
+
|
|
124
|
+
Own connection per operation, always closed — ``with sqlite3.connect(...)``
|
|
125
|
+
commits but never closes (same note as
|
|
126
|
+
:meth:`lattice_brain.conversations.ConversationStore._connect`).
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
def __init__(self, db_path: Path) -> None:
|
|
130
|
+
self.db_path = Path(db_path)
|
|
131
|
+
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
132
|
+
self._init_db()
|
|
133
|
+
|
|
134
|
+
@contextmanager
|
|
135
|
+
def _connect(self) -> Iterator[sqlite3.Connection]:
|
|
136
|
+
conn = sqlite3.connect(str(self.db_path))
|
|
137
|
+
conn.row_factory = sqlite3.Row
|
|
138
|
+
conn.execute("PRAGMA journal_mode=WAL")
|
|
139
|
+
try:
|
|
140
|
+
with conn:
|
|
141
|
+
yield conn
|
|
142
|
+
finally:
|
|
143
|
+
conn.close()
|
|
144
|
+
|
|
145
|
+
def _init_db(self) -> None:
|
|
146
|
+
with self._connect() as conn:
|
|
147
|
+
conn.executescript(
|
|
148
|
+
"""
|
|
149
|
+
CREATE TABLE IF NOT EXISTS ingestion_jobs (
|
|
150
|
+
job_id TEXT PRIMARY KEY,
|
|
151
|
+
status TEXT NOT NULL,
|
|
152
|
+
total INTEGER NOT NULL DEFAULT 0,
|
|
153
|
+
processed INTEGER NOT NULL DEFAULT 0,
|
|
154
|
+
failed INTEGER NOT NULL DEFAULT 0,
|
|
155
|
+
incremental INTEGER NOT NULL DEFAULT 1,
|
|
156
|
+
user_email TEXT,
|
|
157
|
+
max_errors INTEGER NOT NULL DEFAULT 50,
|
|
158
|
+
items_json TEXT NOT NULL DEFAULT '[]',
|
|
159
|
+
done_indices_json TEXT NOT NULL DEFAULT '[]',
|
|
160
|
+
errors_json TEXT NOT NULL DEFAULT '[]',
|
|
161
|
+
created_at TEXT NOT NULL,
|
|
162
|
+
updated_at TEXT NOT NULL
|
|
163
|
+
);
|
|
164
|
+
CREATE INDEX IF NOT EXISTS idx_ingestion_jobs_status
|
|
165
|
+
ON ingestion_jobs(status);
|
|
166
|
+
CREATE INDEX IF NOT EXISTS idx_ingestion_jobs_created
|
|
167
|
+
ON ingestion_jobs(created_at);
|
|
168
|
+
"""
|
|
169
|
+
)
|
|
170
|
+
|
|
171
|
+
def save(self, job: BackgroundIngestionJob) -> None:
|
|
172
|
+
with self._connect() as conn:
|
|
173
|
+
conn.execute(
|
|
174
|
+
"""
|
|
175
|
+
INSERT INTO ingestion_jobs(
|
|
176
|
+
job_id, status, total, processed, failed, incremental, user_email,
|
|
177
|
+
max_errors, items_json, done_indices_json, errors_json,
|
|
178
|
+
created_at, updated_at)
|
|
179
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
180
|
+
ON CONFLICT(job_id) DO UPDATE SET
|
|
181
|
+
status=excluded.status,
|
|
182
|
+
total=excluded.total,
|
|
183
|
+
processed=excluded.processed,
|
|
184
|
+
failed=excluded.failed,
|
|
185
|
+
incremental=excluded.incremental,
|
|
186
|
+
user_email=excluded.user_email,
|
|
187
|
+
max_errors=excluded.max_errors,
|
|
188
|
+
items_json=excluded.items_json,
|
|
189
|
+
done_indices_json=excluded.done_indices_json,
|
|
190
|
+
errors_json=excluded.errors_json,
|
|
191
|
+
updated_at=excluded.updated_at
|
|
192
|
+
""",
|
|
193
|
+
(
|
|
194
|
+
job.job_id,
|
|
195
|
+
job.status,
|
|
196
|
+
int(job.total),
|
|
197
|
+
int(job.processed),
|
|
198
|
+
int(job.failed),
|
|
199
|
+
1 if job.incremental else 0,
|
|
200
|
+
job.user_email,
|
|
201
|
+
int(job.max_errors),
|
|
202
|
+
json.dumps(
|
|
203
|
+
[_item_payload(item) for item in job.items],
|
|
204
|
+
ensure_ascii=False,
|
|
205
|
+
default=str,
|
|
206
|
+
),
|
|
207
|
+
json.dumps(sorted(job.done_indices)),
|
|
208
|
+
json.dumps(list(job.errors), ensure_ascii=False, default=str),
|
|
209
|
+
job.created_at,
|
|
210
|
+
job.updated_at,
|
|
211
|
+
),
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
def load_all(self) -> List[BackgroundIngestionJob]:
|
|
215
|
+
"""Every persisted job, oldest first.
|
|
216
|
+
|
|
217
|
+
A row still marked ``running`` means the process died mid-run. It is
|
|
218
|
+
reported as ``partial``/``queued`` (whichever the recorded progress
|
|
219
|
+
supports) so it is honestly *not* running and so
|
|
220
|
+
``run_background_job`` will pick it up instead of refusing.
|
|
221
|
+
"""
|
|
222
|
+
with self._connect() as conn:
|
|
223
|
+
rows = conn.execute(
|
|
224
|
+
"SELECT * FROM ingestion_jobs ORDER BY created_at ASC, job_id ASC"
|
|
225
|
+
).fetchall()
|
|
226
|
+
jobs: List[BackgroundIngestionJob] = []
|
|
227
|
+
for row in rows:
|
|
228
|
+
try:
|
|
229
|
+
jobs.append(self._row_to_job(row))
|
|
230
|
+
except Exception as exc: # noqa: BLE001 — one bad row must not hide the rest
|
|
231
|
+
logging.warning(
|
|
232
|
+
"ingestion job %s could not be restored: %s", row["job_id"], exc
|
|
233
|
+
)
|
|
234
|
+
return jobs
|
|
235
|
+
|
|
236
|
+
@staticmethod
|
|
237
|
+
def _row_to_job(row: sqlite3.Row) -> BackgroundIngestionJob:
|
|
238
|
+
items = [
|
|
239
|
+
_item_from_payload(payload)
|
|
240
|
+
for payload in json.loads(row["items_json"] or "[]")
|
|
241
|
+
]
|
|
242
|
+
done = {int(i) for i in json.loads(row["done_indices_json"] or "[]")}
|
|
243
|
+
status = str(row["status"] or "queued")
|
|
244
|
+
if status == "running":
|
|
245
|
+
status = "partial" if done else "queued"
|
|
246
|
+
return BackgroundIngestionJob(
|
|
247
|
+
job_id=str(row["job_id"]),
|
|
248
|
+
items=items,
|
|
249
|
+
status=status,
|
|
250
|
+
created_at=str(row["created_at"]),
|
|
251
|
+
updated_at=str(row["updated_at"]),
|
|
252
|
+
processed=len(done),
|
|
253
|
+
failed=int(row["failed"] or 0),
|
|
254
|
+
total=int(row["total"] or 0),
|
|
255
|
+
errors=list(json.loads(row["errors_json"] or "[]")),
|
|
256
|
+
incremental=bool(row["incremental"]),
|
|
257
|
+
user_email=row["user_email"],
|
|
258
|
+
max_errors=int(row["max_errors"] or JOB_ERRORS_CAP),
|
|
259
|
+
done_indices=done,
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
|
|
82
263
|
class BackgroundIngestionQueue:
|
|
83
|
-
"""
|
|
264
|
+
"""Queue for background incremental ingestion, durable when given a db.
|
|
84
265
|
|
|
85
266
|
For large corpus: this is the seam where a real scheduler / worker pool
|
|
86
267
|
(celery, rq, or internal thread) can be plugged later without changing callers.
|
|
87
268
|
Supports incremental (skip duplicates) vs force reindex.
|
|
269
|
+
|
|
270
|
+
``db_path`` makes job state survive a restart. The in-process dict stays
|
|
271
|
+
the authority *within* a process (callers hold job references and mutate
|
|
272
|
+
them), and every mutation is mirrored to SQLite through :meth:`save`.
|
|
273
|
+
Without ``db_path`` — or when the database cannot be opened — the queue is
|
|
274
|
+
memory-only and :meth:`describe` reports that instead of implying
|
|
275
|
+
durability it does not have.
|
|
88
276
|
"""
|
|
89
|
-
|
|
277
|
+
|
|
278
|
+
def __init__(self, db_path: Optional[Any] = None) -> None:
|
|
90
279
|
self._jobs: Dict[str, BackgroundIngestionJob] = {}
|
|
91
280
|
self._counter = 0
|
|
92
|
-
self._lock = threading.
|
|
281
|
+
self._lock = threading.RLock()
|
|
282
|
+
self._store: Optional[IngestionJobStore] = None
|
|
283
|
+
self._persistence_detail: Optional[str] = None
|
|
284
|
+
if db_path is None:
|
|
285
|
+
self._persistence_detail = "no database configured; job state is in-memory only"
|
|
286
|
+
elif not isinstance(db_path, (str, Path)):
|
|
287
|
+
self._persistence_detail = (
|
|
288
|
+
f"unusable db_path {type(db_path).__name__}; job state is in-memory only"
|
|
289
|
+
)
|
|
290
|
+
else:
|
|
291
|
+
try:
|
|
292
|
+
self._store = IngestionJobStore(Path(db_path))
|
|
293
|
+
for job in self._store.load_all():
|
|
294
|
+
self._jobs[job.job_id] = job
|
|
295
|
+
self._counter = max(self._counter, _job_sequence(job.job_id))
|
|
296
|
+
except Exception as exc: # noqa: BLE001 — degrade to memory, never block ingestion
|
|
297
|
+
self._store = None
|
|
298
|
+
self._persistence_detail = f"job persistence unavailable: {exc}"
|
|
299
|
+
logging.warning("ingestion job persistence unavailable: %s", exc)
|
|
300
|
+
|
|
301
|
+
# ── honesty surface ──────────────────────────────────────────────────────
|
|
302
|
+
def describe(self) -> Dict[str, Any]:
|
|
303
|
+
"""Whether resume state actually survives a restart, and where."""
|
|
304
|
+
return {
|
|
305
|
+
"persistent": self._store is not None,
|
|
306
|
+
"db_path": str(self._store.db_path) if self._store is not None else None,
|
|
307
|
+
"jobs": len(self._jobs),
|
|
308
|
+
"detail": self._persistence_detail,
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
def save(self, job: BackgroundIngestionJob) -> None:
|
|
312
|
+
"""Mirror a job's current state to disk (no-op when memory-only).
|
|
313
|
+
|
|
314
|
+
A persistence failure degrades durability, never the run in progress:
|
|
315
|
+
the item work already succeeded and must not be rolled back by a
|
|
316
|
+
bookkeeping error.
|
|
317
|
+
"""
|
|
318
|
+
if self._store is None:
|
|
319
|
+
return
|
|
320
|
+
try:
|
|
321
|
+
self._store.save(job)
|
|
322
|
+
except Exception as exc: # noqa: BLE001 — durability is best-effort mid-run
|
|
323
|
+
logging.warning("ingestion job %s could not be persisted: %s", job.job_id, exc)
|
|
93
324
|
|
|
94
325
|
def schedule(
|
|
95
326
|
self,
|
|
@@ -114,6 +345,7 @@ class BackgroundIngestionQueue:
|
|
|
114
345
|
it.metadata = {**it.metadata, "incremental": incremental, "bg_job": job_id}
|
|
115
346
|
with self._lock:
|
|
116
347
|
self._jobs[job_id] = job
|
|
348
|
+
self.save(job)
|
|
117
349
|
return job
|
|
118
350
|
|
|
119
351
|
def get(self, job_id: str) -> Optional[BackgroundIngestionJob]:
|
|
@@ -130,4 +362,22 @@ class BackgroundIngestionQueue:
|
|
|
130
362
|
return list(reversed(jobs))[:limit]
|
|
131
363
|
|
|
132
364
|
|
|
133
|
-
|
|
365
|
+
def _job_sequence(job_id: str) -> int:
|
|
366
|
+
"""The numeric suffix of ``bg_ingest_0007`` → 7 (0 when unparseable).
|
|
367
|
+
|
|
368
|
+
Restored jobs must not have their ids handed out again after a restart.
|
|
369
|
+
"""
|
|
370
|
+
_, _, suffix = str(job_id or "").rpartition("_")
|
|
371
|
+
try:
|
|
372
|
+
return int(suffix)
|
|
373
|
+
except ValueError:
|
|
374
|
+
return 0
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
__all__ = [
|
|
378
|
+
"JOB_ERRORS_CAP",
|
|
379
|
+
"JOB_STATUSES",
|
|
380
|
+
"BackgroundIngestionJob",
|
|
381
|
+
"BackgroundIngestionQueue",
|
|
382
|
+
"IngestionJobStore",
|
|
383
|
+
]
|
|
@@ -4,8 +4,11 @@ The Knowledge Graph is the user's durable asset, so it must be portable without
|
|
|
4
4
|
any cloud service. Two complementary mechanisms, both fully local:
|
|
5
5
|
|
|
6
6
|
* **Logical export/import** (JSON): nodes/edges/chunks/sources/provenance with a
|
|
7
|
-
versioned header (schema + projection + embed-dim).
|
|
8
|
-
|
|
7
|
+
versioned header (schema + projection + embed-dim). Vectors are not in the
|
|
8
|
+
artifact; the importer re-embeds with its own embedder and reports the
|
|
9
|
+
resulting index state under ``result["index"]`` (``degraded: true`` means the
|
|
10
|
+
content landed but recall is lexical-only until a rebuild succeeds). That is
|
|
11
|
+
what makes it portable across machines.
|
|
9
12
|
* **Binary backup/restore** (ZIP): a faithful snapshot of the SQLite DB (incl.
|
|
10
13
|
vector embeddings) plus the blob directory, integrity-checked, for
|
|
11
14
|
same-machine recovery.
|
|
@@ -47,7 +47,7 @@ from typing import Any, Callable, Dict, List, Optional
|
|
|
47
47
|
from ..utils import now_iso as _now
|
|
48
48
|
from .contracts import multi_agent_contract
|
|
49
49
|
|
|
50
|
-
MULTI_AGENT_VERSION = "10.6.
|
|
50
|
+
MULTI_AGENT_VERSION = "10.6.2"
|
|
51
51
|
|
|
52
52
|
AGENT_ROLES = ("researcher", "planner", "executor", "reviewer", "release")
|
|
53
53
|
CORE_PIPELINE = ("planner", "executor", "reviewer")
|
package/latticeai/__init__.py
CHANGED
package/latticeai/api/browser.py
CHANGED
|
@@ -28,6 +28,7 @@ from pydantic import BaseModel
|
|
|
28
28
|
|
|
29
29
|
from lattice_brain.ingestion import IngestionItem, capture_quality_verdict
|
|
30
30
|
from latticeai import __version__
|
|
31
|
+
from latticeai.api.workspace_scope import resolve_workspace_scope
|
|
31
32
|
from latticeai.core.quiet import quiet
|
|
32
33
|
|
|
33
34
|
MAX_TAB_BYTES = 4 * 1024 * 1024 # 4 MB per captured tab payload
|
|
@@ -399,20 +400,13 @@ def create_browser_router(
|
|
|
399
400
|
raise HTTPException(status_code=503, detail="Knowledge Graph ingestion is disabled.")
|
|
400
401
|
|
|
401
402
|
def _write_workspace(request: Request, body_workspace: Optional[str], user: str) -> Optional[str]:
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
requested = supplied[0] if supplied else None
|
|
410
|
-
if workspace_service is None:
|
|
411
|
-
return requested
|
|
412
|
-
try:
|
|
413
|
-
return workspace_service.resolve_write_scope(requested, user or None)
|
|
414
|
-
except PermissionError as exc:
|
|
415
|
-
raise HTTPException(status_code=403, detail=str(exc)) from exc
|
|
403
|
+
return resolve_workspace_scope(
|
|
404
|
+
request,
|
|
405
|
+
user=user,
|
|
406
|
+
workspace_service=workspace_service,
|
|
407
|
+
write=True,
|
|
408
|
+
body_workspace=body_workspace,
|
|
409
|
+
)
|
|
416
410
|
|
|
417
411
|
@router.post("/api/browser/read-url")
|
|
418
412
|
async def read_url(req: ReadUrlRequest, request: Request):
|
|
@@ -14,6 +14,10 @@ from pydantic import BaseModel
|
|
|
14
14
|
|
|
15
15
|
from lattice_brain.ingestion import IngestionItem
|
|
16
16
|
from latticeai.api.ui_redirects import app_redirect
|
|
17
|
+
from latticeai.api.workspace_scope import (
|
|
18
|
+
resolve_workspace_scope,
|
|
19
|
+
workspace_scope_from_request,
|
|
20
|
+
)
|
|
17
21
|
|
|
18
22
|
|
|
19
23
|
class KnowledgeGraphIngestRequest(BaseModel):
|
|
@@ -43,12 +47,9 @@ class PromotionActionRequest(BaseModel):
|
|
|
43
47
|
ids: Optional[List[str]] = None
|
|
44
48
|
|
|
45
49
|
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
return header.strip()
|
|
50
|
-
query = request.query_params.get("workspace_id")
|
|
51
|
-
return query.strip() if query and query.strip() else None
|
|
50
|
+
# Kept as a module-level name because callers import it from here; the
|
|
51
|
+
# implementation now lives in the shared resolver.
|
|
52
|
+
_workspace_scope_from_request = workspace_scope_from_request
|
|
52
53
|
|
|
53
54
|
|
|
54
55
|
def _format_context(matches: list, limit: int) -> str:
|
|
@@ -127,13 +128,44 @@ def create_knowledge_graph_router(
|
|
|
127
128
|
]
|
|
128
129
|
|
|
129
130
|
def _write_workspace(request: Request, user: str) -> Optional[str]:
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
131
|
+
return resolve_workspace_scope(
|
|
132
|
+
request,
|
|
133
|
+
user=user,
|
|
134
|
+
workspace_service=workspace_service,
|
|
135
|
+
write=True,
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
def _scoped_stats(request: Request) -> Dict[str, Any]:
|
|
139
|
+
"""Store statistics restricted to what the caller may read.
|
|
140
|
+
|
|
141
|
+
``stats()`` counted every row in the database, so a member of one
|
|
142
|
+
organization workspace could read another's node/edge/document volume
|
|
143
|
+
off a "harmless" metrics endpoint. Unscoped mode (single-user / no
|
|
144
|
+
auth) still gets the whole-store counts, which is the same number it
|
|
145
|
+
always was.
|
|
146
|
+
"""
|
|
147
|
+
kg, allowed = _scoped(request)
|
|
148
|
+
if allowed is None:
|
|
149
|
+
return dict(kg.stats())
|
|
133
150
|
try:
|
|
134
|
-
return
|
|
135
|
-
|
|
136
|
-
|
|
151
|
+
return dict(
|
|
152
|
+
kg.stats(allowed_workspaces=allowed, include_legacy_global=False)
|
|
153
|
+
)
|
|
154
|
+
except TypeError:
|
|
155
|
+
# A store predating scoped stats cannot answer the scoped
|
|
156
|
+
# question. Keep the response shape and empty the aggregates it
|
|
157
|
+
# could not restrict, rather than leaking whole-store totals.
|
|
158
|
+
payload = dict(kg.stats())
|
|
159
|
+
empty: Dict[str, Any] = {
|
|
160
|
+
"nodes": {},
|
|
161
|
+
"edges": {},
|
|
162
|
+
"local_sources": 0,
|
|
163
|
+
"local_file_status": {},
|
|
164
|
+
}
|
|
165
|
+
payload.update(
|
|
166
|
+
{key: value for key, value in empty.items() if key in payload}
|
|
167
|
+
)
|
|
168
|
+
return payload
|
|
137
169
|
|
|
138
170
|
@router.get("/graph")
|
|
139
171
|
async def knowledge_graph_page(request: Request):
|
|
@@ -214,13 +246,11 @@ def create_knowledge_graph_router(
|
|
|
214
246
|
|
|
215
247
|
@router.get("/knowledge-graph/stats")
|
|
216
248
|
async def knowledge_graph_stats(request: Request):
|
|
217
|
-
|
|
218
|
-
return graph().stats()
|
|
249
|
+
return _scoped_stats(request)
|
|
219
250
|
|
|
220
251
|
@router.get("/knowledge-graph/schema")
|
|
221
252
|
async def knowledge_graph_schema(request: Request):
|
|
222
|
-
|
|
223
|
-
stats = graph().stats()
|
|
253
|
+
stats = _scoped_stats(request)
|
|
224
254
|
return {
|
|
225
255
|
"legacy_schema_version": stats.get("schema_version"),
|
|
226
256
|
"v2_schema_available": stats.get("v2_schema_available"),
|
|
@@ -21,6 +21,7 @@ from fastapi import APIRouter, HTTPException, Request
|
|
|
21
21
|
from pydantic import BaseModel
|
|
22
22
|
|
|
23
23
|
from latticeai.api.ui_redirects import app_redirect
|
|
24
|
+
from latticeai.api.workspace_scope import workspace_scope_from_request
|
|
24
25
|
from latticeai.services.app_context import AppContext
|
|
25
26
|
|
|
26
27
|
# ── Request models (workspace-only; moved verbatim from server_app) ──────────
|
|
@@ -142,17 +143,9 @@ class WorkspaceActivateRequest(BaseModel):
|
|
|
142
143
|
workspace_id: str
|
|
143
144
|
|
|
144
145
|
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
``None`` lets the service fall back to the active workspace (Personal by
|
|
149
|
-
default), preserving pre-1.1 behaviour for clients that send no header.
|
|
150
|
-
"""
|
|
151
|
-
header = request.headers.get("X-Workspace-Id")
|
|
152
|
-
if header and header.strip():
|
|
153
|
-
return header.strip()
|
|
154
|
-
query = request.query_params.get("workspace_id")
|
|
155
|
-
return query.strip() if query and query.strip() else None
|
|
146
|
+
# Historical name: ``app_factory`` re-exports it as part of the legacy
|
|
147
|
+
# ``server_app`` surface. The implementation is the shared resolver.
|
|
148
|
+
_workspace_scope_from_request = workspace_scope_from_request
|
|
156
149
|
|
|
157
150
|
|
|
158
151
|
def create_workspace_router(context: AppContext) -> APIRouter:
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
"""One answer to "which workspace is this request talking about?".
|
|
2
|
+
|
|
3
|
+
Eight routers used to each re-derive this from raw headers. They did not agree:
|
|
4
|
+
some accepted the ``workspace_id`` query parameter and some only the
|
|
5
|
+
``X-Workspace-Id`` header, and only three of them checked that a body's
|
|
6
|
+
``workspace_id`` matched the header instead of silently letting one win. A
|
|
7
|
+
guard that exists in some handlers and not others is not a guard — this module
|
|
8
|
+
is the single implementation they all call, so the rule is stated once and
|
|
9
|
+
holds everywhere.
|
|
10
|
+
|
|
11
|
+
The rule:
|
|
12
|
+
|
|
13
|
+
* A caller may name a workspace in the ``X-Workspace-Id`` header, the
|
|
14
|
+
``workspace_id`` query parameter, or the request body.
|
|
15
|
+
* If more than one of those is present they must **agree**. Disagreement is a
|
|
16
|
+
``403``, never a silent preference — a request that names two workspaces has
|
|
17
|
+
no single meaning, and picking one is how a scoped write lands in the wrong
|
|
18
|
+
vault.
|
|
19
|
+
* Naming nothing resolves to ``None``, which every caller passes to
|
|
20
|
+
:class:`~latticeai.services.workspace_service.WorkspaceService`, where it
|
|
21
|
+
falls back to the active workspace (Personal by default). That is the
|
|
22
|
+
pre-1.1 single-workspace behaviour and it stays intact.
|
|
23
|
+
|
|
24
|
+
Permission is *not* decided here. ``WorkspaceService`` owns read/write gating;
|
|
25
|
+
this module only decides what was asked for and translates the service's
|
|
26
|
+
``PermissionError`` into the HTTP boundary's ``403``.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
from typing import Any, List, Optional
|
|
32
|
+
|
|
33
|
+
from fastapi import HTTPException, Request
|
|
34
|
+
|
|
35
|
+
__all__ = [
|
|
36
|
+
"WORKSPACE_HEADER",
|
|
37
|
+
"WORKSPACE_PARAM",
|
|
38
|
+
"requested_workspace",
|
|
39
|
+
"resolve_workspace_scope",
|
|
40
|
+
"workspace_scope_from_request",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
WORKSPACE_HEADER = "X-Workspace-Id"
|
|
44
|
+
WORKSPACE_PARAM = "workspace_id"
|
|
45
|
+
|
|
46
|
+
_MISMATCH_DETAIL = "Workspace selectors must match."
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _clean(value: Any) -> Optional[str]:
|
|
50
|
+
"""Normalize one selector to a non-empty string, or ``None``."""
|
|
51
|
+
if value is None:
|
|
52
|
+
return None
|
|
53
|
+
text = str(value).strip()
|
|
54
|
+
return text or None
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def workspace_scope_from_request(request: Request) -> Optional[str]:
|
|
58
|
+
"""Resolve the workspace named by header/query, or ``None``.
|
|
59
|
+
|
|
60
|
+
``None`` lets the service fall back to the active workspace (Personal by
|
|
61
|
+
default), preserving pre-1.1 behaviour for clients that send no header.
|
|
62
|
+
"""
|
|
63
|
+
header = _clean(request.headers.get(WORKSPACE_HEADER))
|
|
64
|
+
if header:
|
|
65
|
+
return header
|
|
66
|
+
return _clean(request.query_params.get(WORKSPACE_PARAM))
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def requested_workspace(
|
|
70
|
+
request: Request,
|
|
71
|
+
*,
|
|
72
|
+
body_workspace: Any = None,
|
|
73
|
+
) -> Optional[str]:
|
|
74
|
+
"""The one workspace this request names, or ``None``.
|
|
75
|
+
|
|
76
|
+
Raises ``403`` when the header, query parameter, and body disagree.
|
|
77
|
+
"""
|
|
78
|
+
selectors: List[str] = []
|
|
79
|
+
for value in (
|
|
80
|
+
_clean(body_workspace),
|
|
81
|
+
_clean(request.headers.get(WORKSPACE_HEADER)),
|
|
82
|
+
_clean(request.query_params.get(WORKSPACE_PARAM)),
|
|
83
|
+
):
|
|
84
|
+
if value is not None:
|
|
85
|
+
selectors.append(value)
|
|
86
|
+
if len(set(selectors)) > 1:
|
|
87
|
+
raise HTTPException(status_code=403, detail=_MISMATCH_DETAIL)
|
|
88
|
+
return selectors[0] if selectors else None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def resolve_workspace_scope(
|
|
92
|
+
request: Request,
|
|
93
|
+
*,
|
|
94
|
+
user: Optional[str],
|
|
95
|
+
workspace_service: Any = None,
|
|
96
|
+
write: bool = True,
|
|
97
|
+
body_workspace: Any = None,
|
|
98
|
+
allow_unscoped_anonymous: bool = False,
|
|
99
|
+
) -> Optional[str]:
|
|
100
|
+
"""Resolve *and authorize* the workspace a handler should act on.
|
|
101
|
+
|
|
102
|
+
``workspace_service=None`` is the standalone/embedded router contract: the
|
|
103
|
+
named workspace passes through ungated, exactly as each local copy of this
|
|
104
|
+
logic did. With a service present, reads are gated on ``read`` and writes
|
|
105
|
+
on ``write``, and a denial surfaces as ``403``.
|
|
106
|
+
|
|
107
|
+
``allow_unscoped_anonymous`` preserves one deliberate exception: a no-auth
|
|
108
|
+
local caller that names no workspace keeps its legacy *unscoped* records
|
|
109
|
+
instead of being resolved onto the active workspace.
|
|
110
|
+
"""
|
|
111
|
+
requested = requested_workspace(request, body_workspace=body_workspace)
|
|
112
|
+
if workspace_service is None:
|
|
113
|
+
return requested
|
|
114
|
+
if allow_unscoped_anonymous and not user and requested is None:
|
|
115
|
+
return None
|
|
116
|
+
resolver = (
|
|
117
|
+
workspace_service.resolve_write_scope
|
|
118
|
+
if write
|
|
119
|
+
else workspace_service.resolve_read_scope
|
|
120
|
+
)
|
|
121
|
+
try:
|
|
122
|
+
scope: Optional[str] = resolver(requested, user or None)
|
|
123
|
+
except PermissionError as exc:
|
|
124
|
+
raise HTTPException(status_code=403, detail=str(exc)) from exc
|
|
125
|
+
return scope
|