answer42 0.3.2__tar.gz → 0.3.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {answer42-0.3.2 → answer42-0.3.3}/PKG-INFO +1 -1
- {answer42-0.3.2 → answer42-0.3.3}/pyproject.toml +1 -1
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/rag/parsers.py +48 -12
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/rag/service.py +100 -4
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/rag/store.py +22 -2
- {answer42-0.3.2 → answer42-0.3.3}/.gitignore +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/LICENSE +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/README.md +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/credentials.example.json +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/docs/agent-installation.md +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/docs/architecture.md +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/docs/assets/answer42-logo.png +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/docs/assets/platform42-logo.svg +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/docs/installation.md +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/scripts/build_cf.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/scripts/build_pages.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/scripts/e2e_stable.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/scripts/openclaw_mcp_autoreload.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/scripts/rag_cli.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/ConfigDumpInfo.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/Configuration.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/DataProcessors/MCPTestManager/Ext/ObjectModule.bsl +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260/Ext/Form/Module.bsl" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260/Ext/Form.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/DataProcessors/MCPTestManager.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/Ext/ManagedApplicationModule.bsl +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/cf/Languages//320/240/321/203/321/201/321/201/320/272/320/270/320/271.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Catalogs//320/237/320/241_/320/222/320/273/320/260/320/264/320/265/320/273/320/265/321/206.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Catalogs//320/237/320/241_/320/241/320/277/321/200/320/260/320/262/320/276/321/207/320/275/320/270/320/2721.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/ConfigDumpInfo.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Configuration.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Languages//320/240/321/203/321/201/321/201/320/272/320/270/320/271.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Ext/ManagerModule.bsl" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Ext/ObjectModule.bsl" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Templates//320/236/321/201/320/275/320/276/320/262/320/275/320/260/321/217/320/241/321/205/320/265/320/274/320/260/320/232/320/276/320/274/320/277/320/276/320/275/320/276/320/262/320/272/320/270/320/224/320/260/320/275/320/275/321/213/321/205/Ext/Template.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Templates//320/236/321/201/320/275/320/276/320/262/320/275/320/260/321/217/320/241/321/205/320/265/320/274/320/260/320/232/320/276/320/274/320/277/320/276/320/275/320/276/320/262/320/272/320/270/320/224/320/260/320/275/320/275/321/213/321/205.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/__init__.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/assets/MCPTestClient.cf +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/assets/MCPTestManager.cf +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/assets/__init__.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/assets/skills/answer42/SKILL.md +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/assets/skills/answer42-rag/SKILL.md +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/bridge.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/credentials.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/os_support.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/platform.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/protocol.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/rag/__init__.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/rag/detect.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/rag/dump.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/rag/model.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/recorder.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/release_helper.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/runtime.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/server.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/skill_installer.py +0 -0
- {answer42-0.3.2 → answer42-0.3.3}/src/mcp_1c/window_control.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: answer42
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.3
|
|
4
4
|
Summary: Answer42 — The Answer to Life, Universe, and 1C — UI Driver. MCP-powered 1C:Enterprise UI automation: click, fill, navigate, test, and introspect managed forms through the test-client API
|
|
5
5
|
Author: Marvin (AI Assistant), 42Clouds, and contributors
|
|
6
6
|
Author-email: "Kosolapov Stanislav (proDOOMman)" <prodoomman@gmail.com>
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "answer42"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.3"
|
|
4
4
|
description = "Answer42 — The Answer to Life, Universe, and 1C — UI Driver. MCP-powered 1C:Enterprise UI automation: click, fill, navigate, test, and introspect managed forms through the test-client API"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11"
|
|
@@ -341,17 +341,20 @@ def _extract_embedded_data_composition_fields(parsed: ParsedSource, root: ET.Ele
|
|
|
341
341
|
|
|
342
342
|
|
|
343
343
|
def _extract_help_pages(parsed: ParsedSource, object_dir: Path, full: str) -> None:
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
for
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
344
|
+
# EDT exports usually keep help as <Object>/Help/*.html, while Designer
|
|
345
|
+
# hierarchical dumps keep it under <Object>/Ext/Help/*.html with
|
|
346
|
+
# <Object>/Ext/Help.xml next to the language-specific pages.
|
|
347
|
+
for help_dir in (object_dir / "Help", object_dir / "Ext" / "Help"):
|
|
348
|
+
if not help_dir.exists():
|
|
349
|
+
continue
|
|
350
|
+
for page in sorted(help_dir.glob("*.html")):
|
|
351
|
+
raw = page.read_text(encoding="utf-8-sig", errors="ignore")
|
|
352
|
+
title_match = re.search(r"<title[^>]*>(.*?)</title>", raw, flags=re.IGNORECASE | re.DOTALL)
|
|
353
|
+
h1_match = re.search(r"<h1[^>]*>(.*?)</h1>", raw, flags=re.IGNORECASE | re.DOTALL)
|
|
354
|
+
title = _html_to_text((title_match or h1_match).group(1)) if (title_match or h1_match) else None
|
|
355
|
+
text = _html_to_text(raw)
|
|
356
|
+
if text:
|
|
357
|
+
parsed.help_pages.append(HelpPage(full, page.stem, title, text, str(page), {"file": page.name}))
|
|
355
358
|
|
|
356
359
|
|
|
357
360
|
def _html_to_text(raw: str) -> str:
|
|
@@ -410,5 +413,38 @@ def navigation_for_object(full_name: str, kind: str) -> tuple[str, str] | None:
|
|
|
410
413
|
return None
|
|
411
414
|
|
|
412
415
|
|
|
416
|
+
def split_1c_identifier(text: str | None) -> str:
|
|
417
|
+
"""Return a human-readable variant of 1C metadata identifiers.
|
|
418
|
+
|
|
419
|
+
1C metadata names are commonly written as CamelCase identifiers, e.g.
|
|
420
|
+
``ОборотноСальдоваяВедомостьПоСчету``. FTS tokenizers do not split those
|
|
421
|
+
words automatically, so business queries like ``оборотно сальдовая`` miss
|
|
422
|
+
chunks that contain only the technical identifier. Keep this helper small
|
|
423
|
+
and deterministic: it does not add domain synonyms, it only exposes words
|
|
424
|
+
already present in the metadata name.
|
|
425
|
+
"""
|
|
426
|
+
if not text:
|
|
427
|
+
return ""
|
|
428
|
+
value = re.sub(r"[._/\\:-]+", " ", text)
|
|
429
|
+
value = re.sub(r"(?<=[а-яёa-z0-9])(?=[А-ЯЁA-Z])", " ", value)
|
|
430
|
+
return re.sub(r"\s+", " ", value).strip()
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def search_text_variants(*values: str | None) -> str:
|
|
434
|
+
parts: list[str] = []
|
|
435
|
+
seen: set[str] = set()
|
|
436
|
+
for value in values:
|
|
437
|
+
if not value:
|
|
438
|
+
continue
|
|
439
|
+
for part in (value, split_1c_identifier(value)):
|
|
440
|
+
normalized = normalize_query(part)
|
|
441
|
+
if normalized and normalized not in seen:
|
|
442
|
+
seen.add(normalized)
|
|
443
|
+
parts.append(part)
|
|
444
|
+
return " ".join(parts)
|
|
445
|
+
|
|
446
|
+
|
|
413
447
|
def normalize_query(text: str) -> str:
|
|
414
|
-
|
|
448
|
+
text = text.replace("ё", "е")
|
|
449
|
+
text = re.sub(r"[\u2010-\u2015–—−-]+", " ", text)
|
|
450
|
+
return re.sub(r"\s+", " ", text.strip().lower())
|
|
@@ -1,16 +1,22 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import math
|
|
3
6
|
import os
|
|
7
|
+
import re
|
|
4
8
|
import subprocess
|
|
5
9
|
from pathlib import Path
|
|
6
10
|
from typing import Any
|
|
7
11
|
|
|
8
12
|
from .detect import detect_source_format, discover_source_roots
|
|
9
13
|
from .model import SourceInfo
|
|
10
|
-
from .parsers import navigation_for_object, normalize_query, parse_source
|
|
14
|
+
from .parsers import navigation_for_object, normalize_query, parse_source, search_text_variants
|
|
11
15
|
from .store import RagStore, json_dumps, row_to_dict
|
|
12
16
|
|
|
13
17
|
DEFAULT_RAG_DB = Path(os.environ.get("MCP_1C_RAG_DB", "build/rag/onec-rag.sqlite"))
|
|
18
|
+
EMBEDDING_MODEL = "answer42-local-hash-v1"
|
|
19
|
+
EMBEDDING_DIM = 256
|
|
14
20
|
|
|
15
21
|
|
|
16
22
|
class RagService:
|
|
@@ -246,9 +252,11 @@ class RagService:
|
|
|
246
252
|
nav = navigation_for_object(obj.full_name, obj.kind)
|
|
247
253
|
if nav:
|
|
248
254
|
con.execute("INSERT INTO navigation_links(object_id, kind, url) VALUES(?,?,?)", (oid, nav[0], nav[1]))
|
|
249
|
-
|
|
255
|
+
object_terms = search_text_variants(obj.full_name, obj.name, obj.synonym)
|
|
256
|
+
chunk_text = f"{obj.full_name} {object_terms} {obj.synonym or ''} {obj.kind} navigation {nav[1] if nav else ''}"
|
|
250
257
|
cid = _insert_chunk(con, snapshot_id, oid, "object", chunk_text, obj.source_path, {"source": src["name"]})
|
|
251
258
|
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, obj.full_name, obj.kind, obj.source_path, cid))
|
|
259
|
+
_insert_embedding(con, cid, chunk_text)
|
|
252
260
|
counts["chunks"] += 1
|
|
253
261
|
if obj.synonym:
|
|
254
262
|
_insert_business_term(con, snapshot_id, obj.synonym, oid, 0.7, "object synonym")
|
|
@@ -295,15 +303,23 @@ class RagService:
|
|
|
295
303
|
chunk_text = " ".join(part for part in text_parts if part)
|
|
296
304
|
cid = _insert_chunk(con, snapshot_id, oid, "dcs_field", chunk_text, dcs_field.source_path, {"source": src["name"], "schema": dcs_field.schema_name})
|
|
297
305
|
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, dcs_field.object_full_name, "СКД", dcs_field.source_path, cid))
|
|
306
|
+
_insert_embedding(con, cid, chunk_text)
|
|
298
307
|
counts["chunks"] += 1
|
|
299
308
|
for help_page in parsed.help_pages:
|
|
300
309
|
oid = object_ids.get(help_page.object_full_name)
|
|
301
310
|
if not oid:
|
|
302
311
|
continue
|
|
303
|
-
|
|
312
|
+
object_row = con.execute("SELECT name, synonym FROM metadata_objects WHERE id=?", (oid,)).fetchone()
|
|
313
|
+
object_terms = search_text_variants(
|
|
314
|
+
help_page.object_full_name,
|
|
315
|
+
object_row["name"] if object_row else None,
|
|
316
|
+
object_row["synonym"] if object_row else None,
|
|
317
|
+
)
|
|
318
|
+
text_parts = [help_page.object_full_name, object_terms, "справка", help_page.language, help_page.title or "", help_page.text or ""]
|
|
304
319
|
chunk_text = "\n".join(part for part in text_parts if part)
|
|
305
320
|
cid = _insert_chunk(con, snapshot_id, oid, "help", chunk_text, help_page.source_path, {"source": src["name"], "language": help_page.language, "title": help_page.title})
|
|
306
321
|
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, help_page.object_full_name, "Справка", help_page.source_path, cid))
|
|
322
|
+
_insert_embedding(con, cid, chunk_text)
|
|
307
323
|
counts["help_pages"] += 1
|
|
308
324
|
counts["chunks"] += 1
|
|
309
325
|
return {"db_path": self.db_path, **counts}
|
|
@@ -497,7 +513,45 @@ class RagService:
|
|
|
497
513
|
seen.add(key)
|
|
498
514
|
if len(fts_rows) >= limit:
|
|
499
515
|
break
|
|
500
|
-
|
|
516
|
+
semantic_rows = self._semantic_query(con, q, snapshot_id, limit)
|
|
517
|
+
return {"query": text, "snapshot": snapshot, "terms": term_rows, "fts": fts_rows, "semantic": semantic_rows}
|
|
518
|
+
|
|
519
|
+
def _semantic_query(self, con, query: str, snapshot_id: int | None, limit: int) -> list[dict[str, Any]]:
|
|
520
|
+
query_vector = _embedding_vector(query)
|
|
521
|
+
params: list[Any] = [EMBEDDING_MODEL]
|
|
522
|
+
snapshot_filter = ""
|
|
523
|
+
if snapshot_id is not None:
|
|
524
|
+
snapshot_filter = "AND rc.snapshot_id=?"
|
|
525
|
+
params.append(snapshot_id)
|
|
526
|
+
rows = con.execute(
|
|
527
|
+
f"""
|
|
528
|
+
SELECT rc.text, rc.source_path, mo.full_name AS object_full_name, mo.kind,
|
|
529
|
+
re.vector_json
|
|
530
|
+
FROM rag_embeddings re
|
|
531
|
+
JOIN rag_chunks rc ON rc.id=re.chunk_id
|
|
532
|
+
LEFT JOIN metadata_objects mo ON mo.id=rc.object_id
|
|
533
|
+
WHERE re.model=? {snapshot_filter}
|
|
534
|
+
""",
|
|
535
|
+
params,
|
|
536
|
+
)
|
|
537
|
+
scored: list[dict[str, Any]] = []
|
|
538
|
+
for row in rows:
|
|
539
|
+
vector = json.loads(row["vector_json"])
|
|
540
|
+
score = _cosine(query_vector, vector)
|
|
541
|
+
if score <= 0:
|
|
542
|
+
continue
|
|
543
|
+
scored.append(
|
|
544
|
+
{
|
|
545
|
+
"text": row["text"],
|
|
546
|
+
"object_full_name": row["object_full_name"],
|
|
547
|
+
"kind": row["kind"],
|
|
548
|
+
"source_path": row["source_path"],
|
|
549
|
+
"score": score,
|
|
550
|
+
"model": EMBEDDING_MODEL,
|
|
551
|
+
}
|
|
552
|
+
)
|
|
553
|
+
scored.sort(key=lambda item: item["score"], reverse=True)
|
|
554
|
+
return scored[:limit]
|
|
501
555
|
|
|
502
556
|
|
|
503
557
|
def _git_commit(path: Path) -> str | None:
|
|
@@ -530,5 +584,47 @@ def _insert_chunk(con, snapshot_id, object_id, chunk_type, text, source_path, me
|
|
|
530
584
|
return cur.lastrowid
|
|
531
585
|
|
|
532
586
|
|
|
587
|
+
def _insert_embedding(con, chunk_id: int, text: str) -> None:
|
|
588
|
+
con.execute(
|
|
589
|
+
"""
|
|
590
|
+
INSERT INTO rag_embeddings(chunk_id,model,dim,vector_json,text_hash)
|
|
591
|
+
VALUES(?,?,?,?,?)
|
|
592
|
+
""",
|
|
593
|
+
(
|
|
594
|
+
chunk_id,
|
|
595
|
+
EMBEDDING_MODEL,
|
|
596
|
+
EMBEDDING_DIM,
|
|
597
|
+
json.dumps(_embedding_vector(text), separators=(",", ":")),
|
|
598
|
+
hashlib.sha256(text.encode("utf-8", errors="ignore")).hexdigest(),
|
|
599
|
+
),
|
|
600
|
+
)
|
|
601
|
+
|
|
602
|
+
|
|
603
|
+
def _embedding_vector(text: str) -> list[float]:
|
|
604
|
+
vector = [0.0] * EMBEDDING_DIM
|
|
605
|
+
normalized = normalize_query(text)
|
|
606
|
+
tokens = re.findall(r"[\w]+", normalized, flags=re.UNICODE)
|
|
607
|
+
features: list[str] = []
|
|
608
|
+
for token in tokens:
|
|
609
|
+
if len(token) <= 2:
|
|
610
|
+
features.append("tok:" + token)
|
|
611
|
+
continue
|
|
612
|
+
features.append("tok:" + token)
|
|
613
|
+
for size in (3, 4):
|
|
614
|
+
if len(token) >= size:
|
|
615
|
+
features.extend(f"ng{size}:" + token[pos : pos + size] for pos in range(len(token) - size + 1))
|
|
616
|
+
for feature in features:
|
|
617
|
+
digest = hashlib.blake2b(feature.encode("utf-8"), digest_size=8).digest()
|
|
618
|
+
bucket = int.from_bytes(digest[:4], "little") % EMBEDDING_DIM
|
|
619
|
+
sign = 1.0 if digest[4] & 1 else -1.0
|
|
620
|
+
vector[bucket] += sign
|
|
621
|
+
norm = math.sqrt(sum(value * value for value in vector)) or 1.0
|
|
622
|
+
return [round(value / norm, 6) for value in vector]
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def _cosine(left: list[float], right: list[float]) -> float:
|
|
626
|
+
return sum(a * b for a, b in zip(left, right, strict=False))
|
|
627
|
+
|
|
628
|
+
|
|
533
629
|
def _insert_business_term(con, snapshot_id, term, object_id, confidence, evidence) -> None:
|
|
534
630
|
con.execute("INSERT INTO business_terms(snapshot_id,term,normalized_term,object_id,confidence,evidence) VALUES(?,?,?,?,?,?)", (snapshot_id, term, normalize_query(term), object_id, confidence, evidence))
|
|
@@ -5,7 +5,7 @@ import sqlite3
|
|
|
5
5
|
from pathlib import Path
|
|
6
6
|
from typing import Any
|
|
7
7
|
|
|
8
|
-
SCHEMA_VERSION =
|
|
8
|
+
SCHEMA_VERSION = 4
|
|
9
9
|
|
|
10
10
|
|
|
11
11
|
class RagStore:
|
|
@@ -25,7 +25,7 @@ class RagStore:
|
|
|
25
25
|
"""
|
|
26
26
|
CREATE TABLE IF NOT EXISTS schema_info(version INTEGER NOT NULL);
|
|
27
27
|
DELETE FROM schema_info;
|
|
28
|
-
INSERT INTO schema_info(version) VALUES (
|
|
28
|
+
INSERT INTO schema_info(version) VALUES (4);
|
|
29
29
|
|
|
30
30
|
CREATE TABLE IF NOT EXISTS sources(
|
|
31
31
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
@@ -195,6 +195,15 @@ class RagStore:
|
|
|
195
195
|
chunk_id UNINDEXED
|
|
196
196
|
);
|
|
197
197
|
|
|
198
|
+
CREATE TABLE IF NOT EXISTS rag_embeddings(
|
|
199
|
+
chunk_id INTEGER PRIMARY KEY,
|
|
200
|
+
model TEXT NOT NULL,
|
|
201
|
+
dim INTEGER NOT NULL,
|
|
202
|
+
vector_json TEXT NOT NULL,
|
|
203
|
+
text_hash TEXT NOT NULL,
|
|
204
|
+
FOREIGN KEY(chunk_id) REFERENCES rag_chunks(id) ON DELETE CASCADE
|
|
205
|
+
);
|
|
206
|
+
|
|
198
207
|
CREATE INDEX IF NOT EXISTS idx_objects_source_name ON metadata_objects(source_id, full_name);
|
|
199
208
|
CREATE INDEX IF NOT EXISTS idx_objects_kind ON metadata_objects(kind);
|
|
200
209
|
CREATE INDEX IF NOT EXISTS idx_attrs_object ON attributes(object_id);
|
|
@@ -210,6 +219,17 @@ class RagStore:
|
|
|
210
219
|
con.execute("ALTER TABLE attributes ADD COLUMN role TEXT NOT NULL DEFAULT 'attribute'")
|
|
211
220
|
|
|
212
221
|
def clear_index_for_source(self, con: sqlite3.Connection, source_id: int) -> None:
|
|
222
|
+
con.execute(
|
|
223
|
+
"""
|
|
224
|
+
DELETE FROM rag_embeddings
|
|
225
|
+
WHERE chunk_id IN (
|
|
226
|
+
SELECT rc.id
|
|
227
|
+
FROM rag_chunks rc JOIN metadata_objects mo ON mo.id=rc.object_id
|
|
228
|
+
WHERE mo.source_id=?
|
|
229
|
+
)
|
|
230
|
+
""",
|
|
231
|
+
(source_id,),
|
|
232
|
+
)
|
|
213
233
|
con.execute(
|
|
214
234
|
"""
|
|
215
235
|
DELETE FROM search_fts
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|