answer42 0.3.2__tar.gz → 0.3.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {answer42-0.3.2 → answer42-0.3.4}/PKG-INFO +1 -1
- {answer42-0.3.2 → answer42-0.3.4}/pyproject.toml +1 -1
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/MCPTestClient.cf +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/MCPTestManager.cf +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/parsers.py +100 -20
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/service.py +119 -4
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/store.py +22 -2
- {answer42-0.3.2 → answer42-0.3.4}/.gitignore +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/LICENSE +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/README.md +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/credentials.example.json +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/docs/agent-installation.md +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/docs/architecture.md +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/docs/assets/answer42-logo.png +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/docs/assets/platform42-logo.svg +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/docs/installation.md +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/scripts/build_cf.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/scripts/build_pages.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/scripts/e2e_stable.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/scripts/openclaw_mcp_autoreload.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/scripts/rag_cli.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/ConfigDumpInfo.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/Configuration.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Ext/ObjectModule.bsl +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260/Ext/Form/Module.bsl" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260/Ext/Form.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/Ext/ManagedApplicationModule.bsl +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/cf/Languages//320/240/321/203/321/201/321/201/320/272/320/270/320/271.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Catalogs//320/237/320/241_/320/222/320/273/320/260/320/264/320/265/320/273/320/265/321/206.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Catalogs//320/237/320/241_/320/241/320/277/321/200/320/260/320/262/320/276/321/207/320/275/320/270/320/2721.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/ConfigDumpInfo.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Configuration.xml +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Languages//320/240/321/203/321/201/321/201/320/272/320/270/320/271.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Ext/ManagerModule.bsl" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Ext/ObjectModule.bsl" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Templates//320/236/321/201/320/275/320/276/320/262/320/275/320/260/321/217/320/241/321/205/320/265/320/274/320/260/320/232/320/276/320/274/320/277/320/276/320/275/320/276/320/262/320/272/320/270/320/224/320/260/320/275/320/275/321/213/321/205/Ext/Template.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Templates//320/236/321/201/320/275/320/276/320/262/320/275/320/260/321/217/320/241/321/205/320/265/320/274/320/260/320/232/320/276/320/274/320/277/320/276/320/275/320/276/320/262/320/272/320/270/320/224/320/260/320/275/320/275/321/213/321/205.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262.xml" +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/__init__.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/__init__.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/skills/answer42/SKILL.md +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/skills/answer42-rag/SKILL.md +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/bridge.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/credentials.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/os_support.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/platform.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/protocol.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/__init__.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/detect.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/dump.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/model.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/recorder.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/release_helper.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/runtime.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/server.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/skill_installer.py +0 -0
- {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/window_control.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: answer42
|
|
3
|
-
Version: 0.3.
|
|
3
|
+
Version: 0.3.4
|
|
4
4
|
Summary: Answer42 — The Answer to Life, Universe, and 1C — UI Driver. MCP-powered 1C:Enterprise UI automation: click, fill, navigate, test, and introspect managed forms through the test-client API
|
|
5
5
|
Author: Marvin (AI Assistant), 42Clouds, and contributors
|
|
6
6
|
Author-email: "Kosolapov Stanislav (proDOOMman)" <prodoomman@gmail.com>
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "answer42"
|
|
3
|
-
version = "0.3.
|
|
3
|
+
version = "0.3.4"
|
|
4
4
|
description = "Answer42 — The Answer to Life, Universe, and 1C — UI Driver. MCP-powered 1C:Enterprise UI automation: click, fill, navigate, test, and introspect managed forms through the test-client API"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11"
|
|
Binary file
|
|
Binary file
|
|
@@ -21,6 +21,8 @@ from .model import (
|
|
|
21
21
|
)
|
|
22
22
|
|
|
23
23
|
KIND_DIRS = {
|
|
24
|
+
"Constants": "Константа",
|
|
25
|
+
"CommonForms": "ОбщаяФорма",
|
|
24
26
|
"Catalogs": "Справочник",
|
|
25
27
|
"Documents": "Документ",
|
|
26
28
|
"DataProcessors": "Обработка",
|
|
@@ -33,6 +35,8 @@ KIND_DIRS = {
|
|
|
33
35
|
"CalculationRegisters": "РегистрРасчета",
|
|
34
36
|
}
|
|
35
37
|
|
|
38
|
+
FORM_FILE_NAMES = {"Form.form", "Form.xml"}
|
|
39
|
+
|
|
36
40
|
|
|
37
41
|
def _strip_ns(tag: str) -> str:
|
|
38
42
|
return tag.rsplit("}", 1)[-1]
|
|
@@ -199,7 +203,10 @@ def parse_edt_source(source: SourceInfo) -> ParsedSource:
|
|
|
199
203
|
parsed.objects.append(MetadataObject(full, ru_kind, obj_name, synonym, str(mdo), source.format, hierarchical, [o for o in owners if o]))
|
|
200
204
|
if root is not None:
|
|
201
205
|
_extract_metadata_children(parsed, root, full, mdo)
|
|
202
|
-
|
|
206
|
+
if ru_kind == "ОбщаяФорма":
|
|
207
|
+
_extract_common_form(parsed, mdo.parent, full)
|
|
208
|
+
else:
|
|
209
|
+
_extract_forms(parsed, mdo.parent, full)
|
|
203
210
|
_extract_data_composition_schemas(parsed, mdo.parent, full)
|
|
204
211
|
_extract_help_pages(parsed, mdo.parent, full)
|
|
205
212
|
return parsed
|
|
@@ -231,12 +238,26 @@ def parse_designer_xml_source(source: SourceInfo) -> ParsedSource:
|
|
|
231
238
|
if root is not None:
|
|
232
239
|
_extract_metadata_children(parsed, root, full, xml)
|
|
233
240
|
object_dir = xml.parent if xml.parent != base else base / xml.stem
|
|
234
|
-
|
|
241
|
+
if ru_kind == "ОбщаяФорма":
|
|
242
|
+
_extract_common_form(parsed, object_dir, full)
|
|
243
|
+
else:
|
|
244
|
+
_extract_forms(parsed, object_dir, full)
|
|
235
245
|
_extract_data_composition_schemas(parsed, object_dir, full)
|
|
236
246
|
_extract_help_pages(parsed, object_dir, full)
|
|
237
247
|
return parsed
|
|
238
248
|
|
|
239
249
|
|
|
250
|
+
def _extract_common_form(parsed: ParsedSource, object_dir: Path, full: str) -> None:
|
|
251
|
+
form_xml = _find_form_file(object_dir)
|
|
252
|
+
parsed.forms.append(Form(full, "Форма", "CommonForm", str(form_xml) if form_xml.exists() else str(object_dir)))
|
|
253
|
+
if not form_xml.exists():
|
|
254
|
+
return
|
|
255
|
+
root = _xml_root(form_xml)
|
|
256
|
+
if root is not None:
|
|
257
|
+
_extract_form_elements(parsed, root, full, "Форма", form_xml)
|
|
258
|
+
_extract_embedded_data_composition_fields(parsed, root, full, "Форма", form_xml)
|
|
259
|
+
|
|
260
|
+
|
|
240
261
|
def _extract_metadata_children(parsed: ParsedSource, root: ET.Element, full: str, source_path: Path) -> None:
|
|
241
262
|
# Heuristic parser covering both EDT and designer XML structures.
|
|
242
263
|
# Keep object attributes and table-part columns separate: XML dumps often
|
|
@@ -273,9 +294,7 @@ def _extract_forms(parsed: ParsedSource, object_dir: Path, full: str) -> None:
|
|
|
273
294
|
return
|
|
274
295
|
for form_dir in sorted([p for p in forms_dir.iterdir() if p.is_dir()]):
|
|
275
296
|
form_name = form_dir.name
|
|
276
|
-
form_xml = form_dir
|
|
277
|
-
if not form_xml.exists():
|
|
278
|
-
form_xml = form_dir / "Form.xml"
|
|
297
|
+
form_xml = _find_form_file(form_dir)
|
|
279
298
|
parsed.forms.append(Form(full, form_name, None, str(form_xml) if form_xml.exists() else str(form_dir)))
|
|
280
299
|
if form_xml.exists():
|
|
281
300
|
root = _xml_root(form_xml)
|
|
@@ -284,23 +303,34 @@ def _extract_forms(parsed: ParsedSource, object_dir: Path, full: str) -> None:
|
|
|
284
303
|
_extract_embedded_data_composition_fields(parsed, root, full, form_name, form_xml)
|
|
285
304
|
|
|
286
305
|
|
|
306
|
+
def _find_form_file(form_dir: Path) -> Path:
|
|
307
|
+
for candidate_dir in (form_dir / "Ext", form_dir):
|
|
308
|
+
for name in FORM_FILE_NAMES:
|
|
309
|
+
form_file = candidate_dir / name
|
|
310
|
+
if form_file.exists():
|
|
311
|
+
return form_file
|
|
312
|
+
return form_dir / "Ext" / "Form.xml"
|
|
313
|
+
|
|
314
|
+
|
|
287
315
|
def _extract_form_elements(parsed: ParsedSource, root: ET.Element, full: str, form_name: str, source_path: Path) -> None:
|
|
288
316
|
for elem in root.iter():
|
|
289
317
|
tag = _strip_ns(elem.tag)
|
|
290
318
|
low = tag.lower()
|
|
291
|
-
if low not in {"element", "item", "items", "button", "command", "field", "table", "group"}:
|
|
319
|
+
if low not in {"element", "item", "items", "button", "command", "field", "table", "group", "extendedtooltip", "contextmenu"}:
|
|
292
320
|
continue
|
|
293
321
|
name = elem.attrib.get("name") or elem.attrib.get("Name") or _first_text(elem, ["name", "Name"])
|
|
294
322
|
if not name:
|
|
295
323
|
continue
|
|
324
|
+
title = elem.attrib.get("title") or elem.attrib.get("Title") or _first_text(elem, ["title", "Title", "caption", "Caption"])
|
|
325
|
+
data_path = _form_data_path(elem)
|
|
296
326
|
parsed.form_elements.append(
|
|
297
327
|
FormElement(
|
|
298
328
|
full,
|
|
299
329
|
form_name,
|
|
300
330
|
name,
|
|
301
331
|
tag,
|
|
302
|
-
|
|
303
|
-
|
|
332
|
+
title,
|
|
333
|
+
data_path,
|
|
304
334
|
elem.attrib.get("commandName") or _first_text(elem, ["commandName", "CommandName"]),
|
|
305
335
|
None,
|
|
306
336
|
str(source_path),
|
|
@@ -308,6 +338,20 @@ def _extract_form_elements(parsed: ParsedSource, root: ET.Element, full: str, fo
|
|
|
308
338
|
)
|
|
309
339
|
|
|
310
340
|
|
|
341
|
+
def _form_data_path(elem: ET.Element) -> str | None:
|
|
342
|
+
direct = elem.attrib.get("dataPath") or elem.attrib.get("path") or _direct_text(elem, ["dataPath", "path", "DataPath"])
|
|
343
|
+
if direct:
|
|
344
|
+
return direct
|
|
345
|
+
for child in list(elem):
|
|
346
|
+
if _strip_ns(child.tag).lower() == "datapath":
|
|
347
|
+
segments = [_text(segment) for segment in child.iter() if _strip_ns(segment.tag).lower() in {"segments", "segment"}]
|
|
348
|
+
values = [value for value in segments if value]
|
|
349
|
+
if values:
|
|
350
|
+
return ".".join(values)
|
|
351
|
+
return _text(child)
|
|
352
|
+
return None
|
|
353
|
+
|
|
354
|
+
|
|
311
355
|
def _extract_embedded_data_composition_fields(parsed: ParsedSource, root: ET.Element, full: str, form_name: str, source_path: Path) -> None:
|
|
312
356
|
dataset_name: str | None = None
|
|
313
357
|
for elem in root.iter():
|
|
@@ -341,17 +385,20 @@ def _extract_embedded_data_composition_fields(parsed: ParsedSource, root: ET.Ele
|
|
|
341
385
|
|
|
342
386
|
|
|
343
387
|
def _extract_help_pages(parsed: ParsedSource, object_dir: Path, full: str) -> None:
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
for
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
388
|
+
# EDT exports usually keep help as <Object>/Help/*.html, while Designer
|
|
389
|
+
# hierarchical dumps keep it under <Object>/Ext/Help/*.html with
|
|
390
|
+
# <Object>/Ext/Help.xml next to the language-specific pages.
|
|
391
|
+
for help_dir in (object_dir / "Help", object_dir / "Ext" / "Help"):
|
|
392
|
+
if not help_dir.exists():
|
|
393
|
+
continue
|
|
394
|
+
for page in sorted(help_dir.glob("*.html")):
|
|
395
|
+
raw = page.read_text(encoding="utf-8-sig", errors="ignore")
|
|
396
|
+
title_match = re.search(r"<title[^>]*>(.*?)</title>", raw, flags=re.IGNORECASE | re.DOTALL)
|
|
397
|
+
h1_match = re.search(r"<h1[^>]*>(.*?)</h1>", raw, flags=re.IGNORECASE | re.DOTALL)
|
|
398
|
+
title = _html_to_text((title_match or h1_match).group(1)) if (title_match or h1_match) else None
|
|
399
|
+
text = _html_to_text(raw)
|
|
400
|
+
if text:
|
|
401
|
+
parsed.help_pages.append(HelpPage(full, page.stem, title, text, str(page), {"file": page.name}))
|
|
355
402
|
|
|
356
403
|
|
|
357
404
|
def _html_to_text(raw: str) -> str:
|
|
@@ -410,5 +457,38 @@ def navigation_for_object(full_name: str, kind: str) -> tuple[str, str] | None:
|
|
|
410
457
|
return None
|
|
411
458
|
|
|
412
459
|
|
|
460
|
+
def split_1c_identifier(text: str | None) -> str:
|
|
461
|
+
"""Return a human-readable variant of 1C metadata identifiers.
|
|
462
|
+
|
|
463
|
+
1C metadata names are commonly written as CamelCase identifiers, e.g.
|
|
464
|
+
``ОборотноСальдоваяВедомостьПоСчету``. FTS tokenizers do not split those
|
|
465
|
+
words automatically, so business queries like ``оборотно сальдовая`` miss
|
|
466
|
+
chunks that contain only the technical identifier. Keep this helper small
|
|
467
|
+
and deterministic: it does not add domain synonyms, it only exposes words
|
|
468
|
+
already present in the metadata name.
|
|
469
|
+
"""
|
|
470
|
+
if not text:
|
|
471
|
+
return ""
|
|
472
|
+
value = re.sub(r"[._/\\:-]+", " ", text)
|
|
473
|
+
value = re.sub(r"(?<=[а-яёa-z0-9])(?=[А-ЯЁA-Z])", " ", value)
|
|
474
|
+
return re.sub(r"\s+", " ", value).strip()
|
|
475
|
+
|
|
476
|
+
|
|
477
|
+
def search_text_variants(*values: str | None) -> str:
|
|
478
|
+
parts: list[str] = []
|
|
479
|
+
seen: set[str] = set()
|
|
480
|
+
for value in values:
|
|
481
|
+
if not value:
|
|
482
|
+
continue
|
|
483
|
+
for part in (value, split_1c_identifier(value)):
|
|
484
|
+
normalized = normalize_query(part)
|
|
485
|
+
if normalized and normalized not in seen:
|
|
486
|
+
seen.add(normalized)
|
|
487
|
+
parts.append(part)
|
|
488
|
+
return " ".join(parts)
|
|
489
|
+
|
|
490
|
+
|
|
413
491
|
def normalize_query(text: str) -> str:
|
|
414
|
-
|
|
492
|
+
text = text.replace("ё", "е")
|
|
493
|
+
text = re.sub(r"[\u2010-\u2015–—−-]+", " ", text)
|
|
494
|
+
return re.sub(r"\s+", " ", text.strip().lower())
|
|
@@ -1,16 +1,22 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import math
|
|
3
6
|
import os
|
|
7
|
+
import re
|
|
4
8
|
import subprocess
|
|
5
9
|
from pathlib import Path
|
|
6
10
|
from typing import Any
|
|
7
11
|
|
|
8
12
|
from .detect import detect_source_format, discover_source_roots
|
|
9
13
|
from .model import SourceInfo
|
|
10
|
-
from .parsers import navigation_for_object, normalize_query, parse_source
|
|
14
|
+
from .parsers import navigation_for_object, normalize_query, parse_source, search_text_variants
|
|
11
15
|
from .store import RagStore, json_dumps, row_to_dict
|
|
12
16
|
|
|
13
17
|
DEFAULT_RAG_DB = Path(os.environ.get("MCP_1C_RAG_DB", "build/rag/onec-rag.sqlite"))
|
|
18
|
+
EMBEDDING_MODEL = "answer42-local-hash-v1"
|
|
19
|
+
EMBEDDING_DIM = 256
|
|
14
20
|
|
|
15
21
|
|
|
16
22
|
class RagService:
|
|
@@ -246,9 +252,11 @@ class RagService:
|
|
|
246
252
|
nav = navigation_for_object(obj.full_name, obj.kind)
|
|
247
253
|
if nav:
|
|
248
254
|
con.execute("INSERT INTO navigation_links(object_id, kind, url) VALUES(?,?,?)", (oid, nav[0], nav[1]))
|
|
249
|
-
|
|
255
|
+
object_terms = search_text_variants(obj.full_name, obj.name, obj.synonym)
|
|
256
|
+
chunk_text = f"{obj.full_name} {object_terms} {obj.synonym or ''} {obj.kind} navigation {nav[1] if nav else ''}"
|
|
250
257
|
cid = _insert_chunk(con, snapshot_id, oid, "object", chunk_text, obj.source_path, {"source": src["name"]})
|
|
251
258
|
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, obj.full_name, obj.kind, obj.source_path, cid))
|
|
259
|
+
_insert_embedding(con, cid, chunk_text)
|
|
252
260
|
counts["chunks"] += 1
|
|
253
261
|
if obj.synonym:
|
|
254
262
|
_insert_business_term(con, snapshot_id, obj.synonym, oid, 0.7, "object synonym")
|
|
@@ -258,6 +266,12 @@ class RagService:
|
|
|
258
266
|
continue
|
|
259
267
|
con.execute("INSERT INTO attributes(object_id,name,type_name,synonym,required,source_path,role,raw_json) VALUES(?,?,?,?,?,?,?,?)", (oid, attr.name, attr.type_name, attr.synonym, attr.required, attr.source_path, attr.role, json_dumps(attr.raw)))
|
|
260
268
|
counts["attributes"] += 1
|
|
269
|
+
text_parts = [attr.object_full_name, attr.role, attr.name, search_text_variants(attr.name, attr.synonym), attr.synonym or "", attr.type_name or ""]
|
|
270
|
+
chunk_text = " ".join(part for part in text_parts if part)
|
|
271
|
+
cid = _insert_chunk(con, snapshot_id, oid, "attribute", chunk_text, attr.source_path, {"source": src["name"], "role": attr.role})
|
|
272
|
+
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, attr.object_full_name, "Реквизит", attr.source_path, cid))
|
|
273
|
+
_insert_embedding(con, cid, chunk_text)
|
|
274
|
+
counts["chunks"] += 1
|
|
261
275
|
for tp in parsed.table_parts:
|
|
262
276
|
oid = object_ids.get(tp.object_full_name)
|
|
263
277
|
if not oid:
|
|
@@ -276,12 +290,25 @@ class RagService:
|
|
|
276
290
|
cur = con.execute("INSERT INTO forms(object_id,name,kind,source_path,raw_json) VALUES(?,?,?,?,?)", (oid, form.name, form.kind, form.source_path, json_dumps(form.raw)))
|
|
277
291
|
form_ids[(form.object_full_name, form.name)] = cur.lastrowid
|
|
278
292
|
counts["forms"] += 1
|
|
293
|
+
text_parts = [form.object_full_name, "форма", form.name, search_text_variants(form.name), form.kind or ""]
|
|
294
|
+
chunk_text = " ".join(part for part in text_parts if part)
|
|
295
|
+
cid = _insert_chunk(con, snapshot_id, oid, "form", chunk_text, form.source_path, {"source": src["name"], "form": form.name})
|
|
296
|
+
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, form.object_full_name, "Форма", form.source_path, cid))
|
|
297
|
+
_insert_embedding(con, cid, chunk_text)
|
|
298
|
+
counts["chunks"] += 1
|
|
279
299
|
for elem in parsed.form_elements:
|
|
280
300
|
fid = form_ids.get((elem.object_full_name, elem.form_name))
|
|
281
301
|
if not fid:
|
|
282
302
|
continue
|
|
283
303
|
con.execute("INSERT INTO form_elements(form_id,name,element_type,title,data_path,command_name,parent_name,source_path,raw_json) VALUES(?,?,?,?,?,?,?,?,?)", (fid, elem.name, elem.element_type, elem.title, elem.data_path, elem.command_name, elem.parent_name, elem.source_path, json_dumps(elem.raw)))
|
|
284
304
|
counts["form_elements"] += 1
|
|
305
|
+
oid = object_ids.get(elem.object_full_name)
|
|
306
|
+
text_parts = [elem.object_full_name, "элемент формы", elem.form_name, elem.name, search_text_variants(elem.name, elem.title, elem.data_path), elem.title or "", elem.data_path or "", elem.command_name or "", elem.element_type or ""]
|
|
307
|
+
chunk_text = " ".join(part for part in text_parts if part)
|
|
308
|
+
cid = _insert_chunk(con, snapshot_id, oid, "form_element", chunk_text, elem.source_path, {"source": src["name"], "form": elem.form_name, "element": elem.name})
|
|
309
|
+
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, elem.object_full_name, "ЭлементФормы", elem.source_path, cid))
|
|
310
|
+
_insert_embedding(con, cid, chunk_text)
|
|
311
|
+
counts["chunks"] += 1
|
|
285
312
|
for dcs_field in parsed.data_composition_fields:
|
|
286
313
|
oid = object_ids.get(dcs_field.object_full_name)
|
|
287
314
|
if not oid:
|
|
@@ -295,15 +322,23 @@ class RagService:
|
|
|
295
322
|
chunk_text = " ".join(part for part in text_parts if part)
|
|
296
323
|
cid = _insert_chunk(con, snapshot_id, oid, "dcs_field", chunk_text, dcs_field.source_path, {"source": src["name"], "schema": dcs_field.schema_name})
|
|
297
324
|
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, dcs_field.object_full_name, "СКД", dcs_field.source_path, cid))
|
|
325
|
+
_insert_embedding(con, cid, chunk_text)
|
|
298
326
|
counts["chunks"] += 1
|
|
299
327
|
for help_page in parsed.help_pages:
|
|
300
328
|
oid = object_ids.get(help_page.object_full_name)
|
|
301
329
|
if not oid:
|
|
302
330
|
continue
|
|
303
|
-
|
|
331
|
+
object_row = con.execute("SELECT name, synonym FROM metadata_objects WHERE id=?", (oid,)).fetchone()
|
|
332
|
+
object_terms = search_text_variants(
|
|
333
|
+
help_page.object_full_name,
|
|
334
|
+
object_row["name"] if object_row else None,
|
|
335
|
+
object_row["synonym"] if object_row else None,
|
|
336
|
+
)
|
|
337
|
+
text_parts = [help_page.object_full_name, object_terms, "справка", help_page.language, help_page.title or "", help_page.text or ""]
|
|
304
338
|
chunk_text = "\n".join(part for part in text_parts if part)
|
|
305
339
|
cid = _insert_chunk(con, snapshot_id, oid, "help", chunk_text, help_page.source_path, {"source": src["name"], "language": help_page.language, "title": help_page.title})
|
|
306
340
|
con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, help_page.object_full_name, "Справка", help_page.source_path, cid))
|
|
341
|
+
_insert_embedding(con, cid, chunk_text)
|
|
307
342
|
counts["help_pages"] += 1
|
|
308
343
|
counts["chunks"] += 1
|
|
309
344
|
return {"db_path": self.db_path, **counts}
|
|
@@ -497,7 +532,45 @@ class RagService:
|
|
|
497
532
|
seen.add(key)
|
|
498
533
|
if len(fts_rows) >= limit:
|
|
499
534
|
break
|
|
500
|
-
|
|
535
|
+
semantic_rows = self._semantic_query(con, q, snapshot_id, limit)
|
|
536
|
+
return {"query": text, "snapshot": snapshot, "terms": term_rows, "fts": fts_rows, "semantic": semantic_rows}
|
|
537
|
+
|
|
538
|
+
def _semantic_query(self, con, query: str, snapshot_id: int | None, limit: int) -> list[dict[str, Any]]:
|
|
539
|
+
query_vector = _embedding_vector(query)
|
|
540
|
+
params: list[Any] = [EMBEDDING_MODEL]
|
|
541
|
+
snapshot_filter = ""
|
|
542
|
+
if snapshot_id is not None:
|
|
543
|
+
snapshot_filter = "AND rc.snapshot_id=?"
|
|
544
|
+
params.append(snapshot_id)
|
|
545
|
+
rows = con.execute(
|
|
546
|
+
f"""
|
|
547
|
+
SELECT rc.text, rc.source_path, mo.full_name AS object_full_name, mo.kind,
|
|
548
|
+
re.vector_json
|
|
549
|
+
FROM rag_embeddings re
|
|
550
|
+
JOIN rag_chunks rc ON rc.id=re.chunk_id
|
|
551
|
+
LEFT JOIN metadata_objects mo ON mo.id=rc.object_id
|
|
552
|
+
WHERE re.model=? {snapshot_filter}
|
|
553
|
+
""",
|
|
554
|
+
params,
|
|
555
|
+
)
|
|
556
|
+
scored: list[dict[str, Any]] = []
|
|
557
|
+
for row in rows:
|
|
558
|
+
vector = json.loads(row["vector_json"])
|
|
559
|
+
score = _cosine(query_vector, vector)
|
|
560
|
+
if score <= 0:
|
|
561
|
+
continue
|
|
562
|
+
scored.append(
|
|
563
|
+
{
|
|
564
|
+
"text": row["text"],
|
|
565
|
+
"object_full_name": row["object_full_name"],
|
|
566
|
+
"kind": row["kind"],
|
|
567
|
+
"source_path": row["source_path"],
|
|
568
|
+
"score": score,
|
|
569
|
+
"model": EMBEDDING_MODEL,
|
|
570
|
+
}
|
|
571
|
+
)
|
|
572
|
+
scored.sort(key=lambda item: item["score"], reverse=True)
|
|
573
|
+
return scored[:limit]
|
|
501
574
|
|
|
502
575
|
|
|
503
576
|
def _git_commit(path: Path) -> str | None:
|
|
@@ -530,5 +603,47 @@ def _insert_chunk(con, snapshot_id, object_id, chunk_type, text, source_path, me
|
|
|
530
603
|
return cur.lastrowid
|
|
531
604
|
|
|
532
605
|
|
|
606
|
+
def _insert_embedding(con, chunk_id: int, text: str) -> None:
|
|
607
|
+
con.execute(
|
|
608
|
+
"""
|
|
609
|
+
INSERT INTO rag_embeddings(chunk_id,model,dim,vector_json,text_hash)
|
|
610
|
+
VALUES(?,?,?,?,?)
|
|
611
|
+
""",
|
|
612
|
+
(
|
|
613
|
+
chunk_id,
|
|
614
|
+
EMBEDDING_MODEL,
|
|
615
|
+
EMBEDDING_DIM,
|
|
616
|
+
json.dumps(_embedding_vector(text), separators=(",", ":")),
|
|
617
|
+
hashlib.sha256(text.encode("utf-8", errors="ignore")).hexdigest(),
|
|
618
|
+
),
|
|
619
|
+
)
|
|
620
|
+
|
|
621
|
+
|
|
622
|
+
def _embedding_vector(text: str) -> list[float]:
|
|
623
|
+
vector = [0.0] * EMBEDDING_DIM
|
|
624
|
+
normalized = normalize_query(text)
|
|
625
|
+
tokens = re.findall(r"[\w]+", normalized, flags=re.UNICODE)
|
|
626
|
+
features: list[str] = []
|
|
627
|
+
for token in tokens:
|
|
628
|
+
if len(token) <= 2:
|
|
629
|
+
features.append("tok:" + token)
|
|
630
|
+
continue
|
|
631
|
+
features.append("tok:" + token)
|
|
632
|
+
for size in (3, 4):
|
|
633
|
+
if len(token) >= size:
|
|
634
|
+
features.extend(f"ng{size}:" + token[pos : pos + size] for pos in range(len(token) - size + 1))
|
|
635
|
+
for feature in features:
|
|
636
|
+
digest = hashlib.blake2b(feature.encode("utf-8"), digest_size=8).digest()
|
|
637
|
+
bucket = int.from_bytes(digest[:4], "little") % EMBEDDING_DIM
|
|
638
|
+
sign = 1.0 if digest[4] & 1 else -1.0
|
|
639
|
+
vector[bucket] += sign
|
|
640
|
+
norm = math.sqrt(sum(value * value for value in vector)) or 1.0
|
|
641
|
+
return [round(value / norm, 6) for value in vector]
|
|
642
|
+
|
|
643
|
+
|
|
644
|
+
def _cosine(left: list[float], right: list[float]) -> float:
|
|
645
|
+
return sum(a * b for a, b in zip(left, right, strict=False))
|
|
646
|
+
|
|
647
|
+
|
|
533
648
|
def _insert_business_term(con, snapshot_id, term, object_id, confidence, evidence) -> None:
|
|
534
649
|
con.execute("INSERT INTO business_terms(snapshot_id,term,normalized_term,object_id,confidence,evidence) VALUES(?,?,?,?,?,?)", (snapshot_id, term, normalize_query(term), object_id, confidence, evidence))
|
|
@@ -5,7 +5,7 @@ import sqlite3
|
|
|
5
5
|
from pathlib import Path
|
|
6
6
|
from typing import Any
|
|
7
7
|
|
|
8
|
-
SCHEMA_VERSION =
|
|
8
|
+
SCHEMA_VERSION = 4
|
|
9
9
|
|
|
10
10
|
|
|
11
11
|
class RagStore:
|
|
@@ -25,7 +25,7 @@ class RagStore:
|
|
|
25
25
|
"""
|
|
26
26
|
CREATE TABLE IF NOT EXISTS schema_info(version INTEGER NOT NULL);
|
|
27
27
|
DELETE FROM schema_info;
|
|
28
|
-
INSERT INTO schema_info(version) VALUES (
|
|
28
|
+
INSERT INTO schema_info(version) VALUES (4);
|
|
29
29
|
|
|
30
30
|
CREATE TABLE IF NOT EXISTS sources(
|
|
31
31
|
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
@@ -195,6 +195,15 @@ class RagStore:
|
|
|
195
195
|
chunk_id UNINDEXED
|
|
196
196
|
);
|
|
197
197
|
|
|
198
|
+
CREATE TABLE IF NOT EXISTS rag_embeddings(
|
|
199
|
+
chunk_id INTEGER PRIMARY KEY,
|
|
200
|
+
model TEXT NOT NULL,
|
|
201
|
+
dim INTEGER NOT NULL,
|
|
202
|
+
vector_json TEXT NOT NULL,
|
|
203
|
+
text_hash TEXT NOT NULL,
|
|
204
|
+
FOREIGN KEY(chunk_id) REFERENCES rag_chunks(id) ON DELETE CASCADE
|
|
205
|
+
);
|
|
206
|
+
|
|
198
207
|
CREATE INDEX IF NOT EXISTS idx_objects_source_name ON metadata_objects(source_id, full_name);
|
|
199
208
|
CREATE INDEX IF NOT EXISTS idx_objects_kind ON metadata_objects(kind);
|
|
200
209
|
CREATE INDEX IF NOT EXISTS idx_attrs_object ON attributes(object_id);
|
|
@@ -210,6 +219,17 @@ class RagStore:
|
|
|
210
219
|
con.execute("ALTER TABLE attributes ADD COLUMN role TEXT NOT NULL DEFAULT 'attribute'")
|
|
211
220
|
|
|
212
221
|
def clear_index_for_source(self, con: sqlite3.Connection, source_id: int) -> None:
|
|
222
|
+
con.execute(
|
|
223
|
+
"""
|
|
224
|
+
DELETE FROM rag_embeddings
|
|
225
|
+
WHERE chunk_id IN (
|
|
226
|
+
SELECT rc.id
|
|
227
|
+
FROM rag_chunks rc JOIN metadata_objects mo ON mo.id=rc.object_id
|
|
228
|
+
WHERE mo.source_id=?
|
|
229
|
+
)
|
|
230
|
+
""",
|
|
231
|
+
(source_id,),
|
|
232
|
+
)
|
|
213
233
|
con.execute(
|
|
214
234
|
"""
|
|
215
235
|
DELETE FROM search_fts
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|