answer42 0.3.2__tar.gz → 0.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. {answer42-0.3.2 → answer42-0.3.4}/PKG-INFO +1 -1
  2. {answer42-0.3.2 → answer42-0.3.4}/pyproject.toml +1 -1
  3. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/MCPTestClient.cf +0 -0
  4. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/MCPTestManager.cf +0 -0
  5. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/parsers.py +100 -20
  6. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/service.py +119 -4
  7. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/store.py +22 -2
  8. {answer42-0.3.2 → answer42-0.3.4}/.gitignore +0 -0
  9. {answer42-0.3.2 → answer42-0.3.4}/LICENSE +0 -0
  10. {answer42-0.3.2 → answer42-0.3.4}/README.md +0 -0
  11. {answer42-0.3.2 → answer42-0.3.4}/credentials.example.json +0 -0
  12. {answer42-0.3.2 → answer42-0.3.4}/docs/agent-installation.md +0 -0
  13. {answer42-0.3.2 → answer42-0.3.4}/docs/architecture.md +0 -0
  14. {answer42-0.3.2 → answer42-0.3.4}/docs/assets/answer42-logo.png +0 -0
  15. {answer42-0.3.2 → answer42-0.3.4}/docs/assets/platform42-logo.svg +0 -0
  16. {answer42-0.3.2 → answer42-0.3.4}/docs/installation.md +0 -0
  17. {answer42-0.3.2 → answer42-0.3.4}/scripts/build_cf.py +0 -0
  18. {answer42-0.3.2 → answer42-0.3.4}/scripts/build_pages.py +0 -0
  19. {answer42-0.3.2 → answer42-0.3.4}/scripts/e2e_stable.py +0 -0
  20. {answer42-0.3.2 → answer42-0.3.4}/scripts/openclaw_mcp_autoreload.py +0 -0
  21. {answer42-0.3.2 → answer42-0.3.4}/scripts/rag_cli.py +0 -0
  22. {answer42-0.3.2 → answer42-0.3.4}/src/cf/ConfigDumpInfo.xml +0 -0
  23. {answer42-0.3.2 → answer42-0.3.4}/src/cf/Configuration.xml +0 -0
  24. {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Ext/ObjectModule.bsl +0 -0
  25. {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260/Ext/Form/Module.bsl" +0 -0
  26. {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260/Ext/Form.xml" +0 -0
  27. {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager/Forms//320/244/320/276/321/200/320/274/320/260.xml" +0 -0
  28. {answer42-0.3.2 → answer42-0.3.4}/src/cf/DataProcessors/MCPTestManager.xml +0 -0
  29. {answer42-0.3.2 → answer42-0.3.4}/src/cf/Ext/ManagedApplicationModule.bsl +0 -0
  30. {answer42-0.3.2 → answer42-0.3.4}/src/cf/Languages//320/240/321/203/321/201/321/201/320/272/320/270/320/271.xml" +0 -0
  31. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Catalogs//320/237/320/241_/320/222/320/273/320/260/320/264/320/265/320/273/320/265/321/206.xml" +0 -0
  32. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Catalogs//320/237/320/241_/320/241/320/277/321/200/320/260/320/262/320/276/321/207/320/275/320/270/320/2721.xml" +0 -0
  33. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/ConfigDumpInfo.xml +0 -0
  34. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Configuration.xml +0 -0
  35. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Languages//320/240/321/203/321/201/321/201/320/272/320/270/320/271.xml" +0 -0
  36. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Ext/ManagerModule.bsl" +0 -0
  37. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Ext/ObjectModule.bsl" +0 -0
  38. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Templates//320/236/321/201/320/275/320/276/320/262/320/275/320/260/321/217/320/241/321/205/320/265/320/274/320/260/320/232/320/276/320/274/320/277/320/276/320/275/320/276/320/262/320/272/320/270/320/224/320/260/320/275/320/275/321/213/321/205/Ext/Template.xml" +0 -0
  39. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262/Templates//320/236/321/201/320/275/320/276/320/262/320/275/320/260/321/217/320/241/321/205/320/265/320/274/320/260/320/232/320/276/320/274/320/277/320/276/320/275/320/276/320/262/320/272/320/270/320/224/320/260/320/275/320/275/321/213/321/205.xml" +0 -0
  40. {answer42-0.3.2 → answer42-0.3.4}/src/client_cf/Reports//320/237/320/241_/320/241/320/277/320/270/321/201/320/276/320/272/320/255/320/273/320/265/320/274/320/265/320/275/321/202/320/276/320/262.xml" +0 -0
  41. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/__init__.py +0 -0
  42. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/__init__.py +0 -0
  43. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/skills/answer42/SKILL.md +0 -0
  44. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/assets/skills/answer42-rag/SKILL.md +0 -0
  45. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/bridge.py +0 -0
  46. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/credentials.py +0 -0
  47. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/os_support.py +0 -0
  48. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/platform.py +0 -0
  49. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/protocol.py +0 -0
  50. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/__init__.py +0 -0
  51. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/detect.py +0 -0
  52. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/dump.py +0 -0
  53. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/rag/model.py +0 -0
  54. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/recorder.py +0 -0
  55. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/release_helper.py +0 -0
  56. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/runtime.py +0 -0
  57. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/server.py +0 -0
  58. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/skill_installer.py +0 -0
  59. {answer42-0.3.2 → answer42-0.3.4}/src/mcp_1c/window_control.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: answer42
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: Answer42 — The Answer to Life, Universe, and 1C — UI Driver. MCP-powered 1C:Enterprise UI automation: click, fill, navigate, test, and introspect managed forms through the test-client API
5
5
  Author: Marvin (AI Assistant), 42Clouds, and contributors
6
6
  Author-email: "Kosolapov Stanislav (proDOOMman)" <prodoomman@gmail.com>
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "answer42"
3
- version = "0.3.2"
3
+ version = "0.3.4"
4
4
  description = "Answer42 — The Answer to Life, Universe, and 1C — UI Driver. MCP-powered 1C:Enterprise UI automation: click, fill, navigate, test, and introspect managed forms through the test-client API"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -21,6 +21,8 @@ from .model import (
21
21
  )
22
22
 
23
23
  KIND_DIRS = {
24
+ "Constants": "Константа",
25
+ "CommonForms": "ОбщаяФорма",
24
26
  "Catalogs": "Справочник",
25
27
  "Documents": "Документ",
26
28
  "DataProcessors": "Обработка",
@@ -33,6 +35,8 @@ KIND_DIRS = {
33
35
  "CalculationRegisters": "РегистрРасчета",
34
36
  }
35
37
 
38
+ FORM_FILE_NAMES = {"Form.form", "Form.xml"}
39
+
36
40
 
37
41
  def _strip_ns(tag: str) -> str:
38
42
  return tag.rsplit("}", 1)[-1]
@@ -199,7 +203,10 @@ def parse_edt_source(source: SourceInfo) -> ParsedSource:
199
203
  parsed.objects.append(MetadataObject(full, ru_kind, obj_name, synonym, str(mdo), source.format, hierarchical, [o for o in owners if o]))
200
204
  if root is not None:
201
205
  _extract_metadata_children(parsed, root, full, mdo)
202
- _extract_forms(parsed, mdo.parent, full)
206
+ if ru_kind == "ОбщаяФорма":
207
+ _extract_common_form(parsed, mdo.parent, full)
208
+ else:
209
+ _extract_forms(parsed, mdo.parent, full)
203
210
  _extract_data_composition_schemas(parsed, mdo.parent, full)
204
211
  _extract_help_pages(parsed, mdo.parent, full)
205
212
  return parsed
@@ -231,12 +238,26 @@ def parse_designer_xml_source(source: SourceInfo) -> ParsedSource:
231
238
  if root is not None:
232
239
  _extract_metadata_children(parsed, root, full, xml)
233
240
  object_dir = xml.parent if xml.parent != base else base / xml.stem
234
- _extract_forms(parsed, object_dir, full)
241
+ if ru_kind == "ОбщаяФорма":
242
+ _extract_common_form(parsed, object_dir, full)
243
+ else:
244
+ _extract_forms(parsed, object_dir, full)
235
245
  _extract_data_composition_schemas(parsed, object_dir, full)
236
246
  _extract_help_pages(parsed, object_dir, full)
237
247
  return parsed
238
248
 
239
249
 
250
+ def _extract_common_form(parsed: ParsedSource, object_dir: Path, full: str) -> None:
251
+ form_xml = _find_form_file(object_dir)
252
+ parsed.forms.append(Form(full, "Форма", "CommonForm", str(form_xml) if form_xml.exists() else str(object_dir)))
253
+ if not form_xml.exists():
254
+ return
255
+ root = _xml_root(form_xml)
256
+ if root is not None:
257
+ _extract_form_elements(parsed, root, full, "Форма", form_xml)
258
+ _extract_embedded_data_composition_fields(parsed, root, full, "Форма", form_xml)
259
+
260
+
240
261
  def _extract_metadata_children(parsed: ParsedSource, root: ET.Element, full: str, source_path: Path) -> None:
241
262
  # Heuristic parser covering both EDT and designer XML structures.
242
263
  # Keep object attributes and table-part columns separate: XML dumps often
@@ -273,9 +294,7 @@ def _extract_forms(parsed: ParsedSource, object_dir: Path, full: str) -> None:
273
294
  return
274
295
  for form_dir in sorted([p for p in forms_dir.iterdir() if p.is_dir()]):
275
296
  form_name = form_dir.name
276
- form_xml = form_dir / "Ext" / "Form.xml"
277
- if not form_xml.exists():
278
- form_xml = form_dir / "Form.xml"
297
+ form_xml = _find_form_file(form_dir)
279
298
  parsed.forms.append(Form(full, form_name, None, str(form_xml) if form_xml.exists() else str(form_dir)))
280
299
  if form_xml.exists():
281
300
  root = _xml_root(form_xml)
@@ -284,23 +303,34 @@ def _extract_forms(parsed: ParsedSource, object_dir: Path, full: str) -> None:
284
303
  _extract_embedded_data_composition_fields(parsed, root, full, form_name, form_xml)
285
304
 
286
305
 
306
+ def _find_form_file(form_dir: Path) -> Path:
307
+ for candidate_dir in (form_dir / "Ext", form_dir):
308
+ for name in FORM_FILE_NAMES:
309
+ form_file = candidate_dir / name
310
+ if form_file.exists():
311
+ return form_file
312
+ return form_dir / "Ext" / "Form.xml"
313
+
314
+
287
315
  def _extract_form_elements(parsed: ParsedSource, root: ET.Element, full: str, form_name: str, source_path: Path) -> None:
288
316
  for elem in root.iter():
289
317
  tag = _strip_ns(elem.tag)
290
318
  low = tag.lower()
291
- if low not in {"element", "item", "items", "button", "command", "field", "table", "group"}:
319
+ if low not in {"element", "item", "items", "button", "command", "field", "table", "group", "extendedtooltip", "contextmenu"}:
292
320
  continue
293
321
  name = elem.attrib.get("name") or elem.attrib.get("Name") or _first_text(elem, ["name", "Name"])
294
322
  if not name:
295
323
  continue
324
+ title = elem.attrib.get("title") or elem.attrib.get("Title") or _first_text(elem, ["title", "Title", "caption", "Caption"])
325
+ data_path = _form_data_path(elem)
296
326
  parsed.form_elements.append(
297
327
  FormElement(
298
328
  full,
299
329
  form_name,
300
330
  name,
301
331
  tag,
302
- elem.attrib.get("title") or elem.attrib.get("Title") or _first_text(elem, ["title", "Title", "caption", "Caption"]),
303
- elem.attrib.get("dataPath") or elem.attrib.get("path") or _first_text(elem, ["dataPath", "path", "DataPath"]),
332
+ title,
333
+ data_path,
304
334
  elem.attrib.get("commandName") or _first_text(elem, ["commandName", "CommandName"]),
305
335
  None,
306
336
  str(source_path),
@@ -308,6 +338,20 @@ def _extract_form_elements(parsed: ParsedSource, root: ET.Element, full: str, fo
308
338
  )
309
339
 
310
340
 
341
+ def _form_data_path(elem: ET.Element) -> str | None:
342
+ direct = elem.attrib.get("dataPath") or elem.attrib.get("path") or _direct_text(elem, ["dataPath", "path", "DataPath"])
343
+ if direct:
344
+ return direct
345
+ for child in list(elem):
346
+ if _strip_ns(child.tag).lower() == "datapath":
347
+ segments = [_text(segment) for segment in child.iter() if _strip_ns(segment.tag).lower() in {"segments", "segment"}]
348
+ values = [value for value in segments if value]
349
+ if values:
350
+ return ".".join(values)
351
+ return _text(child)
352
+ return None
353
+
354
+
311
355
  def _extract_embedded_data_composition_fields(parsed: ParsedSource, root: ET.Element, full: str, form_name: str, source_path: Path) -> None:
312
356
  dataset_name: str | None = None
313
357
  for elem in root.iter():
@@ -341,17 +385,20 @@ def _extract_embedded_data_composition_fields(parsed: ParsedSource, root: ET.Ele
341
385
 
342
386
 
343
387
  def _extract_help_pages(parsed: ParsedSource, object_dir: Path, full: str) -> None:
344
- help_dir = object_dir / "Help"
345
- if not help_dir.exists():
346
- return
347
- for page in sorted(help_dir.glob("*.html")):
348
- raw = page.read_text(encoding="utf-8", errors="ignore")
349
- title_match = re.search(r"<title[^>]*>(.*?)</title>", raw, flags=re.IGNORECASE | re.DOTALL)
350
- h1_match = re.search(r"<h1[^>]*>(.*?)</h1>", raw, flags=re.IGNORECASE | re.DOTALL)
351
- title = _html_to_text((title_match or h1_match).group(1)) if (title_match or h1_match) else None
352
- text = _html_to_text(raw)
353
- if text:
354
- parsed.help_pages.append(HelpPage(full, page.stem, title, text, str(page), {"file": page.name}))
388
+ # EDT exports usually keep help as <Object>/Help/*.html, while Designer
389
+ # hierarchical dumps keep it under <Object>/Ext/Help/*.html with
390
+ # <Object>/Ext/Help.xml next to the language-specific pages.
391
+ for help_dir in (object_dir / "Help", object_dir / "Ext" / "Help"):
392
+ if not help_dir.exists():
393
+ continue
394
+ for page in sorted(help_dir.glob("*.html")):
395
+ raw = page.read_text(encoding="utf-8-sig", errors="ignore")
396
+ title_match = re.search(r"<title[^>]*>(.*?)</title>", raw, flags=re.IGNORECASE | re.DOTALL)
397
+ h1_match = re.search(r"<h1[^>]*>(.*?)</h1>", raw, flags=re.IGNORECASE | re.DOTALL)
398
+ title = _html_to_text((title_match or h1_match).group(1)) if (title_match or h1_match) else None
399
+ text = _html_to_text(raw)
400
+ if text:
401
+ parsed.help_pages.append(HelpPage(full, page.stem, title, text, str(page), {"file": page.name}))
355
402
 
356
403
 
357
404
  def _html_to_text(raw: str) -> str:
@@ -410,5 +457,38 @@ def navigation_for_object(full_name: str, kind: str) -> tuple[str, str] | None:
410
457
  return None
411
458
 
412
459
 
460
+ def split_1c_identifier(text: str | None) -> str:
461
+ """Return a human-readable variant of 1C metadata identifiers.
462
+
463
+ 1C metadata names are commonly written as CamelCase identifiers, e.g.
464
+ ``ОборотноСальдоваяВедомостьПоСчету``. FTS tokenizers do not split those
465
+ words automatically, so business queries like ``оборотно сальдовая`` miss
466
+ chunks that contain only the technical identifier. Keep this helper small
467
+ and deterministic: it does not add domain synonyms, it only exposes words
468
+ already present in the metadata name.
469
+ """
470
+ if not text:
471
+ return ""
472
+ value = re.sub(r"[._/\\:-]+", " ", text)
473
+ value = re.sub(r"(?<=[а-яёa-z0-9])(?=[А-ЯЁA-Z])", " ", value)
474
+ return re.sub(r"\s+", " ", value).strip()
475
+
476
+
477
+ def search_text_variants(*values: str | None) -> str:
478
+ parts: list[str] = []
479
+ seen: set[str] = set()
480
+ for value in values:
481
+ if not value:
482
+ continue
483
+ for part in (value, split_1c_identifier(value)):
484
+ normalized = normalize_query(part)
485
+ if normalized and normalized not in seen:
486
+ seen.add(normalized)
487
+ parts.append(part)
488
+ return " ".join(parts)
489
+
490
+
413
491
  def normalize_query(text: str) -> str:
414
- return re.sub(r"\s+", " ", text.strip().lower().replace("ё", "е"))
492
+ text = text.replace("ё", "е")
493
+ text = re.sub(r"[\u2010-\u2015–—−-]+", " ", text)
494
+ return re.sub(r"\s+", " ", text.strip().lower())
@@ -1,16 +1,22 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import hashlib
4
+ import json
5
+ import math
3
6
  import os
7
+ import re
4
8
  import subprocess
5
9
  from pathlib import Path
6
10
  from typing import Any
7
11
 
8
12
  from .detect import detect_source_format, discover_source_roots
9
13
  from .model import SourceInfo
10
- from .parsers import navigation_for_object, normalize_query, parse_source
14
+ from .parsers import navigation_for_object, normalize_query, parse_source, search_text_variants
11
15
  from .store import RagStore, json_dumps, row_to_dict
12
16
 
13
17
  DEFAULT_RAG_DB = Path(os.environ.get("MCP_1C_RAG_DB", "build/rag/onec-rag.sqlite"))
18
+ EMBEDDING_MODEL = "answer42-local-hash-v1"
19
+ EMBEDDING_DIM = 256
14
20
 
15
21
 
16
22
  class RagService:
@@ -246,9 +252,11 @@ class RagService:
246
252
  nav = navigation_for_object(obj.full_name, obj.kind)
247
253
  if nav:
248
254
  con.execute("INSERT INTO navigation_links(object_id, kind, url) VALUES(?,?,?)", (oid, nav[0], nav[1]))
249
- chunk_text = f"{obj.full_name} {obj.synonym or ''} {obj.kind} navigation {nav[1] if nav else ''}"
255
+ object_terms = search_text_variants(obj.full_name, obj.name, obj.synonym)
256
+ chunk_text = f"{obj.full_name} {object_terms} {obj.synonym or ''} {obj.kind} navigation {nav[1] if nav else ''}"
250
257
  cid = _insert_chunk(con, snapshot_id, oid, "object", chunk_text, obj.source_path, {"source": src["name"]})
251
258
  con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, obj.full_name, obj.kind, obj.source_path, cid))
259
+ _insert_embedding(con, cid, chunk_text)
252
260
  counts["chunks"] += 1
253
261
  if obj.synonym:
254
262
  _insert_business_term(con, snapshot_id, obj.synonym, oid, 0.7, "object synonym")
@@ -258,6 +266,12 @@ class RagService:
258
266
  continue
259
267
  con.execute("INSERT INTO attributes(object_id,name,type_name,synonym,required,source_path,role,raw_json) VALUES(?,?,?,?,?,?,?,?)", (oid, attr.name, attr.type_name, attr.synonym, attr.required, attr.source_path, attr.role, json_dumps(attr.raw)))
260
268
  counts["attributes"] += 1
269
+ text_parts = [attr.object_full_name, attr.role, attr.name, search_text_variants(attr.name, attr.synonym), attr.synonym or "", attr.type_name or ""]
270
+ chunk_text = " ".join(part for part in text_parts if part)
271
+ cid = _insert_chunk(con, snapshot_id, oid, "attribute", chunk_text, attr.source_path, {"source": src["name"], "role": attr.role})
272
+ con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, attr.object_full_name, "Реквизит", attr.source_path, cid))
273
+ _insert_embedding(con, cid, chunk_text)
274
+ counts["chunks"] += 1
261
275
  for tp in parsed.table_parts:
262
276
  oid = object_ids.get(tp.object_full_name)
263
277
  if not oid:
@@ -276,12 +290,25 @@ class RagService:
276
290
  cur = con.execute("INSERT INTO forms(object_id,name,kind,source_path,raw_json) VALUES(?,?,?,?,?)", (oid, form.name, form.kind, form.source_path, json_dumps(form.raw)))
277
291
  form_ids[(form.object_full_name, form.name)] = cur.lastrowid
278
292
  counts["forms"] += 1
293
+ text_parts = [form.object_full_name, "форма", form.name, search_text_variants(form.name), form.kind or ""]
294
+ chunk_text = " ".join(part for part in text_parts if part)
295
+ cid = _insert_chunk(con, snapshot_id, oid, "form", chunk_text, form.source_path, {"source": src["name"], "form": form.name})
296
+ con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, form.object_full_name, "Форма", form.source_path, cid))
297
+ _insert_embedding(con, cid, chunk_text)
298
+ counts["chunks"] += 1
279
299
  for elem in parsed.form_elements:
280
300
  fid = form_ids.get((elem.object_full_name, elem.form_name))
281
301
  if not fid:
282
302
  continue
283
303
  con.execute("INSERT INTO form_elements(form_id,name,element_type,title,data_path,command_name,parent_name,source_path,raw_json) VALUES(?,?,?,?,?,?,?,?,?)", (fid, elem.name, elem.element_type, elem.title, elem.data_path, elem.command_name, elem.parent_name, elem.source_path, json_dumps(elem.raw)))
284
304
  counts["form_elements"] += 1
305
+ oid = object_ids.get(elem.object_full_name)
306
+ text_parts = [elem.object_full_name, "элемент формы", elem.form_name, elem.name, search_text_variants(elem.name, elem.title, elem.data_path), elem.title or "", elem.data_path or "", elem.command_name or "", elem.element_type or ""]
307
+ chunk_text = " ".join(part for part in text_parts if part)
308
+ cid = _insert_chunk(con, snapshot_id, oid, "form_element", chunk_text, elem.source_path, {"source": src["name"], "form": elem.form_name, "element": elem.name})
309
+ con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, elem.object_full_name, "ЭлементФормы", elem.source_path, cid))
310
+ _insert_embedding(con, cid, chunk_text)
311
+ counts["chunks"] += 1
285
312
  for dcs_field in parsed.data_composition_fields:
286
313
  oid = object_ids.get(dcs_field.object_full_name)
287
314
  if not oid:
@@ -295,15 +322,23 @@ class RagService:
295
322
  chunk_text = " ".join(part for part in text_parts if part)
296
323
  cid = _insert_chunk(con, snapshot_id, oid, "dcs_field", chunk_text, dcs_field.source_path, {"source": src["name"], "schema": dcs_field.schema_name})
297
324
  con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, dcs_field.object_full_name, "СКД", dcs_field.source_path, cid))
325
+ _insert_embedding(con, cid, chunk_text)
298
326
  counts["chunks"] += 1
299
327
  for help_page in parsed.help_pages:
300
328
  oid = object_ids.get(help_page.object_full_name)
301
329
  if not oid:
302
330
  continue
303
- text_parts = [help_page.object_full_name, "справка", help_page.language, help_page.title or "", help_page.text or ""]
331
+ object_row = con.execute("SELECT name, synonym FROM metadata_objects WHERE id=?", (oid,)).fetchone()
332
+ object_terms = search_text_variants(
333
+ help_page.object_full_name,
334
+ object_row["name"] if object_row else None,
335
+ object_row["synonym"] if object_row else None,
336
+ )
337
+ text_parts = [help_page.object_full_name, object_terms, "справка", help_page.language, help_page.title or "", help_page.text or ""]
304
338
  chunk_text = "\n".join(part for part in text_parts if part)
305
339
  cid = _insert_chunk(con, snapshot_id, oid, "help", chunk_text, help_page.source_path, {"source": src["name"], "language": help_page.language, "title": help_page.title})
306
340
  con.execute("INSERT INTO search_fts(text, object_full_name, kind, source_path, chunk_id) VALUES(?,?,?,?,?)", (chunk_text, help_page.object_full_name, "Справка", help_page.source_path, cid))
341
+ _insert_embedding(con, cid, chunk_text)
307
342
  counts["help_pages"] += 1
308
343
  counts["chunks"] += 1
309
344
  return {"db_path": self.db_path, **counts}
@@ -497,7 +532,45 @@ class RagService:
497
532
  seen.add(key)
498
533
  if len(fts_rows) >= limit:
499
534
  break
500
- return {"query": text, "snapshot": snapshot, "terms": term_rows, "fts": fts_rows}
535
+ semantic_rows = self._semantic_query(con, q, snapshot_id, limit)
536
+ return {"query": text, "snapshot": snapshot, "terms": term_rows, "fts": fts_rows, "semantic": semantic_rows}
537
+
538
+ def _semantic_query(self, con, query: str, snapshot_id: int | None, limit: int) -> list[dict[str, Any]]:
539
+ query_vector = _embedding_vector(query)
540
+ params: list[Any] = [EMBEDDING_MODEL]
541
+ snapshot_filter = ""
542
+ if snapshot_id is not None:
543
+ snapshot_filter = "AND rc.snapshot_id=?"
544
+ params.append(snapshot_id)
545
+ rows = con.execute(
546
+ f"""
547
+ SELECT rc.text, rc.source_path, mo.full_name AS object_full_name, mo.kind,
548
+ re.vector_json
549
+ FROM rag_embeddings re
550
+ JOIN rag_chunks rc ON rc.id=re.chunk_id
551
+ LEFT JOIN metadata_objects mo ON mo.id=rc.object_id
552
+ WHERE re.model=? {snapshot_filter}
553
+ """,
554
+ params,
555
+ )
556
+ scored: list[dict[str, Any]] = []
557
+ for row in rows:
558
+ vector = json.loads(row["vector_json"])
559
+ score = _cosine(query_vector, vector)
560
+ if score <= 0:
561
+ continue
562
+ scored.append(
563
+ {
564
+ "text": row["text"],
565
+ "object_full_name": row["object_full_name"],
566
+ "kind": row["kind"],
567
+ "source_path": row["source_path"],
568
+ "score": score,
569
+ "model": EMBEDDING_MODEL,
570
+ }
571
+ )
572
+ scored.sort(key=lambda item: item["score"], reverse=True)
573
+ return scored[:limit]
501
574
 
502
575
 
503
576
  def _git_commit(path: Path) -> str | None:
@@ -530,5 +603,47 @@ def _insert_chunk(con, snapshot_id, object_id, chunk_type, text, source_path, me
530
603
  return cur.lastrowid
531
604
 
532
605
 
606
+ def _insert_embedding(con, chunk_id: int, text: str) -> None:
607
+ con.execute(
608
+ """
609
+ INSERT INTO rag_embeddings(chunk_id,model,dim,vector_json,text_hash)
610
+ VALUES(?,?,?,?,?)
611
+ """,
612
+ (
613
+ chunk_id,
614
+ EMBEDDING_MODEL,
615
+ EMBEDDING_DIM,
616
+ json.dumps(_embedding_vector(text), separators=(",", ":")),
617
+ hashlib.sha256(text.encode("utf-8", errors="ignore")).hexdigest(),
618
+ ),
619
+ )
620
+
621
+
622
+ def _embedding_vector(text: str) -> list[float]:
623
+ vector = [0.0] * EMBEDDING_DIM
624
+ normalized = normalize_query(text)
625
+ tokens = re.findall(r"[\w]+", normalized, flags=re.UNICODE)
626
+ features: list[str] = []
627
+ for token in tokens:
628
+ if len(token) <= 2:
629
+ features.append("tok:" + token)
630
+ continue
631
+ features.append("tok:" + token)
632
+ for size in (3, 4):
633
+ if len(token) >= size:
634
+ features.extend(f"ng{size}:" + token[pos : pos + size] for pos in range(len(token) - size + 1))
635
+ for feature in features:
636
+ digest = hashlib.blake2b(feature.encode("utf-8"), digest_size=8).digest()
637
+ bucket = int.from_bytes(digest[:4], "little") % EMBEDDING_DIM
638
+ sign = 1.0 if digest[4] & 1 else -1.0
639
+ vector[bucket] += sign
640
+ norm = math.sqrt(sum(value * value for value in vector)) or 1.0
641
+ return [round(value / norm, 6) for value in vector]
642
+
643
+
644
+ def _cosine(left: list[float], right: list[float]) -> float:
645
+ return sum(a * b for a, b in zip(left, right, strict=False))
646
+
647
+
533
648
  def _insert_business_term(con, snapshot_id, term, object_id, confidence, evidence) -> None:
534
649
  con.execute("INSERT INTO business_terms(snapshot_id,term,normalized_term,object_id,confidence,evidence) VALUES(?,?,?,?,?,?)", (snapshot_id, term, normalize_query(term), object_id, confidence, evidence))
@@ -5,7 +5,7 @@ import sqlite3
5
5
  from pathlib import Path
6
6
  from typing import Any
7
7
 
8
- SCHEMA_VERSION = 3
8
+ SCHEMA_VERSION = 4
9
9
 
10
10
 
11
11
  class RagStore:
@@ -25,7 +25,7 @@ class RagStore:
25
25
  """
26
26
  CREATE TABLE IF NOT EXISTS schema_info(version INTEGER NOT NULL);
27
27
  DELETE FROM schema_info;
28
- INSERT INTO schema_info(version) VALUES (3);
28
+ INSERT INTO schema_info(version) VALUES (4);
29
29
 
30
30
  CREATE TABLE IF NOT EXISTS sources(
31
31
  id INTEGER PRIMARY KEY AUTOINCREMENT,
@@ -195,6 +195,15 @@ class RagStore:
195
195
  chunk_id UNINDEXED
196
196
  );
197
197
 
198
+ CREATE TABLE IF NOT EXISTS rag_embeddings(
199
+ chunk_id INTEGER PRIMARY KEY,
200
+ model TEXT NOT NULL,
201
+ dim INTEGER NOT NULL,
202
+ vector_json TEXT NOT NULL,
203
+ text_hash TEXT NOT NULL,
204
+ FOREIGN KEY(chunk_id) REFERENCES rag_chunks(id) ON DELETE CASCADE
205
+ );
206
+
198
207
  CREATE INDEX IF NOT EXISTS idx_objects_source_name ON metadata_objects(source_id, full_name);
199
208
  CREATE INDEX IF NOT EXISTS idx_objects_kind ON metadata_objects(kind);
200
209
  CREATE INDEX IF NOT EXISTS idx_attrs_object ON attributes(object_id);
@@ -210,6 +219,17 @@ class RagStore:
210
219
  con.execute("ALTER TABLE attributes ADD COLUMN role TEXT NOT NULL DEFAULT 'attribute'")
211
220
 
212
221
  def clear_index_for_source(self, con: sqlite3.Connection, source_id: int) -> None:
222
+ con.execute(
223
+ """
224
+ DELETE FROM rag_embeddings
225
+ WHERE chunk_id IN (
226
+ SELECT rc.id
227
+ FROM rag_chunks rc JOIN metadata_objects mo ON mo.id=rc.object_id
228
+ WHERE mo.source_id=?
229
+ )
230
+ """,
231
+ (source_id,),
232
+ )
213
233
  con.execute(
214
234
  """
215
235
  DELETE FROM search_fts
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes