mcp-okf 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_okf-0.1.0/.gitignore +8 -0
- mcp_okf-0.1.0/LICENSE +21 -0
- mcp_okf-0.1.0/PKG-INFO +68 -0
- mcp_okf-0.1.0/README.md +54 -0
- mcp_okf-0.1.0/mcp_okf/__init__.py +0 -0
- mcp_okf-0.1.0/mcp_okf/bundle.py +150 -0
- mcp_okf-0.1.0/mcp_okf/server.py +52 -0
- mcp_okf-0.1.0/mcp_okf/store.py +190 -0
- mcp_okf-0.1.0/mcp_okf/tools.py +185 -0
- mcp_okf-0.1.0/pyproject.toml +29 -0
- mcp_okf-0.1.0/tests/conftest.py +60 -0
- mcp_okf-0.1.0/tests/test_bundle.py +57 -0
- mcp_okf-0.1.0/tests/test_tools.py +72 -0
mcp_okf-0.1.0/.gitignore
ADDED
mcp_okf-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Daniel Xavier Araújo
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
mcp_okf-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: mcp-okf
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: MCP server de base de conhecimento sobre bundles OKF (alm-sync): indexação, enriquecimento e busca semântica
|
|
5
|
+
Author-email: Daniel Xavier Araújo <danielxaraujo@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Requires-Dist: markdown-it-py>=3
|
|
10
|
+
Requires-Dist: mcp[cli]>=2.2
|
|
11
|
+
Requires-Dist: pyyaml>=6
|
|
12
|
+
Requires-Dist: qdrant-client[fastembed]>=1.12
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
|
|
15
|
+
# mcp-okf
|
|
16
|
+
|
|
17
|
+
Servidor MCP que transforma um bundle OKF (pasta com `index.md` e um `.md` por documento, como o gerado pela
|
|
18
|
+
alm-sync do [mcp-alm](https://github.com/dxaraujo/mcp-alm)) numa base de conhecimento consultável em linguagem natural.
|
|
19
|
+
|
|
20
|
+
- **Índice vetorial:** Qdrant em modo local (pasta no disco, sem servidor), busca híbrida denso + BM25 com fusão RRF.
|
|
21
|
+
- **Embeddings:** fastembed local, `paraphrase-multilingual-mpnet-base-v2` (PT-BR); nada sai da máquina.
|
|
22
|
+
- **Enriquecimento:** feito pela própria LLM do cliente (sem chave de API): resumo, palavras-chave/sinônimos,
|
|
23
|
+
entidades e perguntas que o documento responde.
|
|
24
|
+
- **Links:** `links` do frontmatter (nomes do DOORS Next), links do corpo para outros arquivos do bundle (`cita`) e
|
|
25
|
+
URL do ALM de artefato que está no bundle; backlinks e documentos semanticamente similares na leitura.
|
|
26
|
+
|
|
27
|
+
## Instalação
|
|
28
|
+
|
|
29
|
+
```bash
|
|
30
|
+
uv sync
|
|
31
|
+
# baixa os modelos uma vez (~1 GB), para a 1ª indexação não estourar o timeout do cliente MCP
|
|
32
|
+
uv run python -c "from mcp_okf import store; store.embed_dense(['x']); store.embed_sparse(['x'])"
|
|
33
|
+
claude mcp add okf -- uv --directory /caminho/mcp-okf run mcp-okf
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Dados em `~/.config/mcp-okf/` (`%APPDATA%\mcp-okf` no Windows; ou `MCP_OKF_HOME`): `qdrant/` e
|
|
37
|
+
`state/<sha1(root)>.json` (hash, links e enriquecimento de cada documento). Nada é gravado no bundle.
|
|
38
|
+
O Qdrant local trava a pasta: só um processo do mcp-okf por vez.
|
|
39
|
+
|
|
40
|
+
## Tools
|
|
41
|
+
|
|
42
|
+
`root` é sempre o caminho absoluto do bundle.
|
|
43
|
+
|
|
44
|
+
| Tool | Parâmetros | Devolve |
|
|
45
|
+
|---|---|---|
|
|
46
|
+
| `okf_index` | `root`, `limit=200`, `force=False` | `{documentos, indexados, removidos, restantes, a_enriquecer}`; incremental por sha256; chame até `restantes=0` |
|
|
47
|
+
| `okf_enrich_next` | `root`, `limit=5` | `{restantes, instrucoes, documentos: [{path, id, type, title, description, body, links}]}` |
|
|
48
|
+
| `okf_save_enrichment` | `root`, `items: [{path, summary, keywords, entities, questions}]` | `{gravados, restantes}` |
|
|
49
|
+
| `okf_search` | `root`, `query`, `limit=8`, `folder?`, `type?` | `[{path, id, title, type, folder, score, summary, trechos: [{section, text}]}]` |
|
|
50
|
+
| `okf_list_documents` | `root`, `folder?`, `type?` | `[{path, id, title, type, folder, description, indexado, enriquecido}]` |
|
|
51
|
+
| `okf_get_document` | `root`, `ref` (path ou id) | `{path, markdown, enrichment, links: {saida, entrada, similares}}` |
|
|
52
|
+
|
|
53
|
+
Fluxo: `okf_index` até `restantes=0` → `okf_enrich_next` / `okf_save_enrichment` até `restantes=0` → perguntas
|
|
54
|
+
com `okf_search` e `okf_get_document`. Depois de um novo sincronismo da alm-sync, rode `okf_index` de novo: só o que
|
|
55
|
+
mudou é reindexado, e só o que mudou volta para a fila de enriquecimento.
|
|
56
|
+
|
|
57
|
+
## Como indexa
|
|
58
|
+
|
|
59
|
+
Cada documento vira pontos `chunk` (corpo dividido por heading `#`..`###`, seções longas por parágrafo; cada chunk
|
|
60
|
+
leva o prefixo `<tipo> <id> — <título> | <pasta> | <seção>`) e um ponto `doc` (cabeçalho + enriquecimento). A busca
|
|
61
|
+
funde o ranking denso e o BM25 (RRF) e agrupa por documento.
|
|
62
|
+
|
|
63
|
+
## Testes
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
uv run pytest
|
|
67
|
+
```
|
|
68
|
+
Os testes usam Qdrant em memória e embeddings falsos (não baixam modelo).
|
mcp_okf-0.1.0/README.md
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# mcp-okf
|
|
2
|
+
|
|
3
|
+
Servidor MCP que transforma um bundle OKF (pasta com `index.md` e um `.md` por documento, como o gerado pela
|
|
4
|
+
alm-sync do [mcp-alm](https://github.com/dxaraujo/mcp-alm)) numa base de conhecimento consultável em linguagem natural.
|
|
5
|
+
|
|
6
|
+
- **Índice vetorial:** Qdrant em modo local (pasta no disco, sem servidor), busca híbrida denso + BM25 com fusão RRF.
|
|
7
|
+
- **Embeddings:** fastembed local, `paraphrase-multilingual-mpnet-base-v2` (PT-BR); nada sai da máquina.
|
|
8
|
+
- **Enriquecimento:** feito pela própria LLM do cliente (sem chave de API): resumo, palavras-chave/sinônimos,
|
|
9
|
+
entidades e perguntas que o documento responde.
|
|
10
|
+
- **Links:** `links` do frontmatter (nomes do DOORS Next), links do corpo para outros arquivos do bundle (`cita`) e
|
|
11
|
+
URL do ALM de artefato que está no bundle; backlinks e documentos semanticamente similares na leitura.
|
|
12
|
+
|
|
13
|
+
## Instalação
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
uv sync
|
|
17
|
+
# baixa os modelos uma vez (~1 GB), para a 1ª indexação não estourar o timeout do cliente MCP
|
|
18
|
+
uv run python -c "from mcp_okf import store; store.embed_dense(['x']); store.embed_sparse(['x'])"
|
|
19
|
+
claude mcp add okf -- uv --directory /caminho/mcp-okf run mcp-okf
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
Dados em `~/.config/mcp-okf/` (`%APPDATA%\mcp-okf` no Windows; ou `MCP_OKF_HOME`): `qdrant/` e
|
|
23
|
+
`state/<sha1(root)>.json` (hash, links e enriquecimento de cada documento). Nada é gravado no bundle.
|
|
24
|
+
O Qdrant local trava a pasta: só um processo do mcp-okf por vez.
|
|
25
|
+
|
|
26
|
+
## Tools
|
|
27
|
+
|
|
28
|
+
`root` é sempre o caminho absoluto do bundle.
|
|
29
|
+
|
|
30
|
+
| Tool | Parâmetros | Devolve |
|
|
31
|
+
|---|---|---|
|
|
32
|
+
| `okf_index` | `root`, `limit=200`, `force=False` | `{documentos, indexados, removidos, restantes, a_enriquecer}`; incremental por sha256; chame até `restantes=0` |
|
|
33
|
+
| `okf_enrich_next` | `root`, `limit=5` | `{restantes, instrucoes, documentos: [{path, id, type, title, description, body, links}]}` |
|
|
34
|
+
| `okf_save_enrichment` | `root`, `items: [{path, summary, keywords, entities, questions}]` | `{gravados, restantes}` |
|
|
35
|
+
| `okf_search` | `root`, `query`, `limit=8`, `folder?`, `type?` | `[{path, id, title, type, folder, score, summary, trechos: [{section, text}]}]` |
|
|
36
|
+
| `okf_list_documents` | `root`, `folder?`, `type?` | `[{path, id, title, type, folder, description, indexado, enriquecido}]` |
|
|
37
|
+
| `okf_get_document` | `root`, `ref` (path ou id) | `{path, markdown, enrichment, links: {saida, entrada, similares}}` |
|
|
38
|
+
|
|
39
|
+
Fluxo: `okf_index` até `restantes=0` → `okf_enrich_next` / `okf_save_enrichment` até `restantes=0` → perguntas
|
|
40
|
+
com `okf_search` e `okf_get_document`. Depois de um novo sincronismo da alm-sync, rode `okf_index` de novo: só o que
|
|
41
|
+
mudou é reindexado, e só o que mudou volta para a fila de enriquecimento.
|
|
42
|
+
|
|
43
|
+
## Como indexa
|
|
44
|
+
|
|
45
|
+
Cada documento vira pontos `chunk` (corpo dividido por heading `#`..`###`, seções longas por parágrafo; cada chunk
|
|
46
|
+
leva o prefixo `<tipo> <id> — <título> | <pasta> | <seção>`) e um ponto `doc` (cabeçalho + enriquecimento). A busca
|
|
47
|
+
funde o ranking denso e o BM25 (RRF) e agrupa por documento.
|
|
48
|
+
|
|
49
|
+
## Testes
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
uv run pytest
|
|
53
|
+
```
|
|
54
|
+
Os testes usam Qdrant em memória e embeddings falsos (não baixam modelo).
|
|
File without changes
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
"""Leitura de um bundle OKF (alm-sync do mcp-alm): documentos, frontmatter, links entre eles e chunks.
|
|
2
|
+
Só disco e texto; o índice fica em store.py."""
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import os
|
|
7
|
+
import posixpath
|
|
8
|
+
import re
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from urllib.parse import unquote
|
|
11
|
+
|
|
12
|
+
import yaml
|
|
13
|
+
from markdown_it import MarkdownIt
|
|
14
|
+
|
|
15
|
+
SKIP = {"index.md", "sync.md"} # arquivos do próprio bundle, não documentos
|
|
16
|
+
CHUNK_MAX = 1500
|
|
17
|
+
_HEADING = re.compile(r"^(#{1,3})\s+(.+?)\s*#*\s*$", re.M)
|
|
18
|
+
_LEADING_ID = re.compile(r"(\d+)(?:[ _:-].*)?")
|
|
19
|
+
_LINK_ID = re.compile(r"\s*(\d+)\s*:") # item de `links`: 'id: título'
|
|
20
|
+
_MD = MarkdownIt()
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def root(dest: str) -> str:
|
|
24
|
+
"""Raiz do bundle: `dest` tem de ser absoluto (o cwd do MCP não é o bundle)."""
|
|
25
|
+
if not os.path.isabs(dest):
|
|
26
|
+
raise ValueError(f"root precisa ser caminho absoluto: {dest}")
|
|
27
|
+
path = os.path.realpath(dest)
|
|
28
|
+
if not os.path.isdir(path):
|
|
29
|
+
raise ValueError(f"pasta não encontrada: {dest}")
|
|
30
|
+
return path
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def read_document(text: str) -> tuple[dict, str]:
|
|
34
|
+
"""(frontmatter YAML completo, corpo). Sem cabeçalho ou YAML inválido: ({}, texto)."""
|
|
35
|
+
if not text.startswith("---\n") or (end := text.find("\n---\n", 3)) < 0:
|
|
36
|
+
return {}, text
|
|
37
|
+
try:
|
|
38
|
+
head = yaml.safe_load(text[4:end]) or {}
|
|
39
|
+
except yaml.YAMLError:
|
|
40
|
+
return {}, text
|
|
41
|
+
return (head, text[end + 5:]) if isinstance(head, dict) else ({}, text)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _text(value) -> str | None:
|
|
45
|
+
return " ".join(str(value).split()) if value not in (None, "") else None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def load(root_dir: str, path: str) -> dict:
|
|
49
|
+
"""Documento pelo caminho relativo (posix) ao root."""
|
|
50
|
+
full = os.path.join(root_dir, *path.split("/"))
|
|
51
|
+
with open(full, "rb") as f:
|
|
52
|
+
raw = f.read()
|
|
53
|
+
head, body = read_document(raw.decode("utf-8"))
|
|
54
|
+
name = posixpath.basename(path)[:-3]
|
|
55
|
+
m = _LEADING_ID.fullmatch(name)
|
|
56
|
+
doc_id = head.get("id") if head.get("id") is not None else (m.group(1) if m else None)
|
|
57
|
+
sources = head.get("sources") if isinstance(head.get("sources"), list) else []
|
|
58
|
+
tags = head.get("tags") if isinstance(head.get("tags"), list) else []
|
|
59
|
+
return {"path": path, "id": str(doc_id) if doc_id is not None else None,
|
|
60
|
+
"title": _text(head.get("title")) or name, "type": _text(head.get("type")),
|
|
61
|
+
"folder": path.split("/")[0] if "/" in path else "", "description": _text(head.get("description")),
|
|
62
|
+
"tags": [str(t) for t in tags], "resource": _text(head.get("resource")),
|
|
63
|
+
"last_modified": next((str(s["last_modified"]) for s in sources
|
|
64
|
+
if isinstance(s, dict) and s.get("last_modified")), None),
|
|
65
|
+
"head": head, "body": body, "hash": hashlib.sha256(raw).hexdigest()}
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def scan(root_dir: str) -> dict[str, dict]:
|
|
69
|
+
"""Todos os .md do bundle (menos index.md/sync.md na raiz), por caminho relativo posix."""
|
|
70
|
+
docs = {}
|
|
71
|
+
for full in sorted(Path(root_dir).rglob("*.md")):
|
|
72
|
+
rel = full.relative_to(root_dir).as_posix()
|
|
73
|
+
if rel in SKIP or any(part.startswith(".") for part in full.relative_to(root_dir).parts):
|
|
74
|
+
continue
|
|
75
|
+
docs[rel] = load(root_dir, rel)
|
|
76
|
+
return docs
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _targets(markdown: str) -> list[str]:
|
|
80
|
+
"""Alvos de link do corpo (markdown-it, inclusive '<alvo com espaço>')."""
|
|
81
|
+
out = []
|
|
82
|
+
for token in _MD.parse(markdown):
|
|
83
|
+
for child in token.children or []:
|
|
84
|
+
if child.type == "link_open" and (href := child.attrGet("href")):
|
|
85
|
+
out.append(str(href))
|
|
86
|
+
return out
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def resolve(doc_path: str, target: str, by_id: dict[str, str], docs: dict) -> str | None:
|
|
90
|
+
"""Alvo de link -> caminho de documento do bundle; None se estiver fora dele."""
|
|
91
|
+
target = unquote(target).split("#")[0].strip()
|
|
92
|
+
if not target:
|
|
93
|
+
return None
|
|
94
|
+
if "/rm/resources/" in target: # URL do ALM: só liga se o artefato estiver no bundle
|
|
95
|
+
return next((p for p, d in docs.items() if d["resource"] == target.split("?")[0]), None)
|
|
96
|
+
if re.match(r"[a-z][a-z0-9+.-]*:", target, re.I):
|
|
97
|
+
return None
|
|
98
|
+
base = "" if target.startswith("/") else posixpath.dirname(doc_path)
|
|
99
|
+
path = posixpath.normpath(posixpath.join(base, target.lstrip("/")))
|
|
100
|
+
if path.startswith("..") or path.startswith("/"):
|
|
101
|
+
return None # fora do bundle
|
|
102
|
+
if path in docs:
|
|
103
|
+
return path
|
|
104
|
+
m = _LEADING_ID.fullmatch(posixpath.basename(path).removesuffix(".md"))
|
|
105
|
+
return by_id.get(m.group(1)) if m else None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def links(doc: dict, docs: dict, by_id: dict[str, str]) -> list[dict]:
|
|
109
|
+
"""[{rel, path}] sem repetição: do frontmatter `links {rel: ['id: título']}` e do corpo (rel 'cita')."""
|
|
110
|
+
out: dict[tuple[str, str], dict] = {}
|
|
111
|
+
raw = doc["head"].get("links")
|
|
112
|
+
for rel, items in (raw.items() if isinstance(raw, dict) else []):
|
|
113
|
+
for item in items if isinstance(items, list) else [items]:
|
|
114
|
+
m = _LINK_ID.match(str(item))
|
|
115
|
+
if m and (path := by_id.get(m.group(1))) and path != doc["path"]:
|
|
116
|
+
out[(str(rel), path)] = {"rel": str(rel), "path": path}
|
|
117
|
+
for target in _targets(doc["body"]):
|
|
118
|
+
if (path := resolve(doc["path"], target, by_id, docs)) and path != doc["path"]:
|
|
119
|
+
out.setdefault(("cita", path), {"rel": "cita", "path": path})
|
|
120
|
+
return list(out.values())
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def by_id(docs: dict) -> dict[str, str]:
|
|
124
|
+
return {d["id"]: p for p, d in docs.items() if d["id"]}
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _split(text: str) -> list[str]:
|
|
128
|
+
"""Quebra por parágrafo em pedaços de até CHUNK_MAX (parágrafo maior que isso vai inteiro)."""
|
|
129
|
+
parts, cur = [], ""
|
|
130
|
+
for para in re.split(r"\n\s*\n", text):
|
|
131
|
+
if cur and len(cur) + len(para) > CHUNK_MAX:
|
|
132
|
+
parts.append(cur)
|
|
133
|
+
cur = ""
|
|
134
|
+
cur = f"{cur}\n\n{para}" if cur else para
|
|
135
|
+
return parts + ([cur] if cur.strip() else [])
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def chunks(doc: dict) -> list[dict]:
|
|
139
|
+
"""[{section, text}] por heading (#..###); `text` já com o prefixo de contexto para o embedding."""
|
|
140
|
+
body, sections = doc["body"], []
|
|
141
|
+
marks = list(_HEADING.finditer(body))
|
|
142
|
+
starts = [(0, "")] + [(m.start(), m.group(2)) for m in marks]
|
|
143
|
+
for i, (start, title) in enumerate(starts):
|
|
144
|
+
end = marks[i].start() if i < len(marks) else len(body)
|
|
145
|
+
text = body[start:end].strip()
|
|
146
|
+
if text:
|
|
147
|
+
sections += [(title, part.strip()) for part in _split(text) if part.strip()]
|
|
148
|
+
prefix = " ".join(filter(None, [doc["type"], doc["id"], "—", doc["title"]]))
|
|
149
|
+
return [{"section": s, "text": f"{prefix} | {doc['folder']} | {s}\n{t}"} for s, t in sections] or \
|
|
150
|
+
[{"section": "", "text": f"{prefix} | {doc['folder']}\n{doc['description'] or doc['title']}"}]
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""Servidor MCP de base de conhecimento sobre bundles OKF (alm-sync do mcp-alm).
|
|
2
|
+
|
|
3
|
+
As tools ficam em tools.py, registradas com `@tool`; leitura do bundle em bundle.py e índice em store.py.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import functools
|
|
8
|
+
|
|
9
|
+
from mcp.server.mcpserver import MCPServer
|
|
10
|
+
from mcp.server.mcpserver.exceptions import ToolError
|
|
11
|
+
|
|
12
|
+
mcp = MCPServer(
|
|
13
|
+
"okf",
|
|
14
|
+
instructions=(
|
|
15
|
+
"Base de conhecimento sobre um bundle OKF (pasta com index.md e um .md por documento, como o gerado pela "
|
|
16
|
+
"alm-sync). `root` é sempre o caminho absoluto da pasta. Preparar: okf_index até restantes=0; depois "
|
|
17
|
+
"okf_enrich_next -> gere o enriquecimento de cada documento -> okf_save_enrichment, até restantes=0. "
|
|
18
|
+
"Consultar: okf_search (pergunta em linguagem natural), okf_get_document (documento inteiro + links de "
|
|
19
|
+
"saída, de entrada e similares), okf_list_documents (inventário). A busca funciona sem enriquecimento, mas "
|
|
20
|
+
"fica melhor com ele. Cite os documentos pelo path/id."
|
|
21
|
+
),
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
# RuntimeError: Qdrant travado por outro processo; OSError: disco; ValueError/LookupError: parâmetro inválido
|
|
25
|
+
EXPECTED = (RuntimeError, OSError, LookupError, ValueError)
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def tool(fn):
|
|
29
|
+
"""Registra `fn` como tool e devolve `fn` intacta (testes veem as exceções originais).
|
|
30
|
+
|
|
31
|
+
O MCP SDK 2.x esconde o texto de exceções que não são ToolError."""
|
|
32
|
+
@functools.wraps(fn)
|
|
33
|
+
def wrapper(*args, **kwargs):
|
|
34
|
+
try:
|
|
35
|
+
return fn(*args, **kwargs)
|
|
36
|
+
except EXPECTED as exc:
|
|
37
|
+
raise ToolError(str(exc)) from exc
|
|
38
|
+
|
|
39
|
+
mcp.tool()(wrapper)
|
|
40
|
+
return fn
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
# importado depois de `tool` existir: o módulo registra as tools ao ser carregado
|
|
44
|
+
from . import tools # noqa: E402,F401
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def main() -> None:
|
|
48
|
+
mcp.run()
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
if __name__ == "__main__":
|
|
52
|
+
main()
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Índice do bundle: Qdrant local (vetor denso + BM25, fusão RRF) e o estado por bundle (hash, links, enriquecimento).
|
|
2
|
+
|
|
3
|
+
Estado em <home>/state/<sha1(root)>.json: {path: {hash, indexed_hash, links: [{rel, path}], enrichment}}.
|
|
4
|
+
É o que torna a indexação incremental e mantém o enriquecimento entre reindexações. Nada é gravado no bundle."""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
import functools
|
|
8
|
+
import hashlib
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
import sys
|
|
12
|
+
import threading
|
|
13
|
+
import uuid
|
|
14
|
+
from datetime import datetime, timezone
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
from qdrant_client import QdrantClient, models
|
|
18
|
+
|
|
19
|
+
from . import bundle
|
|
20
|
+
|
|
21
|
+
COLLECTION = "okf"
|
|
22
|
+
DENSE_MODEL = "sentence-transformers/paraphrase-multilingual-mpnet-base-v2"
|
|
23
|
+
DENSE_DIM = 768
|
|
24
|
+
SPARSE_MODEL = "Qdrant/bm25"
|
|
25
|
+
_NS = uuid.UUID("5b0f0c6e-2a51-4f43-9d55-0c2c7f1e9a10")
|
|
26
|
+
# um lock para tudo: tools síncronas podem rodar em threads e o estado é um JSON
|
|
27
|
+
LOCK = threading.RLock()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def home() -> Path:
|
|
31
|
+
if env := os.environ.get("MCP_OKF_HOME"):
|
|
32
|
+
return Path(env)
|
|
33
|
+
if sys.platform == "win32":
|
|
34
|
+
return Path(os.environ["APPDATA"]) / "mcp-okf"
|
|
35
|
+
return Path.home() / ".config" / "mcp-okf"
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@functools.cache
|
|
39
|
+
def client() -> QdrantClient:
|
|
40
|
+
"""Qdrant em modo local (a pasta fica travada para um processo só)."""
|
|
41
|
+
path = home() / "qdrant"
|
|
42
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
43
|
+
try:
|
|
44
|
+
qc = QdrantClient(path=str(path))
|
|
45
|
+
except RuntimeError as exc:
|
|
46
|
+
raise RuntimeError(f"Índice em {path} já está aberto por outro processo do mcp-okf: {exc}") from exc
|
|
47
|
+
_ensure(qc)
|
|
48
|
+
return qc
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _ensure(qc: QdrantClient) -> None:
|
|
52
|
+
if qc.collection_exists(COLLECTION):
|
|
53
|
+
return
|
|
54
|
+
qc.create_collection(
|
|
55
|
+
COLLECTION,
|
|
56
|
+
vectors_config={"dense": models.VectorParams(size=DENSE_DIM, distance=models.Distance.COSINE)},
|
|
57
|
+
sparse_vectors_config={"bm25": models.SparseVectorParams(modifier=models.Modifier.IDF)})
|
|
58
|
+
# shortcut: sem índices de payload (o Qdrant local os ignora); criar em bundle/path/folder/type/kind ao migrar
|
|
59
|
+
# para servidor Qdrant
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@functools.cache
|
|
63
|
+
def _dense():
|
|
64
|
+
from fastembed import TextEmbedding
|
|
65
|
+
return TextEmbedding(DENSE_MODEL)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@functools.cache
|
|
69
|
+
def _sparse():
|
|
70
|
+
from fastembed import SparseTextEmbedding
|
|
71
|
+
return SparseTextEmbedding(SPARSE_MODEL, language="portuguese")
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def embed_dense(texts: list[str]) -> list[list[float]]:
|
|
75
|
+
return [v.tolist() for v in _dense().embed(texts)]
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def embed_sparse(texts: list[str], query: bool = False) -> list[models.SparseVector]:
|
|
79
|
+
fn = _sparse().query_embed if query else _sparse().embed
|
|
80
|
+
return [models.SparseVector(indices=e.indices.tolist(), values=e.values.tolist()) for e in fn(texts)]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
# --- estado
|
|
84
|
+
|
|
85
|
+
def _state_file(root: str) -> Path:
|
|
86
|
+
return home() / "state" / f"{hashlib.sha1(root.encode()).hexdigest()}.json"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def load_state(root: str) -> dict:
|
|
90
|
+
try:
|
|
91
|
+
return json.loads(_state_file(root).read_text(encoding="utf-8"))
|
|
92
|
+
except FileNotFoundError:
|
|
93
|
+
return {}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def save_state(root: str, state: dict) -> None:
|
|
97
|
+
path = _state_file(root)
|
|
98
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
99
|
+
tmp = path.with_suffix(".tmp")
|
|
100
|
+
tmp.write_text(json.dumps(state, ensure_ascii=False, indent=1), encoding="utf-8")
|
|
101
|
+
os.replace(tmp, path) # atômico: estado nunca fica pela metade
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def enriched(entry: dict) -> bool:
|
|
105
|
+
return entry.get("enrichment", {}).get("hash") == entry.get("hash")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def indexed(entry: dict) -> bool:
|
|
109
|
+
return entry.get("indexed_hash") == entry.get("hash")
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
# --- pontos
|
|
113
|
+
|
|
114
|
+
def _pid(root: str, path: str, n) -> str:
|
|
115
|
+
return str(uuid.uuid5(_NS, f"{root}\n{path}\n{n}"))
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _filter(root: str, **eq) -> models.Filter:
|
|
119
|
+
return models.Filter(must=[models.FieldCondition(key=k, match=models.MatchValue(value=v))
|
|
120
|
+
for k, v in {"bundle": root, **eq}.items() if v is not None])
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _payload(root: str, doc: dict, kind: str, section: str = "", text: str = "") -> dict:
|
|
124
|
+
return {"bundle": root, "path": doc["path"], "id": doc["id"], "title": doc["title"], "type": doc["type"],
|
|
125
|
+
"folder": doc["folder"], "kind": kind, "section": section, "text": text}
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def doc_text(doc: dict, enrichment: dict | None) -> str:
|
|
129
|
+
"""Texto do ponto 'doc': cabeçalho + enriquecimento (é ele que casa pergunta em linguagem natural com jargão)."""
|
|
130
|
+
e = enrichment or {}
|
|
131
|
+
parts = [" ".join(filter(None, [doc["type"], doc["id"], "—", doc["title"]])), doc["folder"],
|
|
132
|
+
doc["description"], e.get("summary"), ", ".join(e.get("keywords", [])),
|
|
133
|
+
", ".join(e.get("entities", [])), "\n".join(e.get("questions", []))]
|
|
134
|
+
return "\n".join(p for p in parts if p)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def _doc_point(root: str, doc: dict, enrichment: dict | None) -> models.PointStruct:
|
|
138
|
+
text = doc_text(doc, enrichment)
|
|
139
|
+
return models.PointStruct(id=_pid(root, doc["path"], "doc"),
|
|
140
|
+
vector={"dense": embed_dense([text])[0], "bm25": embed_sparse([text])[0]},
|
|
141
|
+
payload=_payload(root, doc, "doc", text=(enrichment or {}).get("summary") or ""))
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _delete(root: str, path: str) -> None:
|
|
145
|
+
client().delete(COLLECTION, points_selector=models.FilterSelector(filter=_filter(root, path=path)))
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def write_doc(root: str, doc: dict, enrichment: dict | None) -> None:
|
|
149
|
+
"""Apaga os pontos do documento e grava os chunks + o ponto 'doc'."""
|
|
150
|
+
parts = bundle.chunks(doc)
|
|
151
|
+
texts = [c["text"] for c in parts]
|
|
152
|
+
dense, sparse = embed_dense(texts), embed_sparse(texts)
|
|
153
|
+
points = [models.PointStruct(id=_pid(root, doc["path"], n), vector={"dense": d, "bm25": s},
|
|
154
|
+
payload=_payload(root, doc, "chunk", c["section"], c["text"].split("\n", 1)[-1]))
|
|
155
|
+
for n, (c, d, s) in enumerate(zip(parts, dense, sparse))]
|
|
156
|
+
_delete(root, doc["path"])
|
|
157
|
+
client().upsert(COLLECTION, points + [_doc_point(root, doc, enrichment)])
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def write_doc_point(root: str, doc: dict, enrichment: dict | None) -> None:
|
|
161
|
+
client().upsert(COLLECTION, [_doc_point(root, doc, enrichment)])
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def delete_doc(root: str, path: str) -> None:
|
|
165
|
+
_delete(root, path)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def search(root: str, query: str, limit: int, folder: str | None, type_: str | None) -> list[models.ScoredPoint]:
|
|
169
|
+
flt = _filter(root, folder=folder, type=type_)
|
|
170
|
+
wide = limit * 5
|
|
171
|
+
return client().query_points(
|
|
172
|
+
COLLECTION,
|
|
173
|
+
prefetch=[models.Prefetch(query=embed_dense([query])[0], using="dense", filter=flt, limit=wide),
|
|
174
|
+
models.Prefetch(query=embed_sparse([query], query=True)[0], using="bm25", filter=flt, limit=wide)],
|
|
175
|
+
query=models.FusionQuery(fusion=models.Fusion.RRF), limit=wide, with_payload=True).points
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def similar(root: str, path: str, limit: int) -> list[models.ScoredPoint]:
|
|
179
|
+
"""Documentos mais próximos pelo vetor do ponto 'doc' (vazio se o documento não estiver indexado)."""
|
|
180
|
+
found = client().retrieve(COLLECTION, [_pid(root, path, "doc")], with_vectors=["dense"])
|
|
181
|
+
if not found:
|
|
182
|
+
return []
|
|
183
|
+
flt = _filter(root, kind="doc")
|
|
184
|
+
flt.must_not = [models.FieldCondition(key="path", match=models.MatchValue(value=path))]
|
|
185
|
+
return client().query_points(COLLECTION, query=found[0].vector["dense"], using="dense", query_filter=flt,
|
|
186
|
+
limit=limit, with_payload=True).points
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def now() -> str:
|
|
190
|
+
return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
@@ -0,0 +1,185 @@
|
|
|
1
|
+
"""Tools do mcp-okf: indexação, enriquecimento (feito pela LLM do cliente), busca e leitura."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from typing import TypedDict
|
|
5
|
+
|
|
6
|
+
from . import bundle, store
|
|
7
|
+
from .server import tool
|
|
8
|
+
|
|
9
|
+
BODY_MAX = 12000 # corpo devolvido para enriquecer: limita o contexto gasto por documento
|
|
10
|
+
SNIPPET_MAX = 800
|
|
11
|
+
LIST_MAX, ITEM_MAX, SUMMARY_MAX = 30, 300, 2000
|
|
12
|
+
INSTRUCOES = (
|
|
13
|
+
"Para cada documento, gere em PT-BR e chame okf_save_enrichment com items=[{path, summary, keywords, entities, "
|
|
14
|
+
"questions}]: summary = 2-3 frases do que o documento define; keywords = termos de negócio e sinônimos que uma "
|
|
15
|
+
"pessoa usaria para procurá-lo (inclusive os que o texto não usa literalmente); entities = atores, sistemas, "
|
|
16
|
+
"regras e dados citados; questions = 3-5 perguntas que o documento responde. Use só o que está no documento e "
|
|
17
|
+
"nos links; não invente.")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Enrichment(TypedDict):
|
|
21
|
+
path: str
|
|
22
|
+
summary: str
|
|
23
|
+
keywords: list[str]
|
|
24
|
+
entities: list[str]
|
|
25
|
+
questions: list[str]
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _ref(entry: dict, path: str) -> dict:
|
|
29
|
+
return {"path": path, "id": entry.get("id"), "title": entry.get("title")}
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _pending_enrichment(state: dict) -> list[str]:
|
|
33
|
+
return [p for p, e in state.items() if store.indexed(e) and not store.enriched(e)]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@tool
|
|
37
|
+
def okf_index(root: str, limit: int = 200, force: bool = False) -> dict:
|
|
38
|
+
"""Indexa o bundle OKF em `root` (caminho absoluto): lê todos os .md, recalcula os links entre documentos, remove
|
|
39
|
+
do índice os arquivos apagados e (re)indexa até `limit` documentos novos ou alterados (sha256 diferente).
|
|
40
|
+
Chame até `restantes` = 0. `force=True` (só na 1ª chamada) reindexa tudo, mantendo o enriquecimento.
|
|
41
|
+
Devolve {documentos, indexados, removidos, restantes, a_enriquecer}."""
|
|
42
|
+
with store.LOCK:
|
|
43
|
+
rd = bundle.root(root)
|
|
44
|
+
docs = bundle.scan(rd)
|
|
45
|
+
ids = bundle.by_id(docs)
|
|
46
|
+
state = store.load_state(rd)
|
|
47
|
+
removed = [p for p in state if p not in docs]
|
|
48
|
+
for p in removed:
|
|
49
|
+
store.delete_doc(rd, p)
|
|
50
|
+
del state[p]
|
|
51
|
+
for p, d in docs.items():
|
|
52
|
+
e = state.setdefault(p, {})
|
|
53
|
+
e.update(hash=d["hash"], id=d["id"], title=d["title"], type=d["type"], folder=d["folder"],
|
|
54
|
+
links=bundle.links(d, docs, ids))
|
|
55
|
+
if force:
|
|
56
|
+
e.pop("indexed_hash", None)
|
|
57
|
+
pending = [p for p in docs if not store.indexed(state[p])]
|
|
58
|
+
done = 0
|
|
59
|
+
try:
|
|
60
|
+
for p in pending[:limit]:
|
|
61
|
+
store.write_doc(rd, docs[p], state[p].get("enrichment")) # enriquecimento antigo vale até o novo
|
|
62
|
+
state[p]["indexed_hash"] = docs[p]["hash"]
|
|
63
|
+
done += 1
|
|
64
|
+
finally:
|
|
65
|
+
store.save_state(rd, state) # progresso salvo mesmo se um documento falhar
|
|
66
|
+
return {"documentos": len(docs), "indexados": done, "removidos": len(removed),
|
|
67
|
+
"restantes": len(pending) - done, "a_enriquecer": len(_pending_enrichment(state))}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
@tool
|
|
71
|
+
def okf_enrich_next(root: str, limit: int = 5) -> dict:
|
|
72
|
+
"""Próximos documentos indexados sem enriquecimento (ou alterados depois dele), para a LLM enriquecer e gravar
|
|
73
|
+
com okf_save_enrichment. Devolve {restantes, instrucoes, documentos: [{path, id, type, title, body, links}]}."""
|
|
74
|
+
with store.LOCK:
|
|
75
|
+
rd = bundle.root(root)
|
|
76
|
+
state = store.load_state(rd)
|
|
77
|
+
pending = _pending_enrichment(state)
|
|
78
|
+
out = []
|
|
79
|
+
for p in pending[:max(1, limit)]:
|
|
80
|
+
doc = bundle.load(rd, p)
|
|
81
|
+
out.append({"path": p, "id": doc["id"], "type": doc["type"], "title": doc["title"],
|
|
82
|
+
"description": doc["description"], "body": doc["body"][:BODY_MAX],
|
|
83
|
+
"links": [{"rel": l["rel"], **_ref(state.get(l["path"], {}), l["path"])}
|
|
84
|
+
for l in state[p].get("links", [])]})
|
|
85
|
+
return {"restantes": len(pending), "instrucoes": INSTRUCOES, "documentos": out}
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _strings(item: dict, key: str) -> list[str]:
|
|
89
|
+
value = item.get(key) or []
|
|
90
|
+
if not isinstance(value, list) or not all(isinstance(x, str) for x in value):
|
|
91
|
+
raise ValueError(f"{item.get('path')}: {key} deve ser lista de textos")
|
|
92
|
+
return [" ".join(x.split())[:ITEM_MAX] for x in value if x.strip()][:LIST_MAX]
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
@tool
|
|
96
|
+
def okf_save_enrichment(root: str, items: list[Enrichment]) -> dict:
|
|
97
|
+
"""Grava o enriquecimento gerado para os documentos de okf_enrich_next e reindexa o ponto de resumo de cada um.
|
|
98
|
+
items = [{path, summary, keywords[], entities[], questions[]}]. Devolve {gravados, restantes}."""
|
|
99
|
+
with store.LOCK:
|
|
100
|
+
rd = bundle.root(root)
|
|
101
|
+
state = store.load_state(rd)
|
|
102
|
+
valid = []
|
|
103
|
+
for item in items: # valida tudo antes de gravar qualquer coisa
|
|
104
|
+
if not isinstance(item, dict) or item.get("path") not in state:
|
|
105
|
+
path = item.get("path") if isinstance(item, dict) else item
|
|
106
|
+
raise ValueError(f"path fora do índice (rode okf_index): {path}")
|
|
107
|
+
summary = item.get("summary")
|
|
108
|
+
if not isinstance(summary, str) or not summary.strip():
|
|
109
|
+
raise ValueError(f"{item['path']}: summary obrigatório")
|
|
110
|
+
valid.append((item["path"], {"summary": " ".join(summary.split())[:SUMMARY_MAX],
|
|
111
|
+
**{k: _strings(item, k) for k in ("keywords", "entities", "questions")}}))
|
|
112
|
+
for path, enrichment in valid:
|
|
113
|
+
doc = bundle.load(rd, path)
|
|
114
|
+
entry = state[path]
|
|
115
|
+
entry["enrichment"] = {**enrichment, "hash": doc["hash"], "at": store.now()}
|
|
116
|
+
if store.indexed(entry) and entry["hash"] == doc["hash"]:
|
|
117
|
+
store.write_doc_point(rd, doc, entry["enrichment"])
|
|
118
|
+
store.save_state(rd, state)
|
|
119
|
+
return {"gravados": len(valid), "restantes": len(_pending_enrichment(state))}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
@tool
|
|
123
|
+
def okf_search(root: str, query: str, limit: int = 8, folder: str | None = None, type: str | None = None) -> list:
|
|
124
|
+
"""Busca semântica + palavra-chave (híbrida) no bundle indexado, em linguagem natural. Filtros opcionais: `folder`
|
|
125
|
+
(primeira pasta do path) e `type` (tipo do frontmatter). Devolve até `limit` documentos, do mais relevante:
|
|
126
|
+
[{path, id, title, type, folder, score, summary, trechos: [{section, text}]}]. Leia o documento inteiro com
|
|
127
|
+
okf_get_document quando o trecho não bastar."""
|
|
128
|
+
if not query.strip():
|
|
129
|
+
raise ValueError("query vazia")
|
|
130
|
+
with store.LOCK:
|
|
131
|
+
rd = bundle.root(root)
|
|
132
|
+
state = store.load_state(rd)
|
|
133
|
+
out: dict[str, dict] = {}
|
|
134
|
+
for hit in store.search(rd, query, max(1, limit), folder, type):
|
|
135
|
+
p = hit.payload
|
|
136
|
+
r = out.get(p["path"])
|
|
137
|
+
if r is None:
|
|
138
|
+
if len(out) >= limit:
|
|
139
|
+
continue
|
|
140
|
+
r = out[p["path"]] = {
|
|
141
|
+
"path": p["path"], "id": p["id"], "title": p["title"], "type": p["type"], "folder": p["folder"],
|
|
142
|
+
"score": round(hit.score, 4),
|
|
143
|
+
"summary": state.get(p["path"], {}).get("enrichment", {}).get("summary"), "trechos": []}
|
|
144
|
+
if p["kind"] == "chunk" and len(r["trechos"]) < 2:
|
|
145
|
+
r["trechos"].append({"section": p["section"], "text": p["text"][:SNIPPET_MAX]})
|
|
146
|
+
return list(out.values())
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
@tool
|
|
150
|
+
def okf_list_documents(root: str, folder: str | None = None, type: str | None = None) -> list:
|
|
151
|
+
"""Inventário do bundle (sem corpo): [{path, id, title, type, folder, description, indexado, enriquecido}]."""
|
|
152
|
+
with store.LOCK:
|
|
153
|
+
rd = bundle.root(root)
|
|
154
|
+
state = store.load_state(rd)
|
|
155
|
+
out = []
|
|
156
|
+
for p, d in bundle.scan(rd).items():
|
|
157
|
+
if (folder and d["folder"] != folder) or (type and d["type"] != type):
|
|
158
|
+
continue
|
|
159
|
+
e = {**state.get(p, {}), "hash": d["hash"]} # compara com o arquivo de agora
|
|
160
|
+
out.append({"path": p, "id": d["id"], "title": d["title"], "type": d["type"], "folder": d["folder"],
|
|
161
|
+
"description": d["description"], "indexado": store.indexed(e),
|
|
162
|
+
"enriquecido": store.enriched(e)})
|
|
163
|
+
return out
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
@tool
|
|
167
|
+
def okf_get_document(root: str, ref: str) -> dict:
|
|
168
|
+
"""Documento por path (relativo ao root) ou id: {path, markdown (arquivo inteiro), enrichment, links: {saida,
|
|
169
|
+
entrada, similares}}. saida/entrada = [{rel, path, id, title}] (rel = nome do link no DOORS Next ou 'cita' para
|
|
170
|
+
link no texto); similares = documentos mais próximos semanticamente [{path, id, title, score}]."""
|
|
171
|
+
with store.LOCK:
|
|
172
|
+
rd = bundle.root(root)
|
|
173
|
+
state = store.load_state(rd)
|
|
174
|
+
path = ref if ref in state else next((p for p, e in state.items() if e.get("id") == str(ref)), None)
|
|
175
|
+
if path is None:
|
|
176
|
+
raise LookupError(f"documento não encontrado no índice: {ref} (rode okf_index)")
|
|
177
|
+
with open(f"{rd}/{path}", encoding="utf-8") as f:
|
|
178
|
+
markdown = f.read()
|
|
179
|
+
entry = state[path]
|
|
180
|
+
return {"path": path, "markdown": markdown, "enrichment": entry.get("enrichment"), "links": {
|
|
181
|
+
"saida": [{"rel": l["rel"], **_ref(state.get(l["path"], {}), l["path"])} for l in entry.get("links", [])],
|
|
182
|
+
"entrada": [{"rel": l["rel"], **_ref(e, p)} for p, e in state.items()
|
|
183
|
+
for l in e.get("links", []) if l["path"] == path],
|
|
184
|
+
"similares": [{"path": h.payload["path"], "id": h.payload["id"], "title": h.payload["title"],
|
|
185
|
+
"score": round(h.score, 4)} for h in store.similar(rd, path, 5)]}}
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "mcp-okf"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "MCP server de base de conhecimento sobre bundles OKF (alm-sync): indexação, enriquecimento e busca semântica"
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = "MIT"
|
|
7
|
+
license-files = ["LICENSE"]
|
|
8
|
+
authors = [{ name = "Daniel Xavier Araújo", email = "danielxaraujo@gmail.com" }]
|
|
9
|
+
requires-python = ">=3.10"
|
|
10
|
+
dependencies = [
|
|
11
|
+
"mcp[cli]>=2.2",
|
|
12
|
+
"qdrant-client[fastembed]>=1.12",
|
|
13
|
+
"pyyaml>=6",
|
|
14
|
+
"markdown-it-py>=3",
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
[project.scripts]
|
|
18
|
+
mcp-okf = "mcp_okf.server:main"
|
|
19
|
+
|
|
20
|
+
[dependency-groups]
|
|
21
|
+
dev = ["pytest>=8"]
|
|
22
|
+
|
|
23
|
+
[build-system]
|
|
24
|
+
requires = ["hatchling"]
|
|
25
|
+
build-backend = "hatchling.build"
|
|
26
|
+
|
|
27
|
+
[tool.pytest.ini_options]
|
|
28
|
+
testpaths = ["tests"]
|
|
29
|
+
pythonpath = ["."]
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Bundle de exemplo copiado para tmp, Qdrant em memória e embeddings falsos (sem baixar modelo)."""
|
|
2
|
+
import re
|
|
3
|
+
import zlib
|
|
4
|
+
|
|
5
|
+
import pytest
|
|
6
|
+
from qdrant_client import QdrantClient, models
|
|
7
|
+
|
|
8
|
+
from mcp_okf import store
|
|
9
|
+
|
|
10
|
+
# bundle OKF no formato da alm-sync: {caminho relativo: conteúdo}
|
|
11
|
+
BUNDLE = {
|
|
12
|
+
'01-Requisitos/10-login-do-cliente.md': '---\ntype: Requisito\ntitle: Login do cliente\ndescription: "Acesso ao portal com CPF e senha."\nresource: "https://alm.example.com/rm/resources/TX_10"\ntags:\n - "01-Requisitos"\nsources:\n - id: doors-next\n resource: "https://alm.example.com/rm/resources/TX_10"\n last_modified: "2026-01-10T13:05:00Z"\nid: 10\n---\nO cliente acessa o portal informando CPF e senha. O CPF é validado conforme [2001 RN - Validar CPF do cliente](../03-Regras/2001-rn-validar-cpf-do-cliente.md).\n',
|
|
13
|
+
'03-Casos de Uso/2010-uc-cadastrar-cliente.md': '---\ntype: Caso de Uso\ntitle: UC - Cadastrar cliente\nresource: "https://alm.example.com/rm/resources/TX_2010"\ntags:\n - "03-Casos de Uso"\nsources:\n - id: doors-next\n resource: "https://alm.example.com/rm/resources/TX_2010"\n author: "human:ana.souza"\n last_modified: "2026-01-10T13:05:00Z"\ngenerated:\n by: "process:alm-mcp/1.0.14"\n at: "2026-01-11T19:20:00Z"\nid: 2010\nlinks:\n Vincular A:\n - "10: Login do cliente"\n - "9999: Fora do bundle"\n---\n## Fluxo Básico\n\n1. O usuário informa os dados do cliente.\n2. O sistema valida o documento: [2001 RN - Validar CPF do cliente](https://alm.example.com/rm/resources/TX_2001)\n3. Ver também [manual](https://example.com/manual.pdf) e [fora](../../segredo.md).\n',
|
|
14
|
+
'03-Regras/2001-rn-validar-cpf-do-cliente.md': '---\ntype: Regra de Negócio\ntitle: RN - Validar CPF do cliente\nresource: "https://alm.example.com/rm/resources/TX_2001"\ntags:\n - "03-Regras"\nid: 2001\n---\n## Regra\n\nO CPF deve ter 11 dígitos e dígitos verificadores válidos (módulo 11). CPFs com todos os dígitos iguais são rejeitados.\n\n## Mensagem\n\nExibir "CPF inválido" e bloquear o envio do formulário.\n',
|
|
15
|
+
'index.md': '---\nokf_version: "0.2"\n---\n\n# 01-Requisitos\n\n* [10 — Login do cliente](</01-Requisitos/10-login-do-cliente.md>) - Acesso ao portal com CPF e senha.\n\n# 03-Casos de Uso\n\n* [2010 — UC - Cadastrar cliente](</03-Casos de Uso/2010-uc-cadastrar-cliente.md>)\n\n# 03-Regras\n\n* [2001 — RN - Validar CPF do cliente](</03-Regras/2001-rn-validar-cpf-do-cliente.md>)\n\n# Bundle\n\n* [Sincronismo ALM → OKF](/sync.md) - Situação de cada artefato do DOORS Next baixado neste bundle.\n',
|
|
16
|
+
'sync.md': '---\ntype: Relatório de Sincronismo\ntitle: Sincronismo ALM → OKF\n---\n| Artefato | Pasta | Última atualização ALM | Hash | Status |\n|---|---|---|---|---|\n',
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _words(text: str) -> list[str]:
|
|
21
|
+
return re.findall(r"\w+", text.lower())
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def fake_dense(texts):
|
|
25
|
+
out = []
|
|
26
|
+
for t in texts:
|
|
27
|
+
v = [0.0] * store.DENSE_DIM
|
|
28
|
+
for w in _words(t):
|
|
29
|
+
v[zlib.crc32(w.encode()) % store.DENSE_DIM] += 1.0
|
|
30
|
+
out.append(v if any(v) else [1.0] + [0.0] * (store.DENSE_DIM - 1))
|
|
31
|
+
return out
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def fake_sparse(texts, query=False):
|
|
35
|
+
out = []
|
|
36
|
+
for t in texts:
|
|
37
|
+
idx = sorted({zlib.crc32(w.encode()) % 100_000 for w in _words(t)})
|
|
38
|
+
out.append(models.SparseVector(indices=idx, values=[1.0] * len(idx)))
|
|
39
|
+
return out
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@pytest.fixture
|
|
43
|
+
def root(tmp_path):
|
|
44
|
+
dest = tmp_path / "bundle"
|
|
45
|
+
for rel, text in BUNDLE.items():
|
|
46
|
+
(dest / rel).parent.mkdir(parents=True, exist_ok=True)
|
|
47
|
+
(dest / rel).write_text(text, encoding="utf-8")
|
|
48
|
+
return str(dest)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@pytest.fixture
|
|
52
|
+
def okf(tmp_path, monkeypatch):
|
|
53
|
+
monkeypatch.setenv("MCP_OKF_HOME", str(tmp_path / "home"))
|
|
54
|
+
qc = QdrantClient(":memory:")
|
|
55
|
+
store._ensure(qc)
|
|
56
|
+
monkeypatch.setattr(store, "client", lambda: qc)
|
|
57
|
+
monkeypatch.setattr(store, "embed_dense", fake_dense)
|
|
58
|
+
monkeypatch.setattr(store, "embed_sparse", fake_sparse)
|
|
59
|
+
from mcp_okf import tools
|
|
60
|
+
return tools
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
from mcp_okf import bundle
|
|
6
|
+
|
|
7
|
+
UC = "03-Casos de Uso/2010-uc-cadastrar-cliente.md"
|
|
8
|
+
RN = "03-Regras/2001-rn-validar-cpf-do-cliente.md"
|
|
9
|
+
LOGIN = "01-Requisitos/10-login-do-cliente.md"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def test_root_requires_absolute_existing_dir(root):
|
|
13
|
+
with pytest.raises(ValueError):
|
|
14
|
+
bundle.root("relativo/bundle")
|
|
15
|
+
with pytest.raises(ValueError):
|
|
16
|
+
bundle.root(os.path.join(root, "nao-existe"))
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def test_scan_reads_nested_frontmatter_and_skips_bundle_files(root):
|
|
20
|
+
docs = bundle.scan(root)
|
|
21
|
+
assert set(docs) == {UC, RN, LOGIN}
|
|
22
|
+
uc = docs[UC]
|
|
23
|
+
assert (uc["id"], uc["type"], uc["folder"]) == ("2010", "Caso de Uso", "03-Casos de Uso")
|
|
24
|
+
assert uc["head"]["links"] == {"Vincular A": ["10: Login do cliente", "9999: Fora do bundle"]}
|
|
25
|
+
assert uc["last_modified"] == "2026-01-10T13:05:00Z"
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def test_id_falls_back_to_file_name(root):
|
|
29
|
+
with open(os.path.join(root, "01-Requisitos", "77-sem-cabecalho.md"), "w") as f:
|
|
30
|
+
f.write("Texto sem frontmatter.\n")
|
|
31
|
+
doc = bundle.load(root, "01-Requisitos/77-sem-cabecalho.md")
|
|
32
|
+
assert (doc["id"], doc["title"], doc["head"]) == ("77", "77-sem-cabecalho", {})
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def test_links_from_frontmatter_body_and_alm_url(root):
|
|
36
|
+
docs = bundle.scan(root)
|
|
37
|
+
ids = bundle.by_id(docs)
|
|
38
|
+
assert bundle.links(docs[UC], docs, ids) == [{"rel": "Vincular A", "path": LOGIN}, {"rel": "cita", "path": RN}]
|
|
39
|
+
assert bundle.links(docs[LOGIN], docs, ids) == [{"rel": "cita", "path": RN}] # relativo '../03-Regras/...'
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def test_resolve_ignores_targets_outside_bundle(root):
|
|
43
|
+
docs = bundle.scan(root)
|
|
44
|
+
assert bundle.resolve(UC, "../../segredo.md", {}, docs) is None
|
|
45
|
+
assert bundle.resolve(UC, "https://example.com/x.md", {}, docs) is None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_chunks_by_heading_with_context_prefix(root):
|
|
49
|
+
parts = bundle.chunks(bundle.scan(root)[RN])
|
|
50
|
+
assert [c["section"] for c in parts] == ["Regra", "Mensagem"]
|
|
51
|
+
assert parts[0]["text"].startswith("Regra de Negócio 2001 — RN - Validar CPF do cliente | 03-Regras | Regra\n")
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_long_section_is_split_by_paragraph():
|
|
55
|
+
body = "\n\n".join("p" * 900 for _ in range(3))
|
|
56
|
+
doc = {"body": body, "type": None, "id": None, "title": "T", "folder": "", "description": None}
|
|
57
|
+
assert len(bundle.chunks(doc)) == 3
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
import os
|
|
2
|
+
|
|
3
|
+
import pytest
|
|
4
|
+
|
|
5
|
+
UC = "03-Casos de Uso/2010-uc-cadastrar-cliente.md"
|
|
6
|
+
RN = "03-Regras/2001-rn-validar-cpf-do-cliente.md"
|
|
7
|
+
LOGIN = "01-Requisitos/10-login-do-cliente.md"
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def test_index_is_incremental(okf, root):
|
|
11
|
+
assert okf.okf_index(root, limit=2) == {"documentos": 3, "indexados": 2, "removidos": 0, "restantes": 1,
|
|
12
|
+
"a_enriquecer": 2}
|
|
13
|
+
assert okf.okf_index(root)["indexados"] == 1
|
|
14
|
+
assert okf.okf_index(root)["indexados"] == 0
|
|
15
|
+
with open(os.path.join(root, RN), "a") as f:
|
|
16
|
+
f.write("\nNovo parágrafo.\n")
|
|
17
|
+
assert okf.okf_index(root)["indexados"] == 1
|
|
18
|
+
os.remove(os.path.join(root, LOGIN))
|
|
19
|
+
assert okf.okf_index(root)["removidos"] == 1
|
|
20
|
+
assert {d["path"] for d in okf.okf_list_documents(root)} == {UC, RN}
|
|
21
|
+
assert all(r["path"] != LOGIN for r in okf.okf_search(root, "login portal senha"))
|
|
22
|
+
assert okf.okf_index(root, force=True)["indexados"] == 2
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def test_enrichment_round_trip(okf, root):
|
|
26
|
+
okf.okf_index(root)
|
|
27
|
+
batch = okf.okf_enrich_next(root, limit=1)
|
|
28
|
+
assert batch["restantes"] == 3 and len(batch["documentos"]) == 1 and "summary" in batch["instrucoes"]
|
|
29
|
+
uc = okf.okf_enrich_next(root, limit=5)["documentos"]
|
|
30
|
+
assert {"rel": "Vincular A", "path": LOGIN, "id": "10", "title": "Login do cliente"} in \
|
|
31
|
+
next(d for d in uc if d["path"] == UC)["links"]
|
|
32
|
+
items = [{"path": d["path"], "summary": f"Resumo {d['title']}", "keywords": ["onboarding"], "entities": [],
|
|
33
|
+
"questions": []} for d in uc]
|
|
34
|
+
assert okf.okf_save_enrichment(root, items) == {"gravados": 3, "restantes": 0}
|
|
35
|
+
assert okf.okf_enrich_next(root)["documentos"] == []
|
|
36
|
+
assert okf.okf_index(root)["a_enriquecer"] == 0
|
|
37
|
+
# keyword que só existe no enriquecimento já acha o documento
|
|
38
|
+
assert okf.okf_search(root, "onboarding", limit=3)[0]["summary"].startswith("Resumo")
|
|
39
|
+
with open(os.path.join(root, RN), "a") as f:
|
|
40
|
+
f.write("\nMudou.\n")
|
|
41
|
+
assert okf.okf_index(root)["a_enriquecer"] == 1
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def test_save_enrichment_validates_before_writing(okf, root):
|
|
45
|
+
okf.okf_index(root)
|
|
46
|
+
ok = {"path": RN, "summary": "x", "keywords": [], "entities": [], "questions": []}
|
|
47
|
+
with pytest.raises(ValueError):
|
|
48
|
+
okf.okf_save_enrichment(root, [ok, {**ok, "path": "../fora.md"}])
|
|
49
|
+
with pytest.raises(ValueError):
|
|
50
|
+
okf.okf_save_enrichment(root, [{**ok, "keywords": "não é lista"}])
|
|
51
|
+
assert okf.okf_enrich_next(root)["restantes"] == 3
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def test_search_and_filters(okf, root):
|
|
55
|
+
okf.okf_index(root)
|
|
56
|
+
hits = okf.okf_search(root, "CPF dígitos verificadores módulo 11")
|
|
57
|
+
assert hits[0]["path"] == RN and hits[0]["trechos"][0]["section"] == "Regra"
|
|
58
|
+
assert {h["folder"] for h in okf.okf_search(root, "CPF", folder="01-Requisitos")} == {"01-Requisitos"}
|
|
59
|
+
assert okf.okf_search(root, "CPF", type="Caso de Uso")[0]["path"] == UC
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_get_document_links_backlinks_and_similar(okf, root):
|
|
63
|
+
okf.okf_index(root)
|
|
64
|
+
doc = okf.okf_get_document(root, "2001")
|
|
65
|
+
assert doc["path"] == RN and doc["markdown"].startswith("---\ntype: Regra de Negócio")
|
|
66
|
+
assert {(l["rel"], l["path"]) for l in doc["links"]["entrada"]} == {("cita", UC), ("cita", LOGIN)}
|
|
67
|
+
assert doc["links"]["saida"] == []
|
|
68
|
+
assert RN not in {s["path"] for s in doc["links"]["similares"]} and doc["links"]["similares"]
|
|
69
|
+
assert okf.okf_get_document(root, UC)["links"]["saida"][0] == {
|
|
70
|
+
"rel": "Vincular A", "path": LOGIN, "id": "10", "title": "Login do cliente"}
|
|
71
|
+
with pytest.raises(LookupError):
|
|
72
|
+
okf.okf_get_document(root, "../../etc/passwd")
|