mcp-okf 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,8 @@
1
+ __pycache__/
2
+ *.pyc
3
+ .venv/
4
+ .pytest_cache/
5
+ .superpowers/
6
+ dist/
7
+ .DS_Store
8
+ uv.lock
mcp_okf-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Daniel Xavier Araújo
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
mcp_okf-0.1.0/PKG-INFO ADDED
@@ -0,0 +1,68 @@
1
+ Metadata-Version: 2.5
2
+ Name: mcp-okf
3
+ Version: 0.1.0
4
+ Summary: MCP server de base de conhecimento sobre bundles OKF (alm-sync): indexação, enriquecimento e busca semântica
5
+ Author-email: Daniel Xavier Araújo <danielxaraujo@gmail.com>
6
+ License-Expression: MIT
7
+ License-File: LICENSE
8
+ Requires-Python: >=3.10
9
+ Requires-Dist: markdown-it-py>=3
10
+ Requires-Dist: mcp[cli]>=2.2
11
+ Requires-Dist: pyyaml>=6
12
+ Requires-Dist: qdrant-client[fastembed]>=1.12
13
+ Description-Content-Type: text/markdown
14
+
15
+ # mcp-okf
16
+
17
+ Servidor MCP que transforma um bundle OKF (pasta com `index.md` e um `.md` por documento, como o gerado pela
18
+ alm-sync do [mcp-alm](https://github.com/dxaraujo/mcp-alm)) numa base de conhecimento consultável em linguagem natural.
19
+
20
+ - **Índice vetorial:** Qdrant em modo local (pasta no disco, sem servidor), busca híbrida denso + BM25 com fusão RRF.
21
+ - **Embeddings:** fastembed local, `paraphrase-multilingual-mpnet-base-v2` (PT-BR); nada sai da máquina.
22
+ - **Enriquecimento:** feito pela própria LLM do cliente (sem chave de API): resumo, palavras-chave/sinônimos,
23
+ entidades e perguntas que o documento responde.
24
+ - **Links:** `links` do frontmatter (nomes do DOORS Next), links do corpo para outros arquivos do bundle (`cita`) e
25
+ URL do ALM de artefato que está no bundle; backlinks e documentos semanticamente similares na leitura.
26
+
27
+ ## Instalação
28
+
29
+ ```bash
30
+ uv sync
31
+ # baixa os modelos uma vez (~1 GB), para a 1ª indexação não estourar o timeout do cliente MCP
32
+ uv run python -c "from mcp_okf import store; store.embed_dense(['x']); store.embed_sparse(['x'])"
33
+ claude mcp add okf -- uv --directory /caminho/mcp-okf run mcp-okf
34
+ ```
35
+
36
+ Dados em `~/.config/mcp-okf/` (`%APPDATA%\mcp-okf` no Windows; ou `MCP_OKF_HOME`): `qdrant/` e
37
+ `state/<sha1(root)>.json` (hash, links e enriquecimento de cada documento). Nada é gravado no bundle.
38
+ O Qdrant local trava a pasta: só um processo do mcp-okf por vez.
39
+
40
+ ## Tools
41
+
42
+ `root` é sempre o caminho absoluto do bundle.
43
+
44
+ | Tool | Parâmetros | Devolve |
45
+ |---|---|---|
46
+ | `okf_index` | `root`, `limit=200`, `force=False` | `{documentos, indexados, removidos, restantes, a_enriquecer}`; incremental por sha256; chame até `restantes=0` |
47
+ | `okf_enrich_next` | `root`, `limit=5` | `{restantes, instrucoes, documentos: [{path, id, type, title, description, body, links}]}` |
48
+ | `okf_save_enrichment` | `root`, `items: [{path, summary, keywords, entities, questions}]` | `{gravados, restantes}` |
49
+ | `okf_search` | `root`, `query`, `limit=8`, `folder?`, `type?` | `[{path, id, title, type, folder, score, summary, trechos: [{section, text}]}]` |
50
+ | `okf_list_documents` | `root`, `folder?`, `type?` | `[{path, id, title, type, folder, description, indexado, enriquecido}]` |
51
+ | `okf_get_document` | `root`, `ref` (path ou id) | `{path, markdown, enrichment, links: {saida, entrada, similares}}` |
52
+
53
+ Fluxo: `okf_index` até `restantes=0` → `okf_enrich_next` / `okf_save_enrichment` até `restantes=0` → perguntas
54
+ com `okf_search` e `okf_get_document`. Depois de um novo sincronismo da alm-sync, rode `okf_index` de novo: só o que
55
+ mudou é reindexado, e só o que mudou volta para a fila de enriquecimento.
56
+
57
+ ## Como indexa
58
+
59
+ Cada documento vira pontos `chunk` (corpo dividido por heading `#`..`###`, seções longas por parágrafo; cada chunk
60
+ leva o prefixo `<tipo> <id> — <título> | <pasta> | <seção>`) e um ponto `doc` (cabeçalho + enriquecimento). A busca
61
+ funde o ranking denso e o BM25 (RRF) e agrupa por documento.
62
+
63
+ ## Testes
64
+
65
+ ```bash
66
+ uv run pytest
67
+ ```
68
+ Os testes usam Qdrant em memória e embeddings falsos (não baixam modelo).
@@ -0,0 +1,54 @@
1
+ # mcp-okf
2
+
3
+ Servidor MCP que transforma um bundle OKF (pasta com `index.md` e um `.md` por documento, como o gerado pela
4
+ alm-sync do [mcp-alm](https://github.com/dxaraujo/mcp-alm)) numa base de conhecimento consultável em linguagem natural.
5
+
6
+ - **Índice vetorial:** Qdrant em modo local (pasta no disco, sem servidor), busca híbrida denso + BM25 com fusão RRF.
7
+ - **Embeddings:** fastembed local, `paraphrase-multilingual-mpnet-base-v2` (PT-BR); nada sai da máquina.
8
+ - **Enriquecimento:** feito pela própria LLM do cliente (sem chave de API): resumo, palavras-chave/sinônimos,
9
+ entidades e perguntas que o documento responde.
10
+ - **Links:** `links` do frontmatter (nomes do DOORS Next), links do corpo para outros arquivos do bundle (`cita`) e
11
+ URL do ALM de artefato que está no bundle; backlinks e documentos semanticamente similares na leitura.
12
+
13
+ ## Instalação
14
+
15
+ ```bash
16
+ uv sync
17
+ # baixa os modelos uma vez (~1 GB), para a 1ª indexação não estourar o timeout do cliente MCP
18
+ uv run python -c "from mcp_okf import store; store.embed_dense(['x']); store.embed_sparse(['x'])"
19
+ claude mcp add okf -- uv --directory /caminho/mcp-okf run mcp-okf
20
+ ```
21
+
22
+ Dados em `~/.config/mcp-okf/` (`%APPDATA%\mcp-okf` no Windows; ou `MCP_OKF_HOME`): `qdrant/` e
23
+ `state/<sha1(root)>.json` (hash, links e enriquecimento de cada documento). Nada é gravado no bundle.
24
+ O Qdrant local trava a pasta: só um processo do mcp-okf por vez.
25
+
26
+ ## Tools
27
+
28
+ `root` é sempre o caminho absoluto do bundle.
29
+
30
+ | Tool | Parâmetros | Devolve |
31
+ |---|---|---|
32
+ | `okf_index` | `root`, `limit=200`, `force=False` | `{documentos, indexados, removidos, restantes, a_enriquecer}`; incremental por sha256; chame até `restantes=0` |
33
+ | `okf_enrich_next` | `root`, `limit=5` | `{restantes, instrucoes, documentos: [{path, id, type, title, description, body, links}]}` |
34
+ | `okf_save_enrichment` | `root`, `items: [{path, summary, keywords, entities, questions}]` | `{gravados, restantes}` |
35
+ | `okf_search` | `root`, `query`, `limit=8`, `folder?`, `type?` | `[{path, id, title, type, folder, score, summary, trechos: [{section, text}]}]` |
36
+ | `okf_list_documents` | `root`, `folder?`, `type?` | `[{path, id, title, type, folder, description, indexado, enriquecido}]` |
37
+ | `okf_get_document` | `root`, `ref` (path ou id) | `{path, markdown, enrichment, links: {saida, entrada, similares}}` |
38
+
39
+ Fluxo: `okf_index` até `restantes=0` → `okf_enrich_next` / `okf_save_enrichment` até `restantes=0` → perguntas
40
+ com `okf_search` e `okf_get_document`. Depois de um novo sincronismo da alm-sync, rode `okf_index` de novo: só o que
41
+ mudou é reindexado, e só o que mudou volta para a fila de enriquecimento.
42
+
43
+ ## Como indexa
44
+
45
+ Cada documento vira pontos `chunk` (corpo dividido por heading `#`..`###`, seções longas por parágrafo; cada chunk
46
+ leva o prefixo `<tipo> <id> — <título> | <pasta> | <seção>`) e um ponto `doc` (cabeçalho + enriquecimento). A busca
47
+ funde o ranking denso e o BM25 (RRF) e agrupa por documento.
48
+
49
+ ## Testes
50
+
51
+ ```bash
52
+ uv run pytest
53
+ ```
54
+ Os testes usam Qdrant em memória e embeddings falsos (não baixam modelo).
File without changes
@@ -0,0 +1,150 @@
1
+ """Leitura de um bundle OKF (alm-sync do mcp-alm): documentos, frontmatter, links entre eles e chunks.
2
+ Só disco e texto; o índice fica em store.py."""
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import os
7
+ import posixpath
8
+ import re
9
+ from pathlib import Path
10
+ from urllib.parse import unquote
11
+
12
+ import yaml
13
+ from markdown_it import MarkdownIt
14
+
15
+ SKIP = {"index.md", "sync.md"} # arquivos do próprio bundle, não documentos
16
+ CHUNK_MAX = 1500
17
+ _HEADING = re.compile(r"^(#{1,3})\s+(.+?)\s*#*\s*$", re.M)
18
+ _LEADING_ID = re.compile(r"(\d+)(?:[ _:-].*)?")
19
+ _LINK_ID = re.compile(r"\s*(\d+)\s*:") # item de `links`: 'id: título'
20
+ _MD = MarkdownIt()
21
+
22
+
23
+ def root(dest: str) -> str:
24
+ """Raiz do bundle: `dest` tem de ser absoluto (o cwd do MCP não é o bundle)."""
25
+ if not os.path.isabs(dest):
26
+ raise ValueError(f"root precisa ser caminho absoluto: {dest}")
27
+ path = os.path.realpath(dest)
28
+ if not os.path.isdir(path):
29
+ raise ValueError(f"pasta não encontrada: {dest}")
30
+ return path
31
+
32
+
33
+ def read_document(text: str) -> tuple[dict, str]:
34
+ """(frontmatter YAML completo, corpo). Sem cabeçalho ou YAML inválido: ({}, texto)."""
35
+ if not text.startswith("---\n") or (end := text.find("\n---\n", 3)) < 0:
36
+ return {}, text
37
+ try:
38
+ head = yaml.safe_load(text[4:end]) or {}
39
+ except yaml.YAMLError:
40
+ return {}, text
41
+ return (head, text[end + 5:]) if isinstance(head, dict) else ({}, text)
42
+
43
+
44
+ def _text(value) -> str | None:
45
+ return " ".join(str(value).split()) if value not in (None, "") else None
46
+
47
+
48
+ def load(root_dir: str, path: str) -> dict:
49
+ """Documento pelo caminho relativo (posix) ao root."""
50
+ full = os.path.join(root_dir, *path.split("/"))
51
+ with open(full, "rb") as f:
52
+ raw = f.read()
53
+ head, body = read_document(raw.decode("utf-8"))
54
+ name = posixpath.basename(path)[:-3]
55
+ m = _LEADING_ID.fullmatch(name)
56
+ doc_id = head.get("id") if head.get("id") is not None else (m.group(1) if m else None)
57
+ sources = head.get("sources") if isinstance(head.get("sources"), list) else []
58
+ tags = head.get("tags") if isinstance(head.get("tags"), list) else []
59
+ return {"path": path, "id": str(doc_id) if doc_id is not None else None,
60
+ "title": _text(head.get("title")) or name, "type": _text(head.get("type")),
61
+ "folder": path.split("/")[0] if "/" in path else "", "description": _text(head.get("description")),
62
+ "tags": [str(t) for t in tags], "resource": _text(head.get("resource")),
63
+ "last_modified": next((str(s["last_modified"]) for s in sources
64
+ if isinstance(s, dict) and s.get("last_modified")), None),
65
+ "head": head, "body": body, "hash": hashlib.sha256(raw).hexdigest()}
66
+
67
+
68
+ def scan(root_dir: str) -> dict[str, dict]:
69
+ """Todos os .md do bundle (menos index.md/sync.md na raiz), por caminho relativo posix."""
70
+ docs = {}
71
+ for full in sorted(Path(root_dir).rglob("*.md")):
72
+ rel = full.relative_to(root_dir).as_posix()
73
+ if rel in SKIP or any(part.startswith(".") for part in full.relative_to(root_dir).parts):
74
+ continue
75
+ docs[rel] = load(root_dir, rel)
76
+ return docs
77
+
78
+
79
+ def _targets(markdown: str) -> list[str]:
80
+ """Alvos de link do corpo (markdown-it, inclusive '<alvo com espaço>')."""
81
+ out = []
82
+ for token in _MD.parse(markdown):
83
+ for child in token.children or []:
84
+ if child.type == "link_open" and (href := child.attrGet("href")):
85
+ out.append(str(href))
86
+ return out
87
+
88
+
89
+ def resolve(doc_path: str, target: str, by_id: dict[str, str], docs: dict) -> str | None:
90
+ """Alvo de link -> caminho de documento do bundle; None se estiver fora dele."""
91
+ target = unquote(target).split("#")[0].strip()
92
+ if not target:
93
+ return None
94
+ if "/rm/resources/" in target: # URL do ALM: só liga se o artefato estiver no bundle
95
+ return next((p for p, d in docs.items() if d["resource"] == target.split("?")[0]), None)
96
+ if re.match(r"[a-z][a-z0-9+.-]*:", target, re.I):
97
+ return None
98
+ base = "" if target.startswith("/") else posixpath.dirname(doc_path)
99
+ path = posixpath.normpath(posixpath.join(base, target.lstrip("/")))
100
+ if path.startswith("..") or path.startswith("/"):
101
+ return None # fora do bundle
102
+ if path in docs:
103
+ return path
104
+ m = _LEADING_ID.fullmatch(posixpath.basename(path).removesuffix(".md"))
105
+ return by_id.get(m.group(1)) if m else None
106
+
107
+
108
+ def links(doc: dict, docs: dict, by_id: dict[str, str]) -> list[dict]:
109
+ """[{rel, path}] sem repetição: do frontmatter `links {rel: ['id: título']}` e do corpo (rel 'cita')."""
110
+ out: dict[tuple[str, str], dict] = {}
111
+ raw = doc["head"].get("links")
112
+ for rel, items in (raw.items() if isinstance(raw, dict) else []):
113
+ for item in items if isinstance(items, list) else [items]:
114
+ m = _LINK_ID.match(str(item))
115
+ if m and (path := by_id.get(m.group(1))) and path != doc["path"]:
116
+ out[(str(rel), path)] = {"rel": str(rel), "path": path}
117
+ for target in _targets(doc["body"]):
118
+ if (path := resolve(doc["path"], target, by_id, docs)) and path != doc["path"]:
119
+ out.setdefault(("cita", path), {"rel": "cita", "path": path})
120
+ return list(out.values())
121
+
122
+
123
+ def by_id(docs: dict) -> dict[str, str]:
124
+ return {d["id"]: p for p, d in docs.items() if d["id"]}
125
+
126
+
127
+ def _split(text: str) -> list[str]:
128
+ """Quebra por parágrafo em pedaços de até CHUNK_MAX (parágrafo maior que isso vai inteiro)."""
129
+ parts, cur = [], ""
130
+ for para in re.split(r"\n\s*\n", text):
131
+ if cur and len(cur) + len(para) > CHUNK_MAX:
132
+ parts.append(cur)
133
+ cur = ""
134
+ cur = f"{cur}\n\n{para}" if cur else para
135
+ return parts + ([cur] if cur.strip() else [])
136
+
137
+
138
+ def chunks(doc: dict) -> list[dict]:
139
+ """[{section, text}] por heading (#..###); `text` já com o prefixo de contexto para o embedding."""
140
+ body, sections = doc["body"], []
141
+ marks = list(_HEADING.finditer(body))
142
+ starts = [(0, "")] + [(m.start(), m.group(2)) for m in marks]
143
+ for i, (start, title) in enumerate(starts):
144
+ end = marks[i].start() if i < len(marks) else len(body)
145
+ text = body[start:end].strip()
146
+ if text:
147
+ sections += [(title, part.strip()) for part in _split(text) if part.strip()]
148
+ prefix = " ".join(filter(None, [doc["type"], doc["id"], "—", doc["title"]]))
149
+ return [{"section": s, "text": f"{prefix} | {doc['folder']} | {s}\n{t}"} for s, t in sections] or \
150
+ [{"section": "", "text": f"{prefix} | {doc['folder']}\n{doc['description'] or doc['title']}"}]
@@ -0,0 +1,52 @@
1
+ """Servidor MCP de base de conhecimento sobre bundles OKF (alm-sync do mcp-alm).
2
+
3
+ As tools ficam em tools.py, registradas com `@tool`; leitura do bundle em bundle.py e índice em store.py.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ import functools
8
+
9
+ from mcp.server.mcpserver import MCPServer
10
+ from mcp.server.mcpserver.exceptions import ToolError
11
+
12
+ mcp = MCPServer(
13
+ "okf",
14
+ instructions=(
15
+ "Base de conhecimento sobre um bundle OKF (pasta com index.md e um .md por documento, como o gerado pela "
16
+ "alm-sync). `root` é sempre o caminho absoluto da pasta. Preparar: okf_index até restantes=0; depois "
17
+ "okf_enrich_next -> gere o enriquecimento de cada documento -> okf_save_enrichment, até restantes=0. "
18
+ "Consultar: okf_search (pergunta em linguagem natural), okf_get_document (documento inteiro + links de "
19
+ "saída, de entrada e similares), okf_list_documents (inventário). A busca funciona sem enriquecimento, mas "
20
+ "fica melhor com ele. Cite os documentos pelo path/id."
21
+ ),
22
+ )
23
+
24
+ # RuntimeError: Qdrant travado por outro processo; OSError: disco; ValueError/LookupError: parâmetro inválido
25
+ EXPECTED = (RuntimeError, OSError, LookupError, ValueError)
26
+
27
+
28
+ def tool(fn):
29
+ """Registra `fn` como tool e devolve `fn` intacta (testes veem as exceções originais).
30
+
31
+ O MCP SDK 2.x esconde o texto de exceções que não são ToolError."""
32
+ @functools.wraps(fn)
33
+ def wrapper(*args, **kwargs):
34
+ try:
35
+ return fn(*args, **kwargs)
36
+ except EXPECTED as exc:
37
+ raise ToolError(str(exc)) from exc
38
+
39
+ mcp.tool()(wrapper)
40
+ return fn
41
+
42
+
43
+ # importado depois de `tool` existir: o módulo registra as tools ao ser carregado
44
+ from . import tools # noqa: E402,F401
45
+
46
+
47
+ def main() -> None:
48
+ mcp.run()
49
+
50
+
51
+ if __name__ == "__main__":
52
+ main()
@@ -0,0 +1,190 @@
1
+ """Índice do bundle: Qdrant local (vetor denso + BM25, fusão RRF) e o estado por bundle (hash, links, enriquecimento).
2
+
3
+ Estado em <home>/state/<sha1(root)>.json: {path: {hash, indexed_hash, links: [{rel, path}], enrichment}}.
4
+ É o que torna a indexação incremental e mantém o enriquecimento entre reindexações. Nada é gravado no bundle."""
5
+ from __future__ import annotations
6
+
7
+ import functools
8
+ import hashlib
9
+ import json
10
+ import os
11
+ import sys
12
+ import threading
13
+ import uuid
14
+ from datetime import datetime, timezone
15
+ from pathlib import Path
16
+
17
+ from qdrant_client import QdrantClient, models
18
+
19
+ from . import bundle
20
+
21
+ COLLECTION = "okf"
22
+ DENSE_MODEL = "sentence-transformers/paraphrase-multilingual-mpnet-base-v2"
23
+ DENSE_DIM = 768
24
+ SPARSE_MODEL = "Qdrant/bm25"
25
+ _NS = uuid.UUID("5b0f0c6e-2a51-4f43-9d55-0c2c7f1e9a10")
26
+ # um lock para tudo: tools síncronas podem rodar em threads e o estado é um JSON
27
+ LOCK = threading.RLock()
28
+
29
+
30
+ def home() -> Path:
31
+ if env := os.environ.get("MCP_OKF_HOME"):
32
+ return Path(env)
33
+ if sys.platform == "win32":
34
+ return Path(os.environ["APPDATA"]) / "mcp-okf"
35
+ return Path.home() / ".config" / "mcp-okf"
36
+
37
+
38
+ @functools.cache
39
+ def client() -> QdrantClient:
40
+ """Qdrant em modo local (a pasta fica travada para um processo só)."""
41
+ path = home() / "qdrant"
42
+ path.mkdir(parents=True, exist_ok=True)
43
+ try:
44
+ qc = QdrantClient(path=str(path))
45
+ except RuntimeError as exc:
46
+ raise RuntimeError(f"Índice em {path} já está aberto por outro processo do mcp-okf: {exc}") from exc
47
+ _ensure(qc)
48
+ return qc
49
+
50
+
51
+ def _ensure(qc: QdrantClient) -> None:
52
+ if qc.collection_exists(COLLECTION):
53
+ return
54
+ qc.create_collection(
55
+ COLLECTION,
56
+ vectors_config={"dense": models.VectorParams(size=DENSE_DIM, distance=models.Distance.COSINE)},
57
+ sparse_vectors_config={"bm25": models.SparseVectorParams(modifier=models.Modifier.IDF)})
58
+ # shortcut: sem índices de payload (o Qdrant local os ignora); criar em bundle/path/folder/type/kind ao migrar
59
+ # para servidor Qdrant
60
+
61
+
62
+ @functools.cache
63
+ def _dense():
64
+ from fastembed import TextEmbedding
65
+ return TextEmbedding(DENSE_MODEL)
66
+
67
+
68
+ @functools.cache
69
+ def _sparse():
70
+ from fastembed import SparseTextEmbedding
71
+ return SparseTextEmbedding(SPARSE_MODEL, language="portuguese")
72
+
73
+
74
+ def embed_dense(texts: list[str]) -> list[list[float]]:
75
+ return [v.tolist() for v in _dense().embed(texts)]
76
+
77
+
78
+ def embed_sparse(texts: list[str], query: bool = False) -> list[models.SparseVector]:
79
+ fn = _sparse().query_embed if query else _sparse().embed
80
+ return [models.SparseVector(indices=e.indices.tolist(), values=e.values.tolist()) for e in fn(texts)]
81
+
82
+
83
+ # --- estado
84
+
85
+ def _state_file(root: str) -> Path:
86
+ return home() / "state" / f"{hashlib.sha1(root.encode()).hexdigest()}.json"
87
+
88
+
89
+ def load_state(root: str) -> dict:
90
+ try:
91
+ return json.loads(_state_file(root).read_text(encoding="utf-8"))
92
+ except FileNotFoundError:
93
+ return {}
94
+
95
+
96
+ def save_state(root: str, state: dict) -> None:
97
+ path = _state_file(root)
98
+ path.parent.mkdir(parents=True, exist_ok=True)
99
+ tmp = path.with_suffix(".tmp")
100
+ tmp.write_text(json.dumps(state, ensure_ascii=False, indent=1), encoding="utf-8")
101
+ os.replace(tmp, path) # atômico: estado nunca fica pela metade
102
+
103
+
104
+ def enriched(entry: dict) -> bool:
105
+ return entry.get("enrichment", {}).get("hash") == entry.get("hash")
106
+
107
+
108
+ def indexed(entry: dict) -> bool:
109
+ return entry.get("indexed_hash") == entry.get("hash")
110
+
111
+
112
+ # --- pontos
113
+
114
+ def _pid(root: str, path: str, n) -> str:
115
+ return str(uuid.uuid5(_NS, f"{root}\n{path}\n{n}"))
116
+
117
+
118
+ def _filter(root: str, **eq) -> models.Filter:
119
+ return models.Filter(must=[models.FieldCondition(key=k, match=models.MatchValue(value=v))
120
+ for k, v in {"bundle": root, **eq}.items() if v is not None])
121
+
122
+
123
+ def _payload(root: str, doc: dict, kind: str, section: str = "", text: str = "") -> dict:
124
+ return {"bundle": root, "path": doc["path"], "id": doc["id"], "title": doc["title"], "type": doc["type"],
125
+ "folder": doc["folder"], "kind": kind, "section": section, "text": text}
126
+
127
+
128
+ def doc_text(doc: dict, enrichment: dict | None) -> str:
129
+ """Texto do ponto 'doc': cabeçalho + enriquecimento (é ele que casa pergunta em linguagem natural com jargão)."""
130
+ e = enrichment or {}
131
+ parts = [" ".join(filter(None, [doc["type"], doc["id"], "—", doc["title"]])), doc["folder"],
132
+ doc["description"], e.get("summary"), ", ".join(e.get("keywords", [])),
133
+ ", ".join(e.get("entities", [])), "\n".join(e.get("questions", []))]
134
+ return "\n".join(p for p in parts if p)
135
+
136
+
137
+ def _doc_point(root: str, doc: dict, enrichment: dict | None) -> models.PointStruct:
138
+ text = doc_text(doc, enrichment)
139
+ return models.PointStruct(id=_pid(root, doc["path"], "doc"),
140
+ vector={"dense": embed_dense([text])[0], "bm25": embed_sparse([text])[0]},
141
+ payload=_payload(root, doc, "doc", text=(enrichment or {}).get("summary") or ""))
142
+
143
+
144
+ def _delete(root: str, path: str) -> None:
145
+ client().delete(COLLECTION, points_selector=models.FilterSelector(filter=_filter(root, path=path)))
146
+
147
+
148
+ def write_doc(root: str, doc: dict, enrichment: dict | None) -> None:
149
+ """Apaga os pontos do documento e grava os chunks + o ponto 'doc'."""
150
+ parts = bundle.chunks(doc)
151
+ texts = [c["text"] for c in parts]
152
+ dense, sparse = embed_dense(texts), embed_sparse(texts)
153
+ points = [models.PointStruct(id=_pid(root, doc["path"], n), vector={"dense": d, "bm25": s},
154
+ payload=_payload(root, doc, "chunk", c["section"], c["text"].split("\n", 1)[-1]))
155
+ for n, (c, d, s) in enumerate(zip(parts, dense, sparse))]
156
+ _delete(root, doc["path"])
157
+ client().upsert(COLLECTION, points + [_doc_point(root, doc, enrichment)])
158
+
159
+
160
+ def write_doc_point(root: str, doc: dict, enrichment: dict | None) -> None:
161
+ client().upsert(COLLECTION, [_doc_point(root, doc, enrichment)])
162
+
163
+
164
+ def delete_doc(root: str, path: str) -> None:
165
+ _delete(root, path)
166
+
167
+
168
+ def search(root: str, query: str, limit: int, folder: str | None, type_: str | None) -> list[models.ScoredPoint]:
169
+ flt = _filter(root, folder=folder, type=type_)
170
+ wide = limit * 5
171
+ return client().query_points(
172
+ COLLECTION,
173
+ prefetch=[models.Prefetch(query=embed_dense([query])[0], using="dense", filter=flt, limit=wide),
174
+ models.Prefetch(query=embed_sparse([query], query=True)[0], using="bm25", filter=flt, limit=wide)],
175
+ query=models.FusionQuery(fusion=models.Fusion.RRF), limit=wide, with_payload=True).points
176
+
177
+
178
+ def similar(root: str, path: str, limit: int) -> list[models.ScoredPoint]:
179
+ """Documentos mais próximos pelo vetor do ponto 'doc' (vazio se o documento não estiver indexado)."""
180
+ found = client().retrieve(COLLECTION, [_pid(root, path, "doc")], with_vectors=["dense"])
181
+ if not found:
182
+ return []
183
+ flt = _filter(root, kind="doc")
184
+ flt.must_not = [models.FieldCondition(key="path", match=models.MatchValue(value=path))]
185
+ return client().query_points(COLLECTION, query=found[0].vector["dense"], using="dense", query_filter=flt,
186
+ limit=limit, with_payload=True).points
187
+
188
+
189
+ def now() -> str:
190
+ return datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
@@ -0,0 +1,185 @@
1
+ """Tools do mcp-okf: indexação, enriquecimento (feito pela LLM do cliente), busca e leitura."""
2
+ from __future__ import annotations
3
+
4
+ from typing import TypedDict
5
+
6
+ from . import bundle, store
7
+ from .server import tool
8
+
9
+ BODY_MAX = 12000 # corpo devolvido para enriquecer: limita o contexto gasto por documento
10
+ SNIPPET_MAX = 800
11
+ LIST_MAX, ITEM_MAX, SUMMARY_MAX = 30, 300, 2000
12
+ INSTRUCOES = (
13
+ "Para cada documento, gere em PT-BR e chame okf_save_enrichment com items=[{path, summary, keywords, entities, "
14
+ "questions}]: summary = 2-3 frases do que o documento define; keywords = termos de negócio e sinônimos que uma "
15
+ "pessoa usaria para procurá-lo (inclusive os que o texto não usa literalmente); entities = atores, sistemas, "
16
+ "regras e dados citados; questions = 3-5 perguntas que o documento responde. Use só o que está no documento e "
17
+ "nos links; não invente.")
18
+
19
+
20
+ class Enrichment(TypedDict):
21
+ path: str
22
+ summary: str
23
+ keywords: list[str]
24
+ entities: list[str]
25
+ questions: list[str]
26
+
27
+
28
+ def _ref(entry: dict, path: str) -> dict:
29
+ return {"path": path, "id": entry.get("id"), "title": entry.get("title")}
30
+
31
+
32
+ def _pending_enrichment(state: dict) -> list[str]:
33
+ return [p for p, e in state.items() if store.indexed(e) and not store.enriched(e)]
34
+
35
+
36
+ @tool
37
+ def okf_index(root: str, limit: int = 200, force: bool = False) -> dict:
38
+ """Indexa o bundle OKF em `root` (caminho absoluto): lê todos os .md, recalcula os links entre documentos, remove
39
+ do índice os arquivos apagados e (re)indexa até `limit` documentos novos ou alterados (sha256 diferente).
40
+ Chame até `restantes` = 0. `force=True` (só na 1ª chamada) reindexa tudo, mantendo o enriquecimento.
41
+ Devolve {documentos, indexados, removidos, restantes, a_enriquecer}."""
42
+ with store.LOCK:
43
+ rd = bundle.root(root)
44
+ docs = bundle.scan(rd)
45
+ ids = bundle.by_id(docs)
46
+ state = store.load_state(rd)
47
+ removed = [p for p in state if p not in docs]
48
+ for p in removed:
49
+ store.delete_doc(rd, p)
50
+ del state[p]
51
+ for p, d in docs.items():
52
+ e = state.setdefault(p, {})
53
+ e.update(hash=d["hash"], id=d["id"], title=d["title"], type=d["type"], folder=d["folder"],
54
+ links=bundle.links(d, docs, ids))
55
+ if force:
56
+ e.pop("indexed_hash", None)
57
+ pending = [p for p in docs if not store.indexed(state[p])]
58
+ done = 0
59
+ try:
60
+ for p in pending[:limit]:
61
+ store.write_doc(rd, docs[p], state[p].get("enrichment")) # enriquecimento antigo vale até o novo
62
+ state[p]["indexed_hash"] = docs[p]["hash"]
63
+ done += 1
64
+ finally:
65
+ store.save_state(rd, state) # progresso salvo mesmo se um documento falhar
66
+ return {"documentos": len(docs), "indexados": done, "removidos": len(removed),
67
+ "restantes": len(pending) - done, "a_enriquecer": len(_pending_enrichment(state))}
68
+
69
+
70
+ @tool
71
+ def okf_enrich_next(root: str, limit: int = 5) -> dict:
72
+ """Próximos documentos indexados sem enriquecimento (ou alterados depois dele), para a LLM enriquecer e gravar
73
+ com okf_save_enrichment. Devolve {restantes, instrucoes, documentos: [{path, id, type, title, body, links}]}."""
74
+ with store.LOCK:
75
+ rd = bundle.root(root)
76
+ state = store.load_state(rd)
77
+ pending = _pending_enrichment(state)
78
+ out = []
79
+ for p in pending[:max(1, limit)]:
80
+ doc = bundle.load(rd, p)
81
+ out.append({"path": p, "id": doc["id"], "type": doc["type"], "title": doc["title"],
82
+ "description": doc["description"], "body": doc["body"][:BODY_MAX],
83
+ "links": [{"rel": l["rel"], **_ref(state.get(l["path"], {}), l["path"])}
84
+ for l in state[p].get("links", [])]})
85
+ return {"restantes": len(pending), "instrucoes": INSTRUCOES, "documentos": out}
86
+
87
+
88
+ def _strings(item: dict, key: str) -> list[str]:
89
+ value = item.get(key) or []
90
+ if not isinstance(value, list) or not all(isinstance(x, str) for x in value):
91
+ raise ValueError(f"{item.get('path')}: {key} deve ser lista de textos")
92
+ return [" ".join(x.split())[:ITEM_MAX] for x in value if x.strip()][:LIST_MAX]
93
+
94
+
95
+ @tool
96
+ def okf_save_enrichment(root: str, items: list[Enrichment]) -> dict:
97
+ """Grava o enriquecimento gerado para os documentos de okf_enrich_next e reindexa o ponto de resumo de cada um.
98
+ items = [{path, summary, keywords[], entities[], questions[]}]. Devolve {gravados, restantes}."""
99
+ with store.LOCK:
100
+ rd = bundle.root(root)
101
+ state = store.load_state(rd)
102
+ valid = []
103
+ for item in items: # valida tudo antes de gravar qualquer coisa
104
+ if not isinstance(item, dict) or item.get("path") not in state:
105
+ path = item.get("path") if isinstance(item, dict) else item
106
+ raise ValueError(f"path fora do índice (rode okf_index): {path}")
107
+ summary = item.get("summary")
108
+ if not isinstance(summary, str) or not summary.strip():
109
+ raise ValueError(f"{item['path']}: summary obrigatório")
110
+ valid.append((item["path"], {"summary": " ".join(summary.split())[:SUMMARY_MAX],
111
+ **{k: _strings(item, k) for k in ("keywords", "entities", "questions")}}))
112
+ for path, enrichment in valid:
113
+ doc = bundle.load(rd, path)
114
+ entry = state[path]
115
+ entry["enrichment"] = {**enrichment, "hash": doc["hash"], "at": store.now()}
116
+ if store.indexed(entry) and entry["hash"] == doc["hash"]:
117
+ store.write_doc_point(rd, doc, entry["enrichment"])
118
+ store.save_state(rd, state)
119
+ return {"gravados": len(valid), "restantes": len(_pending_enrichment(state))}
120
+
121
+
122
+ @tool
123
+ def okf_search(root: str, query: str, limit: int = 8, folder: str | None = None, type: str | None = None) -> list:
124
+ """Busca semântica + palavra-chave (híbrida) no bundle indexado, em linguagem natural. Filtros opcionais: `folder`
125
+ (primeira pasta do path) e `type` (tipo do frontmatter). Devolve até `limit` documentos, do mais relevante:
126
+ [{path, id, title, type, folder, score, summary, trechos: [{section, text}]}]. Leia o documento inteiro com
127
+ okf_get_document quando o trecho não bastar."""
128
+ if not query.strip():
129
+ raise ValueError("query vazia")
130
+ with store.LOCK:
131
+ rd = bundle.root(root)
132
+ state = store.load_state(rd)
133
+ out: dict[str, dict] = {}
134
+ for hit in store.search(rd, query, max(1, limit), folder, type):
135
+ p = hit.payload
136
+ r = out.get(p["path"])
137
+ if r is None:
138
+ if len(out) >= limit:
139
+ continue
140
+ r = out[p["path"]] = {
141
+ "path": p["path"], "id": p["id"], "title": p["title"], "type": p["type"], "folder": p["folder"],
142
+ "score": round(hit.score, 4),
143
+ "summary": state.get(p["path"], {}).get("enrichment", {}).get("summary"), "trechos": []}
144
+ if p["kind"] == "chunk" and len(r["trechos"]) < 2:
145
+ r["trechos"].append({"section": p["section"], "text": p["text"][:SNIPPET_MAX]})
146
+ return list(out.values())
147
+
148
+
149
+ @tool
150
+ def okf_list_documents(root: str, folder: str | None = None, type: str | None = None) -> list:
151
+ """Inventário do bundle (sem corpo): [{path, id, title, type, folder, description, indexado, enriquecido}]."""
152
+ with store.LOCK:
153
+ rd = bundle.root(root)
154
+ state = store.load_state(rd)
155
+ out = []
156
+ for p, d in bundle.scan(rd).items():
157
+ if (folder and d["folder"] != folder) or (type and d["type"] != type):
158
+ continue
159
+ e = {**state.get(p, {}), "hash": d["hash"]} # compara com o arquivo de agora
160
+ out.append({"path": p, "id": d["id"], "title": d["title"], "type": d["type"], "folder": d["folder"],
161
+ "description": d["description"], "indexado": store.indexed(e),
162
+ "enriquecido": store.enriched(e)})
163
+ return out
164
+
165
+
166
+ @tool
167
+ def okf_get_document(root: str, ref: str) -> dict:
168
+ """Documento por path (relativo ao root) ou id: {path, markdown (arquivo inteiro), enrichment, links: {saida,
169
+ entrada, similares}}. saida/entrada = [{rel, path, id, title}] (rel = nome do link no DOORS Next ou 'cita' para
170
+ link no texto); similares = documentos mais próximos semanticamente [{path, id, title, score}]."""
171
+ with store.LOCK:
172
+ rd = bundle.root(root)
173
+ state = store.load_state(rd)
174
+ path = ref if ref in state else next((p for p, e in state.items() if e.get("id") == str(ref)), None)
175
+ if path is None:
176
+ raise LookupError(f"documento não encontrado no índice: {ref} (rode okf_index)")
177
+ with open(f"{rd}/{path}", encoding="utf-8") as f:
178
+ markdown = f.read()
179
+ entry = state[path]
180
+ return {"path": path, "markdown": markdown, "enrichment": entry.get("enrichment"), "links": {
181
+ "saida": [{"rel": l["rel"], **_ref(state.get(l["path"], {}), l["path"])} for l in entry.get("links", [])],
182
+ "entrada": [{"rel": l["rel"], **_ref(e, p)} for p, e in state.items()
183
+ for l in e.get("links", []) if l["path"] == path],
184
+ "similares": [{"path": h.payload["path"], "id": h.payload["id"], "title": h.payload["title"],
185
+ "score": round(h.score, 4)} for h in store.similar(rd, path, 5)]}}
@@ -0,0 +1,29 @@
1
+ [project]
2
+ name = "mcp-okf"
3
+ version = "0.1.0"
4
+ description = "MCP server de base de conhecimento sobre bundles OKF (alm-sync): indexação, enriquecimento e busca semântica"
5
+ readme = "README.md"
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ authors = [{ name = "Daniel Xavier Araújo", email = "danielxaraujo@gmail.com" }]
9
+ requires-python = ">=3.10"
10
+ dependencies = [
11
+ "mcp[cli]>=2.2",
12
+ "qdrant-client[fastembed]>=1.12",
13
+ "pyyaml>=6",
14
+ "markdown-it-py>=3",
15
+ ]
16
+
17
+ [project.scripts]
18
+ mcp-okf = "mcp_okf.server:main"
19
+
20
+ [dependency-groups]
21
+ dev = ["pytest>=8"]
22
+
23
+ [build-system]
24
+ requires = ["hatchling"]
25
+ build-backend = "hatchling.build"
26
+
27
+ [tool.pytest.ini_options]
28
+ testpaths = ["tests"]
29
+ pythonpath = ["."]
@@ -0,0 +1,60 @@
1
+ """Bundle de exemplo copiado para tmp, Qdrant em memória e embeddings falsos (sem baixar modelo)."""
2
+ import re
3
+ import zlib
4
+
5
+ import pytest
6
+ from qdrant_client import QdrantClient, models
7
+
8
+ from mcp_okf import store
9
+
10
+ # bundle OKF no formato da alm-sync: {caminho relativo: conteúdo}
11
+ BUNDLE = {
12
+ '01-Requisitos/10-login-do-cliente.md': '---\ntype: Requisito\ntitle: Login do cliente\ndescription: "Acesso ao portal com CPF e senha."\nresource: "https://alm.example.com/rm/resources/TX_10"\ntags:\n - "01-Requisitos"\nsources:\n - id: doors-next\n resource: "https://alm.example.com/rm/resources/TX_10"\n last_modified: "2026-01-10T13:05:00Z"\nid: 10\n---\nO cliente acessa o portal informando CPF e senha. O CPF é validado conforme [2001 RN - Validar CPF do cliente](../03-Regras/2001-rn-validar-cpf-do-cliente.md).\n',
13
+ '03-Casos de Uso/2010-uc-cadastrar-cliente.md': '---\ntype: Caso de Uso\ntitle: UC - Cadastrar cliente\nresource: "https://alm.example.com/rm/resources/TX_2010"\ntags:\n - "03-Casos de Uso"\nsources:\n - id: doors-next\n resource: "https://alm.example.com/rm/resources/TX_2010"\n author: "human:ana.souza"\n last_modified: "2026-01-10T13:05:00Z"\ngenerated:\n by: "process:alm-mcp/1.0.14"\n at: "2026-01-11T19:20:00Z"\nid: 2010\nlinks:\n Vincular A:\n - "10: Login do cliente"\n - "9999: Fora do bundle"\n---\n## Fluxo Básico\n\n1. O usuário informa os dados do cliente.\n2. O sistema valida o documento: [2001 RN - Validar CPF do cliente](https://alm.example.com/rm/resources/TX_2001)\n3. Ver também [manual](https://example.com/manual.pdf) e [fora](../../segredo.md).\n',
14
+ '03-Regras/2001-rn-validar-cpf-do-cliente.md': '---\ntype: Regra de Negócio\ntitle: RN - Validar CPF do cliente\nresource: "https://alm.example.com/rm/resources/TX_2001"\ntags:\n - "03-Regras"\nid: 2001\n---\n## Regra\n\nO CPF deve ter 11 dígitos e dígitos verificadores válidos (módulo 11). CPFs com todos os dígitos iguais são rejeitados.\n\n## Mensagem\n\nExibir "CPF inválido" e bloquear o envio do formulário.\n',
15
+ 'index.md': '---\nokf_version: "0.2"\n---\n\n# 01-Requisitos\n\n* [10 — Login do cliente](</01-Requisitos/10-login-do-cliente.md>) - Acesso ao portal com CPF e senha.\n\n# 03-Casos de Uso\n\n* [2010 — UC - Cadastrar cliente](</03-Casos de Uso/2010-uc-cadastrar-cliente.md>)\n\n# 03-Regras\n\n* [2001 — RN - Validar CPF do cliente](</03-Regras/2001-rn-validar-cpf-do-cliente.md>)\n\n# Bundle\n\n* [Sincronismo ALM → OKF](/sync.md) - Situação de cada artefato do DOORS Next baixado neste bundle.\n',
16
+ 'sync.md': '---\ntype: Relatório de Sincronismo\ntitle: Sincronismo ALM → OKF\n---\n| Artefato | Pasta | Última atualização ALM | Hash | Status |\n|---|---|---|---|---|\n',
17
+ }
18
+
19
+
20
+ def _words(text: str) -> list[str]:
21
+ return re.findall(r"\w+", text.lower())
22
+
23
+
24
+ def fake_dense(texts):
25
+ out = []
26
+ for t in texts:
27
+ v = [0.0] * store.DENSE_DIM
28
+ for w in _words(t):
29
+ v[zlib.crc32(w.encode()) % store.DENSE_DIM] += 1.0
30
+ out.append(v if any(v) else [1.0] + [0.0] * (store.DENSE_DIM - 1))
31
+ return out
32
+
33
+
34
+ def fake_sparse(texts, query=False):
35
+ out = []
36
+ for t in texts:
37
+ idx = sorted({zlib.crc32(w.encode()) % 100_000 for w in _words(t)})
38
+ out.append(models.SparseVector(indices=idx, values=[1.0] * len(idx)))
39
+ return out
40
+
41
+
42
+ @pytest.fixture
43
+ def root(tmp_path):
44
+ dest = tmp_path / "bundle"
45
+ for rel, text in BUNDLE.items():
46
+ (dest / rel).parent.mkdir(parents=True, exist_ok=True)
47
+ (dest / rel).write_text(text, encoding="utf-8")
48
+ return str(dest)
49
+
50
+
51
+ @pytest.fixture
52
+ def okf(tmp_path, monkeypatch):
53
+ monkeypatch.setenv("MCP_OKF_HOME", str(tmp_path / "home"))
54
+ qc = QdrantClient(":memory:")
55
+ store._ensure(qc)
56
+ monkeypatch.setattr(store, "client", lambda: qc)
57
+ monkeypatch.setattr(store, "embed_dense", fake_dense)
58
+ monkeypatch.setattr(store, "embed_sparse", fake_sparse)
59
+ from mcp_okf import tools
60
+ return tools
@@ -0,0 +1,57 @@
1
+ import os
2
+
3
+ import pytest
4
+
5
+ from mcp_okf import bundle
6
+
7
+ UC = "03-Casos de Uso/2010-uc-cadastrar-cliente.md"
8
+ RN = "03-Regras/2001-rn-validar-cpf-do-cliente.md"
9
+ LOGIN = "01-Requisitos/10-login-do-cliente.md"
10
+
11
+
12
+ def test_root_requires_absolute_existing_dir(root):
13
+ with pytest.raises(ValueError):
14
+ bundle.root("relativo/bundle")
15
+ with pytest.raises(ValueError):
16
+ bundle.root(os.path.join(root, "nao-existe"))
17
+
18
+
19
+ def test_scan_reads_nested_frontmatter_and_skips_bundle_files(root):
20
+ docs = bundle.scan(root)
21
+ assert set(docs) == {UC, RN, LOGIN}
22
+ uc = docs[UC]
23
+ assert (uc["id"], uc["type"], uc["folder"]) == ("2010", "Caso de Uso", "03-Casos de Uso")
24
+ assert uc["head"]["links"] == {"Vincular A": ["10: Login do cliente", "9999: Fora do bundle"]}
25
+ assert uc["last_modified"] == "2026-01-10T13:05:00Z"
26
+
27
+
28
+ def test_id_falls_back_to_file_name(root):
29
+ with open(os.path.join(root, "01-Requisitos", "77-sem-cabecalho.md"), "w") as f:
30
+ f.write("Texto sem frontmatter.\n")
31
+ doc = bundle.load(root, "01-Requisitos/77-sem-cabecalho.md")
32
+ assert (doc["id"], doc["title"], doc["head"]) == ("77", "77-sem-cabecalho", {})
33
+
34
+
35
+ def test_links_from_frontmatter_body_and_alm_url(root):
36
+ docs = bundle.scan(root)
37
+ ids = bundle.by_id(docs)
38
+ assert bundle.links(docs[UC], docs, ids) == [{"rel": "Vincular A", "path": LOGIN}, {"rel": "cita", "path": RN}]
39
+ assert bundle.links(docs[LOGIN], docs, ids) == [{"rel": "cita", "path": RN}] # relativo '../03-Regras/...'
40
+
41
+
42
+ def test_resolve_ignores_targets_outside_bundle(root):
43
+ docs = bundle.scan(root)
44
+ assert bundle.resolve(UC, "../../segredo.md", {}, docs) is None
45
+ assert bundle.resolve(UC, "https://example.com/x.md", {}, docs) is None
46
+
47
+
48
+ def test_chunks_by_heading_with_context_prefix(root):
49
+ parts = bundle.chunks(bundle.scan(root)[RN])
50
+ assert [c["section"] for c in parts] == ["Regra", "Mensagem"]
51
+ assert parts[0]["text"].startswith("Regra de Negócio 2001 — RN - Validar CPF do cliente | 03-Regras | Regra\n")
52
+
53
+
54
+ def test_long_section_is_split_by_paragraph():
55
+ body = "\n\n".join("p" * 900 for _ in range(3))
56
+ doc = {"body": body, "type": None, "id": None, "title": "T", "folder": "", "description": None}
57
+ assert len(bundle.chunks(doc)) == 3
@@ -0,0 +1,72 @@
1
+ import os
2
+
3
+ import pytest
4
+
5
+ UC = "03-Casos de Uso/2010-uc-cadastrar-cliente.md"
6
+ RN = "03-Regras/2001-rn-validar-cpf-do-cliente.md"
7
+ LOGIN = "01-Requisitos/10-login-do-cliente.md"
8
+
9
+
10
+ def test_index_is_incremental(okf, root):
11
+ assert okf.okf_index(root, limit=2) == {"documentos": 3, "indexados": 2, "removidos": 0, "restantes": 1,
12
+ "a_enriquecer": 2}
13
+ assert okf.okf_index(root)["indexados"] == 1
14
+ assert okf.okf_index(root)["indexados"] == 0
15
+ with open(os.path.join(root, RN), "a") as f:
16
+ f.write("\nNovo parágrafo.\n")
17
+ assert okf.okf_index(root)["indexados"] == 1
18
+ os.remove(os.path.join(root, LOGIN))
19
+ assert okf.okf_index(root)["removidos"] == 1
20
+ assert {d["path"] for d in okf.okf_list_documents(root)} == {UC, RN}
21
+ assert all(r["path"] != LOGIN for r in okf.okf_search(root, "login portal senha"))
22
+ assert okf.okf_index(root, force=True)["indexados"] == 2
23
+
24
+
25
+ def test_enrichment_round_trip(okf, root):
26
+ okf.okf_index(root)
27
+ batch = okf.okf_enrich_next(root, limit=1)
28
+ assert batch["restantes"] == 3 and len(batch["documentos"]) == 1 and "summary" in batch["instrucoes"]
29
+ uc = okf.okf_enrich_next(root, limit=5)["documentos"]
30
+ assert {"rel": "Vincular A", "path": LOGIN, "id": "10", "title": "Login do cliente"} in \
31
+ next(d for d in uc if d["path"] == UC)["links"]
32
+ items = [{"path": d["path"], "summary": f"Resumo {d['title']}", "keywords": ["onboarding"], "entities": [],
33
+ "questions": []} for d in uc]
34
+ assert okf.okf_save_enrichment(root, items) == {"gravados": 3, "restantes": 0}
35
+ assert okf.okf_enrich_next(root)["documentos"] == []
36
+ assert okf.okf_index(root)["a_enriquecer"] == 0
37
+ # keyword que só existe no enriquecimento já acha o documento
38
+ assert okf.okf_search(root, "onboarding", limit=3)[0]["summary"].startswith("Resumo")
39
+ with open(os.path.join(root, RN), "a") as f:
40
+ f.write("\nMudou.\n")
41
+ assert okf.okf_index(root)["a_enriquecer"] == 1
42
+
43
+
44
+ def test_save_enrichment_validates_before_writing(okf, root):
45
+ okf.okf_index(root)
46
+ ok = {"path": RN, "summary": "x", "keywords": [], "entities": [], "questions": []}
47
+ with pytest.raises(ValueError):
48
+ okf.okf_save_enrichment(root, [ok, {**ok, "path": "../fora.md"}])
49
+ with pytest.raises(ValueError):
50
+ okf.okf_save_enrichment(root, [{**ok, "keywords": "não é lista"}])
51
+ assert okf.okf_enrich_next(root)["restantes"] == 3
52
+
53
+
54
+ def test_search_and_filters(okf, root):
55
+ okf.okf_index(root)
56
+ hits = okf.okf_search(root, "CPF dígitos verificadores módulo 11")
57
+ assert hits[0]["path"] == RN and hits[0]["trechos"][0]["section"] == "Regra"
58
+ assert {h["folder"] for h in okf.okf_search(root, "CPF", folder="01-Requisitos")} == {"01-Requisitos"}
59
+ assert okf.okf_search(root, "CPF", type="Caso de Uso")[0]["path"] == UC
60
+
61
+
62
+ def test_get_document_links_backlinks_and_similar(okf, root):
63
+ okf.okf_index(root)
64
+ doc = okf.okf_get_document(root, "2001")
65
+ assert doc["path"] == RN and doc["markdown"].startswith("---\ntype: Regra de Negócio")
66
+ assert {(l["rel"], l["path"]) for l in doc["links"]["entrada"]} == {("cita", UC), ("cita", LOGIN)}
67
+ assert doc["links"]["saida"] == []
68
+ assert RN not in {s["path"] for s in doc["links"]["similares"]} and doc["links"]["similares"]
69
+ assert okf.okf_get_document(root, UC)["links"]["saida"][0] == {
70
+ "rel": "Vincular A", "path": LOGIN, "id": "10", "title": "Login do cliente"}
71
+ with pytest.raises(LookupError):
72
+ okf.okf_get_document(root, "../../etc/passwd")