spec-probe 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- spec_probe/__init__.py +3 -0
- spec_probe/cache/__init__.py +0 -0
- spec_probe/cache/store.py +84 -0
- spec_probe/collab/__init__.py +26 -0
- spec_probe/collab/client.py +248 -0
- spec_probe/collab/converter.py +286 -0
- spec_probe/collab/publisher.py +188 -0
- spec_probe/config_bootstrap.py +102 -0
- spec_probe/config_loader.py +586 -0
- spec_probe/domain_pack.py +416 -0
- spec_probe/graph/__init__.py +0 -0
- spec_probe/graph/pipeline.py +453 -0
- spec_probe/graph/state.py +32 -0
- spec_probe/grounding/__init__.py +0 -0
- spec_probe/grounding/anchor.py +124 -0
- spec_probe/grounding/validator.py +115 -0
- spec_probe/kb/__init__.py +15 -0
- spec_probe/kb/models.py +34 -0
- spec_probe/kb/rag_documents.py +39 -0
- spec_probe/kb/retrieve.py +92 -0
- spec_probe/kb/verification_store.py +96 -0
- spec_probe/llm/__init__.py +72 -0
- spec_probe/llm/copilot_auth.py +165 -0
- spec_probe/llm/copilot_check.py +62 -0
- spec_probe/llm/copilot_client.py +222 -0
- spec_probe/llm/copilot_login.py +159 -0
- spec_probe/llm/copilot_models.py +28 -0
- spec_probe/llm/copilot_usage.py +88 -0
- spec_probe/llm/factory.py +210 -0
- spec_probe/llm/jobs.py +118 -0
- spec_probe/llm_parse.py +50 -0
- spec_probe/models.py +160 -0
- spec_probe/nodes/__init__.py +0 -0
- spec_probe/nodes/caller_context.py +74 -0
- spec_probe/nodes/enrich_requirement.py +77 -0
- spec_probe/nodes/final_review.py +417 -0
- spec_probe/nodes/grounding.py +29 -0
- spec_probe/nodes/load_spec.py +49 -0
- spec_probe/nodes/not_found_rescue.py +43 -0
- spec_probe/nodes/report.py +157 -0
- spec_probe/nodes/retrieve_context.py +144 -0
- spec_probe/nodes/search_evidence.py +309 -0
- spec_probe/nodes/verify.py +645 -0
- spec_probe/rag/__init__.py +5 -0
- spec_probe/rag/code_index.py +129 -0
- spec_probe/rag/config.py +55 -0
- spec_probe/rag/embeddings.py +43 -0
- spec_probe/rag/hybrid.py +125 -0
- spec_probe/rag/index.py +80 -0
- spec_probe/rag/manifest.py +53 -0
- spec_probe/rag/reranker.py +72 -0
- spec_probe/rag/spec_index.py +96 -0
- spec_probe/report/__init__.py +1 -0
- spec_probe/report/candidate_display.py +124 -0
- spec_probe/report/chat_summary.py +80 -0
- spec_probe/report/concise.py +161 -0
- spec_probe/report/evidence_format.py +37 -0
- spec_probe/report/i18n.py +57 -0
- spec_probe/report/links.py +125 -0
- spec_probe/report/llm_report.py +150 -0
- spec_probe/report/metrics.py +115 -0
- spec_probe/report/requirement_block.py +191 -0
- spec_probe/report/session.py +37 -0
- spec_probe/report/write_files.py +73 -0
- spec_probe/report_paths.py +87 -0
- spec_probe/session/__init__.py +5 -0
- spec_probe/session/warmup.py +58 -0
- spec_probe/settings.py +411 -0
- spec_probe/spec/__init__.py +11 -0
- spec_probe/spec/clarify.py +64 -0
- spec_probe/spec/converter.py +102 -0
- spec_probe/spec/inline.py +63 -0
- spec_probe/spec/normalize.py +91 -0
- spec_probe/spec/parse.py +605 -0
- spec_probe/spec/search.py +182 -0
- spec_probe/spec/section_context.py +34 -0
- spec_probe/spec/translate.py +314 -0
- spec_probe/structural/__init__.py +5 -0
- spec_probe/structural/compile.py +134 -0
- spec_probe/structural/fr_templates.py +194 -0
- spec_probe/structural/kb_mining.py +451 -0
- spec_probe/structural/prefilter.py +115 -0
- spec_probe/structural/scan_helpers.py +344 -0
- spec_probe/structural/templates.py +644 -0
- spec_probe/tools/__init__.py +0 -0
- spec_probe/tools/call_chain.py +193 -0
- spec_probe/tools/caller_context.py +236 -0
- spec_probe/tools/candidates.py +346 -0
- spec_probe/tools/codegraph_client.py +266 -0
- spec_probe/tools/codegraph_search.py +103 -0
- spec_probe/tools/discover_multi.py +103 -0
- spec_probe/tools/discover_terms.py +154 -0
- spec_probe/tools/evidence_chain.py +159 -0
- spec_probe/tools/fr_topics.py +54 -0
- spec_probe/tools/grep_fallback.py +60 -0
- spec_probe/tools/locate.py +144 -0
- spec_probe/tools/module_scope.py +38 -0
- spec_probe/tools/not_found_rescue.py +854 -0
- spec_probe/tools/ownership_context.py +248 -0
- spec_probe/tools/rubric_prompt.py +49 -0
- spec_probe/tools/search_agent.py +351 -0
- spec_probe/tools/search_per_file.py +182 -0
- spec_probe/tools/source_tools.py +272 -0
- spec_probe/tools/targeted_grep.py +115 -0
- spec_probe/transports/__init__.py +0 -0
- spec_probe/transports/cli/__init__.py +0 -0
- spec_probe/transports/cli/main.py +334 -0
- spec_probe/user_home.py +81 -0
- spec_probe/wiki/__init__.py +30 -0
- spec_probe/wiki/indexer.py +543 -0
- spec_probe/wiki/models.py +99 -0
- spec_probe/wiki/router.py +279 -0
- spec_probe/wiki/store.py +294 -0
- spec_probe-0.1.0.dist-info/METADATA +529 -0
- spec_probe-0.1.0.dist-info/RECORD +118 -0
- spec_probe-0.1.0.dist-info/WHEEL +5 -0
- spec_probe-0.1.0.dist-info/entry_points.txt +4 -0
- spec_probe-0.1.0.dist-info/top_level.txt +1 -0
spec_probe/__init__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""Persistent cache for per-FR verification results."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import threading
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
from spec_probe.kb.verification_store import entry_from_verified, upsert_entry
|
|
12
|
+
from spec_probe.models import VerifiedCoverage
|
|
13
|
+
from spec_probe.user_home import ensure_cache_dir
|
|
14
|
+
|
|
15
|
+
_CACHE_LOCK = threading.RLock()
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _cache_key(module_path: str, spec_path: str) -> str:
|
|
19
|
+
raw = f"{module_path}|{spec_path}"
|
|
20
|
+
return hashlib.sha256(raw.encode()).hexdigest()[:16]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def cache_path(module_path: str, spec_path: str) -> Path:
|
|
24
|
+
return ensure_cache_dir() / f"coverage_{_cache_key(module_path, spec_path)}.json"
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def load_cache(module_path: str, spec_path: str) -> dict[str, Any]:
|
|
28
|
+
path = cache_path(module_path, spec_path)
|
|
29
|
+
if not path.is_file():
|
|
30
|
+
return {}
|
|
31
|
+
try:
|
|
32
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
33
|
+
return data if isinstance(data, dict) else {}
|
|
34
|
+
except (json.JSONDecodeError, OSError):
|
|
35
|
+
return {}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def save_cache(module_path: str, spec_path: str, entries: dict[str, Any]) -> Path:
|
|
39
|
+
path = cache_path(module_path, spec_path)
|
|
40
|
+
path.write_text(json.dumps(entries, indent=2, ensure_ascii=False), encoding="utf-8")
|
|
41
|
+
return path
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def get_cached_result(
|
|
45
|
+
module_path: str,
|
|
46
|
+
spec_path: str,
|
|
47
|
+
fr_id: str,
|
|
48
|
+
) -> dict[str, Any] | None:
|
|
49
|
+
with _CACHE_LOCK:
|
|
50
|
+
cache = load_cache(module_path, spec_path)
|
|
51
|
+
entry = cache.get(fr_id)
|
|
52
|
+
return entry if isinstance(entry, dict) else None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def set_cached_result(
|
|
56
|
+
module_path: str,
|
|
57
|
+
spec_path: str,
|
|
58
|
+
fr_id: str,
|
|
59
|
+
result: dict[str, Any],
|
|
60
|
+
) -> None:
|
|
61
|
+
with _CACHE_LOCK:
|
|
62
|
+
cache = load_cache(module_path, spec_path)
|
|
63
|
+
cache[fr_id] = result
|
|
64
|
+
save_cache(module_path, spec_path, cache)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def persist_verification_result(
|
|
68
|
+
result: VerifiedCoverage,
|
|
69
|
+
*,
|
|
70
|
+
module_path: str,
|
|
71
|
+
spec_path: str,
|
|
72
|
+
kb_path: str | None = None,
|
|
73
|
+
search_keywords: list[str] | None = None,
|
|
74
|
+
) -> None:
|
|
75
|
+
"""Append successful verification to the cross-project knowledge base."""
|
|
76
|
+
if result.final_verdict.value not in ("IMPLEMENTED", "PARTIAL", "NEEDS_REVIEW"):
|
|
77
|
+
return
|
|
78
|
+
entry = entry_from_verified(
|
|
79
|
+
result,
|
|
80
|
+
module_path=module_path,
|
|
81
|
+
spec_path=spec_path,
|
|
82
|
+
search_keywords=search_keywords,
|
|
83
|
+
)
|
|
84
|
+
upsert_entry(entry, kb_path)
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Collab (Confluence) integration for spec-probe."""
|
|
2
|
+
|
|
3
|
+
from spec_probe.collab.client import (
|
|
4
|
+
extract_base_url,
|
|
5
|
+
find_child_page,
|
|
6
|
+
find_or_create_child_page,
|
|
7
|
+
find_or_create_page,
|
|
8
|
+
make_confluence_client,
|
|
9
|
+
parse_collab_url,
|
|
10
|
+
read_page,
|
|
11
|
+
)
|
|
12
|
+
from spec_probe.collab.converter import md_to_confluence_html
|
|
13
|
+
from spec_probe.collab.publisher import build_page_title, publish_coverage_report
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"build_page_title",
|
|
17
|
+
"extract_base_url",
|
|
18
|
+
"find_child_page",
|
|
19
|
+
"find_or_create_child_page",
|
|
20
|
+
"find_or_create_page",
|
|
21
|
+
"make_confluence_client",
|
|
22
|
+
"md_to_confluence_html",
|
|
23
|
+
"parse_collab_url",
|
|
24
|
+
"publish_coverage_report",
|
|
25
|
+
"read_page",
|
|
26
|
+
]
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Confluence client helper — connect, read, and write pages."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import os
|
|
6
|
+
import re
|
|
7
|
+
from typing import Any, Optional
|
|
8
|
+
from loguru import logger
|
|
9
|
+
|
|
10
|
+
try:
|
|
11
|
+
from atlassian import Confluence
|
|
12
|
+
except ImportError:
|
|
13
|
+
Confluence = None
|
|
14
|
+
|
|
15
|
+
try:
|
|
16
|
+
import requests as _requests
|
|
17
|
+
from requests.adapters import HTTPAdapter as _HTTPAdapter
|
|
18
|
+
|
|
19
|
+
class _TimeoutAdapter(_HTTPAdapter):
|
|
20
|
+
"""HTTP adapter with default 30s timeout to prevent hanging connections."""
|
|
21
|
+
|
|
22
|
+
def __init__(self, timeout: int = 30, *args: Any, **kwargs: Any) -> None:
|
|
23
|
+
self.timeout = timeout
|
|
24
|
+
super().__init__(*args, **kwargs)
|
|
25
|
+
|
|
26
|
+
def send(self, request: Any, **kwargs: Any) -> Any:
|
|
27
|
+
kwargs.setdefault("timeout", self.timeout)
|
|
28
|
+
return super().send(request, **kwargs)
|
|
29
|
+
|
|
30
|
+
except ImportError:
|
|
31
|
+
_requests = None
|
|
32
|
+
_TimeoutAdapter = None
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def parse_collab_url(url: str) -> tuple[Optional[str], Optional[str], Optional[str]]:
|
|
36
|
+
"""
|
|
37
|
+
Extract (space_key, title, page_id) from a Collab/Confluence URL.
|
|
38
|
+
Returns (space_key, title, page_id) — title or page_id may be None.
|
|
39
|
+
"""
|
|
40
|
+
if not url:
|
|
41
|
+
return None, None, None
|
|
42
|
+
|
|
43
|
+
# New-style: /spaces/SPACEKEY/pages/PAGEID/Page+Title
|
|
44
|
+
m = re.search(r"/spaces/([^/?#]+)/pages/(\d+)(?:/([^?#]+))?", url)
|
|
45
|
+
if m:
|
|
46
|
+
title = (m.group(3) or "").replace("+", " ").replace("%20", " ") or None
|
|
47
|
+
return m.group(1), title, m.group(2)
|
|
48
|
+
|
|
49
|
+
# Classic: /display/SPACEKEY/Page+Title
|
|
50
|
+
m = re.search(r"/display/([^/?#]+)/([^?#]+)", url)
|
|
51
|
+
if m:
|
|
52
|
+
return m.group(1), m.group(2).replace("+", " ").replace("%20", " "), None
|
|
53
|
+
|
|
54
|
+
# URL with pageId query parameter: ...?pageId=12345
|
|
55
|
+
m = re.search(r"pageId=(\d+)", url)
|
|
56
|
+
if m:
|
|
57
|
+
return None, None, m.group(1)
|
|
58
|
+
|
|
59
|
+
return None, None, None
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def extract_base_url(url: str) -> str:
|
|
63
|
+
"""Extract Confluence base URL (strips path after /main, /spaces, /display, /pages)."""
|
|
64
|
+
if not url:
|
|
65
|
+
return ""
|
|
66
|
+
base_url = url
|
|
67
|
+
for marker in ["/spaces/", "/display/", "/pages/"]:
|
|
68
|
+
idx = base_url.find(marker)
|
|
69
|
+
if idx != -1:
|
|
70
|
+
base_url = base_url[:idx]
|
|
71
|
+
break
|
|
72
|
+
return base_url.rstrip("/")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def make_confluence_client(
|
|
76
|
+
collab_url: str = "",
|
|
77
|
+
*,
|
|
78
|
+
api_token: str = "",
|
|
79
|
+
timeout: int = 30,
|
|
80
|
+
) -> Optional[Confluence]:
|
|
81
|
+
"""
|
|
82
|
+
Build Confluence client from token in environment (CONFLUENCE_API_TOKEN) or parameter.
|
|
83
|
+
"""
|
|
84
|
+
if Confluence is None:
|
|
85
|
+
logger.error(
|
|
86
|
+
"atlassian-python-api is not installed. Run: pip install atlassian-python-api"
|
|
87
|
+
)
|
|
88
|
+
return None
|
|
89
|
+
|
|
90
|
+
base_url = extract_base_url(collab_url)
|
|
91
|
+
if not base_url:
|
|
92
|
+
logger.error("No valid Collab URL provided.")
|
|
93
|
+
return None
|
|
94
|
+
|
|
95
|
+
token = api_token or os.environ.get("CONFLUENCE_API_TOKEN", "")
|
|
96
|
+
if not token:
|
|
97
|
+
logger.error(
|
|
98
|
+
"Missing Confluence credentials: set CONFLUENCE_API_TOKEN in ~/.config/spec-probe/.env"
|
|
99
|
+
)
|
|
100
|
+
return None
|
|
101
|
+
|
|
102
|
+
kwargs: dict[str, Any] = {}
|
|
103
|
+
if _requests is not None and _TimeoutAdapter is not None:
|
|
104
|
+
session = _requests.Session()
|
|
105
|
+
session.mount("http://", _TimeoutAdapter(timeout))
|
|
106
|
+
session.mount("https://", _TimeoutAdapter(timeout))
|
|
107
|
+
kwargs["session"] = session
|
|
108
|
+
|
|
109
|
+
try:
|
|
110
|
+
return Confluence(url=base_url, token=token, cloud=False, **kwargs)
|
|
111
|
+
except Exception as exc:
|
|
112
|
+
logger.error(f"Failed to initialize Confluence client: {exc}")
|
|
113
|
+
return None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def read_page(
|
|
117
|
+
client: Confluence,
|
|
118
|
+
*,
|
|
119
|
+
page_id: str = "",
|
|
120
|
+
space_key: str = "",
|
|
121
|
+
title: str = "",
|
|
122
|
+
) -> Optional[dict[str, Any]]:
|
|
123
|
+
"""Read a Confluence page by page_id or (space_key, title)."""
|
|
124
|
+
try:
|
|
125
|
+
if page_id:
|
|
126
|
+
return client.get_page_by_id(page_id, expand="space,version,body.storage")
|
|
127
|
+
if space_key and title:
|
|
128
|
+
return client.get_page_by_title(space_key, title, expand="space,version,body.storage")
|
|
129
|
+
except Exception as exc:
|
|
130
|
+
logger.warning(f"Failed to read Confluence page (id={page_id}, space={space_key}, title={title}): {exc}")
|
|
131
|
+
return None
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _normalize_child_page_results(result: Any) -> list:
|
|
135
|
+
if result is None:
|
|
136
|
+
return []
|
|
137
|
+
if isinstance(result, list):
|
|
138
|
+
return result
|
|
139
|
+
if isinstance(result, dict):
|
|
140
|
+
return result.get("results", [])
|
|
141
|
+
try:
|
|
142
|
+
return list(result)
|
|
143
|
+
except TypeError:
|
|
144
|
+
return []
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def find_child_page(
|
|
148
|
+
client: Confluence,
|
|
149
|
+
parent_id: str,
|
|
150
|
+
title: str,
|
|
151
|
+
) -> tuple[Optional[str], int]:
|
|
152
|
+
"""Find a direct child page under parent_id by exact title."""
|
|
153
|
+
try:
|
|
154
|
+
start = 0
|
|
155
|
+
limit = 100
|
|
156
|
+
while True:
|
|
157
|
+
result = client.get_page_child_by_type(
|
|
158
|
+
parent_id, type="page", start=start, limit=limit
|
|
159
|
+
)
|
|
160
|
+
pages = _normalize_child_page_results(result)
|
|
161
|
+
for page in pages:
|
|
162
|
+
if page.get("title") == title:
|
|
163
|
+
return page.get("id", ""), page.get("version", {}).get("number", 1)
|
|
164
|
+
if len(pages) < limit:
|
|
165
|
+
break
|
|
166
|
+
start += limit
|
|
167
|
+
except Exception as exc:
|
|
168
|
+
logger.warning(f"Could not list child pages under {parent_id}: {exc}")
|
|
169
|
+
return None, 0
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def find_page_by_title(
|
|
173
|
+
client: Confluence,
|
|
174
|
+
space_key: str,
|
|
175
|
+
title: str,
|
|
176
|
+
) -> tuple[Optional[str], int]:
|
|
177
|
+
"""Find a page anywhere in the space by exact title."""
|
|
178
|
+
try:
|
|
179
|
+
page = client.get_page_by_title(space_key, title, expand="version")
|
|
180
|
+
if page:
|
|
181
|
+
return page.get("id", ""), page.get("version", {}).get("number", 1)
|
|
182
|
+
except Exception as exc:
|
|
183
|
+
logger.warning(f"Could not find page by title '{title}': {exc}")
|
|
184
|
+
return None, 0
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def find_or_create_child_page(
|
|
188
|
+
client: Confluence,
|
|
189
|
+
parent_id: str,
|
|
190
|
+
space_key: str,
|
|
191
|
+
title: str,
|
|
192
|
+
) -> tuple[str, int]:
|
|
193
|
+
"""Find existing child page under parent or create a new child page."""
|
|
194
|
+
page_id, version = find_child_page(client, parent_id, title)
|
|
195
|
+
if page_id:
|
|
196
|
+
logger.info(f"[collab] Found child page: '{title}' (ID: {page_id}, version: {version})")
|
|
197
|
+
return page_id, version
|
|
198
|
+
|
|
199
|
+
logger.info(f"[collab] Creating child page: '{title}' under parent {parent_id}")
|
|
200
|
+
try:
|
|
201
|
+
result = client.create_page(
|
|
202
|
+
space=space_key,
|
|
203
|
+
title=title,
|
|
204
|
+
body="<p>Loading coverage report...</p>",
|
|
205
|
+
parent_id=parent_id,
|
|
206
|
+
representation="storage",
|
|
207
|
+
)
|
|
208
|
+
page_id = result.get("id", "")
|
|
209
|
+
logger.info(f"[collab] Created child page ID: {page_id}")
|
|
210
|
+
return page_id, 1
|
|
211
|
+
except Exception as exc:
|
|
212
|
+
err = str(exc).lower()
|
|
213
|
+
if "already exists" in err or "same title" in err:
|
|
214
|
+
page_id, version = find_page_by_title(client, space_key, title)
|
|
215
|
+
if page_id:
|
|
216
|
+
logger.info(f"[collab] Page already exists: '{title}' (ID: {page_id}, version: {version})")
|
|
217
|
+
return page_id, version
|
|
218
|
+
raise
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def find_or_create_page(
|
|
222
|
+
client: Confluence,
|
|
223
|
+
space_key: str,
|
|
224
|
+
title: str,
|
|
225
|
+
) -> tuple[str, int]:
|
|
226
|
+
"""Find existing page in space or create a new one."""
|
|
227
|
+
page_id, version = find_page_by_title(client, space_key, title)
|
|
228
|
+
if page_id:
|
|
229
|
+
logger.info(f"[collab] Found existing page: '{title}' (ID: {page_id}, version: {version})")
|
|
230
|
+
return page_id, version
|
|
231
|
+
|
|
232
|
+
logger.info(f"[collab] Creating new page: '{title}' in space {space_key}")
|
|
233
|
+
try:
|
|
234
|
+
result = client.create_page(
|
|
235
|
+
space=space_key,
|
|
236
|
+
title=title,
|
|
237
|
+
body="<p>Loading coverage report...</p>",
|
|
238
|
+
representation="storage",
|
|
239
|
+
)
|
|
240
|
+
page_id = result.get("id", "")
|
|
241
|
+
return page_id, 1
|
|
242
|
+
except Exception as exc:
|
|
243
|
+
err = str(exc).lower()
|
|
244
|
+
if "already exists" in err or "same title" in err:
|
|
245
|
+
page_id, version = find_page_by_title(client, space_key, title)
|
|
246
|
+
if page_id:
|
|
247
|
+
return page_id, version
|
|
248
|
+
raise
|
|
@@ -0,0 +1,286 @@
|
|
|
1
|
+
"""Markdown to Confluence Storage Format XHTML converter."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import html as html_module
|
|
6
|
+
import re
|
|
7
|
+
from loguru import logger
|
|
8
|
+
|
|
9
|
+
try:
|
|
10
|
+
import markdown
|
|
11
|
+
except ImportError:
|
|
12
|
+
markdown = None
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _escape_raw_angle_brackets(md_text: str) -> str:
|
|
16
|
+
"""
|
|
17
|
+
Escape < and > outside fenced code blocks and inline code.
|
|
18
|
+
Prevents malformed XML tags when publishing to Confluence storage format.
|
|
19
|
+
"""
|
|
20
|
+
parts = re.split(r"(```.*?```|`[^`\n]+`)", md_text, flags=re.DOTALL)
|
|
21
|
+
escaped: list[str] = []
|
|
22
|
+
for i, part in enumerate(parts):
|
|
23
|
+
if i % 2 == 1:
|
|
24
|
+
# Code block or inline code
|
|
25
|
+
escaped.append(part)
|
|
26
|
+
else:
|
|
27
|
+
escaped.append(part.replace("<", "<").replace(">", ">"))
|
|
28
|
+
return "".join(escaped)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _md_to_html_fallback(md_text: str) -> str:
|
|
32
|
+
"""Basic fallback parser for markdown to HTML without python-markdown."""
|
|
33
|
+
lines = md_text.split("\n")
|
|
34
|
+
html_lines: list[str] = []
|
|
35
|
+
in_table = False
|
|
36
|
+
in_code = False
|
|
37
|
+
in_list = False
|
|
38
|
+
code_buffer: list[str] = []
|
|
39
|
+
code_lang = ""
|
|
40
|
+
|
|
41
|
+
def _close_list() -> None:
|
|
42
|
+
nonlocal in_list
|
|
43
|
+
if in_list:
|
|
44
|
+
html_lines.append("</ul>")
|
|
45
|
+
in_list = False
|
|
46
|
+
|
|
47
|
+
for line in lines:
|
|
48
|
+
if line.startswith("```"):
|
|
49
|
+
if in_code:
|
|
50
|
+
lang = code_lang or "cpp"
|
|
51
|
+
html_lines.append(
|
|
52
|
+
_confluence_code_macro("".join(code_buffer).rstrip("\n"), lang)
|
|
53
|
+
)
|
|
54
|
+
code_buffer = []
|
|
55
|
+
code_lang = ""
|
|
56
|
+
in_code = False
|
|
57
|
+
else:
|
|
58
|
+
code_lang = line[3:].strip() or "cpp"
|
|
59
|
+
in_code = True
|
|
60
|
+
continue
|
|
61
|
+
if in_code:
|
|
62
|
+
code_buffer.append(line + "\n")
|
|
63
|
+
continue
|
|
64
|
+
|
|
65
|
+
if line.startswith("### "):
|
|
66
|
+
_close_list()
|
|
67
|
+
html_lines.append(f"<h3>{line[4:]}</h3>")
|
|
68
|
+
elif line.startswith("## "):
|
|
69
|
+
_close_list()
|
|
70
|
+
html_lines.append(f"<h2>{line[3:]}</h2>")
|
|
71
|
+
elif line.startswith("# "):
|
|
72
|
+
_close_list()
|
|
73
|
+
html_lines.append(f"<h1>{line[2:]}</h1>")
|
|
74
|
+
elif re.match(r"^[\s\|:-]+$", line) and "|" in line:
|
|
75
|
+
continue
|
|
76
|
+
elif line.startswith("|") and line.endswith("|"):
|
|
77
|
+
cells = [c.strip() for c in line.split("|")[1:-1]]
|
|
78
|
+
if not in_table:
|
|
79
|
+
html_lines.append("<table><tbody>")
|
|
80
|
+
in_table = True
|
|
81
|
+
html_lines.append("<tr>" + "".join(f"<th>{c}</th>" for c in cells) + "</tr>")
|
|
82
|
+
else:
|
|
83
|
+
html_lines.append("<tr>" + "".join(f"<td>{c}</td>" for c in cells) + "</tr>")
|
|
84
|
+
else:
|
|
85
|
+
if in_table:
|
|
86
|
+
html_lines.append("</tbody></table>")
|
|
87
|
+
in_table = False
|
|
88
|
+
if line.strip() == "":
|
|
89
|
+
_close_list()
|
|
90
|
+
continue
|
|
91
|
+
elif line.startswith("- ") or line.startswith("* "):
|
|
92
|
+
if not in_list:
|
|
93
|
+
html_lines.append("<ul>")
|
|
94
|
+
in_list = True
|
|
95
|
+
item = line[2:]
|
|
96
|
+
item = re.sub(r"\*\*(.+?)\*\*", r"<strong>\1</strong>", item)
|
|
97
|
+
item = re.sub(r"`(.+?)`", r"<code>\1</code>", item)
|
|
98
|
+
item = re.sub(r"\[(.+?)\]\((.+?)\)", r'<a href="\2">\1</a>', item)
|
|
99
|
+
html_lines.append(f"<li>{item}</li>")
|
|
100
|
+
elif line.startswith(" - ") or line.startswith(" * "):
|
|
101
|
+
if not in_list:
|
|
102
|
+
html_lines.append("<ul>")
|
|
103
|
+
in_list = True
|
|
104
|
+
html_lines.append(f"<li style='margin-left:20px'>{line[4:]}</li>")
|
|
105
|
+
else:
|
|
106
|
+
_close_list()
|
|
107
|
+
text = re.sub(r"\*\*(.+?)\*\*", r"<strong>\1</strong>", line)
|
|
108
|
+
text = re.sub(r"`(.+?)`", r"<code>\1</code>", text)
|
|
109
|
+
text = re.sub(r"\[(.+?)\]\((.+?)\)", r'<a href="\2">\1</a>', text)
|
|
110
|
+
html_lines.append(f"<p>{text}</p>")
|
|
111
|
+
|
|
112
|
+
_close_list()
|
|
113
|
+
if in_table:
|
|
114
|
+
html_lines.append("</tbody></table>")
|
|
115
|
+
if in_code:
|
|
116
|
+
html_lines.append(
|
|
117
|
+
_confluence_code_macro("".join(code_buffer).rstrip("\n"), code_lang or "cpp")
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
return "\n".join(html_lines)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _confluence_code_macro(code: str, language: str = "cpp") -> str:
|
|
124
|
+
"""Confluence storage-format code block with syntax highlighting."""
|
|
125
|
+
lang = (language or "text").strip().lower()
|
|
126
|
+
body = code.replace("]]>", "]]>")
|
|
127
|
+
return (
|
|
128
|
+
'<ac:structured-macro ac:name="code" ac:schema-version="1">'
|
|
129
|
+
f'<ac:parameter ac:name="language">{lang}</ac:parameter>'
|
|
130
|
+
f"<ac:plain-text-body><![CDATA[{body}]]></ac:plain-text-body>"
|
|
131
|
+
"</ac:structured-macro>"
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _plain_code_from_html_fragment(fragment: str) -> str:
|
|
136
|
+
raw = html_module.unescape(fragment or "")
|
|
137
|
+
return re.sub(r"<[^>]+>", "", raw).strip()
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def _code_pre_blocks_to_confluence_macros(html: str) -> str:
|
|
141
|
+
"""Replace HTML <pre><code> blocks with Confluence ``code`` macros (C++ highlight)."""
|
|
142
|
+
|
|
143
|
+
def repl_pre(m: re.Match[str]) -> str:
|
|
144
|
+
lang = (m.group(1) or "cpp").strip().lower()
|
|
145
|
+
return _confluence_code_macro(_plain_code_from_html_fragment(m.group(2)), lang)
|
|
146
|
+
|
|
147
|
+
# codehilite (python-markdown) wrapper — must run before bare <pre>
|
|
148
|
+
html = re.sub(
|
|
149
|
+
r'<div class="(?:highlight|codehilite)"[^>]*>\s*<pre[^>]*>(?:<span></span>)?'
|
|
150
|
+
r"<code[^>]*>(.*?)</code></pre>\s*</div>",
|
|
151
|
+
lambda m: _confluence_code_macro(_plain_code_from_html_fragment(m.group(1)), "cpp"),
|
|
152
|
+
html,
|
|
153
|
+
flags=re.DOTALL | re.IGNORECASE,
|
|
154
|
+
)
|
|
155
|
+
html = re.sub(
|
|
156
|
+
r'<pre[^>]*><code(?:\s+class="[^"]*language-(\w+)[^"]*")?[^>]*>(.*?)</code></pre>',
|
|
157
|
+
repl_pre,
|
|
158
|
+
html,
|
|
159
|
+
flags=re.DOTALL | re.IGNORECASE,
|
|
160
|
+
)
|
|
161
|
+
return html
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _rewrite_links_for_confluence(html: str) -> str:
|
|
165
|
+
"""OpenGrok/http links in new tab; rewrite IDE file links to OpenGrok when configured."""
|
|
166
|
+
from spec_probe.report.links import code_browser_uri, get_code_link_config
|
|
167
|
+
|
|
168
|
+
cfg = get_code_link_config()
|
|
169
|
+
|
|
170
|
+
def repl(m: re.Match[str]) -> str:
|
|
171
|
+
href = m.group(1)
|
|
172
|
+
text = m.group(2)
|
|
173
|
+
if href.startswith(("http://", "https://")):
|
|
174
|
+
return (
|
|
175
|
+
f'<a href="{href}" target="_blank" rel="noopener noreferrer">{text}</a>'
|
|
176
|
+
)
|
|
177
|
+
if cfg and cfg.code_browser_base_url.strip():
|
|
178
|
+
for prefix in ("vscode://file", "cursor://file"):
|
|
179
|
+
if href.startswith(prefix):
|
|
180
|
+
path_part = href[len(prefix) :]
|
|
181
|
+
match = re.match(r"^(.+?):(\d+)(?::\d+)?$", path_part)
|
|
182
|
+
if match:
|
|
183
|
+
path = match.group(1)
|
|
184
|
+
line = int(match.group(2))
|
|
185
|
+
new_href = code_browser_uri(
|
|
186
|
+
path,
|
|
187
|
+
line,
|
|
188
|
+
base_url=cfg.code_browser_base_url.strip(),
|
|
189
|
+
path_marker=cfg.code_browser_path_marker,
|
|
190
|
+
)
|
|
191
|
+
return (
|
|
192
|
+
f'<a href="{new_href}" target="_blank" '
|
|
193
|
+
f'rel="noopener noreferrer">{text}</a>'
|
|
194
|
+
)
|
|
195
|
+
if href.startswith(("vscode://", "cursor://", "file://")):
|
|
196
|
+
return text
|
|
197
|
+
return m.group(0)
|
|
198
|
+
|
|
199
|
+
return re.sub(
|
|
200
|
+
r'<a href="([^"]+)"[^>]*>(.*?)</a>',
|
|
201
|
+
repl,
|
|
202
|
+
html,
|
|
203
|
+
flags=re.DOTALL,
|
|
204
|
+
)
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def _style_verdicts(html: str) -> str:
|
|
208
|
+
"""Add visual badge styling for verdicts in Confluence tables/headings."""
|
|
209
|
+
replacements = {
|
|
210
|
+
r"\bIMPLEMENTED\b": '<span style="color:#00875A;font-weight:bold;background-color:#E3FCEF;padding:2px 6px;border-radius:3px;">IMPLEMENTED</span>',
|
|
211
|
+
r"\bPARTIAL\b": '<span style="color:#FF8B00;font-weight:bold;background-color:#FFF0B3;padding:2px 6px;border-radius:3px;">PARTIAL</span>',
|
|
212
|
+
r"\bNOT_FOUND\b": '<span style="color:#DE350B;font-weight:bold;background-color:#FFEBE6;padding:2px 6px;border-radius:3px;">NOT_FOUND</span>',
|
|
213
|
+
r"\bNEEDS_REVIEW\b": '<span style="color:#42526E;font-weight:bold;background-color:#EBECF0;padding:2px 6px;border-radius:3px;">NEEDS_REVIEW</span>',
|
|
214
|
+
}
|
|
215
|
+
for pattern, badge in replacements.items():
|
|
216
|
+
# Only replace inside table cells <td> or list items <li> to avoid replacing code blocks
|
|
217
|
+
html = re.sub(
|
|
218
|
+
rf"(<td[^>]*>.*?){pattern}(.*?</td>)",
|
|
219
|
+
rf"\1{badge}\2",
|
|
220
|
+
html,
|
|
221
|
+
flags=re.IGNORECASE,
|
|
222
|
+
)
|
|
223
|
+
return html
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def md_to_confluence_html(
|
|
227
|
+
md_text: str,
|
|
228
|
+
*,
|
|
229
|
+
jira_url: str = "",
|
|
230
|
+
codebeamer_url: str = "",
|
|
231
|
+
) -> str:
|
|
232
|
+
"""
|
|
233
|
+
Convert Markdown to Confluence storage-format HTML.
|
|
234
|
+
Handles linkification, table styling, code blocks, and escaping.
|
|
235
|
+
"""
|
|
236
|
+
safe_md = _escape_raw_angle_brackets(md_text)
|
|
237
|
+
|
|
238
|
+
if markdown is None:
|
|
239
|
+
logger.warning("[collab] markdown library not installed, using fallback HTML converter")
|
|
240
|
+
html = _md_to_html_fallback(safe_md)
|
|
241
|
+
else:
|
|
242
|
+
extensions = [
|
|
243
|
+
"markdown.extensions.extra",
|
|
244
|
+
"markdown.extensions.codehilite",
|
|
245
|
+
"markdown.extensions.tables",
|
|
246
|
+
"markdown.extensions.smarty",
|
|
247
|
+
]
|
|
248
|
+
html = markdown.markdown(safe_md, extensions=extensions)
|
|
249
|
+
|
|
250
|
+
# Linkify raw HTTP(S) URLs if not already inside an <a> tag
|
|
251
|
+
existing_spans = [m.span() for m in re.finditer(r"<a\b[^>]*>.*?</a>", html, flags=re.DOTALL)]
|
|
252
|
+
|
|
253
|
+
def _in_existing_link(pos: int) -> bool:
|
|
254
|
+
return any(start <= pos < end for start, end in existing_spans)
|
|
255
|
+
|
|
256
|
+
def _linkify_url(m: re.Match) -> str:
|
|
257
|
+
if m.group(0).startswith(""") or _in_existing_link(m.start()):
|
|
258
|
+
return m.group(0)
|
|
259
|
+
return f'<a href="{m.group(1)}">{m.group(1)}</a>'
|
|
260
|
+
|
|
261
|
+
html = re.sub(r"(https?://[^\s<)\"]+)", _linkify_url, html)
|
|
262
|
+
|
|
263
|
+
# Jira keys: PROJECT-1234
|
|
264
|
+
if jira_url:
|
|
265
|
+
jira_base = jira_url.rstrip("/")
|
|
266
|
+
html = re.sub(
|
|
267
|
+
r"(<t[dh][^>]*>)\s*([A-Z][A-Z0-9]+-\d+)\s*(</t[dh]>)",
|
|
268
|
+
lambda m: f'{m.group(1)}<a href="{jira_base}/browse/{m.group(2)}">{m.group(2)}</a>{m.group(3)}',
|
|
269
|
+
html,
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
# Codebeamer ticket IDs: 8 digits
|
|
273
|
+
if codebeamer_url:
|
|
274
|
+
cb_base = codebeamer_url.rstrip("/")
|
|
275
|
+
html = re.sub(
|
|
276
|
+
r"(<t[dh][^>]*>)\s*(\d{8})\s*(</t[dh]>)",
|
|
277
|
+
lambda m: f'{m.group(1)}<a href="{cb_base}/issue/{m.group(2)}">{m.group(2)}</a>{m.group(3)}',
|
|
278
|
+
html,
|
|
279
|
+
)
|
|
280
|
+
|
|
281
|
+
# Style verdict badges
|
|
282
|
+
html = _style_verdicts(html)
|
|
283
|
+
html = _code_pre_blocks_to_confluence_macros(html)
|
|
284
|
+
html = _rewrite_links_for_confluence(html)
|
|
285
|
+
|
|
286
|
+
return html
|