spec-probe 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (118) hide show
  1. spec_probe/__init__.py +3 -0
  2. spec_probe/cache/__init__.py +0 -0
  3. spec_probe/cache/store.py +84 -0
  4. spec_probe/collab/__init__.py +26 -0
  5. spec_probe/collab/client.py +248 -0
  6. spec_probe/collab/converter.py +286 -0
  7. spec_probe/collab/publisher.py +188 -0
  8. spec_probe/config_bootstrap.py +102 -0
  9. spec_probe/config_loader.py +586 -0
  10. spec_probe/domain_pack.py +416 -0
  11. spec_probe/graph/__init__.py +0 -0
  12. spec_probe/graph/pipeline.py +453 -0
  13. spec_probe/graph/state.py +32 -0
  14. spec_probe/grounding/__init__.py +0 -0
  15. spec_probe/grounding/anchor.py +124 -0
  16. spec_probe/grounding/validator.py +115 -0
  17. spec_probe/kb/__init__.py +15 -0
  18. spec_probe/kb/models.py +34 -0
  19. spec_probe/kb/rag_documents.py +39 -0
  20. spec_probe/kb/retrieve.py +92 -0
  21. spec_probe/kb/verification_store.py +96 -0
  22. spec_probe/llm/__init__.py +72 -0
  23. spec_probe/llm/copilot_auth.py +165 -0
  24. spec_probe/llm/copilot_check.py +62 -0
  25. spec_probe/llm/copilot_client.py +222 -0
  26. spec_probe/llm/copilot_login.py +159 -0
  27. spec_probe/llm/copilot_models.py +28 -0
  28. spec_probe/llm/copilot_usage.py +88 -0
  29. spec_probe/llm/factory.py +210 -0
  30. spec_probe/llm/jobs.py +118 -0
  31. spec_probe/llm_parse.py +50 -0
  32. spec_probe/models.py +160 -0
  33. spec_probe/nodes/__init__.py +0 -0
  34. spec_probe/nodes/caller_context.py +74 -0
  35. spec_probe/nodes/enrich_requirement.py +77 -0
  36. spec_probe/nodes/final_review.py +417 -0
  37. spec_probe/nodes/grounding.py +29 -0
  38. spec_probe/nodes/load_spec.py +49 -0
  39. spec_probe/nodes/not_found_rescue.py +43 -0
  40. spec_probe/nodes/report.py +157 -0
  41. spec_probe/nodes/retrieve_context.py +144 -0
  42. spec_probe/nodes/search_evidence.py +309 -0
  43. spec_probe/nodes/verify.py +645 -0
  44. spec_probe/rag/__init__.py +5 -0
  45. spec_probe/rag/code_index.py +129 -0
  46. spec_probe/rag/config.py +55 -0
  47. spec_probe/rag/embeddings.py +43 -0
  48. spec_probe/rag/hybrid.py +125 -0
  49. spec_probe/rag/index.py +80 -0
  50. spec_probe/rag/manifest.py +53 -0
  51. spec_probe/rag/reranker.py +72 -0
  52. spec_probe/rag/spec_index.py +96 -0
  53. spec_probe/report/__init__.py +1 -0
  54. spec_probe/report/candidate_display.py +124 -0
  55. spec_probe/report/chat_summary.py +80 -0
  56. spec_probe/report/concise.py +161 -0
  57. spec_probe/report/evidence_format.py +37 -0
  58. spec_probe/report/i18n.py +57 -0
  59. spec_probe/report/links.py +125 -0
  60. spec_probe/report/llm_report.py +150 -0
  61. spec_probe/report/metrics.py +115 -0
  62. spec_probe/report/requirement_block.py +191 -0
  63. spec_probe/report/session.py +37 -0
  64. spec_probe/report/write_files.py +73 -0
  65. spec_probe/report_paths.py +87 -0
  66. spec_probe/session/__init__.py +5 -0
  67. spec_probe/session/warmup.py +58 -0
  68. spec_probe/settings.py +411 -0
  69. spec_probe/spec/__init__.py +11 -0
  70. spec_probe/spec/clarify.py +64 -0
  71. spec_probe/spec/converter.py +102 -0
  72. spec_probe/spec/inline.py +63 -0
  73. spec_probe/spec/normalize.py +91 -0
  74. spec_probe/spec/parse.py +605 -0
  75. spec_probe/spec/search.py +182 -0
  76. spec_probe/spec/section_context.py +34 -0
  77. spec_probe/spec/translate.py +314 -0
  78. spec_probe/structural/__init__.py +5 -0
  79. spec_probe/structural/compile.py +134 -0
  80. spec_probe/structural/fr_templates.py +194 -0
  81. spec_probe/structural/kb_mining.py +451 -0
  82. spec_probe/structural/prefilter.py +115 -0
  83. spec_probe/structural/scan_helpers.py +344 -0
  84. spec_probe/structural/templates.py +644 -0
  85. spec_probe/tools/__init__.py +0 -0
  86. spec_probe/tools/call_chain.py +193 -0
  87. spec_probe/tools/caller_context.py +236 -0
  88. spec_probe/tools/candidates.py +346 -0
  89. spec_probe/tools/codegraph_client.py +266 -0
  90. spec_probe/tools/codegraph_search.py +103 -0
  91. spec_probe/tools/discover_multi.py +103 -0
  92. spec_probe/tools/discover_terms.py +154 -0
  93. spec_probe/tools/evidence_chain.py +159 -0
  94. spec_probe/tools/fr_topics.py +54 -0
  95. spec_probe/tools/grep_fallback.py +60 -0
  96. spec_probe/tools/locate.py +144 -0
  97. spec_probe/tools/module_scope.py +38 -0
  98. spec_probe/tools/not_found_rescue.py +854 -0
  99. spec_probe/tools/ownership_context.py +248 -0
  100. spec_probe/tools/rubric_prompt.py +49 -0
  101. spec_probe/tools/search_agent.py +351 -0
  102. spec_probe/tools/search_per_file.py +182 -0
  103. spec_probe/tools/source_tools.py +272 -0
  104. spec_probe/tools/targeted_grep.py +115 -0
  105. spec_probe/transports/__init__.py +0 -0
  106. spec_probe/transports/cli/__init__.py +0 -0
  107. spec_probe/transports/cli/main.py +334 -0
  108. spec_probe/user_home.py +81 -0
  109. spec_probe/wiki/__init__.py +30 -0
  110. spec_probe/wiki/indexer.py +543 -0
  111. spec_probe/wiki/models.py +99 -0
  112. spec_probe/wiki/router.py +279 -0
  113. spec_probe/wiki/store.py +294 -0
  114. spec_probe-0.1.0.dist-info/METADATA +529 -0
  115. spec_probe-0.1.0.dist-info/RECORD +118 -0
  116. spec_probe-0.1.0.dist-info/WHEEL +5 -0
  117. spec_probe-0.1.0.dist-info/entry_points.txt +4 -0
  118. spec_probe-0.1.0.dist-info/top_level.txt +1 -0
spec_probe/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """spec-probe — functional requirement coverage scanner for C/C++ modules."""
2
+
3
+ __version__ = "0.1.0"
File without changes
@@ -0,0 +1,84 @@
1
+ """Persistent cache for per-FR verification results."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import threading
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ from spec_probe.kb.verification_store import entry_from_verified, upsert_entry
12
+ from spec_probe.models import VerifiedCoverage
13
+ from spec_probe.user_home import ensure_cache_dir
14
+
15
+ _CACHE_LOCK = threading.RLock()
16
+
17
+
18
+ def _cache_key(module_path: str, spec_path: str) -> str:
19
+ raw = f"{module_path}|{spec_path}"
20
+ return hashlib.sha256(raw.encode()).hexdigest()[:16]
21
+
22
+
23
+ def cache_path(module_path: str, spec_path: str) -> Path:
24
+ return ensure_cache_dir() / f"coverage_{_cache_key(module_path, spec_path)}.json"
25
+
26
+
27
+ def load_cache(module_path: str, spec_path: str) -> dict[str, Any]:
28
+ path = cache_path(module_path, spec_path)
29
+ if not path.is_file():
30
+ return {}
31
+ try:
32
+ data = json.loads(path.read_text(encoding="utf-8"))
33
+ return data if isinstance(data, dict) else {}
34
+ except (json.JSONDecodeError, OSError):
35
+ return {}
36
+
37
+
38
+ def save_cache(module_path: str, spec_path: str, entries: dict[str, Any]) -> Path:
39
+ path = cache_path(module_path, spec_path)
40
+ path.write_text(json.dumps(entries, indent=2, ensure_ascii=False), encoding="utf-8")
41
+ return path
42
+
43
+
44
+ def get_cached_result(
45
+ module_path: str,
46
+ spec_path: str,
47
+ fr_id: str,
48
+ ) -> dict[str, Any] | None:
49
+ with _CACHE_LOCK:
50
+ cache = load_cache(module_path, spec_path)
51
+ entry = cache.get(fr_id)
52
+ return entry if isinstance(entry, dict) else None
53
+
54
+
55
+ def set_cached_result(
56
+ module_path: str,
57
+ spec_path: str,
58
+ fr_id: str,
59
+ result: dict[str, Any],
60
+ ) -> None:
61
+ with _CACHE_LOCK:
62
+ cache = load_cache(module_path, spec_path)
63
+ cache[fr_id] = result
64
+ save_cache(module_path, spec_path, cache)
65
+
66
+
67
+ def persist_verification_result(
68
+ result: VerifiedCoverage,
69
+ *,
70
+ module_path: str,
71
+ spec_path: str,
72
+ kb_path: str | None = None,
73
+ search_keywords: list[str] | None = None,
74
+ ) -> None:
75
+ """Append successful verification to the cross-project knowledge base."""
76
+ if result.final_verdict.value not in ("IMPLEMENTED", "PARTIAL", "NEEDS_REVIEW"):
77
+ return
78
+ entry = entry_from_verified(
79
+ result,
80
+ module_path=module_path,
81
+ spec_path=spec_path,
82
+ search_keywords=search_keywords,
83
+ )
84
+ upsert_entry(entry, kb_path)
@@ -0,0 +1,26 @@
1
+ """Collab (Confluence) integration for spec-probe."""
2
+
3
+ from spec_probe.collab.client import (
4
+ extract_base_url,
5
+ find_child_page,
6
+ find_or_create_child_page,
7
+ find_or_create_page,
8
+ make_confluence_client,
9
+ parse_collab_url,
10
+ read_page,
11
+ )
12
+ from spec_probe.collab.converter import md_to_confluence_html
13
+ from spec_probe.collab.publisher import build_page_title, publish_coverage_report
14
+
15
+ __all__ = [
16
+ "build_page_title",
17
+ "extract_base_url",
18
+ "find_child_page",
19
+ "find_or_create_child_page",
20
+ "find_or_create_page",
21
+ "make_confluence_client",
22
+ "md_to_confluence_html",
23
+ "parse_collab_url",
24
+ "publish_coverage_report",
25
+ "read_page",
26
+ ]
@@ -0,0 +1,248 @@
1
+ """Confluence client helper — connect, read, and write pages."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import os
6
+ import re
7
+ from typing import Any, Optional
8
+ from loguru import logger
9
+
10
+ try:
11
+ from atlassian import Confluence
12
+ except ImportError:
13
+ Confluence = None
14
+
15
+ try:
16
+ import requests as _requests
17
+ from requests.adapters import HTTPAdapter as _HTTPAdapter
18
+
19
+ class _TimeoutAdapter(_HTTPAdapter):
20
+ """HTTP adapter with default 30s timeout to prevent hanging connections."""
21
+
22
+ def __init__(self, timeout: int = 30, *args: Any, **kwargs: Any) -> None:
23
+ self.timeout = timeout
24
+ super().__init__(*args, **kwargs)
25
+
26
+ def send(self, request: Any, **kwargs: Any) -> Any:
27
+ kwargs.setdefault("timeout", self.timeout)
28
+ return super().send(request, **kwargs)
29
+
30
+ except ImportError:
31
+ _requests = None
32
+ _TimeoutAdapter = None
33
+
34
+
35
+ def parse_collab_url(url: str) -> tuple[Optional[str], Optional[str], Optional[str]]:
36
+ """
37
+ Extract (space_key, title, page_id) from a Collab/Confluence URL.
38
+ Returns (space_key, title, page_id) — title or page_id may be None.
39
+ """
40
+ if not url:
41
+ return None, None, None
42
+
43
+ # New-style: /spaces/SPACEKEY/pages/PAGEID/Page+Title
44
+ m = re.search(r"/spaces/([^/?#]+)/pages/(\d+)(?:/([^?#]+))?", url)
45
+ if m:
46
+ title = (m.group(3) or "").replace("+", " ").replace("%20", " ") or None
47
+ return m.group(1), title, m.group(2)
48
+
49
+ # Classic: /display/SPACEKEY/Page+Title
50
+ m = re.search(r"/display/([^/?#]+)/([^?#]+)", url)
51
+ if m:
52
+ return m.group(1), m.group(2).replace("+", " ").replace("%20", " "), None
53
+
54
+ # URL with pageId query parameter: ...?pageId=12345
55
+ m = re.search(r"pageId=(\d+)", url)
56
+ if m:
57
+ return None, None, m.group(1)
58
+
59
+ return None, None, None
60
+
61
+
62
+ def extract_base_url(url: str) -> str:
63
+ """Extract Confluence base URL (strips path after /main, /spaces, /display, /pages)."""
64
+ if not url:
65
+ return ""
66
+ base_url = url
67
+ for marker in ["/spaces/", "/display/", "/pages/"]:
68
+ idx = base_url.find(marker)
69
+ if idx != -1:
70
+ base_url = base_url[:idx]
71
+ break
72
+ return base_url.rstrip("/")
73
+
74
+
75
+ def make_confluence_client(
76
+ collab_url: str = "",
77
+ *,
78
+ api_token: str = "",
79
+ timeout: int = 30,
80
+ ) -> Optional[Confluence]:
81
+ """
82
+ Build Confluence client from token in environment (CONFLUENCE_API_TOKEN) or parameter.
83
+ """
84
+ if Confluence is None:
85
+ logger.error(
86
+ "atlassian-python-api is not installed. Run: pip install atlassian-python-api"
87
+ )
88
+ return None
89
+
90
+ base_url = extract_base_url(collab_url)
91
+ if not base_url:
92
+ logger.error("No valid Collab URL provided.")
93
+ return None
94
+
95
+ token = api_token or os.environ.get("CONFLUENCE_API_TOKEN", "")
96
+ if not token:
97
+ logger.error(
98
+ "Missing Confluence credentials: set CONFLUENCE_API_TOKEN in ~/.config/spec-probe/.env"
99
+ )
100
+ return None
101
+
102
+ kwargs: dict[str, Any] = {}
103
+ if _requests is not None and _TimeoutAdapter is not None:
104
+ session = _requests.Session()
105
+ session.mount("http://", _TimeoutAdapter(timeout))
106
+ session.mount("https://", _TimeoutAdapter(timeout))
107
+ kwargs["session"] = session
108
+
109
+ try:
110
+ return Confluence(url=base_url, token=token, cloud=False, **kwargs)
111
+ except Exception as exc:
112
+ logger.error(f"Failed to initialize Confluence client: {exc}")
113
+ return None
114
+
115
+
116
+ def read_page(
117
+ client: Confluence,
118
+ *,
119
+ page_id: str = "",
120
+ space_key: str = "",
121
+ title: str = "",
122
+ ) -> Optional[dict[str, Any]]:
123
+ """Read a Confluence page by page_id or (space_key, title)."""
124
+ try:
125
+ if page_id:
126
+ return client.get_page_by_id(page_id, expand="space,version,body.storage")
127
+ if space_key and title:
128
+ return client.get_page_by_title(space_key, title, expand="space,version,body.storage")
129
+ except Exception as exc:
130
+ logger.warning(f"Failed to read Confluence page (id={page_id}, space={space_key}, title={title}): {exc}")
131
+ return None
132
+
133
+
134
+ def _normalize_child_page_results(result: Any) -> list:
135
+ if result is None:
136
+ return []
137
+ if isinstance(result, list):
138
+ return result
139
+ if isinstance(result, dict):
140
+ return result.get("results", [])
141
+ try:
142
+ return list(result)
143
+ except TypeError:
144
+ return []
145
+
146
+
147
+ def find_child_page(
148
+ client: Confluence,
149
+ parent_id: str,
150
+ title: str,
151
+ ) -> tuple[Optional[str], int]:
152
+ """Find a direct child page under parent_id by exact title."""
153
+ try:
154
+ start = 0
155
+ limit = 100
156
+ while True:
157
+ result = client.get_page_child_by_type(
158
+ parent_id, type="page", start=start, limit=limit
159
+ )
160
+ pages = _normalize_child_page_results(result)
161
+ for page in pages:
162
+ if page.get("title") == title:
163
+ return page.get("id", ""), page.get("version", {}).get("number", 1)
164
+ if len(pages) < limit:
165
+ break
166
+ start += limit
167
+ except Exception as exc:
168
+ logger.warning(f"Could not list child pages under {parent_id}: {exc}")
169
+ return None, 0
170
+
171
+
172
+ def find_page_by_title(
173
+ client: Confluence,
174
+ space_key: str,
175
+ title: str,
176
+ ) -> tuple[Optional[str], int]:
177
+ """Find a page anywhere in the space by exact title."""
178
+ try:
179
+ page = client.get_page_by_title(space_key, title, expand="version")
180
+ if page:
181
+ return page.get("id", ""), page.get("version", {}).get("number", 1)
182
+ except Exception as exc:
183
+ logger.warning(f"Could not find page by title '{title}': {exc}")
184
+ return None, 0
185
+
186
+
187
+ def find_or_create_child_page(
188
+ client: Confluence,
189
+ parent_id: str,
190
+ space_key: str,
191
+ title: str,
192
+ ) -> tuple[str, int]:
193
+ """Find existing child page under parent or create a new child page."""
194
+ page_id, version = find_child_page(client, parent_id, title)
195
+ if page_id:
196
+ logger.info(f"[collab] Found child page: '{title}' (ID: {page_id}, version: {version})")
197
+ return page_id, version
198
+
199
+ logger.info(f"[collab] Creating child page: '{title}' under parent {parent_id}")
200
+ try:
201
+ result = client.create_page(
202
+ space=space_key,
203
+ title=title,
204
+ body="<p>Loading coverage report...</p>",
205
+ parent_id=parent_id,
206
+ representation="storage",
207
+ )
208
+ page_id = result.get("id", "")
209
+ logger.info(f"[collab] Created child page ID: {page_id}")
210
+ return page_id, 1
211
+ except Exception as exc:
212
+ err = str(exc).lower()
213
+ if "already exists" in err or "same title" in err:
214
+ page_id, version = find_page_by_title(client, space_key, title)
215
+ if page_id:
216
+ logger.info(f"[collab] Page already exists: '{title}' (ID: {page_id}, version: {version})")
217
+ return page_id, version
218
+ raise
219
+
220
+
221
+ def find_or_create_page(
222
+ client: Confluence,
223
+ space_key: str,
224
+ title: str,
225
+ ) -> tuple[str, int]:
226
+ """Find existing page in space or create a new one."""
227
+ page_id, version = find_page_by_title(client, space_key, title)
228
+ if page_id:
229
+ logger.info(f"[collab] Found existing page: '{title}' (ID: {page_id}, version: {version})")
230
+ return page_id, version
231
+
232
+ logger.info(f"[collab] Creating new page: '{title}' in space {space_key}")
233
+ try:
234
+ result = client.create_page(
235
+ space=space_key,
236
+ title=title,
237
+ body="<p>Loading coverage report...</p>",
238
+ representation="storage",
239
+ )
240
+ page_id = result.get("id", "")
241
+ return page_id, 1
242
+ except Exception as exc:
243
+ err = str(exc).lower()
244
+ if "already exists" in err or "same title" in err:
245
+ page_id, version = find_page_by_title(client, space_key, title)
246
+ if page_id:
247
+ return page_id, version
248
+ raise
@@ -0,0 +1,286 @@
1
+ """Markdown to Confluence Storage Format XHTML converter."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import html as html_module
6
+ import re
7
+ from loguru import logger
8
+
9
+ try:
10
+ import markdown
11
+ except ImportError:
12
+ markdown = None
13
+
14
+
15
+ def _escape_raw_angle_brackets(md_text: str) -> str:
16
+ """
17
+ Escape < and > outside fenced code blocks and inline code.
18
+ Prevents malformed XML tags when publishing to Confluence storage format.
19
+ """
20
+ parts = re.split(r"(```.*?```|`[^`\n]+`)", md_text, flags=re.DOTALL)
21
+ escaped: list[str] = []
22
+ for i, part in enumerate(parts):
23
+ if i % 2 == 1:
24
+ # Code block or inline code
25
+ escaped.append(part)
26
+ else:
27
+ escaped.append(part.replace("<", "&lt;").replace(">", "&gt;"))
28
+ return "".join(escaped)
29
+
30
+
31
+ def _md_to_html_fallback(md_text: str) -> str:
32
+ """Basic fallback parser for markdown to HTML without python-markdown."""
33
+ lines = md_text.split("\n")
34
+ html_lines: list[str] = []
35
+ in_table = False
36
+ in_code = False
37
+ in_list = False
38
+ code_buffer: list[str] = []
39
+ code_lang = ""
40
+
41
+ def _close_list() -> None:
42
+ nonlocal in_list
43
+ if in_list:
44
+ html_lines.append("</ul>")
45
+ in_list = False
46
+
47
+ for line in lines:
48
+ if line.startswith("```"):
49
+ if in_code:
50
+ lang = code_lang or "cpp"
51
+ html_lines.append(
52
+ _confluence_code_macro("".join(code_buffer).rstrip("\n"), lang)
53
+ )
54
+ code_buffer = []
55
+ code_lang = ""
56
+ in_code = False
57
+ else:
58
+ code_lang = line[3:].strip() or "cpp"
59
+ in_code = True
60
+ continue
61
+ if in_code:
62
+ code_buffer.append(line + "\n")
63
+ continue
64
+
65
+ if line.startswith("### "):
66
+ _close_list()
67
+ html_lines.append(f"<h3>{line[4:]}</h3>")
68
+ elif line.startswith("## "):
69
+ _close_list()
70
+ html_lines.append(f"<h2>{line[3:]}</h2>")
71
+ elif line.startswith("# "):
72
+ _close_list()
73
+ html_lines.append(f"<h1>{line[2:]}</h1>")
74
+ elif re.match(r"^[\s\|:-]+$", line) and "|" in line:
75
+ continue
76
+ elif line.startswith("|") and line.endswith("|"):
77
+ cells = [c.strip() for c in line.split("|")[1:-1]]
78
+ if not in_table:
79
+ html_lines.append("<table><tbody>")
80
+ in_table = True
81
+ html_lines.append("<tr>" + "".join(f"<th>{c}</th>" for c in cells) + "</tr>")
82
+ else:
83
+ html_lines.append("<tr>" + "".join(f"<td>{c}</td>" for c in cells) + "</tr>")
84
+ else:
85
+ if in_table:
86
+ html_lines.append("</tbody></table>")
87
+ in_table = False
88
+ if line.strip() == "":
89
+ _close_list()
90
+ continue
91
+ elif line.startswith("- ") or line.startswith("* "):
92
+ if not in_list:
93
+ html_lines.append("<ul>")
94
+ in_list = True
95
+ item = line[2:]
96
+ item = re.sub(r"\*\*(.+?)\*\*", r"<strong>\1</strong>", item)
97
+ item = re.sub(r"`(.+?)`", r"<code>\1</code>", item)
98
+ item = re.sub(r"\[(.+?)\]\((.+?)\)", r'<a href="\2">\1</a>', item)
99
+ html_lines.append(f"<li>{item}</li>")
100
+ elif line.startswith(" - ") or line.startswith(" * "):
101
+ if not in_list:
102
+ html_lines.append("<ul>")
103
+ in_list = True
104
+ html_lines.append(f"<li style='margin-left:20px'>{line[4:]}</li>")
105
+ else:
106
+ _close_list()
107
+ text = re.sub(r"\*\*(.+?)\*\*", r"<strong>\1</strong>", line)
108
+ text = re.sub(r"`(.+?)`", r"<code>\1</code>", text)
109
+ text = re.sub(r"\[(.+?)\]\((.+?)\)", r'<a href="\2">\1</a>', text)
110
+ html_lines.append(f"<p>{text}</p>")
111
+
112
+ _close_list()
113
+ if in_table:
114
+ html_lines.append("</tbody></table>")
115
+ if in_code:
116
+ html_lines.append(
117
+ _confluence_code_macro("".join(code_buffer).rstrip("\n"), code_lang or "cpp")
118
+ )
119
+
120
+ return "\n".join(html_lines)
121
+
122
+
123
+ def _confluence_code_macro(code: str, language: str = "cpp") -> str:
124
+ """Confluence storage-format code block with syntax highlighting."""
125
+ lang = (language or "text").strip().lower()
126
+ body = code.replace("]]>", "]]&gt;")
127
+ return (
128
+ '<ac:structured-macro ac:name="code" ac:schema-version="1">'
129
+ f'<ac:parameter ac:name="language">{lang}</ac:parameter>'
130
+ f"<ac:plain-text-body><![CDATA[{body}]]></ac:plain-text-body>"
131
+ "</ac:structured-macro>"
132
+ )
133
+
134
+
135
+ def _plain_code_from_html_fragment(fragment: str) -> str:
136
+ raw = html_module.unescape(fragment or "")
137
+ return re.sub(r"<[^>]+>", "", raw).strip()
138
+
139
+
140
+ def _code_pre_blocks_to_confluence_macros(html: str) -> str:
141
+ """Replace HTML <pre><code> blocks with Confluence ``code`` macros (C++ highlight)."""
142
+
143
+ def repl_pre(m: re.Match[str]) -> str:
144
+ lang = (m.group(1) or "cpp").strip().lower()
145
+ return _confluence_code_macro(_plain_code_from_html_fragment(m.group(2)), lang)
146
+
147
+ # codehilite (python-markdown) wrapper — must run before bare <pre>
148
+ html = re.sub(
149
+ r'<div class="(?:highlight|codehilite)"[^>]*>\s*<pre[^>]*>(?:<span></span>)?'
150
+ r"<code[^>]*>(.*?)</code></pre>\s*</div>",
151
+ lambda m: _confluence_code_macro(_plain_code_from_html_fragment(m.group(1)), "cpp"),
152
+ html,
153
+ flags=re.DOTALL | re.IGNORECASE,
154
+ )
155
+ html = re.sub(
156
+ r'<pre[^>]*><code(?:\s+class="[^"]*language-(\w+)[^"]*")?[^>]*>(.*?)</code></pre>',
157
+ repl_pre,
158
+ html,
159
+ flags=re.DOTALL | re.IGNORECASE,
160
+ )
161
+ return html
162
+
163
+
164
+ def _rewrite_links_for_confluence(html: str) -> str:
165
+ """OpenGrok/http links in new tab; rewrite IDE file links to OpenGrok when configured."""
166
+ from spec_probe.report.links import code_browser_uri, get_code_link_config
167
+
168
+ cfg = get_code_link_config()
169
+
170
+ def repl(m: re.Match[str]) -> str:
171
+ href = m.group(1)
172
+ text = m.group(2)
173
+ if href.startswith(("http://", "https://")):
174
+ return (
175
+ f'<a href="{href}" target="_blank" rel="noopener noreferrer">{text}</a>'
176
+ )
177
+ if cfg and cfg.code_browser_base_url.strip():
178
+ for prefix in ("vscode://file", "cursor://file"):
179
+ if href.startswith(prefix):
180
+ path_part = href[len(prefix) :]
181
+ match = re.match(r"^(.+?):(\d+)(?::\d+)?$", path_part)
182
+ if match:
183
+ path = match.group(1)
184
+ line = int(match.group(2))
185
+ new_href = code_browser_uri(
186
+ path,
187
+ line,
188
+ base_url=cfg.code_browser_base_url.strip(),
189
+ path_marker=cfg.code_browser_path_marker,
190
+ )
191
+ return (
192
+ f'<a href="{new_href}" target="_blank" '
193
+ f'rel="noopener noreferrer">{text}</a>'
194
+ )
195
+ if href.startswith(("vscode://", "cursor://", "file://")):
196
+ return text
197
+ return m.group(0)
198
+
199
+ return re.sub(
200
+ r'<a href="([^"]+)"[^>]*>(.*?)</a>',
201
+ repl,
202
+ html,
203
+ flags=re.DOTALL,
204
+ )
205
+
206
+
207
+ def _style_verdicts(html: str) -> str:
208
+ """Add visual badge styling for verdicts in Confluence tables/headings."""
209
+ replacements = {
210
+ r"\bIMPLEMENTED\b": '<span style="color:#00875A;font-weight:bold;background-color:#E3FCEF;padding:2px 6px;border-radius:3px;">IMPLEMENTED</span>',
211
+ r"\bPARTIAL\b": '<span style="color:#FF8B00;font-weight:bold;background-color:#FFF0B3;padding:2px 6px;border-radius:3px;">PARTIAL</span>',
212
+ r"\bNOT_FOUND\b": '<span style="color:#DE350B;font-weight:bold;background-color:#FFEBE6;padding:2px 6px;border-radius:3px;">NOT_FOUND</span>',
213
+ r"\bNEEDS_REVIEW\b": '<span style="color:#42526E;font-weight:bold;background-color:#EBECF0;padding:2px 6px;border-radius:3px;">NEEDS_REVIEW</span>',
214
+ }
215
+ for pattern, badge in replacements.items():
216
+ # Only replace inside table cells <td> or list items <li> to avoid replacing code blocks
217
+ html = re.sub(
218
+ rf"(<td[^>]*>.*?){pattern}(.*?</td>)",
219
+ rf"\1{badge}\2",
220
+ html,
221
+ flags=re.IGNORECASE,
222
+ )
223
+ return html
224
+
225
+
226
+ def md_to_confluence_html(
227
+ md_text: str,
228
+ *,
229
+ jira_url: str = "",
230
+ codebeamer_url: str = "",
231
+ ) -> str:
232
+ """
233
+ Convert Markdown to Confluence storage-format HTML.
234
+ Handles linkification, table styling, code blocks, and escaping.
235
+ """
236
+ safe_md = _escape_raw_angle_brackets(md_text)
237
+
238
+ if markdown is None:
239
+ logger.warning("[collab] markdown library not installed, using fallback HTML converter")
240
+ html = _md_to_html_fallback(safe_md)
241
+ else:
242
+ extensions = [
243
+ "markdown.extensions.extra",
244
+ "markdown.extensions.codehilite",
245
+ "markdown.extensions.tables",
246
+ "markdown.extensions.smarty",
247
+ ]
248
+ html = markdown.markdown(safe_md, extensions=extensions)
249
+
250
+ # Linkify raw HTTP(S) URLs if not already inside an <a> tag
251
+ existing_spans = [m.span() for m in re.finditer(r"<a\b[^>]*>.*?</a>", html, flags=re.DOTALL)]
252
+
253
+ def _in_existing_link(pos: int) -> bool:
254
+ return any(start <= pos < end for start, end in existing_spans)
255
+
256
+ def _linkify_url(m: re.Match) -> str:
257
+ if m.group(0).startswith("&quot;") or _in_existing_link(m.start()):
258
+ return m.group(0)
259
+ return f'<a href="{m.group(1)}">{m.group(1)}</a>'
260
+
261
+ html = re.sub(r"(https?://[^\s<)\"]+)", _linkify_url, html)
262
+
263
+ # Jira keys: PROJECT-1234
264
+ if jira_url:
265
+ jira_base = jira_url.rstrip("/")
266
+ html = re.sub(
267
+ r"(<t[dh][^>]*>)\s*([A-Z][A-Z0-9]+-\d+)\s*(</t[dh]>)",
268
+ lambda m: f'{m.group(1)}<a href="{jira_base}/browse/{m.group(2)}">{m.group(2)}</a>{m.group(3)}',
269
+ html,
270
+ )
271
+
272
+ # Codebeamer ticket IDs: 8 digits
273
+ if codebeamer_url:
274
+ cb_base = codebeamer_url.rstrip("/")
275
+ html = re.sub(
276
+ r"(<t[dh][^>]*>)\s*(\d{8})\s*(</t[dh]>)",
277
+ lambda m: f'{m.group(1)}<a href="{cb_base}/issue/{m.group(2)}">{m.group(2)}</a>{m.group(3)}',
278
+ html,
279
+ )
280
+
281
+ # Style verdict badges
282
+ html = _style_verdicts(html)
283
+ html = _code_pre_blocks_to_confluence_macros(html)
284
+ html = _rewrite_links_for_confluence(html)
285
+
286
+ return html