atlasdocs-metadata 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,24 @@
1
+ # Python
2
+ __pycache__/
3
+ *.py[cod]
4
+ *.egg-info/
5
+ .eggs/
6
+
7
+ # Build / dist
8
+ dist/
9
+ build/
10
+
11
+ # Pixi environments
12
+ .pixi/
13
+
14
+ # macOS
15
+ .DS_Store
16
+
17
+ # Editors
18
+ .idea/
19
+ .vscode/
20
+ *.swp
21
+
22
+ # Secrets / env
23
+ .env
24
+ *.env
@@ -0,0 +1,9 @@
1
+ Metadata-Version: 2.4
2
+ Name: atlasdocs-metadata
3
+ Version: 0.9.0
4
+ Summary: Syncs the atlasdocs/metadata registry (data_<CODE>.yml / stare_<CODE>.json.br, every group) and renders it via GroupMacros/GlanceMacros for atlasdocs-theme sites
5
+ Author-email: Jason Oliver <jason.oliver@cern.ch>
6
+ License-Expression: Apache-2.0
7
+ Requires-Python: >=3.9
8
+ Requires-Dist: brotli>=1.1
9
+ Requires-Dist: pyyaml>=6
@@ -0,0 +1,17 @@
1
+ from __future__ import annotations
2
+
3
+
4
+ def __getattr__(name: str):
5
+ if name == "fetch_metadata":
6
+ from .fetch_metadata import main as fetch_metadata
7
+ return fetch_metadata
8
+ if name == "GroupMacros":
9
+ from .group import GroupMacros
10
+ return GroupMacros
11
+ if name == "GlanceMacros":
12
+ from .glance import GlanceMacros
13
+ return GlanceMacros
14
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
15
+
16
+
17
+ __all__ = ["fetch_metadata", "GroupMacros", "GlanceMacros"]
@@ -0,0 +1,152 @@
1
+ from __future__ import annotations
2
+
3
+ """
4
+ Syncs the whole `data/` tree of atlasdocs/metadata (every group's
5
+ data_<CODE>.yml and stare_<CODE>.json.br — pre-normalized, brotli-compressed
6
+ ATLAS Glance/STARE records) into <hub>/data/metadata/. This is the fetch
7
+ half of the package; atlasdocs_metadata.GroupMacros / .GlanceMacros (see
8
+ group.py / glance.py) are the reader/renderer half, kept in the same
9
+ package deliberately since they read exactly what this writes — splitting
10
+ fetch format and reader across two independently-versioned packages would
11
+ reintroduce the version-skew risk this package exists to avoid.
12
+
13
+ Deliberately all-or-nothing: there is no per-group allowlist. A site either
14
+ opts in (by adding a [metadata] config, see below) and gets every group, or
15
+ it doesn't fetch anything. This mirrors atlas-docs-bphy's old
16
+ scripts/pull_metadata.sh + pull_stare.sh, merged into one clone (both files
17
+ live under the same `data/` prefix upstream, so one sparse-checkout covers
18
+ both) and widened from a single hardcoded CODE to the full registry.
19
+
20
+ Config — reads [metadata] from atlasdocs.toml first, falling back to a
21
+ standalone metadata.toml (flat, no [metadata] wrapper) if that file or
22
+ section isn't present. Presence of either is what opts a site in; a site
23
+ with neither is skipped entirely, matching atlasdocs_gitlab's gitlab.toml
24
+ gating. Recognized keys (all optional):
25
+
26
+ repo = "https://gitlab.cern.ch/atlasdocs/metadata.git" # default shown
27
+ branch = "main" # default: repo default branch
28
+
29
+ Auth mirrors atlasdocs_theme.scripts.import_sources: prefers $GITLAB_TOKEN,
30
+ otherwise clones bare — in CI the global url.insteadOf rewrite (set up in
31
+ every consumer's .gitlab-ci.yml pixi before_script) injects $CI_JOB_TOKEN,
32
+ and locally the ambient git credential helper takes over.
33
+
34
+ A clone failure is non-fatal (warns and exits 0) — same soft-fail behavior
35
+ as the shell scripts it replaces, reinforced by this step being registered
36
+ as required=False in atlasdocs_theme.registry.
37
+
38
+ Usage:
39
+ python -m atlasdocs_metadata.fetch_metadata [hub_dir]
40
+ Pixi (via atlasdocs-theme's step registry):
41
+ pixi run fetch
42
+ """
43
+
44
+ import os
45
+ import shutil
46
+ import subprocess
47
+ import sys
48
+ import tomllib
49
+ from pathlib import Path
50
+
51
+ ATLASDOCS_TOML = "atlasdocs.toml"
52
+ LEGACY_TOML = "metadata.toml"
53
+ DEFAULT_REPO = "https://gitlab.cern.ch/atlasdocs/metadata.git"
54
+ CACHE_DIR = ".metadata-cache"
55
+ CLONE_TIMEOUT = 300
56
+
57
+
58
+ def load_config(hub_dir: Path) -> dict | None:
59
+ """Returns the [metadata] config dict, or None if this site hasn't opted in."""
60
+ unified = hub_dir / ATLASDOCS_TOML
61
+ if unified.is_file():
62
+ with open(unified, "rb") as f:
63
+ data = tomllib.load(f)
64
+ if "metadata" in data:
65
+ return data["metadata"]
66
+
67
+ legacy = hub_dir / LEGACY_TOML
68
+ if legacy.is_file():
69
+ with open(legacy, "rb") as f:
70
+ return tomllib.load(f)
71
+
72
+ return None
73
+
74
+
75
+ def _auth_url(url: str) -> str:
76
+ if token := os.environ.get("GITLAB_TOKEN"):
77
+ return url.replace("https://", f"https://oauth2:{token}@", 1)
78
+ return url
79
+
80
+
81
+ def _git(args: list[str]) -> subprocess.CompletedProcess:
82
+ env = {**os.environ, "GIT_TERMINAL_PROMPT": "0"}
83
+ return subprocess.run(["git", *args], capture_output=True, text=True, env=env, timeout=CLONE_TIMEOUT)
84
+
85
+
86
+ def sync_registry(repo_url: str, branch: str | None, cache: Path) -> bool:
87
+ """Sparse-clone <repo>'s data/ tree (blobless + no-checkout, then sparse
88
+ checkout just `data`) into `cache`. Returns False on any failure."""
89
+ if cache.exists():
90
+ shutil.rmtree(cache)
91
+ clone = ["clone", "--depth=1", "--filter=blob:none", "--no-checkout"]
92
+ if branch:
93
+ clone += ["--branch", branch]
94
+ clone += [_auth_url(repo_url), str(cache)]
95
+ try:
96
+ r = _git(clone)
97
+ if r.returncode != 0:
98
+ print(f"WARNING: could not clone {repo_url}, skipping metadata sync:\n{r.stderr.strip()}")
99
+ return False
100
+ _git(["-C", str(cache), "sparse-checkout", "set", "data"])
101
+ co = _git(["-C", str(cache), "checkout"])
102
+ if co.returncode != 0:
103
+ print(f"WARNING: checkout failed, skipping metadata sync:\n{co.stderr.strip()}")
104
+ return False
105
+ except subprocess.TimeoutExpired:
106
+ print(f"WARNING: clone of {repo_url} timed out after {CLONE_TIMEOUT}s, skipping metadata sync")
107
+ return False
108
+ return True
109
+
110
+
111
+ def replace_tree(src: Path, dest: Path) -> tuple[int, int]:
112
+ """Wholesale-replace dest with src's contents (so groups removed upstream
113
+ disappear locally too, not just accumulate). Returns (n_yml, n_br) copied."""
114
+ if dest.exists():
115
+ shutil.rmtree(dest)
116
+ shutil.copytree(src, dest)
117
+ n_yml = len(list(dest.glob("data_*.yml")))
118
+ n_br = len(list(dest.glob("stare_*.json.br")))
119
+ return n_yml, n_br
120
+
121
+
122
+ def main() -> None:
123
+ hub_dir = Path(sys.argv[1]).resolve() if len(sys.argv) > 1 else Path.cwd()
124
+
125
+ config = load_config(hub_dir)
126
+ if config is None:
127
+ print(f"[fetch_metadata] no {ATLASDOCS_TOML} [metadata] section or {LEGACY_TOML} found, skipping")
128
+ return
129
+
130
+ repo_url = config.get("repo", DEFAULT_REPO)
131
+ branch = config.get("branch")
132
+
133
+ cache = hub_dir / CACHE_DIR
134
+ if not sync_registry(repo_url, branch, cache):
135
+ shutil.rmtree(cache, ignore_errors=True)
136
+ return
137
+
138
+ src = cache / "data"
139
+ if not src.is_dir():
140
+ print(f"WARNING: {repo_url} has no data/ directory, skipping metadata sync")
141
+ shutil.rmtree(cache, ignore_errors=True)
142
+ return
143
+
144
+ dest = hub_dir / "data" / "metadata"
145
+ n_yml, n_br = replace_tree(src, dest)
146
+ print(f"[fetch_metadata] synced {n_yml} data_*.yml + {n_br} stare_*.json.br -> {dest}")
147
+
148
+ shutil.rmtree(cache, ignore_errors=True)
149
+
150
+
151
+ if __name__ == "__main__":
152
+ main()
@@ -0,0 +1,490 @@
1
+ from __future__ import annotations
2
+
3
+ import html
4
+ import json
5
+ import re
6
+ from pathlib import Path
7
+
8
+ import brotli
9
+
10
+ _GLANCE_BASE = "https://atlas-glance.cern.ch/atlas/analysis"
11
+
12
+ # Per-entry-type lifecycle key -> display label, in display order. Mirrors
13
+ # ANALYSIS_LIFECYCLE_LABELS / PAPER_LIFECYCLE_LABELS / NOTE_LIFECYCLE_LABELS
14
+ # in atlas-search-frontend's GlanceStareCard.vue.
15
+ _ANALYSIS_LIFECYCLE_LABELS = {
16
+ "start_date": "Start",
17
+ "meeting_eoi": "EOI",
18
+ "meeting_eb_request": "EB Request",
19
+ "editorial_board_formed_date": "EB Formed",
20
+ "pgc_sgc_sign_off_date": "PGC/SGC",
21
+ "meeting_pre_approval": "Pre-approval",
22
+ "meeting_approval": "Approval",
23
+ }
24
+ _PAPER_LIFECYCLE_LABELS = {
25
+ "phase1_start": "Phase 1 Start",
26
+ "editorial_board_formed": "EB Formed",
27
+ "presentation": "Presentation",
28
+ "pgc_approval": "PGC Approval",
29
+ "draft1_released": "Draft 1",
30
+ "draft2_released": "Draft 2",
31
+ "draft2_sent_to_cern": "Sent to CERN",
32
+ "paper_closure": "Closure",
33
+ "arxiv_submitted": "arXiv Submitted",
34
+ "journal_accepted": "Journal Accepted",
35
+ "published_online": "Published Online",
36
+ }
37
+ _NOTE_LIFECYCLE_LABELS = {
38
+ "start_date": "Start",
39
+ "editorial_board_formed": "EB Formed",
40
+ "presentation": "Presentation",
41
+ "pgc_approval": "PGC Approval",
42
+ "first_sign_off": "First Sign-off",
43
+ "first_reader_sign_off": "First Reader Sign-off",
44
+ "second_sign_off": "Second Sign-off",
45
+ "second_reader_sign_off": "Second Reader Sign-off",
46
+ "group_approval": "Group Approval",
47
+ "atlas_circulation": "ATLAS Circulation",
48
+ "release_date": "Release",
49
+ }
50
+
51
+ _REPO_LABELS = {"INT": "Int. Note", "PAP": "Paper", "PUB": "PUB Note", "CONF": "CONF Note", "CODE": "Code"}
52
+
53
+
54
+ def _e(value) -> str:
55
+ """Escape a value for HTML text content (titles/keywords can contain '->' etc.)."""
56
+ return html.escape(str(value)) if value is not None else ""
57
+
58
+
59
+ class GlanceMacros:
60
+ """Renders ATLAS Glance/STARE personnel + analyses views from data/metadata/stare_<CODE>.json.br
61
+ (pre-normalized, brotli-compressed hit dicts published by atlas-docs-metadata's
62
+ scripts/split_stare_registry.py, synced locally by atlasdocs_metadata.fetch_metadata). A
63
+ hand-written HTML/CSS port of atlas-search-frontend's GlanceStareCard.vue — same look, no
64
+ Vue/OpenSearch involved.
65
+
66
+ Usage in markdown::
67
+
68
+ {{ GLANCE.active_personnel("BPHY") | safe }}
69
+ {{ GLANCE.personnel_history("BPHY") | safe }}
70
+ {{ GLANCE.analyses("BPHY") | safe }}
71
+ {{ GLANCE.analyses("BPHY", view="list", entry_type="Paper") | safe }}
72
+ {{ GLANCE.analyses("BPHY", entry_type="Analysis", ongoing_only=True, show_circle=False) | safe }}
73
+ {{ GLANCE.analysis("ANA-BPHY-2018-14") | safe }}
74
+ """
75
+
76
+ def __init__(self, data_dir: Path):
77
+ self._hits_by_group: dict[str, list[dict]] = {}
78
+ self._by_ref: dict[str, dict] = {}
79
+ for path in sorted(data_dir.glob("stare_*.json.br")):
80
+ code = path.stem[len("stare_"):].removesuffix(".json")
81
+ payload = json.loads(brotli.decompress(path.read_bytes()))
82
+ hits = payload.get("hits", [])
83
+ self._hits_by_group[code] = hits
84
+ for hit in hits:
85
+ ref = hit.get("ref_code")
86
+ if ref:
87
+ self._by_ref.setdefault(ref, hit)
88
+ final_ref = hit.get("final_ref_code")
89
+ if final_ref:
90
+ self._by_ref.setdefault(final_ref, hit)
91
+
92
+ # ── lookups ──────────────────────────────────────────────────────────
93
+
94
+ def _find(self, ref_code: str) -> dict | None:
95
+ if ref_code in self._by_ref:
96
+ return self._by_ref[ref_code]
97
+ if not ref_code.startswith("ANA-"):
98
+ return self._by_ref.get(f"ANA-{ref_code}")
99
+ return None
100
+
101
+ @staticmethod
102
+ def _glance_url(hit: dict) -> str:
103
+ ref = hit.get("ref_code", "")
104
+ et = hit.get("entry_type")
105
+ if et == "ConfNote":
106
+ return f"{_GLANCE_BASE}/confnotes/details?ref_code={ref}"
107
+ if et == "PubNote":
108
+ return f"{_GLANCE_BASE}/pubnotes/details?ref_code={ref}"
109
+ if et == "Paper":
110
+ return f"{_GLANCE_BASE}/papers/details.php?ref_code={ref}"
111
+ return f"{_GLANCE_BASE}/analyses/details.php?ref_code={ref}"
112
+
113
+ @staticmethod
114
+ def _badge(hit: dict) -> tuple[str, str]:
115
+ et = hit.get("entry_type", "")
116
+ if et == "Paper":
117
+ return "Paper", "glance-type-paper"
118
+ if et == "ConfNote":
119
+ return "CONF Note", "glance-type-note"
120
+ if et == "PubNote":
121
+ return "PUB Note", "glance-type-note"
122
+ # Analysis: show the actual status name (e.g. "Phase 0 Active"), as
123
+ # stored by stare_normalize.py's _status() — not a derived "phase0
124
+ # active"-style slug.
125
+ state = hit.get("phase") or ""
126
+ if not state:
127
+ return "ongoing", "glance-type-analysis"
128
+ cls = "glance-type-analysis" if "active" in state.lower() else "glance-type-note"
129
+ return state, cls
130
+
131
+ @staticmethod
132
+ def _display_title(hit: dict) -> str:
133
+ """Raw (unescaped) display title — callers must run it through _e() before
134
+ embedding in HTML; analyses(view="list") also needs the raw form for its
135
+ markdown "|" escaping."""
136
+ return re.sub(r"^\[(?:ANA|CONF|PUB)-", "[", hit.get("title", ""))
137
+
138
+ @staticmethod
139
+ def _lifecycle_labels(entry_type: str) -> dict:
140
+ if entry_type == "Paper":
141
+ return _PAPER_LIFECYCLE_LABELS
142
+ if entry_type in ("ConfNote", "PubNote"):
143
+ return _NOTE_LIFECYCLE_LABELS
144
+ return _ANALYSIS_LIFECYCLE_LABELS
145
+
146
+ # ── card rendering ───────────────────────────────────────────────────
147
+
148
+ def _render_card(self, hit: dict, show_circle: bool = True) -> str:
149
+ ref = hit.get("ref_code", "")
150
+ badge_label, badge_cls = self._badge(hit)
151
+ title = _e(self._display_title(hit))
152
+ glance_url = self._glance_url(hit)
153
+
154
+ leading = hit.get("leading_group")
155
+ subgroups = hit.get("subgroups") or []
156
+ runs = list(dict.fromkeys(s.strip() for s in (hit.get("run") or "").split(",") if s.strip()))
157
+
158
+ tags = []
159
+ if leading:
160
+ tags.append(f'<span class="glance-tag glance-leading-group-tag">{_e(leading)}</span>')
161
+ for sg in subgroups[:3]:
162
+ tags.append(f'<span class="glance-tag glance-leading-group-tag">{_e(sg.upper())}</span>')
163
+ if len(subgroups) > 3:
164
+ tags.append(f'<span class="glance-tag glance-tag-more">+{len(subgroups) - 3}</span>')
165
+ for r in runs:
166
+ tags.append(f'<span class="glance-tag glance-run">{_e(r)}</span>')
167
+
168
+ date_label = hit.get("date", "")
169
+
170
+ # ── lifecycle (left column) ──
171
+ lifecycle = hit.get("lifecycle") or {}
172
+ labels = self._lifecycle_labels(hit.get("entry_type", ""))
173
+ lc_rows = []
174
+ for key, label in labels.items():
175
+ val = lifecycle.get(key)
176
+ if not val:
177
+ continue
178
+ if isinstance(val, dict):
179
+ d = val.get("date")
180
+ if not d:
181
+ continue
182
+ url = val.get("url")
183
+ date_html = f'<a class="glance-lifecycle-date" href="{_e(url)}" target="_blank" rel="noopener">{_e(d)} ↗</a>' if url else f'<span class="glance-lifecycle-date">{_e(d)}</span>'
184
+ else:
185
+ date_html = f'<span class="glance-lifecycle-date">{_e(val)}</span>'
186
+ lc_rows.append(f'<div class="glance-lifecycle-row"><span class="glance-lifecycle-label">{_e(label)}</span>{date_html}</div>')
187
+ has_lifecycle = bool(lc_rows)
188
+
189
+ # ── general info / doclinks (right column) ──
190
+ doclinks = []
191
+ repos = hit.get("repositories") or []
192
+ int_repos = [r for r in repos if (r.get("type") or "").split(".")[-1] == "INT"]
193
+ other_repos = [r for r in repos if (r.get("type") or "").split(".")[-1] != "INT"]
194
+
195
+ def _repo_label(r):
196
+ t = (r.get("type") or "").split(".")[-1]
197
+ if t in ("INT", "PAP", "CONF"):
198
+ return r["url"].rstrip("/").split("/")[-1]
199
+ return ""
200
+
201
+ if int_repos:
202
+ vals = "".join(f'<a href="{_e(r["url"])}" target="_blank" rel="noopener" class="gs-reflink">{_e(_repo_label(r))}</a>' for r in int_repos)
203
+ doclinks.append(f'<span class="gs-doclabel">INT NOTES</span><div class="gs-docvals">{vals}</div>')
204
+
205
+ support_docs = hit.get("supporting_docs") or []
206
+ if support_docs:
207
+ vals = "".join(f'<a href="{_e(u)}" target="_blank" rel="noopener" class="gs-docnum">[{i + 1}]</a>' for i, u in enumerate(support_docs))
208
+ doclinks.append(f'<span class="gs-doclabel">SUPPORT DOCS</span><div class="gs-docvals">{vals}</div>')
209
+
210
+ if other_repos:
211
+ parts = []
212
+ for i, r in enumerate(other_repos):
213
+ label = _repo_label(r)
214
+ t = (r.get("type") or "").split(".")[-1]
215
+ title_attr = _e(r.get("label") or _REPO_LABELS.get(t, t or "Repo"))
216
+ cls = "gs-reflink" if label else "gs-docnum"
217
+ text = _e(label) if label else f"[{i + 1}]"
218
+ parts.append(f'<a href="{_e(r["url"])}" target="_blank" rel="noopener" class="{cls}" title="{title_attr}">{text}</a>')
219
+ doclinks.append(f'<span class="gs-doclabel">REPOS</span><div class="gs-docvals">{"".join(parts)}</div>')
220
+
221
+ doclinks.append(f'<span class="gs-doclabel">GLANCE</span><div class="gs-docvals"><a href="{_e(glance_url)}" target="_blank" rel="noopener" class="gs-reflink">Glance ↗</a></div>')
222
+
223
+ paper_url = hit.get("paper_url")
224
+ if paper_url:
225
+ doclinks.append(f'<span class="gs-doclabel">PAPER</span><div class="gs-docvals"><a href="{_e(paper_url)}" target="_blank" rel="noopener" class="gs-reflink">Paper ↗</a></div>')
226
+
227
+ arxiv_url = hit.get("arxiv_url")
228
+ if arxiv_url:
229
+ doclinks.append(f'<span class="gs-doclabel">ARXIV</span><div class="gs-docvals"><a href="{_e(arxiv_url)}" target="_blank" rel="noopener" class="gs-reflink">arXiv ↗</a></div>')
230
+
231
+ related = hit.get("related_publications") or []
232
+ if related:
233
+ vals = "".join(
234
+ f'<a href="{_e(self._glance_url(self._by_ref[r]) if r in self._by_ref else f"{_GLANCE_BASE}/analyses/details.php?ref_code={r}")}" target="_blank" rel="noopener" class="gs-reflink">{_e(r)}</a>'
235
+ for r in related
236
+ )
237
+ doclinks.append(f'<span class="gs-doclabel">RELATED</span><div class="gs-docvals">{vals}</div>')
238
+
239
+ if hit.get("entry_type") == "Analysis":
240
+ fw = hit.get("analysis_framework") or {}
241
+ for fw_key, fw_label in (("ntupling", "NTUPLING"), ("histogramming", "HISTOGRAMMING")):
242
+ text = fw.get(fw_key)
243
+ if text:
244
+ doclinks.append(f'<span class="gs-doclabel">{fw_label}</span><div class="gs-docvals"><span class="glance-tag glance-tool-tag">{_e(text)}</span></div>')
245
+ else:
246
+ doclinks.append(f'<span class="gs-doclabel">{fw_label}</span><div class="gs-docvals"><span class="glance-tag glance-missing-tag">Missing information</span></div>')
247
+
248
+ stat_tools = hit.get("statistical_tools") or []
249
+ if stat_tools:
250
+ vals = "".join(f'<span class="glance-tag glance-tool-tag" title="{_e(t)}">{_e(t)}</span>' for t in stat_tools)
251
+ doclinks.append(f'<span class="gs-doclabel">STAT TOOLS</span><div class="gs-docvals">{vals}</div>')
252
+ else:
253
+ doclinks.append('<span class="gs-doclabel">STAT TOOLS</span><div class="gs-docvals"><span class="glance-tag glance-missing-tag">Missing information</span></div>')
254
+
255
+ mva_tools = hit.get("mva_ml_tools") or []
256
+ if mva_tools:
257
+ vals = "".join(f'<span class="glance-tag glance-tool-tag" title="{_e(t)}">{_e(t)}</span>' for t in mva_tools)
258
+ doclinks.append(f'<span class="gs-doclabel">MVA/ML</span><div class="gs-docvals">{vals}</div>')
259
+
260
+ keywords = hit.get("keywords") or []
261
+ if keywords:
262
+ vals = "".join(f'<span class="glance-tag glance-keyword-tag">{_e(k)}</span>' for k in keywords)
263
+ doclinks.append(f'<span class="gs-doclabel">KEYWORDS</span><div class="gs-docvals">{vals}</div>')
264
+
265
+ # ── triggers / datasets (full width) ──
266
+ extra_rows = []
267
+ triggers = hit.get("triggers") or []
268
+ if triggers:
269
+ vals = "".join(f'<span class="glance-tag glance-trigger-tag">{_e(t)}</span>' for t in triggers)
270
+ extra_rows.append(f'<div class="gs-triggers-row"><span class="glance-meta-label">Triggers</span> {vals}</div>')
271
+ datasets = hit.get("datasets") or []
272
+ if datasets:
273
+ vals = "".join(f'<li class="gs-plain-list-item" title="{_e(d)}">{_e(d)}</li>' for d in datasets)
274
+ extra_rows.append(f'<div class="gs-datasets-section"><span class="glance-meta-label">Datasets ({len(datasets)})</span><ul class="gs-plain-list">{vals}</ul></div>')
275
+
276
+ # ── team / editorial board ──
277
+ team = hit.get("team") or []
278
+ team_size = hit.get("team_size", len(team))
279
+ team_sorted = sorted(team, key=lambda m: 0 if m.get("is_contact_editor") else 1 if m.get("is_analysis_contact") else 2)
280
+ egroup_team = egroup_eb = None
281
+ if ref.startswith("ANA-"):
282
+ slug = ref[len("ANA-"):].lower()
283
+ egroup_team = f"atlas-ana-{slug}-analysis-team@cern.ch"
284
+ egroup_eb = f"atlas-ana-{slug}-edboard-conveners@cern.ch"
285
+
286
+ team_html = ""
287
+ if team:
288
+ rows = []
289
+ if egroup_team:
290
+ rows.append(f'<a href="mailto:{egroup_team}" class="gs-egroup-plain" title="{egroup_team}">{egroup_team.replace("@cern.ch", "")}</a>')
291
+ for m in team_sorted:
292
+ role = "EDITOR" if m.get("is_contact_editor") else "CONTACT" if m.get("is_analysis_contact") else "MEMBER"
293
+ name = _e(m.get("name", ""))
294
+ if m.get("email"):
295
+ rows.append(f'<div class="gs-team-row"><span class="gs-team-role">{role}</span><a href="mailto:{_e(m["email"])}" class="gs-team-email">{name}</a></div>')
296
+ else:
297
+ rows.append(f'<div class="gs-team-row"><span class="gs-team-role">{role}</span><span>{name}</span></div>')
298
+ eb = hit.get("editorial_board") or []
299
+ eb_rows = []
300
+ if eb:
301
+ if egroup_eb:
302
+ eb_rows.append(f'<a href="mailto:{egroup_eb}" class="gs-egroup-plain" title="{egroup_eb}">{egroup_eb.replace("@cern.ch", "")}</a>')
303
+ for m in eb:
304
+ role = "CHAIR" if m.get("is_chair") else "EX OFFICIO" if m.get("is_ex_officio") else "MEMBER"
305
+ name = _e(m.get("name", ""))
306
+ if m.get("email"):
307
+ eb_rows.append(f'<div class="gs-editor-row"><span class="gs-team-role">{role}</span><a href="mailto:{_e(m["email"])}" class="gs-team-email">{name}</a></div>')
308
+ else:
309
+ eb_rows.append(f'<div class="gs-editor-row"><span class="gs-team-role">{role}</span><span>{name}</span></div>')
310
+ elif hit.get("entry_type") == "Analysis":
311
+ eb_rows.append('<span class="gs-not-appointed">Not currently appointed</span>')
312
+
313
+ team_html = (
314
+ '<hr class="gs-team-divider">'
315
+ '<div class="gs-team-cols">'
316
+ f'<div class="gs-col-team"><div class="gs-section-label">Team ({team_size})</div>{"".join(rows)}</div>'
317
+ f'<div class="gs-col-editors"><div class="gs-section-label">Editorial Board{f" ({len(eb)})" if eb else ""}</div>{"".join(eb_rows)}</div>'
318
+ '</div>'
319
+ )
320
+
321
+ body_cols_cls = "gs-body-cols" if has_lifecycle else ""
322
+ lifecycle_col = (
323
+ f'<div class="gs-col-lifecycle"><div class="gs-section-label">Lifecycle</div>'
324
+ f'<div class="glance-lifecycle-body">{"".join(lc_rows)}</div></div>'
325
+ ) if has_lifecycle else ""
326
+ meta_col = f'<div class="gs-col-meta"><div class="gs-section-label">General Information</div><div class="gs-doclinks">{"".join(doclinks)}</div></div>'
327
+
328
+ circle_html = '<div class="source-circle" style="background: rgb(85, 139, 47);">GL</div>' if show_circle else ''
329
+
330
+ return (
331
+ '<div class="search-result">'
332
+ '<div class="card-row">'
333
+ f'{circle_html}'
334
+ '<div class="card-content">'
335
+ f'<div class="card-title"><a href="{_e(glance_url)}" target="_blank" rel="noopener"><span class="link-text">{title}</span></a>'
336
+ f'<span class="glance-badge {badge_cls}">{_e(badge_label)}</span></div>'
337
+ '<div class="card-breadcrumb glance-breadcrumb">'
338
+ f'<span class="glance-breadcrumb-left">{"".join(tags)}</span>'
339
+ f'<span class="glance-breadcrumb-right">{f"last updated: {_e(date_label)}" if date_label else ""}</span>'
340
+ '</div></div></div>'
341
+ '<hr>'
342
+ f'<div class="{body_cols_cls}">{lifecycle_col}{meta_col}</div>'
343
+ f'{"".join(extra_rows)}'
344
+ f'{team_html}'
345
+ '</div>'
346
+ )
347
+
348
+ # ── public API ───────────────────────────────────────────────────────
349
+
350
+ @staticmethod
351
+ def _year(hit: dict) -> str | None:
352
+ raw = hit.get("date") or hit.get("creation_date") or ""
353
+ return raw[:4] if len(raw) >= 4 and raw[:4].isdigit() else None
354
+
355
+ @staticmethod
356
+ def _is_closed(hit: dict) -> bool:
357
+ return (hit.get("phase") or "").strip().lower() == "closed"
358
+
359
+ @staticmethod
360
+ def _is_ongoing_analysis(hit: dict) -> bool:
361
+ """True for an Analysis genuinely in an active phase (phase text
362
+ contains "Active", e.g. "Phase 0 Active"). Deliberately stricter than
363
+ "not Closed" — plenty of analyses that have practically moved on
364
+ (into a paper/note) sit at "Phase 0 Finished" indefinitely without
365
+ ever being marked Closed in Glance, so "not Closed" alone
366
+ over-counts as active."""
367
+ return hit.get("entry_type") == "Analysis" and "active" in (hit.get("phase") or "").lower()
368
+
369
+ def _collect_personnel(self, code: str) -> dict[str, dict]:
370
+ people: dict[str, dict] = {}
371
+ for hit in self._hits_by_group.get(code, []):
372
+ year = self._year(hit)
373
+ ongoing = self._is_ongoing_analysis(hit)
374
+ for m in hit.get("team", []):
375
+ ccid = m.get("ccid")
376
+ if not ccid:
377
+ continue
378
+ p = people.setdefault(ccid, {"name": m.get("name", ""), "email": m.get("email"), "roles": set(), "count": 0, "years": set(), "active": False})
379
+ p["count"] += 1
380
+ if year:
381
+ p["years"].add(year)
382
+ if ongoing:
383
+ p["active"] = True
384
+ if m.get("is_contact_editor"):
385
+ p["roles"].add("Editor")
386
+ if m.get("is_analysis_contact"):
387
+ p["roles"].add("Contact")
388
+ if not p["roles"]:
389
+ p["roles"].add("Member")
390
+ return people
391
+
392
+ @staticmethod
393
+ def _render_personnel_table(people: list[dict]) -> str:
394
+ rows = sorted(people, key=lambda p: p["name"])
395
+ lines = ["| Name | Email | Roles | # entries | Years |", "|---|---|---|---|---|"]
396
+ for p in rows:
397
+ email = f'[{p["email"]}](mailto:{p["email"]})' if p.get("email") else ""
398
+ years = sorted(p["years"])
399
+ if not years:
400
+ years_label = ""
401
+ elif p["active"]:
402
+ # Still on an open analysis — leave the end open rather than
403
+ # implying they stopped after their latest recorded year.
404
+ years_label = f"{years[0]}–"
405
+ elif len(years) == 1:
406
+ years_label = years[0]
407
+ else:
408
+ years_label = f"{years[0]}–{years[-1]}"
409
+ lines.append(f'| {p["name"]} | {email} | {", ".join(sorted(p["roles"]))} | {p["count"]} | {years_label} |')
410
+ return "\n".join(lines)
411
+
412
+ def active_personnel(self, code: str) -> str:
413
+ """Active members — anyone on at least one genuinely ongoing Analysis
414
+ (see _is_ongoing_analysis). See personnel_history() for the full,
415
+ unfiltered roster."""
416
+ people = self._collect_personnel(code)
417
+ active = [p for p in people.values() if p["active"]]
418
+ return self._render_personnel_table(active)
419
+
420
+ def personnel_history(self, code: str) -> str:
421
+ """Every person who has ever appeared on a BPHY analysis, paper,
422
+ CONF note, or PUB note team — active or not. Whether an entry counts
423
+ as "Closed" is unreliable enough (see active_personnel()) that a
424
+ clean past/active split isn't trustworthy; this is the full roster
425
+ instead of a guessed complement."""
426
+ people = self._collect_personnel(code)
427
+ return self._render_personnel_table(list(people.values()))
428
+
429
+ @staticmethod
430
+ def _repo_links(hit: dict, types: set[str] | None) -> str:
431
+ """Markdown links for hit["repositories"] entries, optionally filtered
432
+ to a set of RepositoryType suffixes (e.g. {"INT"}); None means "all
433
+ types except INT" (the general "git repo" column)."""
434
+ repos = hit.get("repositories") or []
435
+ out = []
436
+ for r in repos:
437
+ t = (r.get("type") or "").split(".")[-1]
438
+ if types is not None and t not in types:
439
+ continue
440
+ if types is None and t == "INT":
441
+ continue
442
+ url = r.get("url")
443
+ if not url:
444
+ continue
445
+ label = url.rstrip("/").split("/")[-1]
446
+ out.append(f"[{label}]({url})")
447
+ return ", ".join(out)
448
+
449
+ @staticmethod
450
+ def _support_doc_links(hit: dict) -> str:
451
+ docs = hit.get("supporting_docs") or []
452
+ return ", ".join(f"[{i + 1}]({url})" for i, url in enumerate(docs))
453
+
454
+ def analyses(
455
+ self,
456
+ code: str,
457
+ view: str = "cards",
458
+ entry_type: str | None = None,
459
+ ongoing_only: bool = False,
460
+ show_circle: bool = True,
461
+ ) -> str:
462
+ hits = self._hits_by_group.get(code, [])
463
+ if entry_type:
464
+ hits = [h for h in hits if h.get("entry_type") == entry_type]
465
+ if ongoing_only:
466
+ hits = [h for h in hits if self._is_ongoing_analysis(h)]
467
+ if view == "list":
468
+ lines = [
469
+ "| Reference code | Title | Status | Leading group | Int Notes | Support Docs | Git Repo |",
470
+ "|---|---|---|---|---|---|---|",
471
+ ]
472
+ for hit in sorted(hits, key=lambda h: h.get("ref_code", "")):
473
+ status = hit.get("phase") or self._badge(hit)[0]
474
+ title = self._display_title(hit).replace("|", "\\|")
475
+ ref = hit.get("ref_code", "")
476
+ int_notes = self._repo_links(hit, {"INT"})
477
+ support_docs = self._support_doc_links(hit)
478
+ git_repo = self._repo_links(hit, None)
479
+ lines.append(
480
+ f'| [{ref}]({self._glance_url(hit)}) | {title} | {status} | {hit.get("leading_group", "")} '
481
+ f'| {int_notes} | {support_docs} | {git_repo} |'
482
+ )
483
+ return "\n".join(lines)
484
+ return "\n".join(self._render_card(hit, show_circle=show_circle) for hit in hits)
485
+
486
+ def analysis(self, ref_code: str) -> str:
487
+ hit = self._find(ref_code)
488
+ if not hit:
489
+ return f'<p><em>No Glance/STARE entry found for {_e(ref_code)}.</em></p>'
490
+ return self._render_card(hit)
@@ -0,0 +1,102 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from pathlib import Path
5
+
6
+ import yaml
7
+
8
+ _LINK_LABELS = {
9
+ "contacts": "Contacts",
10
+ "mandate": "Mandate",
11
+ "indico": "Indico",
12
+ "gitlab": "GitLab",
13
+ "mattermost": "Mattermost",
14
+ "jira": "Jira",
15
+ "glance": "Glance",
16
+ "egroup": "E-Group",
17
+ "atlastalk": "AtlasTalk",
18
+ "recommendations": "Recommendations",
19
+ }
20
+
21
+ _LINK_ICONS = {
22
+ "contacts": "lucide/mail",
23
+ "gitlab": "simple/gitlab",
24
+ "mattermost": "lucide/compass",
25
+ }
26
+
27
+
28
+ def _icon_md(icon: str) -> str:
29
+ """Convert 'lucide/info' -> ':lucide-info: '"""
30
+ if not icon:
31
+ return ""
32
+ return ":" + icon.replace("/", "-") + ": "
33
+
34
+
35
+ class GroupMacros:
36
+ def __init__(self, path: Path | str):
37
+ with open(path) as f:
38
+ data = yaml.safe_load(f)
39
+ group = data.get("group", {})
40
+ self._contacts = {k: v["contacts"] for k, v in group.items() if isinstance(v, dict) and "contacts" in v}
41
+ self._email = {k: v["email"] for k, v in group.items() if isinstance(v, dict) and "email" in v}
42
+ self._links = data.get("links", {})
43
+ self._default_icon = data.get("icon")
44
+
45
+ def contacts(self, key):
46
+ names = self._contacts.get(key, "")
47
+ if not names:
48
+ return ""
49
+ account = self._email.get(key, "")
50
+ if not account:
51
+ return names
52
+ address = f"{account}@cern.ch"
53
+ icon = _icon_md(_LINK_ICONS["contacts"])
54
+ return f'[{icon}{names}](mailto:{address} "{address}")'
55
+
56
+ def email(self, key):
57
+ value = self._email.get(key, "")
58
+ return f"[{value}](mailto:{value}@cern.ch)" if value else ""
59
+
60
+ @staticmethod
61
+ def _sanitize(text):
62
+ """Neutralize characters that would break markdown link/title syntax
63
+ (unescaped quotes ending a title early, brackets ending a label
64
+ early, embedded newlines) when text comes from free-form data
65
+ (Mattermost purposes, CDS policy text, etc.)."""
66
+ text = re.sub(r"\s+", " ", text).strip()
67
+ return text.replace('"', "'").replace("[", "(").replace("]", ")")
68
+
69
+ @classmethod
70
+ def _render_link(cls, value, fallback_label, default_icon=None):
71
+ """Render a link value as markdown. `value` is either a bare URL
72
+ string, or an object with a mandatory `url` (and optional `label` /
73
+ `description` overriding `fallback_label` and adding a hover title).
74
+ `default_icon` (a 'namespace/name' shortcode) is used unless the
75
+ entry itself sets its own `icon`."""
76
+ if isinstance(value, dict):
77
+ url = value.get("url", "")
78
+ if not url:
79
+ return ""
80
+ label = cls._sanitize(value.get("label") or fallback_label)
81
+ desc = value.get("description", "")
82
+ title = f' "{cls._sanitize(desc)}"' if desc else ""
83
+ icon = _icon_md(value.get("icon") or default_icon)
84
+ return f"[{icon}{label}]({url}{title})"
85
+ if not value:
86
+ return ""
87
+ icon = _icon_md(default_icon)
88
+ return f"[{icon}{cls._sanitize(fallback_label)}]({value})"
89
+
90
+ def locations(self, key, sub=None):
91
+ entry = self._links.get(key, "")
92
+ icon = _LINK_ICONS.get(key) or self._default_icon
93
+ if isinstance(entry, dict) and "url" not in entry:
94
+ # nested multi-variant map (e.g. recommendations); default to
95
+ # "latest" when no sub-key is given
96
+ sub = sub or "latest"
97
+ base = _LINK_LABELS.get(key, key.title())
98
+ return self._render_link(entry.get(sub, ""), f"{sub.title()} {base}", icon)
99
+ if sub is not None:
100
+ return ""
101
+ label = _LINK_LABELS.get(key, key.title())
102
+ return self._render_link(entry, label, icon)
@@ -0,0 +1,57 @@
1
+ from __future__ import annotations
2
+
3
+ """
4
+ zensical macros pluglet — loaded via `modules = ["atlasdocs_metadata.macros"]`
5
+ in a site's zensical.toml (see https://zensical.org/docs/setup/extensions/macros/#modules-zensicaltoml),
6
+ on top of that site's own module_name for anything site-specific.
7
+
8
+ Populates one macro variable per data_<CODE>.yml under data/metadata/ (as
9
+ synced by fetch_metadata.py) plus GLANCE if any stare_*.json.br is present.
10
+ A site with no data/metadata/ (fetch never run, or opted out) is a no-op.
11
+
12
+ NOTE: zensical enforces a "no watched paths outside the project folder"
13
+ policy for `modules` targets (and for venvs) — as of zensical 0.0.52,
14
+ violating it panics its watcher thread (crates/zensical/src/watcher.rs:299,
15
+ "invariant") on both `build` and `serve`, instead of a clean error. Confirmed
16
+ via a minimal repro: modules=[...] pointing at a file inside the project
17
+ folder builds fine; the identical config pointing outside it panics every
18
+ time. This only bites a *local* dev setup that installs atlasdocs-metadata
19
+ via a path override outside the consumer's project folder (as
20
+ atlas-docs-bphy/internal did before it had a real PyPI release) — a normal
21
+ install (into .pixi/envs/.../site-packages, which is inside the project
22
+ folder) should be fine. Until atlasdocs-metadata has a real release,
23
+ consumer sites should instead import GroupMacros/GlanceMacros directly from
24
+ their own module_name-loaded file and call the same loading logic as
25
+ define_env() below (see atlas-docs-bphy/internal/macros/atlasdocs_main.py
26
+ for the current workaround). Keep this module correct and in sync with that
27
+ workaround regardless — it's the intended path once there's a real release,
28
+ not dead code.
29
+ """
30
+
31
+ from pathlib import Path
32
+
33
+ import yaml
34
+
35
+ from .glance import GlanceMacros
36
+ from .group import GroupMacros
37
+
38
+ _VENDOR_DIRNAME = "metadata"
39
+
40
+
41
+ def _load_yaml(path: Path) -> dict:
42
+ with open(path) as f:
43
+ return yaml.safe_load(f) or {}
44
+
45
+
46
+ def define_env(env):
47
+ data_dir = Path.cwd() / "data" / _VENDOR_DIRNAME
48
+ if not data_dir.is_dir():
49
+ return
50
+
51
+ for path in sorted(data_dir.glob("data_*.yml")):
52
+ key = path.stem[len("data_"):]
53
+ data = _load_yaml(path)
54
+ env.variables[key] = GroupMacros(path) if isinstance(data, dict) and "group" in data else data
55
+
56
+ if any(data_dir.glob("stare_*.json.br")):
57
+ env.variables["GLANCE"] = GlanceMacros(data_dir)
@@ -0,0 +1,21 @@
1
+ [workspace]
2
+ name = "atlasdocs-metadata-dev"
3
+ description = "Development environment for the atlasdocs-metadata package"
4
+ channels = ["conda-forge"]
5
+ platforms = ["osx-arm64", "linux-64"]
6
+
7
+ [dependencies]
8
+ python = ">=3.11,<3.15"
9
+ uv = ">=0.4"
10
+ git = "*"
11
+ pyyaml = "*"
12
+ brotli-python = "*"
13
+
14
+ [tasks.build]
15
+ cmd = "uv build"
16
+ description = "Build wheel and sdist"
17
+
18
+ [tasks.newrelease]
19
+ cmd = "bash scripts/release.sh"
20
+ depends-on = []
21
+ description = "Bump version in pyproject.toml, build, and publish to PyPI"
@@ -0,0 +1,20 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "atlasdocs-metadata"
7
+ version = "0.9.0"
8
+ description = "Syncs the atlasdocs/metadata registry (data_<CODE>.yml / stare_<CODE>.json.br, every group) and renders it via GroupMacros/GlanceMacros for atlasdocs-theme sites"
9
+ authors = [
10
+ { name = "Jason Oliver", email = "jason.oliver@cern.ch" }
11
+ ]
12
+ license = "Apache-2.0"
13
+ requires-python = ">=3.9"
14
+ dependencies = ["pyyaml>=6", "brotli>=1.1"]
15
+
16
+ [tool.hatch.build.targets.wheel]
17
+ include = ["atlasdocs_metadata"]
18
+
19
+ [project.scripts]
20
+ fetch-metadata = "atlasdocs_metadata.fetch_metadata:main"
@@ -0,0 +1,79 @@
1
+ #!/usr/bin/env bash
2
+ set -euo pipefail
3
+
4
+ PYPROJECT="$(dirname "$0")/../pyproject.toml"
5
+ PYPROJECT="$(realpath "$PYPROJECT")"
6
+ PKG_DIR="$(dirname "$PYPROJECT")"
7
+
8
+ current_version() {
9
+ grep '^version' "$PYPROJECT" | sed 's/version = "\(.*\)"/\1/'
10
+ }
11
+
12
+ bump_version() {
13
+ local part="${1:-patch}"
14
+ local ver
15
+ ver="$(current_version)"
16
+ IFS='.' read -r major minor patch <<< "$ver"
17
+ case "$part" in
18
+ major) major=$((major + 1)); minor=0; patch=0 ;;
19
+ minor) minor=$((minor + 1)); patch=0 ;;
20
+ patch) patch=$((patch + 1)) ;;
21
+ *) echo "Unknown part: $part (use major|minor|patch)"; exit 1 ;;
22
+ esac
23
+ echo "${major}.${minor}.${patch}"
24
+ }
25
+
26
+ set_version() {
27
+ local new="$1"
28
+ sed -i '' "s/^version = \".*\"/version = \"${new}\"/" "$PYPROJECT"
29
+ echo "Version set to ${new}"
30
+ }
31
+
32
+ PART="${1:-}"
33
+ SKIP_PUBLISH="${2:-}"
34
+ CURRENT="$(current_version)"
35
+
36
+ if [[ -z "$PART" ]]; then
37
+ echo "Current version: ${CURRENT}"
38
+ echo "Increment: [1] patch [2] minor [3] major"
39
+ read -r -p "Choice [1]: " choice
40
+ case "${choice:-1}" in
41
+ 1|patch) PART="patch" ;;
42
+ 2|minor) PART="minor" ;;
43
+ 3|major) PART="major" ;;
44
+ *) echo "Invalid choice."; exit 1 ;;
45
+ esac
46
+ fi
47
+
48
+ if [[ "$PART" == exact:* ]]; then
49
+ NEW_VERSION="${PART#exact:}"
50
+ else
51
+ NEW_VERSION="$(bump_version "$PART")"
52
+ fi
53
+
54
+ echo "Current: ${CURRENT} → New: ${NEW_VERSION} (${PART})"
55
+ read -r -p "Continue? [y/N] " confirm
56
+ [[ "$confirm" =~ ^[Yy]$ ]] || { echo "Aborted."; exit 0; }
57
+
58
+ set_version "$NEW_VERSION"
59
+
60
+ echo ""
61
+ echo "▶ uv build"
62
+ uv build --project "$PKG_DIR"
63
+
64
+ if [[ "$SKIP_PUBLISH" != "--no-publish" ]]; then
65
+ echo ""
66
+ echo "▶ uv publish"
67
+ uv publish dist/atlasdocs_metadata-"${NEW_VERSION}"*.whl dist/atlasdocs_metadata-"${NEW_VERSION}".tar.gz
68
+ fi
69
+
70
+ echo ""
71
+ echo "▶ git add + commit + tag"
72
+ REPO_ROOT="$(git -C "$PKG_DIR" rev-parse --show-toplevel)"
73
+ git -C "$REPO_ROOT" add "$PYPROJECT"
74
+ git -C "$REPO_ROOT" commit -m "chore(metadata): release ${NEW_VERSION}"
75
+ git -C "$REPO_ROOT" tag "atlasdocs-metadata/v${NEW_VERSION}"
76
+
77
+ echo ""
78
+ echo "✓ Released atlasdocs-metadata ${NEW_VERSION}"
79
+ echo " Push with: git push && git push --tags"