viva-catalog 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,65 @@
1
+ name: Build ecosystem index
2
+
3
+ # Rebuilds ecosystem-index.json by harvesting every repo's published workbench
4
+ # dashboard, commits it back to main, and publishes modules.json + the index to
5
+ # gh-pages so any published workbench can fetch them same-origin.
6
+ on:
7
+ workflow_dispatch:
8
+ schedule:
9
+ - cron: "0 6 * * *" # daily
10
+ push:
11
+ branches: [main]
12
+ paths:
13
+ - "viva_catalog/modules.json"
14
+ - "scripts/build_ecosystem_index.py"
15
+
16
+ permissions:
17
+ contents: write
18
+
19
+ concurrency:
20
+ group: build-index
21
+ cancel-in-progress: true
22
+
23
+ jobs:
24
+ build:
25
+ runs-on: ubuntu-latest
26
+ steps:
27
+ - uses: actions/checkout@v4
28
+ - uses: actions/setup-python@v5
29
+ with:
30
+ python-version: "3.12"
31
+
32
+ - name: Install PyYAML (for study/investigation/composite YAML scanning)
33
+ run: python -m pip install --quiet pyyaml
34
+
35
+ - name: Discover repos (viva-marketplace topic) + clone/scan → index
36
+ env:
37
+ # discovery queries the org for public repos with the viva-marketplace
38
+ # topic; the default workflow token authenticates the search API.
39
+ GITHUB_TOKEN: ${{ github.token }}
40
+ run: python scripts/build_ecosystem_index.py
41
+
42
+ - name: Commit the refreshed registry + index (if changed)
43
+ run: |
44
+ git config user.name "github-actions[bot]"
45
+ git config user.email "github-actions[bot]@users.noreply.github.com"
46
+ git add viva_catalog/modules.json viva_catalog/ecosystem-index.json
47
+ if ! git diff --cached --quiet; then
48
+ git commit -m "chore: refresh registry (topic discovery) + ecosystem-index [skip ci]"
49
+ git push
50
+ else
51
+ echo "registry + index unchanged"
52
+ fi
53
+
54
+ - name: Assemble the public ledger (modules + index)
55
+ run: |
56
+ mkdir -p public
57
+ cp viva_catalog/modules.json public/modules.json
58
+ cp viva_catalog/ecosystem-index.json public/ecosystem-index.json
59
+
60
+ - name: Publish ledger to gh-pages
61
+ uses: peaceiris/actions-gh-pages@v4
62
+ with:
63
+ github_token: ${{ secrets.GITHUB_TOKEN }}
64
+ publish_dir: ./public
65
+ keep_files: false
@@ -0,0 +1,20 @@
1
+ name: release
2
+ on:
3
+ push:
4
+ tags: ["v*"]
5
+ workflow_dispatch: {} # manual fallback (tag-push events have been unreliable on this repo)
6
+ permissions:
7
+ id-token: write # for PyPI trusted publishing
8
+ jobs:
9
+ publish:
10
+ runs-on: ubuntu-latest
11
+ steps:
12
+ - uses: actions/checkout@v4
13
+ - uses: actions/setup-python@v5
14
+ with: {python-version: "3.11"}
15
+ - name: Install uv
16
+ run: pip install uv
17
+ - name: Build
18
+ run: uv build
19
+ - name: Publish to PyPI
20
+ uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,20 @@
1
+ name: Validate registry
2
+
3
+ # PR gate: any change to the repo registry (modules.json) is validated before it
4
+ # can merge. Adding/updating a repo goes through a pull request — see CONTRIBUTING.md.
5
+ on:
6
+ pull_request:
7
+ paths:
8
+ - "viva_catalog/modules.json"
9
+ - "scripts/validate_modules.py"
10
+
11
+ jobs:
12
+ validate:
13
+ runs-on: ubuntu-latest
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: actions/setup-python@v5
17
+ with:
18
+ python-version: "3.12"
19
+ - name: Validate modules.json
20
+ run: python scripts/validate_modules.py
@@ -0,0 +1,7 @@
1
+ __pycache__/
2
+ *.pyc
3
+ dist/
4
+ build/
5
+ *.egg-info/
6
+ public/
7
+ .venv/
@@ -0,0 +1,65 @@
1
+ # Contributing to viva-catalog
2
+
3
+ viva-catalog is the **ecosystem ledger** for the vivarium / process-bigraph
4
+ workbench. Both files are now **machine-generated** — you don't hand-edit them:
5
+
6
+ - `viva_catalog/modules.json` — the registry of repos, **discovered from
7
+ GitHub topics** (see below).
8
+ - `viva_catalog/ecosystem-index.json` — the aggregated artifact index,
9
+ built by cloning + scanning each discovered repo.
10
+
11
+ ## Publish your repository to the marketplace
12
+
13
+ **Add the `viva-marketplace` GitHub topic to your repo.** That's the whole step:
14
+
15
+ ```bash
16
+ gh repo edit vivarium-collective/<your-repo> --add-topic viva-marketplace
17
+ ```
18
+
19
+ (Or add it via the repo's GitHub page → About → ⚙ → Topics.) Your repo must be
20
+ **public** and not archived.
21
+
22
+ The nightly builder (and any manual re-run) then discovers every public,
23
+ non-archived `vivarium-collective` repo carrying the topic, refreshes
24
+ `modules.json` from that set, and **clones your repo and scans its source** — your
25
+ composites (`@composite_generator` / `*.composite.yaml`), `Process`/`Step`
26
+ subclasses, `studies/<slug>/study.yaml`, and
27
+ `investigations/<slug>/investigation.yaml` all appear in `ecosystem-index.json`
28
+ automatically. You never hand-write your artifact list.
29
+
30
+ To make your artifacts discoverable, follow the usual conventions:
31
+ `@composite_generator(name=…, description=…)` for composites, `Process`/`Step`
32
+ subclasses (with a `description` attribute or docstring) for processes, and the
33
+ `studies/`/`investigations/` YAML for those.
34
+
35
+ ## Update your listing
36
+
37
+ Your `description` and `tags` come straight from your repo's **GitHub description
38
+ and topics** — edit them on GitHub and the next build picks them up. The artifact
39
+ index refreshes from your repo's source, so just keep your source current.
40
+
41
+ ## Remove your repository
42
+
43
+ Remove the `viva-marketplace` topic (or archive / make the repo private) — it
44
+ drops out of the registry on the next build:
45
+
46
+ ```bash
47
+ gh repo edit vivarium-collective/<your-repo> --remove-topic viva-marketplace
48
+ ```
49
+
50
+ ## Ground rules
51
+
52
+ - **Membership = the `viva-marketplace` topic on a public, non-archived repo.**
53
+ No hand-maintained list to drift.
54
+ - Both `modules.json` and `ecosystem-index.json` are generated — don't hand-edit
55
+ them (changes are overwritten on the next build).
56
+
57
+ ## Rebuild locally
58
+
59
+ ```bash
60
+ pip install pyyaml
61
+ export GITHUB_TOKEN=$(gh auth token) # discovery authenticates the search API
62
+ python scripts/build_ecosystem_index.py # discover + clone/scan all
63
+ python scripts/build_ecosystem_index.py --no-discover # use committed modules.json as-is
64
+ python scripts/build_ecosystem_index.py --only viva-biofilm # just one repo
65
+ ```
@@ -0,0 +1,79 @@
1
+ Metadata-Version: 2.5
2
+ Name: viva-catalog
3
+ Version: 0.1.0
4
+ Summary: Ecosystem ledger for the vivarium / process-bigraph workbench: the registry of repos + an aggregated artifact index.
5
+ Project-URL: Homepage, https://github.com/vivarium-collective/viva-catalog
6
+ Author: Vivarium Collective
7
+ License: MIT
8
+ Keywords: marketplace,process-bigraph,vivarium,workbench
9
+ Requires-Python: >=3.9
10
+ Description-Content-Type: text/markdown
11
+
12
+ # viva-catalog
13
+
14
+ The **ecosystem ledger** for the [vivarium](https://github.com/vivarium-collective) /
15
+ [process-bigraph](https://github.com/vivarium-collective/process-bigraph) workbench.
16
+
17
+ It answers two questions for the [vivarium-workbench](https://github.com/vivarium-collective/vivarium-workbench)
18
+ Registry, so a workbench can browse the **whole ecosystem** — not just what's
19
+ installed locally — and offer **Install to use**:
20
+
21
+ | File | What it is |
22
+ |---|---|
23
+ | [`viva_catalog/modules.json`](viva_catalog/modules.json) | The registry of ecosystem repos — `name`, `source`, `description`, `tags`. **Generated** by discovering every public `vivarium-collective` repo with the `viva-marketplace` GitHub topic. |
24
+ | [`viva_catalog/ecosystem-index.json`](viva_catalog/ecosystem-index.json) | The aggregated **artifact index** — every repo's processes / steps / composites / studies / investigations (name + description + counts). Regenerated by CI. |
25
+
26
+ Previously the registry lived in `viva_superpowers/catalog/modules.json` (the
27
+ Claude-Code plugin). It moved here so the ledger is owned by a dedicated repo.
28
+
29
+ **Adding a repo?** Just add the **`viva-marketplace` GitHub topic** to your
30
+ public repo (`gh repo edit vivarium-collective/<repo> --add-topic
31
+ viva-marketplace`) — the nightly builder discovers it, refreshes `modules.json`,
32
+ and scans its source into the index. No PR needed. See
33
+ [CONTRIBUTING.md](CONTRIBUTING.md).
34
+
35
+ ## How the index is built
36
+
37
+ `scripts/build_ecosystem_index.py` reads `modules.json` and, for each repo,
38
+ **shallow-clones it and scans the source** — no published dashboard required:
39
+
40
+ - **composites** — `@composite_generator(name=…, description=…)` decorators (AST)
41
+ + any `*.composite.yaml` files
42
+ - **processes / steps** — top-level classes whose base ends in `Process` / `Step`
43
+ (AST), described by a `description` class attribute or the class docstring
44
+ - **studies** — `**/studies/*/study.yaml` (name + objective/title)
45
+ - **investigations** — `**/investigations/*/investigation.yaml` (name + title)
46
+
47
+ This gives complete coverage across the ecosystem whether or not a repo publishes
48
+ a workbench dashboard. Repos that can't be cloned are still listed (empty
49
+ artifacts, `cloned: false`).
50
+
51
+ ```bash
52
+ python scripts/build_ecosystem_index.py # rebuild the index locally
53
+ python scripts/build_ecosystem_index.py --only Viva-munk,pbg-copasi # subset
54
+ ```
55
+
56
+ Needs `git` + `PyYAML`.
57
+
58
+ ## Consuming the ledger
59
+
60
+ **Python** (viva-superpowers, vivarium-workbench):
61
+
62
+ ```python
63
+ import viva_catalog
64
+ repos = viva_catalog.load_modules() # the repo registry
65
+ index = viva_catalog.load_ecosystem_index() # aggregated artifacts
66
+ ```
67
+
68
+ **Over HTTP** (published to gh-pages, same-origin for any published workbench):
69
+
70
+ ```
71
+ https://vivarium-collective.github.io/viva-catalog/modules.json
72
+ https://vivarium-collective.github.io/viva-catalog/ecosystem-index.json
73
+ ```
74
+
75
+ ## CI
76
+
77
+ [`.github/workflows/build-index.yml`](.github/workflows/build-index.yml) rebuilds
78
+ `ecosystem-index.json` daily (and on `modules.json` changes), commits it to `main`,
79
+ and publishes `modules.json` + the index to `gh-pages`.
@@ -0,0 +1,68 @@
1
+ # viva-catalog
2
+
3
+ The **ecosystem ledger** for the [vivarium](https://github.com/vivarium-collective) /
4
+ [process-bigraph](https://github.com/vivarium-collective/process-bigraph) workbench.
5
+
6
+ It answers two questions for the [vivarium-workbench](https://github.com/vivarium-collective/vivarium-workbench)
7
+ Registry, so a workbench can browse the **whole ecosystem** — not just what's
8
+ installed locally — and offer **Install to use**:
9
+
10
+ | File | What it is |
11
+ |---|---|
12
+ | [`viva_catalog/modules.json`](viva_catalog/modules.json) | The registry of ecosystem repos — `name`, `source`, `description`, `tags`. **Generated** by discovering every public `vivarium-collective` repo with the `viva-marketplace` GitHub topic. |
13
+ | [`viva_catalog/ecosystem-index.json`](viva_catalog/ecosystem-index.json) | The aggregated **artifact index** — every repo's processes / steps / composites / studies / investigations (name + description + counts). Regenerated by CI. |
14
+
15
+ Previously the registry lived in `viva_superpowers/catalog/modules.json` (the
16
+ Claude-Code plugin). It moved here so the ledger is owned by a dedicated repo.
17
+
18
+ **Adding a repo?** Just add the **`viva-marketplace` GitHub topic** to your
19
+ public repo (`gh repo edit vivarium-collective/<repo> --add-topic
20
+ viva-marketplace`) — the nightly builder discovers it, refreshes `modules.json`,
21
+ and scans its source into the index. No PR needed. See
22
+ [CONTRIBUTING.md](CONTRIBUTING.md).
23
+
24
+ ## How the index is built
25
+
26
+ `scripts/build_ecosystem_index.py` reads `modules.json` and, for each repo,
27
+ **shallow-clones it and scans the source** — no published dashboard required:
28
+
29
+ - **composites** — `@composite_generator(name=…, description=…)` decorators (AST)
30
+ + any `*.composite.yaml` files
31
+ - **processes / steps** — top-level classes whose base ends in `Process` / `Step`
32
+ (AST), described by a `description` class attribute or the class docstring
33
+ - **studies** — `**/studies/*/study.yaml` (name + objective/title)
34
+ - **investigations** — `**/investigations/*/investigation.yaml` (name + title)
35
+
36
+ This gives complete coverage across the ecosystem whether or not a repo publishes
37
+ a workbench dashboard. Repos that can't be cloned are still listed (empty
38
+ artifacts, `cloned: false`).
39
+
40
+ ```bash
41
+ python scripts/build_ecosystem_index.py # rebuild the index locally
42
+ python scripts/build_ecosystem_index.py --only Viva-munk,pbg-copasi # subset
43
+ ```
44
+
45
+ Needs `git` + `PyYAML`.
46
+
47
+ ## Consuming the ledger
48
+
49
+ **Python** (viva-superpowers, vivarium-workbench):
50
+
51
+ ```python
52
+ import viva_catalog
53
+ repos = viva_catalog.load_modules() # the repo registry
54
+ index = viva_catalog.load_ecosystem_index() # aggregated artifacts
55
+ ```
56
+
57
+ **Over HTTP** (published to gh-pages, same-origin for any published workbench):
58
+
59
+ ```
60
+ https://vivarium-collective.github.io/viva-catalog/modules.json
61
+ https://vivarium-collective.github.io/viva-catalog/ecosystem-index.json
62
+ ```
63
+
64
+ ## CI
65
+
66
+ [`.github/workflows/build-index.yml`](.github/workflows/build-index.yml) rebuilds
67
+ `ecosystem-index.json` daily (and on `modules.json` changes), commits it to `main`,
68
+ and publishes `modules.json` + the index to `gh-pages`.
@@ -0,0 +1,25 @@
1
+ [build-system]
2
+ requires = ["hatchling"]
3
+ build-backend = "hatchling.build"
4
+
5
+ [project]
6
+ name = "viva-catalog"
7
+ version = "0.1.0"
8
+ description = "Ecosystem ledger for the vivarium / process-bigraph workbench: the registry of repos + an aggregated artifact index."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ authors = [{ name = "Vivarium Collective" }]
13
+ keywords = ["vivarium", "process-bigraph", "workbench", "marketplace"]
14
+
15
+ [project.urls]
16
+ Homepage = "https://github.com/vivarium-collective/viva-catalog"
17
+
18
+ [tool.hatch.build.targets.wheel]
19
+ packages = ["viva_catalog", "viva_marketplace"]
20
+
21
+ # Ship the ledger data files inside the wheel so consumers can read them via
22
+ # viva_catalog.load_modules() / load_ecosystem_index().
23
+ [tool.hatch.build.targets.wheel.force-include]
24
+ "viva_catalog/modules.json" = "viva_catalog/modules.json"
25
+ "viva_catalog/ecosystem-index.json" = "viva_catalog/ecosystem-index.json"
@@ -0,0 +1,304 @@
1
+ #!/usr/bin/env python3
2
+ """Build ``viva_catalog/ecosystem-index.json`` from the repo registry.
3
+
4
+ For every repo in ``modules.json`` we **shallow-clone the repo and scan its
5
+ source** — no published dashboard required. What we extract:
6
+
7
+ - **composites** — ``@composite_generator(name=…, description=…)`` decorators
8
+ (AST) plus any ``*.composite.yaml`` files.
9
+ - **processes / steps** — top-level classes whose base ends in ``Process`` /
10
+ ``Step`` (AST), with the description taken from a ``description`` class
11
+ attribute or the class docstring.
12
+ - **studies** — ``**/studies/*/study.yaml`` (name + objective/title).
13
+ - **investigations** — ``**/investigations/*/investigation.yaml`` (name + title).
14
+
15
+ This gives complete coverage across the ecosystem regardless of whether a repo
16
+ publishes a workbench dashboard. Repos that can't be cloned are still listed
17
+ (empty artifacts, ``cloned: false``).
18
+
19
+ Usage: python scripts/build_ecosystem_index.py [--out PATH] [--timeout N] [--jobs N]
20
+ Needs: git + PyYAML.
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import argparse
25
+ import ast
26
+ import json
27
+ import os
28
+ import re
29
+ import subprocess
30
+ import sys
31
+ import tempfile
32
+ from datetime import datetime, timezone
33
+ from pathlib import Path
34
+
35
+ try:
36
+ import yaml
37
+ except ImportError: # pragma: no cover
38
+ yaml = None
39
+
40
+ ROOT = Path(__file__).resolve().parent.parent
41
+ PKG = ROOT / "viva_catalog"
42
+
43
+ _PROC_BASE = re.compile(r"(Process|Step)$")
44
+
45
+ # A repo "publishes to the marketplace" by adding this GitHub topic. Membership
46
+ # is discovered from the org (no hand-maintained list) — see discover_modules.
47
+ MARKETPLACE_TOPIC = "viva-marketplace"
48
+ ORG = "vivarium-collective"
49
+
50
+
51
+ def discover_modules(timeout: float = 60.0) -> list[dict]:
52
+ """Discover the marketplace registry from GitHub: every PUBLIC, non-archived
53
+ ``vivarium-collective`` repo carrying the ``viva-marketplace`` topic.
54
+
55
+ This replaces the hand-maintained ``modules.json`` membership list — a repo
56
+ joins the marketplace by adding the topic (``gh repo edit --add-topic
57
+ viva-marketplace``), and the daily index build picks it up automatically.
58
+ Returns registry entries in the same shape modules.json used
59
+ (name/source/ref/package/homepage/description/tags), sorted by name.
60
+ """
61
+ import urllib.parse
62
+ import urllib.request
63
+
64
+ token = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
65
+ q = f"org:{ORG} topic:{MARKETPLACE_TOPIC} archived:false"
66
+ modules: list[dict] = []
67
+ page = 1
68
+ while True:
69
+ url = ("https://api.github.com/search/repositories?q="
70
+ + urllib.parse.quote(q)
71
+ + f"&per_page=100&page={page}&sort=full_name&order=asc")
72
+ headers = {"Accept": "application/vnd.github+json",
73
+ "User-Agent": "viva-marketplace-index"}
74
+ if token:
75
+ headers["Authorization"] = f"Bearer {token}"
76
+ req = urllib.request.Request(url, headers=headers)
77
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
78
+ data = json.loads(resp.read())
79
+ items = data.get("items", []) or []
80
+ for it in items:
81
+ modules.append({
82
+ "name": it["name"],
83
+ "source": it.get("clone_url") or f"{it['html_url']}.git",
84
+ "ref": it.get("default_branch") or "main",
85
+ "package": it["name"].replace("-", "_"),
86
+ "homepage": it.get("html_url"),
87
+ "description": it.get("description") or "",
88
+ "tags": sorted(t for t in (it.get("topics") or [])
89
+ if t != MARKETPLACE_TOPIC),
90
+ })
91
+ total = int(data.get("total_count") or 0)
92
+ if len(items) < 100 or len(modules) >= total:
93
+ break
94
+ page += 1
95
+ modules.sort(key=lambda m: m["name"].lower())
96
+ return modules
97
+
98
+
99
+ def _org_repo(source: str) -> tuple[str, str]:
100
+ s = re.sub(r"\.git$", "", (source or "").strip())
101
+ s = re.sub(r"^git@github\.com:", "https://github.com/", s)
102
+ m = re.search(r"github\.com[/:]([^/]+)/([^/]+)/?$", s)
103
+ return (m.group(1), m.group(2)) if m else ("vivarium-collective", s.rsplit("/", 1)[-1])
104
+
105
+
106
+ def _clone(url: str, ref: str, dest: Path, timeout: float) -> bool:
107
+ """Shallow-clone url@ref into dest. Falls back to the default branch if the
108
+ ref doesn't exist. Returns True on success."""
109
+ base = ["git", "clone", "--depth", "1", "--quiet"]
110
+ for args in ([*base, "--branch", ref, url, str(dest)] if ref else None,
111
+ [*base, url, str(dest)]):
112
+ if args is None:
113
+ continue
114
+ try:
115
+ subprocess.run(args, check=True, capture_output=True, text=True, timeout=timeout)
116
+ return True
117
+ except (subprocess.CalledProcessError, subprocess.TimeoutExpired):
118
+ continue
119
+ return False
120
+
121
+
122
+ def _class_description(node: ast.ClassDef) -> str:
123
+ # Prefer a `description = "..."` class attribute (pbg convention), else the
124
+ # first line of the docstring.
125
+ for stmt in node.body:
126
+ if (isinstance(stmt, ast.Assign)
127
+ and any(isinstance(t, ast.Name) and t.id == "description" for t in stmt.targets)
128
+ and isinstance(stmt.value, ast.Constant) and isinstance(stmt.value.value, str)):
129
+ return stmt.value.value.strip().splitlines()[0]
130
+ doc = ast.get_docstring(node) or ""
131
+ return doc.strip().splitlines()[0] if doc.strip() else ""
132
+
133
+
134
+ def _kw_str(call: ast.Call, name: str) -> str:
135
+ for kw in call.keywords:
136
+ if kw.arg == name and isinstance(kw.value, ast.Constant) and isinstance(kw.value.value, str):
137
+ return kw.value.value
138
+ return ""
139
+
140
+
141
+ def _scan_python(root: Path) -> tuple[list, list, list]:
142
+ composites, processes, steps = [], [], []
143
+ seen_c, seen_p, seen_s = set(), set(), set()
144
+ for py in root.rglob("*.py"):
145
+ # Skip vendored / test / build noise.
146
+ parts = set(py.parts)
147
+ if parts & {".git", "tests", "test", "build", "dist", "node_modules", ".venv", "venv"}:
148
+ continue
149
+ try:
150
+ tree = ast.parse(py.read_text(encoding="utf-8", errors="ignore"))
151
+ except (SyntaxError, ValueError):
152
+ continue
153
+ for node in ast.walk(tree):
154
+ if isinstance(node, ast.ClassDef):
155
+ for b in node.bases:
156
+ bname = b.attr if isinstance(b, ast.Attribute) else (b.id if isinstance(b, ast.Name) else "")
157
+ if not _PROC_BASE.search(bname or "") or node.name in ("Process", "Step"):
158
+ continue
159
+ if bname.endswith("Step"):
160
+ if node.name not in seen_s:
161
+ seen_s.add(node.name)
162
+ steps.append({"name": node.name, "description": _class_description(node)})
163
+ else:
164
+ if node.name not in seen_p:
165
+ seen_p.add(node.name)
166
+ processes.append({"name": node.name, "description": _class_description(node)})
167
+ break
168
+ elif isinstance(node, ast.Call):
169
+ fn = node.func
170
+ fname = fn.attr if isinstance(fn, ast.Attribute) else (fn.id if isinstance(fn, ast.Name) else "")
171
+ if fname == "composite_generator":
172
+ nm = _kw_str(node, "name")
173
+ if nm and nm not in seen_c:
174
+ seen_c.add(nm)
175
+ composites.append({"name": nm, "description": _kw_str(node, "description")})
176
+ # *.composite.yaml files
177
+ for cy in root.rglob("*.composite.yaml"):
178
+ if ".git" in cy.parts:
179
+ continue
180
+ nm = cy.name[: -len(".composite.yaml")]
181
+ desc = ""
182
+ if yaml:
183
+ try:
184
+ d = yaml.safe_load(cy.read_text(encoding="utf-8")) or {}
185
+ nm = d.get("name") or nm
186
+ desc = d.get("description") or ""
187
+ except Exception: # noqa: BLE001
188
+ pass
189
+ if nm not in seen_c:
190
+ seen_c.add(nm)
191
+ composites.append({"name": nm, "description": desc})
192
+ return composites, processes, steps
193
+
194
+
195
+ def _scan_specs(root: Path, kind: str, spec_name: str, desc_keys) -> list:
196
+ out, seen = [], set()
197
+ for spec in root.rglob(f"{kind}/*/{spec_name}"):
198
+ if ".git" in spec.parts:
199
+ continue
200
+ name, desc = spec.parent.name, ""
201
+ if yaml:
202
+ try:
203
+ d = yaml.safe_load(spec.read_text(encoding="utf-8")) or {}
204
+ name = d.get("name") or name
205
+ for k in desc_keys:
206
+ v = d.get(k)
207
+ if isinstance(v, dict):
208
+ v = v.get("question") or v.get("objective")
209
+ if v:
210
+ desc = str(v); break
211
+ except Exception: # noqa: BLE001
212
+ pass
213
+ if name not in seen:
214
+ seen.add(name)
215
+ out.append({"name": name, "description": desc})
216
+ return out
217
+
218
+
219
+ def harvest_repo(module: dict, timeout: float) -> dict:
220
+ name = module.get("name") or module.get("package") or ""
221
+ source = module.get("source") or module.get("homepage") or ""
222
+ ref = module.get("ref") or ""
223
+ org, repo = _org_repo(source)
224
+ entry = {
225
+ "name": name, "repo": repo, "source": re.sub(r"\.git$", "", source),
226
+ "homepage": module.get("homepage") or f"https://github.com/{org}/{repo}",
227
+ "description": module.get("description") or "", "tags": module.get("tags") or [],
228
+ "cloned": False, "processes": [], "steps": [], "composites": [],
229
+ "studies": [], "investigations": [],
230
+ }
231
+ with tempfile.TemporaryDirectory() as td:
232
+ dest = Path(td) / repo
233
+ if _clone(re.sub(r"^git@github\.com:", "https://github.com/", source) or f"https://github.com/{org}/{repo}.git",
234
+ ref, dest, timeout):
235
+ entry["cloned"] = True
236
+ comps, procs, steps = _scan_python(dest)
237
+ entry["composites"] = comps
238
+ entry["processes"] = procs
239
+ entry["steps"] = steps
240
+ entry["studies"] = _scan_specs(dest, "studies", "study.yaml",
241
+ ("objective", "purpose", "title", "description"))
242
+ entry["investigations"] = _scan_specs(dest, "investigations", "investigation.yaml",
243
+ ("title", "description", "objective"))
244
+ entry["counts"] = {k: len(entry[k]) for k in
245
+ ("processes", "steps", "composites", "studies", "investigations")}
246
+ return entry
247
+
248
+
249
+ def main(argv=None) -> int:
250
+ ap = argparse.ArgumentParser(description=__doc__)
251
+ ap.add_argument("--out", default=str(PKG / "ecosystem-index.json"))
252
+ ap.add_argument("--timeout", type=float, default=120.0)
253
+ ap.add_argument("--stamp", default=None)
254
+ ap.add_argument("--only", default=None, help="comma-separated repo names to limit (debug)")
255
+ ap.add_argument("--no-discover", action="store_true",
256
+ help="skip GitHub topic discovery; use the committed modules.json as-is "
257
+ "(offline / no token)")
258
+ args = ap.parse_args(argv)
259
+
260
+ # Discover the registry from the `viva-marketplace` GitHub topic and refresh
261
+ # the committed modules.json (the generated cache). Fall back to the committed
262
+ # list if discovery fails (offline / API error) so the build never breaks.
263
+ if not args.no_discover:
264
+ try:
265
+ discovered = discover_modules(args.timeout)
266
+ (PKG / "modules.json").write_text(
267
+ json.dumps(discovered, indent=2) + "\n", encoding="utf-8")
268
+ print(f"discovered {len(discovered)} repos via topic "
269
+ f"'{MARKETPLACE_TOPIC}' -> refreshed modules.json", file=sys.stderr)
270
+ except Exception as e: # noqa: BLE001
271
+ print(f"WARNING: topic discovery failed ({e}); using committed "
272
+ f"modules.json", file=sys.stderr)
273
+
274
+ modules = json.loads((PKG / "modules.json").read_text(encoding="utf-8"))
275
+ if isinstance(modules, dict):
276
+ modules = modules.get("modules") or []
277
+ only = set(args.only.split(",")) if args.only else None
278
+
279
+ repos = []
280
+ for m in modules:
281
+ if not isinstance(m, dict):
282
+ continue
283
+ if only and m.get("name") not in only:
284
+ continue
285
+ entry = harvest_repo(m, args.timeout)
286
+ c = entry["counts"]
287
+ print(f" {entry['name']:24} cloned={str(entry['cloned']):5} "
288
+ f"proc={c['processes']} step={c['steps']} comp={c['composites']} "
289
+ f"study={c['studies']} inv={c['investigations']}", file=sys.stderr)
290
+ repos.append(entry)
291
+
292
+ index = {
293
+ "generated_at": args.stamp or datetime.now(timezone.utc).isoformat(timespec="seconds"),
294
+ "n_repos": len(repos),
295
+ "n_cloned": sum(1 for r in repos if r["cloned"]),
296
+ "repos": repos,
297
+ }
298
+ Path(args.out).write_text(json.dumps(index, indent=2) + "\n", encoding="utf-8")
299
+ print(f"wrote {args.out}: {index['n_repos']} repos, {index['n_cloned']} cloned", file=sys.stderr)
300
+ return 0
301
+
302
+
303
+ if __name__ == "__main__":
304
+ raise SystemExit(main())