gini-toolkit 6.0.1.dev0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (278) hide show
  1. gini/__init__.py +12 -0
  2. gini/__main__.py +107 -0
  3. gini/_version.py +24 -0
  4. gini/agent/__init__.py +17 -0
  5. gini/agent/agent_gamemaster.py +140 -0
  6. gini/agent/api.py +291 -0
  7. gini/agent/ask.py +123 -0
  8. gini/agent/authoring.py +72 -0
  9. gini/agent/blackboard.py +114 -0
  10. gini/agent/contracts.py +142 -0
  11. gini/agent/domains.py +91 -0
  12. gini/agent/embed.py +123 -0
  13. gini/agent/gamemaster.py +256 -0
  14. gini/agent/kb.py +148 -0
  15. gini/agent/lesson_resolver.py +261 -0
  16. gini/agent/llm/__init__.py +5 -0
  17. gini/agent/llm/backend.py +43 -0
  18. gini/agent/llm/fake.py +25 -0
  19. gini/agent/llm/ollama.py +206 -0
  20. gini/agent/loop.py +258 -0
  21. gini/agent/mcp_server.py +86 -0
  22. gini/agent/meaning.py +225 -0
  23. gini/agent/mission.py +210 -0
  24. gini/agent/mission_controller.py +208 -0
  25. gini/agent/narration.py +116 -0
  26. gini/agent/notifier.py +86 -0
  27. gini/agent/personas.py +79 -0
  28. gini/agent/reasoning.py +172 -0
  29. gini/agent/recall.py +248 -0
  30. gini/agent/session.py +79 -0
  31. gini/agent/teaching_center.py +482 -0
  32. gini/agent/tools/__init__.py +3 -0
  33. gini/agent/tools/registry.py +193 -0
  34. gini/agent/twin/__init__.py +28 -0
  35. gini/agent/twin/authoring.py +71 -0
  36. gini/agent/twin/contracts.py +54 -0
  37. gini/agent/twin/dialectic.py +189 -0
  38. gini/agent/twin/harness.py +93 -0
  39. gini/agent/twin/justify.py +156 -0
  40. gini/agent/twin/learner.py +64 -0
  41. gini/agent/twin/mission.py +60 -0
  42. gini/agent/twin/os_coach.py +79 -0
  43. gini/agent/twin/salience.py +30 -0
  44. gini/agent/understand.py +250 -0
  45. gini/agent/verifiers.py +106 -0
  46. gini/agent/wizard.py +178 -0
  47. gini/agent/xv6_pack.py +74 -0
  48. gini/app/__init__.py +3 -0
  49. gini/app/context.py +368 -0
  50. gini/app/paths.py +121 -0
  51. gini/data/README.md +21 -0
  52. gini/domain/__init__.py +9 -0
  53. gini/domain/assembly.py +209 -0
  54. gini/domain/authoring.py +353 -0
  55. gini/domain/blueprints.py +5 -0
  56. gini/domain/capabilities.py +177 -0
  57. gini/domain/catalog.py +85 -0
  58. gini/domain/certify.py +201 -0
  59. gini/domain/compose.py +413 -0
  60. gini/domain/composition.py +88 -0
  61. gini/domain/concepts.py +383 -0
  62. gini/domain/connection_rules.py +269 -0
  63. gini/domain/constraints.py +153 -0
  64. gini/domain/content.py +59 -0
  65. gini/domain/cpu_journey.py +89 -0
  66. gini/domain/devices.py +747 -0
  67. gini/domain/diagnose.py +201 -0
  68. gini/domain/element_guide.py +327 -0
  69. gini/domain/explain.py +90 -0
  70. gini/domain/fingerprint.py +201 -0
  71. gini/domain/firewall.py +34 -0
  72. gini/domain/flowlog.py +61 -0
  73. gini/domain/flowtable.py +179 -0
  74. gini/domain/fragment_yaml.py +230 -0
  75. gini/domain/fragments.py +169 -0
  76. gini/domain/games/__init__.py +2 -0
  77. gini/domain/games/paging_games.py +119 -0
  78. gini/domain/games/policy_game.py +86 -0
  79. gini/domain/games/process_game.py +48 -0
  80. gini/domain/games/thrash_game.py +75 -0
  81. gini/domain/games/translate_game.py +60 -0
  82. gini/domain/games/trap_game.py +86 -0
  83. gini/domain/grader.py +155 -0
  84. gini/domain/grouping.py +67 -0
  85. gini/domain/legality.py +103 -0
  86. gini/domain/lesson.py +241 -0
  87. gini/domain/lexicon.py +150 -0
  88. gini/domain/machine_state.py +410 -0
  89. gini/domain/missions/networking/basic-lan.yaml +32 -0
  90. gini/domain/missions/networking/cache-in-front.yaml +23 -0
  91. gini/domain/missions/networking/decouple-with-queue.yaml +31 -0
  92. gini/domain/missions/networking/drive-load.yaml +20 -0
  93. gini/domain/missions/networking/fix-the-address.yaml +75 -0
  94. gini/domain/missions/networking/fix-the-lan.yaml +43 -0
  95. gini/domain/missions/networking/inspect-flows.yaml +16 -0
  96. gini/domain/missions/networking/k8s-autoscale.yaml +27 -0
  97. gini/domain/missions/networking/least-privilege.yaml +21 -0
  98. gini/domain/missions/networking/load-balanced-web.yaml +29 -0
  99. gini/domain/missions/networking/observe-it.yaml +24 -0
  100. gini/domain/missions/networking/put-in-vpc.yaml +30 -0
  101. gini/domain/missions/networking/reachability-boundary.yaml +56 -0
  102. gini/domain/missions/networking/sdn-reactive.yaml +35 -0
  103. gini/domain/missions/networking/send-request.yaml +19 -0
  104. gini/domain/missions/networking/serverless-api.yaml +25 -0
  105. gini/domain/missions/networking/service-chain.yaml +33 -0
  106. gini/domain/missions/os/lottery-fix.yaml +19 -0
  107. gini/domain/missions/os/priority-fix.yaml +24 -0
  108. gini/domain/missions.py +111 -0
  109. gini/domain/modulechain.py +36 -0
  110. gini/domain/objectives.py +488 -0
  111. gini/domain/os_zoo.py +79 -0
  112. gini/domain/paging_sim.py +141 -0
  113. gini/domain/pricing.py +199 -0
  114. gini/domain/probes.py +226 -0
  115. gini/domain/profile.py +142 -0
  116. gini/domain/recipes.py +738 -0
  117. gini/domain/riders.py +309 -0
  118. gini/domain/router_modules.py +224 -0
  119. gini/domain/routetable.py +67 -0
  120. gini/domain/scoring.py +76 -0
  121. gini/domain/staging.py +122 -0
  122. gini/domain/syscall_builder.py +144 -0
  123. gini/domain/topic_cloud.py +62 -0
  124. gini/domain/topology.py +213 -0
  125. gini/domain/vocabulary.py +51 -0
  126. gini/domain/xv6.py +808 -0
  127. gini/domain/xv6_fs.py +250 -0
  128. gini/domain/xv6_runner.py +113 -0
  129. gini/domain/xv6_vm.py +385 -0
  130. gini/gloader.py +17 -0
  131. gini/runtime/__init__.py +18 -0
  132. gini/runtime/cloudfabric_agent.py +370 -0
  133. gini/runtime/console.py +68 -0
  134. gini/runtime/control.py +70 -0
  135. gini/runtime/frame.py +138 -0
  136. gini/runtime/gbridge.py +638 -0
  137. gini/runtime/grouter.py +223 -0
  138. gini/runtime/hostsim.py +90 -0
  139. gini/runtime/shuttle.py +348 -0
  140. gini/runtime/switch.py +109 -0
  141. gini/runtime/transport.py +77 -0
  142. gini/runtime/xv6_bridge.py +312 -0
  143. gini/server/__init__.py +22 -0
  144. gini/server/__main__.py +74 -0
  145. gini/server/app.py +140 -0
  146. gini/server/auth.py +82 -0
  147. gini/server/policy.py +57 -0
  148. gini/server/session.py +23 -0
  149. gini/services/__init__.py +15 -0
  150. gini/services/boardflash.py +248 -0
  151. gini/services/boardsetup.py +374 -0
  152. gini/services/cloud_catalog.py +143 -0
  153. gini/services/compiler.py +1858 -0
  154. gini/services/discovery.py +324 -0
  155. gini/services/gloader.py +183 -0
  156. gini/services/orchestrator.py +1460 -0
  157. gini/services/persistence.py +28 -0
  158. gini/services/probe_runner.py +149 -0
  159. gini/services/project.py +217 -0
  160. gini/services/remote.py +93 -0
  161. gini/services/rider_runner.py +96 -0
  162. gini/services/rider_session.py +171 -0
  163. gini/services/shadow_store.py +52 -0
  164. gini/services/terminal.py +45 -0
  165. gini/setup/__init__.py +17 -0
  166. gini/setup/cli.py +109 -0
  167. gini/setup/images.py +33 -0
  168. gini/setup/marker.py +43 -0
  169. gini/setup/runtime.py +69 -0
  170. gini/ui/__init__.py +3 -0
  171. gini/ui/assets/app_icon.icns +0 -0
  172. gini/ui/assets/app_icon.ico +0 -0
  173. gini/ui/assets/app_icon.png +0 -0
  174. gini/ui/assets/app_icon_1024.png +0 -0
  175. gini/ui/assets/cue/_w.txt +1 -0
  176. gini/ui/assets/cue/ai.png +0 -0
  177. gini/ui/assets/cue/canvas.png +0 -0
  178. gini/ui/assets/cue/cloud.png +0 -0
  179. gini/ui/assets/cue/cost.png +0 -0
  180. gini/ui/assets/cue/dark/ai.png +0 -0
  181. gini/ui/assets/cue/dark/canvas.png +0 -0
  182. gini/ui/assets/cue/dark/cloud.png +0 -0
  183. gini/ui/assets/cue/dark/cost.png +0 -0
  184. gini/ui/assets/cue/dark/metrics.png +0 -0
  185. gini/ui/assets/cue/dark/router.png +0 -0
  186. gini/ui/assets/cue/dark/run.png +0 -0
  187. gini/ui/assets/cue/dark/serverless.png +0 -0
  188. gini/ui/assets/cue/dark/settings.png +0 -0
  189. gini/ui/assets/cue/dark/welcome.png +0 -0
  190. gini/ui/assets/cue/dark/wizard.png +0 -0
  191. gini/ui/assets/cue/ginibrand/ai.png +0 -0
  192. gini/ui/assets/cue/ginibrand/canvas.png +0 -0
  193. gini/ui/assets/cue/ginibrand/cloud.png +0 -0
  194. gini/ui/assets/cue/ginibrand/cost.png +0 -0
  195. gini/ui/assets/cue/ginibrand/metrics.png +0 -0
  196. gini/ui/assets/cue/ginibrand/router.png +0 -0
  197. gini/ui/assets/cue/ginibrand/run.png +0 -0
  198. gini/ui/assets/cue/ginibrand/serverless.png +0 -0
  199. gini/ui/assets/cue/ginibrand/settings.png +0 -0
  200. gini/ui/assets/cue/ginibrand/welcome.png +0 -0
  201. gini/ui/assets/cue/ginibrand/wizard.png +0 -0
  202. gini/ui/assets/cue/highcontrast/ai.png +0 -0
  203. gini/ui/assets/cue/highcontrast/canvas.png +0 -0
  204. gini/ui/assets/cue/highcontrast/cloud.png +0 -0
  205. gini/ui/assets/cue/highcontrast/cost.png +0 -0
  206. gini/ui/assets/cue/highcontrast/metrics.png +0 -0
  207. gini/ui/assets/cue/highcontrast/router.png +0 -0
  208. gini/ui/assets/cue/highcontrast/run.png +0 -0
  209. gini/ui/assets/cue/highcontrast/serverless.png +0 -0
  210. gini/ui/assets/cue/highcontrast/settings.png +0 -0
  211. gini/ui/assets/cue/highcontrast/welcome.png +0 -0
  212. gini/ui/assets/cue/highcontrast/wizard.png +0 -0
  213. gini/ui/assets/cue/light/ai.png +0 -0
  214. gini/ui/assets/cue/light/canvas.png +0 -0
  215. gini/ui/assets/cue/light/cloud.png +0 -0
  216. gini/ui/assets/cue/light/cost.png +0 -0
  217. gini/ui/assets/cue/light/metrics.png +0 -0
  218. gini/ui/assets/cue/light/router.png +0 -0
  219. gini/ui/assets/cue/light/run.png +0 -0
  220. gini/ui/assets/cue/light/serverless.png +0 -0
  221. gini/ui/assets/cue/light/settings.png +0 -0
  222. gini/ui/assets/cue/light/welcome.png +0 -0
  223. gini/ui/assets/cue/light/wizard.png +0 -0
  224. gini/ui/assets/cue/metrics.png +0 -0
  225. gini/ui/assets/cue/router.png +0 -0
  226. gini/ui/assets/cue/run.png +0 -0
  227. gini/ui/assets/cue/serverless.png +0 -0
  228. gini/ui/assets/cue/settings.png +0 -0
  229. gini/ui/assets/cue/welcome.png +0 -0
  230. gini/ui/assets/cue/wizard.png +0 -0
  231. gini/ui/assistant.py +2111 -0
  232. gini/ui/author_dialog.py +184 -0
  233. gini/ui/board_dialog.py +247 -0
  234. gini/ui/branding.py +21 -0
  235. gini/ui/canvas.py +2007 -0
  236. gini/ui/chat_panel.py +7 -0
  237. gini/ui/cpu_journey.py +212 -0
  238. gini/ui/cpu_lab.py +306 -0
  239. gini/ui/cue_cards.py +214 -0
  240. gini/ui/dashboard.py +222 -0
  241. gini/ui/diagnose_game.py +336 -0
  242. gini/ui/fingerprint_lab.py +219 -0
  243. gini/ui/flash_dialog.py +244 -0
  244. gini/ui/flow_layout.py +63 -0
  245. gini/ui/fragment_manager.py +1415 -0
  246. gini/ui/game_catalog.py +184 -0
  247. gini/ui/game_renderers.py +340 -0
  248. gini/ui/games_lab.py +90 -0
  249. gini/ui/inspector.py +1055 -0
  250. gini/ui/live_metrics.py +130 -0
  251. gini/ui/machine_lab.py +1412 -0
  252. gini/ui/main_window.py +3153 -0
  253. gini/ui/memory_lab.py +371 -0
  254. gini/ui/mission_panel.py +302 -0
  255. gini/ui/mode_indicator.py +227 -0
  256. gini/ui/palette.py +112 -0
  257. gini/ui/peripherals.py +218 -0
  258. gini/ui/process_tree.py +130 -0
  259. gini/ui/reset_dialog.py +179 -0
  260. gini/ui/router_lab.py +776 -0
  261. gini/ui/run_button.py +183 -0
  262. gini/ui/settings_dialog.py +234 -0
  263. gini/ui/signin_dialog.py +111 -0
  264. gini/ui/storage_lab.py +219 -0
  265. gini/ui/syscall_builder.py +235 -0
  266. gini/ui/syscall_lab.py +152 -0
  267. gini/ui/theme/__init__.py +5 -0
  268. gini/ui/theme/icons.py +145 -0
  269. gini/ui/theme/manager.py +291 -0
  270. gini/ui/theme/tokens.py +194 -0
  271. gini/ui/trap_lab.py +270 -0
  272. gini/ui/worker_host.py +102 -0
  273. gini/ui/zoo_lab.py +112 -0
  274. gini_toolkit-6.0.1.dev0.dist-info/METADATA +77 -0
  275. gini_toolkit-6.0.1.dev0.dist-info/RECORD +278 -0
  276. gini_toolkit-6.0.1.dev0.dist-info/WHEEL +5 -0
  277. gini_toolkit-6.0.1.dev0.dist-info/entry_points.txt +3 -0
  278. gini_toolkit-6.0.1.dev0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1858 @@
1
+ """RuntimeCompiler — lower a canvas Topology onto the portable user-space runtime.
2
+
3
+ Generalizes the hand-written R0 wiring to any topology:
4
+ * classify devices (machine / switch / router / grouping),
5
+ * find L2 broadcast domains (segments) and give each a subnet,
6
+ * assign IPs, gateways, MACs, and per-endpoint UDP ports,
7
+ * emit machine/switch/router specs that gini.runtime can run (in-process or Docker).
8
+
9
+ Cloud "grouping" devices (VPC, cloud-subnet, region, cluster, pod, instance-group) are
10
+ organizational in R0 and are skipped as runtime nodes (links touching them are
11
+ dropped, with a note). Cloud endpoints (instances, containers, LBs, …) run as machines.
12
+ (Plain IP subnets are NOT a device: each L2 broadcast domain is auto-assigned a /24.)
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ import re
18
+ from dataclasses import dataclass, field
19
+ from pathlib import Path
20
+
21
+ from ..domain import devices as _dev
22
+ from ..domain.topology import Topology
23
+
24
+
25
+ def _gini_home() -> Path:
26
+ # Same rule as app.paths.gini_home, replicated so this service avoids an `app` import cycle.
27
+ return Path(os.environ.get("GINI_HOME_DIR") or (Path.home() / ".gini")).expanduser()
28
+ from .cloud_catalog import is_service, service_for
29
+
30
+ ROUTERS = {"router", "firewall"}
31
+ SWITCHES = {"switch", "hub"} # plain L2 — live in the shared `fabric` container
32
+ GROUPS = {"vpc", "cloud_subnet", "region"}
33
+ # OS Zoo: emulated historical OSes, one container each, screen served over noVNC (a web port).
34
+ # "BYO-style" elements carry Emulator/Image/Rom properties and boot via the generic BYO path — the
35
+ # generic "Classic OS (your image)" plus the convenience presets (Mac System 7, Windows 3.11) whose
36
+ # properties are simply pre-filled with a download URL (GINI still ships no proprietary image).
37
+ OSZOO_BYO_KEYS = {"oszoo_byo", "msdos", "mac7", "win31"}
38
+ OSZOO_KEYS = {"freedos", "kolibri", "menuet"} | OSZOO_BYO_KEYS
39
+ # Sources/Sinks: instruments that run INSIDE a donor container — they get no runtime node
40
+ # of their own, and their (attach) edges are never wired.
41
+ RIDERS = {k for k, dt in _dev.REGISTRY.items() if getattr(dt, "rider", False)}
42
+ K8S_ROLES = {"k8s_cluster": "k8scluster", "pod": "k8sworkload",
43
+ "instance_group": "hpa", "k8s_node": "k8snode"}
44
+
45
+ # SDN: an OVS is an OpenFlow switch that runs as its OWN container (the gRouter in
46
+ # --openflow mode), programmed by a controller over a management channel. A controller
47
+ # is the control plane — it is NOT a data host and gets no data-plane IP/gateway.
48
+ DEFAULT_OF_PORT = 6633
49
+ DEFAULT_OF_APP = "gini.samples.switch"
50
+
51
+ # GINI32: every real board is served by one shared relay container, so a board's fabric
52
+ # endpoint lives at this service name. (The board itself is out on the physical LAN.)
53
+ GBRIDGE_SVC = "gbridge"
54
+
55
+
56
+ def _role(type_key: str) -> str:
57
+ if type_key in ROUTERS:
58
+ return "router"
59
+ if type_key in SWITCHES:
60
+ return "switch"
61
+ if type_key == "ovs":
62
+ return "ovs"
63
+ if type_key == "controller":
64
+ return "controller"
65
+ if type_key in K8S_ROLES: # kubernetes: real k3s cluster + manifests
66
+ return K8S_ROLES[type_key]
67
+ if type_key == "function": # serverless — a handler in the shared faas runtime
68
+ return "function"
69
+ if is_service(type_key): # cloud managed service (own container, bridge net)
70
+ return "service"
71
+ if type_key in ("instance", "container", "kinstance"): # cloud compute — own bridge container
72
+ return "compute" # (kinstance = VM-isolated via Kata)
73
+ if type_key == "security_group": # a policy, not a container — drives member iptables
74
+ return "secgroup"
75
+ if type_key == "vnf": # NFV: an inline network function (forwarding container)
76
+ return "vnf"
77
+ if type_key == "xv6": # standalone teaching kernel (QEMU-RISC-V) — no fabric
78
+ return "xv6"
79
+ if type_key in OSZOO_KEYS: # OS Zoo: emulated historical OS, embedded via noVNC
80
+ return "oszoo"
81
+ if type_key == "gini32": # a REAL ESP32 board: addressed on the fabric like a
82
+ return "gini32" # host, but reached through the gbridge relay, not a
83
+ # container of its own — see _build_gbridge().
84
+ if type_key in ("terminal", "storage_volume"): # xv6 peripherals: pure UI, no
85
+ return "peripheral" # container; never emitted/addressed
86
+ if type_key in RIDERS: # Sources/Sinks: run on a donor, no container of their own
87
+ return "rider"
88
+ if type_key in GROUPS:
89
+ return "group"
90
+ return "machine" # host = a node on the simulated tun fabric
91
+
92
+
93
+ def _svc(name: str) -> str:
94
+ return re.sub(r"[^a-z0-9]", "", name.lower()) or "node"
95
+
96
+
97
+ def _hostname(name: str) -> str:
98
+ """The hostname to set inside the machine container — the user's element name (e.g.
99
+ 'toronto.on') made into a valid hostname, so `hostname` at the shell matches the label on
100
+ the canvas and the student needs no mental mapping. Falls back to the service name."""
101
+ h = re.sub(r"[^a-zA-Z0-9.-]", "-", (name or "").strip()).strip(".-")
102
+ return h or _svc(name)
103
+
104
+
105
+ def _cpus_for(device) -> float:
106
+ """CPU limit (vCPUs) for a device from its size tier — 0.5/1/2/4 for S/M/L/XL."""
107
+ from ..domain import pricing
108
+ return pricing.size_tier(pricing.size_level(getattr(device, "size", 1)))[1]
109
+
110
+
111
+ def _toolkit_for(device) -> str:
112
+ """Which Machine image this host is built from — read off its `Toolkit` property.
113
+
114
+ LEAN is the default and the one to encourage: an Alpine host with the tools a student actually
115
+ types (ip/ping/tcpdump/dig/curl/nc/iperf3/nmap), an order of magnitude smaller than the Debian
116
+ image. A host only opts into FULL when its experiment genuinely needs the heavy services —
117
+ bind9 for the DNS chapter, postfix for mail, ettercap/dsniff for the spoofing labs.
118
+
119
+ Anything unrecognised means lean: a typo must not silently pull in the 10x image."""
120
+ if getattr(device, "type_key", "") == "desktop": # the headful Desktop element is always gui
121
+ return "gui"
122
+ props = getattr(device, "properties", None) or {}
123
+ want = str(props.get("Toolkit", "")).strip().lower()
124
+ return want if want in ("full", "security", "gui") else "lean"
125
+
126
+
127
+ def _xv6_harts(device) -> int:
128
+ """xv6 QEMU CPU count (-smp), capped at 2: S/M -> 1 hart, L/XL -> 2 harts. (xv6 SMP is kept
129
+ to 1-2 for a clear, legible scheduler demo.)"""
130
+ v = _cpus_for(device) # the shown vCPU count: 0.5/1/2/4
131
+ return max(1, min(2, int(round(v))))
132
+
133
+
134
+ def _int(v, default: int) -> int:
135
+ try:
136
+ return int(float(v))
137
+ except (TypeError, ValueError):
138
+ return default
139
+
140
+
141
+ def _k8s_deployment_yaml(d: dict) -> str:
142
+ return (f"apiVersion: apps/v1\nkind: Deployment\nmetadata:\n name: {d['name']}\n"
143
+ f" labels: {{app: {d['name']}}}\nspec:\n replicas: {d['replicas']}\n"
144
+ f" selector:\n matchLabels: {{app: {d['name']}}}\n template:\n"
145
+ f" metadata:\n labels: {{app: {d['name']}}}\n spec:\n"
146
+ f" containers:\n - name: {d['name']}\n image: {d['image']}\n"
147
+ f" ports:\n - containerPort: {d['port']}\n"
148
+ f" resources:\n requests:\n cpu: 50m")
149
+
150
+
151
+ def _k8s_service_yaml(d: dict) -> str:
152
+ return (f"apiVersion: v1\nkind: Service\nmetadata:\n name: {d['name']}\nspec:\n"
153
+ f" selector: {{app: {d['name']}}}\n ports:\n - port: {d['port']}\n"
154
+ f" targetPort: {d['port']}")
155
+
156
+
157
+ def _k8s_hpa_yaml(d: dict) -> str:
158
+ h = d["hpa"]
159
+ return (f"apiVersion: autoscaling/v2\nkind: HorizontalPodAutoscaler\nmetadata:\n"
160
+ f" name: {d['name']}\nspec:\n scaleTargetRef:\n apiVersion: apps/v1\n"
161
+ f" kind: Deployment\n name: {d['name']}\n minReplicas: {h['min']}\n"
162
+ f" maxReplicas: {h['max']}\n metrics:\n - type: Resource\n"
163
+ f" resource:\n name: cpu\n target:\n"
164
+ f" type: Utilization\n averageUtilization: {h['cpu']}")
165
+
166
+
167
+ def _norm_image(raw: str) -> str:
168
+ """Turn a friendly image property into a real Docker tag.
169
+ 'ubuntu-22.04' -> 'ubuntu:22.04'; an explicit tag/registry is kept as-is."""
170
+ raw = (raw or "").strip()
171
+ if not raw:
172
+ return "ubuntu:22.04"
173
+ if ":" in raw or "/" in raw:
174
+ return raw
175
+ if "-" in raw: # ubuntu-22.04 -> ubuntu:22.04
176
+ n, _, v = raw.partition("-")
177
+ return f"{n}:{v}"
178
+ return raw + ":latest"
179
+
180
+
181
+ class _UF:
182
+ def __init__(self) -> None:
183
+ self.p: dict = {}
184
+
185
+ def find(self, x):
186
+ self.p.setdefault(x, x)
187
+ while self.p[x] != x:
188
+ self.p[x] = self.p[self.p[x]]
189
+ x = self.p[x]
190
+ return x
191
+
192
+ def union(self, a, b):
193
+ self.p[self.find(a)] = self.find(b)
194
+
195
+
196
+ @dataclass
197
+ class Endpoint:
198
+ device: str
199
+ location: str # service name (machine) or "fabric"
200
+ bind_port: int = 0
201
+ peer: "Endpoint | None" = None
202
+
203
+ def peer_host(self, docker: bool) -> str:
204
+ if not docker:
205
+ return "127.0.0.1"
206
+ if self.location == "fabric" and self.peer.location == "fabric":
207
+ return "127.0.0.1"
208
+ return self.peer.location
209
+
210
+ def wiring(self, docker: bool) -> dict:
211
+ return {"bind_host": "0.0.0.0" if docker else "127.0.0.1",
212
+ "bind_port": self.bind_port,
213
+ "peer_host": self.peer_host(docker),
214
+ "peer_port": self.peer.bind_port}
215
+
216
+
217
+ @dataclass
218
+ class MachineSpec:
219
+ name: str
220
+ ifaces: list["IfaceSpec"] # one per segment the machine is on (multi-homing)
221
+ gw: str | None # default gateway (first segment that has a router)
222
+ # --- internet / NAT gateway (the drawn "Internet" element) --------------- #
223
+ gateway: bool = False # this node is the on-fabric NAT gateway to the world
224
+ fabric_default: bool = False # send 0.0.0.0/0 INTO the fabric (egress via the gateway)
225
+ fabric_gw: str | None = None # for the gateway: the local router IP for the return path
226
+ cpus: float = 0.0 # CPU limit from the size tier (0 = unset)
227
+ # Which image this host is built from: "lean" (Alpine, the default — small and fast to boot)
228
+ # or "full" (Debian + bind9/postfix/ettercap…, for the book's heavy experiments). This is a
229
+ # DIFFERENT axis from the size tier above: size = how much CPU it gets and what it costs;
230
+ # toolkit = what software is installed in it. A lean host with an XL cap is perfectly valid.
231
+ toolkit: str = "lean"
232
+ novnc_port: int = 0 # headful ("gui") host: published host port for its noVNC console
233
+ # --- inline VNF (NFV service function) ----------------------------------- #
234
+ forward: bool = False # IP-forward between its interfaces (a transit node)
235
+ nf: str = "" # the network function kind: firewall|block|ids|cache|shaper
236
+ nf_rules: str = "" # the function's config (e.g. firewall: "deny 10.0.3.0/24")
237
+
238
+
239
+ @dataclass
240
+ class IfaceSpec:
241
+ ip: str
242
+ mac: str
243
+ ep: Endpoint
244
+ link_id: str = "" # the topology link this interface sits on
245
+
246
+
247
+ def _is_ipv4(s: str) -> bool:
248
+ parts = (s or "").strip().split(".")
249
+ if len(parts) != 4:
250
+ return False
251
+ try:
252
+ return all(0 <= int(p) <= 255 for p in parts)
253
+ except ValueError:
254
+ return False
255
+
256
+
257
+ def _valid_cidr(s: str) -> bool:
258
+ import ipaddress
259
+ try:
260
+ ipaddress.ip_network((s or "").strip(), strict=False)
261
+ return True
262
+ except (ValueError, TypeError):
263
+ return False
264
+
265
+
266
+ def _parse_ingress(text: str, sg_by_name: dict) -> list:
267
+ """Parse a Security Group's Ingress field into rule dicts {port, cidrs, svcs}. Rules are
268
+ `;`/newline separated, each like '5432 from app', '80 from anywhere', 'from 10.0.0.0/8'
269
+ or '443'. 'from <sg-name>' expands to that SG's member service names (SG→SG rules)."""
270
+ import re as _re
271
+ out = []
272
+ for raw in _re.split(r"[;\n]", text or ""):
273
+ s = raw.strip()
274
+ if not s:
275
+ continue
276
+ m = _re.match(r"(?i)^(?:port\s+)?(\d+|all|any|\*)?\s*(?:from\s+(.+))?$", s)
277
+ if not m:
278
+ continue
279
+ ptok = (m.group(1) or "").lower()
280
+ src = (m.group(2) or "anywhere").strip()
281
+ port = None if ptok in ("", "all", "any", "*") else int(ptok)
282
+ cidrs, svcs, low = [], [], src.lower()
283
+ if low in ("anywhere", "any", "all", "0.0.0.0/0", "public", "internet"):
284
+ cidrs.append("0.0.0.0/0")
285
+ elif "/" in src and _valid_cidr(src):
286
+ cidrs.append(src)
287
+ elif low in sg_by_name:
288
+ svcs.extend(sg_by_name[low])
289
+ else:
290
+ svcs.append(_svc(src)) # treat an unknown name as a service/host
291
+ out.append({"port": port, "cidrs": cidrs, "svcs": svcs})
292
+ return out
293
+
294
+
295
+ def _sg_script(rules: list) -> str:
296
+ """A stateful default-deny-inbound iptables script for one member's union of SG rules."""
297
+ L = [
298
+ "iptables -P INPUT DROP", "iptables -P FORWARD DROP", "iptables -P OUTPUT ACCEPT",
299
+ "iptables -A INPUT -i lo -j ACCEPT",
300
+ "iptables -A INPUT -m conntrack --ctstate ESTABLISHED,RELATED -j ACCEPT",
301
+ # the GINI telemetry agent is infra — always allow it (so the dashboard still polls)
302
+ "for ip in $(getent hosts cloudfabric 2>/dev/null | awk '{print $1}'); do "
303
+ 'iptables -A INPUT -s "$ip" -j ACCEPT; done',
304
+ ]
305
+ for r in rules:
306
+ dport = f" --dport {r['port']}" if r.get("port") else ""
307
+ for cidr in r.get("cidrs", []):
308
+ L.append(f"iptables -A INPUT -p tcp{dport} -s {cidr} -j ACCEPT")
309
+ if r.get("svcs"):
310
+ names = " ".join(r["svcs"])
311
+ L.append(f"for ip in $(getent hosts {names} 2>/dev/null | awk '{{print $1}}'); "
312
+ f'do iptables -A INPUT -p tcp{dport} -s "$ip" -j ACCEPT; done')
313
+ return "\n".join(L) + "\n"
314
+
315
+
316
+ @dataclass
317
+ class SwitchSpec:
318
+ name: str
319
+ eps: list[Endpoint]
320
+ hub: bool = False # True = a Layer-1 hub (flood-all repeater), not a learning switch
321
+
322
+
323
+ @dataclass
324
+ class RouterSpec:
325
+ name: str
326
+ ifaces: list[IfaceSpec]
327
+ routes: list = field(default_factory=list) # static inter-router routes:
328
+ # {net, mask, gw, dev} (dev = tun index)
329
+
330
+
331
+ @dataclass
332
+ class OvsSpec:
333
+ """An OpenFlow switch: the gRouter in --openflow mode, in its own container,
334
+ programmed by `controller` (a service name) over OpenFlow on `controller_port`."""
335
+ name: str
336
+ eps: list[Endpoint]
337
+ controller: str | None = None # controller service name (host), or None
338
+ controller_port: int = DEFAULT_OF_PORT
339
+
340
+
341
+ @dataclass
342
+ class ControllerSpec:
343
+ """An SDN controller: a POX container running `app` on `port`, programming the
344
+ OVS switches in `switches` (their service names)."""
345
+ name: str
346
+ app: str = DEFAULT_OF_APP
347
+ port: int = DEFAULT_OF_PORT
348
+ switches: list[str] = field(default_factory=list)
349
+
350
+
351
+ @dataclass
352
+ class ServiceSpec:
353
+ """A managed cloud service backed by an off-the-shelf container image (MinIO,
354
+ Postgres, …). Runs on the shared bridge network, reachable by its service name.
355
+ `ports` is a list of {container, host, label, web} — host is a unique published
356
+ port so multiple consoles don't collide. `volumes` are compose volume strings;
357
+ `files` are generated config files (relative path -> content) written into the
358
+ project and bind-mounted — this is how the observability stack is auto-wired."""
359
+ name: str
360
+ type_key: str
361
+ image: str
362
+ summary: str
363
+ command: list[str] = field(default_factory=list)
364
+ env: dict[str, str] = field(default_factory=dict)
365
+ ports: list[dict] = field(default_factory=list)
366
+ volumes: list[str] = field(default_factory=list)
367
+ privileged: bool = False
368
+ files: dict[str, str] = field(default_factory=dict)
369
+ cpus: float = 0.0 # CPU limit from the size tier (0 = unset)
370
+ networks: list = field(default_factory=lambda: ["gini"]) # Docker networks to attach to
371
+ runtime: str = "" # OCI runtime override (e.g. "kata" for a Kata Instance)
372
+
373
+
374
+ @dataclass
375
+ class K8sSpec:
376
+ """A real Kubernetes cluster (k3s in a container) + the workloads to deploy in it.
377
+ `deployments` = [{name,image,replicas,port,hpa:{min,max,cpu}|None}]; `manifests` is
378
+ the combined YAML applied via `kubectl apply` once the cluster is up."""
379
+ name: str
380
+ svc: str
381
+ image: str
382
+ deployments: list = field(default_factory=list)
383
+ manifests: str = ""
384
+
385
+
386
+ @dataclass
387
+ class FabricSpec:
388
+ """The GINI Cloud Fabric agent: one container that polls each cloud service's native
389
+ metrics and serves them to gBuilder. `services` = [{name,type,host,port,creds}]."""
390
+ services: list = field(default_factory=list)
391
+ port: int = 9099
392
+
393
+
394
+ @dataclass
395
+ class NetworkSpec:
396
+ """A VPC/subnet rendered as a real Docker bridge network. Containers on different VPC
397
+ networks can't reach each other; within one they can. A VPC's shared net is `internal`
398
+ (no internet of its own — the implicit VPC fabric); a per-VPC *egress* net is a normal
399
+ bridge that public-subnet members also join for real internet + host-published consoles.
400
+ `cidr` is the network's subnet (empty = let Docker auto-assign)."""
401
+ name: str # docker network name (the VPC's slug, or <slug>_egress)
402
+ cidr: str # e.g. 10.0.0.0/16, or "" for auto
403
+ label: str = "" # display name (for notes/UI)
404
+ region: str = ""
405
+ internal: bool = False # True = no external connectivity (the VPC fabric net)
406
+
407
+
408
+ # Fallback resolver when an Internet element carries no DNS of its own (an older saved
409
+ # topology, drawn before the property existed). Google's is used because it is the one
410
+ # public resolver reachable from essentially every network that has internet at all.
411
+ DEFAULT_PUBLIC_DNS = "8.8.8.8"
412
+
413
+
414
+ def _internet_dns(topo) -> str:
415
+ """The resolver to hand out, taken from the Internet element on the canvas.
416
+
417
+ Blanking the property is a legitimate choice — "internet, but resolve names
418
+ yourself" — so an explicitly empty value is honoured rather than back-filled.
419
+ """
420
+ for d in topo.devices.values():
421
+ if d.type_key != "cloud":
422
+ continue
423
+ props = d.properties or {}
424
+ if "DNS" not in props:
425
+ return DEFAULT_PUBLIC_DNS # saved before the property existed
426
+ raw = str(props.get("DNS", "")).strip()
427
+ return raw if _valid_ip(raw) else ""
428
+ return ""
429
+
430
+
431
+ def _valid_ip(text: str) -> bool:
432
+ parts = (text or "").split(".")
433
+ return (len(parts) == 4
434
+ and all(p.isdigit() and len(p) <= 3 and 0 <= int(p) <= 255 for p in parts))
435
+
436
+
437
+ @dataclass
438
+ class GBridgeSpec:
439
+ """One drawn GINI32 element: a real board's end of a fabric link.
440
+
441
+ Unlike every other spec this does NOT become a container. The `gbridge` relay
442
+ owns `ep` (the fabric-side UDP endpoint) and forwards frames over the physical
443
+ LAN to whichever address the board checked in from. Everything here except
444
+ `board_id` is handed to the board in the relay's HELLO_ACK, so the canvas stays
445
+ the single source of truth for the board's fabric identity.
446
+ """
447
+ name: str
448
+ board_id: str # must match the id flashed on the board
449
+ ip: str # fabric address assigned from its segment
450
+ mask: str
451
+ gw: str # its gateway (the router on that segment)
452
+ mac: str
453
+ ep: Endpoint
454
+ mode: str = "nat" # nat = devices hidden behind `ip`; routed = own subnet
455
+ physical_subnet: str = "" # the subnet behind the radio (always allocated)
456
+ mtu: int = 1400
457
+ seg: int = -1 # the segment it sits on (routed-mode route emission)
458
+ # The hotspot the board raises for real devices. Assigned here, not baked into
459
+ # firmware, so two boards never collide and a lab can be renamed without a reflash.
460
+ ap_ssid: str = ""
461
+ ap_pass: str = ""
462
+ # The resolver the board's DHCP server hands to real devices, or "" when the canvas
463
+ # has no Internet element. Empty is meaningful, not missing: with nothing to egress
464
+ # through, offering a resolver would promise name resolution that cannot work.
465
+ dns: str = ""
466
+
467
+
468
+ @dataclass
469
+ class RuntimeConfig:
470
+ machines: list[MachineSpec] = field(default_factory=list)
471
+ gbridge: list[GBridgeSpec] = field(default_factory=list) # real GINI32 boards
472
+ switches: list[SwitchSpec] = field(default_factory=list)
473
+ routers: list[RouterSpec] = field(default_factory=list)
474
+ ovs_switches: list[OvsSpec] = field(default_factory=list)
475
+ controllers: list[ControllerSpec] = field(default_factory=list)
476
+ services: list[ServiceSpec] = field(default_factory=list)
477
+ fabric: "FabricSpec | None" = None # cloud telemetry agent
478
+ k8s: list = field(default_factory=list) # real k3s clusters + manifests
479
+ faas: list = field(default_factory=list) # serverless functions (shared runtime)
480
+ networks: list = field(default_factory=list) # VPCs as isolated Docker networks
481
+ firewalls: list = field(default_factory=list) # security-group iptables per member
482
+ subnets: dict[int, str] = field(default_factory=dict) # seg -> cidr
483
+ notes: list[str] = field(default_factory=list)
484
+
485
+ # -- emit for the in-process simulator / Docker ------------------------- #
486
+ def to_runtime(self, docker: bool) -> dict:
487
+ return {
488
+ "machines": [
489
+ {"name": _svc(m.name), "hostname": _hostname(m.name), "gw": m.gw,
490
+ "gateway": m.gateway, "fabric_default": m.fabric_default,
491
+ "fabric_gw": m.fabric_gw, "cpus": m.cpus, "toolkit": m.toolkit,
492
+ "novnc_port": m.novnc_port,
493
+ "forward": m.forward, "nf": m.nf, "nf_rules": m.nf_rules,
494
+ "ifaces": [{"ip": i.ip, "mac": i.mac, "tap": f"gini{idx}",
495
+ "port": i.ep.wiring(docker)}
496
+ for idx, i in enumerate(m.ifaces)]}
497
+ for m in self.machines
498
+ ],
499
+ "switches": [
500
+ {"name": _svc(s.name), "ports": [e.wiring(docker) for e in s.eps],
501
+ "hub": s.hub}
502
+ for s in self.switches
503
+ ],
504
+ "routers": [
505
+ {"name": _svc(r.name),
506
+ "ifaces": [{"ip": i.ip, "mac": i.mac, "port": i.ep.wiring(docker)}
507
+ for i in r.ifaces],
508
+ "routes": r.routes}
509
+ for r in self.routers
510
+ ],
511
+ # SDN: OVS switches run as their own gRouter --openflow containers; each
512
+ # data port is a cross-container UDP link, just like a router interface.
513
+ # Ports carry a link-local placeholder IP only so the gRouter can bring the
514
+ # tun up — in OpenFlow mode frames are switched by the flow table (shunted
515
+ # before the L3 stack), so the address is never used for forwarding.
516
+ "ovs": [
517
+ {"name": _svc(s.name), "openflow": True,
518
+ "controller": s.controller, "controller_port": s.controller_port,
519
+ "ports": [
520
+ {"ip": f"169.254.{si}.{pi + 1}/16",
521
+ "mac": f"02:00:fe:{si:02x}:00:{pi + 1:02x}",
522
+ "port": e.wiring(docker)}
523
+ for pi, e in enumerate(s.eps)]}
524
+ for si, s in enumerate(self.ovs_switches)
525
+ ],
526
+ "controllers": [
527
+ {"name": _svc(c.name), "app": c.app, "port": c.port,
528
+ "switches": c.switches}
529
+ for c in self.controllers
530
+ ],
531
+ # Managed cloud services — ordinary containers from public images on the
532
+ # shared bridge network, reachable by service name (cloud-style discovery).
533
+ "services": [
534
+ {"name": _svc(s.name), "type": s.type_key, "image": s.image,
535
+ "summary": s.summary, "command": s.command, "env": s.env,
536
+ "ports": s.ports, "volumes": s.volumes, "privileged": s.privileged,
537
+ "files": s.files, "cpus": s.cpus, "networks": s.networks,
538
+ "runtime": s.runtime}
539
+ for s in self.services
540
+ ],
541
+ "fabric": ({"port": self.fabric.port, "services": self.fabric.services}
542
+ if self.fabric else None),
543
+ "k8s": [{"name": _svc(k.name), "image": k.image,
544
+ "deployments": k.deployments} for k in self.k8s],
545
+ "faas": self.faas, # serverless: [{name, handler, code}] -> one faas container
546
+ # VPC/subnet Docker networks (empty -> only the flat `gini` bridge).
547
+ "networks": [{"name": n.name, "cidr": n.cidr, "label": n.label,
548
+ "region": n.region, "internal": n.internal} for n in self.networks],
549
+ # security groups -> per-member iptables (a sidecar in each member's netns).
550
+ "firewalls": self.firewalls,
551
+ # Real GINI32 boards. One `gbridge` relay container serves all of them:
552
+ # `fabric` is its end of the link to the board's router, and the rest is
553
+ # the identity it hands the board when the board announces itself.
554
+ "gbridge": [
555
+ {"board_id": b.board_id, "name": _svc(b.name), "label": b.name,
556
+ "ip": b.ip, "mask": b.mask, "gw": b.gw, "mac": b.mac, "mtu": b.mtu,
557
+ "mode": b.mode, "physical_subnet": b.physical_subnet,
558
+ "ap_ssid": b.ap_ssid, "ap_pass": b.ap_pass, "dns": b.dns,
559
+ "fabric": b.ep.wiring(docker)}
560
+ for b in self.gbridge
561
+ ],
562
+ }
563
+
564
+
565
+ # --- observability auto-wiring config (generated into the project, bind-mounted) --- #
566
+ _PROMETHEUS_YML = (
567
+ "global:\n"
568
+ " scrape_interval: 5s\n"
569
+ "scrape_configs:\n"
570
+ " - job_name: prometheus\n"
571
+ " static_configs:\n"
572
+ " - targets: ['localhost:9090']\n"
573
+ " - job_name: cadvisor\n"
574
+ " static_configs:\n"
575
+ " - targets: ['cadvisor:8080']\n"
576
+ )
577
+ _GRAFANA_DS = (
578
+ "apiVersion: 1\n"
579
+ "datasources:\n"
580
+ " - name: Prometheus\n"
581
+ " type: prometheus\n"
582
+ " uid: prometheus\n" # fixed uid so the dashboard panels bind reliably
583
+ " access: proxy\n"
584
+ " url: http://{prom}:9090\n"
585
+ " isDefault: true\n"
586
+ )
587
+ _GRAFANA_PROVIDER = (
588
+ "apiVersion: 1\n"
589
+ "providers:\n"
590
+ " - name: gini\n"
591
+ " type: file\n"
592
+ " options:\n"
593
+ " path: /var/lib/grafana/dashboards\n"
594
+ )
595
+
596
+
597
+ # Traefik dynamic (file-provider) config: route everything to the drawn backends.
598
+ _TRAEFIK_DYNAMIC = (
599
+ "http:\n"
600
+ " routers:\n"
601
+ " gini:\n"
602
+ " rule: \"PathPrefix(`/`)\"\n"
603
+ " entryPoints: [web]\n"
604
+ " service: gini\n"
605
+ " services:\n"
606
+ " gini:\n"
607
+ " loadBalancer:\n"
608
+ " servers:\n"
609
+ "{servers}\n"
610
+ )
611
+
612
+ # nginx load-balancer config: an upstream over the drawn backends (honoring the chosen
613
+ # algorithm) + a /nginx_status endpoint so the cloud fabric can read its request rate.
614
+ _NGINX_LB = (
615
+ "events {{}}\n"
616
+ "http {{\n"
617
+ " upstream gini_backend {{\n"
618
+ "{algo}"
619
+ "{servers}\n"
620
+ " }}\n"
621
+ " server {{\n"
622
+ " listen 80;\n"
623
+ " location / {{\n"
624
+ " proxy_pass http://gini_backend;\n"
625
+ " proxy_set_header Host $host;\n"
626
+ " }}\n"
627
+ " location /nginx_status {{ stub_status; }}\n"
628
+ " }}\n"
629
+ "}}\n"
630
+ )
631
+
632
+
633
+ def _grafana_dashboard_json() -> str:
634
+ """A starter Grafana dashboard. It leads with Prometheus *pipeline-health* panels
635
+ (targets up, samples scraped) that ALWAYS have data — so the board is never blank and
636
+ doubles as a built-in diagnostic — then shows cAdvisor per-container CPU/mem/network
637
+ (best-effort; cAdvisor can be sparse on Docker Desktop). Built via json.dumps so the
638
+ PromQL (with quotes) is always valid JSON."""
639
+ import json
640
+
641
+ ds = {"type": "prometheus", "uid": "prometheus"} # bind panels to the datasource
642
+
643
+ def ts(pid, title, expr, legend, x, y, w=12, h=8):
644
+ return {"id": pid, "type": "timeseries", "title": title, "datasource": ds,
645
+ "gridPos": {"h": h, "w": w, "x": x, "y": y},
646
+ "fieldConfig": {"defaults": {}, "overrides": []},
647
+ "targets": [{"expr": expr, "legendFormat": legend, "refId": "A",
648
+ "datasource": ds}]}
649
+
650
+ def stat(pid, title, expr, legend, x, y, w=12, h=6):
651
+ return {"id": pid, "type": "stat", "title": title, "datasource": ds,
652
+ "gridPos": {"h": h, "w": w, "x": x, "y": y},
653
+ "options": {"colorMode": "background", "graphMode": "none",
654
+ "textMode": "value_and_name", "reduceOptions":
655
+ {"calcs": ["lastNotNull"]}},
656
+ "fieldConfig": {"defaults": {"mappings": [
657
+ {"type": "value", "options": {"0": {"text": "DOWN", "color": "red"},
658
+ "1": {"text": "UP", "color": "green"}}}],
659
+ "thresholds": {"steps": [{"color": "red", "value": None},
660
+ {"color": "green", "value": 1}]}},
661
+ "overrides": []},
662
+ "targets": [{"expr": expr, "legendFormat": legend, "refId": "A",
663
+ "datasource": ds}]}
664
+
665
+ return json.dumps({
666
+ "title": "GINI lab overview", "uid": "gini-containers",
667
+ "schemaVersion": 39, "version": 1, "refresh": "5s",
668
+ "time": {"from": "now-15m", "to": "now"},
669
+ "panels": [
670
+ # --- pipeline health: always populated (Prometheus knows its own targets) ---
671
+ stat(10, "Scrape targets up", "up", "{{job}}", 0, 0, w=8, h=6),
672
+ ts(11, "Samples scraped / target", "scrape_samples_scraped", "{{job}}",
673
+ 8, 0, w=16, h=6),
674
+ # --- per-container resources (cAdvisor; best-effort on Docker Desktop) ---
675
+ ts(1, "CPU (cores) by container",
676
+ 'rate(container_cpu_usage_seconds_total{name!=""}[1m])', "{{name}}", 0, 6),
677
+ ts(2, "Memory (bytes) by container",
678
+ 'container_memory_usage_bytes{name!=""}', "{{name}}", 12, 6),
679
+ ts(3, "Network RX (bytes/s) by container",
680
+ 'rate(container_network_receive_bytes_total{name!=""}[1m])',
681
+ "{{name}}", 0, 14, w=24),
682
+ ],
683
+ }, indent=2)
684
+
685
+
686
+ class RuntimeCompiler:
687
+ def compile(self, topo: Topology) -> RuntimeConfig:
688
+ cfg = RuntimeConfig()
689
+ role = {d.id: _role(d.type_key) for d in topo.devices.values()}
690
+ name = {d.id: d.name for d in topo.devices.values()}
691
+ # the drawn "Internet" element (type_key "cloud") is the on-fabric NAT gateway:
692
+ # it sits on the fabric like a host but also has an external uplink + does NAT.
693
+ gw_dids = {d.id for d in topo.devices.values() if d.type_key == "cloud"}
694
+
695
+ # 1. keep links not touching grouping devices; pull out SDN control links
696
+ # (controller↔OVS) — they are a management association, not a data segment.
697
+ kept = []
698
+ control_links = []
699
+ ovs_controller: dict[str, str] = {} # ovs did -> controller did
700
+ ctrl_switches: dict[str, list[str]] = {} # controller did -> [ovs did]
701
+ for l in topo.links.values():
702
+ rs, rt = role.get(l.source_id), role.get(l.target_id)
703
+ if getattr(l, "kind", "link") == "attach" or rs == "rider" or rt == "rider":
704
+ # a Source/Sink mount: the rider runs inside the donor, so this is not a cable.
705
+ cfg.notes.append(f"skipped rider attach: "
706
+ f"{name.get(l.source_id)}–{name.get(l.target_id)}")
707
+ elif rs == "group" or rt == "group":
708
+ cfg.notes.append(f"skipped link touching grouping: "
709
+ f"{name.get(l.source_id)}–{name.get(l.target_id)}")
710
+ elif rs in K8S_ROLES.values() or rt in K8S_ROLES.values():
711
+ # kubernetes associations (cluster↔pod, pod↔autoscaler) are intent for
712
+ # manifest generation, not data-plane segments.
713
+ cfg.notes.append(f"k8s link: {name.get(l.source_id)}–{name.get(l.target_id)}")
714
+ elif rs in ("service", "compute") or rt in ("service", "compute"):
715
+ # cloud services + compute live on the bridge net and are reached by
716
+ # name, so a link to one is intent ("uses"), not a data-plane segment.
717
+ cfg.notes.append(f"service link (reach by name): "
718
+ f"{name.get(l.source_id)}–{name.get(l.target_id)}")
719
+ elif rs == "controller" or rt == "controller":
720
+ control_links.append(l)
721
+ ctrl = l.source_id if rs == "controller" else l.target_id
722
+ other = l.target_id if rs == "controller" else l.source_id
723
+ if role.get(other) == "ovs": # only OVS↔controller is a real assoc
724
+ ovs_controller[other] = ctrl
725
+ ctrl_switches.setdefault(ctrl, []).append(other)
726
+ else:
727
+ cfg.notes.append(f"controller {name.get(ctrl)} should attach to an "
728
+ f"OVS, not {name.get(other)}")
729
+ else:
730
+ kept.append(l)
731
+
732
+ # 2. (machines may be multi-homed: a machine keeps ALL its links and gets one
733
+ # interface/IP per segment — see the machine build + shuttle.)
734
+
735
+ # 3. segments: union links that share an L2 (switch) device
736
+ uf = _UF()
737
+ for l in kept:
738
+ uf.find(l.id)
739
+ by_device: dict[str, list] = {}
740
+ for l in kept:
741
+ by_device.setdefault(l.source_id, []).append(l)
742
+ by_device.setdefault(l.target_id, []).append(l)
743
+ for did, links in by_device.items():
744
+ if role[did] in ("switch", "ovs"): # OVS is an L2 domain too
745
+ for l in links[1:]:
746
+ uf.union(links[0].id, l.id)
747
+ seg_of_link = {l.id: uf.find(l.id) for l in kept}
748
+ seg_ids = {}
749
+ for root in dict.fromkeys(seg_of_link.values()):
750
+ seg_ids[root] = len(seg_ids)
751
+ for root, i in seg_ids.items():
752
+ cfg.subnets[i] = f"10.0.{i + 1}.0/24"
753
+
754
+ # 4. endpoints per kept link
755
+ eps: dict[tuple, Endpoint] = {}
756
+
757
+ def endpoint(device_id: str) -> Endpoint:
758
+ # Switches live inside the shared `fabric` container; routers each run as
759
+ # their own `gini-grouter` container (the real C gRouter), so a router's
760
+ # location is its own service name. Machines are their own containers too.
761
+ # This makes every router link a symmetric cross-container UDP link.
762
+ #
763
+ # A GINI32 board is the one endpoint that is NOT a container: it is real
764
+ # hardware on the physical LAN. Its location is the shared `gbridge` relay,
765
+ # which owns this end of the link and forwards to the board over the LAN.
766
+ # The gRouter on the far side therefore needs no notion of hardware at all.
767
+ r = role[device_id]
768
+ loc = ("fabric" if r == "switch"
769
+ else GBRIDGE_SVC if r == "gini32"
770
+ else _svc(name[device_id]))
771
+ return Endpoint(device=name[device_id], location=loc)
772
+
773
+ port = 5000
774
+ link_eps: dict[str, tuple[Endpoint, Endpoint]] = {}
775
+ for l in kept:
776
+ a, b = endpoint(l.source_id), endpoint(l.target_id)
777
+ a.bind_port = port; port += 1
778
+ b.bind_port = port; port += 1
779
+ a.peer, b.peer = b, a
780
+ link_eps[l.id] = (a, b)
781
+ eps[(l.id, l.source_id)] = a
782
+ eps[(l.id, l.target_id)] = b
783
+
784
+ # 5. IP assignment per segment
785
+ seg_hosts: dict[int, int] = {} # next machine host octet
786
+ seg_rtr: dict[int, int] = {} # next router host octet
787
+ seg_gateway: dict[int, str] = {}
788
+ # interfaces on each segment: (device_id, link)
789
+ iface_ip: dict[tuple, str] = {}
790
+ # routers (and inline VNFs) first (so .1 is the gateway), then machines. A VNF is a
791
+ # forwarding node in the path, so it's addressed like a router and becomes the
792
+ # gateway on its point-to-point segments — neighbours then route THROUGH it.
793
+ ordered = sorted(kept, key=lambda l: 0)
794
+ # A GINI32 board is addressed exactly like a host on its segment (it presents one
795
+ # fabric address, behind which its real devices are NATed), so it shares the
796
+ # machine numbering pass.
797
+ for did_role in ("router", "vnf", "machine", "gini32"):
798
+ for l in kept:
799
+ seg = seg_ids[seg_of_link[l.id]]
800
+ base = f"10.0.{seg + 1}."
801
+ for end in (l.source_id, l.target_id):
802
+ if role[end] != did_role:
803
+ continue
804
+ key = (l.id, end)
805
+ if key in iface_ip:
806
+ continue
807
+ if did_role in ("router", "vnf"):
808
+ n = seg_rtr.get(seg, 0) + 1
809
+ seg_rtr[seg] = n
810
+ ip = base + str(n) # .1, .2 ...
811
+ seg_gateway.setdefault(seg, ip)
812
+ else:
813
+ n = seg_hosts.get(seg, 9) + 1
814
+ seg_hosts[seg] = n
815
+ ip = base + str(n) # .10, .11 ...
816
+ iface_ip[key] = ip
817
+
818
+ # 5b. manual addressing: honor each device's typed static IPs, auto-fill the
819
+ # rest. MACs stay auto. Default gateways then follow the (possibly
820
+ # overridden) router address on each segment.
821
+ if getattr(topo, "manual_addressing", False):
822
+ for d in topo.devices.values():
823
+ for lid, sip in (getattr(d, "static_ips", None) or {}).items():
824
+ key = (lid, d.id)
825
+ bare = (sip or "").strip().split("/")[0]
826
+ if key in iface_ip and _is_ipv4(bare):
827
+ iface_ip[key] = bare
828
+ seg_gateway = {}
829
+ for l in kept:
830
+ seg = seg_ids[seg_of_link[l.id]]
831
+ for end in (l.source_id, l.target_id):
832
+ if role[end] in ("router", "vnf") and (l.id, end) in iface_ip:
833
+ seg_gateway.setdefault(seg, iface_ip[(l.id, end)])
834
+
835
+ # the Internet node's IP on each segment it touches; if a segment has an
836
+ # Internet node but no router, hosts there default straight to the Internet node.
837
+ seg_internet_ip: dict[int, str] = {}
838
+ for l in kept:
839
+ seg = seg_ids[seg_of_link[l.id]]
840
+ for end in (l.source_id, l.target_id):
841
+ if end in gw_dids and (l.id, end) in iface_ip:
842
+ seg_internet_ip.setdefault(seg, iface_ip[(l.id, end)])
843
+ for seg, ip in seg_internet_ip.items():
844
+ seg_gateway.setdefault(seg, ip)
845
+ # the segment + IP that everything default-routes toward (first Internet node)
846
+ gw_seg, gw_ip = next(iter(seg_internet_ip.items()), (None, None))
847
+
848
+ def mac(seg: int, kind: int, idx: int) -> str:
849
+ return f"02:00:00:{seg + 1:02x}:{kind:02x}:{idx:02x}"
850
+
851
+ # 6. build specs
852
+ # machines
853
+ midx = 0
854
+ m_ifaces: dict[str, list] = {} # machine did -> [IfaceSpec] (one per segment)
855
+ m_gw: dict[str, str] = {} # machine did -> default gateway
856
+ m_order: list[str] = [] # preserve first-seen order
857
+ for l in kept:
858
+ seg = seg_ids[seg_of_link[l.id]]
859
+ for end in (l.source_id, l.target_id):
860
+ if role[end] != "machine":
861
+ continue
862
+ key = (l.id, end)
863
+ if key not in iface_ip:
864
+ continue
865
+ midx += 1
866
+ if end not in m_ifaces:
867
+ m_ifaces[end] = []
868
+ m_order.append(end)
869
+ m_ifaces[end].append(IfaceSpec(ip=iface_ip[key] + "/24",
870
+ mac=mac(seg, 2, midx), ep=eps[key],
871
+ link_id=l.id))
872
+ gw = seg_gateway.get(seg) # default route via the first router seen
873
+ if gw and end not in m_gw:
874
+ m_gw[end] = gw
875
+ have_internet = bool(gw_dids)
876
+ for did in m_order:
877
+ if did in gw_dids:
878
+ # the Internet element: NAT gateway. It defaults OUT its uplink (set up
879
+ # at runtime), and routes the experiment supernet back via its local
880
+ # router so replies reach the hosts behind it.
881
+ cfg.machines.append(MachineSpec(
882
+ name=name[did], ifaces=m_ifaces[did], gw=None,
883
+ gateway=True, fabric_gw=m_gw.get(did)))
884
+ else:
885
+ # an ordinary host: when an Internet element is on the canvas, its
886
+ # default route goes INTO the fabric so internet egresses through the
887
+ # drawn routers (traceroute then shows the real path).
888
+ cfg.machines.append(MachineSpec(
889
+ name=name[did], ifaces=m_ifaces[did], gw=m_gw.get(did),
890
+ fabric_default=have_internet and bool(m_gw.get(did)),
891
+ cpus=_cpus_for(topo.devices[did]), # size tier -> CPU limit
892
+ toolkit=_toolkit_for(topo.devices[did]))) # lean (default) | full
893
+
894
+ # GINI32 boards: real hardware on the fabric. No container is emitted — the shared
895
+ # `gbridge` relay holds this end of the link and carries frames to the board over
896
+ # the physical LAN. We only need to hand it the identity the canvas assigned.
897
+ for l in kept:
898
+ seg = seg_ids[seg_of_link[l.id]]
899
+ for end in (l.source_id, l.target_id):
900
+ if role[end] != "gini32":
901
+ continue
902
+ key = (l.id, end)
903
+ if key not in iface_ip:
904
+ continue
905
+ midx += 1
906
+ dev = topo.devices[end]
907
+ props = getattr(dev, "properties", None) or {}
908
+ # Blank BoardID falls back to the element's own name. That id will not
909
+ # match any real board — but it is UNIQUE per element, which is the
910
+ # point: the earlier shared default made a second board vanish from the
911
+ # relay's table silently. Emitting nothing instead would be worse still,
912
+ # because the relay is what collects announcing boards for the Inspector
913
+ # to offer: no entry means no relay means an empty picker and no way to
914
+ # fix the very problem. So the element compiles, validate() flags it on
915
+ # the canvas, and the Inspector names the boards that ARE on the air.
916
+ board_id = str(props.get("BoardID", "")).strip() or _svc(name[end])
917
+ mode = str(props.get("Mode", "routed")).strip().lower()
918
+ if mode not in ("nat", "routed"):
919
+ mode = "routed"
920
+
921
+ # The subnet behind this board's radio. Blank means "allocate me one":
922
+ # every board needs a DISTINCT one, or routers end up with two routes
923
+ # to the same network via different next hops and neither works. We
924
+ # skip any third octet the topology's own segments already use.
925
+ phys = str(props.get("PhysicalSubnet", "")).strip()
926
+ if not phys:
927
+ used = {int(c.split(".")[2]) for c in cfg.subnets.values()}
928
+ used |= {int(b.physical_subnet.split(".")[2])
929
+ for b in cfg.gbridge if b.physical_subnet}
930
+ oct3 = 9
931
+ while oct3 in used:
932
+ oct3 += 1
933
+ phys = f"10.0.{oct3}.0/24"
934
+ elif not _valid_cidr(phys):
935
+ cfg.notes.append(
936
+ f"{name[end]}: PhysicalSubnet {phys!r} is not a valid CIDR — "
937
+ f"allocating one automatically")
938
+ phys = ""
939
+ used = {int(c.split(".")[2]) for c in cfg.subnets.values()}
940
+ used |= {int(b.physical_subnet.split(".")[2])
941
+ for b in cfg.gbridge if b.physical_subnet}
942
+ oct3 = 9
943
+ while oct3 in used:
944
+ oct3 += 1
945
+ phys = f"10.0.{oct3}.0/24"
946
+
947
+ # The hotspot real devices join. Named after the element unless the
948
+ # user overrode it, so two boards are distinguishable on a phone.
949
+ ap_ssid = str(props.get("ApSSID", "")).strip()
950
+ if not ap_ssid:
951
+ ap_ssid = f"GINI32-{re.sub(r'[^A-Za-z0-9-]', '', name[end]) or 'board'}"
952
+ ap_pass = str(props.get("ApPassword", "")).strip()
953
+ # Real devices on the board's radio get a resolver ONLY when the canvas
954
+ # has an Internet element to egress through. Without one, an iPad could
955
+ # still be handed 8.8.8.8, would send queries into a topology with no way
956
+ # out, and would sit there timing out — which looks like broken Wi-Fi
957
+ # rather than a network with deliberately no internet in it. This is also
958
+ # why DNS follows the same route as everything else the board is told:
959
+ # the canvas decides, the board obeys.
960
+ dns = _internet_dns(topo) if have_internet else ""
961
+ cfg.gbridge.append(GBridgeSpec(
962
+ name=name[end], board_id=board_id,
963
+ ip=iface_ip[key], mask="255.255.255.0",
964
+ gw=seg_gateway.get(seg, ""), mac=mac(seg, 4, midx),
965
+ ep=eps[key], mode=mode,
966
+ # The board always serves this subnet; `mode` only decides whether
967
+ # the emulated side gets a ROUTE to it or the devices are hidden.
968
+ physical_subnet=phys, seg=seg,
969
+ ap_ssid=ap_ssid, ap_pass=ap_pass, dns=dns))
970
+
971
+ # inline VNFs: a forwarding container that applies a network function between its
972
+ # segments (the Internet-element pattern, but fabric<->fabric + an NF instead of NAT).
973
+ # Addressed like a router above, so it's the gateway on its point-to-point segments;
974
+ # `gw` is a next hop toward egress (a router/gateway on its OTHER segment).
975
+ v_ifaces: dict[str, list] = {}
976
+ v_gw: dict[str, str] = {}
977
+ v_order: list[str] = []
978
+ for l in kept:
979
+ seg = seg_ids[seg_of_link[l.id]]
980
+ for end in (l.source_id, l.target_id):
981
+ if role[end] != "vnf":
982
+ continue
983
+ key = (l.id, end)
984
+ if key not in iface_ip:
985
+ continue
986
+ midx += 1
987
+ my_ip = iface_ip[key]
988
+ if end not in v_ifaces:
989
+ v_ifaces[end] = []
990
+ v_order.append(end)
991
+ v_ifaces[end].append(IfaceSpec(ip=my_ip + "/24", mac=mac(seg, 3, midx),
992
+ ep=eps[key], link_id=l.id))
993
+ g = seg_gateway.get(seg)
994
+ if g and g != my_ip and end not in v_gw: # onward route toward egress
995
+ v_gw[end] = g
996
+ vprops = {d.id: getattr(d, "properties", {}) or {} for d in topo.devices.values()}
997
+ for did in v_order:
998
+ p = vprops[did]
999
+ cfg.machines.append(MachineSpec(
1000
+ name=name[did], ifaces=v_ifaces[did], gw=v_gw.get(did),
1001
+ forward=True, nf=(p.get("Kind") or "firewall"),
1002
+ nf_rules=(p.get("Rules") or ""), cpus=_cpus_for(topo.devices[did])))
1003
+
1004
+ # switches (a Hub is the same fabric node in flood-all mode — no MAC learning)
1005
+ for did, r in role.items():
1006
+ if r != "switch":
1007
+ continue
1008
+ ports = [eps[(l.id, did)] for l in by_device.get(did, []) if l in kept]
1009
+ if ports:
1010
+ is_hub = topo.devices[did].type_key == "hub"
1011
+ cfg.switches.append(SwitchSpec(name=name[did], eps=ports, hub=is_hub))
1012
+
1013
+ # OVS switches — own gRouter --openflow container, programmed by a controller
1014
+ props = {d.id: getattr(d, "properties", {}) or {} for d in topo.devices.values()}
1015
+
1016
+ # VPCs -> isolated Docker networks. Each VPC element with members becomes its own
1017
+ # bridge (unique subnet); every element inside it (via parent_id) attaches to that
1018
+ # network instead of the flat `gini` bridge, so different VPCs can't reach each
1019
+ # other. Elements with no VPC ancestor stay on `gini` (unchanged flat behavior).
1020
+ vpc_net_of = self._build_networks(cfg, topo, name)
1021
+ for did, r in role.items():
1022
+ if r != "ovs":
1023
+ continue
1024
+ ports = [eps[(l.id, did)] for l in by_device.get(did, []) if l in kept]
1025
+ ctrl_did = ovs_controller.get(did)
1026
+ ctrl_name = _svc(name[ctrl_did]) if ctrl_did else None
1027
+ ctrl_port = DEFAULT_OF_PORT
1028
+ if ctrl_did:
1029
+ ctrl_port = int(props[ctrl_did].get("Port") or DEFAULT_OF_PORT)
1030
+ cfg.ovs_switches.append(OvsSpec(name=name[did], eps=ports,
1031
+ controller=ctrl_name,
1032
+ controller_port=ctrl_port))
1033
+
1034
+ # controllers — POX containers; each programs the OVS switches linked to it
1035
+ for did, r in role.items():
1036
+ if r != "controller":
1037
+ continue
1038
+ p = props[did]
1039
+ cfg.controllers.append(ControllerSpec(
1040
+ name=name[did],
1041
+ app=p.get("App") or DEFAULT_OF_APP,
1042
+ port=int(p.get("Port") or DEFAULT_OF_PORT),
1043
+ switches=[_svc(name[o]) for o in ctrl_switches.get(did, [])]))
1044
+
1045
+ # managed cloud services — each backed by an off-the-shelf image. Web consoles
1046
+ # get a unique published host port so several services can coexist.
1047
+ host_port = 38000
1048
+ # headful ("gui") machines publish their noVNC console on a unique host port too, so the
1049
+ # Desktop element can open the embedded screen (the machine dict carries the port).
1050
+ for m in cfg.machines:
1051
+ if m.toolkit == "gui":
1052
+ m.novnc_port = host_port
1053
+ host_port += 1
1054
+ for d in topo.devices.values():
1055
+ if role.get(d.id) != "service":
1056
+ continue
1057
+ svc = service_for(d.type_key)
1058
+ if svc is None:
1059
+ continue
1060
+ ports = []
1061
+ for p in svc.ports:
1062
+ ports.append({"container": p.container, "host": host_port,
1063
+ "label": p.label, "web": p.web, "path": p.path})
1064
+ host_port += 1
1065
+ # some images need to advertise their own service name (e.g. Redpanda's
1066
+ # kafka address); `{svc}` in the catalog command/env is filled in here.
1067
+ sname = _svc(d.name)
1068
+ command = [a.replace("{svc}", sname) for a in svc.command]
1069
+ env = {k: v.replace("{svc}", sname) for k, v in svc.env.items()}
1070
+ cfg.services.append(ServiceSpec(
1071
+ name=d.name, type_key=d.type_key, image=svc.image,
1072
+ summary=svc.summary, command=command, env=env, ports=ports,
1073
+ cpus=_cpus_for(d), # size tier -> CPU limit
1074
+ networks=vpc_net_of.get(d.id, ["gini"]))) # VPC/subnet nets (or flat bridge)
1075
+
1076
+ # cloud compute (instance / container) — a plain bridge container the student can
1077
+ # log into and run an app on, reaching services by name. Image from the element's
1078
+ # property; keep it alive (base OS images would exit) unless a Command is given.
1079
+ import shlex
1080
+ for d in topo.devices.values():
1081
+ if role.get(d.id) != "compute":
1082
+ continue
1083
+ p = props[d.id]
1084
+ if d.type_key == "container":
1085
+ image = _norm_image(p.get("Image") or "alpine:latest")
1086
+ summary = f"Container ({image})."
1087
+ elif d.type_key == "kinstance": # VM-isolated workload via Kata
1088
+ image = _norm_image(p.get("Image") or "ubuntu:22.04")
1089
+ summary = f"Kata Instance — VM-isolated ({image})."
1090
+ else:
1091
+ image = _norm_image(p.get("Image") or "ubuntu:22.04")
1092
+ summary = f"Compute instance ({image}, {p.get('Type', 'vm')})."
1093
+ cmd = p.get("Command") or ""
1094
+ command = shlex.split(cmd) if cmd.strip() else ["tail", "-f", "/dev/null"]
1095
+ is_kata = d.type_key == "kinstance"
1096
+ cfg.services.append(ServiceSpec(
1097
+ name=d.name, type_key=d.type_key, image=image, summary=summary,
1098
+ command=command, env={}, ports=[], cpus=_cpus_for(d), # size -> CPU
1099
+ # Kata Instances stay flat (no VPC) and run under the kata OCI runtime.
1100
+ networks=["gini"] if is_kata else vpc_net_of.get(d.id, ["gini"]),
1101
+ runtime="kata" if is_kata else ""))
1102
+
1103
+ # xv6 teaching kernel — a standalone QEMU-RISC-V machine that boots a real kernel and
1104
+ # exposes its GDB stub (port 1234) so the Machine Lab's bridge can read and steer it.
1105
+ # No fabric wiring (xv6 is standalone); the time-slice tier seeds the kernel quantum.
1106
+ for d in topo.devices.values():
1107
+ if role.get(d.id) != "xv6":
1108
+ continue
1109
+ p = props[d.id]
1110
+ # the Load loop: bind-mount a host folder over kernel/shadows/ so the student edits
1111
+ # gini_sched.c in their own editor and Load rebuilds in-container. The folder can start
1112
+ # empty — the agent seeds the shipped stub into it on boot (see gini_agent.py). Mount a
1113
+ # DIRECTORY (editors save via rename, which breaks a single-file mount). It lives under
1114
+ # the GINI home (~/.gini/xv6-shadows/<name>/) so it's stable + discoverable and the
1115
+ # student's edits PERSIST across Stop/Run (unlike the ephemeral compose workdir).
1116
+ _sane = "".join(c if (c.isalnum() or c in "_.-") else "-" for c in d.name)
1117
+ _shadows_host = _gini_home() / "xv6-shadows" / _sane
1118
+ try:
1119
+ _shadows_host.mkdir(parents=True, exist_ok=True) # exists + user-owned before `up`
1120
+ except OSError:
1121
+ pass
1122
+ cfg.services.append(ServiceSpec(
1123
+ name=d.name, type_key="xv6", image="gini-xv6:latest",
1124
+ summary="xv6 teaching kernel (QEMU-RISC-V); in-container agent serves live state.",
1125
+ command=[], env={"XV6_QUANTUM": str(p.get("Timeslice", "1")),
1126
+ "XV6_CPUS": str(_xv6_harts(d))}, # size stepper -> real harts
1127
+ # agent HTTP (the Machine Lab bridge talks here) + serial (the human console).
1128
+ ports=[{"container": 5000, "host": host_port, "label": "agent",
1129
+ "web": False, "path": ""},
1130
+ {"container": 4444, "host": host_port + 1, "label": "serial",
1131
+ "web": False, "path": ""}],
1132
+ volumes=[f"{_shadows_host}:/opt/xv6-riscv/kernel/shadows"],
1133
+ cpus=_cpus_for(d), networks=["gini"]))
1134
+ host_port += 2
1135
+
1136
+ # OS Zoo — a real historical OS under emulation. One container per element, image
1137
+ # `gini-oszoo:latest`, with ZOO_OS selecting the guest; the container runs the emulator
1138
+ # (`-vnc :0`) + websockify/noVNC and publishes the framebuffer as a web page. The Zoo Lab
1139
+ # embeds that URL in a QWebEngineView. Standalone (no fabric wiring in v1).
1140
+ for d in topo.devices.values():
1141
+ if role.get(d.id) != "oszoo":
1142
+ continue
1143
+ p = props[d.id]
1144
+ is_byo = d.type_key in OSZOO_BYO_KEYS
1145
+ os_id = "byo" if is_byo else d.type_key
1146
+ env = {"ZOO_OS": os_id,
1147
+ "ZOO_PERSIST": "1" if str(p.get("Persist", "false")).lower() == "true" else "0"}
1148
+ if is_byo: # BYO / preset: Emulator + Image (+Rom)
1149
+ env["ZOO_EMULATOR"] = str(p.get("Emulator", "qemu"))
1150
+ env["ZOO_ARCH"] = str(p.get("Arch", "x86"))
1151
+ # Image/Rom may be a local path (bind-mounted below) OR an http(s):// URL that the
1152
+ # container downloads on first boot. Pass the raw value; boot_zoo.sh decides.
1153
+ if str(p.get("Image", "")):
1154
+ env["ZOO_IMAGE"] = str(p.get("Image", ""))
1155
+ if str(p.get("Rom", "")): # Basilisk II: a Mac ROM
1156
+ env["ZOO_ROM"] = str(p.get("Rom", ""))
1157
+ if d.type_key == "win31": # ship Digger Remastered on the Win 3.11 C:
1158
+ env["ZOO_ADDONS"] = "digger" # (run it from the DOS prompt: cd\digger, digger)
1159
+ # persist downloaded guest images on the host so each OS is fetched once, not on
1160
+ # every Run (an anonymous /zoo/cache volume is discarded when the container recreates).
1161
+ # Computed inline (mirrors app.paths.oszoo_cache_dir) to keep the compiler Qt-free.
1162
+ import os
1163
+ from pathlib import Path
1164
+ _home = Path(os.environ.get("GINI_HOME_DIR") or (Path.home() / ".gini")).expanduser()
1165
+ cache = _home / "oszoo-cache"; cache.mkdir(parents=True, exist_ok=True)
1166
+ volumes = [f"{cache}:/zoo/cache"]
1167
+ def _is_url(s: str) -> bool:
1168
+ return s.startswith("http://") or s.startswith("https://")
1169
+ img = str(p.get("Image", "")) if is_byo else ""
1170
+ if img and not _is_url(img): # local path -> bind-mount read-only
1171
+ volumes.append(f"{img}:/zoo/byo.img:ro") # (a URL is downloaded in the container)
1172
+ rom = str(p.get("Rom", "")) if is_byo else ""
1173
+ if rom and not _is_url(rom): # Basilisk II Mac ROM, read-only
1174
+ volumes.append(f"{rom}:/zoo/rom:ro")
1175
+ # Basilisk II creates its 60 Hz timer as a real-time-scheduled thread; Docker's default
1176
+ # sandbox (seccomp + no CAP_SYS_NICE) forbids RT scheduling, so that container needs to
1177
+ # be privileged. QEMU/DOSBox guests don't, so keep them unprivileged.
1178
+ needs_priv = is_byo and str(p.get("Emulator", "")) == "basilisk"
1179
+ cfg.services.append(ServiceSpec(
1180
+ name=d.name, type_key=d.type_key, image="gini-oszoo:latest",
1181
+ summary=f"OS Zoo: {d.type_key} under emulation, screen embedded over noVNC.",
1182
+ command=[], env=env, privileged=needs_priv,
1183
+ # the noVNC web console (the Zoo Lab embeds this) + the raw VNC port.
1184
+ ports=[{"container": 6080, "host": host_port, "label": "screen",
1185
+ "web": True, "path": "/vnc.html?autoconnect=1&resize=remote"},
1186
+ {"container": 5900, "host": host_port + 1, "label": "vnc",
1187
+ "web": False, "path": ""}],
1188
+ volumes=volumes, cpus=_cpus_for(d), networks=["gini"]))
1189
+ host_port += 2
1190
+
1191
+ # make proxies / load balancers actually route to their drawn backends
1192
+ self._wire_proxies(cfg, topo, role, name, props)
1193
+ # serverless: gather Functions into the shared faas runtime + route API Gateways to it
1194
+ self._build_faas(cfg, topo, role, name, props)
1195
+ self._wire_api_gateway(cfg, topo, role, name, props)
1196
+ # security groups: per-member iptables (default-deny inbound + the listed rules)
1197
+ self._build_security_groups(cfg, topo, role, name, props)
1198
+ # auto-wire an observability stack so Prometheus/Grafana actually show data
1199
+ host_port = self._wire_observability(cfg, host_port)
1200
+ # auto-add the cloud-fabric telemetry agent if there are cloud services to watch
1201
+ self._build_fabric(cfg)
1202
+ # real Kubernetes: k3s clusters + generated Deployment/Service/HPA manifests
1203
+ self._build_k8s(cfg, topo, role, name, props)
1204
+
1205
+ # routers
1206
+ ridx = 0
1207
+ spec_of: dict[str, RouterSpec] = {} # router did -> its spec
1208
+ rtr_seg_ip: dict[tuple, str] = {} # (did, seg) -> this router's ip on seg
1209
+ rtr_seg_dev: dict[tuple, int] = {} # (did, seg) -> tun index (1-based)
1210
+ seg_routers: dict[int, list] = {} # seg -> [router dids on it]
1211
+ for did, r in role.items():
1212
+ if r != "router":
1213
+ continue
1214
+ ifaces = []
1215
+ pos = 0
1216
+ for l in by_device.get(did, []):
1217
+ if l not in kept:
1218
+ continue
1219
+ seg = seg_ids[seg_of_link[l.id]]
1220
+ key = (l.id, did)
1221
+ ridx += 1
1222
+ pos += 1
1223
+ ifaces.append(IfaceSpec(ip=iface_ip[key] + "/24",
1224
+ mac=mac(seg, 1, ridx), ep=eps[key],
1225
+ link_id=l.id))
1226
+ rtr_seg_ip[(did, seg)] = iface_ip[key]
1227
+ rtr_seg_dev[(did, seg)] = pos # matches run_grouter's tun{pos}
1228
+ seg_routers.setdefault(seg, []).append(did)
1229
+ if ifaces:
1230
+ spec = RouterSpec(name=name[did], ifaces=ifaces)
1231
+ cfg.routers.append(spec)
1232
+ spec_of[did] = spec
1233
+
1234
+ # static inter-router routes: each router needs a route to every subnet it is
1235
+ # NOT directly on, via the neighbouring router on the shortest path. (There is no
1236
+ # routing protocol between the C routers, so we compute the static routes here.)
1237
+ # A routed-mode GINI32 board fronts a real subnet behind its radio, which no
1238
+ # router knows about. Feed those in as extra destinations reached via the board.
1239
+ extra = [(b.physical_subnet, b.seg, b.ip)
1240
+ for b in cfg.gbridge if b.mode == "routed" and b.physical_subnet
1241
+ and b.seg >= 0]
1242
+ self._add_static_routes(cfg, spec_of, rtr_seg_ip, rtr_seg_dev, seg_routers,
1243
+ gw_seg, gw_ip, extra)
1244
+
1245
+ return cfg
1246
+
1247
+ @staticmethod
1248
+ def _add_static_routes(cfg, spec_of, rtr_seg_ip, rtr_seg_dev, seg_routers,
1249
+ gw_seg=None, gw_ip=None, extra_nets=None) -> None:
1250
+ """extra_nets: [(cidr, seg, via_ip)] — destinations that are not GINI subnets but
1251
+ hang off a node ON `seg` (a routed-mode GINI32 board's physical subnet). Routers on
1252
+ that segment route to them via `via_ip`; others hop toward a router that is."""
1253
+ import ipaddress
1254
+ from collections import deque
1255
+
1256
+ routers = list(spec_of.keys())
1257
+ # router adjacency: two routers are neighbours if they share a segment (a
1258
+ # router-to-router link), which gives the gateway IPs on that link.
1259
+ adj: dict = {d: {} for d in routers}
1260
+ for _seg, rtrs in seg_routers.items():
1261
+ for a in rtrs:
1262
+ for b in rtrs:
1263
+ if a != b:
1264
+ adj[a][b] = _seg
1265
+
1266
+ for did in routers:
1267
+ my_segs = {seg for (d, seg) in rtr_seg_ip if d == did}
1268
+ # BFS: first-hop neighbour toward every reachable router
1269
+ dist = {did: 0}
1270
+ firsthop: dict = {did: None}
1271
+ q = deque([did])
1272
+ while q:
1273
+ cur = q.popleft()
1274
+ for nb in adj[cur]:
1275
+ if nb not in dist:
1276
+ dist[nb] = dist[cur] + 1
1277
+ firsthop[nb] = nb if cur == did else firsthop[cur]
1278
+ q.append(nb)
1279
+ routes = []
1280
+ for seg, cidr in cfg.subnets.items():
1281
+ if seg in my_segs:
1282
+ continue # directly connected
1283
+ cand = [c for c in seg_routers.get(seg, []) if c in dist and c != did]
1284
+ if not cand:
1285
+ continue # unreachable from here
1286
+ best = min(cand, key=lambda c: dist[c])
1287
+ nh = firsthop[best]
1288
+ if nh is None:
1289
+ continue
1290
+ shared = adj[did][nh]
1291
+ net = ipaddress.ip_network(cidr)
1292
+ routes.append({"net": str(net.network_address),
1293
+ "mask": str(net.netmask),
1294
+ "gw": rtr_seg_ip[(nh, shared)],
1295
+ "dev": rtr_seg_dev[(did, shared)]})
1296
+
1297
+ # subnets living behind a routed-mode GINI32 board (real devices on its radio)
1298
+ for cidr, bseg, via_ip in (extra_nets or []):
1299
+ try:
1300
+ net = ipaddress.ip_network(cidr, strict=False)
1301
+ except ValueError:
1302
+ continue
1303
+ if bseg in my_segs: # board is on my segment
1304
+ routes.append({"net": str(net.network_address),
1305
+ "mask": str(net.netmask), "gw": via_ip,
1306
+ "dev": rtr_seg_dev[(did, bseg)]})
1307
+ else: # hop toward its router
1308
+ cand = [c for c in seg_routers.get(bseg, []) if c in dist and c != did]
1309
+ if not cand:
1310
+ continue
1311
+ nh = firsthop[min(cand, key=lambda c: dist[c])]
1312
+ if nh is None:
1313
+ continue
1314
+ shared = adj[did][nh]
1315
+ routes.append({"net": str(net.network_address),
1316
+ "mask": str(net.netmask),
1317
+ "gw": rtr_seg_ip[(nh, shared)],
1318
+ "dev": rtr_seg_dev[(did, shared)]})
1319
+
1320
+ # default route (0.0.0.0/0) toward the Internet/NAT gateway, so internet-
1321
+ # bound traffic leaves the lab through the drawn Internet element.
1322
+ if gw_seg is not None and gw_ip:
1323
+ if gw_seg in my_segs: # gateway is on my segment
1324
+ routes.append({"net": "0.0.0.0", "mask": "0.0.0.0", "gw": gw_ip,
1325
+ "dev": rtr_seg_dev[(did, gw_seg)]})
1326
+ else: # hop toward its router
1327
+ cand = [c for c in seg_routers.get(gw_seg, [])
1328
+ if c in dist and c != did]
1329
+ if cand:
1330
+ best = min(cand, key=lambda c: dist[c])
1331
+ nh = firsthop[best]
1332
+ if nh is not None:
1333
+ shared = adj[did][nh]
1334
+ routes.append({"net": "0.0.0.0", "mask": "0.0.0.0",
1335
+ "gw": rtr_seg_ip[(nh, shared)],
1336
+ "dev": rtr_seg_dev[(did, shared)]})
1337
+ spec_of[did].routes = routes
1338
+
1339
+ # backend elements a proxy / load balancer can route HTTP traffic to
1340
+ _PROXY_BACKENDS = {"web_app", "instance", "container"}
1341
+
1342
+ @classmethod
1343
+ def _wire_proxies(cls, cfg, topo, role, name, props) -> None:
1344
+ """A Reverse Proxy / Load Balancer only forwards if it has a backend config.
1345
+ Build that config from the drawn links: the connected Web Apps / Instances /
1346
+ Containers become its upstreams. Traefik gets a file-provider config; nginx gets
1347
+ an `upstream` block honoring the chosen Scheme + a /nginx_status endpoint."""
1348
+ id_of = {n: i for i, n in name.items()}
1349
+ # adjacency from the drawn links
1350
+ nbrs: dict[str, list] = {d: [] for d in topo.devices}
1351
+ for l in topo.links.values():
1352
+ nbrs[l.source_id].append(l.target_id)
1353
+ nbrs[l.target_id].append(l.source_id)
1354
+
1355
+ svc_by_name = {s.name: s for s in cfg.services}
1356
+ for s in cfg.services:
1357
+ if s.type_key not in ("proxy", "load_balancer"):
1358
+ continue
1359
+ did = id_of.get(s.name)
1360
+ if did is None:
1361
+ continue
1362
+ backends = []
1363
+ for nb in nbrs.get(did, []):
1364
+ tk = topo.devices[nb].type_key
1365
+ if tk in cls._PROXY_BACKENDS:
1366
+ port = 80
1367
+ backends.append((_svc(name[nb]), port))
1368
+ elif role.get(nb) == "service" and tk not in ("proxy", "load_balancer"):
1369
+ bs = svc_by_name.get(name[nb])
1370
+ port = bs.ports[0]["container"] if (bs and bs.ports) else 80
1371
+ backends.append((_svc(name[nb]), port))
1372
+ if not backends:
1373
+ cfg.notes.append(f"{s.name}: no backends wired — connect a Web App to it")
1374
+ continue
1375
+ sname = _svc(s.name)
1376
+ if s.type_key == "proxy":
1377
+ servers = "\n".join(f' - url: "http://{h}:{p}"' for h, p in backends)
1378
+ s.files[f"{sname}/dynamic.yml"] = _TRAEFIK_DYNAMIC.format(servers=servers)
1379
+ s.volumes.append(f"./{sname}/dynamic.yml:/etc/traefik/dynamic/dynamic.yml:ro")
1380
+ s.command = list(s.command) + ["--providers.file.directory=/etc/traefik/dynamic"]
1381
+ else: # nginx load balancer
1382
+ scheme = (props.get(did, {}).get("Scheme") or "round-robin").lower()
1383
+ directive = {"least_conn": " least_conn;\n", "least-conn": " least_conn;\n",
1384
+ "ip_hash": " ip_hash;\n", "ip-hash": " ip_hash;\n"}.get(scheme, "")
1385
+ servers = "\n".join(f" server {h}:{p};" for h, p in backends)
1386
+ s.files[f"{sname}/nginx.conf"] = _NGINX_LB.format(algo=directive, servers=servers)
1387
+ s.volumes.append(f"./{sname}/nginx.conf:/etc/nginx/nginx.conf:ro")
1388
+
1389
+ @classmethod
1390
+ def _build_networks(cls, cfg, topo, name) -> dict:
1391
+ """VPCs + Subnets as real Docker networks (cloud-networking Phase 2).
1392
+
1393
+ A VPC is an **internal** bridge with its CIDR that every member joins — the implicit
1394
+ VPC fabric: members reach each other by name, but it has no internet of its own. A
1395
+ **public** subnet additionally puts its members on a per-VPC **egress** bridge (real
1396
+ internet + host-published consoles); a **private** subnet's members stay on the
1397
+ internal VPC net only — no internet, not reachable from the host (what 'private'
1398
+ means). Membership + tier come from containment: device → Subnet(Tier) → VPC.
1399
+ Returns {device_id -> [docker networks to join]}; non-members default to ["gini"].
1400
+ """
1401
+ parent = {d.id: d.parent_id for d in topo.devices.values()}
1402
+ tkey = {d.id: d.type_key for d in topo.devices.values()}
1403
+ props = {d.id: getattr(d, "properties", {}) or {} for d in topo.devices.values()}
1404
+
1405
+ def ancestors(did):
1406
+ """(vpc_id, subnet_id) — the nearest VPC and Subnet boxes above `did`."""
1407
+ vpc = sub = None
1408
+ seen, cur = set(), parent.get(did)
1409
+ while cur and cur not in seen:
1410
+ seen.add(cur)
1411
+ t = tkey.get(cur)
1412
+ if t == "cloud_subnet" and sub is None:
1413
+ sub = cur
1414
+ if t == "vpc":
1415
+ vpc = cur
1416
+ break
1417
+ cur = parent.get(cur)
1418
+ return vpc, sub
1419
+
1420
+ from ..domain.grouping import BOX_TYPES # VPC/Subnet/Region are containers,
1421
+ member_vpc, member_sub = {}, {} # not workloads — never "members"
1422
+ for d in topo.devices.values():
1423
+ if d.type_key in BOX_TYPES:
1424
+ continue
1425
+ v, s = ancestors(d.id)
1426
+ if v is not None:
1427
+ member_vpc[d.id] = v
1428
+ member_sub[d.id] = s
1429
+ if not member_vpc:
1430
+ return {}
1431
+
1432
+ def is_public(did) -> bool:
1433
+ s = member_sub.get(did)
1434
+ if s is None:
1435
+ return True # in a VPC but not in a subnet -> default public (egress)
1436
+ return (props[s].get("Tier", "private") or "private").strip().lower() == "public"
1437
+
1438
+ used: set[str] = set()
1439
+
1440
+ def unique_cidr(want):
1441
+ want = (want or "").strip()
1442
+ if _valid_cidr(want) and want not in used:
1443
+ used.add(want)
1444
+ return want
1445
+ for i in range(10, 250): # 10.10.0.0/16 … (avoids wan)
1446
+ c = f"10.{i}.0.0/16"
1447
+ if c not in used:
1448
+ used.add(c)
1449
+ return c
1450
+ return "10.249.0.0/16"
1451
+
1452
+ vpc_has_public = {member_vpc[did] for did in member_vpc if is_public(did)}
1453
+ net_of_vpc = {} # vpc_id -> (shared_internal_name, egress_name | None)
1454
+ for vdid in dict.fromkeys(member_vpc.values()): # stable, deduped
1455
+ v = topo.devices[vdid]
1456
+ slug = _svc(name[vdid])
1457
+ cfg.networks.append(NetworkSpec(
1458
+ name=slug, cidr=unique_cidr(props[vdid].get("CIDR")),
1459
+ label=v.name, region=props[vdid].get("Region", ""), internal=True))
1460
+ egress = None
1461
+ if vdid in vpc_has_public:
1462
+ egress = f"{slug}_egress"
1463
+ cfg.networks.append(NetworkSpec(
1464
+ name=egress, cidr="", label=f"{v.name} (public egress)", internal=False))
1465
+ net_of_vpc[vdid] = (slug, egress)
1466
+
1467
+ out = {}
1468
+ for did, vdid in member_vpc.items():
1469
+ shared, egress = net_of_vpc[vdid]
1470
+ nets = [shared]
1471
+ if egress and is_public(did):
1472
+ nets.append(egress)
1473
+ out[did] = nets
1474
+ return out
1475
+
1476
+ @classmethod
1477
+ def _build_security_groups(cls, cfg, topo, role, name, props) -> None:
1478
+ """Security Groups → a stateful per-member firewall (the classic web→app→db least
1479
+ privilege). An SG wired to a workload/datastore makes it default-deny inbound (only
1480
+ stateful replies + the GINI agent allowed) and opens the ports its Ingress lists,
1481
+ from a CIDR or from the members of another SG. Realized as an iptables init sidecar
1482
+ that shares each member's network namespace (so stock images need no changes)."""
1483
+ sgs = [d for d in topo.devices.values() if d.type_key == "security_group"]
1484
+ if not sgs:
1485
+ return
1486
+ nbrs: dict[str, list] = {did: [] for did in topo.devices}
1487
+ for l in topo.links.values():
1488
+ nbrs[l.source_id].append(l.target_id)
1489
+ nbrs[l.target_id].append(l.source_id)
1490
+
1491
+ def members_of(sg):
1492
+ return [nb for nb in nbrs[sg.id] if role.get(nb) in ("service", "compute")]
1493
+
1494
+ sg_by_name: dict[str, list] = {} # "from <sg>" -> that SG's member svc names
1495
+ for sg in sgs:
1496
+ ms = [_svc(name[m]) for m in members_of(sg)]
1497
+ sg_by_name[_svc(sg.name)] = ms
1498
+ sg_by_name[(sg.name or "").strip().lower()] = ms
1499
+
1500
+ per_member: dict[str, list] = {} # member did -> union of its SGs' rules
1501
+ for sg in sgs:
1502
+ rules = _parse_ingress(props.get(sg.id, {}).get("Ingress", ""), sg_by_name)
1503
+ for m in members_of(sg):
1504
+ per_member.setdefault(m, []).extend(rules)
1505
+
1506
+ for did, rules in per_member.items():
1507
+ cfg.firewalls.append({"member": _svc(name[did]), "script": _sg_script(rules)})
1508
+
1509
+ # event sources that can trigger a Function: element type_key -> client port. The
1510
+ # queue/topic/subject the runtime subscribes to is named after the function itself.
1511
+ _EVENT_PORTS = {"queue": 5672, "stream": 9092, "messaging": 4222}
1512
+
1513
+ @classmethod
1514
+ def _build_faas(cls, cfg, topo, role, name, props) -> None:
1515
+ """Gather every Function into the shared faas runtime (one container). A Function
1516
+ node = a handler hosted by the platform, reachable at http://faas:8000/<name>.
1517
+ A Function wired to an event source (Queue/Stream/Pub-Sub) also gets a trigger so
1518
+ the runtime subscribes and invokes the handler on each message (event-driven FaaS)."""
1519
+ nbrs: dict[str, list] = {d: [] for d in topo.devices}
1520
+ for l in topo.links.values():
1521
+ nbrs[l.source_id].append(l.target_id)
1522
+ nbrs[l.target_id].append(l.source_id)
1523
+ funcs = []
1524
+ for did, r in role.items():
1525
+ if r != "function":
1526
+ continue
1527
+ p = props.get(did, {})
1528
+ triggers = []
1529
+ for nb in nbrs.get(did, []):
1530
+ tk = topo.devices[nb].type_key
1531
+ port = cls._EVENT_PORTS.get(tk)
1532
+ if port:
1533
+ triggers.append({"type": tk, "host": _svc(name[nb]), "port": port})
1534
+ funcs.append({"name": _svc(name[did]),
1535
+ "handler": (p.get("Handler") or "echo").strip().lower(),
1536
+ "code": p.get("Code", ""),
1537
+ "triggers": triggers})
1538
+ cfg.faas = funcs
1539
+
1540
+ @classmethod
1541
+ def _wire_api_gateway(cls, cfg, topo, role, name, props) -> None:
1542
+ """An API Gateway (Traefik) routes a URL path to each connected Function: a request
1543
+ to /<fn> is forwarded to the faas runtime, which dispatches to that handler."""
1544
+ id_of = {n: i for i, n in name.items()}
1545
+ nbrs: dict[str, list] = {d: [] for d in topo.devices}
1546
+ for l in topo.links.values():
1547
+ nbrs[l.source_id].append(l.target_id)
1548
+ nbrs[l.target_id].append(l.source_id)
1549
+ for s in cfg.services:
1550
+ if s.type_key != "api_gateway":
1551
+ continue
1552
+ did = id_of.get(s.name)
1553
+ routes = [_svc(name[nb]) for nb in nbrs.get(did, [])
1554
+ if role.get(nb) == "function"]
1555
+ if not routes:
1556
+ cfg.notes.append(f"{s.name}: connect a Function to it to route to one")
1557
+ continue
1558
+ sname = _svc(s.name)
1559
+ routers = "\n".join(
1560
+ f" fn-{fn}:\n rule: \"PathPrefix(`/{fn}`)\"\n"
1561
+ f" service: faas\n entryPoints: [web]" for fn in routes)
1562
+ dyn = ("http:\n routers:\n" + routers +
1563
+ "\n services:\n faas:\n loadBalancer:\n servers:\n"
1564
+ " - url: \"http://faas:8000\"\n")
1565
+ s.files[f"{sname}/dynamic.yml"] = dyn
1566
+ s.volumes.append(f"./{sname}/dynamic.yml:/etc/traefik/dynamic/dynamic.yml:ro")
1567
+ s.command = list(s.command) + ["--providers.file.directory=/etc/traefik/dynamic"]
1568
+
1569
+ # how the cloud-fabric agent probes each service type: (port, creds-kind)
1570
+ _FABRIC_PROBE = {"cache": (6379, None), "queue": (15672, "rabbit"),
1571
+ "database": (5432, "postgres"), "messaging": (8222, None),
1572
+ "proxy": (8080, None), "load_balancer": (80, None)}
1573
+ # infra services the fabric should not monitor (it watches the *app* services)
1574
+ _FABRIC_SKIP = {"_cadvisor", "metrics", "dashboard", "tracing"}
1575
+
1576
+ K3S_IMAGE = "rancher/k3s:v1.30.6-k3s1"
1577
+
1578
+ @classmethod
1579
+ def _build_k8s(cls, cfg: RuntimeConfig, topo, role, name, props) -> None:
1580
+ """Turn each drawn K8s Cluster + its connected Pods/Autoscalers into a real k3s
1581
+ cluster spec with generated Deployment/Service/HPA manifests."""
1582
+ id_of = {n: i for i, n in name.items()}
1583
+ nbrs: dict[str, list] = {d: [] for d in topo.devices}
1584
+ for l in topo.links.values():
1585
+ nbrs[l.source_id].append(l.target_id)
1586
+ nbrs[l.target_id].append(l.source_id)
1587
+
1588
+ for cdid, r in role.items():
1589
+ if r != "k8scluster":
1590
+ continue
1591
+ deployments = []
1592
+ for nb in nbrs.get(cdid, []):
1593
+ if role.get(nb) != "k8sworkload": # a Pod (= a Deployment)
1594
+ continue
1595
+ p = props.get(nb, {})
1596
+ dep = {"name": _svc(name[nb]),
1597
+ "image": _norm_image(p.get("Image") or "nginxdemos/hello:latest"),
1598
+ "replicas": _int(p.get("Replicas"), 2),
1599
+ "port": _int(p.get("Port"), 80), "hpa": None}
1600
+ for nb2 in nbrs.get(nb, []): # an Autoscaling Group on it -> HPA
1601
+ if role.get(nb2) == "hpa":
1602
+ ap = props.get(nb2, {})
1603
+ dep["hpa"] = {"min": _int(ap.get("Min"), 1),
1604
+ "max": _int(ap.get("Max"), 5),
1605
+ "cpu": _int(ap.get("TargetCPU"), 60)}
1606
+ break
1607
+ deployments.append(dep)
1608
+ manifests = "\n---\n".join(
1609
+ m for d in deployments for m in (
1610
+ _k8s_deployment_yaml(d), _k8s_service_yaml(d),
1611
+ *( [_k8s_hpa_yaml(d)] if d["hpa"] else [] )))
1612
+ cfg.k8s.append(K8sSpec(name=name[cdid], svc=_svc(name[cdid]),
1613
+ image=cls.K3S_IMAGE, deployments=deployments,
1614
+ manifests=manifests))
1615
+
1616
+ @classmethod
1617
+ def _build_fabric(cls, cfg: RuntimeConfig) -> None:
1618
+ """List the cloud services the GINI Cloud Fabric agent should watch, with the
1619
+ per-type probe port + credentials pulled from the catalog config."""
1620
+ watched = []
1621
+ for s in cfg.services:
1622
+ if s.type_key in cls._FABRIC_SKIP:
1623
+ continue
1624
+ port, credkind = cls._FABRIC_PROBE.get(
1625
+ s.type_key, (s.ports[0]["container"] if s.ports else 80, None))
1626
+ if credkind == "postgres":
1627
+ creds = {"user": s.env.get("POSTGRES_USER", "gini"),
1628
+ "password": s.env.get("POSTGRES_PASSWORD", "gini"),
1629
+ "db": s.env.get("POSTGRES_DB", "postgres")}
1630
+ elif credkind == "rabbit":
1631
+ creds = {"user": "guest", "password": "guest"}
1632
+ else:
1633
+ creds = {}
1634
+ watched.append({"name": _svc(s.name), "type": s.type_key,
1635
+ "host": _svc(s.name), "port": port, "creds": creds})
1636
+ if watched:
1637
+ cfg.fabric = FabricSpec(services=watched)
1638
+
1639
+ @staticmethod
1640
+ def _wire_observability(cfg: RuntimeConfig, host_port: int) -> int:
1641
+ """If the canvas has Metrics/Dashboards, make them actually observe the lab:
1642
+ add a cAdvisor sidecar (universal per-container metrics), point Prometheus at it,
1643
+ and provision Grafana with the datasource + a starter dashboard. Returns the next
1644
+ free host port."""
1645
+ metrics = [s for s in cfg.services if s.type_key == "metrics"]
1646
+ dashboards = [s for s in cfg.services if s.type_key == "dashboard"]
1647
+ if not metrics and not dashboards:
1648
+ return host_port
1649
+
1650
+ # A Dashboards (Grafana) element with no Prometheus on the canvas: auto-add a
1651
+ # hidden Prometheus so Grafana always has a datasource + data to show (mirrors the
1652
+ # cAdvisor sidecar). Without this, Grafana loads but has nothing to graph.
1653
+ if dashboards and not metrics:
1654
+ from .cloud_catalog import service_for
1655
+ auto = ServiceSpec(
1656
+ name="Prometheus", type_key="metrics", image=service_for("metrics").image,
1657
+ summary="Auto-added Prometheus backing the dashboard (scrapes cAdvisor).",
1658
+ ports=[{"container": 9090, "host": host_port,
1659
+ "label": "console", "web": True}])
1660
+ host_port += 1
1661
+ cfg.services.append(auto)
1662
+ metrics = [auto]
1663
+ cfg.notes.append("auto-added Prometheus + cAdvisor behind Grafana")
1664
+
1665
+ # cAdvisor — exposes CPU/mem/net for EVERY container, so any topology is
1666
+ # observable without the apps exporting anything. It's infra (not a canvas node).
1667
+ cfg.services.append(ServiceSpec(
1668
+ name="cAdvisor", type_key="_cadvisor",
1669
+ image="gcr.io/cadvisor/cadvisor:v0.49.1",
1670
+ summary="Per-container CPU / memory / network metrics for the whole lab.",
1671
+ command=["-housekeeping_interval=2s", "-docker_only=true"], # fresher data
1672
+ ports=[{"container": 8080, "host": host_port, "label": "cadvisor", "web": True}],
1673
+ volumes=["/:/rootfs:ro", "/var/run:/var/run:ro", "/sys:/sys:ro",
1674
+ "/var/lib/docker/:/var/lib/docker:ro", "/dev/disk/:/dev/disk:ro"],
1675
+ privileged=True))
1676
+ host_port += 1
1677
+
1678
+ for prom in metrics: # Prometheus scrapes cAdvisor (+ itself)
1679
+ prom.files["observability/prometheus.yml"] = _PROMETHEUS_YML
1680
+ prom.volumes.append(
1681
+ "./observability/prometheus.yml:/etc/prometheus/prometheus.yml:ro")
1682
+
1683
+ if dashboards and metrics: # Grafana: datasource -> Prometheus + a dashboard
1684
+ prom_svc = _svc(metrics[0].name)
1685
+ for graf in dashboards:
1686
+ graf.files["observability/grafana/ds.yml"] = _GRAFANA_DS.format(prom=prom_svc)
1687
+ graf.files["observability/grafana/dash.yml"] = _GRAFANA_PROVIDER
1688
+ graf.files["observability/grafana/container.json"] = _grafana_dashboard_json()
1689
+ graf.volumes += [
1690
+ "./observability/grafana/ds.yml:"
1691
+ "/etc/grafana/provisioning/datasources/ds.yml:ro",
1692
+ "./observability/grafana/dash.yml:"
1693
+ "/etc/grafana/provisioning/dashboards/dash.yml:ro",
1694
+ "./observability/grafana/container.json:"
1695
+ "/var/lib/grafana/dashboards/container.json:ro",
1696
+ ]
1697
+ # land students on the provisioned dashboard — set ONLY now that the file
1698
+ # exists (else Grafana errors "Failed to load home dashboard").
1699
+ graf.env["GF_DASHBOARDS_DEFAULT_HOME_DASHBOARD_PATH"] = \
1700
+ "/var/lib/grafana/dashboards/container.json"
1701
+ return host_port
1702
+
1703
+
1704
+ def validate(topo: Topology) -> list[dict]:
1705
+ """Advisory topology lint — never blocks, just surfaces issues a student should see.
1706
+
1707
+ Returns a list of {level: 'warn'|'info', device: name|None, message}.
1708
+ """
1709
+ issues: list[dict] = []
1710
+ role = {d.id: _role(d.type_key) for d in topo.devices.values()}
1711
+ name = {d.id: d.name for d in topo.devices.values()}
1712
+ nbrs: dict[str, list] = {d.id: [] for d in topo.devices.values()}
1713
+ for l in topo.links.values():
1714
+ nbrs[l.source_id].append(l.target_id)
1715
+ nbrs[l.target_id].append(l.source_id)
1716
+
1717
+ # 0. GINI32 boards name real hardware. A missing or duplicated BoardID does not
1718
+ # fail loudly at run time — the relay keys its table by that id, so a duplicate
1719
+ # makes one board silently disappear. Say so on the canvas instead.
1720
+ boards = [d for d in topo.devices.values() if d.type_key == "gini32"]
1721
+ seen_ids: dict[str, str] = {}
1722
+ for d in boards:
1723
+ bid = str((d.properties or {}).get("BoardID", "")).strip()
1724
+ if not bid:
1725
+ issues.append({"level": "warn", "device": d.name,
1726
+ "message": "No BoardID set — put the id from the board's "
1727
+ "label here (see `gini32 provision --id`), or no "
1728
+ "hardware will attach to this element."})
1729
+ elif bid in seen_ids:
1730
+ issues.append({"level": "warn", "device": d.name,
1731
+ "message": f"BoardID {bid!r} is also used by "
1732
+ f"{seen_ids[bid]} — two elements cannot share one "
1733
+ f"physical board; one of them will never connect."})
1734
+ else:
1735
+ seen_ids[bid] = d.name
1736
+ # overlapping physical subnets => routers get two routes to one network
1737
+ import ipaddress as _ipa
1738
+ nets: list[tuple] = []
1739
+ for d in boards:
1740
+ raw = str((d.properties or {}).get("PhysicalSubnet", "")).strip()
1741
+ if not raw:
1742
+ continue # blank is fine: allocated automatically
1743
+ try:
1744
+ net = _ipa.ip_network(raw, strict=False)
1745
+ except ValueError:
1746
+ continue # the compiler already notes and replaces it
1747
+ for other, onet in nets:
1748
+ if net.overlaps(onet):
1749
+ issues.append({"level": "warn", "device": d.name,
1750
+ "message": f"PhysicalSubnet {raw} overlaps {other}'s — "
1751
+ f"give each board its own, or leave both blank "
1752
+ f"to have them allocated."})
1753
+ nets.append((d.name, net))
1754
+
1755
+ # 1. isolated devices (degree 0) — not part of any network. xv6 runs standalone (it
1756
+ # has no networking), its peripherals are optional, and OS Zoo guests run in isolation
1757
+ # (display-only, no fabric wiring in v1), so none of them are "islands".
1758
+ tkey = {d.id: d.type_key for d in topo.devices.values()}
1759
+ for did, r in role.items():
1760
+ if r == "group" or nbrs[did]:
1761
+ continue
1762
+ if r == "oszoo" or tkey[did] in ("xv6", "terminal", "storage_volume"):
1763
+ continue
1764
+ issues.append({"level": "warn", "device": name[did],
1765
+ "message": f"{name[did]} isn't connected to anything."})
1766
+
1767
+ # 2. machines with no gateway (no router on any of their subnets) — islands.
1768
+ # A host on a switched/SDN L2 domain is fine without a router (it reaches its
1769
+ # LAN at layer 2), so only warn for hosts NOT on any switch/OVS.
1770
+ cfg = RuntimeCompiler().compile(topo)
1771
+ id_of = {n: i for i, n in name.items()}
1772
+ for m in cfg.machines:
1773
+ if m.gw or getattr(m, "gateway", False): # gateway egresses via its own uplink
1774
+ continue
1775
+ did = id_of.get(m.name)
1776
+ on_lan = did is not None and any(
1777
+ role.get(nb) in ("switch", "ovs") for nb in nbrs[did])
1778
+ if on_lan:
1779
+ continue
1780
+ issues.append({"level": "warn", "device": m.name,
1781
+ "message": f"{m.name} has no gateway (no router on its "
1782
+ f"subnet) — it can only reach hosts on its own subnet."})
1783
+
1784
+ # 2b. SDN advisories — teach the control-plane relationship
1785
+ for o in cfg.ovs_switches:
1786
+ if not o.controller:
1787
+ issues.append({"level": "warn", "device": o.name,
1788
+ "message": f"{o.name} has no controller — it runs fail-secure "
1789
+ f"with an empty flow table, so it drops ALL traffic. "
1790
+ f"Connect a controller to give it switching behavior."})
1791
+ for c in cfg.controllers:
1792
+ if not c.switches:
1793
+ issues.append({"level": "warn", "device": c.name,
1794
+ "message": f"{c.name} isn't programming any switch — connect "
1795
+ f"it to an OVS for it to control."})
1796
+
1797
+ # 3. L2 loop among switches/hubs — our switches don't run STP, so a loop floods
1798
+ l2 = {"switch", "hub", "ovs"}
1799
+ uf = _UF()
1800
+ looped = False
1801
+ for l in topo.links.values():
1802
+ if role.get(l.source_id) in l2 and role.get(l.target_id) in l2:
1803
+ if uf.find(l.source_id) == uf.find(l.target_id):
1804
+ looped = True
1805
+ uf.union(l.source_id, l.target_id)
1806
+ if looped:
1807
+ issues.append({"level": "warn", "device": None,
1808
+ "message": "Switch loop detected — the switches have no spanning "
1809
+ "tree, so a loop will flood broadcasts. Remove a link."})
1810
+
1811
+ # 4. compiler notes (e.g. grouping devices whose links are organizational only)
1812
+ for note in cfg.notes:
1813
+ issues.append({"level": "info", "device": None, "message": note})
1814
+ return issues
1815
+
1816
+
1817
+ def address_map(topo: Topology) -> dict[str, dict]:
1818
+ """Per-device addressing for the inspector / canvas labels.
1819
+
1820
+ Returns {device_name: {role, interfaces:[{name, ip, mac, subnet, gateway, peer}], …}}.
1821
+ IPs/MACs come from compiling the topology, so they exist before anything runs.
1822
+ """
1823
+ import ipaddress
1824
+ cfg = RuntimeCompiler().compile(topo)
1825
+
1826
+ def subnet(cidr: str) -> str:
1827
+ return str(ipaddress.ip_interface(cidr).network)
1828
+
1829
+ out: dict[str, dict] = {}
1830
+ for m in cfg.machines:
1831
+ out[m.name] = {"role": "machine", "interfaces": [
1832
+ {"name": f"eth{i}", "ip": itf.ip, "mac": itf.mac, "subnet": subnet(itf.ip),
1833
+ "gateway": m.gw if i == 0 else None, "peer": itf.ep.peer.device,
1834
+ "link_id": itf.link_id}
1835
+ for i, itf in enumerate(m.ifaces)]}
1836
+ for r in cfg.routers:
1837
+ out[r.name] = {"role": "router", "interfaces": [
1838
+ {"name": f"eth{i}", "ip": itf.ip, "mac": itf.mac, "subnet": subnet(itf.ip),
1839
+ "gateway": None, "peer": itf.ep.peer.device, "link_id": itf.link_id}
1840
+ for i, itf in enumerate(r.ifaces)]}
1841
+ for s in cfg.switches:
1842
+ out[s.name] = {"role": "switch", "ports": len(s.eps), "interfaces": [],
1843
+ "peers": [e.peer.device for e in s.eps]}
1844
+ return out
1845
+
1846
+
1847
+ def overlay_hosts(addressing: dict) -> dict:
1848
+ """device name -> its primary overlay (gini0) IP, from `address_map` output. GINI writes these
1849
+ into each machine's /etc/hosts so names resolve over the DRAWN network (gini0) instead of the
1850
+ Docker bridge — which is what makes DNS/getent/ping/reach ride the overlay."""
1851
+ out: dict[str, str] = {}
1852
+ for name, info in (addressing or {}).items():
1853
+ for itf in info.get("interfaces", []):
1854
+ ip = str(itf.get("ip", "")).split("/")[0].strip()
1855
+ if ip:
1856
+ out[name] = ip
1857
+ break
1858
+ return out