gini-toolkit 6.0.1.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gini/__init__.py +12 -0
- gini/__main__.py +107 -0
- gini/_version.py +24 -0
- gini/agent/__init__.py +17 -0
- gini/agent/agent_gamemaster.py +140 -0
- gini/agent/api.py +291 -0
- gini/agent/ask.py +123 -0
- gini/agent/authoring.py +72 -0
- gini/agent/blackboard.py +114 -0
- gini/agent/contracts.py +142 -0
- gini/agent/domains.py +91 -0
- gini/agent/embed.py +123 -0
- gini/agent/gamemaster.py +256 -0
- gini/agent/kb.py +148 -0
- gini/agent/lesson_resolver.py +261 -0
- gini/agent/llm/__init__.py +5 -0
- gini/agent/llm/backend.py +43 -0
- gini/agent/llm/fake.py +25 -0
- gini/agent/llm/ollama.py +206 -0
- gini/agent/loop.py +258 -0
- gini/agent/mcp_server.py +86 -0
- gini/agent/meaning.py +225 -0
- gini/agent/mission.py +210 -0
- gini/agent/mission_controller.py +208 -0
- gini/agent/narration.py +116 -0
- gini/agent/notifier.py +86 -0
- gini/agent/personas.py +79 -0
- gini/agent/reasoning.py +172 -0
- gini/agent/recall.py +248 -0
- gini/agent/session.py +79 -0
- gini/agent/teaching_center.py +482 -0
- gini/agent/tools/__init__.py +3 -0
- gini/agent/tools/registry.py +193 -0
- gini/agent/twin/__init__.py +28 -0
- gini/agent/twin/authoring.py +71 -0
- gini/agent/twin/contracts.py +54 -0
- gini/agent/twin/dialectic.py +189 -0
- gini/agent/twin/harness.py +93 -0
- gini/agent/twin/justify.py +156 -0
- gini/agent/twin/learner.py +64 -0
- gini/agent/twin/mission.py +60 -0
- gini/agent/twin/os_coach.py +79 -0
- gini/agent/twin/salience.py +30 -0
- gini/agent/understand.py +250 -0
- gini/agent/verifiers.py +106 -0
- gini/agent/wizard.py +178 -0
- gini/agent/xv6_pack.py +74 -0
- gini/app/__init__.py +3 -0
- gini/app/context.py +368 -0
- gini/app/paths.py +121 -0
- gini/data/README.md +21 -0
- gini/domain/__init__.py +9 -0
- gini/domain/assembly.py +209 -0
- gini/domain/authoring.py +353 -0
- gini/domain/blueprints.py +5 -0
- gini/domain/capabilities.py +177 -0
- gini/domain/catalog.py +85 -0
- gini/domain/certify.py +201 -0
- gini/domain/compose.py +413 -0
- gini/domain/composition.py +88 -0
- gini/domain/concepts.py +383 -0
- gini/domain/connection_rules.py +269 -0
- gini/domain/constraints.py +153 -0
- gini/domain/content.py +59 -0
- gini/domain/cpu_journey.py +89 -0
- gini/domain/devices.py +747 -0
- gini/domain/diagnose.py +201 -0
- gini/domain/element_guide.py +327 -0
- gini/domain/explain.py +90 -0
- gini/domain/fingerprint.py +201 -0
- gini/domain/firewall.py +34 -0
- gini/domain/flowlog.py +61 -0
- gini/domain/flowtable.py +179 -0
- gini/domain/fragment_yaml.py +230 -0
- gini/domain/fragments.py +169 -0
- gini/domain/games/__init__.py +2 -0
- gini/domain/games/paging_games.py +119 -0
- gini/domain/games/policy_game.py +86 -0
- gini/domain/games/process_game.py +48 -0
- gini/domain/games/thrash_game.py +75 -0
- gini/domain/games/translate_game.py +60 -0
- gini/domain/games/trap_game.py +86 -0
- gini/domain/grader.py +155 -0
- gini/domain/grouping.py +67 -0
- gini/domain/legality.py +103 -0
- gini/domain/lesson.py +241 -0
- gini/domain/lexicon.py +150 -0
- gini/domain/machine_state.py +410 -0
- gini/domain/missions/networking/basic-lan.yaml +32 -0
- gini/domain/missions/networking/cache-in-front.yaml +23 -0
- gini/domain/missions/networking/decouple-with-queue.yaml +31 -0
- gini/domain/missions/networking/drive-load.yaml +20 -0
- gini/domain/missions/networking/fix-the-address.yaml +75 -0
- gini/domain/missions/networking/fix-the-lan.yaml +43 -0
- gini/domain/missions/networking/inspect-flows.yaml +16 -0
- gini/domain/missions/networking/k8s-autoscale.yaml +27 -0
- gini/domain/missions/networking/least-privilege.yaml +21 -0
- gini/domain/missions/networking/load-balanced-web.yaml +29 -0
- gini/domain/missions/networking/observe-it.yaml +24 -0
- gini/domain/missions/networking/put-in-vpc.yaml +30 -0
- gini/domain/missions/networking/reachability-boundary.yaml +56 -0
- gini/domain/missions/networking/sdn-reactive.yaml +35 -0
- gini/domain/missions/networking/send-request.yaml +19 -0
- gini/domain/missions/networking/serverless-api.yaml +25 -0
- gini/domain/missions/networking/service-chain.yaml +33 -0
- gini/domain/missions/os/lottery-fix.yaml +19 -0
- gini/domain/missions/os/priority-fix.yaml +24 -0
- gini/domain/missions.py +111 -0
- gini/domain/modulechain.py +36 -0
- gini/domain/objectives.py +488 -0
- gini/domain/os_zoo.py +79 -0
- gini/domain/paging_sim.py +141 -0
- gini/domain/pricing.py +199 -0
- gini/domain/probes.py +226 -0
- gini/domain/profile.py +142 -0
- gini/domain/recipes.py +738 -0
- gini/domain/riders.py +309 -0
- gini/domain/router_modules.py +224 -0
- gini/domain/routetable.py +67 -0
- gini/domain/scoring.py +76 -0
- gini/domain/staging.py +122 -0
- gini/domain/syscall_builder.py +144 -0
- gini/domain/topic_cloud.py +62 -0
- gini/domain/topology.py +213 -0
- gini/domain/vocabulary.py +51 -0
- gini/domain/xv6.py +808 -0
- gini/domain/xv6_fs.py +250 -0
- gini/domain/xv6_runner.py +113 -0
- gini/domain/xv6_vm.py +385 -0
- gini/gloader.py +17 -0
- gini/runtime/__init__.py +18 -0
- gini/runtime/cloudfabric_agent.py +370 -0
- gini/runtime/console.py +68 -0
- gini/runtime/control.py +70 -0
- gini/runtime/frame.py +138 -0
- gini/runtime/gbridge.py +638 -0
- gini/runtime/grouter.py +223 -0
- gini/runtime/hostsim.py +90 -0
- gini/runtime/shuttle.py +348 -0
- gini/runtime/switch.py +109 -0
- gini/runtime/transport.py +77 -0
- gini/runtime/xv6_bridge.py +312 -0
- gini/server/__init__.py +22 -0
- gini/server/__main__.py +74 -0
- gini/server/app.py +140 -0
- gini/server/auth.py +82 -0
- gini/server/policy.py +57 -0
- gini/server/session.py +23 -0
- gini/services/__init__.py +15 -0
- gini/services/boardflash.py +248 -0
- gini/services/boardsetup.py +374 -0
- gini/services/cloud_catalog.py +143 -0
- gini/services/compiler.py +1858 -0
- gini/services/discovery.py +324 -0
- gini/services/gloader.py +183 -0
- gini/services/orchestrator.py +1460 -0
- gini/services/persistence.py +28 -0
- gini/services/probe_runner.py +149 -0
- gini/services/project.py +217 -0
- gini/services/remote.py +93 -0
- gini/services/rider_runner.py +96 -0
- gini/services/rider_session.py +171 -0
- gini/services/shadow_store.py +52 -0
- gini/services/terminal.py +45 -0
- gini/setup/__init__.py +17 -0
- gini/setup/cli.py +109 -0
- gini/setup/images.py +33 -0
- gini/setup/marker.py +43 -0
- gini/setup/runtime.py +69 -0
- gini/ui/__init__.py +3 -0
- gini/ui/assets/app_icon.icns +0 -0
- gini/ui/assets/app_icon.ico +0 -0
- gini/ui/assets/app_icon.png +0 -0
- gini/ui/assets/app_icon_1024.png +0 -0
- gini/ui/assets/cue/_w.txt +1 -0
- gini/ui/assets/cue/ai.png +0 -0
- gini/ui/assets/cue/canvas.png +0 -0
- gini/ui/assets/cue/cloud.png +0 -0
- gini/ui/assets/cue/cost.png +0 -0
- gini/ui/assets/cue/dark/ai.png +0 -0
- gini/ui/assets/cue/dark/canvas.png +0 -0
- gini/ui/assets/cue/dark/cloud.png +0 -0
- gini/ui/assets/cue/dark/cost.png +0 -0
- gini/ui/assets/cue/dark/metrics.png +0 -0
- gini/ui/assets/cue/dark/router.png +0 -0
- gini/ui/assets/cue/dark/run.png +0 -0
- gini/ui/assets/cue/dark/serverless.png +0 -0
- gini/ui/assets/cue/dark/settings.png +0 -0
- gini/ui/assets/cue/dark/welcome.png +0 -0
- gini/ui/assets/cue/dark/wizard.png +0 -0
- gini/ui/assets/cue/ginibrand/ai.png +0 -0
- gini/ui/assets/cue/ginibrand/canvas.png +0 -0
- gini/ui/assets/cue/ginibrand/cloud.png +0 -0
- gini/ui/assets/cue/ginibrand/cost.png +0 -0
- gini/ui/assets/cue/ginibrand/metrics.png +0 -0
- gini/ui/assets/cue/ginibrand/router.png +0 -0
- gini/ui/assets/cue/ginibrand/run.png +0 -0
- gini/ui/assets/cue/ginibrand/serverless.png +0 -0
- gini/ui/assets/cue/ginibrand/settings.png +0 -0
- gini/ui/assets/cue/ginibrand/welcome.png +0 -0
- gini/ui/assets/cue/ginibrand/wizard.png +0 -0
- gini/ui/assets/cue/highcontrast/ai.png +0 -0
- gini/ui/assets/cue/highcontrast/canvas.png +0 -0
- gini/ui/assets/cue/highcontrast/cloud.png +0 -0
- gini/ui/assets/cue/highcontrast/cost.png +0 -0
- gini/ui/assets/cue/highcontrast/metrics.png +0 -0
- gini/ui/assets/cue/highcontrast/router.png +0 -0
- gini/ui/assets/cue/highcontrast/run.png +0 -0
- gini/ui/assets/cue/highcontrast/serverless.png +0 -0
- gini/ui/assets/cue/highcontrast/settings.png +0 -0
- gini/ui/assets/cue/highcontrast/welcome.png +0 -0
- gini/ui/assets/cue/highcontrast/wizard.png +0 -0
- gini/ui/assets/cue/light/ai.png +0 -0
- gini/ui/assets/cue/light/canvas.png +0 -0
- gini/ui/assets/cue/light/cloud.png +0 -0
- gini/ui/assets/cue/light/cost.png +0 -0
- gini/ui/assets/cue/light/metrics.png +0 -0
- gini/ui/assets/cue/light/router.png +0 -0
- gini/ui/assets/cue/light/run.png +0 -0
- gini/ui/assets/cue/light/serverless.png +0 -0
- gini/ui/assets/cue/light/settings.png +0 -0
- gini/ui/assets/cue/light/welcome.png +0 -0
- gini/ui/assets/cue/light/wizard.png +0 -0
- gini/ui/assets/cue/metrics.png +0 -0
- gini/ui/assets/cue/router.png +0 -0
- gini/ui/assets/cue/run.png +0 -0
- gini/ui/assets/cue/serverless.png +0 -0
- gini/ui/assets/cue/settings.png +0 -0
- gini/ui/assets/cue/welcome.png +0 -0
- gini/ui/assets/cue/wizard.png +0 -0
- gini/ui/assistant.py +2111 -0
- gini/ui/author_dialog.py +184 -0
- gini/ui/board_dialog.py +247 -0
- gini/ui/branding.py +21 -0
- gini/ui/canvas.py +2007 -0
- gini/ui/chat_panel.py +7 -0
- gini/ui/cpu_journey.py +212 -0
- gini/ui/cpu_lab.py +306 -0
- gini/ui/cue_cards.py +214 -0
- gini/ui/dashboard.py +222 -0
- gini/ui/diagnose_game.py +336 -0
- gini/ui/fingerprint_lab.py +219 -0
- gini/ui/flash_dialog.py +244 -0
- gini/ui/flow_layout.py +63 -0
- gini/ui/fragment_manager.py +1415 -0
- gini/ui/game_catalog.py +184 -0
- gini/ui/game_renderers.py +340 -0
- gini/ui/games_lab.py +90 -0
- gini/ui/inspector.py +1055 -0
- gini/ui/live_metrics.py +130 -0
- gini/ui/machine_lab.py +1412 -0
- gini/ui/main_window.py +3153 -0
- gini/ui/memory_lab.py +371 -0
- gini/ui/mission_panel.py +302 -0
- gini/ui/mode_indicator.py +227 -0
- gini/ui/palette.py +112 -0
- gini/ui/peripherals.py +218 -0
- gini/ui/process_tree.py +130 -0
- gini/ui/reset_dialog.py +179 -0
- gini/ui/router_lab.py +776 -0
- gini/ui/run_button.py +183 -0
- gini/ui/settings_dialog.py +234 -0
- gini/ui/signin_dialog.py +111 -0
- gini/ui/storage_lab.py +219 -0
- gini/ui/syscall_builder.py +235 -0
- gini/ui/syscall_lab.py +152 -0
- gini/ui/theme/__init__.py +5 -0
- gini/ui/theme/icons.py +145 -0
- gini/ui/theme/manager.py +291 -0
- gini/ui/theme/tokens.py +194 -0
- gini/ui/trap_lab.py +270 -0
- gini/ui/worker_host.py +102 -0
- gini/ui/zoo_lab.py +112 -0
- gini_toolkit-6.0.1.dev0.dist-info/METADATA +77 -0
- gini_toolkit-6.0.1.dev0.dist-info/RECORD +278 -0
- gini_toolkit-6.0.1.dev0.dist-info/WHEEL +5 -0
- gini_toolkit-6.0.1.dev0.dist-info/entry_points.txt +3 -0
- gini_toolkit-6.0.1.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,1858 @@
|
|
|
1
|
+
"""RuntimeCompiler — lower a canvas Topology onto the portable user-space runtime.
|
|
2
|
+
|
|
3
|
+
Generalizes the hand-written R0 wiring to any topology:
|
|
4
|
+
* classify devices (machine / switch / router / grouping),
|
|
5
|
+
* find L2 broadcast domains (segments) and give each a subnet,
|
|
6
|
+
* assign IPs, gateways, MACs, and per-endpoint UDP ports,
|
|
7
|
+
* emit machine/switch/router specs that gini.runtime can run (in-process or Docker).
|
|
8
|
+
|
|
9
|
+
Cloud "grouping" devices (VPC, cloud-subnet, region, cluster, pod, instance-group) are
|
|
10
|
+
organizational in R0 and are skipped as runtime nodes (links touching them are
|
|
11
|
+
dropped, with a note). Cloud endpoints (instances, containers, LBs, …) run as machines.
|
|
12
|
+
(Plain IP subnets are NOT a device: each L2 broadcast domain is auto-assigned a /24.)
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import os
|
|
17
|
+
import re
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
|
|
21
|
+
from ..domain import devices as _dev
|
|
22
|
+
from ..domain.topology import Topology
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _gini_home() -> Path:
|
|
26
|
+
# Same rule as app.paths.gini_home, replicated so this service avoids an `app` import cycle.
|
|
27
|
+
return Path(os.environ.get("GINI_HOME_DIR") or (Path.home() / ".gini")).expanduser()
|
|
28
|
+
from .cloud_catalog import is_service, service_for
|
|
29
|
+
|
|
30
|
+
ROUTERS = {"router", "firewall"}
|
|
31
|
+
SWITCHES = {"switch", "hub"} # plain L2 — live in the shared `fabric` container
|
|
32
|
+
GROUPS = {"vpc", "cloud_subnet", "region"}
|
|
33
|
+
# OS Zoo: emulated historical OSes, one container each, screen served over noVNC (a web port).
|
|
34
|
+
# "BYO-style" elements carry Emulator/Image/Rom properties and boot via the generic BYO path — the
|
|
35
|
+
# generic "Classic OS (your image)" plus the convenience presets (Mac System 7, Windows 3.11) whose
|
|
36
|
+
# properties are simply pre-filled with a download URL (GINI still ships no proprietary image).
|
|
37
|
+
OSZOO_BYO_KEYS = {"oszoo_byo", "msdos", "mac7", "win31"}
|
|
38
|
+
OSZOO_KEYS = {"freedos", "kolibri", "menuet"} | OSZOO_BYO_KEYS
|
|
39
|
+
# Sources/Sinks: instruments that run INSIDE a donor container — they get no runtime node
|
|
40
|
+
# of their own, and their (attach) edges are never wired.
|
|
41
|
+
RIDERS = {k for k, dt in _dev.REGISTRY.items() if getattr(dt, "rider", False)}
|
|
42
|
+
K8S_ROLES = {"k8s_cluster": "k8scluster", "pod": "k8sworkload",
|
|
43
|
+
"instance_group": "hpa", "k8s_node": "k8snode"}
|
|
44
|
+
|
|
45
|
+
# SDN: an OVS is an OpenFlow switch that runs as its OWN container (the gRouter in
|
|
46
|
+
# --openflow mode), programmed by a controller over a management channel. A controller
|
|
47
|
+
# is the control plane — it is NOT a data host and gets no data-plane IP/gateway.
|
|
48
|
+
DEFAULT_OF_PORT = 6633
|
|
49
|
+
DEFAULT_OF_APP = "gini.samples.switch"
|
|
50
|
+
|
|
51
|
+
# GINI32: every real board is served by one shared relay container, so a board's fabric
|
|
52
|
+
# endpoint lives at this service name. (The board itself is out on the physical LAN.)
|
|
53
|
+
GBRIDGE_SVC = "gbridge"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _role(type_key: str) -> str:
|
|
57
|
+
if type_key in ROUTERS:
|
|
58
|
+
return "router"
|
|
59
|
+
if type_key in SWITCHES:
|
|
60
|
+
return "switch"
|
|
61
|
+
if type_key == "ovs":
|
|
62
|
+
return "ovs"
|
|
63
|
+
if type_key == "controller":
|
|
64
|
+
return "controller"
|
|
65
|
+
if type_key in K8S_ROLES: # kubernetes: real k3s cluster + manifests
|
|
66
|
+
return K8S_ROLES[type_key]
|
|
67
|
+
if type_key == "function": # serverless — a handler in the shared faas runtime
|
|
68
|
+
return "function"
|
|
69
|
+
if is_service(type_key): # cloud managed service (own container, bridge net)
|
|
70
|
+
return "service"
|
|
71
|
+
if type_key in ("instance", "container", "kinstance"): # cloud compute — own bridge container
|
|
72
|
+
return "compute" # (kinstance = VM-isolated via Kata)
|
|
73
|
+
if type_key == "security_group": # a policy, not a container — drives member iptables
|
|
74
|
+
return "secgroup"
|
|
75
|
+
if type_key == "vnf": # NFV: an inline network function (forwarding container)
|
|
76
|
+
return "vnf"
|
|
77
|
+
if type_key == "xv6": # standalone teaching kernel (QEMU-RISC-V) — no fabric
|
|
78
|
+
return "xv6"
|
|
79
|
+
if type_key in OSZOO_KEYS: # OS Zoo: emulated historical OS, embedded via noVNC
|
|
80
|
+
return "oszoo"
|
|
81
|
+
if type_key == "gini32": # a REAL ESP32 board: addressed on the fabric like a
|
|
82
|
+
return "gini32" # host, but reached through the gbridge relay, not a
|
|
83
|
+
# container of its own — see _build_gbridge().
|
|
84
|
+
if type_key in ("terminal", "storage_volume"): # xv6 peripherals: pure UI, no
|
|
85
|
+
return "peripheral" # container; never emitted/addressed
|
|
86
|
+
if type_key in RIDERS: # Sources/Sinks: run on a donor, no container of their own
|
|
87
|
+
return "rider"
|
|
88
|
+
if type_key in GROUPS:
|
|
89
|
+
return "group"
|
|
90
|
+
return "machine" # host = a node on the simulated tun fabric
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _svc(name: str) -> str:
|
|
94
|
+
return re.sub(r"[^a-z0-9]", "", name.lower()) or "node"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _hostname(name: str) -> str:
|
|
98
|
+
"""The hostname to set inside the machine container — the user's element name (e.g.
|
|
99
|
+
'toronto.on') made into a valid hostname, so `hostname` at the shell matches the label on
|
|
100
|
+
the canvas and the student needs no mental mapping. Falls back to the service name."""
|
|
101
|
+
h = re.sub(r"[^a-zA-Z0-9.-]", "-", (name or "").strip()).strip(".-")
|
|
102
|
+
return h or _svc(name)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _cpus_for(device) -> float:
|
|
106
|
+
"""CPU limit (vCPUs) for a device from its size tier — 0.5/1/2/4 for S/M/L/XL."""
|
|
107
|
+
from ..domain import pricing
|
|
108
|
+
return pricing.size_tier(pricing.size_level(getattr(device, "size", 1)))[1]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _toolkit_for(device) -> str:
|
|
112
|
+
"""Which Machine image this host is built from — read off its `Toolkit` property.
|
|
113
|
+
|
|
114
|
+
LEAN is the default and the one to encourage: an Alpine host with the tools a student actually
|
|
115
|
+
types (ip/ping/tcpdump/dig/curl/nc/iperf3/nmap), an order of magnitude smaller than the Debian
|
|
116
|
+
image. A host only opts into FULL when its experiment genuinely needs the heavy services —
|
|
117
|
+
bind9 for the DNS chapter, postfix for mail, ettercap/dsniff for the spoofing labs.
|
|
118
|
+
|
|
119
|
+
Anything unrecognised means lean: a typo must not silently pull in the 10x image."""
|
|
120
|
+
if getattr(device, "type_key", "") == "desktop": # the headful Desktop element is always gui
|
|
121
|
+
return "gui"
|
|
122
|
+
props = getattr(device, "properties", None) or {}
|
|
123
|
+
want = str(props.get("Toolkit", "")).strip().lower()
|
|
124
|
+
return want if want in ("full", "security", "gui") else "lean"
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def _xv6_harts(device) -> int:
|
|
128
|
+
"""xv6 QEMU CPU count (-smp), capped at 2: S/M -> 1 hart, L/XL -> 2 harts. (xv6 SMP is kept
|
|
129
|
+
to 1-2 for a clear, legible scheduler demo.)"""
|
|
130
|
+
v = _cpus_for(device) # the shown vCPU count: 0.5/1/2/4
|
|
131
|
+
return max(1, min(2, int(round(v))))
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _int(v, default: int) -> int:
|
|
135
|
+
try:
|
|
136
|
+
return int(float(v))
|
|
137
|
+
except (TypeError, ValueError):
|
|
138
|
+
return default
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _k8s_deployment_yaml(d: dict) -> str:
|
|
142
|
+
return (f"apiVersion: apps/v1\nkind: Deployment\nmetadata:\n name: {d['name']}\n"
|
|
143
|
+
f" labels: {{app: {d['name']}}}\nspec:\n replicas: {d['replicas']}\n"
|
|
144
|
+
f" selector:\n matchLabels: {{app: {d['name']}}}\n template:\n"
|
|
145
|
+
f" metadata:\n labels: {{app: {d['name']}}}\n spec:\n"
|
|
146
|
+
f" containers:\n - name: {d['name']}\n image: {d['image']}\n"
|
|
147
|
+
f" ports:\n - containerPort: {d['port']}\n"
|
|
148
|
+
f" resources:\n requests:\n cpu: 50m")
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _k8s_service_yaml(d: dict) -> str:
|
|
152
|
+
return (f"apiVersion: v1\nkind: Service\nmetadata:\n name: {d['name']}\nspec:\n"
|
|
153
|
+
f" selector: {{app: {d['name']}}}\n ports:\n - port: {d['port']}\n"
|
|
154
|
+
f" targetPort: {d['port']}")
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _k8s_hpa_yaml(d: dict) -> str:
|
|
158
|
+
h = d["hpa"]
|
|
159
|
+
return (f"apiVersion: autoscaling/v2\nkind: HorizontalPodAutoscaler\nmetadata:\n"
|
|
160
|
+
f" name: {d['name']}\nspec:\n scaleTargetRef:\n apiVersion: apps/v1\n"
|
|
161
|
+
f" kind: Deployment\n name: {d['name']}\n minReplicas: {h['min']}\n"
|
|
162
|
+
f" maxReplicas: {h['max']}\n metrics:\n - type: Resource\n"
|
|
163
|
+
f" resource:\n name: cpu\n target:\n"
|
|
164
|
+
f" type: Utilization\n averageUtilization: {h['cpu']}")
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _norm_image(raw: str) -> str:
|
|
168
|
+
"""Turn a friendly image property into a real Docker tag.
|
|
169
|
+
'ubuntu-22.04' -> 'ubuntu:22.04'; an explicit tag/registry is kept as-is."""
|
|
170
|
+
raw = (raw or "").strip()
|
|
171
|
+
if not raw:
|
|
172
|
+
return "ubuntu:22.04"
|
|
173
|
+
if ":" in raw or "/" in raw:
|
|
174
|
+
return raw
|
|
175
|
+
if "-" in raw: # ubuntu-22.04 -> ubuntu:22.04
|
|
176
|
+
n, _, v = raw.partition("-")
|
|
177
|
+
return f"{n}:{v}"
|
|
178
|
+
return raw + ":latest"
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
class _UF:
|
|
182
|
+
def __init__(self) -> None:
|
|
183
|
+
self.p: dict = {}
|
|
184
|
+
|
|
185
|
+
def find(self, x):
|
|
186
|
+
self.p.setdefault(x, x)
|
|
187
|
+
while self.p[x] != x:
|
|
188
|
+
self.p[x] = self.p[self.p[x]]
|
|
189
|
+
x = self.p[x]
|
|
190
|
+
return x
|
|
191
|
+
|
|
192
|
+
def union(self, a, b):
|
|
193
|
+
self.p[self.find(a)] = self.find(b)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
@dataclass
|
|
197
|
+
class Endpoint:
|
|
198
|
+
device: str
|
|
199
|
+
location: str # service name (machine) or "fabric"
|
|
200
|
+
bind_port: int = 0
|
|
201
|
+
peer: "Endpoint | None" = None
|
|
202
|
+
|
|
203
|
+
def peer_host(self, docker: bool) -> str:
|
|
204
|
+
if not docker:
|
|
205
|
+
return "127.0.0.1"
|
|
206
|
+
if self.location == "fabric" and self.peer.location == "fabric":
|
|
207
|
+
return "127.0.0.1"
|
|
208
|
+
return self.peer.location
|
|
209
|
+
|
|
210
|
+
def wiring(self, docker: bool) -> dict:
|
|
211
|
+
return {"bind_host": "0.0.0.0" if docker else "127.0.0.1",
|
|
212
|
+
"bind_port": self.bind_port,
|
|
213
|
+
"peer_host": self.peer_host(docker),
|
|
214
|
+
"peer_port": self.peer.bind_port}
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
@dataclass
|
|
218
|
+
class MachineSpec:
|
|
219
|
+
name: str
|
|
220
|
+
ifaces: list["IfaceSpec"] # one per segment the machine is on (multi-homing)
|
|
221
|
+
gw: str | None # default gateway (first segment that has a router)
|
|
222
|
+
# --- internet / NAT gateway (the drawn "Internet" element) --------------- #
|
|
223
|
+
gateway: bool = False # this node is the on-fabric NAT gateway to the world
|
|
224
|
+
fabric_default: bool = False # send 0.0.0.0/0 INTO the fabric (egress via the gateway)
|
|
225
|
+
fabric_gw: str | None = None # for the gateway: the local router IP for the return path
|
|
226
|
+
cpus: float = 0.0 # CPU limit from the size tier (0 = unset)
|
|
227
|
+
# Which image this host is built from: "lean" (Alpine, the default — small and fast to boot)
|
|
228
|
+
# or "full" (Debian + bind9/postfix/ettercap…, for the book's heavy experiments). This is a
|
|
229
|
+
# DIFFERENT axis from the size tier above: size = how much CPU it gets and what it costs;
|
|
230
|
+
# toolkit = what software is installed in it. A lean host with an XL cap is perfectly valid.
|
|
231
|
+
toolkit: str = "lean"
|
|
232
|
+
novnc_port: int = 0 # headful ("gui") host: published host port for its noVNC console
|
|
233
|
+
# --- inline VNF (NFV service function) ----------------------------------- #
|
|
234
|
+
forward: bool = False # IP-forward between its interfaces (a transit node)
|
|
235
|
+
nf: str = "" # the network function kind: firewall|block|ids|cache|shaper
|
|
236
|
+
nf_rules: str = "" # the function's config (e.g. firewall: "deny 10.0.3.0/24")
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
@dataclass
|
|
240
|
+
class IfaceSpec:
|
|
241
|
+
ip: str
|
|
242
|
+
mac: str
|
|
243
|
+
ep: Endpoint
|
|
244
|
+
link_id: str = "" # the topology link this interface sits on
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _is_ipv4(s: str) -> bool:
|
|
248
|
+
parts = (s or "").strip().split(".")
|
|
249
|
+
if len(parts) != 4:
|
|
250
|
+
return False
|
|
251
|
+
try:
|
|
252
|
+
return all(0 <= int(p) <= 255 for p in parts)
|
|
253
|
+
except ValueError:
|
|
254
|
+
return False
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _valid_cidr(s: str) -> bool:
|
|
258
|
+
import ipaddress
|
|
259
|
+
try:
|
|
260
|
+
ipaddress.ip_network((s or "").strip(), strict=False)
|
|
261
|
+
return True
|
|
262
|
+
except (ValueError, TypeError):
|
|
263
|
+
return False
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _parse_ingress(text: str, sg_by_name: dict) -> list:
|
|
267
|
+
"""Parse a Security Group's Ingress field into rule dicts {port, cidrs, svcs}. Rules are
|
|
268
|
+
`;`/newline separated, each like '5432 from app', '80 from anywhere', 'from 10.0.0.0/8'
|
|
269
|
+
or '443'. 'from <sg-name>' expands to that SG's member service names (SG→SG rules)."""
|
|
270
|
+
import re as _re
|
|
271
|
+
out = []
|
|
272
|
+
for raw in _re.split(r"[;\n]", text or ""):
|
|
273
|
+
s = raw.strip()
|
|
274
|
+
if not s:
|
|
275
|
+
continue
|
|
276
|
+
m = _re.match(r"(?i)^(?:port\s+)?(\d+|all|any|\*)?\s*(?:from\s+(.+))?$", s)
|
|
277
|
+
if not m:
|
|
278
|
+
continue
|
|
279
|
+
ptok = (m.group(1) or "").lower()
|
|
280
|
+
src = (m.group(2) or "anywhere").strip()
|
|
281
|
+
port = None if ptok in ("", "all", "any", "*") else int(ptok)
|
|
282
|
+
cidrs, svcs, low = [], [], src.lower()
|
|
283
|
+
if low in ("anywhere", "any", "all", "0.0.0.0/0", "public", "internet"):
|
|
284
|
+
cidrs.append("0.0.0.0/0")
|
|
285
|
+
elif "/" in src and _valid_cidr(src):
|
|
286
|
+
cidrs.append(src)
|
|
287
|
+
elif low in sg_by_name:
|
|
288
|
+
svcs.extend(sg_by_name[low])
|
|
289
|
+
else:
|
|
290
|
+
svcs.append(_svc(src)) # treat an unknown name as a service/host
|
|
291
|
+
out.append({"port": port, "cidrs": cidrs, "svcs": svcs})
|
|
292
|
+
return out
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _sg_script(rules: list) -> str:
|
|
296
|
+
"""A stateful default-deny-inbound iptables script for one member's union of SG rules."""
|
|
297
|
+
L = [
|
|
298
|
+
"iptables -P INPUT DROP", "iptables -P FORWARD DROP", "iptables -P OUTPUT ACCEPT",
|
|
299
|
+
"iptables -A INPUT -i lo -j ACCEPT",
|
|
300
|
+
"iptables -A INPUT -m conntrack --ctstate ESTABLISHED,RELATED -j ACCEPT",
|
|
301
|
+
# the GINI telemetry agent is infra — always allow it (so the dashboard still polls)
|
|
302
|
+
"for ip in $(getent hosts cloudfabric 2>/dev/null | awk '{print $1}'); do "
|
|
303
|
+
'iptables -A INPUT -s "$ip" -j ACCEPT; done',
|
|
304
|
+
]
|
|
305
|
+
for r in rules:
|
|
306
|
+
dport = f" --dport {r['port']}" if r.get("port") else ""
|
|
307
|
+
for cidr in r.get("cidrs", []):
|
|
308
|
+
L.append(f"iptables -A INPUT -p tcp{dport} -s {cidr} -j ACCEPT")
|
|
309
|
+
if r.get("svcs"):
|
|
310
|
+
names = " ".join(r["svcs"])
|
|
311
|
+
L.append(f"for ip in $(getent hosts {names} 2>/dev/null | awk '{{print $1}}'); "
|
|
312
|
+
f'do iptables -A INPUT -p tcp{dport} -s "$ip" -j ACCEPT; done')
|
|
313
|
+
return "\n".join(L) + "\n"
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
@dataclass
|
|
317
|
+
class SwitchSpec:
|
|
318
|
+
name: str
|
|
319
|
+
eps: list[Endpoint]
|
|
320
|
+
hub: bool = False # True = a Layer-1 hub (flood-all repeater), not a learning switch
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
@dataclass
|
|
324
|
+
class RouterSpec:
|
|
325
|
+
name: str
|
|
326
|
+
ifaces: list[IfaceSpec]
|
|
327
|
+
routes: list = field(default_factory=list) # static inter-router routes:
|
|
328
|
+
# {net, mask, gw, dev} (dev = tun index)
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
@dataclass
|
|
332
|
+
class OvsSpec:
|
|
333
|
+
"""An OpenFlow switch: the gRouter in --openflow mode, in its own container,
|
|
334
|
+
programmed by `controller` (a service name) over OpenFlow on `controller_port`."""
|
|
335
|
+
name: str
|
|
336
|
+
eps: list[Endpoint]
|
|
337
|
+
controller: str | None = None # controller service name (host), or None
|
|
338
|
+
controller_port: int = DEFAULT_OF_PORT
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
@dataclass
|
|
342
|
+
class ControllerSpec:
|
|
343
|
+
"""An SDN controller: a POX container running `app` on `port`, programming the
|
|
344
|
+
OVS switches in `switches` (their service names)."""
|
|
345
|
+
name: str
|
|
346
|
+
app: str = DEFAULT_OF_APP
|
|
347
|
+
port: int = DEFAULT_OF_PORT
|
|
348
|
+
switches: list[str] = field(default_factory=list)
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
@dataclass
|
|
352
|
+
class ServiceSpec:
|
|
353
|
+
"""A managed cloud service backed by an off-the-shelf container image (MinIO,
|
|
354
|
+
Postgres, …). Runs on the shared bridge network, reachable by its service name.
|
|
355
|
+
`ports` is a list of {container, host, label, web} — host is a unique published
|
|
356
|
+
port so multiple consoles don't collide. `volumes` are compose volume strings;
|
|
357
|
+
`files` are generated config files (relative path -> content) written into the
|
|
358
|
+
project and bind-mounted — this is how the observability stack is auto-wired."""
|
|
359
|
+
name: str
|
|
360
|
+
type_key: str
|
|
361
|
+
image: str
|
|
362
|
+
summary: str
|
|
363
|
+
command: list[str] = field(default_factory=list)
|
|
364
|
+
env: dict[str, str] = field(default_factory=dict)
|
|
365
|
+
ports: list[dict] = field(default_factory=list)
|
|
366
|
+
volumes: list[str] = field(default_factory=list)
|
|
367
|
+
privileged: bool = False
|
|
368
|
+
files: dict[str, str] = field(default_factory=dict)
|
|
369
|
+
cpus: float = 0.0 # CPU limit from the size tier (0 = unset)
|
|
370
|
+
networks: list = field(default_factory=lambda: ["gini"]) # Docker networks to attach to
|
|
371
|
+
runtime: str = "" # OCI runtime override (e.g. "kata" for a Kata Instance)
|
|
372
|
+
|
|
373
|
+
|
|
374
|
+
@dataclass
|
|
375
|
+
class K8sSpec:
|
|
376
|
+
"""A real Kubernetes cluster (k3s in a container) + the workloads to deploy in it.
|
|
377
|
+
`deployments` = [{name,image,replicas,port,hpa:{min,max,cpu}|None}]; `manifests` is
|
|
378
|
+
the combined YAML applied via `kubectl apply` once the cluster is up."""
|
|
379
|
+
name: str
|
|
380
|
+
svc: str
|
|
381
|
+
image: str
|
|
382
|
+
deployments: list = field(default_factory=list)
|
|
383
|
+
manifests: str = ""
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
@dataclass
|
|
387
|
+
class FabricSpec:
|
|
388
|
+
"""The GINI Cloud Fabric agent: one container that polls each cloud service's native
|
|
389
|
+
metrics and serves them to gBuilder. `services` = [{name,type,host,port,creds}]."""
|
|
390
|
+
services: list = field(default_factory=list)
|
|
391
|
+
port: int = 9099
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
@dataclass
|
|
395
|
+
class NetworkSpec:
|
|
396
|
+
"""A VPC/subnet rendered as a real Docker bridge network. Containers on different VPC
|
|
397
|
+
networks can't reach each other; within one they can. A VPC's shared net is `internal`
|
|
398
|
+
(no internet of its own — the implicit VPC fabric); a per-VPC *egress* net is a normal
|
|
399
|
+
bridge that public-subnet members also join for real internet + host-published consoles.
|
|
400
|
+
`cidr` is the network's subnet (empty = let Docker auto-assign)."""
|
|
401
|
+
name: str # docker network name (the VPC's slug, or <slug>_egress)
|
|
402
|
+
cidr: str # e.g. 10.0.0.0/16, or "" for auto
|
|
403
|
+
label: str = "" # display name (for notes/UI)
|
|
404
|
+
region: str = ""
|
|
405
|
+
internal: bool = False # True = no external connectivity (the VPC fabric net)
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
# Fallback resolver when an Internet element carries no DNS of its own (an older saved
|
|
409
|
+
# topology, drawn before the property existed). Google's is used because it is the one
|
|
410
|
+
# public resolver reachable from essentially every network that has internet at all.
|
|
411
|
+
DEFAULT_PUBLIC_DNS = "8.8.8.8"
|
|
412
|
+
|
|
413
|
+
|
|
414
|
+
def _internet_dns(topo) -> str:
|
|
415
|
+
"""The resolver to hand out, taken from the Internet element on the canvas.
|
|
416
|
+
|
|
417
|
+
Blanking the property is a legitimate choice — "internet, but resolve names
|
|
418
|
+
yourself" — so an explicitly empty value is honoured rather than back-filled.
|
|
419
|
+
"""
|
|
420
|
+
for d in topo.devices.values():
|
|
421
|
+
if d.type_key != "cloud":
|
|
422
|
+
continue
|
|
423
|
+
props = d.properties or {}
|
|
424
|
+
if "DNS" not in props:
|
|
425
|
+
return DEFAULT_PUBLIC_DNS # saved before the property existed
|
|
426
|
+
raw = str(props.get("DNS", "")).strip()
|
|
427
|
+
return raw if _valid_ip(raw) else ""
|
|
428
|
+
return ""
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _valid_ip(text: str) -> bool:
|
|
432
|
+
parts = (text or "").split(".")
|
|
433
|
+
return (len(parts) == 4
|
|
434
|
+
and all(p.isdigit() and len(p) <= 3 and 0 <= int(p) <= 255 for p in parts))
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
@dataclass
|
|
438
|
+
class GBridgeSpec:
|
|
439
|
+
"""One drawn GINI32 element: a real board's end of a fabric link.
|
|
440
|
+
|
|
441
|
+
Unlike every other spec this does NOT become a container. The `gbridge` relay
|
|
442
|
+
owns `ep` (the fabric-side UDP endpoint) and forwards frames over the physical
|
|
443
|
+
LAN to whichever address the board checked in from. Everything here except
|
|
444
|
+
`board_id` is handed to the board in the relay's HELLO_ACK, so the canvas stays
|
|
445
|
+
the single source of truth for the board's fabric identity.
|
|
446
|
+
"""
|
|
447
|
+
name: str
|
|
448
|
+
board_id: str # must match the id flashed on the board
|
|
449
|
+
ip: str # fabric address assigned from its segment
|
|
450
|
+
mask: str
|
|
451
|
+
gw: str # its gateway (the router on that segment)
|
|
452
|
+
mac: str
|
|
453
|
+
ep: Endpoint
|
|
454
|
+
mode: str = "nat" # nat = devices hidden behind `ip`; routed = own subnet
|
|
455
|
+
physical_subnet: str = "" # the subnet behind the radio (always allocated)
|
|
456
|
+
mtu: int = 1400
|
|
457
|
+
seg: int = -1 # the segment it sits on (routed-mode route emission)
|
|
458
|
+
# The hotspot the board raises for real devices. Assigned here, not baked into
|
|
459
|
+
# firmware, so two boards never collide and a lab can be renamed without a reflash.
|
|
460
|
+
ap_ssid: str = ""
|
|
461
|
+
ap_pass: str = ""
|
|
462
|
+
# The resolver the board's DHCP server hands to real devices, or "" when the canvas
|
|
463
|
+
# has no Internet element. Empty is meaningful, not missing: with nothing to egress
|
|
464
|
+
# through, offering a resolver would promise name resolution that cannot work.
|
|
465
|
+
dns: str = ""
|
|
466
|
+
|
|
467
|
+
|
|
468
|
+
@dataclass
|
|
469
|
+
class RuntimeConfig:
|
|
470
|
+
machines: list[MachineSpec] = field(default_factory=list)
|
|
471
|
+
gbridge: list[GBridgeSpec] = field(default_factory=list) # real GINI32 boards
|
|
472
|
+
switches: list[SwitchSpec] = field(default_factory=list)
|
|
473
|
+
routers: list[RouterSpec] = field(default_factory=list)
|
|
474
|
+
ovs_switches: list[OvsSpec] = field(default_factory=list)
|
|
475
|
+
controllers: list[ControllerSpec] = field(default_factory=list)
|
|
476
|
+
services: list[ServiceSpec] = field(default_factory=list)
|
|
477
|
+
fabric: "FabricSpec | None" = None # cloud telemetry agent
|
|
478
|
+
k8s: list = field(default_factory=list) # real k3s clusters + manifests
|
|
479
|
+
faas: list = field(default_factory=list) # serverless functions (shared runtime)
|
|
480
|
+
networks: list = field(default_factory=list) # VPCs as isolated Docker networks
|
|
481
|
+
firewalls: list = field(default_factory=list) # security-group iptables per member
|
|
482
|
+
subnets: dict[int, str] = field(default_factory=dict) # seg -> cidr
|
|
483
|
+
notes: list[str] = field(default_factory=list)
|
|
484
|
+
|
|
485
|
+
# -- emit for the in-process simulator / Docker ------------------------- #
|
|
486
|
+
def to_runtime(self, docker: bool) -> dict:
|
|
487
|
+
return {
|
|
488
|
+
"machines": [
|
|
489
|
+
{"name": _svc(m.name), "hostname": _hostname(m.name), "gw": m.gw,
|
|
490
|
+
"gateway": m.gateway, "fabric_default": m.fabric_default,
|
|
491
|
+
"fabric_gw": m.fabric_gw, "cpus": m.cpus, "toolkit": m.toolkit,
|
|
492
|
+
"novnc_port": m.novnc_port,
|
|
493
|
+
"forward": m.forward, "nf": m.nf, "nf_rules": m.nf_rules,
|
|
494
|
+
"ifaces": [{"ip": i.ip, "mac": i.mac, "tap": f"gini{idx}",
|
|
495
|
+
"port": i.ep.wiring(docker)}
|
|
496
|
+
for idx, i in enumerate(m.ifaces)]}
|
|
497
|
+
for m in self.machines
|
|
498
|
+
],
|
|
499
|
+
"switches": [
|
|
500
|
+
{"name": _svc(s.name), "ports": [e.wiring(docker) for e in s.eps],
|
|
501
|
+
"hub": s.hub}
|
|
502
|
+
for s in self.switches
|
|
503
|
+
],
|
|
504
|
+
"routers": [
|
|
505
|
+
{"name": _svc(r.name),
|
|
506
|
+
"ifaces": [{"ip": i.ip, "mac": i.mac, "port": i.ep.wiring(docker)}
|
|
507
|
+
for i in r.ifaces],
|
|
508
|
+
"routes": r.routes}
|
|
509
|
+
for r in self.routers
|
|
510
|
+
],
|
|
511
|
+
# SDN: OVS switches run as their own gRouter --openflow containers; each
|
|
512
|
+
# data port is a cross-container UDP link, just like a router interface.
|
|
513
|
+
# Ports carry a link-local placeholder IP only so the gRouter can bring the
|
|
514
|
+
# tun up — in OpenFlow mode frames are switched by the flow table (shunted
|
|
515
|
+
# before the L3 stack), so the address is never used for forwarding.
|
|
516
|
+
"ovs": [
|
|
517
|
+
{"name": _svc(s.name), "openflow": True,
|
|
518
|
+
"controller": s.controller, "controller_port": s.controller_port,
|
|
519
|
+
"ports": [
|
|
520
|
+
{"ip": f"169.254.{si}.{pi + 1}/16",
|
|
521
|
+
"mac": f"02:00:fe:{si:02x}:00:{pi + 1:02x}",
|
|
522
|
+
"port": e.wiring(docker)}
|
|
523
|
+
for pi, e in enumerate(s.eps)]}
|
|
524
|
+
for si, s in enumerate(self.ovs_switches)
|
|
525
|
+
],
|
|
526
|
+
"controllers": [
|
|
527
|
+
{"name": _svc(c.name), "app": c.app, "port": c.port,
|
|
528
|
+
"switches": c.switches}
|
|
529
|
+
for c in self.controllers
|
|
530
|
+
],
|
|
531
|
+
# Managed cloud services — ordinary containers from public images on the
|
|
532
|
+
# shared bridge network, reachable by service name (cloud-style discovery).
|
|
533
|
+
"services": [
|
|
534
|
+
{"name": _svc(s.name), "type": s.type_key, "image": s.image,
|
|
535
|
+
"summary": s.summary, "command": s.command, "env": s.env,
|
|
536
|
+
"ports": s.ports, "volumes": s.volumes, "privileged": s.privileged,
|
|
537
|
+
"files": s.files, "cpus": s.cpus, "networks": s.networks,
|
|
538
|
+
"runtime": s.runtime}
|
|
539
|
+
for s in self.services
|
|
540
|
+
],
|
|
541
|
+
"fabric": ({"port": self.fabric.port, "services": self.fabric.services}
|
|
542
|
+
if self.fabric else None),
|
|
543
|
+
"k8s": [{"name": _svc(k.name), "image": k.image,
|
|
544
|
+
"deployments": k.deployments} for k in self.k8s],
|
|
545
|
+
"faas": self.faas, # serverless: [{name, handler, code}] -> one faas container
|
|
546
|
+
# VPC/subnet Docker networks (empty -> only the flat `gini` bridge).
|
|
547
|
+
"networks": [{"name": n.name, "cidr": n.cidr, "label": n.label,
|
|
548
|
+
"region": n.region, "internal": n.internal} for n in self.networks],
|
|
549
|
+
# security groups -> per-member iptables (a sidecar in each member's netns).
|
|
550
|
+
"firewalls": self.firewalls,
|
|
551
|
+
# Real GINI32 boards. One `gbridge` relay container serves all of them:
|
|
552
|
+
# `fabric` is its end of the link to the board's router, and the rest is
|
|
553
|
+
# the identity it hands the board when the board announces itself.
|
|
554
|
+
"gbridge": [
|
|
555
|
+
{"board_id": b.board_id, "name": _svc(b.name), "label": b.name,
|
|
556
|
+
"ip": b.ip, "mask": b.mask, "gw": b.gw, "mac": b.mac, "mtu": b.mtu,
|
|
557
|
+
"mode": b.mode, "physical_subnet": b.physical_subnet,
|
|
558
|
+
"ap_ssid": b.ap_ssid, "ap_pass": b.ap_pass, "dns": b.dns,
|
|
559
|
+
"fabric": b.ep.wiring(docker)}
|
|
560
|
+
for b in self.gbridge
|
|
561
|
+
],
|
|
562
|
+
}
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
# --- observability auto-wiring config (generated into the project, bind-mounted) --- #
|
|
566
|
+
_PROMETHEUS_YML = (
|
|
567
|
+
"global:\n"
|
|
568
|
+
" scrape_interval: 5s\n"
|
|
569
|
+
"scrape_configs:\n"
|
|
570
|
+
" - job_name: prometheus\n"
|
|
571
|
+
" static_configs:\n"
|
|
572
|
+
" - targets: ['localhost:9090']\n"
|
|
573
|
+
" - job_name: cadvisor\n"
|
|
574
|
+
" static_configs:\n"
|
|
575
|
+
" - targets: ['cadvisor:8080']\n"
|
|
576
|
+
)
|
|
577
|
+
_GRAFANA_DS = (
|
|
578
|
+
"apiVersion: 1\n"
|
|
579
|
+
"datasources:\n"
|
|
580
|
+
" - name: Prometheus\n"
|
|
581
|
+
" type: prometheus\n"
|
|
582
|
+
" uid: prometheus\n" # fixed uid so the dashboard panels bind reliably
|
|
583
|
+
" access: proxy\n"
|
|
584
|
+
" url: http://{prom}:9090\n"
|
|
585
|
+
" isDefault: true\n"
|
|
586
|
+
)
|
|
587
|
+
_GRAFANA_PROVIDER = (
|
|
588
|
+
"apiVersion: 1\n"
|
|
589
|
+
"providers:\n"
|
|
590
|
+
" - name: gini\n"
|
|
591
|
+
" type: file\n"
|
|
592
|
+
" options:\n"
|
|
593
|
+
" path: /var/lib/grafana/dashboards\n"
|
|
594
|
+
)
|
|
595
|
+
|
|
596
|
+
|
|
597
|
+
# Traefik dynamic (file-provider) config: route everything to the drawn backends.
|
|
598
|
+
_TRAEFIK_DYNAMIC = (
|
|
599
|
+
"http:\n"
|
|
600
|
+
" routers:\n"
|
|
601
|
+
" gini:\n"
|
|
602
|
+
" rule: \"PathPrefix(`/`)\"\n"
|
|
603
|
+
" entryPoints: [web]\n"
|
|
604
|
+
" service: gini\n"
|
|
605
|
+
" services:\n"
|
|
606
|
+
" gini:\n"
|
|
607
|
+
" loadBalancer:\n"
|
|
608
|
+
" servers:\n"
|
|
609
|
+
"{servers}\n"
|
|
610
|
+
)
|
|
611
|
+
|
|
612
|
+
# nginx load-balancer config: an upstream over the drawn backends (honoring the chosen
|
|
613
|
+
# algorithm) + a /nginx_status endpoint so the cloud fabric can read its request rate.
|
|
614
|
+
_NGINX_LB = (
|
|
615
|
+
"events {{}}\n"
|
|
616
|
+
"http {{\n"
|
|
617
|
+
" upstream gini_backend {{\n"
|
|
618
|
+
"{algo}"
|
|
619
|
+
"{servers}\n"
|
|
620
|
+
" }}\n"
|
|
621
|
+
" server {{\n"
|
|
622
|
+
" listen 80;\n"
|
|
623
|
+
" location / {{\n"
|
|
624
|
+
" proxy_pass http://gini_backend;\n"
|
|
625
|
+
" proxy_set_header Host $host;\n"
|
|
626
|
+
" }}\n"
|
|
627
|
+
" location /nginx_status {{ stub_status; }}\n"
|
|
628
|
+
" }}\n"
|
|
629
|
+
"}}\n"
|
|
630
|
+
)
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
def _grafana_dashboard_json() -> str:
|
|
634
|
+
"""A starter Grafana dashboard. It leads with Prometheus *pipeline-health* panels
|
|
635
|
+
(targets up, samples scraped) that ALWAYS have data — so the board is never blank and
|
|
636
|
+
doubles as a built-in diagnostic — then shows cAdvisor per-container CPU/mem/network
|
|
637
|
+
(best-effort; cAdvisor can be sparse on Docker Desktop). Built via json.dumps so the
|
|
638
|
+
PromQL (with quotes) is always valid JSON."""
|
|
639
|
+
import json
|
|
640
|
+
|
|
641
|
+
ds = {"type": "prometheus", "uid": "prometheus"} # bind panels to the datasource
|
|
642
|
+
|
|
643
|
+
def ts(pid, title, expr, legend, x, y, w=12, h=8):
|
|
644
|
+
return {"id": pid, "type": "timeseries", "title": title, "datasource": ds,
|
|
645
|
+
"gridPos": {"h": h, "w": w, "x": x, "y": y},
|
|
646
|
+
"fieldConfig": {"defaults": {}, "overrides": []},
|
|
647
|
+
"targets": [{"expr": expr, "legendFormat": legend, "refId": "A",
|
|
648
|
+
"datasource": ds}]}
|
|
649
|
+
|
|
650
|
+
def stat(pid, title, expr, legend, x, y, w=12, h=6):
|
|
651
|
+
return {"id": pid, "type": "stat", "title": title, "datasource": ds,
|
|
652
|
+
"gridPos": {"h": h, "w": w, "x": x, "y": y},
|
|
653
|
+
"options": {"colorMode": "background", "graphMode": "none",
|
|
654
|
+
"textMode": "value_and_name", "reduceOptions":
|
|
655
|
+
{"calcs": ["lastNotNull"]}},
|
|
656
|
+
"fieldConfig": {"defaults": {"mappings": [
|
|
657
|
+
{"type": "value", "options": {"0": {"text": "DOWN", "color": "red"},
|
|
658
|
+
"1": {"text": "UP", "color": "green"}}}],
|
|
659
|
+
"thresholds": {"steps": [{"color": "red", "value": None},
|
|
660
|
+
{"color": "green", "value": 1}]}},
|
|
661
|
+
"overrides": []},
|
|
662
|
+
"targets": [{"expr": expr, "legendFormat": legend, "refId": "A",
|
|
663
|
+
"datasource": ds}]}
|
|
664
|
+
|
|
665
|
+
return json.dumps({
|
|
666
|
+
"title": "GINI lab overview", "uid": "gini-containers",
|
|
667
|
+
"schemaVersion": 39, "version": 1, "refresh": "5s",
|
|
668
|
+
"time": {"from": "now-15m", "to": "now"},
|
|
669
|
+
"panels": [
|
|
670
|
+
# --- pipeline health: always populated (Prometheus knows its own targets) ---
|
|
671
|
+
stat(10, "Scrape targets up", "up", "{{job}}", 0, 0, w=8, h=6),
|
|
672
|
+
ts(11, "Samples scraped / target", "scrape_samples_scraped", "{{job}}",
|
|
673
|
+
8, 0, w=16, h=6),
|
|
674
|
+
# --- per-container resources (cAdvisor; best-effort on Docker Desktop) ---
|
|
675
|
+
ts(1, "CPU (cores) by container",
|
|
676
|
+
'rate(container_cpu_usage_seconds_total{name!=""}[1m])', "{{name}}", 0, 6),
|
|
677
|
+
ts(2, "Memory (bytes) by container",
|
|
678
|
+
'container_memory_usage_bytes{name!=""}', "{{name}}", 12, 6),
|
|
679
|
+
ts(3, "Network RX (bytes/s) by container",
|
|
680
|
+
'rate(container_network_receive_bytes_total{name!=""}[1m])',
|
|
681
|
+
"{{name}}", 0, 14, w=24),
|
|
682
|
+
],
|
|
683
|
+
}, indent=2)
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
class RuntimeCompiler:
|
|
687
|
+
def compile(self, topo: Topology) -> RuntimeConfig:
|
|
688
|
+
cfg = RuntimeConfig()
|
|
689
|
+
role = {d.id: _role(d.type_key) for d in topo.devices.values()}
|
|
690
|
+
name = {d.id: d.name for d in topo.devices.values()}
|
|
691
|
+
# the drawn "Internet" element (type_key "cloud") is the on-fabric NAT gateway:
|
|
692
|
+
# it sits on the fabric like a host but also has an external uplink + does NAT.
|
|
693
|
+
gw_dids = {d.id for d in topo.devices.values() if d.type_key == "cloud"}
|
|
694
|
+
|
|
695
|
+
# 1. keep links not touching grouping devices; pull out SDN control links
|
|
696
|
+
# (controller↔OVS) — they are a management association, not a data segment.
|
|
697
|
+
kept = []
|
|
698
|
+
control_links = []
|
|
699
|
+
ovs_controller: dict[str, str] = {} # ovs did -> controller did
|
|
700
|
+
ctrl_switches: dict[str, list[str]] = {} # controller did -> [ovs did]
|
|
701
|
+
for l in topo.links.values():
|
|
702
|
+
rs, rt = role.get(l.source_id), role.get(l.target_id)
|
|
703
|
+
if getattr(l, "kind", "link") == "attach" or rs == "rider" or rt == "rider":
|
|
704
|
+
# a Source/Sink mount: the rider runs inside the donor, so this is not a cable.
|
|
705
|
+
cfg.notes.append(f"skipped rider attach: "
|
|
706
|
+
f"{name.get(l.source_id)}–{name.get(l.target_id)}")
|
|
707
|
+
elif rs == "group" or rt == "group":
|
|
708
|
+
cfg.notes.append(f"skipped link touching grouping: "
|
|
709
|
+
f"{name.get(l.source_id)}–{name.get(l.target_id)}")
|
|
710
|
+
elif rs in K8S_ROLES.values() or rt in K8S_ROLES.values():
|
|
711
|
+
# kubernetes associations (cluster↔pod, pod↔autoscaler) are intent for
|
|
712
|
+
# manifest generation, not data-plane segments.
|
|
713
|
+
cfg.notes.append(f"k8s link: {name.get(l.source_id)}–{name.get(l.target_id)}")
|
|
714
|
+
elif rs in ("service", "compute") or rt in ("service", "compute"):
|
|
715
|
+
# cloud services + compute live on the bridge net and are reached by
|
|
716
|
+
# name, so a link to one is intent ("uses"), not a data-plane segment.
|
|
717
|
+
cfg.notes.append(f"service link (reach by name): "
|
|
718
|
+
f"{name.get(l.source_id)}–{name.get(l.target_id)}")
|
|
719
|
+
elif rs == "controller" or rt == "controller":
|
|
720
|
+
control_links.append(l)
|
|
721
|
+
ctrl = l.source_id if rs == "controller" else l.target_id
|
|
722
|
+
other = l.target_id if rs == "controller" else l.source_id
|
|
723
|
+
if role.get(other) == "ovs": # only OVS↔controller is a real assoc
|
|
724
|
+
ovs_controller[other] = ctrl
|
|
725
|
+
ctrl_switches.setdefault(ctrl, []).append(other)
|
|
726
|
+
else:
|
|
727
|
+
cfg.notes.append(f"controller {name.get(ctrl)} should attach to an "
|
|
728
|
+
f"OVS, not {name.get(other)}")
|
|
729
|
+
else:
|
|
730
|
+
kept.append(l)
|
|
731
|
+
|
|
732
|
+
# 2. (machines may be multi-homed: a machine keeps ALL its links and gets one
|
|
733
|
+
# interface/IP per segment — see the machine build + shuttle.)
|
|
734
|
+
|
|
735
|
+
# 3. segments: union links that share an L2 (switch) device
|
|
736
|
+
uf = _UF()
|
|
737
|
+
for l in kept:
|
|
738
|
+
uf.find(l.id)
|
|
739
|
+
by_device: dict[str, list] = {}
|
|
740
|
+
for l in kept:
|
|
741
|
+
by_device.setdefault(l.source_id, []).append(l)
|
|
742
|
+
by_device.setdefault(l.target_id, []).append(l)
|
|
743
|
+
for did, links in by_device.items():
|
|
744
|
+
if role[did] in ("switch", "ovs"): # OVS is an L2 domain too
|
|
745
|
+
for l in links[1:]:
|
|
746
|
+
uf.union(links[0].id, l.id)
|
|
747
|
+
seg_of_link = {l.id: uf.find(l.id) for l in kept}
|
|
748
|
+
seg_ids = {}
|
|
749
|
+
for root in dict.fromkeys(seg_of_link.values()):
|
|
750
|
+
seg_ids[root] = len(seg_ids)
|
|
751
|
+
for root, i in seg_ids.items():
|
|
752
|
+
cfg.subnets[i] = f"10.0.{i + 1}.0/24"
|
|
753
|
+
|
|
754
|
+
# 4. endpoints per kept link
|
|
755
|
+
eps: dict[tuple, Endpoint] = {}
|
|
756
|
+
|
|
757
|
+
def endpoint(device_id: str) -> Endpoint:
|
|
758
|
+
# Switches live inside the shared `fabric` container; routers each run as
|
|
759
|
+
# their own `gini-grouter` container (the real C gRouter), so a router's
|
|
760
|
+
# location is its own service name. Machines are their own containers too.
|
|
761
|
+
# This makes every router link a symmetric cross-container UDP link.
|
|
762
|
+
#
|
|
763
|
+
# A GINI32 board is the one endpoint that is NOT a container: it is real
|
|
764
|
+
# hardware on the physical LAN. Its location is the shared `gbridge` relay,
|
|
765
|
+
# which owns this end of the link and forwards to the board over the LAN.
|
|
766
|
+
# The gRouter on the far side therefore needs no notion of hardware at all.
|
|
767
|
+
r = role[device_id]
|
|
768
|
+
loc = ("fabric" if r == "switch"
|
|
769
|
+
else GBRIDGE_SVC if r == "gini32"
|
|
770
|
+
else _svc(name[device_id]))
|
|
771
|
+
return Endpoint(device=name[device_id], location=loc)
|
|
772
|
+
|
|
773
|
+
port = 5000
|
|
774
|
+
link_eps: dict[str, tuple[Endpoint, Endpoint]] = {}
|
|
775
|
+
for l in kept:
|
|
776
|
+
a, b = endpoint(l.source_id), endpoint(l.target_id)
|
|
777
|
+
a.bind_port = port; port += 1
|
|
778
|
+
b.bind_port = port; port += 1
|
|
779
|
+
a.peer, b.peer = b, a
|
|
780
|
+
link_eps[l.id] = (a, b)
|
|
781
|
+
eps[(l.id, l.source_id)] = a
|
|
782
|
+
eps[(l.id, l.target_id)] = b
|
|
783
|
+
|
|
784
|
+
# 5. IP assignment per segment
|
|
785
|
+
seg_hosts: dict[int, int] = {} # next machine host octet
|
|
786
|
+
seg_rtr: dict[int, int] = {} # next router host octet
|
|
787
|
+
seg_gateway: dict[int, str] = {}
|
|
788
|
+
# interfaces on each segment: (device_id, link)
|
|
789
|
+
iface_ip: dict[tuple, str] = {}
|
|
790
|
+
# routers (and inline VNFs) first (so .1 is the gateway), then machines. A VNF is a
|
|
791
|
+
# forwarding node in the path, so it's addressed like a router and becomes the
|
|
792
|
+
# gateway on its point-to-point segments — neighbours then route THROUGH it.
|
|
793
|
+
ordered = sorted(kept, key=lambda l: 0)
|
|
794
|
+
# A GINI32 board is addressed exactly like a host on its segment (it presents one
|
|
795
|
+
# fabric address, behind which its real devices are NATed), so it shares the
|
|
796
|
+
# machine numbering pass.
|
|
797
|
+
for did_role in ("router", "vnf", "machine", "gini32"):
|
|
798
|
+
for l in kept:
|
|
799
|
+
seg = seg_ids[seg_of_link[l.id]]
|
|
800
|
+
base = f"10.0.{seg + 1}."
|
|
801
|
+
for end in (l.source_id, l.target_id):
|
|
802
|
+
if role[end] != did_role:
|
|
803
|
+
continue
|
|
804
|
+
key = (l.id, end)
|
|
805
|
+
if key in iface_ip:
|
|
806
|
+
continue
|
|
807
|
+
if did_role in ("router", "vnf"):
|
|
808
|
+
n = seg_rtr.get(seg, 0) + 1
|
|
809
|
+
seg_rtr[seg] = n
|
|
810
|
+
ip = base + str(n) # .1, .2 ...
|
|
811
|
+
seg_gateway.setdefault(seg, ip)
|
|
812
|
+
else:
|
|
813
|
+
n = seg_hosts.get(seg, 9) + 1
|
|
814
|
+
seg_hosts[seg] = n
|
|
815
|
+
ip = base + str(n) # .10, .11 ...
|
|
816
|
+
iface_ip[key] = ip
|
|
817
|
+
|
|
818
|
+
# 5b. manual addressing: honor each device's typed static IPs, auto-fill the
|
|
819
|
+
# rest. MACs stay auto. Default gateways then follow the (possibly
|
|
820
|
+
# overridden) router address on each segment.
|
|
821
|
+
if getattr(topo, "manual_addressing", False):
|
|
822
|
+
for d in topo.devices.values():
|
|
823
|
+
for lid, sip in (getattr(d, "static_ips", None) or {}).items():
|
|
824
|
+
key = (lid, d.id)
|
|
825
|
+
bare = (sip or "").strip().split("/")[0]
|
|
826
|
+
if key in iface_ip and _is_ipv4(bare):
|
|
827
|
+
iface_ip[key] = bare
|
|
828
|
+
seg_gateway = {}
|
|
829
|
+
for l in kept:
|
|
830
|
+
seg = seg_ids[seg_of_link[l.id]]
|
|
831
|
+
for end in (l.source_id, l.target_id):
|
|
832
|
+
if role[end] in ("router", "vnf") and (l.id, end) in iface_ip:
|
|
833
|
+
seg_gateway.setdefault(seg, iface_ip[(l.id, end)])
|
|
834
|
+
|
|
835
|
+
# the Internet node's IP on each segment it touches; if a segment has an
|
|
836
|
+
# Internet node but no router, hosts there default straight to the Internet node.
|
|
837
|
+
seg_internet_ip: dict[int, str] = {}
|
|
838
|
+
for l in kept:
|
|
839
|
+
seg = seg_ids[seg_of_link[l.id]]
|
|
840
|
+
for end in (l.source_id, l.target_id):
|
|
841
|
+
if end in gw_dids and (l.id, end) in iface_ip:
|
|
842
|
+
seg_internet_ip.setdefault(seg, iface_ip[(l.id, end)])
|
|
843
|
+
for seg, ip in seg_internet_ip.items():
|
|
844
|
+
seg_gateway.setdefault(seg, ip)
|
|
845
|
+
# the segment + IP that everything default-routes toward (first Internet node)
|
|
846
|
+
gw_seg, gw_ip = next(iter(seg_internet_ip.items()), (None, None))
|
|
847
|
+
|
|
848
|
+
def mac(seg: int, kind: int, idx: int) -> str:
|
|
849
|
+
return f"02:00:00:{seg + 1:02x}:{kind:02x}:{idx:02x}"
|
|
850
|
+
|
|
851
|
+
# 6. build specs
|
|
852
|
+
# machines
|
|
853
|
+
midx = 0
|
|
854
|
+
m_ifaces: dict[str, list] = {} # machine did -> [IfaceSpec] (one per segment)
|
|
855
|
+
m_gw: dict[str, str] = {} # machine did -> default gateway
|
|
856
|
+
m_order: list[str] = [] # preserve first-seen order
|
|
857
|
+
for l in kept:
|
|
858
|
+
seg = seg_ids[seg_of_link[l.id]]
|
|
859
|
+
for end in (l.source_id, l.target_id):
|
|
860
|
+
if role[end] != "machine":
|
|
861
|
+
continue
|
|
862
|
+
key = (l.id, end)
|
|
863
|
+
if key not in iface_ip:
|
|
864
|
+
continue
|
|
865
|
+
midx += 1
|
|
866
|
+
if end not in m_ifaces:
|
|
867
|
+
m_ifaces[end] = []
|
|
868
|
+
m_order.append(end)
|
|
869
|
+
m_ifaces[end].append(IfaceSpec(ip=iface_ip[key] + "/24",
|
|
870
|
+
mac=mac(seg, 2, midx), ep=eps[key],
|
|
871
|
+
link_id=l.id))
|
|
872
|
+
gw = seg_gateway.get(seg) # default route via the first router seen
|
|
873
|
+
if gw and end not in m_gw:
|
|
874
|
+
m_gw[end] = gw
|
|
875
|
+
have_internet = bool(gw_dids)
|
|
876
|
+
for did in m_order:
|
|
877
|
+
if did in gw_dids:
|
|
878
|
+
# the Internet element: NAT gateway. It defaults OUT its uplink (set up
|
|
879
|
+
# at runtime), and routes the experiment supernet back via its local
|
|
880
|
+
# router so replies reach the hosts behind it.
|
|
881
|
+
cfg.machines.append(MachineSpec(
|
|
882
|
+
name=name[did], ifaces=m_ifaces[did], gw=None,
|
|
883
|
+
gateway=True, fabric_gw=m_gw.get(did)))
|
|
884
|
+
else:
|
|
885
|
+
# an ordinary host: when an Internet element is on the canvas, its
|
|
886
|
+
# default route goes INTO the fabric so internet egresses through the
|
|
887
|
+
# drawn routers (traceroute then shows the real path).
|
|
888
|
+
cfg.machines.append(MachineSpec(
|
|
889
|
+
name=name[did], ifaces=m_ifaces[did], gw=m_gw.get(did),
|
|
890
|
+
fabric_default=have_internet and bool(m_gw.get(did)),
|
|
891
|
+
cpus=_cpus_for(topo.devices[did]), # size tier -> CPU limit
|
|
892
|
+
toolkit=_toolkit_for(topo.devices[did]))) # lean (default) | full
|
|
893
|
+
|
|
894
|
+
# GINI32 boards: real hardware on the fabric. No container is emitted — the shared
|
|
895
|
+
# `gbridge` relay holds this end of the link and carries frames to the board over
|
|
896
|
+
# the physical LAN. We only need to hand it the identity the canvas assigned.
|
|
897
|
+
for l in kept:
|
|
898
|
+
seg = seg_ids[seg_of_link[l.id]]
|
|
899
|
+
for end in (l.source_id, l.target_id):
|
|
900
|
+
if role[end] != "gini32":
|
|
901
|
+
continue
|
|
902
|
+
key = (l.id, end)
|
|
903
|
+
if key not in iface_ip:
|
|
904
|
+
continue
|
|
905
|
+
midx += 1
|
|
906
|
+
dev = topo.devices[end]
|
|
907
|
+
props = getattr(dev, "properties", None) or {}
|
|
908
|
+
# Blank BoardID falls back to the element's own name. That id will not
|
|
909
|
+
# match any real board — but it is UNIQUE per element, which is the
|
|
910
|
+
# point: the earlier shared default made a second board vanish from the
|
|
911
|
+
# relay's table silently. Emitting nothing instead would be worse still,
|
|
912
|
+
# because the relay is what collects announcing boards for the Inspector
|
|
913
|
+
# to offer: no entry means no relay means an empty picker and no way to
|
|
914
|
+
# fix the very problem. So the element compiles, validate() flags it on
|
|
915
|
+
# the canvas, and the Inspector names the boards that ARE on the air.
|
|
916
|
+
board_id = str(props.get("BoardID", "")).strip() or _svc(name[end])
|
|
917
|
+
mode = str(props.get("Mode", "routed")).strip().lower()
|
|
918
|
+
if mode not in ("nat", "routed"):
|
|
919
|
+
mode = "routed"
|
|
920
|
+
|
|
921
|
+
# The subnet behind this board's radio. Blank means "allocate me one":
|
|
922
|
+
# every board needs a DISTINCT one, or routers end up with two routes
|
|
923
|
+
# to the same network via different next hops and neither works. We
|
|
924
|
+
# skip any third octet the topology's own segments already use.
|
|
925
|
+
phys = str(props.get("PhysicalSubnet", "")).strip()
|
|
926
|
+
if not phys:
|
|
927
|
+
used = {int(c.split(".")[2]) for c in cfg.subnets.values()}
|
|
928
|
+
used |= {int(b.physical_subnet.split(".")[2])
|
|
929
|
+
for b in cfg.gbridge if b.physical_subnet}
|
|
930
|
+
oct3 = 9
|
|
931
|
+
while oct3 in used:
|
|
932
|
+
oct3 += 1
|
|
933
|
+
phys = f"10.0.{oct3}.0/24"
|
|
934
|
+
elif not _valid_cidr(phys):
|
|
935
|
+
cfg.notes.append(
|
|
936
|
+
f"{name[end]}: PhysicalSubnet {phys!r} is not a valid CIDR — "
|
|
937
|
+
f"allocating one automatically")
|
|
938
|
+
phys = ""
|
|
939
|
+
used = {int(c.split(".")[2]) for c in cfg.subnets.values()}
|
|
940
|
+
used |= {int(b.physical_subnet.split(".")[2])
|
|
941
|
+
for b in cfg.gbridge if b.physical_subnet}
|
|
942
|
+
oct3 = 9
|
|
943
|
+
while oct3 in used:
|
|
944
|
+
oct3 += 1
|
|
945
|
+
phys = f"10.0.{oct3}.0/24"
|
|
946
|
+
|
|
947
|
+
# The hotspot real devices join. Named after the element unless the
|
|
948
|
+
# user overrode it, so two boards are distinguishable on a phone.
|
|
949
|
+
ap_ssid = str(props.get("ApSSID", "")).strip()
|
|
950
|
+
if not ap_ssid:
|
|
951
|
+
ap_ssid = f"GINI32-{re.sub(r'[^A-Za-z0-9-]', '', name[end]) or 'board'}"
|
|
952
|
+
ap_pass = str(props.get("ApPassword", "")).strip()
|
|
953
|
+
# Real devices on the board's radio get a resolver ONLY when the canvas
|
|
954
|
+
# has an Internet element to egress through. Without one, an iPad could
|
|
955
|
+
# still be handed 8.8.8.8, would send queries into a topology with no way
|
|
956
|
+
# out, and would sit there timing out — which looks like broken Wi-Fi
|
|
957
|
+
# rather than a network with deliberately no internet in it. This is also
|
|
958
|
+
# why DNS follows the same route as everything else the board is told:
|
|
959
|
+
# the canvas decides, the board obeys.
|
|
960
|
+
dns = _internet_dns(topo) if have_internet else ""
|
|
961
|
+
cfg.gbridge.append(GBridgeSpec(
|
|
962
|
+
name=name[end], board_id=board_id,
|
|
963
|
+
ip=iface_ip[key], mask="255.255.255.0",
|
|
964
|
+
gw=seg_gateway.get(seg, ""), mac=mac(seg, 4, midx),
|
|
965
|
+
ep=eps[key], mode=mode,
|
|
966
|
+
# The board always serves this subnet; `mode` only decides whether
|
|
967
|
+
# the emulated side gets a ROUTE to it or the devices are hidden.
|
|
968
|
+
physical_subnet=phys, seg=seg,
|
|
969
|
+
ap_ssid=ap_ssid, ap_pass=ap_pass, dns=dns))
|
|
970
|
+
|
|
971
|
+
# inline VNFs: a forwarding container that applies a network function between its
|
|
972
|
+
# segments (the Internet-element pattern, but fabric<->fabric + an NF instead of NAT).
|
|
973
|
+
# Addressed like a router above, so it's the gateway on its point-to-point segments;
|
|
974
|
+
# `gw` is a next hop toward egress (a router/gateway on its OTHER segment).
|
|
975
|
+
v_ifaces: dict[str, list] = {}
|
|
976
|
+
v_gw: dict[str, str] = {}
|
|
977
|
+
v_order: list[str] = []
|
|
978
|
+
for l in kept:
|
|
979
|
+
seg = seg_ids[seg_of_link[l.id]]
|
|
980
|
+
for end in (l.source_id, l.target_id):
|
|
981
|
+
if role[end] != "vnf":
|
|
982
|
+
continue
|
|
983
|
+
key = (l.id, end)
|
|
984
|
+
if key not in iface_ip:
|
|
985
|
+
continue
|
|
986
|
+
midx += 1
|
|
987
|
+
my_ip = iface_ip[key]
|
|
988
|
+
if end not in v_ifaces:
|
|
989
|
+
v_ifaces[end] = []
|
|
990
|
+
v_order.append(end)
|
|
991
|
+
v_ifaces[end].append(IfaceSpec(ip=my_ip + "/24", mac=mac(seg, 3, midx),
|
|
992
|
+
ep=eps[key], link_id=l.id))
|
|
993
|
+
g = seg_gateway.get(seg)
|
|
994
|
+
if g and g != my_ip and end not in v_gw: # onward route toward egress
|
|
995
|
+
v_gw[end] = g
|
|
996
|
+
vprops = {d.id: getattr(d, "properties", {}) or {} for d in topo.devices.values()}
|
|
997
|
+
for did in v_order:
|
|
998
|
+
p = vprops[did]
|
|
999
|
+
cfg.machines.append(MachineSpec(
|
|
1000
|
+
name=name[did], ifaces=v_ifaces[did], gw=v_gw.get(did),
|
|
1001
|
+
forward=True, nf=(p.get("Kind") or "firewall"),
|
|
1002
|
+
nf_rules=(p.get("Rules") or ""), cpus=_cpus_for(topo.devices[did])))
|
|
1003
|
+
|
|
1004
|
+
# switches (a Hub is the same fabric node in flood-all mode — no MAC learning)
|
|
1005
|
+
for did, r in role.items():
|
|
1006
|
+
if r != "switch":
|
|
1007
|
+
continue
|
|
1008
|
+
ports = [eps[(l.id, did)] for l in by_device.get(did, []) if l in kept]
|
|
1009
|
+
if ports:
|
|
1010
|
+
is_hub = topo.devices[did].type_key == "hub"
|
|
1011
|
+
cfg.switches.append(SwitchSpec(name=name[did], eps=ports, hub=is_hub))
|
|
1012
|
+
|
|
1013
|
+
# OVS switches — own gRouter --openflow container, programmed by a controller
|
|
1014
|
+
props = {d.id: getattr(d, "properties", {}) or {} for d in topo.devices.values()}
|
|
1015
|
+
|
|
1016
|
+
# VPCs -> isolated Docker networks. Each VPC element with members becomes its own
|
|
1017
|
+
# bridge (unique subnet); every element inside it (via parent_id) attaches to that
|
|
1018
|
+
# network instead of the flat `gini` bridge, so different VPCs can't reach each
|
|
1019
|
+
# other. Elements with no VPC ancestor stay on `gini` (unchanged flat behavior).
|
|
1020
|
+
vpc_net_of = self._build_networks(cfg, topo, name)
|
|
1021
|
+
for did, r in role.items():
|
|
1022
|
+
if r != "ovs":
|
|
1023
|
+
continue
|
|
1024
|
+
ports = [eps[(l.id, did)] for l in by_device.get(did, []) if l in kept]
|
|
1025
|
+
ctrl_did = ovs_controller.get(did)
|
|
1026
|
+
ctrl_name = _svc(name[ctrl_did]) if ctrl_did else None
|
|
1027
|
+
ctrl_port = DEFAULT_OF_PORT
|
|
1028
|
+
if ctrl_did:
|
|
1029
|
+
ctrl_port = int(props[ctrl_did].get("Port") or DEFAULT_OF_PORT)
|
|
1030
|
+
cfg.ovs_switches.append(OvsSpec(name=name[did], eps=ports,
|
|
1031
|
+
controller=ctrl_name,
|
|
1032
|
+
controller_port=ctrl_port))
|
|
1033
|
+
|
|
1034
|
+
# controllers — POX containers; each programs the OVS switches linked to it
|
|
1035
|
+
for did, r in role.items():
|
|
1036
|
+
if r != "controller":
|
|
1037
|
+
continue
|
|
1038
|
+
p = props[did]
|
|
1039
|
+
cfg.controllers.append(ControllerSpec(
|
|
1040
|
+
name=name[did],
|
|
1041
|
+
app=p.get("App") or DEFAULT_OF_APP,
|
|
1042
|
+
port=int(p.get("Port") or DEFAULT_OF_PORT),
|
|
1043
|
+
switches=[_svc(name[o]) for o in ctrl_switches.get(did, [])]))
|
|
1044
|
+
|
|
1045
|
+
# managed cloud services — each backed by an off-the-shelf image. Web consoles
|
|
1046
|
+
# get a unique published host port so several services can coexist.
|
|
1047
|
+
host_port = 38000
|
|
1048
|
+
# headful ("gui") machines publish their noVNC console on a unique host port too, so the
|
|
1049
|
+
# Desktop element can open the embedded screen (the machine dict carries the port).
|
|
1050
|
+
for m in cfg.machines:
|
|
1051
|
+
if m.toolkit == "gui":
|
|
1052
|
+
m.novnc_port = host_port
|
|
1053
|
+
host_port += 1
|
|
1054
|
+
for d in topo.devices.values():
|
|
1055
|
+
if role.get(d.id) != "service":
|
|
1056
|
+
continue
|
|
1057
|
+
svc = service_for(d.type_key)
|
|
1058
|
+
if svc is None:
|
|
1059
|
+
continue
|
|
1060
|
+
ports = []
|
|
1061
|
+
for p in svc.ports:
|
|
1062
|
+
ports.append({"container": p.container, "host": host_port,
|
|
1063
|
+
"label": p.label, "web": p.web, "path": p.path})
|
|
1064
|
+
host_port += 1
|
|
1065
|
+
# some images need to advertise their own service name (e.g. Redpanda's
|
|
1066
|
+
# kafka address); `{svc}` in the catalog command/env is filled in here.
|
|
1067
|
+
sname = _svc(d.name)
|
|
1068
|
+
command = [a.replace("{svc}", sname) for a in svc.command]
|
|
1069
|
+
env = {k: v.replace("{svc}", sname) for k, v in svc.env.items()}
|
|
1070
|
+
cfg.services.append(ServiceSpec(
|
|
1071
|
+
name=d.name, type_key=d.type_key, image=svc.image,
|
|
1072
|
+
summary=svc.summary, command=command, env=env, ports=ports,
|
|
1073
|
+
cpus=_cpus_for(d), # size tier -> CPU limit
|
|
1074
|
+
networks=vpc_net_of.get(d.id, ["gini"]))) # VPC/subnet nets (or flat bridge)
|
|
1075
|
+
|
|
1076
|
+
# cloud compute (instance / container) — a plain bridge container the student can
|
|
1077
|
+
# log into and run an app on, reaching services by name. Image from the element's
|
|
1078
|
+
# property; keep it alive (base OS images would exit) unless a Command is given.
|
|
1079
|
+
import shlex
|
|
1080
|
+
for d in topo.devices.values():
|
|
1081
|
+
if role.get(d.id) != "compute":
|
|
1082
|
+
continue
|
|
1083
|
+
p = props[d.id]
|
|
1084
|
+
if d.type_key == "container":
|
|
1085
|
+
image = _norm_image(p.get("Image") or "alpine:latest")
|
|
1086
|
+
summary = f"Container ({image})."
|
|
1087
|
+
elif d.type_key == "kinstance": # VM-isolated workload via Kata
|
|
1088
|
+
image = _norm_image(p.get("Image") or "ubuntu:22.04")
|
|
1089
|
+
summary = f"Kata Instance — VM-isolated ({image})."
|
|
1090
|
+
else:
|
|
1091
|
+
image = _norm_image(p.get("Image") or "ubuntu:22.04")
|
|
1092
|
+
summary = f"Compute instance ({image}, {p.get('Type', 'vm')})."
|
|
1093
|
+
cmd = p.get("Command") or ""
|
|
1094
|
+
command = shlex.split(cmd) if cmd.strip() else ["tail", "-f", "/dev/null"]
|
|
1095
|
+
is_kata = d.type_key == "kinstance"
|
|
1096
|
+
cfg.services.append(ServiceSpec(
|
|
1097
|
+
name=d.name, type_key=d.type_key, image=image, summary=summary,
|
|
1098
|
+
command=command, env={}, ports=[], cpus=_cpus_for(d), # size -> CPU
|
|
1099
|
+
# Kata Instances stay flat (no VPC) and run under the kata OCI runtime.
|
|
1100
|
+
networks=["gini"] if is_kata else vpc_net_of.get(d.id, ["gini"]),
|
|
1101
|
+
runtime="kata" if is_kata else ""))
|
|
1102
|
+
|
|
1103
|
+
# xv6 teaching kernel — a standalone QEMU-RISC-V machine that boots a real kernel and
|
|
1104
|
+
# exposes its GDB stub (port 1234) so the Machine Lab's bridge can read and steer it.
|
|
1105
|
+
# No fabric wiring (xv6 is standalone); the time-slice tier seeds the kernel quantum.
|
|
1106
|
+
for d in topo.devices.values():
|
|
1107
|
+
if role.get(d.id) != "xv6":
|
|
1108
|
+
continue
|
|
1109
|
+
p = props[d.id]
|
|
1110
|
+
# the Load loop: bind-mount a host folder over kernel/shadows/ so the student edits
|
|
1111
|
+
# gini_sched.c in their own editor and Load rebuilds in-container. The folder can start
|
|
1112
|
+
# empty — the agent seeds the shipped stub into it on boot (see gini_agent.py). Mount a
|
|
1113
|
+
# DIRECTORY (editors save via rename, which breaks a single-file mount). It lives under
|
|
1114
|
+
# the GINI home (~/.gini/xv6-shadows/<name>/) so it's stable + discoverable and the
|
|
1115
|
+
# student's edits PERSIST across Stop/Run (unlike the ephemeral compose workdir).
|
|
1116
|
+
_sane = "".join(c if (c.isalnum() or c in "_.-") else "-" for c in d.name)
|
|
1117
|
+
_shadows_host = _gini_home() / "xv6-shadows" / _sane
|
|
1118
|
+
try:
|
|
1119
|
+
_shadows_host.mkdir(parents=True, exist_ok=True) # exists + user-owned before `up`
|
|
1120
|
+
except OSError:
|
|
1121
|
+
pass
|
|
1122
|
+
cfg.services.append(ServiceSpec(
|
|
1123
|
+
name=d.name, type_key="xv6", image="gini-xv6:latest",
|
|
1124
|
+
summary="xv6 teaching kernel (QEMU-RISC-V); in-container agent serves live state.",
|
|
1125
|
+
command=[], env={"XV6_QUANTUM": str(p.get("Timeslice", "1")),
|
|
1126
|
+
"XV6_CPUS": str(_xv6_harts(d))}, # size stepper -> real harts
|
|
1127
|
+
# agent HTTP (the Machine Lab bridge talks here) + serial (the human console).
|
|
1128
|
+
ports=[{"container": 5000, "host": host_port, "label": "agent",
|
|
1129
|
+
"web": False, "path": ""},
|
|
1130
|
+
{"container": 4444, "host": host_port + 1, "label": "serial",
|
|
1131
|
+
"web": False, "path": ""}],
|
|
1132
|
+
volumes=[f"{_shadows_host}:/opt/xv6-riscv/kernel/shadows"],
|
|
1133
|
+
cpus=_cpus_for(d), networks=["gini"]))
|
|
1134
|
+
host_port += 2
|
|
1135
|
+
|
|
1136
|
+
# OS Zoo — a real historical OS under emulation. One container per element, image
|
|
1137
|
+
# `gini-oszoo:latest`, with ZOO_OS selecting the guest; the container runs the emulator
|
|
1138
|
+
# (`-vnc :0`) + websockify/noVNC and publishes the framebuffer as a web page. The Zoo Lab
|
|
1139
|
+
# embeds that URL in a QWebEngineView. Standalone (no fabric wiring in v1).
|
|
1140
|
+
for d in topo.devices.values():
|
|
1141
|
+
if role.get(d.id) != "oszoo":
|
|
1142
|
+
continue
|
|
1143
|
+
p = props[d.id]
|
|
1144
|
+
is_byo = d.type_key in OSZOO_BYO_KEYS
|
|
1145
|
+
os_id = "byo" if is_byo else d.type_key
|
|
1146
|
+
env = {"ZOO_OS": os_id,
|
|
1147
|
+
"ZOO_PERSIST": "1" if str(p.get("Persist", "false")).lower() == "true" else "0"}
|
|
1148
|
+
if is_byo: # BYO / preset: Emulator + Image (+Rom)
|
|
1149
|
+
env["ZOO_EMULATOR"] = str(p.get("Emulator", "qemu"))
|
|
1150
|
+
env["ZOO_ARCH"] = str(p.get("Arch", "x86"))
|
|
1151
|
+
# Image/Rom may be a local path (bind-mounted below) OR an http(s):// URL that the
|
|
1152
|
+
# container downloads on first boot. Pass the raw value; boot_zoo.sh decides.
|
|
1153
|
+
if str(p.get("Image", "")):
|
|
1154
|
+
env["ZOO_IMAGE"] = str(p.get("Image", ""))
|
|
1155
|
+
if str(p.get("Rom", "")): # Basilisk II: a Mac ROM
|
|
1156
|
+
env["ZOO_ROM"] = str(p.get("Rom", ""))
|
|
1157
|
+
if d.type_key == "win31": # ship Digger Remastered on the Win 3.11 C:
|
|
1158
|
+
env["ZOO_ADDONS"] = "digger" # (run it from the DOS prompt: cd\digger, digger)
|
|
1159
|
+
# persist downloaded guest images on the host so each OS is fetched once, not on
|
|
1160
|
+
# every Run (an anonymous /zoo/cache volume is discarded when the container recreates).
|
|
1161
|
+
# Computed inline (mirrors app.paths.oszoo_cache_dir) to keep the compiler Qt-free.
|
|
1162
|
+
import os
|
|
1163
|
+
from pathlib import Path
|
|
1164
|
+
_home = Path(os.environ.get("GINI_HOME_DIR") or (Path.home() / ".gini")).expanduser()
|
|
1165
|
+
cache = _home / "oszoo-cache"; cache.mkdir(parents=True, exist_ok=True)
|
|
1166
|
+
volumes = [f"{cache}:/zoo/cache"]
|
|
1167
|
+
def _is_url(s: str) -> bool:
|
|
1168
|
+
return s.startswith("http://") or s.startswith("https://")
|
|
1169
|
+
img = str(p.get("Image", "")) if is_byo else ""
|
|
1170
|
+
if img and not _is_url(img): # local path -> bind-mount read-only
|
|
1171
|
+
volumes.append(f"{img}:/zoo/byo.img:ro") # (a URL is downloaded in the container)
|
|
1172
|
+
rom = str(p.get("Rom", "")) if is_byo else ""
|
|
1173
|
+
if rom and not _is_url(rom): # Basilisk II Mac ROM, read-only
|
|
1174
|
+
volumes.append(f"{rom}:/zoo/rom:ro")
|
|
1175
|
+
# Basilisk II creates its 60 Hz timer as a real-time-scheduled thread; Docker's default
|
|
1176
|
+
# sandbox (seccomp + no CAP_SYS_NICE) forbids RT scheduling, so that container needs to
|
|
1177
|
+
# be privileged. QEMU/DOSBox guests don't, so keep them unprivileged.
|
|
1178
|
+
needs_priv = is_byo and str(p.get("Emulator", "")) == "basilisk"
|
|
1179
|
+
cfg.services.append(ServiceSpec(
|
|
1180
|
+
name=d.name, type_key=d.type_key, image="gini-oszoo:latest",
|
|
1181
|
+
summary=f"OS Zoo: {d.type_key} under emulation, screen embedded over noVNC.",
|
|
1182
|
+
command=[], env=env, privileged=needs_priv,
|
|
1183
|
+
# the noVNC web console (the Zoo Lab embeds this) + the raw VNC port.
|
|
1184
|
+
ports=[{"container": 6080, "host": host_port, "label": "screen",
|
|
1185
|
+
"web": True, "path": "/vnc.html?autoconnect=1&resize=remote"},
|
|
1186
|
+
{"container": 5900, "host": host_port + 1, "label": "vnc",
|
|
1187
|
+
"web": False, "path": ""}],
|
|
1188
|
+
volumes=volumes, cpus=_cpus_for(d), networks=["gini"]))
|
|
1189
|
+
host_port += 2
|
|
1190
|
+
|
|
1191
|
+
# make proxies / load balancers actually route to their drawn backends
|
|
1192
|
+
self._wire_proxies(cfg, topo, role, name, props)
|
|
1193
|
+
# serverless: gather Functions into the shared faas runtime + route API Gateways to it
|
|
1194
|
+
self._build_faas(cfg, topo, role, name, props)
|
|
1195
|
+
self._wire_api_gateway(cfg, topo, role, name, props)
|
|
1196
|
+
# security groups: per-member iptables (default-deny inbound + the listed rules)
|
|
1197
|
+
self._build_security_groups(cfg, topo, role, name, props)
|
|
1198
|
+
# auto-wire an observability stack so Prometheus/Grafana actually show data
|
|
1199
|
+
host_port = self._wire_observability(cfg, host_port)
|
|
1200
|
+
# auto-add the cloud-fabric telemetry agent if there are cloud services to watch
|
|
1201
|
+
self._build_fabric(cfg)
|
|
1202
|
+
# real Kubernetes: k3s clusters + generated Deployment/Service/HPA manifests
|
|
1203
|
+
self._build_k8s(cfg, topo, role, name, props)
|
|
1204
|
+
|
|
1205
|
+
# routers
|
|
1206
|
+
ridx = 0
|
|
1207
|
+
spec_of: dict[str, RouterSpec] = {} # router did -> its spec
|
|
1208
|
+
rtr_seg_ip: dict[tuple, str] = {} # (did, seg) -> this router's ip on seg
|
|
1209
|
+
rtr_seg_dev: dict[tuple, int] = {} # (did, seg) -> tun index (1-based)
|
|
1210
|
+
seg_routers: dict[int, list] = {} # seg -> [router dids on it]
|
|
1211
|
+
for did, r in role.items():
|
|
1212
|
+
if r != "router":
|
|
1213
|
+
continue
|
|
1214
|
+
ifaces = []
|
|
1215
|
+
pos = 0
|
|
1216
|
+
for l in by_device.get(did, []):
|
|
1217
|
+
if l not in kept:
|
|
1218
|
+
continue
|
|
1219
|
+
seg = seg_ids[seg_of_link[l.id]]
|
|
1220
|
+
key = (l.id, did)
|
|
1221
|
+
ridx += 1
|
|
1222
|
+
pos += 1
|
|
1223
|
+
ifaces.append(IfaceSpec(ip=iface_ip[key] + "/24",
|
|
1224
|
+
mac=mac(seg, 1, ridx), ep=eps[key],
|
|
1225
|
+
link_id=l.id))
|
|
1226
|
+
rtr_seg_ip[(did, seg)] = iface_ip[key]
|
|
1227
|
+
rtr_seg_dev[(did, seg)] = pos # matches run_grouter's tun{pos}
|
|
1228
|
+
seg_routers.setdefault(seg, []).append(did)
|
|
1229
|
+
if ifaces:
|
|
1230
|
+
spec = RouterSpec(name=name[did], ifaces=ifaces)
|
|
1231
|
+
cfg.routers.append(spec)
|
|
1232
|
+
spec_of[did] = spec
|
|
1233
|
+
|
|
1234
|
+
# static inter-router routes: each router needs a route to every subnet it is
|
|
1235
|
+
# NOT directly on, via the neighbouring router on the shortest path. (There is no
|
|
1236
|
+
# routing protocol between the C routers, so we compute the static routes here.)
|
|
1237
|
+
# A routed-mode GINI32 board fronts a real subnet behind its radio, which no
|
|
1238
|
+
# router knows about. Feed those in as extra destinations reached via the board.
|
|
1239
|
+
extra = [(b.physical_subnet, b.seg, b.ip)
|
|
1240
|
+
for b in cfg.gbridge if b.mode == "routed" and b.physical_subnet
|
|
1241
|
+
and b.seg >= 0]
|
|
1242
|
+
self._add_static_routes(cfg, spec_of, rtr_seg_ip, rtr_seg_dev, seg_routers,
|
|
1243
|
+
gw_seg, gw_ip, extra)
|
|
1244
|
+
|
|
1245
|
+
return cfg
|
|
1246
|
+
|
|
1247
|
+
@staticmethod
|
|
1248
|
+
def _add_static_routes(cfg, spec_of, rtr_seg_ip, rtr_seg_dev, seg_routers,
|
|
1249
|
+
gw_seg=None, gw_ip=None, extra_nets=None) -> None:
|
|
1250
|
+
"""extra_nets: [(cidr, seg, via_ip)] — destinations that are not GINI subnets but
|
|
1251
|
+
hang off a node ON `seg` (a routed-mode GINI32 board's physical subnet). Routers on
|
|
1252
|
+
that segment route to them via `via_ip`; others hop toward a router that is."""
|
|
1253
|
+
import ipaddress
|
|
1254
|
+
from collections import deque
|
|
1255
|
+
|
|
1256
|
+
routers = list(spec_of.keys())
|
|
1257
|
+
# router adjacency: two routers are neighbours if they share a segment (a
|
|
1258
|
+
# router-to-router link), which gives the gateway IPs on that link.
|
|
1259
|
+
adj: dict = {d: {} for d in routers}
|
|
1260
|
+
for _seg, rtrs in seg_routers.items():
|
|
1261
|
+
for a in rtrs:
|
|
1262
|
+
for b in rtrs:
|
|
1263
|
+
if a != b:
|
|
1264
|
+
adj[a][b] = _seg
|
|
1265
|
+
|
|
1266
|
+
for did in routers:
|
|
1267
|
+
my_segs = {seg for (d, seg) in rtr_seg_ip if d == did}
|
|
1268
|
+
# BFS: first-hop neighbour toward every reachable router
|
|
1269
|
+
dist = {did: 0}
|
|
1270
|
+
firsthop: dict = {did: None}
|
|
1271
|
+
q = deque([did])
|
|
1272
|
+
while q:
|
|
1273
|
+
cur = q.popleft()
|
|
1274
|
+
for nb in adj[cur]:
|
|
1275
|
+
if nb not in dist:
|
|
1276
|
+
dist[nb] = dist[cur] + 1
|
|
1277
|
+
firsthop[nb] = nb if cur == did else firsthop[cur]
|
|
1278
|
+
q.append(nb)
|
|
1279
|
+
routes = []
|
|
1280
|
+
for seg, cidr in cfg.subnets.items():
|
|
1281
|
+
if seg in my_segs:
|
|
1282
|
+
continue # directly connected
|
|
1283
|
+
cand = [c for c in seg_routers.get(seg, []) if c in dist and c != did]
|
|
1284
|
+
if not cand:
|
|
1285
|
+
continue # unreachable from here
|
|
1286
|
+
best = min(cand, key=lambda c: dist[c])
|
|
1287
|
+
nh = firsthop[best]
|
|
1288
|
+
if nh is None:
|
|
1289
|
+
continue
|
|
1290
|
+
shared = adj[did][nh]
|
|
1291
|
+
net = ipaddress.ip_network(cidr)
|
|
1292
|
+
routes.append({"net": str(net.network_address),
|
|
1293
|
+
"mask": str(net.netmask),
|
|
1294
|
+
"gw": rtr_seg_ip[(nh, shared)],
|
|
1295
|
+
"dev": rtr_seg_dev[(did, shared)]})
|
|
1296
|
+
|
|
1297
|
+
# subnets living behind a routed-mode GINI32 board (real devices on its radio)
|
|
1298
|
+
for cidr, bseg, via_ip in (extra_nets or []):
|
|
1299
|
+
try:
|
|
1300
|
+
net = ipaddress.ip_network(cidr, strict=False)
|
|
1301
|
+
except ValueError:
|
|
1302
|
+
continue
|
|
1303
|
+
if bseg in my_segs: # board is on my segment
|
|
1304
|
+
routes.append({"net": str(net.network_address),
|
|
1305
|
+
"mask": str(net.netmask), "gw": via_ip,
|
|
1306
|
+
"dev": rtr_seg_dev[(did, bseg)]})
|
|
1307
|
+
else: # hop toward its router
|
|
1308
|
+
cand = [c for c in seg_routers.get(bseg, []) if c in dist and c != did]
|
|
1309
|
+
if not cand:
|
|
1310
|
+
continue
|
|
1311
|
+
nh = firsthop[min(cand, key=lambda c: dist[c])]
|
|
1312
|
+
if nh is None:
|
|
1313
|
+
continue
|
|
1314
|
+
shared = adj[did][nh]
|
|
1315
|
+
routes.append({"net": str(net.network_address),
|
|
1316
|
+
"mask": str(net.netmask),
|
|
1317
|
+
"gw": rtr_seg_ip[(nh, shared)],
|
|
1318
|
+
"dev": rtr_seg_dev[(did, shared)]})
|
|
1319
|
+
|
|
1320
|
+
# default route (0.0.0.0/0) toward the Internet/NAT gateway, so internet-
|
|
1321
|
+
# bound traffic leaves the lab through the drawn Internet element.
|
|
1322
|
+
if gw_seg is not None and gw_ip:
|
|
1323
|
+
if gw_seg in my_segs: # gateway is on my segment
|
|
1324
|
+
routes.append({"net": "0.0.0.0", "mask": "0.0.0.0", "gw": gw_ip,
|
|
1325
|
+
"dev": rtr_seg_dev[(did, gw_seg)]})
|
|
1326
|
+
else: # hop toward its router
|
|
1327
|
+
cand = [c for c in seg_routers.get(gw_seg, [])
|
|
1328
|
+
if c in dist and c != did]
|
|
1329
|
+
if cand:
|
|
1330
|
+
best = min(cand, key=lambda c: dist[c])
|
|
1331
|
+
nh = firsthop[best]
|
|
1332
|
+
if nh is not None:
|
|
1333
|
+
shared = adj[did][nh]
|
|
1334
|
+
routes.append({"net": "0.0.0.0", "mask": "0.0.0.0",
|
|
1335
|
+
"gw": rtr_seg_ip[(nh, shared)],
|
|
1336
|
+
"dev": rtr_seg_dev[(did, shared)]})
|
|
1337
|
+
spec_of[did].routes = routes
|
|
1338
|
+
|
|
1339
|
+
# backend elements a proxy / load balancer can route HTTP traffic to
|
|
1340
|
+
_PROXY_BACKENDS = {"web_app", "instance", "container"}
|
|
1341
|
+
|
|
1342
|
+
@classmethod
|
|
1343
|
+
def _wire_proxies(cls, cfg, topo, role, name, props) -> None:
|
|
1344
|
+
"""A Reverse Proxy / Load Balancer only forwards if it has a backend config.
|
|
1345
|
+
Build that config from the drawn links: the connected Web Apps / Instances /
|
|
1346
|
+
Containers become its upstreams. Traefik gets a file-provider config; nginx gets
|
|
1347
|
+
an `upstream` block honoring the chosen Scheme + a /nginx_status endpoint."""
|
|
1348
|
+
id_of = {n: i for i, n in name.items()}
|
|
1349
|
+
# adjacency from the drawn links
|
|
1350
|
+
nbrs: dict[str, list] = {d: [] for d in topo.devices}
|
|
1351
|
+
for l in topo.links.values():
|
|
1352
|
+
nbrs[l.source_id].append(l.target_id)
|
|
1353
|
+
nbrs[l.target_id].append(l.source_id)
|
|
1354
|
+
|
|
1355
|
+
svc_by_name = {s.name: s for s in cfg.services}
|
|
1356
|
+
for s in cfg.services:
|
|
1357
|
+
if s.type_key not in ("proxy", "load_balancer"):
|
|
1358
|
+
continue
|
|
1359
|
+
did = id_of.get(s.name)
|
|
1360
|
+
if did is None:
|
|
1361
|
+
continue
|
|
1362
|
+
backends = []
|
|
1363
|
+
for nb in nbrs.get(did, []):
|
|
1364
|
+
tk = topo.devices[nb].type_key
|
|
1365
|
+
if tk in cls._PROXY_BACKENDS:
|
|
1366
|
+
port = 80
|
|
1367
|
+
backends.append((_svc(name[nb]), port))
|
|
1368
|
+
elif role.get(nb) == "service" and tk not in ("proxy", "load_balancer"):
|
|
1369
|
+
bs = svc_by_name.get(name[nb])
|
|
1370
|
+
port = bs.ports[0]["container"] if (bs and bs.ports) else 80
|
|
1371
|
+
backends.append((_svc(name[nb]), port))
|
|
1372
|
+
if not backends:
|
|
1373
|
+
cfg.notes.append(f"{s.name}: no backends wired — connect a Web App to it")
|
|
1374
|
+
continue
|
|
1375
|
+
sname = _svc(s.name)
|
|
1376
|
+
if s.type_key == "proxy":
|
|
1377
|
+
servers = "\n".join(f' - url: "http://{h}:{p}"' for h, p in backends)
|
|
1378
|
+
s.files[f"{sname}/dynamic.yml"] = _TRAEFIK_DYNAMIC.format(servers=servers)
|
|
1379
|
+
s.volumes.append(f"./{sname}/dynamic.yml:/etc/traefik/dynamic/dynamic.yml:ro")
|
|
1380
|
+
s.command = list(s.command) + ["--providers.file.directory=/etc/traefik/dynamic"]
|
|
1381
|
+
else: # nginx load balancer
|
|
1382
|
+
scheme = (props.get(did, {}).get("Scheme") or "round-robin").lower()
|
|
1383
|
+
directive = {"least_conn": " least_conn;\n", "least-conn": " least_conn;\n",
|
|
1384
|
+
"ip_hash": " ip_hash;\n", "ip-hash": " ip_hash;\n"}.get(scheme, "")
|
|
1385
|
+
servers = "\n".join(f" server {h}:{p};" for h, p in backends)
|
|
1386
|
+
s.files[f"{sname}/nginx.conf"] = _NGINX_LB.format(algo=directive, servers=servers)
|
|
1387
|
+
s.volumes.append(f"./{sname}/nginx.conf:/etc/nginx/nginx.conf:ro")
|
|
1388
|
+
|
|
1389
|
+
@classmethod
|
|
1390
|
+
def _build_networks(cls, cfg, topo, name) -> dict:
|
|
1391
|
+
"""VPCs + Subnets as real Docker networks (cloud-networking Phase 2).
|
|
1392
|
+
|
|
1393
|
+
A VPC is an **internal** bridge with its CIDR that every member joins — the implicit
|
|
1394
|
+
VPC fabric: members reach each other by name, but it has no internet of its own. A
|
|
1395
|
+
**public** subnet additionally puts its members on a per-VPC **egress** bridge (real
|
|
1396
|
+
internet + host-published consoles); a **private** subnet's members stay on the
|
|
1397
|
+
internal VPC net only — no internet, not reachable from the host (what 'private'
|
|
1398
|
+
means). Membership + tier come from containment: device → Subnet(Tier) → VPC.
|
|
1399
|
+
Returns {device_id -> [docker networks to join]}; non-members default to ["gini"].
|
|
1400
|
+
"""
|
|
1401
|
+
parent = {d.id: d.parent_id for d in topo.devices.values()}
|
|
1402
|
+
tkey = {d.id: d.type_key for d in topo.devices.values()}
|
|
1403
|
+
props = {d.id: getattr(d, "properties", {}) or {} for d in topo.devices.values()}
|
|
1404
|
+
|
|
1405
|
+
def ancestors(did):
|
|
1406
|
+
"""(vpc_id, subnet_id) — the nearest VPC and Subnet boxes above `did`."""
|
|
1407
|
+
vpc = sub = None
|
|
1408
|
+
seen, cur = set(), parent.get(did)
|
|
1409
|
+
while cur and cur not in seen:
|
|
1410
|
+
seen.add(cur)
|
|
1411
|
+
t = tkey.get(cur)
|
|
1412
|
+
if t == "cloud_subnet" and sub is None:
|
|
1413
|
+
sub = cur
|
|
1414
|
+
if t == "vpc":
|
|
1415
|
+
vpc = cur
|
|
1416
|
+
break
|
|
1417
|
+
cur = parent.get(cur)
|
|
1418
|
+
return vpc, sub
|
|
1419
|
+
|
|
1420
|
+
from ..domain.grouping import BOX_TYPES # VPC/Subnet/Region are containers,
|
|
1421
|
+
member_vpc, member_sub = {}, {} # not workloads — never "members"
|
|
1422
|
+
for d in topo.devices.values():
|
|
1423
|
+
if d.type_key in BOX_TYPES:
|
|
1424
|
+
continue
|
|
1425
|
+
v, s = ancestors(d.id)
|
|
1426
|
+
if v is not None:
|
|
1427
|
+
member_vpc[d.id] = v
|
|
1428
|
+
member_sub[d.id] = s
|
|
1429
|
+
if not member_vpc:
|
|
1430
|
+
return {}
|
|
1431
|
+
|
|
1432
|
+
def is_public(did) -> bool:
|
|
1433
|
+
s = member_sub.get(did)
|
|
1434
|
+
if s is None:
|
|
1435
|
+
return True # in a VPC but not in a subnet -> default public (egress)
|
|
1436
|
+
return (props[s].get("Tier", "private") or "private").strip().lower() == "public"
|
|
1437
|
+
|
|
1438
|
+
used: set[str] = set()
|
|
1439
|
+
|
|
1440
|
+
def unique_cidr(want):
|
|
1441
|
+
want = (want or "").strip()
|
|
1442
|
+
if _valid_cidr(want) and want not in used:
|
|
1443
|
+
used.add(want)
|
|
1444
|
+
return want
|
|
1445
|
+
for i in range(10, 250): # 10.10.0.0/16 … (avoids wan)
|
|
1446
|
+
c = f"10.{i}.0.0/16"
|
|
1447
|
+
if c not in used:
|
|
1448
|
+
used.add(c)
|
|
1449
|
+
return c
|
|
1450
|
+
return "10.249.0.0/16"
|
|
1451
|
+
|
|
1452
|
+
vpc_has_public = {member_vpc[did] for did in member_vpc if is_public(did)}
|
|
1453
|
+
net_of_vpc = {} # vpc_id -> (shared_internal_name, egress_name | None)
|
|
1454
|
+
for vdid in dict.fromkeys(member_vpc.values()): # stable, deduped
|
|
1455
|
+
v = topo.devices[vdid]
|
|
1456
|
+
slug = _svc(name[vdid])
|
|
1457
|
+
cfg.networks.append(NetworkSpec(
|
|
1458
|
+
name=slug, cidr=unique_cidr(props[vdid].get("CIDR")),
|
|
1459
|
+
label=v.name, region=props[vdid].get("Region", ""), internal=True))
|
|
1460
|
+
egress = None
|
|
1461
|
+
if vdid in vpc_has_public:
|
|
1462
|
+
egress = f"{slug}_egress"
|
|
1463
|
+
cfg.networks.append(NetworkSpec(
|
|
1464
|
+
name=egress, cidr="", label=f"{v.name} (public egress)", internal=False))
|
|
1465
|
+
net_of_vpc[vdid] = (slug, egress)
|
|
1466
|
+
|
|
1467
|
+
out = {}
|
|
1468
|
+
for did, vdid in member_vpc.items():
|
|
1469
|
+
shared, egress = net_of_vpc[vdid]
|
|
1470
|
+
nets = [shared]
|
|
1471
|
+
if egress and is_public(did):
|
|
1472
|
+
nets.append(egress)
|
|
1473
|
+
out[did] = nets
|
|
1474
|
+
return out
|
|
1475
|
+
|
|
1476
|
+
@classmethod
|
|
1477
|
+
def _build_security_groups(cls, cfg, topo, role, name, props) -> None:
|
|
1478
|
+
"""Security Groups → a stateful per-member firewall (the classic web→app→db least
|
|
1479
|
+
privilege). An SG wired to a workload/datastore makes it default-deny inbound (only
|
|
1480
|
+
stateful replies + the GINI agent allowed) and opens the ports its Ingress lists,
|
|
1481
|
+
from a CIDR or from the members of another SG. Realized as an iptables init sidecar
|
|
1482
|
+
that shares each member's network namespace (so stock images need no changes)."""
|
|
1483
|
+
sgs = [d for d in topo.devices.values() if d.type_key == "security_group"]
|
|
1484
|
+
if not sgs:
|
|
1485
|
+
return
|
|
1486
|
+
nbrs: dict[str, list] = {did: [] for did in topo.devices}
|
|
1487
|
+
for l in topo.links.values():
|
|
1488
|
+
nbrs[l.source_id].append(l.target_id)
|
|
1489
|
+
nbrs[l.target_id].append(l.source_id)
|
|
1490
|
+
|
|
1491
|
+
def members_of(sg):
|
|
1492
|
+
return [nb for nb in nbrs[sg.id] if role.get(nb) in ("service", "compute")]
|
|
1493
|
+
|
|
1494
|
+
sg_by_name: dict[str, list] = {} # "from <sg>" -> that SG's member svc names
|
|
1495
|
+
for sg in sgs:
|
|
1496
|
+
ms = [_svc(name[m]) for m in members_of(sg)]
|
|
1497
|
+
sg_by_name[_svc(sg.name)] = ms
|
|
1498
|
+
sg_by_name[(sg.name or "").strip().lower()] = ms
|
|
1499
|
+
|
|
1500
|
+
per_member: dict[str, list] = {} # member did -> union of its SGs' rules
|
|
1501
|
+
for sg in sgs:
|
|
1502
|
+
rules = _parse_ingress(props.get(sg.id, {}).get("Ingress", ""), sg_by_name)
|
|
1503
|
+
for m in members_of(sg):
|
|
1504
|
+
per_member.setdefault(m, []).extend(rules)
|
|
1505
|
+
|
|
1506
|
+
for did, rules in per_member.items():
|
|
1507
|
+
cfg.firewalls.append({"member": _svc(name[did]), "script": _sg_script(rules)})
|
|
1508
|
+
|
|
1509
|
+
# event sources that can trigger a Function: element type_key -> client port. The
|
|
1510
|
+
# queue/topic/subject the runtime subscribes to is named after the function itself.
|
|
1511
|
+
_EVENT_PORTS = {"queue": 5672, "stream": 9092, "messaging": 4222}
|
|
1512
|
+
|
|
1513
|
+
@classmethod
|
|
1514
|
+
def _build_faas(cls, cfg, topo, role, name, props) -> None:
|
|
1515
|
+
"""Gather every Function into the shared faas runtime (one container). A Function
|
|
1516
|
+
node = a handler hosted by the platform, reachable at http://faas:8000/<name>.
|
|
1517
|
+
A Function wired to an event source (Queue/Stream/Pub-Sub) also gets a trigger so
|
|
1518
|
+
the runtime subscribes and invokes the handler on each message (event-driven FaaS)."""
|
|
1519
|
+
nbrs: dict[str, list] = {d: [] for d in topo.devices}
|
|
1520
|
+
for l in topo.links.values():
|
|
1521
|
+
nbrs[l.source_id].append(l.target_id)
|
|
1522
|
+
nbrs[l.target_id].append(l.source_id)
|
|
1523
|
+
funcs = []
|
|
1524
|
+
for did, r in role.items():
|
|
1525
|
+
if r != "function":
|
|
1526
|
+
continue
|
|
1527
|
+
p = props.get(did, {})
|
|
1528
|
+
triggers = []
|
|
1529
|
+
for nb in nbrs.get(did, []):
|
|
1530
|
+
tk = topo.devices[nb].type_key
|
|
1531
|
+
port = cls._EVENT_PORTS.get(tk)
|
|
1532
|
+
if port:
|
|
1533
|
+
triggers.append({"type": tk, "host": _svc(name[nb]), "port": port})
|
|
1534
|
+
funcs.append({"name": _svc(name[did]),
|
|
1535
|
+
"handler": (p.get("Handler") or "echo").strip().lower(),
|
|
1536
|
+
"code": p.get("Code", ""),
|
|
1537
|
+
"triggers": triggers})
|
|
1538
|
+
cfg.faas = funcs
|
|
1539
|
+
|
|
1540
|
+
@classmethod
|
|
1541
|
+
def _wire_api_gateway(cls, cfg, topo, role, name, props) -> None:
|
|
1542
|
+
"""An API Gateway (Traefik) routes a URL path to each connected Function: a request
|
|
1543
|
+
to /<fn> is forwarded to the faas runtime, which dispatches to that handler."""
|
|
1544
|
+
id_of = {n: i for i, n in name.items()}
|
|
1545
|
+
nbrs: dict[str, list] = {d: [] for d in topo.devices}
|
|
1546
|
+
for l in topo.links.values():
|
|
1547
|
+
nbrs[l.source_id].append(l.target_id)
|
|
1548
|
+
nbrs[l.target_id].append(l.source_id)
|
|
1549
|
+
for s in cfg.services:
|
|
1550
|
+
if s.type_key != "api_gateway":
|
|
1551
|
+
continue
|
|
1552
|
+
did = id_of.get(s.name)
|
|
1553
|
+
routes = [_svc(name[nb]) for nb in nbrs.get(did, [])
|
|
1554
|
+
if role.get(nb) == "function"]
|
|
1555
|
+
if not routes:
|
|
1556
|
+
cfg.notes.append(f"{s.name}: connect a Function to it to route to one")
|
|
1557
|
+
continue
|
|
1558
|
+
sname = _svc(s.name)
|
|
1559
|
+
routers = "\n".join(
|
|
1560
|
+
f" fn-{fn}:\n rule: \"PathPrefix(`/{fn}`)\"\n"
|
|
1561
|
+
f" service: faas\n entryPoints: [web]" for fn in routes)
|
|
1562
|
+
dyn = ("http:\n routers:\n" + routers +
|
|
1563
|
+
"\n services:\n faas:\n loadBalancer:\n servers:\n"
|
|
1564
|
+
" - url: \"http://faas:8000\"\n")
|
|
1565
|
+
s.files[f"{sname}/dynamic.yml"] = dyn
|
|
1566
|
+
s.volumes.append(f"./{sname}/dynamic.yml:/etc/traefik/dynamic/dynamic.yml:ro")
|
|
1567
|
+
s.command = list(s.command) + ["--providers.file.directory=/etc/traefik/dynamic"]
|
|
1568
|
+
|
|
1569
|
+
# how the cloud-fabric agent probes each service type: (port, creds-kind)
|
|
1570
|
+
_FABRIC_PROBE = {"cache": (6379, None), "queue": (15672, "rabbit"),
|
|
1571
|
+
"database": (5432, "postgres"), "messaging": (8222, None),
|
|
1572
|
+
"proxy": (8080, None), "load_balancer": (80, None)}
|
|
1573
|
+
# infra services the fabric should not monitor (it watches the *app* services)
|
|
1574
|
+
_FABRIC_SKIP = {"_cadvisor", "metrics", "dashboard", "tracing"}
|
|
1575
|
+
|
|
1576
|
+
K3S_IMAGE = "rancher/k3s:v1.30.6-k3s1"
|
|
1577
|
+
|
|
1578
|
+
@classmethod
|
|
1579
|
+
def _build_k8s(cls, cfg: RuntimeConfig, topo, role, name, props) -> None:
|
|
1580
|
+
"""Turn each drawn K8s Cluster + its connected Pods/Autoscalers into a real k3s
|
|
1581
|
+
cluster spec with generated Deployment/Service/HPA manifests."""
|
|
1582
|
+
id_of = {n: i for i, n in name.items()}
|
|
1583
|
+
nbrs: dict[str, list] = {d: [] for d in topo.devices}
|
|
1584
|
+
for l in topo.links.values():
|
|
1585
|
+
nbrs[l.source_id].append(l.target_id)
|
|
1586
|
+
nbrs[l.target_id].append(l.source_id)
|
|
1587
|
+
|
|
1588
|
+
for cdid, r in role.items():
|
|
1589
|
+
if r != "k8scluster":
|
|
1590
|
+
continue
|
|
1591
|
+
deployments = []
|
|
1592
|
+
for nb in nbrs.get(cdid, []):
|
|
1593
|
+
if role.get(nb) != "k8sworkload": # a Pod (= a Deployment)
|
|
1594
|
+
continue
|
|
1595
|
+
p = props.get(nb, {})
|
|
1596
|
+
dep = {"name": _svc(name[nb]),
|
|
1597
|
+
"image": _norm_image(p.get("Image") or "nginxdemos/hello:latest"),
|
|
1598
|
+
"replicas": _int(p.get("Replicas"), 2),
|
|
1599
|
+
"port": _int(p.get("Port"), 80), "hpa": None}
|
|
1600
|
+
for nb2 in nbrs.get(nb, []): # an Autoscaling Group on it -> HPA
|
|
1601
|
+
if role.get(nb2) == "hpa":
|
|
1602
|
+
ap = props.get(nb2, {})
|
|
1603
|
+
dep["hpa"] = {"min": _int(ap.get("Min"), 1),
|
|
1604
|
+
"max": _int(ap.get("Max"), 5),
|
|
1605
|
+
"cpu": _int(ap.get("TargetCPU"), 60)}
|
|
1606
|
+
break
|
|
1607
|
+
deployments.append(dep)
|
|
1608
|
+
manifests = "\n---\n".join(
|
|
1609
|
+
m for d in deployments for m in (
|
|
1610
|
+
_k8s_deployment_yaml(d), _k8s_service_yaml(d),
|
|
1611
|
+
*( [_k8s_hpa_yaml(d)] if d["hpa"] else [] )))
|
|
1612
|
+
cfg.k8s.append(K8sSpec(name=name[cdid], svc=_svc(name[cdid]),
|
|
1613
|
+
image=cls.K3S_IMAGE, deployments=deployments,
|
|
1614
|
+
manifests=manifests))
|
|
1615
|
+
|
|
1616
|
+
@classmethod
|
|
1617
|
+
def _build_fabric(cls, cfg: RuntimeConfig) -> None:
|
|
1618
|
+
"""List the cloud services the GINI Cloud Fabric agent should watch, with the
|
|
1619
|
+
per-type probe port + credentials pulled from the catalog config."""
|
|
1620
|
+
watched = []
|
|
1621
|
+
for s in cfg.services:
|
|
1622
|
+
if s.type_key in cls._FABRIC_SKIP:
|
|
1623
|
+
continue
|
|
1624
|
+
port, credkind = cls._FABRIC_PROBE.get(
|
|
1625
|
+
s.type_key, (s.ports[0]["container"] if s.ports else 80, None))
|
|
1626
|
+
if credkind == "postgres":
|
|
1627
|
+
creds = {"user": s.env.get("POSTGRES_USER", "gini"),
|
|
1628
|
+
"password": s.env.get("POSTGRES_PASSWORD", "gini"),
|
|
1629
|
+
"db": s.env.get("POSTGRES_DB", "postgres")}
|
|
1630
|
+
elif credkind == "rabbit":
|
|
1631
|
+
creds = {"user": "guest", "password": "guest"}
|
|
1632
|
+
else:
|
|
1633
|
+
creds = {}
|
|
1634
|
+
watched.append({"name": _svc(s.name), "type": s.type_key,
|
|
1635
|
+
"host": _svc(s.name), "port": port, "creds": creds})
|
|
1636
|
+
if watched:
|
|
1637
|
+
cfg.fabric = FabricSpec(services=watched)
|
|
1638
|
+
|
|
1639
|
+
@staticmethod
|
|
1640
|
+
def _wire_observability(cfg: RuntimeConfig, host_port: int) -> int:
|
|
1641
|
+
"""If the canvas has Metrics/Dashboards, make them actually observe the lab:
|
|
1642
|
+
add a cAdvisor sidecar (universal per-container metrics), point Prometheus at it,
|
|
1643
|
+
and provision Grafana with the datasource + a starter dashboard. Returns the next
|
|
1644
|
+
free host port."""
|
|
1645
|
+
metrics = [s for s in cfg.services if s.type_key == "metrics"]
|
|
1646
|
+
dashboards = [s for s in cfg.services if s.type_key == "dashboard"]
|
|
1647
|
+
if not metrics and not dashboards:
|
|
1648
|
+
return host_port
|
|
1649
|
+
|
|
1650
|
+
# A Dashboards (Grafana) element with no Prometheus on the canvas: auto-add a
|
|
1651
|
+
# hidden Prometheus so Grafana always has a datasource + data to show (mirrors the
|
|
1652
|
+
# cAdvisor sidecar). Without this, Grafana loads but has nothing to graph.
|
|
1653
|
+
if dashboards and not metrics:
|
|
1654
|
+
from .cloud_catalog import service_for
|
|
1655
|
+
auto = ServiceSpec(
|
|
1656
|
+
name="Prometheus", type_key="metrics", image=service_for("metrics").image,
|
|
1657
|
+
summary="Auto-added Prometheus backing the dashboard (scrapes cAdvisor).",
|
|
1658
|
+
ports=[{"container": 9090, "host": host_port,
|
|
1659
|
+
"label": "console", "web": True}])
|
|
1660
|
+
host_port += 1
|
|
1661
|
+
cfg.services.append(auto)
|
|
1662
|
+
metrics = [auto]
|
|
1663
|
+
cfg.notes.append("auto-added Prometheus + cAdvisor behind Grafana")
|
|
1664
|
+
|
|
1665
|
+
# cAdvisor — exposes CPU/mem/net for EVERY container, so any topology is
|
|
1666
|
+
# observable without the apps exporting anything. It's infra (not a canvas node).
|
|
1667
|
+
cfg.services.append(ServiceSpec(
|
|
1668
|
+
name="cAdvisor", type_key="_cadvisor",
|
|
1669
|
+
image="gcr.io/cadvisor/cadvisor:v0.49.1",
|
|
1670
|
+
summary="Per-container CPU / memory / network metrics for the whole lab.",
|
|
1671
|
+
command=["-housekeeping_interval=2s", "-docker_only=true"], # fresher data
|
|
1672
|
+
ports=[{"container": 8080, "host": host_port, "label": "cadvisor", "web": True}],
|
|
1673
|
+
volumes=["/:/rootfs:ro", "/var/run:/var/run:ro", "/sys:/sys:ro",
|
|
1674
|
+
"/var/lib/docker/:/var/lib/docker:ro", "/dev/disk/:/dev/disk:ro"],
|
|
1675
|
+
privileged=True))
|
|
1676
|
+
host_port += 1
|
|
1677
|
+
|
|
1678
|
+
for prom in metrics: # Prometheus scrapes cAdvisor (+ itself)
|
|
1679
|
+
prom.files["observability/prometheus.yml"] = _PROMETHEUS_YML
|
|
1680
|
+
prom.volumes.append(
|
|
1681
|
+
"./observability/prometheus.yml:/etc/prometheus/prometheus.yml:ro")
|
|
1682
|
+
|
|
1683
|
+
if dashboards and metrics: # Grafana: datasource -> Prometheus + a dashboard
|
|
1684
|
+
prom_svc = _svc(metrics[0].name)
|
|
1685
|
+
for graf in dashboards:
|
|
1686
|
+
graf.files["observability/grafana/ds.yml"] = _GRAFANA_DS.format(prom=prom_svc)
|
|
1687
|
+
graf.files["observability/grafana/dash.yml"] = _GRAFANA_PROVIDER
|
|
1688
|
+
graf.files["observability/grafana/container.json"] = _grafana_dashboard_json()
|
|
1689
|
+
graf.volumes += [
|
|
1690
|
+
"./observability/grafana/ds.yml:"
|
|
1691
|
+
"/etc/grafana/provisioning/datasources/ds.yml:ro",
|
|
1692
|
+
"./observability/grafana/dash.yml:"
|
|
1693
|
+
"/etc/grafana/provisioning/dashboards/dash.yml:ro",
|
|
1694
|
+
"./observability/grafana/container.json:"
|
|
1695
|
+
"/var/lib/grafana/dashboards/container.json:ro",
|
|
1696
|
+
]
|
|
1697
|
+
# land students on the provisioned dashboard — set ONLY now that the file
|
|
1698
|
+
# exists (else Grafana errors "Failed to load home dashboard").
|
|
1699
|
+
graf.env["GF_DASHBOARDS_DEFAULT_HOME_DASHBOARD_PATH"] = \
|
|
1700
|
+
"/var/lib/grafana/dashboards/container.json"
|
|
1701
|
+
return host_port
|
|
1702
|
+
|
|
1703
|
+
|
|
1704
|
+
def validate(topo: Topology) -> list[dict]:
|
|
1705
|
+
"""Advisory topology lint — never blocks, just surfaces issues a student should see.
|
|
1706
|
+
|
|
1707
|
+
Returns a list of {level: 'warn'|'info', device: name|None, message}.
|
|
1708
|
+
"""
|
|
1709
|
+
issues: list[dict] = []
|
|
1710
|
+
role = {d.id: _role(d.type_key) for d in topo.devices.values()}
|
|
1711
|
+
name = {d.id: d.name for d in topo.devices.values()}
|
|
1712
|
+
nbrs: dict[str, list] = {d.id: [] for d in topo.devices.values()}
|
|
1713
|
+
for l in topo.links.values():
|
|
1714
|
+
nbrs[l.source_id].append(l.target_id)
|
|
1715
|
+
nbrs[l.target_id].append(l.source_id)
|
|
1716
|
+
|
|
1717
|
+
# 0. GINI32 boards name real hardware. A missing or duplicated BoardID does not
|
|
1718
|
+
# fail loudly at run time — the relay keys its table by that id, so a duplicate
|
|
1719
|
+
# makes one board silently disappear. Say so on the canvas instead.
|
|
1720
|
+
boards = [d for d in topo.devices.values() if d.type_key == "gini32"]
|
|
1721
|
+
seen_ids: dict[str, str] = {}
|
|
1722
|
+
for d in boards:
|
|
1723
|
+
bid = str((d.properties or {}).get("BoardID", "")).strip()
|
|
1724
|
+
if not bid:
|
|
1725
|
+
issues.append({"level": "warn", "device": d.name,
|
|
1726
|
+
"message": "No BoardID set — put the id from the board's "
|
|
1727
|
+
"label here (see `gini32 provision --id`), or no "
|
|
1728
|
+
"hardware will attach to this element."})
|
|
1729
|
+
elif bid in seen_ids:
|
|
1730
|
+
issues.append({"level": "warn", "device": d.name,
|
|
1731
|
+
"message": f"BoardID {bid!r} is also used by "
|
|
1732
|
+
f"{seen_ids[bid]} — two elements cannot share one "
|
|
1733
|
+
f"physical board; one of them will never connect."})
|
|
1734
|
+
else:
|
|
1735
|
+
seen_ids[bid] = d.name
|
|
1736
|
+
# overlapping physical subnets => routers get two routes to one network
|
|
1737
|
+
import ipaddress as _ipa
|
|
1738
|
+
nets: list[tuple] = []
|
|
1739
|
+
for d in boards:
|
|
1740
|
+
raw = str((d.properties or {}).get("PhysicalSubnet", "")).strip()
|
|
1741
|
+
if not raw:
|
|
1742
|
+
continue # blank is fine: allocated automatically
|
|
1743
|
+
try:
|
|
1744
|
+
net = _ipa.ip_network(raw, strict=False)
|
|
1745
|
+
except ValueError:
|
|
1746
|
+
continue # the compiler already notes and replaces it
|
|
1747
|
+
for other, onet in nets:
|
|
1748
|
+
if net.overlaps(onet):
|
|
1749
|
+
issues.append({"level": "warn", "device": d.name,
|
|
1750
|
+
"message": f"PhysicalSubnet {raw} overlaps {other}'s — "
|
|
1751
|
+
f"give each board its own, or leave both blank "
|
|
1752
|
+
f"to have them allocated."})
|
|
1753
|
+
nets.append((d.name, net))
|
|
1754
|
+
|
|
1755
|
+
# 1. isolated devices (degree 0) — not part of any network. xv6 runs standalone (it
|
|
1756
|
+
# has no networking), its peripherals are optional, and OS Zoo guests run in isolation
|
|
1757
|
+
# (display-only, no fabric wiring in v1), so none of them are "islands".
|
|
1758
|
+
tkey = {d.id: d.type_key for d in topo.devices.values()}
|
|
1759
|
+
for did, r in role.items():
|
|
1760
|
+
if r == "group" or nbrs[did]:
|
|
1761
|
+
continue
|
|
1762
|
+
if r == "oszoo" or tkey[did] in ("xv6", "terminal", "storage_volume"):
|
|
1763
|
+
continue
|
|
1764
|
+
issues.append({"level": "warn", "device": name[did],
|
|
1765
|
+
"message": f"{name[did]} isn't connected to anything."})
|
|
1766
|
+
|
|
1767
|
+
# 2. machines with no gateway (no router on any of their subnets) — islands.
|
|
1768
|
+
# A host on a switched/SDN L2 domain is fine without a router (it reaches its
|
|
1769
|
+
# LAN at layer 2), so only warn for hosts NOT on any switch/OVS.
|
|
1770
|
+
cfg = RuntimeCompiler().compile(topo)
|
|
1771
|
+
id_of = {n: i for i, n in name.items()}
|
|
1772
|
+
for m in cfg.machines:
|
|
1773
|
+
if m.gw or getattr(m, "gateway", False): # gateway egresses via its own uplink
|
|
1774
|
+
continue
|
|
1775
|
+
did = id_of.get(m.name)
|
|
1776
|
+
on_lan = did is not None and any(
|
|
1777
|
+
role.get(nb) in ("switch", "ovs") for nb in nbrs[did])
|
|
1778
|
+
if on_lan:
|
|
1779
|
+
continue
|
|
1780
|
+
issues.append({"level": "warn", "device": m.name,
|
|
1781
|
+
"message": f"{m.name} has no gateway (no router on its "
|
|
1782
|
+
f"subnet) — it can only reach hosts on its own subnet."})
|
|
1783
|
+
|
|
1784
|
+
# 2b. SDN advisories — teach the control-plane relationship
|
|
1785
|
+
for o in cfg.ovs_switches:
|
|
1786
|
+
if not o.controller:
|
|
1787
|
+
issues.append({"level": "warn", "device": o.name,
|
|
1788
|
+
"message": f"{o.name} has no controller — it runs fail-secure "
|
|
1789
|
+
f"with an empty flow table, so it drops ALL traffic. "
|
|
1790
|
+
f"Connect a controller to give it switching behavior."})
|
|
1791
|
+
for c in cfg.controllers:
|
|
1792
|
+
if not c.switches:
|
|
1793
|
+
issues.append({"level": "warn", "device": c.name,
|
|
1794
|
+
"message": f"{c.name} isn't programming any switch — connect "
|
|
1795
|
+
f"it to an OVS for it to control."})
|
|
1796
|
+
|
|
1797
|
+
# 3. L2 loop among switches/hubs — our switches don't run STP, so a loop floods
|
|
1798
|
+
l2 = {"switch", "hub", "ovs"}
|
|
1799
|
+
uf = _UF()
|
|
1800
|
+
looped = False
|
|
1801
|
+
for l in topo.links.values():
|
|
1802
|
+
if role.get(l.source_id) in l2 and role.get(l.target_id) in l2:
|
|
1803
|
+
if uf.find(l.source_id) == uf.find(l.target_id):
|
|
1804
|
+
looped = True
|
|
1805
|
+
uf.union(l.source_id, l.target_id)
|
|
1806
|
+
if looped:
|
|
1807
|
+
issues.append({"level": "warn", "device": None,
|
|
1808
|
+
"message": "Switch loop detected — the switches have no spanning "
|
|
1809
|
+
"tree, so a loop will flood broadcasts. Remove a link."})
|
|
1810
|
+
|
|
1811
|
+
# 4. compiler notes (e.g. grouping devices whose links are organizational only)
|
|
1812
|
+
for note in cfg.notes:
|
|
1813
|
+
issues.append({"level": "info", "device": None, "message": note})
|
|
1814
|
+
return issues
|
|
1815
|
+
|
|
1816
|
+
|
|
1817
|
+
def address_map(topo: Topology) -> dict[str, dict]:
|
|
1818
|
+
"""Per-device addressing for the inspector / canvas labels.
|
|
1819
|
+
|
|
1820
|
+
Returns {device_name: {role, interfaces:[{name, ip, mac, subnet, gateway, peer}], …}}.
|
|
1821
|
+
IPs/MACs come from compiling the topology, so they exist before anything runs.
|
|
1822
|
+
"""
|
|
1823
|
+
import ipaddress
|
|
1824
|
+
cfg = RuntimeCompiler().compile(topo)
|
|
1825
|
+
|
|
1826
|
+
def subnet(cidr: str) -> str:
|
|
1827
|
+
return str(ipaddress.ip_interface(cidr).network)
|
|
1828
|
+
|
|
1829
|
+
out: dict[str, dict] = {}
|
|
1830
|
+
for m in cfg.machines:
|
|
1831
|
+
out[m.name] = {"role": "machine", "interfaces": [
|
|
1832
|
+
{"name": f"eth{i}", "ip": itf.ip, "mac": itf.mac, "subnet": subnet(itf.ip),
|
|
1833
|
+
"gateway": m.gw if i == 0 else None, "peer": itf.ep.peer.device,
|
|
1834
|
+
"link_id": itf.link_id}
|
|
1835
|
+
for i, itf in enumerate(m.ifaces)]}
|
|
1836
|
+
for r in cfg.routers:
|
|
1837
|
+
out[r.name] = {"role": "router", "interfaces": [
|
|
1838
|
+
{"name": f"eth{i}", "ip": itf.ip, "mac": itf.mac, "subnet": subnet(itf.ip),
|
|
1839
|
+
"gateway": None, "peer": itf.ep.peer.device, "link_id": itf.link_id}
|
|
1840
|
+
for i, itf in enumerate(r.ifaces)]}
|
|
1841
|
+
for s in cfg.switches:
|
|
1842
|
+
out[s.name] = {"role": "switch", "ports": len(s.eps), "interfaces": [],
|
|
1843
|
+
"peers": [e.peer.device for e in s.eps]}
|
|
1844
|
+
return out
|
|
1845
|
+
|
|
1846
|
+
|
|
1847
|
+
def overlay_hosts(addressing: dict) -> dict:
|
|
1848
|
+
"""device name -> its primary overlay (gini0) IP, from `address_map` output. GINI writes these
|
|
1849
|
+
into each machine's /etc/hosts so names resolve over the DRAWN network (gini0) instead of the
|
|
1850
|
+
Docker bridge — which is what makes DNS/getent/ping/reach ride the overlay."""
|
|
1851
|
+
out: dict[str, str] = {}
|
|
1852
|
+
for name, info in (addressing or {}).items():
|
|
1853
|
+
for itf in info.get("interfaces", []):
|
|
1854
|
+
ip = str(itf.get("ip", "")).split("/")[0].strip()
|
|
1855
|
+
if ip:
|
|
1856
|
+
out[name] = ip
|
|
1857
|
+
break
|
|
1858
|
+
return out
|