gini-toolkit 6.0.1.dev0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- gini/__init__.py +12 -0
- gini/__main__.py +107 -0
- gini/_version.py +24 -0
- gini/agent/__init__.py +17 -0
- gini/agent/agent_gamemaster.py +140 -0
- gini/agent/api.py +291 -0
- gini/agent/ask.py +123 -0
- gini/agent/authoring.py +72 -0
- gini/agent/blackboard.py +114 -0
- gini/agent/contracts.py +142 -0
- gini/agent/domains.py +91 -0
- gini/agent/embed.py +123 -0
- gini/agent/gamemaster.py +256 -0
- gini/agent/kb.py +148 -0
- gini/agent/lesson_resolver.py +261 -0
- gini/agent/llm/__init__.py +5 -0
- gini/agent/llm/backend.py +43 -0
- gini/agent/llm/fake.py +25 -0
- gini/agent/llm/ollama.py +206 -0
- gini/agent/loop.py +258 -0
- gini/agent/mcp_server.py +86 -0
- gini/agent/meaning.py +225 -0
- gini/agent/mission.py +210 -0
- gini/agent/mission_controller.py +208 -0
- gini/agent/narration.py +116 -0
- gini/agent/notifier.py +86 -0
- gini/agent/personas.py +79 -0
- gini/agent/reasoning.py +172 -0
- gini/agent/recall.py +248 -0
- gini/agent/session.py +79 -0
- gini/agent/teaching_center.py +482 -0
- gini/agent/tools/__init__.py +3 -0
- gini/agent/tools/registry.py +193 -0
- gini/agent/twin/__init__.py +28 -0
- gini/agent/twin/authoring.py +71 -0
- gini/agent/twin/contracts.py +54 -0
- gini/agent/twin/dialectic.py +189 -0
- gini/agent/twin/harness.py +93 -0
- gini/agent/twin/justify.py +156 -0
- gini/agent/twin/learner.py +64 -0
- gini/agent/twin/mission.py +60 -0
- gini/agent/twin/os_coach.py +79 -0
- gini/agent/twin/salience.py +30 -0
- gini/agent/understand.py +250 -0
- gini/agent/verifiers.py +106 -0
- gini/agent/wizard.py +178 -0
- gini/agent/xv6_pack.py +74 -0
- gini/app/__init__.py +3 -0
- gini/app/context.py +368 -0
- gini/app/paths.py +121 -0
- gini/data/README.md +21 -0
- gini/domain/__init__.py +9 -0
- gini/domain/assembly.py +209 -0
- gini/domain/authoring.py +353 -0
- gini/domain/blueprints.py +5 -0
- gini/domain/capabilities.py +177 -0
- gini/domain/catalog.py +85 -0
- gini/domain/certify.py +201 -0
- gini/domain/compose.py +413 -0
- gini/domain/composition.py +88 -0
- gini/domain/concepts.py +383 -0
- gini/domain/connection_rules.py +269 -0
- gini/domain/constraints.py +153 -0
- gini/domain/content.py +59 -0
- gini/domain/cpu_journey.py +89 -0
- gini/domain/devices.py +747 -0
- gini/domain/diagnose.py +201 -0
- gini/domain/element_guide.py +327 -0
- gini/domain/explain.py +90 -0
- gini/domain/fingerprint.py +201 -0
- gini/domain/firewall.py +34 -0
- gini/domain/flowlog.py +61 -0
- gini/domain/flowtable.py +179 -0
- gini/domain/fragment_yaml.py +230 -0
- gini/domain/fragments.py +169 -0
- gini/domain/games/__init__.py +2 -0
- gini/domain/games/paging_games.py +119 -0
- gini/domain/games/policy_game.py +86 -0
- gini/domain/games/process_game.py +48 -0
- gini/domain/games/thrash_game.py +75 -0
- gini/domain/games/translate_game.py +60 -0
- gini/domain/games/trap_game.py +86 -0
- gini/domain/grader.py +155 -0
- gini/domain/grouping.py +67 -0
- gini/domain/legality.py +103 -0
- gini/domain/lesson.py +241 -0
- gini/domain/lexicon.py +150 -0
- gini/domain/machine_state.py +410 -0
- gini/domain/missions/networking/basic-lan.yaml +32 -0
- gini/domain/missions/networking/cache-in-front.yaml +23 -0
- gini/domain/missions/networking/decouple-with-queue.yaml +31 -0
- gini/domain/missions/networking/drive-load.yaml +20 -0
- gini/domain/missions/networking/fix-the-address.yaml +75 -0
- gini/domain/missions/networking/fix-the-lan.yaml +43 -0
- gini/domain/missions/networking/inspect-flows.yaml +16 -0
- gini/domain/missions/networking/k8s-autoscale.yaml +27 -0
- gini/domain/missions/networking/least-privilege.yaml +21 -0
- gini/domain/missions/networking/load-balanced-web.yaml +29 -0
- gini/domain/missions/networking/observe-it.yaml +24 -0
- gini/domain/missions/networking/put-in-vpc.yaml +30 -0
- gini/domain/missions/networking/reachability-boundary.yaml +56 -0
- gini/domain/missions/networking/sdn-reactive.yaml +35 -0
- gini/domain/missions/networking/send-request.yaml +19 -0
- gini/domain/missions/networking/serverless-api.yaml +25 -0
- gini/domain/missions/networking/service-chain.yaml +33 -0
- gini/domain/missions/os/lottery-fix.yaml +19 -0
- gini/domain/missions/os/priority-fix.yaml +24 -0
- gini/domain/missions.py +111 -0
- gini/domain/modulechain.py +36 -0
- gini/domain/objectives.py +488 -0
- gini/domain/os_zoo.py +79 -0
- gini/domain/paging_sim.py +141 -0
- gini/domain/pricing.py +199 -0
- gini/domain/probes.py +226 -0
- gini/domain/profile.py +142 -0
- gini/domain/recipes.py +738 -0
- gini/domain/riders.py +309 -0
- gini/domain/router_modules.py +224 -0
- gini/domain/routetable.py +67 -0
- gini/domain/scoring.py +76 -0
- gini/domain/staging.py +122 -0
- gini/domain/syscall_builder.py +144 -0
- gini/domain/topic_cloud.py +62 -0
- gini/domain/topology.py +213 -0
- gini/domain/vocabulary.py +51 -0
- gini/domain/xv6.py +808 -0
- gini/domain/xv6_fs.py +250 -0
- gini/domain/xv6_runner.py +113 -0
- gini/domain/xv6_vm.py +385 -0
- gini/gloader.py +17 -0
- gini/runtime/__init__.py +18 -0
- gini/runtime/cloudfabric_agent.py +370 -0
- gini/runtime/console.py +68 -0
- gini/runtime/control.py +70 -0
- gini/runtime/frame.py +138 -0
- gini/runtime/gbridge.py +638 -0
- gini/runtime/grouter.py +223 -0
- gini/runtime/hostsim.py +90 -0
- gini/runtime/shuttle.py +348 -0
- gini/runtime/switch.py +109 -0
- gini/runtime/transport.py +77 -0
- gini/runtime/xv6_bridge.py +312 -0
- gini/server/__init__.py +22 -0
- gini/server/__main__.py +74 -0
- gini/server/app.py +140 -0
- gini/server/auth.py +82 -0
- gini/server/policy.py +57 -0
- gini/server/session.py +23 -0
- gini/services/__init__.py +15 -0
- gini/services/boardflash.py +248 -0
- gini/services/boardsetup.py +374 -0
- gini/services/cloud_catalog.py +143 -0
- gini/services/compiler.py +1858 -0
- gini/services/discovery.py +324 -0
- gini/services/gloader.py +183 -0
- gini/services/orchestrator.py +1460 -0
- gini/services/persistence.py +28 -0
- gini/services/probe_runner.py +149 -0
- gini/services/project.py +217 -0
- gini/services/remote.py +93 -0
- gini/services/rider_runner.py +96 -0
- gini/services/rider_session.py +171 -0
- gini/services/shadow_store.py +52 -0
- gini/services/terminal.py +45 -0
- gini/setup/__init__.py +17 -0
- gini/setup/cli.py +109 -0
- gini/setup/images.py +33 -0
- gini/setup/marker.py +43 -0
- gini/setup/runtime.py +69 -0
- gini/ui/__init__.py +3 -0
- gini/ui/assets/app_icon.icns +0 -0
- gini/ui/assets/app_icon.ico +0 -0
- gini/ui/assets/app_icon.png +0 -0
- gini/ui/assets/app_icon_1024.png +0 -0
- gini/ui/assets/cue/_w.txt +1 -0
- gini/ui/assets/cue/ai.png +0 -0
- gini/ui/assets/cue/canvas.png +0 -0
- gini/ui/assets/cue/cloud.png +0 -0
- gini/ui/assets/cue/cost.png +0 -0
- gini/ui/assets/cue/dark/ai.png +0 -0
- gini/ui/assets/cue/dark/canvas.png +0 -0
- gini/ui/assets/cue/dark/cloud.png +0 -0
- gini/ui/assets/cue/dark/cost.png +0 -0
- gini/ui/assets/cue/dark/metrics.png +0 -0
- gini/ui/assets/cue/dark/router.png +0 -0
- gini/ui/assets/cue/dark/run.png +0 -0
- gini/ui/assets/cue/dark/serverless.png +0 -0
- gini/ui/assets/cue/dark/settings.png +0 -0
- gini/ui/assets/cue/dark/welcome.png +0 -0
- gini/ui/assets/cue/dark/wizard.png +0 -0
- gini/ui/assets/cue/ginibrand/ai.png +0 -0
- gini/ui/assets/cue/ginibrand/canvas.png +0 -0
- gini/ui/assets/cue/ginibrand/cloud.png +0 -0
- gini/ui/assets/cue/ginibrand/cost.png +0 -0
- gini/ui/assets/cue/ginibrand/metrics.png +0 -0
- gini/ui/assets/cue/ginibrand/router.png +0 -0
- gini/ui/assets/cue/ginibrand/run.png +0 -0
- gini/ui/assets/cue/ginibrand/serverless.png +0 -0
- gini/ui/assets/cue/ginibrand/settings.png +0 -0
- gini/ui/assets/cue/ginibrand/welcome.png +0 -0
- gini/ui/assets/cue/ginibrand/wizard.png +0 -0
- gini/ui/assets/cue/highcontrast/ai.png +0 -0
- gini/ui/assets/cue/highcontrast/canvas.png +0 -0
- gini/ui/assets/cue/highcontrast/cloud.png +0 -0
- gini/ui/assets/cue/highcontrast/cost.png +0 -0
- gini/ui/assets/cue/highcontrast/metrics.png +0 -0
- gini/ui/assets/cue/highcontrast/router.png +0 -0
- gini/ui/assets/cue/highcontrast/run.png +0 -0
- gini/ui/assets/cue/highcontrast/serverless.png +0 -0
- gini/ui/assets/cue/highcontrast/settings.png +0 -0
- gini/ui/assets/cue/highcontrast/welcome.png +0 -0
- gini/ui/assets/cue/highcontrast/wizard.png +0 -0
- gini/ui/assets/cue/light/ai.png +0 -0
- gini/ui/assets/cue/light/canvas.png +0 -0
- gini/ui/assets/cue/light/cloud.png +0 -0
- gini/ui/assets/cue/light/cost.png +0 -0
- gini/ui/assets/cue/light/metrics.png +0 -0
- gini/ui/assets/cue/light/router.png +0 -0
- gini/ui/assets/cue/light/run.png +0 -0
- gini/ui/assets/cue/light/serverless.png +0 -0
- gini/ui/assets/cue/light/settings.png +0 -0
- gini/ui/assets/cue/light/welcome.png +0 -0
- gini/ui/assets/cue/light/wizard.png +0 -0
- gini/ui/assets/cue/metrics.png +0 -0
- gini/ui/assets/cue/router.png +0 -0
- gini/ui/assets/cue/run.png +0 -0
- gini/ui/assets/cue/serverless.png +0 -0
- gini/ui/assets/cue/settings.png +0 -0
- gini/ui/assets/cue/welcome.png +0 -0
- gini/ui/assets/cue/wizard.png +0 -0
- gini/ui/assistant.py +2111 -0
- gini/ui/author_dialog.py +184 -0
- gini/ui/board_dialog.py +247 -0
- gini/ui/branding.py +21 -0
- gini/ui/canvas.py +2007 -0
- gini/ui/chat_panel.py +7 -0
- gini/ui/cpu_journey.py +212 -0
- gini/ui/cpu_lab.py +306 -0
- gini/ui/cue_cards.py +214 -0
- gini/ui/dashboard.py +222 -0
- gini/ui/diagnose_game.py +336 -0
- gini/ui/fingerprint_lab.py +219 -0
- gini/ui/flash_dialog.py +244 -0
- gini/ui/flow_layout.py +63 -0
- gini/ui/fragment_manager.py +1415 -0
- gini/ui/game_catalog.py +184 -0
- gini/ui/game_renderers.py +340 -0
- gini/ui/games_lab.py +90 -0
- gini/ui/inspector.py +1055 -0
- gini/ui/live_metrics.py +130 -0
- gini/ui/machine_lab.py +1412 -0
- gini/ui/main_window.py +3153 -0
- gini/ui/memory_lab.py +371 -0
- gini/ui/mission_panel.py +302 -0
- gini/ui/mode_indicator.py +227 -0
- gini/ui/palette.py +112 -0
- gini/ui/peripherals.py +218 -0
- gini/ui/process_tree.py +130 -0
- gini/ui/reset_dialog.py +179 -0
- gini/ui/router_lab.py +776 -0
- gini/ui/run_button.py +183 -0
- gini/ui/settings_dialog.py +234 -0
- gini/ui/signin_dialog.py +111 -0
- gini/ui/storage_lab.py +219 -0
- gini/ui/syscall_builder.py +235 -0
- gini/ui/syscall_lab.py +152 -0
- gini/ui/theme/__init__.py +5 -0
- gini/ui/theme/icons.py +145 -0
- gini/ui/theme/manager.py +291 -0
- gini/ui/theme/tokens.py +194 -0
- gini/ui/trap_lab.py +270 -0
- gini/ui/worker_host.py +102 -0
- gini/ui/zoo_lab.py +112 -0
- gini_toolkit-6.0.1.dev0.dist-info/METADATA +77 -0
- gini_toolkit-6.0.1.dev0.dist-info/RECORD +278 -0
- gini_toolkit-6.0.1.dev0.dist-info/WHEEL +5 -0
- gini_toolkit-6.0.1.dev0.dist-info/entry_points.txt +3 -0
- gini_toolkit-6.0.1.dev0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,193 @@
|
|
|
1
|
+
"""Shared tool registry — the single contract for build / inspect / explain / present.
|
|
2
|
+
|
|
3
|
+
One registry is consumed by BOTH the in-app agent loop (local Ollama) and the MCP
|
|
4
|
+
server (external agents). Register a tool once and it's available everywhere. Each
|
|
5
|
+
tool carries a JSON-Schema parameter spec, so it can be handed to an LLM as an
|
|
6
|
+
OpenAI/Ollama-style function and executed by name.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from collections.abc import Callable
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from ..api import GiniAPI
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class ToolSpec:
|
|
19
|
+
name: str
|
|
20
|
+
description: str
|
|
21
|
+
parameters: dict # JSON Schema (object)
|
|
22
|
+
handler: Callable[..., Any]
|
|
23
|
+
group: str = "general" # build | inspect | explain | present
|
|
24
|
+
|
|
25
|
+
def to_openai(self) -> dict:
|
|
26
|
+
return {"type": "function", "function": {
|
|
27
|
+
"name": self.name,
|
|
28
|
+
"description": self.description,
|
|
29
|
+
"parameters": self.parameters,
|
|
30
|
+
}}
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class ToolRegistry:
|
|
34
|
+
def __init__(self) -> None:
|
|
35
|
+
self._tools: dict[str, ToolSpec] = {}
|
|
36
|
+
|
|
37
|
+
def register(self, spec: ToolSpec) -> None:
|
|
38
|
+
self._tools[spec.name] = spec
|
|
39
|
+
|
|
40
|
+
def specs(self) -> list[ToolSpec]:
|
|
41
|
+
return list(self._tools.values())
|
|
42
|
+
|
|
43
|
+
def openai_tools(self) -> list[dict]:
|
|
44
|
+
return [t.to_openai() for t in self._tools.values()]
|
|
45
|
+
|
|
46
|
+
def names(self) -> list[str]:
|
|
47
|
+
return list(self._tools)
|
|
48
|
+
|
|
49
|
+
def execute(self, name: str, args: dict | None = None) -> Any:
|
|
50
|
+
args = args or {}
|
|
51
|
+
spec = self._tools.get(name)
|
|
52
|
+
if spec is None:
|
|
53
|
+
return {"error": f"unknown tool {name!r}"}
|
|
54
|
+
try:
|
|
55
|
+
return spec.handler(**args)
|
|
56
|
+
except TypeError as e:
|
|
57
|
+
return {"error": f"bad arguments for {name}: {e}"}
|
|
58
|
+
except Exception as e: # surface to the agent rather than crashing
|
|
59
|
+
return {"error": str(e)}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
# JSON-schema fragments
|
|
63
|
+
_STR = {"type": "string"}
|
|
64
|
+
_NUM = {"type": "number"}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def build_registry(api: GiniAPI) -> ToolRegistry:
|
|
68
|
+
"""Register the GiniAPI surface as tools. `present` tools are added in phase A3."""
|
|
69
|
+
r = ToolRegistry()
|
|
70
|
+
|
|
71
|
+
r.register(ToolSpec(
|
|
72
|
+
"list_device_types", "List every device/element type GINI can place (networking + cloud).",
|
|
73
|
+
{"type": "object", "properties": {}},
|
|
74
|
+
lambda: api.list_device_types(), group="inspect"))
|
|
75
|
+
|
|
76
|
+
r.register(ToolSpec(
|
|
77
|
+
"add_device", "Add a device. type_key e.g. 'router','switch','vpc','container','instance'.",
|
|
78
|
+
{"type": "object",
|
|
79
|
+
"properties": {"type_key": _STR, "name": _STR, "x": _NUM, "y": _NUM},
|
|
80
|
+
"required": ["type_key"]},
|
|
81
|
+
lambda type_key, name="", x=0.0, y=0.0:
|
|
82
|
+
api.add_device(type_key, name=name or None, x=x, y=y), group="build"))
|
|
83
|
+
|
|
84
|
+
r.register(ToolSpec(
|
|
85
|
+
"connect_devices", "Create a link between two devices (by name or id).",
|
|
86
|
+
{"type": "object",
|
|
87
|
+
"properties": {"a": _STR, "b": _STR, "label": _STR}, "required": ["a", "b"]},
|
|
88
|
+
lambda a, b, label="": api.connect(a, b, label), group="build"))
|
|
89
|
+
|
|
90
|
+
r.register(ToolSpec(
|
|
91
|
+
"set_property", "Set a property on a device (by name or id).",
|
|
92
|
+
{"type": "object",
|
|
93
|
+
"properties": {"device": _STR, "key": _STR, "value": _STR},
|
|
94
|
+
"required": ["device", "key", "value"]},
|
|
95
|
+
lambda device, key, value: api.set_property(device, key, value), group="build"))
|
|
96
|
+
|
|
97
|
+
r.register(ToolSpec(
|
|
98
|
+
"remove_device", "Remove a device (by name or id).",
|
|
99
|
+
{"type": "object", "properties": {"device": _STR}, "required": ["device"]},
|
|
100
|
+
lambda device: (api.remove_device(device), {"removed": device})[1], group="build"))
|
|
101
|
+
|
|
102
|
+
r.register(ToolSpec(
|
|
103
|
+
"inspect_device", "Inspect a device's type, properties, neighbors, and degree.",
|
|
104
|
+
{"type": "object", "properties": {"device": _STR}, "required": ["device"]},
|
|
105
|
+
lambda device: api.inspect(device), group="inspect"))
|
|
106
|
+
|
|
107
|
+
r.register(ToolSpec(
|
|
108
|
+
"get_topology", "Return the full topology as JSON (devices + links).",
|
|
109
|
+
{"type": "object", "properties": {}},
|
|
110
|
+
lambda: api.get_topology(), group="inspect"))
|
|
111
|
+
|
|
112
|
+
r.register(ToolSpec(
|
|
113
|
+
"summarize_topology", "Counts of devices, links, and categories.",
|
|
114
|
+
{"type": "object", "properties": {}},
|
|
115
|
+
lambda: api.summary(), group="inspect"))
|
|
116
|
+
|
|
117
|
+
r.register(ToolSpec(
|
|
118
|
+
"explain_topology", "Explain the whole topology in plain language for a student.",
|
|
119
|
+
{"type": "object", "properties": {}},
|
|
120
|
+
lambda: api.explain_topology(), group="explain"))
|
|
121
|
+
|
|
122
|
+
r.register(ToolSpec(
|
|
123
|
+
"explain_device", "Explain a single device in plain language for a student.",
|
|
124
|
+
{"type": "object", "properties": {"device": _STR}, "required": ["device"]},
|
|
125
|
+
lambda device: api.explain_device(device), group="explain"))
|
|
126
|
+
|
|
127
|
+
r.register(ToolSpec(
|
|
128
|
+
"explain_element", "Explain a palette element TYPE (e.g. 'router', 'switch', "
|
|
129
|
+
"'hub', 'load_balancer') — what it is and when to use it vs. similar elements. "
|
|
130
|
+
"Use when the student asks about an element from the palette, not a placed device.",
|
|
131
|
+
{"type": "object", "properties": {"type_key": _STR}, "required": ["type_key"]},
|
|
132
|
+
lambda type_key: api.explain_element_type(type_key), group="explain"))
|
|
133
|
+
|
|
134
|
+
r.register(ToolSpec(
|
|
135
|
+
"trace_path", "Get the hop-by-hop device path a packet takes from one device to "
|
|
136
|
+
"another (names). Feed the result to animate_packet to show it on the canvas.",
|
|
137
|
+
{"type": "object", "properties": {"src": _STR, "dst": _STR},
|
|
138
|
+
"required": ["src", "dst"]},
|
|
139
|
+
lambda src, dst: {"path": api.trace_path(src, dst)}, group="inspect"))
|
|
140
|
+
|
|
141
|
+
_register_present(r, api)
|
|
142
|
+
return r
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def _register_present(r: ToolRegistry, api: GiniAPI) -> None:
|
|
146
|
+
"""The `present` verb — the AI tutor's stage. Handlers emit on the event bus;
|
|
147
|
+
the canvas overlay renders them. Available to the in-app loop and external agents."""
|
|
148
|
+
bus = api.ctx.bus
|
|
149
|
+
|
|
150
|
+
def ids(names: list[str]) -> list[str]:
|
|
151
|
+
out = []
|
|
152
|
+
for n in names or []:
|
|
153
|
+
try:
|
|
154
|
+
out.append(api._resolve(n).id)
|
|
155
|
+
except KeyError:
|
|
156
|
+
pass
|
|
157
|
+
return out
|
|
158
|
+
|
|
159
|
+
_ARR = {"type": "array", "items": _STR}
|
|
160
|
+
|
|
161
|
+
r.register(ToolSpec(
|
|
162
|
+
"spotlight", "Spotlight device(s) by name and dim the rest, to focus attention.",
|
|
163
|
+
{"type": "object", "properties": {"targets": _ARR}, "required": ["targets"]},
|
|
164
|
+
lambda targets: (bus.present_spotlight.emit(ids(targets)), {"spotlight": targets})[1],
|
|
165
|
+
group="present"))
|
|
166
|
+
|
|
167
|
+
r.register(ToolSpec(
|
|
168
|
+
"highlight", "Outline/ring device(s) by name without dimming others.",
|
|
169
|
+
{"type": "object", "properties": {"targets": _ARR}, "required": ["targets"]},
|
|
170
|
+
lambda targets: (bus.present_highlight.emit(ids(targets)), {"highlight": targets})[1],
|
|
171
|
+
group="present"))
|
|
172
|
+
|
|
173
|
+
r.register(ToolSpec(
|
|
174
|
+
"callout", "Show an anchored speech bubble on a device with explanatory text.",
|
|
175
|
+
{"type": "object", "properties": {"device": _STR, "text": _STR},
|
|
176
|
+
"required": ["device", "text"]},
|
|
177
|
+
lambda device, text: (bus.present_callout.emit(ids([device])[0] if ids([device]) else "", text),
|
|
178
|
+
{"callout": device})[1], group="present"))
|
|
179
|
+
|
|
180
|
+
r.register(ToolSpec(
|
|
181
|
+
"narrate", "Speak a line of narration to the student (shown on the canvas).",
|
|
182
|
+
{"type": "object", "properties": {"text": _STR}, "required": ["text"]},
|
|
183
|
+
lambda text: (bus.present_narrate.emit(text), {"narrated": True})[1], group="present"))
|
|
184
|
+
|
|
185
|
+
r.register(ToolSpec(
|
|
186
|
+
"animate_packet", "Animate a packet travelling along a path of device names.",
|
|
187
|
+
{"type": "object", "properties": {"path": _ARR}, "required": ["path"]},
|
|
188
|
+
lambda path: (bus.present_packet.emit(ids(path)), {"animated": path})[1], group="present"))
|
|
189
|
+
|
|
190
|
+
r.register(ToolSpec(
|
|
191
|
+
"clear_stage", "Clear all tutor overlays (spotlights, callouts, highlights).",
|
|
192
|
+
{"type": "object", "properties": {}},
|
|
193
|
+
lambda: (bus.present_clear.emit(), {"cleared": True})[1], group="present"))
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""The Reasoning Twin — a deterministic (non-LLM) shadow that twins each structured reasoning
|
|
2
|
+
turn and checks the LLM caught the main points (docs/REASONING_2.0_DESIGN.md).
|
|
3
|
+
|
|
4
|
+
It enumerates the concerns that matter (from GINI's symbolic substrate — blackboard verdicts,
|
|
5
|
+
the predicate explainer, legality flags), asks the Reasoning persona to report coverage against
|
|
6
|
+
them, diffs that report EXACTLY (a set diff, no NLP), poses "why not X?" objections for silent
|
|
7
|
+
misses, and — after one bounded revision round — turns surviving objections into visible flags.
|
|
8
|
+
It is a challenger, never a judge: verdicts still come only from the deterministic oracle, and
|
|
9
|
+
with the Twin disabled every turn behaves exactly as before.
|
|
10
|
+
"""
|
|
11
|
+
from .contracts import Concern, Coverage, Objection, parse_coverage
|
|
12
|
+
from .dialectic import (
|
|
13
|
+
COVERAGE_SCHEMA, Twin, TwinContext, TwinResult, concern_context, coverage_instruction,
|
|
14
|
+
)
|
|
15
|
+
from .authoring import authoring_concerns
|
|
16
|
+
from .harness import GoldenTurn, HarnessReport, replay
|
|
17
|
+
from .justify import Adjudication, adjudicate, classify, state_holds
|
|
18
|
+
from .learner import learner_concerns
|
|
19
|
+
from .mission import mission_concerns
|
|
20
|
+
from .os_coach import coach_concerns, fallback_text, focus_line
|
|
21
|
+
from .salience import MAX_CONCERNS, MUST_ADDRESS, cap
|
|
22
|
+
|
|
23
|
+
__all__ = ["Concern", "Coverage", "Objection", "parse_coverage", "Twin", "TwinContext",
|
|
24
|
+
"TwinResult", "concern_context", "coverage_instruction", "COVERAGE_SCHEMA",
|
|
25
|
+
"Adjudication", "adjudicate", "classify", "state_holds", "mission_concerns",
|
|
26
|
+
"coach_concerns", "focus_line", "fallback_text", "authoring_concerns",
|
|
27
|
+
"learner_concerns", "GoldenTurn", "HarnessReport", "replay",
|
|
28
|
+
"MAX_CONCERNS", "MUST_ADDRESS", "cap"]
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
"""The authoring/compose concern enumerator (REASONING_2.0_DESIGN.md §5, third surface).
|
|
2
|
+
|
|
3
|
+
These concerns run at RATIFY time, teacher-facing: deterministic observations about the gap
|
|
4
|
+
between what the teacher's words said and what the model picked — "the words strongly match Y
|
|
5
|
+
but X was chosen; why not Y?" — plus a check that stated DON'Ts really were honoured. The
|
|
6
|
+
ratify UI renders them as questions beside the Proposal; nothing here blocks or judges."""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from .contracts import Concern
|
|
10
|
+
from .salience import cap
|
|
11
|
+
|
|
12
|
+
_DISAGREE_MARGIN = 3 # the lexical-override threshold lesson_resolver already uses
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def authoring_concerns(intent_text: str, proposal) -> list[Concern]:
|
|
16
|
+
"""Concerns about a compose/resolve Proposal, for the ratify surface. Duck-typed against
|
|
17
|
+
`lesson_resolver.Proposal` (archetype_id, lesson, infeasible, suppressed)."""
|
|
18
|
+
concerns: list[Concern] = []
|
|
19
|
+
|
|
20
|
+
# 1) strong lexical disagreement: the teacher's words match a different archetype hard.
|
|
21
|
+
try:
|
|
22
|
+
from ..lesson_resolver import lexical_scores
|
|
23
|
+
scores = lexical_scores(intent_text)
|
|
24
|
+
if scores:
|
|
25
|
+
top_id, top_score = scores[0]
|
|
26
|
+
if top_id != getattr(proposal, "archetype_id", "") and top_score >= _DISAGREE_MARGIN:
|
|
27
|
+
concerns.append(Concern(
|
|
28
|
+
id=f"authoring:lexical-disagreement:{top_id}",
|
|
29
|
+
kind="composition-gap",
|
|
30
|
+
statement=(f"the request's words strongly match '{top_id}' but "
|
|
31
|
+
f"'{proposal.archetype_id}' was chosen — why not {top_id}?"),
|
|
32
|
+
evidence=f"lexical score {top_score} for {top_id} (defining-field overlap)",
|
|
33
|
+
salience=2, source="lesson_resolver.lexical_scores"))
|
|
34
|
+
except Exception:
|
|
35
|
+
pass
|
|
36
|
+
|
|
37
|
+
# 2) stated DON'Ts must be honoured in the assembled lesson (deterministic negation scan
|
|
38
|
+
# against the elements the lesson actually stages/grades).
|
|
39
|
+
try:
|
|
40
|
+
from ...domain import constraints as _con
|
|
41
|
+
excluded = set(getattr(_con.from_text(intent_text), "exclude", ()) or ())
|
|
42
|
+
lesson = getattr(proposal, "lesson", None)
|
|
43
|
+
if excluded and lesson is not None:
|
|
44
|
+
staged = {getattr(d, "type_key", "") for d in
|
|
45
|
+
getattr(getattr(lesson, "stage", None), "values", lambda: [])()} \
|
|
46
|
+
if hasattr(getattr(lesson, "stage", None), "values") else set()
|
|
47
|
+
violated = sorted(excluded & staged)
|
|
48
|
+
if violated:
|
|
49
|
+
concerns.append(Concern(
|
|
50
|
+
id="authoring:exclusion-violated", kind="composition-gap",
|
|
51
|
+
statement=("the request excluded " + ", ".join(violated) +
|
|
52
|
+
" but the proposal stages them"),
|
|
53
|
+
evidence=f"negation scan: exclude={sorted(excluded)}; staged∩={violated}",
|
|
54
|
+
salience=3, source="constraints.from_text"))
|
|
55
|
+
except Exception:
|
|
56
|
+
pass
|
|
57
|
+
|
|
58
|
+
# 3) infeasibility / suppression the resolver already computed — surfaced as concerns so the
|
|
59
|
+
# ratify UI shows them in the same voice.
|
|
60
|
+
if getattr(proposal, "infeasible", ""):
|
|
61
|
+
concerns.append(Concern(
|
|
62
|
+
id="authoring:infeasible", kind="composition-gap",
|
|
63
|
+
statement=f"the request is infeasible as stated: {proposal.infeasible}",
|
|
64
|
+
evidence=proposal.infeasible, salience=3, source="lesson_resolver"))
|
|
65
|
+
if getattr(proposal, "suppressed", ""):
|
|
66
|
+
concerns.append(Concern(
|
|
67
|
+
id="authoring:suppressed", kind="composition-gap",
|
|
68
|
+
statement=f"something was left out to honour the DON'Ts: {proposal.suppressed}",
|
|
69
|
+
evidence=proposal.suppressed, salience=2, source="lesson_resolver"))
|
|
70
|
+
|
|
71
|
+
return cap(concerns)
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Twin contracts — the Concern (a point that matters), the Coverage report (what the model says
|
|
2
|
+
it addressed), and the Objection (the Twin's challenge). Plain dataclasses; no Qt, no LLM.
|
|
3
|
+
|
|
4
|
+
A Concern is only ever emitted with `evidence` — a deterministic ground fact from the substrate.
|
|
5
|
+
The Twin can only cite what GINI can prove (REASONING_2.0_DESIGN.md §7)."""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass(frozen=True)
|
|
12
|
+
class Concern:
|
|
13
|
+
id: str # stable, e.g. "objective:web-reach", "legality:off_task"
|
|
14
|
+
kind: str # objective | legality | grammar-option | watcher-event | composition-gap
|
|
15
|
+
statement: str # human-readable: what matters ("'Wire the LAN to a router' is unmet")
|
|
16
|
+
evidence: str # the deterministic ground fact backing it (explain why / verdict data)
|
|
17
|
+
salience: int = 1 # 0..3 (rules-only, twin/salience.py); >= MUST_ADDRESS -> must be covered
|
|
18
|
+
source: str = "" # which substrate produced it (debugging / eval)
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Coverage:
|
|
23
|
+
"""The model's self-report, used ONLY as an index for the exact diff — never trusted as
|
|
24
|
+
truth (omission justifications get validated; false 'addressed' claims are the eval
|
|
25
|
+
harness's and verify_claims' business)."""
|
|
26
|
+
addressed: frozenset = field(default_factory=frozenset) # concern ids
|
|
27
|
+
omitted: dict = field(default_factory=dict) # concern id -> one-line why
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def parse_coverage(obj) -> Coverage | None:
|
|
31
|
+
"""The `coverage` member of the persona's JSON reply -> Coverage, or None when absent or
|
|
32
|
+
malformed (-> the coverage-silent posture; must never raise)."""
|
|
33
|
+
if not isinstance(obj, dict):
|
|
34
|
+
return None
|
|
35
|
+
addressed = obj.get("addressed")
|
|
36
|
+
omitted_raw = obj.get("omitted")
|
|
37
|
+
if not isinstance(addressed, list) and not isinstance(omitted_raw, list):
|
|
38
|
+
return None
|
|
39
|
+
omitted: dict = {}
|
|
40
|
+
for entry in omitted_raw if isinstance(omitted_raw, list) else []:
|
|
41
|
+
if isinstance(entry, dict) and entry.get("id"):
|
|
42
|
+
omitted[str(entry["id"])] = str(entry.get("why", ""))
|
|
43
|
+
elif isinstance(entry, str) and entry:
|
|
44
|
+
omitted[entry] = ""
|
|
45
|
+
return Coverage(
|
|
46
|
+
addressed=frozenset(str(a) for a in (addressed or []) if a),
|
|
47
|
+
omitted=omitted)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class Objection:
|
|
52
|
+
"""A challenge the Twin poses for a silently-missed (or unjustified) concern."""
|
|
53
|
+
concern: Concern
|
|
54
|
+
question: str
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
"""The dialectic — the Twin's audit of one reasoning turn (REASONING_2.0_DESIGN.md §3.4).
|
|
2
|
+
|
|
3
|
+
Deterministic control flow throughout: an EXACT set diff of the model's coverage report against
|
|
4
|
+
the must-address concerns; ADJUDICATION of every claimed omission against ground truth
|
|
5
|
+
(twin/justify.py — an objection is defeated only by a VALIDATED justification); template-
|
|
6
|
+
generated objections; a bounded dialectic (max_rounds revisions or the time budget, whichever
|
|
7
|
+
first) driven by the persona's existing `react(note=…)` mechanism; and surviving objections
|
|
8
|
+
turned into a visible flag appended to the move — never a silent ship, never suppression.
|
|
9
|
+
|
|
10
|
+
The Twin also keeps per-mission state: the HISTORY of concern ids the model actually covered
|
|
11
|
+
(what "already addressed" is checked against) and METRICS (phase-E raw material)."""
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import time
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
|
|
17
|
+
from .contracts import Concern, Coverage, Objection
|
|
18
|
+
from .justify import Adjudication, adjudicate
|
|
19
|
+
|
|
20
|
+
# The decoder-constrained reply shape for a covered reasoning turn (via Ollama structured
|
|
21
|
+
# outputs — see agent/llm/ollama.py `schema=`). text = the tutor line; coverage = the report.
|
|
22
|
+
COVERAGE_SCHEMA: dict = {
|
|
23
|
+
"type": "object",
|
|
24
|
+
"properties": {
|
|
25
|
+
"text": {"type": "string"},
|
|
26
|
+
"coverage": {
|
|
27
|
+
"type": "object",
|
|
28
|
+
"properties": {
|
|
29
|
+
"addressed": {"type": "array", "items": {"type": "string"}},
|
|
30
|
+
"omitted": {"type": "array", "items": {
|
|
31
|
+
"type": "object",
|
|
32
|
+
"properties": {"id": {"type": "string"}, "why": {"type": "string"}},
|
|
33
|
+
"required": ["id"],
|
|
34
|
+
}},
|
|
35
|
+
},
|
|
36
|
+
"required": ["addressed", "omitted"],
|
|
37
|
+
},
|
|
38
|
+
},
|
|
39
|
+
"required": ["text", "coverage"],
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def concern_context(concerns: list[Concern]) -> str:
|
|
44
|
+
"""Twin-as-context (phase B): the concern set injected UP FRONT into the grounding, so the
|
|
45
|
+
substrate's guaranteed recall shapes the DRAFT — the same enumeration that later audits it.
|
|
46
|
+
Statements + evidence only; what to do about them stays the model's call."""
|
|
47
|
+
if not concerns:
|
|
48
|
+
return ""
|
|
49
|
+
lines = [f"- {c.statement} ({c.evidence})" for c in concerns]
|
|
50
|
+
return "Things that matter right now (ground truth):\n" + "\n".join(lines)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def coverage_instruction(concerns: list[Concern]) -> str:
|
|
54
|
+
"""The reporting checklist appended to the persona task — the INDEX the exact diff runs
|
|
55
|
+
against (ids the reply must account for, addressed or justified)."""
|
|
56
|
+
lines = [f' {c.id}: {c.statement}' for c in concerns]
|
|
57
|
+
return (
|
|
58
|
+
"\nRespond as ONE JSON object: {\"text\": \"<your line to the student>\", \"coverage\": "
|
|
59
|
+
"{\"addressed\": [<concern ids your line addresses>], \"omitted\": [{\"id\": \"<concern "
|
|
60
|
+
"id>\", \"why\": \"<one line>\"}]}}. Every concern below must appear in either list; "
|
|
61
|
+
"omitting is fine WITH a reason (e.g. it would give the answer away, or it is off this "
|
|
62
|
+
"question's topic).\nConcerns:\n" + "\n".join(lines))
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _question(c: Concern) -> str:
|
|
66
|
+
if c.kind == "legality":
|
|
67
|
+
return (f"You did not address this: {c.statement}. It is an active rule violation "
|
|
68
|
+
f"({c.evidence}) — why not?")
|
|
69
|
+
return (f"You did not address this: {c.statement} ({c.evidence}). "
|
|
70
|
+
"Is it OK to leave that out? If it belongs in your line, work it in.")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _rejected_question(c: Concern, why: str, adj: Adjudication) -> str:
|
|
74
|
+
"""The objection for an omission whose justification failed adjudication — it tells the
|
|
75
|
+
model exactly why its excuse doesn't hold (the ground the Twin checked)."""
|
|
76
|
+
return (f"You left out {c.statement!r} saying {why!r}, but {adj.reason}. "
|
|
77
|
+
"Address it, or give a reason that actually holds.")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass
|
|
81
|
+
class TwinContext:
|
|
82
|
+
"""What omission adjudication needs from the turn (built by the orchestrator)."""
|
|
83
|
+
move_kind: str = "say"
|
|
84
|
+
utterance: str = ""
|
|
85
|
+
world: object = None # the live board (for state-claim checks)
|
|
86
|
+
history: set = field(default_factory=set) # concern ids covered in prior turns
|
|
87
|
+
translate: object = None # callable(why) -> predicate | None (LLM-backed)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass
|
|
91
|
+
class TwinResult:
|
|
92
|
+
"""What one audit produced — for tests, metrics (phase E), and the instructor log."""
|
|
93
|
+
concerns: tuple = ()
|
|
94
|
+
coverage_silent: bool = False
|
|
95
|
+
objections: tuple = () # first-round objections
|
|
96
|
+
surviving: tuple = () # objections still standing after the revision round(s)
|
|
97
|
+
accepted_omissions: dict = field(default_factory=dict) # id -> VALIDATED why
|
|
98
|
+
rejected_omissions: dict = field(default_factory=dict) # id -> why the justification failed
|
|
99
|
+
rounds: int = 0
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
class Twin:
|
|
103
|
+
"""The Reasoning Twin's audit half. `audit` = exact diff + omission adjudication (phase B);
|
|
104
|
+
`note`/`flag` are the two ways an objection re-enters the turn. Per-mission state: covered-
|
|
105
|
+
concern history + metrics. Bounds: `max_rounds` revisions or `budget_s`, whichever first."""
|
|
106
|
+
|
|
107
|
+
def __init__(self, max_rounds: int = 2, budget_s: float = 4.0) -> None:
|
|
108
|
+
self.max_rounds = max_rounds
|
|
109
|
+
self.budget_s = budget_s
|
|
110
|
+
self.history: set = set() # concern ids the model actually covered, ever
|
|
111
|
+
self.metrics: dict = {"turns": 0, "coverage_silent": 0, "objections": 0,
|
|
112
|
+
"defeats": 0, "revisions": 0, "flags": 0}
|
|
113
|
+
self.last_adjudications: dict = {} # id -> Adjudication (of the most recent audit)
|
|
114
|
+
|
|
115
|
+
def diff(self, concerns: list[Concern], coverage: Coverage | None) -> list[Objection]:
|
|
116
|
+
"""Exact diff of the coverage report against the must-address concerns — the SILENT
|
|
117
|
+
misses only (omissions are the `audit` step's business).
|
|
118
|
+
|
|
119
|
+
Coverage-silent (no report — model without schema support, or the offline fallback):
|
|
120
|
+
don't guess what the prose covered; object only about the URGENT tier (salience 3),
|
|
121
|
+
so a degraded model still surfaces rule violations without nagging about the rest."""
|
|
122
|
+
from .salience import MUST_ADDRESS
|
|
123
|
+
if coverage is None:
|
|
124
|
+
return [Objection(c, _question(c)) for c in concerns if c.salience >= 3]
|
|
125
|
+
seen = coverage.addressed | set(coverage.omitted)
|
|
126
|
+
return [Objection(c, _question(c))
|
|
127
|
+
for c in concerns
|
|
128
|
+
if c.salience >= MUST_ADDRESS and c.id not in seen]
|
|
129
|
+
|
|
130
|
+
def audit(self, concerns: list[Concern], coverage: Coverage | None,
|
|
131
|
+
ctx: TwinContext | None = None) -> list[Objection]:
|
|
132
|
+
"""One audit round: silent misses (the diff) PLUS every claimed omission adjudicated
|
|
133
|
+
against ground truth. A justification defeats its objection only if it VALIDATES."""
|
|
134
|
+
from .salience import MUST_ADDRESS
|
|
135
|
+
objections = self.diff(concerns, coverage)
|
|
136
|
+
self.last_adjudications = {}
|
|
137
|
+
if coverage is None or ctx is None:
|
|
138
|
+
return objections
|
|
139
|
+
by_id = {c.id: c for c in concerns}
|
|
140
|
+
for cid, why in coverage.omitted.items():
|
|
141
|
+
c = by_id.get(cid)
|
|
142
|
+
if c is None or c.salience < MUST_ADDRESS:
|
|
143
|
+
continue
|
|
144
|
+
adj = adjudicate(c, why, ctx)
|
|
145
|
+
self.last_adjudications[cid] = adj
|
|
146
|
+
if adj.valid:
|
|
147
|
+
self.metrics["defeats"] += 1
|
|
148
|
+
else:
|
|
149
|
+
objections.append(Objection(c, _rejected_question(c, why, adj)))
|
|
150
|
+
return objections
|
|
151
|
+
|
|
152
|
+
def record(self, coverage: Coverage | None) -> None:
|
|
153
|
+
"""Fold the turn's genuinely-covered concern ids into the mission history (what a later
|
|
154
|
+
'already addressed' justification is checked against). Omissions do NOT count."""
|
|
155
|
+
if coverage is not None:
|
|
156
|
+
self.history |= set(coverage.addressed)
|
|
157
|
+
|
|
158
|
+
def split_omissions(self, concerns: list[Concern], coverage: Coverage | None) -> tuple:
|
|
159
|
+
"""(accepted, rejected) omission maps from the last audit's adjudications."""
|
|
160
|
+
if coverage is None:
|
|
161
|
+
return {}, {}
|
|
162
|
+
ids = {c.id for c in concerns}
|
|
163
|
+
accepted, rejected = {}, {}
|
|
164
|
+
for cid, why in coverage.omitted.items():
|
|
165
|
+
if cid not in ids:
|
|
166
|
+
continue
|
|
167
|
+
adj = self.last_adjudications.get(cid)
|
|
168
|
+
if adj is not None and not adj.valid:
|
|
169
|
+
rejected[cid] = adj.reason or why
|
|
170
|
+
else:
|
|
171
|
+
accepted[cid] = why
|
|
172
|
+
return accepted, rejected
|
|
173
|
+
|
|
174
|
+
@staticmethod
|
|
175
|
+
def note(objections: list[Objection]) -> str:
|
|
176
|
+
"""One revision note carrying every objection (batched — one round trip, not N)."""
|
|
177
|
+
return " ".join(o.question for o in objections)
|
|
178
|
+
|
|
179
|
+
@staticmethod
|
|
180
|
+
def flag(move, objections: list[Objection]):
|
|
181
|
+
"""Surviving objections become a visible, clearly-separated addendum on the move (the
|
|
182
|
+
ratified game-master surfacing: append). The Twin's voice is the concern STATEMENT —
|
|
183
|
+
grounded substrate text, not model prose."""
|
|
184
|
+
if not objections:
|
|
185
|
+
return move
|
|
186
|
+
worth = "; ".join(o.concern.statement for o in objections)
|
|
187
|
+
from ..contracts import Move
|
|
188
|
+
return Move(kind=move.kind, text=(move.text + f"\n(Also worth a look: {worth}.)").strip(),
|
|
189
|
+
refs=move.refs, claims=move.claims)
|
|
@@ -0,0 +1,93 @@
|
|
|
1
|
+
"""The reasoning eval harness (REASONING_2.0_DESIGN.md phase E).
|
|
2
|
+
|
|
3
|
+
Replays a bank of GOLDEN TURNS — (world setup, a scripted model, a trigger, expectations) —
|
|
4
|
+
through the real MissionAgent+Twin and aggregates the metrics the design names: addressed-rate
|
|
5
|
+
of must-address concerns, coverage-silence rate, false-objection rate, flag rate. The harness
|
|
6
|
+
tests the TWIN deterministically (scripted models make every run reproducible) and gives prompt/
|
|
7
|
+
model changes a regression gate: run the same bank, compare the report.
|
|
8
|
+
|
|
9
|
+
A `false objection` is an objection raised on a turn whose golden expectation says the model's
|
|
10
|
+
coverage was complete/justified — i.e. the Twin nagged when it shouldn't have. With scripted
|
|
11
|
+
models this is fully deterministic, so the false-objection rate here is a check on the TWIN's
|
|
12
|
+
rules; against a live model (Mac-side) the same report measures the MODEL instead."""
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from dataclasses import dataclass, field
|
|
16
|
+
|
|
17
|
+
from .salience import MUST_ADDRESS
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass
|
|
21
|
+
class GoldenTurn:
|
|
22
|
+
"""One replayable turn: `make_agent(llm)` builds the MissionAgent (world + twin inside);
|
|
23
|
+
`llm` is the scripted model; `trigger`/`utterance` wake it. Expectations are about the
|
|
24
|
+
TWIN's behavior, not prose."""
|
|
25
|
+
name: str
|
|
26
|
+
make_agent: object # callable(llm) -> MissionAgent (with a Twin)
|
|
27
|
+
llm: object # scripted callable(prompt) -> str
|
|
28
|
+
trigger: object = None
|
|
29
|
+
utterance: str = ""
|
|
30
|
+
expect_flags: bool = False # should the turn end with a visible flag?
|
|
31
|
+
expect_objections: bool = False # should the FIRST audit raise objections?
|
|
32
|
+
expect_clean: bool = False # golden says coverage was complete/justified
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class TurnOutcome:
|
|
37
|
+
name: str
|
|
38
|
+
ok: bool
|
|
39
|
+
flags: bool
|
|
40
|
+
objections: int
|
|
41
|
+
surviving: int
|
|
42
|
+
rounds: int
|
|
43
|
+
coverage_silent: bool
|
|
44
|
+
false_objection: bool
|
|
45
|
+
notes: str = ""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class HarnessReport:
|
|
50
|
+
outcomes: list = field(default_factory=list)
|
|
51
|
+
|
|
52
|
+
@property
|
|
53
|
+
def passed(self) -> bool:
|
|
54
|
+
return all(o.ok for o in self.outcomes)
|
|
55
|
+
|
|
56
|
+
def metrics(self) -> dict:
|
|
57
|
+
n = len(self.outcomes) or 1
|
|
58
|
+
addressed = [o for o in self.outcomes if not o.surviving]
|
|
59
|
+
return {
|
|
60
|
+
"turns": len(self.outcomes),
|
|
61
|
+
"pass_rate": sum(o.ok for o in self.outcomes) / n,
|
|
62
|
+
"addressed_rate": len(addressed) / n, # must-address fully covered by turn end
|
|
63
|
+
"silence_rate": sum(o.coverage_silent for o in self.outcomes) / n,
|
|
64
|
+
"false_objection_rate": sum(o.false_objection for o in self.outcomes) / n,
|
|
65
|
+
"flag_rate": sum(o.flags for o in self.outcomes) / n,
|
|
66
|
+
"mean_rounds": sum(o.rounds for o in self.outcomes) / n,
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def replay(turns: list[GoldenTurn]) -> HarnessReport:
|
|
71
|
+
"""Run every golden turn through the real orchestration; never raises — a broken turn is a
|
|
72
|
+
failed outcome with a note, so the CI report always renders."""
|
|
73
|
+
report = HarnessReport()
|
|
74
|
+
for t in turns:
|
|
75
|
+
try:
|
|
76
|
+
agent = t.make_agent(t.llm)
|
|
77
|
+
move = agent.turn(t.trigger, utterance=t.utterance)
|
|
78
|
+
res = agent.last_twin_result
|
|
79
|
+
flags = "Also worth a look" in (move.text or "")
|
|
80
|
+
objections = len(res.objections) if res else 0
|
|
81
|
+
surviving = len(res.surviving) if res else 0
|
|
82
|
+
silent = bool(res and res.coverage_silent)
|
|
83
|
+
rounds = res.rounds if res else 0
|
|
84
|
+
false_obj = bool(t.expect_clean and objections)
|
|
85
|
+
ok = (flags == t.expect_flags
|
|
86
|
+
and (objections > 0) == t.expect_objections
|
|
87
|
+
and not false_obj)
|
|
88
|
+
report.outcomes.append(TurnOutcome(
|
|
89
|
+
t.name, ok, flags, objections, surviving, rounds, silent, false_obj))
|
|
90
|
+
except Exception as e: # noqa: BLE001 — report, don't die
|
|
91
|
+
report.outcomes.append(TurnOutcome(
|
|
92
|
+
t.name, False, False, 0, 0, 0, False, False, notes=f"error: {e}"))
|
|
93
|
+
return report
|