specmodule 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
module_harness/feed.py ADDED
@@ -0,0 +1,197 @@
1
+ """stdlib 可视化开关:零第三方依赖的极简运行 feed(roadmap 独立线)。
2
+
3
+ 用 ``http.server`` 起一个只读 HTTP 服务,把已落盘的运行数据以 JSON feed
4
+ 暴露给浏览器/脚本轮询查看——运行中即可看(每 tick 落盘 run.sqlite),
5
+ 运行后完整看。数据组合全部走共享查询层(``module_harness.query`` /
6
+ ``status.query_run_status``),本文件只做 HTTP 适配,不重实现查询逻辑。
7
+
8
+ 富交互编辑器/完整 Web UX 属于生态项目 ``SpecModule_webview``;本开关只
9
+ 提供"极简可见"的最低形态,供库使用者零依赖快速查看。
10
+
11
+ 用法(CLI 接线见 cli.py ``feed`` 子命令)::
12
+
13
+ specmodule feed [--host 127.0.0.1] [--port 8000] [--run-id <id>]
14
+
15
+ 端点:
16
+ GET / 极简 HTML 页面(原生 JS 每 2s 轮询 feed)
17
+ GET /feed.json?run_id=... 组合 JSON:status / timeline / checkpoints
18
+ (缺省 run_id = 最新修改的运行)
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import json
24
+ import logging
25
+ from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
26
+ from pathlib import Path
27
+ from urllib.parse import parse_qs, urlparse
28
+
29
+ from .query import (
30
+ build_checkpoints,
31
+ build_timeline,
32
+ checkpoints_to_dict,
33
+ timeline_to_dict,
34
+ )
35
+ from .status import query_run_status
36
+
37
+ log = logging.getLogger(__name__)
38
+
39
+ _HTML = """<!DOCTYPE html>
40
+ <html lang="zh">
41
+ <head>
42
+ <meta charset="utf-8">
43
+ <title>SpecModule 运行 feed</title>
44
+ <style>
45
+ body { font-family: ui-monospace, Consolas, monospace; margin: 2em; }
46
+ h1 { font-size: 1.2em; }
47
+ .phase { font-weight: bold; }
48
+ table { border-collapse: collapse; width: 100%; }
49
+ td, th { border: 1px solid #ccc; padding: 4px 8px; text-align: left; font-size: 0.9em; }
50
+ .err { color: #c00; }
51
+ .ok { color: #080; }
52
+ </style>
53
+ </head>
54
+ <body>
55
+ <h1>SpecModule 运行 feed</h1>
56
+ <p>run_id: <span id="run-id">?</span> · 阶段: <span id="phase" class="phase">?</span>
57
+ · tick: <span id="tick">?</span> · 刷新: <span id="stamp">?</span></p>
58
+ <div id="outputs"></div>
59
+ <h2>时间线</h2>
60
+ <table id="timeline"><tr><th>tick</th><th>节点</th><th>状态</th><th>输出/错误</th></tr></table>
61
+ <h2>检查点</h2>
62
+ <table id="checkpoints"><tr><th>tick</th><th>类型</th><th>fired 节点</th></tr></table>
63
+ <script>
64
+ async function refresh() {
65
+ try {
66
+ const r = await fetch("feed.json" + location.search);
67
+ const d = await r.json();
68
+ const st = d.status;
69
+ document.getElementById("run-id").textContent = d.run_id;
70
+ document.getElementById("phase").textContent = st ? st.phase : "无运行";
71
+ document.getElementById("tick").textContent = st ? (st.tick ?? "—") : "—";
72
+ document.getElementById("stamp").textContent = new Date().toLocaleTimeString();
73
+ let oh = "";
74
+ if (st && st.outputs) {
75
+ for (const [node, out] of Object.entries(st.outputs)) {
76
+ oh += "<p><b>" + node + "</b>: "
77
+ + (typeof out === "string" ? out : JSON.stringify(out)) + "</p>";
78
+ }
79
+ }
80
+ document.getElementById("outputs").innerHTML = oh;
81
+ const tl = document.getElementById("timeline");
82
+ while (tl.rows.length > 1) tl.deleteRow(1);
83
+ for (const e of (d.timeline ? d.timeline.entries : [])) {
84
+ const row = tl.insertRow();
85
+ row.insertCell().textContent = e.tick;
86
+ row.insertCell().textContent = e.node;
87
+ row.insertCell().textContent = e.status || (e.error ? "✗" : "✓");
88
+ row.insertCell().textContent = e.error || (e.output ? JSON.stringify(e.output) : "");
89
+ }
90
+ const cp = document.getElementById("checkpoints");
91
+ while (cp.rows.length > 1) cp.deleteRow(1);
92
+ for (const e of (d.checkpoints ? d.checkpoints.checkpoints : [])) {
93
+ const row = cp.insertRow();
94
+ row.insertCell().textContent = e.tick;
95
+ row.insertCell().textContent = e.label ? e.label : "tick";
96
+ row.insertCell().textContent = (e.fired || []).join(", ");
97
+ }
98
+ } catch (err) {
99
+ document.getElementById("phase").textContent = "feed 读取失败: " + err;
100
+ }
101
+ }
102
+ setInterval(refresh, 2000);
103
+ refresh();
104
+ </script>
105
+ </body>
106
+ </html>
107
+ """
108
+
109
+
110
+ class RunFeedHandler(BaseHTTPRequestHandler):
111
+ """只读 feed:/ 为页面,/feed.json 为 JSON 组合。"""
112
+
113
+ server: "RunFeedServer" # type: ignore[misc] # ThreadingHTTPServer 实例
114
+
115
+ def do_GET(self) -> None: # noqa: N802(http.server 命名约定)
116
+ parsed = urlparse(self.path)
117
+ if parsed.path == "/feed.json":
118
+ self._serve_feed(parsed)
119
+ elif parsed.path in ("/", "/index.html"):
120
+ self._serve_html()
121
+ else:
122
+ self.send_error(404, "not found")
123
+
124
+ # -- feed -----------------------------------------------------------
125
+
126
+ def _serve_feed(self, parsed) -> None:
127
+ qs = parse_qs(parsed.query)
128
+ run_id = (qs.get("run_id") or [None])[0] or self.server.latest_run_id()
129
+ base = self.server.base_dir
130
+ if run_id is None:
131
+ self._send_json({"run_id": None, "error": "无运行记录"}, 404)
132
+ return
133
+
134
+ status = query_run_status(run_id, base_dir=base)
135
+ timeline = build_timeline(run_id, base_dir=base)
136
+ checkpoints = build_checkpoints(run_id, base_dir=base)
137
+ if status is None and timeline is None:
138
+ self._send_json({"run_id": run_id, "error": "无运行记录"}, 404)
139
+ return
140
+ self._send_json({
141
+ "run_id": run_id,
142
+ "status": status and {
143
+ "phase": status.phase,
144
+ "tick": status.tick,
145
+ "fired": status.fired,
146
+ "outputs": status.outputs,
147
+ "error": status.error,
148
+ },
149
+ "timeline": timeline_to_dict(timeline) if timeline else None,
150
+ "checkpoints": checkpoints_to_dict(checkpoints) if checkpoints else None,
151
+ })
152
+
153
+ def _serve_html(self) -> None:
154
+ body = _HTML.encode("utf-8")
155
+ self.send_response(200)
156
+ self.send_header("Content-Type", "text/html; charset=utf-8")
157
+ self.send_header("Content-Length", str(len(body)))
158
+ self.end_headers()
159
+ self.wfile.write(body)
160
+
161
+ def _send_json(self, data: dict, code: int = 200) -> None:
162
+ body = json.dumps(data, ensure_ascii=False).encode("utf-8")
163
+ self.send_response(code)
164
+ self.send_header("Content-Type", "application/json; charset=utf-8")
165
+ self.send_header("Content-Length", str(len(body)))
166
+ self.end_headers()
167
+ self.wfile.write(body)
168
+
169
+ def log_message(self, fmt: str, *args) -> None: # noqa: A003
170
+ log.info("%s - %s", self.address_string(), fmt % args)
171
+
172
+
173
+ class RunFeedServer(ThreadingHTTPServer):
174
+ """极简运行 feed 服务(零第三方依赖,stdlib http.server)。"""
175
+
176
+ daemon_threads = True
177
+
178
+ def __init__(
179
+ self,
180
+ server_address: tuple[str, int],
181
+ base_dir: Path | None = None,
182
+ ) -> None:
183
+ self.base_dir = base_dir
184
+ super().__init__(server_address, RunFeedHandler)
185
+
186
+ def latest_run_id(self) -> str | None:
187
+ """与 CLI ``_latest_run_id`` 同语义:扫描 runs/ 取最新修改子目录。"""
188
+ runs = (self.base_dir or Path.cwd()) / ".specmodule" / "runs"
189
+ if not runs.is_dir():
190
+ return None
191
+ candidates = [
192
+ p for p in runs.iterdir()
193
+ if p.is_dir() and (p / "status.json").exists()
194
+ ]
195
+ if not candidates:
196
+ return None
197
+ return max(candidates, key=lambda p: p.stat().st_mtime).name
@@ -0,0 +1,334 @@
1
+ # module_harness/graph_builder.py
2
+ """Tasklist -> tickflow Graph translator.
3
+
4
+ Translates a :class:`Tasklist` into a :class:`Graph` with namespace-isolated
5
+ body registrations. Each Task's body is registered as ``{module_id}:{key}``
6
+ so modules with overlapping task keys do not collide.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import json
12
+ from typing import Any
13
+
14
+ from tickflow import Failure, Graph, parse as parse_graph
15
+ from tickflow.ir import InputPolicy
16
+
17
+ from .config import HarnessConfig
18
+ from .outputfmt import OutputFormat
19
+ from .registry import HarnessRegistry
20
+ from .spec import SpecValidationError, TaskDefinition, Tasklist
21
+ from .translator import prepare_flow, _suppress_parse_noise
22
+
23
+
24
+ _CONSTANT_TOKENS = frozenset({"{spec}", "{tasklist}", "{node}"})
25
+
26
+
27
+ def _is_constant_ref(producer: Any) -> bool:
28
+ """是否为注册时解析的常量引用({spec}/{tasklist}/{node} 或 {spec.xxx})。"""
29
+ return (
30
+ isinstance(producer, str)
31
+ and (producer in _CONSTANT_TOKENS or producer.startswith("{spec."))
32
+ )
33
+
34
+
35
+ class TasklistTranslator:
36
+ """Translates a :class:`Tasklist` into a parsed :class:`Graph` paired with
37
+ a :class:`HarnessRegistry` that contains all required bodies.
38
+
39
+ Usage::
40
+
41
+ builder = TasklistTranslator(registry, module_id="my_mod")
42
+ graph, out_reg = builder.build(tasklist)
43
+ """
44
+
45
+ def __init__(
46
+ self,
47
+ registry: HarnessRegistry,
48
+ module_id: str,
49
+ *,
50
+ modules: dict[str, Any] | None = None,
51
+ llm_client: Any = None,
52
+ ) -> None:
53
+ self.reg = registry
54
+ self.module_id = module_id
55
+ self.modules = dict(modules or {})
56
+ self._llm_client = llm_client
57
+
58
+ # ------------------------------------------------------------------
59
+ # Public API
60
+ # ------------------------------------------------------------------
61
+
62
+ def build(self, tasklist: Tasklist, spec: Any | None = None) -> tuple[Graph, HarnessRegistry]:
63
+ """Iterate tasks, register bodies, parse flow, attach body names.
64
+
65
+ ``spec``:可选,用于解析 task 中 ``{spec.xxx}`` 字段引用。
66
+ Returns (graph, registry) where *registry* is ``self.reg`` (the same
67
+ object that was passed to the constructor, now populated with the
68
+ isolated body entries).
69
+ """
70
+ spec_dict = spec.to_dict() if spec is not None else {}
71
+ tasklist_dict = tasklist.to_dict()
72
+
73
+ # 1. Register every task's body under an isolated name.
74
+ for key, task in tasklist.tasks.items():
75
+ self._register_body(key, task, spec_dict, tasklist_dict)
76
+
77
+ # 2. Prepare the flow text for tickflow's parser (no body/input
78
+ # declarations -- those are attached programmatically below
79
+ # because tickflow's DSL does not support ``:`` in body names).
80
+ flow_text = prepare_flow(tasklist.flow)
81
+
82
+ # 3. Parse into a Graph. The parser validates that guard names
83
+ # and default placeholder bodies exist in the registry.
84
+ with _suppress_parse_noise():
85
+ graph = parse_graph(flow_text, registry=self.reg)
86
+
87
+ # 4. Assign the real (colon-scoped) body name to each graph node.
88
+ for key, task in tasklist.tasks.items():
89
+ isolated_name = self._isolated(key)
90
+ graph.nodes[key].body = isolated_name
91
+
92
+ # 5. Wire task inputs to graph node inputs.
93
+ # ``task.inputs = {field_name: producer}`` — 同时注册 field 名与
94
+ # producer 名两个 key:body 用 ``view.<producer>.value``(有值),
95
+ # ``view.<field_name>.value`` 为 Missing(不崩溃,脚本旧写法兼容)。
96
+ # harness 的 prompt 占位符 ``{field}`` 经 _register_harness 传入的
97
+ # input_aliases 在运行时解析 producer 输出值。
98
+ # 常量引用({spec}/{tasklist}/{node}、{spec.xxx})已由
99
+ # _register_harness 解析为 spec_inputs,此处跳过。
100
+ for key, task in tasklist.tasks.items():
101
+ if task.inputs:
102
+ for field_name, producer in task.inputs.items():
103
+ if _is_constant_ref(producer):
104
+ continue
105
+ graph.nodes[key].inputs[field_name] = InputPolicy.latest()
106
+ if producer != field_name:
107
+ graph.nodes[key].inputs[producer] = InputPolicy.latest()
108
+
109
+ return graph, self.reg
110
+
111
+ # ------------------------------------------------------------------
112
+ # Body registration
113
+ # ------------------------------------------------------------------
114
+
115
+ def _isolated(self, key: str) -> str:
116
+ """Return the namespace-isolated body name for *key*."""
117
+ return f"{self.module_id}:{key}"
118
+
119
+ def _register_body(self, key: str, task: TaskDefinition,
120
+ spec_dict: dict[str, Any],
121
+ tasklist_dict: dict[str, Any]) -> None:
122
+ """Register one task's body in *self.reg* under an isolated name.
123
+
124
+ Delegates to the appropriate helper based on ``task.type``.
125
+ """
126
+ if task.type == "harness":
127
+ self._register_harness(key, task, spec_dict, tasklist_dict)
128
+ elif task.type == "script":
129
+ self._register_script(key, task)
130
+ elif task.type == "command":
131
+ self._register_command(key, task)
132
+ elif task.type == "submodule":
133
+ self._register_submodule(key, task, spec_dict, tasklist_dict)
134
+ else:
135
+ raise ValueError(f"Task '{key}': unknown type {task.type!r}")
136
+
137
+ @staticmethod
138
+ def _resolve_spec_ref(value: str, spec_dict: dict[str, Any]) -> Any:
139
+ """解析 "{spec.xxx}" 引用。非引用原样返回。"""
140
+ if isinstance(value, str) and value.startswith("{spec.") and value.endswith("}"):
141
+ return spec_dict.get(value[len("{spec."):-1])
142
+ return value
143
+
144
+ @staticmethod
145
+ def _resolve_constant(token: str, node_key: str,
146
+ spec_dict: dict[str, Any],
147
+ tasklist_dict: dict[str, Any]) -> Any:
148
+ """解析常量 token:{spec} → spec JSON,{tasklist} → tasklist JSON,
149
+ {node} → 当前节点 key。未知 token 抛 ValueError。"""
150
+ try:
151
+ if token == "{spec}":
152
+ return json.dumps(spec_dict, ensure_ascii=False)
153
+ if token == "{tasklist}":
154
+ return json.dumps(tasklist_dict, ensure_ascii=False)
155
+ except TypeError as e:
156
+ raise ValueError(
157
+ f"Task '{node_key}': token {token} 含不可 JSON 序列化的值: {e}"
158
+ ) from e
159
+ if token == "{node}":
160
+ return node_key
161
+ raise ValueError(f"未知常量 token: {token}")
162
+
163
+ def _register_harness(self, key: str, task: TaskDefinition,
164
+ spec_dict: dict[str, Any],
165
+ tasklist_dict: dict[str, Any]) -> None:
166
+ """Copy an existing harness config, apply task-level overrides, and
167
+ register under the isolated name."""
168
+ assert task.harness is not None # validated by spec
169
+ existing = self.reg.harness_config(task.harness)
170
+ if existing is None:
171
+ raise ValueError(
172
+ f"Task '{key}': harness '{task.harness}' not found. "
173
+ f"Make sure it was registered via reg.harness()."
174
+ )
175
+
176
+ # Build a new config with task-level overrides.
177
+ output_format: OutputFormat | None
178
+ if task.outputformat is not None:
179
+ output_format = OutputFormat(**task.outputformat)
180
+ else:
181
+ output_format = existing.output_format
182
+
183
+ api_params = dict(existing.api_params)
184
+ if task.api_params:
185
+ api_params.update(task.api_params)
186
+
187
+ cfg = HarnessConfig(
188
+ prompt_core=existing.prompt_core,
189
+ prompt_modes=dict(existing.prompt_modes),
190
+ output_format=output_format,
191
+ notdo=list(task.notdo) if task.notdo is not None else list(existing.notdo),
192
+ model=task.model if task.model is not None else existing.model,
193
+ temperature=(
194
+ task.temperature
195
+ if task.temperature is not None
196
+ else existing.temperature
197
+ ),
198
+ think=task.think if task.think is not None else existing.think,
199
+ api_params=api_params,
200
+ )
201
+
202
+ # 解析常量引用:promptmode 的 "{spec.xxx}" 与 inputs 的常量 token
203
+ # ({spec}/{tasklist}/{node} 与 {spec.xxx})均在注册时解析为字面值
204
+ promptmode = task.promptmode
205
+ if promptmode is not None:
206
+ promptmode = self._resolve_spec_ref(promptmode, spec_dict)
207
+
208
+ spec_inputs: dict[str, Any] = {}
209
+ if task.inputs:
210
+ for field_name, producer in task.inputs.items():
211
+ if isinstance(producer, str) and producer in _CONSTANT_TOKENS:
212
+ spec_inputs[field_name] = self._resolve_constant(
213
+ producer, key, spec_dict, tasklist_dict
214
+ )
215
+ elif isinstance(producer, str) and producer.startswith("{spec."):
216
+ resolved = self._resolve_spec_ref(producer, spec_dict)
217
+ spec_inputs[field_name] = resolved
218
+
219
+ # 跨节点输入别名:非常量引用的 inputs 把 field 名映射到 producer 节点,
220
+ # harness body 运行时据此把 producer 输出渲染进 prompt 的 {field} 占位符。
221
+ input_aliases: dict[str, str] = {}
222
+ if task.inputs:
223
+ for field_name, producer in task.inputs.items():
224
+ if _is_constant_ref(producer):
225
+ continue
226
+ input_aliases[field_name] = producer
227
+
228
+ self.reg.harness(
229
+ self._isolated(key),
230
+ cfg,
231
+ promptmode=promptmode,
232
+ prompt_extra=task.prompt,
233
+ spec_inputs=spec_inputs,
234
+ input_aliases=input_aliases,
235
+ )
236
+
237
+ def _register_script(self, key: str, task: TaskDefinition) -> None:
238
+ """Copy an existing script body and register under the isolated name."""
239
+ assert task.script is not None # validated by spec
240
+ if not self.reg.has_body(task.script):
241
+ raise ValueError(
242
+ f"Task '{key}': script '{task.script}' not found. "
243
+ f"Make sure it was registered via @reg.script()."
244
+ )
245
+ orig_body = self.reg.get_body(task.script)
246
+ self.reg.body(self._isolated(key), orig_body)
247
+
248
+ def _register_command(self, key: str, task: TaskDefinition) -> None:
249
+ """Copy an existing command config, apply task overrides, register under
250
+ the isolated name."""
251
+ assert task.command is not None # validated by spec
252
+ existing = self.reg.command_config(task.command)
253
+ if existing is None:
254
+ raise ValueError(
255
+ f"Task '{key}': command '{task.command}' not found. "
256
+ f"Make sure it was registered via reg.command()."
257
+ )
258
+
259
+ self.reg.command(
260
+ self._isolated(key),
261
+ existing,
262
+ timeout=task.timeout,
263
+ cwd=task.cwd,
264
+ )
265
+
266
+ def _register_submodule(self, key: str, task: TaskDefinition,
267
+ spec_dict: dict[str, Any],
268
+ tasklist_dict: dict[str, Any]) -> None:
269
+ """注册 submodule 节点 body:黑盒嵌入运行子模块(audit=False +
270
+ persist=False,不进审计/落盘),返回子流程终点输出。"""
271
+ assert task.submodule is not None # validated by spec
272
+ child_ref = self.modules.get(task.submodule)
273
+ if child_ref is None:
274
+ raise ValueError(
275
+ f"Task '{key}': submodule '{task.submodule}' not found. "
276
+ f"Make sure it was declared in modules."
277
+ )
278
+ if task.outputs:
279
+ for out_field, child_field in task.outputs.items():
280
+ if child_field not in child_ref.spec_schema.output:
281
+ raise ValueError(
282
+ f"Task '{key}': outputs 字段 '{child_field}' 不在 "
283
+ f"submodule '{task.submodule}' 的 spec_schema.output 中"
284
+ )
285
+
286
+ # 子模块实例:类 → 懒实例化(注入父 client);实例(loader 加载)→ 复用
287
+ if isinstance(child_ref, type):
288
+ child = child_ref(llm_client=self._llm_client)
289
+ else:
290
+ child = child_ref
291
+
292
+ # 节点级 LLM 覆盖 → 传播到子模块内部全部 harness
293
+ overrides: dict[str, Any] = {}
294
+ for k in ("model", "temperature", "think", "api_params"):
295
+ v = getattr(task, k)
296
+ if v is not None:
297
+ overrides[k] = v
298
+
299
+ # 常量引用({spec.xxx} / {spec}/{tasklist}/{node})构建期解析
300
+ const_inputs: dict[str, Any] = {}
301
+ if task.inputs:
302
+ for field_name, producer in task.inputs.items():
303
+ if isinstance(producer, str) and producer in _CONSTANT_TOKENS:
304
+ const_inputs[field_name] = self._resolve_constant(
305
+ producer, key, spec_dict, tasklist_dict
306
+ )
307
+ elif isinstance(producer, str) and producer.startswith("{spec."):
308
+ const_inputs[field_name] = self._resolve_spec_ref(
309
+ producer, spec_dict
310
+ )
311
+
312
+ async def body(view: Any) -> Any:
313
+ spec_input = dict(const_inputs)
314
+ if task.inputs:
315
+ for field_name, producer in task.inputs.items():
316
+ if _is_constant_ref(producer):
317
+ continue
318
+ spec_input[field_name] = view[producer].value
319
+ try:
320
+ firings = await child.run(
321
+ spec_input, audit=False, persist=False,
322
+ harness_overrides=overrides,
323
+ )
324
+ except SpecValidationError as e:
325
+ return Failure(
326
+ f"submodule '{task.submodule}' spec 校验失败: {e}",
327
+ type="infrastructure",
328
+ )
329
+ out = firings[-1].output if firings else {}
330
+ if task.outputs and isinstance(out, dict) and not isinstance(out, Failure):
331
+ out = {k: out.get(v) for k, v in task.outputs.items()}
332
+ return out
333
+
334
+ self.reg.body(self._isolated(key), body)