ltcai 9.9.3 → 9.9.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (81) hide show
  1. package/README.md +43 -38
  2. package/docs/CHANGELOG.md +65 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/ONBOARDING.md +1 -1
  6. package/docs/OPERATIONS.md +1 -1
  7. package/docs/SURFACE_PARITY.md +47 -0
  8. package/docs/TRUST_MODEL.md +1 -1
  9. package/docs/WHY_LATTICE.md +1 -1
  10. package/docs/kg-schema.md +1 -1
  11. package/lattice_brain/__init__.py +1 -1
  12. package/lattice_brain/graph/_kg_common.py +335 -0
  13. package/lattice_brain/graph/discovery_index.py +8 -1
  14. package/lattice_brain/graph/ingest.py +43 -9
  15. package/lattice_brain/graph/projection.py +221 -37
  16. package/lattice_brain/graph/retrieval.py +61 -12
  17. package/lattice_brain/graph/retrieval_policy.py +174 -0
  18. package/lattice_brain/graph/retrieval_vector.py +122 -2
  19. package/lattice_brain/portability.py +41 -12
  20. package/lattice_brain/runtime/multi_agent.py +1 -1
  21. package/latticeai/__init__.py +1 -1
  22. package/latticeai/api/chat.py +6 -0
  23. package/latticeai/api/chat_agent_http.py +175 -6
  24. package/latticeai/api/chat_intents.py +37 -18
  25. package/latticeai/api/chat_stream.py +76 -2
  26. package/latticeai/api/knowledge_graph.py +35 -1
  27. package/latticeai/core/agent.py +264 -9
  28. package/latticeai/core/enterprise.py +5 -0
  29. package/latticeai/core/file_generation.py +130 -5
  30. package/latticeai/core/legacy_compatibility.py +1 -1
  31. package/latticeai/core/marketplace.py +1 -1
  32. package/latticeai/core/run_store.py +243 -0
  33. package/latticeai/core/workspace_os.py +1 -1
  34. package/latticeai/models/router.py +25 -12
  35. package/latticeai/services/architecture_readiness.py +1 -1
  36. package/latticeai/services/command_center.py +90 -1
  37. package/latticeai/services/folder_watch.py +5 -0
  38. package/latticeai/services/funnel_metrics.py +9 -1
  39. package/latticeai/services/product_readiness.py +1 -1
  40. package/latticeai/services/search_service.py +43 -13
  41. package/package.json +1 -1
  42. package/scripts/bench_agent_smoke.py +410 -0
  43. package/scripts/check_current_release_docs.mjs +1 -1
  44. package/scripts/funnel_soft_gate.py +192 -0
  45. package/src-tauri/Cargo.lock +1 -1
  46. package/src-tauri/Cargo.toml +1 -1
  47. package/src-tauri/tauri.conf.json +1 -1
  48. package/static/app/asset-manifest.json +29 -28
  49. package/static/app/assets/Act-BCmTU0E2.js +1 -0
  50. package/static/app/assets/{Brain-DEY9jLVt.js → Brain-CTnjox7w.js} +2 -2
  51. package/static/app/assets/BrainHome-BJ3sFNX0.js +2 -0
  52. package/static/app/assets/BrainSignals-C52lwZVD.js +1 -0
  53. package/static/app/assets/{Capture-CVZ09QXi.js → Capture-B6vBhFa3.js} +1 -1
  54. package/static/app/assets/{CommandPalette-DepwOQFv.js → CommandPalette-90u9FWFH.js} +1 -1
  55. package/static/app/assets/{Library-Bp0n-HlW.js → Library-pKCK0_tk.js} +1 -1
  56. package/static/app/assets/{LivingBrain-DxP4efJF.js → LivingBrain-Jlf2wFqI.js} +1 -1
  57. package/static/app/assets/{ProductFlow-DRbm7NEq.js → ProductFlow-yg1fKP1P.js} +1 -1
  58. package/static/app/assets/{ReviewCard-C4HAO7A3.js → ReviewCard-DWvD7n9h.js} +1 -1
  59. package/static/app/assets/{System-ByQcmJW-.js → System-BAEuHqNY.js} +1 -1
  60. package/static/app/assets/{bot-BNDyZLR7.js → bot-nB_buEZD.js} +1 -1
  61. package/static/app/assets/circle-pause-DLNw6Ucp.js +1 -0
  62. package/static/app/assets/{circle-play-BkhdcHgd.js → circle-play-B-IsFL1y.js} +1 -1
  63. package/static/app/assets/{cpu-C6jjYm6i.js → cpu-CEPBHaBl.js} +1 -1
  64. package/static/app/assets/{folder-open-DjGIvDBQ.js → folder-open-hmN0N9cX.js} +1 -1
  65. package/static/app/assets/{hard-drive-BlSbwSaT.js → hard-drive-CBV_B_Yd.js} +1 -1
  66. package/static/app/assets/{index-Bge3DXW7.css → index-7FAfYm4v.css} +1 -1
  67. package/static/app/assets/{index-CHu7cgj3.js → index-DrmOCySv.js} +3 -3
  68. package/static/app/assets/{input-DVDI0YR3.js → input-gtVCg-ll.js} +1 -1
  69. package/static/app/assets/{navigation-BddhEWA0.js → navigation-Bot0hvuv.js} +1 -1
  70. package/static/app/assets/{network-pYQt5oBu.js → network-jE42eKfT.js} +1 -1
  71. package/static/app/assets/{primitives-D7gCdEvS.js → primitives-CX2Komon.js} +1 -1
  72. package/static/app/assets/{shield-alert-K9RKGQeg.js → shield-alert-BftATuAA.js} +1 -1
  73. package/static/app/assets/{textarea-sqQmoBKL.js → textarea-CiMJfOSI.js} +1 -1
  74. package/static/app/assets/{useFocusTrap-7EV9dFP2.js → useFocusTrap-B7RPGfFy.js} +1 -1
  75. package/static/app/assets/utils-SJUNVOj5.js +7 -0
  76. package/static/app/index.html +4 -4
  77. package/static/sw.js +1 -1
  78. package/static/app/assets/Act-DmdruVKV.js +0 -1
  79. package/static/app/assets/BrainHome-CeNaxjP1.js +0 -2
  80. package/static/app/assets/BrainSignals-CStjIqYi.js +0 -1
  81. package/static/app/assets/utils-uQYKXNeq.js +0 -7
@@ -10,7 +10,10 @@ Tracks the product's core value funnel with honest, local-only counters:
10
10
  * ``needs_review_runs`` — agent runs that ended in ``NEEDS_REVIEW``;
11
11
  * ``ingest_completions`` — successful ingestion pipeline completions;
12
12
  * ``recall_successes`` — chat turns whose context was grounded in at
13
- least one Brain node.
13
+ least one Brain node;
14
+ * ``approval_pauses`` — agent/workflow runs paused for human approval;
15
+ * ``approval_resumes`` — paused runs a human explicitly resumed
16
+ (approved), the ``approval_resume_rate`` numerator.
14
17
 
15
18
  TTFV ("time to first value") derives from two first-occurrence timestamps:
16
19
  the first successful ingest and the first grounded recall/answer.
@@ -44,6 +47,8 @@ COUNTER_NAMES = (
44
47
  "needs_review_runs",
45
48
  "ingest_completions",
46
49
  "recall_successes",
50
+ "approval_pauses",
51
+ "approval_resumes",
47
52
  )
48
53
 
49
54
  _FIRST_NAMES = ("first_ingest_at", "first_value_at")
@@ -193,6 +198,9 @@ class FunnelMetricsService:
193
198
  "needs_review_rate": _rate(
194
199
  counters["needs_review_runs"], counters["agent_runs"]
195
200
  ),
201
+ "approval_resume_rate": _rate(
202
+ counters["approval_resumes"], counters["approval_pauses"]
203
+ ),
196
204
  },
197
205
  "ttfv_seconds": self.ttfv_seconds(),
198
206
  "generated_at": _utc_now_iso(),
@@ -18,7 +18,7 @@ from typing import Any, Dict, List
18
18
 
19
19
  from latticeai.services.architecture_readiness import architecture_readiness
20
20
 
21
- PRODUCT_VERSION_TARGET = "9.9.3"
21
+ PRODUCT_VERSION_TARGET = "9.9.4"
22
22
 
23
23
 
24
24
  @dataclass(frozen=True)
@@ -7,9 +7,11 @@ keyword search into UI-ready contracts without tying routers to store internals.
7
7
  from __future__ import annotations
8
8
 
9
9
  from dataclasses import dataclass
10
- from typing import Any, Dict, Mapping, Optional
10
+ from datetime import datetime
11
+ from typing import Any, Dict, List, Mapping, Optional
11
12
 
12
- from lattice_brain.graph.fusion import fusion_profile
13
+ from lattice_brain.graph._kg_fsutil import _parse_iso, _recency_score
14
+ from lattice_brain.graph.retrieval_policy import resolve_policy
13
15
 
14
16
 
15
17
  DEFAULT_HYBRID_WEIGHTS = {
@@ -404,35 +406,44 @@ class SearchService:
404
406
  allowed_workspaces=None,
405
407
  include_legacy_global: bool = False,
406
408
  ) -> Dict[str, Any]:
407
- # Query-class fusion (backlog #5): when the caller does not pin
408
- # explicit weights, detect the query class (fact/code/person/recency)
409
- # and use its per-class channel weights. Explicit weights still win,
410
- # and the "fact" class equals DEFAULT_HYBRID_WEIGHTS, so pinned-weight
411
- # callers and fact-class queries behave exactly as before.
409
+ # Single retrieval policy (review Wave 0.2): when the caller does not
410
+ # pin explicit weights, resolve the query class + per-class channel
411
+ # weights + deterministic rewrite from lattice_brain.graph.
412
+ # retrieval_policy (the same policy the graph-layer hybrid consults).
413
+ # Explicit weights still win (and disable rewrite/decay), and the
414
+ # "fact" class equals DEFAULT_HYBRID_WEIGHTS, so pinned-weight callers
415
+ # and fact-class queries behave exactly as before.
412
416
  query_class: Optional[str] = None
417
+ search_query = query
418
+ rewrite_rules: List[str] = []
419
+ recency_half_life_days: Optional[float] = None
413
420
  if weights is None:
414
- profile = fusion_profile(query)
415
- query_class = profile["query_class"]
416
- weights = dict(profile["weights"])
421
+ policy = resolve_policy(query)
422
+ query_class = policy["query_class"]
423
+ weights = dict(policy["weights"])
424
+ rewrite_rules = list(policy["rewrite_rules"])
425
+ recency_half_life_days = policy["recency_half_life_days"]
426
+ if policy["search_query"] and policy["search_query"] != policy["original_query"]:
427
+ search_query = policy["search_query"]
417
428
  else:
418
429
  weights = {**DEFAULT_HYBRID_WEIGHTS, **dict(weights)}
419
430
  # Scope each channel at the source so out-of-scope rows never enter the
420
431
  # fusion set (defense-in-depth — the fused result is re-scoped below too).
421
432
  channels = {
422
433
  "keyword": self.keyword_search(
423
- query,
434
+ search_query,
424
435
  limit=keyword_limit,
425
436
  allowed_workspaces=allowed_workspaces,
426
437
  include_legacy_global=include_legacy_global,
427
438
  ),
428
439
  "vector": self.vector_search(
429
- query,
440
+ search_query,
430
441
  limit=vector_limit,
431
442
  allowed_workspaces=allowed_workspaces,
432
443
  include_legacy_global=include_legacy_global,
433
444
  ),
434
445
  "graph": self.graph_search(
435
- query,
446
+ search_query,
436
447
  limit=graph_limit,
437
448
  allowed_workspaces=allowed_workspaces,
438
449
  include_legacy_global=include_legacy_global,
@@ -465,6 +476,24 @@ class SearchService:
465
476
  current.setdefault("graph_context", [])
466
477
  current["graph_context"].extend(result.get("graph_context") or [])
467
478
 
479
+ # Recency-class age decay (retrieval_policy): dampen each fused score
480
+ # into the [0.5, 1.0] band so old-but-relevant items sink without ever
481
+ # being zeroed. Missing/unparseable updated_at keeps multiplier 1.0 —
482
+ # unknown age is not evidence of staleness. Other classes skip this
483
+ # block byte-identically.
484
+ if recency_half_life_days is not None:
485
+ decay_now = datetime.now()
486
+ for item in fused.values():
487
+ stamp = item.get("updated_at")
488
+ if _parse_iso(stamp):
489
+ multiplier = 0.5 + 0.5 * _recency_score(
490
+ stamp, now=decay_now, half_life_days=recency_half_life_days
491
+ )
492
+ else:
493
+ multiplier = 1.0
494
+ item["source_scores"]["age_decay"] = round(multiplier, 6)
495
+ item["score"] = float(item["score"]) * multiplier
496
+
468
497
  matches = self._scope(
469
498
  sorted(fused.values(), key=lambda item: item["score"], reverse=True),
470
499
  allowed_workspaces,
@@ -484,6 +513,7 @@ class SearchService:
484
513
  "mode": "hybrid",
485
514
  "query_class": query_class,
486
515
  "weights": weights,
516
+ "policy": {"search_query": search_query, "rewrite_rules": rewrite_rules},
487
517
  "channels": {
488
518
  name: {
489
519
  key: value
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ltcai",
3
- "version": "9.9.3",
3
+ "version": "9.9.4",
4
4
  "description": "Lattice AI — local-first Digital Brain that keeps your knowledge durable across any AI model.",
5
5
  "homepage": "https://github.com/TaeSooPark-PTS/LatticeAI#readme",
6
6
  "repository": {
@@ -0,0 +1,410 @@
1
+ #!/usr/bin/env python3
2
+ """Weekly real-model agent-loop smoke (review Wave 3.4) — FAIL-OPEN.
3
+
4
+ What it measures
5
+ ================
6
+ ``scripts/agent_eval.py`` proves the loop's state machine against *scripted*
7
+ model replies. This harness closes the remaining gap: it drives a small set of
8
+ canonical agent tasks through the REAL :class:`latticeai.core.agent.
9
+ SingleAgentRuntime` (plan → approve → execute → verify), with the generation
10
+ port wired to each *locally installed* MLX model — real inference, no HTTP
11
+ server. Per model and task it reports:
12
+
13
+ * ``final_state`` — the loop's honest terminal state (DONE / NEEDS_REVIEW /
14
+ FAILED), plus the ``agent_eval`` result-class bucket;
15
+ * ``steps`` — LLM calls the run needed;
16
+ * parse repairs — ``parse_errors`` / ``parse_recovered`` and the LoopTrace
17
+ ``repairs`` histogram (how hard the loop worked to keep this model on rails);
18
+ * ``duration_s`` — true end-to-end wall time including inference.
19
+
20
+ Model access pattern
21
+ ====================
22
+ Identical to ``scripts/bench_models.py --filegen``: installed gemma/qwen/llama
23
+ MLX models are discovered through the product's own model catalog + on-disk HF
24
+ checks, loaded with the real ``LLMRouter``, and unloaded afterwards. The tool
25
+ port is in-memory (canned results, nothing touches disk or network) — this is
26
+ a smoke of the LOOP over real model output, not of the tool implementations.
27
+
28
+ FAIL-OPEN by design
29
+ ===================
30
+ This is a scheduled/weekly report, never a CI gate:
31
+
32
+ * no models installed (or catalog import fails, e.g. on a CI runner) →
33
+ an honest "no models available — skipped (fail-open)" report, exit 0;
34
+ * a model that fails to load → a per-model skip entry, exit 0;
35
+ * any run-level crash → a skip report naming the error, exit 0;
36
+ * weak results are the finding, never a failure.
37
+
38
+ Non-zero exits happen only for real script errors (bad flags → argparse's 2).
39
+
40
+ Usage
41
+ =====
42
+ .venv/bin/python scripts/bench_agent_smoke.py # human table
43
+ .venv/bin/python scripts/bench_agent_smoke.py --json # JSON to stdout
44
+ .venv/bin/python scripts/bench_agent_smoke.py --tasks 1 --max-steps 6
45
+ .venv/bin/python scripts/bench_agent_smoke.py --model mlx-community/...
46
+ """
47
+
48
+ from __future__ import annotations
49
+
50
+ import argparse
51
+ import asyncio
52
+ import json
53
+ import sys
54
+ import tempfile
55
+ import time
56
+ from pathlib import Path
57
+ from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple
58
+
59
+ sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
60
+
61
+ from latticeai.core.agent import ( # noqa: E402
62
+ AgentDeps,
63
+ AgentRunContext,
64
+ AgentState,
65
+ SingleAgentRuntime,
66
+ )
67
+ from latticeai.core.agent_eval import classify_result # noqa: E402
68
+ from latticeai.core.agent_prompts import ( # noqa: E402
69
+ CRITIC_PROMPT,
70
+ EXECUTOR_PROMPT,
71
+ MEMORY_UPDATER_PROMPT,
72
+ PLANNER_PROMPT,
73
+ )
74
+ from latticeai.tools import ToolError # noqa: E402
75
+
76
+
77
+ # ── canonical smoke tasks ────────────────────────────────────────────────
78
+ # Small, file-flavored tasks users actually ask for. The tool port is canned,
79
+ # so the measurement is "can this model steer the real loop to completion",
80
+ # not "is the artifact beautiful".
81
+ SMOKE_TASKS: List[Tuple[str, str]] = [
82
+ ("haiku-file", "Write a short haiku about autumn into haiku.txt."),
83
+ ("hello-script", "Create a Python file hello.py that prints hello."),
84
+ (
85
+ "list-then-summarize",
86
+ "List the files in the workspace, then summarize what you see in one sentence.",
87
+ ),
88
+ ]
89
+
90
+ _SMOKE_MODEL_FAMILIES = ("gemma", "qwen", "llama")
91
+
92
+ _AUTO_POLICY = {
93
+ "auto_approve": True, "risk": "low", "shell": False, "network": False,
94
+ "destructive": False, "sandbox": False, "rollback": "none",
95
+ }
96
+
97
+
98
+ def discover_agent_models() -> List[Dict[str, str]]:
99
+ """Installed local MLX models — the exact discovery pattern of
100
+ ``scripts/bench_models.py --filegen`` (product catalog + on-disk HF
101
+ checks). Any import/probe failure is fail-open: an empty list, never an
102
+ exception. Messages go to stderr so ``--json`` stdout stays parseable."""
103
+ try:
104
+ from latticeai.models.router import (
105
+ _looks_like_hf_model_dir,
106
+ hf_cache_model_dir,
107
+ hf_model_dir,
108
+ )
109
+ from latticeai.services.model_catalog import ENGINE_MODEL_CATALOG
110
+ except Exception as exc: # noqa: BLE001 - discovery must never crash the report
111
+ print(
112
+ f"agent-smoke: model catalog unavailable ({exc}); no models discovered",
113
+ file=sys.stderr,
114
+ )
115
+ return []
116
+ found: List[Dict[str, str]] = []
117
+ for entry in ENGINE_MODEL_CATALOG.get("local_mlx", []):
118
+ model_id = str(entry.get("id") or "")
119
+ lowered = model_id.lower()
120
+ family = next((f for f in _SMOKE_MODEL_FAMILIES if f in lowered), None)
121
+ if not family:
122
+ continue
123
+ try:
124
+ downloaded = (
125
+ _looks_like_hf_model_dir(hf_model_dir(model_id))
126
+ or hf_cache_model_dir(model_id) is not None
127
+ )
128
+ except Exception: # noqa: BLE001
129
+ downloaded = False
130
+ if downloaded:
131
+ found.append({"id": model_id, "family": family})
132
+ return found
133
+
134
+
135
+ # ── HTTP-less single-agent runtime over an in-memory tool port ───────────
136
+
137
+ class _SmokeReq:
138
+ conversation_id = None
139
+ temperature = 0.2
140
+ workspace_id = None
141
+ source = "agent_smoke"
142
+
143
+ def __init__(self, message: str) -> None:
144
+ self.message = message
145
+
146
+
147
+ def _build_smoke_deps(
148
+ generate_as: Callable[..., Awaitable[Any]],
149
+ tool_log: List[Dict[str, Any]],
150
+ ) -> AgentDeps:
151
+ """Real production prompts + a canned in-memory tool port.
152
+
153
+ The port mirrors the fake in ``latticeai.core.agent_eval``: it records
154
+ calls, fails a pathless write like the real dispatcher, and never touches
155
+ disk or network. All tools are auto-approve — governance behaviour is
156
+ agent_eval's job; this smoke measures loop mechanics over real inference.
157
+ """
158
+
159
+ async def generate(**kwargs):
160
+ return '{"action": "noop"}'
161
+
162
+ def execute_tool(name: str, args: dict) -> dict:
163
+ if name in ("write_file", "generate_file") and not args.get("path"):
164
+ raise ToolError(f"{name} requires args.path")
165
+ tool_log.append({"name": name, "args": args})
166
+ if name == "list_dir":
167
+ return {"ok": True, "entries": ["README.md", "haiku.txt", "hello.py"]}
168
+ if name == "read_file":
169
+ return {"ok": True, "path": args.get("path", ""), "content": "smoke fixture file"}
170
+ return {"ok": True, "path": args.get("path", "")}
171
+
172
+ return AgentDeps(
173
+ generate_as=generate_as,
174
+ generate=generate,
175
+ execute_tool=execute_tool,
176
+ policy_for=lambda name, args: dict(_AUTO_POLICY),
177
+ risk_level=lambda p: p["risk"],
178
+ check_role=lambda name, user: None,
179
+ tool_governance={
180
+ "write_file": dict(_AUTO_POLICY),
181
+ "read_file": dict(_AUTO_POLICY),
182
+ "list_dir": dict(_AUTO_POLICY),
183
+ "generate_file": dict(_AUTO_POLICY),
184
+ "knowledge_graph_ingest": dict(_AUTO_POLICY),
185
+ "knowledge_graph_search": dict(_AUTO_POLICY),
186
+ },
187
+ file_create_actions=frozenset({"write_file", "generate_file"}),
188
+ recent_chat_context=lambda **kw: "",
189
+ clear_history=lambda keep: {"ok": True},
190
+ knowledge_save=lambda *a, **kw: None,
191
+ audit=lambda *a, **kw: None,
192
+ planner_prompt=PLANNER_PROMPT,
193
+ executor_prompt=EXECUTOR_PROMPT,
194
+ critic_prompt=CRITIC_PROMPT,
195
+ memory_updater_prompt=MEMORY_UPDATER_PROMPT,
196
+ agent_root=Path(tempfile.gettempdir()) / "agent-smoke",
197
+ )
198
+
199
+
200
+ async def _run_one_task(
201
+ generate_as: Callable[..., Awaitable[Any]],
202
+ task_id: str,
203
+ prompt: str,
204
+ *,
205
+ max_steps: int = 8,
206
+ model_id: Optional[str] = None,
207
+ ) -> Dict[str, Any]:
208
+ """Drive one task through the real state machine; reduce to a report row."""
209
+ tool_log: List[Dict[str, Any]] = []
210
+ runtime = SingleAgentRuntime(_build_smoke_deps(generate_as, tool_log))
211
+ ctx = AgentRunContext()
212
+ ctx.state = AgentState.PLANNING
213
+ ctx.executing_model = model_id
214
+ ctx.reviewing_model = model_id
215
+ req = _SmokeReq(prompt)
216
+
217
+ t0 = time.perf_counter()
218
+ await runtime.plan(ctx, req, "en", "smoke@local", model_id=model_id)
219
+ # The operator launched this run deliberately — the human approval the
220
+ # gate asks for. The gate itself (and denial paths) are covered by
221
+ # agent_eval; a real model's free-form plan must not dead-end the smoke.
222
+ runtime.approve(ctx, "smoke@local", approved_by_human=True)
223
+ if ctx.state == AgentState.EXECUTING:
224
+ await runtime.run_to_completion(
225
+ ctx, req, "en", "smoke@local", max_steps=max_steps, max_retry=2
226
+ )
227
+ duration = round(time.perf_counter() - t0, 2)
228
+
229
+ summary = ctx.trace.summary()
230
+ executed = [call["name"] for call in tool_log]
231
+ return {
232
+ "task": task_id,
233
+ "final_state": ctx.state.value,
234
+ "result_class": classify_result(ctx.state.value, ctx.trace.events, summary, executed),
235
+ "steps": summary["llm_calls"],
236
+ "parse_errors": summary["parse_errors"],
237
+ "parse_recovered": summary["parse_recovered"],
238
+ "repairs_total": sum((summary.get("repairs") or {}).values()),
239
+ "repairs": summary.get("repairs") or {},
240
+ "tool_calls": len(tool_log),
241
+ "duration_s": duration,
242
+ }
243
+
244
+
245
+ def _make_router_generate_as(router: Any, model_id: str) -> Callable[..., Awaitable[str]]:
246
+ """Bridge the runtime's generate_as port to the real LLMRouter (the same
247
+ ``generate_as`` call shape bench_models' filegen mode uses)."""
248
+
249
+ async def generate_as(_model_id, message, context, max_tokens, temperature):
250
+ return str(
251
+ await router.generate_as(
252
+ _model_id or model_id,
253
+ message=message,
254
+ context=context,
255
+ max_tokens=max_tokens,
256
+ temperature=temperature,
257
+ )
258
+ )
259
+
260
+ return generate_as
261
+
262
+
263
+ async def _smoke_run_models(
264
+ models: List[Dict[str, str]],
265
+ tasks: List[Tuple[str, str]],
266
+ max_steps: int,
267
+ ) -> List[Dict[str, Any]]:
268
+ from latticeai.models.router import LLMRouter
269
+
270
+ router = LLMRouter()
271
+ results: List[Dict[str, Any]] = []
272
+ for model in models:
273
+ model_id = model["id"]
274
+ try:
275
+ await router.load_model(model_id)
276
+ except Exception as exc: # noqa: BLE001 - fail-open per model
277
+ results.append({
278
+ "model": model_id, "family": model["family"],
279
+ "skipped": True, "reason": f"load failed: {exc}",
280
+ })
281
+ continue
282
+ generate_as = _make_router_generate_as(router, model_id)
283
+ rows: List[Dict[str, Any]] = []
284
+ for task_id, prompt in tasks:
285
+ try:
286
+ rows.append(await _run_one_task(
287
+ generate_as, task_id, prompt,
288
+ max_steps=max_steps, model_id=model_id,
289
+ ))
290
+ except Exception as exc: # noqa: BLE001 - one broken task never sinks the report
291
+ rows.append({
292
+ "task": task_id, "final_state": "HARNESS_ERROR",
293
+ "result_class": "failed", "error": str(exc),
294
+ })
295
+ results.append({
296
+ "model": model_id,
297
+ "family": model["family"],
298
+ "tasks": rows,
299
+ "tasks_total": len(rows),
300
+ "completed": sum(1 for r in rows if r.get("final_state") == "DONE"),
301
+ "duration_s": round(sum(float(r.get("duration_s") or 0.0) for r in rows), 2),
302
+ })
303
+ try:
304
+ router.unload_model(model_id)
305
+ except Exception: # noqa: BLE001
306
+ pass
307
+ return results
308
+
309
+
310
+ def run_agent_smoke(
311
+ models: Optional[List[Dict[str, str]]] = None,
312
+ tasks: Optional[List[Tuple[str, str]]] = None,
313
+ max_steps: int = 8,
314
+ ) -> Dict[str, Any]:
315
+ """Weekly per-model agent-loop report. Fail-open: never raises, never gates."""
316
+ if models is None:
317
+ models = discover_agent_models()
318
+ selected_tasks = list(tasks or SMOKE_TASKS)
319
+ if not models:
320
+ return {
321
+ "mode": "agent-smoke",
322
+ "status": "skipped",
323
+ "reason": (
324
+ "no models available — install a local gemma/qwen/llama model "
325
+ "via the app's model picker, then re-run"
326
+ ),
327
+ "models": [],
328
+ }
329
+ try:
330
+ results = asyncio.run(_smoke_run_models(models, selected_tasks, max_steps))
331
+ except Exception as exc: # noqa: BLE001 - fail-open at the run level too
332
+ return {
333
+ "mode": "agent-smoke", "status": "skipped",
334
+ "reason": f"smoke run failed: {exc}", "models": [],
335
+ }
336
+ return {"mode": "agent-smoke", "status": "ok", "models": results}
337
+
338
+
339
+ def format_smoke_report(report: Dict[str, Any]) -> str:
340
+ lines = [
341
+ "Weekly agent-loop smoke (real models through SingleAgentRuntime)",
342
+ "=" * 72,
343
+ ]
344
+ if report.get("status") == "skipped":
345
+ lines.append(f"no models available — skipped (fail-open): {report.get('reason')}")
346
+ lines.append("=" * 72)
347
+ return "\n".join(lines)
348
+ for model in report.get("models", []):
349
+ if model.get("skipped"):
350
+ lines.append(f" {model['model']}: SKIPPED — {model.get('reason')}")
351
+ continue
352
+ lines.append(
353
+ f" {model['model']} "
354
+ f"(completed {model['completed']}/{model['tasks_total']}, "
355
+ f"{model['duration_s']}s)"
356
+ )
357
+ for row in model.get("tasks", []):
358
+ if row.get("final_state") == "HARNESS_ERROR":
359
+ lines.append(f" {row['task']:<22} HARNESS_ERROR {row.get('error')}")
360
+ continue
361
+ lines.append(
362
+ f" {row['task']:<22} {row['final_state']:<13} "
363
+ f"steps={row['steps']} parse_errors={row['parse_errors']} "
364
+ f"repairs={row['repairs_total']} tools={row['tool_calls']} "
365
+ f"{row['duration_s']}s"
366
+ )
367
+ lines.append("=" * 72)
368
+ lines.append(
369
+ "note: fail-open weekly report — weak results are the finding, never a failure"
370
+ )
371
+ return "\n".join(lines)
372
+
373
+
374
+ def main(argv: List[str] | None = None) -> int:
375
+ parser = argparse.ArgumentParser(
376
+ description="Weekly real-model agent-loop smoke (fail-open, never a gate)"
377
+ )
378
+ parser.add_argument(
379
+ "--json", dest="json_out", action="store_true",
380
+ help="print the machine-readable report JSON to stdout",
381
+ )
382
+ parser.add_argument(
383
+ "--tasks", type=int, default=len(SMOKE_TASKS),
384
+ help=f"how many smoke tasks to run per model (1-{len(SMOKE_TASKS)}, default all)",
385
+ )
386
+ parser.add_argument(
387
+ "--max-steps", type=int, default=8, help="agent-loop step budget per task",
388
+ )
389
+ parser.add_argument(
390
+ "--model", help="restrict the run to one installed model id",
391
+ )
392
+ args = parser.parse_args(argv)
393
+
394
+ count = max(1, min(int(args.tasks), len(SMOKE_TASKS)))
395
+ models = discover_agent_models()
396
+ if args.model:
397
+ models = [m for m in models if m["id"] == args.model]
398
+ report = run_agent_smoke(
399
+ models=models, tasks=SMOKE_TASKS[:count], max_steps=max(2, int(args.max_steps)),
400
+ )
401
+ if args.json_out:
402
+ print(json.dumps(report, ensure_ascii=False, indent=2))
403
+ else:
404
+ print(format_smoke_report(report))
405
+ # FAIL-OPEN: missing or weak models never produce a non-zero exit.
406
+ return 0
407
+
408
+
409
+ if __name__ == "__main__":
410
+ raise SystemExit(main())
@@ -6,7 +6,7 @@ const root = process.cwd();
6
6
  const pkg = JSON.parse(readFileSync(path.join(root, "package.json"), "utf8"));
7
7
  const version = pkg.version;
8
8
  const releaseDir = `output/release/v${version}`;
9
- const releaseTheme = "Closed Loops";
9
+ const releaseTheme = "Durable Loops";
10
10
  const title = `${version} — ${releaseTheme}`;
11
11
  const escapedVersion = version.replaceAll(".", "\\.");
12
12