ltcai 11.5.0 → 11.5.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +86 -47
- package/docs/CHANGELOG.md +76 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/HYBRID_CLOUD_KG_STREAMING.md +8 -4
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/WORKFLOW_DESIGNER.md +1 -1
- package/docs/architecture.md +4 -4
- package/docs/kg-schema.md +1 -1
- package/docs/v11.5.1_RUST_FULL_LOOP_PLAN.md +85 -0
- package/docs/v11.5.2_TIGHT_SHIP_PLAN.md +144 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/proactive.py +1 -1
- package/lattice_brain/graph/projection/v2_schema.py +0 -28
- package/lattice_brain/graph/vector_index/base.py +0 -3
- package/lattice_brain/graph/vector_index/brute_force.py +0 -4
- package/lattice_brain/graph/vector_index/hnsw.py +0 -3
- package/lattice_brain/graph/vector_index/quantized.py +0 -3
- package/lattice_brain/ingestion_jobs.py +0 -3
- package/lattice_brain/memory.py +0 -27
- package/lattice_brain/portability/sharing.py +0 -4
- package/lattice_brain/quality.py +0 -49
- package/lattice_brain/runtime/agent_runtime.py +1 -1
- package/lattice_brain/runtime/hooks.py +0 -13
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/lattice_brain/sealed_box.py +0 -4
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/admin.py +12 -12
- package/latticeai/api/agent_worker_seam.py +389 -0
- package/latticeai/api/auth.py +25 -2
- package/latticeai/api/chat.py +3 -3
- package/latticeai/api/chat_agent_http.py +3 -12
- package/latticeai/api/chat_helpers.py +10 -13
- package/latticeai/api/computer_use.py +31 -42
- package/latticeai/api/health.py +21 -1
- package/latticeai/api/local_files.py +14 -0
- package/latticeai/api/permissions.py +38 -12
- package/latticeai/api/search.py +24 -0
- package/latticeai/api/static_routes.py +4 -1
- package/latticeai/cli/runtime.py +7 -3
- package/latticeai/core/config.py +47 -3
- package/latticeai/core/csrf.py +28 -2
- package/latticeai/core/embedding_providers/__init__.py +3 -5
- package/latticeai/core/embedding_providers/base.py +1 -1
- package/latticeai/core/embedding_providers/text.py +1 -1
- package/latticeai/core/enterprise.py +1 -4
- package/latticeai/core/http_origin.py +146 -0
- package/latticeai/core/invitations.py +3 -2
- package/latticeai/core/io_utils.py +2 -11
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +40 -0
- package/latticeai/core/model_compat.py +2 -8
- package/latticeai/core/module_probe.py +37 -0
- package/latticeai/core/run_explain.py +5 -2
- package/latticeai/core/run_store.py +2 -2
- package/latticeai/core/security.py +13 -0
- package/latticeai/core/sessions.py +3 -2
- package/latticeai/core/sse.py +25 -0
- package/latticeai/core/users.py +0 -9
- package/latticeai/core/workspace_graph_trace.py +2 -8
- package/latticeai/core/workspace_os.py +0 -10
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/core/workspace_os_utils.py +34 -0
- package/latticeai/core/workspace_review_items.py +2 -8
- package/latticeai/core/workspace_runs.py +2 -8
- package/latticeai/core/workspace_skills.py +2 -7
- package/latticeai/models/router/documents.py +0 -35
- package/latticeai/models/router/registry.py +0 -4
- package/latticeai/runtime/access_runtime.py +16 -4
- package/latticeai/runtime/build_phases/features.py +25 -2
- package/latticeai/runtime/build_phases/web.py +2 -0
- package/latticeai/runtime/feature_toggle_wiring.py +12 -18
- package/latticeai/runtime/network_boundary_wiring.py +8 -19
- package/latticeai/runtime/permission_mode_wiring.py +6 -13
- package/latticeai/runtime/router_registration.py +4 -0
- package/latticeai/runtime/runtime_context.py +1 -13
- package/latticeai/runtime/service_singletons.py +55 -0
- package/latticeai/services/architecture_readiness.py +1 -1
- package/latticeai/services/brain_intelligence/proposals.py +0 -5
- package/latticeai/services/change_proposals.py +2 -36
- package/latticeai/services/chronicle.py +4 -6
- package/latticeai/services/command_center.py +4 -4
- package/latticeai/services/evidence_actions.py +5 -2
- package/latticeai/services/hybrid_chat.py +2 -2
- package/latticeai/services/hybrid_policy.py +8 -57
- package/latticeai/services/mode_store.py +132 -0
- package/latticeai/services/model_capability_registry.py +0 -15
- package/latticeai/services/model_catalog.py +0 -7
- package/latticeai/services/model_loading.py +3 -2
- package/latticeai/services/model_runtime/__init__.py +0 -57
- package/latticeai/services/model_runtime/engines.py +27 -128
- package/latticeai/services/model_runtime/loading.py +2 -2
- package/latticeai/services/network_boundary_service.py +7 -44
- package/latticeai/services/permission_mode_service.py +8 -56
- package/latticeai/services/product_readiness.py +1 -1
- package/latticeai/services/setup_detection.py +67 -1
- package/latticeai/services/tool_dispatch.py +0 -4
- package/latticeai/services/upload_service.py +7 -20
- package/latticeai/services/workspace_service.py +0 -6
- package/latticeai/setup/auto_setup.py +15 -26
- package/latticeai/setup/wizard/catalog.py +5 -2
- package/latticeai/setup/wizard/detect.py +6 -27
- package/latticeai/setup/wizard/paths.py +2 -5
- package/latticeai/setup/wizard/plans.py +2 -2
- package/package.json +2 -4
- package/scripts/bump_version.py +14 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_legacy_debt.mjs +1 -1
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_agent_loop_fixtures.py +994 -0
- package/scripts/generate_rust_parity_fixtures.py +72 -161
- package/scripts/parity_fixture_corpus_context.py +162 -0
- package/scripts/parity_fixture_corpus_docgen.py +341 -0
- package/scripts/release_screen_claims.json +20 -0
- package/src-tauri/Cargo.lock +13 -7
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/src/backend.rs +76 -2
- package/src-tauri/src/main.rs +13 -3
- package/src-tauri/tauri.conf.json +2 -2
- package/static/app/asset-manifest.json +41 -41
- package/static/app/assets/{Act-CWnxSCgN.js → Act-DcQizkl1.js} +1 -1
- package/static/app/assets/{AdminConsole-BEQYU6kF.js → AdminConsole-cf4npybT.js} +1 -1
- package/static/app/assets/{Brain-DWu1BhFg.js → Brain-3VCSHFcn.js} +1 -1
- package/static/app/assets/{BrainHome-95Hilr9R.js → BrainHome-Qm8eaztx.js} +1 -1
- package/static/app/assets/{BrainSignals-QdeqCpAF.js → BrainSignals-DS9BtKOW.js} +1 -1
- package/static/app/assets/{Capture-BHpCxnzb.js → Capture-DiQ219jW.js} +1 -1
- package/static/app/assets/{Chronicle-B4xYKoed.js → Chronicle-BGvuAchH.js} +1 -1
- package/static/app/assets/{CommandPalette-BVXnttSz.js → CommandPalette-Bqhm0Urn.js} +1 -1
- package/static/app/assets/{Library-DgYcHome.js → Library-BV6NnF0a.js} +1 -1
- package/static/app/assets/{LivingBrain-CrJLDbf7.js → LivingBrain-GzenJchP.js} +1 -1
- package/static/app/assets/{ProductFlow-DFlScKoJ.js → ProductFlow-DEP6-vML.js} +1 -1
- package/static/app/assets/{ReviewCard-Cy5f48Pj.js → ReviewCard-CNZ7XjWG.js} +1 -1
- package/static/app/assets/{System-NF8IfhTa.js → System-CieofHQa.js} +1 -1
- package/static/app/assets/arrow-left-kfsrk0mv.js +1 -0
- package/static/app/assets/{bot-CucuhLhm.js → bot-B_K1Tdmw.js} +1 -1
- package/static/app/assets/{brain-BBnSryW_.js → brain-DWyaV1L1.js} +1 -1
- package/static/app/assets/{button-C2GUj2Ai.js → button-aTn4s84A.js} +1 -1
- package/static/app/assets/circle-check-qqLug9nU.js +1 -0
- package/static/app/assets/{circle-pause-CbkWzBmG.js → circle-pause-xKgeGXkT.js} +1 -1
- package/static/app/assets/{circle-play-7lEaqHdJ.js → circle-play-DkT6tYPX.js} +1 -1
- package/static/app/assets/{cpu-DAlCXlIy.js → cpu-85xYObUC.js} +1 -1
- package/static/app/assets/{download-RNhuuJwh.js → download-B5Fm7YXo.js} +1 -1
- package/static/app/assets/{folder-open-CLW4odzM.js → folder-open-kk2Xa52u.js} +1 -1
- package/static/app/assets/{hard-drive-NKEiDIAJ.js → hard-drive-DkA3zBW_.js} +1 -1
- package/static/app/assets/{index-DMurvUuR.js → index-BMPdTmlY.js} +3 -3
- package/static/app/assets/index-DxmOfNRi.css +2 -0
- package/static/app/assets/{input-D2UhPC1X.js → input-B0nRf2jO.js} +1 -1
- package/static/app/assets/{link-2-6amKbP_P.js → link-2-Dwb4gnTc.js} +1 -1
- package/static/app/assets/{permissionCopy-Cu9TZtdR.js → permissionCopy-CQDUBrOZ.js} +1 -1
- package/static/app/assets/{primitives-gPsccucr.js → primitives-SNp0LRJz.js} +1 -1
- package/static/app/assets/search-BcHqkjoy.js +1 -0
- package/static/app/assets/{share-2-Bau7KkPq.js → share-2-BsrxFglO.js} +1 -1
- package/static/app/assets/{shield-alert-BufNYypi.js → shield-alert-5BStfp2_.js} +1 -1
- package/static/app/assets/{textarea-BQnVWhYs.js → textarea-Cg8IUA-k.js} +1 -1
- package/static/app/assets/{useFocusTrap-B3_w60si.js → useFocusTrap-CYKvE46M.js} +1 -1
- package/static/app/assets/{useMutation-BHhCflT6.js → useMutation-CSn9t1op.js} +1 -1
- package/static/app/assets/{useQuery-rBWfI-5t.js → useQuery-CY2OI2uy.js} +1 -1
- package/static/app/assets/{utils-V_5-wxr5.js → utils-Ddol2RWD.js} +1 -1
- package/static/app/assets/{workspace-K1zjYUHj.js → workspace-BqDwOz_p.js} +1 -1
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/desktop/electron/README.md +0 -9
- package/desktop/electron/main.cjs +0 -58
- package/desktop/electron/preload.cjs +0 -5
- package/latticeai/core/graph_curator.py +0 -11
- package/latticeai/core/hooks.py +0 -11
- package/latticeai/core/local_embeddings.py +0 -104
- package/latticeai/core/multi_agent.py +0 -11
- package/latticeai/core/workflow_engine.py +0 -11
- package/latticeai/services/ingestion.py +0 -11
- package/latticeai/services/kg_portability.py +0 -11
- package/latticeai/services/multimodal_streaming.py +0 -129
- package/scripts/measure_brain_home_fill.mjs +0 -141
- package/static/app/assets/arrow-left-DwkSYrjR.js +0 -1
- package/static/app/assets/circle-check-CxOVPwYq.js +0 -1
- package/static/app/assets/index-BLPb5lmE.css +0 -2
- package/static/app/assets/search-Cj_TKk_2.js +0 -1
|
@@ -0,0 +1,994 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build the committed Python↔Rust **agent loop** parity fixtures (v11.5.1).
|
|
3
|
+
|
|
4
|
+
``rust/lattice-agent`` now owns the PLAN → EXECUTE → VERIFY → ROLLBACK
|
|
5
|
+
orchestration that ``latticeai.core.agent`` has always owned, with the Python
|
|
6
|
+
worker behind three seam endpoints. A port of a *state machine* is only worth
|
|
7
|
+
having if something keeps proving it still reaches the same states, by the same
|
|
8
|
+
route, with the same record — so this script is the Python half of that proof.
|
|
9
|
+
|
|
10
|
+
It runs the **real** functions, never a re-description of them:
|
|
11
|
+
|
|
12
|
+
* the deterministic helpers — ``extract_action_details``, ``normalize_plan``,
|
|
13
|
+
``infer_file_target`` / ``infer_project_manifest``, ``requirement_coverage``,
|
|
14
|
+
``artifact_checklist``, ``files_written``, ``compact_transcript``,
|
|
15
|
+
``_truncate_strings``, ``filter_learnings``, ``PhaseBudgets`` /
|
|
16
|
+
``TranscriptBudget``;
|
|
17
|
+
* the verification verdict mapping, by calling the real ``verify()`` over a
|
|
18
|
+
verdict × evidence × coverage × retry grid with a scripted critic;
|
|
19
|
+
* the run store's ``serialize_run_context`` / ``restore_run_context``;
|
|
20
|
+
* and **end-to-end trajectories**: the real :class:`SingleAgentRuntime`, driven
|
|
21
|
+
by a scripted LLM and the real tool registry inside a throwaway
|
|
22
|
+
``AGENT_ROOT``, for seven scenarios that between them exercise every branch
|
|
23
|
+
the loop can take to a terminal state.
|
|
24
|
+
|
|
25
|
+
Two consumers read what it writes:
|
|
26
|
+
|
|
27
|
+
* ``tests/unit/test_agent_loop_parity_contract.py`` re-runs the Python loop over
|
|
28
|
+
the same scripts and asserts the committed goldens still hold — so a change to
|
|
29
|
+
a Python gate fails loudly instead of silently invalidating the contract the
|
|
30
|
+
Rust side is pinned to;
|
|
31
|
+
* ``rust/lattice-agent/tests/agent_loop.rs`` drives the native loop against a
|
|
32
|
+
fake worker replaying the recorded completions and tool results, and asserts
|
|
33
|
+
the same trajectories.
|
|
34
|
+
|
|
35
|
+
Determinism is the design constraint. Four normalisation rules make the record
|
|
36
|
+
machine-independent, and both sides apply them:
|
|
37
|
+
|
|
38
|
+
1. the absolute workspace root becomes ``<AGENT_ROOT>``;
|
|
39
|
+
2. ``at`` keys are dropped (trace timestamps);
|
|
40
|
+
3. ``stderr`` keys are dropped (git's text is version- and locale-specific);
|
|
41
|
+
4. a JSON decoder detail is collapsed to ``<decoder-detail>`` — CPython renamed
|
|
42
|
+
several of those messages in 3.14, and pinning a Python patch release is not
|
|
43
|
+
what this contract is for.
|
|
44
|
+
|
|
45
|
+
Usage::
|
|
46
|
+
|
|
47
|
+
.venv/bin/python scripts/generate_agent_loop_fixtures.py
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
from __future__ import annotations
|
|
51
|
+
|
|
52
|
+
import asyncio
|
|
53
|
+
import copy
|
|
54
|
+
import json
|
|
55
|
+
import os
|
|
56
|
+
import sys
|
|
57
|
+
import tempfile
|
|
58
|
+
from contextlib import contextmanager
|
|
59
|
+
from pathlib import Path
|
|
60
|
+
from typing import Any, Dict, Iterator, List, Optional
|
|
61
|
+
|
|
62
|
+
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
63
|
+
if str(REPO_ROOT) not in sys.path:
|
|
64
|
+
sys.path.insert(0, str(REPO_ROOT))
|
|
65
|
+
|
|
66
|
+
# The tool registry resolves AGENT_ROOT at import; point it somewhere harmless
|
|
67
|
+
# before anything imports it, then patch it per run through `use_workspace`.
|
|
68
|
+
os.environ.setdefault(
|
|
69
|
+
"LATTICEAI_AGENT_ROOT", str(Path(tempfile.gettempdir()) / "agent-loop-fixtures-import")
|
|
70
|
+
)
|
|
71
|
+
|
|
72
|
+
import latticeai.services.tool_dispatch as tool_dispatch # noqa: E402
|
|
73
|
+
import latticeai.tools as tools # noqa: E402
|
|
74
|
+
from latticeai.api.chat_contracts import AgentRequest # noqa: E402
|
|
75
|
+
from latticeai.core.agent import ( # noqa: E402
|
|
76
|
+
AgentRunContext,
|
|
77
|
+
AgentState,
|
|
78
|
+
SingleAgentRuntime,
|
|
79
|
+
)
|
|
80
|
+
from latticeai.core.agent.deps import AgentDeps # noqa: E402
|
|
81
|
+
from latticeai.core.agent_helpers import ( # noqa: E402
|
|
82
|
+
PhaseBudgets,
|
|
83
|
+
TranscriptBudget,
|
|
84
|
+
_truncate_strings,
|
|
85
|
+
artifact_checklist,
|
|
86
|
+
compact_transcript,
|
|
87
|
+
extract_action_details,
|
|
88
|
+
files_written,
|
|
89
|
+
filter_learnings,
|
|
90
|
+
normalize_plan,
|
|
91
|
+
requirement_coverage,
|
|
92
|
+
)
|
|
93
|
+
from latticeai.core.agent_profiles import ( # noqa: E402
|
|
94
|
+
COMPACT_MAX_PARAMS_B,
|
|
95
|
+
model_size_b,
|
|
96
|
+
profile_for_model,
|
|
97
|
+
)
|
|
98
|
+
from latticeai.core.agent_trace import LoopTrace # noqa: E402
|
|
99
|
+
from latticeai.core.file_generation import ( # noqa: E402
|
|
100
|
+
infer_file_target,
|
|
101
|
+
infer_project_manifest,
|
|
102
|
+
sanitize_write_content,
|
|
103
|
+
)
|
|
104
|
+
from latticeai.core.run_store import ( # noqa: E402
|
|
105
|
+
restore_run_context,
|
|
106
|
+
serialize_run_context,
|
|
107
|
+
)
|
|
108
|
+
from latticeai.core.tool_registry import ( # noqa: E402
|
|
109
|
+
FILE_CREATE_ACTIONS,
|
|
110
|
+
LOCAL_WRITE_BLOCKED_PREFIXES,
|
|
111
|
+
SCOPED_KNOWLEDGE_TOOLS,
|
|
112
|
+
TOOL_GOVERNANCE,
|
|
113
|
+
TOOL_GOVERNANCE_DEFAULT,
|
|
114
|
+
)
|
|
115
|
+
from latticeai.tools.documents import document_output_target # noqa: E402
|
|
116
|
+
|
|
117
|
+
FIXTURE_DIR = REPO_ROOT / "rust" / "fixtures" / "agent_loop"
|
|
118
|
+
GOLDEN_DIR = FIXTURE_DIR / "golden"
|
|
119
|
+
SCHEMA = "agent-loop-parity/v1"
|
|
120
|
+
|
|
121
|
+
DECODER_DETAIL_PREFIX = "Agent did not return valid JSON: "
|
|
122
|
+
ROOT_PLACEHOLDER = "<AGENT_ROOT>"
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
# ── normalisation ─────────────────────────────────────────────────────────────
|
|
126
|
+
def normalize(value: Any, root: Path) -> Any:
|
|
127
|
+
"""Apply the four machine-independence rules, recursively."""
|
|
128
|
+
if isinstance(value, str):
|
|
129
|
+
text = value.replace(str(root), ROOT_PLACEHOLDER)
|
|
130
|
+
if text.startswith(DECODER_DETAIL_PREFIX):
|
|
131
|
+
return f"{DECODER_DETAIL_PREFIX}<decoder-detail>"
|
|
132
|
+
return text
|
|
133
|
+
if isinstance(value, dict):
|
|
134
|
+
return {
|
|
135
|
+
key: normalize(item, root)
|
|
136
|
+
for key, item in value.items()
|
|
137
|
+
if key not in ("at", "stderr")
|
|
138
|
+
}
|
|
139
|
+
if isinstance(value, list):
|
|
140
|
+
return [normalize(item, root) for item in value]
|
|
141
|
+
return value
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
@contextmanager
|
|
145
|
+
def use_workspace(root: Path) -> Iterator[Path]:
|
|
146
|
+
"""Point the real tool registry at ``root`` for the duration.
|
|
147
|
+
|
|
148
|
+
Two modules hold the constant: ``latticeai.tools`` defines it and
|
|
149
|
+
``latticeai.services.tool_dispatch`` imported the value, so patching one
|
|
150
|
+
would leave the snapshot/rollback ports reading the real workspace.
|
|
151
|
+
"""
|
|
152
|
+
root = Path(root)
|
|
153
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
154
|
+
resolved = root.resolve()
|
|
155
|
+
previous = (tools.AGENT_ROOT, tool_dispatch.AGENT_ROOT)
|
|
156
|
+
tools.AGENT_ROOT = resolved
|
|
157
|
+
tool_dispatch.AGENT_ROOT = resolved
|
|
158
|
+
try:
|
|
159
|
+
yield resolved
|
|
160
|
+
finally:
|
|
161
|
+
tools.AGENT_ROOT, tool_dispatch.AGENT_ROOT = previous
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# ── the deterministic helper grids ────────────────────────────────────────────
|
|
165
|
+
#: Raw model outputs chosen for the rung of the tolerance chain each one reaches.
|
|
166
|
+
RAW_ACTIONS: Dict[str, str] = {
|
|
167
|
+
"clean": '{"action": "final", "message": "done"}',
|
|
168
|
+
"clean_with_args": '{"thoughts": "t", "action": "write_file", "args": {"path": "a.md"}}',
|
|
169
|
+
"fence_json": '```json\n{"action": "read_file", "args": {"path": "a.md"}}\n```',
|
|
170
|
+
"fence_bare": '```\n{"action": "final"}\n```',
|
|
171
|
+
"fence_with_prose": 'Sure!\n```json\n{"action": "final"}\n```\nHope that helps.',
|
|
172
|
+
"fence_multiline": '```json\n{\n "action": "final",\n "message": "ok"\n}\n```',
|
|
173
|
+
"think_then_json": '<think>hmm {"action": "wrong"}</think>\n{"action": "right"}',
|
|
174
|
+
"thinking_tag": '<thinking>plan</thinking>{"action": "final"}',
|
|
175
|
+
"reasoning_tag": '<reasoning>why</reasoning>\n{"action": "final"}',
|
|
176
|
+
"think_uppercase": '<THINK>x</THINK>{"action": "final"}',
|
|
177
|
+
"think_unclosed": '<think>never closed {"action": "final"}',
|
|
178
|
+
"think_mismatched": '<think>{"action": "a"}</reasoning>',
|
|
179
|
+
"slice_prefix": 'I will call: {"action": "write_file"}',
|
|
180
|
+
"slice_suffix": '{"action": "write_file"} — that is the call.',
|
|
181
|
+
"slice_both": 'Calling {"action": "final", "message": "완료"} now.',
|
|
182
|
+
"slice_nested": 'note {"action": "a", "args": {"b": {"c": 1}}} end',
|
|
183
|
+
"trailing_comma_object": '{"action": "final", "message": "x",}',
|
|
184
|
+
"trailing_comma_array": '{"action": "a", "args": {"items": [1, 2,]}}',
|
|
185
|
+
"trailing_comma_nested": '{"action": "a", "args": {"x": 1,},}',
|
|
186
|
+
"python_literal": "{'action': 'write_file', 'args': {'path': 'a.md'}}",
|
|
187
|
+
"python_literal_true": "{'action': 'a', 'ok': True, 'bad': False, 'none': None}",
|
|
188
|
+
"python_literal_trailing": "{'action': 'final', 'message': 'hi',}",
|
|
189
|
+
"python_literal_nested": "{'action': 'a', 'args': {'items': [1, 2.5, 'x']}}",
|
|
190
|
+
"python_literal_escapes": "{'action': 'a', 'note': 'line\\nbreak'}",
|
|
191
|
+
"python_literal_not_dict": "('a', 'b')",
|
|
192
|
+
"broken_prose": "I think we should start by reading the notes.",
|
|
193
|
+
"broken_empty": "",
|
|
194
|
+
"broken_whitespace": " \n ",
|
|
195
|
+
"broken_unclosed": '{"action": "final"',
|
|
196
|
+
"broken_missing_value": '{"action": }',
|
|
197
|
+
"broken_unquoted_key": "{action: 1}",
|
|
198
|
+
"no_action_key": '{"thoughts": "no action here"}',
|
|
199
|
+
"not_an_object": "[1, 2, 3]",
|
|
200
|
+
"bare_number": "42",
|
|
201
|
+
"bare_string": '"just text"',
|
|
202
|
+
"korean_prose_slice": '작업 계획: {"action": "final", "message": "완료"} 끝.',
|
|
203
|
+
"double_object": '{"action": "a"} {"action": "b"}',
|
|
204
|
+
"unicode_thoughts": '{"action": "final", "thoughts": "가나다라마바사"}',
|
|
205
|
+
}
|
|
206
|
+
|
|
207
|
+
#: Plans chosen for the seven normalisation rules and their interactions.
|
|
208
|
+
PLAN_CASES: Dict[str, Dict[str, Any]] = {
|
|
209
|
+
"complete": {
|
|
210
|
+
"plan": {"goal": "g", "steps": [{"action": "read_file", "args": {"path": "a.md"}}],
|
|
211
|
+
"estimated_steps": 1, "requires_approval": False, "rollback_strategy": "none"},
|
|
212
|
+
"message": "read a.md",
|
|
213
|
+
},
|
|
214
|
+
"not_an_object": {"plan": ["nope"], "message": "hi"},
|
|
215
|
+
"null_plan": {"plan": None, "message": "hi"},
|
|
216
|
+
"string_plan": {"plan": "a plan", "message": "hi"},
|
|
217
|
+
"blank_goal": {"plan": {"goal": " ", "steps": []}, "message": "do the thing"},
|
|
218
|
+
"missing_goal": {"plan": {"steps": []}, "message": "do the thing"},
|
|
219
|
+
"numeric_goal": {"plan": {"goal": 5, "steps": []}, "message": "hi"},
|
|
220
|
+
"junk_steps": {
|
|
221
|
+
"plan": {"goal": "g", "steps": ["x", {"no_action": 1}, {"action": ""},
|
|
222
|
+
{"action": "read_file"}]},
|
|
223
|
+
"message": "g",
|
|
224
|
+
},
|
|
225
|
+
"steps_not_a_list": {"plan": {"goal": "g", "steps": "read a file"}, "message": "g"},
|
|
226
|
+
"empty_steps": {"plan": {"goal": "g", "steps": []}, "message": "g"},
|
|
227
|
+
"manifest_empty_plan": {"plan": {"goal": "g", "steps": []},
|
|
228
|
+
"message": "todo 앱 html css js 만들어줘"},
|
|
229
|
+
"manifest_partial": {
|
|
230
|
+
"plan": {"goal": "g", "steps": [{"action": "write_file", "args": {"path": "index.html"}}]},
|
|
231
|
+
"message": "todo 앱 html css js 만들어줘",
|
|
232
|
+
},
|
|
233
|
+
"manifest_covered": {
|
|
234
|
+
"plan": {"goal": "g", "steps": [
|
|
235
|
+
{"action": "write_file", "args": {"path": "page.HTML"}},
|
|
236
|
+
{"action": "write_file", "args": {"path": "a.css"}},
|
|
237
|
+
{"action": "generate_file", "args": {"path": "b.js"}}]},
|
|
238
|
+
"message": "todo 앱 html css js 만들어줘",
|
|
239
|
+
},
|
|
240
|
+
"manifest_with_read": {
|
|
241
|
+
"plan": {"goal": "g", "steps": [{"action": "read_file", "args": {"path": "spec.md"}},
|
|
242
|
+
{"action": "write_file", "args": {"path": "index.html"}}]},
|
|
243
|
+
"message": "todo 앱 html css js 만들어줘",
|
|
244
|
+
},
|
|
245
|
+
"manifest_react": {"plan": {}, "message": "react 로 todo 앱 만들어줘"},
|
|
246
|
+
"manifest_python": {"plan": {}, "message": "mytool 패키지 파이썬으로 만들어줘"},
|
|
247
|
+
"heuristic_single_file": {"plan": {}, "message": "html 파일 만들어줘"},
|
|
248
|
+
"heuristic_long_message": {"plan": {}, "message": "html 파일 만들어줘 " + "가" * 200},
|
|
249
|
+
"estimated_string": {"plan": {"goal": "g", "estimated_steps": "4"}, "message": "g"},
|
|
250
|
+
"estimated_float": {"plan": {"goal": "g", "estimated_steps": 3.7}, "message": "g"},
|
|
251
|
+
"estimated_invalid": {"plan": {"goal": "g", "estimated_steps": "many"}, "message": "g"},
|
|
252
|
+
"estimated_list": {"plan": {"goal": "g", "estimated_steps": [3]}, "message": "g"},
|
|
253
|
+
"coerced_tail": {"plan": {"goal": "g", "requires_approval": "yes", "rollback_strategy": 7},
|
|
254
|
+
"message": "g"},
|
|
255
|
+
"rollback_kept": {"plan": {"goal": "g", "rollback_strategy": "git"}, "message": "g"},
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
#: Requests the two inference functions are asked about.
|
|
259
|
+
INFERENCE_MESSAGES: List[str] = [
|
|
260
|
+
"html 파일 만들어줘", "write me a python script", "csv 저장해줘", "html이 뭐야?",
|
|
261
|
+
"만들어줘", "", " ", "html과 css 만들어줘", "html and css 만들어줘",
|
|
262
|
+
"todo 앱 html+css+js로 만들어줘", "웹페이지 js로 만들어줘", "웹페이지 css로 만들어줘",
|
|
263
|
+
"react 로 todo 앱 만들어줘", "리액트로 만들어줘", "vite 앱 만들어줘",
|
|
264
|
+
"mytool 패키지 파이썬으로 만들어줘", "my-tool 패키지 파이썬으로 생성",
|
|
265
|
+
"파이썬 패키지 만들어줘", "index.html 이랑 style.css 만들어줘",
|
|
266
|
+
"웹사이트 css js 만들어줘", "마크다운 파일 작성해줘", "yaml 만들어줘",
|
|
267
|
+
]
|
|
268
|
+
|
|
269
|
+
#: ``(tool_name, filename)`` pairs for the document-target resolver. Chosen for
|
|
270
|
+
#: the branches they reach, not for realism: a tool that is not a document
|
|
271
|
+
#: creator, each of the four that are, a name that already carries its suffix,
|
|
272
|
+
#: one that carries the wrong one, a path that must be reduced to its basename,
|
|
273
|
+
#: characters the sanitizer replaces, and the empty names that fall back to
|
|
274
|
+
#: ``artifact<suffix>``. The Rust port takes the basename with its own
|
|
275
|
+
#: ``path_name`` and has its own empty-name fallback, so those two are exactly
|
|
276
|
+
#: where the two implementations could disagree unnoticed.
|
|
277
|
+
DOCUMENT_TARGET_CASES: List[Dict[str, str]] = [
|
|
278
|
+
{"tool": "write_file", "filename": "notes.md"},
|
|
279
|
+
{"tool": "create_docx", "filename": "report.docx"},
|
|
280
|
+
{"tool": "create_docx", "filename": "report"},
|
|
281
|
+
{"tool": "create_docx", "filename": "report.pdf"},
|
|
282
|
+
{"tool": "create_xlsx", "filename": "budget.xlsx"},
|
|
283
|
+
{"tool": "create_pptx", "filename": "deck"},
|
|
284
|
+
{"tool": "create_pdf", "filename": "invoice"},
|
|
285
|
+
{"tool": "create_pdf", "filename": "sub/dir/invoice.pdf"},
|
|
286
|
+
{"tool": "create_pdf", "filename": "../../escape.pdf"},
|
|
287
|
+
{"tool": "create_docx", "filename": "회의 기록.docx"},
|
|
288
|
+
{"tool": "create_docx", "filename": "a/b*c?d.docx"},
|
|
289
|
+
{"tool": "create_docx", "filename": ""},
|
|
290
|
+
{"tool": "create_pptx", "filename": " "},
|
|
291
|
+
{"tool": "create_xlsx", "filename": "REPORT.XLSX"},
|
|
292
|
+
{"tool": "create_pdf", "filename": ".pdf"},
|
|
293
|
+
{"tool": "unknown_tool", "filename": "x.docx"},
|
|
294
|
+
]
|
|
295
|
+
|
|
296
|
+
#: Model ids for the profile dial. The interesting ones are the quantization
|
|
297
|
+
#: suffixes (``4bit`` is not a parameter count), the multi-size ids where the
|
|
298
|
+
#: *smallest* wins, and the boundary at ``COMPACT_MAX_PARAMS_B`` itself.
|
|
299
|
+
PROFILE_MODEL_IDS: List[str] = [
|
|
300
|
+
"mlx-community/gemma-4-12B-it-4bit",
|
|
301
|
+
"qwen2.5-1.5b",
|
|
302
|
+
"llama-3.2-3B",
|
|
303
|
+
"mlx-community/Llama-3.2-3B-Instruct-4bit",
|
|
304
|
+
"phi-4-mini-3.8b-8bit",
|
|
305
|
+
"some-model-4b",
|
|
306
|
+
"some-model-4.0b",
|
|
307
|
+
"some-model-4.1b",
|
|
308
|
+
"gpt-4o",
|
|
309
|
+
"claude-sonnet",
|
|
310
|
+
"",
|
|
311
|
+
" ",
|
|
312
|
+
"model-8bit",
|
|
313
|
+
"abc123b",
|
|
314
|
+
"7b-and-1.5b-mixed",
|
|
315
|
+
"Model-70B-Instruct",
|
|
316
|
+
]
|
|
317
|
+
|
|
318
|
+
#: ``LATTICEAI_AGENT_PROFILE`` values, including the two that must fall through
|
|
319
|
+
#: to the size heuristic rather than failing the run.
|
|
320
|
+
PROFILE_OVERRIDES: List[str] = ["", "standard", "compact", "COMPACT", "nonsense"]
|
|
321
|
+
|
|
322
|
+
#: Transcripts the artifact/coverage helpers are asked about.
|
|
323
|
+
TRANSCRIPT_CASES: Dict[str, Dict[str, Any]] = {
|
|
324
|
+
"empty": {"message": "todo 앱 html css js 만들어줘", "transcript": []},
|
|
325
|
+
"one_write": {
|
|
326
|
+
"message": "todo 앱 html css js 만들어줘",
|
|
327
|
+
"transcript": [{"state": "EXECUTING", "action": "write_file",
|
|
328
|
+
"args": {"path": "index.html"},
|
|
329
|
+
"result": {"path": "index.html", "bytes": 10}}],
|
|
330
|
+
},
|
|
331
|
+
"all_written": {
|
|
332
|
+
"message": "todo 앱 html css js 만들어줘",
|
|
333
|
+
"transcript": [
|
|
334
|
+
{"state": "EXECUTING", "action": "write_file", "args": {"path": "index.html"},
|
|
335
|
+
"result": {"path": "index.html", "bytes": 10}},
|
|
336
|
+
{"state": "EXECUTING", "action": "write_file", "args": {"path": "sub/STYLE.CSS"},
|
|
337
|
+
"result": {"path": "sub/STYLE.CSS", "bytes": 3}},
|
|
338
|
+
{"state": "EXECUTING", "action": "write_file", "args": {"path": "app.js"},
|
|
339
|
+
"result": {"path": "app.js", "bytes": 3},
|
|
340
|
+
"content_sanitize": {"sanitized": True, "repaired": True}},
|
|
341
|
+
],
|
|
342
|
+
},
|
|
343
|
+
"blocked_and_repeated": {
|
|
344
|
+
"message": "make a note",
|
|
345
|
+
"transcript": [
|
|
346
|
+
{"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
|
|
347
|
+
"error": "BLOCKED: nope"},
|
|
348
|
+
{"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
|
|
349
|
+
"result": {"path": "a.md", "bytes": 2}},
|
|
350
|
+
{"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
|
|
351
|
+
"result": {"path": "a.md", "bytes": 2}},
|
|
352
|
+
{"state": "VERIFYING", "action": "write_file", "result": {"path": "ignored.md"}},
|
|
353
|
+
{"state": "EXECUTING", "action": "read_file", "result": {"path": "skip.md"}},
|
|
354
|
+
],
|
|
355
|
+
},
|
|
356
|
+
"requirements_listed": {
|
|
357
|
+
"message": "만들어줘:\n- 다크모드\n* dark mode\n1. 검색 기능\n2) 필터\nfree prose\n- ab",
|
|
358
|
+
"transcript": [],
|
|
359
|
+
},
|
|
360
|
+
"requirements_capped": {
|
|
361
|
+
"message": "\n".join(f"- item number {index}" for index in range(15)),
|
|
362
|
+
"transcript": [],
|
|
363
|
+
},
|
|
364
|
+
"proposed_step": {
|
|
365
|
+
"message": "make a note",
|
|
366
|
+
"transcript": [{"state": "EXECUTING", "action": "write_file", "args": {"path": "a.md"},
|
|
367
|
+
"result": {"proposed": True, "proposal_id": "p1"}}],
|
|
368
|
+
},
|
|
369
|
+
}
|
|
370
|
+
|
|
371
|
+
#: Values `_truncate_strings` is asked about, with the cap each one uses.
|
|
372
|
+
TRUNCATE_CASES: List[Dict[str, Any]] = [
|
|
373
|
+
{"key": "short", "limit": 700, "value": {"a": "hello"}},
|
|
374
|
+
{"key": "exact", "limit": 5, "value": "abcde"},
|
|
375
|
+
{"key": "over", "limit": 5, "value": "abcdefgh"},
|
|
376
|
+
{"key": "korean", "limit": 5, "value": "가" * 10},
|
|
377
|
+
{"key": "nested", "limit": 3, "value": {"a": ["abcdef", {"b": "xyz!"}], "n": 5, "t": True}},
|
|
378
|
+
{"key": "null_and_float", "limit": 2, "value": {"a": None, "b": 1.5, "c": []}},
|
|
379
|
+
]
|
|
380
|
+
|
|
381
|
+
LEARNING_CASES: List[List[Any]] = [
|
|
382
|
+
["short", "파일을 만들었습니다", "Successfully created the file",
|
|
383
|
+
"Vite needs the entry script tag before </body> or the app never mounts",
|
|
384
|
+
"VITE NEEDS THE ENTRY SCRIPT TAG BEFORE </BODY> OR THE APP NEVER MOUNTS", None],
|
|
385
|
+
["작업을 완료했습니다", "task was completed", "file was created",
|
|
386
|
+
"Successfully created the file, but the CSS never loaded because the path was wrong"],
|
|
387
|
+
[],
|
|
388
|
+
[123456789012345, " padded learning that is long enough "],
|
|
389
|
+
]
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
#: A root that matches nothing, for grids that never carry a path.
|
|
393
|
+
NO_ROOT = Path("/__no_workspace_in_this_grid__")
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def helper_rows() -> Dict[str, Any]:
|
|
397
|
+
"""Every deterministic helper, over its grid."""
|
|
398
|
+
actions = []
|
|
399
|
+
for key, raw in RAW_ACTIONS.items():
|
|
400
|
+
try:
|
|
401
|
+
action, repairs = extract_action_details(raw)
|
|
402
|
+
actions.append({"key": key, "raw": raw, "ok": True,
|
|
403
|
+
"action": action, "repairs": repairs})
|
|
404
|
+
except ValueError as exc:
|
|
405
|
+
actions.append({"key": key, "raw": raw, "ok": False, "error": str(exc)})
|
|
406
|
+
|
|
407
|
+
plans = []
|
|
408
|
+
for key, case in PLAN_CASES.items():
|
|
409
|
+
plan, fixes = normalize_plan(copy.deepcopy(case["plan"]), case["message"])
|
|
410
|
+
plans.append({"key": key, "plan": case["plan"], "message": case["message"],
|
|
411
|
+
"normalized": plan, "fixes": fixes})
|
|
412
|
+
|
|
413
|
+
inference = [
|
|
414
|
+
{"message": message,
|
|
415
|
+
"file_target": infer_file_target(message),
|
|
416
|
+
"manifest": infer_project_manifest(message)}
|
|
417
|
+
for message in INFERENCE_MESSAGES
|
|
418
|
+
]
|
|
419
|
+
|
|
420
|
+
transcripts = []
|
|
421
|
+
for key, case in TRANSCRIPT_CASES.items():
|
|
422
|
+
transcripts.append({
|
|
423
|
+
"key": key,
|
|
424
|
+
"message": case["message"],
|
|
425
|
+
"transcript": case["transcript"],
|
|
426
|
+
"files_written": files_written(case["transcript"], FILE_CREATE_ACTIONS),
|
|
427
|
+
"artifact_checklist": artifact_checklist(case["transcript"], FILE_CREATE_ACTIONS),
|
|
428
|
+
"requirement_coverage": requirement_coverage(
|
|
429
|
+
case["message"], case["transcript"], FILE_CREATE_ACTIONS
|
|
430
|
+
),
|
|
431
|
+
"compact_window_2": compact_transcript(case["transcript"], window=2, result_chars=40),
|
|
432
|
+
})
|
|
433
|
+
|
|
434
|
+
truncated = [
|
|
435
|
+
{"key": case["key"], "limit": case["limit"], "value": case["value"],
|
|
436
|
+
"truncated": _truncate_strings(case["value"], case["limit"])}
|
|
437
|
+
for case in TRUNCATE_CASES
|
|
438
|
+
]
|
|
439
|
+
|
|
440
|
+
learnings = [
|
|
441
|
+
{"input": case, "kept": filter_learnings(case)} for case in LEARNING_CASES
|
|
442
|
+
]
|
|
443
|
+
|
|
444
|
+
documents = [
|
|
445
|
+
{**case, "target": document_output_target(case["tool"], case["filename"])}
|
|
446
|
+
for case in DOCUMENT_TARGET_CASES
|
|
447
|
+
]
|
|
448
|
+
|
|
449
|
+
# ``env`` is passed explicitly: the Rust twin reads the ambient process
|
|
450
|
+
# environment, and a golden that inherited this machine's would be a
|
|
451
|
+
# machine-specific value in a committed fixture.
|
|
452
|
+
profiles = []
|
|
453
|
+
for override in PROFILE_OVERRIDES:
|
|
454
|
+
env = {"LATTICEAI_AGENT_PROFILE": override} if override else {}
|
|
455
|
+
for model_id in PROFILE_MODEL_IDS:
|
|
456
|
+
profiles.append({
|
|
457
|
+
"override": override,
|
|
458
|
+
"model_id": model_id,
|
|
459
|
+
"size_b": model_size_b(model_id),
|
|
460
|
+
"profile": profile_for_model(model_id, env=env).__dict__,
|
|
461
|
+
})
|
|
462
|
+
profiles.append({
|
|
463
|
+
"override": "",
|
|
464
|
+
"model_id": None,
|
|
465
|
+
"size_b": model_size_b(""),
|
|
466
|
+
"profile": profile_for_model(None, env={}).__dict__,
|
|
467
|
+
})
|
|
468
|
+
|
|
469
|
+
return normalize({
|
|
470
|
+
"schema": SCHEMA,
|
|
471
|
+
"extract_action_details": actions,
|
|
472
|
+
"normalize_plan": plans,
|
|
473
|
+
"inference": inference,
|
|
474
|
+
"transcript_helpers": transcripts,
|
|
475
|
+
"truncate_strings": truncated,
|
|
476
|
+
"filter_learnings": learnings,
|
|
477
|
+
"document_targets": documents,
|
|
478
|
+
"agent_profiles": profiles,
|
|
479
|
+
"budgets": {
|
|
480
|
+
"phase": PhaseBudgets().__dict__,
|
|
481
|
+
"transcript": TranscriptBudget().__dict__,
|
|
482
|
+
},
|
|
483
|
+
}, NO_ROOT)
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
# ── the scripted ports ────────────────────────────────────────────────────────
|
|
487
|
+
class ScriptedLLM:
|
|
488
|
+
"""`deps.generate_as`, answering a queue of recorded completions."""
|
|
489
|
+
|
|
490
|
+
def __init__(self, outputs: List[str]) -> None:
|
|
491
|
+
self.outputs = list(outputs)
|
|
492
|
+
self.calls: List[Dict[str, Any]] = []
|
|
493
|
+
|
|
494
|
+
async def generate_as(self, model_id: Optional[str] = None, **kwargs: Any) -> str:
|
|
495
|
+
text = self.outputs.pop(0) if self.outputs else ""
|
|
496
|
+
self.calls.append({
|
|
497
|
+
"model_id": model_id,
|
|
498
|
+
"message": kwargs.get("message"),
|
|
499
|
+
"temperature": kwargs.get("temperature"),
|
|
500
|
+
"max_tokens": kwargs.get("max_tokens"),
|
|
501
|
+
})
|
|
502
|
+
return text
|
|
503
|
+
|
|
504
|
+
async def generate(self, **kwargs: Any) -> str:
|
|
505
|
+
return await self.generate_as(None, **kwargs)
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
class ScriptedGovernor:
|
|
509
|
+
"""`deps.change_governor`, answering one fixed verdict."""
|
|
510
|
+
|
|
511
|
+
governed_tools = frozenset({"write_file", "edit_file"})
|
|
512
|
+
|
|
513
|
+
def __init__(self, verdict: Optional[Dict[str, Any]]) -> None:
|
|
514
|
+
self.verdict = verdict
|
|
515
|
+
|
|
516
|
+
def review(self, name: str, args: Dict[str, Any], **kwargs: Any) -> Optional[Dict[str, Any]]:
|
|
517
|
+
return copy.deepcopy(self.verdict)
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def build_deps(
|
|
521
|
+
root: Path,
|
|
522
|
+
llm: ScriptedLLM,
|
|
523
|
+
*,
|
|
524
|
+
governor: Optional[ScriptedGovernor],
|
|
525
|
+
tool_calls: List[Dict[str, Any]],
|
|
526
|
+
audit: List[Dict[str, Any]],
|
|
527
|
+
) -> AgentDeps:
|
|
528
|
+
"""The real ports, wired to the throwaway workspace."""
|
|
529
|
+
registry = tools.DEFAULT_TOOL_REGISTRY
|
|
530
|
+
service = tool_dispatch.DEFAULT_TOOL_DISPATCH_SERVICE
|
|
531
|
+
|
|
532
|
+
def execute_tool(name: str, args: Dict[str, Any]) -> Dict[str, Any]:
|
|
533
|
+
record: Dict[str, Any] = {"tool": name, "args": copy.deepcopy(args)}
|
|
534
|
+
try:
|
|
535
|
+
result = tools.execute_tool(name, args)
|
|
536
|
+
except Exception as exc: # noqa: BLE001 — recorded, then re-raised
|
|
537
|
+
record["error"] = str(exc)
|
|
538
|
+
tool_calls.append(record)
|
|
539
|
+
raise
|
|
540
|
+
record["result"] = result
|
|
541
|
+
tool_calls.append(record)
|
|
542
|
+
return result
|
|
543
|
+
|
|
544
|
+
def record_audit(event: str, **details: Any) -> None:
|
|
545
|
+
audit.append({"event": event, **details})
|
|
546
|
+
|
|
547
|
+
return AgentDeps(
|
|
548
|
+
generate_as=llm.generate_as,
|
|
549
|
+
generate=llm.generate,
|
|
550
|
+
execute_tool=execute_tool,
|
|
551
|
+
policy_for=registry.policy_for,
|
|
552
|
+
risk_level=registry.risk_level,
|
|
553
|
+
check_role=lambda name, user: None,
|
|
554
|
+
tool_governance=TOOL_GOVERNANCE,
|
|
555
|
+
file_create_actions=FILE_CREATE_ACTIONS,
|
|
556
|
+
recent_chat_context=lambda **kwargs: "",
|
|
557
|
+
clear_history=lambda keep_last: {"ok": True, "kept": keep_last},
|
|
558
|
+
knowledge_save=lambda *a, **k: None,
|
|
559
|
+
audit=record_audit,
|
|
560
|
+
planner_prompt="",
|
|
561
|
+
executor_prompt="",
|
|
562
|
+
critic_prompt="",
|
|
563
|
+
memory_updater_prompt="",
|
|
564
|
+
agent_root=root,
|
|
565
|
+
rollback_file=service.rollback_file,
|
|
566
|
+
snapshot_file=service.snapshot_file,
|
|
567
|
+
restore_snapshot=service.restore_snapshot,
|
|
568
|
+
hooks=None,
|
|
569
|
+
change_governor=governor,
|
|
570
|
+
phase_budgets=PhaseBudgets(),
|
|
571
|
+
transcript_budget=TranscriptBudget(),
|
|
572
|
+
)
|
|
573
|
+
|
|
574
|
+
|
|
575
|
+
# ── verification mapping grid ─────────────────────────────────────────────────
|
|
576
|
+
#: `(verdict, next_state)` pairs the mapping table distinguishes.
|
|
577
|
+
VERDICT_PAIRS = [
|
|
578
|
+
("PASS", "DONE"), ("PASS", "COMPLETE"), ("PASS", "EXECUTING"), ("PASS", ""),
|
|
579
|
+
("FAIL", "DONE"), ("FAIL", "EXECUTING"), ("FAIL", "RETRY"), ("FAIL", "ROLLBACK"),
|
|
580
|
+
("FAIL", "FAILED"), ("FAIL", "SOMETHING_ELSE"), ("", "DONE"),
|
|
581
|
+
]
|
|
582
|
+
|
|
583
|
+
EVIDENCE_STEP = {"state": "EXECUTING", "action": "write_file", "args": {"path": "index.html"},
|
|
584
|
+
"result": {"path": "index.html", "bytes": 4}}
|
|
585
|
+
NO_EVIDENCE_STEP = {"state": "EXECUTING", "action": "final", "thoughts": "t"}
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
async def verification_rows(root: Path) -> List[Dict[str, Any]]:
|
|
589
|
+
"""The verdict mapping, from the real `verify()`."""
|
|
590
|
+
rows: List[Dict[str, Any]] = []
|
|
591
|
+
for verdict, next_state in VERDICT_PAIRS:
|
|
592
|
+
for evidence in (True, False):
|
|
593
|
+
for message in ("make a note", "todo 앱 html css js 만들어줘"):
|
|
594
|
+
for retry_count in (0, 3):
|
|
595
|
+
body = json.dumps(
|
|
596
|
+
{"action": "verdict", "verdict": verdict, "next_state": next_state,
|
|
597
|
+
"reason": "because", "corrections": ["be specific"], "confidence": 0.5},
|
|
598
|
+
ensure_ascii=False,
|
|
599
|
+
)
|
|
600
|
+
llm = ScriptedLLM([body])
|
|
601
|
+
ctx = AgentRunContext()
|
|
602
|
+
ctx.trace = LoopTrace()
|
|
603
|
+
ctx.retry_count = retry_count
|
|
604
|
+
ctx.transcript = [copy.deepcopy(
|
|
605
|
+
EVIDENCE_STEP if evidence else NO_EVIDENCE_STEP
|
|
606
|
+
)]
|
|
607
|
+
runtime = SingleAgentRuntime(build_deps(
|
|
608
|
+
root, llm, governor=None, tool_calls=[], audit=[]
|
|
609
|
+
))
|
|
610
|
+
request = AgentRequest(message=message)
|
|
611
|
+
await runtime.verify(ctx, request, "Korean", "owner@example.com", max_retry=3)
|
|
612
|
+
rows.append({
|
|
613
|
+
"verdict": verdict, "next_state": next_state, "evidence": evidence,
|
|
614
|
+
"message": message, "retry_count": retry_count,
|
|
615
|
+
"final_state": ctx.state.value,
|
|
616
|
+
"final_message": ctx.final_message,
|
|
617
|
+
"retry_count_after": ctx.retry_count,
|
|
618
|
+
"transcript": normalize(ctx.transcript, root),
|
|
619
|
+
})
|
|
620
|
+
# The unparseable critic: one strict retry, then fail closed.
|
|
621
|
+
for outputs, key in (
|
|
622
|
+
(["prose", "still prose"], "never_parses"),
|
|
623
|
+
(["prose", '{"action": "v", "verdict": "PASS", "next_state": "DONE", "reason": "r"}'],
|
|
624
|
+
"strict_retry_recovers"),
|
|
625
|
+
):
|
|
626
|
+
llm = ScriptedLLM(list(outputs))
|
|
627
|
+
ctx = AgentRunContext()
|
|
628
|
+
ctx.trace = LoopTrace()
|
|
629
|
+
ctx.transcript = [copy.deepcopy(EVIDENCE_STEP)]
|
|
630
|
+
runtime = SingleAgentRuntime(build_deps(root, llm, governor=None, tool_calls=[], audit=[]))
|
|
631
|
+
await runtime.verify(ctx, AgentRequest(message="make a note"), "Korean", "owner", max_retry=3)
|
|
632
|
+
rows.append({
|
|
633
|
+
"verdict": key, "next_state": "", "evidence": True, "message": "make a note",
|
|
634
|
+
"retry_count": 0, "final_state": ctx.state.value,
|
|
635
|
+
"final_message": ctx.final_message, "retry_count_after": ctx.retry_count,
|
|
636
|
+
"transcript": normalize(ctx.transcript, root),
|
|
637
|
+
"llm_calls": len(llm.calls),
|
|
638
|
+
"temperatures": [call["temperature"] for call in llm.calls],
|
|
639
|
+
})
|
|
640
|
+
return rows
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
# ── run-store round trips ─────────────────────────────────────────────────────
|
|
644
|
+
def run_store_rows() -> List[Dict[str, Any]]:
|
|
645
|
+
"""`serialize_run_context` / `restore_run_context`, field for field."""
|
|
646
|
+
rows: List[Dict[str, Any]] = []
|
|
647
|
+
|
|
648
|
+
full = AgentRunContext()
|
|
649
|
+
full.state = AgentState.WAITING_APPROVAL
|
|
650
|
+
full.plan = {"goal": "g", "steps": [{"action": "write_file"}]}
|
|
651
|
+
full.transcript = [{"state": "PLANNING", "goal": "g"}]
|
|
652
|
+
full.retry_count = 2
|
|
653
|
+
full.state_history = ["PLANNING", "WAITING_APPROVAL"]
|
|
654
|
+
full.corrections = ["reply with JSON"]
|
|
655
|
+
full.final_message = "paused"
|
|
656
|
+
full.rollback_log = [{"path": "a.md", "existed": False}]
|
|
657
|
+
full.executing_model = "m-exec"
|
|
658
|
+
full.reviewing_model = "m-review"
|
|
659
|
+
full.approved_by_human = True
|
|
660
|
+
full.permission_mode = "trusted"
|
|
661
|
+
full.trace = LoopTrace(clock=lambda: "PINNED")
|
|
662
|
+
full.trace.llm_call("plan", model="m-exec")
|
|
663
|
+
full.trace.repair("plan", repairs=["fence"])
|
|
664
|
+
rows.append({"key": "full", "serialized": serialize_run_context(full)})
|
|
665
|
+
|
|
666
|
+
empty = AgentRunContext()
|
|
667
|
+
rows.append({"key": "empty", "serialized": serialize_run_context(empty)})
|
|
668
|
+
|
|
669
|
+
for key, payload in (
|
|
670
|
+
("unknown_state", {"state": "SOMETHING_NEW"}),
|
|
671
|
+
("no_state", {}),
|
|
672
|
+
("null_state", {"state": None}),
|
|
673
|
+
("blank_mode", {"state": "EXECUTING", "permission_mode": ""}),
|
|
674
|
+
("kept_mode", {"state": "EXECUTING", "permission_mode": "bypass"}),
|
|
675
|
+
("truthy_approval", {"approved_by_human": 1}),
|
|
676
|
+
):
|
|
677
|
+
rows.append({"key": f"restore_{key}", "payload": payload,
|
|
678
|
+
"restored": serialize_run_context(restore_run_context(payload))})
|
|
679
|
+
|
|
680
|
+
for row in rows:
|
|
681
|
+
if "serialized" in row:
|
|
682
|
+
row["round_trip"] = serialize_run_context(restore_run_context(row["serialized"]))
|
|
683
|
+
return normalize(rows, NO_ROOT)
|
|
684
|
+
|
|
685
|
+
|
|
686
|
+
# ── end-to-end trajectories ───────────────────────────────────────────────────
|
|
687
|
+
def plan_json(goal: str, steps: List[Dict[str, Any]]) -> str:
|
|
688
|
+
return json.dumps(
|
|
689
|
+
{"action": "plan", "goal": goal, "steps": steps, "estimated_steps": max(1, len(steps)),
|
|
690
|
+
"requires_approval": False, "rollback_strategy": "none"},
|
|
691
|
+
ensure_ascii=False,
|
|
692
|
+
)
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def action_json(**payload: Any) -> str:
|
|
696
|
+
return json.dumps(payload, ensure_ascii=False)
|
|
697
|
+
|
|
698
|
+
|
|
699
|
+
def verdict_json(verdict: str, next_state: str, reason: str = "checked") -> str:
|
|
700
|
+
return json.dumps(
|
|
701
|
+
{"action": "verdict", "verdict": verdict, "next_state": next_state,
|
|
702
|
+
"reason": reason, "corrections": []},
|
|
703
|
+
ensure_ascii=False,
|
|
704
|
+
)
|
|
705
|
+
|
|
706
|
+
|
|
707
|
+
WRITE_STEP = [{"action": "write_file", "args": {"path": "note.md"}, "description": "the note"}]
|
|
708
|
+
|
|
709
|
+
#: Seven trajectories: between them they reach every terminal state by every
|
|
710
|
+
#: route the loop has, under all three permission modes.
|
|
711
|
+
SCENARIOS: Dict[str, Dict[str, Any]] = {
|
|
712
|
+
"clean_done_trusted": {
|
|
713
|
+
"mode": "trusted", "message": "make a note", "seed": {},
|
|
714
|
+
"governor_verdict": None,
|
|
715
|
+
"script": [
|
|
716
|
+
plan_json("make a note", WRITE_STEP),
|
|
717
|
+
action_json(thoughts="writing", action="write_file",
|
|
718
|
+
args={"path": "note.md", "content": "# Note\n\nhello\n"}),
|
|
719
|
+
action_json(action="final", message="파일을 만들었습니다."),
|
|
720
|
+
verdict_json("PASS", "DONE", "the file exists"),
|
|
721
|
+
],
|
|
722
|
+
},
|
|
723
|
+
"strict_proposal_pause": {
|
|
724
|
+
"mode": "strict", "message": "update the note", "seed": {"note.md": "original\n"},
|
|
725
|
+
"governor_verdict": {"decision": "proposed", "proposal": {"id": "prop-1"},
|
|
726
|
+
"classification": {"change_class": "mutation"}},
|
|
727
|
+
"script": [
|
|
728
|
+
plan_json("update the note", WRITE_STEP),
|
|
729
|
+
action_json(thoughts="rewriting", action="write_file",
|
|
730
|
+
args={"path": "note.md", "content": "rewritten\n"}),
|
|
731
|
+
action_json(action="final", message="제안으로 저장했습니다."),
|
|
732
|
+
verdict_json("PASS", "DONE", "staged for review"),
|
|
733
|
+
],
|
|
734
|
+
},
|
|
735
|
+
"parse_budget_exhaustion": {
|
|
736
|
+
"mode": "trusted", "message": "make a note", "seed": {},
|
|
737
|
+
"governor_verdict": None,
|
|
738
|
+
"script": [
|
|
739
|
+
plan_json("make a note", WRITE_STEP),
|
|
740
|
+
"I will now write the file for you.",
|
|
741
|
+
"Writing it now, one moment.",
|
|
742
|
+
"All done, I think.",
|
|
743
|
+
verdict_json("PASS", "DONE", "looks fine"),
|
|
744
|
+
],
|
|
745
|
+
},
|
|
746
|
+
"repeated_create_guard": {
|
|
747
|
+
"mode": "trusted", "message": "make a note", "seed": {},
|
|
748
|
+
"governor_verdict": None,
|
|
749
|
+
"script": [
|
|
750
|
+
plan_json("make a note", WRITE_STEP),
|
|
751
|
+
action_json(action="write_file", args={"path": "note.md", "content": "body\n"}),
|
|
752
|
+
action_json(action="write_file", args={"path": "note.md", "content": "body\n"}),
|
|
753
|
+
verdict_json("PASS", "DONE", "written once"),
|
|
754
|
+
],
|
|
755
|
+
},
|
|
756
|
+
"verify_retry_then_failed": {
|
|
757
|
+
"mode": "trusted", "message": "make a note", "seed": {},
|
|
758
|
+
"governor_verdict": None,
|
|
759
|
+
"script": [
|
|
760
|
+
plan_json("make a note", WRITE_STEP),
|
|
761
|
+
action_json(action="write_file", args={"path": "note.md", "content": "v1\n"}),
|
|
762
|
+
action_json(action="final", message="done"),
|
|
763
|
+
verdict_json("FAIL", "EXECUTING", "not good enough"),
|
|
764
|
+
action_json(action="final", message="done"),
|
|
765
|
+
verdict_json("FAIL", "EXECUTING", "still not"),
|
|
766
|
+
action_json(action="final", message="done"),
|
|
767
|
+
verdict_json("FAIL", "EXECUTING", "no"),
|
|
768
|
+
action_json(action="final", message="done"),
|
|
769
|
+
verdict_json("FAIL", "EXECUTING", "give up"),
|
|
770
|
+
],
|
|
771
|
+
},
|
|
772
|
+
"verify_pass_no_evidence": {
|
|
773
|
+
"mode": "trusted", "message": "tell me about the notes", "seed": {},
|
|
774
|
+
"governor_verdict": None,
|
|
775
|
+
"script": [
|
|
776
|
+
plan_json("answer the question", []),
|
|
777
|
+
action_json(action="final", message="여기 답변입니다."),
|
|
778
|
+
verdict_json("PASS", "DONE", "answered"),
|
|
779
|
+
],
|
|
780
|
+
},
|
|
781
|
+
"rollback_path": {
|
|
782
|
+
"mode": "trusted", "message": "update the note", "seed": {"note.md": "original\n"},
|
|
783
|
+
"governor_verdict": None,
|
|
784
|
+
"script": [
|
|
785
|
+
plan_json("update the note", WRITE_STEP),
|
|
786
|
+
action_json(action="write_file", args={"path": "note.md", "content": "broken\n"}),
|
|
787
|
+
action_json(action="final", message="done"),
|
|
788
|
+
verdict_json("FAIL", "ROLLBACK", "the change is wrong"),
|
|
789
|
+
],
|
|
790
|
+
},
|
|
791
|
+
"blocked_fail_closed_strict": {
|
|
792
|
+
"mode": "strict", "message": "delete the note", "seed": {"note.md": "original\n"},
|
|
793
|
+
"governor_verdict": None,
|
|
794
|
+
"script": [
|
|
795
|
+
# The plan itself is auto-approvable (a read); the *executor* then
|
|
796
|
+
# reaches for a destructive tool, which is where the gate fires.
|
|
797
|
+
plan_json("delete the note", [{"action": "read_file", "args": {"path": "note.md"},
|
|
798
|
+
"description": "look at it first"}]),
|
|
799
|
+
action_json(action="delete_file", args={"path": "note.md"}),
|
|
800
|
+
action_json(action="final", message="삭제하지 못했습니다."),
|
|
801
|
+
verdict_json("PASS", "DONE", "nothing was deleted"),
|
|
802
|
+
],
|
|
803
|
+
},
|
|
804
|
+
"blocked_breaker_bypass": {
|
|
805
|
+
"mode": "bypass", "message": "fix the hosts file", "seed": {},
|
|
806
|
+
"governor_verdict": None,
|
|
807
|
+
"script": [
|
|
808
|
+
plan_json("fix the hosts file", [{"action": "read_file", "args": {"path": "note.md"},
|
|
809
|
+
"description": "look first"}]),
|
|
810
|
+
# The registry rewrites a write aimed at a blocked system prefix
|
|
811
|
+
# into a destructive policy, and the breaker refuses it — in
|
|
812
|
+
# `bypass`, which is the whole point of a mode-invariant gate.
|
|
813
|
+
action_json(action="write_file", args={"path": "/etc/hosts", "content": "x\n"}),
|
|
814
|
+
action_json(action="final", message="시스템 파일은 건드리지 않았습니다."),
|
|
815
|
+
verdict_json("FAIL", "FAILED", "nothing was changed"),
|
|
816
|
+
],
|
|
817
|
+
},
|
|
818
|
+
"approval_pause_strict": {
|
|
819
|
+
"mode": "strict", "message": "run the tests", "seed": {},
|
|
820
|
+
"governor_verdict": None,
|
|
821
|
+
"pause_expected": True,
|
|
822
|
+
"script": [
|
|
823
|
+
plan_json("run the tests", [{"action": "run_command",
|
|
824
|
+
"args": {"command": "ls"}, "description": "list"}]),
|
|
825
|
+
],
|
|
826
|
+
},
|
|
827
|
+
}
|
|
828
|
+
|
|
829
|
+
|
|
830
|
+
async def trajectory(key: str, scenario: Dict[str, Any], base: Path) -> Dict[str, Any]:
|
|
831
|
+
"""Drive the real runtime through one scenario and record what happened."""
|
|
832
|
+
root = base / key / "agent_workspace"
|
|
833
|
+
with use_workspace(root) as resolved:
|
|
834
|
+
for name, body in scenario["seed"].items():
|
|
835
|
+
target = resolved / name
|
|
836
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
837
|
+
target.write_text(body, encoding="utf-8")
|
|
838
|
+
# A scripted write whose content the artifact pipeline would rewrite
|
|
839
|
+
# would make this trajectory untestable against the native loop, where
|
|
840
|
+
# sanitation is the worker's job. Prove it does not, at build time.
|
|
841
|
+
for output in scenario["script"]:
|
|
842
|
+
content = _scripted_write_content(output)
|
|
843
|
+
if content is not None:
|
|
844
|
+
_, meta = sanitize_write_content("note.md", content, user_request=scenario["message"])
|
|
845
|
+
if meta.get("sanitized"):
|
|
846
|
+
raise SystemExit(
|
|
847
|
+
f"scenario {key}: scripted content is rewritten by "
|
|
848
|
+
"sanitize_write_content; pick content the pipeline leaves alone"
|
|
849
|
+
)
|
|
850
|
+
|
|
851
|
+
tool_calls: List[Dict[str, Any]] = []
|
|
852
|
+
audit: List[Dict[str, Any]] = []
|
|
853
|
+
llm = ScriptedLLM(scenario["script"])
|
|
854
|
+
governor = (
|
|
855
|
+
ScriptedGovernor(scenario["governor_verdict"])
|
|
856
|
+
if scenario.get("governor_verdict") is not None or scenario["mode"] == "strict"
|
|
857
|
+
else ScriptedGovernor(None)
|
|
858
|
+
)
|
|
859
|
+
deps = build_deps(resolved, llm, governor=governor, tool_calls=tool_calls, audit=audit)
|
|
860
|
+
runtime = SingleAgentRuntime(deps)
|
|
861
|
+
|
|
862
|
+
request = AgentRequest(message=scenario["message"], user_email="owner@example.com")
|
|
863
|
+
ctx = AgentRunContext()
|
|
864
|
+
ctx.trace = LoopTrace()
|
|
865
|
+
ctx.permission_mode = scenario["mode"]
|
|
866
|
+
ctx.state = AgentState.PLANNING
|
|
867
|
+
ctx.state_history.append(ctx.state.value)
|
|
868
|
+
await runtime.plan(ctx, request, "Korean", "owner@example.com", model_id=None)
|
|
869
|
+
requirements = runtime.approval_requirements(ctx)
|
|
870
|
+
paused = bool(requirements["requires_approval"])
|
|
871
|
+
if paused:
|
|
872
|
+
ctx.state_history.append(AgentState.WAITING_APPROVAL.value)
|
|
873
|
+
else:
|
|
874
|
+
runtime.approve(ctx, "owner@example.com", approved_by_human=False)
|
|
875
|
+
await runtime.run_to_completion(
|
|
876
|
+
ctx, request, "Korean", "owner@example.com",
|
|
877
|
+
max(1, min(request.max_steps, 50)), 3,
|
|
878
|
+
)
|
|
879
|
+
if paused != bool(scenario.get("pause_expected")):
|
|
880
|
+
raise SystemExit(f"scenario {key}: pause={paused}, expected the opposite")
|
|
881
|
+
|
|
882
|
+
return {
|
|
883
|
+
"key": key,
|
|
884
|
+
"mode": scenario["mode"],
|
|
885
|
+
"message": scenario["message"],
|
|
886
|
+
"seed": scenario["seed"],
|
|
887
|
+
"scripted_llm": scenario["script"],
|
|
888
|
+
"governor_verdict": scenario["governor_verdict"],
|
|
889
|
+
"req": {"message": scenario["message"], "user_email": "owner@example.com",
|
|
890
|
+
"max_steps": request.max_steps, "temperature": request.temperature},
|
|
891
|
+
"paused": paused,
|
|
892
|
+
"approval_requirements": normalize(requirements, resolved),
|
|
893
|
+
"final_state": ctx.state.value,
|
|
894
|
+
"final_message": normalize(ctx.final_message, resolved),
|
|
895
|
+
"state_history": ctx.state_history,
|
|
896
|
+
"transcript": normalize(ctx.transcript, resolved),
|
|
897
|
+
"rollback_log": normalize(ctx.rollback_log, resolved),
|
|
898
|
+
"loop": ctx.trace.summary(),
|
|
899
|
+
"tool_calls": normalize(tool_calls, resolved),
|
|
900
|
+
"audit": normalize(audit, resolved),
|
|
901
|
+
"llm_calls": len(llm.calls),
|
|
902
|
+
"unused_script": len(llm.outputs),
|
|
903
|
+
}
|
|
904
|
+
|
|
905
|
+
|
|
906
|
+
def _scripted_write_content(output: str) -> Optional[str]:
|
|
907
|
+
"""The `content` of a scripted `write_file` action, when it is one."""
|
|
908
|
+
try:
|
|
909
|
+
payload = json.loads(output)
|
|
910
|
+
except (json.JSONDecodeError, TypeError):
|
|
911
|
+
return None
|
|
912
|
+
if not isinstance(payload, dict) or payload.get("action") != "write_file":
|
|
913
|
+
return None
|
|
914
|
+
content = (payload.get("args") or {}).get("content")
|
|
915
|
+
return content if isinstance(content, str) else None
|
|
916
|
+
|
|
917
|
+
|
|
918
|
+
def policy_payload() -> Dict[str, Any]:
|
|
919
|
+
"""The real registry, as the data the native loop takes as input."""
|
|
920
|
+
return {
|
|
921
|
+
"tools": {name: dict(policy) for name, policy in sorted(TOOL_GOVERNANCE.items())},
|
|
922
|
+
"default": dict(TOOL_GOVERNANCE_DEFAULT),
|
|
923
|
+
"blocked_write_prefixes": list(LOCAL_WRITE_BLOCKED_PREFIXES),
|
|
924
|
+
}
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def manifest_payload() -> Dict[str, Any]:
|
|
928
|
+
return {
|
|
929
|
+
"schema": SCHEMA,
|
|
930
|
+
"scenarios": sorted(SCENARIOS),
|
|
931
|
+
"raw_actions": sorted(RAW_ACTIONS),
|
|
932
|
+
"plan_cases": sorted(PLAN_CASES),
|
|
933
|
+
"normalization": [
|
|
934
|
+
"the absolute workspace root becomes <AGENT_ROOT>",
|
|
935
|
+
"keys named `at` are dropped (trace timestamps)",
|
|
936
|
+
"keys named `stderr` are dropped (git text is version-specific)",
|
|
937
|
+
f"a string starting with {DECODER_DETAIL_PREFIX!r} keeps only that prefix",
|
|
938
|
+
],
|
|
939
|
+
"constants": {
|
|
940
|
+
"file_create_actions": sorted(FILE_CREATE_ACTIONS),
|
|
941
|
+
"scoped_knowledge_tools": sorted(SCOPED_KNOWLEDGE_TOOLS),
|
|
942
|
+
"governed_tools": sorted(ScriptedGovernor.governed_tools),
|
|
943
|
+
"phase_budgets": PhaseBudgets().__dict__,
|
|
944
|
+
"transcript_budget": TranscriptBudget().__dict__,
|
|
945
|
+
"max_state_history": 200,
|
|
946
|
+
"max_retry": 3,
|
|
947
|
+
"compact_max_params_b": COMPACT_MAX_PARAMS_B,
|
|
948
|
+
},
|
|
949
|
+
}
|
|
950
|
+
|
|
951
|
+
|
|
952
|
+
async def build_async(base: Path) -> Dict[str, Any]:
|
|
953
|
+
"""Everything the goldens hold, as `{filename: payload}`."""
|
|
954
|
+
verify_root = base / "verify" / "agent_workspace"
|
|
955
|
+
with use_workspace(verify_root) as resolved:
|
|
956
|
+
verification = await verification_rows(resolved)
|
|
957
|
+
trajectories = [await trajectory(key, SCENARIOS[key], base) for key in sorted(SCENARIOS)]
|
|
958
|
+
return {
|
|
959
|
+
"manifest.json": manifest_payload(),
|
|
960
|
+
"policies.json": policy_payload(),
|
|
961
|
+
"helpers.json": helper_rows(),
|
|
962
|
+
"verification.json": {"schema": SCHEMA, "cases": verification},
|
|
963
|
+
"run_store.json": {"schema": SCHEMA, "cases": run_store_rows()},
|
|
964
|
+
"trajectories.json": {"schema": SCHEMA, "cases": trajectories},
|
|
965
|
+
}
|
|
966
|
+
|
|
967
|
+
|
|
968
|
+
def build(base: Optional[Path] = None) -> Dict[str, Any]:
|
|
969
|
+
"""Synchronous entry point, for the generator and the contract test alike."""
|
|
970
|
+
if base is None:
|
|
971
|
+
base = Path(tempfile.mkdtemp(prefix="agent-loop-fixtures-"))
|
|
972
|
+
return asyncio.run(build_async(Path(base)))
|
|
973
|
+
|
|
974
|
+
|
|
975
|
+
def write(payloads: Dict[str, Any]) -> List[str]:
|
|
976
|
+
GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
|
|
977
|
+
for name, payload in payloads.items():
|
|
978
|
+
(GOLDEN_DIR / name).write_text(
|
|
979
|
+
json.dumps(payload, ensure_ascii=False, indent=2, sort_keys=True) + "\n",
|
|
980
|
+
encoding="utf-8",
|
|
981
|
+
)
|
|
982
|
+
return sorted(payloads)
|
|
983
|
+
|
|
984
|
+
|
|
985
|
+
def main() -> int:
|
|
986
|
+
written = write(build())
|
|
987
|
+
for name in written:
|
|
988
|
+
path = GOLDEN_DIR / name
|
|
989
|
+
print(f"wrote {path.relative_to(REPO_ROOT)} ({path.stat().st_size:,} bytes)")
|
|
990
|
+
return 0
|
|
991
|
+
|
|
992
|
+
|
|
993
|
+
if __name__ == "__main__":
|
|
994
|
+
raise SystemExit(main())
|