ltcai 11.2.0 → 11.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -53
- package/docs/CHANGELOG.md +87 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +181 -0
- package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/api/index_jobs.py +145 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +14 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +421 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +2 -0
- package/scripts/chunking_parity_corpus.py +449 -0
- package/scripts/generate_agent_parity_fixtures.py +752 -0
- package/scripts/generate_chunking_parity_fixtures.py +259 -0
- package/scripts/generate_rust_parity_fixtures.py +997 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +42 -2
- package/src-tauri/Cargo.lock +404 -3
- package/src-tauri/Cargo.toml +13 -1
- package/src-tauri/src/backend.rs +460 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +109 -399
- package/src-tauri/src/topology.rs +356 -0
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-CWnxSCgN.js +1 -0
- package/static/app/assets/AdminConsole-BEQYU6kF.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-DWu1BhFg.js} +2 -2
- package/static/app/assets/BrainHome-95Hilr9R.js +2 -0
- package/static/app/assets/BrainSignals-QdeqCpAF.js +1 -0
- package/static/app/assets/Capture-BHpCxnzb.js +1 -0
- package/static/app/assets/Chronicle-B4xYKoed.js +1 -0
- package/static/app/assets/CommandPalette-BVXnttSz.js +1 -0
- package/static/app/assets/Library-DgYcHome.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-CrJLDbf7.js} +1 -1
- package/static/app/assets/ProductFlow-DFlScKoJ.js +1 -0
- package/static/app/assets/ReviewCard-Cy5f48Pj.js +3 -0
- package/static/app/assets/System-NF8IfhTa.js +1 -0
- package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-CucuhLhm.js} +1 -1
- package/static/app/assets/brain-BBnSryW_.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-C2GUj2Ai.js} +1 -1
- package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-CbkWzBmG.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-7lEaqHdJ.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DAlCXlIy.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-RNhuuJwh.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CLW4odzM.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-NKEiDIAJ.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-DMurvUuR.js +10 -0
- package/static/app/assets/input-D2UhPC1X.js +1 -0
- package/static/app/assets/link-2-6amKbP_P.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-Cu9TZtdR.js} +1 -1
- package/static/app/assets/primitives-gPsccucr.js +1 -0
- package/static/app/assets/search-Cj_TKk_2.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-Bau7KkPq.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-BufNYypi.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-BQnVWhYs.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-B3_w60si.js} +1 -1
- package/static/app/assets/useMutation-BHhCflT6.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-rBWfI-5t.js} +1 -1
- package/static/app/assets/utils-V_5-wxr5.js +4 -0
- package/static/app/assets/workspace-K1zjYUHj.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,752 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build the committed Python↔Rust **safety kernel** parity fixtures (v11.5.0).
|
|
3
|
+
|
|
4
|
+
``rust/lattice-agent`` owns the diagram's Tools/Sandbox/Permission labels: it
|
|
5
|
+
decides, natively, whether a tool call may run and whether a command string is
|
|
6
|
+
safe to execute. A decision port is only worth having if something keeps proving
|
|
7
|
+
it still decides the same thing, so this script is the Python half of that
|
|
8
|
+
proof. It runs the **real** kernel functions
|
|
9
|
+
|
|
10
|
+
* ``latticeai.core.permission_mode`` — ``normalize_mode``, ``is_circuit_breaker``,
|
|
11
|
+
``effective_auto_approve``, ``should_stage_proposal``, ``plan_requires_approval``,
|
|
12
|
+
``mode_contract``;
|
|
13
|
+
* ``latticeai.core.agent_permission`` — ``block_reason_for_tool``,
|
|
14
|
+
``non_auto_plan_steps``;
|
|
15
|
+
* ``latticeai.core.tool_governor`` — ``classify_tool_call``;
|
|
16
|
+
* ``latticeai.tools.commands.run_command`` — the command sandbox validator, and
|
|
17
|
+
its real execution for a small read-only set;
|
|
18
|
+
|
|
19
|
+
over a decision grid built from the **real** per-tool policy table
|
|
20
|
+
(``latticeai.core.tool_registry.TOOL_GOVERNANCE``, read through
|
|
21
|
+
``ToolRegistry.policy_for`` so the blocked-prefix override is real too), and
|
|
22
|
+
writes every verdict to ``rust/fixtures/agent/golden/``.
|
|
23
|
+
|
|
24
|
+
Two consumers read what it writes:
|
|
25
|
+
|
|
26
|
+
* ``tests/unit/test_agent_kernel_parity_contract.py`` re-runs the Python kernel
|
|
27
|
+
over the same grid and asserts the committed goldens still hold — so loosening
|
|
28
|
+
a gate in Python fails loudly instead of silently invalidating the contract
|
|
29
|
+
the Rust side is pinned to;
|
|
30
|
+
* ``rust/lattice-agent/tests/parity.rs`` runs the Rust kernel against the same
|
|
31
|
+
goldens, exactly.
|
|
32
|
+
|
|
33
|
+
Determinism is the design constraint: no clock, no network, no machine-specific
|
|
34
|
+
path in any golden. The command fixtures run inside a throwaway workspace whose
|
|
35
|
+
layout is *described* in the manifest (``tree``) so the Rust suite can build the
|
|
36
|
+
identical tree, and the absolute root is written back out as ``<AGENT_ROOT>``.
|
|
37
|
+
The ``which(1)`` lookup is deliberately outside the goldens: whether ``rg`` is
|
|
38
|
+
installed is a property of the machine, not of the validator.
|
|
39
|
+
|
|
40
|
+
Usage::
|
|
41
|
+
|
|
42
|
+
.venv/bin/python scripts/generate_agent_parity_fixtures.py
|
|
43
|
+
"""
|
|
44
|
+
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import json
|
|
48
|
+
import os
|
|
49
|
+
import shlex
|
|
50
|
+
import shutil
|
|
51
|
+
import subprocess
|
|
52
|
+
import sys
|
|
53
|
+
import tempfile
|
|
54
|
+
from contextlib import contextmanager
|
|
55
|
+
from pathlib import Path
|
|
56
|
+
from typing import Any, Callable, Dict, Iterator, List, Optional, Tuple
|
|
57
|
+
|
|
58
|
+
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
59
|
+
if str(REPO_ROOT) not in sys.path:
|
|
60
|
+
sys.path.insert(0, str(REPO_ROOT))
|
|
61
|
+
|
|
62
|
+
import latticeai.tools as tools # noqa: E402
|
|
63
|
+
from latticeai.core.agent_permission import ( # noqa: E402
|
|
64
|
+
block_reason_for_tool,
|
|
65
|
+
non_auto_plan_steps,
|
|
66
|
+
)
|
|
67
|
+
from latticeai.core.permission_mode import ( # noqa: E402
|
|
68
|
+
COMPUTER_CONTROL_TOOLS,
|
|
69
|
+
COMPUTER_OBSERVATION_TOOLS,
|
|
70
|
+
HARD_BLOCK_SANDBOXES,
|
|
71
|
+
KNOWLEDGE_READ_TOOLS,
|
|
72
|
+
WORKSPACE_WRITE_TOOLS,
|
|
73
|
+
effective_auto_approve,
|
|
74
|
+
is_circuit_breaker,
|
|
75
|
+
mode_contract,
|
|
76
|
+
normalize_mode,
|
|
77
|
+
plan_requires_approval,
|
|
78
|
+
should_stage_proposal,
|
|
79
|
+
)
|
|
80
|
+
from latticeai.core.tool_governor import ( # noqa: E402
|
|
81
|
+
MUTATING_TOOL_INVENTORY,
|
|
82
|
+
PROPOSAL_CAPABLE_TOOLS,
|
|
83
|
+
classify_tool_call,
|
|
84
|
+
)
|
|
85
|
+
from latticeai.core.tool_registry import TOOL_GOVERNANCE # noqa: E402
|
|
86
|
+
from latticeai.tools import commands as command_tools # noqa: E402
|
|
87
|
+
|
|
88
|
+
FIXTURE_DIR = REPO_ROOT / "rust" / "fixtures" / "agent"
|
|
89
|
+
GOLDEN_DIR = FIXTURE_DIR / "golden"
|
|
90
|
+
|
|
91
|
+
SCHEMA = "agent-kernel-parity/v1"
|
|
92
|
+
MODES: List[str] = ["strict", "trusted", "bypass"]
|
|
93
|
+
|
|
94
|
+
#: LANG/LC_ALL leak into the child environment from the parent, so they are
|
|
95
|
+
#: pinned while the fixtures are built and recorded in the manifest.
|
|
96
|
+
PINNED_ENV: Dict[str, str] = {"LANG": "C.UTF-8", "LC_ALL": "C.UTF-8"}
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
# ── the decision grid ─────────────────────────────────────────────────────────
|
|
100
|
+
#: Every name any kernel table knows about: the dispatchable registry, the
|
|
101
|
+
#: governance table, the mutating inventory, and the four permission-mode tool
|
|
102
|
+
#: sets (which name knowledge-graph tools the registry does not dispatch).
|
|
103
|
+
def tool_universe() -> List[str]:
|
|
104
|
+
return sorted(
|
|
105
|
+
set(tools.registered_tools())
|
|
106
|
+
| set(TOOL_GOVERNANCE)
|
|
107
|
+
| set(MUTATING_TOOL_INVENTORY)
|
|
108
|
+
| set(KNOWLEDGE_READ_TOOLS)
|
|
109
|
+
| set(WORKSPACE_WRITE_TOOLS)
|
|
110
|
+
| set(COMPUTER_OBSERVATION_TOOLS)
|
|
111
|
+
| set(COMPUTER_CONTROL_TOOLS)
|
|
112
|
+
)
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
#: Argument shapes chosen for the branches they reach, not for realism:
|
|
116
|
+
#: the two path keys ``is_circuit_breaker`` reads, both command keys, the
|
|
117
|
+
#: blocked-prefix override that rewrites a write policy into a destructive one,
|
|
118
|
+
#: the ``rstrip("/")`` and backslash-normalisation branches of the root guard,
|
|
119
|
+
#: and a target that exists (mutation) versus one that does not (additive).
|
|
120
|
+
ARG_VARIANTS: Dict[str, Dict[str, Any]] = {
|
|
121
|
+
"none": {},
|
|
122
|
+
"benign_path": {"path": "notes/todo.md"},
|
|
123
|
+
"existing_path": {"path": "notes/existing.md"},
|
|
124
|
+
"filename_doc": {"filename": "report.docx"},
|
|
125
|
+
"blocked_prefix": {"path": "/etc/hosts"},
|
|
126
|
+
"root_path": {"path": "/"},
|
|
127
|
+
"home_tilde_slash": {"path": "~/"},
|
|
128
|
+
"windows_home": {"path": "\\home"},
|
|
129
|
+
"users_root": {"path": "/Users"},
|
|
130
|
+
"traversal": {"path": "../../etc/passwd"},
|
|
131
|
+
"rm_rf_root": {"command": "rm -rf /"},
|
|
132
|
+
"rm_rf_home_upper": {"cmd": "RM -RF $HOME"},
|
|
133
|
+
"long_path": {"path": "deep/" + ("a" * 180) + ".md"},
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
#: Paths ``classify_tool_call``'s injected ``path_exists`` answers True for.
|
|
137
|
+
EXISTING_PATHS = frozenset({"notes/existing.md", "report.docx", "/etc/hosts"})
|
|
138
|
+
|
|
139
|
+
#: ``effective_auto_approve``'s other axis. ``None`` is a member of the trusted
|
|
140
|
+
#: branch's accepted set, so it is a case and not an absence.
|
|
141
|
+
CHANGE_CLASSES: List[Optional[str]] = [
|
|
142
|
+
None, "read", "additive", "mutation", "destructive", "exec",
|
|
143
|
+
]
|
|
144
|
+
|
|
145
|
+
#: Mode inputs, including every alias plus the shapes that reach the
|
|
146
|
+
#: ``str(value or "")`` fallback.
|
|
147
|
+
NORMALIZE_INPUTS: List[Any] = [
|
|
148
|
+
"strict", "default", "manual", "trusted", "acceptedits", "accept_edits",
|
|
149
|
+
"workspace", "bypass", "bypasspermissions", "bypass_permissions", "yolo",
|
|
150
|
+
"dangerously-skip-permissions", "acceptEdits", " TRUSTED ", "BYPASS",
|
|
151
|
+
"Dangerously-Skip-Permissions", "", " ", "junk", "strictly", "read-only",
|
|
152
|
+
None, False, True, 0, 1, [], {},
|
|
153
|
+
]
|
|
154
|
+
|
|
155
|
+
#: Plans exercising the strict governor-tool skip, the missing-``action`` skip,
|
|
156
|
+
#: an unknown tool falling back to the default policy, and the ``plan_flag``.
|
|
157
|
+
PLAN_CASES: List[Dict[str, Any]] = [
|
|
158
|
+
{"key": "empty", "steps": [], "governed": [], "plan_flag": False},
|
|
159
|
+
{"key": "reads_only", "governed": [], "plan_flag": False,
|
|
160
|
+
"steps": [{"action": "read_file"}, {"action": "list_dir"}]},
|
|
161
|
+
{"key": "reads_plan_flag", "governed": [], "plan_flag": True,
|
|
162
|
+
"steps": [{"action": "read_file"}]},
|
|
163
|
+
{"key": "workspace_writes", "governed": [], "plan_flag": False,
|
|
164
|
+
"steps": [{"action": "write_file"}, {"action": "edit_file"}, {"action": "todo_write"}]},
|
|
165
|
+
{"key": "governed_writes", "governed": ["write_file", "edit_file"], "plan_flag": False,
|
|
166
|
+
"steps": [{"action": "write_file"}, {"action": "edit_file"}, {"action": "run_command"}]},
|
|
167
|
+
{"key": "exec_and_desktop", "governed": [], "plan_flag": False,
|
|
168
|
+
"steps": [{"action": "run_command"}, {"action": "computer_click"},
|
|
169
|
+
{"action": "computer_screenshot"}]},
|
|
170
|
+
{"key": "unknown_tool", "governed": [], "plan_flag": False,
|
|
171
|
+
"steps": [{"action": "not_a_tool"}, {"action": "knowledge_search"}]},
|
|
172
|
+
{"key": "missing_action", "governed": [], "plan_flag": False,
|
|
173
|
+
"steps": [{"description": "no action key"}, {"action": ""}, {"action": "local_read"}]},
|
|
174
|
+
]
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
# ── the command-sandbox workspace ────────────────────────────────────────────
|
|
178
|
+
#: The throwaway ``AGENT_ROOT`` the command fixtures run inside, described so the
|
|
179
|
+
#: Rust suite builds the byte-identical tree. ``outside`` entries are written
|
|
180
|
+
#: beside the root — the only way a symlink escape can be a real escape.
|
|
181
|
+
TREE: List[Dict[str, Any]] = [
|
|
182
|
+
{"kind": "outside", "path": "outside_secret.txt", "content": "top secret\n"},
|
|
183
|
+
{"kind": "dir", "path": "notes"},
|
|
184
|
+
{"kind": "file", "path": "notes/a.txt", "content": "alpha\nbeta\ngamma\n"},
|
|
185
|
+
{"kind": "dir", "path": "a b"},
|
|
186
|
+
{"kind": "file", "path": "a b/c.txt", "content": "spaced\n"},
|
|
187
|
+
{"kind": "file", "path": "quoted name.txt", "content": "quoted\n"},
|
|
188
|
+
{"kind": "dir", "path": "노트"},
|
|
189
|
+
{"kind": "file", "path": "노트/메모.txt", "content": "한글\n"},
|
|
190
|
+
{"kind": "dir", "path": "sub"},
|
|
191
|
+
{"kind": "file", "path": "sub/inner.txt", "content": "inner\n"},
|
|
192
|
+
{"kind": "file", "path": "a\\b", "content": "backslash\n"},
|
|
193
|
+
{"kind": "symlink", "path": "inside_link", "target": "notes/a.txt"},
|
|
194
|
+
{"kind": "symlink", "path": "escape_link", "target": "../outside_secret.txt"},
|
|
195
|
+
#: 3,000 lines of ``%07d\n`` = 24,000 characters, so ``cat`` of it is exactly
|
|
196
|
+
#: twice ``MAX_COMMAND_OUTPUT`` and the tail-slice is observable.
|
|
197
|
+
{"kind": "lines", "path": "big.txt", "count": 3000},
|
|
198
|
+
]
|
|
199
|
+
|
|
200
|
+
#: ``(key, command, cwd)``. Validation only — nothing here is executed.
|
|
201
|
+
COMMAND_CASES: List[Tuple[str, str, Optional[str]]] = [
|
|
202
|
+
("empty", "", None),
|
|
203
|
+
("whitespace_only", " ", None),
|
|
204
|
+
("plain_ls", "ls", None),
|
|
205
|
+
("ls_flags", "ls -la", None),
|
|
206
|
+
("pwd", "pwd", None),
|
|
207
|
+
("absolute_executable", "/bin/ls", None),
|
|
208
|
+
("relative_executable", "./ls", None),
|
|
209
|
+
("trailing_slash_executable", "ls/", None),
|
|
210
|
+
("empty_executable", "'' notes", None),
|
|
211
|
+
("blocked_rm", "rm -rf /", None),
|
|
212
|
+
("blocked_sudo", "sudo ls", None),
|
|
213
|
+
("not_allowlisted", "echo hi", None),
|
|
214
|
+
("git_status", "git status", None),
|
|
215
|
+
("git_log", "git log --oneline", None),
|
|
216
|
+
("pipe", "cat notes/a.txt | wc -l", None),
|
|
217
|
+
("and_and", "cat notes/a.txt && ls", None),
|
|
218
|
+
("semicolon", "cat notes/a.txt; ls", None),
|
|
219
|
+
("redirect_out", "cat notes/a.txt > out.txt", None),
|
|
220
|
+
("redirect_in", "wc -l < notes/a.txt", None),
|
|
221
|
+
("dollar_paren", "cat $(ls)", None),
|
|
222
|
+
("backtick", "cat `ls`", None),
|
|
223
|
+
("or_or", "ls || ls", None),
|
|
224
|
+
("find_delete", "find . -delete", None),
|
|
225
|
+
("find_exec", "find . -name x -exec cat {} +", None),
|
|
226
|
+
("find_okdir", "find . -okdir cat", None),
|
|
227
|
+
("find_ok_prefix_only", "find . -name -execute", None),
|
|
228
|
+
("find_plain", "find . -name inner.txt", None),
|
|
229
|
+
("rg_pre", "rg --pre cat foo", None),
|
|
230
|
+
("rg_pre_glob_eq", "rg --pre-glob=*.py foo", None),
|
|
231
|
+
("rg_pretty_is_fine", "rg --pretty foo", None),
|
|
232
|
+
("traversal", "cat ../outside_secret.txt", None),
|
|
233
|
+
("traversal_middle", "cat sub/../../outside_secret.txt", None),
|
|
234
|
+
("dotdot_alone", "cat ..", None),
|
|
235
|
+
("absolute_arg", "cat /etc/passwd", None),
|
|
236
|
+
("tilde_arg", "cat ~", None),
|
|
237
|
+
("tilde_path_arg", "cat ~/secret", None),
|
|
238
|
+
("dev_null_is_exempt", "cat /dev/null", None),
|
|
239
|
+
("bare_dash_flag", "cat -", None),
|
|
240
|
+
("empty_arg", "cat ''", None),
|
|
241
|
+
("dot_arg", "cat .", None),
|
|
242
|
+
("kv_plain_value", "ls --color=auto", None),
|
|
243
|
+
("kv_absolute_value", "ls --color=/etc", None),
|
|
244
|
+
("kv_traversal_value", "ls --color=../x", None),
|
|
245
|
+
("kv_inside_value", "ls --color=notes/a.txt", None),
|
|
246
|
+
("symlink_escape", "cat escape_link", None),
|
|
247
|
+
("symlink_inside", "cat inside_link", None),
|
|
248
|
+
("quoted_space", "cat 'quoted name.txt'", None),
|
|
249
|
+
("double_quoted_dir", 'cat "a b/c.txt"', None),
|
|
250
|
+
("escaped_space", "cat a\\ b/c.txt", None),
|
|
251
|
+
("backslash_value", "cat a\\\\b", None),
|
|
252
|
+
("korean_path", "cat 노트/메모.txt", None),
|
|
253
|
+
("unterminated_quote", "cat 'unterminated", None),
|
|
254
|
+
("dangling_escape", "cat trailing\\", None),
|
|
255
|
+
("missing_file_is_allowed", "cat missing.txt", None),
|
|
256
|
+
("cwd_subdir", "ls", "sub"),
|
|
257
|
+
("cwd_escape", "ls", "../"),
|
|
258
|
+
("cwd_absolute", "ls", "/etc"),
|
|
259
|
+
("cwd_missing", "ls", "nope"),
|
|
260
|
+
("cwd_is_a_file", "ls", "notes/a.txt"),
|
|
261
|
+
]
|
|
262
|
+
|
|
263
|
+
#: Commands that really run, in the throwaway workspace. Read-only, allow-listed,
|
|
264
|
+
#: and chosen so the bytes are identical on macOS and Linux — which rules out
|
|
265
|
+
#: ``wc`` (BSD pads its counts) and any listing whose order is collation
|
|
266
|
+
#: dependent. ``stderr`` is pinned only where it is empty on both.
|
|
267
|
+
EXECUTION_CASES: List[Tuple[str, str, Optional[str], bool]] = [
|
|
268
|
+
("pwd", "pwd", None, True),
|
|
269
|
+
("pwd_in_subdir", "pwd", "sub", True),
|
|
270
|
+
("cat_file", "cat notes/a.txt", None, True),
|
|
271
|
+
("cat_quoted", "cat 'quoted name.txt'", None, True),
|
|
272
|
+
("cat_double_quoted", 'cat "a b/c.txt"', None, True),
|
|
273
|
+
("cat_korean", "cat 노트/메모.txt", None, True),
|
|
274
|
+
("cat_backslash", "cat a\\\\b", None, True),
|
|
275
|
+
("cat_symlink_inside", "cat inside_link", None, True),
|
|
276
|
+
("head_two", "head -n 2 notes/a.txt", None, True),
|
|
277
|
+
("tail_one", "tail -n 1 notes/a.txt", None, True),
|
|
278
|
+
("ls_subdir", "ls notes", None, True),
|
|
279
|
+
("ls_cwd", "ls", "sub", True),
|
|
280
|
+
("find_one", "find . -name inner.txt", None, True),
|
|
281
|
+
("cat_truncates", "cat big.txt", None, True),
|
|
282
|
+
("missing_file_exit_code", "cat missing.txt", None, False),
|
|
283
|
+
]
|
|
284
|
+
|
|
285
|
+
#: ``_resolve_path`` — the file sandbox every file tool goes through.
|
|
286
|
+
PATH_CASES: List[Tuple[str, str]] = [
|
|
287
|
+
("empty", ""),
|
|
288
|
+
("dot", "."),
|
|
289
|
+
("benign", "notes/a.txt"),
|
|
290
|
+
("normalising", "notes/../notes/a.txt"),
|
|
291
|
+
("korean", "노트/메모.txt"),
|
|
292
|
+
("traversal", "../outside_secret.txt"),
|
|
293
|
+
("absolute_outside", "/etc/passwd"),
|
|
294
|
+
("absolute_inside", "<AGENT_ROOT>/notes/a.txt"),
|
|
295
|
+
("absolute_root", "<AGENT_ROOT>"),
|
|
296
|
+
("symlink_inside", "inside_link"),
|
|
297
|
+
("symlink_escape", "escape_link"),
|
|
298
|
+
("missing_but_inside", "notes/missing/deeper.txt"),
|
|
299
|
+
]
|
|
300
|
+
|
|
301
|
+
#: ``shlex.split`` edges. The validator splits before it decides anything, so a
|
|
302
|
+
#: divergence here is a divergence in every rule downstream of it.
|
|
303
|
+
SHLEX_CASES: List[str] = [
|
|
304
|
+
"", " ", "ls", "ls -la", "ls\t-la", "ls\n-la", "a b c",
|
|
305
|
+
"cat 'a b.txt'", 'cat "a b"', "cat a\\ b", 'cat "a\\nb"', "cat 'a\\nb'",
|
|
306
|
+
"cat ''", 'cat ""', "''", '""', 'a"b"c', "x'y'z", "cat \"a\\\"b\"",
|
|
307
|
+
"cat 'a\\'", "cat back\\\\", "cat \\'quoted\\'", "cat a\\\\b",
|
|
308
|
+
"cat 노트/메모.txt", "cat '노트/메모.txt'", " leading and trailing ",
|
|
309
|
+
"cat \"a\"b'c'", "cat '' ''", "cat $(ls)", "cat `ls`", "cat a=b --c=d",
|
|
310
|
+
"cat 'unterminated", 'cat "unterminated', "cat trailing\\", 'cat "trailing\\',
|
|
311
|
+
]
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
# ── helpers ───────────────────────────────────────────────────────────────────
|
|
315
|
+
@contextmanager
|
|
316
|
+
def pinned_environment() -> Iterator[None]:
|
|
317
|
+
"""Pin the environment variables that leak into the sandboxed child."""
|
|
318
|
+
previous = {key: os.environ.get(key) for key in PINNED_ENV}
|
|
319
|
+
os.environ.update(PINNED_ENV)
|
|
320
|
+
try:
|
|
321
|
+
yield
|
|
322
|
+
finally:
|
|
323
|
+
for key, value in previous.items():
|
|
324
|
+
if value is None:
|
|
325
|
+
os.environ.pop(key, None)
|
|
326
|
+
else:
|
|
327
|
+
os.environ[key] = value
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
@contextmanager
|
|
331
|
+
def agent_root(root: Path) -> Iterator[Path]:
|
|
332
|
+
"""Point the real ``AGENT_ROOT`` at ``root`` for the duration.
|
|
333
|
+
|
|
334
|
+
``latticeai.tools`` holds the module global every path helper reads, and
|
|
335
|
+
``commands.py`` reaches it through ``tools.AGENT_ROOT``, so rebinding that
|
|
336
|
+
one name redirects the whole sandbox.
|
|
337
|
+
"""
|
|
338
|
+
original = tools.AGENT_ROOT
|
|
339
|
+
tools.AGENT_ROOT = root.resolve()
|
|
340
|
+
try:
|
|
341
|
+
yield tools.AGENT_ROOT
|
|
342
|
+
finally:
|
|
343
|
+
tools.AGENT_ROOT = original
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def build_tree(root: Path) -> None:
|
|
347
|
+
"""Materialise :data:`TREE`. Same spec the Rust suite builds from."""
|
|
348
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
349
|
+
for node in TREE:
|
|
350
|
+
kind = node["kind"]
|
|
351
|
+
if kind == "outside":
|
|
352
|
+
(root.parent / node["path"]).write_text(node["content"], encoding="utf-8")
|
|
353
|
+
elif kind == "dir":
|
|
354
|
+
(root / node["path"]).mkdir(parents=True, exist_ok=True)
|
|
355
|
+
elif kind == "file":
|
|
356
|
+
(root / node["path"]).write_text(node["content"], encoding="utf-8")
|
|
357
|
+
elif kind == "lines":
|
|
358
|
+
body = "".join(f"{index:07d}\n" for index in range(node["count"]))
|
|
359
|
+
(root / node["path"]).write_text(body, encoding="utf-8")
|
|
360
|
+
elif kind == "symlink":
|
|
361
|
+
link = root / node["path"]
|
|
362
|
+
if link.is_symlink():
|
|
363
|
+
link.unlink()
|
|
364
|
+
link.symlink_to(node["target"])
|
|
365
|
+
else: # pragma: no cover - the spec is closed
|
|
366
|
+
raise ValueError(f"unknown tree node kind: {kind}")
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
class _Spawned(Exception):
|
|
370
|
+
"""Raised in place of ``subprocess.run`` so validation stops at the spawn."""
|
|
371
|
+
|
|
372
|
+
def __init__(self, argv: List[str], cwd: Any, env: Dict[str, str]) -> None:
|
|
373
|
+
super().__init__("spawn")
|
|
374
|
+
self.argv = list(argv)
|
|
375
|
+
self.cwd = cwd
|
|
376
|
+
self.env = dict(env)
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
@contextmanager
|
|
380
|
+
def validation_only() -> Iterator[List[str]]:
|
|
381
|
+
"""Stop ``run_command`` at the spawn, and make ``which`` machine-independent.
|
|
382
|
+
|
|
383
|
+
Whether ``rg`` is installed is a property of the machine; whether ``rg --pre``
|
|
384
|
+
is refused is a property of the validator. Recording the second without the
|
|
385
|
+
first is the whole point of this seam. Every ``which`` call is captured so
|
|
386
|
+
the caller can assert the fixed PATH was the one searched.
|
|
387
|
+
"""
|
|
388
|
+
searched: List[str] = []
|
|
389
|
+
real_subprocess, real_shutil = command_tools.subprocess, command_tools.shutil
|
|
390
|
+
|
|
391
|
+
class _SubprocessShim:
|
|
392
|
+
TimeoutExpired = subprocess.TimeoutExpired
|
|
393
|
+
|
|
394
|
+
@staticmethod
|
|
395
|
+
def run(argv: List[str], **kwargs: Any) -> None:
|
|
396
|
+
raise _Spawned(argv, kwargs.get("cwd"), kwargs.get("env") or {})
|
|
397
|
+
|
|
398
|
+
class _ShutilShim:
|
|
399
|
+
@staticmethod
|
|
400
|
+
def which(cmd: str, path: Optional[str] = None) -> str:
|
|
401
|
+
searched.append(str(path))
|
|
402
|
+
return f"<which>/{cmd}"
|
|
403
|
+
|
|
404
|
+
command_tools.subprocess = _SubprocessShim # type: ignore[assignment]
|
|
405
|
+
command_tools.shutil = _ShutilShim # type: ignore[assignment]
|
|
406
|
+
try:
|
|
407
|
+
yield searched
|
|
408
|
+
finally:
|
|
409
|
+
command_tools.subprocess = real_subprocess
|
|
410
|
+
command_tools.shutil = real_shutil
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def _error(exc: BaseException) -> Dict[str, str]:
|
|
414
|
+
kind = "tool" if isinstance(exc, tools.ToolError) else "shlex"
|
|
415
|
+
return {"kind": kind, "message": str(exc)}
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _relative_to(root: Path, path: Path) -> str:
|
|
419
|
+
return "." if path == root else str(path.relative_to(root))
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
# ── grid builders ─────────────────────────────────────────────────────────────
|
|
423
|
+
def policy_for(tool_name: str, args: Dict[str, Any]) -> Dict[str, Any]:
|
|
424
|
+
"""The **real** per-tool policy, override included.
|
|
425
|
+
|
|
426
|
+
``ToolRegistry.policy_for`` is the single source of truth: the table for the
|
|
427
|
+
ordinary case, and a synthesised destructive policy when a write targets a
|
|
428
|
+
blocked system prefix. Both shapes belong in the goldens.
|
|
429
|
+
"""
|
|
430
|
+
return dict(tools.DEFAULT_TOOL_REGISTRY.policy_for(tool_name, args))
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def policy_table() -> Dict[str, Any]:
|
|
434
|
+
"""The registry table, the default, and every args-dependent override."""
|
|
435
|
+
default = dict(tools.DEFAULT_TOOL_REGISTRY.default_policy)
|
|
436
|
+
table = {name: dict(policy) for name, policy in TOOL_GOVERNANCE.items()}
|
|
437
|
+
overrides: Dict[str, Any] = {}
|
|
438
|
+
for name in tool_universe():
|
|
439
|
+
base = table.get(name, default)
|
|
440
|
+
for variant, args in ARG_VARIANTS.items():
|
|
441
|
+
policy = policy_for(name, args)
|
|
442
|
+
if policy != base:
|
|
443
|
+
overrides[f"{name}|{variant}"] = policy
|
|
444
|
+
return {"schema": SCHEMA, "default": default, "tools": table, "overrides": overrides}
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def policy_key(name: str, variant: str, default: Dict[str, Any]) -> str:
|
|
448
|
+
"""How a case names its policy: table entry, override, or the default."""
|
|
449
|
+
base = TOOL_GOVERNANCE.get(name)
|
|
450
|
+
policy = policy_for(name, ARG_VARIANTS[variant])
|
|
451
|
+
if base is not None and policy == dict(base):
|
|
452
|
+
return name
|
|
453
|
+
if base is None and policy == default:
|
|
454
|
+
return "@default"
|
|
455
|
+
return f"{name}|{variant}"
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def call_rows() -> List[Dict[str, Any]]:
|
|
459
|
+
"""The mode-*invariant* half of the grid: policy, breaker, classification.
|
|
460
|
+
|
|
461
|
+
Circuit breakers and change classification do not read the mode — that is a
|
|
462
|
+
documented property of the kernel, and keeping them in their own file is how
|
|
463
|
+
the fixture states it rather than repeating it three times.
|
|
464
|
+
"""
|
|
465
|
+
default = dict(tools.DEFAULT_TOOL_REGISTRY.default_policy)
|
|
466
|
+
rows: List[Dict[str, Any]] = []
|
|
467
|
+
for name in tool_universe():
|
|
468
|
+
for variant, args in ARG_VARIANTS.items():
|
|
469
|
+
policy = policy_for(name, args)
|
|
470
|
+
rows.append({
|
|
471
|
+
"tool": name,
|
|
472
|
+
"variant": variant,
|
|
473
|
+
"policy": policy_key(name, variant, default),
|
|
474
|
+
"circuit_breaker": is_circuit_breaker(name, policy, args),
|
|
475
|
+
"classification": classify_tool_call(
|
|
476
|
+
name, args, policy=policy, path_exists=lambda p: p in EXISTING_PATHS,
|
|
477
|
+
),
|
|
478
|
+
})
|
|
479
|
+
return rows
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def decision_rows(mode: str) -> List[Dict[str, Any]]:
|
|
483
|
+
"""One mode's half: auto-approve, block reason, proposal staging."""
|
|
484
|
+
rows: List[Dict[str, Any]] = []
|
|
485
|
+
for name in tool_universe():
|
|
486
|
+
for variant, args in ARG_VARIANTS.items():
|
|
487
|
+
policy = policy_for(name, args)
|
|
488
|
+
classification = classify_tool_call(
|
|
489
|
+
name, args, policy=policy, path_exists=lambda p: p in EXISTING_PATHS,
|
|
490
|
+
)
|
|
491
|
+
rows.append({
|
|
492
|
+
"tool": name,
|
|
493
|
+
"variant": variant,
|
|
494
|
+
"auto_approve": effective_auto_approve(mode, name, policy, args=args),
|
|
495
|
+
"block_reason": block_reason_for_tool(mode, name, policy, args),
|
|
496
|
+
"stage_proposal": should_stage_proposal(
|
|
497
|
+
mode, proposal_required=classification["proposal_required"],
|
|
498
|
+
),
|
|
499
|
+
})
|
|
500
|
+
return rows
|
|
501
|
+
|
|
502
|
+
|
|
503
|
+
def change_class_rows(mode: str) -> List[Dict[str, Any]]:
|
|
504
|
+
"""``effective_auto_approve``'s second axis, over the workspace writers."""
|
|
505
|
+
rows: List[Dict[str, Any]] = []
|
|
506
|
+
for name in sorted(WORKSPACE_WRITE_TOOLS | {"local_write", "read_file", "run_command"}):
|
|
507
|
+
policy = policy_for(name, {})
|
|
508
|
+
for change_class in CHANGE_CLASSES:
|
|
509
|
+
rows.append({
|
|
510
|
+
"tool": name,
|
|
511
|
+
"change_class": change_class,
|
|
512
|
+
"auto_approve": effective_auto_approve(
|
|
513
|
+
mode, name, policy, change_class=change_class,
|
|
514
|
+
),
|
|
515
|
+
})
|
|
516
|
+
return rows
|
|
517
|
+
|
|
518
|
+
|
|
519
|
+
def approval_rows(mode: str) -> List[Dict[str, Any]]:
|
|
520
|
+
"""Plan-level gates: which steps stay non-auto, and whether the plan pauses."""
|
|
521
|
+
governance = {name: dict(policy) for name, policy in TOOL_GOVERNANCE.items()}
|
|
522
|
+
rows: List[Dict[str, Any]] = []
|
|
523
|
+
for case in PLAN_CASES:
|
|
524
|
+
non_auto = non_auto_plan_steps(
|
|
525
|
+
mode, case["steps"], governance, governed_tools=case["governed"],
|
|
526
|
+
)
|
|
527
|
+
rows.append({
|
|
528
|
+
"key": case["key"],
|
|
529
|
+
"non_auto_steps": non_auto,
|
|
530
|
+
"requires_approval": plan_requires_approval(
|
|
531
|
+
mode, non_auto_steps=non_auto, plan_flag=case["plan_flag"],
|
|
532
|
+
),
|
|
533
|
+
})
|
|
534
|
+
return rows
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
def normalize_rows() -> List[Dict[str, Any]]:
|
|
538
|
+
return [
|
|
539
|
+
{"input": value, "mode": normalize_mode(value).value}
|
|
540
|
+
for value in NORMALIZE_INPUTS
|
|
541
|
+
]
|
|
542
|
+
|
|
543
|
+
|
|
544
|
+
def contract_payload() -> Dict[str, Any]:
|
|
545
|
+
return {
|
|
546
|
+
"schema": SCHEMA,
|
|
547
|
+
"default_mode": normalize_mode(None).value,
|
|
548
|
+
"contracts": {mode: mode_contract(mode) for mode in MODES},
|
|
549
|
+
}
|
|
550
|
+
|
|
551
|
+
|
|
552
|
+
def shlex_rows() -> List[Dict[str, Any]]:
|
|
553
|
+
rows: List[Dict[str, Any]] = []
|
|
554
|
+
for command in SHLEX_CASES:
|
|
555
|
+
try:
|
|
556
|
+
rows.append({"input": command, "tokens": shlex.split(command)})
|
|
557
|
+
except ValueError as exc:
|
|
558
|
+
rows.append({"input": command, "error": str(exc)})
|
|
559
|
+
return rows
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def command_rows(root: Path) -> Tuple[List[Dict[str, Any]], Dict[str, Any], List[str]]:
|
|
563
|
+
"""Validation verdicts, plus the environment the validator would spawn into."""
|
|
564
|
+
rows: List[Dict[str, Any]] = []
|
|
565
|
+
spawn_env: Dict[str, Any] = {}
|
|
566
|
+
with agent_root(root) as resolved, validation_only() as searched:
|
|
567
|
+
for key, command, cwd in COMMAND_CASES:
|
|
568
|
+
try:
|
|
569
|
+
command_tools.run_command(command, cwd)
|
|
570
|
+
except _Spawned as spawned:
|
|
571
|
+
env = {k: v.replace(str(resolved), "<AGENT_ROOT>") for k, v in spawned.env.items()}
|
|
572
|
+
if spawn_env and env != spawn_env: # pragma: no cover - defensive
|
|
573
|
+
raise AssertionError("the sandbox environment is not constant")
|
|
574
|
+
spawn_env = env
|
|
575
|
+
rows.append({
|
|
576
|
+
"key": key, "command": command, "cwd": cwd, "outcome": "spawn",
|
|
577
|
+
"executable": Path(spawned.argv[0]).name,
|
|
578
|
+
"args": spawned.argv[1:],
|
|
579
|
+
"workdir": _relative_to(resolved, Path(str(spawned.cwd))),
|
|
580
|
+
})
|
|
581
|
+
except (tools.ToolError, ValueError) as exc:
|
|
582
|
+
rows.append({
|
|
583
|
+
"key": key, "command": command, "cwd": cwd,
|
|
584
|
+
"outcome": "error", "error": _error(exc),
|
|
585
|
+
})
|
|
586
|
+
else: # pragma: no cover - the shim always raises
|
|
587
|
+
raise AssertionError(f"{command!r} reached the real subprocess")
|
|
588
|
+
return rows, spawn_env, searched
|
|
589
|
+
|
|
590
|
+
|
|
591
|
+
def execution_rows(root: Path) -> List[Dict[str, Any]]:
|
|
592
|
+
"""The real ``run_command``, really executed, with the answers pinned."""
|
|
593
|
+
rows: List[Dict[str, Any]] = []
|
|
594
|
+
with agent_root(root) as resolved:
|
|
595
|
+
for key, command, cwd, pin_stderr in EXECUTION_CASES:
|
|
596
|
+
result = command_tools.run_command(command, cwd)
|
|
597
|
+
row = {
|
|
598
|
+
"key": key,
|
|
599
|
+
"command": command,
|
|
600
|
+
"cwd": cwd,
|
|
601
|
+
"result_cwd": result["cwd"],
|
|
602
|
+
"returncode": result["returncode"],
|
|
603
|
+
"stdout": result["stdout"].replace(str(resolved), "<AGENT_ROOT>"),
|
|
604
|
+
}
|
|
605
|
+
if pin_stderr:
|
|
606
|
+
row["stderr"] = result["stderr"]
|
|
607
|
+
rows.append(row)
|
|
608
|
+
return rows
|
|
609
|
+
|
|
610
|
+
|
|
611
|
+
def path_rows(root: Path) -> List[Dict[str, Any]]:
|
|
612
|
+
rows: List[Dict[str, Any]] = []
|
|
613
|
+
with agent_root(root) as resolved:
|
|
614
|
+
for key, raw in PATH_CASES:
|
|
615
|
+
candidate = raw.replace("<AGENT_ROOT>", str(resolved))
|
|
616
|
+
try:
|
|
617
|
+
resolved_path = tools.resolve_workspace_path(candidate)
|
|
618
|
+
except tools.ToolError as exc:
|
|
619
|
+
rows.append({"key": key, "input": raw, "outcome": "error",
|
|
620
|
+
"error": _error(exc)})
|
|
621
|
+
else:
|
|
622
|
+
rows.append({"key": key, "input": raw, "outcome": "ok",
|
|
623
|
+
"relative": _relative_to(resolved, resolved_path)})
|
|
624
|
+
return rows
|
|
625
|
+
|
|
626
|
+
|
|
627
|
+
def constants() -> Dict[str, Any]:
|
|
628
|
+
"""Every table the Rust kernel duplicates, so drift is a failing assertion."""
|
|
629
|
+
return {
|
|
630
|
+
"max_file_bytes": tools.MAX_FILE_BYTES,
|
|
631
|
+
"max_command_seconds": tools.MAX_COMMAND_SECONDS,
|
|
632
|
+
"max_command_output": tools.MAX_COMMAND_OUTPUT,
|
|
633
|
+
"safe_executable_path": command_tools._SAFE_EXECUTABLE_PATH,
|
|
634
|
+
"allowed_commands": sorted(tools.ALLOWED_COMMANDS),
|
|
635
|
+
"blocked_commands": sorted(tools.BLOCKED_COMMANDS),
|
|
636
|
+
"allowed_git_subcommands": sorted(tools.ALLOWED_GIT_SUBCOMMANDS),
|
|
637
|
+
"blocked_find_flags": sorted(command_tools._BLOCKED_FIND_FLAGS),
|
|
638
|
+
"blocked_rg_flags": sorted(command_tools._BLOCKED_RG_FLAGS),
|
|
639
|
+
"shell_operators": ["|", "&&", "||", ";", ">", "<", "$(", "`"],
|
|
640
|
+
"hard_block_sandboxes": sorted(HARD_BLOCK_SANDBOXES),
|
|
641
|
+
"knowledge_read_tools": sorted(KNOWLEDGE_READ_TOOLS),
|
|
642
|
+
"workspace_write_tools": sorted(WORKSPACE_WRITE_TOOLS),
|
|
643
|
+
"computer_observation_tools": sorted(COMPUTER_OBSERVATION_TOOLS),
|
|
644
|
+
"computer_control_tools": sorted(COMPUTER_CONTROL_TOOLS),
|
|
645
|
+
"mutating_tool_inventory": dict(sorted(MUTATING_TOOL_INVENTORY.items())),
|
|
646
|
+
"proposal_capable_tools": sorted(PROPOSAL_CAPABLE_TOOLS),
|
|
647
|
+
}
|
|
648
|
+
|
|
649
|
+
|
|
650
|
+
def manifest(searched_paths: List[str]) -> Dict[str, Any]:
|
|
651
|
+
return {
|
|
652
|
+
"schema": SCHEMA,
|
|
653
|
+
"modes": MODES,
|
|
654
|
+
"tools": tool_universe(),
|
|
655
|
+
"arg_variants": ARG_VARIANTS,
|
|
656
|
+
"existing_paths": sorted(EXISTING_PATHS),
|
|
657
|
+
"change_classes": CHANGE_CLASSES,
|
|
658
|
+
"plans": PLAN_CASES,
|
|
659
|
+
"tree": TREE,
|
|
660
|
+
"pinned_env": PINNED_ENV,
|
|
661
|
+
"constants": constants(),
|
|
662
|
+
# Evidence that the allowlisted binary is looked up on the fixed PATH and
|
|
663
|
+
# nowhere else: every which() the validator made, deduplicated.
|
|
664
|
+
"which_paths": sorted(set(searched_paths)),
|
|
665
|
+
}
|
|
666
|
+
|
|
667
|
+
|
|
668
|
+
# ── writing ───────────────────────────────────────────────────────────────────
|
|
669
|
+
def _dump(path: Path, payload: Any) -> None:
|
|
670
|
+
path.write_text(
|
|
671
|
+
json.dumps(payload, ensure_ascii=False, sort_keys=True, indent=2) + "\n",
|
|
672
|
+
encoding="utf-8",
|
|
673
|
+
)
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
def _dump_grid(path: Path, header: Dict[str, Any], groups: Dict[str, List[Any]]) -> None:
|
|
677
|
+
"""Header pretty, cases one per line — a thousand-row diff stays readable."""
|
|
678
|
+
parts = [
|
|
679
|
+
f" {json.dumps(key, ensure_ascii=False)}: "
|
|
680
|
+
+ json.dumps(header[key], ensure_ascii=False, sort_keys=True)
|
|
681
|
+
for key in sorted(header)
|
|
682
|
+
]
|
|
683
|
+
for name in sorted(groups):
|
|
684
|
+
rows = ",\n ".join(
|
|
685
|
+
json.dumps(row, ensure_ascii=False, sort_keys=True) for row in groups[name]
|
|
686
|
+
)
|
|
687
|
+
body = f"[\n {rows}\n ]" if rows else "[]"
|
|
688
|
+
parts.append(f" {json.dumps(name)}: {body}")
|
|
689
|
+
path.write_text("{\n" + ",\n".join(parts) + "\n}\n", encoding="utf-8")
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def build(root: Path) -> Dict[str, Callable[[], None]]:
|
|
693
|
+
"""Every golden, keyed by filename, as a thunk that writes it."""
|
|
694
|
+
commands, spawn_env, searched = command_rows(root)
|
|
695
|
+
return {
|
|
696
|
+
"manifest.json": lambda: _dump(GOLDEN_DIR / "manifest.json", manifest(searched)),
|
|
697
|
+
"policies.json": lambda: _dump(GOLDEN_DIR / "policies.json", policy_table()),
|
|
698
|
+
"contract.json": lambda: _dump(GOLDEN_DIR / "contract.json", contract_payload()),
|
|
699
|
+
"calls.json": lambda: _dump_grid(
|
|
700
|
+
GOLDEN_DIR / "calls.json", {"schema": SCHEMA}, {"cases": call_rows()},
|
|
701
|
+
),
|
|
702
|
+
"normalize.json": lambda: _dump_grid(
|
|
703
|
+
GOLDEN_DIR / "normalize.json", {"schema": SCHEMA}, {"cases": normalize_rows()},
|
|
704
|
+
),
|
|
705
|
+
"shlex.json": lambda: _dump_grid(
|
|
706
|
+
GOLDEN_DIR / "shlex.json", {"schema": SCHEMA}, {"cases": shlex_rows()},
|
|
707
|
+
),
|
|
708
|
+
"commands.json": lambda: _dump_grid(
|
|
709
|
+
GOLDEN_DIR / "commands.json",
|
|
710
|
+
{"schema": SCHEMA, "spawn_env": spawn_env},
|
|
711
|
+
{"cases": commands},
|
|
712
|
+
),
|
|
713
|
+
"execution.json": lambda: _dump_grid(
|
|
714
|
+
GOLDEN_DIR / "execution.json", {"schema": SCHEMA},
|
|
715
|
+
{"cases": execution_rows(root)},
|
|
716
|
+
),
|
|
717
|
+
"paths.json": lambda: _dump_grid(
|
|
718
|
+
GOLDEN_DIR / "paths.json", {"schema": SCHEMA}, {"cases": path_rows(root)},
|
|
719
|
+
),
|
|
720
|
+
**{
|
|
721
|
+
f"decisions__{mode}.json": (lambda mode=mode: _dump_grid(
|
|
722
|
+
GOLDEN_DIR / f"decisions__{mode}.json",
|
|
723
|
+
{"schema": SCHEMA, "mode": mode},
|
|
724
|
+
{
|
|
725
|
+
"cases": decision_rows(mode),
|
|
726
|
+
"change_class_cases": change_class_rows(mode),
|
|
727
|
+
"plan_cases": approval_rows(mode),
|
|
728
|
+
},
|
|
729
|
+
))
|
|
730
|
+
for mode in MODES
|
|
731
|
+
},
|
|
732
|
+
}
|
|
733
|
+
|
|
734
|
+
|
|
735
|
+
def main() -> int:
|
|
736
|
+
if GOLDEN_DIR.exists():
|
|
737
|
+
shutil.rmtree(GOLDEN_DIR)
|
|
738
|
+
GOLDEN_DIR.mkdir(parents=True, exist_ok=True)
|
|
739
|
+
with pinned_environment(), tempfile.TemporaryDirectory() as tmp:
|
|
740
|
+
root = Path(tmp) / "agent_workspace"
|
|
741
|
+
build_tree(root)
|
|
742
|
+
writers = build(root)
|
|
743
|
+
for name in sorted(writers):
|
|
744
|
+
writers[name]()
|
|
745
|
+
total = sum(path.stat().st_size for path in GOLDEN_DIR.glob("*.json"))
|
|
746
|
+
print(f"golden: {len(list(GOLDEN_DIR.glob('*.json')))} files, {total / 1024:.1f} KiB")
|
|
747
|
+
print(f"grid: {len(tool_universe())} tools × {len(ARG_VARIANTS)} variants × {len(MODES)} modes")
|
|
748
|
+
return 0
|
|
749
|
+
|
|
750
|
+
|
|
751
|
+
if __name__ == "__main__":
|
|
752
|
+
raise SystemExit(main())
|