ltcai 11.5.2 → 11.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -148
- package/bin/ltcai.js +234 -24
- package/docs/CHANGELOG.md +119 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +8 -6
- package/docs/ENTERPRISE.md +2 -1
- package/docs/MULTI_AGENT_RUNTIME.md +12 -5
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -3
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/WORKFLOW_DESIGNER.md +3 -2
- package/docs/kg-schema.md +1 -1
- package/docs/v11.6.0_ONE_DOOR_PLAN.md +170 -0
- package/lattice_brain/__init__.py +42 -79
- package/lattice_brain/graph/__init__.py +9 -24
- package/lattice_brain/graph/_kg_common/__init__.py +13 -24
- package/lattice_brain/ingestion/__init__.py +19 -58
- package/lattice_brain/ingestion/pipeline.py +20 -398
- package/lattice_brain/multimodal/__init__.py +7 -18
- package/lattice_brain/multimodal/images.py +6 -264
- package/lattice_brain/multimodal/video.py +8 -247
- package/lattice_brain/runtime/__init__.py +8 -79
- package/lattice_brain/runtime/hooks.py +22 -584
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +7 -80
- package/latticeai/api/health.py +6 -20
- package/latticeai/api/local_files.py +16 -613
- package/latticeai/api/models.py +12 -87
- package/latticeai/api/search.py +17 -255
- package/latticeai/api/tools.py +69 -750
- package/latticeai/api/voice_capture.py +8 -70
- package/latticeai/api/worker_compute.py +842 -0
- package/latticeai/api/worker_seams.py +216 -0
- package/latticeai/app_factory.py +23 -232
- package/latticeai/cli/entrypoint.py +36 -249
- package/latticeai/core/agent_permission.py +16 -91
- package/latticeai/core/messages.py +89 -531
- package/latticeai/runtime/access_runtime.py +0 -14
- package/latticeai/runtime/bootstrap.py +10 -21
- package/latticeai/runtime/brain_runtime.py +35 -43
- package/latticeai/runtime/build_phases/__init__.py +21 -37
- package/latticeai/runtime/build_phases/features.py +99 -389
- package/latticeai/runtime/build_phases/foundation.py +82 -392
- package/latticeai/runtime/build_phases/web.py +91 -408
- package/latticeai/runtime/build_phases/worker_profile.py +244 -0
- package/latticeai/runtime/lifespan_runtime.py +7 -16
- package/latticeai/runtime/platform_services_runtime.py +1 -15
- package/latticeai/runtime/runtime_context.py +13 -130
- package/latticeai/runtime/security_runtime.py +10 -103
- package/latticeai/services/architecture_readiness.py +82 -43
- package/latticeai/services/model_runtime/__init__.py +2 -11
- package/latticeai/services/model_runtime/service.py +4 -28
- package/latticeai/services/p_reinforce.py +10 -261
- package/latticeai/services/product_readiness.py +31 -38
- package/latticeai/services/search_service.py +37 -795
- package/latticeai/services/tool_dispatch.py +30 -386
- package/latticeai/services/voice_capture.py +13 -107
- package/latticeai/tools/__init__.py +22 -49
- package/latticeai/tools/commands.py +0 -163
- package/latticeai/tools/computer.py +0 -39
- package/latticeai/tools/documents.py +1 -134
- package/latticeai/tools/filesystem.py +1 -247
- package/latticeai/tools/knowledge.py +1 -52
- package/latticeai/tools/local_files.py +0 -20
- package/latticeai/worker_app.py +75 -0
- package/package.json +2 -3
- package/requirements.txt +0 -5
- package/scripts/agent_eval.py +16 -28
- package/scripts/brain_quality_eval.py +20 -183
- package/scripts/bump_version.py +5 -5
- package/scripts/check_current_release_docs.mjs +7 -4
- package/scripts/check_openapi_drift.mjs +10 -0
- package/scripts/check_server_i18n.mjs +2 -18
- package/scripts/compose_openapi.py +377 -0
- package/scripts/export_openapi.py +32 -4
- package/scripts/gen_messages_catalog_fixture.py +302 -0
- package/scripts/gen_openapi_fragments.py +365 -0
- package/scripts/gen_redact_fixture.py +196 -0
- package/scripts/gen_worker_allowlist_fixture.py +115 -0
- package/scripts/generate_agent_parity_fixtures.py +33 -14
- package/scripts/openapi_route_families.json +2161 -0
- package/scripts/release_screen_claims.json +54 -0
- package/scripts/run_integration_tests.mjs +134 -29
- package/scripts/run_sidecar_e2e.mjs +104 -11
- package/scripts/wheel_smoke.py +32 -22
- package/src-tauri/Cargo.lock +375 -9
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/src/backend.rs +151 -45
- package/src-tauri/src/main.rs +18 -18
- package/src-tauri/src/topology.rs +58 -88
- package/src-tauri/tauri.conf.json +1 -2
- package/static/app/asset-manifest.json +41 -41
- package/static/app/assets/{Act-DcQizkl1.js → Act-BPcVAbOL.js} +1 -1
- package/static/app/assets/AdminConsole-Bw1ATQL0.js +1 -0
- package/static/app/assets/{Brain-3VCSHFcn.js → Brain-CT92Kos0.js} +2 -2
- package/static/app/assets/{BrainHome-Qm8eaztx.js → BrainHome-CFBkt1K_.js} +1 -1
- package/static/app/assets/{BrainSignals-DS9BtKOW.js → BrainSignals-ReLWF2H8.js} +1 -1
- package/static/app/assets/Capture-BsTokYkk.js +1 -0
- package/static/app/assets/{Chronicle-BGvuAchH.js → Chronicle-B6f0T9id.js} +1 -1
- package/static/app/assets/{CommandPalette-Bqhm0Urn.js → CommandPalette-CuvjTv1u.js} +1 -1
- package/static/app/assets/{Library-BV6NnF0a.js → Library-BGJbG9Hd.js} +1 -1
- package/static/app/assets/{LivingBrain-GzenJchP.js → LivingBrain-DGYK_Jsa.js} +1 -1
- package/static/app/assets/{ProductFlow-DEP6-vML.js → ProductFlow-DXBC6brE.js} +1 -1
- package/static/app/assets/{ReviewCard-CNZ7XjWG.js → ReviewCard-HXRle3qq.js} +2 -2
- package/static/app/assets/System-CMHSO9qM.js +1 -0
- package/static/app/assets/arrow-left-BfmkskWx.js +1 -0
- package/static/app/assets/{bot-B_K1Tdmw.js → bot-Cn8bWRuq.js} +1 -1
- package/static/app/assets/{brain-DWyaV1L1.js → brain-CQJberbE.js} +1 -1
- package/static/app/assets/{button-aTn4s84A.js → button-Ct9f2_oT.js} +1 -1
- package/static/app/assets/circle-check-DruOxB-4.js +1 -0
- package/static/app/assets/{circle-pause-xKgeGXkT.js → circle-pause-CmzC_apg.js} +1 -1
- package/static/app/assets/{circle-play-DkT6tYPX.js → circle-play-D8mW2aQ7.js} +1 -1
- package/static/app/assets/{cpu-85xYObUC.js → cpu-DZcdd0PZ.js} +1 -1
- package/static/app/assets/{download-B5Fm7YXo.js → download-bv1KEPGQ.js} +1 -1
- package/static/app/assets/{folder-open-kk2Xa52u.js → folder-open-d-Pip5gr.js} +1 -1
- package/static/app/assets/{hard-drive-DkA3zBW_.js → hard-drive-D20iavUb.js} +1 -1
- package/static/app/assets/index-D9x-kSNy.css +2 -0
- package/static/app/assets/{index-BMPdTmlY.js → index-Do83hDzJ.js} +3 -3
- package/static/app/assets/{input-B0nRf2jO.js → input-BLXVNmj1.js} +1 -1
- package/static/app/assets/{link-2-Dwb4gnTc.js → link-2-BPJOFlAy.js} +1 -1
- package/static/app/assets/{permissionCopy-CQDUBrOZ.js → permissionCopy-ChdJd493.js} +1 -1
- package/static/app/assets/primitives-Cv5tbZBY.js +1 -0
- package/static/app/assets/search-CT9aho2j.js +1 -0
- package/static/app/assets/{share-2-BsrxFglO.js → share-2-YNX_NtMU.js} +1 -1
- package/static/app/assets/{shield-alert-5BStfp2_.js → shield-alert-DuQ3zrVL.js} +1 -1
- package/static/app/assets/{textarea-Cg8IUA-k.js → textarea-DqwLnli4.js} +1 -1
- package/static/app/assets/{useFocusTrap-CYKvE46M.js → useFocusTrap-ZVI98jaW.js} +1 -1
- package/static/app/assets/{useMutation-CSn9t1op.js → useMutation-CVC4qv_D.js} +1 -1
- package/static/app/assets/{useQuery-CY2OI2uy.js → useQuery-C7BeG4HU.js} +1 -1
- package/static/app/assets/{utils-Ddol2RWD.js → utils-CiFtIdZq.js} +1 -1
- package/static/app/assets/{workspace-BqDwOz_p.js → workspace-DQz9vIId.js} +1 -1
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/archive.py +0 -522
- package/lattice_brain/context.py +0 -325
- package/lattice_brain/conversations.py +0 -382
- package/lattice_brain/core.py +0 -82
- package/lattice_brain/graph/_kg_contract.py +0 -249
- package/lattice_brain/graph/curator.py +0 -676
- package/lattice_brain/graph/discovery.py +0 -595
- package/lattice_brain/graph/discovery_index/__init__.py +0 -35
- package/lattice_brain/graph/discovery_index/cleanup.py +0 -182
- package/lattice_brain/graph/discovery_index/extract.py +0 -137
- package/lattice_brain/graph/discovery_index/scan.py +0 -411
- package/lattice_brain/graph/discovery_index/upsert.py +0 -495
- package/lattice_brain/graph/documents.py +0 -380
- package/lattice_brain/graph/fusion.py +0 -395
- package/lattice_brain/graph/identity.py +0 -175
- package/lattice_brain/graph/image_vectors.py +0 -230
- package/lattice_brain/graph/ingest.py +0 -829
- package/lattice_brain/graph/network.py +0 -205
- package/lattice_brain/graph/proactive.py +0 -724
- package/lattice_brain/graph/projection/__init__.py +0 -42
- package/lattice_brain/graph/projection/curation.py +0 -500
- package/lattice_brain/graph/projection/v2_schema.py +0 -518
- package/lattice_brain/graph/provenance.py +0 -524
- package/lattice_brain/graph/rerank.py +0 -163
- package/lattice_brain/graph/retrieval/__init__.py +0 -54
- package/lattice_brain/graph/retrieval/context.py +0 -197
- package/lattice_brain/graph/retrieval/graph_view.py +0 -319
- package/lattice_brain/graph/retrieval/hybrid.py +0 -488
- package/lattice_brain/graph/retrieval/maintenance.py +0 -121
- package/lattice_brain/graph/retrieval/signals.py +0 -95
- package/lattice_brain/graph/retrieval_docgen.py +0 -253
- package/lattice_brain/graph/retrieval_policy.py +0 -180
- package/lattice_brain/graph/retrieval_reads.py +0 -769
- package/lattice_brain/graph/retrieval_vector/__init__.py +0 -42
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +0 -97
- package/lattice_brain/graph/retrieval_vector/indexing.py +0 -347
- package/lattice_brain/graph/retrieval_vector/search.py +0 -560
- package/lattice_brain/graph/retrieval_vector/status.py +0 -374
- package/lattice_brain/graph/schema.py +0 -792
- package/lattice_brain/graph/store.py +0 -268
- package/lattice_brain/graph/vector_index/__init__.py +0 -85
- package/lattice_brain/graph/vector_index/base.py +0 -167
- package/lattice_brain/graph/vector_index/brute_force.py +0 -110
- package/lattice_brain/graph/vector_index/hnsw.py +0 -290
- package/lattice_brain/graph/vector_index/jobs.py +0 -287
- package/lattice_brain/graph/vector_index/quantized.py +0 -148
- package/lattice_brain/graph/vector_index/selector.py +0 -161
- package/lattice_brain/graph/write_master.py +0 -308
- package/lattice_brain/ingestion/_contract.py +0 -90
- package/lattice_brain/ingestion/folder_scan.py +0 -57
- package/lattice_brain/ingestion/folders.py +0 -258
- package/lattice_brain/ingestion/jobs_api.py +0 -107
- package/lattice_brain/ingestion/routing.py +0 -295
- package/lattice_brain/ingestion_jobs.py +0 -380
- package/lattice_brain/memory.py +0 -75
- package/lattice_brain/portability/__init__.py +0 -90
- package/lattice_brain/portability/_contract.py +0 -42
- package/lattice_brain/portability/backups.py +0 -338
- package/lattice_brain/portability/bundles.py +0 -136
- package/lattice_brain/portability/constants.py +0 -93
- package/lattice_brain/portability/fsops.py +0 -138
- package/lattice_brain/portability/service.py +0 -41
- package/lattice_brain/portability/sharing.py +0 -710
- package/lattice_brain/quality.py +0 -543
- package/lattice_brain/retrieval_benchmark_fixtures.py +0 -92
- package/lattice_brain/runtime/agent_runtime.py +0 -859
- package/lattice_brain/runtime/contracts.py +0 -460
- package/lattice_brain/runtime/multi_agent.py +0 -942
- package/lattice_brain/runtime/statuses.py +0 -10
- package/lattice_brain/sealed_box.py +0 -240
- package/lattice_brain/self_model.py +0 -675
- package/lattice_brain/sensitivity.py +0 -94
- package/lattice_brain/storage/__init__.py +0 -22
- package/lattice_brain/storage/base.py +0 -100
- package/lattice_brain/storage/docker.py +0 -105
- package/lattice_brain/storage/factory.py +0 -31
- package/lattice_brain/storage/migration.py +0 -191
- package/lattice_brain/storage/postgres.py +0 -123
- package/lattice_brain/storage/sqlite.py +0 -143
- package/lattice_brain/synthesis.py +0 -824
- package/lattice_brain/workflow.py +0 -497
- package/latticeai/api/admin.py +0 -471
- package/latticeai/api/agent_registry.py +0 -105
- package/latticeai/api/agents.py +0 -228
- package/latticeai/api/auth.py +0 -383
- package/latticeai/api/automation_intelligence.py +0 -401
- package/latticeai/api/brain_intelligence.py +0 -199
- package/latticeai/api/browser.py +0 -493
- package/latticeai/api/change_proposals.py +0 -89
- package/latticeai/api/chat.py +0 -572
- package/latticeai/api/chat_agent_http.py +0 -892
- package/latticeai/api/chat_contracts.py +0 -73
- package/latticeai/api/chat_documents.py +0 -276
- package/latticeai/api/chat_helpers.py +0 -460
- package/latticeai/api/chat_history.py +0 -101
- package/latticeai/api/chat_hybrid.py +0 -113
- package/latticeai/api/chat_intents.py +0 -646
- package/latticeai/api/chat_stream.py +0 -216
- package/latticeai/api/chronicle.py +0 -63
- package/latticeai/api/command_center.py +0 -51
- package/latticeai/api/computer_use.py +0 -474
- package/latticeai/api/evidence_actions.py +0 -48
- package/latticeai/api/features.py +0 -70
- package/latticeai/api/funnel_metrics.py +0 -31
- package/latticeai/api/garden.py +0 -34
- package/latticeai/api/hooks.py +0 -165
- package/latticeai/api/index_jobs.py +0 -145
- package/latticeai/api/invitations.py +0 -100
- package/latticeai/api/knowledge_graph.py +0 -536
- package/latticeai/api/marketplace.py +0 -105
- package/latticeai/api/mcp.py +0 -482
- package/latticeai/api/memory.py +0 -270
- package/latticeai/api/network.py +0 -81
- package/latticeai/api/network_boundary.py +0 -225
- package/latticeai/api/permission_mode.py +0 -61
- package/latticeai/api/permissions.py +0 -436
- package/latticeai/api/plugins.py +0 -126
- package/latticeai/api/portability.py +0 -391
- package/latticeai/api/project_sessions.py +0 -114
- package/latticeai/api/realtime.py +0 -118
- package/latticeai/api/review_queue.py +0 -364
- package/latticeai/api/security_dashboard.py +0 -604
- package/latticeai/api/setup.py +0 -319
- package/latticeai/api/static_routes.py +0 -354
- package/latticeai/api/ui_redirects.py +0 -26
- package/latticeai/api/workflow_designer.py +0 -394
- package/latticeai/api/workspace.py +0 -856
- package/latticeai/api/workspace_scope.py +0 -125
- package/latticeai/core/agent/__init__.py +0 -93
- package/latticeai/core/agent/_contract.py +0 -79
- package/latticeai/core/agent/context.py +0 -57
- package/latticeai/core/agent/deps.py +0 -125
- package/latticeai/core/agent/execution.py +0 -622
- package/latticeai/core/agent/planning.py +0 -145
- package/latticeai/core/agent/recovery.py +0 -157
- package/latticeai/core/agent/runtime.py +0 -210
- package/latticeai/core/agent/verification.py +0 -231
- package/latticeai/core/agent_eval.py +0 -739
- package/latticeai/core/agent_helpers.py +0 -493
- package/latticeai/core/agent_profiles.py +0 -110
- package/latticeai/core/agent_prompts.py +0 -171
- package/latticeai/core/agent_registry.py +0 -232
- package/latticeai/core/agent_state.py +0 -41
- package/latticeai/core/agent_trace.py +0 -104
- package/latticeai/core/artifact_ledger.py +0 -109
- package/latticeai/core/audit.py +0 -260
- package/latticeai/core/builtin_hooks.py +0 -105
- package/latticeai/core/context_builder.py +0 -394
- package/latticeai/core/document_generator.py +0 -103
- package/latticeai/core/enterprise.py +0 -154
- package/latticeai/core/enterprise_admin.py +0 -158
- package/latticeai/core/file_generation/__init__.py +0 -115
- package/latticeai/core/file_generation/bundles.py +0 -76
- package/latticeai/core/file_generation/extraction.py +0 -154
- package/latticeai/core/file_generation/inference.py +0 -235
- package/latticeai/core/file_generation/orchestration.py +0 -152
- package/latticeai/core/file_generation/prompting.py +0 -117
- package/latticeai/core/file_generation/repair.py +0 -114
- package/latticeai/core/file_generation/sanitize.py +0 -61
- package/latticeai/core/file_generation/validation.py +0 -201
- package/latticeai/core/invitations.py +0 -132
- package/latticeai/core/legacy_compatibility.py +0 -243
- package/latticeai/core/logging_safety.py +0 -46
- package/latticeai/core/marketplace.py +0 -293
- package/latticeai/core/mcp_catalog.py +0 -452
- package/latticeai/core/mcp_registry.py +0 -506
- package/latticeai/core/network_boundary.py +0 -168
- package/latticeai/core/oidc.py +0 -208
- package/latticeai/core/plugins.py +0 -432
- package/latticeai/core/product_hardening.py +0 -218
- package/latticeai/core/project_sessions.py +0 -337
- package/latticeai/core/realtime.py +0 -238
- package/latticeai/core/run_explain.py +0 -426
- package/latticeai/core/run_store.py +0 -252
- package/latticeai/core/timezones.py +0 -80
- package/latticeai/core/workspace_computer_memory.py +0 -84
- package/latticeai/core/workspace_graph_trace.py +0 -155
- package/latticeai/core/workspace_indexing.py +0 -102
- package/latticeai/core/workspace_memory.py +0 -77
- package/latticeai/core/workspace_onboarding.py +0 -104
- package/latticeai/core/workspace_os.py +0 -978
- package/latticeai/core/workspace_os_constants.py +0 -126
- package/latticeai/core/workspace_os_state.py +0 -180
- package/latticeai/core/workspace_os_utils.py +0 -103
- package/latticeai/core/workspace_permissions.py +0 -101
- package/latticeai/core/workspace_plugins.py +0 -97
- package/latticeai/core/workspace_relationships.py +0 -99
- package/latticeai/core/workspace_reorganization.py +0 -335
- package/latticeai/core/workspace_review_items.py +0 -112
- package/latticeai/core/workspace_runs.py +0 -726
- package/latticeai/core/workspace_skills.py +0 -109
- package/latticeai/core/workspace_snapshots.py +0 -198
- package/latticeai/core/workspace_timeline.py +0 -110
- package/latticeai/integrations/__init__.py +0 -0
- package/latticeai/integrations/telegram_bot/__init__.py +0 -123
- package/latticeai/integrations/telegram_bot/__main__.py +0 -17
- package/latticeai/integrations/telegram_bot/config.py +0 -86
- package/latticeai/integrations/telegram_bot/dispatch.py +0 -311
- package/latticeai/integrations/telegram_bot/flows.py +0 -478
- package/latticeai/integrations/telegram_bot/helpers.py +0 -322
- package/latticeai/integrations/telegram_bot/screens.py +0 -394
- package/latticeai/runtime/audit_runtime.py +0 -76
- package/latticeai/runtime/automation_runtime.py +0 -81
- package/latticeai/runtime/chat_wiring.py +0 -141
- package/latticeai/runtime/context_runtime.py +0 -66
- package/latticeai/runtime/feature_toggle_wiring.py +0 -157
- package/latticeai/runtime/history_runtime.py +0 -163
- package/latticeai/runtime/history_writer.py +0 -138
- package/latticeai/runtime/hooks_runtime.py +0 -77
- package/latticeai/runtime/model_wiring.py +0 -68
- package/latticeai/runtime/namespace_runtime.py +0 -163
- package/latticeai/runtime/network_boundary_wiring.py +0 -117
- package/latticeai/runtime/network_config_runtime.py +0 -56
- package/latticeai/runtime/permission_mode_wiring.py +0 -112
- package/latticeai/runtime/persistence_runtime.py +0 -159
- package/latticeai/runtime/platform_runtime_wiring.py +0 -89
- package/latticeai/runtime/review_wiring.py +0 -42
- package/latticeai/runtime/router_registration.py +0 -693
- package/latticeai/runtime/service_singletons.py +0 -55
- package/latticeai/runtime/sso_config_runtime.py +0 -128
- package/latticeai/runtime/user_key_runtime.py +0 -106
- package/latticeai/runtime/web_runtime.py +0 -92
- package/latticeai/server_app.py +0 -51
- package/latticeai/services/app_context.py +0 -130
- package/latticeai/services/automation_execution.py +0 -266
- package/latticeai/services/automation_intelligence.py +0 -614
- package/latticeai/services/brain_automation.py +0 -191
- package/latticeai/services/brain_intelligence/__init__.py +0 -58
- package/latticeai/services/brain_intelligence/_contract.py +0 -71
- package/latticeai/services/brain_intelligence/consistency.py +0 -193
- package/latticeai/services/brain_intelligence/constants.py +0 -47
- package/latticeai/services/brain_intelligence/digest.py +0 -258
- package/latticeai/services/brain_intelligence/health.py +0 -331
- package/latticeai/services/brain_intelligence/proposals.py +0 -259
- package/latticeai/services/brain_intelligence/sampling.py +0 -84
- package/latticeai/services/brain_intelligence/service.py +0 -48
- package/latticeai/services/change_proposals.py +0 -471
- package/latticeai/services/chat_service.py +0 -243
- package/latticeai/services/chronicle.py +0 -555
- package/latticeai/services/cloud_egress_audit.py +0 -85
- package/latticeai/services/cloud_extraction.py +0 -129
- package/latticeai/services/cloud_streaming.py +0 -268
- package/latticeai/services/cloud_token_guard.py +0 -84
- package/latticeai/services/command_center.py +0 -548
- package/latticeai/services/evidence_actions.py +0 -258
- package/latticeai/services/feature_toggles.py +0 -502
- package/latticeai/services/folder_watch.py +0 -520
- package/latticeai/services/funnel_metrics.py +0 -307
- package/latticeai/services/hybrid_chat.py +0 -316
- package/latticeai/services/hybrid_context.py +0 -228
- package/latticeai/services/hybrid_policy.py +0 -129
- package/latticeai/services/interop_bridges.py +0 -978
- package/latticeai/services/local_knowledge.py +0 -465
- package/latticeai/services/memory_service/__init__.py +0 -52
- package/latticeai/services/memory_service/_contract.py +0 -100
- package/latticeai/services/memory_service/brief.py +0 -431
- package/latticeai/services/memory_service/constants.py +0 -57
- package/latticeai/services/memory_service/maintenance.py +0 -138
- package/latticeai/services/memory_service/manager.py +0 -186
- package/latticeai/services/memory_service/proof.py +0 -136
- package/latticeai/services/memory_service/recall.py +0 -225
- package/latticeai/services/memory_service/service.py +0 -48
- package/latticeai/services/memory_service/stores.py +0 -110
- package/latticeai/services/mode_store.py +0 -132
- package/latticeai/services/model_recommendation.py +0 -224
- package/latticeai/services/model_runtime/cloud.py +0 -87
- package/latticeai/services/network_boundary_service.py +0 -117
- package/latticeai/services/obsidian_bridge.py +0 -609
- package/latticeai/services/openai_compatible_adapter.py +0 -101
- package/latticeai/services/permission_mode_service.py +0 -122
- package/latticeai/services/platform_runtime.py +0 -366
- package/latticeai/services/review_queue.py +0 -380
- package/latticeai/services/router_context.py +0 -59
- package/latticeai/services/run_executor.py +0 -387
- package/latticeai/services/self_model_service.py +0 -171
- package/latticeai/services/setup_detection.py +0 -147
- package/latticeai/services/triggers.py +0 -378
- package/latticeai/services/upload_service.py +0 -172
- package/latticeai/services/workspace_service.py +0 -165
- package/latticeai/setup/__init__.py +0 -25
- package/latticeai/setup/auto_setup.py +0 -846
- package/latticeai/setup/demo_corpus.py +0 -98
- package/latticeai/setup/wizard/__init__.py +0 -126
- package/latticeai/setup/wizard/catalog.py +0 -175
- package/latticeai/setup/wizard/detect.py +0 -302
- package/latticeai/setup/wizard/install.py +0 -348
- package/latticeai/setup/wizard/paths.py +0 -165
- package/latticeai/setup/wizard/plans.py +0 -74
- package/latticeai/setup/wizard/recommend.py +0 -320
- package/scripts/bench_agent_smoke.py +0 -409
- package/scripts/bench_models.py +0 -540
- package/scripts/bench_vector_index.py +0 -295
- package/scripts/funnel_soft_gate.py +0 -192
- package/scripts/generate_agent_loop_fixtures.py +0 -994
- package/scripts/generate_rust_parity_fixtures.py +0 -908
- package/scripts/migrate_brain_storage.py +0 -57
- package/scripts/parity_fixture_corpus_context.py +0 -162
- package/scripts/parity_fixture_corpus_docgen.py +0 -341
- package/scripts/profile_kg.py +0 -355
- package/server.py +0 -30
- package/static/app/assets/AdminConsole-cf4npybT.js +0 -1
- package/static/app/assets/Capture-DiQ219jW.js +0 -1
- package/static/app/assets/System-CieofHQa.js +0 -1
- package/static/app/assets/arrow-left-kfsrk0mv.js +0 -1
- package/static/app/assets/circle-check-qqLug9nU.js +0 -1
- package/static/app/assets/index-DxmOfNRi.css +0 -2
- package/static/app/assets/primitives-SNp0LRJz.js +0 -1
- package/static/app/assets/search-BcHqkjoy.js +0 -1
package/scripts/bench_models.py
DELETED
|
@@ -1,540 +0,0 @@
|
|
|
1
|
-
#!/usr/bin/env python3
|
|
2
|
-
"""Model robustness benchmark harness for the Lattice agent loop.
|
|
3
|
-
|
|
4
|
-
What it measures
|
|
5
|
-
================
|
|
6
|
-
The product claim is that the Brain stays durable "across any AI model" — a
|
|
7
|
-
weaker model may emit sloppier output, but the agent loop repairs it and still
|
|
8
|
-
completes the task. This harness turns that claim into a **matrix** of three
|
|
9
|
-
numbers per model tier:
|
|
10
|
-
|
|
11
|
-
* ``success_rate`` — fraction of model outputs the loop could turn into a valid
|
|
12
|
-
action (i.e. the task can proceed).
|
|
13
|
-
* ``repair_rate`` — of the successful parses, the fraction that ONLY succeeded
|
|
14
|
-
because the loop's tolerant parser had to repair the output (fences, prose,
|
|
15
|
-
trailing commas, ``<think>`` blocks, Python-dict literals). High repair_rate
|
|
16
|
-
= "this tier leans hard on the loop's robustness".
|
|
17
|
-
* ``latency_ms`` — see the honesty note below.
|
|
18
|
-
|
|
19
|
-
It measures **real code**: every output is fed through
|
|
20
|
-
``latticeai.core.agent.extract_action_details`` — the exact parser/repair
|
|
21
|
-
function the production loop uses (``agent.py`` calls it in plan/execute/verify).
|
|
22
|
-
An ``agent-loop`` reference row additionally runs the real
|
|
23
|
-
``latticeai.core.agent_eval.run_agent_eval`` state machine over its full
|
|
24
|
-
scripted scenario suite.
|
|
25
|
-
|
|
26
|
-
Three modes
|
|
27
|
-
===========
|
|
28
|
-
1. **scripted** (default, always runnable, no model/network): a curated corpus
|
|
29
|
-
of model-realistic outputs per quality tier (frontier / mid-local /
|
|
30
|
-
weak-local). Proves the harness works and exposes the loop's repair boundary
|
|
31
|
-
deterministically.
|
|
32
|
-
2. **live** (opt-in, ``--live-endpoint``): sends a fixed set of agent prompts to
|
|
33
|
-
an OpenAI-compatible local endpoint (e.g. LM Studio / llama.cpp / vLLM) and
|
|
34
|
-
runs the *real* completions through the same parser, measuring true
|
|
35
|
-
end-to-end generation latency. Falls back to scripted with an honest message
|
|
36
|
-
if the endpoint is unreachable.
|
|
37
|
-
3. **filegen** (opt-in, ``--filegen``): the weekly multi-model file-generation
|
|
38
|
-
report. Discovers *installed* local gemma/qwen/llama MLX models via the
|
|
39
|
-
product's own model catalog + HF download checks, loads each with the real
|
|
40
|
-
``LLMRouter``, and drives the real ``generate_file_content`` pipeline
|
|
41
|
-
(prompt → extract → validate → retry → repair) for each canonical file type
|
|
42
|
-
(html/css/js/py/json/md). Reports a model × filetype success matrix.
|
|
43
|
-
**FAIL-OPEN by design**: no models installed (or a load failure) yields a
|
|
44
|
-
clear skip report and exit code 0 — this mode is a scheduled/manual report,
|
|
45
|
-
never a CI gate.
|
|
46
|
-
|
|
47
|
-
Honesty note on latency
|
|
48
|
-
========================
|
|
49
|
-
In **scripted** mode ``latency_ms`` is the *parse+repair* cost only
|
|
50
|
-
(microseconds); it is NOT model inference time and must not be read as such.
|
|
51
|
-
Real generation latency is only meaningful in **live** mode.
|
|
52
|
-
|
|
53
|
-
Usage
|
|
54
|
-
=====
|
|
55
|
-
.venv/bin/python scripts/bench_models.py # scripted matrix
|
|
56
|
-
.venv/bin/python scripts/bench_models.py --json out.json # + machine output
|
|
57
|
-
.venv/bin/python scripts/bench_models.py \
|
|
58
|
-
--live-endpoint http://127.0.0.1:1234/v1 --model my-local-model
|
|
59
|
-
.venv/bin/python scripts/bench_models.py --filegen # weekly filegen report
|
|
60
|
-
.venv/bin/python scripts/bench_models.py --filegen --json filegen_report.json
|
|
61
|
-
"""
|
|
62
|
-
|
|
63
|
-
from __future__ import annotations
|
|
64
|
-
|
|
65
|
-
import argparse
|
|
66
|
-
import asyncio
|
|
67
|
-
import json
|
|
68
|
-
import statistics
|
|
69
|
-
import sys
|
|
70
|
-
import time
|
|
71
|
-
import urllib.error
|
|
72
|
-
import urllib.request
|
|
73
|
-
from pathlib import Path
|
|
74
|
-
from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple
|
|
75
|
-
|
|
76
|
-
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
|
77
|
-
|
|
78
|
-
from latticeai.core.agent import extract_action_details # noqa: E402
|
|
79
|
-
from latticeai.core.agent_eval import run_agent_eval # noqa: E402
|
|
80
|
-
from latticeai.core.file_generation import ( # noqa: E402
|
|
81
|
-
generate_file_content,
|
|
82
|
-
validate_file_content,
|
|
83
|
-
)
|
|
84
|
-
|
|
85
|
-
# ── Scripted corpora ─────────────────────────────────────────────────────
|
|
86
|
-
# Each entry is one model output for a canonical agent turn. The tiers encode
|
|
87
|
-
# how the SAME intent degrades as model quality drops. All "should_parse=True"
|
|
88
|
-
# entries are recoverable by the real loop; "should_parse=False" entries are
|
|
89
|
-
# genuinely broken (past the repair boundary) and SHOULD fail — that is what
|
|
90
|
-
# makes success_rate < 1.0 meaningful rather than a rigged demo.
|
|
91
|
-
|
|
92
|
-
_CLEAN = [
|
|
93
|
-
('{"action": "plan", "goal": "ingest", "steps": [{"action": "write_file"}]}', True),
|
|
94
|
-
('{"action": "write_file", "args": {"path": "note.txt", "content": "hi"}}', True),
|
|
95
|
-
('{"action": "final", "message": "done"}', True),
|
|
96
|
-
('{"action": "verdict", "verdict": "PASS", "next_state": "DONE", "reason": "ok"}', True),
|
|
97
|
-
('{"action": "knowledge_graph_search", "args": {"query": "roadmap"}}', True),
|
|
98
|
-
('{"action": "read_file", "args": {"path": "README.md"}}', True),
|
|
99
|
-
]
|
|
100
|
-
|
|
101
|
-
_MID = [
|
|
102
|
-
# markdown-fenced JSON (very common)
|
|
103
|
-
('```json\n{"action": "plan", "goal": "ingest", "steps": []}\n```', True),
|
|
104
|
-
('```\n{"action": "write_file", "args": {"path": "a.txt", "content": "x"}}\n```', True),
|
|
105
|
-
# prose preamble then the object
|
|
106
|
-
('Sure! Here is the next step:\n{"action": "final", "message": "done"}', True),
|
|
107
|
-
# trailing comma before closing brace
|
|
108
|
-
('{"action": "read_file", "args": {"path": "README.md",}}', True),
|
|
109
|
-
# trailing comma in array
|
|
110
|
-
('{"action": "plan", "goal": "g", "steps": [{"action": "read_file"},]}', True),
|
|
111
|
-
('```json\n{"action": "verdict", "verdict": "PASS", "next_state": "DONE", "reason": "ok"}\n```', True),
|
|
112
|
-
]
|
|
113
|
-
|
|
114
|
-
_WEAK = [
|
|
115
|
-
# <think> reasoning block that itself contains braces, then the action
|
|
116
|
-
('<think>I should write the file {maybe}</think>\n{"action": "write_file", "args": {"path": "n.txt", "content": "c"}}', True),
|
|
117
|
-
('<reasoning>ok</reasoning> {"action": "final", "message": "done"}', True),
|
|
118
|
-
# Python dict literal (single quotes, True) — ast.literal_eval path
|
|
119
|
-
("{'action': 'read_file', 'args': {'path': 'x.txt'}}", True),
|
|
120
|
-
("{'action': 'verdict', 'verdict': 'PASS', 'next_state': 'DONE', 'reason': 'ok'}", True),
|
|
121
|
-
# fenced + trailing comma + prose all at once
|
|
122
|
-
('Here you go:\n```json\n{"action": "plan", "goal": "g", "steps": [],}\n```', True),
|
|
123
|
-
# genuinely broken: pure prose, no JSON object at all -> must fail
|
|
124
|
-
("I think we are done here, nothing else to do.", False),
|
|
125
|
-
# genuinely broken: object but missing the required "action" field
|
|
126
|
-
('{"message": "done", "status": "ok"}', False),
|
|
127
|
-
# genuinely broken: truncated / unbalanced braces past repair
|
|
128
|
-
('{"action": "write_file", "args": {"path": "n.txt"', False),
|
|
129
|
-
]
|
|
130
|
-
|
|
131
|
-
_PROFILES: Dict[str, List[Tuple[str, bool]]] = {
|
|
132
|
-
"frontier (clean JSON)": _CLEAN,
|
|
133
|
-
"mid-local (fenced/prose/commas)": _MID,
|
|
134
|
-
"weak-local (think/py-literal/broken)": _WEAK,
|
|
135
|
-
}
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
def _bench_corpus(corpus: List[Tuple[str, bool]]) -> Dict[str, Any]:
|
|
139
|
-
total = len(corpus)
|
|
140
|
-
parsed = 0
|
|
141
|
-
repaired = 0
|
|
142
|
-
latencies: List[float] = []
|
|
143
|
-
mismatches: List[str] = []
|
|
144
|
-
repair_kinds: Dict[str, int] = {}
|
|
145
|
-
for raw, should_parse in corpus:
|
|
146
|
-
t0 = time.perf_counter()
|
|
147
|
-
try:
|
|
148
|
-
_action, repairs = extract_action_details(raw)
|
|
149
|
-
ok = True
|
|
150
|
-
except (ValueError, Exception): # noqa: BLE001 - parser raises ValueError
|
|
151
|
-
ok = False
|
|
152
|
-
repairs = []
|
|
153
|
-
latencies.append((time.perf_counter() - t0) * 1000.0)
|
|
154
|
-
if ok:
|
|
155
|
-
parsed += 1
|
|
156
|
-
if repairs:
|
|
157
|
-
repaired += 1
|
|
158
|
-
for kind in repairs:
|
|
159
|
-
repair_kinds[kind] = repair_kinds.get(kind, 0) + 1
|
|
160
|
-
if ok != should_parse:
|
|
161
|
-
mismatches.append(raw[:60])
|
|
162
|
-
return {
|
|
163
|
-
"total": total,
|
|
164
|
-
"parsed": parsed,
|
|
165
|
-
"repaired": repaired,
|
|
166
|
-
"success_rate": round(parsed / total, 4) if total else 0.0,
|
|
167
|
-
"repair_rate": round(repaired / parsed, 4) if parsed else 0.0,
|
|
168
|
-
"latency_ms_mean": round(statistics.fmean(latencies), 4) if latencies else 0.0,
|
|
169
|
-
"repair_kinds": repair_kinds,
|
|
170
|
-
# A healthy harness has zero mismatches: every "should_parse" call
|
|
171
|
-
# matched reality. Non-empty means the corpus/parser disagree.
|
|
172
|
-
"harness_mismatches": mismatches,
|
|
173
|
-
}
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
# ── Live (OpenAI-compatible) path ────────────────────────────────────────
|
|
177
|
-
_LIVE_PROMPTS = [
|
|
178
|
-
"Return ONLY one JSON object for the next agent step: read the file README.md.",
|
|
179
|
-
"Return ONLY one JSON object: plan to ingest a document, then finish.",
|
|
180
|
-
"Return ONLY one JSON object: write 'hello' to notes.txt.",
|
|
181
|
-
"Return ONLY one JSON object: a PASS verdict moving the loop to DONE.",
|
|
182
|
-
"Return ONLY one JSON object: search the knowledge graph for 'roadmap'.",
|
|
183
|
-
]
|
|
184
|
-
|
|
185
|
-
_SYSTEM = (
|
|
186
|
-
"You are the executor of an agent loop. Every reply MUST be exactly one "
|
|
187
|
-
'JSON object with an "action" field and nothing else.'
|
|
188
|
-
)
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
def _openai_chat(base_url: str, model: str, prompt: str, *, timeout: float) -> Tuple[str, float]:
|
|
192
|
-
url = base_url.rstrip("/") + "/chat/completions"
|
|
193
|
-
body = json.dumps(
|
|
194
|
-
{
|
|
195
|
-
"model": model,
|
|
196
|
-
"messages": [
|
|
197
|
-
{"role": "system", "content": _SYSTEM},
|
|
198
|
-
{"role": "user", "content": prompt},
|
|
199
|
-
],
|
|
200
|
-
"temperature": 0.2,
|
|
201
|
-
"max_tokens": 256,
|
|
202
|
-
}
|
|
203
|
-
).encode("utf-8")
|
|
204
|
-
req = urllib.request.Request( # noqa: S310 - operator-supplied local endpoint
|
|
205
|
-
url, data=body, headers={"Content-Type": "application/json"}, method="POST"
|
|
206
|
-
)
|
|
207
|
-
t0 = time.perf_counter()
|
|
208
|
-
with urllib.request.urlopen(req, timeout=timeout) as res: # noqa: S310
|
|
209
|
-
payload = json.loads(res.read().decode("utf-8", errors="replace"))
|
|
210
|
-
latency = (time.perf_counter() - t0) * 1000.0
|
|
211
|
-
text = payload["choices"][0]["message"]["content"]
|
|
212
|
-
return text, latency
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
def _bench_live(base_url: str, model: str, *, timeout: float) -> Dict[str, Any]:
|
|
216
|
-
total = 0
|
|
217
|
-
parsed = 0
|
|
218
|
-
repaired = 0
|
|
219
|
-
latencies: List[float] = []
|
|
220
|
-
for prompt in _LIVE_PROMPTS:
|
|
221
|
-
total += 1
|
|
222
|
-
try:
|
|
223
|
-
text, latency = _openai_chat(base_url, model, prompt, timeout=timeout)
|
|
224
|
-
except (urllib.error.URLError, OSError, KeyError, ValueError) as exc:
|
|
225
|
-
return {"error": f"live endpoint unreachable/invalid: {exc}"}
|
|
226
|
-
latencies.append(latency)
|
|
227
|
-
try:
|
|
228
|
-
_action, repairs = extract_action_details(text)
|
|
229
|
-
parsed += 1
|
|
230
|
-
if repairs:
|
|
231
|
-
repaired += 1
|
|
232
|
-
except Exception: # noqa: BLE001
|
|
233
|
-
pass
|
|
234
|
-
return {
|
|
235
|
-
"total": total,
|
|
236
|
-
"parsed": parsed,
|
|
237
|
-
"repaired": repaired,
|
|
238
|
-
"success_rate": round(parsed / total, 4) if total else 0.0,
|
|
239
|
-
"repair_rate": round(repaired / parsed, 4) if parsed else 0.0,
|
|
240
|
-
"latency_ms_mean": round(statistics.fmean(latencies), 2) if latencies else 0.0,
|
|
241
|
-
}
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
# ── Filegen benchmark (weekly multi-model report, fail-open) ─────────────
|
|
245
|
-
# One canonical request per supported file type. Requests are phrased the way
|
|
246
|
-
# users actually ask, so the report measures the real prompt+pipeline, not a
|
|
247
|
-
# synthetic best case.
|
|
248
|
-
FILEGEN_TARGETS: List[Tuple[str, str]] = [
|
|
249
|
-
("bench_page.html", "간단한 자기소개 웹 페이지를 만들어줘. 제목, 소개 문단, 연락처 목록을 포함해."),
|
|
250
|
-
("bench_styles.css", "Create a stylesheet with body typography, a header rule and a .card class."),
|
|
251
|
-
("bench_app.js", "Create a small browser script that renders a todo list and lets the user add items."),
|
|
252
|
-
("bench_tool.py", "Create a Python script that takes a CSV file path from argv and prints the row count."),
|
|
253
|
-
("bench_data.json", "Create a JSON document describing three sample books with title, author and year."),
|
|
254
|
-
("bench_notes.md", "Create a Markdown document with a title, two sections and a bullet list."),
|
|
255
|
-
]
|
|
256
|
-
|
|
257
|
-
_FILEGEN_FAMILIES = ("gemma", "qwen", "llama")
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
def discover_filegen_models() -> List[Dict[str, str]]:
|
|
261
|
-
"""Installed local MLX models from the product catalog (gemma/qwen/llama).
|
|
262
|
-
|
|
263
|
-
Uses the same catalog + on-disk checks the runtime uses (``~/.ltcai``
|
|
264
|
-
model dir or the HF hub cache). Any import/probe failure is fail-open:
|
|
265
|
-
an empty list, never an exception.
|
|
266
|
-
"""
|
|
267
|
-
try:
|
|
268
|
-
from latticeai.models.router import (
|
|
269
|
-
_looks_like_hf_model_dir,
|
|
270
|
-
hf_cache_model_dir,
|
|
271
|
-
hf_model_dir,
|
|
272
|
-
)
|
|
273
|
-
from latticeai.services.model_catalog import ENGINE_MODEL_CATALOG
|
|
274
|
-
except Exception as exc: # noqa: BLE001 - discovery must never crash the report
|
|
275
|
-
print(f"filegen: model catalog unavailable ({exc}); no models discovered")
|
|
276
|
-
return []
|
|
277
|
-
found: List[Dict[str, str]] = []
|
|
278
|
-
for entry in ENGINE_MODEL_CATALOG.get("local_mlx", []):
|
|
279
|
-
model_id = str(entry.get("id") or "")
|
|
280
|
-
lowered = model_id.lower()
|
|
281
|
-
family = next((f for f in _FILEGEN_FAMILIES if f in lowered), None)
|
|
282
|
-
if not family:
|
|
283
|
-
continue
|
|
284
|
-
try:
|
|
285
|
-
downloaded = (
|
|
286
|
-
_looks_like_hf_model_dir(hf_model_dir(model_id))
|
|
287
|
-
or hf_cache_model_dir(model_id) is not None
|
|
288
|
-
)
|
|
289
|
-
except Exception: # noqa: BLE001
|
|
290
|
-
downloaded = False
|
|
291
|
-
if downloaded:
|
|
292
|
-
found.append({"id": model_id, "family": family})
|
|
293
|
-
return found
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
async def bench_filegen_model(
|
|
297
|
-
model_label: str,
|
|
298
|
-
generate_async: Callable[[str], Awaitable[str]],
|
|
299
|
-
targets: Optional[List[Tuple[str, str]]] = None,
|
|
300
|
-
) -> Dict[str, Any]:
|
|
301
|
-
"""Drive the real file-generation pipeline for every target file type.
|
|
302
|
-
|
|
303
|
-
``generate_async`` is the same ``context -> raw model text`` callable shape
|
|
304
|
-
the chat layer injects, so this measures exactly what production runs
|
|
305
|
-
(extraction, validation, corrective retry, deterministic repair included).
|
|
306
|
-
"""
|
|
307
|
-
rows: List[Dict[str, Any]] = []
|
|
308
|
-
for target_path, user_request in (targets or FILEGEN_TARGETS):
|
|
309
|
-
t0 = time.perf_counter()
|
|
310
|
-
content, meta = await generate_file_content(
|
|
311
|
-
generate_async, target_path=target_path, user_request=user_request,
|
|
312
|
-
)
|
|
313
|
-
elapsed_ms = round((time.perf_counter() - t0) * 1000.0, 1)
|
|
314
|
-
valid, reason = validate_file_content(content, target_path)
|
|
315
|
-
attempts = meta.get("attempts") or []
|
|
316
|
-
rows.append({
|
|
317
|
-
"target": target_path,
|
|
318
|
-
"type": target_path.rsplit(".", 1)[-1],
|
|
319
|
-
"valid": valid, # valid after sanitize/repair
|
|
320
|
-
"reason": reason,
|
|
321
|
-
"clean_first_try": bool(attempts and attempts[0].get("valid")),
|
|
322
|
-
"repaired": bool(meta.get("repaired")),
|
|
323
|
-
"attempts": len(attempts),
|
|
324
|
-
"latency_ms": elapsed_ms,
|
|
325
|
-
"bytes": len(content.encode("utf-8", errors="replace")),
|
|
326
|
-
})
|
|
327
|
-
total = len(rows)
|
|
328
|
-
valid_count = sum(1 for r in rows if r["valid"])
|
|
329
|
-
clean_count = sum(1 for r in rows if r["valid"] and not r["repaired"])
|
|
330
|
-
return {
|
|
331
|
-
"model": model_label,
|
|
332
|
-
"targets": rows,
|
|
333
|
-
"total": total,
|
|
334
|
-
"success_rate": round(valid_count / total, 4) if total else 0.0,
|
|
335
|
-
# Of the valid files, how many did NOT need the deterministic repair
|
|
336
|
-
# fallback — the honest "the model itself produced usable output" rate.
|
|
337
|
-
"clean_rate": round(clean_count / total, 4) if total else 0.0,
|
|
338
|
-
}
|
|
339
|
-
|
|
340
|
-
|
|
341
|
-
def _make_router_generate(router: Any, model_id: str) -> Callable[[str], Awaitable[str]]:
|
|
342
|
-
"""Mirror the direct chat path's generation call (chat_intents)."""
|
|
343
|
-
|
|
344
|
-
async def _generate(context: str) -> str:
|
|
345
|
-
return str(
|
|
346
|
-
await router.generate_as(
|
|
347
|
-
model_id,
|
|
348
|
-
message="Return only the requested file content.",
|
|
349
|
-
context=context,
|
|
350
|
-
max_tokens=4096,
|
|
351
|
-
temperature=0.2,
|
|
352
|
-
)
|
|
353
|
-
)
|
|
354
|
-
|
|
355
|
-
return _generate
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
async def _filegen_run_models(models: List[Dict[str, str]]) -> List[Dict[str, Any]]:
|
|
359
|
-
from latticeai.models.router import LLMRouter
|
|
360
|
-
|
|
361
|
-
router = LLMRouter()
|
|
362
|
-
results: List[Dict[str, Any]] = []
|
|
363
|
-
for model in models:
|
|
364
|
-
model_id = model["id"]
|
|
365
|
-
try:
|
|
366
|
-
await router.load_model(model_id)
|
|
367
|
-
except Exception as exc: # noqa: BLE001 - fail-open per model
|
|
368
|
-
results.append({
|
|
369
|
-
"model": model_id, "family": model["family"],
|
|
370
|
-
"skipped": True, "reason": f"load failed: {exc}",
|
|
371
|
-
})
|
|
372
|
-
continue
|
|
373
|
-
result = await bench_filegen_model(model_id, _make_router_generate(router, model_id))
|
|
374
|
-
result["family"] = model["family"]
|
|
375
|
-
results.append(result)
|
|
376
|
-
try:
|
|
377
|
-
router.unload_model(model_id)
|
|
378
|
-
except Exception: # noqa: BLE001
|
|
379
|
-
pass
|
|
380
|
-
return results
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
def run_filegen_benchmark(
|
|
384
|
-
models: Optional[List[Dict[str, str]]] = None,
|
|
385
|
-
) -> Dict[str, Any]:
|
|
386
|
-
"""Weekly model × filetype report. Fail-open: never raises, never gates."""
|
|
387
|
-
if models is None:
|
|
388
|
-
models = discover_filegen_models()
|
|
389
|
-
if not models:
|
|
390
|
-
return {
|
|
391
|
-
"mode": "filegen",
|
|
392
|
-
"skipped": True,
|
|
393
|
-
"reason": (
|
|
394
|
-
"no local gemma/qwen/llama models installed — install one via "
|
|
395
|
-
"the app's model picker, then re-run"
|
|
396
|
-
),
|
|
397
|
-
"models": [],
|
|
398
|
-
}
|
|
399
|
-
try:
|
|
400
|
-
results = asyncio.run(_filegen_run_models(models))
|
|
401
|
-
except Exception as exc: # noqa: BLE001 - fail-open at the run level too
|
|
402
|
-
return {
|
|
403
|
-
"mode": "filegen", "skipped": True,
|
|
404
|
-
"reason": f"benchmark run failed: {exc}", "models": [],
|
|
405
|
-
}
|
|
406
|
-
return {"mode": "filegen", "skipped": False, "models": results}
|
|
407
|
-
|
|
408
|
-
|
|
409
|
-
def format_filegen_report(report: Dict[str, Any]) -> str:
|
|
410
|
-
lines = [
|
|
411
|
-
"Weekly filegen benchmark (real pipeline: generate_file_content)",
|
|
412
|
-
"=" * 72,
|
|
413
|
-
]
|
|
414
|
-
if report.get("skipped"):
|
|
415
|
-
lines.append(f"SKIPPED (fail-open): {report.get('reason')}")
|
|
416
|
-
lines.append("=" * 72)
|
|
417
|
-
return "\n".join(lines)
|
|
418
|
-
for model in report.get("models", []):
|
|
419
|
-
if model.get("skipped"):
|
|
420
|
-
lines.append(f" {model['model']}: SKIPPED — {model.get('reason')}")
|
|
421
|
-
continue
|
|
422
|
-
lines.append(
|
|
423
|
-
f" {model['model']} "
|
|
424
|
-
f"(success={model['success_rate']}, clean={model['clean_rate']})"
|
|
425
|
-
)
|
|
426
|
-
for row in model.get("targets", []):
|
|
427
|
-
status = "ok" if row["valid"] else "FAIL"
|
|
428
|
-
if row["valid"] and row["repaired"]:
|
|
429
|
-
status = "ok(repaired)"
|
|
430
|
-
elif row["valid"] and not row["clean_first_try"]:
|
|
431
|
-
status = "ok(retry)"
|
|
432
|
-
lines.append(
|
|
433
|
-
f" {row['type']:<5} {status:<12} "
|
|
434
|
-
f"attempts={row['attempts']} latency_ms={row['latency_ms']} "
|
|
435
|
-
f"bytes={row['bytes']}"
|
|
436
|
-
)
|
|
437
|
-
lines.append("=" * 72)
|
|
438
|
-
lines.append("note: success = valid after sanitize/repair; clean = valid without repair fallback")
|
|
439
|
-
return "\n".join(lines)
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
# ── Reporting ────────────────────────────────────────────────────────────
|
|
443
|
-
def _fmt_row(name: str, r: Dict[str, Any]) -> str:
|
|
444
|
-
return (
|
|
445
|
-
f" {name:<40} "
|
|
446
|
-
f"success={r['success_rate']:<7} "
|
|
447
|
-
f"repair={r['repair_rate']:<7} "
|
|
448
|
-
f"n={r.get('total', '-'):<4} "
|
|
449
|
-
f"latency_ms={r['latency_ms_mean']}"
|
|
450
|
-
)
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
def main(argv: List[str] | None = None) -> int:
|
|
454
|
-
parser = argparse.ArgumentParser(description="Model robustness benchmark harness")
|
|
455
|
-
parser.add_argument("--json", dest="json_out", help="write full report JSON to this path")
|
|
456
|
-
parser.add_argument("--live-endpoint", help="OpenAI-compatible base URL, e.g. http://127.0.0.1:1234/v1")
|
|
457
|
-
parser.add_argument("--model", help="model id for the live endpoint")
|
|
458
|
-
parser.add_argument("--timeout", type=float, default=30.0, help="live request timeout seconds")
|
|
459
|
-
parser.add_argument(
|
|
460
|
-
"--filegen", action="store_true",
|
|
461
|
-
help="weekly multi-model file-generation report (fail-open, exit 0 even with no models)",
|
|
462
|
-
)
|
|
463
|
-
args = parser.parse_args(argv)
|
|
464
|
-
|
|
465
|
-
if args.filegen:
|
|
466
|
-
filegen_report = run_filegen_benchmark()
|
|
467
|
-
print(format_filegen_report(filegen_report))
|
|
468
|
-
if args.json_out:
|
|
469
|
-
Path(args.json_out).write_text(
|
|
470
|
-
json.dumps(filegen_report, ensure_ascii=False, indent=2), encoding="utf-8"
|
|
471
|
-
)
|
|
472
|
-
print(f"wrote {args.json_out}")
|
|
473
|
-
# FAIL-OPEN: this is a scheduled report, never a CI gate.
|
|
474
|
-
return 0
|
|
475
|
-
|
|
476
|
-
report: Dict[str, Any] = {"mode": "scripted", "profiles": {}}
|
|
477
|
-
|
|
478
|
-
# Scripted matrix (always).
|
|
479
|
-
for name, corpus in _PROFILES.items():
|
|
480
|
-
report["profiles"][name] = _bench_corpus(corpus)
|
|
481
|
-
|
|
482
|
-
# Real agent-loop reference row.
|
|
483
|
-
loop = run_agent_eval()
|
|
484
|
-
report["agent_loop_reference"] = {
|
|
485
|
-
"scenarios": loop["scenarios"],
|
|
486
|
-
"passed": loop["passed"],
|
|
487
|
-
"success_rate": loop["success_rate"],
|
|
488
|
-
"recovery_rate": loop["recovery_rate"],
|
|
489
|
-
"parse_errors": loop["parse_errors"],
|
|
490
|
-
"parse_recovered": loop["parse_recovered"],
|
|
491
|
-
}
|
|
492
|
-
|
|
493
|
-
# Optional live row.
|
|
494
|
-
if args.live_endpoint:
|
|
495
|
-
if not args.model:
|
|
496
|
-
print("--live-endpoint requires --model", file=sys.stderr)
|
|
497
|
-
return 2
|
|
498
|
-
live = _bench_live(args.live_endpoint, args.model, timeout=args.timeout)
|
|
499
|
-
report["mode"] = "scripted+live"
|
|
500
|
-
report["live"] = {"endpoint": args.live_endpoint, "model": args.model, **live}
|
|
501
|
-
|
|
502
|
-
# Human-readable matrix.
|
|
503
|
-
print("Model robustness benchmark (real parser: extract_action_details)")
|
|
504
|
-
print("=" * 72)
|
|
505
|
-
print("Scripted tiers (latency_ms = parse+repair only, NOT model inference):")
|
|
506
|
-
mismatch_total = 0
|
|
507
|
-
for name, r in report["profiles"].items():
|
|
508
|
-
print(_fmt_row(name, r))
|
|
509
|
-
if r["repair_kinds"]:
|
|
510
|
-
print(f" repairs used: {r['repair_kinds']}")
|
|
511
|
-
mismatch_total += len(r["harness_mismatches"])
|
|
512
|
-
ref = report["agent_loop_reference"]
|
|
513
|
-
print("-" * 72)
|
|
514
|
-
print(
|
|
515
|
-
f" agent-loop (real state machine) "
|
|
516
|
-
f"success={ref['success_rate']:<7} "
|
|
517
|
-
f"recovery={ref['recovery_rate']:<7} "
|
|
518
|
-
f"scenarios={ref['scenarios']}"
|
|
519
|
-
)
|
|
520
|
-
if "live" in report:
|
|
521
|
-
lv = report["live"]
|
|
522
|
-
print("-" * 72)
|
|
523
|
-
if "error" in lv:
|
|
524
|
-
print(f" live [{lv['model']}] SKIPPED: {lv['error']}")
|
|
525
|
-
else:
|
|
526
|
-
print(_fmt_row(f"live [{lv['model']}] (real inference)", lv))
|
|
527
|
-
print("=" * 72)
|
|
528
|
-
print(f"harness self-check: {mismatch_total} corpus/parser mismatches (0 = healthy)")
|
|
529
|
-
|
|
530
|
-
if args.json_out:
|
|
531
|
-
Path(args.json_out).write_text(json.dumps(report, indent=2), encoding="utf-8")
|
|
532
|
-
print(f"wrote {args.json_out}")
|
|
533
|
-
|
|
534
|
-
# Non-zero exit only if the harness itself is inconsistent (a real bug),
|
|
535
|
-
# never because a weak tier scored low — low scores are the finding.
|
|
536
|
-
return 1 if mismatch_total else 0
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
if __name__ == "__main__":
|
|
540
|
-
raise SystemExit(main())
|