ltcai 11.5.2 → 11.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (443) hide show
  1. package/README.md +93 -148
  2. package/bin/ltcai.js +234 -24
  3. package/docs/CHANGELOG.md +119 -0
  4. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  5. package/docs/DEVELOPMENT.md +8 -6
  6. package/docs/ENTERPRISE.md +2 -1
  7. package/docs/MULTI_AGENT_RUNTIME.md +12 -5
  8. package/docs/ONBOARDING.md +1 -1
  9. package/docs/OPERATIONS.md +6 -3
  10. package/docs/REALTIME_COLLABORATION.md +1 -1
  11. package/docs/TRUST_MODEL.md +1 -1
  12. package/docs/WHY_LATTICE.md +1 -1
  13. package/docs/WORKFLOW_DESIGNER.md +3 -2
  14. package/docs/kg-schema.md +1 -1
  15. package/docs/v11.6.0_ONE_DOOR_PLAN.md +170 -0
  16. package/lattice_brain/__init__.py +42 -79
  17. package/lattice_brain/graph/__init__.py +9 -24
  18. package/lattice_brain/graph/_kg_common/__init__.py +13 -24
  19. package/lattice_brain/ingestion/__init__.py +19 -58
  20. package/lattice_brain/ingestion/pipeline.py +20 -398
  21. package/lattice_brain/multimodal/__init__.py +7 -18
  22. package/lattice_brain/multimodal/images.py +6 -264
  23. package/lattice_brain/multimodal/video.py +8 -247
  24. package/lattice_brain/runtime/__init__.py +8 -79
  25. package/lattice_brain/runtime/hooks.py +22 -584
  26. package/latticeai/__init__.py +1 -1
  27. package/latticeai/api/agent_worker_seam.py +7 -80
  28. package/latticeai/api/health.py +6 -20
  29. package/latticeai/api/local_files.py +16 -613
  30. package/latticeai/api/models.py +12 -87
  31. package/latticeai/api/search.py +17 -255
  32. package/latticeai/api/tools.py +69 -750
  33. package/latticeai/api/voice_capture.py +8 -70
  34. package/latticeai/api/worker_compute.py +842 -0
  35. package/latticeai/api/worker_seams.py +216 -0
  36. package/latticeai/app_factory.py +23 -232
  37. package/latticeai/cli/entrypoint.py +36 -249
  38. package/latticeai/core/agent_permission.py +16 -91
  39. package/latticeai/core/messages.py +89 -531
  40. package/latticeai/runtime/access_runtime.py +0 -14
  41. package/latticeai/runtime/bootstrap.py +10 -21
  42. package/latticeai/runtime/brain_runtime.py +35 -43
  43. package/latticeai/runtime/build_phases/__init__.py +21 -37
  44. package/latticeai/runtime/build_phases/features.py +99 -389
  45. package/latticeai/runtime/build_phases/foundation.py +82 -392
  46. package/latticeai/runtime/build_phases/web.py +91 -408
  47. package/latticeai/runtime/build_phases/worker_profile.py +244 -0
  48. package/latticeai/runtime/lifespan_runtime.py +7 -16
  49. package/latticeai/runtime/platform_services_runtime.py +1 -15
  50. package/latticeai/runtime/runtime_context.py +13 -130
  51. package/latticeai/runtime/security_runtime.py +10 -103
  52. package/latticeai/services/architecture_readiness.py +82 -43
  53. package/latticeai/services/model_runtime/__init__.py +2 -11
  54. package/latticeai/services/model_runtime/service.py +4 -28
  55. package/latticeai/services/p_reinforce.py +10 -261
  56. package/latticeai/services/product_readiness.py +31 -38
  57. package/latticeai/services/search_service.py +37 -795
  58. package/latticeai/services/tool_dispatch.py +30 -386
  59. package/latticeai/services/voice_capture.py +13 -107
  60. package/latticeai/tools/__init__.py +22 -49
  61. package/latticeai/tools/commands.py +0 -163
  62. package/latticeai/tools/computer.py +0 -39
  63. package/latticeai/tools/documents.py +1 -134
  64. package/latticeai/tools/filesystem.py +1 -247
  65. package/latticeai/tools/knowledge.py +1 -52
  66. package/latticeai/tools/local_files.py +0 -20
  67. package/latticeai/worker_app.py +75 -0
  68. package/package.json +2 -3
  69. package/requirements.txt +0 -5
  70. package/scripts/agent_eval.py +16 -28
  71. package/scripts/brain_quality_eval.py +20 -183
  72. package/scripts/bump_version.py +5 -5
  73. package/scripts/check_current_release_docs.mjs +7 -4
  74. package/scripts/check_openapi_drift.mjs +10 -0
  75. package/scripts/check_server_i18n.mjs +2 -18
  76. package/scripts/compose_openapi.py +377 -0
  77. package/scripts/export_openapi.py +32 -4
  78. package/scripts/gen_messages_catalog_fixture.py +302 -0
  79. package/scripts/gen_openapi_fragments.py +365 -0
  80. package/scripts/gen_redact_fixture.py +196 -0
  81. package/scripts/gen_worker_allowlist_fixture.py +115 -0
  82. package/scripts/generate_agent_parity_fixtures.py +33 -14
  83. package/scripts/openapi_route_families.json +2161 -0
  84. package/scripts/release_screen_claims.json +54 -0
  85. package/scripts/run_integration_tests.mjs +134 -29
  86. package/scripts/run_sidecar_e2e.mjs +104 -11
  87. package/scripts/wheel_smoke.py +32 -22
  88. package/src-tauri/Cargo.lock +375 -9
  89. package/src-tauri/Cargo.toml +1 -1
  90. package/src-tauri/src/backend.rs +151 -45
  91. package/src-tauri/src/main.rs +18 -18
  92. package/src-tauri/src/topology.rs +58 -88
  93. package/src-tauri/tauri.conf.json +1 -2
  94. package/static/app/asset-manifest.json +41 -41
  95. package/static/app/assets/{Act-DcQizkl1.js → Act-BPcVAbOL.js} +1 -1
  96. package/static/app/assets/AdminConsole-Bw1ATQL0.js +1 -0
  97. package/static/app/assets/{Brain-3VCSHFcn.js → Brain-CT92Kos0.js} +2 -2
  98. package/static/app/assets/{BrainHome-Qm8eaztx.js → BrainHome-CFBkt1K_.js} +1 -1
  99. package/static/app/assets/{BrainSignals-DS9BtKOW.js → BrainSignals-ReLWF2H8.js} +1 -1
  100. package/static/app/assets/Capture-BsTokYkk.js +1 -0
  101. package/static/app/assets/{Chronicle-BGvuAchH.js → Chronicle-B6f0T9id.js} +1 -1
  102. package/static/app/assets/{CommandPalette-Bqhm0Urn.js → CommandPalette-CuvjTv1u.js} +1 -1
  103. package/static/app/assets/{Library-BV6NnF0a.js → Library-BGJbG9Hd.js} +1 -1
  104. package/static/app/assets/{LivingBrain-GzenJchP.js → LivingBrain-DGYK_Jsa.js} +1 -1
  105. package/static/app/assets/{ProductFlow-DEP6-vML.js → ProductFlow-DXBC6brE.js} +1 -1
  106. package/static/app/assets/{ReviewCard-CNZ7XjWG.js → ReviewCard-HXRle3qq.js} +2 -2
  107. package/static/app/assets/System-CMHSO9qM.js +1 -0
  108. package/static/app/assets/arrow-left-BfmkskWx.js +1 -0
  109. package/static/app/assets/{bot-B_K1Tdmw.js → bot-Cn8bWRuq.js} +1 -1
  110. package/static/app/assets/{brain-DWyaV1L1.js → brain-CQJberbE.js} +1 -1
  111. package/static/app/assets/{button-aTn4s84A.js → button-Ct9f2_oT.js} +1 -1
  112. package/static/app/assets/circle-check-DruOxB-4.js +1 -0
  113. package/static/app/assets/{circle-pause-xKgeGXkT.js → circle-pause-CmzC_apg.js} +1 -1
  114. package/static/app/assets/{circle-play-DkT6tYPX.js → circle-play-D8mW2aQ7.js} +1 -1
  115. package/static/app/assets/{cpu-85xYObUC.js → cpu-DZcdd0PZ.js} +1 -1
  116. package/static/app/assets/{download-B5Fm7YXo.js → download-bv1KEPGQ.js} +1 -1
  117. package/static/app/assets/{folder-open-kk2Xa52u.js → folder-open-d-Pip5gr.js} +1 -1
  118. package/static/app/assets/{hard-drive-DkA3zBW_.js → hard-drive-D20iavUb.js} +1 -1
  119. package/static/app/assets/index-D9x-kSNy.css +2 -0
  120. package/static/app/assets/{index-BMPdTmlY.js → index-Do83hDzJ.js} +3 -3
  121. package/static/app/assets/{input-B0nRf2jO.js → input-BLXVNmj1.js} +1 -1
  122. package/static/app/assets/{link-2-Dwb4gnTc.js → link-2-BPJOFlAy.js} +1 -1
  123. package/static/app/assets/{permissionCopy-CQDUBrOZ.js → permissionCopy-ChdJd493.js} +1 -1
  124. package/static/app/assets/primitives-Cv5tbZBY.js +1 -0
  125. package/static/app/assets/search-CT9aho2j.js +1 -0
  126. package/static/app/assets/{share-2-BsrxFglO.js → share-2-YNX_NtMU.js} +1 -1
  127. package/static/app/assets/{shield-alert-5BStfp2_.js → shield-alert-DuQ3zrVL.js} +1 -1
  128. package/static/app/assets/{textarea-Cg8IUA-k.js → textarea-DqwLnli4.js} +1 -1
  129. package/static/app/assets/{useFocusTrap-CYKvE46M.js → useFocusTrap-ZVI98jaW.js} +1 -1
  130. package/static/app/assets/{useMutation-CSn9t1op.js → useMutation-CVC4qv_D.js} +1 -1
  131. package/static/app/assets/{useQuery-CY2OI2uy.js → useQuery-C7BeG4HU.js} +1 -1
  132. package/static/app/assets/{utils-Ddol2RWD.js → utils-CiFtIdZq.js} +1 -1
  133. package/static/app/assets/{workspace-BqDwOz_p.js → workspace-DQz9vIId.js} +1 -1
  134. package/static/app/index.html +4 -4
  135. package/static/sw.js +1 -1
  136. package/lattice_brain/archive.py +0 -522
  137. package/lattice_brain/context.py +0 -325
  138. package/lattice_brain/conversations.py +0 -382
  139. package/lattice_brain/core.py +0 -82
  140. package/lattice_brain/graph/_kg_contract.py +0 -249
  141. package/lattice_brain/graph/curator.py +0 -676
  142. package/lattice_brain/graph/discovery.py +0 -595
  143. package/lattice_brain/graph/discovery_index/__init__.py +0 -35
  144. package/lattice_brain/graph/discovery_index/cleanup.py +0 -182
  145. package/lattice_brain/graph/discovery_index/extract.py +0 -137
  146. package/lattice_brain/graph/discovery_index/scan.py +0 -411
  147. package/lattice_brain/graph/discovery_index/upsert.py +0 -495
  148. package/lattice_brain/graph/documents.py +0 -380
  149. package/lattice_brain/graph/fusion.py +0 -395
  150. package/lattice_brain/graph/identity.py +0 -175
  151. package/lattice_brain/graph/image_vectors.py +0 -230
  152. package/lattice_brain/graph/ingest.py +0 -829
  153. package/lattice_brain/graph/network.py +0 -205
  154. package/lattice_brain/graph/proactive.py +0 -724
  155. package/lattice_brain/graph/projection/__init__.py +0 -42
  156. package/lattice_brain/graph/projection/curation.py +0 -500
  157. package/lattice_brain/graph/projection/v2_schema.py +0 -518
  158. package/lattice_brain/graph/provenance.py +0 -524
  159. package/lattice_brain/graph/rerank.py +0 -163
  160. package/lattice_brain/graph/retrieval/__init__.py +0 -54
  161. package/lattice_brain/graph/retrieval/context.py +0 -197
  162. package/lattice_brain/graph/retrieval/graph_view.py +0 -319
  163. package/lattice_brain/graph/retrieval/hybrid.py +0 -488
  164. package/lattice_brain/graph/retrieval/maintenance.py +0 -121
  165. package/lattice_brain/graph/retrieval/signals.py +0 -95
  166. package/lattice_brain/graph/retrieval_docgen.py +0 -253
  167. package/lattice_brain/graph/retrieval_policy.py +0 -180
  168. package/lattice_brain/graph/retrieval_reads.py +0 -769
  169. package/lattice_brain/graph/retrieval_vector/__init__.py +0 -42
  170. package/lattice_brain/graph/retrieval_vector/fingerprint.py +0 -97
  171. package/lattice_brain/graph/retrieval_vector/indexing.py +0 -347
  172. package/lattice_brain/graph/retrieval_vector/search.py +0 -560
  173. package/lattice_brain/graph/retrieval_vector/status.py +0 -374
  174. package/lattice_brain/graph/schema.py +0 -792
  175. package/lattice_brain/graph/store.py +0 -268
  176. package/lattice_brain/graph/vector_index/__init__.py +0 -85
  177. package/lattice_brain/graph/vector_index/base.py +0 -167
  178. package/lattice_brain/graph/vector_index/brute_force.py +0 -110
  179. package/lattice_brain/graph/vector_index/hnsw.py +0 -290
  180. package/lattice_brain/graph/vector_index/jobs.py +0 -287
  181. package/lattice_brain/graph/vector_index/quantized.py +0 -148
  182. package/lattice_brain/graph/vector_index/selector.py +0 -161
  183. package/lattice_brain/graph/write_master.py +0 -308
  184. package/lattice_brain/ingestion/_contract.py +0 -90
  185. package/lattice_brain/ingestion/folder_scan.py +0 -57
  186. package/lattice_brain/ingestion/folders.py +0 -258
  187. package/lattice_brain/ingestion/jobs_api.py +0 -107
  188. package/lattice_brain/ingestion/routing.py +0 -295
  189. package/lattice_brain/ingestion_jobs.py +0 -380
  190. package/lattice_brain/memory.py +0 -75
  191. package/lattice_brain/portability/__init__.py +0 -90
  192. package/lattice_brain/portability/_contract.py +0 -42
  193. package/lattice_brain/portability/backups.py +0 -338
  194. package/lattice_brain/portability/bundles.py +0 -136
  195. package/lattice_brain/portability/constants.py +0 -93
  196. package/lattice_brain/portability/fsops.py +0 -138
  197. package/lattice_brain/portability/service.py +0 -41
  198. package/lattice_brain/portability/sharing.py +0 -710
  199. package/lattice_brain/quality.py +0 -543
  200. package/lattice_brain/retrieval_benchmark_fixtures.py +0 -92
  201. package/lattice_brain/runtime/agent_runtime.py +0 -859
  202. package/lattice_brain/runtime/contracts.py +0 -460
  203. package/lattice_brain/runtime/multi_agent.py +0 -942
  204. package/lattice_brain/runtime/statuses.py +0 -10
  205. package/lattice_brain/sealed_box.py +0 -240
  206. package/lattice_brain/self_model.py +0 -675
  207. package/lattice_brain/sensitivity.py +0 -94
  208. package/lattice_brain/storage/__init__.py +0 -22
  209. package/lattice_brain/storage/base.py +0 -100
  210. package/lattice_brain/storage/docker.py +0 -105
  211. package/lattice_brain/storage/factory.py +0 -31
  212. package/lattice_brain/storage/migration.py +0 -191
  213. package/lattice_brain/storage/postgres.py +0 -123
  214. package/lattice_brain/storage/sqlite.py +0 -143
  215. package/lattice_brain/synthesis.py +0 -824
  216. package/lattice_brain/workflow.py +0 -497
  217. package/latticeai/api/admin.py +0 -471
  218. package/latticeai/api/agent_registry.py +0 -105
  219. package/latticeai/api/agents.py +0 -228
  220. package/latticeai/api/auth.py +0 -383
  221. package/latticeai/api/automation_intelligence.py +0 -401
  222. package/latticeai/api/brain_intelligence.py +0 -199
  223. package/latticeai/api/browser.py +0 -493
  224. package/latticeai/api/change_proposals.py +0 -89
  225. package/latticeai/api/chat.py +0 -572
  226. package/latticeai/api/chat_agent_http.py +0 -892
  227. package/latticeai/api/chat_contracts.py +0 -73
  228. package/latticeai/api/chat_documents.py +0 -276
  229. package/latticeai/api/chat_helpers.py +0 -460
  230. package/latticeai/api/chat_history.py +0 -101
  231. package/latticeai/api/chat_hybrid.py +0 -113
  232. package/latticeai/api/chat_intents.py +0 -646
  233. package/latticeai/api/chat_stream.py +0 -216
  234. package/latticeai/api/chronicle.py +0 -63
  235. package/latticeai/api/command_center.py +0 -51
  236. package/latticeai/api/computer_use.py +0 -474
  237. package/latticeai/api/evidence_actions.py +0 -48
  238. package/latticeai/api/features.py +0 -70
  239. package/latticeai/api/funnel_metrics.py +0 -31
  240. package/latticeai/api/garden.py +0 -34
  241. package/latticeai/api/hooks.py +0 -165
  242. package/latticeai/api/index_jobs.py +0 -145
  243. package/latticeai/api/invitations.py +0 -100
  244. package/latticeai/api/knowledge_graph.py +0 -536
  245. package/latticeai/api/marketplace.py +0 -105
  246. package/latticeai/api/mcp.py +0 -482
  247. package/latticeai/api/memory.py +0 -270
  248. package/latticeai/api/network.py +0 -81
  249. package/latticeai/api/network_boundary.py +0 -225
  250. package/latticeai/api/permission_mode.py +0 -61
  251. package/latticeai/api/permissions.py +0 -436
  252. package/latticeai/api/plugins.py +0 -126
  253. package/latticeai/api/portability.py +0 -391
  254. package/latticeai/api/project_sessions.py +0 -114
  255. package/latticeai/api/realtime.py +0 -118
  256. package/latticeai/api/review_queue.py +0 -364
  257. package/latticeai/api/security_dashboard.py +0 -604
  258. package/latticeai/api/setup.py +0 -319
  259. package/latticeai/api/static_routes.py +0 -354
  260. package/latticeai/api/ui_redirects.py +0 -26
  261. package/latticeai/api/workflow_designer.py +0 -394
  262. package/latticeai/api/workspace.py +0 -856
  263. package/latticeai/api/workspace_scope.py +0 -125
  264. package/latticeai/core/agent/__init__.py +0 -93
  265. package/latticeai/core/agent/_contract.py +0 -79
  266. package/latticeai/core/agent/context.py +0 -57
  267. package/latticeai/core/agent/deps.py +0 -125
  268. package/latticeai/core/agent/execution.py +0 -622
  269. package/latticeai/core/agent/planning.py +0 -145
  270. package/latticeai/core/agent/recovery.py +0 -157
  271. package/latticeai/core/agent/runtime.py +0 -210
  272. package/latticeai/core/agent/verification.py +0 -231
  273. package/latticeai/core/agent_eval.py +0 -739
  274. package/latticeai/core/agent_helpers.py +0 -493
  275. package/latticeai/core/agent_profiles.py +0 -110
  276. package/latticeai/core/agent_prompts.py +0 -171
  277. package/latticeai/core/agent_registry.py +0 -232
  278. package/latticeai/core/agent_state.py +0 -41
  279. package/latticeai/core/agent_trace.py +0 -104
  280. package/latticeai/core/artifact_ledger.py +0 -109
  281. package/latticeai/core/audit.py +0 -260
  282. package/latticeai/core/builtin_hooks.py +0 -105
  283. package/latticeai/core/context_builder.py +0 -394
  284. package/latticeai/core/document_generator.py +0 -103
  285. package/latticeai/core/enterprise.py +0 -154
  286. package/latticeai/core/enterprise_admin.py +0 -158
  287. package/latticeai/core/file_generation/__init__.py +0 -115
  288. package/latticeai/core/file_generation/bundles.py +0 -76
  289. package/latticeai/core/file_generation/extraction.py +0 -154
  290. package/latticeai/core/file_generation/inference.py +0 -235
  291. package/latticeai/core/file_generation/orchestration.py +0 -152
  292. package/latticeai/core/file_generation/prompting.py +0 -117
  293. package/latticeai/core/file_generation/repair.py +0 -114
  294. package/latticeai/core/file_generation/sanitize.py +0 -61
  295. package/latticeai/core/file_generation/validation.py +0 -201
  296. package/latticeai/core/invitations.py +0 -132
  297. package/latticeai/core/legacy_compatibility.py +0 -243
  298. package/latticeai/core/logging_safety.py +0 -46
  299. package/latticeai/core/marketplace.py +0 -293
  300. package/latticeai/core/mcp_catalog.py +0 -452
  301. package/latticeai/core/mcp_registry.py +0 -506
  302. package/latticeai/core/network_boundary.py +0 -168
  303. package/latticeai/core/oidc.py +0 -208
  304. package/latticeai/core/plugins.py +0 -432
  305. package/latticeai/core/product_hardening.py +0 -218
  306. package/latticeai/core/project_sessions.py +0 -337
  307. package/latticeai/core/realtime.py +0 -238
  308. package/latticeai/core/run_explain.py +0 -426
  309. package/latticeai/core/run_store.py +0 -252
  310. package/latticeai/core/timezones.py +0 -80
  311. package/latticeai/core/workspace_computer_memory.py +0 -84
  312. package/latticeai/core/workspace_graph_trace.py +0 -155
  313. package/latticeai/core/workspace_indexing.py +0 -102
  314. package/latticeai/core/workspace_memory.py +0 -77
  315. package/latticeai/core/workspace_onboarding.py +0 -104
  316. package/latticeai/core/workspace_os.py +0 -978
  317. package/latticeai/core/workspace_os_constants.py +0 -126
  318. package/latticeai/core/workspace_os_state.py +0 -180
  319. package/latticeai/core/workspace_os_utils.py +0 -103
  320. package/latticeai/core/workspace_permissions.py +0 -101
  321. package/latticeai/core/workspace_plugins.py +0 -97
  322. package/latticeai/core/workspace_relationships.py +0 -99
  323. package/latticeai/core/workspace_reorganization.py +0 -335
  324. package/latticeai/core/workspace_review_items.py +0 -112
  325. package/latticeai/core/workspace_runs.py +0 -726
  326. package/latticeai/core/workspace_skills.py +0 -109
  327. package/latticeai/core/workspace_snapshots.py +0 -198
  328. package/latticeai/core/workspace_timeline.py +0 -110
  329. package/latticeai/integrations/__init__.py +0 -0
  330. package/latticeai/integrations/telegram_bot/__init__.py +0 -123
  331. package/latticeai/integrations/telegram_bot/__main__.py +0 -17
  332. package/latticeai/integrations/telegram_bot/config.py +0 -86
  333. package/latticeai/integrations/telegram_bot/dispatch.py +0 -311
  334. package/latticeai/integrations/telegram_bot/flows.py +0 -478
  335. package/latticeai/integrations/telegram_bot/helpers.py +0 -322
  336. package/latticeai/integrations/telegram_bot/screens.py +0 -394
  337. package/latticeai/runtime/audit_runtime.py +0 -76
  338. package/latticeai/runtime/automation_runtime.py +0 -81
  339. package/latticeai/runtime/chat_wiring.py +0 -141
  340. package/latticeai/runtime/context_runtime.py +0 -66
  341. package/latticeai/runtime/feature_toggle_wiring.py +0 -157
  342. package/latticeai/runtime/history_runtime.py +0 -163
  343. package/latticeai/runtime/history_writer.py +0 -138
  344. package/latticeai/runtime/hooks_runtime.py +0 -77
  345. package/latticeai/runtime/model_wiring.py +0 -68
  346. package/latticeai/runtime/namespace_runtime.py +0 -163
  347. package/latticeai/runtime/network_boundary_wiring.py +0 -117
  348. package/latticeai/runtime/network_config_runtime.py +0 -56
  349. package/latticeai/runtime/permission_mode_wiring.py +0 -112
  350. package/latticeai/runtime/persistence_runtime.py +0 -159
  351. package/latticeai/runtime/platform_runtime_wiring.py +0 -89
  352. package/latticeai/runtime/review_wiring.py +0 -42
  353. package/latticeai/runtime/router_registration.py +0 -693
  354. package/latticeai/runtime/service_singletons.py +0 -55
  355. package/latticeai/runtime/sso_config_runtime.py +0 -128
  356. package/latticeai/runtime/user_key_runtime.py +0 -106
  357. package/latticeai/runtime/web_runtime.py +0 -92
  358. package/latticeai/server_app.py +0 -51
  359. package/latticeai/services/app_context.py +0 -130
  360. package/latticeai/services/automation_execution.py +0 -266
  361. package/latticeai/services/automation_intelligence.py +0 -614
  362. package/latticeai/services/brain_automation.py +0 -191
  363. package/latticeai/services/brain_intelligence/__init__.py +0 -58
  364. package/latticeai/services/brain_intelligence/_contract.py +0 -71
  365. package/latticeai/services/brain_intelligence/consistency.py +0 -193
  366. package/latticeai/services/brain_intelligence/constants.py +0 -47
  367. package/latticeai/services/brain_intelligence/digest.py +0 -258
  368. package/latticeai/services/brain_intelligence/health.py +0 -331
  369. package/latticeai/services/brain_intelligence/proposals.py +0 -259
  370. package/latticeai/services/brain_intelligence/sampling.py +0 -84
  371. package/latticeai/services/brain_intelligence/service.py +0 -48
  372. package/latticeai/services/change_proposals.py +0 -471
  373. package/latticeai/services/chat_service.py +0 -243
  374. package/latticeai/services/chronicle.py +0 -555
  375. package/latticeai/services/cloud_egress_audit.py +0 -85
  376. package/latticeai/services/cloud_extraction.py +0 -129
  377. package/latticeai/services/cloud_streaming.py +0 -268
  378. package/latticeai/services/cloud_token_guard.py +0 -84
  379. package/latticeai/services/command_center.py +0 -548
  380. package/latticeai/services/evidence_actions.py +0 -258
  381. package/latticeai/services/feature_toggles.py +0 -502
  382. package/latticeai/services/folder_watch.py +0 -520
  383. package/latticeai/services/funnel_metrics.py +0 -307
  384. package/latticeai/services/hybrid_chat.py +0 -316
  385. package/latticeai/services/hybrid_context.py +0 -228
  386. package/latticeai/services/hybrid_policy.py +0 -129
  387. package/latticeai/services/interop_bridges.py +0 -978
  388. package/latticeai/services/local_knowledge.py +0 -465
  389. package/latticeai/services/memory_service/__init__.py +0 -52
  390. package/latticeai/services/memory_service/_contract.py +0 -100
  391. package/latticeai/services/memory_service/brief.py +0 -431
  392. package/latticeai/services/memory_service/constants.py +0 -57
  393. package/latticeai/services/memory_service/maintenance.py +0 -138
  394. package/latticeai/services/memory_service/manager.py +0 -186
  395. package/latticeai/services/memory_service/proof.py +0 -136
  396. package/latticeai/services/memory_service/recall.py +0 -225
  397. package/latticeai/services/memory_service/service.py +0 -48
  398. package/latticeai/services/memory_service/stores.py +0 -110
  399. package/latticeai/services/mode_store.py +0 -132
  400. package/latticeai/services/model_recommendation.py +0 -224
  401. package/latticeai/services/model_runtime/cloud.py +0 -87
  402. package/latticeai/services/network_boundary_service.py +0 -117
  403. package/latticeai/services/obsidian_bridge.py +0 -609
  404. package/latticeai/services/openai_compatible_adapter.py +0 -101
  405. package/latticeai/services/permission_mode_service.py +0 -122
  406. package/latticeai/services/platform_runtime.py +0 -366
  407. package/latticeai/services/review_queue.py +0 -380
  408. package/latticeai/services/router_context.py +0 -59
  409. package/latticeai/services/run_executor.py +0 -387
  410. package/latticeai/services/self_model_service.py +0 -171
  411. package/latticeai/services/setup_detection.py +0 -147
  412. package/latticeai/services/triggers.py +0 -378
  413. package/latticeai/services/upload_service.py +0 -172
  414. package/latticeai/services/workspace_service.py +0 -165
  415. package/latticeai/setup/__init__.py +0 -25
  416. package/latticeai/setup/auto_setup.py +0 -846
  417. package/latticeai/setup/demo_corpus.py +0 -98
  418. package/latticeai/setup/wizard/__init__.py +0 -126
  419. package/latticeai/setup/wizard/catalog.py +0 -175
  420. package/latticeai/setup/wizard/detect.py +0 -302
  421. package/latticeai/setup/wizard/install.py +0 -348
  422. package/latticeai/setup/wizard/paths.py +0 -165
  423. package/latticeai/setup/wizard/plans.py +0 -74
  424. package/latticeai/setup/wizard/recommend.py +0 -320
  425. package/scripts/bench_agent_smoke.py +0 -409
  426. package/scripts/bench_models.py +0 -540
  427. package/scripts/bench_vector_index.py +0 -295
  428. package/scripts/funnel_soft_gate.py +0 -192
  429. package/scripts/generate_agent_loop_fixtures.py +0 -994
  430. package/scripts/generate_rust_parity_fixtures.py +0 -908
  431. package/scripts/migrate_brain_storage.py +0 -57
  432. package/scripts/parity_fixture_corpus_context.py +0 -162
  433. package/scripts/parity_fixture_corpus_docgen.py +0 -341
  434. package/scripts/profile_kg.py +0 -355
  435. package/server.py +0 -30
  436. package/static/app/assets/AdminConsole-cf4npybT.js +0 -1
  437. package/static/app/assets/Capture-DiQ219jW.js +0 -1
  438. package/static/app/assets/System-CieofHQa.js +0 -1
  439. package/static/app/assets/arrow-left-kfsrk0mv.js +0 -1
  440. package/static/app/assets/circle-check-qqLug9nU.js +0 -1
  441. package/static/app/assets/index-DxmOfNRi.css +0 -2
  442. package/static/app/assets/primitives-SNp0LRJz.js +0 -1
  443. package/static/app/assets/search-BcHqkjoy.js +0 -1
@@ -1,540 +0,0 @@
1
- #!/usr/bin/env python3
2
- """Model robustness benchmark harness for the Lattice agent loop.
3
-
4
- What it measures
5
- ================
6
- The product claim is that the Brain stays durable "across any AI model" — a
7
- weaker model may emit sloppier output, but the agent loop repairs it and still
8
- completes the task. This harness turns that claim into a **matrix** of three
9
- numbers per model tier:
10
-
11
- * ``success_rate`` — fraction of model outputs the loop could turn into a valid
12
- action (i.e. the task can proceed).
13
- * ``repair_rate`` — of the successful parses, the fraction that ONLY succeeded
14
- because the loop's tolerant parser had to repair the output (fences, prose,
15
- trailing commas, ``<think>`` blocks, Python-dict literals). High repair_rate
16
- = "this tier leans hard on the loop's robustness".
17
- * ``latency_ms`` — see the honesty note below.
18
-
19
- It measures **real code**: every output is fed through
20
- ``latticeai.core.agent.extract_action_details`` — the exact parser/repair
21
- function the production loop uses (``agent.py`` calls it in plan/execute/verify).
22
- An ``agent-loop`` reference row additionally runs the real
23
- ``latticeai.core.agent_eval.run_agent_eval`` state machine over its full
24
- scripted scenario suite.
25
-
26
- Three modes
27
- ===========
28
- 1. **scripted** (default, always runnable, no model/network): a curated corpus
29
- of model-realistic outputs per quality tier (frontier / mid-local /
30
- weak-local). Proves the harness works and exposes the loop's repair boundary
31
- deterministically.
32
- 2. **live** (opt-in, ``--live-endpoint``): sends a fixed set of agent prompts to
33
- an OpenAI-compatible local endpoint (e.g. LM Studio / llama.cpp / vLLM) and
34
- runs the *real* completions through the same parser, measuring true
35
- end-to-end generation latency. Falls back to scripted with an honest message
36
- if the endpoint is unreachable.
37
- 3. **filegen** (opt-in, ``--filegen``): the weekly multi-model file-generation
38
- report. Discovers *installed* local gemma/qwen/llama MLX models via the
39
- product's own model catalog + HF download checks, loads each with the real
40
- ``LLMRouter``, and drives the real ``generate_file_content`` pipeline
41
- (prompt → extract → validate → retry → repair) for each canonical file type
42
- (html/css/js/py/json/md). Reports a model × filetype success matrix.
43
- **FAIL-OPEN by design**: no models installed (or a load failure) yields a
44
- clear skip report and exit code 0 — this mode is a scheduled/manual report,
45
- never a CI gate.
46
-
47
- Honesty note on latency
48
- ========================
49
- In **scripted** mode ``latency_ms`` is the *parse+repair* cost only
50
- (microseconds); it is NOT model inference time and must not be read as such.
51
- Real generation latency is only meaningful in **live** mode.
52
-
53
- Usage
54
- =====
55
- .venv/bin/python scripts/bench_models.py # scripted matrix
56
- .venv/bin/python scripts/bench_models.py --json out.json # + machine output
57
- .venv/bin/python scripts/bench_models.py \
58
- --live-endpoint http://127.0.0.1:1234/v1 --model my-local-model
59
- .venv/bin/python scripts/bench_models.py --filegen # weekly filegen report
60
- .venv/bin/python scripts/bench_models.py --filegen --json filegen_report.json
61
- """
62
-
63
- from __future__ import annotations
64
-
65
- import argparse
66
- import asyncio
67
- import json
68
- import statistics
69
- import sys
70
- import time
71
- import urllib.error
72
- import urllib.request
73
- from pathlib import Path
74
- from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple
75
-
76
- sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
77
-
78
- from latticeai.core.agent import extract_action_details # noqa: E402
79
- from latticeai.core.agent_eval import run_agent_eval # noqa: E402
80
- from latticeai.core.file_generation import ( # noqa: E402
81
- generate_file_content,
82
- validate_file_content,
83
- )
84
-
85
- # ── Scripted corpora ─────────────────────────────────────────────────────
86
- # Each entry is one model output for a canonical agent turn. The tiers encode
87
- # how the SAME intent degrades as model quality drops. All "should_parse=True"
88
- # entries are recoverable by the real loop; "should_parse=False" entries are
89
- # genuinely broken (past the repair boundary) and SHOULD fail — that is what
90
- # makes success_rate < 1.0 meaningful rather than a rigged demo.
91
-
92
- _CLEAN = [
93
- ('{"action": "plan", "goal": "ingest", "steps": [{"action": "write_file"}]}', True),
94
- ('{"action": "write_file", "args": {"path": "note.txt", "content": "hi"}}', True),
95
- ('{"action": "final", "message": "done"}', True),
96
- ('{"action": "verdict", "verdict": "PASS", "next_state": "DONE", "reason": "ok"}', True),
97
- ('{"action": "knowledge_graph_search", "args": {"query": "roadmap"}}', True),
98
- ('{"action": "read_file", "args": {"path": "README.md"}}', True),
99
- ]
100
-
101
- _MID = [
102
- # markdown-fenced JSON (very common)
103
- ('```json\n{"action": "plan", "goal": "ingest", "steps": []}\n```', True),
104
- ('```\n{"action": "write_file", "args": {"path": "a.txt", "content": "x"}}\n```', True),
105
- # prose preamble then the object
106
- ('Sure! Here is the next step:\n{"action": "final", "message": "done"}', True),
107
- # trailing comma before closing brace
108
- ('{"action": "read_file", "args": {"path": "README.md",}}', True),
109
- # trailing comma in array
110
- ('{"action": "plan", "goal": "g", "steps": [{"action": "read_file"},]}', True),
111
- ('```json\n{"action": "verdict", "verdict": "PASS", "next_state": "DONE", "reason": "ok"}\n```', True),
112
- ]
113
-
114
- _WEAK = [
115
- # <think> reasoning block that itself contains braces, then the action
116
- ('<think>I should write the file {maybe}</think>\n{"action": "write_file", "args": {"path": "n.txt", "content": "c"}}', True),
117
- ('<reasoning>ok</reasoning> {"action": "final", "message": "done"}', True),
118
- # Python dict literal (single quotes, True) — ast.literal_eval path
119
- ("{'action': 'read_file', 'args': {'path': 'x.txt'}}", True),
120
- ("{'action': 'verdict', 'verdict': 'PASS', 'next_state': 'DONE', 'reason': 'ok'}", True),
121
- # fenced + trailing comma + prose all at once
122
- ('Here you go:\n```json\n{"action": "plan", "goal": "g", "steps": [],}\n```', True),
123
- # genuinely broken: pure prose, no JSON object at all -> must fail
124
- ("I think we are done here, nothing else to do.", False),
125
- # genuinely broken: object but missing the required "action" field
126
- ('{"message": "done", "status": "ok"}', False),
127
- # genuinely broken: truncated / unbalanced braces past repair
128
- ('{"action": "write_file", "args": {"path": "n.txt"', False),
129
- ]
130
-
131
- _PROFILES: Dict[str, List[Tuple[str, bool]]] = {
132
- "frontier (clean JSON)": _CLEAN,
133
- "mid-local (fenced/prose/commas)": _MID,
134
- "weak-local (think/py-literal/broken)": _WEAK,
135
- }
136
-
137
-
138
- def _bench_corpus(corpus: List[Tuple[str, bool]]) -> Dict[str, Any]:
139
- total = len(corpus)
140
- parsed = 0
141
- repaired = 0
142
- latencies: List[float] = []
143
- mismatches: List[str] = []
144
- repair_kinds: Dict[str, int] = {}
145
- for raw, should_parse in corpus:
146
- t0 = time.perf_counter()
147
- try:
148
- _action, repairs = extract_action_details(raw)
149
- ok = True
150
- except (ValueError, Exception): # noqa: BLE001 - parser raises ValueError
151
- ok = False
152
- repairs = []
153
- latencies.append((time.perf_counter() - t0) * 1000.0)
154
- if ok:
155
- parsed += 1
156
- if repairs:
157
- repaired += 1
158
- for kind in repairs:
159
- repair_kinds[kind] = repair_kinds.get(kind, 0) + 1
160
- if ok != should_parse:
161
- mismatches.append(raw[:60])
162
- return {
163
- "total": total,
164
- "parsed": parsed,
165
- "repaired": repaired,
166
- "success_rate": round(parsed / total, 4) if total else 0.0,
167
- "repair_rate": round(repaired / parsed, 4) if parsed else 0.0,
168
- "latency_ms_mean": round(statistics.fmean(latencies), 4) if latencies else 0.0,
169
- "repair_kinds": repair_kinds,
170
- # A healthy harness has zero mismatches: every "should_parse" call
171
- # matched reality. Non-empty means the corpus/parser disagree.
172
- "harness_mismatches": mismatches,
173
- }
174
-
175
-
176
- # ── Live (OpenAI-compatible) path ────────────────────────────────────────
177
- _LIVE_PROMPTS = [
178
- "Return ONLY one JSON object for the next agent step: read the file README.md.",
179
- "Return ONLY one JSON object: plan to ingest a document, then finish.",
180
- "Return ONLY one JSON object: write 'hello' to notes.txt.",
181
- "Return ONLY one JSON object: a PASS verdict moving the loop to DONE.",
182
- "Return ONLY one JSON object: search the knowledge graph for 'roadmap'.",
183
- ]
184
-
185
- _SYSTEM = (
186
- "You are the executor of an agent loop. Every reply MUST be exactly one "
187
- 'JSON object with an "action" field and nothing else.'
188
- )
189
-
190
-
191
- def _openai_chat(base_url: str, model: str, prompt: str, *, timeout: float) -> Tuple[str, float]:
192
- url = base_url.rstrip("/") + "/chat/completions"
193
- body = json.dumps(
194
- {
195
- "model": model,
196
- "messages": [
197
- {"role": "system", "content": _SYSTEM},
198
- {"role": "user", "content": prompt},
199
- ],
200
- "temperature": 0.2,
201
- "max_tokens": 256,
202
- }
203
- ).encode("utf-8")
204
- req = urllib.request.Request( # noqa: S310 - operator-supplied local endpoint
205
- url, data=body, headers={"Content-Type": "application/json"}, method="POST"
206
- )
207
- t0 = time.perf_counter()
208
- with urllib.request.urlopen(req, timeout=timeout) as res: # noqa: S310
209
- payload = json.loads(res.read().decode("utf-8", errors="replace"))
210
- latency = (time.perf_counter() - t0) * 1000.0
211
- text = payload["choices"][0]["message"]["content"]
212
- return text, latency
213
-
214
-
215
- def _bench_live(base_url: str, model: str, *, timeout: float) -> Dict[str, Any]:
216
- total = 0
217
- parsed = 0
218
- repaired = 0
219
- latencies: List[float] = []
220
- for prompt in _LIVE_PROMPTS:
221
- total += 1
222
- try:
223
- text, latency = _openai_chat(base_url, model, prompt, timeout=timeout)
224
- except (urllib.error.URLError, OSError, KeyError, ValueError) as exc:
225
- return {"error": f"live endpoint unreachable/invalid: {exc}"}
226
- latencies.append(latency)
227
- try:
228
- _action, repairs = extract_action_details(text)
229
- parsed += 1
230
- if repairs:
231
- repaired += 1
232
- except Exception: # noqa: BLE001
233
- pass
234
- return {
235
- "total": total,
236
- "parsed": parsed,
237
- "repaired": repaired,
238
- "success_rate": round(parsed / total, 4) if total else 0.0,
239
- "repair_rate": round(repaired / parsed, 4) if parsed else 0.0,
240
- "latency_ms_mean": round(statistics.fmean(latencies), 2) if latencies else 0.0,
241
- }
242
-
243
-
244
- # ── Filegen benchmark (weekly multi-model report, fail-open) ─────────────
245
- # One canonical request per supported file type. Requests are phrased the way
246
- # users actually ask, so the report measures the real prompt+pipeline, not a
247
- # synthetic best case.
248
- FILEGEN_TARGETS: List[Tuple[str, str]] = [
249
- ("bench_page.html", "간단한 자기소개 웹 페이지를 만들어줘. 제목, 소개 문단, 연락처 목록을 포함해."),
250
- ("bench_styles.css", "Create a stylesheet with body typography, a header rule and a .card class."),
251
- ("bench_app.js", "Create a small browser script that renders a todo list and lets the user add items."),
252
- ("bench_tool.py", "Create a Python script that takes a CSV file path from argv and prints the row count."),
253
- ("bench_data.json", "Create a JSON document describing three sample books with title, author and year."),
254
- ("bench_notes.md", "Create a Markdown document with a title, two sections and a bullet list."),
255
- ]
256
-
257
- _FILEGEN_FAMILIES = ("gemma", "qwen", "llama")
258
-
259
-
260
- def discover_filegen_models() -> List[Dict[str, str]]:
261
- """Installed local MLX models from the product catalog (gemma/qwen/llama).
262
-
263
- Uses the same catalog + on-disk checks the runtime uses (``~/.ltcai``
264
- model dir or the HF hub cache). Any import/probe failure is fail-open:
265
- an empty list, never an exception.
266
- """
267
- try:
268
- from latticeai.models.router import (
269
- _looks_like_hf_model_dir,
270
- hf_cache_model_dir,
271
- hf_model_dir,
272
- )
273
- from latticeai.services.model_catalog import ENGINE_MODEL_CATALOG
274
- except Exception as exc: # noqa: BLE001 - discovery must never crash the report
275
- print(f"filegen: model catalog unavailable ({exc}); no models discovered")
276
- return []
277
- found: List[Dict[str, str]] = []
278
- for entry in ENGINE_MODEL_CATALOG.get("local_mlx", []):
279
- model_id = str(entry.get("id") or "")
280
- lowered = model_id.lower()
281
- family = next((f for f in _FILEGEN_FAMILIES if f in lowered), None)
282
- if not family:
283
- continue
284
- try:
285
- downloaded = (
286
- _looks_like_hf_model_dir(hf_model_dir(model_id))
287
- or hf_cache_model_dir(model_id) is not None
288
- )
289
- except Exception: # noqa: BLE001
290
- downloaded = False
291
- if downloaded:
292
- found.append({"id": model_id, "family": family})
293
- return found
294
-
295
-
296
- async def bench_filegen_model(
297
- model_label: str,
298
- generate_async: Callable[[str], Awaitable[str]],
299
- targets: Optional[List[Tuple[str, str]]] = None,
300
- ) -> Dict[str, Any]:
301
- """Drive the real file-generation pipeline for every target file type.
302
-
303
- ``generate_async`` is the same ``context -> raw model text`` callable shape
304
- the chat layer injects, so this measures exactly what production runs
305
- (extraction, validation, corrective retry, deterministic repair included).
306
- """
307
- rows: List[Dict[str, Any]] = []
308
- for target_path, user_request in (targets or FILEGEN_TARGETS):
309
- t0 = time.perf_counter()
310
- content, meta = await generate_file_content(
311
- generate_async, target_path=target_path, user_request=user_request,
312
- )
313
- elapsed_ms = round((time.perf_counter() - t0) * 1000.0, 1)
314
- valid, reason = validate_file_content(content, target_path)
315
- attempts = meta.get("attempts") or []
316
- rows.append({
317
- "target": target_path,
318
- "type": target_path.rsplit(".", 1)[-1],
319
- "valid": valid, # valid after sanitize/repair
320
- "reason": reason,
321
- "clean_first_try": bool(attempts and attempts[0].get("valid")),
322
- "repaired": bool(meta.get("repaired")),
323
- "attempts": len(attempts),
324
- "latency_ms": elapsed_ms,
325
- "bytes": len(content.encode("utf-8", errors="replace")),
326
- })
327
- total = len(rows)
328
- valid_count = sum(1 for r in rows if r["valid"])
329
- clean_count = sum(1 for r in rows if r["valid"] and not r["repaired"])
330
- return {
331
- "model": model_label,
332
- "targets": rows,
333
- "total": total,
334
- "success_rate": round(valid_count / total, 4) if total else 0.0,
335
- # Of the valid files, how many did NOT need the deterministic repair
336
- # fallback — the honest "the model itself produced usable output" rate.
337
- "clean_rate": round(clean_count / total, 4) if total else 0.0,
338
- }
339
-
340
-
341
- def _make_router_generate(router: Any, model_id: str) -> Callable[[str], Awaitable[str]]:
342
- """Mirror the direct chat path's generation call (chat_intents)."""
343
-
344
- async def _generate(context: str) -> str:
345
- return str(
346
- await router.generate_as(
347
- model_id,
348
- message="Return only the requested file content.",
349
- context=context,
350
- max_tokens=4096,
351
- temperature=0.2,
352
- )
353
- )
354
-
355
- return _generate
356
-
357
-
358
- async def _filegen_run_models(models: List[Dict[str, str]]) -> List[Dict[str, Any]]:
359
- from latticeai.models.router import LLMRouter
360
-
361
- router = LLMRouter()
362
- results: List[Dict[str, Any]] = []
363
- for model in models:
364
- model_id = model["id"]
365
- try:
366
- await router.load_model(model_id)
367
- except Exception as exc: # noqa: BLE001 - fail-open per model
368
- results.append({
369
- "model": model_id, "family": model["family"],
370
- "skipped": True, "reason": f"load failed: {exc}",
371
- })
372
- continue
373
- result = await bench_filegen_model(model_id, _make_router_generate(router, model_id))
374
- result["family"] = model["family"]
375
- results.append(result)
376
- try:
377
- router.unload_model(model_id)
378
- except Exception: # noqa: BLE001
379
- pass
380
- return results
381
-
382
-
383
- def run_filegen_benchmark(
384
- models: Optional[List[Dict[str, str]]] = None,
385
- ) -> Dict[str, Any]:
386
- """Weekly model × filetype report. Fail-open: never raises, never gates."""
387
- if models is None:
388
- models = discover_filegen_models()
389
- if not models:
390
- return {
391
- "mode": "filegen",
392
- "skipped": True,
393
- "reason": (
394
- "no local gemma/qwen/llama models installed — install one via "
395
- "the app's model picker, then re-run"
396
- ),
397
- "models": [],
398
- }
399
- try:
400
- results = asyncio.run(_filegen_run_models(models))
401
- except Exception as exc: # noqa: BLE001 - fail-open at the run level too
402
- return {
403
- "mode": "filegen", "skipped": True,
404
- "reason": f"benchmark run failed: {exc}", "models": [],
405
- }
406
- return {"mode": "filegen", "skipped": False, "models": results}
407
-
408
-
409
- def format_filegen_report(report: Dict[str, Any]) -> str:
410
- lines = [
411
- "Weekly filegen benchmark (real pipeline: generate_file_content)",
412
- "=" * 72,
413
- ]
414
- if report.get("skipped"):
415
- lines.append(f"SKIPPED (fail-open): {report.get('reason')}")
416
- lines.append("=" * 72)
417
- return "\n".join(lines)
418
- for model in report.get("models", []):
419
- if model.get("skipped"):
420
- lines.append(f" {model['model']}: SKIPPED — {model.get('reason')}")
421
- continue
422
- lines.append(
423
- f" {model['model']} "
424
- f"(success={model['success_rate']}, clean={model['clean_rate']})"
425
- )
426
- for row in model.get("targets", []):
427
- status = "ok" if row["valid"] else "FAIL"
428
- if row["valid"] and row["repaired"]:
429
- status = "ok(repaired)"
430
- elif row["valid"] and not row["clean_first_try"]:
431
- status = "ok(retry)"
432
- lines.append(
433
- f" {row['type']:<5} {status:<12} "
434
- f"attempts={row['attempts']} latency_ms={row['latency_ms']} "
435
- f"bytes={row['bytes']}"
436
- )
437
- lines.append("=" * 72)
438
- lines.append("note: success = valid after sanitize/repair; clean = valid without repair fallback")
439
- return "\n".join(lines)
440
-
441
-
442
- # ── Reporting ────────────────────────────────────────────────────────────
443
- def _fmt_row(name: str, r: Dict[str, Any]) -> str:
444
- return (
445
- f" {name:<40} "
446
- f"success={r['success_rate']:<7} "
447
- f"repair={r['repair_rate']:<7} "
448
- f"n={r.get('total', '-'):<4} "
449
- f"latency_ms={r['latency_ms_mean']}"
450
- )
451
-
452
-
453
- def main(argv: List[str] | None = None) -> int:
454
- parser = argparse.ArgumentParser(description="Model robustness benchmark harness")
455
- parser.add_argument("--json", dest="json_out", help="write full report JSON to this path")
456
- parser.add_argument("--live-endpoint", help="OpenAI-compatible base URL, e.g. http://127.0.0.1:1234/v1")
457
- parser.add_argument("--model", help="model id for the live endpoint")
458
- parser.add_argument("--timeout", type=float, default=30.0, help="live request timeout seconds")
459
- parser.add_argument(
460
- "--filegen", action="store_true",
461
- help="weekly multi-model file-generation report (fail-open, exit 0 even with no models)",
462
- )
463
- args = parser.parse_args(argv)
464
-
465
- if args.filegen:
466
- filegen_report = run_filegen_benchmark()
467
- print(format_filegen_report(filegen_report))
468
- if args.json_out:
469
- Path(args.json_out).write_text(
470
- json.dumps(filegen_report, ensure_ascii=False, indent=2), encoding="utf-8"
471
- )
472
- print(f"wrote {args.json_out}")
473
- # FAIL-OPEN: this is a scheduled report, never a CI gate.
474
- return 0
475
-
476
- report: Dict[str, Any] = {"mode": "scripted", "profiles": {}}
477
-
478
- # Scripted matrix (always).
479
- for name, corpus in _PROFILES.items():
480
- report["profiles"][name] = _bench_corpus(corpus)
481
-
482
- # Real agent-loop reference row.
483
- loop = run_agent_eval()
484
- report["agent_loop_reference"] = {
485
- "scenarios": loop["scenarios"],
486
- "passed": loop["passed"],
487
- "success_rate": loop["success_rate"],
488
- "recovery_rate": loop["recovery_rate"],
489
- "parse_errors": loop["parse_errors"],
490
- "parse_recovered": loop["parse_recovered"],
491
- }
492
-
493
- # Optional live row.
494
- if args.live_endpoint:
495
- if not args.model:
496
- print("--live-endpoint requires --model", file=sys.stderr)
497
- return 2
498
- live = _bench_live(args.live_endpoint, args.model, timeout=args.timeout)
499
- report["mode"] = "scripted+live"
500
- report["live"] = {"endpoint": args.live_endpoint, "model": args.model, **live}
501
-
502
- # Human-readable matrix.
503
- print("Model robustness benchmark (real parser: extract_action_details)")
504
- print("=" * 72)
505
- print("Scripted tiers (latency_ms = parse+repair only, NOT model inference):")
506
- mismatch_total = 0
507
- for name, r in report["profiles"].items():
508
- print(_fmt_row(name, r))
509
- if r["repair_kinds"]:
510
- print(f" repairs used: {r['repair_kinds']}")
511
- mismatch_total += len(r["harness_mismatches"])
512
- ref = report["agent_loop_reference"]
513
- print("-" * 72)
514
- print(
515
- f" agent-loop (real state machine) "
516
- f"success={ref['success_rate']:<7} "
517
- f"recovery={ref['recovery_rate']:<7} "
518
- f"scenarios={ref['scenarios']}"
519
- )
520
- if "live" in report:
521
- lv = report["live"]
522
- print("-" * 72)
523
- if "error" in lv:
524
- print(f" live [{lv['model']}] SKIPPED: {lv['error']}")
525
- else:
526
- print(_fmt_row(f"live [{lv['model']}] (real inference)", lv))
527
- print("=" * 72)
528
- print(f"harness self-check: {mismatch_total} corpus/parser mismatches (0 = healthy)")
529
-
530
- if args.json_out:
531
- Path(args.json_out).write_text(json.dumps(report, indent=2), encoding="utf-8")
532
- print(f"wrote {args.json_out}")
533
-
534
- # Non-zero exit only if the harness itself is inconsistent (a real bug),
535
- # never because a weak tier scored low — low scores are the finding.
536
- return 1 if mismatch_total else 0
537
-
538
-
539
- if __name__ == "__main__":
540
- raise SystemExit(main())