ltcai 11.2.0 → 11.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (247) hide show
  1. package/README.md +46 -53
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/MULTI_AGENT_RUNTIME.md +1 -1
  6. package/docs/ONBOARDING.md +1 -1
  7. package/docs/OPERATIONS.md +6 -2
  8. package/docs/PERMISSION_MODE.md +1 -1
  9. package/docs/TRUST_MODEL.md +1 -1
  10. package/docs/WHY_LATTICE.md +1 -1
  11. package/docs/kg-schema.md +2 -2
  12. package/docs/v11.3.0_PLAN.md +202 -0
  13. package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/graph/_kg_common/__init__.py +287 -0
  16. package/lattice_brain/graph/_kg_common/extraction.py +516 -0
  17. package/lattice_brain/graph/_kg_common/relations.py +161 -0
  18. package/lattice_brain/graph/_kg_common/text.py +479 -0
  19. package/lattice_brain/graph/discovery_index/__init__.py +35 -0
  20. package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
  21. package/lattice_brain/graph/discovery_index/extract.py +137 -0
  22. package/lattice_brain/graph/discovery_index/scan.py +411 -0
  23. package/lattice_brain/graph/discovery_index/upsert.py +495 -0
  24. package/lattice_brain/graph/projection/__init__.py +42 -0
  25. package/lattice_brain/graph/projection/curation.py +500 -0
  26. package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
  27. package/lattice_brain/graph/retrieval/__init__.py +54 -0
  28. package/lattice_brain/graph/retrieval/context.py +197 -0
  29. package/lattice_brain/graph/retrieval/graph_view.py +319 -0
  30. package/lattice_brain/graph/retrieval/hybrid.py +488 -0
  31. package/lattice_brain/graph/retrieval/maintenance.py +121 -0
  32. package/lattice_brain/graph/retrieval/signals.py +95 -0
  33. package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
  34. package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
  35. package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
  36. package/lattice_brain/graph/retrieval_vector/search.py +560 -0
  37. package/lattice_brain/graph/retrieval_vector/status.py +374 -0
  38. package/lattice_brain/ingestion/__init__.py +130 -0
  39. package/lattice_brain/ingestion/_contract.py +90 -0
  40. package/lattice_brain/ingestion/constants.py +127 -0
  41. package/lattice_brain/ingestion/folder_scan.py +57 -0
  42. package/lattice_brain/ingestion/folders.py +258 -0
  43. package/lattice_brain/ingestion/hashing.py +26 -0
  44. package/lattice_brain/ingestion/jobs_api.py +107 -0
  45. package/lattice_brain/ingestion/models.py +80 -0
  46. package/lattice_brain/ingestion/pipeline.py +486 -0
  47. package/lattice_brain/ingestion/quality.py +209 -0
  48. package/lattice_brain/ingestion/routing.py +295 -0
  49. package/lattice_brain/multimodal/__init__.py +164 -0
  50. package/lattice_brain/multimodal/audio.py +77 -0
  51. package/lattice_brain/multimodal/common.py +118 -0
  52. package/lattice_brain/multimodal/images.py +498 -0
  53. package/lattice_brain/multimodal/ports.py +169 -0
  54. package/lattice_brain/multimodal/video.py +410 -0
  55. package/lattice_brain/portability/__init__.py +90 -0
  56. package/lattice_brain/portability/_contract.py +42 -0
  57. package/lattice_brain/portability/backups.py +338 -0
  58. package/lattice_brain/portability/bundles.py +136 -0
  59. package/lattice_brain/portability/constants.py +93 -0
  60. package/lattice_brain/portability/fsops.py +138 -0
  61. package/lattice_brain/portability/service.py +41 -0
  62. package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
  63. package/lattice_brain/runtime/__init__.py +1 -1
  64. package/lattice_brain/runtime/multi_agent.py +1 -1
  65. package/latticeai/__init__.py +1 -1
  66. package/latticeai/api/chronicle.py +63 -0
  67. package/latticeai/core/agent/__init__.py +93 -0
  68. package/latticeai/core/agent/_contract.py +79 -0
  69. package/latticeai/core/agent/context.py +57 -0
  70. package/latticeai/core/agent/deps.py +125 -0
  71. package/latticeai/core/agent/execution.py +622 -0
  72. package/latticeai/core/agent/planning.py +145 -0
  73. package/latticeai/core/agent/recovery.py +157 -0
  74. package/latticeai/core/agent/runtime.py +210 -0
  75. package/latticeai/core/agent/verification.py +231 -0
  76. package/latticeai/core/embedding_providers/__init__.py +151 -0
  77. package/latticeai/core/embedding_providers/base.py +199 -0
  78. package/latticeai/core/embedding_providers/captions.py +162 -0
  79. package/latticeai/core/embedding_providers/profiles.py +126 -0
  80. package/latticeai/core/embedding_providers/text.py +350 -0
  81. package/latticeai/core/embedding_providers/vision.py +352 -0
  82. package/latticeai/core/file_generation/__init__.py +115 -0
  83. package/latticeai/core/file_generation/bundles.py +76 -0
  84. package/latticeai/core/file_generation/extraction.py +154 -0
  85. package/latticeai/core/file_generation/inference.py +235 -0
  86. package/latticeai/core/file_generation/orchestration.py +152 -0
  87. package/latticeai/core/file_generation/prompting.py +117 -0
  88. package/latticeai/core/file_generation/repair.py +114 -0
  89. package/latticeai/core/file_generation/sanitize.py +61 -0
  90. package/latticeai/core/file_generation/validation.py +201 -0
  91. package/latticeai/core/legacy_compatibility.py +1 -1
  92. package/latticeai/core/marketplace.py +1 -1
  93. package/latticeai/core/messages.py +9 -0
  94. package/latticeai/core/workspace_os_constants.py +1 -1
  95. package/latticeai/integrations/telegram_bot/__init__.py +123 -0
  96. package/latticeai/integrations/telegram_bot/__main__.py +17 -0
  97. package/latticeai/integrations/telegram_bot/config.py +86 -0
  98. package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
  99. package/latticeai/integrations/telegram_bot/flows.py +478 -0
  100. package/latticeai/integrations/telegram_bot/helpers.py +322 -0
  101. package/latticeai/integrations/telegram_bot/screens.py +394 -0
  102. package/latticeai/models/router/__init__.py +88 -0
  103. package/latticeai/models/router/_contract.py +66 -0
  104. package/latticeai/models/router/branding.py +56 -0
  105. package/latticeai/models/router/catalog.py +69 -0
  106. package/latticeai/models/router/documents.py +199 -0
  107. package/latticeai/models/router/errors.py +37 -0
  108. package/latticeai/models/router/generation.py +258 -0
  109. package/latticeai/models/router/loading.py +291 -0
  110. package/latticeai/models/router/local_models.py +85 -0
  111. package/latticeai/models/router/registry.py +147 -0
  112. package/latticeai/runtime/build_phases/__init__.py +82 -0
  113. package/latticeai/runtime/build_phases/features.py +407 -0
  114. package/latticeai/runtime/build_phases/foundation.py +555 -0
  115. package/latticeai/runtime/build_phases/web.py +492 -0
  116. package/latticeai/runtime/runtime_context.py +1 -0
  117. package/latticeai/services/architecture_readiness.py +48 -19
  118. package/latticeai/services/brain_intelligence/__init__.py +58 -0
  119. package/latticeai/services/brain_intelligence/_contract.py +71 -0
  120. package/latticeai/services/brain_intelligence/consistency.py +193 -0
  121. package/latticeai/services/brain_intelligence/constants.py +47 -0
  122. package/latticeai/services/brain_intelligence/digest.py +258 -0
  123. package/latticeai/services/brain_intelligence/health.py +331 -0
  124. package/latticeai/services/brain_intelligence/proposals.py +264 -0
  125. package/latticeai/services/brain_intelligence/sampling.py +84 -0
  126. package/latticeai/services/brain_intelligence/service.py +48 -0
  127. package/latticeai/services/chronicle.py +557 -0
  128. package/latticeai/services/memory_service/__init__.py +52 -0
  129. package/latticeai/services/memory_service/_contract.py +100 -0
  130. package/latticeai/services/memory_service/brief.py +431 -0
  131. package/latticeai/services/memory_service/constants.py +57 -0
  132. package/latticeai/services/memory_service/maintenance.py +138 -0
  133. package/latticeai/services/memory_service/manager.py +186 -0
  134. package/latticeai/services/memory_service/proof.py +136 -0
  135. package/latticeai/services/memory_service/recall.py +225 -0
  136. package/latticeai/services/memory_service/service.py +48 -0
  137. package/latticeai/services/memory_service/stores.py +110 -0
  138. package/latticeai/services/model_runtime/__init__.py +322 -0
  139. package/latticeai/services/model_runtime/cloud.py +87 -0
  140. package/latticeai/services/model_runtime/download.py +282 -0
  141. package/latticeai/services/model_runtime/engines.py +341 -0
  142. package/latticeai/services/model_runtime/loading.py +178 -0
  143. package/latticeai/services/model_runtime/service.py +129 -0
  144. package/latticeai/services/model_runtime/state.py +131 -0
  145. package/latticeai/services/model_runtime/status.py +255 -0
  146. package/latticeai/services/product_readiness.py +15 -7
  147. package/latticeai/setup/wizard/__init__.py +126 -0
  148. package/latticeai/setup/wizard/catalog.py +172 -0
  149. package/latticeai/setup/wizard/detect.py +323 -0
  150. package/latticeai/setup/wizard/install.py +348 -0
  151. package/latticeai/setup/wizard/paths.py +168 -0
  152. package/latticeai/setup/wizard/plans.py +74 -0
  153. package/latticeai/setup/wizard/recommend.py +320 -0
  154. package/package.json +6 -2
  155. package/scripts/bump_version.py +14 -0
  156. package/scripts/capture_release_evidence.mjs +33 -21
  157. package/scripts/check_current_release_docs.mjs +1 -1
  158. package/scripts/check_i18n_namespace_coverage.mjs +41 -4
  159. package/scripts/check_max_file_lines.mjs +102 -0
  160. package/scripts/check_release_evidence_bound.mjs +30 -15
  161. package/scripts/check_screenshot_pixel_delta.py +34 -4
  162. package/scripts/check_server_i18n.mjs +1 -0
  163. package/scripts/generate_rust_parity_fixtures.py +562 -0
  164. package/scripts/lib/mock_server_fingerprint.mjs +94 -0
  165. package/scripts/release_screen_claims.json +31 -2
  166. package/src-tauri/Cargo.lock +361 -3
  167. package/src-tauri/Cargo.toml +6 -1
  168. package/src-tauri/src/backend.rs +349 -0
  169. package/src-tauri/src/folder.rs +33 -0
  170. package/src-tauri/src/main.rs +97 -399
  171. package/src-tauri/tauri.conf.json +1 -1
  172. package/static/app/asset-manifest.json +41 -37
  173. package/static/app/assets/Act-yYpYnn0v.js +1 -0
  174. package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
  175. package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
  176. package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
  177. package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
  178. package/static/app/assets/Capture-CFIRsFNE.js +1 -0
  179. package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
  180. package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
  181. package/static/app/assets/Library-DwO3yZST.js +1 -0
  182. package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
  183. package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
  184. package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
  185. package/static/app/assets/System-DW8F-2xL.js +1 -0
  186. package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
  187. package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
  188. package/static/app/assets/brain-Ci1CkWjM.js +1 -0
  189. package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
  190. package/static/app/assets/circle-check-DfInj-qD.js +1 -0
  191. package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
  192. package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
  193. package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
  194. package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
  195. package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
  196. package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
  197. package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
  198. package/static/app/assets/index-_u5iUHDr.js +10 -0
  199. package/static/app/assets/input-B0lPdRQZ.js +1 -0
  200. package/static/app/assets/link-2-CoFbooHS.js +1 -0
  201. package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
  202. package/static/app/assets/primitives-DEbN-d6p.js +1 -0
  203. package/static/app/assets/search-BybIWPNd.js +1 -0
  204. package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
  205. package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
  206. package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
  207. package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
  208. package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
  209. package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
  210. package/static/app/assets/utils-BlZr7Pd4.js +4 -0
  211. package/static/app/assets/workspace-jJY4RuAV.js +1 -0
  212. package/static/app/index.html +4 -4
  213. package/static/sw.js +1 -1
  214. package/lattice_brain/graph/_kg_common.py +0 -1331
  215. package/lattice_brain/graph/discovery_index.py +0 -1141
  216. package/lattice_brain/graph/retrieval.py +0 -1120
  217. package/lattice_brain/graph/retrieval_vector.py +0 -1293
  218. package/lattice_brain/ingestion.py +0 -1525
  219. package/lattice_brain/multimodal.py +0 -1258
  220. package/latticeai/core/agent.py +0 -1465
  221. package/latticeai/core/embedding_providers.py +0 -1196
  222. package/latticeai/core/file_generation.py +0 -1047
  223. package/latticeai/integrations/telegram_bot.py +0 -1390
  224. package/latticeai/models/router.py +0 -1007
  225. package/latticeai/runtime/build_phases.py +0 -1450
  226. package/latticeai/services/brain_intelligence.py +0 -1083
  227. package/latticeai/services/memory_service.py +0 -1177
  228. package/latticeai/services/model_runtime.py +0 -1281
  229. package/latticeai/setup/wizard.py +0 -1310
  230. package/static/app/assets/Act-AWf0SAKp.js +0 -1
  231. package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
  232. package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
  233. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
  234. package/static/app/assets/Capture-CqOSzyPr.js +0 -1
  235. package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
  236. package/static/app/assets/Library-CX-bbhmK.js +0 -1
  237. package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
  238. package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
  239. package/static/app/assets/System-Bu2t5hn1.js +0 -1
  240. package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
  241. package/static/app/assets/brain-DJMoqrwx.js +0 -1
  242. package/static/app/assets/index-BpYkzcVm.js +0 -10
  243. package/static/app/assets/input-DSlJJxRs.js +0 -1
  244. package/static/app/assets/primitives-BCx6TvfG.js +0 -1
  245. package/static/app/assets/search-Cgy8cCFJ.js +0 -1
  246. package/static/app/assets/utils-zqPZJxdx.js +0 -4
  247. package/static/app/assets/workspace-DXTihhfU.js +0 -1
@@ -1,1047 +0,0 @@
1
- """Model-agnostic file content generation pipeline.
2
-
3
- Small local models (gemma/qwen/llama 7B class) asked to "generate an HTML
4
- file" commonly wrap the payload in chat noise: leading commentary ("Sure!
5
- Here is your page:"), Markdown fences, ``<think>`` reasoning blocks,
6
- trailing explanations, or an incomplete document. The previous direct-write
7
- path saved that reply nearly verbatim, so weak models produced broken files.
8
-
9
- This module makes the file-creation flow robust regardless of which LLM is
10
- loaded, by treating the model as an untrusted content source:
11
-
12
- 1. Prompt — extension-aware instructions anchored with the exact first
13
- line the reply must start with (small models follow examples, not rules).
14
- 2. Extract — strip reasoning blocks and conversational framing, pick the
15
- best fenced block, slice known document boundaries.
16
- 3. Validate — per-extension structural checks (HTML document shape, JSON
17
- parses, CSS has rule blocks, refusal/chat detection).
18
- 4. Retry — one corrective attempt that tells the model what was wrong.
19
- 5. Repair — deterministic scaffolds guarantee the user still gets a valid
20
- file even when the model never produces usable output.
21
-
22
- The pipeline is pure (no I/O, no FastAPI); the chat layer injects an async
23
- ``generate(context) -> str`` callable.
24
- """
25
-
26
- from __future__ import annotations
27
-
28
- import ast
29
- import html as html_lib
30
- import json
31
- import re
32
- from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple
33
-
34
- from latticeai.core.quiet import quiet
35
-
36
- # ── extraction ──────────────────────────────────────────────────────────
37
-
38
- _THINK_BLOCK_RE = re.compile(
39
- r"<(think|thinking|reasoning|reflection)>.*?</\1>",
40
- re.DOTALL | re.IGNORECASE,
41
- )
42
- # Unclosed think block (model hit the token limit mid-reasoning).
43
- _THINK_OPEN_RE = re.compile(r"<(think|thinking|reasoning)>.*\Z", re.DOTALL | re.IGNORECASE)
44
-
45
- _FENCE_RE = re.compile(r"```([\w.+-]*)[ \t]*\n(.*?)```", re.DOTALL)
46
-
47
- # Conversational lines that small models prepend/append around the payload.
48
- _CHAT_LINE_RE = re.compile(
49
- r"^\s*("
50
- r"(sure|of course|certainly|okay|ok|alright|great|absolutely)\b[^\n]*"
51
- r"|here('s| is| are)\b[^\n]*"
52
- r"|i('ve| have) (created|written|generated|made)\b[^\n]*"
53
- r"|(below|following) is\b[^\n]*"
54
- r"|let me know\b[^\n]*"
55
- r"|hope (this|that) helps[^\n]*"
56
- r"|feel free\b[^\n]*"
57
- r"|물론(입니다|이죠|이에요)?[!., ]*[^\n]*"
58
- r"|네[,!. ][^\n]*"
59
- r"|알겠습니다[^\n]*"
60
- r"|다음은[^\n]*(입니다|합니다)[:.]?[^\n]*"
61
- r"|아래는?[^\n]*(입니다|내용)[^\n]*"
62
- r"|(요청하신|원하시는)[^\n]*(입니다|만들었습니다|작성했습니다)[^\n]*"
63
- r"|(파일|내용|코드)[을를]?\s*(생성|작성|만들)[^\n]*"
64
- r"|도움이 (필요하|되)[^\n]*"
65
- r"|추가로[^\n]*(말씀|요청)[^\n]*"
66
- r")\s*$",
67
- re.IGNORECASE,
68
- )
69
-
70
- _REFUSAL_RE = re.compile(
71
- r"(i can('|no)?t|i'?m (sorry|unable)|as an ai|cannot assist"
72
- r"|죄송(하지만|합니다)|할 수 없|불가능합니다|도와드릴 수 없)",
73
- re.IGNORECASE,
74
- )
75
-
76
- # Language tags that identify a fenced block as the payload for an extension.
77
- _EXT_FENCE_LANGS: Dict[str, Tuple[str, ...]] = {
78
- ".html": ("html", "htm", "xhtml"),
79
- ".htm": ("html", "htm", "xhtml"),
80
- ".css": ("css",),
81
- ".js": ("js", "javascript"),
82
- ".jsx": ("jsx", "javascript"),
83
- ".ts": ("ts", "typescript"),
84
- ".tsx": ("tsx", "typescript"),
85
- ".py": ("py", "python"),
86
- ".json": ("json",),
87
- ".yaml": ("yaml", "yml"),
88
- ".yml": ("yaml", "yml"),
89
- ".toml": ("toml",),
90
- ".md": ("md", "markdown"),
91
- ".markdown": ("md", "markdown"),
92
- ".sql": ("sql",),
93
- ".sh": ("sh", "bash", "shell", "zsh"),
94
- ".xml": ("xml", "svg"),
95
- ".csv": ("csv",),
96
- ".txt": ("txt", "text", "plaintext"),
97
- ".vue": ("vue", "html"),
98
- ".svelte": ("svelte", "html"),
99
- }
100
-
101
-
102
- def _ext(path: str) -> str:
103
- dot = path.rfind(".")
104
- return path[dot:].lower() if dot >= 0 else ""
105
-
106
-
107
- def _strip_chat_lines(text: str) -> str:
108
- """Drop leading/trailing conversational lines around the payload."""
109
- lines = text.split("\n")
110
- start, end = 0, len(lines)
111
- while start < end and (not lines[start].strip() or _CHAT_LINE_RE.match(lines[start])):
112
- start += 1
113
- while end > start and (not lines[end - 1].strip() or _CHAT_LINE_RE.match(lines[end - 1])):
114
- end -= 1
115
- stripped = "\n".join(lines[start:end]).strip()
116
- return stripped if stripped else text.strip()
117
-
118
-
119
- def extract_file_content(raw: str, target_path: str) -> str:
120
- """Recover the intended file payload from an arbitrary model reply."""
121
- text = (raw or "").strip()
122
- if not text:
123
- return ""
124
- text = _THINK_BLOCK_RE.sub("", text)
125
- text = _THINK_OPEN_RE.sub("", text).strip()
126
-
127
- ext = _ext(target_path)
128
- fences = _FENCE_RE.findall(text + ("\n```" if text.count("```") % 2 else ""))
129
- if fences:
130
- wanted = _EXT_FENCE_LANGS.get(ext, ())
131
- matching = [body for lang, body in fences if lang.lower() in wanted]
132
- candidates = matching if matching else [body for _, body in fences]
133
- # The payload is the largest block; short blocks are usually usage
134
- # snippets ("run it with: python app.py").
135
- content = max(candidates, key=len).strip()
136
- else:
137
- content = _strip_chat_lines(text)
138
-
139
- if ext in (".html", ".htm"):
140
- content = _slice_html_document(content)
141
- elif ext == ".json":
142
- sliced = _slice_json_document(content)
143
- if sliced is not None:
144
- content = sliced
145
- return content.strip()
146
-
147
-
148
- def _slice_html_document(content: str) -> str:
149
- """Cut a complete HTML document out of surrounding prose when present."""
150
- lower = content.lower()
151
- start = lower.find("<!doctype")
152
- if start < 0:
153
- start = lower.find("<html")
154
- if start > 0:
155
- content = content[start:]
156
- lower = lower[start:]
157
- end = lower.rfind("</html>")
158
- if end >= 0:
159
- content = content[: end + len("</html>")]
160
- return content
161
-
162
-
163
- def _slice_json_document(content: str) -> Optional[str]:
164
- """Return the largest parseable JSON value inside ``content``, if any."""
165
- candidates: List[str] = [content]
166
- for opener, closer in (("{", "}"), ("[", "]")):
167
- start = content.find(opener)
168
- end = content.rfind(closer)
169
- if start >= 0 and end > start:
170
- candidates.append(content[start : end + 1])
171
- best: Optional[str] = None
172
- for candidate in candidates:
173
- try:
174
- json.loads(candidate)
175
- except (ValueError, TypeError):
176
- quiet()
177
- continue
178
- if best is None or len(candidate) > len(best):
179
- best = candidate
180
- return best
181
-
182
-
183
- # ── validation ──────────────────────────────────────────────────────────
184
-
185
- def looks_like_refusal(content: str) -> bool:
186
- head = content[:300]
187
- return bool(_REFUSAL_RE.search(head)) and len(content) < 600
188
-
189
-
190
- # Braced code types validated structurally (balanced delimiters, no fences).
191
- _BRACED_CODE_EXTENSIONS = frozenset({".js", ".jsx", ".ts", ".tsx"})
192
- # Single-file components validated by their block tags being closed.
193
- _COMPONENT_EXTENSIONS = frozenset({".vue", ".svelte"})
194
-
195
-
196
- def _strip_code_literals(text: str) -> str:
197
- """Remove string literals and comments so delimiter counting stays honest.
198
-
199
- A cheap single-pass scanner (not a parser): quotes ('', "", ``), line
200
- comments (//) and block comments (/* */) commonly contain lone braces
201
- that would otherwise false-flag valid JS/TS as unbalanced.
202
- """
203
- out: List[str] = []
204
- i, n = 0, len(text)
205
- while i < n:
206
- ch = text[i]
207
- nxt = text[i + 1] if i + 1 < n else ""
208
- if ch == "\\":
209
- i += 2 # escaped char (inside or outside a literal — always skip)
210
- continue
211
- if ch in ("'", '"', "`"):
212
- quote = ch
213
- i += 1
214
- while i < n:
215
- if text[i] == "\\":
216
- i += 2
217
- continue
218
- if text[i] == quote:
219
- i += 1
220
- break
221
- i += 1
222
- continue
223
- if ch == "/" and nxt == "/":
224
- while i < n and text[i] != "\n":
225
- i += 1
226
- continue
227
- if ch == "/" and nxt == "*":
228
- end = text.find("*/", i + 2)
229
- i = n if end < 0 else end + 2
230
- continue
231
- out.append(ch)
232
- i += 1
233
- return "".join(out)
234
-
235
-
236
- def _check_balanced_delimiters(content: str) -> Tuple[bool, str]:
237
- """Lenient count-based balance check for braced code (js/ts family).
238
-
239
- Only *counts* are compared, never ordering, so valid-but-unusual code is
240
- not rejected; a truncated file with a dangling ``{`` still fails.
241
- """
242
- stripped = _strip_code_literals(content)
243
- for opener, closer, label in (("{", "}", "braces"), ("(", ")", "parentheses"), ("[", "]", "brackets")):
244
- if stripped.count(opener) != stripped.count(closer):
245
- return False, f"unbalanced {label} ({opener}{closer}) — the file looks truncated"
246
- return True, "ok"
247
-
248
-
249
- def _check_component_blocks(content: str) -> Tuple[bool, str]:
250
- """Vue/Svelte SFC sanity: every opened block tag must be closed."""
251
- lower = content.lower()
252
- for tag in ("template", "script", "style"):
253
- opened = len(re.findall(rf"<{tag}(?:\s[^>]*)?>", lower))
254
- closed = lower.count(f"</{tag}>")
255
- if opened != closed:
256
- return False, f"<{tag}> block is not closed — the component looks truncated"
257
- return True, "ok"
258
-
259
-
260
- def validate_file_content(content: str, target_path: str) -> Tuple[bool, str]:
261
- """Structural sanity check per file type. Returns (ok, reason)."""
262
- if not content.strip():
263
- return False, "empty output"
264
- if looks_like_refusal(content):
265
- return False, "the reply was a refusal/chat message, not file content"
266
-
267
- ext = _ext(target_path)
268
- if ext in (".html", ".htm"):
269
- lower = content.lower()
270
- if "<html" not in lower and "<!doctype" not in lower:
271
- return False, "not a complete HTML document (missing <!DOCTYPE html>/<html>)"
272
- if "</html>" not in lower:
273
- return False, "HTML document is truncated (missing </html>)"
274
- # A document that merely *contains* html somewhere is not a valid
275
- # file payload: fenced/chat-wrapped replies must fail here so the
276
- # extraction pass gets a chance to slice out the real document.
277
- stripped = content.lstrip()
278
- if not (stripped.lower().startswith("<!doctype") or stripped.lower().startswith("<html")):
279
- return False, "HTML document is wrapped in prose or fences"
280
- if "```" in content:
281
- return False, "output still contains Markdown fences"
282
- return True, "ok"
283
- if ext == ".json":
284
- try:
285
- json.loads(content)
286
- except (ValueError, TypeError) as exc:
287
- return False, f"invalid JSON: {exc}"
288
- return True, "ok"
289
- if ext == ".css":
290
- if "```" in content:
291
- return False, "output still contains Markdown fences"
292
- if "{" not in content or "}" not in content:
293
- return False, "no CSS rule blocks found"
294
- return True, "ok"
295
- if ext == ".py":
296
- if "```" in content:
297
- return False, "output still contains Markdown fences"
298
- try:
299
- ast.parse(content)
300
- except SyntaxError as exc:
301
- return False, f"invalid Python syntax: {exc.msg} (line {exc.lineno})"
302
- return True, "ok"
303
- if ext in _BRACED_CODE_EXTENSIONS:
304
- if "```" in content:
305
- return False, "output still contains Markdown fences"
306
- return _check_balanced_delimiters(content)
307
- if ext in _COMPONENT_EXTENSIONS:
308
- if "```" in content:
309
- return False, "output still contains Markdown fences"
310
- return _check_component_blocks(content)
311
- if ext in (".sh", ".sql"):
312
- if "```" in content:
313
- return False, "output still contains Markdown fences"
314
- return True, "ok"
315
- # Prose types (.md, .txt, .csv, …) have no grammar to check, which used to
316
- # mean *nothing* was checked: a 1–4B model that answered "Sure! Here is the
317
- # document you asked for:" and stopped had its sentence saved as the file,
318
- # because the fence stripper only removes conversational lines it can
319
- # recognise and the length guard on `looks_like_refusal` lets a wordy
320
- # refusal through. The two checks below are the only ones that generalise
321
- # without inventing a grammar: it must not still be wearing fences, and it
322
- # must not be *only* an answer about the file.
323
- if "```" in content:
324
- return False, "output still contains Markdown fences"
325
- if _looks_like_commentary(content):
326
- return False, "the reply talks about the file instead of being the file"
327
- return True, "ok"
328
-
329
-
330
- # Openers that mean "I am about to give you the thing" — if the whole reply is
331
- # one of these, the thing never arrived.
332
- _COMMENTARY_RE = re.compile(
333
- r"^\s*("
334
- r"(sure|of course|certainly|okay|ok|alright|here|below|the following)\b"
335
- r"|i('ve| have| will|'ll)\b"
336
- r"|(물론|네[,!. ]|알겠|다음은|아래(는|의)?|요청하신|원하시는)"
337
- r")",
338
- re.IGNORECASE,
339
- )
340
-
341
-
342
- def _looks_like_commentary(content: str) -> bool:
343
- """True when the reply reads as an answer *about* a file, not the file.
344
-
345
- Deliberately conservative — a long document that merely opens with "The
346
- following" is a document. Only a short reply that both opens
347
- conversationally and never grows into content is rejected, so a real file
348
- is never thrown away to catch a chat line.
349
- """
350
- stripped = content.strip()
351
- if len(stripped) > 400:
352
- return False
353
- if not _COMMENTARY_RE.match(stripped):
354
- return False
355
- # Structure means content arrived after the preamble: a heading, a list, a
356
- # table row, a delimiter, or simply several lines of body text.
357
- body = stripped.split("\n", 1)[1].strip() if "\n" in stripped else ""
358
- if re.search(r"^\s*(#{1,6}\s|[-*+]\s|\d+[.)]\s|\||>)", body, re.MULTILINE):
359
- return False
360
- return len(body) < 120
361
-
362
-
363
- # ── prompting ───────────────────────────────────────────────────────────
364
-
365
- _FIRST_LINE_HINTS: Dict[str, str] = {
366
- ".html": "<!DOCTYPE html>",
367
- ".htm": "<!DOCTYPE html>",
368
- ".py": "# (python code — imports or code on the first line)",
369
- ".sh": "#!/bin/sh",
370
- ".json": "{",
371
- ".xml": "<?xml version=\"1.0\" encoding=\"UTF-8\"?>",
372
- }
373
-
374
- _TYPE_RULES: Dict[str, str] = {
375
- ".html": (
376
- "Produce ONE complete standalone HTML5 document: <!DOCTYPE html>, <html>, "
377
- "<head> with <meta charset=\"utf-8\"> and a <title>, inline <style> for CSS, "
378
- "and a closed </html> tag. Do not reference external files."
379
- ),
380
- ".htm": (
381
- "Produce ONE complete standalone HTML5 document ending with </html>."
382
- ),
383
- ".json": "Produce strictly valid JSON (double quotes, no comments, no trailing commas).",
384
- ".css": "Produce valid CSS rules only.",
385
- ".md": "Produce well-structured Markdown with headings.",
386
- ".markdown": "Produce well-structured Markdown with headings.",
387
- ".csv": "Produce CSV with a header row; comma-separated, one record per line.",
388
- ".py": "Produce complete runnable Python source code.",
389
- ".js": "Produce complete valid JavaScript source code.",
390
- ".jsx": "Produce one complete React component file in JSX.",
391
- ".ts": "Produce complete valid TypeScript source code.",
392
- ".tsx": "Produce one complete React component file in TSX (TypeScript).",
393
- ".vue": "Produce ONE complete Vue single-file component with closed <template>/<script>/<style> blocks.",
394
- ".svelte": "Produce ONE complete Svelte component; every <script>/<style> block must be closed.",
395
- }
396
-
397
- # Multi-file bundles override the standalone-HTML rule: the page must link
398
- # its sibling files instead of inlining everything.
399
- _BUNDLE_HTML_RULE = (
400
- "Produce ONE complete HTML5 document: <!DOCTYPE html>, <html>, <head> with "
401
- "<meta charset=\"utf-8\"> and a <title>, and a closed </html> tag. "
402
- "This page is part of a multi-file project: link the project stylesheet(s) "
403
- "with <link rel=\"stylesheet\" href=\"...\"> and load the project script(s) "
404
- "with <script src=\"...\"></script> just before </body>. Reference ONLY the "
405
- "project files listed below — no other external files, no inline <style> "
406
- "blocks, no inline behavior scripts."
407
- )
408
-
409
- # Vite/React bundles need a module entry point, not classic script tags.
410
- _BUNDLE_HTML_MODULE_RULE = (
411
- "Produce ONE complete HTML5 document: <!DOCTYPE html>, <html>, <head> with "
412
- "<meta charset=\"utf-8\"> and a <title>, and a closed </html> tag. "
413
- "This page is the Vite entry of a React project: the <body> must contain "
414
- "<div id=\"root\"></div> and load the app with "
415
- "<script type=\"module\" src=\"/src/main.jsx\"></script> just before "
416
- "</body>. No inline <style> blocks, no other scripts, no external files."
417
- )
418
-
419
-
420
- def _bundle_html_rule(bundle_files: List[str]) -> str:
421
- """Pick the HTML bundle rule that matches the bundle's technology."""
422
- if any(str(path).lower().endswith((".jsx", ".tsx")) for path in bundle_files):
423
- return _BUNDLE_HTML_MODULE_RULE
424
- return _BUNDLE_HTML_RULE
425
-
426
-
427
- def build_file_generation_context(
428
- target_path: str,
429
- user_request: str,
430
- feedback: Optional[str] = None,
431
- bundle_files: Optional[List[str]] = None,
432
- ) -> str:
433
- """Strict, extension-aware generation instructions.
434
-
435
- Small models ignore abstract rules but reliably imitate concrete anchors,
436
- so the prompt pins the exact first line of the expected output.
437
- """
438
- ext = _ext(target_path)
439
- parts = [
440
- "You are a file content generator. Your entire reply is saved verbatim "
441
- f"as the file `{target_path}` — it is NOT shown in a chat.",
442
- "Rules:",
443
- "- Output ONLY the raw file content.",
444
- "- No Markdown code fences (```), no explanations, no greetings, "
445
- "no text before or after the content.",
446
- ]
447
- type_rule = _TYPE_RULES.get(ext)
448
- if bundle_files and ext in (".html", ".htm"):
449
- type_rule = _bundle_html_rule(bundle_files)
450
- if type_rule:
451
- parts.append(f"- {type_rule}")
452
- if bundle_files:
453
- listed = ", ".join(bundle_files)
454
- parts.append(f"- Project files in this bundle: {listed}")
455
- first_line = _FIRST_LINE_HINTS.get(ext)
456
- if first_line:
457
- parts.append(f"- The very first line of your reply must be: {first_line}")
458
- if feedback:
459
- parts.append(
460
- "Your previous attempt was rejected: "
461
- f"{feedback}. Fix that and output only the corrected file content."
462
- )
463
- parts.append(f"\nUser request: {user_request}")
464
- return "\n".join(parts)
465
-
466
-
467
- # ── repair (deterministic fallback) ─────────────────────────────────────
468
-
469
- def repair_file_content(content: str, target_path: str, user_request: str) -> str:
470
- """Turn whatever the model produced into a valid file of the target type.
471
-
472
- This is the last resort after retries: the user asked for a file, so the
473
- request must still end in a well-formed file, never an error.
474
- """
475
- ext = _ext(target_path)
476
- salvage = content.strip()
477
- if looks_like_refusal(salvage):
478
- salvage = ""
479
-
480
- if ext in (".html", ".htm"):
481
- return _repair_html(salvage, user_request)
482
- if ext == ".json":
483
- sliced = _slice_json_document(salvage)
484
- if sliced is not None:
485
- return sliced
486
- return json.dumps(
487
- {"request": user_request, "content": salvage},
488
- ensure_ascii=False,
489
- indent=2,
490
- )
491
- if ext == ".py" and salvage:
492
- # The repair guarantee for Python is parseability: unparseable output
493
- # is preserved honestly as a commented-out draft, never as a broken
494
- # module the user has to debug.
495
- try:
496
- ast.parse(salvage)
497
- return salvage
498
- except SyntaxError:
499
- commented = "\n".join(f"# {line}" for line in salvage.splitlines())
500
- return (
501
- f"# TODO: model produced invalid Python for: {user_request}\n"
502
- "# The draft below is preserved as comments — fix and uncomment.\n"
503
- f"{commented}\n"
504
- )
505
- if salvage:
506
- return salvage
507
- # Nothing usable at all — leave an honest placeholder in the right format.
508
- comment = {
509
- ".py": "# TODO: model produced no usable content for: ",
510
- ".js": "// TODO: model produced no usable content for: ",
511
- ".jsx": "// TODO: model produced no usable content for: ",
512
- ".ts": "// TODO: model produced no usable content for: ",
513
- ".tsx": "// TODO: model produced no usable content for: ",
514
- ".css": "/* TODO: model produced no usable content for: ",
515
- ".sh": "# TODO: model produced no usable content for: ",
516
- ".sql": "-- TODO: model produced no usable content for: ",
517
- }.get(ext, "")
518
- if ext == ".css":
519
- return f"{comment}{user_request} */\n"
520
- if comment:
521
- return f"{comment}{user_request}\n"
522
- return f"{user_request}\n"
523
-
524
-
525
- def _repair_html(salvage: str, user_request: str) -> str:
526
- lower = salvage.lower()
527
- if "<html" in lower or "<!doctype" in lower:
528
- # A real document that is merely truncated — close it.
529
- doc = _slice_html_document(salvage)
530
- low = doc.lower()
531
- if "</body>" not in low and "<body" in low:
532
- doc += "\n</body>"
533
- if "</html>" not in low:
534
- doc += "\n</html>"
535
- return doc
536
- if re.search(r"<\w+[^>]*>", salvage):
537
- body = salvage # an HTML fragment — embed as-is
538
- elif salvage:
539
- body = "\n".join(
540
- f" <p>{html_lib.escape(line)}</p>"
541
- for line in salvage.splitlines()
542
- if line.strip()
543
- )
544
- else:
545
- body = f" <p>{html_lib.escape(user_request)}</p>"
546
- title = html_lib.escape(user_request[:60] or "Generated page")
547
- return (
548
- "<!DOCTYPE html>\n"
549
- "<html lang=\"ko\">\n"
550
- "<head>\n"
551
- " <meta charset=\"utf-8\">\n"
552
- " <meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n"
553
- f" <title>{title}</title>\n"
554
- " <style>\n"
555
- " body { font-family: system-ui, sans-serif; margin: 2rem auto; "
556
- "max-width: 720px; line-height: 1.6; padding: 0 1rem; }\n"
557
- " </style>\n"
558
- "</head>\n"
559
- "<body>\n"
560
- f"{body}\n"
561
- "</body>\n"
562
- "</html>"
563
- )
564
-
565
-
566
- # Extensions the Brain UI can render inline (preview) after creation.
567
- PREVIEWABLE_EXTENSIONS = frozenset({
568
- ".html", ".htm", ".md", ".markdown", ".txt", ".json", ".css", ".js",
569
- ".csv", ".py", ".yaml", ".yml", ".xml", ".sql", ".sh",
570
- ".jsx", ".ts", ".tsx", ".vue", ".svelte",
571
- })
572
-
573
-
574
- # ── write-side sanitize (ArtifactWritePipeline) ─────────────────────────
575
-
576
- def sanitize_write_content(
577
- target_path: str,
578
- content: Any,
579
- user_request: str = "",
580
- ) -> Tuple[str, Dict[str, Any]]:
581
- """Single write-side guarantee for model-produced file content.
582
-
583
- The direct chat path already runs the full generate→validate→repair
584
- pipeline, but the agent JSON loop historically wrote ``args.content``
585
- verbatim — weak models routinely put fenced/chatty payloads there. This
586
- conservative sanitizer closes that gap for *any* write entry point:
587
-
588
- 1. content that already validates is returned byte-for-byte unchanged
589
- (trusted/user-authored content is never mangled);
590
- 2. otherwise the extraction pass strips fences/think-blocks/chat noise
591
- and is used only when the extracted payload validates;
592
- 3. otherwise deterministic repair guarantees a structurally valid file.
593
-
594
- Empty content is left untouched (creating an empty file is a legitimate,
595
- intentional action — e.g. ``__init__.py``). Returns ``(content, meta)``
596
- where meta is ``{"sanitized": bool, "repaired": bool, "reason": str}``.
597
- """
598
- raw = str(content or "")
599
- if not raw.strip():
600
- return raw, {"sanitized": False, "repaired": False, "reason": "empty"}
601
- ok, reason = validate_file_content(raw, target_path)
602
- if ok:
603
- return raw, {"sanitized": False, "repaired": False, "reason": "ok"}
604
- extracted = extract_file_content(raw, target_path)
605
- if extracted:
606
- extracted_ok, _ = validate_file_content(extracted, target_path)
607
- if extracted_ok:
608
- return extracted, {"sanitized": True, "repaired": False, "reason": reason}
609
- repaired = repair_file_content(
610
- extracted or raw, target_path, user_request or f"content for {target_path}"
611
- )
612
- return repaired, {"sanitized": True, "repaired": True, "reason": reason}
613
-
614
-
615
- # ── filename inference ──────────────────────────────────────────────────
616
-
617
- _CREATE_VERB_RE = re.compile(
618
- r"(만들|생성|작성|써\s*줘|저장|create|make|write|generate|build|save)",
619
- re.IGNORECASE,
620
- )
621
-
622
- # Explicit type keyword → default filename. Ordered: first match wins.
623
- _TYPE_KEYWORDS: Tuple[Tuple[str, str], ...] = (
624
- (r"\bhtml\b|웹\s*페이지|웹페이지|홈페이지|landing\s*page|web\s*page", "generated_page.html"),
625
- (r"\bcss\b|스타일\s*시트", "styles.css"),
626
- (r"\bjavascript\b|\bjs\b\s*(파일|file)|자바스크립트", "script.js"),
627
- (r"\bpython\b|파이썬", "script.py"),
628
- (r"\bjson\b", "data.json"),
629
- (r"\bcsv\b", "data.csv"),
630
- (r"\byaml\b|\byml\b", "config.yaml"),
631
- (r"\bxml\b", "data.xml"),
632
- (r"\bsql\b", "query.sql"),
633
- (r"마크다운|\bmarkdown\b|\bmd\b\s*(파일|file)", "notes.md"),
634
- (r"텍스트\s*파일|\btext\s*file\b|\btxt\b", "notes.txt"),
635
- )
636
-
637
-
638
- def infer_file_target(message: str) -> Optional[str]:
639
- """Infer a filename for creation requests that name a type but no path.
640
-
641
- "html 파일 만들어줘" previously fell through to the agent JSON loop, which
642
- small models fail at. Inference keeps such requests on the deterministic
643
- direct-write path. Deliberately narrow: requires a creation verb and an
644
- explicit file-type keyword — report/document prose requests keep flowing
645
- to the document generator.
646
- """
647
- text = (message or "").strip()
648
- if not text or not _CREATE_VERB_RE.search(text):
649
- return None
650
- lower = text.lower()
651
- for pattern, filename in _TYPE_KEYWORDS:
652
- if re.search(pattern, lower):
653
- return filename
654
- return None
655
-
656
-
657
- # ── project manifest (multi-file bundles) ───────────────────────────────
658
-
659
- # ``\b`` fails against Korean particles ("js로") because Hangul is ``\w`` —
660
- # use ASCII lookarounds so type keywords match with or without a particle.
661
- _HTML_HINT_RE = re.compile(
662
- r"(?<![a-z0-9])html(?![a-z0-9])"
663
- r"|웹\s*페이지|웹페이지|홈페이지|웹\s*사이트|웹사이트|website|web\s*page|landing\s*page",
664
- )
665
- _CSS_HINT_RE = re.compile(r"(?<![a-z0-9])css(?![a-z0-9])|스타일\s*시트|stylesheet")
666
- _JS_HINT_RE = re.compile(
667
- r"(?<![a-z0-9])js(?![a-z0-9])|javascript|자바스크립트|자바\s*스크립트"
668
- )
669
- # An explicit filename means the user is managing paths — keep the
670
- # deterministic single-file flow untouched.
671
- _EXPLICIT_FILENAME_RE = re.compile(
672
- r"[\w-]+\.(?:html?|css|js|jsx|ts|tsx|py|json|md|txt|csv|vue|svelte)\b",
673
- re.IGNORECASE,
674
- )
675
- _PROJECT_NAME_RE = re.compile(r"([A-Za-z][A-Za-z0-9_-]{1,30})\s*(?:앱|app\b)", re.IGNORECASE)
676
- # React/Vite intent: the react keyword is specific enough on its own.
677
- _REACT_HINT_RE = re.compile(r"(?<![a-z0-9])react(?![a-z0-9])|리액트")
678
- _VITE_HINT_RE = re.compile(r"(?<![a-z0-9])vite(?![a-z0-9])")
679
- # Python package intent: language + package word, both required.
680
- _PYTHON_HINT_RE = re.compile(r"(?<![a-z0-9])python(?![a-z0-9])|파이썬")
681
- _PACKAGE_HINT_RE = re.compile(r"패키지|(?<![a-z0-9])package(?![a-z0-9])")
682
- _PKG_NAME_RE = re.compile(
683
- r"([A-Za-z][A-Za-z0-9_-]{1,30})\s*(?:패키지|package\b)", re.IGNORECASE
684
- )
685
-
686
-
687
- def _react_manifest(text: str) -> Dict[str, Any]:
688
- """Vite + React starter manifest (review Wave 4: manifest 확장)."""
689
- name_match = _PROJECT_NAME_RE.search(text)
690
- name = f"{name_match.group(1).lower()}-app" if name_match else "react-app"
691
- return {
692
- "name": name,
693
- "kind": "react",
694
- "files": [
695
- {
696
- "path": "package.json",
697
- "brief": (
698
- f'Vite React app manifest: strictly valid JSON with "name": "{name}", '
699
- '"private": true, "type": "module", "scripts" {"dev": "vite", '
700
- '"build": "vite build", "preview": "vite preview"}, "dependencies" '
701
- 'with react and react-dom (^18), and "devDependencies" with vite '
702
- "and @vitejs/plugin-react."
703
- ),
704
- },
705
- {
706
- "path": "index.html",
707
- "brief": (
708
- "The Vite entry HTML: <div id=\"root\"></div> in <body> and "
709
- "<script type=\"module\" src=\"/src/main.jsx\"></script> just "
710
- "before </body>. No inline styles or scripts."
711
- ),
712
- },
713
- {
714
- "path": "src/main.jsx",
715
- "brief": (
716
- "React entry: createRoot from react-dom/client rendering <App /> "
717
- "into #root; imports ./App.jsx and ./App.css."
718
- ),
719
- },
720
- {
721
- "path": "src/App.jsx",
722
- "brief": (
723
- "The main App component implementing the user's request as one "
724
- "self-contained React component (hooks allowed, no extra deps)."
725
- ),
726
- },
727
- {
728
- "path": "src/App.css",
729
- "brief": "All visual styles for the App component.",
730
- },
731
- ],
732
- }
733
-
734
-
735
- def _python_package_manifest(text: str) -> Dict[str, Any]:
736
- """Multi-file Python package manifest (review Wave 4: manifest 확장)."""
737
- name_match = _PKG_NAME_RE.search(text)
738
- raw_name = name_match.group(1).lower() if name_match else "my_package"
739
- module = re.sub(r"[^a-z0-9_]", "_", raw_name)
740
- if not re.match(r"[a-z_]", module):
741
- module = f"pkg_{module}"
742
- return {
743
- "name": module,
744
- "kind": "python",
745
- "files": [
746
- {
747
- "path": f"{module}/__init__.py",
748
- "brief": (
749
- f"Package init for {module}: import and re-export the public "
750
- "API from .core with an explicit __all__."
751
- ),
752
- },
753
- {
754
- "path": f"{module}/core.py",
755
- "brief": (
756
- "Implement the user's request as clean, documented functions/"
757
- "classes with type hints. Standard library only."
758
- ),
759
- },
760
- {
761
- "path": f"{module}/cli.py",
762
- "brief": (
763
- "argparse CLI wrapping the core API: a main() function and an "
764
- 'if __name__ == "__main__": main() guard.'
765
- ),
766
- },
767
- {
768
- "path": "README.md",
769
- "brief": (
770
- f"Usage documentation for the {module} package: install, import "
771
- "example, and CLI example."
772
- ),
773
- },
774
- ],
775
- }
776
-
777
-
778
- def infer_project_manifest(message: str) -> Optional[Dict[str, Any]]:
779
- """Infer a multi-file project manifest from a creation request.
780
-
781
- "todo 앱 html+css+js로 만들어줘" should yield real linked files, not one
782
- inlined page. Deliberately narrow and deterministic (weak local models
783
- never see this decision): requires a creation verb, a recognized project
784
- intent (web page + css/js, React/Vite app, or Python package), and no
785
- explicit filename. Single-type requests return ``None`` so the existing
786
- single-file flow is completely unchanged.
787
- """
788
- text = (message or "").strip()
789
- if not text or not _CREATE_VERB_RE.search(text):
790
- return None
791
- if _EXPLICIT_FILENAME_RE.search(text):
792
- return None
793
- lower = text.lower()
794
-
795
- # Most-specific first: React (its own structure), then Python package,
796
- # then the classic html+css/js web bundle.
797
- if _REACT_HINT_RE.search(lower) or _VITE_HINT_RE.search(lower):
798
- return _react_manifest(text)
799
- if _PYTHON_HINT_RE.search(lower) and _PACKAGE_HINT_RE.search(lower):
800
- return _python_package_manifest(text)
801
-
802
- wants_html = bool(_HTML_HINT_RE.search(lower))
803
- wants_css = bool(_CSS_HINT_RE.search(lower))
804
- wants_js = bool(_JS_HINT_RE.search(lower))
805
- if not wants_html or not (wants_css or wants_js):
806
- return None
807
-
808
- name_match = _PROJECT_NAME_RE.search(text)
809
- name = f"{name_match.group(1).lower()}-app" if name_match else "web-project"
810
-
811
- files: List[Dict[str, str]] = []
812
- html_refs: List[str] = []
813
- if wants_css:
814
- html_refs.append('<link rel="stylesheet" href="style.css"> in <head>')
815
- if wants_js:
816
- html_refs.append('<script src="app.js"></script> just before </body>')
817
- files.append({
818
- "path": "index.html",
819
- "brief": (
820
- "The main HTML page of the project. Reference the sibling files: "
821
- + " and ".join(html_refs)
822
- + ". Do not inline styles or behavior scripts."
823
- ),
824
- })
825
- if wants_css:
826
- files.append({
827
- "path": "style.css",
828
- "brief": "All visual styles for index.html (layout, colors, typography).",
829
- })
830
- if wants_js:
831
- files.append({
832
- "path": "app.js",
833
- "brief": (
834
- "All page behavior for index.html as plain browser JavaScript "
835
- "(no build step, no imports of missing files)."
836
- ),
837
- })
838
- return {"name": name, "kind": "web", "files": files}
839
-
840
-
841
- _HTML_LOCAL_REF_RE = re.compile(
842
- r"(?:href|src)\s*=\s*[\"']([^\"'#?]+)[\"']", re.IGNORECASE
843
- )
844
- _EXTERNAL_REF_PREFIXES = ("http://", "https://", "//", "data:", "mailto:", "tel:", "javascript:")
845
-
846
-
847
- def _local_bundle_refs(html: str) -> List[str]:
848
- """File references inside an HTML document that must exist in the bundle."""
849
- refs: List[str] = []
850
- for ref in _HTML_LOCAL_REF_RE.findall(html or ""):
851
- candidate = ref.strip()
852
- if not candidate or candidate.startswith(_EXTERNAL_REF_PREFIXES):
853
- continue
854
- if "." not in candidate.rsplit("/", 1)[-1]:
855
- continue # anchors / routes, not files
856
- refs.append(candidate)
857
- return refs
858
-
859
-
860
- def repair_bundle_references(files: Dict[str, str]) -> Tuple[Dict[str, str], List[str]]:
861
- """Deterministically point dangling HTML refs at real bundle files.
862
-
863
- A weak model asked for ``style.css`` sometimes links ``styles.css``. When
864
- a referenced file is missing but the bundle contains exactly one file of
865
- the same extension, the reference is rewritten. Returns ``(files, fixes)``.
866
- """
867
- names = {p.rsplit("/", 1)[-1] for p in files}
868
- fixes: List[str] = []
869
- repaired = dict(files)
870
- for path, content in files.items():
871
- if _ext(path) not in (".html", ".htm"):
872
- continue
873
- updated = content
874
- for ref in _local_bundle_refs(content):
875
- base = ref.rsplit("/", 1)[-1]
876
- if base in names:
877
- continue
878
- same_ext = [n for n in names if _ext(n) == _ext(base)]
879
- if len(same_ext) == 1:
880
- updated = updated.replace(ref, same_ext[0])
881
- fixes.append(f"{path}: '{ref}' -> '{same_ext[0]}'")
882
- if updated != content:
883
- repaired[path] = updated
884
- return repaired, fixes
885
-
886
-
887
- def validate_project_bundle(files: Dict[str, str]) -> Dict[str, Any]:
888
- """Bundle-level verification: every file valid, every HTML ref resolvable."""
889
- issues: List[str] = []
890
- per_file: Dict[str, Dict[str, Any]] = {}
891
- names = {p.rsplit("/", 1)[-1] for p in files}
892
- for path, content in files.items():
893
- ok, reason = validate_file_content(content, path)
894
- per_file[path] = {"valid": ok, "reason": reason}
895
- if not ok:
896
- issues.append(f"{path}: {reason}")
897
- if _ext(path) in (".html", ".htm"):
898
- for ref in _local_bundle_refs(content):
899
- if ref.rsplit("/", 1)[-1] not in names:
900
- issues.append(f"{path}: references missing file '{ref}'")
901
- return {"ok": not issues, "issues": issues, "files": per_file}
902
-
903
-
904
- # ── orchestration ───────────────────────────────────────────────────────
905
-
906
- async def generate_file_content(
907
- generate: Callable[[str], Awaitable[Any]],
908
- *,
909
- target_path: str,
910
- user_request: str,
911
- max_attempts: int = 2,
912
- bundle_files: Optional[List[str]] = None,
913
- ) -> Tuple[str, Dict[str, Any]]:
914
- """Generate validated file content with any LLM.
915
-
916
- ``generate`` is an async callable ``context -> raw model text``. Runs up
917
- to ``max_attempts`` model calls (each retry carrying corrective feedback),
918
- then falls back to deterministic repair, so the returned content is
919
- always non-empty and structurally valid for the target type.
920
-
921
- One extra call beyond ``max_attempts`` is spent — at most once per
922
- request — when the model has returned a byte-identical rejected reply.
923
- That is the one case where the ordinary retry is known to be dead on
924
- arrival: the corrective feedback did not change the reply, so the budget
925
- is better spent on a prompt that names the repetition than on a third
926
- identical round trip. Small local models hit this constantly; large ones
927
- never do, so the extra call is not charged to models that do not need it.
928
- """
929
- attempts: List[Dict[str, Any]] = []
930
- feedback: Optional[str] = None
931
- best_candidate = ""
932
- best_score = (-1, -1)
933
- seen: set[str] = set()
934
- escalations_left = 1
935
- attempt = 0
936
- budget = max_attempts
937
- while attempt < budget:
938
- attempt += 1
939
- context = build_file_generation_context(
940
- target_path, user_request, feedback=feedback, bundle_files=bundle_files,
941
- )
942
- try:
943
- raw = await generate(context)
944
- except Exception as exc: # model backend hiccup — repair still delivers
945
- attempts.append({"attempt": attempt, "valid": False, "reason": f"generation error: {exc}"})
946
- feedback = "the model call failed"
947
- continue
948
- candidate = extract_file_content(str(raw or ""), target_path)
949
- ok, reason = validate_file_content(candidate, target_path)
950
- record: Dict[str, Any] = {"attempt": attempt, "valid": ok, "reason": reason}
951
- if ok:
952
- attempts.append(record)
953
- return candidate, {"attempts": attempts, "repaired": False}
954
-
955
- # A small model handed the same corrective feedback often replays the
956
- # same reply verbatim. Saying "you sent this before" is the only signal
957
- # left that has any chance of moving it, and it makes the wasted retry
958
- # visible in the trace instead of looking like two genuine tries.
959
- fingerprint = candidate.strip()
960
- repeated = fingerprint in seen and bool(fingerprint)
961
- record["repeated"] = repeated
962
- seen.add(fingerprint)
963
- if repeated and escalations_left and attempt >= budget:
964
- # The retry budget is exhausted and the last thing it bought was a
965
- # duplicate. Buy one more, but only with a prompt that says so.
966
- escalations_left -= 1
967
- budget += 1
968
- record["escalated"] = True
969
- attempts.append(record)
970
-
971
- # Keep the candidate that is *closest to a file*, not the longest one.
972
- # Longest-wins handed repair a 900-character apology in preference to a
973
- # 300-character HTML document that only needed its </html> closing —
974
- # and repair can finish the document but can only bury the apology.
975
- score = _salvage_score(candidate, target_path)
976
- if score > best_score:
977
- best_score, best_candidate = score, candidate
978
-
979
- feedback = (
980
- f"{reason}. You already sent exactly this reply and it was rejected "
981
- "for the same reason — do not repeat it. Output the file itself, "
982
- "starting at its first character."
983
- if repeated else reason
984
- )
985
- repaired = repair_file_content(best_candidate, target_path, user_request)
986
- return repaired, {"attempts": attempts, "repaired": True}
987
-
988
-
989
- def _salvage_score(candidate: str, target_path: str) -> Tuple[int, int]:
990
- """How useful an invalid candidate is as raw material for repair.
991
-
992
- ``(tier, length)`` — tier first, so a short real document always beats a
993
- long non-document; length breaks ties within a tier.
994
-
995
- Tier 2 something of the right shape that repair can finish (an HTML
996
- document missing its close tag, parseable-ish JSON, Python that
997
- at least tokenises).
998
- Tier 1 ordinary text: no structure, but the words may be the content.
999
- Tier 0 a refusal — repair should prefer literally anything else, because
1000
- an apology written into the file is worse than an empty stub.
1001
- """
1002
- text = candidate.strip()
1003
- if not text:
1004
- return (0, 0)
1005
- if looks_like_refusal(text):
1006
- return (0, len(text))
1007
-
1008
- ext = _ext(target_path)
1009
- lower = text.lower()
1010
- if ext in (".html", ".htm"):
1011
- if lower.startswith("<!doctype") or lower.startswith("<html"):
1012
- return (2, len(text))
1013
- elif ext == ".json":
1014
- if _slice_json_document(text) is not None:
1015
- return (2, len(text))
1016
- elif ext == ".py":
1017
- try:
1018
- ast.parse(text)
1019
- except SyntaxError:
1020
- pass
1021
- else:
1022
- return (2, len(text))
1023
- elif ext in _BRACED_CODE_EXTENSIONS:
1024
- if _check_balanced_delimiters(text)[0]:
1025
- return (2, len(text))
1026
- elif ext in _COMPONENT_EXTENSIONS:
1027
- if _check_component_blocks(text)[0]:
1028
- return (2, len(text))
1029
- elif ext == ".css" and "{" in text and "}" in text:
1030
- return (2, len(text))
1031
- return (1, len(text))
1032
-
1033
-
1034
- __all__ = [
1035
- "PREVIEWABLE_EXTENSIONS",
1036
- "build_file_generation_context",
1037
- "extract_file_content",
1038
- "generate_file_content",
1039
- "infer_file_target",
1040
- "infer_project_manifest",
1041
- "looks_like_refusal",
1042
- "repair_bundle_references",
1043
- "repair_file_content",
1044
- "sanitize_write_content",
1045
- "validate_file_content",
1046
- "validate_project_bundle",
1047
- ]