ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -1,1047 +0,0 @@
|
|
|
1
|
-
"""Model-agnostic file content generation pipeline.
|
|
2
|
-
|
|
3
|
-
Small local models (gemma/qwen/llama 7B class) asked to "generate an HTML
|
|
4
|
-
file" commonly wrap the payload in chat noise: leading commentary ("Sure!
|
|
5
|
-
Here is your page:"), Markdown fences, ``<think>`` reasoning blocks,
|
|
6
|
-
trailing explanations, or an incomplete document. The previous direct-write
|
|
7
|
-
path saved that reply nearly verbatim, so weak models produced broken files.
|
|
8
|
-
|
|
9
|
-
This module makes the file-creation flow robust regardless of which LLM is
|
|
10
|
-
loaded, by treating the model as an untrusted content source:
|
|
11
|
-
|
|
12
|
-
1. Prompt — extension-aware instructions anchored with the exact first
|
|
13
|
-
line the reply must start with (small models follow examples, not rules).
|
|
14
|
-
2. Extract — strip reasoning blocks and conversational framing, pick the
|
|
15
|
-
best fenced block, slice known document boundaries.
|
|
16
|
-
3. Validate — per-extension structural checks (HTML document shape, JSON
|
|
17
|
-
parses, CSS has rule blocks, refusal/chat detection).
|
|
18
|
-
4. Retry — one corrective attempt that tells the model what was wrong.
|
|
19
|
-
5. Repair — deterministic scaffolds guarantee the user still gets a valid
|
|
20
|
-
file even when the model never produces usable output.
|
|
21
|
-
|
|
22
|
-
The pipeline is pure (no I/O, no FastAPI); the chat layer injects an async
|
|
23
|
-
``generate(context) -> str`` callable.
|
|
24
|
-
"""
|
|
25
|
-
|
|
26
|
-
from __future__ import annotations
|
|
27
|
-
|
|
28
|
-
import ast
|
|
29
|
-
import html as html_lib
|
|
30
|
-
import json
|
|
31
|
-
import re
|
|
32
|
-
from typing import Any, Awaitable, Callable, Dict, List, Optional, Tuple
|
|
33
|
-
|
|
34
|
-
from latticeai.core.quiet import quiet
|
|
35
|
-
|
|
36
|
-
# ── extraction ──────────────────────────────────────────────────────────
|
|
37
|
-
|
|
38
|
-
_THINK_BLOCK_RE = re.compile(
|
|
39
|
-
r"<(think|thinking|reasoning|reflection)>.*?</\1>",
|
|
40
|
-
re.DOTALL | re.IGNORECASE,
|
|
41
|
-
)
|
|
42
|
-
# Unclosed think block (model hit the token limit mid-reasoning).
|
|
43
|
-
_THINK_OPEN_RE = re.compile(r"<(think|thinking|reasoning)>.*\Z", re.DOTALL | re.IGNORECASE)
|
|
44
|
-
|
|
45
|
-
_FENCE_RE = re.compile(r"```([\w.+-]*)[ \t]*\n(.*?)```", re.DOTALL)
|
|
46
|
-
|
|
47
|
-
# Conversational lines that small models prepend/append around the payload.
|
|
48
|
-
_CHAT_LINE_RE = re.compile(
|
|
49
|
-
r"^\s*("
|
|
50
|
-
r"(sure|of course|certainly|okay|ok|alright|great|absolutely)\b[^\n]*"
|
|
51
|
-
r"|here('s| is| are)\b[^\n]*"
|
|
52
|
-
r"|i('ve| have) (created|written|generated|made)\b[^\n]*"
|
|
53
|
-
r"|(below|following) is\b[^\n]*"
|
|
54
|
-
r"|let me know\b[^\n]*"
|
|
55
|
-
r"|hope (this|that) helps[^\n]*"
|
|
56
|
-
r"|feel free\b[^\n]*"
|
|
57
|
-
r"|물론(입니다|이죠|이에요)?[!., ]*[^\n]*"
|
|
58
|
-
r"|네[,!. ][^\n]*"
|
|
59
|
-
r"|알겠습니다[^\n]*"
|
|
60
|
-
r"|다음은[^\n]*(입니다|합니다)[:.]?[^\n]*"
|
|
61
|
-
r"|아래는?[^\n]*(입니다|내용)[^\n]*"
|
|
62
|
-
r"|(요청하신|원하시는)[^\n]*(입니다|만들었습니다|작성했습니다)[^\n]*"
|
|
63
|
-
r"|(파일|내용|코드)[을를]?\s*(생성|작성|만들)[^\n]*"
|
|
64
|
-
r"|도움이 (필요하|되)[^\n]*"
|
|
65
|
-
r"|추가로[^\n]*(말씀|요청)[^\n]*"
|
|
66
|
-
r")\s*$",
|
|
67
|
-
re.IGNORECASE,
|
|
68
|
-
)
|
|
69
|
-
|
|
70
|
-
_REFUSAL_RE = re.compile(
|
|
71
|
-
r"(i can('|no)?t|i'?m (sorry|unable)|as an ai|cannot assist"
|
|
72
|
-
r"|죄송(하지만|합니다)|할 수 없|불가능합니다|도와드릴 수 없)",
|
|
73
|
-
re.IGNORECASE,
|
|
74
|
-
)
|
|
75
|
-
|
|
76
|
-
# Language tags that identify a fenced block as the payload for an extension.
|
|
77
|
-
_EXT_FENCE_LANGS: Dict[str, Tuple[str, ...]] = {
|
|
78
|
-
".html": ("html", "htm", "xhtml"),
|
|
79
|
-
".htm": ("html", "htm", "xhtml"),
|
|
80
|
-
".css": ("css",),
|
|
81
|
-
".js": ("js", "javascript"),
|
|
82
|
-
".jsx": ("jsx", "javascript"),
|
|
83
|
-
".ts": ("ts", "typescript"),
|
|
84
|
-
".tsx": ("tsx", "typescript"),
|
|
85
|
-
".py": ("py", "python"),
|
|
86
|
-
".json": ("json",),
|
|
87
|
-
".yaml": ("yaml", "yml"),
|
|
88
|
-
".yml": ("yaml", "yml"),
|
|
89
|
-
".toml": ("toml",),
|
|
90
|
-
".md": ("md", "markdown"),
|
|
91
|
-
".markdown": ("md", "markdown"),
|
|
92
|
-
".sql": ("sql",),
|
|
93
|
-
".sh": ("sh", "bash", "shell", "zsh"),
|
|
94
|
-
".xml": ("xml", "svg"),
|
|
95
|
-
".csv": ("csv",),
|
|
96
|
-
".txt": ("txt", "text", "plaintext"),
|
|
97
|
-
".vue": ("vue", "html"),
|
|
98
|
-
".svelte": ("svelte", "html"),
|
|
99
|
-
}
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
def _ext(path: str) -> str:
|
|
103
|
-
dot = path.rfind(".")
|
|
104
|
-
return path[dot:].lower() if dot >= 0 else ""
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
def _strip_chat_lines(text: str) -> str:
|
|
108
|
-
"""Drop leading/trailing conversational lines around the payload."""
|
|
109
|
-
lines = text.split("\n")
|
|
110
|
-
start, end = 0, len(lines)
|
|
111
|
-
while start < end and (not lines[start].strip() or _CHAT_LINE_RE.match(lines[start])):
|
|
112
|
-
start += 1
|
|
113
|
-
while end > start and (not lines[end - 1].strip() or _CHAT_LINE_RE.match(lines[end - 1])):
|
|
114
|
-
end -= 1
|
|
115
|
-
stripped = "\n".join(lines[start:end]).strip()
|
|
116
|
-
return stripped if stripped else text.strip()
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
def extract_file_content(raw: str, target_path: str) -> str:
|
|
120
|
-
"""Recover the intended file payload from an arbitrary model reply."""
|
|
121
|
-
text = (raw or "").strip()
|
|
122
|
-
if not text:
|
|
123
|
-
return ""
|
|
124
|
-
text = _THINK_BLOCK_RE.sub("", text)
|
|
125
|
-
text = _THINK_OPEN_RE.sub("", text).strip()
|
|
126
|
-
|
|
127
|
-
ext = _ext(target_path)
|
|
128
|
-
fences = _FENCE_RE.findall(text + ("\n```" if text.count("```") % 2 else ""))
|
|
129
|
-
if fences:
|
|
130
|
-
wanted = _EXT_FENCE_LANGS.get(ext, ())
|
|
131
|
-
matching = [body for lang, body in fences if lang.lower() in wanted]
|
|
132
|
-
candidates = matching if matching else [body for _, body in fences]
|
|
133
|
-
# The payload is the largest block; short blocks are usually usage
|
|
134
|
-
# snippets ("run it with: python app.py").
|
|
135
|
-
content = max(candidates, key=len).strip()
|
|
136
|
-
else:
|
|
137
|
-
content = _strip_chat_lines(text)
|
|
138
|
-
|
|
139
|
-
if ext in (".html", ".htm"):
|
|
140
|
-
content = _slice_html_document(content)
|
|
141
|
-
elif ext == ".json":
|
|
142
|
-
sliced = _slice_json_document(content)
|
|
143
|
-
if sliced is not None:
|
|
144
|
-
content = sliced
|
|
145
|
-
return content.strip()
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
def _slice_html_document(content: str) -> str:
|
|
149
|
-
"""Cut a complete HTML document out of surrounding prose when present."""
|
|
150
|
-
lower = content.lower()
|
|
151
|
-
start = lower.find("<!doctype")
|
|
152
|
-
if start < 0:
|
|
153
|
-
start = lower.find("<html")
|
|
154
|
-
if start > 0:
|
|
155
|
-
content = content[start:]
|
|
156
|
-
lower = lower[start:]
|
|
157
|
-
end = lower.rfind("</html>")
|
|
158
|
-
if end >= 0:
|
|
159
|
-
content = content[: end + len("</html>")]
|
|
160
|
-
return content
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
def _slice_json_document(content: str) -> Optional[str]:
|
|
164
|
-
"""Return the largest parseable JSON value inside ``content``, if any."""
|
|
165
|
-
candidates: List[str] = [content]
|
|
166
|
-
for opener, closer in (("{", "}"), ("[", "]")):
|
|
167
|
-
start = content.find(opener)
|
|
168
|
-
end = content.rfind(closer)
|
|
169
|
-
if start >= 0 and end > start:
|
|
170
|
-
candidates.append(content[start : end + 1])
|
|
171
|
-
best: Optional[str] = None
|
|
172
|
-
for candidate in candidates:
|
|
173
|
-
try:
|
|
174
|
-
json.loads(candidate)
|
|
175
|
-
except (ValueError, TypeError):
|
|
176
|
-
quiet()
|
|
177
|
-
continue
|
|
178
|
-
if best is None or len(candidate) > len(best):
|
|
179
|
-
best = candidate
|
|
180
|
-
return best
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
# ── validation ──────────────────────────────────────────────────────────
|
|
184
|
-
|
|
185
|
-
def looks_like_refusal(content: str) -> bool:
|
|
186
|
-
head = content[:300]
|
|
187
|
-
return bool(_REFUSAL_RE.search(head)) and len(content) < 600
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
# Braced code types validated structurally (balanced delimiters, no fences).
|
|
191
|
-
_BRACED_CODE_EXTENSIONS = frozenset({".js", ".jsx", ".ts", ".tsx"})
|
|
192
|
-
# Single-file components validated by their block tags being closed.
|
|
193
|
-
_COMPONENT_EXTENSIONS = frozenset({".vue", ".svelte"})
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
def _strip_code_literals(text: str) -> str:
|
|
197
|
-
"""Remove string literals and comments so delimiter counting stays honest.
|
|
198
|
-
|
|
199
|
-
A cheap single-pass scanner (not a parser): quotes ('', "", ``), line
|
|
200
|
-
comments (//) and block comments (/* */) commonly contain lone braces
|
|
201
|
-
that would otherwise false-flag valid JS/TS as unbalanced.
|
|
202
|
-
"""
|
|
203
|
-
out: List[str] = []
|
|
204
|
-
i, n = 0, len(text)
|
|
205
|
-
while i < n:
|
|
206
|
-
ch = text[i]
|
|
207
|
-
nxt = text[i + 1] if i + 1 < n else ""
|
|
208
|
-
if ch == "\\":
|
|
209
|
-
i += 2 # escaped char (inside or outside a literal — always skip)
|
|
210
|
-
continue
|
|
211
|
-
if ch in ("'", '"', "`"):
|
|
212
|
-
quote = ch
|
|
213
|
-
i += 1
|
|
214
|
-
while i < n:
|
|
215
|
-
if text[i] == "\\":
|
|
216
|
-
i += 2
|
|
217
|
-
continue
|
|
218
|
-
if text[i] == quote:
|
|
219
|
-
i += 1
|
|
220
|
-
break
|
|
221
|
-
i += 1
|
|
222
|
-
continue
|
|
223
|
-
if ch == "/" and nxt == "/":
|
|
224
|
-
while i < n and text[i] != "\n":
|
|
225
|
-
i += 1
|
|
226
|
-
continue
|
|
227
|
-
if ch == "/" and nxt == "*":
|
|
228
|
-
end = text.find("*/", i + 2)
|
|
229
|
-
i = n if end < 0 else end + 2
|
|
230
|
-
continue
|
|
231
|
-
out.append(ch)
|
|
232
|
-
i += 1
|
|
233
|
-
return "".join(out)
|
|
234
|
-
|
|
235
|
-
|
|
236
|
-
def _check_balanced_delimiters(content: str) -> Tuple[bool, str]:
|
|
237
|
-
"""Lenient count-based balance check for braced code (js/ts family).
|
|
238
|
-
|
|
239
|
-
Only *counts* are compared, never ordering, so valid-but-unusual code is
|
|
240
|
-
not rejected; a truncated file with a dangling ``{`` still fails.
|
|
241
|
-
"""
|
|
242
|
-
stripped = _strip_code_literals(content)
|
|
243
|
-
for opener, closer, label in (("{", "}", "braces"), ("(", ")", "parentheses"), ("[", "]", "brackets")):
|
|
244
|
-
if stripped.count(opener) != stripped.count(closer):
|
|
245
|
-
return False, f"unbalanced {label} ({opener}{closer}) — the file looks truncated"
|
|
246
|
-
return True, "ok"
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
def _check_component_blocks(content: str) -> Tuple[bool, str]:
|
|
250
|
-
"""Vue/Svelte SFC sanity: every opened block tag must be closed."""
|
|
251
|
-
lower = content.lower()
|
|
252
|
-
for tag in ("template", "script", "style"):
|
|
253
|
-
opened = len(re.findall(rf"<{tag}(?:\s[^>]*)?>", lower))
|
|
254
|
-
closed = lower.count(f"</{tag}>")
|
|
255
|
-
if opened != closed:
|
|
256
|
-
return False, f"<{tag}> block is not closed — the component looks truncated"
|
|
257
|
-
return True, "ok"
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
def validate_file_content(content: str, target_path: str) -> Tuple[bool, str]:
|
|
261
|
-
"""Structural sanity check per file type. Returns (ok, reason)."""
|
|
262
|
-
if not content.strip():
|
|
263
|
-
return False, "empty output"
|
|
264
|
-
if looks_like_refusal(content):
|
|
265
|
-
return False, "the reply was a refusal/chat message, not file content"
|
|
266
|
-
|
|
267
|
-
ext = _ext(target_path)
|
|
268
|
-
if ext in (".html", ".htm"):
|
|
269
|
-
lower = content.lower()
|
|
270
|
-
if "<html" not in lower and "<!doctype" not in lower:
|
|
271
|
-
return False, "not a complete HTML document (missing <!DOCTYPE html>/<html>)"
|
|
272
|
-
if "</html>" not in lower:
|
|
273
|
-
return False, "HTML document is truncated (missing </html>)"
|
|
274
|
-
# A document that merely *contains* html somewhere is not a valid
|
|
275
|
-
# file payload: fenced/chat-wrapped replies must fail here so the
|
|
276
|
-
# extraction pass gets a chance to slice out the real document.
|
|
277
|
-
stripped = content.lstrip()
|
|
278
|
-
if not (stripped.lower().startswith("<!doctype") or stripped.lower().startswith("<html")):
|
|
279
|
-
return False, "HTML document is wrapped in prose or fences"
|
|
280
|
-
if "```" in content:
|
|
281
|
-
return False, "output still contains Markdown fences"
|
|
282
|
-
return True, "ok"
|
|
283
|
-
if ext == ".json":
|
|
284
|
-
try:
|
|
285
|
-
json.loads(content)
|
|
286
|
-
except (ValueError, TypeError) as exc:
|
|
287
|
-
return False, f"invalid JSON: {exc}"
|
|
288
|
-
return True, "ok"
|
|
289
|
-
if ext == ".css":
|
|
290
|
-
if "```" in content:
|
|
291
|
-
return False, "output still contains Markdown fences"
|
|
292
|
-
if "{" not in content or "}" not in content:
|
|
293
|
-
return False, "no CSS rule blocks found"
|
|
294
|
-
return True, "ok"
|
|
295
|
-
if ext == ".py":
|
|
296
|
-
if "```" in content:
|
|
297
|
-
return False, "output still contains Markdown fences"
|
|
298
|
-
try:
|
|
299
|
-
ast.parse(content)
|
|
300
|
-
except SyntaxError as exc:
|
|
301
|
-
return False, f"invalid Python syntax: {exc.msg} (line {exc.lineno})"
|
|
302
|
-
return True, "ok"
|
|
303
|
-
if ext in _BRACED_CODE_EXTENSIONS:
|
|
304
|
-
if "```" in content:
|
|
305
|
-
return False, "output still contains Markdown fences"
|
|
306
|
-
return _check_balanced_delimiters(content)
|
|
307
|
-
if ext in _COMPONENT_EXTENSIONS:
|
|
308
|
-
if "```" in content:
|
|
309
|
-
return False, "output still contains Markdown fences"
|
|
310
|
-
return _check_component_blocks(content)
|
|
311
|
-
if ext in (".sh", ".sql"):
|
|
312
|
-
if "```" in content:
|
|
313
|
-
return False, "output still contains Markdown fences"
|
|
314
|
-
return True, "ok"
|
|
315
|
-
# Prose types (.md, .txt, .csv, …) have no grammar to check, which used to
|
|
316
|
-
# mean *nothing* was checked: a 1–4B model that answered "Sure! Here is the
|
|
317
|
-
# document you asked for:" and stopped had its sentence saved as the file,
|
|
318
|
-
# because the fence stripper only removes conversational lines it can
|
|
319
|
-
# recognise and the length guard on `looks_like_refusal` lets a wordy
|
|
320
|
-
# refusal through. The two checks below are the only ones that generalise
|
|
321
|
-
# without inventing a grammar: it must not still be wearing fences, and it
|
|
322
|
-
# must not be *only* an answer about the file.
|
|
323
|
-
if "```" in content:
|
|
324
|
-
return False, "output still contains Markdown fences"
|
|
325
|
-
if _looks_like_commentary(content):
|
|
326
|
-
return False, "the reply talks about the file instead of being the file"
|
|
327
|
-
return True, "ok"
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
# Openers that mean "I am about to give you the thing" — if the whole reply is
|
|
331
|
-
# one of these, the thing never arrived.
|
|
332
|
-
_COMMENTARY_RE = re.compile(
|
|
333
|
-
r"^\s*("
|
|
334
|
-
r"(sure|of course|certainly|okay|ok|alright|here|below|the following)\b"
|
|
335
|
-
r"|i('ve| have| will|'ll)\b"
|
|
336
|
-
r"|(물론|네[,!. ]|알겠|다음은|아래(는|의)?|요청하신|원하시는)"
|
|
337
|
-
r")",
|
|
338
|
-
re.IGNORECASE,
|
|
339
|
-
)
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
def _looks_like_commentary(content: str) -> bool:
|
|
343
|
-
"""True when the reply reads as an answer *about* a file, not the file.
|
|
344
|
-
|
|
345
|
-
Deliberately conservative — a long document that merely opens with "The
|
|
346
|
-
following" is a document. Only a short reply that both opens
|
|
347
|
-
conversationally and never grows into content is rejected, so a real file
|
|
348
|
-
is never thrown away to catch a chat line.
|
|
349
|
-
"""
|
|
350
|
-
stripped = content.strip()
|
|
351
|
-
if len(stripped) > 400:
|
|
352
|
-
return False
|
|
353
|
-
if not _COMMENTARY_RE.match(stripped):
|
|
354
|
-
return False
|
|
355
|
-
# Structure means content arrived after the preamble: a heading, a list, a
|
|
356
|
-
# table row, a delimiter, or simply several lines of body text.
|
|
357
|
-
body = stripped.split("\n", 1)[1].strip() if "\n" in stripped else ""
|
|
358
|
-
if re.search(r"^\s*(#{1,6}\s|[-*+]\s|\d+[.)]\s|\||>)", body, re.MULTILINE):
|
|
359
|
-
return False
|
|
360
|
-
return len(body) < 120
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
# ── prompting ───────────────────────────────────────────────────────────
|
|
364
|
-
|
|
365
|
-
_FIRST_LINE_HINTS: Dict[str, str] = {
|
|
366
|
-
".html": "<!DOCTYPE html>",
|
|
367
|
-
".htm": "<!DOCTYPE html>",
|
|
368
|
-
".py": "# (python code — imports or code on the first line)",
|
|
369
|
-
".sh": "#!/bin/sh",
|
|
370
|
-
".json": "{",
|
|
371
|
-
".xml": "<?xml version=\"1.0\" encoding=\"UTF-8\"?>",
|
|
372
|
-
}
|
|
373
|
-
|
|
374
|
-
_TYPE_RULES: Dict[str, str] = {
|
|
375
|
-
".html": (
|
|
376
|
-
"Produce ONE complete standalone HTML5 document: <!DOCTYPE html>, <html>, "
|
|
377
|
-
"<head> with <meta charset=\"utf-8\"> and a <title>, inline <style> for CSS, "
|
|
378
|
-
"and a closed </html> tag. Do not reference external files."
|
|
379
|
-
),
|
|
380
|
-
".htm": (
|
|
381
|
-
"Produce ONE complete standalone HTML5 document ending with </html>."
|
|
382
|
-
),
|
|
383
|
-
".json": "Produce strictly valid JSON (double quotes, no comments, no trailing commas).",
|
|
384
|
-
".css": "Produce valid CSS rules only.",
|
|
385
|
-
".md": "Produce well-structured Markdown with headings.",
|
|
386
|
-
".markdown": "Produce well-structured Markdown with headings.",
|
|
387
|
-
".csv": "Produce CSV with a header row; comma-separated, one record per line.",
|
|
388
|
-
".py": "Produce complete runnable Python source code.",
|
|
389
|
-
".js": "Produce complete valid JavaScript source code.",
|
|
390
|
-
".jsx": "Produce one complete React component file in JSX.",
|
|
391
|
-
".ts": "Produce complete valid TypeScript source code.",
|
|
392
|
-
".tsx": "Produce one complete React component file in TSX (TypeScript).",
|
|
393
|
-
".vue": "Produce ONE complete Vue single-file component with closed <template>/<script>/<style> blocks.",
|
|
394
|
-
".svelte": "Produce ONE complete Svelte component; every <script>/<style> block must be closed.",
|
|
395
|
-
}
|
|
396
|
-
|
|
397
|
-
# Multi-file bundles override the standalone-HTML rule: the page must link
|
|
398
|
-
# its sibling files instead of inlining everything.
|
|
399
|
-
_BUNDLE_HTML_RULE = (
|
|
400
|
-
"Produce ONE complete HTML5 document: <!DOCTYPE html>, <html>, <head> with "
|
|
401
|
-
"<meta charset=\"utf-8\"> and a <title>, and a closed </html> tag. "
|
|
402
|
-
"This page is part of a multi-file project: link the project stylesheet(s) "
|
|
403
|
-
"with <link rel=\"stylesheet\" href=\"...\"> and load the project script(s) "
|
|
404
|
-
"with <script src=\"...\"></script> just before </body>. Reference ONLY the "
|
|
405
|
-
"project files listed below — no other external files, no inline <style> "
|
|
406
|
-
"blocks, no inline behavior scripts."
|
|
407
|
-
)
|
|
408
|
-
|
|
409
|
-
# Vite/React bundles need a module entry point, not classic script tags.
|
|
410
|
-
_BUNDLE_HTML_MODULE_RULE = (
|
|
411
|
-
"Produce ONE complete HTML5 document: <!DOCTYPE html>, <html>, <head> with "
|
|
412
|
-
"<meta charset=\"utf-8\"> and a <title>, and a closed </html> tag. "
|
|
413
|
-
"This page is the Vite entry of a React project: the <body> must contain "
|
|
414
|
-
"<div id=\"root\"></div> and load the app with "
|
|
415
|
-
"<script type=\"module\" src=\"/src/main.jsx\"></script> just before "
|
|
416
|
-
"</body>. No inline <style> blocks, no other scripts, no external files."
|
|
417
|
-
)
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
def _bundle_html_rule(bundle_files: List[str]) -> str:
|
|
421
|
-
"""Pick the HTML bundle rule that matches the bundle's technology."""
|
|
422
|
-
if any(str(path).lower().endswith((".jsx", ".tsx")) for path in bundle_files):
|
|
423
|
-
return _BUNDLE_HTML_MODULE_RULE
|
|
424
|
-
return _BUNDLE_HTML_RULE
|
|
425
|
-
|
|
426
|
-
|
|
427
|
-
def build_file_generation_context(
|
|
428
|
-
target_path: str,
|
|
429
|
-
user_request: str,
|
|
430
|
-
feedback: Optional[str] = None,
|
|
431
|
-
bundle_files: Optional[List[str]] = None,
|
|
432
|
-
) -> str:
|
|
433
|
-
"""Strict, extension-aware generation instructions.
|
|
434
|
-
|
|
435
|
-
Small models ignore abstract rules but reliably imitate concrete anchors,
|
|
436
|
-
so the prompt pins the exact first line of the expected output.
|
|
437
|
-
"""
|
|
438
|
-
ext = _ext(target_path)
|
|
439
|
-
parts = [
|
|
440
|
-
"You are a file content generator. Your entire reply is saved verbatim "
|
|
441
|
-
f"as the file `{target_path}` — it is NOT shown in a chat.",
|
|
442
|
-
"Rules:",
|
|
443
|
-
"- Output ONLY the raw file content.",
|
|
444
|
-
"- No Markdown code fences (```), no explanations, no greetings, "
|
|
445
|
-
"no text before or after the content.",
|
|
446
|
-
]
|
|
447
|
-
type_rule = _TYPE_RULES.get(ext)
|
|
448
|
-
if bundle_files and ext in (".html", ".htm"):
|
|
449
|
-
type_rule = _bundle_html_rule(bundle_files)
|
|
450
|
-
if type_rule:
|
|
451
|
-
parts.append(f"- {type_rule}")
|
|
452
|
-
if bundle_files:
|
|
453
|
-
listed = ", ".join(bundle_files)
|
|
454
|
-
parts.append(f"- Project files in this bundle: {listed}")
|
|
455
|
-
first_line = _FIRST_LINE_HINTS.get(ext)
|
|
456
|
-
if first_line:
|
|
457
|
-
parts.append(f"- The very first line of your reply must be: {first_line}")
|
|
458
|
-
if feedback:
|
|
459
|
-
parts.append(
|
|
460
|
-
"Your previous attempt was rejected: "
|
|
461
|
-
f"{feedback}. Fix that and output only the corrected file content."
|
|
462
|
-
)
|
|
463
|
-
parts.append(f"\nUser request: {user_request}")
|
|
464
|
-
return "\n".join(parts)
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
# ── repair (deterministic fallback) ─────────────────────────────────────
|
|
468
|
-
|
|
469
|
-
def repair_file_content(content: str, target_path: str, user_request: str) -> str:
|
|
470
|
-
"""Turn whatever the model produced into a valid file of the target type.
|
|
471
|
-
|
|
472
|
-
This is the last resort after retries: the user asked for a file, so the
|
|
473
|
-
request must still end in a well-formed file, never an error.
|
|
474
|
-
"""
|
|
475
|
-
ext = _ext(target_path)
|
|
476
|
-
salvage = content.strip()
|
|
477
|
-
if looks_like_refusal(salvage):
|
|
478
|
-
salvage = ""
|
|
479
|
-
|
|
480
|
-
if ext in (".html", ".htm"):
|
|
481
|
-
return _repair_html(salvage, user_request)
|
|
482
|
-
if ext == ".json":
|
|
483
|
-
sliced = _slice_json_document(salvage)
|
|
484
|
-
if sliced is not None:
|
|
485
|
-
return sliced
|
|
486
|
-
return json.dumps(
|
|
487
|
-
{"request": user_request, "content": salvage},
|
|
488
|
-
ensure_ascii=False,
|
|
489
|
-
indent=2,
|
|
490
|
-
)
|
|
491
|
-
if ext == ".py" and salvage:
|
|
492
|
-
# The repair guarantee for Python is parseability: unparseable output
|
|
493
|
-
# is preserved honestly as a commented-out draft, never as a broken
|
|
494
|
-
# module the user has to debug.
|
|
495
|
-
try:
|
|
496
|
-
ast.parse(salvage)
|
|
497
|
-
return salvage
|
|
498
|
-
except SyntaxError:
|
|
499
|
-
commented = "\n".join(f"# {line}" for line in salvage.splitlines())
|
|
500
|
-
return (
|
|
501
|
-
f"# TODO: model produced invalid Python for: {user_request}\n"
|
|
502
|
-
"# The draft below is preserved as comments — fix and uncomment.\n"
|
|
503
|
-
f"{commented}\n"
|
|
504
|
-
)
|
|
505
|
-
if salvage:
|
|
506
|
-
return salvage
|
|
507
|
-
# Nothing usable at all — leave an honest placeholder in the right format.
|
|
508
|
-
comment = {
|
|
509
|
-
".py": "# TODO: model produced no usable content for: ",
|
|
510
|
-
".js": "// TODO: model produced no usable content for: ",
|
|
511
|
-
".jsx": "// TODO: model produced no usable content for: ",
|
|
512
|
-
".ts": "// TODO: model produced no usable content for: ",
|
|
513
|
-
".tsx": "// TODO: model produced no usable content for: ",
|
|
514
|
-
".css": "/* TODO: model produced no usable content for: ",
|
|
515
|
-
".sh": "# TODO: model produced no usable content for: ",
|
|
516
|
-
".sql": "-- TODO: model produced no usable content for: ",
|
|
517
|
-
}.get(ext, "")
|
|
518
|
-
if ext == ".css":
|
|
519
|
-
return f"{comment}{user_request} */\n"
|
|
520
|
-
if comment:
|
|
521
|
-
return f"{comment}{user_request}\n"
|
|
522
|
-
return f"{user_request}\n"
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
def _repair_html(salvage: str, user_request: str) -> str:
|
|
526
|
-
lower = salvage.lower()
|
|
527
|
-
if "<html" in lower or "<!doctype" in lower:
|
|
528
|
-
# A real document that is merely truncated — close it.
|
|
529
|
-
doc = _slice_html_document(salvage)
|
|
530
|
-
low = doc.lower()
|
|
531
|
-
if "</body>" not in low and "<body" in low:
|
|
532
|
-
doc += "\n</body>"
|
|
533
|
-
if "</html>" not in low:
|
|
534
|
-
doc += "\n</html>"
|
|
535
|
-
return doc
|
|
536
|
-
if re.search(r"<\w+[^>]*>", salvage):
|
|
537
|
-
body = salvage # an HTML fragment — embed as-is
|
|
538
|
-
elif salvage:
|
|
539
|
-
body = "\n".join(
|
|
540
|
-
f" <p>{html_lib.escape(line)}</p>"
|
|
541
|
-
for line in salvage.splitlines()
|
|
542
|
-
if line.strip()
|
|
543
|
-
)
|
|
544
|
-
else:
|
|
545
|
-
body = f" <p>{html_lib.escape(user_request)}</p>"
|
|
546
|
-
title = html_lib.escape(user_request[:60] or "Generated page")
|
|
547
|
-
return (
|
|
548
|
-
"<!DOCTYPE html>\n"
|
|
549
|
-
"<html lang=\"ko\">\n"
|
|
550
|
-
"<head>\n"
|
|
551
|
-
" <meta charset=\"utf-8\">\n"
|
|
552
|
-
" <meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n"
|
|
553
|
-
f" <title>{title}</title>\n"
|
|
554
|
-
" <style>\n"
|
|
555
|
-
" body { font-family: system-ui, sans-serif; margin: 2rem auto; "
|
|
556
|
-
"max-width: 720px; line-height: 1.6; padding: 0 1rem; }\n"
|
|
557
|
-
" </style>\n"
|
|
558
|
-
"</head>\n"
|
|
559
|
-
"<body>\n"
|
|
560
|
-
f"{body}\n"
|
|
561
|
-
"</body>\n"
|
|
562
|
-
"</html>"
|
|
563
|
-
)
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
# Extensions the Brain UI can render inline (preview) after creation.
|
|
567
|
-
PREVIEWABLE_EXTENSIONS = frozenset({
|
|
568
|
-
".html", ".htm", ".md", ".markdown", ".txt", ".json", ".css", ".js",
|
|
569
|
-
".csv", ".py", ".yaml", ".yml", ".xml", ".sql", ".sh",
|
|
570
|
-
".jsx", ".ts", ".tsx", ".vue", ".svelte",
|
|
571
|
-
})
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
# ── write-side sanitize (ArtifactWritePipeline) ─────────────────────────
|
|
575
|
-
|
|
576
|
-
def sanitize_write_content(
|
|
577
|
-
target_path: str,
|
|
578
|
-
content: Any,
|
|
579
|
-
user_request: str = "",
|
|
580
|
-
) -> Tuple[str, Dict[str, Any]]:
|
|
581
|
-
"""Single write-side guarantee for model-produced file content.
|
|
582
|
-
|
|
583
|
-
The direct chat path already runs the full generate→validate→repair
|
|
584
|
-
pipeline, but the agent JSON loop historically wrote ``args.content``
|
|
585
|
-
verbatim — weak models routinely put fenced/chatty payloads there. This
|
|
586
|
-
conservative sanitizer closes that gap for *any* write entry point:
|
|
587
|
-
|
|
588
|
-
1. content that already validates is returned byte-for-byte unchanged
|
|
589
|
-
(trusted/user-authored content is never mangled);
|
|
590
|
-
2. otherwise the extraction pass strips fences/think-blocks/chat noise
|
|
591
|
-
and is used only when the extracted payload validates;
|
|
592
|
-
3. otherwise deterministic repair guarantees a structurally valid file.
|
|
593
|
-
|
|
594
|
-
Empty content is left untouched (creating an empty file is a legitimate,
|
|
595
|
-
intentional action — e.g. ``__init__.py``). Returns ``(content, meta)``
|
|
596
|
-
where meta is ``{"sanitized": bool, "repaired": bool, "reason": str}``.
|
|
597
|
-
"""
|
|
598
|
-
raw = str(content or "")
|
|
599
|
-
if not raw.strip():
|
|
600
|
-
return raw, {"sanitized": False, "repaired": False, "reason": "empty"}
|
|
601
|
-
ok, reason = validate_file_content(raw, target_path)
|
|
602
|
-
if ok:
|
|
603
|
-
return raw, {"sanitized": False, "repaired": False, "reason": "ok"}
|
|
604
|
-
extracted = extract_file_content(raw, target_path)
|
|
605
|
-
if extracted:
|
|
606
|
-
extracted_ok, _ = validate_file_content(extracted, target_path)
|
|
607
|
-
if extracted_ok:
|
|
608
|
-
return extracted, {"sanitized": True, "repaired": False, "reason": reason}
|
|
609
|
-
repaired = repair_file_content(
|
|
610
|
-
extracted or raw, target_path, user_request or f"content for {target_path}"
|
|
611
|
-
)
|
|
612
|
-
return repaired, {"sanitized": True, "repaired": True, "reason": reason}
|
|
613
|
-
|
|
614
|
-
|
|
615
|
-
# ── filename inference ──────────────────────────────────────────────────
|
|
616
|
-
|
|
617
|
-
_CREATE_VERB_RE = re.compile(
|
|
618
|
-
r"(만들|생성|작성|써\s*줘|저장|create|make|write|generate|build|save)",
|
|
619
|
-
re.IGNORECASE,
|
|
620
|
-
)
|
|
621
|
-
|
|
622
|
-
# Explicit type keyword → default filename. Ordered: first match wins.
|
|
623
|
-
_TYPE_KEYWORDS: Tuple[Tuple[str, str], ...] = (
|
|
624
|
-
(r"\bhtml\b|웹\s*페이지|웹페이지|홈페이지|landing\s*page|web\s*page", "generated_page.html"),
|
|
625
|
-
(r"\bcss\b|스타일\s*시트", "styles.css"),
|
|
626
|
-
(r"\bjavascript\b|\bjs\b\s*(파일|file)|자바스크립트", "script.js"),
|
|
627
|
-
(r"\bpython\b|파이썬", "script.py"),
|
|
628
|
-
(r"\bjson\b", "data.json"),
|
|
629
|
-
(r"\bcsv\b", "data.csv"),
|
|
630
|
-
(r"\byaml\b|\byml\b", "config.yaml"),
|
|
631
|
-
(r"\bxml\b", "data.xml"),
|
|
632
|
-
(r"\bsql\b", "query.sql"),
|
|
633
|
-
(r"마크다운|\bmarkdown\b|\bmd\b\s*(파일|file)", "notes.md"),
|
|
634
|
-
(r"텍스트\s*파일|\btext\s*file\b|\btxt\b", "notes.txt"),
|
|
635
|
-
)
|
|
636
|
-
|
|
637
|
-
|
|
638
|
-
def infer_file_target(message: str) -> Optional[str]:
|
|
639
|
-
"""Infer a filename for creation requests that name a type but no path.
|
|
640
|
-
|
|
641
|
-
"html 파일 만들어줘" previously fell through to the agent JSON loop, which
|
|
642
|
-
small models fail at. Inference keeps such requests on the deterministic
|
|
643
|
-
direct-write path. Deliberately narrow: requires a creation verb and an
|
|
644
|
-
explicit file-type keyword — report/document prose requests keep flowing
|
|
645
|
-
to the document generator.
|
|
646
|
-
"""
|
|
647
|
-
text = (message or "").strip()
|
|
648
|
-
if not text or not _CREATE_VERB_RE.search(text):
|
|
649
|
-
return None
|
|
650
|
-
lower = text.lower()
|
|
651
|
-
for pattern, filename in _TYPE_KEYWORDS:
|
|
652
|
-
if re.search(pattern, lower):
|
|
653
|
-
return filename
|
|
654
|
-
return None
|
|
655
|
-
|
|
656
|
-
|
|
657
|
-
# ── project manifest (multi-file bundles) ───────────────────────────────
|
|
658
|
-
|
|
659
|
-
# ``\b`` fails against Korean particles ("js로") because Hangul is ``\w`` —
|
|
660
|
-
# use ASCII lookarounds so type keywords match with or without a particle.
|
|
661
|
-
_HTML_HINT_RE = re.compile(
|
|
662
|
-
r"(?<![a-z0-9])html(?![a-z0-9])"
|
|
663
|
-
r"|웹\s*페이지|웹페이지|홈페이지|웹\s*사이트|웹사이트|website|web\s*page|landing\s*page",
|
|
664
|
-
)
|
|
665
|
-
_CSS_HINT_RE = re.compile(r"(?<![a-z0-9])css(?![a-z0-9])|스타일\s*시트|stylesheet")
|
|
666
|
-
_JS_HINT_RE = re.compile(
|
|
667
|
-
r"(?<![a-z0-9])js(?![a-z0-9])|javascript|자바스크립트|자바\s*스크립트"
|
|
668
|
-
)
|
|
669
|
-
# An explicit filename means the user is managing paths — keep the
|
|
670
|
-
# deterministic single-file flow untouched.
|
|
671
|
-
_EXPLICIT_FILENAME_RE = re.compile(
|
|
672
|
-
r"[\w-]+\.(?:html?|css|js|jsx|ts|tsx|py|json|md|txt|csv|vue|svelte)\b",
|
|
673
|
-
re.IGNORECASE,
|
|
674
|
-
)
|
|
675
|
-
_PROJECT_NAME_RE = re.compile(r"([A-Za-z][A-Za-z0-9_-]{1,30})\s*(?:앱|app\b)", re.IGNORECASE)
|
|
676
|
-
# React/Vite intent: the react keyword is specific enough on its own.
|
|
677
|
-
_REACT_HINT_RE = re.compile(r"(?<![a-z0-9])react(?![a-z0-9])|리액트")
|
|
678
|
-
_VITE_HINT_RE = re.compile(r"(?<![a-z0-9])vite(?![a-z0-9])")
|
|
679
|
-
# Python package intent: language + package word, both required.
|
|
680
|
-
_PYTHON_HINT_RE = re.compile(r"(?<![a-z0-9])python(?![a-z0-9])|파이썬")
|
|
681
|
-
_PACKAGE_HINT_RE = re.compile(r"패키지|(?<![a-z0-9])package(?![a-z0-9])")
|
|
682
|
-
_PKG_NAME_RE = re.compile(
|
|
683
|
-
r"([A-Za-z][A-Za-z0-9_-]{1,30})\s*(?:패키지|package\b)", re.IGNORECASE
|
|
684
|
-
)
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
def _react_manifest(text: str) -> Dict[str, Any]:
|
|
688
|
-
"""Vite + React starter manifest (review Wave 4: manifest 확장)."""
|
|
689
|
-
name_match = _PROJECT_NAME_RE.search(text)
|
|
690
|
-
name = f"{name_match.group(1).lower()}-app" if name_match else "react-app"
|
|
691
|
-
return {
|
|
692
|
-
"name": name,
|
|
693
|
-
"kind": "react",
|
|
694
|
-
"files": [
|
|
695
|
-
{
|
|
696
|
-
"path": "package.json",
|
|
697
|
-
"brief": (
|
|
698
|
-
f'Vite React app manifest: strictly valid JSON with "name": "{name}", '
|
|
699
|
-
'"private": true, "type": "module", "scripts" {"dev": "vite", '
|
|
700
|
-
'"build": "vite build", "preview": "vite preview"}, "dependencies" '
|
|
701
|
-
'with react and react-dom (^18), and "devDependencies" with vite '
|
|
702
|
-
"and @vitejs/plugin-react."
|
|
703
|
-
),
|
|
704
|
-
},
|
|
705
|
-
{
|
|
706
|
-
"path": "index.html",
|
|
707
|
-
"brief": (
|
|
708
|
-
"The Vite entry HTML: <div id=\"root\"></div> in <body> and "
|
|
709
|
-
"<script type=\"module\" src=\"/src/main.jsx\"></script> just "
|
|
710
|
-
"before </body>. No inline styles or scripts."
|
|
711
|
-
),
|
|
712
|
-
},
|
|
713
|
-
{
|
|
714
|
-
"path": "src/main.jsx",
|
|
715
|
-
"brief": (
|
|
716
|
-
"React entry: createRoot from react-dom/client rendering <App /> "
|
|
717
|
-
"into #root; imports ./App.jsx and ./App.css."
|
|
718
|
-
),
|
|
719
|
-
},
|
|
720
|
-
{
|
|
721
|
-
"path": "src/App.jsx",
|
|
722
|
-
"brief": (
|
|
723
|
-
"The main App component implementing the user's request as one "
|
|
724
|
-
"self-contained React component (hooks allowed, no extra deps)."
|
|
725
|
-
),
|
|
726
|
-
},
|
|
727
|
-
{
|
|
728
|
-
"path": "src/App.css",
|
|
729
|
-
"brief": "All visual styles for the App component.",
|
|
730
|
-
},
|
|
731
|
-
],
|
|
732
|
-
}
|
|
733
|
-
|
|
734
|
-
|
|
735
|
-
def _python_package_manifest(text: str) -> Dict[str, Any]:
|
|
736
|
-
"""Multi-file Python package manifest (review Wave 4: manifest 확장)."""
|
|
737
|
-
name_match = _PKG_NAME_RE.search(text)
|
|
738
|
-
raw_name = name_match.group(1).lower() if name_match else "my_package"
|
|
739
|
-
module = re.sub(r"[^a-z0-9_]", "_", raw_name)
|
|
740
|
-
if not re.match(r"[a-z_]", module):
|
|
741
|
-
module = f"pkg_{module}"
|
|
742
|
-
return {
|
|
743
|
-
"name": module,
|
|
744
|
-
"kind": "python",
|
|
745
|
-
"files": [
|
|
746
|
-
{
|
|
747
|
-
"path": f"{module}/__init__.py",
|
|
748
|
-
"brief": (
|
|
749
|
-
f"Package init for {module}: import and re-export the public "
|
|
750
|
-
"API from .core with an explicit __all__."
|
|
751
|
-
),
|
|
752
|
-
},
|
|
753
|
-
{
|
|
754
|
-
"path": f"{module}/core.py",
|
|
755
|
-
"brief": (
|
|
756
|
-
"Implement the user's request as clean, documented functions/"
|
|
757
|
-
"classes with type hints. Standard library only."
|
|
758
|
-
),
|
|
759
|
-
},
|
|
760
|
-
{
|
|
761
|
-
"path": f"{module}/cli.py",
|
|
762
|
-
"brief": (
|
|
763
|
-
"argparse CLI wrapping the core API: a main() function and an "
|
|
764
|
-
'if __name__ == "__main__": main() guard.'
|
|
765
|
-
),
|
|
766
|
-
},
|
|
767
|
-
{
|
|
768
|
-
"path": "README.md",
|
|
769
|
-
"brief": (
|
|
770
|
-
f"Usage documentation for the {module} package: install, import "
|
|
771
|
-
"example, and CLI example."
|
|
772
|
-
),
|
|
773
|
-
},
|
|
774
|
-
],
|
|
775
|
-
}
|
|
776
|
-
|
|
777
|
-
|
|
778
|
-
def infer_project_manifest(message: str) -> Optional[Dict[str, Any]]:
|
|
779
|
-
"""Infer a multi-file project manifest from a creation request.
|
|
780
|
-
|
|
781
|
-
"todo 앱 html+css+js로 만들어줘" should yield real linked files, not one
|
|
782
|
-
inlined page. Deliberately narrow and deterministic (weak local models
|
|
783
|
-
never see this decision): requires a creation verb, a recognized project
|
|
784
|
-
intent (web page + css/js, React/Vite app, or Python package), and no
|
|
785
|
-
explicit filename. Single-type requests return ``None`` so the existing
|
|
786
|
-
single-file flow is completely unchanged.
|
|
787
|
-
"""
|
|
788
|
-
text = (message or "").strip()
|
|
789
|
-
if not text or not _CREATE_VERB_RE.search(text):
|
|
790
|
-
return None
|
|
791
|
-
if _EXPLICIT_FILENAME_RE.search(text):
|
|
792
|
-
return None
|
|
793
|
-
lower = text.lower()
|
|
794
|
-
|
|
795
|
-
# Most-specific first: React (its own structure), then Python package,
|
|
796
|
-
# then the classic html+css/js web bundle.
|
|
797
|
-
if _REACT_HINT_RE.search(lower) or _VITE_HINT_RE.search(lower):
|
|
798
|
-
return _react_manifest(text)
|
|
799
|
-
if _PYTHON_HINT_RE.search(lower) and _PACKAGE_HINT_RE.search(lower):
|
|
800
|
-
return _python_package_manifest(text)
|
|
801
|
-
|
|
802
|
-
wants_html = bool(_HTML_HINT_RE.search(lower))
|
|
803
|
-
wants_css = bool(_CSS_HINT_RE.search(lower))
|
|
804
|
-
wants_js = bool(_JS_HINT_RE.search(lower))
|
|
805
|
-
if not wants_html or not (wants_css or wants_js):
|
|
806
|
-
return None
|
|
807
|
-
|
|
808
|
-
name_match = _PROJECT_NAME_RE.search(text)
|
|
809
|
-
name = f"{name_match.group(1).lower()}-app" if name_match else "web-project"
|
|
810
|
-
|
|
811
|
-
files: List[Dict[str, str]] = []
|
|
812
|
-
html_refs: List[str] = []
|
|
813
|
-
if wants_css:
|
|
814
|
-
html_refs.append('<link rel="stylesheet" href="style.css"> in <head>')
|
|
815
|
-
if wants_js:
|
|
816
|
-
html_refs.append('<script src="app.js"></script> just before </body>')
|
|
817
|
-
files.append({
|
|
818
|
-
"path": "index.html",
|
|
819
|
-
"brief": (
|
|
820
|
-
"The main HTML page of the project. Reference the sibling files: "
|
|
821
|
-
+ " and ".join(html_refs)
|
|
822
|
-
+ ". Do not inline styles or behavior scripts."
|
|
823
|
-
),
|
|
824
|
-
})
|
|
825
|
-
if wants_css:
|
|
826
|
-
files.append({
|
|
827
|
-
"path": "style.css",
|
|
828
|
-
"brief": "All visual styles for index.html (layout, colors, typography).",
|
|
829
|
-
})
|
|
830
|
-
if wants_js:
|
|
831
|
-
files.append({
|
|
832
|
-
"path": "app.js",
|
|
833
|
-
"brief": (
|
|
834
|
-
"All page behavior for index.html as plain browser JavaScript "
|
|
835
|
-
"(no build step, no imports of missing files)."
|
|
836
|
-
),
|
|
837
|
-
})
|
|
838
|
-
return {"name": name, "kind": "web", "files": files}
|
|
839
|
-
|
|
840
|
-
|
|
841
|
-
_HTML_LOCAL_REF_RE = re.compile(
|
|
842
|
-
r"(?:href|src)\s*=\s*[\"']([^\"'#?]+)[\"']", re.IGNORECASE
|
|
843
|
-
)
|
|
844
|
-
_EXTERNAL_REF_PREFIXES = ("http://", "https://", "//", "data:", "mailto:", "tel:", "javascript:")
|
|
845
|
-
|
|
846
|
-
|
|
847
|
-
def _local_bundle_refs(html: str) -> List[str]:
|
|
848
|
-
"""File references inside an HTML document that must exist in the bundle."""
|
|
849
|
-
refs: List[str] = []
|
|
850
|
-
for ref in _HTML_LOCAL_REF_RE.findall(html or ""):
|
|
851
|
-
candidate = ref.strip()
|
|
852
|
-
if not candidate or candidate.startswith(_EXTERNAL_REF_PREFIXES):
|
|
853
|
-
continue
|
|
854
|
-
if "." not in candidate.rsplit("/", 1)[-1]:
|
|
855
|
-
continue # anchors / routes, not files
|
|
856
|
-
refs.append(candidate)
|
|
857
|
-
return refs
|
|
858
|
-
|
|
859
|
-
|
|
860
|
-
def repair_bundle_references(files: Dict[str, str]) -> Tuple[Dict[str, str], List[str]]:
|
|
861
|
-
"""Deterministically point dangling HTML refs at real bundle files.
|
|
862
|
-
|
|
863
|
-
A weak model asked for ``style.css`` sometimes links ``styles.css``. When
|
|
864
|
-
a referenced file is missing but the bundle contains exactly one file of
|
|
865
|
-
the same extension, the reference is rewritten. Returns ``(files, fixes)``.
|
|
866
|
-
"""
|
|
867
|
-
names = {p.rsplit("/", 1)[-1] for p in files}
|
|
868
|
-
fixes: List[str] = []
|
|
869
|
-
repaired = dict(files)
|
|
870
|
-
for path, content in files.items():
|
|
871
|
-
if _ext(path) not in (".html", ".htm"):
|
|
872
|
-
continue
|
|
873
|
-
updated = content
|
|
874
|
-
for ref in _local_bundle_refs(content):
|
|
875
|
-
base = ref.rsplit("/", 1)[-1]
|
|
876
|
-
if base in names:
|
|
877
|
-
continue
|
|
878
|
-
same_ext = [n for n in names if _ext(n) == _ext(base)]
|
|
879
|
-
if len(same_ext) == 1:
|
|
880
|
-
updated = updated.replace(ref, same_ext[0])
|
|
881
|
-
fixes.append(f"{path}: '{ref}' -> '{same_ext[0]}'")
|
|
882
|
-
if updated != content:
|
|
883
|
-
repaired[path] = updated
|
|
884
|
-
return repaired, fixes
|
|
885
|
-
|
|
886
|
-
|
|
887
|
-
def validate_project_bundle(files: Dict[str, str]) -> Dict[str, Any]:
|
|
888
|
-
"""Bundle-level verification: every file valid, every HTML ref resolvable."""
|
|
889
|
-
issues: List[str] = []
|
|
890
|
-
per_file: Dict[str, Dict[str, Any]] = {}
|
|
891
|
-
names = {p.rsplit("/", 1)[-1] for p in files}
|
|
892
|
-
for path, content in files.items():
|
|
893
|
-
ok, reason = validate_file_content(content, path)
|
|
894
|
-
per_file[path] = {"valid": ok, "reason": reason}
|
|
895
|
-
if not ok:
|
|
896
|
-
issues.append(f"{path}: {reason}")
|
|
897
|
-
if _ext(path) in (".html", ".htm"):
|
|
898
|
-
for ref in _local_bundle_refs(content):
|
|
899
|
-
if ref.rsplit("/", 1)[-1] not in names:
|
|
900
|
-
issues.append(f"{path}: references missing file '{ref}'")
|
|
901
|
-
return {"ok": not issues, "issues": issues, "files": per_file}
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
# ── orchestration ───────────────────────────────────────────────────────
|
|
905
|
-
|
|
906
|
-
async def generate_file_content(
|
|
907
|
-
generate: Callable[[str], Awaitable[Any]],
|
|
908
|
-
*,
|
|
909
|
-
target_path: str,
|
|
910
|
-
user_request: str,
|
|
911
|
-
max_attempts: int = 2,
|
|
912
|
-
bundle_files: Optional[List[str]] = None,
|
|
913
|
-
) -> Tuple[str, Dict[str, Any]]:
|
|
914
|
-
"""Generate validated file content with any LLM.
|
|
915
|
-
|
|
916
|
-
``generate`` is an async callable ``context -> raw model text``. Runs up
|
|
917
|
-
to ``max_attempts`` model calls (each retry carrying corrective feedback),
|
|
918
|
-
then falls back to deterministic repair, so the returned content is
|
|
919
|
-
always non-empty and structurally valid for the target type.
|
|
920
|
-
|
|
921
|
-
One extra call beyond ``max_attempts`` is spent — at most once per
|
|
922
|
-
request — when the model has returned a byte-identical rejected reply.
|
|
923
|
-
That is the one case where the ordinary retry is known to be dead on
|
|
924
|
-
arrival: the corrective feedback did not change the reply, so the budget
|
|
925
|
-
is better spent on a prompt that names the repetition than on a third
|
|
926
|
-
identical round trip. Small local models hit this constantly; large ones
|
|
927
|
-
never do, so the extra call is not charged to models that do not need it.
|
|
928
|
-
"""
|
|
929
|
-
attempts: List[Dict[str, Any]] = []
|
|
930
|
-
feedback: Optional[str] = None
|
|
931
|
-
best_candidate = ""
|
|
932
|
-
best_score = (-1, -1)
|
|
933
|
-
seen: set[str] = set()
|
|
934
|
-
escalations_left = 1
|
|
935
|
-
attempt = 0
|
|
936
|
-
budget = max_attempts
|
|
937
|
-
while attempt < budget:
|
|
938
|
-
attempt += 1
|
|
939
|
-
context = build_file_generation_context(
|
|
940
|
-
target_path, user_request, feedback=feedback, bundle_files=bundle_files,
|
|
941
|
-
)
|
|
942
|
-
try:
|
|
943
|
-
raw = await generate(context)
|
|
944
|
-
except Exception as exc: # model backend hiccup — repair still delivers
|
|
945
|
-
attempts.append({"attempt": attempt, "valid": False, "reason": f"generation error: {exc}"})
|
|
946
|
-
feedback = "the model call failed"
|
|
947
|
-
continue
|
|
948
|
-
candidate = extract_file_content(str(raw or ""), target_path)
|
|
949
|
-
ok, reason = validate_file_content(candidate, target_path)
|
|
950
|
-
record: Dict[str, Any] = {"attempt": attempt, "valid": ok, "reason": reason}
|
|
951
|
-
if ok:
|
|
952
|
-
attempts.append(record)
|
|
953
|
-
return candidate, {"attempts": attempts, "repaired": False}
|
|
954
|
-
|
|
955
|
-
# A small model handed the same corrective feedback often replays the
|
|
956
|
-
# same reply verbatim. Saying "you sent this before" is the only signal
|
|
957
|
-
# left that has any chance of moving it, and it makes the wasted retry
|
|
958
|
-
# visible in the trace instead of looking like two genuine tries.
|
|
959
|
-
fingerprint = candidate.strip()
|
|
960
|
-
repeated = fingerprint in seen and bool(fingerprint)
|
|
961
|
-
record["repeated"] = repeated
|
|
962
|
-
seen.add(fingerprint)
|
|
963
|
-
if repeated and escalations_left and attempt >= budget:
|
|
964
|
-
# The retry budget is exhausted and the last thing it bought was a
|
|
965
|
-
# duplicate. Buy one more, but only with a prompt that says so.
|
|
966
|
-
escalations_left -= 1
|
|
967
|
-
budget += 1
|
|
968
|
-
record["escalated"] = True
|
|
969
|
-
attempts.append(record)
|
|
970
|
-
|
|
971
|
-
# Keep the candidate that is *closest to a file*, not the longest one.
|
|
972
|
-
# Longest-wins handed repair a 900-character apology in preference to a
|
|
973
|
-
# 300-character HTML document that only needed its </html> closing —
|
|
974
|
-
# and repair can finish the document but can only bury the apology.
|
|
975
|
-
score = _salvage_score(candidate, target_path)
|
|
976
|
-
if score > best_score:
|
|
977
|
-
best_score, best_candidate = score, candidate
|
|
978
|
-
|
|
979
|
-
feedback = (
|
|
980
|
-
f"{reason}. You already sent exactly this reply and it was rejected "
|
|
981
|
-
"for the same reason — do not repeat it. Output the file itself, "
|
|
982
|
-
"starting at its first character."
|
|
983
|
-
if repeated else reason
|
|
984
|
-
)
|
|
985
|
-
repaired = repair_file_content(best_candidate, target_path, user_request)
|
|
986
|
-
return repaired, {"attempts": attempts, "repaired": True}
|
|
987
|
-
|
|
988
|
-
|
|
989
|
-
def _salvage_score(candidate: str, target_path: str) -> Tuple[int, int]:
|
|
990
|
-
"""How useful an invalid candidate is as raw material for repair.
|
|
991
|
-
|
|
992
|
-
``(tier, length)`` — tier first, so a short real document always beats a
|
|
993
|
-
long non-document; length breaks ties within a tier.
|
|
994
|
-
|
|
995
|
-
Tier 2 something of the right shape that repair can finish (an HTML
|
|
996
|
-
document missing its close tag, parseable-ish JSON, Python that
|
|
997
|
-
at least tokenises).
|
|
998
|
-
Tier 1 ordinary text: no structure, but the words may be the content.
|
|
999
|
-
Tier 0 a refusal — repair should prefer literally anything else, because
|
|
1000
|
-
an apology written into the file is worse than an empty stub.
|
|
1001
|
-
"""
|
|
1002
|
-
text = candidate.strip()
|
|
1003
|
-
if not text:
|
|
1004
|
-
return (0, 0)
|
|
1005
|
-
if looks_like_refusal(text):
|
|
1006
|
-
return (0, len(text))
|
|
1007
|
-
|
|
1008
|
-
ext = _ext(target_path)
|
|
1009
|
-
lower = text.lower()
|
|
1010
|
-
if ext in (".html", ".htm"):
|
|
1011
|
-
if lower.startswith("<!doctype") or lower.startswith("<html"):
|
|
1012
|
-
return (2, len(text))
|
|
1013
|
-
elif ext == ".json":
|
|
1014
|
-
if _slice_json_document(text) is not None:
|
|
1015
|
-
return (2, len(text))
|
|
1016
|
-
elif ext == ".py":
|
|
1017
|
-
try:
|
|
1018
|
-
ast.parse(text)
|
|
1019
|
-
except SyntaxError:
|
|
1020
|
-
pass
|
|
1021
|
-
else:
|
|
1022
|
-
return (2, len(text))
|
|
1023
|
-
elif ext in _BRACED_CODE_EXTENSIONS:
|
|
1024
|
-
if _check_balanced_delimiters(text)[0]:
|
|
1025
|
-
return (2, len(text))
|
|
1026
|
-
elif ext in _COMPONENT_EXTENSIONS:
|
|
1027
|
-
if _check_component_blocks(text)[0]:
|
|
1028
|
-
return (2, len(text))
|
|
1029
|
-
elif ext == ".css" and "{" in text and "}" in text:
|
|
1030
|
-
return (2, len(text))
|
|
1031
|
-
return (1, len(text))
|
|
1032
|
-
|
|
1033
|
-
|
|
1034
|
-
__all__ = [
|
|
1035
|
-
"PREVIEWABLE_EXTENSIONS",
|
|
1036
|
-
"build_file_generation_context",
|
|
1037
|
-
"extract_file_content",
|
|
1038
|
-
"generate_file_content",
|
|
1039
|
-
"infer_file_target",
|
|
1040
|
-
"infer_project_manifest",
|
|
1041
|
-
"looks_like_refusal",
|
|
1042
|
-
"repair_bundle_references",
|
|
1043
|
-
"repair_file_content",
|
|
1044
|
-
"sanitize_write_content",
|
|
1045
|
-
"validate_file_content",
|
|
1046
|
-
"validate_project_bundle",
|
|
1047
|
-
]
|