cct-cli 0.7.9.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- calc_terminal/__init__.py +14 -0
- calc_terminal/__main__.py +14 -0
- calc_terminal/activity.py +1334 -0
- calc_terminal/agent.py +3387 -0
- calc_terminal/agent_runtime.py +519 -0
- calc_terminal/ai_context.py +447 -0
- calc_terminal/ai_modes.py +752 -0
- calc_terminal/ai_personalization.py +286 -0
- calc_terminal/ai_preview_feedback.py +213 -0
- calc_terminal/aicore.py +2572 -0
- calc_terminal/anim.py +367 -0
- calc_terminal/app.py +3685 -0
- calc_terminal/art.py +639 -0
- calc_terminal/atomsim.py +368 -0
- calc_terminal/attachments.py +743 -0
- calc_terminal/benchmark_system.py +414 -0
- calc_terminal/browser/__init__.py +36 -0
- calc_terminal/browser/browser_state.py +346 -0
- calc_terminal/browser/devserver.py +176 -0
- calc_terminal/browser/engine.py +494 -0
- calc_terminal/browser/navigation.py +84 -0
- calc_terminal/browser/preview.py +429 -0
- calc_terminal/browser/preview_entry.py +95 -0
- calc_terminal/browser/project_detector.py +144 -0
- calc_terminal/browser/server.py +449 -0
- calc_terminal/browser/state.py +75 -0
- calc_terminal/browser/watcher.py +99 -0
- calc_terminal/browser_gui/__init__.py +1 -0
- calc_terminal/browser_gui/__main__.py +3 -0
- calc_terminal/browser_gui/launcher.py +173 -0
- calc_terminal/browser_gui/playwright_browser.py +117 -0
- calc_terminal/browser_gui/qt_browser.py +1501 -0
- calc_terminal/browser_gui/webview_browser.py +57 -0
- calc_terminal/capabilities/__init__.py +35 -0
- calc_terminal/capabilities/adapters/__init__.py +33 -0
- calc_terminal/capabilities/adapters/bioinformatics.py +204 -0
- calc_terminal/capabilities/adapters/browser_adapter.py +205 -0
- calc_terminal/capabilities/adapters/filesystem.py +206 -0
- calc_terminal/capabilities/adapters/git_adapter.py +202 -0
- calc_terminal/capabilities/adapters/jupyter_adapter.py +138 -0
- calc_terminal/capabilities/adapters/ml_frameworks.py +158 -0
- calc_terminal/capabilities/adapters/platforms.py +200 -0
- calc_terminal/capabilities/adapters/python_exec.py +93 -0
- calc_terminal/capabilities/adapters/quantum_adapter.py +150 -0
- calc_terminal/capabilities/adapters/scientific_comp.py +123 -0
- calc_terminal/capabilities/adapters/structural_bio.py +161 -0
- calc_terminal/capabilities/adapters/terminal.py +99 -0
- calc_terminal/capabilities/bus.py +178 -0
- calc_terminal/capabilities/discovery.py +207 -0
- calc_terminal/capabilities/schema.py +221 -0
- calc_terminal/cat.ico +0 -0
- calc_terminal/cat_browser.py +2018 -0
- calc_terminal/chat_store.py +703 -0
- calc_terminal/cli.py +1178 -0
- calc_terminal/code_editor.py +640 -0
- calc_terminal/collaboration.py +723 -0
- calc_terminal/commands_data.py +139 -0
- calc_terminal/compatibility_engine.py +352 -0
- calc_terminal/compute/__init__.py +31 -0
- calc_terminal/compute/fabric.py +350 -0
- calc_terminal/config.py +227 -0
- calc_terminal/core/__init__.py +41 -0
- calc_terminal/core/checkpoint.py +156 -0
- calc_terminal/core/input/__init__.py +45 -0
- calc_terminal/core/mode_registry.py +300 -0
- calc_terminal/core/project_graph.py +172 -0
- calc_terminal/core/recovery.py +129 -0
- calc_terminal/core/security_layer.py +112 -0
- calc_terminal/core/task_graph.py +202 -0
- calc_terminal/core/unified_runtime.py +184 -0
- calc_terminal/core/verification.py +257 -0
- calc_terminal/customization.py +1566 -0
- calc_terminal/derivations.py +153 -0
- calc_terminal/device_control.py +263 -0
- calc_terminal/diagnostics/__init__.py +27 -0
- calc_terminal/diagnostics/doctor_engine.py +382 -0
- calc_terminal/diagnostics/self_test.py +247 -0
- calc_terminal/doctor.py +519 -0
- calc_terminal/easter_eggs.py +274 -0
- calc_terminal/editor/__init__.py +1 -0
- calc_terminal/editor/actions.py +263 -0
- calc_terminal/editor/commands.py +160 -0
- calc_terminal/editor/shortcuts.py +226 -0
- calc_terminal/engine.py +259 -0
- calc_terminal/errors.py +120 -0
- calc_terminal/event_stream.py +146 -0
- calc_terminal/eventbus.py +133 -0
- calc_terminal/extensions.py +733 -0
- calc_terminal/fallback_cli.py +1321 -0
- calc_terminal/first_run.py +265 -0
- calc_terminal/fomoji_auth.py +1043 -0
- calc_terminal/formulas.py +82 -0
- calc_terminal/fs_cache.py +121 -0
- calc_terminal/fs_watcher.py +277 -0
- calc_terminal/game.py +193 -0
- calc_terminal/gen1.py +5 -0
- calc_terminal/generators.py +245 -0
- calc_terminal/gestures/__init__.py +42 -0
- calc_terminal/gestures/bindings.py +175 -0
- calc_terminal/gestures/manager.py +477 -0
- calc_terminal/goodbye.py +363 -0
- calc_terminal/gpu3d.py +290 -0
- calc_terminal/graphs.py +358 -0
- calc_terminal/hardware_analyzer.py +440 -0
- calc_terminal/host/__init__.py +30 -0
- calc_terminal/host/browser_manager.py +187 -0
- calc_terminal/host/desktop.py +1386 -0
- calc_terminal/host/launcher.py +395 -0
- calc_terminal/host/terminal.py +279 -0
- calc_terminal/identity.py +216 -0
- calc_terminal/input/__init__.py +54 -0
- calc_terminal/input/capabilities.py +258 -0
- calc_terminal/input/focus.py +87 -0
- calc_terminal/input/gestures.py +64 -0
- calc_terminal/input/pointer.py +114 -0
- calc_terminal/input/touch.py +345 -0
- calc_terminal/keys.py +84 -0
- calc_terminal/live_automation.py +165 -0
- calc_terminal/mathtext.py +433 -0
- calc_terminal/mcp.py +386 -0
- calc_terminal/memory.py +337 -0
- calc_terminal/memory_v2.py +479 -0
- calc_terminal/metrics.py +333 -0
- calc_terminal/mode_detection.py +146 -0
- calc_terminal/model.py +2431 -0
- calc_terminal/model_router.py +665 -0
- calc_terminal/models/__init__.py +0 -0
- calc_terminal/models/active_state.py +187 -0
- calc_terminal/models/dynamic_registry.py +584 -0
- calc_terminal/models/manager.py +781 -0
- calc_terminal/models/model_metadata.json +3526 -0
- calc_terminal/models/profiles.py +194 -0
- calc_terminal/models/registry.py +265 -0
- calc_terminal/models/schema.py +197 -0
- calc_terminal/models/validator.py +287 -0
- calc_terminal/models/verification_engine.py +368 -0
- calc_terminal/native_picker.py +215 -0
- calc_terminal/ollama_catalog.py +279 -0
- calc_terminal/ollama_download.py +233 -0
- calc_terminal/orchestrator.py +304 -0
- calc_terminal/package_research.py +322 -0
- calc_terminal/packages.py +1024 -0
- calc_terminal/pc_specs.py +116 -0
- calc_terminal/permissions.py +334 -0
- calc_terminal/pet.py +106 -0
- calc_terminal/pipeline.py +505 -0
- calc_terminal/platform/__init__.py +491 -0
- calc_terminal/platform/desktop.py +491 -0
- calc_terminal/platform/web.py +781 -0
- calc_terminal/preview/__init__.py +1 -0
- calc_terminal/preview/dev_server.py +303 -0
- calc_terminal/preview/diagnostics.py +131 -0
- calc_terminal/preview/live_reload.py +66 -0
- calc_terminal/preview/manager.py +129 -0
- calc_terminal/project_stats.py +209 -0
- calc_terminal/projects.py +328 -0
- calc_terminal/providers/__init__.py +0 -0
- calc_terminal/providers/adapters/__init__.py +80 -0
- calc_terminal/providers/adapters/anthropic_adapter.py +127 -0
- calc_terminal/providers/adapters/base.py +106 -0
- calc_terminal/providers/adapters/chinese_adapters.py +420 -0
- calc_terminal/providers/adapters/gemini_adapter.py +101 -0
- calc_terminal/providers/adapters/ollama_adapter.py +83 -0
- calc_terminal/providers/adapters/openai_adapter.py +159 -0
- calc_terminal/providers/adapters/other_adapters.py +246 -0
- calc_terminal/providers/anthropic_provider.py +172 -0
- calc_terminal/providers/auto_update.py +416 -0
- calc_terminal/providers/base_provider.py +105 -0
- calc_terminal/providers/discovery_manager.py +207 -0
- calc_terminal/providers/gemini_provider.py +178 -0
- calc_terminal/providers/lifecycle.py +767 -0
- calc_terminal/providers/ollama_adapter.py +707 -0
- calc_terminal/providers/openai_provider.py +254 -0
- calc_terminal/providers/provider_manager.py +1827 -0
- calc_terminal/providers/providers.json +4075 -0
- calc_terminal/reactionsim.py +279 -0
- calc_terminal/registry.py +337 -0
- calc_terminal/report.py +162 -0
- calc_terminal/research/__init__.py +45 -0
- calc_terminal/research/artifact_intel.py +126 -0
- calc_terminal/research/data_lineage.py +123 -0
- calc_terminal/research/experiment_ledger.py +303 -0
- calc_terminal/research/reproducibility.py +131 -0
- calc_terminal/resilience/__init__.py +47 -0
- calc_terminal/resilience/agent_state.py +121 -0
- calc_terminal/resilience/capability_matcher.py +174 -0
- calc_terminal/resilience/circuit_breaker.py +158 -0
- calc_terminal/resilience/failover_engine.py +230 -0
- calc_terminal/resilience/health_monitor.py +192 -0
- calc_terminal/resilience/ollama_adapter.py +125 -0
- calc_terminal/resilience/orchestrator.py +312 -0
- calc_terminal/resilience/types.py +134 -0
- calc_terminal/sandbox.py +98 -0
- calc_terminal/scires.py +558 -0
- calc_terminal/security_scanner.py +126 -0
- calc_terminal/session.py +294 -0
- calc_terminal/sim3d.py +206 -0
- calc_terminal/solver.py +276 -0
- calc_terminal/sound.py +127 -0
- calc_terminal/task_reports.py +287 -0
- calc_terminal/terminal_host.py +201 -0
- calc_terminal/terminal_identity.py +411 -0
- calc_terminal/test_ai_mode_reliability.py +344 -0
- calc_terminal/test_browser.py +368 -0
- calc_terminal/test_code_editor_upgrade.py +485 -0
- calc_terminal/test_customization.py +1148 -0
- calc_terminal/test_customization_ui.py +612 -0
- calc_terminal/test_dynamic_registry.py +304 -0
- calc_terminal/test_extensions.py +436 -0
- calc_terminal/test_overhaul.py +557 -0
- calc_terminal/test_project_detect.py +255 -0
- calc_terminal/test_root_cause_fix.py +527 -0
- calc_terminal/test_stability.py +532 -0
- calc_terminal/test_terminal_identity.py +132 -0
- calc_terminal/test_v079_speed.py +460 -0
- calc_terminal/theme.py +1107 -0
- calc_terminal/timeline.py +139 -0
- calc_terminal/todos.py +246 -0
- calc_terminal/tool_call_normalizer.py +419 -0
- calc_terminal/tui.py +104 -0
- calc_terminal/ui/__init__.py +8 -0
- calc_terminal/ui/activity_panel.py +231 -0
- calc_terminal/ui/activity_stream_panel.py +238 -0
- calc_terminal/ui/animations.py +122 -0
- calc_terminal/ui/app.py +7271 -0
- calc_terminal/ui/attach_panel.py +597 -0
- calc_terminal/ui/attachments.py +424 -0
- calc_terminal/ui/backup_panel.py +810 -0
- calc_terminal/ui/browser_shell.py +887 -0
- calc_terminal/ui/cat_agent.py +357 -0
- calc_terminal/ui/chats_panel.py +899 -0
- calc_terminal/ui/command_palette.py +125 -0
- calc_terminal/ui/command_palette_modal.py +166 -0
- calc_terminal/ui/composer.py +1141 -0
- calc_terminal/ui/context_menu.py +197 -0
- calc_terminal/ui/conversation.py +1435 -0
- calc_terminal/ui/customization_panel.py +1229 -0
- calc_terminal/ui/dashboard.py +404 -0
- calc_terminal/ui/design_system.py +557 -0
- calc_terminal/ui/diff_panel.py +213 -0
- calc_terminal/ui/editor.py +2102 -0
- calc_terminal/ui/empty_state.py +302 -0
- calc_terminal/ui/events.py +487 -0
- calc_terminal/ui/extensions_panel.py +815 -0
- calc_terminal/ui/footer.py +166 -0
- calc_terminal/ui/gestures_panel.py +383 -0
- calc_terminal/ui/goodbye_screen.py +100 -0
- calc_terminal/ui/header.py +1034 -0
- calc_terminal/ui/help_panel.py +254 -0
- calc_terminal/ui/live_activities.py +914 -0
- calc_terminal/ui/mcp_panel.py +570 -0
- calc_terminal/ui/memory_center.py +524 -0
- calc_terminal/ui/mode_colors_panel.py +525 -0
- calc_terminal/ui/nav_screens.py +747 -0
- calc_terminal/ui/ollama_panel.py +536 -0
- calc_terminal/ui/palette.py +221 -0
- calc_terminal/ui/permission_panel.py +269 -0
- calc_terminal/ui/personalization_panel.py +517 -0
- calc_terminal/ui/personalize_center.py +1568 -0
- calc_terminal/ui/preview_panel.py +441 -0
- calc_terminal/ui/resizers.py +402 -0
- calc_terminal/ui/sidebar.py +1285 -0
- calc_terminal/ui/statusbar.py +168 -0
- calc_terminal/ui/theme_css.py +1396 -0
- calc_terminal/ui/thinking.py +226 -0
- calc_terminal/ui/timeline_panel.py +102 -0
- calc_terminal/ui/todo_panel.py +193 -0
- calc_terminal/ui/viewport.py +136 -0
- calc_terminal/ui/vision_panel.py +489 -0
- calc_terminal/ui/welcome_modal.py +343 -0
- calc_terminal/ui/widgets.py +160 -0
- calc_terminal/ui/workspace.py +831 -0
- calc_terminal/viewers/__init__.py +1 -0
- calc_terminal/viewers/document_viewer.py +252 -0
- calc_terminal/viewers/image_viewer.py +241 -0
- calc_terminal/viewers/pdf_viewer.py +203 -0
- calc_terminal/viewers/presentation_viewer.py +164 -0
- calc_terminal/viewers/registry.py +120 -0
- calc_terminal/viewers/spreadsheet_viewer.py +204 -0
- calc_terminal/vision/__init__.py +89 -0
- calc_terminal/vision/analysis.py +194 -0
- calc_terminal/vision/annotations.py +297 -0
- calc_terminal/vision/capture.py +171 -0
- calc_terminal/vision/context.py +231 -0
- calc_terminal/vision/correlation.py +169 -0
- calc_terminal/vision/cursor.py +258 -0
- calc_terminal/vision/events.py +66 -0
- calc_terminal/vision/frame_pipeline.py +259 -0
- calc_terminal/vision/priority.py +218 -0
- calc_terminal/vision/provider.py +180 -0
- calc_terminal/vision/safety.py +149 -0
- calc_terminal/vision/session.py +281 -0
- calc_terminal/vision/verify.py +162 -0
- calc_terminal/vision.py +514 -0
- calc_terminal/vscode_integration.py +113 -0
- calc_terminal/web/__init__.py +8 -0
- calc_terminal/web/cat_runtime.py +710 -0
- calc_terminal/web/server.py +2891 -0
- calc_terminal/web/static/css/app.css +3152 -0
- calc_terminal/web/static/icons/badge-72.png +0 -0
- calc_terminal/web/static/icons/cat.ico +0 -0
- calc_terminal/web/static/icons/icon-128.png +0 -0
- calc_terminal/web/static/icons/icon-144.png +0 -0
- calc_terminal/web/static/icons/icon-152.png +0 -0
- calc_terminal/web/static/icons/icon-192.png +0 -0
- calc_terminal/web/static/icons/icon-384.png +0 -0
- calc_terminal/web/static/icons/icon-512.png +0 -0
- calc_terminal/web/static/icons/icon-72.png +0 -0
- calc_terminal/web/static/icons/icon-96.png +0 -0
- calc_terminal/web/static/icons/icon.svg +34 -0
- calc_terminal/web/static/icons/new-project.png +0 -0
- calc_terminal/web/static/icons/open-project.png +0 -0
- calc_terminal/web/static/index.html +734 -0
- calc_terminal/web/static/js/app.js +2403 -0
- calc_terminal/web/static/manifest.json +88 -0
- calc_terminal/web/static/sw.js +230 -0
- calc_terminal/workflow_engine.py +769 -0
- calc_terminal/workspace.py +593 -0
- calc_terminal/workspace_index.py +385 -0
- cct_cli-0.7.9.0.dist-info/METADATA +210 -0
- cct_cli-0.7.9.0.dist-info/RECORD +325 -0
- cct_cli-0.7.9.0.dist-info/WHEEL +5 -0
- cct_cli-0.7.9.0.dist-info/entry_points.txt +4 -0
- cct_cli-0.7.9.0.dist-info/licenses/LICENSE +21 -0
- cct_cli-0.7.9.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,743 @@
|
|
|
1
|
+
"""
|
|
2
|
+
CCT attachment pipeline (v0.7.8.1) — the ONE internal representation
|
|
3
|
+
every attachment takes between "user selects a file" and "the model
|
|
4
|
+
receives usable context".
|
|
5
|
+
|
|
6
|
+
Attachment {
|
|
7
|
+
id, name, path, extension, mime_type, size,
|
|
8
|
+
kind, content, metadata, extraction_status, error
|
|
9
|
+
}
|
|
10
|
+
|
|
11
|
+
The pipeline every attachment goes through:
|
|
12
|
+
|
|
13
|
+
SELECT -> VALIDATE -> READ -> DETECT -> EXTRACT
|
|
14
|
+
-> NORMALIZED OBJECT -> CONTEXT BLOCK -> MODEL
|
|
15
|
+
|
|
16
|
+
Rules this module enforces:
|
|
17
|
+
|
|
18
|
+
- NEVER assume a file was read. `extraction_status` is only "ready"
|
|
19
|
+
after validation + reading + type detection + extraction all
|
|
20
|
+
succeeded; anything else is "failed" with an honest `error`, and the
|
|
21
|
+
UI shows the failure instead of a false "Attached" chip.
|
|
22
|
+
- No random path strings flow around the application: the UI chip, the
|
|
23
|
+
message Turn, the context builder and the provider adapter all
|
|
24
|
+
reference the same Attachment object by `id`.
|
|
25
|
+
- If extraction fails, the context block still says so explicitly —
|
|
26
|
+
the model is never silently handed nothing.
|
|
27
|
+
- Large files/projects are capped and summarized, never dumped whole:
|
|
28
|
+
every context block obeys `max_chars`, and folders produce a real
|
|
29
|
+
listing + only the most relevant text files' content.
|
|
30
|
+
|
|
31
|
+
The capability layer (`provider_capabilities`) lives here too: it
|
|
32
|
+
decides whether the active provider can receive native image payloads
|
|
33
|
+
(vision) or must fall back to the extracted textual context. The
|
|
34
|
+
fallback is ALWAYS safe — text context works for every provider.
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
if __name__ == "__main__":
|
|
38
|
+
print("This is a library file and is not meant to be run directly.")
|
|
39
|
+
import sys
|
|
40
|
+
sys.exit(1)
|
|
41
|
+
|
|
42
|
+
import base64
|
|
43
|
+
import csv
|
|
44
|
+
import mimetypes
|
|
45
|
+
import os
|
|
46
|
+
import re
|
|
47
|
+
import time
|
|
48
|
+
import zlib
|
|
49
|
+
|
|
50
|
+
# ---------------------------------------------------------------------------
|
|
51
|
+
# Constants / registry of supported file types (spec section 4)
|
|
52
|
+
# ---------------------------------------------------------------------------
|
|
53
|
+
|
|
54
|
+
TEXT_FILE_MAX_CHARS = 20000
|
|
55
|
+
FOLDER_MAX_FILES = 40
|
|
56
|
+
FOLDER_MAX_BYTES = 30000
|
|
57
|
+
|
|
58
|
+
# Text/code formats read directly (content included verbatim, capped).
|
|
59
|
+
CODE_EXTS = {
|
|
60
|
+
".py", ".js", ".ts", ".tsx", ".jsx", ".html", ".css", ".json", ".yaml",
|
|
61
|
+
".yml", ".toml", ".md", ".txt", ".xml", ".sql", ".sh", ".bat", ".ps1",
|
|
62
|
+
".c", ".cpp", ".h", ".hpp", ".java", ".rs", ".go", ".php", ".rb", ".kt",
|
|
63
|
+
".swift", ".lua", ".r", ".cs", ".pl", ".m", ".vue", ".svelte", ".scss",
|
|
64
|
+
".less", ".ini", ".conf", ".cfg", ".env", ".gitignore", ".dockerfile",
|
|
65
|
+
".lock", ".log", ".csv", ".tsv", ".graphql", ".proto",
|
|
66
|
+
}
|
|
67
|
+
IMAGE_EXTS = {".png", ".jpg", ".jpeg", ".gif", ".webp", ".bmp", ".ico", ".svg", ".tiff", ".tif"}
|
|
68
|
+
ARCHIVE_EXTS = {".zip", ".tar", ".gz", ".bz2", ".tgz", ".rar", ".7z", ".xz", ".zst"}
|
|
69
|
+
PDF_EXTS = {".pdf"}
|
|
70
|
+
AUDIO_EXTS = {".mp3", ".wav", ".ogg", ".flac", ".m4a", ".aac", ".wma", ".opus"}
|
|
71
|
+
VIDEO_EXTS = {".mp4", ".mkv", ".avi", ".mov", ".webm", ".wmv", ".m4v", ".mpg", ".mpeg"}
|
|
72
|
+
BINARY_EXTS = {".exe", ".dll", ".so", ".dylib", ".bin", ".dat", ".db", ".sqlite",
|
|
73
|
+
".sqlite3", ".pyc", ".pdb", ".iso", ".img", ".parquet", ".h5", ".hdf5",
|
|
74
|
+
".doc", ".docx", ".xls", ".xlsx", ".ppt", ".pptx", ".odt"}
|
|
75
|
+
|
|
76
|
+
# Structured formats that get a real parse (not just raw text).
|
|
77
|
+
STRUCTURED_EXTS = {".json", ".yaml", ".yml", ".toml", ".csv", ".tsv", ".xml", ".sql"}
|
|
78
|
+
|
|
79
|
+
KIND_LABELS = {
|
|
80
|
+
"code": "Code", "markdown": "Markdown", "data": "Data",
|
|
81
|
+
"config": "Config", "text": "Text", "image": "Image", "pdf": "PDF",
|
|
82
|
+
"archive": "Archive", "folder": "Folder", "audio": "Audio",
|
|
83
|
+
"video": "Video", "binary": "Binary", "document": "Document",
|
|
84
|
+
"unknown": "File",
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
# Extraction statuses the whole app shares.
|
|
88
|
+
STATUS_SELECTING = "selecting"
|
|
89
|
+
STATUS_READING = "reading"
|
|
90
|
+
STATUS_READY = "ready"
|
|
91
|
+
STATUS_FAILED = "failed"
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class Attachment:
|
|
95
|
+
"""One normalized attachment. `id` is the stable key the UI chip,
|
|
96
|
+
the message Turn, the context builder and the provider adapter all
|
|
97
|
+
share — no random path strings anywhere.
|
|
98
|
+
|
|
99
|
+
`content` is the model-usable extracted payload (text for text-like
|
|
100
|
+
files, a real listing for archives, metadata for binary, etc.).
|
|
101
|
+
`extraction_status` is only `ready` when every stage succeeded;
|
|
102
|
+
otherwise `failed` and `error` says why.
|
|
103
|
+
"""
|
|
104
|
+
|
|
105
|
+
__slots__ = ("id", "name", "path", "extension", "mime_type", "size",
|
|
106
|
+
"kind", "content", "metadata", "extraction_status",
|
|
107
|
+
"error", "created_at", "source")
|
|
108
|
+
|
|
109
|
+
def __init__(self, path, kind="unknown"):
|
|
110
|
+
self.id = f"att-{int(time.time() * 1000)}-{abs(hash(os.path.normpath(path))) % 1000000}"
|
|
111
|
+
self.path = os.path.abspath(os.path.expanduser(path))
|
|
112
|
+
self.name = os.path.basename(self.path.rstrip(os.sep)) or self.path
|
|
113
|
+
self.extension = os.path.splitext(self.name)[1].lower()
|
|
114
|
+
try:
|
|
115
|
+
self.mime_type = mimetypes.guess_type(self.name)[0] or "application/octet-stream"
|
|
116
|
+
except Exception:
|
|
117
|
+
self.mime_type = "application/octet-stream"
|
|
118
|
+
try:
|
|
119
|
+
self.size = os.path.getsize(self.path) if os.path.isfile(self.path) else 0
|
|
120
|
+
except Exception:
|
|
121
|
+
self.size = 0
|
|
122
|
+
self.kind = kind if kind != "unknown" else detect_kind(self.path)
|
|
123
|
+
self.content = None
|
|
124
|
+
self.metadata = {}
|
|
125
|
+
self.extraction_status = STATUS_SELECTING
|
|
126
|
+
self.error = None
|
|
127
|
+
self.created_at = time.time()
|
|
128
|
+
# How this file arrived: "browse" | "drag_and_drop" | "paste" |
|
|
129
|
+
# "sidebar". Recorded for every attachment (requirement #16).
|
|
130
|
+
self.source = "browse"
|
|
131
|
+
|
|
132
|
+
# ---------------------------------------------------------- helpers
|
|
133
|
+
def is_ready(self):
|
|
134
|
+
return self.extraction_status == STATUS_READY
|
|
135
|
+
|
|
136
|
+
def is_failed(self):
|
|
137
|
+
return self.extraction_status == STATUS_FAILED
|
|
138
|
+
|
|
139
|
+
def to_dict(self):
|
|
140
|
+
return {
|
|
141
|
+
"id": self.id, "name": self.name, "path": self.path,
|
|
142
|
+
"extension": self.extension, "mime_type": self.mime_type,
|
|
143
|
+
"size": self.size, "kind": self.kind,
|
|
144
|
+
"extraction_status": self.extraction_status, "error": self.error,
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
@staticmethod
|
|
148
|
+
def from_dict(data):
|
|
149
|
+
if isinstance(data, Attachment):
|
|
150
|
+
return data
|
|
151
|
+
if not isinstance(data, dict) or not data.get("path"):
|
|
152
|
+
return None
|
|
153
|
+
att = Attachment(data["path"])
|
|
154
|
+
for key in ("id", "kind", "extraction_status", "error", "content", "metadata"):
|
|
155
|
+
if key in data:
|
|
156
|
+
setattr(att, key, data[key])
|
|
157
|
+
return att
|
|
158
|
+
|
|
159
|
+
def __repr__(self):
|
|
160
|
+
return (f"<Attachment id={self.id} name={self.name!r} kind={self.kind} "
|
|
161
|
+
f"status={self.extraction_status}>")
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
# ---------------------------------------------------------------------------
|
|
165
|
+
# Type detection
|
|
166
|
+
# ---------------------------------------------------------------------------
|
|
167
|
+
|
|
168
|
+
def detect_kind(path):
|
|
169
|
+
"""Automatic kind detection from the path/extension (spec section 4).
|
|
170
|
+
A directory is a 'folder'; everything else is decided by extension,
|
|
171
|
+
with a final text/unknown fallback."""
|
|
172
|
+
if os.path.isdir(path):
|
|
173
|
+
return "folder"
|
|
174
|
+
ext = os.path.splitext(path)[1].lower()
|
|
175
|
+
if ext in IMAGE_EXTS:
|
|
176
|
+
return "image"
|
|
177
|
+
if ext in PDF_EXTS:
|
|
178
|
+
return "pdf"
|
|
179
|
+
if ext in ARCHIVE_EXTS:
|
|
180
|
+
return "archive"
|
|
181
|
+
if ext in AUDIO_EXTS:
|
|
182
|
+
return "audio"
|
|
183
|
+
if ext in VIDEO_EXTS:
|
|
184
|
+
return "video"
|
|
185
|
+
if ext in BINARY_EXTS:
|
|
186
|
+
return "document"
|
|
187
|
+
if ext == ".md":
|
|
188
|
+
return "markdown"
|
|
189
|
+
if ext in STRUCTURED_EXTS or ext in {".ini", ".conf", ".cfg", ".env", ".gitignore", ".dockerfile", ".lock"}:
|
|
190
|
+
return "data"
|
|
191
|
+
if ext in CODE_EXTS:
|
|
192
|
+
return "code"
|
|
193
|
+
# Extension-less file: sniff the head for printable text.
|
|
194
|
+
try:
|
|
195
|
+
with open(path, "rb") as f:
|
|
196
|
+
head = f.read(2048)
|
|
197
|
+
if head and not bytes(head).translate(None, b"\x00\x01\x02\x03\x04\x05\x06\x07"
|
|
198
|
+
b"\x08\x0e\x0f\x10\x11\x12\x13\x14\x15\x16\x17"
|
|
199
|
+
b"\x18\x19\x1a\x1c\x1d\x1e\x1f").lstrip(b" \t\r\n"):
|
|
200
|
+
return "text"
|
|
201
|
+
except Exception:
|
|
202
|
+
pass
|
|
203
|
+
return "unknown"
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def human_size(n):
|
|
207
|
+
if n < 1024:
|
|
208
|
+
return f"{n} B"
|
|
209
|
+
if n < 1024 ** 2:
|
|
210
|
+
return f"{n / 1024:.1f} KB"
|
|
211
|
+
if n < 1024 ** 3:
|
|
212
|
+
return f"{n / 1024 ** 2:.1f} MB"
|
|
213
|
+
return f"{n / 1024 ** 3:.1f} GB"
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
# ---------------------------------------------------------------------------
|
|
217
|
+
# Extractors — every one returns (content, metadata, error) and NEVER
|
|
218
|
+
# fabricates data. A failure is an explicit error string, never silence.
|
|
219
|
+
# ---------------------------------------------------------------------------
|
|
220
|
+
|
|
221
|
+
def _read_text(path, max_chars=TEXT_FILE_MAX_CHARS):
|
|
222
|
+
"""Read a UTF-8 (or best-effort) text file, capped at max_chars.
|
|
223
|
+
Returns (content, truncated, error)."""
|
|
224
|
+
for enc in ("utf-8", "utf-8-sig", "latin-1"):
|
|
225
|
+
try:
|
|
226
|
+
with open(path, "r", encoding=enc) as f:
|
|
227
|
+
content = f.read(max_chars + 1)
|
|
228
|
+
truncated = len(content) > max_chars
|
|
229
|
+
if truncated:
|
|
230
|
+
content = content[:max_chars]
|
|
231
|
+
return content, truncated, None
|
|
232
|
+
except UnicodeDecodeError:
|
|
233
|
+
continue
|
|
234
|
+
except Exception as e:
|
|
235
|
+
return None, False, str(e)
|
|
236
|
+
return None, False, "could not decode the file as text"
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def _extract_structured(path, ext, max_chars):
|
|
240
|
+
"""Structured formats get a real parse where useful (spec section 4:
|
|
241
|
+
'For structured formats: PARSE WHEN APPROPRIATE') — JSON is
|
|
242
|
+
pretty-printed (and schema-flattened when huge), CSV gets a genuine
|
|
243
|
+
header/sample/count. Unknown structure falls back to raw text."""
|
|
244
|
+
if ext == ".json":
|
|
245
|
+
try:
|
|
246
|
+
with open(path, "r", encoding="utf-8") as f:
|
|
247
|
+
raw = f.read(max_chars + 1)
|
|
248
|
+
try:
|
|
249
|
+
import json
|
|
250
|
+
data = json.loads(raw)
|
|
251
|
+
pretty = json.dumps(data, indent=2, ensure_ascii=False)
|
|
252
|
+
truncated = len(pretty) > max_chars
|
|
253
|
+
if truncated:
|
|
254
|
+
pretty = pretty[:max_chars]
|
|
255
|
+
return pretty, truncated, None
|
|
256
|
+
except Exception:
|
|
257
|
+
truncated = len(raw) > max_chars
|
|
258
|
+
return (raw[:max_chars] if truncated else raw), truncated, None
|
|
259
|
+
except Exception as e:
|
|
260
|
+
return None, False, f"could not parse JSON: {e}"
|
|
261
|
+
if ext in (".csv", ".tsv"):
|
|
262
|
+
delimiter = "\t" if ext == ".tsv" else ","
|
|
263
|
+
header, sample, total = None, [], 0
|
|
264
|
+
capped = False
|
|
265
|
+
try:
|
|
266
|
+
with open(path, "r", encoding="utf-8", errors="replace", newline="") as f:
|
|
267
|
+
for i, row in enumerate(csv.reader(f, delimiter=delimiter)):
|
|
268
|
+
if i == 0:
|
|
269
|
+
header = row
|
|
270
|
+
elif len(sample) < 5:
|
|
271
|
+
sample.append(row)
|
|
272
|
+
total += 1
|
|
273
|
+
if total >= 5000:
|
|
274
|
+
capped = True
|
|
275
|
+
break
|
|
276
|
+
except Exception as e:
|
|
277
|
+
return None, False, f"could not parse CSV: {e}"
|
|
278
|
+
lines = [f"CSV {path}: {len(header) if header else 0} columns, "
|
|
279
|
+
f"{total} rows{' (capped at 5000)' if capped else ''}"]
|
|
280
|
+
if header:
|
|
281
|
+
lines.append("header: " + " | ".join(header))
|
|
282
|
+
if sample:
|
|
283
|
+
lines.append("first rows:")
|
|
284
|
+
lines += [" " + " | ".join(row) for row in sample]
|
|
285
|
+
body = "\n".join(lines)
|
|
286
|
+
truncated = len(body) > max_chars
|
|
287
|
+
if truncated:
|
|
288
|
+
body = body[:max_chars]
|
|
289
|
+
return body, truncated, None
|
|
290
|
+
# yaml/toml/xml/sql/etc. — the plain text is the useful parse here.
|
|
291
|
+
content, truncated, err = _read_text(path, max_chars)
|
|
292
|
+
return content, truncated, err
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _extract_archive(path, ext, max_chars):
|
|
296
|
+
name = os.path.basename(path)
|
|
297
|
+
size_txt = human_size(os.path.getsize(path) if os.path.isfile(path) else 0)
|
|
298
|
+
if ext == ".zip":
|
|
299
|
+
try:
|
|
300
|
+
import zipfile
|
|
301
|
+
with zipfile.ZipFile(path) as zf:
|
|
302
|
+
infos = zf.infolist()
|
|
303
|
+
files = [i for i in infos if not i.is_dir()]
|
|
304
|
+
total = sum(i.file_size for i in files)
|
|
305
|
+
listing = "\n".join(i.filename for i in files[:40])
|
|
306
|
+
more = f"\n\u2026 and {len(files) - 40} more files" if len(files) > 40 else ""
|
|
307
|
+
body = (f"zip archive: {len(files)} files, {human_size(total)} uncompressed\n"
|
|
308
|
+
f"{listing}{more}" if files else "zip archive: empty.")
|
|
309
|
+
return f"[attachment: {name} ({size_txt})]\n{body}"[:max_chars], False, None
|
|
310
|
+
except Exception as e:
|
|
311
|
+
return None, False, f"not a readable zip archive: {e}"
|
|
312
|
+
try:
|
|
313
|
+
import tarfile
|
|
314
|
+
with tarfile.open(path) as tf:
|
|
315
|
+
files = [m for m in tf.getmembers() if m.isfile()]
|
|
316
|
+
listing = "\n".join(m.name for m in files[:40])
|
|
317
|
+
more = f"\n\u2026 and {len(files) - 40} more files" if len(files) > 40 else ""
|
|
318
|
+
body = (f"tar archive: {len(files)} files\n{listing}{more}" if files
|
|
319
|
+
else "tar archive: empty.")
|
|
320
|
+
return f"[attachment: {name} ({size_txt})]\n{body}"[:max_chars], False, None
|
|
321
|
+
except Exception as e:
|
|
322
|
+
return None, False, f"not a readable tar archive: {e}"
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _extract_pdf(path, max_chars):
|
|
326
|
+
name = os.path.basename(path)
|
|
327
|
+
size_txt = human_size(os.path.getsize(path) if os.path.isfile(path) else 0)
|
|
328
|
+
try:
|
|
329
|
+
with open(path, "rb") as f:
|
|
330
|
+
raw = f.read()
|
|
331
|
+
except Exception as e:
|
|
332
|
+
return None, False, f"unreadable PDF: {e}"
|
|
333
|
+
counts = [int(m) for m in re.findall(rb"/Count\s+(\d+)", raw)]
|
|
334
|
+
page_txt = f"{max(counts)} pages" if counts else "page count unknown"
|
|
335
|
+
meta_bits = []
|
|
336
|
+
for key in (b"Title", b"Author", b"Subject", b"Creator", b"Producer"):
|
|
337
|
+
m = re.search(key + rb"\s*\(([^()\\]*(?:\\.[^()\\]*)*)\)", raw[:200000])
|
|
338
|
+
if m:
|
|
339
|
+
val = m.group(1)[:120].decode("latin-1", "replace")
|
|
340
|
+
meta_bits.append(f"{key.decode()}: {val}")
|
|
341
|
+
meta_txt = ("; ".join(meta_bits) + ".") if meta_bits else "no document metadata."
|
|
342
|
+
texts = []
|
|
343
|
+
for m in re.finditer(rb"stream\r?\n(.*?)endstream", raw, re.DOTALL):
|
|
344
|
+
data = m.group(1).lstrip(b"\r\n")
|
|
345
|
+
try:
|
|
346
|
+
payload = zlib.decompress(data)
|
|
347
|
+
except Exception:
|
|
348
|
+
payload = data
|
|
349
|
+
texts += re.findall(rb"\(((?:[^()\\]|\\.)*)\)\s*Tj", payload)
|
|
350
|
+
plain = " ".join(
|
|
351
|
+
t.replace(b"\\(", b"(").replace(b"\\)", b")").replace(b"\\\\", b"\\")
|
|
352
|
+
.decode("latin-1", "replace") for t in texts)
|
|
353
|
+
plain = re.sub(r"\s+", " ", plain).strip()
|
|
354
|
+
if plain:
|
|
355
|
+
if len(plain) > max_chars:
|
|
356
|
+
plain = plain[:max_chars] + " \u2026[truncated]"
|
|
357
|
+
body = f"[attachment: {name} ({size_txt}) \u2014 PDF, {page_txt}; {meta_txt}]\n{plain}"
|
|
358
|
+
else:
|
|
359
|
+
body = (f"[attachment: {name} ({size_txt}) \u2014 PDF, {page_txt}; {meta_txt} "
|
|
360
|
+
f"No extractable text layer (scanned image or glyph-encoded PDF).]")
|
|
361
|
+
return body, False, None
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _extract_image(path):
|
|
365
|
+
"""Images: real dimensions/format metadata when PIL exists; never
|
|
366
|
+
fabricated content. Whether the image itself travels natively is
|
|
367
|
+
decided by the capability layer at request time — this metadata
|
|
368
|
+
block is the always-safe textual fallback."""
|
|
369
|
+
name = os.path.basename(path)
|
|
370
|
+
size_txt = human_size(os.path.getsize(path) if os.path.isfile(path) else 0)
|
|
371
|
+
dims = ""
|
|
372
|
+
try:
|
|
373
|
+
from PIL import Image
|
|
374
|
+
with Image.open(path) as im:
|
|
375
|
+
w, h = im.size
|
|
376
|
+
fmt = (im.format or "").upper()
|
|
377
|
+
dims = f" {w}x{h} {fmt}"
|
|
378
|
+
except Exception:
|
|
379
|
+
pass
|
|
380
|
+
body = (f"[Attached image: {name} ({size_txt}{dims}) \u2014 image metadata; "
|
|
381
|
+
f"if the active model supports vision the image is also sent natively, "
|
|
382
|
+
f"otherwise analyze it from this metadata.]")
|
|
383
|
+
return body, False, None
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _extract_folder(path, max_chars):
|
|
387
|
+
"""Folders: a real recursive listing (capped), plus the content of
|
|
388
|
+
the most relevant text/code files (capped), never the whole tree.
|
|
389
|
+
This is what 'attach a project/folder' means — the model gets the
|
|
390
|
+
shape of the project and its key files, not megabytes of dumps."""
|
|
391
|
+
files = []
|
|
392
|
+
dirs = 0
|
|
393
|
+
for dp, dnames, fnames in os.walk(path):
|
|
394
|
+
dnames[:] = [d for d in dnames
|
|
395
|
+
if d not in (".git", "node_modules", "__pycache__", ".venv",
|
|
396
|
+
"venv", "dist", "build", ".idea", ".vscode")]
|
|
397
|
+
dirs += len(dnames)
|
|
398
|
+
for f in fnames:
|
|
399
|
+
full = os.path.join(dp, f)
|
|
400
|
+
rel = os.path.relpath(full, path)
|
|
401
|
+
files.append(rel)
|
|
402
|
+
files.sort(key=lambda r: (os.path.basename(r).lower(), r))
|
|
403
|
+
listing = files[:FOLDER_MAX_FILES]
|
|
404
|
+
more = len(files) - FOLDER_MAX_FILES
|
|
405
|
+
lines = [
|
|
406
|
+
f"[attachment: folder {os.path.basename(path.rstrip(os.sep))} "
|
|
407
|
+
f"\u2014 {len(files)} files, {dirs} folders]",
|
|
408
|
+
]
|
|
409
|
+
lines += [" " + r for r in listing]
|
|
410
|
+
if more > 0:
|
|
411
|
+
lines.append(f" \u2026 and {more} more files (listing capped)")
|
|
412
|
+
body = "\n".join(lines)
|
|
413
|
+
# Read a few of the most relevant text files so the model can
|
|
414
|
+
# actually answer questions about the project.
|
|
415
|
+
budget = max_chars - len(body) - 512
|
|
416
|
+
included = 0
|
|
417
|
+
if budget > 0:
|
|
418
|
+
for rel in listing:
|
|
419
|
+
ext = os.path.splitext(rel)[1].lower()
|
|
420
|
+
if ext not in CODE_EXTS and ext not in {".md", ".txt", ".json", ".yaml", ".yml", ".toml", ".csv"}:
|
|
421
|
+
continue
|
|
422
|
+
full = os.path.join(path, rel)
|
|
423
|
+
content, truncated, _err = _read_text(full, min(6000, budget))
|
|
424
|
+
if content is None:
|
|
425
|
+
continue
|
|
426
|
+
block = (f"\n--- {rel}{' [truncated]' if truncated else ''} ---\n{content}")
|
|
427
|
+
if len(block) > budget:
|
|
428
|
+
block = block[:budget]
|
|
429
|
+
body += block
|
|
430
|
+
budget -= len(block)
|
|
431
|
+
included += 1
|
|
432
|
+
if budget < 2000 or included >= 6:
|
|
433
|
+
break
|
|
434
|
+
if len(body) > max_chars:
|
|
435
|
+
body = body[:max_chars]
|
|
436
|
+
return body, False, None
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def _extract_media(path, kind, max_chars):
|
|
440
|
+
name = os.path.basename(path)
|
|
441
|
+
size_txt = human_size(os.path.getsize(path) if os.path.isfile(path) else 0)
|
|
442
|
+
return (f"[attachment: {name} ({size_txt}) \u2014 {kind} file; duration and tags "
|
|
443
|
+
f"would need a media library, which isn't available in this environment. "
|
|
444
|
+
f"Only the file's existence/size can be reported honestly.]", False, None)
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def extract_attachment(path, max_chars=TEXT_FILE_MAX_CHARS):
|
|
448
|
+
"""VALIDATE -> READ -> DETECT -> EXTRACT for one path. Returns
|
|
449
|
+
(content, metadata, kind, error). content is None on failure (error
|
|
450
|
+
is set) — never a silent empty string."""
|
|
451
|
+
path = os.path.abspath(os.path.expanduser(path))
|
|
452
|
+
if not os.path.lexists(path):
|
|
453
|
+
return None, {}, "unknown", "the path does not exist on disk"
|
|
454
|
+
if os.path.isdir(path):
|
|
455
|
+
kind = "folder"
|
|
456
|
+
try:
|
|
457
|
+
content, _trunc, err = _extract_folder(path, max_chars)
|
|
458
|
+
except Exception as e:
|
|
459
|
+
content, err = None, f"could not read folder: {e}"
|
|
460
|
+
if content is None:
|
|
461
|
+
return None, {}, kind, err or "could not read folder"
|
|
462
|
+
return content, {"items": "recursive listing"}, kind, None
|
|
463
|
+
if not os.path.isfile(path):
|
|
464
|
+
return None, {}, "unknown", "the path is neither a file nor a folder"
|
|
465
|
+
try:
|
|
466
|
+
size = os.path.getsize(path)
|
|
467
|
+
except Exception as e:
|
|
468
|
+
return None, {}, "unknown", f"could not stat the file: {e}"
|
|
469
|
+
ext = os.path.splitext(path)[1].lower()
|
|
470
|
+
kind = detect_kind(path)
|
|
471
|
+
meta = {"size": size, "mime": mimetypes.guess_type(path)[0] or "application/octet-stream"}
|
|
472
|
+
|
|
473
|
+
if kind == "image":
|
|
474
|
+
content, _trunc, err = _extract_image(path)
|
|
475
|
+
if err:
|
|
476
|
+
return None, meta, kind, err
|
|
477
|
+
return content, meta, kind, None
|
|
478
|
+
if kind == "pdf":
|
|
479
|
+
content, _trunc, err = _extract_pdf(path, max_chars)
|
|
480
|
+
if err:
|
|
481
|
+
return None, meta, kind, err
|
|
482
|
+
return content, meta, kind, None
|
|
483
|
+
if kind == "archive":
|
|
484
|
+
content, _trunc, err = _extract_archive(path, ext, max_chars)
|
|
485
|
+
if err:
|
|
486
|
+
return None, meta, kind, err
|
|
487
|
+
return content, meta, kind, None
|
|
488
|
+
if kind == "audio":
|
|
489
|
+
content, _trunc, err = _extract_media(path, "audio", max_chars)
|
|
490
|
+
return content, meta, kind, err
|
|
491
|
+
if kind == "video":
|
|
492
|
+
content, _trunc, err = _extract_media(path, "video", max_chars)
|
|
493
|
+
return content, meta, kind, err
|
|
494
|
+
if kind in ("code", "markdown", "text", "data"):
|
|
495
|
+
content, truncated, err = _extract_structured(path, ext, max_chars) \
|
|
496
|
+
if ext in STRUCTURED_EXTS else _read_text(path, max_chars)
|
|
497
|
+
if err:
|
|
498
|
+
return None, meta, kind, f"the file exists, but CAT could not extract its contents: {err}"
|
|
499
|
+
if truncated:
|
|
500
|
+
meta["truncated"] = True
|
|
501
|
+
return content, meta, kind, None
|
|
502
|
+
# document/binary/unknown
|
|
503
|
+
return (f"[attachment: {os.path.basename(path)} ({human_size(size)}) \u2014 {kind} "
|
|
504
|
+
f"file, content not readable as text. Only metadata is available.]",
|
|
505
|
+
meta, kind, None)
|
|
506
|
+
|
|
507
|
+
|
|
508
|
+
# ---------------------------------------------------------------------------
|
|
509
|
+
# Manager + context building
|
|
510
|
+
# ---------------------------------------------------------------------------
|
|
511
|
+
|
|
512
|
+
class AttachmentManager:
|
|
513
|
+
"""The single entry point for creating/validating attachments and
|
|
514
|
+
turning them into model context. The UI, the session and aicore all
|
|
515
|
+
use this — never ad-hoc path handling."""
|
|
516
|
+
|
|
517
|
+
_instance = None
|
|
518
|
+
|
|
519
|
+
def __new__(cls):
|
|
520
|
+
if cls._instance is None:
|
|
521
|
+
cls._instance = super().__new__(cls)
|
|
522
|
+
return cls._instance
|
|
523
|
+
|
|
524
|
+
@classmethod
|
|
525
|
+
def create(cls, path, max_chars=TEXT_FILE_MAX_CHARS):
|
|
526
|
+
"""SELECT -> VALIDATE -> READ -> DETECT -> EXTRACT -> object.
|
|
527
|
+
Blocks for IO; call from a worker thread, never the UI thread.
|
|
528
|
+
The returned Attachment is always usable: READY with real
|
|
529
|
+
content, or FAILED with an honest error."""
|
|
530
|
+
path = os.path.abspath(os.path.expanduser(path))
|
|
531
|
+
att = Attachment(path)
|
|
532
|
+
if not os.path.lexists(path):
|
|
533
|
+
att.extraction_status = STATUS_FAILED
|
|
534
|
+
att.error = "the file does not exist"
|
|
535
|
+
return att
|
|
536
|
+
if not (os.path.isfile(path) or os.path.isdir(path)):
|
|
537
|
+
att.extraction_status = STATUS_FAILED
|
|
538
|
+
att.error = "the path is neither a file nor a folder"
|
|
539
|
+
return att
|
|
540
|
+
att.extraction_status = STATUS_READING
|
|
541
|
+
content, meta, kind, err = extract_attachment(path, max_chars=max_chars)
|
|
542
|
+
att.kind = kind
|
|
543
|
+
att.metadata.update(meta or {})
|
|
544
|
+
if err is not None:
|
|
545
|
+
att.extraction_status = STATUS_FAILED
|
|
546
|
+
att.error = err
|
|
547
|
+
return att
|
|
548
|
+
att.content = content
|
|
549
|
+
att.extraction_status = STATUS_READY
|
|
550
|
+
return att
|
|
551
|
+
|
|
552
|
+
@classmethod
|
|
553
|
+
def ensure_extracted(cls, att, max_chars=TEXT_FILE_MAX_CHARS):
|
|
554
|
+
"""Re-run extraction when an attachment is still pending (e.g.
|
|
555
|
+
the user sent the message before the async extraction finished)
|
|
556
|
+
so the request NEVER carries an unverified attachment. Returns
|
|
557
|
+
the same object, updated in place."""
|
|
558
|
+
if att is None:
|
|
559
|
+
return att
|
|
560
|
+
if att.extraction_status == STATUS_READY and att.content is not None:
|
|
561
|
+
return att
|
|
562
|
+
if att.extraction_status == STATUS_FAILED:
|
|
563
|
+
return att
|
|
564
|
+
fresh = cls.create(att.path, max_chars=max_chars)
|
|
565
|
+
att.content = fresh.content
|
|
566
|
+
att.kind = fresh.kind
|
|
567
|
+
att.metadata = fresh.metadata
|
|
568
|
+
att.extraction_status = fresh.extraction_status
|
|
569
|
+
att.error = fresh.error
|
|
570
|
+
return att
|
|
571
|
+
|
|
572
|
+
@classmethod
|
|
573
|
+
def build_context(cls, attachments, max_chars=TEXT_FILE_MAX_CHARS):
|
|
574
|
+
"""Turns attachment objects into one model-readable context
|
|
575
|
+
block each (spec section 5 fallback format). Unverifiable
|
|
576
|
+
attachments produce an explicit '[could not be read]' block —
|
|
577
|
+
NEVER a silent omission."""
|
|
578
|
+
if not attachments:
|
|
579
|
+
return ""
|
|
580
|
+
blocks = []
|
|
581
|
+
for att in attachments:
|
|
582
|
+
att = cls.ensure_extracted(att, max_chars=max_chars)
|
|
583
|
+
if att is None:
|
|
584
|
+
continue
|
|
585
|
+
if att.extraction_status != STATUS_READY or att.content is None:
|
|
586
|
+
blocks.append(
|
|
587
|
+
f"--- ATTACHMENT ---\nFile: {att.name}\nPath: {att.path}\n"
|
|
588
|
+
f"Status: unavailable\n"
|
|
589
|
+
f"Error: {att.error or 'content could not be extracted'}\n"
|
|
590
|
+
f"--- END ATTACHMENT ---")
|
|
591
|
+
continue
|
|
592
|
+
body = att.content
|
|
593
|
+
if len(body) > max_chars:
|
|
594
|
+
body = body[:max_chars] + " \u2026[truncated]"
|
|
595
|
+
blocks.append(
|
|
596
|
+
f"--- ATTACHMENT ---\n"
|
|
597
|
+
f"File: {att.name}\nType: {KIND_LABELS.get(att.kind, att.kind)}\n"
|
|
598
|
+
f"Path: {att.path}\n"
|
|
599
|
+
f"Content:\n{body}\n"
|
|
600
|
+
f"--- END ATTACHMENT ---")
|
|
601
|
+
return "\n\n".join(blocks)
|
|
602
|
+
|
|
603
|
+
@classmethod
|
|
604
|
+
def verify(cls, attachments, log=None):
|
|
605
|
+
"""Attachment context verification (spec section 7): every
|
|
606
|
+
attachment must have a valid id/path/type and successful
|
|
607
|
+
extraction (or a valid native payload) before the request goes
|
|
608
|
+
out. Returns True when every attachment is verified. `log` is a
|
|
609
|
+
callable(message) for debug output — never file contents."""
|
|
610
|
+
if log is None:
|
|
611
|
+
log = lambda _m: None
|
|
612
|
+
if not attachments:
|
|
613
|
+
return True
|
|
614
|
+
log(f"AttachmentManager: verifying {len(attachments)} attachment(s)")
|
|
615
|
+
ok = True
|
|
616
|
+
for att in attachments:
|
|
617
|
+
if att is None or not getattr(att, "id", None):
|
|
618
|
+
log("AttachmentManager: INVALID attachment (no id)")
|
|
619
|
+
ok = False
|
|
620
|
+
continue
|
|
621
|
+
valid_path = bool(getattr(att, "path", "")) and os.path.lexists(att.path)
|
|
622
|
+
cls.ensure_extracted(att)
|
|
623
|
+
if not valid_path:
|
|
624
|
+
log(f"AttachmentManager: {att.id} INVALID path")
|
|
625
|
+
ok = False
|
|
626
|
+
elif att.extraction_status != STATUS_READY or not att.content:
|
|
627
|
+
log(f"AttachmentManager: {att.id} NOT extracted ({att.error or 'unknown'})")
|
|
628
|
+
ok = False
|
|
629
|
+
else:
|
|
630
|
+
log(f"AttachmentManager: {att.id} verified (kind={att.kind}, "
|
|
631
|
+
f"{len(att.content)} chars)")
|
|
632
|
+
log(f"AttachmentManager: {'all attachments verified' if ok else 'attachment verification FAILED'}")
|
|
633
|
+
return ok
|
|
634
|
+
|
|
635
|
+
|
|
636
|
+
# ---------------------------------------------------------------------------
|
|
637
|
+
# Provider/model capability layer (spec section 6)
|
|
638
|
+
# ---------------------------------------------------------------------------
|
|
639
|
+
|
|
640
|
+
# Provider styles that can receive a native image payload in CCT's
|
|
641
|
+
# current transport (see aicore._stream_ai_once).
|
|
642
|
+
_VISION_API_STYLES = ("openai", "anthropic", "gemini")
|
|
643
|
+
|
|
644
|
+
# Well-known model families that do NOT accept image content blocks —
|
|
645
|
+
# anything unknown defaults to "no vision", which is the safe choice
|
|
646
|
+
# (falls back to textual context).
|
|
647
|
+
_NON_VISION_MODEL_HINTS = (
|
|
648
|
+
"gpt-3.5", "llama", "mixtral", "mistral", "deepseek", "phi-3", "phi3",
|
|
649
|
+
"qwen", "command", "gemma", "granite", "codex", "o1-mini", "o3-mini",
|
|
650
|
+
)
|
|
651
|
+
_VISION_MODEL_HINTS = (
|
|
652
|
+
"gpt-4o", "gpt-4.1", "gpt-5", "o3", "o4", "claude-3", "claude-4",
|
|
653
|
+
"gemini-1.5", "gemini-2.0", "gemini-2.5", "gemini-3", "qwen2.5-vl",
|
|
654
|
+
"llava", "llama-3.2-vision",
|
|
655
|
+
)
|
|
656
|
+
|
|
657
|
+
|
|
658
|
+
def provider_capabilities(config=None):
|
|
659
|
+
"""Detect what the active provider/model can do (spec section 6):
|
|
660
|
+
returns a dict with `vision` (native image payloads supported) and
|
|
661
|
+
`native_attachments` (structured file attachments — always False in
|
|
662
|
+
this transport; the extracted-text fallback is universal).
|
|
663
|
+
|
|
664
|
+
v0.7.9.0: the dict is now backed by the model_router's capability
|
|
665
|
+
registry, so it also carries tools/streaming/reasoning/
|
|
666
|
+
context_window/latency_score — the same record the Smart Router
|
|
667
|
+
consults BEFORE a request goes out."""
|
|
668
|
+
caps = {"vision": False, "native_attachments": False,
|
|
669
|
+
"tools": True, "streaming": True, "reasoning": False,
|
|
670
|
+
"context_window": 32000, "latency_score": 0.5}
|
|
671
|
+
try:
|
|
672
|
+
if not config:
|
|
673
|
+
from . import aicore
|
|
674
|
+
config = aicore.load_config()
|
|
675
|
+
config = config or {}
|
|
676
|
+
# Preferred source of truth: the shared capability registry.
|
|
677
|
+
try:
|
|
678
|
+
from . import model_router as _mr
|
|
679
|
+
rc = _mr._capabilities_for(config)
|
|
680
|
+
caps.update({
|
|
681
|
+
"vision": bool(rc.vision),
|
|
682
|
+
"tools": bool(rc.tools),
|
|
683
|
+
"streaming": bool(rc.streaming),
|
|
684
|
+
"reasoning": bool(rc.reasoning),
|
|
685
|
+
"context_window": int(rc.context_window),
|
|
686
|
+
"latency_score": float(rc.latency_score),
|
|
687
|
+
})
|
|
688
|
+
return caps
|
|
689
|
+
except Exception:
|
|
690
|
+
pass
|
|
691
|
+
provider = str(config.get("provider", "")).lower()
|
|
692
|
+
model = str(config.get("model", "")).lower()
|
|
693
|
+
api_style = str(config.get("api_style", "")).lower()
|
|
694
|
+
if not api_style:
|
|
695
|
+
try:
|
|
696
|
+
from . import aicore as _a
|
|
697
|
+
info = _a.PROVIDERS.get(provider)
|
|
698
|
+
api_style = (info["api_style"] if info else "openai") or "openai"
|
|
699
|
+
except Exception:
|
|
700
|
+
api_style = "openai"
|
|
701
|
+
if api_style not in _VISION_API_STYLES:
|
|
702
|
+
return caps
|
|
703
|
+
if any(h in model for h in _NON_VISION_MODEL_HINTS):
|
|
704
|
+
return caps
|
|
705
|
+
if any(h in model for h in _VISION_MODEL_HINTS):
|
|
706
|
+
caps["vision"] = True
|
|
707
|
+
return caps
|
|
708
|
+
# Unknown model on a vision-capable provider: default to no
|
|
709
|
+
# vision (safe fallback to textual context).
|
|
710
|
+
return caps
|
|
711
|
+
except Exception:
|
|
712
|
+
return caps
|
|
713
|
+
|
|
714
|
+
|
|
715
|
+
def attachment_has_image_payload(att):
|
|
716
|
+
return att is not None and att.kind == "image"
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def encode_image_data_url(path):
|
|
720
|
+
"""Base64 data URL for a local image — used by the native vision
|
|
721
|
+
path when the provider supports it.
|
|
722
|
+
|
|
723
|
+
v0.7.9.0 (requirement #15): the image is NORMALIZED first via the
|
|
724
|
+
vision pipeline — EXIF orientation applied, oversized dimensions
|
|
725
|
+
downscaled with a high-quality filter, exotic formats (TIFF/BMP)
|
|
726
|
+
converted to provider-compatible PNG/JPEG — so a 20 MB photo no
|
|
727
|
+
longer ships as a ~27 MB base64 blob. Quality is preserved: images
|
|
728
|
+
already within budget pass through byte-for-byte."""
|
|
729
|
+
ext = os.path.splitext(str(path))[1].lower()
|
|
730
|
+
mime = {"jpg": "image/jpeg", "jpeg": "image/jpeg", "png": "image/png",
|
|
731
|
+
"gif": "image/gif", "webp": "image/webp", "bmp": "image/bmp",
|
|
732
|
+
"svg": "image/svg+xml", "ico": "image/x-icon"}.get(ext, "image/png")
|
|
733
|
+
try:
|
|
734
|
+
from . import vision as _vision
|
|
735
|
+
n_mime, b64, _meta = _vision.encode_for_model(path)
|
|
736
|
+
return n_mime or mime, b64
|
|
737
|
+
except Exception:
|
|
738
|
+
# Honest fallback: send the original bytes untouched for native
|
|
739
|
+
# formats; anything exotic without Pillow fails at the API with
|
|
740
|
+
# its own message rather than here.
|
|
741
|
+
with open(path, "rb") as f:
|
|
742
|
+
b64 = base64.b64encode(f.read()).decode("ascii")
|
|
743
|
+
return mime, b64
|