attachments 0.3.0__tar.gz → 0.5.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- attachments-0.5.2/PKG-INFO +438 -0
- attachments-0.5.2/README.md +342 -0
- attachments-0.5.2/pyproject.toml +258 -0
- attachments-0.5.2/src/attachments/__init__.py +70 -0
- attachments-0.5.2/src/attachments/adapt.py +254 -0
- attachments-0.5.2/src/attachments/core.py +991 -0
- attachments-0.5.2/src/attachments/data/__init__.py +33 -0
- attachments-0.5.2/src/attachments/highest_level_api.py +399 -0
- attachments-0.5.2/src/attachments/load.py +13 -0
- attachments-0.5.2/src/attachments/loaders/__init__.py +17 -0
- attachments-0.5.2/src/attachments/loaders/data/__init__.py +7 -0
- attachments-0.5.2/src/attachments/loaders/data/csv.py +15 -0
- attachments-0.5.2/src/attachments/loaders/documents/__init__.py +14 -0
- attachments-0.5.2/src/attachments/loaders/documents/office.py +37 -0
- attachments-0.5.2/src/attachments/loaders/documents/pdf.py +51 -0
- attachments-0.5.2/src/attachments/loaders/documents/text.py +39 -0
- attachments-0.5.2/src/attachments/loaders/media/__init__.py +9 -0
- attachments-0.5.2/src/attachments/loaders/media/archives.py +48 -0
- attachments-0.5.2/src/attachments/loaders/media/images.py +34 -0
- attachments-0.5.2/src/attachments/loaders/repositories/__init__.py +9 -0
- attachments-0.5.2/src/attachments/loaders/repositories/directories.py +205 -0
- attachments-0.5.2/src/attachments/loaders/repositories/git.py +145 -0
- attachments-0.5.2/src/attachments/loaders/repositories/utils.py +520 -0
- attachments-0.5.2/src/attachments/loaders/web/__init__.py +8 -0
- attachments-0.5.2/src/attachments/loaders/web/urls.py +101 -0
- attachments-0.5.2/src/attachments/matchers.py +73 -0
- attachments-0.5.2/src/attachments/modify.py +361 -0
- attachments-0.5.2/src/attachments/pipelines/__init__.py +205 -0
- attachments-0.5.2/src/attachments/pipelines/csv_processor.py +100 -0
- attachments-0.5.2/src/attachments/pipelines/docx_processor.py +99 -0
- attachments-0.5.2/src/attachments/pipelines/example_processors.py +172 -0
- attachments-0.5.2/src/attachments/pipelines/excel_processor.py +103 -0
- attachments-0.5.2/src/attachments/pipelines/image_processor.py +48 -0
- attachments-0.5.2/src/attachments/pipelines/pdf_processor.py +127 -0
- attachments-0.5.2/src/attachments/pipelines/pptx_processor.py +99 -0
- attachments-0.5.2/src/attachments/pipelines/webpage_processor.py +136 -0
- attachments-0.5.2/src/attachments/present.py +1 -0
- attachments-0.5.2/src/attachments/presenters/__init__.py +15 -0
- attachments-0.5.2/src/attachments/presenters/data/__init__.py +11 -0
- attachments-0.5.2/src/attachments/presenters/data/repositories.py +293 -0
- attachments-0.5.2/src/attachments/presenters/data/summaries.py +77 -0
- attachments-0.5.2/src/attachments/presenters/metadata/__init__.py +7 -0
- attachments-0.5.2/src/attachments/presenters/metadata/info.py +77 -0
- attachments-0.5.2/src/attachments/presenters/text/__init__.py +18 -0
- attachments-0.5.2/src/attachments/presenters/text/markdown.py +301 -0
- attachments-0.5.2/src/attachments/presenters/text/ocr.py +104 -0
- attachments-0.5.2/src/attachments/presenters/text/plain.py +210 -0
- attachments-0.5.2/src/attachments/presenters/text/structured.py +183 -0
- attachments-0.5.2/src/attachments/presenters/visual/__init__.py +7 -0
- attachments-0.5.2/src/attachments/presenters/visual/images.py +862 -0
- attachments-0.5.2/src/attachments/py.typed +4 -0
- attachments-0.5.2/src/attachments/refine.py +555 -0
- attachments-0.5.2/src/attachments/split.py +395 -0
- attachments-0.5.2/src/attachments.egg-info/PKG-INFO +438 -0
- attachments-0.5.2/src/attachments.egg-info/SOURCES.txt +61 -0
- attachments-0.5.2/src/attachments.egg-info/not-zip-safe +1 -0
- attachments-0.5.2/src/attachments.egg-info/requires.txt +76 -0
- attachments-0.5.2/tests/test_api_methods.py +145 -0
- attachments-0.5.2/tests/test_smoke.py +251 -0
- attachments-0.3.0/PKG-INFO +0 -172
- attachments-0.3.0/README.md +0 -132
- attachments-0.3.0/pyproject.toml +0 -68
- attachments-0.3.0/src/attachments/__init__.py +0 -39
- attachments-0.3.0/src/attachments/audio_processing.py +0 -217
- attachments-0.3.0/src/attachments/config.py +0 -35
- attachments-0.3.0/src/attachments/core.py +0 -848
- attachments-0.3.0/src/attachments/detectors.py +0 -144
- attachments-0.3.0/src/attachments/exceptions.py +0 -69
- attachments-0.3.0/src/attachments/image_processing.py +0 -272
- attachments-0.3.0/src/attachments/office_contact_sheet.py +0 -192
- attachments-0.3.0/src/attachments/parsers.py +0 -414
- attachments-0.3.0/src/attachments/renderers.py +0 -140
- attachments-0.3.0/src/attachments/utils.py +0 -389
- attachments-0.3.0/src/attachments.egg-info/PKG-INFO +0 -172
- attachments-0.3.0/src/attachments.egg-info/SOURCES.txt +0 -31
- attachments-0.3.0/src/attachments.egg-info/requires.txt +0 -26
- attachments-0.3.0/tests/test_audio_processing.py +0 -342
- attachments-0.3.0/tests/test_base.py +0 -290
- attachments-0.3.0/tests/test_core_functionality.py +0 -336
- attachments-0.3.0/tests/test_document_parsing.py +0 -87
- attachments-0.3.0/tests/test_html_parsing.py +0 -90
- attachments-0.3.0/tests/test_image_processing.py +0 -193
- attachments-0.3.0/tests/test_indexing_slicing.py +0 -79
- attachments-0.3.0/tests/test_markdown_rendering.py +0 -212
- attachments-0.3.0/tests/test_office_contact_sheet.py +0 -102
- attachments-0.3.0/tests/test_pdf_parsing.py +0 -170
- attachments-0.3.0/tests/test_pptx_parsing.py +0 -119
- attachments-0.3.0/tests/test_simple.py +0 -8
- {attachments-0.3.0 → attachments-0.5.2}/LICENSE +0 -0
- {attachments-0.3.0 → attachments-0.5.2}/setup.cfg +0 -0
- {attachments-0.3.0 → attachments-0.5.2}/src/attachments.egg-info/dependency_links.txt +0 -0
- {attachments-0.3.0 → attachments-0.5.2}/src/attachments.egg-info/top_level.txt +0 -0
|
@@ -0,0 +1,438 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: attachments
|
|
3
|
+
Version: 0.5.2
|
|
4
|
+
Summary: The Python funnel for LLM context - turn any file into model-ready text + images, in one line.
|
|
5
|
+
Author-email: Maxime Rivest <mrive052@gmail.com>
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://maximerivest.github.io/attachments/
|
|
8
|
+
Project-URL: Documentation, https://maximerivest.github.io/attachments/
|
|
9
|
+
Project-URL: Repository, https://github.com/maximrivest/attachments
|
|
10
|
+
Project-URL: Bug Tracker, https://github.com/maximrivest/attachments/issues
|
|
11
|
+
Project-URL: Changelog, https://maximerivest.github.io/attachments/changelog.html
|
|
12
|
+
Keywords: llm,ai,pdf,document,multimodal,openai,claude,context,attachments
|
|
13
|
+
Classifier: Development Status :: 4 - Beta
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
22
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
23
|
+
Classifier: Topic :: Text Processing
|
|
24
|
+
Classifier: Topic :: Multimedia
|
|
25
|
+
Requires-Python: >=3.10
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: requests>=2.25.0
|
|
29
|
+
Requires-Dist: beautifulsoup4>=4.9.0
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest>=8.0.0; extra == "dev"
|
|
32
|
+
Requires-Dist: pytest-randomly>=3.15.0; extra == "dev"
|
|
33
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "dev"
|
|
34
|
+
Requires-Dist: black>=24.0.0; extra == "dev"
|
|
35
|
+
Requires-Dist: flake8>=7.0.0; extra == "dev"
|
|
36
|
+
Requires-Dist: mypy>=1.8.0; extra == "dev"
|
|
37
|
+
Requires-Dist: pre-commit>=3.6.0; extra == "dev"
|
|
38
|
+
Provides-Extra: docs
|
|
39
|
+
Requires-Dist: mystmd>=1.2.0; extra == "docs"
|
|
40
|
+
Requires-Dist: jupyter>=1.0.0; extra == "docs"
|
|
41
|
+
Requires-Dist: jupytext>=1.16.0; extra == "docs"
|
|
42
|
+
Requires-Dist: sphinx>=7.0.0; extra == "docs"
|
|
43
|
+
Requires-Dist: sphinx-autodoc2>=0.5.0; extra == "docs"
|
|
44
|
+
Provides-Extra: common
|
|
45
|
+
Requires-Dist: pandas>=2.0.0; extra == "common"
|
|
46
|
+
Requires-Dist: Pillow>=10.0.0; extra == "common"
|
|
47
|
+
Requires-Dist: pdfplumber>=0.10.0; extra == "common"
|
|
48
|
+
Requires-Dist: python-pptx>=0.6.0; extra == "common"
|
|
49
|
+
Requires-Dist: python-docx>=1.1.0; extra == "common"
|
|
50
|
+
Requires-Dist: openpyxl>=3.1.0; extra == "common"
|
|
51
|
+
Requires-Dist: pypdfium2>=4.0.0; extra == "common"
|
|
52
|
+
Provides-Extra: pdf
|
|
53
|
+
Requires-Dist: pdfplumber>=0.7.0; extra == "pdf"
|
|
54
|
+
Requires-Dist: pypdf>=5.0.0; extra == "pdf"
|
|
55
|
+
Requires-Dist: pypdfium2>=4.26.0; extra == "pdf"
|
|
56
|
+
Provides-Extra: pdf-agpl
|
|
57
|
+
Requires-Dist: PyMuPDF>=1.24.0; extra == "pdf-agpl"
|
|
58
|
+
Provides-Extra: extended
|
|
59
|
+
Requires-Dist: python-magic>=0.4.27; extra == "extended"
|
|
60
|
+
Requires-Dist: mammoth>=1.6.0; extra == "extended"
|
|
61
|
+
Requires-Dist: odfpy>=1.4.1; extra == "extended"
|
|
62
|
+
Requires-Dist: pillow-heif>=0.22.0; extra == "extended"
|
|
63
|
+
Requires-Dist: pydub>=0.25.1; extra == "extended"
|
|
64
|
+
Requires-Dist: SpeechRecognition>=3.10.0; extra == "extended"
|
|
65
|
+
Requires-Dist: markitdown[all]>=0.1.1; extra == "extended"
|
|
66
|
+
Requires-Dist: pytesseract>=0.3.10; extra == "extended"
|
|
67
|
+
Provides-Extra: browser
|
|
68
|
+
Requires-Dist: playwright>=1.43.0; extra == "browser"
|
|
69
|
+
Requires-Dist: selenium>=4.15.0; extra == "browser"
|
|
70
|
+
Provides-Extra: all
|
|
71
|
+
Requires-Dist: pandas>=1.3.0; extra == "all"
|
|
72
|
+
Requires-Dist: Pillow>=8.0.0; extra == "all"
|
|
73
|
+
Requires-Dist: pdfplumber>=0.7.0; extra == "all"
|
|
74
|
+
Requires-Dist: python-pptx>=0.6.21; extra == "all"
|
|
75
|
+
Requires-Dist: python-docx>=0.8.11; extra == "all"
|
|
76
|
+
Requires-Dist: openpyxl>=3.0.9; extra == "all"
|
|
77
|
+
Requires-Dist: python-magic>=0.4.27; extra == "all"
|
|
78
|
+
Requires-Dist: mammoth>=1.6.0; extra == "all"
|
|
79
|
+
Requires-Dist: odfpy>=1.4.1; extra == "all"
|
|
80
|
+
Requires-Dist: pillow-heif>=0.22.0; extra == "all"
|
|
81
|
+
Requires-Dist: pydub>=0.25.1; extra == "all"
|
|
82
|
+
Requires-Dist: SpeechRecognition>=3.10.0; extra == "all"
|
|
83
|
+
Requires-Dist: markitdown[all]>=0.1.1; extra == "all"
|
|
84
|
+
Requires-Dist: pytesseract>=0.3.10; extra == "all"
|
|
85
|
+
Requires-Dist: pypdf>=5.0.0; extra == "all"
|
|
86
|
+
Requires-Dist: pypdfium2>=4.26.0; extra == "all"
|
|
87
|
+
Requires-Dist: playwright>=1.43.0; extra == "all"
|
|
88
|
+
Requires-Dist: selenium>=4.15.0; extra == "all"
|
|
89
|
+
Provides-Extra: test
|
|
90
|
+
Requires-Dist: pytest>=8.0.0; extra == "test"
|
|
91
|
+
Requires-Dist: pytest-randomly>=3.15.0; extra == "test"
|
|
92
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == "test"
|
|
93
|
+
Requires-Dist: tox>=4.0.0; extra == "test"
|
|
94
|
+
Requires-Dist: coverage[toml]>=7.4.0; extra == "test"
|
|
95
|
+
Dynamic: license-file
|
|
96
|
+
|
|
97
|
+
# Attachments – the Python funnel for LLM context
|
|
98
|
+
|
|
99
|
+
### Turn *any* file into model-ready text + images, in one line
|
|
100
|
+
|
|
101
|
+
Most users will not have to learn anything more than: `Attachments("path/to/file.pdf")`
|
|
102
|
+
|
|
103
|
+
> **TL;DR**
|
|
104
|
+
> ```bash
|
|
105
|
+
> pip install attachments
|
|
106
|
+
> ```
|
|
107
|
+
> ```python
|
|
108
|
+
> from attachments import Attachments
|
|
109
|
+
> ctx = Attachments("https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/sample.pdf",
|
|
110
|
+
> "https://github.com/MaximeRivest/attachments/raw/refs/heads/main/src/attachments/data/sample_multipage.pptx")
|
|
111
|
+
> llm_ready_text = str(ctx) # all extracted text, already "prompt-engineered"
|
|
112
|
+
> llm_ready_images = ctx.images # list[str] – base64 PNGs
|
|
113
|
+
> ```
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
Attachments aims to be **the** community funnel from *file → text + base64 images* for LLMs.
|
|
117
|
+
Stop re-writing that plumbing in every project – contribute your *loader / transform / renderer* plugin instead!
|
|
118
|
+
|
|
119
|
+
## Quick-start ⚡
|
|
120
|
+
|
|
121
|
+
```bash
|
|
122
|
+
pip install attachments
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
### Try it now with sample files
|
|
126
|
+
|
|
127
|
+
```python
|
|
128
|
+
from attachments import Attachments
|
|
129
|
+
from attachments.data import get_sample_path
|
|
130
|
+
|
|
131
|
+
# Option 1: Use included sample files (works offline)
|
|
132
|
+
pdf_path = get_sample_path("sample.pdf")
|
|
133
|
+
txt_path = get_sample_path("sample.txt")
|
|
134
|
+
ctx = Attachments(pdf_path, txt_path)
|
|
135
|
+
|
|
136
|
+
print(str(ctx)) # Pretty text view
|
|
137
|
+
print(len(ctx.images)) # Number of extracted images
|
|
138
|
+
|
|
139
|
+
# Try different file types
|
|
140
|
+
docx_path = get_sample_path("test_document.docx")
|
|
141
|
+
csv_path = get_sample_path("test.csv")
|
|
142
|
+
json_path = get_sample_path("sample.json")
|
|
143
|
+
|
|
144
|
+
ctx = Attachments(docx_path, csv_path, json_path)
|
|
145
|
+
print(f"Processed {len(ctx)} files: Word doc, CSV data, and JSON")
|
|
146
|
+
|
|
147
|
+
# Option 2: Use URLs (same API, works with any URL)
|
|
148
|
+
ctx = Attachments(
|
|
149
|
+
"https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/sample.pdf",
|
|
150
|
+
"https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/sample_multipage.pptx"
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
print(str(ctx)) # Pretty text view
|
|
154
|
+
print(len(ctx.images)) # Number of extracted images
|
|
155
|
+
```
|
|
156
|
+
|
|
157
|
+
### Advanced usage with DSL
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
from attachments import Attachments
|
|
161
|
+
|
|
162
|
+
a = Attachments(
|
|
163
|
+
"https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/" \
|
|
164
|
+
"sample_multipage.pptx[3-5]"
|
|
165
|
+
)
|
|
166
|
+
print(a) # pretty text view
|
|
167
|
+
len(a.images) # 👉 base64 PNG list
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
### Send to OpenAI
|
|
171
|
+
|
|
172
|
+
```bash
|
|
173
|
+
pip install openai
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
```python
|
|
177
|
+
from openai import OpenAI
|
|
178
|
+
from attachments import Attachments
|
|
179
|
+
|
|
180
|
+
pdf = Attachments("https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/sample_multipage.pptx[3-5]")
|
|
181
|
+
|
|
182
|
+
client = OpenAI()
|
|
183
|
+
resp = client.chat.completions.create(
|
|
184
|
+
model="gpt-4.1-nano",
|
|
185
|
+
messages=pdf.openai_chat("Analyse the following document:")
|
|
186
|
+
)
|
|
187
|
+
print(resp.choices[0].message.content)
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
or with the response API
|
|
191
|
+
|
|
192
|
+
```python
|
|
193
|
+
from openai import OpenAI
|
|
194
|
+
from attachments import Attachments
|
|
195
|
+
|
|
196
|
+
pdf = Attachments("https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/sample_multipage.pptx[3-5]")
|
|
197
|
+
|
|
198
|
+
client = OpenAI()
|
|
199
|
+
resp = client.responses.create(
|
|
200
|
+
input=pdf.openai_responses("Analyse the following document:"),
|
|
201
|
+
model="gpt-4.1-nano"
|
|
202
|
+
)
|
|
203
|
+
print(resp.output[0].content[0].text)
|
|
204
|
+
```
|
|
205
|
+
|
|
206
|
+
### Send to Anthropic / Claude
|
|
207
|
+
|
|
208
|
+
```bash
|
|
209
|
+
pip install anthropic
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
```python
|
|
213
|
+
import anthropic
|
|
214
|
+
from attachments import Attachments
|
|
215
|
+
|
|
216
|
+
pptx = Attachments("https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/sample_multipage.pptx[3-5]")
|
|
217
|
+
|
|
218
|
+
msg = anthropic.Anthropic().messages.create(
|
|
219
|
+
model="claude-3-5-haiku-20241022",
|
|
220
|
+
max_tokens=8_192,
|
|
221
|
+
messages=pptx.claude("Analyse the slides:")
|
|
222
|
+
)
|
|
223
|
+
print(msg.content)
|
|
224
|
+
```
|
|
225
|
+
|
|
226
|
+
### DSPy Integration
|
|
227
|
+
|
|
228
|
+
```bash
|
|
229
|
+
pip install dspy
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
```python
|
|
233
|
+
import dspy
|
|
234
|
+
from attachments import Attachments
|
|
235
|
+
|
|
236
|
+
dspy.configure(lm=dspy.LM('openai/gpt-4.1-nano'))
|
|
237
|
+
rag = dspy.ChainOfThought("question, document -> answer")
|
|
238
|
+
|
|
239
|
+
result = rag(
|
|
240
|
+
question="What is the main message of the document?",
|
|
241
|
+
document=Attachments("https://github.com/MaximeRivest/attachments/raw/main/src/attachments/data/sample_multipage.pptx[3-5]").dspy()
|
|
242
|
+
)
|
|
243
|
+
print(result.answer)
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
### Advanced Pipeline Processing
|
|
247
|
+
|
|
248
|
+
For power users, use the full grammar system with composable pipelines:
|
|
249
|
+
|
|
250
|
+
```python
|
|
251
|
+
from attachments import attach, load, modify, present, refine, adapt
|
|
252
|
+
|
|
253
|
+
# Custom processing pipeline
|
|
254
|
+
result = (attach("document.pdf[pages:1-5]")
|
|
255
|
+
| load.pdf_to_pdfplumber
|
|
256
|
+
| modify.pages
|
|
257
|
+
| present.markdown + present.images
|
|
258
|
+
| refine.add_headers | refine.truncate
|
|
259
|
+
| adapt.claude("Analyze this content"))
|
|
260
|
+
|
|
261
|
+
# Web scraping pipeline
|
|
262
|
+
title = (attach("https://en.wikipedia.org/wiki/Llama[select:title]")
|
|
263
|
+
| load.url_to_bs4
|
|
264
|
+
| modify.select
|
|
265
|
+
| present.text)
|
|
266
|
+
|
|
267
|
+
# Reusable processors
|
|
268
|
+
csv_analyzer = (load.csv_to_pandas
|
|
269
|
+
| modify.limit
|
|
270
|
+
| present.head + present.summary + present.metadata
|
|
271
|
+
| refine.add_headers)
|
|
272
|
+
|
|
273
|
+
# Use as function
|
|
274
|
+
result = csv_analyzer("data.csv[limit:100]")
|
|
275
|
+
analysis = result.claude("What patterns do you see?")
|
|
276
|
+
```
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
|
|
280
|
+
---
|
|
281
|
+
|
|
282
|
+
## DSL cheatsheet 📝
|
|
283
|
+
|
|
284
|
+
| Piece | Example | Notes |
|
|
285
|
+
| ------------------------- | ------------------------- | --------------------------------------------- |
|
|
286
|
+
| **Select pages / slides** | `report.pdf[1,3-5,-1]` | Supports ranges, negative indices, `N` = last |
|
|
287
|
+
| **Image transforms** | `photo.jpg[rotate:90]` | Any token implemented by a `Transform` plugin |
|
|
288
|
+
| **Data-frame summary** | `table.csv[summary:true]` | Ships with a quick `df.describe()` renderer |
|
|
289
|
+
| **Web content selection** | `url[select:title]` | CSS selectors for web scraping |
|
|
290
|
+
| **Web element highlighting** | `url[select:h1][viewport:1920x1080]` | Visual highlighting in screenshots |
|
|
291
|
+
| **Image processing** | `image.jpg[crop:100,100,400,300][rotate:45]` | Chain multiple transformations |
|
|
292
|
+
| **Content filtering** | `doc.pdf[format:plain][images:false]` | Control text/image extraction |
|
|
293
|
+
| **Repository processing** | `repo[files:false][ignore:standard]` | Smart codebase analysis |
|
|
294
|
+
|
|
295
|
+
---
|
|
296
|
+
|
|
297
|
+
## Supported formats (out of the box)
|
|
298
|
+
|
|
299
|
+
* **Docs**: PDF, PowerPoint (`.pptx`), CSV, TXT, Markdown, HTML
|
|
300
|
+
* **Images**: PNG, JPEG, BMP, GIF, WEBP, HEIC/HEIF, …
|
|
301
|
+
* **Web**: URLs with BeautifulSoup parsing and CSS selection
|
|
302
|
+
* **Archives**: ZIP files → image collections with tiling
|
|
303
|
+
* **Repositories**: Git repos with smart ignore patterns
|
|
304
|
+
* **Data**: CSV with pandas, JSON
|
|
305
|
+
|
|
306
|
+
---
|
|
307
|
+
|
|
308
|
+
## Advanced Examples 🧩
|
|
309
|
+
|
|
310
|
+
### **Multimodal Document Processing**
|
|
311
|
+
```python
|
|
312
|
+
# PDF with image tiling and analysis
|
|
313
|
+
result = Attachments("report.pdf[tile:2x3][resize_images:400]")
|
|
314
|
+
analysis = result.claude("Analyze both text and visual elements")
|
|
315
|
+
|
|
316
|
+
# Multiple file types in one context
|
|
317
|
+
ctx = Attachments("report.pdf", "data.csv", "chart.png")
|
|
318
|
+
comparison = ctx.openai("Compare insights across all documents")
|
|
319
|
+
```
|
|
320
|
+
|
|
321
|
+
### **Repository Analysis**
|
|
322
|
+
```python
|
|
323
|
+
# Codebase structure only
|
|
324
|
+
structure = Attachments("./my-project[mode:structure]")
|
|
325
|
+
|
|
326
|
+
# Full codebase analysis with smart filtering
|
|
327
|
+
codebase = Attachments("./my-project[ignore:standard]")
|
|
328
|
+
review = codebase.claude("Review this code for best practices")
|
|
329
|
+
|
|
330
|
+
# Custom ignore patterns
|
|
331
|
+
filtered = Attachments("./app[ignore:.env,*.log,node_modules]")
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
### **Web Scraping with CSS Selectors**
|
|
335
|
+
```python
|
|
336
|
+
# Extract specific content from web pages
|
|
337
|
+
title = Attachments("https://example.com[select:h1]")
|
|
338
|
+
paragraphs = Attachments("https://example.com[select:p]")
|
|
339
|
+
|
|
340
|
+
# Visual highlighting in screenshots with animations
|
|
341
|
+
highlighted = Attachments("https://example.com[select:h1][viewport:1920x1080]")
|
|
342
|
+
# Creates screenshot with animated highlighting of h1 elements
|
|
343
|
+
|
|
344
|
+
# Multiple element highlighting with counters
|
|
345
|
+
multi_select = Attachments("https://example.com[select:h1, .important][fullpage:true]")
|
|
346
|
+
# Shows "H1 (1/3)", "DIV (2/3)", etc. with different colors for multiple selections
|
|
347
|
+
|
|
348
|
+
# Pipeline approach for complex scraping
|
|
349
|
+
content = (attach("https://en.wikipedia.org/wiki/Llama[select:p]")
|
|
350
|
+
| load.url_to_bs4
|
|
351
|
+
| modify.select
|
|
352
|
+
| present.text
|
|
353
|
+
| refine.truncate)
|
|
354
|
+
```
|
|
355
|
+
|
|
356
|
+
### **Image Processing Chains**
|
|
357
|
+
```python
|
|
358
|
+
# HEIC support with transformations
|
|
359
|
+
processed = Attachments("IMG_2160.HEIC[crop:100,100,400,300][rotate:90]")
|
|
360
|
+
|
|
361
|
+
# Batch image processing with tiling
|
|
362
|
+
collage = Attachments("photos.zip[tile:3x2][resize_images:800]")
|
|
363
|
+
description = collage.claude("Describe this image collage")
|
|
364
|
+
```
|
|
365
|
+
|
|
366
|
+
### **Data Analysis Workflows**
|
|
367
|
+
```python
|
|
368
|
+
# Rich data presentation
|
|
369
|
+
data_summary = Attachments("sales_data.csv[limit:1000][summary:true]")
|
|
370
|
+
|
|
371
|
+
# Pipeline for complex data processing
|
|
372
|
+
result = (attach("data.csv[limit:500]")
|
|
373
|
+
| load.csv_to_pandas
|
|
374
|
+
| modify.limit
|
|
375
|
+
| present.head + present.summary + present.metadata
|
|
376
|
+
| refine.add_headers
|
|
377
|
+
| adapt.claude("What trends do you see?"))
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
---
|
|
381
|
+
|
|
382
|
+
## Extending 🧩
|
|
383
|
+
|
|
384
|
+
```python
|
|
385
|
+
# my_ocr_renderer.py
|
|
386
|
+
from attachments.plugin_api import register_plugin, requires
|
|
387
|
+
from attachments.core import Renderer
|
|
388
|
+
|
|
389
|
+
@register_plugin("renderer_text", priority=50)
|
|
390
|
+
@requires("pytesseract", "PIL")
|
|
391
|
+
class ImageOCR(Renderer):
|
|
392
|
+
content_type = "text"
|
|
393
|
+
|
|
394
|
+
def match(self, obj):
|
|
395
|
+
from PIL import Image
|
|
396
|
+
return isinstance(obj, Image.Image)
|
|
397
|
+
|
|
398
|
+
def render(self, obj, meta):
|
|
399
|
+
import pytesseract
|
|
400
|
+
return pytesseract.image_to_string(obj)
|
|
401
|
+
```
|
|
402
|
+
|
|
403
|
+
1. Put the file somewhere on disk.
|
|
404
|
+
2. `export ATTACHMENTS_PLUGIN_PATH=/abs/path/to/dir_or_file`
|
|
405
|
+
3. `import attachments` – your plugin is auto-discovered, no code changes.
|
|
406
|
+
|
|
407
|
+
---
|
|
408
|
+
|
|
409
|
+
## API reference (essentials)
|
|
410
|
+
|
|
411
|
+
| Object / method | Description |
|
|
412
|
+
| ----------------------- | --------------------------------------------------------------- |
|
|
413
|
+
| `Attachments(*sources)` | Many `Attachment` objects flattened into one container |
|
|
414
|
+
| `Attachments.text` | All text joined with blank lines |
|
|
415
|
+
| `Attachments.images` | Flat list of base64 PNGs |
|
|
416
|
+
| `.claude(prompt="")` | Claude API format with image support |
|
|
417
|
+
| `.openai_chat(prompt="")` | OpenAI Chat Completions API format |
|
|
418
|
+
| `.openai_responses(prompt="")` | OpenAI Responses API format (different structure) |
|
|
419
|
+
| `.openai(prompt="")` | Alias for openai_chat (backwards compatibility) |
|
|
420
|
+
| `.dspy()` | DSPy BaseType-compatible objects |
|
|
421
|
+
|
|
422
|
+
### Grammar System (Advanced)
|
|
423
|
+
|
|
424
|
+
| Namespace | Purpose | Examples |
|
|
425
|
+
|-----------|---------|----------|
|
|
426
|
+
| `load.*` | File format → objects | `pdf_to_pdfplumber`, `csv_to_pandas`, `url_to_bs4` |
|
|
427
|
+
| `modify.*` | Transform objects | `pages`, `limit`, `select`, `crop`, `rotate` |
|
|
428
|
+
| `present.*` | Extract content | `text`, `images`, `markdown`, `summary` |
|
|
429
|
+
| `refine.*` | Post-process | `truncate`, `add_headers`, `tile_images` |
|
|
430
|
+
| `adapt.*` | Format for APIs | `claude`, `openai`, `dspy` |
|
|
431
|
+
|
|
432
|
+
**Operators**: `|` (sequential), `+` (additive)
|
|
433
|
+
|
|
434
|
+
---
|
|
435
|
+
|
|
436
|
+
### Roadmap
|
|
437
|
+
|
|
438
|
+
Join us – file an issue or open a PR! 🚀
|