langchain-decodo 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langchain_decodo-0.1.0/.gitignore +197 -0
- langchain_decodo-0.1.0/LICENSE +21 -0
- langchain_decodo-0.1.0/Makefile +39 -0
- langchain_decodo-0.1.0/PKG-INFO +182 -0
- langchain_decodo-0.1.0/README.md +157 -0
- langchain_decodo-0.1.0/langchain_decodo/__init__.py +10 -0
- langchain_decodo-0.1.0/langchain_decodo/_version.py +1 -0
- langchain_decodo-0.1.0/langchain_decodo/document_loaders.py +267 -0
- langchain_decodo-0.1.0/langchain_decodo/py.typed +0 -0
- langchain_decodo-0.1.0/langchain_decodo/tools.py +446 -0
- langchain_decodo-0.1.0/pyproject.toml +62 -0
- langchain_decodo-0.1.0/tests/__init__.py +0 -0
- langchain_decodo-0.1.0/tests/integration_tests/__init__.py +0 -0
- langchain_decodo-0.1.0/tests/integration_tests/test_tools.py +146 -0
- langchain_decodo-0.1.0/tests/unit_tests/__init__.py +0 -0
- langchain_decodo-0.1.0/tests/unit_tests/test_imports.py +7 -0
- langchain_decodo-0.1.0/tests/unit_tests/test_tools.py +380 -0
- langchain_decodo-0.1.0/uv.lock +2223 -0
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
.vs/
|
|
2
|
+
.claude/
|
|
3
|
+
.idea/
|
|
4
|
+
#Emacs backup
|
|
5
|
+
*~
|
|
6
|
+
# Byte-compiled / optimized / DLL files
|
|
7
|
+
__pycache__/
|
|
8
|
+
*.py[cod]
|
|
9
|
+
*$py.class
|
|
10
|
+
|
|
11
|
+
# C extensions
|
|
12
|
+
*.so
|
|
13
|
+
|
|
14
|
+
# Distribution / packaging
|
|
15
|
+
.Python
|
|
16
|
+
build/
|
|
17
|
+
develop-eggs/
|
|
18
|
+
dist/
|
|
19
|
+
downloads/
|
|
20
|
+
eggs/
|
|
21
|
+
.eggs/
|
|
22
|
+
lib/
|
|
23
|
+
lib64/
|
|
24
|
+
parts/
|
|
25
|
+
sdist/
|
|
26
|
+
var/
|
|
27
|
+
wheels/
|
|
28
|
+
pip-wheel-metadata/
|
|
29
|
+
share/python-wheels/
|
|
30
|
+
*.egg-info/
|
|
31
|
+
.installed.cfg
|
|
32
|
+
*.egg
|
|
33
|
+
MANIFEST
|
|
34
|
+
|
|
35
|
+
# Google GitHub Actions credentials files created by:
|
|
36
|
+
# https://github.com/google-github-actions/auth
|
|
37
|
+
#
|
|
38
|
+
# That action recommends adding this gitignore to prevent accidentally committing keys.
|
|
39
|
+
gha-creds-*.json
|
|
40
|
+
|
|
41
|
+
# PyInstaller
|
|
42
|
+
# Usually these files are written by a python script from a template
|
|
43
|
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
|
44
|
+
*.manifest
|
|
45
|
+
*.spec
|
|
46
|
+
|
|
47
|
+
# Installer logs
|
|
48
|
+
pip-log.txt
|
|
49
|
+
pip-delete-this-directory.txt
|
|
50
|
+
|
|
51
|
+
# Unit test / coverage reports
|
|
52
|
+
htmlcov/
|
|
53
|
+
.tox/
|
|
54
|
+
.nox/
|
|
55
|
+
.coverage
|
|
56
|
+
.coverage.*
|
|
57
|
+
.cache
|
|
58
|
+
nosetests.xml
|
|
59
|
+
coverage.xml
|
|
60
|
+
*.cover
|
|
61
|
+
*.py,cover
|
|
62
|
+
.hypothesis/
|
|
63
|
+
.pytest_cache/
|
|
64
|
+
.codspeed/
|
|
65
|
+
|
|
66
|
+
# Translations
|
|
67
|
+
*.mo
|
|
68
|
+
*.pot
|
|
69
|
+
|
|
70
|
+
# Django stuff:
|
|
71
|
+
*.log
|
|
72
|
+
local_settings.py
|
|
73
|
+
db.sqlite3
|
|
74
|
+
db.sqlite3-journal
|
|
75
|
+
|
|
76
|
+
# Flask stuff:
|
|
77
|
+
instance/
|
|
78
|
+
.webassets-cache
|
|
79
|
+
|
|
80
|
+
# Scrapy stuff:
|
|
81
|
+
.scrapy
|
|
82
|
+
|
|
83
|
+
# PyBuilder
|
|
84
|
+
target/
|
|
85
|
+
|
|
86
|
+
# Jupyter Notebook
|
|
87
|
+
.ipynb_checkpoints
|
|
88
|
+
notebooks/
|
|
89
|
+
|
|
90
|
+
# IPython
|
|
91
|
+
profile_default/
|
|
92
|
+
ipython_config.py
|
|
93
|
+
|
|
94
|
+
# pyenv
|
|
95
|
+
.python-version
|
|
96
|
+
|
|
97
|
+
# pipenv
|
|
98
|
+
# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
|
|
99
|
+
# However, in case of collaboration, if having platform-specific dependencies or dependencies
|
|
100
|
+
# having no cross-platform support, pipenv may install dependencies that don't work, or not
|
|
101
|
+
# install all needed dependencies.
|
|
102
|
+
#Pipfile.lock
|
|
103
|
+
|
|
104
|
+
# PEP 582; used by e.g. github.com/David-OConnor/pyflow
|
|
105
|
+
__pypackages__/
|
|
106
|
+
|
|
107
|
+
# Celery stuff
|
|
108
|
+
celerybeat-schedule
|
|
109
|
+
celerybeat.pid
|
|
110
|
+
|
|
111
|
+
# SageMath parsed files
|
|
112
|
+
*.sage.py
|
|
113
|
+
|
|
114
|
+
# Environments
|
|
115
|
+
.env
|
|
116
|
+
.env.*
|
|
117
|
+
!.env.example
|
|
118
|
+
.envrc
|
|
119
|
+
*.pem
|
|
120
|
+
*.key
|
|
121
|
+
*.crt
|
|
122
|
+
credentials.json
|
|
123
|
+
# SSH private keys (no file extension, so *.key never matches them)
|
|
124
|
+
id_rsa
|
|
125
|
+
id_dsa
|
|
126
|
+
id_ecdsa
|
|
127
|
+
id_ed25519
|
|
128
|
+
*_rsa
|
|
129
|
+
*_dsa
|
|
130
|
+
*_ecdsa
|
|
131
|
+
*_ed25519
|
|
132
|
+
!*.pub
|
|
133
|
+
|
|
134
|
+
# Keystores
|
|
135
|
+
*.p12
|
|
136
|
+
*.pfx
|
|
137
|
+
*.jks
|
|
138
|
+
|
|
139
|
+
# Tokens, cookie jars, and git credential stores
|
|
140
|
+
token.json
|
|
141
|
+
Cookies
|
|
142
|
+
Cookies.db
|
|
143
|
+
cookies.sqlite
|
|
144
|
+
cookies.txt
|
|
145
|
+
.git-credentials
|
|
146
|
+
.venv*
|
|
147
|
+
venv*
|
|
148
|
+
env/
|
|
149
|
+
ENV/
|
|
150
|
+
env.bak/
|
|
151
|
+
|
|
152
|
+
# Spyder project settings
|
|
153
|
+
.spyderproject
|
|
154
|
+
.spyproject
|
|
155
|
+
|
|
156
|
+
# Rope project settings
|
|
157
|
+
.ropeproject
|
|
158
|
+
|
|
159
|
+
# mkdocs documentation
|
|
160
|
+
/site
|
|
161
|
+
|
|
162
|
+
# mypy
|
|
163
|
+
.mypy_cache/
|
|
164
|
+
.mypy_cache_test/
|
|
165
|
+
.dmypy.json
|
|
166
|
+
dmypy.json
|
|
167
|
+
|
|
168
|
+
# Pyre type checker
|
|
169
|
+
.pyre/
|
|
170
|
+
|
|
171
|
+
# macOS display setting files
|
|
172
|
+
.DS_Store
|
|
173
|
+
|
|
174
|
+
# Wandb directory
|
|
175
|
+
wandb/
|
|
176
|
+
|
|
177
|
+
# asdf tool versions
|
|
178
|
+
.tool-versions
|
|
179
|
+
/.ruff_cache/
|
|
180
|
+
|
|
181
|
+
*.pkl
|
|
182
|
+
*.bin
|
|
183
|
+
|
|
184
|
+
# integration test artifacts
|
|
185
|
+
data_map*
|
|
186
|
+
\[('_type', 'fake'), ('stop', None)]
|
|
187
|
+
|
|
188
|
+
# Replit files
|
|
189
|
+
*replit*
|
|
190
|
+
|
|
191
|
+
node_modules
|
|
192
|
+
|
|
193
|
+
prof
|
|
194
|
+
virtualenv/
|
|
195
|
+
scratch/
|
|
196
|
+
|
|
197
|
+
.langgraph_api/
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2024 Decodo
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
|
|
2
|
+
.PHONY: all lint format test tests integration_tests help
|
|
3
|
+
|
|
4
|
+
all: help
|
|
5
|
+
|
|
6
|
+
######################
|
|
7
|
+
# LINTING AND FORMATTING
|
|
8
|
+
######################
|
|
9
|
+
|
|
10
|
+
lint format: PYTHON_FILES=.
|
|
11
|
+
lint:
|
|
12
|
+
ruff check $(PYTHON_FILES)
|
|
13
|
+
ruff format $(PYTHON_FILES) --diff
|
|
14
|
+
mypy $(PYTHON_FILES)
|
|
15
|
+
|
|
16
|
+
format:
|
|
17
|
+
ruff format $(PYTHON_FILES)
|
|
18
|
+
ruff check --fix $(PYTHON_FILES)
|
|
19
|
+
|
|
20
|
+
######################
|
|
21
|
+
# TESTING
|
|
22
|
+
######################
|
|
23
|
+
|
|
24
|
+
tests test:
|
|
25
|
+
pytest tests/unit_tests
|
|
26
|
+
|
|
27
|
+
integration_tests:
|
|
28
|
+
pytest tests/integration_tests
|
|
29
|
+
|
|
30
|
+
######################
|
|
31
|
+
# HELP
|
|
32
|
+
######################
|
|
33
|
+
|
|
34
|
+
help:
|
|
35
|
+
@echo '----'
|
|
36
|
+
@echo 'lint - run linters'
|
|
37
|
+
@echo 'format - run formatters'
|
|
38
|
+
@echo 'tests - run unit tests'
|
|
39
|
+
@echo 'integration_tests - run integration tests (requires DECODO_API_TOKEN)'
|
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: langchain-decodo
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: LangChain integration for the Decodo web scraping API
|
|
5
|
+
Project-URL: Homepage, https://decodo.com
|
|
6
|
+
Project-URL: Documentation, https://developers.decodo.com
|
|
7
|
+
Project-URL: Repository, https://github.com/langchain-ai/langchain/tree/master/libs/partners/decodo
|
|
8
|
+
Author-email: Decodo <support@decodo.com>
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: ai,decodo,langchain,llm,rag,web-scraping
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Internet :: WWW/HTTP
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
21
|
+
Requires-Python: >=3.10
|
|
22
|
+
Requires-Dist: httpx>=0.27.0
|
|
23
|
+
Requires-Dist: langchain-core<2.0.0,>=0.1.0
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
|
|
26
|
+
# langchain-decodo
|
|
27
|
+
|
|
28
|
+
LangChain partner package for the [Decodo](https://decodo.com) web scraping API.
|
|
29
|
+
|
|
30
|
+
Provides two **LangChain tools** and a **document loader** that let LLM agents
|
|
31
|
+
and RAG pipelines fetch live web content without managing proxies, JavaScript
|
|
32
|
+
rendering, or anti-bot protection.
|
|
33
|
+
|
|
34
|
+
## Installation
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install langchain-decodo
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Authentication
|
|
41
|
+
|
|
42
|
+
All classes read your Decodo API token from the `DECODO_API_TOKEN` environment
|
|
43
|
+
variable, or you can pass it explicitly.
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
export DECODO_API_TOKEN="your-decodo-api-token"
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Get a token from the [Decodo Dashboard](https://app.decodo.com).
|
|
50
|
+
|
|
51
|
+
## Components
|
|
52
|
+
|
|
53
|
+
### `DecodoWebScrapeTool`
|
|
54
|
+
|
|
55
|
+
Scrape any URL and return its full content as markdown or plain text.
|
|
56
|
+
Handles JavaScript rendering, CAPTCHAs, and geo-blocking automatically.
|
|
57
|
+
|
|
58
|
+
```python
|
|
59
|
+
from langchain_decodo import DecodoWebScrapeTool
|
|
60
|
+
|
|
61
|
+
tool = DecodoWebScrapeTool() # reads DECODO_API_TOKEN from env
|
|
62
|
+
content = tool.run("https://example.com")
|
|
63
|
+
print(content)
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
Pass explicitly:
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
from pydantic import SecretStr
|
|
70
|
+
from langchain_decodo import DecodoWebScrapeTool
|
|
71
|
+
|
|
72
|
+
tool = DecodoWebScrapeTool(decodo_api_token=SecretStr("YOUR_TOKEN"))
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
### `DecodoSearchTool`
|
|
76
|
+
|
|
77
|
+
Search Google, Amazon, or Reddit and return structured JSON results.
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
from langchain_decodo import DecodoSearchTool
|
|
81
|
+
|
|
82
|
+
tool = DecodoSearchTool()
|
|
83
|
+
|
|
84
|
+
# Google search (default)
|
|
85
|
+
results = tool.run({"query": "LangChain latest release", "engine": "google"})
|
|
86
|
+
|
|
87
|
+
# Amazon product search
|
|
88
|
+
results = tool.run({"query": "Python programming book", "engine": "amazon"})
|
|
89
|
+
|
|
90
|
+
# Reddit discussion search
|
|
91
|
+
results = tool.run({"query": "best web scraping libraries", "engine": "reddit"})
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Returns a JSON string — a list of objects with `content`, `url`, and
|
|
95
|
+
`status_code` fields.
|
|
96
|
+
|
|
97
|
+
Supported engines:
|
|
98
|
+
|
|
99
|
+
| `engine` | Decodo target | Description |
|
|
100
|
+
|---|---|---|
|
|
101
|
+
| `google` | `google_search` | Google SERP |
|
|
102
|
+
| `amazon` | `amazon_search` | Amazon product search |
|
|
103
|
+
| `reddit` | `google_search` + `site:reddit.com` | Reddit via Google |
|
|
104
|
+
|
|
105
|
+
### `DecodoLoader`
|
|
106
|
+
|
|
107
|
+
Load one or more URLs as LangChain `Document` objects for use in RAG pipelines.
|
|
108
|
+
|
|
109
|
+
```python
|
|
110
|
+
from langchain_decodo import DecodoLoader
|
|
111
|
+
|
|
112
|
+
loader = DecodoLoader(
|
|
113
|
+
urls=[
|
|
114
|
+
"https://python.org/about/",
|
|
115
|
+
"https://docs.python.org/3/whatsnew/3.12.html",
|
|
116
|
+
],
|
|
117
|
+
)
|
|
118
|
+
docs = loader.load()
|
|
119
|
+
|
|
120
|
+
for doc in docs:
|
|
121
|
+
print(doc.metadata["url"], "—", len(doc.page_content), "chars")
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Each `Document` has:
|
|
125
|
+
|
|
126
|
+
- `page_content` — scraped text/markdown.
|
|
127
|
+
- `metadata["url"]` — the source URL.
|
|
128
|
+
- `metadata["source"]` — same as `url` (LangChain convention).
|
|
129
|
+
- `metadata["status_code"]` — HTTP status from the target site.
|
|
130
|
+
|
|
131
|
+
## LangChain agent example
|
|
132
|
+
|
|
133
|
+
```python
|
|
134
|
+
from langchain import hub
|
|
135
|
+
from langchain.agents import AgentExecutor, create_react_agent
|
|
136
|
+
from langchain_openai import ChatOpenAI
|
|
137
|
+
from langchain_decodo import DecodoWebScrapeTool, DecodoSearchTool
|
|
138
|
+
|
|
139
|
+
tools = [DecodoWebScrapeTool(), DecodoSearchTool()]
|
|
140
|
+
llm = ChatOpenAI(model="gpt-4o-mini", temperature=0)
|
|
141
|
+
prompt = hub.pull("hwchase17/react")
|
|
142
|
+
|
|
143
|
+
agent = create_react_agent(llm=llm, tools=tools, prompt=prompt)
|
|
144
|
+
executor = AgentExecutor(agent=agent, tools=tools, verbose=True)
|
|
145
|
+
|
|
146
|
+
result = executor.invoke({
|
|
147
|
+
"input": "What is the latest stable version of Python? Check python.org."
|
|
148
|
+
})
|
|
149
|
+
print(result["output"])
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## RAG pipeline example
|
|
153
|
+
|
|
154
|
+
```python
|
|
155
|
+
from langchain.text_splitter import RecursiveCharacterTextSplitter
|
|
156
|
+
from langchain_community.vectorstores import FAISS
|
|
157
|
+
from langchain_openai import ChatOpenAI, OpenAIEmbeddings
|
|
158
|
+
from langchain.chains import RetrievalQA
|
|
159
|
+
from langchain_decodo import DecodoLoader
|
|
160
|
+
|
|
161
|
+
loader = DecodoLoader(urls=["https://python.org/about/"])
|
|
162
|
+
docs = loader.load()
|
|
163
|
+
|
|
164
|
+
splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
|
|
165
|
+
chunks = splitter.split_documents(docs)
|
|
166
|
+
|
|
167
|
+
store = FAISS.from_documents(chunks, OpenAIEmbeddings())
|
|
168
|
+
chain = RetrievalQA.from_chain_type(
|
|
169
|
+
llm=ChatOpenAI(model="gpt-4o-mini"),
|
|
170
|
+
retriever=store.as_retriever(search_kwargs={"k": 4}),
|
|
171
|
+
)
|
|
172
|
+
|
|
173
|
+
result = chain.invoke({"query": "What is Python used for?"})
|
|
174
|
+
print(result["result"])
|
|
175
|
+
```
|
|
176
|
+
|
|
177
|
+
## Links
|
|
178
|
+
|
|
179
|
+
- [Decodo website](https://decodo.com)
|
|
180
|
+
- [Decodo API documentation](https://developers.decodo.com)
|
|
181
|
+
- [Decodo Dashboard](https://app.decodo.com)
|
|
182
|
+
- [LangChain documentation](https://python.langchain.com)
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
# langchain-decodo
|
|
2
|
+
|
|
3
|
+
LangChain partner package for the [Decodo](https://decodo.com) web scraping API.
|
|
4
|
+
|
|
5
|
+
Provides two **LangChain tools** and a **document loader** that let LLM agents
|
|
6
|
+
and RAG pipelines fetch live web content without managing proxies, JavaScript
|
|
7
|
+
rendering, or anti-bot protection.
|
|
8
|
+
|
|
9
|
+
## Installation
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install langchain-decodo
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Authentication
|
|
16
|
+
|
|
17
|
+
All classes read your Decodo API token from the `DECODO_API_TOKEN` environment
|
|
18
|
+
variable, or you can pass it explicitly.
|
|
19
|
+
|
|
20
|
+
```bash
|
|
21
|
+
export DECODO_API_TOKEN="your-decodo-api-token"
|
|
22
|
+
```
|
|
23
|
+
|
|
24
|
+
Get a token from the [Decodo Dashboard](https://app.decodo.com).
|
|
25
|
+
|
|
26
|
+
## Components
|
|
27
|
+
|
|
28
|
+
### `DecodoWebScrapeTool`
|
|
29
|
+
|
|
30
|
+
Scrape any URL and return its full content as markdown or plain text.
|
|
31
|
+
Handles JavaScript rendering, CAPTCHAs, and geo-blocking automatically.
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
from langchain_decodo import DecodoWebScrapeTool
|
|
35
|
+
|
|
36
|
+
tool = DecodoWebScrapeTool() # reads DECODO_API_TOKEN from env
|
|
37
|
+
content = tool.run("https://example.com")
|
|
38
|
+
print(content)
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Pass explicitly:
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from pydantic import SecretStr
|
|
45
|
+
from langchain_decodo import DecodoWebScrapeTool
|
|
46
|
+
|
|
47
|
+
tool = DecodoWebScrapeTool(decodo_api_token=SecretStr("YOUR_TOKEN"))
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
### `DecodoSearchTool`
|
|
51
|
+
|
|
52
|
+
Search Google, Amazon, or Reddit and return structured JSON results.
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from langchain_decodo import DecodoSearchTool
|
|
56
|
+
|
|
57
|
+
tool = DecodoSearchTool()
|
|
58
|
+
|
|
59
|
+
# Google search (default)
|
|
60
|
+
results = tool.run({"query": "LangChain latest release", "engine": "google"})
|
|
61
|
+
|
|
62
|
+
# Amazon product search
|
|
63
|
+
results = tool.run({"query": "Python programming book", "engine": "amazon"})
|
|
64
|
+
|
|
65
|
+
# Reddit discussion search
|
|
66
|
+
results = tool.run({"query": "best web scraping libraries", "engine": "reddit"})
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
Returns a JSON string — a list of objects with `content`, `url`, and
|
|
70
|
+
`status_code` fields.
|
|
71
|
+
|
|
72
|
+
Supported engines:
|
|
73
|
+
|
|
74
|
+
| `engine` | Decodo target | Description |
|
|
75
|
+
|---|---|---|
|
|
76
|
+
| `google` | `google_search` | Google SERP |
|
|
77
|
+
| `amazon` | `amazon_search` | Amazon product search |
|
|
78
|
+
| `reddit` | `google_search` + `site:reddit.com` | Reddit via Google |
|
|
79
|
+
|
|
80
|
+
### `DecodoLoader`
|
|
81
|
+
|
|
82
|
+
Load one or more URLs as LangChain `Document` objects for use in RAG pipelines.
|
|
83
|
+
|
|
84
|
+
```python
|
|
85
|
+
from langchain_decodo import DecodoLoader
|
|
86
|
+
|
|
87
|
+
loader = DecodoLoader(
|
|
88
|
+
urls=[
|
|
89
|
+
"https://python.org/about/",
|
|
90
|
+
"https://docs.python.org/3/whatsnew/3.12.html",
|
|
91
|
+
],
|
|
92
|
+
)
|
|
93
|
+
docs = loader.load()
|
|
94
|
+
|
|
95
|
+
for doc in docs:
|
|
96
|
+
print(doc.metadata["url"], "—", len(doc.page_content), "chars")
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
Each `Document` has:
|
|
100
|
+
|
|
101
|
+
- `page_content` — scraped text/markdown.
|
|
102
|
+
- `metadata["url"]` — the source URL.
|
|
103
|
+
- `metadata["source"]` — same as `url` (LangChain convention).
|
|
104
|
+
- `metadata["status_code"]` — HTTP status from the target site.
|
|
105
|
+
|
|
106
|
+
## LangChain agent example
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
from langchain import hub
|
|
110
|
+
from langchain.agents import AgentExecutor, create_react_agent
|
|
111
|
+
from langchain_openai import ChatOpenAI
|
|
112
|
+
from langchain_decodo import DecodoWebScrapeTool, DecodoSearchTool
|
|
113
|
+
|
|
114
|
+
tools = [DecodoWebScrapeTool(), DecodoSearchTool()]
|
|
115
|
+
llm = ChatOpenAI(model="gpt-4o-mini", temperature=0)
|
|
116
|
+
prompt = hub.pull("hwchase17/react")
|
|
117
|
+
|
|
118
|
+
agent = create_react_agent(llm=llm, tools=tools, prompt=prompt)
|
|
119
|
+
executor = AgentExecutor(agent=agent, tools=tools, verbose=True)
|
|
120
|
+
|
|
121
|
+
result = executor.invoke({
|
|
122
|
+
"input": "What is the latest stable version of Python? Check python.org."
|
|
123
|
+
})
|
|
124
|
+
print(result["output"])
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
## RAG pipeline example
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
from langchain.text_splitter import RecursiveCharacterTextSplitter
|
|
131
|
+
from langchain_community.vectorstores import FAISS
|
|
132
|
+
from langchain_openai import ChatOpenAI, OpenAIEmbeddings
|
|
133
|
+
from langchain.chains import RetrievalQA
|
|
134
|
+
from langchain_decodo import DecodoLoader
|
|
135
|
+
|
|
136
|
+
loader = DecodoLoader(urls=["https://python.org/about/"])
|
|
137
|
+
docs = loader.load()
|
|
138
|
+
|
|
139
|
+
splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
|
|
140
|
+
chunks = splitter.split_documents(docs)
|
|
141
|
+
|
|
142
|
+
store = FAISS.from_documents(chunks, OpenAIEmbeddings())
|
|
143
|
+
chain = RetrievalQA.from_chain_type(
|
|
144
|
+
llm=ChatOpenAI(model="gpt-4o-mini"),
|
|
145
|
+
retriever=store.as_retriever(search_kwargs={"k": 4}),
|
|
146
|
+
)
|
|
147
|
+
|
|
148
|
+
result = chain.invoke({"query": "What is Python used for?"})
|
|
149
|
+
print(result["result"])
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## Links
|
|
153
|
+
|
|
154
|
+
- [Decodo website](https://decodo.com)
|
|
155
|
+
- [Decodo API documentation](https://developers.decodo.com)
|
|
156
|
+
- [Decodo Dashboard](https://app.decodo.com)
|
|
157
|
+
- [LangChain documentation](https://python.langchain.com)
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
from langchain_decodo._version import __version__
|
|
2
|
+
from langchain_decodo.document_loaders import DecodoLoader
|
|
3
|
+
from langchain_decodo.tools import DecodoSearchTool, DecodoWebScrapeTool
|
|
4
|
+
|
|
5
|
+
__all__ = [
|
|
6
|
+
"DecodoLoader",
|
|
7
|
+
"DecodoSearchTool",
|
|
8
|
+
"DecodoWebScrapeTool",
|
|
9
|
+
"__version__",
|
|
10
|
+
]
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|