langchain-decodo 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,197 @@
1
+ .vs/
2
+ .claude/
3
+ .idea/
4
+ #Emacs backup
5
+ *~
6
+ # Byte-compiled / optimized / DLL files
7
+ __pycache__/
8
+ *.py[cod]
9
+ *$py.class
10
+
11
+ # C extensions
12
+ *.so
13
+
14
+ # Distribution / packaging
15
+ .Python
16
+ build/
17
+ develop-eggs/
18
+ dist/
19
+ downloads/
20
+ eggs/
21
+ .eggs/
22
+ lib/
23
+ lib64/
24
+ parts/
25
+ sdist/
26
+ var/
27
+ wheels/
28
+ pip-wheel-metadata/
29
+ share/python-wheels/
30
+ *.egg-info/
31
+ .installed.cfg
32
+ *.egg
33
+ MANIFEST
34
+
35
+ # Google GitHub Actions credentials files created by:
36
+ # https://github.com/google-github-actions/auth
37
+ #
38
+ # That action recommends adding this gitignore to prevent accidentally committing keys.
39
+ gha-creds-*.json
40
+
41
+ # PyInstaller
42
+ # Usually these files are written by a python script from a template
43
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
44
+ *.manifest
45
+ *.spec
46
+
47
+ # Installer logs
48
+ pip-log.txt
49
+ pip-delete-this-directory.txt
50
+
51
+ # Unit test / coverage reports
52
+ htmlcov/
53
+ .tox/
54
+ .nox/
55
+ .coverage
56
+ .coverage.*
57
+ .cache
58
+ nosetests.xml
59
+ coverage.xml
60
+ *.cover
61
+ *.py,cover
62
+ .hypothesis/
63
+ .pytest_cache/
64
+ .codspeed/
65
+
66
+ # Translations
67
+ *.mo
68
+ *.pot
69
+
70
+ # Django stuff:
71
+ *.log
72
+ local_settings.py
73
+ db.sqlite3
74
+ db.sqlite3-journal
75
+
76
+ # Flask stuff:
77
+ instance/
78
+ .webassets-cache
79
+
80
+ # Scrapy stuff:
81
+ .scrapy
82
+
83
+ # PyBuilder
84
+ target/
85
+
86
+ # Jupyter Notebook
87
+ .ipynb_checkpoints
88
+ notebooks/
89
+
90
+ # IPython
91
+ profile_default/
92
+ ipython_config.py
93
+
94
+ # pyenv
95
+ .python-version
96
+
97
+ # pipenv
98
+ # According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control.
99
+ # However, in case of collaboration, if having platform-specific dependencies or dependencies
100
+ # having no cross-platform support, pipenv may install dependencies that don't work, or not
101
+ # install all needed dependencies.
102
+ #Pipfile.lock
103
+
104
+ # PEP 582; used by e.g. github.com/David-OConnor/pyflow
105
+ __pypackages__/
106
+
107
+ # Celery stuff
108
+ celerybeat-schedule
109
+ celerybeat.pid
110
+
111
+ # SageMath parsed files
112
+ *.sage.py
113
+
114
+ # Environments
115
+ .env
116
+ .env.*
117
+ !.env.example
118
+ .envrc
119
+ *.pem
120
+ *.key
121
+ *.crt
122
+ credentials.json
123
+ # SSH private keys (no file extension, so *.key never matches them)
124
+ id_rsa
125
+ id_dsa
126
+ id_ecdsa
127
+ id_ed25519
128
+ *_rsa
129
+ *_dsa
130
+ *_ecdsa
131
+ *_ed25519
132
+ !*.pub
133
+
134
+ # Keystores
135
+ *.p12
136
+ *.pfx
137
+ *.jks
138
+
139
+ # Tokens, cookie jars, and git credential stores
140
+ token.json
141
+ Cookies
142
+ Cookies.db
143
+ cookies.sqlite
144
+ cookies.txt
145
+ .git-credentials
146
+ .venv*
147
+ venv*
148
+ env/
149
+ ENV/
150
+ env.bak/
151
+
152
+ # Spyder project settings
153
+ .spyderproject
154
+ .spyproject
155
+
156
+ # Rope project settings
157
+ .ropeproject
158
+
159
+ # mkdocs documentation
160
+ /site
161
+
162
+ # mypy
163
+ .mypy_cache/
164
+ .mypy_cache_test/
165
+ .dmypy.json
166
+ dmypy.json
167
+
168
+ # Pyre type checker
169
+ .pyre/
170
+
171
+ # macOS display setting files
172
+ .DS_Store
173
+
174
+ # Wandb directory
175
+ wandb/
176
+
177
+ # asdf tool versions
178
+ .tool-versions
179
+ /.ruff_cache/
180
+
181
+ *.pkl
182
+ *.bin
183
+
184
+ # integration test artifacts
185
+ data_map*
186
+ \[('_type', 'fake'), ('stop', None)]
187
+
188
+ # Replit files
189
+ *replit*
190
+
191
+ node_modules
192
+
193
+ prof
194
+ virtualenv/
195
+ scratch/
196
+
197
+ .langgraph_api/
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2024 Decodo
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,39 @@
1
+
2
+ .PHONY: all lint format test tests integration_tests help
3
+
4
+ all: help
5
+
6
+ ######################
7
+ # LINTING AND FORMATTING
8
+ ######################
9
+
10
+ lint format: PYTHON_FILES=.
11
+ lint:
12
+ ruff check $(PYTHON_FILES)
13
+ ruff format $(PYTHON_FILES) --diff
14
+ mypy $(PYTHON_FILES)
15
+
16
+ format:
17
+ ruff format $(PYTHON_FILES)
18
+ ruff check --fix $(PYTHON_FILES)
19
+
20
+ ######################
21
+ # TESTING
22
+ ######################
23
+
24
+ tests test:
25
+ pytest tests/unit_tests
26
+
27
+ integration_tests:
28
+ pytest tests/integration_tests
29
+
30
+ ######################
31
+ # HELP
32
+ ######################
33
+
34
+ help:
35
+ @echo '----'
36
+ @echo 'lint - run linters'
37
+ @echo 'format - run formatters'
38
+ @echo 'tests - run unit tests'
39
+ @echo 'integration_tests - run integration tests (requires DECODO_API_TOKEN)'
@@ -0,0 +1,182 @@
1
+ Metadata-Version: 2.5
2
+ Name: langchain-decodo
3
+ Version: 0.1.0
4
+ Summary: LangChain integration for the Decodo web scraping API
5
+ Project-URL: Homepage, https://decodo.com
6
+ Project-URL: Documentation, https://developers.decodo.com
7
+ Project-URL: Repository, https://github.com/langchain-ai/langchain/tree/master/libs/partners/decodo
8
+ Author-email: Decodo <support@decodo.com>
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: ai,decodo,langchain,llm,rag,web-scraping
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Internet :: WWW/HTTP
20
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
21
+ Requires-Python: >=3.10
22
+ Requires-Dist: httpx>=0.27.0
23
+ Requires-Dist: langchain-core<2.0.0,>=0.1.0
24
+ Description-Content-Type: text/markdown
25
+
26
+ # langchain-decodo
27
+
28
+ LangChain partner package for the [Decodo](https://decodo.com) web scraping API.
29
+
30
+ Provides two **LangChain tools** and a **document loader** that let LLM agents
31
+ and RAG pipelines fetch live web content without managing proxies, JavaScript
32
+ rendering, or anti-bot protection.
33
+
34
+ ## Installation
35
+
36
+ ```bash
37
+ pip install langchain-decodo
38
+ ```
39
+
40
+ ## Authentication
41
+
42
+ All classes read your Decodo API token from the `DECODO_API_TOKEN` environment
43
+ variable, or you can pass it explicitly.
44
+
45
+ ```bash
46
+ export DECODO_API_TOKEN="your-decodo-api-token"
47
+ ```
48
+
49
+ Get a token from the [Decodo Dashboard](https://app.decodo.com).
50
+
51
+ ## Components
52
+
53
+ ### `DecodoWebScrapeTool`
54
+
55
+ Scrape any URL and return its full content as markdown or plain text.
56
+ Handles JavaScript rendering, CAPTCHAs, and geo-blocking automatically.
57
+
58
+ ```python
59
+ from langchain_decodo import DecodoWebScrapeTool
60
+
61
+ tool = DecodoWebScrapeTool() # reads DECODO_API_TOKEN from env
62
+ content = tool.run("https://example.com")
63
+ print(content)
64
+ ```
65
+
66
+ Pass explicitly:
67
+
68
+ ```python
69
+ from pydantic import SecretStr
70
+ from langchain_decodo import DecodoWebScrapeTool
71
+
72
+ tool = DecodoWebScrapeTool(decodo_api_token=SecretStr("YOUR_TOKEN"))
73
+ ```
74
+
75
+ ### `DecodoSearchTool`
76
+
77
+ Search Google, Amazon, or Reddit and return structured JSON results.
78
+
79
+ ```python
80
+ from langchain_decodo import DecodoSearchTool
81
+
82
+ tool = DecodoSearchTool()
83
+
84
+ # Google search (default)
85
+ results = tool.run({"query": "LangChain latest release", "engine": "google"})
86
+
87
+ # Amazon product search
88
+ results = tool.run({"query": "Python programming book", "engine": "amazon"})
89
+
90
+ # Reddit discussion search
91
+ results = tool.run({"query": "best web scraping libraries", "engine": "reddit"})
92
+ ```
93
+
94
+ Returns a JSON string — a list of objects with `content`, `url`, and
95
+ `status_code` fields.
96
+
97
+ Supported engines:
98
+
99
+ | `engine` | Decodo target | Description |
100
+ |---|---|---|
101
+ | `google` | `google_search` | Google SERP |
102
+ | `amazon` | `amazon_search` | Amazon product search |
103
+ | `reddit` | `google_search` + `site:reddit.com` | Reddit via Google |
104
+
105
+ ### `DecodoLoader`
106
+
107
+ Load one or more URLs as LangChain `Document` objects for use in RAG pipelines.
108
+
109
+ ```python
110
+ from langchain_decodo import DecodoLoader
111
+
112
+ loader = DecodoLoader(
113
+ urls=[
114
+ "https://python.org/about/",
115
+ "https://docs.python.org/3/whatsnew/3.12.html",
116
+ ],
117
+ )
118
+ docs = loader.load()
119
+
120
+ for doc in docs:
121
+ print(doc.metadata["url"], "—", len(doc.page_content), "chars")
122
+ ```
123
+
124
+ Each `Document` has:
125
+
126
+ - `page_content` — scraped text/markdown.
127
+ - `metadata["url"]` — the source URL.
128
+ - `metadata["source"]` — same as `url` (LangChain convention).
129
+ - `metadata["status_code"]` — HTTP status from the target site.
130
+
131
+ ## LangChain agent example
132
+
133
+ ```python
134
+ from langchain import hub
135
+ from langchain.agents import AgentExecutor, create_react_agent
136
+ from langchain_openai import ChatOpenAI
137
+ from langchain_decodo import DecodoWebScrapeTool, DecodoSearchTool
138
+
139
+ tools = [DecodoWebScrapeTool(), DecodoSearchTool()]
140
+ llm = ChatOpenAI(model="gpt-4o-mini", temperature=0)
141
+ prompt = hub.pull("hwchase17/react")
142
+
143
+ agent = create_react_agent(llm=llm, tools=tools, prompt=prompt)
144
+ executor = AgentExecutor(agent=agent, tools=tools, verbose=True)
145
+
146
+ result = executor.invoke({
147
+ "input": "What is the latest stable version of Python? Check python.org."
148
+ })
149
+ print(result["output"])
150
+ ```
151
+
152
+ ## RAG pipeline example
153
+
154
+ ```python
155
+ from langchain.text_splitter import RecursiveCharacterTextSplitter
156
+ from langchain_community.vectorstores import FAISS
157
+ from langchain_openai import ChatOpenAI, OpenAIEmbeddings
158
+ from langchain.chains import RetrievalQA
159
+ from langchain_decodo import DecodoLoader
160
+
161
+ loader = DecodoLoader(urls=["https://python.org/about/"])
162
+ docs = loader.load()
163
+
164
+ splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
165
+ chunks = splitter.split_documents(docs)
166
+
167
+ store = FAISS.from_documents(chunks, OpenAIEmbeddings())
168
+ chain = RetrievalQA.from_chain_type(
169
+ llm=ChatOpenAI(model="gpt-4o-mini"),
170
+ retriever=store.as_retriever(search_kwargs={"k": 4}),
171
+ )
172
+
173
+ result = chain.invoke({"query": "What is Python used for?"})
174
+ print(result["result"])
175
+ ```
176
+
177
+ ## Links
178
+
179
+ - [Decodo website](https://decodo.com)
180
+ - [Decodo API documentation](https://developers.decodo.com)
181
+ - [Decodo Dashboard](https://app.decodo.com)
182
+ - [LangChain documentation](https://python.langchain.com)
@@ -0,0 +1,157 @@
1
+ # langchain-decodo
2
+
3
+ LangChain partner package for the [Decodo](https://decodo.com) web scraping API.
4
+
5
+ Provides two **LangChain tools** and a **document loader** that let LLM agents
6
+ and RAG pipelines fetch live web content without managing proxies, JavaScript
7
+ rendering, or anti-bot protection.
8
+
9
+ ## Installation
10
+
11
+ ```bash
12
+ pip install langchain-decodo
13
+ ```
14
+
15
+ ## Authentication
16
+
17
+ All classes read your Decodo API token from the `DECODO_API_TOKEN` environment
18
+ variable, or you can pass it explicitly.
19
+
20
+ ```bash
21
+ export DECODO_API_TOKEN="your-decodo-api-token"
22
+ ```
23
+
24
+ Get a token from the [Decodo Dashboard](https://app.decodo.com).
25
+
26
+ ## Components
27
+
28
+ ### `DecodoWebScrapeTool`
29
+
30
+ Scrape any URL and return its full content as markdown or plain text.
31
+ Handles JavaScript rendering, CAPTCHAs, and geo-blocking automatically.
32
+
33
+ ```python
34
+ from langchain_decodo import DecodoWebScrapeTool
35
+
36
+ tool = DecodoWebScrapeTool() # reads DECODO_API_TOKEN from env
37
+ content = tool.run("https://example.com")
38
+ print(content)
39
+ ```
40
+
41
+ Pass explicitly:
42
+
43
+ ```python
44
+ from pydantic import SecretStr
45
+ from langchain_decodo import DecodoWebScrapeTool
46
+
47
+ tool = DecodoWebScrapeTool(decodo_api_token=SecretStr("YOUR_TOKEN"))
48
+ ```
49
+
50
+ ### `DecodoSearchTool`
51
+
52
+ Search Google, Amazon, or Reddit and return structured JSON results.
53
+
54
+ ```python
55
+ from langchain_decodo import DecodoSearchTool
56
+
57
+ tool = DecodoSearchTool()
58
+
59
+ # Google search (default)
60
+ results = tool.run({"query": "LangChain latest release", "engine": "google"})
61
+
62
+ # Amazon product search
63
+ results = tool.run({"query": "Python programming book", "engine": "amazon"})
64
+
65
+ # Reddit discussion search
66
+ results = tool.run({"query": "best web scraping libraries", "engine": "reddit"})
67
+ ```
68
+
69
+ Returns a JSON string — a list of objects with `content`, `url`, and
70
+ `status_code` fields.
71
+
72
+ Supported engines:
73
+
74
+ | `engine` | Decodo target | Description |
75
+ |---|---|---|
76
+ | `google` | `google_search` | Google SERP |
77
+ | `amazon` | `amazon_search` | Amazon product search |
78
+ | `reddit` | `google_search` + `site:reddit.com` | Reddit via Google |
79
+
80
+ ### `DecodoLoader`
81
+
82
+ Load one or more URLs as LangChain `Document` objects for use in RAG pipelines.
83
+
84
+ ```python
85
+ from langchain_decodo import DecodoLoader
86
+
87
+ loader = DecodoLoader(
88
+ urls=[
89
+ "https://python.org/about/",
90
+ "https://docs.python.org/3/whatsnew/3.12.html",
91
+ ],
92
+ )
93
+ docs = loader.load()
94
+
95
+ for doc in docs:
96
+ print(doc.metadata["url"], "—", len(doc.page_content), "chars")
97
+ ```
98
+
99
+ Each `Document` has:
100
+
101
+ - `page_content` — scraped text/markdown.
102
+ - `metadata["url"]` — the source URL.
103
+ - `metadata["source"]` — same as `url` (LangChain convention).
104
+ - `metadata["status_code"]` — HTTP status from the target site.
105
+
106
+ ## LangChain agent example
107
+
108
+ ```python
109
+ from langchain import hub
110
+ from langchain.agents import AgentExecutor, create_react_agent
111
+ from langchain_openai import ChatOpenAI
112
+ from langchain_decodo import DecodoWebScrapeTool, DecodoSearchTool
113
+
114
+ tools = [DecodoWebScrapeTool(), DecodoSearchTool()]
115
+ llm = ChatOpenAI(model="gpt-4o-mini", temperature=0)
116
+ prompt = hub.pull("hwchase17/react")
117
+
118
+ agent = create_react_agent(llm=llm, tools=tools, prompt=prompt)
119
+ executor = AgentExecutor(agent=agent, tools=tools, verbose=True)
120
+
121
+ result = executor.invoke({
122
+ "input": "What is the latest stable version of Python? Check python.org."
123
+ })
124
+ print(result["output"])
125
+ ```
126
+
127
+ ## RAG pipeline example
128
+
129
+ ```python
130
+ from langchain.text_splitter import RecursiveCharacterTextSplitter
131
+ from langchain_community.vectorstores import FAISS
132
+ from langchain_openai import ChatOpenAI, OpenAIEmbeddings
133
+ from langchain.chains import RetrievalQA
134
+ from langchain_decodo import DecodoLoader
135
+
136
+ loader = DecodoLoader(urls=["https://python.org/about/"])
137
+ docs = loader.load()
138
+
139
+ splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)
140
+ chunks = splitter.split_documents(docs)
141
+
142
+ store = FAISS.from_documents(chunks, OpenAIEmbeddings())
143
+ chain = RetrievalQA.from_chain_type(
144
+ llm=ChatOpenAI(model="gpt-4o-mini"),
145
+ retriever=store.as_retriever(search_kwargs={"k": 4}),
146
+ )
147
+
148
+ result = chain.invoke({"query": "What is Python used for?"})
149
+ print(result["result"])
150
+ ```
151
+
152
+ ## Links
153
+
154
+ - [Decodo website](https://decodo.com)
155
+ - [Decodo API documentation](https://developers.decodo.com)
156
+ - [Decodo Dashboard](https://app.decodo.com)
157
+ - [LangChain documentation](https://python.langchain.com)
@@ -0,0 +1,10 @@
1
+ from langchain_decodo._version import __version__
2
+ from langchain_decodo.document_loaders import DecodoLoader
3
+ from langchain_decodo.tools import DecodoSearchTool, DecodoWebScrapeTool
4
+
5
+ __all__ = [
6
+ "DecodoLoader",
7
+ "DecodoSearchTool",
8
+ "DecodoWebScrapeTool",
9
+ "__version__",
10
+ ]
@@ -0,0 +1 @@
1
+ __version__ = "0.1.0"