de-agentic 2.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- de_agentic-2.1.1/LICENSE +21 -0
- de_agentic-2.1.1/PKG-INFO +650 -0
- de_agentic-2.1.1/README.md +599 -0
- de_agentic-2.1.1/ai/__init__.py +1 -0
- de_agentic-2.1.1/ai/assistant.py +124 -0
- de_agentic-2.1.1/ai/tasks/__init__.py +1 -0
- de_agentic-2.1.1/cli.py +880 -0
- de_agentic-2.1.1/core/__init__.py +1 -0
- de_agentic-2.1.1/core/config.py +208 -0
- de_agentic-2.1.1/core/database.py +201 -0
- de_agentic-2.1.1/core/llm.py +490 -0
- de_agentic-2.1.1/de_agentic.egg-info/PKG-INFO +650 -0
- de_agentic-2.1.1/de_agentic.egg-info/SOURCES.txt +89 -0
- de_agentic-2.1.1/de_agentic.egg-info/dependency_links.txt +1 -0
- de_agentic-2.1.1/de_agentic.egg-info/entry_points.txt +3 -0
- de_agentic-2.1.1/de_agentic.egg-info/requires.txt +34 -0
- de_agentic-2.1.1/de_agentic.egg-info/top_level.txt +6 -0
- de_agentic-2.1.1/harness/README.md +137 -0
- de_agentic-2.1.1/harness/__init__.py +17 -0
- de_agentic-2.1.1/harness/__main__.py +10 -0
- de_agentic-2.1.1/harness/agent.py +661 -0
- de_agentic-2.1.1/harness/artifact.py +483 -0
- de_agentic-2.1.1/harness/checkpoints.py +385 -0
- de_agentic-2.1.1/harness/cli.py +602 -0
- de_agentic-2.1.1/harness/config.py +176 -0
- de_agentic-2.1.1/harness/config.yaml +107 -0
- de_agentic-2.1.1/harness/graph.py +141 -0
- de_agentic-2.1.1/harness/plan.py +612 -0
- de_agentic-2.1.1/harness/prompts/fix.md +46 -0
- de_agentic-2.1.1/harness/prompts/implement.md +76 -0
- de_agentic-2.1.1/harness/prompts/qa.md +56 -0
- de_agentic-2.1.1/harness/prompts.py +113 -0
- de_agentic-2.1.1/harness/runners.py +286 -0
- de_agentic-2.1.1/harness/util.py +163 -0
- de_agentic-2.1.1/harness/verify.py +291 -0
- de_agentic-2.1.1/operations/__init__.py +1 -0
- de_agentic-2.1.1/operations/profiling.py +219 -0
- de_agentic-2.1.1/operations/query.py +95 -0
- de_agentic-2.1.1/operations/sample_data.py +414 -0
- de_agentic-2.1.1/operations/schema.py +96 -0
- de_agentic-2.1.1/pyproject.toml +123 -0
- de_agentic-2.1.1/setup.cfg +4 -0
- de_agentic-2.1.1/src/__init__.py +9 -0
- de_agentic-2.1.1/src/agents/__init__.py +6 -0
- de_agentic-2.1.1/src/agents/base_agent.py +226 -0
- de_agentic-2.1.1/src/agents/de_agent.py +147 -0
- de_agentic-2.1.1/src/cli.py +607 -0
- de_agentic-2.1.1/src/skills/__init__.py +23 -0
- de_agentic-2.1.1/src/skills/architecture_diagram.py +897 -0
- de_agentic-2.1.1/src/skills/base_skill.py +65 -0
- de_agentic-2.1.1/src/skills/data_profiler.py +126 -0
- de_agentic-2.1.1/src/skills/error_analyzer.py +11 -0
- de_agentic-2.1.1/src/skills/lineage_tracker.py +11 -0
- de_agentic-2.1.1/src/skills/query_optimizer.py +11 -0
- de_agentic-2.1.1/src/skills/schema_analyzer.py +148 -0
- de_agentic-2.1.1/src/skills/sql_generator.py +96 -0
- de_agentic-2.1.1/src/tasks/__init__.py +23 -0
- de_agentic-2.1.1/src/tasks/architecture_task.py +94 -0
- de_agentic-2.1.1/src/tasks/base_task.py +57 -0
- de_agentic-2.1.1/src/tasks/data_ingestion_task.py +84 -0
- de_agentic-2.1.1/src/tasks/data_modeling_task.py +82 -0
- de_agentic-2.1.1/src/tasks/data_quality_task.py +84 -0
- de_agentic-2.1.1/src/tasks/debugging_task.py +75 -0
- de_agentic-2.1.1/src/tasks/reverse_engineering_task.py +84 -0
- de_agentic-2.1.1/src/tasks/warehousing_task.py +93 -0
- de_agentic-2.1.1/src/tools/__init__.py +11 -0
- de_agentic-2.1.1/src/tools/database_tool.py +111 -0
- de_agentic-2.1.1/src/tools/file_tool.py +109 -0
- de_agentic-2.1.1/src/tools/profiling_tool.py +12 -0
- de_agentic-2.1.1/src/tools/schema_tool.py +12 -0
- de_agentic-2.1.1/src/utils/__init__.py +9 -0
- de_agentic-2.1.1/src/utils/config_loader.py +123 -0
- de_agentic-2.1.1/src/utils/image_render.py +493 -0
- de_agentic-2.1.1/src/utils/llm_factory.py +130 -0
- de_agentic-2.1.1/src/utils/logger.py +55 -0
- de_agentic-2.1.1/src/workflows/__init__.py +8 -0
- de_agentic-2.1.1/src/workflows/base_workflow.py +25 -0
- de_agentic-2.1.1/src/workflows/workflow_engine.py +118 -0
- de_agentic-2.1.1/tests/test_ai.py +162 -0
- de_agentic-2.1.1/tests/test_architecture_diagram_skill.py +299 -0
- de_agentic-2.1.1/tests/test_cli_validation.py +131 -0
- de_agentic-2.1.1/tests/test_config.py +101 -0
- de_agentic-2.1.1/tests/test_core.py +198 -0
- de_agentic-2.1.1/tests/test_core_config.py +54 -0
- de_agentic-2.1.1/tests/test_harness.py +355 -0
- de_agentic-2.1.1/tests/test_integration.py +95 -0
- de_agentic-2.1.1/tests/test_llm_selection.py +80 -0
- de_agentic-2.1.1/tests/test_operations.py +169 -0
- de_agentic-2.1.1/tests/test_phase1_foundation.py +78 -0
- de_agentic-2.1.1/tests/test_phase1_integration.py +84 -0
- de_agentic-2.1.1/tests/test_sample_data.py +304 -0
de_agentic-2.1.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 DE Agentic Team
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,650 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: de-agentic
|
|
3
|
+
Version: 2.1.1
|
|
4
|
+
Summary: AI-Powered Agentic System for Data Engineers
|
|
5
|
+
Author: DE Agentic Team
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/yourusername/de-agentic
|
|
8
|
+
Project-URL: Documentation, https://github.com/yourusername/de-agentic/docs
|
|
9
|
+
Project-URL: Repository, https://github.com/yourusername/de-agentic
|
|
10
|
+
Keywords: data-engineering,ai-agent,llm,data-quality,etl
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: pyyaml>=6.0
|
|
22
|
+
Requires-Dist: pydantic>=2.0
|
|
23
|
+
Provides-Extra: pretty
|
|
24
|
+
Requires-Dist: rich>=13.0; extra == "pretty"
|
|
25
|
+
Provides-Extra: data
|
|
26
|
+
Requires-Dist: duckdb>=0.9; extra == "data"
|
|
27
|
+
Requires-Dist: pandas>=2.0; extra == "data"
|
|
28
|
+
Provides-Extra: dev
|
|
29
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
30
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
31
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
32
|
+
Requires-Dist: twine>=5.0; extra == "dev"
|
|
33
|
+
Provides-Extra: test
|
|
34
|
+
Requires-Dist: sqlparse>=0.4; extra == "test"
|
|
35
|
+
Requires-Dist: loguru>=0.7; extra == "test"
|
|
36
|
+
Requires-Dist: python-dotenv>=1.0; extra == "test"
|
|
37
|
+
Provides-Extra: legacy
|
|
38
|
+
Requires-Dist: typer>=0.9; extra == "legacy"
|
|
39
|
+
Requires-Dist: rich>=13.0; extra == "legacy"
|
|
40
|
+
Requires-Dist: duckdb>=0.9; extra == "legacy"
|
|
41
|
+
Requires-Dist: pandas>=2.0; extra == "legacy"
|
|
42
|
+
Requires-Dist: langchain>=0.2; extra == "legacy"
|
|
43
|
+
Requires-Dist: langchain-anthropic>=0.1; extra == "legacy"
|
|
44
|
+
Requires-Dist: langchain-openai>=0.1; extra == "legacy"
|
|
45
|
+
Requires-Dist: langchain-community>=0.0; extra == "legacy"
|
|
46
|
+
Requires-Dist: loguru>=0.7; extra == "legacy"
|
|
47
|
+
Requires-Dist: sqlalchemy>=2.0; extra == "legacy"
|
|
48
|
+
Requires-Dist: sqlparse>=0.4; extra == "legacy"
|
|
49
|
+
Requires-Dist: python-dotenv>=1.0; extra == "legacy"
|
|
50
|
+
Dynamic: license-file
|
|
51
|
+
|
|
52
|
+
# 🤖 DE Agentic - AI-Powered Data Engineering Assistant
|
|
53
|
+
|
|
54
|
+
## ▶ Start here
|
|
55
|
+
|
|
56
|
+
**Thirty seconds, no install:**
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
python3 de.py demo
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Creates a sample database, profiles a file, runs a query, reads the schema and
|
|
63
|
+
answers a question — standard library only. No API key, no `pip install`.
|
|
64
|
+
|
|
65
|
+
Prefer a `de` command on your PATH?
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
python3 -m pip install -e . # then: de demo
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
The four commands:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
python3 de.py profile customers.csv # profile a file
|
|
75
|
+
python3 de.py query "SELECT * FROM customers LIMIT 5" # read-only SQL
|
|
76
|
+
python3 de.py schema # tables + relationships
|
|
77
|
+
python3 de.py ask "how do I find null values?" # model, or offline help
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Plug in a real model (llama.cpp, Ollama, OpenAI or Anthropic) whenever you want:
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
python3 de.py doctor # what is reachable?
|
|
84
|
+
python3 de.py setup # detect one and write .env
|
|
85
|
+
python3 de.py demo --ai # full demo, step 5 answered by your model
|
|
86
|
+
python3 de.py tc # end-to-end test: the model writes SQL, we run it
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
llama.cpp, Ollama and OpenAI all speak the same OpenAI-compatible
|
|
90
|
+
`/v1/chat/completions`, so one code path covers them. Precedence is
|
|
91
|
+
**environment > `.env` > `config.yaml` > defaults**.
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
## Cloud warehouses (ClickHouse Cloud and Snowflake)
|
|
95
|
+
|
|
96
|
+
The default remains a local database. To use a configured warehouse, put non-secret
|
|
97
|
+
connection settings in `config.yaml` and keep the password in an environment variable:
|
|
98
|
+
|
|
99
|
+
```yaml
|
|
100
|
+
database:
|
|
101
|
+
type: clickhouse_cloud # or: snowflake
|
|
102
|
+
host: your-service.clickhouse.cloud
|
|
103
|
+
port: 8443 # Snowflake commonly uses 443
|
|
104
|
+
database: analytics
|
|
105
|
+
username: default
|
|
106
|
+
password_env: CLICKHOUSE_PASSWORD
|
|
107
|
+
secure: true
|
|
108
|
+
# Snowflake also accepts warehouse, role and schema.
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Then export the password and use `--db configured`:
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
export CLICKHOUSE_PASSWORD='...'
|
|
115
|
+
python3 -m pip install clickhouse-connect
|
|
116
|
+
python3 de.py profile --db configured
|
|
117
|
+
python3 de.py schema --db configured
|
|
118
|
+
python3 de.py query "SELECT count() FROM events" --db configured
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
For Snowflake, install `snowflake-connector-python` and set
|
|
122
|
+
`SNOWFLAKE_PASSWORD` (or your chosen environment-variable name). Passwords are never
|
|
123
|
+
stored in YAML; the YAML stores only the variable name.
|
|
124
|
+
|
|
125
|
+
📖 **Full getting-started guide: [QUICKSTART.md](QUICKSTART.md)**
|
|
126
|
+
|
|
127
|
+
> **Which docs are current.**
|
|
128
|
+
>
|
|
129
|
+
> Start with **Quick Start** and **Usage** above — they are verified against a
|
|
130
|
+
> real install.
|
|
131
|
+
>
|
|
132
|
+
> `docs/COMMAND_REFERENCE.md`, `docs/TESTING.md`, `docs/QUICK_TEST_REFERENCE.md`,
|
|
133
|
+
> `docs/DEPLOYMENT.md`, `docs/MODEL_SELECTION.md`, `docs/AGENT_MODES_OPTIONS.md`
|
|
134
|
+
> and `docs/FIX_CLI_OPTIONS.md` still document the legacy
|
|
135
|
+
> `python -m src.cli run ...` interface, which needs the optional `legacy`
|
|
136
|
+
> extra. They are kept for reference while that tree is migrated.
|
|
137
|
+
>
|
|
138
|
+
> `docs/MIGRATION.md` is the guide for moving off them, and
|
|
139
|
+
> `docs/QUALITY_REPORT.md` / `docs/USER_TESTING.md` are dated records of the
|
|
140
|
+
> simplification work, not living instructions.
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
A comprehensive agentic system designed to assist data engineers with daily tasks including data ingestion, modeling, quality checks, warehousing, debugging, reverse engineering, and architecture simplification.
|
|
145
|
+
|
|
146
|
+
**🆕 Now with Local LLM Support!** Run completely offline with Ollama, or use OpenAI/Anthropic for cloud-based models. Switch between models per command for cost/quality optimization.
|
|
147
|
+
|
|
148
|
+
*!Note: This is the part of README.md of Data-Agent Project, the project is covering for all topics related to data engineering. The project is still in development, but I would like to publish the documentation for reference and idea brainstorming into the to accelerating the working.*
|
|
149
|
+
|
|
150
|
+
## System Overview
|
|
151
|
+
|
|
152
|
+

|
|
153
|
+
|
|
154
|
+
To understand how the project is going to resolve, run the demo below:
|
|
155
|
+
|
|
156
|
+
> python ./examples/demo_local.py
|
|
157
|
+
|
|
158
|
+
> python ./examples/demo_simple.py
|
|
159
|
+
|
|
160
|
+
## 🌟 Features
|
|
161
|
+
|
|
162
|
+
### Task Categories
|
|
163
|
+
- **Data Ingestion**: Automated data loading from various sources (APIs, databases, files)
|
|
164
|
+
- **Data Modeling**: Schema design, ERD generation, normalization checks
|
|
165
|
+
- **Data Quality**: Profiling, validation, anomaly detection
|
|
166
|
+
- **Warehousing**: Pipeline generation, optimization, dbt assistance
|
|
167
|
+
- **Debugging**: Error analysis, log parsing, performance profiling
|
|
168
|
+
- **Reverse Engineering**: Schema extraction, lineage tracking, documentation generation
|
|
169
|
+
- **Architecture**: Pattern detection, optimization recommendations, diagram generation
|
|
170
|
+
|
|
171
|
+
### Execution Modes
|
|
172
|
+
- **🚀 Core Modes** (Free, Instant, No LLM Required):
|
|
173
|
+
- `query`: Execute SQL queries on local database
|
|
174
|
+
- `reverse`: Analyze database schema and generate documentation
|
|
175
|
+
- `profile`: Profile data files with comprehensive statistics
|
|
176
|
+
- `interactive`: Command-based system for database operations
|
|
177
|
+
|
|
178
|
+
- **🤖 Agent Modes** (LLM-Powered, Flexible Model Selection):
|
|
179
|
+
- `debug`: Analyze errors and provide debugging guidance
|
|
180
|
+
- `quality`: Data quality assessment with recommendations
|
|
181
|
+
- `model`: Data modeling assistance
|
|
182
|
+
- `ingest`: Data ingestion strategy and code generation
|
|
183
|
+
- `warehouse`: Data warehouse architecture guidance
|
|
184
|
+
- `architect`: System architecture analysis and optimization
|
|
185
|
+
|
|
186
|
+
### Mental Model Principles
|
|
187
|
+
- **ReAct (Reasoning + Acting)**: Combines reasoning and action for better decision-making
|
|
188
|
+
- **Chain of Thought**: Step-by-step reasoning for complex problems
|
|
189
|
+
- **Planning**: Task decomposition and strategic planning
|
|
190
|
+
- **Reflection**: Self-evaluation and continuous improvement
|
|
191
|
+
- **Memory**: Context retention across interactions
|
|
192
|
+
|
|
193
|
+
### LLM Flexibility
|
|
194
|
+
- **Multiple Providers**: OpenAI (GPT-4, GPT-4o-mini), Ollama (llama2, mistral, codellama), Anthropic (Claude)
|
|
195
|
+
- **Per-Command Model Selection**: Choose optimal model for each task using `--model` parameter
|
|
196
|
+
- **Cost Optimization**: Use cheap models for daily work, premium for critical tasks
|
|
197
|
+
- **Local-First**: Default to Ollama for free, offline operation
|
|
198
|
+
|
|
199
|
+
## 🚀 Quick Start
|
|
200
|
+
|
|
201
|
+
### Prerequisites
|
|
202
|
+
- Python 3.9 or newer
|
|
203
|
+
- Optional: [Ollama](https://ollama.ai) or another local model server
|
|
204
|
+
- Optional: an API key for OpenAI or Anthropic
|
|
205
|
+
- Optional: Docker, for the container image
|
|
206
|
+
|
|
207
|
+
### Install
|
|
208
|
+
|
|
209
|
+
```bash
|
|
210
|
+
pip install de-agentic
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
That is the whole thing. The base install depends only on `pyyaml` and
|
|
214
|
+
`pydantic`; `profile`, `query`, `schema` and `init-db` then run on the
|
|
215
|
+
standard library alone. Two extras are available:
|
|
216
|
+
|
|
217
|
+
```bash
|
|
218
|
+
pip install "de-agentic[pretty]" # rich tables (falls back to plain text)
|
|
219
|
+
pip install "de-agentic[data]" # DuckDB files, Parquet/Excel profiling
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
`de` is now on your PATH. Check it:
|
|
223
|
+
|
|
224
|
+
```bash
|
|
225
|
+
de --version
|
|
226
|
+
```
|
|
227
|
+
|
|
228
|
+
Prefer not to install anything? `python3 de.py demo` runs the same
|
|
229
|
+
walkthrough from a clone using only the standard library.
|
|
230
|
+
|
|
231
|
+
### Create the sample database
|
|
232
|
+
|
|
233
|
+
The commands below read `sample.sqlite`, which is generated on demand:
|
|
234
|
+
|
|
235
|
+
```bash
|
|
236
|
+
de init-db # customers, products, orders, order_items
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
### Everyday commands
|
|
240
|
+
|
|
241
|
+
```bash
|
|
242
|
+
de profile customers.csv # column types, nulls, ranges
|
|
243
|
+
de query "SELECT * FROM customers LIMIT 5" # read-only SQL
|
|
244
|
+
de schema # tables, columns, relationships
|
|
245
|
+
de ask "how do I find null values?" # answered by your model
|
|
246
|
+
```
|
|
247
|
+
|
|
248
|
+
`ask` works offline too: without a model it falls back to local guidance and
|
|
249
|
+
tells you how to configure one.
|
|
250
|
+
|
|
251
|
+
### Connect a model
|
|
252
|
+
|
|
253
|
+
```bash
|
|
254
|
+
de doctor # which endpoints are reachable?
|
|
255
|
+
de setup # detect one and write .env
|
|
256
|
+
```
|
|
257
|
+
|
|
258
|
+
Then re-run `de ask`, or try the full walkthrough with a real model:
|
|
259
|
+
|
|
260
|
+
```bash
|
|
261
|
+
de demo # end to end
|
|
262
|
+
de demo --ai # same, with step 5 answered by your model
|
|
263
|
+
```
|
|
264
|
+
|
|
265
|
+
llama.cpp, Ollama and OpenAI all speak the same OpenAI-compatible
|
|
266
|
+
`/v1/chat/completions`, so one code path covers them. Precedence is
|
|
267
|
+
**environment > `.env` > `config.yaml` > defaults**.
|
|
268
|
+
|
|
269
|
+
Per-command overrides are available when you want to switch providers
|
|
270
|
+
without editing configuration:
|
|
271
|
+
|
|
272
|
+
```bash
|
|
273
|
+
de ask "..." --provider openai --model gpt-4o-mini
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
### Using Docker
|
|
277
|
+
|
|
278
|
+
#### Quick Start with Docker Compose
|
|
279
|
+
|
|
280
|
+
```bash
|
|
281
|
+
# Build and start all services (app + postgres + ollama)
|
|
282
|
+
docker compose up -d
|
|
283
|
+
|
|
284
|
+
# Check status
|
|
285
|
+
docker compose ps
|
|
286
|
+
|
|
287
|
+
# View logs
|
|
288
|
+
docker compose logs -f de-agentic
|
|
289
|
+
```
|
|
290
|
+
|
|
291
|
+
#### Test the Deployment
|
|
292
|
+
|
|
293
|
+
This stack exists for the legacy agent interface, which needs the optional
|
|
294
|
+
`legacy` extra that the base image does not install. If you only want the
|
|
295
|
+
`de` commands, use the single-container form below instead.
|
|
296
|
+
|
|
297
|
+
```bash
|
|
298
|
+
# Interactive shell with the CLI (requires -it)
|
|
299
|
+
docker compose exec -it de-agentic de --help
|
|
300
|
+
|
|
301
|
+
# Create the sample database and inspect it
|
|
302
|
+
docker compose exec de-agentic de init-db
|
|
303
|
+
docker compose exec de-agentic de schema
|
|
304
|
+
docker compose exec de-agentic de query "SELECT * FROM customers LIMIT 5"
|
|
305
|
+
```
|
|
306
|
+
|
|
307
|
+
#### Single Container (No Dependencies)
|
|
308
|
+
|
|
309
|
+
The image installs this package from source, so the commands inside the
|
|
310
|
+
container are the same `de` commands:
|
|
311
|
+
|
|
312
|
+
```bash
|
|
313
|
+
# Build the image
|
|
314
|
+
docker build -t de-agentic .
|
|
315
|
+
|
|
316
|
+
# Standard-library walkthrough, no model required
|
|
317
|
+
docker run --rm de-agentic de demo
|
|
318
|
+
|
|
319
|
+
# Or use the CLI directly
|
|
320
|
+
docker run --rm de-agentic de schema
|
|
321
|
+
docker run --rm de-agentic de query "SELECT 1 AS x"
|
|
322
|
+
|
|
323
|
+
# Profile a local file (mount volume)
|
|
324
|
+
docker run --rm -v "$PWD/customers.csv:/data/customers.csv" \
|
|
325
|
+
de-agentic de profile /data/customers.csv
|
|
326
|
+
```
|
|
327
|
+
|
|
328
|
+
Note: the container's default command is `de demo`, not the legacy
|
|
329
|
+
`python -m src.cli`, which needs optional dependencies the base image
|
|
330
|
+
does not install.
|
|
331
|
+
|
|
332
|
+
**📚 For complete deployment guide, see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md)**
|
|
333
|
+
|
|
334
|
+
## 📖 Usage
|
|
335
|
+
|
|
336
|
+
### Command Line Interface
|
|
337
|
+
|
|
338
|
+
#### Core modes (free, instant, no model required)
|
|
339
|
+
|
|
340
|
+
These run on the standard library and need no model server:
|
|
341
|
+
|
|
342
|
+
```bash
|
|
343
|
+
de init-db # create sample.sqlite
|
|
344
|
+
de query "SELECT COUNT(*) FROM customers" # read-only SQL
|
|
345
|
+
de schema # tables, columns, relationships
|
|
346
|
+
de profile customers.csv # column types, nulls, ranges
|
|
347
|
+
```
|
|
348
|
+
|
|
349
|
+
#### Model-backed commands
|
|
350
|
+
|
|
351
|
+
`ask` and `demo --ai` route through whichever provider is configured. Use
|
|
352
|
+
`de doctor` to see what is reachable and `de setup` to configure one.
|
|
353
|
+
|
|
354
|
+
```bash
|
|
355
|
+
de ask "which tables have no primary key?" # answered by your model
|
|
356
|
+
de demo --ai # full walkthrough, model answers
|
|
357
|
+
de tc # end-to-end test of the SQL path
|
|
358
|
+
```
|
|
359
|
+
|
|
360
|
+
Select a provider per command without editing any config file:
|
|
361
|
+
|
|
362
|
+
```bash
|
|
363
|
+
de ask "..." --provider openai --model gpt-4o-mini
|
|
364
|
+
de ask "..." --provider ollama --model llama3.2:1b
|
|
365
|
+
```
|
|
366
|
+
|
|
367
|
+
#### Legacy CLI
|
|
368
|
+
|
|
369
|
+
The earlier agent-mode interface (`run query`, `run debug`, `run reverse`,
|
|
370
|
+
...) is still in the tree as `src.cli`, but it is mid-migration and needs the
|
|
371
|
+
optional `legacy` extra:
|
|
372
|
+
|
|
373
|
+
```bash
|
|
374
|
+
pip install "de-agentic[legacy]"
|
|
375
|
+
de-agentic --help
|
|
376
|
+
```
|
|
377
|
+
|
|
378
|
+
It is not required for anything documented above. New code should target the
|
|
379
|
+
`de` commands.
|
|
380
|
+
|
|
381
|
+
#### Legacy agent modes
|
|
382
|
+
|
|
383
|
+
Available only with the `legacy` extra, via `de-agentic`. Listed for
|
|
384
|
+
reference; these are mid-migration and not covered by CI.
|
|
385
|
+
|
|
386
|
+
```bash
|
|
387
|
+
de-agentic run debug --error="Connection timeout"
|
|
388
|
+
de-agentic run quality --file=data.csv --model=gpt-4o-mini
|
|
389
|
+
de-agentic run model --description="E-commerce system"
|
|
390
|
+
de-agentic run ingest --source="API" --target="postgres"
|
|
391
|
+
de-agentic run warehouse --requirements="Real-time analytics"
|
|
392
|
+
de-agentic run architect --description="Current pipeline"
|
|
393
|
+
```
|
|
394
|
+
|
|
395
|
+
Per-command `--model` works the same way here as with `de`.
|
|
396
|
+
|
|
397
|
+
### Python API
|
|
398
|
+
|
|
399
|
+
```python
|
|
400
|
+
from src.agents.de_agent import DEAgent
|
|
401
|
+
from src.tasks import DataQualityTask
|
|
402
|
+
|
|
403
|
+
# Initialize agent
|
|
404
|
+
agent = DEAgent()
|
|
405
|
+
|
|
406
|
+
# Run data quality check
|
|
407
|
+
task = DataQualityTask(
|
|
408
|
+
file_path="customers.csv",
|
|
409
|
+
profile=True,
|
|
410
|
+
validate=True
|
|
411
|
+
)
|
|
412
|
+
|
|
413
|
+
result = agent.execute(task)
|
|
414
|
+
print(result)
|
|
415
|
+
```
|
|
416
|
+
|
|
417
|
+
### Available Models
|
|
418
|
+
|
|
419
|
+
| Provider | Model | Cost | Best For |
|
|
420
|
+
|----------|-------|------|----------|
|
|
421
|
+
| **OpenAI** | gpt-4o-mini | $0.15/1M tokens | Daily work, fast responses |
|
|
422
|
+
| | gpt-4 | $30/1M tokens | Critical issues, complex problems |
|
|
423
|
+
| | gpt-4-turbo | $10/1M tokens | Balance of cost/quality |
|
|
424
|
+
| **Ollama** | llama2 | Free | General purpose, offline |
|
|
425
|
+
| | mistral | Free | Better code understanding |
|
|
426
|
+
| | codellama | Free | Code generation, debugging |
|
|
427
|
+
| **Anthropic** | claude-3.5-sonnet | $3/1M tokens | Long context, analysis |
|
|
428
|
+
| | claude-3-opus | $15/1M tokens | Premium quality |
|
|
429
|
+
|
|
430
|
+
## 🏗️ Architecture
|
|
431
|
+
|
|
432
|
+
```
|
|
433
|
+
de-agentic/
|
|
434
|
+
├── src/
|
|
435
|
+
│ ├── agents/ # Agent implementations
|
|
436
|
+
│ ├── tasks/ # Task definitions
|
|
437
|
+
│ ├── skills/ # Reusable skills
|
|
438
|
+
│ ├── tools/ # Integration tools
|
|
439
|
+
│ ├── workflows/ # Workflow orchestration
|
|
440
|
+
│ └── utils/ # Utilities
|
|
441
|
+
├── config/ # Configuration files
|
|
442
|
+
├── examples/ # Usage examples
|
|
443
|
+
├── tests/ # Test suite
|
|
444
|
+
└── docs/ # Documentation
|
|
445
|
+
```
|
|
446
|
+
|
|
447
|
+
## 🔧 Configuration
|
|
448
|
+
|
|
449
|
+
### LLM Provider Configuration (.env)
|
|
450
|
+
|
|
451
|
+
```bash
|
|
452
|
+
# Default: Local Ollama (free, offline)
|
|
453
|
+
LLM_PROVIDER=ollama
|
|
454
|
+
OLLAMA_BASE_URL=http://localhost:11434
|
|
455
|
+
OLLAMA_MODEL=llama2
|
|
456
|
+
|
|
457
|
+
# OpenAI (requires API key)
|
|
458
|
+
LLM_PROVIDER=openai
|
|
459
|
+
OPENAI_API_KEY=sk-proj-...
|
|
460
|
+
OPENAI_MODEL=gpt-4o-mini
|
|
461
|
+
|
|
462
|
+
# Anthropic (requires API key)
|
|
463
|
+
LLM_PROVIDER=anthropic
|
|
464
|
+
ANTHROPIC_API_KEY=sk-ant-...
|
|
465
|
+
ANTHROPIC_MODEL=claude-3.5-sonnet
|
|
466
|
+
```
|
|
467
|
+
|
|
468
|
+
### Agent Configuration (config/agent_config.yaml)
|
|
469
|
+
|
|
470
|
+
Edit `config/agent_config.yaml` to customize:
|
|
471
|
+
- Enable/disable specific tasks and skills
|
|
472
|
+
- Adjust mental model parameters
|
|
473
|
+
- Configure database connections
|
|
474
|
+
- Set execution limits
|
|
475
|
+
- Fine-tune agent behavior
|
|
476
|
+
|
|
477
|
+
### Database Setup
|
|
478
|
+
|
|
479
|
+
The `demo.db` DuckDB database includes:
|
|
480
|
+
- **customers** table: 10 sample customers
|
|
481
|
+
- **orders** table: 30 sample orders
|
|
482
|
+
- **products** table: 10 sample products
|
|
483
|
+
- Total revenue: $6,813.97
|
|
484
|
+
|
|
485
|
+
Recreate anytime with: `de init-db --force`
|
|
486
|
+
|
|
487
|
+
## 🤝 Contributing
|
|
488
|
+
|
|
489
|
+
Contributions are welcome! Please read our contributing guidelines first.
|
|
490
|
+
|
|
491
|
+
## 📚 Documentation
|
|
492
|
+
|
|
493
|
+
Comprehensive guides available in the `docs/` directory:
|
|
494
|
+
|
|
495
|
+
- **[QUICKSTART.md](QUICKSTART.md)**: 5-minute getting started guide
|
|
496
|
+
- **[DEPLOYMENT.md](docs/DEPLOYMENT.md)**: Complete deployment and testing guide (Docker, remote, production)
|
|
497
|
+
- **[MODEL_SELECTION.md](docs/MODEL_SELECTION.md)**: Complete guide to choosing and using different LLM models
|
|
498
|
+
- **[COMMAND_REFERENCE.md](docs/COMMAND_REFERENCE.md)**: All CLI commands with examples
|
|
499
|
+
- **[AGENT_MODES_OPTIONS.md](docs/AGENT_MODES_OPTIONS.md)**: Detailed comparison of all 10 execution modes
|
|
500
|
+
- **[DESIGN.md](DESIGN.md)**: Architecture and design principles
|
|
501
|
+
|
|
502
|
+
## 🎯 Use Cases
|
|
503
|
+
|
|
504
|
+
### Daily Data Engineering Tasks
|
|
505
|
+
- Query databases without remembering SQL syntax
|
|
506
|
+
- Profile new data files instantly
|
|
507
|
+
- Debug pipeline errors with AI assistance
|
|
508
|
+
- Generate data models from requirements
|
|
509
|
+
|
|
510
|
+
### Cost Optimization Strategy
|
|
511
|
+
```bash
|
|
512
|
+
# Development/Testing (free, local)
|
|
513
|
+
--model=llama2
|
|
514
|
+
|
|
515
|
+
# Daily work (cheap, cloud)
|
|
516
|
+
--model=gpt-4o-mini
|
|
517
|
+
|
|
518
|
+
# Critical issues (premium, best quality)
|
|
519
|
+
--model=gpt-4
|
|
520
|
+
```
|
|
521
|
+
|
|
522
|
+
### Example workflow
|
|
523
|
+
|
|
524
|
+
```bash
|
|
525
|
+
# 1. Profile a new file (free, instant, no model)
|
|
526
|
+
de profile new_data.csv
|
|
527
|
+
|
|
528
|
+
# 2. Read the schema
|
|
529
|
+
de schema
|
|
530
|
+
|
|
531
|
+
# 3. Run a check with a cheap model
|
|
532
|
+
de ask "does new_data.csv have quality issues?" --model gpt-4o-mini
|
|
533
|
+
|
|
534
|
+
# 4. Escalate to a stronger model for a hard question
|
|
535
|
+
de ask "how should I fix the encoding problem?" --model gpt-4
|
|
536
|
+
|
|
537
|
+
# 5. Verify with SQL (free, instant)
|
|
538
|
+
de query "SELECT count(*) FROM new_data"
|
|
539
|
+
```
|
|
540
|
+
|
|
541
|
+
## 🚀 Recent Updates
|
|
542
|
+
|
|
543
|
+
### v2.1.1 - Documentation accuracy
|
|
544
|
+
- ✅ Recent Updates now covers the 2.1.x line instead of stopping at v1.2.0
|
|
545
|
+
- ✅ Added the MIT `LICENSE` file that `pyproject.toml` and the README referenced
|
|
546
|
+
- ✅ Legacy reference docs are banner-marked and point at `MIGRATION.md`
|
|
547
|
+
- ✅ Release workflow publishes only for strict `vX.Y.Z` tags
|
|
548
|
+
|
|
549
|
+
### v2.1.0 - Packaging, CI and install docs
|
|
550
|
+
- ✅ GitHub Actions: lint, a 3.9-3.12 test matrix, and a build that verifies the wheel
|
|
551
|
+
- ✅ Tag-driven release: publish to PyPI via trusted publishing, plus a GitHub release
|
|
552
|
+
- ✅ The wheel now declares its dependencies; `import core.config` works after install
|
|
553
|
+
- ✅ `harness` data files and the `src` package ship correctly
|
|
554
|
+
- ✅ `de doctor` no longer crashes when no model is configured
|
|
555
|
+
- ✅ README and Dockerfile now match what actually installs and runs
|
|
556
|
+
|
|
557
|
+
### v1.2.0 - Model Selection Feature
|
|
558
|
+
- ✅ Added `--model` parameter for per-command model selection
|
|
559
|
+
- ✅ Auto-provider detection (gpt* → openai, claude* → anthropic)
|
|
560
|
+
- ✅ Cost optimization through flexible model selection
|
|
561
|
+
- ✅ Comprehensive documentation (docs/MODEL_SELECTION.md)
|
|
562
|
+
|
|
563
|
+
### v1.1.0 - Local Setup & CLI Enhancement
|
|
564
|
+
- ✅ DuckDB local database with sample data
|
|
565
|
+
- ✅ Ollama integration for local LLM support
|
|
566
|
+
- ✅ Legacy execution modes (4 core + 6 agent, behind the `legacy` extra)
|
|
567
|
+
- ✅ Enhanced CLI with rich terminal output
|
|
568
|
+
- ✅ Upgraded to Typer 0.21.0 for better compatibility
|
|
569
|
+
|
|
570
|
+
### v1.0.0 - Initial Release
|
|
571
|
+
- ✅ 7 task categories with modular architecture
|
|
572
|
+
- ✅ Mental model principles (ReAct, CoT, Planning, Reflection, Memory)
|
|
573
|
+
- ✅ Multiple LLM provider support
|
|
574
|
+
- ✅ Docker deployment support
|
|
575
|
+
|
|
576
|
+
## 📄 License
|
|
577
|
+
|
|
578
|
+
MIT License - see LICENSE file for details
|
|
579
|
+
|
|
580
|
+
## 🙏 Acknowledgments
|
|
581
|
+
|
|
582
|
+
Built with:
|
|
583
|
+
- **LangChain** for LLM orchestration and agent framework
|
|
584
|
+
- **Ollama** for local LLM deployment
|
|
585
|
+
- **OpenAI** & **Anthropic** for cloud LLM options
|
|
586
|
+
- **DuckDB** for in-memory analytics and local database
|
|
587
|
+
- **Typer** for CLI framework
|
|
588
|
+
- **Rich** for beautiful terminal output
|
|
589
|
+
- **Pandas** & **NumPy** for data manipulation
|
|
590
|
+
- **SQLAlchemy** for database connectivity
|
|
591
|
+
- **Great Expectations** for data quality (optional)
|
|
592
|
+
- **SQLGlot** for SQL parsing (optional)
|
|
593
|
+
|
|
594
|
+
## 💡 Pro Tips
|
|
595
|
+
|
|
596
|
+
1. **Start with Core Modes**: Use `query`, `reverse`, `profile`, `interactive` for free, instant results
|
|
597
|
+
2. **Cost Control**: Use `--model=gpt-4o-mini` for daily work, save `--model=gpt-4` for critical issues
|
|
598
|
+
3. **Offline Mode**: Configure `LLM_PROVIDER=ollama` in .env for completely offline operation
|
|
599
|
+
4. **Model Selection**: See `docs/MODEL_SELECTION.md` for detailed guidance on choosing models
|
|
600
|
+
5. **Agent Modes**: Read the first response from agent modes (it's usually complete) before any looping occurs
|
|
601
|
+
|
|
602
|
+
## 🐛 Troubleshooting
|
|
603
|
+
|
|
604
|
+
### Ollama Issues
|
|
605
|
+
```bash
|
|
606
|
+
# Check if Ollama is running
|
|
607
|
+
ollama list
|
|
608
|
+
|
|
609
|
+
# Restart Ollama service
|
|
610
|
+
# Windows: Restart from system tray
|
|
611
|
+
# Linux: systemctl restart ollama
|
|
612
|
+
```
|
|
613
|
+
|
|
614
|
+
### Model Not Found
|
|
615
|
+
```bash
|
|
616
|
+
# Pull the model first
|
|
617
|
+
ollama pull llama2
|
|
618
|
+
ollama pull mistral
|
|
619
|
+
```
|
|
620
|
+
|
|
621
|
+
### Database Issues
|
|
622
|
+
```bash
|
|
623
|
+
# Recreate the sample database
|
|
624
|
+
de init-db --force
|
|
625
|
+
```
|
|
626
|
+
|
|
627
|
+
### Query Syntax
|
|
628
|
+
```bash
|
|
629
|
+
# The SQL goes in quotes, after the subcommand
|
|
630
|
+
de query "SELECT * FROM customers"
|
|
631
|
+
```
|
|
632
|
+
|
|
633
|
+
For more troubleshooting help, see `docs/MODEL_SELECTION.md` and `docs/COMMAND_REFERENCE.md`.
|
|
634
|
+
|
|
635
|
+
## 📞 Support
|
|
636
|
+
|
|
637
|
+
- 📖 Documentation: See `docs/` directory
|
|
638
|
+
- 🐛 Issues: GitHub Issues
|
|
639
|
+
- 💬 Discussions: GitHub Discussions
|
|
640
|
+
|
|
641
|
+
---
|
|
642
|
+
|
|
643
|
+
**Ready to get started?**
|
|
644
|
+
|
|
645
|
+
```bash
|
|
646
|
+
pip install de-agentic
|
|
647
|
+
de init-db
|
|
648
|
+
de demo
|
|
649
|
+
```
|
|
650
|
+
# llm-based-data-engineering-agents
|