de-agentic 2.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. de_agentic-2.1.1/LICENSE +21 -0
  2. de_agentic-2.1.1/PKG-INFO +650 -0
  3. de_agentic-2.1.1/README.md +599 -0
  4. de_agentic-2.1.1/ai/__init__.py +1 -0
  5. de_agentic-2.1.1/ai/assistant.py +124 -0
  6. de_agentic-2.1.1/ai/tasks/__init__.py +1 -0
  7. de_agentic-2.1.1/cli.py +880 -0
  8. de_agentic-2.1.1/core/__init__.py +1 -0
  9. de_agentic-2.1.1/core/config.py +208 -0
  10. de_agentic-2.1.1/core/database.py +201 -0
  11. de_agentic-2.1.1/core/llm.py +490 -0
  12. de_agentic-2.1.1/de_agentic.egg-info/PKG-INFO +650 -0
  13. de_agentic-2.1.1/de_agentic.egg-info/SOURCES.txt +89 -0
  14. de_agentic-2.1.1/de_agentic.egg-info/dependency_links.txt +1 -0
  15. de_agentic-2.1.1/de_agentic.egg-info/entry_points.txt +3 -0
  16. de_agentic-2.1.1/de_agentic.egg-info/requires.txt +34 -0
  17. de_agentic-2.1.1/de_agentic.egg-info/top_level.txt +6 -0
  18. de_agentic-2.1.1/harness/README.md +137 -0
  19. de_agentic-2.1.1/harness/__init__.py +17 -0
  20. de_agentic-2.1.1/harness/__main__.py +10 -0
  21. de_agentic-2.1.1/harness/agent.py +661 -0
  22. de_agentic-2.1.1/harness/artifact.py +483 -0
  23. de_agentic-2.1.1/harness/checkpoints.py +385 -0
  24. de_agentic-2.1.1/harness/cli.py +602 -0
  25. de_agentic-2.1.1/harness/config.py +176 -0
  26. de_agentic-2.1.1/harness/config.yaml +107 -0
  27. de_agentic-2.1.1/harness/graph.py +141 -0
  28. de_agentic-2.1.1/harness/plan.py +612 -0
  29. de_agentic-2.1.1/harness/prompts/fix.md +46 -0
  30. de_agentic-2.1.1/harness/prompts/implement.md +76 -0
  31. de_agentic-2.1.1/harness/prompts/qa.md +56 -0
  32. de_agentic-2.1.1/harness/prompts.py +113 -0
  33. de_agentic-2.1.1/harness/runners.py +286 -0
  34. de_agentic-2.1.1/harness/util.py +163 -0
  35. de_agentic-2.1.1/harness/verify.py +291 -0
  36. de_agentic-2.1.1/operations/__init__.py +1 -0
  37. de_agentic-2.1.1/operations/profiling.py +219 -0
  38. de_agentic-2.1.1/operations/query.py +95 -0
  39. de_agentic-2.1.1/operations/sample_data.py +414 -0
  40. de_agentic-2.1.1/operations/schema.py +96 -0
  41. de_agentic-2.1.1/pyproject.toml +123 -0
  42. de_agentic-2.1.1/setup.cfg +4 -0
  43. de_agentic-2.1.1/src/__init__.py +9 -0
  44. de_agentic-2.1.1/src/agents/__init__.py +6 -0
  45. de_agentic-2.1.1/src/agents/base_agent.py +226 -0
  46. de_agentic-2.1.1/src/agents/de_agent.py +147 -0
  47. de_agentic-2.1.1/src/cli.py +607 -0
  48. de_agentic-2.1.1/src/skills/__init__.py +23 -0
  49. de_agentic-2.1.1/src/skills/architecture_diagram.py +897 -0
  50. de_agentic-2.1.1/src/skills/base_skill.py +65 -0
  51. de_agentic-2.1.1/src/skills/data_profiler.py +126 -0
  52. de_agentic-2.1.1/src/skills/error_analyzer.py +11 -0
  53. de_agentic-2.1.1/src/skills/lineage_tracker.py +11 -0
  54. de_agentic-2.1.1/src/skills/query_optimizer.py +11 -0
  55. de_agentic-2.1.1/src/skills/schema_analyzer.py +148 -0
  56. de_agentic-2.1.1/src/skills/sql_generator.py +96 -0
  57. de_agentic-2.1.1/src/tasks/__init__.py +23 -0
  58. de_agentic-2.1.1/src/tasks/architecture_task.py +94 -0
  59. de_agentic-2.1.1/src/tasks/base_task.py +57 -0
  60. de_agentic-2.1.1/src/tasks/data_ingestion_task.py +84 -0
  61. de_agentic-2.1.1/src/tasks/data_modeling_task.py +82 -0
  62. de_agentic-2.1.1/src/tasks/data_quality_task.py +84 -0
  63. de_agentic-2.1.1/src/tasks/debugging_task.py +75 -0
  64. de_agentic-2.1.1/src/tasks/reverse_engineering_task.py +84 -0
  65. de_agentic-2.1.1/src/tasks/warehousing_task.py +93 -0
  66. de_agentic-2.1.1/src/tools/__init__.py +11 -0
  67. de_agentic-2.1.1/src/tools/database_tool.py +111 -0
  68. de_agentic-2.1.1/src/tools/file_tool.py +109 -0
  69. de_agentic-2.1.1/src/tools/profiling_tool.py +12 -0
  70. de_agentic-2.1.1/src/tools/schema_tool.py +12 -0
  71. de_agentic-2.1.1/src/utils/__init__.py +9 -0
  72. de_agentic-2.1.1/src/utils/config_loader.py +123 -0
  73. de_agentic-2.1.1/src/utils/image_render.py +493 -0
  74. de_agentic-2.1.1/src/utils/llm_factory.py +130 -0
  75. de_agentic-2.1.1/src/utils/logger.py +55 -0
  76. de_agentic-2.1.1/src/workflows/__init__.py +8 -0
  77. de_agentic-2.1.1/src/workflows/base_workflow.py +25 -0
  78. de_agentic-2.1.1/src/workflows/workflow_engine.py +118 -0
  79. de_agentic-2.1.1/tests/test_ai.py +162 -0
  80. de_agentic-2.1.1/tests/test_architecture_diagram_skill.py +299 -0
  81. de_agentic-2.1.1/tests/test_cli_validation.py +131 -0
  82. de_agentic-2.1.1/tests/test_config.py +101 -0
  83. de_agentic-2.1.1/tests/test_core.py +198 -0
  84. de_agentic-2.1.1/tests/test_core_config.py +54 -0
  85. de_agentic-2.1.1/tests/test_harness.py +355 -0
  86. de_agentic-2.1.1/tests/test_integration.py +95 -0
  87. de_agentic-2.1.1/tests/test_llm_selection.py +80 -0
  88. de_agentic-2.1.1/tests/test_operations.py +169 -0
  89. de_agentic-2.1.1/tests/test_phase1_foundation.py +78 -0
  90. de_agentic-2.1.1/tests/test_phase1_integration.py +84 -0
  91. de_agentic-2.1.1/tests/test_sample_data.py +304 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 DE Agentic Team
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,650 @@
1
+ Metadata-Version: 2.4
2
+ Name: de-agentic
3
+ Version: 2.1.1
4
+ Summary: AI-Powered Agentic System for Data Engineers
5
+ Author: DE Agentic Team
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/yourusername/de-agentic
8
+ Project-URL: Documentation, https://github.com/yourusername/de-agentic/docs
9
+ Project-URL: Repository, https://github.com/yourusername/de-agentic
10
+ Keywords: data-engineering,ai-agent,llm,data-quality,etl
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
14
+ Classifier: Programming Language :: Python :: 3.9
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Requires-Python: >=3.9
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: pyyaml>=6.0
22
+ Requires-Dist: pydantic>=2.0
23
+ Provides-Extra: pretty
24
+ Requires-Dist: rich>=13.0; extra == "pretty"
25
+ Provides-Extra: data
26
+ Requires-Dist: duckdb>=0.9; extra == "data"
27
+ Requires-Dist: pandas>=2.0; extra == "data"
28
+ Provides-Extra: dev
29
+ Requires-Dist: pytest>=7.0; extra == "dev"
30
+ Requires-Dist: ruff>=0.6; extra == "dev"
31
+ Requires-Dist: build>=1.0; extra == "dev"
32
+ Requires-Dist: twine>=5.0; extra == "dev"
33
+ Provides-Extra: test
34
+ Requires-Dist: sqlparse>=0.4; extra == "test"
35
+ Requires-Dist: loguru>=0.7; extra == "test"
36
+ Requires-Dist: python-dotenv>=1.0; extra == "test"
37
+ Provides-Extra: legacy
38
+ Requires-Dist: typer>=0.9; extra == "legacy"
39
+ Requires-Dist: rich>=13.0; extra == "legacy"
40
+ Requires-Dist: duckdb>=0.9; extra == "legacy"
41
+ Requires-Dist: pandas>=2.0; extra == "legacy"
42
+ Requires-Dist: langchain>=0.2; extra == "legacy"
43
+ Requires-Dist: langchain-anthropic>=0.1; extra == "legacy"
44
+ Requires-Dist: langchain-openai>=0.1; extra == "legacy"
45
+ Requires-Dist: langchain-community>=0.0; extra == "legacy"
46
+ Requires-Dist: loguru>=0.7; extra == "legacy"
47
+ Requires-Dist: sqlalchemy>=2.0; extra == "legacy"
48
+ Requires-Dist: sqlparse>=0.4; extra == "legacy"
49
+ Requires-Dist: python-dotenv>=1.0; extra == "legacy"
50
+ Dynamic: license-file
51
+
52
+ # 🤖 DE Agentic - AI-Powered Data Engineering Assistant
53
+
54
+ ## ▶ Start here
55
+
56
+ **Thirty seconds, no install:**
57
+
58
+ ```bash
59
+ python3 de.py demo
60
+ ```
61
+
62
+ Creates a sample database, profiles a file, runs a query, reads the schema and
63
+ answers a question — standard library only. No API key, no `pip install`.
64
+
65
+ Prefer a `de` command on your PATH?
66
+
67
+ ```bash
68
+ python3 -m pip install -e . # then: de demo
69
+ ```
70
+
71
+ The four commands:
72
+
73
+ ```bash
74
+ python3 de.py profile customers.csv # profile a file
75
+ python3 de.py query "SELECT * FROM customers LIMIT 5" # read-only SQL
76
+ python3 de.py schema # tables + relationships
77
+ python3 de.py ask "how do I find null values?" # model, or offline help
78
+ ```
79
+
80
+ Plug in a real model (llama.cpp, Ollama, OpenAI or Anthropic) whenever you want:
81
+
82
+ ```bash
83
+ python3 de.py doctor # what is reachable?
84
+ python3 de.py setup # detect one and write .env
85
+ python3 de.py demo --ai # full demo, step 5 answered by your model
86
+ python3 de.py tc # end-to-end test: the model writes SQL, we run it
87
+ ```
88
+
89
+ llama.cpp, Ollama and OpenAI all speak the same OpenAI-compatible
90
+ `/v1/chat/completions`, so one code path covers them. Precedence is
91
+ **environment > `.env` > `config.yaml` > defaults**.
92
+
93
+
94
+ ## Cloud warehouses (ClickHouse Cloud and Snowflake)
95
+
96
+ The default remains a local database. To use a configured warehouse, put non-secret
97
+ connection settings in `config.yaml` and keep the password in an environment variable:
98
+
99
+ ```yaml
100
+ database:
101
+ type: clickhouse_cloud # or: snowflake
102
+ host: your-service.clickhouse.cloud
103
+ port: 8443 # Snowflake commonly uses 443
104
+ database: analytics
105
+ username: default
106
+ password_env: CLICKHOUSE_PASSWORD
107
+ secure: true
108
+ # Snowflake also accepts warehouse, role and schema.
109
+ ```
110
+
111
+ Then export the password and use `--db configured`:
112
+
113
+ ```bash
114
+ export CLICKHOUSE_PASSWORD='...'
115
+ python3 -m pip install clickhouse-connect
116
+ python3 de.py profile --db configured
117
+ python3 de.py schema --db configured
118
+ python3 de.py query "SELECT count() FROM events" --db configured
119
+ ```
120
+
121
+ For Snowflake, install `snowflake-connector-python` and set
122
+ `SNOWFLAKE_PASSWORD` (or your chosen environment-variable name). Passwords are never
123
+ stored in YAML; the YAML stores only the variable name.
124
+
125
+ 📖 **Full getting-started guide: [QUICKSTART.md](QUICKSTART.md)**
126
+
127
+ > **Which docs are current.**
128
+ >
129
+ > Start with **Quick Start** and **Usage** above — they are verified against a
130
+ > real install.
131
+ >
132
+ > `docs/COMMAND_REFERENCE.md`, `docs/TESTING.md`, `docs/QUICK_TEST_REFERENCE.md`,
133
+ > `docs/DEPLOYMENT.md`, `docs/MODEL_SELECTION.md`, `docs/AGENT_MODES_OPTIONS.md`
134
+ > and `docs/FIX_CLI_OPTIONS.md` still document the legacy
135
+ > `python -m src.cli run ...` interface, which needs the optional `legacy`
136
+ > extra. They are kept for reference while that tree is migrated.
137
+ >
138
+ > `docs/MIGRATION.md` is the guide for moving off them, and
139
+ > `docs/QUALITY_REPORT.md` / `docs/USER_TESTING.md` are dated records of the
140
+ > simplification work, not living instructions.
141
+
142
+ ---
143
+
144
+ A comprehensive agentic system designed to assist data engineers with daily tasks including data ingestion, modeling, quality checks, warehousing, debugging, reverse engineering, and architecture simplification.
145
+
146
+ **🆕 Now with Local LLM Support!** Run completely offline with Ollama, or use OpenAI/Anthropic for cloud-based models. Switch between models per command for cost/quality optimization.
147
+
148
+ *!Note: This is the part of README.md of Data-Agent Project, the project is covering for all topics related to data engineering. The project is still in development, but I would like to publish the documentation for reference and idea brainstorming into the to accelerating the working.*
149
+
150
+ ## System Overview
151
+
152
+ ![System Overview](./docs/DE_Agentic_Architecture.png)
153
+
154
+ To understand how the project is going to resolve, run the demo below:
155
+
156
+ > python ./examples/demo_local.py
157
+
158
+ > python ./examples/demo_simple.py
159
+
160
+ ## 🌟 Features
161
+
162
+ ### Task Categories
163
+ - **Data Ingestion**: Automated data loading from various sources (APIs, databases, files)
164
+ - **Data Modeling**: Schema design, ERD generation, normalization checks
165
+ - **Data Quality**: Profiling, validation, anomaly detection
166
+ - **Warehousing**: Pipeline generation, optimization, dbt assistance
167
+ - **Debugging**: Error analysis, log parsing, performance profiling
168
+ - **Reverse Engineering**: Schema extraction, lineage tracking, documentation generation
169
+ - **Architecture**: Pattern detection, optimization recommendations, diagram generation
170
+
171
+ ### Execution Modes
172
+ - **🚀 Core Modes** (Free, Instant, No LLM Required):
173
+ - `query`: Execute SQL queries on local database
174
+ - `reverse`: Analyze database schema and generate documentation
175
+ - `profile`: Profile data files with comprehensive statistics
176
+ - `interactive`: Command-based system for database operations
177
+
178
+ - **🤖 Agent Modes** (LLM-Powered, Flexible Model Selection):
179
+ - `debug`: Analyze errors and provide debugging guidance
180
+ - `quality`: Data quality assessment with recommendations
181
+ - `model`: Data modeling assistance
182
+ - `ingest`: Data ingestion strategy and code generation
183
+ - `warehouse`: Data warehouse architecture guidance
184
+ - `architect`: System architecture analysis and optimization
185
+
186
+ ### Mental Model Principles
187
+ - **ReAct (Reasoning + Acting)**: Combines reasoning and action for better decision-making
188
+ - **Chain of Thought**: Step-by-step reasoning for complex problems
189
+ - **Planning**: Task decomposition and strategic planning
190
+ - **Reflection**: Self-evaluation and continuous improvement
191
+ - **Memory**: Context retention across interactions
192
+
193
+ ### LLM Flexibility
194
+ - **Multiple Providers**: OpenAI (GPT-4, GPT-4o-mini), Ollama (llama2, mistral, codellama), Anthropic (Claude)
195
+ - **Per-Command Model Selection**: Choose optimal model for each task using `--model` parameter
196
+ - **Cost Optimization**: Use cheap models for daily work, premium for critical tasks
197
+ - **Local-First**: Default to Ollama for free, offline operation
198
+
199
+ ## 🚀 Quick Start
200
+
201
+ ### Prerequisites
202
+ - Python 3.9 or newer
203
+ - Optional: [Ollama](https://ollama.ai) or another local model server
204
+ - Optional: an API key for OpenAI or Anthropic
205
+ - Optional: Docker, for the container image
206
+
207
+ ### Install
208
+
209
+ ```bash
210
+ pip install de-agentic
211
+ ```
212
+
213
+ That is the whole thing. The base install depends only on `pyyaml` and
214
+ `pydantic`; `profile`, `query`, `schema` and `init-db` then run on the
215
+ standard library alone. Two extras are available:
216
+
217
+ ```bash
218
+ pip install "de-agentic[pretty]" # rich tables (falls back to plain text)
219
+ pip install "de-agentic[data]" # DuckDB files, Parquet/Excel profiling
220
+ ```
221
+
222
+ `de` is now on your PATH. Check it:
223
+
224
+ ```bash
225
+ de --version
226
+ ```
227
+
228
+ Prefer not to install anything? `python3 de.py demo` runs the same
229
+ walkthrough from a clone using only the standard library.
230
+
231
+ ### Create the sample database
232
+
233
+ The commands below read `sample.sqlite`, which is generated on demand:
234
+
235
+ ```bash
236
+ de init-db # customers, products, orders, order_items
237
+ ```
238
+
239
+ ### Everyday commands
240
+
241
+ ```bash
242
+ de profile customers.csv # column types, nulls, ranges
243
+ de query "SELECT * FROM customers LIMIT 5" # read-only SQL
244
+ de schema # tables, columns, relationships
245
+ de ask "how do I find null values?" # answered by your model
246
+ ```
247
+
248
+ `ask` works offline too: without a model it falls back to local guidance and
249
+ tells you how to configure one.
250
+
251
+ ### Connect a model
252
+
253
+ ```bash
254
+ de doctor # which endpoints are reachable?
255
+ de setup # detect one and write .env
256
+ ```
257
+
258
+ Then re-run `de ask`, or try the full walkthrough with a real model:
259
+
260
+ ```bash
261
+ de demo # end to end
262
+ de demo --ai # same, with step 5 answered by your model
263
+ ```
264
+
265
+ llama.cpp, Ollama and OpenAI all speak the same OpenAI-compatible
266
+ `/v1/chat/completions`, so one code path covers them. Precedence is
267
+ **environment > `.env` > `config.yaml` > defaults**.
268
+
269
+ Per-command overrides are available when you want to switch providers
270
+ without editing configuration:
271
+
272
+ ```bash
273
+ de ask "..." --provider openai --model gpt-4o-mini
274
+ ```
275
+
276
+ ### Using Docker
277
+
278
+ #### Quick Start with Docker Compose
279
+
280
+ ```bash
281
+ # Build and start all services (app + postgres + ollama)
282
+ docker compose up -d
283
+
284
+ # Check status
285
+ docker compose ps
286
+
287
+ # View logs
288
+ docker compose logs -f de-agentic
289
+ ```
290
+
291
+ #### Test the Deployment
292
+
293
+ This stack exists for the legacy agent interface, which needs the optional
294
+ `legacy` extra that the base image does not install. If you only want the
295
+ `de` commands, use the single-container form below instead.
296
+
297
+ ```bash
298
+ # Interactive shell with the CLI (requires -it)
299
+ docker compose exec -it de-agentic de --help
300
+
301
+ # Create the sample database and inspect it
302
+ docker compose exec de-agentic de init-db
303
+ docker compose exec de-agentic de schema
304
+ docker compose exec de-agentic de query "SELECT * FROM customers LIMIT 5"
305
+ ```
306
+
307
+ #### Single Container (No Dependencies)
308
+
309
+ The image installs this package from source, so the commands inside the
310
+ container are the same `de` commands:
311
+
312
+ ```bash
313
+ # Build the image
314
+ docker build -t de-agentic .
315
+
316
+ # Standard-library walkthrough, no model required
317
+ docker run --rm de-agentic de demo
318
+
319
+ # Or use the CLI directly
320
+ docker run --rm de-agentic de schema
321
+ docker run --rm de-agentic de query "SELECT 1 AS x"
322
+
323
+ # Profile a local file (mount volume)
324
+ docker run --rm -v "$PWD/customers.csv:/data/customers.csv" \
325
+ de-agentic de profile /data/customers.csv
326
+ ```
327
+
328
+ Note: the container's default command is `de demo`, not the legacy
329
+ `python -m src.cli`, which needs optional dependencies the base image
330
+ does not install.
331
+
332
+ **📚 For complete deployment guide, see [docs/DEPLOYMENT.md](docs/DEPLOYMENT.md)**
333
+
334
+ ## 📖 Usage
335
+
336
+ ### Command Line Interface
337
+
338
+ #### Core modes (free, instant, no model required)
339
+
340
+ These run on the standard library and need no model server:
341
+
342
+ ```bash
343
+ de init-db # create sample.sqlite
344
+ de query "SELECT COUNT(*) FROM customers" # read-only SQL
345
+ de schema # tables, columns, relationships
346
+ de profile customers.csv # column types, nulls, ranges
347
+ ```
348
+
349
+ #### Model-backed commands
350
+
351
+ `ask` and `demo --ai` route through whichever provider is configured. Use
352
+ `de doctor` to see what is reachable and `de setup` to configure one.
353
+
354
+ ```bash
355
+ de ask "which tables have no primary key?" # answered by your model
356
+ de demo --ai # full walkthrough, model answers
357
+ de tc # end-to-end test of the SQL path
358
+ ```
359
+
360
+ Select a provider per command without editing any config file:
361
+
362
+ ```bash
363
+ de ask "..." --provider openai --model gpt-4o-mini
364
+ de ask "..." --provider ollama --model llama3.2:1b
365
+ ```
366
+
367
+ #### Legacy CLI
368
+
369
+ The earlier agent-mode interface (`run query`, `run debug`, `run reverse`,
370
+ ...) is still in the tree as `src.cli`, but it is mid-migration and needs the
371
+ optional `legacy` extra:
372
+
373
+ ```bash
374
+ pip install "de-agentic[legacy]"
375
+ de-agentic --help
376
+ ```
377
+
378
+ It is not required for anything documented above. New code should target the
379
+ `de` commands.
380
+
381
+ #### Legacy agent modes
382
+
383
+ Available only with the `legacy` extra, via `de-agentic`. Listed for
384
+ reference; these are mid-migration and not covered by CI.
385
+
386
+ ```bash
387
+ de-agentic run debug --error="Connection timeout"
388
+ de-agentic run quality --file=data.csv --model=gpt-4o-mini
389
+ de-agentic run model --description="E-commerce system"
390
+ de-agentic run ingest --source="API" --target="postgres"
391
+ de-agentic run warehouse --requirements="Real-time analytics"
392
+ de-agentic run architect --description="Current pipeline"
393
+ ```
394
+
395
+ Per-command `--model` works the same way here as with `de`.
396
+
397
+ ### Python API
398
+
399
+ ```python
400
+ from src.agents.de_agent import DEAgent
401
+ from src.tasks import DataQualityTask
402
+
403
+ # Initialize agent
404
+ agent = DEAgent()
405
+
406
+ # Run data quality check
407
+ task = DataQualityTask(
408
+ file_path="customers.csv",
409
+ profile=True,
410
+ validate=True
411
+ )
412
+
413
+ result = agent.execute(task)
414
+ print(result)
415
+ ```
416
+
417
+ ### Available Models
418
+
419
+ | Provider | Model | Cost | Best For |
420
+ |----------|-------|------|----------|
421
+ | **OpenAI** | gpt-4o-mini | $0.15/1M tokens | Daily work, fast responses |
422
+ | | gpt-4 | $30/1M tokens | Critical issues, complex problems |
423
+ | | gpt-4-turbo | $10/1M tokens | Balance of cost/quality |
424
+ | **Ollama** | llama2 | Free | General purpose, offline |
425
+ | | mistral | Free | Better code understanding |
426
+ | | codellama | Free | Code generation, debugging |
427
+ | **Anthropic** | claude-3.5-sonnet | $3/1M tokens | Long context, analysis |
428
+ | | claude-3-opus | $15/1M tokens | Premium quality |
429
+
430
+ ## 🏗️ Architecture
431
+
432
+ ```
433
+ de-agentic/
434
+ ├── src/
435
+ │ ├── agents/ # Agent implementations
436
+ │ ├── tasks/ # Task definitions
437
+ │ ├── skills/ # Reusable skills
438
+ │ ├── tools/ # Integration tools
439
+ │ ├── workflows/ # Workflow orchestration
440
+ │ └── utils/ # Utilities
441
+ ├── config/ # Configuration files
442
+ ├── examples/ # Usage examples
443
+ ├── tests/ # Test suite
444
+ └── docs/ # Documentation
445
+ ```
446
+
447
+ ## 🔧 Configuration
448
+
449
+ ### LLM Provider Configuration (.env)
450
+
451
+ ```bash
452
+ # Default: Local Ollama (free, offline)
453
+ LLM_PROVIDER=ollama
454
+ OLLAMA_BASE_URL=http://localhost:11434
455
+ OLLAMA_MODEL=llama2
456
+
457
+ # OpenAI (requires API key)
458
+ LLM_PROVIDER=openai
459
+ OPENAI_API_KEY=sk-proj-...
460
+ OPENAI_MODEL=gpt-4o-mini
461
+
462
+ # Anthropic (requires API key)
463
+ LLM_PROVIDER=anthropic
464
+ ANTHROPIC_API_KEY=sk-ant-...
465
+ ANTHROPIC_MODEL=claude-3.5-sonnet
466
+ ```
467
+
468
+ ### Agent Configuration (config/agent_config.yaml)
469
+
470
+ Edit `config/agent_config.yaml` to customize:
471
+ - Enable/disable specific tasks and skills
472
+ - Adjust mental model parameters
473
+ - Configure database connections
474
+ - Set execution limits
475
+ - Fine-tune agent behavior
476
+
477
+ ### Database Setup
478
+
479
+ The `demo.db` DuckDB database includes:
480
+ - **customers** table: 10 sample customers
481
+ - **orders** table: 30 sample orders
482
+ - **products** table: 10 sample products
483
+ - Total revenue: $6,813.97
484
+
485
+ Recreate anytime with: `de init-db --force`
486
+
487
+ ## 🤝 Contributing
488
+
489
+ Contributions are welcome! Please read our contributing guidelines first.
490
+
491
+ ## 📚 Documentation
492
+
493
+ Comprehensive guides available in the `docs/` directory:
494
+
495
+ - **[QUICKSTART.md](QUICKSTART.md)**: 5-minute getting started guide
496
+ - **[DEPLOYMENT.md](docs/DEPLOYMENT.md)**: Complete deployment and testing guide (Docker, remote, production)
497
+ - **[MODEL_SELECTION.md](docs/MODEL_SELECTION.md)**: Complete guide to choosing and using different LLM models
498
+ - **[COMMAND_REFERENCE.md](docs/COMMAND_REFERENCE.md)**: All CLI commands with examples
499
+ - **[AGENT_MODES_OPTIONS.md](docs/AGENT_MODES_OPTIONS.md)**: Detailed comparison of all 10 execution modes
500
+ - **[DESIGN.md](DESIGN.md)**: Architecture and design principles
501
+
502
+ ## 🎯 Use Cases
503
+
504
+ ### Daily Data Engineering Tasks
505
+ - Query databases without remembering SQL syntax
506
+ - Profile new data files instantly
507
+ - Debug pipeline errors with AI assistance
508
+ - Generate data models from requirements
509
+
510
+ ### Cost Optimization Strategy
511
+ ```bash
512
+ # Development/Testing (free, local)
513
+ --model=llama2
514
+
515
+ # Daily work (cheap, cloud)
516
+ --model=gpt-4o-mini
517
+
518
+ # Critical issues (premium, best quality)
519
+ --model=gpt-4
520
+ ```
521
+
522
+ ### Example workflow
523
+
524
+ ```bash
525
+ # 1. Profile a new file (free, instant, no model)
526
+ de profile new_data.csv
527
+
528
+ # 2. Read the schema
529
+ de schema
530
+
531
+ # 3. Run a check with a cheap model
532
+ de ask "does new_data.csv have quality issues?" --model gpt-4o-mini
533
+
534
+ # 4. Escalate to a stronger model for a hard question
535
+ de ask "how should I fix the encoding problem?" --model gpt-4
536
+
537
+ # 5. Verify with SQL (free, instant)
538
+ de query "SELECT count(*) FROM new_data"
539
+ ```
540
+
541
+ ## 🚀 Recent Updates
542
+
543
+ ### v2.1.1 - Documentation accuracy
544
+ - ✅ Recent Updates now covers the 2.1.x line instead of stopping at v1.2.0
545
+ - ✅ Added the MIT `LICENSE` file that `pyproject.toml` and the README referenced
546
+ - ✅ Legacy reference docs are banner-marked and point at `MIGRATION.md`
547
+ - ✅ Release workflow publishes only for strict `vX.Y.Z` tags
548
+
549
+ ### v2.1.0 - Packaging, CI and install docs
550
+ - ✅ GitHub Actions: lint, a 3.9-3.12 test matrix, and a build that verifies the wheel
551
+ - ✅ Tag-driven release: publish to PyPI via trusted publishing, plus a GitHub release
552
+ - ✅ The wheel now declares its dependencies; `import core.config` works after install
553
+ - ✅ `harness` data files and the `src` package ship correctly
554
+ - ✅ `de doctor` no longer crashes when no model is configured
555
+ - ✅ README and Dockerfile now match what actually installs and runs
556
+
557
+ ### v1.2.0 - Model Selection Feature
558
+ - ✅ Added `--model` parameter for per-command model selection
559
+ - ✅ Auto-provider detection (gpt* → openai, claude* → anthropic)
560
+ - ✅ Cost optimization through flexible model selection
561
+ - ✅ Comprehensive documentation (docs/MODEL_SELECTION.md)
562
+
563
+ ### v1.1.0 - Local Setup & CLI Enhancement
564
+ - ✅ DuckDB local database with sample data
565
+ - ✅ Ollama integration for local LLM support
566
+ - ✅ Legacy execution modes (4 core + 6 agent, behind the `legacy` extra)
567
+ - ✅ Enhanced CLI with rich terminal output
568
+ - ✅ Upgraded to Typer 0.21.0 for better compatibility
569
+
570
+ ### v1.0.0 - Initial Release
571
+ - ✅ 7 task categories with modular architecture
572
+ - ✅ Mental model principles (ReAct, CoT, Planning, Reflection, Memory)
573
+ - ✅ Multiple LLM provider support
574
+ - ✅ Docker deployment support
575
+
576
+ ## 📄 License
577
+
578
+ MIT License - see LICENSE file for details
579
+
580
+ ## 🙏 Acknowledgments
581
+
582
+ Built with:
583
+ - **LangChain** for LLM orchestration and agent framework
584
+ - **Ollama** for local LLM deployment
585
+ - **OpenAI** & **Anthropic** for cloud LLM options
586
+ - **DuckDB** for in-memory analytics and local database
587
+ - **Typer** for CLI framework
588
+ - **Rich** for beautiful terminal output
589
+ - **Pandas** & **NumPy** for data manipulation
590
+ - **SQLAlchemy** for database connectivity
591
+ - **Great Expectations** for data quality (optional)
592
+ - **SQLGlot** for SQL parsing (optional)
593
+
594
+ ## 💡 Pro Tips
595
+
596
+ 1. **Start with Core Modes**: Use `query`, `reverse`, `profile`, `interactive` for free, instant results
597
+ 2. **Cost Control**: Use `--model=gpt-4o-mini` for daily work, save `--model=gpt-4` for critical issues
598
+ 3. **Offline Mode**: Configure `LLM_PROVIDER=ollama` in .env for completely offline operation
599
+ 4. **Model Selection**: See `docs/MODEL_SELECTION.md` for detailed guidance on choosing models
600
+ 5. **Agent Modes**: Read the first response from agent modes (it's usually complete) before any looping occurs
601
+
602
+ ## 🐛 Troubleshooting
603
+
604
+ ### Ollama Issues
605
+ ```bash
606
+ # Check if Ollama is running
607
+ ollama list
608
+
609
+ # Restart Ollama service
610
+ # Windows: Restart from system tray
611
+ # Linux: systemctl restart ollama
612
+ ```
613
+
614
+ ### Model Not Found
615
+ ```bash
616
+ # Pull the model first
617
+ ollama pull llama2
618
+ ollama pull mistral
619
+ ```
620
+
621
+ ### Database Issues
622
+ ```bash
623
+ # Recreate the sample database
624
+ de init-db --force
625
+ ```
626
+
627
+ ### Query Syntax
628
+ ```bash
629
+ # The SQL goes in quotes, after the subcommand
630
+ de query "SELECT * FROM customers"
631
+ ```
632
+
633
+ For more troubleshooting help, see `docs/MODEL_SELECTION.md` and `docs/COMMAND_REFERENCE.md`.
634
+
635
+ ## 📞 Support
636
+
637
+ - 📖 Documentation: See `docs/` directory
638
+ - 🐛 Issues: GitHub Issues
639
+ - 💬 Discussions: GitHub Discussions
640
+
641
+ ---
642
+
643
+ **Ready to get started?**
644
+
645
+ ```bash
646
+ pip install de-agentic
647
+ de init-db
648
+ de demo
649
+ ```
650
+ # llm-based-data-engineering-agents