de-agentic 2.1.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ai/__init__.py +1 -0
- ai/assistant.py +124 -0
- ai/tasks/__init__.py +1 -0
- cli.py +880 -0
- core/__init__.py +1 -0
- core/config.py +208 -0
- core/database.py +201 -0
- core/llm.py +490 -0
- de_agentic-2.1.1.dist-info/METADATA +650 -0
- de_agentic-2.1.1.dist-info/RECORD +73 -0
- de_agentic-2.1.1.dist-info/WHEEL +5 -0
- de_agentic-2.1.1.dist-info/entry_points.txt +3 -0
- de_agentic-2.1.1.dist-info/licenses/LICENSE +21 -0
- de_agentic-2.1.1.dist-info/top_level.txt +6 -0
- harness/README.md +137 -0
- harness/__init__.py +17 -0
- harness/__main__.py +10 -0
- harness/agent.py +661 -0
- harness/artifact.py +483 -0
- harness/checkpoints.py +385 -0
- harness/cli.py +602 -0
- harness/config.py +176 -0
- harness/config.yaml +107 -0
- harness/graph.py +141 -0
- harness/plan.py +612 -0
- harness/prompts/fix.md +46 -0
- harness/prompts/implement.md +76 -0
- harness/prompts/qa.md +56 -0
- harness/prompts.py +113 -0
- harness/runners.py +286 -0
- harness/util.py +163 -0
- harness/verify.py +291 -0
- operations/__init__.py +1 -0
- operations/profiling.py +219 -0
- operations/query.py +95 -0
- operations/sample_data.py +414 -0
- operations/schema.py +96 -0
- src/__init__.py +9 -0
- src/agents/__init__.py +6 -0
- src/agents/base_agent.py +226 -0
- src/agents/de_agent.py +147 -0
- src/cli.py +607 -0
- src/skills/__init__.py +23 -0
- src/skills/architecture_diagram.py +897 -0
- src/skills/base_skill.py +65 -0
- src/skills/data_profiler.py +126 -0
- src/skills/error_analyzer.py +11 -0
- src/skills/lineage_tracker.py +11 -0
- src/skills/query_optimizer.py +11 -0
- src/skills/schema_analyzer.py +148 -0
- src/skills/sql_generator.py +96 -0
- src/tasks/__init__.py +23 -0
- src/tasks/architecture_task.py +94 -0
- src/tasks/base_task.py +57 -0
- src/tasks/data_ingestion_task.py +84 -0
- src/tasks/data_modeling_task.py +82 -0
- src/tasks/data_quality_task.py +84 -0
- src/tasks/debugging_task.py +75 -0
- src/tasks/reverse_engineering_task.py +84 -0
- src/tasks/warehousing_task.py +93 -0
- src/tools/__init__.py +11 -0
- src/tools/database_tool.py +111 -0
- src/tools/file_tool.py +109 -0
- src/tools/profiling_tool.py +12 -0
- src/tools/schema_tool.py +12 -0
- src/utils/__init__.py +9 -0
- src/utils/config_loader.py +123 -0
- src/utils/image_render.py +493 -0
- src/utils/llm_factory.py +130 -0
- src/utils/logger.py +55 -0
- src/workflows/__init__.py +8 -0
- src/workflows/base_workflow.py +25 -0
- src/workflows/workflow_engine.py +118 -0
ai/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""ai package for the simplified DE Agentic layout."""
|
ai/assistant.py
ADDED
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""The AI assistant: a thin, bounded-context layer over ``core.llm``.
|
|
2
|
+
|
|
3
|
+
Two things make it more than a chat wrapper:
|
|
4
|
+
|
|
5
|
+
* **Bounded memory.** The conversation is a deque capped at ``context_window``
|
|
6
|
+
turns, so a long session cannot silently grow the prompt.
|
|
7
|
+
* **Grounded answers.** The schema digest is injected on every question, and
|
|
8
|
+
any SQL the model returns can be executed and fed back as the next turn -
|
|
9
|
+
so "what is the total revenue?" is answered with a real result, not a guess.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections import deque
|
|
15
|
+
from typing import Any, Deque, Dict, List, Optional
|
|
16
|
+
|
|
17
|
+
from core import llm
|
|
18
|
+
from operations import query as query_ops
|
|
19
|
+
from operations import profiling
|
|
20
|
+
|
|
21
|
+
SYSTEM_PROMPT = (
|
|
22
|
+
"You are a concise data engineering assistant. Use only the tables and "
|
|
23
|
+
"columns listed in the schema - never invent others. Missing data is NULL: "
|
|
24
|
+
"test it with IS NULL / IS NOT NULL, never = '' or = 0. Reply with at most "
|
|
25
|
+
"4 short sentences, or one small SQL block with no extra commentary."
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def extract_sql(text: str) -> str:
|
|
30
|
+
"""Pull a SQL statement out of an answer (fenced block or bare)."""
|
|
31
|
+
import re
|
|
32
|
+
|
|
33
|
+
fenced = re.search(r"```(?:sql)?\s*(.+?)```", text or "", re.S | re.I)
|
|
34
|
+
if fenced:
|
|
35
|
+
return fenced.group(1).strip()
|
|
36
|
+
bare = re.search(r"\b(WITH|SELECT)\b.*", text or "", re.S | re.I)
|
|
37
|
+
return bare.group(0).strip().rstrip(";") if bare else ""
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class Assistant:
|
|
41
|
+
"""A question/answer loop with bounded context and optional SQL execution."""
|
|
42
|
+
|
|
43
|
+
def __init__(self, settings: Optional[llm.Settings] = None,
|
|
44
|
+
db: str = "sample.sqlite", context_window: int = 6,
|
|
45
|
+
execute_sql: bool = True):
|
|
46
|
+
self.settings = settings or llm.resolve()
|
|
47
|
+
self.db = db
|
|
48
|
+
self.context_window = max(1, int(context_window))
|
|
49
|
+
self.execute_sql = execute_sql
|
|
50
|
+
self._history: Deque[Dict[str, str]] = deque(maxlen=self.context_window * 2)
|
|
51
|
+
self.last_execution: Optional[Dict[str, Any]] = None
|
|
52
|
+
|
|
53
|
+
# -- context -------------------------------------------------------------
|
|
54
|
+
@property
|
|
55
|
+
def history(self) -> List[Dict[str, str]]:
|
|
56
|
+
return list(self._history)
|
|
57
|
+
|
|
58
|
+
def clear(self) -> None:
|
|
59
|
+
self._history.clear()
|
|
60
|
+
self.last_execution = None
|
|
61
|
+
|
|
62
|
+
def _turns(self) -> List[Dict[str, str]]:
|
|
63
|
+
# deque has no slicing, so materialise before taking the last N turns.
|
|
64
|
+
return list(self._history)[-self.context_window:]
|
|
65
|
+
|
|
66
|
+
def schema_lines(self) -> str:
|
|
67
|
+
return "\n".join(profiling.profile_database(self.db)["lines"])
|
|
68
|
+
|
|
69
|
+
def build_prompt(self, question: str) -> str:
|
|
70
|
+
return (f"Database: {self.db}\n"
|
|
71
|
+
f"Schema:\n{self.schema_lines()}\n\n"
|
|
72
|
+
f"Question: {question}\n\n"
|
|
73
|
+
"Answer briefly and concretely.")
|
|
74
|
+
|
|
75
|
+
# -- asking --------------------------------------------------------------
|
|
76
|
+
def ask(self, question: str, stream: bool = False) -> str:
|
|
77
|
+
"""Answer a question, optionally executing any SQL it returns."""
|
|
78
|
+
messages = [{"role": "system", "content": SYSTEM_PROMPT},
|
|
79
|
+
{"role": "user", "content": self.build_prompt(question)}]
|
|
80
|
+
messages.extend(self._turns())
|
|
81
|
+
answer = llm.chat(self.settings, messages, stream=stream)
|
|
82
|
+
if stream:
|
|
83
|
+
return answer # caller is rendering deltas; history is not updated
|
|
84
|
+
self._history.append({"role": "user", "content": question})
|
|
85
|
+
self._history.append({"role": "assistant", "content": str(answer)})
|
|
86
|
+
self.last_execution = None
|
|
87
|
+
if self.execute_sql:
|
|
88
|
+
self.last_execution = self._maybe_execute(str(answer))
|
|
89
|
+
if self.last_execution and self.last_execution.get("ok"):
|
|
90
|
+
self._history.append({
|
|
91
|
+
"role": "user",
|
|
92
|
+
"content": "Here is the result of running that SQL:\n"
|
|
93
|
+
f"{self.last_execution['summary']}\n"
|
|
94
|
+
"State the answer in one sentence.",
|
|
95
|
+
})
|
|
96
|
+
follow_up = llm.chat(self.settings, messages + [
|
|
97
|
+
{"role": "user", "content": self._history[-1]["content"]}
|
|
98
|
+
], stream=False)
|
|
99
|
+
self._history.append({"role": "assistant", "content": str(follow_up)})
|
|
100
|
+
return str(follow_up)
|
|
101
|
+
return str(answer)
|
|
102
|
+
|
|
103
|
+
# -- SQL execution -------------------------------------------------------
|
|
104
|
+
def _maybe_execute(self, answer: str) -> Optional[Dict[str, Any]]:
|
|
105
|
+
sql = extract_sql(answer)
|
|
106
|
+
if not sql:
|
|
107
|
+
return None
|
|
108
|
+
try:
|
|
109
|
+
result = query_ops.run_query(self.db, sql, max_rows=20)
|
|
110
|
+
except query_ops.QueryError as exc:
|
|
111
|
+
return {"ok": False, "sql": sql, "error": str(exc)}
|
|
112
|
+
rendered = "; ".join(
|
|
113
|
+
f"{column}={value}" for column, value in zip(result["columns"],
|
|
114
|
+
result["rows"][0])
|
|
115
|
+
) if result["rows"] else "(no rows)"
|
|
116
|
+
return {
|
|
117
|
+
"ok": True,
|
|
118
|
+
"sql": sql,
|
|
119
|
+
"row_count": result["row_count"],
|
|
120
|
+
"columns": result["columns"],
|
|
121
|
+
"rows": result["rows"],
|
|
122
|
+
"summary": f"{result['row_count']} row(s). First row: {rendered}"
|
|
123
|
+
if result["rows"] else "0 rows returned.",
|
|
124
|
+
}
|
ai/tasks/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""ai/tasks package for the simplified DE Agentic layout."""
|