adra 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- adra/__init__.py +36 -0
- adra/clients/synthetic/northwind/README.md +44 -0
- adra/clients/synthetic/northwind/adr/ADR-0001-deterministic-grounding.md +21 -0
- adra/clients/synthetic/northwind/adr/ADR-0002-no-stale-base-merges.md +20 -0
- adra/clients/synthetic/northwind/adr/ADR-0003-bundle-resources-stay-yml.md +19 -0
- adra/clients/synthetic/northwind/adr/ADR-0004-test-discovery.md +20 -0
- adra/clients/synthetic/northwind/adr/ADR-0005-experiments-on-warehouse.md +29 -0
- adra/clients/synthetic/northwind/adr/ADR-0006-docs-from-provenance.md +19 -0
- adra/clients/synthetic/northwind/adr/ADR-0007-decision-support-framing.md +19 -0
- adra/clients/synthetic/northwind/adr/ADR-0008-minimum-functional.md +19 -0
- adra/clients/synthetic/northwind/cases/CASE-2024-031-stale-base-destructive-merge.md +19 -0
- adra/clients/synthetic/northwind/cases/CASE-2024-047-coverage-no-data.md +19 -0
- adra/clients/synthetic/northwind/cases/CASE-2024-052-premature-no-access.md +16 -0
- adra/clients/synthetic/northwind/cases/CASE-2024-058-anomaly-verified-live.md +17 -0
- adra/clients/synthetic/northwind/cases/CASE-2024-061-route-blast-radius.md +18 -0
- adra/clients/synthetic/northwind/ci-standards.md +41 -0
- adra/clients/synthetic/northwind/conventions.md +46 -0
- adra/clients/synthetic/northwind/glossary.md +19 -0
- adra/config.py +164 -0
- adra/connectors/__init__.py +74 -0
- adra/connectors/azure.py +87 -0
- adra/connectors/azure_devops.py +210 -0
- adra/connectors/base.py +94 -0
- adra/connectors/databricks.py +118 -0
- adra/connectors/emulator.py +117 -0
- adra/connectors/github.py +110 -0
- adra/critic.py +119 -0
- adra/judge.py +147 -0
- adra/llm.py +180 -0
- adra/nodes.py +22 -0
- adra/orchestrator.py +83 -0
- adra/prompts/code_review.md +33 -0
- adra/prompts/critic.md +21 -0
- adra/prompts/decide.md +21 -0
- adra/prompts/document.md +23 -0
- adra/prompts/experiment.md +21 -0
- adra/prompts/improve.md +19 -0
- adra/prompts/judge.md +15 -0
- adra/prompts/pr_eval.md +29 -0
- adra/provenance.py +74 -0
- adra/rubric.py +182 -0
- adra/skills/__init__.py +26 -0
- adra/skills/base.py +56 -0
- adra/skills/code_review.py +66 -0
- adra/skills/decide.py +58 -0
- adra/skills/document.py +86 -0
- adra/skills/experiment.py +58 -0
- adra/skills/improve.py +48 -0
- adra/skills/pr_eval.py +85 -0
- adra/state.py +178 -0
- adra/tools/__init__.py +19 -0
- adra/tools/bundle_tools.py +52 -0
- adra/tools/ci_tools.py +76 -0
- adra/tools/discovery_tools.py +45 -0
- adra/tools/git_tools.py +89 -0
- adra/tools/lang_tools.py +65 -0
- adra/tools/sql_tools.py +73 -0
- adra/utils.py +86 -0
- adra-0.4.0.dist-info/METADATA +217 -0
- adra-0.4.0.dist-info/RECORD +66 -0
- adra-0.4.0.dist-info/WHEEL +5 -0
- adra-0.4.0.dist-info/entry_points.txt +2 -0
- adra-0.4.0.dist-info/licenses/LICENSE +201 -0
- adra-0.4.0.dist-info/top_level.txt +2 -0
- cli/__init__.py +1 -0
- cli/__main__.py +147 -0
adra/__init__.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""ADRA — Adversarial Dev Review Agent.
|
|
2
|
+
|
|
3
|
+
A client-agnostic, deterministic-first, adversarial-validation engine that supports the
|
|
4
|
+
software lifecycle. It formalizes six capabilities a team runs informally under
|
|
5
|
+
adversarial human direction:
|
|
6
|
+
|
|
7
|
+
code_review | pr_eval | experiment | improve | document | decide
|
|
8
|
+
|
|
9
|
+
The design spine is *adversarial validation*: every generated artifact is passed through
|
|
10
|
+
a blocking, tool-grounded adversarial critic before it is accepted, and every run emits
|
|
11
|
+
an immutable provenance record (the deep change history). Deterministic tools
|
|
12
|
+
(git / CI / SQL / static analysis) are ground truth; the LLM only adds what tools cannot
|
|
13
|
+
settle. Connectors (GitHub / Azure DevOps / Databricks / Azure) and a self-contained
|
|
14
|
+
offline emulator sit behind one Protocol, so the same engine runs against a real
|
|
15
|
+
platform or a synthetic one.
|
|
16
|
+
|
|
17
|
+
The package runs offline with a deterministic ``mock`` provider (no API key required)
|
|
18
|
+
and switches to a real provider (e.g. Anthropic Claude) when its key is present.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from adra.config import Settings, load_settings
|
|
22
|
+
from adra.orchestrator import Orchestrator
|
|
23
|
+
from adra.state import CriticVerdict, Finding, RunState, Severity, ToolResult
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"Settings",
|
|
27
|
+
"load_settings",
|
|
28
|
+
"Orchestrator",
|
|
29
|
+
"Finding",
|
|
30
|
+
"Severity",
|
|
31
|
+
"ToolResult",
|
|
32
|
+
"CriticVerdict",
|
|
33
|
+
"RunState",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
__version__ = "0.4.0" # PEP 440 package version; display/tag version is v0.04.000 (see VERSION)
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Northwind Data Platform — engineering standards (fictional client suite)
|
|
2
|
+
|
|
3
|
+
> **Fictional.** Northwind Trading and everything below are invented for this
|
|
4
|
+
> reference agent. They model a realistic governance baseline so ADRA has a concrete
|
|
5
|
+
> client to reason against, without referencing any real organization. Point the
|
|
6
|
+
> agent at a different client by replacing this `standards/` folder.
|
|
7
|
+
|
|
8
|
+
## Client profile
|
|
9
|
+
|
|
10
|
+
**Northwind Trading** — a fictional B2B commerce company. The **Northwind Data Platform
|
|
11
|
+
(NDP)** runs the analytics and decision-support products for four operating domains.
|
|
12
|
+
|
|
13
|
+
| Item | Value |
|
|
14
|
+
|---|---|
|
|
15
|
+
| Cloud / compute | Azure Databricks |
|
|
16
|
+
| Catalog / governance | Unity Catalog (UC) |
|
|
17
|
+
| Deploy unit | Databricks Asset Bundles (DAB) |
|
|
18
|
+
| Pipelines | Delta Live Tables (DLT) |
|
|
19
|
+
| Source control / CI | Azure DevOps — org `NorthwindNDP`, project `Data Platform`, shared CI templates `ndp-ci` |
|
|
20
|
+
| Integration branch | `main` |
|
|
21
|
+
| Work branches | `task/<NDP-ticket>/<slug>` |
|
|
22
|
+
| Tickets | `NDP-####` |
|
|
23
|
+
| Catalog naming | `<env>_<domain>_<subdomain>` (env ∈ `dev` / `preprod` / `prod`) |
|
|
24
|
+
|
|
25
|
+
### Domains
|
|
26
|
+
|
|
27
|
+
| Domain | Scope | Example catalog |
|
|
28
|
+
|---|---|---|
|
|
29
|
+
| `catalog` | Product catalog & merchandising | `prod_catalog_items` |
|
|
30
|
+
| `orders` | Order management & fulfilment | `prod_orders_fulfilment` |
|
|
31
|
+
| `payments` | Payments, billing & reconciliation | `prod_payments_ledger` |
|
|
32
|
+
| `analytics` | Demand/risk forecasting & decision support | `prod_analytics_forecast` |
|
|
33
|
+
|
|
34
|
+
## Index
|
|
35
|
+
|
|
36
|
+
- `conventions.md` — language, naming, branching, PR body, labels.
|
|
37
|
+
- `ci-standards.md` — the exact CI command, coverage, test discovery, bundle validate.
|
|
38
|
+
- `glossary.md` — domain and platform terms.
|
|
39
|
+
- `adr/` — Architecture Decision Records (`ADR-0001` … `ADR-0008`).
|
|
40
|
+
- `cases/` — anonymized post-incident notes the rubric is learned from (`CASE-*`).
|
|
41
|
+
|
|
42
|
+
These documents are the source of truth the ADRA agent grounds on. The adversarial
|
|
43
|
+
rubric (`adra/rubric.py`) references them by id, and the skill/critic prompts cite
|
|
44
|
+
them, so "what we check" lives in one place.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# ADR-0001 — Deterministic-first grounding and second-method proof
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
Reviews and experiments that rely on a reviewer's narrative ("this looks fine",
|
|
7
|
+
"that's probably because…") let unverified claims through. Plausible explanations
|
|
8
|
+
without verification have caused rework and missed defects.
|
|
9
|
+
|
|
10
|
+
## Decision
|
|
11
|
+
- Deterministic, high-precision checks (the exact CI command, `bundle validate`,
|
|
12
|
+
git merge-base, language scan, SQL probes) run **first** and are **ground truth**.
|
|
13
|
+
- Any cause or outcome must be backed by a **second, independent method** (a probe,
|
|
14
|
+
a re-run of the exact command, a cross-check) or stated as **"unknown"**.
|
|
15
|
+
- An LLM may only add findings the deterministic tools cannot settle; it may not
|
|
16
|
+
contradict them.
|
|
17
|
+
|
|
18
|
+
## Consequences
|
|
19
|
+
- Verdicts carry evidence, not opinion.
|
|
20
|
+
- The agent reproduces the exact thing under test instead of approximating.
|
|
21
|
+
- See `cases/CASE-2024-058` (an anomaly confirmed only after live verification).
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# ADR-0002 — No stale-base merges (merge-base health)
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
A pull request whose branch is based on an outdated `main` can, on merge, silently
|
|
7
|
+
revert or delete work that landed in the meantime — including notebooks and bundle
|
|
8
|
+
resources — because its diff is computed against an old base.
|
|
9
|
+
|
|
10
|
+
## Decision
|
|
11
|
+
- Every PR is checked for **merge-base health**: compute the merge-base and the
|
|
12
|
+
number of commits the branch is **behind** `main`.
|
|
13
|
+
- A branch behind a fresh `main` must be **rebased or recreated** before review.
|
|
14
|
+
- The diff against the merge-base is scanned for the destructive signature:
|
|
15
|
+
**file deletions** and **resource renames** (see `ADR-0003`). Both are blocking
|
|
16
|
+
until explicitly confirmed.
|
|
17
|
+
|
|
18
|
+
## Consequences
|
|
19
|
+
- Destructive merges are caught before they land.
|
|
20
|
+
- See `cases/CASE-2024-031`.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# ADR-0003 — Bundle resource files stay `.yml`
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
A Databricks Asset Bundle includes resources (jobs, schemas, volumes, pipelines)
|
|
7
|
+
declared in `resources/bundle.resources.<kind>.yml`. Renaming such a file to any
|
|
8
|
+
other extension (e.g. `.yml.t`) silently removes the resource from the bundle, so a
|
|
9
|
+
deploy drops the job/schema/volume without an obvious diff signal.
|
|
10
|
+
|
|
11
|
+
## Decision
|
|
12
|
+
- Resource files **must keep the `.yml` extension**.
|
|
13
|
+
- A rename away from `.yml` (e.g. `→ .yml.t`) is a **blocking** finding.
|
|
14
|
+
- Any change under `resources/` requires `databricks bundle validate -t <env>`
|
|
15
|
+
returning `Validation OK!` before review (see `ci-standards.md`).
|
|
16
|
+
|
|
17
|
+
## Consequences
|
|
18
|
+
- Bundle composition stays explicit and validated.
|
|
19
|
+
- See `cases/CASE-2024-031`.
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# ADR-0004 — Test discovery is `test*.py`; product logic must be importable
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
`ndp-ci` runs `unittest discover -s databricks -p "test*.py"` under coverage. Two
|
|
7
|
+
failure modes recur: test files that do not match the discovery pattern, and product
|
|
8
|
+
logic that lives only inside notebooks (excluded from coverage and not importable).
|
|
9
|
+
|
|
10
|
+
## Decision
|
|
11
|
+
- Test files **must match `test*.py`** (prefix). A `*_test.py` (suffix) file is never
|
|
12
|
+
collected — it is dead code and must be renamed or removed.
|
|
13
|
+
- A test directory must contain `__init__.py` to be recursed into.
|
|
14
|
+
- Product logic that needs coverage must live in **importable, non-notebook** modules
|
|
15
|
+
(plain `.py` outside `tests/`).
|
|
16
|
+
- `Ran 0 tests` / "No data was collected" is a **blocking** CI failure.
|
|
17
|
+
|
|
18
|
+
## Consequences
|
|
19
|
+
- Coverage measures real product code.
|
|
20
|
+
- See `cases/CASE-2024-047`.
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# ADR-0005 — Experiments run on the shared SQL warehouse; 8-point access preflight
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
Ad-hoc validation that spins up interactive clusters is slow and costly, and
|
|
7
|
+
"no access to this catalog" is often concluded prematurely (wrong profile, stopped
|
|
8
|
+
warehouse, missing grant) instead of diagnosed.
|
|
9
|
+
|
|
10
|
+
## Decision
|
|
11
|
+
- Validation experiments use the **shared SQL warehouse** via
|
|
12
|
+
`databricks api post /api/2.0/sql/statements`; do not create interactive clusters.
|
|
13
|
+
- A hypothesis is **falsifiable**, carries a probability and an impact-if-true, and
|
|
14
|
+
is tied to a **standalone probe**; raw rows are persisted (`runs/*.json`).
|
|
15
|
+
- Conclude only what the rows support; record discarded hypotheses with data too.
|
|
16
|
+
|
|
17
|
+
### The 8-point access preflight (exhaust before declaring "no access")
|
|
18
|
+
1. Profile matches the catalog env (`prod` for `prod_*`, `dev` for `dev_*`).
|
|
19
|
+
2. `databricks current-user me --profile <p>` returns the expected user.
|
|
20
|
+
3. `warehouse_id` is valid for the profile and is `RUNNING`.
|
|
21
|
+
4. The catalog exists in that workspace (`SHOW CATALOGS`).
|
|
22
|
+
5. The schema exists (`SHOW SCHEMAS IN <catalog>`).
|
|
23
|
+
6. The table exists (`SHOW TABLES IN <catalog>.<schema>`).
|
|
24
|
+
7. `current_user` is a member of the granting group (`is_member(...)`).
|
|
25
|
+
8. If the warehouse runs as a service principal, the SP has the grant.
|
|
26
|
+
|
|
27
|
+
## Consequences
|
|
28
|
+
- Reproducible, cheap experiments; "no access" becomes a diagnosis, not a guess.
|
|
29
|
+
- See `cases/CASE-2024-052`.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# ADR-0006 — Documentation is generated from provenance
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
Documentation written from memory drifts from what actually shipped. Change history
|
|
7
|
+
was shallow: hard to answer *why* a change was made and *on what evidence*.
|
|
8
|
+
|
|
9
|
+
## Decision
|
|
10
|
+
- Documentation is **generated from the run record** (provenance), not authored from
|
|
11
|
+
memory; pages cite evidence files and **commit-pinned** links (`?version=GC<sha>`).
|
|
12
|
+
- Change history has layers: a **PR change-control page** per merged PR, an
|
|
13
|
+
**experiment page** per experiment, and a **methodology-history** that records only
|
|
14
|
+
**architectural milestones** (contract / persistence / strategy / input changes).
|
|
15
|
+
- A **source-of-truth gap table** is kept when a change makes existing docs stale.
|
|
16
|
+
- Pages are **English, third person, no AI-session leak** (see `conventions.md`).
|
|
17
|
+
|
|
18
|
+
## Consequences
|
|
19
|
+
- Docs stay aligned with reality and are auditable.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# ADR-0007 — Decision-support outputs avoid overclaiming
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
`analytics` products are decision support: demand and risk forecasts, anomaly scores,
|
|
7
|
+
prioritization. Language that claims a model "detects", "predicts", "guarantees" or
|
|
8
|
+
"prevents" an outcome overstates what the model does and creates liability exposure when
|
|
9
|
+
the outcome differs from the claim.
|
|
10
|
+
|
|
11
|
+
## Decision
|
|
12
|
+
- `analytics` / decision-support outputs are framed as **likelihoods, risk scores and
|
|
13
|
+
recommendations** that carry their evidence — never as guaranteed detection or prediction.
|
|
14
|
+
- Documentation, code comments, UI strings and PR text use non-overclaiming framing.
|
|
15
|
+
- The high-consequence decision stays **human-owned**; the model prepares the evidence.
|
|
16
|
+
|
|
17
|
+
## Consequences
|
|
18
|
+
- Claims match what the models actually do (calibration honesty).
|
|
19
|
+
- The `overclaim_language` rubric item flags violations on any decision-support deliverable.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# ADR-0008 — Minimum-functional, smallest reversible change
|
|
2
|
+
|
|
3
|
+
**Status:** Accepted
|
|
4
|
+
|
|
5
|
+
## Context
|
|
6
|
+
Changes that copy whole templates or add "just in case" scaffolding accumulate dead
|
|
7
|
+
code and widen blast radius. Larger diffs are harder to review and to roll back.
|
|
8
|
+
|
|
9
|
+
## Decision
|
|
10
|
+
- Include **only what advances the result**; prune filler even when it was copied
|
|
11
|
+
from a team standard or template.
|
|
12
|
+
- Removing code requires **proof it is dead** (not collected by CI discovery,
|
|
13
|
+
unreferenced) — not an assertion.
|
|
14
|
+
- Prefer the **smallest, reversible** diff; name the rollback. Assess **blast radius**
|
|
15
|
+
(shared CI templates, cross-domain libraries, prod data) and prefer the
|
|
16
|
+
smallest-scope route (see `cases/CASE-2024-061`).
|
|
17
|
+
|
|
18
|
+
## Consequences
|
|
19
|
+
- Smaller, safer, defensible changes; less dead code.
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# CASE-2024-031 — Stale-base PR dropped a notebook and bundle resources
|
|
2
|
+
|
|
3
|
+
**Domain:** `orders` · **Relates to:** ADR-0002, ADR-0003
|
|
4
|
+
|
|
5
|
+
## What happened
|
|
6
|
+
A PR for the orders bundle was opened from a branch based on a `main` that was ~12
|
|
7
|
+
commits stale. Because the diff was computed against the old merge-base, the PR:
|
|
8
|
+
- deleted `nb-priority-coverage.py` (which had landed on `main` in the meantime), and
|
|
9
|
+
- renamed `bundle.resources.schemas.yml` and `…volumes.yml` to `.yml.t`, dropping
|
|
10
|
+
both resources from the bundle.
|
|
11
|
+
|
|
12
|
+
`databricks bundle validate` would have failed, but it was not run.
|
|
13
|
+
|
|
14
|
+
## Root cause
|
|
15
|
+
Stale merge-base + no destructive-diff review + no bundle validation.
|
|
16
|
+
|
|
17
|
+
## Fix / rule
|
|
18
|
+
Rebuilt the change cleanly on a fresh `main`. Codified as ADR-0002 (merge-base
|
|
19
|
+
health + destructive-diff scan) and ADR-0003 (resources stay `.yml`).
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# CASE-2024-047 — Coverage CI failed with "No data was collected"
|
|
2
|
+
|
|
3
|
+
**Domain:** `catalog` · **Relates to:** ADR-0004
|
|
4
|
+
|
|
5
|
+
## What happened
|
|
6
|
+
The `catalog` repo's coverage stage exited 1 with `Ran 0 tests` then
|
|
7
|
+
`CoverageWarning: No data was collected`. The only test-shaped file was
|
|
8
|
+
`py_aux_functions_v3_test.py` (a `*_test.py` **suffix**), which `unittest discover -p
|
|
9
|
+
"test*.py"` never collects. The product logic lived inside a notebook (excluded from
|
|
10
|
+
coverage, not importable).
|
|
11
|
+
|
|
12
|
+
## Diagnosis (second method)
|
|
13
|
+
Ran the **exact** CI command locally and confirmed 0 collected tests. Cross-checked a
|
|
14
|
+
sibling repo whose coverage passed: it had a discoverable `test*.py` exercising an
|
|
15
|
+
importable module — proving the difference, not guessing it.
|
|
16
|
+
|
|
17
|
+
## Fix / rule
|
|
18
|
+
Added an importable module + a discoverable `test*.py`; removed the dead suffix file.
|
|
19
|
+
Codified as ADR-0004 (discovery pattern + importable logic).
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
# CASE-2024-052 — "No access" to a catalog, concluded prematurely
|
|
2
|
+
|
|
3
|
+
**Domain:** `payments` · **Relates to:** ADR-0005
|
|
4
|
+
|
|
5
|
+
## What happened
|
|
6
|
+
An experiment reported "no access to `prod_payments_ledger`" and was closed. In fact
|
|
7
|
+
the query had been issued with the `dev` profile against a `prod_*` catalog; the
|
|
8
|
+
profile is bound to the workspace, so the grant did not apply.
|
|
9
|
+
|
|
10
|
+
## Diagnosis (preflight)
|
|
11
|
+
Walking the 8-point preflight surfaced it at step 1 (profile/env mismatch). Re-running
|
|
12
|
+
with `--profile prod` returned rows immediately.
|
|
13
|
+
|
|
14
|
+
## Fix / rule
|
|
15
|
+
Codified the 8-point access preflight in ADR-0005: never declare "no access" without
|
|
16
|
+
exhausting it.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
# CASE-2024-058 — Column-coverage anomaly traced to a config typo (verified live)
|
|
2
|
+
|
|
3
|
+
**Domain:** `orders` · **Relates to:** ADR-0001
|
|
4
|
+
|
|
5
|
+
## What happened
|
|
6
|
+
An orders-pipeline verification flagged 2 of 41 expected source columns as "missing"
|
|
7
|
+
from the refined stream. The tempting conclusion was "the source is incomplete".
|
|
8
|
+
|
|
9
|
+
## Diagnosis (second method)
|
|
10
|
+
A probe against the live table (~130M rows, last 7 days) showed 39/41 columns present
|
|
11
|
+
and the 2 "missing" ones used a malformed literal (absent for those two
|
|
12
|
+
columns) — a **config typo**, not missing data. The conclusion was only made *after*
|
|
13
|
+
the rows confirmed it.
|
|
14
|
+
|
|
15
|
+
## Fix / rule
|
|
16
|
+
Corrected the column literals. Codified ADR-0001: conclude from a second-method proof,
|
|
17
|
+
record confirmed vs discarded with numbers — never assert from the symptom.
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# CASE-2024-061 — Choosing a route by blast radius and precedent
|
|
2
|
+
|
|
3
|
+
**Domain:** `catalog` · **Relates to:** ADR-0008
|
|
4
|
+
|
|
5
|
+
## What happened
|
|
6
|
+
A catalog refresh trigger needed a higher cadence. Two routes were on the table:
|
|
7
|
+
(a) edit the **shared `ndp-ci` trigger template** (touches every consuming repo), or
|
|
8
|
+
(b) change the cadence **in the catalog repo's own trigger**, matching an existing
|
|
9
|
+
precedent already present for a sibling trigger.
|
|
10
|
+
|
|
11
|
+
## Decision
|
|
12
|
+
Route (b) was chosen: smaller **blast radius**, reversible, and **justified against a
|
|
13
|
+
precedent** in the same repo (a sibling trigger already ran at the target cadence).
|
|
14
|
+
Route (a) was recorded as discarded with its trade-off (broad blast radius).
|
|
15
|
+
|
|
16
|
+
## Fix / rule
|
|
17
|
+
Codified ADR-0008: prefer the smallest reversible route; justify against convention
|
|
18
|
+
or a measured gap; assess blast radius explicitly.
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# NDP CI standards
|
|
2
|
+
|
|
3
|
+
The `ndp-ci` shared templates are the source of truth for "green". The agent must
|
|
4
|
+
reproduce the **exact** commands, never an approximation (see `adr/ADR-0001`).
|
|
5
|
+
|
|
6
|
+
## Unit tests + coverage (the exact command)
|
|
7
|
+
|
|
8
|
+
```bash
|
|
9
|
+
python -m coverage run -m unittest discover -s databricks -p "test*.py"
|
|
10
|
+
python -m coverage report --fail-under=80
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
- **Discovery pattern is `test*.py`** (prefix). A file named `*_test.py` (suffix) is
|
|
14
|
+
**never collected** and is dead code (see `adr/ADR-0004`, `cases/CASE-2024-047`).
|
|
15
|
+
- A package without `__init__.py` is **not** recursed into by `unittest discover`.
|
|
16
|
+
- Product logic must be **importable, non-notebook** code (a plain `.py` outside
|
|
17
|
+
`tests/`); notebooks are excluded from coverage and cannot be imported (they call
|
|
18
|
+
`dbutils` / `spark` at module load).
|
|
19
|
+
- `Ran 0 tests` → coverage reports **"No data was collected"** → non-zero exit. This
|
|
20
|
+
is a CI failure, not a warning.
|
|
21
|
+
|
|
22
|
+
## Bundle validation
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
databricks bundle validate -t <env>
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Must print `Validation OK!`. Run it before any PR that touches `resources/` or the
|
|
29
|
+
bundle (see `adr/ADR-0003`).
|
|
30
|
+
|
|
31
|
+
## Experiments / ad-hoc data access
|
|
32
|
+
|
|
33
|
+
Experiments run against the **shared SQL warehouse**, never a fresh interactive
|
|
34
|
+
cluster (see `adr/ADR-0005`):
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
databricks api post /api/2.0/sql/statements --profile <prod|dev> --json '{...}'
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Before concluding "no access" to a catalog, exhaust the 8-point preflight in
|
|
41
|
+
`adr/ADR-0005`.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# NDP conventions
|
|
2
|
+
|
|
3
|
+
## Language & authorship
|
|
4
|
+
- Everything written to disk is **English**: code, docstrings, comments, commit
|
|
5
|
+
messages, PR titles/descriptions, test names, file and folder names.
|
|
6
|
+
- **Third person.** No first-person singular in code, docs, commits or PRs.
|
|
7
|
+
- **No AI-session leak.** Nothing written to a repo may reveal AI authorship
|
|
8
|
+
(no `Claude`, `Anthropic`, `Co-Authored-By`, "generated with AI", etc.). See
|
|
9
|
+
`adr/ADR-0006`.
|
|
10
|
+
|
|
11
|
+
## Naming
|
|
12
|
+
- Catalogs: `<env>_<domain>_<subdomain>` — e.g. `prod_orders_fulfilment`. `env` ∈
|
|
13
|
+
`dev` / `preprod` / `prod`.
|
|
14
|
+
- Schemas follow the medallion split: `landing` / `trusted` / `refined`.
|
|
15
|
+
- DAB resource files: `bundle.resources.<kind>.yml` and they **stay `.yml`** (see
|
|
16
|
+
`adr/ADR-0003`).
|
|
17
|
+
- Notebooks: `nb-<purpose>.py` with the `# Databricks notebook source` header.
|
|
18
|
+
|
|
19
|
+
## Branching & tickets
|
|
20
|
+
- Integration branch: `main`. Never commit product changes directly to `main`.
|
|
21
|
+
- Work branches: `task/<NDP-####>/<short-slug>`, always rebased on a fresh `main`
|
|
22
|
+
(see `adr/ADR-0002`).
|
|
23
|
+
- Tickets are `NDP-####`; every PR links its ticket.
|
|
24
|
+
|
|
25
|
+
## Pull requests
|
|
26
|
+
PR description uses these sections, in order:
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
## Objective
|
|
30
|
+
## Changes
|
|
31
|
+
## What is NOT touched
|
|
32
|
+
## Validation (exact CI command output; `bundle validate` OK)
|
|
33
|
+
## Risks / mitigations
|
|
34
|
+
## Test plan
|
|
35
|
+
## Work Item (NDP-####)
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
- Reference another PR by its **full URL**, never a bare `#NNNN`.
|
|
39
|
+
- Use **commit-pinned** file links (`?version=GC<full-sha>`) so links survive the merge.
|
|
40
|
+
- Apply the owning team's **labels** before completing the PR.
|
|
41
|
+
- A PR with any deterministic blocker is **changes-requested** regardless of opinion.
|
|
42
|
+
|
|
43
|
+
## Decision-support framing
|
|
44
|
+
`analytics` products are **decision support**: outputs are *forecasts, risk scores and
|
|
45
|
+
recommendations* with their evidence — never claims to "guarantee", "prevent",
|
|
46
|
+
"detect" or "predict" an outcome (see `adr/ADR-0007`).
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# NDP glossary
|
|
2
|
+
|
|
3
|
+
| Term | Meaning |
|
|
4
|
+
|---|---|
|
|
5
|
+
| NDP | Northwind Data Platform |
|
|
6
|
+
| Domain | A product area: `catalog`, `orders`, `payments`, `analytics` |
|
|
7
|
+
| UC | Unity Catalog (governance over catalogs / schemas / tables / volumes) |
|
|
8
|
+
| DAB | Databricks Asset Bundle (the deploy unit) |
|
|
9
|
+
| DLT | Delta Live Tables (declarative pipelines) |
|
|
10
|
+
| Medallion | `landing` → `trusted` → `refined` schema split |
|
|
11
|
+
| Warehouse | Shared serverless SQL warehouse for ad-hoc queries / experiments |
|
|
12
|
+
| SKU | Stock-keeping unit — a `catalog` product identifier |
|
|
13
|
+
| Threshold | A configurable decision target in `analytics` (e.g. a fraud-score cutoff) |
|
|
14
|
+
| Conversion | Order conversion rate (%) tracked in `analytics` |
|
|
15
|
+
| Forecast/risk output | `analytics` decision-support output — a likelihood/recommendation, not a guarantee |
|
|
16
|
+
| Data contract | The documented schema + semantics of a published UC table |
|
|
17
|
+
| Provenance run record | The immutable JSON ADRA writes per run (evidence + decisions) |
|
|
18
|
+
| Preflight | The 8-point access checklist before declaring "no access" (`adr/ADR-0005`) |
|
|
19
|
+
| Blast radius | The reach of a change (shared templates, cross-domain libs, prod data) |
|
adra/config.py
ADDED
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Runtime settings for ADRA.
|
|
2
|
+
|
|
3
|
+
Settings come from environment variables (a ``.env`` is loaded if present) with
|
|
4
|
+
safe defaults so the package runs offline out of the box. Nothing here reads or
|
|
5
|
+
stores secrets beyond the LLM key, which is only ever taken from the environment.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
from adra.utils import client_dir as _client_dir
|
|
15
|
+
|
|
16
|
+
# Default Anthropic model — claude-haiku-4-5 per ADR-0053 (cost-appropriate default for
|
|
17
|
+
# private/quality apps; switch to claude-sonnet-4-6 when reasoning depth matters). Overridable via ADRA_MODEL.
|
|
18
|
+
DEFAULT_ANTHROPIC_MODEL = "claude-haiku-4-5"
|
|
19
|
+
|
|
20
|
+
# Built-in providers. Anthropic uses its native SDK; every entry below speaks the
|
|
21
|
+
# OpenAI-compatible Chat Completions API, so OpenAI, Groq, xAI, Mistral, DeepSeek,
|
|
22
|
+
# OpenRouter, Together AND local servers (Ollama / LM Studio / vLLM) all work — bring
|
|
23
|
+
# whatever you have, or run a local model for free. Any other OpenAI-compatible service
|
|
24
|
+
# works via ADRA_BASE_URL + ADRA_API_KEY.
|
|
25
|
+
PROVIDERS: dict[str, dict[str, str]] = {
|
|
26
|
+
"openai": {"base_url": "https://api.openai.com/v1", "key_env": "OPENAI_API_KEY", "default_model": "gpt-4o"},
|
|
27
|
+
"groq": {"base_url": "https://api.groq.com/openai/v1", "key_env": "GROQ_API_KEY", "default_model": "llama-3.3-70b-versatile"},
|
|
28
|
+
"xai": {"base_url": "https://api.x.ai/v1", "key_env": "XAI_API_KEY", "default_model": "grok-4"},
|
|
29
|
+
"mistral": {"base_url": "https://api.mistral.ai/v1", "key_env": "MISTRAL_API_KEY", "default_model": "mistral-large-latest"},
|
|
30
|
+
"deepseek": {"base_url": "https://api.deepseek.com/v1", "key_env": "DEEPSEEK_API_KEY", "default_model": "deepseek-chat"},
|
|
31
|
+
"openrouter": {"base_url": "https://openrouter.ai/api/v1", "key_env": "OPENROUTER_API_KEY", "default_model": "openai/gpt-4o"},
|
|
32
|
+
"together": {"base_url": "https://api.together.xyz/v1", "key_env": "TOGETHER_API_KEY", "default_model": "meta-llama/Llama-3.3-70B-Instruct-Turbo"},
|
|
33
|
+
"ollama": {"base_url": "http://localhost:11434/v1", "key_env": "", "default_model": "llama3.1"},
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
# Auto-detect order when ADRA_PROVIDER is unset: a present key wins; else offline mock.
|
|
37
|
+
_AUTODETECT_ORDER = ("anthropic", "openai", "groq", "xai", "mistral", "deepseek", "openrouter", "together")
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def default_model(provider: str) -> str:
|
|
41
|
+
"""The default model id for a provider (overridable with ADRA_MODEL)."""
|
|
42
|
+
if provider in ("anthropic", "mock"):
|
|
43
|
+
return DEFAULT_ANTHROPIC_MODEL
|
|
44
|
+
info = PROVIDERS.get(provider)
|
|
45
|
+
return info["default_model"] if info else DEFAULT_ANTHROPIC_MODEL
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _autodetect_provider() -> str:
|
|
49
|
+
"""Pick a provider from whichever API key is present; offline mock if none."""
|
|
50
|
+
for name in _AUTODETECT_ORDER:
|
|
51
|
+
key_env = "ANTHROPIC_API_KEY" if name == "anthropic" else PROVIDERS.get(name, {}).get("key_env", "")
|
|
52
|
+
if key_env and os.environ.get(key_env):
|
|
53
|
+
return name
|
|
54
|
+
return "mock"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _load_dotenv(path: Path) -> None:
|
|
58
|
+
"""Minimal .env loader (no external dependency).
|
|
59
|
+
|
|
60
|
+
Only sets keys that are not already present in the environment, so explicit
|
|
61
|
+
environment variables always win.
|
|
62
|
+
"""
|
|
63
|
+
if not path.exists():
|
|
64
|
+
return
|
|
65
|
+
for raw in path.read_text(encoding="utf-8").splitlines():
|
|
66
|
+
line = raw.strip()
|
|
67
|
+
if not line or line.startswith("#") or "=" not in line:
|
|
68
|
+
continue
|
|
69
|
+
key, value = line.split("=", 1)
|
|
70
|
+
key, value = key.strip(), value.strip().strip('"').strip("'")
|
|
71
|
+
os.environ.setdefault(key, value)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass
|
|
75
|
+
class Settings:
|
|
76
|
+
"""Resolved configuration for a run."""
|
|
77
|
+
|
|
78
|
+
# LLM (the deterministic loop runs offline with NO key; a provider adds the semantic
|
|
79
|
+
# layer). provider: mock | anthropic | openai | groq | xai | mistral | deepseek |
|
|
80
|
+
# openrouter | together | ollama (local). See adra.config.PROVIDERS.
|
|
81
|
+
provider: str = "mock"
|
|
82
|
+
model: str = DEFAULT_ANTHROPIC_MODEL
|
|
83
|
+
temperature: float = 0.0
|
|
84
|
+
max_tokens: int = 4096
|
|
85
|
+
# Per-role model overrides so one run can orchestrate across providers (e.g. a strong
|
|
86
|
+
# model for the critic/judge, a cheaper/faster one for generation). Value per role is
|
|
87
|
+
# "provider:model" or just "model" (provider inherits the default). Empty => default.
|
|
88
|
+
role_models: dict[str, str] = field(default_factory=dict)
|
|
89
|
+
|
|
90
|
+
# Adversarial loop
|
|
91
|
+
max_rounds: int = 3 # generate -> critic -> revise iterations before escalation
|
|
92
|
+
judge_swap_average: bool = True # evaluate pairwise comparisons in both orders
|
|
93
|
+
|
|
94
|
+
# Provenance
|
|
95
|
+
runs_dir: Path = field(default_factory=lambda: Path("runs"))
|
|
96
|
+
|
|
97
|
+
# Repo context (for deterministic tools); optional.
|
|
98
|
+
repo_path: Path | None = None
|
|
99
|
+
|
|
100
|
+
# Active client governance suite (conventions / ADRs / CI standards / cases) the
|
|
101
|
+
# engine grounds on. Defaults to the bundled synthetic client; ADRA_CLIENT_DIR overrides.
|
|
102
|
+
client_dir: Path = field(default_factory=_client_dir)
|
|
103
|
+
|
|
104
|
+
# Safety: deterministic tools that mutate or call external services stay off
|
|
105
|
+
# unless explicitly enabled (default is dry-run / read-only).
|
|
106
|
+
allow_external_calls: bool = False
|
|
107
|
+
|
|
108
|
+
def role(self, role: str) -> tuple[str, str]:
|
|
109
|
+
"""Resolve (provider, model) for a flow role: 'plan'|'generate'|'critic'|'judge'.
|
|
110
|
+
|
|
111
|
+
Falls back to the run's default provider/model; overridden per role via
|
|
112
|
+
``role_models`` (``ADRA_MODEL_<ROLE>``), value ``"provider:model"`` or ``"model"``.
|
|
113
|
+
"""
|
|
114
|
+
spec = self.role_models.get(role, "")
|
|
115
|
+
if not spec:
|
|
116
|
+
return self.provider, self.model
|
|
117
|
+
if ":" in spec:
|
|
118
|
+
prov, _, mdl = spec.partition(":")
|
|
119
|
+
return (prov or self.provider), (mdl or self.model)
|
|
120
|
+
return self.provider, spec
|
|
121
|
+
|
|
122
|
+
def model_id(self, role: str) -> str:
|
|
123
|
+
"""``provider:model`` string for the given role (for logging / provenance)."""
|
|
124
|
+
prov, mdl = self.role(role)
|
|
125
|
+
return f"{prov}:{mdl}"
|
|
126
|
+
|
|
127
|
+
@property
|
|
128
|
+
def offline(self) -> bool:
|
|
129
|
+
return self.provider == "mock"
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def load_settings(**overrides) -> Settings:
|
|
133
|
+
"""Build :class:`Settings` from env + overrides.
|
|
134
|
+
|
|
135
|
+
Provider auto-detects: if ``ADRA_PROVIDER`` is unset, the first provider with a
|
|
136
|
+
present API key wins (Anthropic → OpenAI → Groq → xAI → ...); otherwise the offline
|
|
137
|
+
``mock`` provider. Model defaults per provider unless ``ADRA_MODEL`` is set; per-role
|
|
138
|
+
overrides come from ``ADRA_MODEL_{PLAN,GENERATE,CRITIC,JUDGE}``.
|
|
139
|
+
"""
|
|
140
|
+
_load_dotenv(Path(".env"))
|
|
141
|
+
|
|
142
|
+
provider = os.environ.get("ADRA_PROVIDER") or _autodetect_provider()
|
|
143
|
+
|
|
144
|
+
role_models: dict[str, str] = {}
|
|
145
|
+
for role in ("plan", "generate", "critic", "judge"):
|
|
146
|
+
spec = os.environ.get(f"ADRA_MODEL_{role.upper()}")
|
|
147
|
+
if spec:
|
|
148
|
+
role_models[role] = spec
|
|
149
|
+
|
|
150
|
+
settings = Settings(
|
|
151
|
+
provider=provider,
|
|
152
|
+
model=os.environ.get("ADRA_MODEL") or default_model(provider),
|
|
153
|
+
temperature=float(os.environ.get("ADRA_TEMPERATURE", "0.0")),
|
|
154
|
+
max_tokens=int(os.environ.get("ADRA_MAX_TOKENS", "4096")),
|
|
155
|
+
max_rounds=int(os.environ.get("ADRA_MAX_ROUNDS", "3")),
|
|
156
|
+
allow_external_calls=os.environ.get("ADRA_ALLOW_EXTERNAL", "0") == "1",
|
|
157
|
+
role_models=role_models,
|
|
158
|
+
)
|
|
159
|
+
repo = os.environ.get("ADRA_REPO_PATH")
|
|
160
|
+
if repo:
|
|
161
|
+
settings.repo_path = Path(repo)
|
|
162
|
+
for key, value in overrides.items():
|
|
163
|
+
setattr(settings, key, value)
|
|
164
|
+
return settings
|