aiva-agent 0.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
aiva_agent/__init__.py ADDED
@@ -0,0 +1,8 @@
1
+ import os
2
+
3
+ os.environ.setdefault("OPENAI_AGENTS_DISABLE_TRACING", "1")
4
+
5
+ from .agent import run_once
6
+ from .notebook import aiva_agent, reset_session
7
+
8
+ __all__ = ["run_once", "aiva_agent", "reset_session"]
aiva_agent/agent.py ADDED
@@ -0,0 +1,255 @@
1
+ """Agent runtime: builds an OpenAI Agents SDK Agent from --disable, runs it once.
2
+
3
+ Multi-provider works through any OpenAI-compatible endpoint:
4
+ - User passes --model, --base-url, --api-key (or sets LLM_BASE_URL / LLM_API_KEY)
5
+ - We construct AsyncOpenAI(base_url=, api_key=) and OpenAIChatCompletionsModel(model=, ...)
6
+ - Pass that as model= to Agent(...)
7
+
8
+ No PROVIDERS dict, no provider-prefix parsing, no LiteLLM. The user already knows
9
+ their provider's URL and has their key; we don't need to enumerate.
10
+ """
11
+
12
+ import asyncio
13
+ import os
14
+ from pathlib import Path
15
+ from typing import Iterable
16
+
17
+ from agents import Agent, OpenAIChatCompletionsModel, Runner, SQLiteSession, function_tool
18
+ from openai import AsyncOpenAI
19
+
20
+ from .classification_prompts import ACMG_AMP_SYSTEM_PROMPT
21
+ from .prompts import get_main_system_prompt
22
+ from .tools import (
23
+ build_annotate_tool,
24
+ build_literature_tool,
25
+ build_phen2gene_tool,
26
+ build_trials_tool,
27
+ build_vcf_tools,
28
+ build_web_tool,
29
+ )
30
+
31
+ DEFAULT_MODEL = "gpt-5.5"
32
+ DEFAULT_MAX_TURNS = 25
33
+
34
+ ALL_TOOLS = ("vcf", "annotate", "literature", "trials", "phen2gene", "web", "classify")
35
+
36
+
37
+ def _resolve_max_turns() -> int:
38
+ raw = os.environ.get("AIVA_MAX_TURNS", "").strip()
39
+ if not raw:
40
+ return DEFAULT_MAX_TURNS
41
+ try:
42
+ n = int(raw)
43
+ except ValueError:
44
+ return DEFAULT_MAX_TURNS
45
+ return n if n > 0 else DEFAULT_MAX_TURNS
46
+
47
+
48
+ def _resolve_disabled(disabled_tools: Iterable[str] | None) -> list[str]:
49
+ """Return the active tool list = ALL_TOOLS minus the disabled set.
50
+
51
+ Default (None / empty) → every tool is on. Unknown names are ignored
52
+ (the CLI already validates), preserving the lenient behavior of the
53
+ previous `_resolve_enabled`.
54
+ """
55
+ if not disabled_tools:
56
+ return list(ALL_TOOLS)
57
+ drop = {name.strip() for name in disabled_tools if name and name.strip()}
58
+ return [t for t in ALL_TOOLS if t not in drop]
59
+
60
+
61
+ def build_model(
62
+ model: str,
63
+ base_url: str | None = None,
64
+ api_key: str | None = None,
65
+ ) -> OpenAIChatCompletionsModel:
66
+ """Construct an OpenAIChatCompletionsModel pointing at any OpenAI-compatible endpoint.
67
+
68
+ Resolution order for base_url and api_key:
69
+ 1. The argument passed in (CLI flag).
70
+ 2. LLM_BASE_URL / LLM_API_KEY env vars.
71
+ """
72
+ resolved_base_url = base_url or os.environ.get("LLM_BASE_URL")
73
+ resolved_api_key = api_key or os.environ.get("LLM_API_KEY")
74
+ client = AsyncOpenAI(base_url=resolved_base_url, api_key=resolved_api_key)
75
+ return OpenAIChatCompletionsModel(model=model, openai_client=client)
76
+
77
+
78
+ def build_classifier_tool(
79
+ classifier_tools: list,
80
+ model: str,
81
+ base_url: str | None,
82
+ api_key: str | None,
83
+ ):
84
+ """Wrap the variant-classifier sub-agent as a @function_tool callable from the parent.
85
+
86
+ Mirrors genomiq's SubAgentRunner pattern: build a fresh Agent with the classifier's
87
+ own tool palette, run it via Runner.run, return the JSON output.
88
+ """
89
+ sub_model = build_model(model, base_url, api_key)
90
+ sub_agent = Agent(
91
+ name="variant-classifier",
92
+ instructions=ACMG_AMP_SYSTEM_PROMPT,
93
+ model=sub_model,
94
+ tools=classifier_tools,
95
+ )
96
+
97
+ # `variant_data` is a heterogeneous dict (rsid OR hgvs OR chrom/pos/ref/alt) — strict
98
+ # JSON Schema doesn't model that cleanly without a TypedDict per shape, so we relax
99
+ # strict mode for this single tool. The shape contract is documented in the docstring
100
+ # and enforced by the sub-agent's system prompt.
101
+ @function_tool(strict_mode=False)
102
+ async def classify_variant(
103
+ variant_data: dict,
104
+ classification_type: str,
105
+ assembly: str,
106
+ phenotype_terms: str = "",
107
+ description: str = "",
108
+ additional_context: str = "",
109
+ ) -> str:
110
+ """Classify a genomic variant using ACMG/AMP 2015 guidelines (germline) or
111
+ AMP/ASCO/CAP 2017 guidelines (somatic). Runs as a sub-agent that gathers
112
+ annotations, literature, and trial evidence on its own, returning a JSON
113
+ classification with criteria_met, evidence_summary, and sources.
114
+
115
+ Args:
116
+ variant_data: One of {"rsid": "rs..."} or {"hgvs": "..."} or
117
+ {"chrom": "...", "pos": ..., "ref": "...", "alt": "..."}.
118
+ classification_type: 'acmg' for germline or 'amp' for somatic/cancer.
119
+ assembly: 'GRCh37' or 'GRCh38'.
120
+ phenotype_terms: e.g. 'melanoma', 'hereditary breast cancer'.
121
+ description: Sample/patient context (optional).
122
+ additional_context: Verified clinical context (de novo status, family history,
123
+ zygosity, segregation). Apply criteria strictly to what's stated here.
124
+ """
125
+ formatted = (
126
+ "Classify this variant.\n\n"
127
+ f"- variant_data: {variant_data}\n"
128
+ f"- classification_type: {classification_type}\n"
129
+ f"- assembly: {assembly}\n"
130
+ f"- phenotype_terms: {phenotype_terms}\n"
131
+ f"- description: {description}\n"
132
+ f"- additional_context: {additional_context}\n"
133
+ )
134
+ result = await Runner.run(sub_agent, formatted, max_turns=_resolve_max_turns())
135
+ return result.final_output
136
+
137
+ return classify_variant
138
+
139
+
140
+ def build_tools_and_cleanup(
141
+ enabled: list[str],
142
+ vcf_specs: list[tuple[str, Path]] | None,
143
+ model: str,
144
+ base_url: str | None,
145
+ api_key: str | None,
146
+ ) -> tuple[list, list]:
147
+ """Construct the tool list for an agent run plus any cleanup callables.
148
+
149
+ Raises ValueError if 'vcf' is in `enabled` but `vcf_specs` is empty. The CLI
150
+ auto-disables vcf with a warning before reaching this function; the guard
151
+ is the contract for direct library callers.
152
+
153
+ `vcf_specs` is a list of `(alias, path)` pairs; each becomes a named DuckDB
154
+ view. Single-file invocations should pass `[("vcf", path)]`.
155
+
156
+ classify is special: it needs annotate / literature / trials available to its
157
+ sub-agent. We build those tools once and share the instances between the parent's
158
+ palette and the classifier's palette.
159
+ """
160
+ tools: list = []
161
+ cleanups: list = []
162
+
163
+ annotate_tool = literature_tool = trials_tool = None
164
+
165
+ if "vcf" in enabled:
166
+ if not vcf_specs:
167
+ raise ValueError("--vcf is required when 'vcf' is enabled")
168
+ vcf_tools, vcf_cleanup = build_vcf_tools(vcf_specs)
169
+ tools.extend(vcf_tools)
170
+ cleanups.append(vcf_cleanup)
171
+
172
+ if "annotate" in enabled or "classify" in enabled:
173
+ annotate_tool = build_annotate_tool()
174
+ if "annotate" in enabled:
175
+ tools.append(annotate_tool)
176
+
177
+ if "literature" in enabled or "classify" in enabled:
178
+ literature_tool, literature_cleanup = build_literature_tool()
179
+ cleanups.append(literature_cleanup)
180
+ if "literature" in enabled:
181
+ tools.append(literature_tool)
182
+
183
+ if "trials" in enabled or "classify" in enabled:
184
+ trials_tool, trials_cleanup = build_trials_tool()
185
+ cleanups.append(trials_cleanup)
186
+ if "trials" in enabled:
187
+ tools.append(trials_tool)
188
+
189
+ if "phen2gene" in enabled:
190
+ phen2gene_tool, phen2gene_cleanup = build_phen2gene_tool()
191
+ cleanups.append(phen2gene_cleanup)
192
+ tools.append(phen2gene_tool)
193
+
194
+ if "web" in enabled:
195
+ tools.append(build_web_tool())
196
+
197
+ if "classify" in enabled:
198
+ classifier_tools = [t for t in (annotate_tool, literature_tool, trials_tool) if t is not None]
199
+ tools.append(
200
+ build_classifier_tool(
201
+ classifier_tools=classifier_tools,
202
+ model=model,
203
+ base_url=base_url,
204
+ api_key=api_key,
205
+ )
206
+ )
207
+
208
+ return tools, cleanups
209
+
210
+
211
+ async def run_once(
212
+ prompt: str,
213
+ *,
214
+ vcf_specs: list[tuple[str, Path]] | None = None,
215
+ disabled_tools: Iterable[str] | None = None,
216
+ model: str = DEFAULT_MODEL,
217
+ base_url: str | None = None,
218
+ api_key: str | None = None,
219
+ session_id: str | None = None,
220
+ ) -> str:
221
+ enabled = _resolve_disabled(disabled_tools)
222
+
223
+ tools, cleanups = build_tools_and_cleanup(
224
+ enabled=enabled,
225
+ vcf_specs=vcf_specs,
226
+ model=model,
227
+ base_url=base_url,
228
+ api_key=api_key,
229
+ )
230
+
231
+ session = None
232
+ if session_id:
233
+ db_path = Path.home() / ".aiva" / "sessions.db"
234
+ db_path.parent.mkdir(parents=True, exist_ok=True)
235
+ session = SQLiteSession(session_id, str(db_path))
236
+
237
+ try:
238
+ agent = Agent(
239
+ name="aiva",
240
+ instructions=get_main_system_prompt(enabled, vcf_specs=vcf_specs),
241
+ model=build_model(model, base_url, api_key),
242
+ tools=tools,
243
+ )
244
+ result = await Runner.run(
245
+ agent, prompt, max_turns=_resolve_max_turns(), session=session
246
+ )
247
+ return result.final_output
248
+ finally:
249
+ for cleanup in cleanups:
250
+ try:
251
+ result = cleanup()
252
+ if asyncio.iscoroutine(result):
253
+ await result
254
+ except Exception:
255
+ pass
@@ -0,0 +1,143 @@
1
+ """Static system prompt for the variant-classifier sub-agent.
2
+
3
+ The genomiq classification flow built one prompt per call that interleaved the
4
+ variant data, criteria, and instructions. With the OpenAI Agents SDK's
5
+ function_tool sub-agent pattern (a `@function_tool` that internally instantiates
6
+ an `Agent` and calls `Runner.run`) we use a single, static system prompt — the
7
+ variant data is passed in via the first user-turn message from the parent agent.
8
+ """
9
+
10
+ ACMG_AMP_SYSTEM_PROMPT = """You are a clinical genetics expert specializing in variant interpretation
11
+ using ACMG/AMP 2015 guidelines (germline) and AMP/ASCO/CAP 2017 guidelines (somatic/cancer).
12
+
13
+ # Input contract
14
+
15
+ Your invoking prompt will provide these fields (the parent agent is instructed to
16
+ include them). Parse them out before doing any work:
17
+
18
+ - **variant_data** (REQUIRED): one of
19
+ • `{"rsid": "rs113488022"}`
20
+ • `{"hgvs": "chr7:g.140453136A>T"}` or `{"hgvs": "NM_004333.6:c.1799T>A"}`
21
+ • `{"chrom": "7", "pos": 140453136, "ref": "A", "alt": "T"}`
22
+ - **classification_type** (REQUIRED): `"acmg"` (germline) or `"amp"` (somatic/cancer)
23
+ - **assembly** (REQUIRED): `"GRCh37"` or `"GRCh38"`
24
+ - **phenotype_terms**: e.g. "melanoma", "hereditary breast cancer". Use it to scope
25
+ literature/trial searches and to evaluate phenotype-specific criteria (PP4, etc.).
26
+ - **description**: free-form sample/patient context. Use it for clinical framing.
27
+ - **additional_context**: verified clinical context (de novo status, family history,
28
+ zygosity, segregation). Apply ACMG criteria based STRICTLY on what is explicitly
29
+ stated here — do not extrapolate. PS2 only with confirmed de novo; PM6 with
30
+ assumed de novo; neither if no de novo info.
31
+
32
+ If `variant_data`, `classification_type`, or `assembly` is missing or ambiguous,
33
+ return `{"error": "missing required field: <field>", "expected": "..."}` instead of
34
+ guessing.
35
+
36
+ # Available tools
37
+
38
+ - `annotate_variant` (myvariant / vep / civic) — gather variant annotations.
39
+ Pass `include_commercial=true` if you need CADD or dbNSFP scores for PP3/BP4.
40
+ - `search_biomedical_literature` (PubTator3) — supporting literature.
41
+ - `search_clinical_trials` (ClinicalTrials.gov) — therapeutic context for somatic.
42
+
43
+ You have a strict ~120-second budget. Prioritize: (1) ONE annotate_variant call to
44
+ gather all annotations, (2) a few targeted literature searches, (3) clinical trials
45
+ only if relevant for somatic classification. Do NOT call annotate_variant multiple
46
+ times — one call returns all data; only retry on error or rate-limit.
47
+
48
+ # ACMG/AMP 2015 (Germline) — `acmg`
49
+
50
+ **Key Interpretation Rules:**
51
+ - DO NOT defer to ClinVar's existing classification — your classification must be based
52
+ on YOUR evaluation of all criteria, not ClinVar's conclusion.
53
+ - Missing gnomAD = RARE variant (supports PM2 criterion).
54
+ - In-silico priority: CADD and DITTO are primary. CADD≥15 or DITTO≥0.5 strongly
55
+ supports deleterious effect. SIFT, PolyPhen, MutationTaster, REVEL are secondary.
56
+ - Avoid plain "VUS" — use "VUS (leaning pathogenic)" or "VUS (leaning benign)".
57
+
58
+ **Critical Validation Rules:**
59
+ - PP3/BP4 (Computational predictions): ONLY apply with actual scores (CADD, DITTO,
60
+ REVEL, SIFT, PolyPhen). Do NOT apply based on domain location alone.
61
+ - PM1 (Functional domain): Requires BOTH functional domain evidence AND absence of
62
+ benign variation at that location.
63
+ - PP2 (Missense constraint): Requires quantitative metrics (gnomAD mis_z, RVIS).
64
+ - PM5: Requires that other same-position missense variants are TRULY pathogenic in
65
+ ClinVar/literature, not just reported.
66
+ - PS2/PM6 (De novo): PS2 if confirmed; PM6 if assumed; otherwise neither.
67
+ - PP4 (Phenotype specificity): Requires pathognomonic features, not general symptoms.
68
+ - Each criterion is fully met or not_applicable — no partial application.
69
+ - PP5/BP6 are removed per ClinGen 2018; do not apply.
70
+
71
+ **Pathogenic Criteria (apply only when met):**
72
+ PVS1 (LOF in gene where LOF causes disease — strength per Abou Tayoun 2018),
73
+ PS1 (same AA change as established pathogenic), PS2 (confirmed de novo),
74
+ PS3 (well-established functional studies), PS4 (case prevalence OR>5),
75
+ PM1 (hot spot + no benign variation), PM2_Supporting (rare in gnomAD),
76
+ PM3 (in trans with pathogenic, recessive), PM4 (length change in-frame),
77
+ PM5 (novel missense at established pathogenic position), PM6 (assumed de novo),
78
+ PP1 (co-segregation), PP2 (constrained gene with quantitative metrics),
79
+ PP3 (computational evidence supporting deleterious),
80
+ PP4 (pathognomonic phenotype).
81
+
82
+ **Benign Criteria (apply only when met):**
83
+ BA1 (>5% gnomAD), BS1 (frequency > expected), BS2 (healthy adult observed),
84
+ BS3 (functional studies show no damage), BS4 (lack of segregation),
85
+ BP1 (missense in truncating-only gene), BP2 (in trans/cis with pathogenic),
86
+ BP3 (in-frame del/ins in repeat), BP4 (computational benign predictions),
87
+ BP5 (alternate molecular cause), BP7 (synonymous, no splice impact).
88
+
89
+ **Combining Rules:**
90
+ - Pathogenic: 1 VS + 1 S; OR 2 S; OR 1 S + 3 M; OR 1 S + 2 M + 2 P; OR 1 S + 1 M + 4 P
91
+ - Likely Pathogenic: 1 VS + 1 M; OR 1 S + 1-2 M; OR 1 S + 2 P; OR 3 M; OR 2 M + 2 P; OR 1 M + 4 P
92
+ - Benign: 1 stand-alone (BA1) OR 2 strong
93
+ - Likely Benign: 1 strong + 1 supporting OR 2 supporting
94
+ - VUS: criteria don't meet thresholds (use "leaning pathogenic" / "leaning benign")
95
+
96
+ **ACMG output (JSON ONLY):**
97
+ {
98
+ "classification": "Pathogenic | Likely Pathogenic | VUS (leaning pathogenic) | VUS (leaning benign) | Likely Benign | Benign",
99
+ "confidence": "high | medium | low",
100
+ "acmg_score": <int>,
101
+ "criteria_met": ["PM2_Supporting", "PP3", ...],
102
+ "evidence_summary": {"PM2_Supporting": "Absent in gnomAD v4.0 ...", "PP3": "CADD=28.5 ...", "notable": "..."},
103
+ "classification_rationale": "...",
104
+ "sources": ["ClinVar:12215", "PubMed:12345678", "gnomAD v4.0"]
105
+ }
106
+
107
+ # AMP/ASCO/CAP 2017 (Somatic) — `amp`
108
+
109
+ **Tier Framework:**
110
+ - Tier I-A: FDA-approved therapy for this tumor type or in NCCN/ASCO/CAP guidelines.
111
+ - Tier I-B: Well-powered studies with expert consensus on clinical utility.
112
+ - Tier II-C: FDA-approved for different tumor type, or trial eligibility.
113
+ - Tier II-D: Preclinical / case reports / biological rationale.
114
+ - Tier III: Insufficient evidence.
115
+ - Tier IV: Benign / common germline / no functional impact.
116
+
117
+ **Evidence categories:** Therapeutic (TA/TB/TC/TD), Diagnostic (DA/DB/DC/DD),
118
+ Prognostic (PA/PB/PC/PD). Benign: BN1 (>1% gnomAD), BN2 (no functional impact —
119
+ requires actual scores), BN3 (germline polymorphism), BN4 (synonymous, no splice).
120
+
121
+ **Critical Validation Rule:** BN2 only when actual scores show CADD<15 AND DITTO<0.5,
122
+ SIFT=tolerated, PolyPhen=benign. Do NOT apply BN2 from domain inference.
123
+
124
+ **AMP output (JSON ONLY):**
125
+ {
126
+ "classification": "Tier I-A | Tier I-B | Tier II-C | Tier II-D | Tier III | Tier IV",
127
+ "confidence": "high | medium | low",
128
+ "criteria_met": ["TA", "PB", ...],
129
+ "evidence_summary": {"TA": "FDA-approved drug X for this tumor type ...", "PB": "..."},
130
+ "therapeutic_summary": "...",
131
+ "diagnostic_summary": "...",
132
+ "prognostic_summary": "...",
133
+ "sources": ["OncoKB", "ClinVar:12345", "PubMed:...", "NCCN Guidelines"]
134
+ }
135
+
136
+ # General
137
+
138
+ - Internally evaluate ALL criteria but only output the criteria that are actually met.
139
+ - Return ONLY valid JSON — no prose before or after, no markdown fences.
140
+ - If evidence is missing, mark a criterion not_applicable with a brief explanation.
141
+ - DO NOT use web_search for variant-specific data (gnomAD, CADD, ClinVar) — those
142
+ come from annotate_variant.
143
+ """