vantage-core 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- vantage_core/__init__.py +20 -0
- vantage_core/__main__.py +6 -0
- vantage_core/cli.py +255 -0
- vantage_core/contract.py +261 -0
- vantage_core/cost.py +77 -0
- vantage_core/data/model_rates_slim.json +9 -0
- vantage_core/decision.py +290 -0
- vantage_core/fixtures/de_sql_slow_query.sql +13 -0
- vantage_core/library/__init__.py +81 -0
- vantage_core/llm_openrouter.py +126 -0
- vantage_core/monorepo_run.py +181 -0
- vantage_core/py.typed +0 -0
- vantage_core/run_store.py +43 -0
- vantage_core/runner.py +132 -0
- vantage_core/schemas/decision_object.v1.json +117 -0
- vantage_core/scorers/__init__.py +20 -0
- vantage_core/scorers/hard_checks.py +98 -0
- vantage_core/scorers/sql_optimization.py +150 -0
- vantage_core/task_runner.py +124 -0
- vantage_core/trust.py +109 -0
- vantage_core-0.1.0.dist-info/METADATA +175 -0
- vantage_core-0.1.0.dist-info/RECORD +26 -0
- vantage_core-0.1.0.dist-info/WHEEL +5 -0
- vantage_core-0.1.0.dist-info/entry_points.txt +2 -0
- vantage_core-0.1.0.dist-info/licenses/LICENSE +21 -0
- vantage_core-0.1.0.dist-info/top_level.txt +1 -0
vantage_core/__init__.py
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""vantage-core — RuntimeAI check-ride CLI and portable decision artifact."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__version__ = "0.1.0"
|
|
6
|
+
|
|
7
|
+
from vantage_core.decision import (
|
|
8
|
+
SCHEMA_ID,
|
|
9
|
+
build_decision_object,
|
|
10
|
+
payload_sha256,
|
|
11
|
+
validate_decision_object,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"SCHEMA_ID",
|
|
16
|
+
"__version__",
|
|
17
|
+
"build_decision_object",
|
|
18
|
+
"payload_sha256",
|
|
19
|
+
"validate_decision_object",
|
|
20
|
+
]
|
vantage_core/__main__.py
ADDED
vantage_core/cli.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""vantage-core CLI — check-ride run + decision schema helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import os
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from vantage_core import SCHEMA_ID, __version__, validate_decision_object
|
|
13
|
+
from vantage_core.decision import SCHEMA_ID as _SCHEMA
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _load_env() -> None:
|
|
17
|
+
try:
|
|
18
|
+
from dotenv import load_dotenv
|
|
19
|
+
except ImportError:
|
|
20
|
+
return
|
|
21
|
+
load_dotenv(Path.cwd() / ".env")
|
|
22
|
+
home = os.environ.get("VANTAGE_HOME")
|
|
23
|
+
if home:
|
|
24
|
+
load_dotenv(Path(home) / ".env")
|
|
25
|
+
load_dotenv(Path(home) / "server" / ".env")
|
|
26
|
+
# Monorepo convenience when present
|
|
27
|
+
try:
|
|
28
|
+
from vantage_core.monorepo_run import find_server_root
|
|
29
|
+
|
|
30
|
+
root = find_server_root()
|
|
31
|
+
if root is not None:
|
|
32
|
+
load_dotenv(root.parent / ".env")
|
|
33
|
+
load_dotenv(root / ".env")
|
|
34
|
+
except Exception:
|
|
35
|
+
pass
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _print_decision(decision: dict[str, Any], *, as_json: bool) -> None:
|
|
39
|
+
passed = bool(decision.get("passed"))
|
|
40
|
+
score = decision.get("out_of_10")
|
|
41
|
+
cost = decision.get("est_usd")
|
|
42
|
+
cost_s = f"~${cost:.4f}" if isinstance(cost, (int, float)) else "n/a"
|
|
43
|
+
verdict = "PASS" if passed else "FAIL"
|
|
44
|
+
score_s = f"{float(score):.1f}" if isinstance(score, (int, float)) else "n/a"
|
|
45
|
+
|
|
46
|
+
if as_json:
|
|
47
|
+
print(json.dumps(decision, indent=2))
|
|
48
|
+
else:
|
|
49
|
+
print(f"score {score_s} / 10 · {cost_s} · {verdict}")
|
|
50
|
+
print(f"session {decision.get('session_id')}")
|
|
51
|
+
gate = decision.get("pass_gate") if isinstance(decision.get("pass_gate"), dict) else {}
|
|
52
|
+
if gate.get("headline"):
|
|
53
|
+
print(f"gate {gate['headline']}")
|
|
54
|
+
print(f"scorecard re-run with --json ({SCHEMA_ID})")
|
|
55
|
+
if decision.get("status") != "ended" and decision.get("error"):
|
|
56
|
+
print(f"status {decision.get('status')}: {decision.get('error')}", file=sys.stderr)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def cmd_run(args: argparse.Namespace) -> int:
|
|
60
|
+
_load_env()
|
|
61
|
+
|
|
62
|
+
use_monorepo = (
|
|
63
|
+
bool(getattr(args, "monorepo", False))
|
|
64
|
+
or (os.getenv("VANTAGE_USE_MONOREPO") or "").strip() in ("1", "true", "yes")
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
if use_monorepo and not args.contract:
|
|
68
|
+
from vantage_core.monorepo_run import bootstrap_server, run_checkride as mono_run
|
|
69
|
+
|
|
70
|
+
try:
|
|
71
|
+
server = bootstrap_server()
|
|
72
|
+
except RuntimeError as exc:
|
|
73
|
+
print(str(exc), file=sys.stderr)
|
|
74
|
+
return 2
|
|
75
|
+
if not server._openrouter_api_key():
|
|
76
|
+
print("OPENROUTER_API_KEY not configured.", file=sys.stderr)
|
|
77
|
+
return 2
|
|
78
|
+
try:
|
|
79
|
+
decision = mono_run(
|
|
80
|
+
server,
|
|
81
|
+
scenario_id=args.scenario,
|
|
82
|
+
model=args.model,
|
|
83
|
+
turns=args.turns,
|
|
84
|
+
timeout_s=args.timeout,
|
|
85
|
+
fail_under=float(args.fail_under),
|
|
86
|
+
runner_version=__version__,
|
|
87
|
+
)
|
|
88
|
+
except Exception as exc:
|
|
89
|
+
print(f"vantage-core run failed: {exc}", file=sys.stderr)
|
|
90
|
+
return 1
|
|
91
|
+
_print_decision(decision, as_json=args.json)
|
|
92
|
+
return 0 if decision.get("passed") else 1
|
|
93
|
+
|
|
94
|
+
# Standalone path (default)
|
|
95
|
+
from vantage_core.contract import contract_from_library_id, load_contract
|
|
96
|
+
from vantage_core.llm_openrouter import openrouter_api_key
|
|
97
|
+
from vantage_core.runner import run_checkride
|
|
98
|
+
|
|
99
|
+
if not openrouter_api_key():
|
|
100
|
+
print(
|
|
101
|
+
"OPENROUTER_API_KEY not configured. Export it or put it in .env.",
|
|
102
|
+
file=sys.stderr,
|
|
103
|
+
)
|
|
104
|
+
return 2
|
|
105
|
+
|
|
106
|
+
try:
|
|
107
|
+
if args.contract:
|
|
108
|
+
contract = load_contract(args.contract)
|
|
109
|
+
elif args.scenario:
|
|
110
|
+
contract = contract_from_library_id(
|
|
111
|
+
args.scenario,
|
|
112
|
+
fail_under=float(args.fail_under),
|
|
113
|
+
turns=args.turns,
|
|
114
|
+
model=args.model,
|
|
115
|
+
)
|
|
116
|
+
else:
|
|
117
|
+
print("Provide --contract PATH or --scenario ID", file=sys.stderr)
|
|
118
|
+
return 2
|
|
119
|
+
|
|
120
|
+
decision = run_checkride(
|
|
121
|
+
contract,
|
|
122
|
+
model=args.model,
|
|
123
|
+
fail_under=float(args.fail_under) if args.fail_under is not None else None,
|
|
124
|
+
turns=args.turns,
|
|
125
|
+
timeout_s=args.timeout,
|
|
126
|
+
runner_version=__version__,
|
|
127
|
+
)
|
|
128
|
+
except Exception as exc:
|
|
129
|
+
print(f"vantage-core run failed: {exc}", file=sys.stderr)
|
|
130
|
+
return 1
|
|
131
|
+
|
|
132
|
+
_print_decision(decision, as_json=args.json)
|
|
133
|
+
return 0 if decision.get("passed") else 1
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def cmd_schema(_args: argparse.Namespace) -> int:
|
|
137
|
+
pkg_dir = Path(__file__).resolve().parent
|
|
138
|
+
schema_path = pkg_dir / "schemas" / "decision_object.v1.json"
|
|
139
|
+
if not schema_path.is_file():
|
|
140
|
+
schema_path = pkg_dir.parent / "schemas" / "decision_object.v1.json"
|
|
141
|
+
print(f"schema {_SCHEMA}")
|
|
142
|
+
print(f"contract_schema runtimeai.contract/v1")
|
|
143
|
+
print(f"version {__version__}")
|
|
144
|
+
if schema_path.is_file():
|
|
145
|
+
print(f"json_schema {schema_path}")
|
|
146
|
+
print(
|
|
147
|
+
"fields contract + scorecard.pass_gate + usd.est_eval + exit + "
|
|
148
|
+
"integrity.payload_sha256"
|
|
149
|
+
)
|
|
150
|
+
from vantage_core.library import list_library_ids
|
|
151
|
+
|
|
152
|
+
print(f"library {', '.join(list_library_ids())}")
|
|
153
|
+
return 0
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def cmd_validate(args: argparse.Namespace) -> int:
|
|
157
|
+
path = Path(args.path)
|
|
158
|
+
# Decision JSON or contract YAML/JSON
|
|
159
|
+
suffix = path.suffix.lower()
|
|
160
|
+
if suffix in (".yaml", ".yml") or (
|
|
161
|
+
suffix == ".json" and "contract" in path.name.lower()
|
|
162
|
+
):
|
|
163
|
+
try:
|
|
164
|
+
from vantage_core.contract import load_contract
|
|
165
|
+
|
|
166
|
+
contract = load_contract(path)
|
|
167
|
+
except Exception as exc:
|
|
168
|
+
print(f"INVALID contract {path}: {exc}", file=sys.stderr)
|
|
169
|
+
return 1
|
|
170
|
+
print(f"VALID contract {path} ({contract.schema} · {contract.id} · {contract.mode})")
|
|
171
|
+
return 0
|
|
172
|
+
|
|
173
|
+
try:
|
|
174
|
+
data: Any = json.loads(path.read_text(encoding="utf-8"))
|
|
175
|
+
except Exception as exc:
|
|
176
|
+
print(f"failed to read JSON: {exc}", file=sys.stderr)
|
|
177
|
+
return 2
|
|
178
|
+
# Heuristic: contract JSON vs decision JSON
|
|
179
|
+
if isinstance(data, dict) and data.get("schema") == "runtimeai.contract/v1":
|
|
180
|
+
try:
|
|
181
|
+
from vantage_core.contract import resolve_contract
|
|
182
|
+
|
|
183
|
+
contract = resolve_contract(data, source_path=path)
|
|
184
|
+
except Exception as exc:
|
|
185
|
+
print(f"INVALID contract {path}: {exc}", file=sys.stderr)
|
|
186
|
+
return 1
|
|
187
|
+
print(f"VALID contract {path} ({contract.schema} · {contract.id})")
|
|
188
|
+
return 0
|
|
189
|
+
|
|
190
|
+
errors = validate_decision_object(data)
|
|
191
|
+
if errors:
|
|
192
|
+
print(f"INVALID {path} ({len(errors)} error(s))", file=sys.stderr)
|
|
193
|
+
for err in errors:
|
|
194
|
+
print(f" - {err}", file=sys.stderr)
|
|
195
|
+
return 1
|
|
196
|
+
print(f"VALID {path} ({SCHEMA_ID})")
|
|
197
|
+
return 0
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
201
|
+
p = argparse.ArgumentParser(
|
|
202
|
+
prog="vantage-core",
|
|
203
|
+
description=(
|
|
204
|
+
"RuntimeAI check-ride CLI. Emits a portable "
|
|
205
|
+
f"{SCHEMA_ID} decision artifact. Standalone — no monorepo required."
|
|
206
|
+
),
|
|
207
|
+
)
|
|
208
|
+
p.add_argument("--version", action="version", version=f"%(prog)s {__version__}")
|
|
209
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
210
|
+
|
|
211
|
+
run_p = sub.add_parser("run", help="Run a local contract or bundled library scenario")
|
|
212
|
+
run_p.add_argument(
|
|
213
|
+
"--contract",
|
|
214
|
+
help="Path to runtimeai.contract/v1 YAML or JSON",
|
|
215
|
+
)
|
|
216
|
+
run_p.add_argument(
|
|
217
|
+
"--scenario",
|
|
218
|
+
help="Bundled library scenario id (e.g. de_sql_optimization_v1)",
|
|
219
|
+
)
|
|
220
|
+
run_p.add_argument("--model", default="openai/gpt-4o-mini")
|
|
221
|
+
run_p.add_argument("--turns", type=int, default=None, help="Override contract/library turns")
|
|
222
|
+
run_p.add_argument("--fail-under", type=float, default=7.0)
|
|
223
|
+
run_p.add_argument("--timeout", type=float, default=180.0)
|
|
224
|
+
run_p.add_argument("--json", action="store_true")
|
|
225
|
+
run_p.add_argument(
|
|
226
|
+
"--monorepo",
|
|
227
|
+
action="store_true",
|
|
228
|
+
help="Force monorepo server path (dev parity; needs Vantage clone)",
|
|
229
|
+
)
|
|
230
|
+
run_p.set_defaults(func=cmd_run)
|
|
231
|
+
|
|
232
|
+
schema_p = sub.add_parser("schema", help="Show frozen decision + contract schema ids")
|
|
233
|
+
schema_p.set_defaults(func=cmd_schema)
|
|
234
|
+
|
|
235
|
+
val_p = sub.add_parser("validate", help="Validate a decision JSON or contract YAML/JSON")
|
|
236
|
+
val_p.add_argument("path", help="Path to decision or contract file")
|
|
237
|
+
val_p.set_defaults(func=cmd_validate)
|
|
238
|
+
return p
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def main(argv: list[str] | None = None) -> int:
|
|
242
|
+
parser = build_parser()
|
|
243
|
+
args = parser.parse_args(argv)
|
|
244
|
+
if getattr(args, "command", None) == "run":
|
|
245
|
+
if not args.contract and not args.scenario:
|
|
246
|
+
parser.error("run requires --contract or --scenario")
|
|
247
|
+
return int(args.func(args))
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def cli_entry() -> None:
|
|
251
|
+
raise SystemExit(main())
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
if __name__ == "__main__":
|
|
255
|
+
cli_entry()
|
vantage_core/contract.py
ADDED
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
"""runtimeai.contract/v1 — locally authored check-ride contracts."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from dataclasses import dataclass, field
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
CONTRACT_SCHEMA = "runtimeai.contract/v1"
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class HardCheck:
|
|
16
|
+
id: str
|
|
17
|
+
points: int = 5
|
|
18
|
+
any_of: list[str] = field(default_factory=list)
|
|
19
|
+
none_of: list[str] = field(default_factory=list)
|
|
20
|
+
hard_fail: bool = False
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class ResolvedContract:
|
|
25
|
+
schema: str
|
|
26
|
+
id: str
|
|
27
|
+
name: str
|
|
28
|
+
mode: str # library_replay | custom
|
|
29
|
+
fail_under: float
|
|
30
|
+
turns: int
|
|
31
|
+
model: str | None
|
|
32
|
+
agent_system: str
|
|
33
|
+
opening: str
|
|
34
|
+
followups: list[str]
|
|
35
|
+
scorer_kind: str # library:de_sql_optimization_v1 | hard_checks
|
|
36
|
+
hard_checks: list[HardCheck]
|
|
37
|
+
library_scenario_id: str | None
|
|
38
|
+
source_path: Path | None = None
|
|
39
|
+
|
|
40
|
+
def content_sha256(self) -> str:
|
|
41
|
+
payload = {
|
|
42
|
+
"id": self.id,
|
|
43
|
+
"mode": self.mode,
|
|
44
|
+
"agent_system": self.agent_system,
|
|
45
|
+
"opening": self.opening,
|
|
46
|
+
"followups": self.followups,
|
|
47
|
+
"scorer_kind": self.scorer_kind,
|
|
48
|
+
"hard_checks": [
|
|
49
|
+
{
|
|
50
|
+
"id": c.id,
|
|
51
|
+
"points": c.points,
|
|
52
|
+
"any_of": c.any_of,
|
|
53
|
+
"none_of": c.none_of,
|
|
54
|
+
"hard_fail": c.hard_fail,
|
|
55
|
+
}
|
|
56
|
+
for c in self.hard_checks
|
|
57
|
+
],
|
|
58
|
+
"library_scenario_id": self.library_scenario_id,
|
|
59
|
+
}
|
|
60
|
+
raw = json.dumps(payload, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
|
|
61
|
+
return hashlib.sha256(raw.encode("utf-8")).hexdigest()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _load_raw(path: Path) -> dict[str, Any]:
|
|
65
|
+
text = path.read_text(encoding="utf-8")
|
|
66
|
+
suffix = path.suffix.lower()
|
|
67
|
+
if suffix in (".yaml", ".yml"):
|
|
68
|
+
try:
|
|
69
|
+
import yaml
|
|
70
|
+
except ImportError as exc:
|
|
71
|
+
raise RuntimeError(
|
|
72
|
+
"PyYAML is required for .yaml contracts. "
|
|
73
|
+
"Install: pip install 'vantage-core[run]' or pip install pyyaml"
|
|
74
|
+
) from exc
|
|
75
|
+
data = yaml.safe_load(text)
|
|
76
|
+
else:
|
|
77
|
+
data = json.loads(text)
|
|
78
|
+
if not isinstance(data, dict):
|
|
79
|
+
raise ValueError(f"contract root must be a mapping: {path}")
|
|
80
|
+
return data
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _as_str_list(val: Any) -> list[str]:
|
|
84
|
+
if val is None:
|
|
85
|
+
return []
|
|
86
|
+
if isinstance(val, str):
|
|
87
|
+
return [val]
|
|
88
|
+
if isinstance(val, list):
|
|
89
|
+
return [str(x) for x in val if str(x).strip()]
|
|
90
|
+
raise ValueError(f"expected string or list, got {type(val).__name__}")
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _parse_hard_checks(raw: Any) -> list[HardCheck]:
|
|
94
|
+
if not raw:
|
|
95
|
+
return []
|
|
96
|
+
if not isinstance(raw, list):
|
|
97
|
+
raise ValueError("scorer.checks must be a list")
|
|
98
|
+
out: list[HardCheck] = []
|
|
99
|
+
for i, item in enumerate(raw):
|
|
100
|
+
if not isinstance(item, dict):
|
|
101
|
+
raise ValueError(f"scorer.checks[{i}] must be an object")
|
|
102
|
+
cid = str(item.get("id") or f"check_{i}").strip()
|
|
103
|
+
out.append(
|
|
104
|
+
HardCheck(
|
|
105
|
+
id=cid,
|
|
106
|
+
points=int(item.get("points") or 5),
|
|
107
|
+
any_of=_as_str_list(item.get("any_of")),
|
|
108
|
+
none_of=_as_str_list(item.get("none_of")),
|
|
109
|
+
hard_fail=bool(item.get("hard_fail")),
|
|
110
|
+
)
|
|
111
|
+
)
|
|
112
|
+
return out
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def resolve_contract(data: dict[str, Any], *, source_path: Path | None = None) -> ResolvedContract:
|
|
116
|
+
schema = str(data.get("schema") or CONTRACT_SCHEMA).strip()
|
|
117
|
+
if schema != CONTRACT_SCHEMA:
|
|
118
|
+
raise ValueError(f"unsupported contract schema {schema!r}; expected {CONTRACT_SCHEMA!r}")
|
|
119
|
+
|
|
120
|
+
cid = str(data.get("id") or "").strip()
|
|
121
|
+
if not cid:
|
|
122
|
+
raise ValueError("contract.id is required")
|
|
123
|
+
name = str(data.get("name") or cid).strip()
|
|
124
|
+
mode = str(data.get("mode") or "").strip().lower()
|
|
125
|
+
if mode not in ("library_replay", "custom"):
|
|
126
|
+
raise ValueError("contract.mode must be 'library_replay' or 'custom'")
|
|
127
|
+
|
|
128
|
+
fail_under = float(data.get("fail_under") if data.get("fail_under") is not None else 7.0)
|
|
129
|
+
turns = max(1, min(24, int(data.get("turns") or 1)))
|
|
130
|
+
model = str(data.get("model") or "").strip() or None
|
|
131
|
+
|
|
132
|
+
from vantage_core.library import get_library_scenario
|
|
133
|
+
|
|
134
|
+
if mode == "library_replay":
|
|
135
|
+
lib = data.get("library") if isinstance(data.get("library"), dict) else {}
|
|
136
|
+
lib_id = str(lib.get("scenario_id") or "").strip()
|
|
137
|
+
if not lib_id:
|
|
138
|
+
raise ValueError("library_replay requires library.scenario_id")
|
|
139
|
+
bundled = get_library_scenario(lib_id)
|
|
140
|
+
if bundled is None:
|
|
141
|
+
raise ValueError(
|
|
142
|
+
f"unknown library.scenario_id {lib_id!r}. "
|
|
143
|
+
f"Bundled: {', '.join(sorted(_library_ids()))}"
|
|
144
|
+
)
|
|
145
|
+
fixture = lib.get("fixture")
|
|
146
|
+
if fixture is None:
|
|
147
|
+
opening = bundled.default_opening()
|
|
148
|
+
else:
|
|
149
|
+
opening = bundled.opening_template.format(fixture=str(fixture).strip())
|
|
150
|
+
agent_system = str(lib.get("system_prompt") or bundled.agent_system).strip()
|
|
151
|
+
followups = _as_str_list(lib.get("followups")) or list(bundled.followups)
|
|
152
|
+
scorer_kind = f"library:{lib_id}"
|
|
153
|
+
return ResolvedContract(
|
|
154
|
+
schema=schema,
|
|
155
|
+
id=cid,
|
|
156
|
+
name=name,
|
|
157
|
+
mode=mode,
|
|
158
|
+
fail_under=fail_under,
|
|
159
|
+
turns=turns,
|
|
160
|
+
model=model,
|
|
161
|
+
agent_system=agent_system,
|
|
162
|
+
opening=opening,
|
|
163
|
+
followups=followups,
|
|
164
|
+
scorer_kind=scorer_kind,
|
|
165
|
+
hard_checks=[],
|
|
166
|
+
library_scenario_id=lib_id,
|
|
167
|
+
source_path=source_path,
|
|
168
|
+
)
|
|
169
|
+
|
|
170
|
+
# custom
|
|
171
|
+
agent = data.get("agent") if isinstance(data.get("agent"), dict) else {}
|
|
172
|
+
agent_system = str(agent.get("system") or "").strip()
|
|
173
|
+
opening = str(agent.get("opening") or "").strip()
|
|
174
|
+
if not agent_system or not opening:
|
|
175
|
+
raise ValueError("custom mode requires agent.system and agent.opening")
|
|
176
|
+
followups = _as_str_list(agent.get("followups"))
|
|
177
|
+
|
|
178
|
+
scorer = data.get("scorer") if isinstance(data.get("scorer"), dict) else {}
|
|
179
|
+
kind = str(scorer.get("kind") or "hard_checks").strip().lower()
|
|
180
|
+
if kind.startswith("library:"):
|
|
181
|
+
lib_id = kind.split(":", 1)[1].strip()
|
|
182
|
+
if get_library_scenario(lib_id) is None:
|
|
183
|
+
raise ValueError(f"unknown scorer library id {lib_id!r}")
|
|
184
|
+
scorer_kind = kind
|
|
185
|
+
hard_checks: list[HardCheck] = []
|
|
186
|
+
elif kind == "hard_checks":
|
|
187
|
+
hard_checks = _parse_hard_checks(scorer.get("checks"))
|
|
188
|
+
if not hard_checks:
|
|
189
|
+
raise ValueError("hard_checks scorer requires at least one check")
|
|
190
|
+
scorer_kind = "hard_checks"
|
|
191
|
+
elif kind == "heuristic_sql_v1":
|
|
192
|
+
scorer_kind = "library:de_sql_optimization_v1"
|
|
193
|
+
hard_checks = []
|
|
194
|
+
else:
|
|
195
|
+
raise ValueError(
|
|
196
|
+
f"unsupported scorer.kind {kind!r}; use hard_checks or library:<id>"
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
return ResolvedContract(
|
|
200
|
+
schema=schema,
|
|
201
|
+
id=cid,
|
|
202
|
+
name=name,
|
|
203
|
+
mode=mode,
|
|
204
|
+
fail_under=fail_under,
|
|
205
|
+
turns=turns,
|
|
206
|
+
model=model,
|
|
207
|
+
agent_system=agent_system,
|
|
208
|
+
opening=opening,
|
|
209
|
+
followups=followups,
|
|
210
|
+
scorer_kind=scorer_kind,
|
|
211
|
+
hard_checks=hard_checks,
|
|
212
|
+
library_scenario_id=None,
|
|
213
|
+
source_path=source_path,
|
|
214
|
+
)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _library_ids() -> list[str]:
|
|
218
|
+
from vantage_core.library import list_library_ids
|
|
219
|
+
|
|
220
|
+
return list_library_ids()
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def load_contract(path: str | Path) -> ResolvedContract:
|
|
224
|
+
p = Path(path).expanduser().resolve()
|
|
225
|
+
if not p.is_file():
|
|
226
|
+
raise FileNotFoundError(f"contract not found: {p}")
|
|
227
|
+
return resolve_contract(_load_raw(p), source_path=p)
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def contract_from_library_id(
|
|
231
|
+
scenario_id: str,
|
|
232
|
+
*,
|
|
233
|
+
fail_under: float = 7.0,
|
|
234
|
+
turns: int | None = None,
|
|
235
|
+
model: str | None = None,
|
|
236
|
+
) -> ResolvedContract:
|
|
237
|
+
"""Build a resolved contract from a bundled library scenario id."""
|
|
238
|
+
from vantage_core.library import get_library_scenario
|
|
239
|
+
|
|
240
|
+
bundled = get_library_scenario(scenario_id)
|
|
241
|
+
if bundled is None:
|
|
242
|
+
raise ValueError(
|
|
243
|
+
f"unknown scenario {scenario_id!r}. "
|
|
244
|
+
f"Bundled: {', '.join(sorted(_library_ids()))} — or pass --contract"
|
|
245
|
+
)
|
|
246
|
+
return ResolvedContract(
|
|
247
|
+
schema=CONTRACT_SCHEMA,
|
|
248
|
+
id=scenario_id,
|
|
249
|
+
name=bundled.name,
|
|
250
|
+
mode="library_replay",
|
|
251
|
+
fail_under=float(fail_under),
|
|
252
|
+
turns=max(1, min(24, int(turns if turns is not None else bundled.default_turns))),
|
|
253
|
+
model=model,
|
|
254
|
+
agent_system=bundled.agent_system,
|
|
255
|
+
opening=bundled.default_opening(),
|
|
256
|
+
followups=list(bundled.followups),
|
|
257
|
+
scorer_kind=f"library:{scenario_id}",
|
|
258
|
+
hard_checks=[],
|
|
259
|
+
library_scenario_id=scenario_id,
|
|
260
|
+
source_path=None,
|
|
261
|
+
)
|
vantage_core/cost.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Slim USD estimate for standalone runs."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from importlib import resources
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
# Defaults match server task-cost assumptions.
|
|
11
|
+
_AVG_IN = 1500.0
|
|
12
|
+
_AVG_OUT = 350.0
|
|
13
|
+
|
|
14
|
+
_RATES: dict[str, dict[str, float]] | None = None
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _load_rates() -> dict[str, dict[str, float]]:
|
|
18
|
+
global _RATES
|
|
19
|
+
if _RATES is not None:
|
|
20
|
+
return _RATES
|
|
21
|
+
try:
|
|
22
|
+
raw = (resources.files("vantage_core") / "data" / "model_rates_slim.json").read_text(
|
|
23
|
+
encoding="utf-8"
|
|
24
|
+
)
|
|
25
|
+
data = json.loads(raw)
|
|
26
|
+
except Exception:
|
|
27
|
+
path = Path(__file__).resolve().parent / "data" / "model_rates_slim.json"
|
|
28
|
+
data = json.loads(path.read_text(encoding="utf-8")) if path.is_file() else {}
|
|
29
|
+
out: dict[str, dict[str, float]] = {}
|
|
30
|
+
if isinstance(data, dict):
|
|
31
|
+
for mid, row in data.items():
|
|
32
|
+
if not isinstance(row, dict):
|
|
33
|
+
continue
|
|
34
|
+
try:
|
|
35
|
+
out[str(mid)] = {
|
|
36
|
+
"input": float(row.get("input_token_rate_usd") or row.get("input") or 0),
|
|
37
|
+
"output": float(row.get("output_token_rate_usd") or row.get("output") or 0),
|
|
38
|
+
}
|
|
39
|
+
except (TypeError, ValueError):
|
|
40
|
+
continue
|
|
41
|
+
_RATES = out
|
|
42
|
+
return out
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _lookup(model: str) -> dict[str, float] | None:
|
|
46
|
+
rates = _load_rates()
|
|
47
|
+
mid = (model or "").strip()
|
|
48
|
+
if mid in rates:
|
|
49
|
+
return rates[mid]
|
|
50
|
+
# Tail match: openai/gpt-4o-mini ↔ gpt-4o-mini
|
|
51
|
+
if "/" in mid:
|
|
52
|
+
tail = mid.split("/", 1)[1]
|
|
53
|
+
if tail in rates:
|
|
54
|
+
return rates[tail]
|
|
55
|
+
for key, val in rates.items():
|
|
56
|
+
if key.endswith("/" + mid) or key.endswith(mid):
|
|
57
|
+
return val
|
|
58
|
+
return None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def estimate_run_cost_usd(run: dict[str, Any]) -> float | None:
|
|
62
|
+
model = str(run.get("model") or "").strip()
|
|
63
|
+
rates = _lookup(model)
|
|
64
|
+
if not rates:
|
|
65
|
+
return None
|
|
66
|
+
sim = [e for e in (run.get("events") or []) if isinstance(e, dict) and e.get("kind") == "sim"]
|
|
67
|
+
# Count agent (protagonist) turns only — same as server task cost.
|
|
68
|
+
calls = sum(
|
|
69
|
+
1
|
|
70
|
+
for e in sim
|
|
71
|
+
if e.get("role") in ("pm", "salesops", "sales_rep", "assistant")
|
|
72
|
+
and str(e.get("content") or "").strip()
|
|
73
|
+
)
|
|
74
|
+
if calls <= 0:
|
|
75
|
+
return None
|
|
76
|
+
usd = calls * (_AVG_IN * rates["input"] + _AVG_OUT * rates["output"])
|
|
77
|
+
return round(usd, 8)
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
{
|
|
2
|
+
"openai/gpt-4o-mini": {"input_token_rate_usd": 0.00000015, "output_token_rate_usd": 0.0000006},
|
|
3
|
+
"openai/gpt-4o": {"input_token_rate_usd": 0.0000025, "output_token_rate_usd": 0.00001},
|
|
4
|
+
"amazon/nova-micro-v1": {"input_token_rate_usd": 0.000000035, "output_token_rate_usd": 0.00000014},
|
|
5
|
+
"amazon/nova-lite-v1": {"input_token_rate_usd": 0.00000006, "output_token_rate_usd": 0.00000024},
|
|
6
|
+
"anthropic/claude-haiku-4.5": {"input_token_rate_usd": 0.000001, "output_token_rate_usd": 0.000005},
|
|
7
|
+
"anthropic/claude-sonnet-4": {"input_token_rate_usd": 0.000003, "output_token_rate_usd": 0.000015},
|
|
8
|
+
"google/gemini-2.5-flash": {"input_token_rate_usd": 0.0000003, "output_token_rate_usd": 0.0000025}
|
|
9
|
+
}
|