devin-metrics 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- devin_metrics/__init__.py +1 -0
- devin_metrics/aggregate.py +199 -0
- devin_metrics/churn.py +141 -0
- devin_metrics/cli.py +246 -0
- devin_metrics/collect.py +240 -0
- devin_metrics/dashboard/__init__.py +1 -0
- devin_metrics/dashboard/cli.py +137 -0
- devin_metrics/dashboard/collect.py +415 -0
- devin_metrics/dashboard/render.py +287 -0
- devin_metrics/paths.py +81 -0
- devin_metrics/render.py +170 -0
- devin_metrics-0.2.0.dist-info/METADATA +210 -0
- devin_metrics-0.2.0.dist-info/RECORD +17 -0
- devin_metrics-0.2.0.dist-info/WHEEL +5 -0
- devin_metrics-0.2.0.dist-info/entry_points.txt +3 -0
- devin_metrics-0.2.0.dist-info/licenses/LICENSE +21 -0
- devin_metrics-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""Rollups over a :class:`MetricsSnapshot` — pure functions, JSON-ready dicts.
|
|
2
|
+
|
|
3
|
+
Attribution rules:
|
|
4
|
+
|
|
5
|
+
- Sessions group by ``working_directory`` (the project) — free with Devin's
|
|
6
|
+
session store.
|
|
7
|
+
- Day buckets use the session's ``created_at`` in UTC.
|
|
8
|
+
- ``cost_usd_total`` counts **all** acp usage rows (including orphan dbs with
|
|
9
|
+
no session row); ``cost_usd_by_sessions`` counts only matched sessions.
|
|
10
|
+
- ``cost_usd is None`` means *unknown* (no acp data), never zero.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from datetime import datetime, timezone
|
|
16
|
+
from typing import Any, Iterable
|
|
17
|
+
|
|
18
|
+
from devin_metrics.collect import MetricsSnapshot, SessionMetrics, UsageRecord
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _day(ms: int) -> str:
|
|
22
|
+
return datetime.fromtimestamp(ms / 1000, tz=timezone.utc).date().isoformat()
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def session_to_dict(s: SessionMetrics) -> dict[str, Any]:
|
|
26
|
+
return {
|
|
27
|
+
"id": s.id,
|
|
28
|
+
"title": s.title,
|
|
29
|
+
"project": s.working_directory,
|
|
30
|
+
"model": s.model,
|
|
31
|
+
"created": _day(s.created_at),
|
|
32
|
+
"created_at": s.created_at,
|
|
33
|
+
"duration_ms": s.duration_ms,
|
|
34
|
+
"messages": s.n_messages,
|
|
35
|
+
"tool_calls": s.n_tool_calls,
|
|
36
|
+
"cost_usd": s.cost_usd,
|
|
37
|
+
"input_tokens": s.input_tokens,
|
|
38
|
+
"output_tokens": s.output_tokens,
|
|
39
|
+
"context_tokens": s.context_tokens,
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _sum_or_none(values: Iterable[float | int | None]) -> float | int | None:
|
|
44
|
+
known = [v for v in values if v is not None]
|
|
45
|
+
return sum(known) if known else None
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _sum(values: Iterable[float | int]) -> float | int:
|
|
49
|
+
return sum(values)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
# -- summaries ---------------------------------------------------------------
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def summarize(snap: MetricsSnapshot) -> dict[str, Any]:
|
|
56
|
+
sessions = list(snap.sessions)
|
|
57
|
+
n = len(sessions)
|
|
58
|
+
with_cost = [s for s in sessions if s.cost_usd is not None]
|
|
59
|
+
return {
|
|
60
|
+
"sessions": n,
|
|
61
|
+
"sessions_with_cost": len(with_cost),
|
|
62
|
+
"hidden_sessions": sum(1 for s in sessions if s.hidden),
|
|
63
|
+
"first_seen": _day(min(s.created_at for s in sessions)) if n else None,
|
|
64
|
+
"last_seen": _day(max(s.last_activity_at for s in sessions)) if n else None,
|
|
65
|
+
"messages": _sum(s.n_messages for s in sessions),
|
|
66
|
+
"tool_calls": _sum(s.n_tool_calls for s in sessions),
|
|
67
|
+
"duration_ms_total": _sum(s.duration_ms for s in sessions),
|
|
68
|
+
"cost_usd_total": _sum_or_none(u.cost_usd for u in snap.usage),
|
|
69
|
+
"cost_usd_by_sessions": _sum_or_none(s.cost_usd for s in sessions),
|
|
70
|
+
"input_tokens_total": _sum_or_none(u.input_tokens for u in snap.usage),
|
|
71
|
+
"output_tokens_total": _sum_or_none(u.output_tokens for u in snap.usage),
|
|
72
|
+
"context_tokens_peak": max(
|
|
73
|
+
(s.context_tokens for s in sessions
|
|
74
|
+
if s.context_tokens is not None),
|
|
75
|
+
default=None,
|
|
76
|
+
),
|
|
77
|
+
"sessions_with_context": sum(
|
|
78
|
+
1 for s in sessions if s.context_tokens is not None
|
|
79
|
+
),
|
|
80
|
+
"avg_duration_ms": (_sum(s.duration_ms for s in sessions) / n) if n else None,
|
|
81
|
+
"avg_messages": (_sum(s.n_messages for s in sessions) / n) if n else None,
|
|
82
|
+
"avg_cost_usd": (
|
|
83
|
+
_sum(s.cost_usd for s in with_cost) / len(with_cost) # type: ignore[arg-type]
|
|
84
|
+
if with_cost
|
|
85
|
+
else None
|
|
86
|
+
),
|
|
87
|
+
"acp_available": snap.acp_available,
|
|
88
|
+
"acp_db_files": snap.acp_db_files,
|
|
89
|
+
"acp_db_errors": snap.acp_db_errors,
|
|
90
|
+
"matched_sessions": snap.matched_sessions,
|
|
91
|
+
"orphan_dbs": snap.orphan_dbs,
|
|
92
|
+
"top_longest": top_sessions(snap, n=5, key="duration_ms"),
|
|
93
|
+
"top_costliest": top_sessions(snap, n=5, key="cost_usd"),
|
|
94
|
+
"models": by_model(snap),
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def by_project(snap: MetricsSnapshot) -> list[dict[str, Any]]:
|
|
99
|
+
groups: dict[str, list[SessionMetrics]] = {}
|
|
100
|
+
for s in snap.sessions:
|
|
101
|
+
groups.setdefault(s.working_directory, []).append(s)
|
|
102
|
+
rows = [
|
|
103
|
+
{
|
|
104
|
+
"project": wd,
|
|
105
|
+
"sessions": len(members),
|
|
106
|
+
"messages": _sum(s.n_messages for s in members),
|
|
107
|
+
"tool_calls": _sum(s.n_tool_calls for s in members),
|
|
108
|
+
"duration_ms": _sum(s.duration_ms for s in members),
|
|
109
|
+
"cost_usd": _sum_or_none(s.cost_usd for s in members),
|
|
110
|
+
"input_tokens": _sum_or_none(s.input_tokens for s in members),
|
|
111
|
+
"output_tokens": _sum_or_none(s.output_tokens for s in members),
|
|
112
|
+
}
|
|
113
|
+
for wd, members in groups.items()
|
|
114
|
+
]
|
|
115
|
+
rows.sort(
|
|
116
|
+
key=lambda r: (r["cost_usd"] is not None, r["cost_usd"] or 0, r["sessions"]),
|
|
117
|
+
reverse=True,
|
|
118
|
+
)
|
|
119
|
+
return rows
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def by_model(snap: MetricsSnapshot) -> list[dict[str, Any]]:
|
|
123
|
+
"""Per-model stats merging session-declared models and acp usage rows."""
|
|
124
|
+
rows: dict[str, dict[str, Any]] = {}
|
|
125
|
+
|
|
126
|
+
def _row(model: str | None) -> dict[str, Any]:
|
|
127
|
+
name = model or "(unknown)"
|
|
128
|
+
return rows.setdefault(
|
|
129
|
+
name,
|
|
130
|
+
{
|
|
131
|
+
"model": name,
|
|
132
|
+
"sessions": 0,
|
|
133
|
+
"usage_records": 0,
|
|
134
|
+
"cost_usd": None,
|
|
135
|
+
"input_tokens": None,
|
|
136
|
+
"output_tokens": None,
|
|
137
|
+
},
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
for s in snap.sessions:
|
|
141
|
+
_row(s.model)["sessions"] += 1
|
|
142
|
+
for u in snap.usage:
|
|
143
|
+
r = _row(u.model)
|
|
144
|
+
r["usage_records"] += 1
|
|
145
|
+
if u.cost_usd is not None:
|
|
146
|
+
r["cost_usd"] = (r["cost_usd"] or 0.0) + u.cost_usd
|
|
147
|
+
if u.input_tokens is not None:
|
|
148
|
+
r["input_tokens"] = (r["input_tokens"] or 0) + u.input_tokens
|
|
149
|
+
if u.output_tokens is not None:
|
|
150
|
+
r["output_tokens"] = (r["output_tokens"] or 0) + u.output_tokens
|
|
151
|
+
out = list(rows.values())
|
|
152
|
+
out.sort(
|
|
153
|
+
key=lambda r: (r["cost_usd"] is not None, r["cost_usd"] or 0, r["sessions"]),
|
|
154
|
+
reverse=True,
|
|
155
|
+
)
|
|
156
|
+
return out
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def by_day(snap: MetricsSnapshot, days: int | None = None) -> list[dict[str, Any]]:
|
|
160
|
+
"""Sessions bucketed by UTC creation date.
|
|
161
|
+
|
|
162
|
+
``days=N`` keeps the N most recent *activity* days (anchored on the latest
|
|
163
|
+
day in the data, not on wall-clock now — deterministic for fixtures and
|
|
164
|
+
for inspecting old installs alike).
|
|
165
|
+
"""
|
|
166
|
+
groups: dict[str, list[SessionMetrics]] = {}
|
|
167
|
+
for s in snap.sessions:
|
|
168
|
+
groups.setdefault(_day(s.created_at), []).append(s)
|
|
169
|
+
dates = sorted(groups)
|
|
170
|
+
if days is not None:
|
|
171
|
+
dates = dates[-days:] if days > 0 else []
|
|
172
|
+
return [
|
|
173
|
+
{
|
|
174
|
+
"date": d,
|
|
175
|
+
"sessions": len(groups[d]),
|
|
176
|
+
"messages": _sum(s.n_messages for s in groups[d]),
|
|
177
|
+
"tool_calls": _sum(s.n_tool_calls for s in groups[d]),
|
|
178
|
+
"duration_ms": _sum(s.duration_ms for s in groups[d]),
|
|
179
|
+
"cost_usd": _sum_or_none(s.cost_usd for s in groups[d]),
|
|
180
|
+
}
|
|
181
|
+
for d in dates
|
|
182
|
+
]
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def top_sessions(
|
|
186
|
+
snap: MetricsSnapshot, n: int = 5, key: str = "duration_ms"
|
|
187
|
+
) -> list[dict[str, Any]]:
|
|
188
|
+
"""Top-N sessions. ``key``: ``duration_ms`` or ``cost_usd``.
|
|
189
|
+
|
|
190
|
+
For ``cost_usd`` sessions with unknown cost are excluded entirely.
|
|
191
|
+
"""
|
|
192
|
+
if key == "cost_usd":
|
|
193
|
+
pool = [s for s in snap.sessions if s.cost_usd is not None]
|
|
194
|
+
ordered = sorted(pool, key=lambda s: s.cost_usd or 0, reverse=True)
|
|
195
|
+
elif key == "duration_ms":
|
|
196
|
+
ordered = sorted(snap.sessions, key=lambda s: s.duration_ms, reverse=True)
|
|
197
|
+
else:
|
|
198
|
+
raise ValueError(f"top_sessions: unknown key {key!r}")
|
|
199
|
+
return [session_to_dict(s) for s in ordered[:n]]
|
devin_metrics/churn.py
ADDED
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
"""Rework (churn) metrics over a built ``graph.db`` (ME-4).
|
|
2
|
+
|
|
3
|
+
Churn = rework: distinct tool calls in the same session that touched the
|
|
4
|
+
same file again. Read straight from the knowledge graph's
|
|
5
|
+
``file_touched`` edges (each edge is one tool_call → file, owned by a
|
|
6
|
+
session), so this needs a ``devin-graph build`` first — the graph is the
|
|
7
|
+
source of truth, this module never touches sessions.db.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import json
|
|
13
|
+
import sqlite3
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _connect_ro(path: Path) -> sqlite3.Connection:
|
|
19
|
+
con = sqlite3.connect(f"file:{path}?mode=ro", uri=True)
|
|
20
|
+
con.execute("PRAGMA query_only = ON")
|
|
21
|
+
return con
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def churn_report(graph_db: str | Path) -> dict[str, Any]:
|
|
25
|
+
"""Per-session and per-model rework stats from a graph.db.
|
|
26
|
+
|
|
27
|
+
Returns dict with ``sessions`` (one row per session that reworked at
|
|
28
|
+
least one file), ``models`` (aggregated by the session node's model
|
|
29
|
+
attr) and ``top_files`` (most-reworked paths overall).
|
|
30
|
+
"""
|
|
31
|
+
path = Path(graph_db).expanduser()
|
|
32
|
+
if not path.exists():
|
|
33
|
+
raise FileNotFoundError(
|
|
34
|
+
f"{path}: no graph — run 'devin-graph build --graph {path}'")
|
|
35
|
+
con = _connect_ro(path)
|
|
36
|
+
try:
|
|
37
|
+
_require_schema(con)
|
|
38
|
+
reworked = con.execute(
|
|
39
|
+
"SELECT e.owner, e.dst, COUNT(*) AS calls"
|
|
40
|
+
" FROM edges e WHERE e.kind = 'file_touched'"
|
|
41
|
+
" GROUP BY e.owner, e.dst HAVING calls > 1"
|
|
42
|
+
).fetchall()
|
|
43
|
+
touched = con.execute(
|
|
44
|
+
"SELECT owner, COUNT(DISTINCT dst) FROM edges"
|
|
45
|
+
" WHERE kind = 'file_touched' GROUP BY owner"
|
|
46
|
+
).fetchall()
|
|
47
|
+
models = {
|
|
48
|
+
r[0]: _model_of(r[1])
|
|
49
|
+
for r in con.execute(
|
|
50
|
+
"SELECT key, attrs FROM nodes WHERE kind = 'session'"
|
|
51
|
+
).fetchall()
|
|
52
|
+
}
|
|
53
|
+
paths = {
|
|
54
|
+
r[0]: r[1]
|
|
55
|
+
for r in con.execute(
|
|
56
|
+
"SELECT id, key FROM nodes WHERE kind = 'file'"
|
|
57
|
+
).fetchall()
|
|
58
|
+
}
|
|
59
|
+
finally:
|
|
60
|
+
con.close()
|
|
61
|
+
|
|
62
|
+
distinct_files = {owner: n for owner, n in touched}
|
|
63
|
+
per_session: dict[str, dict[str, Any]] = {}
|
|
64
|
+
file_calls: dict[str, int] = {}
|
|
65
|
+
for session_id, file_id, calls in reworked:
|
|
66
|
+
row = per_session.setdefault(session_id, {
|
|
67
|
+
"session_id": session_id,
|
|
68
|
+
"model": models.get(session_id),
|
|
69
|
+
"files_touched": distinct_files.get(session_id, 0),
|
|
70
|
+
"reworked_files": 0,
|
|
71
|
+
"repeat_calls": 0,
|
|
72
|
+
})
|
|
73
|
+
row["reworked_files"] += 1
|
|
74
|
+
row["repeat_calls"] += calls - 1
|
|
75
|
+
file_calls[file_id] = file_calls.get(file_id, 0) + calls - 1
|
|
76
|
+
|
|
77
|
+
sessions = sorted(
|
|
78
|
+
per_session.values(), key=lambda r: -r["repeat_calls"])
|
|
79
|
+
|
|
80
|
+
per_model: dict[str, dict[str, Any]] = {}
|
|
81
|
+
for row in sessions:
|
|
82
|
+
m = row["model"] or "(unknown)"
|
|
83
|
+
agg = per_model.setdefault(m, {
|
|
84
|
+
"model": m, "sessions_with_churn": 0,
|
|
85
|
+
"reworked_files": 0, "repeat_calls": 0})
|
|
86
|
+
agg["sessions_with_churn"] += 1
|
|
87
|
+
agg["reworked_files"] += row["reworked_files"]
|
|
88
|
+
agg["repeat_calls"] += row["repeat_calls"]
|
|
89
|
+
|
|
90
|
+
return {
|
|
91
|
+
"graph": str(path),
|
|
92
|
+
"sessions_with_churn": len(sessions),
|
|
93
|
+
"sessions": sessions,
|
|
94
|
+
"models": sorted(
|
|
95
|
+
per_model.values(), key=lambda r: -r["repeat_calls"]),
|
|
96
|
+
"top_files": [
|
|
97
|
+
{"file": paths.get(fid, fid), "repeat_calls": n}
|
|
98
|
+
for fid, n in sorted(
|
|
99
|
+
file_calls.items(), key=lambda kv: -kv[1])[:20]
|
|
100
|
+
],
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _require_schema(con: sqlite3.Connection) -> None:
|
|
105
|
+
tables = {r[0] for r in con.execute(
|
|
106
|
+
"SELECT name FROM sqlite_master WHERE type = 'table'")}
|
|
107
|
+
cols = {r[1] for r in con.execute("PRAGMA table_info(edges)")}
|
|
108
|
+
if not ({"nodes", "edges"} <= tables and {"kind", "src", "dst", "owner"}
|
|
109
|
+
<= cols):
|
|
110
|
+
raise ValueError(
|
|
111
|
+
f"{con}: not a devin-graph database — "
|
|
112
|
+
"expected nodes/edges(kind, src, dst, owner) from "
|
|
113
|
+
"'devin-graph build'")
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _model_of(attrs_json: str) -> str | None:
|
|
117
|
+
try:
|
|
118
|
+
attrs = json.loads(attrs_json)
|
|
119
|
+
except (TypeError, ValueError):
|
|
120
|
+
return None
|
|
121
|
+
m = attrs.get("model")
|
|
122
|
+
return m if isinstance(m, str) and m else None
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def render_churn(report: dict[str, Any]) -> str:
|
|
126
|
+
"""Human-readable churn table."""
|
|
127
|
+
lines = [
|
|
128
|
+
f"churn report — {report['graph']}",
|
|
129
|
+
f"sessions with rework: {report['sessions_with_churn']}",
|
|
130
|
+
"",
|
|
131
|
+
f"{'MODEL':<24} {'SESS':>5} {'REWORKED FILES':>15} {'REPEAT CALLS':>13}",
|
|
132
|
+
]
|
|
133
|
+
for m in report["models"]:
|
|
134
|
+
lines.append(
|
|
135
|
+
f"{m['model']:<24} {m['sessions_with_churn']:>5} "
|
|
136
|
+
f"{m['reworked_files']:>15} {m['repeat_calls']:>13}")
|
|
137
|
+
if report["top_files"]:
|
|
138
|
+
lines += ["", "most reworked files:"]
|
|
139
|
+
for f in report["top_files"][:10]:
|
|
140
|
+
lines.append(f" {f['repeat_calls']:>4}× {f['file']}")
|
|
141
|
+
return "\n".join(lines)
|
devin_metrics/cli.py
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
"""``devin-metrics`` — thin CLI wrapper; all logic lives in the library.
|
|
2
|
+
|
|
3
|
+
Subcommands (read-only, no network, all accept ``--json``):
|
|
4
|
+
|
|
5
|
+
- ``summary`` headline numbers + per-model table + top-5 longest sessions
|
|
6
|
+
- ``projects`` per-project (``working_directory``) session/activity table
|
|
7
|
+
- ``daily`` per-day activity; ``--days N`` keeps the N most recent days
|
|
8
|
+
|
|
9
|
+
Store locations default to the platform Devin data dir (``paths.py``);
|
|
10
|
+
``--data-dir`` overrides the root, ``--sessions-db``/``--acp-dir`` override
|
|
11
|
+
individual stores. A missing ``acp-messages`` dir degrades gracefully — a
|
|
12
|
+
warning on stderr and ``-`` in the cost columns. A missing ``sessions.db``
|
|
13
|
+
is fatal (exit 1): it is the only source of session rows.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import argparse
|
|
19
|
+
import json
|
|
20
|
+
import sqlite3
|
|
21
|
+
import sys
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Sequence
|
|
24
|
+
|
|
25
|
+
from devin_internals.schema import SchemaError
|
|
26
|
+
|
|
27
|
+
from devin_metrics.aggregate import by_day, by_project, summarize
|
|
28
|
+
from devin_metrics.churn import churn_report, render_churn
|
|
29
|
+
from devin_metrics.collect import MetricsSnapshot, collect
|
|
30
|
+
from devin_metrics.dashboard.collect import collect_stats
|
|
31
|
+
from devin_metrics.dashboard.render import render_html
|
|
32
|
+
from devin_metrics.paths import (
|
|
33
|
+
acp_messages_dir,
|
|
34
|
+
default_acp_messages_dir,
|
|
35
|
+
default_data_dir,
|
|
36
|
+
sessions_db_path,
|
|
37
|
+
)
|
|
38
|
+
from devin_metrics.render import (
|
|
39
|
+
dumps_json,
|
|
40
|
+
render_daily_md,
|
|
41
|
+
render_projects_md,
|
|
42
|
+
render_summary_md,
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _resolve(args: argparse.Namespace) -> tuple[Path, Path]:
|
|
47
|
+
root = Path(args.data_dir).expanduser() if args.data_dir else default_data_dir()
|
|
48
|
+
sessions_db = (
|
|
49
|
+
Path(args.sessions_db).expanduser()
|
|
50
|
+
if args.sessions_db
|
|
51
|
+
else sessions_db_path(root)
|
|
52
|
+
)
|
|
53
|
+
acp_dir = (
|
|
54
|
+
Path(args.acp_dir).expanduser()
|
|
55
|
+
if args.acp_dir is not None
|
|
56
|
+
else acp_messages_dir(root) if args.data_dir
|
|
57
|
+
else default_acp_messages_dir()
|
|
58
|
+
)
|
|
59
|
+
return sessions_db, acp_dir
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _snapshot(args: argparse.Namespace) -> MetricsSnapshot:
|
|
63
|
+
sessions_db, acp_dir = _resolve(args)
|
|
64
|
+
if not acp_dir.is_dir():
|
|
65
|
+
print(
|
|
66
|
+
f"warning: {acp_dir}: no such directory — "
|
|
67
|
+
"cost/token columns will show '-'",
|
|
68
|
+
file=sys.stderr,
|
|
69
|
+
)
|
|
70
|
+
return collect(sessions_db, acp_dir)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def cmd_summary(args: argparse.Namespace) -> int:
|
|
74
|
+
summary = summarize(_snapshot(args))
|
|
75
|
+
print(dumps_json(summary) if args.json else render_summary_md(summary))
|
|
76
|
+
return 0
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def cmd_projects(args: argparse.Namespace) -> int:
|
|
80
|
+
rows = by_project(_snapshot(args))
|
|
81
|
+
print(dumps_json(rows) if args.json else render_projects_md(rows))
|
|
82
|
+
return 0
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def cmd_daily(args: argparse.Namespace) -> int:
|
|
86
|
+
rows = by_day(_snapshot(args), days=args.days)
|
|
87
|
+
print(dumps_json(rows) if args.json else render_daily_md(rows))
|
|
88
|
+
return 0
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def cmd_watch(args: argparse.Namespace) -> int:
|
|
92
|
+
"""ME-2: advisory context guard — warns, never blocks.
|
|
93
|
+
|
|
94
|
+
Cost is unverifiable in local stores (ME-3), so the guard watches
|
|
95
|
+
``context_tokens`` (peak prompt size per session) — the only real
|
|
96
|
+
resource signal that persists. Thresholds are advisory: sessions that
|
|
97
|
+
crossed them are listed, exit stays 0 unless --fail.
|
|
98
|
+
"""
|
|
99
|
+
snap = _snapshot(args)
|
|
100
|
+
findings: list[dict] = []
|
|
101
|
+
by_ctx = sorted(
|
|
102
|
+
(s for s in snap.sessions if s.context_tokens is not None),
|
|
103
|
+
key=lambda s: s.context_tokens or 0, reverse=True)
|
|
104
|
+
over_session = [s for s in by_ctx if s.context_tokens > args.session_warn]
|
|
105
|
+
days: dict[str, int] = {}
|
|
106
|
+
for s in snap.sessions:
|
|
107
|
+
if s.context_tokens:
|
|
108
|
+
from devin_metrics.aggregate import _day
|
|
109
|
+
d = _day(s.created_at)
|
|
110
|
+
days[d] = days.get(d, 0) + s.context_tokens
|
|
111
|
+
over_days = {d: v for d, v in days.items() if v > args.daily_warn}
|
|
112
|
+
for s in over_session:
|
|
113
|
+
findings.append({"kind": "session", "session_id": s.id,
|
|
114
|
+
"context_tokens": s.context_tokens,
|
|
115
|
+
"threshold": args.session_warn})
|
|
116
|
+
for d, v in sorted(over_days.items()):
|
|
117
|
+
findings.append({"kind": "day", "date": d, "context_tokens": v,
|
|
118
|
+
"threshold": args.daily_warn})
|
|
119
|
+
report = {
|
|
120
|
+
"advisory": True,
|
|
121
|
+
"thresholds": {"session_warn": args.session_warn,
|
|
122
|
+
"daily_warn": args.daily_warn},
|
|
123
|
+
"sessions_observed": len(by_ctx),
|
|
124
|
+
"findings": findings,
|
|
125
|
+
"note": "advisory only — context_tokens is peak prompt size, "
|
|
126
|
+
"not cost (local stores have no cost data, verified ME-3)",
|
|
127
|
+
}
|
|
128
|
+
if args.json:
|
|
129
|
+
print(json.dumps(report, indent=2))
|
|
130
|
+
else:
|
|
131
|
+
print(f"watch (advisory): {len(by_ctx)} sessions observed · "
|
|
132
|
+
f"{len(findings)} finding(s)")
|
|
133
|
+
for f in findings:
|
|
134
|
+
if f["kind"] == "session":
|
|
135
|
+
print(f" session {f['session_id'][:16]}… "
|
|
136
|
+
f"{f['context_tokens']:,} tok > {f['threshold']:,}")
|
|
137
|
+
else:
|
|
138
|
+
print(f" day {f['date']} {f['context_tokens']:,} tok "
|
|
139
|
+
f"> {f['threshold']:,}")
|
|
140
|
+
if not findings:
|
|
141
|
+
print(" all clear")
|
|
142
|
+
return 1 if (findings and args.fail) else 0
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def cmd_churn(args: argparse.Namespace) -> int:
|
|
146
|
+
report = churn_report(args.graph)
|
|
147
|
+
print(json.dumps(report, indent=2) if args.json
|
|
148
|
+
else render_churn(report))
|
|
149
|
+
return 0
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def cmd_dashboard(args: argparse.Namespace) -> int:
|
|
153
|
+
sessions_db, acp_dir = _resolve(args)
|
|
154
|
+
if not acp_dir.is_dir():
|
|
155
|
+
print(
|
|
156
|
+
f"warning: {acp_dir}: no such directory — "
|
|
157
|
+
"cost/token fields will be empty",
|
|
158
|
+
file=sys.stderr,
|
|
159
|
+
)
|
|
160
|
+
stats = collect_stats(sessions_db, acp_dir)
|
|
161
|
+
if args.json:
|
|
162
|
+
import json
|
|
163
|
+
|
|
164
|
+
print(json.dumps(stats, indent=2))
|
|
165
|
+
return 0
|
|
166
|
+
out = Path(args.out).expanduser()
|
|
167
|
+
out.write_text(render_html(stats), encoding="utf-8")
|
|
168
|
+
print(f"wrote {out} ({out.stat().st_size:,} bytes) — open it in a browser")
|
|
169
|
+
return 0
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
173
|
+
parser = argparse.ArgumentParser(
|
|
174
|
+
prog="devin-metrics",
|
|
175
|
+
description="Local-only metrics from Devin's session stores.",
|
|
176
|
+
)
|
|
177
|
+
common = argparse.ArgumentParser(add_help=False)
|
|
178
|
+
common.add_argument(
|
|
179
|
+
"--data-dir", metavar="DIR",
|
|
180
|
+
help="Devin data dir (default: platform-specific, see docs/SPEC.md)",
|
|
181
|
+
)
|
|
182
|
+
common.add_argument(
|
|
183
|
+
"--sessions-db", metavar="DB",
|
|
184
|
+
help="path to sessions.db (default: <data-dir>/cli/sessions.db)",
|
|
185
|
+
)
|
|
186
|
+
common.add_argument(
|
|
187
|
+
"--acp-dir", metavar="DIR",
|
|
188
|
+
help="path to acp-messages dir (default: <data-dir>/User/acp-messages)",
|
|
189
|
+
)
|
|
190
|
+
common.add_argument("--json", action="store_true", help="JSON output")
|
|
191
|
+
|
|
192
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
193
|
+
sub.add_parser("summary", parents=[common],
|
|
194
|
+
help="headline numbers").set_defaults(func=cmd_summary)
|
|
195
|
+
sub.add_parser("projects", parents=[common],
|
|
196
|
+
help="per-project session/activity table").set_defaults(func=cmd_projects)
|
|
197
|
+
p_daily = sub.add_parser("daily", parents=[common],
|
|
198
|
+
help="activity over time")
|
|
199
|
+
p_daily.add_argument(
|
|
200
|
+
"--days", type=int, default=None, metavar="N",
|
|
201
|
+
help="keep the N most recent activity days (default: all)",
|
|
202
|
+
)
|
|
203
|
+
p_daily.set_defaults(func=cmd_daily)
|
|
204
|
+
p_dash = sub.add_parser("dashboard", parents=[common],
|
|
205
|
+
help="render the static HTML dashboard")
|
|
206
|
+
p_dash.add_argument(
|
|
207
|
+
"--out", metavar="FILE", default="index.html",
|
|
208
|
+
help="output file (default: index.html)",
|
|
209
|
+
)
|
|
210
|
+
p_dash.set_defaults(func=cmd_dashboard)
|
|
211
|
+
p_churn = sub.add_parser(
|
|
212
|
+
"churn", help="rework stats from a graph.db (devin-graph build first)")
|
|
213
|
+
p_churn.add_argument("--graph", metavar="DB", default="graph.db",
|
|
214
|
+
help="path to a built graph.db (default ./graph.db)")
|
|
215
|
+
p_churn.add_argument("--json", action="store_true")
|
|
216
|
+
p_churn.set_defaults(func=cmd_churn)
|
|
217
|
+
p_watch = sub.add_parser(
|
|
218
|
+
"watch", parents=[common],
|
|
219
|
+
help="advisory context guard — lists sessions/days over a token "
|
|
220
|
+
"threshold; never blocks (ME-2)")
|
|
221
|
+
p_watch.add_argument("--session-warn", type=int, default=400_000,
|
|
222
|
+
metavar="N", help="flag sessions with peak context "
|
|
223
|
+
"> N tokens (default: %(default)s)")
|
|
224
|
+
p_watch.add_argument("--daily-warn", type=int, default=2_000_000,
|
|
225
|
+
metavar="N", help="flag days whose summed context "
|
|
226
|
+
"> N tokens (default: %(default)s)")
|
|
227
|
+
p_watch.add_argument("--fail", action="store_true",
|
|
228
|
+
help="exit 1 when findings exist (CI mode)")
|
|
229
|
+
p_watch.set_defaults(func=cmd_watch)
|
|
230
|
+
return parser
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
234
|
+
args = build_parser().parse_args(argv)
|
|
235
|
+
try:
|
|
236
|
+
return args.func(args)
|
|
237
|
+
except SchemaError as exc:
|
|
238
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
239
|
+
return 1
|
|
240
|
+
except (sqlite3.Error, OSError) as exc:
|
|
241
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
242
|
+
return 1
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
if __name__ == "__main__":
|
|
246
|
+
raise SystemExit(main())
|