kostria 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kostria/__init__.py +3 -0
- kostria/analysis.py +312 -0
- kostria/attribution.py +359 -0
- kostria/badge.py +56 -0
- kostria/cache.py +112 -0
- kostria/cli.py +718 -0
- kostria/collectors/__init__.py +62 -0
- kostria/collectors/claude_code.py +233 -0
- kostria/collectors/codex.py +82 -0
- kostria/collectors/opencode.py +88 -0
- kostria/compare.py +89 -0
- kostria/config.py +214 -0
- kostria/doctor.py +209 -0
- kostria/landing/index.html +648 -0
- kostria/landing/vendor/ScrollTrigger.min.js +11 -0
- kostria/landing/vendor/fonts/Inter.woff2 +0 -0
- kostria/landing/vendor/fonts/SourceSerif4.woff2 +0 -0
- kostria/landing/vendor/gsap.min.js +11 -0
- kostria/landing/vendor/three.core.js +60007 -0
- kostria/landing/vendor/three.module.js +19610 -0
- kostria/login.py +131 -0
- kostria/markdown.py +146 -0
- kostria/period.py +103 -0
- kostria/pricing.py +313 -0
- kostria/render.py +746 -0
- kostria/report.py +372 -0
- kostria/sync.py +226 -0
- kostria/telemetry.py +101 -0
- kostria/watch.py +73 -0
- kostria/web/dashboard.html +1496 -0
- kostria/web/vendor/fonts/Inter.woff2 +0 -0
- kostria/web/vendor/fonts/SourceSerif4.woff2 +0 -0
- kostria/web/vendor/gsap.min.js +11 -0
- kostria-0.2.0.dist-info/METADATA +198 -0
- kostria-0.2.0.dist-info/RECORD +39 -0
- kostria-0.2.0.dist-info/WHEEL +5 -0
- kostria-0.2.0.dist-info/entry_points.txt +2 -0
- kostria-0.2.0.dist-info/licenses/LICENSE +21 -0
- kostria-0.2.0.dist-info/top_level.txt +1 -0
kostria/__init__.py
ADDED
kostria/analysis.py
ADDED
|
@@ -0,0 +1,312 @@
|
|
|
1
|
+
"""Sinais acionaveis: o que fazer com o numero depois de apura-lo.
|
|
2
|
+
|
|
3
|
+
Um total de gasto nao muda decisao nenhuma. Estes sinais mudam:
|
|
4
|
+
|
|
5
|
+
- **reprocessamento**: quanto do custo e releitura de contexto em vez de
|
|
6
|
+
producao de codigo. Cache read barato em unidade, caro em volume.
|
|
7
|
+
- **retrabalho**: arquivos reescritos muitas vezes na mesma sessao, sinal de
|
|
8
|
+
tentativa e erro cara.
|
|
9
|
+
- **eficiencia por modelo**: custo por edicao entregue, que e o que permite
|
|
10
|
+
comparar modelo caro com modelo barato de forma honesta.
|
|
11
|
+
- **exploracao**: sessoes que nao produziram edicao nenhuma.
|
|
12
|
+
- **contas ociosas**: conta paga sem uso recente.
|
|
13
|
+
"""
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from collections import Counter, defaultdict
|
|
17
|
+
from datetime import datetime, timedelta, timezone
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _dt(iso):
|
|
21
|
+
if not iso:
|
|
22
|
+
return None
|
|
23
|
+
try:
|
|
24
|
+
return datetime.fromisoformat(str(iso).replace("Z", "+00:00"))
|
|
25
|
+
except ValueError:
|
|
26
|
+
return None
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def token_mix(sessions):
|
|
30
|
+
"""Composicao do consumo. Precisa dos models por sessao, entao aproxima
|
|
31
|
+
pelo agregado de tokens quando o detalhe nao existe."""
|
|
32
|
+
mix = Counter()
|
|
33
|
+
for s in sessions:
|
|
34
|
+
for kind, n in (s.token_mix or {}).items():
|
|
35
|
+
mix[kind] += n
|
|
36
|
+
total = sum(mix.values()) or 1
|
|
37
|
+
return {k: {"tokens": v, "share": v / total} for k, v in mix.most_common()}
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def rework(sessions, resolver, top=12):
|
|
41
|
+
"""Arquivos reescritos varias vezes dentro da mesma sessao."""
|
|
42
|
+
counts: dict[str, int] = defaultdict(int)
|
|
43
|
+
sess_hits: dict[str, int] = defaultdict(int)
|
|
44
|
+
cost_of: dict[str, float] = defaultdict(float)
|
|
45
|
+
|
|
46
|
+
for s in sessions:
|
|
47
|
+
if not s.edits:
|
|
48
|
+
continue
|
|
49
|
+
per = Counter(e.path for e in s.edits)
|
|
50
|
+
share = s.usd / len(s.edits) if s.edits else 0.0
|
|
51
|
+
for path, n in per.items():
|
|
52
|
+
if n >= 3: # 1-2 escritas e trabalho normal
|
|
53
|
+
counts[path] += n
|
|
54
|
+
sess_hits[path] += 1
|
|
55
|
+
cost_of[path] += share * n
|
|
56
|
+
out = []
|
|
57
|
+
for path, n in sorted(counts.items(), key=lambda kv: -cost_of[kv[0]])[:top]:
|
|
58
|
+
out.append({"path": path, "writes": n, "sessions": sess_hits[path],
|
|
59
|
+
"cost": cost_of[path],
|
|
60
|
+
"project": resolver.of_path(path)})
|
|
61
|
+
return out
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def model_efficiency(sessions):
|
|
65
|
+
"""Custo por edicao entregue, por modelo.
|
|
66
|
+
|
|
67
|
+
Comparar modelos por preco de token e enganoso: o modelo caro pode
|
|
68
|
+
precisar de menos tentativas. O que importa e o custo do resultado.
|
|
69
|
+
"""
|
|
70
|
+
cost: dict[str, float] = defaultdict(float)
|
|
71
|
+
edits: dict[str, int] = defaultdict(int)
|
|
72
|
+
tokens: dict[str, int] = defaultdict(int)
|
|
73
|
+
|
|
74
|
+
for s in sessions:
|
|
75
|
+
if not s.models:
|
|
76
|
+
continue
|
|
77
|
+
total = sum(c for c, _t in s.models.values()) or 0.0
|
|
78
|
+
for model, (c, t) in s.models.items():
|
|
79
|
+
cost[model] += c
|
|
80
|
+
tokens[model] += t
|
|
81
|
+
if total > 0 and s.edits:
|
|
82
|
+
edits[model] += round(len(s.edits) * (c / total))
|
|
83
|
+
|
|
84
|
+
out = []
|
|
85
|
+
for m in cost:
|
|
86
|
+
e = edits[m]
|
|
87
|
+
out.append({"model": m, "cost": cost[m], "tokens": tokens[m],
|
|
88
|
+
"edits": e,
|
|
89
|
+
"cost_per_edit": (cost[m] / e) if e else None})
|
|
90
|
+
return sorted(out, key=lambda x: -x["cost"])
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def exploration_share(sessions):
|
|
94
|
+
"""Custo em sessoes que nao editaram arquivo nenhum."""
|
|
95
|
+
explo = sum(s.usd for s in sessions if not s.edits)
|
|
96
|
+
total = sum(s.usd for s in sessions) or 1
|
|
97
|
+
return {"cost": explo, "share": explo / total}
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def idle_accounts(sessions, days=14):
|
|
101
|
+
"""Contas sem atividade recente: candidatas a revisao de assento."""
|
|
102
|
+
last: dict[str, datetime] = {}
|
|
103
|
+
cost: dict[str, float] = defaultdict(float)
|
|
104
|
+
for s in sessions:
|
|
105
|
+
cost[s.account] += s.usd
|
|
106
|
+
d = _dt(s.started_at)
|
|
107
|
+
if d and (s.account not in last or d > last[s.account]):
|
|
108
|
+
last[s.account] = d
|
|
109
|
+
now = datetime.now(timezone.utc)
|
|
110
|
+
out = []
|
|
111
|
+
for acct, c in cost.items():
|
|
112
|
+
d = last.get(acct)
|
|
113
|
+
idle = (now - d).days if d else None
|
|
114
|
+
out.append({"account": acct, "cost": c,
|
|
115
|
+
"last_seen": d.isoformat(timespec="seconds") if d else None,
|
|
116
|
+
"idle_days": idle,
|
|
117
|
+
"idle": bool(idle is not None and idle >= days)})
|
|
118
|
+
return sorted(out, key=lambda x: -x["cost"])
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def daily_series(sessions):
|
|
122
|
+
"""Custo por dia, para tendencia e projecao."""
|
|
123
|
+
by_day: dict[str, float] = defaultdict(float)
|
|
124
|
+
for s in sessions:
|
|
125
|
+
d = _dt(s.started_at)
|
|
126
|
+
if d:
|
|
127
|
+
by_day[d.date().isoformat()] += s.usd
|
|
128
|
+
return dict(sorted(by_day.items()))
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def projection(sessions, window=7):
|
|
132
|
+
"""Projecao mensal a partir da media dos ultimos dias com atividade."""
|
|
133
|
+
series = daily_series(sessions)
|
|
134
|
+
if not series:
|
|
135
|
+
return None
|
|
136
|
+
days = sorted(series)[-window:]
|
|
137
|
+
if not days:
|
|
138
|
+
return None
|
|
139
|
+
avg = sum(series[d] for d in days) / len(days)
|
|
140
|
+
return {"daily_avg": avg, "monthly": avg * 30, "window_days": len(days),
|
|
141
|
+
"from": days[0], "to": days[-1]}
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def anomalies(sessions, sigma: float = 2.0, min_days: int = 7):
|
|
145
|
+
"""Dias cujo custo foge do padrao do proprio periodo.
|
|
146
|
+
|
|
147
|
+
Media simples nao serve de alarme: um dia 4x acima da media pode ser
|
|
148
|
+
normal numa serie muito irregular. O corte usa desvio padrao da propria
|
|
149
|
+
serie, e so opina quando ha dias suficientes para o desvio significar algo.
|
|
150
|
+
"""
|
|
151
|
+
serie = daily_series(sessions)
|
|
152
|
+
if len(serie) < min_days:
|
|
153
|
+
return {"enough_data": False, "days": len(serie), "items": []}
|
|
154
|
+
|
|
155
|
+
valores = list(serie.values())
|
|
156
|
+
media = sum(valores) / len(valores)
|
|
157
|
+
var = sum((v - media) ** 2 for v in valores) / len(valores)
|
|
158
|
+
desvio = var ** 0.5
|
|
159
|
+
corte = media + sigma * desvio
|
|
160
|
+
|
|
161
|
+
itens = [{"day": d, "cost": v, "times_avg": (v / media) if media else None,
|
|
162
|
+
"excess": v - media}
|
|
163
|
+
for d, v in serie.items() if v > corte]
|
|
164
|
+
itens.sort(key=lambda x: -x["cost"])
|
|
165
|
+
return {"enough_data": True, "days": len(serie), "avg": media,
|
|
166
|
+
"stdev": desvio, "threshold": corte, "sigma": sigma,
|
|
167
|
+
"items": itens}
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def top_sessions(sessions, top: int = 10, resolver=None):
|
|
171
|
+
"""Sessoes mais caras, uma por uma.
|
|
172
|
+
|
|
173
|
+
O agregado por projeto esconde o caso individual: uma sessao que entrou em
|
|
174
|
+
loop pode responder sozinha por uma fatia grande do mes, e e nela que se
|
|
175
|
+
age. Traz o projeto, o volume e o que a sessao produziu, para separar
|
|
176
|
+
trabalho pesado de desperdicio.
|
|
177
|
+
"""
|
|
178
|
+
total = sum(s.usd for s in sessions) or 1
|
|
179
|
+
out = []
|
|
180
|
+
for s in sessions:
|
|
181
|
+
if not s.usd:
|
|
182
|
+
continue
|
|
183
|
+
projeto = None
|
|
184
|
+
if resolver is not None:
|
|
185
|
+
alvo = (s.edits or [None])[0] or (getattr(s, "reads", None) or [None])[0]
|
|
186
|
+
projeto = resolver.of_path(alvo.path if alvo else s.cwd)
|
|
187
|
+
mix = s.token_mix or {}
|
|
188
|
+
reread = mix.get("cache_read", 0)
|
|
189
|
+
out.append({
|
|
190
|
+
"cost": s.usd, "tokens": s.tokens, "share": s.usd / total,
|
|
191
|
+
"project": projeto, "tool": s.tool, "account": s.account,
|
|
192
|
+
"started_at": s.started_at,
|
|
193
|
+
"edits": len(s.edits or []), "reads": len(getattr(s, "reads", None) or []),
|
|
194
|
+
"reread_share": (reread / s.tokens) if s.tokens else None,
|
|
195
|
+
"source": s.source,
|
|
196
|
+
})
|
|
197
|
+
out.sort(key=lambda r: -r["cost"])
|
|
198
|
+
return out[:top]
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def end_of_month_forecast(sessions):
|
|
202
|
+
"""Previsao de fechamento do mes corrente, nao um generico "por mes".
|
|
203
|
+
|
|
204
|
+
A projecao existente (media movel x30) responde "nesse ritmo, quanto por
|
|
205
|
+
mes". Esta responde a pergunta que fecha orcamento de verdade: "quanto eu
|
|
206
|
+
vou ter gasto no dia 1 do mes que vem". Soma o que ja foi gasto desde o
|
|
207
|
+
dia 1 do mes atual com a media recente extrapolada pelos dias que faltam
|
|
208
|
+
no CALENDARIO, nao numa janela movel generica.
|
|
209
|
+
"""
|
|
210
|
+
import calendar
|
|
211
|
+
|
|
212
|
+
serie = daily_series(sessions)
|
|
213
|
+
if not serie:
|
|
214
|
+
return None
|
|
215
|
+
hoje = datetime.now(timezone.utc).date()
|
|
216
|
+
inicio_mes = hoje.replace(day=1)
|
|
217
|
+
dias_no_mes = calendar.monthrange(hoje.year, hoje.month)[1]
|
|
218
|
+
dias_passados = (hoje - inicio_mes).days + 1
|
|
219
|
+
dias_restantes = dias_no_mes - dias_passados
|
|
220
|
+
|
|
221
|
+
ini_iso, hoje_iso = inicio_mes.isoformat(), hoje.isoformat()
|
|
222
|
+
mtd = sum(v for d, v in serie.items() if ini_iso <= d <= hoje_iso)
|
|
223
|
+
|
|
224
|
+
proj = projection(sessions)
|
|
225
|
+
media_recente = proj["daily_avg"] if proj else 0.0
|
|
226
|
+
|
|
227
|
+
return {
|
|
228
|
+
"month": ini_iso[:7], "today": hoje_iso, "month_to_date": mtd,
|
|
229
|
+
"days_elapsed": dias_passados, "days_remaining": dias_restantes,
|
|
230
|
+
"days_in_month": dias_no_mes, "recent_daily_avg": media_recente,
|
|
231
|
+
"forecast_total": mtd + media_recente * dias_restantes,
|
|
232
|
+
}
|
|
233
|
+
|
|
234
|
+
|
|
235
|
+
def _cost_tokens_by_model(sessions):
|
|
236
|
+
cost: dict[str, float] = defaultdict(float)
|
|
237
|
+
tokens: dict[str, int] = defaultdict(int)
|
|
238
|
+
for s in sessions:
|
|
239
|
+
for model, (c, t) in (s.models or {}).items():
|
|
240
|
+
cost[model] += c
|
|
241
|
+
tokens[model] += t
|
|
242
|
+
return cost, tokens
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _downgrade_estimate(sessions, alternative_of, alt_price_of, min_cost):
|
|
246
|
+
"""Nucleo comum de model_downgrade_opportunities/opencode_downgrade_
|
|
247
|
+
opportunities: agrega custo/tokens por modelo, acha a alternativa via
|
|
248
|
+
`alternative_of(model)` e escala o custo pela razao de preco de input.
|
|
249
|
+
|
|
250
|
+
Aproximacao: `Session.models` so guarda custo e tokens TOTAIS por
|
|
251
|
+
modelo, sem quebra input/output/cache — nao da pra recalcular o custo
|
|
252
|
+
exato com a tabela da alternativa. Em vez disso escala o custo real
|
|
253
|
+
pela RAZAO entre os precos de input dos dois modelos, assumindo que a
|
|
254
|
+
mistura de tipo de token e parecida entre ambos (razoavel: e o mesmo
|
|
255
|
+
tipo de sessao). Numero e ESTIMATIVA DE PRECO, nunca diz se o resultado
|
|
256
|
+
teria a mesma qualidade.
|
|
257
|
+
"""
|
|
258
|
+
from . import pricing
|
|
259
|
+
|
|
260
|
+
cost, tokens = _cost_tokens_by_model(sessions)
|
|
261
|
+
out = []
|
|
262
|
+
for model, c in cost.items():
|
|
263
|
+
if c < min_cost:
|
|
264
|
+
continue
|
|
265
|
+
alt = alternative_of(model)
|
|
266
|
+
if not alt:
|
|
267
|
+
continue
|
|
268
|
+
preco_atual, preco_alt = pricing.lookup_any(model), alt_price_of(alt)
|
|
269
|
+
if not preco_atual or not preco_alt:
|
|
270
|
+
continue
|
|
271
|
+
fator = preco_alt[0] / preco_atual[0]
|
|
272
|
+
estimado = c * fator
|
|
273
|
+
out.append({
|
|
274
|
+
"model": model, "alternative": alt,
|
|
275
|
+
"cost": c, "tokens": tokens[model],
|
|
276
|
+
"estimated_cost_with_alternative": estimado,
|
|
277
|
+
"estimated_savings": c - estimado,
|
|
278
|
+
"savings_pct": 1 - fator,
|
|
279
|
+
})
|
|
280
|
+
return sorted(out, key=lambda x: -x["estimated_savings"])
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def model_downgrade_opportunities(sessions, min_cost: float = 1.0):
|
|
284
|
+
"""Economia estimada trocando cada modelo usado pelo mais barato do
|
|
285
|
+
MESMO fornecedor (pricing.cheaper_alternative) — ver _downgrade_estimate
|
|
286
|
+
para a metodologia de estimativa."""
|
|
287
|
+
from . import pricing
|
|
288
|
+
return _downgrade_estimate(
|
|
289
|
+
sessions, pricing.cheaper_alternative, pricing.lookup_any, min_cost)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def opencode_downgrade_opportunities(sessions, min_cost: float = 1.0):
|
|
293
|
+
"""Economia estimada trocando cada modelo usado por um modelo ABERTO
|
|
294
|
+
do catalogo pago OpenCode Zen (pricing.opencode_alternative) —
|
|
295
|
+
CROSS-fornecedor, ver _downgrade_estimate para a metodologia."""
|
|
296
|
+
from . import pricing
|
|
297
|
+
return _downgrade_estimate(
|
|
298
|
+
sessions, pricing.opencode_alternative,
|
|
299
|
+
lambda m: pricing.OPENCODE_ZEN.get(m), min_cost)
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def weekday_profile(sessions):
|
|
303
|
+
"""Custo por dia da semana: revela se o gasto vaza no fim de semana."""
|
|
304
|
+
nomes = ["segunda", "terca", "quarta", "quinta", "sexta", "sabado", "domingo"]
|
|
305
|
+
acc = defaultdict(float)
|
|
306
|
+
for s in sessions:
|
|
307
|
+
d = _dt(s.started_at)
|
|
308
|
+
if d:
|
|
309
|
+
acc[nomes[d.weekday()]] += s.usd
|
|
310
|
+
total = sum(acc.values()) or 1
|
|
311
|
+
return {n: {"cost": acc.get(n, 0.0), "share": acc.get(n, 0.0) / total}
|
|
312
|
+
for n in nomes}
|
kostria/attribution.py
ADDED
|
@@ -0,0 +1,359 @@
|
|
|
1
|
+
"""Atribuicao: de onde o dinheiro saiu, de verdade.
|
|
2
|
+
|
|
3
|
+
Tres descobertas sustentam este modulo:
|
|
4
|
+
|
|
5
|
+
1. **O diretorio da sessao mente.** O agente edita varios repositorios a partir
|
|
6
|
+
de um cwd so (working dirs adicionais, worktrees, pastas temporarias). Medir
|
|
7
|
+
por cwd joga o custo no projeto errado. A atribuicao segue os ARQUIVOS
|
|
8
|
+
efetivamente editados.
|
|
9
|
+
2. **Worktree nao e outro projeto.** `git rev-parse --git-common-dir` aponta o
|
|
10
|
+
repositorio principal, o que resolve worktrees sem depender de convencao de
|
|
11
|
+
nome.
|
|
12
|
+
3. **O PR e uma unidade grosseira.** Um PR de consolidacao carrega varias
|
|
13
|
+
frentes e uma feature grande atravessa varios PRs. A unidade fina e o
|
|
14
|
+
commit, classificado por area (escopo do commit) e por item do board.
|
|
15
|
+
"""
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import os
|
|
19
|
+
import re
|
|
20
|
+
import subprocess
|
|
21
|
+
from bisect import bisect_left
|
|
22
|
+
from collections import defaultdict
|
|
23
|
+
from datetime import datetime
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
|
|
26
|
+
TICKET_RE = re.compile(r"\b[A-Z][A-Z0-9]+-\d+\b")
|
|
27
|
+
SCOPE_RE = re.compile(
|
|
28
|
+
r"^(?:feat|fix|perf|refactor|chore|docs|test|build|ci|style)\(([^)]+)\)")
|
|
29
|
+
RS, US = "\x1e", "\x1f"
|
|
30
|
+
|
|
31
|
+
# pastas que nao representam um projeto de verdade
|
|
32
|
+
SCRATCH = {"scratchpad", "tmp", "temp", "out", "dist", "build", "node_modules",
|
|
33
|
+
"__pycache__", ".cache", "bench", "retro"}
|
|
34
|
+
TMPNAME_RE = re.compile(r"w[a-z]*[A-Z0-9]?\.[A-Za-z0-9]{5,8}")
|
|
35
|
+
UUID_RE = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", re.I)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _git(args, cwd=None):
|
|
39
|
+
try:
|
|
40
|
+
r = subprocess.run(["git", *args], cwd=cwd, capture_output=True,
|
|
41
|
+
text=True, errors="ignore", timeout=30)
|
|
42
|
+
return r.stdout.strip() if r.returncode == 0 else None
|
|
43
|
+
except (OSError, subprocess.SubprocessError):
|
|
44
|
+
return None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class ProjectResolver:
|
|
48
|
+
"""Mapeia um caminho para o nome do projeto, com cache."""
|
|
49
|
+
|
|
50
|
+
def __init__(self, workspaces=None):
|
|
51
|
+
self._cache: dict[str, str] = {}
|
|
52
|
+
self.workspaces = [str(w) for w in (workspaces or
|
|
53
|
+
[Path.home() / "Documentos" / "VS"])]
|
|
54
|
+
self.notes: dict[str, str] = {}
|
|
55
|
+
self._known_cache = None
|
|
56
|
+
self._worktrees = self._index_worktrees()
|
|
57
|
+
|
|
58
|
+
def _index_worktrees(self) -> dict[str, str]:
|
|
59
|
+
"""prefixo de worktree -> repo principal.
|
|
60
|
+
|
|
61
|
+
Worktree apagado nao responde a `git rev-parse`, entao o indice e
|
|
62
|
+
montado a partir dos repos vivos e serve tambem para caminhos mortos.
|
|
63
|
+
"""
|
|
64
|
+
idx: dict[str, str] = {}
|
|
65
|
+
for ws in self.workspaces:
|
|
66
|
+
base = Path(ws)
|
|
67
|
+
if not base.is_dir():
|
|
68
|
+
continue
|
|
69
|
+
for repo in base.iterdir():
|
|
70
|
+
if not (repo / ".git").exists():
|
|
71
|
+
continue
|
|
72
|
+
out = _git(["worktree", "list", "--porcelain"], cwd=repo)
|
|
73
|
+
for line in (out or "").splitlines():
|
|
74
|
+
if line.startswith("worktree "):
|
|
75
|
+
wt = line.split(" ", 1)[1].strip()
|
|
76
|
+
if Path(wt).resolve() != repo.resolve():
|
|
77
|
+
idx[wt] = repo.name
|
|
78
|
+
# convencao comum: <repo>-wt/<branch> guarda os worktrees
|
|
79
|
+
sibling = base / f"{repo.name}-wt"
|
|
80
|
+
if sibling.is_dir():
|
|
81
|
+
idx[str(sibling)] = repo.name
|
|
82
|
+
return idx
|
|
83
|
+
|
|
84
|
+
def _encoded_origin(self, d: str) -> str | None:
|
|
85
|
+
"""Projeto codificado num segmento do caminho (`-home-user-code-app`).
|
|
86
|
+
|
|
87
|
+
O scratch por sessao guarda o cwd de origem no proprio nome da pasta.
|
|
88
|
+
Sem isso, todo esse custo cai num balde generico e some do projeto que
|
|
89
|
+
de fato o gerou.
|
|
90
|
+
|
|
91
|
+
A codificacao troca TODA barra por hifen, o que e ambiguo: o segmento
|
|
92
|
+
`-home-user-code-obsidian-vault` tanto pode ser `.../obsidian/vault`
|
|
93
|
+
quanto `.../obsidian-vault`. Decodificar ao contrario chuta errado,
|
|
94
|
+
entao o casamento e feito contra os nomes de projeto que existem de
|
|
95
|
+
verdade no workspace, do mais longo para o mais curto.
|
|
96
|
+
"""
|
|
97
|
+
for seg in d.split("/"):
|
|
98
|
+
if not seg.startswith("-") or len(seg) < 8:
|
|
99
|
+
continue
|
|
100
|
+
for name in self._known_projects:
|
|
101
|
+
if seg.endswith("-" + name.replace("/", "-")):
|
|
102
|
+
return name
|
|
103
|
+
return None
|
|
104
|
+
|
|
105
|
+
@property
|
|
106
|
+
def _known_projects(self):
|
|
107
|
+
if self._known_cache is None:
|
|
108
|
+
names = []
|
|
109
|
+
for ws in self.workspaces:
|
|
110
|
+
base = Path(ws)
|
|
111
|
+
if base.is_dir():
|
|
112
|
+
names += [p.name for p in base.iterdir() if p.is_dir()]
|
|
113
|
+
# mais longo primeiro: 'obsidian-vault' antes de 'vault'
|
|
114
|
+
self._known_cache = sorted(set(names), key=len, reverse=True)
|
|
115
|
+
return self._known_cache
|
|
116
|
+
|
|
117
|
+
def _from_worktree_index(self, d: str) -> str | None:
|
|
118
|
+
for wt, repo in self._worktrees.items():
|
|
119
|
+
if d == wt or d.startswith(wt.rstrip("/") + "/"):
|
|
120
|
+
return repo
|
|
121
|
+
return None
|
|
122
|
+
|
|
123
|
+
def of_path(self, path: str | None) -> str | None:
|
|
124
|
+
if not path:
|
|
125
|
+
return None
|
|
126
|
+
d = path if os.path.isdir(path) else os.path.dirname(path)
|
|
127
|
+
if not d:
|
|
128
|
+
return None
|
|
129
|
+
if d in self._cache:
|
|
130
|
+
return self._cache[d]
|
|
131
|
+
name = self._resolve(d)
|
|
132
|
+
self._cache[d] = name
|
|
133
|
+
return name
|
|
134
|
+
|
|
135
|
+
def _resolve(self, d: str) -> str:
|
|
136
|
+
# worktree (vivo ou ja apagado) pertence ao repo principal
|
|
137
|
+
wt = self._from_worktree_index(d)
|
|
138
|
+
if wt:
|
|
139
|
+
self.notes[d] = "worktree do repo principal"
|
|
140
|
+
return self._post(wt)
|
|
141
|
+
# scratch de sessao carrega o projeto de origem codificado no caminho:
|
|
142
|
+
# /tmp/claude-1000/-home-user-code-app/<sessao>/scratchpad/...
|
|
143
|
+
enc = self._encoded_origin(d)
|
|
144
|
+
if enc:
|
|
145
|
+
self.notes[d] = "scratch de sessao, projeto lido do caminho"
|
|
146
|
+
return self._post(enc)
|
|
147
|
+
if os.path.isdir(d):
|
|
148
|
+
# git-common-dir aponta o repo principal mesmo dentro de worktree
|
|
149
|
+
common = _git(["rev-parse", "--git-common-dir"], cwd=d)
|
|
150
|
+
if common:
|
|
151
|
+
# git devolve caminho relativo ao cwd (".git", "../../.git"):
|
|
152
|
+
# sem normalizar, o nome do projeto vira ".."
|
|
153
|
+
raw = common if os.path.isabs(common) else os.path.join(d, common)
|
|
154
|
+
p = Path(os.path.normpath(raw))
|
|
155
|
+
if p.name == ".git":
|
|
156
|
+
self.notes[d] = "raiz git"
|
|
157
|
+
return self._post(p.parent.name)
|
|
158
|
+
for ws in self.workspaces:
|
|
159
|
+
if d.startswith(ws + "/"):
|
|
160
|
+
self.notes[d] = "primeiro nivel do workspace"
|
|
161
|
+
return self._post(d[len(ws) + 1:].split("/")[0])
|
|
162
|
+
self.notes[d] = "nome da pasta (sem repo identificavel)"
|
|
163
|
+
return self._post(os.path.basename(d.rstrip("/")) or d)
|
|
164
|
+
|
|
165
|
+
@staticmethod
|
|
166
|
+
def _post(name: str) -> str:
|
|
167
|
+
if UUID_RE.fullmatch(name or ""):
|
|
168
|
+
return "(worktree temporario removido)"
|
|
169
|
+
if name in SCRATCH or TMPNAME_RE.fullmatch(name or ""):
|
|
170
|
+
return "(scratch / temporario)"
|
|
171
|
+
return name or "(nao atribuido)"
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
#: como o custo de uma sessao foi ligado a um projeto, do mais forte ao mais
|
|
175
|
+
#: fraco. A distincao importa: "editou" e prova de entrega, "leu" e prova de
|
|
176
|
+
#: contexto, e "diretorio" e so o que a ferramenta declarou — e o diretorio
|
|
177
|
+
#: mente, porque o agente edita fora dele o tempo todo.
|
|
178
|
+
ATTRIBUTION = {
|
|
179
|
+
"edited": "arquivos escritos na sessao",
|
|
180
|
+
"read": "arquivos lidos na sessao (investigacao, sem entrega)",
|
|
181
|
+
"cwd": "diretorio declarado, sem nenhum arquivo tocado",
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def split_session_cost(session, resolver: ProjectResolver):
|
|
186
|
+
"""Rateia o custo da sessao entre os projetos, em cascata de evidencia.
|
|
187
|
+
|
|
188
|
+
Devolve (custo_por_projeto, tokens_por_projeto, metodo), onde metodo diz
|
|
189
|
+
QUAL evidencia sustentou a atribuicao. Tratar tudo como equivalente
|
|
190
|
+
esconderia que parte do numero repousa em prova fraca.
|
|
191
|
+
"""
|
|
192
|
+
def _rateio(paths):
|
|
193
|
+
projetos = [p for p in (resolver.of_path(x) for x in paths) if p]
|
|
194
|
+
if not projetos:
|
|
195
|
+
return None
|
|
196
|
+
custo: dict[str, float] = defaultdict(float)
|
|
197
|
+
toks: dict[str, int] = defaultdict(int)
|
|
198
|
+
n = len(projetos)
|
|
199
|
+
for p in projetos:
|
|
200
|
+
custo[p] += session.usd / n
|
|
201
|
+
toks[p] += session.tokens // n
|
|
202
|
+
return custo, toks
|
|
203
|
+
|
|
204
|
+
got = _rateio([e.path for e in (getattr(session, "edits", None) or [])])
|
|
205
|
+
if got:
|
|
206
|
+
return got[0], got[1], "edited"
|
|
207
|
+
|
|
208
|
+
got = _rateio([e.path for e in (getattr(session, "reads", None) or [])])
|
|
209
|
+
if got:
|
|
210
|
+
return got[0], got[1], "read"
|
|
211
|
+
|
|
212
|
+
proj = resolver.of_path(session.cwd) or session.cwd or "(sem projeto)"
|
|
213
|
+
return ({proj: session.usd}, {proj: session.tokens}, "cwd")
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
class RepoHistory:
|
|
217
|
+
"""Historico de um repositorio: arquivo -> commits que o tocaram."""
|
|
218
|
+
|
|
219
|
+
def __init__(self, repo: Path, ref: str | None = None):
|
|
220
|
+
self.repo = Path(repo)
|
|
221
|
+
self.ref = ref or self._default_ref()
|
|
222
|
+
self.by_file: dict[str, list] = defaultdict(list)
|
|
223
|
+
self._ts: dict[str, list] = {}
|
|
224
|
+
self.commits = 0
|
|
225
|
+
self._load()
|
|
226
|
+
|
|
227
|
+
def _default_ref(self) -> str:
|
|
228
|
+
for ref in ("origin/dev", "origin/main", "origin/master", "HEAD"):
|
|
229
|
+
if _git(["rev-parse", "--verify", ref], cwd=self.repo):
|
|
230
|
+
return ref
|
|
231
|
+
return "HEAD"
|
|
232
|
+
|
|
233
|
+
def _load(self):
|
|
234
|
+
out = _git(["log", self.ref,
|
|
235
|
+
f"--pretty=format:{RS}%H{US}%ct{US}%s{US}%b{US}",
|
|
236
|
+
"--name-only"], cwd=self.repo)
|
|
237
|
+
if not out:
|
|
238
|
+
return
|
|
239
|
+
for rec in out.split(RS):
|
|
240
|
+
if not rec.strip():
|
|
241
|
+
continue
|
|
242
|
+
parts = rec.split(US)
|
|
243
|
+
if len(parts) < 5:
|
|
244
|
+
continue
|
|
245
|
+
sha, cts, subject, body = parts[0], parts[1], parts[2], parts[3]
|
|
246
|
+
try:
|
|
247
|
+
ts = int(cts)
|
|
248
|
+
except ValueError:
|
|
249
|
+
continue
|
|
250
|
+
self.commits += 1
|
|
251
|
+
tickets = tuple(sorted(set(TICKET_RE.findall(f"{subject}\n{body}"))))
|
|
252
|
+
m = SCOPE_RE.match(subject.strip())
|
|
253
|
+
entry = (ts, tickets, m.group(1) if m else None, subject.strip(), sha)
|
|
254
|
+
for path in (x for x in parts[4].splitlines() if x.strip()):
|
|
255
|
+
self.by_file[path].append(entry)
|
|
256
|
+
for path, lst in self.by_file.items():
|
|
257
|
+
lst.sort(key=lambda x: x[0])
|
|
258
|
+
self._ts[path] = [x[0] for x in lst]
|
|
259
|
+
|
|
260
|
+
def relpath(self, abs_path: str) -> str | None:
|
|
261
|
+
"""Maior sufixo do caminho que exista no historico do repo."""
|
|
262
|
+
parts = abs_path.strip("/").split("/")
|
|
263
|
+
for i in range(len(parts) - 1):
|
|
264
|
+
cand = "/".join(parts[i:])
|
|
265
|
+
if cand in self.by_file:
|
|
266
|
+
# sufixo curto demais colide entre repos
|
|
267
|
+
if cand.count("/") >= 1 or len(parts) - i > 1:
|
|
268
|
+
return cand
|
|
269
|
+
return None
|
|
270
|
+
|
|
271
|
+
def commit_for(self, relpath: str, when_iso: str | None):
|
|
272
|
+
"""Primeiro commit posterior a edicao que tocou aquele arquivo."""
|
|
273
|
+
if not when_iso or relpath not in self._ts:
|
|
274
|
+
return None
|
|
275
|
+
try:
|
|
276
|
+
ts = int(datetime.fromisoformat(
|
|
277
|
+
when_iso.replace("Z", "+00:00")).timestamp())
|
|
278
|
+
except ValueError:
|
|
279
|
+
return None
|
|
280
|
+
i = bisect_left(self._ts[relpath], ts)
|
|
281
|
+
lst = self.by_file[relpath]
|
|
282
|
+
return lst[i] if i < len(lst) else None
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _dia_de(iso_ts):
|
|
286
|
+
"""`2026-07-28T13:45:00Z` -> `2026-07-28`. None se a data nao presta.
|
|
287
|
+
|
|
288
|
+
Mesma conversao do `report._day_of`. Fica repetida aqui de proposito: o
|
|
289
|
+
report importa este modulo, entao importar de volta fecharia um ciclo por
|
|
290
|
+
causa de cinco linhas.
|
|
291
|
+
"""
|
|
292
|
+
if not iso_ts:
|
|
293
|
+
return None
|
|
294
|
+
try:
|
|
295
|
+
return datetime.fromisoformat(
|
|
296
|
+
str(iso_ts).replace("Z", "+00:00")).date().isoformat()
|
|
297
|
+
except ValueError:
|
|
298
|
+
return None
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def features_for_repo(sessions, repo: Path, resolver: ProjectResolver,
|
|
302
|
+
ref: str | None = None):
|
|
303
|
+
"""Custo por area do produto e por item do board, dentro de um repo."""
|
|
304
|
+
if not _git(["rev-parse", "--git-dir"], cwd=repo):
|
|
305
|
+
return {"error": f"{repo} nao e um repositorio git"}
|
|
306
|
+
hist = RepoHistory(repo, ref)
|
|
307
|
+
if not hist.commits:
|
|
308
|
+
return {"error": f"repositorio sem commits alcancaveis em {repo}"}
|
|
309
|
+
|
|
310
|
+
# `days` guarda o custo do escopo POR DIA. A data ja existe (`e.ts`, usada
|
|
311
|
+
# logo abaixo para achar o commit), entao isto nao inventa distribuicao:
|
|
312
|
+
# so para de jogar fora uma quebra que o motor ja tinha em maos. Sem ela o
|
|
313
|
+
# sync so consegue mandar um escopo por projeto, e o board mostra o custo
|
|
314
|
+
# inteiro do projeto colado no escopo dominante.
|
|
315
|
+
by_scope: dict[str, dict] = defaultdict(
|
|
316
|
+
lambda: {"cost": 0.0, "edits": 0, "days": defaultdict(float)})
|
|
317
|
+
by_ticket: dict[str, dict] = defaultdict(
|
|
318
|
+
lambda: {"cost": 0.0, "edits": 0, "subject": "", "days": defaultdict(float)})
|
|
319
|
+
matched = unmatched = 0.0
|
|
320
|
+
|
|
321
|
+
for s in sessions:
|
|
322
|
+
if not s.edits or not s.usd:
|
|
323
|
+
continue
|
|
324
|
+
share = s.usd / len(s.edits)
|
|
325
|
+
for e in s.edits:
|
|
326
|
+
rel = hist.relpath(e.path)
|
|
327
|
+
if not rel:
|
|
328
|
+
continue
|
|
329
|
+
c = hist.commit_for(rel, e.ts)
|
|
330
|
+
if not c:
|
|
331
|
+
unmatched += share
|
|
332
|
+
continue
|
|
333
|
+
_ts, tickets, scope, subject, _sha = c
|
|
334
|
+
matched += share
|
|
335
|
+
key = scope or "(sem escopo)"
|
|
336
|
+
by_scope[key]["cost"] += share
|
|
337
|
+
by_scope[key]["edits"] += 1
|
|
338
|
+
dia = _dia_de(e.ts)
|
|
339
|
+
if dia: # edicao sem data nao vira linha de dia
|
|
340
|
+
by_scope[key]["days"][dia] += share
|
|
341
|
+
if tickets:
|
|
342
|
+
for t in tickets: # commit com N itens: 1/N
|
|
343
|
+
by_ticket[t]["cost"] += share / len(tickets)
|
|
344
|
+
by_ticket[t]["edits"] += 1
|
|
345
|
+
by_ticket[t]["subject"] = by_ticket[t]["subject"] or subject
|
|
346
|
+
if dia:
|
|
347
|
+
by_ticket[t]["days"][dia] += share / len(tickets)
|
|
348
|
+
else:
|
|
349
|
+
by_ticket["(sem item)"]["cost"] += share
|
|
350
|
+
by_ticket["(sem item)"]["edits"] += 1
|
|
351
|
+
if dia:
|
|
352
|
+
by_ticket["(sem item)"]["days"][dia] += share
|
|
353
|
+
|
|
354
|
+
return {"by_scope": {k: {**v, "days": dict(v["days"])}
|
|
355
|
+
for k, v in by_scope.items()},
|
|
356
|
+
"by_ticket": {k: {**v, "days": dict(v["days"])}
|
|
357
|
+
for k, v in by_ticket.items()},
|
|
358
|
+
"matched": matched, "unmatched": unmatched,
|
|
359
|
+
"ref": hist.ref, "commits": hist.commits}
|