dsh-tacit 0.4.0 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/client/client.js +574 -195
- package/docs/README.zh.md +2 -1
- package/lib/analyze.js +144 -40
- package/lib/fold.js +11 -5
- package/lib/redact.js +53 -0
- package/lib/routes.js +1 -0
- package/lib/schema.js +71 -12
- package/lib/service.js +210 -51
- package/lib/store.js +2 -2
- package/lib/usage.js +57 -14
- package/lib/workspace.js +62 -0
- package/package.json +1 -1
package/lib/service.js
CHANGED
|
@@ -32,6 +32,7 @@ import {
|
|
|
32
32
|
bootstrapArgSchema,
|
|
33
33
|
usageArgSchema,
|
|
34
34
|
usageRunArgSchema,
|
|
35
|
+
directiveReceiptArgSchema,
|
|
35
36
|
} from './schema.js'
|
|
36
37
|
import { dayKey } from './store.js'
|
|
37
38
|
import { createUsageTracker, totalTokens } from './usage.js'
|
|
@@ -71,22 +72,26 @@ import {
|
|
|
71
72
|
ANALYSIS_TOOL,
|
|
72
73
|
IMPROVE_TOOL,
|
|
73
74
|
DISTILL_TOOL,
|
|
75
|
+
POLICY_RE,
|
|
74
76
|
clipSafe,
|
|
75
77
|
isMessyTurn,
|
|
76
78
|
looksLikeCorrection,
|
|
77
79
|
looksLikeContinuation,
|
|
80
|
+
looksLikeJobNotification,
|
|
78
81
|
classifyDirectives,
|
|
79
82
|
DIRECTIVE_SYSTEM_PROMPT,
|
|
80
83
|
DIRECTIVE_TOOL,
|
|
81
84
|
DIRECTIVE_MAX_TOKENS,
|
|
82
85
|
DIRECTIVE_TIMEOUT_MS,
|
|
83
86
|
buildDirectiveUserText,
|
|
87
|
+
directiveLabels,
|
|
84
88
|
buildSteeringSection,
|
|
85
89
|
renderSteeringSection,
|
|
86
|
-
workspaceLabel,
|
|
87
90
|
scopeOf,
|
|
88
91
|
capDirectives,
|
|
89
92
|
mergeDirectives,
|
|
93
|
+
isDeadDirective,
|
|
94
|
+
MAX_DIRECTIVE_EVIDENCE,
|
|
90
95
|
ENRICH_SYSTEM_PROMPT,
|
|
91
96
|
ENRICH_TOOL,
|
|
92
97
|
ENRICH_MAX_TOKENS,
|
|
@@ -99,6 +104,7 @@ import {
|
|
|
99
104
|
computeTrend,
|
|
100
105
|
markCorrections,
|
|
101
106
|
} from './analyze.js'
|
|
107
|
+
import { normalizeWorkspace, workspaceContains, workspaceLabel, workspaceLabels } from './workspace.js'
|
|
102
108
|
|
|
103
109
|
/** In-memory rewrite ledger bounds (never persisted). */
|
|
104
110
|
const MAX_REWRITE_RECORDS = 50
|
|
@@ -162,12 +168,13 @@ function turnsOf(service, sessionId) {
|
|
|
162
168
|
}
|
|
163
169
|
}
|
|
164
170
|
|
|
165
|
-
/** The
|
|
171
|
+
/** The normalised workspace directory a session was created in, else undefined. */
|
|
166
172
|
function cwdOf(session) {
|
|
167
173
|
const cwd = session !== null && typeof session === 'object' && session.header !== null && typeof session.header === 'object'
|
|
168
174
|
? session.header.cwd
|
|
169
175
|
: undefined
|
|
170
|
-
|
|
176
|
+
const workspace = normalizeWorkspace(cwd)
|
|
177
|
+
return workspace.length > 0 ? workspace : undefined
|
|
171
178
|
}
|
|
172
179
|
|
|
173
180
|
/** A human label for a session: the workspace directory's basename, else ''. */
|
|
@@ -179,17 +186,30 @@ function sessionLabelOf(service, sessionId) {
|
|
|
179
186
|
/** Every distinct workspace among the live sessions, labelled for the UI. */
|
|
180
187
|
function listWorkspaces(service) {
|
|
181
188
|
const sessions = typeof service.sessions?.list === 'function' ? service.sessions.list() : []
|
|
182
|
-
const
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
189
|
+
const cwds = (Array.isArray(sessions) ? sessions : []).map(cwdOf).filter((cwd) => cwd !== undefined)
|
|
190
|
+
return [...workspaceLabels(cwds)]
|
|
191
|
+
.map(([cwd, label]) => ({ cwd, label }))
|
|
192
|
+
.sort((a, b) => a.label.localeCompare(b.label))
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/** The `workspaceSeenAt` map narrowed to the workspaces a directive still points at. */
|
|
196
|
+
function pruneSeenAt(seenAt, directives) {
|
|
197
|
+
const scopes = new Set(directives.map(scopeOf))
|
|
198
|
+
return Object.fromEntries(Object.entries(seenAt).filter(([scope]) => scopes.has(scope)))
|
|
188
199
|
}
|
|
189
200
|
|
|
190
|
-
/**
|
|
201
|
+
/**
|
|
202
|
+
* Bounded context of the last two turns the user actually wrote, read from the
|
|
203
|
+
* digest. Background-job notifications and bare continuations are finished
|
|
204
|
+
* turns too, and one of them landing after the last real exchange would
|
|
205
|
+
* otherwise hide the facts the rewrite needs.
|
|
206
|
+
*/
|
|
191
207
|
function recentContextOf(turns) {
|
|
192
|
-
const
|
|
208
|
+
const written = (turn) => {
|
|
209
|
+
const prompt = typeof turn.prompt === 'string' ? turn.prompt.trim() : ''
|
|
210
|
+
return prompt.length > 0 && !looksLikeJobNotification(prompt) && !looksLikeContinuation(prompt)
|
|
211
|
+
}
|
|
212
|
+
const finished = (Array.isArray(turns) ? turns : []).filter((turn) => turn?.finished === true && written(turn)).slice(-2)
|
|
193
213
|
if (finished.length === 0) return ''
|
|
194
214
|
return finished.map((turn) => {
|
|
195
215
|
const prompt = typeof turn.prompt === 'string' ? turn.prompt.slice(0, 600) : ''
|
|
@@ -300,9 +320,15 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
300
320
|
|
|
301
321
|
const safeProfile = () => {
|
|
302
322
|
const parsed = profileSchema.safeParse(store.profile())
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
323
|
+
if (!parsed.success) {
|
|
324
|
+
return { analyzedCount: 0, patterns: [], updatedAt: 0, styleRules: [], feedbackLog: [], pendingDistill: 0, directives: [], analysesSinceDirectives: 0, workspaceSeenAt: {} }
|
|
325
|
+
}
|
|
326
|
+
for (const entry of parsed.data.directives) {
|
|
327
|
+
const workspace = normalizeWorkspace(entry.workspace)
|
|
328
|
+
if (workspace.length > 0) entry.workspace = workspace
|
|
329
|
+
else delete entry.workspace
|
|
330
|
+
}
|
|
331
|
+
return parsed.data
|
|
306
332
|
}
|
|
307
333
|
let directiveSeq = 0
|
|
308
334
|
const nextDirectiveId = () => {
|
|
@@ -311,7 +337,9 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
311
337
|
}
|
|
312
338
|
let directivesInFlight = false
|
|
313
339
|
/** Steering `{ text, ids }` frozen per live session object (keeps the model's prefix cache stable within a session). */
|
|
314
|
-
|
|
340
|
+
let steeringFrozen = new WeakMap()
|
|
341
|
+
/** Forget every open session's frozen steering: the next assembly re-freezes from the profile as it is now. */
|
|
342
|
+
const invalidateSteering = () => { steeringFrozen = new WeakMap() }
|
|
315
343
|
|
|
316
344
|
const nextRewriteId = () => {
|
|
317
345
|
rewriteSeq += 1
|
|
@@ -341,13 +369,24 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
341
369
|
return profile
|
|
342
370
|
}
|
|
343
371
|
|
|
372
|
+
/** The normalised workspace of every session loaded right now. */
|
|
373
|
+
const liveCwds = () => listWorkspaces(serviceOf(ctx)).map((entry) => entry.cwd)
|
|
374
|
+
|
|
344
375
|
/** Bound every v2 field, validate, persist; returns the stored profile. */
|
|
345
376
|
const capAndSaveProfile = (profile) => {
|
|
346
377
|
const config = effectiveConfig()
|
|
347
378
|
profile.patterns = profile.patterns.slice(0, config.maxPatterns)
|
|
348
379
|
profile.styleRules = profile.styleRules.slice(-MAX_STYLE_RULES)
|
|
349
380
|
profile.feedbackLog = profile.feedbackLog.slice(-MAX_FEEDBACK_LOG)
|
|
350
|
-
|
|
381
|
+
const cwds = liveCwds()
|
|
382
|
+
const seenAt = profile.workspaceSeenAt ?? {}
|
|
383
|
+
for (const entry of profile.directives) {
|
|
384
|
+
const scope = scopeOf(entry)
|
|
385
|
+
if (scope !== '' && cwds.some((cwd) => workspaceContains(scope, cwd))) seenAt[scope] = Date.now()
|
|
386
|
+
}
|
|
387
|
+
profile.workspaceSeenAt = pruneSeenAt(seenAt, profile.directives)
|
|
388
|
+
profile.directives = capDirectives(profile.directives, { seenAt: profile.workspaceSeenAt })
|
|
389
|
+
profile.workspaceSeenAt = pruneSeenAt(profile.workspaceSeenAt, profile.directives)
|
|
351
390
|
profile.updatedAt = Date.now()
|
|
352
391
|
const validated = profileSchema.parse(profile)
|
|
353
392
|
store.saveProfile(validated)
|
|
@@ -471,15 +510,17 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
471
510
|
const correctionRate = trial.corrected / trial.turns
|
|
472
511
|
const messyRate = trial.messy / trial.turns
|
|
473
512
|
const worse = correctionRate > trial.baselineCorrectionRate + config.directiveWorseBy
|
|
474
|
-
? 'corrections ' + pct(trial.baselineCorrectionRate) + ' → ' + pct(correctionRate)
|
|
513
|
+
? 'corrections rose ' + pct(trial.baselineCorrectionRate) + ' → ' + pct(correctionRate)
|
|
475
514
|
: messyRate > trial.baselineMessyRate + 2 * config.directiveWorseBy
|
|
476
|
-
? 'messy turns ' + pct(trial.baselineMessyRate) + ' → ' + pct(messyRate)
|
|
515
|
+
? 'messy turns rose ' + pct(trial.baselineMessyRate) + ' → ' + pct(messyRate)
|
|
477
516
|
: null
|
|
478
517
|
verdicts += 1
|
|
518
|
+
entry.evaluatedAt = Date.now()
|
|
519
|
+
entry.updatedAt = entry.evaluatedAt
|
|
479
520
|
if (worse !== null) {
|
|
480
521
|
entry.status = 'retired'
|
|
481
522
|
entry.enabled = false
|
|
482
|
-
entry.retiredReason = worse + '
|
|
523
|
+
entry.retiredReason = worse + ' during its trial'
|
|
483
524
|
console.info('[tacit] retired directive (' + entry.retiredReason + '): ' + entry.text)
|
|
484
525
|
} else {
|
|
485
526
|
entry.status = 'active'
|
|
@@ -527,7 +568,7 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
527
568
|
return earlier.length > 0 ? earlier[earlier.length - 1] : null
|
|
528
569
|
}
|
|
529
570
|
|
|
530
|
-
const runAnalysis = (sessionId, turn, { trigger = 'manual', followUp = '', digest = null, previousDigest = null, runId = '' } = {}) => {
|
|
571
|
+
const runAnalysis = (sessionId, turn, { trigger = 'manual', followUp = '', digest = null, previousDigest = null, runId = '', force = false } = {}) => {
|
|
531
572
|
const profile = safeProfile()
|
|
532
573
|
const key = `${sessionId}:${turn}`
|
|
533
574
|
/** The run this analysis is billed to: the caller's batch run, or one of its own. Stays '' on a soft refusal. */
|
|
@@ -547,6 +588,10 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
547
588
|
if (trigger === 'manual' && looksLikeContinuation(record.prompt)) {
|
|
548
589
|
return { ok: false, report: null, profile, code: 'continuation', detail: '' }
|
|
549
590
|
}
|
|
591
|
+
// A report already exists: a second paid look at the same turn only on request.
|
|
592
|
+
if (trigger === 'manual' && !force && store.report(sessionId, turn) !== null) {
|
|
593
|
+
return { ok: false, report: null, profile, code: 'already-analyzed', detail: '' }
|
|
594
|
+
}
|
|
550
595
|
const previous = previousDigest !== null && typeof previousDigest === 'object' ? previousDigest : previousFinishedOf(turns, turn)
|
|
551
596
|
const userText = buildAnalysisUserText(record, { followUp, previous })
|
|
552
597
|
if (userText === null) return { ok: false, report: null, profile, code: 'not-retained', detail: '' }
|
|
@@ -686,6 +731,9 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
686
731
|
if (session === null || session === undefined || typeof session !== 'object') return steeringNow().text
|
|
687
732
|
let frozen = steeringFrozen.get(session)
|
|
688
733
|
if (frozen === undefined) {
|
|
734
|
+
// A workspace seen for the first time may revive a paused trial.
|
|
735
|
+
const profile = safeProfile()
|
|
736
|
+
if (startNextTrial(profile)) capAndSaveProfile(profile)
|
|
689
737
|
frozen = steeringNow(cwdOf(session))
|
|
690
738
|
steeringFrozen.set(session, frozen)
|
|
691
739
|
if (typeof session.id === 'string' && session.id.length > 0) {
|
|
@@ -706,19 +754,39 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
706
754
|
}
|
|
707
755
|
|
|
708
756
|
/**
|
|
709
|
-
* One trial per scope at a time
|
|
710
|
-
*
|
|
711
|
-
*
|
|
757
|
+
* One trial per scope at a time, and only in a workspace some session is
|
|
758
|
+
* loaded in: a candidate whose workspace is gone goes back to the queue and
|
|
759
|
+
* frees the slot, then the oldest enabled queued directive of every free
|
|
760
|
+
* seen scope starts its trial now, with baselines measured at this moment.
|
|
761
|
+
* Idempotent; returns whether any status changed.
|
|
712
762
|
*/
|
|
713
763
|
const startNextTrial = (profile) => {
|
|
764
|
+
const cwds = liveCwds()
|
|
765
|
+
const seen = new Set([''])
|
|
766
|
+
for (const entry of profile.directives) {
|
|
767
|
+
const scope = scopeOf(entry)
|
|
768
|
+
if (scope !== '' && cwds.some((cwd) => workspaceContains(scope, cwd))) seen.add(scope)
|
|
769
|
+
}
|
|
770
|
+
let changed = false
|
|
771
|
+
for (const entry of profile.directives) {
|
|
772
|
+
if (entry.status !== 'candidate' || seen.has(scopeOf(entry))) continue
|
|
773
|
+
entry.status = 'queued'
|
|
774
|
+
delete entry.trial
|
|
775
|
+
changed = true
|
|
776
|
+
console.info('[tacit] trial paused, workspace not loaded: ' + entry.text)
|
|
777
|
+
}
|
|
778
|
+
const review = effectiveConfig().reviewCandidates === true
|
|
714
779
|
const busy = new Set(profile.directives.filter((entry) => entry.status === 'candidate').map(scopeOf))
|
|
715
780
|
for (const entry of profile.directives) {
|
|
716
|
-
if (entry.status !== 'queued' || entry.enabled === false || busy.has(scopeOf(entry))) continue
|
|
781
|
+
if (entry.status !== 'queued' || entry.enabled === false || busy.has(scopeOf(entry)) || !seen.has(scopeOf(entry))) continue
|
|
782
|
+
if (review && entry.source !== 'user' && !(entry.approvedAt > 0)) continue
|
|
717
783
|
busy.add(scopeOf(entry))
|
|
718
784
|
entry.status = 'candidate'
|
|
719
785
|
entry.trial = { turns: 0, messy: 0, corrected: 0, ...baselinesFor(scopeOf(entry)), startedAt: Date.now() }
|
|
786
|
+
changed = true
|
|
720
787
|
console.info('[tacit] directive on trial: ' + entry.text)
|
|
721
788
|
}
|
|
789
|
+
return changed
|
|
722
790
|
}
|
|
723
791
|
|
|
724
792
|
/** ONE small call every `directiveEvery` new analyses (or forced). Soft-fails; never throws. */
|
|
@@ -733,19 +801,19 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
733
801
|
? usage.beginRun({ type: 'directive-distillation', trigger: 'auto', sessionId, model: config.model, provider })
|
|
734
802
|
: runId
|
|
735
803
|
try {
|
|
736
|
-
const recent = store.listAllReports(20)
|
|
804
|
+
const recent = store.listAllReports(20)
|
|
805
|
+
.map((entry) => ({ ...store.report(entry.sessionId, entry.turn), sessionId: entry.sessionId, turn: entry.turn }))
|
|
806
|
+
.filter((report) => report.ok !== undefined || typeof report.problems === 'object')
|
|
807
|
+
// Newest first, before the prompt builder reverses the list for the model.
|
|
808
|
+
const evidence = recent.slice(0, MAX_DIRECTIVE_EVIDENCE).map((report) => ({ sessionId: report.sessionId, turn: report.turn, trigger: typeof report.trigger === 'string' ? report.trigger : 'manual' }))
|
|
737
809
|
// The model sees workspace names only; map them back to the directories they stand for.
|
|
738
|
-
const
|
|
739
|
-
|
|
740
|
-
if (typeof report.cwd !== 'string' || report.cwd.length === 0) continue
|
|
741
|
-
const label = workspaceLabel(report.cwd)
|
|
742
|
-
if (label.length > 0 && !workspaces.has(label)) workspaces.set(label, report.cwd)
|
|
743
|
-
}
|
|
810
|
+
const labels = directiveLabels(profile, recent)
|
|
811
|
+
const workspaces = new Map([...labels].map(([cwd, label]) => [label, cwd]))
|
|
744
812
|
const text = await callCoachModel(ctx, metered(usageRunId, { op: 'directive-distillation', sessionId }, {
|
|
745
813
|
provider,
|
|
746
814
|
model: config.model,
|
|
747
815
|
system: DIRECTIVE_SYSTEM_PROMPT,
|
|
748
|
-
userText: buildDirectiveUserText(profile, recent.reverse()),
|
|
816
|
+
userText: buildDirectiveUserText(profile, recent.reverse(), { labels }),
|
|
749
817
|
maxTokens: DIRECTIVE_MAX_TOKENS,
|
|
750
818
|
timeoutMs: DIRECTIVE_TIMEOUT_MS,
|
|
751
819
|
tool: DIRECTIVE_TOOL,
|
|
@@ -757,12 +825,31 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
757
825
|
console.warn('[tacit] directive distillation returned nothing usable; will retry after the next analysis:', clipSafe(text, 300))
|
|
758
826
|
return
|
|
759
827
|
}
|
|
760
|
-
const items =
|
|
761
|
-
|
|
762
|
-
|
|
763
|
-
|
|
764
|
-
|
|
828
|
+
const items = []
|
|
829
|
+
for (const item of kept) {
|
|
830
|
+
if (item.workspace !== undefined && !workspaces.has(item.workspace)) {
|
|
831
|
+
console.info('[tacit] dropped directive (its workspace "' + item.workspace + '" is not one the evidence came from):', item.text)
|
|
832
|
+
continue
|
|
833
|
+
}
|
|
834
|
+
items.push({
|
|
835
|
+
text: item.text,
|
|
836
|
+
...(item.id === undefined ? {} : { id: item.id }),
|
|
837
|
+
...(item.workspace === undefined ? {} : { workspace: workspaces.get(item.workspace) }),
|
|
838
|
+
})
|
|
839
|
+
}
|
|
840
|
+
if (items.length === 0) return
|
|
841
|
+
const before = new Map(safeProfile().directives.map((entry) => [entry.id, entry.text]))
|
|
765
842
|
profile = mergeDirectives(safeProfile(), items, { nextId: nextDirectiveId })
|
|
843
|
+
const now = Date.now()
|
|
844
|
+
for (const entry of profile.directives) {
|
|
845
|
+
if (entry.source === 'user' || isDeadDirective(entry)) continue
|
|
846
|
+
const previous = before.get(entry.id)
|
|
847
|
+
if (previous === entry.text) continue
|
|
848
|
+
if (previous !== undefined) entry.version = (entry.version ?? 1) + 1
|
|
849
|
+
entry.updatedAt = now
|
|
850
|
+
entry.evidence = evidence
|
|
851
|
+
entry.distillationRunId = usageRunId
|
|
852
|
+
}
|
|
766
853
|
startNextTrial(profile)
|
|
767
854
|
profile.analysesSinceDirectives = 0
|
|
768
855
|
capAndSaveProfile(profile)
|
|
@@ -1019,7 +1106,7 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
1019
1106
|
* an auto trigger or another batch — reports `busy` and costs nothing. A
|
|
1020
1107
|
* bootstrap running elsewhere does NOT block the batch.
|
|
1021
1108
|
*/
|
|
1022
|
-
const analyzeBatch = async ({ sessionId, turns }) => {
|
|
1109
|
+
const analyzeBatch = async ({ sessionId, turns, force = false }) => {
|
|
1023
1110
|
const { session } = turnsOf(serviceOf(ctx), sessionId)
|
|
1024
1111
|
if (session === undefined) return { ok: false, results: [], profile: safeProfile(), run: null, code: 'no-session', detail: '' }
|
|
1025
1112
|
const wanted = [...new Set(turns)].sort((a, b) => a - b)
|
|
@@ -1043,7 +1130,7 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
1043
1130
|
const at = next
|
|
1044
1131
|
next += 1
|
|
1045
1132
|
const turn = wanted[at]
|
|
1046
|
-
const result = await runAnalysis(sessionId, turn, { trigger: 'manual', runId })
|
|
1133
|
+
const result = await runAnalysis(sessionId, turn, { trigger: 'manual', runId, force })
|
|
1047
1134
|
const ok = result !== null && typeof result === 'object' && result.ok === true
|
|
1048
1135
|
if (ok) analyzed += 1
|
|
1049
1136
|
results[at] = { turn, ok, code: result?.code ?? 'call-failed', report: ok ? result.report : null }
|
|
@@ -1056,8 +1143,8 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
1056
1143
|
// made and nothing failed, so this is a real request that succeeded —
|
|
1057
1144
|
// not the zero-attempt `failed` run the default derivation would write.
|
|
1058
1145
|
const entries = results.filter((entry) => entry !== null && entry !== undefined)
|
|
1059
|
-
const
|
|
1060
|
-
closeRun(runId, { requested: wanted.length, analyzed, skipped: wanted.length - analyzed },
|
|
1146
|
+
const allSkipped = analyzed === 0 && entries.length === wanted.length && entries.every((entry) => entry.code === 'busy' || entry.code === 'already-analyzed')
|
|
1147
|
+
closeRun(runId, { requested: wanted.length, analyzed, skipped: wanted.length - analyzed }, allSkipped ? 'success' : undefined)
|
|
1061
1148
|
}
|
|
1062
1149
|
return { ok: true, results, profile: safeProfile(), run: usage.runSummary(runId), code: '', detail: '' }
|
|
1063
1150
|
}
|
|
@@ -1189,7 +1276,7 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
1189
1276
|
if (!parsed.success) {
|
|
1190
1277
|
return { ok: false, report: null, profile: safeProfile(), code: 'bad-request', detail: '', run: null }
|
|
1191
1278
|
}
|
|
1192
|
-
return runAnalysis(parsed.data.sessionId, parsed.data.turn, { trigger: 'manual' })
|
|
1279
|
+
return runAnalysis(parsed.data.sessionId, parsed.data.turn, { trigger: 'manual', force: parsed.data.force === true })
|
|
1193
1280
|
},
|
|
1194
1281
|
|
|
1195
1282
|
/** Analyze a hand-picked set of turns of one session under a single run. */
|
|
@@ -1365,18 +1452,51 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
1365
1452
|
const found = profile.directives.find((entry) => entry.id === input.id)
|
|
1366
1453
|
if (found === undefined) return { ok: false, profile, steering: steeringStatus(), code: 'bad-request', detail: 'id' }
|
|
1367
1454
|
found.enabled = input.enabled
|
|
1455
|
+
found.updatedAt = Date.now()
|
|
1368
1456
|
// Re-enabling a retired directive is an explicit override.
|
|
1369
1457
|
if (input.enabled && found.status === 'retired') {
|
|
1370
1458
|
found.status = 'active'
|
|
1371
1459
|
delete found.retiredReason
|
|
1372
1460
|
}
|
|
1461
|
+
// A candidate is always enabled: switched off, it leaves the trial slot and queues again.
|
|
1462
|
+
if (!input.enabled && found.status === 'candidate') {
|
|
1463
|
+
found.status = 'queued'
|
|
1464
|
+
delete found.trial
|
|
1465
|
+
}
|
|
1466
|
+
if (!input.enabled) invalidateSteering()
|
|
1467
|
+
} else if (input.action === 'start-trial') {
|
|
1468
|
+
const found = profile.directives.find((entry) => entry.id === input.id)
|
|
1469
|
+
if (found === undefined || found.status !== 'queued') return { ok: false, profile, steering: steeringStatus(), code: 'bad-request', detail: 'id' }
|
|
1470
|
+
found.approvedAt = Date.now()
|
|
1471
|
+
found.updatedAt = found.approvedAt
|
|
1373
1472
|
} else if (input.action === 'add') {
|
|
1374
1473
|
const text = clipSafe(input.text.trim(), 220)
|
|
1375
1474
|
if (text.length === 0) return { ok: false, profile, steering: steeringStatus(), code: 'bad-request', detail: 'text' }
|
|
1376
|
-
|
|
1377
|
-
|
|
1475
|
+
if (POLICY_RE.test(text)) return { ok: false, profile, steering: steeringStatus(), code: 'directive-policy', detail: '' }
|
|
1476
|
+
const workspace = normalizeWorkspace(typeof input.workspace === 'string' ? input.workspace.trim() : '')
|
|
1477
|
+
profile.directives.push({ id: nextDirectiveId(), text, enabled: true, source: 'user', createdAt: Date.now(), ...(workspace === '' ? {} : { workspace }) })
|
|
1478
|
+
} else if (input.action === 'rescope') {
|
|
1479
|
+
const found = profile.directives.find((entry) => entry.id === input.id)
|
|
1480
|
+
if (found === undefined) return { ok: false, profile, steering: steeringStatus(), code: 'bad-request', detail: 'id' }
|
|
1481
|
+
const workspace = normalizeWorkspace(input.workspace.trim())
|
|
1482
|
+
if (workspace === '') delete found.workspace
|
|
1483
|
+
else found.workspace = workspace
|
|
1484
|
+
// A trial measured in one workspace says nothing about another.
|
|
1485
|
+
if (found.status === 'candidate') {
|
|
1486
|
+
found.status = 'queued'
|
|
1487
|
+
delete found.trial
|
|
1488
|
+
}
|
|
1378
1489
|
} else {
|
|
1379
|
-
|
|
1490
|
+
// A tombstone, not a delete: the distiller is told about it so it never
|
|
1491
|
+
// proposes the same directive (or a rewording) back.
|
|
1492
|
+
const found = profile.directives.find((entry) => entry.id === input.id)
|
|
1493
|
+
if (found !== undefined) {
|
|
1494
|
+
found.status = 'removed'
|
|
1495
|
+
found.enabled = false
|
|
1496
|
+
found.updatedAt = Date.now()
|
|
1497
|
+
delete found.trial
|
|
1498
|
+
invalidateSteering()
|
|
1499
|
+
}
|
|
1380
1500
|
}
|
|
1381
1501
|
startNextTrial(profile)
|
|
1382
1502
|
const saved = capAndSaveProfile(profile)
|
|
@@ -1440,15 +1560,54 @@ export function createCoachService(ctx, store, effectiveConfig) {
|
|
|
1440
1560
|
async usageReport(args) {
|
|
1441
1561
|
const parsed = usageArgSchema.safeParse(args !== null && typeof args === 'object' ? args : {})
|
|
1442
1562
|
if (!parsed.success) return { ok: false, code: 'bad-request', detail: '' }
|
|
1443
|
-
return
|
|
1444
|
-
|
|
1445
|
-
|
|
1446
|
-
|
|
1447
|
-
|
|
1448
|
-
|
|
1563
|
+
return {
|
|
1564
|
+
...usage.report({
|
|
1565
|
+
config: effectiveConfig(),
|
|
1566
|
+
pricingStatus: pricing.status(),
|
|
1567
|
+
pricingRates: pricing.rates(),
|
|
1568
|
+
filters: parsed.data,
|
|
1569
|
+
}),
|
|
1570
|
+
auto: autoStatus(),
|
|
1571
|
+
}
|
|
1449
1572
|
},
|
|
1450
1573
|
|
|
1451
1574
|
/** One run with its attempt rows (a live run included); expired ids are a soft `unknown-run`. */
|
|
1575
|
+
/** One directive's provenance and cost: ids, counts and money, never prompt text. */
|
|
1576
|
+
async directiveReceipt(args) {
|
|
1577
|
+
const parsed = directiveReceiptArgSchema.safeParse(args)
|
|
1578
|
+
if (!parsed.success) return { ok: false, receipt: null, code: 'bad-request', detail: '' }
|
|
1579
|
+
const entry = safeProfile().directives.find((candidate) => candidate.id === parsed.data.id)
|
|
1580
|
+
if (entry === undefined) return { ok: false, receipt: null, code: 'unknown-directive', detail: '' }
|
|
1581
|
+
const triggers = {}
|
|
1582
|
+
const conversations = new Set()
|
|
1583
|
+
for (const item of entry.evidence) {
|
|
1584
|
+
triggers[item.trigger] = (triggers[item.trigger] ?? 0) + 1
|
|
1585
|
+
conversations.add(item.sessionId)
|
|
1586
|
+
}
|
|
1587
|
+
const run = entry.distillationRunId === '' ? null : usage.run(entry.distillationRunId)
|
|
1588
|
+
const attempts = run !== null && Array.isArray(run.attempts) ? run.attempts.filter((attempt) => attempt.op === 'directive-distillation') : []
|
|
1589
|
+
const priced = attempts.filter((attempt) => attempt.priced !== null && typeof attempt.priced === 'object' && typeof attempt.priced.usd === 'number')
|
|
1590
|
+
const receipt = {
|
|
1591
|
+
id: entry.id,
|
|
1592
|
+
text: entry.text,
|
|
1593
|
+
scope: scopeOf(entry),
|
|
1594
|
+
status: entry.status,
|
|
1595
|
+
source: entry.source,
|
|
1596
|
+
enabled: entry.enabled !== false,
|
|
1597
|
+
createdAt: entry.createdAt,
|
|
1598
|
+
updatedAt: entry.updatedAt,
|
|
1599
|
+
evaluatedAt: entry.evaluatedAt,
|
|
1600
|
+
approvedAt: entry.approvedAt,
|
|
1601
|
+
version: entry.version,
|
|
1602
|
+
trial: entry.trial ?? null,
|
|
1603
|
+
retiredReason: entry.retiredReason ?? '',
|
|
1604
|
+
triggers,
|
|
1605
|
+
evidence: { turns: entry.evidence.length, conversations: conversations.size, items: entry.evidence },
|
|
1606
|
+
cost: { runId: entry.distillationRunId, calls: attempts.length, usd: priced.length === 0 ? null : priced.reduce((sum, attempt) => sum + attempt.priced.usd, 0) },
|
|
1607
|
+
}
|
|
1608
|
+
return { ok: true, receipt, code: '', detail: '' }
|
|
1609
|
+
},
|
|
1610
|
+
|
|
1452
1611
|
async usageRun(args) {
|
|
1453
1612
|
const parsed = usageRunArgSchema.safeParse(args)
|
|
1454
1613
|
if (!parsed.success) return { ok: false, run: null, code: 'bad-request', detail: '' }
|
package/lib/store.js
CHANGED
|
@@ -8,7 +8,7 @@
|
|
|
8
8
|
* profile.json persistent user mistake profile
|
|
9
9
|
* reports/<sessionId>/<turn>.json analysis reports
|
|
10
10
|
* usage/<YYYY-MM-DD>.json per-day usage ledger (runs of metered model calls)
|
|
11
|
-
* usage/summary.json rolling lifetime/byType/byModel/day totals
|
|
11
|
+
* usage/summary.json rolling lifetime/byType/byModel/byProvider/byTrigger/day totals
|
|
12
12
|
*
|
|
13
13
|
* Safety rules (hard constraints):
|
|
14
14
|
* - writes are atomic (temp file + rename) and never truncate an existing
|
|
@@ -62,7 +62,7 @@ export function dayKeyBefore(today, days) {
|
|
|
62
62
|
}
|
|
63
63
|
|
|
64
64
|
function emptyUsageSummaryRaw() {
|
|
65
|
-
return { version: 1, trackingSince: Date.now(), lifetime: {}, byType: {}, byModel: {}, days: {} }
|
|
65
|
+
return { version: 1, trackingSince: Date.now(), lifetime: {}, byType: {}, byModel: {}, byProvider: {}, byTrigger: {}, days: {} }
|
|
66
66
|
}
|
|
67
67
|
|
|
68
68
|
export function emptyProfile() {
|
package/lib/usage.js
CHANGED
|
@@ -27,6 +27,8 @@ import { dayKey, dayKeyBefore } from './store.js'
|
|
|
27
27
|
import { USAGE_ATTEMPT_STATUSES, USAGE_OPS, USAGE_RUN_STATUSES, USAGE_RUN_TYPES } from './schema.js'
|
|
28
28
|
|
|
29
29
|
const TOKEN_KEYS = ['inputTokens', 'outputTokens', 'cacheReadTokens', 'cacheWriteTokens', 'reasoningTokens']
|
|
30
|
+
/** The retry ops a repair pays for, each its own billed attempt. */
|
|
31
|
+
const REPAIR_OPS = ['analysis-repair', 'improve-repair']
|
|
30
32
|
/** Finished runs kept addressable for `runSummary()` after they leave `live`. */
|
|
31
33
|
const MAX_REMEMBERED_RUNS = 50
|
|
32
34
|
const MS_PER_DAY = 24 * 60 * 60 * 1000
|
|
@@ -83,7 +85,7 @@ function emptyTokens() {
|
|
|
83
85
|
}
|
|
84
86
|
|
|
85
87
|
function emptyTotals() {
|
|
86
|
-
return { attempts: 0, billedCalls: 0, unmeteredCalls: 0, unpricedCalls: 0, tokens: emptyTokens(), usdKnown: 0 }
|
|
88
|
+
return { attempts: 0, billedCalls: 0, failedCalls: 0, unmeteredCalls: 0, unpricedCalls: 0, tokens: emptyTokens(), usdKnown: 0, failedUsd: 0 }
|
|
87
89
|
}
|
|
88
90
|
|
|
89
91
|
function emptyDayTotals() {
|
|
@@ -113,9 +115,11 @@ function narrowUsage(usage) {
|
|
|
113
115
|
function addTotals(target, delta) {
|
|
114
116
|
target.attempts += delta.attempts
|
|
115
117
|
target.billedCalls += delta.billedCalls
|
|
118
|
+
target.failedCalls += delta.failedCalls
|
|
116
119
|
target.unmeteredCalls += delta.unmeteredCalls
|
|
117
120
|
target.unpricedCalls += delta.unpricedCalls
|
|
118
121
|
target.usdKnown += delta.usdKnown
|
|
122
|
+
target.failedUsd += delta.failedUsd
|
|
119
123
|
for (const key of TOKEN_KEYS) target.tokens[key] += delta.tokens[key]
|
|
120
124
|
}
|
|
121
125
|
|
|
@@ -177,7 +181,7 @@ function safeTotals(value) {
|
|
|
177
181
|
const source = isPlainObject(value) ? value : {}
|
|
178
182
|
const out = { ...emptyTotals(), ...source }
|
|
179
183
|
out.tokens = { ...emptyTokens(), ...(isPlainObject(source.tokens) ? source.tokens : {}) }
|
|
180
|
-
for (const key of ['attempts', 'billedCalls', 'unmeteredCalls', 'unpricedCalls', 'usdKnown']) out[key] = count(out[key])
|
|
184
|
+
for (const key of ['attempts', 'billedCalls', 'failedCalls', 'unmeteredCalls', 'unpricedCalls', 'usdKnown', 'failedUsd']) out[key] = count(out[key])
|
|
181
185
|
for (const key of TOKEN_KEYS) out.tokens[key] = count(out.tokens[key])
|
|
182
186
|
return out
|
|
183
187
|
}
|
|
@@ -189,10 +193,12 @@ function attemptDelta(attempt) {
|
|
|
189
193
|
return {
|
|
190
194
|
attempts: 1,
|
|
191
195
|
billedCalls: usage !== null ? 1 : 0,
|
|
196
|
+
failedCalls: usage !== null && attempt.status === 'failed' ? 1 : 0,
|
|
192
197
|
unmeteredCalls: usage === null ? 1 : 0,
|
|
193
198
|
unpricedCalls: usage !== null && priced === null ? 1 : 0,
|
|
194
199
|
tokens: usage ?? emptyTokens(),
|
|
195
200
|
usdKnown: count(priced?.usd),
|
|
201
|
+
failedUsd: attempt.status === 'failed' ? count(priced?.usd) : 0,
|
|
196
202
|
}
|
|
197
203
|
}
|
|
198
204
|
|
|
@@ -354,15 +360,19 @@ export function createUsageTracker({ store, config, pricing, now = Date.now, flu
|
|
|
354
360
|
const delta = {
|
|
355
361
|
attempts: 1,
|
|
356
362
|
billedCalls: usage !== null ? 1 : 0,
|
|
363
|
+
failedCalls: usage !== null && attempt.status === 'failed' ? 1 : 0,
|
|
357
364
|
unmeteredCalls: usage === null ? 1 : 0,
|
|
358
365
|
unpricedCalls: usage !== null && attempt.priced === null ? 1 : 0,
|
|
359
366
|
tokens: usage ?? emptyTokens(),
|
|
360
367
|
usdKnown: count(attempt.priced?.usd),
|
|
368
|
+
failedUsd: attempt.status === 'failed' ? count(attempt.priced?.usd) : 0,
|
|
361
369
|
}
|
|
362
370
|
addTotals(run.totals, delta)
|
|
363
371
|
addTotals(bucketOf(summary, 'lifetime', emptyTotals), delta)
|
|
364
372
|
addTotals(bucketOf(summary.byType, run.type, emptyTotals), delta)
|
|
365
373
|
if (model.length > 0) addTotals(bucketOf(summary.byModel, model, emptyTotals), delta)
|
|
374
|
+
if (provider.length > 0) addTotals(bucketOf(summary.byProvider, provider, emptyTotals), delta)
|
|
375
|
+
if (run.trigger.length > 0) addTotals(bucketOf(summary.byTrigger, run.trigger, emptyTotals), delta)
|
|
366
376
|
const day = bucketOf(summary.days, dayKey(startedAt), emptyDayTotals)
|
|
367
377
|
if (!isPlainObject(day.byType)) day.byType = {}
|
|
368
378
|
addTotals(day, delta)
|
|
@@ -408,10 +418,12 @@ export function createUsageTracker({ store, config, pricing, now = Date.now, flu
|
|
|
408
418
|
status: run.status,
|
|
409
419
|
attempts: run.totals.attempts,
|
|
410
420
|
billedCalls: run.totals.billedCalls,
|
|
421
|
+
failedCalls: run.totals.failedCalls,
|
|
411
422
|
unmeteredCalls: run.totals.unmeteredCalls,
|
|
412
423
|
unpricedCalls: run.totals.unpricedCalls,
|
|
413
424
|
tokens: { ...run.totals.tokens },
|
|
414
425
|
usdKnown: run.totals.usdKnown,
|
|
426
|
+
failedUsd: run.totals.failedUsd,
|
|
415
427
|
}
|
|
416
428
|
}
|
|
417
429
|
|
|
@@ -496,7 +508,7 @@ export function createUsageTracker({ store, config, pricing, now = Date.now, flu
|
|
|
496
508
|
}
|
|
497
509
|
|
|
498
510
|
/**
|
|
499
|
-
* One period card: the summary's own totals plus the
|
|
511
|
+
* One period card: the summary's own totals plus the derived figures
|
|
500
512
|
* the panel shows. `avgAnalysisUsd` is a median (a single bootstrap batch
|
|
501
513
|
* must not drag the typical cost of one analysis upwards) over the priced
|
|
502
514
|
* `analysis` attempts of the day files that were loaded — so it is only as
|
|
@@ -504,12 +516,26 @@ export function createUsageTracker({ store, config, pricing, now = Date.now, flu
|
|
|
504
516
|
*/
|
|
505
517
|
function periodOf(totals, attempts) {
|
|
506
518
|
const usds = []
|
|
519
|
+
let repairUsd = 0
|
|
520
|
+
let repairUsdNotFailed = 0
|
|
507
521
|
for (const attempt of attempts) {
|
|
508
|
-
if (attempt.op !== 'analysis') continue
|
|
509
522
|
if (!isPlainObject(attempt.priced)) continue
|
|
510
|
-
|
|
523
|
+
const usd = count(attempt.priced.usd)
|
|
524
|
+
if (attempt.op === 'analysis') usds.push(usd)
|
|
525
|
+
if (!REPAIR_OPS.includes(attempt.op)) continue
|
|
526
|
+
repairUsd += usd
|
|
527
|
+
if (attempt.status !== 'failed') repairUsdNotFailed += usd
|
|
528
|
+
}
|
|
529
|
+
// `failedOrRepairUsd` is the union of "failed" and "repair", counted once.
|
|
530
|
+
// `repairUsd` and `totals.failedUsd` both count a failed repair, so those
|
|
531
|
+
// two halves can sum to more than the union.
|
|
532
|
+
return {
|
|
533
|
+
...totals,
|
|
534
|
+
avgAnalysisUsd: median(usds),
|
|
535
|
+
cachedInputRate: cachedInputRateOf(totals.tokens),
|
|
536
|
+
repairUsd,
|
|
537
|
+
failedOrRepairUsd: totals.failedUsd + repairUsdNotFailed,
|
|
511
538
|
}
|
|
512
|
-
return { ...totals, avgAnalysisUsd: median(usds), cachedInputRate: cachedInputRateOf(totals.tokens) }
|
|
513
539
|
}
|
|
514
540
|
|
|
515
541
|
/**
|
|
@@ -644,10 +670,12 @@ export function createUsageTracker({ store, config, pricing, now = Date.now, flu
|
|
|
644
670
|
const monthPrefix = dayKey(now()).slice(0, 7)
|
|
645
671
|
const monthKeys = Object.keys(summary.days ?? {}).filter((day) => day.startsWith(monthPrefix)).sort()
|
|
646
672
|
|
|
647
|
-
// byType comes straight from the summary's own per-day buckets; byModel
|
|
648
|
-
// folded from the loaded
|
|
649
|
-
//
|
|
650
|
-
//
|
|
673
|
+
// byType comes straight from the summary's own per-day buckets; byModel,
|
|
674
|
+
// byProvider and byTrigger are folded from the loaded runs because the day
|
|
675
|
+
// buckets carry none of those three splits. Model and provider come off the
|
|
676
|
+
// attempt, the trigger off the run around it. All four cover the same
|
|
677
|
+
// last-30-days window, whatever `range` asked for — the detail window above
|
|
678
|
+
// always includes those 30 days.
|
|
651
679
|
const byType = {}
|
|
652
680
|
for (const day of last30Keys) {
|
|
653
681
|
const buckets = summary.days?.[day]?.byType
|
|
@@ -655,10 +683,23 @@ export function createUsageTracker({ store, config, pricing, now = Date.now, flu
|
|
|
655
683
|
for (const [type, totals] of Object.entries(buckets)) addTotals(bucketOf(byType, type, emptyTotals), safeTotals(totals))
|
|
656
684
|
}
|
|
657
685
|
const byModel = {}
|
|
658
|
-
|
|
659
|
-
|
|
660
|
-
|
|
661
|
-
|
|
686
|
+
const byProvider = {}
|
|
687
|
+
const byTrigger = {}
|
|
688
|
+
const last30Days = new Set(last30Keys)
|
|
689
|
+
for (const dayRuns of loaded.values()) {
|
|
690
|
+
for (const run of dayRuns) {
|
|
691
|
+
if (!Array.isArray(run.attempts)) continue
|
|
692
|
+
const trigger = typeof run.trigger === 'string' ? run.trigger : ''
|
|
693
|
+
for (const attempt of run.attempts) {
|
|
694
|
+
if (!isPlainObject(attempt) || !last30Days.has(dayKey(count(attempt.startedAt)))) continue
|
|
695
|
+
const delta = attemptDelta(attempt)
|
|
696
|
+
const model = typeof attempt.model === 'string' ? attempt.model : ''
|
|
697
|
+
const provider = typeof attempt.provider === 'string' ? attempt.provider : ''
|
|
698
|
+
if (model.length > 0) addTotals(bucketOf(byModel, model, emptyTotals), delta)
|
|
699
|
+
if (provider.length > 0) addTotals(bucketOf(byProvider, provider, emptyTotals), delta)
|
|
700
|
+
if (trigger.length > 0) addTotals(bucketOf(byTrigger, trigger, emptyTotals), delta)
|
|
701
|
+
}
|
|
702
|
+
}
|
|
662
703
|
}
|
|
663
704
|
|
|
664
705
|
const today = periodOf(totalsOver(todayKeys), attemptsIn(todayKeys))
|
|
@@ -685,6 +726,8 @@ export function createUsageTracker({ store, config, pricing, now = Date.now, flu
|
|
|
685
726
|
lifetime: periodOf(safeTotals(summary.lifetime), attempts),
|
|
686
727
|
byType,
|
|
687
728
|
byModel,
|
|
729
|
+
byProvider,
|
|
730
|
+
byTrigger,
|
|
688
731
|
series7: seriesOf(7),
|
|
689
732
|
series30: seriesOf(30),
|
|
690
733
|
warnings: {
|