@goodandready/dsh-cron 0.2.34 → 0.2.36

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -186,8 +186,11 @@ Every task picks its own runtime; non-LLM runtimes need no model and consume no
186
186
  ### 7. Cost Control: Fallback Model & Burn Guard
187
187
  * **Fallback Model** — A task can run on the cheap model by default and still finish on the strong one: set `fallbackModel` (and optionally `fallbackProvider`) and a failed run — `error` or `timeout` — is retried **once** on that model before the ordinary retry backoff applies. History records which model produced the result and whether the fallback was used, usage and cost of both attempts are summed, and the `{model}` template variable renders the model that finished the run. Only agent-mediated tasks (`llm`, `skill`, `workflow`) can use a fallback.
188
188
  * **Token & Cost Burn Guard** — Prevent runaway spending by configuring per-task limits: `costLimitUsd` (lifetime spend limit in USD), `dailyCostLimitUsd` (rolling 24-hour spend limit in USD), and `tokenLimit` (lifetime token limit). If a task exceeds any threshold, execution is halted, the task is automatically paused with `pausedReason` (`cost_limit_exceeded`, `daily_cost_limit_exceeded`, or `token_limit_exceeded`), and an alert notification is dispatched across all active channels.
189
+ * **Event-Driven Token & Cost Extraction** — Reads real token usage from DSH session stream events (`assistant/message`, `assistant/chunk` usage, and `assistant/attempt`), accounting for uncached input, cached reads, output tokens, and paid failed attempts across retries.
190
+ * **Rolling 24-Hour Cost Ledger** — Daily burn guard (`dailyCostLimitUsd`) maintains an independent rolling 24h cost ledger per task, preserved in `store.json`. Expenses remain fully counted across history archive rotations (beyond 100 runs) and survive daemon restarts.
189
191
 
190
192
  ### 8. Session Integration & Permissions
193
+ * **Real Assistant Output & Terminal Status Extraction** — Isolates current turn session events, extracts the actual assistant message text (excluding previous turn history in persistent sessions), and validates terminal turn status (`turn/end` errors or interruptions) to guarantee accurate reporting, chaining, and structured action execution.
191
194
  * **Per-task permission presets** — `default`, `read-only`, `workspace-write`, or `full` are applied to the task's agent session before the prompt runs.
192
195
  * **Session auto-archive** — isolated cron sessions are archived after each run (best-effort) so they do not clutter the chat list.
193
196
  * **History → session navigation** — every LLM run records its session; open it straight from the run history entry.
@@ -234,6 +237,8 @@ Prevent rogue processes from stacking concurrent duplicate executions:
234
237
  * **`skip`** (default): drops the overlapping run and records a `skipped` entry in the run history.
235
238
  * **`queue`**: queues the next execution and starts it as soon as the active job completes.
236
239
  * **`replace`**: aborts the active run via `AbortController` and launches a fresh execution.
240
+ * **Croner Overlap Policy Delegation** — Scheduled cron ticks fire without upstream suppression (`Croner protect: false`), allowing `skip` (with history logging), `queue` (delayed execution), and `replace` (clean abort) to govern recurring cron ticks and manual triggers consistently.
241
+ * **Context-Preserving Queues** — Both the global concurrency queue and task overlap queue retain full immutable execution context (`chainDepth`, `prevOutput`, `prevTaskId`, `prevStatus`, `prevCostUsd`), preventing data loss under load.
237
242
 
238
243
  If the daemon was offline at a scheduled time, the run is recorded as `missed` on startup, so gaps in the history stay visible.
239
244
 
package/README.ru.md CHANGED
@@ -186,8 +186,11 @@ cron({
186
186
  ### 7. Экономия: fallback-модель и защита бюджета (Burn Guard)
187
187
  * **Fallback-модель** — задача может идти на дешёвой модели по умолчанию и всё же завершиться на сильной: задайте `fallbackModel` (и при необходимости `fallbackProvider`), и сбойный запуск (`error` или `timeout`) один раз повторится на этой модели, прежде чем включится обычный retry с задержкой. В истории видно, какая модель произвела результат и был ли использован fallback; расход и стоимость обеих попыток суммируются; переменная шаблона `{model}` подставляет модель, завершившую запуск. Fallback доступен только агентским типам (`llm`, `skill`, `workflow`).
188
188
  * **Защита бюджета токенов и расходов (Burn Guard)** — предотвращение неконтролируемых трат через индивидуальные лимиты задачи: `costLimitUsd` (общий лимит расходов в USD), `dailyCostLimitUsd` (суточный лимит за последние 24 часа в USD) и `tokenLimit` (лимит суммарных токенов). При превышении любого порога выполнение прекращается, задача автоматически переводится в паузу с фиксацией `pausedReason` (`cost_limit_exceeded`, `daily_cost_limit_exceeded`, `token_limit_exceeded`), а во все активные каналы отправляется тревожное оповещение.
189
+ * **Учёт токенов и расходов на событиях сессии** — считывание фактического расхода токенов напрямую из событий потока сессии DSH (`assistant/message`, `assistant/chunk` с типом usage, `assistant/attempt`), с разделением некэшированного ввода, чтений кэша, генерации и оплаченных сбойных попыток провайдера при повторах.
190
+ * **Скользящий суточный реестр (24h Cost Ledger)** — суточный лимит `dailyCostLimitUsd` опирается на независимый реестр расходов задачи за последние 24 часа, сохраняемый в `store.json`. Расходы сохраняются при ротации архива истории (свыше 100 запусков) и восстанавливаются после перезапуска службы.
189
191
 
190
192
  ### 8. Интеграция сессий и права
193
+ * **Извлечение реального ответа ассистента и статуса хода** — выделение событий только текущего хода (без подмешивания истории прошлых ходов долговременных сессий), сборка итогового текста сообщения ассистента и валидация признака завершения (`turn/end` с ошибкой или прерыванием) для безошибочной классификации сбоев, цепочек задач и выполнения директив.
191
194
  * **Permission-пресеты на задачу** — `default`, `read-only`, `workspace-write` или `full` применяются к сессии агента перед запуском промпта.
192
195
  * **Автоархивация сессий** — изолированные cron-сессии архивируются после запуска (best-effort), не засоряя список чатов.
193
196
  * **История → сессия** — каждый LLM-запуск хранит свою сессию; открыть диалог можно прямо из записи истории.
@@ -226,6 +229,8 @@ cron({
226
229
  * **`skip`** (по умолчанию): накладывающийся запуск отбрасывается, в истории появляется запись `skipped`;
227
230
  * **`queue`**: следующий запуск ставится в очередь и стартует по завершении активного;
228
231
  * **`replace`**: активный запуск прерывается через `AbortController`, запускается свежий.
232
+ * **Прямая передача тиков расписания политикам наложения** — тики Croner поступают без подавления на уровне планировщика (`Croner protect: false`), гарантируя одинаковую работу политик `skip` (с записью в историю), `queue` (отложенный запуск) и `replace` (чистый abort) как для cron-расписания, так и для ручных вызовов.
233
+ * **Сохранение контекста очереди** — глобальная очередь конкурентности и очередь наложения сохраняют полный контекст выполнения (`chainDepth`, `prevOutput`, `prevTaskId`, `prevStatus`, `prevCostUsd`), предотвращая потерю параметров цепочек под нагрузкой.
229
234
 
230
235
  Если сервис был выключен в момент планового запуска, при старте в истории появится запись `missed` — пробелы в истории остаются видимыми.
231
236
 
package/README.zh.md CHANGED
@@ -186,8 +186,11 @@ cron({
186
186
  ### 7. 成本控制:回退模型与支出保护(Burn Guard)
187
187
  * **回退模型** —— 任务可以默认使用便宜模型,失败时改用更强模型完成:设置 `fallbackModel`(可选 `fallbackProvider`),失败(`error` 或 `timeout`)的运行会在该模型上重试一次,之后才进入常规重试退避。历史记录会标明最终产出结果的模型以及是否使用了回退,两次尝试的用量与成本都会累计,模板变量 `{model}` 渲染完成运行的模型。回退仅适用于智能体类型(`llm`、`skill`、`workflow`)。
188
188
  * **Token 与成本支出保护(Burn Guard)** —— 为任务配置严格预算上限:`costLimitUsd`(总支出美元上限)、`dailyCostLimitUsd`(24小时滚动支出上限)和 `tokenLimit`(Token总数上限)。一旦达到任一阈值,任务将自动暂停并记录 `pausedReason`(`cost_limit_exceeded`、`daily_cost_limit_exceeded` 或 `token_limit_exceeded`),同时向所有配置的通知渠道发送报警通知。
189
+ * **基于会话事件的 Token 与成本统计** —— 直接从 DSH 会话流式事件(`assistant/message`、`assistant/chunk` usage、`assistant/attempt`)中提取实际 Token 消耗,精确计入未命中输入、命中缓存读取、生成输出及重试过程中已计费的失败尝试。
190
+ * **滑动 24 小时成本账本(24h Cost Ledger)** —— 日耗保护(`dailyCostLimitUsd`)在 `store.json` 中为每个任务维护独立的滑动 24 小时成本记录。即使历史记录超过 100 条触发归档,24 小时内的所有花费依然完整保留并能跨服务重启持续生效。
189
191
 
190
192
  ### 8. 会话集成与权限
193
+ * **真实智能体输出与终端状态捕获** —— 隔离当前轮次的会话事件,提取最终真实的智能体回复文本(持久会话中自动排除以往历史轮次),并严格校验轮次终止状态(带有错误或中断的 `turn/end`),确保任务失败能准确反映到执行历史、链路调用及结构化指令中。
191
194
  * **按任务的权限预设** —— `default`、`read-only`、`workspace-write` 或 `full` 在提示词执行前应用于任务会话。
192
195
  * **会话自动归档** —— 隔离的 cron 会话在运行后自动归档(尽力而为),不干扰聊天列表。
193
196
  * **历史 → 会话** —— 每次 LLM 运行都会记录会话,可直接从历史记录打开对话。
@@ -226,6 +229,8 @@ cron({
226
229
  * **`skip`**(默认):丢弃重叠的运行,在历史中记录 `skipped`;
227
230
  * **`queue`**:将下一次运行排队,当前任务完成后自动开始;
228
231
  * **`replace`**:通过 `AbortController` 中止当前运行并启动新的执行。
232
+ * **调度器重叠策略直达** —— Croner 定时触发不再在上游被静默抑制(`Croner protect: false`),确保定时触发的重叠事件能够完整传递至调度器,严格执行 `skip`(记录历史)、`queue`(延迟排队)与 `replace`(优雅终止)。
233
+ * **队列上下文与链路深度延续** —— 全局并发限制队列与任务重叠队列均完整保留不可变执行参数(`chainDepth`、`prevOutput`、`prevTaskId`、`prevStatus`、`prevCostUsd`),防止高负载或排队时任务链路数据丢失。
229
234
 
230
235
  如果守护进程在计划时刻处于离线状态,启动时该次运行会被记录为 `missed`,历史空档始终可见。
231
236
 
@@ -363,7 +368,7 @@ bash deploy.sh verify [exact-version]
363
368
  ### 22. 自动化、任务链与可观测性包(v0.2.10,#137)
364
369
  - **Telegram 双向交互控制**:任务通知附带内嵌操作按钮(`🚀 立即运行`、`⏸️ 暂停/恢复`、`📋 最新日志`)。由 `POST /dsh-cron/telegram/webhook` 处理,严格鉴权 Chat ID 并调用 `answerCallbackQuery` 反馈。
365
370
  - **任务链上下文与动态变量插值**:配置 `onSuccess` 与 `onFailure` 下游触发器。父任务的执行结果与元数据自动传递给子任务,在 Shell 任务中提供 `$DSH_PREV_OUTPUT`、`$DSH_PREV_TASK_ID`、`$DSH_PREV_STATUS` 环境变量,在 LLM Prompt 中支持 `{{prev.output}}`(或 `{{prevOutput}}`)、`{{prev.taskId}}`、`{{prev.status}}` 占位符插值。Prompt 额外支持动态运行时时间与元数据变量:`{{date}}`、`{{time}}`、`{{datetime}}`、`{{timestamp}}`、`{{year}}`、`{{month}}`、`{{day}}`、`{{taskId}}`、`{{taskName}}`、`{{runCount}}`。内置最大 5 级深度递归防护,杜绝死循环。
366
- - **模型结构化动作指令**:自主分析任务可输出 JSON 指令触发级联任务(`trigger_task`)、定向告警(`notify`)或创建 Issue。受 `llmActionsEnabled: false` 严格保护。
371
+ - **模型结构化动作指令**:自主分析任务可输出 JSON 指令触发级联任务(`trigger_task`)、定向告警(`notify`)或创建 Issue。受 `llmActionsEnabled: false` 严格保护。`trigger_task` 指令共享全局链路深度上限(`chainDepth < 4`,最大 5 层调用),彻底阻断自调用死循环与 A → B → A 循环递归,拒绝原因完整记入历史与操作日志中。
367
372
  - **历史归档与延迟洞察**:REST 接口 `GET /dsh-cron/tasks/:id/archive`(支持分页)与 `GET /dsh-cron/tasks/:id/stats`;UI 任务卡片展示耗时彩色徽章(<5s 绿,<30s 黄,≥30s 红)。
368
373
  - **Prometheus 监控增强**:`/dsh-cron/metrics` 导出当前活动并发量 `dsh_cron_concurrent_running`、各任务 Token 计数器及成本预估指标。
369
374
 
package/lib/api.js CHANGED
@@ -218,7 +218,13 @@ async function createOrUpdateTask({ store, scheduler, req, res, apiToken }) {
218
218
  return;
219
219
  }
220
220
  }
221
- const task = store.set(buildTaskRecord(store, body, taskType, parsed));
221
+ let task;
222
+ try {
223
+ task = store.set(buildTaskRecord(store, body, taskType, parsed));
224
+ } catch (err) {
225
+ sendJson(res, 500, { ok: false, error: 'Failed to persist task: ' + err.message });
226
+ return;
227
+ }
222
228
  if (task.status === 'active') {
223
229
  scheduler.scheduleTask(task);
224
230
  } else {
@@ -345,7 +351,12 @@ async function handleItemPost({ store, scheduler, req, res, url, id, action }) {
345
351
  return true;
346
352
  }
347
353
  const copy = buildDuplicateTask(source, { id: randomUUID() });
348
- store.set(copy);
354
+ try {
355
+ store.set(copy);
356
+ } catch (err) {
357
+ sendJson(res, 500, { ok: false, error: 'Failed to persist duplicate task: ' + err.message });
358
+ return true;
359
+ }
349
360
  // A copy starts paused: it must never fire on its own before the user
350
361
  // reviews the schedule (a duplicated one-shot may point at a past time).
351
362
  scheduler.pauseTask(copy.id);
@@ -402,7 +413,7 @@ async function patchTask({ store, scheduler, req, res, id }) {
402
413
  allowCodeSwitch: req.headers[SCRIPT_CONFIRM_HEADER] === 'script',
403
414
  });
404
415
  if (!result.ok) {
405
- const status = result.notFound ? 404 : (result.needsConfirmation ? 403 : 400);
416
+ const status = result.status || (result.notFound ? 404 : (result.needsConfirmation ? 403 : 400));
406
417
  const error = result.needsConfirmation
407
418
  ? `Switching a task to the ${normalizeTaskType(body.type)} type over HTTP requires the ${SCRIPT_CONFIRM_HEADER}: script header`
408
419
  : result.error;
@@ -428,7 +439,12 @@ function deleteTask({ store, scheduler, res, id }) {
428
439
  } else if (scheduler) {
429
440
  scheduler.pauseTask(id);
430
441
  }
431
- store.delete(id);
442
+ try {
443
+ store.delete(id);
444
+ } catch (err) {
445
+ sendJson(res, 500, { ok: false, error: 'Failed to delete task: ' + err.message });
446
+ return;
447
+ }
432
448
  }
433
449
  sendJson(res, 200, { ok: true });
434
450
  }
package/lib/burn-guard.js CHANGED
@@ -14,10 +14,20 @@ import { bestEffort } from './best-effort.js';
14
14
  * @param {number} now
15
15
  * @returns {number}
16
16
  */
17
- export function calculateRollingCost(runs = [], windowMs = 86400000, now = Date.now()) {
17
+ export function calculateRollingCost(runsOrTask = [], windowMs = 86400000, now = Date.now()) {
18
18
  const cutoff = now - windowMs;
19
+ let items = [];
20
+ if (Array.isArray(runsOrTask)) {
21
+ items = runsOrTask;
22
+ } else if (runsOrTask && typeof runsOrTask === 'object') {
23
+ if (Array.isArray(runsOrTask.costLedger) && runsOrTask.costLedger.length > 0) {
24
+ items = runsOrTask.costLedger;
25
+ } else if (Array.isArray(runsOrTask.runs)) {
26
+ items = runsOrTask.runs;
27
+ }
28
+ }
19
29
  let total = 0;
20
- for (const r of runs) {
30
+ for (const r of items) {
21
31
  if (r && typeof r.at === 'number' && r.at >= cutoff) {
22
32
  total += Number(r.costUsd) || 0;
23
33
  }
@@ -32,14 +42,28 @@ export function calculateRollingCost(runs = [], windowMs = 86400000, now = Date.
32
42
  * @param {number} now
33
43
  * @returns {number}
34
44
  */
35
- export function calculateRollingTokens(runs = [], windowMs = 86400000, now = Date.now()) {
45
+ export function calculateRollingTokens(runsOrTask = [], windowMs = 86400000, now = Date.now()) {
36
46
  const cutoff = now - windowMs;
47
+ let items = [];
48
+ if (Array.isArray(runsOrTask)) {
49
+ items = runsOrTask;
50
+ } else if (runsOrTask && typeof runsOrTask === 'object') {
51
+ if (Array.isArray(runsOrTask.costLedger) && runsOrTask.costLedger.length > 0) {
52
+ items = runsOrTask.costLedger;
53
+ } else if (Array.isArray(runsOrTask.runs)) {
54
+ items = runsOrTask.runs;
55
+ }
56
+ }
37
57
  let total = 0;
38
- for (const r of runs) {
58
+ for (const r of items) {
39
59
  if (r && typeof r.at === 'number' && r.at >= cutoff) {
40
- const u = r.usage;
41
- const tokens = (u?.inputTokens || 0) + (u?.outputTokens || 0) + (u?.cacheReadTokens || 0);
42
- total += tokens;
60
+ if (typeof r.tokens === 'number') {
61
+ total += r.tokens;
62
+ } else {
63
+ const u = r.usage;
64
+ const tokens = (u?.inputTokens || 0) + (u?.outputTokens || 0) + (u?.cacheReadTokens || 0);
65
+ total += tokens;
66
+ }
43
67
  }
44
68
  }
45
69
  return total;
@@ -70,10 +94,11 @@ export function checkTaskBudgetLimits(task, runs = [], now = Date.now()) {
70
94
  }
71
95
  }
72
96
 
73
- // 2. Rolling 24h daily cost limit in USD
97
+ // 2. Rolling 24h daily cost limit in USD (#221)
74
98
  const dailyCostLimit = Number(task.dailyCostLimitUsd);
75
99
  if (dailyCostLimit > 0) {
76
- const dailyCost = calculateRollingCost(runs, 86400000, now);
100
+ const source = (Array.isArray(task.costLedger) && task.costLedger.length > 0) ? task.costLedger : runs;
101
+ const dailyCost = calculateRollingCost(source, 86400000, now);
77
102
  if (dailyCost >= dailyCostLimit) {
78
103
  return {
79
104
  exceeded: true,
package/lib/index.js CHANGED
@@ -198,6 +198,7 @@ export function apply(ctx, rawConfig) {
198
198
  return () => {
199
199
  if (heartbeatTimer) clearInterval(heartbeatTimer);
200
200
  scheduler.stopAll();
201
+ bestEffort('lifecycle-store-flush', () => store.flushSync(), logger);
201
202
  };
202
203
  }, 'dsh-cron: lifecycle');
203
204
 
@@ -65,7 +65,7 @@ function collectDirectives(obj, list) {
65
65
  /**
66
66
  * Execute parsed LLM action directives if enabled in plugin settings (#137).
67
67
  */
68
- export async function executeLlmActionDirectives({ directives, task, runInfo, scheduler, store, settings = {}, fetchFn = globalThis.fetch }) {
68
+ export async function executeLlmActionDirectives({ directives, task, runInfo, scheduler, store, settings = {}, fetchFn = globalThis.fetch, options = {} }) {
69
69
  if (!Array.isArray(directives) || directives.length === 0) return [];
70
70
  if (!settings.llmActionsEnabled) {
71
71
  return [];
@@ -78,11 +78,22 @@ export async function executeLlmActionDirectives({ directives, task, runInfo, sc
78
78
  if (type === 'trigger_task' || type === 'run_task') {
79
79
  const targetId = dir.taskId || dir.payload?.taskId;
80
80
  if (targetId && typeof scheduler.runNow === 'function') {
81
+ const currentDepth = options?.chainDepth || 0;
82
+ if (currentDepth >= 4) {
83
+ const limitMsg = `Structured action trigger_task recursion limit reached (depth ${currentDepth}). Target "${targetId}" not triggered.`;
84
+ logger.warn(`[dsh-cron] ${limitMsg}`);
85
+ results.push({ type, targetId, status: 'rejected', error: limitMsg, depth: currentDepth });
86
+ continue;
87
+ }
81
88
  await scheduler.runNow(targetId, {
89
+ chainDepth: currentDepth + 1,
82
90
  prevOutput: runInfo?.output || '',
83
91
  prevTaskId: task?.id || '',
92
+ prevStatus: runInfo?.status || 'success',
93
+ prevCostUsd: runInfo?.costUsd || 0,
94
+ prevDurationMs: runInfo?.durationMs || 0,
84
95
  });
85
- results.push({ type, targetId, status: 'triggered' });
96
+ results.push({ type, targetId, status: 'triggered', depth: currentDepth + 1 });
86
97
  }
87
98
  } else if (type === 'notify') {
88
99
  const msg = dir.message || dir.text || dir.payload?.message;
package/lib/runner.js CHANGED
@@ -324,7 +324,7 @@ export class SessionRunner {
324
324
 
325
325
  /** Create the session, drive one turn, and always clean up. */
326
326
  async _executeAgentTurn(task, prep) {
327
- let handle = null;
327
+ let handle = prep?.handle || null;
328
328
  let isResumed = false;
329
329
  try {
330
330
  // Resolve agent preset (#GH-1, #GH-2): scheduled llm sessions must join an agent preset
@@ -397,6 +397,13 @@ export class SessionRunner {
397
397
 
398
398
  this._applyPermissionPreset(task, handle);
399
399
  await prep.raceAbort(handle.agent.whenIdle());
400
+
401
+ // 1. Capture starting sequence before sending user message (#227, #228)
402
+ const session = handle.agent.session;
403
+ const startSeq = (session && typeof session.seq === 'number')
404
+ ? session.seq
405
+ : (Array.isArray(session?.log) ? session.log.length : 0);
406
+
400
407
  handle.agent.followup(prep.createUserMessage({
401
408
  content: [{ type: 'text', text: prep.agentPrompt }],
402
409
  source: { kind: 'plugin:dsh-cron', form: 'cron-execute' },
@@ -405,12 +412,73 @@ export class SessionRunner {
405
412
  if (typeof handle.agent.whenIdle === 'function') {
406
413
  await prep.raceAbort(handle.agent.whenIdle());
407
414
  }
408
- const usage = this._extractUsage(handle);
409
- const sessionNote = isResumed
410
- ? `resumed: ${prep.targetSessionId}`
411
- : `session: ${handle.agent.session?.id || 'cron'}`;
415
+
416
+ // 2. Snapshot events from this turn only (#227, #228)
417
+ let turnEvents = [];
418
+ if (typeof session?.snapshotEvents === 'function') {
419
+ turnEvents = session.snapshotEvents(startSeq) || [];
420
+ } else if (Array.isArray(session?.log)) {
421
+ turnEvents = session.log.slice(startSeq);
422
+ } else if (Array.isArray(handle.agent?.log)) {
423
+ turnEvents = handle.agent.log.slice(startSeq);
424
+ }
425
+
426
+ // 3. Terminal turn failure check (#227)
427
+ for (const ev of turnEvents) {
428
+ if (!ev) continue;
429
+ if (ev.type === 'turn/end' || ev.type === 'turn/error' || ev.type === 'turn/failure') {
430
+ const reason = ev.data?.reason || ev.reason;
431
+ const reasonKind = (reason && typeof reason === 'object') ? reason.kind : reason;
432
+ if (['error', 'interrupted', 'failed', 'aborted'].includes(reasonKind) || ev.data?.error || ev.error) {
433
+ const errMsg = ev.data?.error?.message || ev.error?.message || reason?.message || (typeof reason === 'string' ? reason : 'Agent turn ended with error');
434
+ const err = new Error(errMsg);
435
+ err.reason = reason;
436
+ err.turnFailed = true;
437
+ throw err;
438
+ }
439
+ }
440
+ }
441
+
442
+ // 4. Extract assistant message text from turn events (#227)
443
+ const messageTexts = [];
444
+ const chunkTexts = [];
445
+ for (const ev of turnEvents) {
446
+ if (!ev) continue;
447
+ if (ev.type === 'assistant/message') {
448
+ const msg = ev.data?.message || ev.data;
449
+ if (msg) {
450
+ if (typeof msg.content === 'string' && msg.content.trim()) {
451
+ messageTexts.push(msg.content.trim());
452
+ } else if (Array.isArray(msg.content)) {
453
+ const parts = msg.content
454
+ .filter((p) => p && p.type === 'text' && typeof p.text === 'string')
455
+ .map((p) => p.text)
456
+ .join('');
457
+ if (parts.trim()) messageTexts.push(parts.trim());
458
+ }
459
+ }
460
+ } else if (ev.type === 'assistant/chunk' && ev.data?.chunk) {
461
+ const chunk = ev.data.chunk;
462
+ if (chunk.type === 'text' && typeof chunk.text === 'string') {
463
+ chunkTexts.push(chunk.text);
464
+ } else if (chunk.type === 'text-delta' && typeof chunk.delta === 'string') {
465
+ chunkTexts.push(chunk.delta);
466
+ }
467
+ }
468
+ }
469
+ let outputText = messageTexts.length > 0 ? messageTexts.join('\n') : chunkTexts.join('');
470
+
471
+ // Fallback if no assistant message text was extracted from turn events
472
+ if (!outputText.trim()) {
473
+ const sessionNote = isResumed
474
+ ? `resumed: ${prep.targetSessionId}`
475
+ : `session: ${handle.agent.session?.id || 'cron'}`;
476
+ outputText = `[dsh-cron] Agent finished turn for task "${task.title}" (${sessionNote})`;
477
+ }
478
+
479
+ const usage = this._extractUsage(handle, turnEvents);
412
480
  return {
413
- output: `[dsh-cron] Agent finished turn for task "${task.title}" (${sessionNote})`,
481
+ output: outputText,
414
482
  usage,
415
483
  costUsd: estimateTokenCost(prep.model, usage),
416
484
  sessionId: handle.agent.session?.id || null,
@@ -433,17 +501,95 @@ export class SessionRunner {
433
501
  }
434
502
  }
435
503
 
436
- /** Token usage reported by the session, when the core exposes it. */
437
- _extractUsage(handle) {
504
+ /** Token usage reported by session events or the handle (#228). */
505
+ _extractUsage(handle, turnEvents = []) {
438
506
  const usage = { inputTokens: 0, outputTokens: 0, cacheReadTokens: 0 };
439
- bestEffort('extract-session-usage', () => {
440
- const sessionUsage = handle.agent.session?.usage || handle.agent.usage;
441
- if (sessionUsage) {
442
- usage.inputTokens = sessionUsage.inputTokens || sessionUsage.promptTokens || 0;
443
- usage.outputTokens = sessionUsage.outputTokens || sessionUsage.completionTokens || 0;
444
- usage.cacheReadTokens = sessionUsage.cacheReadTokens || sessionUsage.cachedTokens || 0;
507
+ let hasEventUsage = false;
508
+
509
+ bestEffort('extract-session-events-usage', () => {
510
+ const stepUsage = new Map();
511
+ const failedAttempts = [];
512
+
513
+ for (const ev of turnEvents) {
514
+ if (!ev || !ev.type) continue;
515
+
516
+ // assistant/chunk with usage type
517
+ if (ev.type === 'assistant/chunk' && ev.data?.chunk?.type === 'usage' && ev.data.chunk.usage) {
518
+ const u = ev.data.chunk.usage;
519
+ const key = `${ev.data.turn || 0}:${ev.data.step || 0}`;
520
+ if (!stepUsage.has(key)) {
521
+ stepUsage.set(key, {
522
+ inputTokens: Number(u.inputTokens || u.promptTokens || u.uncachedInputTokens) || 0,
523
+ outputTokens: Number(u.outputTokens || u.completionTokens) || 0,
524
+ cacheReadTokens: Number(u.cacheReadTokens || u.cachedTokens) || 0,
525
+ });
526
+ hasEventUsage = true;
527
+ }
528
+ }
529
+
530
+ // assistant/message finalized usage overwrites intermediate chunk usage for this step
531
+ if (ev.type === 'assistant/message') {
532
+ const u = ev.data?.usage || ev.data?.message?.usage;
533
+ if (u) {
534
+ const key = `${ev.data.turn || 0}:${ev.data.step || 0}`;
535
+ stepUsage.set(key, {
536
+ inputTokens: Number(u.inputTokens || u.promptTokens || u.uncachedInputTokens) || 0,
537
+ outputTokens: Number(u.outputTokens || u.completionTokens) || 0,
538
+ cacheReadTokens: Number(u.cacheReadTokens || u.cachedTokens) || 0,
539
+ });
540
+ hasEventUsage = true;
541
+ }
542
+ }
543
+
544
+ // assistant/attempt usage (e.g. paid failed attempts or retry attempts)
545
+ if (ev.type === 'assistant/attempt' && ev.data?.usage) {
546
+ const u = ev.data.usage;
547
+ if (ev.data.failed || ev.data.error || ev.data.status === 'failed') {
548
+ failedAttempts.push({
549
+ inputTokens: Number(u.inputTokens || u.promptTokens || u.uncachedInputTokens) || 0,
550
+ outputTokens: Number(u.outputTokens || u.completionTokens) || 0,
551
+ cacheReadTokens: Number(u.cacheReadTokens || u.cachedTokens) || 0,
552
+ });
553
+ hasEventUsage = true;
554
+ } else {
555
+ const key = `${ev.data.turn || 0}:${ev.data.step || 0}`;
556
+ if (!stepUsage.has(key)) {
557
+ stepUsage.set(key, {
558
+ inputTokens: Number(u.inputTokens || u.promptTokens || u.uncachedInputTokens) || 0,
559
+ outputTokens: Number(u.outputTokens || u.completionTokens) || 0,
560
+ cacheReadTokens: Number(u.cacheReadTokens || u.cachedTokens) || 0,
561
+ });
562
+ hasEventUsage = true;
563
+ }
564
+ }
565
+ }
566
+ }
567
+
568
+ if (hasEventUsage) {
569
+ for (const item of stepUsage.values()) {
570
+ usage.inputTokens += item.inputTokens;
571
+ usage.outputTokens += item.outputTokens;
572
+ usage.cacheReadTokens += item.cacheReadTokens;
573
+ }
574
+ for (const item of failedAttempts) {
575
+ usage.inputTokens += item.inputTokens;
576
+ usage.outputTokens += item.outputTokens;
577
+ usage.cacheReadTokens += item.cacheReadTokens;
578
+ }
445
579
  }
446
580
  });
581
+
582
+ if (!hasEventUsage) {
583
+ bestEffort('extract-session-usage-fallback', () => {
584
+ const sessionUsage = handle?.agent?.session?.usage || handle?.agent?.usage;
585
+ if (sessionUsage) {
586
+ usage.inputTokens = Number(sessionUsage.inputTokens || sessionUsage.promptTokens) || 0;
587
+ usage.outputTokens = Number(sessionUsage.outputTokens || sessionUsage.completionTokens) || 0;
588
+ usage.cacheReadTokens = Number(sessionUsage.cacheReadTokens || sessionUsage.cachedTokens) || 0;
589
+ }
590
+ });
591
+ }
592
+
447
593
  return usage;
448
594
  }
449
595
 
@@ -141,11 +141,17 @@ export async function executeWithFallback(scheduler, task, signal, options = {})
141
141
  /**
142
142
  * Begin a task run with concurrency and overlap policy checks.
143
143
  */
144
- export function beginRun(scheduler, task, taskId) {
144
+ export function beginRun(scheduler, task, taskId, options = {}) {
145
+ if (scheduler.isStopped) return null;
145
146
  if (!scheduler.running.has(taskId) && scheduler.maxConcurrent > 0 && scheduler.running.size >= scheduler.maxConcurrent) {
146
147
  if (task.overlapPolicy === 'queue') {
147
148
  logger.info(`[dsh-cron] Task "${task.title}" (${taskId}) queued by concurrency limit (${scheduler.maxConcurrent})`);
148
- scheduler.queue.push({ taskId, priority: task.priority !== undefined ? task.priority : 5, queuedAt: Date.now() });
149
+ scheduler.queue.push({
150
+ taskId,
151
+ options: { ...options },
152
+ priority: task.priority !== undefined ? task.priority : 5,
153
+ queuedAt: Date.now(),
154
+ });
149
155
  scheduler.queue.sort((a, b) => (a.priority - b.priority) || (a.queuedAt - b.queuedAt));
150
156
  return null;
151
157
  }
@@ -168,13 +174,23 @@ export function beginRun(scheduler, task, taskId) {
168
174
  }
169
175
  if (overlapPolicy === 'queue') {
170
176
  logger.info(`[dsh-cron] Task "${task.title}" (${taskId}) is already running, queueing next run (overlapPolicy: queue)`);
171
- active.queueCount = (active.queueCount || 0) + 1;
177
+ active.queuedRuns = active.queuedRuns || [];
178
+ active.queuedRuns.push({ options: { ...options }, queuedAt: Date.now() });
179
+ active.queueCount = active.queuedRuns.length;
172
180
  return null;
173
181
  }
174
182
  }
175
183
 
176
184
  const controller = new AbortController();
177
- const currentRun = { controller, startedAt: Date.now(), queueCount: 0 };
185
+ const runId = 'run_' + Math.random().toString(36).slice(2, 9) + '_' + Date.now();
186
+ const pendingOverlap = Array.isArray(options?.pendingOverlapQueue) ? options.pendingOverlapQueue : [];
187
+ const currentRun = {
188
+ id: runId,
189
+ controller,
190
+ startedAt: Date.now(),
191
+ queuedRuns: pendingOverlap,
192
+ queueCount: pendingOverlap.length,
193
+ };
178
194
  scheduler.running.set(taskId, currentRun);
179
195
  return currentRun;
180
196
  }
@@ -183,6 +199,7 @@ export function beginRun(scheduler, task, taskId) {
183
199
  * Run task by ID.
184
200
  */
185
201
  export async function executeTask(scheduler, taskId, options = {}) {
202
+ if (scheduler.isStopped) return null;
186
203
  const task = scheduler.store.get(taskId);
187
204
  if (!task) return null;
188
205
 
@@ -218,7 +235,7 @@ export async function executeTask(scheduler, taskId, options = {}) {
218
235
  };
219
236
  }
220
237
 
221
- const currentRun = scheduler.beginRun(task, taskId);
238
+ const currentRun = scheduler.beginRun(task, taskId, options);
222
239
  if (currentRun === null) return null;
223
240
 
224
241
  const preflight = await executePreflight(task);
@@ -328,19 +345,32 @@ export async function deliverNotifications(task, runInfo, { store, resolveSecret
328
345
  /**
329
346
  * Post-run bookkeeping: one-shot completion, next-run timestamp, retry backoff, draining.
330
347
  */
331
- export function finishRun(scheduler, task, taskId, status, queuedCount) {
348
+ export function finishRun(scheduler, taskSnapshot, taskId, status, queuedCount, meta = {}) {
332
349
  scheduler.countRun(status);
350
+
351
+ // If this run was superseded by a newer run or scheduler is stopped, skip updating task
352
+ if (meta.isCurrentRun === false || scheduler.isStopped) {
353
+ return;
354
+ }
355
+
356
+ // Refetch live task from store to avoid resurrecting deleted task or overwriting edits (#215)
357
+ const current = scheduler.store.get(taskId);
358
+ if (!current) {
359
+ logger.info(`[dsh-cron] finishRun: task "${taskId}" no longer exists in store; skipping completion write`);
360
+ return;
361
+ }
362
+
333
363
  try {
334
- if (task.oneShot) {
335
- task.status = 'completed';
336
- task.nextRunAt = null;
337
- scheduler.store.set(task);
364
+ if (current.oneShot) {
365
+ current.status = 'completed';
366
+ current.nextRunAt = null;
367
+ scheduler.store.set(current);
338
368
  } else {
339
369
  const job = scheduler.jobs.get(taskId);
340
370
  if (job) {
341
371
  const next = job.nextRun();
342
- task.nextRunAt = next ? next.getTime() : null;
343
- scheduler.store.set(task);
372
+ current.nextRunAt = next ? next.getTime() : null;
373
+ scheduler.store.set(current);
344
374
  }
345
375
  }
346
376
  } catch (err) {
@@ -349,32 +379,42 @@ export function finishRun(scheduler, task, taskId, status, queuedCount) {
349
379
 
350
380
  try {
351
381
  const failed = status === 'error' || status === 'timeout';
352
- const maxRetries = Number(task.maxRetries) > 0 ? Number(task.maxRetries) : 0;
353
- if (failed && maxRetries > 0 && (task.attempts || 0) < maxRetries && task.status !== 'paused') {
354
- task.attempts = (task.attempts || 0) + 1;
355
- const backoffMs = (Number(task.retryBackoffMs) > 0 ? Number(task.retryBackoffMs) : 30000) * Math.pow(2, task.attempts - 1);
356
- const attempt = task.attempts;
357
- scheduler.store.set(task);
382
+ const maxRetries = Number(current.maxRetries) > 0 ? Number(current.maxRetries) : 0;
383
+ if (failed && maxRetries > 0 && (current.attempts || 0) < maxRetries && current.status !== 'paused' && !scheduler.isStopped) {
384
+ current.attempts = (current.attempts || 0) + 1;
385
+ const backoffMs = (Number(current.retryBackoffMs) > 0 ? Number(current.retryBackoffMs) : 30000) * Math.pow(2, current.attempts - 1);
386
+ const attempt = current.attempts;
387
+ scheduler.store.set(current);
358
388
  scheduler.clearRetryTimer(taskId);
359
389
  scheduler.retryTimers.set(taskId, setTimeout(() => {
360
390
  scheduler.retryTimers.delete(taskId);
361
- scheduler.runTask(taskId, { isRetry: true });
391
+ if (!scheduler.isStopped) {
392
+ scheduler.runTask(taskId, { isRetry: true });
393
+ }
362
394
  }, backoffMs));
363
- logger.info(`[dsh-cron] retry ${attempt}/${maxRetries} for task "${task.title}" in ${backoffMs}ms`);
364
- } else if (!failed && task.attempts) {
365
- task.attempts = 0;
366
- scheduler.store.set(task);
367
- } else if (failed && maxRetries > 0 && (task.attempts || 0) >= maxRetries) {
368
- task.attempts = 0;
369
- scheduler.store.set(task);
395
+ logger.info(`[dsh-cron] retry ${attempt}/${maxRetries} for task "${current.title}" in ${backoffMs}ms`);
396
+ } else if (!failed && current.attempts) {
397
+ current.attempts = 0;
398
+ scheduler.store.set(current);
399
+ } else if (failed && maxRetries > 0 && (current.attempts || 0) >= maxRetries) {
400
+ current.attempts = 0;
401
+ scheduler.store.set(current);
370
402
  }
371
403
  } catch (err) {
372
404
  logger.error('[dsh-cron] retry bookkeeping failed for task', taskId + ':', err.message);
373
405
  }
374
406
 
375
- if (queuedCount > 0) {
407
+ if (queuedCount > 0 && !scheduler.isStopped) {
408
+ const queuedRuns = Array.isArray(meta?.queuedRuns) ? [...meta.queuedRuns] : [];
409
+ const nextItem = queuedRuns.shift();
410
+ const nextOptions = {
411
+ ...(nextItem?.options || {}),
412
+ ...(queuedRuns.length > 0 ? { pendingOverlapQueue: queuedRuns } : {}),
413
+ };
376
414
  setImmediate(() => {
377
- scheduler.runTask(taskId);
415
+ if (!scheduler.isStopped) {
416
+ scheduler.runTask(taskId, nextOptions);
417
+ }
378
418
  });
379
419
  }
380
420
  }
@@ -440,24 +480,34 @@ export async function handleCompleteRun(scheduler, task, taskId, currentRun, out
440
480
  }
441
481
 
442
482
  const finishedRun = scheduler.running.get(taskId);
443
- if (finishedRun === currentRun) scheduler.running.delete(taskId);
444
- const queuedCount = finishedRun?.queueCount || 0;
483
+ const isCurrentRun = (finishedRun === currentRun);
484
+ if (isCurrentRun) scheduler.running.delete(taskId);
485
+ const queuedRuns = isCurrentRun ? (finishedRun?.queuedRuns || []) : [];
486
+ const queuedCount = isCurrentRun ? (finishedRun?.queueCount || 0) : 0;
445
487
 
446
488
  if (!silent.skipped) await scheduler.deliverNotifications(task, runInfo);
447
- scheduler.finishRun(task, taskId, outcome.status, queuedCount);
489
+ scheduler.finishRun(task, taskId, outcome.status, queuedCount, {
490
+ isCurrentRun,
491
+ runId: currentRun?.id,
492
+ queuedRuns,
493
+ });
494
+
495
+ if (scheduler.isStopped) return;
448
496
 
449
497
  const guard = await checkAndApplyBurnGuard(scheduler, taskId, outcome);
450
498
  if (guard && guard.exceeded) return;
451
499
 
452
- while (scheduler.queue.length > 0 && scheduler.running.size < scheduler.maxConcurrent) {
500
+ while (!scheduler.isStopped && scheduler.queue.length > 0 && scheduler.running.size < scheduler.maxConcurrent) {
453
501
  const nextItem = scheduler.queue.shift();
454
502
  if (!nextItem) break;
455
503
  const queuedTask = scheduler.store.get(nextItem.taskId);
456
- if (queuedTask && queuedTask.status === 'active') {
504
+ if (queuedTask && queuedTask.status === 'active' && !scheduler.isStopped) {
457
505
  setImmediate(() => {
458
- scheduler.runTask(nextItem.taskId).catch((runErr) => {
459
- bestEffort('drain-queued-task', () => {}, scheduler.logger);
460
- });
506
+ if (!scheduler.isStopped) {
507
+ scheduler.runTask(nextItem.taskId, nextItem.options || {}).catch((runErr) => {
508
+ bestEffort('drain-queued-task', () => {}, scheduler.logger);
509
+ });
510
+ }
461
511
  });
462
512
  break;
463
513
  }
@@ -470,14 +520,18 @@ export async function handleCompleteRun(scheduler, task, taskId, currentRun, out
470
520
  if (settings.llmActionsEnabled) {
471
521
  const directives = parseLlmActionDirectives(outcome.output);
472
522
  if (directives.length > 0) {
473
- await executeLlmActionDirectives({
523
+ const actionResults = await executeLlmActionDirectives({
474
524
  directives,
475
525
  task,
476
526
  runInfo,
477
527
  scheduler,
478
528
  store: scheduler.store,
479
529
  settings,
530
+ options,
480
531
  });
532
+ if (actionResults && actionResults.length > 0) {
533
+ runInfo.actionResults = actionResults;
534
+ }
481
535
  }
482
536
  }
483
537
  } catch (actErr) {
package/lib/scheduler.js CHANGED
@@ -254,6 +254,8 @@ export class TaskScheduler {
254
254
  this.queue = [];
255
255
  this.heartbeatTimer = null;
256
256
  this.logger = opts.logger || null;
257
+ this.isStopped = false;
258
+ this.generation = 0;
257
259
  }
258
260
 
259
261
  isRunning(taskId) {
@@ -290,6 +292,8 @@ export class TaskScheduler {
290
292
  }
291
293
 
292
294
  start() {
295
+ this.isStopped = false;
296
+ this.generation = (this.generation || 0) + 1;
293
297
  const tasks = this.store.list();
294
298
  const now = Date.now();
295
299
  for (const task of tasks) {
@@ -328,7 +332,13 @@ export class TaskScheduler {
328
332
  }
329
333
 
330
334
  stopAll() {
335
+ this.isStopped = true;
336
+ this.generation = (this.generation || 0) + 1;
331
337
  this.stopHeartbeatWatcher();
338
+
339
+ // Clear concurrency queue (#217)
340
+ this.queue = [];
341
+
332
342
  for (const [, job] of this.jobs.entries()) {
333
343
  bestEffort('stop-job', () => job.stop(), this.logger);
334
344
  }
@@ -348,6 +358,10 @@ export class TaskScheduler {
348
358
  bestEffort('abort-running', () => run.controller.abort(new Error('Scheduler stopped')), this.logger);
349
359
  }
350
360
  this.running.clear();
361
+
362
+ if (this.store && typeof this.store.flushSync === 'function') {
363
+ bestEffort('store-flush-on-stop', () => this.store.flushSync(), this.logger);
364
+ }
351
365
  }
352
366
 
353
367
  clearScheduled(taskId) {
@@ -379,7 +393,6 @@ export class TaskScheduler {
379
393
  scheduleCron(task, parsed) {
380
394
  const timezone = task.timezone || this.defaultTimezone || undefined;
381
395
  const job = new Cron(parsed.cronPattern, {
382
- protect: true,
383
396
  sloppyRanges: true,
384
397
  catch: (err) => logger.error(`[dsh-cron] scheduled run of "${task.title}" (${task.id}) failed:`, (err && err.message) || err),
385
398
  ...(timezone ? { timezone } : {}),
@@ -412,8 +425,8 @@ export class TaskScheduler {
412
425
  }
413
426
  }
414
427
 
415
- beginRun(task, taskId) {
416
- return beginRun(this, task, taskId);
428
+ beginRun(task, taskId, options = {}) {
429
+ return beginRun(this, task, taskId, options);
417
430
  }
418
431
 
419
432
  countRun(status) {
@@ -451,6 +464,7 @@ export class TaskScheduler {
451
464
  }
452
465
 
453
466
  async runTask(taskId, options = {}) {
467
+ if (this.isStopped) return null;
454
468
  return executeTask(this, taskId, options);
455
469
  }
456
470
 
@@ -500,8 +514,8 @@ export class TaskScheduler {
500
514
  });
501
515
  }
502
516
 
503
- finishRun(task, taskId, status, queuedCount) {
504
- return finishRun(this, task, taskId, status, queuedCount);
517
+ finishRun(task, taskId, status, queuedCount, meta) {
518
+ return finishRun(this, task, taskId, status, queuedCount, meta);
505
519
  }
506
520
 
507
521
  pauseTask(taskId, reason) {
package/lib/store.js CHANGED
@@ -53,7 +53,22 @@ export class TaskStore {
53
53
  this.tasks.clear();
54
54
  this.history.clear();
55
55
  if (Array.isArray(data.tasks)) {
56
+ const now = Date.now();
57
+ const cutoff24h = now - 86400000;
56
58
  for (const t of data.tasks) {
59
+ if (Array.isArray(t.costLedger) && t.costLedger.length > 0) {
60
+ t.costLedger = t.costLedger.filter(e => e && typeof e.at === 'number' && e.at >= cutoff24h);
61
+ } else {
62
+ // Seed from active history if present (#221)
63
+ const hist = (data.history && Array.isArray(data.history[t.id])) ? data.history[t.id] : [];
64
+ t.costLedger = hist
65
+ .filter(r => r && typeof r.at === 'number' && r.at >= cutoff24h)
66
+ .map(r => ({
67
+ at: r.at,
68
+ costUsd: Number(r.costUsd) || 0,
69
+ tokens: (r.usage?.inputTokens || 0) + (r.usage?.outputTokens || 0) + (r.usage?.cacheReadTokens || 0),
70
+ }));
71
+ }
57
72
  this.tasks.set(t.id, t);
58
73
  }
59
74
  }
@@ -112,7 +127,11 @@ export class TaskStore {
112
127
  if (this._saveTimer) return;
113
128
  this._saveTimer = setTimeout(() => {
114
129
  this._saveTimer = null;
115
- this.save();
130
+ try {
131
+ this.save();
132
+ } catch (err) {
133
+ logger.error('[dsh-cron] debounced save failed:', err.message);
134
+ }
116
135
  }, delay);
117
136
  if (typeof this._saveTimer.unref === 'function') {
118
137
  this._saveTimer.unref();
@@ -160,6 +179,7 @@ export class TaskStore {
160
179
  }
161
180
  } catch (err) {
162
181
  logger.error('[dsh-cron] store save error:', err.message);
182
+ throw err;
163
183
  }
164
184
  }
165
185
 
@@ -254,6 +274,7 @@ export class TaskStore {
254
274
  else if (typeof value === 'string') patch[key] = value.trim();
255
275
  }
256
276
 
277
+ const prevSettings = this.settings;
257
278
  this.settings = {
258
279
  ...this.settings,
259
280
  ...patch,
@@ -265,7 +286,12 @@ export class TaskStore {
265
286
  onlyOnFailure: patch.onlyOnFailure !== undefined ? Boolean(patch.onlyOnFailure) : Boolean(this.settings.onlyOnFailure),
266
287
  llmActionsEnabled: patch.llmActionsEnabled !== undefined ? Boolean(patch.llmActionsEnabled) : Boolean(this.settings.llmActionsEnabled),
267
288
  };
268
- this.save();
289
+ try {
290
+ this.save();
291
+ } catch (err) {
292
+ this.settings = prevSettings;
293
+ throw err;
294
+ }
269
295
  return this.getSettings();
270
296
  }
271
297
 
@@ -292,7 +318,8 @@ export class TaskStore {
292
318
  set(task) {
293
319
  const now = Date.now();
294
320
  const id = task.id || ('cron_' + Math.random().toString(36).slice(2, 9));
295
- const prev = this.tasks.get(id) || {};
321
+ const existing = this.tasks.get(id);
322
+ const prev = existing || {};
296
323
  const record = {
297
324
  ...prev,
298
325
  ...task,
@@ -337,7 +364,13 @@ export class TaskStore {
337
364
  nextRunAt: task.nextRunAt !== undefined ? task.nextRunAt : (prev.nextRunAt || null),
338
365
  };
339
366
  this.tasks.set(id, record);
340
- this.save();
367
+ try {
368
+ this.save();
369
+ } catch (err) {
370
+ if (existing) this.tasks.set(id, existing);
371
+ else this.tasks.delete(id);
372
+ throw err;
373
+ }
341
374
  return record;
342
375
  }
343
376
 
@@ -345,13 +378,25 @@ export class TaskStore {
345
378
  const task = this.tasks.get(id);
346
379
  if (!task) return null;
347
380
  const now = Date.now();
381
+ const prevPing = task.lastPingAt;
382
+ const prevAlerted = task.heartbeatAlerted;
383
+ const prevUpdated = task.updatedAt;
384
+ const prevStatus = task.lastStatus;
348
385
  task.lastPingAt = now;
349
386
  task.heartbeatAlerted = false;
350
387
  task.updatedAt = now;
351
388
  if (task.lastStatus === 'missed') {
352
389
  task.lastStatus = 'active';
353
390
  }
354
- this.save();
391
+ try {
392
+ this.save();
393
+ } catch (err) {
394
+ task.lastPingAt = prevPing;
395
+ task.heartbeatAlerted = prevAlerted;
396
+ task.updatedAt = prevUpdated;
397
+ task.lastStatus = prevStatus;
398
+ throw err;
399
+ }
355
400
  const interval = (task.heartbeatIntervalSeconds || 0) + (task.gracePeriodSeconds || 300);
356
401
  return {
357
402
  ok: true,
@@ -363,10 +408,19 @@ export class TaskStore {
363
408
  }
364
409
 
365
410
  delete(id) {
366
- const existed = this.tasks.delete(id);
411
+ const existing = this.tasks.get(id);
412
+ if (!existing) return false;
413
+ const existingHistory = this.history.get(id);
414
+ this.tasks.delete(id);
367
415
  this.history.delete(id);
368
- if (existed) this.save();
369
- return existed;
416
+ try {
417
+ this.save();
418
+ } catch (err) {
419
+ this.tasks.set(id, existing);
420
+ if (existingHistory) this.history.set(id, existingHistory);
421
+ throw err;
422
+ }
423
+ return true;
370
424
  }
371
425
 
372
426
  recordRun(id, runInfo) {
@@ -385,6 +439,18 @@ export class TaskStore {
385
439
  task.totalCostUsd = Number(((task.totalCostUsd || 0) + costUsd).toFixed(6));
386
440
  task.updatedAt = Date.now();
387
441
 
442
+ // Rolling 24h cost ledger preserved across archive rotation and restarts (#221)
443
+ if (!Array.isArray(task.costLedger)) {
444
+ task.costLedger = [];
445
+ }
446
+ task.costLedger.push({
447
+ at: task.lastRunAt,
448
+ costUsd,
449
+ tokens: runTokens,
450
+ });
451
+ const cutoff24h = Date.now() - 86400000;
452
+ task.costLedger = task.costLedger.filter(e => e && typeof e.at === 'number' && e.at >= cutoff24h);
453
+
388
454
  const runs = this.history.get(id) || [];
389
455
  runs.unshift({
390
456
  id: 'run_' + Math.random().toString(36).slice(2, 9),
package/lib/task-patch.js CHANGED
@@ -114,10 +114,14 @@ export function applyTaskPatch({ store, scheduler, id, body, allowCodeSwitch = f
114
114
  }
115
115
  const built = buildTaskPatch(current, body);
116
116
  if (!built.ok) return built;
117
- const task = store.set({ ...current, ...built.patch, id });
118
- if (task.status === 'active') scheduler.scheduleTask(task);
119
- else scheduler.pauseTask(id);
120
- return { ok: true, task, patch: built.patch };
117
+ try {
118
+ const task = store.set({ ...current, ...built.patch, id });
119
+ if (task.status === 'active') scheduler.scheduleTask(task);
120
+ else scheduler.pauseTask(id);
121
+ return { ok: true, task, patch: built.patch };
122
+ } catch (err) {
123
+ return { ok: false, error: 'Failed to persist task patch: ' + err.message, status: 500 };
124
+ }
121
125
  }
122
126
 
123
127
  /** Human-readable summary of what changed, for tool output and logs. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@goodandready/dsh-cron",
3
- "version": "0.2.34",
3
+ "version": "0.2.36",
4
4
  "description": "Background automation runner for DSH: isolated agent runs, script/HTTP/SSH/Docker runtimes, cost guard, notifications, heartbeats.",
5
5
  "type": "module",
6
6
  "main": "lib/index.js",