@miphamai/cli 0.81.5 → 0.81.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/README.md +9 -9
  2. package/bin/daemon.ts +7 -32
  3. package/bin/mipham.ts +43 -29
  4. package/package.json +5 -2
  5. package/skills/standard/mipham-code-setup.SKILL.md +3 -3
  6. package/src/agent/sub-agent.ts +12 -1
  7. package/src/commands/project.ts +92 -12
  8. package/src/config/keys-manager.ts +3 -3
  9. package/src/config/loader.ts +82 -1
  10. package/src/core/context.ts +10 -2
  11. package/src/core/engine.ts +32 -4
  12. package/src/core/metrics.ts +8 -0
  13. package/src/core/paths.ts +79 -0
  14. package/src/core/permission-rules.ts +145 -13
  15. package/src/core/permission.ts +3 -0
  16. package/src/core/session-log.ts +11 -2
  17. package/src/daemon/engine-capabilities.ts +131 -0
  18. package/src/daemon/index.ts +4 -1
  19. package/src/daemon/launch.ts +287 -0
  20. package/src/daemon/remote-engine.ts +5 -0
  21. package/src/daemon/server.ts +9 -0
  22. package/src/daemon/session-worker.ts +7 -4
  23. package/src/i18n-core/locales/en-US.json +6 -7
  24. package/src/i18n-core/locales/zh-CN.json +6 -7
  25. package/src/index.tsx +79 -0
  26. package/src/mcp/client.ts +109 -8
  27. package/src/providers/anthropic.ts +2 -0
  28. package/src/shared/package-info.ts +1 -1
  29. package/src/shared/types.ts +15 -0
  30. package/src/skills/bundled-skills.ts +1 -1
  31. package/src/telemetry/consent.ts +209 -0
  32. package/src/telemetry/crash.ts +197 -0
  33. package/src/telemetry/endpoint.ts +82 -0
  34. package/src/telemetry/index.ts +153 -0
  35. package/src/telemetry/payload.ts +141 -0
  36. package/src/telemetry/queue.ts +95 -0
  37. package/src/telemetry/redact.ts +127 -0
  38. package/src/telemetry/transport.ts +81 -0
  39. package/src/tools/agent/workflow.ts +11 -4
  40. package/src/tools/exec/bash.ts +6 -4
  41. package/src/tools/exec/enter-worktree.ts +6 -5
  42. package/src/tools/exec/exit-worktree.ts +10 -5
  43. package/src/tools/exec/git.ts +18 -8
  44. package/src/tools/system/config.ts +3 -3
  45. package/src/ui/app.tsx +47 -11
  46. package/src/ui/commands.ts +159 -34
  47. package/src/workflow/primitives/agent.ts +4 -2
  48. package/src/core/task-runner-tasks.json +0 -14
  49. package/src/core/task-runner.ts +0 -163
  50. package/src/skills/mipham/runtime.ts +0 -66
  51. package/src/skills/standard/runtime.ts +0 -62
@@ -16,6 +16,7 @@ import { McpClient } from '../mcp/client'
16
16
  import { buildCapabilityReport } from '../core/capability-inventory'
17
17
  import { InstructionsLoader } from '../core/instructions'
18
18
  import { findDerivableSections, DERIVABLE_HINTS } from '../core/claude-md-audit'
19
+ import { worktreeRoot, workflowScriptDir, workflowScriptDirs } from '../core/paths.ts'
19
20
  import { fixDoctor, fixConfig, fixCache, selectRepoClaudeFiles } from '../core/fix'
20
21
  import { fixCodeTarget } from '../core/fix-code'
21
22
  import { homedir } from 'node:os'
@@ -56,6 +57,15 @@ import { NPM_UPDATE_COMMAND, PACKAGE_VERSION, COAUTHOR_TRAILER } from '../shared
56
57
  import { getPreference } from '../config/preferences'
57
58
  import { loadCrossSessionConfig, tryRestoreFromBackup } from '../config/loader'
58
59
  import { getMemoryManager } from '../core/memory/memory-loader'
60
+ import {
61
+ resolveTelemetry,
62
+ readTelemetrySettings,
63
+ setTelemetryEnabled,
64
+ setTelemetryEndpoint,
65
+ resetInstallId,
66
+ } from '../telemetry/consent'
67
+ import { getTelemetryConsent, enableTelemetryNow } from '../telemetry/index'
68
+ import { NO_ENDPOINT } from '../telemetry/endpoint'
59
69
  import { stripIndent } from './strip-indent.js'
60
70
  import { createT } from '../i18n-core/t'
61
71
  import type { TranslationMap } from '../i18n-core/types'
@@ -251,7 +261,7 @@ const helpCmd: CommandHandler = (ctx) => {
251
261
  /init Initialize .mipham config
252
262
  /setup Guided project setup wizard
253
263
  /recommend Analyze project + recommend setup
254
- /permissions Show permission settings
264
+ /permissions Show or persist permission rules
255
265
  /add-dir <dir> Add workspace directory
256
266
  /security Security review checklist
257
267
  /audit Same as /security
@@ -2076,8 +2086,8 @@ const todosCmd: CommandHandler = (_ctx, args) => {
2076
2086
  return { content: t('commands.todos.usage_create') }
2077
2087
  }
2078
2088
  return {
2079
- content: `${t('commands.todos.create_title')}\n\nCreating task: "${title.trim()}"\n\nPassing to AI for structured task creation with TaskCreate...`,
2080
- forwardToAI: `Create a new task using TaskCreate with subject "${title.trim()}". Set a clear description and activeForm.`,
2089
+ content: `${t('commands.todos.create_title')}\n\nCreating task: "${title.trim()}"\n\nPassing to AI for structured task creation with the Task tool...`,
2090
+ forwardToAI: `Create a new task using the Task tool with action "create" and subject "${title.trim()}". Set a clear description and activeForm.`,
2081
2091
  }
2082
2092
  }
2083
2093
 
@@ -2092,7 +2102,7 @@ const todosCmd: CommandHandler = (_ctx, args) => {
2092
2102
  ${t('commands.todos.item_create')}
2093
2103
  `,
2094
2104
  forwardToAI:
2095
- 'Use TaskList to show all current tasks. Present them in a clear summary grouped by status (pending/in_progress/completed). If there are no tasks, suggest creating one.',
2105
+ 'Use the Task tool with action "list" to show all current tasks. Present them in a clear summary grouped by status (pending/in_progress/completed). If there are no tasks, suggest creating one.',
2096
2106
  }
2097
2107
  }
2098
2108
 
@@ -2104,7 +2114,8 @@ const todosCmd: CommandHandler = (_ctx, args) => {
2104
2114
 
2105
2115
  ${t('commands.todos.default_body')}
2106
2116
  `,
2107
- forwardToAI: 'Use TaskList to show all current tasks, then present them clearly.',
2117
+ forwardToAI:
2118
+ 'Use the Task tool with action "list" to show all current tasks, then present them clearly.',
2108
2119
  }
2109
2120
  }
2110
2121
 
@@ -2192,7 +2203,7 @@ const goalCmd: CommandHandler = (ctx, args) => {
2192
2203
  if (decompose) {
2193
2204
  lines.push(t('commands.goal.decompose_enabled'))
2194
2205
  // Decompose by creating initial subtasks
2195
- const decomposeMsg = `Break down this goal into 3-5 subtasks: "${goal}". For each subtask, use TaskCreate with the subject and description. Mark each as blocked by the previous one to create a dependency chain.`
2206
+ const decomposeMsg = `Break down this goal into 3-5 subtasks: "${goal}". For each subtask, use the Task tool with action "create", giving the subject and description. Mark each as blocked by the previous one to create a dependency chain.`
2196
2207
  return {
2197
2208
  content: lines.join('\n'),
2198
2209
  forwardToAI: decomposeMsg,
@@ -2813,12 +2824,12 @@ const tasksCmd: CommandHandler = (ctx) => {
2813
2824
  const c = ctx.engine.getContext()
2814
2825
  const msgs = c.getMessages()
2815
2826
 
2816
- // Scan for task-related tool uses in message history
2827
+ // Scan for task-related tool uses in message history.
2828
+ // 任务工具只有一个 `Task`,动作走 action 参数 —— 过滤条件必须按真实工具名匹配,
2829
+ // 否则计数恒为 0,「已检测到 N 次任务操作」这条分支永远不可达。
2817
2830
  const toolUses = msgs.flatMap((m) => {
2818
2831
  if (Array.isArray(m.content)) {
2819
- return m.content.filter(
2820
- (b) => b.type === 'tool_use' && ['TaskCreate', 'TaskUpdate', 'TaskList'].includes(b.name),
2821
- )
2832
+ return m.content.filter((b) => b.type === 'tool_use' && b.name === 'Task')
2822
2833
  }
2823
2834
  return []
2824
2835
  })
@@ -2830,12 +2841,13 @@ const tasksCmd: CommandHandler = (ctx) => {
2830
2841
  ${toolUses.length > 0 ? t('commands.task_list.detected', { count: String(toolUses.length) }) : t('commands.task_list.no_tasks')}
2831
2842
 
2832
2843
  ${t('commands.task_list.reference')}
2833
- TaskCreate — create a new task
2834
- TaskList — list all tasks
2835
- TaskUpdate — update task status
2836
- TaskGet — get task details
2837
- TaskOutput — get background task output
2838
- TaskStop — stop a running task
2844
+ Task(action: "create") — create a new task
2845
+ Task(action: "list") — list all tasks
2846
+ Task(action: "update") — update task status
2847
+ Task(action: "get") — get task details
2848
+ Task(action: "delete") — delete a task
2849
+ Task(action: "output") — get background task output
2850
+ Task(action: "stop") — stop a running task
2839
2851
 
2840
2852
  ${t('commands.task_list.legacy_hint')}
2841
2853
  `,
@@ -3807,6 +3819,72 @@ const hooksCmd: CommandHandler = async (ctx) => {
3807
3819
  return { content: lines.join('\n') }
3808
3820
  }
3809
3821
 
3822
+ const telemetryCmd: CommandHandler = async (ctx, args) => {
3823
+ const sub = (args[0] ?? 'status').toLowerCase()
3824
+
3825
+ if (sub === 'on' || sub === 'off') {
3826
+ const enable = sub === 'on'
3827
+ setTelemetryEnabled(enable)
3828
+ if (enable) enableTelemetryNow()
3829
+ return {
3830
+ content: enable
3831
+ ? '✓ Telemetry enabled. Anonymous usage counts will be sent to the configured endpoint.'
3832
+ : '✓ Telemetry disabled. Nothing is collected and no queue file is written.',
3833
+ }
3834
+ }
3835
+
3836
+ if (sub === 'reset-id') {
3837
+ const id = resetInstallId()
3838
+ return { content: `✓ New anonymous install id: ${id}` }
3839
+ }
3840
+
3841
+ if (sub === 'endpoint') {
3842
+ const url = args[1]
3843
+ if (!url) return { content: 'Usage: /telemetry endpoint <url|none>' }
3844
+ setTelemetryEndpoint(url)
3845
+ return {
3846
+ content:
3847
+ url === NO_ENDPOINT
3848
+ ? '✓ Destination cleared. Telemetry stays on, but nothing is ever sent.'
3849
+ : `✓ Endpoint set to ${url}`,
3850
+ }
3851
+ }
3852
+
3853
+ if (sub !== 'status') {
3854
+ return {
3855
+ content: 'Usage: /telemetry [status|on|off|reset-id|endpoint <url|none>]',
3856
+ }
3857
+ }
3858
+
3859
+ // ── status ──
3860
+ const consent = getTelemetryConsent() ?? resolveTelemetry()
3861
+ const settings = readTelemetrySettings('user')
3862
+ // Two ways for `endpoint` to be empty, and they mean different things: never
3863
+ // resolved (the kill switch fired first) or deliberately cleared (the `none`
3864
+ // sentinel). A third — nothing configured — is unreachable now that the
3865
+ // default is a real URL, which is why this used to be a one-line fallback.
3866
+ const destination = consent.endpoint
3867
+ ? consent.endpoint
3868
+ : consent.endpointSource === 'off'
3869
+ ? '_(not resolved — telemetry is off)_'
3870
+ : '_(nothing — the `none` sentinel cleared it)_'
3871
+ const lines = [
3872
+ '## 📡 Telemetry',
3873
+ '',
3874
+ `| Field | Value |`,
3875
+ `| --- | --- |`,
3876
+ `| State | ${consent.enabled ? '🟢 enabled' : '⚪ disabled'} |`,
3877
+ `| Decided by | \`${consent.source}\` |`,
3878
+ `| Endpoint | ${destination} |`,
3879
+ `| Source | \`${consent.endpointSource}\` |`,
3880
+ `| Install id | ${settings.installId ?? '_(not yet generated)_'} |`,
3881
+ `| Prompted | ${settings.promptedAt ?? '_(never)_'} |`,
3882
+ '',
3883
+ 'Full data dictionary: `docs/telemetry.md`',
3884
+ ]
3885
+ return { content: lines.join('\n') }
3886
+ }
3887
+
3810
3888
  const hooksHealthCmd: CommandHandler = (ctx) => {
3811
3889
  const hookEngine = ctx.engine.getHookEngine?.()
3812
3890
  if (!hookEngine) {
@@ -4319,10 +4397,7 @@ const workflowsCmd: CommandHandler = async () => {
4319
4397
  const { existsSync, readdirSync, readFileSync } = await import('node:fs')
4320
4398
  const { join } = await import('node:path')
4321
4399
 
4322
- const locations = [
4323
- join(process.cwd(), '.claude', 'workflows'),
4324
- join(homedir(), '.claude', 'workflows'),
4325
- ]
4400
+ const locations = workflowScriptDirs(process.cwd())
4326
4401
 
4327
4402
  const lines: string[] = ['─ Workflows ─', '']
4328
4403
  let found = 0
@@ -4360,10 +4435,11 @@ const workflowsCmd: CommandHandler = async () => {
4360
4435
  lines.push('No workflow scripts found.')
4361
4436
  lines.push('')
4362
4437
  lines.push('Workflows are multi-agent orchestration scripts stored in:')
4363
- lines.push(' .claude/workflows/ (project-level)')
4364
- lines.push(' ~/.claude/workflows/ (user-level)')
4438
+ lines.push(' .mipham/workflows/ (project-level — new scripts go here)')
4439
+ lines.push(' .claude/workflows/ (project-level, legacy — still read)')
4440
+ lines.push(' ~/.claude/workflows/ (user-level, legacy — still read)')
4365
4441
  lines.push('')
4366
- lines.push('Create a .js file in either location to add a workflow.')
4442
+ lines.push('Create a .js file in either project location to add a workflow.')
4367
4443
  } else {
4368
4444
  lines.push('')
4369
4445
  lines.push(`${found} workflow(s) found.`)
@@ -4381,14 +4457,20 @@ const workflowSaveCmd = async (name: string): Promise<CommandResult> => {
4381
4457
  const { existsSync, mkdirSync, writeFileSync, readFileSync } = await import('node:fs')
4382
4458
  const { join } = await import('node:path')
4383
4459
 
4384
- const targetDir = join(process.cwd(), '.claude', 'workflows')
4460
+ const targetDir = workflowScriptDir(process.cwd())
4385
4461
  if (!existsSync(targetDir)) {
4386
4462
  mkdirSync(targetDir, { recursive: true })
4387
4463
  }
4388
4464
 
4389
- // Read the last-run state persisted by the Workflow tool
4390
- const stateFile = join(targetDir, '.last-run.json')
4391
- if (!existsSync(stateFile)) {
4465
+ // Read the last-run state persisted by the Workflow tool. Every readable
4466
+ // location is checked, not just the writable one: a run that predates the
4467
+ // move to `.mipham/` left its state in `.claude/workflows/`, so looking in
4468
+ // the new directory alone would report "no recent run" on the first save
4469
+ // after upgrading.
4470
+ const stateFile = workflowScriptDirs(process.cwd())
4471
+ .map((dir) => join(dir, '.last-run.json'))
4472
+ .find((file) => existsSync(file))
4473
+ if (!stateFile) {
4392
4474
  return { content: 'No recent workflow run found. Run a workflow first with /workflow <task>.' }
4393
4475
  }
4394
4476
 
@@ -4417,10 +4499,7 @@ const workflowRunCmd = async (name: string): Promise<CommandResult> => {
4417
4499
 
4418
4500
  const safeName = name.replace(/[^a-zA-Z0-9_-]/g, '-')
4419
4501
 
4420
- const locations = [
4421
- join(process.cwd(), '.claude', 'workflows'),
4422
- join(homedir(), '.claude', 'workflows'),
4423
- ]
4502
+ const locations = workflowScriptDirs(process.cwd())
4424
4503
 
4425
4504
  for (const loc of locations) {
4426
4505
  const scriptPath = join(loc, `${safeName}.js`)
@@ -4435,7 +4514,9 @@ const workflowRunCmd = async (name: string): Promise<CommandResult> => {
4435
4514
  }
4436
4515
 
4437
4516
  return {
4438
- content: `Workflow "${safeName}" not found in .claude/workflows/ or ~/.claude/workflows/`,
4517
+ content:
4518
+ `Workflow "${safeName}" not found in .mipham/workflows/, ` +
4519
+ `.claude/workflows/ or ~/.claude/workflows/`,
4439
4520
  }
4440
4521
  }
4441
4522
 
@@ -5003,7 +5084,7 @@ const forkCmd: CommandHandler = async (ctx, args) => {
5003
5084
  .slice(0, 40)
5004
5085
  const name = `${slug}-${Date.now().toString(36)}`
5005
5086
  const branch = `worktree/${name}`
5006
- const wtPath = join(process.cwd(), '.claude', 'worktrees', name)
5087
+ const wtPath = join(worktreeRoot(process.cwd()), name)
5007
5088
 
5008
5089
  try {
5009
5090
  execSync(`git worktree add -b ${branch} ${wtPath} HEAD`, { stdio: 'ignore', timeout: 30_000 })
@@ -5304,6 +5385,7 @@ const commandsListCmd: CommandHandler = () => {
5304
5385
  '/keys audit': 'Account',
5305
5386
  '/keys view': 'Account',
5306
5387
  '/feedback': 'Account',
5388
+ '/telemetry': 'Account',
5307
5389
  '/agents': 'Agents',
5308
5390
  '/bg': 'Agents',
5309
5391
  '/fork': 'Agents',
@@ -5525,6 +5607,7 @@ registry.set('/cd', cdCmd)
5525
5607
  registry.set('/hooks', hooksCmd)
5526
5608
  registry.set('/hooks health', hooksHealthCmd)
5527
5609
  registry.set('/hooks enable', hooksEnableCmd)
5610
+ registry.set('/telemetry', telemetryCmd)
5528
5611
  registry.set('/batch', batchCmd)
5529
5612
 
5530
5613
  // ═══════════════════════════════════════════════════════════════
@@ -5539,6 +5622,47 @@ export function getCommandNames(): string[] {
5539
5622
  return Array.from(registry.keys()).sort()
5540
5623
  }
5541
5624
 
5625
+ /**
5626
+ * The commands `app.tsx` intercepts **before** the registry lookup.
5627
+ *
5628
+ * They return early and never reach `getCommand`, so the registry is not
5629
+ * authoritative for them. `/model-picker` is the one that is not a registry key
5630
+ * at all — the other five are, so listing them here is belt-and-braces rather
5631
+ * than a claim that they are missing.
5632
+ */
5633
+ const PRE_REGISTRY_COMMANDS = ['/switch', '/pick', '/model-picker', '/exit', '/quit', '/focus']
5634
+
5635
+ /**
5636
+ * The bucket an unrecognised command name is recorded under.
5637
+ *
5638
+ * **Why this exists.** `parseSlashCommand` returns `parts[0]` — whatever the user
5639
+ * typed. Without a convergence point, `/foobar` mints a `command_calls./foobar`
5640
+ * series on the spot, and nothing bounds the number of series:
5641
+ * `MAX_LABEL_LENGTH` in `payload.ts` truncates the **value**, not the key count.
5642
+ * `tool_name` really is closed by construction (the registry declares the tool
5643
+ * set); `command_name` is not.
5644
+ */
5645
+ export const UNKNOWN_COMMAND = '/unknown'
5646
+
5647
+ /** The label a command name is recorded under: itself if we ship it, else the bucket. */
5648
+ export function commandLabelFor(command: string): string {
5649
+ if (registry.has(command) || PRE_REGISTRY_COMMANDS.includes(command)) return command
5650
+ return UNKNOWN_COMMAND
5651
+ }
5652
+
5653
+ /**
5654
+ * Every label `command_calls` can carry — the source the collector's allowlist is
5655
+ * generated from.
5656
+ *
5657
+ * Two more than `getCommandNames()`: `/model-picker` (user-typable, not a registry
5658
+ * key) and `UNKNOWN_COMMAND`. Miss either and the allowlist does not contain it,
5659
+ * so the collector folds those events into `__other__` — and `__other__` is
5660
+ * exactly what T4 must not be reading when it votes.
5661
+ */
5662
+ export function getCommandLabelNames(): string[] {
5663
+ return Array.from(new Set([...registry.keys(), ...PRE_REGISTRY_COMMANDS, UNKNOWN_COMMAND])).sort()
5664
+ }
5665
+
5542
5666
  export interface CommandEntry {
5543
5667
  name: string
5544
5668
  description: string
@@ -5640,7 +5764,7 @@ const COMMAND_DESCRIPTIONS: Record<string, string> = {
5640
5764
  '/loop': 'Run prompt on interval',
5641
5765
  '/init': 'Initialize .mipham config',
5642
5766
  '/setup': 'Guided project setup wizard',
5643
- '/permissions': 'Show permission settings',
5767
+ '/permissions': 'Show or persist permission rules',
5644
5768
  '/add-dir': 'Add workspace directory',
5645
5769
  '/recommend': 'Analyze project + recommend skills & setup',
5646
5770
  '/security': 'Security review checklist',
@@ -5681,6 +5805,7 @@ const COMMAND_DESCRIPTIONS: Record<string, string> = {
5681
5805
  '/hooks health': 'Check hook health — see failures, disabled hooks, recovery status',
5682
5806
  '/hooks enable': 'Manually re-enable a hook that was auto-disabled after repeated failures',
5683
5807
  '/batch': 'Apply changes across multiple files',
5808
+ '/telemetry': 'Show or change anonymous usage-reporting settings',
5684
5809
  }
5685
5810
 
5686
5811
  export function getCommandList(): CommandEntry[] {
@@ -1,8 +1,10 @@
1
+ import { join } from 'node:path'
1
2
  import { SubAgent } from '../../agent/sub-agent'
2
3
  import type { ProviderRegistry } from '../../providers/registry'
3
4
  import type { Llm } from '../../providers/llm'
4
5
  import type { ToolDefinition } from '../../shared/index.ts'
5
6
  import type { PermissionSystem } from '../../core/permission'
7
+ import { worktreeRoot } from '../../core/paths.ts'
6
8
  import { validateJSONSchema, formatValidationErrors } from '../schema-validator'
7
9
 
8
10
  export interface WorkflowAgentOpts {
@@ -32,7 +34,7 @@ export interface WorkflowAgentOpts {
32
34
  * 4. Returns the validated object, or { raw, validationErrors } on final failure.
33
35
  *
34
36
  * When `isolation: 'worktree'` is set:
35
- * 1. A git worktree is created at .claude/worktrees/wf-<slug>
37
+ * 1. A git worktree is created at .mipham/worktrees/wf-<slug>
36
38
  * 2. The sub-agent runs with its cwd set to the worktree path
37
39
  * 3. Changes are auto-committed (best-effort)
38
40
  * 4. The worktree is cleaned up after execution
@@ -60,7 +62,7 @@ export async function workflowAgent(
60
62
  if (opts.isolation === 'worktree') {
61
63
  const slug = `wf-${Date.now().toString(36)}-${Math.random().toString(36).slice(2, 6)}`
62
64
  worktreeBranch = `worktree/${slug}`
63
- worktreePath = `.claude/worktrees/${slug}`
65
+ worktreePath = join(worktreeRoot(process.cwd()), slug)
64
66
 
65
67
  const proc = Bun.spawn(['git', 'worktree', 'add', '-b', worktreeBranch, worktreePath, 'HEAD'], {
66
68
  stdout: 'pipe',
@@ -1,14 +0,0 @@
1
- {
2
- "version": 1,
3
- "tasks": [
4
- {
5
- "id": "task-answer-fn",
6
- "instruction": "用 Write 工具在 <taskDir>/solution.ts 写入一个 TypeScript 文件,导出 `export function answer(): number { return 42 }`。",
7
- "groundTruth": {
8
- "kind": "file-contains",
9
- "file": "solution.ts",
10
- "contains": ["export function answer", "42"]
11
- }
12
- }
13
- ]
14
- }
@@ -1,163 +0,0 @@
1
- // apps/cli/src/core/task-runner.ts
2
- // CRSI 端到端任务运行器(C-MVP)——行为效果度量基建。
3
- import { existsSync, readFileSync, mkdirSync, rmSync } from 'node:fs'
4
- import { join } from 'node:path'
5
- import tasksFile from './task-runner-tasks.json' with { type: 'json' }
6
- import { QueryEngine } from './engine'
7
- import { ContextManager } from './context'
8
- import { PermissionSystem } from './permission'
9
- import { ProviderRegistry } from '../providers/registry'
10
- import type { Llm } from '../providers/llm'
11
- import { createToolRegistry } from '../tools'
12
- import type { PermissionLevel } from '../shared'
13
-
14
- export type RunnerGroundTruth = { kind: 'file-contains'; file: string; contains: string[] }
15
-
16
- export interface RunnerTask {
17
- id: string
18
- instruction: string
19
- groundTruth: RunnerGroundTruth
20
- }
21
-
22
- export function loadRunnerTasks(): RunnerTask[] {
23
- return tasksFile.tasks as unknown as RunnerTask[]
24
- }
25
-
26
- export function judgeTask(task: RunnerTask, taskDir: string): { passed: boolean; detail?: string } {
27
- if (task.groundTruth.kind !== 'file-contains') {
28
- return { passed: false, detail: `unsupported groundTruth kind: ${task.groundTruth.kind}` }
29
- }
30
- const filePath = join(taskDir, task.groundTruth.file)
31
- if (!existsSync(filePath)) {
32
- return { passed: false, detail: `file not found: ${task.groundTruth.file}` }
33
- }
34
- const content = readFileSync(filePath, 'utf-8')
35
- for (const needle of task.groundTruth.contains) {
36
- if (!content.includes(needle)) {
37
- return { passed: false, detail: `missing substring: ${needle}` }
38
- }
39
- }
40
- return { passed: true }
41
- }
42
-
43
- export interface TaskRunResult {
44
- taskId: string
45
- passed: boolean
46
- detail?: string
47
- }
48
-
49
- const TASK_DIR_PLACEHOLDER = '<taskDir>'
50
-
51
- function buildEngine(llm: Llm, permission: PermissionLevel, systemPrompt?: string): QueryEngine {
52
- const registry = new ProviderRegistry([], 'test', 'test-model')
53
- // 注册一个永不 chat 的占位 provider——llm 被 setLlm 覆盖,但 process() 内部
54
- // 多处调用 registry.getActive().config.id 记录 provider id,必须能取到。
55
- registry.register('test', {
56
- config: { id: 'test', name: 'Test', protocol: 'openai-compatible', apiKey: 'key', models: [] },
57
- chat: async function* () {
58
- yield { type: 'stop' }
59
- },
60
- listModels: async () => [],
61
- healthCheck: async () => true,
62
- })
63
- const context = new ContextManager({ maxTokens: 100_000, compactionThreshold: 0.9 })
64
- if (systemPrompt !== undefined) context.setSystemPrompt(systemPrompt)
65
- const tools = createToolRegistry()
66
- const engine = new QueryEngine(registry, context, tools, new PermissionSystem(permission))
67
- engine.setLlm(llm)
68
- return engine
69
- }
70
-
71
- export async function runTask(
72
- task: RunnerTask,
73
- llm: Llm,
74
- opts: { taskDir?: string; permission?: PermissionLevel; systemPrompt?: string } = {},
75
- ): Promise<TaskRunResult> {
76
- const taskDir = opts.taskDir ?? join(process.cwd(), '.mipham', 'task-runner')
77
- const permission = opts.permission ?? 'bypassPermissions'
78
-
79
- rmSync(taskDir, { recursive: true, force: true })
80
- mkdirSync(taskDir, { recursive: true })
81
-
82
- const instruction = task.instruction.replaceAll(TASK_DIR_PLACEHOLDER, taskDir)
83
- const engine = buildEngine(llm, permission, opts.systemPrompt)
84
-
85
- for await (const _ of engine.process(instruction)) {
86
- /* drain agentic loop */
87
- }
88
-
89
- const verdict = judgeTask(task, taskDir)
90
- return { taskId: task.id, passed: verdict.passed, detail: verdict.detail }
91
- }
92
-
93
- export interface TaskRunStats {
94
- taskId: string
95
- samples: number
96
- passed: number
97
- /** 0-1 */
98
- passRate: number
99
- }
100
-
101
- export async function runTaskN(
102
- task: RunnerTask,
103
- llm: Llm,
104
- n: number,
105
- opts: { taskDir?: string; permission?: PermissionLevel; systemPrompt?: string } = {},
106
- ): Promise<TaskRunStats> {
107
- let passed = 0
108
- for (let i = 0; i < n; i++) {
109
- const result = await runTask(task, llm, opts)
110
- if (result.passed) passed++
111
- }
112
- return { taskId: task.id, samples: n, passed, passRate: n > 0 ? passed / n : 0 }
113
- }
114
-
115
- export interface RunComparison {
116
- baseline: TaskRunStats
117
- candidate: TaskRunStats
118
- /** 弱判:candidate 不退化(不低于 baseline 且至少 1 次成功) */
119
- notDegraded: boolean
120
- /** candidate 严格更好(通过率更高) */
121
- improved: boolean
122
- }
123
-
124
- export function isNotDegraded(baseline: TaskRunStats, candidate: TaskRunStats): boolean {
125
- return candidate.passRate >= baseline.passRate && candidate.passed >= 1
126
- }
127
-
128
- export function isImproved(baseline: TaskRunStats, candidate: TaskRunStats): boolean {
129
- return candidate.passRate > baseline.passRate
130
- }
131
-
132
- export function compareRuns(baseline: TaskRunStats, candidate: TaskRunStats): RunComparison {
133
- return {
134
- baseline,
135
- candidate,
136
- notDegraded: isNotDegraded(baseline, candidate),
137
- improved: isImproved(baseline, candidate),
138
- }
139
- }
140
-
141
- export async function runBeforeAfter(
142
- task: RunnerTask,
143
- llm: Llm,
144
- n: number,
145
- opts: {
146
- beforePrompt?: string
147
- afterPrompt?: string
148
- taskDir?: string
149
- permission?: PermissionLevel
150
- } = {},
151
- ): Promise<RunComparison> {
152
- const baseline = await runTaskN(task, llm, n, {
153
- taskDir: opts.taskDir,
154
- permission: opts.permission,
155
- systemPrompt: opts.beforePrompt,
156
- })
157
- const candidate = await runTaskN(task, llm, n, {
158
- taskDir: opts.taskDir,
159
- permission: opts.permission,
160
- systemPrompt: opts.afterPrompt,
161
- })
162
- return compareRuns(baseline, candidate)
163
- }
@@ -1,66 +0,0 @@
1
- import type { SkillDefinition } from '../../shared/index.ts'
2
-
3
- export interface MiphamRuntimeContext {
4
- skill: SkillDefinition
5
- cwd: string
6
- sessionId: string
7
- modelId: string
8
- providerId: string
9
- }
10
-
11
- /**
12
- * Mipham exclusive skill runtime.
13
- * Mipham skills use `.mipham-skill.md` extension and have access to
14
- * Mipham-specific features: model optimization, security features, etc.
15
- */
16
- export class MiphamRuntime {
17
- private context: MiphamRuntimeContext
18
-
19
- constructor(context: MiphamRuntimeContext) {
20
- this.context = context
21
- }
22
-
23
- getSkill(): SkillDefinition {
24
- return this.context.skill
25
- }
26
-
27
- getPrompts(): Record<string, string> {
28
- return this.context.skill.prompts || {}
29
- }
30
-
31
- getTools(): SkillDefinition['tools'] {
32
- return this.context.skill.tools || []
33
- }
34
-
35
- getHooks(): SkillDefinition['hooks'] {
36
- return this.context.skill.hooks || []
37
- }
38
-
39
- /**
40
- * Execute a named prompt with Mipham-specific context.
41
- */
42
- async executePrompt(name: string, variables?: Record<string, string>): Promise<string> {
43
- const prompt = this.context.skill.prompts?.[name]
44
- if (!prompt) {
45
- throw new Error(`Prompt "${name}" not found in skill "${this.context.skill.name}"`)
46
- }
47
-
48
- let result = prompt
49
- const allVars: Record<string, string> = {
50
- provider: this.context.providerId,
51
- model: this.context.modelId,
52
- session: this.context.sessionId,
53
- cwd: this.context.cwd,
54
- ...variables,
55
- }
56
-
57
- // Single-pass substitution: replacement values are NOT re-scanned for more
58
- // template markers, so a variable containing `${key}` can't be re-expanded
59
- // into another variable's value.
60
- result = result.replace(/\$\{([^}]+)\}/g, (match, key: string) =>
61
- Object.prototype.hasOwnProperty.call(allVars, key) ? allVars[key]! : match,
62
- )
63
-
64
- return result
65
- }
66
- }
@@ -1,62 +0,0 @@
1
- import type { SkillDefinition } from '../../shared/index.ts'
2
-
3
- export interface StandardRuntimeContext {
4
- skill: SkillDefinition
5
- cwd: string
6
- sessionId: string
7
- }
8
-
9
- /**
10
- * Standard skill runtime — executes SKILL.md definitions.
11
- * Standard skills follow the open-source SKILL.md specification.
12
- */
13
- export class StandardRuntime {
14
- private context: StandardRuntimeContext
15
-
16
- constructor(context: StandardRuntimeContext) {
17
- this.context = context
18
- }
19
-
20
- getSkill(): SkillDefinition {
21
- return this.context.skill
22
- }
23
-
24
- getPrompts(): Record<string, string> {
25
- return this.context.skill.prompts || {}
26
- }
27
-
28
- getPrompt(name: string): string | undefined {
29
- return this.context.skill.prompts?.[name]
30
- }
31
-
32
- getTools(): SkillDefinition['tools'] {
33
- return this.context.skill.tools || []
34
- }
35
-
36
- getHooks(): SkillDefinition['hooks'] {
37
- return this.context.skill.hooks || []
38
- }
39
-
40
- /**
41
- * Execute a named prompt from the skill.
42
- * Returns the prompt text with any variable substitution applied.
43
- */
44
- async executePrompt(name: string, variables?: Record<string, string>): Promise<string> {
45
- const prompt = this.getPrompt(name)
46
- if (!prompt) {
47
- throw new Error(`Prompt "${name}" not found in skill "${this.context.skill.name}"`)
48
- }
49
-
50
- let result = prompt
51
- if (variables) {
52
- // Single-pass substitution: replacement values are NOT re-scanned for
53
- // more template markers, so an argument containing `${key}` can't be
54
- // re-expanded into another variable's value.
55
- result = result.replace(/\$\{([^}]+)\}/g, (match, key: string) =>
56
- Object.prototype.hasOwnProperty.call(variables, key) ? variables[key]! : match,
57
- )
58
- }
59
-
60
- return result
61
- }
62
- }