@outputai/cli 0.1.13-next.91c5d78.0 → 0.1.13-next.934347c.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. package/dist/api/generated/api.d.ts +322 -84
  2. package/dist/api/generated/api.js +67 -9
  3. package/dist/assets/docker/docker-compose-dev.yml +2 -2
  4. package/dist/commands/credentials/set.d.ts +16 -0
  5. package/dist/commands/credentials/set.js +125 -0
  6. package/dist/commands/credentials/set.spec.d.ts +1 -0
  7. package/dist/commands/credentials/set.spec.js +160 -0
  8. package/dist/commands/migrate.d.ts +11 -0
  9. package/dist/commands/migrate.js +40 -0
  10. package/dist/commands/workflow/list.d.ts +1 -0
  11. package/dist/commands/workflow/list.js +12 -6
  12. package/dist/commands/workflow/list.spec.js +21 -0
  13. package/dist/commands/workflow/plan.js +1 -1
  14. package/dist/commands/workflow/run.js +2 -1
  15. package/dist/commands/workflow/run.spec.js +1 -0
  16. package/dist/commands/workflow/start.js +2 -1
  17. package/dist/commands/workflow/start.spec.js +1 -0
  18. package/dist/commands/workflow/test_eval.js +2 -2
  19. package/dist/components/command_footer.d.ts +8 -0
  20. package/dist/components/command_footer.js +4 -0
  21. package/dist/components/status_icon.d.ts +11 -0
  22. package/dist/components/status_icon.js +25 -0
  23. package/dist/components/workflow_summary.d.ts +10 -0
  24. package/dist/components/workflow_summary.js +4 -0
  25. package/dist/generated/framework_version.json +1 -1
  26. package/dist/services/claude_client.d.ts +14 -2
  27. package/dist/services/claude_client.integration.test.js +2 -2
  28. package/dist/services/claude_client.js +42 -6
  29. package/dist/services/claude_client.spec.js +3 -3
  30. package/dist/services/datasets.d.ts +1 -1
  31. package/dist/services/datasets.js +41 -37
  32. package/dist/services/datasets.test.d.ts +1 -0
  33. package/dist/services/datasets.test.js +202 -0
  34. package/dist/services/workflow_builder.d.ts +1 -1
  35. package/dist/services/workflow_builder.js +1 -1
  36. package/dist/utils/date_formatter.d.ts +11 -1
  37. package/dist/utils/date_formatter.js +26 -1
  38. package/dist/utils/format_workflow_result.d.ts +3 -3
  39. package/dist/utils/open_url.d.ts +1 -0
  40. package/dist/utils/open_url.js +12 -0
  41. package/dist/views/dev.js +62 -26
  42. package/dist/views/workflow/list.d.ts +6 -0
  43. package/dist/views/workflow/list.js +127 -0
  44. package/package.json +11 -11
@@ -114,7 +114,7 @@ export default class WorkflowTest extends Command {
114
114
  }
115
115
  };
116
116
  if (save) {
117
- const filePath = join(dir, `${dataset.name}.yml`);
117
+ const filePath = dataset._source ?? join(dir, `${dataset.name}.yml`);
118
118
  await writeDataset(updated, filePath);
119
119
  this.log(` Saved output to ${filePath}`);
120
120
  }
@@ -137,7 +137,7 @@ export default class WorkflowTest extends Command {
137
137
  date: now
138
138
  }
139
139
  };
140
- const filePath = join(dir, `${dataset.name}.yml`);
140
+ const filePath = dataset._source ?? join(dir, `${dataset.name}.yml`);
141
141
  await writeDataset(updated, filePath);
142
142
  this.log(` Saved eval result to ${filePath}`);
143
143
  }
@@ -0,0 +1,8 @@
1
+ import React from 'react';
2
+ export interface CommandHint {
3
+ key: string;
4
+ label: string;
5
+ }
6
+ export declare const CommandFooter: React.FC<{
7
+ hints: CommandHint[];
8
+ }>;
@@ -0,0 +1,4 @@
1
+ import { jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
2
+ import React from 'react';
3
+ import { Box, Text } from 'ink';
4
+ export const CommandFooter = ({ hints }) => (_jsx(Box, { marginTop: 1, children: hints.map((hint, i) => (_jsxs(React.Fragment, { children: [i > 0 && _jsx(Text, { dimColor: true, children: ' | ' }), _jsx(Text, { dimColor: true, children: '(' }), _jsx(Text, { dimColor: true, bold: true, children: hint.key }), _jsx(Text, { dimColor: true, children: ')' }), _jsx(Text, { dimColor: true, children: ` ${hint.label}` })] }, hint.key))) }));
@@ -0,0 +1,11 @@
1
+ import React from 'react';
2
+ interface StatusDisplay {
3
+ icon: string;
4
+ color: string;
5
+ }
6
+ export declare const resolveStatus: (status: string) => StatusDisplay;
7
+ export declare const statusColor: (status: string) => string;
8
+ export declare const StatusIcon: React.FC<{
9
+ status: string;
10
+ }>;
11
+ export {};
@@ -0,0 +1,25 @@
1
+ import { jsx as _jsx } from "react/jsx-runtime";
2
+ import { Text } from 'ink';
3
+ const STATUS_MAP = {
4
+ // Docker service health
5
+ healthy: { icon: '●', color: 'green' },
6
+ unhealthy: { icon: '○', color: 'red' },
7
+ starting: { icon: '◐', color: 'yellow' },
8
+ none: { icon: '●', color: 'blue' },
9
+ exited: { icon: '✗', color: 'red' },
10
+ // Workflow run status
11
+ running: { icon: '●', color: 'blue' },
12
+ completed: { icon: '●', color: 'green' },
13
+ failed: { icon: '✗', color: 'red' },
14
+ canceled: { icon: '○', color: 'gray' },
15
+ terminated: { icon: '✗', color: 'red' },
16
+ timed_out: { icon: '✗', color: 'red' },
17
+ continued: { icon: '↻', color: 'blue' }
18
+ };
19
+ const DEFAULT_DISPLAY = { icon: '?', color: 'white' };
20
+ export const resolveStatus = (status) => STATUS_MAP[status] ?? DEFAULT_DISPLAY;
21
+ export const statusColor = (status) => resolveStatus(status).color;
22
+ export const StatusIcon = ({ status }) => {
23
+ const { icon, color } = resolveStatus(status);
24
+ return _jsx(Text, { color: color, children: icon });
25
+ };
@@ -0,0 +1,10 @@
1
+ import React from 'react';
2
+ export interface WorkflowSummary {
3
+ running: number;
4
+ completed: number;
5
+ failed: number;
6
+ total: number;
7
+ }
8
+ export declare const WorkflowSummarySection: React.FC<{
9
+ summary: WorkflowSummary;
10
+ }>;
@@ -0,0 +1,4 @@
1
+ import { jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
2
+ import { Box, Text } from 'ink';
3
+ import { statusColor } from '#components/status_icon.js';
4
+ export const WorkflowSummarySection = ({ summary }) => (_jsxs(Box, { flexDirection: "column", marginTop: 1, children: [_jsx(Text, { bold: true, children: "\uD83D\uDCCB Workflows" }), _jsxs(Box, { marginTop: 1, children: [_jsxs(Text, { color: statusColor('running'), children: [summary.running, " running"] }), _jsx(Text, { children: ", " }), _jsxs(Text, { color: statusColor('failed'), children: [summary.failed, " failed"] }), _jsx(Text, { children: ", " }), _jsxs(Text, { color: statusColor('completed'), children: [summary.completed, " complete"] })] })] }));
@@ -1,3 +1,3 @@
1
1
  {
2
- "framework": "0.1.13-next.91c5d78.0"
2
+ "framework": "0.1.13-next.934347c.0"
3
3
  }
@@ -5,6 +5,7 @@ import { Options } from '@anthropic-ai/claude-agent-sdk';
5
5
  export declare const ADDITIONAL_INSTRUCTIONS: {
6
6
  readonly PLAN: "\n! IMPORTANT !\n1. Use TodoWrite to track your progress through plan creation.\n\n2. Please respond with only the final version of the plan content.\n\n3. Respond in a markdown format with these metadata headers:\n\n---\ntitle: <plan-title>\ndescription: <plan-description>\ndate: <plan-date>\n---\n\n<plan-content>\n\n4. After you mark all todos as complete, you must respond with the final version of the plan.\n\n5. DO NOT write the plan to disk — the CLI will handle saving the file to the plans directory.\n\n6. DO NOT suggest any next steps, follow-up commands, or instructions for the user — the CLI will inform the user of next steps after saving.\n";
7
7
  readonly BUILD: "\n! IMPORTANT !\n1. Use TodoWrite to track your progress through workflow implementation.\n\n2. Follow the implementation plan exactly as specified in the plan file.\n\n3. Implement all workflow files following Output.ai patterns and best practices.\n\n4. After you mark all todos as complete, provide a summary of what was implemented.\n";
8
+ readonly MIGRATE: "\n! IMPORTANT !\n1. Use TodoWrite to track your progress through the migration.\n\n2. Fetch the migration guide from https://docs.output.ai/migrations — do not invent migration steps from memory.\n\n3. If the specific guide URL 404s, fall back to the /migrations index and chain the guides that cover the version range.\n\n4. Confirm the planned changes with the user before editing files.\n\n5. After you mark all todos as complete, provide a summary of which files changed and whether the type check passed.\n";
8
9
  };
9
10
  export declare const PLAN_COMMAND_OPTIONS: Options;
10
11
  interface ReplyToClaudeOptions {
@@ -12,13 +13,14 @@ interface ReplyToClaudeOptions {
12
13
  applyAdditionalInstructions?: string;
13
14
  }
14
15
  export declare const BUILD_COMMAND_OPTIONS: Options;
16
+ export declare const MIGRATE_COMMAND_OPTIONS: Options;
15
17
  export declare class ClaudeInvocationError extends Error {
16
18
  cause?: Error | undefined;
17
19
  constructor(message: string, cause?: Error | undefined);
18
20
  }
19
21
  export declare function replyToClaude(message: string, { anthropicOpts, applyAdditionalInstructions }?: ReplyToClaudeOptions): Promise<string>;
20
22
  /**
21
- * Invoke claude-code with /outputai:plan_workflow slash command
23
+ * Invoke claude-code with /output-plan-workflow slash command
22
24
  * The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
23
25
  * ensureOutputAISystem() scaffolds the command files to that location.
24
26
  * @param description - Workflow description
@@ -26,7 +28,7 @@ export declare function replyToClaude(message: string, { anthropicOpts, applyAdd
26
28
  */
27
29
  export declare function invokePlanWorkflow(description: string): Promise<string>;
28
30
  /**
29
- * Invoke claude-code with /outputai:build_workflow slash command
31
+ * Invoke claude-code with /output-build-workflow slash command
30
32
  * The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
31
33
  * ensureOutputAISystem() scaffolds the command files to that location.
32
34
  * @param planFilePath - Absolute path to the plan file
@@ -36,4 +38,14 @@ export declare function invokePlanWorkflow(description: string): Promise<string>
36
38
  * @returns Implementation output from claude-code
37
39
  */
38
40
  export declare function invokeBuildWorkflow(planFilePath: string, workflowDir: string, workflowName: string, additionalInstructions?: string): Promise<string>;
41
+ /**
42
+ * Invoke claude-code with /output-migrate slash command (registered via the output-migrate skill).
43
+ * The slash command fetches migration instructions from docs.output.ai —
44
+ * this CLI wrapper just passes through the version arguments.
45
+ * @param fromVersion - Current framework version (empty string = auto-detect)
46
+ * @param toVersion - Target version (empty string = use npm "latest")
47
+ * @param additionalInstructions - Optional user-supplied guidance
48
+ * @returns Migration summary from claude-code
49
+ */
50
+ export declare function invokeMigrate(fromVersion: string, toVersion: string, additionalInstructions?: string): Promise<string>;
39
51
  export {};
@@ -15,7 +15,7 @@ describe('invokePlanWorkflow - Integration Tests', () => {
15
15
  const messages = [];
16
16
  try {
17
17
  for await (const message of query({
18
- prompt: `/outputai:plan_workflow ${description}`,
18
+ prompt: `/output-plan-workflow ${description}`,
19
19
  options: { maxTurns: 1 }
20
20
  })) {
21
21
  console.log('\nReceived message:', JSON.stringify(message, null, 2));
@@ -31,7 +31,7 @@ describe('invokePlanWorkflow - Integration Tests', () => {
31
31
  // This test is just for debugging - we expect messages
32
32
  expect(messages.length).toBeGreaterThan(0);
33
33
  }, 60000); // 60 second timeout
34
- it('should successfully invoke /outputai:plan_workflow slash command and return content', async () => {
34
+ it('should successfully invoke /output-plan-workflow slash command and return content', async () => {
35
35
  const description = 'Simple workflow that takes a number and doubles it';
36
36
  const result = await invokePlanWorkflow(description);
37
37
  console.log('\n===== PLAN RESULT =====');
@@ -38,10 +38,25 @@ date: <plan-date>
38
38
  3. Implement all workflow files following Output.ai patterns and best practices.
39
39
 
40
40
  4. After you mark all todos as complete, provide a summary of what was implemented.
41
+ `,
42
+ MIGRATE: `
43
+ ! IMPORTANT !
44
+ 1. Use TodoWrite to track your progress through the migration.
45
+
46
+ 2. Fetch the migration guide from https://docs.output.ai/migrations — do not invent migration steps from memory.
47
+
48
+ 3. If the specific guide URL 404s, fall back to the /migrations index and chain the guides that cover the version range.
49
+
50
+ 4. Confirm the planned changes with the user before editing files.
51
+
52
+ 5. After you mark all todos as complete, provide a summary of which files changed and whether the type check passed.
41
53
  `
42
54
  };
43
- const PLAN_COMMAND = 'outputai:plan_workflow';
44
- const BUILD_COMMAND = 'outputai:build_workflow';
55
+ // Slash-command naming convention used by the outputai plugin:
56
+ // `outputai:<kebab-name>` — skills under coding_assistants/.../skills/, which surface as top-level slash commands without the plugin prefix.
57
+ const PLAN_COMMAND = 'outputai:output-plan-workflow';
58
+ const BUILD_COMMAND = 'outputai:output-build-workflow';
59
+ const MIGRATE_COMMAND = 'outputai:output-migrate';
45
60
  const GLOBAL_CLAUDE_OPTIONS = {
46
61
  settingSources: ['user', 'project', 'local']
47
62
  };
@@ -51,6 +66,9 @@ export const PLAN_COMMAND_OPTIONS = {
51
66
  export const BUILD_COMMAND_OPTIONS = {
52
67
  permissionMode: 'bypassPermissions'
53
68
  };
69
+ export const MIGRATE_COMMAND_OPTIONS = {
70
+ permissionMode: 'bypassPermissions'
71
+ };
54
72
  export class ClaudeInvocationError extends Error {
55
73
  cause;
56
74
  constructor(message, cause) {
@@ -77,7 +95,7 @@ function validateEnvironment() {
77
95
  }
78
96
  }
79
97
  function validateSystem(systemMessage) {
80
- const requiredCommands = [PLAN_COMMAND, BUILD_COMMAND];
98
+ const requiredCommands = [PLAN_COMMAND, BUILD_COMMAND, MIGRATE_COMMAND];
81
99
  const availableCommands = systemMessage.slash_commands;
82
100
  const missingCommands = requiredCommands.filter(command => !availableCommands.includes(command));
83
101
  return {
@@ -105,7 +123,10 @@ function getTodoWriteMessage(message) {
105
123
  if (message.type !== 'assistant') {
106
124
  return null;
107
125
  }
108
- const todoWriteMessage = message.message.content.find((c) => c.type === 'tool_use' && c.name === 'TodoWrite');
126
+ const todoWriteMessage = message.message.content.find((c) => {
127
+ const block = c;
128
+ return block.type === 'tool_use' && block.name === 'TodoWrite';
129
+ });
109
130
  return todoWriteMessage ?? null;
110
131
  }
111
132
  function applyInstructions(message, instructions) {
@@ -190,7 +211,7 @@ export async function replyToClaude(message, { anthropicOpts, applyAdditionalIns
190
211
  return singleQuery(applyInstructions(message, applyAdditionalInstructions), { continue: true, ...anthropicOpts });
191
212
  }
192
213
  /**
193
- * Invoke claude-code with /outputai:plan_workflow slash command
214
+ * Invoke claude-code with /output-plan-workflow slash command
194
215
  * The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
195
216
  * ensureOutputAISystem() scaffolds the command files to that location.
196
217
  * @param description - Workflow description
@@ -200,7 +221,7 @@ export async function invokePlanWorkflow(description) {
200
221
  return singleQuery(applyInstructions(`/${PLAN_COMMAND} ${description}`, ADDITIONAL_INSTRUCTIONS.PLAN), PLAN_COMMAND_OPTIONS);
201
222
  }
202
223
  /**
203
- * Invoke claude-code with /outputai:build_workflow slash command
224
+ * Invoke claude-code with /output-build-workflow slash command
204
225
  * The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
205
226
  * ensureOutputAISystem() scaffolds the command files to that location.
206
227
  * @param planFilePath - Absolute path to the plan file
@@ -216,3 +237,18 @@ export async function invokeBuildWorkflow(planFilePath, workflowDir, workflowNam
216
237
  `/${BUILD_COMMAND} ${commandArgs}`;
217
238
  return singleQuery(applyInstructions(fullCommand, ADDITIONAL_INSTRUCTIONS.BUILD), BUILD_COMMAND_OPTIONS);
218
239
  }
240
+ /**
241
+ * Invoke claude-code with /output-migrate slash command (registered via the output-migrate skill).
242
+ * The slash command fetches migration instructions from docs.output.ai —
243
+ * this CLI wrapper just passes through the version arguments.
244
+ * @param fromVersion - Current framework version (empty string = auto-detect)
245
+ * @param toVersion - Target version (empty string = use npm "latest")
246
+ * @param additionalInstructions - Optional user-supplied guidance
247
+ * @returns Migration summary from claude-code
248
+ */
249
+ export async function invokeMigrate(fromVersion, toVersion, additionalInstructions) {
250
+ const from = fromVersion || 'auto';
251
+ const to = toVersion || 'latest';
252
+ const commandArgs = [from, to, additionalInstructions].filter(Boolean).join(' ');
253
+ return singleQuery(applyInstructions(`/${MIGRATE_COMMAND} ${commandArgs}`, ADDITIONAL_INSTRUCTIONS.MIGRATE), MIGRATE_COMMAND_OPTIONS);
254
+ }
@@ -13,7 +13,7 @@ describe('invokePlanWorkflow', () => {
13
13
  // Clean up environment variables
14
14
  delete process.env.ANTHROPIC_API_KEY;
15
15
  });
16
- it('should invoke /outputai:plan_workflow slash command with settingSources', async () => {
16
+ it('should invoke /outputai:output-plan-workflow slash command with settingSources', async () => {
17
17
  const { query } = await import('@anthropic-ai/claude-agent-sdk');
18
18
  process.env.ANTHROPIC_API_KEY = 'test-key';
19
19
  async function* mockIterator() {
@@ -22,7 +22,7 @@ describe('invokePlanWorkflow', () => {
22
22
  vi.mocked(query).mockReturnValue(mockIterator());
23
23
  await invokePlanWorkflow('Test workflow');
24
24
  const calls = vi.mocked(query).mock.calls;
25
- expect(calls[0]?.[0]?.prompt).toContain('/outputai:plan_workflow Test workflow');
25
+ expect(calls[0]?.[0]?.prompt).toContain('/outputai:output-plan-workflow Test workflow');
26
26
  expect(calls[0]?.[0]?.options?.settingSources).toEqual(['user', 'project', 'local']);
27
27
  expect(calls[0]?.[0]?.options?.allowedTools).toEqual(['Read', 'Grep', 'WebSearch', 'WebFetch', 'TodoWrite']);
28
28
  });
@@ -36,7 +36,7 @@ describe('invokePlanWorkflow', () => {
36
36
  const description = 'Build a user authentication system';
37
37
  await invokePlanWorkflow(description);
38
38
  const calls = vi.mocked(query).mock.calls;
39
- expect(calls[0]?.[0]?.prompt).toContain(`/outputai:plan_workflow ${description}`);
39
+ expect(calls[0]?.[0]?.prompt).toContain(`/outputai:output-plan-workflow ${description}`);
40
40
  expect(calls[0]?.[0]?.options?.settingSources).toEqual(['user', 'project', 'local']);
41
41
  });
42
42
  it('should return plan output from claude-code', async () => {
@@ -8,7 +8,7 @@ export interface DatasetInfo {
8
8
  }
9
9
  export declare function resolveDatasetsDir(workflowName: string, basePath?: string): string | null;
10
10
  export declare function resolveDefaultDatasetsDir(workflowName: string, basePath?: string): string;
11
- export declare function readDataset(filePath: string): Promise<Dataset>;
11
+ export declare function readDatasetFile(filePath: string): Promise<Dataset[]>;
12
12
  export declare function readAllDatasets(workflowName: string, filterNames?: string[], basePath?: string): Promise<{
13
13
  datasets: Dataset[];
14
14
  dir: string;
@@ -24,10 +24,19 @@ export function resolveDefaultDatasetsDir(workflowName, basePath = process.cwd()
24
24
  // Default to first workflows path
25
25
  return resolve(basePath, WORKFLOWS_PATHS[0], workflowName, DATASETS_DIR);
26
26
  }
27
- export async function readDataset(filePath) {
28
- const content = await readFile(filePath, 'utf-8');
29
- const raw = yaml.load(content);
30
- return DatasetSchema.parse(raw);
27
+ export async function readDatasetFile(filePath) {
28
+ const raw = yaml.load(await readFile(filePath, 'utf-8'));
29
+ if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
30
+ throw new Error(`Invalid dataset file: ${filePath}`);
31
+ }
32
+ return Object.entries(raw).map(([name, body]) => {
33
+ if (!body || typeof body !== 'object' || !('input' in body)) {
34
+ throw new Error(`Dataset case "${name}" in ${filePath} is missing required "input" field`);
35
+ }
36
+ const dataset = DatasetSchema.parse({ name, ...body });
37
+ dataset._source = filePath;
38
+ return dataset;
39
+ });
31
40
  }
32
41
  export async function readAllDatasets(workflowName, filterNames, basePath) {
33
42
  const dir = resolveDatasetsDir(workflowName, basePath);
@@ -36,42 +45,35 @@ export async function readAllDatasets(workflowName, filterNames, basePath) {
36
45
  }
37
46
  const files = await readdir(dir);
38
47
  const ymlFiles = files.filter(f => f.endsWith('.yml') || f.endsWith('.yaml'));
48
+ const seen = new Set();
39
49
  const datasets = [];
40
50
  for (const file of ymlFiles) {
41
- const dataset = await readDataset(join(dir, file));
42
- if (filterNames && !filterNames.includes(dataset.name)) {
43
- continue;
51
+ const cases = await readDatasetFile(join(dir, file));
52
+ for (const dataset of cases) {
53
+ if (seen.has(dataset.name)) {
54
+ throw new Error(`Duplicate dataset case name "${dataset.name}" found in ${file}`);
55
+ }
56
+ seen.add(dataset.name);
57
+ if (filterNames && !filterNames.includes(dataset.name)) {
58
+ continue;
59
+ }
60
+ datasets.push(dataset);
44
61
  }
45
- datasets.push(dataset);
46
62
  }
47
63
  return { datasets, dir };
48
64
  }
49
- async function mergeWithExisting(dataset, filePath) {
50
- if (!existsSync(filePath)) {
51
- return dataset;
52
- }
53
- try {
54
- const existing = await readDataset(filePath);
55
- return {
56
- ...existing,
57
- ...dataset,
58
- ground_truth: dataset.ground_truth ?? existing.ground_truth,
59
- last_output: dataset.last_output ?? existing.last_output,
60
- last_eval: dataset.last_eval ?? existing.last_eval
61
- };
62
- }
63
- catch {
64
- return dataset;
65
- }
66
- }
67
65
  export async function writeDataset(dataset, filePath) {
68
- const merged = await mergeWithExisting(dataset, filePath);
69
66
  const dir = resolve(filePath, '..');
70
67
  if (!existsSync(dir)) {
71
68
  await mkdir(dir, { recursive: true });
72
69
  }
73
- const content = yaml.dump(merged, { lineWidth: 120, noRefs: true, sortKeys: false });
74
- await writeFile(filePath, content, 'utf-8');
70
+ const loaded = existsSync(filePath) ? yaml.load(await readFile(filePath, 'utf-8')) : null;
71
+ const fileObj = (loaded && typeof loaded === 'object' && !Array.isArray(loaded)) ?
72
+ loaded :
73
+ {};
74
+ const { name, _source, ...caseBody } = dataset;
75
+ fileObj[name] = { ...fileObj[name], ...caseBody };
76
+ await writeFile(filePath, yaml.dump(fileObj, { lineWidth: 120, noRefs: true, sortKeys: false }), 'utf-8');
75
77
  }
76
78
  export async function listDatasets(workflowName, basePath) {
77
79
  const dir = resolveDatasetsDir(workflowName, basePath);
@@ -84,14 +86,16 @@ export async function listDatasets(workflowName, basePath) {
84
86
  for (const file of ymlFiles) {
85
87
  const filePath = join(dir, file);
86
88
  try {
87
- const dataset = await readDataset(filePath);
88
- results.push({
89
- name: dataset.name,
90
- path: filePath,
91
- hasLastOutput: dataset.last_output?.output !== undefined,
92
- lastOutputDate: dataset.last_output?.date,
93
- lastEvalDate: dataset.last_eval?.date
94
- });
89
+ const cases = await readDatasetFile(filePath);
90
+ for (const dataset of cases) {
91
+ results.push({
92
+ name: dataset.name,
93
+ path: filePath,
94
+ hasLastOutput: dataset.last_output?.output !== undefined,
95
+ lastOutputDate: dataset.last_output?.date,
96
+ lastEvalDate: dataset.last_eval?.date
97
+ });
98
+ }
95
99
  }
96
100
  catch {
97
101
  results.push({
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,202 @@
1
+ import { describe, it, expect, beforeEach, afterEach } from 'vitest';
2
+ import { mkdtemp, rm, writeFile, readFile, mkdir } from 'node:fs/promises';
3
+ import { join } from 'node:path';
4
+ import { tmpdir } from 'node:os';
5
+ import yaml from 'js-yaml';
6
+ import { readDatasetFile, readAllDatasets, writeDataset, listDatasets } from './datasets.js';
7
+ const ctx = { tmpDir: '' };
8
+ beforeEach(async () => {
9
+ ctx.tmpDir = await mkdtemp(join(tmpdir(), 'output-datasets-test-'));
10
+ });
11
+ afterEach(async () => {
12
+ await rm(ctx.tmpDir, { recursive: true, force: true });
13
+ });
14
+ function writeYaml(filePath, obj) {
15
+ return writeFile(filePath, yaml.dump(obj, { lineWidth: 120, noRefs: true, sortKeys: false }), 'utf-8');
16
+ }
17
+ // ---------------------------------------------------------------------------
18
+ // readDatasetFile
19
+ // ---------------------------------------------------------------------------
20
+ describe('readDatasetFile', () => {
21
+ it('parses a multi-case file and returns all cases', async () => {
22
+ const filePath = join(ctx.tmpDir, 'cases.yml');
23
+ await writeYaml(filePath, {
24
+ case_a: { input: { query: 'foo' }, ground_truth: { expected: 1 } },
25
+ case_b: { input: { query: 'bar' } },
26
+ case_c: { input: { query: 'baz' }, ground_truth: { expected: 3 } }
27
+ });
28
+ const datasets = await readDatasetFile(filePath);
29
+ expect(datasets).toHaveLength(3);
30
+ expect(datasets.map(d => d.name)).toEqual(['case_a', 'case_b', 'case_c']);
31
+ expect(datasets[0].input).toEqual({ query: 'foo' });
32
+ expect(datasets[0].ground_truth).toEqual({ expected: 1 });
33
+ });
34
+ it('attaches _source with the absolute file path to each dataset', async () => {
35
+ const filePath = join(ctx.tmpDir, 'cases.yml');
36
+ await writeYaml(filePath, {
37
+ my_case: { input: { x: 1 } }
38
+ });
39
+ const [dataset] = await readDatasetFile(filePath);
40
+ expect(dataset._source).toBe(filePath);
41
+ });
42
+ it('throws when a case is missing input', async () => {
43
+ const filePath = join(ctx.tmpDir, 'bad.yml');
44
+ await writeYaml(filePath, {
45
+ good_case: { input: { x: 1 } },
46
+ bad_case: { ground_truth: { expected: 42 } }
47
+ });
48
+ await expect(readDatasetFile(filePath)).rejects.toThrow('Dataset case "bad_case" in');
49
+ });
50
+ it('throws when file content is not an object', async () => {
51
+ const filePath = join(ctx.tmpDir, 'bad.yml');
52
+ await writeFile(filePath, 'just a string', 'utf-8');
53
+ await expect(readDatasetFile(filePath)).rejects.toThrow('Invalid dataset file');
54
+ });
55
+ it('throws with clear message when file content is a YAML array', async () => {
56
+ const filePath = join(ctx.tmpDir, 'bad.yml');
57
+ await writeFile(filePath, '- foo\n- bar\n', 'utf-8');
58
+ await expect(readDatasetFile(filePath)).rejects.toThrow('Invalid dataset file');
59
+ });
60
+ it('preserves last_output and last_eval fields', async () => {
61
+ const filePath = join(ctx.tmpDir, 'cases.yml');
62
+ await writeYaml(filePath, {
63
+ cached_case: {
64
+ input: { q: 'hello' },
65
+ last_output: { output: { result: 42 }, executionTimeMs: 100, date: '2026-01-01T00:00:00.000Z' }
66
+ }
67
+ });
68
+ const [dataset] = await readDatasetFile(filePath);
69
+ expect(dataset.last_output?.output).toEqual({ result: 42 });
70
+ expect(dataset.last_output?.executionTimeMs).toBe(100);
71
+ });
72
+ });
73
+ // ---------------------------------------------------------------------------
74
+ // readAllDatasets
75
+ // ---------------------------------------------------------------------------
76
+ describe('readAllDatasets', () => {
77
+ it('flattens cases from multiple files', async () => {
78
+ const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
79
+ await mkdir(datasetsDir, { recursive: true });
80
+ await writeYaml(join(datasetsDir, 'group_a.yml'), {
81
+ case_1: { input: { x: 1 } },
82
+ case_2: { input: { x: 2 } }
83
+ });
84
+ await writeYaml(join(datasetsDir, 'group_b.yml'), {
85
+ case_3: { input: { x: 3 } }
86
+ });
87
+ const { datasets } = await readAllDatasets('my_workflow', undefined, ctx.tmpDir);
88
+ expect(datasets).toHaveLength(3);
89
+ expect(datasets.map(d => d.name).sort()).toEqual(['case_1', 'case_2', 'case_3']);
90
+ });
91
+ it('filters by case name across files', async () => {
92
+ const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
93
+ await mkdir(datasetsDir, { recursive: true });
94
+ await writeYaml(join(datasetsDir, 'group_a.yml'), {
95
+ case_1: { input: { x: 1 } },
96
+ case_2: { input: { x: 2 } }
97
+ });
98
+ await writeYaml(join(datasetsDir, 'group_b.yml'), {
99
+ case_3: { input: { x: 3 } }
100
+ });
101
+ const { datasets } = await readAllDatasets('my_workflow', ['case_2', 'case_3'], ctx.tmpDir);
102
+ expect(datasets).toHaveLength(2);
103
+ expect(datasets.map(d => d.name).sort()).toEqual(['case_2', 'case_3']);
104
+ });
105
+ it('returns empty datasets and a default dir when workflow has no datasets dir', async () => {
106
+ const { datasets, dir } = await readAllDatasets('nonexistent_workflow', undefined, ctx.tmpDir);
107
+ expect(datasets).toHaveLength(0);
108
+ expect(dir).toContain('nonexistent_workflow');
109
+ });
110
+ it('throws when the same case name appears in two different files', async () => {
111
+ const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
112
+ await mkdir(datasetsDir, { recursive: true });
113
+ await writeYaml(join(datasetsDir, 'group_a.yml'), { case_1: { input: { x: 1 } } });
114
+ await writeYaml(join(datasetsDir, 'group_b.yml'), { case_1: { input: { x: 2 } } });
115
+ await expect(readAllDatasets('my_workflow', undefined, ctx.tmpDir)).rejects.toThrow('Duplicate dataset case name "case_1"');
116
+ });
117
+ });
118
+ // ---------------------------------------------------------------------------
119
+ // writeDataset
120
+ // ---------------------------------------------------------------------------
121
+ describe('writeDataset', () => {
122
+ it('creates a new file with one case keyed by name', async () => {
123
+ const filePath = join(ctx.tmpDir, 'cases.yml');
124
+ const dataset = { name: 'new_case', input: { q: 'hello' } };
125
+ await writeDataset(dataset, filePath);
126
+ const raw = yaml.load(await readFile(filePath, 'utf-8'));
127
+ expect(raw).toHaveProperty('new_case');
128
+ expect(raw.new_case.input).toEqual({ q: 'hello' });
129
+ expect(raw.new_case).not.toHaveProperty('name');
130
+ });
131
+ it('does not write _source into the file', async () => {
132
+ const filePath = join(ctx.tmpDir, 'cases.yml');
133
+ const dataset = { name: 'my_case', input: { q: 'x' }, _source: '/some/path.yml' };
134
+ await writeDataset(dataset, filePath);
135
+ const raw = yaml.load(await readFile(filePath, 'utf-8'));
136
+ expect(raw.my_case).not.toHaveProperty('_source');
137
+ });
138
+ it('updates only the target case, leaving other cases untouched', async () => {
139
+ const filePath = join(ctx.tmpDir, 'cases.yml');
140
+ await writeYaml(filePath, {
141
+ case_a: { input: { x: 1 }, ground_truth: { expected: 1 } },
142
+ case_b: { input: { x: 2 }, ground_truth: { expected: 2 } }
143
+ });
144
+ await writeDataset({ name: 'case_a', input: { x: 1 }, last_output: { output: { result: 1 }, date: '2026-01-01T00:00:00.000Z' } }, filePath);
145
+ const raw = yaml.load(await readFile(filePath, 'utf-8'));
146
+ expect(raw).toHaveProperty('case_b');
147
+ expect(raw.case_b.ground_truth).toEqual({ expected: 2 });
148
+ });
149
+ it('preserves existing fields when writing last_output then last_eval', async () => {
150
+ const filePath = join(ctx.tmpDir, 'cases.yml');
151
+ await writeYaml(filePath, {
152
+ my_case: { input: { q: 'hello' }, ground_truth: { expected: 42 } }
153
+ });
154
+ await writeDataset({ name: 'my_case', input: { q: 'hello' }, last_output: { output: { result: 42 }, executionTimeMs: 50, date: '2026-01-01T00:00:00.000Z' } }, filePath);
155
+ await writeDataset({
156
+ name: 'my_case', input: { q: 'hello' },
157
+ last_eval: { output: { datasetName: 'my_case', verdict: 'pass', evaluators: [] }, date: '2026-01-01T00:01:00.000Z' }
158
+ }, filePath);
159
+ const raw = yaml.load(await readFile(filePath, 'utf-8'));
160
+ const caseObj = raw.my_case;
161
+ expect(caseObj).toHaveProperty('last_output');
162
+ expect(caseObj).toHaveProperty('last_eval');
163
+ expect(caseObj).toHaveProperty('ground_truth');
164
+ });
165
+ it('creates parent directories if they do not exist', async () => {
166
+ const filePath = join(ctx.tmpDir, 'deep', 'nested', 'cases.yml');
167
+ await writeDataset({ name: 'my_case', input: { q: 'x' } }, filePath);
168
+ const raw = yaml.load(await readFile(filePath, 'utf-8'));
169
+ expect(raw).toHaveProperty('my_case');
170
+ });
171
+ it('recovers gracefully when existing file contains non-object YAML', async () => {
172
+ const filePath = join(ctx.tmpDir, 'cases.yml');
173
+ await writeFile(filePath, 'just a string', 'utf-8');
174
+ await writeDataset({ name: 'my_case', input: { q: 'x' } }, filePath);
175
+ const raw = yaml.load(await readFile(filePath, 'utf-8'));
176
+ expect(raw).toHaveProperty('my_case');
177
+ });
178
+ });
179
+ // ---------------------------------------------------------------------------
180
+ // listDatasets
181
+ // ---------------------------------------------------------------------------
182
+ describe('listDatasets', () => {
183
+ it('returns one DatasetInfo per case across all files', async () => {
184
+ const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
185
+ await mkdir(datasetsDir, { recursive: true });
186
+ await writeYaml(join(datasetsDir, 'core.yml'), {
187
+ case_1: { input: { x: 1 }, last_output: { output: { r: 1 }, date: '2026-01-01T00:00:00.000Z' } },
188
+ case_2: { input: { x: 2 } }
189
+ });
190
+ const infos = await listDatasets('my_workflow', ctx.tmpDir);
191
+ expect(infos).toHaveLength(2);
192
+ const case1 = infos.find(i => i.name === 'case_1');
193
+ expect(case1.hasLastOutput).toBe(true);
194
+ expect(case1.path).toContain('core.yml');
195
+ const case2 = infos.find(i => i.name === 'case_2');
196
+ expect(case2.hasLastOutput).toBe(false);
197
+ });
198
+ it('returns empty array when no datasets directory exists', async () => {
199
+ const infos = await listDatasets('nonexistent_workflow', ctx.tmpDir);
200
+ expect(infos).toHaveLength(0);
201
+ });
202
+ });
@@ -1,5 +1,5 @@
1
1
  /**
2
- * Build a workflow from a plan file using the /outputai:build_workflow slash command
2
+ * Build a workflow from a plan file using the /output-build-workflow slash command
3
3
  * @param planFilePath - Absolute path to the plan file
4
4
  * @param workflowDir - Absolute path to the workflow directory
5
5
  * @param workflowName - Name of the workflow
@@ -31,7 +31,7 @@ function isEmpty(modification) {
31
31
  return modification.trim() === '';
32
32
  }
33
33
  /**
34
- * Build a workflow from a plan file using the /outputai:build_workflow slash command
34
+ * Build a workflow from a plan file using the /output-build-workflow slash command
35
35
  * @param planFilePath - Absolute path to the plan file
36
36
  * @param workflowDir - Absolute path to the workflow directory
37
37
  * @param workflowName - Name of the workflow