@outputai/cli 0.1.13-next.91c5d78.0 → 0.1.13-next.934347c.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/api/generated/api.d.ts +322 -84
- package/dist/api/generated/api.js +67 -9
- package/dist/assets/docker/docker-compose-dev.yml +2 -2
- package/dist/commands/credentials/set.d.ts +16 -0
- package/dist/commands/credentials/set.js +125 -0
- package/dist/commands/credentials/set.spec.d.ts +1 -0
- package/dist/commands/credentials/set.spec.js +160 -0
- package/dist/commands/migrate.d.ts +11 -0
- package/dist/commands/migrate.js +40 -0
- package/dist/commands/workflow/list.d.ts +1 -0
- package/dist/commands/workflow/list.js +12 -6
- package/dist/commands/workflow/list.spec.js +21 -0
- package/dist/commands/workflow/plan.js +1 -1
- package/dist/commands/workflow/run.js +2 -1
- package/dist/commands/workflow/run.spec.js +1 -0
- package/dist/commands/workflow/start.js +2 -1
- package/dist/commands/workflow/start.spec.js +1 -0
- package/dist/commands/workflow/test_eval.js +2 -2
- package/dist/components/command_footer.d.ts +8 -0
- package/dist/components/command_footer.js +4 -0
- package/dist/components/status_icon.d.ts +11 -0
- package/dist/components/status_icon.js +25 -0
- package/dist/components/workflow_summary.d.ts +10 -0
- package/dist/components/workflow_summary.js +4 -0
- package/dist/generated/framework_version.json +1 -1
- package/dist/services/claude_client.d.ts +14 -2
- package/dist/services/claude_client.integration.test.js +2 -2
- package/dist/services/claude_client.js +42 -6
- package/dist/services/claude_client.spec.js +3 -3
- package/dist/services/datasets.d.ts +1 -1
- package/dist/services/datasets.js +41 -37
- package/dist/services/datasets.test.d.ts +1 -0
- package/dist/services/datasets.test.js +202 -0
- package/dist/services/workflow_builder.d.ts +1 -1
- package/dist/services/workflow_builder.js +1 -1
- package/dist/utils/date_formatter.d.ts +11 -1
- package/dist/utils/date_formatter.js +26 -1
- package/dist/utils/format_workflow_result.d.ts +3 -3
- package/dist/utils/open_url.d.ts +1 -0
- package/dist/utils/open_url.js +12 -0
- package/dist/views/dev.js +62 -26
- package/dist/views/workflow/list.d.ts +6 -0
- package/dist/views/workflow/list.js +127 -0
- package/package.json +11 -11
|
@@ -114,7 +114,7 @@ export default class WorkflowTest extends Command {
|
|
|
114
114
|
}
|
|
115
115
|
};
|
|
116
116
|
if (save) {
|
|
117
|
-
const filePath = join(dir, `${dataset.name}.yml`);
|
|
117
|
+
const filePath = dataset._source ?? join(dir, `${dataset.name}.yml`);
|
|
118
118
|
await writeDataset(updated, filePath);
|
|
119
119
|
this.log(` Saved output to ${filePath}`);
|
|
120
120
|
}
|
|
@@ -137,7 +137,7 @@ export default class WorkflowTest extends Command {
|
|
|
137
137
|
date: now
|
|
138
138
|
}
|
|
139
139
|
};
|
|
140
|
-
const filePath = join(dir, `${dataset.name}.yml`);
|
|
140
|
+
const filePath = dataset._source ?? join(dir, `${dataset.name}.yml`);
|
|
141
141
|
await writeDataset(updated, filePath);
|
|
142
142
|
this.log(` Saved eval result to ${filePath}`);
|
|
143
143
|
}
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import { jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
|
|
2
|
+
import React from 'react';
|
|
3
|
+
import { Box, Text } from 'ink';
|
|
4
|
+
export const CommandFooter = ({ hints }) => (_jsx(Box, { marginTop: 1, children: hints.map((hint, i) => (_jsxs(React.Fragment, { children: [i > 0 && _jsx(Text, { dimColor: true, children: ' | ' }), _jsx(Text, { dimColor: true, children: '(' }), _jsx(Text, { dimColor: true, bold: true, children: hint.key }), _jsx(Text, { dimColor: true, children: ')' }), _jsx(Text, { dimColor: true, children: ` ${hint.label}` })] }, hint.key))) }));
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
import React from 'react';
|
|
2
|
+
interface StatusDisplay {
|
|
3
|
+
icon: string;
|
|
4
|
+
color: string;
|
|
5
|
+
}
|
|
6
|
+
export declare const resolveStatus: (status: string) => StatusDisplay;
|
|
7
|
+
export declare const statusColor: (status: string) => string;
|
|
8
|
+
export declare const StatusIcon: React.FC<{
|
|
9
|
+
status: string;
|
|
10
|
+
}>;
|
|
11
|
+
export {};
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
import { jsx as _jsx } from "react/jsx-runtime";
|
|
2
|
+
import { Text } from 'ink';
|
|
3
|
+
const STATUS_MAP = {
|
|
4
|
+
// Docker service health
|
|
5
|
+
healthy: { icon: '●', color: 'green' },
|
|
6
|
+
unhealthy: { icon: '○', color: 'red' },
|
|
7
|
+
starting: { icon: '◐', color: 'yellow' },
|
|
8
|
+
none: { icon: '●', color: 'blue' },
|
|
9
|
+
exited: { icon: '✗', color: 'red' },
|
|
10
|
+
// Workflow run status
|
|
11
|
+
running: { icon: '●', color: 'blue' },
|
|
12
|
+
completed: { icon: '●', color: 'green' },
|
|
13
|
+
failed: { icon: '✗', color: 'red' },
|
|
14
|
+
canceled: { icon: '○', color: 'gray' },
|
|
15
|
+
terminated: { icon: '✗', color: 'red' },
|
|
16
|
+
timed_out: { icon: '✗', color: 'red' },
|
|
17
|
+
continued: { icon: '↻', color: 'blue' }
|
|
18
|
+
};
|
|
19
|
+
const DEFAULT_DISPLAY = { icon: '?', color: 'white' };
|
|
20
|
+
export const resolveStatus = (status) => STATUS_MAP[status] ?? DEFAULT_DISPLAY;
|
|
21
|
+
export const statusColor = (status) => resolveStatus(status).color;
|
|
22
|
+
export const StatusIcon = ({ status }) => {
|
|
23
|
+
const { icon, color } = resolveStatus(status);
|
|
24
|
+
return _jsx(Text, { color: color, children: icon });
|
|
25
|
+
};
|
|
@@ -0,0 +1,4 @@
|
|
|
1
|
+
import { jsx as _jsx, jsxs as _jsxs } from "react/jsx-runtime";
|
|
2
|
+
import { Box, Text } from 'ink';
|
|
3
|
+
import { statusColor } from '#components/status_icon.js';
|
|
4
|
+
export const WorkflowSummarySection = ({ summary }) => (_jsxs(Box, { flexDirection: "column", marginTop: 1, children: [_jsx(Text, { bold: true, children: "\uD83D\uDCCB Workflows" }), _jsxs(Box, { marginTop: 1, children: [_jsxs(Text, { color: statusColor('running'), children: [summary.running, " running"] }), _jsx(Text, { children: ", " }), _jsxs(Text, { color: statusColor('failed'), children: [summary.failed, " failed"] }), _jsx(Text, { children: ", " }), _jsxs(Text, { color: statusColor('completed'), children: [summary.completed, " complete"] })] })] }));
|
|
@@ -5,6 +5,7 @@ import { Options } from '@anthropic-ai/claude-agent-sdk';
|
|
|
5
5
|
export declare const ADDITIONAL_INSTRUCTIONS: {
|
|
6
6
|
readonly PLAN: "\n! IMPORTANT !\n1. Use TodoWrite to track your progress through plan creation.\n\n2. Please respond with only the final version of the plan content.\n\n3. Respond in a markdown format with these metadata headers:\n\n---\ntitle: <plan-title>\ndescription: <plan-description>\ndate: <plan-date>\n---\n\n<plan-content>\n\n4. After you mark all todos as complete, you must respond with the final version of the plan.\n\n5. DO NOT write the plan to disk — the CLI will handle saving the file to the plans directory.\n\n6. DO NOT suggest any next steps, follow-up commands, or instructions for the user — the CLI will inform the user of next steps after saving.\n";
|
|
7
7
|
readonly BUILD: "\n! IMPORTANT !\n1. Use TodoWrite to track your progress through workflow implementation.\n\n2. Follow the implementation plan exactly as specified in the plan file.\n\n3. Implement all workflow files following Output.ai patterns and best practices.\n\n4. After you mark all todos as complete, provide a summary of what was implemented.\n";
|
|
8
|
+
readonly MIGRATE: "\n! IMPORTANT !\n1. Use TodoWrite to track your progress through the migration.\n\n2. Fetch the migration guide from https://docs.output.ai/migrations — do not invent migration steps from memory.\n\n3. If the specific guide URL 404s, fall back to the /migrations index and chain the guides that cover the version range.\n\n4. Confirm the planned changes with the user before editing files.\n\n5. After you mark all todos as complete, provide a summary of which files changed and whether the type check passed.\n";
|
|
8
9
|
};
|
|
9
10
|
export declare const PLAN_COMMAND_OPTIONS: Options;
|
|
10
11
|
interface ReplyToClaudeOptions {
|
|
@@ -12,13 +13,14 @@ interface ReplyToClaudeOptions {
|
|
|
12
13
|
applyAdditionalInstructions?: string;
|
|
13
14
|
}
|
|
14
15
|
export declare const BUILD_COMMAND_OPTIONS: Options;
|
|
16
|
+
export declare const MIGRATE_COMMAND_OPTIONS: Options;
|
|
15
17
|
export declare class ClaudeInvocationError extends Error {
|
|
16
18
|
cause?: Error | undefined;
|
|
17
19
|
constructor(message: string, cause?: Error | undefined);
|
|
18
20
|
}
|
|
19
21
|
export declare function replyToClaude(message: string, { anthropicOpts, applyAdditionalInstructions }?: ReplyToClaudeOptions): Promise<string>;
|
|
20
22
|
/**
|
|
21
|
-
* Invoke claude-code with /
|
|
23
|
+
* Invoke claude-code with /output-plan-workflow slash command
|
|
22
24
|
* The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
|
|
23
25
|
* ensureOutputAISystem() scaffolds the command files to that location.
|
|
24
26
|
* @param description - Workflow description
|
|
@@ -26,7 +28,7 @@ export declare function replyToClaude(message: string, { anthropicOpts, applyAdd
|
|
|
26
28
|
*/
|
|
27
29
|
export declare function invokePlanWorkflow(description: string): Promise<string>;
|
|
28
30
|
/**
|
|
29
|
-
* Invoke claude-code with /
|
|
31
|
+
* Invoke claude-code with /output-build-workflow slash command
|
|
30
32
|
* The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
|
|
31
33
|
* ensureOutputAISystem() scaffolds the command files to that location.
|
|
32
34
|
* @param planFilePath - Absolute path to the plan file
|
|
@@ -36,4 +38,14 @@ export declare function invokePlanWorkflow(description: string): Promise<string>
|
|
|
36
38
|
* @returns Implementation output from claude-code
|
|
37
39
|
*/
|
|
38
40
|
export declare function invokeBuildWorkflow(planFilePath: string, workflowDir: string, workflowName: string, additionalInstructions?: string): Promise<string>;
|
|
41
|
+
/**
|
|
42
|
+
* Invoke claude-code with /output-migrate slash command (registered via the output-migrate skill).
|
|
43
|
+
* The slash command fetches migration instructions from docs.output.ai —
|
|
44
|
+
* this CLI wrapper just passes through the version arguments.
|
|
45
|
+
* @param fromVersion - Current framework version (empty string = auto-detect)
|
|
46
|
+
* @param toVersion - Target version (empty string = use npm "latest")
|
|
47
|
+
* @param additionalInstructions - Optional user-supplied guidance
|
|
48
|
+
* @returns Migration summary from claude-code
|
|
49
|
+
*/
|
|
50
|
+
export declare function invokeMigrate(fromVersion: string, toVersion: string, additionalInstructions?: string): Promise<string>;
|
|
39
51
|
export {};
|
|
@@ -15,7 +15,7 @@ describe('invokePlanWorkflow - Integration Tests', () => {
|
|
|
15
15
|
const messages = [];
|
|
16
16
|
try {
|
|
17
17
|
for await (const message of query({
|
|
18
|
-
prompt: `/
|
|
18
|
+
prompt: `/output-plan-workflow ${description}`,
|
|
19
19
|
options: { maxTurns: 1 }
|
|
20
20
|
})) {
|
|
21
21
|
console.log('\nReceived message:', JSON.stringify(message, null, 2));
|
|
@@ -31,7 +31,7 @@ describe('invokePlanWorkflow - Integration Tests', () => {
|
|
|
31
31
|
// This test is just for debugging - we expect messages
|
|
32
32
|
expect(messages.length).toBeGreaterThan(0);
|
|
33
33
|
}, 60000); // 60 second timeout
|
|
34
|
-
it('should successfully invoke /
|
|
34
|
+
it('should successfully invoke /output-plan-workflow slash command and return content', async () => {
|
|
35
35
|
const description = 'Simple workflow that takes a number and doubles it';
|
|
36
36
|
const result = await invokePlanWorkflow(description);
|
|
37
37
|
console.log('\n===== PLAN RESULT =====');
|
|
@@ -38,10 +38,25 @@ date: <plan-date>
|
|
|
38
38
|
3. Implement all workflow files following Output.ai patterns and best practices.
|
|
39
39
|
|
|
40
40
|
4. After you mark all todos as complete, provide a summary of what was implemented.
|
|
41
|
+
`,
|
|
42
|
+
MIGRATE: `
|
|
43
|
+
! IMPORTANT !
|
|
44
|
+
1. Use TodoWrite to track your progress through the migration.
|
|
45
|
+
|
|
46
|
+
2. Fetch the migration guide from https://docs.output.ai/migrations — do not invent migration steps from memory.
|
|
47
|
+
|
|
48
|
+
3. If the specific guide URL 404s, fall back to the /migrations index and chain the guides that cover the version range.
|
|
49
|
+
|
|
50
|
+
4. Confirm the planned changes with the user before editing files.
|
|
51
|
+
|
|
52
|
+
5. After you mark all todos as complete, provide a summary of which files changed and whether the type check passed.
|
|
41
53
|
`
|
|
42
54
|
};
|
|
43
|
-
|
|
44
|
-
|
|
55
|
+
// Slash-command naming convention used by the outputai plugin:
|
|
56
|
+
// `outputai:<kebab-name>` — skills under coding_assistants/.../skills/, which surface as top-level slash commands without the plugin prefix.
|
|
57
|
+
const PLAN_COMMAND = 'outputai:output-plan-workflow';
|
|
58
|
+
const BUILD_COMMAND = 'outputai:output-build-workflow';
|
|
59
|
+
const MIGRATE_COMMAND = 'outputai:output-migrate';
|
|
45
60
|
const GLOBAL_CLAUDE_OPTIONS = {
|
|
46
61
|
settingSources: ['user', 'project', 'local']
|
|
47
62
|
};
|
|
@@ -51,6 +66,9 @@ export const PLAN_COMMAND_OPTIONS = {
|
|
|
51
66
|
export const BUILD_COMMAND_OPTIONS = {
|
|
52
67
|
permissionMode: 'bypassPermissions'
|
|
53
68
|
};
|
|
69
|
+
export const MIGRATE_COMMAND_OPTIONS = {
|
|
70
|
+
permissionMode: 'bypassPermissions'
|
|
71
|
+
};
|
|
54
72
|
export class ClaudeInvocationError extends Error {
|
|
55
73
|
cause;
|
|
56
74
|
constructor(message, cause) {
|
|
@@ -77,7 +95,7 @@ function validateEnvironment() {
|
|
|
77
95
|
}
|
|
78
96
|
}
|
|
79
97
|
function validateSystem(systemMessage) {
|
|
80
|
-
const requiredCommands = [PLAN_COMMAND, BUILD_COMMAND];
|
|
98
|
+
const requiredCommands = [PLAN_COMMAND, BUILD_COMMAND, MIGRATE_COMMAND];
|
|
81
99
|
const availableCommands = systemMessage.slash_commands;
|
|
82
100
|
const missingCommands = requiredCommands.filter(command => !availableCommands.includes(command));
|
|
83
101
|
return {
|
|
@@ -105,7 +123,10 @@ function getTodoWriteMessage(message) {
|
|
|
105
123
|
if (message.type !== 'assistant') {
|
|
106
124
|
return null;
|
|
107
125
|
}
|
|
108
|
-
const todoWriteMessage = message.message.content.find((c) =>
|
|
126
|
+
const todoWriteMessage = message.message.content.find((c) => {
|
|
127
|
+
const block = c;
|
|
128
|
+
return block.type === 'tool_use' && block.name === 'TodoWrite';
|
|
129
|
+
});
|
|
109
130
|
return todoWriteMessage ?? null;
|
|
110
131
|
}
|
|
111
132
|
function applyInstructions(message, instructions) {
|
|
@@ -190,7 +211,7 @@ export async function replyToClaude(message, { anthropicOpts, applyAdditionalIns
|
|
|
190
211
|
return singleQuery(applyInstructions(message, applyAdditionalInstructions), { continue: true, ...anthropicOpts });
|
|
191
212
|
}
|
|
192
213
|
/**
|
|
193
|
-
* Invoke claude-code with /
|
|
214
|
+
* Invoke claude-code with /output-plan-workflow slash command
|
|
194
215
|
* The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
|
|
195
216
|
* ensureOutputAISystem() scaffolds the command files to that location.
|
|
196
217
|
* @param description - Workflow description
|
|
@@ -200,7 +221,7 @@ export async function invokePlanWorkflow(description) {
|
|
|
200
221
|
return singleQuery(applyInstructions(`/${PLAN_COMMAND} ${description}`, ADDITIONAL_INSTRUCTIONS.PLAN), PLAN_COMMAND_OPTIONS);
|
|
201
222
|
}
|
|
202
223
|
/**
|
|
203
|
-
* Invoke claude-code with /
|
|
224
|
+
* Invoke claude-code with /output-build-workflow slash command
|
|
204
225
|
* The SDK loads custom commands from .claude/commands/ when settingSources includes 'project'.
|
|
205
226
|
* ensureOutputAISystem() scaffolds the command files to that location.
|
|
206
227
|
* @param planFilePath - Absolute path to the plan file
|
|
@@ -216,3 +237,18 @@ export async function invokeBuildWorkflow(planFilePath, workflowDir, workflowNam
|
|
|
216
237
|
`/${BUILD_COMMAND} ${commandArgs}`;
|
|
217
238
|
return singleQuery(applyInstructions(fullCommand, ADDITIONAL_INSTRUCTIONS.BUILD), BUILD_COMMAND_OPTIONS);
|
|
218
239
|
}
|
|
240
|
+
/**
|
|
241
|
+
* Invoke claude-code with /output-migrate slash command (registered via the output-migrate skill).
|
|
242
|
+
* The slash command fetches migration instructions from docs.output.ai —
|
|
243
|
+
* this CLI wrapper just passes through the version arguments.
|
|
244
|
+
* @param fromVersion - Current framework version (empty string = auto-detect)
|
|
245
|
+
* @param toVersion - Target version (empty string = use npm "latest")
|
|
246
|
+
* @param additionalInstructions - Optional user-supplied guidance
|
|
247
|
+
* @returns Migration summary from claude-code
|
|
248
|
+
*/
|
|
249
|
+
export async function invokeMigrate(fromVersion, toVersion, additionalInstructions) {
|
|
250
|
+
const from = fromVersion || 'auto';
|
|
251
|
+
const to = toVersion || 'latest';
|
|
252
|
+
const commandArgs = [from, to, additionalInstructions].filter(Boolean).join(' ');
|
|
253
|
+
return singleQuery(applyInstructions(`/${MIGRATE_COMMAND} ${commandArgs}`, ADDITIONAL_INSTRUCTIONS.MIGRATE), MIGRATE_COMMAND_OPTIONS);
|
|
254
|
+
}
|
|
@@ -13,7 +13,7 @@ describe('invokePlanWorkflow', () => {
|
|
|
13
13
|
// Clean up environment variables
|
|
14
14
|
delete process.env.ANTHROPIC_API_KEY;
|
|
15
15
|
});
|
|
16
|
-
it('should invoke /outputai:
|
|
16
|
+
it('should invoke /outputai:output-plan-workflow slash command with settingSources', async () => {
|
|
17
17
|
const { query } = await import('@anthropic-ai/claude-agent-sdk');
|
|
18
18
|
process.env.ANTHROPIC_API_KEY = 'test-key';
|
|
19
19
|
async function* mockIterator() {
|
|
@@ -22,7 +22,7 @@ describe('invokePlanWorkflow', () => {
|
|
|
22
22
|
vi.mocked(query).mockReturnValue(mockIterator());
|
|
23
23
|
await invokePlanWorkflow('Test workflow');
|
|
24
24
|
const calls = vi.mocked(query).mock.calls;
|
|
25
|
-
expect(calls[0]?.[0]?.prompt).toContain('/outputai:
|
|
25
|
+
expect(calls[0]?.[0]?.prompt).toContain('/outputai:output-plan-workflow Test workflow');
|
|
26
26
|
expect(calls[0]?.[0]?.options?.settingSources).toEqual(['user', 'project', 'local']);
|
|
27
27
|
expect(calls[0]?.[0]?.options?.allowedTools).toEqual(['Read', 'Grep', 'WebSearch', 'WebFetch', 'TodoWrite']);
|
|
28
28
|
});
|
|
@@ -36,7 +36,7 @@ describe('invokePlanWorkflow', () => {
|
|
|
36
36
|
const description = 'Build a user authentication system';
|
|
37
37
|
await invokePlanWorkflow(description);
|
|
38
38
|
const calls = vi.mocked(query).mock.calls;
|
|
39
|
-
expect(calls[0]?.[0]?.prompt).toContain(`/outputai:
|
|
39
|
+
expect(calls[0]?.[0]?.prompt).toContain(`/outputai:output-plan-workflow ${description}`);
|
|
40
40
|
expect(calls[0]?.[0]?.options?.settingSources).toEqual(['user', 'project', 'local']);
|
|
41
41
|
});
|
|
42
42
|
it('should return plan output from claude-code', async () => {
|
|
@@ -8,7 +8,7 @@ export interface DatasetInfo {
|
|
|
8
8
|
}
|
|
9
9
|
export declare function resolveDatasetsDir(workflowName: string, basePath?: string): string | null;
|
|
10
10
|
export declare function resolveDefaultDatasetsDir(workflowName: string, basePath?: string): string;
|
|
11
|
-
export declare function
|
|
11
|
+
export declare function readDatasetFile(filePath: string): Promise<Dataset[]>;
|
|
12
12
|
export declare function readAllDatasets(workflowName: string, filterNames?: string[], basePath?: string): Promise<{
|
|
13
13
|
datasets: Dataset[];
|
|
14
14
|
dir: string;
|
|
@@ -24,10 +24,19 @@ export function resolveDefaultDatasetsDir(workflowName, basePath = process.cwd()
|
|
|
24
24
|
// Default to first workflows path
|
|
25
25
|
return resolve(basePath, WORKFLOWS_PATHS[0], workflowName, DATASETS_DIR);
|
|
26
26
|
}
|
|
27
|
-
export async function
|
|
28
|
-
const
|
|
29
|
-
|
|
30
|
-
|
|
27
|
+
export async function readDatasetFile(filePath) {
|
|
28
|
+
const raw = yaml.load(await readFile(filePath, 'utf-8'));
|
|
29
|
+
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
|
|
30
|
+
throw new Error(`Invalid dataset file: ${filePath}`);
|
|
31
|
+
}
|
|
32
|
+
return Object.entries(raw).map(([name, body]) => {
|
|
33
|
+
if (!body || typeof body !== 'object' || !('input' in body)) {
|
|
34
|
+
throw new Error(`Dataset case "${name}" in ${filePath} is missing required "input" field`);
|
|
35
|
+
}
|
|
36
|
+
const dataset = DatasetSchema.parse({ name, ...body });
|
|
37
|
+
dataset._source = filePath;
|
|
38
|
+
return dataset;
|
|
39
|
+
});
|
|
31
40
|
}
|
|
32
41
|
export async function readAllDatasets(workflowName, filterNames, basePath) {
|
|
33
42
|
const dir = resolveDatasetsDir(workflowName, basePath);
|
|
@@ -36,42 +45,35 @@ export async function readAllDatasets(workflowName, filterNames, basePath) {
|
|
|
36
45
|
}
|
|
37
46
|
const files = await readdir(dir);
|
|
38
47
|
const ymlFiles = files.filter(f => f.endsWith('.yml') || f.endsWith('.yaml'));
|
|
48
|
+
const seen = new Set();
|
|
39
49
|
const datasets = [];
|
|
40
50
|
for (const file of ymlFiles) {
|
|
41
|
-
const
|
|
42
|
-
|
|
43
|
-
|
|
51
|
+
const cases = await readDatasetFile(join(dir, file));
|
|
52
|
+
for (const dataset of cases) {
|
|
53
|
+
if (seen.has(dataset.name)) {
|
|
54
|
+
throw new Error(`Duplicate dataset case name "${dataset.name}" found in ${file}`);
|
|
55
|
+
}
|
|
56
|
+
seen.add(dataset.name);
|
|
57
|
+
if (filterNames && !filterNames.includes(dataset.name)) {
|
|
58
|
+
continue;
|
|
59
|
+
}
|
|
60
|
+
datasets.push(dataset);
|
|
44
61
|
}
|
|
45
|
-
datasets.push(dataset);
|
|
46
62
|
}
|
|
47
63
|
return { datasets, dir };
|
|
48
64
|
}
|
|
49
|
-
async function mergeWithExisting(dataset, filePath) {
|
|
50
|
-
if (!existsSync(filePath)) {
|
|
51
|
-
return dataset;
|
|
52
|
-
}
|
|
53
|
-
try {
|
|
54
|
-
const existing = await readDataset(filePath);
|
|
55
|
-
return {
|
|
56
|
-
...existing,
|
|
57
|
-
...dataset,
|
|
58
|
-
ground_truth: dataset.ground_truth ?? existing.ground_truth,
|
|
59
|
-
last_output: dataset.last_output ?? existing.last_output,
|
|
60
|
-
last_eval: dataset.last_eval ?? existing.last_eval
|
|
61
|
-
};
|
|
62
|
-
}
|
|
63
|
-
catch {
|
|
64
|
-
return dataset;
|
|
65
|
-
}
|
|
66
|
-
}
|
|
67
65
|
export async function writeDataset(dataset, filePath) {
|
|
68
|
-
const merged = await mergeWithExisting(dataset, filePath);
|
|
69
66
|
const dir = resolve(filePath, '..');
|
|
70
67
|
if (!existsSync(dir)) {
|
|
71
68
|
await mkdir(dir, { recursive: true });
|
|
72
69
|
}
|
|
73
|
-
const
|
|
74
|
-
|
|
70
|
+
const loaded = existsSync(filePath) ? yaml.load(await readFile(filePath, 'utf-8')) : null;
|
|
71
|
+
const fileObj = (loaded && typeof loaded === 'object' && !Array.isArray(loaded)) ?
|
|
72
|
+
loaded :
|
|
73
|
+
{};
|
|
74
|
+
const { name, _source, ...caseBody } = dataset;
|
|
75
|
+
fileObj[name] = { ...fileObj[name], ...caseBody };
|
|
76
|
+
await writeFile(filePath, yaml.dump(fileObj, { lineWidth: 120, noRefs: true, sortKeys: false }), 'utf-8');
|
|
75
77
|
}
|
|
76
78
|
export async function listDatasets(workflowName, basePath) {
|
|
77
79
|
const dir = resolveDatasetsDir(workflowName, basePath);
|
|
@@ -84,14 +86,16 @@ export async function listDatasets(workflowName, basePath) {
|
|
|
84
86
|
for (const file of ymlFiles) {
|
|
85
87
|
const filePath = join(dir, file);
|
|
86
88
|
try {
|
|
87
|
-
const
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
89
|
+
const cases = await readDatasetFile(filePath);
|
|
90
|
+
for (const dataset of cases) {
|
|
91
|
+
results.push({
|
|
92
|
+
name: dataset.name,
|
|
93
|
+
path: filePath,
|
|
94
|
+
hasLastOutput: dataset.last_output?.output !== undefined,
|
|
95
|
+
lastOutputDate: dataset.last_output?.date,
|
|
96
|
+
lastEvalDate: dataset.last_eval?.date
|
|
97
|
+
});
|
|
98
|
+
}
|
|
95
99
|
}
|
|
96
100
|
catch {
|
|
97
101
|
results.push({
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,202 @@
|
|
|
1
|
+
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
|
2
|
+
import { mkdtemp, rm, writeFile, readFile, mkdir } from 'node:fs/promises';
|
|
3
|
+
import { join } from 'node:path';
|
|
4
|
+
import { tmpdir } from 'node:os';
|
|
5
|
+
import yaml from 'js-yaml';
|
|
6
|
+
import { readDatasetFile, readAllDatasets, writeDataset, listDatasets } from './datasets.js';
|
|
7
|
+
const ctx = { tmpDir: '' };
|
|
8
|
+
beforeEach(async () => {
|
|
9
|
+
ctx.tmpDir = await mkdtemp(join(tmpdir(), 'output-datasets-test-'));
|
|
10
|
+
});
|
|
11
|
+
afterEach(async () => {
|
|
12
|
+
await rm(ctx.tmpDir, { recursive: true, force: true });
|
|
13
|
+
});
|
|
14
|
+
function writeYaml(filePath, obj) {
|
|
15
|
+
return writeFile(filePath, yaml.dump(obj, { lineWidth: 120, noRefs: true, sortKeys: false }), 'utf-8');
|
|
16
|
+
}
|
|
17
|
+
// ---------------------------------------------------------------------------
|
|
18
|
+
// readDatasetFile
|
|
19
|
+
// ---------------------------------------------------------------------------
|
|
20
|
+
describe('readDatasetFile', () => {
|
|
21
|
+
it('parses a multi-case file and returns all cases', async () => {
|
|
22
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
23
|
+
await writeYaml(filePath, {
|
|
24
|
+
case_a: { input: { query: 'foo' }, ground_truth: { expected: 1 } },
|
|
25
|
+
case_b: { input: { query: 'bar' } },
|
|
26
|
+
case_c: { input: { query: 'baz' }, ground_truth: { expected: 3 } }
|
|
27
|
+
});
|
|
28
|
+
const datasets = await readDatasetFile(filePath);
|
|
29
|
+
expect(datasets).toHaveLength(3);
|
|
30
|
+
expect(datasets.map(d => d.name)).toEqual(['case_a', 'case_b', 'case_c']);
|
|
31
|
+
expect(datasets[0].input).toEqual({ query: 'foo' });
|
|
32
|
+
expect(datasets[0].ground_truth).toEqual({ expected: 1 });
|
|
33
|
+
});
|
|
34
|
+
it('attaches _source with the absolute file path to each dataset', async () => {
|
|
35
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
36
|
+
await writeYaml(filePath, {
|
|
37
|
+
my_case: { input: { x: 1 } }
|
|
38
|
+
});
|
|
39
|
+
const [dataset] = await readDatasetFile(filePath);
|
|
40
|
+
expect(dataset._source).toBe(filePath);
|
|
41
|
+
});
|
|
42
|
+
it('throws when a case is missing input', async () => {
|
|
43
|
+
const filePath = join(ctx.tmpDir, 'bad.yml');
|
|
44
|
+
await writeYaml(filePath, {
|
|
45
|
+
good_case: { input: { x: 1 } },
|
|
46
|
+
bad_case: { ground_truth: { expected: 42 } }
|
|
47
|
+
});
|
|
48
|
+
await expect(readDatasetFile(filePath)).rejects.toThrow('Dataset case "bad_case" in');
|
|
49
|
+
});
|
|
50
|
+
it('throws when file content is not an object', async () => {
|
|
51
|
+
const filePath = join(ctx.tmpDir, 'bad.yml');
|
|
52
|
+
await writeFile(filePath, 'just a string', 'utf-8');
|
|
53
|
+
await expect(readDatasetFile(filePath)).rejects.toThrow('Invalid dataset file');
|
|
54
|
+
});
|
|
55
|
+
it('throws with clear message when file content is a YAML array', async () => {
|
|
56
|
+
const filePath = join(ctx.tmpDir, 'bad.yml');
|
|
57
|
+
await writeFile(filePath, '- foo\n- bar\n', 'utf-8');
|
|
58
|
+
await expect(readDatasetFile(filePath)).rejects.toThrow('Invalid dataset file');
|
|
59
|
+
});
|
|
60
|
+
it('preserves last_output and last_eval fields', async () => {
|
|
61
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
62
|
+
await writeYaml(filePath, {
|
|
63
|
+
cached_case: {
|
|
64
|
+
input: { q: 'hello' },
|
|
65
|
+
last_output: { output: { result: 42 }, executionTimeMs: 100, date: '2026-01-01T00:00:00.000Z' }
|
|
66
|
+
}
|
|
67
|
+
});
|
|
68
|
+
const [dataset] = await readDatasetFile(filePath);
|
|
69
|
+
expect(dataset.last_output?.output).toEqual({ result: 42 });
|
|
70
|
+
expect(dataset.last_output?.executionTimeMs).toBe(100);
|
|
71
|
+
});
|
|
72
|
+
});
|
|
73
|
+
// ---------------------------------------------------------------------------
|
|
74
|
+
// readAllDatasets
|
|
75
|
+
// ---------------------------------------------------------------------------
|
|
76
|
+
describe('readAllDatasets', () => {
|
|
77
|
+
it('flattens cases from multiple files', async () => {
|
|
78
|
+
const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
|
|
79
|
+
await mkdir(datasetsDir, { recursive: true });
|
|
80
|
+
await writeYaml(join(datasetsDir, 'group_a.yml'), {
|
|
81
|
+
case_1: { input: { x: 1 } },
|
|
82
|
+
case_2: { input: { x: 2 } }
|
|
83
|
+
});
|
|
84
|
+
await writeYaml(join(datasetsDir, 'group_b.yml'), {
|
|
85
|
+
case_3: { input: { x: 3 } }
|
|
86
|
+
});
|
|
87
|
+
const { datasets } = await readAllDatasets('my_workflow', undefined, ctx.tmpDir);
|
|
88
|
+
expect(datasets).toHaveLength(3);
|
|
89
|
+
expect(datasets.map(d => d.name).sort()).toEqual(['case_1', 'case_2', 'case_3']);
|
|
90
|
+
});
|
|
91
|
+
it('filters by case name across files', async () => {
|
|
92
|
+
const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
|
|
93
|
+
await mkdir(datasetsDir, { recursive: true });
|
|
94
|
+
await writeYaml(join(datasetsDir, 'group_a.yml'), {
|
|
95
|
+
case_1: { input: { x: 1 } },
|
|
96
|
+
case_2: { input: { x: 2 } }
|
|
97
|
+
});
|
|
98
|
+
await writeYaml(join(datasetsDir, 'group_b.yml'), {
|
|
99
|
+
case_3: { input: { x: 3 } }
|
|
100
|
+
});
|
|
101
|
+
const { datasets } = await readAllDatasets('my_workflow', ['case_2', 'case_3'], ctx.tmpDir);
|
|
102
|
+
expect(datasets).toHaveLength(2);
|
|
103
|
+
expect(datasets.map(d => d.name).sort()).toEqual(['case_2', 'case_3']);
|
|
104
|
+
});
|
|
105
|
+
it('returns empty datasets and a default dir when workflow has no datasets dir', async () => {
|
|
106
|
+
const { datasets, dir } = await readAllDatasets('nonexistent_workflow', undefined, ctx.tmpDir);
|
|
107
|
+
expect(datasets).toHaveLength(0);
|
|
108
|
+
expect(dir).toContain('nonexistent_workflow');
|
|
109
|
+
});
|
|
110
|
+
it('throws when the same case name appears in two different files', async () => {
|
|
111
|
+
const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
|
|
112
|
+
await mkdir(datasetsDir, { recursive: true });
|
|
113
|
+
await writeYaml(join(datasetsDir, 'group_a.yml'), { case_1: { input: { x: 1 } } });
|
|
114
|
+
await writeYaml(join(datasetsDir, 'group_b.yml'), { case_1: { input: { x: 2 } } });
|
|
115
|
+
await expect(readAllDatasets('my_workflow', undefined, ctx.tmpDir)).rejects.toThrow('Duplicate dataset case name "case_1"');
|
|
116
|
+
});
|
|
117
|
+
});
|
|
118
|
+
// ---------------------------------------------------------------------------
|
|
119
|
+
// writeDataset
|
|
120
|
+
// ---------------------------------------------------------------------------
|
|
121
|
+
describe('writeDataset', () => {
|
|
122
|
+
it('creates a new file with one case keyed by name', async () => {
|
|
123
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
124
|
+
const dataset = { name: 'new_case', input: { q: 'hello' } };
|
|
125
|
+
await writeDataset(dataset, filePath);
|
|
126
|
+
const raw = yaml.load(await readFile(filePath, 'utf-8'));
|
|
127
|
+
expect(raw).toHaveProperty('new_case');
|
|
128
|
+
expect(raw.new_case.input).toEqual({ q: 'hello' });
|
|
129
|
+
expect(raw.new_case).not.toHaveProperty('name');
|
|
130
|
+
});
|
|
131
|
+
it('does not write _source into the file', async () => {
|
|
132
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
133
|
+
const dataset = { name: 'my_case', input: { q: 'x' }, _source: '/some/path.yml' };
|
|
134
|
+
await writeDataset(dataset, filePath);
|
|
135
|
+
const raw = yaml.load(await readFile(filePath, 'utf-8'));
|
|
136
|
+
expect(raw.my_case).not.toHaveProperty('_source');
|
|
137
|
+
});
|
|
138
|
+
it('updates only the target case, leaving other cases untouched', async () => {
|
|
139
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
140
|
+
await writeYaml(filePath, {
|
|
141
|
+
case_a: { input: { x: 1 }, ground_truth: { expected: 1 } },
|
|
142
|
+
case_b: { input: { x: 2 }, ground_truth: { expected: 2 } }
|
|
143
|
+
});
|
|
144
|
+
await writeDataset({ name: 'case_a', input: { x: 1 }, last_output: { output: { result: 1 }, date: '2026-01-01T00:00:00.000Z' } }, filePath);
|
|
145
|
+
const raw = yaml.load(await readFile(filePath, 'utf-8'));
|
|
146
|
+
expect(raw).toHaveProperty('case_b');
|
|
147
|
+
expect(raw.case_b.ground_truth).toEqual({ expected: 2 });
|
|
148
|
+
});
|
|
149
|
+
it('preserves existing fields when writing last_output then last_eval', async () => {
|
|
150
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
151
|
+
await writeYaml(filePath, {
|
|
152
|
+
my_case: { input: { q: 'hello' }, ground_truth: { expected: 42 } }
|
|
153
|
+
});
|
|
154
|
+
await writeDataset({ name: 'my_case', input: { q: 'hello' }, last_output: { output: { result: 42 }, executionTimeMs: 50, date: '2026-01-01T00:00:00.000Z' } }, filePath);
|
|
155
|
+
await writeDataset({
|
|
156
|
+
name: 'my_case', input: { q: 'hello' },
|
|
157
|
+
last_eval: { output: { datasetName: 'my_case', verdict: 'pass', evaluators: [] }, date: '2026-01-01T00:01:00.000Z' }
|
|
158
|
+
}, filePath);
|
|
159
|
+
const raw = yaml.load(await readFile(filePath, 'utf-8'));
|
|
160
|
+
const caseObj = raw.my_case;
|
|
161
|
+
expect(caseObj).toHaveProperty('last_output');
|
|
162
|
+
expect(caseObj).toHaveProperty('last_eval');
|
|
163
|
+
expect(caseObj).toHaveProperty('ground_truth');
|
|
164
|
+
});
|
|
165
|
+
it('creates parent directories if they do not exist', async () => {
|
|
166
|
+
const filePath = join(ctx.tmpDir, 'deep', 'nested', 'cases.yml');
|
|
167
|
+
await writeDataset({ name: 'my_case', input: { q: 'x' } }, filePath);
|
|
168
|
+
const raw = yaml.load(await readFile(filePath, 'utf-8'));
|
|
169
|
+
expect(raw).toHaveProperty('my_case');
|
|
170
|
+
});
|
|
171
|
+
it('recovers gracefully when existing file contains non-object YAML', async () => {
|
|
172
|
+
const filePath = join(ctx.tmpDir, 'cases.yml');
|
|
173
|
+
await writeFile(filePath, 'just a string', 'utf-8');
|
|
174
|
+
await writeDataset({ name: 'my_case', input: { q: 'x' } }, filePath);
|
|
175
|
+
const raw = yaml.load(await readFile(filePath, 'utf-8'));
|
|
176
|
+
expect(raw).toHaveProperty('my_case');
|
|
177
|
+
});
|
|
178
|
+
});
|
|
179
|
+
// ---------------------------------------------------------------------------
|
|
180
|
+
// listDatasets
|
|
181
|
+
// ---------------------------------------------------------------------------
|
|
182
|
+
describe('listDatasets', () => {
|
|
183
|
+
it('returns one DatasetInfo per case across all files', async () => {
|
|
184
|
+
const datasetsDir = join(ctx.tmpDir, 'src', 'workflows', 'my_workflow', 'tests', 'datasets');
|
|
185
|
+
await mkdir(datasetsDir, { recursive: true });
|
|
186
|
+
await writeYaml(join(datasetsDir, 'core.yml'), {
|
|
187
|
+
case_1: { input: { x: 1 }, last_output: { output: { r: 1 }, date: '2026-01-01T00:00:00.000Z' } },
|
|
188
|
+
case_2: { input: { x: 2 } }
|
|
189
|
+
});
|
|
190
|
+
const infos = await listDatasets('my_workflow', ctx.tmpDir);
|
|
191
|
+
expect(infos).toHaveLength(2);
|
|
192
|
+
const case1 = infos.find(i => i.name === 'case_1');
|
|
193
|
+
expect(case1.hasLastOutput).toBe(true);
|
|
194
|
+
expect(case1.path).toContain('core.yml');
|
|
195
|
+
const case2 = infos.find(i => i.name === 'case_2');
|
|
196
|
+
expect(case2.hasLastOutput).toBe(false);
|
|
197
|
+
});
|
|
198
|
+
it('returns empty array when no datasets directory exists', async () => {
|
|
199
|
+
const infos = await listDatasets('nonexistent_workflow', ctx.tmpDir);
|
|
200
|
+
expect(infos).toHaveLength(0);
|
|
201
|
+
});
|
|
202
|
+
});
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Build a workflow from a plan file using the /
|
|
2
|
+
* Build a workflow from a plan file using the /output-build-workflow slash command
|
|
3
3
|
* @param planFilePath - Absolute path to the plan file
|
|
4
4
|
* @param workflowDir - Absolute path to the workflow directory
|
|
5
5
|
* @param workflowName - Name of the workflow
|
|
@@ -31,7 +31,7 @@ function isEmpty(modification) {
|
|
|
31
31
|
return modification.trim() === '';
|
|
32
32
|
}
|
|
33
33
|
/**
|
|
34
|
-
* Build a workflow from a plan file using the /
|
|
34
|
+
* Build a workflow from a plan file using the /output-build-workflow slash command
|
|
35
35
|
* @param planFilePath - Absolute path to the plan file
|
|
36
36
|
* @param workflowDir - Absolute path to the workflow directory
|
|
37
37
|
* @param workflowName - Name of the workflow
|