nah-studio 0.0.1-beta.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/cli.d.ts ADDED
@@ -0,0 +1,13 @@
1
+ export type CliArgs = {
2
+ help: boolean;
3
+ version: boolean;
4
+ port?: number;
5
+ host?: string;
6
+ cwd: string;
7
+ model?: string;
8
+ dbPath?: string;
9
+ nahBin?: string;
10
+ noPublish: boolean;
11
+ };
12
+ /** Hand-rolled, like the CLI it sits beside: a dozen flags do not need a parser. */
13
+ export declare const parseArgs: (argv: string[], env?: NodeJS.ProcessEnv) => CliArgs;
package/dist/cli.js ADDED
@@ -0,0 +1,5 @@
1
+ #!/usr/bin/env node
2
+ import { aI as s } from "./cli-AoWZGiY6.js";
3
+ export {
4
+ s as parseArgs
5
+ };
@@ -0,0 +1,49 @@
1
+ export type StudioEndpoint = {
2
+ /** Base URL, e.g. `http://127.0.0.1:4111`. */
3
+ url: string;
4
+ /** Required only when the Studio is not bound to loopback. */
5
+ token?: string;
6
+ /** The Studio's pid, so a stale file can be told from a live one. */
7
+ pid: number;
8
+ startedAt: number;
9
+ version: string;
10
+ /**
11
+ * The `nah` binary that started this Studio, if it was started by one.
12
+ *
13
+ * Recorded so the Studio launches the same build the user is running. A global
14
+ * `nah` from npm and a `nah` built from source produce different traces, and a
15
+ * dashboard silently watching the wrong one is worse than no dashboard.
16
+ */
17
+ nahBin?: string;
18
+ /** Directory the Studio was pointed at, for the chat tab and experiments. */
19
+ cwd?: string;
20
+ };
21
+ export declare const endpointPath: (home?: string) => string;
22
+ /**
23
+ * Read the endpoint, or null when there is not one.
24
+ *
25
+ * Never throws: every caller treats a missing or corrupt file as "no Studio",
26
+ * because a dashboard that cannot start should not be a dashboard that cannot be
27
+ * read from either.
28
+ */
29
+ export declare const readEndpoint: (path?: string) => Promise<StudioEndpoint | null>;
30
+ /** Write it atomically, so an agent never reads half a file. */
31
+ export declare const writeEndpoint: (endpoint: StudioEndpoint, path?: string) => Promise<void>;
32
+ /**
33
+ * Remove it, but only if it is still ours.
34
+ *
35
+ * Two Studios on one machine is a mistake, not a scenario; whichever exits first
36
+ * would otherwise delete the other's file and leave every agent talking to a
37
+ * closed port.
38
+ */
39
+ export declare const clearEndpoint: (pid: number, path?: string) => Promise<void>;
40
+ /** Whether a process is still running. Signals 0: exists, and we may signal it. */
41
+ export declare const isProcessAlive: (pid: number) => boolean;
42
+ /**
43
+ * The Studio to talk to, or null.
44
+ *
45
+ * A file whose process is gone is not an endpoint. The alternative is every agent
46
+ * on the machine retrying a dead port for as long as the file survives, and a
47
+ * studio the user stopped an hour ago still claiming to be there.
48
+ */
49
+ export declare const liveEndpoint: (path?: string) => Promise<StudioEndpoint | null>;
@@ -0,0 +1,82 @@
1
+ import { LanguageModel } from 'ai';
2
+ import { ExperimentResult, ScoreRecord } from './store.js';
3
+ export type ScoreContext = {
4
+ input: string;
5
+ expected?: string;
6
+ output: string;
7
+ /** Tool names the run called, in order. */
8
+ toolsCalled: string[];
9
+ /** Files the run wrote or edited, repo-relative. */
10
+ filesChanged: string[];
11
+ traceId?: string;
12
+ };
13
+ export type Scorer = {
14
+ id: string;
15
+ name: string;
16
+ description?: string;
17
+ /** `rule` needs no model; `judge` needs one. */
18
+ kind: "rule" | "judge";
19
+ score(context: ScoreContext): Promise<ScoreRecord>;
20
+ };
21
+ /**
22
+ * The one rule worth having by default.
23
+ *
24
+ * Declined rather than zero when the scorer cannot tell, because a scorer that
25
+ * returns 0 for "I don't know" drags an average down with numbers that mean
26
+ * nothing, and the average is the number people quote.
27
+ */
28
+ export declare const includesScorer: (needles: string[], id?: string) => Scorer;
29
+ export declare const calledToolScorer: (toolName: string, id?: string) => Scorer;
30
+ export declare const mentionsFileScorer: (file: string, id?: string) => Scorer;
31
+ /** Refuses to score a refusal, which is often the correct answer. */
32
+ export declare const notRefusedScorer: (id?: string) => Scorer;
33
+ export declare const JUDGE_SYSTEM = "You score one answer to one question against a rubric.\n\nReply with a single JSON object and nothing else:\n{\"score\": <number between 0 and 1>, \"reason\": \"<one or two sentences, citing the specific part of the answer that decided it>\"}";
34
+ /**
35
+ * A model grading another model.
36
+ *
37
+ * Deliberately plain `generateText` with JSON recovery rather than
38
+ * `generateObject`, for the same reason the memory extractor is: models without
39
+ * structured-output support throw `AI_NoObjectGeneratedError`, and a judge that
40
+ * silently fails every time is a score column full of zeros that reads as "the
41
+ * agent is bad at this".
42
+ */
43
+ export declare const judgeScorer: (options: {
44
+ id: string;
45
+ name: string;
46
+ rubric: string;
47
+ model: LanguageModel;
48
+ maxOutputTokens?: number;
49
+ }) => Scorer;
50
+ export type RunOneInput = {
51
+ id: string;
52
+ input: string;
53
+ expected?: string;
54
+ };
55
+ /**
56
+ * Run one dataset item and score it.
57
+ *
58
+ * `execute` is supplied by the caller rather than assumed, so the same engine
59
+ * serves both the CLI (which drives a real session) and a test (which drives a
60
+ * stub) without a branch here.
61
+ *
62
+ * Retried on a provider-level failure. An eval that reports a transient 502 as a
63
+ * failed agent is worse than one that takes a minute longer: the number gets
64
+ * quoted, and it is measuring the provider's Tuesday.
65
+ */
66
+ export declare const runOne: (options: {
67
+ item: RunOneInput;
68
+ scorers: Scorer[];
69
+ execute: (input: RunOneInput) => Promise<{
70
+ output: string;
71
+ toolsCalled: string[];
72
+ filesChanged: string[];
73
+ traceId?: string;
74
+ }>;
75
+ }) => Promise<ExperimentResult>;
76
+ /**
77
+ * Aggregate scores into the shape a comparison needs.
78
+ *
79
+ * Skipped scorers are excluded from the mean, and counted separately, so a judge
80
+ * that failed to answer does not look like a judge that scored badly.
81
+ */
82
+ export declare const summarize: (results: ExperimentResult[]) => Record<string, unknown>;