@telvine/cli 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -6,7 +6,7 @@ Instrument agent plugins with Telvine.
6
6
  npm i -g @telvine/cli@0.1.0
7
7
  telvine login
8
8
  telvine publish ./my-plugin --dry-run
9
- telvine publish ./my-plugin
9
+ telvine publish ./my-plugin --marketplace-url https://github.com/acme/agent-plugin-marketplace.git
10
10
  ```
11
11
 
12
12
  Package: [`@telvine/cli@0.1.0` on npm](https://www.npmjs.com/package/@telvine/cli).
@@ -20,7 +20,8 @@ and assets as versioned components.
20
20
  `.claude-plugin/plugin.json`, discovers Skills and `evals/**/cases.jsonl`, and
21
21
  prints the plugin publish plan without API calls or file writes. Non-dry-run
22
22
  publish registers the plugin, component inventory, Skill versions, eval suites,
23
- and prints a plugin-scoped write key once.
23
+ and prints a plugin-scoped write key once. Use `--marketplace-url` to store the
24
+ Git marketplace or source URL that users should install from for that version.
24
25
 
25
26
  After publishing, set the printed `TELVINE_PLUGIN_ID` and `TELVINE_WRITE_KEY`
26
27
  in the plugin runtime, run the plugin once, then inspect production data:
@@ -32,3 +33,11 @@ telvine plugins events <plugin-id> --limit 20
32
33
 
33
34
  Telemetry uses a closed event envelope. Do not send prompts, file contents,
34
35
  connector payloads, tool arguments, or model outputs.
36
+
37
+ ## Promptfoo evaluation adapter
38
+
39
+ `telvine eval run ./my-plugin --suite SUITE --plugin-id PLUGIN_ID --version VERSION --trials 3 --engine-version PROMPTFOO_VERSION` delegates to an installed Promptfoo executable. Keep native `promptfooconfig.yaml`/`.json`, synthetic `cases.jsonl`, and optional release metadata in `suite.json` under the suite directory. Each test needs `metadata.telvine_case_key`. Full result exports stay local; only allowlisted numeric/boolean result metadata, identities, and hashes reach Telvine.
40
+
41
+ `telvine eval import RESULTS.json --suite-id SUITE_ID --source-hash HASH --version VERSION --trials 3 --engine-version PROMPTFOO_VERSION` records an existing v2/v3 export. CI can supply `TELVINE_READ_TOKEN`, `TELVINE_WRITE_TOKEN`, and `TELVINE_API_BASE_URL`. Prepublish the suite before CI runs. Both commands exit nonzero when the gate fails.
42
+
43
+ See the [evaluation guide](https://telvine.com/learn/how-to-evaluate-agent-plugins-with-telvine) for immutable revisions, trial preservation, calibration, and release checks. Telvine registration does not execute tests or distribute packages.
package/dist/telvine.js CHANGED
@@ -2639,47 +2639,6 @@ var require_wrap_ansi = __commonJS((exports, module) => {
2639
2639
  };
2640
2640
  });
2641
2641
 
2642
- // ../../node_modules/.pnpm/keytar@7.9.0/node_modules/keytar/build/Release/keytar.node
2643
- var require_keytar = __commonJS((exports, module) => {
2644
- module.exports = __require("./keytar-f6bnxfss.node");
2645
- });
2646
-
2647
- // ../../node_modules/.pnpm/keytar@7.9.0/node_modules/keytar/lib/keytar.js
2648
- var require_keytar2 = __commonJS((exports, module) => {
2649
- var keytar = require_keytar();
2650
- function checkRequired(val, name) {
2651
- if (!val || val.length <= 0) {
2652
- throw new Error(name + " is required.");
2653
- }
2654
- }
2655
- module.exports = {
2656
- getPassword: function(service, account) {
2657
- checkRequired(service, "Service");
2658
- checkRequired(account, "Account");
2659
- return keytar.getPassword(service, account);
2660
- },
2661
- setPassword: function(service, account, password) {
2662
- checkRequired(service, "Service");
2663
- checkRequired(account, "Account");
2664
- checkRequired(password, "Password");
2665
- return keytar.setPassword(service, account, password);
2666
- },
2667
- deletePassword: function(service, account) {
2668
- checkRequired(service, "Service");
2669
- checkRequired(account, "Account");
2670
- return keytar.deletePassword(service, account);
2671
- },
2672
- findPassword: function(service) {
2673
- checkRequired(service, "Service");
2674
- return keytar.findPassword(service);
2675
- },
2676
- findCredentials: function(service) {
2677
- checkRequired(service, "Service");
2678
- return keytar.findCredentials(service);
2679
- }
2680
- };
2681
- });
2682
-
2683
2642
  // ../../node_modules/.pnpm/kind-of@6.0.3/node_modules/kind-of/index.js
2684
2643
  var require_kind_of = __commonJS((exports, module) => {
2685
2644
  var toString = Object.prototype.toString;
@@ -9416,7 +9375,7 @@ async function tryKeytar() {
9416
9375
  if (keytar)
9417
9376
  return keytar;
9418
9377
  try {
9419
- const mod = await Promise.resolve().then(() => __toESM(require_keytar2(), 1));
9378
+ const mod = await import("keytar");
9420
9379
  keytar = mod.default ?? mod;
9421
9380
  if (typeof keytar.setPassword !== "function")
9422
9381
  throw new Error("keytar module shape unexpected");
@@ -9498,15 +9457,25 @@ class ApiError extends Error {
9498
9457
  status;
9499
9458
  body;
9500
9459
  constructor(status, body) {
9501
- super(`api_error_${status}`);
9460
+ super(formatApiError(status, body));
9502
9461
  this.status = status;
9503
9462
  this.body = body;
9504
9463
  }
9505
9464
  }
9465
+ function formatApiError(status, body) {
9466
+ if (body && typeof body === "object" && "message" in body && typeof body.message === "string") {
9467
+ return body.message;
9468
+ }
9469
+ if (body && typeof body === "object" && "error" in body && typeof body.error === "string") {
9470
+ return body.error;
9471
+ }
9472
+ return `api_error_${status}`;
9473
+ }
9506
9474
  async function api(path2, init = {}) {
9507
9475
  const cfg = await readConfig();
9508
- const token = await getSecret("clerk_jwt") ?? "";
9509
- const res = await fetch(`${cfg.api_base_url}${path2}`, {
9476
+ const isRead = !init.method || init.method === "GET";
9477
+ const token = (isRead ? process.env.TELVINE_READ_TOKEN : process.env.TELVINE_WRITE_TOKEN) ?? process.env.TELVINE_TOKEN ?? await getSecret("clerk_jwt") ?? "";
9478
+ const res = await fetch(`${process.env.TELVINE_API_BASE_URL ?? cfg.api_base_url}${path2}`, {
9510
9479
  ...init,
9511
9480
  headers: {
9512
9481
  "content-type": "application/json",
@@ -13911,6 +13880,7 @@ var propsByEventType = {
13911
13880
  "feedback.submitted": exports_external.object({
13912
13881
  rating: exports_external.number().int().min(1).max(5).optional(),
13913
13882
  comment: exports_external.string().max(2000).optional(),
13883
+ source: exports_external.enum(["user", "reviewer", "agent"]).optional(),
13914
13884
  task_category: exports_external.enum(TASK_CATEGORIES).optional(),
13915
13885
  session_id: exports_external.string().min(1).max(128).optional()
13916
13886
  }).strict(),
@@ -13955,7 +13925,9 @@ var propsByEventType = {
13955
13925
  score: exports_external.number().min(0).max(1).optional(),
13956
13926
  failure_cluster: exports_external.string().max(160).optional(),
13957
13927
  error_class: exports_external.string().max(128).optional(),
13958
- duration_ms: exports_external.number().int().nonnegative().max(86400000).optional()
13928
+ duration_ms: exports_external.number().int().nonnegative().max(86400000).optional(),
13929
+ agent_experience_score: exports_external.number().int().min(1).max(5).optional(),
13930
+ agent_experience_feedback: exports_external.string().max(2000).optional()
13959
13931
  }).strict(),
13960
13932
  "skill.eval.run.completed": exports_external.object({
13961
13933
  eval_suite_id: exports_external.string().min(1).max(128),
@@ -13992,7 +13964,9 @@ var propsByEventType = {
13992
13964
  score: exports_external.number().min(0).max(1).optional(),
13993
13965
  failure_cluster: exports_external.string().max(160).optional(),
13994
13966
  error_class: exports_external.string().max(128).optional(),
13995
- duration_ms: exports_external.number().int().nonnegative().max(86400000).optional()
13967
+ duration_ms: exports_external.number().int().nonnegative().max(86400000).optional(),
13968
+ agent_experience_score: exports_external.number().int().min(1).max(5).optional(),
13969
+ agent_experience_feedback: exports_external.string().max(2000).optional()
13996
13970
  }).strict(),
13997
13971
  "plugin.eval.run.completed": exports_external.object({
13998
13972
  eval_suite_id: exports_external.string().min(1).max(128),
@@ -14193,9 +14167,135 @@ async function writeSkillMd(path2, content) {
14193
14167
  }
14194
14168
 
14195
14169
  // src/lib/plugin-file.ts
14196
- import { createHash as createHash2, randomUUID } from "node:crypto";
14197
- import { readFile as readFile4, readdir as readdir2, stat as stat2 } from "node:fs/promises";
14170
+ import { createHash as createHash2 } from "node:crypto";
14171
+ import { readFile as readFile4, readdir as readdir2, stat as stat2, realpath } from "node:fs/promises";
14198
14172
  import { basename, join as join4, relative as relative2 } from "node:path";
14173
+
14174
+ // ../../packages/schema/src/evals.ts
14175
+ var evalSuiteStatusSchema = exports_external.enum(["draft", "active", "deprecated"]);
14176
+ var evalRunStatusSchema = exports_external.enum(["queued", "running", "completed", "failed"]);
14177
+ var evalCaseResultStatusSchema = exports_external.enum(["passed", "failed", "skipped", "errored"]);
14178
+ var evalHashSchema = exports_external.string().regex(/^[a-f0-9]{64}$/);
14179
+ var evalMetadataKeySchema = exports_external.string().regex(/^[A-Za-z0-9][A-Za-z0-9_.:-]{0,127}$/);
14180
+ var evalGraderSchema = exports_external.object({
14181
+ id: evalMetadataKeySchema,
14182
+ type: exports_external.enum(["deterministic", "outcome", "model", "pairwise", "human", "similarity"]),
14183
+ target: exports_external.enum(["activation", "trajectory", "output", "state"]).default("output"),
14184
+ version: exports_external.string().min(1).max(64),
14185
+ config_hash: evalHashSchema,
14186
+ model: evalMetadataKeySchema.optional(),
14187
+ required: exports_external.boolean().default(true),
14188
+ threshold: exports_external.number().min(0).max(1).default(1),
14189
+ calibration_ref: evalMetadataKeySchema.optional()
14190
+ }).strict();
14191
+ var evalGatePolicySchema = exports_external.object({
14192
+ required_tags: exports_external.array(evalMetadataKeySchema).max(50).default([]),
14193
+ min_trials: exports_external.number().int().min(1).max(100).default(1),
14194
+ require_calibration: exports_external.boolean().default(false),
14195
+ calibration_min_samples: exports_external.number().int().min(1).default(20),
14196
+ calibration_min_agreement: exports_external.number().min(0).max(1).default(0.9),
14197
+ calibration_max_false_pass_rate: exports_external.number().min(0).max(1).default(0.05),
14198
+ require_baseline: exports_external.boolean().default(false),
14199
+ max_regressions: exports_external.number().int().nonnegative().default(0)
14200
+ }).strict();
14201
+ var evalAssertionResultSchema = exports_external.object({
14202
+ grader_id: evalMetadataKeySchema,
14203
+ status: exports_external.enum(["passed", "failed", "errored", "needs_review"]),
14204
+ score: exports_external.number().min(0).max(1).nullable().optional(),
14205
+ config_hash: evalHashSchema,
14206
+ evidence_ref: evalMetadataKeySchema.optional()
14207
+ }).strict();
14208
+ var evalRubricSchema = exports_external.object({
14209
+ dimensions: exports_external.array(exports_external.object({
14210
+ name: exports_external.string().min(1).max(80),
14211
+ description: exports_external.string().max(1000).optional(),
14212
+ weight: exports_external.number().positive().max(100).optional()
14213
+ }).strict()).max(20).optional()
14214
+ }).catchall(exports_external.unknown());
14215
+ var evalFixtureRefsSchema = exports_external.array(exports_external.object({
14216
+ name: exports_external.string().min(1).max(128),
14217
+ path: exports_external.string().min(1).max(512).refine((path2) => !path2.startsWith("/") && !path2.split("/").includes(".."), "expected plugin-relative path"),
14218
+ synthetic: exports_external.boolean().optional(),
14219
+ content_hash: evalHashSchema.optional()
14220
+ }).strict()).max(100);
14221
+ var evalCaseInputSchema = exports_external.object({
14222
+ case_key: exports_external.string().min(1).max(128),
14223
+ name: exports_external.string().min(1).max(200),
14224
+ task_category: exports_external.string().min(1).max(80).optional(),
14225
+ scenario: exports_external.string().min(1).max(4000),
14226
+ expected_outcome: exports_external.string().min(1).max(4000),
14227
+ rubric: evalRubricSchema.default({}),
14228
+ fixture_refs: evalFixtureRefsSchema.default([]),
14229
+ tags: exports_external.array(exports_external.string().min(1).max(64)).max(50).default([]),
14230
+ weight: exports_external.number().int().min(1).max(100).default(1),
14231
+ required: exports_external.boolean().default(true)
14232
+ }).strict();
14233
+ var evalSuiteInputSchema = exports_external.object({
14234
+ slug: exports_external.string().regex(/^[a-z0-9][a-z0-9-]{1,62}[a-z0-9]$/),
14235
+ name: exports_external.string().min(1).max(200),
14236
+ description: exports_external.string().max(2000).optional(),
14237
+ version: exports_external.string().min(1).max(64),
14238
+ skill_id: exports_external.string().optional(),
14239
+ source_path: exports_external.string().min(1).max(512).refine((path2) => !path2.startsWith("/") && !path2.split("/").includes(".."), "expected plugin-relative path").optional(),
14240
+ source_hash: evalHashSchema.optional(),
14241
+ graders: exports_external.array(evalGraderSchema).max(100).default([]),
14242
+ gate_policy: evalGatePolicySchema.default({}),
14243
+ purpose: exports_external.enum(["regression", "capability", "calibration", "holdout"]).default("regression"),
14244
+ status: evalSuiteStatusSchema.default("draft"),
14245
+ pass_threshold: exports_external.number().min(0).max(1).default(0.8),
14246
+ promotion_gate: exports_external.string().max(2000).optional(),
14247
+ agent_experience_feedback_enabled: exports_external.boolean().default(true),
14248
+ cases: exports_external.array(evalCaseInputSchema).max(1000).default([])
14249
+ }).strict();
14250
+ var evalRunInputSchema = exports_external.object({
14251
+ plugin_version: exports_external.string().min(1).max(64).optional(),
14252
+ skill_version: exports_external.string().min(1).max(64).optional(),
14253
+ baseline_version: exports_external.string().min(1).max(64).optional(),
14254
+ runtime: exports_external.string().max(64).optional(),
14255
+ model: exports_external.string().max(128).optional(),
14256
+ harness: exports_external.string().max(128).optional(),
14257
+ agent_experience_feedback_enabled: exports_external.boolean().optional(),
14258
+ status: exports_external.enum(["queued", "running"]).default("running"),
14259
+ trials: exports_external.number().int().min(1).max(100).default(1),
14260
+ suite_revision_hash: evalHashSchema.optional(),
14261
+ grader_config_hash: evalHashSchema.optional(),
14262
+ engine_version: exports_external.string().max(64).optional(),
14263
+ evaluator_variant: evalMetadataKeySchema.optional(),
14264
+ baseline_run_id: exports_external.string().min(1).optional(),
14265
+ parent_run_id: exports_external.string().min(1).optional(),
14266
+ experiment_id: evalMetadataKeySchema.optional()
14267
+ }).strict();
14268
+ var evalCaseResultInputSchema = exports_external.object({
14269
+ eval_case_id: exports_external.string().min(1),
14270
+ status: evalCaseResultStatusSchema,
14271
+ score: exports_external.number().min(0).max(1).nullable().optional(),
14272
+ rubric_results: exports_external.record(evalMetadataKeySchema, exports_external.union([exports_external.boolean(), exports_external.number().finite(), exports_external.null()])).default({}),
14273
+ trial: exports_external.number().int().min(1).max(100).default(1),
14274
+ assertions: exports_external.array(evalAssertionResultSchema).max(100).default([]),
14275
+ cost_usd: exports_external.number().finite().nonnegative().optional(),
14276
+ failure_cluster: exports_external.string().max(160).optional(),
14277
+ error_class: exports_external.string().max(128).optional(),
14278
+ duration_ms: exports_external.number().int().nonnegative().max(86400000).optional(),
14279
+ evidence_ref: exports_external.string().max(512).optional(),
14280
+ output_hash: exports_external.string().length(64).optional(),
14281
+ agent_experience_score: exports_external.number().int().min(1).max(5).nullable().optional(),
14282
+ agent_experience_feedback: exports_external.string().max(2000).optional()
14283
+ }).strict();
14284
+ var evalCalibrationInputSchema = exports_external.object({
14285
+ grader_id: evalMetadataKeySchema,
14286
+ config_hash: evalHashSchema,
14287
+ rubric_hash: evalHashSchema,
14288
+ dataset_hash: evalHashSchema,
14289
+ split: exports_external.literal("holdout"),
14290
+ samples: exports_external.array(exports_external.object({
14291
+ sample_id: evalMetadataKeySchema,
14292
+ human: exports_external.enum(["passed", "failed"]),
14293
+ judge: exports_external.enum(["passed", "failed", "errored", "needs_review"]),
14294
+ category: evalMetadataKeySchema.default("uncategorized")
14295
+ }).strict()).min(1).max(1e4)
14296
+ }).strict();
14297
+
14298
+ // src/lib/plugin-file.ts
14199
14299
  var sha2562 = (value) => createHash2("sha256").update(value).digest("hex");
14200
14300
  var slugify = (value) => value.toLowerCase().replace(/[^a-z0-9]+/g, "-").replace(/^-+|-+$/g, "").slice(0, 64).replace(/-+$/g, "");
14201
14301
  async function exists(path2) {
@@ -14346,38 +14446,86 @@ async function readEvalSuites(evalsDir, root, version) {
14346
14446
  const cases = casesRaw.split(`
14347
14447
  `).map((line) => line.trim()).filter(Boolean).map((line) => normalizeEvalCase(JSON.parse(line), rubric));
14348
14448
  const sourcePath = relative2(root, suiteDir);
14349
- suites.push({
14350
- slug: slugify(entry.name),
14351
- name: titleize2(entry.name),
14352
- version,
14449
+ const settings = await readJson(join4(suiteDir, "suite.json")).catch((e) => {
14450
+ if (e.code === "ENOENT")
14451
+ return { data: {}, raw: "" };
14452
+ throw e;
14453
+ });
14454
+ for (const c of cases)
14455
+ for (const fixture of c.fixture_refs) {
14456
+ const path2 = await realpath(join4(root, fixture.path));
14457
+ const rel = relative2(await realpath(root), path2);
14458
+ if (rel.startsWith("..") || rel.startsWith("/"))
14459
+ throw new Error("Fixture must stay inside the plugin directory");
14460
+ const contentHash = createHash2("sha256").update(await readFile4(path2)).digest("hex");
14461
+ if (fixture.content_hash && fixture.content_hash !== contentHash)
14462
+ throw new Error("Fixture content hash mismatch");
14463
+ fixture.content_hash = contentHash;
14464
+ }
14465
+ const sourceHash = await hashEvalSuite(suiteDir, root, cases);
14466
+ const graders = Array.isArray(settings.data.graders) ? settings.data.graders.map((g) => ({ ...g, config_hash: g.config_hash ?? sourceHash })) : [];
14467
+ const suite = evalSuiteInputSchema.parse({
14468
+ ...settings.data,
14469
+ slug: settings.data.slug ?? slugify(entry.name),
14470
+ name: settings.data.name ?? titleize2(entry.name),
14471
+ version: settings.data.version ?? version,
14353
14472
  source_path: sourcePath,
14354
- source_hash: sha2562([casesRaw, rubric, readme].join(`
14355
- ---
14356
- `)),
14357
- pass_threshold: 0.8,
14358
- promotion_gate: readme ? readme.slice(0, 2000) : undefined,
14473
+ source_hash: sourceHash,
14474
+ promotion_gate: settings.data.promotion_gate ?? (readme ? readme.slice(0, 2000) : undefined),
14475
+ graders,
14359
14476
  cases
14360
14477
  });
14478
+ if (new Set(suite.cases.map((c) => c.case_key)).size !== suite.cases.length)
14479
+ throw new Error(`Duplicate case keys in ${sourcePath}`);
14480
+ suites.push(suite);
14361
14481
  }
14362
14482
  return suites.sort((a, b) => a.slug.localeCompare(b.slug));
14363
14483
  }
14364
14484
  function normalizeEvalCase(input, rubricText) {
14365
- const key = String(input.case_key ?? input.id ?? input.name ?? randomUUID());
14366
- const expected = String(input.expected_outcome ?? input.expected_behavior ?? input.expected ?? "");
14367
- return {
14368
- case_key: slugify(key),
14369
- name: titleize2(String(input.name ?? key)),
14370
- scenario: String(input.scenario ?? input.prompt ?? input.description ?? ""),
14371
- expected_outcome: expected || "Review with the suite rubric.",
14372
- rubric: rubricText ? { source: "rubric.md", text_hash: sha2562(rubricText) } : {},
14373
- fixture_refs: [],
14374
- tags: Array.isArray(input.tags) ? input.tags.map(String) : [],
14375
- weight: Number.isInteger(input.weight) ? Number(input.weight) : 1
14376
- };
14485
+ const key = input.case_key ?? input.id ?? input.name;
14486
+ if (typeof key !== "string" || !key.trim())
14487
+ throw new Error("Eval cases require a stable case_key, id, or name");
14488
+ return evalCaseInputSchema.parse({
14489
+ case_key: key,
14490
+ name: input.name ?? titleize2(key),
14491
+ task_category: input.task_category,
14492
+ scenario: input.scenario ?? input.prompt ?? input.description,
14493
+ expected_outcome: input.expected_outcome ?? input.expected_behavior ?? input.expected,
14494
+ rubric: input.rubric ?? (rubricText ? { source: "rubric.md", text_hash: sha2562(rubricText) } : {}),
14495
+ fixture_refs: input.fixture_refs ?? [],
14496
+ tags: input.tags ?? [],
14497
+ weight: input.weight ?? 1,
14498
+ required: input.required ?? true
14499
+ });
14500
+ }
14501
+ async function hashEvalDirectory(root) {
14502
+ const entries = [];
14503
+ async function visit(dir2) {
14504
+ for (const item of (await readdir2(dir2, { withFileTypes: true })).sort((a, b) => a.name.localeCompare(b.name))) {
14505
+ if (["node_modules", ".git", ".promptfoo"].includes(item.name))
14506
+ continue;
14507
+ const path2 = join4(dir2, item.name);
14508
+ if (item.isSymbolicLink())
14509
+ throw new Error("Eval suite symlinks are not supported");
14510
+ if (item.isDirectory())
14511
+ await visit(path2);
14512
+ else if (item.isFile())
14513
+ entries.push([relative2(root, path2), createHash2("sha256").update(await readFile4(path2)).digest("hex")]);
14514
+ }
14515
+ }
14516
+ await visit(root);
14517
+ return sha2562(JSON.stringify(entries));
14377
14518
  }
14378
14519
  function titleize2(value) {
14379
14520
  return value.replace(/[-_]+/g, " ").replace(/\b\w/g, (letter) => letter.toUpperCase());
14380
14521
  }
14522
+ async function hashEvalSuite(suiteDir, pluginRoot, cases) {
14523
+ const fixtures = [];
14524
+ for (const c of cases)
14525
+ for (const f of c.fixture_refs)
14526
+ fixtures.push([f.path, createHash2("sha256").update(await readFile4(join4(pluginRoot, f.path))).digest("hex")]);
14527
+ return sha2562(JSON.stringify({ directory: await hashEvalDirectory(suiteDir), fixtures: fixtures.sort(([a], [b]) => a.localeCompare(b)) }));
14528
+ }
14381
14529
 
14382
14530
  // src/commands/publish.ts
14383
14531
  var __dirname3 = dirname(fileURLToPath2(import.meta.url));
@@ -14385,17 +14533,25 @@ var __dirname3 = dirname(fileURLToPath2(import.meta.url));
14385
14533
  class PublishCommand extends Command {
14386
14534
  static paths = [["publish"]];
14387
14535
  static usage = Command.Usage({
14388
- description: "Wrap a plugin's SKILL.md capability and publish a new version. Writes the telemetry footer and script wrappers."
14536
+ description: "Publish a plugin version or wrap a single SKILL.md capability for telemetry."
14389
14537
  });
14390
14538
  skillPath = exports_options.String({ required: true });
14391
14539
  skillId = exports_options.String("--skill-id");
14392
14540
  pluginId = exports_options.String("--plugin-id");
14393
14541
  version = exports_options.String("--version", { description: "semver for this publish; defaults to 0.1.0" });
14542
+ marketplaceUrl = exports_options.String("--marketplace-url", {
14543
+ description: "Git URL for the marketplace or source repo where this plugin version is published"
14544
+ });
14394
14545
  dryRun = exports_options.Boolean("--dry-run", false);
14395
14546
  async execute() {
14396
14547
  const root = resolve(this.skillPath);
14397
14548
  if (await isPluginRoot(root))
14398
14549
  return this.publishPlugin(root);
14550
+ if (this.marketplaceUrl) {
14551
+ this.context.stdout.write(import_picocolors3.default.red(`--marketplace-url is only supported when publishing a plugin directory
14552
+ `));
14553
+ return 1;
14554
+ }
14399
14555
  return this.publishSkill(root);
14400
14556
  }
14401
14557
  async publishSkill(root) {
@@ -14474,6 +14630,12 @@ Write key (store now — not retrievable):
14474
14630
  const plan = await readPlugin(root);
14475
14631
  const version = this.version ?? plan.version;
14476
14632
  const pluginId = this.pluginId;
14633
+ const marketplaceUrl = normalizeMarketplaceUrl(this.marketplaceUrl);
14634
+ if (marketplaceUrl === false) {
14635
+ out.write(import_picocolors3.default.red(`invalid --marketplace-url: expected a Git URL starting with https://, http://, git://, ssh://, or git@
14636
+ `));
14637
+ return 1;
14638
+ }
14477
14639
  for (const component of plan.components)
14478
14640
  pluginComponentInventoryItemSchema.parse(component);
14479
14641
  out.write(`reading plugin ${import_picocolors3.default.cyan(root)}
@@ -14485,6 +14647,9 @@ Write key (store now — not retrievable):
14485
14647
  out.write(` manifest: ${plan.manifestPath} ${import_picocolors3.default.dim(plan.manifestHash)}
14486
14648
  `);
14487
14649
  out.write(` format: ${plan.manifestFormat}
14650
+ `);
14651
+ if (marketplaceUrl)
14652
+ out.write(` marketplace URL: ${marketplaceUrl}
14488
14653
  `);
14489
14654
  out.write(` platforms: ${plan.platforms.length ? plan.platforms.join(", ") : import_picocolors3.default.dim("none")}
14490
14655
  `);
@@ -14505,7 +14670,7 @@ Write key (store now — not retrievable):
14505
14670
  `));
14506
14671
  return 0;
14507
14672
  }
14508
- const published = await publishPluginPlan(plan, version, pluginId);
14673
+ const published = await publishPluginPlan(plan, version, pluginId, marketplaceUrl ?? undefined);
14509
14674
  out.write(import_picocolors3.default.green(`✓ published plugin ${published.plugin.id} ${version}
14510
14675
  `));
14511
14676
  out.write(import_picocolors3.default.bold(`
@@ -14539,7 +14704,7 @@ Imported eval suites:
14539
14704
  return 0;
14540
14705
  }
14541
14706
  }
14542
- async function publishPluginPlan(plan, version, pluginId) {
14707
+ async function publishPluginPlan(plan, version, pluginId, marketplaceUrl) {
14543
14708
  const plugin = pluginId ? await getPlugin(pluginId) : await getOrCreatePlugin(plan);
14544
14709
  const skills = await getOrCreatePluginSkills(plan, plugin.id, version);
14545
14710
  await api(`/v1/plugins/${plugin.id}/versions`, {
@@ -14548,6 +14713,7 @@ async function publishPluginPlan(plan, version, pluginId) {
14548
14713
  version,
14549
14714
  manifest_format: plan.manifestFormat,
14550
14715
  manifest_hash: plan.manifestHash,
14716
+ marketplace_published_git_url: marketplaceUrl,
14551
14717
  components: plan.components
14552
14718
  })
14553
14719
  });
@@ -14562,6 +14728,14 @@ async function publishPluginPlan(plan, version, pluginId) {
14562
14728
  });
14563
14729
  return { plugin, skills, evalSuites, writeKey: keyResp.key };
14564
14730
  }
14731
+ function normalizeMarketplaceUrl(value) {
14732
+ if (!value)
14733
+ return null;
14734
+ const trimmed = value.trim();
14735
+ if (/^(https?:\/\/|git:\/\/|ssh:\/\/|git@)[^\s]+$/i.test(trimmed))
14736
+ return trimmed;
14737
+ return false;
14738
+ }
14565
14739
  async function getPlugin(pluginId) {
14566
14740
  return api(`/v1/plugins/${pluginId}`);
14567
14741
  }
@@ -14620,27 +14794,11 @@ async function getOrCreatePluginSkills(plan, pluginId, version) {
14620
14794
  async function importEvalSuites(plan, pluginId) {
14621
14795
  if (!plan.evalSuites.length)
14622
14796
  return [];
14623
- const existing = await api(`/v1/plugins/${pluginId}/eval-suites`);
14624
14797
  const rows = [];
14625
14798
  for (const suite of plan.evalSuites) {
14626
- const found = existing.data.find((candidate) => candidate.slug === suite.slug);
14627
- if (found) {
14628
- rows.push(found);
14629
- continue;
14630
- }
14631
14799
  rows.push(await api(`/v1/plugins/${pluginId}/eval-suites`, {
14632
14800
  method: "POST",
14633
- body: JSON.stringify({
14634
- slug: suite.slug,
14635
- name: suite.name,
14636
- version: suite.version,
14637
- source_path: suite.source_path,
14638
- source_hash: suite.source_hash,
14639
- status: "active",
14640
- pass_threshold: suite.pass_threshold,
14641
- promotion_gate: suite.promotion_gate,
14642
- cases: suite.cases
14643
- })
14801
+ body: JSON.stringify({ ...suite, status: "active" })
14644
14802
  }));
14645
14803
  }
14646
14804
  return rows;
@@ -14837,18 +14995,209 @@ class WebhooksAddCommand extends Command {
14837
14995
  }
14838
14996
  }
14839
14997
 
14998
+ // src/commands/evals.ts
14999
+ import { spawn, execFile as execFile6 } from "node:child_process";
15000
+ import { promisify as promisify6 } from "node:util";
15001
+ import { mkdtemp, readFile as readFile6, stat as stat3 } from "node:fs/promises";
15002
+ import { join as join6, resolve as resolve2 } from "node:path";
15003
+ import { tmpdir } from "node:os";
15004
+
15005
+ // src/lib/promptfoo.ts
15006
+ import { createHash as createHash3 } from "node:crypto";
15007
+ var object = (v) => v && typeof v === "object" && !Array.isArray(v) ? v : {};
15008
+ var digest = (v) => createHash3("sha256").update(JSON.stringify(v) ?? "null").digest("hex");
15009
+ function importPromptfoo(exported, cases, trials, graders) {
15010
+ const root = object(exported);
15011
+ const summary = Array.isArray(root.results) ? root : object(root.results);
15012
+ const rows = summary.results;
15013
+ if (!Array.isArray(rows) || !rows.length)
15014
+ throw new Error("Expected a non-empty Promptfoo v2/v3 JSON export (results.results)");
15015
+ const byKey = new Map(cases.map((c) => [c.caseKey, c.id]));
15016
+ const groups = new Map;
15017
+ const repetitions = new Map;
15018
+ for (const value of rows) {
15019
+ const row = object(value);
15020
+ const test = object(row.testCase);
15021
+ const metadata = object(test.metadata);
15022
+ const key = metadata.telvine_case_key;
15023
+ if (typeof key !== "string" || !byKey.has(key))
15024
+ throw new Error("Every test must have metadata.telvine_case_key matching a registered case");
15025
+ const provider = object(row.provider);
15026
+ const identity = digest([provider.id, row.promptId ?? row.promptIdx]);
15027
+ const model = typeof provider.id === "string" && evalMetadataKeySchema.safeParse(provider.id).success ? provider.id : `provider:${digest(provider.id).slice(0, 16)}`;
15028
+ const group = groups.get(identity) ?? { identity, model, trials: [] };
15029
+ const repeatKey = `${identity}:${key}`;
15030
+ const trial = (repetitions.get(repeatKey) ?? 0) + 1;
15031
+ if (trial > trials)
15032
+ throw new Error("More results than declared trials; use distinct case keys for expanded test variables");
15033
+ repetitions.set(repeatKey, trial);
15034
+ const grading = object(row.gradingResult);
15035
+ const components = Array.isArray(grading.componentResults) ? grading.componentResults.map(object) : [grading];
15036
+ const assertions = components.flatMap((component) => {
15037
+ const assertion = object(component.assertion);
15038
+ const grader = graders.find((g) => g.id === assertion.metric);
15039
+ if (!grader)
15040
+ return [];
15041
+ return [{
15042
+ grader_id: grader.id,
15043
+ config_hash: grader.config_hash,
15044
+ status: typeof component.pass !== "boolean" ? "errored" : component.pass ? "passed" : "failed",
15045
+ score: typeof component.score === "number" ? component.score : null
15046
+ }];
15047
+ });
15048
+ const isError = row.failureReason === 2 || Boolean(row.error) || typeof row.success !== "boolean";
15049
+ const status = isError ? "errored" : row.success === true ? "passed" : "failed";
15050
+ group.trials.push(evalCaseResultInputSchema.parse({
15051
+ eval_case_id: byKey.get(key),
15052
+ trial,
15053
+ status,
15054
+ score: isError ? null : row.score,
15055
+ rubric_results: {},
15056
+ assertions,
15057
+ error_class: isError ? "promptfoo_execution_error" : undefined,
15058
+ failure_cluster: status === "failed" ? "assertion_failed" : undefined,
15059
+ duration_ms: typeof row.latencyMs === "number" ? Math.round(row.latencyMs) : undefined,
15060
+ cost_usd: typeof row.cost === "number" ? row.cost : undefined,
15061
+ output_hash: row.response ? digest(object(row.response).output) : undefined,
15062
+ evidence_ref: `promptfoo:${digest(row.id ?? [identity, key, trial]).slice(0, 32)}`
15063
+ }));
15064
+ groups.set(identity, group);
15065
+ }
15066
+ return [...groups.values()];
15067
+ }
15068
+
15069
+ // src/commands/evals.ts
15070
+ function runProcess(bin, args, cwd) {
15071
+ return new Promise((res, reject) => {
15072
+ const p = spawn(bin, args, { cwd, stdio: "inherit", shell: false });
15073
+ p.on("error", reject);
15074
+ p.on("exit", (code, signal) => signal ? reject(new Error(`Promptfoo terminated: ${signal}`)) : res(code ?? 1));
15075
+ });
15076
+ }
15077
+ async function submit(suite, exported, options2) {
15078
+ const groups = importPromptfoo(exported, suite.cases, options2.trials, suite.graders);
15079
+ const runs = [];
15080
+ for (const group of groups) {
15081
+ const run2 = await api(`/v1/eval-suites/${suite.id}/runs`, { method: "POST", body: JSON.stringify({
15082
+ plugin_version: options2.version,
15083
+ harness: "promptfoo",
15084
+ model: group.model,
15085
+ trials: options2.trials,
15086
+ suite_revision_hash: suite.revisionHash,
15087
+ grader_config_hash: suite.sourceHash,
15088
+ engine_version: options2.engineVersion,
15089
+ baseline_run_id: options2.baseline,
15090
+ parent_run_id: options2.parent,
15091
+ experiment_id: options2.experiment,
15092
+ evaluator_variant: group.identity
15093
+ }) });
15094
+ try {
15095
+ for (const result of group.trials)
15096
+ await api(`/v1/eval-runs/${run2.id}/cases`, { method: "POST", body: JSON.stringify(result) });
15097
+ runs.push(await api(`/v1/eval-runs/${run2.id}/complete`, { method: "POST", body: JSON.stringify({ status: "completed" }) }));
15098
+ } catch (error) {
15099
+ await api(`/v1/eval-runs/${run2.id}/complete`, { method: "POST", body: JSON.stringify({ status: "failed" }) }).catch(() => {});
15100
+ throw error;
15101
+ }
15102
+ }
15103
+ return runs;
15104
+ }
15105
+
15106
+ class EvalOptions extends Command {
15107
+ version = exports_options.String("--version", { required: true });
15108
+ trials = exports_options.String("--trials", "1");
15109
+ baseline = exports_options.String("--baseline-run-id");
15110
+ parent = exports_options.String("--parent-run-id");
15111
+ experiment = exports_options.String("--experiment-id");
15112
+ trialCount() {
15113
+ const n = Number(this.trials);
15114
+ if (!Number.isInteger(n) || n < 1 || n > 100)
15115
+ throw new Error("--trials must be an integer from 1 to 100");
15116
+ return n;
15117
+ }
15118
+ printRuns(runs) {
15119
+ for (const r of runs)
15120
+ this.context.stdout.write(`${r.id}: gate ${r.gateStatus}; coverage ${Math.round((r.coverage ?? 0) * 100)}%; pass ${Math.round((r.passRate ?? 0) * 100)}%${r.gateReasons.length ? ` (${r.gateReasons.join(", ")})` : ""}
15121
+ `);
15122
+ return runs.every((r) => r.gateStatus === "passed") ? 0 : 1;
15123
+ }
15124
+ }
15125
+
15126
+ class EvalRunCommand extends EvalOptions {
15127
+ static paths = [["eval", "run"]];
15128
+ static usage = Command.Usage({ description: "Run a native Promptfoo suite locally and submit metadata-only evidence to Telvine." });
15129
+ plugin = exports_options.String({ required: true });
15130
+ suiteSlug = exports_options.String("--suite", { required: true });
15131
+ pluginId = exports_options.String("--plugin-id", { required: true });
15132
+ bin = exports_options.String("--promptfoo-bin", "promptfoo");
15133
+ engineVersion = exports_options.String("--engine-version", { required: true });
15134
+ async execute() {
15135
+ const root = resolve2(this.plugin);
15136
+ const plan = await readPlugin(root);
15137
+ const definition = plan.evalSuites.find((s) => s.slug === this.suiteSlug);
15138
+ if (!definition?.source_path)
15139
+ throw new Error("Suite not found under evals/");
15140
+ const suiteDir = join6(root, definition.source_path);
15141
+ let config;
15142
+ for (const name of ["promptfooconfig.yaml", "promptfooconfig.yml", "promptfooconfig.json"])
15143
+ if (await stat3(join6(suiteDir, name)).then((s) => s.isFile()).catch(() => false)) {
15144
+ config = name;
15145
+ break;
15146
+ }
15147
+ if (!config)
15148
+ throw new Error("Add a native promptfooconfig.yaml or .json to the suite directory");
15149
+ const trials = this.trialCount();
15150
+ const { stdout } = await promisify6(execFile6)(this.bin, ["--version"], { cwd: suiteDir, timeout: 30000 });
15151
+ const actualVersion = stdout.trim().match(/(?:^|\s)(\d+\.\d+\.\d+(?:-[A-Za-z0-9.-]+)?)(?:\s|$)/)?.[1];
15152
+ if (!actualVersion || actualVersion !== this.engineVersion)
15153
+ throw new Error(`Promptfoo engine version mismatch: expected ${this.engineVersion}, found ${actualVersion ?? "unknown"}`);
15154
+ const dir2 = await mkdtemp(join6(tmpdir(), "telvine-promptfoo-"));
15155
+ const output = join6(dir2, "results.json");
15156
+ this.context.stdout.write(`Promptfoo runs locally. Full evidence stays at ${output}
15157
+ `);
15158
+ const code = await runProcess(this.bin, ["eval", "-c", config, "--repeat", String(trials), "--no-cache", "--no-share", "--output", output], suiteDir);
15159
+ if (await hashEvalSuite(suiteDir, root, definition.cases) !== definition.source_hash)
15160
+ throw new Error("Suite changed during execution; evidence was not submitted");
15161
+ const exported = JSON.parse(await readFile6(output, "utf8"));
15162
+ const existing = await api(`/v1/plugins/${this.pluginId}/eval-suites`);
15163
+ const found = existing.data.find((s) => s.slug === definition.slug && s.sourceHash === definition.source_hash);
15164
+ const row = found ?? await api(`/v1/plugins/${this.pluginId}/eval-suites`, { method: "POST", body: JSON.stringify({ ...definition, status: "active" }) });
15165
+ const suite = await api(`/v1/eval-suites/${row.id}`);
15166
+ const runs = await submit(suite, exported, { version: this.version, trials, engineVersion: actualVersion, baseline: this.baseline, parent: this.parent, experiment: this.experiment });
15167
+ return this.printRuns(runs) || code;
15168
+ }
15169
+ }
15170
+
15171
+ class EvalImportCommand extends EvalOptions {
15172
+ static paths = [["eval", "import"]];
15173
+ static usage = Command.Usage({ description: "Import only safe metadata from a local Promptfoo JSON export." });
15174
+ file = exports_options.String({ required: true });
15175
+ suiteId = exports_options.String("--suite-id", { required: true });
15176
+ sourceHash = exports_options.String("--source-hash", { required: true });
15177
+ engineVersion = exports_options.String("--engine-version", { required: true });
15178
+ async execute() {
15179
+ const suite = await api(`/v1/eval-suites/${this.suiteId}`);
15180
+ if (suite.sourceHash !== this.sourceHash)
15181
+ throw new Error("Suite source hash mismatch");
15182
+ const exported = JSON.parse(await readFile6(resolve2(this.file), "utf8"));
15183
+ return this.printRuns(await submit(suite, exported, { version: this.version, trials: this.trialCount(), engineVersion: this.engineVersion, baseline: this.baseline, parent: this.parent, experiment: this.experiment }));
15184
+ }
15185
+ }
15186
+
14840
15187
  // src/telvine.ts
14841
15188
  var args = process.argv.slice(2);
14842
15189
  var cli = new Cli({
14843
15190
  binaryLabel: "telvine",
14844
15191
  binaryName: "telvine",
14845
- binaryVersion: "0.1.0"
15192
+ binaryVersion: "0.1.2"
14846
15193
  });
14847
15194
  cli.register(LoginCommand);
14848
15195
  cli.register(exports_builtins.HelpCommand);
14849
15196
  cli.register(exports_builtins.VersionCommand);
14850
15197
  cli.register(InitCommand);
14851
15198
  cli.register(PublishCommand);
15199
+ cli.register(EvalRunCommand);
15200
+ cli.register(EvalImportCommand);
14852
15201
  cli.register(SkillsListCommand);
14853
15202
  cli.register(SkillsMetricsCommand);
14854
15203
  cli.register(PluginsEventsCommand);
package/package.json CHANGED
@@ -1,16 +1,16 @@
1
1
  {
2
2
  "name": "@telvine/cli",
3
- "version": "0.1.0",
4
- "description": "Telvine CLI - instrument SKILL.md capabilities inside agent plugins",
3
+ "version": "0.1.2",
4
+ "description": "Telvine CLI - plugin telemetry, Promptfoo evaluation evidence, and release checks",
5
5
  "homepage": "https://telvine.com",
6
6
  "license": "MIT",
7
7
  "repository": {
8
8
  "type": "git",
9
- "url": "git+https://github.com/telvinehq/telvine.git",
9
+ "url": "git+https://github.com/telvinehq/telvine-cli.git",
10
10
  "directory": "apps/cli"
11
11
  },
12
12
  "bugs": {
13
- "url": "https://github.com/telvinehq/telvine/issues"
13
+ "url": "https://github.com/telvinehq/telvine-cli/issues"
14
14
  },
15
15
  "keywords": [
16
16
  "agent-plugin",
@@ -35,12 +35,6 @@
35
35
  "dist",
36
36
  "src/templates"
37
37
  ],
38
- "scripts": {
39
- "build": "bun build src/telvine.ts --target=node --outdir=dist --banner='#!/usr/bin/env node'",
40
- "dev": "tsx src/telvine.ts",
41
- "typecheck": "tsc --noEmit",
42
- "start": "node dist/telvine.js"
43
- },
44
38
  "dependencies": {
45
39
  "@inquirer/prompts": "^7.0.0",
46
40
  "clipanion": "^4.0.0-rc.4",
@@ -51,9 +45,15 @@
51
45
  "typanion": "^3.14.0"
52
46
  },
53
47
  "devDependencies": {
54
- "@telvine/schema": "workspace:0.1.0",
48
+ "@telvine/schema": "0.1.2",
55
49
  "@types/node": "^22.7.5",
56
50
  "tsx": "^4.19.1",
57
51
  "typescript": "^5.6.3"
52
+ },
53
+ "scripts": {
54
+ "build": "bun build src/telvine.ts --target=node --external=keytar --outdir=dist --banner='#!/usr/bin/env node'",
55
+ "dev": "tsx src/telvine.ts",
56
+ "typecheck": "tsc --noEmit",
57
+ "start": "node dist/telvine.js"
58
58
  }
59
- }
59
+ }
Binary file