ben-ai-models-registry 1.0.49 → 1.0.50

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/index.js CHANGED
@@ -1,6 +1,8 @@
1
1
  // index.js
2
2
  const JobStatus = require("./models/JobStatus");
3
3
  const Run = require("./models/Run");
4
+ const GraphRun = require("./models/GraphRun");
5
+ const GraphEvalResult = require("./models/GraphEvalResult");
4
6
  const Files = require("./models/Files");
5
7
  const Dataset = require("./models/Dataset");
6
8
  const Program = require("./models/Program");
@@ -18,6 +20,8 @@ const ScraperConfig = require("./models/ScraperConfig");
18
20
  module.exports = {
19
21
  JobStatus,
20
22
  Run,
23
+ GraphRun,
24
+ GraphEvalResult,
21
25
  Files,
22
26
  Dataset,
23
27
  State,
@@ -0,0 +1,95 @@
1
+ // GraphEvalResult: ONE flat doc per (question x iteration) of a GraphRun.
2
+ // Keyed by graphRunId (indexed). Inserted independently by the graph-run-worker
3
+ // as each eval completes — no $push into a parent, no shared-doc contention.
4
+ // The graph results view reads these paginated, filtered by graphRunId.
5
+ const mongoose = require("mongoose");
6
+ const { Schema } = mongoose;
7
+
8
+ // One LAAJ judge verdict. Mirrors the shape produced by ragEvaluationModel.
9
+ const JudgeScoreSchema = new Schema(
10
+ {
11
+ score: { type: Number },
12
+ reasoning: { type: String },
13
+ },
14
+ { _id: false, versionKey: false }
15
+ );
16
+
17
+ const SourceSchema = new Schema(
18
+ {
19
+ filename: { type: String },
20
+ url: { type: String },
21
+ citations: [
22
+ {
23
+ // String, not ObjectId: buildSources (eval-run route) sets this to the
24
+ // retrieved doc's id string. A non-ObjectId id (e.g. composite chunk id)
25
+ // would throw on cast; storing as String matches context[]._id too.
26
+ embeddingId: { type: String },
27
+ pageContent: { type: String },
28
+ },
29
+ ],
30
+ },
31
+ { _id: false, versionKey: false }
32
+ );
33
+
34
+ // Human evaluation (added so graph runs can be human-scored in the run view,
35
+ // mirroring the chain Run model's humanEvaluation/humanOverallScore).
36
+ const EvaluationQuestionSchema = new Schema(
37
+ {
38
+ question: { type: String },
39
+ score: { type: Number },
40
+ reasoning: { type: String },
41
+ },
42
+ { _id: false, versionKey: false }
43
+ );
44
+
45
+ const HumanEvaluationSchema = new Schema(
46
+ {
47
+ criterion: { type: String },
48
+ questions: [EvaluationQuestionSchema],
49
+ reasoning: { type: String },
50
+ overallScore: { type: Number },
51
+ },
52
+ { _id: false, versionKey: false }
53
+ );
54
+
55
+ const GraphEvalResultSchema = new Schema(
56
+ {
57
+ graphRunId: {
58
+ type: Schema.Types.ObjectId,
59
+ ref: "GraphRun",
60
+ required: true,
61
+ index: true,
62
+ },
63
+ programId: { type: Schema.Types.ObjectId, ref: "Program" },
64
+ question: { type: String },
65
+ expectedAnswer: { type: String },
66
+ iteration: { type: Number },
67
+ generatedAnswer: { type: String },
68
+ // Classification of generatedAnswer: "substantive" | "refusal" |
69
+ // "conversational". Drives which judges apply (a refusal has no claims to
70
+ // ground; small talk is off the relevance axis) and the run-view rollups.
71
+ answerType: { type: String },
72
+ sources: { type: [SourceSchema], default: [] },
73
+ context: { type: [Schema.Types.Mixed], default: [] },
74
+ // The four RAG judges.
75
+ correctness: { type: JudgeScoreSchema },
76
+ relevance: { type: JudgeScoreSchema },
77
+ groundedness: { type: JudgeScoreSchema },
78
+ retrievalRelevance: { type: JudgeScoreSchema },
79
+ // Human evaluation (filled in from the run view; empty until a human scores it).
80
+ humanEvaluation: { type: [HumanEvaluationSchema], default: [] },
81
+ humanOverallScore: { type: String },
82
+ // Graph-specific signal not present in the chain path.
83
+ retryCount: { type: Number },
84
+ docsRetrieved: { type: Number },
85
+ latencyMs: { type: Number },
86
+ errorMessage: { type: String },
87
+ traceId: { type: String },
88
+ traceUrl: { type: String },
89
+ },
90
+ { timestamps: true, versionKey: false, collection: "graph_eval_results" }
91
+ );
92
+
93
+ module.exports =
94
+ mongoose.models.GraphEvalResult ||
95
+ mongoose.model("GraphEvalResult", GraphEvalResultSchema);
@@ -0,0 +1,66 @@
1
+ // GraphRun: parent doc for a graph-engine eval run. Holds metadata + computed
2
+ // aggregates ONLY. Per-question results live in flat GraphEvalResult docs
3
+ // (one per question x iteration), keyed by graphRunId. This split avoids the
4
+ // 16MB single-doc ceiling, $push races, and unqueryable nesting of the legacy
5
+ // Run.evalResults schema. The chain (single-loop/baseline) path keeps using Run.
6
+ const mongoose = require("mongoose");
7
+ const { Schema } = mongoose;
8
+
9
+ // Snapshot of the config the graph ran with. Empty/undefined fields mean the
10
+ // live production default was used (no override). Present fields = a sweep.
11
+ const ParamsSnapshotSchema = new Schema(
12
+ {
13
+ model: { type: String },
14
+ k: { type: Number },
15
+ fetchK: { type: Number },
16
+ instruction: { type: String },
17
+ embeddingCollection: { type: String },
18
+ },
19
+ { _id: false, versionKey: false }
20
+ );
21
+
22
+ const GraphRunSchema = new Schema(
23
+ {
24
+ stateId: { type: Schema.Types.ObjectId, ref: "State" },
25
+ datasetId: { type: Schema.Types.ObjectId, ref: "Dataset", index: true },
26
+ name: { type: String },
27
+ description: { type: String },
28
+ type: { type: String, default: "graph" },
29
+ programId: { type: Schema.Types.ObjectId, ref: "Program" },
30
+ fileStore: { type: Schema.Types.ObjectId, ref: "FileStore" },
31
+ paramsSnapshot: { type: ParamsSnapshotSchema, default: {} },
32
+ evalModel: { type: String },
33
+ iterations: { type: Number },
34
+ // Aggregates computed on completion by querying GraphEvalResult.
35
+ // Scores are averaged ONLY over results where the metric was applicable
36
+ // (non-null). A metric with zero applicable results stores null, not 0.
37
+ llmPassRate: { type: Number },
38
+ relevanceScore: { type: Number },
39
+ groundednessScore: { type: Number },
40
+ retrievalRelevanceScore: { type: Number },
41
+ // Answer-mix rates (% of all results) so refusals/small talk are visible
42
+ // instead of silently inflating/deflating the scores above.
43
+ refusalRate: { type: Number },
44
+ conversationalRate: { type: Number },
45
+ // How many results actually contributed a score to each metric (the N
46
+ // behind each average — tells you how trustworthy each number is).
47
+ scoredCounts: {
48
+ correctness: { type: Number },
49
+ relevance: { type: Number },
50
+ groundedness: { type: Number },
51
+ retrievalRelevance: { type: Number },
52
+ },
53
+ // Set when a reviewer human-scores the run from the run view. humanPassRate
54
+ // is null (not 0) until at least one row is scored. humanScoredCount = how
55
+ // many result rows have a human score (drives the "unfinished n/N" marker).
56
+ humanPassRate: { type: Number },
57
+ humanScoredCount: { type: Number },
58
+ totalScore: { type: Number },
59
+ totalQuestions: { type: Number },
60
+ totalResults: { type: Number },
61
+ },
62
+ { timestamps: true, versionKey: false, collection: "graph_runs" }
63
+ );
64
+
65
+ module.exports =
66
+ mongoose.models.GraphRun || mongoose.model("GraphRun", GraphRunSchema);
@@ -8,7 +8,7 @@ const JobStatusSchema = new mongoose.Schema(
8
8
  programId: { type: mongoose.Schema.Types.ObjectId, ref: "Program" },
9
9
  jobType: {
10
10
  type: String,
11
- enum: ["run", "ingestion"],
11
+ enum: ["run", "ingestion", "graph-run"],
12
12
  },
13
13
  status: {
14
14
  type: String,
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "ben-ai-models-registry",
3
- "version": "1.0.49",
3
+ "version": "1.0.50",
4
4
  "description": "",
5
5
  "files": [
6
6
  "/models"