@malloy-publisher/server 0.2.2 → 0.2.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/server.mjs CHANGED
@@ -156066,6 +156066,10 @@ function assertDuckDBResourceConfig() {
156066
156066
  if (memoryLimit !== undefined && !/^\d+(\.\d+)?\s*(B|KB|KIB|MB|MIB|GB|GIB|TB|TIB)$/i.test(memoryLimit)) {
156067
156067
  throw new Error(`Invalid value for PUBLISHER_DUCKDB_MEMORY_LIMIT: expected a size like ` + `"1GB" or "512MB" (or "off" to disable), got "${memoryLimit}"`);
156068
156068
  }
156069
+ const rowGroupSizeBytes = getDuckLakeRowGroupSizeBytes();
156070
+ if (rowGroupSizeBytes !== undefined && !/^\d+(\.\d+)?\s*(B|KB|KIB|MB|MIB|GB|GIB)$/i.test(rowGroupSizeBytes)) {
156071
+ throw new Error(`Invalid value for PUBLISHER_DUCKLAKE_ROW_GROUP_SIZE_BYTES: expected a ` + `size like "16MB", got "${rowGroupSizeBytes}"`);
156072
+ }
156069
156073
  const tempDirectory = getDuckDBTempDirectory();
156070
156074
  if (tempDirectory !== undefined) {
156071
156075
  try {
@@ -156225,6 +156229,9 @@ var BUNDLED_DEFAULT_CONFIG_PATH, PER_MODE_COLOR_KEYS, DEFAULT_HIGH_WATER_FRACTIO
156225
156229
  return;
156226
156230
  }
156227
156231
  return raw;
156232
+ }, getDuckLakeRowGroupSizeBytes = () => {
156233
+ const raw = process.env.PUBLISHER_DUCKLAKE_ROW_GROUP_SIZE_BYTES?.trim();
156234
+ return raw === undefined || raw === "" ? undefined : raw;
156228
156235
  }, getDuckDBTempDirectory = () => {
156229
156236
  const raw = process.env.PUBLISHER_DUCKDB_TEMP_DIRECTORY?.trim();
156230
156237
  return raw === undefined || raw === "" ? undefined : raw;
@@ -213188,7 +213195,7 @@ var require_utils74 = __commonJS((exports, module) => {
213188
213195
 
213189
213196
  // ../../node_modules/ssh2/lib/protocol/crypto/build/Release/sshcrypto.node
213190
213197
  var require_sshcrypto = __commonJS((exports, module) => {
213191
- module.exports = __require("./sshcrypto-xqan60jb.node");
213198
+ module.exports = __require("./sshcrypto-vd2k5hq9.node");
213192
213199
  });
213193
213200
 
213194
213201
  // ../../node_modules/ssh2/lib/protocol/crypto/poly1305.js
@@ -233453,6 +233460,19 @@ import {
233453
233460
  MalloyConfig
233454
233461
  } from "@malloydata/malloy";
233455
233462
  import fs3 from "fs/promises";
233463
+ async function applyDuckLakeRowGroupBound(connection, dbName) {
233464
+ const bytes = getDuckLakeRowGroupSizeBytes();
233465
+ if (bytes === undefined) {
233466
+ return;
233467
+ }
233468
+ try {
233469
+ await connection.runSQL("SET preserve_insertion_order=false");
233470
+ await connection.runSQL(`CALL ${dbName}.set_option('parquet_row_group_size_bytes', '${escapeSQL(bytes)}')`);
233471
+ logger.info(`DuckLake row group bound applied to ${dbName}: ${bytes}`);
233472
+ } catch (error) {
233473
+ logger.warn(`Could not set the DuckLake row group bound on ${dbName}; the lake keeps ` + `its existing value: ${error instanceof Error ? error.message : String(error)}`);
233474
+ }
233475
+ }
233456
233476
  async function applySessionResourceLimits(connection, { tempDirectory } = {}) {
233457
233477
  const memoryLimit = getDuckDBMemoryLimit();
233458
233478
  const temp = tempDirectory ?? getDuckDBTempDirectory();
@@ -233801,6 +233821,9 @@ async function attachDuckLakeWithMode(connection, dbName, ducklakeConfig, option
233801
233821
  try {
233802
233822
  await connection.runSQL(attachCommand);
233803
233823
  logger.info(`Successfully attached DuckLake database in ${mode} mode: ${dbName}`);
233824
+ if (!options.readOnly) {
233825
+ await applyDuckLakeRowGroupBound(connection, dbName);
233826
+ }
233804
233827
  } catch (error) {
233805
233828
  if (error instanceof Error && (error.message.includes("already exists") || error.message.includes("already attached"))) {
233806
233829
  logger.info(`DuckLake database ${dbName} is already attached, skipping`);
@@ -294772,6 +294795,1896 @@ function registerGetStatusTool(mcpServer, environmentStore) {
294772
294795
  // src/mcp/skills/skills_bundle.json
294773
294796
  var skills_bundle_default = {
294774
294797
  skills: [
294798
+ {
294799
+ name: "eval-answer",
294800
+ description: "Score one analytical answer against a verified golden, and score which of the entities the golden depends on retrieval delivered to the answerer. Run the contamination checklist, re-execute the submitted query yourself, then spawn a judge subagent per skill:eval-judge. Append attempt, tool_call, score, and retrieval_score events to the file ledger (reference/ledger-schema.md). Never explain the failure (eval-diagnose) or edit the model (eval-improve). Use when asked whether an answer was correct, to score a run, or to baseline a model.",
294801
+ body: `# Evaluate One Answer
294802
+
294803
+ One user intent, answered once. This skill decides whether that answer was
294804
+ correct, records the evidence, and stops.
294805
+
294806
+ **Scope boundary:** verdict and events only. No diagnosis, no model edit.
294807
+
294808
+ ## The unit
294809
+
294810
+ A chat is not the unit. Segment by user intent. Feedback ("break it out by
294811
+ region") is a revision inside the same answer; grade the final accepted
294812
+ revision.
294813
+
294814
+ Take the question from the stored case (\`evals/<set>/cases.jsonl\`), never from
294815
+ memory or a truncated console line. Record \`question_sha\` of the exact text
294816
+ the answerer saw. Record \`servedRevision\` from \`get_context\` or reload, not
294817
+ the package name: a same-named decoy has been measured for hours.
294818
+
294819
+ ## Step 1: Contamination check, before any score
294820
+
294821
+ The answerer can Read or Shell its way to gold. Publisher traces do not see
294822
+ that, so the check runs on the HOST-side tool-use log you kept for the
294823
+ answerer subagent (every tool name and its path or command), plus the MCP
294824
+ call counts the answerer reported.
294825
+
294826
+ The checklist. An attempt is contaminated when its log shows any of:
294827
+
294828
+ 1. a Read, Shell, or any file tool touching \`evals/\` or a gold artifact path;
294829
+ 2. any access to the model file under test through a file tool (the
294830
+ \`modelPath\` argument on an MCP \`execute_query\` is NOT contamination; the
294831
+ server resolves it, the answerer never reads the file);
294832
+ 3. \`reported_calls\` greater than \`host_tool_uses\` (the detectable
294833
+ under-report floor is reported at most total tool uses).
294834
+
294835
+ \`skills/eval-answer/scripts/check_contamination.py\` is a reference aid that
294836
+ mechanizes the same checklist over a JSON log; your reading of the transcript
294837
+ is the check, the script is a second pair of eyes.
294838
+
294839
+ Contaminated attempts get \`verdict: null\` and \`contaminated: true\`. They are
294840
+ excluded from the run aggregates. They are not "wrong answers."
294841
+
294842
+ If you cannot produce a host log, mark \`contaminated: "unknown"\` on both the
294843
+ attempt and its score event, and do not treat the attempt as a clean pass.
294844
+
294845
+ ## Step 2: Re-run the submitted query yourself
294846
+
294847
+ Never score the agent's reported rows. Take its final query, execute it with
294848
+ \`execute_query\`, and write a prediction CSV under the run's
294849
+ \`artifacts/\` directory.
294850
+
294851
+ \`submitted: false\` when there is no final query. That is not a wrong answer.
294852
+ No verdict can be issued (\`verdict: null\`, with the reason) when the attempt
294853
+ is not submitted, when the golden is missing, provisional, invalid, or
294854
+ ambiguous, or when a verified golden has no local artifact to compare.
294855
+
294856
+ ## Step 3: Judge the answer
294857
+
294858
+ Spawn one fresh judge subagent per attempt, following
294859
+ \`skill:eval-judge\` (the rubric, the anchors, and the output shape live
294860
+ there; this skill does not restate them). Give it the question, the golden,
294861
+ your re-executed prediction rows, the canonical query when present, and the
294862
+ relevant source and field definitions from the model. It returns
294863
+ \`{verdict, confidence, why, column_pairing}\`.
294864
+
294865
+ - The judge sees gold. It is therefore never the answerer, and its verdict
294866
+ never leaks back to any answerer.
294867
+ - Confidence 5 or lower records as \`needs_human\`: neither a pass nor a fail,
294868
+ excluded from acceptance arithmetic, queued for a human look.
294869
+ - \`near_match\` is also neither. It means defensibly different, not "nearly a
294870
+ pass", and it stays out of the pass rate and the acceptance check for the same reason
294871
+ \`needs_human\` does. Report the count; do not fold it into either column.
294872
+ - When a human overrules a verdict, append the case to
294873
+ \`evals/<set>/judge-regressions.jsonl\`.
294874
+ - For a scalar golden, the same protocol applies to a one-value prediction.
294875
+ For \`unanswerable\`, a refusal that names the gap is the pass; a confident
294876
+ numeric answer is the fail.
294877
+ - Large row sets are still the judge's job. There is no scripted row oracle:
294878
+ a script that can pass a wrong answer is worse than none, and the rubric's
294879
+ containment and column-pairing rules are what the comparison needs.
294880
+
294881
+ ## Step 4: Score what retrieval delivered
294882
+
294883
+ Per attempt, mechanically, from the ledger -- \`scripts/score_retrieval.py\`. Each
294884
+ case names the entities its answer depends on (\`expectedEntities.required\`, and
294885
+ \`requiredAnyOf\` groups where the model offers more than one route). An entity
294886
+ was delivered if the attempt's \`get_context\` calls returned it as a ranked entity
294887
+ under its id, under the same type and name on a sibling source, or by name inside
294888
+ a returned source's documentation -- text the answerer reads and acts on. Only
294889
+ \`missing\` is a retrieval miss; the route per entity is recorded so the strict
294890
+ count is still there.
294891
+
294892
+ Recall 1.0 with a wrong answer exonerates retrieval: the failure is in the query.
294893
+ Recall below 1.0 and \`coverage: covered\` means the entity existed and search did
294894
+ not surface it; \`derivable\` or \`absent\` means there was nothing to surface. Those
294895
+ look identical in an answer score and have opposite owners, which is what makes
294896
+ this number worth having. It uses the search terms the answerer chose, so it
294897
+ attributes a failure *within* an arm and does not compare retrieval across arms
294898
+ -- that is the engine-side \`eval-retrieval\` skill, which does not ship here.
294899
+
294900
+ ## Step 5: Distrust the golden
294901
+
294902
+ A reference answer can be wrong (parent-column fanout, a join on a shared
294903
+ non-identifying key, or a rubric describing a model that has since been fixed).
294904
+ Fanout is not automatically a defect: \`AVG\` / \`STDDEV\` / \`MIN\` / \`MAX\` survive
294905
+ uniform duplication. Classify \`verified_wrong\` (exclude from scoring) vs
294906
+ \`verified_benign\` (keep).
294907
+
294908
+ **The judge produces this, not you.** It is the only station holding the golden,
294909
+ the re-executed rows and the model source at once, so it is the only one that can
294910
+ see the key contradict any of them; the rules and the four values are in
294911
+ \`skill:eval-judge\`. Carry its \`gold_status\` and \`gold_note\` onto the score
294912
+ event unchanged, and where it says nothing, fall back to the case's standing
294913
+ \`golden.status\`.
294914
+
294915
+ Do not encode a rewrite of a bad golden into the model. A \`suspect\` or
294916
+ \`verified_wrong\`, or a no_match whose why indicts the golden rather than the
294917
+ prediction, routes to the golden side door in \`skill:eval-loop\` as a **dataset**
294918
+ issue. It is never a model failure, and it must be settled before improve runs --
294919
+ otherwise a modelling agent is dispatched to fix a model that is already right.
294920
+
294921
+ ## Step 6: Append events, then stop
294922
+
294923
+ Append to \`evals/<set>/runs/<runId>/events.jsonl\` with \`caseId\` set. Shapes
294924
+ live in \`reference/ledger-schema.md\`.
294925
+
294926
+ 1. \`attempt\`: qid, sample, phase, question_sha, submitted, final_query,
294927
+ served revision, call counts, contamination verdict, transcript path.
294928
+ 2. \`tool_call\`: one per MCP \`get_context\` / \`execute_query\`, with \`traceId\`
294929
+ and the \`rankedSummary\` copied from the trace (per-target ranks included).
294930
+ Do not copy full traces into the event; the trace store holds the body.
294931
+ 3. \`score\`: the judge's verdict object plus \`judge_version\`, \`rubric_sha\`,
294932
+ \`golden_revision\`, \`contaminated\`, \`gold_status\`, and the judge output's
294933
+ artifact path.
294934
+
294935
+ A stage never rewrites another stage's fields. End-of-run numbers come from
294936
+ counting events, not from your arithmetic in prose.
294937
+
294938
+ Sample each case once. Breadth across cases beats repeats of one case; the
294939
+ comparison rule for a before/after is the flip count in \`skill:eval-loop\`
294940
+ Measurement, not a mean over samples.
294941
+
294942
+ ## Re-score after a golden repair
294943
+
294944
+ When \`eval-loop\` has repaired a golden and opened a new run, this skill runs
294945
+ again **without a new answerer**: same stored \`final_query\` (or its saved
294946
+ prediction CSV), new gold artifact, fresh judge, new \`golden_revision\` on the
294947
+ \`score\` event. Contamination does not need to be re-litigated if the attempt
294948
+ was already clean. If you must re-execute, do it yourself; do not ask the
294949
+ original answerer to "try again" with the new key in context.
294950
+
294951
+ ## Related skills
294952
+
294953
+ - \`skill:eval-diagnose\`: why it failed, after this record exists.
294954
+ - \`skill:eval-improve\`: smallest model edit, model-owned issues only.
294955
+ - The \`malloy-analysis-pitfalls\` skill: checks before you trust a result you ran.
294956
+ The judge loads it by name; it is not part of the \`eval\` group.
294957
+
294958
+ ## Reference files over MCP
294959
+
294960
+ This skill's \`reference/\` files are served as separate prompts, one per file, fetched only when you ask for them. Where the text above says to read \`reference/<name>.md\`, get the prompt named \`eval-answer/<name>\` instead.
294961
+
294962
+ Available: ledger-schema.`
294963
+ },
294964
+ {
294965
+ name: "eval-answer/ledger-schema",
294966
+ description: "The eval ledger: files and events. Reference detail for the eval-answer skill.",
294967
+ body: "# The eval ledger: files and events\n\nThe ledger is plain files in the model package's git repository. There is no\neval API and no eval database. The conductor (`skill:eval-loop`) reads and\nwrites these files directly; the stages share them as their contract.\n`eval-answer` writes `attempt`, `tool_call`, and `score`. An engine-side\n`eval-retrieval` skill, which does not ship here, writes `retrieval_score` and\n`probe`; both stay in this contract so one validator covers every run\ndirectory.\n`eval-diagnose` writes `issue` and `issue_status`. `eval-improve` writes\n`candidate`. `eval-loop` writes `acceptance_check`, further `issue_status`, and\n`checkpoint`.\n\n**The contract is code: `eval-answer/scripts/ledger.py`.** Every script that\nwrites `run.json` or `events.jsonl` builds its lines through that module, so a\nfield rename that reaches only one writer fails at write time. This document\nis the human-readable rendering; where the two disagree, the module is right\nand this file is the bug. Check any run directory with:\n\n```\npython3 skills/eval-answer/scripts/ledger.py validate <runDir>\n```\n\nErrors are broken identity (missing `run.json`, a missing required field, two\nscores for one attempt); warnings are missing comparison pins (a run without\n`skillsVersion` still measures, but cannot anchor an A/B on that pin) and\ngrandfathered unknown fields on old runs.\n\n## Layout\n\n```\nevals/<set>/\n set.json # set metadata (below)\n cases.jsonl # one case per line\n judge-regressions.jsonl # judge verdicts a human overruled\n runs/<runId>/\n run.json # run config, the attribution pins\n events.jsonl # append-only event lines\n artifacts/ # prediction CSVs, judge outputs, transcripts\n```\n\nThe set directory lives in the SAME git repository as the model it evaluates,\nso a checkpoint (a git commit) pins the model and the ledger together. Never\nplace `evals/` inside the directory tree the answerer's package serves: gold\nin the served tree is a contamination path.\n\nRules that make the ledger trustworthy:\n\n- `events.jsonl` is append-only. Never edit a line. To void one, append a new\n event whose payload marks the old one `voided`. One sanctioned exception: a\n re-runnable stage (diagnose, improve) replaces ITS OWN prior events, keyed by\n issue id, through `ledger.replace_events()` -- so a crash-and-rerun does not\n duplicate issues. A stage never touches another stage's lines.\n- A stage never rewrites another stage's fields.\n- One artifact directory per attempt (`qid` plus `sample`). Never overwrite a\n previous attempt's prediction CSV.\n- End-of-run numbers come from counting event lines (`jq` over\n `events.jsonl`), never from arithmetic recalled in prose.\n\n## `set.json`\n\n| Field | Notes |\n|---|---|\n| `name` | Set name; also the directory name. |\n| `description` | |\n| `datasetVersion` | Integer. Bump on any golden repair or case change. Runs record the version they scored against. |\n| `targetModelPath` | Model path within the package. |\n| `truthPackage` | Name of the package holding the semantics-free sources every golden is re-derived from -- a package NAME, not an object. `verify_goldens.py` skips every check without it, reporting \"nothing to re-derive against\", so a set that omits it silently has no golden verification at all. `init_truth_package.py` scaffolds the package. |\n| `truthModel` | Model file inside that package. Default `truth.malloy`. |\n| `truthTableRewrite` | Boolean, default false. Rewrites `duckdb.table('data/x.parquet')` refs to bare `x` in canonical queries, for a truth package whose tables are registered rather than read from files. |\n\n## `cases.jsonl`\n\nOne JSON object per line:\n\n| Field | Notes |\n|---|---|\n| `qid` | Stable case id. |\n| `question` | Exact text the answerer will see. |\n| `split` | `dev` or `holdout`. Frozen at import. Diagnose and improve read `dev` only; the acceptance check runs both. |\n| `tags` | list |\n| `state` | `candidate` / `selected` / `excluded`. |\n| `source` | Where the case came from. |\n| `golden` | `status` (`verified` / `provisional` / `invalid` / `ambiguous`), `kind`, `value` or `path` (artifact under the set directory), `canonicalQuery` (runs against the **truth** package, never the model under test), `verifiedBy`, and `verification` -- `{primaryAxis, variesAxis, note}` naming what the second derivation varied. **A golden is written only after two differently shaped derivations agree** (the truth query and a second one that varies a different axis -- group by something else, sum a different way, take a different route through the raw tables); `gold/<qid>.json` holds both (`truthRows`, `verifyRows`, `agreement`). Two derivations that vary the same axis share every other blind spot: that is how a golden that summed three overlapping time slices, 3x too high, passed its check. `verify_goldens.py` re-derives the value, reads the accepting clause of the rubric against the rows, and flags a second derivation that names no axis. **The question asks for data, never for an interpretation of it**, and either fixes the grain or the rubric accepts a correct figure at any stated grain (`judge.md` rules 9 and 10). |\n| `golden.rubric` | **On the golden, not on the case.** The prose the judge is shown as `CASE RUBRIC`, naming what counts as correct and what does not. `run_baseline.py` reads `golden.rubric`; a rubric written at the case's top level is silently not passed, the judge is told `CASE RUBRIC: none`, and it then scores from the golden value and the answer alone -- which reads as a judge that ignores its instructions. |\n| `golden.mustState` | **On the golden, not on the case.** What the answer has to say out loud. Read from `golden.mustState` by the run-package builder. |\n| `goldenRevision` | Integer. Bump on every golden change; `score` events stamp the revision they compared against. |\n| `expectedEntities` | `required`: entity ids (`kind:source:name`) the answer cannot be produced without. `requiredAnyOf`: a list of **groups**, each a list of ids of which any one suffices -- for a case the model can answer through more than one route. Naming only one route scores the other as a retrieval miss and steers diagnosis to \"retrieval ranking\" for a failure that was never retrieval's. `acceptable`: ids that are not noise if returned. An entity counts as delivered when returned under its exact id, under the same type and name on a sibling source, or by name in a returned source's documentation; `score_retrieval.py` records the route per entity as `delivery`. |\n\n## Term files (`eval-retrieval` only)\n\nA customer set has no term file. The engine-side `eval-retrieval` skill keeps\n`intents.jsonl` fixtures -- search term, entity type, a description of what the\nsearcher meant, a validity flag -- beside its own scripts, for evaluating the\nengine against fixed inputs. Their shape and the `retrieval_score` events they\nproduce are documented there and validated here.\n\n## `run.json`\n\nThe attribution pins. Two runs are comparable only when these match where it\nmatters. Two runs with different `target` values are comparable on answer\nverdicts only if the same model version reached both, which is rarely worth\nassuming:\n\nRequired fields are the run's **identity** -- every run ever written carries\nthem. The rest are optional in the schema, but the **comparison pins** are\nwarned about when absent, because a run missing one cannot take part in an A/B\non that axis.\n\n| Field | Notes |\n|---|---|\n| `runId` *(required)* | Directory name. |\n| `target` *(required)* | Which server answered, and how it reached the data: `local`, `local-proxied`, or `platform`. Decides what an improve step is even allowed to do (see `skill:eval-loop`). |\n| `answererModel` *(required)* | |\n| `phase` / `started` *(required)* | |\n| `judgeModel` / `judgeVersion` / `rubricSha` *(pins)* | The judge's model, and the version + content sha of `skill:eval-judge`. |\n| `datasetVersion` *(pin)* | From `set.json` at run time. |\n| `modelSha` *(pin)* | Content sha of the `model.malloy` snapshot in the run directory -- the bytes the answerer actually queried. A git sha is not enough: a snapshot host serves a copy, often of a dirty tree no commit names. |\n| `skillsVersion` *(pin)* | HEAD of the checkout the agents' skills were loaded from, dirty-marked (`ledger.skills_git_sha(root)`). The skills are the doctrine the agents load; a run that cannot name their version cannot anchor a skills A/B. |\n| `skillsRoot` / `harnessVersion` | Which checkout supplied the doctrine (`--skills-root`, e.g. a Publisher checkout for the open-source skills; default this one), and this checkout's own HEAD. The eval-* skills always come from the harness checkout, whatever `skillsRoot` says. Two runs whose `skillsRoot` differ are a skills A/B only if their manifests name the same skills. |\n| `diagnoserModel` / `improverModel` | Written by `diagnose.py` / `improve.py` when those stages run, so the run names every LLM that touched it. Absent on a run that was only answered and judged. |\n| `modelGitSha` | Commit of the model repo, when the target served working files; `-dirty` suffix when the tree had uncommitted changes -- fine for a band measurement, not for an A/B pin. Omit for a platform target, where a local commit pins nothing. |\n| `environment` / `package` / `modelPath` | What was served, and from where. On a platform target `environment` is the organization and `package` is the workspace the MCP URL is scoped to. |\n| `scope` | Platform target only: the `environment/package` the answerer was told to pass as an explicit `scopes` entry on every `get_context` / `execute_query` call. A workspace can serve many packages (a personal workspace serves every package the user can read), so without this the run also measures whether retrieval picks the right package -- a different measurement. Absent means unscoped. |\n| `mcpUrl` / `publisher` | The endpoints the answerer and the re-execution used. |\n| `predictionsReExecuted` | Whether the judge saw re-executed rows (false when the served bytes no longer match the pinned snapshot). |\n| `label` | `<set>-<phase>-<nn>`, assigned by `run_baseline.py` from the set name, the run's phase and the next free number beside it (`ecommerce-baseline-01`, `ecommerce-baseline-02`, `ecommerce-blind_gate-01`). The A/A pair is two runs of the same phase; the post-edit arms are two runs of `blind_gate`. Hand-typed names do not survive one afternoon of runs -- `base`, `rejudged2`, `r3`, `post1` sort wrongly, group not at all, and cannot be matched to an arm. `--label` overrides for a run that genuinely needs a human name. |\n| `effort` | |\n| `answererCostUsd` / `judgeCostUsd` | What the arm cost, split by role. The judge's half was discarded until 2026-09-02, so every \"cost per arm\" quoted before then was the answerer alone. |\n| `goldenCheck` | What `verify_goldens.py` said before the run started: `N ok, M drifted, K other finding(s)`, or why it did not run (no truth server on a platform target; `--skip-golden-check`; rebuild). A run that started on a drifted set says so here rather than pretending its verdicts mean something. |\n| `status` | `complete`, or `aborted` when four consecutive attempts errored or found the server dead and the harness stopped rather than spend the rest of the budget on attempts nobody will trust. |\n| `packageSha` / `servedRevision` *(pins)* | Taken from the server, not recomputed: `sourceContentSha` is a content hash over EVERY model path in the package, so an edit to an imported file moves it where a sha of the one `--model-path` does not; `servedRevision` is minted per load, so it identifies a load rather than content and is a poor pin alone. Measured: `publisher.json`'s `version` moves neither, and nothing in Publisher reads it -- it is not in the Package API schema and never returned, so it pins nothing. |\n| `datasetSha` *(pin)* | Content hash of `set.json` + `cases.jsonl`. Deliberately SEPARATE from the model's sha: a golden repair is not a model change, and one pin covering both would make every answer-key fix read as an edit to the model, which is the distinction an A/B rests on. Automatic, so nobody has to remember it; `datasetVersion` stays beside it as the human-readable sequence. **Local targets only.** A hosted target that publishes IMMUTABLE versions needs none of this: the set rides inside the version, and immutability -- not hashing -- is what makes a pin trustworthy. There, `targetVersion` alone identifies model and set together. |\n| `doubtedGoldens` | The cases whose golden the judge did not believe: `qid`, `gold_status` (`suspect` or `verified_wrong`), `gold_note`. **Read this before diagnose.** These are dataset issues, not model failures, and they go through the golden side door in `skill:eval-loop`. Empty list when the judge believed every key. Written from the same scan that prints the end-of-run warning, because a warning that lives only in console text is one scrollback away from sending a modelling agent at a model that is already right. |\n| `mode` / `setName` / `targetVersion` / `serverVersion` / `traceMode` / `callBudget` / `status` | Defined and accepted, **not yet written by any harness** -- kept in the schema for the platform target and the conductor, which need them. |\n\n## Events\n\nEvery line in `events.jsonl` is **flat**: `{ \"kind\": ..., <fields> }`, one\nJSON object per line, with an optional `at` ISO timestamp. Case-scoped kinds\n(`attempt`, `tool_call`, `score`) carry `qid`, `sample`, `phase` on the line;\nrun-level kinds do not. (An earlier draft of this document nested fields under\na `payload` with a `caseId`; no writer ever did that, and 23 run directories\nexist in the flat shape, so the flat shape is the contract.) `kind` is one of:\n`attempt`, `tool_call`, `score`, `retrieval_score`, `issue`, `issue_status`,\n`candidate`, `acceptance_check`, `checkpoint`.\n\n### `attempt`\n\n| Field | Type | Notes |\n|---|---|---|\n| `qid` | string | |\n| `sample` | int or null | Which repeat. Required even when null. |\n| `phase` | string | `baseline` / `loop` / `blind_gate` / `canary` / `final`. `phase` lives here, on the attempt, not in run config. |\n| `question_sha` | string | Hash of the exact text the answerer saw. |\n| `submitted` | bool | False when there was no final query. Not a wrong answer. |\n| `final_query` | string or null | Required to replay. |\n| `servedRevision` | string or null | From the package actually queried. |\n| `n_get_context` / `n_execute` / `n_execute_errors` | int | |\n| `host_tool_uses` | int | Host-side count, including Read and Shell. |\n| `reported_calls` | int | MCP calls the answerer claimed. |\n| `contaminated` | bool or `\"unknown\"` | `\"unknown\"` when no host log exists. Read it with `ledger.is_contaminated`, never for truthiness: runs written before 2026-09-03 carry the strings `\"true\"`/`\"false\"`, and `bool(\"false\")` is True. `ledger.event` now rejects those strings on write; `validate_run` grandfathers them on read as a warning. |\n| `contamination_reasons` | list | Empty when clean. |\n| `input_tokens` / `output_tokens` / `cache_read_tokens` | int or null | Answerer token usage. Null when the host does not report it. |\n| `cost_usd` | float or null | Answerer cost for this attempt. |\n| `num_turns` / `wall_seconds` | int, float or null | |\n| `answer_text` | string or null | The answer the judge scored. Kept so a verdict can be re-read without the transcript. |\n| `transcriptPath` | string | The answerer's transcript under `artifacts/`. |\n\nToken counts sit here rather than being derived later because the claim a\nsemantic model makes is about cost as well as correctness -- that a documented\nmodel reaches a good answer in fewer turns and tokens than working from raw\nschema. A ledger that counts calls but not tokens can state half of that.\n\n### `tool_call`\n\nOne event per MCP `get_context` or `execute_query` the attempt made.\n\n| Field | Type | Notes |\n|---|---|---|\n| `tool` | string | `get_context` or `execute_query`. |\n| `traceId` | string or null | `get_context` only; look up in your host's trace store. |\n| `targets` | string, list or null | **What the answerer asked for**: the search terms it sent to `get_context`. Null for `execute_query`. |\n| `rankedSummary` | object | Copied at capture from the trace so evidence survives trace eviction: `entityIds`, `ranks`, `resultCount`, and per-target `targets` with within-target ranks. |\n| `error` | string or null | |\n\nNever persist `execute_query` result rows, givens, or credentials.\n\n`targets` records the request; `rankedSummary` records the response. Without\nboth, a low per-attempt recall has two readings that cannot be told apart: the\nanswerer searched for the wrong thing, or it searched well and retrieval ranked\nthe right entity too low. Those have opposite owners -- `agent-skill` and\n`retrieval` -- so a ledger holding only the response cannot attribute the\nfailure, and any recall computed from it is a blend of answerer behaviour and\nretrieval quality.\n\nThat blend is also why per-attempt recall is **not** comparable across arms: a\nstronger answerer searches better and scores higher retrieval recall without\nretrieval having changed. Use it for attribution within an arm. Comparing\nretrieval itself between engine versions is `eval-retrieval`'s job, with fixed\nterms, and is not something a customer run reports.\n\n### `score`\n\nThe answer judge's verdict for one attempt (protocol in\n`skill:eval-judge`). Every attempt in a scored run gets exactly one.\n\n| Field | Type | Notes |\n|---|---|---|\n| `verdict` | string or null | `match` / `near_match` / `no_match` / `needs_human`; null when the attempt is not scorable. Only `match` and `no_match` are decisions; see below. |\n| `reason` | string | Why, from the judge; for a null verdict, why not scorable (`not_submitted`, `golden_missing`, `golden_ambiguous`, `contaminated`). |\n| `confidence` | int or null | 1 to 10. Confidence of 5 or lower forces `needs_human`. |\n| `column_pairing` | object or null | The judge's named gold-to-prediction column correspondence. |\n| `judge_version` / `rubric_sha` | string | Pins which rubric produced this verdict. |\n| `golden_revision` | int | From the case at score time. |\n| `contaminated` | bool or `\"unknown\"` | Copied from the attempt; true or unknown means `verdict: null`. |\n| `artifactPath` | string | The full judge output under `artifacts/`. |\n| `gold_status` | string | `verified` / `verified_benign` / `suspect` / `verified_wrong`. **From the judge**, which scored against the golden as written and reports separately whether it believes it; falls back to the case's standing `golden.status` when the judge does not say. `verified_wrong` excludes the case from run aggregates. `suspect` and `verified_wrong` route to the golden side door as `dataset` issues, never to improve. |\n| `gold_note` | string or null | The judge's evidence for a non-`verified` status: the two values, or the model line against the rubric sentence. Null when `verified`. |\n\nA `submitted: false` attempt gets `verdict: null, reason: \"not_submitted\"`,\nexcept for an `unanswerable` golden, where a refusal that names the gap is the\npass and a confident numeric answer is the fail.\n\nAggregates count decided verdicts only. `match` and `no_match` are the pass and\nthe fail; **`near_match`, `needs_human` and null are none of the above** and stay\nout of acceptance arithmetic. `near_match` is excluded because it means \"defensibly\ndifferent\", and an arguable verdict that moves a pass rate is a measurement\nartefact rather than a result (`skill:eval-judge` rule 7). Report its count:\nit rising is how a set tells you its rubrics are going vague.\n\nThis applies to `score`. On `retrieval_score` below, `near_match` **is** counted\ntowards recall and precision, because an overlapping entity is a genuine\nretrieval success. The two are different questions that share a word.\n\n### `retrieval_score` (written by `eval-retrieval`, not by a customer run)\n\nOne per term judged in a fixed-term replay (run-level; no `caseId`).\n\n| Field | Type | Notes |\n|---|---|---|\n| `intentId` / `term` / `entityType` | string | From the term file. |\n| `in_scope` | bool | Does THIS model version represent the concept? False is a coverage gap, not a retrieval failure. |\n| `judgments` | list | Per returned entity: `entityId`, `rank`, `level` (`match` / `near_match` / `no_match`), `confidence`, `why`. Empty when nothing returned. |\n| `judge_version` / `rubric_sha` | string | |\n| `traceId` | string or null | The `get_context` call judged. |\n\nRun-level metrics fall out by counting:\n\n- `coverage` = in-scope terms / valid terms -- a property of the model and its\n data, reported on its own, never as a retrieval number.\n- `recall` (on in-scope terms) = fraction whose judgments contain a `match`.\n- `precision@N` = `match` judgments / judged, with N stated.\n\n### `issue`\n\n| Field | Type | Notes |\n|---|---|---|\n| `issue_id` | string | Stable across status events. |\n| `qids` | list | Affected cases. |\n| `primary_code` / `contributing_codes` | string / list | From `skill:eval-diagnose`, verbatim. |\n| `component` | string | `dataset` / `agent-call` / `get_context/model` / `get_context/retrieval` / `construction` / `model-definition`. |\n| `owner` | string | `model` / `retrieval` / `agent-skill` / `dataset`. Environment failures stop the run; they are never diagnosed, so there is no environment owner. |\n| `severity` / `confidence` | string | |\n| `sufficiency` | string | `sufficient` / `insufficient` / `unknown`. |\n| `traceIds` | list | |\n| `diagnosis` | string | Written before any edit exists. |\n\n### `issue_status`\n\n`issue_id` plus `status`: `open` / `batched` / `fixed` / `rejected` /\n`deferred`. Readers take the latest event for that `issue_id`. Do not invent\na status column.\n\n### `candidate`\n\nWritten by `eval-improve`, for every proposed edit, accepted or not. A\nrejected direction keeps its record.\n\n| Field | Notes |\n|---|---|\n| `issue_ids` | |\n| `files` | Paths the edit touched. |\n| `diffSummary` | One line per file. |\n| `probes` | Query and result for each factual claim. |\n| `meaningChanged` | Entities whose *meaning* the edit changed; `[]` for a docs-only edit. |\n| `goldenSuspect` | Each `{qid, entity, stored, rederived}`: a golden this edit may have invalidated. Reported by the improver, never repaired by it. **Non-empty halts the acceptance check** until adjudicated through the golden side door. |\n| `goldenAudit` | The set's `verify_goldens.py` run against the edited model: `{ran, clean, model, tail}`. Catches drift and rubric-vs-model contradictions only; a golden whose value silently moved is invisible to it, which is why `goldenSuspect` exists alongside. |\n\n### `acceptance_check`\n\nCalled `gate` before 2026-09-03. `ledger.py` reads the old kind as this one, so\nrun directories written earlier still validate and nothing rewrites them.\n\nWritten by `eval-loop`, one per acceptance check decision, BEFORE any checkpoint commit.\n\n| Field | Notes |\n|---|---|\n| `issue_ids` | |\n| `decision` | `accepted` / `rejected`. |\n| `class` | `docs` / `definition` / `retrieval` / `justified`. |\n| `baselineRunId` / `finalRunIds` | Plural: acceptance needs two independent runs. |\n| `regressions` | Case ids whose verdict got worse vs baseline. Must be empty to accept. |\n| `holdoutDelta` | Confident-verdict delta on the holdout slice. |\n| `reason` | Including independent deterministic justification when that is the basis. |\n\n### `checkpoint`\n\nWritten by `eval-loop` after an accepted acceptance check, or when a restore runs. The\nmodel bytes live in git, not in this payload.\n\n| Field | Notes |\n|---|---|\n| `action` | `created` / `restored`. |\n| `label` | |\n| `modelGitSha` | The commit this checkpoint names (create), or the commit restored to. |\n| `issueIds` | Issues the accepted edit closed. Empty on restore. |\n\n## `judge-regressions.jsonl`\n\nAppend a line whenever a human overrules a judge verdict: the case or intent,\nthe judge's verdict, the human's, and why. Re-run this file against the judge\nwhenever `skill:eval-judge` or the judge model changes; a rubric change that\nflips old human-settled verdicts is a judge regression, not new truth."
294968
+ },
294969
+ {
294970
+ name: "eval-diagnose",
294971
+ description: "Diagnose why a scored answer failed and who owns the fix, then cluster the failures by shared root cause. Walk dataset, agent-call, get_context/model, get_context/retrieval, construction, then model-definition. Append issue events to the file ledger, linked by traceId, one per cluster. Use after eval-answer, when triaging a run, or before changing a model. Does not edit the model (eval-improve).",
294972
+ body: `# Diagnose One Answer
294973
+
294974
+ Consumes a \`score\` event from \`skill:eval-answer\` and answers: why did this fail,
294975
+ and who owns the fix?
294976
+
294977
+ **Scope boundary:** write the diagnosis before any edit exists. This skill never
294978
+ edits a model and never proposes a patch beyond naming the gap. Diagnosis that
294979
+ is allowed to edit becomes justification for an edit somebody already wanted.
294980
+
294981
+ Do not diagnose a contaminated attempt or an environment failure. Those are
294982
+ harness or ops, not model work.
294983
+
294984
+ ## Components, in order
294985
+
294986
+ Walk **in this order** and stop at the first with positive evidence. A later
294987
+ label requires ruling out the earlier ones. Write \`component\` with these strings,
294988
+ never "C1" / "C2" / "C3":
294989
+
294990
+ | \`component\` | Question |
294991
+ |---|---|
294992
+ | \`dataset\` | Bad question, bad or missing golden, or environment drift? |
294993
+ | \`agent-call\` | Did the agent ask for the needed concepts, with the right type and scope? |
294994
+ | \`get_context/model\` | Is the needed entity absent, undocumented, weakly labeled, duplicated, or missing guidance? |
294995
+ | \`get_context/retrieval\` | Was an on-target request against a well-described entity ranked or grouped wrong? |
294996
+ | \`construction\` | Did sufficient context arrive, and the agent still built the wrong query? |
294997
+ | \`model-definition\` | Is a measure, join, filter convention, or source semantically wrong? |
294998
+
294999
+ \`owner\` is separate: \`model\`, \`retrieval\`, \`agent-skill\`, or \`dataset\`. There
295000
+ is no environment owner: an environment failure stops the run before
295001
+ diagnosis (see the boundary above), so no issue can carry it.
295002
+
295003
+ \`construction\` requires proving the needed entities and governing guidance were
295004
+ in the returned context. A server trace proves what Publisher returned, not what
295005
+ the host kept after compaction. If the rendered tool response is gone, mark
295006
+ sufficiency \`unknown\` and do not assign \`construction\`.
295007
+
295008
+ Always report construction eligibility as \`eligible / total\`. That is a
295009
+ diagnostic conditional, not a causal comparison.
295010
+
295011
+ ## Step 1: Extract facts from traces, not from memory
295012
+
295013
+ For each \`get_context\` call, load the stored retrieval trace by the \`traceId\` on the
295014
+ \`tool_call\` event. Write down, before you interpret anything:
295015
+
295016
+ - **Asked:** every retrieval utterance, target types, scopes, and result counts,
295017
+ in order.
295018
+ - **Returned:** for each needed entity, whether it appeared, its best
295019
+ within-target rank, and under which utterance. Read this off the
295020
+ \`rankedSummary\` on the attempt's \`tool_call\` events (its \`targets\` list
295021
+ carries per-target ranks); the full trace body is behind your host's
295022
+ trace lookup.
295023
+ Count from the trace, never from recollection.
295024
+ - **Used:** sources and fields the final query referenced, and needed entities
295025
+ that were returned and then unused.
295026
+
295027
+ Resolve aliases to the real source (\`join_one: bldg is fac_building\` uses
295028
+ \`fac_building\`). Count ranks from the trace, not from recollection.
295029
+
295030
+ The needed set comes from golden metadata or from the entities the corrected
295031
+ answer required. Do not invent it from the question's nouns alone.
295032
+
295033
+ Presence leads; rank refines. Every needed entity present with a wrong
295034
+ answer is prima facie \`construction\`. A needed entity that never appeared is
295035
+ never \`construction\`, no matter how wrong the query looks. Everything
295036
+ present but buried deep under noise the agent reasonably skipped is
295037
+ \`get_context/retrieval\` once the request itself was on-target.
295038
+
295039
+ ## Step 2: Assign one primary code
295040
+
295041
+ Use these codes verbatim. Re-wording them destroys the cross-answer pattern.
295042
+
295043
+ ### Dataset first
295044
+
295045
+ | Code | When | Owner |
295046
+ |---|---|---|
295047
+ | \`BAD-REFERENCE\` | the golden itself is wrong, and you can name the defect | dataset |
295048
+ | \`AMBIGUOUS-REFERENCE\` | the key is untrustworthy as a score, but a replacement is not uniquely determined (two honest replays disagree; later cases may confirm a convention) | dataset |
295049
+ | \`BAD-QUESTION\` | the question is unanswerable as written, or underspecified (ties, rank without order) | dataset |
295050
+ | \`CORRECT-SUPERSET\` | every expected row present, plus extra context | none (it passed) |
295051
+
295052
+ A cheap tell for a bad golden: impossible magnitude; identical values across
295053
+ entities that should differ; \`SUM\` / \`COUNT(*)\` over a join that duplicates on
295054
+ both sides. \`AVG\` / \`STDDEV\` / \`MIN\` / \`MAX\` survive uniform duplication, so
295055
+ fanout alone proves nothing.
295056
+
295057
+ **BAD-REFERENCE and AMBIGUOUS-REFERENCE are first-class outcomes, not awkward
295058
+ misses.** Goldens often encode assumptions we want in the model. They can also
295059
+ be wrong. Flag the case when the key is defective or when two justified
295060
+ replays disagree: do not edit the model to match a bad or unsettled key, and
295061
+ do not invent a replacement number. \`BAD-REFERENCE\` goes to Repair a bad
295062
+ golden. \`AMBIGUOUS-REFERENCE\` goes to Hold an ambiguous golden. Do not leave
295063
+ the run looking like the model failed.
295064
+
295065
+ This skill **classifies and hands off**. Write the issue with
295066
+ \`owner: dataset\` and stop for that case. The conductor (\`skill:eval-loop\`)
295067
+ either repairs the golden or holds it as \`ambiguous\`. Do not capture a
295068
+ replacement golden from inside diagnosis if you are not also conducting; a
295069
+ diagnosis that writes a new key without a version bump silently changes what
295070
+ earlier scores meant.
295071
+
295072
+ Prior \`score\` events are not rewritten. They keep the old \`golden_revision\`.
295073
+
295074
+ ### Agent call
295075
+
295076
+ | Code | When | Owner |
295077
+ |---|---|---|
295078
+ | \`NEVER-ASKED\` | no utterance targeted a needed concept | agent-skill, and model if nothing would have prompted the ask |
295079
+ | \`VAGUE\` | compound or generic utterances, so nothing could rank | agent-skill |
295080
+ | \`QUESTION-VOCAB\` | utterances parroted the question where the data uses other words | agent-skill, and model if that vocabulary is undocumented |
295081
+ | \`NO-DISAMBIG\` | two plausible candidates, never resolved | model: docs should answer, not require the question |
295082
+ | \`ASSUMED\` | assumed a scope or convention instead of checking | model if nothing warned; agent-skill otherwise |
295083
+ | \`WRONG-TYPE-OR-SCOPE\` | asked, but with the wrong target type or an empty/wrong scope | agent-skill |
295084
+
295085
+ If the agent could not reasonably have known to ask, that is a model gap.
295086
+
295087
+ ### get_context / model
295088
+
295089
+ | Code | When | Owner |
295090
+ |---|---|---|
295091
+ | \`COVERAGE\` | no representing entity anywhere | model |
295092
+ | \`NOT-RETURNED\` | it exists, the ask was on target, it never came back | model: labels, docs, synonyms, index |
295093
+ | \`LOW-RANK\` | returned, buried under noise the agent reasonably skipped | model |
295094
+ | \`AMBIGUOUS\` | several near-identical candidates | model: "use X for …, Y when …" |
295095
+ | \`GUIDANCE-NOT-RETRIEVED\` | entities came back, governing guidance did not | model: put guidance on the entities agents search for |
295096
+ | \`GUIDANCE-DECLINED\` | guidance was retrieved and judged inapplicable | model: state the business default, not a caveat |
295097
+
295098
+ A missing join is coverage, not an agent-call miss. The model has to volunteer
295099
+ relationships. A declared join is not a retrieval entity; do not look for it in
295100
+ \`get_context\` results.
295101
+
295102
+ ### get_context / retrieval
295103
+
295104
+ | Code | When | Owner |
295105
+ |---|---|---|
295106
+ | \`RETRIEVAL\` | model looks right, utterance on target, rank or grouping still failed | retrieval |
295107
+
295108
+ Prove it before you use this code: search a distinctive phrase from the entity's
295109
+ own doc. If a rare token retrieves it and ordinary phrasing does not, say so
295110
+ with both queries. Otherwise it is still \`NOT-RETURNED\` / \`LOW-RANK\`.
295111
+
295112
+ ### Construction (only after sufficiency)
295113
+
295114
+ | Code | When | Owner |
295115
+ |---|---|---|
295116
+ | \`WRONG-PICK\` | needed entity returned, used a different one | model if indistinguishable; agent-skill if docs distinguished them |
295117
+ | \`SCOPE\` | right entities, wrong population | model if the scope rule was undocumented |
295118
+ | \`GRAIN\` | right entities, wrong grain | model or agent-skill |
295119
+ | \`FILTER-LITERAL\` | filter literal did not match stored values | model (document the stored form) and agent-skill |
295120
+ | \`CONVENTION\` | right data, wrong statistical or business convention | model: expose a named measure |
295121
+ | \`SYNTAX\` | could not express it; execute errors; never submitted | agent-skill |
295122
+
295123
+ ### model-definition
295124
+
295125
+ Use when the entity was found and used, and the definition or the data behind it
295126
+ is wrong (bad grain, wrong join key, inverted filter). Owner: model. A doc whose
295127
+ factual claim the data contradicts (a population statement, a grain claim) is
295128
+ also model-definition: the SQL may be right while the stated contract is false,
295129
+ and an agent that trusts the doc answers wrongly without ever failing a query.
295130
+ Probe the claim before writing the issue.
295131
+
295132
+ ## Step 3: Read the failure shape
295133
+
295134
+ | Signature | Look here |
295135
+ |---|---|
295136
+ | Extremes match, means do not | Population, filter, or join scope |
295137
+ | Same row count, values differ | Wrong column or literal, not joins |
295138
+ | Row count differs, all expected rows present | Superset; often not an error |
295139
+ | Right keys, wrong aggregates on a minority | Undeclared or wrong-cardinality relationship |
295140
+ | Off by a clean integer multiple | Fanout; which side of the join is non-unique |
295141
+ | Zero errors, few calls, fast, confidently wrong | The model steered it |
295142
+ | Identical high-precision values across entities that should differ | Cross-contamination join |
295143
+ | A magnitude that cannot be true | Fanout, possibly in the golden |
295144
+
295145
+ Mine the agent's prose, not only its calls. It often names the gap.
295146
+
295147
+ ## Step 4: Append issue events, then stop
295148
+
295149
+ Append to \`evals/<set>/runs/<runId>/events.jsonl\` with \`kind: issue\`
295150
+ (shapes in \`skill:eval-answer\` \`reference/ledger-schema.md\`):
295151
+
295152
+ - \`issue_id\`, affected \`qids\`, \`primary_code\`, \`contributing_codes\`
295153
+ - \`component\`, \`owner\`, \`severity\`, \`confidence\`
295154
+ - \`sufficiency\` (\`sufficient\` / \`insufficient\` / \`unknown\`)
295155
+ - \`traceId\`s, not copied trace payloads
295156
+ - \`diagnosis\`: the suspected shared entity, file, or root cause, written
295157
+ before any edit exists
295158
+
295159
+ Then \`issue_status\` with \`status: open\`. Status is always an event. Readers
295160
+ take the latest \`issue_status\` for that \`issue_id\`.
295161
+
295162
+ The issue backlog in the event log is the output, not per-question prose.
295163
+ Diagnose reads dev cases only; a holdout case with a bad score stays
295164
+ undiagnosed so the acceptance check keeps something the improve step never saw.
295165
+
295166
+ ## Step 5: Cluster before anyone improves
295167
+
295168
+ Eight failures are rarely eight problems. They are more often two or three,
295169
+ each surfacing in several cases, and the edit worth making is the one that
295170
+ clears a group. So the unit handed to \`skill:eval-improve\` is the cluster, not
295171
+ the case, and one issue event covers all of its cases rather than one per case.
295172
+
295173
+ **Group on shared cause, not shared symptom.** Two cases that both returned a
295174
+ wrong revenue number belong together only if the same entity, doc gap, or
295175
+ convention explains both. Same \`owner\` and same \`component\` is a hint, never a
295176
+ criterion: two \`COVERAGE\` issues about different missing entities are two
295177
+ clusters, and merging them produces an edit that fixes neither cleanly.
295178
+
295179
+ Order clusters by how many cases they would fix. Cluster the non-model owners
295180
+ too, in their own clusters, so nothing is lost on the way to the backlog --
295181
+ but keep them separate, because only \`owner: model\` may proceed to an edit.
295182
+
295183
+ Say what you considered merging and chose not to. A cluster is a claim that one
295184
+ change fixes N cases, and the near-misses are what a reviewer needs to falsify
295185
+ it.
295186
+
295187
+ Still no patch. Naming the shared root cause precisely enough that someone else
295188
+ can design the edit is the whole job here; the edit itself is
295189
+ \`skill:eval-improve\`, working from this. A cluster carrying an honest open
295190
+ question is more useful than one carrying a remedy nobody probed.
295191
+
295192
+ **Only \`owner: model\` proceeds to \`eval-improve\`.** Skill findings go back into
295193
+ the analysis or phrase-detection skill. Retrieval findings go to the tool.
295194
+ \`BAD-REFERENCE\` and \`AMBIGUOUS-REFERENCE\` go to the golden side door in
295195
+ \`skill:eval-loop\` (repair or hold). Do not send them to improve. Other dataset
295196
+ findings (a bad question, a case worth excluding) go back to the case in
295197
+ \`cases.jsonl\` via the conductor. Routing a skill bug into the model is
295198
+ how models accumulate scar tissue.
295199
+
295200
+ ## Anti-patterns
295201
+
295202
+ - Do not diagnose from the answer alone. Probe why a number differed.
295203
+ - Do not treat a passing answer as uninformative. High call counts on a pass
295204
+ still name gaps.
295205
+ - Do not conclude a model gap from two agents agreeing. They coin-flip onto the
295206
+ same undocumented sibling for the same reason.
295207
+ - Do not assign \`construction\` when sufficiency is unknown.
295208
+
295209
+ ## Related skills
295210
+
295211
+ - \`skill:eval-answer\`: the score this consumes.
295212
+ - \`skill:eval-improve\`: smallest model edit, \`owner: model\` only.
295213
+ - \`skill:eval-loop\`: golden hold/repair, the acceptance check, and checkpoint.
295214
+
295215
+ ## Reference files over MCP
295216
+
295217
+ This skill's \`reference/\` files are served as separate prompts, one per file, fetched only when you ask for them. Where the text above says to read \`reference/<name>.md\`, get the prompt named \`eval-diagnose/<name>\` instead.
295218
+
295219
+ Available: output-contract.`
295220
+ },
295221
+ {
295222
+ name: "eval-diagnose/output-contract",
295223
+ description: "Output contract. Reference detail for the eval-diagnose skill.",
295224
+ body: `<!-- What a diagnose agent must emit. Read this before writing your reply. -->
295225
+
295226
+ # Output contract
295227
+
295228
+ Both shapes below are read by a script. Emit the object as the LAST thing in
295229
+ your reply, with nothing after it. Prose before it is fine and expected -- the
295230
+ reasoning is what the codes have to follow from.
295231
+
295232
+ ## Per case (Step 1-4)
295233
+
295234
+ \`\`\`json
295235
+ {"probes": [{"why": "the claim this checks", "query": "query or search text",
295236
+ "result": "what came back, briefly"}],
295237
+ "reasoning": "how the ladder resolved: what you ruled out, and why",
295238
+ "component": "one of the six",
295239
+ "primary_code": "one code, verbatim from the skill",
295240
+ "contributing_codes": ["zero or more, verbatim"],
295241
+ "owner": "model | retrieval | agent-skill | dataset",
295242
+ "sufficiency": "sufficient | insufficient | unknown",
295243
+ "severity": "high | medium | low",
295244
+ "confidence": "high | medium | low",
295245
+ "diagnosis": "the suspected entity, file, or root cause, in one or two sentences",
295246
+ "sharedWith": "a short phrase naming what other cases would share this cause"}
295247
+
295248
+ \`probes\` must be non-empty: it is the record that you checked rather than
295249
+ assumed. \`reasoning\` precedes the codes because the codes must follow from it.
295250
+ \`\`\`
295251
+
295252
+ ## Per run, clustering (Step 5)
295253
+
295254
+ \`\`\`json
295255
+ {"clusters": [
295256
+ {"cluster_id": "short-kebab-slug",
295257
+ "qids": ["every case in this cluster"],
295258
+ "owner": "model | retrieval | agent-skill | dataset",
295259
+ "component": "the shared component",
295260
+ "codes": ["the primary codes present"],
295261
+ "rootCause": "one or two sentences: the ONE thing explaining all of them",
295262
+ "evidence": "why these belong together, and what would prove it wrong",
295263
+ "confidence": "high | medium | low"}
295264
+ ],
295265
+ "reasoning": "what you considered merging and chose not to, and why"}
295266
+
295267
+ Order clusters by the number of qids, descending. Every diagnosed case must
295268
+ appear in exactly one cluster; a case that shares a cause with nothing else is
295269
+ a cluster of one.
295270
+ \`\`\``
295271
+ },
295272
+ {
295273
+ name: "eval-improve",
295274
+ description: "Make the smallest safe Malloy model edit that closes a diagnosed model-owned gap, with a probe receipt for every factual claim. Use after eval-diagnose, or when asked to fix a model so an agent can discover the right answer. Never accepts its own edit; the acceptance check belongs to eval-loop. Does not decide whether an answer was wrong (eval-answer) or why (eval-diagnose).",
295275
+ body: `# Improve the Model
295276
+
295277
+ Takes an issue with \`owner: model\` and produces **one smallest edit** that closes
295278
+ the gap. Every factual claim is backed by a query you ran.
295279
+
295280
+ **Two hard boundaries:**
295281
+
295282
+ 1. **No diagnosis evidence, no edit.** If the issue cannot name a concrete gap
295283
+ with a trace or probe, record that and stop. Edits from an empty diagnosis
295284
+ have been the inert and wrong ones.
295285
+ 2. **This skill never accepts its own edit.** You propose and verify. The
295286
+ acceptance check
295287
+ in \`skill:eval-loop\` admits or reverts. An improver writing the query it
295288
+ already knows proves the fix is possible, not that the next blind agent
295289
+ will find it.
295290
+
295291
+ ## Step 0: What the evidence entitles you to change
295292
+
295293
+ | Evidence | Edits permitted |
295294
+ |---|---|
295295
+ | Verified golden, or a user who states the answer | Any tier. Probes required. Check the golden first. |
295296
+ | Wrong answer, then a corrected one the user accepted | Prefer docs over structure. The diff between attempts is the missing knowledge. |
295297
+ | User accepted, later contradicted | Docs, labels, index only. No structural change. |
295298
+ | Doubt only, or retrieval-only (no verdict) | Docs, labels, index only, and only where the transcript shows a concrete confusion. |
295299
+ | Silence | **No edit.** |
295300
+
295301
+ Do not edit for \`BAD-REFERENCE\` or \`AMBIGUOUS-REFERENCE\`. Those are the
295302
+ golden side door in \`skill:eval-loop\`: repair or hold the golden, bump
295303
+ \`goldenRevision\` on the case, and open a new baseline run. Being right and unmatched
295304
+ beats encoding a defect or an unsettled key. Do not edit for a skill,
295305
+ retrieval, or dataset owner.
295306
+
295307
+ ## Step 1: What a correct answer may teach
295308
+
295309
+ Encode what a domain expert would volunteer unprompted: systems of record,
295310
+ vocabulary to stored codes, what a metric means and at what grain, which
295311
+ relationship is the real one.
295312
+
295313
+ The expert test, per edit: *would a domain expert have said this about their
295314
+ data with no question in front of them?* Reject:
295315
+
295316
+ - a field that hard-codes this question's filter and serves no other question
295317
+ - this question's text, qid, or expected numbers in a doc, comment, or name
295318
+ - a join copied from gold SQL that you have not probed as a real relationship
295319
+
295320
+ The golden is a hypothesis source. The data is still the verifier.
295321
+
295322
+ ## Step 2: Probe receipts
295323
+
295324
+ Every structural claim needs a query you ran: join key, primary key, filter
295325
+ value existence, snapshot assumption, value space, cardinality.
295326
+
295327
+ \`\`\`sql
295328
+ SELECT COUNT(*), COUNT(DISTINCT col) FROM t;
295329
+ SELECT a.k, COUNT(*) FROM a JOIN b ON … GROUP BY 1 ORDER BY 2 DESC;
295330
+ SELECT col, COUNT(*) FROM t GROUP BY 1 ORDER BY 2 DESC LIMIT 5;
295331
+ \`\`\`
295332
+
295333
+ A false \`primary_key\` compiles and silently corrupts every aggregate. Of one
295334
+ pilot's 11 accepted edits, 4 of 5 wrong ones died to a single
295335
+ \`COUNT(*)\` vs \`COUNT(DISTINCT …)\` probe that was never run.
295336
+
295337
+ Compile-check the edit before saving (scope \`file\` for an edit), then reload
295338
+ the package. Confirm it is not serving a stale model.
295339
+
295340
+ This step needs a target you control: a local server, or a host that can
295341
+ execute a draft. A run whose answerers queried a published model cannot be
295342
+ improved in place, because publishing to score an edit is not something this
295343
+ loop does. \`skill:eval-loop\` picks the target before the run starts, so if you
295344
+ have arrived here against a published target, stop and say so rather than
295345
+ publishing.
295346
+
295347
+ Know which copy of the file the server actually reads. Hosts commonly serve a
295348
+ copy of the package rather than your working tree, so editing the model repo
295349
+ and reloading recompiles the unchanged copy: the reload succeeds, nothing
295350
+ changes, and a verification probe quietly tests the old model. Confirm the
295351
+ edit reached what is served before you trust a probe, and keep the model repo
295352
+ the source of truth that gets committed. On open-source Publisher the served
295353
+ copy lives under \`publisher_data/<env>/<pkg>/\` unless the environment is
295354
+ watch-mounted; other hosts distinguish a draft from a published version.
295355
+
295356
+ ## Step 3: One smallest edit
295357
+
295358
+ Prefer edits that add no entities. New sources compete for retrieval and
295359
+ displace answers that already worked.
295360
+
295361
+ | Rank | Edit |
295362
+ |---|---|
295363
+ | 1 | Disambiguating doc on confusable siblings: "use X for …, Y when …" |
295364
+ | 2 | Named dimension or measure in user vocabulary |
295365
+ | 3 | Doc reword, rename, or \`#(index)\` annotation |
295366
+ | 4 | Declared join on a *probed* key |
295367
+ | 5 | A new source: last resort, at most one |
295368
+
295369
+ Make the correct thing the default. Guidance phrased as a caveat
295370
+ ("pair with X", "note that Y also includes Z") is retrieved, read, and
295371
+ declined. A source parameter or named measure that is already the safe
295372
+ scope does not invite a judgment call.
295373
+
295374
+ You cannot append guidance to every field for free. Doc length trades
295375
+ against the entity's own rank. A declared join is invisible to retrieval;
295376
+ put the rule on the entities agents search for.
295377
+
295378
+ Follow the \`malloy-gotchas-modeling\` skill so the edit does not introduce a
295379
+ new modeling mistake. \`improve.py\` installs it with the rest of the \`modeling\`
295380
+ manifest group, so it is loaded alongside this skill rather than reached from
295381
+ here.
295382
+
295383
+ ## Step 4: Check what your edit did to the answer key
295384
+
295385
+ An edit that changes what a field *means* can silently invalidate goldens for
295386
+ questions you were not working on. The rubric still describes the old meaning,
295387
+ the stored value is still the old number, and **nothing fails** -- the case just
295388
+ starts scoring wrong, against the model, in the direction of your edit. The set's
295389
+ own value re-derivation will not catch it either, because it re-runs a
295390
+ \`canonicalQuery\` that encodes the same stale definition.
295391
+
295392
+ Real instance: fixing \`lifetime_orders\` from line items to distinct orders was
295393
+ correct and targeted. It also silently moved \`top_customer\` (defined over it)
295394
+ from 108 customers to 87, and left two rubrics asserting the pre-fix behaviour.
295395
+ Two correct answers were marked wrong for a full run before anyone noticed.
295396
+
295397
+ So before handing off, for **every entity whose meaning you changed** -- not
295398
+ every entity you touched; a doc reword changes no meaning:
295399
+
295400
+ 1. Grep the case file for the entity name. Any rubric, \`canonicalQuery\` or
295401
+ stated value that mentions it is now in question.
295402
+ 2. For each hit, re-derive the value under the new definition and compare it to
295403
+ the stored golden. Different means the golden is stale, not that you are
295404
+ wrong.
295405
+ 3. Run the set's golden verification if it has one, which catches the mechanical
295406
+ subset (drift, and rubric sentences that contradict the model).
295407
+
295408
+ Report every hit as \`golden_suspect\` in the handoff, with the entity, the case,
295409
+ and the old and new values. **Do not repair them yourself.** Goldens are the
295410
+ side door in \`skill:eval-loop\`, and an improver that edits the answer key its own
295411
+ edit is scored against has removed the only independent check on the edit.
295412
+
295413
+ A non-empty \`golden_suspect\` list blocks the acceptance check until the
295414
+ conductor settles
295415
+ each one, because a rerun against stale goldens measures nothing.
295416
+
295417
+ ## Step 5: Verify, report, hand off
295418
+
295419
+ Compile, reload, run one trivial query against each source you touched.
295420
+ Append a \`candidate\` event to the run's \`events.jsonl\` (shape in
295421
+ \`skill:eval-answer\` \`reference/ledger-schema.md\`): the files touched, a
295422
+ one-line diff summary per file, the issue_ids, probe receipts, and this
295423
+ report. Every proposal gets its event, accepted or not; a rejected direction
295424
+ keeps its record. Then stop and wait for the acceptance check in \`skill:eval-loop\`. That
295425
+ skill writes the \`acceptance_check\` event, accepts or reverts, and **only on accept**
295426
+ checkpoints. This skill never checkpoints and never self-accepts.
295427
+
295428
+ \`\`\`
295429
+ COMPONENT / PRIMARY_CODE / OWNER
295430
+ EVIDENCE: class you worked from, and how it limited the edit
295431
+ DISAGREEMENT: NONE, or anything in the diagnosis probing showed was wrong
295432
+ DIAGNOSIS: 2-3 sentences, written before the edit
295433
+ EDIT: one line, or NONE with why
295434
+ EXPERT-TEST: the business fact this encodes
295435
+ PROBES: each probe query and its result
295436
+ GOLDEN-SUSPECT: NONE, or one line per case: qid, entity, stored -> re-derived
295437
+ \`\`\`
295438
+
295439
+ \`DISAGREEMENT\` is load-bearing. An improver that cannot push back encodes
295440
+ its instructions' mistakes. Report from the files on disk, not from memory.
295441
+
295442
+ ## Related skills
295443
+
295444
+ - \`skill:eval-diagnose\`: the issue this requires.
295445
+ - \`skill:eval-loop\`: the acceptance check that accepts or reverts, then
295446
+ checkpoints on accept. Golden hold/repair lives there, not here.
295447
+ - \`skill:eval-answer\`: scoring after a blind re-answer.
295448
+ - The \`malloy-gotchas-modeling\` skill: mistakes an edit must not introduce.
295449
+ It arrives with the \`modeling\` manifest group, not the \`eval\` group.
295450
+
295451
+ ## Reference files over MCP
295452
+
295453
+ This skill's \`reference/\` files are served as separate prompts, one per file, fetched only when you ask for them. Where the text above says to read \`reference/<name>.md\`, get the prompt named \`eval-improve/<name>\` instead.
295454
+
295455
+ Available: output-contract.`
295456
+ },
295457
+ {
295458
+ name: "eval-improve/output-contract",
295459
+ description: "Output contract. Reference detail for the eval-improve skill.",
295460
+ body: `<!-- What an improve agent must emit, and the golden check it must run first. -->
295461
+
295462
+ # Output contract
295463
+
295464
+ WHAT YOUR EDIT MAY HAVE DONE TO OTHER CASES
295465
+
295466
+ The full case file is {cases_file} -- all of them, not just this cluster's. Per
295467
+ the skill's Step 4, if your edit changed what an entity MEANS, grep that file
295468
+ for the entity name and re-derive every golden that depends on it. Report them;
295469
+ do not repair them. Editing the answer key you are scored against is the one
295470
+ thing this loop cannot let you do.
295471
+
295472
+ Give the skill's report block -- COMPONENT, EVIDENCE, DISAGREEMENT, DIAGNOSIS,
295473
+ EDIT, EXPERT-TEST, PROBES, GOLDEN-SUSPECT -- and then, as the last thing in your
295474
+ reply, ONLY this JSON object:
295475
+
295476
+ {"files": ["paths you changed"],
295477
+ "probes": [{"why": "the claim this checks", "query": "...", "result": "..."}],
295478
+ "edit": "one line, or NONE",
295479
+ "editTier": 1-5 from the skill's table, or null,
295480
+ "disagreement": "NONE, or what the diagnosis got wrong",
295481
+ "compiled": true or false,
295482
+ "syncedShaChanged": true or false,
295483
+ "meaningChanged": ["entities whose meaning changed; [] for a docs-only edit"],
295484
+ "goldenSuspect": [{"qid": "...", "entity": "...",
295485
+ "stored": "...", "rederived": "..."}]}
295486
+ \`\`\``
295487
+ },
295488
+ {
295489
+ name: "eval-judge",
295490
+ description: "Decide whether ONE answer matches its golden, and say whether you believe the golden. Read this before emitting any verdict. Covers containment, column pairing, near_match, refusals, and the gold_status judgement. Use when scoring an attempt in an evaluation run; never to conduct a run (eval-loop), diagnose a failure (eval-diagnose) or edit a model (eval-improve).",
295491
+ body: `# The judge
295492
+
295493
+ JUDGE_VERSION: 4
295494
+
295495
+ This skill IS the judge. One fresh judge subagent is spawned per attempt, with
295496
+ this skill installed in its workspace and the case materials in its prompt. It
295497
+ is loaded, not pasted -- so the prompt carries the case and this carries the
295498
+ doctrine, and a judge that needs to read a Malloy query can reach for the
295499
+ skills beside it rather than being handed a transcription.
295500
+
295501
+ Measured when it stopped being pasted, on the case that had oscillated
295502
+ (a valued golden against a model with no trace of the concept):
295503
+
295504
+ pasted into the prompt match / no_match / match / match
295505
+ loaded as this skill no_match x4, and the reasoning cites the rule
295506
+
295507
+ It costs about 2.5x per verdict, which is the price of the judge actually
295508
+ reading its own rules.
295509
+
295510
+ Record \`judge_version\` and this file's git blob sha
295511
+ (\`git rev-parse HEAD:skills/eval-judge/SKILL.md\`, or the model repo's copy) on
295512
+ every verdict, so a rubric change never silently rewrites what old scores
295513
+ meant.
295514
+
295515
+ The judge is not blind. It sees the golden. It must never be the same
295516
+ subagent that answered, and it never edits anything: it returns a verdict
295517
+ object and stops.
295518
+
295519
+ ## Read one of these before you decide
295520
+
295521
+ This file is the decision procedure. Four situations have their own rules, and
295522
+ each is a file beside this one. Read the file BEFORE emitting a verdict, not
295523
+ after -- these are the cases where judging from the general rubric alone gets it
295524
+ wrong, which is why they are called out rather than summarised.
295525
+
295526
+ | If | Read |
295527
+ |---|---|
295528
+ | the answer declines, or gives no value at all | \`reference/refusal.md\` |
295529
+ | the golden itself looks wrong to you | \`reference/suspect-goldens.md\` |
295530
+ | you are judging retrieval, not an answer | \`reference/retrieval-judge.md\` |
295531
+ | you are AUTHORING a case rather than judging one | \`reference/writing-rubrics.md\` |
295532
+
295533
+ The first row is the one that catches people. A refusal is only exempt from
295534
+ containment when \`golden.kind\` is \`unanswerable\`; against a golden that holds a
295535
+ value, an answer containing none of it is \`no_match\` however well it reasons.
295536
+ \`reference/refusal.md\` is the whole rule.
295537
+
295538
+ ## Answer judge
295539
+
295540
+ Input, all of it (a judge with only two row sets grades formatting, not
295541
+ intent):
295542
+
295543
+ - the question, exactly as the answerer saw it
295544
+ - the golden: rows or scalar, plus \`canonicalQuery\` when present
295545
+ - the prediction: the rows the CONDUCTOR re-executed from the answerer's
295546
+ \`final_query\` (never the answerer's self-reported rows)
295547
+ - the relevant source and field definitions from the model (docs, join list)
295548
+
295549
+ Output, exactly this shape:
295550
+
295551
+ \`\`\`json
295552
+ {
295553
+ "verdict": "match | near_match | no_match",
295554
+ "confidence": 7,
295555
+ "why": "one short paragraph",
295556
+ "column_pairing": { "gold_col": "pred_col", ... },
295557
+ "gold_status": "verified | verified_benign | suspect | verified_wrong",
295558
+ "gold_note": "why, when not verified"
295559
+ }
295560
+ \`\`\`
295561
+
295562
+ ### Rubric
295563
+
295564
+ 1. **Judge intent, not formatting.** The question defines what counts. A
295565
+ result that answers the question in a different but faithful shape is a
295566
+ match.
295567
+ 2. **Gold-subset containment.** The prediction must CONTAIN the gold answer.
295568
+ Extra columns or benign extra context downgrade to \`near_match\` at worst;
295569
+ they never make a containing answer \`no_match\`.
295570
+ 3. **Name the column pairing.** Pair each gold column with the prediction
295571
+ column that carries the same meaning, using names, the question's role for
295572
+ the value, and the values together. Never pair numeric columns by value
295573
+ overlap alone: a year column is not a count column even when magnitudes
295574
+ overlap. If a gold column has no counterpart, say which.
295575
+ 4. **Rows are a multiset.** Order matters only when the question asks for an
295576
+ order. For a "top N" with possible ties, check that the boundary value is
295577
+ right and every returned row legitimately qualifies; any valid tie-break is
295578
+ a match.
295579
+ 5. **Tolerances.** Numeric equality within small rounding (relative 1e-6, or
295580
+ the display precision the golden uses). A percentage and its fraction
295581
+ (50 and 0.5) are the same value in different units when the pairing says
295582
+ the column is a rate.
295583
+ 6. **Confidence 1 to 10.** 5 or lower means the case needs a human:
295584
+ the conductor records \`needs_human\`, which is neither a pass nor a fail.
295585
+ Do not inflate confidence to be helpful; a wrong confident verdict is worse
295586
+ than an abstention.
295587
+ 7. **\`near_match\` is not a soft pass, and it is not a soft fail.** It is a
295588
+ third outcome meaning *defensibly different*: the answer took a reading the
295589
+ rubric allows but did not prefer, broke a tie the other way, or buried a
295590
+ caveat that should have been plain. It is excluded from the pass rate and
295591
+ from the acceptance check, exactly like \`needs_human\`.
295592
+
295593
+ So do not reach for it to avoid a hard call. If the prediction contains the
295594
+ gold answer, that is \`match\` -- extra columns and benign extra context never
295595
+ reduce it (rule 2). If it does not, and the rubric does not sanction the
295596
+ reading that produced it, that is \`no_match\`. Use \`near_match\` only when you
295597
+ can name the rubric clause that makes the difference defensible.
295598
+
295599
+ It is a third outcome because as a pass it was a large share of the measured
295600
+ noise: the same unchanged answer reads \`match\` in one run and \`near_match\`
295601
+ in the next, and the pass rate moves although nothing did. A verdict whose
295602
+ content is "this is arguable" cannot be allowed to decide anything. Its
295603
+ count is still reported, and a rising one means the rubrics are going vague.
295604
+ (What that share was for a given set is in that set's calibration record.)
295605
+ 8. On a large row set, compare it as a set rather than scanning pairwise: state
295606
+ how many gold rows you located in the prediction, name the ones you could
295607
+ not, and say what the mismatched values look like (uniformly scaled, off in
295608
+ one column, a different population). "I checked all 76" without that
295609
+ breakdown is not a comparison.
295610
+ 9. **Score the data, not the insight.** A question that asks for a figure or
295611
+ a series is judged on the figure or the series. Where the question also asks
295612
+ for an interpretation -- "when did it flatten out", "what drove the change"
295613
+ -- that interpretation is not scored unless the rubric marks it \`REQUIRED\`
295614
+ with a criterion that resolves from the data alone. Two analysts reading the
295615
+ same exact curve name different weeks; an eval that scores which week they
295616
+ named is measuring taste, and a run that lost a case that way (13 of 13
295617
+ weekly values exact, plateau named one week outside a window) was measuring
295618
+ nothing. Exact data with a different reading of it is \`match\`.
295619
+ 10. **Do not demand a grain the question did not fix.** When the question names
295620
+ no grain -- by medium, by week, campaign total -- a figure that is correct at
295621
+ the grain the answer states is correct. The golden's grain is \`PREFERRED\`,
295622
+ not the only one: an answer at another grain is \`match\` when the grain is
295623
+ stated and the figures are right at it; \`near_match\` when the grain is left
295624
+ unstated; \`no_match\` only when the figures are wrong at the grain claimed. An
295625
+ answer that named the right segment and showed the index split by medium,
295626
+ every number right, was once scored down for not showing the campaign
295627
+ total; the question had never asked for one. A rubric that means "campaign
295628
+ total only" must say so as \`REQUIRED\`, and the question should say so too.
295629
+
295630
+ ### Anchors
295631
+
295632
+ - **match**: question "total sales by category"; golden 8 rows
295633
+ \`(category, revenue)\`; prediction 8 rows \`(product_category,
295634
+ gross_revenue, order_count)\`. Same categories, revenues equal within
295635
+ rounding; the extra count column does not change what the answer says.
295636
+ Verdict: match, confidence 9.
295637
+ - **near_match**: question "top 5 states by returns"; golden and prediction
295638
+ agree on 4 of 5 states, and the disagreement is at rank 5 where two states
295639
+ tie exactly; the prediction chose the other tie-break. The boundary value
295640
+ is right, the membership defensible, but the golden pinned one tie-break.
295641
+ Verdict: near_match, confidence 7, why names the tie.
295642
+ - **no_match**: question "revenue in 2024, completed orders only"; golden
295643
+ 1.2M; prediction 1.9M and the pairing shows the prediction summed all
295644
+ statuses. Same shape, wrong population. Verdict: no_match, confidence 9.
295645
+
295646
+ Keep the anchor set balanced. A judge shown only matches learns a base rate,
295647
+ not a rubric.
295648
+
295649
+ ### Coverage
295650
+
295651
+ A case may be labelled \`coverage: derivable\`: the model has no entity for the
295652
+ concept and the answer had to be built from the parts that exist. Judge the
295653
+ result exactly as the rubric says -- a derived answer that matches the golden is a
295654
+ \`match\`, and the absence of a named measure is not a deduction. But when the
295655
+ answer states what it built, say so in the why. That sentence is what tells
295656
+ diagnosis the gap is real and lets \`coverage_note\` become a model edit rather
295657
+ than a guess.
295658
+
295659
+ ## Versioning and regressions
295660
+
295661
+ Any change to this file is a judge change: bump JUDGE_VERSION, commit, and
295662
+ re-run \`evals/<set>/judge-regressions.jsonl\` (the human-overruled verdicts)
295663
+ before trusting new scores. Runs record \`judge_version\` and \`rubric_sha\`, so
295664
+ a delta across a rubric change is attributable to the rubric, not the model.
295665
+
295666
+ ## Reference files over MCP
295667
+
295668
+ This skill's \`reference/\` files are served as separate prompts, one per file, fetched only when you ask for them. Where the text above says to read \`reference/<name>.md\`, get the prompt named \`eval-judge/<name>\` instead.
295669
+
295670
+ Available: refusal, retrieval-judge, suspect-goldens, writing-rubrics.`
295671
+ },
295672
+ {
295673
+ name: "eval-judge/refusal",
295674
+ description: "Refusal. Reference detail for the eval-judge skill.",
295675
+ body: `<!-- How to score an answer that declines. Read this WHENEVER the answer gives no value. -->
295676
+
295677
+ # Refusal
295678
+
295679
+ **STOP. Check \`golden.kind\` before reading further.** This section applies ONLY
295680
+ when it is \`unanswerable\`. If the golden carries a value or rows, close this
295681
+ section and score by containment like any other answer: an answer that declines,
295682
+ however well it reasons, contains none of the golden's numbers and is
295683
+ \`no_match\`.
295684
+
295685
+ That the model genuinely lacks the field is NOT a reason to pass a refusal.
295686
+ Whether the model should be able to answer is what \`coverage\` records and what
295687
+ \`eval-diagnose\` decides. Settling it here converts a model gap into a passing
295688
+ case, and the gap then never reaches the backlog.
295689
+
295690
+ This rule is here because refusals against a valued golden are where the judge
295691
+ is least stable, and the instability has been localised rather than guessed at.
295692
+ Holding the answer, the rubric and the golden fixed and varying ONLY the model
295693
+ source shown to the judge, over samples of three to four:
295694
+
295695
+ | model source shown | verdicts |
295696
+ |---|---|
295697
+ | lacks the concept entirely | \`match\` / \`no_match\` / \`match\` / \`match\` -- unstable |
295698
+ | defines something adjacent | \`no_match\` x3 -- stable |
295699
+ | withheld | \`no_match\` x3 -- stable |
295700
+
295701
+ So a model with no trace of the concept is what destabilises the verdict: the
295702
+ judge starts weighing whether the answerer *could* have complied instead of
295703
+ whether it did. Four prompt edits were tried against it -- this rule, deleting
295704
+ the Refusal section, deleting the model-beats-rubric bullet, and splitting that
295705
+ bullet into "the model CONTRADICTS the rubric" versus "the model LACKS what the
295706
+ rubric names" -- and none of them stabilised it.
295707
+
295708
+ Treat a refusal on a coverage case as unstable until that changes: score it with
295709
+ \`check_judge.py --repeat\`, not from one verdict. The rule below is still the
295710
+ rule; it is just not yet enforceable by prompt alone.
295711
+
295712
+ Applies when the case's \`golden.kind\` is \`unanswerable\`. Every rule above assumes
295713
+ a gold result to contain and columns to pair, and here there is neither: no
295714
+ value, no canonical query, no rows. Do not attempt containment. Your input is the
295715
+ question, the case rubric -- which names the missing data and the proxy an agent
295716
+ is expected to reach for -- and the answer text.
295717
+
295718
+ A refusal is not scored on tone or on hedging. It is scored on whether the reader
295719
+ is left correctly informed that the metric is unavailable, and why.
295720
+
295721
+ Rule 7 applies here too: the \`near_match\` below is not a pass. A refusal that is
295722
+ right but useless does not count towards the pass rate, and it does not count
295723
+ against it either.
295724
+
295725
+ - **match** -- declines, and names the specific missing data rather than declining
295726
+ generically. Offering the proxy *with an unmissable label* is also a match:
295727
+ reporting revenue by channel while stating plainly that it is not ROAS because
295728
+ no cost exists is the most useful possible answer and must not be penalised.
295729
+ - **near_match** -- declines correctly but generically. "I don't have enough
295730
+ information" is the right call delivered uselessly; the reader cannot tell
295731
+ whether the data is missing, the question is ambiguous, or the agent gave up.
295732
+ Also \`near_match\` when a proxy is offered but its caveat is buried or hedged.
295733
+ - **no_match** -- produces a number for the metric as asked, whether invented or
295734
+ computed from a proxy presented as the answer. Severity does not fall as the
295735
+ proxy gets closer: a confident 3.51% cart-abandonment rate derived from
295736
+ cancellations is a wrong answer to a question the data cannot answer, and being
295737
+ plausible makes it worse rather than better.
295738
+
295739
+ Two rules override the generosity built into the main rubric.
295740
+
295741
+ 9. **Absence of a number is not evidence of a refusal.** An answer that fails,
295742
+ errors, or wanders off without addressing availability is not a pass. The
295743
+ refusal has to be a stated conclusion. This is distinct from \`submitted:
295744
+ false\`, which the conductor already records as not scorable.
295745
+ 10. **When the case is tagged \`answerable-sounds-unanswerable\`, a refusal is
295746
+ \`no_match\`.** Score it against the golden normally. Rule 2's "benign extra
295747
+ context never makes a containing answer \`no_match\`" does not apply to an
295748
+ answer that contains nothing. A caveat attached to a *correct* answer is
295749
+ welcome and does not reduce the verdict.
295750
+
295751
+ Both directions are scored on purpose. Measuring only whether an agent declines
295752
+ when data is absent rewards one that declines everything, and such an agent is
295753
+ worthless. A set carrying this section should pair the two: in the ecommerce set
295754
+ \`ecom_stockout_revenue_loss\` (refuse) and \`ecom_unsold_stock_value\` (answer)
295755
+ resolve to the same figure, one as an inadmissible proxy for revenue lost to
295756
+ demand that never existed and the other as the honest cost of stock on hand. Only
295757
+ the question distinguishes them, so no policy passes both -- the agent has to read
295758
+ the question.
295759
+
295760
+ #### Anchors
295761
+
295762
+ - **match**: "What is our ROAS?" -- the answer says there is no ad-spend data
295763
+ anywhere in the model, notes that \`traffic_source\` is the account's acquisition
295764
+ channel rather than a cost, and offers revenue by channel explicitly labelled as
295765
+ not being ROAS. Confidence 9.
295766
+ - **near_match**: same question; the answer says "I can't calculate that with the
295767
+ available data" and stops. Correct, and the reader learns nothing about what is
295768
+ missing or whether another source would fix it. Confidence 7.
295769
+ - **no_match**: same question; the answer divides revenue by traffic source and
295770
+ reports a ROAS per channel. Every figure is arithmetically right and the label
295771
+ is false. Confidence 9.
295772
+ - **no_match**: "How much are we sitting on in unsold inventory?", tagged
295773
+ \`answerable-sounds-unanswerable\`; the answer declines for want of an inventory
295774
+ snapshot. The data answers it, and "ever unsold" needs no snapshot -- only
295775
+ "unsold as of a date" would. Confidence 9.`
295776
+ },
295777
+ {
295778
+ name: "eval-judge/retrieval-judge",
295779
+ description: "Retrieval judge. Reference detail for the eval-judge skill.",
295780
+ body: `<!-- A different job from scoring an answer. Read this only when judging retrieval. -->
295781
+
295782
+ # Retrieval judge
295783
+
295784
+ Input:
295785
+
295786
+ - the intent row: \`term\`, \`entityType\`, \`description\` (the rich intent, the
295787
+ thing you actually judge against)
295788
+ - the ranked entities a \`get_context\` call returned for that term, each with
295789
+ its within-target rank and doc
295790
+
295791
+ Two judgments:
295792
+
295793
+ 1. **In scope?** Does THIS model version contain an entity representing the
295794
+ described concept at all, anywhere, regardless of whether it was returned?
295795
+ \`in_scope: false\` is a coverage gap, charged to the model's coverage, not
295796
+ to retrieval.
295797
+ 2. **Per returned entity**: \`match\` (represents the described intent),
295798
+ \`near_match\` (the concept overlaps but the intent might want something
295799
+ broader or narrower; retrieving \`net_revenue\` for the term "revenue" is a
295800
+ near match), or \`no_match\`. Confidence 1 to 10 and a one-line why, each.
295801
+
295802
+ Rule 7 does **not** apply to retrieval. Here \`near_match\` counts towards recall
295803
+ and precision, and should: handing back an overlapping entity is a real
295804
+ retrieval success, since the agent can read the doc and decide. The answer judge
295805
+ excludes it because there the same word means "the answer might be wrong".
295806
+
295807
+ Output:
295808
+
295809
+ \`\`\`json
295810
+ {
295811
+ "in_scope": true,
295812
+ "judgments": [
295813
+ { "entityId": "measure:orders:total_sales", "rank": 1,
295814
+ "level": "match", "confidence": 9, "why": "..." }
295815
+ ]
295816
+ }
295817
+ \`\`\`
295818
+
295819
+ The conductor computes coverage, recall, and precision by counting these
295820
+ (\`reference/ledger-schema.md\`). The judge only judges.`
295821
+ },
295822
+ {
295823
+ name: "eval-judge/suspect-goldens",
295824
+ description: "When the answer key looks wrong. Reference detail for the eval-judge skill.",
295825
+ body: "<!-- How to set gold_status. Read this when the golden itself looks wrong. -->\n\n# When the answer key looks wrong\n\nScore against the golden as written. Then say, separately, whether you believe\nit. Those are two different jobs and `gold_status` is the second one.\n\n**The verdict never bends.** If the prediction does not contain the golden, that\nis `no_match`, whatever you think of the golden. An answer does not pass because\nyou suspect the key. Doubt goes in `gold_status`, and something downstream\nadjudicates it; a judge that quietly graded against its own better answer would\nbe the only record of having done so.\n\n| Value | Meaning |\n|---|---|\n| `verified` | No reason to doubt it. The default, and the honest answer nearly always. |\n| `verified_benign` | Reachable defect that cannot change this verdict -- e.g. join fanout under an `AVG`, `MIN`, `MAX` or `STDDEV`, which uniform duplication does not move. |\n| `suspect` | Something does not add up and you cannot settle it from what you were given. |\n| `verified_wrong` | You can demonstrate the key is wrong, and say how. Excludes the case from run aggregates, so the bar is demonstration, not suspicion. |\n\nWhat earns more than `verified`:\n\n- **The rubric contradicts the model.** You have the model source. A rubric\n saying \"`lifetime_orders` counts line items despite its name\" against a model\n reading `lifetime_orders is count(order_id)` is a rubric written before a fix\n and never revisited. That is `suspect` at least, and the judge is the only\n station positioned to notice -- this exact case failed two correct answers for\n a full run.\n- **The golden and its own `canonicalQuery` disagree**, where you can see both.\n- **The golden is impossible against the re-executed rows** -- a total below one\n of its own parts, a rate outside 0 to 1, a count above the population.\n- **Fanout you can identify**, benign or otherwise, per the classification above.\n\nWhat does not: the answer being more useful, better presented, or more recent\nthan the key. Disagreeing with the question's premise is not a defect in the\nanswer to it.\n\n`gold_note` says what you saw, concretely enough to check -- the two values, or\nthe model line against the rubric sentence. \"Golden looks off\" routes nothing."
295826
+ },
295827
+ {
295828
+ name: "eval-judge/writing-rubrics",
295829
+ description: "Writing a rubric the judge can execute. Reference detail for the eval-judge skill.",
295830
+ body: `<!-- For whoever AUTHORS a case. Not needed to judge one. -->
295831
+
295832
+ # Writing a rubric the judge can execute
295833
+
295834
+ A case rubric is not prose for a human to weigh. It is the part of the judge's
295835
+ instructions that changes per case, so every clause in it must resolve to a
295836
+ verdict. Where one does not, the judge supplies the missing rule itself, and
295837
+ supplies a different one next time -- which reads as model noise and is not.
295838
+
295839
+ Two clause types cause almost all of it. Both must carry their consequence.
295840
+
295841
+ **An alternate reading** -- a second defensible answer to the same question.
295842
+ Mark each one, and never leave the set open:
295843
+
295844
+ | Marker | Verdict | Use when |
295845
+ |---|---|---|
295846
+ | \`PREFERRED\` | \`match\` | The reading the golden encodes. Exactly one. |
295847
+ | \`ACCEPT\` | \`match\` | Equally right. A different but faithful route to the same claim. |
295848
+ | \`DIVERGENT\` | \`near_match\` | Defensible, and not what was asked for. Usually a population or grain the model does not distinguish. |
295849
+ | \`WRONG\` | \`no_match\` | Plausible and incorrect. Name the trap value so the judge can recognise it. |
295850
+
295851
+ **A disclosure** -- something the answer must SAY, beyond the number. Say what
295852
+ silence costs:
295853
+
295854
+ | Marker | Verdict when omitted | Use when |
295855
+ |---|---|---|
295856
+ | \`REQUIRED\` | \`no_match\` | Without it the answer misleads. A year-over-year figure over a truncated year is the case: the number is right and the reader draws a false conclusion from it. |
295857
+ | \`CREDITED\` | \`match\`, no deduction | It adds context a good analyst would give. Its absence leaves the reader correct but less informed. |
295858
+
295859
+ Rules that follow from this:
295860
+
295861
+ - **Write the question so its answer is data.** A question is a request for a
295862
+ figure, a series, or a set of rows -- things a truth query can produce and a
295863
+ judge can compare. "How did reach build week by week" is a question; "and
295864
+ when did it flatten out" is a request for an opinion about the answer, and
295865
+ no golden can hold one. Put interpretation in a \`CREDITED\` clause if it is
295866
+ worth noting, never in the question and never as a scored window.
295867
+ - **Fix the grain in the question, or accept every grain in the rubric.** If the
295868
+ golden is a campaign total and a by-medium answer would be wrong, the question
295869
+ must say "for the campaign as a whole". If it does not, the rubric must accept
295870
+ a correct figure at any stated grain (judge rule 10). A rubric that quietly
295871
+ assumes the golden's grain fails correct answers.
295872
+ - **A right value plus a missing \`CREDITED\` disclosure is a \`match\`.** Not a
295873
+ near match. Do not deduct for it.
295874
+ - **\`DIVERGENT\` is about definitions, not arithmetic.** A clause permitting a
295875
+ different population, grain or convention never excuses a computational
295876
+ error. If a rubric tolerates a shift in the third decimal and the answer is
295877
+ out by a whole unit, that is \`no_match\` however well the narrative reads.
295878
+ - **An unmarked clause is \`CREDITED\`.** The judge must not invent a
295879
+ requirement. A rubric that meant to require something and did not say so is
295880
+ the rubric's bug, and the fix belongs in the case.
295881
+ - **Stable \`near_match\` is a finding, not an outcome.** A case that lands there
295882
+ in run after run is telling you the model cannot distinguish two readings that
295883
+ the question does. That is a coverage gap for \`eval-diagnose\`, and repairing
295884
+ the rubric will not close it.`
295885
+ },
295886
+ {
295887
+ name: "eval-loop",
295888
+ description: "Conduct a local Publisher evaluation loop in five steps: scrape/run, eval, diagnose, improve, checkpoint. You are the conductor: import cases into the file ledger, spawn a blind answerer, then run eval-answer, eval-diagnose, and eval-improve. Persistence is plain files under the model package''s evals/ directory; checkpoints are git commits of the model repo. Use to score a model, diagnose failures, improve behind an acceptance check, or roll back a bad direction.",
295889
+ body: `# The Evaluation Loop
295890
+
295891
+ You conduct this loop. There is no batch orchestrator to start, no eval API,
295892
+ and no eval MCP tools. The ledger is plain files in the model package's git
295893
+ repository (\`reference/ledger-schema.md\` in \`skill:eval-answer\` defines every
295894
+ file and event). Scoring is an LLM judge you spawn per case. There is no
295895
+ scripted scorer, and there will not be one: a script that can pass a wrong
295896
+ answer is worse than none. The scripts under \`scripts/\` run the loop -- they
295897
+ answer, re-execute, spawn the judge, compare runs, and write the ledger -- but
295898
+ none of them decides whether an answer was right.
295899
+
295900
+ \`\`\`
295901
+ scrape/run -> eval -> diagnose -> improve -> checkpoint
295902
+ \`\`\`
295903
+
295904
+ **This skill conducts; it does not restate.** Scoring lives in
295905
+ \`skill:eval-answer\`. Components and owners live in \`skill:eval-diagnose\`.
295906
+ Edit rules live in \`skill:eval-improve\`.
295907
+
295908
+ Do not merge **eval** into **diagnose**. A conductor who scores while
295909
+ explaining writes the explanation into the score. Do not skip the **acceptance
295910
+ check** inside improve. The acceptance check decides whether *this* edit
295911
+ stays. **Checkpoint** decides whether a *sequence* of accepted edits can be
295912
+ undone.
295913
+
295914
+ ## Where the rest of this lives
295915
+
295916
+ This file is the procedure. Five things it used to carry inline are files beside
295917
+ it now, because each is needed at one moment rather than every run, and loading
295918
+ all of them for every run is how a skill stops being read.
295919
+
295920
+ | When | Read |
295921
+ |---|---|
295922
+ | about to run one | \`reference/running-a-run.md\` |
295923
+ | a golden is wrong, doubted, or out of step with the model | \`reference/golden-side-door.md\` |
295924
+ | deciding whether an edit stays | \`reference/acceptance-check.md\` |
295925
+ | about to quote a number, or set the noise band | \`reference/measurement.md\` |
295926
+ | you changed judge doctrine or its inputs | \`reference/checking-the-judge.md\` |
295927
+
295928
+ Read the file, do not work from the summary here. The acceptance-check rules and
295929
+ the golden side door are both places where acting on a half-memory of the rule
295930
+ produces a confident wrong answer rather than an error.
295931
+
295932
+ ## The five steps
295933
+
295934
+ | Step | Job | Writes |
295935
+ |---|---|---|
295936
+ | **a. scrape / run** | Put cases in the ledger; spawn a blind answerer | cases; \`attempt\`, \`tool_call\` |
295937
+ | **b. eval** | Judge the answer; score which required entities retrieval delivered | \`score\` |
295938
+ | **c. diagnose** | Why it failed, who owns it | \`issue\` / \`issue_status\`. Stop. Do not edit. |
295939
+ | **d. improve** | One smallest model edit, then the acceptance check | improve writes \`candidate\`; you write \`acceptance_check\`. Revert on reject. |
295940
+ | **e. checkpoint** | Git commit after an accepted acceptance check | \`checkpoint\` event, then the commit |
295941
+
295942
+ **scrape** and **run** share a letter but are not the same job. Scrape writes
295943
+ cases. Run writes attempts. Do not invent questions and score
295944
+ them in one breath.
295945
+
295946
+ ### Scrape, minimally
295947
+
295948
+ Importing an existing corpus IS the scrape step: copy the set from its home
295949
+ (for example a benchmarks checkout) into \`evals/<set>/\` and convert to the
295950
+ ledger shapes. While importing:
295951
+
295952
+ - Freeze each case's \`split\`: \`dev\` or \`holdout\`. Diagnose and improve read
295953
+ dev cases only; the acceptance check runs both. A set that is all dev cannot defend an
295954
+ accept.
295955
+ - Later, each diagnosed-and-fixed failure becomes a new frozen dev case, so a
295956
+ fixed bug cannot silently return.
295957
+
295958
+ Scraping from production logs (chat transcripts, retrieval traces) is the
295959
+ other supported source, and usually the better one: real traffic asks what
295960
+ people actually ask. Where your logs physically live is a host concern; look
295961
+ for a host-specific log-fetching skill.
295962
+
295963
+ Prefer variety over volume when you sample, from either source. Cases that
295964
+ differ in grain, source, filter shape, and phrasing are what move a
295965
+ measurement; a second sample of the same case is nearly free of new
295966
+ information.
295967
+
295968
+ ### Mode aliases
295969
+
295970
+ Older mode names still work as aliases for how far one run walks:
295971
+
295972
+ | Alias | Steps |
295973
+ |---|---|
295974
+ | \`measure\` | scrape/run + eval |
295975
+ | \`triage\` | plus diagnose |
295976
+ | \`improve\` | plus improve + acceptance check + checkpoint on accept |
295977
+
295978
+ Say which alias (or which steps) you are running before the first question.
295979
+ Record it in \`run.json\`. Do not mix steps in a way that lets the answerer see
295980
+ gold, issues, or the model file.
295981
+
295982
+ Most runs should stop after eval. Diagnose when you need a histogram of
295983
+ components and owners. Improve only for diagnosed *model* gaps, one batch at
295984
+ a time. Checkpoint only after the acceptance check **accepts**.
295985
+
295986
+ ## Roles
295987
+
295988
+ | Role | Sees |
295989
+ |---|---|
295990
+ | **Answerer** | The question and the Malloy tools. Never the golden, \`evals/\`, the model file, or any hint it is being evaluated. |
295991
+ | **Judge** | The golden and the prediction. Never conducts, never answers, never edits. One fresh subagent per verdict (\`skill:eval-judge\`). |
295992
+ | **You (conductor / improver)** | Everything, including goldens and traces. |
295993
+ | **Acceptance check** | The edit and the evidence. Never the improver's self-assessment alone. |
295994
+
295995
+ The answerer stays blind. That is not optional. A grader-visible answerer
295996
+ writes toward the expected answer, and the score is fiction.
295997
+
295998
+ There are no eval MCP tools on purpose. The answerer inherits your tools,
295999
+ including Shell and Read, so any eval convenience surface would also be a
296000
+ gold path for it. Blindness is prevention plus detection, not a guarantee:
296001
+ \`eval-answer\` runs the contamination checklist on every attempt, which is
296002
+ why you keep a host-side tool-use log per answerer.
296003
+
296004
+ ## Pick the target first
296005
+
296006
+ Both a local model server and a hosted platform expose the same two tools the
296007
+ answerer needs, \`get_context\` and \`execute_query\`, so the loop runs against
296008
+ either. What differs is which model is answering and whose data it reads, and
296009
+ those are two separate axes:
296010
+
296011
+ | Target | Model under test | Data | Can edit and re-test? |
296012
+ |---|---|---|---|
296013
+ | **Local (direct)** | your working files | local (for example duckdb), or a direct warehouse connection | yes |
296014
+ | **Local (proxied)** | your working files | the platform's connection, through a proxy connection type | yes |
296015
+ | **Remote** | the published version, through the platform's hosted \`get_context\`/\`execute_query\` | the platform's | no, publishing is not an eval action |
296016
+
296017
+ The middle row is the one worth knowing about: it decouples the two axes, so you
296018
+ can evaluate a model you are still editing against the customer's real data. It
296019
+ is a connection configuration, not a feature.
296020
+
296021
+ Two rules follow, and both are the kind of mistake that produces confident
296022
+ nonsense rather than an error:
296023
+
296024
+ - **The answerer and the conductor must hit the same target.** If the answerer
296025
+ queries the published model and you re-execute its query against your edited
296026
+ local copy, the score describes neither. Decide the target before the first
296027
+ question and record it.
296028
+ - **Pin the version the target actually served, not the one you happen to have.**
296029
+ A local target pins a commit; a platform target pins the published version.
296030
+ Recording a local commit for a run that queried a published model is a pin
296031
+ that means nothing.
296032
+
296033
+ Which target for which job:
296034
+
296035
+ - **Baseline what customers experience:** Remote. It is the deployed model
296036
+ through the deployed engine, which is the thing they actually hit. The judge
296037
+ sees no re-executed rows on a Remote run (there is no local copy of the
296038
+ bytes), so its verdicts rest on the answer text and the golden; say so.
296039
+ - **Improve and accept:** local, because the acceptance check needs compile,
296040
+ reload, and a fresh re-answer between edits. Publishing to a customer
296041
+ environment to score an edit is not something this loop does. Where the host
296042
+ offers draft execution, that counts as local for this purpose.
296043
+ - **Measure real data without touching production:** local proxied.
296044
+
296045
+ So a measure-only run can use any target; a run that includes **improve** needs
296046
+ a local one.
296047
+
296048
+ Two things to check before a platform run, because neither errors and both make
296049
+ the run measure something other than what it names:
296050
+
296051
+ - **The answerer's skills must be written for THIS host.** A shared skill names
296052
+ an MCP tool by its bare name (\`get_context\`) so it reads correctly anywhere,
296053
+ but a host/router skill names its own host's tools directly. Install the
296054
+ latter for the wrong host and the answerer is told to call tools it does not
296055
+ have. \`run_baseline.py\` warns when the manifest it loaded names Publisher-only
296056
+ tools on a platform target; point \`--answerer-manifest\`, or \`--skills-root\`,
296057
+ at the checkout that ships this host's manifest.
296058
+ - **The tool names are configuration.** \`--hosted-mcp-server\` is both the
296059
+ \`mcp__<server>__<tool>\` prefix and the OAuth cache key, so it has to match the
296060
+ name the answerer authenticated under, and \`--hosted-tools\` lists the bare
296061
+ tools that host exposes.
296062
+ - **Get the hosted tools in front of a headless answerer, one of two ways.**
296063
+ A spawned answerer cannot complete an OAuth flow, so the tools have to be
296064
+ reachable before the run starts. \`run_baseline.py\` proves it with one cheap
296065
+ probe and refuses to spend an arm otherwise -- a run whose answerers have no
296066
+ tools does not error, it reads as a terrible model.
296067
+
296068
+ 1. **Authenticate once, interactively.** Works anywhere, including a plain
296069
+ CLI install, and is the route to assume unless you know otherwise. The
296070
+ token is cached per server NAME, so authenticate under the same name the
296071
+ run passes to \`--hosted-mcp-server\`:
296072
+
296073
+ \`\`\`bash
296074
+ claude mcp add --transport http <name> <scoped-url>
296075
+ claude # then /mcp -> <name> -> Authenticate
296076
+ \`\`\`
296077
+
296078
+ Then come back and run. This is a hand-off to a person; there is no
296079
+ headless equivalent, so plan for it rather than discovering it mid-run.
296080
+
296081
+ 2. **A local proxy that already holds the credential.** Some hosts ship an
296082
+ editor extension whose local MCP proxy can expose the hosted
296083
+ \`get_context\` / \`execute_query\` -- often behind a setting that is off by
296084
+ default. Where that exists, point \`--mcp-url\` at the proxy on localhost
296085
+ and no OAuth step is needed, because the extension holds it. Check what
296086
+ the proxy actually exposes before relying on it: the same proxy may serve
296087
+ a local Publisher's \`malloy_*\` tools instead, and then \`--hosted-tools\` is
296088
+ naming tools that are not there. This route is not available to someone
296089
+ running the CLI alone.
296090
+
296091
+ - **Prefer a SCOPED endpoint URL over asking for scope.** A hosted MCP is
296092
+ usually reachable two ways: a global endpoint where every call carries an
296093
+ organization and workspace, and a scoped one where the URL itself is the
296094
+ scope. \`--scope\` and the prompt can only ASK an answerer to stay in one
296095
+ package; a scoped URL enforces it. For an agent being measured that is the
296096
+ difference between a case answered against the package it names and one
296097
+ answered against whatever else the account can see. Authenticate once
296098
+ interactively (\`claude\`, \`/mcp\`) under the same server name the run will use;
296099
+ the token is cached per name, and a spawned headless answerer cannot complete
296100
+ an OAuth flow.
296101
+
296102
+ ## Before you start
296103
+
296104
+ 1. The model package under evaluation must live in a git repository, with
296105
+ \`evals/<set>/\` in the package, beside the model files. Git is the checkpoint
296106
+ mechanism; without it there is no rollback and no run can include improve.
296107
+
296108
+ Keeping the set IN the package is what stops a model edit and its answer key
296109
+ drifting apart: they move in one commit, so fixing a measure and forgetting
296110
+ the golden that depended on it stops being possible. It is safe -- measured
296111
+ on a running server, a \`cases.jsonl\` inside a package appears in no model
296112
+ listing, no notebook listing, no package resource, and 404s over HTTP, so an
296113
+ MCP-only answerer has no route to it.
296114
+
296115
+ What it buys differs by target. On a LOCAL Publisher it does not get you free
296116
+ versioning -- \`sourceContentSha\` hashes model paths only, so the set needs
296117
+ its own \`datasetSha\`. On a hosted target that publishes the whole package
296118
+ directory as an IMMUTABLE version, the set rides inside that version and
296119
+ \`targetVersion\` pins model and answer key together; nothing can be edited
296120
+ under a published version, which is what makes it a pin. Check which you have
296121
+ before deciding how much of this you need.
296122
+
296123
+ 2. The server must be up with retrieval tracing on, so a call's ranked results
296124
+ can be recovered afterwards (open-source Publisher: \`PUBLISHER_MCP_TRACE=retrieval\`).
296125
+ Confirm a trace lookup is available (absent means tracing is off).
296126
+ Refuse to start a scored run without it: failures without traces cannot be
296127
+ attributed.
296128
+
296129
+ 3. Health-check: your host's status check until it reports serving, and inspect
296130
+ \`loadErrors\`. A dead database that still answers HTTP is an environment
296131
+ failure, not a model failure. Stop and fix it. Four consecutive
296132
+ environment or no-result attempts means stop the run.
296133
+
296134
+ 4. Load the set: scrape/import as above, or reuse an existing \`evals/<set>/\`.
296135
+ Never keep two live copies of one set; the set directory in the model repo
296136
+ is the single source of truth, versioned by \`datasetVersion\` in
296137
+ \`set.json\`.
296138
+
296139
+ 5. Review goldens before you score. A verified golden with no local artifact
296140
+ stays verified by provenance and is not scorable until you have rows or a
296141
+ scalar to compare (the judge needs both sides). If diagnosis later marks
296142
+ \`BAD-REFERENCE\` or \`AMBIGUOUS-REFERENCE\`, follow
296143
+ \`reference/golden-side-door.md\`. Both are expected in the wild; both
296144
+ are the golden side door below, not improve, and not a sixth step.
296145
+
296146
+ 6. Create \`runs/<runId>/run.json\` with the attribution pins
296147
+ (\`reference/ledger-schema.md\`): mode, dataset version, **the target and the
296148
+ version it served** (a local target pins a commit, so commit or stash first;
296149
+ answering from a dirty tree pins nothing), server version, judge version and
296150
+ rubric sha, answerer model, call budget, trace mode. Freeze those for the
296151
+ whole run. Raising a call budget mid-run moved mean outcomes on an unchanged
296152
+ model.
296153
+
296154
+ 7. Generate every answerer prompt from the stored case in \`cases.jsonl\`.
296155
+ Never retype the question. A truncated retype is indistinguishable from a
296156
+ real question downstream.
296157
+
296158
+ ## Per question
296159
+
296160
+ 1. Health-check again.
296161
+ 2. Spawn a *fresh* blind subagent. Give it only the question text and the
296162
+ Malloy analysis tools. Tell it to follow the \`malloy-analysis\` skill. Do not
296163
+ mention eval, gold, scoring, or this skill.
296164
+ 3. Keep a host-side tool-use log for that subagent (name, input path or
296165
+ command, MCP tool name). Publisher traces see MCP only; a Read of a gold
296166
+ CSV is invisible server-side.
296167
+ 4. \`skill:eval-answer\`: contamination first, then re-execute, then the judge,
296168
+ then events.
296169
+ 5. \`skill:eval-diagnose\` only when this run includes diagnose, only on dev
296170
+ cases, and only after the score event exists.
296171
+ 6. \`skill:eval-improve\` only when this run includes improve, and only for
296172
+ \`owner: model\`. Then run the acceptance check. On accept, checkpoint.
296173
+
296174
+ ## Checkpoint
296175
+
296176
+ A checkpoint is a git commit of the model repository, taken after an acceptance check
296177
+ accepts, so a bad improve direction can be rolled back. It is not a report,
296178
+ and it is not a remote publish.
296179
+
296180
+ 1. Commit the model files AND the set's ledger in one commit; put the label
296181
+ and the closed issue ids in the message.
296182
+ 2. Append the \`checkpoint\` event (\`action: created\`, label, \`modelGitSha\`
296183
+ from the commit you just made, issueIds). The event line itself rides in
296184
+ the next commit; append-only logs trail by one commit and that is fine.
296185
+ 3. Confirm \`git status\` is clean for the model files.
296186
+
296187
+ **Restore**: \`git checkout <sha> -- <model files>\` (or \`git revert\` the
296188
+ checkpoint commits), then reload the package, then append a \`checkpoint\`
296189
+ event with \`action: restored\` and the sha. Readers return to the model that
296190
+ existed before the bad direction.
296191
+
296192
+ Take a checkpoint of the current model *before* the first improve batch if no
296193
+ commit pins it yet. Rolling back by hand is guesswork.
296194
+
296195
+ If reload reports \`mode: reinstalled\`, the package was re-fetched from its
296196
+ install location and may have overwritten the restored files. Prefer in-place
296197
+ / watch-mounted packages for this loop.
296198
+
296199
+ ## Out of scope
296200
+
296201
+ This loop is local. The ledger is files, the checkpoints are git, you are the
296202
+ conductor. Do not:
296203
+
296204
+ - publish the model to a hosted platform as a "true" checkpoint or learning
296205
+ curve
296206
+ - start a Python orchestrator (\`loop.py\`, \`improve_batch.py\`) that runs the
296207
+ five steps end to end unattended. You conduct; the scripts are the steps,
296208
+ not the sequencing
296209
+ - score by string-diffing rows instead of judging them, or reintroduce a
296210
+ scripted row oracle: one that can pass a wrong answer is worse than none
296211
+ - wait for a bigger gold set before the loop can run; dev/holdout on what
296212
+ exists beats waiting
296213
+ - register eval MCP tools or stand up an eval API
296214
+ - encode unsettled goldens into the model
296215
+
296216
+ ## Prime directives
296217
+
296218
+ - The model is the only thing improve edits. No question text, qids, or
296219
+ expected values in any name, doc, or comment.
296220
+ - When the environment misbehaves, stop. Never diagnose a sick system.
296221
+ - When a subagent disagrees with you, probe. Do not win by authority.
296222
+ - When a rule here is wrong, change this file and note it on the run.
296223
+
296224
+ ## Related skills
296225
+
296226
+ - \`skill:eval-answer\`: contamination, judge protocol, events. Its
296227
+ \`reference/ledger-schema.md\` is the file contract; \`skill:eval-judge\` is
296228
+ the judge.
296229
+ - \`skill:eval-diagnose\`: component, owner, issue events. No edit.
296230
+ - \`skill:eval-improve\`: smallest model edit, probe receipts, no self-accept.
296231
+ - The \`malloy-analysis\` skill: what the blind answerer follows. It is installed
296232
+ from the \`analysis\` manifest group, not the \`eval\` group.
296233
+
296234
+ ## Reference files over MCP
296235
+
296236
+ This skill's \`reference/\` files are served as separate prompts, one per file, fetched only when you ask for them. Where the text above says to read \`reference/<name>.md\`, get the prompt named \`eval-loop/<name>\` instead.
296237
+
296238
+ Available: acceptance-check, checking-the-judge, golden-side-door, measurement, running-a-run.`
296239
+ },
296240
+ {
296241
+ name: "eval-loop/acceptance-check",
296242
+ description: "The acceptance check. Reference detail for the eval-loop skill.",
296243
+ body: `<!-- How to decide whether ONE edit stays. Read this before accepting or reverting anything. -->
296244
+
296245
+ # The acceptance check
296246
+
296247
+ ## The acceptance check (inside improve)
296248
+
296249
+ You own the acceptance check. The improver does not accept its own edit.
296250
+
296251
+ **Before any of it: is the answer key still valid?** An edit that changed what
296252
+ an entity means can have moved goldens for cases nobody was working on, and a
296253
+ rerun against a stale key measures nothing -- it reads as a win or a regression
296254
+ with equal confidence and neither is real. \`skill:eval-improve\` Step 4 reports
296255
+ these as \`golden_suspect\` on the candidate; the judge reports its own doubts as
296256
+ \`gold_status\`. **Any unadjudicated one halts the acceptance check.** Settle
296257
+ each through the golden side door -- repair and bump \`goldenRevision\`, or
296258
+ dismiss it explicitly -- and only then re-answer. Do not net a suspect golden
296259
+ against the flip count; an uncertain key is not noise you can average out.
296260
+
296261
+ Cheap and deterministic, every edit:
296262
+
296263
+ 1. A compile check (scope \`file\` for an edit, \`package\` if importers must
296264
+ survive).
296265
+ 2. Save, then reload the package. Confirm it is not serving a stale model.
296266
+ 3. Replay stored final queries from previously-passing cases. They must still
296267
+ execute, and a judge must still call them a match.
296268
+ 4. A *fresh* blind re-answer of the affected question. The fix must be
296269
+ discoverable, not merely possible. The improver writing the query it
296270
+ already knows proves only that the edit exists.
296271
+
296272
+ Acceptance rules (replacing any vague "results improve"):
296273
+
296274
+ - **Per-case, not aggregate.** No previously-passing case may regress: diff the
296275
+ new run's verdicts against the baseline, case by case (\`jq\` over the two
296276
+ \`events.jsonl\` files). \`regressions\` on the acceptance check event must be
296277
+ empty to accept, and the regressed qids go in the checkpoint commit message
296278
+ if you proceed anyway after a human call.
296279
+ - **Confident verdicts only.** \`needs_human\` and null verdicts are neither
296280
+ passes nor failures; the delta is computed without them.
296281
+ - **Both splits.** The acceptance check runs the affected dev cases AND the holdout
296282
+ slice. Diagnose and improve never saw holdout; that is what makes its delta
296283
+ evidence rather than memorization.
296284
+ - **Twice.** An improvement must survive a second independent run with fresh
296285
+ blind answerers before acceptance. A delta that appears once and vanishes
296286
+ on re-run was answerer or judge variance, not a fix.
296287
+ - Documentation / discoverability edits may accept on a deterministic
296288
+ \`get_context\` probe now returning the entity, provided no replay
296289
+ regresses. Measure, join, or definition edits need the full rules above,
296290
+ including the flip-count bar in Measurement, which means enough affected
296291
+ cases to clear it.
296292
+ - Independent deterministic justification (a probed-wrong definition
296293
+ corrected) may accept without a measured win. Record that as the acceptance check
296294
+ \`reason\`.
296295
+
296296
+ Write the \`acceptance_check\` event (decision, class, baseline and final run ids,
296297
+ regressions, holdout delta, reason) BEFORE any commit, so a rejected
296298
+ direction leaves a record. On reject: revert the files (\`git checkout --\`
296299
+ or \`git restore\`) and reload. On accept: \`issue_status: fixed\` for what the
296300
+ edit actually closed, **then checkpoint**.`
296301
+ },
296302
+ {
296303
+ name: "eval-loop/checking-the-judge",
296304
+ description: "Checking the judge. Reference detail for the eval-loop skill.",
296305
+ body: `<!-- The judge measures the model; this measures the judge. Read after any change to judge doctrine or its inputs. -->
296306
+
296307
+ # Checking the judge
296308
+
296309
+ ## Nothing else checks the judge
296310
+
296311
+ The A/A band measures whether the judge is *repeatable*. It says nothing about
296312
+ whether it is *right* -- a judge answering \`no_match\` every time posts a perfect
296313
+ band. Those come apart in practice, and when they do the loop keeps running and
296314
+ every number it emits is wrong in the same direction.
296315
+
296316
+ So keep a small file of frozen predictions pinned to verdicts a human settled,
296317
+ and re-run them after any edit to the judge prompt, a rubric, or what the judge
296318
+ is given (\`scripts/check_judge.py\`, \`judge-regressions.jsonl\` in the set). Seed it
296319
+ from the cases an A/A pair disagreed on: those are the contested ones, so they
296320
+ are where a change will show first.
296321
+
296322
+ Two things about its shape:
296323
+
296324
+ - **The unit is a prediction, not a question.** One question earns different
296325
+ verdicts for different answers, legitimately. Key the fixture on the answer.
296326
+ - **Judge through the same code path a run uses.** A reimplementation inside the
296327
+ checker can pass while the thing it stands for is broken.
296328
+
296329
+ A fixture that has never failed is not yet known to be a test. Break a rubric on
296330
+ purpose once and confirm the right entry fails.
296331
+
296332
+ When a fixture fails, rule out judge nondeterminism (\`--repeat\`) before you
296333
+ believe it. Then either you moved a verdict you did not mean to, or the fixture
296334
+ was wrong -- re-settle it and record why. Deleting it throws away the only case
296335
+ you had evidence about.`
296336
+ },
296337
+ {
296338
+ name: "eval-loop/golden-side-door",
296339
+ description: "The golden side door. Reference detail for the eval-loop skill.",
296340
+ body: `<!-- Everything about a golden that is wrong, doubted, or out of step with the model. NOT a sixth step, and never improve. -->
296341
+
296342
+ # The golden side door
296343
+
296344
+ ## Golden side door (not a sixth step)
296345
+
296346
+ Bad and ambiguous goldens show up immediately. That is not improve. A
296347
+ checkpoint that mixes model edits and silent golden rewrites is useless for
296348
+ rollback. Keep hold and repair here, outside the five steps.
296349
+
296350
+ ## Repair a bad golden
296351
+
296352
+ This is **your** job as conductor, after \`eval-diagnose\` writes
296353
+ \`BAD-REFERENCE\`. It is not the answerer's job, and it is not a reason to
296354
+ change the model.
296355
+
296356
+ Diagnosis is not the only way one arrives. The judge also reports a
296357
+ \`gold_status\` on every score (\`skill:eval-judge\`), and a
296358
+ \`suspect\` or \`verified_wrong\` comes through this same door -- earlier, because it
296359
+ lands during scoring rather than after. Treat it as a \`BAD-REFERENCE\` with the
296360
+ judge's \`gold_note\` as its evidence. Adjudicate it **before** improve runs: a
296361
+ doubted key sends a modelling agent to fix a model that is already right, which
296362
+ is the most expensive wrong turn this loop can take.
296363
+
296364
+ The judge scored against the golden as written even where it said \`suspect\`, so
296365
+ its verdict is still the verdict. Do not re-open a case merely because the flag
296366
+ is set; open it because you looked and agreed.
296367
+
296368
+ 1. **Replay, yourself.** Take the stored \`final_query\` (or a query you can
296369
+ justify from the model) and run it with \`execute_query\`. Write the
296370
+ rows to a gold artifact under \`evals/<set>/\` (never under the served
296371
+ package tree). If you cannot produce a trusted key, follow **Hold an
296372
+ ambiguous golden** (or mark the golden \`invalid\` if the question itself is
296373
+ unusable). Do not invent a number.
296374
+ 2. **Patch the case** in \`cases.jsonl\`: new \`golden\` (status, kind, value or
296375
+ path, \`canonicalQuery\`, \`verifiedBy: replay\`) and \`goldenRevision\`
296376
+ incremented. Do not edit any old \`score\` event.
296377
+ 3. **Bump \`datasetVersion\`** in \`set.json\`, and commit the ledger change so
296378
+ the repair is attributable.
296379
+ 4. **Close the issue as repaired, not as a model fix**: \`issue_status: fixed\`
296380
+ with a note that the *golden* changed.
296381
+ 5. **Open a new run** whose \`run.json\` records the new \`datasetVersion\`. It
296382
+ is not comparable to runs on the old version without saying so.
296383
+ 6. **Re-score stored queries first**: \`skill:eval-answer\` without a new
296384
+ answerer (saved predictions, fresh judge, new \`golden_revision\` stamps).
296385
+ 7. **Re-answer only if you still need a blind look** (discoverability, or the
296386
+ stored query was itself the thing under test).
296387
+
296388
+ Never mix old-golden and new-golden scores in one aggregate. A before/after
296389
+ that crosses a golden bump is a rebase, not a model delta.
296390
+
296391
+ ## A model fix can invalidate a golden, and nothing will tell you
296392
+
296393
+ A rubric that explains a trap usually has to quote the model -- "this measure
296394
+ counts line items despite its name, so it yields the trap value". That sentence
296395
+ is a claim about the model, and it is false the moment the model is fixed. The
296396
+ judge keeps enforcing it and starts failing correct answers.
296397
+
296398
+ A truth-package check cannot catch this, structurally. It re-derives values from
296399
+ sources that are independent of the model **on purpose**, so a rubric can
296400
+ describe a model that no longer exists while every value still re-derives green.
296401
+
296402
+ Worse, a model fix can move a golden's *value* without touching the data. If a
296403
+ dimension is defined in terms of the measure you fixed, the concept it names now
296404
+ resolves to a different population -- same dimension, same question, different
296405
+ correct answer -- while a canonical truth query still returns the old number
296406
+ because it encoded the old definition.
296407
+
296408
+ So: **after any model edit, re-read the rubrics of every case that names an
296409
+ entity you touched.** \`verify_goldens.py\` audits the mechanical part -- it parses
296410
+ \`X is <expr>\` out of the model and flags any rubric asserting a different
296411
+ definition -- but only for definitions it can parse. Prose claims about grain,
296412
+ population, or convention are still yours to check.
296413
+
296414
+ When one turns up it is \`BAD-REFERENCE\`, and it goes through this side door.
296415
+ Never let it reach improve: the model is right, and an edit would be damage.
296416
+
296417
+ ## A golden must match the state the model is in
296418
+
296419
+ A case whose golden holds a value asserts that the value is obtainable. If the
296420
+ model has no trace of the concept, that assertion is false, and the case is now
296421
+ asking two questions at once: "did the answer contain the golden" (no) and
296422
+ "should the answerer have complied" (no). Both readings are defensible, so the
296423
+ verdict stops being a measurement.
296424
+
296425
+ Measured, holding the answer, the model and the rubric fixed and varying only
296426
+ how the case was authored:
296427
+
296428
+ | the case says | verdicts over four samples |
296429
+ |---|---|
296430
+ | golden holds three counts, model defines no such concept | \`match\` / \`no_match\` / \`match\` / \`near_match\` |
296431
+ | \`golden.kind: unanswerable\`, pass is a refusal that names what is missing | \`match\` x4 |
296432
+
296433
+ The judge is not being unreliable in the first row. It is being asked a question
296434
+ with two right answers.
296435
+
296436
+ So a coverage case has two states and needs a golden for each:
296437
+
296438
+ 1. **Before the model defines the concept.** \`coverage: absent\`,
296439
+ \`golden.kind: unanswerable\`. The pass is a refusal that NAMES what is
296440
+ missing; inventing boundaries and reporting them as the company's is
296441
+ \`no_match\`. This is the state that measures whether the model documents its
296442
+ conventions.
296443
+ 2. **After improve adds it.** Bump \`goldenRevision\`, replace the golden with the
296444
+ real value, bump \`datasetVersion\`. A refusal is now a failure, and the run
296445
+ measures whether the new entity is discoverable.
296446
+
296447
+ Never one case straddling both. The straddle is what produces an oscillating
296448
+ verdict, and no amount of rubric wording fixes it -- four prompt edits were
296449
+ tried against exactly this case and none of them did.
296450
+
296451
+ **This is the mirror of "A model fix can invalidate a golden".** That section
296452
+ warns that adding a definition can move a golden nobody was working on. This one
296453
+ warns of the same seam from the other side: a golden written for a model that
296454
+ does not exist yet is invalid until the model catches up. Both are golden side
296455
+ door work, and neither is improve.
296456
+
296457
+ ## Hold an ambiguous golden
296458
+
296459
+ Use this when the current key is unusable as a score *and* you cannot justify
296460
+ exactly one replacement (two honest replays disagree; a window or tie is
296461
+ unspecified; later samples might confirm a convention).
296462
+
296463
+ 1. **Do not invent a key.** Leave the old artifact on the case for
296464
+ provenance.
296465
+ 2. **Patch the case**: \`golden.status: ambiguous\` with a \`reason\` naming the
296466
+ defect and the competing replacements (not a new number). Increment
296467
+ \`goldenRevision\`.
296468
+ 3. **Do not score** this case until a later sample confirms a convention or a
296469
+ human picks a replacement. Its attempts get \`verdict: null,
296470
+ reason: golden_ambiguous\`.
296471
+ 4. **\`issue_status: deferred\`**, not \`fixed\`. Revisit when another case in
296472
+ the same neighborhood confirms a convention.
296473
+ 5. Old \`score\` events stay. They keep the previous \`golden_revision\` and must
296474
+ not enter an aggregate that claims the model failed.
296475
+
296476
+ If later evidence makes one replacement obvious, then Repair a bad golden.`
296477
+ },
296478
+ {
296479
+ name: "eval-loop/measurement",
296480
+ description: "Measurement. Reference detail for the eval-loop skill.",
296481
+ body: `<!-- Sampling, the flip-count bar, the A/A noise band, and targeted fixes. Read this before quoting any number. -->
296482
+
296483
+ # Measurement
296484
+
296485
+ ## Measurement
296486
+
296487
+ **Sample each case once. Spend the budget on more and more varied cases
296488
+ instead.** Repeats past the first buy very little: variance decompositions of
296489
+ LLM evaluation put the reduction from extra repeats at a small fraction of
296490
+ what extra items buy, and a set of five cases run three times cannot support
296491
+ a claim that fifteen distinct cases can. If a case is genuinely borderline,
296492
+ re-run that case, not the whole set.
296493
+
296494
+ Because a single sample cannot carry a mean, do not report before/after as a
296495
+ score delta. **Count the cases whose verdict changed** between the baseline
296496
+ and the post-edit run, discard the unchanged ones, and read the result off
296497
+ this table:
296498
+
296499
+ | Cases that got worse | Cases that must get better to accept |
296500
+ |---|---|
296501
+ | 0 | 5 |
296502
+ | 1 | 7 |
296503
+ | 2 | 9 |
296504
+ | 3 | 10 |
296505
+
296506
+ Below that bar the change is **unresolved**, not an improvement, and saying
296507
+ so is the honest report. Note the consequence before you scope a run: a set
296508
+ of six cases can essentially never clear this bar, so a set that small can
296509
+ measure a baseline and diagnose failures but cannot defend an edit.
296510
+
296511
+ ## Calibrate the bar before you trust it
296512
+
296513
+ This table was asserted, not measured, and the number it needs is a property of
296514
+ your harness and your set -- not of this skill. Measure it with an **A/A run**:
296515
+ the same model, same config, same set, twice, compared with
296516
+ \`scripts/flip_table.py\`. Every flip it reports is noise by construction, since
296517
+ nothing changed. Record the result with the set, in \`CALIBRATION.md\`, and cite
296518
+ that file when you quote a band.
296519
+
296520
+ Re-measure whenever the model, judge, or set changes. This is not a formality:
296521
+ observed bands have moved by a factor of three across a fortnight of ordinary
296522
+ work, so a band carried over from a previous configuration is a number with no
296523
+ claim on the present one.
296524
+
296525
+ One A/A is one sample of the flip count, not a distribution. It can show a bar
296526
+ is too low; it cannot show one is high enough. Treat any measured band as a
296527
+ floor.
296528
+
296529
+ Two consequences worth separating:
296530
+
296531
+ - **For acceptance**, the band is the threshold untargeted flips must sit under.
296532
+ - **For diagnosis**, it is a warning that a single run's failure list is partly
296533
+ luck. Pick what to fix from the failures that fail in **both** A/A runs.
296534
+ Ranking a backlog by one run's clusters partly ranks which cases were unlucky
296535
+ that afternoon.
296536
+
296537
+ When you inspect the flips, attribute them before you accept them as
296538
+ irreducible. A band dominated by the **judge** re-reading an ambiguous rubric is
296539
+ not answerer noise, and it is not a floor you have to live under: sharpening
296540
+ those rubrics buys more measurement power than any change to the answerer.
296541
+
296542
+ An A/A is not a repeat in the sense the sampling rule forbids. It is a one-off
296543
+ calibration of the instrument, and the loop's whole acceptance rule rests on
296544
+ the constant it produces.
296545
+
296546
+ ## A targeted fix needs a targeted test
296547
+
296548
+ The flip-count table is the right instrument for a broad change and the wrong
296549
+ one for a narrow fix. A fix that repairs three cases on a 49-case set moves the
296550
+ total by three -- inside the noise band an A/A already produces -- so a
296551
+ mechanically-verified repair reports as no effect and gets abandoned.
296552
+
296553
+ This is not a hypothetical failure mode: a mechanically verified repair, where
296554
+ each fixed case now matches its golden exactly, can read as "no effect" on both
296555
+ of two set-total comparisons. Worked examples are in the set's \`CALIBRATION.md\`.
296556
+
296557
+ So for a narrow fix use \`scripts/flip_table.py --targets --noise-band\`:
296558
+
296559
+ 1. **Name the cases before the run.** Pick them from the stable failures of the
296560
+ A/A, never from a single run. Choosing them afterwards is choosing the answer.
296561
+ 2. Accept on the targeted cases: they were failing, they now pass, and none of
296562
+ them broke.
296563
+ 3. Separately require the untargeted flips to sit **at or below the A/A band**.
296564
+ That is what rules out a fix that trades one set of cases for another --
296565
+ above the band, investigate before accepting, however good the targets look.
296566
+ 4. **Run the post-edit arm twice and pass both** (\`--b --b2\`). The band counts
296567
+ flips; it never asks which cases flipped, and that is the hole. Noise
296568
+ scatters, so an untargeted case that breaks in *both* post arms is a real
296569
+ regression however small the count is.
296570
+
296571
+ Report both. A targeted win with untargeted flips above the band is not a win,
296572
+ and a set-total that moved by less than the band is not evidence of anything
296573
+ either way.
296574
+
296575
+ Step 4 exists because the band alone has accepted a real regression: an edit
296576
+ whose untargeted flip count sat inside the band, but where the same untargeted
296577
+ case broke in every post arm. One arm cannot tell that from a coin toss.
296578
+
296579
+ The reason to expect this, rather than treat it as bad luck: **a correct new
296580
+ entity is not a safe one.** Adding a measure changes what agents reach for on
296581
+ questions nobody was thinking about, so a well-named addition can pull a
296582
+ neighbouring question onto the wrong denominator. That makes the untargeted
296583
+ half of the acceptance check the half that matters, and it needs two arms to be
296584
+ readable at all.
296585
+
296586
+ The one retrieval number this loop reports is **per-question entity recall**:
296587
+ of the entities each golden answer depends on, how many did the agent's own
296588
+ \`get_context\` calls deliver (\`skill:eval-answer\`, \`scripts/score_retrieval.py\`;
296589
+ delivered means returned as a ranked entity, under a sibling source, or named
296590
+ in a returned source's documentation). It is measured on the agent's real
296591
+ search text against real questions, so it needs no hand-written terms. Read
296592
+ it within an arm, to attribute a failure; it moves with the answerer, so a
296593
+ cross-arm comparison of retrieval *itself* is not this loop's job -- that is
296594
+ the engine-side \`eval-retrieval\` skill, which ships to no customer.
296595
+ Coverage (can the model answer this at all) is a property of the model and its
296596
+ data and is never reported under a retrieval heading.
296597
+
296598
+ If the contaminated fraction of attempts exceeds 0.1 (or any contamination,
296599
+ on a run smaller than 10), the run is a harness failure. Do not publish a
296600
+ model score.`
296601
+ },
296602
+ {
296603
+ name: "eval-loop/running-a-run",
296604
+ description: "Running a run, concretely. Reference detail for the eval-loop skill.",
296605
+ body: `<!-- The worked command sequence. Read it when you are about to run one. -->
296606
+
296607
+ # Running a run, concretely
296608
+
296609
+ ## Running one, concretely
296610
+
296611
+ \`scripts/run_baseline.py\` does steps 3 and 7 and the whole of **Per question**:
296612
+ one fresh answerer per case with only the Publisher MCP tools, a contamination
296613
+ check, a judge, and a conformant \`events.jsonl\`.
296614
+
296615
+ \`\`\`bash
296616
+ # 1. serve the model under test -- in its own session, so the shell's exit
296617
+ # cannot take it down, and returning only once it answers a query
296618
+ python3 skills/eval-loop/scripts/serve.py --publisher-dir <publisher>/packages/server \\
296619
+ --server-root <root> --port 4811 --mcp-port 4040 --trace-retrieval \\
296620
+ [--allow-proxy] # required for a \`publisher\`-type (proxied) connection
296621
+ # a second server for the TRUTH package, on other ports, that the answerer
296622
+ # has no route to:
296623
+ python3 skills/eval-loop/scripts/serve.py --publisher-dir <publisher>/packages/server \\
296624
+ --server-root <truthroot> --port 4881 --mcp-port 4882 [--allow-proxy]
296625
+
296626
+ # 2. smoke one case first ($0.13), then the arm. Goldens are re-derived from
296627
+ # the truth server before either starts; a drifted set refuses to run.
296628
+ python3 skills/eval-loop/scripts/run_baseline.py \\
296629
+ --set <repo>/evals/ecommerce --out results/smoke --only <qid> --no-judge \\
296630
+ --truth-publisher http://localhost:4881
296631
+ python3 skills/eval-loop/scripts/run_baseline.py \\
296632
+ --set <repo>/evals/ecommerce --out results/<arm> \\
296633
+ --parallel 4 --truth-publisher http://localhost:4881
296634
+ # the run names itself <set>-<phase>-<nn> (ecommerce-baseline-01, then -02
296635
+ # for the second arm of the A/A). Pass --label only for a run that needs a
296636
+ # human name; hand-typed arm names stop being readable within an afternoon.
296637
+
296638
+ # 3. compare two arms, or two runs of one arm
296639
+ python3 skills/eval-loop/scripts/flip_table.py --a results/<a> --b results/<b>
296640
+
296641
+ # 4. FIRST: any golden the judge did not believe. \`jq .doubtedGoldens
296642
+ # results/<arm>/run.json\` -- non-empty means settle those through the golden
296643
+ # side door before diagnosing, or you send a modelling agent at a model that
296644
+ # is already right.
296645
+ python3 skills/eval-diagnose/scripts/diagnose.py \\
296646
+ --run results/<arm> --set <repo>/evals/ecommerce --model-dir <package>
296647
+ # (cluster_failures.py gives a free mechanical first look, as
296648
+ # clusters-mechanical.jsonl; it groups by retrieval outcome and is not a
296649
+ # diagnosis)
296650
+
296651
+ # 5. build the browsable package
296652
+ python3 skills/eval-loop/scripts/build_run_package.py \\
296653
+ --run results/<a> --run results/<b> --set <repo>/evals/ecommerce --out <pkg>
296654
+ \`\`\`
296655
+
296656
+ Order of magnitude for planning, **calibrated on ecommerce over local duckdb**:
296657
+ a Sonnet arm over a few dozen cases costs single-digit dollars and finishes in
296658
+ minutes, at roughly a dime and a handful of turns per case. A proxied warehouse
296659
+ is a different regime: the VideoAmp set ran at $0.33 per case on Sonnet and
296660
+ $0.57–0.71 on Opus, ~100 s per case, driven by warehouse latency and query
296661
+ errors -- budget 4x when the data is not local. Budget **five** such arms for a
296662
+ defensible claim -- a baseline, two for the A/A, and two post-edit -- plus the
296663
+ diagnose and improve agents, which are far cheaper per case but use a larger
296664
+ model. Measured per-arm figures for a given set belong in that set's
296665
+ \`CALIBRATION.md\`.
296666
+
296667
+ \`--rebuild\` re-derives the ledger from saved transcripts without calling a model,
296668
+ and \`--rebuild --rejudge\` re-scores existing answers in place. \`--from <run>
296669
+ --out <new>\` does the same into a NEW run directory -- the answers copied, the
296670
+ judge fresh, the old verdicts untouched -- which is what a golden repair or a
296671
+ rubric change calls for. Use them after a scoring or schema change; re-running
296672
+ the answerers would confound the change you are measuring with fresh answerer
296673
+ variance.
296674
+
296675
+ The scripts import each other by path (\`ledger\`, \`mcp_payload\`,
296676
+ \`score_retrieval\` live in \`eval-answer/scripts\`; the loop scripts insert that
296677
+ path). Run them **in place** from the skills checkout; a copy patched elsewhere
296678
+ chases \`ModuleNotFoundError\` three times.
296679
+
296680
+ Two failure modes worth pre-empting, because both produce a clean-looking run:
296681
+
296682
+ - **Pre-approve the tools.** A headless answerer that has to ask permission for
296683
+ \`malloy_getContext\` stalls until the timeout and lands as a harness error.
296684
+ - **Check the served revision is the one you edited.** Publisher serves a
296685
+ snapshot copy, so a model fix can be absent from the run that is supposed to
296686
+ measure it. Query the changed measure once before spending an arm on it.`
296687
+ },
294775
296688
  {
294776
296689
  name: "malloy",
294777
296690
  description: 'Index of all Malloy skills. Use when user asks "malloy help", "what malloy skills are available", "how do I use malloy", or needs guidance on which Malloy skill to use.',
@@ -294786,6 +296699,8 @@ Say "model my data" and the agent will orchestrate the full modeling workflow au
294786
296699
 
294787
296700
  Every skill in this deployment, by what it is for. Start at a driver; it routes to the rest.
294788
296701
 
296702
+ This table is a catalogue of what exists, not of what is loaded. A host that installs one group takes that group's skills alone: \`analysis\`, \`modeling\`, or \`eval\`. A row naming a skill from a group you did not install says that the skill exists. It is not an instruction to load it, and it is written as a plain name rather than a \`skill:\` reference to say so.
296703
+
294789
296704
  **Start here**
294790
296705
 
294791
296706
  | Skill | Use when... |
@@ -294815,9 +296730,18 @@ Every skill in this deployment, by what it is for. Start at a driver; it routes
294815
296730
  | \`skill:malloy-notebooks\` | Building Malloy notebooks (.malloynb) |
294816
296731
  | \`skill:malloy-analysis-report\` | Combining validated queries into a notebook report or dashboard |
294817
296732
  | \`skill:malloy-analysis-pitfalls\` | Checking a query and its results before presenting an answer |
294818
- | \`skill:malloy-notebook-chat\` | The chat is bound to a notebook or saved report; answer from its cells |
296733
+ | \`malloy-notebook-chat\` | The chat is bound to a notebook or saved report; answer from its cells. Ships in \`analysis\`. |
294819
296734
  | \`skill:malloy-phrase-detection\` | Turning a plain-English question into search targets for the context tool |
294820
296735
 
296736
+ **Evaluating a model** (driven by \`eval-loop\`). These ship in the \`eval\` group, which neither \`analysis\` nor \`modeling\` includes.
296737
+
296738
+ | Skill | Use when... |
296739
+ |-------|-------------|
296740
+ | \`eval-loop\` | Running the loop: baseline, noise band, diagnose, one edit, gate, checkpoint |
296741
+ | \`eval-answer\` | Scoring one answer against a verified golden, and what retrieval delivered |
296742
+ | \`eval-diagnose\` | Deciding why a case failed and which artifact owns the fix |
296743
+ | \`eval-improve\` | The smallest model edit for a diagnosed cluster, with probe receipts |
296744
+
294821
296745
  **Writing correct Malloy** (read before writing, not after failing)
294822
296746
 
294823
296747
  | Skill | Use when... |