npm - @sanity/ailf - Versions diffs - 0.1.0 → 0.1.2 - Mend

@sanity/ailf 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (125) hide show

package/dist/_vendor/ailf-core/examples/index.d.ts +16 -12
package/dist/_vendor/ailf-core/examples/index.js +19 -12
package/dist/_vendor/ailf-core/ports/context.d.ts +4 -0
package/dist/adapters/task-sources/repo-schemas.d.ts +12 -2
package/dist/adapters/task-sources/repo-schemas.js +28 -2
package/dist/cli.js +0 -0
package/dist/commands/init.js +17 -5
package/dist/commands/pipeline-action.js +44 -6
package/dist/commands/publish.js +2 -1
package/dist/commands/validate-tasks.js +4 -1
package/dist/composition-root.js +9 -5
package/dist/orchestration/build-app-context.js +2 -0
package/package.json +1 -1
package/dist/commands/update-quality-scores.d.ts +0 -5
package/dist/commands/update-quality-scores.js +0 -20
package/dist/lib/agent-behavior-report.d.ts +0 -8
package/dist/lib/agent-behavior-report.js +0 -185
package/dist/lib/baseline.d.ts +0 -19
package/dist/lib/baseline.js +0 -153
package/dist/lib/calculate-scores.d.ts +0 -23
package/dist/lib/calculate-scores.js +0 -42
package/dist/lib/compare.d.ts +0 -18
package/dist/lib/compare.js +0 -170
package/dist/lib/coverage-audit.d.ts +0 -4
package/dist/lib/coverage-audit.js +0 -42
package/dist/lib/discovery-report.d.ts +0 -13
package/dist/lib/discovery-report.js +0 -57
package/dist/lib/fetch-docs.d.ts +0 -30
package/dist/lib/fetch-docs.js +0 -171
package/dist/lib/generate-configs.d.ts +0 -25
package/dist/lib/generate-configs.js +0 -42
package/dist/lib/grader-api.d.ts +0 -21
package/dist/lib/grader-api.js +0 -34
package/dist/lib/grader-compare.d.ts +0 -19
package/dist/lib/grader-compare.js +0 -91
package/dist/lib/grader-consistency.d.ts +0 -27
package/dist/lib/grader-consistency.js +0 -79
package/dist/lib/grader-sensitivity.d.ts +0 -19
package/dist/lib/grader-sensitivity.js +0 -75
package/dist/lib/grader-validate.d.ts +0 -19
package/dist/lib/grader-validate.js +0 -78
package/dist/lib/measure-retrieval.d.ts +0 -14
package/dist/lib/measure-retrieval.js +0 -71
package/dist/lib/pr-comment.d.ts +0 -16
package/dist/lib/pr-comment.js +0 -28
package/dist/lib/readiness-report.d.ts +0 -13
package/dist/lib/readiness-report.js +0 -108
package/dist/lib/webhook-server.d.ts +0 -11
package/dist/lib/webhook-server.js +0 -24
package/dist/lib/weekly-digest.d.ts +0 -24
package/dist/lib/weekly-digest.js +0 -148
package/dist/orchestration/env-bridge.d.ts +0 -21
package/dist/orchestration/env-bridge.js +0 -66
package/dist/orchestration/steps/fetch-docs-shell.d.ts +0 -17
package/dist/orchestration/steps/fetch-docs-shell.js +0 -30
package/dist/pipeline/steps/calculate-scores-step.d.ts +0 -11
package/dist/pipeline/steps/calculate-scores-step.js +0 -89
package/dist/pipeline/steps/compare-step.d.ts +0 -18
package/dist/pipeline/steps/compare-step.js +0 -90
package/dist/pipeline/steps/eval-step.d.ts +0 -53
package/dist/pipeline/steps/eval-step.js +0 -347
package/dist/pipeline/steps/fetch-docs-step.d.ts +0 -11
package/dist/pipeline/steps/fetch-docs-step.js +0 -84
package/dist/pipeline/steps/generate-configs-step.d.ts +0 -11
package/dist/pipeline/steps/generate-configs-step.js +0 -98
package/dist/pipeline/steps/grader-consistency-step.d.ts +0 -21
package/dist/pipeline/steps/grader-consistency-step.js +0 -74
package/dist/pipeline/steps/publish-report-step.d.ts +0 -57
package/dist/pipeline/steps/publish-report-step.js +0 -243
package/dist/pipeline/steps/report-step.d.ts +0 -13
package/dist/pipeline/steps/report-step.js +0 -56
package/dist/pipeline/steps/update-scores-step.d.ts +0 -11
package/dist/pipeline/steps/update-scores-step.js +0 -42
package/dist/scripts/agent-behavior-report.d.ts +0 -19
package/dist/scripts/agent-behavior-report.js +0 -315
package/dist/scripts/baseline.d.ts +0 -43
package/dist/scripts/baseline.js +0 -267
package/dist/scripts/calculate-scores.d.ts +0 -166
package/dist/scripts/calculate-scores.js +0 -1296
package/dist/scripts/compare.d.ts +0 -22
package/dist/scripts/compare.js +0 -334
package/dist/scripts/coverage-audit.d.ts +0 -44
package/dist/scripts/coverage-audit.js +0 -209
package/dist/scripts/debug-eval.d.ts +0 -19
package/dist/scripts/debug-eval.js +0 -73
package/dist/scripts/discovery-report.d.ts +0 -58
package/dist/scripts/discovery-report.js +0 -250
package/dist/scripts/fetch-docs.d.ts +0 -35
package/dist/scripts/fetch-docs.js +0 -472
package/dist/scripts/generate-configs.d.ts +0 -66
package/dist/scripts/generate-configs.js +0 -459
package/dist/scripts/grader-api.d.ts +0 -27
package/dist/scripts/grader-api.js +0 -206
package/dist/scripts/grader-compare.d.ts +0 -22
package/dist/scripts/grader-compare.js +0 -368
package/dist/scripts/grader-consistency.d.ts +0 -20
package/dist/scripts/grader-consistency.js +0 -313
package/dist/scripts/grader-sensitivity.d.ts +0 -22
package/dist/scripts/grader-sensitivity.js +0 -354
package/dist/scripts/grader-validate.d.ts +0 -19
package/dist/scripts/grader-validate.js +0 -267
package/dist/scripts/measure-retrieval.d.ts +0 -10
package/dist/scripts/measure-retrieval.js +0 -145
package/dist/scripts/pipeline.d.ts +0 -76
package/dist/scripts/pipeline.js +0 -1031
package/dist/scripts/pr-comment.d.ts +0 -10
package/dist/scripts/pr-comment.js +0 -510
package/dist/scripts/readiness-report.d.ts +0 -88
package/dist/scripts/readiness-report.js +0 -342
package/dist/scripts/update-quality-scores.d.ts +0 -15
package/dist/scripts/update-quality-scores.js +0 -184
package/dist/scripts/validate.d.ts +0 -13
package/dist/scripts/validate.js +0 -79
package/dist/scripts/webhook-server.d.ts +0 -26
package/dist/scripts/webhook-server.js +0 -147
package/dist/scripts/weekly-digest.d.ts +0 -24
package/dist/scripts/weekly-digest.js +0 -144
package/dist/sinks/format-slack.d.ts +0 -64
package/dist/sinks/format-slack.js +0 -306
package/dist/sinks/slack-sink.d.ts +0 -27
package/dist/sinks/slack-sink.js +0 -78
package/dist/sinks/webhook-sink.d.ts +0 -19
package/dist/sinks/webhook-sink.js +0 -50
package/tasks/.expanded.agentic.yaml +0 -51
package/tasks/.expanded.yaml +0 -66

package/dist/_vendor/ailf-core/examples/index.d.ts CHANGED Viewed

@@ -90,9 +90,9 @@ export declare const thresholdYaml = "# Example quality threshold configuration.
 /** Parsed ailf-config example data (JSON-safe) */
 export declare const ailfConfigData: {
     readonly source: {
-        readonly projectId: "your-project-id";
-        readonly dataset: "production";
-        readonly baseUrl: "https://your-site.example.com/docs";
+        readonly projectId: "3do82whm";
+        readonly dataset: "next";
+        readonly baseUrl: "https://www.sanity.io/docs";
     };
     readonly triggers: {
         readonly pr: {
@@ -110,20 +110,21 @@ export declare const ailfConfigData: {
     };
 };
 /** Raw YAML string for ailf-config example (preserves comments) */
-export declare const ailfConfigYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# .ailf/config.yaml \u2014 AI Literacy Framework project configuration\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Documentation source \u2014 where to fetch content for evaluation.\n#\n# projectId \u2014 your Sanity project ID (find it in sanity.io/manage)\n# dataset   \u2014 the dataset to query (e.g., \"production\", \"staging\")\n# baseUrl   \u2014 the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"your-project-id\"\n  dataset: production\n  baseUrl: \"https://your-site.example.com/docs\"\n\n# Trigger configuration \u2014 when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only \u2014 check that task YAML parses correctly (fast, no LLM calls)\n#   eval          \u2014 run the full evaluation pipeline\n#\n# paths \u2014 only trigger when files matching these globs change\n# blocking \u2014 if true, a failing eval blocks the PR merge\n# notify \u2014 if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
+export declare const ailfConfigYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# .ailf/config.yaml \u2014 AI Literacy Framework project configuration\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Evaluations are submitted to the AILF API (ailf-api.sanity.build).\n# The API handles LLM calls, doc fetching, grading, and report\n# publishing. Your repo only needs one secret: AILF_API_KEY.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Documentation source \u2014 which docs are being evaluated.\n#\n# This tells the pipeline which Sanity project and dataset contain\n# the documentation under test. For most users, this is Sanity's own\n# docs project.\n#\n# projectId \u2014 Sanity project ID (find yours at sanity.io/manage)\n# dataset   \u2014 the dataset to query (e.g., \"production\", \"next\")\n# baseUrl   \u2014 the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"3do82whm\"\n  dataset: next\n  baseUrl: \"https://www.sanity.io/docs\"\n\n# Trigger configuration \u2014 when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only \u2014 check that task YAML parses correctly (fast, no LLM calls)\n#   eval          \u2014 run the full evaluation pipeline\n#\n# paths \u2014 only trigger when files matching these globs change\n# blocking \u2014 if true, a failing eval blocks the PR merge\n# notify \u2014 if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
 /** Parsed task data for example-groq-blog-listing (JSON-safe) */
 export declare const exampleGroqBlogListingData: readonly [{
     readonly id: "example-groq-blog-listing";
     readonly description: "Example — Blog listing with GROQ queries";
-    readonly canonical_docs: readonly [{
+    readonly featureArea: "groq";
+    readonly canonicalDocs: readonly [{
         readonly slug: "groq-introduction";
         readonly reason: "Core GROQ syntax and query language reference";
     }, {
         readonly slug: "how-queries-work";
         readonly reason: "Query execution model and best practices";
     }];
-    readonly doc_coverage: true;
-    readonly reference_solution: "canonical/example-groq-blog-listing.ts";
+    readonly docCoverage: true;
+    readonly referenceSolution: "canonical/example-groq-blog-listing.ts";
     readonly vars: {
         readonly task: "Create a Next.js page component that lists blog posts from Sanity\nusing GROQ. The page should display the title, slug, and published\ndate for each post, sorted by most recent first. Use the Sanity\nclient to fetch data.\n";
         readonly docs: "";
@@ -143,17 +144,18 @@ export declare const exampleGroqBlogListingData: readonly [{
     };
 }];
 /** Raw YAML string for example-groq-blog-listing (preserves comments) */
-export declare const exampleGroqBlogListingYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Blog listing with GROQ queries\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# To disable this task without deleting the file, set:\n#   baseline:\n#     enabled: false\n#\n# Full field reference:\n#   https://github.com/sanity-io/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Unique identifier \u2014 lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example \u2014 Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  # featureArea is inferred from the filename by default, but you can\n  # set it explicitly here.\n  # featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   \u2014 the article's URL slug in your docs site\n  #   reason \u2014 why this doc is relevant (helps with auditing)\n  canonical_docs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  doc_coverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  reference_solution: canonical/example-groq-blog-listing.ts\n\n  # vars.task \u2014 the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs \u2014 leave empty (\"\"). The pipeline fills this in:\n  #   \u2022 Gold variant: injected with canonical doc content\n  #   \u2022 Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions \u2014 how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   \u2014 did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  \u2014 is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled \u2014 set to false to skip this task entirely\n  #   rubric  \u2014 \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
+export declare const exampleGroqBlogListingYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Blog listing with GROQ queries\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# To disable this task without deleting the file, set:\n#   baseline:\n#     enabled: false\n#\n# Full field reference:\n#   https://github.com/sanity-io/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Unique identifier \u2014 lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example \u2014 Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   \u2014 the article's URL slug in your docs site\n  #   reason \u2014 why this doc is relevant (helps with auditing)\n  canonicalDocs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  docCoverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  referenceSolution: canonical/example-groq-blog-listing.ts\n\n  # vars.task \u2014 the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs \u2014 leave empty (\"\"). The pipeline fills this in:\n  #   \u2022 Gold variant: injected with canonical doc content\n  #   \u2022 Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions \u2014 how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   \u2014 did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  \u2014 is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled \u2014 set to false to skip this task entirely\n  #   rubric  \u2014 \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
 /** Parsed task data for example-studio-custom-input (JSON-safe) */
 export declare const exampleStudioCustomInputData: readonly [{
     readonly id: "example-studio-custom-input";
     readonly description: "Example — Custom input component in Sanity Studio";
-    readonly canonical_docs: readonly [{
+    readonly featureArea: "studio";
+    readonly canonicalDocs: readonly [{
         readonly slug: "custom-input-components";
         readonly reason: "Guide for building custom form inputs in Sanity Studio";
     }];
-    readonly doc_coverage: true;
-    readonly reference_solution: "canonical/example-studio-custom-input.ts";
+    readonly docCoverage: true;
+    readonly referenceSolution: "canonical/example-studio-custom-input.ts";
     readonly vars: {
         readonly task: "Build a custom string input component for Sanity Studio that shows\na character count below the input field. The component should accept\na maxLength option from the field schema and display a warning when\nthe text exceeds the limit.\n";
         readonly docs: "";
@@ -173,7 +175,7 @@ export declare const exampleStudioCustomInputData: readonly [{
     };
 }];
 /** Raw YAML string for example-studio-custom-input (preserves comments) */
-export declare const exampleStudioCustomInputYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Custom input component in Sanity Studio\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# To disable without deleting:\n#   baseline:\n#     enabled: false\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-studio-custom-input\n  description: \"Example \u2014 Custom input component in Sanity Studio\"\n\n  canonical_docs:\n    - slug: custom-input-components\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n\n  doc_coverage: true\n  reference_solution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
+export declare const exampleStudioCustomInputYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Custom input component in Sanity Studio\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# To disable without deleting:\n#   baseline:\n#     enabled: false\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-studio-custom-input\n  description: \"Example \u2014 Custom input component in Sanity Studio\"\n\n  featureArea: studio\n\n  canonicalDocs:\n    - slug: custom-input-components\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n\n  docCoverage: true\n  referenceSolution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
 /** All task example data as a flat array (JSON-safe) */
 export declare const allTaskData: readonly unknown[];
 /** Map of task ID (filename stem) → raw YAML string (preserves comments) */
@@ -188,3 +190,5 @@ export interface ExampleRecord {
     yaml: string;
 }
 export declare const EXAMPLES: Record<ExampleType, ExampleRecord>;
+/** GitHub Actions workflow template for AI Literacy evaluation */
+export declare const workflowYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# AI Literacy Evaluation \u2014 GitHub Actions workflow\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This workflow submits evaluations to the AILF API when task or config\n# files change in a pull request. The API handles all processing\n# (LLM calls, doc fetching, grading, report publishing).\n#\n# Prerequisites:\n#   Add one secret to your repository (Settings \u2192 Secrets \u2192 Actions):\n#     AILF_API_KEY \u2014 your API key (starts with ailf_live_sk_)\n#\n# Customization:\n#   - Adjust `paths` to match your documentation file locations\n#   - Set full_eval to true for comprehensive (slower) evaluation\n#   - See: https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/API_GATEWAY.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\nname: AI Literacy Eval\n\non:\n  pull_request:\n    branches: [main]\n    paths:\n      - \".ailf/**\"\n\n  # Manual trigger from the Actions tab\n  workflow_dispatch:\n    inputs:\n      full_eval:\n        description: \"Run full evaluation (all tests, slower)\"\n        type: boolean\n        default: false\n\nconcurrency:\n  group: ailf-eval-${{ github.event.pull_request.number || github.ref }}\n  cancel-in-progress: true\n\njobs:\n  evaluate:\n    name: AI Literacy Evaluation\n    runs-on: ubuntu-latest\n    permissions:\n      pull-requests: write\n    steps:\n      # \u2500\u2500\u2500 Submit evaluation to the AILF API \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Submit evaluation\n        id: submit\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          FULL_EVAL: ${{ inputs.full_eval || 'false' }}\n        run: |\n          if [ \"$FULL_EVAL\" = \"true\" ]; then\n            DEBUG_FIELD=\"\"\n          else\n            DEBUG_FIELD='\"debug\": { \"enabled\": true, \"firstN\": 2 },'\n          fi\n\n          PAYLOAD=$(cat <<EOF\n          {\n            \"mode\": \"baseline\",\n            ${DEBUG_FIELD}\n            \"publish\": true,\n            \"compare\": true\n          }\n          EOF\n          )\n\n          RESPONSE=$(curl -sf -X POST \\\n            -H \"Authorization: Bearer $AILF_API_KEY\" \\\n            -H \"Content-Type: application/json\" \\\n            https://ailf-api.sanity.build/v1/pipeline \\\n            -d \"$PAYLOAD\")\n\n          JOB_ID=$(echo \"$RESPONSE\" | jq -r '.jobId')\n          echo \"job_id=$JOB_ID\" >> $GITHUB_OUTPUT\n          echo \"\uD83D\uDCCB Submitted job: $JOB_ID\"\n\n      # \u2500\u2500\u2500 Poll for results (long-polling) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Wait for results\n        id: results\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n        run: |\n          for i in $(seq 1 40); do\n            RESPONSE=$(curl -s \\\n              -H \"Authorization: Bearer $AILF_API_KEY\" \\\n              -H \"Prefer: wait=25\" \\\n              \"https://ailf-api.sanity.build/v1/jobs/$JOB_ID\")\n\n            STATUS=$(echo \"$RESPONSE\" | jq -r '.status')\n\n            case \"$STATUS\" in\n              completed)\n                echo \"status=completed\" >> $GITHUB_OUTPUT\n                echo \"report_id=$(echo $RESPONSE | jq -r '.reportId // empty')\" >> $GITHUB_OUTPUT\n                echo \"score=$(echo $RESPONSE | jq -r '.score // empty')\" >> $GITHUB_OUTPUT\n                echo \"\u2705 Evaluation completed\"\n                exit 0\n                ;;\n              failed|timed-out)\n                echo \"status=$STATUS\" >> $GITHUB_OUTPUT\n                echo \"::error::Evaluation $STATUS\"\n                exit 1\n                ;;\n              *)\n                echo \"\u23F3 [$i/40] $STATUS\"\n                ;;\n            esac\n          done\n\n          echo \"::error::Timed out waiting for evaluation\"\n          exit 1\n\n      # \u2500\u2500\u2500 Post results to PR \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Post PR comment\n        if: >-\n          always() && github.event_name == 'pull_request' &&\n          steps.submit.outputs.job_id != ''\n        uses: actions/github-script@v7\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        with:\n          script: |\n            const marker = '<!-- ailf-score-report -->';\n            const status = process.env.JOB_STATUS;\n            const reportId = process.env.REPORT_ID;\n            const jobId = process.env.JOB_ID;\n            const score = process.env.SCORE;\n\n            let icon, message;\n            if (status === 'completed') {\n              icon = '\u2705';\n              message = score\n                ? `Evaluation completed \u2014 score: **${score}/100**`\n                : 'Evaluation completed successfully.';\n            } else if (status === 'failed' || status === 'timed-out') {\n              icon = '\u26A0\uFE0F';\n              message = `Evaluation ${status}.`;\n            } else {\n              icon = '\u23F3';\n              message = 'Evaluation status unknown (may still be running).';\n            }\n\n            let body = `${marker}\\n## ${icon} AI Literacy Evaluation\\n\\n${message}\\n`;\n            if (reportId) {\n              body += `\\n\uD83D\uDD17 [View detailed report](https://ailf-api.sanity.build/v1/reports/${reportId})\\n`;\n            }\n            body += `\\n<sub>Job: \\`${jobId}\\`</sub>\\n`;\n\n            const { data: comments } = await github.rest.issues.listComments({\n              owner: context.repo.owner,\n              repo: context.repo.repo,\n              issue_number: context.issue.number,\n            });\n            const existing = comments.find(c => c.body?.includes(marker));\n\n            if (existing) {\n              await github.rest.issues.updateComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                comment_id: existing.id,\n                body,\n              });\n            } else {\n              await github.rest.issues.createComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                issue_number: context.issue.number,\n                body,\n              });\n            }\n\n      # \u2500\u2500\u2500 Job summary \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Summary\n        if: always()\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        run: |\n          {\n            echo \"## \uD83D\uDCCA AI Literacy Evaluation\"\n            echo \"\"\n            echo \"| Field | Value |\"\n            echo \"|-------|-------|\"\n            echo \"| Job | \\`$JOB_ID\\` |\"\n            echo \"| Status | $JOB_STATUS |\"\n            [ -n \"$SCORE\" ] && echo \"| Score | $SCORE/100 |\"\n            [ -n \"$REPORT_ID\" ] && echo \"| Report | [$REPORT_ID](https://ailf-api.sanity.build/v1/reports/$REPORT_ID) |\"\n          } >> \"$GITHUB_STEP_SUMMARY\"\n";

package/dist/_vendor/ailf-core/examples/index.js CHANGED Viewed

@@ -119,9 +119,9 @@ export const thresholdYaml = "# Example quality threshold configuration.\n#\n# T
 /** Parsed ailf-config example data (JSON-safe) */
 export const ailfConfigData = {
     "source": {
-        "projectId": "your-project-id",
-        "dataset": "production",
-        "baseUrl": "https://your-site.example.com/docs"
+        "projectId": "3do82whm",
+        "dataset": "next",
+        "baseUrl": "https://www.sanity.io/docs"
     },
     "triggers": {
         "pr": {
@@ -141,13 +141,14 @@ export const ailfConfigData = {
     }
 };
 /** Raw YAML string for ailf-config example (preserves comments) */
-export const ailfConfigYaml = "# ──────────────────────────────────────────────────────────────────────\n# .ailf/config.yaml — AI Literacy Framework project configuration\n# ──────────────────────────────────────────────────────────────────────\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# ──────────────────────────────────────────────────────────────────────\n\n# Documentation source — where to fetch content for evaluation.\n#\n# projectId — your Sanity project ID (find it in sanity.io/manage)\n# dataset   — the dataset to query (e.g., \"production\", \"staging\")\n# baseUrl   — the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"your-project-id\"\n  dataset: production\n  baseUrl: \"https://your-site.example.com/docs\"\n\n# Trigger configuration — when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only — check that task YAML parses correctly (fast, no LLM calls)\n#   eval          — run the full evaluation pipeline\n#\n# paths — only trigger when files matching these globs change\n# blocking — if true, a failing eval blocks the PR merge\n# notify — if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
+export const ailfConfigYaml = "# ──────────────────────────────────────────────────────────────────────\n# .ailf/config.yaml — AI Literacy Framework project configuration\n# ──────────────────────────────────────────────────────────────────────\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Evaluations are submitted to the AILF API (ailf-api.sanity.build).\n# The API handles LLM calls, doc fetching, grading, and report\n# publishing. Your repo only needs one secret: AILF_API_KEY.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# ──────────────────────────────────────────────────────────────────────\n\n# Documentation source — which docs are being evaluated.\n#\n# This tells the pipeline which Sanity project and dataset contain\n# the documentation under test. For most users, this is Sanity's own\n# docs project.\n#\n# projectId — Sanity project ID (find yours at sanity.io/manage)\n# dataset   — the dataset to query (e.g., \"production\", \"next\")\n# baseUrl   — the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"3do82whm\"\n  dataset: next\n  baseUrl: \"https://www.sanity.io/docs\"\n\n# Trigger configuration — when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only — check that task YAML parses correctly (fast, no LLM calls)\n#   eval          — run the full evaluation pipeline\n#\n# paths — only trigger when files matching these globs change\n# blocking — if true, a failing eval blocks the PR merge\n# notify — if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
 /** Parsed task data for example-groq-blog-listing (JSON-safe) */
 export const exampleGroqBlogListingData = [
     {
         "id": "example-groq-blog-listing",
         "description": "Example — Blog listing with GROQ queries",
-        "canonical_docs": [
+        "featureArea": "groq",
+        "canonicalDocs": [
             {
                 "slug": "groq-introduction",
                 "reason": "Core GROQ syntax and query language reference"
@@ -157,8 +158,8 @@ export const exampleGroqBlogListingData = [
                 "reason": "Query execution model and best practices"
             }
         ],
-        "doc_coverage": true,
-        "reference_solution": "canonical/example-groq-blog-listing.ts",
+        "docCoverage": true,
+        "referenceSolution": "canonical/example-groq-blog-listing.ts",
         "vars": {
             "task": "Create a Next.js page component that lists blog posts from Sanity\nusing GROQ. The page should display the title, slug, and published\ndate for each post, sorted by most recent first. Use the Sanity\nclient to fetch data.\n",
             "docs": ""
@@ -189,20 +190,21 @@ export const exampleGroqBlogListingData = [
     }
 ];
 /** Raw YAML string for example-groq-blog-listing (preserves comments) */
-export const exampleGroqBlogListingYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Blog listing with GROQ queries\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# To disable this task without deleting the file, set:\n#   baseline:\n#     enabled: false\n#\n# Full field reference:\n#   https://github.com/sanity-io/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# ──────────────────────────────────────────────────────────────────────\n\n# Unique identifier — lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example — Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  # featureArea is inferred from the filename by default, but you can\n  # set it explicitly here.\n  # featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   — the article's URL slug in your docs site\n  #   reason — why this doc is relevant (helps with auditing)\n  canonical_docs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  doc_coverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  reference_solution: canonical/example-groq-blog-listing.ts\n\n  # vars.task — the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs — leave empty (\"\"). The pipeline fills this in:\n  #   • Gold variant: injected with canonical doc content\n  #   • Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions — how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   — did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  — is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled — set to false to skip this task entirely\n  #   rubric  — \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
+export const exampleGroqBlogListingYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Blog listing with GROQ queries\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# To disable this task without deleting the file, set:\n#   baseline:\n#     enabled: false\n#\n# Full field reference:\n#   https://github.com/sanity-io/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# ──────────────────────────────────────────────────────────────────────\n\n# Unique identifier — lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example — Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   — the article's URL slug in your docs site\n  #   reason — why this doc is relevant (helps with auditing)\n  canonicalDocs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  docCoverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  referenceSolution: canonical/example-groq-blog-listing.ts\n\n  # vars.task — the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs — leave empty (\"\"). The pipeline fills this in:\n  #   • Gold variant: injected with canonical doc content\n  #   • Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions — how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   — did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  — is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled — set to false to skip this task entirely\n  #   rubric  — \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
 /** Parsed task data for example-studio-custom-input (JSON-safe) */
 export const exampleStudioCustomInputData = [
     {
         "id": "example-studio-custom-input",
         "description": "Example — Custom input component in Sanity Studio",
-        "canonical_docs": [
+        "featureArea": "studio",
+        "canonicalDocs": [
             {
                 "slug": "custom-input-components",
                 "reason": "Guide for building custom form inputs in Sanity Studio"
             }
         ],
-        "doc_coverage": true,
-        "reference_solution": "canonical/example-studio-custom-input.ts",
+        "docCoverage": true,
+        "referenceSolution": "canonical/example-studio-custom-input.ts",
         "vars": {
             "task": "Build a custom string input component for Sanity Studio that shows\na character count below the input field. The component should accept\na maxLength option from the field schema and display a warning when\nthe text exceeds the limit.\n",
             "docs": ""
@@ -234,7 +236,7 @@ export const exampleStudioCustomInputData = [
     }
 ];
 /** Raw YAML string for example-studio-custom-input (preserves comments) */
-export const exampleStudioCustomInputYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Custom input component in Sanity Studio\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# To disable without deleting:\n#   baseline:\n#     enabled: false\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-studio-custom-input\n  description: \"Example — Custom input component in Sanity Studio\"\n\n  canonical_docs:\n    - slug: custom-input-components\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n\n  doc_coverage: true\n  reference_solution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
+export const exampleStudioCustomInputYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Custom input component in Sanity Studio\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# To disable without deleting:\n#   baseline:\n#     enabled: false\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-studio-custom-input\n  description: \"Example — Custom input component in Sanity Studio\"\n\n  featureArea: studio\n\n  canonicalDocs:\n    - slug: custom-input-components\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n\n  docCoverage: true\n  referenceSolution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n";
 // ---------------------------------------------------------------------------
 // Aggregate task exports
 // ---------------------------------------------------------------------------
@@ -283,3 +285,8 @@ export const EXAMPLES = {
         yaml: Object.values(taskYamlFiles).join("\n"),
     },
 };
+// ---------------------------------------------------------------------------
+// Raw file exports (non-data files, exported as raw strings)
+// ---------------------------------------------------------------------------
+/** GitHub Actions workflow template for AI Literacy evaluation */
+export const workflowYaml = "# ──────────────────────────────────────────────────────────────────────\n# AI Literacy Evaluation — GitHub Actions workflow\n# ──────────────────────────────────────────────────────────────────────\n#\n# This workflow submits evaluations to the AILF API when task or config\n# files change in a pull request. The API handles all processing\n# (LLM calls, doc fetching, grading, report publishing).\n#\n# Prerequisites:\n#   Add one secret to your repository (Settings → Secrets → Actions):\n#     AILF_API_KEY — your API key (starts with ailf_live_sk_)\n#\n# Customization:\n#   - Adjust `paths` to match your documentation file locations\n#   - Set full_eval to true for comprehensive (slower) evaluation\n#   - See: https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/API_GATEWAY.md\n# ──────────────────────────────────────────────────────────────────────\n\nname: AI Literacy Eval\n\non:\n  pull_request:\n    branches: [main]\n    paths:\n      - \".ailf/**\"\n\n  # Manual trigger from the Actions tab\n  workflow_dispatch:\n    inputs:\n      full_eval:\n        description: \"Run full evaluation (all tests, slower)\"\n        type: boolean\n        default: false\n\nconcurrency:\n  group: ailf-eval-${{ github.event.pull_request.number || github.ref }}\n  cancel-in-progress: true\n\njobs:\n  evaluate:\n    name: AI Literacy Evaluation\n    runs-on: ubuntu-latest\n    permissions:\n      pull-requests: write\n    steps:\n      # ─── Submit evaluation to the AILF API ─────────────────────\n      - name: Submit evaluation\n        id: submit\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          FULL_EVAL: ${{ inputs.full_eval || 'false' }}\n        run: |\n          if [ \"$FULL_EVAL\" = \"true\" ]; then\n            DEBUG_FIELD=\"\"\n          else\n            DEBUG_FIELD='\"debug\": { \"enabled\": true, \"firstN\": 2 },'\n          fi\n\n          PAYLOAD=$(cat <<EOF\n          {\n            \"mode\": \"baseline\",\n            ${DEBUG_FIELD}\n            \"publish\": true,\n            \"compare\": true\n          }\n          EOF\n          )\n\n          RESPONSE=$(curl -sf -X POST \\\n            -H \"Authorization: Bearer $AILF_API_KEY\" \\\n            -H \"Content-Type: application/json\" \\\n            https://ailf-api.sanity.build/v1/pipeline \\\n            -d \"$PAYLOAD\")\n\n          JOB_ID=$(echo \"$RESPONSE\" | jq -r '.jobId')\n          echo \"job_id=$JOB_ID\" >> $GITHUB_OUTPUT\n          echo \"📋 Submitted job: $JOB_ID\"\n\n      # ─── Poll for results (long-polling) ───────────────────────\n      - name: Wait for results\n        id: results\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n        run: |\n          for i in $(seq 1 40); do\n            RESPONSE=$(curl -s \\\n              -H \"Authorization: Bearer $AILF_API_KEY\" \\\n              -H \"Prefer: wait=25\" \\\n              \"https://ailf-api.sanity.build/v1/jobs/$JOB_ID\")\n\n            STATUS=$(echo \"$RESPONSE\" | jq -r '.status')\n\n            case \"$STATUS\" in\n              completed)\n                echo \"status=completed\" >> $GITHUB_OUTPUT\n                echo \"report_id=$(echo $RESPONSE | jq -r '.reportId // empty')\" >> $GITHUB_OUTPUT\n                echo \"score=$(echo $RESPONSE | jq -r '.score // empty')\" >> $GITHUB_OUTPUT\n                echo \"✅ Evaluation completed\"\n                exit 0\n                ;;\n              failed|timed-out)\n                echo \"status=$STATUS\" >> $GITHUB_OUTPUT\n                echo \"::error::Evaluation $STATUS\"\n                exit 1\n                ;;\n              *)\n                echo \"⏳ [$i/40] $STATUS\"\n                ;;\n            esac\n          done\n\n          echo \"::error::Timed out waiting for evaluation\"\n          exit 1\n\n      # ─── Post results to PR ────────────────────────────────────\n      - name: Post PR comment\n        if: >-\n          always() && github.event_name == 'pull_request' &&\n          steps.submit.outputs.job_id != ''\n        uses: actions/github-script@v7\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        with:\n          script: |\n            const marker = '<!-- ailf-score-report -->';\n            const status = process.env.JOB_STATUS;\n            const reportId = process.env.REPORT_ID;\n            const jobId = process.env.JOB_ID;\n            const score = process.env.SCORE;\n\n            let icon, message;\n            if (status === 'completed') {\n              icon = '✅';\n              message = score\n                ? `Evaluation completed — score: **${score}/100**`\n                : 'Evaluation completed successfully.';\n            } else if (status === 'failed' || status === 'timed-out') {\n              icon = '⚠️';\n              message = `Evaluation ${status}.`;\n            } else {\n              icon = '⏳';\n              message = 'Evaluation status unknown (may still be running).';\n            }\n\n            let body = `${marker}\\n## ${icon} AI Literacy Evaluation\\n\\n${message}\\n`;\n            if (reportId) {\n              body += `\\n🔗 [View detailed report](https://ailf-api.sanity.build/v1/reports/${reportId})\\n`;\n            }\n            body += `\\n<sub>Job: \\`${jobId}\\`</sub>\\n`;\n\n            const { data: comments } = await github.rest.issues.listComments({\n              owner: context.repo.owner,\n              repo: context.repo.repo,\n              issue_number: context.issue.number,\n            });\n            const existing = comments.find(c => c.body?.includes(marker));\n\n            if (existing) {\n              await github.rest.issues.updateComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                comment_id: existing.id,\n                body,\n              });\n            } else {\n              await github.rest.issues.createComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                issue_number: context.issue.number,\n                body,\n              });\n            }\n\n      # ─── Job summary ───────────────────────────────────────────\n      - name: Summary\n        if: always()\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        run: |\n          {\n            echo \"## 📊 AI Literacy Evaluation\"\n            echo \"\"\n            echo \"| Field | Value |\"\n            echo \"|-------|-------|\"\n            echo \"| Job | \\`$JOB_ID\\` |\"\n            echo \"| Status | $JOB_STATUS |\"\n            [ -n \"$SCORE\" ] && echo \"| Score | $SCORE/100 |\"\n            [ -n \"$REPORT_ID\" ] && echo \"| Report | [$REPORT_ID](https://ailf-api.sanity.build/v1/reports/$REPORT_ID) |\"\n          } >> \"$GITHUB_STEP_SUMMARY\"\n";

package/dist/_vendor/ailf-core/ports/context.d.ts CHANGED Viewed

@@ -95,6 +95,10 @@ export interface ResolvedConfig {
     taskSourceType?: "content-lake" | "yaml";
     /** Path to repo-based tasks directory (e.g., .ailf/tasks/) */
     repoTasksPath?: string;
+    /** Report store project ID from .ailf/config.yaml reportStore block */
+    reportStoreProjectId?: string;
+    /** Report store dataset from .ailf/config.yaml reportStore block */
+    reportStoreDataset?: string;
     /** Callback URL configuration for API-triggered evaluations */
     callback?: {
         url: string;

package/dist/adapters/task-sources/repo-schemas.d.ts CHANGED Viewed

@@ -185,10 +185,20 @@ export declare const RepoTaskFileSchema: z.ZodArray<z.ZodObject<{
     }, z.core.$strip>>;
 }, z.core.$strip>>;
 /**
- * Zod schema for .ailf/config.yaml — controls how and when evaluations
- * are triggered from an external repository.
+ * Zod schema for .ailf/config.yaml — controls documentation source,
+ * report destination, and trigger behavior for evaluations from an
+ * external repository.
  */
 export declare const RepoConfigSchema: z.ZodObject<{
+    source: z.ZodOptional<z.ZodObject<{
+        projectId: z.ZodOptional<z.ZodString>;
+        dataset: z.ZodOptional<z.ZodString>;
+        baseUrl: z.ZodOptional<z.ZodString>;
+    }, z.core.$strip>>;
+    reportStore: z.ZodOptional<z.ZodObject<{
+        projectId: z.ZodString;
+        dataset: z.ZodString;
+    }, z.core.$strip>>;
     triggers: z.ZodOptional<z.ZodObject<{
         pr: z.ZodOptional<z.ZodObject<{
             mode: z.ZodDefault<z.ZodEnum<{

package/dist/adapters/task-sources/repo-schemas.js CHANGED Viewed

@@ -189,10 +189,36 @@ const ScheduleTriggerSchema = TriggerConfigSchema.extend({
     cron: z.string().min(1),
 });
 /**
- * Zod schema for .ailf/config.yaml — controls how and when evaluations
- * are triggered from an external repository.
+ * Documentation source configuration.
+ * Defines which Sanity project holds the documentation being evaluated.
+ */
+const SourceConfigSchema = z
+    .object({
+    projectId: z.string().min(1).optional(),
+    dataset: z.string().min(1).optional(),
+    baseUrl: z.string().url().optional(),
+})
+    .optional();
+/**
+ * Report store configuration.
+ * Defines which Sanity project receives `ailf.report` documents.
+ * This should match the project/dataset configured in the user's Studio.
+ * The API token comes from the AILF_REPORT_SANITY_API_TOKEN env var.
+ */
+const ReportStoreConfigSchema = z
+    .object({
+    projectId: z.string().min(1),
+    dataset: z.string().min(1),
+})
+    .optional();
+/**
+ * Zod schema for .ailf/config.yaml — controls documentation source,
+ * report destination, and trigger behavior for evaluations from an
+ * external repository.
  */
 export const RepoConfigSchema = z.object({
+    source: SourceConfigSchema,
+    reportStore: ReportStoreConfigSchema,
     triggers: z
         .object({
         pr: TriggerConfigSchema.optional(),

package/dist/cli.js CHANGED Viewed

File without changes

package/dist/commands/init.js CHANGED Viewed

@@ -18,7 +18,7 @@
 import { Command } from "commander";
 import { existsSync, mkdirSync, writeFileSync } from "fs";
 import { resolve, relative } from "path";
-import { ailfConfigData, ailfConfigYaml, taskYamlFiles, TASK_FILE_NAMES, allTaskData, } from "../_vendor/ailf-core/index.js";
+import { ailfConfigData, ailfConfigYaml, taskYamlFiles, TASK_FILE_NAMES, allTaskData, workflowYaml, } from "../_vendor/ailf-core/index.js";
 // ---------------------------------------------------------------------------
 // Command factory
 // ---------------------------------------------------------------------------
@@ -127,7 +127,17 @@ async function runInit(opts) {
     else {
         skipped.push(rel(targetDir, gitignorePath));
     }
-    // 5. Summary
+    // 5. Write GitHub Actions workflow
+    const workflowDir = resolve(targetDir, ".github", "workflows");
+    const workflowPath = resolve(workflowDir, "ailf-eval.yml");
+    mkdirSync(workflowDir, { recursive: true });
+    if (writeIfNew(workflowPath, workflowYaml, force)) {
+        written.push(rel(targetDir, workflowPath));
+    }
+    else {
+        skipped.push(rel(targetDir, workflowPath));
+    }
+    // 6. Summary
     console.log();
     if (written.length > 0) {
         for (const f of written) {
@@ -143,8 +153,10 @@ async function runInit(opts) {
     console.log();
     console.log("  Next steps:");
     console.log();
-    console.log(`  1. Edit ${rel(targetDir, resolve(ailfDir, `config${ext}`))} with your Sanity project settings`);
-    console.log(`  2. Customize the example tasks in ${rel(targetDir, tasksDir)}/`);
-    console.log("  3. Run: ailf pipeline --repo-tasks-path .ailf/tasks/");
+    console.log(`  1. Customize the example tasks in ${rel(targetDir, tasksDir)}/`);
+    console.log("  2. Validate: npx @sanity/ailf validate-tasks .ailf/tasks/");
+    console.log("  3. Set AILF_API_KEY in your environment (e.g. in a local .env file)");
+    console.log("     and add it as a GitHub Actions secret (Settings → Secrets)");
+    console.log("  4. Push — the workflow at .github/workflows/ailf-eval.yml handles the rest");
     console.log();
 }

package/dist/commands/pipeline-action.js CHANGED Viewed

@@ -10,7 +10,7 @@
  *
  * @see packages/eval/src/orchestration/ for the step-based pipeline
  */
-import { writeFileSync } from "fs";
+import { existsSync, readFileSync, writeFileSync } from "fs";
 import { dirname, resolve } from "path";
 import { fileURLToPath } from "url";
 import { classifyUrls } from "../pipeline/classify-url.js";
@@ -18,6 +18,8 @@ import { assessImpact, buildReverseMapping, } from "../pipeline/reverse-mapping.
 import { buildAppContext } from "../orchestration/build-app-context.js";
 import { buildStepSequence } from "../orchestration/build-step-sequence.js";
 import { orchestratePipeline } from "../orchestration/pipeline-orchestrator.js";
+import { load } from "js-yaml";
+import { parseRepoConfig, } from "../adapters/task-sources/repo-schemas.js";
 const __dirname = dirname(fileURLToPath(import.meta.url));
 const ROOT = resolve(__dirname, "..", "..");
 // ---------------------------------------------------------------------------
@@ -32,6 +34,8 @@ const VALID_SEARCH_MODES = ["open", "origin-only", "off"];
  * Exported so the plan builder can call it independently.
  */
 export function computeResolvedOptions(opts) {
+    // Resolve paths relative to the caller's cwd, not the eval package root
+    const callerCwd = process.env.AILF_CALLER_CWD ?? process.cwd();
     // Validate mode
     const mode = opts.mode;
     if (!VALID_MODES.includes(mode)) {
@@ -163,14 +167,21 @@ export function computeResolvedOptions(opts) {
         // Smart default: full runs auto-publish when store is configured
         publishEnabled = reportStoreConfigured && !debugEnabled;
     }
-    // Report store overrides — fall back to the eval dataset so that
-    // perspective evaluations publish reports to the same dataset the
-    // Studio is reading from. AILF_REPORT_DATASET wins when set explicitly.
+    // Report store overrides — resolution order:
+    //   1. Explicit CLI flags (--report-dataset, --report-project)
+    //   2. Environment variables (AILF_REPORT_DATASET, AILF_REPORT_PROJECT_ID)
+    //   3. .ailf/config.yaml reportStore block (when --repo-tasks-path is set)
+    //   4. Eval dataset override (so perspective evals publish to the same dataset)
+    const repoConfig = loadRepoConfigIfPresent(opts.repoTasksPath);
     const reportDataset = opts.reportDataset ??
         process.env.AILF_REPORT_DATASET ??
+        repoConfig?.reportStore?.dataset ??
         datasetOverride ??
         undefined;
-    const reportProjectId = opts.reportProject ?? process.env.AILF_REPORT_PROJECT_ID ?? undefined;
+    const reportProjectId = opts.reportProject ??
+        process.env.AILF_REPORT_PROJECT_ID ??
+        repoConfig?.reportStore?.projectId ??
+        undefined;
     return {
         allowedOriginArgs,
         areaOption,
@@ -206,7 +217,9 @@ export function computeResolvedOptions(opts) {
         skipFetch: opts.skipFetch,
         source: opts.source,
         studioOriginOverride,
-        repoTasksPath: opts.repoTasksPath,
+        repoTasksPath: opts.repoTasksPath
+            ? resolve(callerCwd, opts.repoTasksPath)
+            : undefined,
         taskOption,
         taskSourceType: resolveTaskSourceType(opts.taskSource),
         urlArgs,
@@ -303,3 +316,28 @@ function writePipelineResult(result) {
         // results/latest/ may not exist yet — not critical
     }
 }
+/**
+ * Load .ailf/config.yaml if --repo-tasks-path is set and the config file
+ * exists. Returns null if not applicable.
+ *
+ * The config.yaml lives one level up from the tasks/ directory:
+ *   .ailf/config.yaml  ← config
+ *   .ailf/tasks/       ← repoTasksPath
+ */
+function loadRepoConfigIfPresent(repoTasksPath) {
+    if (!repoTasksPath)
+        return null;
+    // .ailf/tasks/ → .ailf/config.yaml
+    const configPath = resolve(repoTasksPath, "..", "config.yaml");
+    if (!existsSync(configPath))
+        return null;
+    try {
+        const raw = readFileSync(configPath, "utf-8");
+        const parsed = load(raw);
+        return parseRepoConfig(parsed);
+    }
+    catch (err) {
+        console.warn(`  ⚠️  Failed to parse ${configPath}: ${err instanceof Error ? err.message : String(err)}`);
+        return null;
+    }
+}

package/dist/commands/publish.js CHANGED Viewed

@@ -101,7 +101,8 @@ async function runPublishCommand(summaryPath, opts) {
     // -----------------------------------------------------------------------
     // 1. Resolve and read the score summary
     // -----------------------------------------------------------------------
-    const resolvedPath = resolve(summaryPath);
+    const callerCwd = process.env.AILF_CALLER_CWD ?? process.cwd();
+    const resolvedPath = resolve(callerCwd, summaryPath);
     if (!existsSync(resolvedPath)) {
         console.error(`  ✖ File not found: ${resolvedPath}`);
         console.error();

package/dist/commands/validate-tasks.js CHANGED Viewed

@@ -24,7 +24,10 @@ export function createValidateTasksCommand() {
         .argument("[path]", "Path to tasks directory (default: .ailf/tasks/)", ".ailf/tasks")
         .option("--strict", "Treat warnings as errors", false)
         .action(async (tasksPath, opts) => {
-        const resolvedPath = resolve(tasksPath);
+        // Resolve relative to the caller's working directory, not the
+        // eval package root (which differs when run via bin/ailf.js)
+        const callerCwd = process.env.AILF_CALLER_CWD ?? process.cwd();
+        const resolvedPath = resolve(callerCwd, tasksPath);
         if (!existsSync(resolvedPath)) {
             console.error(`❌ Directory not found: ${resolvedPath}`);
             process.exit(1);

package/dist/composition-root.js CHANGED Viewed

@@ -43,7 +43,7 @@ export function createAppContext(config) {
     // Eval runner — Promptfoo subprocess
     const evalRunner = new PromptfooEvalAdapter(config.rootDir);
     // Report store — Sanity Content Lake (for publish + auto-compare)
-    const reportStore = createReportStore();
+    const reportStore = createReportStore(config);
     // Sinks — loaded from config/sinks.yaml
     const sinks = loadSinks();
     return {
@@ -75,7 +75,7 @@ function createCache(config) {
     const token = process.env.AILF_REPORT_SANITY_API_TOKEN ?? process.env.SANITY_API_TOKEN;
     if (!token)
         return local;
-    return new ContentLakeCacheAdapter(local, createReportStore());
+    return new ContentLakeCacheAdapter(local, createReportStore(config));
 }
 function createTaskSource(config) {
     // Primary source — selected by config.taskSourceType
@@ -96,10 +96,14 @@ function createTaskSource(config) {
     }
     return primary;
 }
-function createReportStore() {
+function createReportStore(config) {
     return new ReportStore({
-        dataset: process.env.AILF_REPORT_DATASET ?? undefined,
-        projectId: process.env.AILF_REPORT_PROJECT_ID ?? undefined,
+        dataset: process.env.AILF_REPORT_DATASET ??
+            config?.reportStoreDataset ??
+            undefined,
+        projectId: process.env.AILF_REPORT_PROJECT_ID ??
+            config?.reportStoreProjectId ??
+            undefined,
         token: process.env.AILF_REPORT_SANITY_API_TOKEN ??
             process.env.SANITY_API_TOKEN ??
             undefined,

package/dist/orchestration/build-app-context.js CHANGED Viewed

@@ -67,6 +67,8 @@ export function mapToResolvedConfig(opts, rootDir) {
         beforeOption: opts.beforeOption,
         taskSourceType: opts.taskSourceType,
         repoTasksPath: opts.repoTasksPath,
+        reportStoreProjectId: opts.reportProjectId,
+        reportStoreDataset: opts.reportDataset,
     };
 }
 /**

package/package.json CHANGED Viewed

@@ -1,6 +1,6 @@
 {
   "name": "@sanity/ailf",
-  "version": "0.1.0",
+  "version": "0.1.2",
   "private": false,
   "publishConfig": {
     "access": "restricted"

package/dist/commands/update-quality-scores.d.ts DELETED Viewed

@@ -1,5 +0,0 @@
-/**
- * update-quality-scores command — update QUALITY_SCORE.md from scores.
- */
-import { Command } from "commander";
-export declare function createUpdateQualityScoresCommand(): Command;

package/dist/commands/update-quality-scores.js DELETED Viewed

@@ -1,20 +0,0 @@
-/**
- * update-quality-scores command — update QUALITY_SCORE.md from scores.
- */
-import { Command } from "commander";
-export function createUpdateQualityScoresCommand() {
-    return new Command("update-quality-scores")
-        .description("Update docs/QUALITY_SCORE.md from score-summary.json")
-        .action(async () => {
-        const { updateQualityScores } = await import("../scripts/update-quality-scores.js");
-        console.log("=== Updating QUALITY_SCORE.md from score-summary.json ===\n");
-        const result = updateQualityScores();
-        if (result.success) {
-            console.log(`  ✅ ${result.message}`);
-        }
-        else {
-            console.error(`  ❌ ${result.message}`);
-            process.exit(1);
-        }
-    });
-}

package/dist/lib/agent-behavior-report.d.ts DELETED Viewed

@@ -1,8 +0,0 @@
-/**
- * lib/agent-behavior-report.ts — DEPRECATED re-export shim.
- * @deprecated Import from ../pipeline/agent-behavior-report.js instead.
- */
-import "dotenv/config";
-export { analyzeResults, CANONICAL_DOC_MAP, detectFeatureArea, } from "../pipeline/agent-behavior-report.js";
-export type { AnalysisResult, FeatureAnalysis, TaskBehavior, TestResult, } from "../pipeline/agent-behavior-report.js";
-export declare function main(resultsPathArg?: string): void;