npm - @sanity/ailf - Versions diffs - 0.1.34 → 0.2.0 - Mend

@sanity/ailf 0.1.34 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (26) hide show

package/LICENSE +21 -0
package/config/airbyte/ai_literacy_framework.connector.yaml +6 -0
package/config/bigquery/views/reports.sql +1 -0
package/dist/_vendor/ailf-core/examples/index.d.ts +10 -20
package/dist/_vendor/ailf-core/examples/index.js +10 -20
package/dist/_vendor/ailf-core/ports/task-source.d.ts +2 -0
package/dist/_vendor/ailf-core/types/index.d.ts +12 -0
package/dist/_vendor/ailf-tasks/schemas.d.ts +12 -0
package/dist/_vendor/ailf-tasks/schemas.js +4 -0
package/dist/adapters/task-sources/content-lake-task-source.js +9 -1
package/dist/adapters/task-sources/repo-task-source.js +19 -4
package/dist/commands/calculate-scores.js +5 -1
package/dist/commands/publish.js +3 -0
package/dist/orchestration/steps/calculate-scores-step.js +18 -19
package/dist/orchestration/steps/publish-report-step.js +3 -0
package/dist/pipeline/calculate-scores.d.ts +6 -1
package/dist/pipeline/calculate-scores.js +5 -13
package/dist/pipeline/generate-configs.js +4 -9
package/dist/pipeline/mirror-repo-tasks.d.ts +77 -0
package/dist/pipeline/mirror-repo-tasks.js +141 -27
package/dist/pipeline/report-title.d.ts +66 -0
package/dist/pipeline/report-title.js +118 -0
package/dist/report-store.js +2 -0
package/dist/sinks/bigquery/index.d.ts +1 -0
package/dist/sinks/bigquery/index.js +1 -0
package/package.json +23 -23

package/LICENSE ADDED Viewed

@@ -0,0 +1,21 @@
+MIT License
+Copyright (c) 2025 Sanity.io
+Permission is hereby granted, free of charge, to any person obtaining a copy
+of this software and associated documentation files (the "Software"), to deal
+in the Software without restriction, including without limitation the rights
+to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
+copies of the Software, and to permit persons to whom the Software is
+furnished to do so, subject to the following conditions:
+The above copyright notice and this permission notice shall be included in all
+copies or substantial portions of the Software.
+THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
+FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
+AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
+LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
+OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
+SOFTWARE.

package/config/airbyte/ai_literacy_framework.connector.yaml CHANGED Viewed

@@ -56,6 +56,7 @@ definitions:
                 "completed_at": completedAt,
                 "duration_ms": durationMs,
                 tag,
+                title,
                 "mode": provenance.mode,
                 "source_name": provenance.source.name,
                 "source_base_url": provenance.source.baseUrl,
@@ -318,6 +319,11 @@ schemas:
           - string
           - "null"
         description: Optional human-supplied label
+      title:
+        type:
+          - string
+          - "null"
+        description: Auto-generated descriptive title for discoverability
       mode:
         type:
           - string

package/config/bigquery/views/reports.sql CHANGED Viewed

@@ -19,6 +19,7 @@ SELECT
   TIMESTAMP(completed_at)                         AS completed_at,
   CAST(duration_ms AS INT64)                      AS duration_ms,
   tag,
+  title,
   mode,
   source_name,
   source_base_url,

package/dist/_vendor/ailf-core/examples/index.d.ts CHANGED Viewed

@@ -142,12 +142,10 @@ export declare const exampleGroqBlogListingData: readonly [{
         readonly enabled: true;
         readonly rubric: "abbreviated";
     };
-    readonly execution: {
-        readonly enabled: false;
-    };
+    readonly status: "draft";
 }];
 /** Raw YAML string for example-groq-blog-listing (preserves comments) */
-export declare const exampleGroqBlogListingYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Blog listing with GROQ queries\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# Full field reference:\n#   https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Unique identifier \u2014 lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example \u2014 Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   \u2014 the article's URL slug in your docs site\n  #   reason \u2014 why this doc is relevant (helps with auditing)\n  #\n  # This example uses slug-based references \u2014 the simplest form.\n  # See the other example tasks for path, id, and perspective references.\n  canonicalDocs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  docCoverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  referenceSolution: canonical/example-groq-blog-listing.ts\n\n  # vars.task \u2014 the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs \u2014 leave empty (\"\"). The pipeline fills this in:\n  #   \u2022 Gold variant: injected with canonical doc content\n  #   \u2022 Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions \u2014 how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   \u2014 did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  \u2014 is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled \u2014 set to false to skip this task entirely\n  #   rubric  \u2014 \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Execution configuration.\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export declare const exampleGroqBlogListingYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Blog listing with GROQ queries\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# Full field reference:\n#   https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Unique identifier \u2014 lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example \u2014 Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   \u2014 the article's URL slug in your docs site\n  #   reason \u2014 why this doc is relevant (helps with auditing)\n  #\n  # This example uses slug-based references \u2014 the simplest form.\n  # See the other example tasks for path, id, and perspective references.\n  canonicalDocs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  docCoverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  referenceSolution: canonical/example-groq-blog-listing.ts\n\n  # vars.task \u2014 the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs \u2014 leave empty (\"\"). The pipeline fills this in:\n  #   \u2022 Gold variant: injected with canonical doc content\n  #   \u2022 Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions \u2014 how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   \u2014 did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  \u2014 is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled \u2014 set to false to skip this task entirely\n  #   rubric  \u2014 \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-id-based-ref (JSON-safe) */
 export declare const exampleIdBasedRefData: readonly [{
     readonly id: "example-id-based-ref";
@@ -180,12 +178,10 @@ export declare const exampleIdBasedRefData: readonly [{
         readonly enabled: true;
         readonly rubric: "abbreviated";
     };
-    readonly execution: {
-        readonly enabled: false;
-    };
+    readonly status: "draft";
 }];
 /** Raw YAML string for example-id-based-ref (preserves comments) */
-export declare const exampleIdBasedRefYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Document ID-based canonical doc references\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# Demonstrates using `id` to reference canonical documentation by\n# Sanity document `_id`. This is useful for:\n#   - Draft documents that don't have a stable slug yet\n#   - Programmatic references from imports or migrations\n#   - Documents where you know the _id but not the slug\n#\n# The `id` ref type can also carry optional `slug` and `path` fields\n# as human-readable annotations \u2014 these are NOT used for resolution,\n# only for display in logs and reports.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-id-based-ref\n  description: \"Example \u2014 GROQ feature support (ID-based doc references)\"\n\n  featureArea: groq\n\n  # ID-based canonical doc references.\n  #\n  # Use the Sanity document _id to reference articles directly.\n  # Optional slug/path annotations help humans reading the YAML\n  # but are NOT used for resolution \u2014 only the `id` field matters.\n  #\n  # These IDs reference real articles in the Sanity docs (next dataset):\n  #   0ba88f1b... = \"GROQ feature support across Sanity\"\n  #   5b9c2863... = \"Custom GROQ functions\"\n  canonicalDocs:\n    - id: \"0ba88f1b-d1a7-418a-9267-2e343d01886a\"\n      slug: groq-feature-support-by-context # annotation only \u2014 not used for resolution\n      reason: \"GROQ feature support across different Sanity contexts\"\n    - id: \"5b9c2863-ef01-4565-af8e-ee54e081ee74\"\n      slug: custom-groq-functions # annotation only \u2014 not used for resolution\n      reason: \"Custom GROQ functions and pipelines\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how GROQ is used across different Sanity contexts.\n      Cover the following:\n      1. Which GROQ features are available in each context (API queries,\n         webhooks, custom functions, access control)\n      2. How to create and use custom GROQ functions\n      3. Any differences in GROQ support between contexts\n      Provide examples demonstrating context-specific GROQ patterns.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains GROQ availability across different Sanity contexts\"\n        - \"Describes custom GROQ function creation and usage\"\n        - \"Notes differences in GROQ support between contexts\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"GROQ examples use valid syntax\"\n        - \"Custom function examples follow the correct API pattern\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export declare const exampleIdBasedRefYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Document ID-based canonical doc references\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# Demonstrates using `id` to reference canonical documentation by\n# Sanity document `_id`. This is useful for:\n#   - Draft documents that don't have a stable slug yet\n#   - Programmatic references from imports or migrations\n#   - Documents where you know the _id but not the slug\n#\n# The `id` ref type can also carry optional `slug` and `path` fields\n# as human-readable annotations \u2014 these are NOT used for resolution,\n# only for display in logs and reports.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-id-based-ref\n  description: \"Example \u2014 GROQ feature support (ID-based doc references)\"\n\n  featureArea: groq\n\n  # ID-based canonical doc references.\n  #\n  # Use the Sanity document _id to reference articles directly.\n  # Optional slug/path annotations help humans reading the YAML\n  # but are NOT used for resolution \u2014 only the `id` field matters.\n  #\n  # These IDs reference real articles in the Sanity docs (next dataset):\n  #   0ba88f1b... = \"GROQ feature support across Sanity\"\n  #   5b9c2863... = \"Custom GROQ functions\"\n  canonicalDocs:\n    - id: \"0ba88f1b-d1a7-418a-9267-2e343d01886a\"\n      slug: groq-feature-support-by-context # annotation only \u2014 not used for resolution\n      reason: \"GROQ feature support across different Sanity contexts\"\n    - id: \"5b9c2863-ef01-4565-af8e-ee54e081ee74\"\n      slug: custom-groq-functions # annotation only \u2014 not used for resolution\n      reason: \"Custom GROQ functions and pipelines\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how GROQ is used across different Sanity contexts.\n      Cover the following:\n      1. Which GROQ features are available in each context (API queries,\n         webhooks, custom functions, access control)\n      2. How to create and use custom GROQ functions\n      3. Any differences in GROQ support between contexts\n      Provide examples demonstrating context-specific GROQ patterns.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains GROQ availability across different Sanity contexts\"\n        - \"Describes custom GROQ function creation and usage\"\n        - \"Notes differences in GROQ support between contexts\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"GROQ examples use valid syntax\"\n        - \"Custom function examples follow the correct API pattern\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-path-based-ref (JSON-safe) */
 export declare const examplePathBasedRefData: readonly [{
     readonly id: "example-path-based-ref";
@@ -216,12 +212,10 @@ export declare const examplePathBasedRefData: readonly [{
         readonly enabled: true;
         readonly rubric: "abbreviated";
     };
-    readonly execution: {
-        readonly enabled: false;
-    };
+    readonly status: "draft";
 }];
 /** Raw YAML string for example-path-based-ref (preserves comments) */
-export declare const examplePathBasedRefYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Path-based canonical doc references\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# Demonstrates using `path` to reference canonical documentation.\n# Paths are the preferred reference type because they uniquely identify\n# an article across sections (unlike slugs, which can collide).\n#\n# Path format:\n#   - Simple:   \"webhooks\"              \u2192 resolves by slug lookup\n#   - Sectioned: \"content-lake/webhooks\" \u2192 disambiguates by section + slug\n#\n# This example demonstrates why paths matter: the slug \"documents\"\n# exists in both the \"content-lake\" and \"cli-reference\" sections.\n# Using \"content-lake/documents\" ensures we get the right one.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-path-based-ref\n  description: \"Example \u2014 GROQ mutations (path-based doc references)\"\n\n  featureArea: groq\n\n  # Path-based canonical doc references.\n  #\n  # Use \"section/slug\" format to uniquely identify articles:\n  #   - \"content-lake/mutations-introduction\" \u2192 the mutations article\n  #   - \"content-lake/documents\" \u2192 the documents article in Content Lake\n  #     (not the CLI \"documents\" article in cli-reference section)\n  #\n  # The \"documents\" slug exists in two sections \u2014 this is exactly why\n  # path-based references are preferred over slug-based references.\n  canonicalDocs:\n    - path: content-lake/mutations-introduction\n      reason: \"Introduction to document mutations in the Content Lake\"\n    - path: content-lake/documents\n      reason: \"Document structure and types (Content Lake, not CLI reference)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how to create, update, and delete documents in Sanity's\n      Content Lake using mutations. Cover:\n      1. The different mutation types (create, createOrReplace, patch, delete)\n      2. Document structure and required fields (_id, _type)\n      3. How to use patch operations to update specific fields\n      4. Best practices for mutation patterns\n      Provide working code examples using @sanity/client.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains create, createOrReplace, patch, and delete mutations\"\n        - \"Describes required document fields (_id, _type)\"\n        - \"Shows patch operations for field-level updates\"\n        - \"Includes practical code examples\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses correct @sanity/client mutation API\"\n        - \"Patch operations use valid set/unset/inc syntax\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export declare const examplePathBasedRefYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Path-based canonical doc references\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# Demonstrates using `path` to reference canonical documentation.\n# Paths are the preferred reference type because they uniquely identify\n# an article across sections (unlike slugs, which can collide).\n#\n# Path format:\n#   - Simple:   \"webhooks\"              \u2192 resolves by slug lookup\n#   - Sectioned: \"content-lake/webhooks\" \u2192 disambiguates by section + slug\n#\n# This example demonstrates why paths matter: the slug \"documents\"\n# exists in both the \"content-lake\" and \"cli-reference\" sections.\n# Using \"content-lake/documents\" ensures we get the right one.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-path-based-ref\n  description: \"Example \u2014 GROQ mutations (path-based doc references)\"\n\n  featureArea: groq\n\n  # Path-based canonical doc references.\n  #\n  # Use \"section/slug\" format to uniquely identify articles:\n  #   - \"content-lake/mutations-introduction\" \u2192 the mutations article\n  #   - \"content-lake/documents\" \u2192 the documents article in Content Lake\n  #     (not the CLI \"documents\" article in cli-reference section)\n  #\n  # The \"documents\" slug exists in two sections \u2014 this is exactly why\n  # path-based references are preferred over slug-based references.\n  canonicalDocs:\n    - path: content-lake/mutations-introduction\n      reason: \"Introduction to document mutations in the Content Lake\"\n    - path: content-lake/documents\n      reason: \"Document structure and types (Content Lake, not CLI reference)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how to create, update, and delete documents in Sanity's\n      Content Lake using mutations. Cover:\n      1. The different mutation types (create, createOrReplace, patch, delete)\n      2. Document structure and required fields (_id, _type)\n      3. How to use patch operations to update specific fields\n      4. Best practices for mutation patterns\n      Provide working code examples using @sanity/client.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains create, createOrReplace, patch, and delete mutations\"\n        - \"Describes required document fields (_id, _type)\"\n        - \"Shows patch operations for field-level updates\"\n        - \"Includes practical code examples\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses correct @sanity/client mutation API\"\n        - \"Patch operations use valid set/unset/inc syntax\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-perspective-ref (JSON-safe) */
 export declare const examplePerspectiveRefData: readonly [{
     readonly id: "example-perspective-ref";
@@ -252,12 +246,10 @@ export declare const examplePerspectiveRefData: readonly [{
         readonly enabled: true;
         readonly rubric: "abbreviated";
     };
-    readonly execution: {
-        readonly enabled: false;
-    };
+    readonly status: "draft";
 }];
 /** Raw YAML string for example-perspective-ref (preserves comments) */
-export declare const examplePerspectiveRefYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Perspective / content release doc references\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# Demonstrates using `perspective` to reference all documentation\n# articles within a content release. This is the key capability for\n# evaluating NEW feature documentation before it's published.\n#\n# How it works:\n#   - A perspective ref is one-to-many: the doc fetcher queries the\n#     named release and expands it to ALL articles versioned within it.\n#   - Downstream consumers see the same flat DocContext[] regardless\n#     of how docs were resolved.\n#   - When the release is published, the perspective entry becomes a\n#     no-op (articles are now in published). Migrate to explicit path\n#     or slug refs at your convenience.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-perspective-ref\n  description:\n    \"Example \u2014 GROQ features from content release (perspective-based doc\n    references)\"\n\n  featureArea: groq\n\n  # Perspective-based canonical doc reference.\n  #\n  # The perspective ID references a content release in the Sanity\n  # Content Lake. At evaluation time, the doc fetcher auto-discovers\n  # all articles versioned in this release and includes them as\n  # canonical documentation context.\n  #\n  # Release rE9TSJvR4 contains:\n  #   - \"GROQ-powered webhooks\" (webhooks)\n  #   - \"Query Cheat Sheet - GROQ\" (query-cheat-sheet)\n  #   - \"GROQ joins\" (groq-joins)\n  #\n  # You can combine perspective refs with explicit slug/path/id refs\n  # to include foundational published docs alongside release content.\n  # Here we add groq-data-types as a complementary published reference.\n  canonicalDocs:\n    - perspective: rE9TSJvR4\n      reason: \"All GROQ documentation updates in the test content release\"\n    - slug: groq-data-types\n      reason: \"GROQ data type reference (published, stable)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Using GROQ, demonstrate advanced query patterns including:\n      1. Joining data across document types using references\n      2. Filtering webhook payloads with GROQ projections\n      3. Using the query cheat sheet patterns for common operations\n      4. Working with different GROQ data types in filters\n      Provide working GROQ query examples for each pattern.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Demonstrates GROQ join syntax for cross-document queries\"\n        - \"Shows GROQ filter patterns for webhook configuration\"\n        - \"Includes practical query examples from cheat sheet patterns\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"All GROQ queries use valid syntax\"\n        - \"Reference joins use correct dereference operator (->)\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export declare const examplePerspectiveRefYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Perspective / content release doc references\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# Demonstrates using `perspective` to reference all documentation\n# articles within a content release. This is the key capability for\n# evaluating NEW feature documentation before it's published.\n#\n# How it works:\n#   - A perspective ref is one-to-many: the doc fetcher queries the\n#     named release and expands it to ALL articles versioned within it.\n#   - Downstream consumers see the same flat DocContext[] regardless\n#     of how docs were resolved.\n#   - When the release is published, the perspective entry becomes a\n#     no-op (articles are now in published). Migrate to explicit path\n#     or slug refs at your convenience.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-perspective-ref\n  description:\n    \"Example \u2014 GROQ features from content release (perspective-based doc\n    references)\"\n\n  featureArea: groq\n\n  # Perspective-based canonical doc reference.\n  #\n  # The perspective ID references a content release in the Sanity\n  # Content Lake. At evaluation time, the doc fetcher auto-discovers\n  # all articles versioned in this release and includes them as\n  # canonical documentation context.\n  #\n  # Release rE9TSJvR4 contains:\n  #   - \"GROQ-powered webhooks\" (webhooks)\n  #   - \"Query Cheat Sheet - GROQ\" (query-cheat-sheet)\n  #   - \"GROQ joins\" (groq-joins)\n  #\n  # You can combine perspective refs with explicit slug/path/id refs\n  # to include foundational published docs alongside release content.\n  # Here we add groq-data-types as a complementary published reference.\n  canonicalDocs:\n    - perspective: rE9TSJvR4\n      reason: \"All GROQ documentation updates in the test content release\"\n    - slug: groq-data-types\n      reason: \"GROQ data type reference (published, stable)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Using GROQ, demonstrate advanced query patterns including:\n      1. Joining data across document types using references\n      2. Filtering webhook payloads with GROQ projections\n      3. Using the query cheat sheet patterns for common operations\n      4. Working with different GROQ data types in filters\n      Provide working GROQ query examples for each pattern.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Demonstrates GROQ join syntax for cross-document queries\"\n        - \"Shows GROQ filter patterns for webhook configuration\"\n        - \"Includes practical query examples from cheat sheet patterns\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"All GROQ queries use valid syntax\"\n        - \"Reference joins use correct dereference operator (->)\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-studio-custom-input (JSON-safe) */
 export declare const exampleStudioCustomInputData: readonly [{
     readonly id: "example-studio-custom-input";
@@ -289,12 +281,10 @@ export declare const exampleStudioCustomInputData: readonly [{
         readonly enabled: true;
         readonly rubric: "abbreviated";
     };
-    readonly execution: {
-        readonly enabled: false;
-    };
+    readonly status: "draft";
 }];
 /** Raw YAML string for example-studio-custom-input (preserves comments) */
-export declare const exampleStudioCustomInputYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Custom input component in Sanity Studio\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-studio-custom-input\n  description: \"Example \u2014 Custom input component in Sanity Studio\"\n\n  featureArea: studio\n\n  # Slug-based canonical doc references.\n  canonicalDocs:\n    - slug: custom-input-widgets\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n    - slug: form-components\n      reason: \"Form component API and customization patterns\"\n\n  docCoverage: true\n  referenceSolution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export declare const exampleStudioCustomInputYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# Example Task: Custom input component in Sanity Studio\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This is a starter template \u2014 edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n- id: example-studio-custom-input\n  description: \"Example \u2014 Custom input component in Sanity Studio\"\n\n  featureArea: studio\n\n  # Slug-based canonical doc references.\n  canonicalDocs:\n    - slug: custom-input-widgets\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n    - slug: form-components\n      reason: \"Form component API and customization patterns\"\n\n  docCoverage: true\n  referenceSolution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** All task example data as a flat array (JSON-safe) */
 export declare const allTaskData: readonly unknown[];
 /** Map of task ID (filename stem) → raw YAML string (preserves comments) */

package/dist/_vendor/ailf-core/examples/index.js CHANGED Viewed

@@ -187,13 +187,11 @@ export const exampleGroqBlogListingData = [
             "enabled": true,
             "rubric": "abbreviated"
         },
-        "execution": {
-            "enabled": false
-        }
+        "status": "draft"
     }
 ];
 /** Raw YAML string for example-groq-blog-listing (preserves comments) */
-export const exampleGroqBlogListingYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Blog listing with GROQ queries\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# Full field reference:\n#   https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# ──────────────────────────────────────────────────────────────────────\n\n# Unique identifier — lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example — Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   — the article's URL slug in your docs site\n  #   reason — why this doc is relevant (helps with auditing)\n  #\n  # This example uses slug-based references — the simplest form.\n  # See the other example tasks for path, id, and perspective references.\n  canonicalDocs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  docCoverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  referenceSolution: canonical/example-groq-blog-listing.ts\n\n  # vars.task — the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs — leave empty (\"\"). The pipeline fills this in:\n  #   • Gold variant: injected with canonical doc content\n  #   • Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions — how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   — did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  — is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled — set to false to skip this task entirely\n  #   rubric  — \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Execution configuration.\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export const exampleGroqBlogListingYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Blog listing with GROQ queries\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Each task evaluates whether an AI coding agent can implement a feature\n# using your docs as context. Delete this file or replace it entirely.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# Full field reference:\n#   https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/CONTRIBUTING_TASKS.md\n# ──────────────────────────────────────────────────────────────────────\n\n# Unique identifier — lowercase alphanumeric with hyphens.\n# Must be unique across all task files in .ailf/tasks/.\n- id: example-groq-blog-listing\n\n  # Short human-readable summary. Shown in score tables and reports.\n  description: \"Example — Blog listing with GROQ queries\"\n\n  # Feature area this task belongs to. Tasks with the same area are\n  # grouped together in score summaries. Use a short kebab-case name.\n  featureArea: groq\n\n  # Gold-standard documentation articles for this task. The pipeline\n  # fetches these from Sanity and injects them into the prompt for\n  # baseline evaluation. Each entry needs:\n  #   slug   — the article's URL slug in your docs site\n  #   reason — why this doc is relevant (helps with auditing)\n  #\n  # This example uses slug-based references — the simplest form.\n  # See the other example tasks for path, id, and perspective references.\n  canonicalDocs:\n    - slug: groq-introduction\n      reason: \"Core GROQ syntax and query language reference\"\n    - slug: how-queries-work\n      reason: \"Query execution model and best practices\"\n\n  # When true, the pipeline auto-generates an additional rubric that\n  # checks whether the LLM's response actually used the provided docs.\n  docCoverage: true\n\n  # Path to a gold-standard implementation, relative to canonical/.\n  # The grader uses this as a reference when scoring code correctness.\n  referenceSolution: canonical/example-groq-blog-listing.ts\n\n  # vars.task — the implementation prompt given to the LLM.\n  # Write this as if you're asking a developer to build the feature.\n  # Be specific about requirements so the grader can evaluate clearly.\n  #\n  # vars.docs — leave empty (\"\"). The pipeline fills this in:\n  #   • Gold variant: injected with canonical doc content\n  #   • Baseline variant: left empty (tests model knowledge alone)\n  vars:\n    task: |\n      Create a Next.js page component that lists blog posts from Sanity\n      using GROQ. The page should display the title, slug, and published\n      date for each post, sorted by most recent first. Use the Sanity\n      client to fetch data.\n    docs: \"\"\n\n  # Grading assertions — how the LLM's response is scored.\n  #\n  # \"llm-rubric\" assertions use a grader LLM to score against criteria.\n  # The \"template\" references a rubric from config/rubrics.yaml.\n  # The \"criteria\" are task-specific bullets injected into the template.\n  #\n  # Available templates:\n  #   task-completion   — did the LLM implement the feature? (weight: 0.50)\n  #   code-correctness  — is the code idiomatic and correct? (weight: 0.25)\n  #\n  # You can also use value-based assertions:\n  #   - type: contains\n  #     value: \"client.fetch\"\n  #   - type: contains-any\n  #     value: [\"createClient\", \"sanityClient\"]\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Uses the groq tagged template literal\"\n        - \"Fetches blog posts with title, slug, and publishedAt fields\"\n        - \"Orders results by publishedAt in descending order\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses createClient from @sanity/client or next-sanity\"\n        - \"Exports a valid Next.js page component\"\n\n  # Baseline variant configuration.\n  #   enabled — set to false to skip this task entirely\n  #   rubric  — \"abbreviated\" (faster, default), \"full\", or \"none\"\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-id-based-ref (JSON-safe) */
 export const exampleIdBasedRefData = [
     {
@@ -240,13 +238,11 @@ export const exampleIdBasedRefData = [
             "enabled": true,
             "rubric": "abbreviated"
         },
-        "execution": {
-            "enabled": false
-        }
+        "status": "draft"
     }
 ];
 /** Raw YAML string for example-id-based-ref (preserves comments) */
-export const exampleIdBasedRefYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Document ID-based canonical doc references\n# ──────────────────────────────────────────────────────────────────────\n#\n# Demonstrates using `id` to reference canonical documentation by\n# Sanity document `_id`. This is useful for:\n#   - Draft documents that don't have a stable slug yet\n#   - Programmatic references from imports or migrations\n#   - Documents where you know the _id but not the slug\n#\n# The `id` ref type can also carry optional `slug` and `path` fields\n# as human-readable annotations — these are NOT used for resolution,\n# only for display in logs and reports.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-id-based-ref\n  description: \"Example — GROQ feature support (ID-based doc references)\"\n\n  featureArea: groq\n\n  # ID-based canonical doc references.\n  #\n  # Use the Sanity document _id to reference articles directly.\n  # Optional slug/path annotations help humans reading the YAML\n  # but are NOT used for resolution — only the `id` field matters.\n  #\n  # These IDs reference real articles in the Sanity docs (next dataset):\n  #   0ba88f1b... = \"GROQ feature support across Sanity\"\n  #   5b9c2863... = \"Custom GROQ functions\"\n  canonicalDocs:\n    - id: \"0ba88f1b-d1a7-418a-9267-2e343d01886a\"\n      slug: groq-feature-support-by-context # annotation only — not used for resolution\n      reason: \"GROQ feature support across different Sanity contexts\"\n    - id: \"5b9c2863-ef01-4565-af8e-ee54e081ee74\"\n      slug: custom-groq-functions # annotation only — not used for resolution\n      reason: \"Custom GROQ functions and pipelines\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how GROQ is used across different Sanity contexts.\n      Cover the following:\n      1. Which GROQ features are available in each context (API queries,\n         webhooks, custom functions, access control)\n      2. How to create and use custom GROQ functions\n      3. Any differences in GROQ support between contexts\n      Provide examples demonstrating context-specific GROQ patterns.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains GROQ availability across different Sanity contexts\"\n        - \"Describes custom GROQ function creation and usage\"\n        - \"Notes differences in GROQ support between contexts\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"GROQ examples use valid syntax\"\n        - \"Custom function examples follow the correct API pattern\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export const exampleIdBasedRefYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Document ID-based canonical doc references\n# ──────────────────────────────────────────────────────────────────────\n#\n# Demonstrates using `id` to reference canonical documentation by\n# Sanity document `_id`. This is useful for:\n#   - Draft documents that don't have a stable slug yet\n#   - Programmatic references from imports or migrations\n#   - Documents where you know the _id but not the slug\n#\n# The `id` ref type can also carry optional `slug` and `path` fields\n# as human-readable annotations — these are NOT used for resolution,\n# only for display in logs and reports.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-id-based-ref\n  description: \"Example — GROQ feature support (ID-based doc references)\"\n\n  featureArea: groq\n\n  # ID-based canonical doc references.\n  #\n  # Use the Sanity document _id to reference articles directly.\n  # Optional slug/path annotations help humans reading the YAML\n  # but are NOT used for resolution — only the `id` field matters.\n  #\n  # These IDs reference real articles in the Sanity docs (next dataset):\n  #   0ba88f1b... = \"GROQ feature support across Sanity\"\n  #   5b9c2863... = \"Custom GROQ functions\"\n  canonicalDocs:\n    - id: \"0ba88f1b-d1a7-418a-9267-2e343d01886a\"\n      slug: groq-feature-support-by-context # annotation only — not used for resolution\n      reason: \"GROQ feature support across different Sanity contexts\"\n    - id: \"5b9c2863-ef01-4565-af8e-ee54e081ee74\"\n      slug: custom-groq-functions # annotation only — not used for resolution\n      reason: \"Custom GROQ functions and pipelines\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how GROQ is used across different Sanity contexts.\n      Cover the following:\n      1. Which GROQ features are available in each context (API queries,\n         webhooks, custom functions, access control)\n      2. How to create and use custom GROQ functions\n      3. Any differences in GROQ support between contexts\n      Provide examples demonstrating context-specific GROQ patterns.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains GROQ availability across different Sanity contexts\"\n        - \"Describes custom GROQ function creation and usage\"\n        - \"Notes differences in GROQ support between contexts\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"GROQ examples use valid syntax\"\n        - \"Custom function examples follow the correct API pattern\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-path-based-ref (JSON-safe) */
 export const examplePathBasedRefData = [
     {
@@ -292,13 +288,11 @@ export const examplePathBasedRefData = [
             "enabled": true,
             "rubric": "abbreviated"
         },
-        "execution": {
-            "enabled": false
-        }
+        "status": "draft"
     }
 ];
 /** Raw YAML string for example-path-based-ref (preserves comments) */
-export const examplePathBasedRefYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Path-based canonical doc references\n# ──────────────────────────────────────────────────────────────────────\n#\n# Demonstrates using `path` to reference canonical documentation.\n# Paths are the preferred reference type because they uniquely identify\n# an article across sections (unlike slugs, which can collide).\n#\n# Path format:\n#   - Simple:   \"webhooks\"              → resolves by slug lookup\n#   - Sectioned: \"content-lake/webhooks\" → disambiguates by section + slug\n#\n# This example demonstrates why paths matter: the slug \"documents\"\n# exists in both the \"content-lake\" and \"cli-reference\" sections.\n# Using \"content-lake/documents\" ensures we get the right one.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-path-based-ref\n  description: \"Example — GROQ mutations (path-based doc references)\"\n\n  featureArea: groq\n\n  # Path-based canonical doc references.\n  #\n  # Use \"section/slug\" format to uniquely identify articles:\n  #   - \"content-lake/mutations-introduction\" → the mutations article\n  #   - \"content-lake/documents\" → the documents article in Content Lake\n  #     (not the CLI \"documents\" article in cli-reference section)\n  #\n  # The \"documents\" slug exists in two sections — this is exactly why\n  # path-based references are preferred over slug-based references.\n  canonicalDocs:\n    - path: content-lake/mutations-introduction\n      reason: \"Introduction to document mutations in the Content Lake\"\n    - path: content-lake/documents\n      reason: \"Document structure and types (Content Lake, not CLI reference)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how to create, update, and delete documents in Sanity's\n      Content Lake using mutations. Cover:\n      1. The different mutation types (create, createOrReplace, patch, delete)\n      2. Document structure and required fields (_id, _type)\n      3. How to use patch operations to update specific fields\n      4. Best practices for mutation patterns\n      Provide working code examples using @sanity/client.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains create, createOrReplace, patch, and delete mutations\"\n        - \"Describes required document fields (_id, _type)\"\n        - \"Shows patch operations for field-level updates\"\n        - \"Includes practical code examples\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses correct @sanity/client mutation API\"\n        - \"Patch operations use valid set/unset/inc syntax\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export const examplePathBasedRefYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Path-based canonical doc references\n# ──────────────────────────────────────────────────────────────────────\n#\n# Demonstrates using `path` to reference canonical documentation.\n# Paths are the preferred reference type because they uniquely identify\n# an article across sections (unlike slugs, which can collide).\n#\n# Path format:\n#   - Simple:   \"webhooks\"              → resolves by slug lookup\n#   - Sectioned: \"content-lake/webhooks\" → disambiguates by section + slug\n#\n# This example demonstrates why paths matter: the slug \"documents\"\n# exists in both the \"content-lake\" and \"cli-reference\" sections.\n# Using \"content-lake/documents\" ensures we get the right one.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-path-based-ref\n  description: \"Example — GROQ mutations (path-based doc references)\"\n\n  featureArea: groq\n\n  # Path-based canonical doc references.\n  #\n  # Use \"section/slug\" format to uniquely identify articles:\n  #   - \"content-lake/mutations-introduction\" → the mutations article\n  #   - \"content-lake/documents\" → the documents article in Content Lake\n  #     (not the CLI \"documents\" article in cli-reference section)\n  #\n  # The \"documents\" slug exists in two sections — this is exactly why\n  # path-based references are preferred over slug-based references.\n  canonicalDocs:\n    - path: content-lake/mutations-introduction\n      reason: \"Introduction to document mutations in the Content Lake\"\n    - path: content-lake/documents\n      reason: \"Document structure and types (Content Lake, not CLI reference)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Explain how to create, update, and delete documents in Sanity's\n      Content Lake using mutations. Cover:\n      1. The different mutation types (create, createOrReplace, patch, delete)\n      2. Document structure and required fields (_id, _type)\n      3. How to use patch operations to update specific fields\n      4. Best practices for mutation patterns\n      Provide working code examples using @sanity/client.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Explains create, createOrReplace, patch, and delete mutations\"\n        - \"Describes required document fields (_id, _type)\"\n        - \"Shows patch operations for field-level updates\"\n        - \"Includes practical code examples\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses correct @sanity/client mutation API\"\n        - \"Patch operations use valid set/unset/inc syntax\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-perspective-ref (JSON-safe) */
 export const examplePerspectiveRefData = [
     {
@@ -343,13 +337,11 @@ export const examplePerspectiveRefData = [
             "enabled": true,
             "rubric": "abbreviated"
         },
-        "execution": {
-            "enabled": false
-        }
+        "status": "draft"
     }
 ];
 /** Raw YAML string for example-perspective-ref (preserves comments) */
-export const examplePerspectiveRefYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Perspective / content release doc references\n# ──────────────────────────────────────────────────────────────────────\n#\n# Demonstrates using `perspective` to reference all documentation\n# articles within a content release. This is the key capability for\n# evaluating NEW feature documentation before it's published.\n#\n# How it works:\n#   - A perspective ref is one-to-many: the doc fetcher queries the\n#     named release and expands it to ALL articles versioned within it.\n#   - Downstream consumers see the same flat DocContext[] regardless\n#     of how docs were resolved.\n#   - When the release is published, the perspective entry becomes a\n#     no-op (articles are now in published). Migrate to explicit path\n#     or slug refs at your convenience.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-perspective-ref\n  description:\n    \"Example — GROQ features from content release (perspective-based doc\n    references)\"\n\n  featureArea: groq\n\n  # Perspective-based canonical doc reference.\n  #\n  # The perspective ID references a content release in the Sanity\n  # Content Lake. At evaluation time, the doc fetcher auto-discovers\n  # all articles versioned in this release and includes them as\n  # canonical documentation context.\n  #\n  # Release rE9TSJvR4 contains:\n  #   - \"GROQ-powered webhooks\" (webhooks)\n  #   - \"Query Cheat Sheet - GROQ\" (query-cheat-sheet)\n  #   - \"GROQ joins\" (groq-joins)\n  #\n  # You can combine perspective refs with explicit slug/path/id refs\n  # to include foundational published docs alongside release content.\n  # Here we add groq-data-types as a complementary published reference.\n  canonicalDocs:\n    - perspective: rE9TSJvR4\n      reason: \"All GROQ documentation updates in the test content release\"\n    - slug: groq-data-types\n      reason: \"GROQ data type reference (published, stable)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Using GROQ, demonstrate advanced query patterns including:\n      1. Joining data across document types using references\n      2. Filtering webhook payloads with GROQ projections\n      3. Using the query cheat sheet patterns for common operations\n      4. Working with different GROQ data types in filters\n      Provide working GROQ query examples for each pattern.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Demonstrates GROQ join syntax for cross-document queries\"\n        - \"Shows GROQ filter patterns for webhook configuration\"\n        - \"Includes practical query examples from cheat sheet patterns\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"All GROQ queries use valid syntax\"\n        - \"Reference joins use correct dereference operator (->)\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export const examplePerspectiveRefYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Perspective / content release doc references\n# ──────────────────────────────────────────────────────────────────────\n#\n# Demonstrates using `perspective` to reference all documentation\n# articles within a content release. This is the key capability for\n# evaluating NEW feature documentation before it's published.\n#\n# How it works:\n#   - A perspective ref is one-to-many: the doc fetcher queries the\n#     named release and expands it to ALL articles versioned within it.\n#   - Downstream consumers see the same flat DocContext[] regardless\n#     of how docs were resolved.\n#   - When the release is published, the perspective entry becomes a\n#     no-op (articles are now in published). Migrate to explicit path\n#     or slug refs at your convenience.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n#\n# @see docs/design-docs/canonical-doc-resolution.md\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-perspective-ref\n  description:\n    \"Example — GROQ features from content release (perspective-based doc\n    references)\"\n\n  featureArea: groq\n\n  # Perspective-based canonical doc reference.\n  #\n  # The perspective ID references a content release in the Sanity\n  # Content Lake. At evaluation time, the doc fetcher auto-discovers\n  # all articles versioned in this release and includes them as\n  # canonical documentation context.\n  #\n  # Release rE9TSJvR4 contains:\n  #   - \"GROQ-powered webhooks\" (webhooks)\n  #   - \"Query Cheat Sheet - GROQ\" (query-cheat-sheet)\n  #   - \"GROQ joins\" (groq-joins)\n  #\n  # You can combine perspective refs with explicit slug/path/id refs\n  # to include foundational published docs alongside release content.\n  # Here we add groq-data-types as a complementary published reference.\n  canonicalDocs:\n    - perspective: rE9TSJvR4\n      reason: \"All GROQ documentation updates in the test content release\"\n    - slug: groq-data-types\n      reason: \"GROQ data type reference (published, stable)\"\n\n  docCoverage: true\n\n  vars:\n    task: |\n      Using GROQ, demonstrate advanced query patterns including:\n      1. Joining data across document types using references\n      2. Filtering webhook payloads with GROQ projections\n      3. Using the query cheat sheet patterns for common operations\n      4. Working with different GROQ data types in filters\n      Provide working GROQ query examples for each pattern.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Demonstrates GROQ join syntax for cross-document queries\"\n        - \"Shows GROQ filter patterns for webhook configuration\"\n        - \"Includes practical query examples from cheat sheet patterns\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"All GROQ queries use valid syntax\"\n        - \"Reference joins use correct dereference operator (->)\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 /** Parsed task data for example-studio-custom-input (JSON-safe) */
 export const exampleStudioCustomInputData = [
     {
@@ -396,13 +388,11 @@ export const exampleStudioCustomInputData = [
             "enabled": true,
             "rubric": "abbreviated"
         },
-        "execution": {
-            "enabled": false
-        }
+        "status": "draft"
     }
 ];
 /** Raw YAML string for example-studio-custom-input (preserves comments) */
-export const exampleStudioCustomInputYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Custom input component in Sanity Studio\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# This example task is DISABLED by default. To enable it, either:\n#   1. Remove the execution.enabled: false line below, or\n#   2. Set execution.enabled: true\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-studio-custom-input\n  description: \"Example — Custom input component in Sanity Studio\"\n\n  featureArea: studio\n\n  # Slug-based canonical doc references.\n  canonicalDocs:\n    - slug: custom-input-widgets\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n    - slug: form-components\n      reason: \"Form component API and customization patterns\"\n\n  docCoverage: true\n  referenceSolution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship disabled so they don't run automatically.\n  # Set enabled: true (or remove this block) to activate.\n  execution:\n    enabled: false\n";
+export const exampleStudioCustomInputYaml = "# ──────────────────────────────────────────────────────────────────────\n# Example Task: Custom input component in Sanity Studio\n# ──────────────────────────────────────────────────────────────────────\n#\n# This is a starter template — edit it for your own documentation.\n# Delete this file or replace it with your own tasks.\n#\n# This example task ships as a DRAFT so it does not run in production\n#   evaluations automatically. To activate it, change status to \"active\"\n# or remove the status line entirely (defaults to active).\n# ──────────────────────────────────────────────────────────────────────\n\n- id: example-studio-custom-input\n  description: \"Example — Custom input component in Sanity Studio\"\n\n  featureArea: studio\n\n  # Slug-based canonical doc references.\n  canonicalDocs:\n    - slug: custom-input-widgets\n      reason: \"Guide for building custom form inputs in Sanity Studio\"\n    - slug: form-components\n      reason: \"Form component API and customization patterns\"\n\n  docCoverage: true\n  referenceSolution: canonical/example-studio-custom-input.ts\n\n  vars:\n    task: |\n      Build a custom string input component for Sanity Studio that shows\n      a character count below the input field. The component should accept\n      a maxLength option from the field schema and display a warning when\n      the text exceeds the limit.\n    docs: \"\"\n\n  assert:\n    - type: llm-rubric\n      template: task-completion\n      criteria:\n        - \"Implements a React component that renders a text input\"\n        - \"Displays a live character count\"\n        - \"Reads maxLength from schema options\"\n        - \"Shows a visual warning when limit is exceeded\"\n\n    - type: llm-rubric\n      template: code-correctness\n      criteria:\n        - \"Uses the Sanity UI library for styling\"\n        - \"Calls onChange with patch operations\"\n\n  baseline:\n    enabled: true\n    rubric: abbreviated\n\n  # Example tasks ship as drafts so they don't run in production evals.\n  # Change to \"active\" (or remove this line) to activate.\n  status: draft\n";
 // ---------------------------------------------------------------------------
 // Aggregate task exports
 // ---------------------------------------------------------------------------

package/dist/_vendor/ailf-core/ports/task-source.d.ts CHANGED Viewed

@@ -112,6 +112,8 @@ export interface TaskDefinition {
     baseline?: BaselineConfig;
     /** Additional template variables beyond task (e.g., custom vars) */
     extraVars?: Record<string, unknown>;
+    /** Lifecycle status — controls pipeline inclusion. Absent = "active". */
+    status?: "active" | "archived" | "draft" | "paused";
     /** Freeform labels for filtering and organization */
     tags?: string[];
 }

package/dist/_vendor/ailf-core/types/index.d.ts CHANGED Viewed

@@ -179,6 +179,8 @@ export interface FeatureScore {
 export interface FilterOptions {
     /** Feature areas to include (filename stems, e.g., ["groq", "frameworks"]) */
     areas?: string[];
+    /** Include draft-status tasks in addition to active tasks */
+    includeDrafts?: boolean;
     /** Tags to include — tasks must have at least one matching tag */
     tags?: string[];
     /** Specific task IDs to include (e.g., ["groq-blog-queries"]) */
@@ -452,6 +454,14 @@ export interface PipelineState {
      * Consumed by GenerateConfigsStep and RunEvalStep to narrow scope.
      */
     releaseAutoScope?: ReleaseAutoScope;
+    /**
+     * Feature areas that scored below the critical threshold (40).
+     * Set by CalculateScoresStep, consumed by the orchestrator for
+     * final pipeline result reporting. The pipeline continues running
+     * (gap-analysis, publish, report, compare) even when areas are
+     * below threshold — this is informational, not a hard failure.
+     */
+    belowCritical?: string[];
 }
 /**
  * Release auto-scope metadata — which tasks are affected by a content
@@ -1018,6 +1028,8 @@ export interface Report {
     summary: ScoreSummary;
     /** Optional human-supplied label */
     tag?: string;
+    /** Auto-generated descriptive title for discoverability and sharing */
+    title?: string;
 }
 /** Branded type for report identifiers (UUID v7 for time-sortability) */
 export type ReportId = string & {

package/dist/_vendor/ailf-tasks/schemas.d.ts CHANGED Viewed

@@ -34,6 +34,12 @@ export type RubricTemplateName = (typeof RUBRIC_TEMPLATE_NAMES)[number];
 export declare const RepoTaskSchema: z.ZodObject<{
     id: z.ZodString;
     description: z.ZodString;
+    status: z.ZodDefault<z.ZodOptional<z.ZodEnum<{
+        active: "active";
+        draft: "draft";
+        paused: "paused";
+        archived: "archived";
+    }>>>;
     featureArea: z.ZodString;
     tags: z.ZodOptional<z.ZodArray<z.ZodString>>;
     canonicalDocs: z.ZodDefault<z.ZodOptional<z.ZodArray<z.ZodUnion<readonly [z.ZodObject<{
@@ -113,6 +119,12 @@ export type RepoTask = z.infer<typeof RepoTaskSchema>;
 export declare const RepoTaskFileSchema: z.ZodArray<z.ZodObject<{
     id: z.ZodString;
     description: z.ZodString;
+    status: z.ZodDefault<z.ZodOptional<z.ZodEnum<{
+        active: "active";
+        draft: "draft";
+        paused: "paused";
+        archived: "archived";
+    }>>>;
     featureArea: z.ZodString;
     tags: z.ZodOptional<z.ZodArray<z.ZodString>>;
     canonicalDocs: z.ZodDefault<z.ZodOptional<z.ZodArray<z.ZodUnion<readonly [z.ZodObject<{

package/dist/_vendor/ailf-tasks/schemas.js CHANGED Viewed

@@ -151,6 +151,10 @@ export const RepoTaskSchema = z.object({
         .min(1)
         .regex(/^[a-z0-9][a-z0-9-]*$/, "Task ID must be lowercase alphanumeric with hyphens"),
     description: z.string().min(1),
+    status: z
+        .enum(["active", "draft", "paused", "archived"])
+        .optional()
+        .default("active"),
     featureArea: z
         .string()
         .min(1)

package/dist/adapters/task-sources/content-lake-task-source.js CHANGED Viewed

@@ -31,7 +31,14 @@ const TASKS_QUERY = /* groq */ `
 *[_type == "ailf.task"
   && (!defined($areas) || featureArea->areaId.current in $areas)
   && (!defined($taskIds) || id.current in $taskIds)
-  && (execution.enabled != false)
+  && (
+    // Status-based filtering (unified — replaces execution.enabled)
+    status == "active"
+    || !defined(status)
+    || ($includeDrafts == true && status == "draft")
+    // Explicit --task targeting bypasses draft/paused (but not archived)
+    || (defined($taskIds) && status != "archived")
+  )
   && (!defined($tags) || count((tags)[@ in $tags]) > 0)
 ] | order(featureArea->areaId.current asc, id.current asc) {
   "taskId": id.current,
@@ -92,6 +99,7 @@ function buildGroqParams(filter) {
         areas: filter?.areas && filter.areas.length > 0
             ? filter.areas.map((a) => a.toLowerCase())
             : null,
+        includeDrafts: filter?.includeDrafts ?? false,
         tags: filter?.tags && filter.tags.length > 0 ? filter.tags : null,
         taskIds: filter?.taskIds && filter.taskIds.length > 0 ? filter.taskIds : null,
     };

package/dist/adapters/task-sources/repo-task-source.js CHANGED Viewed

@@ -60,7 +60,8 @@ export class RepoTaskSource {
                 // Filter stages:
                 // 1. Area filter — skip tasks outside requested feature areas
                 // 2. Task ID filter — skip tasks not matching explicit task IDs
-                // 3. Execution.enabled — skip tasks explicitly disabled
+                // 3. Status filter — skip non-active tasks (unless targeting by ID)
+                // 4. Tag filter — skip tasks not matching requested tags
                 // Area filter
                 if (filter?.areas &&
                     filter.areas.length > 0 &&
@@ -75,9 +76,22 @@ export class RepoTaskSource {
                     !filter.taskIds.includes(entry.id)) {
                     continue;
                 }
-                // Execution.enabled filter — skip tasks explicitly disabled
-                if (entry.execution?.enabled === false) {
-                    continue;
+                // Status filter — unified lifecycle control
+                // Resolve effective status: explicit status field wins,
+                // then fall back to execution.enabled for backwards compat
+                const effectiveStatus = entry.status ??
+                    (entry.execution?.enabled === false ? "paused" : "active");
+                const isTargetedById = filter?.taskIds && filter.taskIds.includes(entry.id);
+                if (effectiveStatus === "archived") {
+                    continue; // Archived is always excluded, even with --task
+                }
+                if (effectiveStatus === "paused" && !isTargetedById) {
+                    continue; // Paused skipped unless explicitly targeted
+                }
+                if (effectiveStatus === "draft" &&
+                    !isTargetedById &&
+                    !filter?.includeDrafts) {
+                    continue; // Draft skipped unless targeted or includeDrafts
                 }
                 // Tag filter — skip tasks that don't match any requested tag
                 if (filter?.tags &&
@@ -114,6 +128,7 @@ function mapToTaskDefinition(raw) {
         taskPrompt: typeof task === "string" ? task : "",
         ...(raw.baseline ? { baseline: raw.baseline } : {}),
         ...(extraVars ? { extraVars } : {}),
+        ...(raw.status && raw.status !== "active" ? { status: raw.status } : {}),
         ...(raw.tags?.length ? { tags: raw.tags } : {}),
     };
 }

package/dist/commands/calculate-scores.js CHANGED Viewed

@@ -36,11 +36,15 @@ export function createCalculateScoresCommand() {
                 remote: false,
                 apiUrl: "https://ailf-api.sanity.build",
             });
-            calculateAndWriteScores({
+            const result = calculateAndWriteScores({
                 resultsPath,
                 rootDir: ctx.config.rootDir,
                 source: opts.source,
             });
+            // At the CLI boundary, exit non-zero if areas are below threshold
+            if (result.belowCritical.length > 0) {
+                process.exitCode = 1;
+            }
         }
         catch (err) {
             process.exitCode = 1;

package/dist/commands/publish.js CHANGED Viewed

@@ -24,6 +24,7 @@ import { fileURLToPath } from "url";
 import { Command } from "commander";
 import { createAppContext } from "../composition-root.js";
 import { buildProvenance, } from "../pipeline/provenance.js";
+import { generateReportTitle } from "../pipeline/report-title.js";
 import { generateReportId, } from "../report-store.js";
 import { withRetry } from "../sinks/retry.js";
 const __dirname = dirname(fileURLToPath(import.meta.url));
@@ -166,6 +167,7 @@ async function runPublishCommand(summaryPath, opts) {
         };
     }
     const reportId = generateReportId();
+    const title = generateReportTitle({ provenance });
     const report = {
         comparison: comparison ?? undefined,
         completedAt: now,
@@ -174,6 +176,7 @@ async function runPublishCommand(summaryPath, opts) {
         provenance,
         summary,
         tag: opts.tag,
+        title,
     };
     // -----------------------------------------------------------------------
     // 4. Dry run — print preview and exit