npm - @sanity/ailf - Versions diffs - 0.1.0 → 0.1.1 - Mend

@sanity/ailf 0.1.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Files changed (125) hide show

package/dist/_vendor/ailf-core/examples/index.d.ts +6 -4
package/dist/_vendor/ailf-core/examples/index.js +9 -4
package/dist/_vendor/ailf-core/ports/context.d.ts +4 -0
package/dist/adapters/task-sources/repo-schemas.d.ts +12 -2
package/dist/adapters/task-sources/repo-schemas.js +28 -2
package/dist/cli.js +0 -0
package/dist/commands/init.js +39 -5
package/dist/commands/pipeline-action.js +44 -6
package/dist/commands/publish.js +2 -1
package/dist/commands/validate-tasks.js +4 -1
package/dist/composition-root.js +9 -5
package/dist/orchestration/build-app-context.js +2 -0
package/package.json +1 -1
package/dist/commands/update-quality-scores.d.ts +0 -5
package/dist/commands/update-quality-scores.js +0 -20
package/dist/lib/agent-behavior-report.d.ts +0 -8
package/dist/lib/agent-behavior-report.js +0 -185
package/dist/lib/baseline.d.ts +0 -19
package/dist/lib/baseline.js +0 -153
package/dist/lib/calculate-scores.d.ts +0 -23
package/dist/lib/calculate-scores.js +0 -42
package/dist/lib/compare.d.ts +0 -18
package/dist/lib/compare.js +0 -170
package/dist/lib/coverage-audit.d.ts +0 -4
package/dist/lib/coverage-audit.js +0 -42
package/dist/lib/discovery-report.d.ts +0 -13
package/dist/lib/discovery-report.js +0 -57
package/dist/lib/fetch-docs.d.ts +0 -30
package/dist/lib/fetch-docs.js +0 -171
package/dist/lib/generate-configs.d.ts +0 -25
package/dist/lib/generate-configs.js +0 -42
package/dist/lib/grader-api.d.ts +0 -21
package/dist/lib/grader-api.js +0 -34
package/dist/lib/grader-compare.d.ts +0 -19
package/dist/lib/grader-compare.js +0 -91
package/dist/lib/grader-consistency.d.ts +0 -27
package/dist/lib/grader-consistency.js +0 -79
package/dist/lib/grader-sensitivity.d.ts +0 -19
package/dist/lib/grader-sensitivity.js +0 -75
package/dist/lib/grader-validate.d.ts +0 -19
package/dist/lib/grader-validate.js +0 -78
package/dist/lib/measure-retrieval.d.ts +0 -14
package/dist/lib/measure-retrieval.js +0 -71
package/dist/lib/pr-comment.d.ts +0 -16
package/dist/lib/pr-comment.js +0 -28
package/dist/lib/readiness-report.d.ts +0 -13
package/dist/lib/readiness-report.js +0 -108
package/dist/lib/webhook-server.d.ts +0 -11
package/dist/lib/webhook-server.js +0 -24
package/dist/lib/weekly-digest.d.ts +0 -24
package/dist/lib/weekly-digest.js +0 -148
package/dist/orchestration/env-bridge.d.ts +0 -21
package/dist/orchestration/env-bridge.js +0 -66
package/dist/orchestration/steps/fetch-docs-shell.d.ts +0 -17
package/dist/orchestration/steps/fetch-docs-shell.js +0 -30
package/dist/pipeline/steps/calculate-scores-step.d.ts +0 -11
package/dist/pipeline/steps/calculate-scores-step.js +0 -89
package/dist/pipeline/steps/compare-step.d.ts +0 -18
package/dist/pipeline/steps/compare-step.js +0 -90
package/dist/pipeline/steps/eval-step.d.ts +0 -53
package/dist/pipeline/steps/eval-step.js +0 -347
package/dist/pipeline/steps/fetch-docs-step.d.ts +0 -11
package/dist/pipeline/steps/fetch-docs-step.js +0 -84
package/dist/pipeline/steps/generate-configs-step.d.ts +0 -11
package/dist/pipeline/steps/generate-configs-step.js +0 -98
package/dist/pipeline/steps/grader-consistency-step.d.ts +0 -21
package/dist/pipeline/steps/grader-consistency-step.js +0 -74
package/dist/pipeline/steps/publish-report-step.d.ts +0 -57
package/dist/pipeline/steps/publish-report-step.js +0 -243
package/dist/pipeline/steps/report-step.d.ts +0 -13
package/dist/pipeline/steps/report-step.js +0 -56
package/dist/pipeline/steps/update-scores-step.d.ts +0 -11
package/dist/pipeline/steps/update-scores-step.js +0 -42
package/dist/scripts/agent-behavior-report.d.ts +0 -19
package/dist/scripts/agent-behavior-report.js +0 -315
package/dist/scripts/baseline.d.ts +0 -43
package/dist/scripts/baseline.js +0 -267
package/dist/scripts/calculate-scores.d.ts +0 -166
package/dist/scripts/calculate-scores.js +0 -1296
package/dist/scripts/compare.d.ts +0 -22
package/dist/scripts/compare.js +0 -334
package/dist/scripts/coverage-audit.d.ts +0 -44
package/dist/scripts/coverage-audit.js +0 -209
package/dist/scripts/debug-eval.d.ts +0 -19
package/dist/scripts/debug-eval.js +0 -73
package/dist/scripts/discovery-report.d.ts +0 -58
package/dist/scripts/discovery-report.js +0 -250
package/dist/scripts/fetch-docs.d.ts +0 -35
package/dist/scripts/fetch-docs.js +0 -472
package/dist/scripts/generate-configs.d.ts +0 -66
package/dist/scripts/generate-configs.js +0 -459
package/dist/scripts/grader-api.d.ts +0 -27
package/dist/scripts/grader-api.js +0 -206
package/dist/scripts/grader-compare.d.ts +0 -22
package/dist/scripts/grader-compare.js +0 -368
package/dist/scripts/grader-consistency.d.ts +0 -20
package/dist/scripts/grader-consistency.js +0 -313
package/dist/scripts/grader-sensitivity.d.ts +0 -22
package/dist/scripts/grader-sensitivity.js +0 -354
package/dist/scripts/grader-validate.d.ts +0 -19
package/dist/scripts/grader-validate.js +0 -267
package/dist/scripts/measure-retrieval.d.ts +0 -10
package/dist/scripts/measure-retrieval.js +0 -145
package/dist/scripts/pipeline.d.ts +0 -76
package/dist/scripts/pipeline.js +0 -1031
package/dist/scripts/pr-comment.d.ts +0 -10
package/dist/scripts/pr-comment.js +0 -510
package/dist/scripts/readiness-report.d.ts +0 -88
package/dist/scripts/readiness-report.js +0 -342
package/dist/scripts/update-quality-scores.d.ts +0 -15
package/dist/scripts/update-quality-scores.js +0 -184
package/dist/scripts/validate.d.ts +0 -13
package/dist/scripts/validate.js +0 -79
package/dist/scripts/webhook-server.d.ts +0 -26
package/dist/scripts/webhook-server.js +0 -147
package/dist/scripts/weekly-digest.d.ts +0 -24
package/dist/scripts/weekly-digest.js +0 -144
package/dist/sinks/format-slack.d.ts +0 -64
package/dist/sinks/format-slack.js +0 -306
package/dist/sinks/slack-sink.d.ts +0 -27
package/dist/sinks/slack-sink.js +0 -78
package/dist/sinks/webhook-sink.d.ts +0 -19
package/dist/sinks/webhook-sink.js +0 -50
package/tasks/.expanded.agentic.yaml +0 -51
package/tasks/.expanded.yaml +0 -66

package/dist/_vendor/ailf-core/examples/index.d.ts CHANGED Viewed

@@ -90,9 +90,9 @@ export declare const thresholdYaml = "# Example quality threshold configuration.
 /** Parsed ailf-config example data (JSON-safe) */
 export declare const ailfConfigData: {
     readonly source: {
-        readonly projectId: "your-project-id";
-        readonly dataset: "production";
-        readonly baseUrl: "https://your-site.example.com/docs";
+        readonly projectId: "3do82whm";
+        readonly dataset: "next";
+        readonly baseUrl: "https://www.sanity.io/docs";
     };
     readonly triggers: {
         readonly pr: {
@@ -110,7 +110,7 @@ export declare const ailfConfigData: {
     };
 };
 /** Raw YAML string for ailf-config example (preserves comments) */
-export declare const ailfConfigYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# .ailf/config.yaml \u2014 AI Literacy Framework project configuration\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Documentation source \u2014 where to fetch content for evaluation.\n#\n# projectId \u2014 your Sanity project ID (find it in sanity.io/manage)\n# dataset   \u2014 the dataset to query (e.g., \"production\", \"staging\")\n# baseUrl   \u2014 the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"your-project-id\"\n  dataset: production\n  baseUrl: \"https://your-site.example.com/docs\"\n\n# Trigger configuration \u2014 when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only \u2014 check that task YAML parses correctly (fast, no LLM calls)\n#   eval          \u2014 run the full evaluation pipeline\n#\n# paths \u2014 only trigger when files matching these globs change\n# blocking \u2014 if true, a failing eval blocks the PR merge\n# notify \u2014 if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
+export declare const ailfConfigYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# .ailf/config.yaml \u2014 AI Literacy Framework project configuration\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Evaluations are submitted to the AILF API (ailf-api.sanity.build).\n# The API handles LLM calls, doc fetching, grading, and report\n# publishing. Your repo only needs one secret: AILF_API_KEY.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\n# Documentation source \u2014 which docs are being evaluated.\n#\n# This tells the pipeline which Sanity project and dataset contain\n# the documentation under test. For most users, this is Sanity's own\n# docs project.\n#\n# projectId \u2014 Sanity project ID (find yours at sanity.io/manage)\n# dataset   \u2014 the dataset to query (e.g., \"production\", \"next\")\n# baseUrl   \u2014 the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"3do82whm\"\n  dataset: next\n  baseUrl: \"https://www.sanity.io/docs\"\n\n# Trigger configuration \u2014 when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only \u2014 check that task YAML parses correctly (fast, no LLM calls)\n#   eval          \u2014 run the full evaluation pipeline\n#\n# paths \u2014 only trigger when files matching these globs change\n# blocking \u2014 if true, a failing eval blocks the PR merge\n# notify \u2014 if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
 /** Parsed task data for example-groq-blog-listing (JSON-safe) */
 export declare const exampleGroqBlogListingData: readonly [{
     readonly id: "example-groq-blog-listing";
@@ -188,3 +188,5 @@ export interface ExampleRecord {
     yaml: string;
 }
 export declare const EXAMPLES: Record<ExampleType, ExampleRecord>;
+/** GitHub Actions workflow template for AI Literacy evaluation */
+export declare const workflowYaml = "# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n# AI Literacy Evaluation \u2014 GitHub Actions workflow\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n#\n# This workflow submits evaluations to the AILF API when task or config\n# files change in a pull request. The API handles all processing\n# (LLM calls, doc fetching, grading, report publishing).\n#\n# Prerequisites:\n#   Add one secret to your repository (Settings \u2192 Secrets \u2192 Actions):\n#     AILF_API_KEY \u2014 your API key (starts with ailf_live_sk_)\n#\n# Customization:\n#   - Adjust `paths` to match your documentation file locations\n#   - Set full_eval to true for comprehensive (slower) evaluation\n#   - See: https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/API_GATEWAY.md\n# \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n\nname: AI Literacy Eval\n\non:\n  pull_request:\n    branches: [main]\n    paths:\n      - \".ailf/**\"\n\n  # Manual trigger from the Actions tab\n  workflow_dispatch:\n    inputs:\n      full_eval:\n        description: \"Run full evaluation (all tests, slower)\"\n        type: boolean\n        default: false\n\nconcurrency:\n  group: ailf-eval-${{ github.event.pull_request.number || github.ref }}\n  cancel-in-progress: true\n\njobs:\n  evaluate:\n    name: AI Literacy Evaluation\n    runs-on: ubuntu-latest\n    permissions:\n      pull-requests: write\n    steps:\n      # \u2500\u2500\u2500 Submit evaluation to the AILF API \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Submit evaluation\n        id: submit\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          FULL_EVAL: ${{ inputs.full_eval || 'false' }}\n        run: |\n          if [ \"$FULL_EVAL\" = \"true\" ]; then\n            DEBUG_FIELD=\"\"\n          else\n            DEBUG_FIELD='\"debug\": { \"enabled\": true, \"firstN\": 2 },'\n          fi\n\n          PAYLOAD=$(cat <<EOF\n          {\n            \"mode\": \"baseline\",\n            ${DEBUG_FIELD}\n            \"publish\": true,\n            \"compare\": true\n          }\n          EOF\n          )\n\n          RESPONSE=$(curl -sf -X POST \\\n            -H \"Authorization: Bearer $AILF_API_KEY\" \\\n            -H \"Content-Type: application/json\" \\\n            https://ailf-api.sanity.build/v1/pipeline \\\n            -d \"$PAYLOAD\")\n\n          JOB_ID=$(echo \"$RESPONSE\" | jq -r '.jobId')\n          echo \"job_id=$JOB_ID\" >> $GITHUB_OUTPUT\n          echo \"\uD83D\uDCCB Submitted job: $JOB_ID\"\n\n      # \u2500\u2500\u2500 Poll for results (long-polling) \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Wait for results\n        id: results\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n        run: |\n          for i in $(seq 1 40); do\n            RESPONSE=$(curl -s \\\n              -H \"Authorization: Bearer $AILF_API_KEY\" \\\n              -H \"Prefer: wait=25\" \\\n              \"https://ailf-api.sanity.build/v1/jobs/$JOB_ID\")\n\n            STATUS=$(echo \"$RESPONSE\" | jq -r '.status')\n\n            case \"$STATUS\" in\n              completed)\n                echo \"status=completed\" >> $GITHUB_OUTPUT\n                echo \"report_id=$(echo $RESPONSE | jq -r '.reportId // empty')\" >> $GITHUB_OUTPUT\n                echo \"score=$(echo $RESPONSE | jq -r '.score // empty')\" >> $GITHUB_OUTPUT\n                echo \"\u2705 Evaluation completed\"\n                exit 0\n                ;;\n              failed|timed-out)\n                echo \"status=$STATUS\" >> $GITHUB_OUTPUT\n                echo \"::error::Evaluation $STATUS\"\n                exit 1\n                ;;\n              *)\n                echo \"\u23F3 [$i/40] $STATUS\"\n                ;;\n            esac\n          done\n\n          echo \"::error::Timed out waiting for evaluation\"\n          exit 1\n\n      # \u2500\u2500\u2500 Post results to PR \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Post PR comment\n        if: >-\n          always() && github.event_name == 'pull_request' &&\n          steps.submit.outputs.job_id != ''\n        uses: actions/github-script@v7\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        with:\n          script: |\n            const marker = '<!-- ailf-score-report -->';\n            const status = process.env.JOB_STATUS;\n            const reportId = process.env.REPORT_ID;\n            const jobId = process.env.JOB_ID;\n            const score = process.env.SCORE;\n\n            let icon, message;\n            if (status === 'completed') {\n              icon = '\u2705';\n              message = score\n                ? `Evaluation completed \u2014 score: **${score}/100**`\n                : 'Evaluation completed successfully.';\n            } else if (status === 'failed' || status === 'timed-out') {\n              icon = '\u26A0\uFE0F';\n              message = `Evaluation ${status}.`;\n            } else {\n              icon = '\u23F3';\n              message = 'Evaluation status unknown (may still be running).';\n            }\n\n            let body = `${marker}\\n## ${icon} AI Literacy Evaluation\\n\\n${message}\\n`;\n            if (reportId) {\n              body += `\\n\uD83D\uDD17 [View detailed report](https://ailf-api.sanity.build/v1/reports/${reportId})\\n`;\n            }\n            body += `\\n<sub>Job: \\`${jobId}\\`</sub>\\n`;\n\n            const { data: comments } = await github.rest.issues.listComments({\n              owner: context.repo.owner,\n              repo: context.repo.repo,\n              issue_number: context.issue.number,\n            });\n            const existing = comments.find(c => c.body?.includes(marker));\n\n            if (existing) {\n              await github.rest.issues.updateComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                comment_id: existing.id,\n                body,\n              });\n            } else {\n              await github.rest.issues.createComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                issue_number: context.issue.number,\n                body,\n              });\n            }\n\n      # \u2500\u2500\u2500 Job summary \u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\u2500\n      - name: Summary\n        if: always()\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        run: |\n          {\n            echo \"## \uD83D\uDCCA AI Literacy Evaluation\"\n            echo \"\"\n            echo \"| Field | Value |\"\n            echo \"|-------|-------|\"\n            echo \"| Job | \\`$JOB_ID\\` |\"\n            echo \"| Status | $JOB_STATUS |\"\n            [ -n \"$SCORE\" ] && echo \"| Score | $SCORE/100 |\"\n            [ -n \"$REPORT_ID\" ] && echo \"| Report | [$REPORT_ID](https://ailf-api.sanity.build/v1/reports/$REPORT_ID) |\"\n          } >> \"$GITHUB_STEP_SUMMARY\"\n";

package/dist/_vendor/ailf-core/examples/index.js CHANGED Viewed

@@ -119,9 +119,9 @@ export const thresholdYaml = "# Example quality threshold configuration.\n#\n# T
 /** Parsed ailf-config example data (JSON-safe) */
 export const ailfConfigData = {
     "source": {
-        "projectId": "your-project-id",
-        "dataset": "production",
-        "baseUrl": "https://your-site.example.com/docs"
+        "projectId": "3do82whm",
+        "dataset": "next",
+        "baseUrl": "https://www.sanity.io/docs"
     },
     "triggers": {
         "pr": {
@@ -141,7 +141,7 @@ export const ailfConfigData = {
     }
 };
 /** Raw YAML string for ailf-config example (preserves comments) */
-export const ailfConfigYaml = "# ──────────────────────────────────────────────────────────────────────\n# .ailf/config.yaml — AI Literacy Framework project configuration\n# ──────────────────────────────────────────────────────────────────────\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# ──────────────────────────────────────────────────────────────────────\n\n# Documentation source — where to fetch content for evaluation.\n#\n# projectId — your Sanity project ID (find it in sanity.io/manage)\n# dataset   — the dataset to query (e.g., \"production\", \"staging\")\n# baseUrl   — the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"your-project-id\"\n  dataset: production\n  baseUrl: \"https://your-site.example.com/docs\"\n\n# Trigger configuration — when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only — check that task YAML parses correctly (fast, no LLM calls)\n#   eval          — run the full evaluation pipeline\n#\n# paths — only trigger when files matching these globs change\n# blocking — if true, a failing eval blocks the PR merge\n# notify — if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
+export const ailfConfigYaml = "# ──────────────────────────────────────────────────────────────────────\n# .ailf/config.yaml — AI Literacy Framework project configuration\n# ──────────────────────────────────────────────────────────────────────\n#\n# This file configures how the AILF evaluation pipeline runs in this\n# repository. Place it at .ailf/config.yaml in your project root.\n#\n# Evaluations are submitted to the AILF API (ailf-api.sanity.build).\n# The API handles LLM calls, doc fetching, grading, and report\n# publishing. Your repo only needs one secret: AILF_API_KEY.\n#\n# Docs: https://github.com/sanity-io/ai-literacy-framework\n# ──────────────────────────────────────────────────────────────────────\n\n# Documentation source — which docs are being evaluated.\n#\n# This tells the pipeline which Sanity project and dataset contain\n# the documentation under test. For most users, this is Sanity's own\n# docs project.\n#\n# projectId — Sanity project ID (find yours at sanity.io/manage)\n# dataset   — the dataset to query (e.g., \"production\", \"next\")\n# baseUrl   — the public URL of your documentation site\n#             (used by agentic mode to test agent discoverability)\nsource:\n  projectId: \"3do82whm\"\n  dataset: next\n  baseUrl: \"https://www.sanity.io/docs\"\n\n# Trigger configuration — when evaluations run automatically.\n#\n# Each key is a trigger context. The pipeline checks which trigger\n# matches the current execution context (PR, merge, schedule, etc.)\n# and applies its settings.\n#\n# mode options:\n#   validate-only — check that task YAML parses correctly (fast, no LLM calls)\n#   eval          — run the full evaluation pipeline\n#\n# paths — only trigger when files matching these globs change\n# blocking — if true, a failing eval blocks the PR merge\n# notify — if true, post results to configured notification channels\ntriggers:\n  # On pull requests: just validate task files parse correctly\n  pr:\n    mode: validate-only\n\n  # When .ailf/ files change in a PR: run a real evaluation\n  pr-task-change:\n    mode: eval\n    paths: [\".ailf/**\"]\n\n  # On merge to main: run evaluation (non-blocking)\n  main:\n    mode: eval\n    blocking: false\n    notify: true\n";
 /** Parsed task data for example-groq-blog-listing (JSON-safe) */
 export const exampleGroqBlogListingData = [
     {
@@ -283,3 +283,8 @@ export const EXAMPLES = {
         yaml: Object.values(taskYamlFiles).join("\n"),
     },
 };
+// ---------------------------------------------------------------------------
+// Raw file exports (non-data files, exported as raw strings)
+// ---------------------------------------------------------------------------
+/** GitHub Actions workflow template for AI Literacy evaluation */
+export const workflowYaml = "# ──────────────────────────────────────────────────────────────────────\n# AI Literacy Evaluation — GitHub Actions workflow\n# ──────────────────────────────────────────────────────────────────────\n#\n# This workflow submits evaluations to the AILF API when task or config\n# files change in a pull request. The API handles all processing\n# (LLM calls, doc fetching, grading, report publishing).\n#\n# Prerequisites:\n#   Add one secret to your repository (Settings → Secrets → Actions):\n#     AILF_API_KEY — your API key (starts with ailf_live_sk_)\n#\n# Customization:\n#   - Adjust `paths` to match your documentation file locations\n#   - Set full_eval to true for comprehensive (slower) evaluation\n#   - See: https://github.com/sanity-labs/ai-literacy-framework/blob/main/docs/API_GATEWAY.md\n# ──────────────────────────────────────────────────────────────────────\n\nname: AI Literacy Eval\n\non:\n  pull_request:\n    branches: [main]\n    paths:\n      - \".ailf/**\"\n\n  # Manual trigger from the Actions tab\n  workflow_dispatch:\n    inputs:\n      full_eval:\n        description: \"Run full evaluation (all tests, slower)\"\n        type: boolean\n        default: false\n\nconcurrency:\n  group: ailf-eval-${{ github.event.pull_request.number || github.ref }}\n  cancel-in-progress: true\n\njobs:\n  evaluate:\n    name: AI Literacy Evaluation\n    runs-on: ubuntu-latest\n    permissions:\n      pull-requests: write\n    steps:\n      # ─── Submit evaluation to the AILF API ─────────────────────\n      - name: Submit evaluation\n        id: submit\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          FULL_EVAL: ${{ inputs.full_eval || 'false' }}\n        run: |\n          if [ \"$FULL_EVAL\" = \"true\" ]; then\n            DEBUG_FIELD=\"\"\n          else\n            DEBUG_FIELD='\"debug\": { \"enabled\": true, \"firstN\": 2 },'\n          fi\n\n          PAYLOAD=$(cat <<EOF\n          {\n            \"mode\": \"baseline\",\n            ${DEBUG_FIELD}\n            \"publish\": true,\n            \"compare\": true\n          }\n          EOF\n          )\n\n          RESPONSE=$(curl -sf -X POST \\\n            -H \"Authorization: Bearer $AILF_API_KEY\" \\\n            -H \"Content-Type: application/json\" \\\n            https://ailf-api.sanity.build/v1/pipeline \\\n            -d \"$PAYLOAD\")\n\n          JOB_ID=$(echo \"$RESPONSE\" | jq -r '.jobId')\n          echo \"job_id=$JOB_ID\" >> $GITHUB_OUTPUT\n          echo \"📋 Submitted job: $JOB_ID\"\n\n      # ─── Poll for results (long-polling) ───────────────────────\n      - name: Wait for results\n        id: results\n        env:\n          AILF_API_KEY: ${{ secrets.AILF_API_KEY }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n        run: |\n          for i in $(seq 1 40); do\n            RESPONSE=$(curl -s \\\n              -H \"Authorization: Bearer $AILF_API_KEY\" \\\n              -H \"Prefer: wait=25\" \\\n              \"https://ailf-api.sanity.build/v1/jobs/$JOB_ID\")\n\n            STATUS=$(echo \"$RESPONSE\" | jq -r '.status')\n\n            case \"$STATUS\" in\n              completed)\n                echo \"status=completed\" >> $GITHUB_OUTPUT\n                echo \"report_id=$(echo $RESPONSE | jq -r '.reportId // empty')\" >> $GITHUB_OUTPUT\n                echo \"score=$(echo $RESPONSE | jq -r '.score // empty')\" >> $GITHUB_OUTPUT\n                echo \"✅ Evaluation completed\"\n                exit 0\n                ;;\n              failed|timed-out)\n                echo \"status=$STATUS\" >> $GITHUB_OUTPUT\n                echo \"::error::Evaluation $STATUS\"\n                exit 1\n                ;;\n              *)\n                echo \"⏳ [$i/40] $STATUS\"\n                ;;\n            esac\n          done\n\n          echo \"::error::Timed out waiting for evaluation\"\n          exit 1\n\n      # ─── Post results to PR ────────────────────────────────────\n      - name: Post PR comment\n        if: >-\n          always() && github.event_name == 'pull_request' &&\n          steps.submit.outputs.job_id != ''\n        uses: actions/github-script@v7\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        with:\n          script: |\n            const marker = '<!-- ailf-score-report -->';\n            const status = process.env.JOB_STATUS;\n            const reportId = process.env.REPORT_ID;\n            const jobId = process.env.JOB_ID;\n            const score = process.env.SCORE;\n\n            let icon, message;\n            if (status === 'completed') {\n              icon = '✅';\n              message = score\n                ? `Evaluation completed — score: **${score}/100**`\n                : 'Evaluation completed successfully.';\n            } else if (status === 'failed' || status === 'timed-out') {\n              icon = '⚠️';\n              message = `Evaluation ${status}.`;\n            } else {\n              icon = '⏳';\n              message = 'Evaluation status unknown (may still be running).';\n            }\n\n            let body = `${marker}\\n## ${icon} AI Literacy Evaluation\\n\\n${message}\\n`;\n            if (reportId) {\n              body += `\\n🔗 [View detailed report](https://ailf-api.sanity.build/v1/reports/${reportId})\\n`;\n            }\n            body += `\\n<sub>Job: \\`${jobId}\\`</sub>\\n`;\n\n            const { data: comments } = await github.rest.issues.listComments({\n              owner: context.repo.owner,\n              repo: context.repo.repo,\n              issue_number: context.issue.number,\n            });\n            const existing = comments.find(c => c.body?.includes(marker));\n\n            if (existing) {\n              await github.rest.issues.updateComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                comment_id: existing.id,\n                body,\n              });\n            } else {\n              await github.rest.issues.createComment({\n                owner: context.repo.owner,\n                repo: context.repo.repo,\n                issue_number: context.issue.number,\n                body,\n              });\n            }\n\n      # ─── Job summary ───────────────────────────────────────────\n      - name: Summary\n        if: always()\n        env:\n          JOB_STATUS: ${{ steps.results.outputs.status || 'unknown' }}\n          REPORT_ID: ${{ steps.results.outputs.report_id || '' }}\n          JOB_ID: ${{ steps.submit.outputs.job_id }}\n          SCORE: ${{ steps.results.outputs.score || '' }}\n        run: |\n          {\n            echo \"## 📊 AI Literacy Evaluation\"\n            echo \"\"\n            echo \"| Field | Value |\"\n            echo \"|-------|-------|\"\n            echo \"| Job | \\`$JOB_ID\\` |\"\n            echo \"| Status | $JOB_STATUS |\"\n            [ -n \"$SCORE\" ] && echo \"| Score | $SCORE/100 |\"\n            [ -n \"$REPORT_ID\" ] && echo \"| Report | [$REPORT_ID](https://ailf-api.sanity.build/v1/reports/$REPORT_ID) |\"\n          } >> \"$GITHUB_STEP_SUMMARY\"\n";

package/dist/_vendor/ailf-core/ports/context.d.ts CHANGED Viewed

@@ -95,6 +95,10 @@ export interface ResolvedConfig {
     taskSourceType?: "content-lake" | "yaml";
     /** Path to repo-based tasks directory (e.g., .ailf/tasks/) */
     repoTasksPath?: string;
+    /** Report store project ID from .ailf/config.yaml reportStore block */
+    reportStoreProjectId?: string;
+    /** Report store dataset from .ailf/config.yaml reportStore block */
+    reportStoreDataset?: string;
     /** Callback URL configuration for API-triggered evaluations */
     callback?: {
         url: string;

package/dist/adapters/task-sources/repo-schemas.d.ts CHANGED Viewed

@@ -185,10 +185,20 @@ export declare const RepoTaskFileSchema: z.ZodArray<z.ZodObject<{
     }, z.core.$strip>>;
 }, z.core.$strip>>;
 /**
- * Zod schema for .ailf/config.yaml — controls how and when evaluations
- * are triggered from an external repository.
+ * Zod schema for .ailf/config.yaml — controls documentation source,
+ * report destination, and trigger behavior for evaluations from an
+ * external repository.
  */
 export declare const RepoConfigSchema: z.ZodObject<{
+    source: z.ZodOptional<z.ZodObject<{
+        projectId: z.ZodOptional<z.ZodString>;
+        dataset: z.ZodOptional<z.ZodString>;
+        baseUrl: z.ZodOptional<z.ZodString>;
+    }, z.core.$strip>>;
+    reportStore: z.ZodOptional<z.ZodObject<{
+        projectId: z.ZodString;
+        dataset: z.ZodString;
+    }, z.core.$strip>>;
     triggers: z.ZodOptional<z.ZodObject<{
         pr: z.ZodOptional<z.ZodObject<{
             mode: z.ZodDefault<z.ZodEnum<{

package/dist/adapters/task-sources/repo-schemas.js CHANGED Viewed

@@ -189,10 +189,36 @@ const ScheduleTriggerSchema = TriggerConfigSchema.extend({
     cron: z.string().min(1),
 });
 /**
- * Zod schema for .ailf/config.yaml — controls how and when evaluations
- * are triggered from an external repository.
+ * Documentation source configuration.
+ * Defines which Sanity project holds the documentation being evaluated.
+ */
+const SourceConfigSchema = z
+    .object({
+    projectId: z.string().min(1).optional(),
+    dataset: z.string().min(1).optional(),
+    baseUrl: z.string().url().optional(),
+})
+    .optional();
+/**
+ * Report store configuration.
+ * Defines which Sanity project receives `ailf.report` documents.
+ * This should match the project/dataset configured in the user's Studio.
+ * The API token comes from the AILF_REPORT_SANITY_API_TOKEN env var.
+ */
+const ReportStoreConfigSchema = z
+    .object({
+    projectId: z.string().min(1),
+    dataset: z.string().min(1),
+})
+    .optional();
+/**
+ * Zod schema for .ailf/config.yaml — controls documentation source,
+ * report destination, and trigger behavior for evaluations from an
+ * external repository.
  */
 export const RepoConfigSchema = z.object({
+    source: SourceConfigSchema,
+    reportStore: ReportStoreConfigSchema,
     triggers: z
         .object({
         pr: TriggerConfigSchema.optional(),

package/dist/cli.js CHANGED Viewed

File without changes

package/dist/commands/init.js CHANGED Viewed

@@ -18,7 +18,7 @@
 import { Command } from "commander";
 import { existsSync, mkdirSync, writeFileSync } from "fs";
 import { resolve, relative } from "path";
-import { ailfConfigData, ailfConfigYaml, taskYamlFiles, TASK_FILE_NAMES, allTaskData, } from "../_vendor/ailf-core/index.js";
+import { ailfConfigData, ailfConfigYaml, taskYamlFiles, TASK_FILE_NAMES, allTaskData, workflowYaml, } from "../_vendor/ailf-core/index.js";
 // ---------------------------------------------------------------------------
 // Command factory
 // ---------------------------------------------------------------------------
@@ -127,7 +127,40 @@ async function runInit(opts) {
     else {
         skipped.push(rel(targetDir, gitignorePath));
     }
-    // 5. Summary
+    // 5. Write GitHub Actions workflow
+    const workflowDir = resolve(targetDir, ".github", "workflows");
+    const workflowPath = resolve(workflowDir, "ailf-eval.yml");
+    mkdirSync(workflowDir, { recursive: true });
+    if (writeIfNew(workflowPath, workflowYaml, force)) {
+        written.push(rel(targetDir, workflowPath));
+    }
+    else {
+        skipped.push(rel(targetDir, workflowPath));
+    }
+    // 6. Write .env.example (secrets template — never committed)
+    const envExamplePath = resolve(targetDir, ".env.example");
+    const envExampleContent = `# ═══════════════════════════════════════════════════════════════════
+# AI Literacy Framework — Environment Variables
+# ═══════════════════════════════════════════════════════════════════
+# Copy this file to .env and fill in your values:
+#   cp .env.example .env
+#
+# IMPORTANT: Never commit .env to version control.
+# ═══════════════════════════════════════════════════════════════════
+# ─── AILF API Key (required) ─────────────────────────────────────
+# Authenticates requests to the AILF API (ailf-api.sanity.build).
+# The API handles LLM calls, doc fetching, grading, and publishing.
+# Request a key from the AILF team.
+AILF_API_KEY=ailf_live_sk_...
+`;
+    if (writeIfNew(envExamplePath, envExampleContent, force)) {
+        written.push(rel(targetDir, envExamplePath));
+    }
+    else {
+        skipped.push(rel(targetDir, envExamplePath));
+    }
+    // 7. Summary
     console.log();
     if (written.length > 0) {
         for (const f of written) {
@@ -143,8 +176,9 @@ async function runInit(opts) {
     console.log();
     console.log("  Next steps:");
     console.log();
-    console.log(`  1. Edit ${rel(targetDir, resolve(ailfDir, `config${ext}`))} with your Sanity project settings`);
-    console.log(`  2. Customize the example tasks in ${rel(targetDir, tasksDir)}/`);
-    console.log("  3. Run: ailf pipeline --repo-tasks-path .ailf/tasks/");
+    console.log(`  1. Customize the example tasks in ${rel(targetDir, tasksDir)}/`);
+    console.log("  2. Validate: npx @sanity/ailf validate-tasks .ailf/tasks/");
+    console.log("  3. Add AILF_API_KEY as a GitHub Actions secret (Settings → Secrets)");
+    console.log("  4. Push — the workflow at .github/workflows/ailf-eval.yml handles the rest");
     console.log();
 }

package/dist/commands/pipeline-action.js CHANGED Viewed

@@ -10,7 +10,7 @@
  *
  * @see packages/eval/src/orchestration/ for the step-based pipeline
  */
-import { writeFileSync } from "fs";
+import { existsSync, readFileSync, writeFileSync } from "fs";
 import { dirname, resolve } from "path";
 import { fileURLToPath } from "url";
 import { classifyUrls } from "../pipeline/classify-url.js";
@@ -18,6 +18,8 @@ import { assessImpact, buildReverseMapping, } from "../pipeline/reverse-mapping.
 import { buildAppContext } from "../orchestration/build-app-context.js";
 import { buildStepSequence } from "../orchestration/build-step-sequence.js";
 import { orchestratePipeline } from "../orchestration/pipeline-orchestrator.js";
+import { load } from "js-yaml";
+import { parseRepoConfig, } from "../adapters/task-sources/repo-schemas.js";
 const __dirname = dirname(fileURLToPath(import.meta.url));
 const ROOT = resolve(__dirname, "..", "..");
 // ---------------------------------------------------------------------------
@@ -32,6 +34,8 @@ const VALID_SEARCH_MODES = ["open", "origin-only", "off"];
  * Exported so the plan builder can call it independently.
  */
 export function computeResolvedOptions(opts) {
+    // Resolve paths relative to the caller's cwd, not the eval package root
+    const callerCwd = process.env.AILF_CALLER_CWD ?? process.cwd();
     // Validate mode
     const mode = opts.mode;
     if (!VALID_MODES.includes(mode)) {
@@ -163,14 +167,21 @@ export function computeResolvedOptions(opts) {
         // Smart default: full runs auto-publish when store is configured
         publishEnabled = reportStoreConfigured && !debugEnabled;
     }
-    // Report store overrides — fall back to the eval dataset so that
-    // perspective evaluations publish reports to the same dataset the
-    // Studio is reading from. AILF_REPORT_DATASET wins when set explicitly.
+    // Report store overrides — resolution order:
+    //   1. Explicit CLI flags (--report-dataset, --report-project)
+    //   2. Environment variables (AILF_REPORT_DATASET, AILF_REPORT_PROJECT_ID)
+    //   3. .ailf/config.yaml reportStore block (when --repo-tasks-path is set)
+    //   4. Eval dataset override (so perspective evals publish to the same dataset)
+    const repoConfig = loadRepoConfigIfPresent(opts.repoTasksPath);
     const reportDataset = opts.reportDataset ??
         process.env.AILF_REPORT_DATASET ??
+        repoConfig?.reportStore?.dataset ??
         datasetOverride ??
         undefined;
-    const reportProjectId = opts.reportProject ?? process.env.AILF_REPORT_PROJECT_ID ?? undefined;
+    const reportProjectId = opts.reportProject ??
+        process.env.AILF_REPORT_PROJECT_ID ??
+        repoConfig?.reportStore?.projectId ??
+        undefined;
     return {
         allowedOriginArgs,
         areaOption,
@@ -206,7 +217,9 @@ export function computeResolvedOptions(opts) {
         skipFetch: opts.skipFetch,
         source: opts.source,
         studioOriginOverride,
-        repoTasksPath: opts.repoTasksPath,
+        repoTasksPath: opts.repoTasksPath
+            ? resolve(callerCwd, opts.repoTasksPath)
+            : undefined,
         taskOption,
         taskSourceType: resolveTaskSourceType(opts.taskSource),
         urlArgs,
@@ -303,3 +316,28 @@ function writePipelineResult(result) {
         // results/latest/ may not exist yet — not critical
     }
 }
+/**
+ * Load .ailf/config.yaml if --repo-tasks-path is set and the config file
+ * exists. Returns null if not applicable.
+ *
+ * The config.yaml lives one level up from the tasks/ directory:
+ *   .ailf/config.yaml  ← config
+ *   .ailf/tasks/       ← repoTasksPath
+ */
+function loadRepoConfigIfPresent(repoTasksPath) {
+    if (!repoTasksPath)
+        return null;
+    // .ailf/tasks/ → .ailf/config.yaml
+    const configPath = resolve(repoTasksPath, "..", "config.yaml");
+    if (!existsSync(configPath))
+        return null;
+    try {
+        const raw = readFileSync(configPath, "utf-8");
+        const parsed = load(raw);
+        return parseRepoConfig(parsed);
+    }
+    catch (err) {
+        console.warn(`  ⚠️  Failed to parse ${configPath}: ${err instanceof Error ? err.message : String(err)}`);
+        return null;
+    }
+}

package/dist/commands/publish.js CHANGED Viewed

@@ -101,7 +101,8 @@ async function runPublishCommand(summaryPath, opts) {
     // -----------------------------------------------------------------------
     // 1. Resolve and read the score summary
     // -----------------------------------------------------------------------
-    const resolvedPath = resolve(summaryPath);
+    const callerCwd = process.env.AILF_CALLER_CWD ?? process.cwd();
+    const resolvedPath = resolve(callerCwd, summaryPath);
     if (!existsSync(resolvedPath)) {
         console.error(`  ✖ File not found: ${resolvedPath}`);
         console.error();

package/dist/commands/validate-tasks.js CHANGED Viewed

@@ -24,7 +24,10 @@ export function createValidateTasksCommand() {
         .argument("[path]", "Path to tasks directory (default: .ailf/tasks/)", ".ailf/tasks")
         .option("--strict", "Treat warnings as errors", false)
         .action(async (tasksPath, opts) => {
-        const resolvedPath = resolve(tasksPath);
+        // Resolve relative to the caller's working directory, not the
+        // eval package root (which differs when run via bin/ailf.js)
+        const callerCwd = process.env.AILF_CALLER_CWD ?? process.cwd();
+        const resolvedPath = resolve(callerCwd, tasksPath);
         if (!existsSync(resolvedPath)) {
             console.error(`❌ Directory not found: ${resolvedPath}`);
             process.exit(1);

package/dist/composition-root.js CHANGED Viewed

@@ -43,7 +43,7 @@ export function createAppContext(config) {
     // Eval runner — Promptfoo subprocess
     const evalRunner = new PromptfooEvalAdapter(config.rootDir);
     // Report store — Sanity Content Lake (for publish + auto-compare)
-    const reportStore = createReportStore();
+    const reportStore = createReportStore(config);
     // Sinks — loaded from config/sinks.yaml
     const sinks = loadSinks();
     return {
@@ -75,7 +75,7 @@ function createCache(config) {
     const token = process.env.AILF_REPORT_SANITY_API_TOKEN ?? process.env.SANITY_API_TOKEN;
     if (!token)
         return local;
-    return new ContentLakeCacheAdapter(local, createReportStore());
+    return new ContentLakeCacheAdapter(local, createReportStore(config));
 }
 function createTaskSource(config) {
     // Primary source — selected by config.taskSourceType
@@ -96,10 +96,14 @@ function createTaskSource(config) {
     }
     return primary;
 }
-function createReportStore() {
+function createReportStore(config) {
     return new ReportStore({
-        dataset: process.env.AILF_REPORT_DATASET ?? undefined,
-        projectId: process.env.AILF_REPORT_PROJECT_ID ?? undefined,
+        dataset: process.env.AILF_REPORT_DATASET ??
+            config?.reportStoreDataset ??
+            undefined,
+        projectId: process.env.AILF_REPORT_PROJECT_ID ??
+            config?.reportStoreProjectId ??
+            undefined,
         token: process.env.AILF_REPORT_SANITY_API_TOKEN ??
             process.env.SANITY_API_TOKEN ??
             undefined,

package/dist/orchestration/build-app-context.js CHANGED Viewed

@@ -67,6 +67,8 @@ export function mapToResolvedConfig(opts, rootDir) {
         beforeOption: opts.beforeOption,
         taskSourceType: opts.taskSourceType,
         repoTasksPath: opts.repoTasksPath,
+        reportStoreProjectId: opts.reportProjectId,
+        reportStoreDataset: opts.reportDataset,
     };
 }
 /**

package/package.json CHANGED Viewed

@@ -1,6 +1,6 @@
 {
   "name": "@sanity/ailf",
-  "version": "0.1.0",
+  "version": "0.1.1",
   "private": false,
   "publishConfig": {
     "access": "restricted"

package/dist/commands/update-quality-scores.d.ts DELETED Viewed

@@ -1,5 +0,0 @@
-/**
- * update-quality-scores command — update QUALITY_SCORE.md from scores.
- */
-import { Command } from "commander";
-export declare function createUpdateQualityScoresCommand(): Command;

package/dist/commands/update-quality-scores.js DELETED Viewed

@@ -1,20 +0,0 @@
-/**
- * update-quality-scores command — update QUALITY_SCORE.md from scores.
- */
-import { Command } from "commander";
-export function createUpdateQualityScoresCommand() {
-    return new Command("update-quality-scores")
-        .description("Update docs/QUALITY_SCORE.md from score-summary.json")
-        .action(async () => {
-        const { updateQualityScores } = await import("../scripts/update-quality-scores.js");
-        console.log("=== Updating QUALITY_SCORE.md from score-summary.json ===\n");
-        const result = updateQualityScores();
-        if (result.success) {
-            console.log(`  ✅ ${result.message}`);
-        }
-        else {
-            console.error(`  ❌ ${result.message}`);
-            process.exit(1);
-        }
-    });
-}

package/dist/lib/agent-behavior-report.d.ts DELETED Viewed

@@ -1,8 +0,0 @@
-/**
- * lib/agent-behavior-report.ts — DEPRECATED re-export shim.
- * @deprecated Import from ../pipeline/agent-behavior-report.js instead.
- */
-import "dotenv/config";
-export { analyzeResults, CANONICAL_DOC_MAP, detectFeatureArea, } from "../pipeline/agent-behavior-report.js";
-export type { AnalysisResult, FeatureAnalysis, TaskBehavior, TestResult, } from "../pipeline/agent-behavior-report.js";
-export declare function main(resultsPathArg?: string): void;