akm-cli 0.9.0-rc.8 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +1063 -44
- package/README.md +51 -25
- package/SECURITY.md +14 -1
- package/STABILITY.md +497 -0
- package/dist/akm +148 -35
- package/dist/{akm-migrate-storage → akm-migrate} +6 -9
- package/dist/assets/hints/cli-hints-full.md +223 -95
- package/dist/assets/hints/cli-hints-short.md +85 -22
- package/dist/assets/improve-strategies/default.json +1 -1
- package/dist/assets/improve-strategies/reflect-distill.json +1 -1
- package/dist/assets/prompts/memory-infer-user.md +2 -3
- package/dist/assets/stash-skeleton/README.md +6 -5
- package/dist/assets/stash-skeleton/facts/conventions/assets/agent.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/command.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/fact.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/knowledge.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/lesson.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/memory.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/script.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/skill.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/assets/workflow.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/backlinks.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/domains.md +2 -0
- package/dist/assets/stash-skeleton/facts/conventions/organization.md +20 -9
- package/dist/assets/tasks/core/extract.yml +1 -1
- package/dist/assets/tasks/core/version-check.yml +1 -1
- package/dist/assets/tasks/improve/akm-graph-refresh-weekly.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-catchup.yml +8 -0
- package/dist/assets/tasks/improve/akm-improve-consolidate.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-frequent.yml +5 -0
- package/dist/assets/tasks/improve/akm-improve-nightly.yml +5 -0
- package/dist/assets/templates/html/health.html +1 -3
- package/dist/assets/workflows/workflow-template.md +32 -15
- package/dist/cli/invocation.js +40 -15
- package/dist/cli/parse-args.js +0 -22
- package/dist/cli/retired-commands.js +121 -0
- package/dist/cli/shared.js +154 -22
- package/dist/cli/unknown-flags.js +236 -0
- package/dist/cli-node.mjs +2 -1
- package/dist/cli.js +696 -258
- package/dist/commands/agent/agent-dispatch.js +14 -3
- package/dist/commands/agent/contribute-cli.js +73 -88
- package/dist/commands/completions.js +79 -22
- package/dist/commands/config-cli.js +17 -150
- package/dist/commands/env/env-cli.js +59 -143
- package/dist/commands/env/env.js +12 -163
- package/dist/commands/env/marker-path.js +6 -0
- package/dist/commands/env/secret-cli.js +36 -66
- package/dist/commands/env/secret.js +24 -57
- package/dist/commands/feedback-cli.js +141 -87
- package/dist/commands/health/accept-rate.js +58 -0
- package/dist/commands/health/advisories.js +3 -4
- package/dist/commands/health/checks.js +85 -23
- package/dist/commands/health/html-report.js +7 -10
- package/dist/commands/health/improve-metrics.js +25 -83
- package/dist/commands/health/md-report.js +5 -9
- package/dist/commands/health/metrics.js +62 -20
- package/dist/commands/health/renderers.js +47 -0
- package/dist/commands/health/report-view-model.js +4 -5
- package/dist/commands/health/stash-exposure.js +1 -1
- package/dist/commands/health/surfaces.js +3 -48
- package/dist/commands/health/task-runs.js +3 -67
- package/dist/commands/health/types-improve.js +7 -0
- package/dist/commands/health.js +99 -28
- package/dist/commands/improve/anti-collapse.js +2 -2
- package/dist/commands/improve/autonomy-gate.js +68 -0
- package/dist/commands/improve/collapse-detector.js +41 -40
- package/dist/commands/improve/consolidate/eligibility.js +1 -23
- package/dist/commands/improve/consolidate/merge.js +4 -0
- package/dist/commands/improve/consolidate.js +140 -1000
- package/dist/commands/improve/distill/promote-memory.js +12 -12
- package/dist/commands/improve/distill/quality-gate.js +6 -6
- package/dist/commands/improve/distill.js +58 -69
- package/dist/commands/improve/eligibility.js +105 -57
- package/dist/commands/improve/extract-cli.js +14 -133
- package/dist/commands/improve/improve-cli.js +98 -114
- package/dist/commands/improve/improve-result-file.js +1 -28
- package/dist/commands/improve/improve-strategies.js +8 -5
- package/dist/commands/improve/improve.js +128 -91
- package/dist/commands/improve/loop-stages.js +182 -20
- package/dist/commands/improve/memory/derived-ref.js +45 -43
- package/dist/commands/improve/memory/memory-belief.js +1 -1
- package/dist/commands/improve/memory/memory-contradiction-detect.js +4 -12
- package/dist/commands/improve/memory/memory-improve.js +6 -5
- package/dist/commands/improve/outcome-loop.js +22 -65
- package/dist/commands/improve/preparation.js +114 -123
- package/dist/commands/improve/proactive-maintenance.js +2 -5
- package/dist/commands/improve/reflect.js +56 -160
- package/dist/commands/improve/salience.js +11 -122
- package/dist/commands/improve/source-identity.js +10 -38
- package/dist/commands/lint/base-linter.js +20 -124
- package/dist/commands/lint/env-key-rules.js +31 -47
- package/dist/commands/lint/index.js +249 -43
- package/dist/commands/{events.js → log.js} +33 -38
- package/dist/commands/migrate-cli.js +92 -12
- package/dist/commands/migration-tool.js +46 -0
- package/dist/commands/observability-cli.js +70 -209
- package/dist/commands/proposal/drain.js +101 -29
- package/dist/commands/proposal/proposal-cli.js +76 -48
- package/dist/commands/proposal/proposal.js +54 -18
- package/dist/commands/proposal/propose-cli.js +88 -0
- package/dist/commands/proposal/propose.js +23 -15
- package/dist/commands/proposal/repository.js +701 -278
- package/dist/commands/proposal/validators/proposal-quality-validators.js +2 -8
- package/dist/commands/proposal/validators/proposal-validators.js +55 -7
- package/dist/commands/proposal/validators/proposals.js +4 -7
- package/dist/commands/read/curate.js +34 -53
- package/dist/commands/read/knowledge.js +150 -95
- package/dist/commands/read/registry-search.js +2 -2
- package/dist/commands/read/remember-cli.js +42 -15
- package/dist/commands/read/search-cli.js +180 -78
- package/dist/commands/read/search.js +58 -43
- package/dist/commands/read/show.js +197 -141
- package/dist/commands/registry-cli.js +12 -51
- package/dist/commands/remember.js +14 -57
- package/dist/commands/sources/add-cli.js +100 -31
- package/dist/commands/sources/bundle-cli.js +166 -0
- package/dist/commands/sources/bundle-config-ops.js +7 -2
- package/dist/commands/sources/info.js +18 -5
- package/dist/commands/sources/init.js +12 -12
- package/dist/commands/sources/installed-stashes.js +382 -98
- package/dist/commands/sources/schema-repair.js +3 -2
- package/dist/commands/sources/self-update.js +131 -38
- package/dist/commands/sources/source-add.js +72 -17
- package/dist/commands/sources/source-clone.js +129 -45
- package/dist/commands/sources/source-manage.js +43 -23
- package/dist/commands/sources/sources-cli.js +57 -208
- package/dist/commands/sources/stash-cli.js +46 -53
- package/dist/commands/tasks/tasks-cli.js +91 -97
- package/dist/commands/tasks/tasks.js +276 -421
- package/dist/commands/workflow-cli.js +175 -450
- package/dist/core/adapter/adapters/akm-adapter.js +47 -28
- package/dist/core/adapter/adapters/akm-lint.js +42 -27
- package/dist/core/adapter/adapters/akm-metadata.js +15 -44
- package/dist/core/adapter/adapters/akm-task-adapter.js +15 -13
- package/dist/core/adapter/adapters/akm-workflow-adapter.js +55 -71
- package/dist/core/adapter/adapters/dotenv-adapter.js +1 -1
- package/dist/core/adapter/adapters/generic-files-adapter.js +2 -0
- package/dist/core/adapter/adapters/index.js +6 -6
- package/dist/core/adapter/adapters/llm-wiki-adapter.js +14 -8
- package/dist/core/adapter/adapters/okf-adapter.js +187 -19
- package/dist/core/adapter/adapters/shared.js +3 -19
- package/dist/core/adapter/adapters/tool-dir-shared.js +8 -3
- package/dist/core/adapter/adapters/website-snapshot-adapter.js +1 -0
- package/dist/core/adapter/detect-adapter.js +17 -0
- package/dist/core/adapter/recognize-match.js +6 -4
- package/dist/core/adapter/validate-context.js +214 -0
- package/dist/core/asset/akm-markdown.js +63 -0
- package/dist/core/asset/asset-placement.js +20 -6
- package/dist/core/asset/asset-ref.js +11 -9
- package/dist/core/asset/frontmatter-lint.js +30 -0
- package/dist/core/asset/frontmatter.js +37 -9
- package/dist/core/asset/markdown.js +40 -51
- package/dist/core/asset/resolve-ref.js +89 -18
- package/dist/core/asset/stash-meta.js +1 -1
- package/dist/core/bundle-id.js +51 -0
- package/dist/core/common.js +152 -38
- package/dist/core/config/config-io.js +12 -1
- package/dist/core/config/config-schema.js +35 -8
- package/dist/core/config/config-sources.js +55 -11
- package/dist/core/config/config-walker.js +25 -9
- package/dist/core/config/config.js +9 -48
- package/dist/core/config/experimental.js +21 -0
- package/dist/core/config/schema/embedding.js +5 -1
- package/dist/core/config/schema/experimental.js +30 -0
- package/dist/core/config/schema/improve-processes.js +0 -6
- package/dist/core/config/schema/improve.js +21 -3
- package/dist/core/config/schema/index-config.js +8 -15
- package/dist/core/config/schema/output.js +4 -1
- package/dist/core/config/schema/setup.js +9 -18
- package/dist/core/config/schema/sources-bundles.js +49 -33
- package/dist/core/config/schema/workflow.js +3 -3
- package/dist/core/env-secret-ref.js +76 -46
- package/dist/core/errors.js +18 -12
- package/dist/core/events.js +46 -128
- package/dist/core/file-change.js +6 -5
- package/dist/core/fs-txn.js +83 -7
- package/dist/core/git-message.js +2 -2
- package/dist/core/improve-result.js +1 -100
- package/dist/core/lesson-lint.js +1 -17
- package/dist/core/logs-db.js +2 -1
- package/dist/core/migration-operation.js +16 -0
- package/dist/core/mutation-target.js +78 -0
- package/dist/core/parse.js +4 -1
- package/dist/core/paths.js +17 -20
- package/dist/core/recognition-util.js +12 -14
- package/dist/core/redaction.js +34 -0
- package/dist/core/standards/resolve-standards-context.js +2 -14
- package/dist/core/standards/resolve-stash-standards.js +2 -2
- package/dist/core/standards/resolve-type-conventions.js +2 -2
- package/dist/core/state/migrations.js +41 -18
- package/dist/core/state-db.js +5 -14
- package/dist/core/structured.js +1 -1
- package/dist/core/subprocess.js +6 -4
- package/dist/core/text-truncation.js +9 -5
- package/dist/core/type-presentation.js +3 -3
- package/dist/core/warn.js +0 -3
- package/dist/core/write-source.js +771 -95
- package/dist/indexer/bundle-identity-guard.js +3 -2
- package/dist/indexer/db/graph-db.js +0 -24
- package/dist/indexer/ensure-index.js +1 -0
- package/dist/indexer/graph/graph-boost.js +9 -34
- package/dist/indexer/graph/graph-extraction.js +8 -5
- package/dist/indexer/index-writer-lock.js +53 -17
- package/dist/indexer/index-written-assets.js +16 -22
- package/dist/indexer/indexer.js +497 -239
- package/dist/indexer/installations.js +14 -96
- package/dist/indexer/passes/dir-staleness.js +16 -9
- package/dist/indexer/passes/memory-inference.js +11 -9
- package/dist/indexer/passes/metadata.js +113 -47
- package/dist/indexer/scan/doc-to-entry.js +38 -1
- package/dist/indexer/scan/drain-dir.js +13 -23
- package/dist/indexer/search/db-search.js +99 -54
- package/dist/indexer/search/fts-query.js +47 -24
- package/dist/indexer/search/ranking-contributors.js +42 -20
- package/dist/indexer/search/ranking.js +18 -99
- package/dist/indexer/search/search-fields.js +7 -2
- package/dist/indexer/search/search-source.js +82 -93
- package/dist/indexer/usage/usage-events.js +0 -89
- package/dist/indexer/walk/file-context.js +2 -1
- package/dist/indexer/walk/matchers.js +30 -43
- package/dist/indexer/walk/path-resolver.js +7 -2
- package/dist/indexer/walk/walker.js +38 -12
- package/dist/integrations/agent/builders.js +0 -6
- package/dist/integrations/agent/config.js +2 -2
- package/dist/integrations/agent/detect.js +49 -19
- package/dist/integrations/agent/engine-fallback.js +76 -0
- package/dist/integrations/agent/profiles.js +14 -0
- package/dist/integrations/agent/prompts.js +12 -8
- package/dist/integrations/agent/runner-dispatch.js +4 -2
- package/dist/integrations/agent/runner.js +0 -1
- package/dist/integrations/agent/spawn.js +5 -6
- package/dist/integrations/github.js +1 -1
- package/dist/integrations/harnesses/aider/agent-builder.js +6 -4
- package/dist/integrations/harnesses/amazonq/agent-builder.js +7 -4
- package/dist/integrations/harnesses/claude/session-log.js +0 -10
- package/dist/integrations/harnesses/codex/agent-builder.js +5 -2
- package/dist/integrations/harnesses/copilot/agent-builder.js +5 -3
- package/dist/integrations/harnesses/gemini/agent-builder.js +5 -3
- package/dist/integrations/harnesses/index.js +3 -7
- package/dist/integrations/harnesses/opencode/agent-builder.js +21 -2
- package/dist/integrations/harnesses/opencode/session-log.js +0 -15
- package/dist/integrations/harnesses/opencode-sdk/sdk-runner.js +13 -4
- package/dist/integrations/harnesses/openhands/agent-builder.js +9 -6
- package/dist/integrations/harnesses/pi/agent-builder.js +6 -4
- package/dist/integrations/lockfile.js +101 -6
- package/dist/integrations/session-logs/index.js +3 -28
- package/dist/llm/client.js +136 -100
- package/dist/llm/embedders/remote.js +13 -5
- package/dist/llm/feature-gate.js +4 -12
- package/dist/llm/graph-extract.js +5 -11
- package/dist/llm/memory-infer.js +144 -1
- package/dist/llm/metadata-enhance.js +5 -7
- package/dist/llm/structured-call.js +1 -1
- package/dist/llm/usage-persist.js +26 -5
- package/dist/llm/usage-telemetry.js +25 -2
- package/dist/output/cli-hints.js +1 -2
- package/dist/output/context.js +22 -7
- package/dist/output/format-exempt.js +80 -0
- package/dist/output/generic-render.js +259 -0
- package/dist/output/render-registry.js +57 -0
- package/dist/output/renderers.js +14 -36
- package/dist/output/shapes/curate.js +10 -1
- package/dist/output/shapes/events.js +12 -7
- package/dist/output/shapes/helpers.js +56 -83
- package/dist/output/shapes/migrate.js +8 -0
- package/dist/output/shapes/passthrough.js +7 -41
- package/dist/output/shapes/proposal/producer.js +15 -7
- package/dist/output/shapes.js +2 -9
- package/dist/output/text/{init.js → bundle-create.js} +3 -1
- package/dist/output/text/bundle-show.js +7 -0
- package/dist/output/text/command-format.js +164 -96
- package/dist/output/text/env.js +1 -3
- package/dist/output/text/events.js +8 -7
- package/dist/output/text/health-format.js +103 -0
- package/dist/output/text/health.js +7 -0
- package/dist/output/text/helpers.js +10 -8
- package/dist/output/text/lint-format.js +43 -0
- package/dist/output/text/{save.js → lint.js} +2 -2
- package/dist/output/text/migrate.js +88 -0
- package/dist/output/text/proposal/producer.js +4 -2
- package/dist/output/text/proposal-format.js +44 -72
- package/dist/output/text/registry-commands.js +1 -2
- package/dist/output/text/show-directives.js +15 -7
- package/dist/output/text/status-list.js +32 -0
- package/dist/output/text/sync.js +5 -0
- package/dist/output/text/workflow-format.js +24 -203
- package/dist/output/text/workflow.js +1 -7
- package/dist/output/text.js +16 -17
- package/dist/registry/factory.js +4 -6
- package/dist/registry/origin-resolve.js +16 -27
- package/dist/registry/providers/skills-sh.js +3 -3
- package/dist/registry/providers/static-index.js +13 -23
- package/dist/registry/resolve.js +42 -7
- package/dist/registry/semver.js +34 -84
- package/dist/runtime.js +2 -23
- package/dist/scripts/akm-migrate-node.js +60290 -0
- package/dist/scripts/akm-migrate.js +59628 -0
- package/dist/setup/detect.js +42 -15
- package/dist/setup/registry-stash-loader.js +2 -2
- package/dist/setup/setup.js +236 -136
- package/dist/setup/steps/connection.js +7 -9
- package/dist/setup/steps/platforms.js +9 -9
- package/dist/setup/steps/semantic.js +15 -3
- package/dist/setup/steps/sources.js +12 -13
- package/dist/setup/steps/stashdir.js +2 -3
- package/dist/setup/steps/tasks.js +237 -120
- package/dist/sources/freshness.js +1 -1
- package/dist/sources/provider-factory.js +11 -17
- package/dist/sources/providers/filesystem.js +2 -3
- package/dist/sources/providers/git-install.js +278 -34
- package/dist/sources/providers/git-provider.js +25 -23
- package/dist/sources/providers/git-stash.js +395 -106
- package/dist/sources/providers/git.js +2 -2
- package/dist/sources/providers/npm.js +16 -19
- package/dist/sources/providers/provider-utils.js +7 -4
- package/dist/sources/providers/sync-from-ref.js +3 -9
- package/dist/sources/providers/website.js +6 -1
- package/dist/sources/resolve.js +6 -5
- package/dist/sources/snapshot-fetchers/bluesky.js +146 -0
- package/dist/sources/snapshot-fetchers/content-extract.js +566 -0
- package/dist/sources/snapshot-fetchers/fetcher-util.js +41 -0
- package/dist/sources/snapshot-fetchers/github.js +100 -0
- package/dist/sources/snapshot-fetchers/host-guard.js +291 -0
- package/dist/sources/snapshot-fetchers/registry.js +17 -1
- package/dist/sources/snapshot-fetchers/robots.js +348 -0
- package/dist/sources/snapshot-fetchers/rss.js +282 -0
- package/dist/sources/snapshot-fetchers/secret-seam.js +42 -0
- package/dist/sources/snapshot-fetchers/website-ingest.js +566 -268
- package/dist/sources/snapshot-fetchers/x.js +910 -0
- package/dist/storage/database.js +7 -0
- package/dist/storage/engines/sqlite-migrations.js +23 -111
- package/dist/storage/managed-db.js +2 -2
- package/dist/storage/repositories/canaries-repository.js +1 -1
- package/dist/storage/repositories/events-repository.js +27 -11
- package/dist/storage/repositories/improve-runs-repository.js +6 -12
- package/dist/storage/repositories/index-connection.js +17 -6
- package/dist/storage/repositories/index-entries-repository.js +151 -240
- package/dist/storage/repositories/index-entry-mapper.js +15 -11
- package/dist/storage/repositories/index-fts-repository.js +5 -2
- package/dist/storage/repositories/index-llm-cache-repository.js +0 -1
- package/dist/storage/repositories/index-meta-repository.js +2 -3
- package/dist/storage/repositories/index-schema.js +10 -25
- package/dist/storage/repositories/index-utility-repository.js +15 -28
- package/dist/storage/repositories/index-vec-repository.js +6 -1
- package/dist/storage/repositories/outcome-repository.js +119 -0
- package/dist/storage/repositories/proposals-repository.js +296 -59
- package/dist/storage/repositories/registry-cache.js +19 -0
- package/dist/storage/repositories/salience-repository.js +172 -0
- package/dist/storage/repositories/task-history-repository.js +15 -13
- package/dist/storage/repositories/workflow-runs-repository.js +52 -40
- package/dist/tasks/backends/cron.js +105 -15
- package/dist/tasks/backends/index.js +1 -1
- package/dist/tasks/backends/launchd.js +85 -38
- package/dist/tasks/backends/schtasks.js +135 -15
- package/dist/tasks/embedded.js +56 -40
- package/dist/tasks/parser.js +7 -157
- package/dist/tasks/resolve-akm-bin.js +137 -59
- package/dist/tasks/runner.js +79 -42
- package/dist/tasks/scheduler-invocation.js +220 -10
- package/dist/tasks/schema.js +24 -1
- package/dist/tasks/task-id.js +1 -3
- package/dist/tasks/validator.js +20 -6
- package/dist/workflows/authoring/authoring.js +94 -143
- package/dist/workflows/authoring/scope-key.js +1 -1
- package/dist/workflows/exec/frozen-judge.js +28 -2
- package/dist/workflows/exec/native-executor.js +77 -57
- package/dist/workflows/exec/param-secrets.js +9 -9
- package/dist/workflows/exec/run-workflow.js +133 -79
- package/dist/workflows/exec/step-work.js +219 -346
- package/dist/{migrate-storage-node.mjs → workflows/exec/unit-dispatch.js} +1 -5
- package/dist/workflows/ir/compile.js +141 -270
- package/dist/workflows/ir/freeze.js +40 -30
- package/dist/workflows/ir/params.js +135 -11
- package/dist/workflows/ir/plan-hash.js +1 -1
- package/dist/workflows/ir/schema.js +25 -26
- package/dist/workflows/parser.js +872 -307
- package/dist/workflows/program/expressions.js +20 -208
- package/dist/workflows/program/schema.js +7 -10
- package/dist/workflows/renderer.js +95 -68
- package/dist/workflows/resource-limits.js +2 -0
- package/dist/workflows/runtime/checkin.js +3 -3
- package/dist/workflows/runtime/plan-classifier.js +16 -75
- package/dist/workflows/runtime/runs.js +186 -127
- package/dist/workflows/runtime/unit-checkin.js +1 -1
- package/dist/workflows/runtime/unit-phases.js +2 -2
- package/dist/workflows/runtime/workflow-asset-loader.js +232 -83
- package/dist/workflows/schema.js +1 -11
- package/dist/workflows/validate-summary.js +30 -36
- package/dist/workflows/validator.js +21 -62
- package/docs/README.md +68 -0
- package/docs/migration/README.md +8 -0
- package/docs/migration/release-notes/0.7.0.md +11 -11
- package/docs/migration/release-notes/0.9.0.md +208 -27
- package/docs/migration/v0.7-to-v0.8.md +46 -47
- package/docs/migration/v0.8-to-v0.9.md +564 -208
- package/docs/migration/v0.9.0-troubleshooting.md +561 -0
- package/docs/reference/README.md +12 -0
- package/docs/reference/cli.md +2253 -0
- package/docs/reference/configuration.md +358 -0
- package/docs/reference/data-and-telemetry.md +105 -42
- package/docs/reference/workflows.md +647 -0
- package/package.json +22 -11
- package/schemas/akm-asset-envelope.json +93 -0
- package/schemas/akm-config.json +81 -128
- package/schemas/akm-workflow.json +74 -73
- package/dist/assets/tasks/core/backup.yml +0 -5
- package/dist/assets/tasks/graph-refresh-weekly.yml +0 -10
- package/dist/cli/config-migrate.js +0 -1806
- package/dist/cli/config-validate.js +0 -41
- package/dist/commands/backup-cli.js +0 -56
- package/dist/commands/bundle/bundle-cli.js +0 -68
- package/dist/commands/bundle/bundle.js +0 -219
- package/dist/commands/graph/graph-cli.js +0 -124
- package/dist/commands/graph/graph.js +0 -489
- package/dist/commands/improve/extract-watch.js +0 -140
- package/dist/commands/mv-cli.js +0 -1221
- package/dist/commands/sources/history.js +0 -201
- package/dist/commands/tasks/default-tasks.js +0 -186
- package/dist/core/migration-backup.js +0 -1234
- package/dist/indexer/usage/unmigrated-vaults-guard.js +0 -95
- package/dist/llm/memory-infer-impl.js +0 -138
- package/dist/migrate/legacy/config-source-migration.js +0 -223
- package/dist/migrate/legacy/content-migration.js +0 -305
- package/dist/migrate/legacy/legacy-layout.js +0 -779
- package/dist/migrate/legacy/legacy-paths.js +0 -25
- package/dist/migrate/legacy/legacy-stash-json.js +0 -72
- package/dist/migrate/legacy/proposal-fs-import.js +0 -168
- package/dist/migrate/legacy/task-target-ref-migration.js +0 -272
- package/dist/migrate/legacy/three-db-cutover.js +0 -841
- package/dist/migrate/legacy/workflow-migrations-bodies.js +0 -52
- package/dist/migrate/legacy/workflow-migrations-frozen.js +0 -21
- package/dist/migrate/legacy-ref-grammar.js +0 -214
- package/dist/output/shapes/distill.js +0 -14
- package/dist/output/shapes/history.js +0 -11
- package/dist/output/text/distill.js +0 -6
- package/dist/output/text/enable-disable.js +0 -8
- package/dist/output/text/history.js +0 -6
- package/dist/registry/build-index.js +0 -382
- package/dist/schemas/akm-config.json +0 -4704
- package/dist/schemas/akm-task.json +0 -87
- package/dist/schemas/akm-workflow.json +0 -372
- package/dist/scripts/migrate-storage.js +0 -3816
- package/dist/workflows/authoring/workflow-program-template.yaml +0 -31
- package/dist/workflows/cli.js +0 -53
- package/dist/workflows/exec/brief.js +0 -481
- package/dist/workflows/exec/report.js +0 -1460
- package/dist/workflows/exec/watch.js +0 -116
- package/dist/workflows/program/parser.js +0 -813
- package/dist/workflows/program/project.js +0 -104
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
/**
|
|
5
|
+
* robots.txt parsing, matching, and a crawl-scoped fetch/cache policy.
|
|
6
|
+
*
|
|
7
|
+
* Behavioral reference: docs/plans/specs/p1-robots.md (authoritative). This
|
|
8
|
+
* module is a clean rewrite against akm's contracts, not a port of any
|
|
9
|
+
* upstream source file.
|
|
10
|
+
*
|
|
11
|
+
* Pure by design: no fetching happens here (see `RobotsTxtLoader`/
|
|
12
|
+
* `loadRobotsTxt` in `website-ingest.ts`). There is no crawl-semantic
|
|
13
|
+
* module-level state: the per-origin HTTP cache lives on the object
|
|
14
|
+
* `createRobotsPolicy` returns so it dies with the crawl that created it —
|
|
15
|
+
* a module-level `origin -> RobotsRuleSet` map would leak `disallowAll`
|
|
16
|
+
* results across crawls and across the test suite's process-wide `bun test`
|
|
17
|
+
* run (see spec §6.1). The module-level `WeakMap` below is a different
|
|
18
|
+
* category: it memoizes a pure function of an individual rule object's own
|
|
19
|
+
* `pattern` field, keyed by that object's identity. It cannot leak meaning
|
|
20
|
+
* between origins or tests — distinct `parseRobotsTxt` calls (even for
|
|
21
|
+
* identical robots.txt text) allocate distinct rule objects with distinct
|
|
22
|
+
* cache slots, and slots are reclaimed by the GC once a rule set is
|
|
23
|
+
* unreachable. It is memoization scoped to a single parsed rule set, not
|
|
24
|
+
* shared cross-crawl state.
|
|
25
|
+
*/
|
|
26
|
+
import { warnVerbose } from "../../core/warn.js";
|
|
27
|
+
// ── §2 Constants (pinned by spec) ───────────────────────────────────────────
|
|
28
|
+
/** Group-matching tokens for `User-agent`, compared case-insensitively. */
|
|
29
|
+
export const ROBOTS_PRODUCT_TOKENS = ["akm", "akm-cli"];
|
|
30
|
+
/** 512 KiB. RFC 9309 §2.5 requires parsing at least 500 KiB. */
|
|
31
|
+
export const ROBOTS_BYTE_CAP = 512 * 1024;
|
|
32
|
+
/** Matches the existing website page-fetch timeout. */
|
|
33
|
+
export const ROBOTS_FETCH_TIMEOUT_MS = 15_000;
|
|
34
|
+
/** Body-read deadline; robots.txt is tiny. */
|
|
35
|
+
export const ROBOTS_BODY_TIMEOUT_MS = 15_000;
|
|
36
|
+
/** Ceiling applied to any parsed `Crawl-delay`, so a hostile robots.txt cannot stall the crawl. */
|
|
37
|
+
export const MAX_CRAWL_DELAY_MS = 10_000;
|
|
38
|
+
/** Identical to the existing website-fetch `User-Agent` header. */
|
|
39
|
+
export const ROBOTS_USER_AGENT_HEADER = "akm-cli website provider";
|
|
40
|
+
/**
|
|
41
|
+
* Frozen so a stray mutation (e.g. an accidental `.rules.push(...)` somewhere
|
|
42
|
+
* that borrows one of these as a starting point) can never poison every other
|
|
43
|
+
* crawl sharing this instance (spec §10, Review log 2026-08-01 finding 1).
|
|
44
|
+
*/
|
|
45
|
+
const EMPTY_ROBOTS_RULES = Object.freeze([]);
|
|
46
|
+
export const ALLOW_ALL_RULES = Object.freeze({
|
|
47
|
+
disallowAll: false,
|
|
48
|
+
rules: EMPTY_ROBOTS_RULES,
|
|
49
|
+
crawlDelayMs: null,
|
|
50
|
+
});
|
|
51
|
+
export const DISALLOW_ALL_RULES = Object.freeze({
|
|
52
|
+
disallowAll: true,
|
|
53
|
+
rules: EMPTY_ROBOTS_RULES,
|
|
54
|
+
crawlDelayMs: null,
|
|
55
|
+
});
|
|
56
|
+
/** Strips a trailing `# comment` (RFC 9309 comments run to end of line). */
|
|
57
|
+
function stripRobotsComment(line) {
|
|
58
|
+
const hashIndex = line.indexOf("#");
|
|
59
|
+
return hashIndex === -1 ? line : line.slice(0, hashIndex);
|
|
60
|
+
}
|
|
61
|
+
/**
|
|
62
|
+
* Parses a `Crawl-delay` value (seconds, possibly fractional) into ms.
|
|
63
|
+
* Non-positive, unparseable, and empty values are ignored (`null`).
|
|
64
|
+
* `Infinity` is preserved so the caller's clamp reduces it to the ceiling
|
|
65
|
+
* rather than the value silently vanishing.
|
|
66
|
+
*/
|
|
67
|
+
function parseCrawlDelayMs(raw) {
|
|
68
|
+
if (!raw)
|
|
69
|
+
return null;
|
|
70
|
+
const seconds = Number(raw);
|
|
71
|
+
if (Number.isNaN(seconds) || seconds <= 0)
|
|
72
|
+
return null;
|
|
73
|
+
return Math.round(seconds * 1000);
|
|
74
|
+
}
|
|
75
|
+
/**
|
|
76
|
+
* Pure. Never throws, never fetches.
|
|
77
|
+
*
|
|
78
|
+
* Grouping follows RFC 9309 §2.2.1: consecutive `User-agent` lines (no
|
|
79
|
+
* intervening directive) form one group; a matching *specific* product-token
|
|
80
|
+
* group suppresses the wildcard (`*`) group entirely; rules from all matching
|
|
81
|
+
* groups of the winning kind are unioned; `Crawl-delay` is the maximum across
|
|
82
|
+
* those groups, clamped once at the end.
|
|
83
|
+
*/
|
|
84
|
+
export function parseRobotsTxt(text) {
|
|
85
|
+
const stripped = text.charCodeAt(0) === 0xfeff ? text.slice(1) : text;
|
|
86
|
+
const lines = stripped.split(/\r\n|\r|\n/);
|
|
87
|
+
const groups = [];
|
|
88
|
+
let current = null;
|
|
89
|
+
// True once any non-user-agent directive has been seen for `current`, so
|
|
90
|
+
// the next `User-agent` line starts a NEW group instead of extending it.
|
|
91
|
+
let sawDirectiveSinceLastUserAgent = false;
|
|
92
|
+
for (const rawLine of lines) {
|
|
93
|
+
const line = stripRobotsComment(rawLine).trim();
|
|
94
|
+
if (!line)
|
|
95
|
+
continue;
|
|
96
|
+
const colonIndex = line.indexOf(":");
|
|
97
|
+
if (colonIndex === -1)
|
|
98
|
+
continue; // P-21: no colon => ignored
|
|
99
|
+
const directive = line.slice(0, colonIndex).trim().toLowerCase();
|
|
100
|
+
const value = line.slice(colonIndex + 1).trim();
|
|
101
|
+
if (directive === "user-agent") {
|
|
102
|
+
if (!value)
|
|
103
|
+
continue;
|
|
104
|
+
const token = value.toLowerCase();
|
|
105
|
+
if (current && !sawDirectiveSinceLastUserAgent) {
|
|
106
|
+
current.agents.push(token);
|
|
107
|
+
}
|
|
108
|
+
else {
|
|
109
|
+
current = { agents: [token], rules: [], crawlDelayMsRaw: null };
|
|
110
|
+
groups.push(current);
|
|
111
|
+
sawDirectiveSinceLastUserAgent = false;
|
|
112
|
+
}
|
|
113
|
+
continue;
|
|
114
|
+
}
|
|
115
|
+
if (!current)
|
|
116
|
+
continue; // P-13: directive with no preceding User-agent => ignored
|
|
117
|
+
sawDirectiveSinceLastUserAgent = true;
|
|
118
|
+
if (directive === "disallow" || directive === "allow") {
|
|
119
|
+
if (!value)
|
|
120
|
+
continue; // P-16: empty value matches nothing
|
|
121
|
+
if (!(value.startsWith("/") || value.startsWith("*"))) {
|
|
122
|
+
// P-18: neither "/" nor "*"-rooted => ignored, diagnosed quietly.
|
|
123
|
+
warnVerbose("[akm] robots.txt: ignoring %s value %s (patterns must start with '/' or '*')", directive, value);
|
|
124
|
+
continue;
|
|
125
|
+
}
|
|
126
|
+
const rule = {
|
|
127
|
+
kind: directive === "allow" ? "allow" : "disallow",
|
|
128
|
+
pattern: value,
|
|
129
|
+
specificity: value.length,
|
|
130
|
+
};
|
|
131
|
+
// Compile once, here, at parse time — not on every URL check (spec
|
|
132
|
+
// §6.4, review finding robots.ts:238). See `compiledPatternCache`.
|
|
133
|
+
compiledPatternCache.set(rule, compilePatternSegments(value));
|
|
134
|
+
current.rules.push(rule);
|
|
135
|
+
continue;
|
|
136
|
+
}
|
|
137
|
+
if (directive === "crawl-delay") {
|
|
138
|
+
const parsed = parseCrawlDelayMs(value);
|
|
139
|
+
if (parsed !== null)
|
|
140
|
+
current.crawlDelayMsRaw = parsed;
|
|
141
|
+
}
|
|
142
|
+
// Sitemap / unknown directives: ignored without diagnostic (P-19, P-20).
|
|
143
|
+
}
|
|
144
|
+
const productTokens = ROBOTS_PRODUCT_TOKENS;
|
|
145
|
+
const specificGroups = groups.filter((group) => group.agents.some((agent) => productTokens.includes(agent)));
|
|
146
|
+
const wildcardGroups = groups.filter((group) => group.agents.includes("*"));
|
|
147
|
+
const selected = specificGroups.length > 0 ? specificGroups : wildcardGroups;
|
|
148
|
+
const rules = selected.flatMap((group) => group.rules);
|
|
149
|
+
let maxDelayMs = null;
|
|
150
|
+
for (const group of selected) {
|
|
151
|
+
if (group.crawlDelayMsRaw !== null) {
|
|
152
|
+
maxDelayMs = maxDelayMs === null ? group.crawlDelayMsRaw : Math.max(maxDelayMs, group.crawlDelayMsRaw);
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
const crawlDelayMs = maxDelayMs === null ? null : Math.min(maxDelayMs, MAX_CRAWL_DELAY_MS);
|
|
156
|
+
return { disallowAll: false, rules, crawlDelayMs };
|
|
157
|
+
}
|
|
158
|
+
const COLLAPSE_STARS = /\*+/g;
|
|
159
|
+
const PERCENT_ENCODED_OCTET = /%[0-9a-fA-F]{2}/g;
|
|
160
|
+
/**
|
|
161
|
+
* RFC 3986 §2.3 "unreserved" set: ALPHA / DIGIT / "-" / "." / "_" / "~".
|
|
162
|
+
* A percent-encoded octet in this set is byte-identical to its literal
|
|
163
|
+
* character (`%73` and `s` are the same byte), so it carries no meaning a
|
|
164
|
+
* URL consumer can rely on. Every other octet — `%2F` ('/'), `%3F` ('?'),
|
|
165
|
+
* `%23` ('#'), `%2A` ('*'), non-ASCII bytes like `%C3` — is "reserved" or
|
|
166
|
+
* otherwise structurally significant and must be left percent-encoded:
|
|
167
|
+
* decoding e.g. `%2F` would silently turn one path segment into two.
|
|
168
|
+
*/
|
|
169
|
+
function isUnreservedByte(byte) {
|
|
170
|
+
return ((byte >= 0x41 && byte <= 0x5a) || // A-Z
|
|
171
|
+
(byte >= 0x61 && byte <= 0x7a) || // a-z
|
|
172
|
+
(byte >= 0x30 && byte <= 0x39) || // 0-9
|
|
173
|
+
byte === 0x2d || // -
|
|
174
|
+
byte === 0x2e || // .
|
|
175
|
+
byte === 0x5f || // _
|
|
176
|
+
byte === 0x7e // ~
|
|
177
|
+
);
|
|
178
|
+
}
|
|
179
|
+
/**
|
|
180
|
+
* Normalizes percent-encoding for robots.txt matching, mirroring the RFC
|
|
181
|
+
* 9309 reference matcher: percent-encoded UNRESERVED octets are decoded to
|
|
182
|
+
* their literal character; every other `%XX` escape is left exactly as
|
|
183
|
+
* written (including hex-digit case). Applied identically to rule patterns
|
|
184
|
+
* (via `compilePatternSegments`, at parse time and in the defensive lazy
|
|
185
|
+
* fallback) and match targets (`isPathAllowedByRobots`), so a percent-encoded
|
|
186
|
+
* alias of a disallowed path — e.g. a link written as `/%73ecret/` — cannot
|
|
187
|
+
* bypass `Disallow: /secret/` simply because `URL.pathname` preserves
|
|
188
|
+
* %-escapes verbatim. Reserved-octet patterns like spec §4.3 M-31's
|
|
189
|
+
* `/caf%C3%A9` are untouched on both sides (0xC3/0xA9 are not unreserved),
|
|
190
|
+
* so they keep matching themselves exactly as before.
|
|
191
|
+
*/
|
|
192
|
+
function normalizePercentEncoding(value) {
|
|
193
|
+
if (!value.includes("%"))
|
|
194
|
+
return value;
|
|
195
|
+
return value.replace(PERCENT_ENCODED_OCTET, (octetEscape) => {
|
|
196
|
+
const byte = Number.parseInt(octetEscape.slice(1), 16);
|
|
197
|
+
if (isUnreservedByte(byte))
|
|
198
|
+
return String.fromCharCode(byte);
|
|
199
|
+
// A reserved octet stays encoded, but `%2F` and `%2f` denote the same
|
|
200
|
+
// byte — RFC 3986 §6.2.2.1 makes the hex digits case-insensitive. Without
|
|
201
|
+
// canonicalizing the case, `Disallow: /a%2Fb` would fail to match a link
|
|
202
|
+
// spelled `/a%2fb` and the page would be crawled anyway.
|
|
203
|
+
return octetEscape.toUpperCase();
|
|
204
|
+
});
|
|
205
|
+
}
|
|
206
|
+
function compilePatternSegments(pattern) {
|
|
207
|
+
const anchored = pattern.endsWith("$");
|
|
208
|
+
const body = anchored ? pattern.slice(0, -1) : pattern;
|
|
209
|
+
const collapsed = normalizePercentEncoding(body).replace(COLLAPSE_STARS, "*");
|
|
210
|
+
return { anchored, segments: collapsed.split("*") };
|
|
211
|
+
}
|
|
212
|
+
/**
|
|
213
|
+
* Per-rule-object cache of the compiled pattern (see the file-level doc
|
|
214
|
+
* comment for why this is not "module-level state" in the sense spec §6.1
|
|
215
|
+
* forbids). `parseRobotsTxt` populates this eagerly, once, when it creates
|
|
216
|
+
* each `RobotsPathRule` — the compile-once requirement from spec §6.4 /
|
|
217
|
+
* review finding robots.ts:238. `compiledPatternFor` below is a defensive
|
|
218
|
+
* fallback for `RobotsRuleSet`s assembled by hand (e.g. test fixtures)
|
|
219
|
+
* rather than via `parseRobotsTxt`; it still compiles at most once per rule
|
|
220
|
+
* object, just lazily on first use instead of at parse time.
|
|
221
|
+
*/
|
|
222
|
+
const compiledPatternCache = new WeakMap();
|
|
223
|
+
function compiledPatternFor(rule) {
|
|
224
|
+
const cached = compiledPatternCache.get(rule);
|
|
225
|
+
if (cached)
|
|
226
|
+
return cached;
|
|
227
|
+
const compiled = compilePatternSegments(rule.pattern);
|
|
228
|
+
compiledPatternCache.set(rule, compiled);
|
|
229
|
+
return compiled;
|
|
230
|
+
}
|
|
231
|
+
/**
|
|
232
|
+
* Matches a precompiled pattern against `target` (`pathname + search`).
|
|
233
|
+
* Greedy left-to-right, no backtracking — see the section doc comment above.
|
|
234
|
+
*/
|
|
235
|
+
function matchesCompiledPattern(compiled, target) {
|
|
236
|
+
const { anchored, segments } = compiled;
|
|
237
|
+
const first = segments[0] ?? "";
|
|
238
|
+
if (!target.startsWith(first))
|
|
239
|
+
return false;
|
|
240
|
+
const pos = first.length;
|
|
241
|
+
const lastIndex = segments.length - 1;
|
|
242
|
+
if (lastIndex === 0) {
|
|
243
|
+
// No wildcard at all: already a prefix match; anchored additionally
|
|
244
|
+
// requires the pattern to consume the whole target.
|
|
245
|
+
return anchored ? target.length === first.length : true;
|
|
246
|
+
}
|
|
247
|
+
let cursor = pos;
|
|
248
|
+
for (let i = 1; i < lastIndex; i++) {
|
|
249
|
+
const segment = segments[i] ?? "";
|
|
250
|
+
const found = target.indexOf(segment, cursor);
|
|
251
|
+
if (found === -1)
|
|
252
|
+
return false;
|
|
253
|
+
// Greedy-leftmost is safe here: an earlier match of segment i can only
|
|
254
|
+
// give segment i+1 MORE room to be found later, never less, since the
|
|
255
|
+
// `*` between them absorbs any extra characters. No backtracking needed.
|
|
256
|
+
cursor = found + segment.length;
|
|
257
|
+
}
|
|
258
|
+
const last = segments[lastIndex] ?? "";
|
|
259
|
+
if (anchored) {
|
|
260
|
+
if (last.length > target.length - cursor)
|
|
261
|
+
return false;
|
|
262
|
+
return target.endsWith(last);
|
|
263
|
+
}
|
|
264
|
+
return last === "" || target.indexOf(last, cursor) !== -1;
|
|
265
|
+
}
|
|
266
|
+
/**
|
|
267
|
+
* Pure. `url` is an absolute http(s) URL string. Matching uses
|
|
268
|
+
* `pathname + search` of the parsed URL. Returns true when the URL cannot be
|
|
269
|
+
* parsed (caller has already validated it; do not fail closed on a parse bug).
|
|
270
|
+
*/
|
|
271
|
+
export function isPathAllowedByRobots(rules, url) {
|
|
272
|
+
if (rules.disallowAll)
|
|
273
|
+
return false;
|
|
274
|
+
let parsed;
|
|
275
|
+
try {
|
|
276
|
+
parsed = new URL(url);
|
|
277
|
+
}
|
|
278
|
+
catch {
|
|
279
|
+
return true; // M-32: fail open, never closed, on an unparseable URL.
|
|
280
|
+
}
|
|
281
|
+
// Normalized the same way `compilePatternSegments` normalizes the rule
|
|
282
|
+
// pattern (see `normalizePercentEncoding`'s doc comment) — otherwise a
|
|
283
|
+
// percent-encoded alias of a disallowed path (e.g. `/%73ecret/` for a
|
|
284
|
+
// `Disallow: /secret/` rule) would never match because `URL.pathname`
|
|
285
|
+
// preserves %-escapes verbatim.
|
|
286
|
+
const target = normalizePercentEncoding(`${parsed.pathname}${parsed.search}`);
|
|
287
|
+
let best = null;
|
|
288
|
+
for (const rule of rules.rules) {
|
|
289
|
+
if (!matchesCompiledPattern(compiledPatternFor(rule), target))
|
|
290
|
+
continue;
|
|
291
|
+
const isMoreSpecific = !best || rule.specificity > best.specificity;
|
|
292
|
+
// Tie goes to Allow (least restrictive) — order-independent: this only
|
|
293
|
+
// upgrades a disallow to an allow at equal specificity, never the reverse.
|
|
294
|
+
const isTieBrokenByAllow = best !== null && rule.specificity === best.specificity && rule.kind === "allow" && best.kind === "disallow";
|
|
295
|
+
if (isMoreSpecific || isTieBrokenByAllow)
|
|
296
|
+
best = rule;
|
|
297
|
+
}
|
|
298
|
+
return best ? best.kind === "allow" : true;
|
|
299
|
+
}
|
|
300
|
+
// ── §4.4 createRobotsPolicy / createAllowAllRobotsPolicy ────────────────────
|
|
301
|
+
function outcomeToRuleSet(outcome) {
|
|
302
|
+
switch (outcome.kind) {
|
|
303
|
+
case "body":
|
|
304
|
+
return parseRobotsTxt(outcome.text);
|
|
305
|
+
case "unavailable":
|
|
306
|
+
return ALLOW_ALL_RULES;
|
|
307
|
+
case "unreachable":
|
|
308
|
+
return DISALLOW_ALL_RULES;
|
|
309
|
+
}
|
|
310
|
+
}
|
|
311
|
+
/**
|
|
312
|
+
* Crawl-scoped. Caches per origin (including in-flight promises) so the
|
|
313
|
+
* `load` callback runs at most once per origin regardless of how many pages
|
|
314
|
+
* or concurrent lookups reference it. Deliberately NOT a module-level cache
|
|
315
|
+
* (see the file-level doc comment) — callers construct one instance per crawl.
|
|
316
|
+
*/
|
|
317
|
+
export function createRobotsPolicy(load) {
|
|
318
|
+
const cache = new Map();
|
|
319
|
+
function rulesForOrigin(origin) {
|
|
320
|
+
const cached = cache.get(origin);
|
|
321
|
+
if (cached)
|
|
322
|
+
return cached;
|
|
323
|
+
const robotsUrl = new URL("/robots.txt", origin).toString();
|
|
324
|
+
const pending = load(robotsUrl).then(outcomeToRuleSet);
|
|
325
|
+
// A rejected fetch must not poison the cache: the next lookup for this
|
|
326
|
+
// origin should retry the loader rather than replaying the same failure
|
|
327
|
+
// forever (spec §4.4 L-04).
|
|
328
|
+
pending.catch(() => {
|
|
329
|
+
if (cache.get(origin) === pending)
|
|
330
|
+
cache.delete(origin);
|
|
331
|
+
});
|
|
332
|
+
cache.set(origin, pending);
|
|
333
|
+
return pending;
|
|
334
|
+
}
|
|
335
|
+
return {
|
|
336
|
+
rulesFor: (url) => rulesForOrigin(new URL(url).origin),
|
|
337
|
+
isAllowed: async (url) => isPathAllowedByRobots(await rulesForOrigin(new URL(url).origin), url),
|
|
338
|
+
crawlDelayMs: async (url) => (await rulesForOrigin(new URL(url).origin)).crawlDelayMs ?? 0,
|
|
339
|
+
};
|
|
340
|
+
}
|
|
341
|
+
/** Never fetches. isAllowed => true, crawlDelayMs => 0. Used when respectRobots is false. */
|
|
342
|
+
export function createAllowAllRobotsPolicy() {
|
|
343
|
+
return {
|
|
344
|
+
rulesFor: async () => ALLOW_ALL_RULES,
|
|
345
|
+
isAllowed: async () => true,
|
|
346
|
+
crawlDelayMs: async () => 0,
|
|
347
|
+
};
|
|
348
|
+
}
|
|
@@ -0,0 +1,282 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import { XMLParser } from "fast-xml-parser";
|
|
5
|
+
import { ResponseTooLargeError, readBodyWithByteCap } from "../../core/common.js";
|
|
6
|
+
import { htmlToMarkdown, isSafeLinkUrl } from "./content-extract.js";
|
|
7
|
+
import { fetchGuardedResponse } from "./host-guard.js";
|
|
8
|
+
/**
|
|
9
|
+
* RSS 2.0 / Atom 1.0 / RDF (RSS 1.0) feed fetcher.
|
|
10
|
+
*
|
|
11
|
+
* `matches` is deliberately loose — feed URLs are not reliably identifiable
|
|
12
|
+
* from their path alone — so `fetch` content-sniffs the body and returns
|
|
13
|
+
* `null` for anything that is not actually a feed. Returning `null` lets the
|
|
14
|
+
* registry fall through to the generic website crawler, which is the correct
|
|
15
|
+
* outcome for a `/feed`-shaped URL that serves HTML.
|
|
16
|
+
*/
|
|
17
|
+
/** Feeds far larger than this are aggregators, not knowledge sources. */
|
|
18
|
+
const FEED_BYTE_CAP = 5 * 1024 * 1024;
|
|
19
|
+
const FEED_BODY_TIMEOUT_MS = 30_000;
|
|
20
|
+
const DEFAULT_ITEM_LIMIT = 50;
|
|
21
|
+
/** Paths that usually indicate a feed. Confirmed by sniffing before parsing. */
|
|
22
|
+
const FEED_PATH_PATTERN = /(?:\.(?:rss|atom|xml)$|\/(?:feed|rss|atom)\/?$)/i;
|
|
23
|
+
/**
|
|
24
|
+
* fast-xml-parser resolves no external entities and has no DTD/XXE surface,
|
|
25
|
+
* so the classic billion-laughs and file-disclosure vectors do not apply.
|
|
26
|
+
* The remaining risk is sheer size, which `FEED_BYTE_CAP` bounds before any
|
|
27
|
+
* parsing happens.
|
|
28
|
+
*/
|
|
29
|
+
const xmlParser = new XMLParser({
|
|
30
|
+
ignoreAttributes: false,
|
|
31
|
+
attributeNamePrefix: "@_",
|
|
32
|
+
textNodeName: "#text",
|
|
33
|
+
isArray: (name) => ["item", "entry"].includes(name),
|
|
34
|
+
trimValues: true,
|
|
35
|
+
parseTagValue: false,
|
|
36
|
+
processEntities: true,
|
|
37
|
+
});
|
|
38
|
+
/** Coerce fast-xml-parser's string | {#text} | array shapes to a string. */
|
|
39
|
+
function text(value) {
|
|
40
|
+
if (typeof value === "string")
|
|
41
|
+
return value.trim();
|
|
42
|
+
if (typeof value === "number" || typeof value === "boolean")
|
|
43
|
+
return String(value);
|
|
44
|
+
if (Array.isArray(value))
|
|
45
|
+
return text(value[0]);
|
|
46
|
+
if (value && typeof value === "object") {
|
|
47
|
+
const node = value;
|
|
48
|
+
if ("#text" in node)
|
|
49
|
+
return text(node["#text"]);
|
|
50
|
+
}
|
|
51
|
+
return "";
|
|
52
|
+
}
|
|
53
|
+
/**
|
|
54
|
+
* Atom `<content>` may be type="text"/"html" (a #text body) or type="xhtml"
|
|
55
|
+
* (child elements). fast-xml-parser gives the latter as an object with no
|
|
56
|
+
* `#text`, which would otherwise read as an empty summary and silently drop
|
|
57
|
+
* the whole article body.
|
|
58
|
+
*/
|
|
59
|
+
function atomContent(value) {
|
|
60
|
+
const direct = text(value);
|
|
61
|
+
if (direct)
|
|
62
|
+
return direct;
|
|
63
|
+
if (!value || typeof value !== "object")
|
|
64
|
+
return "";
|
|
65
|
+
const collect = (node) => {
|
|
66
|
+
if (typeof node === "string")
|
|
67
|
+
return node;
|
|
68
|
+
if (Array.isArray(node))
|
|
69
|
+
return node.map(collect).join(" ");
|
|
70
|
+
if (node && typeof node === "object") {
|
|
71
|
+
return Object.entries(node)
|
|
72
|
+
.filter(([key]) => !key.startsWith("@_"))
|
|
73
|
+
.map(([, child]) => collect(child))
|
|
74
|
+
.join(" ");
|
|
75
|
+
}
|
|
76
|
+
return "";
|
|
77
|
+
};
|
|
78
|
+
return collect(value);
|
|
79
|
+
}
|
|
80
|
+
/** Atom links are attribute-shaped and may repeat with different rels. */
|
|
81
|
+
function atomLink(value) {
|
|
82
|
+
const candidates = Array.isArray(value) ? value : [value];
|
|
83
|
+
let fallback = "";
|
|
84
|
+
for (const candidate of candidates) {
|
|
85
|
+
if (typeof candidate === "string") {
|
|
86
|
+
fallback ||= candidate.trim();
|
|
87
|
+
continue;
|
|
88
|
+
}
|
|
89
|
+
if (!candidate || typeof candidate !== "object")
|
|
90
|
+
continue;
|
|
91
|
+
const node = candidate;
|
|
92
|
+
const href = typeof node["@_href"] === "string" ? node["@_href"].trim() : "";
|
|
93
|
+
if (!href)
|
|
94
|
+
continue;
|
|
95
|
+
const rel = typeof node["@_rel"] === "string" ? node["@_rel"] : "";
|
|
96
|
+
if (rel === "alternate" || rel === "")
|
|
97
|
+
return href;
|
|
98
|
+
fallback ||= href;
|
|
99
|
+
}
|
|
100
|
+
return fallback;
|
|
101
|
+
}
|
|
102
|
+
/**
|
|
103
|
+
* Feed item bodies are attacker-controlled HTML (often inside CDATA or
|
|
104
|
+
* entity-encoded). Route them through the same converter the website path
|
|
105
|
+
* uses so dangerous-block removal and link-scheme policy apply here too — a
|
|
106
|
+
* plain tag-strip keeps `<script>` BODIES and leaks attribute text.
|
|
107
|
+
*/
|
|
108
|
+
function htmlToPlainText(value, baseUrl) {
|
|
109
|
+
if (!value)
|
|
110
|
+
return "";
|
|
111
|
+
return htmlToMarkdown(value, baseUrl).replace(/\s+/g, " ").trim();
|
|
112
|
+
}
|
|
113
|
+
/** Collapse whitespace so a value cannot forge markdown structure. */
|
|
114
|
+
function singleLine(value) {
|
|
115
|
+
return value.replace(/\s+/g, " ").trim();
|
|
116
|
+
}
|
|
117
|
+
/**
|
|
118
|
+
* Emit a link only when it resolves to a safe http(s) URL. Relative IRIs are
|
|
119
|
+
* valid and common in Atom, so resolve against the feed URL before the scheme
|
|
120
|
+
* check rather than dropping them.
|
|
121
|
+
*/
|
|
122
|
+
function safeLinkOrEmpty(value, baseUrl) {
|
|
123
|
+
if (!value)
|
|
124
|
+
return "";
|
|
125
|
+
try {
|
|
126
|
+
const resolved = new URL(value, baseUrl);
|
|
127
|
+
return isSafeLinkUrl(resolved) ? resolved.toString() : "";
|
|
128
|
+
}
|
|
129
|
+
catch {
|
|
130
|
+
return "";
|
|
131
|
+
}
|
|
132
|
+
}
|
|
133
|
+
function toIsoDate(value) {
|
|
134
|
+
if (!value)
|
|
135
|
+
return "";
|
|
136
|
+
const parsed = new Date(value);
|
|
137
|
+
return Number.isNaN(parsed.getTime()) ? "" : parsed.toISOString();
|
|
138
|
+
}
|
|
139
|
+
function readRssItems(channel, limit, baseUrl) {
|
|
140
|
+
const items = Array.isArray(channel.item) ? channel.item : [];
|
|
141
|
+
return items.slice(0, limit).map((raw) => {
|
|
142
|
+
const item = (raw ?? {});
|
|
143
|
+
return {
|
|
144
|
+
title: singleLine(text(item.title)),
|
|
145
|
+
link: safeLinkOrEmpty(text(item.link) || text(item.guid), baseUrl),
|
|
146
|
+
date: toIsoDate(text(item.pubDate) || text(item["dc:date"])),
|
|
147
|
+
summary: htmlToPlainText(text(item.description) || text(item["content:encoded"]), baseUrl),
|
|
148
|
+
};
|
|
149
|
+
});
|
|
150
|
+
}
|
|
151
|
+
function readAtomEntries(feed, limit, baseUrl) {
|
|
152
|
+
const entries = Array.isArray(feed.entry) ? feed.entry : [];
|
|
153
|
+
return entries.slice(0, limit).map((raw) => {
|
|
154
|
+
const entry = (raw ?? {});
|
|
155
|
+
return {
|
|
156
|
+
title: singleLine(text(entry.title)),
|
|
157
|
+
link: safeLinkOrEmpty(atomLink(entry.link), baseUrl),
|
|
158
|
+
date: toIsoDate(text(entry.updated) || text(entry.published)),
|
|
159
|
+
summary: htmlToPlainText(text(entry.summary) || atomContent(entry.content), baseUrl),
|
|
160
|
+
};
|
|
161
|
+
});
|
|
162
|
+
}
|
|
163
|
+
/** Returns null when the document is not a recognizable feed. */
|
|
164
|
+
export function parseFeed(xml, limit = DEFAULT_ITEM_LIMIT, baseUrl = "https://feed.invalid/") {
|
|
165
|
+
let doc;
|
|
166
|
+
try {
|
|
167
|
+
doc = xmlParser.parse(xml);
|
|
168
|
+
}
|
|
169
|
+
catch {
|
|
170
|
+
return null;
|
|
171
|
+
}
|
|
172
|
+
if (!doc || typeof doc !== "object")
|
|
173
|
+
return null;
|
|
174
|
+
const rss = doc.rss;
|
|
175
|
+
if (rss?.channel) {
|
|
176
|
+
const channel = rss.channel;
|
|
177
|
+
return { title: singleLine(text(channel.title)), items: readRssItems(channel, limit, baseUrl) };
|
|
178
|
+
}
|
|
179
|
+
const feed = doc.feed;
|
|
180
|
+
if (feed) {
|
|
181
|
+
return { title: singleLine(text(feed.title)), items: readAtomEntries(feed, limit, baseUrl) };
|
|
182
|
+
}
|
|
183
|
+
// RDF / RSS 1.0 hoists <item> to the document root alongside <channel>.
|
|
184
|
+
const rdf = doc["rdf:RDF"];
|
|
185
|
+
if (rdf) {
|
|
186
|
+
const channel = (rdf.channel ?? {});
|
|
187
|
+
const scope = Array.isArray(rdf.item) ? rdf : channel;
|
|
188
|
+
return {
|
|
189
|
+
title: singleLine(text(channel.title)),
|
|
190
|
+
items: readRssItems(scope, limit, baseUrl),
|
|
191
|
+
};
|
|
192
|
+
}
|
|
193
|
+
return null;
|
|
194
|
+
}
|
|
195
|
+
/** Cheap pre-parse check so HTML served at a feed-shaped URL is not parsed. */
|
|
196
|
+
function looksLikeFeed(body) {
|
|
197
|
+
const head = body.slice(0, 16_384);
|
|
198
|
+
return /<(?:rss\b|feed\b|rdf:RDF\b)/i.test(head);
|
|
199
|
+
}
|
|
200
|
+
function slugify(value) {
|
|
201
|
+
return value
|
|
202
|
+
.toLowerCase()
|
|
203
|
+
.replace(/[^a-z0-9._-]+/g, "-")
|
|
204
|
+
.replace(/^-+|-+$/g, "")
|
|
205
|
+
.slice(0, 80);
|
|
206
|
+
}
|
|
207
|
+
function preferredNameFor(url) {
|
|
208
|
+
const host = slugify(url.hostname);
|
|
209
|
+
const pathPart = slugify(url.pathname.replace(/\.(rss|atom|xml)$/i, ""));
|
|
210
|
+
return pathPart ? `feeds/${host}-${pathPart}` : `feeds/${host}`;
|
|
211
|
+
}
|
|
212
|
+
function renderMarkdown(feed, url) {
|
|
213
|
+
const sections = [];
|
|
214
|
+
for (const item of feed.items) {
|
|
215
|
+
const heading = item.title || item.link || "(untitled)";
|
|
216
|
+
sections.push(`## ${heading}`, "");
|
|
217
|
+
const meta = [];
|
|
218
|
+
if (item.date)
|
|
219
|
+
meta.push(item.date);
|
|
220
|
+
if (item.link)
|
|
221
|
+
meta.push(item.link);
|
|
222
|
+
if (meta.length > 0)
|
|
223
|
+
sections.push(meta.join(" — "), "");
|
|
224
|
+
if (item.summary)
|
|
225
|
+
sections.push(item.summary, "");
|
|
226
|
+
}
|
|
227
|
+
if (sections.length === 0)
|
|
228
|
+
sections.push(`No items in feed ${url.toString()}`, "");
|
|
229
|
+
return sections.join("\n").trimEnd();
|
|
230
|
+
}
|
|
231
|
+
const rssFetcher = {
|
|
232
|
+
name: "rss-feed",
|
|
233
|
+
matches(url) {
|
|
234
|
+
if (url.protocol !== "http:" && url.protocol !== "https:")
|
|
235
|
+
return false;
|
|
236
|
+
return FEED_PATH_PATTERN.test(url.pathname) || url.searchParams.has("feed");
|
|
237
|
+
},
|
|
238
|
+
async fetch(url, context) {
|
|
239
|
+
// Full SSRF guard chain: a feed URL is caller-supplied, and without
|
|
240
|
+
// manual redirect handling a public host can bounce the request into the
|
|
241
|
+
// private network with no revalidation.
|
|
242
|
+
const { response, finalUrl } = await fetchGuardedResponse(url.toString(), {
|
|
243
|
+
headers: {
|
|
244
|
+
Accept: "application/rss+xml, application/atom+xml, application/xml;q=0.9, text/xml;q=0.8",
|
|
245
|
+
"User-Agent": "akm-cli rss fetcher",
|
|
246
|
+
},
|
|
247
|
+
signal: context.signal,
|
|
248
|
+
}, { timeoutMs: context.timeoutMs, retries: 1, allowPrivateHosts: context.allowPrivateHosts });
|
|
249
|
+
if (!response.ok) {
|
|
250
|
+
await response.body?.cancel().catch(() => undefined);
|
|
251
|
+
return null;
|
|
252
|
+
}
|
|
253
|
+
let body;
|
|
254
|
+
try {
|
|
255
|
+
body = await readBodyWithByteCap(response, FEED_BYTE_CAP, {
|
|
256
|
+
bodyTimeoutMs: FEED_BODY_TIMEOUT_MS,
|
|
257
|
+
signal: context.signal,
|
|
258
|
+
});
|
|
259
|
+
}
|
|
260
|
+
catch (error) {
|
|
261
|
+
// An oversized feed is not a hard failure — fall through to the generic
|
|
262
|
+
// crawler rather than aborting the whole `akm bundle add`.
|
|
263
|
+
if (error instanceof ResponseTooLargeError)
|
|
264
|
+
return null;
|
|
265
|
+
throw error;
|
|
266
|
+
}
|
|
267
|
+
if (!looksLikeFeed(body))
|
|
268
|
+
return null;
|
|
269
|
+
const resolvedUrl = new URL(finalUrl);
|
|
270
|
+
const feed = parseFeed(body, DEFAULT_ITEM_LIMIT, resolvedUrl.toString());
|
|
271
|
+
if (!feed || feed.items.length === 0)
|
|
272
|
+
return null;
|
|
273
|
+
return {
|
|
274
|
+
url: resolvedUrl.toString(),
|
|
275
|
+
title: feed.title || resolvedUrl.hostname,
|
|
276
|
+
markdown: renderMarkdown(feed, resolvedUrl),
|
|
277
|
+
preferredName: preferredNameFor(resolvedUrl),
|
|
278
|
+
tags: ["rss", "feed", resolvedUrl.hostname],
|
|
279
|
+
};
|
|
280
|
+
},
|
|
281
|
+
};
|
|
282
|
+
export default rssFetcher;
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
// This Source Code Form is subject to the terms of the Mozilla Public
|
|
2
|
+
// License, v. 2.0. If a copy of the MPL was not distributed with this
|
|
3
|
+
// file, You can obtain one at https://mozilla.org/MPL/2.0/.
|
|
4
|
+
import fs from "node:fs";
|
|
5
|
+
import { resolveSecretPath } from "../../core/env-secret-ref.js";
|
|
6
|
+
/**
|
|
7
|
+
* Reads a secret out of akm's secret store for the fetcher `resolveSecret`
|
|
8
|
+
* seam (see `FetcherContext.resolveSecret`).
|
|
9
|
+
*
|
|
10
|
+
* This module — NOT the fetchers themselves — is what imports
|
|
11
|
+
* `core/env-secret-ref`. That module transitively imports the source
|
|
12
|
+
* providers, which import the fetcher registry, so a fetcher importing it
|
|
13
|
+
* directly would close an import cycle. Keeping the dependency here lets the
|
|
14
|
+
* fetchers stay leaves while still reaching the store.
|
|
15
|
+
*
|
|
16
|
+
* Returns null for any failure — missing store, unresolvable ref, unreadable
|
|
17
|
+
* file. The underlying error is deliberately not surfaced: it can embed
|
|
18
|
+
* filesystem paths, and every failure means the same thing to a caller ("no
|
|
19
|
+
* secret"). The value itself is never logged.
|
|
20
|
+
*/
|
|
21
|
+
export function resolveSecretFromStore(ref) {
|
|
22
|
+
try {
|
|
23
|
+
const { absPath } = resolveSecretPath(ref);
|
|
24
|
+
if (!fs.existsSync(absPath))
|
|
25
|
+
return null;
|
|
26
|
+
return fs.readFileSync(absPath, "utf8").trim() || null;
|
|
27
|
+
}
|
|
28
|
+
catch {
|
|
29
|
+
return null;
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
/**
|
|
33
|
+
* The single store-backed {@link SecretResolver}, constructed here because this
|
|
34
|
+
* is the only module in the sources tree sanctioned to import
|
|
35
|
+
* `core/env-secret-ref`. Composition roots above the import cycle
|
|
36
|
+
* (`indexer/indexer.ts` for the `sync()`/bundle-update path, and the
|
|
37
|
+
* `akm import` / `akm bundle add` command handlers) inject this so in-cycle
|
|
38
|
+
* consumers depend only on the `SecretResolver` interface, never on the reader.
|
|
39
|
+
*/
|
|
40
|
+
export const storeSecretResolver = {
|
|
41
|
+
resolveSecret: resolveSecretFromStore,
|
|
42
|
+
};
|