@tenphi/akno-core 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +83 -0
- package/README.md +26 -0
- package/config/default.jsonc +548 -0
- package/dist/bench/answer-corpus.d.ts +48 -0
- package/dist/bench/answer-corpus.d.ts.map +1 -0
- package/dist/bench/answer-corpus.js +491 -0
- package/dist/bench/answer-corpus.js.map +1 -0
- package/dist/bench/answer.d.ts +156 -0
- package/dist/bench/answer.d.ts.map +1 -0
- package/dist/bench/answer.js +608 -0
- package/dist/bench/answer.js.map +1 -0
- package/dist/bench/auto-recall-answer-corpus.d.ts +21 -0
- package/dist/bench/auto-recall-answer-corpus.d.ts.map +1 -0
- package/dist/bench/auto-recall-answer-corpus.js +245 -0
- package/dist/bench/auto-recall-answer-corpus.js.map +1 -0
- package/dist/bench/auto-recall-answer.d.ts +198 -0
- package/dist/bench/auto-recall-answer.d.ts.map +1 -0
- package/dist/bench/auto-recall-answer.js +720 -0
- package/dist/bench/auto-recall-answer.js.map +1 -0
- package/dist/bench/auto-recall-corpus.d.ts +45 -0
- package/dist/bench/auto-recall-corpus.d.ts.map +1 -0
- package/dist/bench/auto-recall-corpus.js +221 -0
- package/dist/bench/auto-recall-corpus.js.map +1 -0
- package/dist/bench/auto-recall.d.ts +171 -0
- package/dist/bench/auto-recall.d.ts.map +1 -0
- package/dist/bench/auto-recall.js +651 -0
- package/dist/bench/auto-recall.js.map +1 -0
- package/dist/bench/entity-resolution.d.ts +37 -0
- package/dist/bench/entity-resolution.d.ts.map +1 -0
- package/dist/bench/entity-resolution.js +121 -0
- package/dist/bench/entity-resolution.js.map +1 -0
- package/dist/bench/graph.d.ts +57 -0
- package/dist/bench/graph.d.ts.map +1 -0
- package/dist/bench/graph.js +573 -0
- package/dist/bench/graph.js.map +1 -0
- package/dist/bench/llm-ranking-probe.d.ts +27 -0
- package/dist/bench/llm-ranking-probe.d.ts.map +1 -0
- package/dist/bench/llm-ranking-probe.js +105 -0
- package/dist/bench/llm-ranking-probe.js.map +1 -0
- package/dist/bench/merge-discovery-corpus.d.ts +26 -0
- package/dist/bench/merge-discovery-corpus.d.ts.map +1 -0
- package/dist/bench/merge-discovery-corpus.js +114 -0
- package/dist/bench/merge-discovery-corpus.js.map +1 -0
- package/dist/bench/merge-discovery-review.d.ts +81 -0
- package/dist/bench/merge-discovery-review.d.ts.map +1 -0
- package/dist/bench/merge-discovery-review.js +149 -0
- package/dist/bench/merge-discovery-review.js.map +1 -0
- package/dist/bench/merge-discovery.d.ts +122 -0
- package/dist/bench/merge-discovery.d.ts.map +1 -0
- package/dist/bench/merge-discovery.js +390 -0
- package/dist/bench/merge-discovery.js.map +1 -0
- package/dist/bench/mixed-retrieval.d.ts +35 -0
- package/dist/bench/mixed-retrieval.d.ts.map +1 -0
- package/dist/bench/mixed-retrieval.js +355 -0
- package/dist/bench/mixed-retrieval.js.map +1 -0
- package/dist/bench/ranking-corpus.d.ts +29 -0
- package/dist/bench/ranking-corpus.d.ts.map +1 -0
- package/dist/bench/ranking-corpus.js +514 -0
- package/dist/bench/ranking-corpus.js.map +1 -0
- package/dist/bench/ranking-end-to-end.d.ts +112 -0
- package/dist/bench/ranking-end-to-end.d.ts.map +1 -0
- package/dist/bench/ranking-end-to-end.js +479 -0
- package/dist/bench/ranking-end-to-end.js.map +1 -0
- package/dist/bench/ranking-latency.d.ts +71 -0
- package/dist/bench/ranking-latency.d.ts.map +1 -0
- package/dist/bench/ranking-latency.js +130 -0
- package/dist/bench/ranking-latency.js.map +1 -0
- package/dist/bench/ranking-matrix.d.ts +141 -0
- package/dist/bench/ranking-matrix.d.ts.map +1 -0
- package/dist/bench/ranking-matrix.js +526 -0
- package/dist/bench/ranking-matrix.js.map +1 -0
- package/dist/bench/ranking-review.d.ts +93 -0
- package/dist/bench/ranking-review.d.ts.map +1 -0
- package/dist/bench/ranking-review.js +249 -0
- package/dist/bench/ranking-review.js.map +1 -0
- package/dist/bench/ranking.d.ts +122 -0
- package/dist/bench/ranking.d.ts.map +1 -0
- package/dist/bench/ranking.js +547 -0
- package/dist/bench/ranking.js.map +1 -0
- package/dist/bench.d.ts +60 -0
- package/dist/bench.d.ts.map +1 -0
- package/dist/bench.js +167 -0
- package/dist/bench.js.map +1 -0
- package/dist/config/jsonc.d.ts +8 -0
- package/dist/config/jsonc.d.ts.map +1 -0
- package/dist/config/jsonc.js +85 -0
- package/dist/config/jsonc.js.map +1 -0
- package/dist/config/load.d.ts +43 -0
- package/dist/config/load.d.ts.map +1 -0
- package/dist/config/load.js +608 -0
- package/dist/config/load.js.map +1 -0
- package/dist/config/paths.d.ts +10 -0
- package/dist/config/paths.d.ts.map +1 -0
- package/dist/config/paths.js +38 -0
- package/dist/config/paths.js.map +1 -0
- package/dist/config/schema.d.ts +698 -0
- package/dist/config/schema.d.ts.map +1 -0
- package/dist/config/schema.js +394 -0
- package/dist/config/schema.js.map +1 -0
- package/dist/config/write-rules.d.ts +41 -0
- package/dist/config/write-rules.d.ts.map +1 -0
- package/dist/config/write-rules.js +225 -0
- package/dist/config/write-rules.js.map +1 -0
- package/dist/config/write-setup.d.ts +42 -0
- package/dist/config/write-setup.d.ts.map +1 -0
- package/dist/config/write-setup.js +358 -0
- package/dist/config/write-setup.js.map +1 -0
- package/dist/context.d.ts +64 -0
- package/dist/context.d.ts.map +1 -0
- package/dist/context.js +14 -0
- package/dist/context.js.map +1 -0
- package/dist/doctor.d.ts +115 -0
- package/dist/doctor.d.ts.map +1 -0
- package/dist/doctor.js +388 -0
- package/dist/doctor.js.map +1 -0
- package/dist/index/chunk.d.ts +62 -0
- package/dist/index/chunk.d.ts.map +1 -0
- package/dist/index/chunk.js +307 -0
- package/dist/index/chunk.js.map +1 -0
- package/dist/index/defer.d.ts +32 -0
- package/dist/index/defer.d.ts.map +1 -0
- package/dist/index/defer.js +66 -0
- package/dist/index/defer.js.map +1 -0
- package/dist/index/derive.d.ts +99 -0
- package/dist/index/derive.d.ts.map +1 -0
- package/dist/index/derive.js +378 -0
- package/dist/index/derive.js.map +1 -0
- package/dist/index/entity-resolution.d.ts +68 -0
- package/dist/index/entity-resolution.d.ts.map +1 -0
- package/dist/index/entity-resolution.js +305 -0
- package/dist/index/entity-resolution.js.map +1 -0
- package/dist/index/graph.d.ts +46 -0
- package/dist/index/graph.d.ts.map +1 -0
- package/dist/index/graph.js +618 -0
- package/dist/index/graph.js.map +1 -0
- package/dist/index/indexer.d.ts +301 -0
- package/dist/index/indexer.d.ts.map +1 -0
- package/dist/index/indexer.js +1398 -0
- package/dist/index/indexer.js.map +1 -0
- package/dist/index/revision-barrier.d.ts +30 -0
- package/dist/index/revision-barrier.d.ts.map +1 -0
- package/dist/index/revision-barrier.js +140 -0
- package/dist/index/revision-barrier.js.map +1 -0
- package/dist/index.d.ts +55 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +37 -0
- package/dist/index.js.map +1 -0
- package/dist/ingest/adoption-eligibility.d.ts +4 -0
- package/dist/ingest/adoption-eligibility.d.ts.map +1 -0
- package/dist/ingest/adoption-eligibility.js +18 -0
- package/dist/ingest/adoption-eligibility.js.map +1 -0
- package/dist/ingest/availability.d.ts +15 -0
- package/dist/ingest/availability.d.ts.map +1 -0
- package/dist/ingest/availability.js +37 -0
- package/dist/ingest/availability.js.map +1 -0
- package/dist/ingest/extract.d.ts +58 -0
- package/dist/ingest/extract.d.ts.map +1 -0
- package/dist/ingest/extract.js +255 -0
- package/dist/ingest/extract.js.map +1 -0
- package/dist/ingest/fetch.d.ts +35 -0
- package/dist/ingest/fetch.d.ts.map +1 -0
- package/dist/ingest/fetch.js +142 -0
- package/dist/ingest/fetch.js.map +1 -0
- package/dist/ingest/inbox.d.ts +53 -0
- package/dist/ingest/inbox.d.ts.map +1 -0
- package/dist/ingest/inbox.js +110 -0
- package/dist/ingest/inbox.js.map +1 -0
- package/dist/ingest/name.d.ts +72 -0
- package/dist/ingest/name.d.ts.map +1 -0
- package/dist/ingest/name.js +208 -0
- package/dist/ingest/name.js.map +1 -0
- package/dist/ingest/parts.d.ts +78 -0
- package/dist/ingest/parts.d.ts.map +1 -0
- package/dist/ingest/parts.js +52 -0
- package/dist/ingest/parts.js.map +1 -0
- package/dist/ingest/rendition.d.ts +75 -0
- package/dist/ingest/rendition.d.ts.map +1 -0
- package/dist/ingest/rendition.js +92 -0
- package/dist/ingest/rendition.js.map +1 -0
- package/dist/ingest/store.d.ts +66 -0
- package/dist/ingest/store.d.ts.map +1 -0
- package/dist/ingest/store.js +92 -0
- package/dist/ingest/store.js.map +1 -0
- package/dist/kb/folders.d.ts +37 -0
- package/dist/kb/folders.d.ts.map +1 -0
- package/dist/kb/folders.js +133 -0
- package/dist/kb/folders.js.map +1 -0
- package/dist/kb/frontmatter.d.ts +91 -0
- package/dist/kb/frontmatter.d.ts.map +1 -0
- package/dist/kb/frontmatter.js +274 -0
- package/dist/kb/frontmatter.js.map +1 -0
- package/dist/kb/line-facts.d.ts +29 -0
- package/dist/kb/line-facts.d.ts.map +1 -0
- package/dist/kb/line-facts.js +23 -0
- package/dist/kb/line-facts.js.map +1 -0
- package/dist/kb/page.d.ts +92 -0
- package/dist/kb/page.d.ts.map +1 -0
- package/dist/kb/page.js +243 -0
- package/dist/kb/page.js.map +1 -0
- package/dist/kb/scan.d.ts +38 -0
- package/dist/kb/scan.d.ts.map +1 -0
- package/dist/kb/scan.js +109 -0
- package/dist/kb/scan.js.map +1 -0
- package/dist/kb/words.d.ts +30 -0
- package/dist/kb/words.d.ts.map +1 -0
- package/dist/kb/words.js +107 -0
- package/dist/kb/words.js.map +1 -0
- package/dist/maintenance/adopt.d.ts +65 -0
- package/dist/maintenance/adopt.d.ts.map +1 -0
- package/dist/maintenance/adopt.js +238 -0
- package/dist/maintenance/adopt.js.map +1 -0
- package/dist/maintenance/budget.d.ts +55 -0
- package/dist/maintenance/budget.d.ts.map +1 -0
- package/dist/maintenance/budget.js +61 -0
- package/dist/maintenance/budget.js.map +1 -0
- package/dist/maintenance/conflicts.d.ts +105 -0
- package/dist/maintenance/conflicts.d.ts.map +1 -0
- package/dist/maintenance/conflicts.js +368 -0
- package/dist/maintenance/conflicts.js.map +1 -0
- package/dist/maintenance/contradictions.d.ts +34 -0
- package/dist/maintenance/contradictions.d.ts.map +1 -0
- package/dist/maintenance/contradictions.js +199 -0
- package/dist/maintenance/contradictions.js.map +1 -0
- package/dist/maintenance/curate.d.ts +156 -0
- package/dist/maintenance/curate.d.ts.map +1 -0
- package/dist/maintenance/curate.js +2070 -0
- package/dist/maintenance/curate.js.map +1 -0
- package/dist/maintenance/dream.d.ts +139 -0
- package/dist/maintenance/dream.d.ts.map +1 -0
- package/dist/maintenance/dream.js +1632 -0
- package/dist/maintenance/dream.js.map +1 -0
- package/dist/maintenance/graph-candidates.d.ts +19 -0
- package/dist/maintenance/graph-candidates.d.ts.map +1 -0
- package/dist/maintenance/graph-candidates.js +120 -0
- package/dist/maintenance/graph-candidates.js.map +1 -0
- package/dist/maintenance/housekeeping.d.ts +81 -0
- package/dist/maintenance/housekeeping.d.ts.map +1 -0
- package/dist/maintenance/housekeeping.js +281 -0
- package/dist/maintenance/housekeeping.js.map +1 -0
- package/dist/maintenance/link-repairs.d.ts +58 -0
- package/dist/maintenance/link-repairs.d.ts.map +1 -0
- package/dist/maintenance/link-repairs.js +281 -0
- package/dist/maintenance/link-repairs.js.map +1 -0
- package/dist/maintenance/log.d.ts +41 -0
- package/dist/maintenance/log.d.ts.map +1 -0
- package/dist/maintenance/log.js +72 -0
- package/dist/maintenance/log.js.map +1 -0
- package/dist/maintenance/managed-item-routing.d.ts +48 -0
- package/dist/maintenance/managed-item-routing.d.ts.map +1 -0
- package/dist/maintenance/managed-item-routing.js +281 -0
- package/dist/maintenance/managed-item-routing.js.map +1 -0
- package/dist/maintenance/managed-item-sources.d.ts +38 -0
- package/dist/maintenance/managed-item-sources.d.ts.map +1 -0
- package/dist/maintenance/managed-item-sources.js +324 -0
- package/dist/maintenance/managed-item-sources.js.map +1 -0
- package/dist/maintenance/managed-items.d.ts +162 -0
- package/dist/maintenance/managed-items.d.ts.map +1 -0
- package/dist/maintenance/managed-items.js +1329 -0
- package/dist/maintenance/managed-items.js.map +1 -0
- package/dist/maintenance/merge-classifier.d.ts +19 -0
- package/dist/maintenance/merge-classifier.d.ts.map +1 -0
- package/dist/maintenance/merge-classifier.js +49 -0
- package/dist/maintenance/merge-classifier.js.map +1 -0
- package/dist/maintenance/model-telemetry.d.ts +45 -0
- package/dist/maintenance/model-telemetry.d.ts.map +1 -0
- package/dist/maintenance/model-telemetry.js +107 -0
- package/dist/maintenance/model-telemetry.js.map +1 -0
- package/dist/maintenance/observe.d.ts +72 -0
- package/dist/maintenance/observe.d.ts.map +1 -0
- package/dist/maintenance/observe.js +265 -0
- package/dist/maintenance/observe.js.map +1 -0
- package/dist/maintenance/path-policy.d.ts +83 -0
- package/dist/maintenance/path-policy.d.ts.map +1 -0
- package/dist/maintenance/path-policy.js +360 -0
- package/dist/maintenance/path-policy.js.map +1 -0
- package/dist/maintenance/plans.d.ts +366 -0
- package/dist/maintenance/plans.d.ts.map +1 -0
- package/dist/maintenance/plans.js +4060 -0
- package/dist/maintenance/plans.js.map +1 -0
- package/dist/maintenance/profile.d.ts +33 -0
- package/dist/maintenance/profile.d.ts.map +1 -0
- package/dist/maintenance/profile.js +99 -0
- package/dist/maintenance/profile.js.map +1 -0
- package/dist/maintenance/recovery.d.ts +38 -0
- package/dist/maintenance/recovery.d.ts.map +1 -0
- package/dist/maintenance/recovery.js +217 -0
- package/dist/maintenance/recovery.js.map +1 -0
- package/dist/maintenance/repair.d.ts +37 -0
- package/dist/maintenance/repair.d.ts.map +1 -0
- package/dist/maintenance/repair.js +183 -0
- package/dist/maintenance/repair.js.map +1 -0
- package/dist/maintenance/rule-drift.d.ts +104 -0
- package/dist/maintenance/rule-drift.d.ts.map +1 -0
- package/dist/maintenance/rule-drift.js +507 -0
- package/dist/maintenance/rule-drift.js.map +1 -0
- package/dist/maintenance/run-verification.d.ts +44 -0
- package/dist/maintenance/run-verification.d.ts.map +1 -0
- package/dist/maintenance/run-verification.js +253 -0
- package/dist/maintenance/run-verification.js.map +1 -0
- package/dist/maintenance/runs.d.ts +142 -0
- package/dist/maintenance/runs.d.ts.map +1 -0
- package/dist/maintenance/runs.js +439 -0
- package/dist/maintenance/runs.js.map +1 -0
- package/dist/maintenance/semantic-merge-discovery.d.ts +54 -0
- package/dist/maintenance/semantic-merge-discovery.d.ts.map +1 -0
- package/dist/maintenance/semantic-merge-discovery.js +308 -0
- package/dist/maintenance/semantic-merge-discovery.js.map +1 -0
- package/dist/maintenance/temporal.d.ts +44 -0
- package/dist/maintenance/temporal.d.ts.map +1 -0
- package/dist/maintenance/temporal.js +295 -0
- package/dist/maintenance/temporal.js.map +1 -0
- package/dist/models/client.d.ts +214 -0
- package/dist/models/client.d.ts.map +1 -0
- package/dist/models/client.js +904 -0
- package/dist/models/client.js.map +1 -0
- package/dist/models/provider-api.d.ts +28 -0
- package/dist/models/provider-api.d.ts.map +1 -0
- package/dist/models/provider-api.js +320 -0
- package/dist/models/provider-api.js.map +1 -0
- package/dist/open.d.ts +163 -0
- package/dist/open.d.ts.map +1 -0
- package/dist/open.js +382 -0
- package/dist/open.js.map +1 -0
- package/dist/ops/adopt.d.ts +12 -0
- package/dist/ops/adopt.d.ts.map +1 -0
- package/dist/ops/adopt.js +138 -0
- package/dist/ops/adopt.js.map +1 -0
- package/dist/ops/answer.d.ts +29 -0
- package/dist/ops/answer.d.ts.map +1 -0
- package/dist/ops/answer.js +557 -0
- package/dist/ops/answer.js.map +1 -0
- package/dist/ops/context.d.ts +18 -0
- package/dist/ops/context.d.ts.map +1 -0
- package/dist/ops/context.js +713 -0
- package/dist/ops/context.js.map +1 -0
- package/dist/ops/folder.d.ts +23 -0
- package/dist/ops/folder.d.ts.map +1 -0
- package/dist/ops/folder.js +130 -0
- package/dist/ops/folder.js.map +1 -0
- package/dist/ops/forget.d.ts +14 -0
- package/dist/ops/forget.d.ts.map +1 -0
- package/dist/ops/forget.js +212 -0
- package/dist/ops/forget.js.map +1 -0
- package/dist/ops/graph.d.ts +10 -0
- package/dist/ops/graph.d.ts.map +1 -0
- package/dist/ops/graph.js +480 -0
- package/dist/ops/graph.js.map +1 -0
- package/dist/ops/ingest.d.ts +40 -0
- package/dist/ops/ingest.d.ts.map +1 -0
- package/dist/ops/ingest.js +434 -0
- package/dist/ops/ingest.js.map +1 -0
- package/dist/ops/list.d.ts +9 -0
- package/dist/ops/list.d.ts.map +1 -0
- package/dist/ops/list.js +217 -0
- package/dist/ops/list.js.map +1 -0
- package/dist/ops/move.d.ts +20 -0
- package/dist/ops/move.d.ts.map +1 -0
- package/dist/ops/move.js +173 -0
- package/dist/ops/move.js.map +1 -0
- package/dist/ops/read.d.ts +9 -0
- package/dist/ops/read.d.ts.map +1 -0
- package/dist/ops/read.js +271 -0
- package/dist/ops/read.js.map +1 -0
- package/dist/ops/recall.d.ts +13 -0
- package/dist/ops/recall.d.ts.map +1 -0
- package/dist/ops/recall.js +297 -0
- package/dist/ops/recall.js.map +1 -0
- package/dist/ops/remember.d.ts +34 -0
- package/dist/ops/remember.d.ts.map +1 -0
- package/dist/ops/remember.js +660 -0
- package/dist/ops/remember.js.map +1 -0
- package/dist/ops/timeline.d.ts +11 -0
- package/dist/ops/timeline.d.ts.map +1 -0
- package/dist/ops/timeline.js +111 -0
- package/dist/ops/timeline.js.map +1 -0
- package/dist/ops/undo.d.ts +9 -0
- package/dist/ops/undo.d.ts.map +1 -0
- package/dist/ops/undo.js +38 -0
- package/dist/ops/undo.js.map +1 -0
- package/dist/ops/write.d.ts +55 -0
- package/dist/ops/write.d.ts.map +1 -0
- package/dist/ops/write.js +463 -0
- package/dist/ops/write.js.map +1 -0
- package/dist/recall/assemble.d.ts +81 -0
- package/dist/recall/assemble.d.ts.map +1 -0
- package/dist/recall/assemble.js +603 -0
- package/dist/recall/assemble.js.map +1 -0
- package/dist/recall/expand.d.ts +55 -0
- package/dist/recall/expand.d.ts.map +1 -0
- package/dist/recall/expand.js +282 -0
- package/dist/recall/expand.js.map +1 -0
- package/dist/recall/graph-arm.d.ts +17 -0
- package/dist/recall/graph-arm.d.ts.map +1 -0
- package/dist/recall/graph-arm.js +227 -0
- package/dist/recall/graph-arm.js.map +1 -0
- package/dist/recall/llm-rerank.d.ts +55 -0
- package/dist/recall/llm-rerank.d.ts.map +1 -0
- package/dist/recall/llm-rerank.js +213 -0
- package/dist/recall/llm-rerank.js.map +1 -0
- package/dist/recall/reranker-calibration.d.ts +23 -0
- package/dist/recall/reranker-calibration.d.ts.map +1 -0
- package/dist/recall/reranker-calibration.js +161 -0
- package/dist/recall/reranker-calibration.js.map +1 -0
- package/dist/recall/search.d.ts +107 -0
- package/dist/recall/search.d.ts.map +1 -0
- package/dist/recall/search.js +530 -0
- package/dist/recall/search.js.map +1 -0
- package/dist/reserved.d.ts +39 -0
- package/dist/reserved.d.ts.map +1 -0
- package/dist/reserved.js +63 -0
- package/dist/reserved.js.map +1 -0
- package/dist/rules/compile.d.ts +27 -0
- package/dist/rules/compile.d.ts.map +1 -0
- package/dist/rules/compile.js +105 -0
- package/dist/rules/compile.js.map +1 -0
- package/dist/setup/model-free.d.ts +15 -0
- package/dist/setup/model-free.d.ts.map +1 -0
- package/dist/setup/model-free.js +26 -0
- package/dist/setup/model-free.js.map +1 -0
- package/dist/setup/openai.d.ts +46 -0
- package/dist/setup/openai.d.ts.map +1 -0
- package/dist/setup/openai.js +165 -0
- package/dist/setup/openai.js.map +1 -0
- package/dist/store/db.d.ts +61 -0
- package/dist/store/db.d.ts.map +1 -0
- package/dist/store/db.js +265 -0
- package/dist/store/db.js.map +1 -0
- package/dist/store/ids.d.ts +29 -0
- package/dist/store/ids.d.ts.map +1 -0
- package/dist/store/ids.js +0 -0
- package/dist/store/ids.js.map +1 -0
- package/dist/store/migrations.d.ts +54 -0
- package/dist/store/migrations.d.ts.map +1 -0
- package/dist/store/migrations.js +923 -0
- package/dist/store/migrations.js.map +1 -0
- package/dist/store/vectors.d.ts +34 -0
- package/dist/store/vectors.d.ts.map +1 -0
- package/dist/store/vectors.js +158 -0
- package/dist/store/vectors.js.map +1 -0
- package/dist/timeline/documents.d.ts +23 -0
- package/dist/timeline/documents.d.ts.map +1 -0
- package/dist/timeline/documents.js +262 -0
- package/dist/timeline/documents.js.map +1 -0
- package/dist/watch/watcher.d.ts +43 -0
- package/dist/watch/watcher.d.ts.map +1 -0
- package/dist/watch/watcher.js +143 -0
- package/dist/watch/watcher.js.map +1 -0
- package/dist/write/atomic.d.ts +34 -0
- package/dist/write/atomic.d.ts.map +1 -0
- package/dist/write/atomic.js +58 -0
- package/dist/write/atomic.js.map +1 -0
- package/dist/write/conflict.d.ts +43 -0
- package/dist/write/conflict.d.ts.map +1 -0
- package/dist/write/conflict.js +149 -0
- package/dist/write/conflict.js.map +1 -0
- package/dist/write/edit.d.ts +59 -0
- package/dist/write/edit.d.ts.map +1 -0
- package/dist/write/edit.js +231 -0
- package/dist/write/edit.js.map +1 -0
- package/dist/write/gate.d.ts +59 -0
- package/dist/write/gate.d.ts.map +1 -0
- package/dist/write/gate.js +99 -0
- package/dist/write/gate.js.map +1 -0
- package/dist/write/journal.d.ts +90 -0
- package/dist/write/journal.d.ts.map +1 -0
- package/dist/write/journal.js +187 -0
- package/dist/write/journal.js.map +1 -0
- package/dist/write/ledger.d.ts +28 -0
- package/dist/write/ledger.d.ts.map +1 -0
- package/dist/write/ledger.js +141 -0
- package/dist/write/ledger.js.map +1 -0
- package/dist/write/placement.d.ts +29 -0
- package/dist/write/placement.d.ts.map +1 -0
- package/dist/write/placement.js +140 -0
- package/dist/write/placement.js.map +1 -0
- package/dist/write/remember-fallback.d.ts +18 -0
- package/dist/write/remember-fallback.d.ts.map +1 -0
- package/dist/write/remember-fallback.js +48 -0
- package/dist/write/remember-fallback.js.map +1 -0
- package/dist/write/retain.d.ts +61 -0
- package/dist/write/retain.d.ts.map +1 -0
- package/dist/write/retain.js +272 -0
- package/dist/write/retain.js.map +1 -0
- package/package.json +50 -0
- package/swift/extract.swift +232 -0
|
@@ -0,0 +1,904 @@
|
|
|
1
|
+
import { createHash } from 'node:crypto';
|
|
2
|
+
import { z } from 'zod';
|
|
3
|
+
const UNSUPPORTED_MAX_TOKENS = /unsupported parameter.*max_tokens|use 'max_completion_tokens'/i;
|
|
4
|
+
/** The rungs in order, so a demotion can be checked for direction rather than assumed. */
|
|
5
|
+
const SCHEMA_RUNGS = ['schema', 'json_schema', 'plain'];
|
|
6
|
+
const UNSUPPORTED_SCHEMA = /response_format|unsupported.*schema|unknown parameter.*schema|invalid.*schema/i;
|
|
7
|
+
/** Statuses worth trying again. Everything else is a configuration error that a
|
|
8
|
+
* second identical request will reproduce exactly. */
|
|
9
|
+
const RETRYABLE_STATUS = new Set([408, 409, 425, 429, 500, 502, 503, 504]);
|
|
10
|
+
/** Provider errors are useful diagnostics but may echo credentials or tenant identifiers. */
|
|
11
|
+
export function redactProviderError(value, exactSecrets = []) {
|
|
12
|
+
let redacted = value;
|
|
13
|
+
for (const secret of [...new Set(exactSecrets.filter((entry) => entry.length > 0))].sort((left, right) => right.length - left.length)) {
|
|
14
|
+
redacted = redacted.replaceAll(secret, '<redacted>');
|
|
15
|
+
}
|
|
16
|
+
return redacted
|
|
17
|
+
.replace(/\bBearer\s+[^\s"']+/gi, 'Bearer <redacted>')
|
|
18
|
+
.replace(/\bsk-[A-Za-z0-9_-]{8,}\b/g, '<redacted>')
|
|
19
|
+
.replace(/\b(proj|org|user|acct)_[A-Za-z0-9_-]+\b/g, '$1_<redacted>');
|
|
20
|
+
}
|
|
21
|
+
export class ModelClient {
|
|
22
|
+
#role;
|
|
23
|
+
#outcomeObserver;
|
|
24
|
+
/** Learned on first rejection and shared by observed views of this role client. */
|
|
25
|
+
#compatibility = { tokenParam: 'max_tokens', schemaMode: 'schema' };
|
|
26
|
+
constructor(role, outcomeObserver = null) {
|
|
27
|
+
this.#role = role;
|
|
28
|
+
this.#outcomeObserver = outcomeObserver;
|
|
29
|
+
}
|
|
30
|
+
/**
|
|
31
|
+
* Isolate accounting to one workflow without mutating the process-wide role client.
|
|
32
|
+
* Compatibility discoveries are shared so instrumentation does not add a probe to a warm client.
|
|
33
|
+
*/
|
|
34
|
+
withOutcomeObserver(observer) {
|
|
35
|
+
const observed = new ModelClient(this.#role, observer);
|
|
36
|
+
observed.#compatibility = this.#compatibility;
|
|
37
|
+
return observed;
|
|
38
|
+
}
|
|
39
|
+
/**
|
|
40
|
+
* Reclassify the preceding successful call when its content cannot satisfy the caller's
|
|
41
|
+
* contract. No response text or validation detail crosses this telemetry boundary.
|
|
42
|
+
*/
|
|
43
|
+
reportInvalidResponse() {
|
|
44
|
+
this.emitObservation({
|
|
45
|
+
event: 'semantic_failure',
|
|
46
|
+
role: this.#role.role,
|
|
47
|
+
modelId: this.#role.id,
|
|
48
|
+
failure: 'bad_response',
|
|
49
|
+
degradedReason: this.degradedReason({ reason: 'bad_response' }),
|
|
50
|
+
});
|
|
51
|
+
}
|
|
52
|
+
get available() {
|
|
53
|
+
return (this.#role.enabled &&
|
|
54
|
+
this.#role.provider !== null &&
|
|
55
|
+
this.#role.id !== null &&
|
|
56
|
+
!this.generativeTransportUnresolved());
|
|
57
|
+
}
|
|
58
|
+
get unavailableReason() {
|
|
59
|
+
if (this.generativeTransportUnresolved()) {
|
|
60
|
+
const provider = this.#role.provider;
|
|
61
|
+
return (provider.apiResolutionError ??
|
|
62
|
+
`provider "${provider.name}" has unresolved api:auto; run a model probe or choose an explicit api`);
|
|
63
|
+
}
|
|
64
|
+
return this.#role.unavailableReason;
|
|
65
|
+
}
|
|
66
|
+
get modelId() {
|
|
67
|
+
return this.#role.id;
|
|
68
|
+
}
|
|
69
|
+
get role() {
|
|
70
|
+
return this.#role.role;
|
|
71
|
+
}
|
|
72
|
+
/** Native cross-encoder endpoint unless the role explicitly opts into prompted ranking. */
|
|
73
|
+
get rerankerMode() {
|
|
74
|
+
return this.#role.rerankerMode ?? 'endpoint';
|
|
75
|
+
}
|
|
76
|
+
get reasoningEffort() {
|
|
77
|
+
return this.#role.reasoningEffort;
|
|
78
|
+
}
|
|
79
|
+
/** Stable without including credentials; used only to key derived calibration data. */
|
|
80
|
+
get endpointFingerprint() {
|
|
81
|
+
if (!this.available || !this.#role.provider || !this.#role.id)
|
|
82
|
+
return null;
|
|
83
|
+
return createHash('sha256')
|
|
84
|
+
.update([
|
|
85
|
+
this.#role.role,
|
|
86
|
+
this.#role.provider.name,
|
|
87
|
+
this.#role.provider.baseUrl,
|
|
88
|
+
this.#role.provider.api,
|
|
89
|
+
this.#role.id,
|
|
90
|
+
].join('\0'))
|
|
91
|
+
.digest('hex');
|
|
92
|
+
}
|
|
93
|
+
/** Embeddings and native cross-encoder reranking do not use the generative adapter. */
|
|
94
|
+
generativeTransportUnresolved() {
|
|
95
|
+
if (this.#role.provider?.api !== 'auto')
|
|
96
|
+
return false;
|
|
97
|
+
return (this.#role.role !== 'embedding' && !(this.#role.role === 'reranker' && this.rerankerMode === 'endpoint'));
|
|
98
|
+
}
|
|
99
|
+
/** True when the user asked for this role, whether or not it resolved. */
|
|
100
|
+
get requested() {
|
|
101
|
+
return this.#role.requested;
|
|
102
|
+
}
|
|
103
|
+
/** Maps an outcome onto the vocabulary a caller branches on. */
|
|
104
|
+
degradedReason(outcome) {
|
|
105
|
+
return degradedReasonFor(this.#role.role, outcome.reason ?? 'unavailable');
|
|
106
|
+
}
|
|
107
|
+
async post(endpoint, body, timeoutMs) {
|
|
108
|
+
const started = performance.now();
|
|
109
|
+
if (!this.available || !this.#role.provider) {
|
|
110
|
+
return {
|
|
111
|
+
ok: false,
|
|
112
|
+
value: null,
|
|
113
|
+
reason: 'unavailable',
|
|
114
|
+
error: this.unavailableReason ?? 'model unavailable',
|
|
115
|
+
latencyMs: 0,
|
|
116
|
+
endpointRequests: 0,
|
|
117
|
+
};
|
|
118
|
+
}
|
|
119
|
+
const headers = {
|
|
120
|
+
'content-type': 'application/json',
|
|
121
|
+
...this.#role.provider.headers,
|
|
122
|
+
};
|
|
123
|
+
if (this.#role.provider.apiKey)
|
|
124
|
+
headers.authorization = `Bearer ${this.#role.provider.apiKey}`;
|
|
125
|
+
const payload = JSON.stringify(body);
|
|
126
|
+
/**
|
|
127
|
+
* **The two deadlines bound different things, so retrying spends them differently.**
|
|
128
|
+
*
|
|
129
|
+
* A `timeoutMs` passed by the caller bounds **felt latency** — something is waiting on this
|
|
130
|
+
* answer. Only `expandQuery` passes one, and it does so precisely because a busy or cold
|
|
131
|
+
* endpoint must cost a weaker search rather than a slow one. So it is the budget for the
|
|
132
|
+
* whole sequence: retries fit inside it or do not happen, and a retrying recall can never
|
|
133
|
+
* outlast one with retrying switched off.
|
|
134
|
+
*
|
|
135
|
+
* The role's `timeout_ms` bounds **an endpoint that has stopped answering** — a backstop,
|
|
136
|
+
* not a target, and nothing is waiting on a background derivation. So it applies per
|
|
137
|
+
* attempt. Making it a total instead quietly rewrote what an operator's 300s meant: a 500
|
|
138
|
+
* arriving late into a long generation would leave the retry a fraction of the budget the
|
|
139
|
+
* number was tuned for, turning "slow error, then retry" into "slow error, then a retry
|
|
140
|
+
* that was set up to fail".
|
|
141
|
+
*
|
|
142
|
+
* What keeps the per-attempt side from running away is that a timeout is never retried, so
|
|
143
|
+
* the common slow path still costs the deadline exactly once. Only HTTP statuses retry, and
|
|
144
|
+
* the ones that occur — a rate limit, a busy llama-server — are refusals returned without
|
|
145
|
+
* doing any work, which makes a real sequence backoff-dominated and measured in seconds.
|
|
146
|
+
*/
|
|
147
|
+
const totalBudget = timeoutMs ?? null;
|
|
148
|
+
const attemptDeadline = timeoutMs ?? this.#role.timeoutMs;
|
|
149
|
+
const maxAttempts = 1 + this.#role.provider.maxRetries;
|
|
150
|
+
let last = null;
|
|
151
|
+
let endpointRequests = 0;
|
|
152
|
+
for (let attempt = 1;; attempt++) {
|
|
153
|
+
// Rounded because `performance.now()` is fractional and `AbortSignal.timeout` rejects a
|
|
154
|
+
// non-integer delay outright — which fails the request before it is sent, on every call.
|
|
155
|
+
const remaining = totalBudget === null ? attemptDeadline : Math.ceil(totalBudget - (performance.now() - started));
|
|
156
|
+
// Only reachable when a backoff consumed the budget between attempts; the
|
|
157
|
+
// pre-sleep check below normally stops it getting here.
|
|
158
|
+
if (remaining <= 0)
|
|
159
|
+
break;
|
|
160
|
+
let status = null;
|
|
161
|
+
let retryAfter = null;
|
|
162
|
+
try {
|
|
163
|
+
endpointRequests += 1;
|
|
164
|
+
const response = await fetch(`${this.#role.provider.baseUrl}${endpoint}`, {
|
|
165
|
+
method: 'POST',
|
|
166
|
+
headers,
|
|
167
|
+
body: payload,
|
|
168
|
+
// `AbortSignal.timeout` is self-clearing. A hand-rolled
|
|
169
|
+
// setTimeout+AbortController leaks a pending timer on every request that
|
|
170
|
+
// resolves before its deadline, which holds the event loop open for the
|
|
171
|
+
// full timeout and turns a 40ms CLI command into a 60-second one.
|
|
172
|
+
signal: AbortSignal.timeout(remaining),
|
|
173
|
+
});
|
|
174
|
+
if (response.ok) {
|
|
175
|
+
return {
|
|
176
|
+
ok: true,
|
|
177
|
+
value: (await response.json()),
|
|
178
|
+
latencyMs: performance.now() - started,
|
|
179
|
+
endpointRequests,
|
|
180
|
+
};
|
|
181
|
+
}
|
|
182
|
+
status = response.status;
|
|
183
|
+
retryAfter = response.headers.get('retry-after');
|
|
184
|
+
const detail = redactProviderError(await response.text().catch(() => ''), [
|
|
185
|
+
this.#role.provider.apiKey ?? '',
|
|
186
|
+
...Object.values(this.#role.provider.headers),
|
|
187
|
+
]).slice(0, 300);
|
|
188
|
+
last = {
|
|
189
|
+
ok: false,
|
|
190
|
+
value: null,
|
|
191
|
+
reason: 'request_failed',
|
|
192
|
+
error: `${this.#role.role} endpoint returned ${status}${detail ? `: ${detail}` : ''}`,
|
|
193
|
+
latencyMs: performance.now() - started,
|
|
194
|
+
endpointRequests,
|
|
195
|
+
};
|
|
196
|
+
}
|
|
197
|
+
catch (err) {
|
|
198
|
+
// `AbortSignal.timeout` rejects with TimeoutError, not AbortError.
|
|
199
|
+
const timedOut = err instanceof Error && (err.name === 'TimeoutError' || err.name === 'AbortError');
|
|
200
|
+
/**
|
|
201
|
+
* **Neither a timeout nor a transport error is retried, and both omissions are
|
|
202
|
+
* deliberate.**
|
|
203
|
+
*
|
|
204
|
+
* A timeout means this attempt spent its whole deadline without an answer, and the
|
|
205
|
+
* callers that care already have a better move than repetition: `derivePage` falls back
|
|
206
|
+
* to asking for the summary alone, which is cheaper *and* likelier to succeed than the
|
|
207
|
+
* identical 2400-token request.
|
|
208
|
+
*
|
|
209
|
+
* A transport error is almost always a refused connection, meaning nothing is
|
|
210
|
+
* listening. Retrying that triples how long `doctor` takes to report the one thing
|
|
211
|
+
* the operator needs to hear.
|
|
212
|
+
*/
|
|
213
|
+
return {
|
|
214
|
+
ok: false,
|
|
215
|
+
value: null,
|
|
216
|
+
reason: timedOut ? 'timeout' : 'request_failed',
|
|
217
|
+
error: withEarlier(timedOut
|
|
218
|
+
? `${this.#role.role} timed out after ${remaining}ms`
|
|
219
|
+
: `${this.#role.role} request failed: ${err instanceof Error ? err.message : String(err)}`, last),
|
|
220
|
+
latencyMs: performance.now() - started,
|
|
221
|
+
endpointRequests,
|
|
222
|
+
};
|
|
223
|
+
}
|
|
224
|
+
if (attempt >= maxAttempts || status === null || !RETRYABLE_STATUS.has(status))
|
|
225
|
+
break;
|
|
226
|
+
const wait = backoffMs(attempt, retryAfter, Math.random());
|
|
227
|
+
// A backoff that would outlast a total budget is not a backoff, it is a slower failure.
|
|
228
|
+
// With no total budget there is nothing for it to overrun — `backoffMs` caps itself.
|
|
229
|
+
if (totalBudget !== null && performance.now() - started + wait >= totalBudget)
|
|
230
|
+
break;
|
|
231
|
+
await sleep(wait);
|
|
232
|
+
}
|
|
233
|
+
// Reached with `last` set on an exhausted or unretryable failure, and without it only when
|
|
234
|
+
// a total budget ran out before a single attempt could be made.
|
|
235
|
+
return last
|
|
236
|
+
? { ...last, endpointRequests }
|
|
237
|
+
: {
|
|
238
|
+
ok: false,
|
|
239
|
+
value: null,
|
|
240
|
+
reason: 'timeout',
|
|
241
|
+
error: `${this.#role.role} had no time left to call: ${attemptDeadline}ms was already spent`,
|
|
242
|
+
latencyMs: performance.now() - started,
|
|
243
|
+
endpointRequests,
|
|
244
|
+
};
|
|
245
|
+
}
|
|
246
|
+
/**
|
|
247
|
+
* Batched because embedding 223 pages one request at a time is dominated by
|
|
248
|
+
* round trips, not by the model.
|
|
249
|
+
*/
|
|
250
|
+
async embed(inputs) {
|
|
251
|
+
if (inputs.length === 0)
|
|
252
|
+
return { ok: true, value: [], latencyMs: 0 };
|
|
253
|
+
const result = await this.post('/embeddings', { model: this.#role.id, input: inputs, encoding_format: 'float' });
|
|
254
|
+
if (!result.ok || !result.value)
|
|
255
|
+
return { ...result, value: null };
|
|
256
|
+
const vectors = new Array(inputs.length);
|
|
257
|
+
for (const entry of result.value.data) {
|
|
258
|
+
const raw = entry.embedding;
|
|
259
|
+
// A base64 body is legal for `encoding_format: "base64"` and some servers
|
|
260
|
+
// send it regardless of what was asked for.
|
|
261
|
+
const values = typeof raw === 'string' ? decodeBase64Floats(raw) : Float32Array.from(raw);
|
|
262
|
+
vectors[entry.index ?? 0] = values;
|
|
263
|
+
}
|
|
264
|
+
if (vectors.some((v) => !v)) {
|
|
265
|
+
return {
|
|
266
|
+
ok: false,
|
|
267
|
+
value: null,
|
|
268
|
+
reason: 'bad_response',
|
|
269
|
+
error: 'embedding response was missing entries',
|
|
270
|
+
latencyMs: result.latencyMs,
|
|
271
|
+
};
|
|
272
|
+
}
|
|
273
|
+
return { ok: true, value: vectors, latencyMs: result.latencyMs };
|
|
274
|
+
}
|
|
275
|
+
/** llama-server, TEI and vLLM all expose `/rerank` with this shape. */
|
|
276
|
+
async rerank(query, documents, topN) {
|
|
277
|
+
if (documents.length === 0)
|
|
278
|
+
return { ok: true, value: [], latencyMs: 0 };
|
|
279
|
+
const result = await this.post('/rerank', { model: this.#role.id, query, documents, top_n: topN ?? documents.length });
|
|
280
|
+
if (!result.ok || !result.value)
|
|
281
|
+
return { ...result, value: null };
|
|
282
|
+
const results = (result.value.results ?? []).map((entry) => ({
|
|
283
|
+
index: entry.index,
|
|
284
|
+
score: entry.relevance_score ?? entry.score ?? 0,
|
|
285
|
+
}));
|
|
286
|
+
return {
|
|
287
|
+
ok: true,
|
|
288
|
+
value: results,
|
|
289
|
+
latencyMs: result.latencyMs,
|
|
290
|
+
endpointRequests: result.endpointRequests,
|
|
291
|
+
};
|
|
292
|
+
}
|
|
293
|
+
async chat(messages, options = {}) {
|
|
294
|
+
const chatStarted = performance.now();
|
|
295
|
+
const wantsJson = options.json || options.schema !== undefined;
|
|
296
|
+
// Built once: `z.toJSONSchema` is cheap but this sits on the recall path, and a
|
|
297
|
+
// retry must send the identical schema rather than a second conversion of it.
|
|
298
|
+
const jsonSchema = options.schema
|
|
299
|
+
? toEndpointSchema(options.schema, { reuseDefinitions: options.reuseSchemaDefinitions ?? false })
|
|
300
|
+
: null;
|
|
301
|
+
if (this.#role.provider?.api === 'responses') {
|
|
302
|
+
return this.responses(messages, options, wantsJson, jsonSchema, chatStarted);
|
|
303
|
+
}
|
|
304
|
+
const build = (tokenParam, schemaMode) => {
|
|
305
|
+
const body = {
|
|
306
|
+
model: this.#role.id,
|
|
307
|
+
messages: options.images?.length ? withImages(messages, options.images) : messages,
|
|
308
|
+
[tokenParam]: tokenCeiling(options.maxTokens, this.#role.maxOutputTokens),
|
|
309
|
+
};
|
|
310
|
+
const reasoningEffort = options.reasoningEffort ?? this.#role.reasoningEffort;
|
|
311
|
+
if (reasoningEffort)
|
|
312
|
+
body.reasoning_effort = reasoningEffort;
|
|
313
|
+
// Some reasoning models reject a non-default temperature outright, and the
|
|
314
|
+
// value buys nothing here — every prompt in this codebase wants determinism.
|
|
315
|
+
if (tokenParam === 'max_tokens')
|
|
316
|
+
body.temperature = options.temperature ?? 0;
|
|
317
|
+
// A small model free-forms its way out of a JSON contract given the chance.
|
|
318
|
+
if (wantsJson)
|
|
319
|
+
body.response_format = responseFormat(jsonSchema, schemaMode);
|
|
320
|
+
return body;
|
|
321
|
+
};
|
|
322
|
+
let result;
|
|
323
|
+
let endpointRequests = 0;
|
|
324
|
+
// Three fixable mistakes at most — the token parameter, and two rungs down the schema
|
|
325
|
+
// ladder — so four passes is the ceiling, not a budget anything grows into.
|
|
326
|
+
for (let pass = 0;; pass++) {
|
|
327
|
+
// **What this attempt sent, captured before it goes out.**
|
|
328
|
+
//
|
|
329
|
+
// The retry decisions below used to read the shared fields back after the call, which is
|
|
330
|
+
// only correct when one call is in flight. Concurrent calls all start with the same wrong
|
|
331
|
+
// parameter and all fail; the first to notice corrects the field; and every other one then
|
|
332
|
+
// asks "did I send `max_tokens`?", reads the field the winner just fixed, sees
|
|
333
|
+
// `max_completion_tokens`, concludes the complaint was about something else, and gives up
|
|
334
|
+
// holding a 400 whose fix was already known.
|
|
335
|
+
//
|
|
336
|
+
// Measured on this install at `derive.concurrency: 4`: a service restart cost the facts of
|
|
337
|
+
// the first pages it derived, every time, because a failed derivation is stamped as derived
|
|
338
|
+
// and never retried. The fix is one call's own state, not the endpoint's.
|
|
339
|
+
const sentTokenParam = this.#compatibility.tokenParam;
|
|
340
|
+
const sentSchemaMode = this.#compatibility.schemaMode;
|
|
341
|
+
result = await this.post('/chat/completions', build(sentTokenParam, sentSchemaMode), options.timeoutMs);
|
|
342
|
+
endpointRequests += result.endpointRequests ?? 0;
|
|
343
|
+
if (result.ok || pass >= 3)
|
|
344
|
+
break;
|
|
345
|
+
// Each of these has a known, mechanical fix, and each is learned once for the
|
|
346
|
+
// life of the process rather than rediscovered per call. Both assignments are safe to
|
|
347
|
+
// repeat: they name the answer rather than stepping towards it.
|
|
348
|
+
if (sentTokenParam === 'max_tokens' && UNSUPPORTED_MAX_TOKENS.test(result.error ?? '')) {
|
|
349
|
+
this.#compatibility.tokenParam = 'max_completion_tokens';
|
|
350
|
+
continue;
|
|
351
|
+
}
|
|
352
|
+
if (jsonSchema && sentSchemaMode !== 'plain' && UNSUPPORTED_SCHEMA.test(result.error ?? '')) {
|
|
353
|
+
// Demoted from the rung *this* attempt used. Reading the shared field instead would skip a
|
|
354
|
+
// rung nobody tried — a call that sent `schema` while a concurrent one had already moved
|
|
355
|
+
// the field to `json_schema` would jump straight to `plain`, and the whole process would
|
|
356
|
+
// lose constrained decoding over a race rather than over an endpoint's actual limits.
|
|
357
|
+
// Monotonic, so a demotion another call has already discovered is never undone.
|
|
358
|
+
const next = sentSchemaMode === 'schema' ? 'json_schema' : 'plain';
|
|
359
|
+
if (SCHEMA_RUNGS.indexOf(next) > SCHEMA_RUNGS.indexOf(this.#compatibility.schemaMode)) {
|
|
360
|
+
this.#compatibility.schemaMode = next;
|
|
361
|
+
}
|
|
362
|
+
continue;
|
|
363
|
+
}
|
|
364
|
+
break;
|
|
365
|
+
}
|
|
366
|
+
if (!result.ok || !result.value) {
|
|
367
|
+
return this.observeChat({
|
|
368
|
+
...result,
|
|
369
|
+
value: null,
|
|
370
|
+
latencyMs: performance.now() - chatStarted,
|
|
371
|
+
endpointRequests,
|
|
372
|
+
});
|
|
373
|
+
}
|
|
374
|
+
const usage = reportedModelUsage(result.value.usage);
|
|
375
|
+
const content = result.value.choices?.[0]?.message?.content;
|
|
376
|
+
if (typeof content !== 'string') {
|
|
377
|
+
return this.observeChat({
|
|
378
|
+
ok: false,
|
|
379
|
+
value: null,
|
|
380
|
+
reason: 'bad_response',
|
|
381
|
+
error: 'chat response had no content',
|
|
382
|
+
latencyMs: performance.now() - chatStarted,
|
|
383
|
+
endpointRequests,
|
|
384
|
+
...(usage ? { usage } : {}),
|
|
385
|
+
});
|
|
386
|
+
}
|
|
387
|
+
return this.observeChat({
|
|
388
|
+
ok: true,
|
|
389
|
+
value: content,
|
|
390
|
+
latencyMs: performance.now() - chatStarted,
|
|
391
|
+
endpointRequests,
|
|
392
|
+
...(usage ? { usage } : {}),
|
|
393
|
+
});
|
|
394
|
+
}
|
|
395
|
+
/**
|
|
396
|
+
* OpenAI's Responses API uses different request keys and nests structured output under
|
|
397
|
+
* `text.format`. This is an explicit provider choice: a failed call never falls through to
|
|
398
|
+
* Chat Completions, which could duplicate cost and change validation semantics.
|
|
399
|
+
*/
|
|
400
|
+
async responses(messages, options, wantsJson, jsonSchema, started) {
|
|
401
|
+
const body = {
|
|
402
|
+
model: this.#role.id,
|
|
403
|
+
input: options.images?.length ? withResponseImages(messages, options.images) : messages,
|
|
404
|
+
max_output_tokens: tokenCeiling(options.maxTokens, this.#role.maxOutputTokens),
|
|
405
|
+
// Akno calls are stateless and routinely contain private memory. Do not create provider-side
|
|
406
|
+
// conversation state merely because Responses supports it.
|
|
407
|
+
store: false,
|
|
408
|
+
};
|
|
409
|
+
const reasoningEffort = options.reasoningEffort ?? this.#role.reasoningEffort;
|
|
410
|
+
if (reasoningEffort)
|
|
411
|
+
body.reasoning = { effort: reasoningEffort };
|
|
412
|
+
if (options.temperature !== undefined)
|
|
413
|
+
body.temperature = options.temperature;
|
|
414
|
+
if (wantsJson)
|
|
415
|
+
body.text = { format: responsesTextFormat(jsonSchema) };
|
|
416
|
+
const result = await this.post('/responses', body, options.timeoutMs);
|
|
417
|
+
const endpointRequests = result.endpointRequests ?? 0;
|
|
418
|
+
if (!result.ok || !result.value) {
|
|
419
|
+
return this.observeChat({
|
|
420
|
+
...result,
|
|
421
|
+
value: null,
|
|
422
|
+
latencyMs: performance.now() - started,
|
|
423
|
+
endpointRequests,
|
|
424
|
+
});
|
|
425
|
+
}
|
|
426
|
+
const usage = reportedModelUsage(result.value.usage);
|
|
427
|
+
const content = responseOutputText(result.value);
|
|
428
|
+
if (content === null) {
|
|
429
|
+
return this.observeChat({
|
|
430
|
+
ok: false,
|
|
431
|
+
value: null,
|
|
432
|
+
reason: 'bad_response',
|
|
433
|
+
error: 'Responses API response had no output text',
|
|
434
|
+
latencyMs: performance.now() - started,
|
|
435
|
+
endpointRequests,
|
|
436
|
+
...(usage ? { usage } : {}),
|
|
437
|
+
});
|
|
438
|
+
}
|
|
439
|
+
return this.observeChat({
|
|
440
|
+
ok: true,
|
|
441
|
+
value: content,
|
|
442
|
+
latencyMs: performance.now() - started,
|
|
443
|
+
endpointRequests,
|
|
444
|
+
...(usage ? { usage } : {}),
|
|
445
|
+
});
|
|
446
|
+
}
|
|
447
|
+
observeChat(outcome) {
|
|
448
|
+
this.emitObservation({
|
|
449
|
+
event: 'call',
|
|
450
|
+
role: this.#role.role,
|
|
451
|
+
modelId: this.#role.id,
|
|
452
|
+
ok: outcome.ok,
|
|
453
|
+
failure: outcome.ok ? null : (outcome.reason ?? 'bad_response'),
|
|
454
|
+
degradedReason: outcome.ok ? null : this.degradedReason(outcome),
|
|
455
|
+
latencyMs: outcome.latencyMs,
|
|
456
|
+
usage: outcome.usage ?? null,
|
|
457
|
+
});
|
|
458
|
+
return outcome;
|
|
459
|
+
}
|
|
460
|
+
emitObservation(observation) {
|
|
461
|
+
if (!this.#outcomeObserver)
|
|
462
|
+
return;
|
|
463
|
+
try {
|
|
464
|
+
this.#outcomeObserver(observation);
|
|
465
|
+
}
|
|
466
|
+
catch {
|
|
467
|
+
// Telemetry must never turn a usable model response into a failed operation.
|
|
468
|
+
}
|
|
469
|
+
}
|
|
470
|
+
/**
|
|
471
|
+
* Model warmth dominates everything else — a cold embedding server costs
|
|
472
|
+
* seconds, three orders of magnitude more than the entire database path. This
|
|
473
|
+
* is what `serve` pings to keep the endpoint from going cold, and what
|
|
474
|
+
* `doctor` measures so model latency is never confused with index latency.
|
|
475
|
+
*/
|
|
476
|
+
async ping() {
|
|
477
|
+
if (!this.available) {
|
|
478
|
+
return {
|
|
479
|
+
ok: false,
|
|
480
|
+
value: null,
|
|
481
|
+
reason: 'unavailable',
|
|
482
|
+
error: this.#role.unavailableReason ?? 'unavailable',
|
|
483
|
+
latencyMs: 0,
|
|
484
|
+
};
|
|
485
|
+
}
|
|
486
|
+
if (this.#role.role === 'embedding') {
|
|
487
|
+
const result = await this.embed(['ping']);
|
|
488
|
+
return {
|
|
489
|
+
ok: result.ok,
|
|
490
|
+
value: result.ok ? result.latencyMs : null,
|
|
491
|
+
...(result.reason ? { reason: result.reason } : {}),
|
|
492
|
+
...(result.error ? { error: result.error } : {}),
|
|
493
|
+
latencyMs: result.latencyMs,
|
|
494
|
+
};
|
|
495
|
+
}
|
|
496
|
+
if (this.#role.role === 'reranker') {
|
|
497
|
+
if (this.rerankerMode === 'llm') {
|
|
498
|
+
const result = await this.chat([{ role: 'user', content: 'Return JSON with exactly one field: {"ok":true}' }], { schema: z.object({ ok: z.boolean() }), maxTokens: 64 });
|
|
499
|
+
return {
|
|
500
|
+
ok: result.ok,
|
|
501
|
+
value: result.ok ? result.latencyMs : null,
|
|
502
|
+
...(result.reason ? { reason: result.reason } : {}),
|
|
503
|
+
...(result.error ? { error: result.error } : {}),
|
|
504
|
+
latencyMs: result.latencyMs,
|
|
505
|
+
};
|
|
506
|
+
}
|
|
507
|
+
const result = await this.rerank('ping', ['ping'], 1);
|
|
508
|
+
return {
|
|
509
|
+
ok: result.ok,
|
|
510
|
+
value: result.ok ? result.latencyMs : null,
|
|
511
|
+
...(result.reason ? { reason: result.reason } : {}),
|
|
512
|
+
...(result.error ? { error: result.error } : {}),
|
|
513
|
+
latencyMs: result.latencyMs,
|
|
514
|
+
};
|
|
515
|
+
}
|
|
516
|
+
// Not 1 token. A reasoning model spends its budget on reasoning before emitting
|
|
517
|
+
// anything, so a 1-token probe comes back as "output limit reached" and the role
|
|
518
|
+
// reads as dead when it is perfectly healthy. 64 is still trivially cheap and
|
|
519
|
+
// only ever runs on `doctor`.
|
|
520
|
+
const result = await this.chat([{ role: 'user', content: 'Reply with: ok' }], { maxTokens: 64 });
|
|
521
|
+
return {
|
|
522
|
+
ok: result.ok,
|
|
523
|
+
value: result.ok ? result.latencyMs : null,
|
|
524
|
+
...(result.reason ? { reason: result.reason } : {}),
|
|
525
|
+
...(result.error ? { error: result.error } : {}),
|
|
526
|
+
latencyMs: result.latencyMs,
|
|
527
|
+
};
|
|
528
|
+
}
|
|
529
|
+
}
|
|
530
|
+
/** OpenAI-compatible servers use either the Chat Completions or Responses token names. */
|
|
531
|
+
function reportedModelUsage(usage) {
|
|
532
|
+
if (!usage)
|
|
533
|
+
return null;
|
|
534
|
+
const inputTokens = tokenCount(usage.prompt_tokens ?? usage.input_tokens);
|
|
535
|
+
const outputTokens = tokenCount(usage.completion_tokens ?? usage.output_tokens);
|
|
536
|
+
const totalTokens = tokenCount(usage.total_tokens);
|
|
537
|
+
const cachedInputTokens = tokenCount(usage.prompt_tokens_details?.cached_tokens ?? usage.input_tokens_details?.cached_tokens);
|
|
538
|
+
const reasoningOutputTokens = tokenCount(usage.completion_tokens_details?.reasoning_tokens ?? usage.output_tokens_details?.reasoning_tokens);
|
|
539
|
+
if (inputTokens === null && outputTokens === null && totalTokens === null)
|
|
540
|
+
return null;
|
|
541
|
+
return {
|
|
542
|
+
inputTokens,
|
|
543
|
+
outputTokens,
|
|
544
|
+
totalTokens,
|
|
545
|
+
...(cachedInputTokens === null ? {} : { cachedInputTokens }),
|
|
546
|
+
...(reasoningOutputTokens === null ? {} : { reasoningOutputTokens }),
|
|
547
|
+
};
|
|
548
|
+
}
|
|
549
|
+
/** Raw HTTP responses do not expose the SDK's computed `output_text` on every server. */
|
|
550
|
+
function responseOutputText(response) {
|
|
551
|
+
if (typeof response.output_text === 'string')
|
|
552
|
+
return response.output_text;
|
|
553
|
+
const parts = [];
|
|
554
|
+
for (const item of response.output ?? []) {
|
|
555
|
+
for (const content of item.content ?? []) {
|
|
556
|
+
if (content.type === 'output_text' && typeof content.text === 'string')
|
|
557
|
+
parts.push(content.text);
|
|
558
|
+
}
|
|
559
|
+
}
|
|
560
|
+
return parts.length > 0 ? parts.join('') : null;
|
|
561
|
+
}
|
|
562
|
+
function tokenCount(value) {
|
|
563
|
+
return typeof value === 'number' && Number.isInteger(value) && value >= 0 ? value : null;
|
|
564
|
+
}
|
|
565
|
+
/**
|
|
566
|
+
* The single place a model failure becomes the vocabulary a caller branches on.
|
|
567
|
+
* `unavailable` means the role was never configured or is switched off — the
|
|
568
|
+
* result is weaker but the knowledge base is intact; everything else means a
|
|
569
|
+
* configured model did not answer, which an operator needs to see.
|
|
570
|
+
*/
|
|
571
|
+
/**
|
|
572
|
+
* The lower of the two ceilings, not one or the other.
|
|
573
|
+
*
|
|
574
|
+
* They mean different things and both are limits. The call's value is what *this task* can possibly
|
|
575
|
+
* need — 64 for a ping, 200 for a conflict verdict, 2400 for a full page derivation. The role's is
|
|
576
|
+
* what *this deployment* is willing to be asked for. Honouring only the call made
|
|
577
|
+
* `max_output_tokens` decorative for the biggest thing that should respect it; honouring only the
|
|
578
|
+
* role would hand a 2400-token budget to a call that needs 200.
|
|
579
|
+
*
|
|
580
|
+
* A role configured below what a task needs truncates that task, which is the honest consequence of
|
|
581
|
+
* configuring it that way — and the reason the committed default is set by what derivation's JSON
|
|
582
|
+
* needs rather than by a round number.
|
|
583
|
+
*/
|
|
584
|
+
function sleep(ms) {
|
|
585
|
+
return new Promise((resolve) => setTimeout(resolve, ms));
|
|
586
|
+
}
|
|
587
|
+
/**
|
|
588
|
+
* Keeps the failure that *caused* the retry attached to the failure that ended the call.
|
|
589
|
+
*
|
|
590
|
+
* Without it a retried rate limit that then times out is reported as a plain timeout, and the
|
|
591
|
+
* two have opposite fixes: a timeout says raise the deadline or use a faster model, a 429 says
|
|
592
|
+
* raise `max_retries` or slow the caller down. This string is what `doctor` prints and what
|
|
593
|
+
* lands in an index warning, so the distinction has to survive as far as a person.
|
|
594
|
+
*/
|
|
595
|
+
function withEarlier(message, earlier) {
|
|
596
|
+
return earlier?.error ? `${message} — a previous attempt was retried after: ${earlier.error}` : message;
|
|
597
|
+
}
|
|
598
|
+
/** Base delay, doubling per attempt, before jitter. */
|
|
599
|
+
const BACKOFF_BASE_MS = 500;
|
|
600
|
+
const BACKOFF_CAP_MS = 20_000;
|
|
601
|
+
/**
|
|
602
|
+
* How long to wait before trying again.
|
|
603
|
+
*
|
|
604
|
+
* `Retry-After` is obeyed when a server sends one, because a rate limiter knows when its
|
|
605
|
+
* window opens and guessing can only be wrong in both directions. Otherwise exponential with
|
|
606
|
+
* **full** jitter — `random × window` rather than `window ± a bit`. That matters here
|
|
607
|
+
* specifically: `derive.concurrency` runs several workers against one endpoint, and workers
|
|
608
|
+
* that back off in lockstep rediscover the same 429 together. Spreading them across the whole
|
|
609
|
+
* window is what actually clears the queue.
|
|
610
|
+
*
|
|
611
|
+
* `jitter` is a parameter rather than a call to `Math.random` so the behaviour is testable
|
|
612
|
+
* without a clock.
|
|
613
|
+
*/
|
|
614
|
+
export function backoffMs(attempt, retryAfter, jitter) {
|
|
615
|
+
const stated = parseRetryAfter(retryAfter);
|
|
616
|
+
if (stated !== null)
|
|
617
|
+
return Math.min(stated, BACKOFF_CAP_MS);
|
|
618
|
+
const window = Math.min(BACKOFF_BASE_MS * 2 ** (attempt - 1), BACKOFF_CAP_MS);
|
|
619
|
+
return Math.round(window * jitter);
|
|
620
|
+
}
|
|
621
|
+
/** `Retry-After` is either delta-seconds or an HTTP date. Both are legal and both appear. */
|
|
622
|
+
export function parseRetryAfter(value) {
|
|
623
|
+
if (!value)
|
|
624
|
+
return null;
|
|
625
|
+
const trimmed = value.trim();
|
|
626
|
+
if (/^\d+$/.test(trimmed))
|
|
627
|
+
return Number(trimmed) * 1000;
|
|
628
|
+
const at = Date.parse(trimmed);
|
|
629
|
+
if (Number.isNaN(at))
|
|
630
|
+
return null;
|
|
631
|
+
// A date already in the past means "now", not a negative sleep.
|
|
632
|
+
return Math.max(0, at - Date.now());
|
|
633
|
+
}
|
|
634
|
+
/**
|
|
635
|
+
* A zod schema as the JSON Schema an endpoint wants.
|
|
636
|
+
*
|
|
637
|
+
* `$schema` is stripped because it describes the *dialect* rather than the value, and a
|
|
638
|
+
* strict endpoint rejects the key outright. draft-7 is asked for because that is what
|
|
639
|
+
* llama.cpp's schema-to-GBNF converter reads.
|
|
640
|
+
*/
|
|
641
|
+
export function toEndpointSchema(schema, options = {}) {
|
|
642
|
+
const { $schema: _dialect, ...rest } = z.toJSONSchema(schema, {
|
|
643
|
+
target: 'draft-7',
|
|
644
|
+
...(options.reuseDefinitions ? { reused: 'ref' } : {}),
|
|
645
|
+
});
|
|
646
|
+
return rest;
|
|
647
|
+
}
|
|
648
|
+
/** The `response_format` for a rung of the ladder. */
|
|
649
|
+
function responseFormat(jsonSchema, mode) {
|
|
650
|
+
if (!jsonSchema || mode === 'plain')
|
|
651
|
+
return { type: 'json_object' };
|
|
652
|
+
if (mode === 'schema')
|
|
653
|
+
return { type: 'json_object', schema: jsonSchema };
|
|
654
|
+
return {
|
|
655
|
+
type: 'json_schema',
|
|
656
|
+
// The name is required and is not addressable from anywhere else, so it is a constant
|
|
657
|
+
// rather than a per-call string nobody would ever look up.
|
|
658
|
+
json_schema: { name: 'akno_response', strict: true, schema: jsonSchema },
|
|
659
|
+
};
|
|
660
|
+
}
|
|
661
|
+
/**
|
|
662
|
+
* Whether a JSON Schema satisfies OpenAI's strict mode: every property of every object listed
|
|
663
|
+
* in that object's `required`, and `additionalProperties: false` throughout.
|
|
664
|
+
*
|
|
665
|
+
* This exists as a *test*, not as a runtime check. Strict mode rejects an optional property
|
|
666
|
+
* outright, so a schema written with `.optional()` would take the `json_schema` rung out of
|
|
667
|
+
* service for that one call site — quietly, because the fallback to plain `json_object` is
|
|
668
|
+
* indistinguishable from success. `.nullable()` is the shape that expresses "may be absent"
|
|
669
|
+
* and stays strict-safe, and every schema here is written that way on purpose.
|
|
670
|
+
*/
|
|
671
|
+
export function strictModeViolations(node, path = '$') {
|
|
672
|
+
if (node === null || typeof node !== 'object')
|
|
673
|
+
return [];
|
|
674
|
+
const out = [];
|
|
675
|
+
const record = node;
|
|
676
|
+
if (record.type === 'object' && record.properties && typeof record.properties === 'object') {
|
|
677
|
+
const properties = Object.keys(record.properties);
|
|
678
|
+
const required = new Set(Array.isArray(record.required) ? record.required : []);
|
|
679
|
+
const missing = properties.filter((property) => !required.has(property));
|
|
680
|
+
if (missing.length > 0)
|
|
681
|
+
out.push(`${path}: optional ${missing.join(', ')}`);
|
|
682
|
+
if (record.additionalProperties !== false)
|
|
683
|
+
out.push(`${path}: additionalProperties not false`);
|
|
684
|
+
}
|
|
685
|
+
for (const [key, value] of Object.entries(record)) {
|
|
686
|
+
if (Array.isArray(value)) {
|
|
687
|
+
value.forEach((entry, index) => out.push(...strictModeViolations(entry, `${path}.${key}[${index}]`)));
|
|
688
|
+
}
|
|
689
|
+
else if (value && typeof value === 'object') {
|
|
690
|
+
out.push(...strictModeViolations(value, `${path}.${key}`));
|
|
691
|
+
}
|
|
692
|
+
}
|
|
693
|
+
return out;
|
|
694
|
+
}
|
|
695
|
+
function tokenCeiling(perCall, perRole) {
|
|
696
|
+
const limits = [perCall, perRole].filter((limit) => typeof limit === 'number' && Number.isFinite(limit) && limit > 0);
|
|
697
|
+
return limits.length > 0 ? Math.min(...limits) : 1024;
|
|
698
|
+
}
|
|
699
|
+
function degradedReasonFor(role, failure) {
|
|
700
|
+
const unconfigured = failure === 'unavailable';
|
|
701
|
+
switch (role) {
|
|
702
|
+
case 'embedding':
|
|
703
|
+
return unconfigured ? 'no_embedding_model' : 'embedding_failed';
|
|
704
|
+
case 'reranker':
|
|
705
|
+
return unconfigured ? 'no_reranker' : 'rerank_failed';
|
|
706
|
+
case 'expansion':
|
|
707
|
+
return unconfigured ? 'no_expansion_model' : 'expansion_failed';
|
|
708
|
+
case 'answer':
|
|
709
|
+
return unconfigured ? 'no_answer_model' : 'answer_failed';
|
|
710
|
+
default:
|
|
711
|
+
return unconfigured ? 'no_derive_model' : 'derive_failed';
|
|
712
|
+
}
|
|
713
|
+
}
|
|
714
|
+
/**
|
|
715
|
+
* Rewrites the last user message into the multipart form a vision endpoint wants.
|
|
716
|
+
* Images go after the text, because a model reads the instruction first and every
|
|
717
|
+
* prompt here tells it what to do with what follows.
|
|
718
|
+
*/
|
|
719
|
+
function withImages(messages, images) {
|
|
720
|
+
const out = messages.map((message) => ({ ...message }));
|
|
721
|
+
for (let i = out.length - 1; i >= 0; i--) {
|
|
722
|
+
const message = out[i];
|
|
723
|
+
if (message.role !== 'user')
|
|
724
|
+
continue;
|
|
725
|
+
out[i] = {
|
|
726
|
+
role: 'user',
|
|
727
|
+
content: [
|
|
728
|
+
{ type: 'text', text: message.content },
|
|
729
|
+
...images.map((image) => ({
|
|
730
|
+
type: 'image_url',
|
|
731
|
+
image_url: { url: `data:${image.mime};base64,${image.data.toString('base64')}` },
|
|
732
|
+
})),
|
|
733
|
+
],
|
|
734
|
+
};
|
|
735
|
+
return out;
|
|
736
|
+
}
|
|
737
|
+
// No user message to attach to: send the images as one.
|
|
738
|
+
out.push({
|
|
739
|
+
role: 'user',
|
|
740
|
+
content: images.map((image) => ({
|
|
741
|
+
type: 'image_url',
|
|
742
|
+
image_url: { url: `data:${image.mime};base64,${image.data.toString('base64')}` },
|
|
743
|
+
})),
|
|
744
|
+
});
|
|
745
|
+
return out;
|
|
746
|
+
}
|
|
747
|
+
/** Responses names input content parts differently from Chat Completions. */
|
|
748
|
+
function withResponseImages(messages, images) {
|
|
749
|
+
const out = messages.map((message) => ({ ...message }));
|
|
750
|
+
for (let i = out.length - 1; i >= 0; i--) {
|
|
751
|
+
const message = out[i];
|
|
752
|
+
if (message.role !== 'user')
|
|
753
|
+
continue;
|
|
754
|
+
out[i] = {
|
|
755
|
+
role: 'user',
|
|
756
|
+
content: [
|
|
757
|
+
{ type: 'input_text', text: message.content },
|
|
758
|
+
...images.map((image) => ({
|
|
759
|
+
type: 'input_image',
|
|
760
|
+
image_url: `data:${image.mime};base64,${image.data.toString('base64')}`,
|
|
761
|
+
})),
|
|
762
|
+
],
|
|
763
|
+
};
|
|
764
|
+
return out;
|
|
765
|
+
}
|
|
766
|
+
out.push({
|
|
767
|
+
role: 'user',
|
|
768
|
+
content: images.map((image) => ({
|
|
769
|
+
type: 'input_image',
|
|
770
|
+
image_url: `data:${image.mime};base64,${image.data.toString('base64')}`,
|
|
771
|
+
})),
|
|
772
|
+
});
|
|
773
|
+
return out;
|
|
774
|
+
}
|
|
775
|
+
/** Responses puts its JSON contract directly under `text.format`. */
|
|
776
|
+
function responsesTextFormat(jsonSchema) {
|
|
777
|
+
if (!jsonSchema)
|
|
778
|
+
return { type: 'json_object' };
|
|
779
|
+
return { type: 'json_schema', name: 'akno_response', strict: true, schema: jsonSchema };
|
|
780
|
+
}
|
|
781
|
+
function decodeBase64Floats(input) {
|
|
782
|
+
const buffer = Buffer.from(input, 'base64');
|
|
783
|
+
const copy = Buffer.from(buffer);
|
|
784
|
+
return new Float32Array(copy.buffer, copy.byteOffset, copy.byteLength / 4);
|
|
785
|
+
}
|
|
786
|
+
/**
|
|
787
|
+
* A 3B instruct model will wrap JSON in prose, in a fence, or both. Extracting
|
|
788
|
+
* the object rather than failing the parse is the difference between summaries
|
|
789
|
+
* working and summaries being a coin flip.
|
|
790
|
+
*/
|
|
791
|
+
export function parseJsonLoose(raw) {
|
|
792
|
+
const trimmed = raw.trim();
|
|
793
|
+
const candidates = [trimmed];
|
|
794
|
+
const fenced = /```(?:json)?\s*([\s\S]*?)```/.exec(trimmed);
|
|
795
|
+
if (fenced?.[1])
|
|
796
|
+
candidates.push(fenced[1].trim());
|
|
797
|
+
const firstBrace = trimmed.indexOf('{');
|
|
798
|
+
const lastBrace = trimmed.lastIndexOf('}');
|
|
799
|
+
if (firstBrace !== -1 && lastBrace > firstBrace)
|
|
800
|
+
candidates.push(trimmed.slice(firstBrace, lastBrace + 1));
|
|
801
|
+
const firstBracket = trimmed.indexOf('[');
|
|
802
|
+
const lastBracket = trimmed.lastIndexOf(']');
|
|
803
|
+
if (firstBracket !== -1 && lastBracket > firstBracket) {
|
|
804
|
+
candidates.push(trimmed.slice(firstBracket, lastBracket + 1));
|
|
805
|
+
}
|
|
806
|
+
// Direct parse *then* repair, per candidate, most-complete candidate first.
|
|
807
|
+
// The order matters: repairing the whole body has to beat parsing a narrower
|
|
808
|
+
// slice of it, or an inner array lifted out of a truncated object wins and the
|
|
809
|
+
// caller silently receives `["lease","rent"]` where it expected the object.
|
|
810
|
+
for (const candidate of candidates) {
|
|
811
|
+
try {
|
|
812
|
+
return JSON.parse(candidate);
|
|
813
|
+
}
|
|
814
|
+
catch {
|
|
815
|
+
// The body was cut off mid-value because the model hit its token ceiling.
|
|
816
|
+
// Everything before the cut is still valid, so closing what is open
|
|
817
|
+
// recovers it — the difference between a long page getting most of its
|
|
818
|
+
// facts and getting none.
|
|
819
|
+
const repaired = closeTruncatedJson(candidate);
|
|
820
|
+
if (repaired === null)
|
|
821
|
+
continue;
|
|
822
|
+
try {
|
|
823
|
+
return JSON.parse(repaired);
|
|
824
|
+
}
|
|
825
|
+
catch {
|
|
826
|
+
continue;
|
|
827
|
+
}
|
|
828
|
+
}
|
|
829
|
+
}
|
|
830
|
+
return null;
|
|
831
|
+
}
|
|
832
|
+
/**
|
|
833
|
+
* Recovers the longest parseable prefix of a truncated JSON body.
|
|
834
|
+
*
|
|
835
|
+
* Rather than guessing where a safe cut is, this records every position where a
|
|
836
|
+
* value just completed — along with the bracket stack owed at that point — then
|
|
837
|
+
* tries them newest-first and returns the first that actually parses. Validating
|
|
838
|
+
* instead of guessing is what makes it correct around the two cases that break a
|
|
839
|
+
* hand-rolled scanner: a bracket inside a string, and a cut immediately after an
|
|
840
|
+
* object *key* (`{"a"` closes to `{"a"}`, which does not parse, so it falls back).
|
|
841
|
+
*
|
|
842
|
+
* Returns null when the input is not truncated JSON — a genuinely malformed body
|
|
843
|
+
* must stay reported as malformed rather than becoming half an object.
|
|
844
|
+
*/
|
|
845
|
+
export function closeTruncatedJson(input) {
|
|
846
|
+
const start = input.search(/[{[]/);
|
|
847
|
+
if (start === -1)
|
|
848
|
+
return null;
|
|
849
|
+
const stack = [];
|
|
850
|
+
const candidates = [];
|
|
851
|
+
let inString = false;
|
|
852
|
+
let escaped = false;
|
|
853
|
+
const mark = (offset) => {
|
|
854
|
+
if (stack.length > 0) {
|
|
855
|
+
candidates.push({ offset, closers: [...stack].reverse().join('') });
|
|
856
|
+
}
|
|
857
|
+
};
|
|
858
|
+
for (let i = start; i < input.length; i++) {
|
|
859
|
+
const char = input[i];
|
|
860
|
+
if (inString) {
|
|
861
|
+
if (escaped)
|
|
862
|
+
escaped = false;
|
|
863
|
+
else if (char === '\\')
|
|
864
|
+
escaped = true;
|
|
865
|
+
else if (char === '"') {
|
|
866
|
+
inString = false;
|
|
867
|
+
// A string just closed. It may be a key, in which case this candidate
|
|
868
|
+
// will fail to parse and a later attempt will use an earlier one.
|
|
869
|
+
mark(i + 1);
|
|
870
|
+
}
|
|
871
|
+
continue;
|
|
872
|
+
}
|
|
873
|
+
if (char === '"')
|
|
874
|
+
inString = true;
|
|
875
|
+
else if (char === '{' || char === '[')
|
|
876
|
+
stack.push(char === '{' ? '}' : ']');
|
|
877
|
+
else if (char === '}' || char === ']') {
|
|
878
|
+
if (stack.pop() !== char)
|
|
879
|
+
return null; // Mismatched: not a truncation.
|
|
880
|
+
mark(i + 1);
|
|
881
|
+
}
|
|
882
|
+
else if (/[\d}\]eln]/.test(char) && !/[\d.eE+-]/.test(input[i + 1] ?? '')) {
|
|
883
|
+
// End of a number or of `true`/`false`/`null`.
|
|
884
|
+
mark(i + 1);
|
|
885
|
+
}
|
|
886
|
+
}
|
|
887
|
+
// Nothing left open means the body was complete, and this function has no
|
|
888
|
+
// business rewriting it.
|
|
889
|
+
if (stack.length === 0)
|
|
890
|
+
return null;
|
|
891
|
+
for (let i = candidates.length - 1; i >= 0; i--) {
|
|
892
|
+
const candidate = candidates[i];
|
|
893
|
+
const repaired = input.slice(start, candidate.offset).replace(/,\s*$/, '') + candidate.closers;
|
|
894
|
+
try {
|
|
895
|
+
JSON.parse(repaired);
|
|
896
|
+
return repaired;
|
|
897
|
+
}
|
|
898
|
+
catch {
|
|
899
|
+
continue;
|
|
900
|
+
}
|
|
901
|
+
}
|
|
902
|
+
return null;
|
|
903
|
+
}
|
|
904
|
+
//# sourceMappingURL=client.js.map
|