@konneal/engine 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (285) hide show
  1. package/LICENSE +29 -0
  2. package/README.md +13 -0
  3. package/dist/admin.d.ts +26 -0
  4. package/dist/ai.d.ts +6 -0
  5. package/dist/anchors.d.ts +6 -0
  6. package/dist/answercache.d.ts +22 -0
  7. package/dist/ask.d.ts +5 -0
  8. package/dist/auth.d.ts +12 -0
  9. package/dist/bubble.d.ts +14 -0
  10. package/dist/chunk-LLWPT2XV.js +49 -0
  11. package/dist/chunk-MB74PTRM.js +114 -0
  12. package/dist/chunk-WOGQM7DJ.js +197 -0
  13. package/dist/chunk-WWNCWKKC.js +42 -0
  14. package/dist/completion.d.ts +5 -0
  15. package/dist/config.d.ts +154 -0
  16. package/dist/config.js +37 -0
  17. package/dist/context.d.ts +115 -0
  18. package/dist/conversations.d.ts +5 -0
  19. package/dist/drafts.d.ts +129 -0
  20. package/dist/env.d.ts +57 -0
  21. package/dist/faithfulness.d.ts +5 -0
  22. package/dist/grader.d.ts +3 -0
  23. package/dist/graph.d.ts +13 -0
  24. package/dist/hybrid.d.ts +7 -0
  25. package/dist/index.d.ts +9 -0
  26. package/dist/index.js +5373 -0
  27. package/dist/internal_gateway.d.ts +14 -0
  28. package/dist/lexical.d.ts +7 -0
  29. package/dist/livedata.d.ts +77 -0
  30. package/dist/memories.d.ts +10 -0
  31. package/dist/modelplane.d.ts +61 -0
  32. package/dist/oidc.d.ts +73 -0
  33. package/dist/pipeline.d.ts +57 -0
  34. package/dist/profile.d.ts +2 -0
  35. package/dist/profile.gen.d.ts +70 -0
  36. package/dist/profile.js +8 -0
  37. package/dist/projects.d.ts +8 -0
  38. package/dist/prompts/conversational.md +8 -0
  39. package/dist/prompts/enrichment.md +3 -0
  40. package/dist/prompts/faithfulness.md +1 -0
  41. package/dist/prompts/grader.md +5 -0
  42. package/dist/prompts/listwise.md +3 -0
  43. package/dist/prompts/precision.md +1 -0
  44. package/dist/prompts/reflect.md +1 -0
  45. package/dist/prompts/relevancy.md +1 -0
  46. package/dist/prompts/research.md +10 -0
  47. package/dist/prompts/section-summary.md +5 -0
  48. package/dist/prompts/summarize.md +1 -0
  49. package/dist/prompts/system.md +18 -0
  50. package/dist/prompts/understanding.md +17 -0
  51. package/dist/quota.d.ts +13 -0
  52. package/dist/reflect.d.ts +5 -0
  53. package/dist/refs.d.ts +40 -0
  54. package/dist/refusal.d.ts +9 -0
  55. package/dist/refusal.js +9 -0
  56. package/dist/requestScope.d.ts +26 -0
  57. package/dist/requestScope.js +10 -0
  58. package/dist/research.d.ts +8 -0
  59. package/dist/search.d.ts +4 -0
  60. package/dist/selfquery.d.ts +7 -0
  61. package/dist/session.d.ts +1 -0
  62. package/dist/share.d.ts +2 -0
  63. package/dist/structural.d.ts +27 -0
  64. package/dist/tablecontext.d.ts +11 -0
  65. package/dist/understand.d.ts +11 -0
  66. package/dist/understandContract.d.ts +29 -0
  67. package/dist/verdict.d.ts +24 -0
  68. package/docs/API.md +451 -0
  69. package/docs/ARCHITECTURE.md +302 -0
  70. package/docs/AUDIT-2026-08-24.md +71 -0
  71. package/docs/CONTRIBUTOR-AUDIT-2026-08-25.md +147 -0
  72. package/docs/INGEST-ARCHITECTURE.md +158 -0
  73. package/docs/MCP.md +92 -0
  74. package/docs/METANORMA-AI-SERIALIZATION.md +247 -0
  75. package/docs/MKO-EXPORT-PIPELINE.md +147 -0
  76. package/docs/REDESIGN-NORMATIVE-RAG-ETSI.md +485 -0
  77. package/docs/RESEARCH-SOTA-2026.md +243 -0
  78. package/docs/ROADMAP-SOTA.md +130 -0
  79. package/docs/SOTA-STAGE-SPECS.md +509 -0
  80. package/docs/annealment/F1-verdict.md +27 -0
  81. package/docs/annealment/F10-notes.md +19 -0
  82. package/docs/annealment/F11-composition.md +17 -0
  83. package/docs/annealment/F12-passport.md +17 -0
  84. package/docs/annealment/F2-counterfactual.md +20 -0
  85. package/docs/annealment/F3-absence.md +21 -0
  86. package/docs/annealment/F4-instance.md +18 -0
  87. package/docs/annealment/F5-workflow.md +21 -0
  88. package/docs/annealment/F6-impact.md +21 -0
  89. package/docs/annealment/F7-editions.md +18 -0
  90. package/docs/annealment/F8-selfverify.md +19 -0
  91. package/docs/annealment/F9-projection-qa.md +17 -0
  92. package/docs/annealment/L0-locate.md +19 -0
  93. package/docs/annealment/L1-extract.md +18 -0
  94. package/docs/annealment/L2-nomenclature.md +22 -0
  95. package/docs/annealment/L3-geometry.md +23 -0
  96. package/docs/annealment/L4-composition.md +21 -0
  97. package/docs/annealment/L5-cross-standard.md +20 -0
  98. package/docs/annealment/L6-diachrony.md +21 -0
  99. package/docs/annealment/L7-perception.md +20 -0
  100. package/docs/annealment/L8-computation.md +22 -0
  101. package/docs/annealment/L9-instance-process.md +23 -0
  102. package/docs/annealment/README.md +10 -0
  103. package/docs/guidelines-metanorma-ai-programme.md +279 -0
  104. package/docs/identity-onboarding-rag.md +65 -0
  105. package/docs/identity-service.md +219 -0
  106. package/docs/knowledge-annealment.md +273 -0
  107. package/docs/konneal-extraction-plan.md +481 -0
  108. package/docs/metanorma-for-ai.md +270 -0
  109. package/docs/mirror-plan.md +36 -0
  110. package/docs/multi-sdo-architecture.md +191 -0
  111. package/docs/paper-annealment-comparison.md +259 -0
  112. package/docs/paper-assets/architecture.svg +94 -0
  113. package/docs/paper-assets/contract-v2.svg +94 -0
  114. package/docs/paper-assets/mko-ingest.svg +91 -0
  115. package/docs/paper-oiml-bulletin.md +402 -0
  116. package/docs/paper-oiml-bulletin.mdx +419 -0
  117. package/docs/product-branding-options.md +172 -0
  118. package/docs/projects-design.md +88 -0
  119. package/docs/sota-mechanisms.md +184 -0
  120. package/docs/spec-api.md +77 -0
  121. package/docs/spec-pipeline.md +126 -0
  122. package/docs/vector-adapter.md +88 -0
  123. package/package.json +70 -0
  124. package/profile/corpora.yaml +5 -0
  125. package/profile/datasets.yaml +14 -0
  126. package/profile/prompts.yaml +5 -0
  127. package/profile/publisher.yaml +17 -0
  128. package/profile/retrieval.yaml +1 -0
  129. package/profile/sources.yaml +5 -0
  130. package/profile/ui.yaml +7 -0
  131. package/scripts/gen_profile.mjs +33 -0
  132. package/workers/shared/ai.ts +21 -0
  133. package/workers/shared/auth.ts +16 -0
  134. package/workers/shared/chunk.ts +108 -0
  135. package/workers/shared/oidc.ts +312 -0
  136. package/workers/shared/router.ts +45 -0
  137. package/workers/shared/session.ts +104 -0
  138. package/workers/worker_internal/src/index.ts +157 -0
  139. package/workers/worker_internal/tsconfig.json +15 -0
  140. package/workers/worker_internal/wrangler.toml +32 -0
  141. package/workers/worker_mcp/src/index.ts +175 -0
  142. package/workers/worker_mcp/tsconfig.json +13 -0
  143. package/workers/worker_mcp/wrangler.toml +18 -0
  144. package/workers/worker_public/migrations/0002_conversations.sql +22 -0
  145. package/workers/worker_public/migrations/0003_shared_conversations.sql +9 -0
  146. package/workers/worker_public/migrations/0004_graph.sql +16 -0
  147. package/workers/worker_public/migrations/0005_documents.sql +19 -0
  148. package/workers/worker_public/migrations/0006_conversation_entities.sql +11 -0
  149. package/workers/worker_public/migrations/0007_chunks_fts.sql +43 -0
  150. package/workers/worker_public/migrations/0008_unit_payloads.sql +15 -0
  151. package/workers/worker_public/migrations/0009_chunks_unit.sql +8 -0
  152. package/workers/worker_public/migrations/0009_message_context.sql +7 -0
  153. package/workers/worker_public/migrations/0010_model_nodes.sql +33 -0
  154. package/workers/worker_public/migrations/0011_schema_union.sql +38 -0
  155. package/workers/worker_public/migrations/0012_memories.sql +15 -0
  156. package/workers/worker_public/migrations/0013_projects.sql +21 -0
  157. package/workers/worker_public/node_modules/@cloudflare/workers-types/2021-11-03/index.d.ts +16306 -0
  158. package/workers/worker_public/node_modules/@cloudflare/workers-types/2021-11-03/index.ts +16261 -0
  159. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-01-31/index.d.ts +16373 -0
  160. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-01-31/index.ts +16328 -0
  161. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-03-21/index.d.ts +16382 -0
  162. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-03-21/index.ts +16337 -0
  163. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-08-04/index.d.ts +16383 -0
  164. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-08-04/index.ts +16338 -0
  165. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-10-31/index.d.ts +16403 -0
  166. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-10-31/index.ts +16358 -0
  167. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-11-30/index.d.ts +16408 -0
  168. package/workers/worker_public/node_modules/@cloudflare/workers-types/2022-11-30/index.ts +16363 -0
  169. package/workers/worker_public/node_modules/@cloudflare/workers-types/2023-03-01/index.d.ts +16414 -0
  170. package/workers/worker_public/node_modules/@cloudflare/workers-types/2023-03-01/index.ts +16369 -0
  171. package/workers/worker_public/node_modules/@cloudflare/workers-types/2023-07-01/index.d.ts +16414 -0
  172. package/workers/worker_public/node_modules/@cloudflare/workers-types/2023-07-01/index.ts +16369 -0
  173. package/workers/worker_public/node_modules/@cloudflare/workers-types/README.md +135 -0
  174. package/workers/worker_public/node_modules/@cloudflare/workers-types/entrypoints.svg +53 -0
  175. package/workers/worker_public/node_modules/@cloudflare/workers-types/experimental/index.d.ts +17095 -0
  176. package/workers/worker_public/node_modules/@cloudflare/workers-types/experimental/index.ts +17050 -0
  177. package/workers/worker_public/node_modules/@cloudflare/workers-types/index.d.ts +16306 -0
  178. package/workers/worker_public/node_modules/@cloudflare/workers-types/index.ts +16261 -0
  179. package/workers/worker_public/node_modules/@cloudflare/workers-types/latest/index.d.ts +16447 -0
  180. package/workers/worker_public/node_modules/@cloudflare/workers-types/latest/index.ts +16402 -0
  181. package/workers/worker_public/node_modules/@cloudflare/workers-types/oldest/index.d.ts +16306 -0
  182. package/workers/worker_public/node_modules/@cloudflare/workers-types/oldest/index.ts +16261 -0
  183. package/workers/worker_public/node_modules/@cloudflare/workers-types/package.json +11 -0
  184. package/workers/worker_public/package.json +13 -0
  185. package/workers/worker_public/prompts/conversational.md +8 -0
  186. package/workers/worker_public/prompts/enrichment.md +3 -0
  187. package/workers/worker_public/prompts/faithfulness.md +1 -0
  188. package/workers/worker_public/prompts/grader.md +5 -0
  189. package/workers/worker_public/prompts/listwise.md +3 -0
  190. package/workers/worker_public/prompts/precision.md +1 -0
  191. package/workers/worker_public/prompts/reflect.md +1 -0
  192. package/workers/worker_public/prompts/relevancy.md +1 -0
  193. package/workers/worker_public/prompts/research.md +10 -0
  194. package/workers/worker_public/prompts/section-summary.md +5 -0
  195. package/workers/worker_public/prompts/summarize.md +1 -0
  196. package/workers/worker_public/prompts/system.md +18 -0
  197. package/workers/worker_public/prompts/understanding.md +17 -0
  198. package/workers/worker_public/public/app.js +166 -0
  199. package/workers/worker_public/public/index.html +48 -0
  200. package/workers/worker_public/public/style.css +147 -0
  201. package/workers/worker_public/schema.sql +248 -0
  202. package/workers/worker_public/src/admin.ts +358 -0
  203. package/workers/worker_public/src/ai.ts +71 -0
  204. package/workers/worker_public/src/anchors.ts +41 -0
  205. package/workers/worker_public/src/answercache.ts +72 -0
  206. package/workers/worker_public/src/ask.ts +1094 -0
  207. package/workers/worker_public/src/auth.ts +252 -0
  208. package/workers/worker_public/src/bubble.ts +111 -0
  209. package/workers/worker_public/src/completion.ts +75 -0
  210. package/workers/worker_public/src/config.ts +238 -0
  211. package/workers/worker_public/src/context.ts +238 -0
  212. package/workers/worker_public/src/conversations.ts +162 -0
  213. package/workers/worker_public/src/drafts.ts +497 -0
  214. package/workers/worker_public/src/env.ts +90 -0
  215. package/workers/worker_public/src/faithfulness.ts +63 -0
  216. package/workers/worker_public/src/grader.ts +89 -0
  217. package/workers/worker_public/src/graph.ts +63 -0
  218. package/workers/worker_public/src/hybrid.ts +77 -0
  219. package/workers/worker_public/src/index.ts +441 -0
  220. package/workers/worker_public/src/internal_gateway.ts +41 -0
  221. package/workers/worker_public/src/lexical.ts +86 -0
  222. package/workers/worker_public/src/lib/hit.ts +4 -0
  223. package/workers/worker_public/src/lib/http.ts +83 -0
  224. package/workers/worker_public/src/lib/router.ts +4 -0
  225. package/workers/worker_public/src/livedata.ts +334 -0
  226. package/workers/worker_public/src/memories.ts +81 -0
  227. package/workers/worker_public/src/modelplane.ts +213 -0
  228. package/workers/worker_public/src/oidc.ts +333 -0
  229. package/workers/worker_public/src/pipeline.ts +377 -0
  230. package/workers/worker_public/src/ports/blobs.ts +7 -0
  231. package/workers/worker_public/src/ports/cloudflare/adapters.ts +177 -0
  232. package/workers/worker_public/src/ports/kv.ts +8 -0
  233. package/workers/worker_public/src/ports/model.ts +28 -0
  234. package/workers/worker_public/src/ports/runtime.ts +13 -0
  235. package/workers/worker_public/src/ports/store.ts +20 -0
  236. package/workers/worker_public/src/ports/vector.ts +26 -0
  237. package/workers/worker_public/src/profile.gen.ts +101 -0
  238. package/workers/worker_public/src/profile.ts +16 -0
  239. package/workers/worker_public/src/projects.ts +108 -0
  240. package/workers/worker_public/src/prompts.d.ts +6 -0
  241. package/workers/worker_public/src/quota.ts +54 -0
  242. package/workers/worker_public/src/reflect.ts +67 -0
  243. package/workers/worker_public/src/refs.ts +107 -0
  244. package/workers/worker_public/src/refusal.ts +65 -0
  245. package/workers/worker_public/src/requestScope.ts +71 -0
  246. package/workers/worker_public/src/research.ts +126 -0
  247. package/workers/worker_public/src/search.ts +56 -0
  248. package/workers/worker_public/src/selfquery.ts +25 -0
  249. package/workers/worker_public/src/session.ts +4 -0
  250. package/workers/worker_public/src/share.ts +53 -0
  251. package/workers/worker_public/src/stages/conceptGraph.ts +68 -0
  252. package/workers/worker_public/src/stages/conceptSteer.ts +39 -0
  253. package/workers/worker_public/src/stages/corpusScope.ts +25 -0
  254. package/workers/worker_public/src/stages/dedup.ts +10 -0
  255. package/workers/worker_public/src/stages/dense.ts +73 -0
  256. package/workers/worker_public/src/stages/diversity.ts +33 -0
  257. package/workers/worker_public/src/stages/editionCover.ts +63 -0
  258. package/workers/worker_public/src/stages/editionSteer.ts +88 -0
  259. package/workers/worker_public/src/stages/familyBoost.ts +22 -0
  260. package/workers/worker_public/src/stages/federate.ts +22 -0
  261. package/workers/worker_public/src/stages/glossary.ts +65 -0
  262. package/workers/worker_public/src/stages/graphLane.ts +31 -0
  263. package/workers/worker_public/src/stages/hyde.ts +29 -0
  264. package/workers/worker_public/src/stages/index.ts +69 -0
  265. package/workers/worker_public/src/stages/lexicalUnion.ts +21 -0
  266. package/workers/worker_public/src/stages/multiQuery.ts +57 -0
  267. package/workers/worker_public/src/stages/overviewDemote.ts +14 -0
  268. package/workers/worker_public/src/stages/poolOpen.ts +10 -0
  269. package/workers/worker_public/src/stages/propagate.ts +15 -0
  270. package/workers/worker_public/src/stages/rerank.ts +47 -0
  271. package/workers/worker_public/src/stages/seal.ts +16 -0
  272. package/workers/worker_public/src/stages/sectionDescent.ts +61 -0
  273. package/workers/worker_public/src/stages/stdRefNudge.ts +35 -0
  274. package/workers/worker_public/src/stages/subQuery.ts +42 -0
  275. package/workers/worker_public/src/stages/termNudge.ts +24 -0
  276. package/workers/worker_public/src/stages/typedPin.ts +131 -0
  277. package/workers/worker_public/src/stages/types.ts +112 -0
  278. package/workers/worker_public/src/stages/windowFloor.ts +23 -0
  279. package/workers/worker_public/src/structural.ts +171 -0
  280. package/workers/worker_public/src/tablecontext.ts +41 -0
  281. package/workers/worker_public/src/understand.ts +72 -0
  282. package/workers/worker_public/src/understandContract.ts +67 -0
  283. package/workers/worker_public/src/verdict.ts +255 -0
  284. package/workers/worker_public/tsconfig.json +18 -0
  285. package/workers/worker_public/wrangler.toml +104 -0
@@ -0,0 +1,248 @@
1
+ CREATE TABLE IF NOT EXISTS api_keys (
2
+ id TEXT PRIMARY KEY,
3
+ name TEXT NOT NULL,
4
+ key_hash TEXT NOT NULL UNIQUE,
5
+ day_limit INTEGER NOT NULL DEFAULT 2000,
6
+ created_at TEXT NOT NULL,
7
+ revoked INTEGER NOT NULL DEFAULT 0
8
+ );
9
+
10
+ CREATE TABLE IF NOT EXISTS queries (
11
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
12
+ ts TEXT NOT NULL,
13
+ day TEXT NOT NULL,
14
+ tier TEXT NOT NULL,
15
+ route TEXT,
16
+ model TEXT,
17
+ ok INTEGER,
18
+ answer_chars INTEGER,
19
+ query_hash TEXT,
20
+ lang TEXT
21
+ );
22
+
23
+ CREATE TABLE IF NOT EXISTS spend (
24
+ day TEXT NOT NULL,
25
+ tier TEXT NOT NULL,
26
+ model TEXT NOT NULL,
27
+ requests INTEGER NOT NULL DEFAULT 0,
28
+ PRIMARY KEY (day, tier, model)
29
+ );
30
+
31
+ CREATE TABLE IF NOT EXISTS feedback (
32
+ query_hash TEXT NOT NULL,
33
+ rating INTEGER NOT NULL,
34
+ ts TEXT NOT NULL
35
+ );
36
+
37
+ -- ── the migration-tracked tables (single union with migrations/*.sql; a
38
+ -- drift test pins the two table sets equal) ──
39
+
40
+ -- Conversations for signed-in members (TODO.impl/15). Anonymous history
41
+ -- is device-local by design; the server stores member conversations only.
42
+ CREATE TABLE IF NOT EXISTS memories (
43
+ id TEXT PRIMARY KEY, -- m:<hex16>
44
+ sub TEXT NOT NULL, -- owner (session sub)
45
+ name TEXT NOT NULL, -- "Lab context"
46
+ content TEXT NOT NULL, -- the CONTEXT.md body (<= 8k chars)
47
+ enabled INTEGER NOT NULL DEFAULT 1,
48
+ created_at INTEGER NOT NULL,
49
+ updated_at INTEGER NOT NULL
50
+ );
51
+ CREATE INDEX IF NOT EXISTS idx_memories_sub ON memories(sub, updated_at);
52
+ CREATE TABLE IF NOT EXISTS projects (
53
+ id TEXT PRIMARY KEY, -- p:<hex16>
54
+ sub TEXT NOT NULL, -- owner (session sub)
55
+ name TEXT NOT NULL,
56
+ created_at INTEGER NOT NULL
57
+ );
58
+ CREATE TABLE IF NOT EXISTS project_files (
59
+ id TEXT PRIMARY KEY, -- pf:<hex16>
60
+ project_id TEXT NOT NULL,
61
+ name TEXT NOT NULL,
62
+ content TEXT NOT NULL,
63
+ created_at INTEGER NOT NULL,
64
+ updated_at INTEGER NOT NULL
65
+ );
66
+ CREATE INDEX IF NOT EXISTS idx_project_files_project ON project_files(project_id, updated_at);
67
+ CREATE TABLE IF NOT EXISTS conversations (
68
+ id TEXT PRIMARY KEY,
69
+ sub TEXT NOT NULL,
70
+ title TEXT NOT NULL DEFAULT '',
71
+ created_at TEXT NOT NULL,
72
+ updated_at TEXT NOT NULL,
73
+ project_id TEXT -- the project this conversation belongs to (NULL = none)
74
+ );
75
+ CREATE INDEX IF NOT EXISTS idx_conversations_sub ON conversations(sub, updated_at DESC);
76
+
77
+ CREATE TABLE IF NOT EXISTS messages (
78
+ id TEXT PRIMARY KEY,
79
+ conversation_id TEXT NOT NULL,
80
+ role TEXT NOT NULL CHECK (role IN ('user','assistant')),
81
+ content TEXT NOT NULL,
82
+ citations TEXT,
83
+ model TEXT,
84
+ created_at TEXT NOT NULL,
85
+ FOREIGN KEY (conversation_id) REFERENCES conversations(id) ON DELETE CASCADE
86
+ );
87
+ CREATE INDEX IF NOT EXISTS idx_messages_conv ON messages(conversation_id, created_at);
88
+ -- Public read-only shared conversations (TODO.rag/09)
89
+ CREATE TABLE IF NOT EXISTS shared_conversations (
90
+ slug TEXT PRIMARY KEY,
91
+ owner_sub TEXT NOT NULL,
92
+ title TEXT NOT NULL DEFAULT '',
93
+ messages TEXT NOT NULL, -- JSON array of {role, content, citations}
94
+ created_at TEXT NOT NULL
95
+ );
96
+ CREATE INDEX IF NOT EXISTS idx_shared_owner ON shared_conversations(owner_sub);
97
+ -- Graph projection (Stage 6 of the SOTA specs): structural edges from
98
+ -- relaton-data-oiml + concept nodes from the Glossarist vocab datasets.
99
+ -- Populated by ingest/graph.py; read by the retrieval graph lane.
100
+ CREATE TABLE IF NOT EXISTS graph_nodes (
101
+ id TEXT PRIMARY KEY, -- doc:OIML-R-60-1-2017 | family:R-60 | concept:<id>
102
+ kind TEXT NOT NULL, -- doc | family | concept
103
+ label TEXT NOT NULL
104
+ );
105
+ CREATE TABLE IF NOT EXISTS graph_edges (
106
+ src TEXT NOT NULL,
107
+ dst TEXT NOT NULL,
108
+ kind TEXT NOT NULL, -- part_of | variant_of | successor | amends | defines
109
+ PRIMARY KEY (src, dst, kind)
110
+ );
111
+ CREATE INDEX IF NOT EXISTS idx_graph_edges_src ON graph_edges(src);
112
+ CREATE INDEX IF NOT EXISTS idx_graph_edges_dst ON graph_edges(dst);
113
+ -- Document identity registry (Stage 1 of the SOTA specs): the SSOT for
114
+ -- edition status. Status is DERIVED, not copied: relaton's status field
115
+ -- is inconsistent (records claim in-force while carrying successors) —
116
+ -- an edition with a successor is superseded regardless; the ACTIVE
117
+ -- edition of a family+part is the terminal node of its successor chain
118
+ -- (no successor, max edition).
119
+ CREATE TABLE IF NOT EXISTS documents (
120
+ canonical_id TEXT PRIMARY KEY, -- doc:OIML-R-60-2021
121
+ docidentifier TEXT NOT NULL, -- OIML R 60:2021
122
+ family TEXT NOT NULL, -- R-60
123
+ part TEXT, -- 1 | 2 | 3 | A | annexes | sup | NULL
124
+ edition TEXT NOT NULL, -- 2021
125
+ status TEXT NOT NULL, -- in-force | superseded | withdrawn
126
+ derived_status TEXT NOT NULL, -- successor-edge derivation result
127
+ active INTEGER NOT NULL, -- 1 = the active edition of family+part
128
+ superseded_by TEXT, -- canonical_id of the successor
129
+ title TEXT
130
+ );
131
+ CREATE INDEX IF NOT EXISTS idx_documents_family ON documents(family, part, active);
132
+ -- Cross-turn entity memory (G5): resolved entities per conversation so
133
+ -- pronouns/ellipsis in follow-ups resolve O(1) instead of re-deriving
134
+ -- from raw history text every turn.
135
+ CREATE TABLE IF NOT EXISTS conversation_entities (
136
+ conversation_id TEXT NOT NULL,
137
+ entity TEXT NOT NULL, -- display form, e.g. "OIML R 60-1:2021"
138
+ kind TEXT NOT NULL, -- document | term
139
+ ts INTEGER NOT NULL,
140
+ PRIMARY KEY (conversation_id, entity)
141
+ );
142
+ CREATE INDEX IF NOT EXISTS idx_conv_entities ON conversation_entities(conversation_id);
143
+ -- G-ETSI-1: full-corpus lexical index for BM25 prefilter (arXiv:2604.09868 §II-B5).
144
+ -- Dense-only Vectorize cannot recover exact-jargon misses; FTS5 is the
145
+ -- corpus-wide sparse stage that feeds RRF alongside dense top-K.
146
+
147
+ CREATE TABLE IF NOT EXISTS chunks (
148
+ id TEXT PRIMARY KEY,
149
+ doc_id TEXT NOT NULL,
150
+ docidentifier TEXT,
151
+ doctype TEXT,
152
+ doc_number TEXT,
153
+ edition TEXT,
154
+ language TEXT,
155
+ clause_anchor TEXT,
156
+ clause_title TEXT,
157
+ status TEXT,
158
+ superseded_by TEXT,
159
+ corpus TEXT,
160
+ tier TEXT,
161
+ -- original body for Hit assembly when dense didn't return this id
162
+ text TEXT NOT NULL,
163
+ -- context preamble + body (Anthropic contextual BM25); what FTS indexes
164
+ fts_text TEXT NOT NULL
165
+ );
166
+
167
+ CREATE VIRTUAL TABLE IF NOT EXISTS chunks_fts USING fts5(
168
+ fts_text,
169
+ content='chunks',
170
+ content_rowid='rowid',
171
+ tokenize = 'porter unicode61'
172
+ );
173
+
174
+ CREATE TRIGGER IF NOT EXISTS chunks_ai AFTER INSERT ON chunks BEGIN
175
+ INSERT INTO chunks_fts(rowid, fts_text) VALUES (new.rowid, new.fts_text);
176
+ END;
177
+ CREATE TRIGGER IF NOT EXISTS chunks_ad AFTER DELETE ON chunks BEGIN
178
+ INSERT INTO chunks_fts(chunks_fts, rowid, fts_text) VALUES ('delete', old.rowid, old.fts_text);
179
+ END;
180
+ CREATE TRIGGER IF NOT EXISTS chunks_au AFTER UPDATE ON chunks BEGIN
181
+ INSERT INTO chunks_fts(chunks_fts, rowid, fts_text) VALUES ('delete', old.rowid, old.fts_text);
182
+ INSERT INTO chunks_fts(rowid, fts_text) VALUES (new.rowid, new.fts_text);
183
+ END;
184
+
185
+ CREATE INDEX IF NOT EXISTS idx_chunks_doc_number ON chunks(doc_number);
186
+ -- Answer contract v2: typed unit payloads for block rendering.
187
+ -- Served blocks NEVER pass through the LLM — the model emits [[u:<id>]]
188
+ -- references; the worker resolves them here and validates against the
189
+ -- passages actually used (producer payloads, ingest-validated).
190
+
191
+ CREATE TABLE IF NOT EXISTS unit_payloads (
192
+ unit_id TEXT PRIMARY KEY,
193
+ doc_id TEXT NOT NULL,
194
+ docidentifier TEXT,
195
+ edition TEXT,
196
+ clause_anchor TEXT,
197
+ type TEXT NOT NULL, -- table | formula | figure | term | requirement
198
+ payload TEXT NOT NULL -- the MN 116 typed payload, JSON-encoded
199
+ );
200
+ CREATE INDEX IF NOT EXISTS idx_unit_payloads_doc ON unit_payloads(doc_id);
201
+ -- Contract v2 over the lexical lane: typed chunks (tables, figures,
202
+ -- formulas, terms) must carry their unit identity through BM25 too —
203
+ -- without these columns, a typed chunk arriving via FTS cannot be
204
+ -- referenced [[u:…]], pinned for doc-scoped queries, or protected by
205
+ -- the retyping check (Vectorize metadata already carries them; D1 did not).
206
+
207
+ ALTER TABLE chunks ADD COLUMN unit_id TEXT;
208
+ ALTER TABLE chunks ADD COLUMN block TEXT;
209
+ -- The per-answer context mark (TODO.ai-platform/02): the transcript marks
210
+ -- the context each answer was grounded in (the panel's declared chip as
211
+ -- the service APPLIED it — the context_applied echo), so a resumed
212
+ -- conversation keeps its honest context lines. NULL means the answer
213
+ -- carries no recorded context (pre-chips history included) — the panel
214
+ -- renders no context line for those rather than guessing one.
215
+ ALTER TABLE messages ADD COLUMN context_applied TEXT;
216
+ -- The model plane (TODO.ai-platform/05): the SMART Recommendation MODELS
217
+ -- join the retrieval — the packages' machine content (the requirements'
218
+ -- constraints, the applicability rules, the acceptance criteria, the
219
+ -- conformance tests, the term definitions, the subject constraints, the
220
+ -- characteristics, the state machines) indexed alongside the prose corpus.
221
+ -- The content DERIVES from the primmel packages (the SSOT) via the smart
222
+ -- repo's model-plane bundles (browser/public/data/model-plane/*.json,
223
+ -- byte-clean-guarded there); the per-standard source_hash pins WHAT the
224
+ -- index derived from — a package change moves the hash and the freshness
225
+ -- gate (ingest model-plane --check) refuses to call the index current
226
+ -- until a re-index lands.
227
+
228
+ CREATE TABLE IF NOT EXISTS model_nodes (
229
+ standard TEXT NOT NULL, -- oiml-r60
230
+ node_id TEXT NOT NULL, -- /req/metrological/mpe
231
+ kind TEXT NOT NULL, -- requirement | conformance_test | term | constraint | characteristic | state_machine | dimension
232
+ name TEXT,
233
+ clause_doc TEXT, -- urn:oiml:pub:r:60-1:2021 (the provenance's document)
234
+ clause_ref TEXT, -- 5.3.2 (the clause inside it)
235
+ content TEXT NOT NULL, -- the node's JSON (the bundle projection, verbatim)
236
+ content_hash TEXT NOT NULL, -- sha256 of content (drift forensics)
237
+ PRIMARY KEY (standard, node_id)
238
+ );
239
+ CREATE INDEX IF NOT EXISTS idx_model_nodes_kind ON model_nodes(kind);
240
+
241
+ CREATE TABLE IF NOT EXISTS model_plane_meta (
242
+ standard TEXT PRIMARY KEY,
243
+ package TEXT NOT NULL, -- oiml-r60 (the primmel package)
244
+ plane TEXT NOT NULL, -- the bundle shape version (model-plane/1)
245
+ source_hash TEXT NOT NULL, -- the package content hash the index derived from
246
+ node_count INTEGER NOT NULL,
247
+ indexed_at TEXT NOT NULL -- when the index landed (operational truth)
248
+ );
@@ -0,0 +1,358 @@
1
+ // Admin-surface handlers: enrichment, section units, captions, vector
2
+ // ops, judging, API-key management — every ADMIN_TOKEN-gated route's
3
+ // behavior lives here (TODO.impl/23); index.ts only registers them.
4
+ import { MODELS, num, sha256Hex, today } from "./config";
5
+ import { embed } from "./ai";
6
+ import { err, json, corsHeaders, readJson } from "./lib/http";
7
+ import type { Env } from "./env";
8
+ import enrichmentPrompt from "../prompts/enrichment.md";
9
+ import sectionSummaryPrompt from "../prompts/section-summary.md";
10
+ import relevancyPrompt from "../prompts/relevancy.md";
11
+ import precisionPrompt from "../prompts/precision.md";
12
+ import { scoreFaithfulness } from "./faithfulness";
13
+ import { scoreJudge } from "./grader";
14
+ import { portModelRunner } from "./env.ts";
15
+
16
+ /** Contextual enrichment (quality-first lane): for each chunk, write a
17
+ * situating context (KV-cached per chunk id), embed context+text, and
18
+ * upsert in place — the enrichment persists into every future retrieval
19
+ * of that chunk. Driven by ingest/enrich.py in resumable batches. */
20
+ export async function handleEnrich(env: Env, ctx: ExecutionContext, req: Request): Promise<Response> {
21
+ if (!env.ADMIN_TOKEN) return err(501, "admin_disabled", "ADMIN_TOKEN secret is not configured");
22
+ const auth = req.headers.get("authorization") ?? "";
23
+ if (auth !== `Bearer ${env.ADMIN_TOKEN}`) return err(401, "unauthorized", "Invalid admin token");
24
+ const body = await readJson(req);
25
+ const chunks = Array.isArray(body?.chunks) ? body.chunks : [];
26
+ if (chunks.length === 0 || chunks.length > 8) return err(400, "invalid_input", "chunks: 1-8 required");
27
+ const force = body?.force === true;
28
+ // mode:"context" generates/returns the situating preamble WITHOUT
29
+ // embedding or upserting — the comparison-lane builders use it so lane
30
+ // chunks can never land in the production index (the 2026-09-02
31
+ // incident: 1,126 lane vectors entered production through this
32
+ // endpoint's upsert side effect). Default mode stays the production
33
+ // enrichment flow (context + embed + upsert in place).
34
+ const contextOnly = body?.mode === "context";
35
+ // mode:"ab" (the enrichment effort experiment): generate WITHOUT any
36
+ // side effect — no KV cache read or write, no embed, no upsert — with
37
+ // an explicit effort. The A/B compares low vs high effort on identical
38
+ // chunks; polluting the production context cache would decide the
39
+ // experiment before the judge does.
40
+ const abMode = body?.mode === "ab";
41
+ const effort = body?.effort === "high" ? "high" : "low";
42
+ // ab-mode only: an admin-gated system-prompt override, so experiment
43
+ // harnesses can run judged comparisons through the binding lane — the
44
+ // REST ai/run token flakes 401 on this account (three waves running)
45
+ const abPrompt = abMode && typeof body?.prompt === "string" ? body.prompt.slice(0, 4000) : null;
46
+ const model = typeof env.ENRICH_MODEL === "string" && env.ENRICH_MODEL ? env.ENRICH_MODEL : MODELS.enrich;
47
+
48
+ const usage = { prompt_tokens: 0, completion_tokens: 0, requests: 0, cache_hits: 0 };
49
+ const results = await Promise.all(
50
+ chunks.map(async (c: any) => {
51
+ if (!c?.id || typeof c?.text !== "string" || !c?.metadata) return { id: c?.id ?? null, ok: false, error: "invalid chunk" };
52
+ try {
53
+ const cacheKey = `e:${c.id}`;
54
+ let context = force || abMode ? null : await env.CACHE.get(cacheKey);
55
+ const cached = !!context;
56
+ if (!context) {
57
+ const m = c.metadata;
58
+ const head = `${m.docidentifier ?? m.doc_id}${m.clause_anchor ? " §" + m.clause_anchor : ""}${m.clause_title ? " — " + m.clause_title : ""}`;
59
+ const res: any = await env.AI.run(model, {
60
+ messages: [
61
+ { role: "system", content: (abMode && abPrompt) || enrichmentPrompt.trimEnd() },
62
+ { role: "user", content: abMode && abPrompt ? String(body?.user_text ?? "").slice(0, 4000) : `${head}\n\n${c.text.slice(0, 1500)}` },
63
+ ],
64
+ max_tokens: 1600,
65
+ // measured (2026-09-12, TODO.impl/62): high effort beats low
66
+ // 58% vs 25% on blind pairwise judging at equal length — the
67
+ // enrichment lane is one-time and quality-first, so the win
68
+ // compounds into every future retrieval
69
+ reasoning_effort: abMode ? effort : "high",
70
+ });
71
+ const raw = typeof res?.response === "string" && res.response.trim()
72
+ ? res.response
73
+ : res?.choices?.[0]?.message?.content;
74
+ context = typeof raw === "string" ? raw.trim().replace(/^["\']|[\"']$/g, "").slice(0, 400) : "";
75
+ if (!context) {
76
+ console.log("enrich raw keys:", Object.keys(res ?? {}), "sample:", JSON.stringify(res).slice(0, 300));
77
+ return { id: c.id, ok: false, error: "empty enrichment" };
78
+ }
79
+ if (res?.usage) {
80
+ usage.prompt_tokens += Number(res.usage.prompt_tokens ?? 0);
81
+ usage.completion_tokens += Number(res.usage.completion_tokens ?? 0);
82
+ }
83
+ usage.requests += 1;
84
+ if (!abMode) ctx.waitUntil(env.CACHE.put(cacheKey, context, { expirationTtl: 2_592_000 }));
85
+ } else {
86
+ usage.cache_hits += 1;
87
+ }
88
+ if (contextOnly) return { id: c.id, ok: true, cached, context };
89
+ const original = typeof c.metadata.chunk_text === "string" && c.metadata.chunk_text ? c.metadata.chunk_text : c.text;
90
+ const enriched = `${context}\n\n${original}`;
91
+ const vector = await embed(portModelRunner(env), MODELS.embed, enriched.slice(0, 6000));
92
+ await env.VECTORIZE.upsert([{ id: c.id, values: vector, metadata: { ...c.metadata, chunk_text: enriched, ctx: "1" } }]);
93
+ return { id: c.id, ok: true, cached, context };
94
+ } catch (e: any) {
95
+ return { id: c.id, ok: false, error: String(e?.message ?? e).slice(0, 200) };
96
+ }
97
+ }),
98
+ );
99
+ const ok = results.filter((r: any) => r.ok).length;
100
+ console.log("enrich:", ok, "/", results.length, "usage:", JSON.stringify(usage));
101
+ if (usage.requests > 0) {
102
+ // ledger: enrichment spend shows up in /v1/admin/stats like serving
103
+ ctx.waitUntil(
104
+ env.DB.prepare(
105
+ "INSERT INTO spend (day, tier, model, requests) VALUES (?1,'enrich',?2,?3) ON CONFLICT(day, tier, model) DO UPDATE SET requests = requests + ?3",
106
+ )
107
+ .bind(today(), model, usage.requests)
108
+ .run(),
109
+ );
110
+ }
111
+ return json({ results, usage });
112
+ }
113
+
114
+ /** Section-summary units (FABLE/BEAR multi-granularity, arXiv:2601.18116):
115
+ * the corpus's clause chunks start at depth 2 ("3.1"), so the tree has no
116
+ * depth-1 nodes. This endpoint writes them: a summary of each top-level
117
+ * clause generated from its child chunks' excerpts (quality-first lane,
118
+ * KV-cached per unit id), embedded as toc-path ⊕ summary à la FABLE's
119
+ * internal-node indexing, and upserted as a navigation node — serving
120
+ * descends from it to quotable leaf clauses (pipeline.ts section
121
+ * descent). Credential, batching and ledger mirror /admin/enrich. */
122
+ export async function handleSectionUnit(env: Env, ctx: ExecutionContext, req: Request): Promise<Response> {
123
+ if (!env.ADMIN_TOKEN) return err(501, "admin_disabled", "ADMIN_TOKEN secret is not configured");
124
+ const auth = req.headers.get("authorization") ?? "";
125
+ if (auth !== `Bearer ${env.ADMIN_TOKEN}`) return err(401, "unauthorized", "Invalid admin token");
126
+ const body = await readJson(req);
127
+ const units = Array.isArray(body?.units) ? body.units : [];
128
+ if (units.length === 0 || units.length > 6) return err(400, "invalid_input", "units: 1-6 required");
129
+ const model = typeof env.ENRICH_MODEL === "string" && env.ENRICH_MODEL ? env.ENRICH_MODEL : MODELS.enrich;
130
+
131
+ const usage = { prompt_tokens: 0, completion_tokens: 0, requests: 0, cache_hits: 0 };
132
+ const results = await Promise.all(
133
+ units.map(async (u: any) => {
134
+ const m = u?.metadata ?? {};
135
+ if (!u?.id || typeof u?.id !== "string" || !m?.doc_id || !m?.clause_anchor || !Array.isArray(u?.children) || u.children.length === 0) {
136
+ return { id: u?.id ?? null, ok: false, error: "invalid unit (id, metadata.doc_id, metadata.clause_anchor, children required)" };
137
+ }
138
+ try {
139
+ const cacheKey = `s:${u.id}`;
140
+ let summary = body?.force === true ? null : await env.CACHE.get(cacheKey);
141
+ const cached = !!summary;
142
+ if (!summary) {
143
+ const head = `${m.docidentifier ?? m.doc_id} §${m.clause_anchor}${m.clause_title ? " — " + m.clause_title : ""}`;
144
+ const listing = u.children
145
+ .slice(0, 12)
146
+ .map((c: any) => `§${c.anchor ?? ""}${c.title ? " " + c.title : ""} — ${String(c.excerpt ?? "").slice(0, 260)}`)
147
+ .join("\n");
148
+ const res: any = await env.AI.run(model, {
149
+ messages: [
150
+ { role: "system", content: sectionSummaryPrompt.trimEnd() },
151
+ { role: "user", content: `${head}\n\nSub-clauses:\n${listing}` },
152
+ ],
153
+ max_tokens: 1600, // parity with the chunk-enrichment call — 900 starved ~40% of section summaries (model-card budget rule)
154
+ reasoning_effort: "low",
155
+ });
156
+ const raw = typeof res?.response === "string" && res.response.trim() ? res.response : res?.choices?.[0]?.message?.content;
157
+ summary = typeof raw === "string" ? raw.trim().replace(/^["']|["']$/g, "").slice(0, 500) : "";
158
+ if (!summary) return { id: u.id, ok: false, error: "empty summary" };
159
+ if (res?.usage) {
160
+ usage.prompt_tokens += Number(res.usage.prompt_tokens ?? 0);
161
+ usage.completion_tokens += Number(res.usage.completion_tokens ?? 0);
162
+ }
163
+ usage.requests += 1;
164
+ ctx.waitUntil(env.CACHE.put(cacheKey, summary, { expirationTtl: 2_592_000 }));
165
+ } else {
166
+ usage.cache_hits += 1;
167
+ }
168
+ const childAnchors = u.children.map((c: any) => c.anchor).filter(Boolean).join(",");
169
+ const text = `§${m.clause_anchor}${m.clause_title ? " " + m.clause_title : ""} — ${summary}\nCovers: ${childAnchors}`;
170
+ const vectorText = `${m.docidentifier ?? m.doc_id} §${m.clause_anchor} ${text}`.slice(0, 2000);
171
+ const vector = await embed(portModelRunner(env), MODELS.embed, vectorText);
172
+ await env.VECTORIZE.upsert([
173
+ { id: u.id, values: vector, metadata: { ...m, chunk_text: text, section_summary: "1", child_anchors: childAnchors, ctx: "1" } },
174
+ ]);
175
+ return { id: u.id, ok: true, cached, children: u.children.length };
176
+ } catch (e: any) {
177
+ return { id: u.id, ok: false, error: String(e?.message ?? e).slice(0, 200) };
178
+ }
179
+ }),
180
+ );
181
+ const ok = results.filter((r: any) => r.ok).length;
182
+ console.log("section units:", ok, "/", results.length, "usage:", JSON.stringify(usage));
183
+ if (usage.requests > 0) {
184
+ ctx.waitUntil(
185
+ env.DB.prepare(
186
+ "INSERT INTO spend (day, tier, model, requests) VALUES (?1,'enrich',?2,?3) ON CONFLICT(day, tier, model) DO UPDATE SET requests = requests + ?3",
187
+ )
188
+ .bind(today(), model, usage.requests)
189
+ .run(),
190
+ );
191
+ }
192
+ return json({ results, usage });
193
+ }
194
+
195
+ /** Ops access to the Vectorize binding (get/upsert by id) for offline
196
+ * passes like embedding smoothing (G-ETSI-4) — the binding is the
197
+ * credential, admin-token gated exactly like /admin/enrich. */
198
+ /** One-time figure captioning (TODO.remaining/03): fetch the unit's asset
199
+ * from R2, describe it with the vision-capable answer model, store the
200
+ * description into unit_payloads. Admin-gated; idempotent. */
201
+ export async function handleCaption(env: Env, req: Request): Promise<Response> {
202
+ if (!env.ADMIN_TOKEN) return err(501, "admin_disabled", "ADMIN_TOKEN secret is not configured");
203
+ const auth = req.headers.get("authorization") ?? "";
204
+ if (auth !== `Bearer ${env.ADMIN_TOKEN}`) return err(401, "unauthorized", "Invalid admin token");
205
+ const body = await readJson(req);
206
+ const unitId = typeof body?.unit_id === "string" ? body.unit_id : "";
207
+ const context = typeof body?.context === "string" ? body.context.slice(0, 400) : "";
208
+ if (!unitId) return err(400, "invalid_input", "unit_id required");
209
+ try {
210
+ const row = await env.DB.prepare("SELECT payload, docidentifier FROM unit_payloads WHERE unit_id = ?1").bind(unitId).first<any>();
211
+ if (!row) return err(404, "not_found", "no unit_payload row for that id");
212
+ const payload = JSON.parse(String(row.payload));
213
+ const uri = payload.uri ?? "";
214
+ const m = uri.match(/^\/assets\/(.+)/);
215
+ if (!m) return err(400, "invalid_input", "payload has no /assets/ uri (upload the asset first)");
216
+ const obj = await env.UNIT_ASSETS.get(m[1]);
217
+ if (!obj) return err(404, "not_found", `asset ${m[1]} not in R2`);
218
+ const buf = await obj.arrayBuffer();
219
+ const ext = m[1].split(".").pop()?.toLowerCase() ?? "png";
220
+ const mime = ext === "svg" ? "image/svg+xml" : `image/${ext === "jpg" ? "jpeg" : ext}`;
221
+ // chunked: spreading the whole byte array blows the V8 stack on large
222
+ // assets (the 389KB u:figure-1)
223
+ const bytes = new Uint8Array(buf);
224
+ let binary = "";
225
+ for (let i = 0; i < bytes.length; i += 8192) binary += String.fromCharCode(...bytes.subarray(i, i + 8192));
226
+ const b64 = btoa(binary);
227
+ let res: any = null;
228
+ for (let attempt = 0; attempt < 2 && !res; attempt++) {
229
+ try {
230
+ res = await env.AI.run(MODELS.member, {
231
+ messages: [
232
+ {
233
+ role: "user",
234
+ content: [
235
+ { type: "text", text: `Describe this figure from ${row.docidentifier}${context ? ` (${context})` : ""} for a reader who cannot see it: what is plotted/shown, the axes or structure, and the normative point it makes. 2-3 plain sentences.` },
236
+ { type: "image_url", image_url: { url: `data:${mime};base64,${b64}` } },
237
+ ],
238
+ },
239
+ ],
240
+ max_tokens: 1024,
241
+ // GLM-5.3-Flash defaults to reasoning_effort "max" when the parameter
242
+ // is absent — max-effort reasoning starves a 1024-token budget and
243
+ // the caption comes back empty (the u:fig-2 straggler)
244
+ reasoning_effort: "low",
245
+ });
246
+ } catch {
247
+ // transient Workers AI flake (8005) — retry once
248
+ }
249
+ }
250
+ const text = typeof res?.response === "string" ? res.response : res?.choices?.[0]?.message?.content;
251
+ if (!text?.trim()) return err(502, "generation_failed", "vision model returned no description");
252
+ const desc = text.trim().slice(0, 600);
253
+ await env.DB.prepare("UPDATE unit_payloads SET payload = json_set(payload, '$.description', ?1) WHERE unit_id = ?2").bind(desc, unitId).run();
254
+ return json({ ok: true, unit_id: unitId, description: desc });
255
+ } catch (e) {
256
+ return err(502, "caption_failed", String(e).slice(0, 200));
257
+ }
258
+ }
259
+
260
+ export async function handleVectors(env: Env, req: Request): Promise<Response> {
261
+ if (!env.ADMIN_TOKEN) return err(501, "admin_disabled", "ADMIN_TOKEN secret is not configured");
262
+ const auth = req.headers.get("authorization") ?? "";
263
+ if (auth !== `Bearer ${env.ADMIN_TOKEN}`) return err(401, "unauthorized", "Invalid admin token");
264
+ const body = await readJson(req);
265
+ const mode = body?.mode;
266
+ try {
267
+ if (mode === "get") {
268
+ const ids = Array.isArray(body?.ids) ? body.ids.filter((x: unknown) => typeof x === "string").slice(0, 100) : [];
269
+ if (!ids.length) return err(400, "invalid_input", "ids: 1-100 required");
270
+ // getByIds above ~20 ids returns EMPTY (observed: 16/20 fine, 24+
271
+ // silently zero) — chunk server-side so the documented 100 actually
272
+ // works instead of lying to callers
273
+ const vectors: any[] = [];
274
+ for (let i = 0; i < ids.length; i += 20) {
275
+ const got = (await env.VECTORIZE.getByIds(ids.slice(i, i + 20))) ?? [];
276
+ for (const v of got) vectors.push({ id: v.id, values: v.values, metadata: v.metadata ?? null });
277
+ }
278
+ return json({ vectors });
279
+ }
280
+ if (mode === "upsert") {
281
+ const vectors = Array.isArray(body?.vectors)
282
+ ? body.vectors.filter((v: any) => v && typeof v.id === "string" && Array.isArray(v.values))
283
+ : [];
284
+ if (!vectors.length || vectors.length > 100) return err(400, "invalid_input", "vectors: 1-100 required");
285
+ await env.VECTORIZE.upsert(vectors);
286
+ return json({ ok: true, upserted: vectors.length });
287
+ }
288
+ if (mode === "embed") {
289
+ // comparison-lane indexing (TODO.model-rag): embed text via the
290
+ // binding's model so lane builders don't need AI REST scope
291
+ const texts = Array.isArray(body?.texts) ? body.texts.filter((t: unknown) => typeof t === "string").slice(0, 16) : [];
292
+ if (!texts.length) return err(400, "invalid_input", "texts: 1-16 required");
293
+ const vectors: number[][] = [];
294
+ for (const t of texts) {
295
+ const v = await embed(portModelRunner(env), MODELS.embed, t.slice(0, 6000));
296
+ vectors.push(v);
297
+ }
298
+ return json({ vectors });
299
+ }
300
+ return err(400, "invalid_input", "mode must be get, upsert, or embed");
301
+ } catch (e) {
302
+ return err(502, "vectorize_failed", String(e).slice(0, 200));
303
+ }
304
+ }
305
+
306
+ export async function handleJudge(env: Env, req: Request): Promise<Response> {
307
+ if (!env.ADMIN_TOKEN) return err(501, "admin_disabled", "ADMIN_TOKEN secret is not configured");
308
+ const auth = req.headers.get("authorization") ?? "";
309
+ if (auth !== `Bearer ${env.ADMIN_TOKEN}`) return err(401, "unauthorized", "Invalid admin token");
310
+ const body = await readJson(req);
311
+ const question = typeof body?.question === "string" ? body.question.slice(0, 2000) : "";
312
+ const answer = typeof body?.answer === "string" ? body.answer.slice(0, 4000) : "";
313
+ const passages = Array.isArray(body?.passages)
314
+ ? body.passages.filter((p: unknown) => typeof p === "string").map((p: string) => p.slice(0, 600)).slice(0, 8)
315
+ : [];
316
+ if (!question || !answer) return err(400, "invalid_input", "question and answer required");
317
+
318
+ const passagesText = passages.map((p: string, i: number) => `[${i + 1}] ${p}`).join("\n");
319
+ const [faith, relevancy, precision] = await Promise.all([
320
+ passages.length ? scoreFaithfulness(env.AI, MODELS.grader, answer, passages) : Promise.resolve(null),
321
+ scoreJudge(env.AI, MODELS.grader, relevancyPrompt, `Question: ${question}\n\nAnswer:\n${answer}`),
322
+ passages.length ? scoreJudge(env.AI, MODELS.grader, precisionPrompt, `Question: ${question}\n\nPassages:\n${passagesText}`) : Promise.resolve(null),
323
+ ]);
324
+ return json({
325
+ question_hash: await sha256Hex(question),
326
+ faithfulness: faith ? faith.score : null,
327
+ answer_relevancy: relevancy,
328
+ context_precision: precision,
329
+ });
330
+ }
331
+
332
+ export async function handleCreateKey(env: Env, req: Request): Promise<Response> {
333
+ if (!env.ADMIN_TOKEN) return err(501, "admin_disabled", "ADMIN_TOKEN secret is not configured");
334
+ const auth = req.headers.get("authorization") ?? "";
335
+ if (auth !== `Bearer ${env.ADMIN_TOKEN}`) return err(401, "unauthorized", "Invalid admin token");
336
+ const body = await readJson(req);
337
+ if (!body?.name || typeof body.name !== "string") return err(400, "invalid_input", "name is required");
338
+ const dayLimit = Number.isFinite(Number(body.day_limit)) && Number(body.day_limit) > 0 ? Number(body.day_limit) : num(env as any, "KEY_DAY_ASK_DEFAULT", 2000);
339
+ const raw = `oiml_${[...crypto.getRandomValues(new Uint8Array(24))].map((b) => b.toString(16).padStart(2, "0")).join("")}`;
340
+ const id = crypto.randomUUID();
341
+ const keyHash = await sha256Hex(raw);
342
+ await env.DB.prepare(
343
+ "INSERT INTO api_keys (id, name, key_hash, day_limit, created_at, revoked) VALUES (?1,?2,?3,?4,?5,0)",
344
+ )
345
+ .bind(id, body.name, keyHash, dayLimit, new Date().toISOString())
346
+ .run();
347
+ return json({ id, name: body.name, day_limit: dayLimit, key: raw, note: "Store this key now — it is not retrievable again." });
348
+ }
349
+
350
+ export async function handleListKeys(env: Env, req: Request): Promise<Response> {
351
+ if (!env.ADMIN_TOKEN) return err(501, "admin_disabled", "ADMIN_TOKEN secret is not configured");
352
+ const auth = req.headers.get("authorization") ?? "";
353
+ if (auth !== `Bearer ${env.ADMIN_TOKEN}`) return err(401, "unauthorized", "Invalid admin token");
354
+ const rows = await env.DB.prepare(
355
+ "SELECT id, name, day_limit, created_at, revoked FROM api_keys ORDER BY created_at DESC",
356
+ ).all();
357
+ return json({ keys: rows.results, ...corsHeaders(req) });
358
+ }
@@ -0,0 +1,71 @@
1
+ import type { ModelRunner } from "./ports/model.ts";
2
+ // Model calls go through the ModelRunner port (ports/model.ts); the
3
+ // request-shape probing that used to live here is adapter behavior now
4
+ // (ports/cloudflare/adapters.ts).
5
+ type AnyAi = ModelRunner;
6
+
7
+
8
+
9
+ const delay = (ms: number) => new Promise((r) => setTimeout(r, ms));
10
+
11
+ export async function embed(ai: ModelRunner, _model: string, text: string): Promise<number[]> {
12
+ // the adapter owns the request shape; two attempts cover transient
13
+ // capacity errors shared with the ingest lane
14
+ let lastError: unknown = null;
15
+ for (let attempt = 0; attempt < 3; attempt++) {
16
+ try {
17
+ const vecs = await ai.embed([text]);
18
+ if (vecs?.[0]?.length) return vecs[0];
19
+ lastError = new Error("adapter returned no vector");
20
+ } catch (e) {
21
+ lastError = e;
22
+ }
23
+ if (attempt < 2) await delay(250 * (attempt + 1));
24
+ }
25
+ // never return an empty vector — the index rejects it with an opaque
26
+ // 40006; a throw surfaces the real failure through the caller's catch
27
+ throw new Error(`embed failed after retries: ${String(lastError)}`);
28
+ }
29
+
30
+ import { answerEffort, effortBudget } from "./config.ts";
31
+
32
+ export async function rerank(
33
+ ai: AnyAi,
34
+ model: string,
35
+ query: string,
36
+ texts: string[],
37
+ ): Promise<number[] | null> {
38
+ // verified REST shape first: contexts are [{text}] objects; the binding
39
+ // may instead want plain strings — try both, then the legacy names
40
+ // the adapter's shape probing handles provider variants; two
41
+ // attempts cover transient capacity errors
42
+ for (let attempt = 0; attempt < 2; attempt++) {
43
+ const scores = await ai.rerank(model, query, texts);
44
+ if (scores && scores.some((s) => Number.isFinite(s))) return scores;
45
+ }
46
+ console.error("rerank failed, using vector order");
47
+ return null;
48
+ }
49
+
50
+ export async function generateOnce(env: any, model: string, messages: any[], effort?: string): Promise<string | null> {
51
+ // one immediate retry: Workers AI intermittently 8005s a call that
52
+ // succeeds unchanged on the second attempt — a flake must not degrade
53
+ // the answer (a multimodal primary falling to a text-only fallback
54
+ // silently blindfolds figure answers)
55
+ for (let attempt = 0; attempt < 2; attempt++) {
56
+ try {
57
+ const res: any = await env.AI.run(model, {
58
+ messages,
59
+ max_tokens: effortBudget(effort ?? answerEffort(env)),
60
+ reasoning_effort: effort ?? answerEffort(env),
61
+ temperature: 0.6,
62
+ top_p: 0.95,
63
+ });
64
+ if (typeof res?.response === "string") return res.response;
65
+ if (typeof res?.choices?.[0]?.message?.content === "string") return res.choices[0].message.content;
66
+ } catch (e) {
67
+ console.error("generate failed:", model, String(e).slice(0, 120));
68
+ }
69
+ }
70
+ return null;
71
+ }