@promptev/context-engine 0.0.3 → 0.0.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +25 -1
- package/dist/cli.js +527 -16
- package/dist/cli.js.map +1 -1
- package/dist/index.cjs +532 -16
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.cts +244 -3
- package/dist/index.d.ts +244 -3
- package/dist/index.js +528 -17
- package/dist/index.js.map +1 -1
- package/dist/{mcp-BKSmxayM.d.cts → mcp-uirRbluA.d.cts} +4 -0
- package/dist/{mcp-BKSmxayM.d.ts → mcp-uirRbluA.d.ts} +4 -0
- package/dist/mcp.cjs +50 -10
- package/dist/mcp.cjs.map +1 -1
- package/dist/mcp.d.cts +1 -1
- package/dist/mcp.d.ts +1 -1
- package/dist/mcp.js +50 -10
- package/dist/mcp.js.map +1 -1
- package/dist/skills/context-engine/SKILL.md +29 -1
- package/package.json +1 -1
- package/src/skills/context-engine/SKILL.md +29 -1
package/dist/mcp.cjs.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/errors.ts","../src/extras.ts","../src/text.ts","../src/hooks.ts","../src/providers/google.ts","../src/providers/llm.ts","../src/crypto.ts","../src/redaction.ts","../src/extraction/json.ts","../src/extraction/files.ts","../src/extraction/index.ts","../src/graph/retrieval.ts","../src/graph/navigation.ts","../src/mcp.ts","../src/graph/entities.ts","../src/chunkers.ts","../src/actions.ts","../src/sandbox.ts","../src/structured.ts","../src/sentinels.ts","../src/ingest.ts","../src/knowledge-tool.ts","../src/routing-core.ts","../src/tools/mcp-tools.ts","../src/version.ts"],"names":["out","z","handler"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAAA,IAGa,mBAQA,qBAAA,EA0CA,iBAAA;AArDb,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,eAAA,GAAA;AAGO,IAAM,iBAAA,GAAN,cAAgC,KAAA,CAAM;AAAA,MAC3C,YAAY,OAAA,EAAiB;AAC3B,QAAA,KAAA,CAAM,OAAO,CAAA;AACb,QAAA,IAAA,CAAK,IAAA,GAAO,mBAAA;AAAA,MACd;AAAA,KACF;AAGO,IAAM,qBAAA,GAAN,cAAoC,KAAA,CAAM;AAAA,MAC/C,YAAY,UAAA,EAAoB;AAC9B,QAAA,KAAA,CAAM,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAE,CAAA;AACzC,QAAA,IAAA,CAAK,IAAA,GAAO,uBAAA;AAAA,MACd;AAAA,KACF;AAqCO,IAAM,iBAAA,GAAN,cAAgC,KAAA,CAAM;AAAA,MAC3C,WAAA,CAAY,KAAA,EAAe,GAAA,EAAa,IAAA,EAAc;AACpD,QAAA,KAAA,CAAM,CAAA,EAAG,IAAI,CAAA,eAAA,EAAkB,GAAG,8BAA8B,KAAK,CAAA,eAAA,EAAkB,GAAG,CAAA,CAAE,CAAA;AAC5F,QAAA,IAAA,CAAK,IAAA,GAAO,mBAAA;AAAA,MACd;AAAA,KACF;AAAA,EAAA;AAAA,CAAA,CAAA;;;AChDA,eAAsB,YAAA,CAA0B,SAAA,EAAmB,KAAA,EAAe,IAAA,EAA0B;AAC1G,EAAA,IAAI;AACF,IAAA,OAAQ,MAAM,OAAO,SAAA,CAAA;AAAA,EACvB,SAAS,IAAA,EAAM;AACb,IAAA,MAAM,IAAI,iBAAA,CAAkB,KAAA,EAAO,SAAA,EAAW,IAAI,CAAA;AAAA,EACpD;AACF;AAhBA,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,eAAA,GAAA;AAAA,IAAA,WAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACAA,IAkRM,mBAAA;AAlRN,IAAA,SAAA,GAAA,KAAA,CAAA;AAAA,EAAA,aAAA,GAAA;AAkRA,IAAM,mBAAA,GAAsB,CAAA,kqaAAA,CAAA;AAI5B,IAAsC,mBAAA,CAAoB,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,CAAC,KAAA,KAAU;AAClF,MAAA,MAAM,CAAC,KAAA,EAAO,GAAA,EAAK,IAAI,CAAA,GAAI,KAAA,CAAM,MAAM,GAAG,CAAA;AAC1C,MAAA,OAAO,EAAE,KAAA,EAAO,MAAA,CAAO,QAAA,CAAS,KAAA,EAAQ,EAAE,CAAA,EAAG,GAAA,EAAK,MAAA,CAAO,QAAA,CAAS,KAAM,EAAE,CAAA,EAAG,IAAA,EAAM,MAAA,CAAO,IAAI,CAAA,EAAE;AAAA,IAClG,CAAC,CAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACzRD,IAAA,UAAA,GAAA,KAAA,CAAA;AAAA,EAAA,cAAA,GAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACAA,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AAmCA,IAAA,WAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACnCA,IAAA,QAAA,GAAA,KAAA,CAAA;AAAA,EAAA,sBAAA,GAAA;AAaA,IAAA,WAAA,EAAA;AAEA,IAAA,WAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACfA,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,eAAA,GAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACAA,IAAA,cAAA,GAAA,KAAA,CAAA;AAAA,EAAA,kBAAA,GAAA;AAAA,IAAA,WAAA,EAAA;AACA,IAAA,UAAA,EAAA;AA0HA,EAAA;AAAA,CAAA,CAAA;AC3HA,IAAA,SAAA,GAAA,KAAA,CAAA;AAAA,EAAA,wBAAA,GAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACAA,IAAA,UAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AAcA,IAAkB,OAAO,IAAA,CAAK,CAAC,IAAM,EAAA,EAAM,CAAA,EAAM,CAAI,CAAC,CAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACdtD,IAAA,eAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AAOA,IAAA,WAAA,EAAA;AACA,IAAA,UAAA,EAAA;AACA,IAAA,UAAA,EAAA;AAoEqC,EAAA;AAAA,CAAA,CAAA;;;ACxCrC,eAAsB,kBAAA,CAAmB,MAAY,SAAA,EAA8C;AACjG,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA,kGAAA,CAAA;AAAA,IAEA,CAAC,SAAS;AAAA,GACZ;AACA,EAAA,OAAO,CAAC,MAAA,CAAO,IAAA,CAAK,CAAC,CAAA,EAAG,OAAA;AAC1B;AAEA,eAAsB,2BAAA,CACpB,MACA,IAAA,EACkB;AAClB,EAAA,IAAI,IAAA,CAAK,UAAA,IAAc,IAAA,EAAM,OAAO,IAAA;AACpC,EAAA,OAAO,kBAAA,CAAmB,IAAA,EAAM,IAAA,CAAK,SAAS,CAAA;AAChD;AApDA,IAAA,cAAA,GAAA,KAAA,CAAA;AAAA,EAAA,wBAAA,GAAA;AAEA,IAAA,UAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACFA,IAAA,kBAAA,GAAA,EAAA;AAAA,QAAA,CAAA,kBAAA,EAAA;AAAA,EAAA,kBAAA,EAAA,MAAA,kBAAA;AAAA,EAAA,gBAAA,EAAA,MAAA,gBAAA;AAAA,EAAA,WAAA,EAAA,MAAA,WAAA;AAAA,EAAA,YAAA,EAAA,MAAA,YAAA;AAAA,EAAA,QAAA,EAAA,MAAA;AAAA,CAAA,CAAA;AAgFA,SAAS,KAAA,CAAM,KAAA,EAAkC,QAAA,EAAkB,OAAA,EAAyB;AAC1F,EAAA,MAAM,CAAA,GAAI,OAAO,KAAA,KAAU,QAAA,IAAY,MAAA,CAAO,QAAA,CAAS,KAAK,CAAA,GAAI,IAAA,CAAK,KAAA,CAAM,KAAK,CAAA,GAAI,QAAA;AACpF,EAAA,OAAO,KAAK,GAAA,CAAI,CAAA,EAAG,KAAK,GAAA,CAAI,CAAA,EAAG,OAAO,CAAC,CAAA;AACzC;AAEA,SAAS,UAAU,IAAA,EAAqE;AACtF,EAAA,OAAO,CAAC,KAAK,SAAA,IAAa,IAAA,EAAM,KAAK,WAAA,IAAe,IAAA,EAAM,IAAA,CAAK,UAAA,IAAc,IAAI,CAAA;AACnF;AAEA,eAAe,aAAA,CACb,IAAA,EACA,IAAA,EACA,IAAA,EAC4D;AAC5D,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA,sCAAA,EACoC,cAAc,CAAA,QAAA,CAAA;AAAA,IAClD,CAAC,GAAG,SAAA,CAAU,IAAI,CAAA,EAAA,CAAI,QAAQ,EAAA,EAAI,IAAA,EAAK,CAAE,WAAA,EAAa;AAAA,GACxD;AACA,EAAA,MAAM,GAAA,GAAM,MAAA,CAAO,IAAA,CAAK,CAAC,CAAA;AACzB,EAAA,OAAO,GAAA,GAAM,EAAE,EAAA,EAAI,MAAA,CAAO,GAAA,CAAI,EAAE,CAAA,EAAG,IAAA,EAAM,GAAA,CAAI,IAAA,EAAM,IAAA,EAAM,GAAA,CAAI,MAAK,GAAI,IAAA;AACxE;AAGA,eAAsB,YAAA,CACpB,MACA,IAAA,EACkC;AAClC,EAAA,MAAM,QAAQ,MAAM,aAAA,CAAc,IAAA,EAAM,IAAA,CAAK,QAAQ,IAAI,CAAA;AACzD,EAAA,IAAI,CAAC,KAAA,EAAO,OAAO,EAAE,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,SAAA,EAAW,SAAA,EAAW,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AAC7E,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,KAAA,EAMG,YAAY;AAAA,iCAAA,CAAA;AAAA,IAEf,CAAC,GAAG,SAAA,CAAU,IAAI,CAAA,EAAG,KAAA,CAAM,EAAA,EAAI,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,EAAA,EAAI,SAAS,CAAC;AAAA,GACjE;AACA,EAAA,MAAM,SAAA,GAAY,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IACxC,MAAM,CAAA,CAAE,IAAA;AAAA,IACR,MAAM,CAAA,CAAE,IAAA;AAAA,IACR,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,OAAO,CAAA,CAAE,KAAA;AAAA,IACT,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,WAAW,CAAA,CAAE;AAAA,GACf,CAAE,CAAA;AACF,EAAA,OAAO;AAAA,IACL,KAAA,EAAO,IAAA;AAAA,IACP,QAAQ,EAAE,IAAA,EAAM,MAAM,IAAA,EAAM,IAAA,EAAM,MAAM,IAAA,EAAK;AAAA,IAC7C,SAAA;AAAA,IACA,OAAO,SAAA,CAAU;AAAA,GACnB;AACF;AAGA,eAAsB,QAAA,CACpB,MACA,IAAA,EACkC;AAClC,EAAA,MAAM,QAAQ,MAAM,aAAA,CAAc,IAAA,EAAM,IAAA,CAAK,QAAQ,IAAI,CAAA;AACzD,EAAA,IAAI,CAAC,KAAA,EAAO,OAAO,EAAE,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,SAAA,EAAW,KAAA,EAAO,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AAGzE,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,OAAA,EAOK,YAAY;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,SAAA,EAWV,YAAY;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,aAAA,CAAA;AAAA,IAQnB;AAAA,MACE,GAAG,UAAU,IAAI,CAAA;AAAA,MACjB,KAAA,CAAM,EAAA;AAAA,MACN,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,CAAA,EAAG,SAAS,CAAA;AAAA,MAC9B,KAAK,QAAA,IAAY,IAAA;AAAA,MACjB,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,EAAA,EAAI,SAAS;AAAA;AACjC,GACF;AACA,EAAA,MAAM,KAAA,GAAQ,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IACpC,QAAQ,CAAA,CAAE,IAAA;AAAA,IACV,aAAa,CAAA,CAAE,IAAA;AAAA,IACf,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,OAAO,CAAA,CAAE,KAAA;AAAA,IACT,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,MAAM,CAAA,CAAE;AAAA,GACV,CAAE,CAAA;AACF,EAAA,KAAA,CAAM,KAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,OAAO,CAAA,CAAE,IAAA,IAAQ,MAAA,CAAO,CAAA,CAAE,MAAM,CAAA,CAAE,aAAA,CAAc,OAAO,CAAA,CAAE,MAAM,CAAC,CAAC,CAAA;AACxF,EAAA,OAAO,EAAE,KAAA,EAAO,IAAA,EAAM,MAAA,EAAQ,EAAE,IAAA,EAAM,KAAA,CAAM,IAAA,EAAM,IAAA,EAAM,MAAM,IAAA,EAAK,EAAG,KAAA,EAAO,KAAA,EAAO,MAAM,MAAA,EAAO;AACnG;AAGA,eAAsB,WAAA,CACpB,MACA,IAAA,EAMkC;AAClC,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,KAAA,EAQG,YAAY;AAAA,yCAAA,CAAA;AAAA,IAEf;AAAA,MACE,GAAG,UAAU,IAAI,CAAA;AAAA,MACjB,KAAK,QAAA,IAAY,IAAA;AAAA,MACjB,KAAK,KAAA,IAAS,IAAA;AAAA,MACd,KAAK,UAAA,IAAc,IAAA;AAAA,MACnB,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,EAAA,EAAI,SAAS;AAAA;AACjC,GACF;AACA,EAAA,MAAM,aAAA,GAAgB,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IAC5C,QAAQ,CAAA,CAAE,QAAA;AAAA,IACV,aAAa,CAAA,CAAE,QAAA;AAAA,IACf,QAAQ,CAAA,CAAE,QAAA;AAAA,IACV,aAAa,CAAA,CAAE,QAAA;AAAA,IACf,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,OAAO,CAAA,CAAE,KAAA;AAAA,IACT,UAAU,CAAA,CAAE;AAAA,GACd,CAAE,CAAA;AACF,EAAA,OAAO,EAAE,KAAA,EAAO,IAAA,EAAM,aAAA,EAAe,KAAA,EAAO,cAAc,MAAA,EAAO;AACnE;AAUA,eAAsB,gBAAA,CACpB,MACA,IAAA,EAOkC;AAClC,EAAA,MAAM,OAAA,GAAU,MAAM,2BAAA,CAA4B,IAAA,EAAM;AAAA,IACtD,SAAA,EAAW,KAAK,SAAA,IAAa,IAAA;AAAA,IAC7B,UAAA,EAAY,KAAK,UAAA,IAAc;AAAA,GAChC,CAAA;AACD,EAAA,IAAI,CAAC,OAAA,EAAS;AACZ,IAAA,OAAO,EAAE,WAAW,KAAA,EAAO,KAAA,EAAO,oBAAoB,WAAA,EAAa,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AAAA,EAClF;AACA,EAAA,MAAM,OAAA,GAAU,MAAM,IAAA,CAAK,QAAA,CAAS,KAAA,CAAM,CAAC,IAAA,CAAK,KAAA,IAAS,EAAE,CAAA,EAAG,OAAO,CAAA;AACrE,EAAA,MAAM,MAAA,GAAS,UAAU,CAAC,CAAA;AAC1B,EAAA,IAAI,CAAC,MAAA,EAAQ,OAAO,EAAE,SAAA,EAAW,MAAM,WAAA,EAAa,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AACjE,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA,uDAAA,CAAA;AAAA,IAKA,CAAC,CAAA,CAAA,EAAI,MAAA,CAAO,IAAI,CAAC,CAAA,KAAM,OAAO,CAAC,CAAC,EAAE,IAAA,CAAK,GAAG,CAAC,CAAA,CAAA,CAAA,EAAK,KAAA,CAAM,KAAK,KAAA,EAAO,CAAA,EAAG,EAAE,CAAC;AAAA,GAC1E;AACA,EAAA,MAAM,WAAA,GAAc,MAAA,CAAO,IAAA,CACxB,MAAA,CAAO,CAAC,CAAA,KAAM,MAAA,CAAO,CAAA,CAAE,GAAG,CAAA,GAAI,uBAAuB,CAAA,CACrD,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IACX,SAAS,CAAA,CAAE,OAAA;AAAA,IACX,cAAc,CAAA,CAAE,YAAA;AAAA,IAChB,oBAAoB,CAAA,CAAE,kBAAA;AAAA,IACtB,iBAAiB,CAAA,CAAE,KAAA;AAAA,IACnB,SAAA,EAAW,KAAK,KAAA,CAAM,MAAA,CAAO,EAAE,GAAG,CAAA,GAAI,GAAK,CAAA,GAAI;AAAA,GACjD,CAAE,CAAA;AACJ,EAAA,OAAO,EAAE,SAAA,EAAW,IAAA,EAAM,WAAA,EAAa,KAAA,EAAO,YAAY,MAAA,EAAO;AACnE;AAxRA,IAoCM,OAOA,YAAA,EAWA,cAAA,EAQA,SAAA,EACA,SAAA,EACA,WACA,uBAAA,EAGO,kBAAA;AApEb,IAAA,eAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AA4BA,IAAA,cAAA,EAAA;AAQA,IAAM,KAAA,GAAQ;AAAA;AAAA;AAAA;AAAA,CAAA;AAOd,IAAM,YAAA,GAAe;AAAA;AAAA;AAAA;AAAA;AAAA,8BAAA,EAKW,KAAK;AAAA;AAAA;AAAA,CAAA;AAMrC,IAAM,cAAA,GAAiB;AAAA;AAAA;AAAA;AAAA,8BAAA,EAIS,KAAK;AAAA;AAAA,CAAA;AAIrC,IAAM,SAAA,GAAY,kBAAA;AAClB,IAAM,SAAA,GAAY,CAAA;AAClB,IAAM,SAAA,GAAY,GAAA;AAClB,IAAM,uBAAA,GAA0B,IAAA;AAGzB,IAAM,kBAAA,GACX,2SAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACpEF,WAAA,EAAA;AACA,WAAA,EAAA;ACIO,IAAM,YAAA,GAAe;AAAA,EAC1B,QAAA;AAAA,EACA,KAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA;AACF,CAAA;AAQO,IAAM,4BAAA,GAA+B;AAAA,EAC1C,cAAA;AAAA,EACA,YAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,WAAA;AAAA,EACA,YAAA;AAAA,EACA;AACF,CAAA;AAEuC,IAAI,GAAA,CAAY,4BAA4B;;;AC/BnF,SAAA,EAAA;AAWkB,IAAI,GAAA;AAAA,EACpB;AACF;AAuIO,SAAS,cAAc,IAAA,EAA0C;AACtE,EAAA,IAAI,CAAC,MAAM,OAAO,KAAA;AAClB,EAAA,MAAM,CAAA,GAAI,KAAK,WAAA,EAAY;AAC3B,EAAA,OAAO,CAAC,KAAA,EAAO,OAAA,EAAS,OAAA,EAAS,eAAA,EAAiB,eAAe,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,KAAM,CAAA,CAAE,QAAA,CAAS,CAAC,CAAC,CAAA;AAC9F;;;ACxIA,WAAA,EAAA;AACA,UAAA,EAAA;AAEA,QAAA,EAAA;AACA,cAAA,EAAA;;;ACGA,WAAA,EAAA;;;ACJA,SAAA,EAAA;AAEA,QAAA,EAAA;AACA,SAAA,EAAA;;;ACNO,IAAM,OAAA,mBAAyB,MAAA,CAAO,GAAA,CAAI,wBAAwB,CAAA;AAYlE,IAAM,QAAA,mBAA0B,MAAA,CAAO,GAAA,CAAI,yBAAyB,CAAA;ACY3E,eAAA,EAAA;AACA,UAAA,EAAA;;;AAGA,cAAA,EAAA;;;AJnBO,IAAM,cAAA,GAAiB,GAAA;;;AKL9B,WAAA,EAAA;AAIO,IAAM,iBAAA,GAAoB;AAAA,EAC/B,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,MAAA;AAAA,EACA,YAAA;AAAA,EACA,SAAA;AAAA,EACA,YAAA;AAAA,EACA,UAAA;AAAA,EACA,YAAA;AAAA,EACA,UAAA;AAAA,EACA,cAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAA;AAGO,IAAM,0BAAA,GACX,yoGAAA;AAiDK,IAAM,wBAAA,GACX,kUAAA;AAOK,IAAM,WAAA,GAAoE;AAAA,EAC/E,MAAA,EAAQ,+EAAA;AAAA,EACR,OAAA,EAAS,mEAAA;AAAA,EACT,IAAA,EAAM,wCAAA;AAAA,EACN,UAAA,EAAY,oEAAA;AAAA,EACZ,OAAA,EACE,kcAAA;AAAA,EAMF,UAAA,EAAY,oEAAA;AAAA,EACZ,QAAA,EAAU,wCAAA;AAAA,EACV,UAAA,EACE,sJAAA;AAAA,EAEF,aAAA,EAAe,qEAAA;AAAA,EACf,QAAA,EAAU,gDAAA;AAAA,EACV,YAAA,EAAc,uEAAA;AAAA,EACd,iBAAA,EAAmB;AACrB,CAAA;AAEA,IAAM,GAAA,GAAM;AAAA,EACV,UAAA,EAAY,2DAAA;AAAA,EACZ,OAAA,EACE,qGAAA;AAAA,EACF,UAAA,EACE,4GAAA;AAAA,EAEF,KAAA,EACE;AAEJ,CAAA;AAEA,IAAM,aAAA,GAAgB,CAAC,UAAA,EAAY,cAAA,EAAgB,iBAAiB,mBAAmB,CAAA;AAMvF,IAAM,mBAAA,GAAsB,CAAC,QAAA,EAAU,SAAA,EAAW,YAAY,YAAY,CAAA;AAG1E,IAAM,sBAAA,GAAyB,GAAA;AASxB,IAAM,gBAAA,GAA4D;AAAA,EACvE,MAAA,EAAQ,EAAE,IAAA,EAAM,QAAA,EAAU,IAAA,EAAM,CAAC,GAAG,iBAAiB,CAAA,EAAG,WAAA,EAAa,wBAAA,EAAyB;AAAA,EAC9F,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,WAAA,EAAa;AAAA,IACX,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,UAAA,EAAY;AAAA,IACV,IAAA,EAAM,OAAA;AAAA,IACN,KAAA,EAAO,EAAE,IAAA,EAAM,QAAA,EAAS;AAAA,IACxB,WAAA,EACE;AAAA,GAIJ;AAAA,EACA,YAAA,EAAc;AAAA,IACZ,IAAA,EAAM,OAAA;AAAA,IACN,KAAA,EAAO,EAAE,IAAA,EAAM,QAAA,EAAS;AAAA,IACxB,WAAA,EACE;AAAA,GAGJ;AAAA,EACA,MAAA,EAAQ;AAAA,IACN,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAIJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EAAa;AAAA,GACf;AAAA,EACA,QAAA,EAAU;AAAA,IACR,IAAA,EAAM,QAAA;AAAA,IACN,IAAA,EAAM,CAAC,GAAG,4BAA4B,CAAA;AAAA,IACtC,WAAA,EACE;AAAA,GAGJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAGJ;AAAA,EACA,WAAA,EAAa;AAAA,IACX,IAAA,EAAM,QAAA;AAAA,IACN,IAAA,EAAM,CAAC,GAAG,YAAY,CAAA;AAAA,IACtB,WAAA,EAAa;AAAA,GACf;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EAAa;AAAA,GACf;AAAA,EACA,IAAA,EAAM;AAAA,IACJ,IAAA,EAAM,QAAA;AAAA,IACN,IAAA,EAAM,CAAC,QAAA,EAAU,OAAO,CAAA;AAAA,IACxB,WAAA,EACE;AAAA,GAKJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA,GAIJ;AAAA,EACA,MAAA,EAAQ;AAAA,IACN,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,GAAA,EAAK;AAAA,IACH,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,SAAA,EAAW;AAAA,IACT,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA;AAGN,CAAA;AAoDO,SAAS,aAAa,KAAA,EAA0B;AACrD,EAAA,IAAI,UAAU,QAAA,EAAU,OAAO,EAAE,SAAA,EAAW,IAAA,EAAM,aAAa,IAAA,EAAK;AACpE,EAAA,IAAI,SAAS,IAAA,EAAM;AACjB,IAAA,MAAM,IAAI,KAAA;AAAA,MACR;AAAA,KAEF;AAAA,EACF;AACA,EAAA,IAAI,KAAA,CAAM,OAAA,CAAQ,KAAK,CAAA,EAAG;AACxB,IAAA,OAAO,EAAE,SAAA,EAAW,CAAC,GAAG,IAAI,GAAA,CAAI,KAAA,CAAM,GAAA,CAAI,MAAM,CAAC,CAAC,CAAA,EAAG,aAAa,IAAA,EAAK;AAAA,EACzE;AACA,EAAA,IAAI,OAAO,UAAU,QAAA,EAAU;AAC7B,IAAA,OAAO;AAAA,MACL,SAAA,EAAW,KAAA,CAAM,SAAA,IAAa,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,KAAA,CAAM,SAAA,CAAU,GAAA,CAAI,MAAM,CAAC,CAAC,CAAA,GAAI,IAAA;AAAA,MACjF,WAAA,EAAa,KAAA,CAAM,WAAA,IAAe,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,KAAA,CAAM,WAAA,CAAY,GAAA,CAAI,MAAM,CAAC,CAAC,CAAA,GAAI;AAAA,KACzF;AAAA,EACF;AACA,EAAA,MAAM,IAAI,KAAA,CAAM,CAAA,sEAAA,EAAoE,OAAO,KAAK,CAAA,CAAE,CAAA;AACpG;AAaO,SAAS,eAAA,CACd,WACA,OAAA,EACiB;AACjB,EAAA,IAAI,OAAA,IAAW,IAAA,EAAM,OAAO,SAAA,IAAa,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,SAAS,CAAC,CAAA,GAAI,IAAA;AAC1E,EAAA,IAAI,SAAA,IAAa,IAAA,EAAM,OAAO,CAAC,GAAG,OAAO,CAAA;AACzC,EAAA,MAAM,OAAA,GAAU,IAAI,GAAA,CAAI,OAAO,CAAA;AAC/B,EAAA,OAAO,CAAC,GAAG,IAAI,GAAA,CAAI,SAAS,CAAC,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,KAAM,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAC,CAAA;AAC7D;AAGA,SAAS,aAAA,CAAc,YAAoB,OAAA,EAAyB;AAClE,EAAA,OAAO,OAAA,CAAQ,eAAe,IAAA,IAAQ,OAAA,CAAQ,YAAY,QAAA,CAAS,MAAA,CAAO,UAAU,CAAC,CAAA;AACvF;AAiBO,SAAS,yBAAA,CACd,MAAA,EAGA,IAAA,GAAqE,EAAC,EACf;AACvD,EAAA,MAAM,MAAA,GAAS,MAAA,CAAO,MAAA,CAAO,GAAA,IAAO,IAAA;AACpC,EAAA,MAAM,OAAA,GAAU,OAAA,CAAQ,MAAA,CAAO,MAAA,CAAO,OAAO,OAAO,CAAA;AACpD,EAAA,OAAO;AAAA,IACL,MAAA,EAAQ,IAAA;AAAA,IACR,OAAA,EAAS,IAAA;AAAA,IACT,IAAA,EAAM,IAAA;AAAA,IACN,UAAA,EAAY,MAAA;AAAA,IACZ,OAAA,EAAS,KAAK,OAAA,IAAW,IAAA,IAAS,UAAU,OAAA,CAAQ,MAAA,CAAO,OAAO,mBAAmB,CAAA;AAAA,IACrF,UAAA,EAAY,IAAA;AAAA,IACZ,QAAA,EAAU,IAAA;AAAA,IACV,UAAA,EAAY,IAAA,CAAK,SAAA,IAAa,IAAA,IAAQ,MAAA;AAAA,IACtC,QAAA,EAAU,OAAA;AAAA,IACV,YAAA,EAAc,OAAA;AAAA,IACd,aAAA,EAAe,OAAA;AAAA,IACf,iBAAA,EAAmB;AAAA,GACrB;AACF;AAOA,SAAS,aAAa,GAAA,EAAsD;AAC1E,EAAA,OAAO,aAAA,CAAc,OAAO,GAAA,CAAI,QAAA,IAAY,IAAI,SAAA,IAAa,EAAE,CAAC,CAAA,GAAI,aAAA,GAAgB,MAAA;AACtF;AAeO,SAAS,YAAY,MAAA,EAAiD;AAC3E,EAAA,IAAI,MAAA,IAAU,IAAA,IAAQ,MAAA,KAAW,EAAA,EAAI,OAAO,IAAA;AAC5C,EAAA,IAAI,OAAO,WAAW,QAAA,IAAY,CAAC,MAAM,OAAA,CAAQ,MAAM,GAAG,OAAO,MAAA;AACjE,EAAA,IAAI,MAAA,GAAkB,IAAA;AACtB,EAAA,IAAI;AACF,IAAA,MAAA,GAAS,IAAA,CAAK,KAAA,CAAM,MAAA,CAAO,MAAM,CAAC,CAAA;AAAA,EACpC,CAAA,CAAA,MAAQ;AACN,IAAA,MAAA,GAAS,IAAA;AAAA,EACX;AACA,EAAA,IAAI,CAAC,UAAU,OAAO,MAAA,KAAW,YAAY,KAAA,CAAM,OAAA,CAAQ,MAAM,CAAA,EAAG;AAClE,IAAA,MAAM,IAAI,KAAA,CAAM,CAAA,gBAAA,EAAmB,KAAK,SAAA,CAAU,MAAM,CAAC,CAAA,CAAE,CAAA;AAAA,EAC7D;AACA,EAAA,OAAO,MAAA;AACT;AAgBA,eAAe,QAAA,CACb,QACA,SAAA,EACA,UAAA,EACA,QACA,KAAA,EACA,MAAA,EACA,SACA,SAAA,EACqB;AACrB,EAAA,MAAM,IAAA,GAAO,MAAM,MAAA,CAAO,aAAA,CAAc;AAAA;AAAA;AAAA,IAGtC,SAAA,EAAW,aAAa,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,SAAS,CAAC,CAAA,GAAI,IAAA;AAAA,IACzD,UAAA;AAAA,IACA,MAAA;AAAA,IACA,KAAA,EAAO,KAAK,GAAA,CAAI,CAAA,EAAG,KAAK,GAAA,CAAI,KAAA,EAAO,cAAc,CAAC,CAAA;AAAA,IAClD;AAAA,GACD,CAAA;AACD,EAAA,IAAI,aAAwB,IAAA,CAAK,SAAA,IAAuD,EAAC,EAAG,GAAA,CAAI,CAAC,GAAA,MAAS;AAAA,IACxG,EAAA,EAAI,MAAA,CAAO,GAAA,CAAI,EAAE,CAAA;AAAA,IACjB,MAAM,GAAA,CAAI,IAAA;AAAA,IACV,SAAA,EAAW,GAAA,CAAI,QAAA,IAAY,GAAA,CAAI,SAAA;AAAA,IAC/B,IAAA,EAAM,aAAa,GAAG;AAAA,GACxB,CAAE,CAAA;AACF,EAAA,IAAI,OAAA,CAAQ,eAAe,IAAA,EAAM;AAC/B,IAAA,MAAM,OAAA,GAAU,IAAI,GAAA,CAAI,OAAA,CAAQ,WAAW,CAAA;AAC3C,IAAA,SAAA,GAAY,SAAA,CAAU,OAAO,CAAC,CAAA,KAAM,QAAQ,GAAA,CAAI,CAAA,CAAE,EAAE,CAAC,CAAA;AAAA,EACvD;AACA,EAAA,MAAM,GAAA,GAAkB;AAAA,IACtB,SAAA;AAAA,IACA,OAAO,SAAA,CAAU,MAAA;AAAA,IACjB,QAAA,EAAU,OAAA,CAAQ,IAAA,CAAK,OAAA,IAAW,KAAK,QAAQ;AAAA,GACjD;AACA,EAAA,MAAM,IAAA,GAAO,IAAA,CAAK,UAAA,IAAc,IAAA,CAAK,WAAA;AACrC,EAAA,IAAI,QAAQ,IAAA,EAAM;AAChB,IAAA,GAAA,CAAI,WAAA,GAAc,IAAA,CAAK,SAAA,CAAU,IAAI,CAAA;AACrC,IAAA,GAAA,CAAI,SAAA,GAAY,EAAE,MAAA,EAAQ,MAAA,EAAQ,IAAI,WAAA,EAAY;AAAA,EACpD;AACA,EAAA,OAAO,GAAA;AACT;AAiCA,eAAsB,iBAAA,CACpB,QACA,IAAA,EACkC;AAClC,EAAA,MAAM,SAAS,IAAA,CAAK,MAAA;AACpB,EAAA,MAAM,aAAa,IAAA,CAAK,UAAA;AACxB,EAAA,MAAM,OAAA,GAAU,YAAA,CAAa,IAAA,CAAK,KAAK,CAAA;AAGvC,EAAA,MAAM,uBAAuB,IAAA,CAAK,YAAA;AAClC,EAAA,MAAM,SAAA,GAAY,eAAA,CAAgB,IAAA,CAAK,UAAA,EAAY,QAAQ,SAAS,CAAA;AACpE,EAAA,MAAM,WAAA,GAAc,eAAA,CAAgB,IAAA,CAAK,YAAA,EAAc,QAAQ,WAAW,CAAA;AAC1E,EAAA,MAAM,OAAO,MAAA,CAAO,IAAA,CAAK,KAAA,IAAS,EAAE,EAAE,IAAA,EAAK;AAC3C,EAAA,MAAM,SAAA,GAAY,0BAA0B,MAAA,EAAQ;AAAA,IAClD,SAAS,IAAA,CAAK,OAAA;AAAA,IACd,WAAW,IAAA,CAAK;AAAA,GACjB,CAAA;AAED,EAAA,IAAI,CAAE,iBAAA,CAAwC,QAAA,CAAS,MAAM,CAAA,EAAG;AAC9D,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,KAAA;AAAA,MACT,OAAO,CAAA,gBAAA,EAAmB,MAAM,wBAAmB,iBAAA,CAAkB,IAAA,CAAK,IAAI,CAAC,CAAA;AAAA,KACjF;AAAA,EACF;AACA,EAAA,IAAI,sBAAsB,MAAA,IAAU,CAAC,mBAAA,CAAoB,QAAA,CAAS,MAAM,CAAA,EAAG;AACzE,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,KAAA;AAAA,MACT,OACE,CAAA,+BAAA,EAAkC,MAAM,mBAAc,mBAAA,CAAoB,IAAA,CAAK,IAAI,CAAC,CAAA,sEAAA;AAAA,KAExF;AAAA,EACF;AACA,EAAA,IAAI,cAAc,QAAA,CAAS,MAAM,KAAK,CAAC,SAAA,CAAU,MAA8C,CAAA,EAAG;AAChG,IAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,IAAI,KAAA,EAAM;AAAA,EAC5C;AAEA,EAAA,IAAI,MAAA,KAAW,UAAA,IAAc,MAAA,KAAW,MAAA,EAAQ;AAC9C,IAAA,MAAM,OAAO,MAAM,QAAA;AAAA,MACjB,MAAA;AAAA,MACA,SAAA;AAAA,MACA,UAAA;AAAA,MACA,MAAA;AAAA,MACA,KAAK,KAAA,IAAS,EAAA;AAAA,MACd,WAAA,CAAY,KAAK,MAAM,CAAA;AAAA,MACvB,OAAA;AAAA,MACA,IAAA,CAAK;AAAA,KACP;AACA,IAAA,IAAI,WAAW,MAAA,EAAQ,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAG,IAAA,EAAK;AACvD,IAAA,MAAM,YAAA,GAAe,IAAA,CAAK,SAAA,CAAU,MAAA,CAAO,CAAC,CAAA,KAAM,CAAA,CAAE,IAAA,KAAS,aAAa,CAAA,CAAE,GAAA,CAAI,CAAC,CAAA,KAAM,EAAE,IAAI,CAAA;AAC7F,IAAA,MAAM,QAAA,GAAW,EAAE,MAAA,EAAQ,QAAA,EAAU,OAAO,gCAAA,EAAiC;AAC7E,IAAA,MAAM,aAAsC,EAAC;AAC7C,IAAA,IAAI,YAAA,CAAa,MAAA,IAAU,SAAA,CAAU,OAAA,EAAS;AAC5C,MAAA,UAAA,CAAW,iCAAiC,CAAA,GAAI;AAAA,QAC9C,MAAA,EAAQ,SAAA;AAAA,QACR,KAAA,EAAO;AAAA,OACT;AAAA,IACF;AACA,IAAA,UAAA,CAAW,wBAAwB,CAAA,GAAI,QAAA;AACvC,IAAA,IAAI,UAAU,aAAA,EAAe;AAC3B,MAAA,UAAA,CAAW,wBAAwB,CAAA,GAAI;AAAA,QACrC,MAAA,EAAQ,eAAA;AAAA,QACR,MAAA,EAAQ;AAAA,OACV;AAAA,IACF;AACA,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,IAAA;AAAA,MACT,GAAG,IAAA;AAAA,MACH,YAAA;AAAA,MACA,mBAAmB,MAAA,CAAO,WAAA;AAAA,QACvB,OAAO,IAAA,CAAK,WAAW,CAAA,CAAsC,GAAA,CAAI,CAAC,IAAA,KAAS;AAAA,UAC1E,IAAA;AAAA,UACA,SAAA,CAAU,IAAI,CAAA,GAAI,WAAA,CAAY,IAAI,CAAA,GAAI,CAAA,EAAG,WAAA,CAAY,IAAI,CAAC,CAAA,MAAA;AAAA,SAC3D;AAAA,OACH;AAAA,MACA,WAAA,EAAa;AAAA,KACf;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,QAAA,EAAU;AACvB,IAAA,IAAI,CAAC,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,sBAAA,EAAuB;AAClE,IAAA,MAAM,MAAA,GAAS,MAAM,MAAA,CAAO,MAAA,CAAO,IAAA,EAAM;AAAA,MACvC,SAAA;AAAA,MACA,WAAA;AAAA,MACA,UAAA;AAAA,MACA,IAAA,EAAM,KAAK,KAAA,IAAS,EAAA;AAAA;AAAA,MAEpB,IAAA,EAAM,KAAK,IAAA,IAAQ,IAAA;AAAA,MACnB,WAAW,IAAA,CAAK;AAAA,KACjB,CAAA;AACD,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,IAAA;AAAA,MACT,OAAO,MAAA,CAAO,IAAA,IAAQ,EAAC,EAAG,GAAA,CAAI,CAAC,GAAA,KAAQ;AACrC,QAAA,MAAM,CAAA,GAAI,GAAA;AACV,QAAA,OAAO;AAAA,UACL,WAAA,EAAa,CAAA,CAAE,WAAA,IAAe,CAAA,CAAE,UAAA;AAAA,UAChC,aAAA,EAAe,CAAA,CAAE,aAAA,IAAiB,CAAA,CAAE,YAAA;AAAA,UACpC,UAAA,EAAY,CAAA,CAAE,UAAA,IAAc,CAAA,CAAE,SAAA;AAAA,UAC9B,OAAO,CAAA,CAAE,KAAA;AAAA,UACT,SAAA,EAAW,CAAA,CAAE,SAAA,IAAa,CAAA,CAAE;AAAA,SAC9B;AAAA,MACF,CAAC,CAAA;AAAA,MACD,OAAO,MAAA,CAAO;AAAA,KAChB;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,SAAA,EAAW;AACxB,IAAA,MAAM,aAAa,IAAA,CAAK,WAAA;AACxB,IAAA,IAAI,CAAC,UAAA,EAAY,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,6BAAA,EAA8B;AAI/E,IAAA,IAAI,CAAC,aAAA,CAAc,UAAA,EAAY,OAAO,CAAA,EAAG;AACvC,MAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,IACtE;AACA,IAAA,IAAI;AACF,MAAA,MAAM,QAAA,GAAW,MAAM,MAAA,CAAO,WAAA,CAAY,UAAA,EAAY;AAAA,QACpD,UAAA;AAAA,QACA,WAAW,IAAA,CAAK;AAAA,OACjB,CAAA;AACD,MAAA,MAAM,GAAA,GAAM,QAAA,EAAU,QAAA,IAAY,QAAA,EAAU,SAAA;AAC5C,MAAA,IAAI,OAAA,CAAQ,SAAA,IAAa,IAAA,IAAQ,CAAC,OAAA,CAAQ,UAAU,QAAA,CAAS,MAAA,CAAO,GAAG,CAAC,CAAA,EAAG;AACzE,QAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,MACtE;AACA,MAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,QAAA,EAAS;AAAA,IACnC,SAAS,GAAA,EAAK;AAGZ,MAAA,IAAI,GAAA,YAAe,uBAAuB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AACtF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,YAAA,EAAc;AAC3B,IAAA,IAAI,CAAC,SAAA,CAAU,UAAA,IAAc,CAAC,MAAA,CAAO,eAAA,EAAiB,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,UAAA,EAAW;AACrG,IAAA,IAAI,CAAC,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,0BAAA,EAA2B;AACtE,IAAA,MAAM,GAAA,GAAM,MAAM,MAAA,CAAO,eAAA,CAAgB,IAAA,EAAM;AAAA,MAC7C,SAAA;AAAA,MACA,UAAA;AAAA,MACA,KAAA,EAAO,IAAA,CAAK,GAAA,CAAI,CAAA,EAAG,IAAA,CAAK,IAAI,IAAA,CAAK,KAAA,IAAS,EAAA,EAAI,GAAG,CAAC,CAAA;AAAA,MAClD,WAAW,IAAA,CAAK;AAAA,KACjB,CAAA;AACD,IAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAI,GAAA,EAAgC;AAAA,EAC9D;AAEA,EAAA,IAAI,WAAW,SAAA,EAAW;AACxB,IAAA,IAAI,CAAC,UAAU,OAAA,EAAS,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,OAAA,EAAQ;AACpE,IAAA,IAAI,CAAC,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,wDAAA,EAAyD;AAKpG,IAAA,MAAM,MAAA,GAAS,IAAA,CAAK,OAAA,IAAW,MAAA,CAAO,OAAA;AACtC,IAAA,IAAI,CAAC,QAAQ,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AACzD,IAAA,IAAI;AACF,MAAA,MAAM,GAAA,GAAM,MAAM,MAAA,CAAO,IAAA,EAAM,EAAE,SAAA,EAAW,WAAA,EAAa,YAAY,CAAA;AACrE,MAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAI,GAAA,EAAgC;AAAA,IAC9D,SAAS,GAAA,EAAK;AAGZ,MAAA,IAAI,GAAA,YAAe,mBAAmB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AAClF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,YAAA,EAAc;AAC3B,IAAA,MAAM,aAAa,IAAA,CAAK,WAAA;AACxB,IAAA,IAAI,CAAC,UAAA,EAAY,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,gCAAA,EAAiC;AAClF,IAAA,IAAI,CAAC,aAAA,CAAc,UAAA,EAAY,OAAO,CAAA,EAAG;AACvC,MAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,IACtE;AACA,IAAA,IAAI;AACF,MAAA,MAAM,IAAA,GAAO,MAAM,MAAA,CAAO,SAAA,CAAW,UAAA,EAAY;AAAA,QAC/C,UAAA;AAAA,QACA,OAAO,IAAA,CAAK,KAAA;AAAA,QACZ,KAAK,IAAA,CAAK,GAAA;AAAA,QACV,WAAW,IAAA,CAAK;AAAA,OACjB,CAAA;AACD,MAAA,IAAI,OAAA,CAAQ,aAAa,IAAA,EAAM;AAG7B,QAAA,MAAM,MAAM,MAAM,MAAA,CAAO,YAAY,UAAA,EAAY,EAAE,YAAY,CAAA;AAC/D,QAAA,MAAM,GAAA,GAAM,GAAA,EAAK,QAAA,IAAY,GAAA,EAAK,SAAA;AAClC,QAAA,IAAI,CAAC,OAAA,CAAQ,SAAA,CAAU,SAAS,MAAA,CAAO,GAAG,CAAC,CAAA,EAAG;AAC5C,UAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,QACtE;AAAA,MACF;AACA,MAAA,MAAM,GAAA,GAA+B,EAAE,OAAA,EAAS,IAAA,EAAM,GAAG,IAAA,EAAK;AAC9D,MAAA,IAAI,KAAK,QAAA,EAAU;AACjB,QAAA,GAAA,CAAI,SAAA,GAAY;AAAA,UACd,MAAA,EAAQ,YAAA;AAAA,UACR,WAAA,EAAa,OAAO,UAAU,CAAA;AAAA,UAC9B,OAAO,IAAA,CAAK;AAAA,SACd;AAAA,MACF;AACA,MAAA,OAAO,GAAA;AAAA,IACT,SAAS,GAAA,EAAK;AACZ,MAAA,IAAI,GAAA,YAAe,uBAAuB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AACtF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,UAAA,EAAY;AACzB,IAAA,IAAI,CAAC,aAAa,MAAA,EAAQ,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,6BAAA,EAA8B;AACxF,IAAA,MAAM,IAAA,GAAO,MAAM,MAAA,CAAO,YAAA,CAAc,WAAA,EAAa,EAAE,UAAA,EAAY,SAAA,EAAW,IAAA,CAAK,SAAA,EAAW,CAAA;AAI9F,IAAA,MAAM,MAAA,GAAS,KAAK,GAAA,CAAI,GAAA,EAAM,KAAK,KAAA,CAAM,IAAA,CAAK,SAAA,IAAa,sBAAsB,CAAC,CAAA;AAClF,IAAA,MAAM,OAAuC,EAAC;AAC9C,IAAA,IAAI,IAAA,GAAO,CAAA;AACX,IAAA,KAAA,MAAW,OAAO,IAAA,EAAM;AACtB,MAAA,MAAM,IAAA,GAAO,MAAA,CAAO,GAAA,CAAI,IAAA,IAAQ,EAAE,CAAA,CAAE,MAAA;AACpC,MAAA,IAAI,IAAA,CAAK,MAAA,IAAU,IAAA,GAAO,IAAA,GAAO,MAAA,EAAQ;AACzC,MAAA,IAAA,CAAK,KAAK,GAAG,CAAA;AACb,MAAA,IAAA,IAAQ,IAAA;AAAA,IACV;AACA,IAAA,MAAM,QAAA,GAAW,IAAI,GAAA,CAAI,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,KAAM,MAAA,CAAO,CAAA,CAAE,EAAE,CAAC,CAAC,CAAA;AACtD,IAAA,MAAM,SAAA,GAAA,CAAa,IAAA,CAAK,YAAA,IAAgB,IAAI,MAAA,CAAO,CAAC,CAAA,KAAM,CAAC,QAAA,CAAS,GAAA,CAAI,MAAA,CAAO,CAAC,CAAC,CAAC,CAAA;AAClF,IAAA,MAAM,GAAA,GAA+B,EAAE,OAAA,EAAS,IAAA,EAAM,WAAW,IAAA,EAAM,KAAA,EAAO,KAAK,MAAA,EAAO;AAC1F,IAAA,IAAI,UAAU,MAAA,EAAQ;AACpB,MAAA,GAAA,CAAI,sBAAA,GAAyB,SAAA;AAC7B,MAAA,GAAA,CAAI,SAAA,GAAY,EAAE,MAAA,EAAQ,UAAA,EAAY,cAAc,SAAA,EAAU;AAAA,IAChE;AACA,IAAA,OAAO,GAAA;AAAA,EACT;AAEA,EAAA,IAAI,WAAW,YAAA,EAAc;AAC3B,IAAA,IAAI,CAAC,UAAU,UAAA,EAAY,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,UAAA,EAAW;AAC1E,IAAA,IAAI,CAAC,IAAA,EAAM;AACT,MAAA,OAAO;AAAA,QACL,OAAA,EAAS,KAAA;AAAA,QACT,KAAA,EAAO;AAAA,OACT;AAAA,IACF;AAGA,IAAA,MAAM,MAAA,GAAS,IAAA,CAAK,UAAA,IAAc,MAAA,CAAO,SAAA;AACzC,IAAA,IAAI,CAAC,QAAQ,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,UAAA,EAAW;AAC5D,IAAA,IAAI;AACF,MAAA,MAAM,GAAA,GAAM,MAAM,MAAA,CAAO,IAAA,EAAM;AAAA,QAC7B,SAAA;AAAA,QACA,WAAA;AAAA,QACA,UAAA;AAAA,QACA,OAAO,IAAA,CAAK;AAAA,OACb,CAAA;AACD,MAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAI,GAAA,EAAgC;AAAA,IAC9D,SAAS,GAAA,EAAK;AACZ,MAAA,IAAI,GAAA,YAAe,mBAAmB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AAClF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,OAAO,gBAAgB,MAAA,EAAQ,IAAA,EAAM,EAAE,SAAA,EAAW,UAAA,EAAY,MAAM,CAAA;AACtE;AAEA,eAAe,eAAA,CACb,MAAA,EACA,IAAA,EACA,GAAA,EACkC;AAGlC,EAAA,MAAM,MAAM,MAAM,OAAA,CAAA,OAAA,EAAA,CAAA,IAAA,CAAA,OAAA,eAAA,EAAA,EAAA,kBAAA,CAAA,CAAA;AAClB,EAAA,MAAM,IAAA,GAAQ,MAAM,MAAA,CAAO,UAAA,IAAa;AAExC,EAAA,MAAM,UAAA,GAAc,IAAI,UAAA,IAAc,IAAA;AACtC,EAAA,MAAM,MAAA,GAAS,EAAE,SAAA,EAAW,GAAA,CAAI,WAAW,UAAA,EAAW;AAEtD,EAAA,IAAI,IAAA,CAAK,WAAW,mBAAA,EAAqB;AACvC,IAAA,IAAI,CAAC,IAAI,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,iCAAA,EAAkC;AACjF,IAAA,MAAMA,IAAAA,GAAM,MAAM,GAAA,CAAI,gBAAA,CAAiB,IAAA,EAAM;AAAA,MAC3C,UAAU,MAAA,CAAO,QAAA;AAAA,MACjB,OAAO,GAAA,CAAI,IAAA;AAAA,MACX,OAAO,IAAA,CAAK,KAAA;AAAA,MACZ,WAAW,GAAA,CAAI,SAAA;AAAA,MACf;AAAA,KACD,CAAA;AACD,IAAA,IAAI,CAACA,KAAI,SAAA,EAAW,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAOA,IAAAA,CAAI,KAAA,EAAgB;AACxE,IAAA,OAAO,EAAE,SAAS,IAAA,EAAM,WAAA,EAAaA,KAAI,WAAA,EAAa,KAAA,EAAOA,KAAI,KAAA,EAAM;AAAA,EACzE;AAEA,EAAA,IAAI,IAAA,CAAK,WAAW,cAAA,EAAgB;AAClC,IAAA,IAAI,EAAE,IAAA,CAAK,QAAA,IAAY,IAAA,CAAK,KAAA,IAAS,KAAK,WAAA,CAAA,EAAc;AACtD,MAAA,OAAO;AAAA,QACL,OAAA,EAAS,KAAA;AAAA,QACT,KAAA,EAAO;AAAA,OACT;AAAA,IACF;AACA,IAAA,MAAMA,IAAAA,GAAM,MAAM,GAAA,CAAI,WAAA,CAAY,IAAA,EAAM;AAAA,MACtC,GAAG,MAAA;AAAA,MACH,UAAU,IAAA,CAAK,QAAA;AAAA,MACf,OAAO,IAAA,CAAK,KAAA;AAAA,MACZ,YAAY,IAAA,CAAK,WAAA;AAAA,MACjB,OAAO,IAAA,CAAK;AAAA,KACb,CAAA;AACD,IAAA,OAAO,EAAE,SAAS,IAAA,EAAM,aAAA,EAAeA,KAAI,aAAA,EAAe,KAAA,EAAOA,KAAI,KAAA,EAAM;AAAA,EAC7E;AAEA,EAAA,IAAI,CAAC,KAAK,MAAA,EAAQ;AAChB,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,KAAA;AAAA,MACT,KAAA,EAAO,CAAA,EAAG,IAAA,CAAK,MAAM,CAAA,mDAAA;AAAA,KACvB;AAAA,EACF;AACA,EAAA,IAAI,IAAA,CAAK,WAAW,eAAA,EAAiB;AACnC,IAAA,MAAMA,IAAAA,GAAM,MAAM,GAAA,CAAI,YAAA,CAAa,MAAM,EAAE,GAAG,MAAA,EAAQ,MAAA,EAAQ,IAAA,CAAK,MAAA,EAAQ,KAAA,EAAO,IAAA,CAAK,OAAO,CAAA;AAC9F,IAAA,IAAI,CAACA,KAAI,KAAA,EAAO,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAOA,IAAAA,CAAI,KAAA,EAAgB;AACpE,IAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,MAAA,EAAQA,IAAAA,CAAI,MAAA,EAAQ,SAAA,EAAWA,IAAAA,CAAI,SAAA,EAAW,KAAA,EAAOA,IAAAA,CAAI,KAAA,EAAM;AAAA,EACzF;AACA,EAAA,MAAM,GAAA,GAAM,MAAM,GAAA,CAAI,QAAA,CAAS,IAAA,EAAM;AAAA,IACnC,GAAG,MAAA;AAAA,IACH,QAAQ,IAAA,CAAK,MAAA;AAAA,IACb,OAAO,IAAA,CAAK,KAAA;AAAA,IACZ,UAAU,IAAA,CAAK,QAAA;AAAA,IACf,OAAO,IAAA,CAAK;AAAA,GACb,CAAA;AACD,EAAA,IAAI,CAAC,IAAI,KAAA,EAAO,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,KAAA,EAAgB;AACpE,EAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,MAAA,EAAQ,GAAA,CAAI,MAAA,EAAQ,KAAA,EAAO,GAAA,CAAI,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,KAAA,EAAM;AACjF;ACv0BO,IAAM,YAAA,GAAN,cAA2B,KAAA,CAAM;AAAA,EACtC,MAAA;AAAA,EACA,MAAA;AAAA,EACA,WAAA,CAAY,QAAgB,MAAA,EAAiB;AAC3C,IAAA,KAAA,CAAM,OAAO,MAAA,KAAW,QAAA,GAAW,SAAS,IAAA,CAAK,SAAA,CAAU,MAAM,CAAC,CAAA;AAClE,IAAA,IAAA,CAAK,IAAA,GAAO,cAAA;AACZ,IAAA,IAAA,CAAK,MAAA,GAAS,MAAA;AACd,IAAA,IAAA,CAAK,MAAA,GAAS,MAAA;AAAA,EAChB;AACF,CAAA;AAEuCC,MACpC,MAAA,CAAO;AAAA,EACN,IAAA,EAAMA,MAAE,MAAA,EAAO;AAAA,EACf,IAAA,EAAMA,MAAE,MAAA,EAAO;AAAA,EACf,WAAWA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC1C,aAAaA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC5C,aAAaA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC5C,SAAA,EAAWA,MAAE,MAAA,CAAOA,KAAA,CAAE,SAAS,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EACrD,GAAA,EAAKA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EAC7C,IAAA,EAAMA,KAAA,CAAE,IAAA,CAAK,CAAC,QAAA,EAAU,OAAO,CAAC,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EACtD,oBAAoBA,KAAA,CAAE,OAAA,GAAU,QAAA,EAAS,CAAE,QAAQ,KAAK,CAAA;AAAA,EACxD,OAAOA,KAAA,CAAE,OAAA,GAAU,QAAA,EAAS,CAAE,QAAQ,KAAK;AAC7C,CAAC,EACA,KAAA;AAEgCA,MAChC,MAAA,CAAO;AAAA,EACN,GAAA,EAAKA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EAC7C,MAAMA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EACrC,aAAaA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC5C,SAAA,EAAWA,MAAE,MAAA,CAAOA,KAAA,CAAE,SAAS,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA;AAC9C,CAAC,EACA,MAAA;AAEgCA,MAChC,MAAA,CAAO;AAAA,EACN,KAAA,EAAOA,MAAE,MAAA,EAAO;AAAA,EAChB,UAAA,EAAYA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA;AAAA;AAAA;AAAA,EAIpD,YAAA,EAAcA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EACtD,KAAA,EAAOA,MAAE,MAAA,EAAO,CAAE,KAAI,CAAE,QAAA,EAAS,CAAE,OAAA,CAAQ,EAAE,CAAA;AAAA,EAC7C,IAAA,EAAMA,KAAA,CAAE,IAAA,CAAK,CAAC,QAAA,EAAU,OAAO,CAAC,CAAA,CAAE,QAAA,EAAS,CAAE,OAAA,CAAQ,QAAQ,CAAA;AAAA,EAC7D,kBAAA,EAAoBA,MAAE,MAAA,EAAO,CAAE,KAAI,CAAE,QAAA,GAAW,QAAA;AAClD,CAAC,EACA,KAAA;AAgBH,IAAM,OAAA,GAAU,iEAAA;AAET,SAAS,iBAAiB,UAAA,EAA0B;AACzD,EAAA,IAAI,CAAC,OAAA,CAAQ,IAAA,CAAK,MAAA,CAAO,UAAU,CAAC,CAAA,EAAG;AACrC,IAAA,MAAM,IAAI,aAAa,GAAA,EAAK,CAAA,qBAAA,EAAwB,KAAK,SAAA,CAAU,UAAU,CAAC,CAAA,CAAE,CAAA;AAAA,EAClF;AACF;AAsBO,SAAS,wBAAA,CAAyB,OAAgB,IAAA,EAA+C;AAKtG,EAAA,IAAI,KAAA,KAAU,SAAS,OAAO,OAAA;AAC9B,EAAA,IAAI,SAAS,IAAA,EAAM;AACjB,IAAA,MAAM,IAAI,YAAA;AAAA,MACR,GAAA;AAAA,MACA,CAAA,EAAG,KAAK,OAAO,CAAA,qWAAA;AAAA,KAMjB;AAAA,EACF;AAIA,EAAA,IAAI,CAAC,KAAA,CAAM,OAAA,CAAQ,KAAK,CAAA,IAAK,KAAA,CAAM,IAAA,CAAK,CAAC,CAAA,KAAM,OAAO,CAAA,KAAM,QAAQ,CAAA,EAAG;AACrE,IAAA,MAAM,IAAI,YAAA;AAAA,MACR,GAAA;AAAA,MACA,CAAA,EAAG,IAAA,CAAK,OAAO,CAAA,2HAAA,EACsD,OAAO,KAAK,CAAA,CAAA;AAAA,KACnF;AAAA,EACF;AACA,EAAA,OAAO,KAAA;AACT;AChHO,IAAM,UAAA,GACX,wLAAA;AAIF,eAAsB,oBAAoB,UAAA,EAAuD;AAC/F,EAAA,MAAM,MAAA,GAAS,MAAM,OAAA,CAAQ,OAAA,CAAQ,YAAY,CAAA;AAQjD,EAAA,IAAI;AACF,IAAA,OAAO,wBAAA,CAAyB,MAAA,EAAQ,EAAE,OAAA,EAAS,mCAAmC,CAAA;AAAA,EACxF,SAAS,GAAA,EAAK;AACZ,IAAA,IAAI,GAAA,YAAe,YAAA,EAAc,MAAM,IAAI,KAAA,CAAM,IAAI,OAAA,EAAS,EAAE,KAAA,EAAO,GAAA,EAAK,CAAA;AAC5E,IAAA,MAAM,GAAA;AAAA,EACR;AACF;AAWO,SAAS,mBAAA,CACd,GAAA,EACA,MAAA,EAQA,UAAA,EAMA,gBAAwC,IAAA,EAClC;AACN,EAAA,GAAA,CAAI,IAAA;AAAA,IACF,cAAA;AAAA,IACA,iPAAA,GAIE,UAAA;AAAA;AAAA,IAEF,EAAE,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,EAAG,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,EAAE;AAAA,IACxD,UAAU,GAAA,KAAiB;AACzB,MAAA,MAAM,IAAA,GAAQ,GAAA,CAAI,CAAC,CAAA,IAAK,EAAC;AACzB,MAAA,MAAM,gBAAA,GAAmB,MAAM,mBAAA,CAAoB,UAAU,CAAA;AAC7D,MAAA,MAAM,KAAA,GAAQ,MAAM,MAAA,CAAO,WAAA,CAAY,OAAO,IAAA,CAAK,KAAA,IAAS,EAAE,CAAA,EAAG;AAAA,QAC/D,UAAA,EAAY,gBAAA;AAAA,QACZ,KAAA,EAAO,KAAK,KAAA,IAAS;AAAA,OACtB,CAAA;AACD,MAAA,OAAO,EAAE,KAAA,EAAM;AAAA,IACjB;AAAA,GACF;AAEA,EAAA,GAAA,CAAI,IAAA;AAAA,IACF,cAAA;AAAA,IACA,iTAAA,GAKE,UAAA;AAAA,IACF,EAAE,IAAA,EAAMA,KAAAA,CAAE,MAAA,EAAO,EAAG,IAAA,EAAMA,KAAAA,CAAE,MAAA,CAAOA,KAAAA,CAAE,OAAA,EAAS,CAAA,CAAE,UAAS,EAAE;AAAA,IAC3D,UAAU,GAAA,KAAiB;AACzB,MAAA,MAAM,IAAA,GAAQ,GAAA,CAAI,CAAC,CAAA,IAAK,EAAC;AACzB,MAAA,MAAM,gBAAA,GAAmB,MAAM,mBAAA,CAAoB,UAAU,CAAA;AAC7D,MAAA,MAAM,QAAQ,aAAA,GAAgB,MAAM,QAAQ,OAAA,CAAQ,aAAA,EAAe,CAAA,GAAI,IAAA;AACvE,MAAA,OAAO,MAAA,CAAO,YAAY,MAAA,CAAO,IAAA,CAAK,IAAI,CAAA,EAAG,IAAA,CAAK,QAAQ,IAAA,EAAM;AAAA,QAC9D,UAAA,EAAY,gBAAA;AAAA,QACZ,MAAA,EAAQ,KAAA;AAAA,QACR,aAAA,EAAe;AAAA,OAChB,CAAA;AAAA,IACH;AAAA,GACF;AACF;;;AClGO,IAAM,WAAA,GAAc,OAAA;;;AXqC3B,SAAS,UAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,MAAA,CAAO,gBAAA,CAAiB,IAAI,CAAA,EAAG,eAAe,EAAE,CAAA;AACzD;AAEA,SAAS,YAAY,IAAA,EAAmE;AACtF,EAAA,OAAO,EAAE,OAAA,EAAS,CAAC,EAAE,IAAA,EAAM,MAAA,EAAQ,IAAA,EAAM,IAAA,CAAK,SAAA,CAAU,IAAI,CAAA,EAAG,CAAA,EAAE;AACnE;AAEA,eAAsB,YAAA,CACpB,QACA,IAAA,EAkCA;AACA,EAAA,IAAI,CAAC,MAAM,UAAA,EAAY;AACrB,IAAA,MAAM,IAAI,UAAU,kCAAkC,CAAA;AAAA,EACxD;AACA,EAAA,IAAI,IAAA,CAAK,KAAA,KAAU,MAAA,IAAa,IAAA,CAAK,UAAU,IAAA,EAAM;AACnD,IAAA,MAAM,IAAI,SAAA;AAAA,MACR;AAAA,KAEF;AAAA,EACF;AAEA,EAAA,IAAI,SAAA;AAOJ,EAAA,IAAI,6BAAA;AAKJ,EAAA,IAAI;AACF,IAAA,MAAM,YAAY,MAAM,YAAA;AAAA,MACtB,yCAAA;AAAA,MACA,KAAA;AAAA,MACA;AAAA,KACF;AACA,IAAA,MAAM,OAAA,GAAU,MAAM,YAAA,CAEnB,oDAAA,EAAsD,OAAO,YAAY,CAAA;AAC5E,IAAA,SAAA,GAAY,SAAA,CAAU,SAAA;AACtB,IAAA,6BAAA,GAAgC,OAAA,CAAQ,6BAAA;AAAA,EAC1C,SAAS,GAAA,EAAK;AACZ,IAAA,IAAI,GAAA,YAAe,mBAAmB,MAAM,GAAA;AAC5C,IAAA,MAAM,IAAI,iBAAA,CAAkB,KAAA,EAAO,2BAAA,EAA6B,YAAY,CAAA;AAAA,EAC9E;AAEA,EAAA,MAAM,GAAA,GAAM,IAAI,SAAA,CAAU,EAAE,MAAM,gBAAA,EAAkB,OAAA,EAAS,aAAa,CAAA;AAE1E,EAAA,MAAM,IAAA,GAAO,CACX,IAAA,EACA,WAAA,EACA,QACAC,QAAAA,KACG;AACH,IAAA,GAAA,CAAI,IAAA;AAAA,MAAK,IAAA;AAAA,MAAM,WAAA;AAAA,MAAa,MAAA;AAAA,MAAQ,OAAO,IAAA,KACzC,WAAA,CAAY,MAAMA,QAAAA,CAAQ,IAAI,CAAC;AAAA,KACjC;AAAA,EACF,CAAA;AAEA,EAAA,IAAA;AAAA,IACE,uBAAA;AAAA,IACA,CAAA,EAAG,0BAA0B,CAAA,EAAG,UAAU,CAAA,CAAA;AAAA,IAC1C;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,MAQE,MAAA,EAAQD,MAAE,IAAA,CAAK,iBAAiB,EAAE,QAAA,CAAS,SAAA,CAAU,QAAQ,CAAC,CAAA;AAAA,MAC9D,KAAA,EAAOA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MACxD,WAAA,EAAaA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,aAAa,CAAC,CAAA;AAAA,MACpE,UAAA,EAAYA,KAAAA,CAAE,KAAA,CAAMA,KAAAA,CAAE,MAAA,EAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,YAAY,CAAC,CAAA;AAAA,MAC3E,YAAA,EAAcA,KAAAA,CAAE,KAAA,CAAMA,KAAAA,CAAE,MAAA,EAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,cAAc,CAAC,CAAA;AAAA,MAC/E,MAAA,EAAQA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,QAAQ,CAAC,CAAA;AAAA,MAC1D,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,QAAA,EAAUA,KAAAA,CAAE,IAAA,CAAK,4BAA4B,CAAA,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,UAAU,CAAC,CAAA;AAAA,MACxF,KAAA,EAAOA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MACxD,WAAA,EAAaA,KAAAA,CAAE,IAAA,CAAK,YAAY,CAAA,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,aAAa,CAAC,CAAA;AAAA,MAC9E,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,IAAA,EAAMA,KAAAA,CAAE,IAAA,CAAK,CAAC,QAAA,EAAU,OAAO,CAAC,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,MAAM,CAAC,CAAA;AAAA,MACvE,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,MAAA,EAAQA,MACL,KAAA,CAAM,CAACA,MAAE,MAAA,EAAO,EAAGA,MAAE,MAAA,CAAOA,KAAAA,CAAE,SAAS,CAAC,CAAC,CAAA,CACzC,QAAA,GACA,QAAA,CAAS,SAAA,CAAU,QAAQ,CAAC,CAAA;AAAA,MAC/B,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,GAAA,EAAKA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,KAAK,CAAC,CAAA;AAAA,MAC1D,SAAA,EAAWA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,WAAW,CAAC;AAAA,KACxE;AAAA,IACA,OAAO,IAAA,KAAS;AAGd,MAAA,MAAM,cAAc,IAAA,CAAK,YAAA;AACzB,MAAA,MAAM,aAAa,IAAA,CAAK,WAAA;AACxB,MAAA,KAAA,MAAW,GAAA,IAAO,CAAC,GAAI,WAAA,IAAe,EAAC,EAAI,GAAI,UAAA,GAAa,CAAC,UAAU,CAAA,GAAI,EAAG,CAAA,EAAG;AAC/E,QAAA,gBAAA,CAAiB,GAAG,CAAA;AAAA,MACtB;AAIA,MAAA,MAAM,gBAAA,GAAmB,MAAM,mBAAA,CAAoB,IAAA,CAAK,UAAU,CAAA;AAClE,MAAA,MAAM,OAAA,GAAU,OAAO,IAAA,CAAK,KAAA,KAAU,aAAa,MAAM,IAAA,CAAK,KAAA,EAAM,GAAI,IAAA,CAAK,KAAA;AAC7E,MAAA,MAAM,MAAA,GAAS,OAAO,IAAA,CAAK,SAAA,KAAc,aAAa,MAAM,IAAA,CAAK,SAAA,EAAU,GAAI,IAAA,CAAK,SAAA;AACpF,MAAA,OAAO,kBAAkB,MAAA,EAAiB;AAAA,QACxC,GAAI,IAAA;AAAA,QACJ,MAAA,EAAQ,MAAA,CAAO,IAAA,CAAK,MAAM,CAAA;AAAA,QAC1B,UAAA,EAAY,gBAAA;AAAA,QACZ,KAAA,EAAO,OAAA;AAAA,QACP,SAAA,EAAW,MAAA;AAAA,QACX,OAAA,EAAS,KAAK,OAAA,IAAW,IAAA;AAAA,QACzB,UAAA,EAAY,KAAK,SAAA,IAAa;AAAA,OAC/B,CAAA;AAAA,IACH;AAAA,GACF;AAEA,EAAA,mBAAA;AAAA,IACE;AAAA,MACE,IAAA,EAAM,CAAC,IAAA,EAAM,WAAA,EAAa,QAAQC,QAAAA,KAAY;AAC5C,QAAA,IAAA,CAAK,MAAM,WAAA,EAAa,MAAA,EAAQ,OAAO,IAAA,KAASA,QAAAA,CAAQ,IAAa,CAAC,CAAA;AAAA,MACxE;AAAA,KACF;AAAA,IACA,MAAA;AAAA,IACA,IAAA,CAAK,UAAA;AAAA,IACL,KAAK,aAAA,IAAiB;AAAA,GACxB;AAEA,EAAA,MAAM,OAAA,IAAW,OACf,GAAA,EACA,GAAA,KACG;AACH,IAAA,MAAM,YAAY,IAAI,6BAAA,CAA8B,EAAE,kBAAA,EAAoB,QAAW,CAAA;AACrF,IAAA,MAAM,GAAA,CAAI,QAAQ,SAAS,CAAA;AAC3B,IAAA,MAAM,SAAA,CAAU,aAAA,CAAc,GAAA,EAAK,GAAG,CAAA;AAAA,EACxC,CAAA,CAAA;AAIA,EAAA,OAAA,CAAQ,GAAA,GAAM,GAAA;AACd,EAAA,OAAO,OAAA;AACT","file":"mcp.cjs","sourcesContent":["/** Action-surface errors — user-facing / data-dependent failures (not found, ACL, bad input).\n * Config/programming errors stay as TypeError / RangeError / Error.\n */\nexport class EngineActionError extends Error {\n constructor(message: string) {\n super(message);\n this.name = \"EngineActionError\";\n }\n}\n\n/** Missing vs forbidden document: identical message so callers cannot distinguish. */\nexport class DocumentNotFoundError extends Error {\n constructor(documentId: string) {\n super(`document not found: ${documentId}`);\n this.name = \"DocumentNotFoundError\";\n }\n}\n\nexport class GraphLegUnavailable extends Error {\n constructor(message = \"graph ranked list was not supplied\") {\n super(message);\n this.name = \"GraphLegUnavailable\";\n }\n}\n\nexport class ApprovalNotPending extends Error {\n constructor(message = \"approval is not pending\") {\n super(message);\n this.name = \"ApprovalNotPending\";\n }\n}\n\nexport class ApprovalExpired extends Error {\n constructor(message = \"approval has expired\") {\n super(message);\n this.name = \"ApprovalExpired\";\n }\n}\n\nexport class CodeExecutionError extends Error {\n constructor(message: string) {\n super(message);\n this.name = \"CodeExecutionError\";\n }\n}\n\nexport class CodeExecutionTimeout extends Error {\n constructor(message = \"code execution timed out\") {\n super(message);\n this.name = \"CodeExecutionTimeout\";\n }\n}\n\nexport class ExtraMissingError extends Error {\n constructor(extra: string, pkg: string, what: string) {\n super(`${what} requires the '${pkg}' package (optional extra: ${extra}): npm install ${pkg}`);\n this.name = \"ExtraMissingError\";\n }\n}\n","import { ExtraMissingError } from \"./errors.js\";\n\nexport async function tryImport<T = unknown>(specifier: string): Promise<T | null> {\n try {\n return (await import(specifier)) as T;\n } catch {\n return null;\n }\n}\n\nexport async function requireExtra<T = unknown>(specifier: string, extra: string, what: string): Promise<T> {\n try {\n return (await import(specifier)) as T;\n } catch (_err) {\n throw new ExtraMissingError(extra, specifier, what);\n }\n}\n","import { francAll } from \"franc\";\n\n// Unicode name first-word table — same logic as unicodedata.name().split()[0].\n// COMBINING marks use the second word, matching Python's _char_script.\n\nconst UNICODE_NAME_WORDS: string[] = [\n \"acute\",\n \"acute-grave-acute\",\n \"acute-macron\",\n \"adlam\",\n \"ahom\",\n \"alef\",\n \"almost\",\n \"anatolian\",\n \"angstrom\",\n \"annuity\",\n \"anticlockwise\",\n \"arabic\",\n \"armenian\",\n \"asterisk\",\n \"avestan\",\n \"balinese\",\n \"bamum\",\n \"bassa\",\n \"batak\",\n \"bengali\",\n \"bet\",\n \"bhaiksuki\",\n \"bindu\",\n \"black-letter\",\n \"bopomofo\",\n \"brahmi\",\n \"breve\",\n \"breve-macron\",\n \"bridge\",\n \"buginese\",\n \"buhid\",\n \"canadian\",\n \"candrabindu\",\n \"carian\",\n \"caron\",\n \"caucasian\",\n \"cedilla\",\n \"chakma\",\n \"cham\",\n \"cherokee\",\n \"chorasmian\",\n \"circumflex\",\n \"cjk\",\n \"clockwise\",\n \"comma\",\n \"conjoining\",\n \"coptic\",\n \"cuneiform\",\n \"cypriot\",\n \"cypro-minoan\",\n \"cyrillic\",\n \"dalet\",\n \"deletion\",\n \"deseret\",\n \"devanagari\",\n \"diaeresis\",\n \"diaeresis-ring\",\n \"dives\",\n \"dogra\",\n \"dot\",\n \"dotted\",\n \"double\",\n \"double-struck\",\n \"doubled\",\n \"down\",\n \"downwards\",\n \"duployan\",\n \"egyptian\",\n \"elbasan\",\n \"elymaic\",\n \"enclosing\",\n \"equals\",\n \"ethiopic\",\n \"euler\",\n \"feminine\",\n \"fermata\",\n \"four\",\n \"fullwidth\",\n \"georgian\",\n \"gimel\",\n \"glagolitic\",\n \"gothic\",\n \"grantha\",\n \"grapheme\",\n \"grave\",\n \"grave-acute-grave\",\n \"grave-macron\",\n \"greek\",\n \"gujarati\",\n \"gunjala\",\n \"gurmukhi\",\n \"halfwidth\",\n \"hangul\",\n \"hanifi\",\n \"hanunoo\",\n \"hatran\",\n \"hebrew\",\n \"hentaigana\",\n \"hiragana\",\n \"homothetic\",\n \"hook\",\n \"horn\",\n \"ideographic\",\n \"imperial\",\n \"infinity\",\n \"information\",\n \"inscriptional\",\n \"inverted\",\n \"is\",\n \"javanese\",\n \"kaithi\",\n \"kannada\",\n \"katakana\",\n \"katakana-hiragana\",\n \"kavyka\",\n \"kawi\",\n \"kayah\",\n \"kelvin\",\n \"kharoshthi\",\n \"khitan\",\n \"khmer\",\n \"khojki\",\n \"khudawadi\",\n \"lao\",\n \"latin\",\n \"left\",\n \"leftwards\",\n \"lepcha\",\n \"ligature\",\n \"light\",\n \"limbu\",\n \"linear\",\n \"lisu\",\n \"long\",\n \"low\",\n \"lycian\",\n \"lydian\",\n \"macron\",\n \"macron-acute\",\n \"macron-breve\",\n \"macron-grave\",\n \"mahajani\",\n \"makasar\",\n \"malayalam\",\n \"mandaic\",\n \"manichaean\",\n \"marchen\",\n \"masaram\",\n \"masculine\",\n \"masu\",\n \"mathematical\",\n \"medefaidrin\",\n \"meetei\",\n \"mende\",\n \"meroitic\",\n \"miao\",\n \"micro\",\n \"minus\",\n \"modi\",\n \"modifier\",\n \"mongolian\",\n \"mro\",\n \"multani\",\n \"musical\",\n \"myanmar\",\n \"nabataean\",\n \"nag\",\n \"nandinagari\",\n \"new\",\n \"newa\",\n \"nko\",\n \"not\",\n \"number\",\n \"nushu\",\n \"nyiakeng\",\n \"ogham\",\n \"ogonek\",\n \"ohm\",\n \"ol\",\n \"old\",\n \"open\",\n \"oriya\",\n \"osage\",\n \"osmanya\",\n \"overline\",\n \"pahawh\",\n \"palatalized\",\n \"palmyrene\",\n \"parentheses\",\n \"pau\",\n \"phags-pa\",\n \"phaistos\",\n \"phoenician\",\n \"planck\",\n \"plus\",\n \"psalter\",\n \"rejang\",\n \"retroflex\",\n \"reverse\",\n \"reversed\",\n \"right\",\n \"rightwards\",\n \"ring\",\n \"roman\",\n \"runic\",\n \"samaritan\",\n \"saurashtra\",\n \"script\",\n \"seagull\",\n \"sharada\",\n \"shavian\",\n \"short\",\n \"siddham\",\n \"signwriting\",\n \"sinhala\",\n \"snake\",\n \"sogdian\",\n \"sora\",\n \"soyombo\",\n \"square\",\n \"strong\",\n \"sundanese\",\n \"superscript\",\n \"suspension\",\n \"syloti\",\n \"syriac\",\n \"tagalog\",\n \"tagbanwa\",\n \"tai\",\n \"takri\",\n \"tamil\",\n \"tangsa\",\n \"tangut\",\n \"telugu\",\n \"thaana\",\n \"thai\",\n \"three\",\n \"tibetan\",\n \"tifinagh\",\n \"tilde\",\n \"tirhuta\",\n \"toto\",\n \"triple\",\n \"turned\",\n \"ugaritic\",\n \"up\",\n \"upwards\",\n \"ur\",\n \"us\",\n \"vai\",\n \"variation\",\n \"vedic\",\n \"vertical\",\n \"vietnamese\",\n \"vithkuqi\",\n \"wancho\",\n \"warang\",\n \"wide\",\n \"wiggly\",\n \"x\",\n \"x-x\",\n \"yezidi\",\n \"yi\",\n \"zanabazar\",\n \"zigzag\",\n \"znamenny\",\n];\n\nconst UNICODE_NAME_RANGES = `41,5a,124;61,7a,124;aa,aa,74;b5,b5,156;ba,ba,148;c0,d6,124;d8,f6,124;f8,2af,124;2b0,2c1,159;2c6,2c6,159;2c7,2c7,34;2c8,2d1,159;2e0,2e4,159;2ec,2ec,159;2ee,2ee,159;300,300,84;301,301,0;302,302,41;303,303,239;304,304,137;305,305,184;306,306,26;307,307,59;308,308,55;309,309,100;30a,30a,202;30b,30b,61;30c,30c,34;30d,30d,252;30e,30f,61;310,310,32;311,311,107;312,312,243;313,313,44;314,314,199;315,315,44;316,316,84;317,317,0;318,318,125;319,319,200;31a,31a,125;31b,31b,101;31c,31c,125;31d,31d,245;31e,31e,64;31f,31f,194;320,320,157;321,321,186;322,322,197;323,323,59;324,324,55;325,325,202;326,326,44;327,327,36;328,328,176;329,329,252;32a,32a,28;32b,32b,107;32c,32c,34;32d,32d,41;32e,32e,26;32f,32f,107;330,330,239;331,331,137;332,332,134;333,333,61;334,334,239;335,335,211;336,336,133;337,337,211;338,338,133;339,339,200;33a,33a,107;33b,33b,219;33c,33c,208;33d,33d,259;33e,33e,252;33f,33f,61;340,340,84;341,341,0;342,345,87;346,346,28;347,347,71;348,348,61;349,349,125;34a,34a,171;34b,34b,99;34c,34c,6;34d,34d,125;34e,34e,246;34f,34f,83;350,350,200;351,351,125;352,352,75;353,353,259;354,354,125;355,357,200;358,358,59;359,359,13;35a,35a,61;35b,35b,264;35c,362,61;363,36f,124;370,374,87;376,377,87;37a,37d,87;37f,37f,87;386,386,87;388,38a,87;38c,38c,87;38e,3a1,87;3a3,3e1,87;3e2,3ef,46;3f0,3f5,87;3f7,3ff,87;400,481,50;483,52f,50;531,556,12;559,559,12;560,588,12;591,5bd,96;5bf,5bf,96;5c1,5c2,96;5c4,5c5,96;5c7,5c7,96;5d0,5ea,96;5ef,5f2,96;610,61a,11;620,65f,11;66e,6d3,11;6d5,6dc,11;6df,6e8,11;6ea,6ef,11;6fa,6fc,11;6ff,6ff,11;710,74a,225;74d,74f,225;750,77f,11;780,7b1,234;7ca,7f5,170;7fa,7fa,170;7fd,7fd,170;800,82d,205;840,85b,144;860,86a,225;870,887,11;889,88e,11;898,8e1,11;8e3,8ff,11;900,963,54;971,97f,54;980,983,19;985,98c,19;98f,990,19;993,9a8,19;9aa,9b0,19;9b2,9b2,19;9b6,9b9,19;9bc,9c4,19;9c7,9c8,19;9cb,9ce,19;9d7,9d7,19;9dc,9dd,19;9df,9e3,19;9f0,9f1,19;9fc,9fc,19;9fe,9fe,19;a01,a03,90;a05,a0a,90;a0f,a10,90;a13,a28,90;a2a,a30,90;a32,a33,90;a35,a36,90;a38,a39,90;a3c,a3c,90;a3e,a42,90;a47,a48,90;a4b,a4d,90;a51,a51,90;a59,a5c,90;a5e,a5e,90;a70,a75,90;a81,a83,88;a85,a8d,88;a8f,a91,88;a93,aa8,88;aaa,ab0,88;ab2,ab3,88;ab5,ab9,88;abc,ac5,88;ac7,ac9,88;acb,acd,88;ad0,ad0,88;ae0,ae3,88;af9,aff,88;b01,b03,181;b05,b0c,181;b0f,b10,181;b13,b28,181;b2a,b30,181;b32,b33,181;b35,b39,181;b3c,b44,181;b47,b48,181;b4b,b4d,181;b55,b57,181;b5c,b5d,181;b5f,b63,181;b71,b71,181;b82,b83,230;b85,b8a,230;b8e,b90,230;b92,b95,230;b99,b9a,230;b9c,b9c,230;b9e,b9f,230;ba3,ba4,230;ba8,baa,230;bae,bb9,230;bbe,bc2,230;bc6,bc8,230;bca,bcd,230;bd0,bd0,230;bd7,bd7,230;c00,c0c,233;c0e,c10,233;c12,c28,233;c2a,c39,233;c3c,c44,233;c46,c48,233;c4a,c4d,233;c55,c56,233;c58,c5a,233;c5d,c5d,233;c60,c63,233;c80,c83,111;c85,c8c,111;c8e,c90,111;c92,ca8,111;caa,cb3,111;cb5,cb9,111;cbc,cc4,111;cc6,cc8,111;cca,ccd,111;cd5,cd6,111;cdd,cde,111;ce0,ce3,111;cf1,cf3,111;d00,d0c,143;d0e,d10,143;d12,d44,143;d46,d48,143;d4a,d4e,143;d54,d57,143;d5f,d63,143;d7a,d7f,143;d81,d83,214;d85,d96,214;d9a,db1,214;db3,dbb,214;dbd,dbd,214;dc0,dc6,214;dca,dca,214;dcf,dd4,214;dd6,dd6,214;dd8,ddf,214;df2,df3,214;e01,e3a,235;e40,e4e,235;e81,e82,123;e84,e84,123;e86,e8a,123;e8c,ea3,123;ea5,ea5,123;ea7,ebd,123;ec0,ec4,123;ec6,ec6,123;ec8,ece,123;edc,edf,123;f00,f00,237;f18,f19,237;f35,f35,237;f37,f37,237;f39,f39,237;f3e,f47,237;f49,f6c,237;f71,f84,237;f86,f97,237;f99,fbc,237;fc6,fc6,237;1000,103f,164;1050,108f,164;109a,109d,164;10a0,10c5,78;10c7,10c7,78;10cd,10cd,78;10d0,10fa,78;10fc,10fc,159;10fd,10ff,78;1100,11ff,92;1200,1248,72;124a,124d,72;1250,1256,72;1258,1258,72;125a,125d,72;1260,1288,72;128a,128d,72;1290,12b0,72;12b2,12b5,72;12b8,12be,72;12c0,12c0,72;12c2,12c5,72;12c8,12d6,72;12d8,1310,72;1312,1315,72;1318,135a,72;135d,135f,72;1380,138f,72;13a0,13f5,39;13f8,13fd,39;1401,166c,31;166f,167f,31;1681,169a,175;16a0,16ea,204;16f1,16f8,204;1700,1715,226;171f,171f,226;1720,1734,94;1740,1753,30;1760,176c,227;176e,1770,227;1772,1773,227;1780,17d3,120;17d7,17d7,120;17dc,17dd,120;180b,180d,160;180f,180f,160;1820,1878,160;1880,18aa,160;18b0,18f5,31;1900,191e,130;1920,192b,130;1930,193b,130;1950,196d,228;1970,1974,228;1980,19ab,168;19b0,19c9,168;1a00,1a1b,29;1a20,1a5e,228;1a60,1a7c,228;1a7f,1a7f,228;1aa7,1aa7,228;1ab0,1ab0,63;1ab1,1ab1,56;1ab2,1ab2,104;1ab3,1ab3,65;1ab4,1ab4,242;1ab5,1ab5,260;1ab6,1ab6,258;1ab7,1ab7,180;1ab8,1ab8,61;1ab9,1ab9,129;1aba,1aba,220;1abb,1abb,188;1abc,1abc,61;1abd,1abe,188;1abf,1ac0,124;1ac1,1ac1,125;1ac2,1ac2,200;1ac3,1ac3,125;1ac4,1ac4,200;1ac5,1ac5,219;1ac6,1ac6,172;1ac7,1ac7,107;1ac8,1ac8,194;1ac9,1aca,61;1acb,1acb,242;1acc,1ace,124;1b00,1b4c,15;1b6b,1b73,15;1b80,1baf,221;1bba,1bbf,221;1bc0,1bf3,18;1c00,1c37,127;1c4d,1c4f,127;1c5a,1c7d,178;1c80,1c88,50;1c90,1cba,78;1cbd,1cbf,78;1cd0,1cd2,251;1cd4,1cfa,251;1d00,1d25,124;1d26,1d2a,87;1d2b,1d2b,50;1d2c,1d61,159;1d62,1d65,124;1d66,1d6a,87;1d6b,1d77,124;1d78,1d78,159;1d79,1d9a,124;1d9b,1dbf,159;1dc0,1dc1,60;1dc2,1dc2,215;1dc3,1dc3,223;1dc4,1dc4,138;1dc5,1dc5,86;1dc6,1dc6,140;1dc7,1dc7,2;1dc8,1dc8,85;1dc9,1dc9,1;1dca,1dca,124;1dcb,1dcb,27;1dcc,1dcc,139;1dcd,1dcd,61;1dce,1dce,176;1dcf,1dcf,264;1dd0,1dd0,108;1dd1,1dd1,247;1dd2,1dd2,248;1dd3,1df4,124;1df5,1df5,245;1df6,1df7,114;1df8,1df8,59;1df9,1df9,257;1dfa,1dfa,59;1dfb,1dfb,52;1dfc,1dfc,61;1dfd,1dfd,6;1dfe,1dfe,125;1dff,1dff,200;1e00,1eff,124;1f00,1f15,87;1f18,1f1d,87;1f20,1f45,87;1f48,1f4d,87;1f50,1f57,87;1f59,1f59,87;1f5b,1f5b,87;1f5d,1f5d,87;1f5f,1f7d,87;1f80,1fb4,87;1fb6,1fbc,87;1fbe,1fbe,87;1fc2,1fc4,87;1fc6,1fcc,87;1fd0,1fd3,87;1fd6,1fdb,87;1fe0,1fec,87;1ff2,1ff4,87;1ff6,1ffc,87;2071,2071,222;207f,207f,222;2090,209c,124;20d0,20d0,125;20d1,20d1,200;20d2,20d2,133;20d3,20d3,211;20d4,20d4,10;20d5,20d5,43;20d6,20d6,125;20d7,20d7,200;20d8,20d8,202;20d9,20d9,43;20da,20da,10;20db,20db,236;20dc,20dc,76;20dd,20e0,70;20e1,20e1,125;20e2,20e4,70;20e5,20e5,198;20e6,20e6,61;20e7,20e7,9;20e8,20e8,242;20e9,20e9,257;20ea,20ea,126;20eb,20eb,133;20ec,20ec,201;20ed,20ed,126;20ee,20ee,125;20ef,20ef,200;20f0,20f0,13;2102,2102,62;2107,2107,73;210a,210b,207;210c,210c,23;210d,210d,62;210e,210f,193;2110,2110,207;2111,2111,23;2112,2113,207;2115,2115,62;2119,211a,62;211b,211b,207;211c,211c,23;211d,211d,62;2124,2124,62;2126,2126,177;2128,2128,23;212a,212a,117;212b,212b,8;212c,212c,207;212d,212d,23;212f,2131,207;2132,2132,243;2133,2134,207;2135,2135,5;2136,2136,20;2137,2137,79;2138,2138,51;2139,2139,105;213c,213f,62;2145,2149,62;214e,214e,243;2183,2183,203;2184,2184,124;2c00,2c5f,80;2c60,2c7c,124;2c7d,2c7d,159;2c7e,2c7f,124;2c80,2ce4,46;2ceb,2cf3,46;2d00,2d25,78;2d27,2d27,78;2d2d,2d2d,78;2d30,2d67,238;2d6f,2d6f,238;2d7f,2d7f,238;2d80,2d96,72;2da0,2da6,72;2da8,2dae,72;2db0,2db6,72;2db8,2dbe,72;2dc0,2dc6,72;2dc8,2dce,72;2dd0,2dd6,72;2dd8,2dde,72;2de0,2dff,50;2e2f,2e2f,252;3005,3006,102;302a,302d,102;302e,302f,92;3031,3035,252;303b,303b,252;303c,303c,149;3041,3096,98;3099,309a,113;309d,309f,98;30a1,30fa,112;30fc,30fc,113;30fd,30ff,112;3105,312f,24;3131,318e,92;31a0,31bf,24;31f0,31ff,112;3400,4dbf,42;4e00,9fff,42;a000,a48c,262;a4d0,a4fd,132;a500,a60c,249;a610,a61f,249;a62a,a62b,249;a640,a672,50;a674,a67d,50;a67f,a69b,50;a69c,a69d,159;a69e,a69f,50;a6a0,a6e5,16;a6f0,a6f1,16;a717,a71f,159;a722,a76f,124;a770,a770,159;a771,a787,124;a788,a788,159;a78b,a7ca,124;a7d0,a7d1,124;a7d3,a7d3,124;a7d5,a7d9,124;a7f2,a7f4,159;a7f5,a7f7,124;a7f8,a7f9,159;a7fa,a7ff,124;a800,a827,224;a82c,a82c,224;a840,a873,190;a880,a8c5,206;a8e0,a8f7,54;a8fb,a8fb,54;a8fd,a8ff,54;a90a,a92d,116;a930,a953,196;a960,a97c,92;a980,a9c0,109;a9cf,a9cf,109;a9e0,a9ef,164;a9fa,a9fe,164;aa00,aa36,38;aa40,aa4d,38;aa60,aa76,164;aa7a,aa7f,164;aa80,aac2,228;aadb,aadd,228;aae0,aaef,152;aaf2,aaf6,152;ab01,ab06,72;ab09,ab0e,72;ab11,ab16,72;ab20,ab26,72;ab28,ab2e,72;ab30,ab5a,124;ab5c,ab5f,159;ab60,ab64,124;ab65,ab65,87;ab66,ab68,124;ab69,ab69,159;ab70,abbf,39;abc0,abea,152;abec,abed,152;ac00,d7a3,92;d7b0,d7c6,92;d7cb,d7fb,92;f900,fa6d,42;fa70,fad9,42;fb00,fb06,124;fb13,fb17,12;fb1d,fb28,96;fb2a,fb36,96;fb38,fb3c,96;fb3e,fb3e,96;fb40,fb41,96;fb43,fb44,96;fb46,fb4f,96;fb50,fbb1,11;fbd3,fd3d,11;fd50,fd8f,11;fd92,fdc7,11;fdf0,fdfb,11;fe00,fe0f,250;fe20,fe21,128;fe22,fe23,61;fe24,fe25,137;fe26,fe26,45;fe27,fe28,128;fe29,fe2a,239;fe2b,fe2c,137;fe2d,fe2d,45;fe2e,fe2f,50;fe70,fe74,11;fe76,fefc,11;ff21,ff3a,77;ff41,ff5a,77;ff66,ffbe,91;ffc2,ffc7,91;ffca,ffcf,91;ffd2,ffd7,91;ffda,ffdc,91;10000,1000b,131;1000d,10026,131;10028,1003a,131;1003c,1003d,131;1003f,1004d,131;10050,1005d,131;10080,100fa,131;101fd,101fd,191;10280,1029c,135;102a0,102d0,33;102e0,102e0,46;10300,1031f,179;1032d,1032f,179;10330,10340,81;10342,10349,81;10350,1037a,179;10380,1039d,244;103a0,103c3,179;103c8,103cf,179;10400,1044f,53;10450,1047f,210;10480,1049d,183;104b0,104d3,182;104d8,104fb,182;10500,10527,68;10530,10563,35;10570,1057a,254;1057c,1058a,254;1058c,10592,254;10594,10595,254;10597,105a1,254;105a3,105b1,254;105b3,105b9,254;105bb,105bc,254;10600,10736,131;10740,10755,131;10760,10767,131;10780,10785,159;10787,107b0,159;107b2,107ba,159;10800,10805,48;10808,10808,48;1080a,10835,48;10837,10838,48;1083c,1083c,48;1083f,1083f,48;10840,10855,103;10860,10876,187;10880,1089e,165;108e0,108f2,95;108f4,108f5,95;10900,10915,192;10920,10939,136;10980,109b7,154;109be,109bf,154;10a00,10a03,118;10a05,10a06,118;10a0c,10a13,118;10a15,10a17,118;10a19,10a35,118;10a38,10a3a,118;10a3f,10a3f,118;10a60,10a7c,179;10a80,10a9c,179;10ac0,10ac7,145;10ac9,10ae6,145;10b00,10b35,14;10b40,10b55,106;10b60,10b72,106;10b80,10b91,195;10c00,10c48,179;10c80,10cb2,179;10cc0,10cf2,179;10d00,10d27,93;10e80,10ea9,261;10eab,10eac,261;10eb0,10eb1,261;10efd,10eff,11;10f00,10f1c,179;10f27,10f27,179;10f30,10f50,216;10f70,10f85,179;10fb0,10fc4,40;10fe0,10ff6,69;11000,11046,25;11070,11075,25;1107f,1107f,25;11080,110ba,110;110c2,110c2,110;110d0,110e8,217;11100,11134,37;11144,11147,37;11150,11173,141;11176,11176,141;11180,111c4,209;111c9,111cc,209;111ce,111cf,209;111da,111da,209;111dc,111dc,209;11200,11211,121;11213,11237,121;1123e,11241,121;11280,11286,162;11288,11288,162;1128a,1128d,162;1128f,1129d,162;1129f,112a8,162;112b0,112ea,122;11300,11303,82;11305,1130c,82;1130f,11310,82;11313,11328,82;1132a,11330,82;11332,11333,82;11335,11339,82;1133b,1133b,22;1133c,11344,82;11347,11348,82;1134b,1134d,82;11350,11350,82;11357,11357,82;1135d,11363,82;11366,1136c,82;11370,11374,82;11400,1144a,169;1145e,11461,169;11480,114c5,240;114c7,114c7,240;11580,115b5,212;115b8,115c0,212;115d8,115dd,212;11600,11640,158;11644,11644,158;11680,116b8,229;11700,1171a,4;1171d,1172b,4;11740,11746,4;11800,1183a,58;118a0,118df,256;118ff,118ff,256;11900,11906,57;11909,11909,57;1190c,11913,57;11915,11916,57;11918,11935,57;11937,11938,57;1193b,11943,57;119a0,119a7,167;119aa,119d7,167;119da,119e1,167;119e3,119e4,167;11a00,11a3e,263;11a47,11a47,263;11a50,11a99,218;11a9d,11a9d,218;11ab0,11abf,31;11ac0,11af8,189;11c00,11c08,21;11c0a,11c36,21;11c38,11c40,21;11c72,11c8f,146;11c92,11ca7,146;11ca9,11cb6,146;11d00,11d06,147;11d08,11d09,147;11d0b,11d36,147;11d3a,11d3a,147;11d3c,11d3d,147;11d3f,11d47,147;11d60,11d65,89;11d67,11d68,89;11d6a,11d8e,89;11d90,11d91,89;11d93,11d98,89;11ee0,11ef6,142;11f00,11f10,115;11f12,11f3a,115;11f3e,11f42,115;11fb0,11fb0,132;12000,12399,47;12480,12543,47;12f90,12ff0,49;13000,1342f,67;13440,13455,67;14400,14646,7;16800,16a38,16;16a40,16a5e,161;16a70,16abe,231;16ad0,16aed,17;16af0,16af4,17;16b00,16b36,185;16b40,16b43,185;16b63,16b77,185;16b7d,16b8f,185;16e40,16e7f,151;16f00,16f4a,155;16f4f,16f87,155;16f8f,16f9f,155;16fe0,16fe0,232;16fe1,16fe1,173;16fe3,16fe3,179;16fe4,16fe4,119;16ff0,16ff1,253;18800,18aff,232;18b00,18cd5,119;1aff0,1aff3,112;1aff5,1affb,112;1affd,1affe,112;1b000,1b000,112;1b001,1b001,98;1b002,1b11e,97;1b11f,1b11f,98;1b120,1b122,112;1b132,1b132,98;1b150,1b152,98;1b155,1b155,112;1b164,1b167,112;1b170,1b2fb,173;1bc00,1bc6a,66;1bc70,1bc7c,66;1bc80,1bc88,66;1bc90,1bc99,66;1bc9d,1bc9e,66;1cf00,1cf2d,265;1cf30,1cf46,265;1d165,1d169,163;1d16d,1d172,163;1d17b,1d182,163;1d185,1d18b,163;1d1aa,1d1ad,163;1d242,1d244,87;1d400,1d454,150;1d456,1d49c,150;1d49e,1d49f,150;1d4a2,1d4a2,150;1d4a5,1d4a6,150;1d4a9,1d4ac,150;1d4ae,1d4b9,150;1d4bb,1d4bb,150;1d4bd,1d4c3,150;1d4c5,1d505,150;1d507,1d50a,150;1d50d,1d514,150;1d516,1d51c,150;1d51e,1d539,150;1d53b,1d53e,150;1d540,1d544,150;1d546,1d546,150;1d54a,1d550,150;1d552,1d6a5,150;1d6a8,1d6c0,150;1d6c2,1d6da,150;1d6dc,1d6fa,150;1d6fc,1d714,150;1d716,1d734,150;1d736,1d74e,150;1d750,1d76e,150;1d770,1d788,150;1d78a,1d7a8,150;1d7aa,1d7c2,150;1d7c4,1d7cb,150;1da00,1da36,213;1da3b,1da6c,213;1da75,1da75,213;1da84,1da84,213;1da9b,1da9f,213;1daa1,1daaf,213;1df00,1df1e,124;1df25,1df2a,124;1e000,1e006,80;1e008,1e018,80;1e01b,1e021,80;1e023,1e024,80;1e026,1e02a,80;1e030,1e050,159;1e051,1e06a,50;1e06b,1e06d,159;1e08f,1e08f,50;1e100,1e12c,174;1e130,1e13d,174;1e14e,1e14e,174;1e290,1e2ae,241;1e2c0,1e2ef,255;1e4d0,1e4ef,166;1e7e0,1e7e6,72;1e7e8,1e7eb,72;1e7ed,1e7ee,72;1e7f0,1e7fe,72;1e800,1e8c4,153;1e8d0,1e8d6,153;1e900,1e94b,3;1ee00,1ee03,11;1ee05,1ee1f,11;1ee21,1ee22,11;1ee24,1ee24,11;1ee27,1ee27,11;1ee29,1ee32,11;1ee34,1ee37,11;1ee39,1ee39,11;1ee3b,1ee3b,11;1ee42,1ee42,11;1ee47,1ee47,11;1ee49,1ee49,11;1ee4b,1ee4b,11;1ee4d,1ee4f,11;1ee51,1ee52,11;1ee54,1ee54,11;1ee57,1ee57,11;1ee59,1ee59,11;1ee5b,1ee5b,11;1ee5d,1ee5d,11;1ee5f,1ee5f,11;1ee61,1ee62,11;1ee64,1ee64,11;1ee67,1ee6a,11;1ee6c,1ee72,11;1ee74,1ee77,11;1ee79,1ee7c,11;1ee7e,1ee7e,11;1ee80,1ee89,11;1ee8b,1ee9b,11;1eea1,1eea3,11;1eea5,1eea9,11;1eeab,1eebb,11;20000,2a6df,42;2a700,2b739,42;2b740,2b81d,42;2b820,2cea1,42;2ceb0,2ebe0,42;2ebf0,2ee5d,42;2f800,2fa1d,42;30000,3134a,42;31350,323af,42;e0100,e01ef,250`;\n\ntype NameRange = { start: number; end: number; word: number };\n\nconst NAME_RANGE_TABLE: NameRange[] = UNICODE_NAME_RANGES.split(\";\").map((entry) => {\n const [start, end, word] = entry.split(\",\");\n return { start: Number.parseInt(start!, 16), end: Number.parseInt(end!, 16), word: Number(word) };\n});\n\nconst SCRIPT_CACHE = new Map<string, string>();\nconst SCRIPT_CACHE_MAX = 4096;\n\nfunction lookupNamePrefix(cp: number): string {\n let lo = 0;\n let hi = NAME_RANGE_TABLE.length - 1;\n while (lo <= hi) {\n const mid = (lo + hi) >> 1;\n const range = NAME_RANGE_TABLE[mid]!;\n if (cp < range.start) hi = mid - 1;\n else if (cp > range.end) lo = mid + 1;\n else return UNICODE_NAME_WORDS[range.word] ?? \"\";\n }\n return \"\";\n}\n\nexport const SPACELESS_SCRIPTS = new Set([\n \"cjk\",\n \"hiragana\",\n \"katakana\",\n \"thai\",\n \"lao\",\n \"khmer\",\n \"myanmar\",\n \"tibetan\",\n \"yi\",\n]);\n\nexport const SENTENCE_TERMINALS = new Set([\n \".\",\n \"!\",\n \"?\",\n \"。\",\n \"!\",\n \"?\",\n \"؟\",\n \"।\",\n \"॥\",\n \"։\",\n \"።\",\n \"။\",\n \"៕\",\n \"᙮\",\n \"⁇\",\n \"⁈\",\n \"⁉\",\n \"꓿\",\n \"᜵\",\n \"᜶\",\n]);\n\n// Audit T1: drop U+003B (ASCII semicolon) — Greek's question mark IS that\n// codepoint, and U+037E NFC-normalizes to it, so including U+003B classified\n// every query with a plain semicolon as a question. Keep U+037E for explicitly\n// encoded Greek question marks.\nexport const QUESTION_MARKS = new Set([\"?\", \"؟\", \"?\", \"\\u037E\", \"⁇\", \"⁈\"]);\n\nexport const SCRIPT_TO_ISO: Record<string, string> = {\n thai: \"th\",\n bengali: \"bn\",\n tamil: \"ta\",\n telugu: \"te\",\n kannada: \"kn\",\n malayalam: \"ml\",\n gujarati: \"gu\",\n gurmukhi: \"pa\",\n sinhala: \"si\",\n lao: \"lo\",\n myanmar: \"my\",\n khmer: \"km\",\n georgian: \"ka\",\n armenian: \"hy\",\n ethiopic: \"am\",\n tibetan: \"bo\",\n greek: \"el\",\n hangul: \"ko\",\n thaana: \"dv\",\n devanagari: \"hi\",\n oriya: \"or\",\n};\n\nexport const AMBIGUOUS_SCRIPTS = new Set([\"latin\", \"cyrillic\", \"arabic\", \"cjk\"]);\n\nconst JA_SCRIPTS = new Set([\"hiragana\", \"katakana\", \"cjk\"]);\n\nconst AMBIGUOUS_TERMINALS = new Set([\".\", \"!\", \"?\"]);\n\nconst ISO_639_3_TO_1: Record<string, string> = {\n afr: \"af\",\n amh: \"am\",\n ara: \"ar\",\n arb: \"ar\",\n arz: \"ar\",\n apc: \"ar\",\n aze: \"az\",\n azj: \"az\",\n bel: \"be\",\n ben: \"bn\",\n bod: \"bo\",\n bos: \"bs\",\n bul: \"bg\",\n cat: \"ca\",\n ces: \"cs\",\n cmn: \"zh\",\n cym: \"cy\",\n dan: \"da\",\n deu: \"de\",\n ell: \"el\",\n eng: \"en\",\n epo: \"eo\",\n est: \"et\",\n ekk: \"et\",\n eus: \"eu\",\n fas: \"fa\",\n pes: \"fa\",\n prs: \"fa\",\n fin: \"fi\",\n fra: \"fr\",\n fry: \"fy\",\n gla: \"gd\",\n gle: \"ga\",\n glg: \"gl\",\n guj: \"gu\",\n hat: \"ht\",\n hau: \"ha\",\n heb: \"he\",\n hin: \"hi\",\n hrv: \"hr\",\n hun: \"hu\",\n hye: \"hy\",\n ibo: \"ig\",\n ind: \"id\",\n isl: \"is\",\n ita: \"it\",\n jav: \"jv\",\n jpn: \"ja\",\n kan: \"kn\",\n kat: \"ka\",\n kaz: \"kk\",\n khm: \"km\",\n khk: \"mn\",\n kin: \"rw\",\n kir: \"ky\",\n kor: \"ko\",\n kur: \"ku\",\n kmr: \"ku\",\n lao: \"lo\",\n lat: \"la\",\n lav: \"lv\",\n lvs: \"lv\",\n lit: \"lt\",\n ltz: \"lb\",\n mal: \"ml\",\n mar: \"mr\",\n mkd: \"mk\",\n mlg: \"mg\",\n plt: \"mg\",\n mlt: \"mt\",\n mon: \"mn\",\n msa: \"ms\",\n zlm: \"ms\",\n mya: \"my\",\n nep: \"ne\",\n nld: \"nl\",\n nno: \"nn\",\n nob: \"nb\",\n nor: \"no\",\n nya: \"ny\",\n ori: \"or\",\n pan: \"pa\",\n pol: \"pl\",\n por: \"pt\",\n pus: \"ps\",\n pbu: \"ps\",\n que: \"qu\",\n qug: \"qu\",\n ron: \"ro\",\n rus: \"ru\",\n sin: \"si\",\n slk: \"sk\",\n slv: \"sl\",\n sna: \"sn\",\n som: \"so\",\n spa: \"es\",\n sqi: \"sq\",\n srp: \"sr\",\n sun: \"su\",\n swa: \"sw\",\n swh: \"sw\",\n swe: \"sv\",\n tam: \"ta\",\n tel: \"te\",\n tgk: \"tg\",\n tgl: \"tl\",\n tha: \"th\",\n tir: \"ti\",\n tuk: \"tk\",\n tur: \"tr\",\n uig: \"ug\",\n ukr: \"uk\",\n urd: \"ur\",\n uzb: \"uz\",\n uzn: \"uz\",\n vie: \"vi\",\n yid: \"yi\",\n yor: \"yo\",\n zho: \"zh\",\n zul: \"zu\",\n};\n\nfunction francToIso6391(code: string): string | null {\n if (!code || code === \"und\") return null;\n if (code.length === 2) return code.toLowerCase();\n return ISO_639_3_TO_1[code] ?? null;\n}\n\n/**\n * Extract script from Unicode character names — same logic as\n * unicodedata.name(ch).split()[0].lower(), with COMBINING → second word.\n * Letters (L) and marks (M) only; everything else returns \"\".\n */\nexport function charScript(ch: string): string {\n const cached = SCRIPT_CACHE.get(ch);\n if (cached !== undefined) return cached;\n let result = \"\";\n if (ch) {\n const cp = ch.codePointAt(0);\n if (cp !== undefined) {\n const letter = String.fromCodePoint(cp);\n if (/^\\p{L}$/u.test(letter) || /^\\p{M}$/u.test(letter)) {\n result = lookupNamePrefix(cp);\n }\n }\n }\n if (SCRIPT_CACHE.size >= SCRIPT_CACHE_MAX) {\n const first = SCRIPT_CACHE.keys().next().value;\n if (first !== undefined) SCRIPT_CACHE.delete(first);\n }\n SCRIPT_CACHE.set(ch, result);\n return result;\n}\n\nfunction mostCommonScripts(scripts: Map<string, number>, n: number): Array<[string, number]> {\n return [...scripts.entries()].sort((a, b) => b[1] - a[1]).slice(0, n);\n}\n\nfunction sliceCodePoints(text: string, max: number): string {\n if (text.length <= max) return text;\n let n = 0;\n let end = 0;\n for (const ch of text) {\n if (n >= max) break;\n end += ch.length;\n n += 1;\n }\n return text.slice(0, end);\n}\n\nfunction scriptOnly(text: string, targetScript: string): string {\n let out = \"\";\n for (const ch of text) {\n if (/^\\s$/u.test(ch)) out += \" \";\n else if (charScript(ch) === targetScript) out += ch;\n }\n return out;\n}\n\nfunction statisticalDetect(sample: string, dominantScript: string, scripts: Map<string, number>): string {\n if (dominantScript === \"hiragana\" || dominantScript === \"katakana\") return \"ja\";\n if (dominantScript === \"cjk\") {\n const totalChars = [...scripts.values()].reduce((a, b) => a + b, 0) || 1;\n const kana = (scripts.get(\"hiragana\") ?? 0) + (scripts.get(\"katakana\") ?? 0);\n if (kana / totalChars >= 0.02) return \"ja\";\n return \"zh\";\n }\n\n try {\n const clean = scriptOnly(sample, dominantScript);\n if (clean.trim().length < 20) return dominantScript;\n const ranked = francAll(clean, { minLength: 10 });\n // franc ranks 639-3 codes; skip ones with no 639-1 mapping (e.g. `sco`\n // outranking `eng` on short English) just as Lingua falls back when\n // iso_code_639_1 is missing — then take the next confident 639-1 hit.\n for (const [code, conf] of ranked) {\n if (conf < 0.5) break;\n const iso = francToIso6391(code);\n if (iso) return iso;\n }\n } catch {\n // franc failed — fall through to script name\n }\n return dominantScript;\n}\n\nexport function detectLanguage(txt: unknown, _minChars = 30): string | null {\n if (!txt) return null;\n let text: string;\n if (typeof txt === \"string\") text = txt;\n else {\n try {\n text = String(txt);\n } catch {\n return null;\n }\n }\n text = text.trim();\n if (!text) return null;\n\n const sample = text.length > 1500 ? sliceCodePoints(text, 1500) : text;\n\n const scripts = new Map<string, number>();\n for (const ch of sample) {\n const s = charScript(ch);\n if (s) scripts.set(s, (scripts.get(s) ?? 0) + 1);\n }\n\n const total = [...scripts.values()].reduce((a, b) => a + b, 0);\n if (total < 20) return null;\n\n const topScripts = mostCommonScripts(scripts, 5).map(\n ([script, count]) => [script, count / total] as [string, number],\n );\n if (!topScripts.length) return null;\n\n const [dominantScript, dominantRatio] = topScripts[0]!;\n const secondRatio = topScripts[1]?.[1] ?? 0.0;\n\n let jaCount = 0;\n for (const [s, c] of scripts) {\n if (JA_SCRIPTS.has(s)) jaCount += c;\n }\n const jaRatio = jaCount / total;\n const kanaRatio = ((scripts.get(\"hiragana\") ?? 0) + (scripts.get(\"katakana\") ?? 0)) / total;\n if (jaRatio >= 0.8 && kanaRatio >= 0.02) return \"ja\";\n\n if (dominantRatio >= 0.2 && secondRatio >= 0.2) return \"mixed\";\n\n if (dominantRatio >= 0.6) {\n const lang = SCRIPT_TO_ISO[dominantScript];\n if (lang) return lang;\n return statisticalDetect(sample, dominantScript, scripts);\n }\n\n return null;\n}\n\nexport function isSpacelessChar(ch: string): boolean {\n return SPACELESS_SCRIPTS.has(charScript(ch));\n}\n\nfunction isCombiningMark(ch: string): boolean {\n return /^\\p{M}$/u.test(ch);\n}\n\nexport function tokenize(text: string): string[] {\n const tokens: string[] = [];\n const currentWord: string[] = [];\n\n for (const ch of text) {\n if (/^\\s$/u.test(ch)) {\n if (currentWord.length) {\n tokens.push(currentWord.join(\"\"));\n currentWord.length = 0;\n }\n continue;\n }\n\n if (isCombiningMark(ch)) {\n if (tokens.length && !currentWord.length) {\n tokens[tokens.length - 1] = tokens[tokens.length - 1]! + ch;\n } else {\n currentWord.push(ch);\n }\n continue;\n }\n\n if (isSpacelessChar(ch)) {\n if (currentWord.length) {\n tokens.push(currentWord.join(\"\"));\n currentWord.length = 0;\n }\n tokens.push(ch);\n } else {\n currentWord.push(ch);\n }\n }\n\n if (currentWord.length) tokens.push(currentWord.join(\"\"));\n return tokens;\n}\n\nexport function tokenCount(text: string): number {\n return tokenize(text || \"\").length;\n}\n\nfunction isSentenceTerminal(ch: string): boolean {\n return SENTENCE_TERMINALS.has(ch);\n}\n\nfunction lstrip(chars: string[]): string[] {\n let i = 0;\n while (i < chars.length && /^\\s$/u.test(chars[i]!)) i += 1;\n return chars.slice(i);\n}\n\nexport function splitSentences(text: string): string[] {\n text = (text || \"\").trim();\n if (!text) return [];\n\n const chars = [...text];\n const parts: string[] = [];\n const current: string[] = [];\n\n for (let i = 0; i < chars.length; i++) {\n const ch = chars[i]!;\n current.push(ch);\n\n if (isSentenceTerminal(ch)) {\n const rest = chars.slice(i + 1);\n const stripped = lstrip(rest);\n\n if (!stripped.length) break;\n\n if (!AMBIGUOUS_TERMINALS.has(ch)) {\n parts.push(current.join(\"\").trim());\n current.length = 0;\n continue;\n }\n\n if (rest.length && /^\\s$/u.test(rest[0]!)) {\n parts.push(current.join(\"\").trim());\n current.length = 0;\n } else if (stripped.length) {\n const currentScript = charScript(ch) || \"\";\n const nextScript = charScript(stripped[0]!) || \"\";\n if (currentScript && nextScript && currentScript !== nextScript) {\n parts.push(current.join(\"\").trim());\n current.length = 0;\n }\n }\n }\n }\n\n if (current.length) {\n const last = current.join(\"\").trim();\n if (last) parts.push(last);\n }\n\n if (parts.length <= 2) return [text];\n return parts;\n}\n\nexport function jaccardSim(a: string, b: string): number {\n const A = new Set(tokenize(a));\n const B = new Set(tokenize(b));\n if (!A.size || !B.size) return 0.0;\n let inter = 0;\n for (const t of A) if (B.has(t)) inter += 1;\n return inter / (A.size + B.size - inter);\n}\n\nexport function isQuestionMark(ch: string): boolean {\n return QUESTION_MARKS.has(ch);\n}\n","import type { UsageEvent } from \"./usage.js\";\n\nconst log = {\n warn: (...args: unknown[]) => console.warn(\"[context-engine]\", ...args),\n error: (...args: unknown[]) => console.error(\"[context-engine]\", ...args),\n};\n\n/**\n * One ingest stage boundary. `documentId` is null until the row is claimed\n * (extraction runs before that, so dedup can decide from the extracted\n * text); `sourceId` + `externalId` are the caller's own identity and exist\n * on every event. `detail` is per stage and always JSON-serialisable.\n */\nexport interface ProgressEvent {\n sourceId: string | null;\n externalId: string | null;\n documentId: string | null;\n name: string;\n stage: \"extract\" | \"redact\" | \"chunk\" | \"embed\" | \"structured\" | \"graph\";\n state: \"started\" | \"done\";\n detail: Record<string, unknown>;\n}\n\nexport interface Hooks {\n onUsage?: ((event: UsageEvent) => void) | null;\n onError?: ((exc: unknown, ctx: Record<string, unknown>) => void) | null;\n onToolCall?: ((event: Record<string, unknown>) => void) | null;\n /** Ingest stage boundaries, for live progress UIs. Never raises outward. */\n onProgress?: ((event: ProgressEvent) => void) | null;\n}\n\nexport function emitUsage(hooks: Hooks | null | undefined, event: UsageEvent): void {\n if (!hooks?.onUsage) return;\n try {\n hooks.onUsage(event);\n } catch {\n log.warn(\"onUsage callback raised; swallowing\");\n }\n}\n\nexport function emitError(hooks: Hooks | null | undefined, exc: unknown, ctx: Record<string, unknown>): void {\n log.error(\"context_engine error:\", exc, \"| ctx=\", ctx);\n if (!hooks?.onError) return;\n try {\n hooks.onError(exc, ctx);\n } catch {\n log.warn(\"onError callback raised; swallowing\");\n }\n}\n\n/** Report an ingest stage boundary to `hooks.onProgress`, if set. Never raises. */\nexport function emitProgress(hooks: Hooks | null | undefined, event: ProgressEvent): void {\n if (!hooks?.onProgress) return;\n try {\n hooks.onProgress(event);\n } catch {\n log.warn(\"onProgress callback raised; swallowing\");\n }\n}\n\nexport function emitToolCall(hooks: Hooks | null | undefined, event: Record<string, unknown>): void {\n if (!hooks?.onToolCall) return;\n try {\n hooks.onToolCall(event);\n } catch {\n log.warn(\"onToolCall callback raised; swallowing\");\n }\n}\n","/**\n * How a `@google/genai` client gets built — the one place that decides.\n *\n * Gemini reaches this package two ways, and they differ only in the client\n * constructor:\n *\n * - **gemini** — the Gemini Developer API, authenticated with an API key.\n * - **vertex_ai** — the same models on Vertex AI, authenticated with\n * Application Default Credentials against a GCP project.\n * `GOOGLE_APPLICATION_CREDENTIALS`, workload identity and\n * `gcloud auth application-default login` all work unchanged; there is\n * deliberately no credentials field of our own.\n *\n * Both the LLM provider and the embeddings provider need this decision, so it\n * lives here rather than in four copies (two per port) that would drift.\n *\n * Two SDK behaviours this module is built around, both read out of the real\n * package rather than assumed:\n *\n * - `project`/`location` and `apiKey` are **mutually exclusive** — the SDK\n * raises if handed both. A `vertex_ai` config that still carries a leftover\n * `apiKey` alongside a project must therefore not forward it, or every call\n * fails on a config that looks entirely reasonable.\n * - Omitting project/location is not automatically a broken config: the SDK\n * reads `GOOGLE_CLOUD_PROJECT` / `GOOGLE_CLOUD_LOCATION`, which is how a GCP\n * deployment is usually already wired. So unset fields are left ABSENT\n * rather than passed as null, and the \"is there enough auth\" question is\n * left to the SDK, the only party that can answer it.\n *\n * **This is where the two ports genuinely differ**, measured rather than\n * assumed: the Python SDK falls back to discovering the project from ADC\n * when neither config nor env supplies one, and this JS SDK does not — it\n * throws at construction. A deployment relying on ADC alone must therefore\n * name the project in config or in the environment for the TS client.\n */\nimport { ExtraMissingError } from \"../errors.js\";\n\n/** The fields `EmbeddingConfig` and `LLMConfig` share here. */\nexport interface GoogleProviderConfig {\n provider: string;\n apiKey?: string | null;\n project?: string | null;\n location?: string | null;\n}\n\ntype GenaiCtorOpts = {\n apiKey?: string | null;\n vertexai?: boolean;\n project?: string;\n location?: string;\n httpOptions?: { timeout?: number };\n};\n\n/**\n * Build a `@google/genai` client for `cfg`, in Gemini or Vertex mode.\n *\n * `timeoutMs` is required: the SDK has NO default timeout of its own, so a\n * missing one means a stuck backend hangs the call forever — on the ingest\n * path a wedged worker rather than a failed document.\n *\n * `purpose` only shapes the `ExtraMissingError` message (\"gemini llm\" vs\n * \"gemini embeddings\").\n */\nexport async function buildGenaiClient<T>(\n cfg: GoogleProviderConfig,\n timeoutMs: number,\n purpose: string,\n): Promise<T> {\n const specifier = \"@google/genai\";\n let mod: {\n GoogleGenAI?: new (opts: GenaiCtorOpts) => T;\n Client?: new (opts: GenaiCtorOpts) => T;\n };\n try {\n mod = (await import(specifier)) as typeof mod;\n } catch {\n throw new ExtraMissingError(\"gemini\", specifier, purpose);\n }\n const Ctor = mod.GoogleGenAI ?? mod.Client;\n if (!Ctor) {\n throw new ExtraMissingError(\"gemini\", specifier, purpose);\n }\n\n const opts: GenaiCtorOpts = { httpOptions: { timeout: timeoutMs } };\n\n if (cfg.provider === \"vertex_ai\") {\n opts.vertexai = true;\n if (cfg.project || cfg.location) {\n // Explicit project/location wins, and the API key is dropped: the SDK\n // refuses the combination outright.\n if (cfg.project) opts.project = cfg.project;\n if (cfg.location) opts.location = cfg.location;\n } else if (cfg.apiKey) {\n // Vertex express mode — an API key and no project. The one combination\n // where `apiKey` means anything on this provider.\n opts.apiKey = cfg.apiKey;\n }\n // Otherwise set neither, so the SDK's own env-var lookup runs untouched.\n } else {\n opts.apiKey = cfg.apiKey ?? null;\n }\n\n try {\n return new Ctor(opts);\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err);\n if (!message.includes(\"Authentication is not set up\")) throw err;\n // The SDK is right, but it cannot know what this config is called. Note\n // the port difference this message has to cover: the Python SDK falls\n // back to discovering the project from ADC, this one does not — it reads\n // the env vars and stops — so an ADC-only deployment must name the\n // project somewhere.\n throw new Error(\n `${message} Set \\`project\\` on the provider config, or export ` +\n \"GOOGLE_CLOUD_PROJECT and GOOGLE_CLOUD_LOCATION. Credentials \" +\n \"themselves come from Application Default Credentials.\",\n );\n }\n}\n","/**\n * Pluggable LLM provider.\n *\n * `callLlm(cfg, { system, user, jsonMode, images, client })` is the single entry\n * point. Returns `[text, { input, output }]`.\n *\n * Images are always PNG. 240s call timeout (every provider), no retries.\n * Injectable `client` / `fetch`.\n * `callLlm` owns and closes the client it builds unless one is passed in.\n */\n\nimport OpenAI from \"openai\";\nimport type { LLMConfig } from \"../config.js\";\nimport { ExtraMissingError } from \"../errors.js\";\nimport type { FetchImpl } from \"./embeddings.js\";\nimport { buildGenaiClient } from \"./google.js\";\n\n/** Hard cap per LLM call, EVERY provider. A non-streaming multi-page vision\n * transcription routinely generates for well over 30s (real batches run\n * minutes), so the old 30s default made every vision attempt on\n * anthropic/openai/bedrock time out and the pages land as `failed` — only\n * gemini had been given this cap. It also bounds the hang: `@google/genai`\n * has NO default timeout at all, and with the bounded render queue a hung\n * batch holds a worker slot for good. */\nexport const CALL_TIMEOUT_MS = 240_000;\n\n/** Back-compat aliases — the cap stopped being gemini-specific. */\nexport const TIMEOUT_MS = CALL_TIMEOUT_MS;\nexport const GEMINI_CALL_TIMEOUT_MS = CALL_TIMEOUT_MS;\nconst ANTHROPIC_VERSION = \"2023-06-01\";\nconst ANTHROPIC_MAX_TOKENS = 4096;\n\nconst OPENAI_FAMILY = new Set([\"openai\", \"azure_openai\", \"custom\"]);\n\n/**\n * Both reach the same models through the same SDK; only the client\n * constructor differs (API key vs ADC against a GCP project), which is why\n * they share every line below `buildGenaiClient`.\n */\nconst GOOGLE_FAMILY = new Set([\"gemini\", \"vertex_ai\"]);\n\nexport type TokenUsage = { input: number; output: number };\n\nexport type ImageBytes = Uint8Array;\n\nexport type ChatClient = {\n chat: {\n completions: {\n create(body: { model: string; messages: unknown[]; response_format?: { type: string } }): Promise<{\n choices: Array<{ message?: { content?: string | null } }>;\n usage?: { prompt_tokens?: number; completion_tokens?: number } | null;\n }>;\n };\n };\n close?: () => void | Promise<void>;\n timeout?: number;\n baseURL?: string;\n};\n\nexport type LlmCallOpts = {\n system: string;\n user: string;\n jsonMode?: boolean;\n images?: ImageBytes[] | null;\n maxTokens?: number | null;\n /** Reasoning tokens are drawn from `maxTokens`, so a caller that wants\n * transcription rather than deliberation must be able to spend the whole\n * budget on output. Omitted = the provider's default. Applied per provider\n * capability: gemini and anthropic take it directly (anthropic thinking is\n * opt-in, so 0 sends nothing and a positive budget drops `temperature` —\n * the API rejects the combination); bedrock forwards a positive budget as\n * the anthropic-style passthrough; the openai family has no portable\n * thinking control. */\n thinkingBudget?: number | null;\n temperature?: number | null;\n};\n\nexport type LLMClientOpts = {\n client?: ChatClient | null;\n fetch?: FetchImpl | null;\n fetchImpl?: FetchImpl | null;\n};\n\ntype GeminiChatClient = {\n models: {\n generateContent(args: { model: string; contents: unknown; config?: Record<string, unknown> }): Promise<{\n text?: string;\n usageMetadata?: {\n promptTokenCount?: number;\n candidatesTokenCount?: number;\n prompt_token_count?: number;\n candidates_token_count?: number;\n };\n }>;\n };\n};\n\ntype BedrockConverseResult = {\n output?: { message?: { content?: Array<{ text?: string }> } };\n usage?: { inputTokens?: number; outputTokens?: number };\n};\n\nfunction toBase64(bytes: Uint8Array): string {\n return Buffer.from(bytes).toString(\"base64\");\n}\n\nasync function postJson(\n fetchImpl: FetchImpl,\n url: string,\n body: unknown,\n headers: Record<string, string>,\n): Promise<any> {\n const resp = await fetchImpl(url, {\n method: \"POST\",\n headers: { \"content-type\": \"application/json\", ...headers },\n body: JSON.stringify(body),\n signal: AbortSignal.timeout(TIMEOUT_MS),\n });\n if (!resp.ok) {\n const text = await resp.text().catch(() => \"\");\n throw new Error(`HTTP ${resp.status} ${resp.statusText}${text ? `: ${text}` : \"\"}`);\n }\n return resp.json();\n}\n\nexport function buildOpenAIChatClient(cfg: LLMConfig): OpenAI {\n return new OpenAI({\n apiKey: cfg.apiKey ?? undefined,\n baseURL: cfg.baseUrl ?? undefined,\n timeout: TIMEOUT_MS,\n maxRetries: 0,\n });\n}\n\nexport class LLMClient {\n cfg: LLMConfig;\n provider: LLMConfig[\"provider\"];\n model: string;\n client: ChatClient | null;\n fetchImpl: FetchImpl | null;\n private genaiClient: GeminiChatClient | null = null;\n\n constructor(cfg: LLMConfig, opts: LLMClientOpts = {}) {\n this.cfg = cfg;\n this.provider = cfg.provider;\n this.model = cfg.model;\n this.client = opts.client ?? null;\n this.fetchImpl = opts.fetch ?? opts.fetchImpl ?? null;\n }\n\n async aclose(): Promise<void> {\n const closer = this.client?.close;\n if (typeof closer === \"function\") {\n await closer.call(this.client);\n }\n }\n\n async [Symbol.asyncDispose](): Promise<void> {\n await this.aclose();\n }\n\n async call(opts: LlmCallOpts): Promise<[string, TokenUsage]> {\n const {\n system,\n user,\n jsonMode = false,\n images = null,\n maxTokens = null,\n thinkingBudget = null,\n temperature = null,\n } = opts;\n if (this.provider === \"anthropic\") {\n return this.callAnthropic(system, user, jsonMode, images, maxTokens, thinkingBudget, temperature);\n }\n if (OPENAI_FAMILY.has(this.provider)) {\n return this.callOpenAI(system, user, jsonMode, images, maxTokens, temperature);\n }\n if (GOOGLE_FAMILY.has(this.provider)) {\n return this.callGemini(system, user, jsonMode, images, maxTokens, thinkingBudget, temperature);\n }\n if (this.provider === \"bedrock\") {\n return this.callBedrock(system, user, jsonMode, images, maxTokens, thinkingBudget, temperature);\n }\n throw new Error(`unknown llm provider: ${JSON.stringify(this.provider)}`);\n }\n\n private async callAnthropic(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n thinkingBudget: number | null = null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n if (!this.fetchImpl) {\n throw new Error(\"anthropic llm client has no fetch implementation\");\n }\n if (jsonMode) {\n system = `${system}\\n\\nRespond with valid JSON only.`;\n }\n const content: Array<Record<string, unknown>> = [];\n for (const image of images ?? []) {\n content.push({\n type: \"image\",\n source: {\n type: \"base64\",\n media_type: \"image/png\",\n data: toBase64(image),\n },\n });\n }\n content.push({ type: \"text\", text: user });\n\n const body: Record<string, unknown> = {\n model: this.model,\n max_tokens: maxTokens ?? ANTHROPIC_MAX_TOKENS,\n system,\n messages: [{ role: \"user\", content }],\n };\n if (thinkingBudget) {\n // Anthropic thinking is OPT-IN, so `thinkingBudget: 0` (the vision\n // transcription contract) correctly maps to sending nothing. The API\n // rejects a temperature alongside enabled thinking, so the budget wins\n // when a caller passes both.\n body.thinking = { type: \"enabled\", budget_tokens: thinkingBudget };\n } else if (temperature !== null) {\n // Determinism contract — the same scan must transcribe to the same\n // text twice. This was silently dropped here, so the guarantee only\n // held on gemini.\n body.temperature = temperature;\n }\n\n const data = await postJson(this.fetchImpl, \"https://api.anthropic.com/v1/messages\", body, {\n \"x-api-key\": this.cfg.apiKey ?? \"\",\n \"anthropic-version\": ANTHROPIC_VERSION,\n \"content-type\": \"application/json\",\n });\n const text = data.content[0].text as string;\n const usage = data.usage ?? {};\n return [text, { input: usage.input_tokens ?? 0, output: usage.output_tokens ?? 0 }];\n }\n\n private async callOpenAI(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n if (!this.client) {\n throw new Error(\"openai-family llm client has no client\");\n }\n const content: Array<Record<string, unknown>> = [{ type: \"text\", text: user }];\n for (const image of images ?? []) {\n content.push({\n type: \"image_url\",\n image_url: { url: `data:image/png;base64,${toBase64(image)}` },\n });\n }\n const body: {\n model: string;\n messages: unknown[];\n response_format?: { type: string };\n max_completion_tokens?: number;\n temperature?: number;\n } = {\n model: this.model,\n messages: [\n { role: \"system\", content: system },\n { role: \"user\", content },\n ],\n };\n if (jsonMode) {\n body.response_format = { type: \"json_object\" };\n }\n if (maxTokens) {\n // `max_completion_tokens`, not the deprecated `max_tokens`: the\n // reasoning models reject the old name outright.\n body.max_completion_tokens = maxTokens;\n }\n if (temperature !== null) {\n // NOTE: the o-series reasoning models reject a temperature — but a\n // clear API error beats the silent nondeterminism of dropping it.\n // (`thinkingBudget` has no portable mapping here — see LlmCallOpts.)\n body.temperature = temperature;\n }\n const resp = await this.client.chat.completions.create(body);\n const text = resp.choices[0]?.message?.content ?? \"\";\n const usage = resp.usage;\n return [text, { input: usage?.prompt_tokens ?? 0, output: usage?.completion_tokens ?? 0 }];\n }\n\n private async callGemini(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n thinkingBudget: number | null = null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n if (!this.genaiClient) {\n // `buildGenaiClient` picks API-key or Vertex/ADC mode and carries the\n // timeout: the SDK has NO default one, so a dropped connection hangs\n // the call forever — and with the bounded render queue that means one\n // hung batch holds a worker slot for good, so the document simply\n // stops with nothing to show for it.\n this.genaiClient = await buildGenaiClient<GeminiChatClient>(\n this.cfg,\n GEMINI_CALL_TIMEOUT_MS,\n \"gemini llm\",\n );\n }\n const parts: unknown[] = [];\n for (const img of images ?? []) {\n parts.push({ inlineData: { mimeType: \"image/png\", data: toBase64(img) } });\n }\n parts.push({ text: user });\n\n const config: Record<string, unknown> = {\n systemInstruction: system,\n // ALWAYS off, and not as a preference. Automatic function calling means\n // the SDK itself EXECUTES a callable it was handed as a tool and loops\n // on the result — up to ten round trips — before returning anything.\n // This package passes declarations only, so today nothing is executable;\n // but that depends on every future caller continuing to do the same, and\n // an application that gates tool execution behind human approval would\n // have that gate bypassed silently, by a library, with the loop already\n // run before it could object.\n automaticFunctionCalling: { disable: true },\n };\n if (jsonMode) {\n config.responseMimeType = \"application/json\";\n }\n if (maxTokens) {\n config.maxOutputTokens = maxTokens;\n }\n if (temperature !== null) {\n config.temperature = temperature;\n }\n if (thinkingBudget !== null) {\n // Only when the CALLER asks. Reasoning tokens come out of\n // `maxOutputTokens`, so a long transcription can otherwise spend its\n // allowance deliberating and truncate part-way through. Entity\n // extraction and structured output may legitimately want the reasoning,\n // so there is no silent default.\n config.thinkingConfig = { thinkingBudget };\n }\n\n const resp = await this.genaiClient.models.generateContent({\n model: this.model,\n contents: parts,\n config,\n });\n const text = resp.text ?? \"\";\n const usageMeta = resp.usageMetadata;\n const inputTokens = usageMeta?.promptTokenCount ?? usageMeta?.prompt_token_count ?? 0;\n const outputTokens = usageMeta?.candidatesTokenCount ?? usageMeta?.candidates_token_count ?? 0;\n return [text, { input: inputTokens || 0, output: outputTokens || 0 }];\n }\n\n private async callBedrock(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n thinkingBudget: number | null = null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n const { BedrockRuntimeClient, ConverseCommand } = await loadBedrockSdk();\n if (jsonMode) {\n system = `${system}\\n\\nRespond with valid JSON only.`;\n }\n const content: Array<Record<string, unknown>> = [];\n for (const image of images ?? []) {\n content.push({ image: { format: \"png\", source: { bytes: image } } });\n }\n content.push({ text: user });\n\n const runtime = new BedrockRuntimeClient({\n maxAttempts: 1,\n requestHandler: {\n requestTimeout: TIMEOUT_MS,\n connectionTimeout: TIMEOUT_MS,\n },\n });\n try {\n const result = (await runtime.send(\n new ConverseCommand({\n modelId: this.model,\n system: [{ text: system }],\n messages: [{ role: \"user\", content }],\n ...(maxTokens || temperature !== null\n ? {\n inferenceConfig: {\n ...(maxTokens ? { maxTokens } : {}),\n ...(temperature !== null ? { temperature } : {}),\n },\n }\n : {}),\n // Anthropic-style passthrough — Converse forwards it to the model.\n // Zero (the vision transcription contract) sends nothing: thinking\n // is opt-in for the anthropic models bedrock hosts.\n ...(thinkingBudget\n ? {\n additionalModelRequestFields: {\n thinking: { type: \"enabled\", budget_tokens: thinkingBudget },\n },\n }\n : {}),\n }),\n )) as BedrockConverseResult;\n const text = result.output?.message?.content?.[0]?.text ?? \"\";\n const usage = result.usage ?? {};\n return [text, { input: usage.inputTokens ?? 0, output: usage.outputTokens ?? 0 }];\n } finally {\n runtime.destroy?.();\n }\n }\n}\n\nasync function loadBedrockSdk(): Promise<{\n BedrockRuntimeClient: new (\n cfg: Record<string, unknown>,\n ) => {\n send(cmd: unknown): Promise<unknown>;\n destroy?: () => void;\n };\n ConverseCommand: new (input: Record<string, unknown>) => unknown;\n}> {\n const specifier = \"@aws-sdk/client-bedrock-runtime\";\n try {\n return (await import(specifier)) as Awaited<ReturnType<typeof loadBedrockSdk>>;\n } catch {\n throw new ExtraMissingError(\"bedrock\", specifier, \"bedrock llm\");\n }\n}\n\nexport function buildLlmClient(cfg: LLMConfig, opts?: LLMClientOpts): LLMClient {\n if (opts?.client || opts?.fetch || opts?.fetchImpl) {\n return new LLMClient(cfg, opts);\n }\n if (cfg.provider === \"anthropic\") {\n return new LLMClient(cfg, { fetch: globalThis.fetch });\n }\n if (OPENAI_FAMILY.has(cfg.provider)) {\n return new LLMClient(cfg, { client: buildOpenAIChatClient(cfg) });\n }\n if (GOOGLE_FAMILY.has(cfg.provider) || cfg.provider === \"bedrock\") {\n return new LLMClient(cfg);\n }\n throw new Error(`unknown llm provider: ${JSON.stringify(cfg.provider)}`);\n}\n\n/**\n * Call the LLM configured by `cfg`. Returns `[text, tokenUsage]`.\n *\n * With no `client`, one is built for this call and CLOSED afterwards,\n * success or failure. Pass `client` to reuse a connection across many calls;\n * the caller then owns `aclose()`.\n */\nexport async function callLlm(\n cfg: LLMConfig,\n opts: LlmCallOpts & { client?: LLMClient | null },\n): Promise<[string, TokenUsage]> {\n const { client, ...callOpts } = opts;\n if (client) {\n return client.call(callOpts);\n }\n const owned = buildLlmClient(cfg);\n try {\n return await owned.call(callOpts);\n } finally {\n await owned.aclose();\n }\n}\n","import { createCipheriv, createDecipheriv, createHmac, randomBytes } from \"node:crypto\";\n\n/**\n * AES-256-GCM crypto seam. Wire format is a compatibility promise with the\n * Python library and promptev-connectors:\n * base64(nonce[12] || AES-256-GCM ciphertext+tag)\n */\n\nexport function encryptDict(data: Record<string, unknown>, key: Buffer): string {\n const nonce = randomBytes(12);\n const cipher = createCipheriv(\"aes-256-gcm\", key, nonce);\n const plaintext = Buffer.from(JSON.stringify(data), \"utf8\");\n const ciphertext = Buffer.concat([cipher.update(plaintext), cipher.final()]);\n const tag = cipher.getAuthTag();\n return Buffer.concat([nonce, ciphertext, tag]).toString(\"base64\");\n}\n\nexport function decryptDict(token: string, key: Buffer): Record<string, unknown> {\n const raw = Buffer.from(token, \"base64\");\n const nonce = raw.subarray(0, 12);\n const tag = raw.subarray(raw.length - 16);\n const ciphertext = raw.subarray(12, raw.length - 16);\n const decipher = createDecipheriv(\"aes-256-gcm\", key, nonce);\n decipher.setAuthTag(tag);\n const plaintext = Buffer.concat([decipher.update(ciphertext), decipher.final()]);\n return JSON.parse(plaintext.toString(\"utf8\")) as Record<string, unknown>;\n}\n\nexport function getSecretKey(config: { secretKey?: string | null }): Buffer {\n const secretKey = config.secretKey;\n if (!secretKey) {\n throw new Error(\n \"ContextEngineConfig.secretKey (env CE_SECRET_KEY) is not configured — \" +\n \"a base64url-encoded 32-byte AES key is required to encrypt/decrypt \" +\n \"tool configs that hold secrets.\",\n );\n }\n return Buffer.from(secretKey, \"base64url\");\n}\n\nexport function hmacSha256Hex(key: string | Buffer, value: string): string {\n const k = typeof key === \"string\" ? Buffer.from(key) : key;\n return createHmac(\"sha256\", k).update(value, \"utf8\").digest(\"hex\");\n}\n","import { hmacSha256Hex } from \"./crypto.js\";\nimport { emitError, type Hooks } from \"./hooks.js\";\n\nexport type Span = [number, number];\nexport type DetectorFn = (text: string) => Span[];\nexport type RedactionAction = \"mask\" | \"hash\" | \"remove\";\nexport type ApplyAt = \"ingest\" | \"output\" | \"both\";\nexport type RedactionPhase = \"ingest\" | \"output\";\n\nexport const BUILTIN_DETECTOR_NAMES = new Set([\"email\", \"phone\", \"ssn\", \"credit_card\", \"iban\", \"api_key\"]);\n\nexport interface RedactionRuleInit {\n name: string;\n detector?: string | null;\n pattern?: string | null;\n field?: string | null;\n action?: RedactionAction;\n placeholder?: string | null;\n applyAt?: ApplyAt;\n unless?: string[];\n}\n\nexport class RedactionRule {\n name: string;\n detector: string | null;\n pattern: string | null;\n field: string | null;\n action: RedactionAction;\n placeholder: string | null;\n applyAt: ApplyAt;\n unless: string[];\n private compiled: RegExp | null = null;\n\n constructor(init: RedactionRuleInit) {\n this.name = init.name;\n this.detector = init.detector ?? null;\n this.pattern = init.pattern ?? null;\n this.field = init.field ?? null;\n this.action = init.action ?? \"mask\";\n this.placeholder = init.placeholder ?? null;\n this.applyAt = init.applyAt ?? \"output\";\n this.unless = init.unless ?? [];\n this.validate();\n }\n\n private validate(): void {\n if (this.field !== null) {\n throw new Error(\n `rule '${this.name}': field-targeted rules are not implemented yet; use detector or pattern instead`,\n );\n }\n const targets = [this.detector, this.pattern].filter((t) => t !== null);\n if (targets.length !== 1) {\n throw new Error(`rule '${this.name}': exactly one of detector/pattern must be set`);\n }\n if (this.pattern !== null) {\n try {\n this.compiled = new RegExp(this.pattern, \"g\");\n } catch (exc) {\n throw new Error(`rule '${this.name}': invalid regex: ${exc}`);\n }\n }\n if (this.unless.length && this.applyAt === \"ingest\") {\n throw new Error(\n `rule '${this.name}': unless is output-time only and cannot be set on an applyAt='ingest' rule`,\n );\n }\n }\n\n patternRe(): RegExp | null {\n return this.compiled;\n }\n\n effectivePlaceholder(): string {\n return this.placeholder || `[${this.name.toUpperCase()}]`;\n }\n}\n\nexport interface RedactionPolicyInit {\n rules?: RedactionRule[] | RedactionRuleInit[];\n customDetectors?: Record<string, DetectorFn>;\n}\n\nexport class RedactionPolicy {\n rules: RedactionRule[];\n customDetectors: Record<string, DetectorFn>;\n\n constructor(init: RedactionPolicyInit = {}) {\n this.rules = (init.rules ?? []).map((r) => (r instanceof RedactionRule ? r : new RedactionRule(r)));\n this.customDetectors = init.customDetectors ?? {};\n const seen = new Set<string>();\n for (const rule of this.rules) {\n if (seen.has(rule.name)) throw new Error(`duplicate rule name: '${rule.name}'`);\n seen.add(rule.name);\n if (rule.detector !== null) {\n const known = BUILTIN_DETECTOR_NAMES.has(rule.detector) || rule.detector in this.customDetectors;\n if (!known) {\n throw new Error(\n `rule '${rule.name}': unknown detector '${rule.detector}' (not built-in and not in customDetectors)`,\n );\n }\n }\n }\n }\n\n isEmpty(): boolean {\n return this.rules.length === 0;\n }\n}\n\nconst EMAIL_RE = /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\\.[A-Za-z]{2,}/g;\nconst PHONE_RE = /\\+\\d[\\d\\s-]{7,17}\\d/g;\nconst SSN_RE = /(?<!\\d)\\d{3}-\\d{2}-\\d{4}(?!\\d)/g;\nconst CARD_RE = /(?<!\\d)(?:\\d[ -]?){12,18}\\d(?!\\d)/g;\nconst IBAN_RE = /(?<![A-Za-z0-9])[A-Z]{2}\\d{2}[A-Z0-9]{10,30}(?![A-Za-z0-9])/g;\nconst API_KEY_ALNUM = \"A-Za-z0-9_\\\\-+/=\";\nconst API_KEY_PREFIX_RE = new RegExp(\n `(?<![${API_KEY_ALNUM}])(?:(?:AKIA|ASIA)[0-9A-Z]{16}|sk-[A-Za-z0-9]{20,}|gh[opsu]_[A-Za-z0-9]{20,}|xox[baprs]-[A-Za-z0-9\\\\-]{10,}|AIza[0-9A-Za-z_\\\\-]{35})(?![${API_KEY_ALNUM}])`,\n \"g\",\n);\nconst API_KEY_GENERIC_RE = new RegExp(\n `(?<![${API_KEY_ALNUM}])[${API_KEY_ALNUM}]{24,}(?![${API_KEY_ALNUM}])`,\n \"g\",\n);\nconst UUID_RE = /^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$/;\n\nfunction spansFrom(re: RegExp, text: string): Span[] {\n const out: Span[] = [];\n re.lastIndex = 0;\n let m = re.exec(text);\n while (m !== null) {\n out.push([m.index, m.index + m[0].length]);\n if (m[0].length === 0) re.lastIndex++;\n m = re.exec(text);\n }\n return out;\n}\n\nfunction luhnOk(digits: string): boolean {\n let total = 0;\n const rev = [...digits].reverse();\n for (let i = 0; i < rev.length; i++) {\n let d = Number(rev[i]);\n if (i % 2 === 1) {\n d *= 2;\n if (d > 9) d -= 9;\n }\n total += d;\n }\n return total % 10 === 0;\n}\n\nfunction detectCreditCard(text: string): Span[] {\n const spans: Span[] = [];\n for (const [start, end] of spansFrom(CARD_RE, text)) {\n const digits = text.slice(start, end).replace(/[ -]/g, \"\");\n if (digits.length >= 13 && digits.length <= 19 && luhnOk(digits)) {\n spans.push([start, end]);\n }\n }\n return spans;\n}\n\nfunction looksLikeGenericSecret(token: string): boolean {\n if (UUID_RE.test(token)) return false;\n let hasUpper = false;\n let hasLower = false;\n let hasDigit = false;\n for (const c of token) {\n if (c >= \"A\" && c <= \"Z\") hasUpper = true;\n else if (c >= \"a\" && c <= \"z\") hasLower = true;\n else if (c >= \"0\" && c <= \"9\") hasDigit = true;\n }\n return hasUpper && hasLower && hasDigit;\n}\n\nfunction detectApiKey(text: string): Span[] {\n const spans = spansFrom(API_KEY_PREFIX_RE, text);\n for (const [start, end] of spansFrom(API_KEY_GENERIC_RE, text)) {\n if (spans.some(([s, e]) => start < e && s < end)) continue;\n if (looksLikeGenericSecret(text.slice(start, end))) spans.push([start, end]);\n }\n return spans;\n}\n\nconst BUILTIN: Record<string, DetectorFn> = {\n email: (t) => spansFrom(EMAIL_RE, t),\n phone: (t) => spansFrom(PHONE_RE, t),\n ssn: (t) => spansFrom(SSN_RE, t),\n credit_card: detectCreditCard,\n iban: (t) => spansFrom(IBAN_RE, t),\n api_key: detectApiKey,\n};\n\nexport function detectBuiltin(name: string, text: string): Span[] {\n const detector = BUILTIN[name];\n if (!detector) throw new Error(`unknown built-in detector: ${name}`);\n if (!text) return [];\n return detector(text).sort((a, b) => a[0] - b[0]);\n}\n\nexport function ruleApplies(\n rule: RedactionRule,\n opts: { phase: RedactionPhase; principals: readonly string[] | null },\n): boolean {\n if (rule.applyAt !== \"both\" && rule.applyAt !== opts.phase) return false;\n if (opts.phase === \"ingest\") return true;\n if (!rule.unless.length) return true;\n if (opts.principals === null) return false;\n const held = new Set(opts.principals);\n return !rule.unless.some((p) => held.has(p));\n}\n\nfunction spansForRule(rule: RedactionRule, text: string, policy: RedactionPolicy): Span[] {\n if (rule.pattern !== null) {\n const re = rule.patternRe() ?? new RegExp(rule.pattern, \"g\");\n return spansFrom(re, text);\n }\n if (rule.detector !== null) {\n const custom = policy.customDetectors[rule.detector];\n if (custom) return [...custom(text)];\n return detectBuiltin(rule.detector, text);\n }\n return [];\n}\n\nconst HASH_TOKEN_CHARS = 16;\n\nfunction hashToken(value: string, secretKey: string | Buffer | null | undefined): string {\n const key = Buffer.isBuffer(secretKey) ? secretKey : Buffer.from(String(secretKey ?? \"\"));\n return hmacSha256Hex(key, value).slice(0, HASH_TOKEN_CHARS);\n}\n\nexport interface RedactionNote {\n rules_fired?: string[];\n spans?: number;\n rules_failed?: string[];\n}\n\nexport function applyRedaction(\n text: string,\n policy: RedactionPolicy,\n opts: {\n phase: RedactionPhase;\n principals?: readonly string[] | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks | null;\n },\n): [string, RedactionNote] {\n if (!text || policy.isEmpty()) return [text, {}];\n const principals = opts.principals ?? null;\n const collected: Array<[number, number, RedactionRule]> = [];\n const fired: string[] = [];\n const failed: string[] = [];\n\n for (const rule of policy.rules) {\n if (!ruleApplies(rule, { phase: opts.phase, principals })) continue;\n if (rule.action === \"hash\" && !opts.secretKey) {\n throw new Error(\n `rule '${rule.name}': action='hash' requires a non-empty secretKey ` +\n `(an unkeyed HMAC is a reversible pseudonym, not a redaction)`,\n );\n }\n try {\n const raw = spansForRule(rule, text, policy);\n const ruleSpans: Span[] = [];\n let invalid = false;\n for (const item of raw) {\n const start = item?.[0];\n const end = item?.[1];\n if (typeof start !== \"number\" || typeof end !== \"number\") {\n invalid = true;\n continue;\n }\n if (!(start >= 0 && start < end && end <= text.length)) {\n invalid = true;\n continue;\n }\n ruleSpans.push([start, end]);\n }\n if (invalid) failed.push(rule.name);\n for (const [s, e] of ruleSpans) collected.push([s, e, rule]);\n } catch (exc) {\n failed.push(rule.name);\n if (opts.hooks) emitError(opts.hooks, exc, { stage: \"redaction\", rule: rule.name });\n }\n }\n\n if (!collected.length) {\n if (failed.length) return [text, { rules_fired: [], spans: 0, rules_failed: failed }];\n return [text, {}];\n }\n\n collected.sort((a, b) => a[0] - b[0] || b[1] - b[0] - (a[1] - a[0]));\n const merged: Array<[number, number, RedactionRule]> = [];\n for (const [start, end, rule] of collected) {\n const last = merged[merged.length - 1];\n if (last && start < last[1]) {\n if (end > last[1]) last[1] = end;\n continue;\n }\n merged.push([start, end, rule]);\n }\n\n const out: string[] = [];\n let cursor = 0;\n for (const [start, end, rule] of merged) {\n out.push(text.slice(cursor, start));\n const original = text.slice(start, end);\n if (rule.action === \"mask\") out.push(rule.effectivePlaceholder());\n else if (rule.action === \"hash\") {\n out.push(`[${rule.name.toUpperCase()}:${hashToken(original, opts.secretKey)}]`);\n }\n if (!fired.includes(rule.name)) fired.push(rule.name);\n cursor = end;\n }\n out.push(text.slice(cursor));\n\n const note: RedactionNote = { rules_fired: fired, spans: merged.length };\n if (failed.length) note.rules_failed = failed;\n return [out.join(\"\"), note];\n}\n","/**\n * Unified JSON parsing with jsonrepair fallback for LLM output.\n *\n * Handles markdown fences, truncated output, trailing commas, unclosed\n * strings/brackets, and other common LLM JSON errors.\n */\nimport { jsonrepair } from \"jsonrepair\";\n\nexport type JsonExpected = \"object\" | \"array\";\n\nfunction stripFences(text: string): string {\n let t = text.trim();\n const fenced = /^```(?:json)?\\s*([\\s\\S]*?)\\s*```$/i.exec(t);\n if (fenced?.[1]) t = fenced[1].trim();\n return t;\n}\n\nfunction isExpected(value: unknown, expectedType: JsonExpected): boolean {\n if (expectedType === \"array\") return Array.isArray(value);\n return typeof value === \"object\" && value !== null && !Array.isArray(value);\n}\n\n/**\n * Parse JSON with automatic repair fallback for LLM output.\n *\n * @throws {SyntaxError} If parsing and repair both fail to produce `expectedType`.\n */\nexport function safeJsonParse(text: string, expectedType: JsonExpected = \"object\"): unknown {\n const stripped = stripFences(text);\n\n try {\n const result: unknown = JSON.parse(stripped);\n if (isExpected(result, expectedType)) return result;\n } catch {\n /* fall through to repair */\n }\n\n let repaired: unknown;\n try {\n repaired = JSON.parse(jsonrepair(stripped));\n } catch {\n throw new SyntaxError(`Expected ${expectedType}, jsonrepair could not parse`);\n }\n if (isExpected(repaired, expectedType)) {\n console.info(`json_repair recovered ${Array.isArray(repaired) ? \"array\" : \"object\"}`);\n return repaired;\n }\n throw new SyntaxError(\n `Expected ${expectedType}, got ${Array.isArray(repaired) ? \"array\" : typeof repaired}`,\n );\n}\n","/**\n * File-format text extraction — dispatch helpers + concrete extractors for\n * the mainstream office/text formats (docx/pptx/xlsx/csv/html/eml).\n *\n * Legacy binary formats (.doc/.ppt/.xls/.msg/.odt/...) are not parsed here:\n * `extract()` falls through to a raw utf-8 decode, matching the Python\n * package's worst-case behavior.\n */\nimport { load } from \"cheerio\";\nimport ExcelJS from \"exceljs\";\nimport JSZip from \"jszip\";\nimport mammoth from \"mammoth\";\nimport PostalMime from \"postal-mime\";\n\nconst ZIP_MAGIC = Buffer.from([0x50, 0x4b, 0x03, 0x04]);\n\nexport const NON_INGESTIBLE_MEDIA_EXTS = new Set([\n \".ogg\",\n \".oga\",\n \".opus\",\n \".mp3\",\n \".wav\",\n \".m4a\",\n \".aac\",\n \".flac\",\n \".wma\",\n \".amr\",\n \".mp4\",\n \".m4v\",\n \".mov\",\n \".avi\",\n \".mkv\",\n \".webm\",\n \".wmv\",\n \".flv\",\n \".3gp\",\n \".mpg\",\n \".mpeg\",\n]);\n\nexport const DOCX_MIME = \"application/vnd.openxmlformats-officedocument.wordprocessingml.document\";\nexport const PPTX_MIME = \"application/vnd.openxmlformats-officedocument.presentationml.presentation\";\nexport const XLSX_MIME = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\";\n\nexport function getFileExtension(filename: string | null | undefined): string {\n if (!filename?.includes(\".\")) return \"\";\n return filename.slice(filename.lastIndexOf(\".\")).toLowerCase();\n}\n\n/**\n * True for audio/video files, which must never be treated as text-bearing.\n * Bare `application/octet-stream` does NOT match.\n */\nexport function isNonIngestibleMedia(\n filename: string | null | undefined,\n mime: string | null | undefined,\n): boolean {\n const m = (mime || \"\").toLowerCase();\n if (m.startsWith(\"audio/\") || m.startsWith(\"video/\")) return true;\n return NON_INGESTIBLE_MEDIA_EXTS.has(getFileExtension(filename));\n}\n\nfunction isZip(content: Buffer): boolean {\n return content.length >= 4 && content.subarray(0, 4).equals(ZIP_MAGIC);\n}\n\nfunction decodeXmlEntities(s: string): string {\n return s\n .replace(/</g, \"<\")\n .replace(/>/g, \">\")\n .replace(/"/g, '\"')\n .replace(/'/g, \"'\")\n .replace(/&/g, \"&\");\n}\n\nfunction collectTagText(xml: string, tag: string): string[] {\n const out: string[] = [];\n const re = new RegExp(`<${tag}(?:\\\\s[^>]*)?>([^<]*)</${tag}>`, \"g\");\n for (const m of xml.matchAll(re)) {\n const t = decodeXmlEntities(m[1] ?? \"\").trim();\n if (t) out.push(t);\n }\n return out;\n}\n\nfunction htmlToPlain(html: string, extraRemove = \"\"): string {\n try {\n const $ = load(html);\n $(`script, style, head${extraRemove ? `, ${extraRemove}` : \"\"}`).remove();\n $(\"br\").replaceWith(\"\\n\");\n $(\"p, div, h1, h2, h3, h4, h5, h6, tr, li, blockquote\").each((_, el) => {\n $(el).prepend(\"\\n\");\n });\n const lines = $.root()\n .text()\n .split(/\\n/)\n .map((l) => l.trim())\n .filter(Boolean);\n return lines.join(\"\\n\");\n } catch {\n return html;\n }\n}\n\nfunction mammothHtmlToText(html: string): string {\n const $ = load(html);\n $(\"table\").each((_, table) => {\n const rows: string[] = [];\n $(table)\n .find(\"tr\")\n .each((__, tr) => {\n const cells = $(tr)\n .find(\"th, td\")\n .map((___, td) => $(td).text().trim())\n .get()\n .filter(Boolean);\n if (cells.length) rows.push(cells.join(\" | \"));\n });\n $(table).replaceWith(`${rows.join(\"\\n\")}\\n`);\n });\n $(\"br\").replaceWith(\"\\n\");\n $(\"p, div, h1, h2, h3, h4, h5, h6, li\").each((_, el) => {\n $(el).prepend(\"\\n\");\n });\n const lines = $.root()\n .text()\n .split(/\\n/)\n .map((l) => l.trim())\n .filter(Boolean);\n return lines.join(\"\\n\");\n}\n\nasync function docxHeaderFooterText(content: Buffer): Promise<string[]> {\n try {\n const zip = await JSZip.loadAsync(content);\n const parts: string[] = [];\n const names = Object.keys(zip.files).filter((n) => /^word\\/(header|footer)\\d*\\.xml$/i.test(n));\n for (const name of names) {\n const xml = await zip.files[name]!.async(\"string\");\n parts.push(...collectTagText(xml, \"w:t\"));\n }\n return parts;\n } catch {\n return [];\n }\n}\n\nexport async function extractDocxText(content: Buffer): Promise<string> {\n if (!isZip(content)) {\n console.warn(\"extractDocxText: not a ZIP archive, decoding as plain text\");\n return content.toString(\"utf8\");\n }\n const { value } = await mammoth.convertToHtml({ buffer: content });\n const body = mammothHtmlToText(value || \"\");\n const extras = await docxHeaderFooterText(content);\n const parts = [body, ...extras].filter((p) => p.trim());\n return parts.join(\"\\n\");\n}\n\nexport async function countDocxPages(content: Buffer): Promise<number> {\n try {\n const { value } = await mammoth.extractRawText({ buffer: content });\n const wordCount = (value || \"\").trim() ? (value || \"\").trim().split(/\\s+/).length : 0;\n return Math.max(1, Math.round(wordCount / 250));\n } catch {\n return 1;\n }\n}\n\nfunction cellString(value: unknown): string {\n if (value == null) return \"\";\n if (value instanceof Date) return value.toISOString();\n if (typeof value === \"object\") {\n const rec = value as { result?: unknown; text?: unknown; richText?: Array<{ text?: string }> };\n if (\"result\" in rec) return cellString(rec.result);\n if (Array.isArray(rec.richText)) return rec.richText.map((t) => t.text ?? \"\").join(\"\");\n if (typeof rec.text === \"string\") return rec.text;\n }\n return String(value);\n}\n\nexport async function extractXlsxText(content: Buffer): Promise<string> {\n if (!isZip(content)) {\n console.warn(\"extractXlsxText: not a ZIP archive, decoding as plain text\");\n return content.toString(\"utf8\");\n }\n try {\n const probe = new ExcelJS.Workbook();\n await probe.xlsx.load(content as unknown as ArrayBuffer);\n const colCaps = new Map<string, number>();\n for (const ws of probe.worksheets) {\n const widths: number[] = [];\n let i = 0;\n ws.eachRow({ includeEmpty: true }, (row) => {\n if (i === 0) {\n i += 1;\n return;\n }\n if (i > 20) return;\n i += 1;\n const values = (row.values as unknown[]) ?? [];\n let lastData = 0;\n for (let j = 1; j < values.length; j++) {\n const v = values[j];\n if (v != null && cellString(v).trim()) lastData = j;\n }\n if (lastData > 0) widths.push(lastData);\n });\n colCaps.set(ws.name, widths.length ? Math.max(...widths) : 50);\n }\n\n const wb = new ExcelJS.Workbook();\n await wb.xlsx.load(content as unknown as ArrayBuffer);\n const out: string[] = [];\n for (const ws of wb.worksheets) {\n out.push(`[Sheet: ${ws.name}]`);\n const cap = colCaps.get(ws.name) ?? 50;\n ws.eachRow({ includeEmpty: true }, (row) => {\n const values = (row.values as unknown[]) ?? [];\n const trimmed: string[] = [];\n const last = Math.min(values.length - 1, cap);\n for (let j = 1; j <= last; j++) trimmed.push(cellString(values[j]));\n while (trimmed.length && !trimmed[trimmed.length - 1]!.trim()) trimmed.pop();\n if (trimmed.length) out.push(trimmed.join(\", \"));\n });\n }\n return out.join(\"\\n\");\n } catch (err) {\n const msg = err instanceof Error ? err.message : String(err);\n if (/zip|corrupt|invalid/i.test(msg)) return content.toString(\"utf8\");\n throw err;\n }\n}\n\nfunction pptxTableRows(tblXml: string): string[] {\n const rows: string[] = [];\n for (const tr of tblXml.matchAll(/<a:tr\\b[\\s\\S]*?<\\/a:tr>/g)) {\n const cells: string[] = [];\n for (const tc of tr[0]!.matchAll(/<a:tc\\b[\\s\\S]*?<\\/a:tc>/g)) {\n const texts = collectTagText(tc[0]!, \"a:t\");\n const joined = texts.join(\" \").trim();\n if (joined) cells.push(joined);\n }\n if (cells.length) rows.push(cells.join(\" | \"));\n }\n return rows;\n}\n\nfunction pptxSlideBody(xml: string): string[] {\n const parts: string[] = [];\n const withoutTables = xml.replace(/<a:tbl\\b[\\s\\S]*?<\\/a:tbl>/g, (tbl) => {\n parts.push(...pptxTableRows(tbl));\n return \"\";\n });\n parts.push(...collectTagText(withoutTables, \"a:t\"));\n return parts.filter(Boolean);\n}\n\nfunction parseRels(xml: string): Map<string, string> {\n const map = new Map<string, string>();\n const re = /Id=\"([^\"]+)\"[^>]*Target=\"([^\"]+)\"|Target=\"([^\"]+)\"[^>]*Id=\"([^\"]+)\"/g;\n for (const m of xml.matchAll(re)) {\n const id = m[1] || m[4];\n const target = m[2] || m[3];\n if (id && target) map.set(id, target);\n }\n return map;\n}\n\nasync function pptxSlideOrder(zip: JSZip): Promise<string[]> {\n const pres = zip.file(\"ppt/presentation.xml\");\n const rels = zip.file(\"ppt/_rels/presentation.xml.rels\");\n if (!pres || !rels) {\n return Object.keys(zip.files)\n .filter((n) => /^ppt\\/slides\\/slide\\d+\\.xml$/.test(n))\n .sort((a, b) => {\n const na = Number(/slide(\\d+)/.exec(a)?.[1] ?? 0);\n const nb = Number(/slide(\\d+)/.exec(b)?.[1] ?? 0);\n return na - nb;\n });\n }\n const presXml = await pres.async(\"string\");\n const relsMap = parseRels(await rels.async(\"string\"));\n const ids: string[] = [];\n for (const m of presXml.matchAll(/<p:sldId\\b[^>]*r:id=\"([^\"]+)\"/g)) {\n ids.push(m[1]!);\n }\n const names: string[] = [];\n for (const id of ids) {\n const target = relsMap.get(id);\n if (!target) continue;\n const path = target.replace(/^\\.\\//, \"\");\n names.push(path.startsWith(\"ppt/\") ? path : `ppt/${path}`);\n }\n return names;\n}\n\nexport async function extractPptxText(content: Buffer): Promise<string> {\n if (!isZip(content)) {\n console.warn(\"extractPptxText: not a ZIP archive, decoding as plain text\");\n return content.toString(\"utf8\");\n }\n const zip = await JSZip.loadAsync(content);\n const slideFiles = await pptxSlideOrder(zip);\n const parts: string[] = [];\n for (let i = 0; i < slideFiles.length; i++) {\n const name = slideFiles[i]!;\n const file = zip.file(name);\n if (!file) continue;\n parts.push(`[Slide ${i + 1}]`);\n const xml = await file.async(\"string\");\n parts.push(...pptxSlideBody(xml));\n const relsName = name.replace(/slides\\/(slide\\d+\\.xml)$/, \"slides/_rels/$1.rels\");\n const relsFile = zip.file(relsName);\n if (relsFile) {\n const relsXml = await relsFile.async(\"string\");\n const notesTarget = [...relsXml.matchAll(/Target=\"([^\"]*notesSlide[^\"]*)\"/gi)][0]?.[1];\n if (notesTarget) {\n const notesPath = notesTarget.startsWith(\"/\")\n ? notesTarget.slice(1)\n : name.replace(/slides\\/slide\\d+\\.xml$/, \"\") +\n notesTarget.replace(/^\\.\\.\\//, \"\").replace(/^\\.\\//, \"\");\n const notesFile = zip.file(notesPath) ?? zip.file(`ppt/${notesTarget.replace(/^(\\.\\.\\/)+/, \"\")}`);\n if (notesFile) {\n const notesXml = await notesFile.async(\"string\");\n const notes = collectTagText(notesXml, \"a:t\").join(\" \").trim();\n if (notes) parts.push(`[Notes] ${notes}`);\n }\n }\n }\n parts.push(\"\");\n }\n return parts.join(\"\\n\");\n}\n\nexport async function countPptxSlides(content: Buffer): Promise<number> {\n try {\n const zip = await JSZip.loadAsync(content);\n const n = (await pptxSlideOrder(zip)).length;\n return n || 1;\n } catch {\n return 1;\n }\n}\n\nexport function extractCsvText(content: Buffer, delimiter = \",\"): string {\n const txt = content.toString(\"utf8\");\n try {\n const rows = parseCsv(txt, delimiter);\n const out: string[] = [];\n for (const row of rows) {\n const trimmed = [...row];\n while (trimmed.length && !String(trimmed[trimmed.length - 1]).trim()) trimmed.pop();\n if (trimmed.length) out.push(trimmed.map((c) => String(c)).join(\", \"));\n }\n return out.join(\"\\n\");\n } catch {\n return txt;\n }\n}\n\nfunction parseCsv(text: string, delimiter: string): string[][] {\n const rows: string[][] = [];\n let row: string[] = [];\n let cell = \"\";\n let i = 0;\n let inQuotes = false;\n while (i < text.length) {\n const ch = text[i]!;\n if (inQuotes) {\n if (ch === '\"') {\n if (text[i + 1] === '\"') {\n cell += '\"';\n i += 2;\n continue;\n }\n inQuotes = false;\n i += 1;\n continue;\n }\n cell += ch;\n i += 1;\n continue;\n }\n if (ch === '\"') {\n inQuotes = true;\n i += 1;\n continue;\n }\n if (ch === delimiter) {\n row.push(cell);\n cell = \"\";\n i += 1;\n continue;\n }\n if (ch === \"\\r\") {\n i += 1;\n continue;\n }\n if (ch === \"\\n\") {\n row.push(cell);\n rows.push(row);\n row = [];\n cell = \"\";\n i += 1;\n continue;\n }\n cell += ch;\n i += 1;\n }\n if (cell.length || row.length) {\n row.push(cell);\n rows.push(row);\n }\n return rows;\n}\n\nexport function extractHtmlText(content: Buffer): string {\n try {\n return htmlToPlain(content.toString(\"utf8\"));\n } catch (exc) {\n console.debug(\"HTML extraction failed:\", exc);\n return content.toString(\"utf8\");\n }\n}\n\nfunction formatAddress(\n addr:\n | { name?: string; address?: string; text?: string }\n | Array<{ name?: string; address?: string; text?: string }>\n | undefined,\n): string {\n if (!addr) return \"\";\n const one = (a: { name?: string; address?: string; text?: string }) =>\n a.text || [a.name, a.address].filter(Boolean).join(\" \") || \"\";\n return Array.isArray(addr) ? addr.map(one).filter(Boolean).join(\", \") : one(addr);\n}\n\nexport async function extractEmlText(content: Buffer): Promise<string> {\n try {\n const msg = await PostalMime.parse(content);\n const parts: string[] = [];\n const from = formatAddress(msg.from);\n const to = formatAddress(msg.to);\n if (from) parts.push(`From: ${from}`);\n if (to) parts.push(`To: ${to}`);\n if (msg.subject) parts.push(`Subject: ${msg.subject}`);\n const date = (msg as { date?: string }).date;\n if (date) parts.push(`Date: ${date}`);\n parts.push(\"\");\n if (msg.text) {\n parts.push(msg.text);\n } else if (msg.html) {\n parts.push(htmlToPlain(msg.html, \"head\"));\n }\n return parts.join(\"\\n\");\n } catch (exc) {\n console.debug(\"EML extraction failed:\", exc);\n return content.toString(\"utf8\");\n }\n}\n","/**\n * File -> text extraction. `extract()` is the single public entry point.\n *\n * Never throws for extraction failures internal to a single file — those are\n * reported via `hooks.onError` and degrade to whatever partial text was recovered.\n */\nimport type { ExtractionConfig, LLMConfig } from \"../config.js\";\nimport { tryImport } from \"../extras.js\";\nimport { emitError, type Hooks } from \"../hooks.js\";\nimport * as files from \"./files.js\";\n\nconst IMAGE_EXTS = new Set([\".png\", \".jpg\", \".jpeg\", \".gif\", \".bmp\", \".tiff\", \".tif\", \".webp\"]);\n\n/**\n * WHY a document came back with less text than it has, when the answer is\n * actionable. Mirrors the Python client's `Extracted.unreadable_reason` —\n * that docstring is the canonical rationale; only the value semantics are\n * restated here:\n *\n * - `needs_vision` — pages had no usable text layer and no vision model was\n * configured. FIXABLE by configuring one (the OCR fallback may have tried\n * and read nothing; a vision model is still the actionable upgrade).\n * - `vision_failed` — a vision model was configured and produced nothing for\n * those pages. An ops problem, not a configuration one.\n *\n * `null` means nothing was lost — a page a reader READ and found empty is\n * blank, not unreadable. A closed set on purpose (callers branch on it), and\n * NOT an error channel; `mediaOnly` stays set alongside it for whole files.\n */\nexport type UnreadableReason = \"needs_vision\" | \"vision_failed\";\n\nexport class Extracted {\n text: string;\n pages: number | null;\n slides: number | null;\n mediaOnly: boolean;\n providerTokens: Record<string, number>;\n llmPictures: number;\n isMarkdown: boolean;\n /** See `UnreadableReason`. `mediaOnly` is the same idea for whole files and\n * stays set alongside this, so existing callers keep working. */\n unreadableReason: UnreadableReason | null;\n unreadablePages: number;\n\n constructor(init: {\n text: string;\n pages?: number | null;\n slides?: number | null;\n mediaOnly?: boolean;\n providerTokens?: Record<string, number>;\n llmPictures?: number;\n isMarkdown?: boolean;\n unreadableReason?: UnreadableReason | null;\n unreadablePages?: number;\n }) {\n this.text = init.text;\n this.pages = init.pages ?? null;\n this.slides = init.slides ?? null;\n this.mediaOnly = init.mediaOnly ?? false;\n this.providerTokens = init.providerTokens ?? {};\n this.llmPictures = init.llmPictures ?? 0;\n this.isMarkdown = init.isMarkdown ?? false;\n this.unreadableReason = init.unreadableReason ?? null;\n this.unreadablePages = init.unreadablePages ?? 0;\n }\n}\n\nfunction isPdf(ext: string, mime: string): boolean {\n return ext === \".pdf\" || mime === \"application/pdf\";\n}\n\nfunction isImage(ext: string, mime: string): boolean {\n return mime.startsWith(\"image/\") || IMAGE_EXTS.has(ext);\n}\n\n// Embedded DOCX/PPTX pictures below either floor are logos, bullets and\n// dividers — a vision call would cost credits to describe nothing.\nconst MIN_EMBEDDED_IMAGE_BYTES = 3 * 1024;\nconst MIN_EMBEDDED_IMAGE_PX = 100;\n\n/**\n * Reason for an image-only Office document that produced no text — the\n * DOCX/PPTX twins of the PDF cases: a screenshot deck with no vision model\n * used to ingest as an empty success. Only images past the negligibility\n * floors count — a letterhead logo in an otherwise empty file is not\n * unreadable content.\n */\nasync function officeUnreadable(\n text: string,\n images: Array<{ imageBytes: Buffer }>,\n visionLlm: LLMConfig | null,\n): Promise<{ reason: UnreadableReason | null; unread: number }> {\n if (text.trim()) return { reason: null, unread: 0 };\n let readable = 0;\n for (const img of images) {\n if (!(await embeddedImageIsNegligible(img.imageBytes))) readable += 1;\n }\n if (!readable) return { reason: null, unread: 0 };\n return { reason: visionLlm ? \"vision_failed\" : \"needs_vision\", unread: readable };\n}\n\n/**\n * Too small to carry content. Anything undecodable (or without the optional\n * canvas to decode with) is NOT negligible past the byte floor — an\n * unreadable image is the provider's business to reject.\n */\nasync function embeddedImageIsNegligible(data: Buffer): Promise<boolean> {\n if (data.length < MIN_EMBEDDED_IMAGE_BYTES) return true;\n const canvas = await tryImport<{\n loadImage: (src: Buffer) => Promise<{ width: number; height: number }>;\n }>(\"@napi-rs/canvas\");\n if (!canvas) return false;\n try {\n const img = await canvas.loadImage(data);\n return Math.max(img.width, img.height) < MIN_EMBEDDED_IMAGE_PX;\n } catch {\n return false;\n }\n}\n\n/**\n * Run embedded DOCX/PPTX pictures through the vision LLM. Never throws.\n *\n * The returned list is always the same length as `images` and positionally\n * aligned (\"\" for skipped pictures) — callers enumerate it to label\n * `[Image N]` / `[Slide N Image]`, so a shift would mislabel every one.\n * Negligible pictures are skipped and identical bytes are read once (a\n * deck's per-slide logo is one vision call, not forty), with the result\n * fanned back to every occurrence.\n */\nexport async function extractEmbeddedImagesText(\n images: Array<{ imageBytes: Buffer }>,\n visionLlm: LLMConfig,\n hooks: Hooks,\n extraction?: ExtractionConfig | null,\n): Promise<[string[], Record<string, number>]> {\n if (!images.length) return [[], {}];\n try {\n const vision = await import(\"./vision.js\");\n const { createHash } = await import(\"node:crypto\");\n\n const keys: Array<string | null> = [];\n const firstAt = new Map<string, number>(); // digest -> index into `send`\n const send: Buffer[] = [];\n for (const img of images) {\n const blob = img.imageBytes;\n if (await embeddedImageIsNegligible(blob)) {\n keys.push(null);\n continue;\n }\n const digest = createHash(\"sha256\").update(blob).digest(\"hex\");\n if (!firstAt.has(digest)) {\n firstAt.set(digest, send.length);\n send.push(blob);\n }\n keys.push(digest);\n }\n\n if (!send.length) return [images.map(() => \"\"), {}];\n\n const [texts, tokens] = await vision.extractTextFromImages(send, { visionLlm, extraction });\n return [keys.map((k) => (k == null ? \"\" : (texts[firstAt.get(k)!] ?? \"\"))), tokens];\n } catch (exc) {\n emitError(hooks, exc, { stage: \"embedded_image_vision\" });\n return [[], {}];\n }\n}\n\n/**\n * Extract text from `content` (raw file bytes). Never throws for per-file\n * extraction failures — those go through `hooks.onError` and degrade.\n */\nexport async function extract(\n content: Buffer,\n filename: string,\n mime: string | null,\n opts: { visionLlm?: LLMConfig | null; hooks: Hooks; extraction?: ExtractionConfig | null },\n): Promise<Extracted> {\n const m = mime || \"\";\n const ext = files.getFileExtension(filename);\n const visionLlm = opts.visionLlm ?? null;\n const hooks = opts.hooks;\n const extraction = opts.extraction ?? null;\n\n try {\n if (files.isNonIngestibleMedia(filename, m)) {\n return new Extracted({ text: \"\", mediaOnly: true });\n }\n\n if (isPdf(ext, m)) {\n const pdf = await import(\"./pdf.js\");\n return await pdf.extract(content, { visionLlm, hooks, extraction });\n }\n\n if (isImage(ext, m)) {\n if (!visionLlm) {\n return new Extracted({\n text: \"\",\n mediaOnly: true,\n unreadableReason: \"needs_vision\",\n unreadablePages: 1,\n });\n }\n try {\n const vision = await import(\"./vision.js\");\n return await vision.extractImage(content, { visionLlm, extraction });\n } catch (exc) {\n emitError(hooks, exc, { stage: \"image_vision\", filename });\n // A reader WAS configured and it blew up — a different answer from\n // \"no reader\", and a different fix.\n return new Extracted({\n text: \"\",\n mediaOnly: true,\n unreadableReason: \"vision_failed\",\n unreadablePages: 1,\n });\n }\n }\n\n if (ext === \".docx\" || m === files.DOCX_MIME) {\n const text = await files.extractDocxText(content);\n let providerTokens: Record<string, number> = {};\n let llmPictures = 0;\n let combined = text;\n let images: Array<{ imageBytes: Buffer }> = [];\n if (visionLlm) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractDocxImages(content);\n const [texts, tokens] = await extractEmbeddedImagesText(images, visionLlm, hooks, extraction);\n for (let i = 0; i < texts.length; i++) {\n const imgText = texts[i];\n if (imgText) {\n combined += `\\n[Image ${i + 1}]\\n${imgText}`;\n llmPictures += 1;\n }\n }\n providerTokens = tokens;\n } else if (!text.trim()) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractDocxImages(content);\n }\n const { reason, unread } = await officeUnreadable(combined, images, visionLlm);\n return new Extracted({\n text: combined,\n pages: await files.countDocxPages(content),\n providerTokens,\n llmPictures,\n unreadableReason: reason,\n unreadablePages: unread,\n });\n }\n\n if (ext === \".pptx\" || m === files.PPTX_MIME) {\n const text = await files.extractPptxText(content);\n let providerTokens: Record<string, number> = {};\n let llmPictures = 0;\n let combined = text;\n let images: Array<{ imageBytes: Buffer; slide?: number }> = [];\n if (visionLlm) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractPptxImages(content);\n const [texts, tokens] = await extractEmbeddedImagesText(images, visionLlm, hooks, extraction);\n for (let i = 0; i < texts.length; i++) {\n const imgText = texts[i];\n if (imgText) {\n const slideNum = images[i]?.slide ?? i + 1;\n combined += `\\n[Slide ${slideNum} Image]\\n${imgText}`;\n llmPictures += 1;\n }\n }\n providerTokens = tokens;\n } else if (!text.trim()) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractPptxImages(content);\n }\n const { reason, unread } = await officeUnreadable(combined, images, visionLlm);\n return new Extracted({\n text: combined,\n slides: await files.countPptxSlides(content),\n providerTokens,\n llmPictures,\n unreadableReason: reason,\n unreadablePages: unread,\n });\n }\n\n if (ext === \".xlsx\" || m === files.XLSX_MIME) {\n return new Extracted({ text: await files.extractXlsxText(content) });\n }\n\n if (ext === \".tsv\" || m === \"text/tab-separated-values\") {\n return new Extracted({ text: files.extractCsvText(content, \"\\t\") });\n }\n\n if (ext === \".csv\" || m === \"text/csv\") {\n return new Extracted({ text: files.extractCsvText(content) });\n }\n\n if (ext === \".html\" || ext === \".htm\" || m === \"text/html\") {\n return new Extracted({ text: files.extractHtmlText(content) });\n }\n\n if (ext === \".eml\" || m === \"message/rfc822\") {\n return new Extracted({ text: await files.extractEmlText(content) });\n }\n\n return new Extracted({ text: content.toString(\"utf8\") });\n } catch (exc) {\n emitError(hooks, exc, { stage: \"extract\", filename });\n return new Extracted({ text: content.toString(\"utf8\") });\n }\n}\n","import type { Pool } from \"pg\";\nimport type { ContextEngineConfig } from \"../config.js\";\nimport { emitError, type Hooks } from \"../hooks.js\";\nimport type { Embedder } from \"../providers/embeddings.js\";\nimport type { AnnTunable } from \"../storage.js\";\nimport type { GraphStore } from \"./neo4j-client.js\";\n\nconst DEFAULT_WEIGHTS = {\n vector_score: 0.3,\n entity_match: 0.3,\n relationship_relevance: 0.2,\n community_match: 0.1,\n graph_connectivity: 0.1,\n};\n\nconst SEED_LIMIT = 200;\nconst ENTITY_LIMIT = 30;\nconst TOP_SEEDS_FOR_ENTITIES = 20;\n\n// The scope predicate. Textually a copy of storage.ts's SCOPE — which is the\n// canonical one — because these queries number their placeholders differently\n// ($2/$3/$4 here, $1/$2/$3 there) and pg binds are POSITIONAL, so the string\n// cannot be shared without renumbering every query on one side. Any change to\n// the canonical predicate must be made here too. Never widen it.\n//\n// $3 is cast to uuid[] so the predicate can use the document_id index —\n// same reasoning as storage.ts's SCOPE.\nconst SCOPE = `\n AND ($2::text[] IS NULL OR c.source_id = ANY($2::text[]))\n AND ($3::uuid[] IS NULL OR c.document_id = ANY($3::uuid[]))\n AND ($4::text[] IS NULL OR c.acl IS NULL OR c.acl && $4::text[])\n`;\n\nfunction vecLiteral(vector: number[]): string {\n return `[${vector.map((x) => Number(x)).join(\",\")}]`;\n}\n\nexport async function corpusIsAclUniform(pool: Pool, sourceIds: string[] | null): Promise<boolean> {\n const result = await pool.query(\n `SELECT EXISTS(SELECT 1 FROM context_engine_chunks c\n WHERE c.acl IS NOT NULL AND ($1::text[] IS NULL OR c.source_id = ANY($1::text[]))) AS has_acl`,\n [sourceIds],\n );\n return !result.rows[0]?.has_acl;\n}\n\nexport async function shouldUseCommunitySummaries(\n pool: Pool,\n opts: { sourceIds: string[] | null; principals: string[] | null },\n): Promise<boolean> {\n if (opts.principals == null) return true;\n return corpusIsAclUniform(pool, opts.sourceIds);\n}\n\nasync function deriveQueryEntities(\n pool: Pool,\n chunkIds: string[],\n opts: { sourceIds: string[] | null; documentIds?: string[] | null; principals: string[] | null },\n): Promise<Array<Record<string, unknown>>> {\n if (!chunkIds.length) return [];\n const result = await pool.query(\n `SELECT e.normalized_name, e.name, e.type, COUNT(*) AS freq\n FROM context_engine_chunk_entities ce\n JOIN context_engine_entities e ON e.id = ce.entity_id\n JOIN context_engine_chunks c ON c.id = ce.chunk_id\n WHERE ce.chunk_id = ANY($1::uuid[]) ${SCOPE}\n GROUP BY e.normalized_name, e.name, e.type\n ORDER BY freq DESC LIMIT $5`,\n [chunkIds, opts.sourceIds, opts.documentIds ?? null, opts.principals, ENTITY_LIMIT],\n );\n return result.rows.map((r) => ({\n normalized_name: r.normalized_name,\n name: r.name,\n type: r.type,\n }));\n}\n\nasync function computeVectorScores(\n pool: Pool,\n chunkIds: string[],\n vector: number[],\n): Promise<Record<string, number>> {\n if (!chunkIds.length || !vector) return {};\n const result = await pool.query(\n `SELECT c.id::text, 1 - (c.embedding <=> CAST($2 AS vector))::float AS sim\n FROM context_engine_chunks c\n WHERE c.id = ANY($1::uuid[]) AND c.embedding IS NOT NULL`,\n [chunkIds, vecLiteral(vector)],\n );\n let scores = Object.fromEntries(result.rows.map((r) => [String(r.id), Number(r.sim)]));\n const vs = Object.values(scores);\n if (vs.length) {\n const lo = Math.min(...vs);\n const hi = Math.max(...vs);\n if (hi > lo)\n scores = Object.fromEntries(Object.entries(scores).map(([k, v]) => [k, (v - lo) / (hi - lo)]));\n }\n return scores;\n}\n\nasync function computeRelationshipScores(\n pool: Pool,\n chunkIds: string[],\n queryEntityNames: string[],\n): Promise<Record<string, number>> {\n if (!chunkIds.length || !queryEntityNames.length) return {};\n const result = await pool.query(\n `SELECT r.chunk_id::text, COUNT(*) AS n\n FROM context_engine_entity_relationships r\n JOIN context_engine_entities se ON se.id = r.source_entity_id\n JOIN context_engine_entities te ON te.id = r.target_entity_id\n WHERE r.chunk_id = ANY($1::uuid[])\n AND (se.normalized_name = ANY($2::text[]) OR te.normalized_name = ANY($2::text[]))\n GROUP BY r.chunk_id`,\n [chunkIds, queryEntityNames],\n );\n const counts = Object.fromEntries(result.rows.map((r) => [String(r.chunk_id), Number(r.n)]));\n if (!Object.keys(counts).length) return {};\n const mx = Math.max(...Object.values(counts));\n return mx ? Object.fromEntries(Object.entries(counts).map(([cid, n]) => [cid, n / mx])) : {};\n}\n\nasync function computeCommunityScores(\n pool: Pool,\n chunkIds: string[],\n vector: number[],\n opts: {\n sourceIds: string[] | null;\n documentIds?: string[] | null;\n principals: string[] | null;\n useSummaries: boolean;\n },\n): Promise<Record<string, number>> {\n if (!opts.useSummaries || !chunkIds.length || !vector) return {};\n const commRows = await pool.query(\n `SELECT entity_ids, 1 - (embedding <=> CAST($1 AS vector))::float AS sim\n FROM context_engine_communities WHERE embedding IS NOT NULL\n ORDER BY embedding <=> CAST($1 AS vector) LIMIT 5`,\n [vecLiteral(vector)],\n );\n if (!commRows.rows.length) return {};\n const entityScore: Record<string, number> = {};\n for (const row of commRows.rows) {\n const sim = Number(row.sim);\n if (sim < 0.15) continue;\n for (const eid of row.entity_ids ?? []) {\n entityScore[String(eid)] = Math.max(entityScore[String(eid)] ?? 0, sim);\n }\n }\n if (!Object.keys(entityScore).length) return {};\n const result = await pool.query(\n `SELECT ce.chunk_id::text, ce.entity_id::text\n FROM context_engine_chunk_entities ce\n JOIN context_engine_chunks c ON c.id = ce.chunk_id\n WHERE ce.chunk_id = ANY($1::uuid[]) AND ce.entity_id = ANY($5::uuid[]) ${SCOPE}`,\n [chunkIds, opts.sourceIds, opts.documentIds ?? null, opts.principals, Object.keys(entityScore)],\n );\n const chunkScores: Record<string, number> = {};\n for (const row of result.rows) {\n chunkScores[row.chunk_id] = Math.max(chunkScores[row.chunk_id] ?? 0, entityScore[row.entity_id] ?? 0);\n }\n return chunkScores;\n}\n\n/** ACL-scoped vector seed search → top chunk ids (best first). Exported for\n * tests.\n *\n * `documentIds` participates at GENERATION time, not only in the caller's\n * post-filter: with a narrow document scope inside a large source,\n * off-document seeds would fill the whole SEED_LIMIT budget, the post-filter\n * would empty the leg, and the search would silently degrade to hybrid — the\n * post-filter recall-collapse class the ANN/ACL leg already fixed.\n *\n * `backend` is the chunk-plane backend the caller already owns, and it is here\n * for one reason: this is an ANN scan with `SCOPE` applied ON TOP of the index\n * walk, the identical shape `PostgresBackend.tuneAnnScan` corrects for the\n * vector leg. Untuned, the walk keeps at most `hnsw.ef_search` (40) candidates\n * before the filter runs, so a principal with a small visible slice gets few or\n * zero seeds and the whole graph leg quietly degrades to hybrid. The tuning is\n * `SET LOCAL`, hence the explicit client and transaction: it has to be the\n * connection the seed query itself runs on. A backend without the knobs\n * (non-Postgres, or none passed) is skipped.\n */\nexport async function vectorSeedIds(\n pool: Pool,\n vector: number[],\n opts: {\n sourceIds: string[] | null;\n documentIds?: string[] | null;\n principals: string[] | null;\n backend?: Partial<AnnTunable> | null;\n },\n): Promise<string[]> {\n const binds = [opts.sourceIds, opts.documentIds ?? null, opts.principals];\n // Same rule the vector leg derives `scoped` by, read off the ONE bind array\n // so a new scope dimension cannot be silently left out of it.\n const scoped = binds.some((v) => v !== null);\n const client = await pool.connect();\n try {\n await client.query(\"BEGIN\");\n await opts.backend?.tuneAnnScan?.(client, SEED_LIMIT, scoped);\n const result = await client.query(\n `SELECT c.id::text FROM context_engine_chunks c\n WHERE c.embedding IS NOT NULL ${SCOPE}\n ORDER BY c.embedding <=> CAST($1 AS vector) LIMIT $5`,\n [vecLiteral(vector), ...binds, SEED_LIMIT],\n );\n await client.query(\"COMMIT\");\n return result.rows.map((r) => String(r.id));\n } catch (err) {\n try {\n await client.query(\"ROLLBACK\");\n } catch {\n /* ignore */\n }\n throw err;\n } finally {\n client.release();\n }\n}\n\nexport async function buildGraphRanked(\n query: string,\n opts: {\n config: ContextEngineConfig;\n pool: Pool;\n embedder: Embedder;\n graphStore: GraphStore;\n hooks?: Hooks | null;\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n principals?: string[] | null;\n maxDepth?: number;\n backend?: Partial<AnnTunable> | null;\n },\n): Promise<string[]> {\n try {\n const result = await opts.embedder.embed([query], { kind: \"query\" });\n const vectors = Array.isArray(result)\n ? (result[0] as number[][])\n : ((result as { vectors?: number[][] }).vectors ?? []);\n const vector = vectors[0] ? [...vectors[0]] : null;\n if (!vector) return [];\n const sourceIds = opts.sourceIds ?? null;\n const documentIds = opts.documentIds ?? null;\n const principals = opts.principals ?? null;\n const seeds = await vectorSeedIds(opts.pool, vector, {\n sourceIds,\n documentIds,\n principals,\n backend: opts.backend,\n });\n if (!seeds.length) return [];\n const queryEntities = await deriveQueryEntities(opts.pool, seeds.slice(0, TOP_SEEDS_FOR_ENTITIES), {\n sourceIds,\n documentIds,\n principals,\n });\n const entityNorms = queryEntities.map((e) => String(e.normalized_name));\n\n let expanded: string[] = seeds;\n let connectivity: Record<string, number> = {};\n try {\n await opts.graphStore.connect();\n expanded = await opts.graphStore.expandChunkSet(seeds, entityNorms, opts.maxDepth ?? 2);\n connectivity = await opts.graphStore.getChunkConnectivityScores(expanded.length ? expanded : seeds);\n } catch (exc) {\n if (opts.hooks) emitError(opts.hooks, exc, { stage: \"graph_expand\" });\n else console.warn(\"graph leg: expansion failed:\", exc);\n expanded = seeds;\n connectivity = {};\n }\n const allIds = expanded.length ? expanded : seeds;\n const useSummaries = await shouldUseCommunitySummaries(opts.pool, { sourceIds, principals });\n const vecScores = await computeVectorScores(opts.pool, allIds, vector);\n const relScores = await computeRelationshipScores(opts.pool, allIds, entityNorms);\n const commScores = await computeCommunityScores(opts.pool, allIds, vector, {\n sourceIds,\n documentIds,\n principals,\n useSummaries,\n });\n const textRows = await opts.pool.query(\n `SELECT id::text, lower(coalesce(text,'')) AS t FROM context_engine_chunks WHERE id = ANY($1::uuid[])`,\n [allIds],\n );\n const texts = Object.fromEntries(textRows.rows.map((r) => [String(r.id), String(r.t)]));\n const weights = {\n ...DEFAULT_WEIGHTS,\n ...((\n opts.config.graph as {\n rerankWeights?: Record<string, number>;\n rerank_weights?: Record<string, number>;\n }\n ).rerankWeights ??\n (opts.config.graph as { rerank_weights?: Record<string, number> }).rerank_weights ??\n {}),\n };\n const entityNameSet = new Set(entityNorms);\n const scored: Array<[string, number]> = [];\n for (const cidRaw of allIds) {\n const cid = String(cidRaw);\n if (!(cid in texts)) continue;\n const v = vecScores[cid] ?? 0.5;\n const chunkText = texts[cid] ?? \"\";\n const matches = [...entityNameSet].filter((n) => n && chunkText.includes(n)).length;\n const e = entityNameSet.size ? Math.min(1.0, matches / Math.max(1, entityNameSet.size)) : 0;\n const r = relScores[cid] ?? 0;\n const cm = commScores[cid] ?? 0;\n const gc = connectivity[cid] ?? 0.5;\n const final =\n v * weights.vector_score +\n e * weights.entity_match +\n r * weights.relationship_relevance +\n cm * weights.community_match +\n gc * weights.graph_connectivity;\n scored.push([cid, final]);\n }\n scored.sort((a, b) => b[1] - a[1]);\n return scored.map(([cid]) => cid);\n } catch (exc) {\n if (opts.hooks) emitError(opts.hooks, exc, { stage: \"graph_embed_query\" });\n else console.warn(\"graph leg: query embed failed:\", exc);\n return [];\n }\n}\n","/**\n * Graph NAVIGATION — walking the entity graph, instead of ranking chunks by it.\n *\n * `retrieval.ts` uses the graph to ORDER chunks: entities and communities\n * become scores in an RRF leg and the caller never sees them. Navigation is\n * the other question — *what is connected to what* — and answers with entity\n * names, relationship labels and the evidence sentence behind each edge.\n *\n * **It reads the Postgres mirror, never Neo4j.** Not an optimisation: the\n * mirror is where access control lives. A chunk carries the `acl`, and the\n * scope predicate is the one thing that enforces it across every leg. The\n * Neo4j copy has no ACL data at all, so traversing it would mean\n * re-implementing visibility in Cypher — a second enforcement point that can\n * only drift from the first.\n *\n * **The visibility rule is the EDGE, not the node.** `evidence` is a sentence\n * quoted from the chunk a relationship was extracted from, so returning an\n * edge whose evidence chunk is hidden hands the caller that chunk's text.\n * Every query joins the relationship to its evidence chunk and scopes it. An\n * edge with no evidence chunk has no provenance to check, so it is withheld\n * from an access-controlled caller and shown only to a trusted one.\n *\n * A start entity that resolves to nothing and one hidden behind an ACL give\n * the SAME answer, for the reason `getDocument` gives an absent and a\n * forbidden document one message.\n */\n\nimport type { Pool } from \"pg\";\nimport { shouldUseCommunitySummaries } from \"./retrieval.js\";\n\n/**\n * The scope predicate, numbered for the queries below. Textually a copy of\n * storage.ts's SCOPE — the canonical one — because pg binds are POSITIONAL\n * and these queries number their placeholders differently. Any change to the\n * canonical predicate must be made here too. Never widen it.\n */\nconst SCOPE = `\n AND ($1::text[] IS NULL OR c.source_id = ANY($1::text[]))\n AND ($2::uuid[] IS NULL OR c.document_id = ANY($2::uuid[]))\n AND ($3::text[] IS NULL OR c.acl IS NULL OR c.acl && $3::text[])\n`;\n\n/** An edge is visible when its evidence chunk is. No chunk, no provenance. */\nconst EDGE_VISIBLE = `\n AND (\n ($3::text[] IS NULL AND r.chunk_id IS NULL)\n OR EXISTS (\n SELECT 1 FROM context_engine_chunks c\n WHERE c.id = r.chunk_id ${SCOPE}\n )\n )\n`;\n\n/** Rule 2 for the entity a walk STARTS from. */\nconst ENTITY_VISIBLE = `\n EXISTS (\n SELECT 1 FROM context_engine_chunk_entities ce\n JOIN context_engine_chunks c ON c.id = ce.chunk_id\n WHERE ce.entity_id = e.id ${SCOPE}\n )\n`;\n\nconst NOT_FOUND = \"entity not found\";\nconst MAX_DEPTH = 5;\nconst MAX_LIMIT = 200;\nconst MIN_COMMUNITY_RELEVANCE = 0.25;\n\n/** Rule 3, said to the model rather than returned as a mysterious empty list. */\nexport const SUMMARIES_WITHHELD =\n \"community summaries are withheld: this corpus mixes access-controlled and \" +\n \"open documents, and a community summary is computed over the whole corpus, \" +\n \"so it cannot be shown to a caller who can only see part of it. Use search, \" +\n \"traverse or get_neighbors instead — those are filtered per document.\";\n\nexport type NavScope = {\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n principals?: string[] | null;\n};\n\nfunction clamp(value: number | null | undefined, fallback: number, ceiling: number): number {\n const n = typeof value === \"number\" && Number.isFinite(value) ? Math.trunc(value) : fallback;\n return Math.max(1, Math.min(n, ceiling));\n}\n\nfunction scopeArgs(opts: NavScope): [string[] | null, string[] | null, string[] | null] {\n return [opts.sourceIds ?? null, opts.documentIds ?? null, opts.principals ?? null];\n}\n\nasync function resolveEntity(\n pool: Pool,\n name: string,\n opts: NavScope,\n): Promise<{ id: string; name: string; type: string } | null> {\n const result = await pool.query(\n `SELECT e.id::text, e.name, e.type FROM context_engine_entities e\n WHERE e.normalized_name = $4 AND ${ENTITY_VISIBLE} LIMIT 1`,\n [...scopeArgs(opts), (name ?? \"\").trim().toLowerCase()],\n );\n const row = result.rows[0];\n return row ? { id: String(row.id), name: row.name, type: row.type } : null;\n}\n\n/** Everything one hop from `entity`, each with the direction of its edge. */\nexport async function getNeighbors(\n pool: Pool,\n opts: NavScope & { entity: string; limit?: number | null },\n): Promise<Record<string, unknown>> {\n const start = await resolveEntity(pool, opts.entity, opts);\n if (!start) return { found: false, error: NOT_FOUND, neighbors: [], count: 0 };\n const result = await pool.query(\n `SELECT other.name, other.type, r.category, r.label, r.evidence,\n CASE WHEN r.source_entity_id = $4::uuid THEN 'outgoing' ELSE 'incoming' END AS direction\n FROM context_engine_entity_relationships r\n JOIN context_engine_entities other ON other.id =\n CASE WHEN r.source_entity_id = $4::uuid THEN r.target_entity_id ELSE r.source_entity_id END\n WHERE (r.source_entity_id = $4::uuid OR r.target_entity_id = $4::uuid)\n ${EDGE_VISIBLE}\n ORDER BY other.name LIMIT $5`,\n [...scopeArgs(opts), start.id, clamp(opts.limit, 20, MAX_LIMIT)],\n );\n const neighbors = result.rows.map((r) => ({\n name: r.name,\n type: r.type,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n direction: r.direction,\n }));\n return {\n found: true,\n entity: { name: start.name, type: start.type },\n neighbors,\n count: neighbors.length,\n };\n}\n\n/** Every entity reachable from `entity` within `depth` hops. */\nexport async function traverse(\n pool: Pool,\n opts: NavScope & { entity: string; depth?: number | null; category?: string | null; limit?: number | null },\n): Promise<Record<string, unknown>> {\n const start = await resolveEntity(pool, opts.entity, opts);\n if (!start) return { found: false, error: NOT_FOUND, paths: [], count: 0 };\n // Undirected — a relationship is a connection whichever way it was written —\n // and `visited` carries the ids already on this path so a cycle cannot loop.\n const result = await pool.query(\n `WITH RECURSIVE walk AS (\n SELECT r.id AS rel_id,\n CASE WHEN r.source_entity_id = $4::uuid THEN r.target_entity_id ELSE r.source_entity_id END AS node_id,\n r.category, r.label, r.evidence, 1 AS hop,\n ARRAY[r.source_entity_id, r.target_entity_id] AS visited\n FROM context_engine_entity_relationships r\n WHERE (r.source_entity_id = $4::uuid OR r.target_entity_id = $4::uuid)\n ${EDGE_VISIBLE}\n UNION ALL\n SELECT r.id,\n CASE WHEN r.source_entity_id = w.node_id THEN r.target_entity_id ELSE r.source_entity_id END,\n r.category, r.label, r.evidence, w.hop + 1,\n w.visited || CASE WHEN r.source_entity_id = w.node_id THEN r.target_entity_id ELSE r.source_entity_id END\n FROM walk w\n JOIN context_engine_entity_relationships r\n ON (r.source_entity_id = w.node_id OR r.target_entity_id = w.node_id)\n WHERE w.hop < $5\n AND NOT (CASE WHEN r.source_entity_id = w.node_id THEN r.target_entity_id ELSE r.source_entity_id END = ANY(w.visited))\n ${EDGE_VISIBLE}\n )\n SELECT DISTINCT ON (w.node_id, w.category, w.label)\n e.name, e.type, w.category, w.label, w.evidence, w.hop\n FROM walk w JOIN context_engine_entities e ON e.id = w.node_id\n WHERE ($6::text IS NULL OR w.category = $6::text)\n ORDER BY w.node_id, w.category, w.label, w.hop ASC\n LIMIT $7`,\n [\n ...scopeArgs(opts),\n start.id,\n clamp(opts.depth, 2, MAX_DEPTH),\n opts.category ?? null,\n clamp(opts.limit, 50, MAX_LIMIT),\n ],\n );\n const paths = result.rows.map((r) => ({\n target: r.name,\n target_type: r.type,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n hops: r.hop,\n }));\n paths.sort((a, b) => a.hops - b.hops || String(a.target).localeCompare(String(b.target)));\n return { found: true, entity: { name: start.name, type: start.type }, paths, count: paths.length };\n}\n\n/** Relationships of a given kind, without naming a start entity. */\nexport async function findRelated(\n pool: Pool,\n opts: NavScope & {\n category?: string | null;\n label?: string | null;\n entityType?: string | null;\n limit?: number | null;\n },\n): Promise<Record<string, unknown>> {\n const result = await pool.query(\n `SELECT src.name AS src_name, src.type AS src_type, tgt.name AS tgt_name, tgt.type AS tgt_type,\n r.category, r.label, r.evidence\n FROM context_engine_entity_relationships r\n JOIN context_engine_entities src ON src.id = r.source_entity_id\n JOIN context_engine_entities tgt ON tgt.id = r.target_entity_id\n WHERE ($4::text IS NULL OR r.category = $4::text)\n AND ($5::text IS NULL OR r.label = $5::text)\n AND ($6::text IS NULL OR src.type = $6::text OR tgt.type = $6::text)\n ${EDGE_VISIBLE}\n ORDER BY src.name, tgt.name LIMIT $7`,\n [\n ...scopeArgs(opts),\n opts.category ?? null,\n opts.label ?? null,\n opts.entityType ?? null,\n clamp(opts.limit, 50, MAX_LIMIT),\n ],\n );\n const relationships = result.rows.map((r) => ({\n source: r.src_name,\n source_type: r.src_type,\n target: r.tgt_name,\n target_type: r.tgt_type,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n }));\n return { found: true, relationships, count: relationships.length };\n}\n\n/**\n * The themes of the corpus: LLM-written summaries of entity communities,\n * ranked against `query`. Withheld entirely from an access-controlled caller\n * over a corpus that mixes open and restricted documents (rule 3).\n *\n * `documentIds` is deliberately not honoured: a community spans the corpus, so\n * narrowing it to some documents would describe a thing never computed.\n */\nexport async function communitySummary(\n pool: Pool,\n opts: {\n embedder: { embed: (texts: string[], kind?: string) => Promise<number[][]> };\n query: string;\n limit?: number | null;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n },\n): Promise<Record<string, unknown>> {\n const allowed = await shouldUseCommunitySummaries(pool, {\n sourceIds: opts.sourceIds ?? null,\n principals: opts.principals ?? null,\n });\n if (!allowed) {\n return { available: false, error: SUMMARIES_WITHHELD, communities: [], count: 0 };\n }\n const vectors = await opts.embedder.embed([opts.query ?? \"\"], \"query\");\n const vector = vectors?.[0];\n if (!vector) return { available: true, communities: [], count: 0 };\n const result = await pool.query(\n `SELECT summary, entity_count, relationship_count, level,\n 1 - (embedding <=> CAST($1 AS vector))::float AS sim\n FROM context_engine_communities\n WHERE embedding IS NOT NULL AND summary IS NOT NULL\n ORDER BY embedding <=> CAST($1 AS vector) LIMIT $2`,\n [`[${vector.map((x) => Number(x)).join(\",\")}]`, clamp(opts.limit, 3, 10)],\n );\n const communities = result.rows\n .filter((r) => Number(r.sim) > MIN_COMMUNITY_RELEVANCE)\n .map((r) => ({\n summary: r.summary,\n entity_count: r.entity_count,\n relationship_count: r.relationship_count,\n hierarchy_level: r.level,\n relevance: Math.round(Number(r.sim) * 10000) / 10000,\n }));\n return { available: true, communities, count: communities.length };\n}\n","import { z } from \"zod\";\nimport { ExtraMissingError } from \"./errors.js\";\nimport { requireExtra } from \"./extras.js\";\nimport { ENTITY_TYPES, RELATIONSHIP_CATEGORY_VALUES } from \"./graph/entities.js\";\nimport {\n callKnowledgeTool,\n INPUT_PROPERTIES,\n KNOWLEDGE_ACTIONS,\n KNOWLEDGE_TOOL_DESCRIPTION,\n type KnowledgeComputeFn,\n type ScopeInput,\n} from \"./knowledge-tool.js\";\nimport { requireValidUuid } from \"./routing-core.js\";\nimport {\n type ApprovalScopeFn,\n type PrincipalsFn,\n registerToolGateway,\n resolvePrincipalsFn,\n SCOPE_NOTE,\n} from \"./tools/mcp-tools.js\";\nimport { __version__ } from \"./version.js\";\n\ntype Engine = {\n config: { llm?: unknown; enableCodeExecution?: boolean };\n search: (query: string, opts?: Record<string, unknown>) => Promise<{ hits: unknown[]; usage?: unknown }>;\n getDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n listDocuments: (opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n queryStructured?: (question: string, opts?: Record<string, unknown>) => Promise<unknown>;\n compute?: (instruction: string, opts?: Record<string, unknown>) => Promise<unknown>;\n searchTools: (query: string, opts?: Record<string, unknown>) => Promise<unknown[]>;\n executeTool: (\n name: string,\n args: Record<string, unknown> | null,\n opts?: Record<string, unknown>,\n ) => Promise<Record<string, unknown>>;\n};\n\n/** The description the shared property table gives this parameter. */\nfunction described(name: string): string {\n return String(INPUT_PROPERTIES[name]?.description ?? \"\");\n}\n\nfunction asMcpResult(data: unknown): { content: Array<{ type: \"text\"; text: string }> } {\n return { content: [{ type: \"text\", text: JSON.stringify(data) }] };\n}\n\nexport async function createMcpApp(\n engine: Engine,\n opts: {\n principals: PrincipalsFn;\n /**\n * REQUIRED ceiling of source ids (or a `Scope`) this mounted tool may\n * ever reach — resolved fresh per call like `principals` and NEVER a tool\n * argument. A model can ask for any source id it likes, and a tool that\n * believed it would let one caller read another's documents. A host with\n * one shared corpus says so on purpose with `scope: UNSCOPED`.\n */\n scope: ScopeInput | (() => ScopeInput | Promise<ScopeInput>);\n /**\n * Overrides `config.redaction` per call, so a per-project or\n * per-customer policy reaches this tool the way it already reaches\n * `search`. Never a tool argument.\n */\n redaction?: unknown | (() => unknown | Promise<unknown>);\n /**\n * Optionally REPLACES the built-in compute action with the host's own\n * callable. Running generated code is where a host has its own rules\n * about permission, billing and approval; supplying one here means it\n * does not have to intercept the action before the tool is reached, and\n * makes the action available whatever `enableCodeExecution` says.\n */\n compute?: KnowledgeComputeFn | null;\n /** The same seam for the other expensive action — one LLM call per document. */\n mapReduce?:\n | ((instruction: string, opts: Record<string, unknown>) => Promise<Record<string, unknown>>)\n | null;\n approvalScope?: ApprovalScopeFn | null;\n },\n): Promise<\n ((req: import(\"node:http\").IncomingMessage, res: import(\"node:http\").ServerResponse) => Promise<void>) & {\n mcp: unknown;\n }\n> {\n if (!opts?.principals) {\n throw new TypeError(\"createMcpApp requires principals\");\n }\n if (opts.scope === undefined || opts.scope === null) {\n throw new TypeError(\n \"createMcpApp requires scope: the source ids this tool may reach, or UNSCOPED \" +\n \"(from @promptev/context-engine) to say the whole corpus on purpose.\",\n );\n }\n\n let McpServer: new (info: {\n name: string;\n version: string;\n }) => {\n tool: (...args: unknown[]) => unknown;\n connect: (t: unknown) => Promise<void>;\n };\n let StreamableHTTPServerTransport: new (opts: {\n sessionIdGenerator?: undefined;\n }) => {\n handleRequest: (req: unknown, res: unknown, body?: unknown) => Promise<void>;\n };\n try {\n const serverMod = await requireExtra<{ McpServer: typeof McpServer }>(\n \"@modelcontextprotocol/sdk/server/mcp.js\",\n \"mcp\",\n \"MCP server\",\n );\n const httpMod = await requireExtra<{\n StreamableHTTPServerTransport: typeof StreamableHTTPServerTransport;\n }>(\"@modelcontextprotocol/sdk/server/streamableHttp.js\", \"mcp\", \"MCP server\");\n McpServer = serverMod.McpServer;\n StreamableHTTPServerTransport = httpMod.StreamableHTTPServerTransport;\n } catch (exc) {\n if (exc instanceof ExtraMissingError) throw exc;\n throw new ExtraMissingError(\"mcp\", \"@modelcontextprotocol/sdk\", \"MCP server\");\n }\n\n const mcp = new McpServer({ name: \"context-engine\", version: __version__ });\n\n const tool = (\n name: string,\n description: string,\n schema: Record<string, unknown>,\n handler: (args: Record<string, unknown>) => Promise<unknown>,\n ) => {\n mcp.tool(name, description, schema, async (args: Record<string, unknown>) =>\n asMcpResult(await handler(args)),\n );\n };\n\n tool(\n \"search_knowledge_base\",\n `${KNOWLEDGE_TOOL_DESCRIPTION}${SCOPE_NOTE}`,\n {\n // Zod RAW SHAPES, not JSON schema: the SDK's isZodRawShape test\n // rejects a plain schema object — on current SDK versions that made\n // registration THROW at startup, and on 1.12.0 the object was consumed\n // as annotations and every handler ran with NO arguments.\n // Every `.describe()` comes from the ONE property table, so the schema\n // a model reads over the protocol and the exported JSON Schema cannot\n // say different things.\n action: z.enum(KNOWLEDGE_ACTIONS).describe(described(\"action\")),\n query: z.string().optional().describe(described(\"query\")),\n document_id: z.string().optional().describe(described(\"document_id\")),\n source_ids: z.array(z.string()).optional().describe(described(\"source_ids\")),\n document_ids: z.array(z.string()).optional().describe(described(\"document_ids\")),\n entity: z.string().optional().describe(described(\"entity\")),\n depth: z.number().int().optional().describe(described(\"depth\")),\n category: z.enum(RELATIONSHIP_CATEGORY_VALUES).optional().describe(described(\"category\")),\n label: z.string().optional().describe(described(\"label\")),\n entity_type: z.enum(ENTITY_TYPES).optional().describe(described(\"entity_type\")),\n top_k: z.number().int().optional().describe(described(\"top_k\")),\n mode: z.enum([\"hybrid\", \"graph\"]).optional().describe(described(\"mode\")),\n limit: z.number().int().optional().describe(described(\"limit\")),\n cursor: z\n .union([z.string(), z.record(z.unknown())])\n .optional()\n .describe(described(\"cursor\")),\n start: z.number().int().optional().describe(described(\"start\")),\n end: z.number().int().optional().describe(described(\"end\")),\n max_chars: z.number().int().optional().describe(described(\"max_chars\")),\n },\n async (args) => {\n // Early, mirroring the HTTP handler: a malformed id must surface as a\n // clean tool error, not a Postgres 22P02 from inside every leg.\n const documentIds = args.document_ids as string[] | undefined;\n const documentId = args.document_id as string | undefined;\n for (const did of [...(documentIds ?? []), ...(documentId ? [documentId] : [])]) {\n requireValidUuid(did);\n }\n // Everything below is `knowledge-tool`, the same function\n // `engine.searchKnowledgeBase` calls — this layer only turns an MCP\n // request into a caller identity and a scope ceiling.\n const callerPrincipals = await resolvePrincipalsFn(opts.principals);\n const ceiling = typeof opts.scope === \"function\" ? await opts.scope() : opts.scope;\n const policy = typeof opts.redaction === \"function\" ? await opts.redaction() : opts.redaction;\n return callKnowledgeTool(engine as never, {\n ...(args as Record<string, unknown>),\n action: String(args.action),\n principals: callerPrincipals,\n scope: ceiling as never,\n redaction: policy,\n compute: opts.compute ?? null,\n map_reduce: opts.mapReduce ?? null,\n });\n },\n );\n\n registerToolGateway(\n {\n tool: (name, description, schema, handler) => {\n tool(name, description, schema, async (args) => handler(args as never));\n },\n },\n engine,\n opts.principals,\n opts.approvalScope ?? null,\n );\n\n const handler = (async (\n req: import(\"node:http\").IncomingMessage,\n res: import(\"node:http\").ServerResponse,\n ) => {\n const transport = new StreamableHTTPServerTransport({ sessionIdGenerator: undefined });\n await mcp.connect(transport);\n await transport.handleRequest(req, res);\n }) as ((\n req: import(\"node:http\").IncomingMessage,\n res: import(\"node:http\").ServerResponse,\n ) => Promise<void>) & { mcp: unknown };\n handler.mcp = mcp;\n return handler;\n}\n","import { randomUUID } from \"node:crypto\";\nimport { jsonrepair } from \"jsonrepair\";\nimport type { Pool } from \"pg\";\nimport type { ContextEngineConfig } from \"../config.js\";\nimport type { GraphStore } from \"./neo4j-client.js\";\n\nexport const ENTITY_TYPES = [\n \"PERSON\",\n \"ORG\",\n \"PRODUCT\",\n \"LOCATION\",\n \"REFERENCE\",\n \"TEMPORAL\",\n \"CONCEPT\",\n] as const;\nexport type EntityType = (typeof ENTITY_TYPES)[number];\n\n/**\n * Ordered so a schema can publish it (a Set has no order to publish). The\n * prompt below is built from it, so the enum a caller is steered toward and\n * the values a row can actually hold are one list.\n */\nexport const RELATIONSHIP_CATEGORY_VALUES = [\n \"HIERARCHICAL\",\n \"MEMBERSHIP\",\n \"CREATION\",\n \"TEMPORAL\",\n \"SPATIAL\",\n \"REFERENCE\",\n \"FUNCTIONAL\",\n \"QUANTITATIVE\",\n] as const;\n\nexport const RELATIONSHIP_CATEGORIES = new Set<string>(RELATIONSHIP_CATEGORY_VALUES);\n\nexport const CHUNKS_PER_BATCH = 11;\n\nconst EXTRACTION_SYSTEM = \"You are a STRICT JSON entity and relationship extraction engine.\";\n\nexport function normalizeEntityName(name: string): string {\n return (name || \"\").toLowerCase().trim().replace(/ {2}/g, \" \");\n}\n\nfunction sharedSignificantTokens(a: string, b: string, minLen = 4): boolean {\n const tokensA = new Set(a.split(/\\s+/).filter((t) => t.length >= minLen));\n const tokensB = b.split(/\\s+/).filter((t) => t.length >= minLen);\n return tokensB.some((t) => tokensA.has(t));\n}\n\n/** Exported so a test can hold the schema enums and this prompt to one list. */\nexport function buildPrompt(fullText: string): string {\n return `Extract ALL entities and relationships from the following document text.\n\nDOCUMENT TEXT:\n${fullText}\n\nENTITY TYPES: ${ENTITY_TYPES.join(\", \")}\n\nRELATIONSHIP CATEGORIES: ${RELATIONSHIP_CATEGORY_VALUES.join(\", \")}\n\nOUTPUT FORMAT (STRICT JSON):\n{\n \"entities\": [\n {\"name\": \"exact text\", \"type\": \"PERSON|ORG|PRODUCT|LOCATION|REFERENCE|TEMPORAL|CONCEPT\"}\n ],\n \"relationships\": [\n {\"source\": \"entity name\", \"target\": \"entity name\", \"category\": \"CATEGORY\", \"label\": \"verb\", \"evidence\": \"brief quote\"}\n ]\n}\n\nRULES:\n- Extract EVERY meaningful entity (people, organizations, products, locations, codes, dates, concepts)\n- Choose the MOST SPECIFIC type; only use CONCEPT when no other type fits\n- Be exhaustive but do NOT invent entities not present in the text\n- Deduplicate: each entity appears once\n- Relationships: source/target MUST be entities you extracted; category MUST be one of the 8\n- Return ONLY the JSON object\n`;\n}\n\nfunction safeJsonParse(text: string, expected: \"object\" | \"array\"): unknown {\n const stripped = text.trim();\n try {\n const result = JSON.parse(stripped);\n if (expected === \"object\" && result && typeof result === \"object\" && !Array.isArray(result))\n return result;\n if (expected === \"array\" && Array.isArray(result)) return result;\n } catch {\n /* repair */\n }\n const repaired = JSON.parse(jsonrepair(stripped));\n return repaired;\n}\n\ntype Querier = { query: Pool[\"query\"] };\n\nexport async function resolveEntity(\n db: Querier,\n _name: string,\n normalizedName: string,\n entityType: string,\n): Promise<Record<string, unknown> | null> {\n const exact = await db.query(`SELECT * FROM context_engine_entities WHERE normalized_name = $1 LIMIT 1`, [\n normalizedName,\n ]);\n if (exact.rows[0]) return exact.rows[0];\n\n const trigram = await db.query(\n `SELECT id FROM context_engine_entities\n WHERE similarity(normalized_name, $1) > 0.75\n ORDER BY similarity(normalized_name, $1) DESC LIMIT 1`,\n [normalizedName],\n );\n if (trigram.rows[0]) {\n const row = await db.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [trigram.rows[0].id]);\n return row.rows[0] ?? null;\n }\n\n const typed = await db.query(\n `SELECT id, normalized_name FROM context_engine_entities\n WHERE type = $2 AND similarity(normalized_name, $1) > 0.45\n ORDER BY similarity(normalized_name, $1) DESC LIMIT 5`,\n [normalizedName, entityType],\n );\n for (const row of typed.rows) {\n if (sharedSignificantTokens(normalizedName, String(row.normalized_name))) {\n const full = await db.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [row.id]);\n return full.rows[0] ?? null;\n }\n }\n return null;\n}\n\nasync function extractFromText(\n fullText: string,\n config: ContextEngineConfig,\n): Promise<{\n entities: Array<Record<string, unknown>>;\n relationships: Array<Record<string, unknown>>;\n tokens: Record<string, number>;\n}> {\n try {\n const llm = config.graph.extractionLlm;\n if (!llm) return { entities: [], relationships: [], tokens: {} };\n const { callLlm } = await import(\"../providers/llm.js\");\n const out = await callLlm(llm, {\n system: EXTRACTION_SYSTEM,\n user: buildPrompt(fullText),\n jsonMode: true,\n });\n const raw = Array.isArray(out) ? out[0] : ((out as { text?: string }).text ?? String(out));\n const tokens = (Array.isArray(out) ? out[1] : (out as { tokens?: Record<string, number> }).tokens) ?? {};\n let result: Record<string, unknown> = {};\n try {\n result = (safeJsonParse(String(raw), \"object\") as Record<string, unknown>) ?? {};\n } catch {\n result = {};\n }\n const entities = ((result.entities as unknown[]) ?? []).filter((e): e is Record<string, unknown> =>\n Boolean(e && typeof e === \"object\" && (e as { name?: unknown }).name && (e as { type?: unknown }).type),\n );\n const relationships = ((result.relationships as unknown[]) ?? []).filter(\n (r): r is Record<string, unknown> =>\n Boolean(\n r &&\n typeof r === \"object\" &&\n (r as { source?: unknown }).source &&\n (r as { target?: unknown }).target,\n ),\n );\n return { entities, relationships, tokens: tokens as Record<string, number> };\n } catch (exc) {\n console.error(\"entity extraction LLM call failed:\", exc);\n return { entities: [], relationships: [], tokens: {} };\n }\n}\n\nasync function loadDocumentChunks(pool: Pool, documentId: string): Promise<Array<Record<string, unknown>>> {\n const result = await pool.query(\n `SELECT id, text, idx, source_id, meta_data FROM context_engine_chunks WHERE document_id = $1 ORDER BY idx`,\n [documentId],\n );\n return result.rows.map((c) => ({\n id: c.id,\n text: c.text || \"\",\n text_lower: String(c.text || \"\").toLowerCase(),\n idx: c.idx || 0,\n source_id: c.source_id,\n token_count:\n (c.meta_data as { token_count?: number } | null)?.token_count ??\n Math.floor(String(c.text || \"\").length / 4),\n }));\n}\n\nasync function persistExtraction(\n pool: Pool,\n chunkSnaps: Array<Record<string, unknown>>,\n extracted: { entities: Array<Record<string, unknown>>; relationships: Array<Record<string, unknown>> },\n): Promise<{ entities_created: number; entities_found: number; relationships_created: number }> {\n const stats = { entities_created: 0, entities_found: 0, relationships_created: 0 };\n const byNorm = new Map<string, Record<string, unknown>>();\n const client = await pool.connect();\n try {\n await client.query(\"BEGIN\");\n for (const entityData of extracted.entities) {\n const normalized = normalizeEntityName(String(entityData.name));\n if (!normalized) continue;\n const entityType = String(entityData.type);\n let entity = await resolveEntity(client, String(entityData.name), normalized, entityType);\n if (entity) {\n await client.query(\n `UPDATE context_engine_entities SET frequency = COALESCE(frequency, 1) + 1,\n name = CASE WHEN length($2) > length(name) THEN $2 ELSE name END, updated_at = now()\n WHERE id = $1`,\n [entity.id, entityData.name],\n );\n entity = (await client.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [entity.id]))\n .rows[0];\n } else {\n const id = randomUUID();\n await client.query(\n `INSERT INTO context_engine_entities (id, name, normalized_name, type, frequency)\n VALUES ($1, $2, $3, $4, 1)`,\n [id, entityData.name, normalized, entityType],\n );\n entity = (await client.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [id])).rows[0];\n stats.entities_created += 1;\n }\n byNorm.set(normalized, entity!);\n byNorm.set(String(entity!.normalized_name), entity!);\n const nameLower = String(entityData.name).toLowerCase();\n for (const snap of chunkSnaps) {\n if (nameLower && String(snap.text_lower).includes(nameLower)) {\n const exists = await client.query(\n `SELECT 1 FROM context_engine_chunk_entities WHERE chunk_id = $1 AND entity_id = $2`,\n [snap.id, entity!.id],\n );\n if (!exists.rows.length) {\n await client.query(\n `INSERT INTO context_engine_chunk_entities (chunk_id, entity_id) VALUES ($1, $2)`,\n [snap.id, entity!.id],\n );\n stats.entities_found += 1;\n }\n }\n }\n }\n for (const rel of extracted.relationships) {\n const sourceNorm = normalizeEntityName(String(rel.source ?? \"\"));\n const targetNorm = normalizeEntityName(String(rel.target ?? \"\"));\n const category = String(rel.category ?? \"\").toUpperCase();\n const label = String(rel.label ?? \"\");\n const evidence = String(rel.evidence ?? \"\");\n if (!RELATIONSHIP_CATEGORIES.has(category)) continue;\n let src = byNorm.get(sourceNorm);\n let tgt = byNorm.get(targetNorm);\n if (!src) {\n src = (\n await client.query(`SELECT * FROM context_engine_entities WHERE normalized_name = $1`, [sourceNorm])\n ).rows[0];\n }\n if (!tgt) {\n tgt = (\n await client.query(`SELECT * FROM context_engine_entities WHERE normalized_name = $1`, [targetNorm])\n ).rows[0];\n }\n if (!src || !tgt) continue;\n const exists = await client.query(\n `SELECT 1 FROM context_engine_entity_relationships\n WHERE source_entity_id = $1 AND target_entity_id = $2 AND category = $3 AND label = $4`,\n [src.id, tgt.id, category, label],\n );\n if (!exists.rows.length) {\n await client.query(\n `INSERT INTO context_engine_entity_relationships\n (id, source_entity_id, target_entity_id, category, label, evidence)\n VALUES ($1, $2, $3, $4, $5, $6)`,\n [randomUUID(), src.id, tgt.id, category, label, evidence.slice(0, 200)],\n );\n stats.relationships_created += 1;\n }\n }\n await client.query(\"COMMIT\");\n } catch (e) {\n await client.query(\"ROLLBACK\");\n throw e;\n } finally {\n client.release();\n }\n return stats;\n}\n\nasync function syncToNeo4j(\n pool: Pool,\n documentId: string,\n chunkSnaps: Array<Record<string, unknown>>,\n graphStore: GraphStore,\n): Promise<void> {\n const chunkRows = chunkSnaps.map((s) => ({\n id: String(s.id),\n document_id: String(documentId),\n source_id: s.source_id,\n text_preview: String(s.text).slice(0, 200),\n position: s.idx,\n token_count: s.token_count,\n }));\n try {\n await graphStore.connect();\n const entities = (await pool.query(`SELECT * FROM context_engine_entities`)).rows;\n const entityById = new Map(entities.map((e) => [String(e.id), e]));\n const entityRows = entities.map((e) => ({\n name: e.name,\n normalized_name: e.normalized_name,\n entity_type: e.type,\n }));\n const ids = chunkSnaps.map((s) => s.id);\n const links = ids.length\n ? (\n await pool.query(`SELECT * FROM context_engine_chunk_entities WHERE chunk_id = ANY($1::uuid[])`, [\n ids,\n ])\n ).rows\n : [];\n const mentionRows = links\n .filter((l) => entityById.has(String(l.entity_id)))\n .map((l) => ({\n chunk_id: String(l.chunk_id),\n entity_name: entityById.get(String(l.entity_id))!.normalized_name,\n }));\n const rels = (await pool.query(`SELECT * FROM context_engine_entity_relationships`)).rows;\n const relRows = [];\n for (const r of rels) {\n const src = entityById.get(String(r.source_entity_id));\n const tgt = entityById.get(String(r.target_entity_id));\n if (src && tgt) {\n relRows.push({\n source_name: src.normalized_name,\n target_name: tgt.normalized_name,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n chunk_id: r.chunk_id ? String(r.chunk_id) : null,\n });\n }\n }\n await graphStore.upsertChunksBatch(chunkRows);\n await graphStore.linkSequentialChunks(String(documentId));\n await graphStore.upsertEntitiesBatch(entityRows);\n await graphStore.linkChunksToEntitiesBatch(mentionRows);\n await graphStore.upsertEntityRelationshipsBatch(relRows);\n } catch (exc) {\n console.warn(\"Neo4j sync failed (continuing on PG mirror):\", exc);\n }\n}\n\nexport async function extractEntitiesForDocument(\n documentId: string,\n opts: {\n config: ContextEngineConfig;\n pool: Pool;\n graphStore?: GraphStore | null;\n hooks?: unknown;\n },\n): Promise<Record<string, unknown>> {\n const chunkSnaps = await loadDocumentChunks(opts.pool, documentId);\n const stats: Record<string, unknown> = {\n chunk_count: chunkSnaps.length,\n entities_created: 0,\n entities_found: 0,\n relationships_created: 0,\n provider_tokens: {},\n };\n if (!chunkSnaps.length) return stats;\n\n const batches = [];\n for (let i = 0; i < chunkSnaps.length; i += CHUNKS_PER_BATCH) {\n batches.push(chunkSnaps.slice(i, i + CHUNKS_PER_BATCH));\n }\n let llmInput = 0;\n let llmOutput = 0;\n for (const batch of batches) {\n const fullText = batch.map((s) => String(s.text)).join(\"\\n\\n\");\n const extracted = await extractFromText(fullText, opts.config);\n llmInput += Number(extracted.tokens.input ?? 0);\n llmOutput += Number(extracted.tokens.output ?? 0);\n const persisted = await persistExtraction(opts.pool, batch, extracted);\n stats.entities_created = Number(stats.entities_created) + persisted.entities_created;\n stats.entities_found = Number(stats.entities_found) + persisted.entities_found;\n stats.relationships_created = Number(stats.relationships_created) + persisted.relationships_created;\n }\n const tokens = stats.provider_tokens as Record<string, number>;\n if (llmInput) tokens.graph_llm_input = llmInput;\n if (llmOutput) tokens.graph_llm_output = llmOutput;\n if (opts.graphStore) await syncToNeo4j(opts.pool, documentId, chunkSnaps, opts.graphStore);\n return stats;\n}\n","import { createHash } from \"node:crypto\";\nimport { closeSync, openSync, readSync } from \"node:fs\";\nimport { detectLanguage, jaccardSim, splitSentences, tokenize } from \"./text.js\";\n\nexport { detectLanguage };\n\nexport const MAX_CHUNK_CHARS = 6000;\n\nconst PAGE_MARKER_RE = /^---\\s*Page\\s+(\\d+)\\s*---$/gm;\nconst PAGE_MARKER_STRIP_RE = /---\\s*Page\\s+\\d+\\s*---\\n?/g;\nconst SHEET_MARKER_RE = /^\\[Sheet:\\s*(.+?)\\]\\s*$/;\nconst HEADING_OR_BULLET_RE = /^\\s*([#*\\-•]|\\d+[.)])\\s+/;\n\nconst PRINTABLE = new Set(\n \"0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ!\\\"#$%&'()*+,-./:;<=>?@[\\\\]^_`{|}~ \\t\\n\\r\\v\\f\",\n);\n\nconst log = {\n warn: (...args: unknown[]) => console.warn(\"[context-engine]\", ...args),\n debug: (...args: unknown[]) => console.debug(\"[context-engine]\", ...args),\n};\n\nexport function normalizeText(s: string | null | undefined): string {\n return (s ?? \"\")\n .split(/\\s+/)\n .filter((p) => p.length > 0)\n .join(\" \");\n}\n\nexport function sha256Bytes(s: string): Buffer {\n return createHash(\"sha256\").update(normalizeText(s), \"utf8\").digest();\n}\n\nexport function calculateContentHash(\n opts: { docBytes?: Buffer | null; fullText?: string | null; filePath?: string | null } = {},\n): string | null {\n const { docBytes = null, fullText = null, filePath = null } = opts;\n if (filePath) {\n const h = createHash(\"md5\");\n const fd = openSync(filePath, \"r\");\n try {\n const buf = Buffer.alloc(8192);\n let n = readSync(fd, buf, 0, 8192, null);\n while (n > 0) {\n h.update(buf.subarray(0, n));\n n = readSync(fd, buf, 0, 8192, null);\n }\n } finally {\n closeSync(fd);\n }\n return h.digest(\"hex\");\n }\n if (docBytes) return createHash(\"md5\").update(docBytes).digest(\"hex\");\n if (fullText) return createHash(\"md5\").update(fullText, \"utf8\").digest(\"hex\");\n return null;\n}\n\nexport function sanitizeText(text: string): string {\n if (!text) return text;\n text = text.replaceAll(\"\\x00\", \"\");\n const allowedControl = new Set([\"\\n\", \"\\r\", \"\\t\"]);\n let out = \"\";\n for (const c of text) {\n if (allowedControl.has(c) || PRINTABLE.has(c) || c.codePointAt(0)! >= 32) out += c;\n }\n return out;\n}\n\nexport function normLine(s: string): string {\n s = (s || \"\").replace(/\\s+/g, \" \").trim();\n s = s.replace(/\\s*,\\s*/g, \", \");\n return s;\n}\n\nexport function dedupLines(text: string, minLen = 24): string {\n const seen = new Set<string>();\n const out: string[] = [];\n for (const raw of (text || \"\").split(/\\r\\n|\\n|\\r/)) {\n const line = normLine(raw);\n if (!line) continue;\n if (line.length >= minLen) {\n const h = createHash(\"sha1\").update(line, \"utf8\").digest(\"hex\");\n if (seen.has(h)) continue;\n seen.add(h);\n }\n out.push(line);\n }\n return out.join(\"\\n\");\n}\n\nexport function looksMostlyBoilerplate(_text: string): boolean {\n return false;\n}\n\nexport function inferMime(\n content: string | null | undefined,\n filename: string | null | undefined,\n): string | null {\n if (filename) {\n const fn = filename.toLowerCase();\n if (fn.endsWith(\".csv\")) return \"text/csv\";\n if (fn.endsWith(\".json\")) return \"application/json\";\n if (fn.endsWith(\".md\") || fn.endsWith(\".markdown\")) return \"text/markdown\";\n if (fn.endsWith(\".xlsx\") || fn.endsWith(\".xls\")) {\n return \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\";\n }\n if (fn.endsWith(\".doc\")) return \"application/msword\";\n if (fn.endsWith(\".rtf\")) return \"application/rtf\";\n if (fn.endsWith(\".html\") || fn.endsWith(\".htm\")) return \"text/html\";\n if (fn.endsWith(\".ppt\")) return \"application/vnd.ms-powerpoint\";\n if (fn.endsWith(\".eml\")) return \"message/rfc822\";\n if (fn.endsWith(\".msg\")) return \"application/vnd.ms-outlook\";\n if (fn.endsWith(\".odt\")) return \"application/vnd.oasis.opendocument.text\";\n if (fn.endsWith(\".ods\")) return \"application/vnd.oasis.opendocument.spreadsheet\";\n if (fn.endsWith(\".odp\")) return \"application/vnd.oasis.opendocument.presentation\";\n if (fn.endsWith(\".epub\")) return \"application/epub+zip\";\n }\n\n if (content) {\n const trimmed = content.trimStart().slice(0, 500);\n\n if (trimmed.startsWith(\"{\") || trimmed.startsWith(\"[\")) {\n try {\n JSON.parse(content);\n return \"application/json\";\n } catch {\n // not valid JSON\n }\n }\n\n if (\n trimmed.startsWith(\"#\") ||\n trimmed.slice(0, 1000).includes(\"\\n## \") ||\n trimmed.slice(0, 500).includes(\"\\n# \")\n ) {\n return \"text/markdown\";\n }\n\n const lines = trimmed.split(\"\\n\").slice(0, 5);\n if (lines.length >= 2) {\n const commaCounts = lines.filter((line) => line.trim()).map((line) => (line.match(/,/g) || []).length);\n if (commaCounts.length >= 2 && new Set(commaCounts).size === 1 && commaCounts[0]! >= 2) {\n return \"text/csv\";\n }\n }\n }\n\n return null;\n}\n\nexport function isTabularMime(mime: string | null | undefined): boolean {\n if (!mime) return false;\n const m = mime.toLowerCase();\n return [\"csv\", \"sheet\", \"excel\", \"spreadsheetml\", \"tab-separated\"].some((k) => m.includes(k));\n}\n\nexport function isMarkdownMime(mime: string | null | undefined): boolean {\n if (!mime) return false;\n const m = mime.toLowerCase();\n return m.includes(\"markdown\") || m === \"text/md\";\n}\n\nexport function isJsonMime(mime: string | null | undefined): boolean {\n if (!mime) return false;\n return mime.toLowerCase().includes(\"json\");\n}\n\nfunction csvNeedsQuote(field: string): boolean {\n return /[\",\\r\\n]/.test(field);\n}\n\nfunction csvEscapeField(field: string): string {\n if (csvNeedsQuote(field)) return `\"${field.replaceAll('\"', '\"\"')}\"`;\n return field;\n}\n\nfunction csvWrite(rows: string[][]): string {\n return `${rows.map((row) => row.map(csvEscapeField).join(\",\")).join(\"\\r\\n\")}\\r\\n`;\n}\n\nfunction parseCsv(text: string): string[][] {\n const rows: string[][] = [];\n let row: string[] = [];\n let field = \"\";\n let i = 0;\n let inQuotes = false;\n while (i < text.length) {\n const ch = text[i]!;\n if (inQuotes) {\n if (ch === '\"') {\n if (text[i + 1] === '\"') {\n field += '\"';\n i += 2;\n continue;\n }\n inQuotes = false;\n i += 1;\n continue;\n }\n field += ch;\n i += 1;\n continue;\n }\n if (ch === '\"') {\n inQuotes = true;\n i += 1;\n continue;\n }\n if (ch === \",\") {\n row.push(field);\n field = \"\";\n i += 1;\n continue;\n }\n if (ch === \"\\r\") {\n i += 1;\n continue;\n }\n if (ch === \"\\n\") {\n row.push(field);\n rows.push(row);\n row = [];\n field = \"\";\n i += 1;\n continue;\n }\n field += ch;\n i += 1;\n }\n if (inQuotes || field.length > 0 || row.length > 0) {\n row.push(field);\n rows.push(row);\n }\n return rows;\n}\n\nfunction jsonDumps(value: unknown): string {\n return JSON.stringify(value, null, 2).replace(/[\\u007f-\\uffff]/g, (ch) => {\n return `\\\\u${ch.charCodeAt(0).toString(16).padStart(4, \"0\")}`;\n });\n}\n\nexport type ChunkMeta = Record<string, unknown>;\n\nfunction splitSheetBlocks(rawText: string): Array<[string | null, string[]]> {\n const lines = (rawText || \"\").split(/\\r\\n|\\n|\\r/);\n const blocks: Array<[string | null, string[]]> = [];\n let currentTitle: string | null = null;\n let currentLines: string[] = [];\n\n for (const line of lines) {\n const m = SHEET_MARKER_RE.exec(line.trim());\n SHEET_MARKER_RE.lastIndex = 0;\n if (m) {\n if (currentLines.length) {\n blocks.push([currentTitle, currentLines]);\n currentLines = [];\n }\n currentTitle = m[1]!.trim();\n continue;\n }\n if (line.trim()) currentLines.push(line);\n }\n\n if (currentLines.length) blocks.push([currentTitle, currentLines]);\n if (!blocks.length && lines.length) {\n return [[null, lines.filter((ln) => ln.trim())]];\n }\n return blocks;\n}\n\nexport function chunkTabular(text: string, rowsPerChunk = 25): [string[], ChunkMeta[]] | [null, null] {\n try {\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n\n for (const [sheetTitle, sheetLines] of splitSheetBlocks(text)) {\n if (!sheetLines.length) continue;\n\n const rows = parseCsv(sheetLines.join(\"\\n\"));\n\n if (rows.length <= 1) {\n let chunkText = sheetLines.join(\"\\n\").trim();\n if (chunkText) {\n if (sheetTitle) chunkText = `[Sheet: ${sheetTitle}]\\n${chunkText}`;\n chunks.push(chunkText);\n const meta: ChunkMeta = { rows_range: \"all\" };\n if (sheetTitle) meta.sheet_title = sheetTitle;\n chunkMetas.push(meta);\n }\n continue;\n }\n\n const header = rows[0]!;\n const dataRows = rows.slice(1);\n\n let effectiveRowsPerChunk = rowsPerChunk;\n const numCols = header.length;\n if (numCols > 100) effectiveRowsPerChunk = 5;\n else if (numCols > 40) effectiveRowsPerChunk = 10;\n else if (numCols > 20) effectiveRowsPerChunk = 15;\n\n for (let i = 0; i < dataRows.length; i += effectiveRowsPerChunk) {\n const batch = dataRows.slice(i, i + effectiveRowsPerChunk);\n let chunkText = csvWrite([header, ...batch]).trim();\n if (sheetTitle && chunkText) chunkText = `[Sheet: ${sheetTitle}]\\n${chunkText}`;\n\n if (chunkText.length > MAX_CHUNK_CHARS && batch.length > 1) {\n const half = Math.floor(batch.length / 2);\n const subBatches: Array<[string[][], number]> = [\n [batch.slice(0, half), i],\n [batch.slice(half), i + half],\n ];\n for (const [subBatch, subStart] of subBatches) {\n let subChunk = csvWrite([header, ...subBatch]).trim();\n if (sheetTitle && subChunk) subChunk = `[Sheet: ${sheetTitle}]\\n${subChunk}`;\n if (subChunk) {\n chunks.push(subChunk);\n const meta: ChunkMeta = {\n rows_range: `${subStart + 2}-${subStart + subBatch.length + 1}`,\n total_rows: dataRows.length,\n };\n if (sheetTitle) meta.sheet_title = sheetTitle;\n chunkMetas.push(meta);\n }\n }\n } else if (chunkText) {\n chunks.push(chunkText);\n const meta: ChunkMeta = {\n rows_range: `${i + 2}-${i + batch.length + 1}`,\n total_rows: dataRows.length,\n };\n if (sheetTitle) meta.sheet_title = sheetTitle;\n chunkMetas.push(meta);\n }\n }\n }\n\n return chunks.length ? [chunks, chunkMetas] : [[text], [{ rows_range: \"all\" }]];\n } catch (exc) {\n log.warn(\"CSV parsing failed, falling back to generic chunker:\", exc);\n return [null, null];\n }\n}\n\nexport function chunkMarkdown(text: string): [string[], ChunkMeta[]] | [null, null] {\n if (!text?.trim()) return [[], []];\n\n const lines = text.split(\"\\n\");\n\n let docTitle: string | null = null;\n for (const line of lines) {\n if (line.startsWith(\"# \") && !line.startsWith(\"## \")) {\n docTitle = line.slice(2).trim();\n break;\n }\n }\n\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n let currentChunkLines: string[] = [];\n let currentH2: string | null = null;\n let titleConsumed = false;\n\n const flushSection = (sectionFallback: string) => {\n if (!currentChunkLines.length) return;\n let prefix = \"\";\n if (docTitle) prefix = `# ${docTitle}\\n`;\n if (currentH2) prefix += `## ${currentH2}\\n`;\n const chunkText = prefix + currentChunkLines.join(\"\\n\").trim();\n\n if (chunkText.length > MAX_CHUNK_CHARS) {\n const subChunks = splitIntoChunks(chunkText);\n for (const sc of subChunks) {\n chunks.push(sc);\n chunkMetas.push({ section_title: currentH2 || docTitle || sectionFallback });\n }\n } else if (chunkText.trim()) {\n chunks.push(chunkText);\n chunkMetas.push({ section_title: currentH2 || docTitle || sectionFallback });\n }\n };\n\n for (const line of lines) {\n if (!titleConsumed && docTitle !== null && line.startsWith(\"# \") && line.slice(2).trim() === docTitle) {\n titleConsumed = true;\n continue;\n }\n if (line.startsWith(\"## \")) {\n flushSection(\"intro\");\n currentH2 = line.slice(3).trim();\n currentChunkLines = [];\n } else {\n currentChunkLines.push(line);\n }\n }\n\n flushSection(\"final\");\n\n const foundH2Headers = lines.some((line) => line.startsWith(\"## \"));\n if (!foundH2Headers || chunks.length === 0) return [null, null];\n return [chunks, chunkMetas];\n}\n\nexport function chunkJson(text: string, itemsPerChunk = 30): [string[], ChunkMeta[]] | [null, null] {\n try {\n if (text.startsWith(\"\\uFEFF\")) text = text.slice(1);\n const data: unknown = JSON.parse(text);\n\n if (Array.isArray(data)) {\n if (data.length <= itemsPerChunk) {\n return [[text], [{ json_path: \"[*]\", item_count: data.length }]];\n }\n\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n for (let i = 0; i < data.length; i += itemsPerChunk) {\n let batch = data.slice(i, i + itemsPerChunk);\n let chunkText = jsonDumps(batch);\n let actualEnd = i + batch.length;\n\n if (chunkText.length > MAX_CHUNK_CHARS && batch.length > 1) {\n let reducedSize = Math.floor(batch.length / 2);\n while (reducedSize > 0) {\n const smallerBatch = batch.slice(0, reducedSize);\n chunkText = jsonDumps(smallerBatch);\n if (chunkText.length <= MAX_CHUNK_CHARS) {\n batch = smallerBatch;\n actualEnd = i + batch.length;\n break;\n }\n reducedSize = Math.floor(reducedSize / 2);\n }\n }\n\n chunkText = `# Items ${i + 1}-${actualEnd} of ${data.length}\\n${chunkText}`;\n chunks.push(chunkText);\n chunkMetas.push({\n json_path: `[${i}:${actualEnd}]`,\n items_range: `${i + 1}-${actualEnd}`,\n });\n }\n return [chunks, chunkMetas];\n }\n\n if (data !== null && typeof data === \"object\" && !Array.isArray(data)) {\n const obj = data as Record<string, unknown>;\n const keys = Object.keys(obj);\n if (keys.length <= 5) {\n return [[text], [{ json_path: \"{*}\", key_count: keys.length }]];\n }\n\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n for (const key of keys) {\n const value = obj[key];\n let chunkText = jsonDumps({ [key]: value });\n if (chunkText.length > MAX_CHUNK_CHARS) {\n chunkText = `# Key: ${key} (truncated)\\n${jsonDumps(value).slice(0, MAX_CHUNK_CHARS - 100)}...`;\n }\n chunks.push(chunkText);\n chunkMetas.push({ json_path: `.${key}`, top_level_key: key });\n }\n return [chunks, chunkMetas];\n }\n\n return [[text], [{ json_path: \"\" }]];\n } catch (exc) {\n log.debug(\"JSON parsing failed, falling back to generic chunker:\", exc);\n return [null, null];\n }\n}\n\nexport function assignPageNumbers(text: string, chunks: string[], chunkMetas: ChunkMeta[]): void {\n const pageMap: Array<[number, number]> = [];\n const re = new RegExp(PAGE_MARKER_RE.source, \"gm\");\n for (const m of (text || \"\").matchAll(re)) {\n pageMap.push([m.index, Number(m[1])]);\n }\n if (!pageMap.length || !chunks.length) return;\n\n const lineMarker = /^---\\s*Page\\s+(\\d+)\\s*---$/;\n\n for (let idx = 0; idx < chunks.length; idx++) {\n const chunkText = chunks[idx]!;\n let anchor = \"\";\n for (const line of chunkText.split(\"\\n\")) {\n const stripped = line.trim();\n if (stripped && !lineMarker.test(stripped) && stripped.length > 10) {\n anchor = stripped.slice(0, 60);\n break;\n }\n }\n if (!anchor) anchor = chunkText.trim().slice(0, 60);\n if (!anchor) continue;\n\n const pos = text.indexOf(anchor);\n if (pos < 0) continue;\n\n let lo = 0;\n let hi = pageMap.length - 1;\n let page: number | null = null;\n while (lo <= hi) {\n const mid = (lo + hi) >> 1;\n if (pageMap[mid]![0] <= pos) {\n page = pageMap[mid]![1];\n lo = mid + 1;\n } else {\n hi = mid - 1;\n }\n }\n\n if (page !== null && idx < chunkMetas.length) {\n chunkMetas[idx]!.page = page;\n }\n }\n}\n\nexport function routeToChunker(\n text: string,\n mime: string | null | undefined,\n title: string | null | undefined,\n opts: { isMarkdown?: boolean } = {},\n): [string[], string | null, ChunkMeta[]] {\n const isMarkdown = opts.isMarkdown ?? false;\n const effectiveMime = mime || inferMime(text, title);\n\n let chunks: string[] | null = null;\n let chunkMetas: ChunkMeta[] | null = null;\n\n if (isTabularMime(effectiveMime)) {\n [chunks, chunkMetas] = chunkTabular(text);\n } else if (isJsonMime(effectiveMime)) {\n [chunks, chunkMetas] = chunkJson(text);\n } else if (isMarkdown || isMarkdownMime(effectiveMime)) {\n [chunks, chunkMetas] = chunkMarkdown(text);\n }\n\n if (chunks === null || chunks.length === 0) {\n chunks = splitIntoChunks(text);\n chunkMetas = Array.from({ length: chunks.length }, () => ({}));\n }\n chunkMetas = chunkMetas ?? [];\n\n assignPageNumbers(text, chunks, chunkMetas);\n\n chunks = chunks.map((c) => c.replace(PAGE_MARKER_STRIP_RE, \"\").trim());\n const paired = chunks.map((c, i) => [c, chunkMetas![i]!] as [string, ChunkMeta]).filter(([c]) => c.trim());\n if (paired.length) {\n chunks = paired.map(([c]) => c);\n chunkMetas = paired.map(([, m]) => m);\n } else {\n chunks = [];\n chunkMetas = [];\n }\n\n return [chunks, effectiveMime, chunkMetas];\n}\n\nexport function chunkerName(mime: string | null | undefined): string {\n if (isTabularMime(mime)) return \"tabular\";\n if (isJsonMime(mime)) return \"json\";\n if (isMarkdownMime(mime)) return \"markdown\";\n return \"generic\";\n}\n\ntype TextUnit = [sep: string, text: string, tokens: number];\n\nfunction assembleUnits(unitList: TextUnit[]): string {\n const parts: string[] = [];\n for (let i = 0; i < unitList.length; i++) {\n const [sep, uText] = unitList[i]!;\n if (i > 0) parts.push(sep);\n parts.push(uText);\n }\n return parts.join(\"\").trim();\n}\n\nfunction codePointLength(s: string): number {\n let n = 0;\n for (const _ of s) n += 1;\n return n;\n}\n\nexport function splitIntoChunks(text: string, maxWords = 180, overlap = 18): string[] {\n const units: TextUnit[] = [];\n let pendingBlank = false;\n for (const ln of (text || \"\").split(/\\r\\n|\\n|\\r/)) {\n const stripped = ln.trim();\n if (!stripped) {\n pendingBlank = units.length > 0;\n continue;\n }\n const lineSep = pendingBlank ? \"\\n\\n\" : \"\\n\";\n pendingBlank = false;\n if (HEADING_OR_BULLET_RE.test(ln) || stripped.length <= 120) {\n units.push([lineSep, stripped, tokenize(stripped).length]);\n } else {\n let first = true;\n for (const sent of splitSentences(ln)) {\n const s = sent.trim();\n if (s) {\n units.push([first ? lineSep : \" \", s, tokenize(s).length]);\n first = false;\n }\n }\n }\n }\n\n const normalizedUnits: TextUnit[] = [];\n for (const [uSep, uText, uTokens] of units) {\n if (uTokens <= maxWords) {\n normalizedUnits.push([uSep, uText, uTokens]);\n continue;\n }\n const windowChars = maxWords * 4;\n const uChars = [...uText];\n let pos = 0;\n let pieceSep = uSep;\n while (pos < uChars.length) {\n let pieceChars = uChars.slice(pos, pos + windowChars);\n let piece = pieceChars.join(\"\");\n if (pos + windowChars < uChars.length) {\n const ws = piece.lastIndexOf(\" \");\n if (ws > Math.floor(windowChars / 2)) {\n piece = piece.slice(0, ws);\n pieceChars = [...piece];\n }\n }\n let pieceTokens = tokenize(piece);\n if (pieceTokens.length > maxWords) {\n piece = pieceChars.slice(0, maxWords).join(\"\");\n pieceTokens = tokenize(piece);\n }\n normalizedUnits.push([pieceSep, piece, pieceTokens.length]);\n pos += codePointLength(piece);\n pieceSep = \"\";\n }\n }\n\n if (!normalizedUnits.some(([, , t]) => t)) return [];\n\n const chunks: string[] = [];\n let lastKept: string | null = null;\n let cur: TextUnit[] = [];\n let curTokens = 0;\n\n const flush = () => {\n const chunk = assembleUnits(cur);\n if (chunk && (lastKept === null || jaccardSim(chunk, lastKept) < 0.92)) {\n chunks.push(chunk);\n lastKept = chunk;\n }\n const carried: TextUnit[] = [];\n let carriedTokens = 0;\n for (let i = cur.length - 1; i >= 0; i--) {\n const prev = cur[i]!;\n if (prev[2] === 0 || carriedTokens + prev[2] > overlap) break;\n carried.unshift(prev);\n carriedTokens += prev[2];\n }\n cur = carried;\n curTokens = carriedTokens;\n };\n\n for (const unit of normalizedUnits) {\n if (cur.length && curTokens + unit[2] > maxWords) flush();\n cur.push(unit);\n curTokens += unit[2];\n }\n\n if (cur.length) {\n const chunk = assembleUnits(cur);\n if (chunk && (lastKept === null || jaccardSim(chunk, lastKept) < 0.92)) {\n chunks.push(chunk);\n }\n }\n\n return chunks;\n}\n","/**\n * CPSE action surface — document access, structured query, compute.\n *\n * **ACL/source scoping.** Every read here goes through `scopeDocumentsSql`,\n * the Document-table equivalent of storage's chunk-table predicate:\n * `sourceIds=null` means the whole corpus, a list scopes to it;\n * `principals=null` means a trusted internal caller (no ACL filtering),\n * `principals=[]` means an anonymous caller and matches only `acl IS NULL`\n * documents. Same non-interchangeable pair as `search()` — see that\n * module's docstring.\n *\n * Missing vs forbidden documents raise `DocumentNotFoundError` with the\n * identical message `document not found: ${id}` so a caller can never\n * distinguish \"doesn't exist\" from \"exists but you can't see it\".\n */\nimport type { Pool } from \"pg\";\nimport { isTabularMime } from \"./chunkers.js\";\nimport type { ContextEngineConfig, LLMConfig } from \"./config.js\";\nimport { DocumentNotFoundError, EngineActionError, ExtraMissingError } from \"./errors.js\";\nimport { emitError, type Hooks } from \"./hooks.js\";\nimport type { Embedder } from \"./providers/embeddings.js\";\nimport { callLlm } from \"./providers/llm.js\";\nimport { applyRedaction, type RedactionPolicy } from \"./redaction.js\";\nimport { executeSafeCode } from \"./sandbox.js\";\nimport { normalizeKey, resolveFields } from \"./structured.js\";\nimport { aclVisible } from \"./tools/acl.js\";\n\nexport const MAX_LIST_LIMIT = 200;\nexport const MAX_COMPUTE_TEXT_CHARS = 2_000_000;\nexport const DEFAULT_COMPUTE_TIMEOUT = 30;\nexport const MAX_COMPUTE_DOCUMENTS = 50;\n\nconst TABULAR_MIME_PATTERNS = [\"%csv%\", \"%sheet%\", \"%excel%\", \"%spreadsheetml%\", \"%tab-separated%\"];\n\nconst UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;\n\nconst WORD_RE = /[^\\W\\d_]+/gu;\n\nconst STOPWORDS = new Set([\n \"the\",\n \"a\",\n \"an\",\n \"is\",\n \"are\",\n \"was\",\n \"were\",\n \"am\",\n \"be\",\n \"been\",\n \"being\",\n \"what\",\n \"whats\",\n \"who\",\n \"whom\",\n \"which\",\n \"how\",\n \"when\",\n \"where\",\n \"why\",\n \"this\",\n \"that\",\n \"these\",\n \"those\",\n \"there\",\n \"here\",\n \"of\",\n \"in\",\n \"on\",\n \"at\",\n \"to\",\n \"for\",\n \"with\",\n \"and\",\n \"or\",\n \"but\",\n \"not\",\n \"do\",\n \"does\",\n \"did\",\n \"done\",\n \"has\",\n \"have\",\n \"had\",\n \"having\",\n \"can\",\n \"could\",\n \"will\",\n \"would\",\n \"shall\",\n \"should\",\n \"may\",\n \"might\",\n \"must\",\n \"its\",\n \"your\",\n \"you\",\n \"about\",\n \"all\",\n \"any\",\n \"some\",\n \"like\",\n]);\n\nconst SHEET_MARKER_RE = /^\\[Sheet:\\s*(.+?)\\]\\s*$/;\nconst CODE_FENCE_RE = /^```(?:javascript|js|python)?\\s*\\n?|\\n?```\\s*$/gm;\n\nconst COMPUTE_SYSTEM_PROMPT = `You are a data analyst. Write JavaScript code that computes the answer to the user's instruction over the provided tables.\n\nAvailable in your code:\n- \\`dfs\\`: an object mapping sheet/document name to an array of row objects (one object per row, keys are column names)\n- \\`documents\\`: an array of objects (\\`id\\`, \\`name\\`, \\`sourceId\\`) — the source documents \\`dfs\\` was built from\n- Language: JSON, Math, Date, Array, Object, Number, String, Boolean, parseInt, parseFloat, isFinite, isNaN, console.log\n\nRules:\n1. Set a variable named \\`result\\` to the final answer (a number, string, array, or object).\n2. No file I/O, no network calls, no imports, no require, no process.\n3. Return ONLY the JavaScript code — no markdown fences, no explanation, no commentary.`;\n\nfunction parseDocumentId(documentId: unknown): string {\n const raw = String(documentId ?? \"\")\n .replace(/[ \\n\\t]/g, \"\")\n .trim();\n if (!UUID_RE.test(raw)) {\n throw new EngineActionError(`invalid document id: ${JSON.stringify(documentId)}`);\n }\n return raw;\n}\n\n/**\n * ACL visibility for one document row.\n *\n * - `principals=null` — trusted internal caller; every document is visible.\n * - `principals=[]` — anonymous; only rows whose `acl IS NULL` (unrestricted).\n * - a non-empty list — unrestricted rows plus those whose ACL overlaps.\n *\n * `[]` and `null` are NOT interchangeable. An empty ACL array on the row\n * (distinct from NULL) overlaps nothing, so it is invisible to anonymous\n * and named callers alike — only a trusted caller sees it.\n *\n * One rule, one implementation — `tools/acl.ts` is the canonical predicate\n * (its Python twin's docstring: a second, divergent implementation is\n * exactly how ACL semantics drift between surfaces). These names are\n * re-exports, never a reimplementation.\n */\nexport const visible = aclVisible;\n\n/** Python-named alias used by engine/tests that expect `_visible`. */\nexport const _visible = aclVisible;\n\nexport { aclVisible };\n\nexport function scopeSql(\n sourceIds: string[] | null | undefined,\n principals: string[] | null | undefined,\n params: unknown[],\n): string {\n const where: string[] = [];\n if (sourceIds != null) {\n params.push(sourceIds);\n where.push(`source_id = ANY($${params.length}::text[])`);\n }\n if (principals != null) {\n // `acl && $principals` — overlap-of-empty-array is FALSE for every\n // non-empty acl, which is exactly the `principals=[]` (anonymous\n // caller) semantics this package uses everywhere. NULL acl still\n // matches via `acl IS NULL`.\n params.push(principals);\n where.push(`(acl IS NULL OR acl && $${params.length}::text[])`);\n }\n return where.length ? `WHERE ${where.join(\" AND \")}` : \"\";\n}\n\nfunction parseCursor(cursorValue: unknown): Record<string, unknown> | null {\n if (cursorValue == null) return null;\n if (typeof cursorValue === \"object\" && !Array.isArray(cursorValue)) {\n return cursorValue as Record<string, unknown>;\n }\n if (typeof cursorValue === \"string\") {\n try {\n const parsed: unknown = JSON.parse(cursorValue);\n if (parsed && typeof parsed === \"object\" && !Array.isArray(parsed)) {\n return parsed as Record<string, unknown>;\n }\n } catch {\n /* fall through */\n }\n }\n console.warn(\"listDocuments: could not parse cursor value %s\", cursorValue);\n return null;\n}\n\n/**\n * Redact every string leaf of `value` — a bare string, or a JSON-ish\n * dict/list structure, walked recursively. Dict KEYS are never redacted\n * (they are field names; rewriting them would silently change the\n * result's schema).\n *\n * `policy=null` or an empty policy is an exact no-op — `value` is returned\n * unchanged, not even copied.\n */\nexport function redactValueRecursive(\n value: unknown,\n policy: RedactionPolicy | null | undefined,\n opts: { principals: string[] | null; secretKey: string | Buffer | null; hooks?: Hooks | null },\n): unknown {\n if (policy == null || policy.isEmpty()) return value;\n\n const walk = (node: unknown): unknown => {\n if (typeof node === \"string\") {\n return applyRedaction(node, policy, {\n phase: \"output\",\n principals: opts.principals,\n secretKey: opts.secretKey,\n hooks: opts.hooks,\n })[0];\n }\n if (Array.isArray(node)) return node.map(walk);\n if (node && typeof node === \"object\") {\n return Object.fromEntries(\n Object.entries(node as Record<string, unknown>).map(([k, v]) => [k, walk(v)]),\n );\n }\n return node;\n };\n return walk(value);\n}\n\nexport async function requireVisible(\n pool: Pool,\n documentId: string,\n principals: string[] | null,\n): Promise<void> {\n const id = parseDocumentId(documentId);\n const { rows } = await pool.query(`SELECT acl FROM context_engine_documents WHERE id = $1`, [id]);\n if (!rows[0] || !visible(rows[0].acl, principals)) {\n throw new DocumentNotFoundError(id);\n }\n}\n\nexport async function getDocumentRow(\n pool: Pool,\n documentId: string,\n principals: string[] | null,\n): Promise<Record<string, unknown>> {\n const id = parseDocumentId(documentId);\n const { rows } = await pool.query(`SELECT * FROM context_engine_documents WHERE id = $1`, [id]);\n const doc = rows[0];\n if (!doc || !visible(doc.acl, principals)) throw new DocumentNotFoundError(id);\n const count = await pool.query(\n `SELECT count(*)::int AS n FROM context_engine_chunks WHERE document_id = $1`,\n [id],\n );\n return {\n id: String(doc.id),\n sourceId: doc.source_id,\n externalId: doc.external_id,\n name: doc.name,\n description: doc.description,\n metaData: doc.meta_data ?? {},\n // `[]` (an explicitly empty principal list) is NOT the same as\n // NULL (unrestricted) — never collapse one into the other.\n acl: doc.acl != null ? [...doc.acl] : null,\n mode: doc.mode,\n mimeType: doc.mime_type,\n lang: doc.lang,\n text: doc.text,\n status: doc.status,\n error: doc.error,\n chunks: count.rows[0]?.n ?? 0,\n documentType: doc.document_type,\n structuredData: doc.structured_data,\n createdAt: doc.created_at,\n updatedAt: doc.updated_at,\n startedAt: doc.started_at,\n completedAt: doc.completed_at,\n unreadableReason: doc.meta_data?.unreadable_reason ?? null,\n unreadablePages: doc.meta_data?.unreadable_pages ?? 0,\n };\n}\n\nexport async function stats(pool: Pool, sourceId: string | null): Promise<Record<string, unknown>> {\n const where = sourceId != null ? \"WHERE source_id = $1\" : \"\";\n const params = sourceId != null ? [sourceId] : [];\n const docs = await pool.query(`SELECT count(*)::int AS n FROM context_engine_documents ${where}`, params);\n const chunks = await pool.query(\n `SELECT count(*)::int AS n FROM context_engine_chunks ${sourceId != null ? \"WHERE source_id = $1\" : \"\"}`,\n params,\n );\n const statuses = await pool.query(\n `SELECT status, count(*)::int AS n FROM context_engine_documents ${where} GROUP BY status`,\n params,\n );\n const meta = await pool.query(\n `SELECT embedding_provider, embedding_model, embedding_dim FROM context_engine_meta WHERE id = 1`,\n );\n const m = meta.rows[0];\n return {\n sourceId,\n documents: docs.rows[0]?.n ?? 0,\n chunks: chunks.rows[0]?.n ?? 0,\n byStatus: Object.fromEntries(statuses.rows.map((r) => [r.status, r.n])),\n embedding: {\n provider: m?.embedding_provider ?? null,\n model: m?.embedding_model ?? null,\n dim: m?.embedding_dim ?? null,\n },\n };\n}\n\n/**\n * Return one document's full text, ACL-checked.\n *\n * Unlike an LLM-tool payload helper, this returns the WHOLE text\n * untruncated — token-budget policy belongs to whatever calls this.\n *\n * Raises `EngineActionError` for an unparseable id and\n * `DocumentNotFoundError` for a missing document or one the caller's\n * `principals` can't see — the same message for the last two, deliberately.\n *\n * `redaction` (an `apply_at=\"output\"` policy) masks the fetched text\n * before it is returned. This is the ONLY protection for a corpus ingested\n * WITHOUT an `apply_at=\"ingest\"` policy.\n */\nexport async function getDocumentText(\n documentId: string,\n opts: {\n pool: Pool;\n principals?: string[] | null;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<string> {\n const principals = opts.principals ?? null;\n const doc = await getDocumentRow(opts.pool, documentId, principals);\n return redactValueRecursive(String(doc.text ?? \"\"), opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }) as string;\n}\n\n/**\n * Page through documents in scope, newest first.\n *\n * `redaction` masks `documentType` — unconstrained free text an LLM wrote\n * after reading the document body. `name`/`description` remain\n * intentionally unmasked: both are CALLER-set at ingest time, not text the\n * pipeline derived from the document body.\n */\n/**\n * One EXISTS: is anything in scope ingested with `mode: \"graph\"`?\n *\n * This is what a mode-less search reads to decide whether the graph leg is\n * worth building. Scoped exactly like the search itself — a linked-up document\n * in another source must not turn the leg on for a search that cannot see it.\n */\nexport async function hasGraphDocuments(\n pool: Pool,\n opts: {\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n principals?: string[] | null;\n },\n): Promise<boolean> {\n const params: unknown[] = [];\n const where = scopeSql(opts.sourceIds ?? null, opts.principals ?? null, params);\n let sql = `SELECT 1 FROM context_engine_documents ${where ? `${where} AND` : \"WHERE\"} mode = 'graph'`;\n if (opts.documentIds != null) {\n // `!= null`, NOT truthiness: an EMPTY list means nothing is in scope, and\n // collapsing it would ask about the whole corpus.\n params.push(opts.documentIds);\n sql += ` AND id = ANY($${params.length}::uuid[])`;\n }\n const { rows } = await pool.query(`${sql} LIMIT 1`, params);\n return rows.length > 0;\n}\n\nexport const MAX_CHUNKS_PER_READ = 25;\n\n/**\n * One LLM call per document, in-process. Far below the queue-backed numbers a\n * task runner can afford: this library has no task queue and is not growing\n * one, so the cap is what a request can honestly finish.\n */\nexport const MAX_MAP_REDUCE_DOCS = 200;\nexport const DEFAULT_MAP_REDUCE_DOCS = 25;\nexport const MAX_MAP_REDUCE_CONCURRENCY = 20;\n\n/**\n * Read one document in order, a range of chunks at a time.\n *\n * The document is ACL-checked FIRST, with the same one message an absent and a\n * forbidden document share: reading a document a chunk at a time must not be\n * the way around the ACL on the document itself.\n */\nexport async function getChunks(\n documentId: string,\n opts: {\n pool: Pool;\n principals?: string[] | null;\n start?: number | null;\n end?: number | null;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<Record<string, unknown>> {\n const parsedId = parseDocumentId(documentId);\n const principals = opts.principals ?? null;\n const first = Math.max(0, Math.trunc(opts.start ?? 0));\n const requestedEnd = opts.end == null ? first + MAX_CHUNKS_PER_READ - 1 : Math.trunc(opts.end);\n const last = Math.max(first, Math.min(requestedEnd, first + MAX_CHUNKS_PER_READ - 1));\n\n const visible = await opts.pool.query(\n `SELECT id FROM context_engine_documents\n WHERE id = $1::uuid AND ($2::text[] IS NULL OR acl IS NULL OR acl && $2::text[])`,\n [parsedId, principals],\n );\n if (!visible.rows.length) throw new DocumentNotFoundError(documentId);\n\n const totalRow = await opts.pool.query(\n `SELECT COUNT(*)::int AS total FROM context_engine_chunks WHERE document_id = $1::uuid`,\n [parsedId],\n );\n const total = Number(totalRow.rows[0]?.total ?? 0);\n const { rows } = await opts.pool.query(\n `SELECT idx, text, lang FROM context_engine_chunks\n WHERE document_id = $1::uuid AND idx >= $2 AND idx <= $3\n ORDER BY idx ASC`,\n [parsedId, first, last],\n );\n const chunks = rows.map((r) => ({\n position: r.idx,\n text: redactValueRecursive(r.text, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n language: r.lang,\n }));\n const nextStart = chunks.length ? Number(chunks[chunks.length - 1]?.position ?? first) + 1 : first;\n return {\n document_id: parsedId,\n total_chunks: total,\n start: first,\n end: last,\n chunks,\n has_more: nextStart < total,\n next_start: nextStart,\n };\n}\n\n/**\n * Ask the same question of every document in scope, one call each.\n *\n * The map step only: each document's answer comes back beside its name, and\n * the caller does the reducing. There is no second LLM call summarising the\n * summaries, because that is the step that invents a number nobody wrote.\n *\n * A document that fails is REPORTED, not thrown — one provider hiccup in a\n * fan-out of twenty must not throw away the nineteen that worked.\n */\nexport async function mapReduce(\n instruction: string,\n opts: {\n pool: Pool;\n config: ContextEngineConfig;\n principals?: string[] | null;\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n limit?: number | null;\n maxConcurrency?: number | null;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<Record<string, unknown>> {\n if (opts.config.llm == null) {\n throw new EngineActionError(\"map_reduce needs an LLM configured on this deployment\");\n }\n const principals = opts.principals ?? null;\n const n = Math.max(1, Math.min(Math.trunc(opts.limit || DEFAULT_MAP_REDUCE_DOCS), MAX_MAP_REDUCE_DOCS));\n const concurrency = Math.max(1, Math.min(Math.trunc(opts.maxConcurrency || 5), MAX_MAP_REDUCE_CONCURRENCY));\n\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n if (opts.documentIds != null) {\n // INTERSECTS with the source scope — a named id outside it can never\n // widen the set, only narrow it.\n const parsedIds: string[] = [];\n for (const did of opts.documentIds) {\n try {\n parsedIds.push(parseDocumentId(did));\n } catch {\n console.warn(\"map_reduce: skipping invalid document id %s\", did);\n }\n }\n params.push(parsedIds);\n where = where\n ? `${where} AND id = ANY($${params.length}::uuid[])`\n : `WHERE id = ANY($${params.length}::uuid[])`;\n }\n params.push(n);\n const { rows } = await opts.pool.query(\n `SELECT id::text, name, text FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n\n const results: Array<Record<string, unknown>> = new Array(rows.length);\n let cursor = 0;\n async function worker() {\n while (cursor < rows.length) {\n const index = cursor++;\n const row = rows[index];\n // Redaction runs BEFORE the text reaches the model, never after: the\n // provider is an egress, and masking the answer would be too late.\n const body = redactValueRecursive(String(row.text ?? \"\").slice(0, 20000), opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n });\n try {\n const [raw] = await callLlm(opts.config.llm as never, {\n system:\n \"You extract one answer from one document. Reply with STRICT JSON only. \" +\n 'If the document does not answer, reply {\"found\": false}.',\n user: `INSTRUCTION:\\n${instruction}\\n\\nDOCUMENT (${row.name}):\\n${body}`,\n jsonMode: true,\n temperature: 0,\n });\n let data: unknown;\n try {\n data = JSON.parse(raw);\n } catch {\n data = { raw };\n }\n results[index] = { document_id: row.id, document_name: row.name, data };\n } catch (err) {\n results[index] = {\n document_id: row.id,\n document_name: row.name,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n }\n }\n await Promise.all(Array.from({ length: Math.min(concurrency, rows.length) }, worker));\n\n const failed = results.filter((r) => r?.error).length;\n return {\n results,\n processed: results.length - failed,\n failed,\n considered: rows.length,\n };\n}\n\nexport async function listDocuments(opts: {\n pool: Pool;\n sourceId?: string | null;\n /** OR-of-many, ONE keyset-paged query across all of them; with `sourceId`, their union. */\n sourceIds?: string[] | null;\n principals?: string[] | null;\n cursor?: unknown;\n limit?: number;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n}): Promise<Record<string, unknown>> {\n const limit = Math.max(1, Math.min(Math.trunc(opts.limit || 50), MAX_LIST_LIMIT));\n const parsedCursor = parseCursor(opts.cursor);\n // `!= null`, NOT truthiness: no scoping given at all means the whole corpus,\n // but an EMPTY array means nothing is in scope, and collapsing the two would\n // turn a scoped listing into a listing of everything.\n const sourceIds =\n opts.sourceIds == null && opts.sourceId == null\n ? null\n : [...new Set([...(opts.sourceIds ?? []), ...(opts.sourceId != null ? [opts.sourceId] : [])])];\n const principals = opts.principals ?? null;\n\n const params: unknown[] = [];\n let where = scopeSql(sourceIds, principals, params);\n const cursorTime = parsedCursor?.time ?? parsedCursor?.created_at;\n const cursorId = parsedCursor?.id;\n if (cursorTime && cursorId && UUID_RE.test(String(cursorId))) {\n params.push(cursorTime, cursorId);\n const extra = `(created_at < $${params.length - 1} OR (created_at = $${params.length - 1} AND id < $${params.length}::uuid))`;\n where = where ? `${where} AND ${extra}` : `WHERE ${extra}`;\n }\n params.push(limit + 1);\n const { rows } = await opts.pool.query(\n `SELECT id, source_id, external_id, name, description, mode, mime_type, lang, status,\n document_type, created_at, updated_at, acl, meta_data\n FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n\n const hasMore = rows.length > limit;\n const page = rows.slice(0, limit);\n const documents = page.map((doc) => ({\n id: String(doc.id),\n sourceId: doc.source_id,\n externalId: doc.external_id,\n name: doc.name,\n description: doc.description,\n mode: doc.mode,\n mimeType: doc.mime_type,\n lang: doc.lang,\n status: doc.status,\n documentType: redactValueRecursive(doc.document_type, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n createdAt: doc.created_at instanceof Date ? doc.created_at.toISOString() : doc.created_at,\n updatedAt: doc.updated_at instanceof Date ? doc.updated_at.toISOString() : doc.updated_at,\n unreadableReason: doc.meta_data?.unreadable_reason ?? null,\n unreadablePages: doc.meta_data?.unreadable_pages ?? 0,\n // WHO can see this. Stored, enforced on every query and editable through\n // updateDocument — and, until this line, invisible to every caller,\n // because this serializer builds a FIXED object. `null` is UNRESTRICTED\n // and must stay null; an empty array would read as \"nobody\", which is the\n // opposite claim.\n acl: doc.acl ?? null,\n }));\n\n const result: Record<string, unknown> = {\n documents,\n count: documents.length,\n hasMore,\n };\n if (hasMore && page.length) {\n const last = page[page.length - 1]!;\n result.nextCursor = {\n time: last.created_at instanceof Date ? last.created_at.toISOString() : last.created_at,\n id: String(last.id),\n };\n }\n return result;\n}\n\nfunction candidateFieldTokens(question: string): string[] {\n const candidates: string[] = [];\n const seen = new Set<string>();\n\n const whole = normalizeKey(question);\n if (whole) {\n candidates.push(question.trim());\n seen.add(whole);\n }\n\n for (const match of question.matchAll(WORD_RE)) {\n const word = match[0]!;\n if (word.length < 3) continue;\n const norm = normalizeKey(word);\n if (!norm || seen.has(norm) || STOPWORDS.has(norm)) continue;\n seen.add(norm);\n candidates.push(word);\n }\n return candidates;\n}\n\n/**\n * Answer a natural-language question against documents' `structuredData`.\n *\n * When NOTHING resolves — every candidate fell below the similarity floor\n * and had no exact/synonym match either — this returns ZERO documents, not\n * the whole in-scope corpus. An off-topic question has no business getting\n * an invoice back just because the invoice happens to have SOME structured\n * fields.\n *\n * `redaction` masks each result's `documentType` and the STRING VALUES of\n * `structuredData`. KEYS are left unredacted — they are the canonical\n * field names this function's `resolveFields` matching keys off of.\n */\nexport async function queryStructured(\n question: string,\n opts: {\n pool: Pool;\n embedder: Embedder;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n docType?: string | null;\n limit?: number;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<Record<string, unknown>> {\n if (!question?.trim()) throw new EngineActionError(\"question must not be empty\");\n\n const limit = Math.max(1, Math.min(Math.trunc(opts.limit || 20), MAX_LIST_LIMIT));\n const principals = opts.principals ?? null;\n const candidates = candidateFieldTokens(question);\n const resolved = await resolveFields(candidates, {\n pool: opts.pool,\n embedder: opts.embedder,\n sourceIds: opts.sourceIds ?? null,\n docType: opts.docType ?? null,\n });\n const canonicalKeys = [...new Set(Object.values(resolved))].sort();\n\n if (!canonicalKeys.length) {\n return { question, resolvedFields: resolved, documents: [], count: 0 };\n }\n\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n const extras = [`structured_data IS NOT NULL`];\n if (opts.docType != null) {\n params.push(opts.docType);\n extras.push(`document_type = $${params.length}`);\n }\n params.push(canonicalKeys);\n extras.push(`structured_data ?| $${params.length}::text[]`);\n where = where ? `${where} AND ${extras.join(\" AND \")}` : `WHERE ${extras.join(\" AND \")}`;\n params.push(limit);\n\n const { rows } = await opts.pool.query(\n `SELECT id, source_id, name, document_type, structured_data\n FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n\n const documents = rows.map((doc) => ({\n id: String(doc.id),\n sourceId: doc.source_id,\n name: doc.name,\n documentType: redactValueRecursive(doc.document_type, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n structuredData: redactValueRecursive(doc.structured_data ?? {}, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n }));\n\n return { question, resolvedFields: resolved, documents, count: documents.length };\n}\n\nfunction stripCodeFences(text: string): string {\n return text.replace(CODE_FENCE_RE, \"\").trim();\n}\n\ntype DanfoMod = {\n readCSV: (input: string) => { columns: string[]; shape: [number, number]; values: unknown[][] };\n};\n\nasync function requireDanfo(): Promise<DanfoMod> {\n const specifier = \"danfojs-node\";\n try {\n return (await import(specifier)) as DanfoMod;\n } catch {\n throw new ExtraMissingError(\"compute\", \"danfojs-node\", \"compute() dataframes\");\n }\n}\n\n/**\n * Parse `[Sheet: name]`-delimited (or bare) CSV text into `{sheet: rows[]}`.\n * Unparseable sheets are skipped, never fatal.\n */\nfunction parseSpreadsheetText(\n dfd: { readCSV: (input: string) => { columns: string[]; shape: [number, number]; values: unknown[][] } },\n text: string | null | undefined,\n): Record<string, Record<string, unknown>[]> {\n if (!text) return {};\n const sheets: Record<string, string[]> = {};\n let current = \"sheet1\";\n for (const line of text.replace(/\\r\\n/g, \"\\n\").replace(/\\r/g, \"\\n\").split(\"\\n\")) {\n const m = SHEET_MARKER_RE.exec(line.trim());\n if (m) {\n current = m[1]!.trim() || current;\n sheets[current] ??= [];\n continue;\n }\n let bucket = sheets[current];\n if (!bucket) {\n bucket = [];\n sheets[current] = bucket;\n }\n bucket.push(line);\n }\n\n const out: Record<string, Record<string, unknown>[]> = {};\n for (const [name, lines] of Object.entries(sheets)) {\n const body = lines.join(\"\\n\").trim();\n if (!body) continue;\n try {\n const df = dfd.readCSV(body);\n const cols = (df.columns ?? []).map((c) => String(c).trim());\n if (!df.shape || df.shape[0] <= 0 || df.shape[1] <= 0) continue;\n const rows: Record<string, unknown>[] = [];\n for (const vals of df.values as unknown[][]) {\n const row: Record<string, unknown> = {};\n for (let i = 0; i < cols.length; i++) row[cols[i]!] = vals[i];\n rows.push(row);\n }\n if (rows.length) out[name] = rows;\n } catch (exc) {\n console.warn(\"compute: sheet '%s' not parseable as CSV: %s\", name, exc);\n }\n }\n return out;\n}\n\n/**\n * Compute an answer to `instruction` over in-scope spreadsheet documents.\n *\n * **Disabled by default** (`config.enableCodeExecution`): this executes\n * LLM-GENERATED code. Raises `EngineActionError` before ANY work — no LLM\n * call, no sandbox run — when the flag is `false`. That flag, not the\n * isolate, is the actual security boundary.\n *\n * Only the document half lives here: fetch the in-scope tabular documents\n * and parse their stored text into `{sheet: rows[]}`. Everything after that\n * — the guards, both redaction surfaces below, the prompt → code → sandbox\n * path and the swept result — is `computeOverFrames`, the seam a host calls\n * when it already holds the frames and has no document to point at; this\n * function builds its frames from documents and delegates.\n *\n * `config.redaction` is applied TWICE:\n * 1. Each document's raw `.text` is masked BEFORE it is parsed into a\n * table, so the data the LLM-authored code runs against is built from\n * masked cells.\n * 2. The final returned dict is swept whole through `redactValueRecursive`\n * — including `code`.\n */\nexport async function compute(\n instruction: string,\n opts: {\n pool: Pool;\n config: ContextEngineConfig;\n hooks: Hooks;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n documentIds?: string[] | null;\n modelCfg?: LLMConfig | null;\n timeout?: number;\n },\n): Promise<Record<string, unknown>> {\n // The same three guards `computeOverFrames` runs — repeated here so a\n // disabled flag, a missing LLM or a blank instruction is refused BEFORE\n // any DB work, not after the documents have been fetched and parsed.\n checkComputePreconditions(opts.config, opts.modelCfg, instruction);\n\n const dfd = await requireDanfo();\n const principals = opts.principals ?? null;\n\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n const mimeClause = TABULAR_MIME_PATTERNS.map((p) => {\n params.push(p);\n return `mime_type ILIKE $${params.length}`;\n }).join(\" OR \");\n where = where ? `${where} AND (${mimeClause})` : `WHERE (${mimeClause})`;\n if (opts.documentIds?.length) {\n const parsedIds: string[] = [];\n for (const did of opts.documentIds) {\n try {\n parsedIds.push(parseDocumentId(did));\n } catch {\n console.warn(\"compute: skipping invalid doc_id %s\", did);\n }\n }\n if (!parsedIds.length) {\n throw new EngineActionError(\"no tabular (CSV/TSV/XLSX) documents found in scope for compute()\");\n }\n params.push(parsedIds);\n where += ` AND id = ANY($${params.length}::uuid[])`;\n }\n params.push(MAX_COMPUTE_DOCUMENTS + 1);\n const { rows } = await opts.pool.query(\n `SELECT id, name, source_id, mime_type, text\n FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n const tabular = rows.filter((d) => isTabularMime(d.mime_type));\n if (!tabular.length) {\n throw new EngineActionError(\"no tabular (CSV/TSV/XLSX) documents found in scope for compute()\");\n }\n if (tabular.length > MAX_COMPUTE_DOCUMENTS) {\n throw new EngineActionError(\n `more than ${MAX_COMPUTE_DOCUMENTS} tabular documents are in scope for compute() — ` +\n \"narrow the request with documentIds or sourceIds\",\n );\n }\n\n const totalChars = tabular.reduce((n, d) => n + String(d.text ?? \"\").length, 0);\n if (totalChars > MAX_COMPUTE_TEXT_CHARS) {\n throw new EngineActionError(\n `in-scope spreadsheet text too large to load (${totalChars} chars > ` +\n `${MAX_COMPUTE_TEXT_CHARS} cap) — narrow the request with documentIds or sourceIds`,\n );\n }\n\n const dfs: Record<string, Record<string, unknown>[]> = {};\n const multiDoc = tabular.length > 1;\n for (const doc of tabular) {\n // `computeOverFrames` masks the frames again below. That second pass is\n // idempotent — a mask placeholder (`[EMAIL]`) or a hash (lowercase hex)\n // never re-matches a builtin detector — and it is kept, rather than\n // skipped for this path, because the seam must be safe for a host that\n // hands it RAW frames; the pre-parse mask here is kept because it is the\n // documented contract and is the only pass that sees the `[Sheet: …]`\n // markers that become the frame keys.\n const maskedText = redactValueRecursive(doc.text, opts.config.redaction, {\n principals,\n secretKey: opts.config.secretKey,\n hooks: opts.hooks,\n }) as string | null;\n const parsed = parseSpreadsheetText(dfd, maskedText);\n for (const [sheetName, table] of Object.entries(parsed)) {\n const key = multiDoc ? `${doc.name}:${sheetName}` : sheetName;\n dfs[key] = table;\n }\n }\n\n const documents = tabular.map((doc) => ({\n id: String(doc.id),\n name: doc.name,\n sourceId: doc.source_id,\n }));\n\n if (!Object.keys(dfs).length) {\n throw new EngineActionError(\"in-scope documents did not parse into any usable dataframe\");\n }\n\n return computeOverFrames(dfs, instruction, {\n config: opts.config,\n modelCfg: opts.modelCfg,\n timeout: opts.timeout,\n hooks: opts.hooks,\n principals,\n documents,\n });\n}\n\n/** What `compute()` builds and `computeOverFrames` consumes: sheet name → rows. */\nexport type ComputeFrames = Record<string, Record<string, unknown>[]>;\n\n/** A source document behind a frame, as `compute()` reports it in `documentsUsed`. */\nexport interface ComputeDocument {\n id: string;\n name?: string | null;\n sourceId?: string | null;\n}\n\n/**\n * The guards every compute entry point runs before doing ANY work.\n *\n * Order matters and is asserted by the tests: the code-execution flag is\n * checked first so a disabled deployment never even reaches the LLM.\n * Returns the LLM config to use (`modelCfg`, else `config.llm`).\n */\nfunction checkComputePreconditions(\n config: ContextEngineConfig,\n modelCfg: LLMConfig | null | undefined,\n instruction: string,\n): LLMConfig {\n if (!config.enableCodeExecution) {\n throw new EngineActionError(\n \"compute() executes generated code and is disabled by default; set \" +\n \"enableCodeExecution=true only in a deployment with out-of-process/\" +\n \"container isolation.\",\n );\n }\n\n const llmCfg = modelCfg ?? config.llm;\n if (llmCfg == null) {\n throw new Error(\"compute() requires an LLM: pass modelCfg= or configure ContextEngineConfig.llm\");\n }\n if (!instruction?.trim()) {\n throw new EngineActionError(\"instruction must not be empty\");\n }\n return llmCfg;\n}\n\n/**\n * `redactValueRecursive` for `{sheet: rows[]}`: mask every string CELL and\n * every COLUMN NAME, leave every other value untouched.\n *\n * Column names ARE masked here even though `redactValueRecursive` leaves\n * object keys alone: an object key is a field name the caller chose, but a\n * spreadsheet header is document content (a sheet of one column per\n * customer e-mail is an ordinary shape) and it reaches the LLM verbatim in\n * the schema summary. Frame KEYS (sheet names) are the caller's choice and\n * are not masked — same rule as object keys.\n *\n * Numbers, booleans and dates are never rewritten: masking works on text,\n * and every non-string value is carried across as it is, so the arithmetic\n * the LLM is about to write still finds numbers.\n *\n * Two headers that mask to the same label (`[EMAIL]` twice) cannot both be\n * keys of one row object, so the second and later get a positional suffix\n * (`[EMAIL]`, `[EMAIL]_2`, …): every column survives and none silently\n * overwrites another. The Python port keeps them as duplicate labels —\n * a DataFrame can hold those, a plain object cannot.\n *\n * `policy=null`/empty is an exact no-op — the caller's own arrays are\n * returned, not copies (the `redactValueRecursive` contract).\n */\nfunction maskFrames(\n frames: ComputeFrames,\n policy: RedactionPolicy | null | undefined,\n opts: { principals: string[] | null; secretKey: string | Buffer | null; hooks?: Hooks | null },\n): ComputeFrames {\n if (policy == null || policy.isEmpty()) return frames;\n const mask = (text: string): string => redactValueRecursive(text, policy, opts) as string;\n\n const masked: ComputeFrames = {};\n for (const [name, rows] of Object.entries(frames)) {\n // One header mapping per sheet, in first-seen column order, so every\n // row of that sheet is relabelled the same way.\n const labels = new Map<string, string>();\n const taken = new Set<string>();\n for (const row of rows) {\n for (const column of Object.keys(row)) {\n if (labels.has(column)) continue;\n const base = mask(column);\n let label = base;\n for (let n = 2; taken.has(label); n++) label = `${base}_${n}`;\n taken.add(label);\n labels.set(column, label);\n }\n }\n masked[name] = rows.map((row) =>\n Object.fromEntries(\n Object.entries(row).map(([column, value]) => [\n labels.get(column) ?? column,\n typeof value === \"string\" ? mask(value) : value,\n ]),\n ),\n );\n }\n return masked;\n}\n\n/**\n * Compute an answer to `instruction` over caller-supplied tables.\n *\n * The frame-level seam under `compute()`: everything `compute()` does AFTER\n * it has turned its documents into `{sheet: rows[]}` lives here, so a host\n * that already holds the frames — an uploaded workbook, a connector's\n * sheet, a query result — can run the same prompt → code → sandbox path\n * without first ingesting a document to point at.\n *\n * `frames` keys are the sheet names the caller chose; the LLM sees them and\n * each row's columns exactly as it sees a parsed document's.\n *\n * The guards are `compute()`'s, in the same order and all BEFORE any LLM\n * call: `config.enableCodeExecution` off → `EngineActionError`; no LLM\n * (`modelCfg` or `config.llm`) → `Error`; blank instruction or empty\n * `frames` → `EngineActionError`; `timeout` clamped to 1..300 seconds.\n *\n * `config.redaction` is applied at the same two surfaces as `compute()`:\n *\n * 1. Every string cell and every column name is masked BEFORE the prompt is\n * built (`maskFrames`), so neither the schema summary the LLM reads nor\n * the rows its code runs against carry a raw value. Same intended\n * trade-off as `compute()`'s point 1 — a masked cell can change a\n * computed result, and that is correct.\n * 2. The returned object is swept whole through `redactValueRecursive`,\n * `code` included — `compute()`'s point 2, deliberately blunt.\n *\n * `hooks`, `principals` and `documents` are how `compute()` threads its own\n * context through; a host calling the seam directly normally leaves them at\n * their defaults (no error hook, trusted-internal `unless` evaluation, no\n * source documents — `documentsUsed` comes back empty).\n */\nexport async function computeOverFrames(\n frames: ComputeFrames,\n instruction: string,\n opts: {\n config: ContextEngineConfig;\n modelCfg?: LLMConfig | null;\n timeout?: number;\n hooks?: Hooks | null;\n principals?: string[] | null;\n documents?: ComputeDocument[] | null;\n },\n): Promise<Record<string, unknown>> {\n const llmCfg = checkComputePreconditions(opts.config, opts.modelCfg, instruction);\n if (!frames || !Object.keys(frames).length) {\n throw new EngineActionError(\"no tabular data to compute over\");\n }\n\n const hooks = opts.hooks ?? {};\n const principals = opts.principals ?? null;\n const documents = [...(opts.documents ?? [])];\n const timeout = Math.max(1, Math.min(Math.trunc(opts.timeout || DEFAULT_COMPUTE_TIMEOUT), 300));\n const redactOpts = { principals, secretKey: opts.config.secretKey, hooks };\n\n // Point 1 of this function's docstring: the LLM never sees a raw value.\n const dfs = maskFrames(frames, opts.config.redaction, redactOpts);\n\n const schemaLines = Object.entries(dfs).map(\n ([name, table]) =>\n `- ${name}: columns=${JSON.stringify(Object.keys(table[0] ?? {}))}, rows=${table.length}`,\n );\n const userPrompt = `Instruction: ${instruction}\\n\\nAvailable dataframes:\\n${schemaLines.join(\"\\n\")}`;\n\n let rawCode: string;\n let tokens: { input: number; output: number };\n try {\n [rawCode, tokens] = await callLlm(llmCfg, {\n system: COMPUTE_SYSTEM_PROMPT,\n user: userPrompt,\n jsonMode: false,\n });\n } catch (exc) {\n emitError(hooks, exc, { stage: \"compute_codegen\", instruction: instruction.slice(0, 200) });\n throw exc;\n }\n\n const code = stripCodeFences(rawCode);\n const execResult = await executeSafeCode(code, { dfs, documents }, { timeout });\n if (!execResult.success) {\n let maskedCode: string;\n try {\n maskedCode = redactValueRecursive(code.slice(0, 500), opts.config.redaction, {\n principals,\n secretKey: opts.config.secretKey,\n hooks: hooks,\n }) as string;\n } catch {\n maskedCode = \"<redaction failed: code omitted>\";\n }\n emitError(hooks, new Error(execResult.error || \"compute execution failed\"), {\n stage: \"compute_exec\",\n code: maskedCode,\n });\n }\n\n const result: Record<string, unknown> = {\n success: execResult.success,\n result: execResult.result,\n code,\n stdout: execResult.stdout ?? \"\",\n error: execResult.error,\n executionTime: execResult.executionTime,\n documentsUsed: documents.map((d) => d.id),\n providerTokens: {\n llm_input: tokens?.input ?? 0,\n llm_output: tokens?.output ?? 0,\n },\n };\n return redactValueRecursive(result, opts.config.redaction, {\n principals,\n secretKey: opts.config.secretKey,\n hooks: hooks,\n }) as Record<string, unknown>;\n}\n\nexport { applyAttributeUpdates } from \"./ingest.js\";\n\nexport function requirePandas(): never {\n throw new ExtraMissingError(\"compute\", \"danfojs-node\", \"compute() dataframes\");\n}\n","/**\n * Isolated-vm JavaScript sandbox — the compute-tool's execution engine.\n *\n * THIS IS BEST-EFFORT, NOT A SECURITY BOUNDARY — read before relying on it.\n *\n * `actions.compute()` (the only caller of this module) is **disabled by\n * default** (`ContextEngineConfig.enableCodeExecution`, default `false`).\n * That flag, not anything in this file, is the actual security boundary:\n * no deployment ships reachable code execution unless an operator\n * explicitly turns it on, and the only responsible reason to turn it on is\n * having put REAL isolation around the process that calls `compute()` — a\n * subprocess with dropped privileges, a container, gVisor, a WASM runtime.\n * Everything below raises the cost of an escape; none of it is a substitute\n * for that.\n *\n * The isolate has:\n * - a wall-clock timeout (`script.run({ timeout })`)\n * - no Node `fs` / `net` / `child_process` / `process` (the isolate does\n * not receive them; user code that names them is rejected up front)\n * - a whitelist of language builtins (JSON, Math, Date, Array, …)\n *\n * Returns `{ result, stdout? }` plus success/error/executionTime so a\n * failed run is a data result, not an thrown exception — matching the\n * Python sandbox's contract that `compute()` inspects `success`.\n */\nimport { CodeExecutionError, ExtraMissingError } from \"./errors.js\";\n\ntype IsolatedVm = {\n Isolate: new (opts: {\n memoryLimit: number;\n }) => {\n createContext(): Promise<{\n global: { set(k: string, v: unknown): Promise<void>; derefInto(): unknown };\n eval(code: string): Promise<unknown>;\n }>;\n compileScript(code: string): Promise<{\n run(ctx: unknown, opts: { timeout: number; copy: boolean }): Promise<unknown>;\n }>;\n dispose(): void;\n };\n};\n\nexport const DEFAULT_TIMEOUT = 30;\nexport const MAX_TIMEOUT = 300;\nexport const DEFAULT_MAX_OUTPUT_LENGTH = 50_000;\nexport const ISOLATE_MEMORY_MB = 128;\n\nconst BLOCKED_IDENTIFIERS =\n /\\b(process|require|module|exports|globalThis|global|Buffer|__dirname|__filename|child_process|worker_threads|fs|net|http|https|dgram|dns|tls|cluster|os|vm|v8|crypto|fetch|WebAssembly|Function|eval|importScripts|SharedArrayBuffer|Atomics|XMLHttpRequest|WebSocket|Worker)\\b/;\n\nconst DUNDER = /__proto__|constructor\\s*\\[|constructor\\s*\\./;\n\nexport interface SandboxResult {\n success: boolean;\n result: unknown;\n stdout?: string;\n stderr?: string;\n executionTime: number;\n error: string | null;\n}\n\nexport async function ensureSandboxAvailable(): Promise<IsolatedVm> {\n const specifier = \"isolated-vm\";\n try {\n return (await import(specifier)) as IsolatedVm;\n } catch {\n throw new ExtraMissingError(\"compute\", \"isolated-vm\", \"sandboxed compute()\");\n }\n}\n\n/** Static reject of blocked identifiers. Returns `[ok, error]`. */\nexport function isSafeCode(code: string): [boolean, string] {\n if (!code?.trim()) return [false, \"empty code\"];\n if (BLOCKED_IDENTIFIERS.test(code)) {\n return [false, \"code references a blocked identifier (fs/net/process/require/eval/…)\"];\n }\n if (DUNDER.test(code)) {\n return [false, \"code references a blocked constructor/prototype path\"];\n }\n if (/\\bimport\\s*\\(|^\\s*import\\s/m.test(code) || /\\bexport\\s/.test(code)) {\n return [false, \"ESM import/export is not allowed in the sandbox\"];\n }\n return [true, \"\"];\n}\n\nexport function validateCodeOnly(code: string): void {\n const [ok, err] = isSafeCode(code);\n if (!ok) throw new CodeExecutionError(err);\n}\n\n/**\n * Execute JavaScript in an isolated-vm isolate.\n *\n * `context` values are copied in (plain JSON-able data only). The isolate\n * does not receive Node's `fs`/`net`/`process`. A `result` binding is\n * expected; `console.log` is captured into `stdout`.\n *\n * Timeout is milliseconds of isolate CPU, clamped to 1–300 seconds.\n * Does not throw on user-code failure — returns `success: false`.\n */\nexport async function executeSafeCode(\n code: string,\n context: Record<string, unknown> = {},\n timeoutOrOpts: number | { timeout?: number; maxOutputLength?: number } = DEFAULT_TIMEOUT,\n): Promise<SandboxResult> {\n const timeoutSec =\n typeof timeoutOrOpts === \"number\" ? timeoutOrOpts : (timeoutOrOpts.timeout ?? DEFAULT_TIMEOUT);\n const maxOutputLength =\n typeof timeoutOrOpts === \"number\"\n ? DEFAULT_MAX_OUTPUT_LENGTH\n : (timeoutOrOpts.maxOutputLength ?? DEFAULT_MAX_OUTPUT_LENGTH);\n const timeout = Math.max(1, Math.min(timeoutSec, MAX_TIMEOUT));\n\n const empty = (error: string, executionTime = 0): SandboxResult => ({\n success: false,\n result: null,\n stdout: \"\",\n stderr: \"\",\n executionTime,\n error,\n });\n\n const [ok, err] = isSafeCode(code);\n if (!ok) {\n console.warn(\"[sandbox] Unsafe code detected: %s\", err);\n return empty(`Code validation failed: ${err}`);\n }\n\n const ivm = await ensureSandboxAvailable();\n const started = Date.now();\n const isolate = new ivm.Isolate({ memoryLimit: ISOLATE_MEMORY_MB });\n try {\n const jail = await isolate.createContext();\n const jailGlobal = jail.global;\n await jailGlobal.set(\"global\", jailGlobal.derefInto());\n\n // Whitelist: language builtins already exist inside the isolate.\n // Do NOT copy Node's process/require/fs. Copy caller context as JSON.\n let contextJson: string;\n try {\n contextJson = JSON.stringify(context ?? {});\n } catch (exc) {\n return empty(`context is not JSON-serializable: ${String(exc)}`);\n }\n\n await jail.eval(\n `const __ctx = ${contextJson};\n for (const k of Object.keys(__ctx)) { global[k] = __ctx[k]; }\n global.__stdout = [];\n global.console = {\n log: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n info: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n warn: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n error: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n };\n void 0;`,\n );\n\n const wrapped = `\"use strict\";\nvar result = null;\n${code}\n({ result: result, stdout: (global.__stdout || []).join(\"\\\\n\") })`;\n\n const script = await isolate.compileScript(wrapped);\n const out = (await script.run(jail, {\n timeout: timeout * 1000,\n copy: true,\n })) as { result?: unknown; stdout?: string } | null;\n\n const executionTime = (Date.now() - started) / 1000;\n let result = out?.result ?? null;\n let stdout = typeof out?.stdout === \"string\" ? out.stdout : \"\";\n if (stdout.length > maxOutputLength) stdout = stdout.slice(0, maxOutputLength);\n try {\n const serialized = JSON.stringify(result);\n if (serialized && serialized.length > maxOutputLength) {\n result = { _truncated: true, _original_size: serialized.length };\n }\n } catch {\n result = String(result).slice(0, maxOutputLength);\n }\n\n return {\n success: true,\n result,\n stdout,\n stderr: \"\",\n executionTime,\n error: null,\n };\n } catch (exc) {\n const executionTime = (Date.now() - started) / 1000;\n const msg = String(exc);\n if (/timed out|timeout/i.test(msg)) {\n return empty(`Code execution timeout after ${timeout} seconds`, executionTime);\n }\n return empty(msg, executionTime);\n } finally {\n isolate.dispose();\n }\n}\n","/**\n * Structured data extraction, key normalization, and field resolution.\n *\n * extract — extractStructuredData() — one LLM call for type + fields\n * normalize— normalizeKey / inferDataType / validateAndNormalize / extractSynonyms\n * register — upsertRegistry() — the learned field-name registry, with embeddings\n * resolve — resolveFields() — question wording to canonical field name\n *\n * extractStructuredData never special-cases a provider: by the time ingest\n * calls it, extraction has already turned the document into text, so this\n * always runs on that text through callLlm. It never throws — an LLM\n * failure or unparseable response comes back as quality=\"failed\".\n *\n * resolveFields: exact key_norm match → synonym-array match → nearest\n * registry embedding by cosine distance, gated by\n * FIELD_RESOLUTION_SIMILARITY_FLOOR (0.55). Unresolved candidates are\n * simply absent from the returned mapping.\n */\nimport { randomUUID } from \"node:crypto\";\nimport type { Pool } from \"pg\";\nimport type { LLMConfig } from \"./config.js\";\nimport { safeJsonParse } from \"./extraction/json.js\";\nimport type { Embedder } from \"./providers/embeddings.js\";\nimport { callLlm } from \"./providers/llm.js\";\nimport { charScript } from \"./text.js\";\n\nexport const MAX_KEY_LENGTH = 100;\nexport const MAX_VALUE_LENGTH = 500;\nexport const MAX_EXAMPLE_VALUES = 10;\n\n/** Extraction schema version stamped on `documents.extraction_version`. */\nexport const EXTRACTION_VERSION = \"ce-structured-v1\";\n\n/** Cap on how much document text is sent to the LLM for one extraction call. */\nexport const MAX_EXTRACTION_CHARS = 120_000;\n\n/**\n * Cosine-similarity floor for the embedding fallback leg: below this,\n * \"nearest\" does not mean \"related\" — with no floor at all, EVERY candidate\n * resolved to SOME field once the registry had at least one embedded row.\n * Similarity = 1 - cosine distance (pgvector's `<=>`). 0.55 is deliberately\n * conservative for short-label vs short-label comparison.\n */\nexport const FIELD_RESOLUTION_SIMILARITY_FLOOR = 0.55;\n\nexport interface ExtractionResult {\n documentType: string | null;\n structuredDataRaw: Record<string, unknown>;\n structuredData: Record<string, unknown>;\n keysRaw: string[];\n keysNormalized: string[];\n quality: \"high\" | \"medium\" | \"low\" | \"failed\" | string;\n error: string | null;\n providerTokens: Record<string, number>;\n}\n\nconst EXTRACTION_SYSTEM_PROMPT = `You are an extraction engine. Analyze the document text you are given and return ONLY valid JSON (no markdown, no commentary, no backticks).\n\nGOALS:\nA) Extract structured information as FLAT KEY-VALUE pairs (industry-agnostic).\nB) Identify document_type and provide a brief quality observation.\n\nHARD RULES:\n1) Output MUST be STRICT JSON matching the schema below (no extra top-level keys).\n2) structured_data_raw MUST be a flat key-value object — no nested objects or arrays.\n If the source has lists/tables, flatten with indexed dot keys: items.0.description, items.0.value, items.1.description, ...\n3) Unknown/unreadable values MUST be null (never guess).\n4) Numeric fields: only use a number when you can read it from the text; store the raw string too if it needs disambiguation.\n5) Always include document_type (a short descriptive type, or \"unknown\").\n\nRETURN JSON SCHEMA (EXACT):\n{\n \"document_type\": \"<descriptive_type | unknown>\",\n \"structured_data_raw\": {\"<key>\": \"<value or null>\"},\n \"structured_quality\": {\"readability\": \"<low|medium|high>\"}\n}`;\n\nfunction buildExtractionPrompt(\n fieldHints: Record<string, unknown>[] | null | undefined,\n attempt = 0,\n): string {\n let prompt = EXTRACTION_SYSTEM_PROMPT;\n if (fieldHints?.length) {\n const lines: string[] = [];\n for (const hint of fieldHints) {\n if (!hint || typeof hint !== \"object\") continue;\n const key = hint.key ?? hint.name;\n if (!key) continue;\n const desc = hint.description ?? hint.type;\n lines.push(`- ${String(key)}${desc ? ` (${String(desc)})` : \"\"}`);\n }\n if (lines.length) {\n prompt +=\n \"\\n\\nPRIORITIZE extracting these fields if present (use these exact \" +\n \"names as keys in structured_data_raw), in addition to anything else \" +\n \"you find:\\n\" +\n lines.join(\"\\n\");\n }\n }\n if (attempt > 0) {\n prompt +=\n \"\\n\\nIMPORTANT: The previous attempt was not valid JSON. Double-check \" +\n \"escaping of quotes/newlines and return JSON only.\";\n }\n return prompt;\n}\n\n/**\n * Extract structured key/value data from already-extracted document text.\n *\n * Never throws — an LLM failure or unparseable response comes back as\n * `ExtractionResult(quality=\"failed\", error=...)` so ingest can complete\n * the rest of ingestion instead of losing already-computed chunks.\n *\n * Tokens are tracked CUMULATIVELY across every attempt whose `callLlm`\n * actually completed, even one whose response then failed to parse —\n * that attempt still consumed real, billable provider tokens.\n */\nexport async function extractStructuredData(\n text: string,\n opts: {\n llmCfg: LLMConfig;\n fieldHints?: Record<string, unknown>[] | null;\n maxChars?: number;\n maxRetries?: number;\n },\n): Promise<ExtractionResult> {\n if (!text?.trim()) {\n return {\n documentType: null,\n structuredDataRaw: {},\n structuredData: {},\n keysRaw: [],\n keysNormalized: [],\n quality: \"failed\",\n error: \"empty document text\",\n providerTokens: {},\n };\n }\n\n const maxChars = opts.maxChars ?? MAX_EXTRACTION_CHARS;\n const body = text.length <= maxChars ? text : text.slice(0, maxChars);\n const maxRetries = opts.maxRetries ?? 1;\n let lastError: string | null = null;\n let totalInputTokens = 0;\n let totalOutputTokens = 0;\n\n for (let attempt = 0; attempt <= maxRetries; attempt++) {\n const prompt = buildExtractionPrompt(opts.fieldHints, attempt);\n let raw: string;\n let tokens: { input: number; output: number };\n try {\n [raw, tokens] = await callLlm(opts.llmCfg, {\n system: prompt,\n user: `Document text:\\n\\n${body}`,\n jsonMode: true,\n });\n } catch (exc) {\n lastError = String(exc);\n console.warn(\"structured extraction attempt %d: callLlm failed: %s\", attempt, exc);\n continue;\n }\n\n totalInputTokens += tokens?.input ?? 0;\n totalOutputTokens += tokens?.output ?? 0;\n\n let parsed: Record<string, unknown>;\n try {\n parsed = safeJsonParse(raw, \"object\") as Record<string, unknown>;\n } catch (exc) {\n lastError = String(exc);\n console.warn(\"structured extraction attempt %d: unparseable response: %s\", attempt, exc);\n continue;\n }\n\n const docType = (parsed.document_type as string) || \"unknown\";\n let rawData = parsed.structured_data_raw;\n if (!rawData || typeof rawData !== \"object\" || Array.isArray(rawData)) rawData = {};\n const qualityInfo = parsed.structured_quality;\n const readability =\n qualityInfo && typeof qualityInfo === \"object\" && !Array.isArray(qualityInfo)\n ? (qualityInfo as { readability?: string }).readability\n : undefined;\n const quality =\n readability === \"high\" || readability === \"medium\" || readability === \"low\" ? readability : \"medium\";\n\n const normalized = validateAndNormalize(rawData as Record<string, unknown>);\n\n return {\n documentType: docType,\n structuredDataRaw: rawData as Record<string, unknown>,\n structuredData: normalized,\n keysRaw: Object.keys(rawData as Record<string, unknown>),\n keysNormalized: Object.keys(normalized),\n quality,\n error: null,\n providerTokens: { llm_input: totalInputTokens, llm_output: totalOutputTokens },\n };\n }\n\n return {\n documentType: null,\n structuredDataRaw: {},\n structuredData: {},\n keysRaw: [],\n keysNormalized: [],\n quality: \"failed\",\n error: lastError || \"extraction failed\",\n providerTokens: { llm_input: totalInputTokens, llm_output: totalOutputTokens },\n };\n}\n\n/**\n * True if `ch` should be KEPT verbatim by `normalizeKey`.\n *\n * Unicode-aware: an ASCII `[^a-z0-9_]` class silently DROPPED every\n * non-Latin field name — \"税额\"/\"مبلغ الضريبة\"/\"Сумма\" all normalized to\n * `\"\"` and got skipped entirely. `charScript` reads a character's Unicode\n * category (letters AND combining marks, any script) instead of a\n * hardcoded Latin range. `/\\d/u` is Unicode-aware on its own.\n */\nfunction isKeyChar(ch: string): boolean {\n return Boolean(charScript(ch)) || /\\d/u.test(ch) || ch === \"_\";\n}\n\n/**\n * Normalize a metadata key: lowercase, non-key-chars -> `_`, dot-paths\n * kept, collapsed/stripped underscores, truncated to `MAX_KEY_LENGTH`.\n *\n * Unicode letters/digits in ANY script are preserved verbatim (only\n * separators/punctuation become `_`) — see `isKeyChar`.\n */\nexport function normalizeKey(key: unknown): string {\n if (!key || typeof key !== \"string\") return \"\";\n\n const segments = key.toLowerCase().split(\".\");\n const processed: string[] = [];\n for (const raw of segments) {\n let segment = \"\";\n for (const ch of raw) segment += isKeyChar(ch) ? ch : \"_\";\n segment = segment.replace(/_+/g, \"_\").replace(/^_|_$/g, \"\");\n if (segment) processed.push(segment);\n }\n let normalized = processed.join(\".\");\n if (normalized.length > MAX_KEY_LENGTH) normalized = normalized.slice(0, MAX_KEY_LENGTH);\n return normalized;\n}\n\n/** Infer 'number' | 'date' | 'bool' | 'string' from a key's suffix, then the value itself. */\nexport function inferDataType(key: string, value: unknown): string {\n if (typeof key !== \"string\") return \"string\";\n\n const keyLower = key.toLowerCase();\n if (keyLower.endsWith(\"_num\") || keyLower.endsWith(\"_number\")) return \"number\";\n if (keyLower.endsWith(\"_date\") || keyLower.endsWith(\"_datetime\")) return \"date\";\n if (keyLower.endsWith(\"_bool\") || keyLower.endsWith(\"_boolean\")) return \"bool\";\n\n if (value == null) return \"string\";\n if (typeof value === \"boolean\") return \"bool\";\n if (typeof value === \"string\") {\n const valueLower = value.toLowerCase().trim();\n if ([\"true\", \"false\", \"yes\", \"no\", \"1\", \"0\"].includes(valueLower)) return \"bool\";\n try {\n const d = new Date(value.replace(\"Z\", \"+00:00\"));\n if (!Number.isNaN(d.getTime()) && /\\d{4}-\\d{2}-\\d{2}/.test(value)) return \"date\";\n } catch {\n /* not a date */\n }\n if (value !== \"\" && !Number.isNaN(Number(value))) return \"number\";\n }\n if (typeof value === \"number\") return \"number\";\n if (value instanceof Date) return \"date\";\n return \"string\";\n}\n\n/**\n * Normalize keys, validate/convert values by inferred type, reject nested\n * structures, truncate over-long strings.\n */\nexport function validateAndNormalize(rawData: Record<string, unknown>): Record<string, unknown> {\n if (!rawData || typeof rawData !== \"object\" || Array.isArray(rawData)) {\n throw new Error(\"Input must be a dictionary\");\n }\n\n const normalized: Record<string, unknown> = {};\n\n for (const [rawKey, rawValue] of Object.entries(rawData)) {\n if (rawValue && typeof rawValue === \"object\") {\n console.warn(\"skipping nested structure for key '%s'\", rawKey);\n continue;\n }\n\n const normKey = normalizeKey(rawKey);\n if (!normKey) continue;\n\n const dataType = inferDataType(normKey, rawValue);\n\n try {\n if (dataType === \"number\") {\n if (rawValue == null || rawValue === \"\") {\n normalized[normKey] = null;\n } else {\n const asNum = Number(rawValue);\n if (!Number.isNaN(asNum)) {\n normalized[normKey] = asNum;\n } else {\n const match = String(rawValue).match(/[-+]?\\d[\\d,]*\\.?\\d*/);\n if (match) {\n const n = Number(match[0].replace(/,/g, \"\"));\n normalized[normKey] = Number.isNaN(n) ? null : n;\n } else {\n normalized[normKey] = null;\n }\n }\n }\n } else if (dataType === \"date\") {\n if (rawValue == null || rawValue === \"\") {\n normalized[normKey] = null;\n } else if (rawValue instanceof Date) {\n normalized[normKey] = rawValue.toISOString();\n } else {\n const d = new Date(String(rawValue).replace(\"Z\", \"+00:00\"));\n normalized[normKey] = Number.isNaN(d.getTime()) ? String(rawValue) : d.toISOString();\n }\n } else if (dataType === \"bool\") {\n if (rawValue == null || rawValue === \"\") {\n normalized[normKey] = null;\n } else if (typeof rawValue === \"boolean\") {\n normalized[normKey] = rawValue;\n } else {\n const valueStr = String(rawValue).toLowerCase().trim();\n if ([\"true\", \"yes\", \"1\"].includes(valueStr)) normalized[normKey] = true;\n else if ([\"false\", \"no\", \"0\"].includes(valueStr)) normalized[normKey] = false;\n else normalized[normKey] = String(rawValue);\n }\n } else {\n if (rawValue == null) {\n normalized[normKey] = null;\n } else {\n let strValue = String(rawValue);\n if (strValue.length > MAX_VALUE_LENGTH) strValue = strValue.slice(0, MAX_VALUE_LENGTH);\n normalized[normKey] = strValue;\n }\n }\n } catch (exc) {\n console.error(\"error processing field '%s': %s\", rawKey, exc);\n normalized[normKey] = rawValue != null ? String(rawValue) : null;\n }\n }\n\n return normalized;\n}\n\n/** Build `{normKey: [synonym variations]}` from raw/normalized key pairs. */\nexport function extractSynonyms(rawKeys: string[], normKeys: string[]): Record<string, string[]> {\n if (rawKeys.length !== normKeys.length) {\n console.error(\"extractSynonyms: length mismatch raw=%d norm=%d\", rawKeys.length, normKeys.length);\n return {};\n }\n\n const synonymMap: Record<string, string[]> = {};\n\n for (let i = 0; i < rawKeys.length; i++) {\n const rawKey = rawKeys[i]!;\n const normKey = normKeys[i]!;\n if (!normKey) continue;\n synonymMap[normKey] ??= [];\n const synonyms = new Set<string>();\n\n if (rawKey) {\n synonyms.add(rawKey);\n synonyms.add(rawKey.toLowerCase());\n synonyms.add(rawKey.toUpperCase());\n const camel = rawKey.replace(/[ _.]/g, \"\");\n if (camel) {\n synonyms.add(camel[0]!.toLowerCase() + camel.slice(1));\n synonyms.add(camel[0]!.toUpperCase() + camel.slice(1));\n }\n }\n\n synonyms.add(normKey);\n synonyms.add(normKey.replace(/_/g, \"\"));\n synonyms.add(normKey.replace(/_/g, \" \"));\n synonyms.add(normKey.replace(/\\./g, \"_\"));\n synonyms.add(normKey.replace(/_/g, \".\"));\n synonyms.add(normKey.toUpperCase());\n synonyms.add(\n normKey\n .split(\"_\")\n .map((w) => (w ? w[0]!.toUpperCase() + w.slice(1) : w))\n .join(\" \"),\n );\n\n synonyms.delete(\"\");\n synonyms.delete(normKey);\n\n const seen = new Set<string>();\n const unique: string[] = [];\n for (const syn of [...[...synonyms].sort(), ...synonymMap[normKey]!]) {\n if (!seen.has(syn)) {\n seen.add(syn);\n unique.push(syn);\n }\n }\n synonymMap[normKey] = unique;\n }\n\n return synonymMap;\n}\n\nfunction embeddingText(keyNorm: string, synonyms: readonly string[]): string {\n return `${keyNorm} ${synonyms.join(\" \")}`.trim();\n}\n\nfunction vectorLiteral(vector: readonly number[]): string {\n return `[${vector.map((x) => Number(x)).join(\",\")}]`;\n}\n\ninterface RegistryRow {\n id: string;\n key_norm: string;\n synonyms: string[] | null;\n embedding: unknown;\n example_values: string[] | null;\n frequency: number | null;\n}\n\n/**\n * Upsert `structuredData`'s keys into the field registry.\n *\n * Generates a field-name embedding (`Embedder.embed(..., kind=\"document\")`)\n * for any NEW field, or one whose synonym set changed, or one that somehow\n * has no embedding yet — batched into one `embed()` call per ingest rather\n * than one per field.\n *\n * Returns the embedding provider's token count (0 if nothing needed embedding).\n */\nexport async function upsertRegistry(opts: {\n pool: Pool;\n embedder: Embedder;\n sourceId: string | null;\n docType: string;\n structuredData: Record<string, unknown>;\n keysRaw: string[];\n keysNormalized: string[];\n}): Promise<number> {\n if (!opts.structuredData || !opts.docType) return 0;\n\n const synonymsMap = extractSynonyms(opts.keysRaw, opts.keysNormalized);\n const toEmbed: Array<[string, string]> = [];\n\n for (const [keyNorm, value] of Object.entries(opts.structuredData)) {\n const dataType = inferDataType(keyNorm, value);\n const exampleValue = value == null ? null : String(value).slice(0, 200);\n const newSynonyms = [\n ...new Set((synonymsMap[keyNorm] ?? []).map((s) => normalizeKey(s)).filter(Boolean)),\n ].sort();\n\n const existing = await opts.pool.query<RegistryRow>(\n `SELECT id, key_norm, synonyms, embedding, example_values, frequency\n FROM context_engine_structured_keys\n WHERE doc_type = $1 AND key_norm = $2\n AND (($3::text IS NULL AND source_id IS NULL) OR source_id = $3)\n LIMIT 1`,\n [opts.docType, keyNorm, opts.sourceId],\n );\n\n let rowId: string;\n let needsEmbedding = false;\n let synonymsForEmbed: string[] = newSynonyms;\n\n if (!existing.rows[0]) {\n rowId = randomUUID();\n await opts.pool.query(\n `INSERT INTO context_engine_structured_keys\n (id, source_id, doc_type, key_norm, data_type, synonyms, example_values, frequency)\n VALUES ($1,$2,$3,$4,$5,$6,$7,1)`,\n [\n rowId,\n opts.sourceId,\n opts.docType,\n keyNorm,\n dataType,\n newSynonyms,\n exampleValue ? [exampleValue] : [],\n ],\n );\n needsEmbedding = true;\n } else {\n const row = existing.rows[0];\n rowId = row.id;\n const freq = (row.frequency ?? 0) + 1;\n let exampleValues = [...(row.example_values ?? [])];\n if (exampleValue && !exampleValues.includes(exampleValue)) {\n exampleValues = [...exampleValues.slice(0, MAX_EXAMPLE_VALUES - 1), exampleValue];\n }\n const currentSyn = [...(row.synonyms ?? [])].sort();\n let merged = currentSyn;\n if (newSynonyms.length) {\n merged = [...new Set([...currentSyn, ...newSynonyms])].sort();\n if (merged.join(\"\\0\") !== currentSyn.join(\"\\0\")) needsEmbedding = true;\n }\n if (row.embedding == null) needsEmbedding = true;\n synonymsForEmbed = merged;\n await opts.pool.query(\n `UPDATE context_engine_structured_keys\n SET frequency = $1, example_values = $2, synonyms = $3, updated_at = now()\n WHERE id = $4`,\n [freq, exampleValues, merged, rowId],\n );\n }\n\n if (needsEmbedding) toEmbed.push([rowId, embeddingText(keyNorm, synonymsForEmbed)]);\n }\n\n if (!toEmbed.length) return 0;\n\n const [vectors, tokens] = await opts.embedder.embed(\n toEmbed.map(([, text]) => text),\n { kind: \"document\" },\n );\n for (let i = 0; i < toEmbed.length; i++) {\n const vec = vectors[i];\n if (!vec) continue;\n await opts.pool.query(\n `UPDATE context_engine_structured_keys SET embedding = $1::vector, updated_at = now() WHERE id = $2`,\n [vectorLiteral(vec), toEmbed[i]![0]],\n );\n }\n return tokens || 0;\n}\n\n/**\n * Resolve free-text field-name candidates to canonical registry keys.\n *\n * Cheapest-first: normalize → exact `key_norm` match → synonym-array\n * match → nearest registry embedding (cosine, gated by\n * FIELD_RESOLUTION_SIMILARITY_FLOOR). Unresolved candidates are simply\n * absent from the returned `{candidate: canonicalKey}` mapping.\n *\n * All still-unresolved candidates are embedded in ONE batched `embed()` call.\n *\n * No ACL: registry rows are field NAMES, not document content, so source_id\n * scoping is enough.\n */\nexport async function resolveFields(\n candidates: readonly string[],\n opts: {\n pool: Pool;\n embedder: Embedder;\n sourceIds?: string[] | null;\n docType?: string | null;\n },\n): Promise<Record<string, string>> {\n const normByCandidate: Record<string, string> = {};\n for (const c of candidates) {\n const n = normalizeKey(c);\n if (n) normByCandidate[c] = n;\n }\n if (!Object.keys(normByCandidate).length) return {};\n\n const { rows } = await opts.pool.query<{\n key_norm: string;\n synonyms: string[] | null;\n embedding: unknown;\n }>(\n `SELECT key_norm, synonyms, embedding\n FROM context_engine_structured_keys\n WHERE ($1::text[] IS NULL OR source_id = ANY($1))\n AND ($2::text IS NULL OR doc_type = $2)`,\n [opts.sourceIds ?? null, opts.docType ?? null],\n );\n\n const resolved: Record<string, string> = {};\n const unresolved: Array<[string, string]> = [];\n for (const [candidate, norm] of Object.entries(normByCandidate)) {\n const exact = rows.find((r) => r.key_norm === norm);\n if (exact) {\n resolved[candidate] = exact.key_norm;\n continue;\n }\n const synonymHit = rows.find((r) => (r.synonyms ?? []).includes(norm));\n if (synonymHit) {\n resolved[candidate] = synonymHit.key_norm;\n continue;\n }\n unresolved.push([candidate, norm]);\n }\n\n if (unresolved.length && rows.some((r) => r.embedding != null)) {\n let vectors: number[][] = [];\n try {\n [vectors] = await opts.embedder.embed(\n unresolved.map(([, norm]) => norm),\n { kind: \"query\" },\n );\n } catch (exc) {\n console.warn(\"resolveFields: batch embedding failed: %s\", exc);\n vectors = [];\n }\n\n for (let i = 0; i < unresolved.length; i++) {\n const vec = vectors[i];\n if (!vec) continue;\n const match = await nearestField(opts.pool, vec, opts.sourceIds ?? null, opts.docType ?? null);\n if (match) resolved[unresolved[i]![0]] = match;\n }\n }\n\n return resolved;\n}\n\nasync function nearestField(\n pool: Pool,\n vector: readonly number[],\n sourceIds: string[] | null,\n docType: string | null,\n): Promise<string | null> {\n const { rows } = await pool.query<{ key_norm: string; distance: number }>(\n `SELECT key_norm, (embedding <=> $1::vector) AS distance\n FROM context_engine_structured_keys\n WHERE embedding IS NOT NULL\n AND ($2::text[] IS NULL OR source_id = ANY($2))\n AND ($3::text IS NULL OR doc_type = $3)\n ORDER BY embedding <=> $1::vector\n LIMIT 1`,\n [vectorLiteral(vector), sourceIds, docType],\n );\n const row = rows[0];\n if (!row) return null;\n const similarity = 1.0 - Number(row.distance);\n if (similarity < FIELD_RESOLUTION_SIMILARITY_FLOOR) return null;\n return String(row.key_norm);\n}\n","/**\n * Sentinels distinguishing omitted arguments from real values, including null.\n *\n * UNSET: \"argument omitted\" vs null (e.g. updateDocument acl=null means unrestricted).\n * TRUSTED: trusted caller, ACL filtering disabled. Truthy on purpose so\n * `if (principals)` does not treat a trusted caller as anonymous.\n */\n\nexport const UNSET: unique symbol = Symbol.for(\"context_engine.UNSET\");\nexport type Unset = typeof UNSET;\n\n// Symbol.for (the process-global registry), NEVER a class instance: the\n// package ships per-entry bundles (tsup splitting:false), so dist/index.js\n// and dist/mcp.js each carry their own copy of this module. A class instance\n// is a different object per copy, and `result === TRUSTED` then fails for an\n// app importing TRUSTED from one entry while mounting another — every tool\n// call on a correctly configured trusted mount errors. UNSET above survives\n// bundle duplication for exactly this reason; TRUSTED must too.\nexport const TRUSTED: unique symbol = Symbol.for(\"context_engine.TRUSTED\");\n\n// The SCOPE ceiling's \"no ceiling at all\". Symbol.for for the same reason\n// TRUSTED is: a class instance breaks `===` across per-entry bundles.\n//\n// Scope has the same shape of danger `principals` has, so it gets the same\n// treatment. A host maps its own word — project, matter, workspace, customer —\n// onto a list of source ids, and that ceiling is what the mounted tool may\n// ever reach. `null`/omitted is the spelling of an oversight, and would\n// otherwise mean a corpus-wide search: exactly the accident worth making\n// impossible. A host running one shared private corpus types UNSCOPED and\n// thereby says it meant it.\nexport const UNSCOPED: unique symbol = Symbol.for(\"context_engine.UNSCOPED\");\nexport type Trusted = typeof TRUSTED;\n\nexport type Principals = string[] | null | typeof TRUSTED | undefined;\n\n// Once per method, matching Python's default warnings filter — before the\n// dedup, a legacy caller re-triggered the warning on EVERY request.\nconst warnedMethods = new Set<string>();\n\nexport function resolvePrincipals(value: Principals, method: string): string[] | null {\n if (value === TRUSTED) return null;\n if (value === undefined || value === null) {\n // `undefined` (omitted) is the same legacy trusted spelling as `null` —\n // Python's omitted `principals` defaults to None and warns identically.\n if (!warnedMethods.has(method)) {\n warnedMethods.add(method);\n process.emitWarning(\n `${method}(principals=${value === null ? \"null\" : \"undefined\"}) means TRUSTED CALLER — ` +\n `access control is disabled and every document is returned. If that is what ` +\n `you want, pass principals=TRUSTED (from @promptev/context-engine) to say so ` +\n `explicitly. If you meant 'no authenticated user', pass principals=[] instead. ` +\n `Passing null/omitting will raise in 1.0.`,\n { type: \"DeprecationWarning\", code: \"CE_PRINCIPALS_NULL\" },\n );\n }\n return null;\n }\n if (!Array.isArray(value)) {\n // NEVER coerce: a bare string (or any other object) coercing to null\n // used to mean full-corpus TRUSTED — an ACL bypass a caller could hit\n // with one wrong type. Fail loudly instead.\n throw new TypeError(\n `${method}(principals=...) must be an array of principal strings, [] for an ` +\n `anonymous caller, or TRUSTED — got ${typeof value}. A non-array value must ` +\n `never silently disable ACL filtering.`,\n );\n }\n return value;\n}\n\nexport function isTrusted(principals: string[] | null | undefined): boolean {\n return principals === null || principals === undefined;\n}\n","/**\n * The ingestion pipeline: bytes/text in, document + chunk rows out.\n *\n * `ContextEngine.ingest()` is a thin shell over `runIngest()` here. One call\n * ingests ONE document through these stages:\n *\n * resolve input → extract → hygiene → dedup check → claim row (queued →\n * processing) → chunk → embed → upsert chunks → graph stage → complete\n *\n * Three rules worth stating up front:\n *\n * 1. **Dedup is by `ingest_hash` + mode.** Re-ingesting identical content at\n * the same-or-lower retrieval mode returns a `skipped` report row, costs\n * zero units, and does not re-process the content. Re-ingesting at a\n * HIGHER mode (hybrid → graph) re-processes fully so the graph stage can\n * run. **`skipped` is about CONTENT, not about the caller's declared\n * attributes**: `acl` / `name` / `description` / `metaData` are not part\n * of the hash, so a skip still applies them (`refreshOnSkip`).\n * 2. **The database is locked to one embedding model.**\n * 3. **Per-document failures are reported, not raised.** Only pre-flight\n * errors (bad arguments, graph disabled, embedding-model mismatch) raise.\n * Hash-redaction misconfig is the exception: finalize to `failed` THEN\n * rethrow so the row is not stuck `processing`.\n * 4. **`batch=true` stops after chunking.** Chunks are stored with\n * `embedding IS NULL`, the document lands at `status=\"batch_pending\"`.\n */\nimport { randomUUID } from \"node:crypto\";\nimport { readFileSync } from \"node:fs\";\nimport { basename } from \"node:path\";\nimport type { Pool } from \"pg\";\nimport {\n calculateContentHash,\n chunkerName,\n dedupLines,\n detectLanguage,\n inferMime,\n looksMostlyBoilerplate,\n routeToChunker,\n sanitizeText,\n sha256Bytes,\n} from \"./chunkers.js\";\nimport type { ContextEngineConfig } from \"./config.js\";\nimport { extract } from \"./extraction/index.js\";\nimport { emitError, emitProgress, emitUsage, type Hooks, type ProgressEvent } from \"./hooks.js\";\nimport { submitEmbeddingBatch } from \"./providers/batch.js\";\nimport type { Embedder } from \"./providers/embeddings.js\";\nimport { applyRedaction, type RedactionPolicy } from \"./redaction.js\";\nimport type { ChunkRow, StorageBackend } from \"./storage.js\";\nimport {\n extractStructuredData,\n EXTRACTION_VERSION as STRUCTURED_EXTRACTION_VERSION,\n upsertRegistry,\n} from \"./structured.js\";\nimport { type DocumentReport, type IngestReport, type UsageEvent, unitsForFile } from \"./usage.js\";\n\nexport type Mode = \"hybrid\" | \"graph\";\n\n/** Retrieval modes are ordered: ingesting at a higher mode than the stored one is a backfill. */\nexport const MODE_RANK: Record<string, number> = { hybrid: 0, graph: 1 };\n\nexport const EMBED_BATCH = 256;\nexport const DOCX_CHARS_PER_PAGE = 1500;\nexport const DEFAULT_MIME = \"text/plain\";\n\nconst DOCX_MIMES = new Set([\n \"application/vnd.openxmlformats-officedocument.wordprocessingml.document\",\n \"application/msword\",\n]);\n\nconst EXTRA_EXT_MIMES: Record<string, string> = {\n \".pdf\": \"application/pdf\",\n \".docx\": \"application/vnd.openxmlformats-officedocument.wordprocessingml.document\",\n \".pptx\": \"application/vnd.openxmlformats-officedocument.presentationml.presentation\",\n \".png\": \"image/png\",\n \".jpg\": \"image/jpeg\",\n \".jpeg\": \"image/jpeg\",\n \".gif\": \"image/gif\",\n \".bmp\": \"image/bmp\",\n \".tif\": \"image/tiff\",\n \".tiff\": \"image/tiff\",\n \".webp\": \"image/webp\",\n \".txt\": \"text/plain\",\n \".tsv\": \"text/tab-separated-values\",\n};\n\nconst MIME_BY_EXT: Record<string, string> = { ...EXTRA_EXT_MIMES };\n\nexport interface IngestRequest {\n content?: Buffer | null;\n filename?: string | null;\n text?: string | null;\n name?: string | null;\n description?: string | null;\n sourceId?: string | null;\n externalId?: string | null;\n metaData?: Record<string, unknown> | null;\n acl?: string[] | null;\n mode?: Mode;\n batch?: boolean;\n extractStructured?: boolean;\n fieldHints?: Record<string, unknown>[] | null;\n}\n\nexport interface Prepared {\n text: string;\n mime: string;\n pages: number | null;\n slides: number | null;\n sizeBytes: number;\n llmPictures: number;\n mediaOnly: boolean;\n providerTokens: Record<string, number>;\n contentHash: string | null;\n isMarkdown: boolean;\n // Carried from `Extracted` so the document plane can persist and report\n // it. null for a caller-supplied `text` ingest: nothing was extracted, so\n // there is nothing to have failed.\n unreadableReason: string | null;\n unreadablePages: number;\n}\n\nexport interface Decision {\n action: \"process\" | \"skip\" | \"upgrade\";\n documentId: string | null;\n}\n\ninterface ContentPayload {\n text: string;\n mime: string;\n lang: string | null;\n ingestHash: Buffer;\n mode: string;\n metaUpdates: Record<string, unknown>;\n documentType?: string | null;\n structuredData?: Record<string, unknown> | null;\n structuredKeys?: string[] | null;\n structuredQuality?: Record<string, unknown> | null;\n extractionVersion?: string | null;\n}\n\nexport interface GraphStageArgs {\n documentId: string;\n chunks: ChunkRow[] | null;\n config: ContextEngineConfig;\n hooks: Hooks;\n}\n\nexport type GraphStage = (args: GraphStageArgs) => Promise<number> | number;\n\nfunction isUniqueViolation(exc: unknown): boolean {\n return (\n typeof exc === \"object\" && exc !== null && \"code\" in exc && (exc as { code?: string }).code === \"23505\"\n );\n}\n\nexport function mimeForFilename(filename: string | null | undefined): string | null {\n if (!filename) return null;\n const upstream = inferMime(null, filename);\n if (upstream) return upstream;\n const dot = filename.lastIndexOf(\".\");\n const ext = dot >= 0 ? filename.slice(dot).toLowerCase() : \"\";\n if (ext && MIME_BY_EXT[ext]) return MIME_BY_EXT[ext]!;\n return null;\n}\n\n/**\n * Normalize the three input modes to `[content, filename, text]`.\n *\n * Exactly one of `file` / `content` / `text` must be given — passing none\n * or more than one is a programming error, so it raises before any work.\n */\nexport function resolveSource(opts: {\n file?: unknown;\n content?: Buffer | Uint8Array | null;\n filename?: string | null;\n text?: string | null;\n}): [Buffer | null, string | null, string | null] {\n const provided = (\n [\n [\"file\", opts.file],\n [\"content\", opts.content],\n [\"text\", opts.text],\n ] as const\n ).filter(([, v]) => v != null);\n if (provided.length !== 1) {\n throw new Error(\n \"exactly one of file=, content= or text= must be provided \" +\n `(got: ${provided.map(([n]) => n).join(\", \") || \"none\"})`,\n );\n }\n\n if (opts.text != null) return [null, opts.filename ?? null, opts.text];\n if (opts.content != null) {\n const buf = Buffer.isBuffer(opts.content) ? opts.content : Buffer.from(opts.content);\n return [buf, opts.filename ?? null, null];\n }\n\n const file = opts.file;\n if (typeof file === \"string\") {\n const data = readFileSync(file);\n return [data, opts.filename || basename(file), null];\n }\n if (\n file &&\n typeof file === \"object\" &&\n \"buffer\" in (file as object) &&\n Buffer.isBuffer((file as { buffer: Buffer }).buffer)\n ) {\n const f = file as { buffer: Buffer; originalname?: string; name?: string };\n return [f.buffer, opts.filename || f.originalname || f.name || null, null];\n }\n if (Buffer.isBuffer(file)) return [file, opts.filename ?? null, null];\n if (file && typeof file === \"object\" && typeof (file as { read?: unknown }).read === \"function\") {\n const data = (file as { read: () => Buffer | string | Uint8Array }).read();\n const buf = typeof data === \"string\" ? Buffer.from(data, \"utf8\") : Buffer.from(data);\n const handleName = (file as { name?: unknown }).name;\n const inferred = typeof handleName === \"string\" ? basename(handleName) : null;\n return [buf, opts.filename || inferred, null];\n }\n throw new Error(\"file= must be a path or an object with a .read() method\");\n}\n\n/** Produce the document's final text + the numbers billing/chunking need. */\nexport async function prepare(opts: {\n content: Buffer | null;\n filename: string | null;\n text: string | null;\n name: string | null;\n config: ContextEngineConfig;\n hooks: Hooks;\n}): Promise<Prepared> {\n if (opts.text != null) {\n const clean = sanitizeText(opts.text) || \"\";\n const mime = inferMime(clean, opts.filename || opts.name) || DEFAULT_MIME;\n return {\n text: clean,\n mime,\n pages: null,\n slides: null,\n sizeBytes: Buffer.byteLength(clean, \"utf8\"),\n llmPictures: 0,\n mediaOnly: false,\n providerTokens: {},\n contentHash: calculateContentHash({ fullText: clean }),\n isMarkdown: false,\n unreadableReason: null,\n unreadablePages: 0,\n };\n }\n\n const content = opts.content!;\n const fname = opts.filename || opts.name || \"\";\n const mime = mimeForFilename(fname) || \"\";\n const extracted = await extract(content, fname, mime || null, {\n visionLlm: opts.config.visionLlm,\n extraction: opts.config.extraction,\n hooks: opts.hooks,\n });\n\n let body = sanitizeText(extracted.text || \"\") || \"\";\n const effectiveMime = mime || inferMime(body, fname) || DEFAULT_MIME;\n // PDFs are the one input whose extraction concatenates pages and therefore\n // repeats headers/footers verbatim, so they get the boilerplate-line dedup\n // — but NOT when the extraction produced Markdown (vision / pdf-inspector):\n // the same heuristic strips table header/separator rows after the first\n // page and flattens fenced code and nested lists.\n if (effectiveMime === \"application/pdf\" && body && !extracted.isMarkdown) body = dedupLines(body);\n\n return {\n text: body,\n mime: effectiveMime,\n pages: extracted.pages ?? null,\n slides: extracted.slides ?? null,\n sizeBytes: content.length,\n llmPictures: extracted.llmPictures ?? 0,\n mediaOnly: Boolean(extracted.mediaOnly),\n providerTokens: { ...(extracted.providerTokens ?? {}) },\n contentHash: calculateContentHash({ docBytes: content }),\n isMarkdown: Boolean(extracted.isMarkdown),\n unreadableReason: extracted.unreadableReason ?? null,\n unreadablePages: extracted.unreadablePages ?? 0,\n };\n}\n\nexport function computeUnits(prepared: Prepared): number {\n let units = unitsForFile(prepared.mime, {\n pages: prepared.pages,\n slides: prepared.slides,\n sizeBytes: prepared.sizeBytes,\n });\n if (DOCX_MIMES.has(prepared.mime) && !prepared.pages && prepared.text) {\n units = Math.max(1, Math.ceil(prepared.text.length / DOCX_CHARS_PER_PAGE));\n }\n return units + Math.max(0, prepared.llmPictures);\n}\n\nfunction mergeFailed(failed: string[] | null | undefined, note: { rules_failed?: string[] }): void {\n if (!failed) return;\n for (const name of note.rules_failed ?? []) {\n if (!failed.includes(name)) failed.push(name);\n }\n}\n\n/**\n * Redact `text` for the DOCUMENT plane using `phase=\"ingest\"` rules.\n *\n * This is the document-level counterpart to the per-chunk redaction inside\n * `buildChunks`: it is what `runIngest` calls to compute `documentText` —\n * the value used for `Document.text`, `Document.lang`, and the input to\n * structured extraction. Without this, those document-plane values kept\n * deriving from the raw `prepared.text`, so an `apply_at=\"ingest\"` policy's\n * \"never-store\" promise held for chunks but not for the document row\n * `getDocument()` / `getDocumentText()` serve whole.\n *\n * `null`/empty `policy` is an exact no-op.\n */\nexport function redactDocumentText(\n text: string,\n policy: RedactionPolicy | null | undefined,\n secretKey: string | Buffer | null,\n hooks?: Hooks | null,\n failed?: string[] | null,\n): string {\n if (policy == null || policy.isEmpty()) return text;\n const [redacted, note] = applyRedaction(text, policy, {\n phase: \"ingest\",\n secretKey,\n hooks,\n });\n mergeFailed(failed, note);\n return redacted;\n}\n\n/**\n * Redact the STRING VALUES of one chunk's structural metadata.\n *\n * KEYS are never touched — they're field names (`section_title`,\n * `sheet_title`, ...), and redacting one would corrupt the result schema,\n * not its content. FLAT walk, deliberately — same trade-off as\n * `search.redactHits`' meta loop.\n */\nexport function redactChunkMeta(\n meta: Record<string, unknown>,\n policy: RedactionPolicy | null | undefined,\n secretKey: string | Buffer | null,\n hooks?: Hooks | null,\n failed?: string[] | null,\n): Record<string, unknown> {\n if (policy == null || policy.isEmpty()) return meta;\n const redacted: Record<string, unknown> = {};\n for (const [key, value] of Object.entries(meta)) {\n if (typeof value === \"string\") {\n const [nv, note] = applyRedaction(value, policy, { phase: \"ingest\", secretKey, hooks });\n mergeFailed(failed, note);\n redacted[key] = nv;\n } else {\n redacted[key] = value;\n }\n }\n return redacted;\n}\n\n/**\n * Route text through the structure-aware chunkers → `ChunkRow`s.\n *\n * When `policy` carries ingest-phase rules, each chunk is redacted BEFORE\n * the `ChunkRow` is built — so the stored text, the detected language, and\n * (later) the embedding all derive from the masked string. An embedding of\n * unmasked text would itself be a partial leak.\n */\nexport function buildChunks(\n prepared: Prepared,\n name: string | null | undefined,\n opts: {\n policy?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks | null;\n failed?: string[] | null;\n } = {},\n): [ChunkRow[], string] {\n const [chunks, effectiveMime, chunkMetas] = routeToChunker(prepared.text, prepared.mime, name, {\n isMarkdown: prepared.isMarkdown,\n });\n let kind = chunkerName(effectiveMime);\n if (prepared.isMarkdown && kind === \"generic\") kind = \"markdown\";\n\n const rows: ChunkRow[] = [];\n for (let idx = 0; idx < chunks.length; idx++) {\n let clean = sanitizeText(chunks[idx]!);\n if (opts.policy != null && !opts.policy.isEmpty()) {\n const [redacted, note] = applyRedaction(clean, opts.policy, {\n phase: \"ingest\",\n secretKey: opts.secretKey,\n hooks: opts.hooks,\n });\n mergeFailed(opts.failed, note);\n clean = redacted;\n }\n if (!clean?.trim()) continue;\n const meta: Record<string, unknown> = { chunker: kind };\n if (idx < chunkMetas.length && chunkMetas[idx] && Object.keys(chunkMetas[idx]!).length) {\n Object.assign(\n meta,\n redactChunkMeta(chunkMetas[idx]!, opts.policy, opts.secretKey ?? null, opts.hooks, opts.failed),\n );\n }\n rows.push({ idx: rows.length, text: clean, lang: detectLanguage(clean), meta });\n }\n return [rows, effectiveMime || prepared.mime || DEFAULT_MIME];\n}\n\n/** Embed `rows` in place (batched). Returns the provider token count. */\nexport async function embedChunks(embedder: Embedder, rows: ChunkRow[]): Promise<number> {\n if (!rows.length) return 0;\n let totalTokens = 0;\n for (let start = 0; start < rows.length; start += EMBED_BATCH) {\n const batch = rows.slice(start, start + EMBED_BATCH);\n const [vectors, tokens] = await embedder.embed(\n batch.map((r) => r.text),\n { kind: \"document\" },\n );\n if (vectors.length !== batch.length) {\n throw new Error(`embedding provider returned ${vectors.length} vectors for ${batch.length} inputs`);\n }\n for (let i = 0; i < batch.length; i++) batch[i]!.embedding = [...vectors[i]!];\n totalTokens += tokens || 0;\n }\n return totalTokens;\n}\n\n/** Bind this database to one embedding provider/model, or raise. */\nexport async function checkEmbeddingLock(pool: Pool, config: ContextEngineConfig): Promise<void> {\n const provider = config.embedding.provider;\n const model = config.embedding.model;\n const { rows } = await pool.query(\n `SELECT embedding_provider, embedding_model, embedding_dim FROM context_engine_meta WHERE id = 1`,\n );\n let row = rows[0];\n if (!row) {\n await pool.query(`INSERT INTO context_engine_meta (id) VALUES (1) ON CONFLICT (id) DO NOTHING`);\n row = { embedding_provider: null, embedding_model: null, embedding_dim: null };\n }\n if (row.embedding_model == null && row.embedding_provider == null) {\n await pool.query(\n `UPDATE context_engine_meta\n SET embedding_provider = $1, embedding_model = $2,\n embedding_dim = COALESCE(embedding_dim, $3)\n WHERE id = 1`,\n [provider, model, config.embedding.dim ?? null],\n );\n return;\n }\n if (row.embedding_model !== model || row.embedding_provider !== provider) {\n throw new Error(\n \"embedding model mismatch: this database was ingested with \" +\n `provider=${JSON.stringify(row.embedding_provider)} model=${JSON.stringify(row.embedding_model)}, but the engine ` +\n `is configured with provider=${JSON.stringify(provider)} model=${JSON.stringify(model)}. Re-embedding an existing ` +\n \"corpus is not automatic — use a separate database for a different embedding model.\",\n );\n }\n}\n\n/** Record (or validate) the observed vector width after the first embed. */\nexport async function recordEmbeddingDim(pool: Pool, dim: number | null | undefined): Promise<void> {\n if (dim == null) return;\n const { rows } = await pool.query(`SELECT embedding_dim FROM context_engine_meta WHERE id = 1`);\n const row = rows[0];\n if (!row) return;\n if (row.embedding_dim == null) {\n await pool.query(`UPDATE context_engine_meta SET embedding_dim = $1 WHERE id = 1`, [dim]);\n } else if (Number(row.embedding_dim) !== dim) {\n throw new Error(\n `embedding dimension mismatch: this database's vector columns are ${row.embedding_dim}-wide ` +\n `(set by \\`context-engine migrate --dim\\`), but the configured embedding model returned ` +\n `${dim}-dimensional vectors.`,\n );\n }\n}\n\nasync function loadExisting(\n pool: Pool,\n opts: {\n sourceId: string | null;\n externalId: string | null;\n name: string | null;\n ingestHash: Buffer;\n },\n): Promise<Record<string, unknown> | null> {\n const scope = opts.sourceId == null ? `source_id IS NULL` : `source_id = $1`;\n const params: unknown[] = opts.sourceId == null ? [] : [opts.sourceId];\n let extra: string;\n if (opts.externalId != null) {\n params.push(opts.externalId);\n extra = `external_id = $${params.length}`;\n } else if (opts.name != null) {\n params.push(opts.name);\n extra = `name = $${params.length}`;\n } else {\n params.push(opts.ingestHash);\n extra = `ingest_hash = $${params.length}`;\n }\n const { rows } = await pool.query(\n `SELECT * FROM context_engine_documents WHERE ${scope} AND ${extra} ORDER BY created_at ASC LIMIT 1`,\n params,\n );\n return rows[0] ?? null;\n}\n\n/**\n * Compare an incoming hash against the stored document's.\n *\n * Called TWICE per ingest, with different hashes, on purpose:\n *\n * - `hashKind=\"content\"` runs BEFORE extraction, on the MD5 of the raw\n * bytes. An unchanged file must not re-pay for vision/OCR extraction.\n * - `hashKind=\"ingest\"` runs after extraction, on the SHA-256 of the\n * normalized text, and is the authoritative check.\n *\n * `skipped` rows (media-only / no extractable text) are terminal too,\n * EXCEPT one whose `meta_data.unreadable_reason` is set (a scan that needs\n * vision, or whose vision extraction failed): re-ingesting the same bytes\n * must re-extract, since the documented remedy — configure a vision model\n * and re-ingest — has no other way to take effect.\n */\nexport async function decide(\n pool: Pool,\n opts: {\n request: IngestRequest;\n hashKind: \"content\" | \"ingest\";\n hashValue: Buffer | string | null;\n },\n): Promise<Decision> {\n if (opts.hashValue == null) return { action: \"process\", documentId: null };\n if (opts.hashKind === \"content\" && opts.request.externalId == null && opts.request.name == null) {\n return { action: \"process\", documentId: null };\n }\n\n const existing = await loadExisting(pool, {\n sourceId: opts.request.sourceId ?? null,\n externalId: opts.request.externalId ?? null,\n name: opts.request.name ?? null,\n ingestHash:\n opts.hashKind === \"ingest\" && Buffer.isBuffer(opts.hashValue) ? opts.hashValue : Buffer.alloc(0),\n });\n if (!existing) return { action: \"process\", documentId: null };\n\n const existingId = String(existing.id);\n if (existing.status !== \"completed\" && existing.status !== \"skipped\") {\n return { action: \"process\", documentId: existingId };\n }\n\n let stored: unknown;\n let incoming: unknown;\n if (opts.hashKind === \"ingest\") {\n stored = existing.ingest_hash != null ? Buffer.from(existing.ingest_hash as Buffer) : null;\n incoming = Buffer.isBuffer(opts.hashValue) ? opts.hashValue : Buffer.from(opts.hashValue as string);\n } else {\n stored = (existing.meta_data as Record<string, unknown> | null)?.content_hash;\n incoming = opts.hashValue;\n }\n\n const same =\n stored != null &&\n incoming != null &&\n (Buffer.isBuffer(stored) && Buffer.isBuffer(incoming)\n ? Buffer.compare(stored, incoming) === 0\n : stored === incoming);\n\n if (!same) return { action: \"process\", documentId: existingId };\n if (existing.status === \"skipped\") {\n const metaData = existing.meta_data as Record<string, unknown> | null;\n if (metaData?.unreadable_reason) return { action: \"process\", documentId: existingId };\n return { action: \"skip\", documentId: existingId };\n }\n const reqMode = opts.request.mode ?? \"hybrid\";\n if ((MODE_RANK[reqMode] ?? 0) <= (MODE_RANK[(existing.mode as string) || \"hybrid\"] ?? 0)) {\n return { action: \"skip\", documentId: existingId };\n }\n return { action: \"upgrade\", documentId: existingId };\n}\n\n/**\n * Insert-or-reuse the row and move it `queued` → `processing`.\n *\n * The two states are committed separately and in order, so `queued` is\n * genuinely observable by a concurrent reader rather than being an\n * in-memory value that only ever hits the database as `processing`.\n *\n * Only caller-declared attributes are written here. Everything derived\n * from the content lands at completion.\n */\nexport async function claimDocument(\n pool: Pool,\n opts: { request: IngestRequest; documentId: string | null },\n): Promise<string> {\n const client = await pool.connect();\n try {\n let id = opts.documentId;\n const req = opts.request;\n if (id) {\n const { rows } = await client.query(`SELECT id FROM context_engine_documents WHERE id = $1`, [id]);\n if (!rows[0]) id = null;\n }\n if (!id) {\n id = randomUUID();\n await client.query(\n `INSERT INTO context_engine_documents\n (id, source_id, external_id, name, description, acl, status, error, started_at, completed_at)\n VALUES ($1,$2,$3,$4,$5,$6,'queued',NULL,NULL,NULL)`,\n [\n id,\n req.sourceId ?? null,\n req.externalId ?? null,\n req.name ?? null,\n req.description ?? null,\n req.acl ?? null,\n ],\n );\n } else {\n await client.query(\n `UPDATE context_engine_documents\n SET source_id=$1, external_id=$2, name=$3, description=$4, acl=$5,\n error=NULL, status='queued', started_at=NULL, completed_at=NULL, updated_at=now()\n WHERE id=$6`,\n [\n req.sourceId ?? null,\n req.externalId ?? null,\n req.name ?? null,\n req.description ?? null,\n req.acl ?? null,\n id,\n ],\n );\n }\n await client.query(\n `UPDATE context_engine_documents SET status='processing', started_at=now(), updated_at=now() WHERE id=$1`,\n [id],\n );\n return id;\n } finally {\n client.release();\n }\n}\n\nexport async function finalizeDocument(\n pool: Pool,\n documentId: string,\n opts: { status: string; error?: string | null; content?: ContentPayload | null },\n): Promise<void> {\n if (opts.content) {\n const c = opts.content;\n const { rows } = await pool.query(`SELECT meta_data FROM context_engine_documents WHERE id = $1`, [\n documentId,\n ]);\n if (!rows[0]) return;\n const meta = { ...((rows[0].meta_data as Record<string, unknown>) ?? {}), ...c.metaUpdates };\n await pool.query(\n `UPDATE context_engine_documents SET\n text=$1, mime_type=$2, lang=$3, ingest_hash=$4, mode=$5, meta_data=$6::jsonb,\n document_type=COALESCE($7, document_type),\n structured_data=COALESCE($8::jsonb, structured_data),\n structured_keys=COALESCE($9::text[], structured_keys),\n structured_quality=COALESCE($10::jsonb, structured_quality),\n extraction_version=COALESCE($11, extraction_version),\n status=$12, error=$13, completed_at=now(), updated_at=now()\n WHERE id=$14`,\n [\n c.text,\n c.mime,\n c.lang,\n c.ingestHash,\n c.mode,\n JSON.stringify(meta),\n c.documentType ?? null,\n c.structuredData != null ? JSON.stringify(c.structuredData) : null,\n c.structuredKeys ?? null,\n c.structuredQuality != null ? JSON.stringify(c.structuredQuality) : null,\n c.extractionVersion ?? null,\n opts.status,\n opts.error ?? null,\n documentId,\n ],\n );\n return;\n }\n await pool.query(\n `UPDATE context_engine_documents SET status=$1, error=$2, completed_at=now(), updated_at=now() WHERE id=$3`,\n [opts.status, opts.error ?? null, documentId],\n );\n}\n\nexport async function refreshContentHash(\n pool: Pool,\n documentId: string | null,\n contentHash: string | null,\n): Promise<void> {\n if (!contentHash || !documentId) return;\n const { rows } = await pool.query(`SELECT meta_data FROM context_engine_documents WHERE id = $1`, [\n documentId,\n ]);\n if (!rows[0]) return;\n const meta = { ...((rows[0].meta_data as Record<string, unknown>) ?? {}) };\n if (meta.content_hash === contentHash) return;\n meta.content_hash = contentHash;\n await pool.query(`UPDATE context_engine_documents SET meta_data=$1::jsonb, updated_at=now() WHERE id=$2`, [\n JSON.stringify(meta),\n documentId,\n ]);\n}\n\n/**\n * Apply caller-declared attribute changes to a document row.\n *\n * `updates` contains ONLY the fields the caller provided: `acl`/`name`/\n * `description` are REPLACED with the given value (including `null`);\n * `meta_data`/`metaData` is MERGED. Shared by the skip-path refresh and\n * `ContextEngine.updateDocument` (partial PATCH).\n *\n * Throws if the document is absent. Returns the names of the fields that\n * actually changed. `\"acl\"` in that list is the caller's signal to push\n * the new value onto the chunk plane.\n */\nexport async function applyAttributeUpdates(\n pool: Pool,\n documentId: string,\n updates: Record<string, unknown>,\n): Promise<string[]> {\n const { rows } = await pool.query(`SELECT * FROM context_engine_documents WHERE id = $1`, [documentId]);\n const doc = rows[0];\n if (!doc) {\n const err = new Error(`document not found: ${JSON.stringify(documentId)}`);\n err.name = \"KeyError\";\n throw err;\n }\n\n const changed: string[] = [];\n const sets: string[] = [];\n const vals: unknown[] = [];\n let i = 1;\n\n if (\"acl\" in updates) {\n const storedAcl = doc.acl != null ? [...(doc.acl as string[])] : null;\n const next = updates.acl as string[] | null;\n const same =\n storedAcl === next ||\n (Array.isArray(storedAcl) &&\n Array.isArray(next) &&\n storedAcl.length === next.length &&\n storedAcl.every((v, idx) => v === next[idx]));\n if (!same) {\n sets.push(`acl = $${i++}`);\n vals.push(next);\n changed.push(\"acl\");\n }\n }\n for (const field of [\"name\", \"description\"] as const) {\n if (field in updates && doc[field] !== updates[field]) {\n sets.push(`${field} = $${i++}`);\n vals.push(updates[field]);\n changed.push(field);\n }\n }\n const metaUpdate = (updates.meta_data ?? updates.metaData) as Record<string, unknown> | null | undefined;\n if (metaUpdate && typeof metaUpdate === \"object\") {\n const meta = { ...((doc.meta_data as Record<string, unknown>) ?? {}) };\n const merged = { ...meta, ...metaUpdate };\n if (JSON.stringify(merged) !== JSON.stringify(meta)) {\n sets.push(`meta_data = $${i++}::jsonb`);\n vals.push(JSON.stringify(merged));\n changed.push(\"meta_data\");\n }\n }\n\n if (!changed.length) return [];\n sets.push(\"updated_at = now()\");\n vals.push(documentId);\n await pool.query(`UPDATE context_engine_documents SET ${sets.join(\", \")} WHERE id = $${i}`, vals);\n return changed;\n}\n\n/**\n * Apply the caller's declared attributes to a document being SKIPPED.\n *\n * **This is a security fix, not a nicety.** Dedup decides from content, but\n * `acl` / `name` / `description` / `metaData` are declared by the CALLER\n * and are not part of the hash. Without this, re-ingesting identical text\n * with a NEW `acl` returned `skipped` before anything was written, the\n * document (and its denormalized chunk copies) kept the OLD ACL, and a\n * tightening the caller believed had applied silently did nothing.\n *\n * - `name` / `description` / `acl` are REPLACED, including with `null`.\n * `acl=null` means \"unrestricted\", not \"leave whatever was there\".\n * - `metaData` is MERGED.\n */\nexport async function refreshDeclaredAttributes(\n pool: Pool,\n documentId: string,\n request: IngestRequest,\n): Promise<string[]> {\n try {\n return await applyAttributeUpdates(pool, documentId, {\n acl: request.acl ?? null,\n name: request.name ?? null,\n description: request.description ?? null,\n meta_data: request.metaData ?? null,\n });\n } catch (exc) {\n if (exc instanceof Error && exc.name === \"KeyError\") return [];\n throw exc;\n }\n}\n\nasync function backfillStructuredExtraction(\n request: IngestRequest,\n opts: {\n pool: Pool;\n embedder: Embedder;\n hooks: Hooks;\n llmCfg: NonNullable<ContextEngineConfig[\"llm\"]>;\n documentId: string;\n },\n): Promise<void> {\n const { rows } = await opts.pool.query(\n `SELECT text, structured_data, source_id FROM context_engine_documents WHERE id = $1`,\n [opts.documentId],\n );\n const doc = rows[0];\n if (!doc) return;\n if (doc.structured_data != null || !doc.text || !String(doc.text).trim()) return;\n\n const extraction = await extractStructuredData(String(doc.text), {\n llmCfg: opts.llmCfg,\n fieldHints: request.fieldHints,\n });\n\n const providerTokens: Record<string, number> = {};\n if (extraction.providerTokens.llm_input)\n providerTokens.structured_llm_input = extraction.providerTokens.llm_input;\n if (extraction.providerTokens.llm_output)\n providerTokens.structured_llm_output = extraction.providerTokens.llm_output;\n\n if (extraction.documentType != null) {\n let embedTokens = 0;\n if (Object.keys(extraction.structuredData).length) {\n try {\n embedTokens = await upsertRegistry({\n pool: opts.pool,\n embedder: opts.embedder,\n sourceId: doc.source_id ?? null,\n docType: extraction.documentType,\n structuredData: extraction.structuredData,\n keysRaw: extraction.keysRaw,\n keysNormalized: extraction.keysNormalized,\n });\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"structured_registry_backfill\", document_id: opts.documentId });\n embedTokens = 0;\n }\n }\n if (embedTokens) providerTokens.structured_embedding_tokens = embedTokens;\n await opts.pool.query(\n `UPDATE context_engine_documents SET\n document_type=$1, structured_data=$2::jsonb, structured_keys=$3,\n structured_quality=$4::jsonb, extraction_version=$5, updated_at=now()\n WHERE id=$6`,\n [\n extraction.documentType,\n JSON.stringify(extraction.structuredData),\n extraction.keysNormalized,\n JSON.stringify({ quality: extraction.quality }),\n STRUCTURED_EXTRACTION_VERSION,\n opts.documentId,\n ],\n );\n } else if (extraction.error) {\n emitError(opts.hooks, new Error(extraction.error), {\n stage: \"structured_extraction_backfill\",\n document_id: opts.documentId,\n });\n }\n\n if (Object.keys(providerTokens).length) {\n emitUsage(opts.hooks, {\n kind: \"ingest\",\n units: 0,\n detail: { document_id: opts.documentId, backfill: \"structured_extraction\" },\n providerTokens,\n } satisfies UsageEvent);\n }\n}\n\n/**\n * Refresh declared attributes across BOTH planes for a skipped document.\n *\n * Called from every skip path — the cheap pre-extraction one, the\n * post-extraction one, and the lost-race one — because all three return\n * without writing anything, and all three are reachable with a changed ACL.\n *\n * Also the choke point for structured-extraction backfill on a skip:\n * `extractStructured=true` against unchanged content used to be a silent\n * no-op because extraction only ran on the `process` branch.\n */\nexport async function refreshOnSkip(\n request: IngestRequest,\n opts: {\n pool: Pool;\n backend: StorageBackend;\n documentId: string | null;\n config: ContextEngineConfig;\n embedder: Embedder;\n hooks: Hooks;\n },\n): Promise<string[]> {\n if (!opts.documentId) return [];\n const changed = await refreshDeclaredAttributes(opts.pool, opts.documentId, request);\n if (changed.includes(\"acl\")) {\n await opts.backend.updateChunkAcl(opts.documentId, request.acl ?? null);\n }\n if (request.extractStructured && opts.config.llm != null) {\n await backfillStructuredExtraction(request, {\n pool: opts.pool,\n embedder: opts.embedder,\n hooks: opts.hooks,\n llmCfg: opts.config.llm,\n documentId: opts.documentId,\n });\n }\n return changed;\n}\n\nasync function upgradeMode(pool: Pool, documentId: string, mode: string): Promise<number> {\n await pool.query(`UPDATE context_engine_documents SET mode=$1, updated_at=now() WHERE id=$2`, [\n mode,\n documentId,\n ]);\n const { rows } = await pool.query(\n `SELECT count(*)::int AS n FROM context_engine_chunks WHERE document_id=$1`,\n [documentId],\n );\n return rows[0]?.n ?? 0;\n}\n\nfunction skippedReport(documentId: string | null, name: string, pages: number | null = null): DocumentReport {\n return {\n documentId: String(documentId ?? \"\"),\n name,\n status: \"skipped\",\n pages,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens: {},\n error: null,\n redactionFailed: [],\n };\n}\n\nfunction singleReport(doc: DocumentReport): IngestReport {\n return {\n documents: [doc],\n totals: {\n files: 1,\n failed: doc.status === \"failed\" ? 1 : 0,\n units: doc.units + doc.graphUnits,\n graphUnits: doc.graphUnits,\n },\n };\n}\n\nasync function maybeAwait(value: Promise<number> | number): Promise<number> {\n return value;\n}\n\n/**\n * Fire one `onProgress` boundary event (see `hooks.emitProgress`). Every\n * stage fires this twice — `started` then `done` — so a live UI can show\n * what is happening as it happens. `documentId` is null on the extract\n * events: the row is only claimed AFTER extraction, because dedup decides\n * from the extracted text. A stage that throws fires no `done`; the failure\n * reaches the caller through the report/exception.\n */\nfunction progress(\n hooks: Hooks,\n request: IngestRequest,\n displayName: string,\n documentId: string | null,\n stage: ProgressEvent[\"stage\"],\n state: ProgressEvent[\"state\"],\n detail: Record<string, unknown> = {},\n): void {\n emitProgress(hooks, {\n sourceId: request.sourceId ?? null,\n externalId: request.externalId ?? null,\n documentId: documentId == null ? null : String(documentId),\n name: displayName,\n stage,\n state,\n detail: { ...detail },\n });\n}\n\n/** How the text was obtained: caller-supplied, a vision model, or a parser. */\nfunction extractMethod(prepared: Prepared, text: string | null): string {\n if (text != null) return \"text\";\n if (Object.keys(prepared.providerTokens).length || prepared.isMarkdown) return \"vision\";\n return \"parser\";\n}\n\nasync function runModeUpgrade(\n request: IngestRequest,\n opts: {\n config: ContextEngineConfig;\n hooks: Hooks;\n pool: Pool;\n backend: StorageBackend;\n embedder: Embedder;\n graphStage: GraphStage | null | undefined;\n documentId: string;\n displayName: string;\n },\n): Promise<IngestReport> {\n await refreshOnSkip(request, {\n pool: opts.pool,\n backend: opts.backend,\n documentId: opts.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n });\n\n let graphUnits = 0;\n let chunkCount = 0;\n try {\n if (request.mode === \"graph\" && opts.graphStage) {\n progress(opts.hooks, request, opts.displayName, opts.documentId, \"graph\", \"started\");\n graphUnits = Number(\n (await maybeAwait(\n opts.graphStage({\n documentId: opts.documentId,\n chunks: null,\n config: opts.config,\n hooks: opts.hooks,\n }),\n )) || 0,\n );\n progress(opts.hooks, request, opts.displayName, opts.documentId, \"graph\", \"done\", {\n graph_units: graphUnits,\n });\n }\n chunkCount = await upgradeMode(opts.pool, opts.documentId, request.mode ?? \"hybrid\");\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"mode_upgrade\", document: opts.displayName });\n return singleReport({\n documentId: String(opts.documentId),\n name: opts.displayName,\n status: \"failed\",\n pages: null,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens: {},\n error: String(exc),\n redactionFailed: [],\n });\n }\n\n if (graphUnits) {\n emitUsage(opts.hooks, {\n kind: \"ingest\",\n units: graphUnits,\n detail: {\n document_id: String(opts.documentId),\n name: opts.displayName,\n mode: request.mode,\n upgrade: true,\n chunks: chunkCount,\n graph_units: graphUnits,\n },\n } satisfies UsageEvent);\n }\n\n return singleReport({\n documentId: String(opts.documentId),\n name: opts.displayName,\n status: \"upgraded\",\n pages: null,\n chunks: chunkCount,\n units: 0,\n graphUnits,\n providerTokens: {},\n error: null,\n redactionFailed: [],\n });\n}\n\n/**\n * Ingest one document end to end and return its report.\n *\n * `contentSource` is the already-resolved `[content, filename, text]`\n * triple from `resolveSource`; the engine resolves it first so a bad\n * argument combination raises before any DB or network work.\n */\nexport async function runIngest(\n request: IngestRequest,\n opts: {\n config: ContextEngineConfig;\n hooks: Hooks;\n backend: StorageBackend;\n embedder: Embedder;\n pool: Pool;\n graphStage?: GraphStage | null;\n contentSource?: [Buffer | null, string | null, string | null] | null;\n },\n): Promise<IngestReport> {\n const [content, filename, text] = opts.contentSource ?? [\n request.content ?? null,\n request.filename ?? null,\n request.text ?? null,\n ];\n const displayName = request.name || filename || request.externalId || \"(unnamed)\";\n const mode: Mode = request.mode ?? opts.config.defaultMode;\n\n await checkEmbeddingLock(opts.pool, opts.config);\n\n if (content != null) {\n const pre = await decide(opts.pool, {\n request: { ...request, mode },\n hashKind: \"content\",\n hashValue: calculateContentHash({ docBytes: content }),\n });\n if (pre.action === \"skip\") {\n await refreshOnSkip(\n { ...request, mode },\n {\n pool: opts.pool,\n backend: opts.backend,\n documentId: pre.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n },\n );\n return singleReport(skippedReport(pre.documentId, displayName));\n }\n if (pre.action === \"upgrade\") {\n return runModeUpgrade(\n { ...request, mode },\n {\n config: opts.config,\n hooks: opts.hooks,\n pool: opts.pool,\n backend: opts.backend,\n embedder: opts.embedder,\n graphStage: opts.graphStage,\n documentId: pre.documentId!,\n displayName,\n },\n );\n }\n }\n\n progress(opts.hooks, request, displayName, null, \"extract\", \"started\");\n const prepared = await prepare({\n content,\n filename,\n text,\n name: request.name ?? null,\n config: opts.config,\n hooks: opts.hooks,\n });\n progress(opts.hooks, request, displayName, null, \"extract\", \"done\", {\n pages: prepared.pages,\n method: extractMethod(prepared, text),\n mime: prepared.mime,\n });\n\n const ingestHash = sha256Bytes(prepared.text || \"\");\n const decision = await decide(opts.pool, {\n request: { ...request, mode },\n hashKind: \"ingest\",\n hashValue: ingestHash,\n });\n\n if (decision.action === \"skip\") {\n await refreshContentHash(opts.pool, decision.documentId, prepared.contentHash);\n await refreshOnSkip(\n { ...request, mode },\n {\n pool: opts.pool,\n backend: opts.backend,\n documentId: decision.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n },\n );\n return singleReport(skippedReport(decision.documentId, displayName, prepared.pages));\n }\n\n if (decision.action === \"upgrade\") {\n return runModeUpgrade(\n { ...request, mode },\n {\n config: opts.config,\n hooks: opts.hooks,\n pool: opts.pool,\n backend: opts.backend,\n embedder: opts.embedder,\n graphStage: opts.graphStage,\n documentId: decision.documentId!,\n displayName,\n },\n );\n }\n\n let documentId: string;\n try {\n documentId = await claimDocument(opts.pool, {\n request: { ...request, mode },\n documentId: decision.documentId,\n });\n } catch (exc) {\n if (!isUniqueViolation(exc)) throw exc;\n const retry = await decide(opts.pool, {\n request: { ...request, mode },\n hashKind: \"ingest\",\n hashValue: ingestHash,\n });\n if (retry.action === \"skip\") {\n await refreshOnSkip(\n { ...request, mode },\n {\n pool: opts.pool,\n backend: opts.backend,\n documentId: retry.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n },\n );\n return singleReport(skippedReport(retry.documentId, displayName, prepared.pages));\n }\n if (retry.action === \"upgrade\") {\n return runModeUpgrade(\n { ...request, mode },\n {\n config: opts.config,\n hooks: opts.hooks,\n pool: opts.pool,\n backend: opts.backend,\n embedder: opts.embedder,\n graphStage: opts.graphStage,\n documentId: retry.documentId!,\n displayName,\n },\n );\n }\n documentId = await claimDocument(opts.pool, {\n request: { ...request, mode },\n documentId: retry.documentId,\n });\n }\n\n const metaUpdates: Record<string, unknown> = { ...(request.metaData ?? {}) };\n metaUpdates.mime_type = prepared.mime;\n if (prepared.contentHash) metaUpdates.content_hash = prepared.contentHash;\n // UNCONDITIONAL, null included: metaUpdates MERGES into the stored\n // meta_data, so a conditional write would leave a stale reason on a\n // document that has since become readable — e.g. re-ingested successfully\n // after a vision model was configured.\n metaUpdates.unreadable_reason = prepared.unreadableReason;\n metaUpdates.unreadable_pages = prepared.unreadablePages;\n\n const redactionFailed: string[] = [];\n let documentText: string;\n progress(opts.hooks, request, displayName, documentId, \"redact\", \"started\");\n try {\n documentText = redactDocumentText(\n prepared.text,\n opts.config.redaction,\n opts.config.secretKey,\n opts.hooks,\n redactionFailed,\n );\n } catch (exc) {\n // A misconfigured policy (a `hash` rule with no `secretKey`) must still\n // raise loudly: swallowing it would silently store unmasked text.\n // `claimDocument` already committed this row to `status=\"processing\"`.\n // Finalize to `failed` first, THEN rethrow — propagation AND terminal\n // state, not one or the other.\n await finalizeDocument(opts.pool, documentId, { status: \"failed\", error: String(exc) });\n throw exc;\n }\n progress(opts.hooks, request, displayName, documentId, \"redact\", \"done\", {\n rules_failed: redactionFailed.length,\n });\n\n if (!prepared.text.trim() || prepared.mediaOnly || looksMostlyBoilerplate(prepared.text)) {\n const reason = prepared.mediaOnly ? \"media-only (no extractable text)\" : \"no extractable text\";\n await opts.backend.upsertChunks(documentId, request.sourceId ?? null, request.acl ?? null, []);\n await finalizeDocument(opts.pool, documentId, {\n status: \"skipped\",\n error: reason,\n content: {\n text: documentText,\n mime: prepared.mime,\n lang: null,\n ingestHash,\n mode,\n metaUpdates,\n },\n });\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"skipped\",\n pages: prepared.pages,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens: {},\n error: reason,\n redactionFailed,\n // Extraction RAN here — an unreadable scan is exactly what lands in\n // this branch, so this is the report that has to say why. The\n // dedup-skip/upgrade sites stay at the defaults: they never extracted\n // and would be claiming blind.\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n }\n\n const units = computeUnits(prepared);\n const providerTokens: Record<string, number> = {};\n if (prepared.providerTokens.input) providerTokens.llm_input = prepared.providerTokens.input;\n if (prepared.providerTokens.output) providerTokens.llm_output = prepared.providerTokens.output;\n\n let rows: ChunkRow[] = [];\n let effectiveMime = prepared.mime;\n try {\n progress(opts.hooks, request, displayName, documentId, \"chunk\", \"started\");\n [rows, effectiveMime] = buildChunks(prepared, request.name, {\n policy: opts.config.redaction,\n secretKey: opts.config.secretKey,\n hooks: opts.hooks,\n failed: redactionFailed,\n });\n progress(opts.hooks, request, displayName, documentId, \"chunk\", \"done\", { chunks: rows.length });\n\n progress(opts.hooks, request, displayName, documentId, \"embed\", \"started\");\n if (request.batch) {\n await opts.backend.upsertChunks(documentId, request.sourceId ?? null, request.acl ?? null, rows);\n const batchId = await submitEmbeddingBatch(\n opts.config.embedding,\n rows.map((r) => r.text),\n );\n progress(opts.hooks, request, displayName, documentId, \"embed\", \"done\", {\n batch: true,\n batch_id: batchId,\n chunks: rows.length,\n });\n metaUpdates.mime_type = effectiveMime;\n metaUpdates.batch = {\n id: batchId,\n units,\n chunks: rows.length,\n mode,\n provider_tokens: providerTokens,\n };\n await finalizeDocument(opts.pool, documentId, {\n status: \"batch_pending\",\n content: {\n text: documentText,\n mime: effectiveMime,\n lang: detectLanguage(documentText),\n ingestHash,\n mode,\n metaUpdates,\n },\n });\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"batch_pending\",\n pages: prepared.pages,\n chunks: rows.length,\n units: 0,\n graphUnits: 0,\n providerTokens,\n error: null,\n redactionFailed,\n // Extraction RAN here just like the skipped/failed/completed sites —\n // this report has to say why pages were unreadable too, not leave\n // the caller to re-derive it later from listDocuments.\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n }\n\n const embeddingTokens = await embedChunks(opts.embedder, rows);\n if (embeddingTokens) providerTokens.embedding_tokens = embeddingTokens;\n await recordEmbeddingDim(opts.pool, opts.embedder.dim);\n\n await opts.backend.upsertChunks(documentId, request.sourceId ?? null, request.acl ?? null, rows);\n progress(opts.hooks, request, displayName, documentId, \"embed\", \"done\", {\n batch: false,\n chunks: rows.length,\n tokens: embeddingTokens,\n });\n\n const structuredKw: Partial<ContentPayload> = {};\n if (request.extractStructured && opts.config.llm) {\n progress(opts.hooks, request, displayName, documentId, \"structured\", \"started\");\n const extraction = await extractStructuredData(documentText, {\n llmCfg: opts.config.llm,\n fieldHints: request.fieldHints,\n });\n if (extraction.providerTokens.llm_input) {\n providerTokens.structured_llm_input = extraction.providerTokens.llm_input;\n }\n if (extraction.providerTokens.llm_output) {\n providerTokens.structured_llm_output = extraction.providerTokens.llm_output;\n }\n if (extraction.documentType != null) {\n let embedTokens = 0;\n if (Object.keys(extraction.structuredData).length) {\n try {\n embedTokens = await upsertRegistry({\n pool: opts.pool,\n embedder: opts.embedder,\n sourceId: request.sourceId ?? null,\n docType: extraction.documentType,\n structuredData: extraction.structuredData,\n keysRaw: extraction.keysRaw,\n keysNormalized: extraction.keysNormalized,\n });\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"structured_registry\", document: displayName });\n embedTokens = 0;\n }\n }\n structuredKw.documentType = extraction.documentType;\n structuredKw.structuredData = extraction.structuredData;\n structuredKw.structuredKeys = extraction.keysNormalized;\n structuredKw.structuredQuality = { quality: extraction.quality };\n structuredKw.extractionVersion = STRUCTURED_EXTRACTION_VERSION;\n if (embedTokens) providerTokens.structured_embedding_tokens = embedTokens;\n } else if (extraction.error) {\n emitError(opts.hooks, new Error(extraction.error), {\n stage: \"structured_extraction\",\n document: displayName,\n });\n }\n progress(opts.hooks, request, displayName, documentId, \"structured\", \"done\", {\n document_type: extraction.documentType,\n keys: extraction.keysNormalized.length,\n quality: extraction.quality,\n });\n }\n\n let graphUnits = 0;\n if (mode === \"graph\" && opts.graphStage) {\n progress(opts.hooks, request, displayName, documentId, \"graph\", \"started\");\n graphUnits = Number(\n (await maybeAwait(\n opts.graphStage({\n documentId,\n chunks: rows,\n config: opts.config,\n hooks: opts.hooks,\n }),\n )) || 0,\n );\n progress(opts.hooks, request, displayName, documentId, \"graph\", \"done\", { graph_units: graphUnits });\n }\n\n metaUpdates.mime_type = effectiveMime;\n await finalizeDocument(opts.pool, documentId, {\n status: \"completed\",\n content: {\n text: documentText,\n mime: effectiveMime,\n lang: detectLanguage(documentText),\n ingestHash,\n mode,\n metaUpdates,\n ...structuredKw,\n },\n });\n\n emitUsage(opts.hooks, {\n kind: \"ingest\",\n units: units + graphUnits,\n detail: {\n document_id: String(documentId),\n name: displayName,\n mime: effectiveMime,\n mode,\n pages: prepared.pages,\n slides: prepared.slides,\n chunks: rows.length,\n graph_units: graphUnits,\n },\n providerTokens,\n } satisfies UsageEvent);\n\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"completed\",\n pages: prepared.pages,\n chunks: rows.length,\n units,\n graphUnits,\n providerTokens,\n error: null,\n redactionFailed,\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"ingest\", document: displayName });\n await finalizeDocument(opts.pool, documentId, { status: \"failed\", error: String(exc) });\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"failed\",\n pages: prepared.pages,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens,\n error: String(exc),\n redactionFailed,\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n }\n}\n","/**\n * `search_knowledge_base` — ONE tool with actions, as a plain library call.\n *\n * The four knowledge tools became sub-actions of one because that is better\n * for the model calling them: one description carries the decision rules once,\n * `discover` says what exists before it guesses, and there is one name to\n * route through. Nothing here knows what MCP is.\n *\n * **Why it lives here and not in `mcp.ts`.** MCP is one way to reach the tool,\n * not the only one. `engine.searchKnowledgeBase(...)` calls this directly;\n * `mcp.ts` registers a tool that resolves the caller's identity and scope and\n * then calls exactly the same function. Two doors, one implementation, so they\n * cannot drift apart.\n *\n * **Identity and scope are arguments, never defaults.** This module invents\n * neither a caller nor a ceiling. Both come from the program that mounted the\n * tool — a model may ask for any source id it likes, and a tool that believed\n * it would let one customer read another's files.\n */\n\nimport { MAX_LIST_LIMIT } from \"./actions.js\";\nimport { isTabularMime } from \"./chunkers.js\";\nimport { DocumentNotFoundError, EngineActionError } from \"./errors.js\";\nimport { ENTITY_TYPES, RELATIONSHIP_CATEGORY_VALUES } from \"./graph/entities.js\";\nimport { UNSCOPED } from \"./sentinels.js\";\n\nexport const KNOWLEDGE_ACTIONS = [\n \"discover\",\n \"search\",\n \"get_doc\",\n \"list\",\n \"query_meta\",\n \"compute\",\n \"get_chunks\",\n \"get_docs\",\n \"map_reduce\",\n \"traverse\",\n \"find_related\",\n \"get_neighbors\",\n \"community_summary\",\n] as const;\nexport type KnowledgeAction = (typeof KNOWLEDGE_ACTIONS)[number];\n\nexport const KNOWLEDGE_TOOL_DESCRIPTION =\n \"The knowledge base — the ingested documents and data files (PDF, Word, \" +\n \"Excel, CSV and the rest) — as ONE tool with actions. Nothing is searched \" +\n \"for you: use it before answering anything that should come from those \" +\n \"documents, and do not use it for general knowledge. \" +\n \"Call action 'discover' FIRST when you do not already know what is there: \" +\n \"it lists the documents, says which are spreadsheets, and says which \" +\n \"actions this deployment can run. Then match the action to the task. \" +\n \"'search' finds passages by meaning or keywords — for questions answered \" +\n \"by reading text. Looking up an identifier (an ID, code, SKU or invoice \" +\n \"number) is the exception: search the BARE identifier alone, e.g. '2525', \" +\n \"never the whole question. Those are CONTENT identifiers — written inside a \" +\n \"document — and they belong in a query; a document_id is a system id (a \" +\n \"uuid) that only discover, list or a search hit can give you, so never \" +\n \"search a uuid as text and never hand an invoice number to get_doc. \" +\n \"'get_doc' reads one whole document by id, 'get_docs' reads several at \" +\n \"once, and 'get_chunks' walks one long document in order a piece at a \" +\n \"time when you need more of it than an excerpt; 'list' browses the \" +\n \"documents without searching. 'map_reduce' asks the SAME question of \" +\n \"every document in scope and answers once per document — for 'which \" +\n \"contracts mention X', where search would return a handful of passages \" +\n \"and miss the rest. \" +\n \"'query_meta' filters and aggregates documents by their structured fields \" +\n \"(dates, amounts, categories) — usually the right action for a question \" +\n \"about spreadsheet data. \" +\n \"'compute' runs code over the spreadsheets for any figure DERIVED from \" +\n \"them — a total, average, count, ranking, margin or comparison across \" +\n \"rows — AND for finding the exact row matching one id or value. In a large \" +\n \"table, search cannot reliably locate an individual row; compute can. \" +\n \"When the question is about how things are CONNECTED rather than what a \" +\n \"document says — who works with whom, what belongs to what, what a change \" +\n \"touches — use the graph actions: 'get_neighbors' for what is one step \" +\n \"from one thing, 'traverse' for everything within a few steps of it, \" +\n \"'find_related' for connections of a kind across the corpus, and \" +\n \"'community_summary' for the themes the corpus groups into. They are \" +\n \"available only where a graph was built; discover says so. Searching with \" +\n \"mode='graph' ranks passages by those same connections instead of by \" +\n \"wording alone, which finds a passage that never repeats your words. \" +\n \"Rules that decide answers: search returns EXCERPTS, and rows of a \" +\n \"spreadsheet are not arithmetic — never add up, average or rank rows \" +\n \"yourself from what search returned, and never say a figure is not \" +\n \"available before computing over the sheet that holds it. A listing that \" +\n \"says has_more has MORE: send its next_cursor back as 'cursor' for the \" +\n \"next page, and never conclude a document is absent from a first page that \" +\n \"was truncated. Search again with different words before saying a document \" +\n \"is missing. A result may carry a next_action (or next_page): it is the \" +\n \"call to make next, already filled in — follow it rather than guessing \" +\n \"the next step. Cite document names.\";\n\n/** The one-line routing rule, carried on the `action` parameter itself. */\nexport const ACTION_PARAM_DESCRIPTION =\n \"Match it to the task: reading questions -> search (a bare identifier for \" +\n \"an ID or code); spreadsheet analysis, or the exact row for one id or \" +\n \"value -> query_meta or compute; how things connect -> get_neighbors, \" +\n \"traverse, find_related or community_summary; browse everything -> list; \" +\n \"unsure what exists -> discover first.\";\n\n/** What `discover` says each action is for. */\nexport const ACTION_HELP: Record<Exclude<KnowledgeAction, \"discover\">, string> = {\n search: \"passages by meaning or keywords; an ID, code or number as the BARE identifier\",\n get_doc: \"one whole document by id — a clause, a policy, the full text\",\n list: \"browse the documents without searching\",\n query_meta: \"filter documents by structured fields (dates, amounts, categories)\",\n compute:\n \"a figure DERIVED from the spreadsheets — total, average, count, ranking, \" +\n \"margin, comparison, or the one row matching a value. Say what you want in \" +\n \"plain words, not code: it runs over EVERY ROW of the sheets in scope, not \" +\n \"a sample and not the excerpts search returned, so name the columns the way \" +\n \"the sheet spells them and say which sheet if more than one could match. \" +\n \"Never retype a figure from memory or from an excerpt — ask for it here\",\n get_chunks: \"read one document in order, a range at a time — the long ones\",\n get_docs: \"several whole documents at once, by id\",\n map_reduce:\n \"ask the SAME question of every document in scope, one answer per document \" +\n \"— for 'which contracts mention X', not for a figure (that is compute)\",\n get_neighbors: \"what is ONE step from a named thing, and which way each link points\",\n traverse: \"everything within a few steps of a named thing\",\n find_related: \"connections of a given kind across the corpus, with no starting thing\",\n community_summary: \"the themes the corpus groups into, ranked against a question\",\n};\n\nconst OFF = {\n query_meta: \"query_meta is off: this deployment has no LLM configured.\",\n compute:\n \"compute is off: code execution is not enabled on this deployment. Quote the rows you found instead.\",\n map_reduce:\n \"map_reduce is off: this deployment has no LLM configured. Use search, or \" +\n \"read the documents with get_docs.\",\n graph:\n \"the graph actions are off: this deployment has no graph configured, so \" +\n \"nothing has been linked up. Use search, get_doc or list instead.\",\n};\n\nconst GRAPH_ACTIONS = [\"traverse\", \"find_related\", \"get_neighbors\", \"community_summary\"];\n\n/**\n * The actions that can narrow by document id. Every other one refuses it\n * rather than answering over the whole corpus as if it had been honoured.\n */\nconst DOCUMENT_ID_ACTIONS = [\"search\", \"compute\", \"get_docs\", \"map_reduce\"];\n\n/** Whole documents are what blows a context window; the tool owns that budget. */\nconst DEFAULT_GET_DOCS_CHARS = 200_000;\n\n/**\n * Every input the MODEL may set — the ONE property table. Both the exported\n * JSON Schema and the protocol server's own Zod shape are built from it.\n * `principals`, `scope` and `redaction` are\n * deliberately absent: they are host-supplied on every surface, and a model\n * that could set them could cross a tenant or turn masking off.\n */\nexport const INPUT_PROPERTIES: Record<string, Record<string, unknown>> = {\n action: { type: \"string\", enum: [...KNOWLEDGE_ACTIONS], description: ACTION_PARAM_DESCRIPTION },\n query: {\n type: \"string\",\n description:\n \"What you are looking for, in words — or a BARE content identifier \" +\n \"when you are looking one up. (for search, query_meta, compute, community_summary)\",\n },\n document_id: {\n type: \"string\",\n description:\n \"One document's id, as returned by discover, list, or a search hit. \" +\n \"A system id, never something written inside a document. (for get_doc)\",\n },\n source_ids: {\n type: \"array\",\n items: { type: \"string\" },\n description:\n \"Narrows to these sources, using ids from discover or a search hit. \" +\n \"Intersected with what this deployment allows: ids outside it are \" +\n \"ignored, not refused. (for discover, list, search, query_meta, compute, \" +\n \"and the graph actions)\",\n },\n document_ids: {\n type: \"array\",\n items: { type: \"string\" },\n description:\n \"Narrows to these documents, using ids from discover, list or a search \" +\n \"hit. It INTERSECTS with source_ids, so a document outside the sources \" +\n \"you named returns nothing. (for search, compute)\",\n },\n entity: {\n type: \"string\",\n description:\n \"The NAME of a thing to start from — a person, an organisation, a \" +\n \"product — spelled as it appears in the documents. A name, NOT AN ID: \" +\n \"take one from discover's graph section, from a search hit, or from an \" +\n \"earlier get_neighbors answer. (for traverse, get_neighbors)\",\n },\n depth: {\n type: \"integer\",\n description: \"How many hops out to walk, 1 to 5. Default 2. (for traverse)\",\n },\n category: {\n type: \"string\",\n enum: [...RELATIONSHIP_CATEGORY_VALUES],\n description:\n \"The kind of connection to follow or look for. These are the only \" +\n \"values the graph holds; see discover for which occur in this corpus. \" +\n \"(for traverse, find_related)\",\n },\n label: {\n type: \"string\",\n description:\n \"The exact relationship wording, as extracted — for example reports_to, \" +\n \"works_for, owns. Narrower than category; use it when you know the \" +\n \"phrasing, otherwise use category. (for find_related)\",\n },\n entity_type: {\n type: \"string\",\n enum: [...ENTITY_TYPES],\n description: \"Keep only relationships with an entity of this type on one end. (for find_related)\",\n },\n top_k: {\n type: \"integer\",\n description: \"How many passages to return. Default 10. (for search)\",\n },\n mode: {\n type: \"string\",\n enum: [\"hybrid\", \"graph\"],\n description:\n \"How to rank. OMIT IT and the right one is worked out from the \" +\n \"documents in scope: wording and meaning, plus how things are \" +\n \"connected whenever those documents were linked up — which finds a \" +\n \"passage that never repeats your words. Name 'hybrid' to force wording \" +\n \"only, or 'graph' to force the connection signal on. (for search)\",\n },\n limit: {\n type: \"integer\",\n description:\n \"How many rows to return: documents for discover and list (default 50), \" +\n \"candidates for query_meta (20), relationships or paths for the graph \" +\n \"actions (50), communities for community_summary (3). (for discover, \" +\n \"list, query_meta, and the graph actions)\",\n },\n cursor: {\n type: \"string\",\n description:\n \"The next_cursor a previous page returned, sent back verbatim to read \" +\n \"the rest. (for discover, list)\",\n },\n start: {\n type: \"integer\",\n description:\n \"First chunk position to read, 0-based. Use the next_start a previous \" +\n \"answer returned to continue. (for get_chunks)\",\n },\n end: {\n type: \"integer\",\n description:\n \"Last chunk position to read, inclusive. At most 25 chunks come back at \" +\n \"a time whatever you ask for. (for get_chunks)\",\n },\n max_chars: {\n type: \"integer\",\n description:\n \"How much document text to return in total before stopping and handing \" +\n \"back the ids it did not reach. (for get_docs)\",\n },\n};\n\n/**\n * The tool as data: name, description and input schema as JSON Schema.\n *\n * Exported so a host wiring this into its own agent loop, or into a protocol\n * this library does not speak, never hand-writes the schema — that would be a\n * third copy of the action list, and a third place for it to fall behind.\n *\n * `readOnlyHint` is false on purpose. Most actions only read, but `compute`\n * runs generated code, and a host deciding whether to auto-approve a call must\n * not read \"knowledge base\" and assume a reader.\n */\nexport function knowledgeToolDefinition(): {\n name: string;\n description: string;\n input_schema: Record<string, unknown>;\n annotations: Record<string, boolean>;\n} {\n return {\n name: \"search_knowledge_base\",\n description: KNOWLEDGE_TOOL_DESCRIPTION,\n input_schema: {\n type: \"object\",\n properties: Object.fromEntries(Object.entries(INPUT_PROPERTIES).map(([k, v]) => [k, { ...v }])),\n required: [\"action\"],\n },\n annotations: { readOnlyHint: false, openWorldHint: false },\n };\n}\n\n// ---------------------------------------------------------------------------\n// Scope — the ceiling the HOST sets, which the model can narrow but never widen\n// ---------------------------------------------------------------------------\n\n/**\n * The documents a mounted tool may ever reach.\n *\n * Deliberately only the nouns the engine already knows: source ids, and\n * optionally some document ids within them. Not project, workspace, tenant or\n * agent — those are a host's words, and baking one in would make every other\n * host translate its word into ours.\n */\nexport type Scope = { sourceIds: string[] | null; documentIds: string[] | null };\n\nexport type ScopeInput = typeof UNSCOPED | string[] | Partial<Scope> | null | undefined;\n\n/**\n * Normalise what a host supplies into a `Scope`. `UNSCOPED` means no ceiling.\n * `null`/omitted throws: it is what a host that forgot looks like, and\n * defaulting it to \"no ceiling\" is the corpus-wide search this exists to stop.\n */\nexport function resolveScope(value: ScopeInput): Scope {\n if (value === UNSCOPED) return { sourceIds: null, documentIds: null };\n if (value == null) {\n throw new Error(\n \"scope is required: pass the source ids this tool may reach, or UNSCOPED \" +\n \"(from @promptev/context-engine) to say the whole corpus on purpose.\",\n );\n }\n if (Array.isArray(value)) {\n return { sourceIds: [...new Set(value.map(String))], documentIds: null };\n }\n if (typeof value === \"object\") {\n return {\n sourceIds: value.sourceIds != null ? [...new Set(value.sourceIds.map(String))] : null,\n documentIds: value.documentIds != null ? [...new Set(value.documentIds.map(String))] : null,\n };\n }\n throw new Error(`scope must be UNSCOPED, an array of source ids, or a Scope — got ${typeof value}`);\n}\n\n/**\n * Intersect what the caller asked for with what the host allows.\n *\n * Narrow, never widen. `null` requested means the whole ceiling, never the\n * whole corpus. An id outside the ceiling is DROPPED rather than refused,\n * because an error would tell the caller which ids are real.\n *\n * The return value distinguishes two things that must never be confused: an\n * array (possibly EMPTY — the caller asked only for things it may not have, so\n * the answer is nothing) from `null` (no ceiling and no narrowing, no filter).\n */\nexport function narrowToCeiling(\n requested: string[] | null | undefined,\n ceiling: string[] | null,\n): string[] | null {\n if (ceiling == null) return requested != null ? [...new Set(requested)] : null;\n if (requested == null) return [...ceiling];\n const allowed = new Set(ceiling);\n return [...new Set(requested)].filter((r) => allowed.has(r));\n}\n\n/** A document-id ceiling is a whitelist: anything not named is outside. */\nfunction withinCeiling(documentId: string, ceiling: Scope): boolean {\n return ceiling.documentIds == null || ceiling.documentIds.includes(String(documentId));\n}\n\n/**\n * A host's own compute. Running generated code is where a host has its own\n * rules about permission, billing and approval, so it can replace ours rather\n * than intercept the action before the tool is reached.\n */\nexport type KnowledgeComputeFn = (\n instruction: string,\n opts: { sourceIds: string[] | null; documentIds: string[] | null; principals: unknown },\n) => Promise<Record<string, unknown>>;\n\n/**\n * Which actions this deployment can service. A host-supplied `compute` makes\n * that action available whatever `enableCodeExecution` says: the host is the\n * one running the code.\n */\nexport function knowledgeActionsAvailable(\n engine: {\n config: { llm?: unknown; enableCodeExecution?: boolean; graph?: { enabled?: boolean } };\n },\n opts: { compute?: KnowledgeComputeFn | null; mapReduce?: unknown } = {},\n): Record<Exclude<KnowledgeAction, \"discover\">, boolean> {\n const hasLlm = engine.config.llm != null;\n const graphOn = Boolean(engine.config.graph?.enabled);\n return {\n search: true,\n get_doc: true,\n list: true,\n query_meta: hasLlm,\n compute: opts.compute != null || (hasLlm && Boolean(engine.config.enableCodeExecution)),\n get_chunks: true,\n get_docs: true,\n map_reduce: opts.mapReduce != null || hasLlm,\n traverse: graphOn,\n find_related: graphOn,\n get_neighbors: graphOn,\n community_summary: graphOn,\n };\n}\n\n/**\n * Spreadsheet exactly when `compute` can read it — `compute` selects on MIME\n * alone and never looks at the name, so judging by filename here would make\n * `discover`'s promise a lie in both directions.\n */\nfunction documentKind(doc: Record<string, unknown>): \"spreadsheet\" | \"text\" {\n return isTabularMime(String(doc.mimeType ?? doc.mime_type ?? \"\")) ? \"spreadsheet\" : \"text\";\n}\n\ntype Listed = { id: string; name: unknown; source_id: unknown; kind: \"spreadsheet\" | \"text\" };\ntype ListedPage = {\n documents: Listed[];\n count: number;\n has_more: boolean;\n next_cursor?: string;\n next_page?: { action: string; cursor: string };\n};\n\n/**\n * A cursor is the `next_cursor` string a previous page returned. A malformed\n * one is refused, never silently treated as page one.\n */\nexport function parseCursor(cursor: unknown): Record<string, unknown> | null {\n if (cursor == null || cursor === \"\") return null;\n if (typeof cursor === \"object\" && !Array.isArray(cursor)) return cursor as Record<string, unknown>;\n let parsed: unknown = null;\n try {\n parsed = JSON.parse(String(cursor));\n } catch {\n parsed = null;\n }\n if (!parsed || typeof parsed !== \"object\" || Array.isArray(parsed)) {\n throw new Error(`invalid cursor: ${JSON.stringify(cursor)}`);\n }\n return parsed as Record<string, unknown>;\n}\n\ntype Engine = {\n config: { llm?: unknown; enableCodeExecution?: boolean; graph?: { enabled?: boolean } };\n search: (query: string, opts?: Record<string, unknown>) => Promise<{ hits: unknown[]; usage?: unknown }>;\n getDocument: (id: string, opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n listDocuments: (opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n queryStructured?: (question: string, opts?: Record<string, unknown>) => Promise<unknown>;\n compute?: (instruction: string, opts?: Record<string, unknown>) => Promise<unknown>;\n ensurePool?: () => Promise<unknown>;\n embedder?: unknown;\n getChunks?: (id: string, opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n getDocuments?: (ids: string[], opts?: Record<string, unknown>) => Promise<Array<Record<string, unknown>>>;\n mapReduce?: (instruction: string, opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n};\n\nasync function listPage(\n engine: Engine,\n sourceIds: string[] | null,\n principals: unknown,\n action: string,\n limit: number,\n cursor: Record<string, unknown> | null,\n ceiling: Scope,\n redaction: unknown,\n): Promise<ListedPage> {\n const page = await engine.listDocuments({\n // `!= null`, NOT truthiness: an EMPTY array means \"nothing is in scope\"\n // and collapsing it to null would list the whole corpus.\n sourceIds: sourceIds != null ? [...new Set(sourceIds)] : null,\n principals,\n cursor,\n limit: Math.max(1, Math.min(limit, MAX_LIST_LIMIT)),\n redaction,\n });\n let documents: Listed[] = ((page.documents as Record<string, unknown>[] | undefined) ?? []).map((raw) => ({\n id: String(raw.id),\n name: raw.name,\n source_id: raw.sourceId ?? raw.source_id,\n kind: documentKind(raw),\n }));\n if (ceiling.documentIds != null) {\n const allowed = new Set(ceiling.documentIds);\n documents = documents.filter((d) => allowed.has(d.id));\n }\n const out: ListedPage = {\n documents,\n count: documents.length,\n has_more: Boolean(page.hasMore ?? page.has_more),\n };\n const next = page.nextCursor ?? page.next_cursor;\n if (next != null) {\n out.next_cursor = JSON.stringify(next);\n out.next_page = { action, cursor: out.next_cursor };\n }\n return out;\n}\n\nexport type KnowledgeToolArgs = {\n action: string;\n principals: unknown;\n scope: ScopeInput;\n query?: string | null;\n document_id?: string | null;\n source_ids?: string[] | null;\n document_ids?: string[] | null;\n entity?: string | null;\n depth?: number | null;\n category?: string | null;\n label?: string | null;\n entity_type?: string | null;\n top_k?: number | null;\n mode?: string | null;\n limit?: number | null;\n cursor?: unknown;\n /** Overrides the deployment's configured policy for this call only. */\n redaction?: unknown;\n /** Replaces the built-in compute action. */\n compute?: KnowledgeComputeFn | null;\n /** Replaces the built-in map_reduce action — the other expensive one. */\n map_reduce?:\n | ((instruction: string, opts: Record<string, unknown>) => Promise<Record<string, unknown>>)\n | null;\n start?: number | null;\n end?: number | null;\n max_chars?: number | null;\n};\n\n/** Run one action of the knowledge tool. */\nexport async function callKnowledgeTool(\n engine: Engine,\n args: KnowledgeToolArgs,\n): Promise<Record<string, unknown>> {\n const action = args.action;\n const principals = args.principals;\n const ceiling = resolveScope(args.scope);\n // What the MODEL asked for, before the host's ceiling is folded in — the\n // refusal below is about the caller's request, never about the ceiling.\n const requestedDocumentIds = args.document_ids;\n const sourceIds = narrowToCeiling(args.source_ids, ceiling.sourceIds);\n const documentIds = narrowToCeiling(args.document_ids, ceiling.documentIds);\n const text = String(args.query ?? \"\").trim();\n const available = knowledgeActionsAvailable(engine, {\n compute: args.compute,\n mapReduce: args.map_reduce,\n });\n\n if (!(KNOWLEDGE_ACTIONS as readonly string[]).includes(action)) {\n return {\n success: false,\n error: `unknown action '${action}' — use one of: ${KNOWLEDGE_ACTIONS.join(\", \")}`,\n };\n }\n if (requestedDocumentIds?.length && !DOCUMENT_ID_ACTIONS.includes(action)) {\n return {\n success: false,\n error:\n `document_ids is not honored by ${action} — only by ${DOCUMENT_ID_ACTIONS.join(\", \")}. ` +\n \"Narrow a listing with source_ids, or read one document with get_doc.\",\n };\n }\n if (GRAPH_ACTIONS.includes(action) && !available[action as Exclude<KnowledgeAction, \"discover\">]) {\n return { success: false, error: OFF.graph };\n }\n\n if (action === \"discover\" || action === \"list\") {\n const page = await listPage(\n engine,\n sourceIds,\n principals,\n action,\n args.limit ?? 50,\n parseCursor(args.cursor),\n ceiling,\n args.redaction,\n );\n if (action === \"list\") return { success: true, ...page };\n const spreadsheets = page.documents.filter((d) => d.kind === \"spreadsheet\").map((d) => d.name);\n const forAFact = { action: \"search\", query: \"<bare identifier or key words>\" };\n const nextAction: Record<string, unknown> = {};\n if (spreadsheets.length && available.compute) {\n nextAction[\"for a figure from a spreadsheet\"] = {\n action: \"compute\",\n query: \"<what to compute, columns as named>\",\n };\n }\n nextAction[\"for a clause or a fact\"] = forAFact;\n if (available.get_neighbors) {\n nextAction[\"for how things connect\"] = {\n action: \"get_neighbors\",\n entity: \"<a name that appears in the documents>\",\n };\n }\n return {\n success: true,\n ...page,\n spreadsheets,\n available_actions: Object.fromEntries(\n (Object.keys(ACTION_HELP) as Array<keyof typeof ACTION_HELP>).map((name) => [\n name,\n available[name] ? ACTION_HELP[name] : `${ACTION_HELP[name]} (OFF)`,\n ]),\n ),\n next_action: nextAction,\n };\n }\n\n if (action === \"search\") {\n if (!text) return { success: false, error: \"search needs a query\" };\n const result = await engine.search(text, {\n sourceIds,\n documentIds,\n principals,\n topK: args.top_k ?? 10,\n // Omitted means \"work it out from the documents in scope\".\n mode: args.mode ?? null,\n redaction: args.redaction,\n });\n return {\n success: true,\n hits: (result.hits ?? []).map((hit) => {\n const h = hit as Record<string, unknown>;\n return {\n document_id: h.document_id ?? h.documentId,\n document_name: h.document_name ?? h.documentName,\n chunk_text: h.chunk_text ?? h.chunkText,\n score: h.score,\n source_id: h.source_id ?? h.sourceId,\n };\n }),\n usage: result.usage,\n };\n }\n\n if (action === \"get_doc\") {\n const documentId = args.document_id;\n if (!documentId) return { success: false, error: \"get_doc needs a document_id\" };\n // Reading one document by id is the obvious way around a source ceiling,\n // so the ceiling is checked here too — and a document outside it reads\n // exactly like one that does not exist.\n if (!withinCeiling(documentId, ceiling)) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n try {\n const document = await engine.getDocument(documentId, {\n principals,\n redaction: args.redaction,\n });\n const src = document?.sourceId ?? document?.source_id;\n if (ceiling.sourceIds != null && !ceiling.sourceIds.includes(String(src))) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n return { success: true, document };\n } catch (err) {\n // `getDocument` gives an absent and a forbidden document the SAME\n // message on purpose, so a caller can never tell them apart.\n if (err instanceof DocumentNotFoundError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n if (action === \"query_meta\") {\n if (!available.query_meta || !engine.queryStructured) return { success: false, error: OFF.query_meta };\n if (!text) return { success: false, error: \"query_meta needs a query\" };\n const out = await engine.queryStructured(text, {\n sourceIds,\n principals,\n limit: Math.max(1, Math.min(args.limit ?? 20, 100)),\n redaction: args.redaction,\n });\n return { success: true, ...(out as Record<string, unknown>) };\n }\n\n if (action === \"compute\") {\n if (!available.compute) return { success: false, error: OFF.compute };\n if (!text) return { success: false, error: \"compute needs a query: what to compute, in plain words\" };\n // A host-supplied `compute` REPLACES the built-in one. Running code is\n // where a host has its own rules about permission, billing and approval,\n // and it should not have to intercept the action before the tool is\n // reached to apply them.\n const runner = args.compute ?? engine.compute;\n if (!runner) return { success: false, error: OFF.compute };\n try {\n const out = await runner(text, { sourceIds, documentIds, principals });\n return { success: true, ...(out as Record<string, unknown>) };\n } catch (err) {\n // Nothing tabular in scope is the model's to recover from — it can quote\n // the rows it found instead — not a transport failure.\n if (err instanceof EngineActionError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n if (action === \"get_chunks\") {\n const documentId = args.document_id;\n if (!documentId) return { success: false, error: \"get_chunks needs a document_id\" };\n if (!withinCeiling(documentId, ceiling)) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n try {\n const page = await engine.getChunks!(documentId, {\n principals,\n start: args.start,\n end: args.end,\n redaction: args.redaction,\n });\n if (ceiling.sourceIds != null) {\n // A source ceiling has to hold here too, and the refusal reads like a\n // document that is not there.\n const doc = await engine.getDocument(documentId, { principals });\n const src = doc?.sourceId ?? doc?.source_id;\n if (!ceiling.sourceIds.includes(String(src))) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n }\n const out: Record<string, unknown> = { success: true, ...page };\n if (page.has_more) {\n out.next_page = {\n action: \"get_chunks\",\n document_id: String(documentId),\n start: page.next_start,\n };\n }\n return out;\n } catch (err) {\n if (err instanceof DocumentNotFoundError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n if (action === \"get_docs\") {\n if (!documentIds?.length) return { success: false, error: \"get_docs needs document_ids\" };\n const docs = await engine.getDocuments!(documentIds, { principals, redaction: args.redaction });\n // The LIBRARY refuses to truncate — that is policy for whatever calls it —\n // and this tool IS that caller, so the budget lives here. What it could not\n // fit comes back as the call to make next.\n const budget = Math.max(1000, Math.trunc(args.max_chars || DEFAULT_GET_DOCS_CHARS));\n const kept: Array<Record<string, unknown>> = [];\n let used = 0;\n for (const doc of docs) {\n const size = String(doc.text ?? \"\").length;\n if (kept.length && used + size > budget) break;\n kept.push(doc);\n used += size;\n }\n const returned = new Set(kept.map((d) => String(d.id)));\n const remaining = (args.document_ids ?? []).filter((d) => !returned.has(String(d)));\n const out: Record<string, unknown> = { success: true, documents: kept, count: kept.length };\n if (remaining.length) {\n out.remaining_document_ids = remaining;\n out.next_page = { action: \"get_docs\", document_ids: remaining };\n }\n return out;\n }\n\n if (action === \"map_reduce\") {\n if (!available.map_reduce) return { success: false, error: OFF.map_reduce };\n if (!text) {\n return {\n success: false,\n error: \"map_reduce needs a query: the question to ask of each document\",\n };\n }\n // A host-supplied `map_reduce` REPLACES the built-in one, for the same\n // reason `compute` has that seam.\n const runner = args.map_reduce ?? engine.mapReduce;\n if (!runner) return { success: false, error: OFF.map_reduce };\n try {\n const out = await runner(text, {\n sourceIds,\n documentIds,\n principals,\n limit: args.limit,\n });\n return { success: true, ...(out as Record<string, unknown>) };\n } catch (err) {\n if (err instanceof EngineActionError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n return callGraphAction(engine, args, { sourceIds, principals, text });\n}\n\nasync function callGraphAction(\n engine: Engine,\n args: KnowledgeToolArgs,\n ctx: { sourceIds: string[] | null; principals: unknown; text: string },\n): Promise<Record<string, unknown>> {\n // Imported lazily so importing the package never pulls the graph module in\n // for a deployment that has no graph.\n const nav = await import(\"./graph/navigation.js\");\n const pool = (await engine.ensurePool?.()) as never;\n // These take the INTERNAL spelling (null = trusted).\n const principals = (ctx.principals ?? null) as string[] | null;\n const scoped = { sourceIds: ctx.sourceIds, principals };\n\n if (args.action === \"community_summary\") {\n if (!ctx.text) return { success: false, error: \"community_summary needs a query\" };\n const out = await nav.communitySummary(pool, {\n embedder: engine.embedder as never,\n query: ctx.text,\n limit: args.limit,\n sourceIds: ctx.sourceIds,\n principals,\n });\n if (!out.available) return { success: false, error: out.error as string };\n return { success: true, communities: out.communities, count: out.count };\n }\n\n if (args.action === \"find_related\") {\n if (!(args.category || args.label || args.entity_type)) {\n return {\n success: false,\n error: \"find_related needs a category, a label or an entity_type to look for\",\n };\n }\n const out = await nav.findRelated(pool, {\n ...scoped,\n category: args.category,\n label: args.label,\n entityType: args.entity_type,\n limit: args.limit,\n });\n return { success: true, relationships: out.relationships, count: out.count };\n }\n\n if (!args.entity) {\n return {\n success: false,\n error: `${args.action} needs an entity: the name of a thing to start from`,\n };\n }\n if (args.action === \"get_neighbors\") {\n const out = await nav.getNeighbors(pool, { ...scoped, entity: args.entity, limit: args.limit });\n if (!out.found) return { success: false, error: out.error as string };\n return { success: true, entity: out.entity, neighbors: out.neighbors, count: out.count };\n }\n const out = await nav.traverse(pool, {\n ...scoped,\n entity: args.entity,\n depth: args.depth,\n category: args.category,\n limit: args.limit,\n });\n if (!out.found) return { success: false, error: out.error as string };\n return { success: true, entity: out.entity, paths: out.paths, count: out.count };\n}\n","import { z } from \"zod\";\nimport { TRUSTED, type Trusted } from \"./sentinels.js\";\n\nexport class HandlerError extends Error {\n status: number;\n detail: unknown;\n constructor(status: number, detail: unknown) {\n super(typeof detail === \"string\" ? detail : JSON.stringify(detail));\n this.name = \"HandlerError\";\n this.status = status;\n this.detail = detail;\n }\n}\n\nexport const ingestJsonRequestSchema = z\n .object({\n text: z.string(),\n name: z.string(),\n source_id: z.string().nullable().optional(),\n external_id: z.string().nullable().optional(),\n description: z.string().nullable().optional(),\n meta_data: z.record(z.unknown()).nullable().optional(),\n acl: z.array(z.string()).nullable().optional(),\n mode: z.enum([\"hybrid\", \"graph\"]).nullable().optional(),\n extract_structured: z.boolean().optional().default(false),\n batch: z.boolean().optional().default(false),\n })\n .strip();\n\nexport const documentPatchSchema = z\n .object({\n acl: z.array(z.string()).nullable().optional(),\n name: z.string().nullable().optional(),\n description: z.string().nullable().optional(),\n meta_data: z.record(z.unknown()).nullable().optional(),\n })\n .strict();\n\nexport const searchRequestSchema = z\n .object({\n query: z.string(),\n source_ids: z.array(z.string()).nullable().optional(),\n // Narrows WITHIN a source and INTERSECTS with source_ids — it can only\n // shrink the result set (the ACL predicate still applies in the same SQL\n // conjunction), so exposing it needs no authorizeAcl-style grant check.\n document_ids: z.array(z.string()).nullable().optional(),\n top_k: z.number().int().optional().default(10),\n mode: z.enum([\"hybrid\", \"graph\"]).optional().default(\"hybrid\"),\n compress_to_tokens: z.number().int().nullable().optional(),\n })\n .strip();\n\nexport type IngestJSONRequest = z.infer<typeof ingestJsonRequestSchema>;\nexport type DocumentPatch = z.infer<typeof documentPatchSchema>;\nexport type SearchRequest = z.infer<typeof searchRequestSchema>;\n\nexport type ContextEngineLike = {\n ingest: (args: Record<string, unknown>) => Promise<unknown>;\n search: (query: string, opts?: Record<string, unknown>) => Promise<{ hits?: unknown[]; usage?: unknown }>;\n listDocuments: (opts: Record<string, unknown>) => Promise<unknown>;\n getDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n deleteDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n updateDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n stats: (sourceId?: string | null) => Promise<unknown>;\n};\n\nconst UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;\n\nexport function requireValidUuid(documentId: string): void {\n if (!UUID_RE.test(String(documentId))) {\n throw new HandlerError(400, `invalid document id: ${JSON.stringify(documentId)}`);\n }\n}\n\nexport function formBool(value: unknown): boolean {\n if (typeof value === \"boolean\") return value;\n if (value == null) return false;\n return [\"1\", \"true\", \"yes\", \"on\"].includes(String(value).trim().toLowerCase());\n}\n\nexport function formJson(value: unknown): unknown {\n if (value === undefined || value === null || value === \"\") return null;\n if (typeof value === \"object\") return value;\n try {\n return JSON.parse(String(value));\n } catch {\n throw new HandlerError(400, `invalid JSON in field: ${JSON.stringify(value)}`);\n }\n}\n\nexport function formStr(value: unknown): string | null {\n return value === undefined || value === null || value === \"\" ? null : String(value);\n}\n\nexport function resolveRequestPrincipals(value: unknown, opts: { surface: string }): string[] | Trusted {\n // Hand the SENTINEL through, not null: every engine method resolves TRUSTED\n // silently, while `principals=null` triggers the deprecation warning aimed\n // at legacy in-process callers — collapsing here made a correctly\n // configured trusted mount re-trigger that warning on every request.\n if (value === TRUSTED) return TRUSTED;\n if (value == null) {\n throw new HandlerError(\n 500,\n `${opts.surface}: the \\`principals\\` dependency returned null, which means ` +\n `TRUSTED CALLER — it disables ACL filtering and skips the check that ` +\n `stops a caller filing documents under groups it does not hold. This ` +\n `mount is misconfigured: return [] for an unauthenticated caller, or ` +\n `pass TRUSTED explicitly (from @promptev/context-engine) if the surface really ` +\n `is trusted.`,\n );\n }\n // A runtime check, not a cast: `as string[]` let a plain-JS dependency\n // return a bare string that flowed into resolvePrincipals and (before its\n // own guard) coerced to full-corpus TRUSTED.\n if (!Array.isArray(value) || value.some((p) => typeof p !== \"string\")) {\n throw new HandlerError(\n 500,\n `${opts.surface}: the \\`principals\\` dependency must return a list of ` +\n `principal strings, [] for an anonymous caller, or TRUSTED — got ${typeof value}.`,\n );\n }\n return value;\n}\n\nexport function authorizeAcl(requested: unknown, principals: string[] | null | Trusted): void {\n if (requested == null || principals == null || principals === TRUSTED) return;\n if (!Array.isArray(requested)) {\n throw new HandlerError(422, \"acl must be a list of principal strings\");\n }\n // TRUSTED was excluded above, so what remains is the plain list.\n const held = new Set(principals as string[]);\n const ungranted = [...new Set(requested.map(String))].filter((p) => !held.has(p)).sort();\n if (ungranted.length) {\n throw new HandlerError(403, `cannot grant access to principals you do not hold: ${ungranted}`);\n }\n}\n\nfunction asDict(obj: unknown): unknown {\n if (obj == null) return obj;\n if (typeof obj === \"object\" && typeof (obj as { toJSON?: () => unknown }).toJSON === \"function\") {\n return (obj as { toJSON: () => unknown }).toJSON();\n }\n return obj;\n}\n\nfunction zodError(err: z.ZodError): HandlerError {\n return new HandlerError(\n 422,\n err.issues.map((i) => ({ loc: i.path, msg: i.message, type: i.code })),\n );\n}\n\nasync function runIngest(engine: ContextEngineLike, kwargs: Record<string, unknown>): Promise<unknown> {\n try {\n return asDict(await engine.ingest(kwargs));\n } catch (exc) {\n if (exc instanceof TypeError || (exc instanceof Error && exc.name === \"TypeError\")) {\n throw new HandlerError(400, String(exc));\n }\n if (exc instanceof Error && exc.name === \"Error\" && /invalid|required/i.test(exc.message)) {\n throw new HandlerError(400, String(exc));\n }\n throw exc;\n }\n}\n\nexport async function handleIngestJson(\n engine: ContextEngineLike,\n payload: unknown,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_ingest_json\" });\n const parsed = ingestJsonRequestSchema.safeParse(payload);\n if (!parsed.success) throw zodError(parsed.error);\n const body = parsed.data;\n authorizeAcl(body.acl, resolved);\n return runIngest(engine, {\n text: body.text,\n name: body.name,\n description: body.description,\n sourceId: body.source_id,\n externalId: body.external_id,\n metaData: body.meta_data,\n acl: body.acl,\n mode: body.mode,\n extractStructured: body.extract_structured,\n batch: body.batch,\n });\n}\n\nexport async function handleIngestFile(\n engine: ContextEngineLike,\n opts: {\n content: Buffer | Uint8Array;\n filename: string | null | undefined;\n form: Record<string, unknown> | Map<string, unknown>;\n principals: unknown;\n },\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(opts.principals, { surface: \"handle_ingest_file\" });\n const form = opts.form instanceof Map ? Object.fromEntries(opts.form) : opts.form;\n const acl = formJson(form.acl);\n authorizeAcl(acl, resolved);\n return runIngest(engine, {\n content: opts.content,\n filename: opts.filename,\n name: formStr(form.name) || opts.filename,\n description: formStr(form.description),\n sourceId: formStr(form.source_id),\n externalId: formStr(form.external_id),\n metaData: formJson(form.meta_data),\n acl,\n mode: formStr(form.mode),\n extractStructured: formBool(form.extract_structured),\n batch: formBool(form.batch),\n });\n}\n\nexport async function handleListDocuments(\n engine: ContextEngineLike,\n opts: {\n sourceId: string | null | undefined;\n limit: number;\n cursor: string | null | undefined;\n principals: unknown;\n },\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(opts.principals, { surface: \"handle_list_documents\" });\n let parsedCursor: unknown = null;\n if (opts.cursor) {\n try {\n parsedCursor = JSON.parse(opts.cursor);\n } catch {\n throw new HandlerError(400, `invalid cursor: ${JSON.stringify(opts.cursor)}`);\n }\n }\n return engine.listDocuments({\n sourceId: opts.sourceId ?? null,\n principals: resolved,\n cursor: parsedCursor,\n limit: opts.limit,\n });\n}\n\nexport async function handleGetDocument(\n engine: ContextEngineLike,\n documentId: string,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_get_document\" });\n requireValidUuid(documentId);\n try {\n return await engine.getDocument(documentId, { principals: resolved });\n } catch (exc) {\n if (\n exc instanceof Error &&\n (exc.name === \"DocumentNotFoundError\" || exc instanceof TypeError === false)\n ) {\n const msg = String(exc);\n if (/not found/i.test(msg) || exc.name === \"DocumentNotFoundError\") {\n throw new HandlerError(404, msg.replace(/^Error:\\s*/, \"\"));\n }\n }\n throw exc;\n }\n}\n\nexport async function handleDeleteDocument(\n engine: ContextEngineLike,\n documentId: string,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_delete_document\" });\n requireValidUuid(documentId);\n try {\n await engine.deleteDocument(documentId, { principals: resolved });\n } catch (exc) {\n const msg = String(exc);\n if (/not found/i.test(msg) || (exc instanceof Error && exc.name === \"DocumentNotFoundError\")) {\n throw new HandlerError(404, msg.replace(/^Error:\\s*/, \"\"));\n }\n throw exc;\n }\n return { deleted: documentId };\n}\n\nexport async function handleUpdateDocument(\n engine: ContextEngineLike,\n documentId: string,\n payload: unknown,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_update_document\" });\n requireValidUuid(documentId);\n const parsed = documentPatchSchema.safeParse(payload);\n if (!parsed.success) throw zodError(parsed.error);\n const provided: Record<string, unknown> = {};\n const raw = payload as Record<string, unknown>;\n for (const key of [\"acl\", \"name\", \"description\", \"meta_data\"] as const) {\n if (Object.hasOwn(raw, key)) {\n provided[key === \"meta_data\" ? \"metaData\" : key] = raw[key];\n }\n }\n if (\"acl\" in provided) authorizeAcl(provided.acl, resolved);\n try {\n const changed = await engine.updateDocument(documentId, { principals: resolved, ...provided });\n return { changed };\n } catch (exc) {\n const msg = String(exc);\n if (/not found/i.test(msg) || (exc instanceof Error && exc.name === \"DocumentNotFoundError\")) {\n throw new HandlerError(404, msg.replace(/^Error:\\s*/, \"\"));\n }\n throw exc;\n }\n}\n\nexport async function handleSearch(\n engine: ContextEngineLike,\n payload: unknown,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_search\" });\n const parsed = searchRequestSchema.safeParse(payload);\n if (!parsed.success) throw zodError(parsed.error);\n const body = parsed.data;\n // Early, like every sibling document endpoint: a malformed id must be a\n // clean 400 here, not a Postgres 22P02 raised inside every search leg —\n // whose 400-vs-500 mapping would otherwise hang on the ENGLISH pg error\n // text, with the raw DatabaseError echoed to the caller.\n for (const did of body.document_ids ?? []) requireValidUuid(did);\n try {\n const result = await engine.search(body.query, {\n sourceIds: body.source_ids,\n documentIds: body.document_ids,\n principals: resolved,\n topK: body.top_k,\n mode: body.mode,\n compressToTokens: body.compress_to_tokens,\n });\n const hits = (result.hits ?? []).map((hit) => asDict(hit));\n return { hits, usage: result.usage };\n } catch (exc) {\n if (exc instanceof Error && /invalid|required|graph/i.test(exc.message)) {\n throw new HandlerError(400, String(exc));\n }\n throw exc;\n }\n}\n\nexport async function handleStats(\n engine: ContextEngineLike,\n sourceId: string | null | undefined,\n): Promise<unknown> {\n return engine.stats(sourceId ?? null);\n}\n","import { z } from \"zod\";\nimport { HandlerError, resolveRequestPrincipals } from \"../routing-core.js\";\nimport type { Trusted } from \"../sentinels.js\";\n\nexport type PrincipalsFn = () => unknown | Promise<unknown>;\n/** Zero-argument, sync or async, resolved fresh per execution — the same\n * contract as `PrincipalsFn`. Returns the opaque approval scope (a run id is\n * the expected shape) or `null` for the unscoped legacy path. */\nexport type ApprovalScopeFn = () => unknown | Promise<unknown>;\n\nexport const SCOPE_NOTE =\n \" Results are scoped to the caller's permissions (resolved by the server's \" +\n \"injected principals dependency — never a caller-supplied argument) and to \" +\n \"the given source_ids, if any.\";\n\nexport async function resolvePrincipalsFn(principals: PrincipalsFn): Promise<string[] | Trusted> {\n const result = await Promise.resolve(principals());\n // Validation DELEGATES to routing-core's resolveRequestPrincipals — the\n // single canonical remote-surface validator (TRUSTED passes through, null\n // and non-arrays are rejected) — rather than hand-copying its rules: a\n // rule added to the canonical validator must not silently miss the MCP\n // surface. Its HandlerError is an HTTP shape, so it is re-raised as a\n // plain Error with the same detail — the MCP host converts that to the\n // isError=true result this surface's contract promises.\n try {\n return resolveRequestPrincipals(result, { surface: \"the MCP `principals` dependency\" });\n } catch (exc) {\n if (exc instanceof HandlerError) throw new Error(exc.message, { cause: exc });\n throw exc;\n }\n}\n\nexport type McpToolHost = {\n tool(\n name: string,\n description: string,\n schema: Record<string, unknown>,\n handler: (...args: never[]) => Promise<unknown>,\n ): unknown;\n};\n\nexport function registerToolGateway(\n mcp: McpToolHost,\n engine: {\n searchTools: (query: string, opts?: Record<string, unknown>) => Promise<unknown[]>;\n executeTool: (\n name: string,\n args: Record<string, unknown> | null,\n opts?: Record<string, unknown>,\n ) => Promise<Record<string, unknown>>;\n },\n principals: PrincipalsFn,\n /** Optional server-side resolver for the approval claim scope (see\n * governance `executeTool`). Like `principals`, NEVER a tool argument: a\n * model that could name its own scope could claim another run's\n * approvals. Validation is the library's, so a wrong type surfaces as the\n * same isError=true result any other host-side wiring bug does. */\n approvalScope: ApprovalScopeFn | null = null,\n): void {\n mcp.tool(\n \"search_tools\",\n \"Keyword search over the tools this deployment has registered \" +\n \"(http/db/mcp/function) — the ACL-visible name, kind, description, \" +\n \"and JSON Schema params of each match, so a caller can discover what \" +\n \"it can then invoke with `execute_tool`.\" +\n SCOPE_NOTE,\n // Zod raw shape — see the note on the search tool in mcp.ts.\n { query: z.string(), limit: z.number().int().optional() },\n async (...raw: never[]) => {\n const args = (raw[0] ?? {}) as { query?: string; limit?: number };\n const callerPrincipals = await resolvePrincipalsFn(principals);\n const tools = await engine.searchTools(String(args.query ?? \"\"), {\n principals: callerPrincipals,\n limit: args.limit ?? 10,\n });\n return { tools };\n },\n );\n\n mcp.tool(\n \"execute_tool\",\n \"Governed execution of a tool previously discovered via \" +\n \"`search_tools` (ACL check, approval gate, audit). Returns \" +\n \"`{result, usage}` on success, or an `approval_required` payload \" +\n \"when the call needs a human approval that isn't already granted — \" +\n \"the caller must not retry until that approval resolves.\" +\n SCOPE_NOTE,\n { name: z.string(), args: z.record(z.unknown()).optional() },\n async (...raw: never[]) => {\n const body = (raw[0] ?? {}) as { name?: string; args?: Record<string, unknown> };\n const callerPrincipals = await resolvePrincipalsFn(principals);\n const scope = approvalScope ? await Promise.resolve(approvalScope()) : null;\n return engine.executeTool(String(body.name), body.args ?? null, {\n principals: callerPrincipals,\n source: \"mcp\",\n approvalScope: scope,\n });\n },\n );\n}\n","/** Bumped by CI on every main merge; 0.0.0 = pre-first-release. */\nexport const __version__ = \"0.0.0\";\n"]}
|
|
1
|
+
{"version":3,"sources":["../src/errors.ts","../src/extras.ts","../src/text.ts","../src/hooks.ts","../src/providers/google.ts","../src/providers/llm.ts","../src/crypto.ts","../src/redaction.ts","../src/extraction/json.ts","../src/extraction/files.ts","../src/extraction/index.ts","../src/graph/retrieval.ts","../src/graph/navigation.ts","../src/mcp.ts","../src/graph/entities.ts","../src/chunkers.ts","../src/actions.ts","../src/sandbox.ts","../src/structured.ts","../src/sentinels.ts","../src/ingest.ts","../src/knowledge-tool.ts","../src/routing-core.ts","../src/tools/mcp-tools.ts","../src/version.ts"],"names":["out","z","handler"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;AAAA,IAGa,mBAQA,qBAAA,EA0CA,iBAAA;AArDb,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,eAAA,GAAA;AAGO,IAAM,iBAAA,GAAN,cAAgC,KAAA,CAAM;AAAA,MAC3C,YAAY,OAAA,EAAiB;AAC3B,QAAA,KAAA,CAAM,OAAO,CAAA;AACb,QAAA,IAAA,CAAK,IAAA,GAAO,mBAAA;AAAA,MACd;AAAA,KACF;AAGO,IAAM,qBAAA,GAAN,cAAoC,KAAA,CAAM;AAAA,MAC/C,YAAY,UAAA,EAAoB;AAC9B,QAAA,KAAA,CAAM,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAE,CAAA;AACzC,QAAA,IAAA,CAAK,IAAA,GAAO,uBAAA;AAAA,MACd;AAAA,KACF;AAqCO,IAAM,iBAAA,GAAN,cAAgC,KAAA,CAAM;AAAA,MAC3C,WAAA,CAAY,KAAA,EAAe,GAAA,EAAa,IAAA,EAAc;AACpD,QAAA,KAAA,CAAM,CAAA,EAAG,IAAI,CAAA,eAAA,EAAkB,GAAG,8BAA8B,KAAK,CAAA,eAAA,EAAkB,GAAG,CAAA,CAAE,CAAA;AAC5F,QAAA,IAAA,CAAK,IAAA,GAAO,mBAAA;AAAA,MACd;AAAA,KACF;AAAA,EAAA;AAAA,CAAA,CAAA;;;AChDA,eAAsB,YAAA,CAA0B,SAAA,EAAmB,KAAA,EAAe,IAAA,EAA0B;AAC1G,EAAA,IAAI;AACF,IAAA,OAAQ,MAAM,OAAO,SAAA,CAAA;AAAA,EACvB,SAAS,IAAA,EAAM;AACb,IAAA,MAAM,IAAI,iBAAA,CAAkB,KAAA,EAAO,SAAA,EAAW,IAAI,CAAA;AAAA,EACpD;AACF;AAhBA,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,eAAA,GAAA;AAAA,IAAA,WAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACAA,IAkRM,mBAAA;AAlRN,IAAA,SAAA,GAAA,KAAA,CAAA;AAAA,EAAA,aAAA,GAAA;AAkRA,IAAM,mBAAA,GAAsB,CAAA,kqaAAA,CAAA;AAI5B,IAAsC,mBAAA,CAAoB,KAAA,CAAM,GAAG,CAAA,CAAE,GAAA,CAAI,CAAC,KAAA,KAAU;AAClF,MAAA,MAAM,CAAC,KAAA,EAAO,GAAA,EAAK,IAAI,CAAA,GAAI,KAAA,CAAM,MAAM,GAAG,CAAA;AAC1C,MAAA,OAAO,EAAE,KAAA,EAAO,MAAA,CAAO,QAAA,CAAS,KAAA,EAAQ,EAAE,CAAA,EAAG,GAAA,EAAK,MAAA,CAAO,QAAA,CAAS,KAAM,EAAE,CAAA,EAAG,IAAA,EAAM,MAAA,CAAO,IAAI,CAAA,EAAE;AAAA,IAClG,CAAC,CAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACzRD,IAAA,UAAA,GAAA,KAAA,CAAA;AAAA,EAAA,cAAA,GAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACAA,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AAmCA,IAAA,WAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACnCA,IAAA,QAAA,GAAA,KAAA,CAAA;AAAA,EAAA,sBAAA,GAAA;AAaA,IAAA,WAAA,EAAA;AAEA,IAAA,WAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACfA,IAAA,WAAA,GAAA,KAAA,CAAA;AAAA,EAAA,eAAA,GAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACAA,IAAA,cAAA,GAAA,KAAA,CAAA;AAAA,EAAA,kBAAA,GAAA;AAAA,IAAA,WAAA,EAAA;AACA,IAAA,UAAA,EAAA;AA0HA,EAAA;AAAA,CAAA,CAAA;AC3HA,IAAA,SAAA,GAAA,KAAA,CAAA;AAAA,EAAA,wBAAA,GAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACAA,IAAA,UAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AAcA,IAAkB,OAAO,IAAA,CAAK,CAAC,IAAM,EAAA,EAAM,CAAA,EAAM,CAAI,CAAC,CAAA;AAAA,EAAA;AAAA,CAAA,CAAA;ACdtD,IAAA,eAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AAOA,IAAA,WAAA,EAAA;AACA,IAAA,UAAA,EAAA;AACA,IAAA,UAAA,EAAA;AAoEqC,EAAA;AAAA,CAAA,CAAA;;;ACxCrC,eAAsB,kBAAA,CAAmB,MAAY,SAAA,EAA8C;AACjG,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA,kGAAA,CAAA;AAAA,IAEA,CAAC,SAAS;AAAA,GACZ;AACA,EAAA,OAAO,CAAC,MAAA,CAAO,IAAA,CAAK,CAAC,CAAA,EAAG,OAAA;AAC1B;AAEA,eAAsB,2BAAA,CACpB,MACA,IAAA,EACkB;AAClB,EAAA,IAAI,IAAA,CAAK,UAAA,IAAc,IAAA,EAAM,OAAO,IAAA;AACpC,EAAA,OAAO,kBAAA,CAAmB,IAAA,EAAM,IAAA,CAAK,SAAS,CAAA;AAChD;AApDA,IAAA,cAAA,GAAA,KAAA,CAAA;AAAA,EAAA,wBAAA,GAAA;AAEA,IAAA,UAAA,EAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACFA,IAAA,kBAAA,GAAA,EAAA;AAAA,QAAA,CAAA,kBAAA,EAAA;AAAA,EAAA,kBAAA,EAAA,MAAA,kBAAA;AAAA,EAAA,gBAAA,EAAA,MAAA,gBAAA;AAAA,EAAA,WAAA,EAAA,MAAA,WAAA;AAAA,EAAA,YAAA,EAAA,MAAA,YAAA;AAAA,EAAA,QAAA,EAAA,MAAA;AAAA,CAAA,CAAA;AAgFA,SAAS,KAAA,CAAM,KAAA,EAAkC,QAAA,EAAkB,OAAA,EAAyB;AAC1F,EAAA,MAAM,CAAA,GAAI,OAAO,KAAA,KAAU,QAAA,IAAY,MAAA,CAAO,QAAA,CAAS,KAAK,CAAA,GAAI,IAAA,CAAK,KAAA,CAAM,KAAK,CAAA,GAAI,QAAA;AACpF,EAAA,OAAO,KAAK,GAAA,CAAI,CAAA,EAAG,KAAK,GAAA,CAAI,CAAA,EAAG,OAAO,CAAC,CAAA;AACzC;AAEA,SAAS,UAAU,IAAA,EAAqE;AACtF,EAAA,OAAO,CAAC,KAAK,SAAA,IAAa,IAAA,EAAM,KAAK,WAAA,IAAe,IAAA,EAAM,IAAA,CAAK,UAAA,IAAc,IAAI,CAAA;AACnF;AAEA,eAAe,aAAA,CACb,IAAA,EACA,IAAA,EACA,IAAA,EAC4D;AAC5D,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA,sCAAA,EACoC,cAAc,CAAA,QAAA,CAAA;AAAA,IAClD,CAAC,GAAG,SAAA,CAAU,IAAI,CAAA,EAAA,CAAI,QAAQ,EAAA,EAAI,IAAA,EAAK,CAAE,WAAA,EAAa;AAAA,GACxD;AACA,EAAA,MAAM,GAAA,GAAM,MAAA,CAAO,IAAA,CAAK,CAAC,CAAA;AACzB,EAAA,OAAO,GAAA,GAAM,EAAE,EAAA,EAAI,MAAA,CAAO,GAAA,CAAI,EAAE,CAAA,EAAG,IAAA,EAAM,GAAA,CAAI,IAAA,EAAM,IAAA,EAAM,GAAA,CAAI,MAAK,GAAI,IAAA;AACxE;AAGA,eAAsB,YAAA,CACpB,MACA,IAAA,EACkC;AAClC,EAAA,MAAM,QAAQ,MAAM,aAAA,CAAc,IAAA,EAAM,IAAA,CAAK,QAAQ,IAAI,CAAA;AACzD,EAAA,IAAI,CAAC,KAAA,EAAO,OAAO,EAAE,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,SAAA,EAAW,SAAA,EAAW,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AAC7E,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,KAAA,EAMG,YAAY;AAAA,iCAAA,CAAA;AAAA,IAEf,CAAC,GAAG,SAAA,CAAU,IAAI,CAAA,EAAG,KAAA,CAAM,EAAA,EAAI,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,EAAA,EAAI,SAAS,CAAC;AAAA,GACjE;AACA,EAAA,MAAM,SAAA,GAAY,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IACxC,MAAM,CAAA,CAAE,IAAA;AAAA,IACR,MAAM,CAAA,CAAE,IAAA;AAAA,IACR,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,OAAO,CAAA,CAAE,KAAA;AAAA,IACT,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,WAAW,CAAA,CAAE;AAAA,GACf,CAAE,CAAA;AACF,EAAA,OAAO;AAAA,IACL,KAAA,EAAO,IAAA;AAAA,IACP,QAAQ,EAAE,IAAA,EAAM,MAAM,IAAA,EAAM,IAAA,EAAM,MAAM,IAAA,EAAK;AAAA,IAC7C,SAAA;AAAA,IACA,OAAO,SAAA,CAAU;AAAA,GACnB;AACF;AAGA,eAAsB,QAAA,CACpB,MACA,IAAA,EACkC;AAClC,EAAA,MAAM,QAAQ,MAAM,aAAA,CAAc,IAAA,EAAM,IAAA,CAAK,QAAQ,IAAI,CAAA;AACzD,EAAA,IAAI,CAAC,KAAA,EAAO,OAAO,EAAE,KAAA,EAAO,KAAA,EAAO,KAAA,EAAO,SAAA,EAAW,KAAA,EAAO,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AAGzE,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,OAAA,EAOK,YAAY;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,SAAA,EAWV,YAAY;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,aAAA,CAAA;AAAA,IAQnB;AAAA,MACE,GAAG,UAAU,IAAI,CAAA;AAAA,MACjB,KAAA,CAAM,EAAA;AAAA,MACN,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,CAAA,EAAG,SAAS,CAAA;AAAA,MAC9B,KAAK,QAAA,IAAY,IAAA;AAAA,MACjB,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,EAAA,EAAI,SAAS;AAAA;AACjC,GACF;AACA,EAAA,MAAM,KAAA,GAAQ,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IACpC,QAAQ,CAAA,CAAE,IAAA;AAAA,IACV,aAAa,CAAA,CAAE,IAAA;AAAA,IACf,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,OAAO,CAAA,CAAE,KAAA;AAAA,IACT,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,MAAM,CAAA,CAAE;AAAA,GACV,CAAE,CAAA;AACF,EAAA,KAAA,CAAM,KAAK,CAAC,CAAA,EAAG,CAAA,KAAM,CAAA,CAAE,OAAO,CAAA,CAAE,IAAA,IAAQ,MAAA,CAAO,CAAA,CAAE,MAAM,CAAA,CAAE,aAAA,CAAc,OAAO,CAAA,CAAE,MAAM,CAAC,CAAC,CAAA;AACxF,EAAA,OAAO,EAAE,KAAA,EAAO,IAAA,EAAM,MAAA,EAAQ,EAAE,IAAA,EAAM,KAAA,CAAM,IAAA,EAAM,IAAA,EAAM,MAAM,IAAA,EAAK,EAAG,KAAA,EAAO,KAAA,EAAO,MAAM,MAAA,EAAO;AACnG;AAGA,eAAsB,WAAA,CACpB,MACA,IAAA,EAMkC;AAClC,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,KAAA,EAQG,YAAY;AAAA,yCAAA,CAAA;AAAA,IAEf;AAAA,MACE,GAAG,UAAU,IAAI,CAAA;AAAA,MACjB,KAAK,QAAA,IAAY,IAAA;AAAA,MACjB,KAAK,KAAA,IAAS,IAAA;AAAA,MACd,KAAK,UAAA,IAAc,IAAA;AAAA,MACnB,KAAA,CAAM,IAAA,CAAK,KAAA,EAAO,EAAA,EAAI,SAAS;AAAA;AACjC,GACF;AACA,EAAA,MAAM,aAAA,GAAgB,MAAA,CAAO,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IAC5C,QAAQ,CAAA,CAAE,QAAA;AAAA,IACV,aAAa,CAAA,CAAE,QAAA;AAAA,IACf,QAAQ,CAAA,CAAE,QAAA;AAAA,IACV,aAAa,CAAA,CAAE,QAAA;AAAA,IACf,UAAU,CAAA,CAAE,QAAA;AAAA,IACZ,OAAO,CAAA,CAAE,KAAA;AAAA,IACT,UAAU,CAAA,CAAE;AAAA,GACd,CAAE,CAAA;AACF,EAAA,OAAO,EAAE,KAAA,EAAO,IAAA,EAAM,aAAA,EAAe,KAAA,EAAO,cAAc,MAAA,EAAO;AACnE;AAUA,eAAsB,gBAAA,CACpB,MACA,IAAA,EAOkC;AAClC,EAAA,MAAM,OAAA,GAAU,MAAM,2BAAA,CAA4B,IAAA,EAAM;AAAA,IACtD,SAAA,EAAW,KAAK,SAAA,IAAa,IAAA;AAAA,IAC7B,UAAA,EAAY,KAAK,UAAA,IAAc;AAAA,GAChC,CAAA;AACD,EAAA,IAAI,CAAC,OAAA,EAAS;AACZ,IAAA,OAAO,EAAE,WAAW,KAAA,EAAO,KAAA,EAAO,oBAAoB,WAAA,EAAa,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AAAA,EAClF;AACA,EAAA,MAAM,OAAA,GAAU,MAAM,IAAA,CAAK,QAAA,CAAS,KAAA,CAAM,CAAC,IAAA,CAAK,KAAA,IAAS,EAAE,CAAA,EAAG,OAAO,CAAA;AACrE,EAAA,MAAM,MAAA,GAAS,UAAU,CAAC,CAAA;AAC1B,EAAA,IAAI,CAAC,MAAA,EAAQ,OAAO,EAAE,SAAA,EAAW,MAAM,WAAA,EAAa,EAAC,EAAG,KAAA,EAAO,CAAA,EAAE;AACjE,EAAA,MAAM,MAAA,GAAS,MAAM,IAAA,CAAK,KAAA;AAAA,IACxB,CAAA;AAAA;AAAA;AAAA;AAAA,uDAAA,CAAA;AAAA,IAKA,CAAC,CAAA,CAAA,EAAI,MAAA,CAAO,IAAI,CAAC,CAAA,KAAM,OAAO,CAAC,CAAC,EAAE,IAAA,CAAK,GAAG,CAAC,CAAA,CAAA,CAAA,EAAK,KAAA,CAAM,KAAK,KAAA,EAAO,CAAA,EAAG,EAAE,CAAC;AAAA,GAC1E;AACA,EAAA,MAAM,WAAA,GAAc,MAAA,CAAO,IAAA,CACxB,MAAA,CAAO,CAAC,CAAA,KAAM,MAAA,CAAO,CAAA,CAAE,GAAG,CAAA,GAAI,uBAAuB,CAAA,CACrD,GAAA,CAAI,CAAC,CAAA,MAAO;AAAA,IACX,SAAS,CAAA,CAAE,OAAA;AAAA,IACX,cAAc,CAAA,CAAE,YAAA;AAAA,IAChB,oBAAoB,CAAA,CAAE,kBAAA;AAAA,IACtB,iBAAiB,CAAA,CAAE,KAAA;AAAA,IACnB,SAAA,EAAW,KAAK,KAAA,CAAM,MAAA,CAAO,EAAE,GAAG,CAAA,GAAI,GAAK,CAAA,GAAI;AAAA,GACjD,CAAE,CAAA;AACJ,EAAA,OAAO,EAAE,SAAA,EAAW,IAAA,EAAM,WAAA,EAAa,KAAA,EAAO,YAAY,MAAA,EAAO;AACnE;AAxRA,IAoCM,OAOA,YAAA,EAWA,cAAA,EAQA,SAAA,EACA,SAAA,EACA,WACA,uBAAA,EAGO,kBAAA;AApEb,IAAA,eAAA,GAAA,KAAA,CAAA;AAAA,EAAA,yBAAA,GAAA;AA4BA,IAAA,cAAA,EAAA;AAQA,IAAM,KAAA,GAAQ;AAAA;AAAA;AAAA;AAAA,CAAA;AAOd,IAAM,YAAA,GAAe;AAAA;AAAA;AAAA;AAAA;AAAA,8BAAA,EAKW,KAAK;AAAA;AAAA;AAAA,CAAA;AAMrC,IAAM,cAAA,GAAiB;AAAA;AAAA;AAAA;AAAA,8BAAA,EAIS,KAAK;AAAA;AAAA,CAAA;AAIrC,IAAM,SAAA,GAAY,kBAAA;AAClB,IAAM,SAAA,GAAY,CAAA;AAClB,IAAM,SAAA,GAAY,GAAA;AAClB,IAAM,uBAAA,GAA0B,IAAA;AAGzB,IAAM,kBAAA,GACX,2SAAA;AAAA,EAAA;AAAA,CAAA,CAAA;;;ACpEF,WAAA,EAAA;AACA,WAAA,EAAA;ACIO,IAAM,YAAA,GAAe;AAAA,EAC1B,QAAA;AAAA,EACA,KAAA;AAAA,EACA,SAAA;AAAA,EACA,UAAA;AAAA,EACA,WAAA;AAAA,EACA,UAAA;AAAA,EACA;AACF,CAAA;AAQO,IAAM,4BAAA,GAA+B;AAAA,EAC1C,cAAA;AAAA,EACA,YAAA;AAAA,EACA,UAAA;AAAA,EACA,UAAA;AAAA,EACA,SAAA;AAAA,EACA,WAAA;AAAA,EACA,YAAA;AAAA,EACA;AACF,CAAA;AAEuC,IAAI,GAAA,CAAY,4BAA4B;;;AC/BnF,SAAA,EAAA;AAWkB,IAAI,GAAA;AAAA,EACpB;AACF;AAuIO,SAAS,cAAc,IAAA,EAA0C;AACtE,EAAA,IAAI,CAAC,MAAM,OAAO,KAAA;AAClB,EAAA,MAAM,CAAA,GAAI,KAAK,WAAA,EAAY;AAC3B,EAAA,OAAO,CAAC,KAAA,EAAO,OAAA,EAAS,OAAA,EAAS,eAAA,EAAiB,eAAe,CAAA,CAAE,IAAA,CAAK,CAAC,CAAA,KAAM,CAAA,CAAE,QAAA,CAAS,CAAC,CAAC,CAAA;AAC9F;;;ACxIA,WAAA,EAAA;AACA,UAAA,EAAA;AAEA,QAAA,EAAA;AACA,cAAA,EAAA;;;ACGA,WAAA,EAAA;;;ACJA,SAAA,EAAA;AAEA,QAAA,EAAA;AACA,SAAA,EAAA;;;ACNO,IAAM,OAAA,mBAAyB,MAAA,CAAO,GAAA,CAAI,wBAAwB,CAAA;AAYlE,IAAM,QAAA,mBAA0B,MAAA,CAAO,GAAA,CAAI,yBAAyB,CAAA;ACY3E,eAAA,EAAA;AACA,UAAA,EAAA;;;AAGA,cAAA,EAAA;;;AJnBO,IAAM,cAAA,GAAiB,GAAA;;;AKL9B,WAAA,EAAA;AAIO,IAAM,iBAAA,GAAoB;AAAA,EAC/B,UAAA;AAAA,EACA,QAAA;AAAA,EACA,SAAA;AAAA,EACA,MAAA;AAAA,EACA,YAAA;AAAA,EACA,SAAA;AAAA,EACA,YAAA;AAAA,EACA,UAAA;AAAA,EACA,YAAA;AAAA,EACA,UAAA;AAAA,EACA,cAAA;AAAA,EACA,eAAA;AAAA,EACA;AACF,CAAA;AAGO,IAAM,0BAAA,GACX,q+HAAA;AA6DK,IAAM,wBAAA,GACX,kUAAA;AAOK,IAAM,WAAA,GAAoE;AAAA,EAC/E,MAAA,EAAQ,+EAAA;AAAA,EACR,OAAA,EAAS,mEAAA;AAAA,EACT,IAAA,EAAM,wCAAA;AAAA,EACN,UAAA,EAAY,oEAAA;AAAA,EACZ,OAAA,EACE,kcAAA;AAAA,EAMF,UAAA,EAAY,oEAAA;AAAA,EACZ,QAAA,EAAU,wCAAA;AAAA,EACV,UAAA,EACE,sJAAA;AAAA,EAEF,aAAA,EAAe,qEAAA;AAAA,EACf,QAAA,EAAU,gDAAA;AAAA,EACV,YAAA,EAAc,uEAAA;AAAA,EACd,iBAAA,EAAmB;AACrB,CAAA;AAEA,IAAM,GAAA,GAAM;AAAA,EACV,UAAA,EAAY,2DAAA;AAAA,EACZ,OAAA,EACE,qGAAA;AAAA,EACF,UAAA,EACE,4GAAA;AAAA,EAEF,KAAA,EACE;AAEJ,CAAA;AAEA,IAAM,aAAA,GAAgB,CAAC,UAAA,EAAY,cAAA,EAAgB,iBAAiB,mBAAmB,CAAA;AAWvF,IAAM,sBAAsB,CAAC,QAAA,EAAU,WAAW,UAAA,EAAY,YAAA,EAAc,YAAY,MAAM,CAAA;AAG9F,IAAM,sBAAA,GAAyB,GAAA;AASxB,IAAM,gBAAA,GAA4D;AAAA,EACvE,MAAA,EAAQ,EAAE,IAAA,EAAM,QAAA,EAAU,IAAA,EAAM,CAAC,GAAG,iBAAiB,CAAA,EAAG,WAAA,EAAa,wBAAA,EAAyB;AAAA,EAC9F,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,WAAA,EAAa;AAAA,IACX,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,UAAA,EAAY;AAAA,IACV,IAAA,EAAM,OAAA;AAAA,IACN,KAAA,EAAO,EAAE,IAAA,EAAM,QAAA,EAAS;AAAA,IACxB,WAAA,EACE;AAAA,GAIJ;AAAA,EACA,YAAA,EAAc;AAAA,IACZ,IAAA,EAAM,OAAA;AAAA,IACN,KAAA,EAAO,EAAE,IAAA,EAAM,QAAA,EAAS;AAAA,IACxB,WAAA,EACE;AAAA,GAMJ;AAAA,EACA,MAAA,EAAQ;AAAA,IACN,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAIJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EAAa;AAAA,GACf;AAAA,EACA,QAAA,EAAU;AAAA,IACR,IAAA,EAAM,QAAA;AAAA,IACN,IAAA,EAAM,CAAC,GAAG,4BAA4B,CAAA;AAAA,IACtC,WAAA,EACE;AAAA,GAGJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAGJ;AAAA,EACA,WAAA,EAAa;AAAA,IACX,IAAA,EAAM,QAAA;AAAA,IACN,IAAA,EAAM,CAAC,GAAG,YAAY,CAAA;AAAA,IACtB,WAAA,EAAa;AAAA,GACf;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EAAa;AAAA,GACf;AAAA,EACA,IAAA,EAAM;AAAA,IACJ,IAAA,EAAM,QAAA;AAAA,IACN,IAAA,EAAM,CAAC,QAAA,EAAU,OAAO,CAAA;AAAA,IACxB,WAAA,EACE;AAAA,GAKJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA,GAIJ;AAAA,EACA,MAAA,EAAQ;AAAA,IACN,IAAA,EAAM,QAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,KAAA,EAAO;AAAA,IACL,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,GAAA,EAAK;AAAA,IACH,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA,GAEJ;AAAA,EACA,SAAA,EAAW;AAAA,IACT,IAAA,EAAM,SAAA;AAAA,IACN,WAAA,EACE;AAAA;AAGN,CAAA;AAoDO,SAAS,aAAa,KAAA,EAA0B;AACrD,EAAA,IAAI,UAAU,QAAA,EAAU,OAAO,EAAE,SAAA,EAAW,IAAA,EAAM,aAAa,IAAA,EAAK;AACpE,EAAA,IAAI,SAAS,IAAA,EAAM;AACjB,IAAA,MAAM,IAAI,KAAA;AAAA,MACR;AAAA,KAEF;AAAA,EACF;AACA,EAAA,IAAI,KAAA,CAAM,OAAA,CAAQ,KAAK,CAAA,EAAG;AACxB,IAAA,OAAO,EAAE,SAAA,EAAW,CAAC,GAAG,IAAI,GAAA,CAAI,KAAA,CAAM,GAAA,CAAI,MAAM,CAAC,CAAC,CAAA,EAAG,aAAa,IAAA,EAAK;AAAA,EACzE;AACA,EAAA,IAAI,OAAO,UAAU,QAAA,EAAU;AAC7B,IAAA,OAAO;AAAA,MACL,SAAA,EAAW,KAAA,CAAM,SAAA,IAAa,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,KAAA,CAAM,SAAA,CAAU,GAAA,CAAI,MAAM,CAAC,CAAC,CAAA,GAAI,IAAA;AAAA,MACjF,WAAA,EAAa,KAAA,CAAM,WAAA,IAAe,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,KAAA,CAAM,WAAA,CAAY,GAAA,CAAI,MAAM,CAAC,CAAC,CAAA,GAAI;AAAA,KACzF;AAAA,EACF;AACA,EAAA,MAAM,IAAI,KAAA,CAAM,CAAA,sEAAA,EAAoE,OAAO,KAAK,CAAA,CAAE,CAAA;AACpG;AAaO,SAAS,eAAA,CACd,WACA,OAAA,EACiB;AACjB,EAAA,IAAI,OAAA,IAAW,IAAA,EAAM,OAAO,SAAA,IAAa,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,SAAS,CAAC,CAAA,GAAI,IAAA;AAC1E,EAAA,IAAI,SAAA,IAAa,IAAA,EAAM,OAAO,CAAC,GAAG,OAAO,CAAA;AACzC,EAAA,MAAM,OAAA,GAAU,IAAI,GAAA,CAAI,OAAO,CAAA;AAC/B,EAAA,OAAO,CAAC,GAAG,IAAI,GAAA,CAAI,SAAS,CAAC,CAAA,CAAE,MAAA,CAAO,CAAC,CAAA,KAAM,OAAA,CAAQ,GAAA,CAAI,CAAC,CAAC,CAAA;AAC7D;AAGA,SAAS,aAAA,CAAc,YAAoB,OAAA,EAAyB;AAClE,EAAA,OAAO,OAAA,CAAQ,eAAe,IAAA,IAAQ,OAAA,CAAQ,YAAY,QAAA,CAAS,MAAA,CAAO,UAAU,CAAC,CAAA;AACvF;AAiBO,SAAS,yBAAA,CACd,MAAA,EAGA,IAAA,GAAqE,EAAC,EACf;AACvD,EAAA,MAAM,MAAA,GAAS,MAAA,CAAO,MAAA,CAAO,GAAA,IAAO,IAAA;AACpC,EAAA,MAAM,OAAA,GAAU,OAAA,CAAQ,MAAA,CAAO,MAAA,CAAO,OAAO,OAAO,CAAA;AACpD,EAAA,OAAO;AAAA,IACL,MAAA,EAAQ,IAAA;AAAA,IACR,OAAA,EAAS,IAAA;AAAA,IACT,IAAA,EAAM,IAAA;AAAA,IACN,UAAA,EAAY,MAAA;AAAA,IACZ,OAAA,EAAS,KAAK,OAAA,IAAW,IAAA,IAAS,UAAU,OAAA,CAAQ,MAAA,CAAO,OAAO,mBAAmB,CAAA;AAAA,IACrF,UAAA,EAAY,IAAA;AAAA,IACZ,QAAA,EAAU,IAAA;AAAA,IACV,UAAA,EAAY,IAAA,CAAK,SAAA,IAAa,IAAA,IAAQ,MAAA;AAAA,IACtC,QAAA,EAAU,OAAA;AAAA,IACV,YAAA,EAAc,OAAA;AAAA,IACd,aAAA,EAAe,OAAA;AAAA,IACf,iBAAA,EAAmB;AAAA,GACrB;AACF;AAOA,SAAS,aAAa,GAAA,EAAsD;AAC1E,EAAA,OAAO,aAAA,CAAc,OAAO,GAAA,CAAI,QAAA,IAAY,IAAI,SAAA,IAAa,EAAE,CAAC,CAAA,GAAI,aAAA,GAAgB,MAAA;AACtF;AAkCO,SAAS,YAAY,MAAA,EAAiD;AAC3E,EAAA,IAAI,MAAA,IAAU,IAAA,IAAQ,MAAA,KAAW,EAAA,EAAI,OAAO,IAAA;AAC5C,EAAA,IAAI,OAAO,WAAW,QAAA,IAAY,CAAC,MAAM,OAAA,CAAQ,MAAM,GAAG,OAAO,MAAA;AACjE,EAAA,IAAI,MAAA,GAAkB,IAAA;AACtB,EAAA,IAAI;AACF,IAAA,MAAA,GAAS,IAAA,CAAK,KAAA,CAAM,MAAA,CAAO,MAAM,CAAC,CAAA;AAAA,EACpC,CAAA,CAAA,MAAQ;AACN,IAAA,MAAA,GAAS,IAAA;AAAA,EACX;AACA,EAAA,IAAI,CAAC,UAAU,OAAO,MAAA,KAAW,YAAY,KAAA,CAAM,OAAA,CAAQ,MAAM,CAAA,EAAG;AAClE,IAAA,MAAM,IAAI,KAAA,CAAM,CAAA,gBAAA,EAAmB,KAAK,SAAA,CAAU,MAAM,CAAC,CAAA,CAAE,CAAA;AAAA,EAC7D;AACA,EAAA,OAAO,MAAA;AACT;AAoBA,eAAe,QAAA,CACb,QACA,SAAA,EACA,UAAA,EACA,QACA,KAAA,EACA,MAAA,EACA,WAAA,EACA,OAAA,EACA,SAAA,EACqB;AACrB,EAAA,MAAM,IAAA,GAAO,MAAM,MAAA,CAAO,aAAA,CAAc;AAAA;AAAA;AAAA,IAGtC,SAAA,EAAW,aAAa,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,SAAS,CAAC,CAAA,GAAI,IAAA;AAAA;AAAA;AAAA;AAAA;AAAA,IAKzD,WAAA,EAAa,eAAe,IAAA,GAAO,CAAC,GAAG,IAAI,GAAA,CAAI,WAAW,CAAC,CAAA,GAAI,IAAA;AAAA,IAC/D,UAAA;AAAA,IACA,MAAA;AAAA,IACA,KAAA,EAAO,KAAK,GAAA,CAAI,CAAA,EAAG,KAAK,GAAA,CAAI,KAAA,EAAO,cAAc,CAAC,CAAA;AAAA,IAClD;AAAA,GACD,CAAA;AACD,EAAA,IAAI,aAAwB,IAAA,CAAK,SAAA,IAAuD,EAAC,EAAG,GAAA,CAAI,CAAC,GAAA,MAAS;AAAA,IACxG,EAAA,EAAI,MAAA,CAAO,GAAA,CAAI,EAAE,CAAA;AAAA,IACjB,MAAM,GAAA,CAAI,IAAA;AAAA,IACV,SAAA,EAAW,GAAA,CAAI,QAAA,IAAY,GAAA,CAAI,SAAA;AAAA,IAC/B,IAAA,EAAM,aAAa,GAAG,CAAA;AAAA,IACtB,aAAA,EAAe,GAAA,CAAI,YAAA,IAAgB,GAAA,CAAI,aAAA,IAAiB,IAAA;AAAA,IACxD,IAAA,EAAM,IAAI,IAAA,IAAQ;AAAA,GACpB,CAAE,CAAA;AACF,EAAA,IAAI,OAAA,CAAQ,eAAe,IAAA,EAAM;AAC/B,IAAA,MAAM,OAAA,GAAU,IAAI,GAAA,CAAI,OAAA,CAAQ,WAAW,CAAA;AAC3C,IAAA,SAAA,GAAY,SAAA,CAAU,OAAO,CAAC,CAAA,KAAM,QAAQ,GAAA,CAAI,CAAA,CAAE,EAAE,CAAC,CAAA;AAAA,EACvD;AACA,EAAA,MAAM,GAAA,GAAkB;AAAA,IACtB,SAAA;AAAA,IACA,OAAO,SAAA,CAAU,MAAA;AAAA,IACjB,QAAA,EAAU,OAAA,CAAQ,IAAA,CAAK,OAAA,IAAW,KAAK,QAAQ;AAAA,GACjD;AACA,EAAA,MAAM,IAAA,GAAO,IAAA,CAAK,UAAA,IAAc,IAAA,CAAK,WAAA;AACrC,EAAA,IAAI,QAAQ,IAAA,EAAM;AAChB,IAAA,GAAA,CAAI,WAAA,GAAc,IAAA,CAAK,SAAA,CAAU,IAAI,CAAA;AACrC,IAAA,GAAA,CAAI,SAAA,GAAY,EAAE,MAAA,EAAQ,MAAA,EAAQ,IAAI,WAAA,EAAY;AAAA,EACpD;AACA,EAAA,OAAO,GAAA;AACT;AAiCA,eAAsB,iBAAA,CACpB,QACA,IAAA,EACkC;AAClC,EAAA,MAAM,SAAS,IAAA,CAAK,MAAA;AACpB,EAAA,MAAM,aAAa,IAAA,CAAK,UAAA;AACxB,EAAA,MAAM,OAAA,GAAU,YAAA,CAAa,IAAA,CAAK,KAAK,CAAA;AAGvC,EAAA,MAAM,uBAAuB,IAAA,CAAK,YAAA;AAClC,EAAA,MAAM,SAAA,GAAY,eAAA,CAAgB,IAAA,CAAK,UAAA,EAAY,QAAQ,SAAS,CAAA;AACpE,EAAA,MAAM,WAAA,GAAc,eAAA,CAAgB,IAAA,CAAK,YAAA,EAAc,QAAQ,WAAW,CAAA;AAC1E,EAAA,MAAM,OAAO,MAAA,CAAO,IAAA,CAAK,KAAA,IAAS,EAAE,EAAE,IAAA,EAAK;AAC3C,EAAA,MAAM,SAAA,GAAY,0BAA0B,MAAA,EAAQ;AAAA,IAClD,SAAS,IAAA,CAAK,OAAA;AAAA,IACd,WAAW,IAAA,CAAK;AAAA,GACjB,CAAA;AAED,EAAA,IAAI,CAAE,iBAAA,CAAwC,QAAA,CAAS,MAAM,CAAA,EAAG;AAC9D,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,KAAA;AAAA,MACT,OAAO,CAAA,gBAAA,EAAmB,MAAM,wBAAmB,iBAAA,CAAkB,IAAA,CAAK,IAAI,CAAC,CAAA;AAAA,KACjF;AAAA,EACF;AACA,EAAA,IAAI,sBAAsB,MAAA,IAAU,CAAC,mBAAA,CAAoB,QAAA,CAAS,MAAM,CAAA,EAAG;AACzE,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,KAAA;AAAA,MACT,OACE,CAAA,+BAAA,EAAkC,MAAM,mBAAc,mBAAA,CAAoB,IAAA,CAAK,IAAI,CAAC,CAAA,sEAAA;AAAA,KAExF;AAAA,EACF;AACA,EAAA,IAAI,cAAc,QAAA,CAAS,MAAM,KAAK,CAAC,SAAA,CAAU,MAA8C,CAAA,EAAG;AAChG,IAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,IAAI,KAAA,EAAM;AAAA,EAC5C;AAEA,EAAA,IAAI,MAAA,KAAW,UAAA,IAAc,MAAA,KAAW,MAAA,EAAQ;AAC9C,IAAA,MAAM,OAAO,MAAM,QAAA;AAAA,MACjB,MAAA;AAAA,MACA,SAAA;AAAA,MACA,UAAA;AAAA,MACA,MAAA;AAAA,MACA,KAAK,KAAA,IAAS,EAAA;AAAA,MACd,WAAA,CAAY,KAAK,MAAM,CAAA;AAAA,MACvB,WAAA;AAAA,MACA,OAAA;AAAA,MACA,IAAA,CAAK;AAAA,KACP;AACA,IAAA,IAAI,WAAW,MAAA,EAAQ,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAG,IAAA,EAAK;AAOvD,IAAA,MAAM,UAAA,GAAa,MAAM,MAAA,CAAO,iBAAA,CAAkB;AAAA,MAChD,aAAa,IAAA,CAAK,SAAA,CAAU,IAAI,CAAC,CAAA,KAAM,EAAE,EAAE,CAAA;AAAA,MAC3C,SAAA;AAAA,MACA,UAAA;AAAA;AAAA;AAAA;AAAA;AAAA,MAKA,OAAA,EAAS,CAAC,oBAAA,EAAsB,MAAA;AAAA,MAChC,WAAW,IAAA,CAAK;AAAA,KACjB,CAAA;AACD,IAAA,KAAA,MAAW,GAAA,IAAO,KAAK,SAAA,EAAW,GAAA,CAAI,YAAY,UAAA,CAAW,GAAA,CAAI,EAAE,CAAA,IAAK,EAAC;AACzE,IAAA,MAAM,cAAA,GAAiB,KAAK,SAAA,CAAU,IAAA,CAAK,CAAC,CAAA,KAAM,CAAA,CAAE,SAAS,aAAa,CAAA;AAI1E,IAAA,MAAM,YAAA,GAAe,MAAM,MAAA,CAAO,YAAA,CAAa;AAAA,MAC7C,SAAA;AAAA,MACA,WAAA;AAAA,MACA,UAAA;AAAA,MACA,WAAW,IAAA,CAAK;AAAA,KACjB,CAAA;AACD,IAAA,MAAM,MAAA,GAAS,MAAM,MAAA,CAAO,aAAA,CAAc;AAAA,MACxC,SAAA;AAAA,MACA,WAAA;AAAA,MACA,UAAA;AAAA,MACA,WAAW,IAAA,CAAK;AAAA,KACjB,CAAA;AACD,IAAA,MAAM,QAAA,GAAW,EAAE,MAAA,EAAQ,QAAA,EAAU,OAAO,gCAAA,EAAiC;AAC7E,IAAA,MAAM,aAAsC,EAAC;AAC7C,IAAA,IAAI,cAAA,IAAkB,UAAU,OAAA,EAAS;AACvC,MAAA,UAAA,CAAW,iCAAiC,CAAA,GAAI;AAAA,QAC9C,MAAA,EAAQ,SAAA;AAAA,QACR,KAAA,EAAO;AAAA,OACT;AAAA,IACF;AACA,IAAA,UAAA,CAAW,wBAAwB,CAAA,GAAI,QAAA;AACvC,IAAA,IAAI,YAAA,CAAa,MAAA,IAAU,SAAA,CAAU,UAAA,EAAY;AAC/C,MAAA,UAAA,CAAW,gCAAgC,CAAA,GAAI;AAAA,QAC7C,MAAA,EAAQ,YAAA;AAAA,QACR,KAAA,EAAO;AAAA,OACT;AAAA,IACF;AACA,IAAA,IAAI,UAAU,aAAA,EAAe;AAC3B,MAAA,UAAA,CAAW,wBAAwB,CAAA,GAAI;AAAA,QACrC,MAAA,EAAQ,eAAA;AAAA,QACR,MAAA,EAAQ;AAAA,OACV;AAAA,IACF;AACA,IAAA,MAAM,UAAA,GAAsC;AAAA,MAC1C,OAAA,EAAS,IAAA;AAAA,MACT,GAAG,IAAA;AAAA,MACH,cAAA,EAAgB,MAAA;AAAA,MAChB,cAAA,EAAgB,YAAA;AAAA,MAChB,mBAAmB,MAAA,CAAO,WAAA;AAAA,QACvB,OAAO,IAAA,CAAK,WAAW,CAAA,CAAsC,GAAA,CAAI,CAAC,IAAA,KAAS;AAAA,UAC1E,IAAA;AAAA,UACA,SAAA,CAAU,IAAI,CAAA,GAAI,WAAA,CAAY,IAAI,CAAA,GAAI,CAAA,EAAG,WAAA,CAAY,IAAI,CAAC,CAAA,MAAA;AAAA,SAC3D;AAAA,OACH;AAAA,MACA,WAAA,EAAa;AAAA,KACf;AACA,IAAA,OAAO,UAAA;AAAA,EACT;AAEA,EAAA,IAAI,WAAW,QAAA,EAAU;AACvB,IAAA,IAAI,CAAC,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,sBAAA,EAAuB;AAClE,IAAA,MAAM,MAAA,GAAS,MAAM,MAAA,CAAO,MAAA,CAAO,IAAA,EAAM;AAAA,MACvC,SAAA;AAAA,MACA,WAAA;AAAA,MACA,UAAA;AAAA,MACA,IAAA,EAAM,KAAK,KAAA,IAAS,EAAA;AAAA;AAAA,MAEpB,IAAA,EAAM,KAAK,IAAA,IAAQ,IAAA;AAAA,MACnB,WAAW,IAAA,CAAK;AAAA,KACjB,CAAA;AACD,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,IAAA;AAAA,MACT,OAAO,MAAA,CAAO,IAAA,IAAQ,EAAC,EAAG,GAAA,CAAI,CAAC,GAAA,KAAQ;AACrC,QAAA,MAAM,CAAA,GAAI,GAAA;AACV,QAAA,OAAO;AAAA,UACL,WAAA,EAAa,CAAA,CAAE,WAAA,IAAe,CAAA,CAAE,UAAA;AAAA,UAChC,aAAA,EAAe,CAAA,CAAE,aAAA,IAAiB,CAAA,CAAE,YAAA;AAAA,UACpC,UAAA,EAAY,CAAA,CAAE,UAAA,IAAc,CAAA,CAAE,SAAA;AAAA,UAC9B,OAAO,CAAA,CAAE,KAAA;AAAA,UACT,SAAA,EAAW,CAAA,CAAE,SAAA,IAAa,CAAA,CAAE;AAAA,SAC9B;AAAA,MACF,CAAC,CAAA;AAAA,MACD,OAAO,MAAA,CAAO;AAAA,KAChB;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,SAAA,EAAW;AACxB,IAAA,MAAM,aAAa,IAAA,CAAK,WAAA;AACxB,IAAA,IAAI,CAAC,UAAA,EAAY,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,6BAAA,EAA8B;AAI/E,IAAA,IAAI,CAAC,aAAA,CAAc,UAAA,EAAY,OAAO,CAAA,EAAG;AACvC,MAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,IACtE;AACA,IAAA,IAAI;AACF,MAAA,MAAM,QAAA,GAAW,MAAM,MAAA,CAAO,WAAA,CAAY,UAAA,EAAY;AAAA,QACpD,UAAA;AAAA,QACA,WAAW,IAAA,CAAK;AAAA,OACjB,CAAA;AACD,MAAA,MAAM,GAAA,GAAM,QAAA,EAAU,QAAA,IAAY,QAAA,EAAU,SAAA;AAC5C,MAAA,IAAI,OAAA,CAAQ,SAAA,IAAa,IAAA,IAAQ,CAAC,OAAA,CAAQ,UAAU,QAAA,CAAS,MAAA,CAAO,GAAG,CAAC,CAAA,EAAG;AACzE,QAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,MACtE;AACA,MAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,QAAA,EAAS;AAAA,IACnC,SAAS,GAAA,EAAK;AAGZ,MAAA,IAAI,GAAA,YAAe,uBAAuB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AACtF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,YAAA,EAAc;AAC3B,IAAA,IAAI,CAAC,SAAA,CAAU,UAAA,IAAc,CAAC,MAAA,CAAO,eAAA,EAAiB,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,UAAA,EAAW;AACrG,IAAA,IAAI,CAAC,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,0BAAA,EAA2B;AACtE,IAAA,MAAM,GAAA,GAAM,MAAM,MAAA,CAAO,eAAA,CAAgB,IAAA,EAAM;AAAA,MAC7C,SAAA;AAAA,MACA,UAAA;AAAA,MACA,KAAA,EAAO,IAAA,CAAK,GAAA,CAAI,CAAA,EAAG,IAAA,CAAK,IAAI,IAAA,CAAK,KAAA,IAAS,EAAA,EAAI,GAAG,CAAC,CAAA;AAAA,MAClD,WAAW,IAAA,CAAK;AAAA,KACjB,CAAA;AACD,IAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAI,GAAA,EAAgC;AAAA,EAC9D;AAEA,EAAA,IAAI,WAAW,SAAA,EAAW;AACxB,IAAA,IAAI,CAAC,UAAU,OAAA,EAAS,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,OAAA,EAAQ;AACpE,IAAA,IAAI,CAAC,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,wDAAA,EAAyD;AAKpG,IAAA,MAAM,MAAA,GAAS,IAAA,CAAK,OAAA,IAAW,MAAA,CAAO,OAAA;AACtC,IAAA,IAAI,CAAC,QAAQ,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AACzD,IAAA,IAAI;AACF,MAAA,MAAM,GAAA,GAAM,MAAM,MAAA,CAAO,IAAA,EAAM,EAAE,SAAA,EAAW,WAAA,EAAa,YAAY,CAAA;AACrE,MAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAI,GAAA,EAAgC;AAAA,IAC9D,SAAS,GAAA,EAAK;AAGZ,MAAA,IAAI,GAAA,YAAe,mBAAmB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AAClF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,YAAA,EAAc;AAC3B,IAAA,MAAM,aAAa,IAAA,CAAK,WAAA;AACxB,IAAA,IAAI,CAAC,UAAA,EAAY,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,gCAAA,EAAiC;AAClF,IAAA,IAAI,CAAC,aAAA,CAAc,UAAA,EAAY,OAAO,CAAA,EAAG;AACvC,MAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,IACtE;AACA,IAAA,IAAI;AACF,MAAA,MAAM,IAAA,GAAO,MAAM,MAAA,CAAO,SAAA,CAAW,UAAA,EAAY;AAAA,QAC/C,UAAA;AAAA,QACA,OAAO,IAAA,CAAK,KAAA;AAAA,QACZ,KAAK,IAAA,CAAK,GAAA;AAAA,QACV,WAAW,IAAA,CAAK;AAAA,OACjB,CAAA;AACD,MAAA,IAAI,OAAA,CAAQ,aAAa,IAAA,EAAM;AAG7B,QAAA,MAAM,MAAM,MAAM,MAAA,CAAO,YAAY,UAAA,EAAY,EAAE,YAAY,CAAA;AAC/D,QAAA,MAAM,GAAA,GAAM,GAAA,EAAK,QAAA,IAAY,GAAA,EAAK,SAAA;AAClC,QAAA,IAAI,CAAC,OAAA,CAAQ,SAAA,CAAU,SAAS,MAAA,CAAO,GAAG,CAAC,CAAA,EAAG;AAC5C,UAAA,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,CAAA,oBAAA,EAAuB,UAAU,CAAA,CAAA,EAAG;AAAA,QACtE;AAAA,MACF;AACA,MAAA,MAAM,GAAA,GAA+B,EAAE,OAAA,EAAS,IAAA,EAAM,GAAG,IAAA,EAAK;AAC9D,MAAA,IAAI,KAAK,QAAA,EAAU;AACjB,QAAA,GAAA,CAAI,SAAA,GAAY;AAAA,UACd,MAAA,EAAQ,YAAA;AAAA,UACR,WAAA,EAAa,OAAO,UAAU,CAAA;AAAA,UAC9B,OAAO,IAAA,CAAK;AAAA,SACd;AAAA,MACF;AACA,MAAA,OAAO,GAAA;AAAA,IACT,SAAS,GAAA,EAAK;AACZ,MAAA,IAAI,GAAA,YAAe,uBAAuB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AACtF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,IAAI,WAAW,UAAA,EAAY;AACzB,IAAA,IAAI,CAAC,aAAa,MAAA,EAAQ,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,6BAAA,EAA8B;AACxF,IAAA,MAAM,IAAA,GAAO,MAAM,MAAA,CAAO,YAAA,CAAc,WAAA,EAAa,EAAE,UAAA,EAAY,SAAA,EAAW,IAAA,CAAK,SAAA,EAAW,CAAA;AAI9F,IAAA,MAAM,MAAA,GAAS,KAAK,GAAA,CAAI,GAAA,EAAM,KAAK,KAAA,CAAM,IAAA,CAAK,SAAA,IAAa,sBAAsB,CAAC,CAAA;AAClF,IAAA,MAAM,OAAuC,EAAC;AAC9C,IAAA,IAAI,IAAA,GAAO,CAAA;AACX,IAAA,KAAA,MAAW,OAAO,IAAA,EAAM;AACtB,MAAA,MAAM,IAAA,GAAO,MAAA,CAAO,GAAA,CAAI,IAAA,IAAQ,EAAE,CAAA,CAAE,MAAA;AACpC,MAAA,IAAI,IAAA,CAAK,MAAA,IAAU,IAAA,GAAO,IAAA,GAAO,MAAA,EAAQ;AACzC,MAAA,IAAA,CAAK,KAAK,GAAG,CAAA;AACb,MAAA,IAAA,IAAQ,IAAA;AAAA,IACV;AACA,IAAA,MAAM,QAAA,GAAW,IAAI,GAAA,CAAI,IAAA,CAAK,GAAA,CAAI,CAAC,CAAA,KAAM,MAAA,CAAO,CAAA,CAAE,EAAE,CAAC,CAAC,CAAA;AACtD,IAAA,MAAM,SAAA,GAAA,CAAa,IAAA,CAAK,YAAA,IAAgB,IAAI,MAAA,CAAO,CAAC,CAAA,KAAM,CAAC,QAAA,CAAS,GAAA,CAAI,MAAA,CAAO,CAAC,CAAC,CAAC,CAAA;AAClF,IAAA,MAAM,GAAA,GAA+B,EAAE,OAAA,EAAS,IAAA,EAAM,WAAW,IAAA,EAAM,KAAA,EAAO,KAAK,MAAA,EAAO;AAC1F,IAAA,IAAI,UAAU,MAAA,EAAQ;AACpB,MAAA,GAAA,CAAI,sBAAA,GAAyB,SAAA;AAC7B,MAAA,GAAA,CAAI,SAAA,GAAY,EAAE,MAAA,EAAQ,UAAA,EAAY,cAAc,SAAA,EAAU;AAAA,IAChE;AACA,IAAA,OAAO,GAAA;AAAA,EACT;AAEA,EAAA,IAAI,WAAW,YAAA,EAAc;AAC3B,IAAA,IAAI,CAAC,UAAU,UAAA,EAAY,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,UAAA,EAAW;AAC1E,IAAA,IAAI,CAAC,IAAA,EAAM;AACT,MAAA,OAAO;AAAA,QACL,OAAA,EAAS,KAAA;AAAA,QACT,KAAA,EAAO;AAAA,OACT;AAAA,IACF;AAGA,IAAA,MAAM,MAAA,GAAS,IAAA,CAAK,UAAA,IAAc,MAAA,CAAO,SAAA;AACzC,IAAA,IAAI,CAAC,QAAQ,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,UAAA,EAAW;AAC5D,IAAA,IAAI;AACF,MAAA,MAAM,GAAA,GAAM,MAAM,MAAA,CAAO,IAAA,EAAM;AAAA,QAC7B,SAAA;AAAA,QACA,WAAA;AAAA,QACA,UAAA;AAAA,QACA,OAAO,IAAA,CAAK;AAAA,OACb,CAAA;AACD,MAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,GAAI,GAAA,EAAgC;AAAA,IAC9D,SAAS,GAAA,EAAK;AACZ,MAAA,IAAI,GAAA,YAAe,mBAAmB,OAAO,EAAE,SAAS,KAAA,EAAO,KAAA,EAAO,IAAI,OAAA,EAAQ;AAClF,MAAA,MAAM,GAAA;AAAA,IACR;AAAA,EACF;AAEA,EAAA,OAAO,gBAAgB,MAAA,EAAQ,IAAA,EAAM,EAAE,SAAA,EAAW,UAAA,EAAY,MAAM,CAAA;AACtE;AAEA,eAAe,eAAA,CACb,MAAA,EACA,IAAA,EACA,GAAA,EACkC;AAGlC,EAAA,MAAM,MAAM,MAAM,OAAA,CAAA,OAAA,EAAA,CAAA,IAAA,CAAA,OAAA,eAAA,EAAA,EAAA,kBAAA,CAAA,CAAA;AAClB,EAAA,MAAM,IAAA,GAAQ,MAAM,MAAA,CAAO,UAAA,IAAa;AAExC,EAAA,MAAM,UAAA,GAAc,IAAI,UAAA,IAAc,IAAA;AACtC,EAAA,MAAM,MAAA,GAAS,EAAE,SAAA,EAAW,GAAA,CAAI,WAAW,UAAA,EAAW;AAEtD,EAAA,IAAI,IAAA,CAAK,WAAW,mBAAA,EAAqB;AACvC,IAAA,IAAI,CAAC,IAAI,IAAA,EAAM,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,OAAO,iCAAA,EAAkC;AACjF,IAAA,MAAMA,IAAAA,GAAM,MAAM,GAAA,CAAI,gBAAA,CAAiB,IAAA,EAAM;AAAA,MAC3C,UAAU,MAAA,CAAO,QAAA;AAAA,MACjB,OAAO,GAAA,CAAI,IAAA;AAAA,MACX,OAAO,IAAA,CAAK,KAAA;AAAA,MACZ,WAAW,GAAA,CAAI,SAAA;AAAA,MACf;AAAA,KACD,CAAA;AACD,IAAA,IAAI,CAACA,KAAI,SAAA,EAAW,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAOA,IAAAA,CAAI,KAAA,EAAgB;AACxE,IAAA,OAAO,EAAE,SAAS,IAAA,EAAM,WAAA,EAAaA,KAAI,WAAA,EAAa,KAAA,EAAOA,KAAI,KAAA,EAAM;AAAA,EACzE;AAEA,EAAA,IAAI,IAAA,CAAK,WAAW,cAAA,EAAgB;AAClC,IAAA,IAAI,EAAE,IAAA,CAAK,QAAA,IAAY,IAAA,CAAK,KAAA,IAAS,KAAK,WAAA,CAAA,EAAc;AACtD,MAAA,OAAO;AAAA,QACL,OAAA,EAAS,KAAA;AAAA,QACT,KAAA,EAAO;AAAA,OACT;AAAA,IACF;AACA,IAAA,MAAMA,IAAAA,GAAM,MAAM,GAAA,CAAI,WAAA,CAAY,IAAA,EAAM;AAAA,MACtC,GAAG,MAAA;AAAA,MACH,UAAU,IAAA,CAAK,QAAA;AAAA,MACf,OAAO,IAAA,CAAK,KAAA;AAAA,MACZ,YAAY,IAAA,CAAK,WAAA;AAAA,MACjB,OAAO,IAAA,CAAK;AAAA,KACb,CAAA;AACD,IAAA,OAAO,EAAE,SAAS,IAAA,EAAM,aAAA,EAAeA,KAAI,aAAA,EAAe,KAAA,EAAOA,KAAI,KAAA,EAAM;AAAA,EAC7E;AAEA,EAAA,IAAI,CAAC,KAAK,MAAA,EAAQ;AAChB,IAAA,OAAO;AAAA,MACL,OAAA,EAAS,KAAA;AAAA,MACT,KAAA,EAAO,CAAA,EAAG,IAAA,CAAK,MAAM,CAAA,mDAAA;AAAA,KACvB;AAAA,EACF;AACA,EAAA,IAAI,IAAA,CAAK,WAAW,eAAA,EAAiB;AACnC,IAAA,MAAMA,IAAAA,GAAM,MAAM,GAAA,CAAI,YAAA,CAAa,MAAM,EAAE,GAAG,MAAA,EAAQ,MAAA,EAAQ,IAAA,CAAK,MAAA,EAAQ,KAAA,EAAO,IAAA,CAAK,OAAO,CAAA;AAC9F,IAAA,IAAI,CAACA,KAAI,KAAA,EAAO,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAOA,IAAAA,CAAI,KAAA,EAAgB;AACpE,IAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,MAAA,EAAQA,IAAAA,CAAI,MAAA,EAAQ,SAAA,EAAWA,IAAAA,CAAI,SAAA,EAAW,KAAA,EAAOA,IAAAA,CAAI,KAAA,EAAM;AAAA,EACzF;AACA,EAAA,MAAM,GAAA,GAAM,MAAM,GAAA,CAAI,QAAA,CAAS,IAAA,EAAM;AAAA,IACnC,GAAG,MAAA;AAAA,IACH,QAAQ,IAAA,CAAK,MAAA;AAAA,IACb,OAAO,IAAA,CAAK,KAAA;AAAA,IACZ,UAAU,IAAA,CAAK,QAAA;AAAA,IACf,OAAO,IAAA,CAAK;AAAA,GACb,CAAA;AACD,EAAA,IAAI,CAAC,IAAI,KAAA,EAAO,OAAO,EAAE,OAAA,EAAS,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,KAAA,EAAgB;AACpE,EAAA,OAAO,EAAE,OAAA,EAAS,IAAA,EAAM,MAAA,EAAQ,GAAA,CAAI,MAAA,EAAQ,KAAA,EAAO,GAAA,CAAI,KAAA,EAAO,KAAA,EAAO,GAAA,CAAI,KAAA,EAAM;AACjF;ACp6BO,IAAM,YAAA,GAAN,cAA2B,KAAA,CAAM;AAAA,EACtC,MAAA;AAAA,EACA,MAAA;AAAA,EACA,WAAA,CAAY,QAAgB,MAAA,EAAiB;AAC3C,IAAA,KAAA,CAAM,OAAO,MAAA,KAAW,QAAA,GAAW,SAAS,IAAA,CAAK,SAAA,CAAU,MAAM,CAAC,CAAA;AAClE,IAAA,IAAA,CAAK,IAAA,GAAO,cAAA;AACZ,IAAA,IAAA,CAAK,MAAA,GAAS,MAAA;AACd,IAAA,IAAA,CAAK,MAAA,GAAS,MAAA;AAAA,EAChB;AACF,CAAA;AAEuCC,MACpC,MAAA,CAAO;AAAA,EACN,IAAA,EAAMA,MAAE,MAAA,EAAO;AAAA,EACf,IAAA,EAAMA,MAAE,MAAA,EAAO;AAAA,EACf,WAAWA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC1C,aAAaA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC5C,aAAaA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC5C,SAAA,EAAWA,MAAE,MAAA,CAAOA,KAAA,CAAE,SAAS,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EACrD,GAAA,EAAKA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EAC7C,IAAA,EAAMA,KAAA,CAAE,IAAA,CAAK,CAAC,QAAA,EAAU,OAAO,CAAC,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EACtD,oBAAoBA,KAAA,CAAE,OAAA,GAAU,QAAA,EAAS,CAAE,QAAQ,KAAK,CAAA;AAAA,EACxD,OAAOA,KAAA,CAAE,OAAA,GAAU,QAAA,EAAS,CAAE,QAAQ,KAAK;AAC7C,CAAC,EACA,KAAA;AAEgCA,MAChC,MAAA,CAAO;AAAA,EACN,GAAA,EAAKA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EAC7C,MAAMA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EACrC,aAAaA,KAAA,CAAE,MAAA,EAAO,CAAE,QAAA,GAAW,QAAA,EAAS;AAAA,EAC5C,SAAA,EAAWA,MAAE,MAAA,CAAOA,KAAA,CAAE,SAAS,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA;AAC9C,CAAC,EACA,MAAA;AAEgCA,MAChC,MAAA,CAAO;AAAA,EACN,KAAA,EAAOA,MAAE,MAAA,EAAO;AAAA,EAChB,UAAA,EAAYA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA;AAAA;AAAA;AAAA,EAIpD,YAAA,EAAcA,MAAE,KAAA,CAAMA,KAAA,CAAE,QAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,EAAS;AAAA,EACtD,KAAA,EAAOA,MAAE,MAAA,EAAO,CAAE,KAAI,CAAE,QAAA,EAAS,CAAE,OAAA,CAAQ,EAAE,CAAA;AAAA,EAC7C,IAAA,EAAMA,KAAA,CAAE,IAAA,CAAK,CAAC,QAAA,EAAU,OAAO,CAAC,CAAA,CAAE,QAAA,EAAS,CAAE,OAAA,CAAQ,QAAQ,CAAA;AAAA,EAC7D,kBAAA,EAAoBA,MAAE,MAAA,EAAO,CAAE,KAAI,CAAE,QAAA,GAAW,QAAA;AAClD,CAAC,EACA,KAAA;AAgBH,IAAM,OAAA,GAAU,iEAAA;AAET,SAAS,iBAAiB,UAAA,EAA0B;AACzD,EAAA,IAAI,CAAC,OAAA,CAAQ,IAAA,CAAK,MAAA,CAAO,UAAU,CAAC,CAAA,EAAG;AACrC,IAAA,MAAM,IAAI,aAAa,GAAA,EAAK,CAAA,qBAAA,EAAwB,KAAK,SAAA,CAAU,UAAU,CAAC,CAAA,CAAE,CAAA;AAAA,EAClF;AACF;AAsBO,SAAS,wBAAA,CAAyB,OAAgB,IAAA,EAA+C;AAKtG,EAAA,IAAI,KAAA,KAAU,SAAS,OAAO,OAAA;AAC9B,EAAA,IAAI,SAAS,IAAA,EAAM;AACjB,IAAA,MAAM,IAAI,YAAA;AAAA,MACR,GAAA;AAAA,MACA,CAAA,EAAG,KAAK,OAAO,CAAA,qWAAA;AAAA,KAMjB;AAAA,EACF;AAIA,EAAA,IAAI,CAAC,KAAA,CAAM,OAAA,CAAQ,KAAK,CAAA,IAAK,KAAA,CAAM,IAAA,CAAK,CAAC,CAAA,KAAM,OAAO,CAAA,KAAM,QAAQ,CAAA,EAAG;AACrE,IAAA,MAAM,IAAI,YAAA;AAAA,MACR,GAAA;AAAA,MACA,CAAA,EAAG,IAAA,CAAK,OAAO,CAAA,2HAAA,EACsD,OAAO,KAAK,CAAA,CAAA;AAAA,KACnF;AAAA,EACF;AACA,EAAA,OAAO,KAAA;AACT;AChHO,IAAM,UAAA,GACX,wLAAA;AAIF,eAAsB,oBAAoB,UAAA,EAAuD;AAC/F,EAAA,MAAM,MAAA,GAAS,MAAM,OAAA,CAAQ,OAAA,CAAQ,YAAY,CAAA;AAQjD,EAAA,IAAI;AACF,IAAA,OAAO,wBAAA,CAAyB,MAAA,EAAQ,EAAE,OAAA,EAAS,mCAAmC,CAAA;AAAA,EACxF,SAAS,GAAA,EAAK;AACZ,IAAA,IAAI,GAAA,YAAe,YAAA,EAAc,MAAM,IAAI,KAAA,CAAM,IAAI,OAAA,EAAS,EAAE,KAAA,EAAO,GAAA,EAAK,CAAA;AAC5E,IAAA,MAAM,GAAA;AAAA,EACR;AACF;AAWO,SAAS,mBAAA,CACd,GAAA,EACA,MAAA,EAQA,UAAA,EAMA,gBAAwC,IAAA,EAClC;AACN,EAAA,GAAA,CAAI,IAAA;AAAA,IACF,cAAA;AAAA,IACA,iPAAA,GAIE,UAAA;AAAA;AAAA,IAEF,EAAE,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,EAAG,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,EAAE;AAAA,IACxD,UAAU,GAAA,KAAiB;AACzB,MAAA,MAAM,IAAA,GAAQ,GAAA,CAAI,CAAC,CAAA,IAAK,EAAC;AACzB,MAAA,MAAM,gBAAA,GAAmB,MAAM,mBAAA,CAAoB,UAAU,CAAA;AAC7D,MAAA,MAAM,KAAA,GAAQ,MAAM,MAAA,CAAO,WAAA,CAAY,OAAO,IAAA,CAAK,KAAA,IAAS,EAAE,CAAA,EAAG;AAAA,QAC/D,UAAA,EAAY,gBAAA;AAAA,QACZ,KAAA,EAAO,KAAK,KAAA,IAAS;AAAA,OACtB,CAAA;AACD,MAAA,OAAO,EAAE,KAAA,EAAM;AAAA,IACjB;AAAA,GACF;AAEA,EAAA,GAAA,CAAI,IAAA;AAAA,IACF,cAAA;AAAA,IACA,iTAAA,GAKE,UAAA;AAAA,IACF,EAAE,IAAA,EAAMA,KAAAA,CAAE,MAAA,EAAO,EAAG,IAAA,EAAMA,KAAAA,CAAE,MAAA,CAAOA,KAAAA,CAAE,OAAA,EAAS,CAAA,CAAE,UAAS,EAAE;AAAA,IAC3D,UAAU,GAAA,KAAiB;AACzB,MAAA,MAAM,IAAA,GAAQ,GAAA,CAAI,CAAC,CAAA,IAAK,EAAC;AACzB,MAAA,MAAM,gBAAA,GAAmB,MAAM,mBAAA,CAAoB,UAAU,CAAA;AAC7D,MAAA,MAAM,QAAQ,aAAA,GAAgB,MAAM,QAAQ,OAAA,CAAQ,aAAA,EAAe,CAAA,GAAI,IAAA;AACvE,MAAA,OAAO,MAAA,CAAO,YAAY,MAAA,CAAO,IAAA,CAAK,IAAI,CAAA,EAAG,IAAA,CAAK,QAAQ,IAAA,EAAM;AAAA,QAC9D,UAAA,EAAY,gBAAA;AAAA,QACZ,MAAA,EAAQ,KAAA;AAAA,QACR,aAAA,EAAe;AAAA,OAChB,CAAA;AAAA,IACH;AAAA,GACF;AACF;;;AClGO,IAAM,WAAA,GAAc,OAAA;;;AXqC3B,SAAS,UAAU,IAAA,EAAsB;AACvC,EAAA,OAAO,MAAA,CAAO,gBAAA,CAAiB,IAAI,CAAA,EAAG,eAAe,EAAE,CAAA;AACzD;AAEA,SAAS,YAAY,IAAA,EAAmE;AACtF,EAAA,OAAO,EAAE,OAAA,EAAS,CAAC,EAAE,IAAA,EAAM,MAAA,EAAQ,IAAA,EAAM,IAAA,CAAK,SAAA,CAAU,IAAI,CAAA,EAAG,CAAA,EAAE;AACnE;AAEA,eAAsB,YAAA,CACpB,QACA,IAAA,EAkCA;AACA,EAAA,IAAI,CAAC,MAAM,UAAA,EAAY;AACrB,IAAA,MAAM,IAAI,UAAU,kCAAkC,CAAA;AAAA,EACxD;AACA,EAAA,IAAI,IAAA,CAAK,KAAA,KAAU,MAAA,IAAa,IAAA,CAAK,UAAU,IAAA,EAAM;AACnD,IAAA,MAAM,IAAI,SAAA;AAAA,MACR;AAAA,KAEF;AAAA,EACF;AAEA,EAAA,IAAI,SAAA;AAOJ,EAAA,IAAI,6BAAA;AAKJ,EAAA,IAAI;AACF,IAAA,MAAM,YAAY,MAAM,YAAA;AAAA,MACtB,yCAAA;AAAA,MACA,KAAA;AAAA,MACA;AAAA,KACF;AACA,IAAA,MAAM,OAAA,GAAU,MAAM,YAAA,CAEnB,oDAAA,EAAsD,OAAO,YAAY,CAAA;AAC5E,IAAA,SAAA,GAAY,SAAA,CAAU,SAAA;AACtB,IAAA,6BAAA,GAAgC,OAAA,CAAQ,6BAAA;AAAA,EAC1C,SAAS,GAAA,EAAK;AACZ,IAAA,IAAI,GAAA,YAAe,mBAAmB,MAAM,GAAA;AAC5C,IAAA,MAAM,IAAI,iBAAA,CAAkB,KAAA,EAAO,2BAAA,EAA6B,YAAY,CAAA;AAAA,EAC9E;AAEA,EAAA,MAAM,GAAA,GAAM,IAAI,SAAA,CAAU,EAAE,MAAM,gBAAA,EAAkB,OAAA,EAAS,aAAa,CAAA;AAE1E,EAAA,MAAM,IAAA,GAAO,CACX,IAAA,EACA,WAAA,EACA,QACAC,QAAAA,KACG;AACH,IAAA,GAAA,CAAI,IAAA;AAAA,MAAK,IAAA;AAAA,MAAM,WAAA;AAAA,MAAa,MAAA;AAAA,MAAQ,OAAO,IAAA,KACzC,WAAA,CAAY,MAAMA,QAAAA,CAAQ,IAAI,CAAC;AAAA,KACjC;AAAA,EACF,CAAA;AAEA,EAAA,IAAA;AAAA,IACE,uBAAA;AAAA,IACA,CAAA,EAAG,0BAA0B,CAAA,EAAG,UAAU,CAAA,CAAA;AAAA,IAC1C;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA,MAQE,MAAA,EAAQD,MAAE,IAAA,CAAK,iBAAiB,EAAE,QAAA,CAAS,SAAA,CAAU,QAAQ,CAAC,CAAA;AAAA,MAC9D,KAAA,EAAOA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MACxD,WAAA,EAAaA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,aAAa,CAAC,CAAA;AAAA,MACpE,UAAA,EAAYA,KAAAA,CAAE,KAAA,CAAMA,KAAAA,CAAE,MAAA,EAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,YAAY,CAAC,CAAA;AAAA,MAC3E,YAAA,EAAcA,KAAAA,CAAE,KAAA,CAAMA,KAAAA,CAAE,MAAA,EAAQ,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,cAAc,CAAC,CAAA;AAAA,MAC/E,MAAA,EAAQA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,QAAQ,CAAC,CAAA;AAAA,MAC1D,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,QAAA,EAAUA,KAAAA,CAAE,IAAA,CAAK,4BAA4B,CAAA,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,UAAU,CAAC,CAAA;AAAA,MACxF,KAAA,EAAOA,MAAE,MAAA,EAAO,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MACxD,WAAA,EAAaA,KAAAA,CAAE,IAAA,CAAK,YAAY,CAAA,CAAE,UAAS,CAAE,QAAA,CAAS,SAAA,CAAU,aAAa,CAAC,CAAA;AAAA,MAC9E,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,IAAA,EAAMA,KAAAA,CAAE,IAAA,CAAK,CAAC,QAAA,EAAU,OAAO,CAAC,CAAA,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,MAAM,CAAC,CAAA;AAAA,MACvE,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,MAAA,EAAQA,MACL,KAAA,CAAM,CAACA,MAAE,MAAA,EAAO,EAAGA,MAAE,MAAA,CAAOA,KAAAA,CAAE,SAAS,CAAC,CAAC,CAAA,CACzC,QAAA,GACA,QAAA,CAAS,SAAA,CAAU,QAAQ,CAAC,CAAA;AAAA,MAC/B,KAAA,EAAOA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,OAAO,CAAC,CAAA;AAAA,MAC9D,GAAA,EAAKA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,KAAK,CAAC,CAAA;AAAA,MAC1D,SAAA,EAAWA,KAAAA,CAAE,MAAA,EAAO,CAAE,GAAA,EAAI,CAAE,QAAA,EAAS,CAAE,QAAA,CAAS,SAAA,CAAU,WAAW,CAAC;AAAA,KACxE;AAAA,IACA,OAAO,IAAA,KAAS;AAGd,MAAA,MAAM,cAAc,IAAA,CAAK,YAAA;AACzB,MAAA,MAAM,aAAa,IAAA,CAAK,WAAA;AACxB,MAAA,KAAA,MAAW,GAAA,IAAO,CAAC,GAAI,WAAA,IAAe,EAAC,EAAI,GAAI,UAAA,GAAa,CAAC,UAAU,CAAA,GAAI,EAAG,CAAA,EAAG;AAC/E,QAAA,gBAAA,CAAiB,GAAG,CAAA;AAAA,MACtB;AAIA,MAAA,MAAM,gBAAA,GAAmB,MAAM,mBAAA,CAAoB,IAAA,CAAK,UAAU,CAAA;AAClE,MAAA,MAAM,OAAA,GAAU,OAAO,IAAA,CAAK,KAAA,KAAU,aAAa,MAAM,IAAA,CAAK,KAAA,EAAM,GAAI,IAAA,CAAK,KAAA;AAC7E,MAAA,MAAM,MAAA,GAAS,OAAO,IAAA,CAAK,SAAA,KAAc,aAAa,MAAM,IAAA,CAAK,SAAA,EAAU,GAAI,IAAA,CAAK,SAAA;AACpF,MAAA,OAAO,kBAAkB,MAAA,EAAiB;AAAA,QACxC,GAAI,IAAA;AAAA,QACJ,MAAA,EAAQ,MAAA,CAAO,IAAA,CAAK,MAAM,CAAA;AAAA,QAC1B,UAAA,EAAY,gBAAA;AAAA,QACZ,KAAA,EAAO,OAAA;AAAA,QACP,SAAA,EAAW,MAAA;AAAA,QACX,OAAA,EAAS,KAAK,OAAA,IAAW,IAAA;AAAA,QACzB,UAAA,EAAY,KAAK,SAAA,IAAa;AAAA,OAC/B,CAAA;AAAA,IACH;AAAA,GACF;AAEA,EAAA,mBAAA;AAAA,IACE;AAAA,MACE,IAAA,EAAM,CAAC,IAAA,EAAM,WAAA,EAAa,QAAQC,QAAAA,KAAY;AAC5C,QAAA,IAAA,CAAK,MAAM,WAAA,EAAa,MAAA,EAAQ,OAAO,IAAA,KAASA,QAAAA,CAAQ,IAAa,CAAC,CAAA;AAAA,MACxE;AAAA,KACF;AAAA,IACA,MAAA;AAAA,IACA,IAAA,CAAK,UAAA;AAAA,IACL,KAAK,aAAA,IAAiB;AAAA,GACxB;AAEA,EAAA,MAAM,OAAA,IAAW,OACf,GAAA,EACA,GAAA,KACG;AACH,IAAA,MAAM,YAAY,IAAI,6BAAA,CAA8B,EAAE,kBAAA,EAAoB,QAAW,CAAA;AACrF,IAAA,MAAM,GAAA,CAAI,QAAQ,SAAS,CAAA;AAC3B,IAAA,MAAM,SAAA,CAAU,aAAA,CAAc,GAAA,EAAK,GAAG,CAAA;AAAA,EACxC,CAAA,CAAA;AAIA,EAAA,OAAA,CAAQ,GAAA,GAAM,GAAA;AACd,EAAA,OAAO,OAAA;AACT","file":"mcp.cjs","sourcesContent":["/** Action-surface errors — user-facing / data-dependent failures (not found, ACL, bad input).\n * Config/programming errors stay as TypeError / RangeError / Error.\n */\nexport class EngineActionError extends Error {\n constructor(message: string) {\n super(message);\n this.name = \"EngineActionError\";\n }\n}\n\n/** Missing vs forbidden document: identical message so callers cannot distinguish. */\nexport class DocumentNotFoundError extends Error {\n constructor(documentId: string) {\n super(`document not found: ${documentId}`);\n this.name = \"DocumentNotFoundError\";\n }\n}\n\nexport class GraphLegUnavailable extends Error {\n constructor(message = \"graph ranked list was not supplied\") {\n super(message);\n this.name = \"GraphLegUnavailable\";\n }\n}\n\nexport class ApprovalNotPending extends Error {\n constructor(message = \"approval is not pending\") {\n super(message);\n this.name = \"ApprovalNotPending\";\n }\n}\n\nexport class ApprovalExpired extends Error {\n constructor(message = \"approval has expired\") {\n super(message);\n this.name = \"ApprovalExpired\";\n }\n}\n\nexport class CodeExecutionError extends Error {\n constructor(message: string) {\n super(message);\n this.name = \"CodeExecutionError\";\n }\n}\n\nexport class CodeExecutionTimeout extends Error {\n constructor(message = \"code execution timed out\") {\n super(message);\n this.name = \"CodeExecutionTimeout\";\n }\n}\n\nexport class ExtraMissingError extends Error {\n constructor(extra: string, pkg: string, what: string) {\n super(`${what} requires the '${pkg}' package (optional extra: ${extra}): npm install ${pkg}`);\n this.name = \"ExtraMissingError\";\n }\n}\n","import { ExtraMissingError } from \"./errors.js\";\n\nexport async function tryImport<T = unknown>(specifier: string): Promise<T | null> {\n try {\n return (await import(specifier)) as T;\n } catch {\n return null;\n }\n}\n\nexport async function requireExtra<T = unknown>(specifier: string, extra: string, what: string): Promise<T> {\n try {\n return (await import(specifier)) as T;\n } catch (_err) {\n throw new ExtraMissingError(extra, specifier, what);\n }\n}\n","import { francAll } from \"franc\";\n\n// Unicode name first-word table — same logic as unicodedata.name().split()[0].\n// COMBINING marks use the second word, matching Python's _char_script.\n\nconst UNICODE_NAME_WORDS: string[] = [\n \"acute\",\n \"acute-grave-acute\",\n \"acute-macron\",\n \"adlam\",\n \"ahom\",\n \"alef\",\n \"almost\",\n \"anatolian\",\n \"angstrom\",\n \"annuity\",\n \"anticlockwise\",\n \"arabic\",\n \"armenian\",\n \"asterisk\",\n \"avestan\",\n \"balinese\",\n \"bamum\",\n \"bassa\",\n \"batak\",\n \"bengali\",\n \"bet\",\n \"bhaiksuki\",\n \"bindu\",\n \"black-letter\",\n \"bopomofo\",\n \"brahmi\",\n \"breve\",\n \"breve-macron\",\n \"bridge\",\n \"buginese\",\n \"buhid\",\n \"canadian\",\n \"candrabindu\",\n \"carian\",\n \"caron\",\n \"caucasian\",\n \"cedilla\",\n \"chakma\",\n \"cham\",\n \"cherokee\",\n \"chorasmian\",\n \"circumflex\",\n \"cjk\",\n \"clockwise\",\n \"comma\",\n \"conjoining\",\n \"coptic\",\n \"cuneiform\",\n \"cypriot\",\n \"cypro-minoan\",\n \"cyrillic\",\n \"dalet\",\n \"deletion\",\n \"deseret\",\n \"devanagari\",\n \"diaeresis\",\n \"diaeresis-ring\",\n \"dives\",\n \"dogra\",\n \"dot\",\n \"dotted\",\n \"double\",\n \"double-struck\",\n \"doubled\",\n \"down\",\n \"downwards\",\n \"duployan\",\n \"egyptian\",\n \"elbasan\",\n \"elymaic\",\n \"enclosing\",\n \"equals\",\n \"ethiopic\",\n \"euler\",\n \"feminine\",\n \"fermata\",\n \"four\",\n \"fullwidth\",\n \"georgian\",\n \"gimel\",\n \"glagolitic\",\n \"gothic\",\n \"grantha\",\n \"grapheme\",\n \"grave\",\n \"grave-acute-grave\",\n \"grave-macron\",\n \"greek\",\n \"gujarati\",\n \"gunjala\",\n \"gurmukhi\",\n \"halfwidth\",\n \"hangul\",\n \"hanifi\",\n \"hanunoo\",\n \"hatran\",\n \"hebrew\",\n \"hentaigana\",\n \"hiragana\",\n \"homothetic\",\n \"hook\",\n \"horn\",\n \"ideographic\",\n \"imperial\",\n \"infinity\",\n \"information\",\n \"inscriptional\",\n \"inverted\",\n \"is\",\n \"javanese\",\n \"kaithi\",\n \"kannada\",\n \"katakana\",\n \"katakana-hiragana\",\n \"kavyka\",\n \"kawi\",\n \"kayah\",\n \"kelvin\",\n \"kharoshthi\",\n \"khitan\",\n \"khmer\",\n \"khojki\",\n \"khudawadi\",\n \"lao\",\n \"latin\",\n \"left\",\n \"leftwards\",\n \"lepcha\",\n \"ligature\",\n \"light\",\n \"limbu\",\n \"linear\",\n \"lisu\",\n \"long\",\n \"low\",\n \"lycian\",\n \"lydian\",\n \"macron\",\n \"macron-acute\",\n \"macron-breve\",\n \"macron-grave\",\n \"mahajani\",\n \"makasar\",\n \"malayalam\",\n \"mandaic\",\n \"manichaean\",\n \"marchen\",\n \"masaram\",\n \"masculine\",\n \"masu\",\n \"mathematical\",\n \"medefaidrin\",\n \"meetei\",\n \"mende\",\n \"meroitic\",\n \"miao\",\n \"micro\",\n \"minus\",\n \"modi\",\n \"modifier\",\n \"mongolian\",\n \"mro\",\n \"multani\",\n \"musical\",\n \"myanmar\",\n \"nabataean\",\n \"nag\",\n \"nandinagari\",\n \"new\",\n \"newa\",\n \"nko\",\n \"not\",\n \"number\",\n \"nushu\",\n \"nyiakeng\",\n \"ogham\",\n \"ogonek\",\n \"ohm\",\n \"ol\",\n \"old\",\n \"open\",\n \"oriya\",\n \"osage\",\n \"osmanya\",\n \"overline\",\n \"pahawh\",\n \"palatalized\",\n \"palmyrene\",\n \"parentheses\",\n \"pau\",\n \"phags-pa\",\n \"phaistos\",\n \"phoenician\",\n \"planck\",\n \"plus\",\n \"psalter\",\n \"rejang\",\n \"retroflex\",\n \"reverse\",\n \"reversed\",\n \"right\",\n \"rightwards\",\n \"ring\",\n \"roman\",\n \"runic\",\n \"samaritan\",\n \"saurashtra\",\n \"script\",\n \"seagull\",\n \"sharada\",\n \"shavian\",\n \"short\",\n \"siddham\",\n \"signwriting\",\n \"sinhala\",\n \"snake\",\n \"sogdian\",\n \"sora\",\n \"soyombo\",\n \"square\",\n \"strong\",\n \"sundanese\",\n \"superscript\",\n \"suspension\",\n \"syloti\",\n \"syriac\",\n \"tagalog\",\n \"tagbanwa\",\n \"tai\",\n \"takri\",\n \"tamil\",\n \"tangsa\",\n \"tangut\",\n \"telugu\",\n \"thaana\",\n \"thai\",\n \"three\",\n \"tibetan\",\n \"tifinagh\",\n \"tilde\",\n \"tirhuta\",\n \"toto\",\n \"triple\",\n \"turned\",\n \"ugaritic\",\n \"up\",\n \"upwards\",\n \"ur\",\n \"us\",\n \"vai\",\n \"variation\",\n \"vedic\",\n \"vertical\",\n \"vietnamese\",\n \"vithkuqi\",\n \"wancho\",\n \"warang\",\n \"wide\",\n \"wiggly\",\n \"x\",\n \"x-x\",\n \"yezidi\",\n \"yi\",\n \"zanabazar\",\n \"zigzag\",\n \"znamenny\",\n];\n\nconst UNICODE_NAME_RANGES = `41,5a,124;61,7a,124;aa,aa,74;b5,b5,156;ba,ba,148;c0,d6,124;d8,f6,124;f8,2af,124;2b0,2c1,159;2c6,2c6,159;2c7,2c7,34;2c8,2d1,159;2e0,2e4,159;2ec,2ec,159;2ee,2ee,159;300,300,84;301,301,0;302,302,41;303,303,239;304,304,137;305,305,184;306,306,26;307,307,59;308,308,55;309,309,100;30a,30a,202;30b,30b,61;30c,30c,34;30d,30d,252;30e,30f,61;310,310,32;311,311,107;312,312,243;313,313,44;314,314,199;315,315,44;316,316,84;317,317,0;318,318,125;319,319,200;31a,31a,125;31b,31b,101;31c,31c,125;31d,31d,245;31e,31e,64;31f,31f,194;320,320,157;321,321,186;322,322,197;323,323,59;324,324,55;325,325,202;326,326,44;327,327,36;328,328,176;329,329,252;32a,32a,28;32b,32b,107;32c,32c,34;32d,32d,41;32e,32e,26;32f,32f,107;330,330,239;331,331,137;332,332,134;333,333,61;334,334,239;335,335,211;336,336,133;337,337,211;338,338,133;339,339,200;33a,33a,107;33b,33b,219;33c,33c,208;33d,33d,259;33e,33e,252;33f,33f,61;340,340,84;341,341,0;342,345,87;346,346,28;347,347,71;348,348,61;349,349,125;34a,34a,171;34b,34b,99;34c,34c,6;34d,34d,125;34e,34e,246;34f,34f,83;350,350,200;351,351,125;352,352,75;353,353,259;354,354,125;355,357,200;358,358,59;359,359,13;35a,35a,61;35b,35b,264;35c,362,61;363,36f,124;370,374,87;376,377,87;37a,37d,87;37f,37f,87;386,386,87;388,38a,87;38c,38c,87;38e,3a1,87;3a3,3e1,87;3e2,3ef,46;3f0,3f5,87;3f7,3ff,87;400,481,50;483,52f,50;531,556,12;559,559,12;560,588,12;591,5bd,96;5bf,5bf,96;5c1,5c2,96;5c4,5c5,96;5c7,5c7,96;5d0,5ea,96;5ef,5f2,96;610,61a,11;620,65f,11;66e,6d3,11;6d5,6dc,11;6df,6e8,11;6ea,6ef,11;6fa,6fc,11;6ff,6ff,11;710,74a,225;74d,74f,225;750,77f,11;780,7b1,234;7ca,7f5,170;7fa,7fa,170;7fd,7fd,170;800,82d,205;840,85b,144;860,86a,225;870,887,11;889,88e,11;898,8e1,11;8e3,8ff,11;900,963,54;971,97f,54;980,983,19;985,98c,19;98f,990,19;993,9a8,19;9aa,9b0,19;9b2,9b2,19;9b6,9b9,19;9bc,9c4,19;9c7,9c8,19;9cb,9ce,19;9d7,9d7,19;9dc,9dd,19;9df,9e3,19;9f0,9f1,19;9fc,9fc,19;9fe,9fe,19;a01,a03,90;a05,a0a,90;a0f,a10,90;a13,a28,90;a2a,a30,90;a32,a33,90;a35,a36,90;a38,a39,90;a3c,a3c,90;a3e,a42,90;a47,a48,90;a4b,a4d,90;a51,a51,90;a59,a5c,90;a5e,a5e,90;a70,a75,90;a81,a83,88;a85,a8d,88;a8f,a91,88;a93,aa8,88;aaa,ab0,88;ab2,ab3,88;ab5,ab9,88;abc,ac5,88;ac7,ac9,88;acb,acd,88;ad0,ad0,88;ae0,ae3,88;af9,aff,88;b01,b03,181;b05,b0c,181;b0f,b10,181;b13,b28,181;b2a,b30,181;b32,b33,181;b35,b39,181;b3c,b44,181;b47,b48,181;b4b,b4d,181;b55,b57,181;b5c,b5d,181;b5f,b63,181;b71,b71,181;b82,b83,230;b85,b8a,230;b8e,b90,230;b92,b95,230;b99,b9a,230;b9c,b9c,230;b9e,b9f,230;ba3,ba4,230;ba8,baa,230;bae,bb9,230;bbe,bc2,230;bc6,bc8,230;bca,bcd,230;bd0,bd0,230;bd7,bd7,230;c00,c0c,233;c0e,c10,233;c12,c28,233;c2a,c39,233;c3c,c44,233;c46,c48,233;c4a,c4d,233;c55,c56,233;c58,c5a,233;c5d,c5d,233;c60,c63,233;c80,c83,111;c85,c8c,111;c8e,c90,111;c92,ca8,111;caa,cb3,111;cb5,cb9,111;cbc,cc4,111;cc6,cc8,111;cca,ccd,111;cd5,cd6,111;cdd,cde,111;ce0,ce3,111;cf1,cf3,111;d00,d0c,143;d0e,d10,143;d12,d44,143;d46,d48,143;d4a,d4e,143;d54,d57,143;d5f,d63,143;d7a,d7f,143;d81,d83,214;d85,d96,214;d9a,db1,214;db3,dbb,214;dbd,dbd,214;dc0,dc6,214;dca,dca,214;dcf,dd4,214;dd6,dd6,214;dd8,ddf,214;df2,df3,214;e01,e3a,235;e40,e4e,235;e81,e82,123;e84,e84,123;e86,e8a,123;e8c,ea3,123;ea5,ea5,123;ea7,ebd,123;ec0,ec4,123;ec6,ec6,123;ec8,ece,123;edc,edf,123;f00,f00,237;f18,f19,237;f35,f35,237;f37,f37,237;f39,f39,237;f3e,f47,237;f49,f6c,237;f71,f84,237;f86,f97,237;f99,fbc,237;fc6,fc6,237;1000,103f,164;1050,108f,164;109a,109d,164;10a0,10c5,78;10c7,10c7,78;10cd,10cd,78;10d0,10fa,78;10fc,10fc,159;10fd,10ff,78;1100,11ff,92;1200,1248,72;124a,124d,72;1250,1256,72;1258,1258,72;125a,125d,72;1260,1288,72;128a,128d,72;1290,12b0,72;12b2,12b5,72;12b8,12be,72;12c0,12c0,72;12c2,12c5,72;12c8,12d6,72;12d8,1310,72;1312,1315,72;1318,135a,72;135d,135f,72;1380,138f,72;13a0,13f5,39;13f8,13fd,39;1401,166c,31;166f,167f,31;1681,169a,175;16a0,16ea,204;16f1,16f8,204;1700,1715,226;171f,171f,226;1720,1734,94;1740,1753,30;1760,176c,227;176e,1770,227;1772,1773,227;1780,17d3,120;17d7,17d7,120;17dc,17dd,120;180b,180d,160;180f,180f,160;1820,1878,160;1880,18aa,160;18b0,18f5,31;1900,191e,130;1920,192b,130;1930,193b,130;1950,196d,228;1970,1974,228;1980,19ab,168;19b0,19c9,168;1a00,1a1b,29;1a20,1a5e,228;1a60,1a7c,228;1a7f,1a7f,228;1aa7,1aa7,228;1ab0,1ab0,63;1ab1,1ab1,56;1ab2,1ab2,104;1ab3,1ab3,65;1ab4,1ab4,242;1ab5,1ab5,260;1ab6,1ab6,258;1ab7,1ab7,180;1ab8,1ab8,61;1ab9,1ab9,129;1aba,1aba,220;1abb,1abb,188;1abc,1abc,61;1abd,1abe,188;1abf,1ac0,124;1ac1,1ac1,125;1ac2,1ac2,200;1ac3,1ac3,125;1ac4,1ac4,200;1ac5,1ac5,219;1ac6,1ac6,172;1ac7,1ac7,107;1ac8,1ac8,194;1ac9,1aca,61;1acb,1acb,242;1acc,1ace,124;1b00,1b4c,15;1b6b,1b73,15;1b80,1baf,221;1bba,1bbf,221;1bc0,1bf3,18;1c00,1c37,127;1c4d,1c4f,127;1c5a,1c7d,178;1c80,1c88,50;1c90,1cba,78;1cbd,1cbf,78;1cd0,1cd2,251;1cd4,1cfa,251;1d00,1d25,124;1d26,1d2a,87;1d2b,1d2b,50;1d2c,1d61,159;1d62,1d65,124;1d66,1d6a,87;1d6b,1d77,124;1d78,1d78,159;1d79,1d9a,124;1d9b,1dbf,159;1dc0,1dc1,60;1dc2,1dc2,215;1dc3,1dc3,223;1dc4,1dc4,138;1dc5,1dc5,86;1dc6,1dc6,140;1dc7,1dc7,2;1dc8,1dc8,85;1dc9,1dc9,1;1dca,1dca,124;1dcb,1dcb,27;1dcc,1dcc,139;1dcd,1dcd,61;1dce,1dce,176;1dcf,1dcf,264;1dd0,1dd0,108;1dd1,1dd1,247;1dd2,1dd2,248;1dd3,1df4,124;1df5,1df5,245;1df6,1df7,114;1df8,1df8,59;1df9,1df9,257;1dfa,1dfa,59;1dfb,1dfb,52;1dfc,1dfc,61;1dfd,1dfd,6;1dfe,1dfe,125;1dff,1dff,200;1e00,1eff,124;1f00,1f15,87;1f18,1f1d,87;1f20,1f45,87;1f48,1f4d,87;1f50,1f57,87;1f59,1f59,87;1f5b,1f5b,87;1f5d,1f5d,87;1f5f,1f7d,87;1f80,1fb4,87;1fb6,1fbc,87;1fbe,1fbe,87;1fc2,1fc4,87;1fc6,1fcc,87;1fd0,1fd3,87;1fd6,1fdb,87;1fe0,1fec,87;1ff2,1ff4,87;1ff6,1ffc,87;2071,2071,222;207f,207f,222;2090,209c,124;20d0,20d0,125;20d1,20d1,200;20d2,20d2,133;20d3,20d3,211;20d4,20d4,10;20d5,20d5,43;20d6,20d6,125;20d7,20d7,200;20d8,20d8,202;20d9,20d9,43;20da,20da,10;20db,20db,236;20dc,20dc,76;20dd,20e0,70;20e1,20e1,125;20e2,20e4,70;20e5,20e5,198;20e6,20e6,61;20e7,20e7,9;20e8,20e8,242;20e9,20e9,257;20ea,20ea,126;20eb,20eb,133;20ec,20ec,201;20ed,20ed,126;20ee,20ee,125;20ef,20ef,200;20f0,20f0,13;2102,2102,62;2107,2107,73;210a,210b,207;210c,210c,23;210d,210d,62;210e,210f,193;2110,2110,207;2111,2111,23;2112,2113,207;2115,2115,62;2119,211a,62;211b,211b,207;211c,211c,23;211d,211d,62;2124,2124,62;2126,2126,177;2128,2128,23;212a,212a,117;212b,212b,8;212c,212c,207;212d,212d,23;212f,2131,207;2132,2132,243;2133,2134,207;2135,2135,5;2136,2136,20;2137,2137,79;2138,2138,51;2139,2139,105;213c,213f,62;2145,2149,62;214e,214e,243;2183,2183,203;2184,2184,124;2c00,2c5f,80;2c60,2c7c,124;2c7d,2c7d,159;2c7e,2c7f,124;2c80,2ce4,46;2ceb,2cf3,46;2d00,2d25,78;2d27,2d27,78;2d2d,2d2d,78;2d30,2d67,238;2d6f,2d6f,238;2d7f,2d7f,238;2d80,2d96,72;2da0,2da6,72;2da8,2dae,72;2db0,2db6,72;2db8,2dbe,72;2dc0,2dc6,72;2dc8,2dce,72;2dd0,2dd6,72;2dd8,2dde,72;2de0,2dff,50;2e2f,2e2f,252;3005,3006,102;302a,302d,102;302e,302f,92;3031,3035,252;303b,303b,252;303c,303c,149;3041,3096,98;3099,309a,113;309d,309f,98;30a1,30fa,112;30fc,30fc,113;30fd,30ff,112;3105,312f,24;3131,318e,92;31a0,31bf,24;31f0,31ff,112;3400,4dbf,42;4e00,9fff,42;a000,a48c,262;a4d0,a4fd,132;a500,a60c,249;a610,a61f,249;a62a,a62b,249;a640,a672,50;a674,a67d,50;a67f,a69b,50;a69c,a69d,159;a69e,a69f,50;a6a0,a6e5,16;a6f0,a6f1,16;a717,a71f,159;a722,a76f,124;a770,a770,159;a771,a787,124;a788,a788,159;a78b,a7ca,124;a7d0,a7d1,124;a7d3,a7d3,124;a7d5,a7d9,124;a7f2,a7f4,159;a7f5,a7f7,124;a7f8,a7f9,159;a7fa,a7ff,124;a800,a827,224;a82c,a82c,224;a840,a873,190;a880,a8c5,206;a8e0,a8f7,54;a8fb,a8fb,54;a8fd,a8ff,54;a90a,a92d,116;a930,a953,196;a960,a97c,92;a980,a9c0,109;a9cf,a9cf,109;a9e0,a9ef,164;a9fa,a9fe,164;aa00,aa36,38;aa40,aa4d,38;aa60,aa76,164;aa7a,aa7f,164;aa80,aac2,228;aadb,aadd,228;aae0,aaef,152;aaf2,aaf6,152;ab01,ab06,72;ab09,ab0e,72;ab11,ab16,72;ab20,ab26,72;ab28,ab2e,72;ab30,ab5a,124;ab5c,ab5f,159;ab60,ab64,124;ab65,ab65,87;ab66,ab68,124;ab69,ab69,159;ab70,abbf,39;abc0,abea,152;abec,abed,152;ac00,d7a3,92;d7b0,d7c6,92;d7cb,d7fb,92;f900,fa6d,42;fa70,fad9,42;fb00,fb06,124;fb13,fb17,12;fb1d,fb28,96;fb2a,fb36,96;fb38,fb3c,96;fb3e,fb3e,96;fb40,fb41,96;fb43,fb44,96;fb46,fb4f,96;fb50,fbb1,11;fbd3,fd3d,11;fd50,fd8f,11;fd92,fdc7,11;fdf0,fdfb,11;fe00,fe0f,250;fe20,fe21,128;fe22,fe23,61;fe24,fe25,137;fe26,fe26,45;fe27,fe28,128;fe29,fe2a,239;fe2b,fe2c,137;fe2d,fe2d,45;fe2e,fe2f,50;fe70,fe74,11;fe76,fefc,11;ff21,ff3a,77;ff41,ff5a,77;ff66,ffbe,91;ffc2,ffc7,91;ffca,ffcf,91;ffd2,ffd7,91;ffda,ffdc,91;10000,1000b,131;1000d,10026,131;10028,1003a,131;1003c,1003d,131;1003f,1004d,131;10050,1005d,131;10080,100fa,131;101fd,101fd,191;10280,1029c,135;102a0,102d0,33;102e0,102e0,46;10300,1031f,179;1032d,1032f,179;10330,10340,81;10342,10349,81;10350,1037a,179;10380,1039d,244;103a0,103c3,179;103c8,103cf,179;10400,1044f,53;10450,1047f,210;10480,1049d,183;104b0,104d3,182;104d8,104fb,182;10500,10527,68;10530,10563,35;10570,1057a,254;1057c,1058a,254;1058c,10592,254;10594,10595,254;10597,105a1,254;105a3,105b1,254;105b3,105b9,254;105bb,105bc,254;10600,10736,131;10740,10755,131;10760,10767,131;10780,10785,159;10787,107b0,159;107b2,107ba,159;10800,10805,48;10808,10808,48;1080a,10835,48;10837,10838,48;1083c,1083c,48;1083f,1083f,48;10840,10855,103;10860,10876,187;10880,1089e,165;108e0,108f2,95;108f4,108f5,95;10900,10915,192;10920,10939,136;10980,109b7,154;109be,109bf,154;10a00,10a03,118;10a05,10a06,118;10a0c,10a13,118;10a15,10a17,118;10a19,10a35,118;10a38,10a3a,118;10a3f,10a3f,118;10a60,10a7c,179;10a80,10a9c,179;10ac0,10ac7,145;10ac9,10ae6,145;10b00,10b35,14;10b40,10b55,106;10b60,10b72,106;10b80,10b91,195;10c00,10c48,179;10c80,10cb2,179;10cc0,10cf2,179;10d00,10d27,93;10e80,10ea9,261;10eab,10eac,261;10eb0,10eb1,261;10efd,10eff,11;10f00,10f1c,179;10f27,10f27,179;10f30,10f50,216;10f70,10f85,179;10fb0,10fc4,40;10fe0,10ff6,69;11000,11046,25;11070,11075,25;1107f,1107f,25;11080,110ba,110;110c2,110c2,110;110d0,110e8,217;11100,11134,37;11144,11147,37;11150,11173,141;11176,11176,141;11180,111c4,209;111c9,111cc,209;111ce,111cf,209;111da,111da,209;111dc,111dc,209;11200,11211,121;11213,11237,121;1123e,11241,121;11280,11286,162;11288,11288,162;1128a,1128d,162;1128f,1129d,162;1129f,112a8,162;112b0,112ea,122;11300,11303,82;11305,1130c,82;1130f,11310,82;11313,11328,82;1132a,11330,82;11332,11333,82;11335,11339,82;1133b,1133b,22;1133c,11344,82;11347,11348,82;1134b,1134d,82;11350,11350,82;11357,11357,82;1135d,11363,82;11366,1136c,82;11370,11374,82;11400,1144a,169;1145e,11461,169;11480,114c5,240;114c7,114c7,240;11580,115b5,212;115b8,115c0,212;115d8,115dd,212;11600,11640,158;11644,11644,158;11680,116b8,229;11700,1171a,4;1171d,1172b,4;11740,11746,4;11800,1183a,58;118a0,118df,256;118ff,118ff,256;11900,11906,57;11909,11909,57;1190c,11913,57;11915,11916,57;11918,11935,57;11937,11938,57;1193b,11943,57;119a0,119a7,167;119aa,119d7,167;119da,119e1,167;119e3,119e4,167;11a00,11a3e,263;11a47,11a47,263;11a50,11a99,218;11a9d,11a9d,218;11ab0,11abf,31;11ac0,11af8,189;11c00,11c08,21;11c0a,11c36,21;11c38,11c40,21;11c72,11c8f,146;11c92,11ca7,146;11ca9,11cb6,146;11d00,11d06,147;11d08,11d09,147;11d0b,11d36,147;11d3a,11d3a,147;11d3c,11d3d,147;11d3f,11d47,147;11d60,11d65,89;11d67,11d68,89;11d6a,11d8e,89;11d90,11d91,89;11d93,11d98,89;11ee0,11ef6,142;11f00,11f10,115;11f12,11f3a,115;11f3e,11f42,115;11fb0,11fb0,132;12000,12399,47;12480,12543,47;12f90,12ff0,49;13000,1342f,67;13440,13455,67;14400,14646,7;16800,16a38,16;16a40,16a5e,161;16a70,16abe,231;16ad0,16aed,17;16af0,16af4,17;16b00,16b36,185;16b40,16b43,185;16b63,16b77,185;16b7d,16b8f,185;16e40,16e7f,151;16f00,16f4a,155;16f4f,16f87,155;16f8f,16f9f,155;16fe0,16fe0,232;16fe1,16fe1,173;16fe3,16fe3,179;16fe4,16fe4,119;16ff0,16ff1,253;18800,18aff,232;18b00,18cd5,119;1aff0,1aff3,112;1aff5,1affb,112;1affd,1affe,112;1b000,1b000,112;1b001,1b001,98;1b002,1b11e,97;1b11f,1b11f,98;1b120,1b122,112;1b132,1b132,98;1b150,1b152,98;1b155,1b155,112;1b164,1b167,112;1b170,1b2fb,173;1bc00,1bc6a,66;1bc70,1bc7c,66;1bc80,1bc88,66;1bc90,1bc99,66;1bc9d,1bc9e,66;1cf00,1cf2d,265;1cf30,1cf46,265;1d165,1d169,163;1d16d,1d172,163;1d17b,1d182,163;1d185,1d18b,163;1d1aa,1d1ad,163;1d242,1d244,87;1d400,1d454,150;1d456,1d49c,150;1d49e,1d49f,150;1d4a2,1d4a2,150;1d4a5,1d4a6,150;1d4a9,1d4ac,150;1d4ae,1d4b9,150;1d4bb,1d4bb,150;1d4bd,1d4c3,150;1d4c5,1d505,150;1d507,1d50a,150;1d50d,1d514,150;1d516,1d51c,150;1d51e,1d539,150;1d53b,1d53e,150;1d540,1d544,150;1d546,1d546,150;1d54a,1d550,150;1d552,1d6a5,150;1d6a8,1d6c0,150;1d6c2,1d6da,150;1d6dc,1d6fa,150;1d6fc,1d714,150;1d716,1d734,150;1d736,1d74e,150;1d750,1d76e,150;1d770,1d788,150;1d78a,1d7a8,150;1d7aa,1d7c2,150;1d7c4,1d7cb,150;1da00,1da36,213;1da3b,1da6c,213;1da75,1da75,213;1da84,1da84,213;1da9b,1da9f,213;1daa1,1daaf,213;1df00,1df1e,124;1df25,1df2a,124;1e000,1e006,80;1e008,1e018,80;1e01b,1e021,80;1e023,1e024,80;1e026,1e02a,80;1e030,1e050,159;1e051,1e06a,50;1e06b,1e06d,159;1e08f,1e08f,50;1e100,1e12c,174;1e130,1e13d,174;1e14e,1e14e,174;1e290,1e2ae,241;1e2c0,1e2ef,255;1e4d0,1e4ef,166;1e7e0,1e7e6,72;1e7e8,1e7eb,72;1e7ed,1e7ee,72;1e7f0,1e7fe,72;1e800,1e8c4,153;1e8d0,1e8d6,153;1e900,1e94b,3;1ee00,1ee03,11;1ee05,1ee1f,11;1ee21,1ee22,11;1ee24,1ee24,11;1ee27,1ee27,11;1ee29,1ee32,11;1ee34,1ee37,11;1ee39,1ee39,11;1ee3b,1ee3b,11;1ee42,1ee42,11;1ee47,1ee47,11;1ee49,1ee49,11;1ee4b,1ee4b,11;1ee4d,1ee4f,11;1ee51,1ee52,11;1ee54,1ee54,11;1ee57,1ee57,11;1ee59,1ee59,11;1ee5b,1ee5b,11;1ee5d,1ee5d,11;1ee5f,1ee5f,11;1ee61,1ee62,11;1ee64,1ee64,11;1ee67,1ee6a,11;1ee6c,1ee72,11;1ee74,1ee77,11;1ee79,1ee7c,11;1ee7e,1ee7e,11;1ee80,1ee89,11;1ee8b,1ee9b,11;1eea1,1eea3,11;1eea5,1eea9,11;1eeab,1eebb,11;20000,2a6df,42;2a700,2b739,42;2b740,2b81d,42;2b820,2cea1,42;2ceb0,2ebe0,42;2ebf0,2ee5d,42;2f800,2fa1d,42;30000,3134a,42;31350,323af,42;e0100,e01ef,250`;\n\ntype NameRange = { start: number; end: number; word: number };\n\nconst NAME_RANGE_TABLE: NameRange[] = UNICODE_NAME_RANGES.split(\";\").map((entry) => {\n const [start, end, word] = entry.split(\",\");\n return { start: Number.parseInt(start!, 16), end: Number.parseInt(end!, 16), word: Number(word) };\n});\n\nconst SCRIPT_CACHE = new Map<string, string>();\nconst SCRIPT_CACHE_MAX = 4096;\n\nfunction lookupNamePrefix(cp: number): string {\n let lo = 0;\n let hi = NAME_RANGE_TABLE.length - 1;\n while (lo <= hi) {\n const mid = (lo + hi) >> 1;\n const range = NAME_RANGE_TABLE[mid]!;\n if (cp < range.start) hi = mid - 1;\n else if (cp > range.end) lo = mid + 1;\n else return UNICODE_NAME_WORDS[range.word] ?? \"\";\n }\n return \"\";\n}\n\nexport const SPACELESS_SCRIPTS = new Set([\n \"cjk\",\n \"hiragana\",\n \"katakana\",\n \"thai\",\n \"lao\",\n \"khmer\",\n \"myanmar\",\n \"tibetan\",\n \"yi\",\n]);\n\nexport const SENTENCE_TERMINALS = new Set([\n \".\",\n \"!\",\n \"?\",\n \"。\",\n \"!\",\n \"?\",\n \"؟\",\n \"।\",\n \"॥\",\n \"։\",\n \"።\",\n \"။\",\n \"៕\",\n \"᙮\",\n \"⁇\",\n \"⁈\",\n \"⁉\",\n \"꓿\",\n \"᜵\",\n \"᜶\",\n]);\n\n// Audit T1: drop U+003B (ASCII semicolon) — Greek's question mark IS that\n// codepoint, and U+037E NFC-normalizes to it, so including U+003B classified\n// every query with a plain semicolon as a question. Keep U+037E for explicitly\n// encoded Greek question marks.\nexport const QUESTION_MARKS = new Set([\"?\", \"؟\", \"?\", \"\\u037E\", \"⁇\", \"⁈\"]);\n\nexport const SCRIPT_TO_ISO: Record<string, string> = {\n thai: \"th\",\n bengali: \"bn\",\n tamil: \"ta\",\n telugu: \"te\",\n kannada: \"kn\",\n malayalam: \"ml\",\n gujarati: \"gu\",\n gurmukhi: \"pa\",\n sinhala: \"si\",\n lao: \"lo\",\n myanmar: \"my\",\n khmer: \"km\",\n georgian: \"ka\",\n armenian: \"hy\",\n ethiopic: \"am\",\n tibetan: \"bo\",\n greek: \"el\",\n hangul: \"ko\",\n thaana: \"dv\",\n devanagari: \"hi\",\n oriya: \"or\",\n};\n\nexport const AMBIGUOUS_SCRIPTS = new Set([\"latin\", \"cyrillic\", \"arabic\", \"cjk\"]);\n\nconst JA_SCRIPTS = new Set([\"hiragana\", \"katakana\", \"cjk\"]);\n\nconst AMBIGUOUS_TERMINALS = new Set([\".\", \"!\", \"?\"]);\n\nconst ISO_639_3_TO_1: Record<string, string> = {\n afr: \"af\",\n amh: \"am\",\n ara: \"ar\",\n arb: \"ar\",\n arz: \"ar\",\n apc: \"ar\",\n aze: \"az\",\n azj: \"az\",\n bel: \"be\",\n ben: \"bn\",\n bod: \"bo\",\n bos: \"bs\",\n bul: \"bg\",\n cat: \"ca\",\n ces: \"cs\",\n cmn: \"zh\",\n cym: \"cy\",\n dan: \"da\",\n deu: \"de\",\n ell: \"el\",\n eng: \"en\",\n epo: \"eo\",\n est: \"et\",\n ekk: \"et\",\n eus: \"eu\",\n fas: \"fa\",\n pes: \"fa\",\n prs: \"fa\",\n fin: \"fi\",\n fra: \"fr\",\n fry: \"fy\",\n gla: \"gd\",\n gle: \"ga\",\n glg: \"gl\",\n guj: \"gu\",\n hat: \"ht\",\n hau: \"ha\",\n heb: \"he\",\n hin: \"hi\",\n hrv: \"hr\",\n hun: \"hu\",\n hye: \"hy\",\n ibo: \"ig\",\n ind: \"id\",\n isl: \"is\",\n ita: \"it\",\n jav: \"jv\",\n jpn: \"ja\",\n kan: \"kn\",\n kat: \"ka\",\n kaz: \"kk\",\n khm: \"km\",\n khk: \"mn\",\n kin: \"rw\",\n kir: \"ky\",\n kor: \"ko\",\n kur: \"ku\",\n kmr: \"ku\",\n lao: \"lo\",\n lat: \"la\",\n lav: \"lv\",\n lvs: \"lv\",\n lit: \"lt\",\n ltz: \"lb\",\n mal: \"ml\",\n mar: \"mr\",\n mkd: \"mk\",\n mlg: \"mg\",\n plt: \"mg\",\n mlt: \"mt\",\n mon: \"mn\",\n msa: \"ms\",\n zlm: \"ms\",\n mya: \"my\",\n nep: \"ne\",\n nld: \"nl\",\n nno: \"nn\",\n nob: \"nb\",\n nor: \"no\",\n nya: \"ny\",\n ori: \"or\",\n pan: \"pa\",\n pol: \"pl\",\n por: \"pt\",\n pus: \"ps\",\n pbu: \"ps\",\n que: \"qu\",\n qug: \"qu\",\n ron: \"ro\",\n rus: \"ru\",\n sin: \"si\",\n slk: \"sk\",\n slv: \"sl\",\n sna: \"sn\",\n som: \"so\",\n spa: \"es\",\n sqi: \"sq\",\n srp: \"sr\",\n sun: \"su\",\n swa: \"sw\",\n swh: \"sw\",\n swe: \"sv\",\n tam: \"ta\",\n tel: \"te\",\n tgk: \"tg\",\n tgl: \"tl\",\n tha: \"th\",\n tir: \"ti\",\n tuk: \"tk\",\n tur: \"tr\",\n uig: \"ug\",\n ukr: \"uk\",\n urd: \"ur\",\n uzb: \"uz\",\n uzn: \"uz\",\n vie: \"vi\",\n yid: \"yi\",\n yor: \"yo\",\n zho: \"zh\",\n zul: \"zu\",\n};\n\nfunction francToIso6391(code: string): string | null {\n if (!code || code === \"und\") return null;\n if (code.length === 2) return code.toLowerCase();\n return ISO_639_3_TO_1[code] ?? null;\n}\n\n/**\n * Extract script from Unicode character names — same logic as\n * unicodedata.name(ch).split()[0].lower(), with COMBINING → second word.\n * Letters (L) and marks (M) only; everything else returns \"\".\n */\nexport function charScript(ch: string): string {\n const cached = SCRIPT_CACHE.get(ch);\n if (cached !== undefined) return cached;\n let result = \"\";\n if (ch) {\n const cp = ch.codePointAt(0);\n if (cp !== undefined) {\n const letter = String.fromCodePoint(cp);\n if (/^\\p{L}$/u.test(letter) || /^\\p{M}$/u.test(letter)) {\n result = lookupNamePrefix(cp);\n }\n }\n }\n if (SCRIPT_CACHE.size >= SCRIPT_CACHE_MAX) {\n const first = SCRIPT_CACHE.keys().next().value;\n if (first !== undefined) SCRIPT_CACHE.delete(first);\n }\n SCRIPT_CACHE.set(ch, result);\n return result;\n}\n\nfunction mostCommonScripts(scripts: Map<string, number>, n: number): Array<[string, number]> {\n return [...scripts.entries()].sort((a, b) => b[1] - a[1]).slice(0, n);\n}\n\nfunction sliceCodePoints(text: string, max: number): string {\n if (text.length <= max) return text;\n let n = 0;\n let end = 0;\n for (const ch of text) {\n if (n >= max) break;\n end += ch.length;\n n += 1;\n }\n return text.slice(0, end);\n}\n\nfunction scriptOnly(text: string, targetScript: string): string {\n let out = \"\";\n for (const ch of text) {\n if (/^\\s$/u.test(ch)) out += \" \";\n else if (charScript(ch) === targetScript) out += ch;\n }\n return out;\n}\n\nfunction statisticalDetect(sample: string, dominantScript: string, scripts: Map<string, number>): string {\n if (dominantScript === \"hiragana\" || dominantScript === \"katakana\") return \"ja\";\n if (dominantScript === \"cjk\") {\n const totalChars = [...scripts.values()].reduce((a, b) => a + b, 0) || 1;\n const kana = (scripts.get(\"hiragana\") ?? 0) + (scripts.get(\"katakana\") ?? 0);\n if (kana / totalChars >= 0.02) return \"ja\";\n return \"zh\";\n }\n\n try {\n const clean = scriptOnly(sample, dominantScript);\n if (clean.trim().length < 20) return dominantScript;\n const ranked = francAll(clean, { minLength: 10 });\n // franc ranks 639-3 codes; skip ones with no 639-1 mapping (e.g. `sco`\n // outranking `eng` on short English) just as Lingua falls back when\n // iso_code_639_1 is missing — then take the next confident 639-1 hit.\n for (const [code, conf] of ranked) {\n if (conf < 0.5) break;\n const iso = francToIso6391(code);\n if (iso) return iso;\n }\n } catch {\n // franc failed — fall through to script name\n }\n return dominantScript;\n}\n\nexport function detectLanguage(txt: unknown, _minChars = 30): string | null {\n if (!txt) return null;\n let text: string;\n if (typeof txt === \"string\") text = txt;\n else {\n try {\n text = String(txt);\n } catch {\n return null;\n }\n }\n text = text.trim();\n if (!text) return null;\n\n const sample = text.length > 1500 ? sliceCodePoints(text, 1500) : text;\n\n const scripts = new Map<string, number>();\n for (const ch of sample) {\n const s = charScript(ch);\n if (s) scripts.set(s, (scripts.get(s) ?? 0) + 1);\n }\n\n const total = [...scripts.values()].reduce((a, b) => a + b, 0);\n if (total < 20) return null;\n\n const topScripts = mostCommonScripts(scripts, 5).map(\n ([script, count]) => [script, count / total] as [string, number],\n );\n if (!topScripts.length) return null;\n\n const [dominantScript, dominantRatio] = topScripts[0]!;\n const secondRatio = topScripts[1]?.[1] ?? 0.0;\n\n let jaCount = 0;\n for (const [s, c] of scripts) {\n if (JA_SCRIPTS.has(s)) jaCount += c;\n }\n const jaRatio = jaCount / total;\n const kanaRatio = ((scripts.get(\"hiragana\") ?? 0) + (scripts.get(\"katakana\") ?? 0)) / total;\n if (jaRatio >= 0.8 && kanaRatio >= 0.02) return \"ja\";\n\n if (dominantRatio >= 0.2 && secondRatio >= 0.2) return \"mixed\";\n\n if (dominantRatio >= 0.6) {\n const lang = SCRIPT_TO_ISO[dominantScript];\n if (lang) return lang;\n return statisticalDetect(sample, dominantScript, scripts);\n }\n\n return null;\n}\n\nexport function isSpacelessChar(ch: string): boolean {\n return SPACELESS_SCRIPTS.has(charScript(ch));\n}\n\nfunction isCombiningMark(ch: string): boolean {\n return /^\\p{M}$/u.test(ch);\n}\n\nexport function tokenize(text: string): string[] {\n const tokens: string[] = [];\n const currentWord: string[] = [];\n\n for (const ch of text) {\n if (/^\\s$/u.test(ch)) {\n if (currentWord.length) {\n tokens.push(currentWord.join(\"\"));\n currentWord.length = 0;\n }\n continue;\n }\n\n if (isCombiningMark(ch)) {\n if (tokens.length && !currentWord.length) {\n tokens[tokens.length - 1] = tokens[tokens.length - 1]! + ch;\n } else {\n currentWord.push(ch);\n }\n continue;\n }\n\n if (isSpacelessChar(ch)) {\n if (currentWord.length) {\n tokens.push(currentWord.join(\"\"));\n currentWord.length = 0;\n }\n tokens.push(ch);\n } else {\n currentWord.push(ch);\n }\n }\n\n if (currentWord.length) tokens.push(currentWord.join(\"\"));\n return tokens;\n}\n\nexport function tokenCount(text: string): number {\n return tokenize(text || \"\").length;\n}\n\nfunction isSentenceTerminal(ch: string): boolean {\n return SENTENCE_TERMINALS.has(ch);\n}\n\nfunction lstrip(chars: string[]): string[] {\n let i = 0;\n while (i < chars.length && /^\\s$/u.test(chars[i]!)) i += 1;\n return chars.slice(i);\n}\n\nexport function splitSentences(text: string): string[] {\n text = (text || \"\").trim();\n if (!text) return [];\n\n const chars = [...text];\n const parts: string[] = [];\n const current: string[] = [];\n\n for (let i = 0; i < chars.length; i++) {\n const ch = chars[i]!;\n current.push(ch);\n\n if (isSentenceTerminal(ch)) {\n const rest = chars.slice(i + 1);\n const stripped = lstrip(rest);\n\n if (!stripped.length) break;\n\n if (!AMBIGUOUS_TERMINALS.has(ch)) {\n parts.push(current.join(\"\").trim());\n current.length = 0;\n continue;\n }\n\n if (rest.length && /^\\s$/u.test(rest[0]!)) {\n parts.push(current.join(\"\").trim());\n current.length = 0;\n } else if (stripped.length) {\n const currentScript = charScript(ch) || \"\";\n const nextScript = charScript(stripped[0]!) || \"\";\n if (currentScript && nextScript && currentScript !== nextScript) {\n parts.push(current.join(\"\").trim());\n current.length = 0;\n }\n }\n }\n }\n\n if (current.length) {\n const last = current.join(\"\").trim();\n if (last) parts.push(last);\n }\n\n if (parts.length <= 2) return [text];\n return parts;\n}\n\nexport function jaccardSim(a: string, b: string): number {\n const A = new Set(tokenize(a));\n const B = new Set(tokenize(b));\n if (!A.size || !B.size) return 0.0;\n let inter = 0;\n for (const t of A) if (B.has(t)) inter += 1;\n return inter / (A.size + B.size - inter);\n}\n\nexport function isQuestionMark(ch: string): boolean {\n return QUESTION_MARKS.has(ch);\n}\n","import type { UsageEvent } from \"./usage.js\";\n\nconst log = {\n warn: (...args: unknown[]) => console.warn(\"[context-engine]\", ...args),\n error: (...args: unknown[]) => console.error(\"[context-engine]\", ...args),\n};\n\n/**\n * One ingest stage boundary. `documentId` is null until the row is claimed\n * (extraction runs before that, so dedup can decide from the extracted\n * text); `sourceId` + `externalId` are the caller's own identity and exist\n * on every event. `detail` is per stage and always JSON-serialisable.\n */\nexport interface ProgressEvent {\n sourceId: string | null;\n externalId: string | null;\n documentId: string | null;\n name: string;\n stage: \"extract\" | \"redact\" | \"chunk\" | \"embed\" | \"structured\" | \"graph\";\n state: \"started\" | \"done\";\n detail: Record<string, unknown>;\n}\n\nexport interface Hooks {\n onUsage?: ((event: UsageEvent) => void) | null;\n onError?: ((exc: unknown, ctx: Record<string, unknown>) => void) | null;\n onToolCall?: ((event: Record<string, unknown>) => void) | null;\n /** Ingest stage boundaries, for live progress UIs. Never raises outward. */\n onProgress?: ((event: ProgressEvent) => void) | null;\n}\n\nexport function emitUsage(hooks: Hooks | null | undefined, event: UsageEvent): void {\n if (!hooks?.onUsage) return;\n try {\n hooks.onUsage(event);\n } catch {\n log.warn(\"onUsage callback raised; swallowing\");\n }\n}\n\nexport function emitError(hooks: Hooks | null | undefined, exc: unknown, ctx: Record<string, unknown>): void {\n log.error(\"context_engine error:\", exc, \"| ctx=\", ctx);\n if (!hooks?.onError) return;\n try {\n hooks.onError(exc, ctx);\n } catch {\n log.warn(\"onError callback raised; swallowing\");\n }\n}\n\n/** Report an ingest stage boundary to `hooks.onProgress`, if set. Never raises. */\nexport function emitProgress(hooks: Hooks | null | undefined, event: ProgressEvent): void {\n if (!hooks?.onProgress) return;\n try {\n hooks.onProgress(event);\n } catch {\n log.warn(\"onProgress callback raised; swallowing\");\n }\n}\n\nexport function emitToolCall(hooks: Hooks | null | undefined, event: Record<string, unknown>): void {\n if (!hooks?.onToolCall) return;\n try {\n hooks.onToolCall(event);\n } catch {\n log.warn(\"onToolCall callback raised; swallowing\");\n }\n}\n","/**\n * How a `@google/genai` client gets built — the one place that decides.\n *\n * Gemini reaches this package two ways, and they differ only in the client\n * constructor:\n *\n * - **gemini** — the Gemini Developer API, authenticated with an API key.\n * - **vertex_ai** — the same models on Vertex AI, authenticated with\n * Application Default Credentials against a GCP project.\n * `GOOGLE_APPLICATION_CREDENTIALS`, workload identity and\n * `gcloud auth application-default login` all work unchanged; there is\n * deliberately no credentials field of our own.\n *\n * Both the LLM provider and the embeddings provider need this decision, so it\n * lives here rather than in four copies (two per port) that would drift.\n *\n * Two SDK behaviours this module is built around, both read out of the real\n * package rather than assumed:\n *\n * - `project`/`location` and `apiKey` are **mutually exclusive** — the SDK\n * raises if handed both. A `vertex_ai` config that still carries a leftover\n * `apiKey` alongside a project must therefore not forward it, or every call\n * fails on a config that looks entirely reasonable.\n * - Omitting project/location is not automatically a broken config: the SDK\n * reads `GOOGLE_CLOUD_PROJECT` / `GOOGLE_CLOUD_LOCATION`, which is how a GCP\n * deployment is usually already wired. So unset fields are left ABSENT\n * rather than passed as null, and the \"is there enough auth\" question is\n * left to the SDK, the only party that can answer it.\n *\n * **This is where the two ports genuinely differ**, measured rather than\n * assumed: the Python SDK falls back to discovering the project from ADC\n * when neither config nor env supplies one, and this JS SDK does not — it\n * throws at construction. A deployment relying on ADC alone must therefore\n * name the project in config or in the environment for the TS client.\n */\nimport { ExtraMissingError } from \"../errors.js\";\n\n/** The fields `EmbeddingConfig` and `LLMConfig` share here. */\nexport interface GoogleProviderConfig {\n provider: string;\n apiKey?: string | null;\n project?: string | null;\n location?: string | null;\n}\n\ntype GenaiCtorOpts = {\n apiKey?: string | null;\n vertexai?: boolean;\n project?: string;\n location?: string;\n httpOptions?: { timeout?: number };\n};\n\n/**\n * Build a `@google/genai` client for `cfg`, in Gemini or Vertex mode.\n *\n * `timeoutMs` is required: the SDK has NO default timeout of its own, so a\n * missing one means a stuck backend hangs the call forever — on the ingest\n * path a wedged worker rather than a failed document.\n *\n * `purpose` only shapes the `ExtraMissingError` message (\"gemini llm\" vs\n * \"gemini embeddings\").\n */\nexport async function buildGenaiClient<T>(\n cfg: GoogleProviderConfig,\n timeoutMs: number,\n purpose: string,\n): Promise<T> {\n const specifier = \"@google/genai\";\n let mod: {\n GoogleGenAI?: new (opts: GenaiCtorOpts) => T;\n Client?: new (opts: GenaiCtorOpts) => T;\n };\n try {\n mod = (await import(specifier)) as typeof mod;\n } catch {\n throw new ExtraMissingError(\"gemini\", specifier, purpose);\n }\n const Ctor = mod.GoogleGenAI ?? mod.Client;\n if (!Ctor) {\n throw new ExtraMissingError(\"gemini\", specifier, purpose);\n }\n\n const opts: GenaiCtorOpts = { httpOptions: { timeout: timeoutMs } };\n\n if (cfg.provider === \"vertex_ai\") {\n opts.vertexai = true;\n if (cfg.project || cfg.location) {\n // Explicit project/location wins, and the API key is dropped: the SDK\n // refuses the combination outright.\n if (cfg.project) opts.project = cfg.project;\n if (cfg.location) opts.location = cfg.location;\n } else if (cfg.apiKey) {\n // Vertex express mode — an API key and no project. The one combination\n // where `apiKey` means anything on this provider.\n opts.apiKey = cfg.apiKey;\n }\n // Otherwise set neither, so the SDK's own env-var lookup runs untouched.\n } else {\n opts.apiKey = cfg.apiKey ?? null;\n }\n\n try {\n return new Ctor(opts);\n } catch (err) {\n const message = err instanceof Error ? err.message : String(err);\n if (!message.includes(\"Authentication is not set up\")) throw err;\n // The SDK is right, but it cannot know what this config is called. Note\n // the port difference this message has to cover: the Python SDK falls\n // back to discovering the project from ADC, this one does not — it reads\n // the env vars and stops — so an ADC-only deployment must name the\n // project somewhere.\n throw new Error(\n `${message} Set \\`project\\` on the provider config, or export ` +\n \"GOOGLE_CLOUD_PROJECT and GOOGLE_CLOUD_LOCATION. Credentials \" +\n \"themselves come from Application Default Credentials.\",\n );\n }\n}\n","/**\n * Pluggable LLM provider.\n *\n * `callLlm(cfg, { system, user, jsonMode, images, client })` is the single entry\n * point. Returns `[text, { input, output }]`.\n *\n * Images are always PNG. 240s call timeout (every provider), no retries.\n * Injectable `client` / `fetch`.\n * `callLlm` owns and closes the client it builds unless one is passed in.\n */\n\nimport OpenAI from \"openai\";\nimport type { LLMConfig } from \"../config.js\";\nimport { ExtraMissingError } from \"../errors.js\";\nimport type { FetchImpl } from \"./embeddings.js\";\nimport { buildGenaiClient } from \"./google.js\";\n\n/** Hard cap per LLM call, EVERY provider. A non-streaming multi-page vision\n * transcription routinely generates for well over 30s (real batches run\n * minutes), so the old 30s default made every vision attempt on\n * anthropic/openai/bedrock time out and the pages land as `failed` — only\n * gemini had been given this cap. It also bounds the hang: `@google/genai`\n * has NO default timeout at all, and with the bounded render queue a hung\n * batch holds a worker slot for good. */\nexport const CALL_TIMEOUT_MS = 240_000;\n\n/** Back-compat aliases — the cap stopped being gemini-specific. */\nexport const TIMEOUT_MS = CALL_TIMEOUT_MS;\nexport const GEMINI_CALL_TIMEOUT_MS = CALL_TIMEOUT_MS;\nconst ANTHROPIC_VERSION = \"2023-06-01\";\nconst ANTHROPIC_MAX_TOKENS = 4096;\n\nconst OPENAI_FAMILY = new Set([\"openai\", \"azure_openai\", \"custom\"]);\n\n/**\n * Both reach the same models through the same SDK; only the client\n * constructor differs (API key vs ADC against a GCP project), which is why\n * they share every line below `buildGenaiClient`.\n */\nconst GOOGLE_FAMILY = new Set([\"gemini\", \"vertex_ai\"]);\n\nexport type TokenUsage = { input: number; output: number };\n\nexport type ImageBytes = Uint8Array;\n\nexport type ChatClient = {\n chat: {\n completions: {\n create(body: { model: string; messages: unknown[]; response_format?: { type: string } }): Promise<{\n choices: Array<{ message?: { content?: string | null } }>;\n usage?: { prompt_tokens?: number; completion_tokens?: number } | null;\n }>;\n };\n };\n close?: () => void | Promise<void>;\n timeout?: number;\n baseURL?: string;\n};\n\nexport type LlmCallOpts = {\n system: string;\n user: string;\n jsonMode?: boolean;\n images?: ImageBytes[] | null;\n maxTokens?: number | null;\n /** Reasoning tokens are drawn from `maxTokens`, so a caller that wants\n * transcription rather than deliberation must be able to spend the whole\n * budget on output. Omitted = the provider's default. Applied per provider\n * capability: gemini and anthropic take it directly (anthropic thinking is\n * opt-in, so 0 sends nothing and a positive budget drops `temperature` —\n * the API rejects the combination); bedrock forwards a positive budget as\n * the anthropic-style passthrough; the openai family has no portable\n * thinking control. */\n thinkingBudget?: number | null;\n temperature?: number | null;\n};\n\nexport type LLMClientOpts = {\n client?: ChatClient | null;\n fetch?: FetchImpl | null;\n fetchImpl?: FetchImpl | null;\n};\n\ntype GeminiChatClient = {\n models: {\n generateContent(args: { model: string; contents: unknown; config?: Record<string, unknown> }): Promise<{\n text?: string;\n usageMetadata?: {\n promptTokenCount?: number;\n candidatesTokenCount?: number;\n prompt_token_count?: number;\n candidates_token_count?: number;\n };\n }>;\n };\n};\n\ntype BedrockConverseResult = {\n output?: { message?: { content?: Array<{ text?: string }> } };\n usage?: { inputTokens?: number; outputTokens?: number };\n};\n\nfunction toBase64(bytes: Uint8Array): string {\n return Buffer.from(bytes).toString(\"base64\");\n}\n\nasync function postJson(\n fetchImpl: FetchImpl,\n url: string,\n body: unknown,\n headers: Record<string, string>,\n): Promise<any> {\n const resp = await fetchImpl(url, {\n method: \"POST\",\n headers: { \"content-type\": \"application/json\", ...headers },\n body: JSON.stringify(body),\n signal: AbortSignal.timeout(TIMEOUT_MS),\n });\n if (!resp.ok) {\n const text = await resp.text().catch(() => \"\");\n throw new Error(`HTTP ${resp.status} ${resp.statusText}${text ? `: ${text}` : \"\"}`);\n }\n return resp.json();\n}\n\nexport function buildOpenAIChatClient(cfg: LLMConfig): OpenAI {\n return new OpenAI({\n apiKey: cfg.apiKey ?? undefined,\n baseURL: cfg.baseUrl ?? undefined,\n timeout: TIMEOUT_MS,\n maxRetries: 0,\n });\n}\n\nexport class LLMClient {\n cfg: LLMConfig;\n provider: LLMConfig[\"provider\"];\n model: string;\n client: ChatClient | null;\n fetchImpl: FetchImpl | null;\n private genaiClient: GeminiChatClient | null = null;\n\n constructor(cfg: LLMConfig, opts: LLMClientOpts = {}) {\n this.cfg = cfg;\n this.provider = cfg.provider;\n this.model = cfg.model;\n this.client = opts.client ?? null;\n this.fetchImpl = opts.fetch ?? opts.fetchImpl ?? null;\n }\n\n async aclose(): Promise<void> {\n const closer = this.client?.close;\n if (typeof closer === \"function\") {\n await closer.call(this.client);\n }\n }\n\n async [Symbol.asyncDispose](): Promise<void> {\n await this.aclose();\n }\n\n async call(opts: LlmCallOpts): Promise<[string, TokenUsage]> {\n const {\n system,\n user,\n jsonMode = false,\n images = null,\n maxTokens = null,\n thinkingBudget = null,\n temperature = null,\n } = opts;\n if (this.provider === \"anthropic\") {\n return this.callAnthropic(system, user, jsonMode, images, maxTokens, thinkingBudget, temperature);\n }\n if (OPENAI_FAMILY.has(this.provider)) {\n return this.callOpenAI(system, user, jsonMode, images, maxTokens, temperature);\n }\n if (GOOGLE_FAMILY.has(this.provider)) {\n return this.callGemini(system, user, jsonMode, images, maxTokens, thinkingBudget, temperature);\n }\n if (this.provider === \"bedrock\") {\n return this.callBedrock(system, user, jsonMode, images, maxTokens, thinkingBudget, temperature);\n }\n throw new Error(`unknown llm provider: ${JSON.stringify(this.provider)}`);\n }\n\n private async callAnthropic(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n thinkingBudget: number | null = null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n if (!this.fetchImpl) {\n throw new Error(\"anthropic llm client has no fetch implementation\");\n }\n if (jsonMode) {\n system = `${system}\\n\\nRespond with valid JSON only.`;\n }\n const content: Array<Record<string, unknown>> = [];\n for (const image of images ?? []) {\n content.push({\n type: \"image\",\n source: {\n type: \"base64\",\n media_type: \"image/png\",\n data: toBase64(image),\n },\n });\n }\n content.push({ type: \"text\", text: user });\n\n const body: Record<string, unknown> = {\n model: this.model,\n max_tokens: maxTokens ?? ANTHROPIC_MAX_TOKENS,\n system,\n messages: [{ role: \"user\", content }],\n };\n if (thinkingBudget) {\n // Anthropic thinking is OPT-IN, so `thinkingBudget: 0` (the vision\n // transcription contract) correctly maps to sending nothing. The API\n // rejects a temperature alongside enabled thinking, so the budget wins\n // when a caller passes both.\n body.thinking = { type: \"enabled\", budget_tokens: thinkingBudget };\n } else if (temperature !== null) {\n // Determinism contract — the same scan must transcribe to the same\n // text twice. This was silently dropped here, so the guarantee only\n // held on gemini.\n body.temperature = temperature;\n }\n\n const data = await postJson(this.fetchImpl, \"https://api.anthropic.com/v1/messages\", body, {\n \"x-api-key\": this.cfg.apiKey ?? \"\",\n \"anthropic-version\": ANTHROPIC_VERSION,\n \"content-type\": \"application/json\",\n });\n const text = data.content[0].text as string;\n const usage = data.usage ?? {};\n return [text, { input: usage.input_tokens ?? 0, output: usage.output_tokens ?? 0 }];\n }\n\n private async callOpenAI(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n if (!this.client) {\n throw new Error(\"openai-family llm client has no client\");\n }\n const content: Array<Record<string, unknown>> = [{ type: \"text\", text: user }];\n for (const image of images ?? []) {\n content.push({\n type: \"image_url\",\n image_url: { url: `data:image/png;base64,${toBase64(image)}` },\n });\n }\n const body: {\n model: string;\n messages: unknown[];\n response_format?: { type: string };\n max_completion_tokens?: number;\n temperature?: number;\n } = {\n model: this.model,\n messages: [\n { role: \"system\", content: system },\n { role: \"user\", content },\n ],\n };\n if (jsonMode) {\n body.response_format = { type: \"json_object\" };\n }\n if (maxTokens) {\n // `max_completion_tokens`, not the deprecated `max_tokens`: the\n // reasoning models reject the old name outright.\n body.max_completion_tokens = maxTokens;\n }\n if (temperature !== null) {\n // NOTE: the o-series reasoning models reject a temperature — but a\n // clear API error beats the silent nondeterminism of dropping it.\n // (`thinkingBudget` has no portable mapping here — see LlmCallOpts.)\n body.temperature = temperature;\n }\n const resp = await this.client.chat.completions.create(body);\n const text = resp.choices[0]?.message?.content ?? \"\";\n const usage = resp.usage;\n return [text, { input: usage?.prompt_tokens ?? 0, output: usage?.completion_tokens ?? 0 }];\n }\n\n private async callGemini(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n thinkingBudget: number | null = null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n if (!this.genaiClient) {\n // `buildGenaiClient` picks API-key or Vertex/ADC mode and carries the\n // timeout: the SDK has NO default one, so a dropped connection hangs\n // the call forever — and with the bounded render queue that means one\n // hung batch holds a worker slot for good, so the document simply\n // stops with nothing to show for it.\n this.genaiClient = await buildGenaiClient<GeminiChatClient>(\n this.cfg,\n GEMINI_CALL_TIMEOUT_MS,\n \"gemini llm\",\n );\n }\n const parts: unknown[] = [];\n for (const img of images ?? []) {\n parts.push({ inlineData: { mimeType: \"image/png\", data: toBase64(img) } });\n }\n parts.push({ text: user });\n\n const config: Record<string, unknown> = {\n systemInstruction: system,\n // ALWAYS off, and not as a preference. Automatic function calling means\n // the SDK itself EXECUTES a callable it was handed as a tool and loops\n // on the result — up to ten round trips — before returning anything.\n // This package passes declarations only, so today nothing is executable;\n // but that depends on every future caller continuing to do the same, and\n // an application that gates tool execution behind human approval would\n // have that gate bypassed silently, by a library, with the loop already\n // run before it could object.\n automaticFunctionCalling: { disable: true },\n };\n if (jsonMode) {\n config.responseMimeType = \"application/json\";\n }\n if (maxTokens) {\n config.maxOutputTokens = maxTokens;\n }\n if (temperature !== null) {\n config.temperature = temperature;\n }\n if (thinkingBudget !== null) {\n // Only when the CALLER asks. Reasoning tokens come out of\n // `maxOutputTokens`, so a long transcription can otherwise spend its\n // allowance deliberating and truncate part-way through. Entity\n // extraction and structured output may legitimately want the reasoning,\n // so there is no silent default.\n config.thinkingConfig = { thinkingBudget };\n }\n\n const resp = await this.genaiClient.models.generateContent({\n model: this.model,\n contents: parts,\n config,\n });\n const text = resp.text ?? \"\";\n const usageMeta = resp.usageMetadata;\n const inputTokens = usageMeta?.promptTokenCount ?? usageMeta?.prompt_token_count ?? 0;\n const outputTokens = usageMeta?.candidatesTokenCount ?? usageMeta?.candidates_token_count ?? 0;\n return [text, { input: inputTokens || 0, output: outputTokens || 0 }];\n }\n\n private async callBedrock(\n system: string,\n user: string,\n jsonMode: boolean,\n images: ImageBytes[] | null,\n maxTokens: number | null,\n thinkingBudget: number | null = null,\n temperature: number | null = null,\n ): Promise<[string, TokenUsage]> {\n const { BedrockRuntimeClient, ConverseCommand } = await loadBedrockSdk();\n if (jsonMode) {\n system = `${system}\\n\\nRespond with valid JSON only.`;\n }\n const content: Array<Record<string, unknown>> = [];\n for (const image of images ?? []) {\n content.push({ image: { format: \"png\", source: { bytes: image } } });\n }\n content.push({ text: user });\n\n const runtime = new BedrockRuntimeClient({\n maxAttempts: 1,\n requestHandler: {\n requestTimeout: TIMEOUT_MS,\n connectionTimeout: TIMEOUT_MS,\n },\n });\n try {\n const result = (await runtime.send(\n new ConverseCommand({\n modelId: this.model,\n system: [{ text: system }],\n messages: [{ role: \"user\", content }],\n ...(maxTokens || temperature !== null\n ? {\n inferenceConfig: {\n ...(maxTokens ? { maxTokens } : {}),\n ...(temperature !== null ? { temperature } : {}),\n },\n }\n : {}),\n // Anthropic-style passthrough — Converse forwards it to the model.\n // Zero (the vision transcription contract) sends nothing: thinking\n // is opt-in for the anthropic models bedrock hosts.\n ...(thinkingBudget\n ? {\n additionalModelRequestFields: {\n thinking: { type: \"enabled\", budget_tokens: thinkingBudget },\n },\n }\n : {}),\n }),\n )) as BedrockConverseResult;\n const text = result.output?.message?.content?.[0]?.text ?? \"\";\n const usage = result.usage ?? {};\n return [text, { input: usage.inputTokens ?? 0, output: usage.outputTokens ?? 0 }];\n } finally {\n runtime.destroy?.();\n }\n }\n}\n\nasync function loadBedrockSdk(): Promise<{\n BedrockRuntimeClient: new (\n cfg: Record<string, unknown>,\n ) => {\n send(cmd: unknown): Promise<unknown>;\n destroy?: () => void;\n };\n ConverseCommand: new (input: Record<string, unknown>) => unknown;\n}> {\n const specifier = \"@aws-sdk/client-bedrock-runtime\";\n try {\n return (await import(specifier)) as Awaited<ReturnType<typeof loadBedrockSdk>>;\n } catch {\n throw new ExtraMissingError(\"bedrock\", specifier, \"bedrock llm\");\n }\n}\n\nexport function buildLlmClient(cfg: LLMConfig, opts?: LLMClientOpts): LLMClient {\n if (opts?.client || opts?.fetch || opts?.fetchImpl) {\n return new LLMClient(cfg, opts);\n }\n if (cfg.provider === \"anthropic\") {\n return new LLMClient(cfg, { fetch: globalThis.fetch });\n }\n if (OPENAI_FAMILY.has(cfg.provider)) {\n return new LLMClient(cfg, { client: buildOpenAIChatClient(cfg) });\n }\n if (GOOGLE_FAMILY.has(cfg.provider) || cfg.provider === \"bedrock\") {\n return new LLMClient(cfg);\n }\n throw new Error(`unknown llm provider: ${JSON.stringify(cfg.provider)}`);\n}\n\n/**\n * Call the LLM configured by `cfg`. Returns `[text, tokenUsage]`.\n *\n * With no `client`, one is built for this call and CLOSED afterwards,\n * success or failure. Pass `client` to reuse a connection across many calls;\n * the caller then owns `aclose()`.\n */\nexport async function callLlm(\n cfg: LLMConfig,\n opts: LlmCallOpts & { client?: LLMClient | null },\n): Promise<[string, TokenUsage]> {\n const { client, ...callOpts } = opts;\n if (client) {\n return client.call(callOpts);\n }\n const owned = buildLlmClient(cfg);\n try {\n return await owned.call(callOpts);\n } finally {\n await owned.aclose();\n }\n}\n","import { createCipheriv, createDecipheriv, createHmac, randomBytes } from \"node:crypto\";\n\n/**\n * AES-256-GCM crypto seam. Wire format is a compatibility promise with the\n * Python library and promptev-connectors:\n * base64(nonce[12] || AES-256-GCM ciphertext+tag)\n */\n\nexport function encryptDict(data: Record<string, unknown>, key: Buffer): string {\n const nonce = randomBytes(12);\n const cipher = createCipheriv(\"aes-256-gcm\", key, nonce);\n const plaintext = Buffer.from(JSON.stringify(data), \"utf8\");\n const ciphertext = Buffer.concat([cipher.update(plaintext), cipher.final()]);\n const tag = cipher.getAuthTag();\n return Buffer.concat([nonce, ciphertext, tag]).toString(\"base64\");\n}\n\nexport function decryptDict(token: string, key: Buffer): Record<string, unknown> {\n const raw = Buffer.from(token, \"base64\");\n const nonce = raw.subarray(0, 12);\n const tag = raw.subarray(raw.length - 16);\n const ciphertext = raw.subarray(12, raw.length - 16);\n const decipher = createDecipheriv(\"aes-256-gcm\", key, nonce);\n decipher.setAuthTag(tag);\n const plaintext = Buffer.concat([decipher.update(ciphertext), decipher.final()]);\n return JSON.parse(plaintext.toString(\"utf8\")) as Record<string, unknown>;\n}\n\nexport function getSecretKey(config: { secretKey?: string | null }): Buffer {\n const secretKey = config.secretKey;\n if (!secretKey) {\n throw new Error(\n \"ContextEngineConfig.secretKey (env CE_SECRET_KEY) is not configured — \" +\n \"a base64url-encoded 32-byte AES key is required to encrypt/decrypt \" +\n \"tool configs that hold secrets.\",\n );\n }\n return Buffer.from(secretKey, \"base64url\");\n}\n\nexport function hmacSha256Hex(key: string | Buffer, value: string): string {\n const k = typeof key === \"string\" ? Buffer.from(key) : key;\n return createHmac(\"sha256\", k).update(value, \"utf8\").digest(\"hex\");\n}\n","import { hmacSha256Hex } from \"./crypto.js\";\nimport { emitError, type Hooks } from \"./hooks.js\";\n\nexport type Span = [number, number];\nexport type DetectorFn = (text: string) => Span[];\nexport type RedactionAction = \"mask\" | \"hash\" | \"remove\";\nexport type ApplyAt = \"ingest\" | \"output\" | \"both\";\nexport type RedactionPhase = \"ingest\" | \"output\";\n\nexport const BUILTIN_DETECTOR_NAMES = new Set([\"email\", \"phone\", \"ssn\", \"credit_card\", \"iban\", \"api_key\"]);\n\nexport interface RedactionRuleInit {\n name: string;\n detector?: string | null;\n pattern?: string | null;\n field?: string | null;\n action?: RedactionAction;\n placeholder?: string | null;\n applyAt?: ApplyAt;\n unless?: string[];\n}\n\nexport class RedactionRule {\n name: string;\n detector: string | null;\n pattern: string | null;\n field: string | null;\n action: RedactionAction;\n placeholder: string | null;\n applyAt: ApplyAt;\n unless: string[];\n private compiled: RegExp | null = null;\n\n constructor(init: RedactionRuleInit) {\n this.name = init.name;\n this.detector = init.detector ?? null;\n this.pattern = init.pattern ?? null;\n this.field = init.field ?? null;\n this.action = init.action ?? \"mask\";\n this.placeholder = init.placeholder ?? null;\n this.applyAt = init.applyAt ?? \"output\";\n this.unless = init.unless ?? [];\n this.validate();\n }\n\n private validate(): void {\n if (this.field !== null) {\n throw new Error(\n `rule '${this.name}': field-targeted rules are not implemented yet; use detector or pattern instead`,\n );\n }\n const targets = [this.detector, this.pattern].filter((t) => t !== null);\n if (targets.length !== 1) {\n throw new Error(`rule '${this.name}': exactly one of detector/pattern must be set`);\n }\n if (this.pattern !== null) {\n try {\n this.compiled = new RegExp(this.pattern, \"g\");\n } catch (exc) {\n throw new Error(`rule '${this.name}': invalid regex: ${exc}`);\n }\n }\n if (this.unless.length && this.applyAt === \"ingest\") {\n throw new Error(\n `rule '${this.name}': unless is output-time only and cannot be set on an applyAt='ingest' rule`,\n );\n }\n }\n\n patternRe(): RegExp | null {\n return this.compiled;\n }\n\n effectivePlaceholder(): string {\n return this.placeholder || `[${this.name.toUpperCase()}]`;\n }\n}\n\nexport interface RedactionPolicyInit {\n rules?: RedactionRule[] | RedactionRuleInit[];\n customDetectors?: Record<string, DetectorFn>;\n}\n\nexport class RedactionPolicy {\n rules: RedactionRule[];\n customDetectors: Record<string, DetectorFn>;\n\n constructor(init: RedactionPolicyInit = {}) {\n this.rules = (init.rules ?? []).map((r) => (r instanceof RedactionRule ? r : new RedactionRule(r)));\n this.customDetectors = init.customDetectors ?? {};\n const seen = new Set<string>();\n for (const rule of this.rules) {\n if (seen.has(rule.name)) throw new Error(`duplicate rule name: '${rule.name}'`);\n seen.add(rule.name);\n if (rule.detector !== null) {\n const known = BUILTIN_DETECTOR_NAMES.has(rule.detector) || rule.detector in this.customDetectors;\n if (!known) {\n throw new Error(\n `rule '${rule.name}': unknown detector '${rule.detector}' (not built-in and not in customDetectors)`,\n );\n }\n }\n }\n }\n\n isEmpty(): boolean {\n return this.rules.length === 0;\n }\n}\n\nconst EMAIL_RE = /[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\\.[A-Za-z]{2,}/g;\nconst PHONE_RE = /\\+\\d[\\d\\s-]{7,17}\\d/g;\nconst SSN_RE = /(?<!\\d)\\d{3}-\\d{2}-\\d{4}(?!\\d)/g;\nconst CARD_RE = /(?<!\\d)(?:\\d[ -]?){12,18}\\d(?!\\d)/g;\nconst IBAN_RE = /(?<![A-Za-z0-9])[A-Z]{2}\\d{2}[A-Z0-9]{10,30}(?![A-Za-z0-9])/g;\nconst API_KEY_ALNUM = \"A-Za-z0-9_\\\\-+/=\";\nconst API_KEY_PREFIX_RE = new RegExp(\n `(?<![${API_KEY_ALNUM}])(?:(?:AKIA|ASIA)[0-9A-Z]{16}|sk-[A-Za-z0-9]{20,}|gh[opsu]_[A-Za-z0-9]{20,}|xox[baprs]-[A-Za-z0-9\\\\-]{10,}|AIza[0-9A-Za-z_\\\\-]{35})(?![${API_KEY_ALNUM}])`,\n \"g\",\n);\nconst API_KEY_GENERIC_RE = new RegExp(\n `(?<![${API_KEY_ALNUM}])[${API_KEY_ALNUM}]{24,}(?![${API_KEY_ALNUM}])`,\n \"g\",\n);\nconst UUID_RE = /^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$/;\n\nfunction spansFrom(re: RegExp, text: string): Span[] {\n const out: Span[] = [];\n re.lastIndex = 0;\n let m = re.exec(text);\n while (m !== null) {\n out.push([m.index, m.index + m[0].length]);\n if (m[0].length === 0) re.lastIndex++;\n m = re.exec(text);\n }\n return out;\n}\n\nfunction luhnOk(digits: string): boolean {\n let total = 0;\n const rev = [...digits].reverse();\n for (let i = 0; i < rev.length; i++) {\n let d = Number(rev[i]);\n if (i % 2 === 1) {\n d *= 2;\n if (d > 9) d -= 9;\n }\n total += d;\n }\n return total % 10 === 0;\n}\n\nfunction detectCreditCard(text: string): Span[] {\n const spans: Span[] = [];\n for (const [start, end] of spansFrom(CARD_RE, text)) {\n const digits = text.slice(start, end).replace(/[ -]/g, \"\");\n if (digits.length >= 13 && digits.length <= 19 && luhnOk(digits)) {\n spans.push([start, end]);\n }\n }\n return spans;\n}\n\nfunction looksLikeGenericSecret(token: string): boolean {\n if (UUID_RE.test(token)) return false;\n let hasUpper = false;\n let hasLower = false;\n let hasDigit = false;\n for (const c of token) {\n if (c >= \"A\" && c <= \"Z\") hasUpper = true;\n else if (c >= \"a\" && c <= \"z\") hasLower = true;\n else if (c >= \"0\" && c <= \"9\") hasDigit = true;\n }\n return hasUpper && hasLower && hasDigit;\n}\n\nfunction detectApiKey(text: string): Span[] {\n const spans = spansFrom(API_KEY_PREFIX_RE, text);\n for (const [start, end] of spansFrom(API_KEY_GENERIC_RE, text)) {\n if (spans.some(([s, e]) => start < e && s < end)) continue;\n if (looksLikeGenericSecret(text.slice(start, end))) spans.push([start, end]);\n }\n return spans;\n}\n\nconst BUILTIN: Record<string, DetectorFn> = {\n email: (t) => spansFrom(EMAIL_RE, t),\n phone: (t) => spansFrom(PHONE_RE, t),\n ssn: (t) => spansFrom(SSN_RE, t),\n credit_card: detectCreditCard,\n iban: (t) => spansFrom(IBAN_RE, t),\n api_key: detectApiKey,\n};\n\nexport function detectBuiltin(name: string, text: string): Span[] {\n const detector = BUILTIN[name];\n if (!detector) throw new Error(`unknown built-in detector: ${name}`);\n if (!text) return [];\n return detector(text).sort((a, b) => a[0] - b[0]);\n}\n\nexport function ruleApplies(\n rule: RedactionRule,\n opts: { phase: RedactionPhase; principals: readonly string[] | null },\n): boolean {\n if (rule.applyAt !== \"both\" && rule.applyAt !== opts.phase) return false;\n if (opts.phase === \"ingest\") return true;\n if (!rule.unless.length) return true;\n if (opts.principals === null) return false;\n const held = new Set(opts.principals);\n return !rule.unless.some((p) => held.has(p));\n}\n\nfunction spansForRule(rule: RedactionRule, text: string, policy: RedactionPolicy): Span[] {\n if (rule.pattern !== null) {\n const re = rule.patternRe() ?? new RegExp(rule.pattern, \"g\");\n return spansFrom(re, text);\n }\n if (rule.detector !== null) {\n const custom = policy.customDetectors[rule.detector];\n if (custom) return [...custom(text)];\n return detectBuiltin(rule.detector, text);\n }\n return [];\n}\n\nconst HASH_TOKEN_CHARS = 16;\n\nfunction hashToken(value: string, secretKey: string | Buffer | null | undefined): string {\n const key = Buffer.isBuffer(secretKey) ? secretKey : Buffer.from(String(secretKey ?? \"\"));\n return hmacSha256Hex(key, value).slice(0, HASH_TOKEN_CHARS);\n}\n\nexport interface RedactionNote {\n rules_fired?: string[];\n spans?: number;\n rules_failed?: string[];\n}\n\nexport function applyRedaction(\n text: string,\n policy: RedactionPolicy,\n opts: {\n phase: RedactionPhase;\n principals?: readonly string[] | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks | null;\n },\n): [string, RedactionNote] {\n if (!text || policy.isEmpty()) return [text, {}];\n const principals = opts.principals ?? null;\n const collected: Array<[number, number, RedactionRule]> = [];\n const fired: string[] = [];\n const failed: string[] = [];\n\n for (const rule of policy.rules) {\n if (!ruleApplies(rule, { phase: opts.phase, principals })) continue;\n if (rule.action === \"hash\" && !opts.secretKey) {\n throw new Error(\n `rule '${rule.name}': action='hash' requires a non-empty secretKey ` +\n `(an unkeyed HMAC is a reversible pseudonym, not a redaction)`,\n );\n }\n try {\n const raw = spansForRule(rule, text, policy);\n const ruleSpans: Span[] = [];\n let invalid = false;\n for (const item of raw) {\n const start = item?.[0];\n const end = item?.[1];\n if (typeof start !== \"number\" || typeof end !== \"number\") {\n invalid = true;\n continue;\n }\n if (!(start >= 0 && start < end && end <= text.length)) {\n invalid = true;\n continue;\n }\n ruleSpans.push([start, end]);\n }\n if (invalid) failed.push(rule.name);\n for (const [s, e] of ruleSpans) collected.push([s, e, rule]);\n } catch (exc) {\n failed.push(rule.name);\n if (opts.hooks) emitError(opts.hooks, exc, { stage: \"redaction\", rule: rule.name });\n }\n }\n\n if (!collected.length) {\n if (failed.length) return [text, { rules_fired: [], spans: 0, rules_failed: failed }];\n return [text, {}];\n }\n\n collected.sort((a, b) => a[0] - b[0] || b[1] - b[0] - (a[1] - a[0]));\n const merged: Array<[number, number, RedactionRule]> = [];\n for (const [start, end, rule] of collected) {\n const last = merged[merged.length - 1];\n if (last && start < last[1]) {\n if (end > last[1]) last[1] = end;\n continue;\n }\n merged.push([start, end, rule]);\n }\n\n const out: string[] = [];\n let cursor = 0;\n for (const [start, end, rule] of merged) {\n out.push(text.slice(cursor, start));\n const original = text.slice(start, end);\n if (rule.action === \"mask\") out.push(rule.effectivePlaceholder());\n else if (rule.action === \"hash\") {\n out.push(`[${rule.name.toUpperCase()}:${hashToken(original, opts.secretKey)}]`);\n }\n if (!fired.includes(rule.name)) fired.push(rule.name);\n cursor = end;\n }\n out.push(text.slice(cursor));\n\n const note: RedactionNote = { rules_fired: fired, spans: merged.length };\n if (failed.length) note.rules_failed = failed;\n return [out.join(\"\"), note];\n}\n","/**\n * Unified JSON parsing with jsonrepair fallback for LLM output.\n *\n * Handles markdown fences, truncated output, trailing commas, unclosed\n * strings/brackets, and other common LLM JSON errors.\n */\nimport { jsonrepair } from \"jsonrepair\";\n\nexport type JsonExpected = \"object\" | \"array\";\n\nfunction stripFences(text: string): string {\n let t = text.trim();\n const fenced = /^```(?:json)?\\s*([\\s\\S]*?)\\s*```$/i.exec(t);\n if (fenced?.[1]) t = fenced[1].trim();\n return t;\n}\n\nfunction isExpected(value: unknown, expectedType: JsonExpected): boolean {\n if (expectedType === \"array\") return Array.isArray(value);\n return typeof value === \"object\" && value !== null && !Array.isArray(value);\n}\n\n/**\n * Parse JSON with automatic repair fallback for LLM output.\n *\n * @throws {SyntaxError} If parsing and repair both fail to produce `expectedType`.\n */\nexport function safeJsonParse(text: string, expectedType: JsonExpected = \"object\"): unknown {\n const stripped = stripFences(text);\n\n try {\n const result: unknown = JSON.parse(stripped);\n if (isExpected(result, expectedType)) return result;\n } catch {\n /* fall through to repair */\n }\n\n let repaired: unknown;\n try {\n repaired = JSON.parse(jsonrepair(stripped));\n } catch {\n throw new SyntaxError(`Expected ${expectedType}, jsonrepair could not parse`);\n }\n if (isExpected(repaired, expectedType)) {\n console.info(`json_repair recovered ${Array.isArray(repaired) ? \"array\" : \"object\"}`);\n return repaired;\n }\n throw new SyntaxError(\n `Expected ${expectedType}, got ${Array.isArray(repaired) ? \"array\" : typeof repaired}`,\n );\n}\n","/**\n * File-format text extraction — dispatch helpers + concrete extractors for\n * the mainstream office/text formats (docx/pptx/xlsx/csv/html/eml).\n *\n * Legacy binary formats (.doc/.ppt/.xls/.msg/.odt/...) are not parsed here:\n * `extract()` falls through to a raw utf-8 decode, matching the Python\n * package's worst-case behavior.\n */\nimport { load } from \"cheerio\";\nimport ExcelJS from \"exceljs\";\nimport JSZip from \"jszip\";\nimport mammoth from \"mammoth\";\nimport PostalMime from \"postal-mime\";\n\nconst ZIP_MAGIC = Buffer.from([0x50, 0x4b, 0x03, 0x04]);\n\nexport const NON_INGESTIBLE_MEDIA_EXTS = new Set([\n \".ogg\",\n \".oga\",\n \".opus\",\n \".mp3\",\n \".wav\",\n \".m4a\",\n \".aac\",\n \".flac\",\n \".wma\",\n \".amr\",\n \".mp4\",\n \".m4v\",\n \".mov\",\n \".avi\",\n \".mkv\",\n \".webm\",\n \".wmv\",\n \".flv\",\n \".3gp\",\n \".mpg\",\n \".mpeg\",\n]);\n\nexport const DOCX_MIME = \"application/vnd.openxmlformats-officedocument.wordprocessingml.document\";\nexport const PPTX_MIME = \"application/vnd.openxmlformats-officedocument.presentationml.presentation\";\nexport const XLSX_MIME = \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\";\n\nexport function getFileExtension(filename: string | null | undefined): string {\n if (!filename?.includes(\".\")) return \"\";\n return filename.slice(filename.lastIndexOf(\".\")).toLowerCase();\n}\n\n/**\n * True for audio/video files, which must never be treated as text-bearing.\n * Bare `application/octet-stream` does NOT match.\n */\nexport function isNonIngestibleMedia(\n filename: string | null | undefined,\n mime: string | null | undefined,\n): boolean {\n const m = (mime || \"\").toLowerCase();\n if (m.startsWith(\"audio/\") || m.startsWith(\"video/\")) return true;\n return NON_INGESTIBLE_MEDIA_EXTS.has(getFileExtension(filename));\n}\n\nfunction isZip(content: Buffer): boolean {\n return content.length >= 4 && content.subarray(0, 4).equals(ZIP_MAGIC);\n}\n\nfunction decodeXmlEntities(s: string): string {\n return s\n .replace(/</g, \"<\")\n .replace(/>/g, \">\")\n .replace(/"/g, '\"')\n .replace(/'/g, \"'\")\n .replace(/&/g, \"&\");\n}\n\nfunction collectTagText(xml: string, tag: string): string[] {\n const out: string[] = [];\n const re = new RegExp(`<${tag}(?:\\\\s[^>]*)?>([^<]*)</${tag}>`, \"g\");\n for (const m of xml.matchAll(re)) {\n const t = decodeXmlEntities(m[1] ?? \"\").trim();\n if (t) out.push(t);\n }\n return out;\n}\n\nfunction htmlToPlain(html: string, extraRemove = \"\"): string {\n try {\n const $ = load(html);\n $(`script, style, head${extraRemove ? `, ${extraRemove}` : \"\"}`).remove();\n $(\"br\").replaceWith(\"\\n\");\n $(\"p, div, h1, h2, h3, h4, h5, h6, tr, li, blockquote\").each((_, el) => {\n $(el).prepend(\"\\n\");\n });\n const lines = $.root()\n .text()\n .split(/\\n/)\n .map((l) => l.trim())\n .filter(Boolean);\n return lines.join(\"\\n\");\n } catch {\n return html;\n }\n}\n\nfunction mammothHtmlToText(html: string): string {\n const $ = load(html);\n $(\"table\").each((_, table) => {\n const rows: string[] = [];\n $(table)\n .find(\"tr\")\n .each((__, tr) => {\n const cells = $(tr)\n .find(\"th, td\")\n .map((___, td) => $(td).text().trim())\n .get()\n .filter(Boolean);\n if (cells.length) rows.push(cells.join(\" | \"));\n });\n $(table).replaceWith(`${rows.join(\"\\n\")}\\n`);\n });\n $(\"br\").replaceWith(\"\\n\");\n $(\"p, div, h1, h2, h3, h4, h5, h6, li\").each((_, el) => {\n $(el).prepend(\"\\n\");\n });\n const lines = $.root()\n .text()\n .split(/\\n/)\n .map((l) => l.trim())\n .filter(Boolean);\n return lines.join(\"\\n\");\n}\n\nasync function docxHeaderFooterText(content: Buffer): Promise<string[]> {\n try {\n const zip = await JSZip.loadAsync(content);\n const parts: string[] = [];\n const names = Object.keys(zip.files).filter((n) => /^word\\/(header|footer)\\d*\\.xml$/i.test(n));\n for (const name of names) {\n const xml = await zip.files[name]!.async(\"string\");\n parts.push(...collectTagText(xml, \"w:t\"));\n }\n return parts;\n } catch {\n return [];\n }\n}\n\nexport async function extractDocxText(content: Buffer): Promise<string> {\n if (!isZip(content)) {\n console.warn(\"extractDocxText: not a ZIP archive, decoding as plain text\");\n return content.toString(\"utf8\");\n }\n const { value } = await mammoth.convertToHtml({ buffer: content });\n const body = mammothHtmlToText(value || \"\");\n const extras = await docxHeaderFooterText(content);\n const parts = [body, ...extras].filter((p) => p.trim());\n return parts.join(\"\\n\");\n}\n\nexport async function countDocxPages(content: Buffer): Promise<number> {\n try {\n const { value } = await mammoth.extractRawText({ buffer: content });\n const wordCount = (value || \"\").trim() ? (value || \"\").trim().split(/\\s+/).length : 0;\n return Math.max(1, Math.round(wordCount / 250));\n } catch {\n return 1;\n }\n}\n\nfunction cellString(value: unknown): string {\n if (value == null) return \"\";\n if (value instanceof Date) return value.toISOString();\n if (typeof value === \"object\") {\n const rec = value as { result?: unknown; text?: unknown; richText?: Array<{ text?: string }> };\n if (\"result\" in rec) return cellString(rec.result);\n if (Array.isArray(rec.richText)) return rec.richText.map((t) => t.text ?? \"\").join(\"\");\n if (typeof rec.text === \"string\") return rec.text;\n }\n return String(value);\n}\n\nexport async function extractXlsxText(content: Buffer): Promise<string> {\n if (!isZip(content)) {\n console.warn(\"extractXlsxText: not a ZIP archive, decoding as plain text\");\n return content.toString(\"utf8\");\n }\n try {\n const probe = new ExcelJS.Workbook();\n await probe.xlsx.load(content as unknown as ArrayBuffer);\n const colCaps = new Map<string, number>();\n for (const ws of probe.worksheets) {\n const widths: number[] = [];\n let i = 0;\n ws.eachRow({ includeEmpty: true }, (row) => {\n if (i === 0) {\n i += 1;\n return;\n }\n if (i > 20) return;\n i += 1;\n const values = (row.values as unknown[]) ?? [];\n let lastData = 0;\n for (let j = 1; j < values.length; j++) {\n const v = values[j];\n if (v != null && cellString(v).trim()) lastData = j;\n }\n if (lastData > 0) widths.push(lastData);\n });\n colCaps.set(ws.name, widths.length ? Math.max(...widths) : 50);\n }\n\n const wb = new ExcelJS.Workbook();\n await wb.xlsx.load(content as unknown as ArrayBuffer);\n const out: string[] = [];\n for (const ws of wb.worksheets) {\n out.push(`[Sheet: ${ws.name}]`);\n const cap = colCaps.get(ws.name) ?? 50;\n ws.eachRow({ includeEmpty: true }, (row) => {\n const values = (row.values as unknown[]) ?? [];\n const trimmed: string[] = [];\n const last = Math.min(values.length - 1, cap);\n for (let j = 1; j <= last; j++) trimmed.push(cellString(values[j]));\n while (trimmed.length && !trimmed[trimmed.length - 1]!.trim()) trimmed.pop();\n if (trimmed.length) out.push(trimmed.join(\", \"));\n });\n }\n return out.join(\"\\n\");\n } catch (err) {\n const msg = err instanceof Error ? err.message : String(err);\n if (/zip|corrupt|invalid/i.test(msg)) return content.toString(\"utf8\");\n throw err;\n }\n}\n\nfunction pptxTableRows(tblXml: string): string[] {\n const rows: string[] = [];\n for (const tr of tblXml.matchAll(/<a:tr\\b[\\s\\S]*?<\\/a:tr>/g)) {\n const cells: string[] = [];\n for (const tc of tr[0]!.matchAll(/<a:tc\\b[\\s\\S]*?<\\/a:tc>/g)) {\n const texts = collectTagText(tc[0]!, \"a:t\");\n const joined = texts.join(\" \").trim();\n if (joined) cells.push(joined);\n }\n if (cells.length) rows.push(cells.join(\" | \"));\n }\n return rows;\n}\n\nfunction pptxSlideBody(xml: string): string[] {\n const parts: string[] = [];\n const withoutTables = xml.replace(/<a:tbl\\b[\\s\\S]*?<\\/a:tbl>/g, (tbl) => {\n parts.push(...pptxTableRows(tbl));\n return \"\";\n });\n parts.push(...collectTagText(withoutTables, \"a:t\"));\n return parts.filter(Boolean);\n}\n\nfunction parseRels(xml: string): Map<string, string> {\n const map = new Map<string, string>();\n const re = /Id=\"([^\"]+)\"[^>]*Target=\"([^\"]+)\"|Target=\"([^\"]+)\"[^>]*Id=\"([^\"]+)\"/g;\n for (const m of xml.matchAll(re)) {\n const id = m[1] || m[4];\n const target = m[2] || m[3];\n if (id && target) map.set(id, target);\n }\n return map;\n}\n\nasync function pptxSlideOrder(zip: JSZip): Promise<string[]> {\n const pres = zip.file(\"ppt/presentation.xml\");\n const rels = zip.file(\"ppt/_rels/presentation.xml.rels\");\n if (!pres || !rels) {\n return Object.keys(zip.files)\n .filter((n) => /^ppt\\/slides\\/slide\\d+\\.xml$/.test(n))\n .sort((a, b) => {\n const na = Number(/slide(\\d+)/.exec(a)?.[1] ?? 0);\n const nb = Number(/slide(\\d+)/.exec(b)?.[1] ?? 0);\n return na - nb;\n });\n }\n const presXml = await pres.async(\"string\");\n const relsMap = parseRels(await rels.async(\"string\"));\n const ids: string[] = [];\n for (const m of presXml.matchAll(/<p:sldId\\b[^>]*r:id=\"([^\"]+)\"/g)) {\n ids.push(m[1]!);\n }\n const names: string[] = [];\n for (const id of ids) {\n const target = relsMap.get(id);\n if (!target) continue;\n const path = target.replace(/^\\.\\//, \"\");\n names.push(path.startsWith(\"ppt/\") ? path : `ppt/${path}`);\n }\n return names;\n}\n\nexport async function extractPptxText(content: Buffer): Promise<string> {\n if (!isZip(content)) {\n console.warn(\"extractPptxText: not a ZIP archive, decoding as plain text\");\n return content.toString(\"utf8\");\n }\n const zip = await JSZip.loadAsync(content);\n const slideFiles = await pptxSlideOrder(zip);\n const parts: string[] = [];\n for (let i = 0; i < slideFiles.length; i++) {\n const name = slideFiles[i]!;\n const file = zip.file(name);\n if (!file) continue;\n parts.push(`[Slide ${i + 1}]`);\n const xml = await file.async(\"string\");\n parts.push(...pptxSlideBody(xml));\n const relsName = name.replace(/slides\\/(slide\\d+\\.xml)$/, \"slides/_rels/$1.rels\");\n const relsFile = zip.file(relsName);\n if (relsFile) {\n const relsXml = await relsFile.async(\"string\");\n const notesTarget = [...relsXml.matchAll(/Target=\"([^\"]*notesSlide[^\"]*)\"/gi)][0]?.[1];\n if (notesTarget) {\n const notesPath = notesTarget.startsWith(\"/\")\n ? notesTarget.slice(1)\n : name.replace(/slides\\/slide\\d+\\.xml$/, \"\") +\n notesTarget.replace(/^\\.\\.\\//, \"\").replace(/^\\.\\//, \"\");\n const notesFile = zip.file(notesPath) ?? zip.file(`ppt/${notesTarget.replace(/^(\\.\\.\\/)+/, \"\")}`);\n if (notesFile) {\n const notesXml = await notesFile.async(\"string\");\n const notes = collectTagText(notesXml, \"a:t\").join(\" \").trim();\n if (notes) parts.push(`[Notes] ${notes}`);\n }\n }\n }\n parts.push(\"\");\n }\n return parts.join(\"\\n\");\n}\n\nexport async function countPptxSlides(content: Buffer): Promise<number> {\n try {\n const zip = await JSZip.loadAsync(content);\n const n = (await pptxSlideOrder(zip)).length;\n return n || 1;\n } catch {\n return 1;\n }\n}\n\nexport function extractCsvText(content: Buffer, delimiter = \",\"): string {\n const txt = content.toString(\"utf8\");\n try {\n const rows = parseCsv(txt, delimiter);\n const out: string[] = [];\n for (const row of rows) {\n const trimmed = [...row];\n while (trimmed.length && !String(trimmed[trimmed.length - 1]).trim()) trimmed.pop();\n if (trimmed.length) out.push(trimmed.map((c) => String(c)).join(\", \"));\n }\n return out.join(\"\\n\");\n } catch {\n return txt;\n }\n}\n\nfunction parseCsv(text: string, delimiter: string): string[][] {\n const rows: string[][] = [];\n let row: string[] = [];\n let cell = \"\";\n let i = 0;\n let inQuotes = false;\n while (i < text.length) {\n const ch = text[i]!;\n if (inQuotes) {\n if (ch === '\"') {\n if (text[i + 1] === '\"') {\n cell += '\"';\n i += 2;\n continue;\n }\n inQuotes = false;\n i += 1;\n continue;\n }\n cell += ch;\n i += 1;\n continue;\n }\n if (ch === '\"') {\n inQuotes = true;\n i += 1;\n continue;\n }\n if (ch === delimiter) {\n row.push(cell);\n cell = \"\";\n i += 1;\n continue;\n }\n if (ch === \"\\r\") {\n i += 1;\n continue;\n }\n if (ch === \"\\n\") {\n row.push(cell);\n rows.push(row);\n row = [];\n cell = \"\";\n i += 1;\n continue;\n }\n cell += ch;\n i += 1;\n }\n if (cell.length || row.length) {\n row.push(cell);\n rows.push(row);\n }\n return rows;\n}\n\nexport function extractHtmlText(content: Buffer): string {\n try {\n return htmlToPlain(content.toString(\"utf8\"));\n } catch (exc) {\n console.debug(\"HTML extraction failed:\", exc);\n return content.toString(\"utf8\");\n }\n}\n\nfunction formatAddress(\n addr:\n | { name?: string; address?: string; text?: string }\n | Array<{ name?: string; address?: string; text?: string }>\n | undefined,\n): string {\n if (!addr) return \"\";\n const one = (a: { name?: string; address?: string; text?: string }) =>\n a.text || [a.name, a.address].filter(Boolean).join(\" \") || \"\";\n return Array.isArray(addr) ? addr.map(one).filter(Boolean).join(\", \") : one(addr);\n}\n\nexport async function extractEmlText(content: Buffer): Promise<string> {\n try {\n const msg = await PostalMime.parse(content);\n const parts: string[] = [];\n const from = formatAddress(msg.from);\n const to = formatAddress(msg.to);\n if (from) parts.push(`From: ${from}`);\n if (to) parts.push(`To: ${to}`);\n if (msg.subject) parts.push(`Subject: ${msg.subject}`);\n const date = (msg as { date?: string }).date;\n if (date) parts.push(`Date: ${date}`);\n parts.push(\"\");\n if (msg.text) {\n parts.push(msg.text);\n } else if (msg.html) {\n parts.push(htmlToPlain(msg.html, \"head\"));\n }\n return parts.join(\"\\n\");\n } catch (exc) {\n console.debug(\"EML extraction failed:\", exc);\n return content.toString(\"utf8\");\n }\n}\n","/**\n * File -> text extraction. `extract()` is the single public entry point.\n *\n * Never throws for extraction failures internal to a single file — those are\n * reported via `hooks.onError` and degrade to whatever partial text was recovered.\n */\nimport type { ExtractionConfig, LLMConfig } from \"../config.js\";\nimport { tryImport } from \"../extras.js\";\nimport { emitError, type Hooks } from \"../hooks.js\";\nimport * as files from \"./files.js\";\n\nconst IMAGE_EXTS = new Set([\".png\", \".jpg\", \".jpeg\", \".gif\", \".bmp\", \".tiff\", \".tif\", \".webp\"]);\n\n/**\n * WHY a document came back with less text than it has, when the answer is\n * actionable. Mirrors the Python client's `Extracted.unreadable_reason` —\n * that docstring is the canonical rationale; only the value semantics are\n * restated here:\n *\n * - `needs_vision` — pages had no usable text layer and no vision model was\n * configured. FIXABLE by configuring one (the OCR fallback may have tried\n * and read nothing; a vision model is still the actionable upgrade).\n * - `vision_failed` — a vision model was configured and produced nothing for\n * those pages. An ops problem, not a configuration one.\n *\n * `null` means nothing was lost — a page a reader READ and found empty is\n * blank, not unreadable. A closed set on purpose (callers branch on it), and\n * NOT an error channel; `mediaOnly` stays set alongside it for whole files.\n */\nexport type UnreadableReason = \"needs_vision\" | \"vision_failed\";\n\nexport class Extracted {\n text: string;\n pages: number | null;\n slides: number | null;\n mediaOnly: boolean;\n providerTokens: Record<string, number>;\n llmPictures: number;\n isMarkdown: boolean;\n /** See `UnreadableReason`. `mediaOnly` is the same idea for whole files and\n * stays set alongside this, so existing callers keep working. */\n unreadableReason: UnreadableReason | null;\n unreadablePages: number;\n\n constructor(init: {\n text: string;\n pages?: number | null;\n slides?: number | null;\n mediaOnly?: boolean;\n providerTokens?: Record<string, number>;\n llmPictures?: number;\n isMarkdown?: boolean;\n unreadableReason?: UnreadableReason | null;\n unreadablePages?: number;\n }) {\n this.text = init.text;\n this.pages = init.pages ?? null;\n this.slides = init.slides ?? null;\n this.mediaOnly = init.mediaOnly ?? false;\n this.providerTokens = init.providerTokens ?? {};\n this.llmPictures = init.llmPictures ?? 0;\n this.isMarkdown = init.isMarkdown ?? false;\n this.unreadableReason = init.unreadableReason ?? null;\n this.unreadablePages = init.unreadablePages ?? 0;\n }\n}\n\nfunction isPdf(ext: string, mime: string): boolean {\n return ext === \".pdf\" || mime === \"application/pdf\";\n}\n\nfunction isImage(ext: string, mime: string): boolean {\n return mime.startsWith(\"image/\") || IMAGE_EXTS.has(ext);\n}\n\n// Embedded DOCX/PPTX pictures below either floor are logos, bullets and\n// dividers — a vision call would cost credits to describe nothing.\nconst MIN_EMBEDDED_IMAGE_BYTES = 3 * 1024;\nconst MIN_EMBEDDED_IMAGE_PX = 100;\n\n/**\n * Reason for an image-only Office document that produced no text — the\n * DOCX/PPTX twins of the PDF cases: a screenshot deck with no vision model\n * used to ingest as an empty success. Only images past the negligibility\n * floors count — a letterhead logo in an otherwise empty file is not\n * unreadable content.\n */\nasync function officeUnreadable(\n text: string,\n images: Array<{ imageBytes: Buffer }>,\n visionLlm: LLMConfig | null,\n): Promise<{ reason: UnreadableReason | null; unread: number }> {\n if (text.trim()) return { reason: null, unread: 0 };\n let readable = 0;\n for (const img of images) {\n if (!(await embeddedImageIsNegligible(img.imageBytes))) readable += 1;\n }\n if (!readable) return { reason: null, unread: 0 };\n return { reason: visionLlm ? \"vision_failed\" : \"needs_vision\", unread: readable };\n}\n\n/**\n * Too small to carry content. Anything undecodable (or without the optional\n * canvas to decode with) is NOT negligible past the byte floor — an\n * unreadable image is the provider's business to reject.\n */\nasync function embeddedImageIsNegligible(data: Buffer): Promise<boolean> {\n if (data.length < MIN_EMBEDDED_IMAGE_BYTES) return true;\n const canvas = await tryImport<{\n loadImage: (src: Buffer) => Promise<{ width: number; height: number }>;\n }>(\"@napi-rs/canvas\");\n if (!canvas) return false;\n try {\n const img = await canvas.loadImage(data);\n return Math.max(img.width, img.height) < MIN_EMBEDDED_IMAGE_PX;\n } catch {\n return false;\n }\n}\n\n/**\n * Run embedded DOCX/PPTX pictures through the vision LLM. Never throws.\n *\n * The returned list is always the same length as `images` and positionally\n * aligned (\"\" for skipped pictures) — callers enumerate it to label\n * `[Image N]` / `[Slide N Image]`, so a shift would mislabel every one.\n * Negligible pictures are skipped and identical bytes are read once (a\n * deck's per-slide logo is one vision call, not forty), with the result\n * fanned back to every occurrence.\n */\nexport async function extractEmbeddedImagesText(\n images: Array<{ imageBytes: Buffer }>,\n visionLlm: LLMConfig,\n hooks: Hooks,\n extraction?: ExtractionConfig | null,\n): Promise<[string[], Record<string, number>]> {\n if (!images.length) return [[], {}];\n try {\n const vision = await import(\"./vision.js\");\n const { createHash } = await import(\"node:crypto\");\n\n const keys: Array<string | null> = [];\n const firstAt = new Map<string, number>(); // digest -> index into `send`\n const send: Buffer[] = [];\n for (const img of images) {\n const blob = img.imageBytes;\n if (await embeddedImageIsNegligible(blob)) {\n keys.push(null);\n continue;\n }\n const digest = createHash(\"sha256\").update(blob).digest(\"hex\");\n if (!firstAt.has(digest)) {\n firstAt.set(digest, send.length);\n send.push(blob);\n }\n keys.push(digest);\n }\n\n if (!send.length) return [images.map(() => \"\"), {}];\n\n const [texts, tokens] = await vision.extractTextFromImages(send, { visionLlm, extraction });\n return [keys.map((k) => (k == null ? \"\" : (texts[firstAt.get(k)!] ?? \"\"))), tokens];\n } catch (exc) {\n emitError(hooks, exc, { stage: \"embedded_image_vision\" });\n return [[], {}];\n }\n}\n\n/**\n * Extract text from `content` (raw file bytes). Never throws for per-file\n * extraction failures — those go through `hooks.onError` and degrade.\n */\nexport async function extract(\n content: Buffer,\n filename: string,\n mime: string | null,\n opts: { visionLlm?: LLMConfig | null; hooks: Hooks; extraction?: ExtractionConfig | null },\n): Promise<Extracted> {\n const m = mime || \"\";\n const ext = files.getFileExtension(filename);\n const visionLlm = opts.visionLlm ?? null;\n const hooks = opts.hooks;\n const extraction = opts.extraction ?? null;\n\n try {\n if (files.isNonIngestibleMedia(filename, m)) {\n return new Extracted({ text: \"\", mediaOnly: true });\n }\n\n if (isPdf(ext, m)) {\n const pdf = await import(\"./pdf.js\");\n return await pdf.extract(content, { visionLlm, hooks, extraction });\n }\n\n if (isImage(ext, m)) {\n if (!visionLlm) {\n return new Extracted({\n text: \"\",\n mediaOnly: true,\n unreadableReason: \"needs_vision\",\n unreadablePages: 1,\n });\n }\n try {\n const vision = await import(\"./vision.js\");\n return await vision.extractImage(content, { visionLlm, extraction });\n } catch (exc) {\n emitError(hooks, exc, { stage: \"image_vision\", filename });\n // A reader WAS configured and it blew up — a different answer from\n // \"no reader\", and a different fix.\n return new Extracted({\n text: \"\",\n mediaOnly: true,\n unreadableReason: \"vision_failed\",\n unreadablePages: 1,\n });\n }\n }\n\n if (ext === \".docx\" || m === files.DOCX_MIME) {\n const text = await files.extractDocxText(content);\n let providerTokens: Record<string, number> = {};\n let llmPictures = 0;\n let combined = text;\n let images: Array<{ imageBytes: Buffer }> = [];\n if (visionLlm) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractDocxImages(content);\n const [texts, tokens] = await extractEmbeddedImagesText(images, visionLlm, hooks, extraction);\n for (let i = 0; i < texts.length; i++) {\n const imgText = texts[i];\n if (imgText) {\n combined += `\\n[Image ${i + 1}]\\n${imgText}`;\n llmPictures += 1;\n }\n }\n providerTokens = tokens;\n } else if (!text.trim()) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractDocxImages(content);\n }\n const { reason, unread } = await officeUnreadable(combined, images, visionLlm);\n return new Extracted({\n text: combined,\n pages: await files.countDocxPages(content),\n providerTokens,\n llmPictures,\n unreadableReason: reason,\n unreadablePages: unread,\n });\n }\n\n if (ext === \".pptx\" || m === files.PPTX_MIME) {\n const text = await files.extractPptxText(content);\n let providerTokens: Record<string, number> = {};\n let llmPictures = 0;\n let combined = text;\n let images: Array<{ imageBytes: Buffer; slide?: number }> = [];\n if (visionLlm) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractPptxImages(content);\n const [texts, tokens] = await extractEmbeddedImagesText(images, visionLlm, hooks, extraction);\n for (let i = 0; i < texts.length; i++) {\n const imgText = texts[i];\n if (imgText) {\n const slideNum = images[i]?.slide ?? i + 1;\n combined += `\\n[Slide ${slideNum} Image]\\n${imgText}`;\n llmPictures += 1;\n }\n }\n providerTokens = tokens;\n } else if (!text.trim()) {\n const vision = await import(\"./vision.js\");\n images = await vision.extractPptxImages(content);\n }\n const { reason, unread } = await officeUnreadable(combined, images, visionLlm);\n return new Extracted({\n text: combined,\n slides: await files.countPptxSlides(content),\n providerTokens,\n llmPictures,\n unreadableReason: reason,\n unreadablePages: unread,\n });\n }\n\n if (ext === \".xlsx\" || m === files.XLSX_MIME) {\n return new Extracted({ text: await files.extractXlsxText(content) });\n }\n\n if (ext === \".tsv\" || m === \"text/tab-separated-values\") {\n return new Extracted({ text: files.extractCsvText(content, \"\\t\") });\n }\n\n if (ext === \".csv\" || m === \"text/csv\") {\n return new Extracted({ text: files.extractCsvText(content) });\n }\n\n if (ext === \".html\" || ext === \".htm\" || m === \"text/html\") {\n return new Extracted({ text: files.extractHtmlText(content) });\n }\n\n if (ext === \".eml\" || m === \"message/rfc822\") {\n return new Extracted({ text: await files.extractEmlText(content) });\n }\n\n return new Extracted({ text: content.toString(\"utf8\") });\n } catch (exc) {\n emitError(hooks, exc, { stage: \"extract\", filename });\n return new Extracted({ text: content.toString(\"utf8\") });\n }\n}\n","import type { Pool } from \"pg\";\nimport type { ContextEngineConfig } from \"../config.js\";\nimport { emitError, type Hooks } from \"../hooks.js\";\nimport type { Embedder } from \"../providers/embeddings.js\";\nimport type { AnnTunable } from \"../storage.js\";\nimport type { GraphStore } from \"./neo4j-client.js\";\n\nconst DEFAULT_WEIGHTS = {\n vector_score: 0.3,\n entity_match: 0.3,\n relationship_relevance: 0.2,\n community_match: 0.1,\n graph_connectivity: 0.1,\n};\n\nconst SEED_LIMIT = 200;\nconst ENTITY_LIMIT = 30;\nconst TOP_SEEDS_FOR_ENTITIES = 20;\n\n// The scope predicate. Textually a copy of storage.ts's SCOPE — which is the\n// canonical one — because these queries number their placeholders differently\n// ($2/$3/$4 here, $1/$2/$3 there) and pg binds are POSITIONAL, so the string\n// cannot be shared without renumbering every query on one side. Any change to\n// the canonical predicate must be made here too. Never widen it.\n//\n// $3 is cast to uuid[] so the predicate can use the document_id index —\n// same reasoning as storage.ts's SCOPE.\nconst SCOPE = `\n AND ($2::text[] IS NULL OR c.source_id = ANY($2::text[]))\n AND ($3::uuid[] IS NULL OR c.document_id = ANY($3::uuid[]))\n AND ($4::text[] IS NULL OR c.acl IS NULL OR c.acl && $4::text[])\n`;\n\nfunction vecLiteral(vector: number[]): string {\n return `[${vector.map((x) => Number(x)).join(\",\")}]`;\n}\n\nexport async function corpusIsAclUniform(pool: Pool, sourceIds: string[] | null): Promise<boolean> {\n const result = await pool.query(\n `SELECT EXISTS(SELECT 1 FROM context_engine_chunks c\n WHERE c.acl IS NOT NULL AND ($1::text[] IS NULL OR c.source_id = ANY($1::text[]))) AS has_acl`,\n [sourceIds],\n );\n return !result.rows[0]?.has_acl;\n}\n\nexport async function shouldUseCommunitySummaries(\n pool: Pool,\n opts: { sourceIds: string[] | null; principals: string[] | null },\n): Promise<boolean> {\n if (opts.principals == null) return true;\n return corpusIsAclUniform(pool, opts.sourceIds);\n}\n\nasync function deriveQueryEntities(\n pool: Pool,\n chunkIds: string[],\n opts: { sourceIds: string[] | null; documentIds?: string[] | null; principals: string[] | null },\n): Promise<Array<Record<string, unknown>>> {\n if (!chunkIds.length) return [];\n const result = await pool.query(\n `SELECT e.normalized_name, e.name, e.type, COUNT(*) AS freq\n FROM context_engine_chunk_entities ce\n JOIN context_engine_entities e ON e.id = ce.entity_id\n JOIN context_engine_chunks c ON c.id = ce.chunk_id\n WHERE ce.chunk_id = ANY($1::uuid[]) ${SCOPE}\n GROUP BY e.normalized_name, e.name, e.type\n ORDER BY freq DESC LIMIT $5`,\n [chunkIds, opts.sourceIds, opts.documentIds ?? null, opts.principals, ENTITY_LIMIT],\n );\n return result.rows.map((r) => ({\n normalized_name: r.normalized_name,\n name: r.name,\n type: r.type,\n }));\n}\n\nasync function computeVectorScores(\n pool: Pool,\n chunkIds: string[],\n vector: number[],\n): Promise<Record<string, number>> {\n if (!chunkIds.length || !vector) return {};\n const result = await pool.query(\n `SELECT c.id::text, 1 - (c.embedding <=> CAST($2 AS vector))::float AS sim\n FROM context_engine_chunks c\n WHERE c.id = ANY($1::uuid[]) AND c.embedding IS NOT NULL`,\n [chunkIds, vecLiteral(vector)],\n );\n let scores = Object.fromEntries(result.rows.map((r) => [String(r.id), Number(r.sim)]));\n const vs = Object.values(scores);\n if (vs.length) {\n const lo = Math.min(...vs);\n const hi = Math.max(...vs);\n if (hi > lo)\n scores = Object.fromEntries(Object.entries(scores).map(([k, v]) => [k, (v - lo) / (hi - lo)]));\n }\n return scores;\n}\n\nasync function computeRelationshipScores(\n pool: Pool,\n chunkIds: string[],\n queryEntityNames: string[],\n): Promise<Record<string, number>> {\n if (!chunkIds.length || !queryEntityNames.length) return {};\n const result = await pool.query(\n `SELECT r.chunk_id::text, COUNT(*) AS n\n FROM context_engine_entity_relationships r\n JOIN context_engine_entities se ON se.id = r.source_entity_id\n JOIN context_engine_entities te ON te.id = r.target_entity_id\n WHERE r.chunk_id = ANY($1::uuid[])\n AND (se.normalized_name = ANY($2::text[]) OR te.normalized_name = ANY($2::text[]))\n GROUP BY r.chunk_id`,\n [chunkIds, queryEntityNames],\n );\n const counts = Object.fromEntries(result.rows.map((r) => [String(r.chunk_id), Number(r.n)]));\n if (!Object.keys(counts).length) return {};\n const mx = Math.max(...Object.values(counts));\n return mx ? Object.fromEntries(Object.entries(counts).map(([cid, n]) => [cid, n / mx])) : {};\n}\n\nasync function computeCommunityScores(\n pool: Pool,\n chunkIds: string[],\n vector: number[],\n opts: {\n sourceIds: string[] | null;\n documentIds?: string[] | null;\n principals: string[] | null;\n useSummaries: boolean;\n },\n): Promise<Record<string, number>> {\n if (!opts.useSummaries || !chunkIds.length || !vector) return {};\n const commRows = await pool.query(\n `SELECT entity_ids, 1 - (embedding <=> CAST($1 AS vector))::float AS sim\n FROM context_engine_communities WHERE embedding IS NOT NULL\n ORDER BY embedding <=> CAST($1 AS vector) LIMIT 5`,\n [vecLiteral(vector)],\n );\n if (!commRows.rows.length) return {};\n const entityScore: Record<string, number> = {};\n for (const row of commRows.rows) {\n const sim = Number(row.sim);\n if (sim < 0.15) continue;\n for (const eid of row.entity_ids ?? []) {\n entityScore[String(eid)] = Math.max(entityScore[String(eid)] ?? 0, sim);\n }\n }\n if (!Object.keys(entityScore).length) return {};\n const result = await pool.query(\n `SELECT ce.chunk_id::text, ce.entity_id::text\n FROM context_engine_chunk_entities ce\n JOIN context_engine_chunks c ON c.id = ce.chunk_id\n WHERE ce.chunk_id = ANY($1::uuid[]) AND ce.entity_id = ANY($5::uuid[]) ${SCOPE}`,\n [chunkIds, opts.sourceIds, opts.documentIds ?? null, opts.principals, Object.keys(entityScore)],\n );\n const chunkScores: Record<string, number> = {};\n for (const row of result.rows) {\n chunkScores[row.chunk_id] = Math.max(chunkScores[row.chunk_id] ?? 0, entityScore[row.entity_id] ?? 0);\n }\n return chunkScores;\n}\n\n/** ACL-scoped vector seed search → top chunk ids (best first). Exported for\n * tests.\n *\n * `documentIds` participates at GENERATION time, not only in the caller's\n * post-filter: with a narrow document scope inside a large source,\n * off-document seeds would fill the whole SEED_LIMIT budget, the post-filter\n * would empty the leg, and the search would silently degrade to hybrid — the\n * post-filter recall-collapse class the ANN/ACL leg already fixed.\n *\n * `backend` is the chunk-plane backend the caller already owns, and it is here\n * for one reason: this is an ANN scan with `SCOPE` applied ON TOP of the index\n * walk, the identical shape `PostgresBackend.tuneAnnScan` corrects for the\n * vector leg. Untuned, the walk keeps at most `hnsw.ef_search` (40) candidates\n * before the filter runs, so a principal with a small visible slice gets few or\n * zero seeds and the whole graph leg quietly degrades to hybrid. The tuning is\n * `SET LOCAL`, hence the explicit client and transaction: it has to be the\n * connection the seed query itself runs on. A backend without the knobs\n * (non-Postgres, or none passed) is skipped.\n */\nexport async function vectorSeedIds(\n pool: Pool,\n vector: number[],\n opts: {\n sourceIds: string[] | null;\n documentIds?: string[] | null;\n principals: string[] | null;\n backend?: Partial<AnnTunable> | null;\n },\n): Promise<string[]> {\n const binds = [opts.sourceIds, opts.documentIds ?? null, opts.principals];\n // Same rule the vector leg derives `scoped` by, read off the ONE bind array\n // so a new scope dimension cannot be silently left out of it.\n const scoped = binds.some((v) => v !== null);\n const client = await pool.connect();\n try {\n await client.query(\"BEGIN\");\n await opts.backend?.tuneAnnScan?.(client, SEED_LIMIT, scoped);\n const result = await client.query(\n `SELECT c.id::text FROM context_engine_chunks c\n WHERE c.embedding IS NOT NULL ${SCOPE}\n ORDER BY c.embedding <=> CAST($1 AS vector) LIMIT $5`,\n [vecLiteral(vector), ...binds, SEED_LIMIT],\n );\n await client.query(\"COMMIT\");\n return result.rows.map((r) => String(r.id));\n } catch (err) {\n try {\n await client.query(\"ROLLBACK\");\n } catch {\n /* ignore */\n }\n throw err;\n } finally {\n client.release();\n }\n}\n\nexport async function buildGraphRanked(\n query: string,\n opts: {\n config: ContextEngineConfig;\n pool: Pool;\n embedder: Embedder;\n graphStore: GraphStore;\n hooks?: Hooks | null;\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n principals?: string[] | null;\n maxDepth?: number;\n backend?: Partial<AnnTunable> | null;\n },\n): Promise<string[]> {\n try {\n const result = await opts.embedder.embed([query], { kind: \"query\" });\n const vectors = Array.isArray(result)\n ? (result[0] as number[][])\n : ((result as { vectors?: number[][] }).vectors ?? []);\n const vector = vectors[0] ? [...vectors[0]] : null;\n if (!vector) return [];\n const sourceIds = opts.sourceIds ?? null;\n const documentIds = opts.documentIds ?? null;\n const principals = opts.principals ?? null;\n const seeds = await vectorSeedIds(opts.pool, vector, {\n sourceIds,\n documentIds,\n principals,\n backend: opts.backend,\n });\n if (!seeds.length) return [];\n const queryEntities = await deriveQueryEntities(opts.pool, seeds.slice(0, TOP_SEEDS_FOR_ENTITIES), {\n sourceIds,\n documentIds,\n principals,\n });\n const entityNorms = queryEntities.map((e) => String(e.normalized_name));\n\n let expanded: string[] = seeds;\n let connectivity: Record<string, number> = {};\n try {\n await opts.graphStore.connect();\n expanded = await opts.graphStore.expandChunkSet(seeds, entityNorms, opts.maxDepth ?? 2);\n connectivity = await opts.graphStore.getChunkConnectivityScores(expanded.length ? expanded : seeds);\n } catch (exc) {\n if (opts.hooks) emitError(opts.hooks, exc, { stage: \"graph_expand\" });\n else console.warn(\"graph leg: expansion failed:\", exc);\n expanded = seeds;\n connectivity = {};\n }\n const allIds = expanded.length ? expanded : seeds;\n const useSummaries = await shouldUseCommunitySummaries(opts.pool, { sourceIds, principals });\n const vecScores = await computeVectorScores(opts.pool, allIds, vector);\n const relScores = await computeRelationshipScores(opts.pool, allIds, entityNorms);\n const commScores = await computeCommunityScores(opts.pool, allIds, vector, {\n sourceIds,\n documentIds,\n principals,\n useSummaries,\n });\n const textRows = await opts.pool.query(\n `SELECT id::text, lower(coalesce(text,'')) AS t FROM context_engine_chunks WHERE id = ANY($1::uuid[])`,\n [allIds],\n );\n const texts = Object.fromEntries(textRows.rows.map((r) => [String(r.id), String(r.t)]));\n const weights = {\n ...DEFAULT_WEIGHTS,\n ...((\n opts.config.graph as {\n rerankWeights?: Record<string, number>;\n rerank_weights?: Record<string, number>;\n }\n ).rerankWeights ??\n (opts.config.graph as { rerank_weights?: Record<string, number> }).rerank_weights ??\n {}),\n };\n const entityNameSet = new Set(entityNorms);\n const scored: Array<[string, number]> = [];\n for (const cidRaw of allIds) {\n const cid = String(cidRaw);\n if (!(cid in texts)) continue;\n const v = vecScores[cid] ?? 0.5;\n const chunkText = texts[cid] ?? \"\";\n const matches = [...entityNameSet].filter((n) => n && chunkText.includes(n)).length;\n const e = entityNameSet.size ? Math.min(1.0, matches / Math.max(1, entityNameSet.size)) : 0;\n const r = relScores[cid] ?? 0;\n const cm = commScores[cid] ?? 0;\n const gc = connectivity[cid] ?? 0.5;\n const final =\n v * weights.vector_score +\n e * weights.entity_match +\n r * weights.relationship_relevance +\n cm * weights.community_match +\n gc * weights.graph_connectivity;\n scored.push([cid, final]);\n }\n scored.sort((a, b) => b[1] - a[1]);\n return scored.map(([cid]) => cid);\n } catch (exc) {\n if (opts.hooks) emitError(opts.hooks, exc, { stage: \"graph_embed_query\" });\n else console.warn(\"graph leg: query embed failed:\", exc);\n return [];\n }\n}\n","/**\n * Graph NAVIGATION — walking the entity graph, instead of ranking chunks by it.\n *\n * `retrieval.ts` uses the graph to ORDER chunks: entities and communities\n * become scores in an RRF leg and the caller never sees them. Navigation is\n * the other question — *what is connected to what* — and answers with entity\n * names, relationship labels and the evidence sentence behind each edge.\n *\n * **It reads the Postgres mirror, never Neo4j.** Not an optimisation: the\n * mirror is where access control lives. A chunk carries the `acl`, and the\n * scope predicate is the one thing that enforces it across every leg. The\n * Neo4j copy has no ACL data at all, so traversing it would mean\n * re-implementing visibility in Cypher — a second enforcement point that can\n * only drift from the first.\n *\n * **The visibility rule is the EDGE, not the node.** `evidence` is a sentence\n * quoted from the chunk a relationship was extracted from, so returning an\n * edge whose evidence chunk is hidden hands the caller that chunk's text.\n * Every query joins the relationship to its evidence chunk and scopes it. An\n * edge with no evidence chunk has no provenance to check, so it is withheld\n * from an access-controlled caller and shown only to a trusted one.\n *\n * A start entity that resolves to nothing and one hidden behind an ACL give\n * the SAME answer, for the reason `getDocument` gives an absent and a\n * forbidden document one message.\n */\n\nimport type { Pool } from \"pg\";\nimport { shouldUseCommunitySummaries } from \"./retrieval.js\";\n\n/**\n * The scope predicate, numbered for the queries below. Textually a copy of\n * storage.ts's SCOPE — the canonical one — because pg binds are POSITIONAL\n * and these queries number their placeholders differently. Any change to the\n * canonical predicate must be made here too. Never widen it.\n */\nconst SCOPE = `\n AND ($1::text[] IS NULL OR c.source_id = ANY($1::text[]))\n AND ($2::uuid[] IS NULL OR c.document_id = ANY($2::uuid[]))\n AND ($3::text[] IS NULL OR c.acl IS NULL OR c.acl && $3::text[])\n`;\n\n/** An edge is visible when its evidence chunk is. No chunk, no provenance. */\nconst EDGE_VISIBLE = `\n AND (\n ($3::text[] IS NULL AND r.chunk_id IS NULL)\n OR EXISTS (\n SELECT 1 FROM context_engine_chunks c\n WHERE c.id = r.chunk_id ${SCOPE}\n )\n )\n`;\n\n/** Rule 2 for the entity a walk STARTS from. */\nconst ENTITY_VISIBLE = `\n EXISTS (\n SELECT 1 FROM context_engine_chunk_entities ce\n JOIN context_engine_chunks c ON c.id = ce.chunk_id\n WHERE ce.entity_id = e.id ${SCOPE}\n )\n`;\n\nconst NOT_FOUND = \"entity not found\";\nconst MAX_DEPTH = 5;\nconst MAX_LIMIT = 200;\nconst MIN_COMMUNITY_RELEVANCE = 0.25;\n\n/** Rule 3, said to the model rather than returned as a mysterious empty list. */\nexport const SUMMARIES_WITHHELD =\n \"community summaries are withheld: this corpus mixes access-controlled and \" +\n \"open documents, and a community summary is computed over the whole corpus, \" +\n \"so it cannot be shown to a caller who can only see part of it. Use search, \" +\n \"traverse or get_neighbors instead — those are filtered per document.\";\n\nexport type NavScope = {\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n principals?: string[] | null;\n};\n\nfunction clamp(value: number | null | undefined, fallback: number, ceiling: number): number {\n const n = typeof value === \"number\" && Number.isFinite(value) ? Math.trunc(value) : fallback;\n return Math.max(1, Math.min(n, ceiling));\n}\n\nfunction scopeArgs(opts: NavScope): [string[] | null, string[] | null, string[] | null] {\n return [opts.sourceIds ?? null, opts.documentIds ?? null, opts.principals ?? null];\n}\n\nasync function resolveEntity(\n pool: Pool,\n name: string,\n opts: NavScope,\n): Promise<{ id: string; name: string; type: string } | null> {\n const result = await pool.query(\n `SELECT e.id::text, e.name, e.type FROM context_engine_entities e\n WHERE e.normalized_name = $4 AND ${ENTITY_VISIBLE} LIMIT 1`,\n [...scopeArgs(opts), (name ?? \"\").trim().toLowerCase()],\n );\n const row = result.rows[0];\n return row ? { id: String(row.id), name: row.name, type: row.type } : null;\n}\n\n/** Everything one hop from `entity`, each with the direction of its edge. */\nexport async function getNeighbors(\n pool: Pool,\n opts: NavScope & { entity: string; limit?: number | null },\n): Promise<Record<string, unknown>> {\n const start = await resolveEntity(pool, opts.entity, opts);\n if (!start) return { found: false, error: NOT_FOUND, neighbors: [], count: 0 };\n const result = await pool.query(\n `SELECT other.name, other.type, r.category, r.label, r.evidence,\n CASE WHEN r.source_entity_id = $4::uuid THEN 'outgoing' ELSE 'incoming' END AS direction\n FROM context_engine_entity_relationships r\n JOIN context_engine_entities other ON other.id =\n CASE WHEN r.source_entity_id = $4::uuid THEN r.target_entity_id ELSE r.source_entity_id END\n WHERE (r.source_entity_id = $4::uuid OR r.target_entity_id = $4::uuid)\n ${EDGE_VISIBLE}\n ORDER BY other.name LIMIT $5`,\n [...scopeArgs(opts), start.id, clamp(opts.limit, 20, MAX_LIMIT)],\n );\n const neighbors = result.rows.map((r) => ({\n name: r.name,\n type: r.type,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n direction: r.direction,\n }));\n return {\n found: true,\n entity: { name: start.name, type: start.type },\n neighbors,\n count: neighbors.length,\n };\n}\n\n/** Every entity reachable from `entity` within `depth` hops. */\nexport async function traverse(\n pool: Pool,\n opts: NavScope & { entity: string; depth?: number | null; category?: string | null; limit?: number | null },\n): Promise<Record<string, unknown>> {\n const start = await resolveEntity(pool, opts.entity, opts);\n if (!start) return { found: false, error: NOT_FOUND, paths: [], count: 0 };\n // Undirected — a relationship is a connection whichever way it was written —\n // and `visited` carries the ids already on this path so a cycle cannot loop.\n const result = await pool.query(\n `WITH RECURSIVE walk AS (\n SELECT r.id AS rel_id,\n CASE WHEN r.source_entity_id = $4::uuid THEN r.target_entity_id ELSE r.source_entity_id END AS node_id,\n r.category, r.label, r.evidence, 1 AS hop,\n ARRAY[r.source_entity_id, r.target_entity_id] AS visited\n FROM context_engine_entity_relationships r\n WHERE (r.source_entity_id = $4::uuid OR r.target_entity_id = $4::uuid)\n ${EDGE_VISIBLE}\n UNION ALL\n SELECT r.id,\n CASE WHEN r.source_entity_id = w.node_id THEN r.target_entity_id ELSE r.source_entity_id END,\n r.category, r.label, r.evidence, w.hop + 1,\n w.visited || CASE WHEN r.source_entity_id = w.node_id THEN r.target_entity_id ELSE r.source_entity_id END\n FROM walk w\n JOIN context_engine_entity_relationships r\n ON (r.source_entity_id = w.node_id OR r.target_entity_id = w.node_id)\n WHERE w.hop < $5\n AND NOT (CASE WHEN r.source_entity_id = w.node_id THEN r.target_entity_id ELSE r.source_entity_id END = ANY(w.visited))\n ${EDGE_VISIBLE}\n )\n SELECT DISTINCT ON (w.node_id, w.category, w.label)\n e.name, e.type, w.category, w.label, w.evidence, w.hop\n FROM walk w JOIN context_engine_entities e ON e.id = w.node_id\n WHERE ($6::text IS NULL OR w.category = $6::text)\n ORDER BY w.node_id, w.category, w.label, w.hop ASC\n LIMIT $7`,\n [\n ...scopeArgs(opts),\n start.id,\n clamp(opts.depth, 2, MAX_DEPTH),\n opts.category ?? null,\n clamp(opts.limit, 50, MAX_LIMIT),\n ],\n );\n const paths = result.rows.map((r) => ({\n target: r.name,\n target_type: r.type,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n hops: r.hop,\n }));\n paths.sort((a, b) => a.hops - b.hops || String(a.target).localeCompare(String(b.target)));\n return { found: true, entity: { name: start.name, type: start.type }, paths, count: paths.length };\n}\n\n/** Relationships of a given kind, without naming a start entity. */\nexport async function findRelated(\n pool: Pool,\n opts: NavScope & {\n category?: string | null;\n label?: string | null;\n entityType?: string | null;\n limit?: number | null;\n },\n): Promise<Record<string, unknown>> {\n const result = await pool.query(\n `SELECT src.name AS src_name, src.type AS src_type, tgt.name AS tgt_name, tgt.type AS tgt_type,\n r.category, r.label, r.evidence\n FROM context_engine_entity_relationships r\n JOIN context_engine_entities src ON src.id = r.source_entity_id\n JOIN context_engine_entities tgt ON tgt.id = r.target_entity_id\n WHERE ($4::text IS NULL OR r.category = $4::text)\n AND ($5::text IS NULL OR r.label = $5::text)\n AND ($6::text IS NULL OR src.type = $6::text OR tgt.type = $6::text)\n ${EDGE_VISIBLE}\n ORDER BY src.name, tgt.name LIMIT $7`,\n [\n ...scopeArgs(opts),\n opts.category ?? null,\n opts.label ?? null,\n opts.entityType ?? null,\n clamp(opts.limit, 50, MAX_LIMIT),\n ],\n );\n const relationships = result.rows.map((r) => ({\n source: r.src_name,\n source_type: r.src_type,\n target: r.tgt_name,\n target_type: r.tgt_type,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n }));\n return { found: true, relationships, count: relationships.length };\n}\n\n/**\n * The themes of the corpus: LLM-written summaries of entity communities,\n * ranked against `query`. Withheld entirely from an access-controlled caller\n * over a corpus that mixes open and restricted documents (rule 3).\n *\n * `documentIds` is deliberately not honoured: a community spans the corpus, so\n * narrowing it to some documents would describe a thing never computed.\n */\nexport async function communitySummary(\n pool: Pool,\n opts: {\n embedder: { embed: (texts: string[], kind?: string) => Promise<number[][]> };\n query: string;\n limit?: number | null;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n },\n): Promise<Record<string, unknown>> {\n const allowed = await shouldUseCommunitySummaries(pool, {\n sourceIds: opts.sourceIds ?? null,\n principals: opts.principals ?? null,\n });\n if (!allowed) {\n return { available: false, error: SUMMARIES_WITHHELD, communities: [], count: 0 };\n }\n const vectors = await opts.embedder.embed([opts.query ?? \"\"], \"query\");\n const vector = vectors?.[0];\n if (!vector) return { available: true, communities: [], count: 0 };\n const result = await pool.query(\n `SELECT summary, entity_count, relationship_count, level,\n 1 - (embedding <=> CAST($1 AS vector))::float AS sim\n FROM context_engine_communities\n WHERE embedding IS NOT NULL AND summary IS NOT NULL\n ORDER BY embedding <=> CAST($1 AS vector) LIMIT $2`,\n [`[${vector.map((x) => Number(x)).join(\",\")}]`, clamp(opts.limit, 3, 10)],\n );\n const communities = result.rows\n .filter((r) => Number(r.sim) > MIN_COMMUNITY_RELEVANCE)\n .map((r) => ({\n summary: r.summary,\n entity_count: r.entity_count,\n relationship_count: r.relationship_count,\n hierarchy_level: r.level,\n relevance: Math.round(Number(r.sim) * 10000) / 10000,\n }));\n return { available: true, communities, count: communities.length };\n}\n","import { z } from \"zod\";\nimport { ExtraMissingError } from \"./errors.js\";\nimport { requireExtra } from \"./extras.js\";\nimport { ENTITY_TYPES, RELATIONSHIP_CATEGORY_VALUES } from \"./graph/entities.js\";\nimport {\n callKnowledgeTool,\n INPUT_PROPERTIES,\n KNOWLEDGE_ACTIONS,\n KNOWLEDGE_TOOL_DESCRIPTION,\n type KnowledgeComputeFn,\n type ScopeInput,\n} from \"./knowledge-tool.js\";\nimport { requireValidUuid } from \"./routing-core.js\";\nimport {\n type ApprovalScopeFn,\n type PrincipalsFn,\n registerToolGateway,\n resolvePrincipalsFn,\n SCOPE_NOTE,\n} from \"./tools/mcp-tools.js\";\nimport { __version__ } from \"./version.js\";\n\ntype Engine = {\n config: { llm?: unknown; enableCodeExecution?: boolean };\n search: (query: string, opts?: Record<string, unknown>) => Promise<{ hits: unknown[]; usage?: unknown }>;\n getDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n listDocuments: (opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n queryStructured?: (question: string, opts?: Record<string, unknown>) => Promise<unknown>;\n compute?: (instruction: string, opts?: Record<string, unknown>) => Promise<unknown>;\n searchTools: (query: string, opts?: Record<string, unknown>) => Promise<unknown[]>;\n executeTool: (\n name: string,\n args: Record<string, unknown> | null,\n opts?: Record<string, unknown>,\n ) => Promise<Record<string, unknown>>;\n};\n\n/** The description the shared property table gives this parameter. */\nfunction described(name: string): string {\n return String(INPUT_PROPERTIES[name]?.description ?? \"\");\n}\n\nfunction asMcpResult(data: unknown): { content: Array<{ type: \"text\"; text: string }> } {\n return { content: [{ type: \"text\", text: JSON.stringify(data) }] };\n}\n\nexport async function createMcpApp(\n engine: Engine,\n opts: {\n principals: PrincipalsFn;\n /**\n * REQUIRED ceiling of source ids (or a `Scope`) this mounted tool may\n * ever reach — resolved fresh per call like `principals` and NEVER a tool\n * argument. A model can ask for any source id it likes, and a tool that\n * believed it would let one caller read another's documents. A host with\n * one shared corpus says so on purpose with `scope: UNSCOPED`.\n */\n scope: ScopeInput | (() => ScopeInput | Promise<ScopeInput>);\n /**\n * Overrides `config.redaction` per call, so a per-project or\n * per-customer policy reaches this tool the way it already reaches\n * `search`. Never a tool argument.\n */\n redaction?: unknown | (() => unknown | Promise<unknown>);\n /**\n * Optionally REPLACES the built-in compute action with the host's own\n * callable. Running generated code is where a host has its own rules\n * about permission, billing and approval; supplying one here means it\n * does not have to intercept the action before the tool is reached, and\n * makes the action available whatever `enableCodeExecution` says.\n */\n compute?: KnowledgeComputeFn | null;\n /** The same seam for the other expensive action — one LLM call per document. */\n mapReduce?:\n | ((instruction: string, opts: Record<string, unknown>) => Promise<Record<string, unknown>>)\n | null;\n approvalScope?: ApprovalScopeFn | null;\n },\n): Promise<\n ((req: import(\"node:http\").IncomingMessage, res: import(\"node:http\").ServerResponse) => Promise<void>) & {\n mcp: unknown;\n }\n> {\n if (!opts?.principals) {\n throw new TypeError(\"createMcpApp requires principals\");\n }\n if (opts.scope === undefined || opts.scope === null) {\n throw new TypeError(\n \"createMcpApp requires scope: the source ids this tool may reach, or UNSCOPED \" +\n \"(from @promptev/context-engine) to say the whole corpus on purpose.\",\n );\n }\n\n let McpServer: new (info: {\n name: string;\n version: string;\n }) => {\n tool: (...args: unknown[]) => unknown;\n connect: (t: unknown) => Promise<void>;\n };\n let StreamableHTTPServerTransport: new (opts: {\n sessionIdGenerator?: undefined;\n }) => {\n handleRequest: (req: unknown, res: unknown, body?: unknown) => Promise<void>;\n };\n try {\n const serverMod = await requireExtra<{ McpServer: typeof McpServer }>(\n \"@modelcontextprotocol/sdk/server/mcp.js\",\n \"mcp\",\n \"MCP server\",\n );\n const httpMod = await requireExtra<{\n StreamableHTTPServerTransport: typeof StreamableHTTPServerTransport;\n }>(\"@modelcontextprotocol/sdk/server/streamableHttp.js\", \"mcp\", \"MCP server\");\n McpServer = serverMod.McpServer;\n StreamableHTTPServerTransport = httpMod.StreamableHTTPServerTransport;\n } catch (exc) {\n if (exc instanceof ExtraMissingError) throw exc;\n throw new ExtraMissingError(\"mcp\", \"@modelcontextprotocol/sdk\", \"MCP server\");\n }\n\n const mcp = new McpServer({ name: \"context-engine\", version: __version__ });\n\n const tool = (\n name: string,\n description: string,\n schema: Record<string, unknown>,\n handler: (args: Record<string, unknown>) => Promise<unknown>,\n ) => {\n mcp.tool(name, description, schema, async (args: Record<string, unknown>) =>\n asMcpResult(await handler(args)),\n );\n };\n\n tool(\n \"search_knowledge_base\",\n `${KNOWLEDGE_TOOL_DESCRIPTION}${SCOPE_NOTE}`,\n {\n // Zod RAW SHAPES, not JSON schema: the SDK's isZodRawShape test\n // rejects a plain schema object — on current SDK versions that made\n // registration THROW at startup, and on 1.12.0 the object was consumed\n // as annotations and every handler ran with NO arguments.\n // Every `.describe()` comes from the ONE property table, so the schema\n // a model reads over the protocol and the exported JSON Schema cannot\n // say different things.\n action: z.enum(KNOWLEDGE_ACTIONS).describe(described(\"action\")),\n query: z.string().optional().describe(described(\"query\")),\n document_id: z.string().optional().describe(described(\"document_id\")),\n source_ids: z.array(z.string()).optional().describe(described(\"source_ids\")),\n document_ids: z.array(z.string()).optional().describe(described(\"document_ids\")),\n entity: z.string().optional().describe(described(\"entity\")),\n depth: z.number().int().optional().describe(described(\"depth\")),\n category: z.enum(RELATIONSHIP_CATEGORY_VALUES).optional().describe(described(\"category\")),\n label: z.string().optional().describe(described(\"label\")),\n entity_type: z.enum(ENTITY_TYPES).optional().describe(described(\"entity_type\")),\n top_k: z.number().int().optional().describe(described(\"top_k\")),\n mode: z.enum([\"hybrid\", \"graph\"]).optional().describe(described(\"mode\")),\n limit: z.number().int().optional().describe(described(\"limit\")),\n cursor: z\n .union([z.string(), z.record(z.unknown())])\n .optional()\n .describe(described(\"cursor\")),\n start: z.number().int().optional().describe(described(\"start\")),\n end: z.number().int().optional().describe(described(\"end\")),\n max_chars: z.number().int().optional().describe(described(\"max_chars\")),\n },\n async (args) => {\n // Early, mirroring the HTTP handler: a malformed id must surface as a\n // clean tool error, not a Postgres 22P02 from inside every leg.\n const documentIds = args.document_ids as string[] | undefined;\n const documentId = args.document_id as string | undefined;\n for (const did of [...(documentIds ?? []), ...(documentId ? [documentId] : [])]) {\n requireValidUuid(did);\n }\n // Everything below is `knowledge-tool`, the same function\n // `engine.searchKnowledgeBase` calls — this layer only turns an MCP\n // request into a caller identity and a scope ceiling.\n const callerPrincipals = await resolvePrincipalsFn(opts.principals);\n const ceiling = typeof opts.scope === \"function\" ? await opts.scope() : opts.scope;\n const policy = typeof opts.redaction === \"function\" ? await opts.redaction() : opts.redaction;\n return callKnowledgeTool(engine as never, {\n ...(args as Record<string, unknown>),\n action: String(args.action),\n principals: callerPrincipals,\n scope: ceiling as never,\n redaction: policy,\n compute: opts.compute ?? null,\n map_reduce: opts.mapReduce ?? null,\n });\n },\n );\n\n registerToolGateway(\n {\n tool: (name, description, schema, handler) => {\n tool(name, description, schema, async (args) => handler(args as never));\n },\n },\n engine,\n opts.principals,\n opts.approvalScope ?? null,\n );\n\n const handler = (async (\n req: import(\"node:http\").IncomingMessage,\n res: import(\"node:http\").ServerResponse,\n ) => {\n const transport = new StreamableHTTPServerTransport({ sessionIdGenerator: undefined });\n await mcp.connect(transport);\n await transport.handleRequest(req, res);\n }) as ((\n req: import(\"node:http\").IncomingMessage,\n res: import(\"node:http\").ServerResponse,\n ) => Promise<void>) & { mcp: unknown };\n handler.mcp = mcp;\n return handler;\n}\n","import { randomUUID } from \"node:crypto\";\nimport { jsonrepair } from \"jsonrepair\";\nimport type { Pool } from \"pg\";\nimport type { ContextEngineConfig } from \"../config.js\";\nimport type { GraphStore } from \"./neo4j-client.js\";\n\nexport const ENTITY_TYPES = [\n \"PERSON\",\n \"ORG\",\n \"PRODUCT\",\n \"LOCATION\",\n \"REFERENCE\",\n \"TEMPORAL\",\n \"CONCEPT\",\n] as const;\nexport type EntityType = (typeof ENTITY_TYPES)[number];\n\n/**\n * Ordered so a schema can publish it (a Set has no order to publish). The\n * prompt below is built from it, so the enum a caller is steered toward and\n * the values a row can actually hold are one list.\n */\nexport const RELATIONSHIP_CATEGORY_VALUES = [\n \"HIERARCHICAL\",\n \"MEMBERSHIP\",\n \"CREATION\",\n \"TEMPORAL\",\n \"SPATIAL\",\n \"REFERENCE\",\n \"FUNCTIONAL\",\n \"QUANTITATIVE\",\n] as const;\n\nexport const RELATIONSHIP_CATEGORIES = new Set<string>(RELATIONSHIP_CATEGORY_VALUES);\n\nexport const CHUNKS_PER_BATCH = 11;\n\nconst EXTRACTION_SYSTEM = \"You are a STRICT JSON entity and relationship extraction engine.\";\n\nexport function normalizeEntityName(name: string): string {\n return (name || \"\").toLowerCase().trim().replace(/ {2}/g, \" \");\n}\n\nfunction sharedSignificantTokens(a: string, b: string, minLen = 4): boolean {\n const tokensA = new Set(a.split(/\\s+/).filter((t) => t.length >= minLen));\n const tokensB = b.split(/\\s+/).filter((t) => t.length >= minLen);\n return tokensB.some((t) => tokensA.has(t));\n}\n\n/** Exported so a test can hold the schema enums and this prompt to one list. */\nexport function buildPrompt(fullText: string): string {\n return `Extract ALL entities and relationships from the following document text.\n\nDOCUMENT TEXT:\n${fullText}\n\nENTITY TYPES: ${ENTITY_TYPES.join(\", \")}\n\nRELATIONSHIP CATEGORIES: ${RELATIONSHIP_CATEGORY_VALUES.join(\", \")}\n\nOUTPUT FORMAT (STRICT JSON):\n{\n \"entities\": [\n {\"name\": \"exact text\", \"type\": \"PERSON|ORG|PRODUCT|LOCATION|REFERENCE|TEMPORAL|CONCEPT\"}\n ],\n \"relationships\": [\n {\"source\": \"entity name\", \"target\": \"entity name\", \"category\": \"CATEGORY\", \"label\": \"verb\", \"evidence\": \"brief quote\"}\n ]\n}\n\nRULES:\n- Extract EVERY meaningful entity (people, organizations, products, locations, codes, dates, concepts)\n- Choose the MOST SPECIFIC type; only use CONCEPT when no other type fits\n- Be exhaustive but do NOT invent entities not present in the text\n- Deduplicate: each entity appears once\n- Relationships: source/target MUST be entities you extracted; category MUST be one of the 8\n- Return ONLY the JSON object\n`;\n}\n\nfunction safeJsonParse(text: string, expected: \"object\" | \"array\"): unknown {\n const stripped = text.trim();\n try {\n const result = JSON.parse(stripped);\n if (expected === \"object\" && result && typeof result === \"object\" && !Array.isArray(result))\n return result;\n if (expected === \"array\" && Array.isArray(result)) return result;\n } catch {\n /* repair */\n }\n const repaired = JSON.parse(jsonrepair(stripped));\n return repaired;\n}\n\ntype Querier = { query: Pool[\"query\"] };\n\nexport async function resolveEntity(\n db: Querier,\n _name: string,\n normalizedName: string,\n entityType: string,\n): Promise<Record<string, unknown> | null> {\n const exact = await db.query(`SELECT * FROM context_engine_entities WHERE normalized_name = $1 LIMIT 1`, [\n normalizedName,\n ]);\n if (exact.rows[0]) return exact.rows[0];\n\n const trigram = await db.query(\n `SELECT id FROM context_engine_entities\n WHERE similarity(normalized_name, $1) > 0.75\n ORDER BY similarity(normalized_name, $1) DESC LIMIT 1`,\n [normalizedName],\n );\n if (trigram.rows[0]) {\n const row = await db.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [trigram.rows[0].id]);\n return row.rows[0] ?? null;\n }\n\n const typed = await db.query(\n `SELECT id, normalized_name FROM context_engine_entities\n WHERE type = $2 AND similarity(normalized_name, $1) > 0.45\n ORDER BY similarity(normalized_name, $1) DESC LIMIT 5`,\n [normalizedName, entityType],\n );\n for (const row of typed.rows) {\n if (sharedSignificantTokens(normalizedName, String(row.normalized_name))) {\n const full = await db.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [row.id]);\n return full.rows[0] ?? null;\n }\n }\n return null;\n}\n\nasync function extractFromText(\n fullText: string,\n config: ContextEngineConfig,\n): Promise<{\n entities: Array<Record<string, unknown>>;\n relationships: Array<Record<string, unknown>>;\n tokens: Record<string, number>;\n}> {\n try {\n const llm = config.graph.extractionLlm;\n if (!llm) return { entities: [], relationships: [], tokens: {} };\n const { callLlm } = await import(\"../providers/llm.js\");\n const out = await callLlm(llm, {\n system: EXTRACTION_SYSTEM,\n user: buildPrompt(fullText),\n jsonMode: true,\n });\n const raw = Array.isArray(out) ? out[0] : ((out as { text?: string }).text ?? String(out));\n const tokens = (Array.isArray(out) ? out[1] : (out as { tokens?: Record<string, number> }).tokens) ?? {};\n let result: Record<string, unknown> = {};\n try {\n result = (safeJsonParse(String(raw), \"object\") as Record<string, unknown>) ?? {};\n } catch {\n result = {};\n }\n const entities = ((result.entities as unknown[]) ?? []).filter((e): e is Record<string, unknown> =>\n Boolean(e && typeof e === \"object\" && (e as { name?: unknown }).name && (e as { type?: unknown }).type),\n );\n const relationships = ((result.relationships as unknown[]) ?? []).filter(\n (r): r is Record<string, unknown> =>\n Boolean(\n r &&\n typeof r === \"object\" &&\n (r as { source?: unknown }).source &&\n (r as { target?: unknown }).target,\n ),\n );\n return { entities, relationships, tokens: tokens as Record<string, number> };\n } catch (exc) {\n console.error(\"entity extraction LLM call failed:\", exc);\n return { entities: [], relationships: [], tokens: {} };\n }\n}\n\nasync function loadDocumentChunks(pool: Pool, documentId: string): Promise<Array<Record<string, unknown>>> {\n const result = await pool.query(\n `SELECT id, text, idx, source_id, meta_data FROM context_engine_chunks WHERE document_id = $1 ORDER BY idx`,\n [documentId],\n );\n return result.rows.map((c) => ({\n id: c.id,\n text: c.text || \"\",\n text_lower: String(c.text || \"\").toLowerCase(),\n idx: c.idx || 0,\n source_id: c.source_id,\n token_count:\n (c.meta_data as { token_count?: number } | null)?.token_count ??\n Math.floor(String(c.text || \"\").length / 4),\n }));\n}\n\nasync function persistExtraction(\n pool: Pool,\n chunkSnaps: Array<Record<string, unknown>>,\n extracted: { entities: Array<Record<string, unknown>>; relationships: Array<Record<string, unknown>> },\n): Promise<{ entities_created: number; entities_found: number; relationships_created: number }> {\n const stats = { entities_created: 0, entities_found: 0, relationships_created: 0 };\n const byNorm = new Map<string, Record<string, unknown>>();\n const client = await pool.connect();\n try {\n await client.query(\"BEGIN\");\n for (const entityData of extracted.entities) {\n const normalized = normalizeEntityName(String(entityData.name));\n if (!normalized) continue;\n const entityType = String(entityData.type);\n let entity = await resolveEntity(client, String(entityData.name), normalized, entityType);\n if (entity) {\n await client.query(\n `UPDATE context_engine_entities SET frequency = COALESCE(frequency, 1) + 1,\n name = CASE WHEN length($2) > length(name) THEN $2 ELSE name END, updated_at = now()\n WHERE id = $1`,\n [entity.id, entityData.name],\n );\n entity = (await client.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [entity.id]))\n .rows[0];\n } else {\n const id = randomUUID();\n await client.query(\n `INSERT INTO context_engine_entities (id, name, normalized_name, type, frequency)\n VALUES ($1, $2, $3, $4, 1)`,\n [id, entityData.name, normalized, entityType],\n );\n entity = (await client.query(`SELECT * FROM context_engine_entities WHERE id = $1`, [id])).rows[0];\n stats.entities_created += 1;\n }\n byNorm.set(normalized, entity!);\n byNorm.set(String(entity!.normalized_name), entity!);\n const nameLower = String(entityData.name).toLowerCase();\n for (const snap of chunkSnaps) {\n if (nameLower && String(snap.text_lower).includes(nameLower)) {\n const exists = await client.query(\n `SELECT 1 FROM context_engine_chunk_entities WHERE chunk_id = $1 AND entity_id = $2`,\n [snap.id, entity!.id],\n );\n if (!exists.rows.length) {\n await client.query(\n `INSERT INTO context_engine_chunk_entities (chunk_id, entity_id) VALUES ($1, $2)`,\n [snap.id, entity!.id],\n );\n stats.entities_found += 1;\n }\n }\n }\n }\n for (const rel of extracted.relationships) {\n const sourceNorm = normalizeEntityName(String(rel.source ?? \"\"));\n const targetNorm = normalizeEntityName(String(rel.target ?? \"\"));\n const category = String(rel.category ?? \"\").toUpperCase();\n const label = String(rel.label ?? \"\");\n const evidence = String(rel.evidence ?? \"\");\n if (!RELATIONSHIP_CATEGORIES.has(category)) continue;\n let src = byNorm.get(sourceNorm);\n let tgt = byNorm.get(targetNorm);\n if (!src) {\n src = (\n await client.query(`SELECT * FROM context_engine_entities WHERE normalized_name = $1`, [sourceNorm])\n ).rows[0];\n }\n if (!tgt) {\n tgt = (\n await client.query(`SELECT * FROM context_engine_entities WHERE normalized_name = $1`, [targetNorm])\n ).rows[0];\n }\n if (!src || !tgt) continue;\n const exists = await client.query(\n `SELECT 1 FROM context_engine_entity_relationships\n WHERE source_entity_id = $1 AND target_entity_id = $2 AND category = $3 AND label = $4`,\n [src.id, tgt.id, category, label],\n );\n if (!exists.rows.length) {\n await client.query(\n `INSERT INTO context_engine_entity_relationships\n (id, source_entity_id, target_entity_id, category, label, evidence)\n VALUES ($1, $2, $3, $4, $5, $6)`,\n [randomUUID(), src.id, tgt.id, category, label, evidence.slice(0, 200)],\n );\n stats.relationships_created += 1;\n }\n }\n await client.query(\"COMMIT\");\n } catch (e) {\n await client.query(\"ROLLBACK\");\n throw e;\n } finally {\n client.release();\n }\n return stats;\n}\n\nasync function syncToNeo4j(\n pool: Pool,\n documentId: string,\n chunkSnaps: Array<Record<string, unknown>>,\n graphStore: GraphStore,\n): Promise<void> {\n const chunkRows = chunkSnaps.map((s) => ({\n id: String(s.id),\n document_id: String(documentId),\n source_id: s.source_id,\n text_preview: String(s.text).slice(0, 200),\n position: s.idx,\n token_count: s.token_count,\n }));\n try {\n await graphStore.connect();\n const entities = (await pool.query(`SELECT * FROM context_engine_entities`)).rows;\n const entityById = new Map(entities.map((e) => [String(e.id), e]));\n const entityRows = entities.map((e) => ({\n name: e.name,\n normalized_name: e.normalized_name,\n entity_type: e.type,\n }));\n const ids = chunkSnaps.map((s) => s.id);\n const links = ids.length\n ? (\n await pool.query(`SELECT * FROM context_engine_chunk_entities WHERE chunk_id = ANY($1::uuid[])`, [\n ids,\n ])\n ).rows\n : [];\n const mentionRows = links\n .filter((l) => entityById.has(String(l.entity_id)))\n .map((l) => ({\n chunk_id: String(l.chunk_id),\n entity_name: entityById.get(String(l.entity_id))!.normalized_name,\n }));\n const rels = (await pool.query(`SELECT * FROM context_engine_entity_relationships`)).rows;\n const relRows = [];\n for (const r of rels) {\n const src = entityById.get(String(r.source_entity_id));\n const tgt = entityById.get(String(r.target_entity_id));\n if (src && tgt) {\n relRows.push({\n source_name: src.normalized_name,\n target_name: tgt.normalized_name,\n category: r.category,\n label: r.label,\n evidence: r.evidence,\n chunk_id: r.chunk_id ? String(r.chunk_id) : null,\n });\n }\n }\n await graphStore.upsertChunksBatch(chunkRows);\n await graphStore.linkSequentialChunks(String(documentId));\n await graphStore.upsertEntitiesBatch(entityRows);\n await graphStore.linkChunksToEntitiesBatch(mentionRows);\n await graphStore.upsertEntityRelationshipsBatch(relRows);\n } catch (exc) {\n console.warn(\"Neo4j sync failed (continuing on PG mirror):\", exc);\n }\n}\n\nexport async function extractEntitiesForDocument(\n documentId: string,\n opts: {\n config: ContextEngineConfig;\n pool: Pool;\n graphStore?: GraphStore | null;\n hooks?: unknown;\n },\n): Promise<Record<string, unknown>> {\n const chunkSnaps = await loadDocumentChunks(opts.pool, documentId);\n const stats: Record<string, unknown> = {\n chunk_count: chunkSnaps.length,\n entities_created: 0,\n entities_found: 0,\n relationships_created: 0,\n provider_tokens: {},\n };\n if (!chunkSnaps.length) return stats;\n\n const batches = [];\n for (let i = 0; i < chunkSnaps.length; i += CHUNKS_PER_BATCH) {\n batches.push(chunkSnaps.slice(i, i + CHUNKS_PER_BATCH));\n }\n let llmInput = 0;\n let llmOutput = 0;\n for (const batch of batches) {\n const fullText = batch.map((s) => String(s.text)).join(\"\\n\\n\");\n const extracted = await extractFromText(fullText, opts.config);\n llmInput += Number(extracted.tokens.input ?? 0);\n llmOutput += Number(extracted.tokens.output ?? 0);\n const persisted = await persistExtraction(opts.pool, batch, extracted);\n stats.entities_created = Number(stats.entities_created) + persisted.entities_created;\n stats.entities_found = Number(stats.entities_found) + persisted.entities_found;\n stats.relationships_created = Number(stats.relationships_created) + persisted.relationships_created;\n }\n const tokens = stats.provider_tokens as Record<string, number>;\n if (llmInput) tokens.graph_llm_input = llmInput;\n if (llmOutput) tokens.graph_llm_output = llmOutput;\n if (opts.graphStore) await syncToNeo4j(opts.pool, documentId, chunkSnaps, opts.graphStore);\n return stats;\n}\n","import { createHash } from \"node:crypto\";\nimport { closeSync, openSync, readSync } from \"node:fs\";\nimport { detectLanguage, jaccardSim, splitSentences, tokenize } from \"./text.js\";\n\nexport { detectLanguage };\n\nexport const MAX_CHUNK_CHARS = 6000;\n\nconst PAGE_MARKER_RE = /^---\\s*Page\\s+(\\d+)\\s*---$/gm;\nconst PAGE_MARKER_STRIP_RE = /---\\s*Page\\s+\\d+\\s*---\\n?/g;\nconst SHEET_MARKER_RE = /^\\[Sheet:\\s*(.+?)\\]\\s*$/;\nconst HEADING_OR_BULLET_RE = /^\\s*([#*\\-•]|\\d+[.)])\\s+/;\n\nconst PRINTABLE = new Set(\n \"0123456789abcdefghijklmnopqrstuvwxyzABCDEFGHIJKLMNOPQRSTUVWXYZ!\\\"#$%&'()*+,-./:;<=>?@[\\\\]^_`{|}~ \\t\\n\\r\\v\\f\",\n);\n\nconst log = {\n warn: (...args: unknown[]) => console.warn(\"[context-engine]\", ...args),\n debug: (...args: unknown[]) => console.debug(\"[context-engine]\", ...args),\n};\n\nexport function normalizeText(s: string | null | undefined): string {\n return (s ?? \"\")\n .split(/\\s+/)\n .filter((p) => p.length > 0)\n .join(\" \");\n}\n\nexport function sha256Bytes(s: string): Buffer {\n return createHash(\"sha256\").update(normalizeText(s), \"utf8\").digest();\n}\n\nexport function calculateContentHash(\n opts: { docBytes?: Buffer | null; fullText?: string | null; filePath?: string | null } = {},\n): string | null {\n const { docBytes = null, fullText = null, filePath = null } = opts;\n if (filePath) {\n const h = createHash(\"md5\");\n const fd = openSync(filePath, \"r\");\n try {\n const buf = Buffer.alloc(8192);\n let n = readSync(fd, buf, 0, 8192, null);\n while (n > 0) {\n h.update(buf.subarray(0, n));\n n = readSync(fd, buf, 0, 8192, null);\n }\n } finally {\n closeSync(fd);\n }\n return h.digest(\"hex\");\n }\n if (docBytes) return createHash(\"md5\").update(docBytes).digest(\"hex\");\n if (fullText) return createHash(\"md5\").update(fullText, \"utf8\").digest(\"hex\");\n return null;\n}\n\nexport function sanitizeText(text: string): string {\n if (!text) return text;\n text = text.replaceAll(\"\\x00\", \"\");\n const allowedControl = new Set([\"\\n\", \"\\r\", \"\\t\"]);\n let out = \"\";\n for (const c of text) {\n if (allowedControl.has(c) || PRINTABLE.has(c) || c.codePointAt(0)! >= 32) out += c;\n }\n return out;\n}\n\nexport function normLine(s: string): string {\n s = (s || \"\").replace(/\\s+/g, \" \").trim();\n s = s.replace(/\\s*,\\s*/g, \", \");\n return s;\n}\n\nexport function dedupLines(text: string, minLen = 24): string {\n const seen = new Set<string>();\n const out: string[] = [];\n for (const raw of (text || \"\").split(/\\r\\n|\\n|\\r/)) {\n const line = normLine(raw);\n if (!line) continue;\n if (line.length >= minLen) {\n const h = createHash(\"sha1\").update(line, \"utf8\").digest(\"hex\");\n if (seen.has(h)) continue;\n seen.add(h);\n }\n out.push(line);\n }\n return out.join(\"\\n\");\n}\n\nexport function looksMostlyBoilerplate(_text: string): boolean {\n return false;\n}\n\nexport function inferMime(\n content: string | null | undefined,\n filename: string | null | undefined,\n): string | null {\n if (filename) {\n const fn = filename.toLowerCase();\n if (fn.endsWith(\".csv\")) return \"text/csv\";\n if (fn.endsWith(\".json\")) return \"application/json\";\n if (fn.endsWith(\".md\") || fn.endsWith(\".markdown\")) return \"text/markdown\";\n if (fn.endsWith(\".xlsx\") || fn.endsWith(\".xls\")) {\n return \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\";\n }\n if (fn.endsWith(\".doc\")) return \"application/msword\";\n if (fn.endsWith(\".rtf\")) return \"application/rtf\";\n if (fn.endsWith(\".html\") || fn.endsWith(\".htm\")) return \"text/html\";\n if (fn.endsWith(\".ppt\")) return \"application/vnd.ms-powerpoint\";\n if (fn.endsWith(\".eml\")) return \"message/rfc822\";\n if (fn.endsWith(\".msg\")) return \"application/vnd.ms-outlook\";\n if (fn.endsWith(\".odt\")) return \"application/vnd.oasis.opendocument.text\";\n if (fn.endsWith(\".ods\")) return \"application/vnd.oasis.opendocument.spreadsheet\";\n if (fn.endsWith(\".odp\")) return \"application/vnd.oasis.opendocument.presentation\";\n if (fn.endsWith(\".epub\")) return \"application/epub+zip\";\n }\n\n if (content) {\n const trimmed = content.trimStart().slice(0, 500);\n\n if (trimmed.startsWith(\"{\") || trimmed.startsWith(\"[\")) {\n try {\n JSON.parse(content);\n return \"application/json\";\n } catch {\n // not valid JSON\n }\n }\n\n if (\n trimmed.startsWith(\"#\") ||\n trimmed.slice(0, 1000).includes(\"\\n## \") ||\n trimmed.slice(0, 500).includes(\"\\n# \")\n ) {\n return \"text/markdown\";\n }\n\n const lines = trimmed.split(\"\\n\").slice(0, 5);\n if (lines.length >= 2) {\n const commaCounts = lines.filter((line) => line.trim()).map((line) => (line.match(/,/g) || []).length);\n if (commaCounts.length >= 2 && new Set(commaCounts).size === 1 && commaCounts[0]! >= 2) {\n return \"text/csv\";\n }\n }\n }\n\n return null;\n}\n\nexport function isTabularMime(mime: string | null | undefined): boolean {\n if (!mime) return false;\n const m = mime.toLowerCase();\n return [\"csv\", \"sheet\", \"excel\", \"spreadsheetml\", \"tab-separated\"].some((k) => m.includes(k));\n}\n\nexport function isMarkdownMime(mime: string | null | undefined): boolean {\n if (!mime) return false;\n const m = mime.toLowerCase();\n return m.includes(\"markdown\") || m === \"text/md\";\n}\n\nexport function isJsonMime(mime: string | null | undefined): boolean {\n if (!mime) return false;\n return mime.toLowerCase().includes(\"json\");\n}\n\nfunction csvNeedsQuote(field: string): boolean {\n return /[\",\\r\\n]/.test(field);\n}\n\nfunction csvEscapeField(field: string): string {\n if (csvNeedsQuote(field)) return `\"${field.replaceAll('\"', '\"\"')}\"`;\n return field;\n}\n\nfunction csvWrite(rows: string[][]): string {\n return `${rows.map((row) => row.map(csvEscapeField).join(\",\")).join(\"\\r\\n\")}\\r\\n`;\n}\n\nexport function parseCsv(text: string): string[][] {\n const rows: string[][] = [];\n let row: string[] = [];\n let field = \"\";\n let i = 0;\n let inQuotes = false;\n while (i < text.length) {\n const ch = text[i]!;\n if (inQuotes) {\n if (ch === '\"') {\n if (text[i + 1] === '\"') {\n field += '\"';\n i += 2;\n continue;\n }\n inQuotes = false;\n i += 1;\n continue;\n }\n field += ch;\n i += 1;\n continue;\n }\n if (ch === '\"') {\n inQuotes = true;\n i += 1;\n continue;\n }\n if (ch === \",\") {\n row.push(field);\n field = \"\";\n i += 1;\n continue;\n }\n if (ch === \"\\r\") {\n i += 1;\n continue;\n }\n if (ch === \"\\n\") {\n row.push(field);\n rows.push(row);\n row = [];\n field = \"\";\n i += 1;\n continue;\n }\n field += ch;\n i += 1;\n }\n if (inQuotes || field.length > 0 || row.length > 0) {\n row.push(field);\n rows.push(row);\n }\n return rows;\n}\n\nfunction jsonDumps(value: unknown): string {\n return JSON.stringify(value, null, 2).replace(/[\\u007f-\\uffff]/g, (ch) => {\n return `\\\\u${ch.charCodeAt(0).toString(16).padStart(4, \"0\")}`;\n });\n}\n\nexport type ChunkMeta = Record<string, unknown>;\n\nfunction splitSheetBlocks(rawText: string): Array<[string | null, string[]]> {\n const lines = (rawText || \"\").split(/\\r\\n|\\n|\\r/);\n const blocks: Array<[string | null, string[]]> = [];\n let currentTitle: string | null = null;\n let currentLines: string[] = [];\n\n for (const line of lines) {\n const m = SHEET_MARKER_RE.exec(line.trim());\n SHEET_MARKER_RE.lastIndex = 0;\n if (m) {\n if (currentLines.length) {\n blocks.push([currentTitle, currentLines]);\n currentLines = [];\n }\n currentTitle = m[1]!.trim();\n continue;\n }\n if (line.trim()) currentLines.push(line);\n }\n\n if (currentLines.length) blocks.push([currentTitle, currentLines]);\n if (!blocks.length && lines.length) {\n return [[null, lines.filter((ln) => ln.trim())]];\n }\n return blocks;\n}\n\nexport function chunkTabular(text: string, rowsPerChunk = 25): [string[], ChunkMeta[]] | [null, null] {\n try {\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n\n for (const [sheetTitle, sheetLines] of splitSheetBlocks(text)) {\n if (!sheetLines.length) continue;\n\n const rows = parseCsv(sheetLines.join(\"\\n\"));\n\n if (rows.length <= 1) {\n let chunkText = sheetLines.join(\"\\n\").trim();\n if (chunkText) {\n if (sheetTitle) chunkText = `[Sheet: ${sheetTitle}]\\n${chunkText}`;\n chunks.push(chunkText);\n const meta: ChunkMeta = { rows_range: \"all\" };\n if (sheetTitle) meta.sheet_title = sheetTitle;\n chunkMetas.push(meta);\n }\n continue;\n }\n\n const header = rows[0]!;\n const dataRows = rows.slice(1);\n\n let effectiveRowsPerChunk = rowsPerChunk;\n const numCols = header.length;\n if (numCols > 100) effectiveRowsPerChunk = 5;\n else if (numCols > 40) effectiveRowsPerChunk = 10;\n else if (numCols > 20) effectiveRowsPerChunk = 15;\n\n for (let i = 0; i < dataRows.length; i += effectiveRowsPerChunk) {\n const batch = dataRows.slice(i, i + effectiveRowsPerChunk);\n let chunkText = csvWrite([header, ...batch]).trim();\n if (sheetTitle && chunkText) chunkText = `[Sheet: ${sheetTitle}]\\n${chunkText}`;\n\n if (chunkText.length > MAX_CHUNK_CHARS && batch.length > 1) {\n const half = Math.floor(batch.length / 2);\n const subBatches: Array<[string[][], number]> = [\n [batch.slice(0, half), i],\n [batch.slice(half), i + half],\n ];\n for (const [subBatch, subStart] of subBatches) {\n let subChunk = csvWrite([header, ...subBatch]).trim();\n if (sheetTitle && subChunk) subChunk = `[Sheet: ${sheetTitle}]\\n${subChunk}`;\n if (subChunk) {\n chunks.push(subChunk);\n const meta: ChunkMeta = {\n rows_range: `${subStart + 2}-${subStart + subBatch.length + 1}`,\n total_rows: dataRows.length,\n };\n if (sheetTitle) meta.sheet_title = sheetTitle;\n chunkMetas.push(meta);\n }\n }\n } else if (chunkText) {\n chunks.push(chunkText);\n const meta: ChunkMeta = {\n rows_range: `${i + 2}-${i + batch.length + 1}`,\n total_rows: dataRows.length,\n };\n if (sheetTitle) meta.sheet_title = sheetTitle;\n chunkMetas.push(meta);\n }\n }\n }\n\n return chunks.length ? [chunks, chunkMetas] : [[text], [{ rows_range: \"all\" }]];\n } catch (exc) {\n log.warn(\"CSV parsing failed, falling back to generic chunker:\", exc);\n return [null, null];\n }\n}\n\nexport function chunkMarkdown(text: string): [string[], ChunkMeta[]] | [null, null] {\n if (!text?.trim()) return [[], []];\n\n const lines = text.split(\"\\n\");\n\n let docTitle: string | null = null;\n for (const line of lines) {\n if (line.startsWith(\"# \") && !line.startsWith(\"## \")) {\n docTitle = line.slice(2).trim();\n break;\n }\n }\n\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n let currentChunkLines: string[] = [];\n let currentH2: string | null = null;\n let titleConsumed = false;\n\n const flushSection = (sectionFallback: string) => {\n if (!currentChunkLines.length) return;\n let prefix = \"\";\n if (docTitle) prefix = `# ${docTitle}\\n`;\n if (currentH2) prefix += `## ${currentH2}\\n`;\n const chunkText = prefix + currentChunkLines.join(\"\\n\").trim();\n\n if (chunkText.length > MAX_CHUNK_CHARS) {\n const subChunks = splitIntoChunks(chunkText);\n for (const sc of subChunks) {\n chunks.push(sc);\n chunkMetas.push({ section_title: currentH2 || docTitle || sectionFallback });\n }\n } else if (chunkText.trim()) {\n chunks.push(chunkText);\n chunkMetas.push({ section_title: currentH2 || docTitle || sectionFallback });\n }\n };\n\n for (const line of lines) {\n if (!titleConsumed && docTitle !== null && line.startsWith(\"# \") && line.slice(2).trim() === docTitle) {\n titleConsumed = true;\n continue;\n }\n if (line.startsWith(\"## \")) {\n flushSection(\"intro\");\n currentH2 = line.slice(3).trim();\n currentChunkLines = [];\n } else {\n currentChunkLines.push(line);\n }\n }\n\n flushSection(\"final\");\n\n const foundH2Headers = lines.some((line) => line.startsWith(\"## \"));\n if (!foundH2Headers || chunks.length === 0) return [null, null];\n return [chunks, chunkMetas];\n}\n\nexport function chunkJson(text: string, itemsPerChunk = 30): [string[], ChunkMeta[]] | [null, null] {\n try {\n if (text.startsWith(\"\\uFEFF\")) text = text.slice(1);\n const data: unknown = JSON.parse(text);\n\n if (Array.isArray(data)) {\n if (data.length <= itemsPerChunk) {\n return [[text], [{ json_path: \"[*]\", item_count: data.length }]];\n }\n\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n for (let i = 0; i < data.length; i += itemsPerChunk) {\n let batch = data.slice(i, i + itemsPerChunk);\n let chunkText = jsonDumps(batch);\n let actualEnd = i + batch.length;\n\n if (chunkText.length > MAX_CHUNK_CHARS && batch.length > 1) {\n let reducedSize = Math.floor(batch.length / 2);\n while (reducedSize > 0) {\n const smallerBatch = batch.slice(0, reducedSize);\n chunkText = jsonDumps(smallerBatch);\n if (chunkText.length <= MAX_CHUNK_CHARS) {\n batch = smallerBatch;\n actualEnd = i + batch.length;\n break;\n }\n reducedSize = Math.floor(reducedSize / 2);\n }\n }\n\n chunkText = `# Items ${i + 1}-${actualEnd} of ${data.length}\\n${chunkText}`;\n chunks.push(chunkText);\n chunkMetas.push({\n json_path: `[${i}:${actualEnd}]`,\n items_range: `${i + 1}-${actualEnd}`,\n });\n }\n return [chunks, chunkMetas];\n }\n\n if (data !== null && typeof data === \"object\" && !Array.isArray(data)) {\n const obj = data as Record<string, unknown>;\n const keys = Object.keys(obj);\n if (keys.length <= 5) {\n return [[text], [{ json_path: \"{*}\", key_count: keys.length }]];\n }\n\n const chunks: string[] = [];\n const chunkMetas: ChunkMeta[] = [];\n for (const key of keys) {\n const value = obj[key];\n let chunkText = jsonDumps({ [key]: value });\n if (chunkText.length > MAX_CHUNK_CHARS) {\n chunkText = `# Key: ${key} (truncated)\\n${jsonDumps(value).slice(0, MAX_CHUNK_CHARS - 100)}...`;\n }\n chunks.push(chunkText);\n chunkMetas.push({ json_path: `.${key}`, top_level_key: key });\n }\n return [chunks, chunkMetas];\n }\n\n return [[text], [{ json_path: \"\" }]];\n } catch (exc) {\n log.debug(\"JSON parsing failed, falling back to generic chunker:\", exc);\n return [null, null];\n }\n}\n\nexport function assignPageNumbers(text: string, chunks: string[], chunkMetas: ChunkMeta[]): void {\n const pageMap: Array<[number, number]> = [];\n const re = new RegExp(PAGE_MARKER_RE.source, \"gm\");\n for (const m of (text || \"\").matchAll(re)) {\n pageMap.push([m.index, Number(m[1])]);\n }\n if (!pageMap.length || !chunks.length) return;\n\n const lineMarker = /^---\\s*Page\\s+(\\d+)\\s*---$/;\n\n for (let idx = 0; idx < chunks.length; idx++) {\n const chunkText = chunks[idx]!;\n let anchor = \"\";\n for (const line of chunkText.split(\"\\n\")) {\n const stripped = line.trim();\n if (stripped && !lineMarker.test(stripped) && stripped.length > 10) {\n anchor = stripped.slice(0, 60);\n break;\n }\n }\n if (!anchor) anchor = chunkText.trim().slice(0, 60);\n if (!anchor) continue;\n\n const pos = text.indexOf(anchor);\n if (pos < 0) continue;\n\n let lo = 0;\n let hi = pageMap.length - 1;\n let page: number | null = null;\n while (lo <= hi) {\n const mid = (lo + hi) >> 1;\n if (pageMap[mid]![0] <= pos) {\n page = pageMap[mid]![1];\n lo = mid + 1;\n } else {\n hi = mid - 1;\n }\n }\n\n if (page !== null && idx < chunkMetas.length) {\n chunkMetas[idx]!.page = page;\n }\n }\n}\n\nexport function routeToChunker(\n text: string,\n mime: string | null | undefined,\n title: string | null | undefined,\n opts: { isMarkdown?: boolean } = {},\n): [string[], string | null, ChunkMeta[]] {\n const isMarkdown = opts.isMarkdown ?? false;\n const effectiveMime = mime || inferMime(text, title);\n\n let chunks: string[] | null = null;\n let chunkMetas: ChunkMeta[] | null = null;\n\n if (isTabularMime(effectiveMime)) {\n [chunks, chunkMetas] = chunkTabular(text);\n } else if (isJsonMime(effectiveMime)) {\n [chunks, chunkMetas] = chunkJson(text);\n } else if (isMarkdown || isMarkdownMime(effectiveMime)) {\n [chunks, chunkMetas] = chunkMarkdown(text);\n }\n\n if (chunks === null || chunks.length === 0) {\n chunks = splitIntoChunks(text);\n chunkMetas = Array.from({ length: chunks.length }, () => ({}));\n }\n chunkMetas = chunkMetas ?? [];\n\n assignPageNumbers(text, chunks, chunkMetas);\n\n chunks = chunks.map((c) => c.replace(PAGE_MARKER_STRIP_RE, \"\").trim());\n const paired = chunks.map((c, i) => [c, chunkMetas![i]!] as [string, ChunkMeta]).filter(([c]) => c.trim());\n if (paired.length) {\n chunks = paired.map(([c]) => c);\n chunkMetas = paired.map(([, m]) => m);\n } else {\n chunks = [];\n chunkMetas = [];\n }\n\n return [chunks, effectiveMime, chunkMetas];\n}\n\nexport function chunkerName(mime: string | null | undefined): string {\n if (isTabularMime(mime)) return \"tabular\";\n if (isJsonMime(mime)) return \"json\";\n if (isMarkdownMime(mime)) return \"markdown\";\n return \"generic\";\n}\n\ntype TextUnit = [sep: string, text: string, tokens: number];\n\nfunction assembleUnits(unitList: TextUnit[]): string {\n const parts: string[] = [];\n for (let i = 0; i < unitList.length; i++) {\n const [sep, uText] = unitList[i]!;\n if (i > 0) parts.push(sep);\n parts.push(uText);\n }\n return parts.join(\"\").trim();\n}\n\nfunction codePointLength(s: string): number {\n let n = 0;\n for (const _ of s) n += 1;\n return n;\n}\n\nexport function splitIntoChunks(text: string, maxWords = 180, overlap = 18): string[] {\n const units: TextUnit[] = [];\n let pendingBlank = false;\n for (const ln of (text || \"\").split(/\\r\\n|\\n|\\r/)) {\n const stripped = ln.trim();\n if (!stripped) {\n pendingBlank = units.length > 0;\n continue;\n }\n const lineSep = pendingBlank ? \"\\n\\n\" : \"\\n\";\n pendingBlank = false;\n if (HEADING_OR_BULLET_RE.test(ln) || stripped.length <= 120) {\n units.push([lineSep, stripped, tokenize(stripped).length]);\n } else {\n let first = true;\n for (const sent of splitSentences(ln)) {\n const s = sent.trim();\n if (s) {\n units.push([first ? lineSep : \" \", s, tokenize(s).length]);\n first = false;\n }\n }\n }\n }\n\n const normalizedUnits: TextUnit[] = [];\n for (const [uSep, uText, uTokens] of units) {\n if (uTokens <= maxWords) {\n normalizedUnits.push([uSep, uText, uTokens]);\n continue;\n }\n const windowChars = maxWords * 4;\n const uChars = [...uText];\n let pos = 0;\n let pieceSep = uSep;\n while (pos < uChars.length) {\n let pieceChars = uChars.slice(pos, pos + windowChars);\n let piece = pieceChars.join(\"\");\n if (pos + windowChars < uChars.length) {\n const ws = piece.lastIndexOf(\" \");\n if (ws > Math.floor(windowChars / 2)) {\n piece = piece.slice(0, ws);\n pieceChars = [...piece];\n }\n }\n let pieceTokens = tokenize(piece);\n if (pieceTokens.length > maxWords) {\n piece = pieceChars.slice(0, maxWords).join(\"\");\n pieceTokens = tokenize(piece);\n }\n normalizedUnits.push([pieceSep, piece, pieceTokens.length]);\n pos += codePointLength(piece);\n pieceSep = \"\";\n }\n }\n\n if (!normalizedUnits.some(([, , t]) => t)) return [];\n\n const chunks: string[] = [];\n let lastKept: string | null = null;\n let cur: TextUnit[] = [];\n let curTokens = 0;\n\n const flush = () => {\n const chunk = assembleUnits(cur);\n if (chunk && (lastKept === null || jaccardSim(chunk, lastKept) < 0.92)) {\n chunks.push(chunk);\n lastKept = chunk;\n }\n const carried: TextUnit[] = [];\n let carriedTokens = 0;\n for (let i = cur.length - 1; i >= 0; i--) {\n const prev = cur[i]!;\n if (prev[2] === 0 || carriedTokens + prev[2] > overlap) break;\n carried.unshift(prev);\n carriedTokens += prev[2];\n }\n cur = carried;\n curTokens = carriedTokens;\n };\n\n for (const unit of normalizedUnits) {\n if (cur.length && curTokens + unit[2] > maxWords) flush();\n cur.push(unit);\n curTokens += unit[2];\n }\n\n if (cur.length) {\n const chunk = assembleUnits(cur);\n if (chunk && (lastKept === null || jaccardSim(chunk, lastKept) < 0.92)) {\n chunks.push(chunk);\n }\n }\n\n return chunks;\n}\n","/**\n * CPSE action surface — document access, structured query, compute.\n *\n * **ACL/source scoping.** Every read here goes through `scopeDocumentsSql`,\n * the Document-table equivalent of storage's chunk-table predicate:\n * `sourceIds=null` means the whole corpus, a list scopes to it;\n * `principals=null` means a trusted internal caller (no ACL filtering),\n * `principals=[]` means an anonymous caller and matches only `acl IS NULL`\n * documents. Same non-interchangeable pair as `search()` — see that\n * module's docstring.\n *\n * Missing vs forbidden documents raise `DocumentNotFoundError` with the\n * identical message `document not found: ${id}` so a caller can never\n * distinguish \"doesn't exist\" from \"exists but you can't see it\".\n */\nimport type { Pool } from \"pg\";\nimport { isTabularMime, parseCsv } from \"./chunkers.js\";\nimport type { ContextEngineConfig, LLMConfig } from \"./config.js\";\nimport { DocumentNotFoundError, EngineActionError, ExtraMissingError } from \"./errors.js\";\nimport { emitError, type Hooks } from \"./hooks.js\";\nimport type { Embedder } from \"./providers/embeddings.js\";\nimport { callLlm } from \"./providers/llm.js\";\nimport { applyRedaction, type RedactionPolicy } from \"./redaction.js\";\nimport { executeSafeCode } from \"./sandbox.js\";\nimport { inferDataType, normalizeKey, resolveFields } from \"./structured.js\";\nimport { aclVisible } from \"./tools/acl.js\";\n\nexport const MAX_LIST_LIMIT = 200;\nexport const MAX_COMPUTE_TEXT_CHARS = 2_000_000;\nexport const DEFAULT_COMPUTE_TIMEOUT = 30;\nexport const MAX_COMPUTE_DOCUMENTS = 50;\n\n/**\n * `discover` reads the stored text of the spreadsheets on the page it is\n * describing. Same budget `compute()` works to, for the same reason: one\n * question must not pull an unbounded slice of the corpus into memory.\n */\nexport const MAX_SCHEMA_TEXT_CHARS = MAX_COMPUTE_TEXT_CHARS;\n/**\n * Per document type, how many fields `discover` describes in full before the\n * rest are listed by name only. A corpus can learn thousands of keys; a model\n * that has to read all of them to find one has been handed the problem again.\n */\nexport const MAX_FIELDS_PER_TYPE = 40;\n/** Hard ceiling on the grouped field scan, across every type. */\nexport const MAX_FIELD_ROWS = 2000;\n/**\n * How many items ONE list inside a `structure` spells out before the rest is\n * reported as a remainder count. `discover` is the FIRST call of every\n * conversation, so an unbounded list here fills the model's context before it\n * has asked anything — and a model only needs to RECOGNISE the columns and\n * sections it will go on to name, not read all two hundred of them.\n */\nexport const MAX_STRUCTURE_ITEMS = 40;\n/**\n * The largest REMAINDER `discover` will invite a caller to come back for.\n * Past it the filled-in follow-up is withheld and replaced by what to do\n * instead.\n *\n * A model reading a capped response cannot know what the full list costs\n * until it has paid for it, so offering the call is the tool making the\n * decision for it. Withholding the invitation is not withholding the answer:\n * a caller that names the document deliberately still gets everything\n * (`bounded: false`), and that cost is then genuinely its own choice rather\n * than something the payload talked it into. A list this long is also a\n * signal in itself — a table that wide is one to COMPUTE over, never one to\n * read back.\n */\nexport const MAX_INVITED_ITEMS = 400;\n/** Buckets in the type census, and the raw (documentType, mime) rows behind it. */\nexport const MAX_DOCUMENT_TYPES = 30;\nexport const MAX_CENSUS_ROWS = 500;\n\nconst TABULAR_MIME_PATTERNS = [\"%csv%\", \"%sheet%\", \"%excel%\", \"%spreadsheetml%\", \"%tab-separated%\"];\n\nconst UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;\n\nconst WORD_RE = /[^\\W\\d_]+/gu;\n\nconst STOPWORDS = new Set([\n \"the\",\n \"a\",\n \"an\",\n \"is\",\n \"are\",\n \"was\",\n \"were\",\n \"am\",\n \"be\",\n \"been\",\n \"being\",\n \"what\",\n \"whats\",\n \"who\",\n \"whom\",\n \"which\",\n \"how\",\n \"when\",\n \"where\",\n \"why\",\n \"this\",\n \"that\",\n \"these\",\n \"those\",\n \"there\",\n \"here\",\n \"of\",\n \"in\",\n \"on\",\n \"at\",\n \"to\",\n \"for\",\n \"with\",\n \"and\",\n \"or\",\n \"but\",\n \"not\",\n \"do\",\n \"does\",\n \"did\",\n \"done\",\n \"has\",\n \"have\",\n \"had\",\n \"having\",\n \"can\",\n \"could\",\n \"will\",\n \"would\",\n \"shall\",\n \"should\",\n \"may\",\n \"might\",\n \"must\",\n \"its\",\n \"your\",\n \"you\",\n \"about\",\n \"all\",\n \"any\",\n \"some\",\n \"like\",\n]);\n\nconst SHEET_MARKER_RE = /^\\[Sheet:\\s*(.+?)\\]\\s*$/;\nconst CODE_FENCE_RE = /^```(?:javascript|js|python)?\\s*\\n?|\\n?```\\s*$/gm;\n\nconst COMPUTE_SYSTEM_PROMPT = `You are a data analyst. Write JavaScript code that computes the answer to the user's instruction over the provided tables.\n\nAvailable in your code:\n- \\`dfs\\`: an object mapping sheet/document name to an array of row objects (one object per row, keys are column names)\n- \\`documents\\`: an array of objects (\\`id\\`, \\`name\\`, \\`sourceId\\`) — the source documents \\`dfs\\` was built from\n- Language: JSON, Math, Date, Array, Object, Number, String, Boolean, parseInt, parseFloat, isFinite, isNaN, console.log\n\nRules:\n1. Set a variable named \\`result\\` to the final answer (a number, string, array, or object).\n2. No file I/O, no network calls, no imports, no require, no process.\n3. Return ONLY the JavaScript code — no markdown fences, no explanation, no commentary.`;\n\nfunction parseDocumentId(documentId: unknown): string {\n const raw = String(documentId ?? \"\")\n .replace(/[ \\n\\t]/g, \"\")\n .trim();\n if (!UUID_RE.test(raw)) {\n throw new EngineActionError(`invalid document id: ${JSON.stringify(documentId)}`);\n }\n return raw;\n}\n\n/**\n * ACL visibility for one document row.\n *\n * - `principals=null` — trusted internal caller; every document is visible.\n * - `principals=[]` — anonymous; only rows whose `acl IS NULL` (unrestricted).\n * - a non-empty list — unrestricted rows plus those whose ACL overlaps.\n *\n * `[]` and `null` are NOT interchangeable. An empty ACL array on the row\n * (distinct from NULL) overlaps nothing, so it is invisible to anonymous\n * and named callers alike — only a trusted caller sees it.\n *\n * One rule, one implementation — `tools/acl.ts` is the canonical predicate\n * (its Python twin's docstring: a second, divergent implementation is\n * exactly how ACL semantics drift between surfaces). These names are\n * re-exports, never a reimplementation.\n */\nexport const visible = aclVisible;\n\n/** Python-named alias used by engine/tests that expect `_visible`. */\nexport const _visible = aclVisible;\n\nexport { aclVisible };\n\nexport function scopeSql(\n sourceIds: string[] | null | undefined,\n principals: string[] | null | undefined,\n params: unknown[],\n /**\n * Table alias to qualify the columns with. Needed only where the query\n * joins a second table that also has `source_id`/`acl` — the chunk outline\n * in `documentStructure` — where unqualified names would be ambiguous.\n * Passing an alias is what lets that query reuse THIS predicate instead of\n * growing a second copy of the ACL rule.\n */\n alias = \"\",\n): string {\n const col = alias ? `${alias}.` : \"\";\n const where: string[] = [];\n if (sourceIds != null) {\n params.push(sourceIds);\n where.push(`${col}source_id = ANY($${params.length}::text[])`);\n }\n if (principals != null) {\n // `acl && $principals` — overlap-of-empty-array is FALSE for every\n // non-empty acl, which is exactly the `principals=[]` (anonymous\n // caller) semantics this package uses everywhere. NULL acl still\n // matches via `acl IS NULL`.\n params.push(principals);\n where.push(`(${col}acl IS NULL OR ${col}acl && $${params.length}::text[])`);\n }\n return where.length ? `WHERE ${where.join(\" AND \")}` : \"\";\n}\n\nfunction parseCursor(cursorValue: unknown): Record<string, unknown> | null {\n if (cursorValue == null) return null;\n if (typeof cursorValue === \"object\" && !Array.isArray(cursorValue)) {\n return cursorValue as Record<string, unknown>;\n }\n if (typeof cursorValue === \"string\") {\n try {\n const parsed: unknown = JSON.parse(cursorValue);\n if (parsed && typeof parsed === \"object\" && !Array.isArray(parsed)) {\n return parsed as Record<string, unknown>;\n }\n } catch {\n /* fall through */\n }\n }\n console.warn(\"listDocuments: could not parse cursor value %s\", cursorValue);\n return null;\n}\n\n/**\n * Redact every string leaf of `value` — a bare string, or a JSON-ish\n * dict/list structure, walked recursively. Dict KEYS are never redacted\n * (they are field names; rewriting them would silently change the\n * result's schema).\n *\n * `policy=null` or an empty policy is an exact no-op — `value` is returned\n * unchanged, not even copied.\n */\nexport function redactValueRecursive(\n value: unknown,\n policy: RedactionPolicy | null | undefined,\n opts: { principals: string[] | null; secretKey: string | Buffer | null; hooks?: Hooks | null },\n): unknown {\n if (policy == null || policy.isEmpty()) return value;\n\n const walk = (node: unknown): unknown => {\n if (typeof node === \"string\") {\n return applyRedaction(node, policy, {\n phase: \"output\",\n principals: opts.principals,\n secretKey: opts.secretKey,\n hooks: opts.hooks,\n })[0];\n }\n if (Array.isArray(node)) return node.map(walk);\n if (node && typeof node === \"object\") {\n return Object.fromEntries(\n Object.entries(node as Record<string, unknown>).map(([k, v]) => [k, walk(v)]),\n );\n }\n return node;\n };\n return walk(value);\n}\n\nexport async function requireVisible(\n pool: Pool,\n documentId: string,\n principals: string[] | null,\n): Promise<void> {\n const id = parseDocumentId(documentId);\n const { rows } = await pool.query(`SELECT acl FROM context_engine_documents WHERE id = $1`, [id]);\n if (!rows[0] || !visible(rows[0].acl, principals)) {\n throw new DocumentNotFoundError(id);\n }\n}\n\nexport async function getDocumentRow(\n pool: Pool,\n documentId: string,\n principals: string[] | null,\n): Promise<Record<string, unknown>> {\n const id = parseDocumentId(documentId);\n const { rows } = await pool.query(`SELECT * FROM context_engine_documents WHERE id = $1`, [id]);\n const doc = rows[0];\n if (!doc || !visible(doc.acl, principals)) throw new DocumentNotFoundError(id);\n const count = await pool.query(\n `SELECT count(*)::int AS n FROM context_engine_chunks WHERE document_id = $1`,\n [id],\n );\n return {\n id: String(doc.id),\n sourceId: doc.source_id,\n externalId: doc.external_id,\n name: doc.name,\n description: doc.description,\n metaData: doc.meta_data ?? {},\n // `[]` (an explicitly empty principal list) is NOT the same as\n // NULL (unrestricted) — never collapse one into the other.\n acl: doc.acl != null ? [...doc.acl] : null,\n mode: doc.mode,\n mimeType: doc.mime_type,\n lang: doc.lang,\n text: doc.text,\n status: doc.status,\n error: doc.error,\n chunks: count.rows[0]?.n ?? 0,\n documentType: doc.document_type,\n structuredData: doc.structured_data,\n createdAt: doc.created_at,\n updatedAt: doc.updated_at,\n startedAt: doc.started_at,\n completedAt: doc.completed_at,\n unreadableReason: doc.meta_data?.unreadable_reason ?? null,\n unreadablePages: doc.meta_data?.unreadable_pages ?? 0,\n };\n}\n\nexport async function stats(pool: Pool, sourceId: string | null): Promise<Record<string, unknown>> {\n const where = sourceId != null ? \"WHERE source_id = $1\" : \"\";\n const params = sourceId != null ? [sourceId] : [];\n const docs = await pool.query(`SELECT count(*)::int AS n FROM context_engine_documents ${where}`, params);\n const chunks = await pool.query(\n `SELECT count(*)::int AS n FROM context_engine_chunks ${sourceId != null ? \"WHERE source_id = $1\" : \"\"}`,\n params,\n );\n const statuses = await pool.query(\n `SELECT status, count(*)::int AS n FROM context_engine_documents ${where} GROUP BY status`,\n params,\n );\n const meta = await pool.query(\n `SELECT embedding_provider, embedding_model, embedding_dim FROM context_engine_meta WHERE id = 1`,\n );\n const m = meta.rows[0];\n return {\n sourceId,\n documents: docs.rows[0]?.n ?? 0,\n chunks: chunks.rows[0]?.n ?? 0,\n byStatus: Object.fromEntries(statuses.rows.map((r) => [r.status, r.n])),\n embedding: {\n provider: m?.embedding_provider ?? null,\n model: m?.embedding_model ?? null,\n dim: m?.embedding_dim ?? null,\n },\n };\n}\n\n/**\n * Return one document's full text, ACL-checked.\n *\n * Unlike an LLM-tool payload helper, this returns the WHOLE text\n * untruncated — token-budget policy belongs to whatever calls this.\n *\n * Raises `EngineActionError` for an unparseable id and\n * `DocumentNotFoundError` for a missing document or one the caller's\n * `principals` can't see — the same message for the last two, deliberately.\n *\n * `redaction` (an `apply_at=\"output\"` policy) masks the fetched text\n * before it is returned. This is the ONLY protection for a corpus ingested\n * WITHOUT an `apply_at=\"ingest\"` policy.\n */\nexport async function getDocumentText(\n documentId: string,\n opts: {\n pool: Pool;\n principals?: string[] | null;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<string> {\n const principals = opts.principals ?? null;\n const doc = await getDocumentRow(opts.pool, documentId, principals);\n return redactValueRecursive(String(doc.text ?? \"\"), opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }) as string;\n}\n\n/**\n * Page through documents in scope, newest first.\n *\n * `redaction` masks `documentType` — unconstrained free text an LLM wrote\n * after reading the document body. `name`/`description` remain\n * intentionally unmasked: both are CALLER-set at ingest time, not text the\n * pipeline derived from the document body.\n */\n/**\n * One EXISTS: is anything in scope ingested with `mode: \"graph\"`?\n *\n * This is what a mode-less search reads to decide whether the graph leg is\n * worth building. Scoped exactly like the search itself — a linked-up document\n * in another source must not turn the leg on for a search that cannot see it.\n */\nexport async function hasGraphDocuments(\n pool: Pool,\n opts: {\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n principals?: string[] | null;\n },\n): Promise<boolean> {\n const params: unknown[] = [];\n const where = scopeSql(opts.sourceIds ?? null, opts.principals ?? null, params);\n let sql = `SELECT 1 FROM context_engine_documents ${where ? `${where} AND` : \"WHERE\"} mode = 'graph'`;\n if (opts.documentIds != null) {\n // `!= null`, NOT truthiness: an EMPTY list means nothing is in scope, and\n // collapsing it would ask about the whole corpus.\n params.push(opts.documentIds);\n sql += ` AND id = ANY($${params.length}::uuid[])`;\n }\n const { rows } = await pool.query(`${sql} LIMIT 1`, params);\n return rows.length > 0;\n}\n\nexport const MAX_CHUNKS_PER_READ = 25;\n\n/**\n * One LLM call per document, in-process. Far below the queue-backed numbers a\n * task runner can afford: this library has no task queue and is not growing\n * one, so the cap is what a request can honestly finish.\n */\nexport const MAX_MAP_REDUCE_DOCS = 200;\nexport const DEFAULT_MAP_REDUCE_DOCS = 25;\nexport const MAX_MAP_REDUCE_CONCURRENCY = 20;\n\n/**\n * Read one document in order, a range of chunks at a time.\n *\n * The document is ACL-checked FIRST, with the same one message an absent and a\n * forbidden document share: reading a document a chunk at a time must not be\n * the way around the ACL on the document itself.\n */\nexport async function getChunks(\n documentId: string,\n opts: {\n pool: Pool;\n principals?: string[] | null;\n start?: number | null;\n end?: number | null;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<Record<string, unknown>> {\n const parsedId = parseDocumentId(documentId);\n const principals = opts.principals ?? null;\n const first = Math.max(0, Math.trunc(opts.start ?? 0));\n const requestedEnd = opts.end == null ? first + MAX_CHUNKS_PER_READ - 1 : Math.trunc(opts.end);\n const last = Math.max(first, Math.min(requestedEnd, first + MAX_CHUNKS_PER_READ - 1));\n\n const visible = await opts.pool.query(\n `SELECT id FROM context_engine_documents\n WHERE id = $1::uuid AND ($2::text[] IS NULL OR acl IS NULL OR acl && $2::text[])`,\n [parsedId, principals],\n );\n if (!visible.rows.length) throw new DocumentNotFoundError(documentId);\n\n const totalRow = await opts.pool.query(\n `SELECT COUNT(*)::int AS total FROM context_engine_chunks WHERE document_id = $1::uuid`,\n [parsedId],\n );\n const total = Number(totalRow.rows[0]?.total ?? 0);\n const { rows } = await opts.pool.query(\n `SELECT idx, text, lang FROM context_engine_chunks\n WHERE document_id = $1::uuid AND idx >= $2 AND idx <= $3\n ORDER BY idx ASC`,\n [parsedId, first, last],\n );\n const chunks = rows.map((r) => ({\n position: r.idx,\n text: redactValueRecursive(r.text, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n language: r.lang,\n }));\n const nextStart = chunks.length ? Number(chunks[chunks.length - 1]?.position ?? first) + 1 : first;\n return {\n document_id: parsedId,\n total_chunks: total,\n start: first,\n end: last,\n chunks,\n has_more: nextStart < total,\n next_start: nextStart,\n };\n}\n\n/**\n * Ask the same question of every document in scope, one call each.\n *\n * The map step only: each document's answer comes back beside its name, and\n * the caller does the reducing. There is no second LLM call summarising the\n * summaries, because that is the step that invents a number nobody wrote.\n *\n * A document that fails is REPORTED, not thrown — one provider hiccup in a\n * fan-out of twenty must not throw away the nineteen that worked.\n */\nexport async function mapReduce(\n instruction: string,\n opts: {\n pool: Pool;\n config: ContextEngineConfig;\n principals?: string[] | null;\n sourceIds?: string[] | null;\n documentIds?: string[] | null;\n limit?: number | null;\n maxConcurrency?: number | null;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<Record<string, unknown>> {\n if (opts.config.llm == null) {\n throw new EngineActionError(\"map_reduce needs an LLM configured on this deployment\");\n }\n const principals = opts.principals ?? null;\n const n = Math.max(1, Math.min(Math.trunc(opts.limit || DEFAULT_MAP_REDUCE_DOCS), MAX_MAP_REDUCE_DOCS));\n const concurrency = Math.max(1, Math.min(Math.trunc(opts.maxConcurrency || 5), MAX_MAP_REDUCE_CONCURRENCY));\n\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n if (opts.documentIds != null) {\n // INTERSECTS with the source scope — a named id outside it can never\n // widen the set, only narrow it.\n const parsedIds: string[] = [];\n for (const did of opts.documentIds) {\n try {\n parsedIds.push(parseDocumentId(did));\n } catch {\n console.warn(\"map_reduce: skipping invalid document id %s\", did);\n }\n }\n params.push(parsedIds);\n where = where\n ? `${where} AND id = ANY($${params.length}::uuid[])`\n : `WHERE id = ANY($${params.length}::uuid[])`;\n }\n params.push(n);\n const { rows } = await opts.pool.query(\n `SELECT id::text, name, text FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n\n const results: Array<Record<string, unknown>> = new Array(rows.length);\n let cursor = 0;\n async function worker() {\n while (cursor < rows.length) {\n const index = cursor++;\n const row = rows[index];\n // Redaction runs BEFORE the text reaches the model, never after: the\n // provider is an egress, and masking the answer would be too late.\n const body = redactValueRecursive(String(row.text ?? \"\").slice(0, 20000), opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n });\n try {\n const [raw] = await callLlm(opts.config.llm as never, {\n system:\n \"You extract one answer from one document. Reply with STRICT JSON only. \" +\n 'If the document does not answer, reply {\"found\": false}.',\n user: `INSTRUCTION:\\n${instruction}\\n\\nDOCUMENT (${row.name}):\\n${body}`,\n jsonMode: true,\n temperature: 0,\n });\n let data: unknown;\n try {\n data = JSON.parse(raw);\n } catch {\n data = { raw };\n }\n results[index] = { document_id: row.id, document_name: row.name, data };\n } catch (err) {\n results[index] = {\n document_id: row.id,\n document_name: row.name,\n error: err instanceof Error ? err.message : String(err),\n };\n }\n }\n }\n await Promise.all(Array.from({ length: Math.min(concurrency, rows.length) }, worker));\n\n const failed = results.filter((r) => r?.error).length;\n return {\n results,\n processed: results.length - failed,\n failed,\n considered: rows.length,\n };\n}\n\nexport async function listDocuments(opts: {\n pool: Pool;\n sourceId?: string | null;\n /** OR-of-many, ONE keyset-paged query across all of them; with `sourceId`, their union. */\n sourceIds?: string[] | null;\n /**\n * Narrows to specific documents IN SQL, not after the page is built: three\n * named documents that happen to sit on page four would otherwise come back\n * as an empty page rather than as themselves. An empty array means nothing\n * is in scope and stays empty.\n */\n documentIds?: string[] | null;\n principals?: string[] | null;\n cursor?: unknown;\n limit?: number;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n}): Promise<Record<string, unknown>> {\n const limit = Math.max(1, Math.min(Math.trunc(opts.limit || 50), MAX_LIST_LIMIT));\n const parsedCursor = parseCursor(opts.cursor);\n // `!= null`, NOT truthiness: no scoping given at all means the whole corpus,\n // but an EMPTY array means nothing is in scope, and collapsing the two would\n // turn a scoped listing into a listing of everything.\n const sourceIds =\n opts.sourceIds == null && opts.sourceId == null\n ? null\n : [...new Set([...(opts.sourceIds ?? []), ...(opts.sourceId != null ? [opts.sourceId] : [])])];\n const principals = opts.principals ?? null;\n\n const params: unknown[] = [];\n let where = scopeSql(sourceIds, principals, params);\n if (opts.documentIds != null) {\n const parsedIds: string[] = [];\n for (const did of opts.documentIds) {\n try {\n parsedIds.push(parseDocumentId(did));\n } catch {\n console.warn(\"listDocuments: skipping invalid document id %s\", did);\n }\n }\n params.push(parsedIds);\n const clause = `id = ANY($${params.length}::uuid[])`;\n where = where ? `${where} AND ${clause}` : `WHERE ${clause}`;\n }\n const cursorTime = parsedCursor?.time ?? parsedCursor?.created_at;\n const cursorId = parsedCursor?.id;\n if (cursorTime && cursorId && UUID_RE.test(String(cursorId))) {\n params.push(cursorTime, cursorId);\n const extra = `(created_at < $${params.length - 1} OR (created_at = $${params.length - 1} AND id < $${params.length}::uuid))`;\n where = where ? `${where} AND ${extra}` : `WHERE ${extra}`;\n }\n params.push(limit + 1);\n const { rows } = await opts.pool.query(\n `SELECT id, source_id, external_id, name, description, mode, mime_type, lang, status,\n document_type, created_at, updated_at, acl, meta_data\n FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n\n const hasMore = rows.length > limit;\n const page = rows.slice(0, limit);\n const documents = page.map((doc) => ({\n id: String(doc.id),\n sourceId: doc.source_id,\n externalId: doc.external_id,\n name: doc.name,\n description: doc.description,\n mode: doc.mode,\n mimeType: doc.mime_type,\n lang: doc.lang,\n status: doc.status,\n documentType: redactValueRecursive(doc.document_type, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n createdAt: doc.created_at instanceof Date ? doc.created_at.toISOString() : doc.created_at,\n updatedAt: doc.updated_at instanceof Date ? doc.updated_at.toISOString() : doc.updated_at,\n unreadableReason: doc.meta_data?.unreadable_reason ?? null,\n unreadablePages: doc.meta_data?.unreadable_pages ?? 0,\n // WHO can see this. Stored, enforced on every query and editable through\n // updateDocument — and, until this line, invisible to every caller,\n // because this serializer builds a FIXED object. `null` is UNRESTRICTED\n // and must stay null; an empty array would read as \"nobody\", which is the\n // opposite claim.\n acl: doc.acl ?? null,\n }));\n\n const result: Record<string, unknown> = {\n documents,\n count: documents.length,\n hasMore,\n };\n if (hasMore && page.length) {\n const last = page[page.length - 1]!;\n result.nextCursor = {\n time: last.created_at instanceof Date ? last.created_at.toISOString() : last.created_at,\n id: String(last.id),\n };\n }\n return result;\n}\n\n// ---------------------------------------------------------------------------\n// schema summaries — what `discover` says beyond the inventory\n// ---------------------------------------------------------------------------\n\nexport type SpreadsheetDescription = {\n document_id: string;\n name: unknown;\n source_id: unknown;\n sheets: SheetSchema[] | null;\n schema_unavailable?: string;\n};\n\n/**\n * Sheet names, column headers and row counts for the named spreadsheets.\n *\n * `documentIds` comes from a listing the caller has already been shown, but\n * the ACL and the source scope are applied AGAIN here: \"we already filtered\n * the list\" is exactly how a batch read ends up reading one row it should\n * not have. A document that is not visible, is not in scope, or is not\n * tabular simply does not appear in the result.\n *\n * Reading document bodies is the cost, so it is bounded twice: the caller\n * passes only the page it is describing, and `budget` caps the total\n * characters of stored text this will pull. Documents past the budget are\n * still LISTED, with `sheets: null` and a reason — a silently shortened\n * schema would read as \"that workbook has no sheets\".\n *\n * `redaction` masks sheet names and column headers, which are derived from\n * the document body exactly like `search`'s chunk text. The document `name`\n * is not masked, matching `listDocuments`: it is caller-set at ingest.\n */\nexport async function spreadsheetSchema(opts: {\n pool: Pool;\n documentIds: string[];\n sourceIds?: string[] | null;\n principals?: string[] | null;\n budget?: number;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n}): Promise<SpreadsheetDescription[]> {\n const ids: string[] = [];\n for (const did of opts.documentIds ?? []) {\n try {\n ids.push(parseDocumentId(did));\n } catch {\n console.warn(\"spreadsheetSchema: skipping invalid document id %s\", did);\n }\n }\n if (!ids.length) return [];\n\n const principals = opts.principals ?? null;\n const budget = opts.budget ?? MAX_SCHEMA_TEXT_CHARS;\n\n // Two queries on purpose: `length(text)` first, bodies second. Deciding\n // what fits the budget only after the bodies are already in memory would\n // make the budget a report rather than a limit.\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n const mimeClause = TABULAR_MIME_PATTERNS.map((pattern) => {\n params.push(pattern);\n return `mime_type ILIKE $${params.length}`;\n }).join(\" OR \");\n where = where ? `${where} AND (${mimeClause})` : `WHERE (${mimeClause})`;\n params.push(ids);\n where += ` AND id = ANY($${params.length}::uuid[])`;\n const sized = await opts.pool.query(\n `SELECT id, name, source_id, mime_type, COALESCE(LENGTH(text), 0) AS text_len\n FROM context_engine_documents\n ${where}`,\n params,\n );\n\n // The ILIKE list is a prefilter so non-spreadsheets are never fetched;\n // `isTabularMime` stays the one definition of \"tabular\", the same split\n // `compute()` keeps.\n const byId = new Map<string, Record<string, unknown>>();\n for (const row of sized.rows) {\n if (isTabularMime(row.mime_type)) byId.set(String(row.id), row);\n }\n // The caller's order is the order it showed the model; keep it.\n const rows = ids.map((id) => byId.get(id)).filter((r): r is Record<string, unknown> => r != null);\n\n const chosen: string[] = [];\n let spent = 0;\n for (const row of rows) {\n const len = Number(row.text_len ?? 0);\n // The first spreadsheet is always read, however big: a page whose one\n // workbook is larger than the budget would otherwise describe nothing.\n if (chosen.length && spent + len > budget) break;\n chosen.push(String(row.id));\n spent += len;\n }\n\n const texts = new Map<string, string | null>();\n if (chosen.length) {\n const bodyParams: unknown[] = [];\n const bodyWhere = scopeSql(opts.sourceIds ?? null, principals, bodyParams);\n bodyParams.push(chosen);\n const clause = `id = ANY($${bodyParams.length}::uuid[])`;\n const { rows: bodies } = await opts.pool.query(\n `SELECT id, text FROM context_engine_documents\n ${bodyWhere ? `${bodyWhere} AND ${clause}` : `WHERE ${clause}`}`,\n bodyParams,\n );\n for (const row of bodies) texts.set(String(row.id), row.text ?? null);\n }\n\n return rows.map((row) => {\n const id = String(row.id);\n if (!texts.has(id)) {\n return {\n document_id: id,\n name: row.name,\n source_id: row.source_id,\n sheets: null,\n schema_unavailable:\n \"not read: this page of spreadsheets is past the text budget — \" +\n \"narrow with sourceIds, or ask for a smaller limit\",\n };\n }\n return {\n document_id: id,\n name: row.name,\n source_id: row.source_id,\n sheets: redactValueRecursive(spreadsheetSchemaFromText(texts.get(id)), opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }) as SheetSchema[],\n };\n });\n}\n\nexport type DocumentStructure = Record<string, unknown>;\n\n/**\n * Cap the sheets, and the columns within each one, at the same limit.\n *\n * A workbook is two unbounded lists, not one: twenty sheets of two hundred\n * columns is four thousand names in the first call of a conversation. The\n * bound is applied HERE rather than in `spreadsheetSchema`, which stays a\n * complete answer for a caller that asked for one document's schema on\n * purpose — it is `discover`'s payload that has a context budget.\n */\n/**\n * How many items every list in this structure lost to the cap, together.\n *\n * The SUM, not the largest: it is what a caller would actually be handed if\n * it came back for this document, and that total is what decides whether the\n * payload should offer the trip at all.\n */\nfunction remainder(structure: DocumentStructure): number {\n let total = 0;\n for (const [key, value] of Object.entries(structure)) {\n if (key.startsWith(\"more_\") && typeof value === \"number\") total += value;\n }\n const sheets = structure.sheets;\n if (Array.isArray(sheets)) {\n for (const sheet of sheets) {\n if (sheet && typeof sheet === \"object\" && typeof sheet.more_columns === \"number\") {\n total += sheet.more_columns;\n }\n }\n }\n return total;\n}\n\nfunction boundedSheets(sheets: SheetSchema[] | null): SheetSchema[] | null {\n if (!sheets?.length) return sheets;\n return sheets.slice(0, MAX_STRUCTURE_ITEMS).map((sheet) => {\n const columns = sheet.columns ?? [];\n const trimmed: SheetSchema & { more_columns?: number } = {\n ...sheet,\n columns: columns.slice(0, MAX_STRUCTURE_ITEMS),\n };\n if (columns.length > MAX_STRUCTURE_ITEMS) {\n trimmed.more_columns = columns.length - MAX_STRUCTURE_ITEMS;\n }\n return trimmed;\n });\n}\n\n/**\n * What is INSIDE each of the named documents, whatever its type.\n *\n * Returns `{documentId: structure}`. The shape of `structure` follows the\n * document, because the thing a model needs to know differs by type:\n *\n * | document | structure |\n * |---|---|\n * | CSV / XLSX / TSV | `sheets: [{name, columns, row_count}]` |\n * | markdown, Word, HTML, a transcribed PDF | `sections: [...]`, `last_page` |\n * | JSON | `keys: [...]` |\n * | anything else | just `chunks` |\n *\n * `chunks` is on every one of them, and is the floor: a plain `.txt` has no\n * headings to report, but a model still has to know whether `get_chunks` is\n * worth calling. A document that is present but described by nothing at all\n * reads as an empty document, which is why there is no \"no structure\" case.\n *\n * Everything but the spreadsheet columns comes from `meta_data` on the chunk\n * rows, which the chunkers already wrote at ingest (`section_title`, `page`,\n * `top_level_key`) — one grouped query over small rows, no document body read\n * and no LLM. Columns are the exception: the header row is not in chunk\n * metadata, so the tabular half still parses stored text, under the same\n * budget (see `spreadsheetSchema`).\n */\nexport async function documentStructure(opts: {\n pool: Pool;\n documentIds: string[];\n sourceIds?: string[] | null;\n principals?: string[] | null;\n budget?: number;\n /**\n * Cap every list at `MAX_STRUCTURE_ITEMS` and report the remainder as a\n * count. One rule decides it, and the caller above applies it: **the cap is\n * for the call that did NOT name its documents.** A `discover` describing a\n * whole page has a context budget to keep; a caller that asked about one\n * document asked for all of it, exactly like `spreadsheetSchema`.\n */\n bounded?: boolean;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n}): Promise<Record<string, DocumentStructure>> {\n const ids: string[] = [];\n for (const did of opts.documentIds ?? []) {\n try {\n ids.push(parseDocumentId(did));\n } catch {\n console.warn(\"documentStructure: skipping invalid document id %s\", did);\n }\n }\n if (!ids.length) return {};\n\n const principals = opts.principals ?? null;\n const bounded = opts.bounded ?? true;\n const described = await spreadsheetSchema({\n pool: opts.pool,\n documentIds: ids,\n sourceIds: opts.sourceIds,\n principals,\n budget: opts.budget,\n redaction: opts.redaction,\n secretKey: opts.secretKey,\n hooks: opts.hooks,\n });\n const tabular = new Map(described.map((entry) => [entry.document_id, entry]));\n\n // Scoped by joining to the documents table and applying the SAME predicate\n // every other read here uses, not by `chunks.acl`: the document row is the\n // authority on who may read a document (the rule `getChunks` enforces\n // before it reads a single chunk), and reusing the one predicate is what\n // keeps a chunk-level read from drifting away from the document-level\n // answer.\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params, \"d\");\n params.push(ids);\n const clause = `c.document_id = ANY($${params.length}::uuid[])`;\n where = where ? `${where} AND ${clause}` : `WHERE ${clause}`;\n const { rows } = await opts.pool.query(\n `SELECT c.document_id AS document_id,\n c.meta_data->>'section_title' AS section,\n c.meta_data->>'top_level_key' AS key,\n MIN(c.idx) AS first_idx,\n COUNT(*) AS chunks,\n -- Only the page-marker pass writes the page number, and it writes an int\n -- — but a cast that meets anything else raises for the whole\n -- query, so the type is checked in SQL rather than assumed.\n MAX(CASE WHEN jsonb_typeof(c.meta_data->'page') = 'number'\n THEN (c.meta_data->>'page')::int END) AS page\n FROM context_engine_chunks c\n JOIN context_engine_documents d ON d.id = c.document_id\n ${where}\n GROUP BY c.document_id, section, key`,\n params,\n );\n\n type Outline = {\n chunks: number;\n page: number | null;\n parts: Array<[number, string | null, string | null]>;\n };\n const outline = new Map<string, Outline>();\n for (const row of rows) {\n const id = String(row.document_id);\n let found = outline.get(id);\n if (!found) {\n found = { chunks: 0, page: null, parts: [] };\n outline.set(id, found);\n }\n found.chunks += Number(row.chunks ?? 0);\n if (row.page != null) found.page = Math.max(found.page ?? 0, Number(row.page));\n found.parts.push([Number(row.first_idx ?? 0), row.section ?? null, row.key ?? null]);\n }\n\n /** Distinct values in the order the document itself puts them. */\n const ordered = (\n parts: Outline[\"parts\"],\n pick: (part: Outline[\"parts\"][number]) => string | null,\n cap: number | null,\n ): [string[], number] => {\n const seen: string[] = [];\n for (const part of [...parts].sort((a, b) => a[0] - b[0])) {\n const value = pick(part);\n if (value && !seen.includes(value)) seen.push(value);\n }\n return [cap === null ? seen : seen.slice(0, cap), seen.length];\n };\n\n const out: Record<string, DocumentStructure> = {};\n for (const id of ids) {\n const found = outline.get(id);\n // A document with no visible chunks still gets a structure, because the\n // caller was shown the document: `{chunks: 0}` says \"nothing indexed\",\n // which is a fact, where an absent key says nothing at all.\n const structure: DocumentStructure = { chunks: found ? found.chunks : 0 };\n\n const sheetDoc = tabular.get(id);\n if (sheetDoc) {\n const sheets = sheetDoc.sheets ?? null;\n structure.sheets = bounded ? boundedSheets(sheets) : sheets;\n if (bounded && sheets && sheets.length > MAX_STRUCTURE_ITEMS) {\n structure.more_sheets = sheets.length - MAX_STRUCTURE_ITEMS;\n }\n if (sheetDoc.schema_unavailable) structure.schema_unavailable = sheetDoc.schema_unavailable;\n out[id] = structure;\n continue;\n }\n\n if (found) {\n if (found.page != null) {\n // `last_page`, not `pages`: chunk metadata records the page a chunk\n // STARTS on, so this is the highest page any chunk reaches and\n // therefore a LOWER BOUND on the document's real page count. The true\n // count is never stored, so naming this `pages` would be a number\n // that reads as a fact and is not one.\n structure.last_page = found.page;\n }\n const cap = bounded ? MAX_STRUCTURE_ITEMS : null;\n const [sections, totalSections] = ordered(found.parts, (p) => p[1], cap);\n if (sections.length) {\n structure.sections = sections;\n // The REMAINDER, and only when there is one: a `more_sections: 0`\n // reads as a truncation that happened to stop at the end.\n if (totalSections > sections.length) structure.more_sections = totalSections - sections.length;\n }\n const [keys, totalKeys] = ordered(found.parts, (p) => p[2], cap);\n if (keys.length) {\n structure.keys = keys;\n if (totalKeys > keys.length) structure.more_keys = totalKeys - keys.length;\n }\n }\n out[id] = structure;\n }\n\n for (const [id, structure] of Object.entries(out)) {\n const left = remainder(structure);\n if (!left) continue;\n if (left <= MAX_INVITED_ITEMS) {\n // A remainder the model cannot act on is just a number. This is the\n // same contract `next_page` has on a truncated listing: the call to\n // make next, already filled in.\n structure.next_action = { action: \"discover\", document_ids: [id] };\n } else if (\"sheets\" in structure) {\n structure.instead =\n \"too many to list: a sheet this wide is one to COMPUTE over, not to read. \" +\n \"Use action 'compute' and name the columns above, or search for a value \" +\n \"to find the one row you want.\";\n } else {\n structure.instead =\n \"too many to list: read the document with get_chunks, or search it for \" + \"the part you need.\";\n }\n }\n\n // Section titles and JSON keys are lifted straight out of the document\n // body, exactly like a chunk of its text. The spreadsheet half was already\n // masked inside `spreadsheetSchema`, and masking it twice is a no-op.\n const redactOpts = { principals, secretKey: opts.secretKey ?? null, hooks: opts.hooks };\n return Object.fromEntries(\n Object.entries(out).map(([id, structure]) => [\n id,\n redactValueRecursive(structure, opts.redaction, redactOpts) as DocumentStructure,\n ]),\n );\n}\n\nexport type DocumentTypeCount = {\n kind: \"spreadsheet\" | \"text\";\n type: string | null;\n documents: number;\n with_fields: number;\n};\n\n/**\n * The corpus census: `[{kind, type, documents, with_fields}]`.\n *\n * Keyed by TWO things, because only one of them is always known:\n *\n * - `kind` is derived from the mime (`isTabularMime`, the same predicate\n * `compute` selects on), so it is known for every document ever ingested.\n * - `type` is the document's `document_type`, which is unconstrained free\n * text an LLM wrote during structured extraction — and extraction is\n * opt-in (`extractStructured`, off by default). On a deployment that never\n * enabled it, `type` is null on every row. A census keyed on `type` alone\n * would therefore be a single null bucket, which tells a model nothing\n * about a corpus it is about to query.\n *\n * `with_fields` counts documents whose `structured_data` is a non-empty\n * object. Extraction that ran and legitimately found nothing stores `{}`, and\n * counting that as coverage would point a model at `query_meta` for a type\n * that has nothing to filter on.\n */\nexport async function documentTypes(opts: {\n pool: Pool;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n documentIds?: string[] | null;\n limit?: number;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n}): Promise<DocumentTypeCount[]> {\n const principals = opts.principals ?? null;\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n if (opts.documentIds != null) {\n const parsedIds: string[] = [];\n for (const did of opts.documentIds) {\n try {\n parsedIds.push(parseDocumentId(did));\n } catch {\n console.warn(\"documentTypes: skipping invalid document id %s\", did);\n }\n }\n params.push(parsedIds);\n const clause = `id = ANY($${params.length}::uuid[])`;\n where = where ? `${where} AND ${clause}` : `WHERE ${clause}`;\n }\n params.push(MAX_CENSUS_ROWS);\n const { rows } = await opts.pool.query(\n `SELECT document_type AS doc_type,\n mime_type,\n COUNT(*) AS documents,\n COUNT(*) FILTER (\n WHERE structured_data IS NOT NULL AND structured_data::text <> '{}'\n ) AS with_fields\n FROM context_engine_documents\n ${where}\n GROUP BY document_type, mime_type\n ORDER BY COUNT(*) DESC\n LIMIT $${params.length}`,\n params,\n );\n\n // The mime -> kind fold happens HERE, not in SQL, so `isTabularMime` stays\n // the single definition of what a spreadsheet is.\n const buckets = new Map<string, DocumentTypeCount>();\n for (const row of rows) {\n const kind: \"spreadsheet\" | \"text\" = isTabularMime(row.mime_type) ? \"spreadsheet\" : \"text\";\n const type = (row.doc_type as string | null) ?? null;\n const bucketKey = `${kind}\\u0000${type ?? \"\"}`;\n let bucket = buckets.get(bucketKey);\n if (!bucket) {\n bucket = { kind, type, documents: 0, with_fields: 0 };\n buckets.set(bucketKey, bucket);\n }\n bucket.documents += Number(row.documents ?? 0);\n bucket.with_fields += Number(row.with_fields ?? 0);\n }\n\n const census = [...buckets.values()]\n .sort(\n (a, b) =>\n b.documents - a.documents ||\n a.kind.localeCompare(b.kind) ||\n (a.type ?? \"\").localeCompare(b.type ?? \"\"),\n )\n .slice(0, opts.limit ?? MAX_DOCUMENT_TYPES);\n return redactValueRecursive(census, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }) as DocumentTypeCount[];\n}\n\nexport type FieldDescription = { field: string; type: string; documents: number };\n\nexport type FieldGroup = {\n kind: \"spreadsheet\" | \"text\";\n type: string | null;\n fields: FieldDescription[];\n more_fields?: string[];\n};\n\n/**\n * Extracted structured field names grouped by document kind AND type.\n *\n * Returns `[{kind, type, fields: [{field, type, documents}], more_fields}]` —\n * one row per group, keyed by the SAME pair `documentTypes` is keyed by. That\n * pairing is the point: `document_type` is a label an LLM wrote, so two\n * different kinds can share one (\"invoice\" for both a PDF and a CSV of\n * invoice rows). Keyed on the label alone, their fields merge into one group\n * and a model reads a spreadsheet's columns as a PDF's extracted fields.\n *\n * `more_fields` holds the NAMES the per-group detail cap left out, not a\n * count: nothing is dropped silently, and a name is all a caller needs to\n * reach a field through `query_meta`. It is absent when nothing was left out.\n *\n * Counted over the documents the CALLER can see, not over the learned field\n * registry (`context_engine_structured_keys`), which carries no ACL: a\n * registry-wide answer would tell an anonymous caller the field names of\n * every restricted document in the corpus. Counting from `structured_data`\n * also makes `documents` mean what it says — how many in-scope documents\n * actually carry the field.\n *\n * The type is `inferDataType`, the same function that labelled the registry,\n * so the vocabulary a model reads here is the one the rest of the engine\n * uses.\n */\nexport async function fieldSummary(opts: {\n pool: Pool;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n documentIds?: string[] | null;\n maxFieldsPerType?: number;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n}): Promise<FieldGroup[]> {\n const principals = opts.principals ?? null;\n const maxFieldsPerType = opts.maxFieldsPerType ?? MAX_FIELDS_PER_TYPE;\n\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n if (opts.documentIds != null) {\n const parsedIds: string[] = [];\n for (const did of opts.documentIds) {\n try {\n parsedIds.push(parseDocumentId(did));\n } catch {\n console.warn(\"fieldSummary: skipping invalid document id %s\", did);\n }\n }\n params.push(parsedIds);\n const clause = `id = ANY($${params.length}::uuid[])`;\n where = where ? `${where} AND ${clause}` : `WHERE ${clause}`;\n }\n const typed = \"jsonb_typeof(structured_data) = 'object'\";\n where = where ? `${where} AND ${typed}` : `WHERE ${typed}`;\n\n // The aggregation happens in Postgres: the alternative — pulling every\n // visible document's `structured_data` and counting in JavaScript — ships\n // the whole corpus's extracted values across the wire to produce a few\n // hundred rows of key names. One representative value per key is enough to\n // infer the data type.\n params.push(MAX_FIELD_ROWS);\n const { rows } = await opts.pool.query(\n `SELECT d.document_type AS doc_type,\n d.mime_type AS mime_type,\n kv.key AS key,\n COUNT(*) AS documents,\n MAX(kv.value::text) AS sample\n FROM context_engine_documents d\n JOIN LATERAL jsonb_each(d.structured_data) AS kv(key, value) ON TRUE\n ${where}\n GROUP BY d.document_type, d.mime_type, kv.key\n ORDER BY d.document_type, COUNT(*) DESC, kv.key\n LIMIT $${params.length}`,\n params,\n );\n\n const groups = new Map<string, FieldGroup>();\n for (const row of rows) {\n // The mime -> kind fold happens HERE, not in SQL, so `isTabularMime`\n // stays the single definition of what a spreadsheet is — the same split\n // `documentTypes` makes.\n const kind: \"spreadsheet\" | \"text\" = isTabularMime(row.mime_type) ? \"spreadsheet\" : \"text\";\n const type = (row.doc_type as string | null) ?? null;\n const groupKey = `${kind}\\u0000${type ?? \"\"}`;\n let group = groups.get(groupKey);\n if (!group) {\n group = { kind, type, fields: [] };\n groups.set(groupKey, group);\n }\n let sample: unknown = null;\n try {\n sample = row.sample == null ? null : JSON.parse(String(row.sample));\n } catch {\n sample = row.sample;\n }\n if (group.fields.length < maxFieldsPerType) {\n group.fields.push({\n field: String(row.key),\n type: inferDataType(String(row.key), sample),\n documents: Number(row.documents),\n });\n } else {\n group.more_fields ??= [];\n group.more_fields.push(String(row.key));\n }\n }\n\n return redactValueRecursive([...groups.values()], opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }) as FieldGroup[];\n}\n\nfunction candidateFieldTokens(question: string): string[] {\n const candidates: string[] = [];\n const seen = new Set<string>();\n\n const whole = normalizeKey(question);\n if (whole) {\n candidates.push(question.trim());\n seen.add(whole);\n }\n\n for (const match of question.matchAll(WORD_RE)) {\n const word = match[0]!;\n if (word.length < 3) continue;\n const norm = normalizeKey(word);\n if (!norm || seen.has(norm) || STOPWORDS.has(norm)) continue;\n seen.add(norm);\n candidates.push(word);\n }\n return candidates;\n}\n\n/**\n * Answer a natural-language question against documents' `structuredData`.\n *\n * When NOTHING resolves — every candidate fell below the similarity floor\n * and had no exact/synonym match either — this returns ZERO documents, not\n * the whole in-scope corpus. An off-topic question has no business getting\n * an invoice back just because the invoice happens to have SOME structured\n * fields.\n *\n * `redaction` masks each result's `documentType` and the STRING VALUES of\n * `structuredData`. KEYS are left unredacted — they are the canonical\n * field names this function's `resolveFields` matching keys off of.\n */\nexport async function queryStructured(\n question: string,\n opts: {\n pool: Pool;\n embedder: Embedder;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n docType?: string | null;\n limit?: number;\n redaction?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks;\n },\n): Promise<Record<string, unknown>> {\n if (!question?.trim()) throw new EngineActionError(\"question must not be empty\");\n\n const limit = Math.max(1, Math.min(Math.trunc(opts.limit || 20), MAX_LIST_LIMIT));\n const principals = opts.principals ?? null;\n const candidates = candidateFieldTokens(question);\n const resolved = await resolveFields(candidates, {\n pool: opts.pool,\n embedder: opts.embedder,\n sourceIds: opts.sourceIds ?? null,\n docType: opts.docType ?? null,\n });\n const canonicalKeys = [...new Set(Object.values(resolved))].sort();\n\n if (!canonicalKeys.length) {\n return { question, resolvedFields: resolved, documents: [], count: 0 };\n }\n\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n const extras = [`structured_data IS NOT NULL`];\n if (opts.docType != null) {\n params.push(opts.docType);\n extras.push(`document_type = $${params.length}`);\n }\n params.push(canonicalKeys);\n extras.push(`structured_data ?| $${params.length}::text[]`);\n where = where ? `${where} AND ${extras.join(\" AND \")}` : `WHERE ${extras.join(\" AND \")}`;\n params.push(limit);\n\n const { rows } = await opts.pool.query(\n `SELECT id, source_id, name, document_type, structured_data\n FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n\n const documents = rows.map((doc) => ({\n id: String(doc.id),\n sourceId: doc.source_id,\n name: doc.name,\n documentType: redactValueRecursive(doc.document_type, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n structuredData: redactValueRecursive(doc.structured_data ?? {}, opts.redaction, {\n principals,\n secretKey: opts.secretKey ?? null,\n hooks: opts.hooks,\n }),\n }));\n\n return { question, resolvedFields: resolved, documents, count: documents.length };\n}\n\nfunction stripCodeFences(text: string): string {\n return text.replace(CODE_FENCE_RE, \"\").trim();\n}\n\ntype DanfoMod = {\n readCSV: (input: string) => { columns: string[]; shape: [number, number]; values: unknown[][] };\n};\n\nasync function requireDanfo(): Promise<DanfoMod> {\n const specifier = \"danfojs-node\";\n try {\n return (await import(specifier)) as DanfoMod;\n } catch {\n throw new ExtraMissingError(\"compute\", \"danfojs-node\", \"compute() dataframes\");\n }\n}\n\n/**\n * Split `[Sheet: name]`-delimited (or bare) text into `{sheet: lines}`.\n *\n * Shared by `parseSpreadsheetText` and `spreadsheetSchemaFromText` on\n * purpose. The sheet names `discover` shows a model are the keys it will\n * index `dfs` by in the `compute` call it writes next, so the two must come\n * from ONE splitter — including the `sheet1` default an unmarked CSV gets,\n * which is a name no part of the document actually contains.\n */\nfunction splitSheets(text: string): Record<string, string[]> {\n const sheets: Record<string, string[]> = {};\n let current = \"sheet1\";\n for (const line of text.replace(/\\r\\n/g, \"\\n\").replace(/\\r/g, \"\\n\").split(\"\\n\")) {\n const m = SHEET_MARKER_RE.exec(line.trim());\n if (m) {\n current = m[1]!.trim() || current;\n sheets[current] ??= [];\n continue;\n }\n let bucket = sheets[current];\n if (!bucket) {\n bucket = [];\n sheets[current] = bucket;\n }\n bucket.push(line);\n }\n return sheets;\n}\n\nexport type SheetSchema = { name: string; columns: string[]; row_count: number };\n\n/**\n * The schema of one stored spreadsheet: `[{name, columns, row_count}]`.\n *\n * Header row and row count per SHEET, not unioned across the workbook — two\n * sheets that both have an `amount` column are two different frames to\n * `compute`, and a union would hide which one has the column a question\n * needs. `row_count` counts DATA rows (the header is not one of them).\n *\n * Needs no danfo: `discover` must stay callable on an install that never\n * enabled code execution, which is exactly the install where a model most\n * needs the columns before it writes a query.\n */\nexport function spreadsheetSchemaFromText(text: string | null | undefined): SheetSchema[] {\n if (!text) return [];\n const out: SheetSchema[] = [];\n for (const [name, lines] of Object.entries(splitSheets(text))) {\n const body = lines.join(\"\\n\").trim();\n if (!body) continue;\n let rows: string[][];\n try {\n rows = parseCsv(body);\n } catch (exc) {\n // Same rule as `parseSpreadsheetText`: one malformed sheet loses that\n // sheet, never the whole workbook's schema.\n console.warn(\"discover: sheet '%s' not parseable as CSV: %s\", name, exc);\n continue;\n }\n rows = rows.filter((row) => row.some((cell) => (cell ?? \"\").trim()));\n if (!rows.length) continue;\n out.push({\n name,\n columns: rows[0]!.map((cell) => String(cell).trim()),\n row_count: rows.length - 1,\n });\n }\n return out;\n}\n\n/**\n * Parse `[Sheet: name]`-delimited (or bare) CSV text into `{sheet: rows[]}`.\n * Unparseable sheets are skipped, never fatal.\n */\nfunction parseSpreadsheetText(\n dfd: { readCSV: (input: string) => { columns: string[]; shape: [number, number]; values: unknown[][] } },\n text: string | null | undefined,\n): Record<string, Record<string, unknown>[]> {\n if (!text) return {};\n const sheets = splitSheets(text);\n\n const out: Record<string, Record<string, unknown>[]> = {};\n for (const [name, lines] of Object.entries(sheets)) {\n const body = lines.join(\"\\n\").trim();\n if (!body) continue;\n try {\n const df = dfd.readCSV(body);\n const cols = (df.columns ?? []).map((c) => String(c).trim());\n if (!df.shape || df.shape[0] <= 0 || df.shape[1] <= 0) continue;\n const rows: Record<string, unknown>[] = [];\n for (const vals of df.values as unknown[][]) {\n const row: Record<string, unknown> = {};\n for (let i = 0; i < cols.length; i++) row[cols[i]!] = vals[i];\n rows.push(row);\n }\n if (rows.length) out[name] = rows;\n } catch (exc) {\n console.warn(\"compute: sheet '%s' not parseable as CSV: %s\", name, exc);\n }\n }\n return out;\n}\n\n/**\n * Compute an answer to `instruction` over in-scope spreadsheet documents.\n *\n * **Disabled by default** (`config.enableCodeExecution`): this executes\n * LLM-GENERATED code. Raises `EngineActionError` before ANY work — no LLM\n * call, no sandbox run — when the flag is `false`. That flag, not the\n * isolate, is the actual security boundary.\n *\n * Only the document half lives here: fetch the in-scope tabular documents\n * and parse their stored text into `{sheet: rows[]}`. Everything after that\n * — the guards, both redaction surfaces below, the prompt → code → sandbox\n * path and the swept result — is `computeOverFrames`, the seam a host calls\n * when it already holds the frames and has no document to point at; this\n * function builds its frames from documents and delegates.\n *\n * `config.redaction` is applied TWICE:\n * 1. Each document's raw `.text` is masked BEFORE it is parsed into a\n * table, so the data the LLM-authored code runs against is built from\n * masked cells.\n * 2. The final returned dict is swept whole through `redactValueRecursive`\n * — including `code`.\n */\nexport async function compute(\n instruction: string,\n opts: {\n pool: Pool;\n config: ContextEngineConfig;\n hooks: Hooks;\n sourceIds?: string[] | null;\n principals?: string[] | null;\n documentIds?: string[] | null;\n modelCfg?: LLMConfig | null;\n timeout?: number;\n },\n): Promise<Record<string, unknown>> {\n // The same three guards `computeOverFrames` runs — repeated here so a\n // disabled flag, a missing LLM or a blank instruction is refused BEFORE\n // any DB work, not after the documents have been fetched and parsed.\n checkComputePreconditions(opts.config, opts.modelCfg, instruction);\n\n const dfd = await requireDanfo();\n const principals = opts.principals ?? null;\n\n const params: unknown[] = [];\n let where = scopeSql(opts.sourceIds ?? null, principals, params);\n const mimeClause = TABULAR_MIME_PATTERNS.map((p) => {\n params.push(p);\n return `mime_type ILIKE $${params.length}`;\n }).join(\" OR \");\n where = where ? `${where} AND (${mimeClause})` : `WHERE (${mimeClause})`;\n if (opts.documentIds?.length) {\n const parsedIds: string[] = [];\n for (const did of opts.documentIds) {\n try {\n parsedIds.push(parseDocumentId(did));\n } catch {\n console.warn(\"compute: skipping invalid doc_id %s\", did);\n }\n }\n if (!parsedIds.length) {\n throw new EngineActionError(\"no tabular (CSV/TSV/XLSX) documents found in scope for compute()\");\n }\n params.push(parsedIds);\n where += ` AND id = ANY($${params.length}::uuid[])`;\n }\n params.push(MAX_COMPUTE_DOCUMENTS + 1);\n const { rows } = await opts.pool.query(\n `SELECT id, name, source_id, mime_type, text\n FROM context_engine_documents\n ${where}\n ORDER BY created_at DESC, id DESC\n LIMIT $${params.length}`,\n params,\n );\n const tabular = rows.filter((d) => isTabularMime(d.mime_type));\n if (!tabular.length) {\n throw new EngineActionError(\"no tabular (CSV/TSV/XLSX) documents found in scope for compute()\");\n }\n if (tabular.length > MAX_COMPUTE_DOCUMENTS) {\n throw new EngineActionError(\n `more than ${MAX_COMPUTE_DOCUMENTS} tabular documents are in scope for compute() — ` +\n \"narrow the request with documentIds or sourceIds\",\n );\n }\n\n const totalChars = tabular.reduce((n, d) => n + String(d.text ?? \"\").length, 0);\n if (totalChars > MAX_COMPUTE_TEXT_CHARS) {\n throw new EngineActionError(\n `in-scope spreadsheet text too large to load (${totalChars} chars > ` +\n `${MAX_COMPUTE_TEXT_CHARS} cap) — narrow the request with documentIds or sourceIds`,\n );\n }\n\n const dfs: Record<string, Record<string, unknown>[]> = {};\n const multiDoc = tabular.length > 1;\n for (const doc of tabular) {\n // `computeOverFrames` masks the frames again below. That second pass is\n // idempotent — a mask placeholder (`[EMAIL]`) or a hash (lowercase hex)\n // never re-matches a builtin detector — and it is kept, rather than\n // skipped for this path, because the seam must be safe for a host that\n // hands it RAW frames; the pre-parse mask here is kept because it is the\n // documented contract and is the only pass that sees the `[Sheet: …]`\n // markers that become the frame keys.\n const maskedText = redactValueRecursive(doc.text, opts.config.redaction, {\n principals,\n secretKey: opts.config.secretKey,\n hooks: opts.hooks,\n }) as string | null;\n const parsed = parseSpreadsheetText(dfd, maskedText);\n for (const [sheetName, table] of Object.entries(parsed)) {\n const key = multiDoc ? `${doc.name}:${sheetName}` : sheetName;\n dfs[key] = table;\n }\n }\n\n const documents = tabular.map((doc) => ({\n id: String(doc.id),\n name: doc.name,\n sourceId: doc.source_id,\n }));\n\n if (!Object.keys(dfs).length) {\n throw new EngineActionError(\"in-scope documents did not parse into any usable dataframe\");\n }\n\n return computeOverFrames(dfs, instruction, {\n config: opts.config,\n modelCfg: opts.modelCfg,\n timeout: opts.timeout,\n hooks: opts.hooks,\n principals,\n documents,\n });\n}\n\n/** What `compute()` builds and `computeOverFrames` consumes: sheet name → rows. */\nexport type ComputeFrames = Record<string, Record<string, unknown>[]>;\n\n/** A source document behind a frame, as `compute()` reports it in `documentsUsed`. */\nexport interface ComputeDocument {\n id: string;\n name?: string | null;\n sourceId?: string | null;\n}\n\n/**\n * The guards every compute entry point runs before doing ANY work.\n *\n * Order matters and is asserted by the tests: the code-execution flag is\n * checked first so a disabled deployment never even reaches the LLM.\n * Returns the LLM config to use (`modelCfg`, else `config.llm`).\n */\nfunction checkComputePreconditions(\n config: ContextEngineConfig,\n modelCfg: LLMConfig | null | undefined,\n instruction: string,\n): LLMConfig {\n if (!config.enableCodeExecution) {\n throw new EngineActionError(\n \"compute() executes generated code and is disabled by default; set \" +\n \"enableCodeExecution=true only in a deployment with out-of-process/\" +\n \"container isolation.\",\n );\n }\n\n const llmCfg = modelCfg ?? config.llm;\n if (llmCfg == null) {\n throw new Error(\"compute() requires an LLM: pass modelCfg= or configure ContextEngineConfig.llm\");\n }\n if (!instruction?.trim()) {\n throw new EngineActionError(\"instruction must not be empty\");\n }\n return llmCfg;\n}\n\n/**\n * `redactValueRecursive` for `{sheet: rows[]}`: mask every string CELL and\n * every COLUMN NAME, leave every other value untouched.\n *\n * Column names ARE masked here even though `redactValueRecursive` leaves\n * object keys alone: an object key is a field name the caller chose, but a\n * spreadsheet header is document content (a sheet of one column per\n * customer e-mail is an ordinary shape) and it reaches the LLM verbatim in\n * the schema summary. Frame KEYS (sheet names) are the caller's choice and\n * are not masked — same rule as object keys.\n *\n * Numbers, booleans and dates are never rewritten: masking works on text,\n * and every non-string value is carried across as it is, so the arithmetic\n * the LLM is about to write still finds numbers.\n *\n * Two headers that mask to the same label (`[EMAIL]` twice) cannot both be\n * keys of one row object, so the second and later get a positional suffix\n * (`[EMAIL]`, `[EMAIL]_2`, …): every column survives and none silently\n * overwrites another. The Python port keeps them as duplicate labels —\n * a DataFrame can hold those, a plain object cannot.\n *\n * `policy=null`/empty is an exact no-op — the caller's own arrays are\n * returned, not copies (the `redactValueRecursive` contract).\n */\nfunction maskFrames(\n frames: ComputeFrames,\n policy: RedactionPolicy | null | undefined,\n opts: { principals: string[] | null; secretKey: string | Buffer | null; hooks?: Hooks | null },\n): ComputeFrames {\n if (policy == null || policy.isEmpty()) return frames;\n const mask = (text: string): string => redactValueRecursive(text, policy, opts) as string;\n\n const masked: ComputeFrames = {};\n for (const [name, rows] of Object.entries(frames)) {\n // One header mapping per sheet, in first-seen column order, so every\n // row of that sheet is relabelled the same way.\n const labels = new Map<string, string>();\n const taken = new Set<string>();\n for (const row of rows) {\n for (const column of Object.keys(row)) {\n if (labels.has(column)) continue;\n const base = mask(column);\n let label = base;\n for (let n = 2; taken.has(label); n++) label = `${base}_${n}`;\n taken.add(label);\n labels.set(column, label);\n }\n }\n masked[name] = rows.map((row) =>\n Object.fromEntries(\n Object.entries(row).map(([column, value]) => [\n labels.get(column) ?? column,\n typeof value === \"string\" ? mask(value) : value,\n ]),\n ),\n );\n }\n return masked;\n}\n\n/**\n * Compute an answer to `instruction` over caller-supplied tables.\n *\n * The frame-level seam under `compute()`: everything `compute()` does AFTER\n * it has turned its documents into `{sheet: rows[]}` lives here, so a host\n * that already holds the frames — an uploaded workbook, a connector's\n * sheet, a query result — can run the same prompt → code → sandbox path\n * without first ingesting a document to point at.\n *\n * `frames` keys are the sheet names the caller chose; the LLM sees them and\n * each row's columns exactly as it sees a parsed document's.\n *\n * The guards are `compute()`'s, in the same order and all BEFORE any LLM\n * call: `config.enableCodeExecution` off → `EngineActionError`; no LLM\n * (`modelCfg` or `config.llm`) → `Error`; blank instruction or empty\n * `frames` → `EngineActionError`; `timeout` clamped to 1..300 seconds.\n *\n * `config.redaction` is applied at the same two surfaces as `compute()`:\n *\n * 1. Every string cell and every column name is masked BEFORE the prompt is\n * built (`maskFrames`), so neither the schema summary the LLM reads nor\n * the rows its code runs against carry a raw value. Same intended\n * trade-off as `compute()`'s point 1 — a masked cell can change a\n * computed result, and that is correct.\n * 2. The returned object is swept whole through `redactValueRecursive`,\n * `code` included — `compute()`'s point 2, deliberately blunt.\n *\n * `hooks`, `principals` and `documents` are how `compute()` threads its own\n * context through; a host calling the seam directly normally leaves them at\n * their defaults (no error hook, trusted-internal `unless` evaluation, no\n * source documents — `documentsUsed` comes back empty).\n */\nexport async function computeOverFrames(\n frames: ComputeFrames,\n instruction: string,\n opts: {\n config: ContextEngineConfig;\n modelCfg?: LLMConfig | null;\n timeout?: number;\n hooks?: Hooks | null;\n principals?: string[] | null;\n documents?: ComputeDocument[] | null;\n },\n): Promise<Record<string, unknown>> {\n const llmCfg = checkComputePreconditions(opts.config, opts.modelCfg, instruction);\n if (!frames || !Object.keys(frames).length) {\n throw new EngineActionError(\"no tabular data to compute over\");\n }\n\n const hooks = opts.hooks ?? {};\n const principals = opts.principals ?? null;\n const documents = [...(opts.documents ?? [])];\n const timeout = Math.max(1, Math.min(Math.trunc(opts.timeout || DEFAULT_COMPUTE_TIMEOUT), 300));\n const redactOpts = { principals, secretKey: opts.config.secretKey, hooks };\n\n // Point 1 of this function's docstring: the LLM never sees a raw value.\n const dfs = maskFrames(frames, opts.config.redaction, redactOpts);\n\n const schemaLines = Object.entries(dfs).map(\n ([name, table]) =>\n `- ${name}: columns=${JSON.stringify(Object.keys(table[0] ?? {}))}, rows=${table.length}`,\n );\n const userPrompt = `Instruction: ${instruction}\\n\\nAvailable dataframes:\\n${schemaLines.join(\"\\n\")}`;\n\n let rawCode: string;\n let tokens: { input: number; output: number };\n try {\n [rawCode, tokens] = await callLlm(llmCfg, {\n system: COMPUTE_SYSTEM_PROMPT,\n user: userPrompt,\n jsonMode: false,\n });\n } catch (exc) {\n emitError(hooks, exc, { stage: \"compute_codegen\", instruction: instruction.slice(0, 200) });\n throw exc;\n }\n\n const code = stripCodeFences(rawCode);\n const execResult = await executeSafeCode(code, { dfs, documents }, { timeout });\n if (!execResult.success) {\n let maskedCode: string;\n try {\n maskedCode = redactValueRecursive(code.slice(0, 500), opts.config.redaction, {\n principals,\n secretKey: opts.config.secretKey,\n hooks: hooks,\n }) as string;\n } catch {\n maskedCode = \"<redaction failed: code omitted>\";\n }\n emitError(hooks, new Error(execResult.error || \"compute execution failed\"), {\n stage: \"compute_exec\",\n code: maskedCode,\n });\n }\n\n const result: Record<string, unknown> = {\n success: execResult.success,\n result: execResult.result,\n code,\n stdout: execResult.stdout ?? \"\",\n error: execResult.error,\n executionTime: execResult.executionTime,\n documentsUsed: documents.map((d) => d.id),\n providerTokens: {\n llm_input: tokens?.input ?? 0,\n llm_output: tokens?.output ?? 0,\n },\n };\n return redactValueRecursive(result, opts.config.redaction, {\n principals,\n secretKey: opts.config.secretKey,\n hooks: hooks,\n }) as Record<string, unknown>;\n}\n\nexport { applyAttributeUpdates } from \"./ingest.js\";\n\nexport function requirePandas(): never {\n throw new ExtraMissingError(\"compute\", \"danfojs-node\", \"compute() dataframes\");\n}\n","/**\n * Isolated-vm JavaScript sandbox — the compute-tool's execution engine.\n *\n * THIS IS BEST-EFFORT, NOT A SECURITY BOUNDARY — read before relying on it.\n *\n * `actions.compute()` (the only caller of this module) is **disabled by\n * default** (`ContextEngineConfig.enableCodeExecution`, default `false`).\n * That flag, not anything in this file, is the actual security boundary:\n * no deployment ships reachable code execution unless an operator\n * explicitly turns it on, and the only responsible reason to turn it on is\n * having put REAL isolation around the process that calls `compute()` — a\n * subprocess with dropped privileges, a container, gVisor, a WASM runtime.\n * Everything below raises the cost of an escape; none of it is a substitute\n * for that.\n *\n * The isolate has:\n * - a wall-clock timeout (`script.run({ timeout })`)\n * - no Node `fs` / `net` / `child_process` / `process` (the isolate does\n * not receive them; user code that names them is rejected up front)\n * - a whitelist of language builtins (JSON, Math, Date, Array, …)\n *\n * Returns `{ result, stdout? }` plus success/error/executionTime so a\n * failed run is a data result, not an thrown exception — matching the\n * Python sandbox's contract that `compute()` inspects `success`.\n */\nimport { CodeExecutionError, ExtraMissingError } from \"./errors.js\";\n\ntype IsolatedVm = {\n Isolate: new (opts: {\n memoryLimit: number;\n }) => {\n createContext(): Promise<{\n global: { set(k: string, v: unknown): Promise<void>; derefInto(): unknown };\n eval(code: string): Promise<unknown>;\n }>;\n compileScript(code: string): Promise<{\n run(ctx: unknown, opts: { timeout: number; copy: boolean }): Promise<unknown>;\n }>;\n dispose(): void;\n };\n};\n\nexport const DEFAULT_TIMEOUT = 30;\nexport const MAX_TIMEOUT = 300;\nexport const DEFAULT_MAX_OUTPUT_LENGTH = 50_000;\nexport const ISOLATE_MEMORY_MB = 128;\n\nconst BLOCKED_IDENTIFIERS =\n /\\b(process|require|module|exports|globalThis|global|Buffer|__dirname|__filename|child_process|worker_threads|fs|net|http|https|dgram|dns|tls|cluster|os|vm|v8|crypto|fetch|WebAssembly|Function|eval|importScripts|SharedArrayBuffer|Atomics|XMLHttpRequest|WebSocket|Worker)\\b/;\n\nconst DUNDER = /__proto__|constructor\\s*\\[|constructor\\s*\\./;\n\nexport interface SandboxResult {\n success: boolean;\n result: unknown;\n stdout?: string;\n stderr?: string;\n executionTime: number;\n error: string | null;\n}\n\nexport async function ensureSandboxAvailable(): Promise<IsolatedVm> {\n const specifier = \"isolated-vm\";\n try {\n return (await import(specifier)) as IsolatedVm;\n } catch {\n throw new ExtraMissingError(\"compute\", \"isolated-vm\", \"sandboxed compute()\");\n }\n}\n\n/** Static reject of blocked identifiers. Returns `[ok, error]`. */\nexport function isSafeCode(code: string): [boolean, string] {\n if (!code?.trim()) return [false, \"empty code\"];\n if (BLOCKED_IDENTIFIERS.test(code)) {\n return [false, \"code references a blocked identifier (fs/net/process/require/eval/…)\"];\n }\n if (DUNDER.test(code)) {\n return [false, \"code references a blocked constructor/prototype path\"];\n }\n if (/\\bimport\\s*\\(|^\\s*import\\s/m.test(code) || /\\bexport\\s/.test(code)) {\n return [false, \"ESM import/export is not allowed in the sandbox\"];\n }\n return [true, \"\"];\n}\n\nexport function validateCodeOnly(code: string): void {\n const [ok, err] = isSafeCode(code);\n if (!ok) throw new CodeExecutionError(err);\n}\n\n/**\n * Execute JavaScript in an isolated-vm isolate.\n *\n * `context` values are copied in (plain JSON-able data only). The isolate\n * does not receive Node's `fs`/`net`/`process`. A `result` binding is\n * expected; `console.log` is captured into `stdout`.\n *\n * Timeout is milliseconds of isolate CPU, clamped to 1–300 seconds.\n * Does not throw on user-code failure — returns `success: false`.\n */\nexport async function executeSafeCode(\n code: string,\n context: Record<string, unknown> = {},\n timeoutOrOpts: number | { timeout?: number; maxOutputLength?: number } = DEFAULT_TIMEOUT,\n): Promise<SandboxResult> {\n const timeoutSec =\n typeof timeoutOrOpts === \"number\" ? timeoutOrOpts : (timeoutOrOpts.timeout ?? DEFAULT_TIMEOUT);\n const maxOutputLength =\n typeof timeoutOrOpts === \"number\"\n ? DEFAULT_MAX_OUTPUT_LENGTH\n : (timeoutOrOpts.maxOutputLength ?? DEFAULT_MAX_OUTPUT_LENGTH);\n const timeout = Math.max(1, Math.min(timeoutSec, MAX_TIMEOUT));\n\n const empty = (error: string, executionTime = 0): SandboxResult => ({\n success: false,\n result: null,\n stdout: \"\",\n stderr: \"\",\n executionTime,\n error,\n });\n\n const [ok, err] = isSafeCode(code);\n if (!ok) {\n console.warn(\"[sandbox] Unsafe code detected: %s\", err);\n return empty(`Code validation failed: ${err}`);\n }\n\n const ivm = await ensureSandboxAvailable();\n const started = Date.now();\n const isolate = new ivm.Isolate({ memoryLimit: ISOLATE_MEMORY_MB });\n try {\n const jail = await isolate.createContext();\n const jailGlobal = jail.global;\n await jailGlobal.set(\"global\", jailGlobal.derefInto());\n\n // Whitelist: language builtins already exist inside the isolate.\n // Do NOT copy Node's process/require/fs. Copy caller context as JSON.\n let contextJson: string;\n try {\n contextJson = JSON.stringify(context ?? {});\n } catch (exc) {\n return empty(`context is not JSON-serializable: ${String(exc)}`);\n }\n\n await jail.eval(\n `const __ctx = ${contextJson};\n for (const k of Object.keys(__ctx)) { global[k] = __ctx[k]; }\n global.__stdout = [];\n global.console = {\n log: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n info: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n warn: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n error: (...args) => { global.__stdout.push(args.map(String).join(\" \")); },\n };\n void 0;`,\n );\n\n const wrapped = `\"use strict\";\nvar result = null;\n${code}\n({ result: result, stdout: (global.__stdout || []).join(\"\\\\n\") })`;\n\n const script = await isolate.compileScript(wrapped);\n const out = (await script.run(jail, {\n timeout: timeout * 1000,\n copy: true,\n })) as { result?: unknown; stdout?: string } | null;\n\n const executionTime = (Date.now() - started) / 1000;\n let result = out?.result ?? null;\n let stdout = typeof out?.stdout === \"string\" ? out.stdout : \"\";\n if (stdout.length > maxOutputLength) stdout = stdout.slice(0, maxOutputLength);\n try {\n const serialized = JSON.stringify(result);\n if (serialized && serialized.length > maxOutputLength) {\n result = { _truncated: true, _original_size: serialized.length };\n }\n } catch {\n result = String(result).slice(0, maxOutputLength);\n }\n\n return {\n success: true,\n result,\n stdout,\n stderr: \"\",\n executionTime,\n error: null,\n };\n } catch (exc) {\n const executionTime = (Date.now() - started) / 1000;\n const msg = String(exc);\n if (/timed out|timeout/i.test(msg)) {\n return empty(`Code execution timeout after ${timeout} seconds`, executionTime);\n }\n return empty(msg, executionTime);\n } finally {\n isolate.dispose();\n }\n}\n","/**\n * Structured data extraction, key normalization, and field resolution.\n *\n * extract — extractStructuredData() — one LLM call for type + fields\n * normalize— normalizeKey / inferDataType / validateAndNormalize / extractSynonyms\n * register — upsertRegistry() — the learned field-name registry, with embeddings\n * resolve — resolveFields() — question wording to canonical field name\n *\n * extractStructuredData never special-cases a provider: by the time ingest\n * calls it, extraction has already turned the document into text, so this\n * always runs on that text through callLlm. It never throws — an LLM\n * failure or unparseable response comes back as quality=\"failed\".\n *\n * resolveFields: exact key_norm match → synonym-array match → nearest\n * registry embedding by cosine distance, gated by\n * FIELD_RESOLUTION_SIMILARITY_FLOOR (0.55). Unresolved candidates are\n * simply absent from the returned mapping.\n */\nimport { randomUUID } from \"node:crypto\";\nimport type { Pool } from \"pg\";\nimport type { LLMConfig } from \"./config.js\";\nimport { safeJsonParse } from \"./extraction/json.js\";\nimport type { Embedder } from \"./providers/embeddings.js\";\nimport { callLlm } from \"./providers/llm.js\";\nimport { charScript } from \"./text.js\";\n\nexport const MAX_KEY_LENGTH = 100;\nexport const MAX_VALUE_LENGTH = 500;\nexport const MAX_EXAMPLE_VALUES = 10;\n\n/** Extraction schema version stamped on `documents.extraction_version`. */\nexport const EXTRACTION_VERSION = \"ce-structured-v1\";\n\n/** Cap on how much document text is sent to the LLM for one extraction call. */\nexport const MAX_EXTRACTION_CHARS = 120_000;\n\n/**\n * Cosine-similarity floor for the embedding fallback leg: below this,\n * \"nearest\" does not mean \"related\" — with no floor at all, EVERY candidate\n * resolved to SOME field once the registry had at least one embedded row.\n * Similarity = 1 - cosine distance (pgvector's `<=>`). 0.55 is deliberately\n * conservative for short-label vs short-label comparison.\n */\nexport const FIELD_RESOLUTION_SIMILARITY_FLOOR = 0.55;\n\nexport interface ExtractionResult {\n documentType: string | null;\n structuredDataRaw: Record<string, unknown>;\n structuredData: Record<string, unknown>;\n keysRaw: string[];\n keysNormalized: string[];\n quality: \"high\" | \"medium\" | \"low\" | \"failed\" | string;\n error: string | null;\n providerTokens: Record<string, number>;\n}\n\nconst EXTRACTION_SYSTEM_PROMPT = `You are an extraction engine. Analyze the document text you are given and return ONLY valid JSON (no markdown, no commentary, no backticks).\n\nGOALS:\nA) Extract structured information as FLAT KEY-VALUE pairs (industry-agnostic).\nB) Identify document_type and provide a brief quality observation.\n\nHARD RULES:\n1) Output MUST be STRICT JSON matching the schema below (no extra top-level keys).\n2) structured_data_raw MUST be a flat key-value object — no nested objects or arrays.\n If the source has lists/tables, flatten with indexed dot keys: items.0.description, items.0.value, items.1.description, ...\n3) Unknown/unreadable values MUST be null (never guess).\n4) Numeric fields: only use a number when you can read it from the text; store the raw string too if it needs disambiguation.\n5) Always include document_type (a short descriptive type, or \"unknown\").\n\nRETURN JSON SCHEMA (EXACT):\n{\n \"document_type\": \"<descriptive_type | unknown>\",\n \"structured_data_raw\": {\"<key>\": \"<value or null>\"},\n \"structured_quality\": {\"readability\": \"<low|medium|high>\"}\n}`;\n\nfunction buildExtractionPrompt(\n fieldHints: Record<string, unknown>[] | null | undefined,\n attempt = 0,\n): string {\n let prompt = EXTRACTION_SYSTEM_PROMPT;\n if (fieldHints?.length) {\n const lines: string[] = [];\n for (const hint of fieldHints) {\n if (!hint || typeof hint !== \"object\") continue;\n const key = hint.key ?? hint.name;\n if (!key) continue;\n const desc = hint.description ?? hint.type;\n lines.push(`- ${String(key)}${desc ? ` (${String(desc)})` : \"\"}`);\n }\n if (lines.length) {\n prompt +=\n \"\\n\\nPRIORITIZE extracting these fields if present (use these exact \" +\n \"names as keys in structured_data_raw), in addition to anything else \" +\n \"you find:\\n\" +\n lines.join(\"\\n\");\n }\n }\n if (attempt > 0) {\n prompt +=\n \"\\n\\nIMPORTANT: The previous attempt was not valid JSON. Double-check \" +\n \"escaping of quotes/newlines and return JSON only.\";\n }\n return prompt;\n}\n\n/**\n * Extract structured key/value data from already-extracted document text.\n *\n * Never throws — an LLM failure or unparseable response comes back as\n * `ExtractionResult(quality=\"failed\", error=...)` so ingest can complete\n * the rest of ingestion instead of losing already-computed chunks.\n *\n * Tokens are tracked CUMULATIVELY across every attempt whose `callLlm`\n * actually completed, even one whose response then failed to parse —\n * that attempt still consumed real, billable provider tokens.\n */\nexport async function extractStructuredData(\n text: string,\n opts: {\n llmCfg: LLMConfig;\n fieldHints?: Record<string, unknown>[] | null;\n maxChars?: number;\n maxRetries?: number;\n },\n): Promise<ExtractionResult> {\n if (!text?.trim()) {\n return {\n documentType: null,\n structuredDataRaw: {},\n structuredData: {},\n keysRaw: [],\n keysNormalized: [],\n quality: \"failed\",\n error: \"empty document text\",\n providerTokens: {},\n };\n }\n\n const maxChars = opts.maxChars ?? MAX_EXTRACTION_CHARS;\n const body = text.length <= maxChars ? text : text.slice(0, maxChars);\n const maxRetries = opts.maxRetries ?? 1;\n let lastError: string | null = null;\n let totalInputTokens = 0;\n let totalOutputTokens = 0;\n\n for (let attempt = 0; attempt <= maxRetries; attempt++) {\n const prompt = buildExtractionPrompt(opts.fieldHints, attempt);\n let raw: string;\n let tokens: { input: number; output: number };\n try {\n [raw, tokens] = await callLlm(opts.llmCfg, {\n system: prompt,\n user: `Document text:\\n\\n${body}`,\n jsonMode: true,\n });\n } catch (exc) {\n lastError = String(exc);\n console.warn(\"structured extraction attempt %d: callLlm failed: %s\", attempt, exc);\n continue;\n }\n\n totalInputTokens += tokens?.input ?? 0;\n totalOutputTokens += tokens?.output ?? 0;\n\n let parsed: Record<string, unknown>;\n try {\n parsed = safeJsonParse(raw, \"object\") as Record<string, unknown>;\n } catch (exc) {\n lastError = String(exc);\n console.warn(\"structured extraction attempt %d: unparseable response: %s\", attempt, exc);\n continue;\n }\n\n const docType = (parsed.document_type as string) || \"unknown\";\n let rawData = parsed.structured_data_raw;\n if (!rawData || typeof rawData !== \"object\" || Array.isArray(rawData)) rawData = {};\n const qualityInfo = parsed.structured_quality;\n const readability =\n qualityInfo && typeof qualityInfo === \"object\" && !Array.isArray(qualityInfo)\n ? (qualityInfo as { readability?: string }).readability\n : undefined;\n const quality =\n readability === \"high\" || readability === \"medium\" || readability === \"low\" ? readability : \"medium\";\n\n const normalized = validateAndNormalize(rawData as Record<string, unknown>);\n\n return {\n documentType: docType,\n structuredDataRaw: rawData as Record<string, unknown>,\n structuredData: normalized,\n keysRaw: Object.keys(rawData as Record<string, unknown>),\n keysNormalized: Object.keys(normalized),\n quality,\n error: null,\n providerTokens: { llm_input: totalInputTokens, llm_output: totalOutputTokens },\n };\n }\n\n return {\n documentType: null,\n structuredDataRaw: {},\n structuredData: {},\n keysRaw: [],\n keysNormalized: [],\n quality: \"failed\",\n error: lastError || \"extraction failed\",\n providerTokens: { llm_input: totalInputTokens, llm_output: totalOutputTokens },\n };\n}\n\n/**\n * True if `ch` should be KEPT verbatim by `normalizeKey`.\n *\n * Unicode-aware: an ASCII `[^a-z0-9_]` class silently DROPPED every\n * non-Latin field name — \"税额\"/\"مبلغ الضريبة\"/\"Сумма\" all normalized to\n * `\"\"` and got skipped entirely. `charScript` reads a character's Unicode\n * category (letters AND combining marks, any script) instead of a\n * hardcoded Latin range. `/\\d/u` is Unicode-aware on its own.\n */\nfunction isKeyChar(ch: string): boolean {\n return Boolean(charScript(ch)) || /\\d/u.test(ch) || ch === \"_\";\n}\n\n/**\n * Normalize a metadata key: lowercase, non-key-chars -> `_`, dot-paths\n * kept, collapsed/stripped underscores, truncated to `MAX_KEY_LENGTH`.\n *\n * Unicode letters/digits in ANY script are preserved verbatim (only\n * separators/punctuation become `_`) — see `isKeyChar`.\n */\nexport function normalizeKey(key: unknown): string {\n if (!key || typeof key !== \"string\") return \"\";\n\n const segments = key.toLowerCase().split(\".\");\n const processed: string[] = [];\n for (const raw of segments) {\n let segment = \"\";\n for (const ch of raw) segment += isKeyChar(ch) ? ch : \"_\";\n segment = segment.replace(/_+/g, \"_\").replace(/^_|_$/g, \"\");\n if (segment) processed.push(segment);\n }\n let normalized = processed.join(\".\");\n if (normalized.length > MAX_KEY_LENGTH) normalized = normalized.slice(0, MAX_KEY_LENGTH);\n return normalized;\n}\n\n/** Infer 'number' | 'date' | 'bool' | 'string' from a key's suffix, then the value itself. */\nexport function inferDataType(key: string, value: unknown): string {\n if (typeof key !== \"string\") return \"string\";\n\n const keyLower = key.toLowerCase();\n if (keyLower.endsWith(\"_num\") || keyLower.endsWith(\"_number\")) return \"number\";\n if (keyLower.endsWith(\"_date\") || keyLower.endsWith(\"_datetime\")) return \"date\";\n if (keyLower.endsWith(\"_bool\") || keyLower.endsWith(\"_boolean\")) return \"bool\";\n\n if (value == null) return \"string\";\n if (typeof value === \"boolean\") return \"bool\";\n if (typeof value === \"string\") {\n const valueLower = value.toLowerCase().trim();\n if ([\"true\", \"false\", \"yes\", \"no\", \"1\", \"0\"].includes(valueLower)) return \"bool\";\n try {\n const d = new Date(value.replace(\"Z\", \"+00:00\"));\n if (!Number.isNaN(d.getTime()) && /\\d{4}-\\d{2}-\\d{2}/.test(value)) return \"date\";\n } catch {\n /* not a date */\n }\n if (value !== \"\" && !Number.isNaN(Number(value))) return \"number\";\n }\n if (typeof value === \"number\") return \"number\";\n if (value instanceof Date) return \"date\";\n return \"string\";\n}\n\n/**\n * Normalize keys, validate/convert values by inferred type, reject nested\n * structures, truncate over-long strings.\n */\nexport function validateAndNormalize(rawData: Record<string, unknown>): Record<string, unknown> {\n if (!rawData || typeof rawData !== \"object\" || Array.isArray(rawData)) {\n throw new Error(\"Input must be a dictionary\");\n }\n\n const normalized: Record<string, unknown> = {};\n\n for (const [rawKey, rawValue] of Object.entries(rawData)) {\n if (rawValue && typeof rawValue === \"object\") {\n console.warn(\"skipping nested structure for key '%s'\", rawKey);\n continue;\n }\n\n const normKey = normalizeKey(rawKey);\n if (!normKey) continue;\n\n const dataType = inferDataType(normKey, rawValue);\n\n try {\n if (dataType === \"number\") {\n if (rawValue == null || rawValue === \"\") {\n normalized[normKey] = null;\n } else {\n const asNum = Number(rawValue);\n if (!Number.isNaN(asNum)) {\n normalized[normKey] = asNum;\n } else {\n const match = String(rawValue).match(/[-+]?\\d[\\d,]*\\.?\\d*/);\n if (match) {\n const n = Number(match[0].replace(/,/g, \"\"));\n normalized[normKey] = Number.isNaN(n) ? null : n;\n } else {\n normalized[normKey] = null;\n }\n }\n }\n } else if (dataType === \"date\") {\n if (rawValue == null || rawValue === \"\") {\n normalized[normKey] = null;\n } else if (rawValue instanceof Date) {\n normalized[normKey] = rawValue.toISOString();\n } else {\n const d = new Date(String(rawValue).replace(\"Z\", \"+00:00\"));\n normalized[normKey] = Number.isNaN(d.getTime()) ? String(rawValue) : d.toISOString();\n }\n } else if (dataType === \"bool\") {\n if (rawValue == null || rawValue === \"\") {\n normalized[normKey] = null;\n } else if (typeof rawValue === \"boolean\") {\n normalized[normKey] = rawValue;\n } else {\n const valueStr = String(rawValue).toLowerCase().trim();\n if ([\"true\", \"yes\", \"1\"].includes(valueStr)) normalized[normKey] = true;\n else if ([\"false\", \"no\", \"0\"].includes(valueStr)) normalized[normKey] = false;\n else normalized[normKey] = String(rawValue);\n }\n } else {\n if (rawValue == null) {\n normalized[normKey] = null;\n } else {\n let strValue = String(rawValue);\n if (strValue.length > MAX_VALUE_LENGTH) strValue = strValue.slice(0, MAX_VALUE_LENGTH);\n normalized[normKey] = strValue;\n }\n }\n } catch (exc) {\n console.error(\"error processing field '%s': %s\", rawKey, exc);\n normalized[normKey] = rawValue != null ? String(rawValue) : null;\n }\n }\n\n return normalized;\n}\n\n/** Build `{normKey: [synonym variations]}` from raw/normalized key pairs. */\nexport function extractSynonyms(rawKeys: string[], normKeys: string[]): Record<string, string[]> {\n if (rawKeys.length !== normKeys.length) {\n console.error(\"extractSynonyms: length mismatch raw=%d norm=%d\", rawKeys.length, normKeys.length);\n return {};\n }\n\n const synonymMap: Record<string, string[]> = {};\n\n for (let i = 0; i < rawKeys.length; i++) {\n const rawKey = rawKeys[i]!;\n const normKey = normKeys[i]!;\n if (!normKey) continue;\n synonymMap[normKey] ??= [];\n const synonyms = new Set<string>();\n\n if (rawKey) {\n synonyms.add(rawKey);\n synonyms.add(rawKey.toLowerCase());\n synonyms.add(rawKey.toUpperCase());\n const camel = rawKey.replace(/[ _.]/g, \"\");\n if (camel) {\n synonyms.add(camel[0]!.toLowerCase() + camel.slice(1));\n synonyms.add(camel[0]!.toUpperCase() + camel.slice(1));\n }\n }\n\n synonyms.add(normKey);\n synonyms.add(normKey.replace(/_/g, \"\"));\n synonyms.add(normKey.replace(/_/g, \" \"));\n synonyms.add(normKey.replace(/\\./g, \"_\"));\n synonyms.add(normKey.replace(/_/g, \".\"));\n synonyms.add(normKey.toUpperCase());\n synonyms.add(\n normKey\n .split(\"_\")\n .map((w) => (w ? w[0]!.toUpperCase() + w.slice(1) : w))\n .join(\" \"),\n );\n\n synonyms.delete(\"\");\n synonyms.delete(normKey);\n\n const seen = new Set<string>();\n const unique: string[] = [];\n for (const syn of [...[...synonyms].sort(), ...synonymMap[normKey]!]) {\n if (!seen.has(syn)) {\n seen.add(syn);\n unique.push(syn);\n }\n }\n synonymMap[normKey] = unique;\n }\n\n return synonymMap;\n}\n\nfunction embeddingText(keyNorm: string, synonyms: readonly string[]): string {\n return `${keyNorm} ${synonyms.join(\" \")}`.trim();\n}\n\nfunction vectorLiteral(vector: readonly number[]): string {\n return `[${vector.map((x) => Number(x)).join(\",\")}]`;\n}\n\ninterface RegistryRow {\n id: string;\n key_norm: string;\n synonyms: string[] | null;\n embedding: unknown;\n example_values: string[] | null;\n frequency: number | null;\n}\n\n/**\n * Upsert `structuredData`'s keys into the field registry.\n *\n * Generates a field-name embedding (`Embedder.embed(..., kind=\"document\")`)\n * for any NEW field, or one whose synonym set changed, or one that somehow\n * has no embedding yet — batched into one `embed()` call per ingest rather\n * than one per field.\n *\n * Returns the embedding provider's token count (0 if nothing needed embedding).\n */\nexport async function upsertRegistry(opts: {\n pool: Pool;\n embedder: Embedder;\n sourceId: string | null;\n docType: string;\n structuredData: Record<string, unknown>;\n keysRaw: string[];\n keysNormalized: string[];\n}): Promise<number> {\n if (!opts.structuredData || !opts.docType) return 0;\n\n const synonymsMap = extractSynonyms(opts.keysRaw, opts.keysNormalized);\n const toEmbed: Array<[string, string]> = [];\n\n for (const [keyNorm, value] of Object.entries(opts.structuredData)) {\n const dataType = inferDataType(keyNorm, value);\n const exampleValue = value == null ? null : String(value).slice(0, 200);\n const newSynonyms = [\n ...new Set((synonymsMap[keyNorm] ?? []).map((s) => normalizeKey(s)).filter(Boolean)),\n ].sort();\n\n const existing = await opts.pool.query<RegistryRow>(\n `SELECT id, key_norm, synonyms, embedding, example_values, frequency\n FROM context_engine_structured_keys\n WHERE doc_type = $1 AND key_norm = $2\n AND (($3::text IS NULL AND source_id IS NULL) OR source_id = $3)\n LIMIT 1`,\n [opts.docType, keyNorm, opts.sourceId],\n );\n\n let rowId: string;\n let needsEmbedding = false;\n let synonymsForEmbed: string[] = newSynonyms;\n\n if (!existing.rows[0]) {\n rowId = randomUUID();\n await opts.pool.query(\n `INSERT INTO context_engine_structured_keys\n (id, source_id, doc_type, key_norm, data_type, synonyms, example_values, frequency)\n VALUES ($1,$2,$3,$4,$5,$6,$7,1)`,\n [\n rowId,\n opts.sourceId,\n opts.docType,\n keyNorm,\n dataType,\n newSynonyms,\n exampleValue ? [exampleValue] : [],\n ],\n );\n needsEmbedding = true;\n } else {\n const row = existing.rows[0];\n rowId = row.id;\n const freq = (row.frequency ?? 0) + 1;\n let exampleValues = [...(row.example_values ?? [])];\n if (exampleValue && !exampleValues.includes(exampleValue)) {\n exampleValues = [...exampleValues.slice(0, MAX_EXAMPLE_VALUES - 1), exampleValue];\n }\n const currentSyn = [...(row.synonyms ?? [])].sort();\n let merged = currentSyn;\n if (newSynonyms.length) {\n merged = [...new Set([...currentSyn, ...newSynonyms])].sort();\n if (merged.join(\"\\0\") !== currentSyn.join(\"\\0\")) needsEmbedding = true;\n }\n if (row.embedding == null) needsEmbedding = true;\n synonymsForEmbed = merged;\n await opts.pool.query(\n `UPDATE context_engine_structured_keys\n SET frequency = $1, example_values = $2, synonyms = $3, updated_at = now()\n WHERE id = $4`,\n [freq, exampleValues, merged, rowId],\n );\n }\n\n if (needsEmbedding) toEmbed.push([rowId, embeddingText(keyNorm, synonymsForEmbed)]);\n }\n\n if (!toEmbed.length) return 0;\n\n const [vectors, tokens] = await opts.embedder.embed(\n toEmbed.map(([, text]) => text),\n { kind: \"document\" },\n );\n for (let i = 0; i < toEmbed.length; i++) {\n const vec = vectors[i];\n if (!vec) continue;\n await opts.pool.query(\n `UPDATE context_engine_structured_keys SET embedding = $1::vector, updated_at = now() WHERE id = $2`,\n [vectorLiteral(vec), toEmbed[i]![0]],\n );\n }\n return tokens || 0;\n}\n\n/**\n * Resolve free-text field-name candidates to canonical registry keys.\n *\n * Cheapest-first: normalize → exact `key_norm` match → synonym-array\n * match → nearest registry embedding (cosine, gated by\n * FIELD_RESOLUTION_SIMILARITY_FLOOR). Unresolved candidates are simply\n * absent from the returned `{candidate: canonicalKey}` mapping.\n *\n * All still-unresolved candidates are embedded in ONE batched `embed()` call.\n *\n * No ACL: registry rows are field NAMES, not document content, so source_id\n * scoping is enough.\n */\nexport async function resolveFields(\n candidates: readonly string[],\n opts: {\n pool: Pool;\n embedder: Embedder;\n sourceIds?: string[] | null;\n docType?: string | null;\n },\n): Promise<Record<string, string>> {\n const normByCandidate: Record<string, string> = {};\n for (const c of candidates) {\n const n = normalizeKey(c);\n if (n) normByCandidate[c] = n;\n }\n if (!Object.keys(normByCandidate).length) return {};\n\n const { rows } = await opts.pool.query<{\n key_norm: string;\n synonyms: string[] | null;\n embedding: unknown;\n }>(\n `SELECT key_norm, synonyms, embedding\n FROM context_engine_structured_keys\n WHERE ($1::text[] IS NULL OR source_id = ANY($1))\n AND ($2::text IS NULL OR doc_type = $2)`,\n [opts.sourceIds ?? null, opts.docType ?? null],\n );\n\n const resolved: Record<string, string> = {};\n const unresolved: Array<[string, string]> = [];\n for (const [candidate, norm] of Object.entries(normByCandidate)) {\n const exact = rows.find((r) => r.key_norm === norm);\n if (exact) {\n resolved[candidate] = exact.key_norm;\n continue;\n }\n const synonymHit = rows.find((r) => (r.synonyms ?? []).includes(norm));\n if (synonymHit) {\n resolved[candidate] = synonymHit.key_norm;\n continue;\n }\n unresolved.push([candidate, norm]);\n }\n\n if (unresolved.length && rows.some((r) => r.embedding != null)) {\n let vectors: number[][] = [];\n try {\n [vectors] = await opts.embedder.embed(\n unresolved.map(([, norm]) => norm),\n { kind: \"query\" },\n );\n } catch (exc) {\n console.warn(\"resolveFields: batch embedding failed: %s\", exc);\n vectors = [];\n }\n\n for (let i = 0; i < unresolved.length; i++) {\n const vec = vectors[i];\n if (!vec) continue;\n const match = await nearestField(opts.pool, vec, opts.sourceIds ?? null, opts.docType ?? null);\n if (match) resolved[unresolved[i]![0]] = match;\n }\n }\n\n return resolved;\n}\n\nasync function nearestField(\n pool: Pool,\n vector: readonly number[],\n sourceIds: string[] | null,\n docType: string | null,\n): Promise<string | null> {\n const { rows } = await pool.query<{ key_norm: string; distance: number }>(\n `SELECT key_norm, (embedding <=> $1::vector) AS distance\n FROM context_engine_structured_keys\n WHERE embedding IS NOT NULL\n AND ($2::text[] IS NULL OR source_id = ANY($2))\n AND ($3::text IS NULL OR doc_type = $3)\n ORDER BY embedding <=> $1::vector\n LIMIT 1`,\n [vectorLiteral(vector), sourceIds, docType],\n );\n const row = rows[0];\n if (!row) return null;\n const similarity = 1.0 - Number(row.distance);\n if (similarity < FIELD_RESOLUTION_SIMILARITY_FLOOR) return null;\n return String(row.key_norm);\n}\n","/**\n * Sentinels distinguishing omitted arguments from real values, including null.\n *\n * UNSET: \"argument omitted\" vs null (e.g. updateDocument acl=null means unrestricted).\n * TRUSTED: trusted caller, ACL filtering disabled. Truthy on purpose so\n * `if (principals)` does not treat a trusted caller as anonymous.\n */\n\nexport const UNSET: unique symbol = Symbol.for(\"context_engine.UNSET\");\nexport type Unset = typeof UNSET;\n\n// Symbol.for (the process-global registry), NEVER a class instance: the\n// package ships per-entry bundles (tsup splitting:false), so dist/index.js\n// and dist/mcp.js each carry their own copy of this module. A class instance\n// is a different object per copy, and `result === TRUSTED` then fails for an\n// app importing TRUSTED from one entry while mounting another — every tool\n// call on a correctly configured trusted mount errors. UNSET above survives\n// bundle duplication for exactly this reason; TRUSTED must too.\nexport const TRUSTED: unique symbol = Symbol.for(\"context_engine.TRUSTED\");\n\n// The SCOPE ceiling's \"no ceiling at all\". Symbol.for for the same reason\n// TRUSTED is: a class instance breaks `===` across per-entry bundles.\n//\n// Scope has the same shape of danger `principals` has, so it gets the same\n// treatment. A host maps its own word — project, matter, workspace, customer —\n// onto a list of source ids, and that ceiling is what the mounted tool may\n// ever reach. `null`/omitted is the spelling of an oversight, and would\n// otherwise mean a corpus-wide search: exactly the accident worth making\n// impossible. A host running one shared private corpus types UNSCOPED and\n// thereby says it meant it.\nexport const UNSCOPED: unique symbol = Symbol.for(\"context_engine.UNSCOPED\");\nexport type Trusted = typeof TRUSTED;\n\nexport type Principals = string[] | null | typeof TRUSTED | undefined;\n\n// Once per method, matching Python's default warnings filter — before the\n// dedup, a legacy caller re-triggered the warning on EVERY request.\nconst warnedMethods = new Set<string>();\n\nexport function resolvePrincipals(value: Principals, method: string): string[] | null {\n if (value === TRUSTED) return null;\n if (value === undefined || value === null) {\n // `undefined` (omitted) is the same legacy trusted spelling as `null` —\n // Python's omitted `principals` defaults to None and warns identically.\n if (!warnedMethods.has(method)) {\n warnedMethods.add(method);\n process.emitWarning(\n `${method}(principals=${value === null ? \"null\" : \"undefined\"}) means TRUSTED CALLER — ` +\n `access control is disabled and every document is returned. If that is what ` +\n `you want, pass principals=TRUSTED (from @promptev/context-engine) to say so ` +\n `explicitly. If you meant 'no authenticated user', pass principals=[] instead. ` +\n `Passing null/omitting will raise in 1.0.`,\n { type: \"DeprecationWarning\", code: \"CE_PRINCIPALS_NULL\" },\n );\n }\n return null;\n }\n if (!Array.isArray(value)) {\n // NEVER coerce: a bare string (or any other object) coercing to null\n // used to mean full-corpus TRUSTED — an ACL bypass a caller could hit\n // with one wrong type. Fail loudly instead.\n throw new TypeError(\n `${method}(principals=...) must be an array of principal strings, [] for an ` +\n `anonymous caller, or TRUSTED — got ${typeof value}. A non-array value must ` +\n `never silently disable ACL filtering.`,\n );\n }\n return value;\n}\n\nexport function isTrusted(principals: string[] | null | undefined): boolean {\n return principals === null || principals === undefined;\n}\n","/**\n * The ingestion pipeline: bytes/text in, document + chunk rows out.\n *\n * `ContextEngine.ingest()` is a thin shell over `runIngest()` here. One call\n * ingests ONE document through these stages:\n *\n * resolve input → extract → hygiene → dedup check → claim row (queued →\n * processing) → chunk → embed → upsert chunks → graph stage → complete\n *\n * Three rules worth stating up front:\n *\n * 1. **Dedup is by `ingest_hash` + mode.** Re-ingesting identical content at\n * the same-or-lower retrieval mode returns a `skipped` report row, costs\n * zero units, and does not re-process the content. Re-ingesting at a\n * HIGHER mode (hybrid → graph) re-processes fully so the graph stage can\n * run. **`skipped` is about CONTENT, not about the caller's declared\n * attributes**: `acl` / `name` / `description` / `metaData` are not part\n * of the hash, so a skip still applies them (`refreshOnSkip`).\n * 2. **The database is locked to one embedding model.**\n * 3. **Per-document failures are reported, not raised.** Only pre-flight\n * errors (bad arguments, graph disabled, embedding-model mismatch) raise.\n * Hash-redaction misconfig is the exception: finalize to `failed` THEN\n * rethrow so the row is not stuck `processing`.\n * 4. **`batch=true` stops after chunking.** Chunks are stored with\n * `embedding IS NULL`, the document lands at `status=\"batch_pending\"`.\n */\nimport { randomUUID } from \"node:crypto\";\nimport { readFileSync } from \"node:fs\";\nimport { basename } from \"node:path\";\nimport type { Pool } from \"pg\";\nimport {\n calculateContentHash,\n chunkerName,\n dedupLines,\n detectLanguage,\n inferMime,\n looksMostlyBoilerplate,\n routeToChunker,\n sanitizeText,\n sha256Bytes,\n} from \"./chunkers.js\";\nimport type { ContextEngineConfig } from \"./config.js\";\nimport { extract } from \"./extraction/index.js\";\nimport { emitError, emitProgress, emitUsage, type Hooks, type ProgressEvent } from \"./hooks.js\";\nimport { submitEmbeddingBatch } from \"./providers/batch.js\";\nimport type { Embedder } from \"./providers/embeddings.js\";\nimport { applyRedaction, type RedactionPolicy } from \"./redaction.js\";\nimport type { ChunkRow, StorageBackend } from \"./storage.js\";\nimport {\n extractStructuredData,\n EXTRACTION_VERSION as STRUCTURED_EXTRACTION_VERSION,\n upsertRegistry,\n} from \"./structured.js\";\nimport { type DocumentReport, type IngestReport, type UsageEvent, unitsForFile } from \"./usage.js\";\n\nexport type Mode = \"hybrid\" | \"graph\";\n\n/** Retrieval modes are ordered: ingesting at a higher mode than the stored one is a backfill. */\nexport const MODE_RANK: Record<string, number> = { hybrid: 0, graph: 1 };\n\nexport const EMBED_BATCH = 256;\nexport const DOCX_CHARS_PER_PAGE = 1500;\nexport const DEFAULT_MIME = \"text/plain\";\n\nconst DOCX_MIMES = new Set([\n \"application/vnd.openxmlformats-officedocument.wordprocessingml.document\",\n \"application/msword\",\n]);\n\nconst EXTRA_EXT_MIMES: Record<string, string> = {\n \".pdf\": \"application/pdf\",\n \".docx\": \"application/vnd.openxmlformats-officedocument.wordprocessingml.document\",\n \".pptx\": \"application/vnd.openxmlformats-officedocument.presentationml.presentation\",\n \".png\": \"image/png\",\n \".jpg\": \"image/jpeg\",\n \".jpeg\": \"image/jpeg\",\n \".gif\": \"image/gif\",\n \".bmp\": \"image/bmp\",\n \".tif\": \"image/tiff\",\n \".tiff\": \"image/tiff\",\n \".webp\": \"image/webp\",\n \".txt\": \"text/plain\",\n \".tsv\": \"text/tab-separated-values\",\n};\n\nconst MIME_BY_EXT: Record<string, string> = { ...EXTRA_EXT_MIMES };\n\nexport interface IngestRequest {\n content?: Buffer | null;\n filename?: string | null;\n text?: string | null;\n name?: string | null;\n description?: string | null;\n sourceId?: string | null;\n externalId?: string | null;\n metaData?: Record<string, unknown> | null;\n acl?: string[] | null;\n mode?: Mode;\n batch?: boolean;\n extractStructured?: boolean;\n fieldHints?: Record<string, unknown>[] | null;\n}\n\nexport interface Prepared {\n text: string;\n mime: string;\n pages: number | null;\n slides: number | null;\n sizeBytes: number;\n llmPictures: number;\n mediaOnly: boolean;\n providerTokens: Record<string, number>;\n contentHash: string | null;\n isMarkdown: boolean;\n // Carried from `Extracted` so the document plane can persist and report\n // it. null for a caller-supplied `text` ingest: nothing was extracted, so\n // there is nothing to have failed.\n unreadableReason: string | null;\n unreadablePages: number;\n}\n\nexport interface Decision {\n action: \"process\" | \"skip\" | \"upgrade\";\n documentId: string | null;\n}\n\ninterface ContentPayload {\n text: string;\n mime: string;\n lang: string | null;\n ingestHash: Buffer;\n mode: string;\n metaUpdates: Record<string, unknown>;\n documentType?: string | null;\n structuredData?: Record<string, unknown> | null;\n structuredKeys?: string[] | null;\n structuredQuality?: Record<string, unknown> | null;\n extractionVersion?: string | null;\n}\n\nexport interface GraphStageArgs {\n documentId: string;\n chunks: ChunkRow[] | null;\n config: ContextEngineConfig;\n hooks: Hooks;\n}\n\nexport type GraphStage = (args: GraphStageArgs) => Promise<number> | number;\n\nfunction isUniqueViolation(exc: unknown): boolean {\n return (\n typeof exc === \"object\" && exc !== null && \"code\" in exc && (exc as { code?: string }).code === \"23505\"\n );\n}\n\nexport function mimeForFilename(filename: string | null | undefined): string | null {\n if (!filename) return null;\n const upstream = inferMime(null, filename);\n if (upstream) return upstream;\n const dot = filename.lastIndexOf(\".\");\n const ext = dot >= 0 ? filename.slice(dot).toLowerCase() : \"\";\n if (ext && MIME_BY_EXT[ext]) return MIME_BY_EXT[ext]!;\n return null;\n}\n\n/**\n * Normalize the three input modes to `[content, filename, text]`.\n *\n * Exactly one of `file` / `content` / `text` must be given — passing none\n * or more than one is a programming error, so it raises before any work.\n */\nexport function resolveSource(opts: {\n file?: unknown;\n content?: Buffer | Uint8Array | null;\n filename?: string | null;\n text?: string | null;\n}): [Buffer | null, string | null, string | null] {\n const provided = (\n [\n [\"file\", opts.file],\n [\"content\", opts.content],\n [\"text\", opts.text],\n ] as const\n ).filter(([, v]) => v != null);\n if (provided.length !== 1) {\n throw new Error(\n \"exactly one of file=, content= or text= must be provided \" +\n `(got: ${provided.map(([n]) => n).join(\", \") || \"none\"})`,\n );\n }\n\n if (opts.text != null) return [null, opts.filename ?? null, opts.text];\n if (opts.content != null) {\n const buf = Buffer.isBuffer(opts.content) ? opts.content : Buffer.from(opts.content);\n return [buf, opts.filename ?? null, null];\n }\n\n const file = opts.file;\n if (typeof file === \"string\") {\n const data = readFileSync(file);\n return [data, opts.filename || basename(file), null];\n }\n if (\n file &&\n typeof file === \"object\" &&\n \"buffer\" in (file as object) &&\n Buffer.isBuffer((file as { buffer: Buffer }).buffer)\n ) {\n const f = file as { buffer: Buffer; originalname?: string; name?: string };\n return [f.buffer, opts.filename || f.originalname || f.name || null, null];\n }\n if (Buffer.isBuffer(file)) return [file, opts.filename ?? null, null];\n if (file && typeof file === \"object\" && typeof (file as { read?: unknown }).read === \"function\") {\n const data = (file as { read: () => Buffer | string | Uint8Array }).read();\n const buf = typeof data === \"string\" ? Buffer.from(data, \"utf8\") : Buffer.from(data);\n const handleName = (file as { name?: unknown }).name;\n const inferred = typeof handleName === \"string\" ? basename(handleName) : null;\n return [buf, opts.filename || inferred, null];\n }\n throw new Error(\"file= must be a path or an object with a .read() method\");\n}\n\n/** Produce the document's final text + the numbers billing/chunking need. */\nexport async function prepare(opts: {\n content: Buffer | null;\n filename: string | null;\n text: string | null;\n name: string | null;\n config: ContextEngineConfig;\n hooks: Hooks;\n}): Promise<Prepared> {\n if (opts.text != null) {\n const clean = sanitizeText(opts.text) || \"\";\n const mime = inferMime(clean, opts.filename || opts.name) || DEFAULT_MIME;\n return {\n text: clean,\n mime,\n pages: null,\n slides: null,\n sizeBytes: Buffer.byteLength(clean, \"utf8\"),\n llmPictures: 0,\n mediaOnly: false,\n providerTokens: {},\n contentHash: calculateContentHash({ fullText: clean }),\n isMarkdown: false,\n unreadableReason: null,\n unreadablePages: 0,\n };\n }\n\n const content = opts.content!;\n const fname = opts.filename || opts.name || \"\";\n const mime = mimeForFilename(fname) || \"\";\n const extracted = await extract(content, fname, mime || null, {\n visionLlm: opts.config.visionLlm,\n extraction: opts.config.extraction,\n hooks: opts.hooks,\n });\n\n let body = sanitizeText(extracted.text || \"\") || \"\";\n const effectiveMime = mime || inferMime(body, fname) || DEFAULT_MIME;\n // PDFs are the one input whose extraction concatenates pages and therefore\n // repeats headers/footers verbatim, so they get the boilerplate-line dedup\n // — but NOT when the extraction produced Markdown (vision / pdf-inspector):\n // the same heuristic strips table header/separator rows after the first\n // page and flattens fenced code and nested lists.\n if (effectiveMime === \"application/pdf\" && body && !extracted.isMarkdown) body = dedupLines(body);\n\n return {\n text: body,\n mime: effectiveMime,\n pages: extracted.pages ?? null,\n slides: extracted.slides ?? null,\n sizeBytes: content.length,\n llmPictures: extracted.llmPictures ?? 0,\n mediaOnly: Boolean(extracted.mediaOnly),\n providerTokens: { ...(extracted.providerTokens ?? {}) },\n contentHash: calculateContentHash({ docBytes: content }),\n isMarkdown: Boolean(extracted.isMarkdown),\n unreadableReason: extracted.unreadableReason ?? null,\n unreadablePages: extracted.unreadablePages ?? 0,\n };\n}\n\nexport function computeUnits(prepared: Prepared): number {\n let units = unitsForFile(prepared.mime, {\n pages: prepared.pages,\n slides: prepared.slides,\n sizeBytes: prepared.sizeBytes,\n });\n if (DOCX_MIMES.has(prepared.mime) && !prepared.pages && prepared.text) {\n units = Math.max(1, Math.ceil(prepared.text.length / DOCX_CHARS_PER_PAGE));\n }\n return units + Math.max(0, prepared.llmPictures);\n}\n\nfunction mergeFailed(failed: string[] | null | undefined, note: { rules_failed?: string[] }): void {\n if (!failed) return;\n for (const name of note.rules_failed ?? []) {\n if (!failed.includes(name)) failed.push(name);\n }\n}\n\n/**\n * Redact `text` for the DOCUMENT plane using `phase=\"ingest\"` rules.\n *\n * This is the document-level counterpart to the per-chunk redaction inside\n * `buildChunks`: it is what `runIngest` calls to compute `documentText` —\n * the value used for `Document.text`, `Document.lang`, and the input to\n * structured extraction. Without this, those document-plane values kept\n * deriving from the raw `prepared.text`, so an `apply_at=\"ingest\"` policy's\n * \"never-store\" promise held for chunks but not for the document row\n * `getDocument()` / `getDocumentText()` serve whole.\n *\n * `null`/empty `policy` is an exact no-op.\n */\nexport function redactDocumentText(\n text: string,\n policy: RedactionPolicy | null | undefined,\n secretKey: string | Buffer | null,\n hooks?: Hooks | null,\n failed?: string[] | null,\n): string {\n if (policy == null || policy.isEmpty()) return text;\n const [redacted, note] = applyRedaction(text, policy, {\n phase: \"ingest\",\n secretKey,\n hooks,\n });\n mergeFailed(failed, note);\n return redacted;\n}\n\n/**\n * Redact the STRING VALUES of one chunk's structural metadata.\n *\n * KEYS are never touched — they're field names (`section_title`,\n * `sheet_title`, ...), and redacting one would corrupt the result schema,\n * not its content. FLAT walk, deliberately — same trade-off as\n * `search.redactHits`' meta loop.\n */\nexport function redactChunkMeta(\n meta: Record<string, unknown>,\n policy: RedactionPolicy | null | undefined,\n secretKey: string | Buffer | null,\n hooks?: Hooks | null,\n failed?: string[] | null,\n): Record<string, unknown> {\n if (policy == null || policy.isEmpty()) return meta;\n const redacted: Record<string, unknown> = {};\n for (const [key, value] of Object.entries(meta)) {\n if (typeof value === \"string\") {\n const [nv, note] = applyRedaction(value, policy, { phase: \"ingest\", secretKey, hooks });\n mergeFailed(failed, note);\n redacted[key] = nv;\n } else {\n redacted[key] = value;\n }\n }\n return redacted;\n}\n\n/**\n * Route text through the structure-aware chunkers → `ChunkRow`s.\n *\n * When `policy` carries ingest-phase rules, each chunk is redacted BEFORE\n * the `ChunkRow` is built — so the stored text, the detected language, and\n * (later) the embedding all derive from the masked string. An embedding of\n * unmasked text would itself be a partial leak.\n */\nexport function buildChunks(\n prepared: Prepared,\n name: string | null | undefined,\n opts: {\n policy?: RedactionPolicy | null;\n secretKey?: string | Buffer | null;\n hooks?: Hooks | null;\n failed?: string[] | null;\n } = {},\n): [ChunkRow[], string] {\n const [chunks, effectiveMime, chunkMetas] = routeToChunker(prepared.text, prepared.mime, name, {\n isMarkdown: prepared.isMarkdown,\n });\n let kind = chunkerName(effectiveMime);\n if (prepared.isMarkdown && kind === \"generic\") kind = \"markdown\";\n\n const rows: ChunkRow[] = [];\n for (let idx = 0; idx < chunks.length; idx++) {\n let clean = sanitizeText(chunks[idx]!);\n if (opts.policy != null && !opts.policy.isEmpty()) {\n const [redacted, note] = applyRedaction(clean, opts.policy, {\n phase: \"ingest\",\n secretKey: opts.secretKey,\n hooks: opts.hooks,\n });\n mergeFailed(opts.failed, note);\n clean = redacted;\n }\n if (!clean?.trim()) continue;\n const meta: Record<string, unknown> = { chunker: kind };\n if (idx < chunkMetas.length && chunkMetas[idx] && Object.keys(chunkMetas[idx]!).length) {\n Object.assign(\n meta,\n redactChunkMeta(chunkMetas[idx]!, opts.policy, opts.secretKey ?? null, opts.hooks, opts.failed),\n );\n }\n rows.push({ idx: rows.length, text: clean, lang: detectLanguage(clean), meta });\n }\n return [rows, effectiveMime || prepared.mime || DEFAULT_MIME];\n}\n\n/** Embed `rows` in place (batched). Returns the provider token count. */\nexport async function embedChunks(embedder: Embedder, rows: ChunkRow[]): Promise<number> {\n if (!rows.length) return 0;\n let totalTokens = 0;\n for (let start = 0; start < rows.length; start += EMBED_BATCH) {\n const batch = rows.slice(start, start + EMBED_BATCH);\n const [vectors, tokens] = await embedder.embed(\n batch.map((r) => r.text),\n { kind: \"document\" },\n );\n if (vectors.length !== batch.length) {\n throw new Error(`embedding provider returned ${vectors.length} vectors for ${batch.length} inputs`);\n }\n for (let i = 0; i < batch.length; i++) batch[i]!.embedding = [...vectors[i]!];\n totalTokens += tokens || 0;\n }\n return totalTokens;\n}\n\n/** Bind this database to one embedding provider/model, or raise. */\nexport async function checkEmbeddingLock(pool: Pool, config: ContextEngineConfig): Promise<void> {\n const provider = config.embedding.provider;\n const model = config.embedding.model;\n const { rows } = await pool.query(\n `SELECT embedding_provider, embedding_model, embedding_dim FROM context_engine_meta WHERE id = 1`,\n );\n let row = rows[0];\n if (!row) {\n await pool.query(`INSERT INTO context_engine_meta (id) VALUES (1) ON CONFLICT (id) DO NOTHING`);\n row = { embedding_provider: null, embedding_model: null, embedding_dim: null };\n }\n if (row.embedding_model == null && row.embedding_provider == null) {\n await pool.query(\n `UPDATE context_engine_meta\n SET embedding_provider = $1, embedding_model = $2,\n embedding_dim = COALESCE(embedding_dim, $3)\n WHERE id = 1`,\n [provider, model, config.embedding.dim ?? null],\n );\n return;\n }\n if (row.embedding_model !== model || row.embedding_provider !== provider) {\n throw new Error(\n \"embedding model mismatch: this database was ingested with \" +\n `provider=${JSON.stringify(row.embedding_provider)} model=${JSON.stringify(row.embedding_model)}, but the engine ` +\n `is configured with provider=${JSON.stringify(provider)} model=${JSON.stringify(model)}. Re-embedding an existing ` +\n \"corpus is not automatic — use a separate database for a different embedding model.\",\n );\n }\n}\n\n/** Record (or validate) the observed vector width after the first embed. */\nexport async function recordEmbeddingDim(pool: Pool, dim: number | null | undefined): Promise<void> {\n if (dim == null) return;\n const { rows } = await pool.query(`SELECT embedding_dim FROM context_engine_meta WHERE id = 1`);\n const row = rows[0];\n if (!row) return;\n if (row.embedding_dim == null) {\n await pool.query(`UPDATE context_engine_meta SET embedding_dim = $1 WHERE id = 1`, [dim]);\n } else if (Number(row.embedding_dim) !== dim) {\n throw new Error(\n `embedding dimension mismatch: this database's vector columns are ${row.embedding_dim}-wide ` +\n `(set by \\`context-engine migrate --dim\\`), but the configured embedding model returned ` +\n `${dim}-dimensional vectors.`,\n );\n }\n}\n\nasync function loadExisting(\n pool: Pool,\n opts: {\n sourceId: string | null;\n externalId: string | null;\n name: string | null;\n ingestHash: Buffer;\n },\n): Promise<Record<string, unknown> | null> {\n const scope = opts.sourceId == null ? `source_id IS NULL` : `source_id = $1`;\n const params: unknown[] = opts.sourceId == null ? [] : [opts.sourceId];\n let extra: string;\n if (opts.externalId != null) {\n params.push(opts.externalId);\n extra = `external_id = $${params.length}`;\n } else if (opts.name != null) {\n params.push(opts.name);\n extra = `name = $${params.length}`;\n } else {\n params.push(opts.ingestHash);\n extra = `ingest_hash = $${params.length}`;\n }\n const { rows } = await pool.query(\n `SELECT * FROM context_engine_documents WHERE ${scope} AND ${extra} ORDER BY created_at ASC LIMIT 1`,\n params,\n );\n return rows[0] ?? null;\n}\n\n/**\n * Compare an incoming hash against the stored document's.\n *\n * Called TWICE per ingest, with different hashes, on purpose:\n *\n * - `hashKind=\"content\"` runs BEFORE extraction, on the MD5 of the raw\n * bytes. An unchanged file must not re-pay for vision/OCR extraction.\n * - `hashKind=\"ingest\"` runs after extraction, on the SHA-256 of the\n * normalized text, and is the authoritative check.\n *\n * `skipped` rows (media-only / no extractable text) are terminal too,\n * EXCEPT one whose `meta_data.unreadable_reason` is set (a scan that needs\n * vision, or whose vision extraction failed): re-ingesting the same bytes\n * must re-extract, since the documented remedy — configure a vision model\n * and re-ingest — has no other way to take effect.\n */\nexport async function decide(\n pool: Pool,\n opts: {\n request: IngestRequest;\n hashKind: \"content\" | \"ingest\";\n hashValue: Buffer | string | null;\n },\n): Promise<Decision> {\n if (opts.hashValue == null) return { action: \"process\", documentId: null };\n if (opts.hashKind === \"content\" && opts.request.externalId == null && opts.request.name == null) {\n return { action: \"process\", documentId: null };\n }\n\n const existing = await loadExisting(pool, {\n sourceId: opts.request.sourceId ?? null,\n externalId: opts.request.externalId ?? null,\n name: opts.request.name ?? null,\n ingestHash:\n opts.hashKind === \"ingest\" && Buffer.isBuffer(opts.hashValue) ? opts.hashValue : Buffer.alloc(0),\n });\n if (!existing) return { action: \"process\", documentId: null };\n\n const existingId = String(existing.id);\n if (existing.status !== \"completed\" && existing.status !== \"skipped\") {\n return { action: \"process\", documentId: existingId };\n }\n\n let stored: unknown;\n let incoming: unknown;\n if (opts.hashKind === \"ingest\") {\n stored = existing.ingest_hash != null ? Buffer.from(existing.ingest_hash as Buffer) : null;\n incoming = Buffer.isBuffer(opts.hashValue) ? opts.hashValue : Buffer.from(opts.hashValue as string);\n } else {\n stored = (existing.meta_data as Record<string, unknown> | null)?.content_hash;\n incoming = opts.hashValue;\n }\n\n const same =\n stored != null &&\n incoming != null &&\n (Buffer.isBuffer(stored) && Buffer.isBuffer(incoming)\n ? Buffer.compare(stored, incoming) === 0\n : stored === incoming);\n\n if (!same) return { action: \"process\", documentId: existingId };\n if (existing.status === \"skipped\") {\n const metaData = existing.meta_data as Record<string, unknown> | null;\n if (metaData?.unreadable_reason) return { action: \"process\", documentId: existingId };\n return { action: \"skip\", documentId: existingId };\n }\n const reqMode = opts.request.mode ?? \"hybrid\";\n if ((MODE_RANK[reqMode] ?? 0) <= (MODE_RANK[(existing.mode as string) || \"hybrid\"] ?? 0)) {\n return { action: \"skip\", documentId: existingId };\n }\n return { action: \"upgrade\", documentId: existingId };\n}\n\n/**\n * Insert-or-reuse the row and move it `queued` → `processing`.\n *\n * The two states are committed separately and in order, so `queued` is\n * genuinely observable by a concurrent reader rather than being an\n * in-memory value that only ever hits the database as `processing`.\n *\n * Only caller-declared attributes are written here. Everything derived\n * from the content lands at completion.\n */\nexport async function claimDocument(\n pool: Pool,\n opts: { request: IngestRequest; documentId: string | null },\n): Promise<string> {\n const client = await pool.connect();\n try {\n let id = opts.documentId;\n const req = opts.request;\n if (id) {\n const { rows } = await client.query(`SELECT id FROM context_engine_documents WHERE id = $1`, [id]);\n if (!rows[0]) id = null;\n }\n if (!id) {\n id = randomUUID();\n await client.query(\n `INSERT INTO context_engine_documents\n (id, source_id, external_id, name, description, acl, status, error, started_at, completed_at)\n VALUES ($1,$2,$3,$4,$5,$6,'queued',NULL,NULL,NULL)`,\n [\n id,\n req.sourceId ?? null,\n req.externalId ?? null,\n req.name ?? null,\n req.description ?? null,\n req.acl ?? null,\n ],\n );\n } else {\n await client.query(\n `UPDATE context_engine_documents\n SET source_id=$1, external_id=$2, name=$3, description=$4, acl=$5,\n error=NULL, status='queued', started_at=NULL, completed_at=NULL, updated_at=now()\n WHERE id=$6`,\n [\n req.sourceId ?? null,\n req.externalId ?? null,\n req.name ?? null,\n req.description ?? null,\n req.acl ?? null,\n id,\n ],\n );\n }\n await client.query(\n `UPDATE context_engine_documents SET status='processing', started_at=now(), updated_at=now() WHERE id=$1`,\n [id],\n );\n return id;\n } finally {\n client.release();\n }\n}\n\nexport async function finalizeDocument(\n pool: Pool,\n documentId: string,\n opts: { status: string; error?: string | null; content?: ContentPayload | null },\n): Promise<void> {\n if (opts.content) {\n const c = opts.content;\n const { rows } = await pool.query(`SELECT meta_data FROM context_engine_documents WHERE id = $1`, [\n documentId,\n ]);\n if (!rows[0]) return;\n const meta = { ...((rows[0].meta_data as Record<string, unknown>) ?? {}), ...c.metaUpdates };\n await pool.query(\n `UPDATE context_engine_documents SET\n text=$1, mime_type=$2, lang=$3, ingest_hash=$4, mode=$5, meta_data=$6::jsonb,\n document_type=COALESCE($7, document_type),\n structured_data=COALESCE($8::jsonb, structured_data),\n structured_keys=COALESCE($9::text[], structured_keys),\n structured_quality=COALESCE($10::jsonb, structured_quality),\n extraction_version=COALESCE($11, extraction_version),\n status=$12, error=$13, completed_at=now(), updated_at=now()\n WHERE id=$14`,\n [\n c.text,\n c.mime,\n c.lang,\n c.ingestHash,\n c.mode,\n JSON.stringify(meta),\n c.documentType ?? null,\n c.structuredData != null ? JSON.stringify(c.structuredData) : null,\n c.structuredKeys ?? null,\n c.structuredQuality != null ? JSON.stringify(c.structuredQuality) : null,\n c.extractionVersion ?? null,\n opts.status,\n opts.error ?? null,\n documentId,\n ],\n );\n return;\n }\n await pool.query(\n `UPDATE context_engine_documents SET status=$1, error=$2, completed_at=now(), updated_at=now() WHERE id=$3`,\n [opts.status, opts.error ?? null, documentId],\n );\n}\n\nexport async function refreshContentHash(\n pool: Pool,\n documentId: string | null,\n contentHash: string | null,\n): Promise<void> {\n if (!contentHash || !documentId) return;\n const { rows } = await pool.query(`SELECT meta_data FROM context_engine_documents WHERE id = $1`, [\n documentId,\n ]);\n if (!rows[0]) return;\n const meta = { ...((rows[0].meta_data as Record<string, unknown>) ?? {}) };\n if (meta.content_hash === contentHash) return;\n meta.content_hash = contentHash;\n await pool.query(`UPDATE context_engine_documents SET meta_data=$1::jsonb, updated_at=now() WHERE id=$2`, [\n JSON.stringify(meta),\n documentId,\n ]);\n}\n\n/**\n * Apply caller-declared attribute changes to a document row.\n *\n * `updates` contains ONLY the fields the caller provided: `acl`/`name`/\n * `description` are REPLACED with the given value (including `null`);\n * `meta_data`/`metaData` is MERGED. Shared by the skip-path refresh and\n * `ContextEngine.updateDocument` (partial PATCH).\n *\n * Throws if the document is absent. Returns the names of the fields that\n * actually changed. `\"acl\"` in that list is the caller's signal to push\n * the new value onto the chunk plane.\n */\nexport async function applyAttributeUpdates(\n pool: Pool,\n documentId: string,\n updates: Record<string, unknown>,\n): Promise<string[]> {\n const { rows } = await pool.query(`SELECT * FROM context_engine_documents WHERE id = $1`, [documentId]);\n const doc = rows[0];\n if (!doc) {\n const err = new Error(`document not found: ${JSON.stringify(documentId)}`);\n err.name = \"KeyError\";\n throw err;\n }\n\n const changed: string[] = [];\n const sets: string[] = [];\n const vals: unknown[] = [];\n let i = 1;\n\n if (\"acl\" in updates) {\n const storedAcl = doc.acl != null ? [...(doc.acl as string[])] : null;\n const next = updates.acl as string[] | null;\n const same =\n storedAcl === next ||\n (Array.isArray(storedAcl) &&\n Array.isArray(next) &&\n storedAcl.length === next.length &&\n storedAcl.every((v, idx) => v === next[idx]));\n if (!same) {\n sets.push(`acl = $${i++}`);\n vals.push(next);\n changed.push(\"acl\");\n }\n }\n for (const field of [\"name\", \"description\"] as const) {\n if (field in updates && doc[field] !== updates[field]) {\n sets.push(`${field} = $${i++}`);\n vals.push(updates[field]);\n changed.push(field);\n }\n }\n const metaUpdate = (updates.meta_data ?? updates.metaData) as Record<string, unknown> | null | undefined;\n if (metaUpdate && typeof metaUpdate === \"object\") {\n const meta = { ...((doc.meta_data as Record<string, unknown>) ?? {}) };\n const merged = { ...meta, ...metaUpdate };\n if (JSON.stringify(merged) !== JSON.stringify(meta)) {\n sets.push(`meta_data = $${i++}::jsonb`);\n vals.push(JSON.stringify(merged));\n changed.push(\"meta_data\");\n }\n }\n\n if (!changed.length) return [];\n sets.push(\"updated_at = now()\");\n vals.push(documentId);\n await pool.query(`UPDATE context_engine_documents SET ${sets.join(\", \")} WHERE id = $${i}`, vals);\n return changed;\n}\n\n/**\n * Apply the caller's declared attributes to a document being SKIPPED.\n *\n * **This is a security fix, not a nicety.** Dedup decides from content, but\n * `acl` / `name` / `description` / `metaData` are declared by the CALLER\n * and are not part of the hash. Without this, re-ingesting identical text\n * with a NEW `acl` returned `skipped` before anything was written, the\n * document (and its denormalized chunk copies) kept the OLD ACL, and a\n * tightening the caller believed had applied silently did nothing.\n *\n * - `name` / `description` / `acl` are REPLACED, including with `null`.\n * `acl=null` means \"unrestricted\", not \"leave whatever was there\".\n * - `metaData` is MERGED.\n */\nexport async function refreshDeclaredAttributes(\n pool: Pool,\n documentId: string,\n request: IngestRequest,\n): Promise<string[]> {\n try {\n return await applyAttributeUpdates(pool, documentId, {\n acl: request.acl ?? null,\n name: request.name ?? null,\n description: request.description ?? null,\n meta_data: request.metaData ?? null,\n });\n } catch (exc) {\n if (exc instanceof Error && exc.name === \"KeyError\") return [];\n throw exc;\n }\n}\n\nasync function backfillStructuredExtraction(\n request: IngestRequest,\n opts: {\n pool: Pool;\n embedder: Embedder;\n hooks: Hooks;\n llmCfg: NonNullable<ContextEngineConfig[\"llm\"]>;\n documentId: string;\n },\n): Promise<void> {\n const { rows } = await opts.pool.query(\n `SELECT text, structured_data, source_id FROM context_engine_documents WHERE id = $1`,\n [opts.documentId],\n );\n const doc = rows[0];\n if (!doc) return;\n if (doc.structured_data != null || !doc.text || !String(doc.text).trim()) return;\n\n const extraction = await extractStructuredData(String(doc.text), {\n llmCfg: opts.llmCfg,\n fieldHints: request.fieldHints,\n });\n\n const providerTokens: Record<string, number> = {};\n if (extraction.providerTokens.llm_input)\n providerTokens.structured_llm_input = extraction.providerTokens.llm_input;\n if (extraction.providerTokens.llm_output)\n providerTokens.structured_llm_output = extraction.providerTokens.llm_output;\n\n if (extraction.documentType != null) {\n let embedTokens = 0;\n if (Object.keys(extraction.structuredData).length) {\n try {\n embedTokens = await upsertRegistry({\n pool: opts.pool,\n embedder: opts.embedder,\n sourceId: doc.source_id ?? null,\n docType: extraction.documentType,\n structuredData: extraction.structuredData,\n keysRaw: extraction.keysRaw,\n keysNormalized: extraction.keysNormalized,\n });\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"structured_registry_backfill\", document_id: opts.documentId });\n embedTokens = 0;\n }\n }\n if (embedTokens) providerTokens.structured_embedding_tokens = embedTokens;\n await opts.pool.query(\n `UPDATE context_engine_documents SET\n document_type=$1, structured_data=$2::jsonb, structured_keys=$3,\n structured_quality=$4::jsonb, extraction_version=$5, updated_at=now()\n WHERE id=$6`,\n [\n extraction.documentType,\n JSON.stringify(extraction.structuredData),\n extraction.keysNormalized,\n JSON.stringify({ quality: extraction.quality }),\n STRUCTURED_EXTRACTION_VERSION,\n opts.documentId,\n ],\n );\n } else if (extraction.error) {\n emitError(opts.hooks, new Error(extraction.error), {\n stage: \"structured_extraction_backfill\",\n document_id: opts.documentId,\n });\n }\n\n if (Object.keys(providerTokens).length) {\n emitUsage(opts.hooks, {\n kind: \"ingest\",\n units: 0,\n detail: { document_id: opts.documentId, backfill: \"structured_extraction\" },\n providerTokens,\n } satisfies UsageEvent);\n }\n}\n\n/**\n * Refresh declared attributes across BOTH planes for a skipped document.\n *\n * Called from every skip path — the cheap pre-extraction one, the\n * post-extraction one, and the lost-race one — because all three return\n * without writing anything, and all three are reachable with a changed ACL.\n *\n * Also the choke point for structured-extraction backfill on a skip:\n * `extractStructured=true` against unchanged content used to be a silent\n * no-op because extraction only ran on the `process` branch.\n */\nexport async function refreshOnSkip(\n request: IngestRequest,\n opts: {\n pool: Pool;\n backend: StorageBackend;\n documentId: string | null;\n config: ContextEngineConfig;\n embedder: Embedder;\n hooks: Hooks;\n },\n): Promise<string[]> {\n if (!opts.documentId) return [];\n const changed = await refreshDeclaredAttributes(opts.pool, opts.documentId, request);\n if (changed.includes(\"acl\")) {\n await opts.backend.updateChunkAcl(opts.documentId, request.acl ?? null);\n }\n if (request.extractStructured && opts.config.llm != null) {\n await backfillStructuredExtraction(request, {\n pool: opts.pool,\n embedder: opts.embedder,\n hooks: opts.hooks,\n llmCfg: opts.config.llm,\n documentId: opts.documentId,\n });\n }\n return changed;\n}\n\nasync function upgradeMode(pool: Pool, documentId: string, mode: string): Promise<number> {\n await pool.query(`UPDATE context_engine_documents SET mode=$1, updated_at=now() WHERE id=$2`, [\n mode,\n documentId,\n ]);\n const { rows } = await pool.query(\n `SELECT count(*)::int AS n FROM context_engine_chunks WHERE document_id=$1`,\n [documentId],\n );\n return rows[0]?.n ?? 0;\n}\n\nfunction skippedReport(documentId: string | null, name: string, pages: number | null = null): DocumentReport {\n return {\n documentId: String(documentId ?? \"\"),\n name,\n status: \"skipped\",\n pages,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens: {},\n error: null,\n redactionFailed: [],\n };\n}\n\nfunction singleReport(doc: DocumentReport): IngestReport {\n return {\n documents: [doc],\n totals: {\n files: 1,\n failed: doc.status === \"failed\" ? 1 : 0,\n units: doc.units + doc.graphUnits,\n graphUnits: doc.graphUnits,\n },\n };\n}\n\nasync function maybeAwait(value: Promise<number> | number): Promise<number> {\n return value;\n}\n\n/**\n * Fire one `onProgress` boundary event (see `hooks.emitProgress`). Every\n * stage fires this twice — `started` then `done` — so a live UI can show\n * what is happening as it happens. `documentId` is null on the extract\n * events: the row is only claimed AFTER extraction, because dedup decides\n * from the extracted text. A stage that throws fires no `done`; the failure\n * reaches the caller through the report/exception.\n */\nfunction progress(\n hooks: Hooks,\n request: IngestRequest,\n displayName: string,\n documentId: string | null,\n stage: ProgressEvent[\"stage\"],\n state: ProgressEvent[\"state\"],\n detail: Record<string, unknown> = {},\n): void {\n emitProgress(hooks, {\n sourceId: request.sourceId ?? null,\n externalId: request.externalId ?? null,\n documentId: documentId == null ? null : String(documentId),\n name: displayName,\n stage,\n state,\n detail: { ...detail },\n });\n}\n\n/** How the text was obtained: caller-supplied, a vision model, or a parser. */\nfunction extractMethod(prepared: Prepared, text: string | null): string {\n if (text != null) return \"text\";\n if (Object.keys(prepared.providerTokens).length || prepared.isMarkdown) return \"vision\";\n return \"parser\";\n}\n\nasync function runModeUpgrade(\n request: IngestRequest,\n opts: {\n config: ContextEngineConfig;\n hooks: Hooks;\n pool: Pool;\n backend: StorageBackend;\n embedder: Embedder;\n graphStage: GraphStage | null | undefined;\n documentId: string;\n displayName: string;\n },\n): Promise<IngestReport> {\n await refreshOnSkip(request, {\n pool: opts.pool,\n backend: opts.backend,\n documentId: opts.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n });\n\n let graphUnits = 0;\n let chunkCount = 0;\n try {\n if (request.mode === \"graph\" && opts.graphStage) {\n progress(opts.hooks, request, opts.displayName, opts.documentId, \"graph\", \"started\");\n graphUnits = Number(\n (await maybeAwait(\n opts.graphStage({\n documentId: opts.documentId,\n chunks: null,\n config: opts.config,\n hooks: opts.hooks,\n }),\n )) || 0,\n );\n progress(opts.hooks, request, opts.displayName, opts.documentId, \"graph\", \"done\", {\n graph_units: graphUnits,\n });\n }\n chunkCount = await upgradeMode(opts.pool, opts.documentId, request.mode ?? \"hybrid\");\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"mode_upgrade\", document: opts.displayName });\n return singleReport({\n documentId: String(opts.documentId),\n name: opts.displayName,\n status: \"failed\",\n pages: null,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens: {},\n error: String(exc),\n redactionFailed: [],\n });\n }\n\n if (graphUnits) {\n emitUsage(opts.hooks, {\n kind: \"ingest\",\n units: graphUnits,\n detail: {\n document_id: String(opts.documentId),\n name: opts.displayName,\n mode: request.mode,\n upgrade: true,\n chunks: chunkCount,\n graph_units: graphUnits,\n },\n } satisfies UsageEvent);\n }\n\n return singleReport({\n documentId: String(opts.documentId),\n name: opts.displayName,\n status: \"upgraded\",\n pages: null,\n chunks: chunkCount,\n units: 0,\n graphUnits,\n providerTokens: {},\n error: null,\n redactionFailed: [],\n });\n}\n\n/**\n * Ingest one document end to end and return its report.\n *\n * `contentSource` is the already-resolved `[content, filename, text]`\n * triple from `resolveSource`; the engine resolves it first so a bad\n * argument combination raises before any DB or network work.\n */\nexport async function runIngest(\n request: IngestRequest,\n opts: {\n config: ContextEngineConfig;\n hooks: Hooks;\n backend: StorageBackend;\n embedder: Embedder;\n pool: Pool;\n graphStage?: GraphStage | null;\n contentSource?: [Buffer | null, string | null, string | null] | null;\n },\n): Promise<IngestReport> {\n const [content, filename, text] = opts.contentSource ?? [\n request.content ?? null,\n request.filename ?? null,\n request.text ?? null,\n ];\n const displayName = request.name || filename || request.externalId || \"(unnamed)\";\n const mode: Mode = request.mode ?? opts.config.defaultMode;\n\n await checkEmbeddingLock(opts.pool, opts.config);\n\n if (content != null) {\n const pre = await decide(opts.pool, {\n request: { ...request, mode },\n hashKind: \"content\",\n hashValue: calculateContentHash({ docBytes: content }),\n });\n if (pre.action === \"skip\") {\n await refreshOnSkip(\n { ...request, mode },\n {\n pool: opts.pool,\n backend: opts.backend,\n documentId: pre.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n },\n );\n return singleReport(skippedReport(pre.documentId, displayName));\n }\n if (pre.action === \"upgrade\") {\n return runModeUpgrade(\n { ...request, mode },\n {\n config: opts.config,\n hooks: opts.hooks,\n pool: opts.pool,\n backend: opts.backend,\n embedder: opts.embedder,\n graphStage: opts.graphStage,\n documentId: pre.documentId!,\n displayName,\n },\n );\n }\n }\n\n progress(opts.hooks, request, displayName, null, \"extract\", \"started\");\n const prepared = await prepare({\n content,\n filename,\n text,\n name: request.name ?? null,\n config: opts.config,\n hooks: opts.hooks,\n });\n progress(opts.hooks, request, displayName, null, \"extract\", \"done\", {\n pages: prepared.pages,\n method: extractMethod(prepared, text),\n mime: prepared.mime,\n });\n\n const ingestHash = sha256Bytes(prepared.text || \"\");\n const decision = await decide(opts.pool, {\n request: { ...request, mode },\n hashKind: \"ingest\",\n hashValue: ingestHash,\n });\n\n if (decision.action === \"skip\") {\n await refreshContentHash(opts.pool, decision.documentId, prepared.contentHash);\n await refreshOnSkip(\n { ...request, mode },\n {\n pool: opts.pool,\n backend: opts.backend,\n documentId: decision.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n },\n );\n return singleReport(skippedReport(decision.documentId, displayName, prepared.pages));\n }\n\n if (decision.action === \"upgrade\") {\n return runModeUpgrade(\n { ...request, mode },\n {\n config: opts.config,\n hooks: opts.hooks,\n pool: opts.pool,\n backend: opts.backend,\n embedder: opts.embedder,\n graphStage: opts.graphStage,\n documentId: decision.documentId!,\n displayName,\n },\n );\n }\n\n let documentId: string;\n try {\n documentId = await claimDocument(opts.pool, {\n request: { ...request, mode },\n documentId: decision.documentId,\n });\n } catch (exc) {\n if (!isUniqueViolation(exc)) throw exc;\n const retry = await decide(opts.pool, {\n request: { ...request, mode },\n hashKind: \"ingest\",\n hashValue: ingestHash,\n });\n if (retry.action === \"skip\") {\n await refreshOnSkip(\n { ...request, mode },\n {\n pool: opts.pool,\n backend: opts.backend,\n documentId: retry.documentId,\n config: opts.config,\n embedder: opts.embedder,\n hooks: opts.hooks,\n },\n );\n return singleReport(skippedReport(retry.documentId, displayName, prepared.pages));\n }\n if (retry.action === \"upgrade\") {\n return runModeUpgrade(\n { ...request, mode },\n {\n config: opts.config,\n hooks: opts.hooks,\n pool: opts.pool,\n backend: opts.backend,\n embedder: opts.embedder,\n graphStage: opts.graphStage,\n documentId: retry.documentId!,\n displayName,\n },\n );\n }\n documentId = await claimDocument(opts.pool, {\n request: { ...request, mode },\n documentId: retry.documentId,\n });\n }\n\n const metaUpdates: Record<string, unknown> = { ...(request.metaData ?? {}) };\n metaUpdates.mime_type = prepared.mime;\n if (prepared.contentHash) metaUpdates.content_hash = prepared.contentHash;\n // UNCONDITIONAL, null included: metaUpdates MERGES into the stored\n // meta_data, so a conditional write would leave a stale reason on a\n // document that has since become readable — e.g. re-ingested successfully\n // after a vision model was configured.\n metaUpdates.unreadable_reason = prepared.unreadableReason;\n metaUpdates.unreadable_pages = prepared.unreadablePages;\n\n const redactionFailed: string[] = [];\n let documentText: string;\n progress(opts.hooks, request, displayName, documentId, \"redact\", \"started\");\n try {\n documentText = redactDocumentText(\n prepared.text,\n opts.config.redaction,\n opts.config.secretKey,\n opts.hooks,\n redactionFailed,\n );\n } catch (exc) {\n // A misconfigured policy (a `hash` rule with no `secretKey`) must still\n // raise loudly: swallowing it would silently store unmasked text.\n // `claimDocument` already committed this row to `status=\"processing\"`.\n // Finalize to `failed` first, THEN rethrow — propagation AND terminal\n // state, not one or the other.\n await finalizeDocument(opts.pool, documentId, { status: \"failed\", error: String(exc) });\n throw exc;\n }\n progress(opts.hooks, request, displayName, documentId, \"redact\", \"done\", {\n rules_failed: redactionFailed.length,\n });\n\n if (!prepared.text.trim() || prepared.mediaOnly || looksMostlyBoilerplate(prepared.text)) {\n const reason = prepared.mediaOnly ? \"media-only (no extractable text)\" : \"no extractable text\";\n await opts.backend.upsertChunks(documentId, request.sourceId ?? null, request.acl ?? null, []);\n await finalizeDocument(opts.pool, documentId, {\n status: \"skipped\",\n error: reason,\n content: {\n text: documentText,\n mime: prepared.mime,\n lang: null,\n ingestHash,\n mode,\n metaUpdates,\n },\n });\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"skipped\",\n pages: prepared.pages,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens: {},\n error: reason,\n redactionFailed,\n // Extraction RAN here — an unreadable scan is exactly what lands in\n // this branch, so this is the report that has to say why. The\n // dedup-skip/upgrade sites stay at the defaults: they never extracted\n // and would be claiming blind.\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n }\n\n const units = computeUnits(prepared);\n const providerTokens: Record<string, number> = {};\n if (prepared.providerTokens.input) providerTokens.llm_input = prepared.providerTokens.input;\n if (prepared.providerTokens.output) providerTokens.llm_output = prepared.providerTokens.output;\n\n let rows: ChunkRow[] = [];\n let effectiveMime = prepared.mime;\n try {\n progress(opts.hooks, request, displayName, documentId, \"chunk\", \"started\");\n [rows, effectiveMime] = buildChunks(prepared, request.name, {\n policy: opts.config.redaction,\n secretKey: opts.config.secretKey,\n hooks: opts.hooks,\n failed: redactionFailed,\n });\n progress(opts.hooks, request, displayName, documentId, \"chunk\", \"done\", { chunks: rows.length });\n\n progress(opts.hooks, request, displayName, documentId, \"embed\", \"started\");\n if (request.batch) {\n await opts.backend.upsertChunks(documentId, request.sourceId ?? null, request.acl ?? null, rows);\n const batchId = await submitEmbeddingBatch(\n opts.config.embedding,\n rows.map((r) => r.text),\n );\n progress(opts.hooks, request, displayName, documentId, \"embed\", \"done\", {\n batch: true,\n batch_id: batchId,\n chunks: rows.length,\n });\n metaUpdates.mime_type = effectiveMime;\n metaUpdates.batch = {\n id: batchId,\n units,\n chunks: rows.length,\n mode,\n provider_tokens: providerTokens,\n };\n await finalizeDocument(opts.pool, documentId, {\n status: \"batch_pending\",\n content: {\n text: documentText,\n mime: effectiveMime,\n lang: detectLanguage(documentText),\n ingestHash,\n mode,\n metaUpdates,\n },\n });\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"batch_pending\",\n pages: prepared.pages,\n chunks: rows.length,\n units: 0,\n graphUnits: 0,\n providerTokens,\n error: null,\n redactionFailed,\n // Extraction RAN here just like the skipped/failed/completed sites —\n // this report has to say why pages were unreadable too, not leave\n // the caller to re-derive it later from listDocuments.\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n }\n\n const embeddingTokens = await embedChunks(opts.embedder, rows);\n if (embeddingTokens) providerTokens.embedding_tokens = embeddingTokens;\n await recordEmbeddingDim(opts.pool, opts.embedder.dim);\n\n await opts.backend.upsertChunks(documentId, request.sourceId ?? null, request.acl ?? null, rows);\n progress(opts.hooks, request, displayName, documentId, \"embed\", \"done\", {\n batch: false,\n chunks: rows.length,\n tokens: embeddingTokens,\n });\n\n const structuredKw: Partial<ContentPayload> = {};\n if (request.extractStructured && opts.config.llm) {\n progress(opts.hooks, request, displayName, documentId, \"structured\", \"started\");\n const extraction = await extractStructuredData(documentText, {\n llmCfg: opts.config.llm,\n fieldHints: request.fieldHints,\n });\n if (extraction.providerTokens.llm_input) {\n providerTokens.structured_llm_input = extraction.providerTokens.llm_input;\n }\n if (extraction.providerTokens.llm_output) {\n providerTokens.structured_llm_output = extraction.providerTokens.llm_output;\n }\n if (extraction.documentType != null) {\n let embedTokens = 0;\n if (Object.keys(extraction.structuredData).length) {\n try {\n embedTokens = await upsertRegistry({\n pool: opts.pool,\n embedder: opts.embedder,\n sourceId: request.sourceId ?? null,\n docType: extraction.documentType,\n structuredData: extraction.structuredData,\n keysRaw: extraction.keysRaw,\n keysNormalized: extraction.keysNormalized,\n });\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"structured_registry\", document: displayName });\n embedTokens = 0;\n }\n }\n structuredKw.documentType = extraction.documentType;\n structuredKw.structuredData = extraction.structuredData;\n structuredKw.structuredKeys = extraction.keysNormalized;\n structuredKw.structuredQuality = { quality: extraction.quality };\n structuredKw.extractionVersion = STRUCTURED_EXTRACTION_VERSION;\n if (embedTokens) providerTokens.structured_embedding_tokens = embedTokens;\n } else if (extraction.error) {\n emitError(opts.hooks, new Error(extraction.error), {\n stage: \"structured_extraction\",\n document: displayName,\n });\n }\n progress(opts.hooks, request, displayName, documentId, \"structured\", \"done\", {\n document_type: extraction.documentType,\n keys: extraction.keysNormalized.length,\n quality: extraction.quality,\n });\n }\n\n let graphUnits = 0;\n if (mode === \"graph\" && opts.graphStage) {\n progress(opts.hooks, request, displayName, documentId, \"graph\", \"started\");\n graphUnits = Number(\n (await maybeAwait(\n opts.graphStage({\n documentId,\n chunks: rows,\n config: opts.config,\n hooks: opts.hooks,\n }),\n )) || 0,\n );\n progress(opts.hooks, request, displayName, documentId, \"graph\", \"done\", { graph_units: graphUnits });\n }\n\n metaUpdates.mime_type = effectiveMime;\n await finalizeDocument(opts.pool, documentId, {\n status: \"completed\",\n content: {\n text: documentText,\n mime: effectiveMime,\n lang: detectLanguage(documentText),\n ingestHash,\n mode,\n metaUpdates,\n ...structuredKw,\n },\n });\n\n emitUsage(opts.hooks, {\n kind: \"ingest\",\n units: units + graphUnits,\n detail: {\n document_id: String(documentId),\n name: displayName,\n mime: effectiveMime,\n mode,\n pages: prepared.pages,\n slides: prepared.slides,\n chunks: rows.length,\n graph_units: graphUnits,\n },\n providerTokens,\n } satisfies UsageEvent);\n\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"completed\",\n pages: prepared.pages,\n chunks: rows.length,\n units,\n graphUnits,\n providerTokens,\n error: null,\n redactionFailed,\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n } catch (exc) {\n emitError(opts.hooks, exc, { stage: \"ingest\", document: displayName });\n await finalizeDocument(opts.pool, documentId, { status: \"failed\", error: String(exc) });\n return singleReport({\n documentId: String(documentId),\n name: displayName,\n status: \"failed\",\n pages: prepared.pages,\n chunks: 0,\n units: 0,\n graphUnits: 0,\n providerTokens,\n error: String(exc),\n redactionFailed,\n unreadableReason: prepared.unreadableReason,\n unreadablePages: prepared.unreadablePages,\n });\n }\n}\n","/**\n * `search_knowledge_base` — ONE tool with actions, as a plain library call.\n *\n * The four knowledge tools became sub-actions of one because that is better\n * for the model calling them: one description carries the decision rules once,\n * `discover` says what exists before it guesses, and there is one name to\n * route through. Nothing here knows what MCP is.\n *\n * **Why it lives here and not in `mcp.ts`.** MCP is one way to reach the tool,\n * not the only one. `engine.searchKnowledgeBase(...)` calls this directly;\n * `mcp.ts` registers a tool that resolves the caller's identity and scope and\n * then calls exactly the same function. Two doors, one implementation, so they\n * cannot drift apart.\n *\n * **Identity and scope are arguments, never defaults.** This module invents\n * neither a caller nor a ceiling. Both come from the program that mounted the\n * tool — a model may ask for any source id it likes, and a tool that believed\n * it would let one customer read another's files.\n */\n\nimport { MAX_LIST_LIMIT } from \"./actions.js\";\nimport { isTabularMime } from \"./chunkers.js\";\nimport { DocumentNotFoundError, EngineActionError } from \"./errors.js\";\nimport { ENTITY_TYPES, RELATIONSHIP_CATEGORY_VALUES } from \"./graph/entities.js\";\nimport { UNSCOPED } from \"./sentinels.js\";\n\nexport const KNOWLEDGE_ACTIONS = [\n \"discover\",\n \"search\",\n \"get_doc\",\n \"list\",\n \"query_meta\",\n \"compute\",\n \"get_chunks\",\n \"get_docs\",\n \"map_reduce\",\n \"traverse\",\n \"find_related\",\n \"get_neighbors\",\n \"community_summary\",\n] as const;\nexport type KnowledgeAction = (typeof KNOWLEDGE_ACTIONS)[number];\n\nexport const KNOWLEDGE_TOOL_DESCRIPTION =\n \"The knowledge base — the ingested documents and data files (PDF, Word, \" +\n \"Excel, CSV and the rest) — as ONE tool with actions. Nothing is searched \" +\n \"for you: use it before answering anything that should come from those \" +\n \"documents, and do not use it for general knowledge. \" +\n \"Call action 'discover' FIRST when you do not already know what is there: \" +\n \"it lists the documents and says what is INSIDE each one — a spreadsheet's \" +\n \"sheet names, column headers and row counts; a document's section titles \" +\n \"and last page; a JSON file's top-level keys — plus the extracted field \" +\n \"names grouped by document type with each field's data type and how many \" +\n \"documents carry it, a census of the whole corpus, and which actions this \" +\n \"deployment can run. Use those names verbatim: they are what compute, \" +\n \"query_meta and get_chunks match on, so one discover is enough and you \" +\n \"never have to go looking for a column, a section or a field name. A long \" +\n \"list is shortened there, with the rest reported as a count and, when it \" +\n \"is worth the trip, the exact call that returns it in full — discover \" +\n \"again with document_ids set to that one document, which answers whole. \" +\n \"When it is not worth the trip the payload says so and says what to do \" +\n \"instead: a sheet with hundreds of columns is one to compute over, never \" +\n \"one to read back. Then match the action to the task. \" +\n \"'search' finds passages by meaning or keywords — for questions answered \" +\n \"by reading text. Looking up an identifier (an ID, code, SKU or invoice \" +\n \"number) is the exception: search the BARE identifier alone, e.g. '2525', \" +\n \"never the whole question. Those are CONTENT identifiers — written inside a \" +\n \"document — and they belong in a query; a document_id is a system id (a \" +\n \"uuid) that only discover, list or a search hit can give you, so never \" +\n \"search a uuid as text and never hand an invoice number to get_doc. \" +\n \"'get_doc' reads one whole document by id, 'get_docs' reads several at \" +\n \"once, and 'get_chunks' walks one long document in order a piece at a \" +\n \"time when you need more of it than an excerpt; 'list' browses the \" +\n \"documents without searching. 'map_reduce' asks the SAME question of \" +\n \"every document in scope and answers once per document — for 'which \" +\n \"contracts mention X', where search would return a handful of passages \" +\n \"and miss the rest. \" +\n \"'query_meta' filters and aggregates documents by their structured fields \" +\n \"(dates, amounts, categories) — usually the right action for a question \" +\n \"about spreadsheet data. \" +\n \"'compute' runs code over the spreadsheets for any figure DERIVED from \" +\n \"them — a total, average, count, ranking, margin or comparison across \" +\n \"rows — AND for finding the exact row matching one id or value. In a large \" +\n \"table, search cannot reliably locate an individual row; compute can. \" +\n \"When the question is about how things are CONNECTED rather than what a \" +\n \"document says — who works with whom, what belongs to what, what a change \" +\n \"touches — use the graph actions: 'get_neighbors' for what is one step \" +\n \"from one thing, 'traverse' for everything within a few steps of it, \" +\n \"'find_related' for connections of a kind across the corpus, and \" +\n \"'community_summary' for the themes the corpus groups into. They are \" +\n \"available only where a graph was built; discover says so. Searching with \" +\n \"mode='graph' ranks passages by those same connections instead of by \" +\n \"wording alone, which finds a passage that never repeats your words. \" +\n \"Rules that decide answers: search returns EXCERPTS, and rows of a \" +\n \"spreadsheet are not arithmetic — never add up, average or rank rows \" +\n \"yourself from what search returned, and never say a figure is not \" +\n \"available before computing over the sheet that holds it. A listing that \" +\n \"says has_more has MORE: send its next_cursor back as 'cursor' for the \" +\n \"next page, and never conclude a document is absent from a first page that \" +\n \"was truncated. Search again with different words before saying a document \" +\n \"is missing. A result may carry a next_action (or next_page): it is the \" +\n \"call to make next, already filled in — follow it rather than guessing \" +\n \"the next step. Cite document names.\";\n\n/** The one-line routing rule, carried on the `action` parameter itself. */\nexport const ACTION_PARAM_DESCRIPTION =\n \"Match it to the task: reading questions -> search (a bare identifier for \" +\n \"an ID or code); spreadsheet analysis, or the exact row for one id or \" +\n \"value -> query_meta or compute; how things connect -> get_neighbors, \" +\n \"traverse, find_related or community_summary; browse everything -> list; \" +\n \"unsure what exists -> discover first.\";\n\n/** What `discover` says each action is for. */\nexport const ACTION_HELP: Record<Exclude<KnowledgeAction, \"discover\">, string> = {\n search: \"passages by meaning or keywords; an ID, code or number as the BARE identifier\",\n get_doc: \"one whole document by id — a clause, a policy, the full text\",\n list: \"browse the documents without searching\",\n query_meta: \"filter documents by structured fields (dates, amounts, categories)\",\n compute:\n \"a figure DERIVED from the spreadsheets — total, average, count, ranking, \" +\n \"margin, comparison, or the one row matching a value. Say what you want in \" +\n \"plain words, not code: it runs over EVERY ROW of the sheets in scope, not \" +\n \"a sample and not the excerpts search returned, so name the columns the way \" +\n \"the sheet spells them and say which sheet if more than one could match. \" +\n \"Never retype a figure from memory or from an excerpt — ask for it here\",\n get_chunks: \"read one document in order, a range at a time — the long ones\",\n get_docs: \"several whole documents at once, by id\",\n map_reduce:\n \"ask the SAME question of every document in scope, one answer per document \" +\n \"— for 'which contracts mention X', not for a figure (that is compute)\",\n get_neighbors: \"what is ONE step from a named thing, and which way each link points\",\n traverse: \"everything within a few steps of a named thing\",\n find_related: \"connections of a given kind across the corpus, with no starting thing\",\n community_summary: \"the themes the corpus groups into, ranked against a question\",\n};\n\nconst OFF = {\n query_meta: \"query_meta is off: this deployment has no LLM configured.\",\n compute:\n \"compute is off: code execution is not enabled on this deployment. Quote the rows you found instead.\",\n map_reduce:\n \"map_reduce is off: this deployment has no LLM configured. Use search, or \" +\n \"read the documents with get_docs.\",\n graph:\n \"the graph actions are off: this deployment has no graph configured, so \" +\n \"nothing has been linked up. Use search, get_doc or list instead.\",\n};\n\nconst GRAPH_ACTIONS = [\"traverse\", \"find_related\", \"get_neighbors\", \"community_summary\"];\n\n/**\n * The actions that can narrow by document id. Every other one refuses it\n * rather than answering over the whole corpus as if it had been honoured.\n */\n// `discover`/`list` are here because they share ONE query with one filter:\n// honouring the argument on one of them and refusing it on the other is a\n// difference a caller would have to memorise for no reason. On `discover` it\n// does double duty — naming the documents is also how a caller asks for the\n// structure lists in full, past the cap the whole-page call applies.\nconst DOCUMENT_ID_ACTIONS = [\"search\", \"compute\", \"get_docs\", \"map_reduce\", \"discover\", \"list\"];\n\n/** Whole documents are what blows a context window; the tool owns that budget. */\nconst DEFAULT_GET_DOCS_CHARS = 200_000;\n\n/**\n * Every input the MODEL may set — the ONE property table. Both the exported\n * JSON Schema and the protocol server's own Zod shape are built from it.\n * `principals`, `scope` and `redaction` are\n * deliberately absent: they are host-supplied on every surface, and a model\n * that could set them could cross a tenant or turn masking off.\n */\nexport const INPUT_PROPERTIES: Record<string, Record<string, unknown>> = {\n action: { type: \"string\", enum: [...KNOWLEDGE_ACTIONS], description: ACTION_PARAM_DESCRIPTION },\n query: {\n type: \"string\",\n description:\n \"What you are looking for, in words — or a BARE content identifier \" +\n \"when you are looking one up. (for search, query_meta, compute, community_summary)\",\n },\n document_id: {\n type: \"string\",\n description:\n \"One document's id, as returned by discover, list, or a search hit. \" +\n \"A system id, never something written inside a document. (for get_doc)\",\n },\n source_ids: {\n type: \"array\",\n items: { type: \"string\" },\n description:\n \"Narrows to these sources, using ids from discover or a search hit. \" +\n \"Intersected with what this deployment allows: ids outside it are \" +\n \"ignored, not refused. (for discover, list, search, query_meta, compute, \" +\n \"and the graph actions)\",\n },\n document_ids: {\n type: \"array\",\n items: { type: \"string\" },\n description:\n \"Narrows to these documents, using ids from discover, list or a search \" +\n \"hit. It INTERSECTS with source_ids, so a document outside the sources \" +\n \"you named returns nothing. On discover it also asks for those \" +\n \"documents' structure IN FULL, past the shortening a whole-page \" +\n \"discover applies. (for search, compute, get_docs, map_reduce, \" +\n \"discover, list)\",\n },\n entity: {\n type: \"string\",\n description:\n \"The NAME of a thing to start from — a person, an organisation, a \" +\n \"product — spelled as it appears in the documents. A name, NOT AN ID: \" +\n \"take one from discover's graph section, from a search hit, or from an \" +\n \"earlier get_neighbors answer. (for traverse, get_neighbors)\",\n },\n depth: {\n type: \"integer\",\n description: \"How many hops out to walk, 1 to 5. Default 2. (for traverse)\",\n },\n category: {\n type: \"string\",\n enum: [...RELATIONSHIP_CATEGORY_VALUES],\n description:\n \"The kind of connection to follow or look for. These are the only \" +\n \"values the graph holds; see discover for which occur in this corpus. \" +\n \"(for traverse, find_related)\",\n },\n label: {\n type: \"string\",\n description:\n \"The exact relationship wording, as extracted — for example reports_to, \" +\n \"works_for, owns. Narrower than category; use it when you know the \" +\n \"phrasing, otherwise use category. (for find_related)\",\n },\n entity_type: {\n type: \"string\",\n enum: [...ENTITY_TYPES],\n description: \"Keep only relationships with an entity of this type on one end. (for find_related)\",\n },\n top_k: {\n type: \"integer\",\n description: \"How many passages to return. Default 10. (for search)\",\n },\n mode: {\n type: \"string\",\n enum: [\"hybrid\", \"graph\"],\n description:\n \"How to rank. OMIT IT and the right one is worked out from the \" +\n \"documents in scope: wording and meaning, plus how things are \" +\n \"connected whenever those documents were linked up — which finds a \" +\n \"passage that never repeats your words. Name 'hybrid' to force wording \" +\n \"only, or 'graph' to force the connection signal on. (for search)\",\n },\n limit: {\n type: \"integer\",\n description:\n \"How many rows to return: documents for discover and list (default 50), \" +\n \"candidates for query_meta (20), relationships or paths for the graph \" +\n \"actions (50), communities for community_summary (3). (for discover, \" +\n \"list, query_meta, and the graph actions)\",\n },\n cursor: {\n type: \"string\",\n description:\n \"The next_cursor a previous page returned, sent back verbatim to read \" +\n \"the rest. (for discover, list)\",\n },\n start: {\n type: \"integer\",\n description:\n \"First chunk position to read, 0-based. Use the next_start a previous \" +\n \"answer returned to continue. (for get_chunks)\",\n },\n end: {\n type: \"integer\",\n description:\n \"Last chunk position to read, inclusive. At most 25 chunks come back at \" +\n \"a time whatever you ask for. (for get_chunks)\",\n },\n max_chars: {\n type: \"integer\",\n description:\n \"How much document text to return in total before stopping and handing \" +\n \"back the ids it did not reach. (for get_docs)\",\n },\n};\n\n/**\n * The tool as data: name, description and input schema as JSON Schema.\n *\n * Exported so a host wiring this into its own agent loop, or into a protocol\n * this library does not speak, never hand-writes the schema — that would be a\n * third copy of the action list, and a third place for it to fall behind.\n *\n * `readOnlyHint` is false on purpose. Most actions only read, but `compute`\n * runs generated code, and a host deciding whether to auto-approve a call must\n * not read \"knowledge base\" and assume a reader.\n */\nexport function knowledgeToolDefinition(): {\n name: string;\n description: string;\n input_schema: Record<string, unknown>;\n annotations: Record<string, boolean>;\n} {\n return {\n name: \"search_knowledge_base\",\n description: KNOWLEDGE_TOOL_DESCRIPTION,\n input_schema: {\n type: \"object\",\n properties: Object.fromEntries(Object.entries(INPUT_PROPERTIES).map(([k, v]) => [k, { ...v }])),\n required: [\"action\"],\n },\n annotations: { readOnlyHint: false, openWorldHint: false },\n };\n}\n\n// ---------------------------------------------------------------------------\n// Scope — the ceiling the HOST sets, which the model can narrow but never widen\n// ---------------------------------------------------------------------------\n\n/**\n * The documents a mounted tool may ever reach.\n *\n * Deliberately only the nouns the engine already knows: source ids, and\n * optionally some document ids within them. Not project, workspace, tenant or\n * agent — those are a host's words, and baking one in would make every other\n * host translate its word into ours.\n */\nexport type Scope = { sourceIds: string[] | null; documentIds: string[] | null };\n\nexport type ScopeInput = typeof UNSCOPED | string[] | Partial<Scope> | null | undefined;\n\n/**\n * Normalise what a host supplies into a `Scope`. `UNSCOPED` means no ceiling.\n * `null`/omitted throws: it is what a host that forgot looks like, and\n * defaulting it to \"no ceiling\" is the corpus-wide search this exists to stop.\n */\nexport function resolveScope(value: ScopeInput): Scope {\n if (value === UNSCOPED) return { sourceIds: null, documentIds: null };\n if (value == null) {\n throw new Error(\n \"scope is required: pass the source ids this tool may reach, or UNSCOPED \" +\n \"(from @promptev/context-engine) to say the whole corpus on purpose.\",\n );\n }\n if (Array.isArray(value)) {\n return { sourceIds: [...new Set(value.map(String))], documentIds: null };\n }\n if (typeof value === \"object\") {\n return {\n sourceIds: value.sourceIds != null ? [...new Set(value.sourceIds.map(String))] : null,\n documentIds: value.documentIds != null ? [...new Set(value.documentIds.map(String))] : null,\n };\n }\n throw new Error(`scope must be UNSCOPED, an array of source ids, or a Scope — got ${typeof value}`);\n}\n\n/**\n * Intersect what the caller asked for with what the host allows.\n *\n * Narrow, never widen. `null` requested means the whole ceiling, never the\n * whole corpus. An id outside the ceiling is DROPPED rather than refused,\n * because an error would tell the caller which ids are real.\n *\n * The return value distinguishes two things that must never be confused: an\n * array (possibly EMPTY — the caller asked only for things it may not have, so\n * the answer is nothing) from `null` (no ceiling and no narrowing, no filter).\n */\nexport function narrowToCeiling(\n requested: string[] | null | undefined,\n ceiling: string[] | null,\n): string[] | null {\n if (ceiling == null) return requested != null ? [...new Set(requested)] : null;\n if (requested == null) return [...ceiling];\n const allowed = new Set(ceiling);\n return [...new Set(requested)].filter((r) => allowed.has(r));\n}\n\n/** A document-id ceiling is a whitelist: anything not named is outside. */\nfunction withinCeiling(documentId: string, ceiling: Scope): boolean {\n return ceiling.documentIds == null || ceiling.documentIds.includes(String(documentId));\n}\n\n/**\n * A host's own compute. Running generated code is where a host has its own\n * rules about permission, billing and approval, so it can replace ours rather\n * than intercept the action before the tool is reached.\n */\nexport type KnowledgeComputeFn = (\n instruction: string,\n opts: { sourceIds: string[] | null; documentIds: string[] | null; principals: unknown },\n) => Promise<Record<string, unknown>>;\n\n/**\n * Which actions this deployment can service. A host-supplied `compute` makes\n * that action available whatever `enableCodeExecution` says: the host is the\n * one running the code.\n */\nexport function knowledgeActionsAvailable(\n engine: {\n config: { llm?: unknown; enableCodeExecution?: boolean; graph?: { enabled?: boolean } };\n },\n opts: { compute?: KnowledgeComputeFn | null; mapReduce?: unknown } = {},\n): Record<Exclude<KnowledgeAction, \"discover\">, boolean> {\n const hasLlm = engine.config.llm != null;\n const graphOn = Boolean(engine.config.graph?.enabled);\n return {\n search: true,\n get_doc: true,\n list: true,\n query_meta: hasLlm,\n compute: opts.compute != null || (hasLlm && Boolean(engine.config.enableCodeExecution)),\n get_chunks: true,\n get_docs: true,\n map_reduce: opts.mapReduce != null || hasLlm,\n traverse: graphOn,\n find_related: graphOn,\n get_neighbors: graphOn,\n community_summary: graphOn,\n };\n}\n\n/**\n * Spreadsheet exactly when `compute` can read it — `compute` selects on MIME\n * alone and never looks at the name, so judging by filename here would make\n * `discover`'s promise a lie in both directions.\n */\nfunction documentKind(doc: Record<string, unknown>): \"spreadsheet\" | \"text\" {\n return isTabularMime(String(doc.mimeType ?? doc.mime_type ?? \"\")) ? \"spreadsheet\" : \"text\";\n}\n\ntype Listed = {\n id: string;\n name: unknown;\n source_id: unknown;\n kind: \"spreadsheet\" | \"text\";\n /**\n * What `discover`'s `fields_by_type` and `document_types` are keyed by.\n * Without it the grouping names types the caller cannot match to any\n * document, so it cannot narrow a follow-up call by one.\n */\n document_type: unknown;\n /**\n * Which documents the graph actions can actually reach. A graph action\n * over a corpus ingested `hybrid` returns nothing, and without this a\n * model cannot tell that from \"no connections\".\n */\n mode: unknown;\n /** Set by `discover` only — see the action body. `list` does not pay for it. */\n structure?: Record<string, unknown>;\n};\ntype ListedPage = {\n documents: Listed[];\n count: number;\n has_more: boolean;\n next_cursor?: string;\n next_page?: { action: string; cursor: string };\n};\n\n/**\n * A cursor is the `next_cursor` string a previous page returned. A malformed\n * one is refused, never silently treated as page one.\n */\nexport function parseCursor(cursor: unknown): Record<string, unknown> | null {\n if (cursor == null || cursor === \"\") return null;\n if (typeof cursor === \"object\" && !Array.isArray(cursor)) return cursor as Record<string, unknown>;\n let parsed: unknown = null;\n try {\n parsed = JSON.parse(String(cursor));\n } catch {\n parsed = null;\n }\n if (!parsed || typeof parsed !== \"object\" || Array.isArray(parsed)) {\n throw new Error(`invalid cursor: ${JSON.stringify(cursor)}`);\n }\n return parsed as Record<string, unknown>;\n}\n\ntype Engine = {\n config: { llm?: unknown; enableCodeExecution?: boolean; graph?: { enabled?: boolean } };\n search: (query: string, opts?: Record<string, unknown>) => Promise<{ hits: unknown[]; usage?: unknown }>;\n getDocument: (id: string, opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n listDocuments: (opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n queryStructured?: (question: string, opts?: Record<string, unknown>) => Promise<unknown>;\n compute?: (instruction: string, opts?: Record<string, unknown>) => Promise<unknown>;\n ensurePool?: () => Promise<unknown>;\n embedder?: unknown;\n getChunks?: (id: string, opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n getDocuments?: (ids: string[], opts?: Record<string, unknown>) => Promise<Array<Record<string, unknown>>>;\n mapReduce?: (instruction: string, opts?: Record<string, unknown>) => Promise<Record<string, unknown>>;\n spreadsheetSchema: (opts: Record<string, unknown>) => Promise<unknown[]>;\n documentStructure: (opts: Record<string, unknown>) => Promise<Record<string, Record<string, unknown>>>;\n documentTypes: (opts?: Record<string, unknown>) => Promise<unknown[]>;\n fieldSummary: (opts?: Record<string, unknown>) => Promise<unknown[]>;\n};\n\nasync function listPage(\n engine: Engine,\n sourceIds: string[] | null,\n principals: unknown,\n action: string,\n limit: number,\n cursor: Record<string, unknown> | null,\n documentIds: string[] | null,\n ceiling: Scope,\n redaction: unknown,\n): Promise<ListedPage> {\n const page = await engine.listDocuments({\n // `!= null`, NOT truthiness: an EMPTY array means \"nothing is in scope\"\n // and collapsing it to null would list the whole corpus.\n sourceIds: sourceIds != null ? [...new Set(sourceIds)] : null,\n // Already intersected with the host's ceiling by `narrowToCeiling`, and\n // filtered in SQL rather than after the page is built — three named\n // documents sitting on page four must come back as themselves, not as an\n // empty page.\n documentIds: documentIds != null ? [...new Set(documentIds)] : null,\n principals,\n cursor,\n limit: Math.max(1, Math.min(limit, MAX_LIST_LIMIT)),\n redaction,\n });\n let documents: Listed[] = ((page.documents as Record<string, unknown>[] | undefined) ?? []).map((raw) => ({\n id: String(raw.id),\n name: raw.name,\n source_id: raw.sourceId ?? raw.source_id,\n kind: documentKind(raw),\n document_type: raw.documentType ?? raw.document_type ?? null,\n mode: raw.mode ?? null,\n }));\n if (ceiling.documentIds != null) {\n const allowed = new Set(ceiling.documentIds);\n documents = documents.filter((d) => allowed.has(d.id));\n }\n const out: ListedPage = {\n documents,\n count: documents.length,\n has_more: Boolean(page.hasMore ?? page.has_more),\n };\n const next = page.nextCursor ?? page.next_cursor;\n if (next != null) {\n out.next_cursor = JSON.stringify(next);\n out.next_page = { action, cursor: out.next_cursor };\n }\n return out;\n}\n\nexport type KnowledgeToolArgs = {\n action: string;\n principals: unknown;\n scope: ScopeInput;\n query?: string | null;\n document_id?: string | null;\n source_ids?: string[] | null;\n document_ids?: string[] | null;\n entity?: string | null;\n depth?: number | null;\n category?: string | null;\n label?: string | null;\n entity_type?: string | null;\n top_k?: number | null;\n mode?: string | null;\n limit?: number | null;\n cursor?: unknown;\n /** Overrides the deployment's configured policy for this call only. */\n redaction?: unknown;\n /** Replaces the built-in compute action. */\n compute?: KnowledgeComputeFn | null;\n /** Replaces the built-in map_reduce action — the other expensive one. */\n map_reduce?:\n | ((instruction: string, opts: Record<string, unknown>) => Promise<Record<string, unknown>>)\n | null;\n start?: number | null;\n end?: number | null;\n max_chars?: number | null;\n};\n\n/** Run one action of the knowledge tool. */\nexport async function callKnowledgeTool(\n engine: Engine,\n args: KnowledgeToolArgs,\n): Promise<Record<string, unknown>> {\n const action = args.action;\n const principals = args.principals;\n const ceiling = resolveScope(args.scope);\n // What the MODEL asked for, before the host's ceiling is folded in — the\n // refusal below is about the caller's request, never about the ceiling.\n const requestedDocumentIds = args.document_ids;\n const sourceIds = narrowToCeiling(args.source_ids, ceiling.sourceIds);\n const documentIds = narrowToCeiling(args.document_ids, ceiling.documentIds);\n const text = String(args.query ?? \"\").trim();\n const available = knowledgeActionsAvailable(engine, {\n compute: args.compute,\n mapReduce: args.map_reduce,\n });\n\n if (!(KNOWLEDGE_ACTIONS as readonly string[]).includes(action)) {\n return {\n success: false,\n error: `unknown action '${action}' — use one of: ${KNOWLEDGE_ACTIONS.join(\", \")}`,\n };\n }\n if (requestedDocumentIds?.length && !DOCUMENT_ID_ACTIONS.includes(action)) {\n return {\n success: false,\n error:\n `document_ids is not honored by ${action} — only by ${DOCUMENT_ID_ACTIONS.join(\", \")}. ` +\n \"Narrow a listing with source_ids, or read one document with get_doc.\",\n };\n }\n if (GRAPH_ACTIONS.includes(action) && !available[action as Exclude<KnowledgeAction, \"discover\">]) {\n return { success: false, error: OFF.graph };\n }\n\n if (action === \"discover\" || action === \"list\") {\n const page = await listPage(\n engine,\n sourceIds,\n principals,\n action,\n args.limit ?? 50,\n parseCursor(args.cursor),\n documentIds,\n ceiling,\n args.redaction,\n );\n if (action === \"list\") return { success: true, ...page };\n // The schema, not just the inventory — and for EVERY document, not only\n // the spreadsheets. A model given a list of file names still has to spend\n // two turns finding the sheet names, the column headers, the section\n // titles and the field names before it can write one correct call. `list`\n // deliberately does NOT pay for this: browsing is not the step that needs\n // the shape of everything it lists.\n const structures = await engine.documentStructure({\n documentIds: page.documents.map((d) => d.id),\n sourceIds,\n principals,\n // The cap is for the call that did NOT name its documents. A caller\n // that asked about specific documents asked for all of them, and the\n // truncated payload tells it to make exactly this call — so answering\n // it truncated again would be a loop.\n bounded: !requestedDocumentIds?.length,\n redaction: args.redaction,\n });\n for (const doc of page.documents) doc.structure = structures[doc.id] ?? {};\n const hasSpreadsheet = page.documents.some((d) => d.kind === \"spreadsheet\");\n // Over the whole scope, not just this page: the field vocabulary is a\n // property of the corpus, and a `query_meta` call written from page one\n // must still be right on page four.\n const fieldsByType = await engine.fieldSummary({\n sourceIds,\n documentIds,\n principals,\n redaction: args.redaction,\n });\n const census = await engine.documentTypes({\n sourceIds,\n documentIds,\n principals,\n redaction: args.redaction,\n });\n const forAFact = { action: \"search\", query: \"<bare identifier or key words>\" };\n const nextAction: Record<string, unknown> = {};\n if (hasSpreadsheet && available.compute) {\n nextAction[\"for a figure from a spreadsheet\"] = {\n action: \"compute\",\n query: \"<what to compute, columns as named above>\",\n };\n }\n nextAction[\"for a clause or a fact\"] = forAFact;\n if (fieldsByType.length && available.query_meta) {\n nextAction[\"for documents by a field value\"] = {\n action: \"query_meta\",\n query: \"<a question naming a field from fields_by_type>\",\n };\n }\n if (available.get_neighbors) {\n nextAction[\"for how things connect\"] = {\n action: \"get_neighbors\",\n entity: \"<a name that appears in the documents>\",\n };\n }\n const discovered: Record<string, unknown> = {\n success: true,\n ...page,\n document_types: census,\n fields_by_type: fieldsByType,\n available_actions: Object.fromEntries(\n (Object.keys(ACTION_HELP) as Array<keyof typeof ACTION_HELP>).map((name) => [\n name,\n available[name] ? ACTION_HELP[name] : `${ACTION_HELP[name]} (OFF)`,\n ]),\n ),\n next_action: nextAction,\n };\n return discovered;\n }\n\n if (action === \"search\") {\n if (!text) return { success: false, error: \"search needs a query\" };\n const result = await engine.search(text, {\n sourceIds,\n documentIds,\n principals,\n topK: args.top_k ?? 10,\n // Omitted means \"work it out from the documents in scope\".\n mode: args.mode ?? null,\n redaction: args.redaction,\n });\n return {\n success: true,\n hits: (result.hits ?? []).map((hit) => {\n const h = hit as Record<string, unknown>;\n return {\n document_id: h.document_id ?? h.documentId,\n document_name: h.document_name ?? h.documentName,\n chunk_text: h.chunk_text ?? h.chunkText,\n score: h.score,\n source_id: h.source_id ?? h.sourceId,\n };\n }),\n usage: result.usage,\n };\n }\n\n if (action === \"get_doc\") {\n const documentId = args.document_id;\n if (!documentId) return { success: false, error: \"get_doc needs a document_id\" };\n // Reading one document by id is the obvious way around a source ceiling,\n // so the ceiling is checked here too — and a document outside it reads\n // exactly like one that does not exist.\n if (!withinCeiling(documentId, ceiling)) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n try {\n const document = await engine.getDocument(documentId, {\n principals,\n redaction: args.redaction,\n });\n const src = document?.sourceId ?? document?.source_id;\n if (ceiling.sourceIds != null && !ceiling.sourceIds.includes(String(src))) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n return { success: true, document };\n } catch (err) {\n // `getDocument` gives an absent and a forbidden document the SAME\n // message on purpose, so a caller can never tell them apart.\n if (err instanceof DocumentNotFoundError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n if (action === \"query_meta\") {\n if (!available.query_meta || !engine.queryStructured) return { success: false, error: OFF.query_meta };\n if (!text) return { success: false, error: \"query_meta needs a query\" };\n const out = await engine.queryStructured(text, {\n sourceIds,\n principals,\n limit: Math.max(1, Math.min(args.limit ?? 20, 100)),\n redaction: args.redaction,\n });\n return { success: true, ...(out as Record<string, unknown>) };\n }\n\n if (action === \"compute\") {\n if (!available.compute) return { success: false, error: OFF.compute };\n if (!text) return { success: false, error: \"compute needs a query: what to compute, in plain words\" };\n // A host-supplied `compute` REPLACES the built-in one. Running code is\n // where a host has its own rules about permission, billing and approval,\n // and it should not have to intercept the action before the tool is\n // reached to apply them.\n const runner = args.compute ?? engine.compute;\n if (!runner) return { success: false, error: OFF.compute };\n try {\n const out = await runner(text, { sourceIds, documentIds, principals });\n return { success: true, ...(out as Record<string, unknown>) };\n } catch (err) {\n // Nothing tabular in scope is the model's to recover from — it can quote\n // the rows it found instead — not a transport failure.\n if (err instanceof EngineActionError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n if (action === \"get_chunks\") {\n const documentId = args.document_id;\n if (!documentId) return { success: false, error: \"get_chunks needs a document_id\" };\n if (!withinCeiling(documentId, ceiling)) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n try {\n const page = await engine.getChunks!(documentId, {\n principals,\n start: args.start,\n end: args.end,\n redaction: args.redaction,\n });\n if (ceiling.sourceIds != null) {\n // A source ceiling has to hold here too, and the refusal reads like a\n // document that is not there.\n const doc = await engine.getDocument(documentId, { principals });\n const src = doc?.sourceId ?? doc?.source_id;\n if (!ceiling.sourceIds.includes(String(src))) {\n return { success: false, error: `document not found: ${documentId}` };\n }\n }\n const out: Record<string, unknown> = { success: true, ...page };\n if (page.has_more) {\n out.next_page = {\n action: \"get_chunks\",\n document_id: String(documentId),\n start: page.next_start,\n };\n }\n return out;\n } catch (err) {\n if (err instanceof DocumentNotFoundError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n if (action === \"get_docs\") {\n if (!documentIds?.length) return { success: false, error: \"get_docs needs document_ids\" };\n const docs = await engine.getDocuments!(documentIds, { principals, redaction: args.redaction });\n // The LIBRARY refuses to truncate — that is policy for whatever calls it —\n // and this tool IS that caller, so the budget lives here. What it could not\n // fit comes back as the call to make next.\n const budget = Math.max(1000, Math.trunc(args.max_chars || DEFAULT_GET_DOCS_CHARS));\n const kept: Array<Record<string, unknown>> = [];\n let used = 0;\n for (const doc of docs) {\n const size = String(doc.text ?? \"\").length;\n if (kept.length && used + size > budget) break;\n kept.push(doc);\n used += size;\n }\n const returned = new Set(kept.map((d) => String(d.id)));\n const remaining = (args.document_ids ?? []).filter((d) => !returned.has(String(d)));\n const out: Record<string, unknown> = { success: true, documents: kept, count: kept.length };\n if (remaining.length) {\n out.remaining_document_ids = remaining;\n out.next_page = { action: \"get_docs\", document_ids: remaining };\n }\n return out;\n }\n\n if (action === \"map_reduce\") {\n if (!available.map_reduce) return { success: false, error: OFF.map_reduce };\n if (!text) {\n return {\n success: false,\n error: \"map_reduce needs a query: the question to ask of each document\",\n };\n }\n // A host-supplied `map_reduce` REPLACES the built-in one, for the same\n // reason `compute` has that seam.\n const runner = args.map_reduce ?? engine.mapReduce;\n if (!runner) return { success: false, error: OFF.map_reduce };\n try {\n const out = await runner(text, {\n sourceIds,\n documentIds,\n principals,\n limit: args.limit,\n });\n return { success: true, ...(out as Record<string, unknown>) };\n } catch (err) {\n if (err instanceof EngineActionError) return { success: false, error: err.message };\n throw err;\n }\n }\n\n return callGraphAction(engine, args, { sourceIds, principals, text });\n}\n\nasync function callGraphAction(\n engine: Engine,\n args: KnowledgeToolArgs,\n ctx: { sourceIds: string[] | null; principals: unknown; text: string },\n): Promise<Record<string, unknown>> {\n // Imported lazily so importing the package never pulls the graph module in\n // for a deployment that has no graph.\n const nav = await import(\"./graph/navigation.js\");\n const pool = (await engine.ensurePool?.()) as never;\n // These take the INTERNAL spelling (null = trusted).\n const principals = (ctx.principals ?? null) as string[] | null;\n const scoped = { sourceIds: ctx.sourceIds, principals };\n\n if (args.action === \"community_summary\") {\n if (!ctx.text) return { success: false, error: \"community_summary needs a query\" };\n const out = await nav.communitySummary(pool, {\n embedder: engine.embedder as never,\n query: ctx.text,\n limit: args.limit,\n sourceIds: ctx.sourceIds,\n principals,\n });\n if (!out.available) return { success: false, error: out.error as string };\n return { success: true, communities: out.communities, count: out.count };\n }\n\n if (args.action === \"find_related\") {\n if (!(args.category || args.label || args.entity_type)) {\n return {\n success: false,\n error: \"find_related needs a category, a label or an entity_type to look for\",\n };\n }\n const out = await nav.findRelated(pool, {\n ...scoped,\n category: args.category,\n label: args.label,\n entityType: args.entity_type,\n limit: args.limit,\n });\n return { success: true, relationships: out.relationships, count: out.count };\n }\n\n if (!args.entity) {\n return {\n success: false,\n error: `${args.action} needs an entity: the name of a thing to start from`,\n };\n }\n if (args.action === \"get_neighbors\") {\n const out = await nav.getNeighbors(pool, { ...scoped, entity: args.entity, limit: args.limit });\n if (!out.found) return { success: false, error: out.error as string };\n return { success: true, entity: out.entity, neighbors: out.neighbors, count: out.count };\n }\n const out = await nav.traverse(pool, {\n ...scoped,\n entity: args.entity,\n depth: args.depth,\n category: args.category,\n limit: args.limit,\n });\n if (!out.found) return { success: false, error: out.error as string };\n return { success: true, entity: out.entity, paths: out.paths, count: out.count };\n}\n","import { z } from \"zod\";\nimport { TRUSTED, type Trusted } from \"./sentinels.js\";\n\nexport class HandlerError extends Error {\n status: number;\n detail: unknown;\n constructor(status: number, detail: unknown) {\n super(typeof detail === \"string\" ? detail : JSON.stringify(detail));\n this.name = \"HandlerError\";\n this.status = status;\n this.detail = detail;\n }\n}\n\nexport const ingestJsonRequestSchema = z\n .object({\n text: z.string(),\n name: z.string(),\n source_id: z.string().nullable().optional(),\n external_id: z.string().nullable().optional(),\n description: z.string().nullable().optional(),\n meta_data: z.record(z.unknown()).nullable().optional(),\n acl: z.array(z.string()).nullable().optional(),\n mode: z.enum([\"hybrid\", \"graph\"]).nullable().optional(),\n extract_structured: z.boolean().optional().default(false),\n batch: z.boolean().optional().default(false),\n })\n .strip();\n\nexport const documentPatchSchema = z\n .object({\n acl: z.array(z.string()).nullable().optional(),\n name: z.string().nullable().optional(),\n description: z.string().nullable().optional(),\n meta_data: z.record(z.unknown()).nullable().optional(),\n })\n .strict();\n\nexport const searchRequestSchema = z\n .object({\n query: z.string(),\n source_ids: z.array(z.string()).nullable().optional(),\n // Narrows WITHIN a source and INTERSECTS with source_ids — it can only\n // shrink the result set (the ACL predicate still applies in the same SQL\n // conjunction), so exposing it needs no authorizeAcl-style grant check.\n document_ids: z.array(z.string()).nullable().optional(),\n top_k: z.number().int().optional().default(10),\n mode: z.enum([\"hybrid\", \"graph\"]).optional().default(\"hybrid\"),\n compress_to_tokens: z.number().int().nullable().optional(),\n })\n .strip();\n\nexport type IngestJSONRequest = z.infer<typeof ingestJsonRequestSchema>;\nexport type DocumentPatch = z.infer<typeof documentPatchSchema>;\nexport type SearchRequest = z.infer<typeof searchRequestSchema>;\n\nexport type ContextEngineLike = {\n ingest: (args: Record<string, unknown>) => Promise<unknown>;\n search: (query: string, opts?: Record<string, unknown>) => Promise<{ hits?: unknown[]; usage?: unknown }>;\n listDocuments: (opts: Record<string, unknown>) => Promise<unknown>;\n getDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n deleteDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n updateDocument: (id: string, opts?: Record<string, unknown>) => Promise<unknown>;\n stats: (sourceId?: string | null) => Promise<unknown>;\n};\n\nconst UUID_RE = /^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}$/i;\n\nexport function requireValidUuid(documentId: string): void {\n if (!UUID_RE.test(String(documentId))) {\n throw new HandlerError(400, `invalid document id: ${JSON.stringify(documentId)}`);\n }\n}\n\nexport function formBool(value: unknown): boolean {\n if (typeof value === \"boolean\") return value;\n if (value == null) return false;\n return [\"1\", \"true\", \"yes\", \"on\"].includes(String(value).trim().toLowerCase());\n}\n\nexport function formJson(value: unknown): unknown {\n if (value === undefined || value === null || value === \"\") return null;\n if (typeof value === \"object\") return value;\n try {\n return JSON.parse(String(value));\n } catch {\n throw new HandlerError(400, `invalid JSON in field: ${JSON.stringify(value)}`);\n }\n}\n\nexport function formStr(value: unknown): string | null {\n return value === undefined || value === null || value === \"\" ? null : String(value);\n}\n\nexport function resolveRequestPrincipals(value: unknown, opts: { surface: string }): string[] | Trusted {\n // Hand the SENTINEL through, not null: every engine method resolves TRUSTED\n // silently, while `principals=null` triggers the deprecation warning aimed\n // at legacy in-process callers — collapsing here made a correctly\n // configured trusted mount re-trigger that warning on every request.\n if (value === TRUSTED) return TRUSTED;\n if (value == null) {\n throw new HandlerError(\n 500,\n `${opts.surface}: the \\`principals\\` dependency returned null, which means ` +\n `TRUSTED CALLER — it disables ACL filtering and skips the check that ` +\n `stops a caller filing documents under groups it does not hold. This ` +\n `mount is misconfigured: return [] for an unauthenticated caller, or ` +\n `pass TRUSTED explicitly (from @promptev/context-engine) if the surface really ` +\n `is trusted.`,\n );\n }\n // A runtime check, not a cast: `as string[]` let a plain-JS dependency\n // return a bare string that flowed into resolvePrincipals and (before its\n // own guard) coerced to full-corpus TRUSTED.\n if (!Array.isArray(value) || value.some((p) => typeof p !== \"string\")) {\n throw new HandlerError(\n 500,\n `${opts.surface}: the \\`principals\\` dependency must return a list of ` +\n `principal strings, [] for an anonymous caller, or TRUSTED — got ${typeof value}.`,\n );\n }\n return value;\n}\n\nexport function authorizeAcl(requested: unknown, principals: string[] | null | Trusted): void {\n if (requested == null || principals == null || principals === TRUSTED) return;\n if (!Array.isArray(requested)) {\n throw new HandlerError(422, \"acl must be a list of principal strings\");\n }\n // TRUSTED was excluded above, so what remains is the plain list.\n const held = new Set(principals as string[]);\n const ungranted = [...new Set(requested.map(String))].filter((p) => !held.has(p)).sort();\n if (ungranted.length) {\n throw new HandlerError(403, `cannot grant access to principals you do not hold: ${ungranted}`);\n }\n}\n\nfunction asDict(obj: unknown): unknown {\n if (obj == null) return obj;\n if (typeof obj === \"object\" && typeof (obj as { toJSON?: () => unknown }).toJSON === \"function\") {\n return (obj as { toJSON: () => unknown }).toJSON();\n }\n return obj;\n}\n\nfunction zodError(err: z.ZodError): HandlerError {\n return new HandlerError(\n 422,\n err.issues.map((i) => ({ loc: i.path, msg: i.message, type: i.code })),\n );\n}\n\nasync function runIngest(engine: ContextEngineLike, kwargs: Record<string, unknown>): Promise<unknown> {\n try {\n return asDict(await engine.ingest(kwargs));\n } catch (exc) {\n if (exc instanceof TypeError || (exc instanceof Error && exc.name === \"TypeError\")) {\n throw new HandlerError(400, String(exc));\n }\n if (exc instanceof Error && exc.name === \"Error\" && /invalid|required/i.test(exc.message)) {\n throw new HandlerError(400, String(exc));\n }\n throw exc;\n }\n}\n\nexport async function handleIngestJson(\n engine: ContextEngineLike,\n payload: unknown,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_ingest_json\" });\n const parsed = ingestJsonRequestSchema.safeParse(payload);\n if (!parsed.success) throw zodError(parsed.error);\n const body = parsed.data;\n authorizeAcl(body.acl, resolved);\n return runIngest(engine, {\n text: body.text,\n name: body.name,\n description: body.description,\n sourceId: body.source_id,\n externalId: body.external_id,\n metaData: body.meta_data,\n acl: body.acl,\n mode: body.mode,\n extractStructured: body.extract_structured,\n batch: body.batch,\n });\n}\n\nexport async function handleIngestFile(\n engine: ContextEngineLike,\n opts: {\n content: Buffer | Uint8Array;\n filename: string | null | undefined;\n form: Record<string, unknown> | Map<string, unknown>;\n principals: unknown;\n },\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(opts.principals, { surface: \"handle_ingest_file\" });\n const form = opts.form instanceof Map ? Object.fromEntries(opts.form) : opts.form;\n const acl = formJson(form.acl);\n authorizeAcl(acl, resolved);\n return runIngest(engine, {\n content: opts.content,\n filename: opts.filename,\n name: formStr(form.name) || opts.filename,\n description: formStr(form.description),\n sourceId: formStr(form.source_id),\n externalId: formStr(form.external_id),\n metaData: formJson(form.meta_data),\n acl,\n mode: formStr(form.mode),\n extractStructured: formBool(form.extract_structured),\n batch: formBool(form.batch),\n });\n}\n\nexport async function handleListDocuments(\n engine: ContextEngineLike,\n opts: {\n sourceId: string | null | undefined;\n limit: number;\n cursor: string | null | undefined;\n principals: unknown;\n },\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(opts.principals, { surface: \"handle_list_documents\" });\n let parsedCursor: unknown = null;\n if (opts.cursor) {\n try {\n parsedCursor = JSON.parse(opts.cursor);\n } catch {\n throw new HandlerError(400, `invalid cursor: ${JSON.stringify(opts.cursor)}`);\n }\n }\n return engine.listDocuments({\n sourceId: opts.sourceId ?? null,\n principals: resolved,\n cursor: parsedCursor,\n limit: opts.limit,\n });\n}\n\nexport async function handleGetDocument(\n engine: ContextEngineLike,\n documentId: string,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_get_document\" });\n requireValidUuid(documentId);\n try {\n return await engine.getDocument(documentId, { principals: resolved });\n } catch (exc) {\n if (\n exc instanceof Error &&\n (exc.name === \"DocumentNotFoundError\" || exc instanceof TypeError === false)\n ) {\n const msg = String(exc);\n if (/not found/i.test(msg) || exc.name === \"DocumentNotFoundError\") {\n throw new HandlerError(404, msg.replace(/^Error:\\s*/, \"\"));\n }\n }\n throw exc;\n }\n}\n\nexport async function handleDeleteDocument(\n engine: ContextEngineLike,\n documentId: string,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_delete_document\" });\n requireValidUuid(documentId);\n try {\n await engine.deleteDocument(documentId, { principals: resolved });\n } catch (exc) {\n const msg = String(exc);\n if (/not found/i.test(msg) || (exc instanceof Error && exc.name === \"DocumentNotFoundError\")) {\n throw new HandlerError(404, msg.replace(/^Error:\\s*/, \"\"));\n }\n throw exc;\n }\n return { deleted: documentId };\n}\n\nexport async function handleUpdateDocument(\n engine: ContextEngineLike,\n documentId: string,\n payload: unknown,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_update_document\" });\n requireValidUuid(documentId);\n const parsed = documentPatchSchema.safeParse(payload);\n if (!parsed.success) throw zodError(parsed.error);\n const provided: Record<string, unknown> = {};\n const raw = payload as Record<string, unknown>;\n for (const key of [\"acl\", \"name\", \"description\", \"meta_data\"] as const) {\n if (Object.hasOwn(raw, key)) {\n provided[key === \"meta_data\" ? \"metaData\" : key] = raw[key];\n }\n }\n if (\"acl\" in provided) authorizeAcl(provided.acl, resolved);\n try {\n const changed = await engine.updateDocument(documentId, { principals: resolved, ...provided });\n return { changed };\n } catch (exc) {\n const msg = String(exc);\n if (/not found/i.test(msg) || (exc instanceof Error && exc.name === \"DocumentNotFoundError\")) {\n throw new HandlerError(404, msg.replace(/^Error:\\s*/, \"\"));\n }\n throw exc;\n }\n}\n\nexport async function handleSearch(\n engine: ContextEngineLike,\n payload: unknown,\n principals: unknown,\n): Promise<unknown> {\n const resolved = resolveRequestPrincipals(principals, { surface: \"handle_search\" });\n const parsed = searchRequestSchema.safeParse(payload);\n if (!parsed.success) throw zodError(parsed.error);\n const body = parsed.data;\n // Early, like every sibling document endpoint: a malformed id must be a\n // clean 400 here, not a Postgres 22P02 raised inside every search leg —\n // whose 400-vs-500 mapping would otherwise hang on the ENGLISH pg error\n // text, with the raw DatabaseError echoed to the caller.\n for (const did of body.document_ids ?? []) requireValidUuid(did);\n try {\n const result = await engine.search(body.query, {\n sourceIds: body.source_ids,\n documentIds: body.document_ids,\n principals: resolved,\n topK: body.top_k,\n mode: body.mode,\n compressToTokens: body.compress_to_tokens,\n });\n const hits = (result.hits ?? []).map((hit) => asDict(hit));\n return { hits, usage: result.usage };\n } catch (exc) {\n if (exc instanceof Error && /invalid|required|graph/i.test(exc.message)) {\n throw new HandlerError(400, String(exc));\n }\n throw exc;\n }\n}\n\nexport async function handleStats(\n engine: ContextEngineLike,\n sourceId: string | null | undefined,\n): Promise<unknown> {\n return engine.stats(sourceId ?? null);\n}\n","import { z } from \"zod\";\nimport { HandlerError, resolveRequestPrincipals } from \"../routing-core.js\";\nimport type { Trusted } from \"../sentinels.js\";\n\nexport type PrincipalsFn = () => unknown | Promise<unknown>;\n/** Zero-argument, sync or async, resolved fresh per execution — the same\n * contract as `PrincipalsFn`. Returns the opaque approval scope (a run id is\n * the expected shape) or `null` for the unscoped legacy path. */\nexport type ApprovalScopeFn = () => unknown | Promise<unknown>;\n\nexport const SCOPE_NOTE =\n \" Results are scoped to the caller's permissions (resolved by the server's \" +\n \"injected principals dependency — never a caller-supplied argument) and to \" +\n \"the given source_ids, if any.\";\n\nexport async function resolvePrincipalsFn(principals: PrincipalsFn): Promise<string[] | Trusted> {\n const result = await Promise.resolve(principals());\n // Validation DELEGATES to routing-core's resolveRequestPrincipals — the\n // single canonical remote-surface validator (TRUSTED passes through, null\n // and non-arrays are rejected) — rather than hand-copying its rules: a\n // rule added to the canonical validator must not silently miss the MCP\n // surface. Its HandlerError is an HTTP shape, so it is re-raised as a\n // plain Error with the same detail — the MCP host converts that to the\n // isError=true result this surface's contract promises.\n try {\n return resolveRequestPrincipals(result, { surface: \"the MCP `principals` dependency\" });\n } catch (exc) {\n if (exc instanceof HandlerError) throw new Error(exc.message, { cause: exc });\n throw exc;\n }\n}\n\nexport type McpToolHost = {\n tool(\n name: string,\n description: string,\n schema: Record<string, unknown>,\n handler: (...args: never[]) => Promise<unknown>,\n ): unknown;\n};\n\nexport function registerToolGateway(\n mcp: McpToolHost,\n engine: {\n searchTools: (query: string, opts?: Record<string, unknown>) => Promise<unknown[]>;\n executeTool: (\n name: string,\n args: Record<string, unknown> | null,\n opts?: Record<string, unknown>,\n ) => Promise<Record<string, unknown>>;\n },\n principals: PrincipalsFn,\n /** Optional server-side resolver for the approval claim scope (see\n * governance `executeTool`). Like `principals`, NEVER a tool argument: a\n * model that could name its own scope could claim another run's\n * approvals. Validation is the library's, so a wrong type surfaces as the\n * same isError=true result any other host-side wiring bug does. */\n approvalScope: ApprovalScopeFn | null = null,\n): void {\n mcp.tool(\n \"search_tools\",\n \"Keyword search over the tools this deployment has registered \" +\n \"(http/db/mcp/function) — the ACL-visible name, kind, description, \" +\n \"and JSON Schema params of each match, so a caller can discover what \" +\n \"it can then invoke with `execute_tool`.\" +\n SCOPE_NOTE,\n // Zod raw shape — see the note on the search tool in mcp.ts.\n { query: z.string(), limit: z.number().int().optional() },\n async (...raw: never[]) => {\n const args = (raw[0] ?? {}) as { query?: string; limit?: number };\n const callerPrincipals = await resolvePrincipalsFn(principals);\n const tools = await engine.searchTools(String(args.query ?? \"\"), {\n principals: callerPrincipals,\n limit: args.limit ?? 10,\n });\n return { tools };\n },\n );\n\n mcp.tool(\n \"execute_tool\",\n \"Governed execution of a tool previously discovered via \" +\n \"`search_tools` (ACL check, approval gate, audit). Returns \" +\n \"`{result, usage}` on success, or an `approval_required` payload \" +\n \"when the call needs a human approval that isn't already granted — \" +\n \"the caller must not retry until that approval resolves.\" +\n SCOPE_NOTE,\n { name: z.string(), args: z.record(z.unknown()).optional() },\n async (...raw: never[]) => {\n const body = (raw[0] ?? {}) as { name?: string; args?: Record<string, unknown> };\n const callerPrincipals = await resolvePrincipalsFn(principals);\n const scope = approvalScope ? await Promise.resolve(approvalScope()) : null;\n return engine.executeTool(String(body.name), body.args ?? null, {\n principals: callerPrincipals,\n source: \"mcp\",\n approvalScope: scope,\n });\n },\n );\n}\n","/** Bumped by CI on every main merge; 0.0.0 = pre-first-release. */\nexport const __version__ = \"0.0.0\";\n"]}
|