failproofai 1.0.4-beta.1 → 1.0.4-beta.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.next/standalone/.next/BUILD_ID +1 -1
- package/.next/standalone/.next/build-manifest.json +5 -5
- package/.next/standalone/.next/prerender-manifest.json +5 -5
- package/.next/standalone/.next/required-server-files.json +1 -1
- package/.next/standalone/.next/server/app/_global-error/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/_global-error/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/_global-error/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/_global-error/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/_global-error.html +1 -1
- package/.next/standalone/.next/server/app/_global-error.rsc +7 -7
- package/.next/standalone/.next/server/app/_global-error.segments/__PAGE__.segment.rsc +6 -6
- package/.next/standalone/.next/server/app/_global-error.segments/_full.segment.rsc +7 -7
- package/.next/standalone/.next/server/app/_global-error.segments/_tree.segment.rsc +1 -1
- package/.next/standalone/.next/server/app/_not-found/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/_not-found/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/_not-found/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/_not-found/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/_not-found.html +1 -1
- package/.next/standalone/.next/server/app/_not-found.rsc +15 -15
- package/.next/standalone/.next/server/app/_not-found.segments/_full.segment.rsc +15 -15
- package/.next/standalone/.next/server/app/_not-found.segments/_not-found/__PAGE__.segment.rsc +14 -14
- package/.next/standalone/.next/server/app/_not-found.segments/_tree.segment.rsc +2 -2
- package/.next/standalone/.next/server/app/api/audit/invite/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/audit/invite/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/api/audit/run/route.js +4 -4
- package/.next/standalone/.next/server/app/api/audit/run/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/audit/run/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/api/audit/status/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/audit/status/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/api/auth/login-request/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/login-request/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/api/auth/login-verify/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/login-verify/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/api/auth/logout/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/logout/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/api/auth/status/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/auth/status/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/api/download/[project]/[session]/route.js +1 -1
- package/.next/standalone/.next/server/app/api/download/[project]/[session]/route.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/api/download/[project]/[session]/route_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/audit/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/audit/page/server-reference-manifest.json +37 -4
- package/.next/standalone/.next/server/app/audit/page.js +2 -2
- package/.next/standalone/.next/server/app/audit/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/audit/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/index.html +1 -1
- package/.next/standalone/.next/server/app/index.rsc +15 -15
- package/.next/standalone/.next/server/app/index.segments/__PAGE__.segment.rsc +14 -14
- package/.next/standalone/.next/server/app/index.segments/_full.segment.rsc +15 -15
- package/.next/standalone/.next/server/app/index.segments/_tree.segment.rsc +2 -2
- package/.next/standalone/.next/server/app/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/policies/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/policies/page/server-reference-manifest.json +14 -14
- package/.next/standalone/.next/server/app/policies/page.js +1 -1
- package/.next/standalone/.next/server/app/policies/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/policies/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/project/[name]/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/project/[name]/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/project/[name]/page.js +3 -3
- package/.next/standalone/.next/server/app/project/[name]/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/project/[name]/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page/react-loadable-manifest.json +2 -2
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page/server-reference-manifest.json +2 -2
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page.js +2 -2
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/project/[name]/session/[sessionId]/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/projects/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/projects/page/server-reference-manifest.json +1 -1
- package/.next/standalone/.next/server/app/projects/page.js +1 -1
- package/.next/standalone/.next/server/app/projects/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/projects/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/app/settings/page/build-manifest.json +2 -2
- package/.next/standalone/.next/server/app/settings/page/server-reference-manifest.json +4 -4
- package/.next/standalone/.next/server/app/settings/page.js.nft.json +1 -1
- package/.next/standalone/.next/server/app/settings/page_client-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/chunks/{[externals]__1zndo1i._.js → [externals]__1lh7m5d._.js} +1 -1
- package/.next/standalone/.next/server/chunks/{[externals]__0_gzk8q._.js → [externals]__1rqkg_y._.js} +1 -1
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__08dchfu._.js +1 -1
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0cuho4x._.js +1 -1
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0o07qi9._.js +1 -1
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__1adacul._.js +1 -1
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__1le9jqc._.js +3 -0
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__1p-qi2t._.js +5 -0
- package/.next/standalone/.next/server/chunks/_08w6xzm._.js +3 -0
- package/.next/standalone/.next/server/chunks/_09dz7xv._.js +1 -1
- package/.next/standalone/.next/server/chunks/_0tovk6q._.js +1 -1
- package/.next/standalone/.next/server/chunks/_0trp3yc._.js +1 -1
- package/.next/standalone/.next/server/chunks/_1ek68ln._.js +14 -21
- package/.next/standalone/.next/server/chunks/lib_telemetry_ts_0dctyyw._.js +1 -1
- package/.next/standalone/.next/server/chunks/node_modules_posthog-node_dist_entrypoints_index_node_mjs_01r25oi._.js +2 -2
- package/.next/standalone/.next/server/chunks/node_modules_posthog-node_dist_entrypoints_index_node_mjs_09z9-p7._.js +2 -2
- package/.next/standalone/.next/server/chunks/package_json_[json]_cjs_1nxcc4v._.js +1 -1
- package/.next/standalone/.next/server/chunks/src_hooks_0iu54mz._.js +1 -1
- package/.next/standalone/.next/server/chunks/src_hooks_fp-home_ts_09kv0bn._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__00-s7h8._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__013jr2b._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__01wy8d-._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__02npjtd._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0da85px._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0ftmoxc._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/{[root-of-the-server]__0wvusu0._.js → [root-of-the-server]__0n_lxhg._.js} +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0oa1lav._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0p-5p8u._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0rgu2r3._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0s740oi._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__190won8._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1ctpynv._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1p2otjt._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1qapotl._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1sc84bb._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1zl_r0t._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/_0-oij9d._.js +23 -0
- package/.next/standalone/.next/server/chunks/ssr/_0-yi74u._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/_08x1r5t._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/_0l2hi_d._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/_0oq1dh6._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/_0wso5d3._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/_1es2j7i._.js +48 -168
- package/.next/standalone/.next/server/chunks/ssr/{_08f0jd2._.js → _1if9ixi._.js} +1 -1
- package/.next/standalone/.next/server/chunks/ssr/_1u8-lu2._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/_next-internal_server_app_policies_page_actions_1sp2-yo.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/app_audit__components_audit-dashboard_tsx_0p9ud47._.js +1 -69
- package/.next/standalone/.next/server/chunks/ssr/app_audit__components_rerun-button_tsx_0blihgw._.js +3 -0
- package/.next/standalone/.next/server/chunks/ssr/app_global-error_tsx_1kp6l3x._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/app_policies_hooks-client_tsx_19dqvpc._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/app_settings_settings-client_tsx_20lq-mq._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/node_modules_13n822a._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/node_modules_posthog-node_dist_entrypoints_index_node_mjs_11bnuzn._.js +2 -2
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_builtin-policies_ts_09j2ndl._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_fp-config_ts_04t589g._.js +1 -1
- package/.next/standalone/.next/server/chunks/ssr/src_hooks_fp-home_ts_0je3xkv._.js +1 -1
- package/.next/standalone/.next/server/middleware-build-manifest.js +5 -5
- package/.next/standalone/.next/server/pages/404.html +1 -1
- package/.next/standalone/.next/server/pages/500.html +1 -1
- package/.next/standalone/.next/server/server-reference-manifest.js +1 -1
- package/.next/standalone/.next/server/server-reference-manifest.json +54 -21
- package/.next/standalone/.next/static/chunks/{1_bxotmf9qngs.js → 010bv1w6j171t.js} +1 -1
- package/.next/standalone/.next/static/chunks/04r6ch8uf_n8m.js +1 -0
- package/.next/standalone/.next/static/chunks/{29-iqnp9v_axt.css → 0o-hh5_turzlz.css} +1 -1
- package/.next/standalone/.next/static/chunks/0qmsunv04x4na.css +2 -0
- package/.next/standalone/.next/static/chunks/{00gv0v91mycm-.js → 0wz8yftk18ts2.js} +1 -1
- package/.next/standalone/.next/static/chunks/0zebh1n9jkfbt.js +1 -0
- package/.next/standalone/.next/static/chunks/{0qn9khv0oo9ex.js → 1pb1oztsbwcss.js} +1 -1
- package/.next/standalone/.next/static/chunks/2aquitk72k2op.js +1 -0
- package/.next/standalone/.next/static/chunks/2bi_1y0a_smt7.js +6 -0
- package/.next/standalone/.next/static/chunks/{1po3lirgk2riw.js → 2ej3b8gk5ittu.js} +1 -1
- package/.next/standalone/.next/static/chunks/{01hvwwrhvbzyh.js → 2zafgs90r_leq.js} +1 -1
- package/.next/standalone/.next/static/chunks/{1_2jsatfuthrw.js → 32spub4wqjem-.js} +1 -1
- package/.next/standalone/.next/static/chunks/{3_--3ibia6ba1.js → 3l36d73kj18zw.js} +1 -1
- package/.next/standalone/.next/static/chunks/{2bvwsyu6x__ia.js → 3m4upvybtrexd.js} +1 -1
- package/.next/standalone/.next/static/chunks/{2k9f4tyv04809.css → 3uyhmm01g64k4.css} +1 -0
- package/.next/standalone/.next/static/chunks/{2bnf_4pgyk1rq.js → 40lpt2o2b283e.js} +1 -1
- package/.next/standalone/.next/static/chunks/{turbopack-18qf4jodtft6b.js → turbopack-3s3c-u_u_l0u9.js} +1 -1
- package/.next/standalone/.opencode/plugins/failproofai.mjs +51 -27
- package/.next/standalone/app/actions/get-leaks.ts +100 -0
- package/.next/standalone/app/audit/_components/audit-dashboard.tsx +112 -33
- package/.next/standalone/app/audit/_components/audit-poster.tsx +29 -10
- package/.next/standalone/app/audit/_components/come-back-better-section.tsx +6 -1
- package/.next/standalone/app/audit/_components/empty-state.tsx +4 -1
- package/.next/standalone/app/audit/_components/how-to-improve-section.tsx +19 -5
- package/.next/standalone/app/audit/_components/leak-section.tsx +179 -0
- package/.next/standalone/app/audit/_components/share-templates.ts +110 -44
- package/.next/standalone/app/audit/audit-styles.css +114 -1
- package/.next/standalone/app/project/[name]/page.tsx +4 -24
- package/.next/standalone/fp-cloud-cli/CHANGELOG.md +26 -3
- package/.next/standalone/fp-cloud-cli/fp_cli/_click_compat.py +31 -8
- package/.next/standalone/fp-cloud-cli/fp_cli/permissions.py +1 -0
- package/.next/standalone/fp-cloud-cli/pyproject.toml +9 -2
- package/.next/standalone/fp-cloud-cli/tests/test_click_compat.py +54 -0
- package/.next/standalone/fp-cloud-cli/uv.lock +9 -12
- package/.next/standalone/lib/auth/api-server-client.ts +28 -0
- package/.next/standalone/lib/claude-sessions.ts +101 -27
- package/.next/standalone/lib/cli-registry.ts +1 -33
- package/.next/standalone/lib/download-session.ts +0 -37
- package/.next/standalone/lib/projects.ts +3 -27
- package/.next/standalone/node_modules/@next/env/package.json +1 -1
- package/.next/standalone/node_modules/next/dist/build/swc/index.js +1 -1
- package/.next/standalone/node_modules/next/dist/compiled/next-server/pages-turbo.runtime.prod.js +1 -1
- package/.next/standalone/node_modules/next/dist/experimental/testmode/fetch.js +10 -2
- package/.next/standalone/node_modules/next/dist/lib/patch-incorrect-lockfile.js +3 -3
- package/.next/standalone/node_modules/next/dist/lib/typescript/runTypeScriptCli.js +2 -2
- package/.next/standalone/node_modules/next/dist/lib/verify-typescript-setup.js +3 -1
- package/.next/standalone/node_modules/next/dist/server/config.js +1 -1
- package/.next/standalone/node_modules/next/dist/server/dev/hot-reloader-turbopack.js +2 -2
- package/.next/standalone/node_modules/next/dist/server/dev/hot-reloader-webpack.js +1 -1
- package/.next/standalone/node_modules/next/dist/server/image-optimizer.js +2 -2
- package/.next/standalone/node_modules/next/dist/server/lib/app-info-log.js +1 -1
- package/.next/standalone/node_modules/next/dist/server/lib/start-server.js +1 -1
- package/.next/standalone/node_modules/next/dist/telemetry/anonymous-meta.js +1 -1
- package/.next/standalone/node_modules/next/dist/telemetry/events/swc-load-failure.js +1 -1
- package/.next/standalone/node_modules/next/dist/telemetry/events/version.js +2 -2
- package/.next/standalone/node_modules/next/package.json +11 -11
- package/.next/standalone/package.json +15 -14
- package/.next/standalone/sdk/python/CHANGELOG.md +155 -0
- package/.next/standalone/sdk/python/README.md +8 -0
- package/.next/standalone/sdk/python/examples/evaluator_worker.py +121 -0
- package/.next/standalone/sdk/python/failproofai_sdk/_version.py +1 -1
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/__init__.py +101 -0
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/__main__.py +49 -0
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/_sandbox_runner.py +63 -0
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/authoring.py +404 -0
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/client.py +299 -0
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/protocol.py +754 -0
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/runtime.py +936 -0
- package/.next/standalone/sdk/python/failproofai_sdk/evaluator/source.py +684 -0
- package/.next/standalone/sdk/python/tests/fixtures/evaluator_v2/README.md +28 -0
- package/.next/standalone/sdk/python/tests/fixtures/evaluator_v2/contract.json +252 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_authoring.py +124 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_client.py +253 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_example.py +35 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_http_e2e.py +636 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_main.py +47 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_protocol.py +246 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_review_fixes.py +164 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_runtime.py +1103 -0
- package/.next/standalone/sdk/python/tests/test_evaluator_source.py +430 -0
- package/.next/standalone/sdk/python/tests/test_zero_dependencies.py +17 -0
- package/.next/standalone/sdk/python/uv.lock +18 -18
- package/.next/standalone/server.js +1 -1
- package/README.md +19 -95
- package/bin/failproofai.mjs +7 -24
- package/dist/cli.mjs +19910 -25855
- package/dist/worker.mjs +4818 -6344
- package/lib/auth/api-server-client.ts +28 -0
- package/lib/claude-sessions.ts +101 -27
- package/lib/cli-registry.ts +1 -33
- package/lib/download-session.ts +0 -37
- package/lib/projects.ts +3 -27
- package/package.json +15 -14
- package/pi-extension/index.ts +42 -4
- package/scripts/changelog-open.py +115 -0
- package/src/audit/cli-adapters/index.ts +0 -24
- package/src/audit/cli.ts +134 -0
- package/src/audit/desktop-notify.ts +420 -0
- package/src/audit/harm-report.ts +101 -0
- package/src/audit/index.ts +141 -13
- package/src/audit/leak-fingerprint.ts +200 -0
- package/src/audit/leak-notice.ts +161 -0
- package/src/audit/leak-record.ts +232 -0
- package/src/audit/leak-scan.ts +292 -0
- package/src/audit/leak-store.ts +217 -0
- package/src/audit/macos-notifier.ts +310 -0
- package/src/audit/redact-example.ts +266 -14
- package/src/audit/report-harm.ts +12 -1
- package/src/audit/report.ts +6 -3
- package/src/audit/schedule-cli.ts +37 -0
- package/src/audit/scoring.ts +49 -0
- package/src/audit/types.ts +41 -0
- package/src/hooks/builtin-policies.ts +71 -8
- package/src/hooks/configure-wizard.ts +20 -0
- package/src/hooks/enforcement-capability.ts +0 -97
- package/src/hooks/fp-config.ts +71 -11
- package/src/hooks/fp-home.ts +27 -0
- package/src/hooks/handler.ts +64 -62
- package/src/hooks/harness-cli.ts +1 -3
- package/src/hooks/integrations.ts +49 -769
- package/src/hooks/normalize-cli-payload.ts +0 -126
- package/src/hooks/notice.ts +155 -0
- package/src/hooks/policy-evaluator.ts +6 -297
- package/src/hooks/resolve-transcript-path.ts +0 -10
- package/src/hooks/tool-name-canonicalize.ts +0 -64
- package/src/hooks/types.ts +2 -533
- package/src/hooks/uninstall-cli.ts +15 -0
- package/.next/standalone/.grok/hooks/failproofai.json +0 -172
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__07hor7h._.js +0 -5
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0elsuli._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0gnj_qa._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0ljuzm_._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0w23bpg._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0x5c7sh._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__0yik9_w._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__1jg6s9m._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__1s6qci9._.js +0 -3
- package/.next/standalone/.next/server/chunks/[root-of-the-server]__1xfm05z._.js +0 -3
- package/.next/standalone/.next/server/chunks/_0tdy_6l._.js +0 -3
- package/.next/standalone/.next/server/chunks/lib_grok-projects_ts_1i2epl_._.js +0 -3
- package/.next/standalone/.next/server/chunks/lib_qwen-projects_ts_0wu_a8e._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0_4gcz8._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0d--wmy._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0qqrlf_._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__0s11fx1._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__16gamwq._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1bbk-si._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1c3ybxr._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__1izi7zn._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/[root-of-the-server]__21bj7uw._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/_00z5uuq._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/_1byck5v._.js +0 -23
- package/.next/standalone/.next/server/chunks/ssr/_1ylae7o._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/_1zopuov._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/lib_cline-projects_ts_13_hbdd._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/lib_grok-projects_ts_0rc-d0o._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/lib_ori-projects_ts_04tofqo._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/lib_qwen-projects_ts_0w3i9o7._.js +0 -3
- package/.next/standalone/.next/server/chunks/ssr/node_modules_html-to-image_es_index_1ao30b1.js +0 -3
- package/.next/standalone/.next/static/chunks/0hzmxwb2pvk5p.js +0 -6
- package/.next/standalone/.next/static/chunks/0j884dwnwtyrt.js +0 -1
- package/.next/standalone/.next/static/chunks/0mr9dg3bnloa1.js +0 -1
- package/.next/standalone/.next/static/chunks/0u64yey446k4f.css +0 -1
- package/.next/standalone/.next/static/chunks/3gf1jg-5m-u1t.css +0 -2
- package/.next/standalone/.next/static/chunks/3qfoeqhy5l10x.js +0 -1
- package/.next/standalone/.next/static/chunks/3ws7eljkq25wg.js +0 -69
- package/.next/standalone/.next/static/chunks/3zkg2s2vzxc3d.js +0 -1
- package/.next/standalone/.qwen/settings.json +0 -251
- package/.next/standalone/assets/logos/cline.png +0 -0
- package/.next/standalone/assets/logos/grok-dark.svg +0 -1
- package/.next/standalone/assets/logos/grok-light.svg +0 -1
- package/.next/standalone/assets/logos/ori-dark.svg +0 -1
- package/.next/standalone/assets/logos/ori-light.svg +0 -1
- package/.next/standalone/assets/logos/qwen.svg +0 -1
- package/.next/standalone/lib/cline-projects.ts +0 -155
- package/.next/standalone/lib/cline-sessions.ts +0 -268
- package/.next/standalone/lib/grok-projects.ts +0 -123
- package/.next/standalone/lib/grok-sessions.ts +0 -355
- package/.next/standalone/lib/ori-projects.ts +0 -141
- package/.next/standalone/lib/ori-sessions.ts +0 -221
- package/.next/standalone/lib/qwen-projects.ts +0 -106
- package/.next/standalone/lib/qwen-sessions.ts +0 -313
- package/lib/cline-projects.ts +0 -155
- package/lib/cline-sessions.ts +0 -268
- package/lib/grok-projects.ts +0 -123
- package/lib/grok-sessions.ts +0 -355
- package/lib/ori-projects.ts +0 -141
- package/lib/ori-sessions.ts +0 -221
- package/lib/qwen-projects.ts +0 -106
- package/lib/qwen-sessions.ts +0 -313
- package/src/audit/cli-adapters/cline.ts +0 -50
- package/src/audit/cli-adapters/grok.ts +0 -70
- package/src/audit/cli-adapters/ori.ts +0 -55
- package/src/audit/cli-adapters/qwen.ts +0 -71
- package/src/hooks/batch-expand.ts +0 -380
- package/src/hooks/batch-fanout.ts +0 -142
- package/src/hooks/risk-patterns.ts +0 -26
- /package/.next/standalone/.next/static/{qJx3mbOQV-5ElrcRb4Sp7 → aKNv4Kmv98Xpwns0Xa-C-}/_buildManifest.js +0 -0
- /package/.next/standalone/.next/static/{qJx3mbOQV-5ElrcRb4Sp7 → aKNv4Kmv98Xpwns0Xa-C-}/_clientMiddlewareManifest.js +0 -0
- /package/.next/standalone/.next/static/{qJx3mbOQV-5ElrcRb4Sp7 → aKNv4Kmv98Xpwns0Xa-C-}/_ssgManifest.js +0 -0
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Authoring and worker primitives for FailproofAI Evaluator v2.
|
|
2
|
+
|
|
3
|
+
This namespace is intentionally lazy relative to :mod:`failproofai_sdk`: users
|
|
4
|
+
who only emit telemetry do not import evaluator networking or runtime code.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from failproofai_sdk.evaluator.authoring import (
|
|
8
|
+
Assertion,
|
|
9
|
+
ConditionResult,
|
|
10
|
+
EvalDefinition,
|
|
11
|
+
EvalResult,
|
|
12
|
+
Evaluator,
|
|
13
|
+
Metric,
|
|
14
|
+
Score,
|
|
15
|
+
)
|
|
16
|
+
from failproofai_sdk.evaluator.client import EvaluatorAPIError, EvaluatorClient
|
|
17
|
+
from failproofai_sdk.evaluator.protocol import (
|
|
18
|
+
Assignment,
|
|
19
|
+
AssignmentDefinition,
|
|
20
|
+
CatalogDefinition,
|
|
21
|
+
ClaimRequest,
|
|
22
|
+
ClaimResponse,
|
|
23
|
+
ErrorResponse,
|
|
24
|
+
EvalSelection,
|
|
25
|
+
ExecutionMode,
|
|
26
|
+
EvaluatorKind,
|
|
27
|
+
HeartbeatRequest,
|
|
28
|
+
HeartbeatResponse,
|
|
29
|
+
HeartbeatRun,
|
|
30
|
+
PlannedRun,
|
|
31
|
+
PlanRequest,
|
|
32
|
+
PlanResponse,
|
|
33
|
+
DefinitionsResponse,
|
|
34
|
+
ProtocolError,
|
|
35
|
+
RegisterRequest,
|
|
36
|
+
RegisterResponse,
|
|
37
|
+
RemoteError,
|
|
38
|
+
ResultItem,
|
|
39
|
+
ResultKind,
|
|
40
|
+
ResultRequest,
|
|
41
|
+
ResultResponse,
|
|
42
|
+
SessionTranscript,
|
|
43
|
+
SkippedEval,
|
|
44
|
+
TerminalRunStatus,
|
|
45
|
+
TranscriptEvent,
|
|
46
|
+
UnsupportedProtocolVersion,
|
|
47
|
+
)
|
|
48
|
+
from failproofai_sdk.evaluator.runtime import WorkerConfig, WorkerRuntime
|
|
49
|
+
from failproofai_sdk.evaluator.source import (
|
|
50
|
+
UnsafeEvaluatorSource,
|
|
51
|
+
compile_condition,
|
|
52
|
+
compile_evaluator,
|
|
53
|
+
source_checksum,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
__all__ = [
|
|
57
|
+
"Assertion",
|
|
58
|
+
"Assignment",
|
|
59
|
+
"AssignmentDefinition",
|
|
60
|
+
"CatalogDefinition",
|
|
61
|
+
"ClaimRequest",
|
|
62
|
+
"ClaimResponse",
|
|
63
|
+
"ConditionResult",
|
|
64
|
+
"ErrorResponse",
|
|
65
|
+
"EvalDefinition",
|
|
66
|
+
"EvalResult",
|
|
67
|
+
"EvalSelection",
|
|
68
|
+
"ExecutionMode",
|
|
69
|
+
"Evaluator",
|
|
70
|
+
"EvaluatorAPIError",
|
|
71
|
+
"EvaluatorClient",
|
|
72
|
+
"EvaluatorKind",
|
|
73
|
+
"HeartbeatRequest",
|
|
74
|
+
"HeartbeatResponse",
|
|
75
|
+
"HeartbeatRun",
|
|
76
|
+
"Metric",
|
|
77
|
+
"PlanRequest",
|
|
78
|
+
"PlanResponse",
|
|
79
|
+
"DefinitionsResponse",
|
|
80
|
+
"PlannedRun",
|
|
81
|
+
"ProtocolError",
|
|
82
|
+
"RegisterRequest",
|
|
83
|
+
"RegisterResponse",
|
|
84
|
+
"RemoteError",
|
|
85
|
+
"ResultItem",
|
|
86
|
+
"ResultKind",
|
|
87
|
+
"ResultRequest",
|
|
88
|
+
"ResultResponse",
|
|
89
|
+
"Score",
|
|
90
|
+
"SessionTranscript",
|
|
91
|
+
"SkippedEval",
|
|
92
|
+
"TerminalRunStatus",
|
|
93
|
+
"TranscriptEvent",
|
|
94
|
+
"UnsupportedProtocolVersion",
|
|
95
|
+
"WorkerConfig",
|
|
96
|
+
"WorkerRuntime",
|
|
97
|
+
"UnsafeEvaluatorSource",
|
|
98
|
+
"compile_condition",
|
|
99
|
+
"compile_evaluator",
|
|
100
|
+
"source_checksum",
|
|
101
|
+
]
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""Run an evaluator declared as ``module:attribute``."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import importlib
|
|
7
|
+
import os
|
|
8
|
+
from collections.abc import Sequence
|
|
9
|
+
|
|
10
|
+
from failproofai_sdk.evaluator.authoring import Evaluator
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def load_evaluator(spec: str) -> Evaluator:
|
|
14
|
+
module_name, separator, attribute = spec.partition(":")
|
|
15
|
+
if not module_name:
|
|
16
|
+
raise ValueError("evaluator module must not be empty")
|
|
17
|
+
if not separator:
|
|
18
|
+
attribute = "app"
|
|
19
|
+
if not attribute:
|
|
20
|
+
raise ValueError("evaluator attribute must not be empty")
|
|
21
|
+
module = importlib.import_module(module_name)
|
|
22
|
+
try:
|
|
23
|
+
evaluator = getattr(module, attribute)
|
|
24
|
+
except AttributeError as error:
|
|
25
|
+
raise ValueError(f"{spec!r} does not define {attribute!r}") from error
|
|
26
|
+
if not isinstance(evaluator, Evaluator):
|
|
27
|
+
raise TypeError(
|
|
28
|
+
f"{spec!r} resolved to {type(evaluator).__name__}, not Evaluator"
|
|
29
|
+
)
|
|
30
|
+
return evaluator
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
34
|
+
parser = argparse.ArgumentParser(prog="python -m failproofai_sdk.evaluator")
|
|
35
|
+
parser.add_argument(
|
|
36
|
+
"module",
|
|
37
|
+
nargs="?",
|
|
38
|
+
default=os.environ.get("FAILPROOFAI_EVALUATOR_MODULE"),
|
|
39
|
+
help="Python module and optional attribute (for example my_evals:app)",
|
|
40
|
+
)
|
|
41
|
+
args = parser.parse_args(argv)
|
|
42
|
+
if not args.module:
|
|
43
|
+
parser.error("module is required (or set FAILPROOFAI_EVALUATOR_MODULE)")
|
|
44
|
+
load_evaluator(args.module).run_from_env()
|
|
45
|
+
return 0
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
if __name__ == "__main__":
|
|
49
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Subprocess entry point for the managed-source sandbox.
|
|
2
|
+
|
|
3
|
+
Invoked as ``python -m failproofai_sdk.evaluator._sandbox_runner <input-file>`` by
|
|
4
|
+
``source._run_sandboxed``. The input file holds a pickled
|
|
5
|
+
``(kind, source, session_wire, cpu_seconds, mem_bytes, eval_key)`` tuple. This
|
|
6
|
+
process installs hard ``RLIMIT_CPU`` + ``RLIMIT_AS`` limits ON ITSELF, evaluates
|
|
7
|
+
the re-validated server-authored source against the reconstructed transcript,
|
|
8
|
+
validates + bounds the result, and writes a pickled ``("ok", result)`` /
|
|
9
|
+
``("err", type_name, message)`` outcome to stdout.
|
|
10
|
+
|
|
11
|
+
This is a FRESH exec'd process — never a fork of the multi-threaded worker — so
|
|
12
|
+
there is no inherited-lock deadlock (forking a process that has an asyncio loop,
|
|
13
|
+
an executor pool and a writer daemon hangs the child). The parent enforces the
|
|
14
|
+
wall-clock bound and the output-size bound by reading only so far and killing
|
|
15
|
+
this process on timeout or overflow.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import pickle
|
|
21
|
+
import sys
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _main() -> int:
|
|
25
|
+
with open(sys.argv[1], "rb") as handle:
|
|
26
|
+
kind, source, session_wire, cpu_seconds, mem_bytes, eval_key = pickle.loads(
|
|
27
|
+
handle.read()
|
|
28
|
+
)
|
|
29
|
+
# Imported here, in the child, so the import cost is never on the worker's path.
|
|
30
|
+
from failproofai_sdk.evaluator.protocol import SessionTranscript
|
|
31
|
+
from failproofai_sdk.evaluator.source import (
|
|
32
|
+
SANDBOX_MAX_RESULT_BYTES,
|
|
33
|
+
_install_limits,
|
|
34
|
+
_raw_eval,
|
|
35
|
+
)
|
|
36
|
+
|
|
37
|
+
try:
|
|
38
|
+
session = SessionTranscript.from_wire(session_wire)
|
|
39
|
+
# Compile + re-validate BEFORE the limits so validation cost is not charged
|
|
40
|
+
# against the eval's CPU budget; the limits bind the eval itself.
|
|
41
|
+
run = _raw_eval(source, kind)
|
|
42
|
+
_install_limits(cpu_seconds, mem_bytes)
|
|
43
|
+
result = run(session)
|
|
44
|
+
# Bound the result INSIDE the sandbox before it crosses back: result_items
|
|
45
|
+
# enforces the 25-result limit + field validation, so a huge result
|
|
46
|
+
# (`metrics={str(x):1 for x in range(100000)}`) raises here instead of
|
|
47
|
+
# being serialized and shipped to the parent.
|
|
48
|
+
if kind == "evaluator":
|
|
49
|
+
result.result_items(eval_key or "result")
|
|
50
|
+
payload = pickle.dumps(("ok", result))
|
|
51
|
+
if len(payload) > SANDBOX_MAX_RESULT_BYTES:
|
|
52
|
+
payload = pickle.dumps(
|
|
53
|
+
("err", "ResultTooLarge", "evaluation result exceeds the size limit")
|
|
54
|
+
)
|
|
55
|
+
except BaseException as error: # noqa: BLE001 - relay type+msg, this process is the boundary
|
|
56
|
+
payload = pickle.dumps(("err", type(error).__name__, str(error)[:500]))
|
|
57
|
+
sys.stdout.buffer.write(payload)
|
|
58
|
+
sys.stdout.buffer.flush()
|
|
59
|
+
return 0
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
if __name__ == "__main__":
|
|
63
|
+
raise SystemExit(_main())
|
|
@@ -0,0 +1,404 @@
|
|
|
1
|
+
"""Evaluator definition registry and typed author results."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import inspect
|
|
7
|
+
import json
|
|
8
|
+
import math
|
|
9
|
+
import re
|
|
10
|
+
from collections.abc import Awaitable, Callable, Mapping
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from failproofai_sdk.evaluator.protocol import (
|
|
15
|
+
MAX_CATALOG_DEFINITIONS,
|
|
16
|
+
MAX_DESCRIPTION_BYTES,
|
|
17
|
+
MAX_DISPLAY_NAME_BYTES,
|
|
18
|
+
MAX_DISPLAY_VALUE_BYTES,
|
|
19
|
+
MAX_EVAL_KEY_BYTES,
|
|
20
|
+
MAX_LABEL_BYTES,
|
|
21
|
+
MAX_LABELS_PER_RESULT,
|
|
22
|
+
MAX_REASONING_BYTES,
|
|
23
|
+
MAX_RESULTS_PER_RUN,
|
|
24
|
+
MAX_SUMMARY_BYTES,
|
|
25
|
+
MAX_UNIT_BYTES,
|
|
26
|
+
MAX_VERSION_BYTES,
|
|
27
|
+
CatalogDefinition,
|
|
28
|
+
ResultItem,
|
|
29
|
+
ResultKind,
|
|
30
|
+
SessionTranscript,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
_KEY = re.compile(r"^[a-z][a-z0-9_]*$")
|
|
34
|
+
EvalFunction = Callable[[SessionTranscript], "EvalResult | Awaitable[EvalResult]"]
|
|
35
|
+
ConditionFunction = Callable[
|
|
36
|
+
[SessionTranscript], "bool | ConditionResult | Awaitable[bool | ConditionResult]"
|
|
37
|
+
]
|
|
38
|
+
CancellationFunction = Callable[[SessionTranscript], "Any | Awaitable[Any]"]
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _bounded(value: str, *, field_name: str, maximum: int) -> str:
|
|
42
|
+
if not isinstance(value, str):
|
|
43
|
+
raise TypeError(f"{field_name} must be a string")
|
|
44
|
+
if not value:
|
|
45
|
+
raise ValueError(f"{field_name} must not be empty")
|
|
46
|
+
size = len(value.encode("utf-8"))
|
|
47
|
+
if size > maximum:
|
|
48
|
+
raise ValueError(f"{field_name} is {size} bytes; maximum is {maximum}")
|
|
49
|
+
# Reject C0 control characters and DEL, matching the server's `check_bounded`
|
|
50
|
+
# (server/src/evaluator/protocol.rs). Without this the SDK accepts a string —
|
|
51
|
+
# e.g. reasoning/summary quoting transcript text that contains an ANSI escape
|
|
52
|
+
# or NUL — that the server then rejects with a NON-RETRYABLE 422, so a
|
|
53
|
+
# successful evaluation is silently lost and its assignment dead-letters.
|
|
54
|
+
# TAB, LF and CR are kept because real multi-line reasoning uses them.
|
|
55
|
+
bad = next(
|
|
56
|
+
(
|
|
57
|
+
ch
|
|
58
|
+
for ch in value
|
|
59
|
+
if (ord(ch) < 0x20 and ch not in "\t\n\r") or ord(ch) == 0x7F
|
|
60
|
+
),
|
|
61
|
+
None,
|
|
62
|
+
)
|
|
63
|
+
if bad is not None:
|
|
64
|
+
raise ValueError(
|
|
65
|
+
f"{field_name} must not contain control characters "
|
|
66
|
+
f"(found U+{ord(bad):04X})"
|
|
67
|
+
)
|
|
68
|
+
return value
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _finite(value: float, field_name: str) -> float:
|
|
72
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
73
|
+
raise TypeError(f"{field_name} must be a number")
|
|
74
|
+
result = float(value)
|
|
75
|
+
if not math.isfinite(result):
|
|
76
|
+
raise ValueError(f"{field_name} must be finite")
|
|
77
|
+
return result
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _labels(values: tuple[str, ...] | list[str]) -> tuple[str, ...]:
|
|
81
|
+
if len(values) > MAX_LABELS_PER_RESULT:
|
|
82
|
+
raise ValueError(f"at most {MAX_LABELS_PER_RESULT} labels are allowed")
|
|
83
|
+
normalized = []
|
|
84
|
+
for label in values:
|
|
85
|
+
normalized.append(_bounded(label, field_name="label", maximum=MAX_LABEL_BYTES))
|
|
86
|
+
if len(set(normalized)) != len(normalized):
|
|
87
|
+
raise ValueError("labels must be unique")
|
|
88
|
+
return tuple(sorted(normalized))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
@dataclass(frozen=True)
|
|
92
|
+
class Score:
|
|
93
|
+
value: float
|
|
94
|
+
passed: bool | None = None
|
|
95
|
+
unit: str = "ratio"
|
|
96
|
+
display_value: str | None = None
|
|
97
|
+
description: str | None = None
|
|
98
|
+
|
|
99
|
+
def __post_init__(self) -> None:
|
|
100
|
+
value = _finite(self.value, "score value")
|
|
101
|
+
if not 0 <= value <= 1:
|
|
102
|
+
raise ValueError("score value must be between 0 and 1")
|
|
103
|
+
object.__setattr__(self, "value", value)
|
|
104
|
+
if self.passed is not None and not isinstance(self.passed, bool):
|
|
105
|
+
raise TypeError("score passed must be a boolean or None")
|
|
106
|
+
_validate_result_text(self.unit, self.display_value, self.description)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass(frozen=True)
|
|
110
|
+
class Metric:
|
|
111
|
+
value: float
|
|
112
|
+
unit: str = ""
|
|
113
|
+
display_value: str | None = None
|
|
114
|
+
description: str | None = None
|
|
115
|
+
|
|
116
|
+
def __post_init__(self) -> None:
|
|
117
|
+
object.__setattr__(self, "value", _finite(self.value, "metric value"))
|
|
118
|
+
_validate_result_text(self.unit, self.display_value, self.description)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@dataclass(frozen=True)
|
|
122
|
+
class Assertion:
|
|
123
|
+
passed: bool
|
|
124
|
+
description: str | None = None
|
|
125
|
+
|
|
126
|
+
def __post_init__(self) -> None:
|
|
127
|
+
if not isinstance(self.passed, bool):
|
|
128
|
+
raise TypeError("assertion passed must be a boolean")
|
|
129
|
+
if self.description is not None:
|
|
130
|
+
_bounded(
|
|
131
|
+
self.description,
|
|
132
|
+
field_name="description",
|
|
133
|
+
maximum=MAX_DESCRIPTION_BYTES,
|
|
134
|
+
)
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass(frozen=True)
|
|
138
|
+
class ConditionResult:
|
|
139
|
+
applicable: bool
|
|
140
|
+
reason_code: str = "condition_false"
|
|
141
|
+
|
|
142
|
+
def __post_init__(self) -> None:
|
|
143
|
+
if not isinstance(self.applicable, bool):
|
|
144
|
+
raise TypeError("condition applicable must be a boolean")
|
|
145
|
+
_validate_key(self.reason_code, "condition reason code")
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _validate_result_text(
|
|
149
|
+
unit: str, display_value: str | None, description: str | None
|
|
150
|
+
) -> None:
|
|
151
|
+
if unit:
|
|
152
|
+
_bounded(unit, field_name="unit", maximum=MAX_UNIT_BYTES)
|
|
153
|
+
if display_value is not None:
|
|
154
|
+
_bounded(
|
|
155
|
+
display_value,
|
|
156
|
+
field_name="display value",
|
|
157
|
+
maximum=MAX_DISPLAY_VALUE_BYTES,
|
|
158
|
+
)
|
|
159
|
+
if description is not None:
|
|
160
|
+
_bounded(
|
|
161
|
+
description,
|
|
162
|
+
field_name="description",
|
|
163
|
+
maximum=MAX_DESCRIPTION_BYTES,
|
|
164
|
+
)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
@dataclass(frozen=True)
|
|
168
|
+
class EvalResult:
|
|
169
|
+
score: Score | None = None
|
|
170
|
+
metrics: Mapping[str, Metric | float] = field(default_factory=dict)
|
|
171
|
+
assertions: Mapping[str, Assertion | bool] = field(default_factory=dict)
|
|
172
|
+
reasoning: str | None = None
|
|
173
|
+
summary: str | None = None
|
|
174
|
+
labels: tuple[str, ...] = ()
|
|
175
|
+
|
|
176
|
+
def __post_init__(self) -> None:
|
|
177
|
+
if self.reasoning is not None:
|
|
178
|
+
_bounded(
|
|
179
|
+
self.reasoning,
|
|
180
|
+
field_name="reasoning",
|
|
181
|
+
maximum=MAX_REASONING_BYTES,
|
|
182
|
+
)
|
|
183
|
+
if self.summary is not None:
|
|
184
|
+
_bounded(self.summary, field_name="summary", maximum=MAX_SUMMARY_BYTES)
|
|
185
|
+
object.__setattr__(self, "labels", _labels(list(self.labels)))
|
|
186
|
+
|
|
187
|
+
def result_items(self, eval_key: str) -> tuple[ResultItem, ...]:
|
|
188
|
+
items: list[ResultItem] = []
|
|
189
|
+
if self.score is not None:
|
|
190
|
+
items.append(
|
|
191
|
+
ResultItem(
|
|
192
|
+
result_key=eval_key,
|
|
193
|
+
result_kind=ResultKind.SCORE,
|
|
194
|
+
numeric_value=self.score.value,
|
|
195
|
+
bool_value=self.score.passed,
|
|
196
|
+
unit=self.score.unit,
|
|
197
|
+
display_value=self.score.display_value,
|
|
198
|
+
description=self.score.description,
|
|
199
|
+
reasoning=self.reasoning,
|
|
200
|
+
labels=self.labels,
|
|
201
|
+
)
|
|
202
|
+
)
|
|
203
|
+
for key, raw_metric in sorted(self.metrics.items()):
|
|
204
|
+
_validate_key(key, "metric key")
|
|
205
|
+
metric = (
|
|
206
|
+
raw_metric if isinstance(raw_metric, Metric) else Metric(raw_metric)
|
|
207
|
+
)
|
|
208
|
+
items.append(
|
|
209
|
+
ResultItem(
|
|
210
|
+
result_key=key,
|
|
211
|
+
result_kind=ResultKind.METRIC,
|
|
212
|
+
numeric_value=metric.value,
|
|
213
|
+
unit=metric.unit,
|
|
214
|
+
display_value=metric.display_value,
|
|
215
|
+
description=metric.description,
|
|
216
|
+
# A metric-kind eval's primary result IS the metric whose
|
|
217
|
+
# key equals eval_key; attach the eval's reasoning there so
|
|
218
|
+
# it is not silently dropped for non-score evals.
|
|
219
|
+
reasoning=self.reasoning if key == eval_key else None,
|
|
220
|
+
labels=self.labels,
|
|
221
|
+
)
|
|
222
|
+
)
|
|
223
|
+
for key, raw_assertion in sorted(self.assertions.items()):
|
|
224
|
+
_validate_key(key, "assertion key")
|
|
225
|
+
assertion = (
|
|
226
|
+
raw_assertion
|
|
227
|
+
if isinstance(raw_assertion, Assertion)
|
|
228
|
+
else Assertion(raw_assertion)
|
|
229
|
+
)
|
|
230
|
+
items.append(
|
|
231
|
+
ResultItem(
|
|
232
|
+
result_key=key,
|
|
233
|
+
result_kind=ResultKind.ASSERTION,
|
|
234
|
+
bool_value=assertion.passed,
|
|
235
|
+
description=assertion.description,
|
|
236
|
+
# An assertion-kind eval's primary result is the assertion
|
|
237
|
+
# whose key equals eval_key; carry the eval's reasoning there
|
|
238
|
+
# so a non-score eval does not lose it.
|
|
239
|
+
reasoning=self.reasoning if key == eval_key else None,
|
|
240
|
+
labels=self.labels,
|
|
241
|
+
)
|
|
242
|
+
)
|
|
243
|
+
if not items:
|
|
244
|
+
raise ValueError("an EvalResult must contain a score, metric, or assertion")
|
|
245
|
+
if len(items) > MAX_RESULTS_PER_RUN:
|
|
246
|
+
raise ValueError(
|
|
247
|
+
f"an EvalResult may contain at most {MAX_RESULTS_PER_RUN} results"
|
|
248
|
+
)
|
|
249
|
+
keys = [item.result_key for item in items]
|
|
250
|
+
if len(keys) != len(set(keys)):
|
|
251
|
+
raise ValueError("result keys must be unique within one evaluation run")
|
|
252
|
+
return tuple(items)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def _validate_key(value: str, field_name: str = "eval_key") -> str:
|
|
256
|
+
_bounded(value, field_name=field_name, maximum=MAX_EVAL_KEY_BYTES)
|
|
257
|
+
if not _KEY.fullmatch(value):
|
|
258
|
+
raise ValueError(f"{field_name} must match {_KEY.pattern}")
|
|
259
|
+
return value
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
@dataclass(frozen=True)
|
|
263
|
+
class EvalDefinition:
|
|
264
|
+
eval_key: str
|
|
265
|
+
display_name: str
|
|
266
|
+
eval_version: str
|
|
267
|
+
result_kind: ResultKind
|
|
268
|
+
labels: tuple[str, ...]
|
|
269
|
+
function: EvalFunction
|
|
270
|
+
condition: ConditionFunction | None
|
|
271
|
+
on_cancel: CancellationFunction | None
|
|
272
|
+
timeout_seconds: float | None
|
|
273
|
+
|
|
274
|
+
def catalog_definition(self) -> CatalogDefinition:
|
|
275
|
+
return CatalogDefinition(
|
|
276
|
+
eval_key=self.eval_key,
|
|
277
|
+
display_name=self.display_name,
|
|
278
|
+
eval_version=self.eval_version,
|
|
279
|
+
result_kind=self.result_kind,
|
|
280
|
+
labels=self.labels,
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
class Evaluator:
|
|
285
|
+
"""A process-local collection of explicitly versioned evaluations."""
|
|
286
|
+
|
|
287
|
+
def __init__(self, *, name: str, version: str) -> None:
|
|
288
|
+
self.name = _bounded(name, field_name="name", maximum=MAX_DISPLAY_NAME_BYTES)
|
|
289
|
+
self.version = _bounded(
|
|
290
|
+
version, field_name="version", maximum=MAX_VERSION_BYTES
|
|
291
|
+
)
|
|
292
|
+
self._definitions: dict[str, EvalDefinition] = {}
|
|
293
|
+
|
|
294
|
+
def eval(
|
|
295
|
+
self,
|
|
296
|
+
eval_key: str,
|
|
297
|
+
*,
|
|
298
|
+
version: str,
|
|
299
|
+
display_name: str | None = None,
|
|
300
|
+
result_kind: ResultKind | str = ResultKind.SCORE,
|
|
301
|
+
labels: tuple[str, ...] | list[str] = (),
|
|
302
|
+
when: ConditionFunction | None = None,
|
|
303
|
+
on_cancel: CancellationFunction | None = None,
|
|
304
|
+
timeout_seconds: float | None = None,
|
|
305
|
+
) -> Callable[[EvalFunction], EvalFunction]:
|
|
306
|
+
key = _validate_key(eval_key)
|
|
307
|
+
eval_version = _bounded(
|
|
308
|
+
version, field_name="eval version", maximum=MAX_VERSION_BYTES
|
|
309
|
+
)
|
|
310
|
+
display = _bounded(
|
|
311
|
+
display_name or eval_key.replace("_", " ").capitalize(),
|
|
312
|
+
field_name="display name",
|
|
313
|
+
maximum=MAX_DISPLAY_NAME_BYTES,
|
|
314
|
+
)
|
|
315
|
+
kind = ResultKind(result_kind)
|
|
316
|
+
normalized_labels = _labels(list(labels))
|
|
317
|
+
if timeout_seconds is not None:
|
|
318
|
+
timeout_seconds = _finite(timeout_seconds, "timeout_seconds")
|
|
319
|
+
if timeout_seconds <= 0:
|
|
320
|
+
raise ValueError("timeout_seconds must be greater than zero")
|
|
321
|
+
|
|
322
|
+
def register(function: EvalFunction) -> EvalFunction:
|
|
323
|
+
if key in self._definitions:
|
|
324
|
+
raise ValueError(f"duplicate eval key: {key}")
|
|
325
|
+
if len(self._definitions) >= MAX_CATALOG_DEFINITIONS:
|
|
326
|
+
raise ValueError(
|
|
327
|
+
f"an evaluator may define at most {MAX_CATALOG_DEFINITIONS} evaluations"
|
|
328
|
+
)
|
|
329
|
+
if not callable(function):
|
|
330
|
+
raise TypeError("evaluation must be callable")
|
|
331
|
+
if when is not None and not callable(when):
|
|
332
|
+
raise TypeError("when must be callable")
|
|
333
|
+
if on_cancel is not None and not callable(on_cancel):
|
|
334
|
+
raise TypeError("on_cancel must be callable")
|
|
335
|
+
self._definitions[key] = EvalDefinition(
|
|
336
|
+
eval_key=key,
|
|
337
|
+
display_name=display,
|
|
338
|
+
eval_version=eval_version,
|
|
339
|
+
result_kind=kind,
|
|
340
|
+
labels=normalized_labels,
|
|
341
|
+
function=function,
|
|
342
|
+
condition=when,
|
|
343
|
+
on_cancel=on_cancel,
|
|
344
|
+
timeout_seconds=timeout_seconds,
|
|
345
|
+
)
|
|
346
|
+
return function
|
|
347
|
+
|
|
348
|
+
return register
|
|
349
|
+
|
|
350
|
+
@property
|
|
351
|
+
def definitions(self) -> tuple[EvalDefinition, ...]:
|
|
352
|
+
return tuple(self._definitions[key] for key in sorted(self._definitions))
|
|
353
|
+
|
|
354
|
+
def catalog(self) -> tuple[CatalogDefinition, ...]:
|
|
355
|
+
return tuple(definition.catalog_definition() for definition in self.definitions)
|
|
356
|
+
|
|
357
|
+
@property
|
|
358
|
+
def catalog_revision(self) -> str:
|
|
359
|
+
payload = [item.to_wire() for item in self.catalog()]
|
|
360
|
+
canonical = json.dumps(
|
|
361
|
+
payload,
|
|
362
|
+
ensure_ascii=False,
|
|
363
|
+
allow_nan=False,
|
|
364
|
+
sort_keys=True,
|
|
365
|
+
separators=(",", ":"),
|
|
366
|
+
).encode("utf-8")
|
|
367
|
+
return "sha256:" + hashlib.sha256(canonical).hexdigest()
|
|
368
|
+
|
|
369
|
+
def definition(self, eval_key: str) -> EvalDefinition:
|
|
370
|
+
try:
|
|
371
|
+
return self._definitions[eval_key]
|
|
372
|
+
except KeyError as error:
|
|
373
|
+
raise KeyError(f"unknown eval key: {eval_key}") from error
|
|
374
|
+
|
|
375
|
+
def run_from_env(self) -> None:
|
|
376
|
+
"""Run this evaluator until the process receives a stop request."""
|
|
377
|
+
import asyncio
|
|
378
|
+
import signal
|
|
379
|
+
|
|
380
|
+
from failproofai_sdk.evaluator.runtime import WorkerConfig, WorkerRuntime
|
|
381
|
+
|
|
382
|
+
async def run() -> None:
|
|
383
|
+
runtime = WorkerRuntime(self, WorkerConfig.from_env())
|
|
384
|
+
loop = asyncio.get_running_loop()
|
|
385
|
+
for name in ("SIGINT", "SIGTERM"):
|
|
386
|
+
process_signal = getattr(signal, name, None)
|
|
387
|
+
if process_signal is None:
|
|
388
|
+
continue
|
|
389
|
+
try:
|
|
390
|
+
loop.add_signal_handler(process_signal, runtime.stop)
|
|
391
|
+
except (NotImplementedError, RuntimeError):
|
|
392
|
+
pass
|
|
393
|
+
await runtime.run_forever()
|
|
394
|
+
|
|
395
|
+
asyncio.run(run())
|
|
396
|
+
|
|
397
|
+
@staticmethod
|
|
398
|
+
async def call(
|
|
399
|
+
function: EvalFunction | ConditionFunction, session: SessionTranscript
|
|
400
|
+
) -> Any:
|
|
401
|
+
result = function(session)
|
|
402
|
+
if inspect.isawaitable(result):
|
|
403
|
+
return await result
|
|
404
|
+
return result
|