@hybridlabor-api/aos 4.4.1 → 4.4.2-beta.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/skills/firecrawl/SKILL.md +149 -0
- package/.agents/skills/firecrawl/rules/install.md +82 -0
- package/.agents/skills/firecrawl/rules/security.md +26 -0
- package/.agents/skills/firecrawl-agent/SKILL.md +58 -0
- package/.agents/skills/firecrawl-build/SKILL.md +39 -0
- package/.agents/skills/firecrawl-build-interact/SKILL.md +68 -0
- package/.agents/skills/firecrawl-build-onboarding/SKILL.md +103 -0
- package/.agents/skills/firecrawl-build-onboarding/references/auth-flow.md +39 -0
- package/.agents/skills/firecrawl-build-onboarding/references/project-setup.md +20 -0
- package/.agents/skills/firecrawl-build-onboarding/references/sdk-installation.md +17 -0
- package/.agents/skills/firecrawl-build-scrape/SKILL.md +69 -0
- package/.agents/skills/firecrawl-build-search/SKILL.md +69 -0
- package/.agents/skills/firecrawl-crawl/SKILL.md +59 -0
- package/.agents/skills/firecrawl-download/SKILL.md +70 -0
- package/.agents/skills/firecrawl-interact/SKILL.md +84 -0
- package/.agents/skills/firecrawl-map/SKILL.md +51 -0
- package/.agents/skills/firecrawl-scrape/SKILL.md +69 -0
- package/.agents/skills/firecrawl-search/SKILL.md +60 -0
- package/CLAUDE.md +573 -0
- package/README.de.md +0 -8
- package/README.md +0 -8
- package/README.pt.md +0 -8
- package/assets/brand/ao-ant-mark.svg +91 -0
- package/assets/brand/bdb-core-mark.svg +34 -0
- package/assets/brand/heimdall-guardian-mark.svg +16 -0
- package/assets/brand/memb-crystal-mark.svg +18 -0
- package/bin/aos-config.mjs +89 -0
- package/bin/aos-dashboard.mjs +608 -0
- package/bin/aos-uninstall.mjs +227 -0
- package/docs/sessions/audit-2026-09-14-silent-failures.md +114 -0
- package/docs/skills_table.md +0 -3
- package/installer.js +387 -237
- package/mcps/RhinoMCP/cc-plugin/.claude/settings.json +10 -0
- package/mcps/after-effects-mcp/build/index.js +840 -0
- package/mcps/after-effects-mcp/build/scripts/applyEffect.jsx +153 -0
- package/mcps/after-effects-mcp/build/scripts/applyEffectTemplate.jsx +218 -0
- package/mcps/after-effects-mcp/build/scripts/createComposition.jsx +71 -0
- package/mcps/after-effects-mcp/build/scripts/createShapeLayer.jsx +147 -0
- package/mcps/after-effects-mcp/build/scripts/createSolidLayer.jsx +114 -0
- package/mcps/after-effects-mcp/build/scripts/createTextLayer.jsx +115 -0
- package/mcps/after-effects-mcp/build/scripts/getLayerInfo.jsx +192 -0
- package/mcps/after-effects-mcp/build/scripts/getProjectInfo.jsx +90 -0
- package/mcps/after-effects-mcp/build/scripts/listCompositions.jsx +50 -0
- package/mcps/after-effects-mcp/build/scripts/mcp-bridge-auto.jsx +1773 -0
- package/mcps/after-effects-mcp/build/scripts/setLayerProperties.jsx +160 -0
- package/mcps/computer-use-mcp/dist/client.d.ts +150 -0
- package/mcps/computer-use-mcp/dist/client.js +136 -0
- package/mcps/computer-use-mcp/dist/entrypoint.d.ts +16 -0
- package/mcps/computer-use-mcp/dist/entrypoint.js +26 -0
- package/mcps/computer-use-mcp/dist/native.d.ts +212 -0
- package/mcps/computer-use-mcp/dist/native.js +50 -0
- package/mcps/computer-use-mcp/dist/server.d.ts +32 -0
- package/mcps/computer-use-mcp/dist/server.js +342 -0
- package/mcps/computer-use-mcp/dist/session.d.ts +101 -0
- package/mcps/computer-use-mcp/dist/session.js +2372 -0
- package/package.json +13 -4
- package/skills/global_config/aos-project-init/SKILL.md +97 -130
- package/skills/global_config/aos-project-init/assets/openwikiignore.template +62 -0
- package/skills/global_config/aos-project-init/scripts/aos-project-doctor.mjs +38 -1
- package/skills/global_config/aos-setup/SKILL.md +98 -8
- package/skills/global_config/aos-setup/scripts/aos-doctor.mjs +21 -2
- package/skills/global_config/ask-tim/SKILL.md +2 -7
- package/skills/global_config/openwiki-skill/scripts/openwiki_daemon.py +65 -12
- package/bin/setup-saas.mjs +0 -792
- package/mcps/blender-mcp-server/.dockerignore +0 -19
- package/mcps/blender-mcp-server/.github/workflows/ci.yml +0 -67
- package/mcps/blender-mcp-server/.github/workflows/publish-pypi.yml +0 -54
- package/mcps/blender-mcp-server/CHANGELOG.md +0 -64
- package/mcps/blender-mcp-server/CONTRIBUTING.md +0 -107
- package/mcps/blender-mcp-server/Dockerfile +0 -19
- package/mcps/blender-mcp-server/LICENSE +0 -21
- package/mcps/blender-mcp-server/README.md +0 -409
- package/mcps/blender-mcp-server/addon/__init__.py +0 -1440
- package/mcps/blender-mcp-server/addon/models.py +0 -223
- package/mcps/blender-mcp-server/pyproject.toml +0 -120
- package/mcps/blender-mcp-server/scripts/blender_bridge_request.py +0 -71
- package/mcps/blender-mcp-server/scripts/blender_create_test_cube.py +0 -49
- package/mcps/blender-mcp-server/scripts/blender_scene_info.py +0 -38
- package/mcps/blender-mcp-server/scripts/build_addon_zip.sh +0 -29
- package/mcps/blender-mcp-server/scripts/demos/README.md +0 -186
- package/mcps/blender-mcp-server/scripts/demos/dam_break_scene.py +0 -276
- package/mcps/blender-mcp-server/scripts/demos/open_dam_break_gui.py +0 -33
- package/mcps/blender-mcp-server/scripts/demos/pipe_movement_study.py +0 -518
- package/mcps/blender-mcp-server/scripts/demos/pipe_pressure_failure_scene.py +0 -806
- package/mcps/blender-mcp-server/scripts/demos/procedural_dam_break_scene.py +0 -600
- package/mcps/blender-mcp-server/scripts/demos/run_dam_break.py +0 -510
- package/mcps/blender-mcp-server/scripts/launch_blender_gui.py +0 -41
- package/mcps/blender-mcp-server/scripts/library/README.md +0 -64
- package/mcps/blender-mcp-server/scripts/library/apply_transforms.py +0 -46
- package/mcps/blender-mcp-server/scripts/library/camera.py +0 -51
- package/mcps/blender-mcp-server/scripts/library/collections.py +0 -65
- package/mcps/blender-mcp-server/scripts/library/create_mesh.py +0 -41
- package/mcps/blender-mcp-server/scripts/library/effector.py +0 -48
- package/mcps/blender-mcp-server/scripts/library/fluid_domain.py +0 -80
- package/mcps/blender-mcp-server/scripts/library/fluid_inflow.py +0 -98
- package/mcps/blender-mcp-server/scripts/library/frame_range.py +0 -35
- package/mcps/blender-mcp-server/scripts/library/keyframes.py +0 -59
- package/mcps/blender-mcp-server/scripts/library/rigid_body.py +0 -52
- package/mcps/blender-mcp-server/scripts/library/save_blend.py +0 -40
- package/mcps/blender-mcp-server/scripts/server_start.sh +0 -26
- package/mcps/blender-mcp-server/src/blender_mcp_server/__init__.py +0 -3
- package/mcps/blender-mcp-server/src/blender_mcp_server/headless.py +0 -330
- package/mcps/blender-mcp-server/src/blender_mcp_server/server.py +0 -606
- package/mcps/davinci-mcp-professional/.bandit +0 -35
- package/mcps/davinci-mcp-professional/.claude/commands/security-review.md +0 -191
- package/mcps/davinci-mcp-professional/.claude/settings.json +0 -1
- package/mcps/davinci-mcp-professional/.editorconfig +0 -27
- package/mcps/davinci-mcp-professional/.gitattributes +0 -207
- package/mcps/davinci-mcp-professional/.github/CODEOWNERS +0 -2
- package/mcps/davinci-mcp-professional/.github/ISSUE_TEMPLATE/bug_report.md +0 -33
- package/mcps/davinci-mcp-professional/.github/ISSUE_TEMPLATE/feature_request.md +0 -19
- package/mcps/davinci-mcp-professional/.github/PULL_REQUEST_TEMPLATE.md +0 -20
- package/mcps/davinci-mcp-professional/.github/SECURITY.md +0 -97
- package/mcps/davinci-mcp-professional/.github/dependabot.yml +0 -27
- package/mcps/davinci-mcp-professional/.github/workflows/ci.yml +0 -39
- package/mcps/davinci-mcp-professional/ATTRIBUTION.md +0 -4
- package/mcps/davinci-mcp-professional/AUTHORS.md +0 -9
- package/mcps/davinci-mcp-professional/BUGS.md +0 -55
- package/mcps/davinci-mcp-professional/CLAUDE.md +0 -229
- package/mcps/davinci-mcp-professional/CONTRIBUTING.md +0 -51
- package/mcps/davinci-mcp-professional/COPYING.md +0 -228
- package/mcps/davinci-mcp-professional/Doxyfile +0 -379
- package/mcps/davinci-mcp-professional/README.md +0 -143
- package/mcps/davinci-mcp-professional/SECURITY.md +0 -40
- package/mcps/davinci-mcp-professional/USING.md +0 -301
- package/mcps/davinci-mcp-professional/VERSION.md +0 -13
- package/mcps/davinci-mcp-professional/build.py +0 -169
- package/mcps/davinci-mcp-professional/claude_desktop_config_template.json +0 -11
- package/mcps/davinci-mcp-professional/hooks/ci-check.sh +0 -29
- package/mcps/davinci-mcp-professional/hooks/pre-push +0 -46
- package/mcps/davinci-mcp-professional/main.py +0 -40
- package/mcps/davinci-mcp-professional/mcp_server.py +0 -30
- package/mcps/davinci-mcp-professional/pyproject.toml +0 -115
- package/mcps/davinci-mcp-professional/security_audit.py +0 -379
- package/mcps/davinci-mcp-professional/src/davinci_mcp/__init__.py +0 -19
- package/mcps/davinci-mcp-professional/src/davinci_mcp/cli.py +0 -138
- package/mcps/davinci-mcp-professional/src/davinci_mcp/resolve_client.py +0 -342
- package/mcps/davinci-mcp-professional/src/davinci_mcp/resources/__init__.py +0 -60
- package/mcps/davinci-mcp-professional/src/davinci_mcp/server.py +0 -185
- package/mcps/davinci-mcp-professional/src/davinci_mcp/tools/__init__.py +0 -146
- package/mcps/davinci-mcp-professional/src/davinci_mcp/types.py +0 -125
- package/mcps/davinci-mcp-professional/src/davinci_mcp/utils/__init__.py +0 -19
- package/mcps/davinci-mcp-professional/src/davinci_mcp/utils/platform.py +0 -239
- package/mcps/davinci-mcp-professional/test.py +0 -186
- package/mcps/davinci-mcp-professional/uv.lock +0 -2073
- package/mcps/davinci-resolve-mcp/.claude/skills/media-analysis.md +0 -44
- package/mcps/davinci-resolve-mcp/.clinerules +0 -17
- package/mcps/davinci-resolve-mcp/.cursorrules +0 -17
- package/mcps/davinci-resolve-mcp/.github/copilot-instructions.md +0 -17
- package/mcps/davinci-resolve-mcp/.github/workflows/npm-publish.yml +0 -109
- package/mcps/davinci-resolve-mcp/.windsurfrules +0 -17
- package/mcps/davinci-resolve-mcp/AGENTS.md +0 -114
- package/mcps/davinci-resolve-mcp/CHANGELOG.md +0 -2648
- package/mcps/davinci-resolve-mcp/CLAUDE.md +0 -15
- package/mcps/davinci-resolve-mcp/LICENSE +0 -21
- package/mcps/davinci-resolve-mcp/README.md +0 -173
- package/mcps/davinci-resolve-mcp/SECURITY.md +0 -53
- package/mcps/davinci-resolve-mcp/bin/davinci-resolve-mcp.mjs +0 -450
- package/mcps/davinci-resolve-mcp/examples/README.md +0 -53
- package/mcps/davinci-resolve-mcp/examples/markers/README.md +0 -81
- package/mcps/davinci-resolve-mcp/examples/media/README.md +0 -94
- package/mcps/davinci-resolve-mcp/examples/timeline/README.md +0 -98
- package/mcps/davinci-resolve-mcp/install.py +0 -1687
- package/mcps/davinci-resolve-mcp/package.json +0 -52
- package/mcps/davinci-resolve-mcp/release-notes/v2.22.0.md +0 -51
- package/mcps/davinci-resolve-mcp/release-notes/v2.23.0.md +0 -51
- package/mcps/davinci-resolve-mcp/release-notes/v2.23.1.md +0 -35
- package/mcps/davinci-resolve-mcp/release-notes/v2.24.1.md +0 -32
- package/mcps/davinci-resolve-mcp/scripts/audit_api_parity.py +0 -275
- package/mcps/davinci-resolve-mcp/scripts/audit_readwrite_symmetry.py +0 -80
- package/mcps/davinci-resolve-mcp/scripts/doctor.py +0 -227
- package/mcps/davinci-resolve-mcp/scripts/gen_api_limitations.py +0 -144
- package/mcps/davinci-resolve-mcp/scripts/live_media_analysis_polish_probe.py +0 -65
- package/mcps/davinci-resolve-mcp/scripts/measure_bridge_cost.py +0 -65
- package/mcps/davinci-resolve-mcp/scripts/regen_panel_screenshots.py +0 -76
- package/mcps/davinci-resolve-mcp/src/__init__.py +0 -3
- package/mcps/davinci-resolve-mcp/src/analysis_dashboard.py +0 -15505
- package/mcps/davinci-resolve-mcp/src/batch_cli.py +0 -540
- package/mcps/davinci-resolve-mcp/src/control_panel.py +0 -13
- package/mcps/davinci-resolve-mcp/src/granular/__init__.py +0 -17
- package/mcps/davinci-resolve-mcp/src/granular/common.py +0 -741
- package/mcps/davinci-resolve-mcp/src/granular/folder.py +0 -420
- package/mcps/davinci-resolve-mcp/src/granular/gallery.py +0 -306
- package/mcps/davinci-resolve-mcp/src/granular/graph.py +0 -309
- package/mcps/davinci-resolve-mcp/src/granular/media_pool.py +0 -679
- package/mcps/davinci-resolve-mcp/src/granular/media_pool_item.py +0 -1020
- package/mcps/davinci-resolve-mcp/src/granular/media_storage.py +0 -179
- package/mcps/davinci-resolve-mcp/src/granular/project.py +0 -1654
- package/mcps/davinci-resolve-mcp/src/granular/resolve_control.py +0 -538
- package/mcps/davinci-resolve-mcp/src/granular/timeline.py +0 -1075
- package/mcps/davinci-resolve-mcp/src/granular/timeline_item.py +0 -2251
- package/mcps/davinci-resolve-mcp/src/resolve_mcp_server.py +0 -43
- package/mcps/davinci-resolve-mcp/src/server.py +0 -24116
- package/mcps/davinci-resolve-mcp/src/utils/__init__.py +0 -3
- package/mcps/davinci-resolve-mcp/src/utils/actor_identity.py +0 -46
- package/mcps/davinci-resolve-mcp/src/utils/analysis_caps.py +0 -678
- package/mcps/davinci-resolve-mcp/src/utils/analysis_memory.py +0 -755
- package/mcps/davinci-resolve-mcp/src/utils/analysis_runs.py +0 -307
- package/mcps/davinci-resolve-mcp/src/utils/analysis_store.py +0 -945
- package/mcps/davinci-resolve-mcp/src/utils/api_truth.py +0 -525
- package/mcps/davinci-resolve-mcp/src/utils/app_control.py +0 -319
- package/mcps/davinci-resolve-mcp/src/utils/audio_fairlight_live_probe.py +0 -263
- package/mcps/davinci-resolve-mcp/src/utils/brain_edits.py +0 -387
- package/mcps/davinci-resolve-mcp/src/utils/bridge_metrics.py +0 -76
- package/mcps/davinci-resolve-mcp/src/utils/cdl.py +0 -20
- package/mcps/davinci-resolve-mcp/src/utils/clip_query.py +0 -85
- package/mcps/davinci-resolve-mcp/src/utils/cloud_operations.py +0 -192
- package/mcps/davinci-resolve-mcp/src/utils/color_grade_live_probe.py +0 -444
- package/mcps/davinci-resolve-mcp/src/utils/contracts.py +0 -102
- package/mcps/davinci-resolve-mcp/src/utils/cut_ir.py +0 -115
- package/mcps/davinci-resolve-mcp/src/utils/dctl_templates.py +0 -368
- package/mcps/davinci-resolve-mcp/src/utils/deep_vision.py +0 -643
- package/mcps/davinci-resolve-mcp/src/utils/destructive_hook.py +0 -606
- package/mcps/davinci-resolve-mcp/src/utils/edit_engine.py +0 -737
- package/mcps/davinci-resolve-mcp/src/utils/embeddings.py +0 -922
- package/mcps/davinci-resolve-mcp/src/utils/entities.py +0 -579
- package/mcps/davinci-resolve-mcp/src/utils/extension_authoring_live_probe.py +0 -292
- package/mcps/davinci-resolve-mcp/src/utils/failure_tracker.py +0 -119
- package/mcps/davinci-resolve-mcp/src/utils/fuse_templates.py +0 -1968
- package/mcps/davinci-resolve-mcp/src/utils/fusion_composition_live_probe.py +0 -284
- package/mcps/davinci-resolve-mcp/src/utils/fusion_group_settings.py +0 -334
- package/mcps/davinci-resolve-mcp/src/utils/layout_presets.py +0 -333
- package/mcps/davinci-resolve-mcp/src/utils/mcp_stdio.py +0 -32
- package/mcps/davinci-resolve-mcp/src/utils/mcp_transport.py +0 -128
- package/mcps/davinci-resolve-mcp/src/utils/media_analysis.py +0 -7690
- package/mcps/davinci-resolve-mcp/src/utils/media_analysis_jobs.py +0 -805
- package/mcps/davinci-resolve-mcp/src/utils/media_pool_changes.py +0 -121
- package/mcps/davinci-resolve-mcp/src/utils/media_pool_ingest_live_probe.py +0 -592
- package/mcps/davinci-resolve-mcp/src/utils/multicam.py +0 -393
- package/mcps/davinci-resolve-mcp/src/utils/object_inspection.py +0 -301
- package/mcps/davinci-resolve-mcp/src/utils/page_lock.py +0 -75
- package/mcps/davinci-resolve-mcp/src/utils/platform.py +0 -157
- package/mcps/davinci-resolve-mcp/src/utils/proc.py +0 -26
- package/mcps/davinci-resolve-mcp/src/utils/project_cleanup.py +0 -73
- package/mcps/davinci-resolve-mcp/src/utils/project_lifecycle_live_probe.py +0 -376
- package/mcps/davinci-resolve-mcp/src/utils/project_lint.py +0 -131
- package/mcps/davinci-resolve-mcp/src/utils/project_properties.py +0 -601
- package/mcps/davinci-resolve-mcp/src/utils/project_spec.py +0 -459
- package/mcps/davinci-resolve-mcp/src/utils/readback.py +0 -91
- package/mcps/davinci-resolve-mcp/src/utils/render_deliver_live_probe.py +0 -384
- package/mcps/davinci-resolve-mcp/src/utils/resolve_ai_governance.py +0 -214
- package/mcps/davinci-resolve-mcp/src/utils/resolve_ai_ledger.py +0 -244
- package/mcps/davinci-resolve-mcp/src/utils/resolve_busy.py +0 -171
- package/mcps/davinci-resolve-mcp/src/utils/resolve_connection.py +0 -77
- package/mcps/davinci-resolve-mcp/src/utils/review_annotation_live_probe.py +0 -352
- package/mcps/davinci-resolve-mcp/src/utils/script_templates.py +0 -1193
- package/mcps/davinci-resolve-mcp/src/utils/shot_relationships.py +0 -544
- package/mcps/davinci-resolve-mcp/src/utils/structural_diff.py +0 -175
- package/mcps/davinci-resolve-mcp/src/utils/sync_detection.py +0 -889
- package/mcps/davinci-resolve-mcp/src/utils/timeline_brain_db.py +0 -826
- package/mcps/davinci-resolve-mcp/src/utils/timeline_conform_live_probe.py +0 -280
- package/mcps/davinci-resolve-mcp/src/utils/timeline_kernel_live_probe.py +0 -1092
- package/mcps/davinci-resolve-mcp/src/utils/timeline_kernel_probe.py +0 -185
- package/mcps/davinci-resolve-mcp/src/utils/timeline_title_text.py +0 -87
- package/mcps/davinci-resolve-mcp/src/utils/timeline_versioning.py +0 -730
- package/mcps/davinci-resolve-mcp/src/utils/update_check.py +0 -685
- package/mcps/vectorworks-mcp/.dockerignore +0 -8
- package/mcps/vectorworks-mcp/Dockerfile +0 -28
- package/mcps/vectorworks-mcp/README.md +0 -156
- package/mcps/vectorworks-mcp/app/__init__.py +0 -2
- package/mcps/vectorworks-mcp/app/api.py +0 -52
- package/mcps/vectorworks-mcp/app/chunking.py +0 -56
- package/mcps/vectorworks-mcp/app/config.py +0 -29
- package/mcps/vectorworks-mcp/app/embeddings.py +0 -28
- package/mcps/vectorworks-mcp/app/indexer.py +0 -106
- package/mcps/vectorworks-mcp/app/main.py +0 -33
- package/mcps/vectorworks-mcp/app/mcp_server.py +0 -71
- package/mcps/vectorworks-mcp/app/search.py +0 -54
- package/mcps/vectorworks-mcp/app/utils.py +0 -14
- package/mcps/vectorworks-mcp/data/.gitkeep +0 -1
- package/mcps/vectorworks-mcp/data/vw-app-help/VW2022_PluginScript.htm +0 -95
- package/mcps/vectorworks-mcp/data/vw-app-help/VW2022_ScriptEditor.htm +0 -266
- package/mcps/vectorworks-mcp/data/vw-app-help/VW2022_Scripting.htm +0 -106
- package/mcps/vectorworks-mcp/data/vw-app-help/VW2023_CreatingScriptedPlugins.htm +0 -327
- package/mcps/vectorworks-mcp/data/vw-app-help/VW2024_RunningScripts.htm +0 -116
- package/mcps/vectorworks-mcp/data/vw-devwiki/VS_Function_Reference_category.html +0 -2129
- package/mcps/vectorworks-mcp/data/vw-jp-ref/ObjectsSymbols.html +0 -2035
- package/mcps/vectorworks-mcp/data/vw-jp-ref/ScriptFunctionReference.html +0 -29
- package/mcps/vectorworks-mcp/design.md +0 -57
- package/mcps/vectorworks-mcp/docker-compose.yml +0 -18
- package/mcps/vectorworks-mcp/index/.gitkeep +0 -1
- package/mcps/vectorworks-mcp/index/meta.jsonl +0 -3956
- package/mcps/vectorworks-mcp/index/vw.faiss +0 -0
- package/mcps/vectorworks-mcp/requirements.txt +0 -10
- package/mcps/vectorworks-mcp/scripts/fetch_docs_minimal.sh +0 -84
- package/mcps/vectorworks-mcp/scripts/fetch_github_vectorworks.sh +0 -85
- package/mcps/vectorworks-mcp/templates/index.html +0 -40
- package/scripts/ecosystem-health-audit.js +0 -257
- package/skills/bdb-dev-os-skill/SKILL.md +0 -49
- package/skills/bdb-dev-os-skill/package.json +0 -11
- package/skills/bdbsaastraining/SKILL.md +0 -316
- package/skills/bdbsaastraining/references/exam-pool.md +0 -313
- package/skills/bdbsaastraining/references/interaction-modes.md +0 -94
- package/skills/bdbsaastraining/references/track-a-ai-agent.md +0 -111
- package/skills/bdbsaastraining/references/track-b-wordpress.md +0 -111
- package/skills/bdbsaastraining/references/track-c-mailserver.md +0 -107
- package/skills/bdbsaastraining/references/track-d-debian.md +0 -109
- package/skills/bdbsaastraining/references/track-e-custom.md +0 -83
- package/skills/bdbsaastraining/scripts/build_profile.py +0 -529
- package/skills/bdbsaastraining/scripts/generate_certificate.py +0 -161
- package/skills/bdbsaastraining/scripts/preflight_check.sh +0 -150
- package/skills/bdbsaastraining/templates/certificate_template.html +0 -508
- package/skills/global_config/bdb-ecosystem-health/SKILL.md +0 -68
- package/skills/global_config/bdbsaas-ops/SKILL.md +0 -16
- package/skills/global_config/bdbsaashost/SKILL.md +0 -125
|
@@ -1,922 +0,0 @@
|
|
|
1
|
-
"""Embeddings + similarity search (Phase C of the analysis program).
|
|
2
|
-
|
|
3
|
-
Text vectors over clip/shot summaries and transcript segments, CLIP image
|
|
4
|
-
vectors over sampled frames, CLAP audio vectors over per-shot audio windows.
|
|
5
|
-
Backends are detected, never installed (the whisper pattern):
|
|
6
|
-
|
|
7
|
-
- text: ollama serving ``nomic-embed-text`` (preferred — works on Apple
|
|
8
|
-
Silicon via Metal), or ``sentence_transformers`` when installed.
|
|
9
|
-
- visual: ``open_clip_torch`` (ViT-B-32) when installed alongside torch.
|
|
10
|
-
- audio: ``transformers`` ClapModel (laion/clap-htsat-unfused; preferred) or
|
|
11
|
-
``laion_clap`` when installed alongside torch; needs ffmpeg. Audio windows
|
|
12
|
-
are piped from the source media as raw PCM — read-only, no temp files.
|
|
13
|
-
|
|
14
|
-
Vectors live in the per-project DB (schema v10 ``embeddings`` table) as
|
|
15
|
-
float32 BLOBs, one row per (entity, kind, model). Similarity is brute-force
|
|
16
|
-
cosine — numpy when present, pure Python otherwise; this is thousands of
|
|
17
|
-
vectors, not millions. Embedding is local compute, so nothing here touches
|
|
18
|
-
the vision-token caps ledger; responses report wall-clock + counts instead.
|
|
19
|
-
"""
|
|
20
|
-
|
|
21
|
-
from __future__ import annotations
|
|
22
|
-
|
|
23
|
-
import array
|
|
24
|
-
import hashlib
|
|
25
|
-
import importlib.util
|
|
26
|
-
import json
|
|
27
|
-
import math
|
|
28
|
-
import os
|
|
29
|
-
import shutil
|
|
30
|
-
import sqlite3
|
|
31
|
-
import time
|
|
32
|
-
import urllib.error
|
|
33
|
-
import urllib.request
|
|
34
|
-
from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
|
|
35
|
-
|
|
36
|
-
from src.utils import timeline_brain_db
|
|
37
|
-
|
|
38
|
-
OLLAMA_URL = os.environ.get("DAVINCI_RESOLVE_MCP_OLLAMA_URL", "http://127.0.0.1:11434")
|
|
39
|
-
OLLAMA_TEXT_MODEL = os.environ.get("DAVINCI_RESOLVE_MCP_EMBED_MODEL", "nomic-embed-text")
|
|
40
|
-
SENTENCE_TRANSFORMERS_MODEL = "all-MiniLM-L6-v2"
|
|
41
|
-
OPEN_CLIP_MODEL = ("ViT-B-32", "laion2b_s34b_b79k")
|
|
42
|
-
|
|
43
|
-
CLAP_HF_MODEL = "laion/clap-htsat-unfused"
|
|
44
|
-
CLAP_SAMPLE_RATE = 48000
|
|
45
|
-
AUDIO_WINDOW_SECONDS = 10.0 # CLAP works on ~10s windows; longer shots are center-cropped
|
|
46
|
-
|
|
47
|
-
_PROBE_TIMEOUT_SECONDS = 2.0
|
|
48
|
-
_EMBED_TIMEOUT_SECONDS = 120.0
|
|
49
|
-
|
|
50
|
-
# Lazy singletons for heavyweight local models.
|
|
51
|
-
_ST_MODEL = None
|
|
52
|
-
_CLIP_STATE: Optional[Tuple[Any, Any, Any]] = None # (model, preprocess, torch)
|
|
53
|
-
_CLAP_STATE: Optional[Tuple[str, Any]] = None # (backend, state)
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
def _now() -> str:
|
|
57
|
-
return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
def _content_hash(value: str) -> str:
|
|
61
|
-
return hashlib.sha256(value.encode("utf-8")).hexdigest()[:16]
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
def pack_vector(vector: Sequence[float]) -> bytes:
|
|
65
|
-
return array.array("f", vector).tobytes()
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
def unpack_vector(blob: bytes) -> List[float]:
|
|
69
|
-
arr = array.array("f")
|
|
70
|
-
arr.frombytes(blob)
|
|
71
|
-
return list(arr)
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
def cosine_similarity(a: Sequence[float], b: Sequence[float]) -> float:
|
|
75
|
-
try:
|
|
76
|
-
import numpy as np
|
|
77
|
-
|
|
78
|
-
va, vb = np.asarray(a, dtype="float32"), np.asarray(b, dtype="float32")
|
|
79
|
-
denom = float(np.linalg.norm(va) * np.linalg.norm(vb))
|
|
80
|
-
if denom == 0.0:
|
|
81
|
-
return 0.0
|
|
82
|
-
return float(np.dot(va, vb) / denom)
|
|
83
|
-
except ImportError:
|
|
84
|
-
dot = sum(x * y for x, y in zip(a, b))
|
|
85
|
-
na = math.sqrt(sum(x * x for x in a))
|
|
86
|
-
nb = math.sqrt(sum(y * y for y in b))
|
|
87
|
-
if na == 0.0 or nb == 0.0:
|
|
88
|
-
return 0.0
|
|
89
|
-
return dot / (na * nb)
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
# ── backend detection (no installs, ever) ────────────────────────────────────
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
def _ollama_state() -> Dict[str, Any]:
|
|
96
|
-
binary = shutil.which("ollama")
|
|
97
|
-
state: Dict[str, Any] = {
|
|
98
|
-
"binary": binary,
|
|
99
|
-
"serving": False,
|
|
100
|
-
"model_present": False,
|
|
101
|
-
"model": OLLAMA_TEXT_MODEL,
|
|
102
|
-
}
|
|
103
|
-
if not binary:
|
|
104
|
-
return state
|
|
105
|
-
try:
|
|
106
|
-
with urllib.request.urlopen(f"{OLLAMA_URL}/api/tags", timeout=_PROBE_TIMEOUT_SECONDS) as resp:
|
|
107
|
-
payload = json.load(resp)
|
|
108
|
-
state["serving"] = True
|
|
109
|
-
models = [str(m.get("name") or "") for m in payload.get("models") or []]
|
|
110
|
-
state["model_present"] = any(
|
|
111
|
-
name == OLLAMA_TEXT_MODEL or name.startswith(f"{OLLAMA_TEXT_MODEL}:")
|
|
112
|
-
for name in models
|
|
113
|
-
)
|
|
114
|
-
except (urllib.error.URLError, OSError, ValueError, TimeoutError):
|
|
115
|
-
pass
|
|
116
|
-
return state
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
def detect_embedding_capabilities() -> Dict[str, Any]:
|
|
120
|
-
"""Availability of text/visual embedding backends, with install guidance."""
|
|
121
|
-
ollama = _ollama_state()
|
|
122
|
-
st_available = importlib.util.find_spec("sentence_transformers") is not None
|
|
123
|
-
torch_available = importlib.util.find_spec("torch") is not None
|
|
124
|
-
open_clip_available = importlib.util.find_spec("open_clip") is not None
|
|
125
|
-
|
|
126
|
-
text_backends = []
|
|
127
|
-
if ollama["binary"] and ollama["serving"] and ollama["model_present"]:
|
|
128
|
-
text_backends.append("ollama")
|
|
129
|
-
if st_available:
|
|
130
|
-
text_backends.append("sentence_transformers")
|
|
131
|
-
|
|
132
|
-
guidance: Dict[str, str] = {}
|
|
133
|
-
if not text_backends:
|
|
134
|
-
if ollama["binary"] and ollama["serving"] and not ollama["model_present"]:
|
|
135
|
-
guidance["text"] = f"Ask the user before running: ollama pull {OLLAMA_TEXT_MODEL}"
|
|
136
|
-
elif ollama["binary"] and not ollama["serving"]:
|
|
137
|
-
guidance["text"] = f"Start ollama (ollama serve), then: ollama pull {OLLAMA_TEXT_MODEL}"
|
|
138
|
-
else:
|
|
139
|
-
guidance["text"] = (
|
|
140
|
-
f"Install ollama (https://ollama.com) and pull {OLLAMA_TEXT_MODEL}, "
|
|
141
|
-
"or pip install sentence-transformers."
|
|
142
|
-
)
|
|
143
|
-
visual_available = torch_available and open_clip_available
|
|
144
|
-
if not visual_available:
|
|
145
|
-
guidance["visual"] = (
|
|
146
|
-
"pip install open_clip_torch"
|
|
147
|
-
if torch_available
|
|
148
|
-
else "pip install torch open_clip_torch"
|
|
149
|
-
)
|
|
150
|
-
|
|
151
|
-
transformers_available = importlib.util.find_spec("transformers") is not None
|
|
152
|
-
laion_clap_available = importlib.util.find_spec("laion_clap") is not None
|
|
153
|
-
ffmpeg_available = bool(shutil.which("ffmpeg"))
|
|
154
|
-
audio_backends: List[str] = []
|
|
155
|
-
if torch_available and ffmpeg_available:
|
|
156
|
-
if transformers_available:
|
|
157
|
-
audio_backends.append("transformers_clap")
|
|
158
|
-
if laion_clap_available:
|
|
159
|
-
audio_backends.append("laion_clap")
|
|
160
|
-
if not audio_backends:
|
|
161
|
-
if not ffmpeg_available:
|
|
162
|
-
guidance["audio"] = "Install ffmpeg (audio windows are extracted with it)."
|
|
163
|
-
else:
|
|
164
|
-
guidance["audio"] = (
|
|
165
|
-
"pip install transformers (CLAP via laion/clap-htsat-unfused), "
|
|
166
|
-
"or pip install laion_clap"
|
|
167
|
-
if torch_available
|
|
168
|
-
else "pip install torch transformers"
|
|
169
|
-
)
|
|
170
|
-
|
|
171
|
-
return {
|
|
172
|
-
"success": True,
|
|
173
|
-
"no_auto_install": True,
|
|
174
|
-
"text": {
|
|
175
|
-
"available": bool(text_backends),
|
|
176
|
-
"backends": text_backends,
|
|
177
|
-
"ollama": ollama,
|
|
178
|
-
"model": OLLAMA_TEXT_MODEL if "ollama" in text_backends else (
|
|
179
|
-
SENTENCE_TRANSFORMERS_MODEL if st_available else None
|
|
180
|
-
),
|
|
181
|
-
},
|
|
182
|
-
"visual": {
|
|
183
|
-
"available": visual_available,
|
|
184
|
-
"backends": ["open_clip"] if visual_available else [],
|
|
185
|
-
"model": "-".join(OPEN_CLIP_MODEL) if visual_available else None,
|
|
186
|
-
},
|
|
187
|
-
"audio": {
|
|
188
|
-
"available": bool(audio_backends),
|
|
189
|
-
"backends": audio_backends,
|
|
190
|
-
"model": CLAP_HF_MODEL if audio_backends else None,
|
|
191
|
-
"ffmpeg": ffmpeg_available,
|
|
192
|
-
},
|
|
193
|
-
"install_guidance": guidance,
|
|
194
|
-
}
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
# ── embedding calls ──────────────────────────────────────────────────────────
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
def _embed_texts_ollama(texts: List[str]) -> Tuple[List[List[float]], str]:
|
|
201
|
-
body = json.dumps({"model": OLLAMA_TEXT_MODEL, "input": texts}).encode("utf-8")
|
|
202
|
-
request = urllib.request.Request(
|
|
203
|
-
f"{OLLAMA_URL}/api/embed",
|
|
204
|
-
data=body,
|
|
205
|
-
headers={"Content-Type": "application/json"},
|
|
206
|
-
)
|
|
207
|
-
with urllib.request.urlopen(request, timeout=_EMBED_TIMEOUT_SECONDS) as resp:
|
|
208
|
-
payload = json.load(resp)
|
|
209
|
-
vectors = payload.get("embeddings")
|
|
210
|
-
if not isinstance(vectors, list) or len(vectors) != len(texts):
|
|
211
|
-
raise ValueError(f"ollama returned {len(vectors or [])} embeddings for {len(texts)} inputs")
|
|
212
|
-
return vectors, f"ollama:{OLLAMA_TEXT_MODEL}"
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
def _embed_texts_sentence_transformers(texts: List[str]) -> Tuple[List[List[float]], str]:
|
|
216
|
-
global _ST_MODEL
|
|
217
|
-
if _ST_MODEL is None:
|
|
218
|
-
from sentence_transformers import SentenceTransformer
|
|
219
|
-
|
|
220
|
-
_ST_MODEL = SentenceTransformer(SENTENCE_TRANSFORMERS_MODEL)
|
|
221
|
-
vectors = _ST_MODEL.encode(texts, convert_to_numpy=True)
|
|
222
|
-
return [list(map(float, v)) for v in vectors], f"sentence_transformers:{SENTENCE_TRANSFORMERS_MODEL}"
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
def embed_texts(texts: List[str], *, backend: Optional[str] = None) -> Dict[str, Any]:
|
|
226
|
-
"""Embed a batch of texts with the best available local backend."""
|
|
227
|
-
if not texts:
|
|
228
|
-
return {"success": True, "vectors": [], "model": None}
|
|
229
|
-
caps = detect_embedding_capabilities()
|
|
230
|
-
backends = caps["text"]["backends"]
|
|
231
|
-
chosen = backend or (backends[0] if backends else None)
|
|
232
|
-
if chosen not in backends:
|
|
233
|
-
return {
|
|
234
|
-
"success": False,
|
|
235
|
-
"error": "No text-embedding backend available",
|
|
236
|
-
"install_guidance": caps["install_guidance"].get("text"),
|
|
237
|
-
}
|
|
238
|
-
try:
|
|
239
|
-
if chosen == "ollama":
|
|
240
|
-
vectors, model = _embed_texts_ollama(texts)
|
|
241
|
-
else:
|
|
242
|
-
vectors, model = _embed_texts_sentence_transformers(texts)
|
|
243
|
-
except Exception as exc: # noqa: BLE001 — backend failures surface as data
|
|
244
|
-
return {"success": False, "error": f"{type(exc).__name__}: {exc}", "backend": chosen}
|
|
245
|
-
return {"success": True, "vectors": vectors, "model": model, "backend": chosen}
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
def _clip_model():
|
|
249
|
-
global _CLIP_STATE
|
|
250
|
-
if _CLIP_STATE is None:
|
|
251
|
-
import open_clip
|
|
252
|
-
import torch
|
|
253
|
-
|
|
254
|
-
model, _, preprocess = open_clip.create_model_and_transforms(
|
|
255
|
-
OPEN_CLIP_MODEL[0], pretrained=OPEN_CLIP_MODEL[1]
|
|
256
|
-
)
|
|
257
|
-
model.eval()
|
|
258
|
-
_CLIP_STATE = (model, preprocess, torch)
|
|
259
|
-
return _CLIP_STATE
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
def embed_images(paths: List[str]) -> Dict[str, Any]:
|
|
263
|
-
"""Embed image files with open_clip. Skips missing files."""
|
|
264
|
-
caps = detect_embedding_capabilities()
|
|
265
|
-
if not caps["visual"]["available"]:
|
|
266
|
-
return {
|
|
267
|
-
"success": False,
|
|
268
|
-
"error": "No visual-embedding backend available",
|
|
269
|
-
"install_guidance": caps["install_guidance"].get("visual"),
|
|
270
|
-
}
|
|
271
|
-
try:
|
|
272
|
-
from PIL import Image
|
|
273
|
-
|
|
274
|
-
model, preprocess, torch = _clip_model()
|
|
275
|
-
vectors: List[Optional[List[float]]] = []
|
|
276
|
-
with torch.no_grad():
|
|
277
|
-
for path in paths:
|
|
278
|
-
if not path or not os.path.isfile(path):
|
|
279
|
-
vectors.append(None)
|
|
280
|
-
continue
|
|
281
|
-
image = preprocess(Image.open(path).convert("RGB")).unsqueeze(0)
|
|
282
|
-
features = model.encode_image(image)
|
|
283
|
-
features = features / features.norm(dim=-1, keepdim=True)
|
|
284
|
-
vectors.append([float(x) for x in features[0].tolist()])
|
|
285
|
-
except Exception as exc: # noqa: BLE001
|
|
286
|
-
return {"success": False, "error": f"{type(exc).__name__}: {exc}"}
|
|
287
|
-
return {"success": True, "vectors": vectors, "model": f"open_clip:{'-'.join(OPEN_CLIP_MODEL)}"}
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
def embed_text_for_visual_query(text: str) -> Dict[str, Any]:
|
|
291
|
-
"""CLIP text encoder — lets a free-text query search visual vectors."""
|
|
292
|
-
caps = detect_embedding_capabilities()
|
|
293
|
-
if not caps["visual"]["available"]:
|
|
294
|
-
return {
|
|
295
|
-
"success": False,
|
|
296
|
-
"error": "No visual-embedding backend available",
|
|
297
|
-
"install_guidance": caps["install_guidance"].get("visual"),
|
|
298
|
-
}
|
|
299
|
-
try:
|
|
300
|
-
import open_clip
|
|
301
|
-
|
|
302
|
-
model, _preprocess, torch = _clip_model()
|
|
303
|
-
tokenizer = open_clip.get_tokenizer(OPEN_CLIP_MODEL[0])
|
|
304
|
-
with torch.no_grad():
|
|
305
|
-
features = model.encode_text(tokenizer([text]))
|
|
306
|
-
features = features / features.norm(dim=-1, keepdim=True)
|
|
307
|
-
return {
|
|
308
|
-
"success": True,
|
|
309
|
-
"vector": [float(x) for x in features[0].tolist()],
|
|
310
|
-
"model": f"open_clip:{'-'.join(OPEN_CLIP_MODEL)}",
|
|
311
|
-
}
|
|
312
|
-
except Exception as exc: # noqa: BLE001
|
|
313
|
-
return {"success": False, "error": f"{type(exc).__name__}: {exc}"}
|
|
314
|
-
|
|
315
|
-
|
|
316
|
-
# ── audio (CLAP) ─────────────────────────────────────────────────────────────
|
|
317
|
-
|
|
318
|
-
|
|
319
|
-
def _clap_state() -> Tuple[str, Any]:
|
|
320
|
-
"""Lazy-load the preferred CLAP backend: transformers (HF-cached, kept
|
|
321
|
-
current) over the laion_clap package."""
|
|
322
|
-
global _CLAP_STATE
|
|
323
|
-
if _CLAP_STATE is None:
|
|
324
|
-
caps = detect_embedding_capabilities()["audio"]
|
|
325
|
-
backends = caps["backends"]
|
|
326
|
-
if "transformers_clap" in backends:
|
|
327
|
-
from transformers import ClapModel, ClapProcessor
|
|
328
|
-
|
|
329
|
-
model = ClapModel.from_pretrained(CLAP_HF_MODEL)
|
|
330
|
-
model.eval()
|
|
331
|
-
processor = ClapProcessor.from_pretrained(CLAP_HF_MODEL)
|
|
332
|
-
_CLAP_STATE = ("transformers_clap", (model, processor))
|
|
333
|
-
elif "laion_clap" in backends:
|
|
334
|
-
import laion_clap
|
|
335
|
-
|
|
336
|
-
module = laion_clap.CLAP_Module(enable_fusion=False)
|
|
337
|
-
module.load_ckpt() # default pretrained checkpoint
|
|
338
|
-
_CLAP_STATE = ("laion_clap", module)
|
|
339
|
-
else:
|
|
340
|
-
raise RuntimeError("No audio-embedding backend available")
|
|
341
|
-
return _CLAP_STATE
|
|
342
|
-
|
|
343
|
-
|
|
344
|
-
def _extract_audio_window(file_path: str, start_seconds: float, duration_seconds: float):
|
|
345
|
-
"""Mono 48 kHz float32 samples piped straight from ffmpeg — read-only on
|
|
346
|
-
the source media, no temp files. Returns None when the window is empty
|
|
347
|
-
(e.g. video-only media)."""
|
|
348
|
-
import subprocess
|
|
349
|
-
|
|
350
|
-
import numpy as np
|
|
351
|
-
|
|
352
|
-
command = [
|
|
353
|
-
"ffmpeg", "-v", "error",
|
|
354
|
-
"-ss", f"{max(0.0, float(start_seconds)):.3f}",
|
|
355
|
-
"-t", f"{max(0.1, float(duration_seconds)):.3f}",
|
|
356
|
-
"-i", file_path,
|
|
357
|
-
"-vn", "-ac", "1", "-ar", str(CLAP_SAMPLE_RATE),
|
|
358
|
-
"-f", "f32le", "-",
|
|
359
|
-
]
|
|
360
|
-
try:
|
|
361
|
-
proc = subprocess.run(command, capture_output=True, timeout=120, check=False)
|
|
362
|
-
except (OSError, subprocess.TimeoutExpired):
|
|
363
|
-
return None
|
|
364
|
-
if proc.returncode != 0 or not proc.stdout:
|
|
365
|
-
return None
|
|
366
|
-
samples = np.frombuffer(proc.stdout, dtype=np.float32)
|
|
367
|
-
return samples if samples.size else None
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
def _clap_audio_vectors(waveforms: List[Any]) -> Tuple[List[List[float]], str]:
|
|
371
|
-
backend, state = _clap_state()
|
|
372
|
-
if backend == "transformers_clap":
|
|
373
|
-
import torch
|
|
374
|
-
|
|
375
|
-
model, processor = state
|
|
376
|
-
inputs = processor(audios=waveforms, sampling_rate=CLAP_SAMPLE_RATE, return_tensors="pt", padding=True)
|
|
377
|
-
with torch.no_grad():
|
|
378
|
-
features = model.get_audio_features(**inputs)
|
|
379
|
-
features = features / features.norm(dim=-1, keepdim=True)
|
|
380
|
-
return [[float(x) for x in row.tolist()] for row in features], f"transformers_clap:{CLAP_HF_MODEL}"
|
|
381
|
-
import numpy as np
|
|
382
|
-
|
|
383
|
-
module = state
|
|
384
|
-
vectors = module.get_audio_embedding_from_data(x=np.stack(waveforms), use_tensor=False)
|
|
385
|
-
return [[float(x) for x in row] for row in vectors], "laion_clap:default"
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
def embed_audio_windows(file_path: str, windows: List[Tuple[float, float]]) -> Dict[str, Any]:
|
|
389
|
-
"""Embed (start_seconds, end_seconds) windows of one media file with CLAP.
|
|
390
|
-
|
|
391
|
-
Windows longer than AUDIO_WINDOW_SECONDS are center-cropped. Vector slots
|
|
392
|
-
are None for windows with no decodable audio.
|
|
393
|
-
"""
|
|
394
|
-
caps = detect_embedding_capabilities()
|
|
395
|
-
if not caps["audio"]["available"]:
|
|
396
|
-
return {
|
|
397
|
-
"success": False,
|
|
398
|
-
"error": "No audio-embedding backend available",
|
|
399
|
-
"install_guidance": caps["install_guidance"].get("audio"),
|
|
400
|
-
}
|
|
401
|
-
if not os.path.isfile(file_path):
|
|
402
|
-
return {"success": False, "error": f"media not found: {file_path}"}
|
|
403
|
-
waveforms: List[Any] = []
|
|
404
|
-
slots: List[Optional[int]] = []
|
|
405
|
-
for start, end in windows:
|
|
406
|
-
duration = max(0.1, float(end) - float(start))
|
|
407
|
-
if duration > AUDIO_WINDOW_SECONDS:
|
|
408
|
-
start = float(start) + (duration - AUDIO_WINDOW_SECONDS) / 2.0
|
|
409
|
-
duration = AUDIO_WINDOW_SECONDS
|
|
410
|
-
samples = _extract_audio_window(file_path, float(start), duration)
|
|
411
|
-
if samples is None:
|
|
412
|
-
slots.append(None)
|
|
413
|
-
continue
|
|
414
|
-
slots.append(len(waveforms))
|
|
415
|
-
waveforms.append(samples)
|
|
416
|
-
if not waveforms:
|
|
417
|
-
return {"success": True, "vectors": [None] * len(windows), "model": None,
|
|
418
|
-
"note": "no decodable audio in any window"}
|
|
419
|
-
try:
|
|
420
|
-
vectors, model = _clap_audio_vectors(waveforms)
|
|
421
|
-
except Exception as exc: # noqa: BLE001 — backend failures surface as data
|
|
422
|
-
return {"success": False, "error": f"{type(exc).__name__}: {exc}"}
|
|
423
|
-
return {
|
|
424
|
-
"success": True,
|
|
425
|
-
"vectors": [vectors[slot] if slot is not None else None for slot in slots],
|
|
426
|
-
"model": model,
|
|
427
|
-
}
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
def embed_text_for_audio_query(text: str) -> Dict[str, Any]:
|
|
431
|
-
"""CLAP text encoder — lets a free-text query ('engine revving') search
|
|
432
|
-
audio vectors."""
|
|
433
|
-
caps = detect_embedding_capabilities()
|
|
434
|
-
if not caps["audio"]["available"]:
|
|
435
|
-
return {
|
|
436
|
-
"success": False,
|
|
437
|
-
"error": "No audio-embedding backend available",
|
|
438
|
-
"install_guidance": caps["install_guidance"].get("audio"),
|
|
439
|
-
}
|
|
440
|
-
try:
|
|
441
|
-
backend, state = _clap_state()
|
|
442
|
-
if backend == "transformers_clap":
|
|
443
|
-
import torch
|
|
444
|
-
|
|
445
|
-
model, processor = state
|
|
446
|
-
inputs = processor(text=[str(text)], return_tensors="pt", padding=True)
|
|
447
|
-
with torch.no_grad():
|
|
448
|
-
features = model.get_text_features(**inputs)
|
|
449
|
-
features = features / features.norm(dim=-1, keepdim=True)
|
|
450
|
-
return {"success": True, "vector": [float(x) for x in features[0].tolist()],
|
|
451
|
-
"model": f"transformers_clap:{CLAP_HF_MODEL}"}
|
|
452
|
-
vectors = state.get_text_embedding([str(text), ""], use_tensor=False)
|
|
453
|
-
return {"success": True, "vector": [float(x) for x in vectors[0]], "model": "laion_clap:default"}
|
|
454
|
-
except Exception as exc: # noqa: BLE001
|
|
455
|
-
return {"success": False, "error": f"{type(exc).__name__}: {exc}"}
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
# ── content builders ─────────────────────────────────────────────────────────
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
def _shot_embed_text(shot: Dict[str, Any]) -> str:
|
|
462
|
-
parts: List[str] = []
|
|
463
|
-
if shot.get("description"):
|
|
464
|
-
parts.append(str(shot["description"]))
|
|
465
|
-
extra = shot.get("extra_json")
|
|
466
|
-
if extra:
|
|
467
|
-
try:
|
|
468
|
-
groups = json.loads(extra)
|
|
469
|
-
except (TypeError, ValueError):
|
|
470
|
-
groups = {}
|
|
471
|
-
for group in ("visual", "content", "editorial", "cuttability", "production"):
|
|
472
|
-
block = groups.get(group)
|
|
473
|
-
if isinstance(block, dict):
|
|
474
|
-
for key, value in sorted(block.items()):
|
|
475
|
-
if value in (None, "", [], {}):
|
|
476
|
-
continue
|
|
477
|
-
parts.append(f"{key}: {json.dumps(value, sort_keys=True, default=str)}")
|
|
478
|
-
return "\n".join(parts).strip()
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
def _clip_embed_text(clip: Dict[str, Any], conn: sqlite3.Connection) -> str:
|
|
482
|
-
parts: List[str] = []
|
|
483
|
-
for key in ("clip_name", "summary"):
|
|
484
|
-
if clip.get(key):
|
|
485
|
-
parts.append(str(clip[key]))
|
|
486
|
-
row = conn.execute(
|
|
487
|
-
"""
|
|
488
|
-
SELECT value_json FROM subjective_fields
|
|
489
|
-
WHERE entity_type='clip' AND entity_uuid=? AND field_path='clip_summary'
|
|
490
|
-
AND superseded_at IS NULL
|
|
491
|
-
""",
|
|
492
|
-
(clip["clip_uuid"],),
|
|
493
|
-
).fetchone()
|
|
494
|
-
if row:
|
|
495
|
-
try:
|
|
496
|
-
parts.append(str(json.loads(row["value_json"])))
|
|
497
|
-
except (TypeError, ValueError):
|
|
498
|
-
pass
|
|
499
|
-
tags = conn.execute(
|
|
500
|
-
"""
|
|
501
|
-
SELECT value_json FROM subjective_fields
|
|
502
|
-
WHERE entity_type='clip' AND entity_uuid=? AND field_path='editing_notes.search_tags'
|
|
503
|
-
AND superseded_at IS NULL
|
|
504
|
-
""",
|
|
505
|
-
(clip["clip_uuid"],),
|
|
506
|
-
).fetchone()
|
|
507
|
-
if tags:
|
|
508
|
-
try:
|
|
509
|
-
parts.append("tags: " + ", ".join(map(str, json.loads(tags["value_json"]) or [])))
|
|
510
|
-
except (TypeError, ValueError):
|
|
511
|
-
pass
|
|
512
|
-
return "\n".join(parts).strip()
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
# ── build ────────────────────────────────────────────────────────────────────
|
|
516
|
-
|
|
517
|
-
|
|
518
|
-
def _existing_rows(conn: sqlite3.Connection, kind: str) -> Dict[Tuple[str, str], str]:
|
|
519
|
-
rows = conn.execute(
|
|
520
|
-
"SELECT entity_type, entity_uuid, content_hash FROM embeddings WHERE embedding_kind = ?",
|
|
521
|
-
(kind,),
|
|
522
|
-
).fetchall()
|
|
523
|
-
return {(str(r["entity_type"]), str(r["entity_uuid"])): str(r["content_hash"] or "") for r in rows}
|
|
524
|
-
|
|
525
|
-
|
|
526
|
-
def _store_vectors(
|
|
527
|
-
project_root: str,
|
|
528
|
-
kind: str,
|
|
529
|
-
model: str,
|
|
530
|
-
items: List[Tuple[str, str, str, List[float]]],
|
|
531
|
-
) -> int:
|
|
532
|
-
"""items: (entity_type, entity_uuid, content_hash, vector)."""
|
|
533
|
-
if not items:
|
|
534
|
-
return 0
|
|
535
|
-
now = _now()
|
|
536
|
-
with timeline_brain_db.transaction(project_root) as conn:
|
|
537
|
-
for entity_type, entity_uuid, content_hash, vector in items:
|
|
538
|
-
conn.execute(
|
|
539
|
-
"""
|
|
540
|
-
INSERT INTO embeddings
|
|
541
|
-
(entity_type, entity_uuid, embedding_kind, model_name,
|
|
542
|
-
dimension, vector, content_hash, computed_at)
|
|
543
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?)
|
|
544
|
-
ON CONFLICT(entity_type, entity_uuid, embedding_kind, model_name)
|
|
545
|
-
DO UPDATE SET vector=excluded.vector, dimension=excluded.dimension,
|
|
546
|
-
content_hash=excluded.content_hash, computed_at=excluded.computed_at
|
|
547
|
-
""",
|
|
548
|
-
(entity_type, entity_uuid, kind, model, len(vector), pack_vector(vector), content_hash, now),
|
|
549
|
-
)
|
|
550
|
-
return len(items)
|
|
551
|
-
|
|
552
|
-
|
|
553
|
-
def build_embeddings(
|
|
554
|
-
project_root: str,
|
|
555
|
-
*,
|
|
556
|
-
kinds: Sequence[str] = ("text",),
|
|
557
|
-
clip_ref: Any = None,
|
|
558
|
-
include_segments: bool = True,
|
|
559
|
-
max_frames_per_clip: int = 16,
|
|
560
|
-
) -> Dict[str, Any]:
|
|
561
|
-
"""Build/refresh embeddings for a project (or one clip). Idempotent:
|
|
562
|
-
entities whose content hash is unchanged are skipped."""
|
|
563
|
-
from src.utils import analysis_store
|
|
564
|
-
|
|
565
|
-
started = time.time()
|
|
566
|
-
conn = timeline_brain_db.connect(project_root)
|
|
567
|
-
clip_filter: Optional[str] = None
|
|
568
|
-
if clip_ref:
|
|
569
|
-
clip_filter = analysis_store.resolve_clip_uuid(conn, clip_ref)
|
|
570
|
-
if not clip_filter:
|
|
571
|
-
return {"success": False, "error": f"clip not found in DB: {clip_ref!r} (run db_ingest first)"}
|
|
572
|
-
|
|
573
|
-
result: Dict[str, Any] = {"success": True, "project_root": project_root, "kinds": list(kinds)}
|
|
574
|
-
where = " WHERE clip_uuid = ?" if clip_filter else ""
|
|
575
|
-
args: Tuple[Any, ...] = (clip_filter,) if clip_filter else ()
|
|
576
|
-
|
|
577
|
-
if "text" in kinds:
|
|
578
|
-
texts: List[str] = []
|
|
579
|
-
meta: List[Tuple[str, str, str]] = [] # (entity_type, uuid, hash)
|
|
580
|
-
existing = _existing_rows(conn, "text")
|
|
581
|
-
for clip in conn.execute(f"SELECT * FROM clips{where}", args).fetchall():
|
|
582
|
-
clip = dict(clip)
|
|
583
|
-
text = _clip_embed_text(clip, conn)
|
|
584
|
-
if text:
|
|
585
|
-
h = _content_hash(text)
|
|
586
|
-
if existing.get(("clip", clip["clip_uuid"])) != h:
|
|
587
|
-
texts.append(text)
|
|
588
|
-
meta.append(("clip", clip["clip_uuid"], h))
|
|
589
|
-
for shot in conn.execute(
|
|
590
|
-
f"SELECT * FROM shots{where} ORDER BY clip_uuid, shot_index", args
|
|
591
|
-
).fetchall():
|
|
592
|
-
shot = dict(shot)
|
|
593
|
-
text = _shot_embed_text(shot)
|
|
594
|
-
if text:
|
|
595
|
-
h = _content_hash(text)
|
|
596
|
-
if existing.get(("shot", shot["shot_uuid"])) != h:
|
|
597
|
-
texts.append(text)
|
|
598
|
-
meta.append(("shot", shot["shot_uuid"], h))
|
|
599
|
-
if include_segments:
|
|
600
|
-
for seg in conn.execute(
|
|
601
|
-
f"SELECT * FROM transcript_segments{where} ORDER BY clip_uuid, segment_index", args
|
|
602
|
-
).fetchall():
|
|
603
|
-
seg = dict(seg)
|
|
604
|
-
text = str(seg.get("text") or "").strip()
|
|
605
|
-
if len(text) < 8:
|
|
606
|
-
continue
|
|
607
|
-
uuid = f"{seg['clip_uuid']}:{seg['segment_index']}"
|
|
608
|
-
h = _content_hash(text)
|
|
609
|
-
if existing.get(("segment", uuid)) != h:
|
|
610
|
-
texts.append(text)
|
|
611
|
-
meta.append(("segment", uuid, h))
|
|
612
|
-
if texts:
|
|
613
|
-
embedded = embed_texts(texts)
|
|
614
|
-
if not embedded.get("success"):
|
|
615
|
-
result["text"] = embedded
|
|
616
|
-
result["success"] = False
|
|
617
|
-
return result
|
|
618
|
-
stored = _store_vectors(
|
|
619
|
-
project_root,
|
|
620
|
-
"text",
|
|
621
|
-
str(embedded["model"]),
|
|
622
|
-
[(m[0], m[1], m[2], v) for m, v in zip(meta, embedded["vectors"])],
|
|
623
|
-
)
|
|
624
|
-
result["text"] = {"success": True, "embedded": stored, "skipped_unchanged": None, "model": embedded["model"]}
|
|
625
|
-
else:
|
|
626
|
-
result["text"] = {"success": True, "embedded": 0, "note": "all up to date"}
|
|
627
|
-
|
|
628
|
-
if "visual" in kinds:
|
|
629
|
-
existing = _existing_rows(conn, "visual")
|
|
630
|
-
frame_rows = conn.execute(
|
|
631
|
-
f"SELECT * FROM frames{where} ORDER BY clip_uuid, frame_index", args
|
|
632
|
-
).fetchall()
|
|
633
|
-
by_clip: Dict[str, List[Dict[str, Any]]] = {}
|
|
634
|
-
for row in frame_rows:
|
|
635
|
-
by_clip.setdefault(str(row["clip_uuid"]), []).append(dict(row))
|
|
636
|
-
paths: List[str] = []
|
|
637
|
-
meta_v: List[Tuple[str, str, str]] = []
|
|
638
|
-
frame_shot: List[Optional[str]] = []
|
|
639
|
-
for clip_uuid, rows in by_clip.items():
|
|
640
|
-
on_disk = [r for r in rows if r.get("frame_path") and os.path.isfile(str(r["frame_path"]))]
|
|
641
|
-
if len(on_disk) > max_frames_per_clip:
|
|
642
|
-
step = len(on_disk) / float(max_frames_per_clip)
|
|
643
|
-
on_disk = [on_disk[int(i * step)] for i in range(max_frames_per_clip)]
|
|
644
|
-
for r in on_disk:
|
|
645
|
-
uuid = f"{clip_uuid}:{r['frame_index']}"
|
|
646
|
-
h = _content_hash(str(r["frame_path"]))
|
|
647
|
-
if existing.get(("frame", uuid)) == h:
|
|
648
|
-
continue
|
|
649
|
-
paths.append(str(r["frame_path"]))
|
|
650
|
-
meta_v.append(("frame", uuid, h))
|
|
651
|
-
frame_shot.append(str(r["shot_uuid"]) if r.get("shot_uuid") else None)
|
|
652
|
-
if paths:
|
|
653
|
-
embedded = embed_images(paths)
|
|
654
|
-
if not embedded.get("success"):
|
|
655
|
-
result["visual"] = embedded
|
|
656
|
-
result["success"] = False
|
|
657
|
-
return result
|
|
658
|
-
items = []
|
|
659
|
-
shot_acc: Dict[str, List[List[float]]] = {}
|
|
660
|
-
for m, shot_uuid, vector in zip(meta_v, frame_shot, embedded["vectors"]):
|
|
661
|
-
if vector is None:
|
|
662
|
-
continue
|
|
663
|
-
items.append((m[0], m[1], m[2], vector))
|
|
664
|
-
if shot_uuid:
|
|
665
|
-
shot_acc.setdefault(shot_uuid, []).append(vector)
|
|
666
|
-
# Per-shot visual vector = mean of its frames'.
|
|
667
|
-
for shot_uuid, vectors in shot_acc.items():
|
|
668
|
-
dim = len(vectors[0])
|
|
669
|
-
mean = [sum(v[i] for v in vectors) / len(vectors) for i in range(dim)]
|
|
670
|
-
items.append(("shot", shot_uuid, _content_hash(json.dumps(sorted(len(v) for v in vectors)) + shot_uuid), mean))
|
|
671
|
-
stored = _store_vectors(project_root, "visual", str(embedded["model"]), items)
|
|
672
|
-
result["visual"] = {"success": True, "embedded": stored, "model": embedded["model"]}
|
|
673
|
-
else:
|
|
674
|
-
result["visual"] = {"success": True, "embedded": 0, "note": "all up to date"}
|
|
675
|
-
|
|
676
|
-
if "audio" in kinds:
|
|
677
|
-
caps = detect_embedding_capabilities()
|
|
678
|
-
if not caps["audio"]["available"]:
|
|
679
|
-
result["audio"] = {
|
|
680
|
-
"success": False,
|
|
681
|
-
"skipped": True,
|
|
682
|
-
"error": "No audio-embedding backend available",
|
|
683
|
-
"install_guidance": caps["install_guidance"].get("audio"),
|
|
684
|
-
}
|
|
685
|
-
result["success"] = False
|
|
686
|
-
return result
|
|
687
|
-
existing = _existing_rows(conn, "audio")
|
|
688
|
-
embedded_count = 0
|
|
689
|
-
missing_media: List[str] = []
|
|
690
|
-
audio_model: Optional[str] = None
|
|
691
|
-
for clip in conn.execute(f"SELECT * FROM clips{where}", args).fetchall():
|
|
692
|
-
clip = dict(clip)
|
|
693
|
-
file_path = str(clip.get("file_path") or "")
|
|
694
|
-
if not file_path or not os.path.isfile(file_path):
|
|
695
|
-
missing_media.append(str(clip.get("clip_name") or clip["clip_uuid"]))
|
|
696
|
-
continue
|
|
697
|
-
shots = [dict(s) for s in conn.execute(
|
|
698
|
-
"SELECT shot_uuid, time_seconds_start, time_seconds_end FROM shots "
|
|
699
|
-
"WHERE clip_uuid = ? ORDER BY shot_index",
|
|
700
|
-
(clip["clip_uuid"],),
|
|
701
|
-
).fetchall()]
|
|
702
|
-
windows: List[Tuple[float, float]] = []
|
|
703
|
-
window_meta: List[Tuple[str, str, str]] = []
|
|
704
|
-
for shot in shots:
|
|
705
|
-
start, end = shot.get("time_seconds_start"), shot.get("time_seconds_end")
|
|
706
|
-
if start is None or end is None or float(end) - float(start) < 0.2:
|
|
707
|
-
continue
|
|
708
|
-
h = _content_hash(f"{file_path}|{start}|{end}")
|
|
709
|
-
if existing.get(("shot", str(shot["shot_uuid"]))) == h:
|
|
710
|
-
continue
|
|
711
|
-
windows.append((float(start), float(end)))
|
|
712
|
-
window_meta.append(("shot", str(shot["shot_uuid"]), h))
|
|
713
|
-
if not windows:
|
|
714
|
-
continue
|
|
715
|
-
embedded = embed_audio_windows(file_path, windows)
|
|
716
|
-
if not embedded.get("success"):
|
|
717
|
-
result["audio"] = embedded
|
|
718
|
-
result["success"] = False
|
|
719
|
-
return result
|
|
720
|
-
if not embedded.get("model"):
|
|
721
|
-
continue # no decodable audio in this clip (e.g. video-only)
|
|
722
|
-
audio_model = str(embedded["model"])
|
|
723
|
-
items: List[Tuple[str, str, str, List[float]]] = []
|
|
724
|
-
clip_vectors: List[List[float]] = []
|
|
725
|
-
for m, vector in zip(window_meta, embedded["vectors"]):
|
|
726
|
-
if vector is None:
|
|
727
|
-
continue
|
|
728
|
-
items.append((m[0], m[1], m[2], vector))
|
|
729
|
-
clip_vectors.append(vector)
|
|
730
|
-
# Clip-level audio vector = mean of its shot windows' (the
|
|
731
|
-
# per-shot-visual pattern).
|
|
732
|
-
if clip_vectors:
|
|
733
|
-
dim = len(clip_vectors[0])
|
|
734
|
-
mean = [sum(v[i] for v in clip_vectors) / len(clip_vectors) for i in range(dim)]
|
|
735
|
-
items.append((
|
|
736
|
-
"clip", str(clip["clip_uuid"]),
|
|
737
|
-
_content_hash(f"{file_path}|clip|{len(clip_vectors)}"), mean,
|
|
738
|
-
))
|
|
739
|
-
embedded_count += _store_vectors(project_root, "audio", audio_model, items)
|
|
740
|
-
result["audio"] = {
|
|
741
|
-
"success": True,
|
|
742
|
-
"embedded": embedded_count,
|
|
743
|
-
"model": audio_model,
|
|
744
|
-
**({"skipped_missing_media": missing_media} if missing_media else {}),
|
|
745
|
-
**({"note": "all up to date"} if not embedded_count and not missing_media else {}),
|
|
746
|
-
}
|
|
747
|
-
|
|
748
|
-
result["wall_clock_ms"] = int((time.time() - started) * 1000)
|
|
749
|
-
counts = conn.execute(
|
|
750
|
-
"SELECT embedding_kind, COUNT(*) AS n FROM embeddings GROUP BY embedding_kind"
|
|
751
|
-
).fetchall()
|
|
752
|
-
result["totals"] = {str(r["embedding_kind"]): int(r["n"]) for r in counts}
|
|
753
|
-
return result
|
|
754
|
-
|
|
755
|
-
|
|
756
|
-
# ── similarity search ────────────────────────────────────────────────────────
|
|
757
|
-
|
|
758
|
-
|
|
759
|
-
def _hydrate(conn: sqlite3.Connection, entity_type: str, entity_uuid: str) -> Dict[str, Any]:
|
|
760
|
-
if entity_type == "clip":
|
|
761
|
-
row = conn.execute(
|
|
762
|
-
"SELECT clip_uuid, clip_name, clip_dir, summary FROM clips WHERE clip_uuid = ?",
|
|
763
|
-
(entity_uuid,),
|
|
764
|
-
).fetchone()
|
|
765
|
-
return dict(row) if row else {}
|
|
766
|
-
if entity_type == "shot":
|
|
767
|
-
row = conn.execute(
|
|
768
|
-
"""
|
|
769
|
-
SELECT s.shot_uuid, s.clip_uuid, s.shot_index, s.time_seconds_start,
|
|
770
|
-
s.time_seconds_end, s.description, c.clip_name
|
|
771
|
-
FROM shots s LEFT JOIN clips c ON c.clip_uuid = s.clip_uuid
|
|
772
|
-
WHERE s.shot_uuid = ?
|
|
773
|
-
""",
|
|
774
|
-
(entity_uuid,),
|
|
775
|
-
).fetchone()
|
|
776
|
-
return dict(row) if row else {}
|
|
777
|
-
if entity_type == "segment":
|
|
778
|
-
clip_uuid, _, seg_index = entity_uuid.rpartition(":")
|
|
779
|
-
row = conn.execute(
|
|
780
|
-
"""
|
|
781
|
-
SELECT t.clip_uuid, t.segment_index, t.start_seconds, t.end_seconds,
|
|
782
|
-
t.text, c.clip_name
|
|
783
|
-
FROM transcript_segments t LEFT JOIN clips c ON c.clip_uuid = t.clip_uuid
|
|
784
|
-
WHERE t.clip_uuid = ? AND t.segment_index = ?
|
|
785
|
-
""",
|
|
786
|
-
(clip_uuid, int(seg_index) if seg_index.isdigit() else -1),
|
|
787
|
-
).fetchone()
|
|
788
|
-
return dict(row) if row else {}
|
|
789
|
-
if entity_type == "frame":
|
|
790
|
-
clip_uuid, _, frame_index = entity_uuid.rpartition(":")
|
|
791
|
-
row = conn.execute(
|
|
792
|
-
"""
|
|
793
|
-
SELECT f.clip_uuid, f.frame_index, f.time_seconds, f.frame_path,
|
|
794
|
-
f.shot_uuid, c.clip_name
|
|
795
|
-
FROM frames f LEFT JOIN clips c ON c.clip_uuid = f.clip_uuid
|
|
796
|
-
WHERE f.clip_uuid = ? AND f.frame_index = ?
|
|
797
|
-
""",
|
|
798
|
-
(clip_uuid, int(frame_index) if frame_index.lstrip("-").isdigit() else -1),
|
|
799
|
-
).fetchone()
|
|
800
|
-
return dict(row) if row else {}
|
|
801
|
-
return {}
|
|
802
|
-
|
|
803
|
-
|
|
804
|
-
def find_similar(
|
|
805
|
-
project_root: str,
|
|
806
|
-
*,
|
|
807
|
-
text: Optional[str] = None,
|
|
808
|
-
clip_ref: Any = None,
|
|
809
|
-
shot_index: Optional[int] = None,
|
|
810
|
-
shot_uuid: Optional[str] = None,
|
|
811
|
-
kind: str = "text",
|
|
812
|
-
entity_types: Optional[Sequence[str]] = None,
|
|
813
|
-
limit: int = 10,
|
|
814
|
-
) -> Dict[str, Any]:
|
|
815
|
-
"""Brute-force cosine search. Query by free text, a clip, or a shot."""
|
|
816
|
-
from src.utils import analysis_store
|
|
817
|
-
|
|
818
|
-
conn = timeline_brain_db.connect(project_root)
|
|
819
|
-
kind = (kind or "text").strip().lower()
|
|
820
|
-
if kind not in ("text", "visual", "audio"):
|
|
821
|
-
return {"success": False, "error": f"kind must be 'text', 'visual', or 'audio', got {kind!r}"}
|
|
822
|
-
|
|
823
|
-
exclude: Optional[Tuple[str, str]] = None
|
|
824
|
-
query_vector: Optional[List[float]] = None
|
|
825
|
-
query_model: Optional[str] = None
|
|
826
|
-
|
|
827
|
-
if text:
|
|
828
|
-
if kind == "text":
|
|
829
|
-
embedded = embed_texts([str(text)])
|
|
830
|
-
if not embedded.get("success"):
|
|
831
|
-
return embedded
|
|
832
|
-
query_vector = embedded["vectors"][0]
|
|
833
|
-
query_model = str(embedded["model"])
|
|
834
|
-
elif kind == "audio":
|
|
835
|
-
encoded = embed_text_for_audio_query(str(text))
|
|
836
|
-
if not encoded.get("success"):
|
|
837
|
-
return encoded
|
|
838
|
-
query_vector = encoded["vector"]
|
|
839
|
-
query_model = str(encoded["model"])
|
|
840
|
-
else:
|
|
841
|
-
encoded = embed_text_for_visual_query(str(text))
|
|
842
|
-
if not encoded.get("success"):
|
|
843
|
-
return encoded
|
|
844
|
-
query_vector = encoded["vector"]
|
|
845
|
-
query_model = str(encoded["model"])
|
|
846
|
-
else:
|
|
847
|
-
entity_type: Optional[str] = None
|
|
848
|
-
entity_uuid: Optional[str] = None
|
|
849
|
-
if shot_uuid:
|
|
850
|
-
entity_type, entity_uuid = "shot", str(shot_uuid)
|
|
851
|
-
elif clip_ref is not None and shot_index is not None:
|
|
852
|
-
clip_uuid = analysis_store.resolve_clip_uuid(conn, clip_ref)
|
|
853
|
-
if not clip_uuid:
|
|
854
|
-
return {"success": False, "error": f"clip not found in DB: {clip_ref!r}"}
|
|
855
|
-
row = conn.execute(
|
|
856
|
-
"SELECT shot_uuid FROM shots WHERE clip_uuid = ? AND shot_index = ?",
|
|
857
|
-
(clip_uuid, int(shot_index)),
|
|
858
|
-
).fetchone()
|
|
859
|
-
if not row:
|
|
860
|
-
return {"success": False, "error": f"shot_index {shot_index} not found"}
|
|
861
|
-
entity_type, entity_uuid = "shot", str(row["shot_uuid"])
|
|
862
|
-
elif clip_ref is not None:
|
|
863
|
-
clip_uuid = analysis_store.resolve_clip_uuid(conn, clip_ref)
|
|
864
|
-
if not clip_uuid:
|
|
865
|
-
return {"success": False, "error": f"clip not found in DB: {clip_ref!r}"}
|
|
866
|
-
entity_type, entity_uuid = "clip", clip_uuid
|
|
867
|
-
else:
|
|
868
|
-
return {"success": False, "error": "find_similar requires text, clip_id/clip_dir, or shot"}
|
|
869
|
-
row = conn.execute(
|
|
870
|
-
"""
|
|
871
|
-
SELECT vector, model_name FROM embeddings
|
|
872
|
-
WHERE entity_type = ? AND entity_uuid = ? AND embedding_kind = ?
|
|
873
|
-
""",
|
|
874
|
-
(entity_type, entity_uuid, kind),
|
|
875
|
-
).fetchone()
|
|
876
|
-
if not row:
|
|
877
|
-
return {
|
|
878
|
-
"success": False,
|
|
879
|
-
"error": (
|
|
880
|
-
f"No {kind} embedding for that {entity_type} yet — run "
|
|
881
|
-
"media_analysis(action='build_embeddings') first."
|
|
882
|
-
),
|
|
883
|
-
}
|
|
884
|
-
query_vector = unpack_vector(row["vector"])
|
|
885
|
-
query_model = str(row["model_name"])
|
|
886
|
-
exclude = (entity_type, entity_uuid)
|
|
887
|
-
|
|
888
|
-
where = "embedding_kind = ? AND model_name = ?"
|
|
889
|
-
args: List[Any] = [kind, query_model]
|
|
890
|
-
if entity_types:
|
|
891
|
-
placeholders = ",".join("?" for _ in entity_types)
|
|
892
|
-
where += f" AND entity_type IN ({placeholders})"
|
|
893
|
-
args.extend(entity_types)
|
|
894
|
-
rows = conn.execute(
|
|
895
|
-
f"SELECT entity_type, entity_uuid, vector FROM embeddings WHERE {where}", args
|
|
896
|
-
).fetchall()
|
|
897
|
-
|
|
898
|
-
scored: List[Tuple[float, str, str]] = []
|
|
899
|
-
for row in rows:
|
|
900
|
-
key = (str(row["entity_type"]), str(row["entity_uuid"]))
|
|
901
|
-
if exclude and key == exclude:
|
|
902
|
-
continue
|
|
903
|
-
score = cosine_similarity(query_vector, unpack_vector(row["vector"]))
|
|
904
|
-
scored.append((score, key[0], key[1]))
|
|
905
|
-
scored.sort(key=lambda item: item[0], reverse=True)
|
|
906
|
-
|
|
907
|
-
results = []
|
|
908
|
-
for score, entity_type, entity_uuid in scored[: max(1, int(limit))]:
|
|
909
|
-
entry = {
|
|
910
|
-
"score": round(score, 4),
|
|
911
|
-
"entity_type": entity_type,
|
|
912
|
-
"entity_uuid": entity_uuid,
|
|
913
|
-
}
|
|
914
|
-
entry.update(_hydrate(conn, entity_type, entity_uuid))
|
|
915
|
-
results.append(entry)
|
|
916
|
-
return {
|
|
917
|
-
"success": True,
|
|
918
|
-
"kind": kind,
|
|
919
|
-
"model": query_model,
|
|
920
|
-
"candidates_scanned": len(rows),
|
|
921
|
-
"results": results,
|
|
922
|
-
}
|