fde-framework 0.1.3__tar.gz → 0.1.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fde_framework-0.1.3 → fde_framework-0.1.4}/CHANGELOG.md +27 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/PKG-INFO +3 -1
- {fde_framework-0.1.3 → fde_framework-0.1.4}/README.md +2 -0
- fde_framework-0.1.4/framework/approaches/graph-expanded-retrieval.md +24 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/graph-retrieval.md +6 -3
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/local-embedding.md +5 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/managed-embedding.md +4 -1
- fde_framework-0.1.4/framework/dimensions/corpus_churn.md +24 -0
- fde_framework-0.1.4/framework/patterns/graph-expanded-retrieval.md +15 -0
- fde_framework-0.1.4/framework/templates/retrieval/graph-expanded-retrieval.pgvector.py.j2 +117 -0
- fde_framework-0.1.4/framework/templates/retrieval/graph-expanded-retrieval.plain.py.j2 +65 -0
- fde_framework-0.1.4/framework/templates/retrieval/graph-expanded-retrieval.qdrant.py.j2 +77 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/pyproject.toml +1 -1
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/emit.py +147 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/ops.py +119 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_coverage.py +29 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_gates.py +3 -1
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_ops.py +24 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_prose.py +16 -0
- fde_framework-0.1.4/tests/test_retrieval_eval.py +137 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/.gitignore +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/LICENSE +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/examples/invoice-extraction/README.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/ansible-playbook.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/assisted-deterministic.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/audit-only.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/boundary-and-audit.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/cascade.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/classical-ml.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/compose.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/decision-log.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/deterministic-masking.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/deterministic.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/direct-call.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/episodic-store.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/explainability-record.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/field-match.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/finetune.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/fixed-sequence.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/gitops.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/governed-tools.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/hybrid-search.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/judged.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/keyword-search.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/kubernetes-manifests.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/labelled-metrics.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/llm-extraction.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/llm-scrubbing.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/llm.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/managed-api.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/manual-runbook.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/model-planner.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/ocr-pipeline.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/optimisation-reasoning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/optimisation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/passthrough.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/reranked-retrieval.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/role-scoped-authority.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/segmentation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/self-hosted.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/serverless-gpu.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/speech-transcription.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/structured-logs.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/systemd-unit.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/terraform-module.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/text-extraction.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/traced.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/vector-search.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/video-ingestion.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/windowed-ingestion.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/approaches/working-state.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/cases/churn-scoring.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/cases/route-planning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/cases/structured-extraction.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/cases/studio-style.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/accountability.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/deployment.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/embedding.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/evaluation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/governance.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/integration.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/memory.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/observability.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/perception.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/planning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/provisioning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/reasoning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/redaction.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/representation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/retrieval.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/components/serving.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/accelerator.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/access_model.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/arrival_rate.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/availability_target.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/cheap_path_coverage.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/confidence_calibrated.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/container_competence.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/corpus_size.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/data_residency.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/environment_lifetime.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/existing_cluster.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/existing_iac_tool.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/external_systems.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/hosting.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/human_waiting.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/input_format.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/interpretability_required.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/labelled_count.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/latency_budget_ms.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/licence_posture.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/operates_after_handover.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/output_shape.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/provisioning_api.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/query_pattern.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/recall_span.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/dimensions/sensitivity_present.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Generator.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Guard.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Mapper.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/ModelServer.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Parser.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Planner.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Retriever.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Scorer.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Store.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/ToolBoundary.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/interfaces/Tracer.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/locales/eu-gdpr.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/locales/in-dpdp.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/ansible-playbook.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/assisted-deterministic.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/audit-only.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/boundary-and-audit.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/cascade-reasoning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/cascade-representation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/cascade-retrieval.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/classical-ml-reasoning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/classical-ml.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/compose.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/decision-log.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/deterministic-masking.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/deterministic.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/direct-call.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/episodic-store.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/explainability-record.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/field-match.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/finetune-representation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/finetune.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/fixed-sequence.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/gitops.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/governed-tools.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/graph-retrieval.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/hybrid-search.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/judged.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/keyword-search.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/kubernetes-manifests.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/labelled-metrics.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/llm-representation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/llm-scrubbing.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/llm.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/local-embedding.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/managed-api.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/managed-embedding.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/manual-runbook.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/model-planner.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/ocr-pipeline.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/optimisation-reasoning.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/optimisation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/passthrough.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/reranked-retrieval.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/role-scoped-authority.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/segmentation.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/self-hosted.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/serverless-gpu.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/speech-transcription.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/structured-logs.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/systemd-unit.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/terraform-module.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/text-extraction.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/traced.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/vector-search.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/video-ingestion.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/windowed-ingestion.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/patterns/working-state.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/langgraph.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/llamaindex.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/local-judge.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/mcp.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/ollama.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/openai-judge.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/opentelemetry.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/ortools.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/pgvector.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/plain-python.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/qdrant.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/tesseract.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/vllm.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/whisper.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/stacks/xgboost.md +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/accountability/decision-log.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/accountability/explainability-record.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/deployment/compose.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/deployment/kubernetes-manifests.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/deployment/systemd-unit.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/embedding/local-embedding.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/embedding/managed-embedding.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/evaluation/field-match.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/evaluation/judged.local-judge.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/evaluation/judged.openai-judge.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/evaluation/judged.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/evaluation/labelled-metrics.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/evaluation/labelled-metrics.xgboost.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/governance/audit-only.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/governance/boundary-and-audit.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/governance/role-scoped-authority.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/integration/direct-call.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/integration/governed-tools.mcp.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/integration/governed-tools.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/memory/episodic-store.pgvector.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/memory/episodic-store.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/memory/working-state.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/observability/structured-logs.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/observability/traced.opentelemetry.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/observability/traced.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/ocr-pipeline.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/ocr-pipeline.tesseract.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/passthrough.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/speech-transcription.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/speech-transcription.whisper.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/text-extraction.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/video-ingestion.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/perception/windowed-ingestion.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/planning/fixed-sequence.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/planning/model-planner.langgraph.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/planning/model-planner.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/planning/optimisation.ortools.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/planning/optimisation.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/provisioning/ansible-playbook.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/provisioning/gitops.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/provisioning/manual-runbook.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/provisioning/terraform-module.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/reasoning/cascade.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/reasoning/classical-ml.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/reasoning/classical-ml.xgboost.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/reasoning/finetune.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/reasoning/llm.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/reasoning/optimisation.ortools.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/reasoning/optimisation.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/redaction/deterministic-masking.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/redaction/llm-scrubbing.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/assisted.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/cascade.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/classical-ml.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/classical-ml.xgboost.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/deterministic.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/finetune.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/llm.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/representation/segmentation.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/cascade.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/graph-retrieval.pgvector.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/graph-retrieval.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/graph-retrieval.qdrant.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/hybrid-search.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/keyword-search.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/reranked-retrieval.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/vector-search.llamaindex.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/vector-search.pgvector.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/vector-search.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/retrieval/vector-search.qdrant.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/serving/managed-api.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/serving/self-hosted.ollama.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/serving/self-hosted.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/serving/self-hosted.vllm.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/framework/templates/serving/serverless-gpu.plain.py.j2 +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/__init__.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/architect.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/cli.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/costing.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/decide.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/decompose.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/deploy.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/evolution.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/factlog.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/gates.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/graph.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/implement.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/intake/__init__.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/intake/answers.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/intake/documents.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/intake/interview.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/intake/llm_reader.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/intake/prose.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/intake/samples.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/models/__init__.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/models/base.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/models/fact.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/models/profile.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/models/respondent.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/models/schema.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/moves.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/predicate.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/realization.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/registry.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/scan.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/space.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/src/fde/workflow.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/__init__.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/conftest.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/models/__init__.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/models/test_profile.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_access_and_sensitivity.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_answers.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cascade.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cli_engagement.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cli_evolution.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cli_gates.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cli_intake.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cli_kb.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cli_robustness.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_cli_samples.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_costing.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_dead_zones.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_decide.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_decompose.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_deploy.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_documents.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_embedding.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_emit.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_evolution.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_example_transcript.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_factlog.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_implement.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_interview.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_llm_reader.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_locales.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_memory_retrieval.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_more_templates.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_moves.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_multimodal.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_package.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_realization.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_registry.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_remaining.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_reversibility.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_samples.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_sanitisation.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_scan.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_schema.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_serving.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_shipped_registry.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_space.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_stack_swap.py +0 -0
- {fde_framework-0.1.3 → fde_framework-0.1.4}/tests/test_templates.py +0 -0
|
@@ -5,6 +5,33 @@ the project is pre-release, so everything sits under 0.1.0 until the first tag.
|
|
|
5
5
|
|
|
6
6
|
## [Unreleased]
|
|
7
7
|
|
|
8
|
+
## [0.1.4] — 2026-09-09
|
|
9
|
+
|
|
10
|
+
The measurement release: claims the framework already made, turned into
|
|
11
|
+
numbers and walks.
|
|
12
|
+
|
|
13
|
+
- Projects with a retrieval component ship `evals/retrieval.py`: recall@10
|
|
14
|
+
and recall@50 against golden queries in `retrieval_cases.jsonl`, gated in
|
|
15
|
+
CI, refusing an empty case set. The embedding and index choices set a
|
|
16
|
+
ceiling nothing downstream recovers; this measures the ceiling by itself,
|
|
17
|
+
no model in the loop. The diagnosis walk's evidence step now ends at this
|
|
18
|
+
number, and both embedding approaches record the doctrine: chosen by
|
|
19
|
+
measured recall on the engagement's own queries, never by leaderboard.
|
|
20
|
+
- Emitted projects ship `ops/diagnosis.md`: the walk that finds which layer
|
|
21
|
+
a failure lives in -- definitions, evidence, tools, loop, model -- ordered
|
|
22
|
+
cheapest-to-check first, sections adapted to the components actually in
|
|
23
|
+
the system. An unclear definition can look like a model error; the
|
|
24
|
+
expensive habit is re-prompting before finding out.
|
|
25
|
+
- Retrieval corpus: `corpus_churn` dimension (static/periodic/continuous —
|
|
26
|
+
how fast the document stock turns over, distinct from corpus size and
|
|
27
|
+
query arrival) and a graph-expanded-retrieval approach: vector entry,
|
|
28
|
+
entity expansion, rerank exit, for multi-hop questions on a corpus that
|
|
29
|
+
keeps changing. Full graph-retrieval now steps aside on continuous churn,
|
|
30
|
+
by name, with the reason on the record. Realizations for plain-python,
|
|
31
|
+
pgvector and qdrant.
|
|
32
|
+
- README quickstart shows the gate refusal before the passing build, so the
|
|
33
|
+
first run's refusal is announced rather than a surprise.
|
|
34
|
+
|
|
8
35
|
## [0.1.3] — 2026-09-08
|
|
9
36
|
|
|
10
37
|
A post-launch funnel audit replayed every public transcript against the
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: fde-framework
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.4
|
|
4
4
|
Summary: A framework for Forward Deployed Engineers: from problem statement to a deployable AI project, every decision traced to a fact.
|
|
5
5
|
Project-URL: Homepage, https://github.com/atulkapoor/fde-framework
|
|
6
6
|
Project-URL: Repository, https://github.com/atulkapoor/fde-framework
|
|
@@ -76,6 +76,8 @@ pip install fde-framework
|
|
|
76
76
|
fde start acme --statement "Extract fields from supplier invoices."
|
|
77
77
|
fde ask acme --role admin # role-scoped discovery interview
|
|
78
78
|
fde architect acme # the design, with cited rationale
|
|
79
|
+
fde build acme --out project # refuses: seven gates guard the build
|
|
80
|
+
# ...verify data access, capture the baseline (each gate prints its remedy), then:
|
|
79
81
|
fde build acme --out project # code + evals + deploy assets + runbook
|
|
80
82
|
```
|
|
81
83
|
|
|
@@ -38,6 +38,8 @@ pip install fde-framework
|
|
|
38
38
|
fde start acme --statement "Extract fields from supplier invoices."
|
|
39
39
|
fde ask acme --role admin # role-scoped discovery interview
|
|
40
40
|
fde architect acme # the design, with cited rationale
|
|
41
|
+
fde build acme --out project # refuses: seven gates guard the build
|
|
42
|
+
# ...verify data access, capture the baseline (each gate prints its remedy), then:
|
|
41
43
|
fde build acme --out project # code + evals + deploy assets + runbook
|
|
42
44
|
```
|
|
43
45
|
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
---
|
|
2
|
+
id: graph-expanded-retrieval
|
|
3
|
+
name: Graph-expanded retrieval
|
|
4
|
+
complexity: 3
|
|
5
|
+
components: [retrieval]
|
|
6
|
+
applies_when: [query_pattern == multi_hop and corpus_churn == continuous]
|
|
7
|
+
avoid_when: [query_pattern == lookup]
|
|
8
|
+
evidence: {case_ids: [structured-extraction], confidence: low, last_verified: 2026-09-09}
|
|
9
|
+
---
|
|
10
|
+
Vector search finds the entry points, the entities they mention open the
|
|
11
|
+
graph, a bounded expansion collects what connects, and a reranker orders the
|
|
12
|
+
merged pool. Most of what a full graph buys on multi-hop questions -- without
|
|
13
|
+
asking the graph to carry recall.
|
|
14
|
+
|
|
15
|
+
That one reassignment is what survives churn. With vector search carrying
|
|
16
|
+
recall, the graph keeps a lighter contract -- entities and first-class edges,
|
|
17
|
+
rebuilt incrementally -- instead of the exhaustive index whose super-linear
|
|
18
|
+
growth and drifting entity resolution turn a moving corpus's graph
|
|
19
|
+
confidently wrong.
|
|
20
|
+
|
|
21
|
+
Where the corpus holds still, the full graph still wins: richer edges, deeper
|
|
22
|
+
traversal, no rerank pass to pay for. This exists for the corpus that will
|
|
23
|
+
not hold still. And like the reranker, it is adopted on measurement rather
|
|
24
|
+
than fashion: run the golden multi-hop queries and let the recall gap argue.
|
|
@@ -4,8 +4,8 @@ name: Graph retrieval
|
|
|
4
4
|
complexity: 3
|
|
5
5
|
components: [retrieval]
|
|
6
6
|
applies_when: [query_pattern == multi_hop]
|
|
7
|
-
avoid_when: [query_pattern == lookup, query_pattern == comparative]
|
|
8
|
-
evidence: {case_ids: [structured-extraction], confidence: medium, last_verified: 2026-
|
|
7
|
+
avoid_when: [query_pattern == lookup, query_pattern == comparative, corpus_churn == continuous]
|
|
8
|
+
evidence: {case_ids: [structured-extraction], confidence: medium, last_verified: 2026-09-09}
|
|
9
9
|
---
|
|
10
10
|
Explicit entities and edges, traversed at query time.
|
|
11
11
|
|
|
@@ -16,7 +16,10 @@ rather than an upgrade.
|
|
|
16
16
|
|
|
17
17
|
The costs are real: multi-pass extraction to build it, two to three times the
|
|
18
18
|
end-to-end latency to use it, and an index that grows super-linearly, which is
|
|
19
|
-
what makes incremental updates painful on a corpus that changes.
|
|
19
|
+
what makes incremental updates painful on a corpus that changes. Where the
|
|
20
|
+
corpus changes continuously, that maintenance is the deciding cost --
|
|
21
|
+
graph-expanded retrieval keeps the multi-hop benefit by letting vector search
|
|
22
|
+
carry recall, so the graph can stay small enough to rebuild.
|
|
20
23
|
|
|
21
24
|
Published gains also warrant scepticism -- judge position bias has been shown to
|
|
22
25
|
swing reported win rates by tens of points. Measure on your own traffic.
|
|
@@ -19,6 +19,11 @@ Ruled out where nobody is named to operate it. A model with no owner is a
|
|
|
19
19
|
liability handed over with a bow on it, and an embedding model whose host dies
|
|
20
20
|
takes the index with it.
|
|
21
21
|
|
|
22
|
+
Whichever model, it sets the ceiling on retrieval: nothing downstream
|
|
23
|
+
recovers a document that was never encoded well enough to surface. The
|
|
24
|
+
choice is made by measured recall on the engagement's own golden queries --
|
|
25
|
+
the emitted `evals/retrieval.py` is that measurement -- not by leaderboard.
|
|
26
|
+
|
|
22
27
|
The handover concern is stated precisely rather than broadly: with nobody
|
|
23
28
|
named to operate and data free to leave, the managed alternative is
|
|
24
29
|
strictly simpler and wins. But where data cannot leave, the plain
|
|
@@ -19,4 +19,7 @@ to their source, so this is not a way of anonymising anything on the way out.
|
|
|
19
19
|
|
|
20
20
|
One-way, like any vendor call. And expensive to reverse for a second reason --
|
|
21
21
|
changing embedding model means reindexing everything, so leaving is a
|
|
22
|
-
reindexing project rather than a configuration change.
|
|
22
|
+
reindexing project rather than a configuration change. Which is exactly why
|
|
23
|
+
the model is chosen by measured recall on the engagement's own golden
|
|
24
|
+
queries (the emitted `evals/retrieval.py`) rather than by leaderboard: the
|
|
25
|
+
reindexing bill for a wrong guess arrives after the index is full.
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
---
|
|
2
|
+
id: corpus_churn
|
|
3
|
+
type: enum
|
|
4
|
+
scope: data
|
|
5
|
+
kind: requirement
|
|
6
|
+
weight: 1.0
|
|
7
|
+
asks: "How often does the document corpus itself change?"
|
|
8
|
+
ask_role: [admin, user]
|
|
9
|
+
values: [static, periodic, continuous]
|
|
10
|
+
recognises:
|
|
11
|
+
static: [historical archive, frozen corpus, fixed set of documents, one-time snapshot, corpus is static, closed cases]
|
|
12
|
+
periodic: [refreshed monthly, monthly refresh, refreshed weekly, quarterly refresh, batch refresh, re-indexed every]
|
|
13
|
+
continuous: [updated continuously, constantly changing, changes daily, keeps changing, new documents arrive daily, updated throughout the day, live document feed]
|
|
14
|
+
---
|
|
15
|
+
`corpus_size` is stock and `arrival_rate` is the flow of questions; this is
|
|
16
|
+
how fast the stock itself turns over. Nobody volunteers it, because no demo
|
|
17
|
+
runs long enough to feel it.
|
|
18
|
+
|
|
19
|
+
Any index that is expensive to rebuild pays this rate forever, and the
|
|
20
|
+
expensive ones fail politely: retrieval keeps answering, fluently, from a
|
|
21
|
+
corpus that is no longer the client's. An entity graph pays the rate worst --
|
|
22
|
+
extraction is multi-pass, the index grows super-linearly, and entity
|
|
23
|
+
resolution drifts as names and structures change, which is how a graph goes
|
|
24
|
+
from wrong to confidently wrong without an error in any log.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
---
|
|
2
|
+
id: graph-expanded-retrieval
|
|
3
|
+
component: retrieval
|
|
4
|
+
approach: graph-expanded-retrieval
|
|
5
|
+
realizations:
|
|
6
|
+
- {stack: plain-python, template: retrieval/graph-expanded-retrieval.plain.py.j2, provides: Retriever}
|
|
7
|
+
- {stack: pgvector, template: retrieval/graph-expanded-retrieval.pgvector.py.j2, provides: Retriever}
|
|
8
|
+
- {stack: qdrant, template: retrieval/graph-expanded-retrieval.qdrant.py.j2, provides: Retriever}
|
|
9
|
+
evidence: {case_ids: [structured-extraction], confidence: low, last_verified: 2026-09-09}
|
|
10
|
+
---
|
|
11
|
+
Implements graph-expanded-retrieval for retrieval, satisfying Retriever.
|
|
12
|
+
|
|
13
|
+
Vector entry, entity expansion, rerank exit. The graph is deliberately the
|
|
14
|
+
junior partner: it contributes connections, never recall, which is what lets
|
|
15
|
+
it stay small enough to rebuild as the corpus moves.
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
"""{{ component }}: {{ approach }}, via {{ stack }}.
|
|
2
|
+
|
|
3
|
+
{{ rationale }}
|
|
4
|
+
|
|
5
|
+
Vector entry, entity expansion, rerank exit -- with both halves in the one
|
|
6
|
+
database already being operated. Chunks and their embeddings in one table,
|
|
7
|
+
entity mentions in another, and the expansion is a self-join on mentions
|
|
8
|
+
rather than a graph engine: for co-mention edges at bounded depth, Postgres
|
|
9
|
+
does this well enough that a separate system has to be argued for.
|
|
10
|
+
|
|
11
|
+
The mentions table is the whole graph, which is the point. Rebuilding it for
|
|
12
|
+
a changed document is a delete and an insert, not a multi-pass re-extraction
|
|
13
|
+
-- the property that lets this survive a corpus that keeps changing.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
from collections.abc import Callable
|
|
19
|
+
from typing import Any
|
|
20
|
+
|
|
21
|
+
interface = "{{ interface }}"
|
|
22
|
+
approach = "{{ approach }}"
|
|
23
|
+
stack = "{{ stack }}"
|
|
24
|
+
|
|
25
|
+
SCHEMA_SQL = """
|
|
26
|
+
CREATE EXTENSION IF NOT EXISTS vector;
|
|
27
|
+
CREATE TABLE IF NOT EXISTS {table} (
|
|
28
|
+
id text PRIMARY KEY,
|
|
29
|
+
body text NOT NULL,
|
|
30
|
+
embedding vector({dimensions})
|
|
31
|
+
);
|
|
32
|
+
CREATE TABLE IF NOT EXISTS {table}_mentions (
|
|
33
|
+
entity text NOT NULL,
|
|
34
|
+
chunk_id text REFERENCES {table}(id) ON DELETE CASCADE,
|
|
35
|
+
PRIMARY KEY (entity, chunk_id)
|
|
36
|
+
);
|
|
37
|
+
CREATE INDEX IF NOT EXISTS {table}_mentions_entity_idx
|
|
38
|
+
ON {table}_mentions (entity);
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
SEED_SQL = """
|
|
42
|
+
SELECT id, body FROM {table}
|
|
43
|
+
ORDER BY embedding <=> %(query)s::vector
|
|
44
|
+
LIMIT %(k)s;
|
|
45
|
+
"""
|
|
46
|
+
|
|
47
|
+
# One hop per query, looped in the application: co-mentioned chunks of the
|
|
48
|
+
# frontier, minus everything already found. Bounded by construction.
|
|
49
|
+
EXPAND_SQL = """
|
|
50
|
+
SELECT DISTINCT m2.chunk_id, c.body
|
|
51
|
+
FROM {table}_mentions m1
|
|
52
|
+
JOIN {table}_mentions m2 ON m2.entity = m1.entity
|
|
53
|
+
JOIN {table} c ON c.id = m2.chunk_id
|
|
54
|
+
WHERE m1.chunk_id = ANY(%(frontier)s)
|
|
55
|
+
AND NOT m2.chunk_id = ANY(%(found)s);
|
|
56
|
+
"""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
class {{ class_name }}:
|
|
60
|
+
"""{{ interface }}, as {{ approach }}."""
|
|
61
|
+
|
|
62
|
+
interface = "{{ interface }}"
|
|
63
|
+
approach = "{{ approach }}"
|
|
64
|
+
stack = "{{ stack }}"
|
|
65
|
+
|
|
66
|
+
def __init__(
|
|
67
|
+
self,
|
|
68
|
+
connection: Any = None,
|
|
69
|
+
embed: Callable[[str], list[float]] | None = None,
|
|
70
|
+
rerank: Callable[[str, str], float] | None = None,
|
|
71
|
+
table: str = "documents",
|
|
72
|
+
dimensions: int = 768,
|
|
73
|
+
max_hops: int = 2,
|
|
74
|
+
seeds: int = 20,
|
|
75
|
+
) -> None:
|
|
76
|
+
self.connection = connection
|
|
77
|
+
self.embed = embed
|
|
78
|
+
self.rerank = rerank
|
|
79
|
+
self.table = table
|
|
80
|
+
self.dimensions = dimensions
|
|
81
|
+
self.max_hops = max_hops
|
|
82
|
+
self.seeds = seeds
|
|
83
|
+
|
|
84
|
+
def schema(self) -> str:
|
|
85
|
+
return SCHEMA_SQL.format(table=self.table, dimensions=self.dimensions)
|
|
86
|
+
|
|
87
|
+
def retrieve(self, query: str, top_k: int = 8) -> list[dict[str, Any]]:
|
|
88
|
+
found: dict[str, dict[str, Any]] = {}
|
|
89
|
+
with self.connection.cursor() as cursor:
|
|
90
|
+
cursor.execute(
|
|
91
|
+
SEED_SQL.format(table=self.table),
|
|
92
|
+
{"query": self.embed(query), "k": self.seeds},
|
|
93
|
+
)
|
|
94
|
+
for row in cursor.fetchall():
|
|
95
|
+
found[row[0]] = {"id": row[0], "text": row[1], "hop": 0}
|
|
96
|
+
frontier = list(found)
|
|
97
|
+
for hop in range(1, self.max_hops + 1):
|
|
98
|
+
if not frontier:
|
|
99
|
+
break
|
|
100
|
+
cursor.execute(
|
|
101
|
+
EXPAND_SQL.format(table=self.table),
|
|
102
|
+
{"frontier": frontier, "found": list(found)},
|
|
103
|
+
)
|
|
104
|
+
fresh = cursor.fetchall()
|
|
105
|
+
if not fresh:
|
|
106
|
+
break
|
|
107
|
+
for row in fresh:
|
|
108
|
+
found[row[0]] = {"id": row[0], "text": row[1], "hop": hop}
|
|
109
|
+
frontier = [row[0] for row in fresh]
|
|
110
|
+
scored = [
|
|
111
|
+
{**c, "rerank_score": self.rerank(query, c["text"])}
|
|
112
|
+
for c in found.values()
|
|
113
|
+
]
|
|
114
|
+
return sorted(scored, key=lambda c: -c["rerank_score"])[:top_k]
|
|
115
|
+
|
|
116
|
+
def run(self, payload: dict[str, Any]) -> Any:
|
|
117
|
+
return self.retrieve(payload["query"], top_k=payload.get("k", 8))
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""{{ component }}: {{ approach }}, via {{ stack }}.
|
|
2
|
+
|
|
3
|
+
{{ rationale }}
|
|
4
|
+
|
|
5
|
+
Vector entry, entity expansion, rerank exit. Vector search carries recall,
|
|
6
|
+
the adjacency map contributes connections, and the reranker earns the final
|
|
7
|
+
order. The graph stays a junior partner on purpose -- entities and co-mention
|
|
8
|
+
edges only, cheap to rebuild incrementally -- because on a corpus that keeps
|
|
9
|
+
changing, the exhaustive graph is the part that rots first.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
from collections.abc import Callable
|
|
15
|
+
|
|
16
|
+
from app.contract import RefusedInput
|
|
17
|
+
|
|
18
|
+
interface = "{{ interface }}"
|
|
19
|
+
approach = "{{ approach }}"
|
|
20
|
+
stack = "{{ stack }}"
|
|
21
|
+
|
|
22
|
+
TOP_K = 8
|
|
23
|
+
SEEDS = 20 # vector recall opens the graph; it does not need to finish the job
|
|
24
|
+
MAX_HOPS = 2 # each hop multiplies the pool, and two answers most questions
|
|
25
|
+
|
|
26
|
+
# Injected at wiring time: retrieve(query, k) -> seed chunks (each a dict with
|
|
27
|
+
# at least id, text, entities); rerank(query, text) -> relevance, higher is
|
|
28
|
+
# better.
|
|
29
|
+
retrieve: Callable | None = None
|
|
30
|
+
rerank: Callable | None = None
|
|
31
|
+
|
|
32
|
+
# entity -> chunk records that mention it, rebuilt incrementally at index
|
|
33
|
+
# time. Co-mention edges only: the point of this variant is a graph small
|
|
34
|
+
# enough to rebuild as the corpus moves.
|
|
35
|
+
adjacency: dict[str, list[dict]] = {}
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class {{ class_name }}:
|
|
39
|
+
def run(self, query: str, top_k: int = TOP_K) -> list[dict]:
|
|
40
|
+
if retrieve is None or rerank is None:
|
|
41
|
+
raise RefusedInput(
|
|
42
|
+
"wire retrieve(query, k) and rerank(query, text) -- an "
|
|
43
|
+
"expansion over unranked seeds would wear a precision it "
|
|
44
|
+
"does not have"
|
|
45
|
+
)
|
|
46
|
+
pool = {c["id"]: {**c, "hop": 0} for c in retrieve(query, SEEDS)}
|
|
47
|
+
frontier = list(pool)
|
|
48
|
+
for hop in range(1, MAX_HOPS + 1):
|
|
49
|
+
connected = [
|
|
50
|
+
chunk
|
|
51
|
+
for chunk_id in frontier
|
|
52
|
+
for entity in pool[chunk_id].get("entities", [])
|
|
53
|
+
for chunk in adjacency.get(entity, [])
|
|
54
|
+
]
|
|
55
|
+
fresh = {c["id"]: {**c, "hop": hop}
|
|
56
|
+
for c in connected if c["id"] not in pool}
|
|
57
|
+
if not fresh:
|
|
58
|
+
break
|
|
59
|
+
pool.update(fresh)
|
|
60
|
+
frontier = list(fresh)
|
|
61
|
+
scored = [
|
|
62
|
+
{**c, "rerank_score": rerank(query, c.get("text", ""))}
|
|
63
|
+
for c in pool.values()
|
|
64
|
+
]
|
|
65
|
+
return sorted(scored, key=lambda c: -c["rerank_score"])[:top_k]
|
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""{{ component }}: {{ approach }}, via {{ stack }}.
|
|
2
|
+
|
|
3
|
+
{{ rationale }}
|
|
4
|
+
|
|
5
|
+
Vector entry, entity expansion, rerank exit. Qdrant answers "what resembles
|
|
6
|
+
this text", the adjacency map answers "what connects to what", and a reranker
|
|
7
|
+
earns the final order over the merged pool -- expanded chunks arrive without
|
|
8
|
+
a similarity score, and ordering them by hop count would pretend distance is
|
|
9
|
+
relevance.
|
|
10
|
+
|
|
11
|
+
The graph stays entities and co-mention edges only, rebuilt incrementally,
|
|
12
|
+
because this variant exists for the corpus that keeps changing.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from collections.abc import Callable
|
|
18
|
+
from typing import Any
|
|
19
|
+
|
|
20
|
+
interface = "{{ interface }}"
|
|
21
|
+
approach = "{{ approach }}"
|
|
22
|
+
stack = "{{ stack }}"
|
|
23
|
+
|
|
24
|
+
COLLECTION = "{{ component }}_chunks"
|
|
25
|
+
TOP_K = 8
|
|
26
|
+
SEEDS = 20
|
|
27
|
+
MAX_HOPS = 2 # bounded: each hop multiplies latency, and two answers most questions
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class QdrantUnwired(RuntimeError):
|
|
31
|
+
"""No client is wired; retrieval refuses rather than returning nothing."""
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
client: Any = None
|
|
35
|
+
embed: Callable[[str], list[float]] | None = None
|
|
36
|
+
rerank: Callable[[str, str], float] | None = None
|
|
37
|
+
|
|
38
|
+
# entity -> connected chunk ids, rebuilt incrementally from co-mentions.
|
|
39
|
+
adjacency: dict[str, list[str]] = {}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def run(query: str, top_k: int = TOP_K) -> list[dict]:
|
|
43
|
+
"""Seed by similarity, expand along entities, let the reranker order."""
|
|
44
|
+
if client is None or embed is None or rerank is None:
|
|
45
|
+
raise QdrantUnwired(
|
|
46
|
+
"wire a QdrantClient, an embed(text) callable and a "
|
|
47
|
+
"rerank(query, text) callable before running -- an empty result "
|
|
48
|
+
"from an unwired store would read as 'no evidence'"
|
|
49
|
+
)
|
|
50
|
+
seeds = client.search(
|
|
51
|
+
collection_name=COLLECTION, query_vector=embed(query),
|
|
52
|
+
limit=SEEDS, with_payload=True,
|
|
53
|
+
)
|
|
54
|
+
found = {str(hit.id): {"id": str(hit.id), "hop": 0, **(hit.payload or {})}
|
|
55
|
+
for hit in seeds}
|
|
56
|
+
frontier = list(found)
|
|
57
|
+
for hop in range(1, MAX_HOPS + 1):
|
|
58
|
+
connected = [
|
|
59
|
+
chunk_id
|
|
60
|
+
for seed_id in frontier
|
|
61
|
+
for entity in found[seed_id].get("entities", [])
|
|
62
|
+
for chunk_id in adjacency.get(entity, [])
|
|
63
|
+
]
|
|
64
|
+
fresh = [c for c in dict.fromkeys(connected) if c not in found]
|
|
65
|
+
if not fresh:
|
|
66
|
+
break
|
|
67
|
+
records = client.retrieve(collection_name=COLLECTION, ids=fresh,
|
|
68
|
+
with_payload=True)
|
|
69
|
+
for record in records:
|
|
70
|
+
found[str(record.id)] = {"id": str(record.id), "hop": hop,
|
|
71
|
+
**(record.payload or {})}
|
|
72
|
+
frontier = fresh
|
|
73
|
+
scored = [
|
|
74
|
+
{**c, "rerank_score": rerank(query, c.get("text", ""))}
|
|
75
|
+
for c in found.values()
|
|
76
|
+
]
|
|
77
|
+
return sorted(scored, key=lambda c: -c["rerank_score"])[:top_k]
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
# Distribution name matches the repository and stays unique on any index;
|
|
3
3
|
# the import package and the CLI command remain the short `fde`.
|
|
4
4
|
name = "fde-framework"
|
|
5
|
-
version = "0.1.
|
|
5
|
+
version = "0.1.4"
|
|
6
6
|
description = "A framework for Forward Deployed Engineers: from problem statement to a deployable AI project, every decision traced to a fact."
|
|
7
7
|
readme = "README.md"
|
|
8
8
|
license = {text = "Apache-2.0"}
|
|
@@ -734,6 +734,15 @@ def _write_evals(architecture: Architecture, out: Path, pairs_path: Path | None)
|
|
|
734
734
|
_HARNESS.format(metrics=json.dumps(metrics), judged=judged)
|
|
735
735
|
)
|
|
736
736
|
|
|
737
|
+
if "retrieval" in architecture.decisions.decided():
|
|
738
|
+
# The retrieval layer measured alone: the embedding and index set a
|
|
739
|
+
# ceiling nothing downstream recovers, and an empty case file is a
|
|
740
|
+
# gap somebody can see rather than a measurement nobody took.
|
|
741
|
+
(evals / "retrieval.py").write_text(_RETRIEVAL_EVAL)
|
|
742
|
+
cases_path = evals / "retrieval_cases.jsonl"
|
|
743
|
+
if not cases_path.exists():
|
|
744
|
+
cases_path.write_text("")
|
|
745
|
+
|
|
737
746
|
golden_count = len(suite.golden) if suite else 0
|
|
738
747
|
(evals / "acceptance.md").write_text(_acceptance(architecture, golden_count))
|
|
739
748
|
|
|
@@ -864,6 +873,144 @@ def classify(expected, actual, context=None):
|
|
|
864
873
|
'''
|
|
865
874
|
|
|
866
875
|
|
|
876
|
+
_RETRIEVAL_EVAL = '''#!/usr/bin/env python3
|
|
877
|
+
"""Measure the retrieval layer alone. Exits non-zero on an empty case set,
|
|
878
|
+
an erroring retriever, or recall below the threshold -- so CI can gate on it.
|
|
879
|
+
|
|
880
|
+
The embedding and index choices set a ceiling on everything downstream: no
|
|
881
|
+
reranking, prompting or model upgrade recovers a document that was never
|
|
882
|
+
retrieved. End-to-end scores blur that ceiling into "quality"; this number is
|
|
883
|
+
the ceiling by itself. When a failing case's answer is missing from the
|
|
884
|
+
evidence, this -- not the prompt -- is the layer to fix (ops/diagnosis.md
|
|
885
|
+
walks the order).
|
|
886
|
+
|
|
887
|
+
Cases live in retrieval_cases.jsonl beside this file, one JSON object per
|
|
888
|
+
line:
|
|
889
|
+
|
|
890
|
+
{"id": "case-1", "query": "...", "relevant": ["doc-7", "doc-12"]}
|
|
891
|
+
|
|
892
|
+
`relevant` lists the document ids a correct top-K must surface; recall@K is
|
|
893
|
+
the share of them that did. Seed cases from the engagement's own golden
|
|
894
|
+
queries -- the ones the eval owner would grade -- never from a benchmark.
|
|
895
|
+
"""
|
|
896
|
+
|
|
897
|
+
import argparse
|
|
898
|
+
import json
|
|
899
|
+
import sys
|
|
900
|
+
from pathlib import Path
|
|
901
|
+
|
|
902
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
|
903
|
+
|
|
904
|
+
HERE = Path(__file__).parent
|
|
905
|
+
KS = (10, 50)
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
def load_cases():
|
|
909
|
+
path = HERE / "retrieval_cases.jsonl"
|
|
910
|
+
if not path.exists():
|
|
911
|
+
return []
|
|
912
|
+
return [json.loads(line) for line in path.read_text().splitlines() if line.strip()]
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def resolve_retriever():
|
|
916
|
+
"""The deployed retriever, whatever shape its realization took.
|
|
917
|
+
|
|
918
|
+
Returns a callable (query, k) -> ranked results. Wiring problems surface
|
|
919
|
+
as the retriever's own refusal, which is the correct failure: an eval
|
|
920
|
+
that silently skipped an unwired store would grade a system that cannot
|
|
921
|
+
retrieve as though it could.
|
|
922
|
+
"""
|
|
923
|
+
from app.components import retrieval as mod
|
|
924
|
+
|
|
925
|
+
run = getattr(mod, "run", None)
|
|
926
|
+
if callable(run) and not isinstance(run, type):
|
|
927
|
+
return lambda query, k: run(query, top_k=k)
|
|
928
|
+
for name in sorted(dir(mod)):
|
|
929
|
+
obj = getattr(mod, name)
|
|
930
|
+
if isinstance(obj, type):
|
|
931
|
+
if callable(getattr(obj, "retrieve", None)):
|
|
932
|
+
return lambda query, k, _cls=obj: _cls().retrieve(query, k)
|
|
933
|
+
if callable(getattr(obj, "run", None)):
|
|
934
|
+
return lambda query, k, _cls=obj: _cls().run({"query": query, "k": k})
|
|
935
|
+
raise SystemExit(
|
|
936
|
+
"no retriever found in app/components/retrieval.py -- expected a "
|
|
937
|
+
"module-level run(query, top_k=...) or a class with retrieve() or run()"
|
|
938
|
+
)
|
|
939
|
+
|
|
940
|
+
|
|
941
|
+
def surfaced_ids(result):
|
|
942
|
+
if not isinstance(result, list):
|
|
943
|
+
return []
|
|
944
|
+
return [str(r.get("id")) for r in result if isinstance(r, dict) and "id" in r]
|
|
945
|
+
|
|
946
|
+
|
|
947
|
+
def main():
|
|
948
|
+
parser = argparse.ArgumentParser()
|
|
949
|
+
parser.add_argument("--min-recall", type=float, default=0.0,
|
|
950
|
+
help="fail below this mean recall@%d" % KS[0])
|
|
951
|
+
args = parser.parse_args()
|
|
952
|
+
|
|
953
|
+
cases = load_cases()
|
|
954
|
+
if not cases:
|
|
955
|
+
print("retrieval_cases.jsonl is empty -- nothing was measured, so "
|
|
956
|
+
"nothing passed. Seed it with golden queries and the document "
|
|
957
|
+
"ids a correct top-K must surface.", file=sys.stderr)
|
|
958
|
+
return 1
|
|
959
|
+
|
|
960
|
+
retrieve = resolve_retriever()
|
|
961
|
+
errors = 0
|
|
962
|
+
recalls = {k: [] for k in KS}
|
|
963
|
+
misses = []
|
|
964
|
+
for case in cases:
|
|
965
|
+
relevant = [str(r) for r in case.get("relevant", [])]
|
|
966
|
+
if not relevant:
|
|
967
|
+
errors += 1
|
|
968
|
+
misses.append({"id": case.get("id"), "note": "case lists no relevant ids"})
|
|
969
|
+
continue
|
|
970
|
+
try:
|
|
971
|
+
got = surfaced_ids(retrieve(case["query"], max(KS)))
|
|
972
|
+
except Exception as exc: # noqa: BLE001
|
|
973
|
+
errors += 1
|
|
974
|
+
misses.append({"id": case.get("id"), "note": f"retriever errored: {exc}"})
|
|
975
|
+
continue
|
|
976
|
+
for k in KS:
|
|
977
|
+
top = set(got[:k])
|
|
978
|
+
recalls[k].append(sum(1 for r in relevant if r in top) / len(relevant))
|
|
979
|
+
absent = [r for r in relevant if r not in set(got[: max(KS)])]
|
|
980
|
+
if absent:
|
|
981
|
+
misses.append({"id": case.get("id"), "missing": absent})
|
|
982
|
+
|
|
983
|
+
for k in KS:
|
|
984
|
+
scored = recalls[k]
|
|
985
|
+
mean = sum(scored) / len(scored) if scored else 0.0
|
|
986
|
+
print(f" recall@{k:<3} {len(scored):>4} cases {mean:.1%}")
|
|
987
|
+
for miss in misses[:10]:
|
|
988
|
+
print(f" {miss}", file=sys.stderr)
|
|
989
|
+
|
|
990
|
+
if errors:
|
|
991
|
+
print(f"{errors} case(s) errored -- the retriever is not wired end "
|
|
992
|
+
f"to end yet", file=sys.stderr)
|
|
993
|
+
return 1
|
|
994
|
+
gate = recalls[KS[0]]
|
|
995
|
+
mean = sum(gate) / len(gate) if gate else 0.0
|
|
996
|
+
if mean <= 0:
|
|
997
|
+
print("nothing relevant surfaced for any case. Look at the index and "
|
|
998
|
+
"the embedding model before anything downstream.",
|
|
999
|
+
file=sys.stderr)
|
|
1000
|
+
return 1
|
|
1001
|
+
if mean < args.min_recall:
|
|
1002
|
+
print(f"recall@{KS[0]} below {args.min_recall:.1%} -- fix retrieval "
|
|
1003
|
+
f"before touching prompts or models: nothing downstream "
|
|
1004
|
+
f"recovers a document that never surfaced", file=sys.stderr)
|
|
1005
|
+
return 1
|
|
1006
|
+
return 0
|
|
1007
|
+
|
|
1008
|
+
|
|
1009
|
+
if __name__ == "__main__":
|
|
1010
|
+
sys.exit(main())
|
|
1011
|
+
'''
|
|
1012
|
+
|
|
1013
|
+
|
|
867
1014
|
_HARNESS = '''#!/usr/bin/env python3
|
|
868
1015
|
"""Run the evaluation. Exits non-zero below the threshold, so CI can gate on it.
|
|
869
1016
|
|