xrefkit 0.5.1__tar.gz → 0.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {xrefkit-0.5.1 → xrefkit-0.6.0}/PKG-INFO +8 -3
- {xrefkit-0.5.1 → xrefkit-0.6.0}/README.md +7 -2
- {xrefkit-0.5.1 → xrefkit-0.6.0}/pyproject.toml +1 -1
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_attention_pet_client_protocol.py +97 -2
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_attention_pet_codex_session.py +3 -3
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_attention_pet_fit.py +69 -12
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/__init__.py +1 -1
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/client_protocol.py +7 -2
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/codex_session.py +2 -2
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/evaluator.py +31 -21
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/model.py +21 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/presentation.py +10 -10
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/profiles.py +25 -5
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/server.py +30 -9
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/attention_pet/pet.css +2 -2
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/attention_pet/pet.html +9 -9
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/attention_pet/pet.js +58 -16
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit.egg-info/PKG-INFO +8 -3
- {xrefkit-0.5.1 → xrefkit-0.6.0}/LICENSE +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/setup.cfg +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_artifact_selected_family.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_attention_pet.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_boundary_analysis.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_brownfield_csharp_structure_definitions.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_brownfield_file_editing_protocol.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_business_intake_conversation_definitions.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_business_intake_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_calibration_lint.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_catalog_preparation_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_check_skill_knowledge_xids.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_cli.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_code_constraint_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_collect_analyzer_sarif.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_constraint_derivation_family_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_constraint_derivation_selected_family.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_contribution_returns.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_convert_to_xrefkit_skill.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_cs_scope_probe.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_csharp_commonality.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_csharp_naming_profile.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_ctx.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_cutover_readiness.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_dashboard.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_database_batch_definition_migration.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_decision_trace.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_design_business_intake_selected_family.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_design_planning_definition_migration.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_dotnet_definition_candidate.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_editorial_family_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_editorial_intake_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_editorial_selected_family.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_embedded_startup_pack.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_error_policy_audit.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_error_policy_locator.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_execution_binding.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_existing_skill_definition_batch.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_final_skilldefinition_batch.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_flow_selected_family.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_fm_multiroot.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_gate.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_gateway.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_gateway_mcp.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_gateway_skill_adapter.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_gateway_work_items.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_goal_desired_state.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_host_precheck.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_human_evaluation.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_implementation_flow_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_implementation_review_selected_family.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_inbound_uploads.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_instruction_workflow.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_knowledge_relations_validator.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_language_review_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_mcp_package_routing.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_mcp_setup.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_mcp_subagent_startup.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_mcp_subagent_startup_integration.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_os_family_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_os_knowledge_family_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_os_selected_family.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_ownership.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_packmeta.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_project_quality_baseline.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_qa_report_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_resource_provider.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_runtime_contracts.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_sarif_to_locator.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_security_definition_migration.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skill_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skill_definition_governance.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skill_definition_mcp.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skill_definition_runtime.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skill_edits.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skill_maturity_return.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skill_runtime_audit.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skillmeta.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_skills_sync.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_structure_catalog.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_subagent_startup.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_subagent_startup_boundaries.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_wbs.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_workflow_family_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_xref.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_xrefkit_instance.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_xrefkit_tools.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_xrefkit_v2_discovery.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_xrefkit_v2_models.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/tests/test_xrefkit_v2_pipeline.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/__main__.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/__init__.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/__main__.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/client.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/scenarios.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/attention_pet/store.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/boundary_analysis.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/catalog_cli.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/cli.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/contracts.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/ctx.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/dashboard.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/decision_trace.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/discovery.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/execution_binding.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/execution_binding_cli.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/gate.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/gateway.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/goalstate.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/hashing.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/host_precheck.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/import_skill.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/instance.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/loaders.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/__init__.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/audit.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/bootstrap.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/catalog.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/cli.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/client_cache.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/client_flow.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/context_registry.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/context_token.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/contracts.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/contribution_adoption.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/contribution_returns.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/dist.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/gateway.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/inbound_uploads.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/knowledge_edits.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/ownership.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/repository.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/schemas.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/server.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/setup.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/skill_edits.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/skill_maturity.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/startup_contract_pack.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp/subagent_startup.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/mcp_tools.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/__init__.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/common.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/effective_bundle.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/human_evaluation.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/local_manifest.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/package_manifest.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/run_log.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/server_config.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/models/skill_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/operations_cli.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/ownership.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/packmeta.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/registry.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resolver.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resource_provider.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/attention_pet/client-state.schema.json +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/contracts.json +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/current.json +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/generations/7a682a5272907354/contracts.json +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/generations/7a682a5272907354/model_body.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/generations/9929294385ccb7b0/contracts.json +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/generations/9929294385ccb7b0/model_body.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/model_body.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/startup_sources/000_agent_entry.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/startup_sources/011_startup_xref_routing.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/startup_sources/015_shared_memory_operations.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/startup_sources/016_uncertainty_protocol.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/startup_sources/017_base_and_xref_layering.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/resources/base/startup_sources/053_context_direction_security_guard.md +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/runlog.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/skill_definition.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/skill_definition_catalog.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/skill_definition_governance.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/skillmeta.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/skillrun.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/skills_sync.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/structure_catalog.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/subagent_startup.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/tools/__init__.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/tools/__main__.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/v2_cli.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/wbs.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/workspace.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit/xref.py +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit.egg-info/SOURCES.txt +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit.egg-info/dependency_links.txt +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit.egg-info/entry_points.txt +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit.egg-info/requires.txt +0 -0
- {xrefkit-0.5.1 → xrefkit-0.6.0}/xrefkit.egg-info/top_level.txt +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: xrefkit
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.6.0
|
|
4
4
|
Summary: Portable XID, Skill, Knowledge, workflow, and MCP runtime for XRefKit
|
|
5
5
|
Author: synthaicode
|
|
6
6
|
License: MIT License
|
|
@@ -47,7 +47,7 @@ record evidence, preserve human judgment, and apply explicit completion checks.
|
|
|
47
47
|
|
|
48
48
|
## Why XRefKit?
|
|
49
49
|
|
|
50
|
-
Experimental local tool: [Attention Pet](projects/attention-pet/README.md) helps
|
|
50
|
+
Experimental local tool: [Attention Pet](projects/attention-pet/README.en.md) helps
|
|
51
51
|
people assess capability sufficiency and lower inference-cost candidates separately.
|
|
52
52
|
Its Model Fit / Cost Fit estimates use uncalibrated profiles; it does not measure internal
|
|
53
53
|
attention or switch models automatically.
|
|
@@ -61,6 +61,11 @@ pet. It also shows lower inference-cost comparison candidates when the
|
|
|
61
61
|
experimental profiles support one. The estimate is observational and does not
|
|
62
62
|
change the Codex model, reasoning level, or chat.
|
|
63
63
|
|
|
64
|
+
It does not measure token usage. It estimates the structural complexity of the
|
|
65
|
+
work context the AI must handle, including retained items, dependencies,
|
|
66
|
+
constraints, decision depth, conflicts, and dispersed evidence. It does not
|
|
67
|
+
read model-internal attention or remaining context-window capacity.
|
|
68
|
+
|
|
64
69
|
Run it from a Codex terminal in this repository so `CODEX_THREAD_ID` identifies
|
|
65
70
|
the current chat:
|
|
66
71
|
|
|
@@ -77,7 +82,7 @@ This Codex preview is bound to the chat that launched it. Selecting another
|
|
|
77
82
|
chat does not move the Pet automatically; launch it again from that chat when
|
|
78
83
|
you want a separate view. The model and reasoning dropdowns only compare
|
|
79
84
|
estimates and do not change the active Codex settings. See the
|
|
80
|
-
[Attention Pet guide](projects/attention-pet/README.md) for interpretation,
|
|
85
|
+
[Attention Pet guide](projects/attention-pet/README.en.md) for interpretation,
|
|
81
86
|
limitations, client integration, and API details.
|
|
82
87
|
|
|
83
88
|
Using AI for real work creates recurring operating problems:
|
|
@@ -8,7 +8,7 @@ record evidence, preserve human judgment, and apply explicit completion checks.
|
|
|
8
8
|
|
|
9
9
|
## Why XRefKit?
|
|
10
10
|
|
|
11
|
-
Experimental local tool: [Attention Pet](projects/attention-pet/README.md) helps
|
|
11
|
+
Experimental local tool: [Attention Pet](projects/attention-pet/README.en.md) helps
|
|
12
12
|
people assess capability sufficiency and lower inference-cost candidates separately.
|
|
13
13
|
Its Model Fit / Cost Fit estimates use uncalibrated profiles; it does not measure internal
|
|
14
14
|
attention or switch models automatically.
|
|
@@ -22,6 +22,11 @@ pet. It also shows lower inference-cost comparison candidates when the
|
|
|
22
22
|
experimental profiles support one. The estimate is observational and does not
|
|
23
23
|
change the Codex model, reasoning level, or chat.
|
|
24
24
|
|
|
25
|
+
It does not measure token usage. It estimates the structural complexity of the
|
|
26
|
+
work context the AI must handle, including retained items, dependencies,
|
|
27
|
+
constraints, decision depth, conflicts, and dispersed evidence. It does not
|
|
28
|
+
read model-internal attention or remaining context-window capacity.
|
|
29
|
+
|
|
25
30
|
Run it from a Codex terminal in this repository so `CODEX_THREAD_ID` identifies
|
|
26
31
|
the current chat:
|
|
27
32
|
|
|
@@ -38,7 +43,7 @@ This Codex preview is bound to the chat that launched it. Selecting another
|
|
|
38
43
|
chat does not move the Pet automatically; launch it again from that chat when
|
|
39
44
|
you want a separate view. The model and reasoning dropdowns only compare
|
|
40
45
|
estimates and do not change the active Codex settings. See the
|
|
41
|
-
[Attention Pet guide](projects/attention-pet/README.md) for interpretation,
|
|
46
|
+
[Attention Pet guide](projects/attention-pet/README.en.md) for interpretation,
|
|
42
47
|
limitations, client integration, and API details.
|
|
43
48
|
|
|
44
49
|
Using AI for real work creates recurring operating problems:
|
|
@@ -101,10 +101,16 @@ def test_handshake_authentication_and_active_session_switching(tmp_path):
|
|
|
101
101
|
get("/api/state?lang=fr")
|
|
102
102
|
assert error.value.code == 400
|
|
103
103
|
|
|
104
|
-
|
|
105
|
-
|
|
104
|
+
next_session = payload("session-b", 2, 2)
|
|
105
|
+
next_session["model"] = "gpt-6-astra"
|
|
106
|
+
next_session["reasoning"] = "high"
|
|
107
|
+
post("/api/active-session", next_session)
|
|
108
|
+
second = get("/api/state?model=&reasoning=medium")
|
|
106
109
|
assert second["source"]["sessionId"] == "session-b"
|
|
110
|
+
assert second["fit"]["selectedProfile"] == {"model": "astra", "reasoning": "high"}
|
|
107
111
|
assert second["state"]["features"]["active_items"] == 2
|
|
112
|
+
manual_comparison = get("/api/state?model=sol&reasoning=medium")
|
|
113
|
+
assert manual_comparison["fit"]["selectedProfile"] == {"model": "sol", "reasoning": "medium"}
|
|
108
114
|
|
|
109
115
|
with pytest.raises(HTTPError) as error:
|
|
110
116
|
post("/api/active-session", payload("session-a", 1, 1))
|
|
@@ -116,6 +122,95 @@ def test_handshake_authentication_and_active_session_switching(tmp_path):
|
|
|
116
122
|
thread.join(timeout=5)
|
|
117
123
|
|
|
118
124
|
|
|
125
|
+
def test_state_response_keeps_one_session_when_activation_follows_snapshot(tmp_path, monkeypatch):
|
|
126
|
+
source = ClientStateSource(tmp_path / "sessions", Weights())
|
|
127
|
+
source.activate(ClientState.model_validate(payload("session-a", 1)))
|
|
128
|
+
next_session = payload("session-b", 2)
|
|
129
|
+
next_session["model"] = "gpt-6-astra"
|
|
130
|
+
next_session["reasoning"] = "high"
|
|
131
|
+
next_state = ClientState.model_validate(next_session)
|
|
132
|
+
original_snapshot = source.snapshot
|
|
133
|
+
|
|
134
|
+
def switch_after_snapshot(fallback_store):
|
|
135
|
+
snapshot = original_snapshot(fallback_store)
|
|
136
|
+
source.activate(next_state)
|
|
137
|
+
return snapshot
|
|
138
|
+
|
|
139
|
+
monkeypatch.setattr(source, "snapshot", switch_after_snapshot)
|
|
140
|
+
server, launch = make_server(Store(), source=source)
|
|
141
|
+
thread = threading.Thread(target=server.serve_forever, daemon=True)
|
|
142
|
+
thread.start()
|
|
143
|
+
try:
|
|
144
|
+
with urlopen(launch + "api/state?model=&reasoning=medium", timeout=5) as response:
|
|
145
|
+
before = json.load(response)
|
|
146
|
+
assert before["state"]["taskId"] == "task-session-a"
|
|
147
|
+
assert before["source"]["sessionId"] == "session-a"
|
|
148
|
+
assert before["fit"]["selectedProfile"] == {"model": "sol", "reasoning": "medium"}
|
|
149
|
+
|
|
150
|
+
monkeypatch.setattr(source, "snapshot", original_snapshot)
|
|
151
|
+
with urlopen(launch + "api/state?model=&reasoning=medium", timeout=5) as response:
|
|
152
|
+
after = json.load(response)
|
|
153
|
+
assert after["state"]["taskId"] == "task-session-b"
|
|
154
|
+
assert after["source"]["sessionId"] == "session-b"
|
|
155
|
+
assert after["fit"]["selectedProfile"] == {"model": "astra", "reasoning": "high"}
|
|
156
|
+
finally:
|
|
157
|
+
server.shutdown()
|
|
158
|
+
server.server_close()
|
|
159
|
+
thread.join(timeout=5)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def test_client_snapshot_blocks_activation_until_all_session_fields_are_read(tmp_path, monkeypatch):
|
|
163
|
+
source = ClientStateSource(tmp_path / "sessions", Weights())
|
|
164
|
+
source.activate(ClientState.model_validate(payload("session-a", 1)))
|
|
165
|
+
next_session = payload("session-b", 2)
|
|
166
|
+
next_session["model"] = "gpt-6-astra"
|
|
167
|
+
next_session["reasoning"] = "high"
|
|
168
|
+
next_state = ClientState.model_validate(next_session)
|
|
169
|
+
view_read = threading.Event()
|
|
170
|
+
resume_view = threading.Event()
|
|
171
|
+
activation_started = threading.Event()
|
|
172
|
+
activation_done = threading.Event()
|
|
173
|
+
captured = {}
|
|
174
|
+
original_view = source.view
|
|
175
|
+
|
|
176
|
+
def pause_after_view(fallback_store):
|
|
177
|
+
result = original_view(fallback_store)
|
|
178
|
+
view_read.set()
|
|
179
|
+
assert resume_view.wait(timeout=5)
|
|
180
|
+
return result
|
|
181
|
+
|
|
182
|
+
def read_snapshot():
|
|
183
|
+
captured["snapshot"] = source.snapshot(Store())
|
|
184
|
+
|
|
185
|
+
def activate_next():
|
|
186
|
+
activation_started.set()
|
|
187
|
+
source.activate(next_state)
|
|
188
|
+
activation_done.set()
|
|
189
|
+
|
|
190
|
+
monkeypatch.setattr(source, "view", pause_after_view)
|
|
191
|
+
reader = threading.Thread(target=read_snapshot)
|
|
192
|
+
activator = threading.Thread(target=activate_next)
|
|
193
|
+
reader.start()
|
|
194
|
+
try:
|
|
195
|
+
assert view_read.wait(timeout=5)
|
|
196
|
+
activator.start()
|
|
197
|
+
assert activation_started.wait(timeout=5)
|
|
198
|
+
assert not activation_done.wait(timeout=0.1)
|
|
199
|
+
finally:
|
|
200
|
+
resume_view.set()
|
|
201
|
+
reader.join(timeout=5)
|
|
202
|
+
if activator.ident is not None:
|
|
203
|
+
activator.join(timeout=5)
|
|
204
|
+
|
|
205
|
+
assert not reader.is_alive() and not activator.is_alive()
|
|
206
|
+
result, selection, status = captured["snapshot"]
|
|
207
|
+
assert result["state"]["taskId"] == "task-session-a"
|
|
208
|
+
assert selection == ("sol", "medium")
|
|
209
|
+
assert status["sessionId"] == "session-a"
|
|
210
|
+
assert activation_done.is_set()
|
|
211
|
+
assert source.status()["sessionId"] == "session-b"
|
|
212
|
+
|
|
213
|
+
|
|
119
214
|
def test_client_contract_rejects_mismatched_session_and_manual_mode(tmp_path):
|
|
120
215
|
source = ClientStateSource(tmp_path / "sessions", Weights())
|
|
121
216
|
server, launch = make_server(Store(), source=source)
|
|
@@ -38,7 +38,7 @@ def test_one_bound_chat_updates_incrementally_without_saving_text(tmp_path):
|
|
|
38
38
|
store = Store(saved)
|
|
39
39
|
source.sync(store)
|
|
40
40
|
first = store.view()
|
|
41
|
-
assert source.selection() == ("sol", "
|
|
41
|
+
assert source.selection() == ("sol", "medium")
|
|
42
42
|
assert source.status()["observedUserTurns"] == 1
|
|
43
43
|
assert first["state"]["coverage"] == "partial"
|
|
44
44
|
assert len(first["workingSet"]["items"]) == 1
|
|
@@ -55,7 +55,7 @@ def test_one_bound_chat_updates_incrementally_without_saving_text(tmp_path):
|
|
|
55
55
|
assert "この修正は不要" not in saved.read_text(encoding="utf-8")
|
|
56
56
|
append(log, "2026-09-26T12:00:03Z", "turn_context", {"model": "gpt-6-terra", "effort": "low"})
|
|
57
57
|
source.sync(store)
|
|
58
|
-
assert source.selection() == ("terra", "
|
|
58
|
+
assert source.selection() == ("terra", "low")
|
|
59
59
|
assert store.view() == second
|
|
60
60
|
append(log, "2026-09-26T12:00:04Z", "turn_context", {"model": "unrecognized-model", "effort": "medium"})
|
|
61
61
|
source.sync(store)
|
|
@@ -80,7 +80,7 @@ def test_live_api_defaults_to_observed_model_allows_comparison_and_rejects_mutat
|
|
|
80
80
|
assert value["source"]["mode"] == "codex-chat"
|
|
81
81
|
assert value["source"]["observedUserTurns"] == 1
|
|
82
82
|
assert value["source"]["profile"] == "sol"
|
|
83
|
-
assert value["source"]["reasoning"] == "
|
|
83
|
+
assert value["source"]["reasoning"] == "medium"
|
|
84
84
|
assert value["fit"]["selected"]["model"] == "astra"
|
|
85
85
|
assert value["fit"]["selected"]["reasoning"] == "high"
|
|
86
86
|
with urlopen(Request(base + "/api/state?model=&reasoning=standard"), timeout=5) as response:
|
|
@@ -16,6 +16,50 @@ from xrefkit.attention_pet.server import make_server
|
|
|
16
16
|
from xrefkit.attention_pet.store import Store
|
|
17
17
|
|
|
18
18
|
|
|
19
|
+
def test_execution_profile_search_crosses_models_and_efforts(monkeypatch):
|
|
20
|
+
# Controlled hypotheses make the two cross-axis cases independent of scenario thresholds.
|
|
21
|
+
state = {"ral": 50, "causes": [], "features": {"active_items": 1},
|
|
22
|
+
"coverage": "reviewed", "trajectoryEvidence": []}
|
|
23
|
+
for model, base in (("luna", 20), ("sol", 40)):
|
|
24
|
+
monkeypatch.setitem(PROFILES, model, PROFILES[model].model_copy(
|
|
25
|
+
update={"capability": Capability(reasoning=base, constraint_tracking=base,
|
|
26
|
+
evidence_handling=base)}))
|
|
27
|
+
for model in ("terra", "astra"):
|
|
28
|
+
monkeypatch.setitem(PROFILES, model, PROFILES[model].model_copy(
|
|
29
|
+
update={"relative_inference_cost": 100.0}))
|
|
30
|
+
result = evaluate_fit(state, "sol", "high")
|
|
31
|
+
assert result["modelFit"] == "Sufficient"
|
|
32
|
+
assert result["costFit"] == "LowerCostCandidateAvailable"
|
|
33
|
+
assert { (c["model"], c["reasoning"]) for c in result["lowerCostCandidates"] } >= {
|
|
34
|
+
("luna", "xhigh"), ("sol", "medium")}
|
|
35
|
+
assert not next(c for c in result["alternatives"] if
|
|
36
|
+
(c["model"], c["reasoning"]) == ("luna", "high"))["meetsRequirements"]
|
|
37
|
+
assert result["lowerCostCandidate"] == {"model": "sol", "reasoning": "medium"}
|
|
38
|
+
assert result["selectedProfile"] == {"model": "sol", "reasoning": "high"}
|
|
39
|
+
assert result["expectedTotalCost"] is None
|
|
40
|
+
|
|
41
|
+
underpowered = evaluate_fit(state, "sol", "low")
|
|
42
|
+
assert underpowered["modelFit"] == "Underpowered"
|
|
43
|
+
assert underpowered["costFit"] == "RetryRisk"
|
|
44
|
+
assert underpowered["lowerCostCandidate"] is None
|
|
45
|
+
assert underpowered["lowerCostCandidates"] == []
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def test_unknown_and_legacy_execution_profiles():
|
|
49
|
+
state = evaluate(task())
|
|
50
|
+
for model, reasoning in (("absent", "high"), ("sol", "absent"),
|
|
51
|
+
("terra", "xhigh")):
|
|
52
|
+
result = evaluate_fit(state, model, reasoning)
|
|
53
|
+
assert result["modelFit"] == result["costFit"] == "Unknown"
|
|
54
|
+
assert result["presentation"]["petState"] == "Unknown"
|
|
55
|
+
assert evaluate_fit(state, "sol", "light")["selected"]["reasoning"] == "light"
|
|
56
|
+
assert evaluate_fit(state, "sol", "standard")["selected"]["reasoning"] == "standard"
|
|
57
|
+
assert evaluate_fit(state, "sol", "light")["selectedProfile"] == {"model": "sol", "reasoning": "low"}
|
|
58
|
+
assert evaluate_fit(state, "sol", "standard")["selectedProfile"] == {"model": "sol", "reasoning": "medium"}
|
|
59
|
+
assert evaluate_fit(state, "sol", "standard")["selected"]["capability"] == \
|
|
60
|
+
evaluate_fit(state, "sol", "medium")["selected"]["capability"]
|
|
61
|
+
|
|
62
|
+
|
|
19
63
|
def task(name="A", step=5):
|
|
20
64
|
return WorkingSet.model_validate(scenario(name, step)["workingSet"])
|
|
21
65
|
|
|
@@ -70,10 +114,10 @@ def test_comparison_is_pure_and_total_cost_is_unknown():
|
|
|
70
114
|
assert {r["baseRal"] for r in results} == {state["ral"]}
|
|
71
115
|
for result in results:
|
|
72
116
|
assert result["expectedTotalCost"] is None
|
|
73
|
-
assert all(result[k] is None for k in ("inferenceCost", "retryCost", "correctionCost", "failureRiskCost"))
|
|
117
|
+
assert all(result[k] is None for k in ("inferenceCost", "retryCost", "correctionCost", "latencyCost", "failureRiskCost"))
|
|
74
118
|
assert result["calibration"] == "uncalibrated"
|
|
75
119
|
assert results[0]["costFit"] == "RetryRisk"
|
|
76
|
-
assert results[1]["costFit"] == "
|
|
120
|
+
assert results[1]["costFit"] == "LowerCostCandidateAvailable"
|
|
77
121
|
assert results[2]["costFit"] == "LowerCostCandidateAvailable"
|
|
78
122
|
|
|
79
123
|
|
|
@@ -121,9 +165,9 @@ def test_schema_matches_python_cost_fit_contract():
|
|
|
121
165
|
|
|
122
166
|
|
|
123
167
|
@pytest.mark.parametrize("name,step,costs,faces", [
|
|
124
|
-
("A", 5, ["
|
|
125
|
-
("B", 3, ["RetryRisk"
|
|
126
|
-
("B", 5, ["RetryRisk"] * 3 + ["
|
|
168
|
+
("A", 5, ["LowerCostCandidateAvailable"] * 4, ["Relaxed"] * 4),
|
|
169
|
+
("B", 3, ["RetryRisk"] + ["LowerCostCandidateAvailable"] * 3, ["Strained"] + ["Relaxed"] * 3),
|
|
170
|
+
("B", 5, ["RetryRisk"] * 3 + ["LowerCostCandidateAvailable"], ["Strained"] * 3 + ["Relaxed"]),
|
|
127
171
|
("D", 5, ["RetryRisk"] * 4, ["Strained"] * 4),
|
|
128
172
|
("C", 3, ["ReviewNeeded"] * 4, ["Review"] * 4),
|
|
129
173
|
])
|
|
@@ -141,11 +185,11 @@ def test_scenarios_separate_cost_and_presentation(name, step, costs, faces):
|
|
|
141
185
|
|
|
142
186
|
def test_other_prices_do_not_change_model_fit(monkeypatch):
|
|
143
187
|
state = evaluate(task())
|
|
144
|
-
before = evaluate_fit(state, "astra")
|
|
188
|
+
before = evaluate_fit(state, "astra", "low")
|
|
145
189
|
assert before["costFit"] == "LowerCostCandidateAvailable"
|
|
146
190
|
for name in ("luna", "terra", "sol"):
|
|
147
191
|
monkeypatch.setitem(PROFILES, name, PROFILES[name].model_copy(update={"relative_inference_cost": 100.0}))
|
|
148
|
-
after = evaluate_fit(state, "astra")
|
|
192
|
+
after = evaluate_fit(state, "astra", "low")
|
|
149
193
|
assert before["modelFit"] == after["modelFit"] == "Sufficient"
|
|
150
194
|
assert after["costFit"] == "NoLowerCostCandidate"
|
|
151
195
|
assert after["lowerCostCandidates"] == []
|
|
@@ -153,9 +197,9 @@ def test_other_prices_do_not_change_model_fit(monkeypatch):
|
|
|
153
197
|
|
|
154
198
|
@pytest.mark.parametrize("name,step,model,status,target", [
|
|
155
199
|
("A", 5, "astra", "Available", "Luna"),
|
|
156
|
-
("B", 3, "astra", "Available", "
|
|
157
|
-
("B", 3, "sol", "Available", "
|
|
158
|
-
("B", 3, "terra", "
|
|
200
|
+
("B", 3, "astra", "Available", "Sol"),
|
|
201
|
+
("B", 3, "sol", "Available", "Sol"),
|
|
202
|
+
("B", 3, "terra", "Available", "Sol"),
|
|
159
203
|
("B", 3, "luna", "Underpowered", None),
|
|
160
204
|
("C", 3, "astra", "HoldForReview", None),
|
|
161
205
|
("D", 5, "astra", "Underpowered", None),
|
|
@@ -168,7 +212,7 @@ def test_lowest_sufficient_downgrade_guidance(name, step, model, status, target)
|
|
|
168
212
|
if target:
|
|
169
213
|
assert f"低コスト比較候補: {target}" in guide["modelGuide"]
|
|
170
214
|
assert target in guide["modelGuideShort"]
|
|
171
|
-
assert "
|
|
215
|
+
assert " / " in guide["modelGuideShort"]
|
|
172
216
|
assert "再試行・修正時間・失敗損失を含む総コストは未比較" in guide["modelGuideDetail"]
|
|
173
217
|
assert "仮の必要能力3軸を満たす試算" in guide["modelGuideDetail"]
|
|
174
218
|
lowest = min(fit["lowerCostCandidates"], key=lambda c: c["relativeInferenceCost"])
|
|
@@ -241,7 +285,7 @@ def test_english_presentation_preserves_evaluation_meaning():
|
|
|
241
285
|
("Unknown", "Unknown", "Unknown", "まだ評価できません", "情報が不足"),
|
|
242
286
|
("Underpowered", "RetryRisk", "Strained", "能力が不足する可能性", "一部を満たしていません"),
|
|
243
287
|
("Sufficient", "NoLowerCostCandidate", "Balanced", "必要な能力を満たす試算", "候補は確認されていません"),
|
|
244
|
-
("Sufficient", "LowerCostCandidateAvailable", "Relaxed", "より低い推論コストの候補", "
|
|
288
|
+
("Sufficient", "LowerCostCandidateAvailable", "Relaxed", "より低い推論コストの候補", "現在の実行プロファイルでも必要能力を満たす試算"),
|
|
245
289
|
("Sufficient", "ReviewNeeded", "Review", "実際の結果", "モデル能力だけを原因とは判断していません"),
|
|
246
290
|
("Underpowered", "ReviewNeeded", "Review", "実際の結果", "モデル能力だけを原因とは判断していません"),
|
|
247
291
|
])
|
|
@@ -309,6 +353,13 @@ def test_profiles_reject_invalid_costs(field, value):
|
|
|
309
353
|
ModelProfile.model_validate(data)
|
|
310
354
|
|
|
311
355
|
|
|
356
|
+
def test_profiles_reject_reasoning_without_depth_definition():
|
|
357
|
+
data = PROFILES["luna"].model_dump()
|
|
358
|
+
data["supported_reasoning"] = ("low", "ultra")
|
|
359
|
+
with pytest.raises(ValidationError, match="unsupported reasoning levels: ultra"):
|
|
360
|
+
ModelProfile.model_validate(data)
|
|
361
|
+
|
|
362
|
+
|
|
312
363
|
def test_api_model_selection_is_read_only_and_invalid_queries_cannot_mutate():
|
|
313
364
|
store = Store()
|
|
314
365
|
before = store.submit(task("B", 3))
|
|
@@ -324,6 +375,12 @@ def test_api_model_selection_is_read_only_and_invalid_queries_cannot_mutate():
|
|
|
324
375
|
return json.load(response)
|
|
325
376
|
|
|
326
377
|
try:
|
|
378
|
+
catalog = request("/api/profiles")
|
|
379
|
+
assert next(p for p in catalog["models"] if p["id"] == "luna")["reasoning"] == [
|
|
380
|
+
"low", "medium", "high", "xhigh"]
|
|
381
|
+
assert catalog["legacyAliases"] == {"light": "low", "standard": "medium"}
|
|
382
|
+
unknown = request("/api/state?model=absent&reasoning=xhigh")
|
|
383
|
+
assert unknown["fit"]["modelFit"] == "Unknown"
|
|
327
384
|
for model, expected in zip(PROFILES, ["Underpowered", "Sufficient", "Sufficient", "Sufficient"], strict=True):
|
|
328
385
|
value = request(f"/api/state?model={model}&reasoning=standard")
|
|
329
386
|
assert value["fit"]["modelFit"] == expected
|
|
@@ -93,6 +93,11 @@ class ClientStateSource:
|
|
|
93
93
|
with self.lock:
|
|
94
94
|
return self._store(self.active).view() if self.active else _fallback_store.view()
|
|
95
95
|
|
|
96
|
+
def snapshot(self, fallback_store: Store) -> tuple[dict, tuple[str, str], dict]:
|
|
97
|
+
"""Read the active work, execution profile, and source under one lock."""
|
|
98
|
+
with self.lock:
|
|
99
|
+
return self.view(fallback_store), self.selection(), self.status()
|
|
100
|
+
|
|
96
101
|
def selection(self) -> tuple[str, str]:
|
|
97
102
|
with self.lock:
|
|
98
103
|
state = self.states.get(self.active) if self.active else None
|
|
@@ -100,8 +105,8 @@ class ClientStateSource:
|
|
|
100
105
|
return "", "standard"
|
|
101
106
|
family = state.model.rsplit("-", 1)[-1]
|
|
102
107
|
model = family if state.model.startswith("gpt-") and family in {"luna", "terra", "sol", "astra"} else ""
|
|
103
|
-
depth = {"none": "
|
|
104
|
-
"high": "high", "xhigh": "
|
|
108
|
+
depth = {"none": "low", "minimal": "low", "low": "low", "medium": "medium",
|
|
109
|
+
"high": "high", "xhigh": "xhigh", "max": "max", "ultra": "max"}.get(state.reasoning)
|
|
105
110
|
return (model, depth) if model and depth else ("", "standard")
|
|
106
111
|
|
|
107
112
|
def status(self) -> dict:
|
|
@@ -20,8 +20,8 @@ _AMBIENT = re.compile(r"<in-app-browser-context\b[^>]*>.*?</in-app-browser-conte
|
|
|
20
20
|
_REFERENCE = re.compile(r"(?:^|\s)(?:この|これ|それ|その|同じ|つづけて|続けて)")
|
|
21
21
|
_CONSTRAINT = ("不要", "削除", "しない", "なくす", "やめ", "違う", "おかしい", "修正", "変更")
|
|
22
22
|
_MODEL_SUFFIX = {"luna": "luna", "terra": "terra", "sol": "sol", "astra": "astra"}
|
|
23
|
-
_DEPTH = {"none": "
|
|
24
|
-
"high": "high", "xhigh": "
|
|
23
|
+
_DEPTH = {"none": "low", "minimal": "low", "low": "low", "medium": "medium",
|
|
24
|
+
"high": "high", "xhigh": "xhigh", "max": "max", "ultra": "max"}
|
|
25
25
|
|
|
26
26
|
|
|
27
27
|
def find_rollout(thread_id: str, codex_home: Path) -> Path:
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
"""Explainable heuristic v1. Thresholds are hypotheses, not calibrated probabilities."""
|
|
2
2
|
from dataclasses import dataclass, asdict
|
|
3
3
|
|
|
4
|
-
from .model import AttentionState, WorkingSet, Capability, FitCandidate, FitEvaluation, ModelProfile, ReasoningDepth
|
|
5
|
-
from .profiles import PROFILES, DEPTHS
|
|
4
|
+
from .model import AttentionState, WorkingSet, Capability, ExecutionProfile, FitCandidate, FitEvaluation, ModelProfile, ReasoningDepth
|
|
5
|
+
from .profiles import PROFILES, DEPTHS, canonical_reasoning, execution_profiles
|
|
6
6
|
from .presentation import present_fit
|
|
7
7
|
|
|
8
8
|
|
|
@@ -110,6 +110,12 @@ def evaluate(ws: WorkingSet, previous: dict | None = None, history: list[dict] |
|
|
|
110
110
|
return AttentionState.model_validate(result).model_dump()
|
|
111
111
|
|
|
112
112
|
|
|
113
|
+
def profile_capability(profile: ModelProfile, depth: ReasoningDepth) -> Capability:
|
|
114
|
+
"""Replaceable, uncalibrated capability hypothesis for one execution profile."""
|
|
115
|
+
return Capability(**{k: min(100, max(0, v + depth.capability_modifier))
|
|
116
|
+
for k, v in profile.capability.model_dump().items()})
|
|
117
|
+
|
|
118
|
+
|
|
113
119
|
def fit_candidate(state: dict, profile: ModelProfile, depth: ReasoningDepth) -> FitCandidate:
|
|
114
120
|
"""Model-generated expansion is a projection, never added to actual input/history."""
|
|
115
121
|
base = state["ral"]
|
|
@@ -134,8 +140,7 @@ def fit_candidate(state: dict, profile: ModelProfile, depth: ReasoningDepth) ->
|
|
|
134
140
|
constraint_tracking=min(100, round(.65 * effective + 25 * constraints + 10 * unresolved, 1)),
|
|
135
141
|
evidence_handling=min(100, round(.65 * effective + 35 * evidence, 1)),
|
|
136
142
|
)
|
|
137
|
-
capability =
|
|
138
|
-
for k, v in profile.capability.model_dump().items()})
|
|
143
|
+
capability = profile_capability(profile, depth)
|
|
139
144
|
shortfalls = [k for k, need in required.model_dump().items() if capability.model_dump()[k] < need]
|
|
140
145
|
return FitCandidate(model=profile.id, reasoning=depth.id, expansion=expansion,
|
|
141
146
|
effectiveRal=effective, capability=capability, requiredCapability=required,
|
|
@@ -148,14 +153,12 @@ def evaluate_fit(state: dict | None, model: str = "", reasoning: str = "standard
|
|
|
148
153
|
"""Relative quality/cost allocation hypothesis; never a routing decision."""
|
|
149
154
|
if locale not in {"ja", "en"}:
|
|
150
155
|
raise ValueError("unsupported display language")
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
if
|
|
154
|
-
|
|
155
|
-
if not model or state is None or not state["features"]["active_items"]:
|
|
156
|
-
reason = (("Select a model to compare." if not model else "There is no work data to evaluate.")
|
|
156
|
+
normalized = canonical_reasoning(reasoning)
|
|
157
|
+
known = model in PROFILES and normalized in DEPTHS and normalized in PROFILES[model].supported_reasoning
|
|
158
|
+
if not known or state is None or not state["features"]["active_items"]:
|
|
159
|
+
reason = (("Select a supported model and reasoning level." if not known else "There is no work data to evaluate.")
|
|
157
160
|
if locale == "en" else
|
|
158
|
-
("
|
|
161
|
+
("対応するモデルと考える深さを選んでください。" if not known else "判定に使う作業内容がありません。"))
|
|
159
162
|
cost_fit = "ReviewNeeded" if state and any(o["kind"] != "validated" for o in state["trajectoryEvidence"]) else "Unknown"
|
|
160
163
|
reasons = [reason]
|
|
161
164
|
if cost_fit == "ReviewNeeded":
|
|
@@ -169,10 +172,12 @@ def evaluate_fit(state: dict | None, model: str = "", reasoning: str = "standard
|
|
|
169
172
|
baseRal=state["ral"] if state else None).model_dump()
|
|
170
173
|
profile, depth = PROFILES[model], DEPTHS[reasoning]
|
|
171
174
|
selected = fit_candidate(state, profile, depth)
|
|
172
|
-
candidates = [fit_candidate(state, p,
|
|
175
|
+
candidates = [fit_candidate(state, p, d) for p, d in execution_profiles()
|
|
176
|
+
if (p.id, d.id) != (model, normalized)]
|
|
173
177
|
cheaper = [c for c in candidates if c.meetsRequirements
|
|
174
178
|
and c.relativeInferenceCost < selected.relativeInferenceCost]
|
|
175
|
-
|
|
179
|
+
cheaper.sort(key=lambda c: (c.relativeInferenceCost, c.model, c.reasoning))
|
|
180
|
+
lowest_sufficient = cheaper[0] if cheaper else None
|
|
176
181
|
axes = ({"reasoning": "reasoning", "constraint_tracking": "constraint tracking",
|
|
177
182
|
"evidence_handling": "evidence handling"} if locale == "en" else
|
|
178
183
|
{"reasoning": "推論", "constraint_tracking": "制約の保持", "evidence_handling": "根拠の扱い"})
|
|
@@ -187,21 +192,21 @@ def evaluate_fit(state: dict | None, model: str = "", reasoning: str = "standard
|
|
|
187
192
|
"再試行や修正が増える可能性があります。回数・損失は未推定です。"])
|
|
188
193
|
elif cheaper:
|
|
189
194
|
cost_fit = "LowerCostCandidateAvailable"
|
|
190
|
-
labels = "
|
|
191
|
-
reasons = (["The selected
|
|
192
|
-
f"
|
|
195
|
+
labels = ", ".join(f"{PROFILES[c.model].label} / {c.reasoning}" for c in cheaper[:3])
|
|
196
|
+
reasons = (["The selected execution profile meets all three estimated capability requirements.",
|
|
197
|
+
f"Across compatible execution profiles, {labels} also meets the requirements at a lower estimated inference-cost index.",
|
|
193
198
|
"Retries, correction time, and failure losses are not compared. This does not establish equal quality, lower total cost, or a need to change models."]
|
|
194
199
|
if locale == "en" else
|
|
195
|
-
|
|
196
|
-
f"
|
|
200
|
+
["選択した実行プロファイルは仮の必要能力3軸を満たしています。",
|
|
201
|
+
f"対応する実行プロファイルを横断すると、{labels} も必要能力を満たし、より低い推論コスト指数となる試算です。",
|
|
197
202
|
"再試行・修正・失敗損失はまだ比較していません。実品質の同等性、総コストの低下、モデル変更の必要性を示すものではありません。"])
|
|
198
203
|
else:
|
|
199
204
|
cost_fit = "NoLowerCostCandidate"
|
|
200
205
|
reasons = (["All estimated capability requirements are met.",
|
|
201
|
-
"No registered
|
|
206
|
+
"No registered execution profile meets the requirements at a lower estimated inference-cost index. Quality and total-cost advantages are unverified."]
|
|
202
207
|
if locale == "en" else
|
|
203
208
|
["仮の必要能力を全項目で満たしています。",
|
|
204
|
-
"
|
|
209
|
+
"登録された実行プロファイルには、より低い推論コスト指数で必要能力を満たすものがありません。実品質や総コストの優位性は未確認です。"])
|
|
205
210
|
if any(o["kind"] != "validated" for o in state["trajectoryEvidence"]):
|
|
206
211
|
cost_fit = "ReviewNeeded"
|
|
207
212
|
reasons.append("Failure or correction records require a separate total-cost review including retries and corrections."
|
|
@@ -217,6 +222,11 @@ def evaluate_fit(state: dict | None, model: str = "", reasoning: str = "standard
|
|
|
217
222
|
presentation=present_fit(model_fit, cost_fit, state["coverage"],
|
|
218
223
|
selected=selected, lowest_sufficient=lowest_sufficient,
|
|
219
224
|
locale=locale),
|
|
220
|
-
lowerCostCandidates=cheaper
|
|
225
|
+
lowerCostCandidates=(cheaper if cost_fit == "LowerCostCandidateAvailable" else []),
|
|
226
|
+
inferenceCostIndex=selected.relativeInferenceCost,
|
|
227
|
+
selectedProfile=ExecutionProfile(model=model, reasoning=normalized),
|
|
228
|
+
lowerCostCandidate=(ExecutionProfile(model=lowest_sufficient.model,
|
|
229
|
+
reasoning=lowest_sufficient.reasoning)
|
|
230
|
+
if cost_fit == "LowerCostCandidateAvailable" and lowest_sufficient else None),
|
|
221
231
|
selected=selected, profile=profile, depth=depth,
|
|
222
232
|
alternatives=candidates, reasons=reasons).model_dump()
|
|
@@ -100,14 +100,27 @@ class Behavior(Contract):
|
|
|
100
100
|
compression: float = Field(ge=0, le=1)
|
|
101
101
|
|
|
102
102
|
|
|
103
|
+
CANONICAL_REASONING_LEVELS = frozenset({"low", "medium", "high", "xhigh", "max"})
|
|
104
|
+
|
|
105
|
+
|
|
103
106
|
class ModelProfile(Contract):
|
|
104
107
|
id: str
|
|
105
108
|
label: str
|
|
106
109
|
capability: Capability
|
|
107
110
|
behavior: Behavior
|
|
108
111
|
relative_inference_cost: float = Field(gt=0, allow_inf_nan=False)
|
|
112
|
+
supported_reasoning: tuple[str, ...] = ("low", "medium", "high")
|
|
109
113
|
calibration: Literal["uncalibrated"] = "uncalibrated"
|
|
110
114
|
|
|
115
|
+
@model_validator(mode="after")
|
|
116
|
+
def supported_levels_are_distinct(self):
|
|
117
|
+
if not self.supported_reasoning or len(set(self.supported_reasoning)) != len(self.supported_reasoning):
|
|
118
|
+
raise ValueError("supported reasoning levels must be nonempty and distinct")
|
|
119
|
+
unsupported = set(self.supported_reasoning) - CANONICAL_REASONING_LEVELS
|
|
120
|
+
if unsupported:
|
|
121
|
+
raise ValueError(f"unsupported reasoning levels: {', '.join(sorted(unsupported))}")
|
|
122
|
+
return self
|
|
123
|
+
|
|
111
124
|
|
|
112
125
|
class ReasoningDepth(Contract):
|
|
113
126
|
id: str
|
|
@@ -116,6 +129,11 @@ class ReasoningDepth(Contract):
|
|
|
116
129
|
cost_modifier: float = Field(gt=0, allow_inf_nan=False)
|
|
117
130
|
|
|
118
131
|
|
|
132
|
+
class ExecutionProfile(Contract):
|
|
133
|
+
model: str
|
|
134
|
+
reasoning: str
|
|
135
|
+
|
|
136
|
+
|
|
119
137
|
class FitCandidate(Contract):
|
|
120
138
|
model: str
|
|
121
139
|
reasoning: str
|
|
@@ -157,6 +175,8 @@ class FitEvaluation(Contract):
|
|
|
157
175
|
presentation: Presentation
|
|
158
176
|
baseRal: int | None = Field(default=None, ge=0, le=100)
|
|
159
177
|
selected: FitCandidate | None = None
|
|
178
|
+
selectedProfile: ExecutionProfile | None = None
|
|
179
|
+
lowerCostCandidate: ExecutionProfile | None = None
|
|
160
180
|
profile: ModelProfile | None = None
|
|
161
181
|
depth: ReasoningDepth | None = None
|
|
162
182
|
alternatives: list[FitCandidate] = Field(default_factory=list)
|
|
@@ -168,6 +188,7 @@ class FitEvaluation(Contract):
|
|
|
168
188
|
inferenceCost: None = None
|
|
169
189
|
retryCost: None = None
|
|
170
190
|
correctionCost: None = None
|
|
191
|
+
latencyCost: None = None
|
|
171
192
|
failureRiskCost: None = None
|
|
172
193
|
|
|
173
194
|
|