precis-cli 0.1.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- app/__init__.py +16 -0
- app/api/__init__.py +32 -0
- app/api/dependencies.py +185 -0
- app/api/main.py +301 -0
- app/api/middleware/__init__.py +34 -0
- app/api/middleware/exception_handler.py +126 -0
- app/api/middleware/request_logging.py +97 -0
- app/api/middleware/token_auth.py +215 -0
- app/api/models/__init__.py +151 -0
- app/api/models/connection_rules.py +98 -0
- app/api/models/files.py +99 -0
- app/api/models/full_validation.py +554 -0
- app/api/models/project.py +117 -0
- app/api/models/projects.py +80 -0
- app/api/models/schema.py +121 -0
- app/api/models/v2_responses.py +46 -0
- app/api/models/validation.py +303 -0
- app/api/models/workspace.py +118 -0
- app/api/routers/__init__.py +56 -0
- app/api/routers/ai/__init__.py +36 -0
- app/api/routers/ai/chat.py +243 -0
- app/api/routers/ai/generate.py +156 -0
- app/api/routers/ai/hardware.py +199 -0
- app/api/routers/ai/jobs.py +544 -0
- app/api/routers/ai/migrate.py +471 -0
- app/api/routers/ai/models.py +352 -0
- app/api/routers/ai/ollama.py +108 -0
- app/api/routers/ai/providers.py +491 -0
- app/api/routers/ai/router.py +44 -0
- app/api/routers/ai/stream.py +281 -0
- app/api/routers/ai/utils.py +182 -0
- app/api/routers/core/__init__.py +34 -0
- app/api/routers/core/connection_rules.py +188 -0
- app/api/routers/core/data_sources.py +374 -0
- app/api/routers/core/regex.py +333 -0
- app/api/routers/core/reporting.py +263 -0
- app/api/routers/files/__init__.py +24 -0
- app/api/routers/files/ops.py +179 -0
- app/api/routers/files/transfer.py +101 -0
- app/api/routers/preview/__init__.py +35 -0
- app/api/routers/preview/content_mode.py +264 -0
- app/api/routers/preview/header_row.py +154 -0
- app/api/routers/preview/models.py +81 -0
- app/api/routers/preview/path_mode.py +440 -0
- app/api/routers/preview/router.py +30 -0
- app/api/routers/project/__init__.py +61 -0
- app/api/routers/project/base.py +111 -0
- app/api/routers/project/constraint.py +345 -0
- app/api/routers/project/full_config.py +411 -0
- app/api/routers/project/full_config_writer.py +327 -0
- app/api/routers/project/helpers.py +167 -0
- app/api/routers/project/inspection_fix.py +330 -0
- app/api/routers/project/manifest.py +605 -0
- app/api/routers/project/models.py +177 -0
- app/api/routers/project/pattern.py +268 -0
- app/api/routers/project/regex.py +358 -0
- app/api/routers/project/scanner.py +157 -0
- app/api/routers/project/schema.py +567 -0
- app/api/routers/project/schema_helpers.py +149 -0
- app/api/routers/project/settings.py +399 -0
- app/api/routers/project/template.py +325 -0
- app/api/routers/project/validation.py +227 -0
- app/api/routers/project/view.py +177 -0
- app/api/routers/project/workspaces.py +167 -0
- app/api/routers/projects/__init__.py +25 -0
- app/api/routers/projects/create.py +121 -0
- app/api/routers/projects/open.py +103 -0
- app/api/routers/projects/scan.py +100 -0
- app/api/routers/validation/__init__.py +39 -0
- app/api/routers/validation/common.py +263 -0
- app/api/routers/validation/content_mode.py +401 -0
- app/api/routers/validation/history.py +235 -0
- app/api/routers/validation/inline_mode.py +196 -0
- app/api/routers/validation/path_mode.py +314 -0
- app/api/routers/validation/router.py +51 -0
- app/api/services/full_validation_response_builder.py +412 -0
- app/api/services/io_error_messages.py +41 -0
- app/api/services/preview_service.py +248 -0
- app/cli/__init__.py +18 -0
- app/cli/__main__.py +30 -0
- app/cli/shared_services/__init__.py +42 -0
- app/cli/shared_services/config_ops.py +581 -0
- app/cli/shared_services/generation_ops.py +160 -0
- app/cli/shared_services/project_ops.py +288 -0
- app/cli/shell/__init__.py +48 -0
- app/cli/shell/commands/__init__.py +63 -0
- app/cli/shell/commands/ai/__init__.py +260 -0
- app/cli/shell/commands/ai/chat.py +331 -0
- app/cli/shell/commands/ai/delete.py +165 -0
- app/cli/shell/commands/ai/diff.py +70 -0
- app/cli/shell/commands/ai/display.py +428 -0
- app/cli/shell/commands/ai/executor.py +225 -0
- app/cli/shell/commands/ai/executor_utils.py +271 -0
- app/cli/shell/commands/ai/generate.py +285 -0
- app/cli/shell/commands/ai/interaction.py +262 -0
- app/cli/shell/commands/ai/migrate.py +281 -0
- app/cli/shell/commands/ai/resolver.py +224 -0
- app/cli/shell/commands/ai/status.py +101 -0
- app/cli/shell/commands/ai/switch.py +158 -0
- app/cli/shell/commands/ai/utils.py +52 -0
- app/cli/shell/commands/base.py +343 -0
- app/cli/shell/commands/config/__init__.py +131 -0
- app/cli/shell/commands/config/base.py +80 -0
- app/cli/shell/commands/config/check.py +246 -0
- app/cli/shell/commands/config/edit.py +148 -0
- app/cli/shell/commands/config/get.py +111 -0
- app/cli/shell/commands/config/init.py +123 -0
- app/cli/shell/commands/config/inspect.py +164 -0
- app/cli/shell/commands/config/list.py +104 -0
- app/cli/shell/commands/config/set.py +119 -0
- app/cli/shell/commands/config/show.py +150 -0
- app/cli/shell/commands/exit.py +67 -0
- app/cli/shell/commands/help.py +104 -0
- app/cli/shell/commands/infer_schema.py +156 -0
- app/cli/shell/commands/open.py +201 -0
- app/cli/shell/commands/project.py +215 -0
- app/cli/shell/commands/provider.py +619 -0
- app/cli/shell/commands/system.py +170 -0
- app/cli/shell/commands/validate.py +455 -0
- app/cli/shell/completer.py +231 -0
- app/cli/shell/config_storage.py +191 -0
- app/cli/shell/exceptions.py +90 -0
- app/cli/shell/formatter.py +371 -0
- app/cli/shell/interactive_menu.py +348 -0
- app/cli/shell/main.py +286 -0
- app/cli/shell/parser.py +266 -0
- app/cli/start.py +180 -0
- app/cli_main.py +52 -0
- app/mcp_server.py +311 -0
- app/shared/core/__init__.py +18 -0
- app/shared/core/app_version.py +59 -0
- app/shared/core/config/__init__.py +119 -0
- app/shared/core/config/server.py +193 -0
- app/shared/core/data_source/__init__.py +106 -0
- app/shared/core/data_source/loader.py +348 -0
- app/shared/core/data_source/loaders/__init__.py +213 -0
- app/shared/core/data_source/loaders/base.py +215 -0
- app/shared/core/data_source/loaders/converter.py +337 -0
- app/shared/core/data_source/loaders/csv_loader.py +252 -0
- app/shared/core/data_source/loaders/excel_loader.py +666 -0
- app/shared/core/data_source/loaders/extractor.py +268 -0
- app/shared/core/data_source/loaders/json_loader.py +354 -0
- app/shared/core/data_source/loaders/registry.py +88 -0
- app/shared/core/data_source/loaders/sql_loader.py +313 -0
- app/shared/core/data_source/loaders/strategies/__init__.py +100 -0
- app/shared/core/data_source/loaders/strategies/array_parser.py +277 -0
- app/shared/core/data_source/loaders/strategies/lines_parser.py +355 -0
- app/shared/core/data_source/loaders/strategies/object_parser.py +278 -0
- app/shared/core/data_source/schema_info.py +50 -0
- app/shared/core/data_source/specs/__init__.py +128 -0
- app/shared/core/data_source/specs/base.py +258 -0
- app/shared/core/data_source/specs/csv_source.py +137 -0
- app/shared/core/data_source/specs/excel_source.py +141 -0
- app/shared/core/data_source/specs/file_base.py +252 -0
- app/shared/core/data_source/specs/json_source.py +226 -0
- app/shared/core/data_source/specs/sql_source.py +161 -0
- app/shared/core/encoding.py +36 -0
- app/shared/core/io/__init__.py +24 -0
- app/shared/core/io/yaml.py +379 -0
- app/shared/core/manifest_schema/__init__.py +52 -0
- app/shared/core/manifest_schema/types.py +99 -0
- app/shared/core/manifest_schema/version.py +175 -0
- app/shared/core/patterns/__init__.py +24 -0
- app/shared/core/patterns/loader.py +134 -0
- app/shared/core/patterns/writer.py +244 -0
- app/shared/core/project/__init__.py +47 -0
- app/shared/core/project/constraint/builders/__init__.py +50 -0
- app/shared/core/project/constraint/builders/base.py +100 -0
- app/shared/core/project/constraint/builders/composite.py +77 -0
- app/shared/core/project/constraint/builders/conditional.py +67 -0
- app/shared/core/project/constraint/builders/foreign_key.py +53 -0
- app/shared/core/project/constraint/builders/registry.py +64 -0
- app/shared/core/project/constraint/builders/scripted.py +51 -0
- app/shared/core/project/constraint/builders/single_column.py +86 -0
- app/shared/core/project/constraint/builders/unique.py +53 -0
- app/shared/core/project/constraint/factory.py +170 -0
- app/shared/core/project/constraint/reader.py +214 -0
- app/shared/core/project/constraint/registry.py +233 -0
- app/shared/core/project/constraint/types/__init__.py +63 -0
- app/shared/core/project/constraint/types/constraint_file.py +261 -0
- app/shared/core/project/constraint/types/refs.py +460 -0
- app/shared/core/project/constraint/types.py +28 -0
- app/shared/core/project/constraint/writer.py +181 -0
- app/shared/core/project/loader/__init__.py +56 -0
- app/shared/core/project/loader/loader.py +30 -0
- app/shared/core/project/loader/loader_parts/config_inspector.py +137 -0
- app/shared/core/project/loader/loader_parts/embedded_constraints.py +224 -0
- app/shared/core/project/loader/loader_parts/file_loaders.py +58 -0
- app/shared/core/project/loader/loader_parts/inspection_ids.py +85 -0
- app/shared/core/project/loader/loader_parts/inspector_helpers.py +312 -0
- app/shared/core/project/loader/loader_parts/inspector_id_checks.py +274 -0
- app/shared/core/project/loader/loader_parts/inspector_reference_checks.py +690 -0
- app/shared/core/project/loader/loader_parts/inspector_uniqueness_checks.py +286 -0
- app/shared/core/project/loader/loader_parts/loading_error_messages.py +298 -0
- app/shared/core/project/loader/loader_parts/main.py +432 -0
- app/shared/core/project/loader/loader_parts/path_validation.py +117 -0
- app/shared/core/project/loader/loader_parts/runtime.py +125 -0
- app/shared/core/project/loader/types.py +246 -0
- app/shared/core/project/manifest/coverage.py +391 -0
- app/shared/core/project/manifest/reader.py +260 -0
- app/shared/core/project/manifest/types.py +91 -0
- app/shared/core/project/manifest/types_parts/__init__.py +60 -0
- app/shared/core/project/manifest/types_parts/constants.py +40 -0
- app/shared/core/project/manifest/types_parts/data_source.py +68 -0
- app/shared/core/project/manifest/types_parts/info.py +59 -0
- app/shared/core/project/manifest/types_parts/manifest.py +294 -0
- app/shared/core/project/manifest/types_parts/refs.py +153 -0
- app/shared/core/project/manifest/types_parts/settings.py +92 -0
- app/shared/core/project/manifest/types_parts/settings_file_processing.py +64 -0
- app/shared/core/project/manifest/types_parts/settings_script_security.py +78 -0
- app/shared/core/project/manifest/types_parts/settings_validation.py +83 -0
- app/shared/core/project/manifest/types_parts/template.py +66 -0
- app/shared/core/project/manifest/writer.py +262 -0
- app/shared/core/project/manual_data/__init__.py +27 -0
- app/shared/core/project/manual_data/types.py +75 -0
- app/shared/core/project/regex/reader.py +197 -0
- app/shared/core/project/regex/types.py +405 -0
- app/shared/core/project/regex/writer.py +123 -0
- app/shared/core/project/schema/reader.py +170 -0
- app/shared/core/project/schema/types.py +47 -0
- app/shared/core/project/schema/types_parts/__init__.py +36 -0
- app/shared/core/project/schema/types_parts/column.py +174 -0
- app/shared/core/project/schema/types_parts/column_utils.py +72 -0
- app/shared/core/project/schema/types_parts/constraint.py +165 -0
- app/shared/core/project/schema/types_parts/schema_id.py +66 -0
- app/shared/core/project/schema/types_parts/source.py +255 -0
- app/shared/core/project/schema/types_parts/source_options.py +347 -0
- app/shared/core/project/schema/types_parts/table.py +230 -0
- app/shared/core/project/schema/writer.py +139 -0
- app/shared/core/project/schema_ref_check.py +95 -0
- app/shared/core/project/template/__init__.py +27 -0
- app/shared/core/project/template/expander.py +263 -0
- app/shared/core/project/template/reader.py +120 -0
- app/shared/core/project/template/types.py +114 -0
- app/shared/core/project/transform/reader.py +76 -0
- app/shared/core/project/transform/types.py +116 -0
- app/shared/core/project/transform/writer.py +84 -0
- app/shared/core/pydantic_messages.py +59 -0
- app/shared/core/reporter/__init__.py +46 -0
- app/shared/core/reporter/reporter.py +220 -0
- app/shared/core/reporter/reporters/__init__.py +65 -0
- app/shared/core/reporter/reporters/base.py +188 -0
- app/shared/core/reporter/reporters/dingtalk_app_reporter.py +274 -0
- app/shared/core/reporter/reporters/email_reporter.py +271 -0
- app/shared/core/reporter/reporters/feishu_app_reporter.py +467 -0
- app/shared/core/reporter/reporters/local_file_reporter.py +208 -0
- app/shared/core/reporter/reporters/wecom_app_reporter.py +268 -0
- app/shared/core/utils/__init__.py +18 -0
- app/shared/core/utils/path_utils.py +60 -0
- app/shared/core/utils/regex_extract.py +113 -0
- app/shared/domain/__init__.py +114 -0
- app/shared/domain/constraints/__init__.py +76 -0
- app/shared/domain/constraints/allowed_values.py +211 -0
- app/shared/domain/constraints/base.py +141 -0
- app/shared/domain/constraints/charset.py +337 -0
- app/shared/domain/constraints/composite.py +174 -0
- app/shared/domain/constraints/condition_registry.py +130 -0
- app/shared/domain/constraints/conditional.py +629 -0
- app/shared/domain/constraints/date_logic.py +731 -0
- app/shared/domain/constraints/foreign_key.py +261 -0
- app/shared/domain/constraints/key_normalization.py +56 -0
- app/shared/domain/constraints/not_null.py +185 -0
- app/shared/domain/constraints/range.py +360 -0
- app/shared/domain/constraints/regex.py +218 -0
- app/shared/domain/constraints/scripted.py +276 -0
- app/shared/domain/constraints/unique.py +233 -0
- app/shared/domain/data_engine.py +384 -0
- app/shared/domain/data_types.py +113 -0
- app/shared/domain/data_types_parts/__init__.py +67 -0
- app/shared/domain/data_types_parts/base.py +204 -0
- app/shared/domain/data_types_parts/composite.py +250 -0
- app/shared/domain/data_types_parts/expression.py +202 -0
- app/shared/domain/data_types_parts/extracted.py +106 -0
- app/shared/domain/data_types_parts/json_types.py +234 -0
- app/shared/domain/data_types_parts/scalars.py +719 -0
- app/shared/domain/data_types_parts/sequence.py +129 -0
- app/shared/domain/dataset_schema.py +48 -0
- app/shared/domain/eval_sandbox.py +63 -0
- app/shared/domain/expression_system.py +366 -0
- app/shared/domain/regex_flags.py +53 -0
- app/shared/domain/schema/builder.py +223 -0
- app/shared/domain/schema/models.py +339 -0
- app/shared/domain/transforms/__init__.py +28 -0
- app/shared/domain/transforms/aggregate.py +141 -0
- app/shared/domain/transforms/base.py +171 -0
- app/shared/domain/transforms/cast_type.py +120 -0
- app/shared/domain/transforms/concat.py +103 -0
- app/shared/domain/transforms/conditional_assign.py +114 -0
- app/shared/domain/transforms/date_format.py +81 -0
- app/shared/domain/transforms/digits.py +77 -0
- app/shared/domain/transforms/drop_duplicates.py +94 -0
- app/shared/domain/transforms/fill_na.py +94 -0
- app/shared/domain/transforms/filter_rows.py +97 -0
- app/shared/domain/transforms/lookup.py +78 -0
- app/shared/domain/transforms/lower_case.py +70 -0
- app/shared/domain/transforms/map_value.py +90 -0
- app/shared/domain/transforms/math_expr.py +112 -0
- app/shared/domain/transforms/modulo.py +82 -0
- app/shared/domain/transforms/regex_extract.py +103 -0
- app/shared/domain/transforms/registry.py +95 -0
- app/shared/domain/transforms/replace.py +88 -0
- app/shared/domain/transforms/sort_rows.py +94 -0
- app/shared/domain/transforms/string_split.py +84 -0
- app/shared/domain/transforms/strip.py +73 -0
- app/shared/domain/transforms/substring.py +92 -0
- app/shared/domain/transforms/upper_case.py +70 -0
- app/shared/domain/transforms/weighted_sum.py +104 -0
- app/shared/domain/validation_constraints.py +69 -0
- app/shared/services/__init__.py +62 -0
- app/shared/services/ai/__init__.py +48 -0
- app/shared/services/ai/agent/__init__.py +40 -0
- app/shared/services/ai/agent/chat_tools/__init__.py +47 -0
- app/shared/services/ai/agent/chat_tools/apply_actions.py +592 -0
- app/shared/services/ai/agent/chat_tools/ask_user.py +237 -0
- app/shared/services/ai/agent/chat_tools/read_canvas.py +140 -0
- app/shared/services/ai/agent/chat_tools/read_project.py +160 -0
- app/shared/services/ai/agent/chat_tools/read_table.py +301 -0
- app/shared/services/ai/agent/chat_tools/schemas.py +105 -0
- app/shared/services/ai/agent/chat_tools/validate_table.py +131 -0
- app/shared/services/ai/agent/executor.py +554 -0
- app/shared/services/ai/agent/memory.py +256 -0
- app/shared/services/ai/agent/planner.py +282 -0
- app/shared/services/ai/agent/tool_registry.py +296 -0
- app/shared/services/ai/agent/tools/__init__.py +32 -0
- app/shared/services/ai/agent/tools/config_generate.py +133 -0
- app/shared/services/ai/agent/tools/config_refine.py +109 -0
- app/shared/services/ai/agent/tools/config_validate.py +403 -0
- app/shared/services/ai/agent/tools/merge_results.py +136 -0
- app/shared/services/ai/agent/tools/plan_chunks.py +79 -0
- app/shared/services/ai/agent/tools/script_parse.py +359 -0
- app/shared/services/ai/agent/types.py +138 -0
- app/shared/services/ai/chat_agent_runner.py +596 -0
- app/shared/services/ai/chat_orchestrator.py +725 -0
- app/shared/services/ai/failure_messages.py +65 -0
- app/shared/services/ai/job_storage.py +274 -0
- app/shared/services/ai/migrate_service.py +550 -0
- app/shared/services/ai/streaming/__init__.py +65 -0
- app/shared/services/ai/streaming/event_journal.py +197 -0
- app/shared/services/ai/streaming/orchestrator.py +262 -0
- app/shared/services/ai/streaming/pending_interaction_store.py +227 -0
- app/shared/services/ai/streaming/sse_response.py +148 -0
- app/shared/services/ai/streaming/types.py +73 -0
- app/shared/services/ai/types.py +213 -0
- app/shared/services/ai/utils.py +322 -0
- app/shared/services/diff/config_diff.py +320 -0
- app/shared/services/hardware.py +237 -0
- app/shared/services/llm/__init__.py +62 -0
- app/shared/services/llm/actions/__init__.py +42 -0
- app/shared/services/llm/actions/_canvas_validator.py +147 -0
- app/shared/services/llm/actions/_constraint_validator.py +401 -0
- app/shared/services/llm/actions/_regex_validator.py +85 -0
- app/shared/services/llm/actions/_schema_validator.py +82 -0
- app/shared/services/llm/actions/_settings_validator.py +76 -0
- app/shared/services/llm/actions/_transform_validator.py +78 -0
- app/shared/services/llm/actions/action_handlers.py +400 -0
- app/shared/services/llm/actions/action_parser.py +63 -0
- app/shared/services/llm/actions/action_processor.py +468 -0
- app/shared/services/llm/actions/action_validator.py +311 -0
- app/shared/services/llm/actions/diff_compute.py +195 -0
- app/shared/services/llm/actions/regex_handlers.py +284 -0
- app/shared/services/llm/actions/registry.py +336 -0
- app/shared/services/llm/actions/schema_handlers.py +335 -0
- app/shared/services/llm/actions/settings_handlers.py +180 -0
- app/shared/services/llm/actions/specs.py +259 -0
- app/shared/services/llm/actions/transform_handlers.py +279 -0
- app/shared/services/llm/actions/validation_types.py +147 -0
- app/shared/services/llm/cache/__init__.py +18 -0
- app/shared/services/llm/cache/response_cache.py +80 -0
- app/shared/services/llm/chat/__init__.py +35 -0
- app/shared/services/llm/chat/chat_system_prompt.py +561 -0
- app/shared/services/llm/chat/response_parser.py +296 -0
- app/shared/services/llm/config/__init__.py +45 -0
- app/shared/services/llm/config/crypto.py +135 -0
- app/shared/services/llm/config/loader.py +258 -0
- app/shared/services/llm/config/models.py +202 -0
- app/shared/services/llm/config/presets.py +128 -0
- app/shared/services/llm/config_generator.py +163 -0
- app/shared/services/llm/constraints/__init__.py +41 -0
- app/shared/services/llm/constraints/constraint_builder.py +177 -0
- app/shared/services/llm/constraints/constraint_deletion.py +84 -0
- app/shared/services/llm/constraints/constraint_id.py +174 -0
- app/shared/services/llm/constraints/frontend_instructions.py +333 -0
- app/shared/services/llm/constraints/inline_batch.py +279 -0
- app/shared/services/llm/discovery/__init__.py +35 -0
- app/shared/services/llm/discovery/scanner.py +181 -0
- app/shared/services/llm/generation/__init__.py +42 -0
- app/shared/services/llm/generation/agent_wiring.py +156 -0
- app/shared/services/llm/generation/config_builder.py +386 -0
- app/shared/services/llm/generation/errors.py +36 -0
- app/shared/services/llm/generation/existing_config.py +99 -0
- app/shared/services/llm/generation/profiler.py +279 -0
- app/shared/services/llm/generation/prompt_builder.py +199 -0
- app/shared/services/llm/generation/response_parser.py +144 -0
- app/shared/services/llm/generation/service.py +622 -0
- app/shared/services/llm/models.py +104 -0
- app/shared/services/llm/providers/__init__.py +48 -0
- app/shared/services/llm/providers/base.py +310 -0
- app/shared/services/llm/providers/cached_provider.py +90 -0
- app/shared/services/llm/providers/ollama.py +498 -0
- app/shared/services/llm/providers/openai.py +334 -0
- app/shared/services/llm/providers/registry.py +108 -0
- app/shared/services/llm/schema_resolver.py +141 -0
- app/shared/services/llm/suggestion_utils.py +219 -0
- app/shared/services/llm/validate_executor.py +163 -0
- app/shared/services/llm/yaml_io.py +406 -0
- app/shared/services/preview/__init__.py +18 -0
- app/shared/services/preview/loader.py +145 -0
- app/shared/services/preview/path_validation.py +138 -0
- app/shared/services/project_loader.py +57 -0
- app/shared/services/schema_inference.py +260 -0
- app/shared/services/schema_runtime_builder.py +120 -0
- app/shared/services/validation/__init__.py +52 -0
- app/shared/services/validation/chunked_loader.py +509 -0
- app/shared/services/validation/dag/__init__.py +35 -0
- app/shared/services/validation/dag/builder.py +141 -0
- app/shared/services/validation/dag/executor.py +225 -0
- app/shared/services/validation/dag/sorter.py +77 -0
- app/shared/services/validation/data_loader.py +231 -0
- app/shared/services/validation/engine.py +446 -0
- app/shared/services/validation/executor.py +1030 -0
- app/shared/services/validation/extractors.py +266 -0
- app/shared/services/validation/history.py +209 -0
- app/shared/services/validation/json_payload.py +136 -0
- app/shared/services/validation/loader.py +152 -0
- app/shared/services/validation/memory_monitor.py +179 -0
- app/shared/services/validation/postprocess.py +208 -0
- app/shared/services/validation/progress.py +64 -0
- app/shared/services/validation/report_export.py +222 -0
- app/shared/services/validation/resolver.py +180 -0
- app/shared/services/validation/service.py +418 -0
- app/shared/services/validation/types.py +238 -0
- app/shared/services/validation/validators/__init__.py +36 -0
- app/shared/services/validation/validators/adapter.py +182 -0
- app/shared/services/validation/validators/base.py +332 -0
- app/shared/services/validation/validators/composite.py +233 -0
- app/shared/services/validation/validators/date_logic.py +276 -0
- app/start_server.py +133 -0
- precis_cli-0.1.3.dist-info/METADATA +180 -0
- precis_cli-0.1.3.dist-info/RECORD +442 -0
- precis_cli-0.1.3.dist-info/WHEEL +5 -0
- precis_cli-0.1.3.dist-info/entry_points.txt +4 -0
- precis_cli-0.1.3.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,666 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2026 Precis Team
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
6
|
+
# you may not use this file except in compliance with the License.
|
|
7
|
+
# You may obtain a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
13
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
14
|
+
# See the License for the specific language governing permissions and
|
|
15
|
+
# limitations under the License.
|
|
16
|
+
"""
|
|
17
|
+
@fileoverview Excel 数据源加载器模块
|
|
18
|
+
|
|
19
|
+
功能概述:
|
|
20
|
+
- 加载 .xlsx / .xls 格式 Excel 文件为 pandas DataFrame
|
|
21
|
+
- 支持单 sheet 加载(load)和多 sheet 批量加载(load_multi_sheet)
|
|
22
|
+
- 空文件提前检查,给出清晰的 DataLoadError(B13)
|
|
23
|
+
- header_row 配置错误检测:若 header_row 之后无数据行则输出警告(B10)
|
|
24
|
+
- 使用 openpyxl 前向填充合并单元格,消除 NotNull/Unique 假阳性(B7)
|
|
25
|
+
- 支持 dtype_inference、skip_rows、nrows 等参数(B9)
|
|
26
|
+
|
|
27
|
+
架构设计:
|
|
28
|
+
- 继承 DataSourceLoader[ExcelSourceSpec],通过注册表自动发现
|
|
29
|
+
- load() 针对单 sheet,返回单个 DataFrame
|
|
30
|
+
- load_multi_sheet() 针对项目多表场景,返回 {schema_id: DataFrame}
|
|
31
|
+
- _apply_merged_cell_fill() 为 openpyxl 级别的合并单元格填充逻辑
|
|
32
|
+
|
|
33
|
+
输入示例:
|
|
34
|
+
spec = ExcelSourceSpec(
|
|
35
|
+
path="data/users.xlsx",
|
|
36
|
+
sheet="Sheet1",
|
|
37
|
+
header_row=0,
|
|
38
|
+
dtype_inference=True
|
|
39
|
+
)
|
|
40
|
+
loader = ExcelLoader(spec)
|
|
41
|
+
|
|
42
|
+
输出示例:
|
|
43
|
+
df = loader.load()
|
|
44
|
+
# 返回 pandas.DataFrame,合并单元格已按 Excel 显示值填充
|
|
45
|
+
# 若 sheet 不存在则抛出 DataLoadError(B8)
|
|
46
|
+
"""
|
|
47
|
+
|
|
48
|
+
from __future__ import annotations
|
|
49
|
+
|
|
50
|
+
import logging
|
|
51
|
+
from pathlib import Path
|
|
52
|
+
from typing import Any
|
|
53
|
+
|
|
54
|
+
import pandas as pd
|
|
55
|
+
|
|
56
|
+
from ..specs.excel_source import ExcelSourceSpec
|
|
57
|
+
from .base import DataLoadError, DataSourceLoader
|
|
58
|
+
from .registry import register_loader
|
|
59
|
+
|
|
60
|
+
logger = logging.getLogger(__name__)
|
|
61
|
+
|
|
62
|
+
# 扩展名 → 唯一可用引擎:.xls 仅 xlrd 支持、.xlsx/.xlsm 仅 openpyxl 支持
|
|
63
|
+
# (xlrd>=2.0 已移除 .xlsx 支持,openpyxl 从不支持 .xls)
|
|
64
|
+
_ENGINE_BY_SUFFIX = {".xls": "xlrd", ".xlsx": "openpyxl", ".xlsm": "openpyxl"}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def resolve_excel_engine(file_path: str) -> str:
|
|
68
|
+
"""按文件扩展名解析 Excel 读取引擎(.xls→xlrd、.xlsx/.xlsm→openpyxl)。
|
|
69
|
+
|
|
70
|
+
供仅需引擎名的场景(如读取工作表列表)使用,与 ExcelLoader 的
|
|
71
|
+
_effective_engine 共用同一份映射,保证两处引擎选择永不漂移。
|
|
72
|
+
"""
|
|
73
|
+
return _ENGINE_BY_SUFFIX.get(Path(file_path).suffix.lower(), "openpyxl")
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def get_excel_sheet_names(file_path: str) -> list[str]:
|
|
77
|
+
"""读取 Excel 文件的全部工作表名,引擎按扩展名自适应,句柄确保关闭。
|
|
78
|
+
|
|
79
|
+
供预览等只需工作表列表的场景使用。预览路由曾各自硬编码 openpyxl 导致
|
|
80
|
+
.xls 必然 500,且 ExcelFile 打开后未 close 使 Windows 下数据文件被锁定。
|
|
81
|
+
"""
|
|
82
|
+
excel_file = pd.ExcelFile(file_path, engine=resolve_excel_engine(file_path))
|
|
83
|
+
try:
|
|
84
|
+
return list(excel_file.sheet_names)
|
|
85
|
+
finally:
|
|
86
|
+
excel_file.close()
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# OOXML 命名空间:xlsx 内部 XML 的主命名空间、officeDocument 关系 id 属性、
|
|
90
|
+
# 以及 *_rels 关系文件元素所属的 package 关系命名空间(注意 .iter() 不支持
|
|
91
|
+
# "{*}" 通配——find/findall 的路径通配语法在标签匹配上是字面量,恒不命中)
|
|
92
|
+
_XLSX_MAIN_NS = "{http://schemas.openxmlformats.org/spreadsheetml/2006/main}"
|
|
93
|
+
_XLSX_REL_ID = "{http://schemas.openxmlformats.org/officeDocument/2006/relationships}id"
|
|
94
|
+
_XLSX_PKG_REL_NS = "{http://schemas.openxmlformats.org/package/2006/relationships}"
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def _xlsx_col_letters_to_index(letters: str) -> int:
|
|
98
|
+
"""Excel 列字母(A/B/.../AA)转 1-based 列号。"""
|
|
99
|
+
index = 0
|
|
100
|
+
for ch in letters.upper():
|
|
101
|
+
index = index * 26 + (ord(ch) - ord("A") + 1)
|
|
102
|
+
return index
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _xlsx_parse_cell_ref(ref: str) -> tuple[int, int]:
|
|
106
|
+
"""单元格引用(如 B5)转 (row, col),均 1-based。"""
|
|
107
|
+
i = 0
|
|
108
|
+
while i < len(ref) and ref[i].isalpha():
|
|
109
|
+
i += 1
|
|
110
|
+
return int(ref[i:]), _xlsx_col_letters_to_index(ref[:i])
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _xlsx_cell_value(elem: Any, shared: list[str] | None) -> Any:
|
|
114
|
+
"""从 <c> 元素解析单元格值,对齐 openpyxl data_only=True 的缓存值语义。"""
|
|
115
|
+
cell_type = elem.get("t")
|
|
116
|
+
if cell_type == "inlineStr":
|
|
117
|
+
return "".join(t.text or "" for t in elem.iter(f"{_XLSX_MAIN_NS}t"))
|
|
118
|
+
v_elem = elem.find(f"{_XLSX_MAIN_NS}v")
|
|
119
|
+
if v_elem is None or v_elem.text is None:
|
|
120
|
+
return None
|
|
121
|
+
raw = v_elem.text
|
|
122
|
+
if cell_type == "s":
|
|
123
|
+
# 共享字符串:v 为 sharedStrings.xml 中的下标
|
|
124
|
+
if shared is None:
|
|
125
|
+
return None
|
|
126
|
+
idx = int(raw)
|
|
127
|
+
return shared[idx] if 0 <= idx < len(shared) else None
|
|
128
|
+
if cell_type == "b":
|
|
129
|
+
return raw == "1"
|
|
130
|
+
if cell_type == "str":
|
|
131
|
+
return raw # 公式字符串结果
|
|
132
|
+
# 数值(默认/公式缓存数值结果):int→float→原样回退
|
|
133
|
+
try:
|
|
134
|
+
return int(raw)
|
|
135
|
+
except ValueError:
|
|
136
|
+
try:
|
|
137
|
+
return float(raw)
|
|
138
|
+
except ValueError:
|
|
139
|
+
return raw
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def read_merged_ranges_from_xlsx(
|
|
143
|
+
file_path: str | Path, sheet_name: str
|
|
144
|
+
) -> list[tuple[int, int, int, int, list[Any]]] | None:
|
|
145
|
+
"""流式读取 xlsx 目标 sheet 的合并单元格区域及各区域首行值(§1.27 分块路径专用)。
|
|
146
|
+
|
|
147
|
+
返回 (min_row, min_col, max_row, max_col, values) 五元组列表,values 为区域首行
|
|
148
|
+
min_col..max_col 各列的值(供跨块悬挂区域直接赋值续填);sheet 不存在返回 None;
|
|
149
|
+
无合并区域返回 []。
|
|
150
|
+
|
|
151
|
+
为何不走 openpyxl:read_only 模式的 ReadOnlyWorksheet 不解析 merged_cells(无该
|
|
152
|
+
属性),普通模式则把全部 sheet 的 Cell 对象图整体物化——分块路径本是为 >500MB
|
|
153
|
+
大文件省内存而设,普通模式峰值内存约为 read_only 的 120 倍(0.5MB/15 万格实测
|
|
154
|
+
55MB vs ≈0MB)。故按 OOXML 结构直接 zipfile+ElementTree 流式解析:workbook.xml
|
|
155
|
+
定位目标 sheet → 第一遍扫描 <mergeCells> 取区域 → 第二遍流式扫描 <sheetData>
|
|
156
|
+
仅物化各区域首行的取值。共享字符串表按 openpyxl read_only 同口径全量载入
|
|
157
|
+
(上界为去重字符串数)。.xls(非 zip 容器)会抛 BadZipFile,由调用方降级处理。
|
|
158
|
+
"""
|
|
159
|
+
import zipfile
|
|
160
|
+
from xml.etree import ElementTree as ET
|
|
161
|
+
|
|
162
|
+
with zipfile.ZipFile(file_path) as zf:
|
|
163
|
+
# 1) sheet 名 → 工作表 XML 条目路径(workbook.xml 的 r:id + 其关系文件)
|
|
164
|
+
rid: str | None = None
|
|
165
|
+
wb_root = ET.fromstring(zf.read("xl/workbook.xml"))
|
|
166
|
+
for sheet in wb_root.iter(f"{_XLSX_MAIN_NS}sheet"):
|
|
167
|
+
if sheet.get("name") == sheet_name:
|
|
168
|
+
rid = sheet.get(_XLSX_REL_ID)
|
|
169
|
+
break
|
|
170
|
+
if not rid:
|
|
171
|
+
return None
|
|
172
|
+
target: str | None = None
|
|
173
|
+
rels_root = ET.fromstring(zf.read("xl/_rels/workbook.xml.rels"))
|
|
174
|
+
for rel in rels_root.iter(f"{_XLSX_PKG_REL_NS}Relationship"):
|
|
175
|
+
if rel.get("Id") == rid:
|
|
176
|
+
target = rel.get("Target")
|
|
177
|
+
break
|
|
178
|
+
if not target:
|
|
179
|
+
return None
|
|
180
|
+
entry = target.lstrip("/") if target.startswith("/") else f"xl/{target}"
|
|
181
|
+
|
|
182
|
+
# 2) 第一遍:mergeCells 区域引用(行元素即扫即弃,内存有界)
|
|
183
|
+
ranges: list[tuple[int, int, int, int]] = []
|
|
184
|
+
with zf.open(entry) as stream:
|
|
185
|
+
for _event, elem in ET.iterparse(stream, events=("end",)):
|
|
186
|
+
if elem.tag == f"{_XLSX_MAIN_NS}mergeCell":
|
|
187
|
+
ref = elem.get("ref") or ""
|
|
188
|
+
if ":" in ref:
|
|
189
|
+
start, end = ref.split(":", 1)
|
|
190
|
+
r1, c1 = _xlsx_parse_cell_ref(start)
|
|
191
|
+
r2, c2 = _xlsx_parse_cell_ref(end)
|
|
192
|
+
ranges.append((min(r1, r2), min(c1, c2), max(r1, r2), max(c1, c2)))
|
|
193
|
+
elem.clear()
|
|
194
|
+
elif elem.tag == f"{_XLSX_MAIN_NS}row":
|
|
195
|
+
elem.clear()
|
|
196
|
+
if not ranges:
|
|
197
|
+
return []
|
|
198
|
+
|
|
199
|
+
# 3) 共享字符串表(存在时全量载入;openpyxl read_only 取值同口径)
|
|
200
|
+
shared: list[str] | None = None
|
|
201
|
+
if "xl/sharedStrings.xml" in zf.namelist():
|
|
202
|
+
shared = []
|
|
203
|
+
with zf.open("xl/sharedStrings.xml") as stream:
|
|
204
|
+
for _event, si in ET.iterparse(stream, events=("end",)):
|
|
205
|
+
if si.tag == f"{_XLSX_MAIN_NS}si":
|
|
206
|
+
shared.append("".join(t.text or "" for t in si.iter(f"{_XLSX_MAIN_NS}t")))
|
|
207
|
+
si.clear()
|
|
208
|
+
|
|
209
|
+
# 4) 第二遍:仅物化各区域首行的单元格值
|
|
210
|
+
needed_rows = {r[0] for r in ranges}
|
|
211
|
+
row_values: dict[int, dict[int, Any]] = {}
|
|
212
|
+
with zf.open(entry) as stream:
|
|
213
|
+
for _event, row in ET.iterparse(stream, events=("end",)):
|
|
214
|
+
if row.tag != f"{_XLSX_MAIN_NS}row":
|
|
215
|
+
continue
|
|
216
|
+
row_num = row.get("r")
|
|
217
|
+
if row_num is not None and int(row_num) in needed_rows:
|
|
218
|
+
cells: dict[int, Any] = {}
|
|
219
|
+
fallback_col = 0
|
|
220
|
+
for cell in row:
|
|
221
|
+
cell_ref = cell.get("r")
|
|
222
|
+
col = _xlsx_parse_cell_ref(cell_ref)[1] if cell_ref else fallback_col + 1
|
|
223
|
+
fallback_col = col
|
|
224
|
+
cells[col] = _xlsx_cell_value(cell, shared)
|
|
225
|
+
row_values[int(row_num)] = cells
|
|
226
|
+
row.clear()
|
|
227
|
+
|
|
228
|
+
result: list[tuple[int, int, int, int, list[Any]]] = []
|
|
229
|
+
for min_row, min_col, max_row, max_col in ranges:
|
|
230
|
+
cells = row_values.get(min_row, {})
|
|
231
|
+
values = [cells.get(c) for c in range(min_col, max_col + 1)]
|
|
232
|
+
result.append((min_row, min_col, max_row, max_col, values))
|
|
233
|
+
return result
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def apply_merged_ranges_fill(
|
|
237
|
+
df: pd.DataFrame,
|
|
238
|
+
merged_ranges: Any,
|
|
239
|
+
*,
|
|
240
|
+
header_row: int = 0,
|
|
241
|
+
skip_rows: int = 0,
|
|
242
|
+
global_row_offset: int = 0,
|
|
243
|
+
) -> tuple[pd.DataFrame, int, int]:
|
|
244
|
+
"""对合并单元格区域做列向前向填充(B7 核心,标准路径与分块路径单一事实源)。
|
|
245
|
+
|
|
246
|
+
参数:
|
|
247
|
+
df: 待填充的 DataFrame(标准路径=整表;分块路径=单块,此时传 global_row_offset)
|
|
248
|
+
merged_ranges: openpyxl merged_cells.ranges 对象(标准路径),或
|
|
249
|
+
(min_row, min_col, max_row, max_col, values) 五元组列表(分块路径——values 为
|
|
250
|
+
区域首行各列的值,供跨块悬挂区域直接赋值,合并单元格区域内所有格本就同值)
|
|
251
|
+
header_row: 表头行号(0-based,不含 skip_rows)
|
|
252
|
+
skip_rows: 表头前跳过的行数(openpyxl→df 行号换算基准;分块路径读取不带 skip_rows,保持 0)
|
|
253
|
+
global_row_offset: 本 df 首行对应的全局数据行号(0-based;标准路径整表为 0)
|
|
254
|
+
|
|
255
|
+
返回:
|
|
256
|
+
(填充后的 df 副本, 填充的区域数, 无法填充的悬挂区域数)
|
|
257
|
+
"""
|
|
258
|
+
df_filled = df.copy()
|
|
259
|
+
df_start_global = global_row_offset
|
|
260
|
+
filled_count = 0
|
|
261
|
+
skipped_cross = 0
|
|
262
|
+
for merged in merged_ranges:
|
|
263
|
+
if isinstance(merged, tuple):
|
|
264
|
+
min_row, min_col, max_row, max_col, values = (
|
|
265
|
+
merged[0],
|
|
266
|
+
merged[1],
|
|
267
|
+
merged[2],
|
|
268
|
+
merged[3],
|
|
269
|
+
merged[4] if len(merged) > 4 else None,
|
|
270
|
+
)
|
|
271
|
+
else:
|
|
272
|
+
min_row, min_col, max_row, max_col = (
|
|
273
|
+
merged.min_row,
|
|
274
|
+
merged.min_col,
|
|
275
|
+
merged.max_row,
|
|
276
|
+
merged.max_col,
|
|
277
|
+
)
|
|
278
|
+
values = None
|
|
279
|
+
# openpyxl 是 1-based;pandas 读取时先跳过 skip_rows 行再把第 header_row
|
|
280
|
+
# 行作为表头,因此全局数据行 0 对应 Excel 第 (skip_rows + header_row + 2) 行。
|
|
281
|
+
start_global = min_row - header_row - skip_rows - 2
|
|
282
|
+
end_global = max_row - header_row - skip_rows - 2
|
|
283
|
+
start_df_col = min_col - 1
|
|
284
|
+
end_df_col = max_col - 1
|
|
285
|
+
|
|
286
|
+
if start_df_col < 0:
|
|
287
|
+
continue
|
|
288
|
+
# 区域在本 df 范围之前(分块场景:整个区域属于更早的块)
|
|
289
|
+
if end_global < df_start_global:
|
|
290
|
+
continue
|
|
291
|
+
# 区域首行不在本 df 内(跨块悬挂):携带首行值时直接赋值(区域内所有格同值),
|
|
292
|
+
# 否则跳过并计数(标准路径整表调用不会走到这里)
|
|
293
|
+
if start_global < df_start_global:
|
|
294
|
+
if end_global >= df_start_global:
|
|
295
|
+
if values is not None:
|
|
296
|
+
seg_end = min(end_global - df_start_global, len(df_filled) - 1)
|
|
297
|
+
actual_end_col = min(end_df_col, len(df_filled.columns) - 1)
|
|
298
|
+
if seg_end >= 0 and actual_end_col >= start_df_col:
|
|
299
|
+
for k, col_idx in enumerate(range(start_df_col, actual_end_col + 1)):
|
|
300
|
+
if k < len(values) and values[k] is not None:
|
|
301
|
+
df_filled.iloc[0 : seg_end + 1, col_idx] = values[k]
|
|
302
|
+
filled_count += 1
|
|
303
|
+
else:
|
|
304
|
+
skipped_cross += 1
|
|
305
|
+
continue
|
|
306
|
+
# 区域首行在本 df 内:换算为 df 内行号
|
|
307
|
+
start_df_row = start_global - df_start_global
|
|
308
|
+
if start_df_row >= len(df_filled):
|
|
309
|
+
continue
|
|
310
|
+
|
|
311
|
+
# 限定区域边界(防止越界;分块场景区域尾部超出本块属正常——下一块经悬挂赋值续填)
|
|
312
|
+
actual_end_row = min(end_global - df_start_global, len(df_filled) - 1)
|
|
313
|
+
actual_end_col = min(end_df_col, len(df_filled.columns) - 1)
|
|
314
|
+
if actual_end_row < start_df_row or actual_end_col < start_df_col:
|
|
315
|
+
continue
|
|
316
|
+
|
|
317
|
+
# 按列向量化前向填充:区域内 NaN 被区域起点方向的最近非空值填充,
|
|
318
|
+
# 已有非空值不动。ffill 从区域起点开始,不跨越区域边界。
|
|
319
|
+
for col_idx in range(start_df_col, actual_end_col + 1):
|
|
320
|
+
region = df_filled.iloc[start_df_row : actual_end_row + 1, col_idx]
|
|
321
|
+
df_filled.iloc[start_df_row : actual_end_row + 1, col_idx] = region.ffill()
|
|
322
|
+
filled_count += 1
|
|
323
|
+
return df_filled, filled_count, skipped_cross
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
@register_loader("excel")
|
|
327
|
+
class ExcelLoader(DataSourceLoader[ExcelSourceSpec]):
|
|
328
|
+
"""
|
|
329
|
+
@classdesc Excel 文件加载器
|
|
330
|
+
|
|
331
|
+
支持 .xlsx 和 .xls 格式的 Excel 文件。
|
|
332
|
+
使用 pandas.read_excel 进行读取。
|
|
333
|
+
|
|
334
|
+
支持两种模式:
|
|
335
|
+
- 单 sheet 加载:通过 load() 返回单个 DataFrame
|
|
336
|
+
- 多 sheet 批量加载:通过 load_multi_sheet() 返回 {sheet_name: DataFrame}
|
|
337
|
+
"""
|
|
338
|
+
|
|
339
|
+
spec_class = ExcelSourceSpec
|
|
340
|
+
|
|
341
|
+
def _effective_engine(self) -> str:
|
|
342
|
+
"""
|
|
343
|
+
@methoddesc 按文件扩展名解析实际读取引擎,与 spec.engine 不符时以扩展名为准
|
|
344
|
+
|
|
345
|
+
业务用途:
|
|
346
|
+
- spec.engine 默认 openpyxl,但 .xls 文件 openpyxl 无法读取(抛
|
|
347
|
+
InvalidFileException/BadZipFile)——此前"宣告支持 .xls 却必炸"的根因之一。
|
|
348
|
+
引擎与扩展名唯一对应,冲突时按扩展名纠正并提示,避免必然失败的组合。
|
|
349
|
+
- 无扩展名(或不识别的扩展名)时尊重 spec.engine 原值。
|
|
350
|
+
"""
|
|
351
|
+
expected = _ENGINE_BY_SUFFIX.get(Path(self.spec.path).suffix.lower())
|
|
352
|
+
if expected and expected != self.spec.engine:
|
|
353
|
+
logger.info(
|
|
354
|
+
"引擎配置(%s)与文件扩展名不符,已按扩展名使用 %s: %s",
|
|
355
|
+
self.spec.engine,
|
|
356
|
+
expected,
|
|
357
|
+
self.spec.path,
|
|
358
|
+
)
|
|
359
|
+
return expected
|
|
360
|
+
return self.spec.engine
|
|
361
|
+
|
|
362
|
+
def load(self) -> pd.DataFrame:
|
|
363
|
+
"""
|
|
364
|
+
@methoddesc 加载 Excel 文件并返回 DataFrame。
|
|
365
|
+
|
|
366
|
+
根据 spec 中的 sheet 名称或索引读取指定工作表,
|
|
367
|
+
支持 header_row、skip_rows、nrows 等配置。
|
|
368
|
+
读取后会应用合并单元格前向填充(仅 openpyxl 引擎)。
|
|
369
|
+
|
|
370
|
+
Returns:
|
|
371
|
+
加载的 DataFrame
|
|
372
|
+
|
|
373
|
+
Raises:
|
|
374
|
+
DataLoadError: 文件不存在、为空、加载失败或 sheet 不存在时抛出
|
|
375
|
+
|
|
376
|
+
示例:
|
|
377
|
+
>>> spec = ExcelSourceSpec(path="data.xlsx", sheet="Sheet1", header_row=0)
|
|
378
|
+
>>> loader = ExcelLoader(spec)
|
|
379
|
+
>>> df = loader.load()
|
|
380
|
+
"""
|
|
381
|
+
try:
|
|
382
|
+
# 空文件提前检查,给出清晰错误(B13)
|
|
383
|
+
path = Path(self.spec.path)
|
|
384
|
+
if path.exists() and path.stat().st_size == 0:
|
|
385
|
+
raise DataLoadError(f"Excel 文件为空: {self.spec.path}", self.spec)
|
|
386
|
+
|
|
387
|
+
read_kwargs = self._build_read_kwargs()
|
|
388
|
+
|
|
389
|
+
df = pd.read_excel(self.spec.path, **read_kwargs)
|
|
390
|
+
|
|
391
|
+
if not self.spec.header_enabled:
|
|
392
|
+
df.columns = [f"col_{i}" for i in range(len(df.columns))]
|
|
393
|
+
|
|
394
|
+
# header_row 配置错误导致空数据时给出警告(B10)
|
|
395
|
+
if len(df) == 0 and self.spec.header_row > 0:
|
|
396
|
+
logger.warning(
|
|
397
|
+
f"Excel 表 '{self.spec.sheet or self.spec.sheet_index}' "
|
|
398
|
+
f"在 header_row={self.spec.header_row} 之后没有数据行,"
|
|
399
|
+
f"请检查 header_row 配置是否正确"
|
|
400
|
+
)
|
|
401
|
+
|
|
402
|
+
df = self._apply_merged_cell_fill(df, self.spec.sheet, self.spec.header_row, self.spec.skip_rows)
|
|
403
|
+
return df
|
|
404
|
+
|
|
405
|
+
except FileNotFoundError as e:
|
|
406
|
+
raise DataLoadError(f"文件不存在: {self.spec.path}", self.spec, e)
|
|
407
|
+
except DataLoadError:
|
|
408
|
+
raise
|
|
409
|
+
except Exception as e:
|
|
410
|
+
raise DataLoadError(f"Excel 加载失败: {e}", self.spec, e)
|
|
411
|
+
|
|
412
|
+
def load_multi_sheet(
|
|
413
|
+
self,
|
|
414
|
+
sheet_configs: dict[str, dict[str, Any]],
|
|
415
|
+
) -> dict[str, pd.DataFrame]:
|
|
416
|
+
"""
|
|
417
|
+
@methoddesc 批量加载多个 sheet。
|
|
418
|
+
|
|
419
|
+
:param sheet_configs: 字典,键为 schema_id,值包含:
|
|
420
|
+
- sheet_name: sheet 名称
|
|
421
|
+
- header_row: 表头行号
|
|
422
|
+
:return: 字典,键为 schema_id,值为对应的 DataFrame
|
|
423
|
+
"""
|
|
424
|
+
if not sheet_configs:
|
|
425
|
+
return {}
|
|
426
|
+
|
|
427
|
+
sheet_names = list({cfg["sheet_name"] for cfg in sheet_configs.values() if cfg.get("sheet_name")})
|
|
428
|
+
|
|
429
|
+
if not sheet_names:
|
|
430
|
+
return {}
|
|
431
|
+
|
|
432
|
+
try:
|
|
433
|
+
loaded_sheets: dict = pd.read_excel(
|
|
434
|
+
self.spec.path,
|
|
435
|
+
sheet_name=sheet_names,
|
|
436
|
+
header=None,
|
|
437
|
+
engine=self._effective_engine(),
|
|
438
|
+
**self._engine_kwargs(),
|
|
439
|
+
)
|
|
440
|
+
|
|
441
|
+
results: dict[str, pd.DataFrame] = {}
|
|
442
|
+
for schema_id, cfg in sheet_configs.items():
|
|
443
|
+
sheet_name = cfg.get("sheet_name")
|
|
444
|
+
header_row = cfg.get("header_row", 0)
|
|
445
|
+
skip_rows = cfg.get("skip_rows", 0)
|
|
446
|
+
nrows = cfg.get("nrows")
|
|
447
|
+
dtype_inference = cfg.get("dtype_inference", True)
|
|
448
|
+
|
|
449
|
+
if not sheet_name:
|
|
450
|
+
continue
|
|
451
|
+
# 缺失 sheet 时显式报错,避免静默跳过导致假阴性(B8)
|
|
452
|
+
if sheet_name not in loaded_sheets:
|
|
453
|
+
raise DataLoadError(
|
|
454
|
+
f"Sheet '{sheet_name}' 不存在于文件 {self.spec.path} 中",
|
|
455
|
+
self.spec,
|
|
456
|
+
)
|
|
457
|
+
|
|
458
|
+
df = loaded_sheets[sheet_name]
|
|
459
|
+
effective_header = header_row + skip_rows
|
|
460
|
+
df.columns = df.iloc[effective_header]
|
|
461
|
+
df = df.drop(index=range(effective_header + 1)).reset_index(drop=True)
|
|
462
|
+
|
|
463
|
+
# 应用 nrows 限制(B9)
|
|
464
|
+
if nrows is not None:
|
|
465
|
+
df = df.head(nrows)
|
|
466
|
+
|
|
467
|
+
# 应用 dtype_inference(B9)
|
|
468
|
+
if not dtype_inference:
|
|
469
|
+
df = df.astype(str)
|
|
470
|
+
|
|
471
|
+
# header_row 配置错误导致空数据时给出警告(B10)
|
|
472
|
+
if len(df) == 0:
|
|
473
|
+
logger.warning(
|
|
474
|
+
f"Sheet '{sheet_name}' 在 header_row={header_row} 之后没有数据行,"
|
|
475
|
+
f"请检查 header_row 配置是否正确"
|
|
476
|
+
)
|
|
477
|
+
df = self._apply_merged_cell_fill(df, sheet_name, effective_header)
|
|
478
|
+
results[schema_id] = df
|
|
479
|
+
|
|
480
|
+
return results
|
|
481
|
+
|
|
482
|
+
except FileNotFoundError as e:
|
|
483
|
+
raise DataLoadError(f"文件不存在: {self.spec.path}", self.spec, e)
|
|
484
|
+
except Exception as e:
|
|
485
|
+
raise DataLoadError(f"Excel 多 sheet 加载失败: {e}", self.spec, e)
|
|
486
|
+
|
|
487
|
+
def _engine_kwargs(self) -> dict[str, Any]:
|
|
488
|
+
"""@methoddesc 构造 pandas.read_excel 的 engine_kwargs。
|
|
489
|
+
|
|
490
|
+
data_only=True 让 openpyxl 读取公式计算结果而非公式本身。
|
|
491
|
+
仅 openpyxl 引擎支持该参数——xlrd(.xls)的 open_workbook 没有
|
|
492
|
+
data_only 形参,pandas 原样转发会 TypeError,导致 .xls 必炸。
|
|
493
|
+
因此 xlrd 分支不传任何 engine_kwargs。
|
|
494
|
+
"""
|
|
495
|
+
if self._effective_engine() == "openpyxl":
|
|
496
|
+
return {"engine_kwargs": {"data_only": True}}
|
|
497
|
+
return {}
|
|
498
|
+
|
|
499
|
+
def _build_read_kwargs(self) -> dict[str, Any]:
|
|
500
|
+
"""
|
|
501
|
+
@methoddesc 构造 pandas.read_excel 的参数字典
|
|
502
|
+
|
|
503
|
+
业务用途:
|
|
504
|
+
- 根据 self.spec 配置(header 行、engine、dtype、sheet 索引、skip_rows、nrows)组装参数
|
|
505
|
+
- engine_kwargs.data_only=True 用于读取公式结果而非公式本身
|
|
506
|
+
|
|
507
|
+
返回:
|
|
508
|
+
可直接传入 pd.read_excel 的参数字典
|
|
509
|
+
"""
|
|
510
|
+
read_kwargs: dict[str, Any] = {
|
|
511
|
+
"header": self.spec.header_row if self.spec.header_enabled else None,
|
|
512
|
+
"engine": self._effective_engine(),
|
|
513
|
+
"dtype": None if self.spec.dtype_inference else str,
|
|
514
|
+
**self._engine_kwargs(),
|
|
515
|
+
}
|
|
516
|
+
|
|
517
|
+
if self.spec.sheet:
|
|
518
|
+
read_kwargs["sheet_name"] = self.spec.sheet
|
|
519
|
+
else:
|
|
520
|
+
read_kwargs["sheet_name"] = self.spec.sheet_index
|
|
521
|
+
|
|
522
|
+
if self.spec.skip_rows > 0:
|
|
523
|
+
read_kwargs["skiprows"] = self.spec.skip_rows
|
|
524
|
+
if self.spec.nrows:
|
|
525
|
+
read_kwargs["nrows"] = self.spec.nrows
|
|
526
|
+
|
|
527
|
+
return read_kwargs
|
|
528
|
+
|
|
529
|
+
def _apply_merged_cell_fill(
|
|
530
|
+
self,
|
|
531
|
+
df: pd.DataFrame,
|
|
532
|
+
sheet_name: str | None = None,
|
|
533
|
+
header_row: int = 0,
|
|
534
|
+
skip_rows: int = 0,
|
|
535
|
+
) -> pd.DataFrame:
|
|
536
|
+
"""@methoddesc 对合并单元格进行前向填充,避免 NotNull/Unique 误报(B7)。
|
|
537
|
+
|
|
538
|
+
使用 openpyxl 检测合并单元格区域,然后仅对区域内的 NaN 进行填充。
|
|
539
|
+
填充核心在模块级 apply_merged_ranges_fill(§1.27 起与分块路径共用)。
|
|
540
|
+
|
|
541
|
+
参数:
|
|
542
|
+
df: 已加载的 DataFrame
|
|
543
|
+
sheet_name: 数据来源的 sheet 名称;None 时按 spec 的 sheet/sheet_index 解析
|
|
544
|
+
header_row: 表头行号(0-based,不含 skip_rows)
|
|
545
|
+
skip_rows: 表头前跳过的行数(load_multi_sheet 路径已把 skip_rows 折入
|
|
546
|
+
effective_header,此时应保持默认 0,避免重复扣减)
|
|
547
|
+
"""
|
|
548
|
+
if self._effective_engine() != "openpyxl":
|
|
549
|
+
return df
|
|
550
|
+
try:
|
|
551
|
+
from openpyxl import load_workbook
|
|
552
|
+
|
|
553
|
+
wb = load_workbook(self.spec.path, data_only=True)
|
|
554
|
+
try:
|
|
555
|
+
if sheet_name is None:
|
|
556
|
+
# 回归修复: 填充逻辑必须定位到与读取一致的工作表。读取时 sheet 未指定
|
|
557
|
+
# 名称会按 sheet_index 选表(见 _build_read_kwargs),此处解析需保持
|
|
558
|
+
# 同一优先级;过去无条件回退到第一张表,sheet_index>0 时填错表。
|
|
559
|
+
if self.spec.sheet:
|
|
560
|
+
sheet_name = self.spec.sheet
|
|
561
|
+
elif 0 <= self.spec.sheet_index < len(wb.sheetnames):
|
|
562
|
+
sheet_name = wb.sheetnames[self.spec.sheet_index]
|
|
563
|
+
else:
|
|
564
|
+
sheet_name = wb.sheetnames[0]
|
|
565
|
+
ws = wb[sheet_name]
|
|
566
|
+
df_filled, _filled, _skipped = apply_merged_ranges_fill(
|
|
567
|
+
df,
|
|
568
|
+
ws.merged_cells.ranges,
|
|
569
|
+
header_row=header_row,
|
|
570
|
+
skip_rows=skip_rows,
|
|
571
|
+
)
|
|
572
|
+
return df_filled
|
|
573
|
+
finally:
|
|
574
|
+
wb.close()
|
|
575
|
+
except Exception as e:
|
|
576
|
+
# B27: 合并单元格前向填充失败时降级为返回原始 df(不中断加载),
|
|
577
|
+
# 但必须留有诊断痕迹——否则 NotNull/Unique 会因未填充的合并单元格 NaN 静默误报,
|
|
578
|
+
# 且无法定位根因(损坏的工作簿、权限错误、openpyxl 版本差异等都会落到这里)。
|
|
579
|
+
logger.warning(
|
|
580
|
+
"合并单元格前向填充失败,降级返回未填充数据(可能导致 NotNull/Unique 误报),file=%s: %s",
|
|
581
|
+
self.spec.path,
|
|
582
|
+
e,
|
|
583
|
+
exc_info=True,
|
|
584
|
+
)
|
|
585
|
+
return df
|
|
586
|
+
|
|
587
|
+
def validate(self) -> list[str]:
|
|
588
|
+
"""
|
|
589
|
+
@methoddesc 验证 Excel 文件配置和文件本身。
|
|
590
|
+
|
|
591
|
+
检查项:
|
|
592
|
+
- 文件是否存在
|
|
593
|
+
- 文件扩展名是否为 .xlsx 或 .xls
|
|
594
|
+
- 文件大小是否超过 100MB(发出警告)
|
|
595
|
+
|
|
596
|
+
Returns:
|
|
597
|
+
错误信息列表,空列表表示验证通过
|
|
598
|
+
|
|
599
|
+
示例:
|
|
600
|
+
>>> errors = loader.validate()
|
|
601
|
+
>>> if errors:
|
|
602
|
+
... print("验证失败:", errors)
|
|
603
|
+
"""
|
|
604
|
+
errors = []
|
|
605
|
+
path = Path(self.spec.path)
|
|
606
|
+
|
|
607
|
+
if not path.exists():
|
|
608
|
+
errors.append(f"文件不存在: {self.spec.path}")
|
|
609
|
+
return errors
|
|
610
|
+
|
|
611
|
+
ext = path.suffix.lower()
|
|
612
|
+
# 与 _ENGINE_BY_SUFFIX/加载路径同口径:.xlsm 同为 openpyxl 可读格式
|
|
613
|
+
if ext not in [".xlsx", ".xls", ".xlsm"]:
|
|
614
|
+
errors.append(f"不支持的 Excel 格式: {ext}")
|
|
615
|
+
|
|
616
|
+
size_mb = path.stat().st_size / (1024 * 1024)
|
|
617
|
+
if size_mb > 100:
|
|
618
|
+
errors.append(f"警告: 文件较大 ({size_mb:.1f}MB)")
|
|
619
|
+
|
|
620
|
+
return errors
|
|
621
|
+
|
|
622
|
+
def preview(self, nrows: int = 10) -> pd.DataFrame:
|
|
623
|
+
"""
|
|
624
|
+
@methoddesc 预览 Excel 文件的前 n 行数据。
|
|
625
|
+
|
|
626
|
+
通过限制 nrows 参数快速加载文件头部数据,
|
|
627
|
+
比加载完整文件更高效。同样会应用合并单元格填充。
|
|
628
|
+
|
|
629
|
+
Args:
|
|
630
|
+
nrows: 要预览的行数,默认为 10
|
|
631
|
+
|
|
632
|
+
Returns:
|
|
633
|
+
包含前 n 行数据的 DataFrame
|
|
634
|
+
|
|
635
|
+
示例:
|
|
636
|
+
>>> df = loader.preview(nrows=5)
|
|
637
|
+
>>> print(df.head())
|
|
638
|
+
"""
|
|
639
|
+
try:
|
|
640
|
+
read_kwargs: dict[str, Any] = {
|
|
641
|
+
"nrows": nrows,
|
|
642
|
+
"header": self.spec.header_row if self.spec.header_enabled else None,
|
|
643
|
+
"engine": self._effective_engine(),
|
|
644
|
+
**self._engine_kwargs(),
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
if self.spec.sheet:
|
|
648
|
+
read_kwargs["sheet_name"] = self.spec.sheet
|
|
649
|
+
else:
|
|
650
|
+
read_kwargs["sheet_name"] = self.spec.sheet_index
|
|
651
|
+
|
|
652
|
+
# 预览同样要跳过表头前的说明行,且合并单元格填充的行号换算依赖 skip_rows,
|
|
653
|
+
# 读取与填充必须使用同一值,否则填充区域错位
|
|
654
|
+
if self.spec.skip_rows > 0:
|
|
655
|
+
read_kwargs["skiprows"] = self.spec.skip_rows
|
|
656
|
+
|
|
657
|
+
df = pd.read_excel(self.spec.path, **read_kwargs)
|
|
658
|
+
|
|
659
|
+
if not self.spec.header_enabled:
|
|
660
|
+
df.columns = [f"col_{i}" for i in range(len(df.columns))]
|
|
661
|
+
|
|
662
|
+
df = self._apply_merged_cell_fill(df, self.spec.sheet, self.spec.header_row, self.spec.skip_rows)
|
|
663
|
+
return df
|
|
664
|
+
|
|
665
|
+
except Exception:
|
|
666
|
+
return super().preview(nrows)
|