precis-cli 0.1.3__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- app/__init__.py +16 -0
- app/api/__init__.py +32 -0
- app/api/dependencies.py +185 -0
- app/api/main.py +301 -0
- app/api/middleware/__init__.py +34 -0
- app/api/middleware/exception_handler.py +126 -0
- app/api/middleware/request_logging.py +97 -0
- app/api/middleware/token_auth.py +215 -0
- app/api/models/__init__.py +151 -0
- app/api/models/connection_rules.py +98 -0
- app/api/models/files.py +99 -0
- app/api/models/full_validation.py +554 -0
- app/api/models/project.py +117 -0
- app/api/models/projects.py +80 -0
- app/api/models/schema.py +121 -0
- app/api/models/v2_responses.py +46 -0
- app/api/models/validation.py +303 -0
- app/api/models/workspace.py +118 -0
- app/api/routers/__init__.py +56 -0
- app/api/routers/ai/__init__.py +36 -0
- app/api/routers/ai/chat.py +243 -0
- app/api/routers/ai/generate.py +156 -0
- app/api/routers/ai/hardware.py +199 -0
- app/api/routers/ai/jobs.py +544 -0
- app/api/routers/ai/migrate.py +471 -0
- app/api/routers/ai/models.py +352 -0
- app/api/routers/ai/ollama.py +108 -0
- app/api/routers/ai/providers.py +491 -0
- app/api/routers/ai/router.py +44 -0
- app/api/routers/ai/stream.py +281 -0
- app/api/routers/ai/utils.py +182 -0
- app/api/routers/core/__init__.py +34 -0
- app/api/routers/core/connection_rules.py +188 -0
- app/api/routers/core/data_sources.py +374 -0
- app/api/routers/core/regex.py +333 -0
- app/api/routers/core/reporting.py +263 -0
- app/api/routers/files/__init__.py +24 -0
- app/api/routers/files/ops.py +179 -0
- app/api/routers/files/transfer.py +101 -0
- app/api/routers/preview/__init__.py +35 -0
- app/api/routers/preview/content_mode.py +264 -0
- app/api/routers/preview/header_row.py +154 -0
- app/api/routers/preview/models.py +81 -0
- app/api/routers/preview/path_mode.py +440 -0
- app/api/routers/preview/router.py +30 -0
- app/api/routers/project/__init__.py +61 -0
- app/api/routers/project/base.py +111 -0
- app/api/routers/project/constraint.py +345 -0
- app/api/routers/project/full_config.py +411 -0
- app/api/routers/project/full_config_writer.py +327 -0
- app/api/routers/project/helpers.py +167 -0
- app/api/routers/project/inspection_fix.py +330 -0
- app/api/routers/project/manifest.py +605 -0
- app/api/routers/project/models.py +177 -0
- app/api/routers/project/pattern.py +268 -0
- app/api/routers/project/regex.py +358 -0
- app/api/routers/project/scanner.py +157 -0
- app/api/routers/project/schema.py +567 -0
- app/api/routers/project/schema_helpers.py +149 -0
- app/api/routers/project/settings.py +399 -0
- app/api/routers/project/template.py +325 -0
- app/api/routers/project/validation.py +227 -0
- app/api/routers/project/view.py +177 -0
- app/api/routers/project/workspaces.py +167 -0
- app/api/routers/projects/__init__.py +25 -0
- app/api/routers/projects/create.py +121 -0
- app/api/routers/projects/open.py +103 -0
- app/api/routers/projects/scan.py +100 -0
- app/api/routers/validation/__init__.py +39 -0
- app/api/routers/validation/common.py +263 -0
- app/api/routers/validation/content_mode.py +401 -0
- app/api/routers/validation/history.py +235 -0
- app/api/routers/validation/inline_mode.py +196 -0
- app/api/routers/validation/path_mode.py +314 -0
- app/api/routers/validation/router.py +51 -0
- app/api/services/full_validation_response_builder.py +412 -0
- app/api/services/io_error_messages.py +41 -0
- app/api/services/preview_service.py +248 -0
- app/cli/__init__.py +18 -0
- app/cli/__main__.py +30 -0
- app/cli/shared_services/__init__.py +42 -0
- app/cli/shared_services/config_ops.py +581 -0
- app/cli/shared_services/generation_ops.py +160 -0
- app/cli/shared_services/project_ops.py +288 -0
- app/cli/shell/__init__.py +48 -0
- app/cli/shell/commands/__init__.py +63 -0
- app/cli/shell/commands/ai/__init__.py +260 -0
- app/cli/shell/commands/ai/chat.py +331 -0
- app/cli/shell/commands/ai/delete.py +165 -0
- app/cli/shell/commands/ai/diff.py +70 -0
- app/cli/shell/commands/ai/display.py +428 -0
- app/cli/shell/commands/ai/executor.py +225 -0
- app/cli/shell/commands/ai/executor_utils.py +271 -0
- app/cli/shell/commands/ai/generate.py +285 -0
- app/cli/shell/commands/ai/interaction.py +262 -0
- app/cli/shell/commands/ai/migrate.py +281 -0
- app/cli/shell/commands/ai/resolver.py +224 -0
- app/cli/shell/commands/ai/status.py +101 -0
- app/cli/shell/commands/ai/switch.py +158 -0
- app/cli/shell/commands/ai/utils.py +52 -0
- app/cli/shell/commands/base.py +343 -0
- app/cli/shell/commands/config/__init__.py +131 -0
- app/cli/shell/commands/config/base.py +80 -0
- app/cli/shell/commands/config/check.py +246 -0
- app/cli/shell/commands/config/edit.py +148 -0
- app/cli/shell/commands/config/get.py +111 -0
- app/cli/shell/commands/config/init.py +123 -0
- app/cli/shell/commands/config/inspect.py +164 -0
- app/cli/shell/commands/config/list.py +104 -0
- app/cli/shell/commands/config/set.py +119 -0
- app/cli/shell/commands/config/show.py +150 -0
- app/cli/shell/commands/exit.py +67 -0
- app/cli/shell/commands/help.py +104 -0
- app/cli/shell/commands/infer_schema.py +156 -0
- app/cli/shell/commands/open.py +201 -0
- app/cli/shell/commands/project.py +215 -0
- app/cli/shell/commands/provider.py +619 -0
- app/cli/shell/commands/system.py +170 -0
- app/cli/shell/commands/validate.py +455 -0
- app/cli/shell/completer.py +231 -0
- app/cli/shell/config_storage.py +191 -0
- app/cli/shell/exceptions.py +90 -0
- app/cli/shell/formatter.py +371 -0
- app/cli/shell/interactive_menu.py +348 -0
- app/cli/shell/main.py +286 -0
- app/cli/shell/parser.py +266 -0
- app/cli/start.py +180 -0
- app/cli_main.py +52 -0
- app/mcp_server.py +311 -0
- app/shared/core/__init__.py +18 -0
- app/shared/core/app_version.py +59 -0
- app/shared/core/config/__init__.py +119 -0
- app/shared/core/config/server.py +193 -0
- app/shared/core/data_source/__init__.py +106 -0
- app/shared/core/data_source/loader.py +348 -0
- app/shared/core/data_source/loaders/__init__.py +213 -0
- app/shared/core/data_source/loaders/base.py +215 -0
- app/shared/core/data_source/loaders/converter.py +337 -0
- app/shared/core/data_source/loaders/csv_loader.py +252 -0
- app/shared/core/data_source/loaders/excel_loader.py +666 -0
- app/shared/core/data_source/loaders/extractor.py +268 -0
- app/shared/core/data_source/loaders/json_loader.py +354 -0
- app/shared/core/data_source/loaders/registry.py +88 -0
- app/shared/core/data_source/loaders/sql_loader.py +313 -0
- app/shared/core/data_source/loaders/strategies/__init__.py +100 -0
- app/shared/core/data_source/loaders/strategies/array_parser.py +277 -0
- app/shared/core/data_source/loaders/strategies/lines_parser.py +355 -0
- app/shared/core/data_source/loaders/strategies/object_parser.py +278 -0
- app/shared/core/data_source/schema_info.py +50 -0
- app/shared/core/data_source/specs/__init__.py +128 -0
- app/shared/core/data_source/specs/base.py +258 -0
- app/shared/core/data_source/specs/csv_source.py +137 -0
- app/shared/core/data_source/specs/excel_source.py +141 -0
- app/shared/core/data_source/specs/file_base.py +252 -0
- app/shared/core/data_source/specs/json_source.py +226 -0
- app/shared/core/data_source/specs/sql_source.py +161 -0
- app/shared/core/encoding.py +36 -0
- app/shared/core/io/__init__.py +24 -0
- app/shared/core/io/yaml.py +379 -0
- app/shared/core/manifest_schema/__init__.py +52 -0
- app/shared/core/manifest_schema/types.py +99 -0
- app/shared/core/manifest_schema/version.py +175 -0
- app/shared/core/patterns/__init__.py +24 -0
- app/shared/core/patterns/loader.py +134 -0
- app/shared/core/patterns/writer.py +244 -0
- app/shared/core/project/__init__.py +47 -0
- app/shared/core/project/constraint/builders/__init__.py +50 -0
- app/shared/core/project/constraint/builders/base.py +100 -0
- app/shared/core/project/constraint/builders/composite.py +77 -0
- app/shared/core/project/constraint/builders/conditional.py +67 -0
- app/shared/core/project/constraint/builders/foreign_key.py +53 -0
- app/shared/core/project/constraint/builders/registry.py +64 -0
- app/shared/core/project/constraint/builders/scripted.py +51 -0
- app/shared/core/project/constraint/builders/single_column.py +86 -0
- app/shared/core/project/constraint/builders/unique.py +53 -0
- app/shared/core/project/constraint/factory.py +170 -0
- app/shared/core/project/constraint/reader.py +214 -0
- app/shared/core/project/constraint/registry.py +233 -0
- app/shared/core/project/constraint/types/__init__.py +63 -0
- app/shared/core/project/constraint/types/constraint_file.py +261 -0
- app/shared/core/project/constraint/types/refs.py +460 -0
- app/shared/core/project/constraint/types.py +28 -0
- app/shared/core/project/constraint/writer.py +181 -0
- app/shared/core/project/loader/__init__.py +56 -0
- app/shared/core/project/loader/loader.py +30 -0
- app/shared/core/project/loader/loader_parts/config_inspector.py +137 -0
- app/shared/core/project/loader/loader_parts/embedded_constraints.py +224 -0
- app/shared/core/project/loader/loader_parts/file_loaders.py +58 -0
- app/shared/core/project/loader/loader_parts/inspection_ids.py +85 -0
- app/shared/core/project/loader/loader_parts/inspector_helpers.py +312 -0
- app/shared/core/project/loader/loader_parts/inspector_id_checks.py +274 -0
- app/shared/core/project/loader/loader_parts/inspector_reference_checks.py +690 -0
- app/shared/core/project/loader/loader_parts/inspector_uniqueness_checks.py +286 -0
- app/shared/core/project/loader/loader_parts/loading_error_messages.py +298 -0
- app/shared/core/project/loader/loader_parts/main.py +432 -0
- app/shared/core/project/loader/loader_parts/path_validation.py +117 -0
- app/shared/core/project/loader/loader_parts/runtime.py +125 -0
- app/shared/core/project/loader/types.py +246 -0
- app/shared/core/project/manifest/coverage.py +391 -0
- app/shared/core/project/manifest/reader.py +260 -0
- app/shared/core/project/manifest/types.py +91 -0
- app/shared/core/project/manifest/types_parts/__init__.py +60 -0
- app/shared/core/project/manifest/types_parts/constants.py +40 -0
- app/shared/core/project/manifest/types_parts/data_source.py +68 -0
- app/shared/core/project/manifest/types_parts/info.py +59 -0
- app/shared/core/project/manifest/types_parts/manifest.py +294 -0
- app/shared/core/project/manifest/types_parts/refs.py +153 -0
- app/shared/core/project/manifest/types_parts/settings.py +92 -0
- app/shared/core/project/manifest/types_parts/settings_file_processing.py +64 -0
- app/shared/core/project/manifest/types_parts/settings_script_security.py +78 -0
- app/shared/core/project/manifest/types_parts/settings_validation.py +83 -0
- app/shared/core/project/manifest/types_parts/template.py +66 -0
- app/shared/core/project/manifest/writer.py +262 -0
- app/shared/core/project/manual_data/__init__.py +27 -0
- app/shared/core/project/manual_data/types.py +75 -0
- app/shared/core/project/regex/reader.py +197 -0
- app/shared/core/project/regex/types.py +405 -0
- app/shared/core/project/regex/writer.py +123 -0
- app/shared/core/project/schema/reader.py +170 -0
- app/shared/core/project/schema/types.py +47 -0
- app/shared/core/project/schema/types_parts/__init__.py +36 -0
- app/shared/core/project/schema/types_parts/column.py +174 -0
- app/shared/core/project/schema/types_parts/column_utils.py +72 -0
- app/shared/core/project/schema/types_parts/constraint.py +165 -0
- app/shared/core/project/schema/types_parts/schema_id.py +66 -0
- app/shared/core/project/schema/types_parts/source.py +255 -0
- app/shared/core/project/schema/types_parts/source_options.py +347 -0
- app/shared/core/project/schema/types_parts/table.py +230 -0
- app/shared/core/project/schema/writer.py +139 -0
- app/shared/core/project/schema_ref_check.py +95 -0
- app/shared/core/project/template/__init__.py +27 -0
- app/shared/core/project/template/expander.py +263 -0
- app/shared/core/project/template/reader.py +120 -0
- app/shared/core/project/template/types.py +114 -0
- app/shared/core/project/transform/reader.py +76 -0
- app/shared/core/project/transform/types.py +116 -0
- app/shared/core/project/transform/writer.py +84 -0
- app/shared/core/pydantic_messages.py +59 -0
- app/shared/core/reporter/__init__.py +46 -0
- app/shared/core/reporter/reporter.py +220 -0
- app/shared/core/reporter/reporters/__init__.py +65 -0
- app/shared/core/reporter/reporters/base.py +188 -0
- app/shared/core/reporter/reporters/dingtalk_app_reporter.py +274 -0
- app/shared/core/reporter/reporters/email_reporter.py +271 -0
- app/shared/core/reporter/reporters/feishu_app_reporter.py +467 -0
- app/shared/core/reporter/reporters/local_file_reporter.py +208 -0
- app/shared/core/reporter/reporters/wecom_app_reporter.py +268 -0
- app/shared/core/utils/__init__.py +18 -0
- app/shared/core/utils/path_utils.py +60 -0
- app/shared/core/utils/regex_extract.py +113 -0
- app/shared/domain/__init__.py +114 -0
- app/shared/domain/constraints/__init__.py +76 -0
- app/shared/domain/constraints/allowed_values.py +211 -0
- app/shared/domain/constraints/base.py +141 -0
- app/shared/domain/constraints/charset.py +337 -0
- app/shared/domain/constraints/composite.py +174 -0
- app/shared/domain/constraints/condition_registry.py +130 -0
- app/shared/domain/constraints/conditional.py +629 -0
- app/shared/domain/constraints/date_logic.py +731 -0
- app/shared/domain/constraints/foreign_key.py +261 -0
- app/shared/domain/constraints/key_normalization.py +56 -0
- app/shared/domain/constraints/not_null.py +185 -0
- app/shared/domain/constraints/range.py +360 -0
- app/shared/domain/constraints/regex.py +218 -0
- app/shared/domain/constraints/scripted.py +276 -0
- app/shared/domain/constraints/unique.py +233 -0
- app/shared/domain/data_engine.py +384 -0
- app/shared/domain/data_types.py +113 -0
- app/shared/domain/data_types_parts/__init__.py +67 -0
- app/shared/domain/data_types_parts/base.py +204 -0
- app/shared/domain/data_types_parts/composite.py +250 -0
- app/shared/domain/data_types_parts/expression.py +202 -0
- app/shared/domain/data_types_parts/extracted.py +106 -0
- app/shared/domain/data_types_parts/json_types.py +234 -0
- app/shared/domain/data_types_parts/scalars.py +719 -0
- app/shared/domain/data_types_parts/sequence.py +129 -0
- app/shared/domain/dataset_schema.py +48 -0
- app/shared/domain/eval_sandbox.py +63 -0
- app/shared/domain/expression_system.py +366 -0
- app/shared/domain/regex_flags.py +53 -0
- app/shared/domain/schema/builder.py +223 -0
- app/shared/domain/schema/models.py +339 -0
- app/shared/domain/transforms/__init__.py +28 -0
- app/shared/domain/transforms/aggregate.py +141 -0
- app/shared/domain/transforms/base.py +171 -0
- app/shared/domain/transforms/cast_type.py +120 -0
- app/shared/domain/transforms/concat.py +103 -0
- app/shared/domain/transforms/conditional_assign.py +114 -0
- app/shared/domain/transforms/date_format.py +81 -0
- app/shared/domain/transforms/digits.py +77 -0
- app/shared/domain/transforms/drop_duplicates.py +94 -0
- app/shared/domain/transforms/fill_na.py +94 -0
- app/shared/domain/transforms/filter_rows.py +97 -0
- app/shared/domain/transforms/lookup.py +78 -0
- app/shared/domain/transforms/lower_case.py +70 -0
- app/shared/domain/transforms/map_value.py +90 -0
- app/shared/domain/transforms/math_expr.py +112 -0
- app/shared/domain/transforms/modulo.py +82 -0
- app/shared/domain/transforms/regex_extract.py +103 -0
- app/shared/domain/transforms/registry.py +95 -0
- app/shared/domain/transforms/replace.py +88 -0
- app/shared/domain/transforms/sort_rows.py +94 -0
- app/shared/domain/transforms/string_split.py +84 -0
- app/shared/domain/transforms/strip.py +73 -0
- app/shared/domain/transforms/substring.py +92 -0
- app/shared/domain/transforms/upper_case.py +70 -0
- app/shared/domain/transforms/weighted_sum.py +104 -0
- app/shared/domain/validation_constraints.py +69 -0
- app/shared/services/__init__.py +62 -0
- app/shared/services/ai/__init__.py +48 -0
- app/shared/services/ai/agent/__init__.py +40 -0
- app/shared/services/ai/agent/chat_tools/__init__.py +47 -0
- app/shared/services/ai/agent/chat_tools/apply_actions.py +592 -0
- app/shared/services/ai/agent/chat_tools/ask_user.py +237 -0
- app/shared/services/ai/agent/chat_tools/read_canvas.py +140 -0
- app/shared/services/ai/agent/chat_tools/read_project.py +160 -0
- app/shared/services/ai/agent/chat_tools/read_table.py +301 -0
- app/shared/services/ai/agent/chat_tools/schemas.py +105 -0
- app/shared/services/ai/agent/chat_tools/validate_table.py +131 -0
- app/shared/services/ai/agent/executor.py +554 -0
- app/shared/services/ai/agent/memory.py +256 -0
- app/shared/services/ai/agent/planner.py +282 -0
- app/shared/services/ai/agent/tool_registry.py +296 -0
- app/shared/services/ai/agent/tools/__init__.py +32 -0
- app/shared/services/ai/agent/tools/config_generate.py +133 -0
- app/shared/services/ai/agent/tools/config_refine.py +109 -0
- app/shared/services/ai/agent/tools/config_validate.py +403 -0
- app/shared/services/ai/agent/tools/merge_results.py +136 -0
- app/shared/services/ai/agent/tools/plan_chunks.py +79 -0
- app/shared/services/ai/agent/tools/script_parse.py +359 -0
- app/shared/services/ai/agent/types.py +138 -0
- app/shared/services/ai/chat_agent_runner.py +596 -0
- app/shared/services/ai/chat_orchestrator.py +725 -0
- app/shared/services/ai/failure_messages.py +65 -0
- app/shared/services/ai/job_storage.py +274 -0
- app/shared/services/ai/migrate_service.py +550 -0
- app/shared/services/ai/streaming/__init__.py +65 -0
- app/shared/services/ai/streaming/event_journal.py +197 -0
- app/shared/services/ai/streaming/orchestrator.py +262 -0
- app/shared/services/ai/streaming/pending_interaction_store.py +227 -0
- app/shared/services/ai/streaming/sse_response.py +148 -0
- app/shared/services/ai/streaming/types.py +73 -0
- app/shared/services/ai/types.py +213 -0
- app/shared/services/ai/utils.py +322 -0
- app/shared/services/diff/config_diff.py +320 -0
- app/shared/services/hardware.py +237 -0
- app/shared/services/llm/__init__.py +62 -0
- app/shared/services/llm/actions/__init__.py +42 -0
- app/shared/services/llm/actions/_canvas_validator.py +147 -0
- app/shared/services/llm/actions/_constraint_validator.py +401 -0
- app/shared/services/llm/actions/_regex_validator.py +85 -0
- app/shared/services/llm/actions/_schema_validator.py +82 -0
- app/shared/services/llm/actions/_settings_validator.py +76 -0
- app/shared/services/llm/actions/_transform_validator.py +78 -0
- app/shared/services/llm/actions/action_handlers.py +400 -0
- app/shared/services/llm/actions/action_parser.py +63 -0
- app/shared/services/llm/actions/action_processor.py +468 -0
- app/shared/services/llm/actions/action_validator.py +311 -0
- app/shared/services/llm/actions/diff_compute.py +195 -0
- app/shared/services/llm/actions/regex_handlers.py +284 -0
- app/shared/services/llm/actions/registry.py +336 -0
- app/shared/services/llm/actions/schema_handlers.py +335 -0
- app/shared/services/llm/actions/settings_handlers.py +180 -0
- app/shared/services/llm/actions/specs.py +259 -0
- app/shared/services/llm/actions/transform_handlers.py +279 -0
- app/shared/services/llm/actions/validation_types.py +147 -0
- app/shared/services/llm/cache/__init__.py +18 -0
- app/shared/services/llm/cache/response_cache.py +80 -0
- app/shared/services/llm/chat/__init__.py +35 -0
- app/shared/services/llm/chat/chat_system_prompt.py +561 -0
- app/shared/services/llm/chat/response_parser.py +296 -0
- app/shared/services/llm/config/__init__.py +45 -0
- app/shared/services/llm/config/crypto.py +135 -0
- app/shared/services/llm/config/loader.py +258 -0
- app/shared/services/llm/config/models.py +202 -0
- app/shared/services/llm/config/presets.py +128 -0
- app/shared/services/llm/config_generator.py +163 -0
- app/shared/services/llm/constraints/__init__.py +41 -0
- app/shared/services/llm/constraints/constraint_builder.py +177 -0
- app/shared/services/llm/constraints/constraint_deletion.py +84 -0
- app/shared/services/llm/constraints/constraint_id.py +174 -0
- app/shared/services/llm/constraints/frontend_instructions.py +333 -0
- app/shared/services/llm/constraints/inline_batch.py +279 -0
- app/shared/services/llm/discovery/__init__.py +35 -0
- app/shared/services/llm/discovery/scanner.py +181 -0
- app/shared/services/llm/generation/__init__.py +42 -0
- app/shared/services/llm/generation/agent_wiring.py +156 -0
- app/shared/services/llm/generation/config_builder.py +386 -0
- app/shared/services/llm/generation/errors.py +36 -0
- app/shared/services/llm/generation/existing_config.py +99 -0
- app/shared/services/llm/generation/profiler.py +279 -0
- app/shared/services/llm/generation/prompt_builder.py +199 -0
- app/shared/services/llm/generation/response_parser.py +144 -0
- app/shared/services/llm/generation/service.py +622 -0
- app/shared/services/llm/models.py +104 -0
- app/shared/services/llm/providers/__init__.py +48 -0
- app/shared/services/llm/providers/base.py +310 -0
- app/shared/services/llm/providers/cached_provider.py +90 -0
- app/shared/services/llm/providers/ollama.py +498 -0
- app/shared/services/llm/providers/openai.py +334 -0
- app/shared/services/llm/providers/registry.py +108 -0
- app/shared/services/llm/schema_resolver.py +141 -0
- app/shared/services/llm/suggestion_utils.py +219 -0
- app/shared/services/llm/validate_executor.py +163 -0
- app/shared/services/llm/yaml_io.py +406 -0
- app/shared/services/preview/__init__.py +18 -0
- app/shared/services/preview/loader.py +145 -0
- app/shared/services/preview/path_validation.py +138 -0
- app/shared/services/project_loader.py +57 -0
- app/shared/services/schema_inference.py +260 -0
- app/shared/services/schema_runtime_builder.py +120 -0
- app/shared/services/validation/__init__.py +52 -0
- app/shared/services/validation/chunked_loader.py +509 -0
- app/shared/services/validation/dag/__init__.py +35 -0
- app/shared/services/validation/dag/builder.py +141 -0
- app/shared/services/validation/dag/executor.py +225 -0
- app/shared/services/validation/dag/sorter.py +77 -0
- app/shared/services/validation/data_loader.py +231 -0
- app/shared/services/validation/engine.py +446 -0
- app/shared/services/validation/executor.py +1030 -0
- app/shared/services/validation/extractors.py +266 -0
- app/shared/services/validation/history.py +209 -0
- app/shared/services/validation/json_payload.py +136 -0
- app/shared/services/validation/loader.py +152 -0
- app/shared/services/validation/memory_monitor.py +179 -0
- app/shared/services/validation/postprocess.py +208 -0
- app/shared/services/validation/progress.py +64 -0
- app/shared/services/validation/report_export.py +222 -0
- app/shared/services/validation/resolver.py +180 -0
- app/shared/services/validation/service.py +418 -0
- app/shared/services/validation/types.py +238 -0
- app/shared/services/validation/validators/__init__.py +36 -0
- app/shared/services/validation/validators/adapter.py +182 -0
- app/shared/services/validation/validators/base.py +332 -0
- app/shared/services/validation/validators/composite.py +233 -0
- app/shared/services/validation/validators/date_logic.py +276 -0
- app/start_server.py +133 -0
- precis_cli-0.1.3.dist-info/METADATA +180 -0
- precis_cli-0.1.3.dist-info/RECORD +442 -0
- precis_cli-0.1.3.dist-info/WHEEL +5 -0
- precis_cli-0.1.3.dist-info/entry_points.txt +4 -0
- precis_cli-0.1.3.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2026 Precis Team
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
6
|
+
# you may not use this file except in compliance with the License.
|
|
7
|
+
# You may obtain a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
13
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
14
|
+
# See the License for the specific language governing permissions and
|
|
15
|
+
# limitations under the License.
|
|
16
|
+
"""
|
|
17
|
+
@fileoverview Aggregate 转换运行器
|
|
18
|
+
|
|
19
|
+
功能概述:
|
|
20
|
+
- 对 DataFrame 进行聚合操作
|
|
21
|
+
- 支持分组聚合(group_by)和整表聚合
|
|
22
|
+
- 支持 count/sum/avg/min/max 聚合函数
|
|
23
|
+
|
|
24
|
+
参数:
|
|
25
|
+
aggregations: 聚合配置列表 [{column, func}]
|
|
26
|
+
- column: 聚合目标列名
|
|
27
|
+
- func: 聚合函数 ("count"|"sum"|"avg"|"min"|"max")
|
|
28
|
+
group_by: 分组列名,支持逗号分隔字符串或列表,留空则整表聚合
|
|
29
|
+
|
|
30
|
+
说明:
|
|
31
|
+
- input_column 被忽略(操作整表而非单列)
|
|
32
|
+
- output_columns 定义聚合结果的列名(按 aggregations 顺序对应)
|
|
33
|
+
如果 output_columns 未提供或不足,则自动生成 "func_column" 格式列名
|
|
34
|
+
- 输出 DataFrame 索引重置为连续 0-based
|
|
35
|
+
"""
|
|
36
|
+
|
|
37
|
+
from __future__ import annotations
|
|
38
|
+
|
|
39
|
+
from typing import Any
|
|
40
|
+
|
|
41
|
+
import pandas as pd
|
|
42
|
+
|
|
43
|
+
from .base import TransformRunner
|
|
44
|
+
|
|
45
|
+
# pandas agg 函数名映射
|
|
46
|
+
_FUNC_MAP = {
|
|
47
|
+
"count": "count",
|
|
48
|
+
"sum": "sum",
|
|
49
|
+
"avg": "mean",
|
|
50
|
+
"min": "min",
|
|
51
|
+
"max": "max",
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class AggregateRunner(TransformRunner):
|
|
56
|
+
"""@classdesc 聚合转换运行器"""
|
|
57
|
+
|
|
58
|
+
def execute(
|
|
59
|
+
self,
|
|
60
|
+
df: pd.DataFrame,
|
|
61
|
+
input_column: str,
|
|
62
|
+
params: dict[str, Any],
|
|
63
|
+
output_columns: list[str],
|
|
64
|
+
) -> pd.DataFrame:
|
|
65
|
+
"""
|
|
66
|
+
@methoddesc 执行聚合转换
|
|
67
|
+
|
|
68
|
+
业务用途:
|
|
69
|
+
- 从 params.aggregations 读取 [{column, func}, ...] 配置
|
|
70
|
+
- 若指定 group_by 则按其分组聚合;否则整表聚合
|
|
71
|
+
- 输出列名优先取 output_columns[i],缺省按 func_column 自动生成
|
|
72
|
+
|
|
73
|
+
参数:
|
|
74
|
+
df: 源 DataFrame
|
|
75
|
+
input_column: 输入列名(聚合场景忽略)
|
|
76
|
+
params: 包含 aggregations / group_by 的参数字典
|
|
77
|
+
output_columns: 用户指定的输出列名列表
|
|
78
|
+
|
|
79
|
+
返回:
|
|
80
|
+
聚合后的 DataFrame
|
|
81
|
+
|
|
82
|
+
异常:
|
|
83
|
+
ValueError: aggregations 为空或列名不存在
|
|
84
|
+
"""
|
|
85
|
+
aggregations = params.get("aggregations", [])
|
|
86
|
+
group_by_str = params.get("group_by", "")
|
|
87
|
+
|
|
88
|
+
if not aggregations:
|
|
89
|
+
raise ValueError("Aggregate 需要至少一个 aggregation 配置")
|
|
90
|
+
|
|
91
|
+
# 解析 group_by:兼容逗号分隔字符串与列表两种格式
|
|
92
|
+
# 前端 tags 产出的是数组,旧配置/手写 YAML 可能是逗号分隔字符串
|
|
93
|
+
group_by = None
|
|
94
|
+
if group_by_str:
|
|
95
|
+
if isinstance(group_by_str, list):
|
|
96
|
+
parsed = [str(col).strip() for col in group_by_str if str(col).strip()]
|
|
97
|
+
else:
|
|
98
|
+
parsed = [col.strip() for col in str(group_by_str).split(",") if col.strip()]
|
|
99
|
+
# §1.14: 拼错列名报配置错误——原实现静默剔除后走整表聚合,
|
|
100
|
+
# "每组一行"变"全表一行",下游约束在错误粒度上继续跑
|
|
101
|
+
missing = [col for col in parsed if col not in df.columns]
|
|
102
|
+
if missing:
|
|
103
|
+
raise ValueError(f"group_by 列不存在: {missing}(表列: {list(df.columns)})")
|
|
104
|
+
if parsed:
|
|
105
|
+
group_by = parsed
|
|
106
|
+
|
|
107
|
+
# 构建 agg 字典:{输出列名: (源列名, pandas函数名)}
|
|
108
|
+
agg_dict: dict[str, tuple[str, str]] = {}
|
|
109
|
+
col_names: list[str] = []
|
|
110
|
+
|
|
111
|
+
for i, agg in enumerate(aggregations):
|
|
112
|
+
col = agg.get("column", "")
|
|
113
|
+
func = agg.get("func", "count")
|
|
114
|
+
|
|
115
|
+
if col not in df.columns:
|
|
116
|
+
raise ValueError(f"聚合列不存在: {col}")
|
|
117
|
+
|
|
118
|
+
# §1.14: 未知 func 报配置错误——原实现静默按 count 聚合,求和结果变计数
|
|
119
|
+
if func not in _FUNC_MAP:
|
|
120
|
+
raise ValueError(f"未知的聚合函数 '{func}',支持: {sorted(_FUNC_MAP)}")
|
|
121
|
+
pandas_func = _FUNC_MAP[func]
|
|
122
|
+
|
|
123
|
+
# 确定输出列名:优先使用 output_columns,否则自动生成
|
|
124
|
+
if output_columns and i < len(output_columns):
|
|
125
|
+
out_name = output_columns[i]
|
|
126
|
+
else:
|
|
127
|
+
out_name = f"{func}_{col}"
|
|
128
|
+
|
|
129
|
+
agg_dict[out_name] = pd.NamedAgg(column=col, aggfunc=pandas_func)
|
|
130
|
+
col_names.append(out_name)
|
|
131
|
+
|
|
132
|
+
# 执行聚合
|
|
133
|
+
if group_by:
|
|
134
|
+
result = df.groupby(group_by, dropna=False).agg(**agg_dict).reset_index()
|
|
135
|
+
else:
|
|
136
|
+
# 整表聚合:groupby(None) 产生单行结果
|
|
137
|
+
result = df.agg(**agg_dict).to_frame().T
|
|
138
|
+
|
|
139
|
+
# 确保列顺序:group_by 列 + 聚合结果列
|
|
140
|
+
result = result.reset_index(drop=True)
|
|
141
|
+
return result
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2026 Precis Team
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
6
|
+
# you may not use this file except in compliance with the License.
|
|
7
|
+
# You may obtain a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
13
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
14
|
+
# See the License for the specific language governing permissions and
|
|
15
|
+
# limitations under the License.
|
|
16
|
+
"""
|
|
17
|
+
@fileoverview Transform 运行器抽象基类模块
|
|
18
|
+
|
|
19
|
+
功能概述:
|
|
20
|
+
- 定义所有 Transform 运行器的统一接口
|
|
21
|
+
- 提供 execute(df, input_column, params) -> pd.DataFrame 契约
|
|
22
|
+
- 提供条件求值共享函数 evaluate_condition
|
|
23
|
+
|
|
24
|
+
架构设计:
|
|
25
|
+
- 抽象基类模式: TransformRunner 定义接口,子类实现具体转换逻辑
|
|
26
|
+
- 单列输入: 每个 transform 操作单列,输出新列到同一 DataFrame
|
|
27
|
+
- evaluate_condition: 条件类 Transform(conditional_assign / filter_rows)共享的条件求值逻辑
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
from __future__ import annotations
|
|
31
|
+
|
|
32
|
+
from abc import ABC, abstractmethod
|
|
33
|
+
from typing import Any
|
|
34
|
+
|
|
35
|
+
import pandas as pd
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def stringify_preserve_null(series: pd.Series) -> pd.Series:
|
|
39
|
+
"""转字符串并保留空值:NaN/None → None,其余值 astype(str)。
|
|
40
|
+
|
|
41
|
+
各字符串型 runner 的输出列统一用此助手——裸 astype(str) 会把空值变成
|
|
42
|
+
"nan"/"None" 字面量写入输出列,污染下游数据(如拼接出 "Nonez"、
|
|
43
|
+
条件匹配命中 "nan")。
|
|
44
|
+
"""
|
|
45
|
+
result = series.astype(str)
|
|
46
|
+
return result.where(series.notna(), None)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _norm_to_str(v: Any) -> str:
|
|
50
|
+
"""条件比较用字符串归一化(A7,与 allowed_values D5 的 _norm_to_str 同语义)。
|
|
51
|
+
|
|
52
|
+
含空值的整数列经类型处理后会整列升为 float64(100 → 100.0),裸 astype(str)
|
|
53
|
+
得 "100.0",与配置值 100 的 "100" 不命中。对 is_integer() 的 float 回收为
|
|
54
|
+
整数字符串,使 float64 的 100.0 与配置 100 等价。
|
|
55
|
+
"""
|
|
56
|
+
if isinstance(v, float) and v.is_integer():
|
|
57
|
+
return str(int(v))
|
|
58
|
+
return str(v)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def evaluate_condition(df: pd.DataFrame, cond: dict[str, Any]) -> pd.Series:
|
|
62
|
+
"""评估单个条件,返回布尔 Series。
|
|
63
|
+
|
|
64
|
+
支持 eq/ne/gt/gte/lt/lte/contains/startswith/endswith/in/not_in/is_null/is_not_null 操作符。
|
|
65
|
+
被 ConditionalAssignRunner 和 FilterRowsRunner 共享。
|
|
66
|
+
|
|
67
|
+
空值语义:NaN/None 行对除 is_null/is_not_null 外的所有操作符一律返回 False
|
|
68
|
+
(对齐 SQL 的 NULL 比较语义)——裸 astype(str) 会把空值字符串化成 "nan",
|
|
69
|
+
导致 eq "nan"、contains "an" 等误命中。
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
df: 输入 DataFrame
|
|
73
|
+
cond: 条件字典,包含 column, op, value
|
|
74
|
+
|
|
75
|
+
Returns:
|
|
76
|
+
布尔 Series;列不存在或操作符未知时返回全 False
|
|
77
|
+
"""
|
|
78
|
+
column = cond.get("column", "")
|
|
79
|
+
op = cond.get("op", "eq")
|
|
80
|
+
value = cond.get("value")
|
|
81
|
+
|
|
82
|
+
if column not in df.columns:
|
|
83
|
+
return pd.Series([False] * len(df), index=df.index)
|
|
84
|
+
|
|
85
|
+
series = df[column]
|
|
86
|
+
|
|
87
|
+
if op == "is_null":
|
|
88
|
+
return series.isna() | (series.astype(str).str.strip() == "")
|
|
89
|
+
elif op == "is_not_null":
|
|
90
|
+
return ~(series.isna() | (series.astype(str).str.strip() == ""))
|
|
91
|
+
|
|
92
|
+
result: pd.Series
|
|
93
|
+
if op == "eq":
|
|
94
|
+
result = series.map(_norm_to_str) == _norm_to_str(value)
|
|
95
|
+
elif op == "ne":
|
|
96
|
+
result = series.map(_norm_to_str) != _norm_to_str(value)
|
|
97
|
+
elif op in ("gt", "gte", "lt", "lte"):
|
|
98
|
+
# §1.5: 比较类阈值不可转数值/缺省 → 配置错误(原实现 NaN 参与比较恒 False,
|
|
99
|
+
# filter 结果为空表、条件赋值零命中,用户对着空输出排查不到是阈值写错)
|
|
100
|
+
if value is None:
|
|
101
|
+
raise ValueError(f"条件阈值未配置(op={op}),无法比较")
|
|
102
|
+
numeric_series = pd.to_numeric(series, errors="coerce")
|
|
103
|
+
numeric_value = pd.to_numeric(value, errors="coerce")
|
|
104
|
+
if pd.isna(numeric_value):
|
|
105
|
+
raise ValueError(f"条件阈值 '{value}' 无法转换为数值")
|
|
106
|
+
if op == "gt":
|
|
107
|
+
result = numeric_series > numeric_value
|
|
108
|
+
elif op == "gte":
|
|
109
|
+
result = numeric_series >= numeric_value
|
|
110
|
+
elif op == "lt":
|
|
111
|
+
result = numeric_series < numeric_value
|
|
112
|
+
else:
|
|
113
|
+
result = numeric_series <= numeric_value
|
|
114
|
+
elif op == "contains":
|
|
115
|
+
# regex=False:contains 按字面量子串匹配。过去默认按正则解释,用户数据含
|
|
116
|
+
# "(未分类)"、"a.b" 等字符时会报 re.error 崩溃或产生错误匹配。
|
|
117
|
+
result = series.astype(str).str.contains(str(value), na=False, regex=False)
|
|
118
|
+
elif op == "startswith":
|
|
119
|
+
result = series.astype(str).str.startswith(str(value), na=False)
|
|
120
|
+
elif op == "endswith":
|
|
121
|
+
result = series.astype(str).str.endswith(str(value), na=False)
|
|
122
|
+
elif op == "in":
|
|
123
|
+
values = value if isinstance(value, list) else [value]
|
|
124
|
+
str_values = [_norm_to_str(v) for v in values]
|
|
125
|
+
result = series.map(_norm_to_str).isin(str_values)
|
|
126
|
+
elif op == "not_in":
|
|
127
|
+
values = value if isinstance(value, list) else [value]
|
|
128
|
+
str_values = [_norm_to_str(v) for v in values]
|
|
129
|
+
result = ~series.map(_norm_to_str).isin(str_values)
|
|
130
|
+
else:
|
|
131
|
+
return pd.Series([False] * len(df), index=df.index)
|
|
132
|
+
|
|
133
|
+
# 空值行不参与条件匹配(对齐 SQL NULL 语义)
|
|
134
|
+
result = result.copy()
|
|
135
|
+
result[series.isna()] = False
|
|
136
|
+
return result
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class TransformRunner(ABC):
|
|
140
|
+
"""@classdesc Transform 运行器抽象基类
|
|
141
|
+
|
|
142
|
+
所有具体转换类型必须继承此类并实现 execute 方法。
|
|
143
|
+
|
|
144
|
+
使用示例:
|
|
145
|
+
class StringSplitRunner(TransformRunner):
|
|
146
|
+
def execute(self, df, input_column, params):
|
|
147
|
+
delimiter = params.get("delimiter", " ")
|
|
148
|
+
df["output"] = df[input_column].str.split(delimiter).str[0]
|
|
149
|
+
return df
|
|
150
|
+
"""
|
|
151
|
+
|
|
152
|
+
@abstractmethod
|
|
153
|
+
def execute(
|
|
154
|
+
self,
|
|
155
|
+
df: pd.DataFrame,
|
|
156
|
+
input_column: str,
|
|
157
|
+
params: dict[str, Any],
|
|
158
|
+
output_columns: list[str],
|
|
159
|
+
) -> pd.DataFrame:
|
|
160
|
+
"""@methoddesc 执行数据转换
|
|
161
|
+
|
|
162
|
+
参数:
|
|
163
|
+
df: 输入 DataFrame
|
|
164
|
+
input_column: 输入列名
|
|
165
|
+
params: 转换参数
|
|
166
|
+
output_columns: 输出列名列表
|
|
167
|
+
|
|
168
|
+
返回:
|
|
169
|
+
转换后的 DataFrame(可能包含新列)
|
|
170
|
+
"""
|
|
171
|
+
...
|
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2026 Precis Team
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
6
|
+
# you may not use this file except in compliance with the License.
|
|
7
|
+
# You may obtain a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
13
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
14
|
+
# See the License for the specific language governing permissions and
|
|
15
|
+
# limitations under the License.
|
|
16
|
+
"""
|
|
17
|
+
@fileoverview CastType 转换运行器
|
|
18
|
+
|
|
19
|
+
功能概述:
|
|
20
|
+
- 将列值转换为目标数据类型
|
|
21
|
+
- 支持 int、float、bool、datetime、string 五种目标类型
|
|
22
|
+
|
|
23
|
+
参数:
|
|
24
|
+
target_type: 目标类型 ("int"|"float"|"bool"|"datetime"|"string")
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
from typing import Any
|
|
30
|
+
|
|
31
|
+
import pandas as pd
|
|
32
|
+
|
|
33
|
+
from .base import TransformRunner
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
class CastTypeRunner(TransformRunner):
|
|
37
|
+
"""@classdesc 类型转换运行器"""
|
|
38
|
+
|
|
39
|
+
def execute(
|
|
40
|
+
self,
|
|
41
|
+
df: pd.DataFrame,
|
|
42
|
+
input_column: str,
|
|
43
|
+
params: dict[str, Any],
|
|
44
|
+
output_columns: list[str],
|
|
45
|
+
) -> pd.DataFrame:
|
|
46
|
+
"""
|
|
47
|
+
@methoddesc 执行 类型转换转换
|
|
48
|
+
|
|
49
|
+
业务用途:
|
|
50
|
+
- TransformRunner 协议的标准入口,由 transform 节点调用
|
|
51
|
+
- 读取 params 中的转换参数,对 input_column 应用转换,输出到 output_columns
|
|
52
|
+
|
|
53
|
+
参数:
|
|
54
|
+
df: 源 DataFrame
|
|
55
|
+
input_column: 输入列名
|
|
56
|
+
params: 转换参数字典
|
|
57
|
+
output_columns: 目标输出列名列表
|
|
58
|
+
|
|
59
|
+
返回:
|
|
60
|
+
转换后的 DataFrame
|
|
61
|
+
"""
|
|
62
|
+
target_type = params.get("target_type", "string")
|
|
63
|
+
|
|
64
|
+
if input_column not in df.columns:
|
|
65
|
+
raise ValueError(f"输入列不存在: {input_column}")
|
|
66
|
+
|
|
67
|
+
if not output_columns:
|
|
68
|
+
raise ValueError("CastType 需要至少一个 output_columns")
|
|
69
|
+
|
|
70
|
+
output_col = output_columns[0]
|
|
71
|
+
series = df[input_column].copy()
|
|
72
|
+
|
|
73
|
+
if target_type in ("int", "integer"):
|
|
74
|
+
# §1.15: fail-fast——原实现 try/except 吞掉 "cannot safely cast" 后静默保留 float,
|
|
75
|
+
# 用户以为已转整数,下游按整数假设配约束全错。不取截断语义(1.9→1 不允许)。
|
|
76
|
+
numeric = pd.to_numeric(series, errors="coerce")
|
|
77
|
+
# 非空但不可转数值("abc"),或可转但非整数(1.5)→ 报错并带首个代表值
|
|
78
|
+
bad_mask = series.notna() & (numeric.isna() | (numeric != numeric.round()))
|
|
79
|
+
bad_indices = series.index[bad_mask]
|
|
80
|
+
if len(bad_indices) > 0:
|
|
81
|
+
first = bad_indices[0]
|
|
82
|
+
raise ValueError(f"无法将值 '{series[first]}'(第 {first} 行)转换为 int:存在非整数或非数值的值")
|
|
83
|
+
# 转为可空整数类型,避免 NaN 导致 float 强转(整数浮点如 1.0 合法转入)
|
|
84
|
+
series = numeric.astype("Int64")
|
|
85
|
+
elif target_type == "float":
|
|
86
|
+
series = pd.to_numeric(series, errors="coerce")
|
|
87
|
+
elif target_type == "bool":
|
|
88
|
+
series = series.apply(self._to_bool)
|
|
89
|
+
elif target_type == "datetime":
|
|
90
|
+
series = pd.to_datetime(series, errors="coerce")
|
|
91
|
+
elif target_type == "string":
|
|
92
|
+
series = series.astype(str)
|
|
93
|
+
series = series.where(~pd.isna(df[input_column]), None)
|
|
94
|
+
else:
|
|
95
|
+
raise ValueError(f"不支持的目标类型: {target_type}")
|
|
96
|
+
|
|
97
|
+
df[output_col] = series
|
|
98
|
+
return df
|
|
99
|
+
|
|
100
|
+
@staticmethod
|
|
101
|
+
def _to_bool(value: Any) -> Any:
|
|
102
|
+
"""将单个值转换为布尔值。
|
|
103
|
+
|
|
104
|
+
支持常见的布尔值字符串表示。
|
|
105
|
+
空值(None/NaN)透传为 None,不参与类型判定——与 string 分支的
|
|
106
|
+
`where(~isna, None)` 口径一致,避免 bool(nan)=True 把空值断言为 True。
|
|
107
|
+
无法识别的值返回 None。
|
|
108
|
+
"""
|
|
109
|
+
if value is None or (isinstance(value, float) and value != value):
|
|
110
|
+
return None
|
|
111
|
+
if isinstance(value, bool):
|
|
112
|
+
return value
|
|
113
|
+
if isinstance(value, (int, float)):
|
|
114
|
+
return bool(value)
|
|
115
|
+
s = str(value).strip().lower()
|
|
116
|
+
if s in ("true", "1", "yes", "y", "on"):
|
|
117
|
+
return True
|
|
118
|
+
if s in ("false", "0", "no", "n", "off"):
|
|
119
|
+
return False
|
|
120
|
+
return None
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2026 Precis Team
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
6
|
+
# you may not use this file except in compliance with the License.
|
|
7
|
+
# You may obtain a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
13
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
14
|
+
# See the License for the specific language governing permissions and
|
|
15
|
+
# limitations under the License.
|
|
16
|
+
"""
|
|
17
|
+
@fileoverview Concat 转换运行器
|
|
18
|
+
|
|
19
|
+
功能概述:
|
|
20
|
+
- 将多个列的值拼接为一个新列
|
|
21
|
+
- 支持自定义分隔符
|
|
22
|
+
|
|
23
|
+
参数:
|
|
24
|
+
columns: 要拼接的列名,支持逗号分隔字符串(如 "first_name,last_name")或列表(如 ["first_name","last_name"])
|
|
25
|
+
separator: 分隔符(默认空字符串)
|
|
26
|
+
output_column: 输出列名(可选)
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
from typing import Any
|
|
32
|
+
|
|
33
|
+
import pandas as pd
|
|
34
|
+
|
|
35
|
+
from .base import TransformRunner, stringify_preserve_null
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class ConcatRunner(TransformRunner):
|
|
39
|
+
"""@classdesc 列拼接转换运行器"""
|
|
40
|
+
|
|
41
|
+
def execute(
|
|
42
|
+
self,
|
|
43
|
+
df: pd.DataFrame,
|
|
44
|
+
input_column: str,
|
|
45
|
+
params: dict[str, Any],
|
|
46
|
+
output_columns: list[str],
|
|
47
|
+
) -> pd.DataFrame:
|
|
48
|
+
"""
|
|
49
|
+
@methoddesc 执行 字符串拼接转换
|
|
50
|
+
|
|
51
|
+
业务用途:
|
|
52
|
+
- TransformRunner 协议的标准入口,由 transform 节点调用
|
|
53
|
+
- 读取 params 中的转换参数,对 input_column 应用转换,输出到 output_columns
|
|
54
|
+
|
|
55
|
+
参数:
|
|
56
|
+
df: 源 DataFrame
|
|
57
|
+
input_column: 输入列名
|
|
58
|
+
params: 转换参数字典
|
|
59
|
+
output_columns: 目标输出列名列表
|
|
60
|
+
|
|
61
|
+
返回:
|
|
62
|
+
转换后的 DataFrame
|
|
63
|
+
"""
|
|
64
|
+
columns_raw = params.get("columns", "")
|
|
65
|
+
separator = params.get("separator", "")
|
|
66
|
+
output_column = params.get("output_column", None)
|
|
67
|
+
|
|
68
|
+
# 解析列名列表:兼容字符串(逗号分隔)与列表两种格式
|
|
69
|
+
# 前端 TagsRenderer 产出的是数组,旧配置/手写 YAML 可能是逗号分隔字符串
|
|
70
|
+
if isinstance(columns_raw, list):
|
|
71
|
+
column_list = [str(col).strip() for col in columns_raw if str(col).strip()]
|
|
72
|
+
else:
|
|
73
|
+
column_list = [col.strip() for col in str(columns_raw).split(",") if col.strip()]
|
|
74
|
+
|
|
75
|
+
if not column_list:
|
|
76
|
+
raise ValueError("Concat 需要 columns 参数(要拼接的列名,逗号分隔)")
|
|
77
|
+
|
|
78
|
+
# 验证所有列都存在
|
|
79
|
+
for col in column_list:
|
|
80
|
+
if col not in df.columns:
|
|
81
|
+
raise ValueError(f"列不存在: {col}")
|
|
82
|
+
|
|
83
|
+
# 确定输出列名
|
|
84
|
+
if output_column:
|
|
85
|
+
output_col = output_column
|
|
86
|
+
elif output_columns:
|
|
87
|
+
output_col = output_columns[0]
|
|
88
|
+
else:
|
|
89
|
+
output_col = "concat_result"
|
|
90
|
+
|
|
91
|
+
# 执行拼接
|
|
92
|
+
if len(column_list) == 1:
|
|
93
|
+
# 单列情况:直接复制(空值透传为 None,不产生 "nan"/"None" 字面量)
|
|
94
|
+
df[output_col] = stringify_preserve_null(df[column_list[0]])
|
|
95
|
+
else:
|
|
96
|
+
# 多列情况:向量化拼接。任一源单元格为空 → 结果为 None
|
|
97
|
+
# (对齐 SQL CONCAT 的 NULL 语义),避免 astype(str) 拼出 "x-None" 之类的污染值
|
|
98
|
+
null_mask = df[column_list].isna().any(axis=1)
|
|
99
|
+
joined = df[column_list].astype(str).agg(separator.join, axis=1)
|
|
100
|
+
joined = joined.where(~null_mask, None)
|
|
101
|
+
df[output_col] = joined
|
|
102
|
+
|
|
103
|
+
return df
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
# SPDX-License-Identifier: Apache-2.0
|
|
2
|
+
#
|
|
3
|
+
# Copyright 2026 Precis Team
|
|
4
|
+
#
|
|
5
|
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
6
|
+
# you may not use this file except in compliance with the License.
|
|
7
|
+
# You may obtain a copy of the License at
|
|
8
|
+
#
|
|
9
|
+
# http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
#
|
|
11
|
+
# Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
13
|
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
14
|
+
# See the License for the specific language governing permissions and
|
|
15
|
+
# limitations under the License.
|
|
16
|
+
"""
|
|
17
|
+
@fileoverview ConditionalAssign 转换运行器
|
|
18
|
+
|
|
19
|
+
功能概述:
|
|
20
|
+
- 根据条件判断对列进行条件赋值
|
|
21
|
+
- 支持多条件组合(AND/OR 逻辑)
|
|
22
|
+
- 支持 eq/ne/gt/gte/lt/lte/contains/startswith/endswith/in/not_in/is_null/is_not_null 操作符
|
|
23
|
+
|
|
24
|
+
参数:
|
|
25
|
+
conditions: 条件列表 [{column, op, value}]
|
|
26
|
+
logic: 条件组合逻辑 ("and"|"or")
|
|
27
|
+
then_value: 条件满足时的赋值
|
|
28
|
+
else_value: 条件不满足时的赋值(可选,不提供则保留原值)
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
from typing import Any
|
|
34
|
+
|
|
35
|
+
import pandas as pd
|
|
36
|
+
|
|
37
|
+
from .base import TransformRunner, evaluate_condition
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class ConditionalAssignRunner(TransformRunner):
|
|
41
|
+
"""@classdesc 条件赋值转换运行器"""
|
|
42
|
+
|
|
43
|
+
def execute(
|
|
44
|
+
self,
|
|
45
|
+
df: pd.DataFrame,
|
|
46
|
+
input_column: str,
|
|
47
|
+
params: dict[str, Any],
|
|
48
|
+
output_columns: list[str],
|
|
49
|
+
) -> pd.DataFrame:
|
|
50
|
+
"""
|
|
51
|
+
@methoddesc 执行 条件赋值转换
|
|
52
|
+
|
|
53
|
+
业务用途:
|
|
54
|
+
- TransformRunner 协议的标准入口,由 transform 节点调用
|
|
55
|
+
- 读取 params 中的转换参数,对 input_column 应用转换,输出到 output_columns
|
|
56
|
+
|
|
57
|
+
参数:
|
|
58
|
+
df: 源 DataFrame
|
|
59
|
+
input_column: 输入列名
|
|
60
|
+
params: 转换参数字典
|
|
61
|
+
output_columns: 目标输出列名列表
|
|
62
|
+
|
|
63
|
+
返回:
|
|
64
|
+
转换后的 DataFrame
|
|
65
|
+
"""
|
|
66
|
+
conditions = params.get("conditions", [])
|
|
67
|
+
# logic 归一大小写不敏感("OR"/"And" 均合法),未知值 fail-fast——不再静默按 AND 组合
|
|
68
|
+
raw_logic = params.get("logic", "and")
|
|
69
|
+
logic = str(raw_logic).lower()
|
|
70
|
+
if logic not in ("and", "or"):
|
|
71
|
+
raise ValueError(f"未知的 logic '{raw_logic}',支持的值为: and, or")
|
|
72
|
+
# §1.18: then_value 缺省报配置错误——原实现缺省 "" 把命中行静默写空串。
|
|
73
|
+
# 显式空串 then_value: '' 是合法赋值(写空串),只拒绝 None。
|
|
74
|
+
if not output_columns:
|
|
75
|
+
raise ValueError("ConditionalAssign 需要至少一个 output_columns")
|
|
76
|
+
then_value = params.get("then_value")
|
|
77
|
+
if then_value is None:
|
|
78
|
+
raise ValueError("ConditionalAssign 缺少必填参数 then_value")
|
|
79
|
+
|
|
80
|
+
output_col = output_columns[0]
|
|
81
|
+
|
|
82
|
+
if not conditions:
|
|
83
|
+
# 无条件时直接赋 then_value
|
|
84
|
+
df[output_col] = then_value
|
|
85
|
+
return df
|
|
86
|
+
|
|
87
|
+
# 逐行评估所有条件,生成布尔掩码
|
|
88
|
+
masks = []
|
|
89
|
+
for cond in conditions:
|
|
90
|
+
mask = evaluate_condition(df, cond)
|
|
91
|
+
masks.append(mask)
|
|
92
|
+
|
|
93
|
+
# 组合条件
|
|
94
|
+
if logic == "or":
|
|
95
|
+
combined = masks[0]
|
|
96
|
+
for m in masks[1:]:
|
|
97
|
+
combined = combined | m
|
|
98
|
+
else:
|
|
99
|
+
combined = masks[0]
|
|
100
|
+
for m in masks[1:]:
|
|
101
|
+
combined = combined & m
|
|
102
|
+
|
|
103
|
+
# 赋值:输出列初始化为 object 类型以支持混合类型赋值(如 int 列赋 str 值)
|
|
104
|
+
if input_column in df.columns:
|
|
105
|
+
df[output_col] = df[input_column].copy().astype(object)
|
|
106
|
+
else:
|
|
107
|
+
df[output_col] = None
|
|
108
|
+
df.loc[combined, output_col] = then_value
|
|
109
|
+
# §1.18: 用键存在性区分"显式 null(置空)"与"未提供(保留原值)"——
|
|
110
|
+
# params.get 对两者都返回 None,原实现 else 分支永不执行,想置空做不到
|
|
111
|
+
if "else_value" in params:
|
|
112
|
+
df.loc[~combined, output_col] = params["else_value"]
|
|
113
|
+
|
|
114
|
+
return df
|