cehrbert-data 0.1.1__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/PKG-INFO +1 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/apps/generate_included_concept_list.py +5 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/apps/generate_training_data.py +21 -27
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/spark_app_base.py +10 -2
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/connect_omop_visit.py +5 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/convert_prediction_time_to_str.py +2 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/download_omop_tables.py +5 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/ehrshot_to_omop.py +5 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/extract_features.py +1 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/sample_omop_tables.py +1 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/update_omop_visit.py +5 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/utils/spark_utils.py +9 -13
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data.egg-info/PKG-INFO +1 -1
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data.egg-info/SOURCES.txt +2 -0
- cehrbert_data-0.2.0/src/cehrbert_data.egg-info/scm_file_list.json +141 -0
- cehrbert_data-0.2.0/src/cehrbert_data.egg-info/scm_version.json +8 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/.github/workflows/python-build.yml +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/.github/workflows/tests.yml +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/.gitignore +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/.pre-commit-config.yaml +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/LICENSE +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/README.md +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/pyproject.toml +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/.part-00000-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/.part-00003-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/.part-00010-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/part-00000-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/part-00003-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept/part-00010-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/.part-00000-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/.part-00002-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/.part-00006-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/.part-00011-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/.part-00013-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/part-00000-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/part-00002-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/part-00006-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/part-00011-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/part-00013-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/.part-00000-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/.part-00002-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/.part-00007-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/.part-00012-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/part-00000-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/part-00002-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/part-00007-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/part-00012-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/condition_occurrence/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/condition_occurrence/.part-00000-4eff03a1-cdcf-4c89-b0cd-9ce590b9b1eb-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/condition_occurrence/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/condition_occurrence/part-00000-4eff03a1-cdcf-4c89-b0cd-9ce590b9b1eb-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/drug_exposure/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/drug_exposure/.part-00000-10bbf1a4-a7da-416e-9703-58609c7edfad-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/drug_exposure/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/drug_exposure/part-00000-10bbf1a4-a7da-416e-9703-58609c7edfad-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/observation_period/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/observation_period/.part-00000-694316e5-cc95-49f1-9fad-5a7f377e2602-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/observation_period/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/observation_period/part-00000-694316e5-cc95-49f1-9fad-5a7f377e2602-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/person/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/person/.part-00000-7d789011-f361-48da-af6f-cfe102978b3a-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/person/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/person/part-00000-7d789011-f361-48da-af6f-cfe102978b3a-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/procedure_occurrence/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/procedure_occurrence/.part-00000-e73003c1-aed5-41c0-b2d4-eaccaccf044a-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/procedure_occurrence/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/procedure_occurrence/part-00000-e73003c1-aed5-41c0-b2d4-eaccaccf044a-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/visit_occurrence/._SUCCESS.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/visit_occurrence/.part-00000-e874b5f1-bf9e-4cb9-93bf-c309a47b0476-c000.snappy.parquet.crc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/visit_occurrence/_SUCCESS +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/visit_occurrence/part-00000-e874b5f1-bf9e-4cb9-93bf-c309a47b0476-c000.snappy.parquet +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/scripts/extract_features_bert.sh +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/scripts/extract_features_gpt.sh +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/scripts/process_cohorts.sh +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/setup.cfg +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/apps/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/atrial_fibrillation.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/cabg.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/coronary_artery_disease.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/covid.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/covid_inpatient.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/death.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/heart_failure.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/ischemic_stroke.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/last_visit_discharged_home.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/query_builder.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/type_two_diabietes.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/ventilation.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/config/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/config/output_names.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/const/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/const/__pycache__/__init__.cpython-311.pyc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/const/__pycache__/common.cpython-311.pyc +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/const/artificial_tokens.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/const/common.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/artificial_time_token_decorator.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/clinical_event_decorator.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/death_event_decorator.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/demographic_event_decorator.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/patient_event_decorator_base.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/prediction_token_decorator.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/token_priority.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/afib_ischemic_stroke.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/cad_cabg_cohort.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/cad_hf_cohort.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/copd_readmission.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/covid_death.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/covid_ventilation.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/discharge_home_death.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/hf_readmission.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/hospitalization.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/hospitalization_mortality.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/readmission.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/t2dm_hf_cohort.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/queries/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/queries/measurement_queries.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/convert_prediction_time_to_local.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/prepare_ehrshot_cohorts.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/utils/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/utils/logging_utils.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/utils/spark_parse_args.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/utils/vocab_utils.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data.egg-info/dependency_links.txt +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data.egg-info/requires.txt +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data.egg-info/top_level.txt +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/integration_tests/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/integration_tests/test_generate_training_data.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/integration_tests/test_hf_readmission.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/integration_tests/test_hf_readmission_cohort_meds.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/pyspark_test_base.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/unit_tests/__init__.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/unit_tests/test_ehrshot_to_omop.py +0 -0
- {cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/unit_tests/test_spark_utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cehrbert_data
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: The Spark ETL tools for generating the CEHR-BERT and CEHR-GPT pre-training and finetuning data
|
|
5
5
|
Author-email: Chao Pang <chaopang229@gmail.com>, Xinzhuo Jiang <xj2193@cumc.columbia.edu>, Krishna Kalluri <kk3326@cumc.columbia.edu>, Nishanth Parameshwar Pavinkurve <np2689@cumc.columbia.edu>, Karthik Natarajan <kn2174@cumc.columbia.edu>
|
|
6
6
|
License: MIT License
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/apps/generate_included_concept_list.py
RENAMED
|
@@ -53,7 +53,11 @@ def main(
|
|
|
53
53
|
The function processes patient event data across various domain tables, excludes low-frequency
|
|
54
54
|
concepts, and saves the filtered concepts to a specified output folder.
|
|
55
55
|
"""
|
|
56
|
-
spark =
|
|
56
|
+
spark = (
|
|
57
|
+
SparkSession.builder.appName("Generate concept list")
|
|
58
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
59
|
+
.getOrCreate()
|
|
60
|
+
)
|
|
57
61
|
|
|
58
62
|
# Exclude measurement from domain_table_list if exists because we need to process measurement
|
|
59
63
|
# in a different way
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/apps/generate_training_data.py
RENAMED
|
@@ -54,7 +54,11 @@ def main(
|
|
|
54
54
|
duplicate_records: bool = False,
|
|
55
55
|
disconnect_problem_list_records: bool = False,
|
|
56
56
|
):
|
|
57
|
-
spark =
|
|
57
|
+
spark = (
|
|
58
|
+
SparkSession.builder.appName("Generate CEHR-BERT Training Data")
|
|
59
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
60
|
+
.getOrCreate()
|
|
61
|
+
)
|
|
58
62
|
|
|
59
63
|
logger = logging.getLogger(__name__)
|
|
60
64
|
logger.info(
|
|
@@ -120,6 +124,7 @@ def main(
|
|
|
120
124
|
"person_id",
|
|
121
125
|
"discharged_to_concept_id",
|
|
122
126
|
)
|
|
127
|
+
|
|
123
128
|
person = preprocess_domain_table(spark, input_folder, PERSON)
|
|
124
129
|
birth_datetime_udf = F.coalesce("birth_datetime", F.concat("year_of_birth", F.lit("-01-01")).cast("timestamp"))
|
|
125
130
|
person = person.select(
|
|
@@ -128,22 +133,12 @@ def main(
|
|
|
128
133
|
"race_concept_id",
|
|
129
134
|
"gender_concept_id",
|
|
130
135
|
)
|
|
131
|
-
|
|
132
136
|
visit_occurrence_person = visit_occurrence.join(person, "person_id").withColumn(
|
|
133
137
|
"age",
|
|
134
138
|
F.ceil(F.months_between(F.col("visit_start_date"), F.col("birth_datetime")) / F.lit(12)),
|
|
135
139
|
)
|
|
136
140
|
visit_occurrence_person = visit_occurrence_person.drop("birth_datetime")
|
|
137
141
|
|
|
138
|
-
death = preprocess_domain_table(spark, input_folder, DEATH) if include_death else None
|
|
139
|
-
|
|
140
|
-
if include_concept_list and patient_ehr_events:
|
|
141
|
-
# Filter out concepts
|
|
142
|
-
qualified_concepts = preprocess_domain_table(spark, input_folder, "qualified_concept_list")
|
|
143
|
-
patient_ehr_events = patient_ehr_events.join(
|
|
144
|
-
qualified_concepts.select("standard_concept_id"), "standard_concept_id"
|
|
145
|
-
)
|
|
146
|
-
|
|
147
142
|
patient_ehr_events = (
|
|
148
143
|
patient_ehr_events.join(visit_occurrence_person, "visit_occurrence_id")
|
|
149
144
|
.select(
|
|
@@ -153,10 +148,12 @@ def main(
|
|
|
153
148
|
.withColumn("cohort_member_id", F.col("person_id"))
|
|
154
149
|
)
|
|
155
150
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
patient_ehr_events = patient_ehr_events.
|
|
151
|
+
if include_concept_list and patient_ehr_events:
|
|
152
|
+
# Filter out concepts
|
|
153
|
+
qualified_concepts = preprocess_domain_table(spark, input_folder, "qualified_concept_list")
|
|
154
|
+
patient_ehr_events = patient_ehr_events.join(
|
|
155
|
+
qualified_concepts.select("standard_concept_id"), "standard_concept_id"
|
|
156
|
+
)
|
|
160
157
|
|
|
161
158
|
if not continue_from_events:
|
|
162
159
|
patient_ehr_events.write.mode("overwrite").parquet(os.path.join(output_folder, "all_patient_events"))
|
|
@@ -164,24 +161,21 @@ def main(
|
|
|
164
161
|
patient_ehr_events = spark.read.parquet(os.path.join(output_folder, "all_patient_events"))
|
|
165
162
|
if should_construct_artificial_visits:
|
|
166
163
|
# Construct artificial visits or re-link the visits for the problem list events
|
|
167
|
-
patient_ehr_events,
|
|
164
|
+
patient_ehr_events, visit_occurrence = construct_artificial_visits(
|
|
168
165
|
patient_ehr_events,
|
|
169
|
-
|
|
166
|
+
visit_occurrence,
|
|
170
167
|
spark=spark,
|
|
171
168
|
persistence_folder=output_folder,
|
|
172
169
|
duplicate_records=duplicate_records,
|
|
173
170
|
disconnect_problem_list_records=disconnect_problem_list_records
|
|
174
171
|
)
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
"age",
|
|
183
|
-
F.ceil(F.months_between(F.col("visit_start_date"), F.col("birth_datetime")) / F.lit(12))
|
|
184
|
-
).drop("visit_start_date", "birth_datetime")
|
|
172
|
+
|
|
173
|
+
death = preprocess_domain_table(spark, input_folder, DEATH) if include_death else None
|
|
174
|
+
|
|
175
|
+
# Apply the age security measure
|
|
176
|
+
# We only keep the patient records, whose corresponding age is less than 90
|
|
177
|
+
if apply_age_filter:
|
|
178
|
+
patient_ehr_events = patient_ehr_events.where(F.col("age") < 90)
|
|
185
179
|
|
|
186
180
|
if is_new_patient_representation:
|
|
187
181
|
patient_sequence_data = create_sequence_data_with_att(
|
|
@@ -137,7 +137,11 @@ class BaseCohortBuilder(ABC):
|
|
|
137
137
|
# Validate if the data folders exist
|
|
138
138
|
validate_date_folder(self._input_folder, self._query_builder.get_dependency_list())
|
|
139
139
|
|
|
140
|
-
self.spark =
|
|
140
|
+
self.spark = (
|
|
141
|
+
SparkSession.builder.appName(f"Generate {self._query_builder.get_cohort_name()}")
|
|
142
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
143
|
+
.getOrCreate()
|
|
144
|
+
)
|
|
141
145
|
|
|
142
146
|
self._dependency_dict = instantiate_dependencies(
|
|
143
147
|
self.spark, self._input_folder, self._query_builder.get_dependency_list()
|
|
@@ -409,7 +413,11 @@ class NestedCohortBuilder:
|
|
|
409
413
|
f"disconnect_problem_list_records: {disconnect_problem_list_records}\n"
|
|
410
414
|
)
|
|
411
415
|
|
|
412
|
-
self.spark =
|
|
416
|
+
self.spark = (
|
|
417
|
+
SparkSession.builder.appName(f"Generate {self._cohort_name}")
|
|
418
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
419
|
+
.getOrCreate()
|
|
420
|
+
)
|
|
413
421
|
self._dependency_dict = instantiate_dependencies(self.spark, self._input_folder, DEFAULT_DEPENDENCY)
|
|
414
422
|
|
|
415
423
|
# Validate the input and output folders
|
|
@@ -226,7 +226,11 @@ def step_2_connect_outpatient_to_inpatient(
|
|
|
226
226
|
|
|
227
227
|
|
|
228
228
|
def main(args):
|
|
229
|
-
spark =
|
|
229
|
+
spark = (
|
|
230
|
+
SparkSession.builder.appName("Clean up visit_occurrence")
|
|
231
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
232
|
+
.getOrCreate()
|
|
233
|
+
)
|
|
230
234
|
visit_occurrence = spark.read.parquet(os.path.join(args.input_folder, "visit_occurrence"))
|
|
231
235
|
visit_occurrence_step_1, in_to_in_visit_mapping = step_1_consolidate_inpatient_visits(
|
|
232
236
|
spark,
|
|
@@ -26,9 +26,10 @@ def convert_file(file_pair):
|
|
|
26
26
|
# Read with Polars
|
|
27
27
|
df = pl.read_parquet(input_file)
|
|
28
28
|
|
|
29
|
-
# Convert datetime to string (ISO format)
|
|
29
|
+
# Convert datetime to string (ISO format) in UTC to avoid timezone issues
|
|
30
30
|
df = df.with_columns([
|
|
31
31
|
pl.col("prediction_time")
|
|
32
|
+
.dt.convert_time_zone("UTC")
|
|
32
33
|
.cast(pl.Datetime("us"))
|
|
33
34
|
.dt.strftime("%Y-%m-%d %H:%M:%S.%f")
|
|
34
35
|
.alias("prediction_time")
|
|
@@ -107,7 +107,11 @@ if __name__ == "__main__":
|
|
|
107
107
|
)
|
|
108
108
|
|
|
109
109
|
ARGS = parser.parse_args()
|
|
110
|
-
spark =
|
|
110
|
+
spark = (
|
|
111
|
+
SparkSession.builder.appName("Download OMOP tables")
|
|
112
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
113
|
+
.getOrCreate()
|
|
114
|
+
)
|
|
111
115
|
domain_table_list = ARGS.domain_table_list
|
|
112
116
|
credential_path = ARGS.credential_path
|
|
113
117
|
download_folder = ARGS.output_folder
|
|
@@ -784,7 +784,11 @@ def drop_duplicate_visits(data: DataFrame) -> DataFrame:
|
|
|
784
784
|
|
|
785
785
|
|
|
786
786
|
def main(args):
|
|
787
|
-
spark =
|
|
787
|
+
spark = (
|
|
788
|
+
SparkSession.builder.appName("Convert EHRShot Data")
|
|
789
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
790
|
+
.getOrCreate()
|
|
791
|
+
)
|
|
788
792
|
|
|
789
793
|
logger.info(
|
|
790
794
|
f"ehr_shot_file: {args.ehr_shot_file}\n"
|
|
@@ -23,6 +23,7 @@ def main(args):
|
|
|
23
23
|
.config("spark.sql.legacy.parquet.int96RebaseModeInWrite", "CORRECTED")
|
|
24
24
|
.config("spark.sql.legacy.parquet.datetimeRebaseModeInRead", "CORRECTED")
|
|
25
25
|
.config("spark.sql.legacy.parquet.datetimeRebaseModeInWrite", "CORRECTED")
|
|
26
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
26
27
|
.getOrCreate()
|
|
27
28
|
)
|
|
28
29
|
patient_sample = spark.read.parquet(args.person_sample)
|
|
@@ -7,7 +7,11 @@ from pyspark.sql import SparkSession
|
|
|
7
7
|
from pyspark.sql import functions as f
|
|
8
8
|
|
|
9
9
|
def main(args):
|
|
10
|
-
spark =
|
|
10
|
+
spark = (
|
|
11
|
+
SparkSession.builder.appName("Clean up visit_occurrence")
|
|
12
|
+
.config("spark.sql.session.timeZone", "UTC")
|
|
13
|
+
.getOrCreate()
|
|
14
|
+
)
|
|
11
15
|
visit_mapping = spark.read.parquet(
|
|
12
16
|
os.path.join(args.output_folder, "visit_mapping")
|
|
13
17
|
)
|
|
@@ -44,7 +44,7 @@ DOMAIN_KEY_FIELDS = {
|
|
|
44
44
|
"condition_concept_id",
|
|
45
45
|
"condition_start_date",
|
|
46
46
|
"condition_start_datetime",
|
|
47
|
-
"
|
|
47
|
+
"condition_occurrence"
|
|
48
48
|
)
|
|
49
49
|
],
|
|
50
50
|
"procedure_occurrence_id": [
|
|
@@ -52,7 +52,7 @@ DOMAIN_KEY_FIELDS = {
|
|
|
52
52
|
"procedure_concept_id",
|
|
53
53
|
"procedure_date",
|
|
54
54
|
"procedure_datetime",
|
|
55
|
-
"
|
|
55
|
+
"procedure_occurrence"
|
|
56
56
|
)
|
|
57
57
|
],
|
|
58
58
|
"drug_exposure_id": [
|
|
@@ -60,7 +60,7 @@ DOMAIN_KEY_FIELDS = {
|
|
|
60
60
|
"drug_concept_id",
|
|
61
61
|
"drug_exposure_start_date",
|
|
62
62
|
"drug_exposure_start_datetime",
|
|
63
|
-
"
|
|
63
|
+
"drug_exposure"
|
|
64
64
|
)
|
|
65
65
|
],
|
|
66
66
|
"measurement_id": [
|
|
@@ -84,13 +84,13 @@ DOMAIN_KEY_FIELDS = {
|
|
|
84
84
|
"device_concept_id",
|
|
85
85
|
"device_exposure_start_date",
|
|
86
86
|
"device_exposure_start_datetime",
|
|
87
|
-
"
|
|
87
|
+
"device_exposure"
|
|
88
88
|
)
|
|
89
89
|
],
|
|
90
90
|
"death_date": [("cause_concept_id", "death_date", "death_datetime", "death")],
|
|
91
91
|
"visit_concept_id": [
|
|
92
|
-
("visit_concept_id", "visit_start_date", "
|
|
93
|
-
("discharged_to_concept_id", "visit_end_date", "
|
|
92
|
+
("visit_concept_id", "visit_start_date", "visit_occurrence"),
|
|
93
|
+
("discharged_to_concept_id", "visit_end_date", "visit_occurrence"),
|
|
94
94
|
],
|
|
95
95
|
}
|
|
96
96
|
|
|
@@ -180,7 +180,7 @@ def extract_events_by_domain(
|
|
|
180
180
|
) in get_key_fields(domain_table):
|
|
181
181
|
|
|
182
182
|
if is_domain_numeric(domain_table_name):
|
|
183
|
-
concept = kwargs.get("concept")
|
|
183
|
+
concept: DataFrame = kwargs.get("concept")
|
|
184
184
|
spark = kwargs.get("spark", None)
|
|
185
185
|
persistence_folder = kwargs.get("persistence_folder", None)
|
|
186
186
|
refresh = kwargs.get("refresh_measurement", False)
|
|
@@ -210,7 +210,7 @@ def extract_events_by_domain(
|
|
|
210
210
|
domain_records = domain_table.where(F.col(date_field).isNotNull()).where(
|
|
211
211
|
F.col(concept_id_field).isNotNull()
|
|
212
212
|
)
|
|
213
|
-
datetime_field_udf = F.to_timestamp(F.coalesce(datetime_field, date_field)
|
|
213
|
+
datetime_field_udf = F.to_timestamp(F.coalesce(datetime_field, date_field))
|
|
214
214
|
domain_records = (
|
|
215
215
|
domain_records.where(F.col(concept_id_field).cast("string") != "0")
|
|
216
216
|
.withColumn("date", F.to_date(F.col(date_field)))
|
|
@@ -222,17 +222,13 @@ def extract_events_by_domain(
|
|
|
222
222
|
domain_records["date"].cast("date"),
|
|
223
223
|
domain_records["datetime"].cast(T.TimestampType()),
|
|
224
224
|
domain_records["visit_occurrence_id"],
|
|
225
|
-
F.lit(domain_table_name).alias("domain"),
|
|
225
|
+
F.lit(domain_table_name.split("_")[0]).alias("domain"),
|
|
226
226
|
F.lit(None).cast("string").alias("event_group_id"),
|
|
227
227
|
F.lit(None).cast("float").alias("number_as_value"),
|
|
228
228
|
F.lit(None).cast("string").alias("concept_as_value"),
|
|
229
229
|
F.col("unit") if domain_has_unit(domain_records) else F.lit(NA).alias("unit"),
|
|
230
230
|
).distinct()
|
|
231
231
|
|
|
232
|
-
# Remove "Patient Died" from condition_occurrence
|
|
233
|
-
if domain_table_name == "condition_occurrence":
|
|
234
|
-
domain_records = domain_records.where("condition_concept_id != 4216643")
|
|
235
|
-
|
|
236
232
|
if ehr_events is None:
|
|
237
233
|
ehr_events = domain_records
|
|
238
234
|
else:
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cehrbert_data
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: The Spark ETL tools for generating the CEHR-BERT and CEHR-GPT pre-training and finetuning data
|
|
5
5
|
Author-email: Chao Pang <chaopang229@gmail.com>, Xinzhuo Jiang <xj2193@cumc.columbia.edu>, Krishna Kalluri <kk3326@cumc.columbia.edu>, Nishanth Parameshwar Pavinkurve <np2689@cumc.columbia.edu>, Karthik Natarajan <kn2174@cumc.columbia.edu>
|
|
6
6
|
License: MIT License
|
|
@@ -68,6 +68,8 @@ src/cehrbert_data.egg-info/PKG-INFO
|
|
|
68
68
|
src/cehrbert_data.egg-info/SOURCES.txt
|
|
69
69
|
src/cehrbert_data.egg-info/dependency_links.txt
|
|
70
70
|
src/cehrbert_data.egg-info/requires.txt
|
|
71
|
+
src/cehrbert_data.egg-info/scm_file_list.json
|
|
72
|
+
src/cehrbert_data.egg-info/scm_version.json
|
|
71
73
|
src/cehrbert_data.egg-info/top_level.txt
|
|
72
74
|
src/cehrbert_data/apps/__init__.py
|
|
73
75
|
src/cehrbert_data/apps/generate_included_concept_list.py
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
{
|
|
2
|
+
"files": [
|
|
3
|
+
".github/workflows/python-build.yml",
|
|
4
|
+
".github/workflows/tests.yml",
|
|
5
|
+
".gitignore",
|
|
6
|
+
".pre-commit-config.yaml",
|
|
7
|
+
"LICENSE",
|
|
8
|
+
"README.md",
|
|
9
|
+
"pyproject.toml",
|
|
10
|
+
"sample_data/omop_sample/concept/._SUCCESS.crc",
|
|
11
|
+
"sample_data/omop_sample/concept/.part-00000-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet.crc",
|
|
12
|
+
"sample_data/omop_sample/concept/.part-00003-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet.crc",
|
|
13
|
+
"sample_data/omop_sample/concept/.part-00010-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet.crc",
|
|
14
|
+
"sample_data/omop_sample/concept/_SUCCESS",
|
|
15
|
+
"sample_data/omop_sample/concept/part-00000-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet",
|
|
16
|
+
"sample_data/omop_sample/concept/part-00003-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet",
|
|
17
|
+
"sample_data/omop_sample/concept/part-00010-4b12270c-f6c8-4b59-8e0f-fd588bd79386-c000.snappy.parquet",
|
|
18
|
+
"sample_data/omop_sample/concept_ancestor/._SUCCESS.crc",
|
|
19
|
+
"sample_data/omop_sample/concept_ancestor/.part-00000-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc",
|
|
20
|
+
"sample_data/omop_sample/concept_ancestor/.part-00002-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc",
|
|
21
|
+
"sample_data/omop_sample/concept_ancestor/.part-00006-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc",
|
|
22
|
+
"sample_data/omop_sample/concept_ancestor/.part-00011-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc",
|
|
23
|
+
"sample_data/omop_sample/concept_ancestor/.part-00013-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet.crc",
|
|
24
|
+
"sample_data/omop_sample/concept_ancestor/_SUCCESS",
|
|
25
|
+
"sample_data/omop_sample/concept_ancestor/part-00000-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet",
|
|
26
|
+
"sample_data/omop_sample/concept_ancestor/part-00002-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet",
|
|
27
|
+
"sample_data/omop_sample/concept_ancestor/part-00006-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet",
|
|
28
|
+
"sample_data/omop_sample/concept_ancestor/part-00011-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet",
|
|
29
|
+
"sample_data/omop_sample/concept_ancestor/part-00013-eafbd8be-3337-46da-89d3-20f79c2565d4-c000.snappy.parquet",
|
|
30
|
+
"sample_data/omop_sample/concept_relationship/._SUCCESS.crc",
|
|
31
|
+
"sample_data/omop_sample/concept_relationship/.part-00000-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc",
|
|
32
|
+
"sample_data/omop_sample/concept_relationship/.part-00002-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc",
|
|
33
|
+
"sample_data/omop_sample/concept_relationship/.part-00007-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc",
|
|
34
|
+
"sample_data/omop_sample/concept_relationship/.part-00012-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet.crc",
|
|
35
|
+
"sample_data/omop_sample/concept_relationship/_SUCCESS",
|
|
36
|
+
"sample_data/omop_sample/concept_relationship/part-00000-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet",
|
|
37
|
+
"sample_data/omop_sample/concept_relationship/part-00002-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet",
|
|
38
|
+
"sample_data/omop_sample/concept_relationship/part-00007-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet",
|
|
39
|
+
"sample_data/omop_sample/concept_relationship/part-00012-5752b472-8ba7-4189-ab69-8c92e46443aa-c000.snappy.parquet",
|
|
40
|
+
"sample_data/omop_sample/condition_occurrence/._SUCCESS.crc",
|
|
41
|
+
"sample_data/omop_sample/condition_occurrence/.part-00000-4eff03a1-cdcf-4c89-b0cd-9ce590b9b1eb-c000.snappy.parquet.crc",
|
|
42
|
+
"sample_data/omop_sample/condition_occurrence/_SUCCESS",
|
|
43
|
+
"sample_data/omop_sample/condition_occurrence/part-00000-4eff03a1-cdcf-4c89-b0cd-9ce590b9b1eb-c000.snappy.parquet",
|
|
44
|
+
"sample_data/omop_sample/drug_exposure/._SUCCESS.crc",
|
|
45
|
+
"sample_data/omop_sample/drug_exposure/.part-00000-10bbf1a4-a7da-416e-9703-58609c7edfad-c000.snappy.parquet.crc",
|
|
46
|
+
"sample_data/omop_sample/drug_exposure/_SUCCESS",
|
|
47
|
+
"sample_data/omop_sample/drug_exposure/part-00000-10bbf1a4-a7da-416e-9703-58609c7edfad-c000.snappy.parquet",
|
|
48
|
+
"sample_data/omop_sample/observation_period/._SUCCESS.crc",
|
|
49
|
+
"sample_data/omop_sample/observation_period/.part-00000-694316e5-cc95-49f1-9fad-5a7f377e2602-c000.snappy.parquet.crc",
|
|
50
|
+
"sample_data/omop_sample/observation_period/_SUCCESS",
|
|
51
|
+
"sample_data/omop_sample/observation_period/part-00000-694316e5-cc95-49f1-9fad-5a7f377e2602-c000.snappy.parquet",
|
|
52
|
+
"sample_data/omop_sample/person/._SUCCESS.crc",
|
|
53
|
+
"sample_data/omop_sample/person/.part-00000-7d789011-f361-48da-af6f-cfe102978b3a-c000.snappy.parquet.crc",
|
|
54
|
+
"sample_data/omop_sample/person/_SUCCESS",
|
|
55
|
+
"sample_data/omop_sample/person/part-00000-7d789011-f361-48da-af6f-cfe102978b3a-c000.snappy.parquet",
|
|
56
|
+
"sample_data/omop_sample/procedure_occurrence/._SUCCESS.crc",
|
|
57
|
+
"sample_data/omop_sample/procedure_occurrence/.part-00000-e73003c1-aed5-41c0-b2d4-eaccaccf044a-c000.snappy.parquet.crc",
|
|
58
|
+
"sample_data/omop_sample/procedure_occurrence/_SUCCESS",
|
|
59
|
+
"sample_data/omop_sample/procedure_occurrence/part-00000-e73003c1-aed5-41c0-b2d4-eaccaccf044a-c000.snappy.parquet",
|
|
60
|
+
"sample_data/omop_sample/visit_occurrence/._SUCCESS.crc",
|
|
61
|
+
"sample_data/omop_sample/visit_occurrence/.part-00000-e874b5f1-bf9e-4cb9-93bf-c309a47b0476-c000.snappy.parquet.crc",
|
|
62
|
+
"sample_data/omop_sample/visit_occurrence/_SUCCESS",
|
|
63
|
+
"sample_data/omop_sample/visit_occurrence/part-00000-e874b5f1-bf9e-4cb9-93bf-c309a47b0476-c000.snappy.parquet",
|
|
64
|
+
"scripts/extract_features_bert.sh",
|
|
65
|
+
"scripts/extract_features_gpt.sh",
|
|
66
|
+
"scripts/process_cohorts.sh",
|
|
67
|
+
"src/__init__.py",
|
|
68
|
+
"src/cehrbert_data/__init__.py",
|
|
69
|
+
"src/cehrbert_data/apps/__init__.py",
|
|
70
|
+
"src/cehrbert_data/apps/generate_included_concept_list.py",
|
|
71
|
+
"src/cehrbert_data/apps/generate_training_data.py",
|
|
72
|
+
"src/cehrbert_data/cohorts/__init__.py",
|
|
73
|
+
"src/cehrbert_data/cohorts/atrial_fibrillation.py",
|
|
74
|
+
"src/cehrbert_data/cohorts/cabg.py",
|
|
75
|
+
"src/cehrbert_data/cohorts/coronary_artery_disease.py",
|
|
76
|
+
"src/cehrbert_data/cohorts/covid.py",
|
|
77
|
+
"src/cehrbert_data/cohorts/covid_inpatient.py",
|
|
78
|
+
"src/cehrbert_data/cohorts/death.py",
|
|
79
|
+
"src/cehrbert_data/cohorts/heart_failure.py",
|
|
80
|
+
"src/cehrbert_data/cohorts/ischemic_stroke.py",
|
|
81
|
+
"src/cehrbert_data/cohorts/last_visit_discharged_home.py",
|
|
82
|
+
"src/cehrbert_data/cohorts/query_builder.py",
|
|
83
|
+
"src/cehrbert_data/cohorts/spark_app_base.py",
|
|
84
|
+
"src/cehrbert_data/cohorts/type_two_diabietes.py",
|
|
85
|
+
"src/cehrbert_data/cohorts/ventilation.py",
|
|
86
|
+
"src/cehrbert_data/config/__init__.py",
|
|
87
|
+
"src/cehrbert_data/config/output_names.py",
|
|
88
|
+
"src/cehrbert_data/const/__init__.py",
|
|
89
|
+
"src/cehrbert_data/const/__pycache__/__init__.cpython-311.pyc",
|
|
90
|
+
"src/cehrbert_data/const/__pycache__/common.cpython-311.pyc",
|
|
91
|
+
"src/cehrbert_data/const/artificial_tokens.py",
|
|
92
|
+
"src/cehrbert_data/const/common.py",
|
|
93
|
+
"src/cehrbert_data/decorators/__init__.py",
|
|
94
|
+
"src/cehrbert_data/decorators/artificial_time_token_decorator.py",
|
|
95
|
+
"src/cehrbert_data/decorators/clinical_event_decorator.py",
|
|
96
|
+
"src/cehrbert_data/decorators/death_event_decorator.py",
|
|
97
|
+
"src/cehrbert_data/decorators/demographic_event_decorator.py",
|
|
98
|
+
"src/cehrbert_data/decorators/patient_event_decorator_base.py",
|
|
99
|
+
"src/cehrbert_data/decorators/prediction_token_decorator.py",
|
|
100
|
+
"src/cehrbert_data/decorators/token_priority.py",
|
|
101
|
+
"src/cehrbert_data/prediction_cohorts/__init__.py",
|
|
102
|
+
"src/cehrbert_data/prediction_cohorts/afib_ischemic_stroke.py",
|
|
103
|
+
"src/cehrbert_data/prediction_cohorts/cad_cabg_cohort.py",
|
|
104
|
+
"src/cehrbert_data/prediction_cohorts/cad_hf_cohort.py",
|
|
105
|
+
"src/cehrbert_data/prediction_cohorts/copd_readmission.py",
|
|
106
|
+
"src/cehrbert_data/prediction_cohorts/covid_death.py",
|
|
107
|
+
"src/cehrbert_data/prediction_cohorts/covid_ventilation.py",
|
|
108
|
+
"src/cehrbert_data/prediction_cohorts/discharge_home_death.py",
|
|
109
|
+
"src/cehrbert_data/prediction_cohorts/hf_readmission.py",
|
|
110
|
+
"src/cehrbert_data/prediction_cohorts/hospitalization.py",
|
|
111
|
+
"src/cehrbert_data/prediction_cohorts/hospitalization_mortality.py",
|
|
112
|
+
"src/cehrbert_data/prediction_cohorts/readmission.py",
|
|
113
|
+
"src/cehrbert_data/prediction_cohorts/t2dm_hf_cohort.py",
|
|
114
|
+
"src/cehrbert_data/queries/__init__.py",
|
|
115
|
+
"src/cehrbert_data/queries/measurement_queries.py",
|
|
116
|
+
"src/cehrbert_data/tools/__init__.py",
|
|
117
|
+
"src/cehrbert_data/tools/connect_omop_visit.py",
|
|
118
|
+
"src/cehrbert_data/tools/convert_prediction_time_to_local.py",
|
|
119
|
+
"src/cehrbert_data/tools/convert_prediction_time_to_str.py",
|
|
120
|
+
"src/cehrbert_data/tools/download_omop_tables.py",
|
|
121
|
+
"src/cehrbert_data/tools/ehrshot_to_omop.py",
|
|
122
|
+
"src/cehrbert_data/tools/extract_features.py",
|
|
123
|
+
"src/cehrbert_data/tools/prepare_ehrshot_cohorts.py",
|
|
124
|
+
"src/cehrbert_data/tools/sample_omop_tables.py",
|
|
125
|
+
"src/cehrbert_data/tools/update_omop_visit.py",
|
|
126
|
+
"src/cehrbert_data/utils/__init__.py",
|
|
127
|
+
"src/cehrbert_data/utils/logging_utils.py",
|
|
128
|
+
"src/cehrbert_data/utils/spark_parse_args.py",
|
|
129
|
+
"src/cehrbert_data/utils/spark_utils.py",
|
|
130
|
+
"src/cehrbert_data/utils/vocab_utils.py",
|
|
131
|
+
"tests/__init__.py",
|
|
132
|
+
"tests/integration_tests/__init__.py",
|
|
133
|
+
"tests/integration_tests/test_generate_training_data.py",
|
|
134
|
+
"tests/integration_tests/test_hf_readmission.py",
|
|
135
|
+
"tests/integration_tests/test_hf_readmission_cohort_meds.py",
|
|
136
|
+
"tests/pyspark_test_base.py",
|
|
137
|
+
"tests/unit_tests/__init__.py",
|
|
138
|
+
"tests/unit_tests/test_ehrshot_to_omop.py",
|
|
139
|
+
"tests/unit_tests/test_spark_utils.py"
|
|
140
|
+
]
|
|
141
|
+
}
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/._SUCCESS.crc
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_ancestor/_SUCCESS
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/concept_relationship/_SUCCESS
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/condition_occurrence/_SUCCESS
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/drug_exposure/._SUCCESS.crc
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/observation_period/._SUCCESS.crc
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/observation_period/_SUCCESS
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/procedure_occurrence/_SUCCESS
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/visit_occurrence/._SUCCESS.crc
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/sample_data/omop_sample/visit_occurrence/_SUCCESS
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/atrial_fibrillation.py
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/coronary_artery_disease.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/cohorts/last_visit_discharged_home.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/clinical_event_decorator.py
RENAMED
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/decorators/death_event_decorator.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/cad_cabg_cohort.py
RENAMED
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/cad_hf_cohort.py
RENAMED
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/copd_readmission.py
RENAMED
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/covid_death.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/hf_readmission.py
RENAMED
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/hospitalization.py
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/readmission.py
RENAMED
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/prediction_cohorts/t2dm_hf_cohort.py
RENAMED
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/queries/measurement_queries.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/src/cehrbert_data/tools/prepare_ehrshot_cohorts.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cehrbert_data-0.1.1 → cehrbert_data-0.2.0}/tests/integration_tests/test_generate_training_data.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|