ltc-code 0.2.28__tar.gz → 0.2.30__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ltc_code-0.2.28 → ltc_code-0.2.30}/PKG-INFO +1 -1
- {ltc_code-0.2.28 → ltc_code-0.2.30}/pyproject.toml +5 -1
- ltc_code-0.2.30/src/ltc_code/nsc/NEW_CROSSWALK.md +26 -0
- ltc_code-0.2.30/src/ltc_code/nsc/OUTCOMES.md +59 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/build_nsc_outcomes.py +74 -128
- ltc_code-0.2.30/src/ltc_code/nsc/build_nsc_outcomes_new.py +2084 -0
- ltc_code-0.2.30/src/ltc_code/nsc/naics.csv +25 -0
- ltc_code-0.2.30/src/ltc_code/nsc/naics_raw.xlsx +0 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +7647 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx +0 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
- ltc_code-0.2.30/src/ltc_code/nsc/raw/directory.dta +0 -0
- ltc_code-0.2.28/pyproject.toml.orig +0 -19
- ltc_code-0.2.28/src/ltc_code/nsc/input/.gitkeep +0 -0
- ltc_code-0.2.28/src/ltc_code/nsc/output/.gitkeep +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/README.md +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/__init__.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/aspire.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/check_cmo_apps.do +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/christel_house.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/green_dot.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/helpers.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/june13.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/june2.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/june30.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/june5.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/june7.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/kipp_nj.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/kipp_tx.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/main.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/make_summary_stats_table.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/mappings.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/may27.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/__init__.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/dhs_stem/extract_dhs_stem_cips.R +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/dhs_stem/stemList2024.pdf +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/naics.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/plot_bars.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/polars_dates.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/rocketship.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/schema_mapping.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/signal_var_calcs.py +0 -0
- {ltc_code-0.2.28 → ltc_code-0.2.30}/src/ltc_code/yes_prep.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "ltc-code"
|
|
3
|
-
version = "0.2.
|
|
3
|
+
version = "0.2.30"
|
|
4
4
|
description = "Add your description here"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.9"
|
|
@@ -17,3 +17,7 @@ ltc-summary-stats = "ltc_code.make_summary_stats_table:main"
|
|
|
17
17
|
[build-system]
|
|
18
18
|
requires = ["uv_build>=0.11.16,<0.12.0"]
|
|
19
19
|
build-backend = "uv_build"
|
|
20
|
+
|
|
21
|
+
[tool.uv.build-backend]
|
|
22
|
+
source-exclude = ["**/.DS_Store", "**/__pycache__/**", "**/*.pyc", "src/ltc_code/nsc/input/**", "src/ltc_code/nsc/output/**"]
|
|
23
|
+
wheel-exclude = ["**/.DS_Store", "**/__pycache__/**", "**/*.pyc", "ltc_code/nsc/input/**", "ltc_code/nsc/output/**"]
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# New NSC crosswalk build
|
|
2
|
+
|
|
3
|
+
Run the new variant explicitly with `python -m ltc_code.nsc.build_nsc_outcomes_new` after providing the student input files. The original build remains available. The new variant writes to `nsc/output/new_crosswalk/` to keep its outputs separate.
|
|
4
|
+
|
|
5
|
+
The official April 2023 NSC crosswalk is primary. Append mapped legacy codes only when their full lookup key is absent from the official file. Do not fill the 108 present-but-blank official keys using Sarah's mappings. Existing 18 sole-rate candidate choices and 24 Calhoun branch assignments are explicit and documented in the code. Preserve full branch codes and deterministic legacy selection.
|
|
6
|
+
|
|
7
|
+
Institution type uses IPEDS 2013, then the latest nonmissing directory classification, then three inline exceptions formerly loaded from the manual spreadsheet. Less-than-two-year labels are parsed before broader two-year labels. Treat Missing/not reported as null before choosing the latest directory record. The existing two-year-or-less outcome grouping is retained. The 5–8 tier bucket lookup counts Nevada State once despite its tier 5/7 aliases.
|
|
8
|
+
|
|
9
|
+
All student outcome formulas, eligibility windows, completion-rate imputation, and the existing within-bucket coarse-rate exercise remain unchanged. The change affects their institutional inputs. No annual enrollment-year classification is introduced.
|
|
10
|
+
|
|
11
|
+
Expected counts with the audited files: 24,166 keys; 21,315 with UNITID; 21,204 matching a UNITID in the 2013/directory files (99.479%); 7,867 matched institutions with no missing type. Distinct institution types: 3,329 four-year, 2,440 two-year, 2,098 less-than-two-year. These are reference-table counts, not secure-sample match rates or completeness of completion rates.
|
|
12
|
+
|
|
13
|
+
Required files in `nsc/raw/`:
|
|
14
|
+
- NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx (official workbook, NSC_to_IPEDS_UNIT_ID sheet)
|
|
15
|
+
- college_crosswalk.xls (legacy fallback)
|
|
16
|
+
- IPEDS_IC_2013.csv
|
|
17
|
+
- directory.dta
|
|
18
|
+
- CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx
|
|
19
|
+
- ipeds_data.dta
|
|
20
|
+
- chetty/mrc_table2.dta and chetty/mrc_table11.dta
|
|
21
|
+
|
|
22
|
+
The old IPEDS_IC_manual.xlsx is not needed by this variant. Student inputs remain input/apps.csv, input/nsc_records_old.dta, and input/nsc_records_new.csv. Do not supply synthetic records for actual analysis. Institutional support files are bundled in the locally built wheel and source distribution but are Git-ignored. GitHub Actions publishes these verified distributions from release assets; it does not rebuild from a checkout that lacks the data. Student inputs are never bundled.
|
|
23
|
+
|
|
24
|
+
Intermediate diagnostic CSVs and printed audit tables are not generated. Only the full and selected final outcome files are written, each in Parquet and CSV format. The 99.48% matched subset is an audit denominator, not a filter that drops student records from the build.
|
|
25
|
+
|
|
26
|
+
Version 0.2.30 adds final audit exports and selectivity/BA sensitivity outcomes to both builds; see [OUTCOMES.md](OUTCOMES.md) for names and definitions.
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# NSC outcomes added in 0.2.30
|
|
2
|
+
|
|
3
|
+
Both `build_nsc_outcomes.py` and `build_nsc_outcomes_new.py` provide these fields.
|
|
4
|
+
The new-crosswalk variant remains an explicit separate build.
|
|
5
|
+
|
|
6
|
+
## Final institution audit fields
|
|
7
|
+
|
|
8
|
+
`nsc_outcomes_final` retains `ID_FSC_firstinst`, `college_name_firstinst`,
|
|
9
|
+
`unitid_firstinst`, `tier_firstinst`, `college_years_firstinst`, and
|
|
10
|
+
`completion_rate_150pct_firstinst`. Institution years are 4 (four-year),
|
|
11
|
+
2 (two-year), 1 (less than two-year), or null if unknown.
|
|
12
|
+
|
|
13
|
+
The first institution remains the earliest retained enrollment spell starting
|
|
14
|
+
on or after July 1 of the year the student turns 18, with the existing tie rule.
|
|
15
|
+
It need not meet the attendance-status rule or fall within the by-year-four
|
|
16
|
+
window. Attendance indicators describe qualifying enrollment in their window;
|
|
17
|
+
they do not describe the first institution. A transfer can therefore have a
|
|
18
|
+
two-year first institution and four-year attendance. Prediction formulas and
|
|
19
|
+
first-institution selection have not changed.
|
|
20
|
+
|
|
21
|
+
The build writes the selected table to `nsc_outcomes_final.parquet` and `.csv`
|
|
22
|
+
alongside the existing full `nsc_outcomes.parquet` and `.csv` outputs. These
|
|
23
|
+
are the only four files written; intermediate diagnostic exports and printed
|
|
24
|
+
audit tables have been removed.
|
|
25
|
+
|
|
26
|
+
## Selectivity outcomes
|
|
27
|
+
|
|
28
|
+
Selective means tiers 3–4. The existing `elite` name means highly selective,
|
|
29
|
+
tiers 1–2. These institution groups do not overlap. A student can attend or
|
|
30
|
+
earn degrees at institutions in both groups, so their student indicators
|
|
31
|
+
are not forced to be mutually exclusive.
|
|
32
|
+
|
|
33
|
+
| Example | Meaning |
|
|
34
|
+
| --- | --- |
|
|
35
|
+
| `att_selective_byY4` | Qualifying attendance at a tier 3–4 institution by year 4 |
|
|
36
|
+
| `att_elite_byY4` | Qualifying attendance at a tier 1–2 institution by year 4 |
|
|
37
|
+
| `cmp_selective_byY8` | AA or BA from a tier 3–4 institution by year 8 |
|
|
38
|
+
| `cmp_elite_byY8` | Existing AA-or-BA measure for tiers 1–2 |
|
|
39
|
+
| `cmp_BA_selective_byY8` | BA from a tier 3–4 institution by year 8 |
|
|
40
|
+
| `cmp_BA_elite_byY8` | BA from a tier 1–2 institution by year 8 |
|
|
41
|
+
| `cmp_BA_noimpute_byY8` | Any BA without institution-type credential imputation |
|
|
42
|
+
| `cmp_BA_selective_noimpute_byY8` | Same sensitivity definition for tiers 3–4 |
|
|
43
|
+
| `cmp_BA_elite_noimpute_byY8` | Same sensitivity definition for tiers 1–2 |
|
|
44
|
+
|
|
45
|
+
Attendance is available for `inY1`–`inY8`, `byY1`–`byY8`, their `fall` and
|
|
46
|
+
`spring` variants, and calendar ages 18–26 (e.g. `att_selective_20`). Completion
|
|
47
|
+
is available for `byY1`–`byY8`. All these selectivity and sensitivity fields are
|
|
48
|
+
retained in the selected final table. They follow the existing observation
|
|
49
|
+
windows and zero/null rules. Unknown institutional tiers do not count as
|
|
50
|
+
selective. Completion uses the degree-granting institution's tier, not the
|
|
51
|
+
first institution's tier.
|
|
52
|
+
|
|
53
|
+
The default BA measures retain the existing institution-type imputation.
|
|
54
|
+
The `noimpute` versions use credential-lookup/title evidence before that step:
|
|
55
|
+
an unknown award at a four-year institution alone does not count as a BA.
|
|
56
|
+
Students remain in the sample, and another identified BA can still qualify.
|
|
57
|
+
This does not change institutional IPEDS rates or coarse predictions.
|
|
58
|
+
|
|
59
|
+
Verification: run `uv run python tests/verify_nsc_outcomes.py` from the repository root with the local institutional support files present. This creates temporary synthetic records and does not require secured student inputs.
|
|
@@ -93,8 +93,8 @@ RANDOM_SEED = 3852804
|
|
|
93
93
|
SARAH_STEM_CIP_FAMILIES = {"11", "14", "15", "26", "27", "40", "41"}
|
|
94
94
|
DHS_STEM_CIP_FAMILIES = {"14", "26", "27", "40"}
|
|
95
95
|
STEM_DEFINITIONS = ["sarah", "dhs"]
|
|
96
|
-
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
|
|
97
|
-
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
|
|
96
|
+
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
97
|
+
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
98
98
|
AGE_ATTENDANCE_RANGE = range(18, 27)
|
|
99
99
|
|
|
100
100
|
# These ordered rules exactly reproduce Sarah's Stata replacements. Every rule
|
|
@@ -573,7 +573,9 @@ college_ref = (
|
|
|
573
573
|
.with_columns(
|
|
574
574
|
(pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
|
|
575
575
|
pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
|
|
576
|
+
# Institution groups: highly selective (elite) is tiers 1–2; selective is 3–4.
|
|
576
577
|
pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
|
|
578
|
+
pl.col("tier").is_in([3, 4]).cast(pl.Int8).alias("college_selective"),
|
|
577
579
|
)
|
|
578
580
|
.with_columns(
|
|
579
581
|
(pl.col("college_4yr") + pl.col("college_2yr") > 0)
|
|
@@ -592,6 +594,7 @@ college_ref = (
|
|
|
592
594
|
"college_2yr",
|
|
593
595
|
"tier",
|
|
594
596
|
"college_elite",
|
|
597
|
+
"college_selective",
|
|
595
598
|
"k_mean",
|
|
596
599
|
"completion_rate_150pct_ip",
|
|
597
600
|
)
|
|
@@ -1061,6 +1064,7 @@ enroll_weeks = (
|
|
|
1061
1064
|
"college_2yr",
|
|
1062
1065
|
"tier",
|
|
1063
1066
|
"college_elite",
|
|
1067
|
+
"college_selective",
|
|
1064
1068
|
"k_mean",
|
|
1065
1069
|
"completion_rate_150pct_ip",
|
|
1066
1070
|
"term_start_date",
|
|
@@ -1131,6 +1135,7 @@ enroll = (
|
|
|
1131
1135
|
pl.col("college_2yr").max(),
|
|
1132
1136
|
pl.col("tier").drop_nulls().min(),
|
|
1133
1137
|
pl.col("college_elite").max(),
|
|
1138
|
+
pl.col("college_selective").max(),
|
|
1134
1139
|
pl.col("k_mean").drop_nulls().first(),
|
|
1135
1140
|
pl.col("completion_rate_150pct_ip").drop_nulls().first(),
|
|
1136
1141
|
pl.col("_term_att").max(),
|
|
@@ -1453,6 +1458,8 @@ degrees = (
|
|
|
1453
1458
|
.otherwise(pl.lit(None))
|
|
1454
1459
|
.alias("_degree")
|
|
1455
1460
|
)
|
|
1461
|
+
# Preserve credential/title evidence before filling unknown awards by sector.
|
|
1462
|
+
.with_columns(pl.col("_degree").alias("_degree_no_type_fill"))
|
|
1456
1463
|
.with_columns(
|
|
1457
1464
|
pl.when(pl.col("_degree").is_null() & (pl.col("college_years") == 4))
|
|
1458
1465
|
.then(pl.lit("BA"))
|
|
@@ -1504,6 +1511,12 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1504
1511
|
(pl.col("_by_end") * (pl.col("_degree") == "BA").cast(pl.Int8))
|
|
1505
1512
|
.max()
|
|
1506
1513
|
.alias(f"cmp_BA_byY{year}"),
|
|
1514
|
+
(
|
|
1515
|
+
pl.col("_by_end")
|
|
1516
|
+
* (pl.col("_degree_no_type_fill") == "BA").fill_null(False).cast(pl.Int8)
|
|
1517
|
+
)
|
|
1518
|
+
.max()
|
|
1519
|
+
.alias(f"cmp_BA_noimpute_byY{year}"),
|
|
1507
1520
|
(pl.col("_by_end") * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8))
|
|
1508
1521
|
.max()
|
|
1509
1522
|
.alias(f"cmp_any_byY{year}"),
|
|
@@ -1514,6 +1527,28 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1514
1527
|
)
|
|
1515
1528
|
.max()
|
|
1516
1529
|
.alias(f"cmp_elite_byY{year}"),
|
|
1530
|
+
(
|
|
1531
|
+
pl.col("_by_end")
|
|
1532
|
+
* pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8)
|
|
1533
|
+
* pl.col("college_selective").fill_null(0).cast(pl.Int8)
|
|
1534
|
+
)
|
|
1535
|
+
.max()
|
|
1536
|
+
.alias(f"cmp_selective_byY{year}"),
|
|
1537
|
+
# BA-only outcomes support comparisons with any-BA completion.
|
|
1538
|
+
*[
|
|
1539
|
+
(
|
|
1540
|
+
pl.col("_by_end")
|
|
1541
|
+
* (pl.col(degree_column) == "BA").fill_null(False).cast(pl.Int8)
|
|
1542
|
+
* pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
|
|
1543
|
+
)
|
|
1544
|
+
.max()
|
|
1545
|
+
.alias(f"cmp_BA_{college_type}{suffix}_byY{year}")
|
|
1546
|
+
for college_type in ["selective", "elite"]
|
|
1547
|
+
for degree_column, suffix in [
|
|
1548
|
+
("_degree", ""),
|
|
1549
|
+
("_degree_no_type_fill", "_noimpute"),
|
|
1550
|
+
]
|
|
1551
|
+
],
|
|
1517
1552
|
*[
|
|
1518
1553
|
expression
|
|
1519
1554
|
for definition in STEM_DEFINITIONS
|
|
@@ -1898,8 +1933,14 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
1898
1933
|
.alias("adj_cmp_rate_coarse_2yr"),
|
|
1899
1934
|
)
|
|
1900
1935
|
|
|
1901
|
-
#
|
|
1902
|
-
|
|
1936
|
+
# Only tiers 1-8 enter these four-year bucket averages. Other-tier branch
|
|
1937
|
+
# mappings can share a UNITID and must not duplicate the institution's cohort.
|
|
1938
|
+
ipeds_tiers = (
|
|
1939
|
+
college_ref.select("unitid", "tier")
|
|
1940
|
+
.drop_nulls()
|
|
1941
|
+
.filter(pl.col("tier").is_in(range(1, 9)))
|
|
1942
|
+
.unique()
|
|
1943
|
+
)
|
|
1903
1944
|
ipeds_tier_completion = ipeds_completion_with_sector.join(
|
|
1904
1945
|
ipeds_tiers, on="unitid", how="left", validate="1:1"
|
|
1905
1946
|
).filter(pl.col("college_sector") == "4yr")
|
|
@@ -1926,7 +1967,6 @@ for label, tiers in tier_buckets.items():
|
|
|
1926
1967
|
.otherwise(pl.col("adj_cmp_rate_4yr"))
|
|
1927
1968
|
.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
1928
1969
|
)
|
|
1929
|
-
print(f"Four-year IPEDS-cohort-weighted completion rate, {label}: {bucket_rate}")
|
|
1930
1970
|
|
|
1931
1971
|
# Match the available 1098-T calendar years, including the missing 2015 year.
|
|
1932
1972
|
tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
|
|
@@ -2008,6 +2048,16 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
2008
2048
|
.alias("recovered_nsc_outcome")
|
|
2009
2049
|
)
|
|
2010
2050
|
|
|
2051
|
+
# Export every selectivity window, plus the BA sensitivity outcomes. Keep the
|
|
2052
|
+
# existing recovered-outcome definition above independent of this export list.
|
|
2053
|
+
selectivity_outcome_columns = [
|
|
2054
|
+
column for column in nsc_outcomes.columns
|
|
2055
|
+
if column.startswith((
|
|
2056
|
+
"att_selective_", "att_elite_", "cmp_selective_", "cmp_elite_",
|
|
2057
|
+
"cmp_BA_selective_", "cmp_BA_elite_", "cmp_BA_noimpute_",
|
|
2058
|
+
))
|
|
2059
|
+
]
|
|
2060
|
+
|
|
2011
2061
|
keep_columns = [
|
|
2012
2062
|
"sid_cepr",
|
|
2013
2063
|
"k_mean",
|
|
@@ -2026,140 +2076,36 @@ keep_columns = [
|
|
|
2026
2076
|
"adj_cmp_rate_coarsen_2yr",
|
|
2027
2077
|
"ID_FSC_firstinst",
|
|
2028
2078
|
"college_name_firstinst",
|
|
2079
|
+
"unitid_firstinst",
|
|
2080
|
+
"college_years_firstinst",
|
|
2029
2081
|
"tier_firstinst",
|
|
2030
2082
|
"completion_rate_150pct_firstinst",
|
|
2031
|
-
] + outcome_columns + stem_outcome_columns
|
|
2083
|
+
] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
|
|
2084
|
+
keep_columns = list(dict.fromkeys(keep_columns))
|
|
2032
2085
|
|
|
2033
2086
|
nsc_outcomes_final = nsc_outcomes.select(keep_columns)
|
|
2034
2087
|
|
|
2088
|
+
# Retain the existing completeness check without generating audit tables.
|
|
2089
|
+
if nsc_outcomes.filter(
|
|
2090
|
+
pl.col("att_any_byY4").is_not_null() & pl.col("k_mean_coarse").is_null()
|
|
2091
|
+
).height > 0:
|
|
2092
|
+
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2093
|
+
|
|
2035
2094
|
OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
|
|
2036
2095
|
nsc_outcomes.write_parquet(OUTCOMES)
|
|
2037
2096
|
nsc_outcomes.with_columns(
|
|
2038
2097
|
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2039
2098
|
).write_csv(OUTCOMES.with_suffix(".csv"))
|
|
2040
2099
|
|
|
2100
|
+
# Save the selected analysis/audit table as well as the existing full output.
|
|
2101
|
+
final_path = OUTCOMES.with_name(f"{OUTCOMES.stem}_final.parquet")
|
|
2102
|
+
nsc_outcomes_final.write_parquet(final_path)
|
|
2103
|
+
nsc_outcomes_final.with_columns(
|
|
2104
|
+
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2105
|
+
).write_csv(final_path.with_suffix(".csv"))
|
|
2106
|
+
|
|
2107
|
+
print(f"Wrote {final_path}")
|
|
2108
|
+
print(f"Wrote {final_path.with_suffix('.csv')}")
|
|
2041
2109
|
print(f"Wrote {OUTCOMES}")
|
|
2042
2110
|
print(f"Wrote {OUTCOMES.with_suffix('.csv')}")
|
|
2043
2111
|
print(f"Rows: {nsc_outcomes.height}, columns: {len(nsc_outcomes.columns)}")
|
|
2044
|
-
|
|
2045
|
-
first_college_coverage = (
|
|
2046
|
-
nsc_outcomes.filter(pl.col("ID_FSC_firstinst").is_not_null())
|
|
2047
|
-
.group_by("ID_FSC_firstinst")
|
|
2048
|
-
.agg(
|
|
2049
|
-
pl.col("k_mean_firstinst").is_not_null().any().alias("has_k_mean"),
|
|
2050
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2051
|
-
.is_not_null()
|
|
2052
|
-
.any()
|
|
2053
|
-
.alias("has_completion_rate"),
|
|
2054
|
-
)
|
|
2055
|
-
)
|
|
2056
|
-
print(
|
|
2057
|
-
"First-college coverage: "
|
|
2058
|
-
f"{first_college_coverage['has_k_mean'].sum()}/"
|
|
2059
|
-
f"{first_college_coverage.height} with college-specific k_mean; "
|
|
2060
|
-
f"{first_college_coverage['has_completion_rate'].sum()}/"
|
|
2061
|
-
f"{first_college_coverage.height} with IPEDS completion rate"
|
|
2062
|
-
)
|
|
2063
|
-
print(
|
|
2064
|
-
"National student-weighted coarse values: "
|
|
2065
|
-
f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
|
|
2066
|
-
f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
|
|
2067
|
-
f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
|
|
2068
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}; "
|
|
2069
|
-
"observed by-Y4 nonattender completion=0"
|
|
2070
|
-
)
|
|
2071
|
-
print(
|
|
2072
|
-
"Sample-institution IPEDS-cohort-weighted coarse values: "
|
|
2073
|
-
f"completion 4yr={cmp_rate_4yr_coarse_sample:.4f} "
|
|
2074
|
-
f"({sample_4yr_institutions} institutions), "
|
|
2075
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse_sample:.4f} "
|
|
2076
|
-
f"({sample_2yr_institutions} institutions)"
|
|
2077
|
-
)
|
|
2078
|
-
|
|
2079
|
-
observable_y4 = nsc_outcomes.filter(pl.col("att_any_byY4").is_not_null())
|
|
2080
|
-
coarse_audit = observable_y4.select(
|
|
2081
|
-
pl.len().alias("students"),
|
|
2082
|
-
pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
|
|
2083
|
-
pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
|
|
2084
|
-
pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
|
|
2085
|
-
pl.col("adj_cmp_rate_coarse")
|
|
2086
|
-
.null_count()
|
|
2087
|
-
.alias("adj_cmp_rate_coarse_missing"),
|
|
2088
|
-
pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
|
|
2089
|
-
pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
|
|
2090
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2091
|
-
.null_count()
|
|
2092
|
-
.alias("adj_cmp_rate_coarse_sample_missing"),
|
|
2093
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2094
|
-
.min()
|
|
2095
|
-
.alias("adj_cmp_rate_coarse_sample_min"),
|
|
2096
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2097
|
-
.max()
|
|
2098
|
-
.alias("adj_cmp_rate_coarse_sample_max"),
|
|
2099
|
-
)
|
|
2100
|
-
|
|
2101
|
-
if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
|
|
2102
|
-
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2103
|
-
# Attendees without a classified first college can have missing coarse rates.
|
|
2104
|
-
# Report those counts below rather than treating them as a build failure.
|
|
2105
|
-
print("Coarse outcome audit:")
|
|
2106
|
-
print(coarse_audit)
|
|
2107
|
-
|
|
2108
|
-
attendee_completion_audit = (
|
|
2109
|
-
observable_y4.filter(pl.col("att_any_byY4") == 1)
|
|
2110
|
-
.select(
|
|
2111
|
-
pl.len().alias("by_y4_attendees"),
|
|
2112
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2113
|
-
.is_null()
|
|
2114
|
-
.sum()
|
|
2115
|
-
.alias("missing_first_institution_rate"),
|
|
2116
|
-
(
|
|
2117
|
-
pl.col("completion_rate_150pct_firstinst").is_null()
|
|
2118
|
-
& pl.col("completion_rate_150pct_ip").is_not_null()
|
|
2119
|
-
)
|
|
2120
|
-
.sum()
|
|
2121
|
-
.alias("filled_by_tier_median"),
|
|
2122
|
-
pl.col("adj_cmp_rate")
|
|
2123
|
-
.is_null()
|
|
2124
|
-
.sum()
|
|
2125
|
-
.alias("missing_after_tier_median"),
|
|
2126
|
-
pl.col("adj_cmp_rate")
|
|
2127
|
-
.is_null()
|
|
2128
|
-
.mean()
|
|
2129
|
-
.alias("missing_after_tier_median_share"),
|
|
2130
|
-
(
|
|
2131
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2132
|
-
& pl.col("tier_firstinst").is_null()
|
|
2133
|
-
)
|
|
2134
|
-
.sum()
|
|
2135
|
-
.alias("missing_after_tier_median_no_tier"),
|
|
2136
|
-
(
|
|
2137
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2138
|
-
& pl.col("tier_firstinst").is_not_null()
|
|
2139
|
-
)
|
|
2140
|
-
.sum()
|
|
2141
|
-
.alias("missing_after_tier_median_with_tier"),
|
|
2142
|
-
)
|
|
2143
|
-
)
|
|
2144
|
-
print("By-Y4 attendee completion-rate audit:")
|
|
2145
|
-
print(attendee_completion_audit)
|
|
2146
|
-
print(
|
|
2147
|
-
nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
|
|
2148
|
-
"sid_cepr"
|
|
2149
|
-
)
|
|
2150
|
-
)
|
|
2151
|
-
|
|
2152
|
-
# Nonmissing year-eight completion identifies students with the full Y8 window.
|
|
2153
|
-
year8_attendance_audit = (
|
|
2154
|
-
nsc_outcomes.filter(pl.col("cmp_BA_byY8").is_not_null())
|
|
2155
|
-
.select(
|
|
2156
|
-
pl.len().alias("N_year8_sample"),
|
|
2157
|
-
(pl.col("att_any_byY4") == 0).sum().alias("N_no_attendance_byY4"),
|
|
2158
|
-
(
|
|
2159
|
-
(pl.col("att_any_byY4") == 0)
|
|
2160
|
-
& (pl.col("cmp_BA_byY8") == 1)
|
|
2161
|
-
).sum().alias("N_no_attendance_byY4_BA_byY8"),
|
|
2162
|
-
)
|
|
2163
|
-
)
|
|
2164
|
-
print("Year-eight sample attendance and late BA completion check:")
|
|
2165
|
-
print(year8_attendance_audit)
|