ltc-code 0.2.29__tar.gz → 0.2.31__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {ltc_code-0.2.29 → ltc_code-0.2.31}/PKG-INFO +1 -1
- {ltc_code-0.2.29 → ltc_code-0.2.31}/pyproject.toml +1 -1
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/NEW_CROSSWALK.md +3 -1
- ltc_code-0.2.31/src/ltc_code/nsc/OUTCOMES.md +59 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/build_nsc_outcomes.py +66 -126
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/build_nsc_outcomes_new.py +94 -149
- {ltc_code-0.2.29 → ltc_code-0.2.31}/README.md +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/aspire.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/helpers.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/ilt.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/mappings.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/aspire.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/christel_house.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/democracy_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/green_dot.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/ilt.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/kipp.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/kipp_nj.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/rocketship.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/yes_prep.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/BALANCE.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/FS.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/ITT.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/TOT.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/apps_helpers.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/harmony.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260706_ceprscripts/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260712_uncommon_scripts/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260712_uncommon_scripts/mappings.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260712_uncommon_scripts/uncommon.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/__init__.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/aspire.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/check_cmo_apps.do +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/christel_house.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/green_dot.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/helpers.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june13.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june2.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june30.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june5.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/june7.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/kipp_nj.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/kipp_tx.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/main.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/make_summary_stats_table.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/mappings.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/may27.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/__init__.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/dhs_stem/extract_dhs_stem_cips.R +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/dhs_stem/stemList2024.pdf +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/naics.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/naics.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/naics_raw.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/CREDENTIAL_LEVEL_LOOKUP_TABLE.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/IPEDS_IC_2013.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/IPEDS_IC_manual.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/NSC_SCHOOL_CODE_TO_IPEDS_UNIT_ID_XWALK_APR-2023.xlsx +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/chetty/mrc_table11.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/chetty/mrc_table2.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/college_crosswalk.xls +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/directory.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/raw/ipeds_data.dta +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/run_nsc_outcomes.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/plot_bars.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/polars_dates.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/rocketship.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/schema_mapping.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/school_name_xwalk/__init__.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/school_name_xwalk/all_schools_with_ccd.csv +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/school_name_xwalk/merge_school_ccd.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/signal_var_calcs.py +0 -0
- {ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/yes_prep.py +0 -0
|
@@ -21,4 +21,6 @@ Required files in `nsc/raw/`:
|
|
|
21
21
|
|
|
22
22
|
The old IPEDS_IC_manual.xlsx is not needed by this variant. Student inputs remain input/apps.csv, input/nsc_records_old.dta, and input/nsc_records_new.csv. Do not supply synthetic records for actual analysis. Institutional support files are bundled in the locally built wheel and source distribution but are Git-ignored. GitHub Actions publishes these verified distributions from release assets; it does not rebuild from a checkout that lacks the data. Student inputs are never bundled.
|
|
23
23
|
|
|
24
|
-
|
|
24
|
+
Intermediate diagnostic CSVs and printed audit tables are not generated. Only the full and selected final outcome files are written, each in Parquet and CSV format. The 99.48% matched subset is an audit denominator, not a filter that drops student records from the build.
|
|
25
|
+
|
|
26
|
+
Version 0.2.30 adds final audit exports and selectivity/BA sensitivity outcomes to both builds; see [OUTCOMES.md](OUTCOMES.md) for names and definitions.
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
# NSC outcomes added in 0.2.30
|
|
2
|
+
|
|
3
|
+
Both `build_nsc_outcomes.py` and `build_nsc_outcomes_new.py` provide these fields.
|
|
4
|
+
The new-crosswalk variant remains an explicit separate build.
|
|
5
|
+
|
|
6
|
+
## Final institution audit fields
|
|
7
|
+
|
|
8
|
+
`nsc_outcomes_final` retains `ID_FSC_firstinst`, `college_name_firstinst`,
|
|
9
|
+
`unitid_firstinst`, `tier_firstinst`, `college_years_firstinst`, and
|
|
10
|
+
`completion_rate_150pct_firstinst`. Institution years are 4 (four-year),
|
|
11
|
+
2 (two-year), 1 (less than two-year), or null if unknown.
|
|
12
|
+
|
|
13
|
+
The first institution remains the earliest retained enrollment spell starting
|
|
14
|
+
on or after July 1 of the year the student turns 18, with the existing tie rule.
|
|
15
|
+
It need not meet the attendance-status rule or fall within the by-year-four
|
|
16
|
+
window. Attendance indicators describe qualifying enrollment in their window;
|
|
17
|
+
they do not describe the first institution. A transfer can therefore have a
|
|
18
|
+
two-year first institution and four-year attendance. Prediction formulas and
|
|
19
|
+
first-institution selection have not changed.
|
|
20
|
+
|
|
21
|
+
The build writes the selected table to `nsc_outcomes_final.parquet` and `.csv`
|
|
22
|
+
alongside the existing full `nsc_outcomes.parquet` and `.csv` outputs. These
|
|
23
|
+
are the only four files written; intermediate diagnostic exports and printed
|
|
24
|
+
audit tables have been removed.
|
|
25
|
+
|
|
26
|
+
## Selectivity outcomes
|
|
27
|
+
|
|
28
|
+
Selective means tiers 3–4. The existing `elite` name means highly selective,
|
|
29
|
+
tiers 1–2. These institution groups do not overlap. A student can attend or
|
|
30
|
+
earn degrees at institutions in both groups, so their student indicators
|
|
31
|
+
are not forced to be mutually exclusive.
|
|
32
|
+
|
|
33
|
+
| Example | Meaning |
|
|
34
|
+
| --- | --- |
|
|
35
|
+
| `att_selective_byY4` | Qualifying attendance at a tier 3–4 institution by year 4 |
|
|
36
|
+
| `att_elite_byY4` | Qualifying attendance at a tier 1–2 institution by year 4 |
|
|
37
|
+
| `cmp_selective_byY8` | AA or BA from a tier 3–4 institution by year 8 |
|
|
38
|
+
| `cmp_elite_byY8` | Existing AA-or-BA measure for tiers 1–2 |
|
|
39
|
+
| `cmp_BA_selective_byY8` | BA from a tier 3–4 institution by year 8 |
|
|
40
|
+
| `cmp_BA_elite_byY8` | BA from a tier 1–2 institution by year 8 |
|
|
41
|
+
| `cmp_BA_noimpute_byY8` | Any BA without institution-type credential imputation |
|
|
42
|
+
| `cmp_BA_selective_noimpute_byY8` | Same sensitivity definition for tiers 3–4 |
|
|
43
|
+
| `cmp_BA_elite_noimpute_byY8` | Same sensitivity definition for tiers 1–2 |
|
|
44
|
+
|
|
45
|
+
Attendance is available for `inY1`–`inY8`, `byY1`–`byY8`, their `fall` and
|
|
46
|
+
`spring` variants, and calendar ages 18–26 (e.g. `att_selective_20`). Completion
|
|
47
|
+
is available for `byY1`–`byY8`. All these selectivity and sensitivity fields are
|
|
48
|
+
retained in the selected final table. They follow the existing observation
|
|
49
|
+
windows and zero/null rules. Unknown institutional tiers do not count as
|
|
50
|
+
selective. Completion uses the degree-granting institution's tier, not the
|
|
51
|
+
first institution's tier.
|
|
52
|
+
|
|
53
|
+
The default BA measures retain the existing institution-type imputation.
|
|
54
|
+
The `noimpute` versions use credential-lookup/title evidence before that step:
|
|
55
|
+
an unknown award at a four-year institution alone does not count as a BA.
|
|
56
|
+
Students remain in the sample, and another identified BA can still qualify.
|
|
57
|
+
This does not change institutional IPEDS rates or coarse predictions.
|
|
58
|
+
|
|
59
|
+
Verification: run `uv run python tests/verify_nsc_outcomes.py` from the repository root with the local institutional support files present. This creates temporary synthetic records and does not require secured student inputs.
|
|
@@ -93,8 +93,8 @@ RANDOM_SEED = 3852804
|
|
|
93
93
|
SARAH_STEM_CIP_FAMILIES = {"11", "14", "15", "26", "27", "40", "41"}
|
|
94
94
|
DHS_STEM_CIP_FAMILIES = {"14", "26", "27", "40"}
|
|
95
95
|
STEM_DEFINITIONS = ["sarah", "dhs"]
|
|
96
|
-
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
|
|
97
|
-
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
|
|
96
|
+
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
97
|
+
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
98
98
|
AGE_ATTENDANCE_RANGE = range(18, 27)
|
|
99
99
|
|
|
100
100
|
# These ordered rules exactly reproduce Sarah's Stata replacements. Every rule
|
|
@@ -573,7 +573,9 @@ college_ref = (
|
|
|
573
573
|
.with_columns(
|
|
574
574
|
(pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
|
|
575
575
|
pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
|
|
576
|
+
# Institution groups: highly selective (elite) is tiers 1–2; selective is 3–4.
|
|
576
577
|
pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
|
|
578
|
+
pl.col("tier").is_in([3, 4]).cast(pl.Int8).alias("college_selective"),
|
|
577
579
|
)
|
|
578
580
|
.with_columns(
|
|
579
581
|
(pl.col("college_4yr") + pl.col("college_2yr") > 0)
|
|
@@ -592,6 +594,7 @@ college_ref = (
|
|
|
592
594
|
"college_2yr",
|
|
593
595
|
"tier",
|
|
594
596
|
"college_elite",
|
|
597
|
+
"college_selective",
|
|
595
598
|
"k_mean",
|
|
596
599
|
"completion_rate_150pct_ip",
|
|
597
600
|
)
|
|
@@ -1061,6 +1064,7 @@ enroll_weeks = (
|
|
|
1061
1064
|
"college_2yr",
|
|
1062
1065
|
"tier",
|
|
1063
1066
|
"college_elite",
|
|
1067
|
+
"college_selective",
|
|
1064
1068
|
"k_mean",
|
|
1065
1069
|
"completion_rate_150pct_ip",
|
|
1066
1070
|
"term_start_date",
|
|
@@ -1131,6 +1135,7 @@ enroll = (
|
|
|
1131
1135
|
pl.col("college_2yr").max(),
|
|
1132
1136
|
pl.col("tier").drop_nulls().min(),
|
|
1133
1137
|
pl.col("college_elite").max(),
|
|
1138
|
+
pl.col("college_selective").max(),
|
|
1134
1139
|
pl.col("k_mean").drop_nulls().first(),
|
|
1135
1140
|
pl.col("completion_rate_150pct_ip").drop_nulls().first(),
|
|
1136
1141
|
pl.col("_term_att").max(),
|
|
@@ -1453,6 +1458,8 @@ degrees = (
|
|
|
1453
1458
|
.otherwise(pl.lit(None))
|
|
1454
1459
|
.alias("_degree")
|
|
1455
1460
|
)
|
|
1461
|
+
# Preserve credential/title evidence before filling unknown awards by sector.
|
|
1462
|
+
.with_columns(pl.col("_degree").alias("_degree_no_type_fill"))
|
|
1456
1463
|
.with_columns(
|
|
1457
1464
|
pl.when(pl.col("_degree").is_null() & (pl.col("college_years") == 4))
|
|
1458
1465
|
.then(pl.lit("BA"))
|
|
@@ -1504,6 +1511,12 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1504
1511
|
(pl.col("_by_end") * (pl.col("_degree") == "BA").cast(pl.Int8))
|
|
1505
1512
|
.max()
|
|
1506
1513
|
.alias(f"cmp_BA_byY{year}"),
|
|
1514
|
+
(
|
|
1515
|
+
pl.col("_by_end")
|
|
1516
|
+
* (pl.col("_degree_no_type_fill") == "BA").fill_null(False).cast(pl.Int8)
|
|
1517
|
+
)
|
|
1518
|
+
.max()
|
|
1519
|
+
.alias(f"cmp_BA_noimpute_byY{year}"),
|
|
1507
1520
|
(pl.col("_by_end") * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8))
|
|
1508
1521
|
.max()
|
|
1509
1522
|
.alias(f"cmp_any_byY{year}"),
|
|
@@ -1514,6 +1527,28 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1514
1527
|
)
|
|
1515
1528
|
.max()
|
|
1516
1529
|
.alias(f"cmp_elite_byY{year}"),
|
|
1530
|
+
(
|
|
1531
|
+
pl.col("_by_end")
|
|
1532
|
+
* pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8)
|
|
1533
|
+
* pl.col("college_selective").fill_null(0).cast(pl.Int8)
|
|
1534
|
+
)
|
|
1535
|
+
.max()
|
|
1536
|
+
.alias(f"cmp_selective_byY{year}"),
|
|
1537
|
+
# BA-only outcomes support comparisons with any-BA completion.
|
|
1538
|
+
*[
|
|
1539
|
+
(
|
|
1540
|
+
pl.col("_by_end")
|
|
1541
|
+
* (pl.col(degree_column) == "BA").fill_null(False).cast(pl.Int8)
|
|
1542
|
+
* pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
|
|
1543
|
+
)
|
|
1544
|
+
.max()
|
|
1545
|
+
.alias(f"cmp_BA_{college_type}{suffix}_byY{year}")
|
|
1546
|
+
for college_type in ["selective", "elite"]
|
|
1547
|
+
for degree_column, suffix in [
|
|
1548
|
+
("_degree", ""),
|
|
1549
|
+
("_degree_no_type_fill", "_noimpute"),
|
|
1550
|
+
]
|
|
1551
|
+
],
|
|
1517
1552
|
*[
|
|
1518
1553
|
expression
|
|
1519
1554
|
for definition in STEM_DEFINITIONS
|
|
@@ -1932,7 +1967,6 @@ for label, tiers in tier_buckets.items():
|
|
|
1932
1967
|
.otherwise(pl.col("adj_cmp_rate_4yr"))
|
|
1933
1968
|
.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
1934
1969
|
)
|
|
1935
|
-
print(f"Four-year IPEDS-cohort-weighted completion rate, {label}: {bucket_rate}")
|
|
1936
1970
|
|
|
1937
1971
|
# Match the available 1098-T calendar years, including the missing 2015 year.
|
|
1938
1972
|
tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
|
|
@@ -2014,6 +2048,16 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
2014
2048
|
.alias("recovered_nsc_outcome")
|
|
2015
2049
|
)
|
|
2016
2050
|
|
|
2051
|
+
# Export every selectivity window, plus the BA sensitivity outcomes. Keep the
|
|
2052
|
+
# existing recovered-outcome definition above independent of this export list.
|
|
2053
|
+
selectivity_outcome_columns = [
|
|
2054
|
+
column for column in nsc_outcomes.columns
|
|
2055
|
+
if column.startswith((
|
|
2056
|
+
"att_selective_", "att_elite_", "cmp_selective_", "cmp_elite_",
|
|
2057
|
+
"cmp_BA_selective_", "cmp_BA_elite_", "cmp_BA_noimpute_",
|
|
2058
|
+
))
|
|
2059
|
+
]
|
|
2060
|
+
|
|
2017
2061
|
keep_columns = [
|
|
2018
2062
|
"sid_cepr",
|
|
2019
2063
|
"k_mean",
|
|
@@ -2032,140 +2076,36 @@ keep_columns = [
|
|
|
2032
2076
|
"adj_cmp_rate_coarsen_2yr",
|
|
2033
2077
|
"ID_FSC_firstinst",
|
|
2034
2078
|
"college_name_firstinst",
|
|
2079
|
+
"unitid_firstinst",
|
|
2080
|
+
"college_years_firstinst",
|
|
2035
2081
|
"tier_firstinst",
|
|
2036
2082
|
"completion_rate_150pct_firstinst",
|
|
2037
|
-
] + outcome_columns + stem_outcome_columns
|
|
2083
|
+
] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
|
|
2084
|
+
keep_columns = list(dict.fromkeys(keep_columns))
|
|
2038
2085
|
|
|
2039
2086
|
nsc_outcomes_final = nsc_outcomes.select(keep_columns)
|
|
2040
2087
|
|
|
2088
|
+
# Retain the existing completeness check without generating audit tables.
|
|
2089
|
+
if nsc_outcomes.filter(
|
|
2090
|
+
pl.col("att_any_byY4").is_not_null() & pl.col("k_mean_coarse").is_null()
|
|
2091
|
+
).height > 0:
|
|
2092
|
+
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2093
|
+
|
|
2041
2094
|
OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
|
|
2042
2095
|
nsc_outcomes.write_parquet(OUTCOMES)
|
|
2043
2096
|
nsc_outcomes.with_columns(
|
|
2044
2097
|
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2045
2098
|
).write_csv(OUTCOMES.with_suffix(".csv"))
|
|
2046
2099
|
|
|
2100
|
+
# Save the selected analysis/audit table as well as the existing full output.
|
|
2101
|
+
final_path = OUTCOMES.with_name(f"{OUTCOMES.stem}_final.parquet")
|
|
2102
|
+
nsc_outcomes_final.write_parquet(final_path)
|
|
2103
|
+
nsc_outcomes_final.with_columns(
|
|
2104
|
+
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2105
|
+
).write_csv(final_path.with_suffix(".csv"))
|
|
2106
|
+
|
|
2107
|
+
print(f"Wrote {final_path}")
|
|
2108
|
+
print(f"Wrote {final_path.with_suffix('.csv')}")
|
|
2047
2109
|
print(f"Wrote {OUTCOMES}")
|
|
2048
2110
|
print(f"Wrote {OUTCOMES.with_suffix('.csv')}")
|
|
2049
2111
|
print(f"Rows: {nsc_outcomes.height}, columns: {len(nsc_outcomes.columns)}")
|
|
2050
|
-
|
|
2051
|
-
first_college_coverage = (
|
|
2052
|
-
nsc_outcomes.filter(pl.col("ID_FSC_firstinst").is_not_null())
|
|
2053
|
-
.group_by("ID_FSC_firstinst")
|
|
2054
|
-
.agg(
|
|
2055
|
-
pl.col("k_mean_firstinst").is_not_null().any().alias("has_k_mean"),
|
|
2056
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2057
|
-
.is_not_null()
|
|
2058
|
-
.any()
|
|
2059
|
-
.alias("has_completion_rate"),
|
|
2060
|
-
)
|
|
2061
|
-
)
|
|
2062
|
-
print(
|
|
2063
|
-
"First-college coverage: "
|
|
2064
|
-
f"{first_college_coverage['has_k_mean'].sum()}/"
|
|
2065
|
-
f"{first_college_coverage.height} with college-specific k_mean; "
|
|
2066
|
-
f"{first_college_coverage['has_completion_rate'].sum()}/"
|
|
2067
|
-
f"{first_college_coverage.height} with IPEDS completion rate"
|
|
2068
|
-
)
|
|
2069
|
-
print(
|
|
2070
|
-
"National student-weighted coarse values: "
|
|
2071
|
-
f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
|
|
2072
|
-
f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
|
|
2073
|
-
f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
|
|
2074
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}; "
|
|
2075
|
-
"observed by-Y4 nonattender completion=0"
|
|
2076
|
-
)
|
|
2077
|
-
print(
|
|
2078
|
-
"Sample-institution IPEDS-cohort-weighted coarse values: "
|
|
2079
|
-
f"completion 4yr={cmp_rate_4yr_coarse_sample:.4f} "
|
|
2080
|
-
f"({sample_4yr_institutions} institutions), "
|
|
2081
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse_sample:.4f} "
|
|
2082
|
-
f"({sample_2yr_institutions} institutions)"
|
|
2083
|
-
)
|
|
2084
|
-
|
|
2085
|
-
observable_y4 = nsc_outcomes.filter(pl.col("att_any_byY4").is_not_null())
|
|
2086
|
-
coarse_audit = observable_y4.select(
|
|
2087
|
-
pl.len().alias("students"),
|
|
2088
|
-
pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
|
|
2089
|
-
pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
|
|
2090
|
-
pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
|
|
2091
|
-
pl.col("adj_cmp_rate_coarse")
|
|
2092
|
-
.null_count()
|
|
2093
|
-
.alias("adj_cmp_rate_coarse_missing"),
|
|
2094
|
-
pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
|
|
2095
|
-
pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
|
|
2096
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2097
|
-
.null_count()
|
|
2098
|
-
.alias("adj_cmp_rate_coarse_sample_missing"),
|
|
2099
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2100
|
-
.min()
|
|
2101
|
-
.alias("adj_cmp_rate_coarse_sample_min"),
|
|
2102
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2103
|
-
.max()
|
|
2104
|
-
.alias("adj_cmp_rate_coarse_sample_max"),
|
|
2105
|
-
)
|
|
2106
|
-
|
|
2107
|
-
if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
|
|
2108
|
-
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2109
|
-
# Attendees without a classified first college can have missing coarse rates.
|
|
2110
|
-
# Report those counts below rather than treating them as a build failure.
|
|
2111
|
-
print("Coarse outcome audit:")
|
|
2112
|
-
print(coarse_audit)
|
|
2113
|
-
|
|
2114
|
-
attendee_completion_audit = (
|
|
2115
|
-
observable_y4.filter(pl.col("att_any_byY4") == 1)
|
|
2116
|
-
.select(
|
|
2117
|
-
pl.len().alias("by_y4_attendees"),
|
|
2118
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2119
|
-
.is_null()
|
|
2120
|
-
.sum()
|
|
2121
|
-
.alias("missing_first_institution_rate"),
|
|
2122
|
-
(
|
|
2123
|
-
pl.col("completion_rate_150pct_firstinst").is_null()
|
|
2124
|
-
& pl.col("completion_rate_150pct_ip").is_not_null()
|
|
2125
|
-
)
|
|
2126
|
-
.sum()
|
|
2127
|
-
.alias("filled_by_tier_median"),
|
|
2128
|
-
pl.col("adj_cmp_rate")
|
|
2129
|
-
.is_null()
|
|
2130
|
-
.sum()
|
|
2131
|
-
.alias("missing_after_tier_median"),
|
|
2132
|
-
pl.col("adj_cmp_rate")
|
|
2133
|
-
.is_null()
|
|
2134
|
-
.mean()
|
|
2135
|
-
.alias("missing_after_tier_median_share"),
|
|
2136
|
-
(
|
|
2137
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2138
|
-
& pl.col("tier_firstinst").is_null()
|
|
2139
|
-
)
|
|
2140
|
-
.sum()
|
|
2141
|
-
.alias("missing_after_tier_median_no_tier"),
|
|
2142
|
-
(
|
|
2143
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2144
|
-
& pl.col("tier_firstinst").is_not_null()
|
|
2145
|
-
)
|
|
2146
|
-
.sum()
|
|
2147
|
-
.alias("missing_after_tier_median_with_tier"),
|
|
2148
|
-
)
|
|
2149
|
-
)
|
|
2150
|
-
print("By-Y4 attendee completion-rate audit:")
|
|
2151
|
-
print(attendee_completion_audit)
|
|
2152
|
-
print(
|
|
2153
|
-
nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
|
|
2154
|
-
"sid_cepr"
|
|
2155
|
-
)
|
|
2156
|
-
)
|
|
2157
|
-
|
|
2158
|
-
# Nonmissing year-eight completion identifies students with the full Y8 window.
|
|
2159
|
-
year8_attendance_audit = (
|
|
2160
|
-
nsc_outcomes.filter(pl.col("cmp_BA_byY8").is_not_null())
|
|
2161
|
-
.select(
|
|
2162
|
-
pl.len().alias("N_year8_sample"),
|
|
2163
|
-
(pl.col("att_any_byY4") == 0).sum().alias("N_no_attendance_byY4"),
|
|
2164
|
-
(
|
|
2165
|
-
(pl.col("att_any_byY4") == 0)
|
|
2166
|
-
& (pl.col("cmp_BA_byY8") == 1)
|
|
2167
|
-
).sum().alias("N_no_attendance_byY4_BA_byY8"),
|
|
2168
|
-
)
|
|
2169
|
-
)
|
|
2170
|
-
print("Year-eight sample attendance and late BA completion check:")
|
|
2171
|
-
print(year8_attendance_audit)
|
|
@@ -60,7 +60,6 @@ INPUT_NSC = PACKAGE_NSC / "input"
|
|
|
60
60
|
# Folder for secured Census inputs; no student records are packaged.
|
|
61
61
|
|
|
62
62
|
OUTPUT_NSC = PACKAGE_NSC / "output" / "new_crosswalk"
|
|
63
|
-
OUTPUT_NSC.mkdir(parents=True, exist_ok=True)
|
|
64
63
|
# Folder for final student-level NSC outcomes.
|
|
65
64
|
|
|
66
65
|
APPS = INPUT_NSC / "apps.csv"
|
|
@@ -93,8 +92,8 @@ RANDOM_SEED = 3852804
|
|
|
93
92
|
SARAH_STEM_CIP_FAMILIES = {"11", "14", "15", "26", "27", "40", "41"}
|
|
94
93
|
DHS_STEM_CIP_FAMILIES = {"14", "26", "27", "40"}
|
|
95
94
|
STEM_DEFINITIONS = ["sarah", "dhs"]
|
|
96
|
-
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite"]
|
|
97
|
-
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr"]
|
|
95
|
+
COLLEGE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
96
|
+
AGE_ATTENDANCE_TYPES = ["any", "4yr", "2yr", "elite", "selective"]
|
|
98
97
|
AGE_ATTENDANCE_RANGE = range(18, 27)
|
|
99
98
|
|
|
100
99
|
# These ordered rules exactly reproduce Sarah's Stata replacements. Every rule
|
|
@@ -249,14 +248,12 @@ legacy_absent_candidates = legacy_raw.join(
|
|
|
249
248
|
official_crosswalk.select("ID_FSC"), on="ID_FSC", how="anti"
|
|
250
249
|
)
|
|
251
250
|
# Reproducible legacy fallback selection; approved sole-rate exceptions below
|
|
252
|
-
# still take precedence.
|
|
253
|
-
legacy_absent_candidates.write_csv(OUTPUT_NSC / "legacy_absent_candidates.csv")
|
|
251
|
+
# still take precedence.
|
|
254
252
|
legacy_append = legacy_absent_candidates.sort(
|
|
255
253
|
["ID_FSC", "legacy_unitid", "ID_OPE", "name"], nulls_last=True
|
|
256
254
|
).unique("ID_FSC", keep="first", maintain_order=True).select(
|
|
257
255
|
"ID_FSC", "legacy_unitid", pl.col("name").alias("college_name_crosswalk")
|
|
258
256
|
)
|
|
259
|
-
legacy_append.write_csv(OUTPUT_NSC / "legacy_appended_codes.csv")
|
|
260
257
|
combined_crosswalk = pl.concat([
|
|
261
258
|
official_crosswalk.with_columns(pl.lit("official").alias("mapping_source")),
|
|
262
259
|
legacy_append.with_columns(pl.lit("legacy_absent_code").alias("mapping_source")),
|
|
@@ -300,7 +297,6 @@ college_crosswalk = (
|
|
|
300
297
|
pl.col("ID_FSC").str.slice(0, 6).cast(pl.Int64, strict=False).alias("opeid"),
|
|
301
298
|
)
|
|
302
299
|
)
|
|
303
|
-
college_crosswalk.write_csv(OUTPUT_NSC / "mapping_decisions.csv")
|
|
304
300
|
|
|
305
301
|
|
|
306
302
|
###########################################################
|
|
@@ -616,7 +612,9 @@ college_ref = (
|
|
|
616
612
|
.with_columns(
|
|
617
613
|
(pl.col("college_years") == 4).cast(pl.Int8).alias("college_4yr"),
|
|
618
614
|
pl.col("college_years").is_in([1, 2]).cast(pl.Int8).alias("college_2yr"),
|
|
615
|
+
# Institution groups: highly selective (elite) is tiers 1–2; selective is 3–4.
|
|
619
616
|
pl.col("tier").is_in([1, 2]).cast(pl.Int8).alias("college_elite"),
|
|
617
|
+
pl.col("tier").is_in([3, 4]).cast(pl.Int8).alias("college_selective"),
|
|
620
618
|
)
|
|
621
619
|
.with_columns(
|
|
622
620
|
(pl.col("college_4yr") + pl.col("college_2yr") > 0)
|
|
@@ -635,26 +633,13 @@ college_ref = (
|
|
|
635
633
|
"college_2yr",
|
|
636
634
|
"tier",
|
|
637
635
|
"college_elite",
|
|
636
|
+
"college_selective",
|
|
638
637
|
"k_mean",
|
|
639
638
|
"completion_rate_150pct_ip",
|
|
640
639
|
)
|
|
641
640
|
)
|
|
642
641
|
|
|
643
642
|
|
|
644
|
-
# Institution-universe coverage: code counts, not student counts.
|
|
645
|
-
coverage = college_crosswalk.join(
|
|
646
|
-
college_ref.select("ID_FSC", "college_years", "completion_rate_150pct_ip"),
|
|
647
|
-
on="ID_FSC", how="left", validate="1:1",
|
|
648
|
-
)
|
|
649
|
-
coverage.write_csv(OUTPUT_NSC / "institution_coverage.csv")
|
|
650
|
-
coverage.filter(pl.col("unitid").is_null()).write_csv(OUTPUT_NSC / "codes_without_unitid.csv")
|
|
651
|
-
coverage.filter(pl.col("unitid").is_not_null() & pl.col("college_years").is_null()).write_csv(OUTPUT_NSC / "mapped_codes_without_level.csv")
|
|
652
|
-
print("CROSSWALK COVERAGE", coverage.select(
|
|
653
|
-
pl.len().alias("codes"), pl.col("unitid").is_not_null().sum().alias("mapped"),
|
|
654
|
-
pl.col("college_years").is_not_null().sum().alias("classified"),
|
|
655
|
-
pl.col("completion_rate_150pct_ip").is_not_null().sum().alias("with_rate"),
|
|
656
|
-
))
|
|
657
|
-
|
|
658
643
|
###########################################################
|
|
659
644
|
# Load student universe and NSC records
|
|
660
645
|
###########################################################
|
|
@@ -1042,6 +1027,7 @@ enroll_weeks = (
|
|
|
1042
1027
|
"college_2yr",
|
|
1043
1028
|
"tier",
|
|
1044
1029
|
"college_elite",
|
|
1030
|
+
"college_selective",
|
|
1045
1031
|
"k_mean",
|
|
1046
1032
|
"completion_rate_150pct_ip",
|
|
1047
1033
|
"term_start_date",
|
|
@@ -1112,6 +1098,7 @@ enroll = (
|
|
|
1112
1098
|
pl.col("college_2yr").max(),
|
|
1113
1099
|
pl.col("tier").drop_nulls().min(),
|
|
1114
1100
|
pl.col("college_elite").max(),
|
|
1101
|
+
pl.col("college_selective").max(),
|
|
1115
1102
|
pl.col("k_mean").drop_nulls().first(),
|
|
1116
1103
|
pl.col("completion_rate_150pct_ip").drop_nulls().first(),
|
|
1117
1104
|
pl.col("_term_att").max(),
|
|
@@ -1306,11 +1293,34 @@ for column in enrollment_outcomes.columns:
|
|
|
1306
1293
|
.alias(column)
|
|
1307
1294
|
)
|
|
1308
1295
|
|
|
1309
|
-
#
|
|
1310
|
-
#
|
|
1296
|
+
# Use exactly the status and window-overlap rules for att_any_byY4. A spell
|
|
1297
|
+
# starting before July 1 can count if it continues into the attendance window.
|
|
1298
|
+
qualifying_enroll = enroll.with_columns(
|
|
1299
|
+
pl.date(pl.col("cohort_18"), 7, 1).alias("_window_start"),
|
|
1300
|
+
pl.date(pl.col("cohort_18") + 4, 6, 30).alias("_window_end"),
|
|
1301
|
+
).filter(
|
|
1302
|
+
(pl.col("_term_att") == 1)
|
|
1303
|
+
& (pl.col("term_end_date") >= pl.col("_window_start"))
|
|
1304
|
+
)
|
|
1305
|
+
|
|
1306
|
+
# Before excluding late starts, count students whose first observed qualifying
|
|
1307
|
+
# enrollment is after Y4. Denominator: observed qualifying enrollees with a
|
|
1308
|
+
# fully observable Y4 window; not all applicants or an eventual-enrollment rate.
|
|
1309
|
+
first_enrollment_timing = qualifying_enroll.group_by("sid_cepr").agg(
|
|
1310
|
+
pl.col("term_start_date").min().alias("first_start"),
|
|
1311
|
+
pl.col("_window_end").first().alias("y4_end"),
|
|
1312
|
+
).filter(pl.col("y4_end") <= NSC_CUTOFF_DATE)
|
|
1313
|
+
|
|
1314
|
+
print("FIRST QUALIFYING ENROLLMENT AFTER Y4", first_enrollment_timing.select(
|
|
1315
|
+
pl.len().alias("observed_enrollees_with_Y4_followup"),
|
|
1316
|
+
(pl.col("first_start") > pl.col("y4_end")).sum().alias("first_after_Y4"),
|
|
1317
|
+
(100 * (pl.col("first_start") > pl.col("y4_end")).mean()).alias("percent_after_Y4"),
|
|
1318
|
+
))
|
|
1319
|
+
|
|
1320
|
+
# Select the earliest qualifying spell overlapping the by-Y4 window. Preserve
|
|
1321
|
+
# Sarah's tie-break: highest ID_FSC when enrollment start dates are equal.
|
|
1311
1322
|
first_institution = (
|
|
1312
|
-
|
|
1313
|
-
.filter(pl.col("term_start_date") >= pl.col("_hs_grad_start"))
|
|
1323
|
+
qualifying_enroll.filter(pl.col("term_start_date") <= pl.col("_window_end"))
|
|
1314
1324
|
.sort(["sid_cepr", "term_start_date", "ID_FSC"], descending=[False, False, True])
|
|
1315
1325
|
.group_by("sid_cepr", maintain_order=True)
|
|
1316
1326
|
.agg(
|
|
@@ -1434,6 +1444,8 @@ degrees = (
|
|
|
1434
1444
|
.otherwise(pl.lit(None))
|
|
1435
1445
|
.alias("_degree")
|
|
1436
1446
|
)
|
|
1447
|
+
# Preserve credential/title evidence before filling unknown awards by sector.
|
|
1448
|
+
.with_columns(pl.col("_degree").alias("_degree_no_type_fill"))
|
|
1437
1449
|
.with_columns(
|
|
1438
1450
|
pl.when(pl.col("_degree").is_null() & (pl.col("college_years") == 4))
|
|
1439
1451
|
.then(pl.lit("BA"))
|
|
@@ -1485,6 +1497,12 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1485
1497
|
(pl.col("_by_end") * (pl.col("_degree") == "BA").cast(pl.Int8))
|
|
1486
1498
|
.max()
|
|
1487
1499
|
.alias(f"cmp_BA_byY{year}"),
|
|
1500
|
+
(
|
|
1501
|
+
pl.col("_by_end")
|
|
1502
|
+
* (pl.col("_degree_no_type_fill") == "BA").fill_null(False).cast(pl.Int8)
|
|
1503
|
+
)
|
|
1504
|
+
.max()
|
|
1505
|
+
.alias(f"cmp_BA_noimpute_byY{year}"),
|
|
1488
1506
|
(pl.col("_by_end") * pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8))
|
|
1489
1507
|
.max()
|
|
1490
1508
|
.alias(f"cmp_any_byY{year}"),
|
|
@@ -1495,6 +1513,28 @@ for year in range(1, N_YEARS_OUT + 1):
|
|
|
1495
1513
|
)
|
|
1496
1514
|
.max()
|
|
1497
1515
|
.alias(f"cmp_elite_byY{year}"),
|
|
1516
|
+
(
|
|
1517
|
+
pl.col("_by_end")
|
|
1518
|
+
* pl.col("_degree").is_in(["AA", "BA"]).cast(pl.Int8)
|
|
1519
|
+
* pl.col("college_selective").fill_null(0).cast(pl.Int8)
|
|
1520
|
+
)
|
|
1521
|
+
.max()
|
|
1522
|
+
.alias(f"cmp_selective_byY{year}"),
|
|
1523
|
+
# BA-only outcomes support comparisons with any-BA completion.
|
|
1524
|
+
*[
|
|
1525
|
+
(
|
|
1526
|
+
pl.col("_by_end")
|
|
1527
|
+
* (pl.col(degree_column) == "BA").fill_null(False).cast(pl.Int8)
|
|
1528
|
+
* pl.col(f"college_{college_type}").fill_null(0).cast(pl.Int8)
|
|
1529
|
+
)
|
|
1530
|
+
.max()
|
|
1531
|
+
.alias(f"cmp_BA_{college_type}{suffix}_byY{year}")
|
|
1532
|
+
for college_type in ["selective", "elite"]
|
|
1533
|
+
for degree_column, suffix in [
|
|
1534
|
+
("_degree", ""),
|
|
1535
|
+
("_degree_no_type_fill", "_noimpute"),
|
|
1536
|
+
]
|
|
1537
|
+
],
|
|
1498
1538
|
*[
|
|
1499
1539
|
expression
|
|
1500
1540
|
for definition in STEM_DEFINITIONS
|
|
@@ -1923,7 +1963,6 @@ for label, tiers in tier_buckets.items():
|
|
|
1923
1963
|
.otherwise(pl.col("adj_cmp_rate_4yr"))
|
|
1924
1964
|
.alias(f"adj_cmp_rate_4yr_coarse_{label}")
|
|
1925
1965
|
)
|
|
1926
|
-
print(f"Four-year IPEDS-cohort-weighted completion rate, {label}: {bucket_rate}")
|
|
1927
1966
|
|
|
1928
1967
|
# Match the available 1098-T calendar years, including the missing 2015 year.
|
|
1929
1968
|
tax_years = list(range(2011, 2015)) + list(range(2016, 2023))
|
|
@@ -2005,6 +2044,16 @@ nsc_outcomes = nsc_outcomes.with_columns(
|
|
|
2005
2044
|
.alias("recovered_nsc_outcome")
|
|
2006
2045
|
)
|
|
2007
2046
|
|
|
2047
|
+
# Export every selectivity window, plus the BA sensitivity outcomes. Keep the
|
|
2048
|
+
# existing recovered-outcome definition above independent of this export list.
|
|
2049
|
+
selectivity_outcome_columns = [
|
|
2050
|
+
column for column in nsc_outcomes.columns
|
|
2051
|
+
if column.startswith((
|
|
2052
|
+
"att_selective_", "att_elite_", "cmp_selective_", "cmp_elite_",
|
|
2053
|
+
"cmp_BA_selective_", "cmp_BA_elite_", "cmp_BA_noimpute_",
|
|
2054
|
+
))
|
|
2055
|
+
]
|
|
2056
|
+
|
|
2008
2057
|
keep_columns = [
|
|
2009
2058
|
"sid_cepr",
|
|
2010
2059
|
"k_mean",
|
|
@@ -2023,140 +2072,36 @@ keep_columns = [
|
|
|
2023
2072
|
"adj_cmp_rate_coarsen_2yr",
|
|
2024
2073
|
"ID_FSC_firstinst",
|
|
2025
2074
|
"college_name_firstinst",
|
|
2075
|
+
"unitid_firstinst",
|
|
2076
|
+
"college_years_firstinst",
|
|
2026
2077
|
"tier_firstinst",
|
|
2027
2078
|
"completion_rate_150pct_firstinst",
|
|
2028
|
-
] + outcome_columns + stem_outcome_columns
|
|
2079
|
+
] + outcome_columns + stem_outcome_columns + selectivity_outcome_columns
|
|
2080
|
+
keep_columns = list(dict.fromkeys(keep_columns))
|
|
2029
2081
|
|
|
2030
2082
|
nsc_outcomes_final = nsc_outcomes.select(keep_columns)
|
|
2031
2083
|
|
|
2084
|
+
# Retain the existing completeness check without generating audit tables.
|
|
2085
|
+
if nsc_outcomes.filter(
|
|
2086
|
+
pl.col("att_any_byY4").is_not_null() & pl.col("k_mean_coarse").is_null()
|
|
2087
|
+
).height > 0:
|
|
2088
|
+
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2089
|
+
|
|
2032
2090
|
OUTCOMES.parent.mkdir(parents=True, exist_ok=True)
|
|
2033
2091
|
nsc_outcomes.write_parquet(OUTCOMES)
|
|
2034
2092
|
nsc_outcomes.with_columns(
|
|
2035
2093
|
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2036
2094
|
).write_csv(OUTCOMES.with_suffix(".csv"))
|
|
2037
2095
|
|
|
2096
|
+
# Save the selected analysis/audit table as well as the existing full output.
|
|
2097
|
+
final_path = OUTCOMES.with_name(f"{OUTCOMES.stem}_final.parquet")
|
|
2098
|
+
nsc_outcomes_final.write_parquet(final_path)
|
|
2099
|
+
nsc_outcomes_final.with_columns(
|
|
2100
|
+
pl.all().exclude(pl.Date).cast(pl.String, strict=False)
|
|
2101
|
+
).write_csv(final_path.with_suffix(".csv"))
|
|
2102
|
+
|
|
2103
|
+
print(f"Wrote {final_path}")
|
|
2104
|
+
print(f"Wrote {final_path.with_suffix('.csv')}")
|
|
2038
2105
|
print(f"Wrote {OUTCOMES}")
|
|
2039
2106
|
print(f"Wrote {OUTCOMES.with_suffix('.csv')}")
|
|
2040
2107
|
print(f"Rows: {nsc_outcomes.height}, columns: {len(nsc_outcomes.columns)}")
|
|
2041
|
-
|
|
2042
|
-
first_college_coverage = (
|
|
2043
|
-
nsc_outcomes.filter(pl.col("ID_FSC_firstinst").is_not_null())
|
|
2044
|
-
.group_by("ID_FSC_firstinst")
|
|
2045
|
-
.agg(
|
|
2046
|
-
pl.col("k_mean_firstinst").is_not_null().any().alias("has_k_mean"),
|
|
2047
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2048
|
-
.is_not_null()
|
|
2049
|
-
.any()
|
|
2050
|
-
.alias("has_completion_rate"),
|
|
2051
|
-
)
|
|
2052
|
-
)
|
|
2053
|
-
print(
|
|
2054
|
-
"First-college coverage: "
|
|
2055
|
-
f"{first_college_coverage['has_k_mean'].sum()}/"
|
|
2056
|
-
f"{first_college_coverage.height} with college-specific k_mean; "
|
|
2057
|
-
f"{first_college_coverage['has_completion_rate'].sum()}/"
|
|
2058
|
-
f"{first_college_coverage.height} with IPEDS completion rate"
|
|
2059
|
-
)
|
|
2060
|
-
print(
|
|
2061
|
-
"National student-weighted coarse values: "
|
|
2062
|
-
f"k_mean 4yr={k_mean_4yr_coarse:.2f}, "
|
|
2063
|
-
f"k_mean 2yr-or-less={k_mean_2yr_coarse:.2f}, "
|
|
2064
|
-
f"completion 4yr={cmp_rate_4yr_coarse:.4f}, "
|
|
2065
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse:.4f}; "
|
|
2066
|
-
"observed by-Y4 nonattender completion=0"
|
|
2067
|
-
)
|
|
2068
|
-
print(
|
|
2069
|
-
"Sample-institution IPEDS-cohort-weighted coarse values: "
|
|
2070
|
-
f"completion 4yr={cmp_rate_4yr_coarse_sample:.4f} "
|
|
2071
|
-
f"({sample_4yr_institutions} institutions), "
|
|
2072
|
-
f"completion 2yr-or-less={cmp_rate_2yr_coarse_sample:.4f} "
|
|
2073
|
-
f"({sample_2yr_institutions} institutions)"
|
|
2074
|
-
)
|
|
2075
|
-
|
|
2076
|
-
observable_y4 = nsc_outcomes.filter(pl.col("att_any_byY4").is_not_null())
|
|
2077
|
-
coarse_audit = observable_y4.select(
|
|
2078
|
-
pl.len().alias("students"),
|
|
2079
|
-
pl.col("k_mean_coarse").null_count().alias("k_mean_coarse_missing"),
|
|
2080
|
-
pl.col("k_mean_coarse").min().alias("k_mean_coarse_min"),
|
|
2081
|
-
pl.col("k_mean_coarse").max().alias("k_mean_coarse_max"),
|
|
2082
|
-
pl.col("adj_cmp_rate_coarse")
|
|
2083
|
-
.null_count()
|
|
2084
|
-
.alias("adj_cmp_rate_coarse_missing"),
|
|
2085
|
-
pl.col("adj_cmp_rate_coarse").min().alias("adj_cmp_rate_coarse_min"),
|
|
2086
|
-
pl.col("adj_cmp_rate_coarse").max().alias("adj_cmp_rate_coarse_max"),
|
|
2087
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2088
|
-
.null_count()
|
|
2089
|
-
.alias("adj_cmp_rate_coarse_sample_missing"),
|
|
2090
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2091
|
-
.min()
|
|
2092
|
-
.alias("adj_cmp_rate_coarse_sample_min"),
|
|
2093
|
-
pl.col("adj_cmp_rate_coarse_sample")
|
|
2094
|
-
.max()
|
|
2095
|
-
.alias("adj_cmp_rate_coarse_sample_max"),
|
|
2096
|
-
)
|
|
2097
|
-
|
|
2098
|
-
if coarse_audit.item(0, "k_mean_coarse_missing") > 0:
|
|
2099
|
-
raise ValueError("k_mean_coarse is unexpectedly missing for observable students.")
|
|
2100
|
-
# Attendees without a classified first college can have missing coarse rates.
|
|
2101
|
-
# Report those counts below rather than treating them as a build failure.
|
|
2102
|
-
print("Coarse outcome audit:")
|
|
2103
|
-
print(coarse_audit)
|
|
2104
|
-
|
|
2105
|
-
attendee_completion_audit = (
|
|
2106
|
-
observable_y4.filter(pl.col("att_any_byY4") == 1)
|
|
2107
|
-
.select(
|
|
2108
|
-
pl.len().alias("by_y4_attendees"),
|
|
2109
|
-
pl.col("completion_rate_150pct_firstinst")
|
|
2110
|
-
.is_null()
|
|
2111
|
-
.sum()
|
|
2112
|
-
.alias("missing_first_institution_rate"),
|
|
2113
|
-
(
|
|
2114
|
-
pl.col("completion_rate_150pct_firstinst").is_null()
|
|
2115
|
-
& pl.col("completion_rate_150pct_ip").is_not_null()
|
|
2116
|
-
)
|
|
2117
|
-
.sum()
|
|
2118
|
-
.alias("filled_by_tier_median"),
|
|
2119
|
-
pl.col("adj_cmp_rate")
|
|
2120
|
-
.is_null()
|
|
2121
|
-
.sum()
|
|
2122
|
-
.alias("missing_after_tier_median"),
|
|
2123
|
-
pl.col("adj_cmp_rate")
|
|
2124
|
-
.is_null()
|
|
2125
|
-
.mean()
|
|
2126
|
-
.alias("missing_after_tier_median_share"),
|
|
2127
|
-
(
|
|
2128
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2129
|
-
& pl.col("tier_firstinst").is_null()
|
|
2130
|
-
)
|
|
2131
|
-
.sum()
|
|
2132
|
-
.alias("missing_after_tier_median_no_tier"),
|
|
2133
|
-
(
|
|
2134
|
-
pl.col("adj_cmp_rate").is_null()
|
|
2135
|
-
& pl.col("tier_firstinst").is_not_null()
|
|
2136
|
-
)
|
|
2137
|
-
.sum()
|
|
2138
|
-
.alias("missing_after_tier_median_with_tier"),
|
|
2139
|
-
)
|
|
2140
|
-
)
|
|
2141
|
-
print("By-Y4 attendee completion-rate audit:")
|
|
2142
|
-
print(attendee_completion_audit)
|
|
2143
|
-
print(
|
|
2144
|
-
nsc_outcomes.select("sid_cepr", "cohort_lottery", "recovered_nsc_outcome").sort(
|
|
2145
|
-
"sid_cepr"
|
|
2146
|
-
)
|
|
2147
|
-
)
|
|
2148
|
-
|
|
2149
|
-
# Nonmissing year-eight completion identifies students with the full Y8 window.
|
|
2150
|
-
year8_attendance_audit = (
|
|
2151
|
-
nsc_outcomes.filter(pl.col("cmp_BA_byY8").is_not_null())
|
|
2152
|
-
.select(
|
|
2153
|
-
pl.len().alias("N_year8_sample"),
|
|
2154
|
-
(pl.col("att_any_byY4") == 0).sum().alias("N_no_attendance_byY4"),
|
|
2155
|
-
(
|
|
2156
|
-
(pl.col("att_any_byY4") == 0)
|
|
2157
|
-
& (pl.col("cmp_BA_byY8") == 1)
|
|
2158
|
-
).sum().alias("N_no_attendance_byY4_BA_byY8"),
|
|
2159
|
-
)
|
|
2160
|
-
)
|
|
2161
|
-
print("Year-eight sample attendance and late BA completion check:")
|
|
2162
|
-
print(year8_attendance_audit)
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/aspire.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/check_cmo_apps.do
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/christel_house.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/democracy_prep.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/green_dot.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/helpers.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/kipp_nj.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/mappings.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/rocketship.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260614_new_build_scripts_update/yes_prep.py
RENAMED
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/christel_house.py
RENAMED
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/20260630_census_disclosure/democracy_prep.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{ltc_code-0.2.29 → ltc_code-0.2.31}/src/ltc_code/nsc/dhs_stem/dhs_stem_cip_additions_2024.csv
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|