diff --git a/coverage-badge.svg b/coverage-badge.svg index 2969548..a357f5b 100644 --- a/coverage-badge.svg +++ b/coverage-badge.svg @@ -1 +1 @@ - \ No newline at end of file + \ No newline at end of file diff --git a/data/Average Year-over-Year Change in ER Visits.png b/data/Average Year-over-Year Change in ER Visits.png deleted file mode 100644 index 3bdd21c..0000000 Binary files a/data/Average Year-over-Year Change in ER Visits.png and /dev/null differ diff --git a/data/Capacity (Visits per Station) vs Demand (Total Visits).png b/data/Capacity (Visits per Station) vs Demand (Total Visits).png deleted file mode 100644 index e49a8f7..0000000 Binary files a/data/Capacity (Visits per Station) vs Demand (Total Visits).png and /dev/null differ diff --git a/data/load_distribution_HospitalOwnership.png b/data/load_distribution_HospitalOwnership.png deleted file mode 100644 index 864340e..0000000 Binary files a/data/load_distribution_HospitalOwnership.png and /dev/null differ diff --git a/data/load_distribution_hospital_ownership.png b/data/load_distribution_hospital_ownership.png deleted file mode 100644 index 72af8a7..0000000 Binary files a/data/load_distribution_hospital_ownership.png and /dev/null differ diff --git a/data/urban_rural_map_California.html b/data/urban_rural_map_California.html deleted file mode 100644 index cf10cc8..0000000 --- a/data/urban_rural_map_California.html +++ /dev/null @@ -1,20560 +0,0 @@ - - -
- - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - \ No newline at end of file diff --git a/data/urban_rural_map_california.html b/data/urban_rural_map_california.html deleted file mode 100644 index cf10cc8..0000000 --- a/data/urban_rural_map_california.html +++ /dev/null @@ -1,20560 +0,0 @@ - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - \ No newline at end of file diff --git a/demo_outputs/california_emergency_data_sample.csv b/demo_outputs/california_emergency_data_sample.csv new file mode 100644 index 0000000..3b896c2 --- /dev/null +++ b/demo_outputs/california_emergency_data_sample.csv @@ -0,0 +1,101 @@ +facility_id,facility_name,county_name,hospital_system,year,licensed_bed_size,hospital_ownership,urban_rural_designation,teaching_designation,category,total_ed_visits,ed_stations,ed_burden,latitude,longitude,primary_care_shortage_area,mental_health_shortage_area,visits_per_station +106010735,Alameda Hospital,Alameda,Alameda Health System,2022,100-149,Government,Urban,Non-Teaching,Active COVID-19,13579,12.0,520,37.76266,-122.253991,No,No,43.3333333333333 +106010739,Alta Bates Summit Medical Center – Alta Bates Campus,Alameda,Sutter Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,29760,22.0,1177,37.85645,-122.25743,No,No,53.5 +106010776,UC San Francisco Benioff Children's Hospital Oakland,Alameda,,2022,150-199,Nonprofit,Urban,Teaching,Active COVID-19,35062,40.0,1822,37.83722,-122.26747,No,No,45.55 +106010846,Highland Hospital,Alameda,Alameda Health System,2022,200-299,Government,Urban,Non-Teaching,Active COVID-19,66095,53.0,2109,37.79925,-122.23138,No,No,39.7924528301887 +106010937,Alta Bates Summit Medical Center,Alameda,Sutter Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,29336,30.0,1325,37.82106,-122.26257,No,No,44.1666666666667 +106010987,Washington Hospital – Fremont,Alameda,,2022,300-499,Government,Urban,Non-Teaching,Active COVID-19,43964,39.0,3013,37.55847,-121.98006,No,No,77.2564102564103 +106014050,Stanford Health Care Tri-Valley,Alameda,,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,28620,17.0,1378,37.69206,-121.88095,No,No,81.0588235294118 +106014132,Kaiser Foundation Hospital – Fremont,Alameda,Kaiser Foundation Hospitals,2022,100-149,Nonprofit,Urban,Non-Teaching,Active COVID-19,38960,18.0,1487,37.55055,-121.97483,No,No,82.6111111111111 +106014233,Eden Medical Center,Alameda,Sutter Health,2022,100-149,Nonprofit,Urban,Non-Teaching,Active COVID-19,31062,22.0,1467,37.6983769,-122.0874055,No,No,66.6818181818182 +106014326,Kaiser Foundation Hospital – Oakland/Richmond,Alameda,Kaiser Foundation Hospitals,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,120088,80.0,4883,37.82402,-122.25791,No,No,61.0375 +106014337,Kaiser Foundation Hospital – San Leandro,Alameda,Kaiser Foundation Hospitals,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,69343,40.0,2893,37.70611,-122.16917,No,No,72.325 +106034002,Sutter Amador Hospital,Amador,Sutter Health,2022,50-99,Nonprofit,Rural,Non-Teaching,Active COVID-19,24996,14.0,1392,38.34743,-120.77272,Yes,No,99.4285714285714 +106040802,Orchard Hospital,Butte,,2022,100-149,Nonprofit,Rural,Non-Teaching,Active COVID-19,10325,5.0,366,39.36701,-121.68955,Yes,Yes,73.2 +106040937,Oroville Hospital,Butte,,2022,100-149,Nonprofit,Rural,Non-Teaching,Active COVID-19,9329,14.0,377,39.50485,-121.5428,Yes,Yes,26.9285714285714 +106040962,Enloe Medical Center– Esplanade,Butte,,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,56376,50.0,1902,39.74226,-121.84918,No,Yes,38.04 +106050932,Mark Twain Medical Center,Calaveras,Dignity Health,2022,1-49,Nonprofit,Rural,Non-Teaching,Active COVID-19,10190,8.0,497,38.19057,-120.67095,Yes,Yes,62.125 +106070934,Sutter Delta Medical Center,Contra Costa,Sutter Health,2022,100-149,Nonprofit,Urban,Non-Teaching,Active COVID-19,41182,30.0,2082,37.98222,-121.8054,No,No,69.4 +106070988,John Muir Medical Center – Walnut Creek Campus,Contra Costa,John Muir Health,2022,500+,Nonprofit,Urban,Non-Teaching,Active COVID-19,47848,44.0,2205,37.91304,-122.040781,No,No,50.1136363636364 +106070990,Kaiser Foundation Hospital – Walnut Creek,Contra Costa,Kaiser Foundation Hospitals,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,65573,52.0,3007,37.8923,-122.05828,No,No,57.8269230769231 +106071018,John Muir Medical Center – Concord Campus,Contra Costa,John Muir Health,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,52369,32.0,2404,37.98615,-122.03874,No,No,75.125 +106074017,San Ramon Regional Medical Center,Contra Costa,Tenet Healthcare Corporation,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,18054,12.0,1009,37.77549,-121.95884,No,No,84.0833333333333 +106074097,Kaiser Foundation Hospital – Antioch,Contra Costa,Kaiser Foundation Hospitals,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,66644,36.0,3385,37.951975,-121.776854,No,No,94.0277777777778 +106090793,Barton Memorial Hospital,El Dorado,,2022,100-149,Nonprofit,Rural,Non-Teaching,Active COVID-19,17483,12.0,596,38.91228,-119.99758,Yes,Yes,49.6666666666667 +106100005,Clovis Community Medical Center,Fresno,,2022,300-499,Nonprofit,Rural,Non-Teaching,Active COVID-19,56532,59.0,2051,36.83745,-119.66072,Yes,Yes,34.7627118644068 +106100697,Coalinga Regional Medical Center,Fresno,,2022,100-149,Investor Owned,Rural,Non-Teaching,Active COVID-19,8870,9.0,361,36.15152,-120.34035,Yes,Yes,40.1111111111111 +106100717,Community Regional Medical Center – Fresno,Fresno,,2022,500+,Nonprofit,Urban,Teaching,Active COVID-19,84815,73.0,2960,36.74252,-119.784071,Yes,No,40.5479452054795 +106100797,Adventist Health Reedley,Fresno,Adventist Health Systems,2022,1-49,Nonprofit,Rural,Non-Teaching,Active COVID-19,38319,10.0,2518,36.60789,-119.45145,Yes,Yes,251.8 +106100899,Saint Agnes Medical Center,Fresno,,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,58162,48.0,2182,36.8370782,-119.7649963,No,Yes,45.4583333333333 +106104062,Kaiser Foundation Hospital – Fresno,Fresno,Kaiser Foundation Hospitals,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,49785,21.0,2289,36.84246,-119.7832,No,Yes,109.0 +106110889,Glenn Medical Center,Glenn,,2022,1-49,Investor Owned,Frontier,Non-Teaching,Active COVID-19,6404,4.0,177,39.52055,-122.20791,Yes,Yes,44.25 +106121051,Providence Redwood Memorial Hospital,Humboldt,Providence St. Joseph Health,2022,1-49,Investor Owned,Rural,Non-Teaching,Active COVID-19,10531,8.0,353,40.58267,-124.13549,Yes,Yes,44.125 +106121080,Providence St. Joseph Hospital – Eureka,Humboldt,Providence St. Joseph Health,2022,100-149,Investor Owned,Rural,Non-Teaching,Active COVID-19,27953,20.0,882,40.7832,-124.14216,Yes,Yes,44.1 +106130760,Pioneers Memorial Healthcare District,Imperial,,2022,100-149,Government,Rural,Non-Teaching,Active COVID-19,42926,16.0,2435,32.96003,-115.55183,Yes,Yes,152.1875 +106141273,Northern Inyo Hospital,Inyo,,2022,1-49,Government,Rural,Non-Teaching,Active COVID-19,8301,8.0,368,37.36207,-118.40756,No,No,46.0 +106141338,Southern Inyo Hospital,Inyo,,2022,1-49,Government,Frontier,Non-Teaching,Active COVID-19,1668,2.0,63,36.6084,-118.05816,Yes,No,31.5 +106150706,Adventist Health Delano,Kern,,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,28475,10.0,945,35.76143,-119.23706,Yes,Yes,94.5 +106150737,Kern Valley Healthcare District,Kern,,2022,100-149,Government,Frontier,Non-Teaching,Active COVID-19,6607,8.0,280,35.63486,-118.40509,Yes,Yes,35.0 +106150761,Mercy Hospital – Bakersfield,Kern,Dignity Health,2022,100-149,Nonprofit,Urban,Non-Teaching,Active COVID-19,62403,45.0,2746,35.37323,-119.02715,No,Yes,61.0222222222222 +106150782,Ridgecrest Regional Hospital,Kern,,2022,50-99,Nonprofit,Rural,Non-Teaching,Active COVID-19,16279,12.0,738,35.6404,-117.66996,Yes,Yes,61.5 +106154168,Adventist Health Tehachapi Valley,Kern,Adventist Health Systems,2022,1-49,Nonprofit,Rural,Non-Teaching,Active COVID-19,21135,13.0,1130,35.14302,-118.44924,Yes,Yes,86.9230769230769 +106164029,Adventist Health Hanford,Kings,Adventist Health Systems,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,92128,44.0,4787,36.3256247,-119.6683912,Yes,Yes,108.795454545455 +106171049,Adventist Health Clearlake,Lake,Adventist Health Systems,2022,1-49,Nonprofit,Rural,Non-Teaching,Active COVID-19,19592,8.0,771,38.93619,-122.6202,Yes,Yes,96.375 +106171395,Sutter Lakeside Hospital,Lake,Sutter Health,2022,1-49,Nonprofit,Rural,Non-Teaching,Active COVID-19,19410,12.0,868,39.10232,-122.91003,Yes,Yes,72.3333333333333 +106184008,Banner Lassen Medical Center,Lassen,,2022,1-49,Nonprofit,Rural,Non-Teaching,Active COVID-19,10145,8.0,436,40.42306,-120.64609,Yes,Yes,54.5 +106190017,Alhambra Hospital Medical Center,Los Angeles,,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,16020,8.0,802,34.08988,-118.1449,No,No,100.25 +106190034,Antelope Valley Hospital,Los Angeles,,2022,300-499,Government,Urban,Non-Teaching,Active COVID-19,107707,45.0,4604,34.6878,-118.157981,No,Yes,102.311111111111 +106190053,St. Mary Medical Center – Long Beach,Los Angeles,Dignity Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,37582,26.0,1223,33.7802376,-118.1866412,No,No,47.0384615384615 +106190110,Southern California Hospital at Culver City,Los Angeles,Alta Hospitals System,2022,300-499,Investor Owned,Urban,Non-Teaching,Active COVID-19,12830,17.0,436,34.0231953,-118.3969429,No,No,25.6470588235294 +106190125,California Hospital Medical Center – Los Angeles,Los Angeles,Dignity Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,57917,35.0,1411,34.03721,-118.26551,No,No,40.3142857142857 +106190148,Centinela Hospital Medical Center,Los Angeles,Prime Healthcare Services,2022,300-499,Investor Owned,Urban,Non-Teaching,Active COVID-19,35849,50.0,712,33.94917,-118.34788,Yes,No,14.24 +106190170,Children's Hospital of Los Angeles,Los Angeles,,2022,300-499,Nonprofit,Urban,Teaching,Active COVID-19,68844,46.0,1716,34.0978692,-118.2909547,Yes,No,37.304347826087 +106190197,Community Hospital of Huntington Park,Los Angeles,Avanti Hospitals,2022,50-99,Investor Owned,Urban,Non-Teaching,Active COVID-19,31164,14.0,388,33.98929,-118.22466,No,Yes,27.7142857142857 +106190198,Los Angeles Community Hospital,Los Angeles,Alta Hospitals System,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,3857,4.0,71,34.01897,-118.1875,Yes,Yes,17.75 +106190200,San Gabriel Valley Medical Center,Los Angeles,"AHMC Healthcare, Inc.",2022,200-299,Investor Owned,Urban,Non-Teaching,Active COVID-19,21304,12.0,800,34.1024,-118.10503,No,No,66.6666666666667 +106190240,Lakewood Regional Medical Center,Los Angeles,Tenet Healthcare Corporation,2022,150-199,Investor Owned,Urban,Non-Teaching,Active COVID-19,34069,14.0,1077,33.859731,-118.1492012,No,No,76.9285714285714 +106190243,PIH Health Hospital – Downey,Los Angeles,,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,48690,26.0,1557,33.93515,-118.13196,No,No,59.8846153846154 +106190256,East Los Angeles Doctors Hospital,Los Angeles,Avanti Hospitals,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,10400,8.0,251,34.02383,-118.18377,Yes,Yes,31.375 +106190280,Encino Hospital Medical Center,Los Angeles,Prime Healthcare Services,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,6940,9.0,183,34.15686,-118.48662,No,No,20.3333333333333 +106190315,Garfield Medical Center,Los Angeles,"AHMC Healthcare, Inc.",2022,200-299,Investor Owned,Urban,Non-Teaching,Active COVID-19,15968,21.0,629,34.06826,-118.12302,No,No,29.952380952381 +106190323,Adventist Health Glendale,Los Angeles,Adventist Health Systems,2022,500+,Nonprofit,Urban,Non-Teaching,Active COVID-19,37767,39.0,2248,34.14951,-118.23109,No,No,57.6410256410256 +106190352,Greater El Monte Community Hospital,Los Angeles,"AHMC Healthcare, Inc.",2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,20914,,635,34.04827,-118.04257,Yes,No, +106190385,Providence Holy Cross Medical Center,Los Angeles,Providence St. Joseph Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,82929,32.0,2956,34.2791,-118.45951,No,No,92.375 +106190422,Torrance Memorial Medical Center,Los Angeles,Cedars-Sinai Health System,2022,500+,Nonprofit,Urban,Non-Teaching,Active COVID-19,71601,29.0,3585,33.8119157,-118.3434945,Yes,No,123.620689655172 +106190429,Kaiser Foundation Hospital – Los Angeles,Los Angeles,Kaiser Foundation Hospitals,2022,300-499,Nonprofit,Urban,Teaching,Active COVID-19,55882,57.0,3477,34.09875,-118.295371,Yes,No,61.0 +106190431,Kaiser Foundation Hospital – South Bay,Los Angeles,Kaiser Foundation Hospitals,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,59992,52.0,3467,33.78926,-118.29373,Yes,No,66.6730769230769 +106190432,Kaiser Foundation Hospital – Panorama City,Los Angeles,Kaiser Foundation Hospitals,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,62888,41.0,4248,34.2198,-118.43103,No,No,103.609756097561 +106190434,Kaiser Foundation Hospital – West LA,Los Angeles,Kaiser Foundation Hospitals,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,80860,53.0,4385,34.03793,-118.37568,No,No,82.7358490566038 +106190470,Providence Little Company of Mary Medical Center – Torrance,Los Angeles,Providence St. Joseph Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,40497,29.0,2155,33.83891,-118.356771,No,No,74.3103448275862 +106190500,Cedars – Sinai Marina Del Rey Hospital,Los Angeles,Cedars-Sinai Health System,2022,100-149,Nonprofit,Urban,Non-Teaching,Active COVID-19,32685,18.0,1313,33.98125,-118.43971,No,No,72.9444444444444 +106190517,Providence Cedars – Sinai Tarzana Medical Center,Los Angeles,Providence St. Joseph Health,2022,200-299,Investor Owned,Urban,Non-Teaching,Active COVID-19,40962,15.0,2034,34.17067,-118.532001,No,No,135.6 +106190521,Memorial Hospital of Gardena,Los Angeles,Avanti Hospitals,2022,150-199,Investor Owned,Urban,Non-Teaching,Active COVID-19,29497,10.0,389,33.89246,-118.29493,Yes,No,38.9 +106190525,MemorialCare Long Beach Medical Center,Los Angeles,MemorialCare,2022,300-499,Nonprofit,Urban,Teaching,Active COVID-19,72326,64.0,4344,33.80801,-118.1852,No,No,67.875 +106190529,USC Arcadia Hospital,Los Angeles,,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,35962,26.0,1638,34.13603,-118.03879,No,No,63.0 +106190547,Monterey Park Hospital,Los Angeles,"AHMC Healthcare, Inc.",2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,16386,6.0,471,34.05308,-118.13665,No,No,78.5 +106190555,Cedars – Sinai Medical Center,Los Angeles,Cedars-Sinai Health System,2022,500+,Nonprofit,Urban,Teaching,Active COVID-19,54587,60.0,2303,34.07681,-118.38061,No,No,38.3833333333333 +106190568,Northridge Hospital Medical Center,Los Angeles,Dignity Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,47234,37.0,2174,34.22075,-118.53186,No,No,58.7567567567568 +106190570,Norwalk Community Hospital,Los Angeles,Alta Hospitals System,2022,50-99,Investor Owned,Urban,Non-Teaching,Active COVID-19,5026,4.0,163,33.91108,-118.0644,No,No,40.75 +106190587,College Medical Center,Los Angeles,,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,6654,7.0,174,33.80759,-118.19355,No,No,24.8571428571429 +106190630,Pomona Valley Hospital Medical Center,Los Angeles,,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,78066,72.0,4775,34.07709,-117.750621,No,No,66.3194444444444 +106190631,PIH Health Hospital – Whittier,Los Angeles,,2022,500+,Nonprofit,Urban,Non-Teaching,Active COVID-19,58298,61.0,2422,33.9697943,-118.049255,No,No,39.7049180327869 +106190673,San Dimas Community Hospital,Los Angeles,Prime Healthcare Services,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,12256,8.0,417,34.099076,-117.834328,No,No,52.125 +106190680,Providence Little Company of Mary Medical Center – San Pedro,Los Angeles,Providence St. Joseph Health,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,33407,16.0,1674,33.737939,-118.303893,No,No,104.625 +106190687,Santa Monica – UCLA Medical Center and Orthopaedic Hospital,Los Angeles,University of California,2022,200-299,Government,Urban,Non-Teaching,Active COVID-19,33209,21.0,1381,34.027539,-118.486123,No,No,65.7619047619048 +106190696,Pacifica Hospital of the Valley,Los Angeles,,2022,200-299,Investor Owned,Urban,Non-Teaching,Active COVID-19,6908,7.0,186,34.24088,-118.39555,Yes,No,26.5714285714286 +106190708,Sherman Oaks Hospital,Los Angeles,Prime Healthcare Services,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,18208,12.0,423,34.15995,-118.4488,No,No,35.25 +106190754,St. Francis Medical Center,Los Angeles,Verity Health System,2022,300-499,Investor Owned,Urban,Non-Teaching,Active COVID-19,48363,42.0,1593,33.93085,-118.20415,Yes,Yes,37.9285714285714 +106190756,Providence Saint John's Health Center,Los Angeles,Providence St. Joseph Health,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,24761,27.0,1002,34.029804,-118.478668,No,No,37.1111111111111 +106190758,Providence Saint Joseph Medical Center,Los Angeles,Providence St. Joseph Health,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,50489,48.0,2375,34.15486,-118.32655,No,No,49.4791666666667 +106190766,Coast Plaza Hospital,Los Angeles,Avanti Hospitals,2022,100-149,Investor Owned,Urban,Non-Teaching,Active COVID-19,9366,8.0,138,33.91255,-118.09915,No,No,17.25 +106190812,Valley Presbyterian Hospital,Los Angeles,,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,44727,30.0,1491,34.19396,-118.4632,No,No,49.7 +106190818,USC Verdugo Hills Hospital,Los Angeles,University of Southern California,2022,150-199,Nonprofit,Urban,Non-Teaching,Active COVID-19,22041,13.0,1265,34.20557,-118.2152,No,No,97.3076923076923 +106190859,West Hills Hospital and Medical Center,Los Angeles,HCA Healthcare Corporation,2022,200-299,Investor Owned,Urban,Non-Teaching,Active COVID-19,46662,30.0,914,34.20337,-118.62939,No,No,30.4666666666667 +106190878,Adventist Health White Memorial,Los Angeles,Adventist Health Systems,2022,300-499,Nonprofit,Urban,Teaching,Active COVID-19,52148,28.0,1607,34.05085,-118.21691,No,No,57.3928571428571 +106190883,Whittier Hospital Medical Center,Los Angeles,"AHMC Healthcare, Inc.",2022,150-199,Investor Owned,Urban,Non-Teaching,Active COVID-19,22499,12.0,760,33.95379,-118.00303,No,No,63.3333333333333 +106190949,Henry Mayo Newhall Hospital,Los Angeles,,2022,300-499,Nonprofit,Urban,Non-Teaching,Active COVID-19,54067,30.0,2380,34.39545,-118.55476,No,No,79.3333333333333 +106191227,Los Angeles County/Harbor – UCLA Medical Center,Los Angeles,County of Los Angeles,2022,300-499,Government,Urban,Teaching,Active COVID-19,77616,73.0,2183,33.83155,-118.29353,Yes,No,29.9041095890411 +106191228,Los Angeles County+USC Medical Center,Los Angeles,County of Los Angeles,2022,500+,Government,Urban,Teaching,Active COVID-19,104935,106.0,1846,34.05982,-118.21031,No,No,17.4150943396226 +106191230,"Martin Luther King, Jr. Community Hospital",Los Angeles,,2022,100-149,Nonprofit,Urban,Teaching,Active COVID-19,99493,29.0,4238,33.92453,-118.2435,Yes,Yes,146.137931034483 +106191231,Los Angeles County Olive View – UCLA Medical Center,Los Angeles,County of Los Angeles,2022,300-499,Government,Urban,Teaching,Active COVID-19,47977,51.0,1112,34.32418,-118.45255,No,No,21.8039215686275 +106191450,Kaiser Foundation Hospital – Woodland Hills,Los Angeles,Kaiser Foundation Hospitals,2022,200-299,Nonprofit,Urban,Non-Teaching,Active COVID-19,40713,36.0,2674,34.17198,-118.58829,No,No,74.2777777777778 diff --git a/data/category_visits_plot.png b/demo_outputs/category_visits_plot.png similarity index 100% rename from data/category_visits_plot.png rename to demo_outputs/category_visits_plot.png diff --git a/demo_outputs/facility_category_visits_plot.png b/demo_outputs/facility_category_visits_plot.png new file mode 100644 index 0000000..117187a Binary files /dev/null and b/demo_outputs/facility_category_visits_plot.png differ diff --git a/demo_outputs/facility_trend.png b/demo_outputs/facility_trend.png new file mode 100644 index 0000000..0475729 Binary files /dev/null and b/demo_outputs/facility_trend.png differ diff --git a/demo_outputs/hospital_load_distribution.png b/demo_outputs/hospital_load_distribution.png new file mode 100644 index 0000000..9f02204 Binary files /dev/null and b/demo_outputs/hospital_load_distribution.png differ diff --git a/demo_outputs/urban_rural_map.html b/demo_outputs/urban_rural_map.html new file mode 100644 index 0000000..975b6c0 --- /dev/null +++ b/demo_outputs/urban_rural_map.html @@ -0,0 +1,20545 @@ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + \ No newline at end of file diff --git a/run_demo.py b/run_demo.py index c5c3994..6b53d62 100644 --- a/run_demo.py +++ b/run_demo.py @@ -1,83 +1,486 @@ from pathlib import Path -from ertimes.stats_visualization import plot_category_visits, plot_hospital_load_distribution, plot_urban_rural_map -from ertimes.io import download_emergency_data, load_emergency_data -from ertimes.stats_analysis import find_capacity_volume_mismatch, county_capacity_summary, rank_counties_by_burden -from ertimes.stats_ranking import rank_hospitals_by_visits_per_station -from ertimes.stats_reports import summarize_by_ownership import pandas as pd +from ertimes.io import download_emergency_data, load_emergency_data + +from ertimes.stats_analysis import ( + county_capacity_summary, + find_capacity_volume_mismatch, + compute_capacity_pressure_score, + mental_health_shortage_analysis, + calculate_growth, + run_er_analysis, + county_facility_counts, + spike_frequency_pivot, + load_california_income_data, + get_income_by_zip, + get_income_by_county, +) + +from ertimes.stats_ranking import ( + rank_counties_by_burden, + rank_hospitals_by_visits_per_station, +) -df = download_emergency_data("california") +from ertimes.stats_reports import ( + generate_county_report, + summarize_by_ownership, + find_duplicates, + per_category_burden_report, +) -#raises error -result = find_capacity_volume_mismatch( - df, - high_visit_quantile=0.65, - low_capacity_quantile=0.35, +from ertimes.stats_visualization import ( + plot_hospital_load_distribution, + plot_facility_trend, + plot_category_visits, + plot_category_visits_by_facility, + plot_urban_rural_map, ) -print(result.head(10)) -print(f"Rows returned: {len(result)}") +from ertimes.demographics import ( + get_income_statistics, + filter_by_income_range, +) -print("Downloading data for test...") -df = download_emergency_data("california") +OUTPUT_DIR = Path("demo_outputs") +OUTPUT_DIR.mkdir(exist_ok=True) -print("Generating plot...") -plot_hospital_load_distribution(df, group_col='hospital_ownership', save=True) -print("Generating urban-rural map...") -plot_urban_rural_map("california", save=True) +def print_section(title: str) -> None: + """Print a clear section header for demo output.""" + print("\n" + "=" * 80) + print(title) + print("=" * 80) -print("Check your 'data/' folder for the new files!") -#test for health conditions bar plot -def demo_plot_category_visits_downloaded() -> None: - """Download emergency data for California and render the category visits plot.""" - print("Downloading emergency data for California...") +def demo_download_emergency_data() -> pd.DataFrame: + """Demo downloading and cleaning emergency department data.""" + print_section("1. Download Emergency Department Data") + + print("Downloading California emergency department data...") df = download_emergency_data("california") - print("Generating category visits plot from downloaded data...") - plot_category_visits(df, save=True) + print("Download complete.") + print(f"Shape: {df.shape}") + print("\nColumns:") + print(df.columns.tolist()) + + print("\nFirst 5 rows:") + print(df.head()) + + return df + + +def demo_load_emergency_data_from_saved_csv(df: pd.DataFrame) -> pd.DataFrame: + """Demo saving then loading emergency data locally.""" + print_section("2. Save and Reload Emergency Data Locally") + + csv_path = OUTPUT_DIR / "california_emergency_data_sample.csv" + df.head(100).to_csv(csv_path, index=False) + + print(f"Saved first 100 rows to: {csv_path}") + loaded_df = load_emergency_data(csv_path) + + print("Reloaded local CSV successfully.") + print(f"Reloaded shape: {loaded_df.shape}") + print(loaded_df.head()) + + return loaded_df + + +def demo_capacity_volume_mismatch(df: pd.DataFrame) -> pd.DataFrame: + """Demo identifying hospitals with high demand and low capacity.""" + print_section("3. Capacity-Volume Mismatch Analysis") + + result = find_capacity_volume_mismatch( + df, + high_visit_quantile=0.65, + low_capacity_quantile=0.35, + facility_col="facility_name", + county_col="county_name", + year_col="year", + visits_col="total_ed_visits", + stations_col="ed_stations", + bed_col="licensed_bed_size", + ) + + print("Hospitals with high visits per station and relatively low bed capacity:") + print(result.head(10)) + print(f"\nRows returned: {len(result)}") + + return result + + +def demo_county_capacity_summary() -> pd.DataFrame: + """Demo county-level capacity summary.""" + print_section("4. County Capacity Summary") + + summary = county_capacity_summary( + "california", + county_col="county_name", + visits_col="total_ed_visits", + stations_col="ed_stations", + bed_col="licensed_bed_size", + ) + + print("County-level capacity summary:") + print(summary.head(10)) + print(f"\nRows returned: {len(summary)}") -def demo_county_capacity_summary(): - """Demo county capacity summary.""" - print("Generating county capacity summary...") - summary = county_capacity_summary("california") - print(summary.head()) return summary -def demo_rank_counties(): - """Demo ranking counties by burden.""" - print("Ranking counties by burden...") - summary = demo_county_capacity_summary() - ranked = rank_counties_by_burden(summary) - print(ranked.head()) + +def demo_rank_counties(summary: pd.DataFrame) -> pd.DataFrame: + """Demo ranking counties by emergency department burden.""" + print_section("5. Rank Counties by Burden") + + ranked = rank_counties_by_burden( + summary, + visits_col="visits_per_station", + ) + + print("Top 10 counties by visits per station:") + print(ranked.head(10)) + return ranked -def demo_rank_hospitals(): + +def demo_generate_county_report(summary: pd.DataFrame) -> pd.DataFrame: + """Demo generating a single-county report.""" + print_section("6. Generate Single County Report") + + if summary.empty: + print("County summary is empty; skipping county report.") + return pd.DataFrame() + + county_name = summary.iloc[0]["county_name"] + + report = generate_county_report( + summary, + county_name=county_name, + county_col="county_name", + visits_col="tot_ed_visits", + stations_col="ed_stations", + beds_col="licensed_bed_size", + visits_per_station_col="visits_per_station", + ) + + print(f"Generated report for county: {county_name}") + print(report) + + return report + + +def demo_rank_hospitals(df: pd.DataFrame) -> pd.DataFrame: """Demo ranking hospitals by visits per station.""" - print("Ranking hospitals by visits per station...") - df = download_emergency_data("california") - ranked = rank_hospitals_by_visits_per_station(df, top_n=5) + print_section("7. Rank Hospitals by Visits per Station") + + ranked = rank_hospitals_by_visits_per_station( + df, + agg="median", + top_n=10, + facility_col="facility_name", + visits_col="visits_per_station", + ) + + print("Top 10 hospitals by median visits per station:") print(ranked) + return ranked -def demo_summarize_by_ownership(): - """Demo summarizing by ownership.""" - print("Summarizing by ownership...") - df = download_emergency_data("california") - summary = summarize_by_ownership(df) - print(summary) + +def demo_summarize_by_ownership(df: pd.DataFrame) -> pd.DataFrame: + """Demo summarizing emergency data by hospital ownership type.""" + print_section("8. Summarize by Hospital Ownership") + + summary = summarize_by_ownership( + df, + column_map={ + "hospital_ownership": "hospital_ownership", + "tot_ed_visits": "total_ed_visits", + "ed_stations": "ed_stations", + "visits_per_station": "visits_per_station", + }, + ) + + print("Ownership-level summary:") + print(summary.head(10)) + return summary -demo_plot_category_visits_downloaded() -# Run demos +def demo_find_duplicates(df: pd.DataFrame) -> pd.DataFrame: + """Demo duplicate detection.""" + print_section("9. Find Duplicate Rows") + + duplicate_rows = find_duplicates( + df, + subset=["facility_name", "county_name", "year", "category"], + ) + + print("Potential duplicate rows using facility, county, year, and category:") + print(duplicate_rows.head(10)) + print(f"\nDuplicate rows found: {len(duplicate_rows)}") + + return duplicate_rows + + +def demo_per_category_burden_report(df: pd.DataFrame) -> dict[str, list[str]]: + """Demo top burdened facilities by condition category.""" + print_section("10. Per-Category Burden Report") + + report = per_category_burden_report( + df, + top_n=3, + facility_col="facility_name", + category_col="category", + visits_col="visits_per_station", + ) + + print("Top 3 facilities by visits per station for each category:") + for category, facilities in list(report.items())[:10]: + print(f"{category}: {facilities}") + + return report + + +def demo_capacity_pressure_score(df: pd.DataFrame) -> pd.DataFrame: + """Demo composite capacity pressure score.""" + print_section("11. Capacity Pressure Score") + + scores = compute_capacity_pressure_score( + df, + facility_col="facility_name", + visits_col="visits_per_station", + bed_col="licensed_bed_size", + primary_shortage_col="primary_care_shortage_area", + mental_shortage_col="mental_health_shortage_area", + ) + + print("Top 10 facilities by capacity pressure score:") + print(scores.head(10)) + + return scores + + +def demo_mental_health_shortage_analysis(df: pd.DataFrame) -> pd.DataFrame: + """Demo high-burden facilities in mental health shortage areas.""" + print_section("12. Mental Health Shortage Analysis") + + result = mental_health_shortage_analysis( + df, + percentile_threshold=80, + visits_col="total_ed_visits", + stations_col="ed_stations", + shortage_col="mental_health_shortage_area", + year_col="year", + ) + + print("Facilities in mental health shortage areas with high ED burden:") + print(result.head(10)) + print(f"\nRows returned: {len(result)}") + + return result + + +def demo_growth_analysis(df: pd.DataFrame) -> pd.DataFrame: + """Demo year-over-year growth calculations.""" + print_section("13. Year-over-Year Growth Analysis") + + result = calculate_growth( + df, + value_col="total_ed_visits", + group_cols=["facility_name"], + time_col="year", + pct=True, + ) + + print("Growth analysis sample:") + print(result[["facility_name", "year", "total_ed_visits", "growth"]].head(15)) + + return result + + +def demo_run_er_analysis(df: pd.DataFrame) -> pd.DataFrame: + """Demo general ER analysis helper.""" + print_section("14. General ER Analysis") + + result = run_er_analysis( + df, + facility_col="facility_name", + year_col="year", + visits_col="total_ed_visits", + visits_per_station_col="visits_per_station", + ) + + print("ER analysis sample:") + print(result[["facility_name", "year", "YoY_Visits", "Utilization", "Mismatch"]].head(15)) + + print("\nMismatch counts:") + print(result["Mismatch"].value_counts(dropna=False)) + + return result + + +def demo_county_facility_counts(df: pd.DataFrame) -> pd.DataFrame: + """Demo counting unique facilities by county.""" + print_section("15. County Facility Counts") + + result = county_facility_counts( + df, + county_col="county_name", + facility_col="facility_name", + ) + + print("Top 10 counties by number of unique facilities:") + print(result.head(10)) + + return result + + +def demo_spike_frequency_pivot(df: pd.DataFrame) -> pd.DataFrame: + """Demo detecting year-over-year burden spikes by category.""" + print_section("16. Spike Frequency Pivot") + + result = spike_frequency_pivot( + df, + threshold_pct=20.0, + facility_col="facility_name", + category_col="category", + year_col="year", + visits_col="visits_per_station", + ) + + print("Categories with the most year-over-year spikes:") + print(result.head(10)) + + return result + + +def demo_income_data() -> pd.DataFrame: + """Demo California income data utilities.""" + print_section("17. California Income Data Utilities") + + income_path = Path("data/california_median_income_by_zipcode.csv") + + if income_path.exists(): + income_df = load_california_income_data(income_path) + print(f"Loaded income data from: {income_path}") + else: + income_df = load_california_income_data() + print("Income CSV not found. Loaded built-in sample income data.") + + print("\nIncome data sample:") + print(income_df.head()) + + print("\nIncome statistics:") + print(get_income_statistics(income_df)) + + print("\nIncome statistics by county:") + print(get_income_statistics(income_df, group_by="county")) + + zip_code = str(income_df.iloc[0]["zip_code"]) + print(f"\nLookup for ZIP code {zip_code}:") + print(get_income_by_zip(income_df, zip_code)) + + county = str(income_df.iloc[0]["county"]) + print(f"\nFirst 5 ZIP codes in county: {county}") + print(get_income_by_county(income_df, county).head()) + + print("\nZIP codes with median income between $70,000 and $120,000:") + print(filter_by_income_range(income_df, 70000, 120000).head(10)) + + return income_df + + +def demo_visualizations(df: pd.DataFrame) -> None: + """Demo visualization functions and save outputs.""" + print_section("18. Visualization Demos") + + load_plot_path = OUTPUT_DIR / "hospital_load_distribution.png" + category_plot_path = OUTPUT_DIR / "category_visits_plot.png" + facility_category_plot_path = OUTPUT_DIR / "facility_category_visits_plot.png" + trend_plot_path = OUTPUT_DIR / "facility_trend.png" + + print("Generating hospital load distribution plot...") + fig = plot_hospital_load_distribution( + df, + visits_col="total_ed_visits", + stations_col="ed_stations", + save_path=str(load_plot_path), + ) + print(f"Saved: {load_plot_path}") + + print("\nGenerating category visits plot...") + fig = plot_category_visits(df, save=False) + fig.savefig(category_plot_path) + print(f"Saved: {category_plot_path}") + + first_facility = df["facility_name"].dropna().iloc[0] + + print(f"\nGenerating facility trend plot for: {first_facility}") + fig = plot_facility_trend( + df, + first_facility, + facility_col="facility_name", + year_col="year", + visits_col="total_ed_visits", + ) + fig.savefig(trend_plot_path) + print(f"Saved: {trend_plot_path}") + + print(f"\nGenerating category visits plot for facility: {first_facility}") + fig = plot_category_visits_by_facility(df, first_facility, save=False) + fig.savefig(facility_category_plot_path) + print(f"Saved: {facility_category_plot_path}") + + print("\nGenerating urban-rural map...") + map_obj = plot_urban_rural_map( + "california", + save=False, + latitude_col="latitude", + longitude_col="longitude", + designation_col="urban_rural_designation", + facility_col="facility_name", + ) + map_path = OUTPUT_DIR / "urban_rural_map.html" + map_obj.save(map_path) + print(f"Saved: {map_path}") + + +def main() -> None: + """Run all demos.""" + print_section("ERTIMES PACKAGE DEMO") + + df = demo_download_emergency_data() + + demo_load_emergency_data_from_saved_csv(df) + + demo_capacity_volume_mismatch(df) + + county_summary = demo_county_capacity_summary() + demo_rank_counties(county_summary) + demo_generate_county_report(county_summary) + + demo_rank_hospitals(df) + demo_summarize_by_ownership(df) + demo_find_duplicates(df) + demo_per_category_burden_report(df) + demo_capacity_pressure_score(df) + demo_mental_health_shortage_analysis(df) + demo_growth_analysis(df) + demo_run_er_analysis(df) + demo_county_facility_counts(df) + demo_spike_frequency_pivot(df) + demo_income_data() + demo_visualizations(df) + + print_section("DEMO COMPLETE") + print(f"Check the '{OUTPUT_DIR}/' folder for saved CSVs, plots, and maps.") + + if __name__ == "__main__": - demo_county_capacity_summary() - demo_rank_counties() - demo_rank_hospitals() - demo_summarize_by_ownership() \ No newline at end of file + main() \ No newline at end of file diff --git a/src/ertimes/democaligraphs1.py b/src/ertimes/democaligraphs1.py deleted file mode 100644 index 0b5f6cf..0000000 --- a/src/ertimes/democaligraphs1.py +++ /dev/null @@ -1,326 +0,0 @@ -import pandas as pd -import matplotlib.pyplot as plt -import numpy as np - - -def load_merged_data(filepath: str) -> pd.DataFrame: - """ - This function loads the merged hospital/demographic data dataset from a CSV file. - - Parameters - ---------- - filepath : str - This is the path to the merged CSV file. - - Returns - ------- - pd.DataFrame - This is a cleaned dataframe with standardized county names. - """ - df = pd.read_csv(filepath) - - # Standardize county names so grouping works even if capitalization/spaces differ. - df["county"] = ( - df["county"] - .astype(str) - .str.strip() - .str.lower() - .str.replace(" county", "", regex=False) - ) - - return df - - -def clean_numeric_column(df: pd.DataFrame, column: str) -> pd.Series: - """ - This function converts each column to numeric values, removing commas if needed. - - Parameters - ---------- - df : pd.DataFrame - Input dataframe. - column : str - Column name to clean. - - Returns - ------- - pd.Series - Numeric version of the column. - """ - return pd.to_numeric( - df[column].astype(str).str.replace(",", "", regex=False), - errors="coerce" - ) - - -def summarize_by_county(df: pd.DataFrame) -> pd.DataFrame: - """ - This function create sone row per county. - - Demographic variables are averaged or taken as first values because they are - county-level values repeated across hospital rows. Hospital variables are - averaged across hospitals/years within each county. - """ - - # Columns that should be numeric before aggregation. - numeric_cols = [ - "cfambelowpovperc", - "cpersbelowpovperc", - "aperclessHS", - "apercbachplus", - "bmedianfamincome", - "bmedianHHincome", - "dpercnonenglish", - "epop>65perc", - "fhispperc", - "fblackperc", - "fwhiteperc", - "fforbornperc", - "tot_ed_nmbvsts", - "visits_per_station", - "edstations", - ] - - # Make a copy so the original dataframe is not changed unexpectedly. - df = df.copy() - - # Clean all selected numeric columns. - for col in numeric_cols: - if col in df.columns: - df[col] = clean_numeric_column(df, col) - - # Group by county to create county-level data. - county_df = df.groupby("county", as_index=False).agg({ - # Demographic variables - "cfambelowpovperc": "first", - "cpersbelowpovperc": "first", - "aperclessHS": "first", - "apercbachplus": "first", - "bmedianfamincome": "first", - "bmedianHHincome": "first", - "dpercnonenglish": "first", - "epop>65perc": "first", - "fhispperc": "first", - "fblackperc": "first", - "fwhiteperc": "first", - "fforbornperc": "first", - - # Hospital shortage area categories - "mentalhealthshortagearea": "first", - "primarycareshortagearea": "first", - - # Hospital/ED variables - "tot_ed_nmbvsts": "mean", - "visits_per_station": "mean", - "edstations": "mean", - }) - - return county_df - - -def graph_boxplot_by_category( - county_df: pd.DataFrame, - numeric_col: str, - category_col: str, - x_label: str, - y_label: str, - title: str -) -> None: - """ - This makes a boxplot comparing a numeric demographic variable across a categorical - hospital shortage-area variable. - """ - - # Keep only needed columns and remove missing values. - plot_df = county_df[[numeric_col, category_col]].dropna().copy() - - # Converts the numeric column safely. - plot_df[numeric_col] = pd.to_numeric(plot_df[numeric_col], errors="coerce") - plot_df = plot_df.dropna() - - groups = [] - labels = [] - - # Creates one boxplot group for each shortage-area category. - for category in sorted(plot_df[category_col].unique()): - group = plot_df.loc[plot_df[category_col] == category, numeric_col] - if len(group) > 0: - groups.append(group) - labels.append(str(category)) - - plt.figure(figsize=(8, 6)) - plt.boxplot(groups, tick_labels=labels) - plt.xlabel(x_label) - plt.ylabel(y_label) - plt.title(title) - plt.tight_layout() - plt.show() - - -def graph_scatter( - county_df: pd.DataFrame, - x_col: str, - y_col: str, - x_label: str, - y_label: str, - title: str -) -> None: - """ - This makes a scatterplot comparing one demographic variable to one hospital variable. - - A simple trendline is added when there are at least two valid points. - """ - - # Keep only the columns needed for the graph. - plot_df = county_df[[x_col, y_col]].dropna().copy() - - # Convert both columns to numeric values. - plot_df[x_col] = pd.to_numeric(plot_df[x_col], errors="coerce") - plot_df[y_col] = pd.to_numeric(plot_df[y_col], errors="coerce") - plot_df = plot_df.dropna() - - plt.figure(figsize=(8, 6)) - plt.scatter(plot_df[x_col], plot_df[y_col]) - - # Add a linear trendline to make the overall relationship easier to see. - if len(plot_df) >= 2: - slope, intercept = np.polyfit(plot_df[x_col], plot_df[y_col], 1) - x_vals = np.linspace(plot_df[x_col].min(), plot_df[x_col].max(), 100) - y_vals = slope * x_vals + intercept - plt.plot(x_vals, y_vals) - - plt.xlabel(x_label) - plt.ylabel(y_label) - plt.title(title) - plt.tight_layout() - plt.show() - - -def graph_correlation_heatmap(county_df: pd.DataFrame) -> None: - """ - This makes a simple correlation heatmap for selected demographic and hospital variables. - """ - - selected_cols = [ - "cfambelowpovperc", - "cpersbelowpovperc", - "aperclessHS", - "apercbachplus", - "bmedianHHincome", - "dpercnonenglish", - "epop>65perc", - "fforbornperc", - "tot_ed_nmbvsts", - "visits_per_station", - "edstations", - ] - - # Keep only columns that are actually present. - selected_cols = [col for col in selected_cols if col in county_df.columns] - - # Create numeric-only correlation dataframe. - corr_df = county_df[selected_cols].apply(pd.to_numeric, errors="coerce").corr() - - plt.figure(figsize=(10, 8)) - plt.imshow(corr_df) - plt.colorbar(label="Correlation") - plt.xticks(range(len(corr_df.columns)), corr_df.columns, rotation=90) - plt.yticks(range(len(corr_df.columns)), corr_df.columns) - plt.title("Correlation Heatmap: Demographics vs. Hospital Variables") - plt.tight_layout() - plt.show() - - -if __name__ == "__main__": - # Load the merged dataset. - df = load_merged_data("merged_output.csv") - - # Summarize to one row per county so demographic data is not repeated by hospital row. - county_df = summarize_by_county(df) - - # Print a quick preview to confirm the county-level dataset was created. - print(county_df.head()) - print("\nCounty-level shape:", county_df.shape) - - # Graph: poverty vs mental health shortage area. - graph_boxplot_by_category( - county_df, - numeric_col="cfambelowpovperc", - category_col="mentalhealthshortagearea", - x_label="Mental Health Shortage Area", - y_label="Families Below Poverty (%)", - title="County Family Poverty vs. Mental Health Shortage Area" - ) - - # Graph: poverty vs visits per station. - graph_scatter( - county_df, - x_col="cfambelowpovperc", - y_col="visits_per_station", - x_label="Families Below Poverty (%)", - y_label="Average Visits per Station", - title="County Family Poverty vs. Average Visits per Station" - ) - - # Graph 1: household income vs visits per station. - graph_scatter( - county_df, - x_col="bmedianHHincome", - y_col="visits_per_station", - x_label="Median Household Income ($)", - y_label="Average Visits per Station", - title="Median Household Income vs. Average Visits per Station" - ) - - # Graph 2: percent non-English-speaking households vs ED visits. - graph_scatter( - county_df, - x_col="dpercnonenglish", - y_col="tot_ed_nmbvsts", - x_label="Non-English-Speaking Households (%)", - y_label="Average Total ED Visits", - title="Non-English-Speaking Households vs. Average ED Visits" - ) - - # Graph 3: percent with bachelor's degree or higher vs visits per station. - graph_scatter( - county_df, - x_col="apercbachplus", - y_col="visits_per_station", - x_label="Bachelor's Degree or Higher (%)", - y_label="Average Visits per Station", - title="Education Level vs. Average Visits per Station" - ) - - # Graph 4: percent age 65+ vs ED stations. - graph_scatter( - county_df, - x_col="epop>65perc", - y_col="edstations", - x_label="Population Age 65+ (%)", - y_label="Average Number of ED Stations", - title="Older Adult Population vs. ED Stations" - ) - - # Graph 5: percent foreign-born vs total ED visits. - graph_scatter( - county_df, - x_col="fforbornperc", - y_col="tot_ed_nmbvsts", - x_label="Foreign-Born Population (%)", - y_label="Average Total ED Visits", - title="Foreign-Born Population vs. Average ED Visits" - ) - - # Graph 6: personal poverty vs primary care shortage area. - graph_boxplot_by_category( - county_df, - numeric_col="cpersbelowpovperc", - category_col="primarycareshortagearea", - x_label="Primary Care Shortage Area", - y_label="People Below Poverty (%)", - title="County Poverty vs. Primary Care Shortage Area" - ) - - # Graph 7: correlation heatmap for selected demographic and hospital variables. - graph_correlation_heatmap(county_df) diff --git a/src/ertimes/demographics.py b/src/ertimes/demographics.py index 18f6f93..01f6301 100644 --- a/src/ertimes/demographics.py +++ b/src/ertimes/demographics.py @@ -3,6 +3,7 @@ from pathlib import Path import pandas as pd +from typing import Optional DATA_URLS = { @@ -358,4 +359,47 @@ def merge_datasets(hospital_filepath: str | Path) -> pd.DataFrame: Prefer merge_with_demographics in new code. """ - return merge_with_demographics(hospital_filepath) \ No newline at end of file + return merge_with_demographics(hospital_filepath) + +def filter_by_income_range( + df: pd.DataFrame, + min_income: float, + max_income: float, +) -> pd.DataFrame: + """ + Filter zip codes by median income range. + """ + return df[ + (df["median_income"] >= min_income) + & (df["median_income"] <= max_income) + ].copy() + +def get_income_statistics( + df: pd.DataFrame, + group_by: Optional[str] = None, +) -> pd.DataFrame: + """ + Calculate median income statistics overall or by county/city. + """ + if group_by: + stats = df.groupby(group_by)["median_income"].agg( + [ + ("mean_income", "mean"), + ("median_income", "median"), + ("min_income", "min"), + ("max_income", "max"), + ("count", "count"), + ] + ).round(2) + else: + stats = pd.DataFrame( + { + "mean_income": [df["median_income"].mean()], + "median_income": [df["median_income"].median()], + "min_income": [df["median_income"].min()], + "max_income": [df["median_income"].max()], + "count": [len(df)], + } + ).round(2) + + return stats \ No newline at end of file diff --git a/src/ertimes/demotest.py b/src/ertimes/demotest.py deleted file mode 100644 index 1bec19a..0000000 --- a/src/ertimes/demotest.py +++ /dev/null @@ -1,36 +0,0 @@ -import pandas as pd -from ertimes.demodata import download_data - -# Tests whether a dataset can be successfully downloaded and loaded, -# verifies it is a valid DataFrame, and prints basic structure and content checks -def test_data_reading(dataset: str) -> bool: - """"Tests the ability to read and clean the specified dataset, - printing out key information about the resulting DataFrame. - Returns True if successful, False otherwise.""" - try: - print(f"Attempting to read data for {dataset}...") - df = download_data(dataset) - - if isinstance(df, pd.DataFrame): - print("Success: Data is a valid Pandas DataFrame.") - print(f"Success: Found {len(df)} rows of data.") - print(f"Success: Found {len(df.columns)} columns.") - print("\nColumns:") - print(df.columns.tolist()) - print("\nFirst 5 rows:") - print(df.head()) - - if "county" in df.columns: - print("\nSuccess: 'county' column detected.") - else: - print("\nWarning: 'county' column not found.") - - return True - - except Exception as e: - print(f"Failed to read data: {e}") - return False - - -if __name__ == "__main__": - test_data_reading("calidemodata") diff --git a/src/ertimes/mental_health.py b/src/ertimes/mental_health.py deleted file mode 100644 index 343a393..0000000 --- a/src/ertimes/mental_health.py +++ /dev/null @@ -1,110 +0,0 @@ -import pandas as pd -import numpy as np -from clean import clean_data - - - -df = pd.read_csv("data/Emergency Department Volume and Capacity - Catalog - ED_COMBINE_AL.csv") - -df = clean_data(df) - -""" -This function highlights facilities that experience both an above-average burden (namely, a higher-than-normal demand relative to available resources) and a shortage of mental health resources, identifying them at high-risk. -""" - -def mental_health_shortage_analysis(df): - """Analyzes the emergency department data to identify facilities that are - at high risk due to a combination of high burden and mental health - resource shortages. - The function calculates a burden score for each facility, - determines the average burden, and flags facilities""" - - - # Creates a copy to prevent modifying original dataframe - df = df.copy() - - # Converts to numeric using the cleaned column names - df['tot_ed_nmb_vsts'] = pd.to_numeric(df['tot_ed_nmb_vsts'], errors='coerce') - df['ed_stations'] = pd.to_numeric(df['ed_stations'], errors='coerce').replace(0, 0.0001) - - # Calculates burden (# of visits/# of ED stations) + finds the burden average - df['burden_score'] = df['tot_ed_nmb_vsts'] / df['ed_stations'] - avg_burden = df['burden_score'].mean() - - # Identifies + filters high-risk areas, where high risk is defined as a facility being classified as a "mental health shortage area" and the burden score being above-average (namely, higher than the mean burden score for the data set) - df['high_risk'] = ( - (df['mental_health_shortage_area'] == 'Yes') & - (df['burden_score'] > avg_burden) - ) - - # Returns the facilities that are high risk - return df[df['high_risk'] == True] - -result = mental_health_shortage_analysis(df) -print(result) - -""" -Identifies high-risk facilities based on emergency department (ED) burden -and mental health resource shortages. - -A facility is classified as "high risk" if: -1. It is located in a mental health shortage area, AND -2. Its burden score (visits per station) exceeds the average burden - within its group (which in this instance is set to year). - -Parameters: - df (pd.DataFrame): Input dataset - visit_col (str): Column name for total ED visits - station_col (str): Column name for number of ED stations - shortage_col (str): Column indicating shortage status - shortage_value (str): Value indicating a shortage (e.g., "Yes") - group_col (str): Column used to compute group averages (e.g., year) - -Returns: - pd.DataFrame: Subset of the original DataFrame containing only - high-risk facilities, with additional columns: - - burden_score - - avg_burden - - high_risk (boolean) -""" - -def mental_health_shortage_analysis( - df, - visit_col="tot_ed_nmb_vsts", - station_col="ed_stations", - shortage_col="mental_health_shortage_area", - shortage_value="Yes", - group_col="year" -): - - # CALCULATION PREP (1): Convert key columns to numeric - # Ensures calculations work properly and prevents errors - df[visit_col] = pd.to_numeric(df[visit_col], errors="raise") - df[station_col] = pd.to_numeric(df[station_col], errors="raise") - - # CALCULATION PREP (2): Prevent division by zero - # Replace 0 stations with NaN - df[station_col] = df[station_col].replace(0, np.nan) - - # TYPE CHECKING: Ensure columns are numeric - if not np.issubdtype(df[visit_col].dtype, np.number): - raise TypeError(f"{visit_col} must be numeric") - if not np.issubdtype(df[station_col].dtype, np.number): - raise TypeError(f"{station_col} must be numeric") - - # BURDEN SCORE: ED visits per station - df["burden_score"] = df[visit_col] / df[station_col] - - # GROUP BENCHMARK: Average burden within group (e.g., year) - df["avg_burden"] = df.groupby(group_col)["burden_score"].transform("mean") - - # HIGH-RISK FLAG: - # 1. In mental health shortage area - # 2. Burden above group average - df["high_risk"] = ( - (df[shortage_col] == shortage_value) & - (df["burden_score"] > df["avg_burden"]) - ) - - # Returns the facilities that are high risk - return df[df['high_risk'] == True] \ No newline at end of file diff --git a/src/ertimes/stats_analysis.py b/src/ertimes/stats_analysis.py index d1b8a3c..d7f08f7 100644 --- a/src/ertimes/stats_analysis.py +++ b/src/ertimes/stats_analysis.py @@ -439,100 +439,6 @@ def spike_frequency_pivot( return pivot -# Census data functions -CENSUS_API_KEY = os.getenv("CENSUS_API_KEY", "YOUR_API_KEY_HERE") -ACS_YEAR = 2024 -TABLE_ID = "S1903" -BASE_URL = f"https://api.census.gov/data/{ACS_YEAR}/acs/acs5" - - -def get_census_data_for_zipcode() -> pd.DataFrame | None: - """ - Fetch California median household income data by zip code from the Census API. - """ - if CENSUS_API_KEY == "YOUR_API_KEY_HERE": - print("Error: Census API key not set!") - print("Get a free API key at: https://api.census.gov/data/key_signup.html") - print("Then set it with: export CENSUS_API_KEY='your_key_here'") - return None - - variables = "S1903_C03_001E,NAME" - params = { - "get": variables, - "for": "zip code tabulation area:*", - "in": "state:06", - "key": CENSUS_API_KEY, - } - - try: - response = requests.get(BASE_URL, params=params) - response.raise_for_status() - data = response.json() - - if len(data) < 2: - print("No data returned from Census Bureau API") - return None - - headers = data[0] - rows = data[1:] - df = pd.DataFrame(rows, columns=headers) - - df = df.rename( - columns={ - "S1903_C03_001E": "median_income", - "NAME": "name", - "zip code tabulation area": "zip_code", - } - ) - - df["median_income"] = pd.to_numeric(df["median_income"], errors="coerce") - df = df[df["median_income"].notna()].copy() - - df["county"] = df["name"].str.extract( - r",\\s*([A-Za-z\\s]+)\\s+County,\\s+California", - expand=False, - ) - df["city"] = df["name"].str.extract(r"^(.*?),", expand=False) - - df = df[["zip_code", "median_income", "county", "city"]].copy() - df["median_income"] = df["median_income"].astype(int) - - return df - - except requests.exceptions.RequestException as e: - print(f"Error fetching data from Census API: {e}") - return None - except Exception as e: - print(f"Error processing Census data: {e}") - return None - - -def save_to_csv( - df: pd.DataFrame, - output_path: str = "data/california_median_income_by_zipcode.csv", -) -> None: - """ - Save the census data to a CSV file. - """ - Path(output_path).parent.mkdir(parents=True, exist_ok=True) - df.to_csv(output_path, index=False) - - -def display_statistics(df: pd.DataFrame) -> None: - """ - Display basic statistics about the income data. - """ - print("\\n" + "=" * 70) - print("CALIFORNIA MEDIAN INCOME STATISTICS") - print("=" * 70) - print(f"Total zip codes: {len(df)}") - print(f"Mean median income: ${df['median_income'].mean():,.2f}") - print(f"Median income: ${df['median_income'].median():,.2f}") - print(f"Min income: ${df['median_income'].min():,}") - print(f"Max income: ${df['median_income'].max():,}") - print(f"Counties: {df['county'].nunique()}") - print("=" * 70 + "\\n") - def load_california_income_data(filepath: Optional[str] = None) -> pd.DataFrame: """ @@ -607,64 +513,3 @@ def get_income_by_county(df: pd.DataFrame, county: str) -> pd.DataFrame: Get median income data for all zip codes in a county. """ return df[df["county"].str.lower() == county.lower()].copy() - - -def get_income_statistics( - df: pd.DataFrame, - group_by: Optional[str] = None, -) -> pd.DataFrame: - """ - Calculate median income statistics overall or by county/city. - """ - if group_by: - stats = df.groupby(group_by)["median_income"].agg( - [ - ("mean_income", "mean"), - ("median_income", "median"), - ("min_income", "min"), - ("max_income", "max"), - ("count", "count"), - ] - ).round(2) - else: - stats = pd.DataFrame( - { - "mean_income": [df["median_income"].mean()], - "median_income": [df["median_income"].median()], - "min_income": [df["median_income"].min()], - "max_income": [df["median_income"].max()], - "count": [len(df)], - } - ).round(2) - - return stats - - -def filter_by_income_range( - df: pd.DataFrame, - min_income: float, - max_income: float, -) -> pd.DataFrame: - """ - Filter zip codes by median income range. - """ - return df[ - (df["median_income"] >= min_income) - & (df["median_income"] <= max_income) - ].copy() - - -def display_income_summary(df: pd.DataFrame) -> None: - """ - Print a formatted summary of income data. - """ - print("\\n" + "=" * 60) - print("CALIFORNIA MEDIAN INCOME SUMMARY") - print("=" * 60) - print(f"Total zip codes: {len(df)}") - print(f"Mean income: ${df['median_income'].mean():,.2f}") - print(f"Median income: ${df['median_income'].median():,.2f}") - print(f"Min income: ${df['median_income'].min():,}") - print(f"Max income: ${df['median_income'].max():,}") - print(f"Counties: {df['county'].nunique()}") - print("=" * 60 + "\\n") \ No newline at end of file diff --git a/src/ertimes/create_sample_census_data.py b/src/scripts/create_sample_census_data.py similarity index 100% rename from src/ertimes/create_sample_census_data.py rename to src/scripts/create_sample_census_data.py diff --git a/src/ertimes/median_income_demo.py b/src/scripts/median_income_demo.py similarity index 81% rename from src/ertimes/median_income_demo.py rename to src/scripts/median_income_demo.py index 646a0c4..fd64b8d 100644 --- a/src/ertimes/median_income_demo.py +++ b/src/scripts/median_income_demo.py @@ -7,17 +7,14 @@ import sys import importlib.util -# Load the module directly without going through __init__.py -spec = importlib.util.spec_from_file_location("Median_income", "/Users/maggie29755/ertimes/src/ertimes/Median_income.py") -module = importlib.util.module_from_spec(spec) -spec.loader.exec_module(module) - -load_california_income_data = module.load_california_income_data -get_income_by_zip = module.get_income_by_zip -get_income_by_county = module.get_income_by_county -get_income_statistics = module.get_income_statistics -filter_by_income_range = module.filter_by_income_range -display_income_summary = module.display_income_summary +from ertimes.stats_analysis import ( + load_california_income_data, + get_income_by_zip, + get_income_by_county, + get_income_statistics, + filter_by_income_range, + display_income_summary, +) def main(): diff --git a/tests/test_stats.py b/tests/test_stats.py index a5a690b..e34d602 100644 --- a/tests/test_stats.py +++ b/tests/test_stats.py @@ -11,8 +11,8 @@ # Use this clean import now that 'pip install -e .' worked! from ertimes.io import download_emergency_data -from ertimes.stats_analysis import _bed_size_to_numeric, find_capacity_volume_mismatch, compute_capacity_pressure_score, spike_frequency_pivot, mental_health_shortage_analysis, calculate_growth, county_facility_counts -from ertimes.stats_reports import generate_county_report, per_category_burden_report, summarize_by_ownership, find_duplicates +from ertimes.stats_analysis import find_capacity_volume_mismatch, compute_capacity_pressure_score, spike_frequency_pivot, mental_health_shortage_analysis +from ertimes.stats_reports import generate_county_report, summarize_by_ownership from ertimes.stats_ranking import rank_counties_by_burden, rank_hospitals_by_visits_per_station from ertimes.stats_visualization import plot_facility_trend, plot_urban_rural_map @@ -412,8 +412,8 @@ def test_generate_county_report_missing_county(): #Median_Income Tests -from ertimes.stats_analysis import load_california_income_data, get_income_by_zip, get_income_statistics - +from ertimes.stats_analysis import load_california_income_data, get_income_by_zip +from ertimes.demographics import get_income_statistics def test_load_california_income_data(): # Test loading data from a valid file @@ -986,4 +986,284 @@ def test_calculate_growth_percent_multiple_groups(): # Group 2 expectations assert np.isnan(result.loc[(result["oshpd_id"] == 2) & (result["year"] == 2020), "growth"].iloc[0]) - assert result.loc[(result["oshpd_id"] == 2) & (result["year"] == 2021), "growth"].iloc[0] == -50 \ No newline at end of file + assert result.loc[(result["oshpd_id"] == 2) & (result["year"] == 2021), "growth"].iloc[0] == -50 + +def test_rank_counties_by_burden_bad_input_type(): + """rank_counties_by_burden should raise TypeError for non-DataFrame input.""" + with pytest.raises(TypeError, match="summary must be a pandas DataFrame"): + rank_counties_by_burden([1, 2, 3]) + + +def test_rank_counties_by_burden_custom_visits_column(): + """rank_counties_by_burden should support a custom burden column name.""" + summary = pd.DataFrame({ + "county": ["A", "B", "C"], + "burden": [5, 15, 10], + }) + + result = rank_counties_by_burden(summary, visits_col="burden") + + assert list(result["county"]) == ["B", "C", "A"] + + +def test_rank_counties_by_burden_resets_index(): + """rank_counties_by_burden should reset the index after sorting.""" + summary = pd.DataFrame({ + "county": ["A", "B", "C"], + "visits_per_station": [5, 15, 10], + }, index=[10, 20, 30]) + + result = rank_counties_by_burden(summary) + + assert list(result.index) == [0, 1, 2] + + +def test_rank_hospitals_by_visits_per_station_bad_input_type(): + """rank_hospitals_by_visits_per_station should raise TypeError for non-DataFrame input.""" + with pytest.raises(TypeError, match="df must be a pandas DataFrame"): + rank_hospitals_by_visits_per_station([1, 2, 3]) + + +def test_rank_hospitals_by_visits_per_station_invalid_agg(): + """rank_hospitals_by_visits_per_station should reject unsupported aggregation methods.""" + df = pd.DataFrame({ + "facility_name": ["A", "B"], + "visits_per_station": [10, 20], + }) + + with pytest.raises(ValueError, match="agg must be 'mean' or 'median'"): + rank_hospitals_by_visits_per_station(df, agg="sum") + + +def test_rank_hospitals_by_visits_per_station_negative_top_n(): + """rank_hospitals_by_visits_per_station should reject negative top_n values.""" + df = pd.DataFrame({ + "facility_name": ["A", "B"], + "visits_per_station": [10, 20], + }) + + with pytest.raises(ValueError, match="top_n must be nonnegative"): + rank_hospitals_by_visits_per_station(df, top_n=-1) + + +def test_rank_hospitals_by_visits_per_station_numeric_coercion(): + """String numeric values should be coerced before ranking.""" + df = pd.DataFrame({ + "facility_name": ["A", "B", "C"], + "visits_per_station": ["10", "30", "20"], + }) + + result = rank_hospitals_by_visits_per_station(df) + + assert list(result["facility_name"]) == ["B", "C", "A"] + assert pd.api.types.is_numeric_dtype(result["visits_per_station"]) + + +def test_rank_hospitals_by_visits_per_station_custom_columns(): + """Function should support custom facility and visits column names.""" + df = pd.DataFrame({ + "Hospital": ["A", "A", "B"], + "Burden": [10, 30, 50], + }) + + result = rank_hospitals_by_visits_per_station( + df, + facility_col="Hospital", + visits_col="Burden", + agg="mean", + ) + + assert list(result["Hospital"]) == ["B", "A"] + assert result.loc[0, "Burden"] == 50 + + +def test_rank_hospitals_by_visits_per_station_top_n_zero(): + """top_n=0 should return an empty DataFrame with the expected columns.""" + df = pd.DataFrame({ + "facility_name": ["A", "B"], + "visits_per_station": [10, 20], + }) + + result = rank_hospitals_by_visits_per_station(df, top_n=0) + + assert result.empty + assert list(result.columns) == ["facility_name", "visits_per_station"] + +from ertimes.stats_visualization import ( + plot_hospital_load_distribution, + year_range, + create_ed_map, + plot_category_visits_by_facility, +) + + +def test_plot_hospital_load_distribution_returns_figure(): + """plot_hospital_load_distribution should return a matplotlib Figure.""" + df = pd.DataFrame({ + "Tot_ED_NmbVsts": [100, 200, 300], + "EDStations": [10, 20, 30], + }) + + fig = plot_hospital_load_distribution(df) + + assert fig is not None + assert hasattr(fig, "savefig") + + +def test_plot_hospital_load_distribution_missing_columns(): + """plot_hospital_load_distribution should raise ValueError when required columns are missing.""" + df = pd.DataFrame({ + "Tot_ED_NmbVsts": [100, 200], + }) + + with pytest.raises(ValueError, match="Missing required columns"): + plot_hospital_load_distribution(df) + + +def test_plot_hospital_load_distribution_bad_input_type(): + """plot_hospital_load_distribution should raise TypeError for non-DataFrame input.""" + with pytest.raises(TypeError, match="df must be a pandas DataFrame"): + plot_hospital_load_distribution([1, 2, 3]) + + +def test_plot_hospital_load_distribution_save_path(tmp_path): + """plot_hospital_load_distribution should save a figure when save_path is provided.""" + df = pd.DataFrame({ + "Tot_ED_NmbVsts": [100, 200, 300], + "EDStations": [10, 20, 30], + }) + output_path = tmp_path / "load_distribution.png" + + fig = plot_hospital_load_distribution(df, save_path=str(output_path)) + + assert fig is not None + assert output_path.exists() + + +def test_year_range_basic(tmp_path): + """year_range should return the minimum and maximum valid year from a CSV.""" + csv_path = tmp_path / "years.csv" + pd.DataFrame({ + "year": [2020, 2021, 2023], + "value": [1, 2, 3], + }).to_csv(csv_path, index=False) + + assert year_range(str(csv_path)) == (2020, 2023) + + +def test_year_range_missing_year_column(tmp_path): + """year_range should raise ValueError when the CSV has no year column.""" + csv_path = tmp_path / "no_year.csv" + pd.DataFrame({ + "value": [1, 2, 3], + }).to_csv(csv_path, index=False) + + with pytest.raises(ValueError, match="year"): + year_range(str(csv_path)) + + +def test_year_range_no_valid_year_values(tmp_path): + """year_range should raise ValueError when year column has no valid numeric values.""" + csv_path = tmp_path / "bad_years.csv" + pd.DataFrame({ + "year": ["bad", None, "unknown"], + }).to_csv(csv_path, index=False) + + with pytest.raises(ValueError, match="valid year"): + year_range(str(csv_path)) + + +def test_plot_facility_trend_bad_input_type(): + """plot_facility_trend should raise TypeError for non-DataFrame input.""" + with pytest.raises(TypeError, match="df must be a pandas DataFrame"): + plot_facility_trend([1, 2, 3], "A") + + +def test_plot_urban_rural_map_missing_coordinate_rows(monkeypatch): + """plot_urban_rural_map should raise ValueError when no valid coordinates remain.""" + fake_df = pd.DataFrame({ + "LATITUDE": [None, None], + "LONGITUDE": [None, None], + "UrbanRuralDesi": ["Urban", "Rural"], + "FacilityName2": ["A", "B"], + }) + + def fake_download(state): + return fake_df + + monkeypatch.setattr("ertimes.stats_analysis.download_emergency_data", fake_download) + + with pytest.raises(ValueError, match="No valid latitude/longitude"): + plot_urban_rural_map("California") + + +def test_create_ed_map_filters_year(): + """create_ed_map should return a Plotly figure for the requested year.""" + df = pd.DataFrame({ + "year": [2020, 2021], + "latitude": [34.1, 35.2], + "longitude": [-118.2, -119.3], + "total_ed_visits": [100, 200], + "primary_care_shortage": ["Yes", "No"], + "mental_health_shortage": ["No", "Yes"], + "county": ["A", "B"], + "ed_name": ["Hospital A", "Hospital B"], + }) + + fig = create_ed_map(df, 2021) + + assert fig is not None + assert hasattr(fig, "to_dict") + + +def test_create_ed_map_missing_columns(): + """create_ed_map should raise ValueError when required columns are missing.""" + df = pd.DataFrame({ + "year": [2021], + "latitude": [34.1], + }) + + with pytest.raises(ValueError, match="Missing required columns"): + create_ed_map(df, 2021) + + +def test_create_ed_map_bad_input_type(): + """create_ed_map should raise TypeError for non-DataFrame input.""" + with pytest.raises(TypeError, match="df must be a pandas DataFrame"): + create_ed_map([1, 2, 3], 2021) + + +def test_plot_category_visits_by_facility_returns_figure(monkeypatch, capsys): + """plot_category_visits_by_facility should return a Figure and print the category summary.""" + monkeypatch.setattr(plt, "show", lambda: None) + + df = pd.DataFrame({ + "facility_name": ["A", "A", "B"], + "category": ["Stroke", "All ED Visits", "Stroke"], + "ed_burden": [20, 999, 40], + }) + + fig = plot_category_visits_by_facility(df, "A") + captured = capsys.readouterr() + + assert fig is not None + assert hasattr(fig, "savefig") + assert "Stroke" in captured.out + assert "All ED Visits" not in captured.out + + +def test_plot_category_visits_by_facility_missing_columns(): + """plot_category_visits_by_facility should raise ValueError if required columns are missing.""" + df = pd.DataFrame({ + "facility_name": ["A"], + "category": ["Stroke"], + }) + + with pytest.raises(ValueError, match="Missing required columns"): + plot_category_visits_by_facility(df, "A") + + +def test_plot_category_visits_bad_input_type(): + """plot_category_visits should raise TypeError for non-DataFrame input.""" + with pytest.raises(TypeError, match="df must be a pandas DataFrame"): + plot_category_visits([1, 2, 3]) \ No newline at end of file