Submitted by Uday Bhanu Singh
Loading the hospital data, and creating a list of counties covered in the data
data <- st_read("https://raw.githubusercontent.com/ujhwang/urban-analytics-2026/main/Assignment/mini_3/hospital_11counties.geojson",
quiet = TRUE)
georgia <- counties(
state = "GA",
year = 2024,
progress_bar = FALSE,
class = 'sf'
)
georgia <- st_transform(georgia, st_crs(data))
atl_counties <- georgia |>
st_filter(data, .predicate = st_intersects)
county_list <- atl_counties$NAME
#DISCLOSURE: I used AI to create a list of all the variables I wanted to include (acs_vars) to avoid typing out each variable code manually
acs_vars <- c(
# 1. Income
median_hh_income = "B19013_001",
# 2. Race and Ethnicity
total_population = "B03002_001",
white_nh = "B03002_003",
black_nh = "B03002_004",
asian_nh = "B03002_006",
hispanic = "B03002_012",
# 3. Vehicle Availability
total_households = "B08201_001",
no_vehicle = "B08201_002",
# 4. Population Aged 65+
setNames(
sprintf("B01001_%03d", c(20:25, 44:49)),
paste0("age65_", 1:12)
),
# 5. Health Insurance
insurance_total = "B27001_001",
setNames(
sprintf("B27001_%03d", c(seq(5, 29, 3), seq(33, 57, 3))),
paste0("uninsured_", 1:18)
),
# 7. Disability Status
disability_total = "B18101_001",
setNames(
sprintf("B18101_%03d", c(4, 7, 10, 13, 16, 19,
23, 26, 29, 32, 35, 38)),
paste0("disabled_", 1:12)
),
# 8. Population Density
pop_total = "B01001_001"
)
acs_data <- get_acs(
geography = "tract",
variables = acs_vars,
state = "GA",
county = county_list,
year = 2024,
survey = "acs5",
output = "wide",
geometry = TRUE
)
acs_cleaned <- acs_data %>%
mutate(
# Population aged 65+
pop_65_plus = rowSums(pick(matches("^age65_\\d+E$"))), #regex to sum any groups over 65
pct_65_plus = 100 * pop_65_plus / total_populationE,
# Race and ethnicity
pct_white_nh = 100 * white_nhE / total_populationE,
pct_black_nh = 100 * black_nhE / total_populationE,
pct_asian_nh = 100 * asian_nhE / total_populationE,
pct_hispanic = 100 * hispanicE / total_populationE,
pct_minority = 100 * (total_populationE - white_nhE) / total_populationE,
# Vehicle availability
pct_no_vehicle = 100 * no_vehicleE / total_householdsE,
# Health insurance
uninsured_total = rowSums(pick(matches("^uninsured_\\d+E$"))), #regex to sum all uninsured groups
pct_uninsured = 100 * uninsured_total / insurance_totalE,
pct_insured = 100 - pct_uninsured,
# Disability
disabled_total = rowSums(pick(matches("^disabled_\\d+E$"))), #regex to sum all disabled groups
pct_disabled = 100 * disabled_total / disability_totalE,
#Population Density
pop_density = total_populationE / as.numeric(set_units(st_area(geometry), km^2))
) %>%
select(
GEOID, NAME,
median_hh_income = median_hh_incomeE,
total_population = total_populationE,
pop_density = pop_density,
total_households = total_householdsE,
pop_65_plus, pct_65_plus,
pct_white_nh, pct_black_nh, pct_asian_nh,
pct_hispanic, pct_minority,
pct_no_vehicle,
uninsured_total, pct_uninsured, pct_insured,
disabled_total, pct_disabled,
geometry
)
acs_cleaned <- st_transform(acs_cleaned, 26916)
data <- st_transform(data, 26916)
pois_joined <- st_join(
acs_cleaned,
data,
join = st_intersects,
left = TRUE
)
acs_cleaned <- acs_cleaned |>
mutate(
hospital_count = lengths(st_intersects(acs_cleaned, data))
)
#Exploratory map
baseMap <- tm_shape(acs_cleaned) + tm_polygons(fill_alpha = 0.7,
col = "gray50", lwd = 0.3
)+
tm_shape(atl_counties) + tm_borders(lwd=0.85, fill_alpha = 0, col='steelblue')+
tm_shape(data) + tm_symbols(size=0.75, lwd = 0.75,
fill="yellow", col="black") + tm_add_legend(type = "symbols",
labels = "Hospital",
fill="yellow",
col="black")
baseMap
hospital_buffers <- st_buffer(data, dist = 5000)
acs_cleaned <- acs_cleaned |>
mutate(
within_buffer = lengths(st_intersects(st_centroid(acs_cleaned), hospital_buffers)) > 0
)
acs_cleaned <- acs_cleaned %>%
mutate(
hospitals_within_5km = lengths(
st_intersects(st_centroid(acs_cleaned), hospital_buffers)
)
)
tm_shape(acs_cleaned) + tm_polygons(fill="within_buffer",
fill_alpha = 0.7,
col = "gray50", lwd = 0.3
)+
tm_shape(atl_counties) + tm_borders(lwd=0.85, fill_alpha = 0, col='steelblue')+
tm_shape(data) + tm_symbols(size=0.75, lwd = 0.75,
fill="yellow", col="black") + tm_add_legend(type = "symbols",
labels = "Hospital",
fill="yellow",
col="black")+
tm_shape(hospital_buffers) + tm_polygons(fill = "lightblue",
fill_alpha = 0.2,
col = "blue",
lwd = 1) + tm_add_legend(labels = "5km Buffer", fill_alpha = 0.2,
col = "blue",
lwd = 1)
acs_cleaned <- acs_cleaned |>
mutate(
nearest_hospital_km =
as.numeric(apply((st_distance(st_centroid(acs_cleaned), data)/1000), 1, min)
))
tm_shape(acs_cleaned) + tm_polygons(fill="nearest_hospital_km",
fill.scale = tm_scale_continuous(values = 'reds'),
fill_alpha = 0.7,
col = "gray50", lwd = 0.3
)+
tm_shape(atl_counties) + tm_borders(lwd=0.85, fill_alpha = 0, col='steelblue')+
tm_shape(data) + tm_symbols(size=0.75, lwd = 0.75,
fill="yellow", col="black") + tm_add_legend(type = "symbols",
labels = "Hospital",
fill="yellow",
col="black")
acs_cleaned |>
st_drop_geometry() |>
pivot_longer(
cols = c(pct_white_nh, pct_black_nh, pct_asian_nh, pct_hispanic),
names_to = "Race",
values_to = "Percentage"
) |>
mutate(
Race = factor(
Race,
levels = c("pct_white_nh", "pct_black_nh",
"pct_asian_nh", "pct_hispanic"),
labels = c("Non-Hispanic White", "Non-Hispanic Black",
"Non-Hispanic Asian", "Hispanic or Latino")
),
within_buffer = factor(
within_buffer,
levels = c(FALSE, TRUE),
labels = c("No", "Yes")
)
) |>
ggplot(aes(x = within_buffer, y = Percentage, fill = within_buffer)) +
geom_boxplot(
alpha = 0.75,
outlier.alpha = 0.25,
outlier.size = 0.8,
width = 0.6
) +
facet_wrap(~ Race, ncol = 2) +
scale_fill_manual(values = c("No" = "#E58A85", "Yes" = "#55B8BB")) +
scale_y_continuous(breaks = seq(0, 100, 20)) +
labs(
title = "Racial and Ethnic Composition by Hospital Proximity",
subtitle = "Census tracts in Metro Atlanta's 11-county study area",
x = "Hospital within 5 km of tract centroid",
y = "Share of tract population (%)"
) +
theme_minimal(base_size = 12) +
theme(
legend.position = "none",
strip.text = element_text(face = "bold"),
plot.title = element_text(face = "bold"),
panel.grid.minor = element_blank()
)
Census tracts without a hospital within 5 km generally have higher proportions of non-Hispanic Black residents (median ~38% versus ~26%). Tracts with nearby hospitals tend to have higher proportions of Asian and Hispanic/Latino residents, while non-Hispanic White shares are relatively similar. These patterns suggest potential racial disparities in geographic hospital access, although substantial overlap between groups indicates considerable variation.
ggplot(acs_cleaned, aes(x = median_hh_income, y = nearest_hospital_km)) +
geom_point(alpha = 0.3, size = 1.5) +
geom_smooth(method = "lm", se = TRUE, color = "red") +
scale_x_continuous(labels = scales::label_dollar()) +
labs(
title = "Household Income and Hospital Proximity",
x = "Median Household Income",
y = "Distance to Nearest Hospital (km)"
) +
theme_minimal()
The scatterplot shows a very weak positive relationship between median household income and distance to the nearest hospital. Household income alone is not a strong predictor of geographic hospital proximity.
plot_data <- acs_cleaned |>
st_drop_geometry() |>
pivot_longer(
cols = c(
pct_no_vehicle,
pct_uninsured,
pct_disabled,
pct_65_plus
),
names_to = "Variable",
values_to = "Percentage"
)
plot_data |>
ggplot(aes(x = Percentage, y = nearest_hospital_km)) +
geom_point(alpha = 0.25, size = 1) +
geom_smooth(method = "lm", se = TRUE, color = "red") +
facet_wrap(~ Variable, scales = "free_x", ncol = 2) +
labs(
title = "Socioeconomic Characteristics and Hospital Proximity",
x = "Population or Household Share (%)",
y = "Distance to Nearest Hospital (km)"
) +
theme_minimal()
The scatterplots show weak positive associations between hospital
distance and the proportions of elderly and disabled residents,
suggesting potentially poorer geographic access for populations with
greater healthcare needs. Tracts with higher shares of households
without vehicles tend to be closer to hospitals, possibly reflecting
greater urban density. The uninsured population share shows a slight
negative association with hospital distance. However, substantial
variation across all four indicators suggests no one indicator alone
explains variation.
model_black <- lm(
nearest_hospital_km ~
median_hh_income +
pct_black_nh +
pct_no_vehicle +
pct_uninsured +
pct_disabled +
pct_65_plus +
pop_density,
data = st_drop_geometry(acs_cleaned)
)
summary(model_black)
##
## Call:
## lm(formula = nearest_hospital_km ~ median_hh_income + pct_black_nh +
## pct_no_vehicle + pct_uninsured + pct_disabled + pct_65_plus +
## pop_density, data = st_drop_geometry(acs_cleaned))
##
## Residuals:
## Min 1Q Median 3Q Max
## -4.8538 -1.9094 -0.5693 1.1234 16.3868
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 5.132e+00 5.733e-01 8.952 < 2e-16 ***
## median_hh_income -2.352e-07 2.832e-06 -0.083 0.93383
## pct_black_nh 1.034e-02 3.662e-03 2.823 0.00484 **
## pct_no_vehicle -6.045e-02 1.251e-02 -4.833 1.51e-06 ***
## pct_uninsured -1.646e-02 1.141e-02 -1.443 0.14924
## pct_disabled 3.151e-02 2.070e-02 1.522 0.12817
## pct_65_plus -1.217e-02 1.408e-02 -0.864 0.38765
## pop_density -6.809e-04 6.373e-05 -10.684 < 2e-16 ***
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.835 on 1247 degrees of freedom
## (15 observations deleted due to missingness)
## Multiple R-squared: 0.153, Adjusted R-squared: 0.1482
## F-statistic: 32.17 on 7 and 1247 DF, p-value: < 2.2e-16
The regression model is statistically significant (p < 0.001) but explains only 15.3% of the variation in hospital distance. Higher population density and greater proportions of households without vehicles are significantly associated with shorter distances to hospitals. Higher Black population shares are associated with greater distances, even after controlling for other socioeconomic factors. This suggests potential racial disparities in hospital proximity, although much of the spatial variation remains unexplained.
model_data <- acs_cleaned |>
st_drop_geometry() |>
select(
GEOID,
nearest_hospital_km,
pop_density,
median_hh_income,
pct_no_vehicle,
pct_uninsured,
pct_disabled,
pct_65_plus,
pct_white_nh,
pct_black_nh,
pct_hispanic,
pct_asian_nh
) %>%
drop_na() |>
mutate(
income_10k = median_hh_income / 10000,
density_1000 = pop_density / 1000
)
# Model A: Geographic baseline
model_a <- lm(
nearest_hospital_km ~ density_1000,
data = model_data
)
# Model B: Socioeconomic characteristics and healthcare needs
model_b <- lm(
nearest_hospital_km ~ density_1000 +
income_10k + pct_no_vehicle +
pct_uninsured + pct_disabled + pct_65_plus,
data = model_data
)
# Model C: Racial/ethnic variables + baseline
model_c <- lm(
nearest_hospital_km ~ density_1000 +
pct_white_nh + pct_black_nh +
pct_hispanic + pct_asian_nh,
data = model_data
)
# Model D: Racial/ethnic variables + Socioeconomic
model_d <- lm(
nearest_hospital_km ~ income_10k + pct_no_vehicle +
pct_uninsured + pct_disabled + pct_65_plus +
pct_white_nh + pct_black_nh +
pct_hispanic + pct_asian_nh,
data = model_data
)
#Model E: All combined
model_e <- lm(
nearest_hospital_km ~ density_1000 + income_10k + pct_no_vehicle +
pct_uninsured + pct_disabled + pct_65_plus +
pct_white_nh + pct_black_nh +
pct_hispanic + pct_asian_nh,
data = model_data
)
models <- list(
A = model_a,
B = model_b,
C = model_c,
D = model_d,
E = model_e
)
model_comparison <- tibble(
Model = names(models),
R_squared = sapply(models, \(m) summary(m)$r.squared),
Adjusted_R2 = sapply(models, \(m) summary(m)$adj.r.squared),
AIC = sapply(models, AIC),
RMSE = sapply(models, \(m) sqrt(mean(residuals(m)^2)))
)
model_comparison
## # A tibble: 5 Ă— 5
## Model R_squared Adjusted_R2 AIC RMSE
## <chr> <dbl> <dbl> <dbl> <dbl>
## 1 A 0.127 0.126 6214. 2.87
## 2 B 0.148 0.143 6193. 2.84
## 3 C 0.172 0.168 6155. 2.79
## 4 D 0.120 0.114 6239. 2.88
## 5 E 0.193 0.187 6132. 2.76
I constructed five linear regression models to examine how geographic, socioeconomic, and racial/ethnic characteristics relate to hospital proximity. Model A establishes a population density baseline, Models B and C introduce socioeconomic and racial/ethnic variables respectively, Model D combines these characteristics without density, and Model E includes all predictors. Models were compared using R², adjusted R², AIC, and RMSE.
Model E performs best, achieving the highest explanatory power and lowest AIC (6132.3) and RMSE (2.76 km). Model C outperforms Model B, suggesting that racial/ethnic composition provides greater explanatory value than the selected socioeconomic variables when combined with population density. However, even the strongest model explains less than 20% of variation in hospital distance, indicating that important geographic and structural factors remain unaccounted for.
summary(model_e)
##
## Call:
## lm(formula = nearest_hospital_km ~ density_1000 + income_10k +
## pct_no_vehicle + pct_uninsured + pct_disabled + pct_65_plus +
## pct_white_nh + pct_black_nh + pct_hispanic + pct_asian_nh,
## data = model_data)
##
## Residuals:
## Min 1Q Median 3Q Max
## -5.3463 -1.9211 -0.4837 1.2747 15.4965
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 3.566786 1.964951 1.815 0.0697 .
## density_1000 -0.662049 0.062325 -10.623 < 2e-16 ***
## income_10k -0.031716 0.028681 -1.106 0.2690
## pct_no_vehicle -0.059162 0.012258 -4.826 1.56e-06 ***
## pct_uninsured -0.001494 0.014380 -0.104 0.9173
## pct_disabled 0.016126 0.020363 0.792 0.4286
## pct_65_plus -0.029317 0.013984 -2.096 0.0362 *
## pct_white_nh 0.036734 0.020994 1.750 0.0804 .
## pct_black_nh 0.031216 0.020495 1.523 0.1280
## pct_hispanic 0.010148 0.021258 0.477 0.6332
## pct_asian_nh -0.031993 0.022073 -1.449 0.1475
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 2.771 on 1244 degrees of freedom
## Multiple R-squared: 0.1932, Adjusted R-squared: 0.1867
## F-statistic: 29.78 on 10 and 1244 DF, p-value: < 2.2e-16
Model E explains about 19.3% of variation in hospital distance. Population density is the strongest predictor, with an additional 1,000 residents per unit area associated with a 0.66 km reduction in distance. Higher proportions of households without vehicles and elderly residents are also significantly associated with shorter distances. None of the racial/ethnic variables remain statistically significant at the 5% level after controlling for other factors, suggesting that previously observed racial disparities may partly reflect underlying geographic and socioeconomic patterns. Overall, the model highlights the importance of population density while leaving substantial variation unexplained.
model_data <- model_data |>
mutate(residual_e = residuals(model_e))
acs_cleaned <- acs_cleaned |>
left_join(
model_data |>
select(GEOID, residual_e),
by = "GEOID"
)
tm_shape(acs_cleaned) +
tm_polygons(
fill = "residual_e",
fill.scale = tm_scale_continuous(
values = "-RdBu",
midpoint = 0
),
fill.legend = tm_legend("Regression Residual (km)"),
col = "gray50",
lwd = 0.3
) +
tm_shape(atl_counties) +
tm_borders(lwd = 0.85, col = "steelblue") +
tm_layout(
title = "Spatial Distribution of Regression Residuals"
)
I mapped the residuals from Model E to see where it performs best and where deviations are largest. The residual map reveals clear spatial clustering, suggesting that Model E does not fully capture geographic variation in hospital proximity. Positive residuals (red) are concentrated along the metropolitan periphery, indicating hospitals are farther away than predicted, while negative residuals (blue) are more common in central areas, where hospitals are closer than predicted. This suggests that additional spatial factors influence hospital accessibility. Some extreme residuals near the study area’s boundaries may also reflect a boundary effect, as hospitals outside the 11-county study area were excluded from distance calculations, potentially overestimating distances for peripheral tracts.
acs_cleaned <- acs_cleaned |>
mutate(
high_transport_disadvantage = pct_no_vehicle >
median(pct_no_vehicle, na.rm = TRUE),
low_income = median_hh_income <
median(median_hh_income, na.rm = TRUE),
high_uninsured = pct_uninsured >
median(pct_uninsured, na.rm = TRUE),
high_disability = pct_disabled >
median(pct_disabled, na.rm = TRUE),
high_elderly = pct_65_plus >
median(pct_65_plus, na.rm = TRUE),
high_minority = pct_minority >
median(pct_minority, na.rm = TRUE),
disadvantage_count = rowSums(pick(
high_transport_disadvantage,
low_income,
high_uninsured,
high_disability,
high_elderly
)),
multiple_disadvantage = disadvantage_count >= 2,
limited_hospital_proximity = nearest_hospital_km > 5,
multiple_disadvantage_minority =
multiple_disadvantage & high_minority,
underserved = multiple_disadvantage & limited_hospital_proximity
)
tm_shape(acs_cleaned) +
tm_polygons(
fill = "disadvantage_count",
fill.scale = tm_scale_categorical(
levels = 0:5,
values = c("#FFF7EC", "#FEE8C8", "#FDBB84",
"#FC8D59", "#E34A33", "#B30000")
),
fill.legend = tm_legend("Vulnerability Indicators"),
fill_alpha = 0.85,
col = "gray65",
lwd = 0.15
) +
tm_shape(atl_counties) +
tm_borders(lwd = 1.2, col = "steelblue") +
tm_shape(data) +
tm_symbols(
size = 0.75,
fill = "yellow",
col = "black",
lwd = 0.5
) +
tm_add_legend(
type = "symbols",
labels = "Hospital",
fill = "yellow",
col = "black"
) +
tm_layout(
title = "Socioeconomic Vulnerability and Hospital Locations",
legend.outside = TRUE
)
I created a variable to compile low household income, high vehicle unavailability, uninsured population share, disability prevalence, and elderly population share into one vulnerability index. Each tract received a score from 0 to 5 based on the number of indicators exceeding their respective median-based vulnerability thresholds. The map reveals concentrations of higher vulnerability in southern Metro Atlanta, particularly south and southwest of the urban core, while hospitals are more clustered in central and northern areas. Although some vulnerable neighborhoods have nearby hospitals, the spatial mismatch in several areas suggests potential inequities in geographic healthcare access.
acs_cleaned |>
st_drop_geometry() |>
select(
hospitals_within_5km,
high_transport_disadvantage,
low_income,
high_uninsured,
high_disability,
high_elderly,
high_minority
) |>
pivot_longer(
cols = -hospitals_within_5km,
names_to = "disadvantage_type",
values_to = "disadvantaged"
) |>
filter(!is.na(disadvantaged), !is.na(hospitals_within_5km)) |>
mutate(
disadvantage_type = case_match(
disadvantage_type,
"high_transport_disadvantage" ~ "No Vehicle Access",
"low_income" ~ "Low Income",
"high_uninsured" ~ "High Uninsured",
"high_disability" ~ "High Disability",
"high_elderly" ~ "High Elderly Population",
"high_minority" ~ "High Minority Population"
),
disadvantaged = factor(
disadvantaged,
levels = c(FALSE, TRUE),
labels = c("No", "Yes")
)
) |>
ggplot(aes(x = disadvantaged, y = hospitals_within_5km,
fill = disadvantaged)) +
geom_boxplot(
alpha = 0.6,
outlier.shape = NA,
width = 0.6
) +
facet_wrap(~ disadvantage_type, ncol = 3) +
scale_y_continuous(limits = c(0, 6), breaks = seq(0, 6, 1)) +
scale_fill_manual(values = c("No" = "steelblue", "Yes" = "salmon")) +
labs(
title = "Hospital Choice by Neighborhood Characteristics",
subtitle = "Comparison of census tracts above and below median vulnerability thresholds",
x = "Meets Indicator Threshold",
y = "Hospitals Within 5 km"
) +
theme_minimal() +
theme(
legend.position = "none",
strip.text = element_text(face = "bold"),
panel.grid.minor = element_blank()
)
The boxplots compare hospital choice, measured by the number of hospitals within 5 km, across six neighborhood characteristics. Tracts with higher elderly and disabled population shares have fewer nearby hospitals on average in terms of median counts (1 versus 2), suggesting potential accessibility disadvantages for populations with greater healthcare needs. Tracts with higher vehicle unavailability have more nearby hospitals (median 2 versus 1), potentially reflecting urban density patterns. Income, insurance status, and minority population share show little difference in median hospital availability, although variation within groups remains substantial. Overall, hospital choice appears uneven across certain vulnerable populations.
The analysis suggests that hospital distribution in Metro Atlanta is not entirely equitable. Hospitals are concentrated in more densely populated central areas, while peripheral communities often experience greater distances. Tracts with higher proportions of elderly and disabled residents also tend to have fewer nearby hospital options, suggesting potential mismatches between healthcare needs and availability.
Racial disparities are evident in the exploratory analysis, particularly for predominantly Black communities, although these associations become statistically insignificant after controlling for geographic and socioeconomic factors. Population density emerges as the strongest predictor of hospital proximity, suggesting that urban development patterns substantially influence accessibility. This does not imply inequity, as car-dependancy grows radially outward from the Downtown, so people in outskirts are likely to use cars anyway.
The relatively low explanatory power of the regression models indicates that important factors remain unaccounted for. Straight-line distances and the exclusion of hospitals outside the study boundary limit the analysis significantly. The study also doesn’t cover quality of care or diversity of services.
Overall, the findings suggest that hospital locations reflect population concentration more strongly than equitable access across communities with differing healthcare needs.