data <- read.csv("analytic_data2025_v3.csv", skip = 1)
dim(data)
## [1] 3204  796
texas <- subset(data, state == "TX" & county != "Texas")
health_data <- texas[, c("county",
                         "v002_rawvalue",
                         "v003_rawvalue")]
dim(health_data)
## [1] 254   3
names(health_data) <- c("county",
                        "poor_fair_health",
                        "uninsured_adults")
names(health_data)
## [1] "county"           "poor_fair_health" "uninsured_adults"
str(health_data)
## 'data.frame':    254 obs. of  3 variables:
##  $ county          : chr  "Anderson County" "Andrews County" "Angelina County" "Aransas County" ...
##  $ poor_fair_health: num  0.224 0.254 0.245 0.236 0.176 0.176 0.25 0.208 0.227 0.184 ...
##  $ uninsured_adults: num  0.217 0.273 0.239 0.235 0.188 ...
nrow(health_data)
## [1] 254
sum(is.na(health_data$poor_fair_health))
## [1] 0
sum(is.na(health_data$uninsured_adults))
## [1] 0
summary(health_data$poor_fair_health)
##    Min. 1st Qu.  Median    Mean 3rd Qu.    Max. 
##  0.1240  0.2050  0.2280  0.2355  0.2605  0.4650
summary(health_data$uninsured_adults)
##    Min. 1st Qu.  Median    Mean 3rd Qu.    Max. 
## 0.03226 0.20192 0.23041 0.23303 0.26161 0.40732
hist(health_data$poor_fair_health,
     main = "Poor or Fair Health in Texas Counties",
     xlab = "Proportion Reporting Poor or Fair Health")

hist(health_data$uninsured_adults,
     main = "Uninsured Adults in Texas Counties",
     xlab = "Proportion of Uninsured Adults")

plot(health_data$uninsured_adults,
     health_data$poor_fair_health,
     main = "Uninsured Adults and Poor or Fair Health",
     xlab = "Proportion of Uninsured Adults",
     ylab = "Proportion Reporting Poor or Fair Health")

cor(health_data$uninsured_adults,
    health_data$poor_fair_health)
## [1] 0.6570203