#install packages
library(haven)
## Warning: package 'haven' was built under R version 4.6.1
library(tidyverse)
## Warning: package 'tidyverse' was built under R version 4.6.1
## Warning: package 'dplyr' was built under R version 4.6.1
## Warning: package 'lubridate' was built under R version 4.6.1
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.2.0
## ✔ forcats 1.0.1 ✔ stringr 1.6.0
## ✔ ggplot2 4.0.3 ✔ tibble 3.3.1
## ✔ lubridate 1.9.5 ✔ tidyr 1.3.2
## ✔ purrr 1.2.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
summary(cars)
## speed dist
## Min. : 4.0 Min. : 2.00
## 1st Qu.:12.0 1st Qu.: 26.00
## Median :15.0 Median : 36.00
## Mean :15.4 Mean : 42.98
## 3rd Qu.:19.0 3rd Qu.: 56.00
## Max. :25.0 Max. :120.00
cdc_wonder <- read_csv("C:/Users/mn Technology Group/Downloads/week7a_material/week7a_materials/cdc_wonder_state_deaths_2018_2023.csv")
## Rows: 306 Columns: 8
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (1): state
## dbl (7): state_code, year, year_code, deaths, population, crude_rate, age_ad...
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
cdc_wonder$state <- as.factor(cdc_wonder$state)
getwd()
## [1] "C:/Users/mn Technology Group/Downloads/week7a_material"
save(cdc_wonder, file = "C:/Users/mn Technology Group/Downloads/week7a_material/week7a_materials/cdc_wonder.Rdata")
census <- read_sas("C:/Users/mn Technology Group/Downloads/week7a_material/week7a_materials/census.sas7bdat")
getwd()
## [1] "C:/Users/mn Technology Group/Downloads/week7a_material"
setwd("C:/Users/mn Technology Group/Downloads/week7a_material/week7a_materials")
write.csv(census, file = "census.csv", row.names = FALSE)
summary(census)
## community_area per_capita_income hardship_index
## Min. : 1 Min. : 8201 Min. : 1.00
## 1st Qu.:20 1st Qu.:15805 1st Qu.:25.00
## Median :39 Median :21669 Median :50.00
## Mean :39 Mean :25597 Mean :49.51
## 3rd Qu.:58 3rd Qu.:28716 3rd Qu.:74.00
## Max. :77 Max. :88669 Max. :98.00
## NAs :1 NAs :1
## percent_aged_under_18_or_over_64 percent_of_housing_crowded
## Min. :13.50 Min. : 0.300
## 1st Qu.:32.15 1st Qu.: 2.325
## Median :38.05 Median : 3.850
## Mean :35.72 Mean : 4.921
## 3rd Qu.:40.50 3rd Qu.: 6.800
## Max. :51.50 Max. :15.800
##
## percent_aged_25_without_high_sch percent_aged_16_unemployed
## Min. : 2.50 Min. : 4.70
## 1st Qu.:12.07 1st Qu.: 9.20
## Median :18.65 Median :13.85
## Mean :20.33 Mean :15.34
## 3rd Qu.:26.60 3rd Qu.:20.00
## Max. :54.80 Max. :35.90
##
## percent_households_below_poverty
## Min. : 3.30
## 1st Qu.:13.35
## Median :19.05
## Mean :21.74
## 3rd Qu.:29.15
## Max. :56.50
##
head(census)
## # A tibble: 6 × 8
## community_area per_capita_income hardship_index percent_aged_under_18_or_ove…¹
## <dbl> <dbl> <dbl> <dbl>
## 1 NA 28202 NA 33.5
## 2 1 23939 39 27.5
## 3 10 32875 21 39.5
## 4 11 27751 25 35.5
## 5 12 44164 11 40.5
## 6 13 26576 33 39
## # ℹ abbreviated name: ¹percent_aged_under_18_or_over_64
## # ℹ 4 more variables: percent_of_housing_crowded <dbl>,
## # percent_aged_25_without_high_sch <dbl>, percent_aged_16_unemployed <dbl>,
## # percent_households_below_poverty <dbl>