Rows: 59919 Columns: 7
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (2): serialno, stcode
dbl (5): hwt, n_persons, hhtype, hhinc, n_kids
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
Rows: 216048 Columns: 7
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (1): serialno
dbl (6): stcode, hwt, n_persons, hhtype, hhinc, n_kids
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
Rows: 119650 Columns: 10
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (2): SERIALNO, STCODE
dbl (8): SPORDER, PWT, AGEP, SEX, PPINC, IS_CHILD, POVPIP, RACEETH3
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
Rows: 447327 Columns: 10
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (1): SERIALNO
dbl (9): SPORDER, STCODE, PWT, AGEP, SEX, PPINC, IS_CHILD, POVPIP, RACEETH3
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
Rows: 61368 Columns: 7
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (1): serialno
dbl (6): stcode, hwt, n_persons, hhtype, hhinc, n_kids
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
Rows: 220070 Columns: 7
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (1): serialno
dbl (6): stcode, hwt, n_persons, hhtype, hhinc, n_kids
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
Rows: 123656 Columns: 10
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (2): serialno, stcode
dbl (8): sporder, pwt, agep, sex, ppinc, is_child, povpip, raceeth3
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
Rows: 454249 Columns: 10
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (1): serialno
dbl (9): sporder, stcode, pwt, agep, sex, ppinc, is_child, povpip, raceeth3
ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
2.
#combine household datasets into onehouseholdfile2019<-rbind(ptone2019hh, pttwo2019hh)householdfile2021<-rbind(ptone2021hh, pttwo2021hh)householdfileall<-rbind(householdfile2019,householdfile2021)#show layouthead(householdfileall)
# A tibble: 6 × 7
serialno stcode hwt n_persons hhtype hhinc n_kids
<chr> <chr> <dbl> <dbl> <dbl> <dbl> <dbl>
1 2019GQ0000032 09 0 1 NA NA NA
2 2019GQ0000222 09 0 1 NA NA NA
3 2019GQ0000384 09 0 1 NA NA NA
4 2019GQ0000494 09 0 1 NA NA NA
5 2019GQ0000680 09 0 1 NA NA NA
6 2019GQ0000699 09 0 1 NA NA NA
#make a year identifier using the year listed in the serial codelibrary(dplyr)
Attaching package: 'dplyr'
The following objects are masked from 'package:stats':
filter, lag
The following objects are masked from 'package:base':
intersect, setdiff, setequal, union
#convert to numerichouseholdfileall$year<-as.numeric(householdfileall$year)
#combine person datasets into onepersonfile2019<-rbind(ptone2019p, pttwo2019p)#make all the variable names for the 2019 person file lowercase to make appending possiblepersonfile2019 <- personfile2019 %>%rename_with(tolower)personfile2021<-rbind(ptone2021p, pttwo2021p)personfileall<-rbind(personfile2019,personfile2021)#show layouthead(personfileall)
# A tibble: 6 × 10
serialno sporder stcode pwt agep sex ppinc is_child povpip raceeth3
<chr> <dbl> <chr> <dbl> <dbl> <dbl> <dbl> <dbl> <dbl> <dbl>
1 2019GQ0000032 1 09 9 20 1 0 NA NA 1
2 2019GQ0000222 1 09 53 20 1 1200 NA NA 1
3 2019GQ0000384 1 09 44 21 2 2000 NA NA 1
4 2019GQ0000494 1 09 11 87 2 0 NA NA 1
5 2019GQ0000680 1 09 11 95 2 0 NA NA 1
6 2019GQ0000699 1 09 79 21 2 1000 NA NA 1
#make a year identifier using the year listed in the serial codelibrary(dplyr)personfileall$year<-substr(personfileall$serialno, 1, 4)class(householdfileall$year)
[1] "numeric"
#convert to numericpersonfileall$year<-as.numeric(personfileall$year)
3.
#remove any observations that are not familiesclass(householdfileall$hhtype)
#standardize income: $1.00 in 2019 is equivalent to $1.06 in 2021#perform conversion for 2021; use case_whenhouseholdfileallfilter<-householdfileallfilter%>%mutate(adjustedincome=case_when(year==2021~ hhinc/1.06,TRUE~ hhinc))
5.
Note: number six is missing, instead number 5 has two parts.
#cap the adjusted income variable at the 99th percentile, opting to ignore missing vluaeshouseholdfileallfilter_capped<-householdfileallfilter%>%filter(adjustedincome<=quantile(adjustedincome, 0.99, na.rm=TRUE))#did family income change bewtween 2019 an 2021?householdfileallfilter_capped2019<-householdfileallfilter_capped%>%filter(year==2019)householdfileallfilter_capped2021<-householdfileallfilter_capped%>%filter(year==2021)mean(householdfileallfilter_capped2019$adjustedincome)
#check for statistical significance using a t-testttest<-t.test(adjustedincome~year, data=householdfileallfilter_capped)ttest
Welch Two Sample t-test
data: adjustedincome by year
t = 7.757, df = 289738, p-value = 8.723e-15
alternative hypothesis: true difference in means between group 2019 and group 2021 is not equal to 0
95 percent confidence interval:
2270.543 3805.876
sample estimates:
mean in group 2019 mean in group 2021
124075.8 121037.6
124075.8-121037.6
[1] 3038.2
The t-test confirms that the change in mean adjusted income from 2019 to 2021 is significant. The mean income was $124,075.80 in 2019 and $121,037.60 in 2021, a drop of 3,038 dollars.
Note: I’m going to continue working with the dataframe that dropped any income values above the 99th percentile.
7.
#merge adjusted income variable with the person-level data so each person has their adjusted household income#create dataframe with only household identifier and adjusted incomehhandid<-householdfileallfilter_capped%>%select(serialno, adjustedincome)personfileallincome<-merge(personfileall, hhandid, by= ("serialno"))#note: this dataframe is not going to ONLY have information for individuals living in a household considered a family household that have an income not above the 99th percentile of the original family income data. #show layouthead(personfileallincome)
#make a variable that indicates whether a person lives in a household the earns less than 150% of the federal poverty level#check distributionsummary(personfileallincome$povpip)
Min. 1st Qu. Median Mean 3rd Qu. Max. NA's
0.0 252.0 449.0 371.6 501.0 501.0 1870
personfileallincome<-personfileallincome%>%mutate(below150=(povpip<150))#this makes TRUE/FALSE variable in which TRUE indicates that individuals live in a household that earns less than 150% of FPL#see distributionlibrary(janitor)
Attaching package: 'janitor'
The following objects are masked from 'package:stats':
chisq.test, fisher.test
tabyl(personfileallincome, below150)
below150 n percent valid_percent
FALSE 764437 0.872437823 0.8743038
TRUE 109901 0.125427981 0.1256962
NA 1870 0.002134196 NA
#note, I haven't dropped any NA's from this dataframe as the instructions have not indicated whether I should. Eventually I would do this during my data cleaning process. This table shows that 1870 values for my newly created varaible are NA's.
9.
#among children aged 3 or younger, how did the percent of children living below 150% poverty change between 2019 and 2021?#filter for only observations of children aged 3 or youngerpersonfileallincomebelow3<-personfileallincome%>%filter(agep<=3)#the is_child variable is equal to 0 then the specific is not a child, but if it is equal to 1 it is a child.tabyl(personfileallincomebelow3, is_child)
is_child n percent
0 4790 0.1221253
1 34432 0.8778747
#since the question refers to children only, I'm going to filter for is_child==1 because some of these cases may be adults with childrenchildrenonly<-personfileallincomebelow3%>%filter(is_child==1)
#create visualization of how the percent of children living below poverty changed from 2019 and 2021visualinfo<-childrenonly%>%group_by(year) %>%summarize(total_children =n(),below150 =sum(below150, na.rm =TRUE),percentlivingbelow150 =mean(below150/ total_children, na.rm =TRUE) *100 )head(visualinfo)
#make a bar graphlibrary(ggplot2)ggplot(visualinfo, aes(x =factor(year), y = percentlivingbelow150, fill =factor(year))) +geom_col(width =0.5, color ="black", alpha =0.8) +geom_text(aes(label =paste0(round(percentlivingbelow150, 1), "%")), vjust =-0.5, fontface ="bold", size =5) +scale_fill_manual(values =c("2019"="orange", "2021"="blue")) +# Sets the axis limit slightly higher than the max percentage for spacingscale_y_continuous(labels =function(x) paste0(x, "%"), limits =c(0, 30)) +labs(title ="% of children living in a household with an income below 150% of the federal poverty line",x ="Year",y ="% of children" ) +theme_minimal(base_size =8) +theme(plot.title =element_text(face ="bold", hjust =0.5),plot.subtitle =element_text(hjust =0.5),legend.position ="none" )
10.
#for each state estiamte the proportion of white, black and hispanic children aged 3 or younger who lived below 150% of poverty in 2021 using survey-weightslibrary(srvyr)
Attaching package: 'srvyr'
The following object is masked from 'package:stats':
filter
#note, I still haven't dropped NA's so they are included in this list.
11.
#make a variable that shows whether a household contains a person aged 3 or younger#use the orignal all person with adjusted income herepersonfileallincomegroupedbyhh<-personfileallincome%>%mutate(child3oryounger= agep<=3)