library("data.table")
##
## Attaching package: 'data.table'
## The following object is masked from 'package:base':
##
## %notin%
library("bit64")
## ********************************************************
## R-core is collecting use cases for 64-bit integers as they explore native support for these vectors.
##
## See https://stat.ethz.ch/pipermail/r-devel/2026-July/084631.html and reach out to Luke Tierney (luke-tierney@uiowa.edu).
## ********************************************************
##
## Attaching package: 'bit64'
## The following object is masked from 'package:utils':
##
## hashtab
## The following objects are masked from 'package:base':
##
## %in%, :, array, as.factor, as.ordered, colSums, factor, intersect,
## is.double, is.element, match, matrix, order, rank, rowSums,
## setdiff, setequal, table, union
library("dplyr")
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:bit64':
##
## intersect, setdiff, setequal, union
## The following objects are masked from 'package:data.table':
##
## between, first, last
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library("pastecs")
##
## Attaching package: 'pastecs'
## The following objects are masked from 'package:dplyr':
##
## first, last
## The following objects are masked from 'package:data.table':
##
## first, last
# Store the text "pu2025.csv" in a name called ds
ds <- c("pu2025.csv")
# The file is pipe-delimited even though it's named .csv
pu <- fread(ds, sep = "|", select = c("SSUID", "ERESIDENCEID", "PNUM", "MONTHCODE", "SPANEL", "WPFINWGT", "ESEEING", "RFOODS", "RFOODR", "TAGE", "ESEX", "THHLDSTATUS"))
# Out of 5,203 variables, 12 variables were selected. The first six in the above line represent ID and structure variables. SSUID is sample unit ID with PNUM identifies a person and ERESIDENCEID identifies a household. PNUM is person number within the sample unit. MONTHCODE is reference month (1–12). SPANEL is panel year (2022-2025). WPFINWGT is the final person weight.The next six represent ESEEING blind or serious difficulty seeing even with glasses/contacts. RFOODS food security status recode: 1 high/marginal, 2 low, 3 very low. RFOODR is the raw food security score a count of affirmative responses (0–6). TAGE is age as of last birthday (capped at 90 for confidentiality). ESEX is sex with 1 Male, 2 Female. THHLDSTATUS represents household status and was added in the midst of coding due to needing to prove why I received 5,423 NAs.
# Make sure all the column names are upper-case
names(pu) <- toupper(names(pu))
# Preview the data
head(pu, 20)
## SSUID ERESIDENCEID PNUM MONTHCODE SPANEL WPFINWGT ESEEING RFOODS
## <i64> <int> <int> <int> <int> <num> <int> <int>
## 1: 11428525324 100001 101 1 2024 8290.772 2 1
## 2: 11428525324 100001 101 2 2024 8476.619 2 1
## 3: 11428525324 100001 101 3 2024 9238.426 2 1
## 4: 11428525324 100001 101 4 2024 9371.330 2 1
## 5: 11428525324 100001 101 5 2024 9547.140 2 1
## 6: 11428525324 100001 101 6 2024 9337.827 2 1
## 7: 11428525324 100001 101 7 2024 9096.493 2 1
## 8: 11428525324 100001 101 8 2024 9310.043 2 1
## 9: 11428525324 100001 101 9 2024 9209.630 2 1
## 10: 11428525324 100001 101 10 2024 9150.450 2 1
## 11: 11428525324 100001 101 11 2024 9195.871 2 1
## 12: 11428525324 100001 101 12 2024 9433.126 2 1
## 13: 11428534325 100001 101 1 2025 4473.576 2 3
## 14: 11428534325 100001 101 2 2025 4561.484 2 3
## 15: 11428534325 100001 101 3 2025 4761.070 2 3
## 16: 11428534325 100001 101 4 2025 5111.557 2 3
## 17: 11428534325 100001 101 5 2025 5001.163 2 3
## 18: 11428534325 100001 101 6 2025 4933.283 2 3
## 19: 11428534325 100001 101 7 2025 4894.891 2 3
## 20: 11428534325 100001 101 8 2025 4716.378 2 3
## SSUID ERESIDENCEID PNUM MONTHCODE SPANEL WPFINWGT ESEEING RFOODS
## <i64> <int> <int> <int> <int> <num> <int> <int>
## RFOODR TAGE ESEX THHLDSTATUS
## <int> <int> <int> <int>
## 1: 0 70 1 1
## 2: 0 70 1 1
## 3: 0 70 1 1
## 4: 0 70 1 1
## 5: 0 70 1 1
## 6: 0 70 1 1
## 7: 0 70 1 1
## 8: 0 70 1 1
## 9: 0 70 1 1
## 10: 0 70 1 1
## 11: 0 70 1 1
## 12: 0 70 1 1
## 13: 6 50 1 2
## 14: 6 50 1 2
## 15: 6 50 1 2
## 16: 6 50 1 2
## 17: 6 50 1 2
## 18: 6 50 1 2
## 19: 6 50 1 2
## 20: 6 50 1 2
## RFOODR TAGE ESEX THHLDSTATUS
## <int> <int> <int> <int>
table(pu$SPANEL)
##
## 2022 2023 2024 2025
## 120652 69613 75519 113431
spanel_2025 <- pu %>% filter(SPANEL == 2025)
nrow(spanel_2025)
## [1] 113431
# Reference years differ across panels; weights are designed within panel. Kept 2025 for recency, sample from 2025 is 113,431 person-months.
n_distinct(spanel_2025$SSUID, spanel_2025$PNUM)
## [1] 9490
month_1 <- spanel_2025 %>% filter(MONTHCODE == 1)
nrow(month_1)
## [1] 9419
# 9,490 people in the 2025 panel, one row per person is set, nrow produces 71-person gap.
first_months <- spanel_2025 %>% group_by(SSUID, PNUM) %>% summarise(first_month =min(MONTHCODE))
## `summarise()` has regrouped the output.
## ℹ Summaries were computed grouped by SSUID and PNUM.
## ℹ Output is grouped by SSUID.
## ℹ Use `summarise(.groups = "drop_last")` to silence this message.
## ℹ Use `summarise(.by = c(SSUID, PNUM))` for per-operation grouping
## (`?dplyr::dplyr_by`) instead.
table(first_months$first_month)
##
## 1 2 3 5 6 7 8 9 10 11 12
## 9419 5 6 6 13 8 8 7 5 5 8
# All 71 missing people have their first record in the data after January, spread from February to December. None of them have a January row, so filtering to January drops them. Data doesn't show why (e.g., birth, moving, etc.)
nrow(month_1)
## [1] 9419
stat.desc(month_1$RFOODR)
## nbr.val nbr.null nbr.na min max range
## 9.419000e+03 7.743000e+03 0.000000e+00 0.000000e+00 6.000000e+00 6.000000e+00
## sum median mean SE.mean CI.mean.0.95 var
## 4.921000e+03 0.000000e+00 5.224546e-01 1.423524e-02 2.790414e-02 1.908686e+00
## std.dev coef.var
## 1.381552e+00 2.644348e+00
# no NAs produced
hist(month_1$RFOODR)
# zero-inflated right-tail skew
month_1 <- month_1 %>% mutate(sqrttrnsfrm = sqrt(RFOODR))
# log doesn't like zeroes so I'm using sqrt
head(month_1)
## SSUID ERESIDENCEID PNUM MONTHCODE SPANEL WPFINWGT ESEEING RFOODS
## <i64> <int> <int> <int> <int> <num> <int> <int>
## 1: 11428534325 100001 101 1 2025 4473.576 2 3
## 2: 11428534325 100001 102 1 2025 4473.576 2 3
## 3: 11428534325 100001 103 1 2025 6465.566 2 3
## 4: 11428534325 100001 104 1 2025 6086.613 2 3
## 5: 11428534325 100001 105 1 2025 6539.197 2 3
## 6: 11486013425 100001 101 1 2025 8086.215 2 1
## RFOODR TAGE ESEX THHLDSTATUS sqrttrnsfrm
## <int> <int> <int> <int> <num>
## 1: 6 50 1 2 2.44949
## 2: 6 38 2 2 2.44949
## 3: 6 20 1 2 2.44949
## 4: 6 15 1 2 2.44949
## 5: 6 6 1 2 2.44949
## 6: 0 43 2 2 0.00000
hist(month_1$sqrttrnsfrm)
# the variable is still not normal after the transformation because of all the zeros.
RFOODR is the raw food security score; it is a count of affirmative responses (0–6). The survey asked the household six yes/no questions about running short on food because of money. RFOODR is how many of those six they said yes to. A score of 0 means they said no to all six. A score of 6 means they said yes to all six. Stat.desc yielded no NAs, 7,743 people scored 0 or roughly 81.6 percent of the total. The median is 0 and the mean is 0.5224546. The difference between these two values proves a skewness.As seen in the last assignment, a histogram of RFOODR gives a zero-inflated right-tail skew. Because the log function returns -Inf, I am using square root. But after transforming the variable, there is still a not normal distribution due to the amount of zeroes.