library("data.table")
## 
## Attaching package: 'data.table'
## The following object is masked from 'package:base':
## 
##     %notin%
library("bit64")
## ********************************************************
## R-core is collecting use cases for 64-bit integers as they explore native support for these vectors.
## 
## See https://stat.ethz.ch/pipermail/r-devel/2026-July/084631.html and reach out to Luke Tierney (luke-tierney@uiowa.edu).
## ********************************************************
## 
## Attaching package: 'bit64'
## The following object is masked from 'package:utils':
## 
##     hashtab
## The following objects are masked from 'package:base':
## 
##     %in%, :, array, as.factor, as.ordered, colSums, factor, intersect,
##     is.double, is.element, match, matrix, order, rank, rowSums,
##     setdiff, setequal, table, union
library("dplyr")
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:bit64':
## 
##     intersect, setdiff, setequal, union
## The following objects are masked from 'package:data.table':
## 
##     between, first, last
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library("pastecs")
## 
## Attaching package: 'pastecs'
## The following objects are masked from 'package:dplyr':
## 
##     first, last
## The following objects are masked from 'package:data.table':
## 
##     first, last
#   Store the text "pu2025.csv" in a name called ds
ds <- c("pu2025.csv")
#   The file is pipe-delimited even though it's named .csv
pu <- fread(ds, sep = "|", select = c("SSUID", "ERESIDENCEID", "PNUM", "MONTHCODE", "SPANEL", "WPFINWGT", "ESEEING", "RFOODS", "RFOODR", "TAGE", "ESEX", "THHLDSTATUS"))
#   Out of 5,203 variables, 12 variables were selected. The first six in the above line represent ID and structure variables. SSUID is sample unit ID with PNUM identifies a person and ERESIDENCEID identifies a household. PNUM is person number within the sample unit. MONTHCODE is reference month (1–12). SPANEL is panel year (2022-2025). WPFINWGT is the final person weight.The next six represent ESEEING blind or serious difficulty seeing even with glasses/contacts. RFOODS food security status recode: 1 high/marginal, 2 low, 3 very low. RFOODR is the raw food security score a count of affirmative responses (0–6). TAGE is age as of last birthday (capped at 90 for confidentiality). ESEX is sex with 1 Male, 2 Female. THHLDSTATUS represents household status and was added in the midst of coding due to needing to prove why I received 5,423 NAs.
#   Make sure all the column names are upper-case
names(pu) <- toupper(names(pu))
#   Preview the data
head(pu, 20)
##           SSUID ERESIDENCEID  PNUM MONTHCODE SPANEL WPFINWGT ESEEING RFOODS
##           <i64>        <int> <int>     <int>  <int>    <num>   <int>  <int>
##  1: 11428525324       100001   101         1   2024 8290.772       2      1
##  2: 11428525324       100001   101         2   2024 8476.619       2      1
##  3: 11428525324       100001   101         3   2024 9238.426       2      1
##  4: 11428525324       100001   101         4   2024 9371.330       2      1
##  5: 11428525324       100001   101         5   2024 9547.140       2      1
##  6: 11428525324       100001   101         6   2024 9337.827       2      1
##  7: 11428525324       100001   101         7   2024 9096.493       2      1
##  8: 11428525324       100001   101         8   2024 9310.043       2      1
##  9: 11428525324       100001   101         9   2024 9209.630       2      1
## 10: 11428525324       100001   101        10   2024 9150.450       2      1
## 11: 11428525324       100001   101        11   2024 9195.871       2      1
## 12: 11428525324       100001   101        12   2024 9433.126       2      1
## 13: 11428534325       100001   101         1   2025 4473.576       2      3
## 14: 11428534325       100001   101         2   2025 4561.484       2      3
## 15: 11428534325       100001   101         3   2025 4761.070       2      3
## 16: 11428534325       100001   101         4   2025 5111.557       2      3
## 17: 11428534325       100001   101         5   2025 5001.163       2      3
## 18: 11428534325       100001   101         6   2025 4933.283       2      3
## 19: 11428534325       100001   101         7   2025 4894.891       2      3
## 20: 11428534325       100001   101         8   2025 4716.378       2      3
##           SSUID ERESIDENCEID  PNUM MONTHCODE SPANEL WPFINWGT ESEEING RFOODS
##           <i64>        <int> <int>     <int>  <int>    <num>   <int>  <int>
##     RFOODR  TAGE  ESEX THHLDSTATUS
##      <int> <int> <int>       <int>
##  1:      0    70     1           1
##  2:      0    70     1           1
##  3:      0    70     1           1
##  4:      0    70     1           1
##  5:      0    70     1           1
##  6:      0    70     1           1
##  7:      0    70     1           1
##  8:      0    70     1           1
##  9:      0    70     1           1
## 10:      0    70     1           1
## 11:      0    70     1           1
## 12:      0    70     1           1
## 13:      6    50     1           2
## 14:      6    50     1           2
## 15:      6    50     1           2
## 16:      6    50     1           2
## 17:      6    50     1           2
## 18:      6    50     1           2
## 19:      6    50     1           2
## 20:      6    50     1           2
##     RFOODR  TAGE  ESEX THHLDSTATUS
##      <int> <int> <int>       <int>
table(pu$SPANEL)
## 
##   2022   2023   2024   2025 
## 120652  69613  75519 113431
spanel_2025 <- pu %>% filter(SPANEL == 2025)
nrow(spanel_2025)
## [1] 113431
#   Reference years differ across panels; weights are designed within panel. Kept 2025 for recency, sample from 2025 is 113,431 person-months.
n_distinct(spanel_2025$SSUID, spanel_2025$PNUM)
## [1] 9490
month_1 <- spanel_2025 %>% filter(MONTHCODE == 1)
nrow(month_1)
## [1] 9419
#   9,490 people in the 2025 panel, one row per person is set, nrow produces 71-person gap.
first_months <- spanel_2025 %>% group_by(SSUID, PNUM) %>%  summarise(first_month =min(MONTHCODE))
## `summarise()` has regrouped the output.
## ℹ Summaries were computed grouped by SSUID and PNUM.
## ℹ Output is grouped by SSUID.
## ℹ Use `summarise(.groups = "drop_last")` to silence this message.
## ℹ Use `summarise(.by = c(SSUID, PNUM))` for per-operation grouping
##   (`?dplyr::dplyr_by`) instead.
table(first_months$first_month)
## 
##    1    2    3    5    6    7    8    9   10   11   12 
## 9419    5    6    6   13    8    8    7    5    5    8
# All 71 missing people have their first record in the data after January, spread from February to December. None of them have a January row, so filtering to January drops them. Data doesn't show why (e.g., birth, moving, etc.)
nrow(month_1)
## [1] 9419
stat.desc(month_1$RFOODR)
##      nbr.val     nbr.null       nbr.na          min          max        range 
## 9.419000e+03 7.743000e+03 0.000000e+00 0.000000e+00 6.000000e+00 6.000000e+00 
##          sum       median         mean      SE.mean CI.mean.0.95          var 
## 4.921000e+03 0.000000e+00 5.224546e-01 1.423524e-02 2.790414e-02 1.908686e+00 
##      std.dev     coef.var 
## 1.381552e+00 2.644348e+00
# no NAs produced
hist(month_1$RFOODR)

# zero-inflated right-tail skew
month_1 <- month_1 %>% mutate(sqrttrnsfrm = sqrt(RFOODR))
# log doesn't like zeroes so I'm using sqrt
head(month_1)
##          SSUID ERESIDENCEID  PNUM MONTHCODE SPANEL WPFINWGT ESEEING RFOODS
##          <i64>        <int> <int>     <int>  <int>    <num>   <int>  <int>
## 1: 11428534325       100001   101         1   2025 4473.576       2      3
## 2: 11428534325       100001   102         1   2025 4473.576       2      3
## 3: 11428534325       100001   103         1   2025 6465.566       2      3
## 4: 11428534325       100001   104         1   2025 6086.613       2      3
## 5: 11428534325       100001   105         1   2025 6539.197       2      3
## 6: 11486013425       100001   101         1   2025 8086.215       2      1
##    RFOODR  TAGE  ESEX THHLDSTATUS sqrttrnsfrm
##     <int> <int> <int>       <int>       <num>
## 1:      6    50     1           2     2.44949
## 2:      6    38     2           2     2.44949
## 3:      6    20     1           2     2.44949
## 4:      6    15     1           2     2.44949
## 5:      6     6     1           2     2.44949
## 6:      0    43     2           2     0.00000
hist(month_1$sqrttrnsfrm)

# the variable is still not normal after the transformation because of all the zeros.

RFOODR is the raw food security score; it is a count of affirmative responses (0–6). The survey asked the household six yes/no questions about running short on food because of money. RFOODR is how many of those six they said yes to. A score of 0 means they said no to all six. A score of 6 means they said yes to all six. Stat.desc yielded no NAs, 7,743 people scored 0 or roughly 81.6 percent of the total. The median is 0 and the mean is 0.5224546. The difference between these two values proves a skewness.As seen in the last assignment, a histogram of RFOODR gives a zero-inflated right-tail skew. Because the log function returns -Inf, I am using square root. But after transforming the variable, there is still a not normal distribution due to the amount of zeroes.