install.packages(“glmnet”) install.packages(“dplyr”) install.packages(“ranger”) #Problem 1

mydata <- read.csv("sipp1991.csv")

##part a

nfa <- glm(net_tfa ~ age + inc + e401, data = mydata)
summary(nfa)
## 
## Call:
## glm(formula = net_tfa ~ age + inc + e401, data = mydata)
## 
## Coefficients:
##               Estimate Std. Error t value Pr(>|t|)    
## (Intercept) -5.551e+04  3.404e+03 -16.310  < 2e-16 ***
## age          9.417e+02  7.781e+01  12.103  < 2e-16 ***
## inc          8.748e-01  3.505e-02  24.959  < 2e-16 ***
## e401         5.135e+03  1.754e+03   2.927  0.00343 ** 
## ---
## Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
## 
## (Dispersion parameter for gaussian family taken to be 3236949127)
## 
##     Null deviance: 1.9335e+13  on 4999  degrees of freedom
## Residual deviance: 1.6172e+13  on 4996  degrees of freedom
## AIC: 123685
## 
## Number of Fisher Scoring iterations: 2
cat("Since the coefficient on the term for offer of enrollment into a 401k plan is positive it tells us economically that working at an employer offering enrollment is positively corellated with an increase in net financial assets.")
## Since the coefficient on the term for offer of enrollment into a 401k plan is positive it tells us economically that working at an employer offering enrollment is positively corellated with an increase in net financial assets.

##part b

cat("The coefficient for the offering of a 401k plan should not be seen as a causal effect in this regression because of other unseen factors such as companies possibly offering other advantages and benefits such as stock options causing OVB, as well as the potential for reverse causuality since employees who are more concious about saving might seek out employers that offer a 401k.")
## The coefficient for the offering of a 401k plan should not be seen as a causal effect in this regression because of other unseen factors such as companies possibly offering other advantages and benefits such as stock options causing OVB, as well as the potential for reverse causuality since employees who are more concious about saving might seek out employers that offer a 401k.

##part c

cat("This helps with the concerns raised in part b because if individuals at the time focused on pay and income then there could be less concern for OVB since the factor of income would be the main concern. Also the factors of selection bias and reverse causality would be mitigated if income is the main driver in employees choosing their employer.")
## This helps with the concerns raised in part b because if individuals at the time focused on pay and income then there could be less concern for OVB since the factor of income would be the main concern. Also the factors of selection bias and reverse causality would be mitigated if income is the main driver in employees choosing their employer.

##part d

cat("The argument would not help when considering enrollment in a 401k plan rather than offer of enrollment since the choice to enroll in a 401k plan is affected by individual characteristics such as financial literacy and saving habits. This means that the previous concerns of selection bias and omitted variable bias come back since there could be other factors involved with employees who choose to enroll. Therefore, compared with an offer of enrollment being exogenous, the choice to enroll is not and could cause issues in the linear regression.")
## The argument would not help when considering enrollment in a 401k plan rather than offer of enrollment since the choice to enroll in a 401k plan is affected by individual characteristics such as financial literacy and saving habits. This means that the previous concerns of selection bias and omitted variable bias come back since there could be other factors involved with employees who choose to enroll. Therefore, compared with an offer of enrollment being exogenous, the choice to enroll is not and could cause issues in the linear regression.

#Problem 2

##part a

cat("The coefficient on age is 9.417e+02, or around 1,000. This means that with each year of age their nfa, or net financial assets, would be expected to rise around 1,000 dollars. This might not be as plausible for older people since they might be spending some of their savings as they transition into retirement, or may be expected to have more healthcare related costs which would effect the cycle of their savings and cause a nonlinear pattern in their saving.")
## The coefficient on age is 9.417e+02, or around 1,000. This means that with each year of age their nfa, or net financial assets, would be expected to rise around 1,000 dollars. This might not be as plausible for older people since they might be spending some of their savings as they transition into retirement, or may be expected to have more healthcare related costs which would effect the cycle of their savings and cause a nonlinear pattern in their saving.

##part b

library(glmnet)
## Loading required package: Matrix
## Loaded glmnet 4.1-8
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
set.seed(333)

mydata$age <- as.factor(mydata$age)
mydata$educ <- as.factor(mydata$educ)

y_model <- glmnetUtils::cv.glmnet(net_tfa ~ age + educ +
    poly(inc, 3, raw = TRUE) + poly(fsize, 3, raw = TRUE) +
    db + marr + twoearn + e401 + pira + hown + (age + educ +
    poly(inc, 3, raw = TRUE) + poly(fsize, 3, raw = TRUE) +
    db + marr + twoearn + e401 + pira + hown)^2, data = mydata,
    use.model.frame = TRUE, nfold = 3)

d_model <- glmnetUtils::cv.glmnet(p401 ~ age + educ + poly(inc,
    3, raw = TRUE) + poly(fsize, 3, raw = TRUE) + db + marr +
    twoearn + e401 + pira + hown + (age + educ + poly(inc,
    3, raw = TRUE) + poly(fsize, 3, raw = TRUE) + db + marr +
    twoearn + e401 + pira + hown)^2, data = mydata, use.model.frame = TRUE,
    nfold = 3)

##part c

d_nonzero_coef_ind <- which(coef(d_model, s = "lambda.min")[-1,
    ] != 0)
y_nonzero_coef_ind <- which(coef(y_model, s = "lambda.min")[-1] !=
    0)
nonzero_coef_ind <- union(y_nonzero_coef_ind, d_nonzero_coef_ind)

controls = model.matrix(~age + educ + poly(inc, 3, raw = TRUE) +
    poly(fsize, 3, raw = TRUE) + db + marr + twoearn + e401 +
    pira + hown + (age + educ + poly(inc, 3, raw = TRUE) +
    poly(fsize, 3, raw = TRUE) + db + marr + twoearn + e401 +
    pira + hown)^2, data = mydata)[, -1]

data_matrix <- data.frame(net_tfa = mydata$net_tfa, p401 = mydata$p401,
    controls[, nonzero_coef_ind])

cat("thirteen variables were actually selected by the post-lasso. We could have included 1,463 variables given by ncol(controls).")
## thirteen variables were actually selected by the post-lasso. We could have included 1,463 variables given by ncol(controls).

##part d

nfa_post_lasso <- glm(net_tfa ~ ., data = data_matrix)

coef(summary(nfa_post_lasso))["p401", ]
##     Estimate   Std. Error      t value     Pr(>|t|) 
## 1.406927e+04 2.621494e+03 5.366891e+00 8.373312e-08

##part e

cat("The average savings in the dataset are:")
## The average savings in the dataset are:
mean(mydata$net_tfa)
## [1] 17282.32
counterfactual <- data_matrix
counterfactual$e401 <- 1

predictedsavings <- predict(nfa_post_lasso, newdata = counterfactual)
averagepredicted <- mean(predictedsavings)

cat("The average predicted savings considering the counterfactual are:")
## The average predicted savings considering the counterfactual are:
mean(predictedsavings)
## [1] 17172.28
cat("As can be seen, the average predicted savings considering the counterfactual of all employers offering a 401k and the average savings from the dataset are fairly close in value which could mean the employer offering a 401k has little effect on the actual net financial assets.")
## As can be seen, the average predicted savings considering the counterfactual of all employers offering a 401k and the average savings from the dataset are fairly close in value which could mean the employer offering a 401k has little effect on the actual net financial assets.

##part f

cat("If the controls were not valid the comparison between the post-lasso and the OLS regressions would most likely also not be valid because with controls that are not valid the post-lasso regression would be similarly affected by potential bias and the results would not be able to be interpreted causally.")
## If the controls were not valid the comparison between the post-lasso and the OLS regressions would most likely also not be valid because with controls that are not valid the post-lasso regression would be similarly affected by potential bias and the results would not be able to be interpreted causally.

#Problem 3

##part a

library(ranger)
set.seed(777)

tfa_tree <- ranger(net_tfa ~ ., data = mydata, write.forest = TRUE,
    num.tree = 200, min.node.size = 10, importance = "impurity")

cat("The OOS R^2 is:")
## The OOS R^2 is:
tfa_tree$r.squared
## [1] 0.230476
predict(tfa_tree, mydata[1:10, ])$predictions
##  [1] 17436.1890  -452.2101 30115.5987  3667.5960  6473.9077 -2242.0618
##  [7]  7292.0118 36082.1958 -3242.8384 22255.0445