install.packages(“formatR”) install.packages(“glmnetUtils”) install.packages(“glmnet”) library(glmnetUtils) library(glmnet)
set.seed(333)
mydata <- read.csv(‘NBA.csv’)
clean_data <- na.omit(mydata)
nba_imputed <- mydata for (i in 9:ncol(nba_imputed)) { missing_rows <- is.na(nba_imputed[, i]) nba_imputed[missing_rows, i] <- mean(nba_imputed[, i], na.rm = TRUE) }
max_games <- max(nba_imputed\(G) new_nba <- nba_imputed[nba_imputed\)G >= 50, ]
nba_US <- new_nba[new_nba$NBA_Country == “USA”, ]
##1
##a
nba_US <- nba_US[, !(names(nba_US) %in% c(“NBA_Country”, “Player”))] nba_US\(Log_Salary <- log(nba_US\)Salary) lasso <- glmnetUtils::cv.glmnet(Log_salary ~ ., data = nba_US, family = “binomial, alpha = 1, nfold = 5)
par(mfrow=c(1, 2)) plot(lasso$glmnet, xvar = ‘lambda’, label = TRUE) plot(lasso)
##b
cat(“As we move from the left side to the right, the coefficients move towards zero, or become smaller. This is because in a lasso regression, as lambda increases the further right we move the coefficients are penalized and decrease in order to make a simpler model with the subset of the original dataset we are using.Yes, the coefficients at the rightmost and leftmost sides will be different because of the decrease of coefficients as the graph moves right.”)
##c
cat(“The higher lambda should have a higher bias, since as the coefficients decrease with a higher lambda this should increase the bias. With a lower lambda comes a higher variance, since the lower lambda has a lower penalty on the coefficients which decreases bias but increase variance.”)
##2
set.seed(333)
cv_lasso <- 1 - lasso\(cvm[lasso\)index[1, 1]]/lasso$cvm[1]
plot(cv_lasso)
cat(“The shape of the plot shows the high variance on the left side of the curve and the high bias on the right side of the curve which are trade-offs of the lasso regression. Therefore, at the bottom of the curve towards the middle is the optimal trade-off between variance and bias thus giving the lowest cross-validation error.”)
##3
lambda_low <- cv_lasso$lambda.min lasso_coef <- coef(cv_lasso, s = lambda_low)
print(“lasso coefficients”) print(lasso_coef)
nonzero_coefficients <- names(lasso_coef)[lasso_coef = 0] print(“Nonzero coefficients:”) print(nonzero_coefficients)
##4
salary_change <- lm(Salary ~ NBA_DraftNumber, data = nba_US) draft_number_coef <- coef(salary_change)[“NBA_DraftNumber”] percentage_change <- draft_number_coef * 100 print(paste(“Percentage change in salary associated with a one-unit increase in draft number:”, round(percentage_change, 2), “%”))
cat(“This is very surprising since you would expect being drafted higher to give a player a higher salary, however when looking at the dataset it makes more sense since some players who were drafted higher have considerably lower salaries than those drafted lower. Also considering that some players may have been drafted lower but played in the league for longer or performed better this also makes more sense.”)
##5 set.seed(333)
nba_US\(Log_Salary <- log(nba_US\)Salary) ridge <- glmnetUtils::cv.glmnet(Log_salary ~ ., data = nba_US, family = “binomial, alpha = 0, nfold = 5)
lambda_min <- ridge$lambda.min ridge_coef <- coef(ridge, s = lambda_min) print(ridge_coef)
cv_lasso <- 1 - lasso\(cvm[lasso\)index[1, 1]]/lasso\(cvm[1] cv_ridge <- 1 - ridge\)cvm[ridge\(index[1, 1]]/ridge\)cvm[1]
ridge_coef <- coef(ridge, s = lambda_min) lasso_coef <- coef(lasso, s = lambda_min)
cat(“The OOS R^2 of the ridge and lasso are lower than the OLS.”)
##7
##a
cat(“I would consider interacting points scored and teams because a player who scores more points would typically have a higher salary, however the team they play for could also affect how high their salary is since if they score many points but play for a smaller team or a team that has less money they may not be paid the same as a similar scoring player who plays for a larger team.”)
##b
##d
cat(“Adding more regressors could cause a greater risk of overfitting as well as the risk of multicollinearity by adding more regressors which could be correlated and cause the regression to have trouble differentiating between the coefficients and inflating or fluctuating the coefficients.”)