library(readr)
MLB2 <- read_csv("C:/Users/jimmy/Downloads/mlb_teams.csv")
## Rows: 2784 Columns: 42
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (8): league_id, division_id, division_winner, wild_card_winner, league_...
## dbl (34): rownames, year, rank, games_played, home_games, wins, losses, runs...
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
names(MLB2)
## [1] "rownames" "year" "league_id"
## [4] "division_id" "rank" "games_played"
## [7] "home_games" "wins" "losses"
## [10] "division_winner" "wild_card_winner" "league_winner"
## [13] "world_series_winner" "runs_scored" "at_bats"
## [16] "hits" "doubles" "triples"
## [19] "homeruns" "walks" "strikeouts_by_batters"
## [22] "stolen_bases" "caught_stealing" "batters_hit_by_pitch"
## [25] "sacrifice_flies" "opponents_runs_scored" "earned_runs_allowed"
## [28] "earned_run_average" "complete_games" "shutouts"
## [31] "saves" "outs_pitches" "hits_allowed"
## [34] "homeruns_allowed" "walks_allowed" "strikeouts_by_pitchers"
## [37] "errors" "double_plays" "fielding_percentage"
## [40] "team_name" "ball_park" "home_attendance"
relevant_vars <- MLB2[, c("wins", "homeruns", "hits", "walks")]
cor_matrix <- cor(relevant_vars, use = "complete.obs")
print("Correlation Matrix:")
## [1] "Correlation Matrix:"
print(cor_matrix)
## wins homeruns hits walks
## wins 1.0000000 0.4172266 0.6434393 0.5837370
## homeruns 0.4172266 1.0000000 0.4301023 0.5421865
## hits 0.6434393 0.4301023 1.0000000 0.6383888
## walks 0.5837370 0.5421865 0.6383888 1.0000000
fit <- lm(wins ~ homeruns + hits + walks, data = MLB2)
library(car)
## Warning: package 'car' was built under R version 4.4.2
## Loading required package: carData
##
## Attaching package: 'car'
## The following object is masked from 'package:dplyr':
##
## recode
vif_values <- vif(fit)
print("Variance Inflation Factors (VIFs):")
## [1] "Variance Inflation Factors (VIFs):"
print(vif_values)
## homeruns hits walks
## 1.440651 1.716822 1.981818
refined_fit <- lm(wins ~ homeruns + hits + walks, data = relevant_vars)
summary(refined_fit)
##
## Call:
## lm(formula = wins ~ homeruns + hits + walks, data = relevant_vars)
##
## Residuals:
## Min 1Q Median 3Q Max
## -45.605 -7.932 0.192 8.085 45.245
##
## Coefficients:
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 4.439767 1.664137 2.668 0.00768 **
## homeruns 0.023240 0.004287 5.421 6.45e-08 ***
## hits 0.038027 0.001548 24.562 < 2e-16 ***
## walks 0.035282 0.002728 12.934 < 2e-16 ***
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 11.59 on 2780 degrees of freedom
## Multiple R-squared: 0.4701, Adjusted R-squared: 0.4695
## F-statistic: 822.1 on 3 and 2780 DF, p-value: < 2.2e-16
suppressWarnings(
stargazer(refined_fit, type = "text",
title = "Refined Regression Results: Wins vs. Homeruns, Hits, and Walks",
covariate.labels = c("Homeruns", "Hits", "Walks"),
dep.var.labels = "wins")
)
##
## Refined Regression Results: Wins vs. Homeruns, Hits, and Walks
## ===============================================
## Dependent variable:
## ---------------------------
## wins
## -----------------------------------------------
## Homeruns 0.023***
## (0.004)
##
## Hits 0.038***
## (0.002)
##
## Walks 0.035***
## (0.003)
##
## Constant 4.440***
## (1.664)
##
## -----------------------------------------------
## Observations 2,784
## R2 0.470
## Adjusted R2 0.470
## Residual Std. Error 11.587 (df = 2780)
## F Statistic 822.141*** (df = 3; 2780)
## ===============================================
## Note: *p<0.1; **p<0.05; ***p<0.01
library(ggplot2)
ggplot(data = relevant_vars, aes(x = homeruns, y = wins)) +
geom_point(alpha = 0.6) +
geom_smooth(method = "lm", color = "blue", se = FALSE) +
labs(
title = "Relationship Between Home Runs and Wins",
x = "Home Runs",
y = "Wins"
) +
theme_minimal()
## `geom_smooth()` using formula = 'y ~ x'
ggplot(data = relevant_vars, aes(x = hits, y = wins)) +
geom_point(alpha = 0.6) +
geom_smooth(method = "lm", color = "green", se = FALSE) +
labs(title = "Relationship Between Hits and Wins",
x = "Hits",
y = "Wins") +
theme_minimal()
## `geom_smooth()` using formula = 'y ~ x'
ggplot(data = relevant_vars, aes(x = walks, y = wins)) +
geom_point(alpha = 0.6) +
geom_smooth(method = "lm", color = "red", se = FALSE) +
labs(title = "Relationship Between Walks and Wins",
x = "Walks",
y = "Wins") +
theme_minimal()
## `geom_smooth()` using formula = 'y ~ x'
conf_intervals <- confint(refined_fit, level = 0.95)
coefficients <- summary(refined_fit)$coefficients
conf_table <- data.frame(
Variable = rownames(coefficients),
Coefficient = coefficients[, "Estimate"],
`Lower Bound` = conf_intervals[, 1],
`Upper Bound` = conf_intervals[, 2],
`P-Value` = coefficients[, "Pr(>|t|)"]
)
print(conf_table)
## Variable Coefficient Lower.Bound Upper.Bound P.Value
## (Intercept) (Intercept) 4.43976741 1.17669887 7.70283595 7.676760e-03
## homeruns homeruns 0.02323967 0.01483310 0.03164623 6.448836e-08
## hits hits 0.03802699 0.03499129 0.04106269 9.788583e-121
## walks walks 0.03528242 0.02993357 0.04063126 3.347624e-37
anova_result <- anova(refined_fit)
anova_result
## Analysis of Variance Table
##
## Response: wins
## Df Sum Sq Mean Sq F value Pr(>F)
## homeruns 1 122620 122620 913.29 < 2.2e-16 ***
## hits 1 186067 186067 1385.85 < 2.2e-16 ***
## walks 1 22461 22461 167.29 < 2.2e-16 ***
## Residuals 2780 373249 134
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
library(ggplot2)
residuals <- residuals(refined_fit)
residuals_df <- data.frame(Residuals = residuals)
ggplot(residuals_df, aes(x = Residuals)) +
geom_density(fill = "blue", alpha = 0.4) +
labs(title = "Density Plot of Regression Residuals",
x = "Residuals",
y = "Density") +
theme_minimal()
So far, I investigated the relationship between home runs and team success, focusing on how home runs impact the number of games a team wins during a season. To answer this question, I conducted a simple linear regression analysis where the dependent variable was the number of wins, and the independent variable was the total number of home runs hit by a team. The dataset consisted of official MLB team statistics, encompassing 2,784 team seasons, which provided a large pool of coverage. The variables analyzed included home runs as an indicator of power-hitting ability and wins as the measure of team success. This analysis aimed to quantify the effect of each additional home run on the expected number of wins. The initial results revealed that the coefficient for home runs was 0.108, meaning each additional home run increased expected wins by approximately 0.108 games. This relationship was statistically significant, with a p-value of less than 2e-16 and a 95% confidence interval of 0.0992 to 0.1167, providing strong evidence that the relationship is not due to random chance. However, the R-squared value of 0.174 indicated that home runs explained only 17.4% of the variation in wins, suggesting that other factors contribute significantly to team success. Additionally, the intercept of 64.287 suggested that teams with zero home runs could still expect to win around 64 games, implying that factors outside the model played a major role in determining outcomes. Residual analysis showed substantial unexplained variability, with a standard error of 14.46. Despite these findings, the analysis faced several limitations. Omitted variable bias was a concern, as factors like pitching performance, defensive metrics, and on-base percentage were not included in the model but likely influenced both home runs and wins. Potential heteroskedasticity might also affect the reliability of standard errors, and the inability to establish causality raised concerns about endogeneity. For instance, unobserved factors such as overall team quality could drive both home runs and wins, leading to biased estimates of the relationship. To improve inference, addressing the limitations of the initial model would significantly enhance its reliability and explanatory power. Expanding the model to include additional variables such as ERA, fielding percentage, on-base percentage, and injury rates would provide a more comprehensive understanding of team success by accounting for key factors beyond home runs. Testing for endogeneity and using instrumental variables, such as ballpark dimensions or weather conditions, could help isolate the causal effect of home runs on wins, reducing potential biases in the estimates. Addressing heteroskedasticity with robust standard errors and testing for multicollinearity among predictors would further improve the model’s accuracy and robustness. Additionally, exploring non-linear relationships and interaction effects would capture more nuanced dynamics between home runs and wins, leading to a deeper understanding of the factors contributing to team success. Implementing these improvements would yield a more robust model, offering more reliable insights into the impact of home runs on team performance.
Including hits and walks in the regression model provides a more comprehensive understanding of a team’s offensive performance. Hits reflect overall hitting success and consistency, focusing specifically on a team’s ability to make contact and generate base hits. This metric is critical for producing runs and ultimately winning games, as consistent hitting often drives offensive production. Walks, on the other hand, measure a team’s ability to reach base without relying solely on hits, capturing a broader aspect of offensive strategy. Walks emphasize the importance of plate discipline and on-base opportunities, complementing power-hitting by consistently putting runners on base to create scoring chances. Together, hits and walks provide a holistic view of offensive effectiveness, helping to explain how teams translate offensive success into wins and capturing nuances beyond the impact of home runs alone.
The results from the correlation matrix and Variance Inflation Factors indicate that multicollinearity is not a significant concern in this model. The correlation matrix shows moderate relationships between the variables: home runs and hits have a correlation of 0.43, home runs and walks have a correlation of 0.54, and hits and walks have a correlation of 0.64. These values suggest that while the variables are somewhat related, they are distinct enough to contribute unique information to the model. None of the pairwise correlations exceed 0.8, which is a common threshold for multicollinearity concerns. The VIF values further support this conclusion. The VIFs for home runs, hits, and walks are 1.44, 1.72, and 1.98, respectively, all well below the commonly used threshold of 10. VIF values under 2 indicate that the independent variables are relatively independent in this regression model. Together, these results confirm that the variables are suitable for inclusion in the model without significant risk of multicollinearity distorting the regression estimates. This ensures that the model can provide reliable insights into the relationships between home runs, hits, walks, and team success.
Run a linear regression to estimate the target relationship. Report regression results as a stargazer table, including coefficients, standard errors, R2, and number of observations.
Interpret the results from the regression above: what do the coefficient estimates mean? What does the R2 value indicate?
The regression results indicate that all three predictors—home runs, hits, and walks—have a statistically significant and positive relationship with the number of team wins. The coefficient for home runs is 0.023, meaning that for every additional home run a team hits, they are expected to win approximately 0.023 more games on average, holding other factors constant. While significant, this relatively small effect suggests that home runs alone are not as impactful on wins as other factors. Hits, with a coefficient of 0.038, have a greater impact, indicating that each additional hit contributes 0.038 more wins on average. Similarly, walks have a strong positive effect, with a coefficient of 0.035, showing that plate discipline and the ability to get on base play a significant role in a team’s success. The intercept of 4.44 represents the baseline number of wins a team would expect if they recorded zero home runs, hits, or walks, although this scenario is not realistic. The R-squared value of 0.470 indicates that the model explains approximately 47% of the variation in team wins, suggesting a moderate level of explanatory power. The adjusted R-squared is identical at 0.470, confirming that the predictors are meaningful and not overfitting the data. However, this also implies that 53% of the variation in wins is driven by other factors not included in the model, such as pitching performance, defensive ability, or managerial decisions. The residual standard error of 11.59 suggests that, on average, the model’s predictions deviate from the actual number of wins by about 11.6 wins. Finally, the F-statistic of 822.1, with a p-value less than 0.001, confirms that the overall model is statistically significant, meaning the predictors collectively have a strong relationship with team wins. Overall, the results emphasize the importance of consistent offensive performance, particularly through hits and walks, in contributing to team success, while home runs play a smaller but still meaningful role.
The 95% confidence intervals for the regression coefficients reveal important insights about their statistical significance. For all variables in the model (homeruns, hits, and walks), the confidence intervals do not include zero. For example, the confidence interval for homeruns ranges from 0.0148 to 0.0316, indicating that even at the lower bound, home runs have a positive impact on wins. Similarly, hits have a confidence interval ranging from 0.0349 to 0.0411, and walks have an interval of 0.0299 to 0.0406, both of which confirm their positive and statistically significant contribution to wins. The p-values for all variables are extremely small (p < 0.001), which further reinforces their statistical significance. This means we can reject the null hypothesis that these coefficients are equal to zero, with a very high level of confidence. In summary, the confidence intervals and p-values together indicate that homeruns, hits, and walks are all significant predictors of team wins, with consistent positive effects.
The F-test demonstrates that the coefficients on the independent variables (homeruns, hits, and walks) are jointly significant. This means that these variables, together, have a meaningful and statistically significant relationship with the number of team wins. In other words, they collectively explain a significant portion of the variation in wins, supporting the validity of the regression model. 4o
Yes they are normally distributed. Compared to the simple regression analysis, the inclusion of additional predictors (hits and walks) improves the distribution’s normality, reducing skewness and better explaining the variance in wins. This highlights the advantages of using multiple predictors for a more accurate model.
The multiple regression analysis demonstrates that homeruns, hits, and walks are all statistically significant predictors of team wins, with hits and walks having a greater impact than homeruns. The adjusted R² value of 0.47 indicates that the model explains 47% of the variation in team wins, which is a substantial improvement over the simple regression analysis. This highlights the importance of consistent hitting and plate discipline, as captured by hits and walks, in contributing to team success. While homeruns remain significant, their smaller coefficient suggests that power-hitting alone is less influential than broader offensive capabilities. In comparison to the earlier simple regression analysis, which likely included only homeruns as a predictor, the multiple regression analysis provides a more comprehensive understanding of team performance. The simple regression had a lower R² value, indicating that home runs alone explained less of the variation in wins. By including hits and walks, the multiple regression model reduces omitted variable bias and highlights the multidimensional nature of offensive success. These differences underscore the value of a more holistic approach in modeling team performance, capturing critical aspects that the simple regression could not.
Moving forward, the project can take several directions. One approach is expanding the model by incorporating additional variables such as pitching metrics like ERA and strikeouts, defensive stats like errors and fielding percentage, or team characteristics like roster depth. These additions would provide a fuller picture of the factors influencing wins. Another avenue is to investigate interaction effects, such as how combinations of variables influence wins, to uncover more nuanced relationships. The model could also be applied for predictive purposes, testing its accuracy on out-of-sample data, such as future seasons. Finally, the project could explore causal analysis by employing advanced techniques like instrumental variables or fixed effects better to isolate the causal relationships between predictors and team success. These next steps would deepen the insights from the analysis and enhance its practical applications for team management and strategy.
AI Statement Throughout this project, I utilized AI to assist with several aspects of the analysis and troubleshooting process. Specifically, I used AI tools to help with downloading necessary R packages and ensuring that my environment was correctly set up. AI also played a role in troubleshooting specific sections of code by analyzing screenshots of errors and providing targeted suggestions to resolve them. For example, I encountered challenges with the stargazer package, particularly in rendering properly formatted regression tables, and used AI to identify and implement alternative solutions while iterating on the code. These tools significantly streamlined the process, helping me to focus on the analysis and interpretation of results rather than getting stuck on technical issue