library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(effectsize)
library(effsize)
library(readxl)
library(ggpubr)
## Loading required package: ggplot2
A6Q4 <- read_excel("C:/Users/nehab/OneDrive/A5221/Assignment--6/A6Q4.xlsx")

# Descriptive statistics for each weightlifting group
A6Q4 %>%
  group_by(Exercise) %>%
  summarise(
    Mean = mean(Weight, na.rm = TRUE),
    Median = median(Weight, na.rm = TRUE),
    SD = sd(Weight, na.rm = TRUE),
    N = n()
  )
## # A tibble: 2 × 5
##   Exercise  Mean Median    SD     N
##   <chr>    <dbl>  <dbl> <dbl> <int>
## 1 lift     120.   116.   53.3    25
## 2 nolift    33.0   40.8  56.7    25
# I corrected the descriptive-statistics code.
# I combined the mean, median, standard deviation, and sample size
# into one summary table and added na.rm = TRUE.

# Examine the no-lifting distribution
hist(
  A6Q4$Weight[A6Q4$Exercise == "nolift"],
  breaks = 15,
  col = "skyblue",
  border = "white",
  main = "Weight Distribution: No Weightlifting",
  xlab = "Body Weight"
)

# Examine the weightlifting distribution
hist(
  A6Q4$Weight[A6Q4$Exercise == "lift"],
  breaks = 15,
  col = "firebrick",
  border = "white",
  main = "Weight Distribution: Weightlifting",
  xlab = "Body Weight"
)

# Data for the no-lifting group appears abnormally distributed.
# Data for the weightlifting group appears abnormally distributed.
# I added separate histograms for both groups as shown in the answer key.

# Create the boxplot
ggboxplot(
  A6Q4,
  x = "Exercise",
  y = "Weight",
  color = "Exercise",
  palette = "jco",
  add = "jitter"
)

# The no-lifting group has outliers.
# The weightlifting group has outliers.
# I replaced the original basic boxplot with the answer-key boxplot.

# Test normality separately for both groups
shapiro.test(
  A6Q4$Weight[A6Q4$Exercise == "nolift"]
)
## 
##  Shapiro-Wilk normality test
## 
## data:  A6Q4$Weight[A6Q4$Exercise == "nolift"]
## W = 0.70002, p-value = 7.294e-06
shapiro.test(
  A6Q4$Weight[A6Q4$Exercise == "lift"]
)
## 
##  Shapiro-Wilk normality test
## 
## data:  A6Q4$Weight[A6Q4$Exercise == "lift"]
## W = 0.78786, p-value = 0.0001436
# Both groups are abnormally distributed because both p-values are below .05.
# Therefore, a Mann-Whitney U Test will be used.

# Conduct the Mann-Whitney U Test
wilcox.test(
  Weight ~ Exercise,
  data = A6Q4
)
## 
##  Wilcoxon rank sum exact test
## 
## data:  Weight by Exercise
## W = 603, p-value = 7.132e-11
## alternative hypothesis: true location shift is not equal to 0
# Calculate Cliff's Delta effect size
mw_effect <- cliff.delta(
  Weight ~ Exercise,
  data = A6Q4
)

print(mw_effect)
## 
## Cliff's Delta
## 
## delta estimate: 0.9296 (large)
## 95 percent confidence interval:
##     lower     upper 
## 0.7993841 0.9764036
# I corrected the effect-size calculation.
# I originally used wilcoxonR(), which reported r = .80.
# I used cliff.delta() from the answer key, which reports
# a large effect with an absolute Cliff's Delta of .930.

Interpretation

A Mann-Whitney U Test was conducted to determine whether there was a difference in body weight between participants who regularly lift weights and participants who do not lift weights. Participants who lift weights (Mdn = 115.59) had significantly different body weight than participants who do not lift weights (Mdn = 40.84), W = 603, p < .001. The effect size was large, with an absolute Cliff’s Delta of .930. Therefore, the null hypothesis was rejected.

I corrected the descriptive statistics, added separate histograms, updated the boxplot, and replaced the original effect-size calculation with Cliff’s Delta from the answer key.