library(readxl)
library(ggpubr)
## Loading required package: ggplot2
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(effectsize)
library(effsize)
library(rstatix)
## 
## Attaching package: 'rstatix'
## The following objects are masked from 'package:effectsize':
## 
##     cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:stats':
## 
##     filter
Between_Subjects <- read_excel("C:/Users/USER/OneDrive - Saint Louis University/AA 5221/Final Project/Between Subjects.xlsx")
View(Between_Subjects)

Within_Subjects <- read_excel("C:/Users/USER/OneDrive - Saint Louis University/AA 5221/Final Project/Within Subjects.xlsx")
View(Within_Subjects)

#Research Question 1: Is there an association between students' gender and whether they regularly use AI tools?
#Null hypothesis: There is no association between students' gender and regularly using AI tools. 
#Alternative hypothesis: There is an association between students' gender and regularly using AI tools. 

#Chi-Square Analysis 

#frequency table

observed_F <- table(Between_Subjects$Gender)
observed_F
## 
## Female   Male 
##     78     72
#Bar Chart

barplot(observed_F,
        main = "Gender",
        xlab = "Gender",
        ylab = "Count",
        col = rainbow(length(observed_F)))

expectedF <- c(.50, .50)

#Chi-Square goodness of fit test

Chi_results <- chisq.test(x=observed_F, p=expectedF)
Chi_results
## 
##  Chi-squared test for given probabilities
## 
## data:  observed_F
## X-squared = 0.24, df = 1, p-value = 0.6242
#A Chi-Square Goodness-of-Fit test was conducted to determine if there was a difference between the observed gender frequencies and the expected frequencies. 
# The results showed that there was no significant difference, χ²(1) = 0.24, p > .05. 


#Research Question 2: Is there a relationship between the number of hours students spend using AI tools for studying per week and their assignment scores?
#Null Hypothesis: There is no relationship between the number of hours students spend using AI tools for studying per week and their assignment scores.
#Alternative hypothesis: There is a relationship between the number of hours students spend using AI tools for studying per week and their assignment scores.

#Pearson correlation or Spearman correlation

ggscatter(
  Between_Subjects,
  x = "AI_Study_Hours",
  y = "Assignment_Score_After",
  add = "reg.line",
  xlab = "AI_Study_Hours",
  ylab = "Assignment_Score_After"
)

# The relationship is linear.
# The relationship is positive.
# There are no outliers.

#Descriptive statistics
mean(Between_Subjects$AI_Study_Hours)
## [1] 3.690667
median(Between_Subjects$AI_Study_Hours)
## [1] 3.35
sd(Between_Subjects$AI_Study_Hours)
## [1] 2.801566
mean(Between_Subjects$Assignment_Score_After)
## [1] 76.04
sd(Between_Subjects$Assignment_Score_After)
## [1] 7.150238
median(Between_Subjects$Assignment_Score_After)
## [1] 75
#Histograms
#For AI Study Hours

hist(Between_Subjects$AI_Study_Hours,
     main = "AI_Study_Hours",
     breaks = 20,
     col = "lightblue",
     border = "white",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

#For Assignment Score After

hist(Between_Subjects$Assignment_Score_After,
     main = "Assignment_Score_After",
     breaks = 20,
     col = "lightcoral",
     border = "white",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

#Histograms Interpretations 

# Variable 1: AI_Study_Hours
# The AI_Study_Hours looks abnormally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.

# Variable 2: Assignment_Score_After
# The Assignment_Score_After looks normally distributed.
# The data is Symmetrical.
# The data has a proper bell curve.

#Shapiro Wilk test
shapiro.test(Between_Subjects$AI_Study_Hours)
## 
##  Shapiro-Wilk normality test
## 
## data:  Between_Subjects$AI_Study_Hours
## W = 0.89975, p-value = 1.281e-08
shapiro.test(Between_Subjects$Assignment_Score_After)
## 
##  Shapiro-Wilk normality test
## 
## data:  Between_Subjects$Assignment_Score_After
## W = 0.98333, p-value = 0.06655
#Shapiro Wilk Test Interpretations

# Variable 1: AI_Study_Hours
# AI_Study_Hours is abnormally distributed (p < .001).

# Variable 2: Assignment_Score_After
# Assignment_Score_After is normally distributed (p > .05).

#Conduct a Spearman Correlation

cor.test(
  Between_Subjects$AI_Study_Hours,
  Between_Subjects$Assignment_Score_After,
  method = "spearman",
  exact = FALSE
)
## 
##  Spearman's rank correlation rho
## 
## data:  Between_Subjects$AI_Study_Hours and Between_Subjects$Assignment_Score_After
## S = 82888, p-value < 2.2e-16
## alternative hypothesis: true rho is not equal to 0
## sample estimates:
##       rho 
## 0.8526378
# A Spearman correlation was conducted to test the relationship between AI Study Hours (Mdn = 3.35, SD = 2.80) and Assignment Score After (Mdn = 75.00, SD = 7.15).

# There was a statistically significant relationship between the two variables, rho = 0.85, p < .001.

# The relationship was positive and strong.

# As AI_Study_Hours increased, Assignment_Score_After increased.


#Research Question 3:Is there a difference in mean assignment scores between students who regularly use AI tools and those who do not? 
#Null hypothesis:There is no difference in mean assignment scores between students who regularly use AI tools and those who do not. 
#Alternative hypothesis:There is a difference in mean assignment scores between students who regularly use AI tools and those who do not.

#Independent t-test or Mann-Whitney U

#Descriptive Statistics

Between_Subjects %>%
  group_by(Gender) %>%
  summarise(
    Mean = mean(Assignment_Score_After, na.rm = TRUE),
    Median = median(Assignment_Score_After, na.rm = TRUE),
    SD = sd(Assignment_Score_After, na.rm = TRUE),
    N = n()
  )
## # A tibble: 2 × 5
##   Gender  Mean Median    SD     N
##   <chr>  <dbl>  <dbl> <dbl> <int>
## 1 Female  76.2   75    7.09    78
## 2 Male    75.9   75.5  7.27    72
#Histograms

hist(Between_Subjects$Assignment_Score_After[Between_Subjects$Gender == "Female"],
     breaks = 15,
     col = "skyblue",
     border = "white")

hist(Between_Subjects$Assignment_Score_After[Between_Subjects$Gender == "Male"],
     breaks = 15,
     col = "firebrick",
     border = "white")

#Data for Female appears abnormally distributed.
#Data for Male appears abnormally distributed.

#Boxplots

ggboxplot(Between_Subjects, x = "Gender", y = "Assignment_Score_After",
          color = "Gender",
          palette = "jco",
          add = "jitter")

# The Female boxplot does not have outliers.
# The Male boxplot does not have outliers.

#Shapiro Wilk Normality Test

shapiro.test(Between_Subjects$Assignment_Score_After[Between_Subjects$Gender == "Female"])
## 
##  Shapiro-Wilk normality test
## 
## data:  Between_Subjects$Assignment_Score_After[Between_Subjects$Gender == "Female"]
## W = 0.97918, p-value = 0.2308
shapiro.test(Between_Subjects$Assignment_Score_After[Between_Subjects$Gender == "Male"])
## 
##  Shapiro-Wilk normality test
## 
## data:  Between_Subjects$Assignment_Score_After[Between_Subjects$Gender == "Male"]
## W = 0.98016, p-value = 0.3138
#The Female group is normally distributed, (p > .05).
#The Male group is normally distributed, (p > .05).

#Conduct the Independent T-Test

t.test(Assignment_Score_After ~ Gender, data = Between_Subjects, var.equal = TRUE)
## 
##  Two Sample t-test
## 
## data:  Assignment_Score_After by Gender
## t = 0.20231, df = 148, p-value = 0.84
## alternative hypothesis: true difference in means between group Female and group Male is not equal to 0
## 95 percent confidence interval:
##  -2.079515  2.553874
## sample estimates:
## mean in group Female   mean in group Male 
##             76.15385             75.91667
#Cohen's d (Effect Size)

cohens_d_result <- cohens_d(Assignment_Score_After ~ Gender, data = Between_Subjects,)
print(cohens_d_result)
## # A tibble: 1 × 7
##   .y.                    group1 group2 effsize    n1    n2 magnitude 
## * <chr>                  <chr>  <chr>    <dbl> <int> <int> <ord>     
## 1 Assignment_Score_After Female Male    0.0330    78    72 negligible
#An Independent T-Test was conducted to determine if there was a difference in Assignment Score between Female and Male.
#Female score (M = 76.20, SD = 7.09) were significantly different from Male score (M = 75.90, SD = 7.27), t(148) = 0.20, p > .05.
#The effect size was medium, Cohen's d = .52.


#Research Question 4:Is there a mean difference in students' assignment scores before versus after using an AI writing tool? 
#Null Hypothesis: There is no mean difference in students' assignment scores before and after using an AI writing tool.
#Alternative hypothesis: There is a mean difference in students' assignment scores before and after using an AI writing tool.

#Dependent t-test or Wilcoxon Signed-Rank test

Before_A <- Within_Subjects$Assignment_Score_Before
After_A <- Within_Subjects$Assignment_Score_After

Differences <- After_A - Before_A

#Descriptive Statistics

mean(Before_A, na.rm = TRUE)
## [1] 74.62667
median(Before_A, na.rm = TRUE)
## [1] 74
sd(Before_A, na.rm = TRUE)
## [1] 6.540064
mean(After_A, na.rm = TRUE)
## [1] 81.8
median(After_A, na.rm = TRUE)
## [1] 83
sd(After_A, na.rm = TRUE)
## [1] 6.794194
#Histogram

hist(Differences,
     breaks = 15,
     col = "blue",
     border = "white")

#Data for the difference scores appears normally distributed.

#Boxplot

boxplot(Differences,
        main = "Distribution of Score Differences (After_A - Before_A)",
        ylab = "Difference in Scores",
        col = "blue",
        border = "darkblue")

# The difference scores boxplot has one outlier.

#Shapiro Wilk Test 
shapiro.test(Differences)
## 
##  Shapiro-Wilk normality test
## 
## data:  Differences
## W = 0.97346, p-value = 0.005296
#Shapiro-Wilk Difference Scores
#The data is abnormally distributed, (p < .05).


#Conduct the Wilcoxon Signed-Rank Test

wilcox.test(Before_A, After_A, paired = TRUE, na.action = na.omit)
## 
##  Wilcoxon signed rank test with continuity correction
## 
## data:  Before_A and After_A
## V = 0, p-value < 2.2e-16
## alternative hypothesis: true location shift is not equal to 0
#Effect Size

df_long <- data.frame(id = rep(1:length(Before_A), 2), time = rep(c("Before_A", "After_A"), each = length(Before_A)), score = c(Before_A, After_A))

wilcox_effsize(df_long, score ~ time, paired = TRUE)
## # A tibble: 1 × 7
##   .y.   group1  group2   effsize    n1    n2 magnitude
## * <chr> <chr>   <chr>      <dbl> <int> <int> <ord>    
## 1 score After_A Before_A   0.870   150   150 large
#A Wilcoxon Signed-Rank Test was conducted to determine if there was a difference in Assignment Score before Using AI writing tool versus after Using AI writing tool.
#Before scores (Mdn = 74.00) were significantly different from after scores (Mdn = 83.00), V = 0, p < .001.
#The effect size was large, r = .87.