#--------------------
# Load The Packages |
#--------------------
library(readxl)
library(rcompanion)
library(ggplot2)
library(ggpubr)
library(effsize)
library(rstatix)
## 
## Attaching package: 'rstatix'
## The following object is masked from 'package:stats':
## 
##     filter
library(effectsize)
## 
## Attaching package: 'effectsize'
## The following objects are masked from 'package:rstatix':
## 
##     cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:rcompanion':
## 
##     phi
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(ggpubr)
library(misty)
## |-------------------------------------|
## | misty 0.8.3 (2026-08-02)            |
## | Miscellaneous Functions T. Yanagida |
## |-------------------------------------|
#+++++++++++++++++++++++++++++

#----------------------
# IMPORTING DATA SETS |
#----------------------
BYOD_1 <- read_excel("C:/Users/tawan/OneDrive - Saint Louis University/AA 5221/100526/School_A_BYOD_Security_Awareness_Data.xlsx")
View(BYOD_1)
BYOD_2<- read_excel("C:/Users/tawan/OneDrive - Saint Louis University/AA 5221/100526/School_A_vs_School_B_Independent_TTest_Data.xlsx")
View(BYOD_2)
#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++

#---------------------
# 1. Chi-square Test |
#---------------------

#------------------------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Junior students in a school bring their own devices to school and connect to the school network
#                    The administration is concerned about the number of security incidences experienced since students 
#                    were allowed to bring their devices. They realised the need to measure the security awareness of
#                    the students before and after providing security awareness training. After training was conducted
#                    there is need to determine if student found the training usefull and if it is dependent on gender.
#
# VARIABLES BEING ANALYSED: Gender, Training Usefullness (Yes/No)
#
# STATISTICAL TEST: Chi-squared Test of Independence
#
# H0: There is no relationship between Gender and finding the training useful.
# H1: There is a relationship between Gender and finding the training useful.
#------------------------------------------------------------------------------------------------------------------------

table(BYOD_1$Gender, BYOD_1$Is_Training_Useful)
##         
##          No Yes
##   Female 10  13
##   Male    4  23
Usefullness_of_Training <- table(BYOD_1$Gender, BYOD_1$Is_Training_Useful)

barplot(Usefullness_of_Training,
        main = "Was Training Found Usefull",
        ylab = "Count",
        xlab = "Gender",
        beside = TRUE,
        col = rainbow(nrow(Usefullness_of_Training)),
        legend = rownames(Usefullness_of_Training),
        args.legend = list(x = "topright"))

#Conducting a Chi-Square Test of Independence : Determining statistical significance
chi_result <- chisq.test(Usefullness_of_Training)
chi_result
## 
##  Pearson's Chi-squared test with Yates' continuity correction
## 
## data:  Usefullness_of_Training
## X-squared = 3.7396, df = 1, p-value = 0.05314
# rcompanion::cramerV(Usefullness_of_Training)

#---------------------------------
# A Chi-Square Test of Independence was conducted to determine if there was an association between gender (male or female) and finding the training useful (yes or no).
# The results showed that there was not an association between the two variables, χ²(1) = 3.74, p > .05.
#+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++


#------------------------------------
# 2. (pearson/spearman) Correlation |
#------------------------------------

#------------------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Student's Level of security awareness is measured after going through training. The training
#                     is self paced and students are encouraged to complete as much modules as they can to improve
#                     their level of security awareness. There is need to find out if hours of training impact the
#                     level of awareness.
# VARIABLES BEING ANALYSED: Training Hours, Awareness Improvement
#
# STATISTICAL TEST: Pearson / Spearman Correlation.
#
# H0: There is no relationship between Training Hours and Security Awareness Improvement.
# H1: There is a relationship between Training Hours and Security Awareness Improvement.
#------------------------------------------------------------------------------------------------------------------

ggscatter(
  BYOD_1,
  x = "Training_Hours",
  y = "Awareness_Improvement",
  add = "reg.line",
  color = "purple",
  xlab = "Hours",
  ylab = "Awareness Improvement"
)

# The relationship is curved.
# The relationship is positive.
# There are no outliers.

#Calculating the Descriptive Statistics
mean(BYOD_1$Training_Hours)
## [1] 8.008
sd(BYOD_1$Training_Hours)
## [1] 2.121401
median(BYOD_1$Training_Hours)
## [1] 7.8
mean(BYOD_1$Awareness_Improvement)
## [1] 25.972
sd(BYOD_1$Awareness_Improvement)
## [1] 6.315806
median(BYOD_1$Awareness_Improvement)
## [1] 25.65
#variable 1
hist(BYOD_1$Training_Hours,
     main = "Training Hours",
     breaks = 20,
     col = "lightblue",
     border = "white",
     xlab = "Hours",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

# Variable 1: Training Hours
# The variable looks normally distributed.
# The data is symmetrical.
# The data has a proper bell curve.

#variable 2
hist(BYOD_1$Awareness_Improvement,
     main = "Awareness",
     breaks = 20,
     col = "lightcoral",
     border = "white",
     xlab = "Hours",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

# Variable 2: Awareness_Improvement
# The variable looks normally distributed.
# The data is symmetrical.
# The data has a proper bell curve.

# Check normality statistically (Shapiro-Wilk Test)
shapiro.test(BYOD_1$Training_Hours)
## 
##  Shapiro-Wilk normality test
## 
## data:  BYOD_1$Training_Hours
## W = 0.98428, p-value = 0.7401
shapiro.test(BYOD_1$Awareness_Improvement)
## 
##  Shapiro-Wilk normality test
## 
## data:  BYOD_1$Awareness_Improvement
## W = 0.99049, p-value = 0.957
# Variable 1: Training_Hours
# The variable is normally distributed (p > .05).

# Variable 2: Awareness_Improvement
# The variable is normally distributed (p > .05).

#Determining which correlation to use
cor.test(
  BYOD_1$Training_Hours,
  BYOD_1$Awareness_Improvement,
  method = "pearson",
  #exact = FALSE
)
## 
##  Pearson's product-moment correlation
## 
## data:  BYOD_1$Training_Hours and BYOD_1$Awareness_Improvement
## t = 4.1696, df = 48, p-value = 0.0001269
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
##  0.2770642 0.6943359
## sample estimates:
##       cor 
## 0.5156453
#--------------------------
# A Pearson correlation was conducted to test the relationship between Training Hours (M = 8.0, SD = 2.12) and Awareness Improvement (M = 25.97, SD = 6.32).

# There was a statistically significant relationship between the two variables, r(48) = 4.16, p < .001.

# The relationship was positive and strong.

# As Training Hours increased, Awareness Improvement increased.

#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++


#--------------------------------------------------------------------------------
# 3. Paired t-test or Wilcoxon Signed-Rank / Dependent (within-subjects design) |
#--------------------------------------------------------------------------------

#----------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Is there a difference in the number of security incidences recorded before and after 
#                     students underwent security awareness training?
# VARIABLES BEING ANALYSED: "Security Incidences Before Training", "Security Incidences Before Training"
#
# STATISTICAL TEST: Paired t-test / Wilcoxon Signed-Rank
#
# H0: There is no difference in the number of security incidences observed before training and after training.
# H1: There is a difference in the number of security incidences observed before training and after training.
#
#------------------------------------------------------------------------------------------------------------

Incidence_Before <- BYOD_1$Security_Incidents_Before
Incidence_After <- BYOD_1$Security_Incidents_After

Differences <- Incidence_Before - Incidence_After

#Descriptive statistics
mean(Incidence_Before, na.rm = TRUE)
## [1] 8.16
median(Incidence_Before, na.rm = TRUE)
## [1] 7
sd(Incidence_Before, na.rm = TRUE)
## [1] 3.053001
mean(Incidence_After, na.rm = TRUE)
## [1] 4.72
median(Incidence_After, na.rm = TRUE)
## [1] 4
sd(Incidence_After, na.rm = TRUE)
## [1] 1.959175
# Normality check 1: Histogram
hist(Differences,
     breaks = 15,
     col = "blue",
     border = "white")

#Data for the difference scores appears normally distributed.

#Normality check 2: Boxplot
boxplot(Differences,
        main = "Distribution of Incidences (After - Before)",
        ylab = "Difference in Incidences",
        col = "blue",
        border = "darkblue")

# The difference scores boxplot does not have outliers.

#Normality check 3: Shapiro-Wilk
shapiro.test(Differences)
## 
##  Shapiro-Wilk normality test
## 
## data:  Differences
## W = 0.84885, p-value = 1.441e-05
#Shapiro-Wilk Difference Scores
#The data is abnormally distributed, (p < .05).

#Conducting the Wilcoxon Signed-Rank Test
wilcox.test(Incidence_Before, Incidence_After, paired = TRUE, na.action = na.omit)
## 
##  Wilcoxon signed rank test with continuity correction
## 
## data:  Incidence_Before and Incidence_After
## V = 1275, p-value = 5.871e-10
## alternative hypothesis: true location shift is not equal to 0
#Calculating the effect size
df_long <- data.frame(id = rep(1:length(Incidence_Before), 2), time = rep(c("Incidence_Before", "Incidence_After"), each = length(Incidence_Before)), score = c(Incidence_Before, Incidence_After))

wilcox_effsize(df_long, score ~ time, paired = TRUE)
## # A tibble: 1 × 7
##   .y.   group1          group2           effsize    n1    n2 magnitude
## * <chr> <chr>           <chr>              <dbl> <int> <int> <ord>    
## 1 score Incidence_After Incidence_Before   0.877    50    50 large
#A Wilcoxon Signed-Rank Test was conducted to determine if there was a difference in Security Incidences before Training versus after Training.
#Before scores (Mdn = 8.16) were significantly different from after scores (Mdn = 4), V = 1275, p < .001.
#The effect size was large, r = .87.
#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++


#----------------------------------------------
# 4. Independent T-Test & Mann-Whitney U Test |
#----------------------------------------------

#----------------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Is there a difference in the security level awareness between a Group of students 
#                     that has received training and a separete group of students that has not received training.
# VARIABLES BEING ANALYSED: "Training Status", "Security Awareness Score"
#
# STATISTICAL TEST: Independent T-Test & Mann-Whitney U Test
#
# H0: There is no difference in the security awareness between a trained group of students and untrained group.
# H1: There is a difference in the security awareness between a trained group of students and untrained group.
#----------------------------------------------------------------------------------------------------------------

BYOD_2 %>%
  group_by(Training_Status) %>%
  summarise(
    Mean = mean(Security_Awareness_Score, na.rm = TRUE),
    Median = median(Security_Awareness_Score, na.rm = TRUE),
    SD = sd(Security_Awareness_Score, na.rm = TRUE),
    N = n()
  )
## # A tibble: 2 × 5
##   Training_Status  Mean Median    SD     N
##   <chr>           <dbl>  <dbl> <dbl> <int>
## 1 Not_Trained      53.8   53.4  8.57    50
## 2 Trained          78.8   77.8  7.64    50
# Normality Check 1: Histograms
hist(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Trained"],
     breaks = 15,
     main = "Histogram of Awareness For School A Students",
     xlab = "Trained",
     col = "skyblue",
     border = "white")

hist(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Not_Trained"],
     breaks = 15,
     main = "Histogram of Awareness For School B Students",
     xlab = "Not_Trained",
     col = "firebrick",
     border = "white")

#Data for Trained students appears normally distributed.
#Data for Untrained students appears normally distributed.

#Normality Check 2: Boxplots
ggboxplot(BYOD_2, x = "Training_Status", y = "Security_Awareness_Score",
          color = "green",
          palette = "jco",
          add = "jitter")

# The Trained students boxplot have outliers.
# The Untrained students boxplot has outliers.

# Normality Check 3: Shapiro-Wilk
shapiro.test(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Trained"])
## 
##  Shapiro-Wilk normality test
## 
## data:  BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Trained"]
## W = 0.98838, p-value = 0.9013
shapiro.test(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Not_Trained"])
## 
##  Shapiro-Wilk normality test
## 
## data:  BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Not_Trained"]
## W = 0.96148, p-value = 0.1026
#The Trained students group is normally distributed, (p > .05).
#The Untrained students group is normally distributed, (p > .05).

#Conduct the Independent T-Test (Normal Data)
t.test(Security_Awareness_Score ~ Training_Status, data = BYOD_2)
## 
##  Welch Two Sample t-test
## 
## data:  Security_Awareness_Score by Training_Status
## t = -15.345, df = 96.725, p-value < 2.2e-16
## alternative hypothesis: true difference in means between group Not_Trained and group Trained is not equal to 0
## 95 percent confidence interval:
##  -28.1388 -21.6932
## sample estimates:
## mean in group Not_Trained     mean in group Trained 
##                    53.836                    78.752
cohens_d_result <- cohens.d(Security_Awareness_Score ~ Training_Status, data = BYOD_2, pooled_sd = TRUE)
##  Cohen's d: Two-Sample Design with Two-Sided 95% CI
## 
##   Between       n nNA     M M.Diff   SD    d   SE  Low  Upp
##    Not_Trained 50   0 53.84                                
##    Trained     50   0 78.75  24.92 8.12 3.07 1.62 2.48 3.65
## 
##   Note. SD = weighted pooled standard deviation
print(cohens_d_result)
##  Cohen's d: Two-Sample Design with Two-Sided 95% CI
## 
##   Between       n nNA     M M.Diff   SD    d   SE  Low  Upp
##    Not_Trained 50   0 53.84                                
##    Trained     50   0 78.75  24.92 8.12 3.07 1.62 2.48 3.65
## 
##   Note. SD = weighted pooled standard deviation
# An Independent T-Test was conducted to determine if there was a difference in Security Awareness between Trained and Untrained students.
# Trained students scores (M = 78.8, SD = 7.64) were significantly different from Untrained students scores (M = 53.8, SD = 8.57), t(98) = 15.34, p < .001.
# The effect size was medium, Cohen's d = 3.07.

#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++