#--------------------
# Load The Packages |
#--------------------
library(readxl)
library(rcompanion)
library(ggplot2)
library(ggpubr)
library(effsize)
library(rstatix)
##
## Attaching package: 'rstatix'
## The following object is masked from 'package:stats':
##
## filter
library(effectsize)
##
## Attaching package: 'effectsize'
## The following objects are masked from 'package:rstatix':
##
## cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:rcompanion':
##
## phi
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(ggpubr)
library(misty)
## |-------------------------------------|
## | misty 0.8.3 (2026-08-02) |
## | Miscellaneous Functions T. Yanagida |
## |-------------------------------------|
#+++++++++++++++++++++++++++++
#----------------------
# IMPORTING DATA SETS |
#----------------------
BYOD_1 <- read_excel("C:/Users/tawan/OneDrive - Saint Louis University/AA 5221/100526/School_A_BYOD_Security_Awareness_Data.xlsx")
View(BYOD_1)
BYOD_2<- read_excel("C:/Users/tawan/OneDrive - Saint Louis University/AA 5221/100526/School_A_vs_School_B_Independent_TTest_Data.xlsx")
View(BYOD_2)
#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
#---------------------
# 1. Chi-square Test |
#---------------------
#------------------------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Junior students in a school bring their own devices to school and connect to the school network
# The administration is concerned about the number of security incidences experienced since students
# were allowed to bring their devices. They realised the need to measure the security awareness of
# the students before and after providing security awareness training. After training was conducted
# there is need to determine if student found the training usefull and if it is dependent on gender.
#
# VARIABLES BEING ANALYSED: Gender, Training Usefullness (Yes/No)
#
# STATISTICAL TEST: Chi-squared Test of Independence
#
# H0: There is no relationship between Gender and finding the training useful.
# H1: There is a relationship between Gender and finding the training useful.
#------------------------------------------------------------------------------------------------------------------------
table(BYOD_1$Gender, BYOD_1$Is_Training_Useful)
##
## No Yes
## Female 10 13
## Male 4 23
Usefullness_of_Training <- table(BYOD_1$Gender, BYOD_1$Is_Training_Useful)
barplot(Usefullness_of_Training,
main = "Was Training Found Usefull",
ylab = "Count",
xlab = "Gender",
beside = TRUE,
col = rainbow(nrow(Usefullness_of_Training)),
legend = rownames(Usefullness_of_Training),
args.legend = list(x = "topright"))

#Conducting a Chi-Square Test of Independence : Determining statistical significance
chi_result <- chisq.test(Usefullness_of_Training)
chi_result
##
## Pearson's Chi-squared test with Yates' continuity correction
##
## data: Usefullness_of_Training
## X-squared = 3.7396, df = 1, p-value = 0.05314
# rcompanion::cramerV(Usefullness_of_Training)
#---------------------------------
# A Chi-Square Test of Independence was conducted to determine if there was an association between gender (male or female) and finding the training useful (yes or no).
# The results showed that there was not an association between the two variables, χ²(1) = 3.74, p > .05.
#+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
#------------------------------------
# 2. (pearson/spearman) Correlation |
#------------------------------------
#------------------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Student's Level of security awareness is measured after going through training. The training
# is self paced and students are encouraged to complete as much modules as they can to improve
# their level of security awareness. There is need to find out if hours of training impact the
# level of awareness.
# VARIABLES BEING ANALYSED: Training Hours, Awareness Improvement
#
# STATISTICAL TEST: Pearson / Spearman Correlation.
#
# H0: There is no relationship between Training Hours and Security Awareness Improvement.
# H1: There is a relationship between Training Hours and Security Awareness Improvement.
#------------------------------------------------------------------------------------------------------------------
ggscatter(
BYOD_1,
x = "Training_Hours",
y = "Awareness_Improvement",
add = "reg.line",
color = "purple",
xlab = "Hours",
ylab = "Awareness Improvement"
)

# The relationship is curved.
# The relationship is positive.
# There are no outliers.
#Calculating the Descriptive Statistics
mean(BYOD_1$Training_Hours)
## [1] 8.008
sd(BYOD_1$Training_Hours)
## [1] 2.121401
median(BYOD_1$Training_Hours)
## [1] 7.8
mean(BYOD_1$Awareness_Improvement)
## [1] 25.972
sd(BYOD_1$Awareness_Improvement)
## [1] 6.315806
median(BYOD_1$Awareness_Improvement)
## [1] 25.65
#variable 1
hist(BYOD_1$Training_Hours,
main = "Training Hours",
breaks = 20,
col = "lightblue",
border = "white",
xlab = "Hours",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

# Variable 1: Training Hours
# The variable looks normally distributed.
# The data is symmetrical.
# The data has a proper bell curve.
#variable 2
hist(BYOD_1$Awareness_Improvement,
main = "Awareness",
breaks = 20,
col = "lightcoral",
border = "white",
xlab = "Hours",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

# Variable 2: Awareness_Improvement
# The variable looks normally distributed.
# The data is symmetrical.
# The data has a proper bell curve.
# Check normality statistically (Shapiro-Wilk Test)
shapiro.test(BYOD_1$Training_Hours)
##
## Shapiro-Wilk normality test
##
## data: BYOD_1$Training_Hours
## W = 0.98428, p-value = 0.7401
shapiro.test(BYOD_1$Awareness_Improvement)
##
## Shapiro-Wilk normality test
##
## data: BYOD_1$Awareness_Improvement
## W = 0.99049, p-value = 0.957
# Variable 1: Training_Hours
# The variable is normally distributed (p > .05).
# Variable 2: Awareness_Improvement
# The variable is normally distributed (p > .05).
#Determining which correlation to use
cor.test(
BYOD_1$Training_Hours,
BYOD_1$Awareness_Improvement,
method = "pearson",
#exact = FALSE
)
##
## Pearson's product-moment correlation
##
## data: BYOD_1$Training_Hours and BYOD_1$Awareness_Improvement
## t = 4.1696, df = 48, p-value = 0.0001269
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
## 0.2770642 0.6943359
## sample estimates:
## cor
## 0.5156453
#--------------------------
# A Pearson correlation was conducted to test the relationship between Training Hours (M = 8.0, SD = 2.12) and Awareness Improvement (M = 25.97, SD = 6.32).
# There was a statistically significant relationship between the two variables, r(48) = 4.16, p < .001.
# The relationship was positive and strong.
# As Training Hours increased, Awareness Improvement increased.
#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
#--------------------------------------------------------------------------------
# 3. Paired t-test or Wilcoxon Signed-Rank / Dependent (within-subjects design) |
#--------------------------------------------------------------------------------
#----------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Is there a difference in the number of security incidences recorded before and after
# students underwent security awareness training?
# VARIABLES BEING ANALYSED: "Security Incidences Before Training", "Security Incidences Before Training"
#
# STATISTICAL TEST: Paired t-test / Wilcoxon Signed-Rank
#
# H0: There is no difference in the number of security incidences observed before training and after training.
# H1: There is a difference in the number of security incidences observed before training and after training.
#
#------------------------------------------------------------------------------------------------------------
Incidence_Before <- BYOD_1$Security_Incidents_Before
Incidence_After <- BYOD_1$Security_Incidents_After
Differences <- Incidence_Before - Incidence_After
#Descriptive statistics
mean(Incidence_Before, na.rm = TRUE)
## [1] 8.16
median(Incidence_Before, na.rm = TRUE)
## [1] 7
sd(Incidence_Before, na.rm = TRUE)
## [1] 3.053001
mean(Incidence_After, na.rm = TRUE)
## [1] 4.72
median(Incidence_After, na.rm = TRUE)
## [1] 4
sd(Incidence_After, na.rm = TRUE)
## [1] 1.959175
# Normality check 1: Histogram
hist(Differences,
breaks = 15,
col = "blue",
border = "white")

#Data for the difference scores appears normally distributed.
#Normality check 2: Boxplot
boxplot(Differences,
main = "Distribution of Incidences (After - Before)",
ylab = "Difference in Incidences",
col = "blue",
border = "darkblue")

# The difference scores boxplot does not have outliers.
#Normality check 3: Shapiro-Wilk
shapiro.test(Differences)
##
## Shapiro-Wilk normality test
##
## data: Differences
## W = 0.84885, p-value = 1.441e-05
#Shapiro-Wilk Difference Scores
#The data is abnormally distributed, (p < .05).
#Conducting the Wilcoxon Signed-Rank Test
wilcox.test(Incidence_Before, Incidence_After, paired = TRUE, na.action = na.omit)
##
## Wilcoxon signed rank test with continuity correction
##
## data: Incidence_Before and Incidence_After
## V = 1275, p-value = 5.871e-10
## alternative hypothesis: true location shift is not equal to 0
#Calculating the effect size
df_long <- data.frame(id = rep(1:length(Incidence_Before), 2), time = rep(c("Incidence_Before", "Incidence_After"), each = length(Incidence_Before)), score = c(Incidence_Before, Incidence_After))
wilcox_effsize(df_long, score ~ time, paired = TRUE)
## # A tibble: 1 × 7
## .y. group1 group2 effsize n1 n2 magnitude
## * <chr> <chr> <chr> <dbl> <int> <int> <ord>
## 1 score Incidence_After Incidence_Before 0.877 50 50 large
#A Wilcoxon Signed-Rank Test was conducted to determine if there was a difference in Security Incidences before Training versus after Training.
#Before scores (Mdn = 8.16) were significantly different from after scores (Mdn = 4), V = 1275, p < .001.
#The effect size was large, r = .87.
#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
#----------------------------------------------
# 4. Independent T-Test & Mann-Whitney U Test |
#----------------------------------------------
#----------------------------------------------------------------------------------------------------------------
# RESEARCH QUESTION : Is there a difference in the security level awareness between a Group of students
# that has received training and a separete group of students that has not received training.
# VARIABLES BEING ANALYSED: "Training Status", "Security Awareness Score"
#
# STATISTICAL TEST: Independent T-Test & Mann-Whitney U Test
#
# H0: There is no difference in the security awareness between a trained group of students and untrained group.
# H1: There is a difference in the security awareness between a trained group of students and untrained group.
#----------------------------------------------------------------------------------------------------------------
BYOD_2 %>%
group_by(Training_Status) %>%
summarise(
Mean = mean(Security_Awareness_Score, na.rm = TRUE),
Median = median(Security_Awareness_Score, na.rm = TRUE),
SD = sd(Security_Awareness_Score, na.rm = TRUE),
N = n()
)
## # A tibble: 2 × 5
## Training_Status Mean Median SD N
## <chr> <dbl> <dbl> <dbl> <int>
## 1 Not_Trained 53.8 53.4 8.57 50
## 2 Trained 78.8 77.8 7.64 50
# Normality Check 1: Histograms
hist(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Trained"],
breaks = 15,
main = "Histogram of Awareness For School A Students",
xlab = "Trained",
col = "skyblue",
border = "white")

hist(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Not_Trained"],
breaks = 15,
main = "Histogram of Awareness For School B Students",
xlab = "Not_Trained",
col = "firebrick",
border = "white")

#Data for Trained students appears normally distributed.
#Data for Untrained students appears normally distributed.
#Normality Check 2: Boxplots
ggboxplot(BYOD_2, x = "Training_Status", y = "Security_Awareness_Score",
color = "green",
palette = "jco",
add = "jitter")

# The Trained students boxplot have outliers.
# The Untrained students boxplot has outliers.
# Normality Check 3: Shapiro-Wilk
shapiro.test(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Trained"])
##
## Shapiro-Wilk normality test
##
## data: BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Trained"]
## W = 0.98838, p-value = 0.9013
shapiro.test(BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Not_Trained"])
##
## Shapiro-Wilk normality test
##
## data: BYOD_2$Security_Awareness_Score[BYOD_2$Training_Status == "Not_Trained"]
## W = 0.96148, p-value = 0.1026
#The Trained students group is normally distributed, (p > .05).
#The Untrained students group is normally distributed, (p > .05).
#Conduct the Independent T-Test (Normal Data)
t.test(Security_Awareness_Score ~ Training_Status, data = BYOD_2)
##
## Welch Two Sample t-test
##
## data: Security_Awareness_Score by Training_Status
## t = -15.345, df = 96.725, p-value < 2.2e-16
## alternative hypothesis: true difference in means between group Not_Trained and group Trained is not equal to 0
## 95 percent confidence interval:
## -28.1388 -21.6932
## sample estimates:
## mean in group Not_Trained mean in group Trained
## 53.836 78.752
cohens_d_result <- cohens.d(Security_Awareness_Score ~ Training_Status, data = BYOD_2, pooled_sd = TRUE)
## Cohen's d: Two-Sample Design with Two-Sided 95% CI
##
## Between n nNA M M.Diff SD d SE Low Upp
## Not_Trained 50 0 53.84
## Trained 50 0 78.75 24.92 8.12 3.07 1.62 2.48 3.65
##
## Note. SD = weighted pooled standard deviation
print(cohens_d_result)
## Cohen's d: Two-Sample Design with Two-Sided 95% CI
##
## Between n nNA M M.Diff SD d SE Low Upp
## Not_Trained 50 0 53.84
## Trained 50 0 78.75 24.92 8.12 3.07 1.62 2.48 3.65
##
## Note. SD = weighted pooled standard deviation
# An Independent T-Test was conducted to determine if there was a difference in Security Awareness between Trained and Untrained students.
# Trained students scores (M = 78.8, SD = 7.64) were significantly different from Untrained students scores (M = 53.8, SD = 8.57), t(98) = 15.34, p < .001.
# The effect size was medium, Cohen's d = 3.07.
#++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++