#----------
#Question 1
#----------
#Open packages
library(readxl)
library(rcompanion)
library(ggplot2)
#Import dataset
dataset1 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset1.xlsx")
View(dataset1)
#Create frequency table
table(dataset1$Department, dataset1$Security_Training)
##
## Completed Not Completed
## Finance 17 35
## HR 12 25
## IT 25 12
## Sales 11 13
#name frequency table
Cybersecurity_awareness_status<- table(dataset1$Department, dataset1$Security_Training)
Cybersecurity_awareness_status
##
## Completed Not Completed
## Finance 17 35
## HR 12 25
## IT 25 12
## Sales 11 13
#Create a bar chart
barplot(Cybersecurity_awareness_status,
beside = TRUE,
col = rainbow(nrow(Cybersecurity_awareness_status)),
legend = rownames(Cybersecurity_awareness_status))

#conduct the chi-square test of independence
chi_result <- chisq.test(Cybersecurity_awareness_status)
chi_result
##
## Pearson's Chi-squared test
##
## data: Cybersecurity_awareness_status
## X-squared = 13.099, df = 3, p-value = 0.004428
#Determine statistical significance: It is statistically significant
#calculate effect size
rcompanion::cramerV(Cybersecurity_awareness_status)
## Cramer V
## 0.2955
#The research question: Is there an association between an employee's department and whether they completed cybersecurity awareness training?
#The variables being analyzed; variable 1: Department, variable 2: Security_Training
#The appropriate statistical test:Chi-Squared Test of Independence
#The null hypothesis: There is no association between employee's department and whether they completed cybersecurity awareness training.
#The alternative hypothesis: There is an association between employee's department and whether they completed cybersecurity awareness training.
# Interpretation of the results:
# A Chi-Square Test of Independence was conducted to determine if there was association between Department and employee's comletion of training.
# The results showed that there was an association between the scholarship status and nationality, χ²(3) = 13.10, p = .004
# The association was moderate, (Cramer's V = .30).
#-------------------------------------------------------------------------
#-----------
# Question 2
#-----------
#Open packages
library(readxl)
library(ggpubr)
#import dataset
dataset1 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset1.xlsx")
View(dataset1)
#Create scatterplot
ggscatter(
dataset1,
x = "Security_Awareness_Score",
y = "Risky_Clicks",
add = "reg.line",
xlab = "Security Awareness Score",
ylab = "Risky Clicks"
)

#The relationship is linear.
# The relationship is negative.
# There are outliers.
#Calculate descriptive statistics
mean(dataset1$Security_Awareness_Score)
## [1] 67.09467
sd(dataset1$Security_Awareness_Score)
## [1] 13.02658
median(dataset1$Security_Awareness_Score)
## [1] 66.6
mean(dataset1$Risky_Clicks)
## [1] 3.44
sd(dataset1$Risky_Clicks)
## [1] 1.569337
median(dataset1$Risky_Clicks)
## [1] 4
#Create histograms
hist(dataset1$Security_Awareness_Score,
main = "Security Awareness Score",
breaks = 20,
col = "lightblue",
border = "white",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

hist(dataset1$Risky_Clicks,
main = "Risky Clicks ",
breaks = 20,
col = "lightcoral",
border = "white",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

# Variable 1: Security Awareness Score
# The variable looks abnormally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.
# Variable 2: Risky Clicks
# The variable looks abnormally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.
#Shapiro-WilkTest
shapiro.test(dataset1$Security_Awareness_Score)
##
## Shapiro-Wilk normality test
##
## data: dataset1$Security_Awareness_Score
## W = 0.98109, p-value = 0.0369
shapiro.test(dataset1$Risky_Clicks)
##
## Shapiro-Wilk normality test
##
## data: dataset1$Risky_Clicks
## W = 0.95167, p-value = 4.522e-05
# Variable 1: Security Awareness Score
# The variable is abnormally distributed (p = .037).
# Variable 2: Risky Click
# The variable is abnormally distributed (p < .001).
#Spearman correlation
cor.test(
dataset1$Security_Awareness_Score,
dataset1$Risky_Clicks,
method = "spearman", exact = FALSE)
##
## Spearman's rank correlation rho
##
## data: dataset1$Security_Awareness_Score and dataset1$Risky_Clicks
## S = 929970, p-value < 2.2e-16
## alternative hypothesis: true rho is not equal to 0
## sample estimates:
## rho
## -0.653354
#The research question: Is there a relationship between employees' cybersecurity awareness scores and the number of risky links they click in a simulated phishing test?
#The variables being analyzed; Independent variable: Security awareness scores, dependent variable: Risky Clicks
#The appropriate statistical test: Spearman Correlation
#The null hypothesis: There is no relationship between employees' Cybersecurity awareness scores and the number of risky links they click in a simulated phishing test.
#The alternative hypothesis: There is a relationship between employees' Cybersecurity awareness scores and the number of risky links they click in a simulated phishing test.
# Interpretation of the results:
# A Spearman correlation was conducted to test the relationship between employee's security awareness score (Mdn = 66.6) and Risky link clicks in phishing text (Mdn = 4).
# There was a statistically significant relationship between the two variables, ρ =-.65, p < .001.
# The relationship was negative and strong.
# As the Security Awareness Score increased, the Risky Clicks decreased.
#-------------------------------------------------------------------------
#-----------
# Question 3
#-----------
#Open packages
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(effectsize)
##
## Attaching package: 'effectsize'
## The following object is masked from 'package:rcompanion':
##
## phi
library(effsize)
library(readxl)
library(ggpubr)
#import dataset
dataset1 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset1.xlsx")
View(dataset1)
#Calculate descriptive statistics
dataset1 %>%
group_by(Security_Training) %>%
summarise(
Mean = mean(Risky_Clicks, na.rm = TRUE),
Median = median(Risky_Clicks, na.rm = TRUE),
SD = sd(Risky_Clicks, na.rm = TRUE),
N = n()
)
## # A tibble: 2 × 5
## Security_Training Mean Median SD N
## <chr> <dbl> <dbl> <dbl> <int>
## 1 Completed 2.77 3 1.58 65
## 2 Not Completed 3.95 4 1.36 85
#Histogram for normality check 1
hist(dataset1$Risky_Clicks[dataset1$Security_Training == "Completed"],
breaks = 15,
col = "skyblue",
border = "white")

hist(dataset1$Risky_Clicks[dataset1$Security_Training == "Not Completed"],
breaks = 15,
col = "firebrick",
border = "white")

#Data for department Completed appears normally distributed.
#Data for department not Completed appears normally distributed.
#Boxplot for normality check 2
ggboxplot(dataset1, x = "Security_Training", y = "Risky_Clicks",
color = "Security_Training",
palette = "jco",
add = "jitter")

# Data for Completed group does have outliers.
# Data for Not Completed group does have outliers.
#Shapiro test for normality check 3
shapiro.test(dataset1$Risky_Clicks[dataset1$Security_Training == "Completed"])
##
## Shapiro-Wilk normality test
##
## data: dataset1$Risky_Clicks[dataset1$Security_Training == "Completed"]
## W = 0.94191, p-value = 0.004278
shapiro.test(dataset1$Risky_Clicks[dataset1$Security_Training == "Not Completed"])
##
## Shapiro-Wilk normality test
##
## data: dataset1$Risky_Clicks[dataset1$Security_Training == "Not Completed"]
## W = 0.93788, p-value = 0.0004869
#The Completed group is abnormally distributed, (p= .004).
#The Not Completed group abnormally distributed, (p <.001).
#Conduct the Mann-Whitney U Test (Abnormal Data)
wilcox.test(Risky_Clicks ~ Security_Training, data = dataset1)
##
## Wilcoxon rank sum test with continuity correction
##
## data: Risky_Clicks by Security_Training
## W = 1631, p-value = 1.199e-05
## alternative hypothesis: true location shift is not equal to 0
#Calculate Cliff's Delta (Effect Size)
mw_effect <- cliff.delta(Risky_Clicks ~ Security_Training, data = dataset1)
print(mw_effect)
##
## Cliff's Delta
##
## delta estimate: -0.4095928 (medium)
## 95 percent confidence interval:
## lower upper
## -0.5602691 -0.2326774
#The research question: Is there a mean difference in the number of risky links clicked between employees who completed cybersecurity training and those who did not?
#The variables being analyzed;Group variable: Security Training, Outcome variable: Risky Clicks
#The appropriate statistical test:Mann-Whitney U Test
#The null hypothesis: There is no mean difference in the number of risky links clicked between employees who completed cybersecurity training and those who did not.
#The alternative hypothesis: There is a mean difference in the number of risky links clicked between employees who completed cybersecurity training and those who did not.
# Interpretation of the results:
#A Mann-Whitney U test was conducted to determine if there was a difference in Risky link clicks between employees that completed and those not Completed cybersecurity training.
#The employees that completed scores (Mdn = 3) were significantly different from those not completed scores (Mdn = 4), W = 1631, p< .001.
#The effect size was medium, Cliff's Delta = -.409.
#-------------------------------------------------------------------------
#----------
#Question 4
#----------
#Open packages
library(readxl)
library(ggpubr)
library(effsize)
library(rstatix)
##
## Attaching package: 'rstatix'
## The following objects are masked from 'package:effectsize':
##
## cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:stats':
##
## filter
#import dataset
dataset2 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset2.xlsx")
View(dataset2)
#Create groups for before and after groups
Before <- dataset2$Awareness_Score_Pre
After <- dataset2$Awareness_Score_Post
Differences <- After - Before
#Calculate descriptive statistics
mean(Before, na.rm = TRUE)
## [1] 59.088
median(Before, na.rm = TRUE)
## [1] 59.05
sd(Before, na.rm = TRUE)
## [1] 8.942849
mean(After, na.rm = TRUE)
## [1] 67.324
median(After, na.rm = TRUE)
## [1] 69.15
sd(After, na.rm = TRUE)
## [1] 10.25009
# Histogram for normality check1
hist(Differences,
breaks = 15,
col = "blue",
border = "white")

#Data for the difference scores appears abnormally distributed.
#Boxplot for normality check 2
boxplot(Differences,
main = "Awareness Score Differences (After - Before)",
ylab = "Difference in Scores",
col = "blue",
border = "darkblue")

# The difference scores boxplot does not have outliers.
#Shapiro_Wilk Test for normality check3
shapiro.test(Differences)
##
## Shapiro-Wilk normality test
##
## data: Differences
## W = 0.95901, p-value = 0.08079
#Shapiro-Wilk Difference Scores
#The data is normally distributed, (p> .05).
#Conduct Dependent T test
t.test(Before, After, paired = TRUE, na.action = na.omit)
##
## Paired t-test
##
## data: Before and After
## t = -10.733, df = 49, p-value = 1.826e-14
## alternative hypothesis: true mean difference is not equal to 0
## 95 percent confidence interval:
## -9.778075 -6.693925
## sample estimates:
## mean difference
## -8.236
#Calculate Cohen's d (Effect Size)
cohen.d(Before, After, paired = TRUE)
##
## Cohen's d
##
## d estimate: -0.8348924 (large)
## 95 percent confidence interval:
## lower upper
## -1.0141545 -0.6556303
#The research question: Is there a mean difference in employees' cybersecurity awareness scores before versus after completing the training intervention?
#The variables being analyzed; variable 1:Awareness_Score_Pre, variable 2:Awareness_Score_Post
#The appropriate statistical test:Chi-Squared Test of Dependence
#The null hypothesis: There is no mean difference in employees' cybersecurity awareness scores before versus after completing the training intervention.
#The alternative hypothesis:There is a mean difference in employees' cybersecurity awareness scores before versus after completing the training intervention.
# Interpretation of the results:
# A Dependent T-Test was conducted to determine if there was a difference in Awareness Score between Awareness Score Pre and Awareness Score Post.
# Awareness Score Pre (M = 59.09, SD = 8.94) is not significantly different from Awareness Score Post (M = 67.32, SD = 10.25), (t(49) = -10.73, p< .001)
#The effect size was large, Cohen's d = -.835.