#----------
#Question 1
#----------

#Open packages
library(readxl)
library(rcompanion)
library(ggplot2)

#Import dataset
dataset1 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset1.xlsx")
View(dataset1)

#Create frequency table
table(dataset1$Department, dataset1$Security_Training)
##          
##           Completed Not Completed
##   Finance        17            35
##   HR             12            25
##   IT             25            12
##   Sales          11            13
#name frequency table
Cybersecurity_awareness_status<- table(dataset1$Department, dataset1$Security_Training)
Cybersecurity_awareness_status
##          
##           Completed Not Completed
##   Finance        17            35
##   HR             12            25
##   IT             25            12
##   Sales          11            13
#Create a bar chart
barplot(Cybersecurity_awareness_status,
        beside = TRUE,
        col = rainbow(nrow(Cybersecurity_awareness_status)),
        legend = rownames(Cybersecurity_awareness_status))

#conduct the chi-square test of independence
chi_result <- chisq.test(Cybersecurity_awareness_status)
chi_result
## 
##  Pearson's Chi-squared test
## 
## data:  Cybersecurity_awareness_status
## X-squared = 13.099, df = 3, p-value = 0.004428
#Determine statistical significance: It is statistically significant

#calculate effect size
rcompanion::cramerV(Cybersecurity_awareness_status)
## Cramer V 
##   0.2955
#The research question: Is there an association between an employee's department and whether they completed cybersecurity awareness training?

#The variables being analyzed; variable 1: Department, variable 2: Security_Training

#The appropriate statistical test:Chi-Squared Test of Independence

#The null hypothesis: There is no association between employee's department and whether they completed cybersecurity awareness training.

#The alternative hypothesis: There is an association between employee's department and whether they completed cybersecurity awareness training.


# Interpretation of the results:
# A Chi-Square Test of Independence was conducted to determine if there was association between Department and employee's comletion of training.
# The results showed that there was an association between the scholarship status and nationality, χ²(3) = 13.10, p = .004
# The association was moderate, (Cramer's V = .30).
#-------------------------------------------------------------------------

#-----------
# Question 2
#-----------

#Open packages
library(readxl)
library(ggpubr)

#import dataset
dataset1 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset1.xlsx")
View(dataset1)

#Create scatterplot
ggscatter(
  dataset1,
  x = "Security_Awareness_Score",
  y = "Risky_Clicks",
  add = "reg.line",
  xlab = "Security Awareness Score",
  ylab = "Risky Clicks"
)

#The relationship is linear.
# The relationship is negative.
# There are  outliers.

#Calculate descriptive statistics
mean(dataset1$Security_Awareness_Score)
## [1] 67.09467
sd(dataset1$Security_Awareness_Score)
## [1] 13.02658
median(dataset1$Security_Awareness_Score)
## [1] 66.6
mean(dataset1$Risky_Clicks)
## [1] 3.44
sd(dataset1$Risky_Clicks)
## [1] 1.569337
median(dataset1$Risky_Clicks)
## [1] 4
#Create histograms
hist(dataset1$Security_Awareness_Score,
     main = "Security Awareness Score",
     breaks = 20,
     col = "lightblue",
     border = "white",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

hist(dataset1$Risky_Clicks,
     main = "Risky Clicks ",
     breaks = 20,
     col = "lightcoral",
     border = "white",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

# Variable 1: Security Awareness Score
# The variable looks abnormally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.

# Variable 2: Risky Clicks
# The variable looks abnormally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.


#Shapiro-WilkTest
shapiro.test(dataset1$Security_Awareness_Score)
## 
##  Shapiro-Wilk normality test
## 
## data:  dataset1$Security_Awareness_Score
## W = 0.98109, p-value = 0.0369
shapiro.test(dataset1$Risky_Clicks)
## 
##  Shapiro-Wilk normality test
## 
## data:  dataset1$Risky_Clicks
## W = 0.95167, p-value = 4.522e-05
# Variable 1: Security Awareness Score
# The variable is abnormally distributed (p = .037).

# Variable 2: Risky Click
# The variable is abnormally distributed (p < .001).


#Spearman correlation
cor.test(
  dataset1$Security_Awareness_Score,
  dataset1$Risky_Clicks,
  method = "spearman", exact = FALSE)
## 
##  Spearman's rank correlation rho
## 
## data:  dataset1$Security_Awareness_Score and dataset1$Risky_Clicks
## S = 929970, p-value < 2.2e-16
## alternative hypothesis: true rho is not equal to 0
## sample estimates:
##       rho 
## -0.653354
#The research question: Is there a relationship between employees' cybersecurity awareness scores and the number of risky links they click in a simulated phishing test?

#The variables being analyzed; Independent variable: Security awareness scores, dependent variable: Risky Clicks

#The appropriate statistical test: Spearman Correlation

#The null hypothesis: There is no relationship between employees' Cybersecurity awareness scores and the number of risky links they click in a simulated phishing test.

#The alternative hypothesis: There is a relationship between employees' Cybersecurity awareness scores and the number of risky links they click in a simulated phishing test.

# Interpretation of the results:
# A Spearman correlation was conducted to test the relationship between employee's security awareness score (Mdn = 66.6) and Risky link clicks in phishing text (Mdn = 4).

# There was a statistically significant relationship between the two variables, ρ =-.65, p < .001.

# The relationship was negative and strong.

# As the Security Awareness Score increased, the Risky Clicks decreased.

#-------------------------------------------------------------------------

#-----------
# Question 3
#-----------

#Open packages
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(effectsize)
## 
## Attaching package: 'effectsize'
## The following object is masked from 'package:rcompanion':
## 
##     phi
library(effsize)
library(readxl)
library(ggpubr)

#import dataset
dataset1 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset1.xlsx")
View(dataset1)

#Calculate descriptive statistics
dataset1 %>%
  group_by(Security_Training) %>%
  summarise(
    Mean = mean(Risky_Clicks, na.rm = TRUE),
    Median = median(Risky_Clicks, na.rm = TRUE),
    SD = sd(Risky_Clicks, na.rm = TRUE),
    N = n()
  )
## # A tibble: 2 × 5
##   Security_Training  Mean Median    SD     N
##   <chr>             <dbl>  <dbl> <dbl> <int>
## 1 Completed          2.77      3  1.58    65
## 2 Not Completed      3.95      4  1.36    85
#Histogram for normality check 1
hist(dataset1$Risky_Clicks[dataset1$Security_Training == "Completed"],
     breaks = 15,
     col = "skyblue",
     border = "white")

hist(dataset1$Risky_Clicks[dataset1$Security_Training == "Not Completed"],
     breaks = 15,
     col = "firebrick",
     border = "white")

#Data for department Completed appears normally distributed.
#Data for department not Completed appears normally distributed.

#Boxplot for normality check 2
ggboxplot(dataset1, x = "Security_Training", y = "Risky_Clicks",
          color = "Security_Training",
          palette = "jco",
          add = "jitter")

# Data for Completed group does have  outliers.
# Data for Not Completed group does have  outliers.

#Shapiro test for normality check 3
shapiro.test(dataset1$Risky_Clicks[dataset1$Security_Training == "Completed"])
## 
##  Shapiro-Wilk normality test
## 
## data:  dataset1$Risky_Clicks[dataset1$Security_Training == "Completed"]
## W = 0.94191, p-value = 0.004278
shapiro.test(dataset1$Risky_Clicks[dataset1$Security_Training == "Not Completed"])
## 
##  Shapiro-Wilk normality test
## 
## data:  dataset1$Risky_Clicks[dataset1$Security_Training == "Not Completed"]
## W = 0.93788, p-value = 0.0004869
#The Completed group is abnormally distributed, (p= .004).
#The Not Completed group abnormally distributed, (p <.001).

#Conduct the Mann-Whitney U Test (Abnormal Data)
wilcox.test(Risky_Clicks ~ Security_Training, data = dataset1)
## 
##  Wilcoxon rank sum test with continuity correction
## 
## data:  Risky_Clicks by Security_Training
## W = 1631, p-value = 1.199e-05
## alternative hypothesis: true location shift is not equal to 0
#Calculate Cliff's Delta (Effect Size)
mw_effect <- cliff.delta(Risky_Clicks ~ Security_Training, data = dataset1)
print(mw_effect)
## 
## Cliff's Delta
## 
## delta estimate: -0.4095928 (medium)
## 95 percent confidence interval:
##      lower      upper 
## -0.5602691 -0.2326774
#The research question: Is there a mean difference in the number of risky links clicked between employees who completed cybersecurity training and those who did not?

#The variables being analyzed;Group variable: Security Training, Outcome variable: Risky Clicks

#The appropriate statistical test:Mann-Whitney U Test

#The null hypothesis: There  is no  mean difference in the number of risky links clicked between employees who completed cybersecurity training and those who did not.

#The alternative hypothesis: There is a mean difference in the number of risky links clicked between employees who completed cybersecurity training and those who did not.

# Interpretation of the results:
#A Mann-Whitney U test was conducted to determine if there was a difference in Risky link clicks between employees that completed and those not Completed cybersecurity training.
#The employees that completed scores (Mdn = 3) were significantly different from those not completed scores (Mdn = 4), W = 1631, p< .001.
#The effect size was medium, Cliff's Delta = -.409.
#-------------------------------------------------------------------------

#----------
#Question 4
#----------

#Open packages
library(readxl)
library(ggpubr)
library(effsize)
library(rstatix)
## 
## Attaching package: 'rstatix'
## The following objects are masked from 'package:effectsize':
## 
##     cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:stats':
## 
##     filter
#import dataset
dataset2 <- read_excel("C:/Users/mashimwenizeyima/OneDrive - Saint Louis University/AA 5221/Final Project/dataset2.xlsx")
View(dataset2)

#Create groups for before and after groups
Before <- dataset2$Awareness_Score_Pre
After <- dataset2$Awareness_Score_Post

Differences <- After - Before

#Calculate descriptive statistics
mean(Before, na.rm = TRUE)
## [1] 59.088
median(Before, na.rm = TRUE)
## [1] 59.05
sd(Before, na.rm = TRUE)
## [1] 8.942849
mean(After, na.rm = TRUE)
## [1] 67.324
median(After, na.rm = TRUE)
## [1] 69.15
sd(After, na.rm = TRUE)
## [1] 10.25009
# Histogram for normality check1
hist(Differences,
     breaks = 15,
     col = "blue",
     border = "white")

#Data for the difference scores appears abnormally distributed.

#Boxplot for normality check 2
boxplot(Differences,
        main = "Awareness Score Differences (After - Before)",
        ylab = "Difference in Scores",
        col = "blue",
        border = "darkblue")

# The difference scores boxplot does not have outliers.

#Shapiro_Wilk Test for normality check3
shapiro.test(Differences)
## 
##  Shapiro-Wilk normality test
## 
## data:  Differences
## W = 0.95901, p-value = 0.08079
#Shapiro-Wilk Difference Scores
#The data is normally distributed, (p> .05).

#Conduct Dependent T test
t.test(Before, After, paired = TRUE, na.action = na.omit)
## 
##  Paired t-test
## 
## data:  Before and After
## t = -10.733, df = 49, p-value = 1.826e-14
## alternative hypothesis: true mean difference is not equal to 0
## 95 percent confidence interval:
##  -9.778075 -6.693925
## sample estimates:
## mean difference 
##          -8.236
#Calculate Cohen's d (Effect Size)
cohen.d(Before, After, paired = TRUE)
## 
## Cohen's d
## 
## d estimate: -0.8348924 (large)
## 95 percent confidence interval:
##      lower      upper 
## -1.0141545 -0.6556303
#The research question:  Is there a mean difference in employees' cybersecurity awareness scores before versus after completing the training intervention?

#The variables being analyzed; variable 1:Awareness_Score_Pre, variable 2:Awareness_Score_Post

#The appropriate statistical test:Chi-Squared Test of Dependence

#The null hypothesis: There is no mean difference in employees' cybersecurity awareness scores before versus after completing the training intervention.

#The alternative hypothesis:There is a mean difference in employees' cybersecurity awareness scores before versus after completing the training intervention.


# Interpretation of the results:
# A Dependent T-Test was conducted to determine if there was a difference in Awareness Score between Awareness Score Pre and Awareness Score Post.
# Awareness Score Pre (M = 59.09, SD = 8.94) is not significantly different from Awareness Score Post (M = 67.32, SD = 10.25), (t(49) = -10.73, p< .001)
#The effect size was large, Cohen's d = -.835.