#============================================================================
# AA 5221 Final Project: Employee Cybersecurity Behavior Analysis
#============================================================================
#Topic: Employee Cybersecurity Behavior: Security Awareness and Incident Response Performance Across Shifts
#Overall Focus: Investigating how employee security awareness changes following a simulated security incident and how incident response performance and compliance differ between morning and night shifts.

library("readxl")
library("ggplot2")
library("rcompanion")
Dataset1 <- read_excel("C:/Users/siqoe/OneDrive - Saint Louis University/AA 5221/Final Project/Dataset1_BetweenSubjects.xlsx")
(Dataset1)
## # A tibble: 120 × 6
##    Employee_ID Shift   Policy_Compliance Security_Awareness_Score
##    <chr>       <chr>   <chr>                                <dbl>
##  1 E001        Morning Compliant                               78
##  2 E002        Night   Non-Compliant                           58
##  3 E003        Morning Compliant                               85
##  4 E004        Night   Compliant                               69
##  5 E005        Morning Compliant                               74
##  6 E006        Night   Non-Compliant                           52
##  7 E007        Morning Compliant                               91
##  8 E008        Night   Compliant                               66
##  9 E009        Morning Non-Compliant                           62
## 10 E010        Night   Non-Compliant                           55
## # ℹ 110 more rows
## # ℹ 2 more variables: Incident_Response_Score <dbl>, Training_Hours <dbl>
Dataset1_BetweenSubjects<-table(Dataset1$Shift,Dataset1$Policy_Compliance)
Dataset1_BetweenSubjects
##          
##           Compliant Non-Compliant
##   Morning        53             7
##   Night          23            37
barplot(Dataset1_BetweenSubjects) 

        beside = TRUE
        col = rainbow(nrow(Dataset1_BetweenSubjects))
        legend = rownames(Dataset1_BetweenSubjects)

chi_result <- chisq.test(Dataset1_BetweenSubjects) 
chi_result
## 
##  Pearson's Chi-squared test with Yates' continuity correction
## 
## data:  Dataset1_BetweenSubjects
## X-squared = 30.179, df = 1, p-value = 3.939e-08
rcompanion::cramerV(Dataset1_BetweenSubjects)
## Cramer V 
##   0.5188
#A Chi-Square Test of Independence was conducted to determine if there was an association between sHIFT allocation and Policy_Compliance. 
#The results showed that there was a significant association between the two variables, χ²(1) = 30.179, p < .001. 
#The association was strong, Cramer's V = .5188.

#======================SpearmanCorrelation===================================
#Is security awareness related to employee incident-response performance?

library(readxl)
library(ggpubr)

Dataset1 <- read_excel("C:/Users/siqoe/OneDrive - Saint Louis University/AA 5221/Final Project/Dataset1_BetweenSubjects.xlsx")
ggscatter(
  Dataset1,
  x = "Security_Awareness_Score",
  y = "Incident_Response_Score",
  add = "reg.line",
  xlab = "Security_Awareness_Score",
  ylab = "Incident_Response_Score"
)

# The relationship is linear.
# The relationship is positive.
# There are  no outliers.

mean(Dataset1$Security_Awareness_Score)
## [1] 70.49167
sd(Dataset1$Security_Awareness_Score)
## [1] 13.70088
median(Dataset1$Security_Awareness_Score)
## [1] 70.5
mean(Dataset1$Incident_Response_Score)
## [1] 73.75833
sd(Dataset1$Incident_Response_Score)
## [1] 13.5758
median(Dataset1$Incident_Response_Score)
## [1] 73
hist(Dataset1$Security_Awareness_Score)

hist(Dataset1$Incident_Response_Score)

# Variable 1:Security_Awareness_Score 
# The variable is abnormally distributed.
# The data is asymmetrical.
# The data has no proper bell curve.

# Variable 2: Incident_Response_Score
# The variable is abnormally distributed.
# The data is asymmetrical.
# The data has  no proper bell curve.

shapiro.test(Dataset1$Security_Awareness_Score)
## 
##  Shapiro-Wilk normality test
## 
## data:  Dataset1$Security_Awareness_Score
## W = 0.95268, p-value = 0.0003414
shapiro.test(Dataset1$Incident_Response_Score)
## 
##  Shapiro-Wilk normality test
## 
## data:  Dataset1$Incident_Response_Score
## W = 0.94439, p-value = 8.562e-05
# Variable 1: Security_Awareness_Score
# The variable is abnormally distributed (p = < .001 ).

# Variable 2: Incident_Response_Score
# The variable is abnormally distributed (p = < .001).

cor.test(
  Dataset1$Security_Awareness_Score,
  Dataset1$Incident_Response_Score,
  method = "spearman")
## Warning in cor.test.default(Dataset1$Security_Awareness_Score,
## Dataset1$Incident_Response_Score, : cannot compute exact p-value with ties
## 
##  Spearman's rank correlation rho
## 
## data:  Dataset1$Security_Awareness_Score and Dataset1$Incident_Response_Score
## S = 333.16, p-value < 2.2e-16
## alternative hypothesis: true rho is not equal to 0
## sample estimates:
##       rho 
## 0.9988431
#A Spearman correlation was conducted to test the relationship between Security_Awareness_Score (Mdn = 70.50) and Incident_Response_Score (Mdn = 73.00).
#There was a statistically significant relationship between the two variables, ρ = .999, p < .001.
#The relationship was positive and extremely strong.
#As Security_Awareness_Score increased, Incident_Response_Score also increased.

#=====================Independent t.test====================================
#Does incident-response performance differ between morning- and night-shift employees?

library(readxl)
library(ggpubr)
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(effectsize)
## 
## Attaching package: 'effectsize'
## The following object is masked from 'package:rcompanion':
## 
##     phi
library(effsize)


Dataset1<- read_excel("C:/Users/siqoe/OneDrive - Saint Louis University/AA 5221/Final Project/Dataset1_BetweenSubjects.xlsx")

Dataset1 %>%
  group_by(Shift)%>%
  summarise(
    Mean = mean(Incident_Response_Score, na.rm = TRUE),
    Median = median(Incident_Response_Score, na.rm = TRUE),
    SD = sd(Incident_Response_Score, na.rm = TRUE),
    N = n()
  )
## # A tibble: 2 × 5
##   Shift    Mean Median    SD     N
##   <chr>   <dbl>  <dbl> <dbl> <int>
## 1 Morning  84.6   86    8.65    60
## 2 Night    62.9   61.5  7.56    60
hist(Dataset1$Incident_Response_Score[Dataset1$Shift == "Morning"],
     breaks = 15,
     col = "skyblue",
     border = "white")

hist(Dataset1$Incident_Response_Score[Dataset1$Shift == "Night"],
     breaks = 15,
     col = "firebrick",
     border = "white")

#Data for Morning is abnormally distributed.
#Data for Night is  abnormally distributed.

ggboxplot(Dataset1, x = "Shift", y = "Incident_Response_Score",
          color = "Shift",
          palette = "jco",
          add = "jitter")

# The  Morning shift boxplot has outliers.
# The Night shift boxplot has outliers.

shapiro.test(Dataset1$Incident_Response_Score[Dataset1$Shift == "Morning"])
## 
##  Shapiro-Wilk normality test
## 
## data:  Dataset1$Incident_Response_Score[Dataset1$Shift == "Morning"]
## W = 0.90861, p-value = 0.0002751
shapiro.test(Dataset1$Incident_Response_Score[Dataset1$Shift == "Night"])
## 
##  Shapiro-Wilk normality test
## 
## data:  Dataset1$Incident_Response_Score[Dataset1$Shift == "Night"]
## W = 0.9274, p-value = 0.001543
#The Morning shift is abnormally distributed, (p < .05).
#The Night shift is abnormally distributed, (p < .05).

wilcox.test(Incident_Response_Score ~ Shift, data = Dataset1)
## 
##  Wilcoxon rank sum test with continuity correction
## 
## data:  Incident_Response_Score by Shift
## W = 3441, p-value < 2.2e-16
## alternative hypothesis: true location shift is not equal to 0
mw_effect <- cliff.delta(Incident_Response_Score ~ Shift, data = Dataset1)
print(mw_effect)
## 
## Cliff's Delta
## 
## delta estimate: 0.9116667 (large)
## 95 percent confidence interval:
##     lower     upper 
## 0.8156832 0.9588006
#A Mann-Whitney U test was conducted to determine if there was a difference in Incident_Response_Scores between Morning and Night shifts.
#Morning shift scores (Mdn = 86.00) were significantly higher than Night shift scores (Mdn = 61.50), W = 3441.00, p < .001.
#The effect size was large, Cliff's Delta = .91.

#================================Dependant t.test============================
#Does employee security awareness change after experiencing a cybersecurity incident?

library(ggpubr)
library(effsize)
library(rstatix)
## 
## Attaching package: 'rstatix'
## The following objects are masked from 'package:effectsize':
## 
##     cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:stats':
## 
##     filter
Dataset2_WithinSubjects <- read_excel("C:/Users/siqoe/OneDrive - Saint Louis University/AA 5221/Final Project/Dataset2_WithinSubjects.xlsx")

Before <- Dataset2_WithinSubjects$Pre_Attack_Awareness
After <- Dataset2_WithinSubjects$Post_Attack_Awareness

Differences <- After - Before

mean(Before, na.rm = TRUE)
## [1] 62.46667
median(Before, na.rm = TRUE)
## [1] 62
sd(Before, na.rm = TRUE)
## [1] 6.502846
mean(After, na.rm = TRUE)
## [1] 73.9
median(After, na.rm = TRUE)
## [1] 74
sd(After, na.rm = TRUE)
## [1] 6.688467
hist(Differences,
     breaks = 15,
     col = "blue",
     border = "white")

boxplot(Differences,
        main = "Security Awareness (After - Before)",
        ylab = "Difference in Awareness",
        col = "blue",
        border = "darkblue")

# The difference in Awareness boxplot has no outliers.

shapiro.test(Differences)
## 
##  Shapiro-Wilk normality test
## 
## data:  Differences
## W = 0.82201, p-value = 4.992e-07
#The data is abnormally distributed, (p = 4.992e-07).

wilcox.test(Before, After, paired = TRUE, na.action = na.omit)
## 
##  Wilcoxon signed rank test with continuity correction
## 
## data:  Before and After
## V = 0, p-value = 6.04e-12
## alternative hypothesis: true location shift is not equal to 0
df_long <- data.frame(id = rep(1:length(Before), 2), time = rep(c("Before", "After"), each = length(Before)), score = c(Before, After))

wilcox_effsize(df_long, score ~ time, paired = TRUE)
## # A tibble: 1 × 7
##   .y.   group1 group2 effsize    n1    n2 magnitude
## * <chr> <chr>  <chr>    <dbl> <int> <int> <ord>    
## 1 score After  Before   0.889    60    60 large
#A Wilcoxon Signed-Rank Test was conducted to determine if there was a difference in Security Awareness scores before the attack intervention versus after the attack intervention.
#Before scores (Mdn = 62.00) were significantly different from after scores (Mdn = 74.00), V = 0, p < .001.
#The effect size was large, r = .889.