library(readxl)
library(rcompanion)
library(ggplot2)
library(ggpubr)
library(effsize)
library(rstatix)
## 
## Attaching package: 'rstatix'
## The following object is masked from 'package:stats':
## 
##     filter
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
PDBO_100 <- read_excel("C:/Users/onhau01/OneDrive - Saint Louis University/AA 5221/Final Project/PDBO_100.xlsx")

PDWS_100 <- read_excel("C:/Users/onhau01/OneDrive - Saint Louis University/AA 5221/Final Project/PDWS_100.xlsx")

#1.Chi-Square Test of Independence - Between Objects/Groups
#  Research Question: Is the presence of an email attachment associated with whether an email is classified as phishing or legitimate?
#  Variables: Attachment: No / Yes
#             Email Classification: Legitimate / Phishing


#Create and View a Frequency Table

email_table <- table(PDBO_100$Email_Classification, PDBO_100$Attachment)
email_table 
##             
##              No Yes
##   Legitimate 37  13
##   Phishing   15  35
#Create a Bar Chart
barplot(email_table,
        beside = TRUE,
        col = rainbow(nrow(email_table)),
        legend = rownames(email_table),
        ylab = "Frequency")

#Pearson's Chi-squared test
chi_result <- chisq.test(email_table)

chi_result
## 
##  Pearson's Chi-squared test with Yates' continuity correction
## 
## data:  email_table
## X-squared = 17.668, df = 1, p-value = 2.63e-05
#Calculating the Effect Size
rcompanion::cramerV(email_table)
## Cramer V 
##   0.4404
# A Chi-Square Test of Independence was conducted to determine if there was an association between Email Classification (Legitimate or Phishing) and Attachment.
# The results showed that there was an association between the two variables, χ²(1) = 17.67, p < .001.
# The association was moderate, (Cramer's V = .44).

#2.Pearson Correlation - Within the Dataset
 # Research Question: Is email length related to the number of links contained in an email?
 # Variables: Email Length: Number of words in the email
 #            Number of Links: Number of links contained in the email


#Create a Scatterplot
ggscatter(
  PDBO_100,
  x = "Email_Length_Words",
  y = "Number_of_Links",
  add = "reg.line",
  xlab = "Email_Length_Words",
  ylab = "Number_of_Links"
)

#Interpret the Scatterplot
# The relationship is linear 
# The relationship is positive.
# There are no outliers.

#Calculate Descriptive Statistics

mean(PDBO_100$Email_Length_Words)
## [1] 110.66
sd(PDBO_100$Email_Length_Words)
## [1] 50.59489
median(PDBO_100$Email_Length_Words)
## [1] 100
mean(PDBO_100$Number_of_Links)
## [1] 3.45
sd(PDBO_100$Number_of_Links)
## [1] 2.328458
median(PDBO_100$Number_of_Links)
## [1] 3
#Create Histograms

hist(PDBO_100$Email_Length_Words,
     main = "Email Length",
     breaks = 20,
     col = "lightblue",
     border = "white",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

hist(PDBO_100$Number_of_Links,
     main = "Number of Links",
     breaks = 20,
     col = "lightcoral",
     border = "white",
     cex.main = 1,
     cex.axis = 1,
     cex.lab = 1)

#Interpret the Histograms
# Variable 1: Email Length
# The variable does not look normally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.

# Variable 2: Number of Links
# The variable does not look normally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.

#Check Normality Statistically (Shapiro-Wilk Test)
shapiro.test (PDBO_100$Email_Length_Words)
## 
##  Shapiro-Wilk normality test
## 
## data:  PDBO_100$Email_Length_Words
## W = 0.95806, p-value = 0.002931
shapiro.test (PDBO_100$Number_of_Links)
## 
##  Shapiro-Wilk normality test
## 
## data:  PDBO_100$Number_of_Links
## W = 0.93508, p-value = 9.776e-05
#Interpret the Shapiro-Wilk Test
# Variable 1: Email_Length_Words
# The variable is abnormally distributed (p = .003).

# Variable 2: Number_of_Links
# The variable is abnormally distributed (p < .001).

#Spearman Correlation
cor.test(
  PDBO_100$Email_Length_Words,
  PDBO_100$Number_of_Links,
  method = "spearman",
  exact = FALSE
)
## 
##  Spearman's rank correlation rho
## 
## data:  PDBO_100$Email_Length_Words and PDBO_100$Number_of_Links
## S = 24340, p-value < 2.2e-16
## alternative hypothesis: true rho is not equal to 0
## sample estimates:
##      rho 
## 0.853945
#Spearman Reporting

# Report the Results
# A Spearman correlation was conducted to test the relationship between Email Length (Mdn = 100.00) and Number of Links (Mdn = 3.00).
# There was a statistically significant relationship between the two variables, ρ = .85, p < .001.
# The relationship was positive and strong.
# As the independent variable increased, the dependent variable increased.

#3.Independent Samples t-Test - Between Objects/Groups
#Research Question: Does the average number of links differ between phishing and legitimate emails?

#Calculate the Descriptive Statistics

PDBO_100%>%
  group_by(Email_Classification) %>%
  summarise(
    Mean = mean(Number_of_Links, na.rm = TRUE),
    Median = median(Number_of_Links, na.rm = TRUE),
    SD = sd(Number_of_Links, na.rm = TRUE),
    N = n()
  )
## # A tibble: 2 × 5
##   Email_Classification  Mean Median    SD     N
##   <chr>                <dbl>  <dbl> <dbl> <int>
## 1 Legitimate            1.92      2  1.21    50
## 2 Phishing              4.98      5  2.17    50
#Normality Check 1: Histograms

hist(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Legitimate"],
     breaks = 15,
     col = "skyblue",
     border = "white")

hist(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Phishing"],
     breaks = 15,
     col = "firebrick",
     border = "white")

#Interpret the Histograms
#Data for Legitimate emails appears abnormally distributed.
#Data for Phishing emails appears normally distributed.

#Normality Check 2: Boxplots

ggboxplot(PDBO_100, x = "Email_Classification", y = "Number_of_Links",
          color = "Email_Classification",
          palette = "jco",
          add = "jitter")

#Interpret the Boxplots
# The Legitimate boxplot does not have outliers.
# The Phishing boxplot does have outliers.

#Normality Check 3: Shapiro-Wilk

shapiro.test(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Legitimate"])
## 
##  Shapiro-Wilk normality test
## 
## data:  PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Legitimate"]
## W = 0.92083, p-value = 0.002528
shapiro.test(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Phishing"])
## 
##  Shapiro-Wilk normality test
## 
## data:  PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Phishing"]
## W = 0.96975, p-value = 0.2261
#Interpret the Shapiro-Wilk Test

#Data for Legitimate emails appears abnormally distributed.p < .05
#Data for Phishing emails appears normally distributed.p > .05

#Conduct the Mann-Whitney U Test (Abnormal Data)

wilcox.test(Number_of_Links ~ Email_Classification, data = PDBO_100)
## 
##  Wilcoxon rank sum test with continuity correction
## 
## data:  Number_of_Links by Email_Classification
## W = 273, p-value = 9.637e-12
## alternative hypothesis: true location shift is not equal to 0
#Calculate Cliff's Delta (Effect Size)

mw_effect <- cliff.delta(Number_of_Links ~ Email_Classification, data = PDBO_100)
print(mw_effect)
## 
## Cliff's Delta
## 
## delta estimate: -0.7816 (large)
## 95 percent confidence interval:
##      lower      upper 
## -0.9031751 -0.5439521
#A Mann-Whitney U test was conducted to determine if there was a difference in Number_of_Links between Legitimate and Phishing.
#Legitimate scores (Mdn = 2.00) were significantly different from Phishing scores (Mdn = 5.00), W = 273.00, p < .001.
#The effect size was large, Cliff's Delta = -.782.


#4. Dependent Samples t-Test - Within Objects/Subjects
##Research Question: Does an individual's ability to identify phishing emails improve after cybersecurity awareness training?
#Variables: Phishing Detection Score Before: Number of phishing emails correctly identified before training    
#           Phishing Detection Score After : Number of phishing emails correctly identified after training


#Create Groups for Before & After
Before <- PDWS_100$Phishing_Detection_Score_Before
After <- PDWS_100$Phishing_Detection_Score_After

Differences <- After - Before

#Calculate the Descriptive Statistics
mean(Before, na.rm = TRUE)
## [1] 10.75
median(Before, na.rm = TRUE)
## [1] 11
sd(Before, na.rm = TRUE)
## [1] 2.528125
mean(After, na.rm = TRUE)
## [1] 14.03
median(After, na.rm = TRUE)
## [1] 14
sd(After, na.rm = TRUE)
## [1] 2.879587
#Normality Check 1: Histogram
hist(Differences,
     breaks = 15,
     col = "blue",
     border = "white")

#Data for the difference scores appears abnormally distributed.

#Normality Check 2: Boxplot

boxplot(Differences,
        main = "Distribution of Score Differences (After - Before)",
        ylab = "Difference in Scores",
        col = "blue",
        border = "darkblue")

# The difference scores boxplot does not have outliers.

#Normality Check 3: Shapiro-Wilk

shapiro.test(Differences)
## 
##  Shapiro-Wilk normality test
## 
## data:  Differences
## W = 0.92443, p-value = 2.438e-05
#Shapiro-Wilk Difference Scores
#The data is abnormally distributed, (p < .001).

#Conduct the Wilcoxon Signed-Rank Test

wilcox.test(Before, After, paired = TRUE, na.action = na.omit)
## 
##  Wilcoxon signed rank test with continuity correction
## 
## data:  Before and After
## V = 0, p-value < 2.2e-16
## alternative hypothesis: true location shift is not equal to 0
#Calculate the Effect Size

df_long <- data.frame(id = rep(1:length(Before), 2), time = rep(c("Before", "After"), each = length(Before)), score = c(Before, After))

wilcox_effsize(df_long, score ~ time, paired = TRUE)
## # A tibble: 1 × 7
##   .y.   group1 group2 effsize    n1    n2 magnitude
## * <chr> <chr>  <chr>    <dbl> <int> <int> <ord>    
## 1 score After  Before   0.874   100   100 large
#Report the Wilcoxon Signed-Rank Test
#A Wilcoxon Signed-Rank Test was conducted to determine if there was a difference in Phishing Detection Score before cybersecurity awareness training versus after cybersecurity awareness training.
#Before scores (Mdn = 11.00) were significantly different from after scores (Mdn = 14.00), V = 0, p < .001.
#The effect size was large, r = .874.