library(readxl)
library(rcompanion)
library(ggplot2)
library(ggpubr)
library(effsize)
library(rstatix)
##
## Attaching package: 'rstatix'
## The following object is masked from 'package:stats':
##
## filter
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
PDBO_100 <- read_excel("C:/Users/onhau01/OneDrive - Saint Louis University/AA 5221/Final Project/PDBO_100.xlsx")
PDWS_100 <- read_excel("C:/Users/onhau01/OneDrive - Saint Louis University/AA 5221/Final Project/PDWS_100.xlsx")
#1.Chi-Square Test of Independence - Between Objects/Groups
# Research Question: Is the presence of an email attachment associated with whether an email is classified as phishing or legitimate?
# Variables: Attachment: No / Yes
# Email Classification: Legitimate / Phishing
#Create and View a Frequency Table
email_table <- table(PDBO_100$Email_Classification, PDBO_100$Attachment)
email_table
##
## No Yes
## Legitimate 37 13
## Phishing 15 35
#Create a Bar Chart
barplot(email_table,
beside = TRUE,
col = rainbow(nrow(email_table)),
legend = rownames(email_table),
ylab = "Frequency")

#Pearson's Chi-squared test
chi_result <- chisq.test(email_table)
chi_result
##
## Pearson's Chi-squared test with Yates' continuity correction
##
## data: email_table
## X-squared = 17.668, df = 1, p-value = 2.63e-05
#Calculating the Effect Size
rcompanion::cramerV(email_table)
## Cramer V
## 0.4404
# A Chi-Square Test of Independence was conducted to determine if there was an association between Email Classification (Legitimate or Phishing) and Attachment.
# The results showed that there was an association between the two variables, χ²(1) = 17.67, p < .001.
# The association was moderate, (Cramer's V = .44).
#2.Pearson Correlation - Within the Dataset
# Research Question: Is email length related to the number of links contained in an email?
# Variables: Email Length: Number of words in the email
# Number of Links: Number of links contained in the email
#Create a Scatterplot
ggscatter(
PDBO_100,
x = "Email_Length_Words",
y = "Number_of_Links",
add = "reg.line",
xlab = "Email_Length_Words",
ylab = "Number_of_Links"
)

#Interpret the Scatterplot
# The relationship is linear
# The relationship is positive.
# There are no outliers.
#Calculate Descriptive Statistics
mean(PDBO_100$Email_Length_Words)
## [1] 110.66
sd(PDBO_100$Email_Length_Words)
## [1] 50.59489
median(PDBO_100$Email_Length_Words)
## [1] 100
mean(PDBO_100$Number_of_Links)
## [1] 3.45
sd(PDBO_100$Number_of_Links)
## [1] 2.328458
median(PDBO_100$Number_of_Links)
## [1] 3
#Create Histograms
hist(PDBO_100$Email_Length_Words,
main = "Email Length",
breaks = 20,
col = "lightblue",
border = "white",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

hist(PDBO_100$Number_of_Links,
main = "Number of Links",
breaks = 20,
col = "lightcoral",
border = "white",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

#Interpret the Histograms
# Variable 1: Email Length
# The variable does not look normally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.
# Variable 2: Number of Links
# The variable does not look normally distributed.
# The data is positively skewed.
# The data does not have a proper bell curve.
#Check Normality Statistically (Shapiro-Wilk Test)
shapiro.test (PDBO_100$Email_Length_Words)
##
## Shapiro-Wilk normality test
##
## data: PDBO_100$Email_Length_Words
## W = 0.95806, p-value = 0.002931
shapiro.test (PDBO_100$Number_of_Links)
##
## Shapiro-Wilk normality test
##
## data: PDBO_100$Number_of_Links
## W = 0.93508, p-value = 9.776e-05
#Interpret the Shapiro-Wilk Test
# Variable 1: Email_Length_Words
# The variable is abnormally distributed (p = .003).
# Variable 2: Number_of_Links
# The variable is abnormally distributed (p < .001).
#Spearman Correlation
cor.test(
PDBO_100$Email_Length_Words,
PDBO_100$Number_of_Links,
method = "spearman",
exact = FALSE
)
##
## Spearman's rank correlation rho
##
## data: PDBO_100$Email_Length_Words and PDBO_100$Number_of_Links
## S = 24340, p-value < 2.2e-16
## alternative hypothesis: true rho is not equal to 0
## sample estimates:
## rho
## 0.853945
#Spearman Reporting
# Report the Results
# A Spearman correlation was conducted to test the relationship between Email Length (Mdn = 100.00) and Number of Links (Mdn = 3.00).
# There was a statistically significant relationship between the two variables, ρ = .85, p < .001.
# The relationship was positive and strong.
# As the independent variable increased, the dependent variable increased.
#3.Independent Samples t-Test - Between Objects/Groups
#Research Question: Does the average number of links differ between phishing and legitimate emails?
#Calculate the Descriptive Statistics
PDBO_100%>%
group_by(Email_Classification) %>%
summarise(
Mean = mean(Number_of_Links, na.rm = TRUE),
Median = median(Number_of_Links, na.rm = TRUE),
SD = sd(Number_of_Links, na.rm = TRUE),
N = n()
)
## # A tibble: 2 × 5
## Email_Classification Mean Median SD N
## <chr> <dbl> <dbl> <dbl> <int>
## 1 Legitimate 1.92 2 1.21 50
## 2 Phishing 4.98 5 2.17 50
#Normality Check 1: Histograms
hist(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Legitimate"],
breaks = 15,
col = "skyblue",
border = "white")

hist(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Phishing"],
breaks = 15,
col = "firebrick",
border = "white")

#Interpret the Histograms
#Data for Legitimate emails appears abnormally distributed.
#Data for Phishing emails appears normally distributed.
#Normality Check 2: Boxplots
ggboxplot(PDBO_100, x = "Email_Classification", y = "Number_of_Links",
color = "Email_Classification",
palette = "jco",
add = "jitter")

#Interpret the Boxplots
# The Legitimate boxplot does not have outliers.
# The Phishing boxplot does have outliers.
#Normality Check 3: Shapiro-Wilk
shapiro.test(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Legitimate"])
##
## Shapiro-Wilk normality test
##
## data: PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Legitimate"]
## W = 0.92083, p-value = 0.002528
shapiro.test(PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Phishing"])
##
## Shapiro-Wilk normality test
##
## data: PDBO_100$Number_of_Links[PDBO_100$Email_Classification == "Phishing"]
## W = 0.96975, p-value = 0.2261
#Interpret the Shapiro-Wilk Test
#Data for Legitimate emails appears abnormally distributed.p < .05
#Data for Phishing emails appears normally distributed.p > .05
#Conduct the Mann-Whitney U Test (Abnormal Data)
wilcox.test(Number_of_Links ~ Email_Classification, data = PDBO_100)
##
## Wilcoxon rank sum test with continuity correction
##
## data: Number_of_Links by Email_Classification
## W = 273, p-value = 9.637e-12
## alternative hypothesis: true location shift is not equal to 0
#Calculate Cliff's Delta (Effect Size)
mw_effect <- cliff.delta(Number_of_Links ~ Email_Classification, data = PDBO_100)
print(mw_effect)
##
## Cliff's Delta
##
## delta estimate: -0.7816 (large)
## 95 percent confidence interval:
## lower upper
## -0.9031751 -0.5439521
#A Mann-Whitney U test was conducted to determine if there was a difference in Number_of_Links between Legitimate and Phishing.
#Legitimate scores (Mdn = 2.00) were significantly different from Phishing scores (Mdn = 5.00), W = 273.00, p < .001.
#The effect size was large, Cliff's Delta = -.782.
#4. Dependent Samples t-Test - Within Objects/Subjects
##Research Question: Does an individual's ability to identify phishing emails improve after cybersecurity awareness training?
#Variables: Phishing Detection Score Before: Number of phishing emails correctly identified before training
# Phishing Detection Score After : Number of phishing emails correctly identified after training
#Create Groups for Before & After
Before <- PDWS_100$Phishing_Detection_Score_Before
After <- PDWS_100$Phishing_Detection_Score_After
Differences <- After - Before
#Calculate the Descriptive Statistics
mean(Before, na.rm = TRUE)
## [1] 10.75
median(Before, na.rm = TRUE)
## [1] 11
sd(Before, na.rm = TRUE)
## [1] 2.528125
mean(After, na.rm = TRUE)
## [1] 14.03
median(After, na.rm = TRUE)
## [1] 14
sd(After, na.rm = TRUE)
## [1] 2.879587
#Normality Check 1: Histogram
hist(Differences,
breaks = 15,
col = "blue",
border = "white")

#Data for the difference scores appears abnormally distributed.
#Normality Check 2: Boxplot
boxplot(Differences,
main = "Distribution of Score Differences (After - Before)",
ylab = "Difference in Scores",
col = "blue",
border = "darkblue")

# The difference scores boxplot does not have outliers.
#Normality Check 3: Shapiro-Wilk
shapiro.test(Differences)
##
## Shapiro-Wilk normality test
##
## data: Differences
## W = 0.92443, p-value = 2.438e-05
#Shapiro-Wilk Difference Scores
#The data is abnormally distributed, (p < .001).
#Conduct the Wilcoxon Signed-Rank Test
wilcox.test(Before, After, paired = TRUE, na.action = na.omit)
##
## Wilcoxon signed rank test with continuity correction
##
## data: Before and After
## V = 0, p-value < 2.2e-16
## alternative hypothesis: true location shift is not equal to 0
#Calculate the Effect Size
df_long <- data.frame(id = rep(1:length(Before), 2), time = rep(c("Before", "After"), each = length(Before)), score = c(Before, After))
wilcox_effsize(df_long, score ~ time, paired = TRUE)
## # A tibble: 1 × 7
## .y. group1 group2 effsize n1 n2 magnitude
## * <chr> <chr> <chr> <dbl> <int> <int> <ord>
## 1 score After Before 0.874 100 100 large
#Report the Wilcoxon Signed-Rank Test
#A Wilcoxon Signed-Rank Test was conducted to determine if there was a difference in Phishing Detection Score before cybersecurity awareness training versus after cybersecurity awareness training.
#Before scores (Mdn = 11.00) were significantly different from after scores (Mdn = 14.00), V = 0, p < .001.
#The effect size was large, r = .874.