library(readxl)
library(rcompanion)
# Chi-Square Test of Independence
# Research Question: Is there an association between a customer's banking type and whether they have experienced a security concern?
# Variables: Banking type (Online banking, In-person banking), Security concern (Yes, No)
# Test: Chi-Square Test of Independence
# H0: There is no association between banking type and security concern.
# Ha: There is an association between banking type and security concern.
banking_data <- read_excel("C:/Users/DELL/OneDrive - Saint Louis University/AA 5221/Final Project/banking_data.xlsx")
View(banking_data)
observed <- table(banking_data$Banking_Type, banking_data$Security_Concern)
observed
##
## No Yes
## In-person banking 95 36
## Online banking 69 100
barplot(
observed,
beside = TRUE,
col = rainbow(nrow(observed)),
legend = rownames(observed)
)

chi_result <- chisq.test(observed)
chi_result
##
## Pearson's Chi-squared test with Yates' continuity correction
##
## data: observed
## X-squared = 28.641, df = 1, p-value = 8.712e-08
cramer_v <- rcompanion::cramerV(observed)
cramer_v
## Cramer V
## 0.3157
# A Chi-Square Test of Independence was conducted to determine if there was an association between banking type (online banking or in-person banking) and security concern (yes or no).
# The results showed that there was an association between the two variables, χ²(1) = 28.64, p < .001.
# The association was moderate (Cramer's V = .32).
library(ggpubr)
## Loading required package: ggplot2
# Pearson or Spearman Correlation
# Research Question: Is there a relationship between a customer's age and their perceived security?
# Variables: Age (years), Perceived security (1 to 10)
# Test: Spearman Correlation
# H0: There is no relationship between age and perceived security.
# Ha: There is a relationship between age and perceived security.
ggscatter(
banking_data,
x = "Age",
y = "Perceived_Security",
add = "reg.line",
)

# The relationship is linear.
# The relationship is negative.
# There are no outliers.
mean(banking_data$Age)
## [1] 45.43333
sd(banking_data$Age)
## [1] 14.33751
median(banking_data$Age)
## [1] 45
mean(banking_data$Perceived_Security)
## [1] 6.659
sd(banking_data$Perceived_Security)
## [1] 1.450851
median(banking_data$Perceived_Security)
## [1] 6.8
hist(banking_data$Age,
breaks = 15,
col = "skyblue",
border = "white")

hist(banking_data$Perceived_Security,
breaks = 15,
col = "firebrick",
border = "white")

#Data for Age appears abnormally distributed.
#Data for Perceived_Security appears normally distributed.
shapiro.test(banking_data$Age)
##
## Shapiro-Wilk normality test
##
## data: banking_data$Age
## W = 0.98716, p-value = 0.009103
shapiro.test(banking_data$Perceived_Security)
##
## Shapiro-Wilk normality test
##
## data: banking_data$Perceived_Security
## W = 0.99428, p-value = 0.3216
#Age was abnormally distributed, W = 0.987, p = .009.
#Perceived_Security was normally distributed, W = 0.994, p = .322.
cor.test(
banking_data$Age,
banking_data$Perceived_Security,
method = "spearman"
)
## Warning in cor.test.default(banking_data$Age, banking_data$Perceived_Security,
## : cannot compute exact p-value with ties
##
## Spearman's rank correlation rho
##
## data: banking_data$Age and banking_data$Perceived_Security
## S = 5856707, p-value = 1.01e-07
## alternative hypothesis: true rho is not equal to 0
## sample estimates:
## rho
## -0.3015049
# A Spearman correlation was conducted to test the relationship between age (Mdn = 45.00) and perceived security (Mdn = 6.80).
# There was a statistically significant relationship between the two variables, ρ = -.30, p < .001.
# The relationship was negative and weak.
# As age increased, perceived security decreased.
# Independent T-Test or Mann-Whitney U
# Research Question: Is there a difference in perceived security between online banking customers and in-person banking customers?
# Variables: Banking type (Online banking, In-person banking), Perceived security (1 to 10)
# Test: Independent T-Test
# H0: There is no difference in perceived security between online banking and in-person banking customers.
# Ha: There is a difference in perceived security between online banking and in-person banking customers.
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(effectsize)
##
## Attaching package: 'effectsize'
## The following object is masked from 'package:rcompanion':
##
## phi
library(effsize)
banking_data %>%
group_by(Banking_Type) %>%
summarise(
Mean = mean(Perceived_Security, na.rm = TRUE),
Median = median(Perceived_Security, na.rm = TRUE),
SD = sd(Perceived_Security, na.rm = TRUE),
N = n()
)
## # A tibble: 2 × 5
## Banking_Type Mean Median SD N
## <chr> <dbl> <dbl> <dbl> <int>
## 1 In-person banking 7.47 7.6 1.18 131
## 2 Online banking 6.03 6.1 1.32 169
hist(banking_data$Perceived_Security[banking_data$Banking_Type == "In-person banking"],
breaks = 15,
col = "skyblue",
border = "white",
main = "Perceived Security: In-person Banking Customers",
xlab = "Perceived Security (1 to 10)")

hist(banking_data$Perceived_Security[banking_data$Banking_Type == "Online banking"],
breaks = 15,
col = "firebrick",
border = "white",
main = "Perceived Security: Online Banking Customers",
xlab = "Perceived Security (1 to 10)")

#Data for In-person_banking appears normally distributed.
#Data for Online_banking appears normally distributed.
ggboxplot(banking_data, x = "Banking_Type", y = "Perceived_Security",
color = "Banking_Type",
palette = "jco",
add = "jitter")

# The Online_banking boxplot does not have outliers.
# The In-person_banking boxplot does not have outliers.
shapiro.test(banking_data$Perceived_Security[banking_data$Banking_Type == "In-person banking"])
##
## Shapiro-Wilk normality test
##
## data: banking_data$Perceived_Security[banking_data$Banking_Type == "In-person banking"]
## W = 0.98985, p-value = 0.4541
shapiro.test(banking_data$Perceived_Security[banking_data$Banking_Type == "Online banking"])
##
## Shapiro-Wilk normality test
##
## data: banking_data$Perceived_Security[banking_data$Banking_Type == "Online banking"]
## W = 0.9935, p-value = 0.6572
#The In-person_banking group is normally distributed, (p > .05).
#The Online_banking group is normally distributed, (p > .05).
t.test(Perceived_Security ~ Banking_Type, data = banking_data, var.equal = TRUE)
##
## Two Sample t-test
##
## data: Perceived_Security by Banking_Type
## t = 9.8089, df = 298, p-value < 2.2e-16
## alternative hypothesis: true difference in means between group In-person banking and group Online banking is not equal to 0
## 95 percent confidence interval:
## 1.153301 1.732222
## sample estimates:
## mean in group In-person banking mean in group Online banking
## 7.471756 6.028994
cohens_d_result <- effectsize::cohens_d(Perceived_Security ~ Banking_Type, data = banking_data, pooled_sd = TRUE)
print(cohens_d_result)
## Cohen's d | 95% CI
## ------------------------
## 1.14 | [0.90, 1.39]
##
## - Estimated using pooled SD.
# An Independent T-Test was conducted to determine if there was a difference in perceived security between in-person banking and online banking customers.
# In-person banking scores (M = 7.47, SD = 1.18) were significantly different from online banking scores (M = 6.03, SD = 1.32), t(298) = 9.81, p < .001.
# The effect size was large, Cohen's d = 1.14.
# Dependent T-Test or Wilcoxon Signed-Rank
# Research Question: Is there a difference in perceived security before versus after exposure to a cybersecurity awareness message?
# Variables: Security_Before (1 to 10), Security_After (1 to 10)
# Test: Wilcoxon Signed-Rank
# H0: There is no difference in perceived security before versus after the cybersecurity awareness message.
# Ha: There is a difference in perceived security before versus after the cybersecurity awareness message.
library(effsize)
library(rstatix)
##
## Attaching package: 'rstatix'
## The following objects are masked from 'package:effectsize':
##
## cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:stats':
##
## filter
banking_data2 <- read_excel("C:/Users/DELL/OneDrive - Saint Louis University/AA 5221/Final Project/banking_data2.xlsx")
View(banking_data2)
Before <- banking_data2$Security_Before
After <- banking_data2$Security_After
Differences <- After - Before
mean(Before, na.rm = TRUE)
## [1] 5.941667
median(Before, na.rm = TRUE)
## [1] 5.9
sd(Before, na.rm = TRUE)
## [1] 0.9881144
mean(After, na.rm = TRUE)
## [1] 6.823333
median(After, na.rm = TRUE)
## [1] 6.8
sd(After, na.rm = TRUE)
## [1] 0.922665
hist(Differences,
breaks = 15,
col = "blue",
border = "white")

#Data for the difference scores appears abnormally distributed.
boxplot(Differences,
main = "Distribution of Score Differences (After - Before)",
ylab = "Difference in Scores",
col = "blue",
border = "darkblue")

# The difference scores boxplot has two outliers.
shapiro.test(Differences)
##
## Shapiro-Wilk normality test
##
## data: Differences
## W = 0.90366, p-value = 0.0001793
#Shapiro-Wilk Difference Scores
#The data is abnormally distributed, (p < .001).
wilcox.test(Before, After, paired = TRUE, na.action = na.omit)
##
## Wilcoxon signed rank test with continuity correction
##
## data: Before and After
## V = 13, p-value = 2.91e-11
## alternative hypothesis: true location shift is not equal to 0
df_long <- data.frame(id = rep(1:length(Before), 2), time = rep(c("Before", "After"), each = length(Before)), score = c(Before, After))
wilcox_effsize(df_long, score ~ time, paired = TRUE)
## # A tibble: 1 × 7
## .y. group1 group2 effsize n1 n2 magnitude
## * <chr> <chr> <chr> <dbl> <int> <int> <ord>
## 1 score After Before 0.859 60 60 large
#A Wilcoxon Signed-Rank Test was conducted to determine if there was a difference in perceived security before exposure to a cybersecurity awareness message versus after exposure.
#Before scores (Mdn = 5.90) were significantly different from after scores (Mdn = 6.80), V = 13, p < .001.
#The effect size was large, r = .86.