# FINAL PROJECT AA 5221
#
# Research Problem Statement 1: Seasonal Household Electricity Consumption
#
# Household electricity use may vary between summer and winter because cooling
# and heating create different energy demands. However, a seasonal difference
# in average consumption and a relationship between households’ consumption
# across seasons are separate questions. This study uses the synthetic
# Household_Electricity.xlsx dataset, containing paired summer and winter
# consumption measurements for 300 households in a 2026 Saint Louis, Missouri
# scenario. It examines whether households with higher summer consumption also
# tend to have higher winter consumption and whether mean consumption differs
# between the two seasons. Consumption is measured in kilowatt-hours over
# comparable periods.
#
# We will be using Pearson or Spearman correlation to examine the relationship
# and a paired-samples t-test or Wilcoxon signed-rank test to examine seasonal
# differences, depending on the relevant assumptions.
#
# Data Sets
#
# The synthetic data sets where modelled using Chat GPT
#
# Question 1 — Dataset: Household_Electricity.xlsx, using Summer_kWh and
# Winter_kWh to examine correlation.
#
# Question 2 — Dataset: Household_Electricity.xlsx, using paired Summer_kWh
# and Winter_kWh measurements to compare seasonal consumption.
#
# Question 3 — Dataset: Ransomware_Recovery.xlsx, using Response_Procedure
# and Recovery_Time_Minutes to compare recovery times between independent groups.
#
# Question 4 — Dataset: Ransomware_Recovery.xlsx, using Response_Procedure
# and Recovery_Outcome to examine association with meeting the assumed
# 150-minute benchmark.
# ----------------------------------------------------------------
# RESEARCH QUESTION 1
# ----------------------------------------------------------------
# QUESTION 1
#
# Household Electricity Data
#
# Is there a relationship between 2026 summer and winter electricity consumption
# among 300 households in Saint Louis, Missouri?
#
# In practical terms: Do households that use more electricity in summer also
# tend to use more electricity in winter?
#
# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: There is no correlation between summer and winter household electricity
# consumption.
# H₁: There is a correlation between summer and winter household electricity
# consumption.
#
# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Each household contributes two numerical measurements: summer consumption
# and winter consumption, measured in comparable kWh periods.
#
# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 1: Statistical Test and Justification
#
# A Pearson correlation will examine the strength and direction of the linear
# relationship between summer and winter electricity consumption. Both
# variables are numerical, and each household provides one paired set of
# measurements. Following the master guideline, descriptive statistics,
# histograms, boxplots, Shapiro–Wilk tests, and a scatterplot will be examined
# before selecting the test. Pearson correlation is appropriate when the
# relationship is linear and its assumptions are reasonably satisfied.
# Spearman correlation will be used if the data are unsuitable for Pearson
# but show a monotonic relationship. A two-sided test with an alpha level of
# .05 will match the hypothesis that a correlation exists in either direction.
# Report the correlation coefficient, p-value, and strength and direction
# of the relationship.
#
# ----------------------------------------------------------------
# 4. RESULTS — DATA, GRAPHS, DIAGNOSTICS AND TESTS
# ----------------------------------------------------------------
library(readxl)
library(ggpubr)
## Loading required package: ggplot2
library(readxl)
data <- read_excel("C:/Users/hp/Desktop/digital Forensics/AA 5221/FInal_Project/Household_Electricity.xlsx")
View(data)
ggscatter(
data,
x = "Summer_kWh",
y = "Winter_kWh",
add = "reg.line",
xlab = "Summer Electricity Consumption (kWh)",
ylab = "Winter Electricity Consumption (kWh)"
)

# Describe whether the relationship is linear.
# Describe whether the relationship is positive.
# Check for potential outliers.
mean(data$Summer_kWh)
## [1] 1110.149
sd(data$Summer_kWh)
## [1] 194.8264
median(data$Summer_kWh)
## [1] 1107.355
mean(data$Winter_kWh)
## [1] 910.9583
sd(data$Winter_kWh)
## [1] 188.4099
median(data$Winter_kWh)
## [1] 920.315
hist(data$Summer_kWh)

hist(data$Winter_kWh)

# Variable 1: Summer Electricity Consumption
# Summer Electricity Consumption looks normally distributed.
# Summer Electricity Consumption data is symmetrical.
# Summer Electricity Consumption data has a proper bell curve.
# Variable 2: Winter Electricity Consumption
# Winter Electricity Consumption data looks normally distributed.
# Winter Electricity Consumption data is symmetrical.
# Winter Electricity Consumption data has a proper bell curve.
shapiro.test(data$Summer_kWh)
##
## Shapiro-Wilk normality test
##
## data: data$Summer_kWh
## W = 0.9956, p-value = 0.5612
shapiro.test(data$Winter_kWh)
##
## Shapiro-Wilk normality test
##
## data: data$Winter_kWh
## W = 0.99191, p-value = 0.1004
# Variable 1: Summer Electricity Consumption
# The variable is normally distributed (p = .5612).
# Variable 2: Winter Electricity Consumption
# The variable is normally distributed (p = .1004).
cor.test(data$Summer_kWh, data$Winter_kWh, method = "pearson")
##
## Pearson's product-moment correlation
##
## data: data$Summer_kWh and data$Winter_kWh
## t = 19.246, df = 298, p-value < 2.2e-16
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
## 0.6892941 0.7909885
## sample estimates:
## cor
## 0.7444277
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Pearson correlation was conducted to test the relationship between summer and winter electricity consumption.
# There was a statistically significant relationship between the two variables,
# r(298) = .74, p < .001.
# The relationship was positive and strong.
# As summer electricity consumption increased, winter electricity consumption increased
# ----------------------------------------------------------------
# RESEARCH QUESTION 2
# ----------------------------------------------------------------
# QUESTION 2
#
# House hold Electricity Data
#
# Is there a difference in mean electricity consumption between summer and
# winter among 300 households in Saint Louis, Missouri?
#
# In practical terms: Do the same households use more electricity, on average,
# in summer or winter?
#
# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: The population mean difference between summer and winter electricity
# consumption is zero
# H₁: The population mean difference between summer and winter electricity
# consumption is not zero
#
# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Each household contributes two measurements: summer consumption and winter
# consumption, measured in kWh over comparable periods.
#
# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 2: Statistical Test and Justification
#
# A dependent t-test, also called a paired-samples t-test, will compare mean
# summer and winter electricity consumption because the same 300 households
# are measured in both seasons. Following the master guideline, a difference
# score will be calculated for each household as summer consumption minus
# winter consumption. A histogram, boxplot, and Shapiro–Wilk test will assess
# these difference scores rather than the two seasonal distributions
# separately. If the assumptions are reasonably satisfied, a two-sided paired
# t-test will be conducted at an alpha level of .05. Report each season’s mean
# and standard deviation, the mean difference and its confidence interval,
# t, degrees of freedom, p-value, and paired Cohen’s d. If the difference
# scores are unsuitable for a t-test, Wilcoxon signed-rank may be used,
# provided its symmetry assumption is reasonable; this alternative does
# not directly test a difference in means.
#
# ----------------------------------------------------------------
# 4. RESULTS — DATA, GRAPHS, DIAGNOSTICS AND TESTS
# ----------------------------------------------------------------
library(readxl)
library(ggpubr)
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(effectsize)
library(effsize)
library(rstatix)
##
## Attaching package: 'rstatix'
## The following objects are masked from 'package:effectsize':
##
## cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:stats':
##
## filter
View(data)
Winter <- data$Winter_kWh
Summer <- data$Summer_kWh
Differences <- Summer - Winter
mean(Winter, na.rm = TRUE)
## [1] 910.9583
median(Winter, na.rm = TRUE)
## [1] 920.315
sd(Winter, na.rm = TRUE)
## [1] 188.4099
mean(Summer, na.rm = TRUE)
## [1] 1110.149
median(Summer, na.rm = TRUE)
## [1] 1107.355
sd(Summer, na.rm = TRUE)
## [1] 194.8264
hist(
Differences,
breaks = 15,
col = "blue",
border = "white"
)

# Describe the distribution of the summer-minus-winter difference scores.
boxplot(
Differences,
main = "Distribution of Electricity Differences (Summer - Winter)",
ylab = "Difference in Consumption (kWh)",
col = "blue",
border = "darkblue"
)

# No outliers in the difference scores.
shapiro.test(Differences)
##
## Shapiro-Wilk normality test
##
## data: Differences
## W = 0.99755, p-value = 0.9363
# The the distribution of differences in Summer and Winter is normal.
# The normality assumption concerns the difference scores.
t.test(
Summer,
Winter,
paired = TRUE,
na.action = na.omit
)
##
## Paired t-test
##
## data: Summer and Winter
## t = 25.16, df = 299, p-value < 2.2e-16
## alternative hypothesis: true mean difference is not equal to 0
## 95 percent confidence interval:
## 183.6104 214.7708
## sample estimates:
## mean difference
## 199.1906
# Paired Cohen's d: mean difference divided by the SD of differences.
cohens_d <- mean(Differences, na.rm = TRUE) /
sd(Differences, na.rm = TRUE)
cohens_d
## [1] 1.452597
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Dependent T-Test was conducted to determine if there was a difference in electricity consumption between summer and winter.
# Summer consumption (M = 1110.15, SD = 194.83) was significantly higher than winter consumption (M = 910.96, SD = 188.41), t(299) = 25.16, p < .001.
# The effect size was very large, paired Cohen's d = 1.453.
# Research Problem Statement 2: Ransomware Response Procedures and Data Recovery
#
# Organizations need to understand whether an optimized ransomware response
# playbook improves data recovery compared with existing procedures. Recovery
# performance can be examined through both the time teams take to restore
# data and whether they meet a defined recovery target. This study uses the
# synthetic Ransomware_Recovery.xlsx dataset, containing recovery times for
# 100 independent cybersecurity teams: 50 using an optimized playbook and
# 50 using old procedures under assumed comparable exercise conditions.
# It examines whether recovery-time distributions differ between the groups
# and whether response procedure is associated with meeting an assumed
# 150-minute recovery benchmark. The benchmark is a hypothetical study target,
# not a universal NIST requirement.
#
# ----------------------------------------------------------------
# RESEARCH QUESTION 3
# ----------------------------------------------------------------
# Question 3
#
# Do data recovery times after a simulated ransomware breach differ between
# cybersecurity teams using an optimized playbook and teams using old procedures?
#
# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Response_Procedure: optimized playbook or old procedures.
# Recovery_Time_Minutes: numerical recovery time in minutes.
# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: The recovery-time distributions are the same for both groups.
# H₁: The recovery-time distributions differ between the groups.
#
# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 3: Statistical Test and Justification
#
# A Mann–Whitney U test will compare recovery times between 50 teams using
# an optimized playbook and 50 different teams using old procedures. The
# groups are independent, each team contributes one numerical recovery-time
# measurement, and the synthetic recovery times are right-skewed. Following
# the master guideline, descriptive statistics, histograms, boxplots, and
# Shapiro–Wilk tests will document the distributions before analysis.
# A two-sided Mann–Whitney U test will be conducted at an alpha level of .05
# because the stated alternative hypothesis asks whether the distributions
# differ in either direction. Report each group’s median and interquartile
# range, the test statistic, p-value, and Cliff’s delta. Because group shapes
# and spreads may differ, interpret the result as a distributional comparison
# rather than solely a difference in medians.
#
# Alignment note: The previously supplied example used a one-sided test for
# faster optimized-playbook recovery. Your current Question 3 requires a
# two-sided test, so its p-value must be recalculated.
#
# ----------------------------------------------------------------
# 4. RESULTS — DATA, GRAPHS, DIAGNOSTICS AND TESTS
# ----------------------------------------------------------------
library(readxl)
library(ggpubr)
library(dplyr)
library(effsize)
data2 <- read_excel("C:/Users/hp/Desktop/digital Forensics/AA 5221/FInal_Project/Ransomware_Recovery.xlsx")
View(data2)
unique(data2$Response_Procedure)
## [1] "Optimized playbook" "Old procedures"
# Set the group order for the test and effect size.
data2$Response_Procedure <- factor(
data2$Response_Procedure,
levels = c("Optimized playbook", "Old procedures")
)
# Descriptive statistics
data2 %>%
group_by(Response_Procedure) %>%
summarise(
Mean = mean(Recovery_Time_Minutes, na.rm = TRUE),
Median = median(Recovery_Time_Minutes, na.rm = TRUE),
SD = sd(Recovery_Time_Minutes, na.rm = TRUE),
N = sum(!is.na(Recovery_Time_Minutes)),
.groups = "drop"
)
## # A tibble: 2 × 5
## Response_Procedure Mean Median SD N
## <fct> <dbl> <dbl> <dbl> <int>
## 1 Optimized playbook 121. 116. 65.9 50
## 2 Old procedures 156. 147. 89.8 50
hist(
data2$Recovery_Time_Minutes[
data2$Response_Procedure == "Optimized playbook"
],
breaks = 15,
col = "skyblue",
border = "white"
)

hist(
data2$Recovery_Time_Minutes[
data2$Response_Procedure == "Old procedures"
],
breaks = 15,
col = "firebrick",
border = "white"
)

# Describe the recovery-time distribution for each group.
ggboxplot(
data2,
x = "Response_Procedure",
y = "Recovery_Time_Minutes",
color = "Response_Procedure",
palette = "jco",
add = "jitter"
)

# Identify potential outliers in each group.
shapiro.test(
data2$Recovery_Time_Minutes[
data2$Response_Procedure == "Optimized playbook"
]
)
##
## Shapiro-Wilk normality test
##
## data: data2$Recovery_Time_Minutes[data2$Response_Procedure == "Optimized playbook"]
## W = 0.94907, p-value = 0.03126
shapiro.test(
data2$Recovery_Time_Minutes[
data2$Response_Procedure == "Old procedures"
]
)
##
## Shapiro-Wilk normality test
##
## data: data2$Recovery_Time_Minutes[data2$Response_Procedure == "Old procedures"]
## W = 0.94992, p-value = 0.03386
# Report the Shapiro-Wilk results for each group.
wilcox.test(
Recovery_Time_Minutes ~ Response_Procedure,
data = data2,
alternative = "two.sided",
exact = FALSE
)
##
## Wilcoxon rank sum test with continuity correction
##
## data: Recovery_Time_Minutes by Response_Procedure
## W = 971, p-value = 0.05487
## alternative hypothesis: true location shift is not equal to 0
mw_effect <- cliff.delta(
Recovery_Time_Minutes ~ Response_Procedure,
data = data2
)
print(mw_effect)
##
## Cliff's Delta
##
## delta estimate: -0.2232 (small)
## 95 percent confidence interval:
## lower upper
## -0.424451967 -0.000932864
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Mann-Whitney U test was conducted to determine if there was a difference in recovery times between optimized-playbook and old-procedure teams.
# Optimized-playbook recovery times (Mdn ≈ 116 minutes) were not significantly different from old-procedure recovery times (Mdn ≈ 147 minutes), W = 971, p = .055.
# The effect size was small, Cliff's Delta = −0.223.
# ----------------------------------------------------------------
# RESEARCH QUESTION 4
# ----------------------------------------------------------------
# Question 4
#
# Is ransomware response procedure associated with meeting the assumed
# 150-minute recovery benchmark?
#
# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Response_Procedure: optimized playbook or old procedures.
# Recovery_Outcome: met (<=150 minutes) or failed (>150 minutes).
# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: Response procedure and meeting the recovery benchmark are independent.
# H₁: Response procedure and meeting the recovery benchmark are associated.
#
# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 4: Statistical Test and Justification
#
# A Chi-Square Test of Independence will determine whether ransomware response
# procedure is associated with meeting the assumed 150-minute recovery
# benchmark. Both variables are categorical: procedure is classified as
# optimized playbook or old procedures, and recovery outcome as met the
# benchmark (≤150 minutes) or failed to meet the benchmark (>150 minutes).
# Following the master guideline, a frequency table and bar chart will
# summarize the categories before analysis. Each team must contribute to
# only one cell, and expected frequencies will be checked; the synthetic
# dataset has expected counts above five in every cell. The test will use
# an alpha level of .05. Report observed frequencies, group percentages,
# χ², degrees of freedom, p-value, and Cramer’s V. The 150-minute threshold
# is a hypothetical study benchmark, not a universal NIST requirement.
#
# ----------------------------------------------------------------
# 4. RESULTS — FIRST ORIGINAL CHI-SQUARE BLOCK
# ----------------------------------------------------------------
library(readxl)
library(ggplot2)
library(rcompanion)
##
## Attaching package: 'rcompanion'
## The following object is masked from 'package:effectsize':
##
## phi
# Chi-Square Test for Independence
MyTable <- table(
data2$Response_Procedure,
data2$Recovery_Outcome
)
MyTable
##
## Over 150 minutes Within 150 minutes
## Optimized playbook 14 36
## Old procedures 25 25
# Create a grouped bar chart.
barplot(
MyTable,
beside = TRUE,
col = rainbow(nrow(MyTable)),
legend.text = rownames(MyTable),
args.legend = list(x = "topright", bty = "n"),
main = "Recovery Outcome by Response Procedure",
xlab = "Recovery Outcome",
ylab = "Number of Teams"
)

# Chi-square test of independence.
# R applies Yates' continuity correction to this 2 × 2 table.
chi_result <- chisq.test(MyTable)
chi_result
##
## Pearson's Chi-squared test with Yates' continuity correction
##
## data: MyTable
## X-squared = 4.2034, df = 1, p-value = 0.04034
# Check expected counts.
chi_result$expected
##
## Over 150 minutes Within 150 minutes
## Optimized playbook 19.5 30.5
## Old procedures 19.5 30.5
# Cramer's V uses the uncorrected Pearson chi-square statistic.
rcompanion::cramerV(MyTable)
## Cramer V
## 0.2255
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Chi-Square Test of Independence was conducted to determine
# if there was an association between response procedure and
# recovery outcome.
# The results showed a significant association between the
# two variables, χ²(1, N = 100) = 4.203, p = .040,
# with Yates' continuity correction.
# The association was small, Cramer's V = .226.
# These data are synthetic.
# ----------------------------------------------------------------
# 4. RESULTS — SECOND ORIGINAL CHI-SQUARE BLOCK
# ----------------------------------------------------------------
library(readxl)
library(ggplot2)
library(rcompanion)
data2 <- read_excel("C:/Users/hp/Desktop/digital Forensics/AA 5221/FInal_Project/Ransomware_Recovery.xlsx")
View(data2)
# Create the contingency table.
MyTable <- table(
data2$Response_Procedure,
data2$Recovery_Outcome
)
MyTable
##
## Over 150 minutes Within 150 minutes
## Old procedures 25 25
## Optimized playbook 14 36
# Create a grouped bar chart.
barplot(
MyTable,
beside = TRUE,
col = rainbow(nrow(MyTable)),
legend.text = rownames(MyTable),
args.legend = list(x = "topright", bty = "n"),
main = "Recovery Outcome by Response Procedure",
xlab = "Recovery Outcome",
ylab = "Number of Teams"
)

# Chi-square test of independence.
# R applies Yates' continuity correction to this 2 × 2 table.
chi_result <- chisq.test(MyTable)
chi_result
##
## Pearson's Chi-squared test with Yates' continuity correction
##
## data: MyTable
## X-squared = 4.2034, df = 1, p-value = 0.04034
# Check expected counts.
chi_result$expected
##
## Over 150 minutes Within 150 minutes
## Old procedures 19.5 30.5
## Optimized playbook 19.5 30.5
# Cramer's V uses the uncorrected Pearson chi-square statistic.
rcompanion::cramerV(MyTable)
## Cramer V
## 0.2255
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Chi-Square Test of Independence was conducted to determine
# if there was an association between response procedure and
# recovery outcome.
# The results showed a significant association between the
# two variables, χ²(1, N = 100) = 4.203, p = .040,
# with Yates' continuity correction.
# The association was small, Cramer's V = .226.
# These data are synthetic.