# FINAL PROJECT AA 5221
#
# Research Problem Statement 1: Seasonal Household Electricity Consumption
#
# Household electricity use may vary between summer and winter because cooling
# and heating create different energy demands. However, a seasonal difference
# in average consumption and a relationship between households’ consumption
# across seasons are separate questions. This study uses the synthetic
# Household_Electricity.xlsx dataset, containing paired summer and winter
# consumption measurements for 300 households in a 2026 Saint Louis, Missouri
# scenario. It examines whether households with higher summer consumption also
# tend to have higher winter consumption and whether mean consumption differs
# between the two seasons. Consumption is measured in kilowatt-hours over
# comparable periods.
#
# We will be using Pearson or Spearman correlation to examine the relationship
# and a paired-samples t-test or Wilcoxon signed-rank test to examine seasonal
# differences, depending on the relevant assumptions.
#
# Data Sets
#
# The synthetic data sets where modelled using Chat GPT
#
# Question 1 — Dataset: Household_Electricity.xlsx, using Summer_kWh and
# Winter_kWh to examine correlation.
#
# Question 2 — Dataset: Household_Electricity.xlsx, using paired Summer_kWh
# and Winter_kWh measurements to compare seasonal consumption.
#
# Question 3 — Dataset: Ransomware_Recovery.xlsx, using Response_Procedure
# and Recovery_Time_Minutes to compare recovery times between independent groups.
#
# Question 4 — Dataset: Ransomware_Recovery.xlsx, using Response_Procedure
# and Recovery_Outcome to examine association with meeting the assumed
# 150-minute benchmark.





# ----------------------------------------------------------------
# RESEARCH QUESTION 1
# ----------------------------------------------------------------
# QUESTION 1
#
# Household Electricity Data
#
# Is there a relationship between 2026 summer and winter electricity consumption
# among 300 households in Saint Louis, Missouri?
#
# In practical terms: Do households that use more electricity in summer also
# tend to use more electricity in winter?
#

# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: There is no correlation between summer and winter household electricity
# consumption.
# H₁: There is a correlation between summer and winter household electricity
# consumption.
#

# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Each household contributes two numerical measurements: summer consumption
# and winter consumption, measured in comparable kWh periods.
#

# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 1: Statistical Test and Justification
#
# A Pearson correlation will examine the strength and direction of the linear
# relationship between summer and winter electricity consumption. Both
# variables are numerical, and each household provides one paired set of
# measurements. Following the master guideline, descriptive statistics,
# histograms, boxplots, Shapiro–Wilk tests, and a scatterplot will be examined
# before selecting the test. Pearson correlation is appropriate when the
# relationship is linear and its assumptions are reasonably satisfied.
# Spearman correlation will be used if the data are unsuitable for Pearson
# but show a monotonic relationship. A two-sided test with an alpha level of
# .05 will match the hypothesis that a correlation exists in either direction.
# Report the correlation coefficient, p-value, and strength and direction
# of the relationship.
#

# ----------------------------------------------------------------
# 4. RESULTS — DATA, GRAPHS, DIAGNOSTICS AND TESTS
# ----------------------------------------------------------------


library(readxl)
library(ggpubr)
## Loading required package: ggplot2
library(readxl)
data <- read_excel("C:/Users/hp/Desktop/digital Forensics/AA 5221/FInal_Project/Household_Electricity.xlsx")
View(data)

ggscatter(
  data,
  x = "Summer_kWh",
  y = "Winter_kWh",
  add = "reg.line",
  xlab = "Summer Electricity Consumption (kWh)",
  ylab = "Winter Electricity Consumption (kWh)"
)

# Describe whether the relationship is linear.
# Describe whether the relationship is positive.
# Check for potential outliers.

mean(data$Summer_kWh)
## [1] 1110.149
sd(data$Summer_kWh)
## [1] 194.8264
median(data$Summer_kWh)
## [1] 1107.355
mean(data$Winter_kWh)
## [1] 910.9583
sd(data$Winter_kWh)
## [1] 188.4099
median(data$Winter_kWh)
## [1] 920.315
hist(data$Summer_kWh)

hist(data$Winter_kWh)

# Variable 1: Summer Electricity Consumption
# Summer Electricity Consumption looks normally distributed.
# Summer Electricity Consumption data is symmetrical.
# Summer Electricity Consumption data has a proper bell curve.

# Variable 2: Winter Electricity Consumption
# Winter Electricity Consumption data looks normally distributed.
# Winter Electricity Consumption data is symmetrical.
# Winter Electricity Consumption data has a proper bell curve.

shapiro.test(data$Summer_kWh)
## 
##  Shapiro-Wilk normality test
## 
## data:  data$Summer_kWh
## W = 0.9956, p-value = 0.5612
shapiro.test(data$Winter_kWh)
## 
##  Shapiro-Wilk normality test
## 
## data:  data$Winter_kWh
## W = 0.99191, p-value = 0.1004
# Variable 1: Summer Electricity Consumption
# The variable is normally distributed (p = .5612).

# Variable 2: Winter Electricity Consumption
# The variable is normally distributed (p = .1004).

cor.test(data$Summer_kWh, data$Winter_kWh, method = "pearson")
## 
##  Pearson's product-moment correlation
## 
## data:  data$Summer_kWh and data$Winter_kWh
## t = 19.246, df = 298, p-value < 2.2e-16
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
##  0.6892941 0.7909885
## sample estimates:
##       cor 
## 0.7444277
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Pearson correlation was conducted to test the relationship between summer and winter electricity consumption.
# There was a statistically significant relationship between the two variables,
# r(298) = .74, p < .001.
# The relationship was positive and strong.
# As summer electricity consumption increased, winter electricity consumption increased


# ----------------------------------------------------------------
# RESEARCH QUESTION 2
# ----------------------------------------------------------------
# QUESTION 2
#
# House hold Electricity Data
#
# Is there a difference in mean electricity consumption between summer and
# winter among 300 households in Saint Louis, Missouri?
#
# In practical terms: Do the same households use more electricity, on average,
# in summer or winter?
#

# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: The population mean difference between summer and winter electricity
# consumption is zero
# H₁: The population mean difference between summer and winter electricity
# consumption is not zero
#

# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Each household contributes two measurements: summer consumption and winter
# consumption, measured in kWh over comparable periods.
#

# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 2: Statistical Test and Justification
#
# A dependent t-test, also called a paired-samples t-test, will compare mean
# summer and winter electricity consumption because the same 300 households
# are measured in both seasons. Following the master guideline, a difference
# score will be calculated for each household as summer consumption minus
# winter consumption. A histogram, boxplot, and Shapiro–Wilk test will assess
# these difference scores rather than the two seasonal distributions
# separately. If the assumptions are reasonably satisfied, a two-sided paired
# t-test will be conducted at an alpha level of .05. Report each season’s mean
# and standard deviation, the mean difference and its confidence interval,
# t, degrees of freedom, p-value, and paired Cohen’s d. If the difference
# scores are unsuitable for a t-test, Wilcoxon signed-rank may be used,
# provided its symmetry assumption is reasonable; this alternative does
# not directly test a difference in means.
#

# ----------------------------------------------------------------
# 4. RESULTS — DATA, GRAPHS, DIAGNOSTICS AND TESTS
# ----------------------------------------------------------------
library(readxl)
library(ggpubr)
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(effectsize)
library(effsize)
library(rstatix)
## 
## Attaching package: 'rstatix'
## The following objects are masked from 'package:effectsize':
## 
##     cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:stats':
## 
##     filter
View(data)

Winter <- data$Winter_kWh
Summer <- data$Summer_kWh

Differences <- Summer - Winter

mean(Winter, na.rm = TRUE)
## [1] 910.9583
median(Winter, na.rm = TRUE)
## [1] 920.315
sd(Winter, na.rm = TRUE)
## [1] 188.4099
mean(Summer, na.rm = TRUE)
## [1] 1110.149
median(Summer, na.rm = TRUE)
## [1] 1107.355
sd(Summer, na.rm = TRUE)
## [1] 194.8264
hist(
  Differences,
  breaks = 15,
  col = "blue",
  border = "white"
)

# Describe the distribution of the summer-minus-winter difference scores.

boxplot(
  Differences,
  main = "Distribution of Electricity Differences (Summer - Winter)",
  ylab = "Difference in Consumption (kWh)",
  col = "blue",
  border = "darkblue"
)

# No outliers in the difference scores.

shapiro.test(Differences)
## 
##  Shapiro-Wilk normality test
## 
## data:  Differences
## W = 0.99755, p-value = 0.9363
# The the distribution of  differences in Summer and Winter is normal.
# The normality assumption concerns the difference scores.

t.test(
  Summer,
  Winter,
  paired = TRUE,
  na.action = na.omit
)
## 
##  Paired t-test
## 
## data:  Summer and Winter
## t = 25.16, df = 299, p-value < 2.2e-16
## alternative hypothesis: true mean difference is not equal to 0
## 95 percent confidence interval:
##  183.6104 214.7708
## sample estimates:
## mean difference 
##        199.1906
# Paired Cohen's d: mean difference divided by the SD of differences.
cohens_d <- mean(Differences, na.rm = TRUE) /
  sd(Differences, na.rm = TRUE)

cohens_d
## [1] 1.452597
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Dependent T-Test was conducted to determine if there was a difference in electricity consumption between summer and winter.
# Summer consumption (M = 1110.15, SD = 194.83) was significantly higher than winter consumption (M = 910.96, SD = 188.41), t(299) = 25.16, p < .001.
# The effect size was very large, paired Cohen's d = 1.453.

# Research Problem Statement 2: Ransomware Response Procedures and Data Recovery
#
# Organizations need to understand whether an optimized ransomware response
# playbook improves data recovery compared with existing procedures. Recovery
# performance can be examined through both the time teams take to restore
# data and whether they meet a defined recovery target. This study uses the
# synthetic Ransomware_Recovery.xlsx dataset, containing recovery times for
# 100 independent cybersecurity teams: 50 using an optimized playbook and
# 50 using old procedures under assumed comparable exercise conditions.
# It examines whether recovery-time distributions differ between the groups
# and whether response procedure is associated with meeting an assumed
# 150-minute recovery benchmark. The benchmark is a hypothetical study target,
# not a universal NIST requirement.
#

# ----------------------------------------------------------------
# RESEARCH QUESTION 3
# ----------------------------------------------------------------
# Question 3
#
# Do data recovery times after a simulated ransomware breach differ between
# cybersecurity teams using an optimized playbook and teams using old procedures?
#

# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Response_Procedure: optimized playbook or old procedures.
# Recovery_Time_Minutes: numerical recovery time in minutes.

# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: The recovery-time distributions are the same for both groups.
# H₁: The recovery-time distributions differ between the groups.
#

# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 3: Statistical Test and Justification
#
# A Mann–Whitney U test will compare recovery times between 50 teams using
# an optimized playbook and 50 different teams using old procedures. The
# groups are independent, each team contributes one numerical recovery-time
# measurement, and the synthetic recovery times are right-skewed. Following
# the master guideline, descriptive statistics, histograms, boxplots, and
# Shapiro–Wilk tests will document the distributions before analysis.
# A two-sided Mann–Whitney U test will be conducted at an alpha level of .05
# because the stated alternative hypothesis asks whether the distributions
# differ in either direction. Report each group’s median and interquartile
# range, the test statistic, p-value, and Cliff’s delta. Because group shapes
# and spreads may differ, interpret the result as a distributional comparison
# rather than solely a difference in medians.
#
# Alignment note: The previously supplied example used a one-sided test for
# faster optimized-playbook recovery. Your current Question 3 requires a
# two-sided test, so its p-value must be recalculated.
#

# ----------------------------------------------------------------
# 4. RESULTS — DATA, GRAPHS, DIAGNOSTICS AND TESTS
# ----------------------------------------------------------------
library(readxl)
library(ggpubr)
library(dplyr)
library(effsize)

data2 <- read_excel("C:/Users/hp/Desktop/digital Forensics/AA 5221/FInal_Project/Ransomware_Recovery.xlsx")
View(data2)


unique(data2$Response_Procedure)
## [1] "Optimized playbook" "Old procedures"
# Set the group order for the test and effect size.
data2$Response_Procedure <- factor(
  data2$Response_Procedure,
  levels = c("Optimized playbook", "Old procedures")
)

# Descriptive statistics
data2 %>%
  group_by(Response_Procedure) %>%
  summarise(
    Mean = mean(Recovery_Time_Minutes, na.rm = TRUE),
    Median = median(Recovery_Time_Minutes, na.rm = TRUE),
    SD = sd(Recovery_Time_Minutes, na.rm = TRUE),
    N = sum(!is.na(Recovery_Time_Minutes)),
    .groups = "drop"
  )
## # A tibble: 2 × 5
##   Response_Procedure  Mean Median    SD     N
##   <fct>              <dbl>  <dbl> <dbl> <int>
## 1 Optimized playbook  121.   116.  65.9    50
## 2 Old procedures      156.   147.  89.8    50
hist(
  data2$Recovery_Time_Minutes[
    data2$Response_Procedure == "Optimized playbook"
  ],
  breaks = 15,
  col = "skyblue",
  border = "white"
)

hist(
  data2$Recovery_Time_Minutes[
    data2$Response_Procedure == "Old procedures"
  ],
  breaks = 15,
  col = "firebrick",
  border = "white"
)

# Describe the recovery-time distribution for each group.

ggboxplot(
  data2,
  x = "Response_Procedure",
  y = "Recovery_Time_Minutes",
  color = "Response_Procedure",
  palette = "jco",
  add = "jitter"
)

# Identify potential outliers in each group.

shapiro.test(
  data2$Recovery_Time_Minutes[
    data2$Response_Procedure == "Optimized playbook"
  ]
)
## 
##  Shapiro-Wilk normality test
## 
## data:  data2$Recovery_Time_Minutes[data2$Response_Procedure == "Optimized playbook"]
## W = 0.94907, p-value = 0.03126
shapiro.test(
  data2$Recovery_Time_Minutes[
    data2$Response_Procedure == "Old procedures"
  ]
)
## 
##  Shapiro-Wilk normality test
## 
## data:  data2$Recovery_Time_Minutes[data2$Response_Procedure == "Old procedures"]
## W = 0.94992, p-value = 0.03386
# Report the Shapiro-Wilk results for each group.

wilcox.test(
  Recovery_Time_Minutes ~ Response_Procedure,
  data = data2,
  alternative = "two.sided",
  exact = FALSE
)
## 
##  Wilcoxon rank sum test with continuity correction
## 
## data:  Recovery_Time_Minutes by Response_Procedure
## W = 971, p-value = 0.05487
## alternative hypothesis: true location shift is not equal to 0
mw_effect <- cliff.delta(
  Recovery_Time_Minutes ~ Response_Procedure,
  data = data2
)

print(mw_effect)
## 
## Cliff's Delta
## 
## delta estimate: -0.2232 (small)
## 95 percent confidence interval:
##        lower        upper 
## -0.424451967 -0.000932864
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Mann-Whitney U test was conducted to determine if there was a difference in recovery times between optimized-playbook and old-procedure teams.
# Optimized-playbook recovery times (Mdn ≈ 116 minutes) were not significantly different from old-procedure recovery times (Mdn ≈ 147 minutes), W = 971, p = .055.
# The effect size was small, Cliff's Delta = −0.223.



# ----------------------------------------------------------------
# RESEARCH QUESTION 4
# ----------------------------------------------------------------
# Question 4
#
# Is ransomware response procedure associated with meeting the assumed
# 150-minute recovery benchmark?
#

# ----------------------------------------------------------------
# 1. VARIABLES
# ----------------------------------------------------------------
# Response_Procedure: optimized playbook or old procedures.
# Recovery_Outcome: met (<=150 minutes) or failed (>150 minutes).

# ----------------------------------------------------------------
# 2. HYPOTHESES
# ----------------------------------------------------------------
# H₀: Response procedure and meeting the recovery benchmark are independent.
# H₁: Response procedure and meeting the recovery benchmark are associated.
#

# ----------------------------------------------------------------
# 3. STATISTICAL TEST
# ----------------------------------------------------------------
# Question 4: Statistical Test and Justification
#
# A Chi-Square Test of Independence will determine whether ransomware response
# procedure is associated with meeting the assumed 150-minute recovery
# benchmark. Both variables are categorical: procedure is classified as
# optimized playbook or old procedures, and recovery outcome as met the
# benchmark (≤150 minutes) or failed to meet the benchmark (>150 minutes).
# Following the master guideline, a frequency table and bar chart will
# summarize the categories before analysis. Each team must contribute to
# only one cell, and expected frequencies will be checked; the synthetic
# dataset has expected counts above five in every cell. The test will use
# an alpha level of .05. Report observed frequencies, group percentages,
# χ², degrees of freedom, p-value, and Cramer’s V. The 150-minute threshold
# is a hypothetical study benchmark, not a universal NIST requirement.
#

# ----------------------------------------------------------------
# 4. RESULTS — FIRST ORIGINAL CHI-SQUARE BLOCK
# ----------------------------------------------------------------
library(readxl)
library(ggplot2)
library(rcompanion)
## 
## Attaching package: 'rcompanion'
## The following object is masked from 'package:effectsize':
## 
##     phi
# Chi-Square Test for Independence

MyTable <- table(
  data2$Response_Procedure,
  data2$Recovery_Outcome
)

MyTable
##                     
##                      Over 150 minutes Within 150 minutes
##   Optimized playbook               14                 36
##   Old procedures                   25                 25
# Create a grouped bar chart.
barplot(
  MyTable,
  beside = TRUE,
  col = rainbow(nrow(MyTable)),
  legend.text = rownames(MyTable),
  args.legend = list(x = "topright", bty = "n"),
  main = "Recovery Outcome by Response Procedure",
  xlab = "Recovery Outcome",
  ylab = "Number of Teams"
)

# Chi-square test of independence.
# R applies Yates' continuity correction to this 2 × 2 table.
chi_result <- chisq.test(MyTable)
chi_result
## 
##  Pearson's Chi-squared test with Yates' continuity correction
## 
## data:  MyTable
## X-squared = 4.2034, df = 1, p-value = 0.04034
# Check expected counts.
chi_result$expected
##                     
##                      Over 150 minutes Within 150 minutes
##   Optimized playbook             19.5               30.5
##   Old procedures                 19.5               30.5
# Cramer's V uses the uncorrected Pearson chi-square statistic.
rcompanion::cramerV(MyTable)
## Cramer V 
##   0.2255
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Chi-Square Test of Independence was conducted to determine
# if there was an association between response procedure and
# recovery outcome.
# The results showed a significant association between the
# two variables, χ²(1, N = 100) = 4.203, p = .040,
# with Yates' continuity correction.
# The association was small, Cramer's V = .226.
# These data are synthetic.




# ----------------------------------------------------------------
# 4. RESULTS — SECOND ORIGINAL CHI-SQUARE BLOCK
# ----------------------------------------------------------------
library(readxl)
library(ggplot2)
library(rcompanion)

data2 <- read_excel("C:/Users/hp/Desktop/digital Forensics/AA 5221/FInal_Project/Ransomware_Recovery.xlsx")
View(data2)

# Create the contingency table.
MyTable <- table(
  data2$Response_Procedure,
  data2$Recovery_Outcome
)

MyTable
##                     
##                      Over 150 minutes Within 150 minutes
##   Old procedures                   25                 25
##   Optimized playbook               14                 36
# Create a grouped bar chart.
barplot(
  MyTable,
  beside = TRUE,
  col = rainbow(nrow(MyTable)),
  legend.text = rownames(MyTable),
  args.legend = list(x = "topright", bty = "n"),
  main = "Recovery Outcome by Response Procedure",
  xlab = "Recovery Outcome",
  ylab = "Number of Teams"
)

# Chi-square test of independence.
# R applies Yates' continuity correction to this 2 × 2 table.
chi_result <- chisq.test(MyTable)
chi_result
## 
##  Pearson's Chi-squared test with Yates' continuity correction
## 
## data:  MyTable
## X-squared = 4.2034, df = 1, p-value = 0.04034
# Check expected counts.
chi_result$expected
##                     
##                      Over 150 minutes Within 150 minutes
##   Old procedures                 19.5               30.5
##   Optimized playbook             19.5               30.5
# Cramer's V uses the uncorrected Pearson chi-square statistic.
rcompanion::cramerV(MyTable)
## Cramer V 
##   0.2255
# ----------------------------------------------------------------
# 5. INTERPRETATION — EXTRACTED ORIGINAL REPORTING COMMENTS
# ----------------------------------------------------------------
# A Chi-Square Test of Independence was conducted to determine
# if there was an association between response procedure and
# recovery outcome.
# The results showed a significant association between the
# two variables, χ²(1, N = 100) = 4.203, p = .040,
# with Yates' continuity correction.
# The association was small, Cramer's V = .226.
# These data are synthetic.