library(readxl)
library(rcompanion)
library(ggplot2)
library(ggpubr)
library(effsize)
library(rstatix)
##
## Attaching package: 'rstatix'
## The following object is masked from 'package:stats':
##
## filter
library(effectsize)
##
## Attaching package: 'effectsize'
## The following objects are masked from 'package:rstatix':
##
## cohens_d, eta_squared, omega_squared
## The following object is masked from 'package:rcompanion':
##
## phi
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(ggpubr)
PGMT1 <- read_excel("C:/Users/winni/OneDrive - Saint Louis University/AA 5221/Project/Project_Management_Synthetic_Data.xlsx")
View(PGMT1)
PGMT1_1 <- read_excel("C:/Users/winni/OneDrive - Saint Louis University/AA 5221/Project/PGMT1_1.xlsx")
View(PGMT1_1)
PGMT2 <- read_excel("C:/Users/winni/OneDrive - Saint Louis University/AA 5221/Project/Independent_TTest_Project_Management_Data.xlsx")
View(PGMT2)
table(PGMT1$Approach, PGMT1$On_Time_Delivery)
##
## No Yes
## Agile 1 9
## Traditional 3 7
Project_Delivered_On_Time <- table(PGMT1$Approach, PGMT1$On_Time_Delivery)
Project_Delivered_On_Time
##
## No Yes
## Agile 1 9
## Traditional 3 7
barplot(Project_Delivered_On_Time,
main = "Timely Delivery of Project",
ylab = "Count",
xlab = "Approach",
beside = TRUE,
col = rainbow(nrow(Project_Delivered_On_Time)),
legend = rownames(Project_Delivered_On_Time),
args.legend = list(x = "topright"))

chi_result <- chisq.test(Project_Delivered_On_Time)
## Warning in chisq.test(Project_Delivered_On_Time): Chi-squared approximation may
## be incorrect
chi_result
##
## Pearson's Chi-squared test with Yates' continuity correction
##
## data: Project_Delivered_On_Time
## X-squared = 0.3125, df = 1, p-value = 0.5762
rcompanion::cramerV(Project_Delivered_On_Time)
## Cramer V
## 0.25
# A Chi-Square Test of Independence was conducted to determine if there was an association between Approach (Agile or Traditional) and On-Time Delivery of Project.
# The results showed that there was no statistically significant association between the two variables, χ²(1) = 0.3125, p > .05
# The association was moderate (Cramer's V = .25).
#+++++++++++++++++++++++++
#Correlation
ggscatter(
PGMT1,
x = "Training_Hours",
y = "Productivity_Improvement",
add = "reg.line",
xlab = "Hours",
ylab = "Productivity Improvement"
)

#variable 1
hist(PGMT1$Training_Hours,
main = "Training Hours",
breaks = 20,
col = "lightblue",
border = "white",
xlab = "Hours",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

#variable 2
hist(PGMT1$Productivity_Improvement,
main = "Training Hours",
breaks = 20,
col = "lightcoral",
border = "white",
xlab = "Hours",
cex.main = 1,
cex.axis = 1,
cex.lab = 1)

shapiro.test(PGMT1$Training_Hours)
##
## Shapiro-Wilk normality test
##
## data: PGMT1$Training_Hours
## W = 0.93809, p-value = 0.2206
shapiro.test(PGMT1$Productivity_Improvement)
##
## Shapiro-Wilk normality test
##
## data: PGMT1$Productivity_Improvement
## W = 0.94314, p-value = 0.2746
# Variable 1: Training Hours
# The variable is normally distributed (p = .22).
# Variable 2: Productivity Improvement
# The variable is normally distributed (p = .27).
cor.test(
PGMT1$Training_Hours,
PGMT1$Productivity_Improvement,
method = "pearson",
#exact = FALSE
)
##
## Pearson's product-moment correlation
##
## data: PGMT1$Training_Hours and PGMT1$Productivity_Improvement
## t = 16.378, df = 18, p-value = 2.938e-12
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
## 0.9193612 0.9875286
## sample estimates:
## cor
## 0.9680457
# A Pearson correlation was conducted to test the relationship between Training Hours and Productivity Improvement.
# There was a statistically significant relationship between the two variables,r(18) = .97, p < .001.
# The relationship was positive and strong.
# As Training Hours increased, Productivity Improvement increased.
#++++++++++++++++++++++++++++
Productivity_Before <- PGMT1_1$Productivity_Pre
Productivity_After <- PGMT1_1$Productivity_Post
Differences <- Productivity_After - Productivity_Before
mean(Productivity_After, na.rm = TRUE)
## [1] 86.81
median(Productivity_After, na.rm = TRUE)
## [1] 86.2
sd(Productivity_After, na.rm = TRUE)
## [1] 8.121501
mean(Productivity_Before, na.rm = TRUE)
## [1] 61.1
median(Productivity_Before, na.rm = TRUE)
## [1] 61.65
sd(Productivity_Before, na.rm = TRUE)
## [1] 4.956253
# Productivity After:
# M = 86.81, Mdn = 86.20, SD = 8.12
# Productivity Before:
# M = 61.10, Mdn = 61.65, SD = 4.96
hist(Differences,
breaks = 15,
col = "blue",
border = "white")

boxplot(Differences,
main = "Distribution of Productivity (After - Before)",
ylab = "Difference in Productivity",
col = "blue",
border = "darkblue")

shapiro.test(Differences)
##
## Shapiro-Wilk normality test
##
## data: Differences
## W = 0.95746, p-value = 0.7566
# The difference scores are normally distributed (p = .76).
# Therefore, a paired-samples t-test was conducted.
t.test(Productivity_After, Productivity_Before, paired = TRUE, na.action = na.omit)
##
## Paired t-test
##
## data: Productivity_After and Productivity_Before
## t = 17.942, df = 9, p-value = 2.361e-08
## alternative hypothesis: true mean difference is not equal to 0
## 95 percent confidence interval:
## 22.46837 28.95163
## sample estimates:
## mean difference
## 25.71
# A paired-samples t-test was conducted to determine if there was a difference in productivity before training and after training.
#Productivity after training (M = 86.81, SD = 8.12) was significantly higher than productivity before training (M = 61.10, SD = 4.96), t(9) = 17.94, p < .001.
#+++++++++++++++++++++++
PGMT2 %>%
group_by(Approach) %>%
summarise(
Mean = mean(Productivity_Post, na.rm = TRUE),
Median = median(Productivity_Post, na.rm = TRUE),
SD = sd(Productivity_Post, na.rm = TRUE),
N = n()
)
## # A tibble: 2 × 5
## Approach Mean Median SD N
## <chr> <dbl> <dbl> <dbl> <int>
## 1 Agile 77.1 76.8 5.56 10
## 2 Traditional 62.8 61.0 6.09 10
hist(PGMT2$Productivity_Post[PGMT2$Approach == "Agile"],
breaks = 15,
main = "Histogram of Agile Productivity Post Training",
xlab = "Agile",
col = "skyblue",
border = "white")

hist(PGMT2$Productivity_Post[PGMT2$Approach == "Traditional"],
breaks = 15,
main = "Histogram of Traditional Productivity Post Training",
xlab = "Traditional",
col = "firebrick",
border = "white")

ggboxplot(PGMT2, x = "Approach", y = "Productivity_Post",
color = "blue",
palette = "jco",
add = "jitter")

shapiro.test(PGMT2$Productivity_Post[PGMT2$Approach == "Agile"])
##
## Shapiro-Wilk normality test
##
## data: PGMT2$Productivity_Post[PGMT2$Approach == "Agile"]
## W = 0.94297, p-value = 0.5865
shapiro.test(PGMT2$Productivity_Post[PGMT2$Approach == "Traditional"])
##
## Shapiro-Wilk normality test
##
## data: PGMT2$Productivity_Post[PGMT2$Approach == "Traditional"]
## W = 0.95521, p-value = 0.7302
# Group 1: Agile
# The variable is normally distributed (p = .59).
# Group 2: Traditional
# The variable is normally distributed (p = .73).
# Since both groups are normally distributed, an independent-samples t-test will be conducted.
t.test(Productivity_Post ~ Approach, data = PGMT2)
##
## Welch Two Sample t-test
##
## data: Productivity_Post by Approach
## t = 5.4899, df = 17.854, p-value = 3.351e-05
## alternative hypothesis: true difference in means between group Agile and group Traditional is not equal to 0
## 95 percent confidence interval:
## 8.830553 19.789447
## sample estimates:
## mean in group Agile mean in group Traditional
## 77.06 62.75
cohen.d(PGMT2$Productivity_Post, PGMT2$Approach)
##
## Cohen's d
##
## d estimate: 2.45517 (large)
## 95 percent confidence interval:
## lower upper
## 1.211012 3.699328
# An Independent T-Test was conducted to determine if there was a difference in Productivity Post scores between Agile and Traditional approaches.
# Agile scores (M = 77.06, SD = 5.56) were significantly different from Traditional scores (M = 62.75, SD = 6.09), t(17.85) = 5.49, p < .001.
# The effect size was large, Cohen's d = 2.46.