library(readxl)
library(ggpubr)
## Loading required package: ggplot2
A5Q1 <- read_excel("C:/Users/nehab/OneDrive/A5221/Assignment 5/A5Q1.xlsx")
ggscatter(
A5Q1,
x = "age",
y = "education",
add = "reg.line",
xlab = "Age",
ylab = "Years of Education Completed"
)

# The relationship is linear.
# The relationship is positive.
# There are no meaningful outliers.
mean(A5Q1$age)
## [1] 35.32634
sd(A5Q1$age)
## [1] 11.45344
median(A5Q1$age)
## [1] 35.79811
mean(A5Q1$education)
## [1] 13.82705
sd(A5Q1$education)
## [1] 2.595901
median(A5Q1$education)
## [1] 14.02915
hist(
A5Q1$age,
main = "Distribution of Age",
xlab = "Age"
)

hist(
A5Q1$education,
main = "Distribution of Education",
xlab = "Years of Education Completed"
)

# Age looks approximately normally distributed.
# The age distribution is reasonably symmetrical.
# Education looks approximately normally distributed.
# The education distribution is reasonably symmetrical.
shapiro.test(A5Q1$age)
##
## Shapiro-Wilk normality test
##
## data: A5Q1$age
## W = 0.99194, p-value = 0.5581
shapiro.test(A5Q1$education)
##
## Shapiro-Wilk normality test
##
## data: A5Q1$education
## W = 0.9908, p-value = 0.4385
# Age is normally distributed because p = .558.
# Education is normally distributed because p = .439.
# Both p-values are greater than .05.
# Therefore, a Pearson correlation will be used.
cor.test(A5Q1$age, A5Q1$education, method = "pearson")
##
## Pearson's product-moment correlation
##
## data: A5Q1$age and A5Q1$education
## t = 7.4066, df = 148, p-value = 9.113e-12
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
## 0.3924728 0.6279534
## sample estimates:
## cor
## 0.5200256
# A Pearson correlation was conducted to examine the relationship
# between age and years of education completed.
# There was a statistically significant positive relationship,
# r(148) = .52, p < .001.
# The relationship was strong.
# As age increased, years of education completed tended to increase.