library(readxl)
library(ggpubr)
## Loading required package: ggplot2
A5Q1 <- read_excel("//apporto.com/dfs/SLU/Users/hannahsmith3_slu/Downloads/A5Q1.xlsx")
ggscatter(
A5Q1,
x = "age",
y = "education",
add = "reg.line",
xlab = "age",
ylab = "education"
)

#The relationship is linear.
#The relationship is positive.
#There are outliers.
mean(A5Q1$age)
## [1] 35.32634
sd(A5Q1$age)
## [1] 11.45344
median(A5Q1$age)
## [1] 35.79811
mean(A5Q1$education)
## [1] 13.82705
sd(A5Q1$education)
## [1] 2.595901
median(A5Q1$education)
## [1] 14.02915
hist(A5Q1$age,
main="age",
breaks=20,
col="lightblue",
border="white")

#Variable 1: Age
#The variable looks normally distrubuted.
#The data is negatively skewed.
#The data has a proper bell curve.
hist(A5Q1$education,
main="education",
breaks=20,
col="lightblue",
border="white")

#Variable 2: Education
#The variable looks normally distrubuted.
#The data is negatively skewed.
#The data has a proper bell curve.
shapiro.test(A5Q1$age)
##
## Shapiro-Wilk normality test
##
## data: A5Q1$age
## W = 0.99194, p-value = 0.5581
#Variable 1: Age
#The variable is normally distrubuted (p=.55).
shapiro.test(A5Q1$education)
##
## Shapiro-Wilk normality test
##
## data: A5Q1$education
## W = 0.9908, p-value = 0.4385
#Variable 2: Education
#The variable is normally distrubuted (p=.43).
cor.test(
A5Q1$age,
A5Q1$education,
method = "pearson"
)
##
## Pearson's product-moment correlation
##
## data: A5Q1$age and A5Q1$education
## t = 7.4066, df = 148, p-value = 9.113e-12
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
## 0.3924728 0.6279534
## sample estimates:
## cor
## 0.5200256
#A Pearson correlation was conducted to test the relationship between age (M = 35.33, SD = 11.45) and education (M = 13.83, SD = 2.60)
#There was a statistically significant relationship between the two variables, r(148) = .52, p = <.001
#The relationship was positive and strong.
#As age increased, education increased.