library(readxl)
library(ggpubr)
## Loading required package: ggplot2
data <- read_excel("C:/Users/hp/Desktop/digital Forensics/AA 5221/Asignment_5/A5Q1(Sheet1).xls")
View(data)
ggscatter(
  data,
  x = "age",
  y = "education",
  add = "reg.line",
  xlab = "Age",
  ylab = "Education"
)

# The relationship is linear.
# The relationship is positive.
# There are no outliers.

mean(data$age)
## [1] 35.32634
sd(data$age)
## [1] 11.45344
median(data$age)
## [1] 35.79811
mean(data$education)
## [1] 13.82705
sd(data$education)
## [1] 2.595901
median(data$education)
## [1] 14.02915
hist(data$age)

hist(data$education)

# Variable 1: Age
# The variable looks normally distributed.
# The data is symmetrical.
# The data has a proper bell curve.

# Variable 2: Education
# The variable looks normally distributed.
# The data is symmetrical.
# The data has a proper bell curve.

shapiro.test(data$age)
## 
##  Shapiro-Wilk normality test
## 
## data:  data$age
## W = 0.99194, p-value = 0.5581
shapiro.test(data$education)
## 
##  Shapiro-Wilk normality test
## 
## data:  data$education
## W = 0.9908, p-value = 0.4385
# Variable 1: Age
# The variable is normally distributed (p = .5581).

# Variable 2: Education
# The variable is normally distributed (p = .4385).

cor.test(data$age, data$education, method = "pearson")
## 
##  Pearson's product-moment correlation
## 
## data:  data$age and data$education
## t = 7.4066, df = 148, p-value = 9.113e-12
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
##  0.3924728 0.6279534
## sample estimates:
##       cor 
## 0.5200256
# A Pearson correlation was conducted to test the relationship between age and education.

# There was a statistically significant relationship between the two variables,
# r(148) = .52, p < .001.

# The relationship was positive and moderate.

# As age increased, education increased.