library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.2.0
## ✔ forcats 1.0.1 ✔ stringr 1.6.0
## ✔ ggplot2 4.0.3 ✔ tibble 3.3.1
## ✔ lubridate 1.9.5 ✔ tidyr 1.3.2
## ✔ purrr 1.2.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(readxl)
district <-read_excel("district (2).xls")
district_example<-district%>% select(DA0AT21R,DA0CSA21R) %>% arrange(-DA0AT21R,DA0CSA21R)
district_example
## # A tibble: 1,207 × 2
## DA0AT21R DA0CSA21R
## <dbl> <dbl>
## 1 100 1241
## 2 100 NA
## 3 99.9 NA
## 4 99.7 NA
## 5 99.6 1048
## 6 99.6 1056
## 7 99.3 NA
## 8 99.3 NA
## 9 99.1 NA
## 10 99 997
## # ℹ 1,197 more rows
cor(district_example)
## DA0AT21R DA0CSA21R
## DA0AT21R 1 NA
## DA0CSA21R NA 1
cor(district_example, use = "pairwise.complete.obs")
## DA0AT21R DA0CSA21R
## DA0AT21R 1.0000000 0.1247348
## DA0CSA21R 0.1247348 1.0000000
There is a weak correlation between the attendance rate and the SAT average score.
district_example1<-district%>% select(DPSTKIDR,DA0CSA21R) %>% arrange(-DPSTKIDR,DA0CSA21R)
district_example1
## # A tibble: 1,207 × 2
## DPSTKIDR DA0CSA21R
## <dbl> <dbl>
## 1 37.3 971
## 2 26.9 NA
## 3 26.3 NA
## 4 24.9 -1
## 5 24.3 NA
## 6 23.2 923
## 7 23 1010
## 8 22.6 NA
## 9 22.4 NA
## 10 21.2 NA
## # ℹ 1,197 more rows
cor(district_example1, use = "pairwise.complete.obs")
## DPSTKIDR DA0CSA21R
## DPSTKIDR 1.0000000 0.2659224
## DA0CSA21R 0.2659224 1.0000000
There is a weak correlation between the number of students per teacher and the SAT average score.
district_example2<-district%>% select(DPETECOP,DA0CSA21R) %>% arrange(-DPETECOP,DA0CSA21R)
district_example2
## # A tibble: 1,207 × 2
## DPETECOP DA0CSA21R
## <dbl> <dbl>
## 1 100 894
## 2 100 1012
## 3 100 NA
## 4 100 NA
## 5 100 NA
## 6 100 NA
## 7 100 NA
## 8 100 NA
## 9 100 NA
## 10 100 NA
## # ℹ 1,197 more rows
cor(district_example2, use = "pairwise.complete.obs")
## DPETECOP DA0CSA21R
## DPETECOP 1.0000000 -0.2015739
## DA0CSA21R -0.2015739 1.0000000
There is a weak correlation between the STUDENTS: % ECONOMICALLY DISADVANTAGED and the SAT average score.
district_example3<-district%>% select(DPSTURNR,DA0CSA21R) %>% arrange(-DPSTURNR,DA0CSA21R)
district_example3
## # A tibble: 1,207 × 2
## DPSTURNR DA0CSA21R
## <dbl> <dbl>
## 1 100 NA
## 2 81.2 NA
## 3 80 1006
## 4 79.5 NA
## 5 77.8 NA
## 6 69.8 NA
## 7 67.8 962
## 8 67 922
## 9 66.7 NA
## 10 66.5 NA
## # ℹ 1,197 more rows
cor(district_example3, use = "pairwise.complete.obs")
## DPSTURNR DA0CSA21R
## DPSTURNR 1.00000000 -0.07310558
## DA0CSA21R -0.07310558 1.00000000
There is no correlation between teacher turnover rate and the SAT average score.
pairs(~DA0AT21R+DPSTKIDR+DPETECOP+DPFVTOTK+DPFRAALLK,data=district)
cor.test(district_example$DA0AT21R,district_example$DA0CSA21R,method = "pearson")
##
## Pearson's product-moment correlation
##
## data: district_example$DA0AT21R and district_example$DA0CSA21R
## t = 3.8585, df = 942, p-value = 0.0001219
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
## 0.0614174 0.1870523
## sample estimates:
## cor
## 0.1247348
cor.test(district_example1$DPSTKIDR,district_example1$DA0CSA21R,method = "pearson")
##
## Pearson's product-moment correlation
##
## data: district_example1$DPSTKIDR and district_example1$DA0CSA21R
## t = 8.4575, df = 940, p-value < 2.2e-16
## alternative hypothesis: true correlation is not equal to 0
## 95 percent confidence interval:
## 0.2055397 0.3242881
## sample estimates:
## cor
## 0.2659224
I tried 4 different variables I chose as my independent variables and I compared them to my dependent variable. All of them have a weak correlation to my dependent variable and i did the pearsons product moment correlation to the one with the most corenation which is .206 still weak but not as weak as the other ones. I also used this one because it has normal data. I did not use Kendall because it is for small sample sizes.