library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr     1.2.1     ✔ readr     2.2.0
## ✔ forcats   1.0.1     ✔ stringr   1.6.0
## ✔ ggplot2   4.0.3     ✔ tibble    3.3.1
## ✔ lubridate 1.9.5     ✔ tidyr     1.3.2
## ✔ purrr     1.2.2     
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag()    masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(readxl)
library(pastecs)
## 
## Attaching package: 'pastecs'
## 
## The following objects are masked from 'package:dplyr':
## 
##     first, last
## 
## The following object is masked from 'package:tidyr':
## 
##     extract
district <- read_excel("district.xls")
pastecs::stat.desc(district$DA0CC23R)
##       nbr.val      nbr.null        nbr.na           min           max 
##  1074.0000000    47.0000000   133.0000000    -1.0000000    97.7000000 
##         range           sum        median          mean       SE.mean 
##    98.7000000 24939.3000000    20.7500000    23.2209497     0.5168674 
##  CI.mean.0.95           var       std.dev      coef.var 
##     1.0141855   286.9211357    16.9387466     0.7294597

This data shows the percentage of College Admissions At/Above Criterion within Texas School Districts for the year 2022-2023.

clean <- district |> drop_na(DA0CC23R) %>% filter(DA0CC23R>0)
hist(clean$DA0CC23R)

clean <- clean |> mutate(DA0CC23R_SQRT=sqrt(DA0CC23R))
hist(clean$DA0CC23R_SQRT)