library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr     1.2.1     ✔ readr     2.2.0
## ✔ forcats   1.0.1     ✔ stringr   1.6.0
## ✔ ggplot2   4.0.3     ✔ tibble    3.3.1
## ✔ lubridate 1.9.5     ✔ tidyr     1.3.2
## ✔ purrr     1.2.2     
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag()    masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(readxl)
library(pastecs)
## 
## Attaching package: 'pastecs'
## 
## The following objects are masked from 'package:dplyr':
## 
##     first, last
## 
## The following object is masked from 'package:tidyr':
## 
##     extract
setwd("C:/Users/Kaitlyn/Downloads/Data")

#saved as homework 4 in the wd

tab1<-read.delim("PUDF_base1_1q2020_tab.txt")
stat.desc(tab1$LENGTH_OF_STAY)
##      nbr.val     nbr.null       nbr.na          min          max        range 
## 7.743210e+05 0.000000e+00 1.000000e+00 1.000000e+00 5.433000e+03 5.432000e+03 
##          sum       median         mean      SE.mean CI.mean.0.95          var 
## 4.228602e+06 3.000000e+00 5.461045e+00 2.114732e-02 4.144805e-02 3.462834e+02 
##      std.dev     coef.var 
## 1.860869e+01 3.407533e+00

this is question 1 for the homework. This isn’t one of my variables but the other ones are not numerical like length of stay.

#There is only one missing value and one 0 in this data set for this value. The range is from 1-5,433 with an average of 5.46 days

tabclean0<- tab1 |> drop_na(LENGTH_OF_STAY) %>% filter(LENGTH_OF_STAY>0)
tabclean100<- tab1 |> drop_na(LENGTH_OF_STAY) %>% filter(LENGTH_OF_STAY<100)
hist(tabclean0$LENGTH_OF_STAY)

hist(tabclean100$LENGTH_OF_STAY)

tabclean100<-tabclean100 %>% mutate(LENGTH_OF_STAY_SQRT=sqrt(LENGTH_OF_STAY))
hist(tabclean100$LENGTH_OF_STAY_SQRT)