Basic Statistics

Load Libraries

# if you haven't run this code before, you'll need to download the below packages first
# instructions on how to do this are included in the video
# but as a reminder, you use the packages tab to the right

library(psych) # for the describe() command
library(expss) # for the cross_cases() command
## Loading required package: maditr
## 
## To aggregate several columns with one summary: take(mtcars, mpg, hp, fun = mean, by = am)
## 
## Attaching package: 'maditr'
## The following object is masked from 'package:base':
## 
##     sort_by
## 
## Use 'expss_output_viewer()' to display tables in the RStudio Viewer.
##  To return to the console output, use 'expss_output_default()'.

Import Data

# import our data for the lab
# for the homework, you will import the mydata.csv that we created in the Data Prep Lab

d2 <- read.csv(file="Data/mydata.csv", header = T)

Univariate Plots: Histograms & Tables

table(d2$race_rc) #the table command shows us what the levels of this variable are, and how many participants are in each level
## 
##       asian       black    hispanic multiracial  nativeamer       other 
##         143         185         230         209           8          80 
##       white 
##        1297
table(d2$age)
## 
## 1 between 18 and 25 2 between 26 and 35 3 between 36 and 45           4 over 45 
##                1981                 115                  38                  18
hist(d2$belong)#the hist command creates a histogram of the variable

hist(d2$npi)

hist(d2$stress)

hist(d2$exploit)

Univariate Normality

We analyzed the skew and kurtosis of our continuous variables and all were within the accepted range (-2/+2). (True for the lab!!!! May not be true for the HW!!!!!)

We analyzed the skew and kurtosis of our … and most were within the accepted range (-2/+2). However, some variables (list them in parentheses) were outside of the accepted range. For this analysis, we will use them anyway, but outside of this class this is bad practice.

describe(d2) #we use this to check univariate normality ... skew and kurtosis (-2/+2)
##          vars    n mean   sd median trimmed  mad min max range  skew kurtosis
## race_rc*    1 2152 5.41 2.16   7.00    5.72 0.00 1.0 7.0   6.0 -0.84    -0.94
## age*        2 2152 1.11 0.43   1.00    1.00 0.00 1.0 4.0   3.0  4.41    21.07
## belong      3 2152 3.21 0.61   3.20    3.23 0.59 1.3 5.0   3.7 -0.27    -0.10
## npi         4 2152 0.27 0.30   0.15    0.23 0.23 0.0 1.0   1.0  0.99    -0.56
## stress      5 2152 3.07 0.60   3.10    3.07 0.59 1.3 4.6   3.3 -0.02    -0.15
## exploit     6 2152 2.37 1.38   2.00    2.18 1.48 1.0 7.0   6.0  0.96     0.38
##            se
## race_rc* 0.05
## age*     0.01
## belong   0.01
## npi      0.01
## stress   0.01
## exploit  0.03

Bivariate Plots

Crosstabs

cross_cases(d2, race_rc, age) #update variable2 and variable3 with tour categorical variable names
 age 
 1 between 18 and 25   2 between 26 and 35   3 between 36 and 45   4 over 45 
 race_rc 
   asian  137 4 2
   black  150 28 3 4
   hispanic  206 18 6
   multiracial  192 13 4
   nativeamer  8
   other  72 5 3
   white  1216 47 20 14
   #Total cases  1981 115 38 18

Scatterplots

plot(d2$stress, d2$npi,
     main="Scatterplot of stress and npi",
     xlab = "stress",
     ylab = "npi")

plot(d2$belong, d2$exploit,
     main="Scatterplot of belong and exploit",
     xlab = "belong",
     ylab = "exploit")

Boxplots

# boxplots use ONE CATEGORICAL and ONE CONTINUOUS variable
# make sure that you enter them in the right order!!!!!!!!!!!!!!!!!!!!
# categorical variable goes BEFORE the tilde ~
# continuous variable goes AFTER the tilde ~

boxplot(data=d2, stress~race_rc,
        main="Boxplot of stress and race_rc",
        xlab = "stress",
        ylab = "race_rc")

boxplot(data=d2, npi~age,
        main="Boxplot of npi and age",
        xlab = "npi",
        ylab = "age")