# if you haven't run this code before, you'll need to download the below packages first
# instructions on how to do this are included in the video
# but as a reminder, you use the packages tab to the right
library(psych) # for the describe() command
library(expss) # for the cross_cases() command
## Loading required package: maditr
##
## To select rows from data: rows(mtcars, am==0)
##
## Attaching package: 'maditr'
## The following object is masked from 'package:base':
##
## sort_by
# Import our data doe the lab
# for the homework you will import the mydata.csv that we created in the Data Prep Lab
d2 <- read.csv(file="Data/mydata.csv", header =)
table(d2$age) #the table command shows us what the levels of this variable are in each level
##
## 1 between 18 and 25 2 between 26 and 35 3 between 36 and 45 4 over 45
## 1934 112 37 17
table(d2$gender)
##
## f m nb
## 1541 528 31
hist(d2$moa_independence) #the hist command creates a histogram of the variable
hist(d2$npi)
hist(d2$belong)
hist(d2$idea)
We analyzed the skew and kurtosis of our continuous varibales and most were within the accepted range (-2/+2). However, some variables (moa_independence and idea) were outside of the accepted range. For this analysis, we will use them anyway, but outside of this class this is bad practice.
describe(d2) #we use this to check univariate normality ... skew and kurtosis, (-2/+2)
## vars n mean sd median trimmed mad min max range skew
## age* 1 2100 1.11 0.43 1.00 1.00 0.00 1.0 4 3.0 4.42
## gender* 2 2100 1.28 0.48 1.00 1.21 0.00 1.0 3 2.0 1.36
## moa_independence 3 2100 3.54 0.47 3.67 3.61 0.49 1.0 4 3.0 -1.49
## npi 4 2100 0.27 0.30 0.15 0.23 0.23 0.0 1 1.0 0.99
## belong 5 2100 3.21 0.61 3.20 3.23 0.59 1.3 5 3.7 -0.27
## idea 6 2100 3.57 0.38 3.62 3.62 0.37 1.0 4 3.0 -1.48
## kurtosis se
## age* 21.19 0.01
## gender* 0.74 0.01
## moa_independence 2.74 0.01
## npi -0.57 0.01
## belong -0.12 0.01
## idea 3.95 0.01
cross_cases(d2, age, gender) #update variable2 and variable3 with your categorical variable names
|  gender | |||
|---|---|---|---|
|  f |  m |  nb | |
|  age | |||
| Â Â Â 1 between 18 and 25Â | 1435 | 469 | 30 |
| Â Â Â 2 between 26 and 35Â | 67 | 45 | |
| Â Â Â 3 between 36 and 45Â | 27 | 9 | 1 |
| Â Â Â 4 over 45Â | 12 | 5 | |
|    #Total cases | 1541 | 528 | 31 |
plot(d2$moa_independence, d2$npi,
main="Scatterplot of moa_independence and npi",
xlab = "moa_independence",
ylab = "npi")
plot(d2$belong, d2$idea,
main="Scatterplot of belong and idea",
xlab = "belong",
ylab = "idea")
# boxplots use ONE CATEGORICAL and ONE CONTNIUOUS variable
# make sure that you enter them in the right order!!!!
# categorical variable goes BEFORE the tilde ~
# continuous variable goes AFTER the tilde ~
boxplot(data=d2, moa_independence~age,
main="Boxplot of moa_independence and age",
xlab = "moa_independence",
ylab = "age")
boxplot(data=d2, npi~gender,
main="Boxplot of npi and gender",
xlab = "npi",
ylab = "gender")