# if you haven't used a given package before, you'll need to download it first
# delete the "#" before the install function and run it to download
# then run the library function calling that package
#install.packages("naniar")
library("naniar") # for the gg_miss-upset() command
Import the full project data into a dataframe, call it “df”. Replace ‘DOWLOADED FILE NAME’ with the actual file name of your dataset (either for the ARC or EAMMi2).
Note: If you named your folder something else, you will also need to replace ‘Data’ with whatever the name of your folder is where you saved the dataset in.
df <- read.csv(file="Data/eammi2_data_final_SP26.csv", header=T)
These are commands useful for viewing a data frame.
# you can also click the object (the little table picture) in the environment tab to view it in a new window
names(df) # all the variable name in the data frame
## [1] "ResponseID" "gender" "race_rc" "age"
## [5] "income" "edu" "sibling" "party_rc"
## [9] "disability" "marriage5" "phys_sym" "pipwd"
## [13] "moa_independence" "moa_role" "moa_safety" "moa_maturity"
## [17] "idea" "swb" "mindful" "belong"
## [21] "efficacy" "support" "socmeduse" "usdream"
## [25] "npi" "exploit" "stress"
head(df) # first 6 lines of data in the data frame
## ResponseID gender race_rc age income
## 1 R_BJN3bQqi1zUMid3 f white 1 between 18 and 25 1 low
## 2 R_2TGbiBXmAtxywsD m white 1 between 18 and 25 1 low
## 3 R_12G7bIqN2wB2N65 m white 1 between 18 and 25 rather not say
## 4 R_39pldNoon8CePfP f other 1 between 18 and 25 rather not say
## 5 R_1QiKb2LdJo1Bhvv m white 1 between 18 and 25 2 middle
## 6 R_pmwDTZyCyCycXwB f white 1 between 18 and 25 rather not say
## edu sibling party_rc disability
## 1 2 Currently in college at least one sibling democrat <NA>
## 2 5 Completed Bachelors Degree at least one sibling independent <NA>
## 3 2 Currently in college at least one sibling apolitical psychiatric
## 4 2 Currently in college at least one sibling apolitical <NA>
## 5 2 Currently in college at least one sibling apolitical <NA>
## 6 2 Currently in college at least one sibling apolitical <NA>
## marriage5 phys_sym pipwd
## 1 are currently divorced from one another high number of symptoms NA
## 2 are currently married to one another high number of symptoms NA
## 3 are currently married to one another high number of symptoms 2.333333
## 4 are currently married to one another high number of symptoms NA
## 5 are currently married to one another low number of symptoms NA
## 6 are currently married to one another high number of symptoms NA
## moa_independence moa_role moa_safety moa_maturity idea swb mindful
## 1 3.666667 3.000000 2.75 3.666667 3.750 4.333333 2.4
## 2 3.666667 2.666667 3.25 3.333333 3.875 4.166667 1.8
## 3 3.500000 2.500000 3.00 3.666667 3.750 1.833333 2.2
## 4 3.000000 2.000000 1.25 3.000000 3.750 5.166667 2.2
## 5 3.833333 2.666667 2.25 3.666667 3.500 3.666667 3.2
## 6 3.500000 3.333333 2.50 4.000000 3.250 4.000000 3.4
## belong efficacy support socmeduse
## 1 2.8 3.4 6.000000 47
## 2 4.2 3.4 6.750000 23
## 3 3.6 2.2 5.166667 34
## 4 4.0 2.8 5.583333 35
## 5 3.4 3.0 6.000000 37
## 6 4.2 2.4 4.500000 13
## usdream npi
## 1 american dream is important and achievable for me 0.69230769
## 2 american dream is important and achievable for me 0.15384615
## 3 american dream is not important and maybe not achievable for me 0.07692308
## 4 american dream is not important and maybe not achievable for me 0.07692308
## 5 not sure if american dream important 0.76923077
## 6 american dream is not important and maybe not achievable for me 0.23076923
## exploit stress
## 1 2.000000 3.3
## 2 3.666667 3.3
## 3 4.333333 4.0
## 4 1.666667 3.2
## 5 4.000000 3.1
## 6 1.333333 3.5
str(df) # shows all the variables in the data frame and their classification type (e.g., numeric, string, character,etc.)
## 'data.frame': 3182 obs. of 27 variables:
## $ ResponseID : chr "R_BJN3bQqi1zUMid3" "R_2TGbiBXmAtxywsD" "R_12G7bIqN2wB2N65" "R_39pldNoon8CePfP" ...
## $ gender : chr "f" "m" "m" "f" ...
## $ race_rc : chr "white" "white" "white" "other" ...
## $ age : chr "1 between 18 and 25" "1 between 18 and 25" "1 between 18 and 25" "1 between 18 and 25" ...
## $ income : chr "1 low" "1 low" "rather not say" "rather not say" ...
## $ edu : chr "2 Currently in college" "5 Completed Bachelors Degree" "2 Currently in college" "2 Currently in college" ...
## $ sibling : chr "at least one sibling" "at least one sibling" "at least one sibling" "at least one sibling" ...
## $ party_rc : chr "democrat" "independent" "apolitical" "apolitical" ...
## $ disability : chr NA NA "psychiatric" NA ...
## $ marriage5 : chr "are currently divorced from one another" "are currently married to one another" "are currently married to one another" "are currently married to one another" ...
## $ phys_sym : chr "high number of symptoms" "high number of symptoms" "high number of symptoms" "high number of symptoms" ...
## $ pipwd : num NA NA 2.33 NA NA ...
## $ moa_independence: num 3.67 3.67 3.5 3 3.83 ...
## $ moa_role : num 3 2.67 2.5 2 2.67 ...
## $ moa_safety : num 2.75 3.25 3 1.25 2.25 2.5 4 3.25 2.75 3.5 ...
## $ moa_maturity : num 3.67 3.33 3.67 3 3.67 ...
## $ idea : num 3.75 3.88 3.75 3.75 3.5 ...
## $ swb : num 4.33 4.17 1.83 5.17 3.67 ...
## $ mindful : num 2.4 1.8 2.2 2.2 3.2 ...
## $ belong : num 2.8 4.2 3.6 4 3.4 4.2 3.9 3.6 2.9 2.5 ...
## $ efficacy : num 3.4 3.4 2.2 2.8 3 2.4 2.3 3 3 3.7 ...
## $ support : num 6 6.75 5.17 5.58 6 ...
## $ socmeduse : int 47 23 34 35 37 13 37 43 37 29 ...
## $ usdream : chr "american dream is important and achievable for me" "american dream is important and achievable for me" "american dream is not important and maybe not achievable for me" "american dream is not important and maybe not achievable for me" ...
## $ npi : num 0.6923 0.1538 0.0769 0.0769 0.7692 ...
## $ exploit : num 2 3.67 4.33 1.67 4 ...
## $ stress : num 3.3 3.3 4 3.2 3.1 3.5 3.3 2.4 2.9 2.7 ...
Open your mini codebook and get the names of your variables (first column). Then enter this list of names within the “select=c()” argument to subset those columns from the dataframe “df” into a new one “d”.
Replace “variable1, variable2,…” with your variables names.
# Make sure to keep the "ResponseID" variable first in the "select" argument.
# NOTE: you will need to replace "ResponseID" with "X" if you are using the ARC data.
d <- subset(df, select=c(ResponseID, sibling, marriage5, belong, support, stress, exploit))
# Your new data frame should contain 7 variables (ResponseID, + your 2 categorical, + your 4 continuous)
names(d) # all the variable name in the data frame
## [1] "ResponseID" "sibling" "marriage5" "belong" "support"
## [6] "stress" "exploit"
head(d) # first 6 lines of data in the data frame
## ResponseID sibling
## 1 R_BJN3bQqi1zUMid3 at least one sibling
## 2 R_2TGbiBXmAtxywsD at least one sibling
## 3 R_12G7bIqN2wB2N65 at least one sibling
## 4 R_39pldNoon8CePfP at least one sibling
## 5 R_1QiKb2LdJo1Bhvv at least one sibling
## 6 R_pmwDTZyCyCycXwB at least one sibling
## marriage5 belong support stress exploit
## 1 are currently divorced from one another 2.8 6.000000 3.3 2.000000
## 2 are currently married to one another 4.2 6.750000 3.3 3.666667
## 3 are currently married to one another 3.6 5.166667 4.0 4.333333
## 4 are currently married to one another 4.0 5.583333 3.2 1.666667
## 5 are currently married to one another 3.4 6.000000 3.1 4.000000
## 6 are currently married to one another 4.2 4.500000 3.5 1.333333
str(d)
## 'data.frame': 3182 obs. of 7 variables:
## $ ResponseID: chr "R_BJN3bQqi1zUMid3" "R_2TGbiBXmAtxywsD" "R_12G7bIqN2wB2N65" "R_39pldNoon8CePfP" ...
## $ sibling : chr "at least one sibling" "at least one sibling" "at least one sibling" "at least one sibling" ...
## $ marriage5 : chr "are currently divorced from one another" "are currently married to one another" "are currently married to one another" "are currently married to one another" ...
## $ belong : num 2.8 4.2 3.6 4 3.4 4.2 3.9 3.6 2.9 2.5 ...
## $ support : num 6 6.75 5.17 5.58 6 ...
## $ stress : num 3.3 3.3 4 3.2 3.1 3.5 3.3 2.4 2.9 2.7 ...
## $ exploit : num 2 3.67 4.33 1.67 4 ...
# use the gg_miss_upset() command for a visualization of your missing data
gg_miss_upset(d[-1], nsets = 6)
## Warning: `aes_string()` was deprecated in ggplot2 3.0.0.
## ℹ Please use tidy evaluation idioms with `aes()`.
## ℹ See also `vignette("ggplot2-in-packages")` for more information.
## ℹ The deprecated feature was likely used in the UpSetR package.
## Please report the issue at <https://github.com/hms-dbmi/UpSetR/issues>.
## This warning is displayed once per session.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.
## Warning: Using `size` aesthetic for lines was deprecated in ggplot2 3.4.0.
## ℹ Please use `linewidth` instead.
## ℹ The deprecated feature was likely used in the UpSetR package.
## Please report the issue at <https://github.com/hms-dbmi/UpSetR/issues>.
## This warning is displayed once per session.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.
## Warning: The `size` argument of `element_line()` is deprecated as of ggplot2 3.4.0.
## ℹ Please use the `linewidth` argument instead.
## ℹ The deprecated feature was likely used in the UpSetR package.
## Please report the issue at <https://github.com/hms-dbmi/UpSetR/issues>.
## This warning is displayed once per session.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.
# the [-1] tells the function to ignore the first column i.e., variable -- we are doing this because here it is just the ID variable, we don't need to check it for missingness because everyone was assigned a random ID
# use the na.omit() command to create a new dataframe in which any participants with missing data are dropped from the dataframe
d2 <- na.omit(d)
# Next, calc the total number of participants dropped, then convert to % and insert both the number and % in the text below.
# insert the total number of participants in your d2 in the text where it says N = #.
3182-3164
## [1] 18
18/3182
## [1] 0.00565682
# Then fill in the paragraph below with your numbers
We looked at the missing data in our dataset, and found that 18, or about .57%, of the participants in our sample skipped at least one item. We dropped these participants from our analysis, which is not advisable and runs the risk of dropping vulnerable groups or skewing results. However, we will proceed for the sake of this class using the reduced dataset, N = 3164.
Our last step is to export the data frame after we’ve dropped NAs so that it can be used in future HWs.
# use the "write.cvs" function to export the cleaned data
# please keep the file name as 'projectdata'
# note: you only need to change 'Data' before the slash if you named your folder something else
write.csv(d2, file="Data/projectdata.csv", row.names = F)
# if you haven't used a given package before, you'll need to download it first
# insert the "#" before the install function so that the file will Knit later
# then run the library function calling that package
#install.packages("")
library(psych) # for the describe() command
# Import the "fakedata.csv" file for this Lab
d2 <- read.csv("Data/projectdata.csv")
# Note: for your Project version, you will import "projectdata.csv" that you created and exported at the end of the Project Data Prep assignment.
# use tables to visualize 2 categorical variables
table(d2$sibling)
##
## at least one sibling only child
## 2860 304
table(d2$marriage5)
##
## are currently divorced from one another
## 735
## are currently married to one another
## 2134
## never married each other and are not together
## 248
## never married each other but are currently together
## 47
# use histograms to visualize 4 continuous variables
hist(d2$belong)
hist(d2$support)
hist(d2$stress)
hist(d2$exploit)
# quick & dirty normality test: check skew & kurtosis values
describe(d2)
## vars n mean sd median trimmed mad min max range
## ResponseID* 1 3164 1582.50 913.51 1582.50 1582.50 1172.74 1.0 3164.0 3163.0
## sibling* 2 3164 1.10 0.29 1.00 1.00 0.00 1.0 2.0 1.0
## marriage5* 3 3164 1.88 0.60 2.00 1.83 0.00 1.0 4.0 3.0
## belong 4 3164 3.23 0.61 3.30 3.25 0.59 1.3 5.0 3.7
## support 5 3164 5.53 1.14 5.75 5.65 0.99 0.0 7.0 7.0
## stress 6 3164 3.05 0.60 3.00 3.05 0.59 1.3 4.7 3.4
## exploit 7 3164 2.39 1.37 2.00 2.21 1.48 1.0 7.0 6.0
## skew kurtosis se
## ResponseID* 0.00 -1.20 16.24
## sibling* 2.74 5.51 0.01
## marriage5* 0.47 1.48 0.01
## belong -0.26 -0.12 0.01
## support -1.11 1.48 0.02
## stress 0.04 -0.17 0.01
## exploit 0.95 0.38 0.02
## For the required write-up below, choose one of these options to paste and edit below based on your output.
## OPTION 1
# We analyzed the skew and kurtosis of our four continuous variables and all were within the acceptable range (-2/+2).
## OPTION 2
# We analyzed the skew and kurtosis of our four continuous variables and (#) were within the acceptable range (-2/+2). However, (#) variables (list variable name(s) here) were outside of the acceptable range. For our analysis we will use them anyway, but outside of class this is bad practice.
# Objective normality test: Shapiro-Wilks test
# We want the test to be NON-SIG (aka > .05) to indicate normality
options(scipen = 999)
# the above code turns off scientific notation
shapiro.test(d2$belong)
##
## Shapiro-Wilk normality test
##
## data: d2$belong
## W = 0.99249, p-value = 0.000000000008691
shapiro.test(d2$support)
##
## Shapiro-Wilk normality test
##
## data: d2$support
## W = 0.92247, p-value < 0.00000000000000022
shapiro.test(d2$stress)
##
## Shapiro-Wilk normality test
##
## data: d2$stress
## W = 0.9962, p-value = 0.0000003434
shapiro.test(d2$exploit)
##
## Shapiro-Wilk normality test
##
## data: d2$exploit
## W = 0.88257, p-value < 0.00000000000000022
We analyzed the skew and kurtosis of our four continuous variables and (5) were within the acceptable range (-2/+2). However, (1) variables (sibling) were outside of the acceptable range. For our analysis we will use them anyway, but outside of class this is bad practice.
Scatterplots are used to visualize combinations of two continuous variables.
plot(d2$support, d2$belong,
main="Scatterplot of Perceived Support and Sense of Belonging",
xlab = "Perceived Support",
ylab = "Sense of Belonging")
plot(d2$exploit, d2$stress,
main="Scatterplot of Exploitativeness and Stress",
xlab = "Exploitativeness",
ylab = "Stress")
# Note: for HW, you will choose to plot 2 combos of your 4 continuous variables, based on your hypotheses. You may repeat 1 variable to see its association with 2 others. You will need replace the variable names on the first line of the function as well as the 'main' (aka plot title), 'xlab' and 'ylab' lines to correctly label the graphs -- remember to use the actual variable names, not their scales, so someone reading your plots can understand them.
Boxplots are used to visualize combinations of one categorical and one continuous variable.
# ORDER MATTERS HERE: 'continuous variable' ~ 'categorical variable'
boxplot(data=d2, support~marriage5,
main="Boxplot of Perceived Support and Parents Marital Status",
xlab = "Parents Marital Status",
ylab = "Perceived Support")
boxplot(data=d2, belong~sibling,
main="Boxplot of Sense of Beloning by Amount of Siblings",
xlab = "Amount of Siblings",
ylab = "Sense of Beloning")