Exploring Diversity in Parental Involvement in the USA - Data Preprocessing, Factor Analysis

In this project, we explore the patterns of parental involvement in their children’s education across diverse racial groups, leveraging data from the High School Longitudinal Study (HSLS) conducted by the National Center for Education Statistics (NCES).

To uncover meaningful latent constructs within the parental involvement survey data, we preprocess the dataset and apply Exploratory Factor Analysis (EFA). This approach allows us to transform complex measures of home- and school-based involvement into more interpretable and actionable factors.

Our analysis identifies the following latent constructs:

We also create composite variables to quantify these constructs, enabling clear and effective visual storytelling. Ultimately, this analysis provides valuable insights into how racial differences influence parental educational involvement, fostering evidence-based discussions about equity in education and supporting initiatives to address disparities.

We start by:

Loading and Cleaning the Raw Data

#Load the libraries
library(dplyr)
library(psych)
library(readr)

Loading the Data

#Load the data
#National Center for Educational Statistics, High School Longitudinal Study
setwd("C:/Users/nhshannoniv/Downloads")
data <- read_csv("hsls_17_student_pets_sr_v1_0.csv")

# Create vector of selected variables
selectvars <- cbind(data$STU_ID,           # Unique Student Identifier
  data$X1PAR1RACE,       # Race of Parent 1
  data$X1PAR2RACE,       # Race of Parent 2
  data$X1P1RELATION,     # Parent 1 relation to child
  data$X1P2RELATION,     # Parent 2 relation to child
  data$X1PAREDU,         # Parent Education Level
  data$X1SES,            # Socioeconomic Status (SES)
  data$X1SESQ5,          # SES Quintile
  data$X3ELLSTATUS,      # English Language Learner (ELL) Status
  data$P1HOMELANG,       # Language Spoken at Home
  data$P1SCHMTG,         # Parent Attended School Meeting
  data$P1PTOMTG,         # Parent Attended PTA Meeting
  data$P1PTCONFER,       # Parent-Teacher Conferences
  data$P1SCHEVENT,       # Parent Attended School Events
  data$P1VOLUNTEER,      # Parent Volunteered at School
  data$P1FUNDRAISE,      # Parent Participated in Fundraising
  data$P1COUNSELOR,      # Parent Met with School Counselor
  data$P1MUSEUM,         # Parent Took Student to a Museum
  data$P1LIBRARY,        # Parent Took Student to a Library
  data$P1SCIFAIR,        # Parent Took Student to a Science Fair
  data$P1COMPUTER,       # Parent Took Student to a Computer-Related Event
  data$P1SHOW,           # Parent Took Student to a Show or Performance
  data$P1FIXED,          # Parent Helped Fix or Build Something with Student
  data$P1SCIPROJ,        # Parent Helped with Science Projects
  data$P1STEMDISC,       # Parent Discussed STEM Careers with Student
  data$X3HSCOMPSTAT,     # High School Completion Status
  data$X3TGPASCI,        # Cumulative GPA in Science
  data$X3TGPAMAT,        # Cumulative GPA in Mathematics
  data$X3TGPASTEM)       # Cumulative GPA in STEM Subjects



colnames(selectvars) <- c(
  "STU_ID",        # Unique Student Identifier
  "PAR1_RACE",     # Race of Parent 1
  "PAR2_RACE",     # Race of Parent 2
  "PAR1_Relation", # Parent 1 relation to child
  "PAR2_Relation", # Parent 2 relation to child
  "PAREDU",        # Parent Education Level
  "SES",           # Socioeconomic Status (SES)
  "SES_Q5",        # SES Quintiles
  "ELL_STATUS",    # English Language Learner (ELL) Status
  "HOME_LANG",     # Language Spoken at Home
  
  "SBI_SCH_MTG",   # Parent Attended School Meeting
  "SBI_PTO",       # Parent Attended PTA Meeting
  "SBI_PT_CONF",   # Parent-Teacher Conferences
  "SBI_SCH_EVNT",  # Parent Attended School Events
  "SBI_VOLNTR",    # Parent Volunteered at School
  "SBI_FUNDR",     # Parent Participated in Fundraising
  "SBI_COUNSL",    # Parent Met with School Counselor
  
  "HBI_MUSEUM",    # Parent Took Student to a Museum
  "HBI_LIBRARY",   # Parent Took Student to a Library
  "HBI_SCIFAIR",   # Parent Took Student to a Science Fair
  "HBI_COMPUTE",   # Parent Took Student to a Computer-Related Event
  "HBI_SHOW",      # Parent Took Student to a Show or Performance
  "HBI_FIXED",     # Parent Helped Fix or Build Something with Student
  "HBI_SCIPROJ",   # Parent Helped with Science Projects
  "HBI_STEMDISC",  # Parent Discussed STEM Careers with Student
  
  "HSCOMP_STAT",   # High School Completion Status
  "GPA_SCI",       # Cumulative GPA in Science
  "GPA_MAT",       # Cumulative GPA in Mathematics
  "GPA_STEM")      # Cumulative GPA in STEM Subjects

Cleaning The Data

selectvars <- as.data.frame(selectvars)
# Replace -9, -8, and -7 with NA in the `selectvars` data frame
selectvars[selectvars == -9 | selectvars == -8 | selectvars == -7] <- NA

###  negatice values with NA in specified variables, more than just (neg 7,8,9)
columns_to_modify <- c("GPA_SCI", "GPA_MAT", "GPA_STEM")

# Values <= 0 with NA for the specified columns
selectvars[columns_to_modify] <- lapply(selectvars[columns_to_modify], function(x) {
  x[x <= 0] <- NA
  return(x)
})

###Race
#combine "Hispanic, mo race specified" and "Hispanic, race specified" (4 & 5)
#removing "Other, non hispanic", non-descript and .6% of data, criticism of this surveys design

hispanic <- function(column) {
  column[!is.na(column) & column %in% c(4, 5)] <- 4
  column[column == 9] <- NA 
  return(column)
}

# Apply the function to both columns
selectvars$PAR1_RACE <- hispanic(selectvars$PAR1_RACE)
selectvars$PAR2_RACE <- hispanic(selectvars$PAR2_RACE)


###Relation to child is a WOMAN, for parent 1 and 2
##combining 1,3,5,15 - Women guardians (Bio Mom, adoptive, step, other)
women_relation <- function(column) {
  column[!is.na(column) & column %in% c(1, 3, 5, 15)] <- 1
  return(column)
}

# Apply the function to both columns
selectvars$PAR1_Relation <- women_relation(selectvars$PAR1_Relation)
selectvars$PAR2_Relation <- women_relation(selectvars$PAR2_Relation)


###Relation to child is a MAN, for parent 1 and 2
##combining 2,4,6,16 - Men guardians (Bio Mom, adoptive, step, other)
men_relation <- function(column) {
  column[!is.na(column) & column %in% c(2, 4, 6, 16)] <- 2
  return(column)
}
# Apply the function to both columns
selectvars$PAR1_Relation <- men_relation(selectvars$PAR1_Relation)
selectvars$PAR2_Relation <- men_relation(selectvars$PAR2_Relation)


# Variables to convert to factors
variables_to_factor <- c(
  "PAR1_RACE", "PAR2_RACE", "PAR1_Relation",
  "PAR2_Relation",  "ELL_STATUS", "HOME_LANG")
selectvars[variables_to_factor] <- lapply(selectvars[variables_to_factor], as.factor)

Exploratory Factor Analysis for Home-Based and School-Based Involvement Surveys

What are the types of involvement measured in these surveys?

In this section, we will investigate the different types of involvement measure within two surveys of school- and home involvement.

#First lets make some survey specific datasets
  #Loading the dataset
  EFA_SBI_data <- na.omit(selectvars[11:17])
  EFA_HBI_data <- na.omit(selectvars[18:24])

Testing the Assumptions of Factor Analysis

Are the data normally distrbuted, is there variance to exlpore with a factor analysis, and are the variables too correlated?

We will test if:

##Bartlett's test of sphericity, normality assumption
cortest.bartlett(EFA_SBI_data)  #p < .05, more evidence of factorability
## R was not square, finding R from data

## $chisq
## [1] 13860.01
## 
## $p.value
## [1] 0
## 
## $df
## [1] 21
#We want a positive determinant
#.46, not close to 1/0, no multicolinearity issues
det(cor(EFA_SBI_data)) 
## [1] 0.4033661
##Bartlett's test of sphericity
cortest.bartlett(EFA_HBI_data) #a significant findings signifies that the data has sufficient shared variance to justify factor analysis
## R was not square, finding R from data

## $chisq
## [1] 6430.166
## 
## $p.value
## [1] 0
## 
## $df
## [1] 21
#We want a positive determinant
det(cor(EFA_HBI_data)) # There is no severe multicollinearity in the dataset.
## [1] 0.659444

Correlation table for SBI and HBI

Both correlation tables show that the data from both surveys are not overly correlated (none above .8). This suggests that the variables exhibit sufficient independence to avoid multicollinearity issues, which is critical for the stability and interpretability of factor analysis.

## terachoric correlation (because the outcome is binary)
tetra_cor_SBI = tetrachoric(EFA_SBI_data)
#correlation matrix
cor.plot(tetra_cor_SBI$rho, numbers=T, upper=FALSE, main = "Correlation Table - SBI", show.legend = FALSE)

## terachoric correlation
tetra_cor_hbi = tetrachoric(EFA_HBI_data)
#correlation matrix
cor.plot(tetra_cor_hbi$rho, numbers=T, upper=FALSE, main = "Correlation Table - HBI", show.legend = FALSE) 

Determining the Latent Factors in the Involvement Surveys

Scree plots will tell us the optimal number of factors that should be included in the model by visualizing the eigenvalues associated with each factor. The key is to identify the “elbow point,” where the eigenvalues start to level off, indicating a diminishing explanatory power of additional factors.

The first plot shows two latent factors.

# Scree plot - SBI
fa.parallel(EFA_SBI_data,main = "SBI Factor Loadings") #suggests we have 2 latent factors

## Parallel analysis suggests that the number of factors =  3  and the number of components =  2

The second plot shows 3 latent factors. (spoiler, one factor only contains a single variable :/ )

# Scree plot - HBI
fa.parallel(EFA_HBI_data, main = "HBI Factor Loadings")

## Parallel analysis suggests that the number of factors =  3  and the number of components =  2

Factor Analysis Results

Attending volunteer events, fundraisers, and school events aligns with a school-based involvement construct we will call “Attending School Activities.” The second latent factor, involving parent-teacher conferences, PTO meetings, general school meetings, and meetings with the school counselor, represents a construct we will call “School-Parent Communication.” These findings are consistent with existing literature on types of parental involvement (Duppong Hurley, Lambert, January, & Huscroft D’Angelo, 2016).

#model
tetra_model_sbi = fa(EFA_SBI_data, nfactor=2, cor="tet", fm="mle", rotate = "varimax")
#tetra_model_sbi #evidence for adequate model fit, I left this output out to avoid a wall of information

# Cluster analysis plot SBI
#Factor #1 Attend School Activities
#Factor #2 School Parent Communication
fa.diagram(tetra_model_sbi)

For the HBI factor loadings, we will focus on the second, more stable factor, which includes involvement actions such as visiting a museum, attending a show, going to the library, and engaging in computer-related activities. We will describe this factor as “Home-Based Educational Enrichment.”

#model
tetra_model_hbi = fa(EFA_HBI_data, nfactor=2, cor="tet", fm="mle", rotate = "varimax")
#tetra_model_hbi #also shows evidence for adequate model fit

#Cluster analysis plot HBI
fa.diagram(tetra_model_hbi)

Creating Composite Variables

Now that we know our latent factors and have given them names, we can calculate new variable composites for our “selectvars” dataset. Then, we can save that new dataset and tell our story with some visuals.

You’ll find this in the second attached file.

#Overall Involvement Totals
selectvars$SBI_Tot <- rowSums(selectvars[, c(
  "SBI_SCH_MTG", "SBI_PTO", "SBI_PT_CONF", 
  "SBI_SCH_EVNT", "SBI_VOLNTR", "SBI_FUNDR",
  "SBI_COUNSL")], na.rm = TRUE)

selectvars$HBI_Tot <- rowSums(selectvars[, c(
  "HBI_MUSEUM", "HBI_LIBRARY", "HBI_SCIFAIR", 
  "HBI_COMPUTE", "HBI_SHOW", "HBI_FIXED", 
  "HBI_SCIPROJ", "HBI_STEMDISC")], na.rm = TRUE)

#Involvement Composite Variables
selectvars$School_Activities <- rowSums(selectvars[, c(
  "SBI_SCH_EVNT", "SBI_VOLNTR", "SBI_FUNDR")], na.rm = TRUE)

selectvars$Parent_Sch_Comm <- rowSums(selectvars[, c(
  "SBI_SCH_MTG", "SBI_PTO", "SBI_PT_CONF", 
  "SBI_COUNSL")], na.rm = TRUE)

selectvars$home_enrichment <- rowSums(selectvars[, c(
  "HBI_MUSEUM", "HBI_LIBRARY", "HBI_COMPUTE", 
  "HBI_SHOW")], na.rm = TRUE)

Save the updated dataset. Go forth and tell the story!

write.csv(selectvars, "involvement_data")