Varsity Tutors Analyses

Author

Alexis Davila

Import

Packages to conduct analyses.

library(tidyverse)
── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
✔ dplyr     1.2.1     ✔ readr     2.2.0
✔ forcats   1.0.1     ✔ stringr   1.6.0
✔ ggplot2   4.0.3     ✔ tibble    3.3.1
✔ lubridate 1.9.5     ✔ tidyr     1.3.2
✔ purrr     1.2.2     
── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
✖ dplyr::filter() masks stats::filter()
✖ dplyr::lag()    masks stats::lag()
ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(sandwich)
library(lmtest)
Loading required package: zoo

Attaching package: 'zoo'

The following objects are masked from 'package:base':

    as.Date, as.Date.numeric
library(psych)

Attaching package: 'psych'

The following objects are masked from 'package:ggplot2':

    %+%, alpha
library(purrr)
library(tibble)
library(stargazer)

Please cite as: 

 Hlavac, Marek (2022). stargazer: Well-Formatted Regression and Summary Statistics Tables.
 R package version 5.2.3. https://CRAN.R-project.org/package=stargazer 

Import Kent data.

kent = read_csv("vt_kent_all_students.csv")
Rows: 5634 Columns: 45
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr  (4): school, school_grade, race, race_recoded
dbl (41): student_id, school_num, grade, grade6, grade7, grade8, male, sped,...

ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
colnames(kent)
 [1] "student_id"                     "school_num"                    
 [3] "school"                         "grade"                         
 [5] "school_grade"                   "grade6"                        
 [7] "grade7"                         "grade8"                        
 [9] "male"                           "sped"                          
[11] "mll"                            "low_inc"                       
[13] "low_ap"                         "race"                          
[15] "race_recoded"                   "race_A"                        
[17] "race_B"                         "race_H"                        
[19] "race_O"                         "race_W"                        
[21] "moy_math_comp"                  "moy_math_z"                    
[23] "moy_ela_comp"                   "moy_ela_z"                     
[25] "eoy_math_comp"                  "eoy_math_z"                    
[27] "eoy_ela_comp"                   "eoy_ela_z"                     
[29] "vt_acct_id"                     "vt_present"                    
[31] "num_days_active"                "num_days_active_recoded"       
[33] "num_weeks"                      "num_sessions_week"             
[35] "total_sessions"                 "num_sessions_AI"               
[37] "num_sessions_human"             "num_classesviewed"             
[39] "num_essaysreviewed"             "num_AI_practiceproblemsessions"
[41] "both_math"                      "both_ela"                      
[43] "all_tests"                      "math_moy_missing_eoy_present"  
[45] "ela_moy_missing_eoy_present"   

Import Penn data.

penn = read_csv("vt_penn_all_students.csv")
Rows: 1279 Columns: 46
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr  (4): school, school_grade, race, race_recoded
dbl (42): student_id, school_num, grade, grade3, grade4, grade5, grade6, mal...

ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
colnames(penn)
 [1] "student_id"                     "school_num"                    
 [3] "school"                         "grade"                         
 [5] "school_grade"                   "grade3"                        
 [7] "grade4"                         "grade5"                        
 [9] "grade6"                         "male"                          
[11] "iep"                            "ell"                           
[13] "frpl"                           "race"                          
[15] "race_recoded"                   "race_A"                        
[17] "race_B"                         "race_H"                        
[19] "race_O"                         "race_W"                        
[21] "moy_math_comp"                  "moy_math_z"                    
[23] "moy_ela_comp"                   "moy_ela_z"                     
[25] "eoy_math_comp"                  "eoy_math_z"                    
[27] "eoy_ela_comp"                   "eoy_ela_z"                     
[29] "treat"                          "vt_acct_id"                    
[31] "vt_present"                     "num_days_active"               
[33] "num_days_active_recoded"        "num_weeks"                     
[35] "num_sessions_week"              "total_sessions"                
[37] "num_sessions_AI"                "num_sessions_human"            
[39] "num_classesviewed"              "num_essaysreviewed"            
[41] "num_AI_practiceproblemsessions" "both_math"                     
[43] "both_ela"                       "all_tests"                     
[45] "math_moy_missing_eoy_present"   "ela_moy_missing_eoy_present"   

Import Cumberland data.

cumber = read_csv("vt_cumber_all_students.csv")
Rows: 8639 Columns: 46
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr  (4): school, school_grade, race, race_recoded
dbl (41): student_id, school_num, grade, grade6, grade7, grade8, male, iep, ...
lgl  (1): frpl

ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
colnames(cumber)
 [1] "student_id"                     "school_num"                    
 [3] "school"                         "grade"                         
 [5] "school_grade"                   "grade6"                        
 [7] "grade7"                         "grade8"                        
 [9] "male"                           "iep"                           
[11] "ell"                            "frpl"                          
[13] "at_risk"                        "race"                          
[15] "race_recoded"                   "race_B"                        
[17] "race_H"                         "race_Other"                    
[19] "race_Multi"                     "race_W"                        
[21] "moy_math_comp"                  "moy_math_z"                    
[23] "moy_ela_comp"                   "moy_ela_z"                     
[25] "eoy_math_comp"                  "eoy_math_z"                    
[27] "eoy_ela_comp"                   "eoy_ela_z"                     
[29] "treat"                          "vt_acct_id"                    
[31] "vt_present"                     "num_days_active"               
[33] "num_days_active_recoded"        "num_weeks"                     
[35] "num_sessions_week"              "total_sessions"                
[37] "num_sessions_AI"                "num_sessions_human"            
[39] "num_classesviewed"              "num_essaysreviewed"            
[41] "num_AI_practiceproblemsessions" "both_math"                     
[43] "both_ela"                       "all_tests"                     
[45] "math_moy_missing_eoy_present"   "ela_moy_missing_eoy_present"   

Preparing for pooled variables

Create district columns

penn$district <- "Penn"
kent$district <- "Kent"
cumber$district <- "Cumberland"

Clean race

Inspect similar categories for cleaning pooled variables later (after appending).

table(penn$race)

    AMERICAN INDIAN/ALASKAN                       ASIAN 
                         22                          31 
     BLACK/AFRICAN AMERICAN                    HISPANIC 
                       1101                          43 
MULTI RACIAL - NON HISPANIC             NATIVE HAWAIIAN 
                          1                           8 
            WHITE/CAUCASIAN 
                         73 
table(kent$race)

         American Indian or Alaska Native 
                                       21 
                                    Asian 
                                     1365 
                Black or African American 
                                      760 
                          Hispanic/Latino 
                                     1312 
Native Hawaiian or Other Pacific Islander 
                                      189 
                        Two or More Races 
                                      512 
                                    White 
                                     1475 
table(cumber$race)

    A     B     H     I Multi     P     W 
  181  4069  1465   100   816    46  1962 

Clean IEP

Inspect Kent’s Special Ed (sped) variable.

table(kent$sped)

   0    1 
4803  831 

Decision: Treat Kent’s SPED variable as an IEP to align with Cumberland and Penn.

kent$iep = kent$sped
table(kent$iep)

   0    1 
4803  831 

Clean ELL

Inspect Kent’s MLL variable.

table(kent$mll)

   0    1 
4023 1611 

Decision: Treat Kent’s MLL variable as ELL to align with Cumberland and Penn.

kent$ell= kent$mll
table(kent$ell)

   0    1 
4023 1611 

Clean FRPL

Free/ reduced price lunch as proxy for low income status. FRPL for Penn, Low-income for Kent, but FRPL missing for Cumberland.

Inspect frpl for Penn.

table(penn$frpl)

  0   1 
353 926 

Decision: Treat Penn’s FRPL as low income variable to align with Kent.

penn$low_inc = penn$frpl
table(penn$low_inc)

  0   1 
353 926 

Decision: Treat Cumberland’s FRPL as low income variable to align with Kent. Note, since FRPL data was all missing, Cumberland will not have low-income status available and thus will be excluded from this subgroup analysis.

table(cumber$frpl)
< table of extent 0 >
cumber$low_inc = cumber$frpl
table(cumber$low_inc)
< table of extent 0 >

Clean At-risk

Inspect Kent’s variable for low academic performance

table(kent$low_ap)

   0    1 
4157 1477 

Decision: Treat Kent’s variable for low academic performance as at-risk variable to align with Cumberland. Since nothing similar is present fr Penn, create at-risk variable for Penn where all entries at NA. Note, since at-risk data was not present, Penn will not have at-risk status available and thus will be excluded from this subgroup analysis.

kent$at_risk = kent$low_ap
table(kent$at_risk)

   0    1 
4157 1477 
penn$at_risk <- NA
table(penn$at_risk)
< table of extent 0 >

Check Cumberland coding for at-risk.

table(cumber$at_risk)

   0    1 
5973 2666 

Pool Data

Select columns for appending

#PENN
penn_to_pool = penn %>%
  dplyr::select(student_id,
                school_num, school, district, 
                grade, school_grade,
                
                male, iep, ell, low_inc, at_risk, race,
                
                moy_math_comp, moy_math_z,
                eoy_math_comp, eoy_math_z,
                
                moy_ela_comp, moy_ela_z,
                eoy_ela_comp, eoy_ela_z, 
                
                vt_acct_id, vt_present,
                num_days_active, num_days_active_recoded, 
                num_weeks, num_sessions_week,total_sessions,
                num_sessions_AI, num_sessions_human,
                num_classesviewed, num_essaysreviewed, num_AI_practiceproblemsessions, 
                
                both_math, both_ela, all_tests, 
                math_moy_missing_eoy_present, ela_moy_missing_eoy_present)

#KENT
kent_to_pool = kent %>%
  dplyr::select(student_id,
                school_num, school, district,  
                grade, school_grade,
                
                male, iep, ell, low_inc, at_risk, race,
                
                moy_math_comp, moy_math_z,
                eoy_math_comp, eoy_math_z,
                
                moy_ela_comp, moy_ela_z,
                eoy_ela_comp, eoy_ela_z, 
                
                vt_acct_id, vt_present,
                num_days_active, num_days_active_recoded, 
                num_weeks, num_sessions_week,total_sessions,
                num_sessions_AI, num_sessions_human,
                num_classesviewed, num_essaysreviewed, num_AI_practiceproblemsessions, 
                
                both_math, both_ela, all_tests, 
                math_moy_missing_eoy_present, ela_moy_missing_eoy_present)

#CUMBERLAND
cumber_to_pool = cumber %>%
  dplyr::select(student_id,
                school_num, school, district, 
                grade, school_grade,
                
                male, iep, ell, low_inc, at_risk, race,
                
                moy_math_comp, moy_math_z,
                eoy_math_comp, eoy_math_z,
                
                moy_ela_comp, moy_ela_z,
                eoy_ela_comp, eoy_ela_z, 
                
                vt_acct_id, vt_present,
                num_days_active, num_days_active_recoded, 
                num_weeks, num_sessions_week,total_sessions,
                num_sessions_AI, num_sessions_human,
                num_classesviewed, num_essaysreviewed, num_AI_practiceproblemsessions, 
                
                both_math, both_ela, all_tests, 
                math_moy_missing_eoy_present, ela_moy_missing_eoy_present)

Append data

Use rbind since columns matched by name and by order.

pooled = rbind(
  penn_to_pool,
  kent_to_pool,
  cumber_to_pool
)

Finish pooled race

table(pooled$race)

                                        A 
                                      181 
         American Indian or Alaska Native 
                                       21 
                  AMERICAN INDIAN/ALASKAN 
                                       22 
                                    Asian 
                                     1365 
                                    ASIAN 
                                       31 
                                        B 
                                     4069 
                Black or African American 
                                      760 
                   BLACK/AFRICAN AMERICAN 
                                     1101 
                                        H 
                                     1465 
                                 HISPANIC 
                                       43 
                          Hispanic/Latino 
                                     1312 
                                        I 
                                      100 
                                    Multi 
                                      816 
              MULTI RACIAL - NON HISPANIC 
                                        1 
                          NATIVE HAWAIIAN 
                                        8 
Native Hawaiian or Other Pacific Islander 
                                      189 
                                        P 
                                       46 
                        Two or More Races 
                                      512 
                                        W 
                                     1962 
                                    White 
                                     1475 
                          WHITE/CAUCASIAN 
                                       73 

Decision: Combine categories where they appears across all districts: White, Black, Hispanic, Asian, American Indian/Native, Multi-racial, Pacific Islander.

pooled = pooled %>%
mutate(
race_pooled = case_when(
race %in% c("WHITE/CAUCASIAN", "White", "W") ~ "White",
race %in% c("BLACK/AFRICAN AMERICAN","Black or African American", "B") ~ "Black",
race %in% c("ASIAN", "Asian", "A") ~ "Asian",
race %in% c("HISPANIC", "Hispanic/Latino","H") ~ "Hispanic",
race %in% c("MULTI RACIAL - NON HISPANIC","Two or More Races", "Multi") ~ "Multi-racial",
race %in% c("AMERICAN INDIAN/ALASKAN", "American Indian or Alaska Native", "I") ~ "Native American",#for American Indian
race %in% c("NATIVE HAWAIIAN", "Native Hawaiian or Other Pacific Islander", "P") ~ "Pacific Islander",
TRUE ~ NA_character_ 

)
)

table(pooled$race_pooled)

           Asian            Black         Hispanic     Multi-racial 
            1577             5930             2820             1329 
 Native American Pacific Islander            White 
             143              243             3510 

###Create race dummy

Collapse race_pooled so that Multi-racial, Native American, Pacific Islander, and Asian fall under Other Races.

pooled = pooled %>%
  mutate(race_pooled = recode(race_pooled,
      "Multi-racial" = "Other Races",
      "Native American" = "Other Races",
      "Pacific Islander" = "Other Races",
      "Asian" = "Other Races"
    )
  )
table(pooled$race_pooled)

      Black    Hispanic Other Races       White 
       5930        2820        3292        3510 
pooled = pooled %>%
  mutate(
    race_B     = ifelse(race_pooled == "Black", 1, ifelse(is.na(race_pooled), NA, 0)),
    race_H     = ifelse(race_pooled == "Hispanic", 1, ifelse(is.na(race_pooled), NA, 0)),
    race_O     = ifelse(race_pooled == "Other Races", 1, ifelse(is.na(race_pooled), NA, 0)),
    race_W     = ifelse(race_pooled == "White", 1, ifelse(is.na(race_pooled), NA, 0))
  )
table(pooled$race_O) #quick check coding on race_O

    0     1 
12260  3292 

###Create grade dummy

table(pooled$grade)

   3    4    5    6    7    8 
 304  315  314 5207 4775 4637 

Create dummies for each grade.

pooled = pooled %>%
  mutate(
    grade3 = if_else(grade == 3, 1, 0),
    grade4 = if_else(grade == 4, 1, 0),
    grade5 = if_else(grade == 5, 1, 0),
    grade6 = if_else(grade == 6, 1, 0), 
    grade7 = if_else(grade == 7, 1, 0),
    grade8 = if_else(grade == 8, 1, 0)
  )

Check coding for grade 6.

table(pooled$grade6)

    0     1 
10345  5207 

Create district grade

Create new categorical variable for grade by school

pooled = pooled %>%
  mutate(district_grade = paste(sub(" .*", "", district), grade))

table(pooled$district_grade)

Cumberland 6 Cumberland 7 Cumberland 8       Kent 6       Kent 7       Kent 8 
        2985         2892         2762         1876         1883         1875 
      Penn 3       Penn 4       Penn 5       Penn 6 
         304          315          314          346 

Create school_treat

“school_treat” will equal 1 if the school is a Live+AI school, and 0 will represent a comparison school.

treatment_schools <- c(
  "Spring Lake Middle",
  "Luther Nick Jeralds Middle",
  "Aldan Elementary School",
  "Bell Avenue Elementary School",
  "Colwyn Elementary School",
  "Walnut Street Elementary School",
  "Meridian Middle"
)

pooled$school_treat = ifelse(pooled$school %in% treatment_schools, 1, 0)
table(pooled$school_treat) #provides enrollment between treat and comp schools

    0     1 
13508  2044 

Construct treat_itt (based on Live+AI offering)

Variable vt_present indicates whether a student has a VT Account ID (i.e., was offered Live+AI tutoring)

table(pooled$vt_present)

    0     1 
14358  1194 

Double check coding on vt_present. Ensure all students with VT Account IDs were only at treatment school sites.

pooled %>%
  filter(vt_present == 1) %>%
  count(school_treat, name = "n") %>%
  mutate(
    pct = round(100 * n / sum(n), 1)
  )
# A tibble: 1 × 3
  school_treat     n   pct
         <dbl> <int> <dbl>
1            1  1194   100
  1. Filter out “treatment” students at treatment schools who were not offered Live+AI. They are not part of the final sample group and should not be used as comparison student under RQ1. Among students where vt_present = 1, exclude those where num_sessions_week is less than once per week.
pooled_treat_comp = pooled %>%
  filter(
    #keep students at comp schools
    school_treat == 0 |
      #keep students at treat schools IF
      (school_treat == 1 & 
         # VT ID is present aka offered Live+AI
         vt_present == 1) )
  1. Create treat: Equals 1 if students had more than 2 sessions. All else equals 0.

Based treat_itt grouping based on vt_present. Equal 1 if student was offered Live+AI, i.e., intended to be treated (regardless of usage).

pooled_treat_comp$treat_itt = pooled_treat_comp$vt_present
table(pooled_treat_comp$treat_itt)

    0     1 
13508  1194 

Construct treat_usage (based on usage)

Variable “treat” will equal 1 if a student offered Live+AI meets the threshold of at least 1 session per week.

  1. Filter out “treatment” students who do not meet usage threshold as well as students at treatment schools who were not offered Live+AI. They are not part of the final dosage sample group and should not be used as comparison student under RQ4. Among students where vt_present = 1, exclude those where num_sessions_week is less than once per week.
pooled_treat_comp_usage = pooled_treat_comp %>%
  filter(
    #keep students at comp schools
    school_treat == 0 |
      #keep students at treat schools IF
      (school_treat == 1 & 
         # meet usage threshold & VT ID is present
         num_sessions_week >= 1 &
         vt_present == 1) )
  1. Create treat: Equals 1 if students had at least one session logged per week. All else equals 0. Kent = 87, Cumberland = 350, Kent = 99, Total = 536.
pooled_treat_comp_usage = pooled_treat_comp_usage %>%
  mutate(treat_usage = ifelse(!is.na(num_sessions_week) &  num_sessions_week >= 1, 1, 0))
table(pooled_treat_comp_usage$treat_usage)

    0     1 
13508   536 

Pooled export

write.csv(pooled, file = "vt_pooled_all_students.csv", row.names = FALSE)

MATH e2i Coach exports

ITT files (treat_itt)

Save e2i file with only containing matched students from the merge with complete test info for Math

table(pooled_treat_comp$both_math)

    0     1 
 2432 12270 
pooled_treat_comp_math = pooled_treat_comp %>% filter(both_math == 1)
print(pooled_treat_comp_math) 
# A tibble: 12,270 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
10     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 12,260 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_math, file = "vt_pooled_e2i_complete_MATH_tests.csv", row.names = FALSE)

Race - Black

pooled_treat_comp_math_black = pooled_treat_comp_math %>% filter(race_pooled == "Black")
write.csv(pooled_treat_comp_math_black, file = "vt_pooled_e2i_complete_math_black.csv", row.names = FALSE)

Race - White

pooled_treat_comp_math_white = pooled_treat_comp_math %>% filter(race_pooled == "White")
write.csv(pooled_treat_comp_math_white, file = "vt_pooled_e2i_complete_math_white.csv", row.names = FALSE)

Race - Hispanic

pooled_treat_comp_math_hisp = pooled_treat_comp_math %>% filter(race_pooled == "Hispanic")
write.csv(pooled_treat_comp_math_hisp, file = "vt_pooled_e2i_complete_math_hisp.csv", row.names = FALSE)

Race - Other

pooled_treat_comp_math_other = pooled_treat_comp_math %>% filter(race_pooled == "Other Races")
write.csv(pooled_treat_comp_math_other, file = "vt_pooled_e2i_complete_math_other.csv", row.names = FALSE)

Low-income

pooled_treat_comp_low_inc = pooled_treat_comp_math %>% filter(low_inc == 1)
print(pooled_treat_comp_low_inc) 
# A tibble: 2,843 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 8     204851          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204895          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
10     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
# ℹ 2,833 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_low_inc, file = "vt_pooled_e2i_complete_math_low-inc.csv", row.names = FALSE)

Non low-income

pooled_treat_comp_non_low_inc = pooled_treat_comp_math %>% filter(low_inc == 0)
print(pooled_treat_comp_non_low_inc) 
# A tibble: 2,543 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 2     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 3     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     204672          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     205714          4 Colwyn E… Penn         4 Colwyn 4         0     0     0
 6     205977          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 7     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 8     206663          4 Colwyn E… Penn         3 Colwyn 3         1     0     0
 9     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
10     208446          4 Colwyn E… Penn         4 Colwyn 4         1     0     0
# ℹ 2,533 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_low_inc, file = "vt_pooled_e2i_complete_math_low-inc_non.csv", row.names = FALSE)

Male

pooled_treat_comp_male = pooled_treat_comp_math %>% filter(male == 1)
print(pooled_treat_comp_male) 
# A tibble: 6,292 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204899          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 8     204948          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 9     204670          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
10     204672          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
# ℹ 6,282 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_male, file = "vt_pooled_e2i_complete_math_males.csv", row.names = FALSE)

Female

pooled_treat_comp_female = pooled_treat_comp_math %>% filter(male == 0)
print(pooled_treat_comp_female) 
# A tibble: 5,975 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 3     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 6     204851          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204895          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
10     205342          4 Colwyn E… Penn         4 Colwyn 4         0     0     0
# ℹ 5,965 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_female, file = "vt_pooled_e2i_complete_math_females.csv", row.names = FALSE)

ELL

pooled_treat_comp_ell = pooled_treat_comp_math %>% filter(ell == 1)
print(pooled_treat_comp_ell) 
# A tibble: 1,625 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
 2     204026          1 Aldan El… Penn         6 Aldan 6          0     0     1
 3     205865          1 Aldan El… Penn         4 Aldan 4          0     0     1
 4     203671          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 5     203826          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 6     203838          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 7     203839          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 8     204068          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 9     204642          2 Ardmore … Penn         6 Ardmore 6        0     1     1
10     204665          2 Ardmore … Penn         5 Ardmore 5        1     0     1
# ℹ 1,615 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_ell, file = "vt_pooled_e2i_complete_math_ell.csv", row.names = FALSE)

Non ELL

pooled_treat_comp_non_ell = pooled_treat_comp_math %>% filter(ell == 0)
print(pooled_treat_comp_non_ell) 
# A tibble: 10,645 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
10     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 10,635 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_ell, file = "vt_pooled_e2i_complete_math_ell_non.csv", row.names = FALSE)

IEP

pooled_treat_comp_iep = pooled_treat_comp_math %>% filter(iep == 1)
print(pooled_treat_comp_iep) 
# A tibble: 1,761 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 3     204899          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 4     205255          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 5     205323          4 Colwyn E… Penn         4 Colwyn 4         1     1     0
 6     205977          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 7     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 8     206840          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 9     208983          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
10     209090          4 Colwyn E… Penn         4 Colwyn 4         1     1     0
# ℹ 1,751 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_iep, file = "vt_pooled_e2i_complete_math_iep.csv", row.names = FALSE)

Non IEP

pooled_treat_comp_non_iep = pooled_treat_comp_math %>% filter(iep == 0)
print(pooled_treat_comp_non_iep) 
# A tibble: 10,509 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 4     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 5     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 6     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204851          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
10     204895          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 10,499 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_iep, file = "vt_pooled_e2i_complete_math_iep_non.csv", row.names = FALSE)

At-risk

pooled_treat_comp_at_risk = pooled_treat_comp_math %>% filter(at_risk == 1)
print(pooled_treat_comp_at_risk) 
# A tibble: 3,000 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     375851          5 Meridian… Kent         8 Meridian 8       1     0     0
 2     416127          5 Meridian… Kent         8 Meridian 8       1     0     0
 3     384433          5 Meridian… Kent         7 Meridian 7       0     0     1
 4     405321          5 Meridian… Kent         7 Meridian 7       1     0     0
 5     405300          5 Meridian… Kent         6 Meridian 6       0     0     0
 6     415714          5 Meridian… Kent         7 Meridian 7       0     0     0
 7     401708          5 Meridian… Kent         6 Meridian 6       0     0     0
 8     390733          5 Meridian… Kent         7 Meridian 7       1     1     0
 9     410498          5 Meridian… Kent         6 Meridian 6       0     0     1
10     395563          5 Meridian… Kent         6 Meridian 6       1     0     0
# ℹ 2,990 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_at_risk, file = "vt_pooled_e2i_complete_math_at-risk.csv", row.names = FALSE)

Non At-risk

pooled_treat_comp_non_at_risk = pooled_treat_comp_math %>% filter(at_risk == 0)
print(pooled_treat_comp_non_at_risk) 
# A tibble: 8,156 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     404464          2 Cedar He… Kent         8 Cedar 8          1     0     1
 2     388550          2 Cedar He… Kent         7 Cedar 7          0     0     0
 3     415487          2 Cedar He… Kent         7 Cedar 7          1     0     0
 4     373617          2 Cedar He… Kent         7 Cedar 7          0     0     0
 5     390483          2 Cedar He… Kent         7 Cedar 7          0     0     0
 6     397051          2 Cedar He… Kent         8 Cedar 8          1     0     0
 7     383688          2 Cedar He… Kent         8 Cedar 8          0     0     0
 8     384247          2 Cedar He… Kent         8 Cedar 8          0     0     0
 9     424874          2 Cedar He… Kent         6 Cedar 6          0     0     0
10     378571          2 Cedar He… Kent         8 Cedar 8          1     1     0
# ℹ 8,146 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_at_risk, file = "vt_pooled_e2i_complete_math_at-risk-non.csv", row.names = FALSE)

Dosage files (treat_usage)

Save e2i file with only containing matched students from the merge with complete test info for Math

table(pooled_treat_comp_usage$both_math)

    0     1 
 2376 11668 
pooled_treat_comp_usage_math = pooled_treat_comp_usage %>% filter(both_math == 1)
print(pooled_treat_comp_usage_math) 
# A tibble: 11,668 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
10     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
# ℹ 11,658 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_math, file = "vt_pooled_e2i_complete_MATH_tests_dosage.csv", row.names = FALSE)

Low-income

pooled_treat_comp_usage_low_inc = pooled_treat_comp_usage_math %>% filter(low_inc == 1)
print(pooled_treat_comp_usage_low_inc) 
# A tibble: 2,653 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
10     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
# ℹ 2,643 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_low_inc, file = "vt_pooled_e2i_complete_math_low-inc_dosage.csv", row.names = FALSE)

Non low-income

pooled_treat_comp_usage_non_low_inc = pooled_treat_comp_usage_math %>% filter(low_inc == 0)
print(pooled_treat_comp_usage_non_low_inc) 
# A tibble: 2,483 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 2     206663          4 Colwyn E… Penn         3 Colwyn 3         1     0     0
 3     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
 4     209842          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 5     209843          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 6     210124          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     203640          1 Aldan El… Penn         6 Aldan 6          0     0     0
 8     205300          1 Aldan El… Penn         4 Aldan 4          0     0     0
 9     206766          1 Aldan El… Penn         3 Aldan 3          1     0     0
10     207763          1 Aldan El… Penn         3 Aldan 3          1     0     0
# ℹ 2,473 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_low_inc, file = "vt_pooled_e2i_complete_math_low-inc_non_dosage.csv", row.names = FALSE)

Male

pooled_treat_comp_usage_male = pooled_treat_comp_usage_math %>% filter(male == 1)
print(pooled_treat_comp_usage_male) 
# A tibble: 6,005 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 4     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 7     206566          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     206663          4 Colwyn E… Penn         3 Colwyn 3         1     0     0
 9     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
10     208983          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
# ℹ 5,995 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_male, file = "vt_pooled_e2i_complete_math_males_dosage.csv", row.names = FALSE)

Female

pooled_treat_comp_usage_female = pooled_treat_comp_usage_math %>% filter(male == 0)
print(pooled_treat_comp_usage_female) 
# A tibble: 5,660 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 3     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 6     206439          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
 7     206678          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     206839          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
 9     206841          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
10     207735          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
# ℹ 5,650 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_female, file = "vt_pooled_e2i_complete_math_females_dosage.csv", row.names = FALSE)

ELL

pooled_treat_comp_usage_ell = pooled_treat_comp_usage_math %>% filter(ell == 1)
print(pooled_treat_comp_usage_ell) 
# A tibble: 1,611 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
 2     205865          1 Aldan El… Penn         4 Aldan 4          0     0     1
 3     203671          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 4     203826          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 5     203838          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 6     203839          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 7     204068          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 8     204642          2 Ardmore … Penn         6 Ardmore 6        0     1     1
 9     204665          2 Ardmore … Penn         5 Ardmore 5        1     0     1
10     204683          2 Ardmore … Penn         5 Ardmore 5        1     0     1
# ℹ 1,601 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_ell, file = "vt_pooled_e2i_complete_math_ell_dosage.csv", row.names = FALSE)

Non ELL

pooled_treat_comp_usage_non_ell = pooled_treat_comp_usage_math %>% filter(ell == 0)
print(pooled_treat_comp_usage_non_ell) 
# A tibble: 10,057 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
10     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
# ℹ 10,047 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_ell, file = "vt_pooled_e2i_complete_math_ell_non_dosage.csv", row.names = FALSE)

IEP

pooled_treat_comp_usage_iep = pooled_treat_comp_usage_math %>% filter(iep == 1)
print(pooled_treat_comp_usage_iep) 
# A tibble: 1,670 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 3     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 4     208983          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 5     209842          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 6     209843          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 7     205411          1 Aldan El… Penn         4 Aldan 4          1     1     0
 8     205448          1 Aldan El… Penn         4 Aldan 4          1     1     0
 9     207331          1 Aldan El… Penn         4 Aldan 4          1     1     0
10     203693          2 Ardmore … Penn         6 Ardmore 6        0     1     0
# ℹ 1,660 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_iep, file = "vt_pooled_e2i_complete_math_iep_dosage.csv", row.names = FALSE)

Non IEP

pooled_treat_comp_usage_non_iep = pooled_treat_comp_usage_math %>% filter(iep == 0)
print(pooled_treat_comp_usage_non_iep) 
# A tibble: 9,998 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 6     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 8     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 9     206439          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
10     206566          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
# ℹ 9,988 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_iep, file = "vt_pooled_e2i_complete_math_iep_non_dosage.csv", row.names = FALSE)

At-risk

pooled_treat_comp_usage_at_risk = pooled_treat_comp_usage_math %>% filter(at_risk == 1)
print(pooled_treat_comp_usage_at_risk) 
# A tibble: 2,872 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     375851          5 Meridian… Kent         8 Meridian 8       1     0     0
 2     416127          5 Meridian… Kent         8 Meridian 8       1     0     0
 3     384433          5 Meridian… Kent         7 Meridian 7       0     0     1
 4     405321          5 Meridian… Kent         7 Meridian 7       1     0     0
 5     405300          5 Meridian… Kent         6 Meridian 6       0     0     0
 6     415714          5 Meridian… Kent         7 Meridian 7       0     0     0
 7     401708          5 Meridian… Kent         6 Meridian 6       0     0     0
 8     390733          5 Meridian… Kent         7 Meridian 7       1     1     0
 9     410498          5 Meridian… Kent         6 Meridian 6       0     0     1
10     395563          5 Meridian… Kent         6 Meridian 6       1     0     0
# ℹ 2,862 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_at_risk, file = "vt_pooled_e2i_complete_math_at-risk_dosage.csv", row.names = FALSE)

Non At-risk

pooled_treat_comp_usage_non_at_risk = pooled_treat_comp_usage_math %>% filter(at_risk == 0)
print(pooled_treat_comp_usage_non_at_risk) 
# A tibble: 7,931 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     404464          2 Cedar He… Kent         8 Cedar 8          1     0     1
 2     388550          2 Cedar He… Kent         7 Cedar 7          0     0     0
 3     415487          2 Cedar He… Kent         7 Cedar 7          1     0     0
 4     373617          2 Cedar He… Kent         7 Cedar 7          0     0     0
 5     390483          2 Cedar He… Kent         7 Cedar 7          0     0     0
 6     397051          2 Cedar He… Kent         8 Cedar 8          1     0     0
 7     383688          2 Cedar He… Kent         8 Cedar 8          0     0     0
 8     384247          2 Cedar He… Kent         8 Cedar 8          0     0     0
 9     424874          2 Cedar He… Kent         6 Cedar 6          0     0     0
10     378571          2 Cedar He… Kent         8 Cedar 8          1     1     0
# ℹ 7,921 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_at_risk, file = "vt_pooled_e2i_complete_math_at-risk-non_dosage.csv", row.names = FALSE)

ELA e2i Coach exports

ITT files (treat_itt)

Save e2i file with only containing matched students from the merge with complete test info for ELA

table(pooled_treat_comp$both_ela)

    0     1 
 1950 12752 
#subset: keep only students with complete test info
pooled_treat_comp_ela = pooled_treat_comp %>% filter(both_ela == 1)
print(pooled_treat_comp_ela) 
# A tibble: 12,752 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
10     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 12,742 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_ela, file = "vt_pooled_e2i_complete_ELA_tests.csv", row.names = FALSE)

Race - Black

pooled_treat_comp_ela_black = pooled_treat_comp_ela %>% filter(race_pooled == "Black")
write.csv(pooled_treat_comp_ela_black, file = "vt_pooled_e2i_complete_ela_black.csv", row.names = FALSE)

Race - White

pooled_treat_comp_ela_white = pooled_treat_comp_ela %>% filter(race_pooled == "White")
write.csv(pooled_treat_comp_ela_white, file = "vt_pooled_e2i_complete_ela_white.csv", row.names = FALSE)

Race - Hispanic

pooled_treat_comp_ela_hisp = pooled_treat_comp_ela %>% filter(race_pooled == "Hispanic")
write.csv(pooled_treat_comp_ela_hisp, file = "vt_pooled_e2i_complete_ela_hisp.csv", row.names = FALSE)

Race - Other

pooled_treat_comp_ela_other = pooled_treat_comp_ela %>% filter(race_pooled == "Other Races")
write.csv(pooled_treat_comp_ela_other, file = "vt_pooled_e2i_complete_ela_other.csv", row.names = FALSE)

Low-income

pooled_treat_comp_low_inc = pooled_treat_comp_ela %>% filter(low_inc == 1)
print(pooled_treat_comp_low_inc) 
# A tibble: 2,898 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 8     204851          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204895          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
10     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
# ℹ 2,888 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_low_inc, file = "vt_pooled_e2i_complete_ela_low-inc.csv", row.names = FALSE)

Non low-income

pooled_treat_comp_non_low_inc = pooled_treat_comp_ela %>% filter(low_inc == 0)
print(pooled_treat_comp_non_low_inc) 
# A tibble: 2,564 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 2     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 3     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     204672          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     205714          4 Colwyn E… Penn         4 Colwyn 4         0     0     0
 6     205977          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 7     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 8     206663          4 Colwyn E… Penn         3 Colwyn 3         1     0     0
 9     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
10     208446          4 Colwyn E… Penn         4 Colwyn 4         1     0     0
# ℹ 2,554 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_low_inc, file = "vt_pooled_e2i_complete_ela_low-inc_non.csv", row.names = FALSE)

Male

pooled_treat_comp_male = pooled_treat_comp_ela %>% filter(male == 1)
print(pooled_treat_comp_male) 
# A tibble: 6,525 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204899          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 8     204948          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 9     204670          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
10     204672          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
# ℹ 6,515 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_male, file = "vt_pooled_e2i_complete_ela_males.csv", row.names = FALSE)

Female

pooled_treat_comp_female = pooled_treat_comp_ela %>% filter(male == 0)
print(pooled_treat_comp_female) 
# A tibble: 6,224 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 3     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 6     204851          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204895          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
10     205342          4 Colwyn E… Penn         4 Colwyn 4         0     0     0
# ℹ 6,214 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_female, file = "vt_pooled_e2i_complete_ela_females.csv", row.names = FALSE)

ELL

pooled_treat_comp_ell = pooled_treat_comp_ela %>% filter(ell == 1)
print(pooled_treat_comp_ell) 
# A tibble: 1,665 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
 2     204026          1 Aldan El… Penn         6 Aldan 6          0     0     1
 3     205865          1 Aldan El… Penn         4 Aldan 4          0     0     1
 4     203671          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 5     203826          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 6     203838          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 7     203839          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 8     204068          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 9     204642          2 Ardmore … Penn         6 Ardmore 6        0     1     1
10     204665          2 Ardmore … Penn         5 Ardmore 5        1     0     1
# ℹ 1,655 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_ell, file = "vt_pooled_e2i_complete_ela_ell.csv", row.names = FALSE)

Non ELL

pooled_treat_comp_non_ell = pooled_treat_comp_ela %>% filter(ell == 0)
print(pooled_treat_comp_non_ell) 
# A tibble: 11,087 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
10     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 11,077 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_ell, file = "vt_pooled_e2i_complete_ela_ell_non.csv", row.names = FALSE)

IEP

pooled_treat_comp_iep = pooled_treat_comp_ela %>% filter(iep == 1)
print(pooled_treat_comp_iep) 
# A tibble: 1,823 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 3     204899          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 4     205255          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 5     205323          4 Colwyn E… Penn         4 Colwyn 4         1     1     0
 6     205977          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 7     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 8     206840          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 9     208983          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
10     209090          4 Colwyn E… Penn         4 Colwyn 4         1     1     0
# ℹ 1,813 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_iep, file = "vt_pooled_e2i_complete_ela_iep.csv", row.names = FALSE)

Non IEP

pooled_treat_comp_non_iep = pooled_treat_comp_ela %>% filter(iep == 0)
print(pooled_treat_comp_non_iep) 
# A tibble: 10,929 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 4     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 5     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 6     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204851          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
10     204895          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 10,919 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_iep, file = "vt_pooled_e2i_complete_ela_iep_non.csv", row.names = FALSE)

At-risk

pooled_treat_comp_at_risk = pooled_treat_comp_ela %>% filter(at_risk == 1)
print(pooled_treat_comp_at_risk) 
# A tibble: 3,051 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     425285          5 Meridian… Kent         6 Meridian 6       1     0     0
 2     375851          5 Meridian… Kent         8 Meridian 8       1     0     0
 3     416127          5 Meridian… Kent         8 Meridian 8       1     0     0
 4     384433          5 Meridian… Kent         7 Meridian 7       0     0     1
 5     405321          5 Meridian… Kent         7 Meridian 7       1     0     0
 6     405300          5 Meridian… Kent         6 Meridian 6       0     0     0
 7     415714          5 Meridian… Kent         7 Meridian 7       0     0     0
 8     401708          5 Meridian… Kent         6 Meridian 6       0     0     0
 9     390733          5 Meridian… Kent         7 Meridian 7       1     1     0
10     410498          5 Meridian… Kent         6 Meridian 6       0     0     1
# ℹ 3,041 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_at_risk, file = "vt_pooled_e2i_complete_ela_at-risk.csv", row.names = FALSE)

Non At-risk

pooled_treat_comp_non_at_risk = pooled_treat_comp_ela %>% filter(at_risk == 0)
print(pooled_treat_comp_non_at_risk) 
# A tibble: 8,595 × 51
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     404464          2 Cedar He… Kent         8 Cedar 8          1     0     1
 2     415487          2 Cedar He… Kent         7 Cedar 7          1     0     0
 3     373617          2 Cedar He… Kent         7 Cedar 7          0     0     0
 4     390483          2 Cedar He… Kent         7 Cedar 7          0     0     0
 5     394182          2 Cedar He… Kent         6 Cedar 6          1     0     0
 6     397051          2 Cedar He… Kent         8 Cedar 8          1     0     0
 7     383688          2 Cedar He… Kent         8 Cedar 8          0     0     0
 8     384247          2 Cedar He… Kent         8 Cedar 8          0     0     0
 9     424874          2 Cedar He… Kent         6 Cedar 6          0     0     0
10     378571          2 Cedar He… Kent         8 Cedar 8          1     1     0
# ℹ 8,585 more rows
# ℹ 42 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_non_at_risk, file = "vt_pooled_e2i_complete_ela_at-risk-non.csv", row.names = FALSE)

Dosage files (treat_usage)

Save e2i file with only containing matched students from the merge with complete test info for ELA

table(pooled_treat_comp_usage$both_ela)

    0     1 
 1867 12177 
#subset: keep only students with complete test info
pooled_treat_comp_usage_ela = pooled_treat_comp_usage %>% filter(both_ela == 1)
print(pooled_treat_comp_usage_ela) 
# A tibble: 12,177 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
10     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
# ℹ 12,167 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_ela, file = "vt_pooled_e2i_complete_ELA_tests_dosage.csv", row.names = FALSE)

Low-income

pooled_treat_comp_usage_low_inc = pooled_treat_comp_usage_ela %>% filter(low_inc == 1)
print(pooled_treat_comp_usage_low_inc) 
# A tibble: 2,708 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
10     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
# ℹ 2,698 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_low_inc, file = "vt_pooled_e2i_complete_ela_low-inc_dosage.csv", row.names = FALSE)

Non low-income

pooled_treat_comp_usage_non_low_inc = pooled_treat_comp_usage_ela %>% filter(low_inc == 0)
print(pooled_treat_comp_usage_non_low_inc) 
# A tibble: 2,505 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 2     206663          4 Colwyn E… Penn         3 Colwyn 3         1     0     0
 3     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
 4     209842          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 5     209843          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 6     210124          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     203640          1 Aldan El… Penn         6 Aldan 6          0     0     0
 8     205300          1 Aldan El… Penn         4 Aldan 4          0     0     0
 9     206766          1 Aldan El… Penn         3 Aldan 3          1     0     0
10     207763          1 Aldan El… Penn         3 Aldan 3          1     0     0
# ℹ 2,495 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_low_inc, file = "vt_pooled_e2i_complete_ela_low-inc_non_dosage.csv", row.names = FALSE)

Male

pooled_treat_comp_usage_male = pooled_treat_comp_usage_ela %>% filter(male == 1)
print(pooled_treat_comp_usage_male) 
# A tibble: 6,256 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 4     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 7     206566          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     206663          4 Colwyn E… Penn         3 Colwyn 3         1     0     0
 9     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
10     208983          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
# ℹ 6,246 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_male, file = "vt_pooled_e2i_complete_ela_males_dosage.csv", row.names = FALSE)

Female

pooled_treat_comp_usage_female = pooled_treat_comp_usage_ela %>% filter(male == 0)
print(pooled_treat_comp_usage_female) 
# A tibble: 5,918 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 3     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 6     206439          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
 7     206678          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     206839          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
 9     206841          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
10     207735          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
# ℹ 5,908 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_female, file = "vt_pooled_e2i_complete_ela_females_dosage.csv", row.names = FALSE)

ELL

pooled_treat_comp_usage_ell = pooled_treat_comp_usage_ela %>% filter(ell == 1)
print(pooled_treat_comp_usage_ell) 
# A tibble: 1,651 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     206895          4 Colwyn E… Penn         3 Colwyn 3         1     0     1
 2     205865          1 Aldan El… Penn         4 Aldan 4          0     0     1
 3     203671          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 4     203826          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 5     203838          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 6     203839          2 Ardmore … Penn         6 Ardmore 6        1     0     1
 7     204068          2 Ardmore … Penn         6 Ardmore 6        0     0     1
 8     204642          2 Ardmore … Penn         6 Ardmore 6        0     1     1
 9     204665          2 Ardmore … Penn         5 Ardmore 5        1     0     1
10     204683          2 Ardmore … Penn         5 Ardmore 5        1     0     1
# ℹ 1,641 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_ell, file = "vt_pooled_e2i_complete_ela_ell_dosage.csv", row.names = FALSE)

Non ELL

pooled_treat_comp_usage_non_ell = pooled_treat_comp_usage_ela %>% filter(ell == 0)
print(pooled_treat_comp_usage_non_ell) 
# A tibble: 10,526 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 5     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 6     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 7     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 8     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
10     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
# ℹ 10,516 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_ell, file = "vt_pooled_e2i_complete_ela_ell_non_dosage.csv", row.names = FALSE)

IEP

pooled_treat_comp_usage_iep = pooled_treat_comp_usage_ela %>% filter(iep == 1)
print(pooled_treat_comp_usage_iep) 
# A tibble: 1,736 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
 3     206541          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 4     208983          4 Colwyn E… Penn         3 Colwyn 3         1     1     0
 5     209842          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 6     209843          4 Colwyn E… Penn         6 Colwyn 6         1     1     0
 7     205411          1 Aldan El… Penn         4 Aldan 4          1     1     0
 8     205448          1 Aldan El… Penn         4 Aldan 4          1     1     0
 9     207331          1 Aldan El… Penn         4 Aldan 4          1     1     0
10     203693          2 Ardmore … Penn         6 Ardmore 6        0     1     0
# ℹ 1,726 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_iep, file = "vt_pooled_e2i_complete_ela_iep_dosage.csv", row.names = FALSE)

Non IEP

pooled_treat_comp_usage_non_iep = pooled_treat_comp_usage_ela %>% filter(iep == 0)
print(pooled_treat_comp_usage_non_iep) 
# A tibble: 10,441 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 2     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 4     204896          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 5     204671          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 6     205051          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     205898          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 8     206039          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 9     206439          4 Colwyn E… Penn         3 Colwyn 3         0     0     0
10     206566          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
# ℹ 10,431 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_iep, file = "vt_pooled_e2i_complete_ela_iep_non_dosage.csv", row.names = FALSE)

At-risk

pooled_treat_comp_usage_at_risk = pooled_treat_comp_usage_ela %>% filter(at_risk == 1)
print(pooled_treat_comp_usage_at_risk) 
# A tibble: 2,928 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     425285          5 Meridian… Kent         6 Meridian 6       1     0     0
 2     375851          5 Meridian… Kent         8 Meridian 8       1     0     0
 3     416127          5 Meridian… Kent         8 Meridian 8       1     0     0
 4     384433          5 Meridian… Kent         7 Meridian 7       0     0     1
 5     405321          5 Meridian… Kent         7 Meridian 7       1     0     0
 6     405300          5 Meridian… Kent         6 Meridian 6       0     0     0
 7     415714          5 Meridian… Kent         7 Meridian 7       0     0     0
 8     401708          5 Meridian… Kent         6 Meridian 6       0     0     0
 9     390733          5 Meridian… Kent         7 Meridian 7       1     1     0
10     410498          5 Meridian… Kent         6 Meridian 6       0     0     1
# ℹ 2,918 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_at_risk, file = "vt_pooled_e2i_complete_ela_at-risk_dosage.csv", row.names = FALSE)

Non At-risk

pooled_treat_comp_usage_non_at_risk = pooled_treat_comp_usage_ela %>% filter(at_risk == 0)
print(pooled_treat_comp_usage_non_at_risk) 
# A tibble: 8,391 × 52
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     404464          2 Cedar He… Kent         8 Cedar 8          1     0     1
 2     415487          2 Cedar He… Kent         7 Cedar 7          1     0     0
 3     373617          2 Cedar He… Kent         7 Cedar 7          0     0     0
 4     390483          2 Cedar He… Kent         7 Cedar 7          0     0     0
 5     394182          2 Cedar He… Kent         6 Cedar 6          1     0     0
 6     397051          2 Cedar He… Kent         8 Cedar 8          1     0     0
 7     383688          2 Cedar He… Kent         8 Cedar 8          0     0     0
 8     384247          2 Cedar He… Kent         8 Cedar 8          0     0     0
 9     424874          2 Cedar He… Kent         6 Cedar 6          0     0     0
10     378571          2 Cedar He… Kent         8 Cedar 8          1     1     0
# ℹ 8,381 more rows
# ℹ 43 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …
write.csv(pooled_treat_comp_usage_non_at_risk, file = "vt_pooled_e2i_complete_ela_at-risk-non_dosage.csv", row.names = FALSE)

Subset students offered Live + AI only

“vt_present” will equal 1 if a student was offered Live+AI, as shown by the presence of a VT account ID.

Create pooled data frame containing only students with VT Account IDs. This will also include students from treatment schools with no usage or usage falling below the threshold (for RQ2 later).

pooled_vt = pooled %>%
  filter(vt_present == 1)
print(pooled_vt)
# A tibble: 1,194 × 50
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
10     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 1,184 more rows
# ℹ 41 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …

Attrition

Percent of students who did not take the EOY assessment, per subject, by study condition.

summary_attrition = pooled_treat_comp %>%
  group_by(treat_itt) %>%
  summarise(
    n_total = n(),
    n_missing_math = sum(is.na(eoy_math_comp)),
    n_missing_ela  = sum(is.na(eoy_ela_comp)),
    .groups = "drop"
  ) %>%
  pivot_longer(
    cols = c(n_missing_math, n_missing_ela),
    names_to = "subject",
    values_to = "n_missing"
  ) %>%
  mutate(
    subject = recode(
      subject,
      n_missing_math = "Math",
      n_missing_ela  = "ELA"
    ),
    pct_missing = 100 * n_missing / n_total
  ) %>%
  select(treat_itt, subject, n_total, n_missing, pct_missing)

summary_attrition
# A tibble: 4 × 5
  treat_itt subject n_total n_missing pct_missing
      <dbl> <chr>     <int>     <int>       <dbl>
1         0 Math      13508      1454       10.8 
2         0 ELA       13508       927        6.86
3         1 Math       1194        50        4.19
4         1 ELA        1194        82        6.87
write.csv(summary_attrition, file = "vt_pooled_attrition.csv", row.names = FALSE)

Alternative: denominator restricted to students who had the corresponding MOY subject score present.

summary_attrition = pooled_treat_comp %>%
  group_by(treat_itt) %>%
  summarise(
    # Math: restrict to students with MOY math score
    n_moy_math = sum(!is.na(moy_math_comp)),
    n_missing_math = sum(!is.na(moy_math_comp) & is.na(eoy_math_comp)),
    
    # ELA: restrict to students with MOY ELA score
    n_moy_ela = sum(!is.na(moy_ela_comp)),
    n_missing_ela = sum(!is.na(moy_ela_comp) & is.na(eoy_ela_comp)),
    
    .groups = "drop"
  ) %>%
  pivot_longer(
    cols = c(n_moy_math, n_missing_math, n_moy_ela, n_missing_ela),
    names_to = c(".value", "subject"),
    names_pattern = "n_(moy|missing)_(math|ela)"
  ) %>%
  mutate(
    subject = recode(
      subject,
      math = "Math",
      ela = "ELA"
    ),
    treatment_group = if_else(treat_itt == 1, "Treatment", "Comparison"),
    pct_missing = round(100 * missing / moy, 1)
  ) %>%
  rename(
    n_moy = moy,
    n_missing = missing
  ) %>%
  select(
    treatment_group,
    subject,
    n_moy,
    n_missing,
    pct_missing
  )

summary_attrition
# A tibble: 4 × 5
  treatment_group subject n_moy n_missing pct_missing
  <chr>           <chr>   <int>     <int>       <dbl>
1 Comparison      Math    12322      1142         9.3
2 Comparison      ELA     12360       638         5.2
3 Treatment       Math     1127        37         3.3
4 Treatment       ELA      1089        59         5.4
write.csv(summary_attrition, file = "vt_pooled_attrition_moy present.csv", row.names = FALSE)

NEW: Count of students EOY who were offered Live+AI at treatment schools (treat_itt = 1).

summary_attrition_eoy = pooled_treat_comp %>%
  filter(treat_itt == 1) %>%
  group_by(school)%>%
  summarise(
    n_treatment = n(),
    n_complete_math = sum(both_math == 1, na.rm = TRUE),
    n_complete_ela = sum(both_ela == 1, na.rm = TRUE)
  )

summary_attrition_eoy
# A tibble: 7 × 4
  school                          n_treatment n_complete_math n_complete_ela
  <chr>                                 <int>           <int>          <int>
1 Aldan Elementary School                  85              83             82
2 Bell Avenue Elementary School           115             115            114
3 Colwyn Elementary School                 71              70             70
4 Luther Nick Jeralds Middle              392             343            341
5 Meridian Middle                          88              83             87
6 Spring Lake Middle                      361             316            255
7 Walnut Street Elementary School          82              80             81
write.csv(summary_attrition_eoy, file = "vt_pooled_attrition_eoy_n.csv", row.names = FALSE)

MDES

Calculating inputs for MDES

Inspect count of groups (schools)

pooled_treat_comp %>%
  count(school, sort = TRUE)
# A tibble: 32 × 2
   school                   n
   <chr>                <int>
 1 Gray's Creek Middle    979
 2 Northwood Middle       884
 3 Cedar Heights Middle   844
 4 Mill Creek Middle      840
 5 Canyon Ridge Middle    834
 6 Meeker Middle          832
 7 Mac Williams Middle    804
 8 Mattson Middle         763
 9 R Max Abbott Middle    699
10 Westover Middle        645
# ℹ 22 more rows

MATH: SD of outcomes = 1.13

pooled_treat_comp %>%
  filter(both_math == 1) %>%
  summarise(
    sd_eoy_math_z = sd(eoy_math_z, na.rm = TRUE)
  )
# A tibble: 1 × 1
  sd_eoy_math_z
          <dbl>
1          1.13

MATH: SD of outcomes = 1.10

pooled_treat_comp %>%
  filter(both_ela == 1) %>%
  summarise(
    sd_eoy_ela_z = sd(eoy_ela_z, na.rm = TRUE)
  )
# A tibble: 1 × 1
  sd_eoy_ela_z
         <dbl>
1         1.10

Baseline EQ tables

Math

Import matched dataset.

math_matched = read_csv("vt-pooled-math-primary-analysis-itt.matching.output.csv")
Rows: 2180 Columns: 54
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr  (6): district_grade, school, district, school_grade, race, race_pooled
dbl (48): treat_itt, iep, moy_math_z, distance, weights, subclass, student_i...

ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
colnames(math_matched)
 [1] "treat_itt"                      "district_grade"                
 [3] "iep"                            "moy_math_z"                    
 [5] "distance"                       "weights"                       
 [7] "subclass"                       "student_id"                    
 [9] "school_num"                     "school"                        
[11] "district"                       "grade"                         
[13] "school_grade"                   "male"                          
[15] "ell"                            "low_inc"                       
[17] "at_risk"                        "race"                          
[19] "moy_math_comp"                  "eoy_math_comp"                 
[21] "eoy_math_z"                     "moy_ela_comp"                  
[23] "moy_ela_z"                      "eoy_ela_comp"                  
[25] "eoy_ela_z"                      "vt_acct_id"                    
[27] "vt_present"                     "num_days_active"               
[29] "num_days_active_recoded"        "num_weeks"                     
[31] "num_sessions_week"              "total_sessions"                
[33] "num_sessions_AI"                "num_sessions_human"            
[35] "num_classesviewed"              "num_essaysreviewed"            
[37] "num_AI_practiceproblemsessions" "both_math"                     
[39] "both_ela"                       "all_tests"                     
[41] "math_moy_missing_eoy_present"   "ela_moy_missing_eoy_present"   
[43] "race_pooled"                    "race_B"                        
[45] "race_H"                         "race_O"                        
[47] "race_W"                         "grade3"                        
[49] "grade4"                         "grade5"                        
[51] "grade6"                         "grade7"                        
[53] "grade8"                         "school_treat"                  

Tabulate stats for baseline EQ table - math sample. Use for narrative in RQ4, student level findings.

library(dplyr)
library(purrr)
library(tibble)

#--------------------------------------------------
# WWC standardized mean difference
#--------------------------------------------------
wwc_smd <- function(x_treat, x_comp) {

  x_treat <- x_treat[!is.na(x_treat)]
  x_comp  <- x_comp[!is.na(x_comp)]

  n_t <- length(x_treat)
  n_c <- length(x_comp)

  pooled_sd <- sqrt(
    (((n_t - 1) * sd(x_treat)^2) +
       ((n_c - 1) * sd(x_comp)^2)) /
      (n_t + n_c - 2)
  )

  (mean(x_treat) - mean(x_comp)) / pooled_sd
}

#--------------------------------------------------
# Create female indicator
#--------------------------------------------------
math_matched <- math_matched %>%
  mutate(female = 1 - male)

#--------------------------------------------------
# Variables to tabulate
#--------------------------------------------------
vars <- c(
  "moy_math_z",
  "race_B",
  "race_H",
  "race_W",
  "race_O",
  "low_inc",
  "iep",
  "ell",
  "male",
  "female",
  "grade3",
  "grade4",
  "grade5",
  "grade6",
  "grade7",
  "grade8"
)

labels <- c(
  "MOY Math z-score",
  "Black",
  "Hispanic",
  "White",
  "Other Race",
  "FRPL",
  "IEP",
  "ELL",
  "Male",
  "Female",
  "Grade 3",
  "Grade 4",
  "Grade 5",
  "Grade 6",
  "Grade 7",
  "Grade 8"
)

#--------------------------------------------------
# One row per characteristic
#--------------------------------------------------
math_baseline_tbl <- map2_dfr(vars, labels, function(v, lbl) {

  trt <- math_matched %>%
    filter(treat_itt == 1) %>%
    pull(all_of(v))

  cmp <- math_matched %>%
    filter(treat_itt == 0) %>%
    pull(all_of(v))

  tibble(
    Characteristic = lbl,

    Treatment = mean(trt, na.rm = TRUE),
    Comparison = mean(cmp, na.rm = TRUE),

    SMD = wwc_smd(trt, cmp),

    Median_Treat = if (v == "moy_math_z")
      median(trt, na.rm = TRUE) else NA_real_,

    Median_Comp = if (v == "moy_math_z")
      median(cmp, na.rm = TRUE) else NA_real_,

    SD_Treat = if (v == "moy_math_z")
      sd(trt, na.rm = TRUE) else NA_real_,

    SD_Comp = if (v == "moy_math_z")
      sd(cmp, na.rm = TRUE) else NA_real_
  )
}) %>%
  mutate(
    Abs_SMD = abs(SMD),
    WWC_Equivalence = case_when(
      Abs_SMD <= .05 ~ "Equivalent",
      Abs_SMD <= .25 ~ "Equivalent with adjustment",
      TRUE ~ "Not equivalent"
    )
  ) %>%
  select(
    Characteristic,
    Treatment,
    Comparison,
    SMD,
    Abs_SMD,
    Median_Treat,
    Median_Comp,
    SD_Treat,
    SD_Comp,
    WWC_Equivalence
  ) %>%
  mutate(
    across(where(is.numeric), ~ round(.x, 3))
  )

math_baseline_tbl
# A tibble: 16 × 10
   Characteristic   Treatment Comparison    SMD Abs_SMD Median_Treat Median_Comp
   <chr>                <dbl>      <dbl>  <dbl>   <dbl>        <dbl>       <dbl>
 1 MOY Math z-score    -0.199     -0.204  0.005   0.005        -0.23       -0.23
 2 Black                0.685      0.619  0.139   0.139        NA          NA   
 3 Hispanic             0.141      0.15  -0.023   0.023        NA          NA   
 4 White                0.076      0.128 -0.17    0.17         NA          NA   
 5 Other Race           0.097      0.104 -0.021   0.021        NA          NA   
 6 FRPL                 0.705      0.659  0.1     0.1          NA          NA   
 7 IEP                  0.123      0.122  0.003   0.003        NA          NA   
 8 ELL                  0.061      0.11  -0.174   0.174        NA          NA   
 9 Male                 0.484      0.507 -0.046   0.046        NA          NA   
10 Female               0.516      0.493  0.046   0.046        NA          NA   
11 Grade 3              0.077      0.077  0       0            NA          NA   
12 Grade 4              0.09       0.09   0       0            NA          NA   
13 Grade 5              0.061      0.061  0       0            NA          NA   
14 Grade 6              0.35       0.35   0       0            NA          NA   
15 Grade 7              0.212      0.212  0       0            NA          NA   
16 Grade 8              0.209      0.209  0       0            NA          NA   
# ℹ 3 more variables: SD_Treat <dbl>, SD_Comp <dbl>, WWC_Equivalence <chr>
write.csv(math_baseline_tbl, file = "baseline_eq_math.csv", row.names = FALSE)

Full sample description - Math.

# MOY math score summary
score_stats <- math_matched %>%
  summarise(
    Mean = mean(moy_math_z, na.rm = TRUE),
    SD   = sd(moy_math_z, na.rm = TRUE)
  ) %>%
  pivot_longer(
    cols = everything(),
    names_to = "Statistic",
    values_to = "Value"
  ) %>%
  mutate(
    Characteristic = "MOY Math Score"
  ) %>%
  select(Characteristic, Statistic, Value)

# Demographic and grade characteristics
demo_stats <- tribble(
  ~Characteristic,           ~Variable, ~Target,
  "Black",                   "race_B",  1,
  "Hispanic",                "race_H",  1,
  "White",                   "race_W",  1,
  "Other Race",              "race_O",  1,
  "FRPL Eligible",           "low_inc", 1,
  "IEP Status",              "iep",     1,
  "ELL Status",              "ell",     1,
  "Male",                    "male",    1,
  "Female",                  "male",    0,
  "Grade 3",                 "grade3",  1,
  "Grade 4",                 "grade4",  1,
  "Grade 5",                 "grade5",  1,
  "Grade 6",                 "grade6",  1,
  "Grade 7",                 "grade7",  1,
  "Grade 8",                 "grade8",  1
) %>%
  rowwise() %>%
  mutate(
    Count = sum(math_matched[[Variable]] == Target, na.rm = TRUE),
    Proportion = mean(math_matched[[Variable]] == Target, na.rm = TRUE)
  ) %>%
  ungroup() %>%
  select(Characteristic, Count, Proportion) %>%
  pivot_longer(
    cols = c(Count, Proportion),
    names_to = "Statistic",
    values_to = "Value"
  )

# Final neat tibble
full_sample_math <- bind_rows(
  score_stats,
  demo_stats
) %>%
  mutate(
    Value = round(Value, 3)
  )

full_sample_math
# A tibble: 32 × 3
   Characteristic Statistic     Value
   <chr>          <chr>         <dbl>
 1 MOY Math Score Mean         -0.201
 2 MOY Math Score SD            1.01 
 3 Black          Count      1422    
 4 Black          Proportion    0.652
 5 Hispanic       Count       317    
 6 Hispanic       Proportion    0.145
 7 White          Count       222    
 8 White          Proportion    0.102
 9 Other Race     Count       219    
10 Other Race     Proportion    0.1  
# ℹ 22 more rows
write.csv(full_sample_math, file = "full_sample_math.csv", row.names = FALSE)

Count of treamment students by school - math

math_matched %>%
  filter(school_treat == 1) %>%
  group_by(school) %>%
  summarise(
    n_students = n(),
    .groups = "drop"
  ) %>%
  arrange(desc(n_students))
# A tibble: 7 × 2
  school                          n_students
  <chr>                                <int>
1 Luther Nick Jeralds Middle             343
2 Spring Lake Middle                     316
3 Bell Avenue Elementary School          115
4 Aldan Elementary School                 83
5 Meridian Middle                         83
6 Walnut Street Elementary School         80
7 Colwyn Elementary School                70

ELA

Import matched dataset.

ela_matched = read_csv("vt-pooled-ela-primary-analysis-itt.matching.output.csv")
Rows: 2060 Columns: 55
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr  (6): district_grade, school, district, school_grade, race, race_pooled
dbl (49): treat_itt, iep, moy_ela_z, distance, weights, subclass, student_id...

ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
colnames(ela_matched)
 [1] "treat_itt"                      "district_grade"                
 [3] "iep"                            "moy_ela_z"                     
 [5] "distance"                       "weights"                       
 [7] "subclass"                       "student_id"                    
 [9] "school_num"                     "school"                        
[11] "district"                       "grade"                         
[13] "school_grade"                   "male"                          
[15] "ell"                            "low_inc"                       
[17] "at_risk"                        "race"                          
[19] "moy_math_comp"                  "moy_math_z"                    
[21] "eoy_math_comp"                  "eoy_math_z"                    
[23] "moy_ela_comp"                   "eoy_ela_comp"                  
[25] "eoy_ela_z"                      "vt_acct_id"                    
[27] "vt_present"                     "num_days_active"               
[29] "num_days_active_recoded"        "num_weeks"                     
[31] "num_sessions_week"              "total_sessions"                
[33] "num_sessions_AI"                "num_sessions_human"            
[35] "num_classesviewed"              "num_essaysreviewed"            
[37] "num_AI_practiceproblemsessions" "both_math"                     
[39] "both_ela"                       "all_tests"                     
[41] "math_moy_missing_eoy_present"   "ela_moy_missing_eoy_present"   
[43] "race_pooled"                    "race_A"                        
[45] "race_B"                         "race_H"                        
[47] "race_O"                         "race_W"                        
[49] "grade3"                         "grade4"                        
[51] "grade5"                         "grade6"                        
[53] "grade7"                         "grade8"                        
[55] "school_treat"                  

Tabulate stats for baseline EQ table - math sample. Use for narrative in RQ4, student level findings.

library(dplyr)
library(purrr)
library(tibble)

#--------------------------------------------------
# WWC standardized mean difference
#--------------------------------------------------
wwc_smd <- function(x_treat, x_comp) {

  x_treat <- x_treat[!is.na(x_treat)]
  x_comp  <- x_comp[!is.na(x_comp)]

  n_t <- length(x_treat)
  n_c <- length(x_comp)

  pooled_sd <- sqrt(
    (((n_t - 1) * sd(x_treat)^2) +
       ((n_c - 1) * sd(x_comp)^2)) /
      (n_t + n_c - 2)
  )

  (mean(x_treat) - mean(x_comp)) / pooled_sd
}

#--------------------------------------------------
# Create female indicator
#--------------------------------------------------
ela_matched <- ela_matched %>%
  mutate(female = 1 - male)

#--------------------------------------------------
# Variables to tabulate
#--------------------------------------------------
vars <- c(
  "moy_ela_z",
  "race_B",
  "race_H",
  "race_W",
  "race_O",
  "low_inc",
  "iep",
  "ell",
  "male",
  "female",
  "grade3",
  "grade4",
  "grade5",
  "grade6",
  "grade7",
  "grade8"
)

labels <- c(
  "MOY ela z-score",
  "Black",
  "Hispanic",
  "White",
  "Other Race",
  "FRPL",
  "IEP",
  "ELL",
  "Male",
  "Female",
  "Grade 3",
  "Grade 4",
  "Grade 5",
  "Grade 6",
  "Grade 7",
  "Grade 8"
)

#--------------------------------------------------
# One row per characteristic
#--------------------------------------------------
ela_baseline_tbl <- map2_dfr(vars, labels, function(v, lbl) {

  trt <- ela_matched %>%
    filter(treat_itt == 1) %>%
    pull(all_of(v))

  cmp <- ela_matched %>%
    filter(treat_itt == 0) %>%
    pull(all_of(v))

  tibble(
    Characteristic = lbl,

    Treatment = mean(trt, na.rm = TRUE),
    Comparison = mean(cmp, na.rm = TRUE),

    SMD = wwc_smd(trt, cmp),

    Median_Treat = if (v == "moy_ela_z")
      median(trt, na.rm = TRUE) else NA_real_,

    Median_Comp = if (v == "moy_ela_z")
      median(cmp, na.rm = TRUE) else NA_real_,

    SD_Treat = if (v == "moy_ela_z")
      sd(trt, na.rm = TRUE) else NA_real_,

    SD_Comp = if (v == "moy_ela_z")
      sd(cmp, na.rm = TRUE) else NA_real_
  )
}) %>%
  mutate(
    Abs_SMD = abs(SMD),
    WWC_Equivalence = case_when(
      Abs_SMD <= .05 ~ "Equivalent",
      Abs_SMD <= .25 ~ "Equivalent with adjustment",
      TRUE ~ "Not equivalent"
    )
  ) %>%
  select(
    Characteristic,
    Treatment,
    Comparison,
    SMD,
    Abs_SMD,
    Median_Treat,
    Median_Comp,
    SD_Treat,
    SD_Comp,
    WWC_Equivalence
  ) %>%
  mutate(
    across(where(is.numeric), ~ round(.x, 3))
  )

ela_baseline_tbl
# A tibble: 16 × 10
   Characteristic  Treatment Comparison    SMD Abs_SMD Median_Treat Median_Comp
   <chr>               <dbl>      <dbl>  <dbl>   <dbl>        <dbl>       <dbl>
 1 MOY ela z-score    -0.162     -0.157 -0.006   0.006         -0.1       -0.11
 2 Black               0.688      0.618  0.147   0.147         NA         NA   
 3 Hispanic            0.134      0.15  -0.045   0.045         NA         NA   
 4 White               0.078      0.126 -0.161   0.161         NA         NA   
 5 Other Race          0.1        0.106 -0.019   0.019         NA         NA   
 6 FRPL                0.705      0.696  0.02    0.02          NA         NA   
 7 IEP                 0.127      0.122  0.015   0.015         NA         NA   
 8 ELL                 0.061      0.103 -0.152   0.152         NA         NA   
 9 Male                0.475      0.497 -0.045   0.045         NA         NA   
10 Female              0.525      0.503  0.045   0.045         NA         NA   
11 Grade 3             0.082      0.082  0       0             NA         NA   
12 Grade 4             0.094      0.094  0       0             NA         NA   
13 Grade 5             0.065      0.065  0       0             NA         NA   
14 Grade 6             0.368      0.368  0       0             NA         NA   
15 Grade 7             0.22       0.22   0       0             NA         NA   
16 Grade 8             0.171      0.171  0       0             NA         NA   
# ℹ 3 more variables: SD_Treat <dbl>, SD_Comp <dbl>, WWC_Equivalence <chr>
write.csv(ela_baseline_tbl, file = "baseline_eq_ela.csv", row.names = FALSE)

Full sample description - ela.

# MOY ela score summary
score_stats <- ela_matched %>%
  summarise(
    Mean = mean(moy_ela_z, na.rm = TRUE),
    SD   = sd(moy_ela_z, na.rm = TRUE)
  ) %>%
  pivot_longer(
    cols = everything(),
    names_to = "Statistic",
    values_to = "Value"
  ) %>%
  mutate(
    Characteristic = "MOY ela Score"
  ) %>%
  select(Characteristic, Statistic, Value)

# Demographic and grade characteristics
demo_stats <- tribble(
  ~Characteristic,           ~Variable, ~Target,
  "Black",                   "race_B",  1,
  "Hispanic",                "race_H",  1,
  "White",                   "race_W",  1,
  "Other Race",              "race_O",  1,
  "FRPL Eligible",           "low_inc", 1,
  "IEP Status",              "iep",     1,
  "ELL Status",              "ell",     1,
  "Male",                    "male",    1,
  "Female",                  "male",    0,
  "Grade 3",                 "grade3",  1,
  "Grade 4",                 "grade4",  1,
  "Grade 5",                 "grade5",  1,
  "Grade 6",                 "grade6",  1,
  "Grade 7",                 "grade7",  1,
  "Grade 8",                 "grade8",  1
) %>%
  rowwise() %>%
  mutate(
    Count = sum(ela_matched[[Variable]] == Target, na.rm = TRUE),
    Proportion = mean(ela_matched[[Variable]] == Target, na.rm = TRUE)
  ) %>%
  ungroup() %>%
  select(Characteristic, Count, Proportion) %>%
  pivot_longer(
    cols = c(Count, Proportion),
    names_to = "Statistic",
    values_to = "Value"
  )

# Final neat tibble
full_sample_ela <- bind_rows(
  score_stats,
  demo_stats
) %>%
  mutate(
    Value = round(Value, 3)
  )

full_sample_ela
# A tibble: 32 × 3
   Characteristic Statistic     Value
   <chr>          <chr>         <dbl>
 1 MOY ela Score  Mean         -0.16 
 2 MOY ela Score  SD            0.864
 3 Black          Count      1346    
 4 Black          Proportion    0.653
 5 Hispanic       Count       292    
 6 Hispanic       Proportion    0.142
 7 White          Count       210    
 8 White          Proportion    0.102
 9 Other Race     Count       212    
10 Other Race     Proportion    0.103
# ℹ 22 more rows
write.csv(full_sample_ela, file = "full_sample_ela.csv", row.names = FALSE)

Count of treamment students by school - ela

ela_matched %>%
  filter(school_treat == 1) %>%
  group_by(school) %>%
  summarise(
    n_students = n(),
    .groups = "drop"
  ) %>%
  arrange(desc(n_students))
# A tibble: 7 × 2
  school                          n_students
  <chr>                                <int>
1 Luther Nick Jeralds Middle             341
2 Spring Lake Middle                     255
3 Bell Avenue Elementary School          114
4 Meridian Middle                         87
5 Aldan Elementary School                 82
6 Walnut Street Elementary School         81
7 Colwyn Elementary School                70

Research Question 1

What are the demographic and academic characteristics of students and schools in LIVE+ AI schools and non-LIVE+ AI schools? We will summarize the demographic and academic characteristics of students and schools in the treatment and comparison groups by reporting descriptive statistics (mean, median, SDs) across both groups. In accordance with What Works Clearinghouse v5.0 standards, we will also report standardized mean differences between the treatment and comparison schools.

Function for data summary

#create function to streamline calculations
calc_pct = function(x) {
  prop.table(table(x)) * 100
}

School level characteristics

Characteristics at the school level include MOY math and ELA scores, enrollment, and the percentage of students who are Black, Hispanic, White, FRPL-eligible, have IEPs, and are female. Note, each school is either a treatment or comparison school; this data frame does not contain students in comparison schools with unexpected usage (clean groupings by school).

Using pooled data dataframe, which includes students at treatment schools who were not offered Live+AI, students from treatment school who were offered Live+AI, and all students from comparison schools.

# Create female indicator
pooled <- pooled %>%
  mutate(
    female = case_when(
      male == 0 ~ 1,
      male == 1 ~ 0,
      TRUE ~ NA_real_
    )
  )

# Variables
continuous_vars <- c(
  "moy_math_z",
  "moy_ela_z"
)

binary_vars <- c(
  "race_B",
  "race_H",
  "race_W",
  "race_O",
  "low_inc",
  "iep",
  "ell",
  "male",
  "female"
)

# Labels
var_labels <- c(
  moy_math_z = "MOY math score",
  moy_ela_z  = "MOY ELA score",
  race_B     = "Black (%)",
  race_H     = "Hispanic (%)",
  race_W     = "White (%)",
  race_O     = "Other race (%)",
  low_inc    = "FRPL-eligible (%)",
  iep        = "IEP (%)",
  ell        = "ELL (%)",
  male       = "Male (%)",
  female     = "Female (%)"
)

# Function for continuous variables
summarize_continuous <- function(data, var) {

  stats <- data %>%
    filter(
      school_treat %in% c(0, 1),
      !is.na(.data[[var]])
    ) %>%
    group_by(school_treat) %>%
    summarise(
      n = n(),
      mean = mean(.data[[var]]),
      median = median(.data[[var]]),
      sd = sd(.data[[var]]),
      .groups = "drop"
    )

  comparison <- stats %>% filter(school_treat == 0)
  treatment  <- stats %>% filter(school_treat == 1)

  pooled_sd <- sqrt(
    (
      (treatment$n - 1) * treatment$sd^2 +
      (comparison$n - 1) * comparison$sd^2
    ) /
    (treatment$n + comparison$n - 2)
  )

  smd <- (treatment$mean - comparison$mean) / pooled_sd

  tibble(
    Characteristics = unname(var_labels[var]),
    Comparison_N = comparison$n,
    Comparison_Mean = comparison$mean,
    Comparison_Median = comparison$median,
    Comparison_SD = comparison$sd,
    Treatment_N = treatment$n,
    Treatment_Mean = treatment$mean,
    Treatment_Median = treatment$median,
    Treatment_SD = treatment$sd,
    Standardized_Difference = smd
  )
}

# Function for binary variables
summarize_binary <- function(data, var) {

  stats <- data %>%
    filter(
      school_treat %in% c(0, 1),
      !is.na(.data[[var]])
    ) %>%
    group_by(school_treat) %>%
    summarise(
      n = n(),
      mean = mean(.data[[var]]),
      median = median(.data[[var]]),
      sd = sd(.data[[var]]),
      .groups = "drop"
    )

  comparison <- stats %>% filter(school_treat == 0)
  treatment  <- stats %>% filter(school_treat == 1)

  pooled_sd <- sqrt(
    (
      (treatment$n - 1) * treatment$sd^2 +
      (comparison$n - 1) * comparison$sd^2
    ) /
    (treatment$n + comparison$n - 2)
  )

  smd <- (treatment$mean - comparison$mean) / pooled_sd

  tibble(
    Characteristics = unname(var_labels[var]),
    Comparison_N = comparison$n,
    Comparison_Mean = comparison$mean * 100,
    Comparison_Median = comparison$median,
    Comparison_SD = comparison$sd,
    Treatment_N = treatment$n,
    Treatment_Mean = treatment$mean * 100,
    Treatment_Median = treatment$median,
    Treatment_SD = treatment$sd,
    Standardized_Difference = smd
  )
}

# Final tibble
school_baseline_table <- bind_rows(
  map_dfr(
    continuous_vars,
    ~ summarize_continuous(pooled, .x)
  ),
  map_dfr(
    binary_vars,
    ~ summarize_binary(pooled, .x)
  )
) %>%
  mutate(
    across(
      c(
        Comparison_Mean,
        Comparison_Median,
        Comparison_SD,
        Treatment_Mean,
        Treatment_Median,
        Treatment_SD,
        Standardized_Difference
      ),
      ~ round(.x, 3)
    )
  )

print(school_baseline_table)
# A tibble: 11 × 10
   Characteristics  Comparison_N Comparison_Mean Comparison_Median Comparison_SD
   <chr>                   <int>           <dbl>             <dbl>         <dbl>
 1 MOY math score          12322          -0.108             -0.06         1.11 
 2 MOY ELA score           12360          -0.139              0.02         1.10 
 3 Black (%)               13508          35.5                0            0.479
 4 Hispanic (%)            13508          18.5                0            0.389
 5 White (%)               13508          24.5                0            0.43 
 6 Other race (%)          13508          21.5                0            0.411
 7 FRPL-eligible (…         5812          52.5                1            0.499
 8 IEP (%)                 13508          14.8                0            0.355
 9 ELL (%)                 13508          13.9                0            0.346
10 Male (%)                13504          51.4                1            0.5  
11 Female (%)              13504          48.6                0            0.5  
# ℹ 5 more variables: Treatment_N <int>, Treatment_Mean <dbl>,
#   Treatment_Median <dbl>, Treatment_SD <dbl>, Standardized_Difference <dbl>
write.csv(school_baseline_table, "vt_pooled_RQ1_school_summary_by_treat.csv", row.names = FALSE)

Random check on proportion of black students in comparison schools

pooled %>%
  filter(school_treat == 0) %>%
  summarise(
    n_students = n(),
    n_nonmissing_race_B = sum(!is.na(race_B)),
    n_black = sum(race_B == 1, na.rm = TRUE),
    prop_black = mean(race_B == 1, na.rm = TRUE),
    pct_black = mean(race_B == 1, na.rm = TRUE) * 100
  )
# A tibble: 1 × 5
  n_students n_nonmissing_race_B n_black prop_black pct_black
       <int>               <int>   <int>      <dbl>     <dbl>
1      13508               13508    4800      0.355      35.5

NEW

library(dplyr)
library(tidyr)

# Test score summary
score_summary <- pooled %>%
  group_by(school_treat) %>%
  summarise(
    moy_math_n      = sum(!is.na(moy_math_z)),
    moy_math_mean   = mean(moy_math_z, na.rm = TRUE),
    moy_math_median = median(moy_math_z, na.rm = TRUE),
    moy_math_sd     = sd(moy_math_z, na.rm = TRUE),

    moy_ela_n       = sum(!is.na(moy_ela_z)),
    moy_ela_mean    = mean(moy_ela_z, na.rm = TRUE),
    moy_ela_median  = median(moy_ela_z, na.rm = TRUE),
    moy_ela_sd      = sd(moy_ela_z, na.rm = TRUE),
    .groups = "drop"
  )

# Demographic summary
demo_summary <- pooled %>%
  group_by(school_treat) %>%
  summarise(
    n_students = n(),

    black_n    = sum(race_B == 1, na.rm = TRUE),
    black_prop = mean(race_B == 1, na.rm = TRUE),

    hisp_n     = sum(race_H == 1, na.rm = TRUE),
    hisp_prop  = mean(race_H == 1, na.rm = TRUE),

    white_n    = sum(race_W == 1, na.rm = TRUE),
    white_prop = mean(race_W == 1, na.rm = TRUE),

    other_n    = sum(race_O == 1, na.rm = TRUE),
    other_prop = mean(race_O == 1, na.rm = TRUE),

    frpl_n     = sum(low_inc == 1, na.rm = TRUE),
    frpl_prop  = mean(low_inc == 1, na.rm = TRUE),

    iep_n      = sum(iep == 1, na.rm = TRUE),
    iep_prop   = mean(iep == 1, na.rm = TRUE),

    ell_n      = sum(ell == 1, na.rm = TRUE),
    ell_prop   = mean(ell == 1, na.rm = TRUE),

    male_n     = sum(male == 1, na.rm = TRUE),
    male_prop  = mean(male == 1, na.rm = TRUE),

    female_n   = sum(male == 0, na.rm = TRUE),
    female_prop= mean(male == 0, na.rm = TRUE),

    .groups = "drop"
  )

# Combine into one wide table
school_characteristics <- left_join(
  score_summary,
  demo_summary,
  by = "school_treat"
)
school_characteristics
# A tibble: 2 × 28
  school_treat moy_math_n moy_math_mean moy_math_median moy_math_sd moy_ela_n
         <dbl>      <int>         <dbl>           <dbl>       <dbl>     <int>
1            0      12322        -0.108           -0.06        1.11     12360
2            1       1905        -0.205           -0.17        1.11      1859
# ℹ 22 more variables: moy_ela_mean <dbl>, moy_ela_median <dbl>,
#   moy_ela_sd <dbl>, n_students <int>, black_n <int>, black_prop <dbl>,
#   hisp_n <int>, hisp_prop <dbl>, white_n <int>, white_prop <dbl>,
#   other_n <int>, other_prop <dbl>, frpl_n <int>, frpl_prop <dbl>,
#   iep_n <int>, iep_prop <dbl>, ell_n <int>, ell_prop <dbl>, male_n <int>,
#   male_prop <dbl>, female_n <int>, female_prop <dbl>
write.csv(school_characteristics, "vt_pooled_RQ1_school_summary_by_treat.csv", row.names = FALSE)
library(dplyr)
library(purrr)
library(tibble)

# WWC standardized mean difference
wwc_smd <- function(x, treat) {

  dat <- tibble(x = x, treat = treat) %>%
    filter(!is.na(x), !is.na(treat))

  trt  <- dat %>% filter(treat == 1)
  comp <- dat %>% filter(treat == 0)

  n_t <- nrow(trt)
  n_c <- nrow(comp)

  m_t <- mean(trt$x)
  m_c <- mean(comp$x)

  sd_t <- sd(trt$x)
  sd_c <- sd(comp$x)

  sd_pool <- sqrt(
    ((n_t - 1) * sd_t^2 + (n_c - 1) * sd_c^2) /
      (n_t + n_c - 2)
  )

  (m_t - m_c) / sd_pool
}

#------------------------------------------
# Test score variables
#------------------------------------------

score_vars <- c(
  "moy_math_z",
  "moy_ela_z"
)

score_tbl <- map_dfr(score_vars, function(v) {

  trt <- pooled %>% filter(school_treat == 1)
  cmp <- pooled %>% filter(school_treat == 0)

  tibble(
    characteristic = v,

    trt_n      = sum(!is.na(trt[[v]])),
    trt_mean   = mean(trt[[v]], na.rm = TRUE),
    trt_median = median(trt[[v]], na.rm = TRUE),
    trt_sd     = sd(trt[[v]], na.rm = TRUE),

    cmp_n      = sum(!is.na(cmp[[v]])),
    cmp_mean   = mean(cmp[[v]], na.rm = TRUE),
    cmp_median = median(cmp[[v]], na.rm = TRUE),
    cmp_sd     = sd(cmp[[v]], na.rm = TRUE),

    smd = wwc_smd(pooled[[v]], pooled$school_treat)
  )
})

#------------------------------------------
# Binary demographic variables
#------------------------------------------

demo_vars <- c(
  race_B  = "Black",
  race_H  = "Hispanic",
  race_W  = "White",
  race_O  = "Other race",
  low_inc = "FRPL eligible",
  iep     = "IEP",
  ell     = "ELL",
  male    = "Male"
)

demo_tbl <- map_dfr(names(demo_vars), function(v) {

  trt <- pooled %>% filter(school_treat == 1)
  cmp <- pooled %>% filter(school_treat == 0)

  tibble(
    characteristic = demo_vars[[v]],

    trt_count = sum(trt[[v]] == 1, na.rm = TRUE),
    trt_prop  = mean(trt[[v]] == 1, na.rm = TRUE),

    cmp_count = sum(cmp[[v]] == 1, na.rm = TRUE),
    cmp_prop  = mean(cmp[[v]] == 1, na.rm = TRUE),

    smd = wwc_smd(pooled[[v]], pooled$school_treat)
  )
})

# Female (derived from male == 0)

female_tbl <- {
  female <- ifelse(is.na(pooled$male), NA, 1 - pooled$male)

  trt <- pooled %>% filter(school_treat == 1)
  cmp <- pooled %>% filter(school_treat == 0)

  tibble(
    characteristic = "Female",

    trt_count = sum(trt$male == 0, na.rm = TRUE),
    trt_prop  = mean(trt$male == 0, na.rm = TRUE),

    cmp_count = sum(cmp$male == 0, na.rm = TRUE),
    cmp_prop  = mean(cmp$male == 0, na.rm = TRUE),

    smd = wwc_smd(female, pooled$school_treat)
  )
}

demo_tbl <- bind_rows(demo_tbl, female_tbl)

#------------------------------------------
# Final table
#------------------------------------------

characteristics_tbl <- bind_rows(
  score_tbl,
  demo_tbl
) %>%
  mutate(
    abs_smd = abs(smd),
    wwc_equivalence = case_when(
      abs_smd <= 0.05 ~ "Equivalent",
      abs_smd <= 0.25 ~ "Adjustment Required",
      TRUE ~ "Not Equivalent"
    )
  )

characteristics_tbl
# A tibble: 11 × 16
   characteristic trt_n trt_mean trt_median trt_sd cmp_n cmp_mean cmp_median
   <chr>          <int>    <dbl>      <dbl>  <dbl> <int>    <dbl>      <dbl>
 1 moy_math_z      1905   -0.205      -0.17   1.11 12322   -0.108      -0.06
 2 moy_ela_z       1859   -0.198      -0.06   1.04 12360   -0.139       0.02
 3 Black             NA   NA          NA     NA       NA   NA          NA   
 4 Hispanic          NA   NA          NA     NA       NA   NA          NA   
 5 White             NA   NA          NA     NA       NA   NA          NA   
 6 Other race        NA   NA          NA     NA       NA   NA          NA   
 7 FRPL eligible     NA   NA          NA     NA       NA   NA          NA   
 8 IEP               NA   NA          NA     NA       NA   NA          NA   
 9 ELL               NA   NA          NA     NA       NA   NA          NA   
10 Male              NA   NA          NA     NA       NA   NA          NA   
11 Female            NA   NA          NA     NA       NA   NA          NA   
# ℹ 8 more variables: cmp_sd <dbl>, smd <dbl>, trt_count <int>, trt_prop <dbl>,
#   cmp_count <int>, cmp_prop <dbl>, abs_smd <dbl>, wwc_equivalence <chr>
write.csv(characteristics_tbl, "vt_pooled_RQ1_school_summary_by_treat.csv", row.names = FALSE)

Student level characteristics

Characteristics at the student level include MOY math and ELA scores and the percentage of students who are Black, Hispanic, White, FRPL-eligible, have IEPs and are female.

# Create female indicator
pooled_treat_comp <- pooled_treat_comp %>%
  mutate(
    female = case_when(
      male == 0 ~ 1,
      male == 1 ~ 0,
      TRUE ~ NA_real_ ) )

# Variables
continuous_vars <- c(
  "moy_math_z",
  "moy_ela_z")

binary_vars <- c(
  "race_B",
  "race_H",
  "race_W",
  "race_O",
  "low_inc",
  "iep",
  "ell",
  "male",
  "female")

# Labels
var_labels <- c(
  moy_math_z = "MOY math score",
  moy_ela_z  = "MOY ELA score",
  race_B     = "Black (%)",
  race_H     = "Hispanic (%)",
  race_W     = "White (%)",
  race_O     = "Other race (%)",
  low_inc    = "FRPL-eligible (%)",
  iep        = "IEP (%)",
  ell        = "ELL (%)",
  male       = "Male (%)",
  female     = "Female (%)"
)

# Continuous variables: Hedges' g
summarize_continuous <- function(data, var) {
  
  stats <- data %>%
    filter(
      treat_itt %in% c(0, 1),
      !is.na(.data[[var]])
    ) %>%
    group_by(treat_itt) %>%
    summarise(
      n = n(),
      mean = mean(.data[[var]]),
      median = median(.data[[var]]),
      sd = sd(.data[[var]]),
      .groups = "drop"
    )
  
  comp  <- stats %>% filter(treat_itt == 0)
  treat <- stats %>% filter(treat_itt == 1)
  
  ## Pooled SD
  pooled_sd <- sqrt(
    ((treat$n - 1) * treat$sd^2 +
       (comp$n - 1) * comp$sd^2) /
      (treat$n + comp$n - 2)
  )
  
  ## Hedges' g
  d <- (treat$mean - comp$mean) / pooled_sd
  df <- treat$n + comp$n - 2
  correction <- 1 - (3 / (4 * df - 1))
  hedges_g <- correction * d
  
  tibble(
    characteristic = unname(var_labels[var]),
    comparison_n = comp$n,
    comparison_mean = comp$mean,
    comparison_median = comp$median,
    comparison_sd = comp$sd,
    treatment_n = treat$n,
    treatment_mean = treat$mean,
    treatment_median = treat$median,
    treatment_sd = treat$sd,
    standardized_difference = hedges_g
  )
}

# Binary variables: Cox's Index
summarize_binary <- function(data, var) {
  
  stats <- data %>%
    filter(
      treat_itt %in% c(0, 1),
      !is.na(.data[[var]])
    ) %>%
    group_by(treat_itt) %>%
    summarise(
      n = n(),
      proportion = mean(.data[[var]]),
      median = median(.data[[var]]),
      sd = sd(.data[[var]]),
      .groups = "drop"
    )
  
  comp  <- stats %>% filter(treat_itt == 0)
  treat <- stats %>% filter(treat_itt == 1)
  
  
  ## Prevent undefined logits when proportion = 0 or 1
  eps <- 1e-6
  p_t <- pmin(pmax(treat$proportion, eps), 1 - eps)
  p_c <- pmin(pmax(comp$proportion, eps), 1 - eps)
  
  ## Cox's Index
  cox_index <- (
    log(p_t / (1 - p_t)) -
      log(p_c / (1 - p_c))
  ) / 1.65
  
  tibble(
    characteristic = unname(var_labels[var]),
    comparison_n = comp$n,
    comparison_mean = comp$proportion * 100,
    comparison_median = comp$median,
    comparison_sd = comp$sd,
    treatment_n = treat$n,
    treatment_mean = treat$proportion * 100,
    treatment_median = treat$median,
    treatment_sd = treat$sd,
    standardized_difference = cox_index)
}

# Create final tibble
baseline_table <- bind_rows(
  map_dfr(
    continuous_vars,
    ~ summarize_continuous(pooled_treat_comp, .x)
),
  map_dfr(
    binary_vars,
    ~ summarize_binary(pooled_treat_comp, .x) )
) %>%
  mutate(across(where(is.numeric), ~ round(.x, 3)))

print(baseline_table)
# A tibble: 11 × 10
   characteristic   comparison_n comparison_mean comparison_median comparison_sd
   <chr>                   <dbl>           <dbl>             <dbl>         <dbl>
 1 MOY math score          12322          -0.108             -0.06         1.11 
 2 MOY ELA score           12360          -0.139              0.02         1.10 
 3 Black (%)               13508          35.5                0            0.479
 4 Hispanic (%)            13508          18.5                0            0.389
 5 White (%)               13508          24.5                0            0.43 
 6 Other race (%)          13508          21.5                0            0.411
 7 FRPL-eligible (…         5812          52.5                1            0.499
 8 IEP (%)                 13508          14.8                0            0.355
 9 ELL (%)                 13508          13.9                0            0.346
10 Male (%)                13504          51.4                1            0.5  
11 Female (%)              13504          48.6                0            0.5  
# ℹ 5 more variables: treatment_n <dbl>, treatment_mean <dbl>,
#   treatment_median <dbl>, treatment_sd <dbl>, standardized_difference <dbl>
write.csv(baseline_table, "vt_pooled_RQ1_student_summary_by_treat.csv", row.names = FALSE)

Simplified table for sample description in the report.

# function to help count and proportion in one cell
fmt_np <- function(x) {
  n_yes <- sum(x == 1, na.rm = TRUE)
  n_tot <- sum(!is.na(x))
  pct <- 100 * n_yes / n_tot
  sprintf("%d (%.1f%%)", n_yes, pct)
}

# Helper function: mean test score
fmt_mean <- function(x) {
  sprintf("%.2f", mean(x, na.rm = TRUE))
}

sample_desc <- tibble(
  Characteristics = c(
    "MOY Math Score",
    "MOY ELA Score",
    "Black",
    "Hispanic",
    "White",
    "Other Race",
    "FRPL Eligible",
    "IEP",
    "ELL",
    "Male",
    "Female"
  ),

  `Full Sample` = c(
    fmt_mean(pooled_treat_comp$moy_math_z),
    fmt_mean(pooled_treat_comp$moy_ela_z),
    fmt_np(pooled_treat_comp$race_B),
    fmt_np(pooled_treat_comp$race_H),
    fmt_np(pooled_treat_comp$race_W),
    fmt_np(pooled_treat_comp$race_O),
    fmt_np(pooled_treat_comp$low_inc),
    fmt_np(pooled_treat_comp$iep),
    fmt_np(pooled_treat_comp$ell),
    fmt_np(pooled_treat_comp$male),
    sprintf(
      "%d (%.1f%%)",
      sum(pooled_treat_comp$male == 0, na.rm = TRUE),
      100 * mean(pooled_treat_comp$male == 0, na.rm = TRUE)
    )
  ),

  Treatment = c(
    fmt_mean(filter(pooled_treat_comp, treat_itt == 1)$moy_math_z),
    fmt_mean(filter(pooled_treat_comp, treat_itt == 1)$moy_ela_z),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$race_B),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$race_H),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$race_W),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$race_O),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$low_inc),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$iep),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$ell),
    fmt_np(filter(pooled_treat_comp, treat_itt == 1)$male),
    sprintf(
      "%d (%.1f%%)",
      sum(filter(pooled_treat_comp, treat_itt == 1)$male == 0, na.rm = TRUE),
      100 * mean(filter(pooled_treat_comp, treat_itt == 1)$male == 0, na.rm = TRUE)
    )
  ),

  Comparison = c(
    fmt_mean(filter(pooled_treat_comp, treat_itt == 0)$moy_math_z),
    fmt_mean(filter(pooled_treat_comp, treat_itt == 0)$moy_ela_z),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$race_B),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$race_H),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$race_W),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$race_O),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$low_inc),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$iep),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$ell),
    fmt_np(filter(pooled_treat_comp, treat_itt == 0)$male),
    sprintf(
      "%d (%.1f%%)",
      sum(filter(pooled_treat_comp, treat_itt == 0)$male == 0, na.rm = TRUE),
      100 * mean(filter(pooled_treat_comp, treat_itt == 0)$male == 0, na.rm = TRUE)
    )
  )
)

print(sample_desc)
# A tibble: 11 × 4
   Characteristics `Full Sample` Treatment   Comparison  
   <chr>           <chr>         <chr>       <chr>       
 1 MOY Math Score  -0.12         -0.20       -0.11       
 2 MOY ELA Score   -0.14         -0.15       -0.14       
 3 Black           5619 (38.2%)  819 (68.6%) 4800 (35.5%)
 4 Hispanic        2671 (18.2%)  167 (14.0%) 2504 (18.5%)
 5 White           3395 (23.1%)  91 (7.6%)   3304 (24.5%)
 6 Other Race      3017 (20.5%)  117 (9.8%)  2900 (21.5%)
 7 FRPL Eligible   3358 (53.7%)  307 (69.6%) 3051 (52.5%)
 8 IEP             2137 (14.5%)  142 (11.9%) 1995 (14.8%)
 9 ELL             1946 (13.2%)  71 (5.9%)   1875 (13.9%)
10 Male            7512 (51.1%)  575 (48.2%) 6937 (51.4%)
11 Female          7186 (48.9%)  619 (51.8%) 6567 (48.6%)
write.csv(sample_desc, "vt_pooled_student_sample_description.csv", row.names = FALSE)

Total sample count between Treatment (ITT) and Comparison

table(pooled_treat_comp$treat_itt)

    0     1 
13508  1194 

Mean and SD for MOY scores.

moy_summary <- tibble(
  Measure = c(
    "MOY math score",
    "MOY ELA score"
  ),

  `Full Sample` = c(
    sprintf(
      "%.2f (%.2f)",
      mean(pooled_treat_comp$moy_math_z, na.rm = TRUE),
      sd(pooled_treat_comp$moy_math_z, na.rm = TRUE)
    ),
    sprintf(
      "%.2f (%.2f)",
      mean(pooled_treat_comp$moy_ela_z, na.rm = TRUE),
      sd(pooled_treat_comp$moy_ela_z, na.rm = TRUE)
    )
  ),

  Treatment = c(
    with(
      filter(pooled_treat_comp, treat_itt == 1),
      sprintf("%.2f (%.2f)",
              mean(moy_math_z, na.rm = TRUE),
              sd(moy_math_z, na.rm = TRUE))
    ),
    with(
      filter(pooled_treat_comp, treat_itt == 1),
      sprintf("%.2f (%.2f)",
              mean(moy_ela_z, na.rm = TRUE),
              sd(moy_ela_z, na.rm = TRUE))
    )
  ),

  Comparison = c(
    with(
      filter(pooled_treat_comp, treat_itt == 0),
      sprintf("%.2f (%.2f)",
              mean(moy_math_z, na.rm = TRUE),
              sd(moy_math_z, na.rm = TRUE))
    ),
    with(
      filter(pooled_treat_comp, treat_itt == 0),
      sprintf("%.2f (%.2f)",
              mean(moy_ela_z, na.rm = TRUE),
              sd(moy_ela_z, na.rm = TRUE))
    )
  )
)

moy_summary
# A tibble: 2 × 4
  Measure        `Full Sample` Treatment    Comparison  
  <chr>          <chr>         <chr>        <chr>       
1 MOY math score -0.12 (1.10)  -0.20 (1.01) -0.11 (1.11)
2 MOY ELA score  -0.14 (1.08)  -0.15 (0.87) -0.14 (1.10)
write.csv(moy_summary, "vt_pooled_student_sample_description_test_info.csv", row.names = FALSE)

Research Question 2

What is the average usage and engagement for students using Live+ AI? Does usage vary meaningfully by student characteristics or school?

Descriptive Statistics

*Pulled together statistics that mimic we want did for Table 2 on the mid year report.

By school.

RQ2_school_summary = pooled_vt %>%
  group_by(school) %>%
  summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_school_summary
# A tibble: 7 × 8
  school                     n_students n_students_active n_students_meet_thre…¹
  <chr>                           <int>             <int>                  <int>
1 Aldan Elementary School            85                76                     21
2 Bell Avenue Elementary Sc…        115               102                     42
3 Colwyn Elementary School           71                69                     28
4 Luther Nick Jeralds Middle        392               376                    125
5 Meridian Middle                    88                88                     87
6 Spring Lake Middle                361               358                    225
7 Walnut Street Elementary …         82                81                      8
# ℹ abbreviated name: ¹​n_students_meet_threshold
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_school_summary, file = "vt_pooled_RQ2_school_summary.csv", row.names = FALSE)

By grade.

RQ2_grade_summary = pooled_vt %>%
  group_by(grade) %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_grade_summary
# A tibble: 6 × 8
  grade n_students n_students_active n_students_meet_threshold
  <dbl>      <int>             <int>                     <int>
1     3         84                73                        24
2     4         99                98                        34
3     5         67                64                        26
4     6        439               417                       194
5     7        256               252                       117
6     8        249               246                       141
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_grade_summary, file = "vt_pooled_RQ2_grade_summary.csv", row.names = FALSE)

By gender.

RQ2_gender_summary = pooled_vt %>%
    group_by(male) %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_gender_summary
# A tibble: 2 × 8
   male n_students n_students_active n_students_meet_threshold
  <dbl>      <int>             <int>                     <int>
1     0        619               602                       273
2     1        575               548                       263
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_gender_summary, file = "vt_pooled_RQ2_gender_summary.csv", row.names = FALSE)

By race.

RQ2_race_summary = pooled_vt %>%
    group_by(race_pooled) %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_race_summary
# A tibble: 4 × 8
  race_pooled n_students n_students_active n_students_meet_threshold
  <chr>            <int>             <int>                     <int>
1 Black              819               785                       327
2 Hispanic           167               163                       102
3 Other Races        117               114                        58
4 White               91                88                        49
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_race_summary, file = "vt_pooled_RQ2_race_summary.csv", row.names = FALSE)

By IEP.

RQ2_iep_summary = pooled_vt %>%
    group_by(iep) %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_iep_summary
# A tibble: 2 × 8
    iep n_students n_students_active n_students_meet_threshold
  <dbl>      <int>             <int>                     <int>
1     0       1052              1014                       490
2     1        142               136                        46
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_iep_summary, file = "vt_pooled_RQ2_iep_summary.csv", row.names = FALSE)

by ELL.

RQ2_ell_summary = pooled_vt %>%
    group_by(ell) %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_iep_summary
# A tibble: 2 × 8
    iep n_students n_students_active n_students_meet_threshold
  <dbl>      <int>             <int>                     <int>
1     0       1052              1014                       490
2     1        142               136                        46
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_ell_summary, file = "vt_pooled_RQ2_ell_summary.csv", row.names = FALSE)

By FRPL.

RQ2_frpl_summary = pooled_vt %>%
    group_by(low_inc) %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_frpl_summary
# A tibble: 3 × 8
  low_inc n_students n_students_active n_students_meet_threshold
    <dbl>      <int>             <int>                     <int>
1       0        134               123                        69
2       1        307               293                       117
3      NA        753               734                       350
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_frpl_summary, file = "vt_pooled_RQ2_frpl_summary.csv", row.names = FALSE)

By at risk.

RQ2_at_risk_summary = pooled_vt %>%
    group_by(at_risk) %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_at_risk_summary
# A tibble: 3 × 8
  at_risk n_students n_students_active n_students_meet_threshold
    <dbl>      <int>             <int>                     <int>
1       0        511               498                       268
2       1        330               324                       169
3      NA        353               328                        99
# ℹ 4 more variables: prop_students_meet_threshold <dbl>,
#   avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_at_risk_summary, file = "vt_pooled_RQ2_at-risk_summary.csv", row.names = FALSE)

Full sample.

RQ2_full_summary = pooled_vt %>%
    summarize(
    n_students = n(),
    n_students_active = sum(total_sessions > 0, na.rm = TRUE),
    n_students_meet_threshold = sum(num_sessions_week >= 1, na.rm = TRUE),
    prop_students_meet_threshold = n_students_meet_threshold / n_students,
    avg_sessions_per_week = mean(num_sessions_week, na.rm = TRUE),
    sd_sessions_per_week = sd(num_sessions_week, na.rm = TRUE),
    med_sessions_per_week = median(num_sessions_week, na.rm = TRUE),
    
    
    .groups = "drop"
  )

RQ2_full_summary
# A tibble: 1 × 7
  n_students n_students_active n_students_meet_threshold prop_students_meet_th…¹
       <int>             <int>                     <int>                   <dbl>
1       1194              1150                       536                   0.449
# ℹ abbreviated name: ¹​prop_students_meet_threshold
# ℹ 3 more variables: avg_sessions_per_week <dbl>, sd_sessions_per_week <dbl>,
#   med_sessions_per_week <dbl>
write.csv(RQ2_full_summary, file = "vt_pooled_RQ2_full_summary.csv", row.names = FALSE)

Number of active sessions among st students with least one active day logged.

pooled_vt %>% 
  filter(num_sessions_week > 0) %>%
  summarise(
    avg_sessions_per_week =  mean(num_sessions_week, na.rm = TRUE))
# A tibble: 1 × 1
  avg_sessions_per_week
                  <dbl>
1                  1.60

Data frame for reference For regressions, only work with those students who had VT Account IDs (those who were offered Live+AI). Pull from data frame where all students were included, regardless of meeting usage threshold.

print(pooled_vt)
# A tibble: 1,194 × 50
   student_id school_num school    district grade school_grade  male   iep   ell
        <dbl>      <dbl> <chr>     <chr>    <dbl> <chr>        <dbl> <dbl> <dbl>
 1     203802          4 Colwyn E… Penn         6 Colwyn 6         0     1     0
 2     203862          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 3     204106          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 4     204166          4 Colwyn E… Penn         6 Colwyn 6         0     0     0
 5     204211          4 Colwyn E… Penn         6 Colwyn 6         1     0     0
 6     204755          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 7     204759          4 Colwyn E… Penn         5 Colwyn 5         1     0     0
 8     204769          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
 9     204787          4 Colwyn E… Penn         5 Colwyn 5         1     1     0
10     204843          4 Colwyn E… Penn         5 Colwyn 5         0     0     0
# ℹ 1,184 more rows
# ℹ 41 more variables: low_inc <dbl>, at_risk <dbl>, race <chr>,
#   moy_math_comp <dbl>, moy_math_z <dbl>, eoy_math_comp <dbl>,
#   eoy_math_z <dbl>, moy_ela_comp <dbl>, moy_ela_z <dbl>, eoy_ela_comp <dbl>,
#   eoy_ela_z <dbl>, vt_acct_id <dbl>, vt_present <dbl>, num_days_active <dbl>,
#   num_days_active_recoded <dbl>, num_weeks <dbl>, num_sessions_week <dbl>,
#   total_sessions <dbl>, num_sessions_AI <dbl>, num_sessions_human <dbl>, …

Pooled - Regress sessions per week ~ School

# Full model
q2_model_pooled_by_school = lm( #before clustering by school
  num_sessions_week ~
    factor(school), 
  data = pooled_vt
)
summary(q2_model_pooled_by_school)

Call:
lm(formula = num_sessions_week ~ factor(school), data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-3.6435 -0.8018 -0.3831  0.2982 17.3861 

Coefficients:
                                              Estimate Std. Error t value
(Intercept)                                     0.7288     0.2016   3.614
factor(school)Bell Avenue Elementary School     0.2607     0.2659   0.980
factor(school)Colwyn Elementary School          0.4091     0.2989   1.369
factor(school)Luther Nick Jeralds Middle        0.1730     0.2224   0.778
factor(school)Meridian Middle                   3.4147     0.2827  12.078
factor(school)Spring Lake Middle                1.5774     0.2241   7.038
factor(school)Walnut Street Elementary School  -0.3167     0.2878  -1.101
                                              Pr(>|t|)    
(Intercept)                                   0.000314 ***
factor(school)Bell Avenue Elementary School   0.327118    
factor(school)Colwyn Elementary School        0.171344    
factor(school)Luther Nick Jeralds Middle      0.436865    
factor(school)Meridian Middle                  < 2e-16 ***
factor(school)Spring Lake Middle              3.29e-12 ***
factor(school)Walnut Street Elementary School 0.271254    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 1.859 on 1187 degrees of freedom
Multiple R-squared:  0.2226,    Adjusted R-squared:  0.2187 
F-statistic: 56.65 on 6 and 1187 DF,  p-value: < 2.2e-16

Reference group is Aldan Elementary School. Results: Aldan had an average a 0.73 sessions/week (p<0.001). Bell, Colwyn, and Luther observed slightly higher usage than Aldan but it was not significant. Meridian observed 3.41 sessions/week more than Aldan (p<0.001). Spring Lake observed 1.60 sessions/week more than Aldan (p<0.001). Walnut Street students logged 0.31 sessions/week less than Aldan but it was not signifcantly different.

Self QA - Passed

q2_qa_usage_by_school = pooled_vt %>%
  group_by(school) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_school)
# A tibble: 7 × 2
  school                          mean_num_sessions_week
  <chr>                                            <dbl>
1 Aldan Elementary School                          0.729
2 Bell Avenue Elementary School                    0.989
3 Colwyn Elementary School                         1.14 
4 Luther Nick Jeralds Middle                       0.902
5 Meridian Middle                                  4.14 
6 Spring Lake Middle                               2.31 
7 Walnut Street Elementary School                  0.412

Pooled - Regress sessions per week ~ District

# Full model
q2_model_pooled_by_district = lm( #before clustering by school
  num_sessions_week ~
    factor(district), 
  data = pooled_vt
)
summary(q2_model_pooled_by_district)

Call:
lm(formula = num_sessions_week ~ factor(district), data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-3.6435 -1.0366 -0.5593  0.3355 18.1172 

Coefficients:
                     Estimate Std. Error t value Pr(>|t|)    
(Intercept)           1.57508    0.07082  22.242  < 2e-16 ***
factor(district)Kent  2.56838    0.21892  11.732  < 2e-16 ***
factor(district)Penn -0.75266    0.12535  -6.004 2.55e-09 ***
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 1.943 on 1191 degrees of freedom
Multiple R-squared:  0.1477,    Adjusted R-squared:  0.1462 
F-statistic: 103.2 on 2 and 1191 DF,  p-value: < 2.2e-16
###################################
##Cluster-robust standard errors by school
q2_pooled_district_clustered = vcovCL(
  q2_model_pooled_by_district,
  cluster = ~ school_num
)

coeftest(
  q2_model_pooled_by_district,
  vcov = q2_pooled_district_clustered #after clustering by school
)

t test of coefficients:

                     Estimate Std. Error t value Pr(>|t|)    
(Intercept)           1.57508    0.53586  2.9394 0.003352 ** 
factor(district)Kent  2.56838    0.53586  4.7930 1.85e-06 ***
factor(district)Penn -0.75266    0.55365 -1.3595 0.174259    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Cumberland as reference group. Results, after school clustering: Average student at Cumberland has 1.57 sessions/week logged (p<0.01), Penn had 0.75 sessions/week less than Cumberland (not significant) and Kent had 2.56 session/week more than Cumberland (p<0.001).

Self QA - Passed

q2_qa_usage_by_district = pooled_vt %>%
  group_by(district) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_district)
# A tibble: 3 × 2
  district   mean_num_sessions_week
  <chr>                       <dbl>
1 Cumberland                  1.58 
2 Kent                        4.14 
3 Penn                        0.822

Pooled - Regress sessions per week ~ Male

# Full model
q2_model_pooled_by_male = lm(
  num_sessions_week ~
    male, 
  data = pooled_vt
)
summary(q2_model_pooled_by_male) #before clustering by school

Call:
lm(formula = num_sessions_week ~ male, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-1.5675 -1.2180 -0.7812  0.4233 18.1743 

Coefficients:
            Estimate Std. Error t value Pr(>|t|)    
(Intercept)  1.51804    0.08456  17.952   <2e-16 ***
male         0.04945    0.12185   0.406    0.685    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 2.104 on 1192 degrees of freedom
Multiple R-squared:  0.0001381, Adjusted R-squared:  -0.0007007 
F-statistic: 0.1647 on 1 and 1192 DF,  p-value: 0.6849
###################################
##Cluster-robust standard errors by school
q2_pooled_male_clustered = vcovCL(
  q2_model_pooled_by_male,
  cluster = ~ school_num
)

coeftest(
  q2_model_pooled_by_male,
  vcov = q2_pooled_male_clustered #after clustering
)

t test of coefficients:

            Estimate Std. Error t value  Pr(>|t|)    
(Intercept) 1.518044   0.404230  3.7554 0.0001814 ***
male        0.049451   0.041079  1.2038 0.2289055    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Female as reference group. Results, after school clustering: Average female has 1.51 sessions/week logged (p<0.001); males had 0.05 sessions/week (not significant).

Self QA - Passed

q2_qa_usage_by_male = pooled_vt %>%
  group_by(male) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_male)
# A tibble: 2 × 2
   male mean_num_sessions_week
  <dbl>                  <dbl>
1     0                   1.52
2     1                   1.57

Pooled - Regress sessions per week ~ Race

# Set White as the reference group
pooled_vt$race_pooled = relevel(
  factor(pooled_vt$race_pooled),
  ref = "White"
)

# Full model
q2_model_pooled_by_race = lm(
  num_sessions_week ~
    race_pooled, 
  data = pooled_vt
)
summary(q2_model_pooled_by_race) #before clustering by school

Call:
lm(formula = num_sessions_week ~ race_pooled, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-2.2383 -1.1371 -0.6108  0.4419 18.3973 

Coefficients:
                       Estimate Std. Error t value Pr(>|t|)    
(Intercept)              1.6933     0.2170   7.803 1.32e-14 ***
race_pooledBlack        -0.3983     0.2287  -1.741   0.0819 .  
race_pooledHispanic      0.4889     0.2697   1.813   0.0701 .  
race_pooledOther Races   0.5450     0.2893   1.883   0.0599 .  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 2.07 on 1190 degrees of freedom
Multiple R-squared:  0.03359,   Adjusted R-squared:  0.03115 
F-statistic: 13.79 on 3 and 1190 DF,  p-value: 7.683e-09
###################################
##Cluster-robust standard errors by school
q2_pooled_race_clustered = vcovCL(
  q2_model_pooled_by_race,
  cluster = ~ school_num
)

coeftest(
  q2_model_pooled_by_race,
  vcov = q2_pooled_race_clustered #after clustering
)

t test of coefficients:

                       Estimate Std. Error t value  Pr(>|t|)    
(Intercept)             1.69329    0.45620  3.7117 0.0002154 ***
race_pooledBlack       -0.39831    0.19004 -2.0960 0.0362954 *  
race_pooledHispanic     0.48890    0.29531  1.6556 0.0980750 .  
race_pooledOther Races  0.54496    0.50049  1.0888 0.2764441    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Reference group is White. Results after school clustering: Average White student had 1.69 sessions/week (p<0.001). Asian students used 2.38 sessions more per week (p<0.001) compared to White students. Black students used 0.39 sessions less per week (p<0.05) compared to White students. Hispanic students used 0.49 sessions more per week than White students (p<0.1). All other groups’ differences in usage was not significantly difference than Whites.

QA - Passed

q2_qa_usage_by_race = pooled_vt %>%
  group_by(race_pooled) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_race)
# A tibble: 4 × 2
  race_pooled mean_num_sessions_week
  <fct>                        <dbl>
1 White                         1.69
2 Black                         1.29
3 Hispanic                      2.18
4 Other Races                   2.24

Pooled - Regress sessions per week ~ IEP

# Full model
q2_model_pooled_by_iep = lm(
  num_sessions_week ~
    iep, 
  data = pooled_vt
)
summary(q2_model_pooled_by_iep) #before clustering by school

Call:
lm(formula = num_sessions_week ~ iep, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-1.5812 -1.1966 -0.7790  0.4188 18.4418 

Coefficients:
            Estimate Std. Error t value Pr(>|t|)    
(Intercept)  1.58119    0.06478   24.41   <2e-16 ***
iep         -0.33069    0.18786   -1.76   0.0786 .  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 2.101 on 1192 degrees of freedom
Multiple R-squared:  0.002593,  Adjusted R-squared:  0.001756 
F-statistic: 3.099 on 1 and 1192 DF,  p-value: 0.07861
###################################
##Cluster-robust standard errors by school
q2_pooled_iep_clustered = vcovCL(
  q2_model_pooled_by_iep,
  cluster = ~ school_num
)

coeftest(
  q2_model_pooled_by_iep,
  vcov = q2_pooled_iep_clustered #after clustering
)

t test of coefficients:

            Estimate Std. Error t value  Pr(>|t|)    
(Intercept)  1.58119    0.42502  3.7203 0.0002083 ***
iep         -0.33069    0.22998 -1.4379 0.1507233    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Reference group is students without IEP. Results after school clustering, Students without IEP averaged 1.58 sessions/week (p<0.001). Compared to students without IEPs, students with IEPs had 0.33 sessions less per week (not significant).

Self QA - Passed

q2_qa_usage_by_iep = pooled_vt %>%
  group_by(iep) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_iep)
# A tibble: 2 × 2
    iep mean_num_sessions_week
  <dbl>                  <dbl>
1     0                   1.58
2     1                   1.25

Pooled - Regress sessions per week ~ ELL

# Full model
q2_model_pooled_by_ell = lm(
  num_sessions_week ~
    ell, 
  data = pooled_vt
)
summary(q2_model_pooled_by_ell) #before clustering by school

Call:
lm(formula = num_sessions_week ~ ell, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-3.0966 -1.1372 -0.7372  0.4235 18.2551 

Coefficients:
            Estimate Std. Error t value Pr(>|t|)    
(Intercept)  1.43724    0.06154  23.354  < 2e-16 ***
ell          1.75939    0.25238   6.971 5.19e-12 ***
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 2.062 on 1192 degrees of freedom
Multiple R-squared:  0.03917,   Adjusted R-squared:  0.03837 
F-statistic:  48.6 on 1 and 1192 DF,  p-value: 5.188e-12
###################################
##Cluster-robust standard errors by school
q2_pooled_ell_clustered = vcovCL(
  q2_model_pooled_by_ell,
  cluster = ~ school_num
)

coeftest(
  q2_model_pooled_by_ell,
  vcov = q2_pooled_ell_clustered #after clustering
)

t test of coefficients:

            Estimate Std. Error t value  Pr(>|t|)    
(Intercept)  1.43724    0.37065  3.8776 0.0001112 ***
ell          1.75939    0.47191  3.7283 0.0002018 ***
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Reference group are student who are not ELLs. Results after clustering: Student who are not ELL have 1.44 session per week (p<0.001). ELLs logged 1.76 sessions more per week than non ELLs (p<0.001).

Self QA - Passed

q2_qa_usage_by_ell = pooled_vt %>%
  group_by(ell) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_ell)
# A tibble: 2 × 2
    ell mean_num_sessions_week
  <dbl>                  <dbl>
1     0                   1.44
2     1                   3.20

Pooled - Regress sessions per week ~ Low-Income

# Full model
q2_model_pooled_by_low_inc = lm(
  num_sessions_week ~
    low_inc, 
  data = pooled_vt
)
summary(q2_model_pooled_by_low_inc) #before clustering by school

Call:
lm(formula = num_sessions_week ~ low_inc, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-2.0000 -1.1551 -0.7341  0.6344 14.2500 

Coefficients:
            Estimate Std. Error t value Pr(>|t|)    
(Intercept)   2.0000     0.1716  11.656  < 2e-16 ***
low_inc      -0.7397     0.2057  -3.597 0.000359 ***
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 1.986 on 439 degrees of freedom
  (753 observations deleted due to missingness)
Multiple R-squared:  0.02862,   Adjusted R-squared:  0.02641 
F-statistic: 12.94 on 1 and 439 DF,  p-value: 0.0003591
###################################
##Cluster-robust standard errors by school
q2_pooled_low_inc_clustered = vcovCL(
  q2_model_pooled_by_low_inc,
  cluster = ~ school_num
)

coeftest(
  q2_model_pooled_by_low_inc,
  vcov = q2_pooled_low_inc_clustered #after clustering
)

t test of coefficients:

            Estimate Std. Error t value Pr(>|t|)  
(Intercept)  2.00005    0.88805  2.2522  0.02480 *
low_inc     -0.73968    0.38053 -1.9438  0.05255 .
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Reference group is students who are not low income. Results after clustering: Students who are not low income logged 2 sessions per week (p>0.05). Students who are low income logged 0.74 sessions less per week compared to students who are not low income (p<0.1).

Self QA - Passed

NAs are from Cumberland not having FRPL data

q2_qa_usage_by_low_inc = pooled_vt %>%
  group_by(low_inc) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_low_inc)
# A tibble: 3 × 2
  low_inc mean_num_sessions_week
    <dbl>                  <dbl>
1       0                   2.00
2       1                   1.26
3      NA                   1.58

Pooled - Regress sessions per week ~ At-Risk

# Full model
q2_model_pooled_by_at_risk = lm(
  num_sessions_week ~
    at_risk, 
  data = pooled_vt
)
summary(q2_model_pooled_by_at_risk) #before clustering by school

Call:
lm(formula = num_sessions_week ~ at_risk, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-1.9539 -1.4539 -0.7728  0.5272 17.9195 

Coefficients:
            Estimate Std. Error t value Pr(>|t|)    
(Intercept)   1.7728     0.1030  17.217   <2e-16 ***
at_risk       0.1811     0.1644   1.102    0.271    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 2.328 on 839 degrees of freedom
  (353 observations deleted due to missingness)
Multiple R-squared:  0.001444,  Adjusted R-squared:  0.000254 
F-statistic: 1.213 on 1 and 839 DF,  p-value: 0.271
###################################
##Cluster-robust standard errors by school
q2_pooled_at_risk_clustered = vcovCL(
  q2_model_pooled_by_at_risk,
  cluster = ~ school_num
)

coeftest(
  q2_model_pooled_by_at_risk,
  vcov = q2_pooled_at_risk_clustered #after clustering
)

t test of coefficients:

            Estimate Std. Error t value Pr(>|t|)  
(Intercept)  1.77278    0.71607  2.4757  0.01349 *
at_risk      0.18107    0.97133  0.1864  0.85217  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Reference group are student who are not academically at-risk. Results after clustering: students who are not at-risk logged 1.77 sessions/week (p<0.05); compared to at-risk students who logged 0.18 sessions more per week (not significant).

Self QA - Passed Note: Penn had no at-risk status data and was excluded from analysis.

q2_qa_usage_by_at_risk = pooled_vt %>%
  group_by(at_risk) %>%
  summarise(
    mean_num_sessions_week = mean(num_sessions_week, na.rm = TRUE),
    .groups = "drop"
  )
print(q2_qa_usage_by_at_risk)
# A tibble: 3 × 2
  at_risk mean_num_sessions_week
    <dbl>                  <dbl>
1       0                  1.77 
2       1                  1.95 
3      NA                  0.822

Pooled - F Test Usage ~ School

One-way ANOVA F-test.

q2_anova_school = aov(num_sessions_week ~ school, data = pooled_vt)
summary(q2_anova_school)
              Df Sum Sq Mean Sq F value Pr(>F)    
school         6   1175  195.78   56.65 <2e-16 ***
Residuals   1187   4102    3.46                   
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Difference between schools’ usage is significant.

#Signficant results, Tukey 
TukeyHSD(q2_anova_school)
  Tukey multiple comparisons of means
    95% family-wise confidence level

Fit: aov(formula = num_sessions_week ~ school, data = pooled_vt)

$school
                                                                     diff
Bell Avenue Elementary School-Aldan Elementary School          0.26068111
Colwyn Elementary School-Aldan Elementary School               0.40908734
Luther Nick Jeralds Middle-Aldan Elementary School             0.17299314
Meridian Middle-Aldan Elementary School                        3.41467334
Spring Lake Middle-Aldan Elementary School                     1.57740815
Walnut Street Elementary School-Aldan Elementary School       -0.31672582
Colwyn Elementary School-Bell Avenue Elementary School         0.14840623
Luther Nick Jeralds Middle-Bell Avenue Elementary School      -0.08768797
Meridian Middle-Bell Avenue Elementary School                  3.15399222
Spring Lake Middle-Bell Avenue Elementary School               1.31672704
Walnut Street Elementary School-Bell Avenue Elementary School -0.57740693
Luther Nick Jeralds Middle-Colwyn Elementary School           -0.23609420
Meridian Middle-Colwyn Elementary School                       3.00558600
Spring Lake Middle-Colwyn Elementary School                    1.16832081
Walnut Street Elementary School-Colwyn Elementary School      -0.72581316
Meridian Middle-Luther Nick Jeralds Middle                     3.24168019
Spring Lake Middle-Luther Nick Jeralds Middle                  1.40441501
Walnut Street Elementary School-Luther Nick Jeralds Middle    -0.48971896
Spring Lake Middle-Meridian Middle                            -1.83726518
Walnut Street Elementary School-Meridian Middle               -3.73139916
Walnut Street Elementary School-Spring Lake Middle            -1.89413397
                                                                     lwr
Bell Avenue Elementary School-Aldan Elementary School         -0.5246530
Colwyn Elementary School-Aldan Elementary School              -0.4736297
Luther Nick Jeralds Middle-Aldan Elementary School            -0.4839148
Meridian Middle-Aldan Elementary School                        2.5797048
Spring Lake Middle-Aldan Elementary School                     0.9154932
Walnut Street Elementary School-Aldan Elementary School       -1.1665708
Colwyn Elementary School-Bell Avenue Elementary School        -0.6802535
Luther Nick Jeralds Middle-Bell Avenue Elementary School      -0.6699385
Meridian Middle-Bell Avenue Elementary School                  2.3763934
Spring Lake Middle-Bell Avenue Elementary School               0.7288333
Walnut Street Elementary School-Bell Avenue Elementary School -1.3709584
Luther Nick Jeralds Middle-Colwyn Elementary School           -0.9442293
Meridian Middle-Colwyn Elementary School                       2.1297437
Spring Lake Middle-Colwyn Elementary School                    0.4555384
Walnut Street Elementary School-Colwyn Elementary School      -1.6158489
Meridian Middle-Luther Nick Jeralds Middle                     2.5940395
Spring Lake Middle-Luther Nick Jeralds Middle                  1.0039185
Walnut Street Elementary School-Luther Nick Jeralds Middle    -1.1564291
Spring Lake Middle-Meridian Middle                            -2.4899839
Walnut Street Elementary School-Meridian Middle               -4.5741012
Walnut Street Elementary School-Spring Lake Middle            -2.5657780
                                                                     upr
Bell Avenue Elementary School-Aldan Elementary School          1.0460152
Colwyn Elementary School-Aldan Elementary School               1.2918044
Luther Nick Jeralds Middle-Aldan Elementary School             0.8299011
Meridian Middle-Aldan Elementary School                        4.2496419
Spring Lake Middle-Aldan Elementary School                     2.2393231
Walnut Street Elementary School-Aldan Elementary School        0.5331191
Colwyn Elementary School-Bell Avenue Elementary School         0.9770659
Luther Nick Jeralds Middle-Bell Avenue Elementary School       0.4945625
Meridian Middle-Bell Avenue Elementary School                  3.9315911
Spring Lake Middle-Bell Avenue Elementary School               1.9046207
Walnut Street Elementary School-Bell Avenue Elementary School  0.2161446
Luther Nick Jeralds Middle-Colwyn Elementary School            0.4720409
Meridian Middle-Colwyn Elementary School                       3.8814283
Spring Lake Middle-Colwyn Elementary School                    1.8811032
Walnut Street Elementary School-Colwyn Elementary School       0.1642226
Meridian Middle-Luther Nick Jeralds Middle                     3.8893208
Spring Lake Middle-Luther Nick Jeralds Middle                  1.8049115
Walnut Street Elementary School-Luther Nick Jeralds Middle     0.1769912
Spring Lake Middle-Meridian Middle                            -1.1845464
Walnut Street Elementary School-Meridian Middle               -2.8886971
Walnut Street Elementary School-Spring Lake Middle            -1.2224899
                                                                  p adj
Bell Avenue Elementary School-Aldan Elementary School         0.9582422
Colwyn Elementary School-Aldan Elementary School              0.8184563
Luther Nick Jeralds Middle-Aldan Elementary School            0.9870536
Meridian Middle-Aldan Elementary School                       0.0000000
Spring Lake Middle-Aldan Elementary School                    0.0000000
Walnut Street Elementary School-Aldan Elementary School       0.9279905
Colwyn Elementary School-Bell Avenue Elementary School        0.9984299
Luther Nick Jeralds Middle-Bell Avenue Elementary School      0.9994151
Meridian Middle-Bell Avenue Elementary School                 0.0000000
Spring Lake Middle-Bell Avenue Elementary School              0.0000000
Walnut Street Elementary School-Bell Avenue Elementary School 0.3246248
Luther Nick Jeralds Middle-Colwyn Elementary School           0.9573429
Meridian Middle-Colwyn Elementary School                      0.0000000
Spring Lake Middle-Colwyn Elementary School                   0.0000301
Walnut Street Elementary School-Colwyn Elementary School      0.1957775
Meridian Middle-Luther Nick Jeralds Middle                    0.0000000
Spring Lake Middle-Luther Nick Jeralds Middle                 0.0000000
Walnut Street Elementary School-Luther Nick Jeralds Middle    0.3130775
Spring Lake Middle-Meridian Middle                            0.0000000
Walnut Street Elementary School-Meridian Middle               0.0000000
Walnut Street Elementary School-Spring Lake Middle            0.0000000

Aligns with results shown in q2_model_pooled_by_school.

Pooled - F Test Usage ~ District

One-way ANOVA F-test.

q2_anova_district = aov(num_sessions_week ~ district, data = pooled_vt)
summary(q2_anova_district)
              Df Sum Sq Mean Sq F value Pr(>F)    
district       2    779   389.6   103.2 <2e-16 ***
Residuals   1191   4498     3.8                   
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Difference between districts’ usage is significant.

#Signficant results, Tukey 
TukeyHSD(q2_anova_district)
  Tukey multiple comparisons of means
    95% family-wise confidence level

Fit: aov(formula = num_sessions_week ~ district, data = pooled_vt)

$district
                      diff       lwr        upr p adj
Kent-Cumberland  2.5683816  2.054648  3.0821157     0
Penn-Cumberland -0.7526599 -1.046812 -0.4585079     0
Penn-Kent       -3.3210416 -3.864379 -2.7777039     0

Aligns with results shown in q2_model_pooled_by_district.

Pooled - T Test Usage ~ Male

Welch’s T-test.

t.test(num_sessions_week ~ male, data = pooled_vt)

    Welch Two Sample t-test

data:  num_sessions_week by male
t = -0.4061, df = 1188.4, p-value = 0.6847
alternative hypothesis: true difference in means between group 0 and group 1 is not equal to 0
95 percent confidence interval:
 -0.2883592  0.1894579
sample estimates:
mean in group 0 mean in group 1 
       1.518044        1.567495 

Difference between the genders’ usage is not significant.

Pooled - F Test Usage ~ Race

One-way ANOVA F-test.

q2_anova_race = aov(num_sessions_week ~ race_pooled, data = pooled_vt)
summary(q2_anova_race)
              Df Sum Sq Mean Sq F value   Pr(>F)    
race_pooled    3    177   59.07   13.79 7.68e-09 ***
Residuals   1190   5099    4.29                     
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Difference between race subgroups’ usage is significant.

#Signficant results, Tukey 
TukeyHSD(q2_anova_race)
  Tukey multiple comparisons of means
    95% family-wise confidence level

Fit: aov(formula = num_sessions_week ~ race_pooled, data = pooled_vt)

$race_pooled
                            diff        lwr       upr     p adj
Black-White          -0.39831482 -0.9867946 0.1901650 0.3027168
Hispanic-White        0.48889622 -0.2050158 1.1828083 0.2678254
Other Races-White     0.54495793 -0.1994167 1.2893325 0.2356754
Hispanic-Black        0.88721104  0.4350306 1.3393915 0.0000031
Other Races-Black     0.94327275  0.4169204 1.4696251 0.0000264
Other Races-Hispanic  0.05606171 -0.5860070 0.6981304 0.9960027

Align with results shown in q2_model_pooled_by_race.

Pooled - T Test Usage ~ IEP

Welch’s T-test.

t.test(num_sessions_week ~ iep, data = pooled_vt)

    Welch Two Sample t-test

data:  num_sessions_week by iep
t = 1.695, df = 177.22, p-value = 0.09183
alternative hypothesis: true difference in means between group 0 and group 1 is not equal to 0
95 percent confidence interval:
 -0.05432281  0.71569669
sample estimates:
mean in group 0 mean in group 1 
       1.581187        1.250500 

Difference between the groups is not significant.

Pooled - T Test Usage ~ ELL

Welch’s T-test.

t.test(num_sessions_week ~ ell, data = pooled_vt)

    Welch Two Sample t-test

data:  num_sessions_week by ell
t = -4.9749, df = 74.139, p-value = 4.103e-06
alternative hypothesis: true difference in means between group 0 and group 1 is not equal to 0
95 percent confidence interval:
 -2.464042 -1.054747
sample estimates:
mean in group 0 mean in group 1 
       1.437238        3.196633 

Difference between the groups is significant.

Pooled - T Test Usage ~ Low-income

Welch’s T-test.

t.test(num_sessions_week ~ low_inc, data = pooled_vt)

    Welch Two Sample t-test

data:  num_sessions_week by low_inc
t = 3.0915, df = 186.92, p-value = 0.002296
alternative hypothesis: true difference in means between group 0 and group 1 is not equal to 0
95 percent confidence interval:
 0.2676821 1.2116721
sample estimates:
mean in group 0 mean in group 1 
       2.000049        1.260372 

Difference between the groups is significant.

Pooled - T Test Usage ~ At-risk

Welch’s T-test.

t.test(num_sessions_week ~ at_risk, data = pooled_vt)

    Welch Two Sample t-test

data:  num_sessions_week by at_risk
t = -1.1113, df = 722.66, p-value = 0.2668
alternative hypothesis: true difference in means between group 0 and group 1 is not equal to 0
95 percent confidence interval:
 -0.5009584  0.1388212
sample estimates:
mean in group 0 mean in group 1 
       1.772783        1.953852 

Difference between the groups is not significant.

Research Question 3

How is the usability of Live + AI perceived by teachers? How do teachers perceive its impact on student confidence and learning?

Analyze teacher survey outside of R.

Research Question 4

What is the effect of offering LIVE+ AI on student learning compared to students who are not offered the tool? Do improvements in student learning vary meaningfully by student characteristics?

Conduct matched comparisons in e2i Coach.

Research Question 5

Among students who are offered LIVE+ AI, do students who use the tool more show greater gains in learning? PAP: Examine the relationships between student usage and learning outcomes using correlational analyses…Regression models adjusting for school clustering will explore associations, with subgroup analyses conducted as appropriate.

*Repeat for each subject.

POOLED - Regress delta MATH ~ Number of sessions

Change in math achievement

Calculate the change in Math scores between MOY and EOY.

pooled_vt$delta_math_z = pooled_vt$eoy_math_z - pooled_vt$moy_math_z

Descriptive statistics

Number of sessions (total). Among students who were offered tutoring, recoded NA to 0.

summary(pooled_vt$total_sessions)
   Min. 1st Qu.  Median    Mean 3rd Qu.    Max. 
   0.00    4.00   10.50   19.09   24.00  256.00 

MOY Math Z score.

summary(pooled_vt$moy_math_z)
   Min. 1st Qu.  Median    Mean 3rd Qu.    Max.    NA's 
-2.9800 -0.8400 -0.2300 -0.2049  0.3800  4.7800      67 

EOY Math Z Score.

summary(pooled_vt$eoy_math_z)
   Min. 1st Qu.  Median    Mean 3rd Qu.    Max.    NA's 
-3.2100 -0.7800 -0.1400 -0.1227  0.5500  4.3700      50 

Delta Math Z Score.

summary(pooled_vt$delta_math_z)
    Min.  1st Qu.   Median     Mean  3rd Qu.     Max.     NA's 
-4.99000 -0.25000  0.10000  0.09393  0.45000  3.74000      104 

—- MAIN —-

Regression model - Overall

# Full model
q5_model_pooled_overall = lm(
  delta_math_z ~
    total_sessions, 
  data = pooled_vt
)
summary(q5_model_pooled_overall) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-5.0848 -0.3395  0.0114  0.3642  3.6410 

Coefficients:
                 Estimate Std. Error t value Pr(>|t|)    
(Intercept)     0.1003572  0.0271974    3.69 0.000235 ***
total_sessions -0.0003294  0.0008244   -0.40 0.689585    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.7238 on 1088 degrees of freedom
  (104 observations deleted due to missingness)
Multiple R-squared:  0.0001467, Adjusted R-squared:  -0.0007723 
F-statistic: 0.1596 on 1 and 1088 DF,  p-value: 0.6896
###################################
##Cluster-robust standard errors by school
q5_pooled_overall_clustered = vcovCL(
  q5_model_pooled_overall,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_overall,
  vcov = q5_pooled_overall_clustered #after clustering
)

t test of coefficients:

                  Estimate  Std. Error t value  Pr(>|t|)    
(Intercept)     0.10035715  0.02990995  3.3553 0.0008201 ***
total_sessions -0.00032938  0.00043148 -0.7634 0.4454003    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_overall, type = "text", out = "q5_math_model_overall.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                           delta_math_z        
-----------------------------------------------
total_sessions                -0.0003          
                              (0.001)          
                                               
Constant                     0.100***          
                              (0.027)          
                                               
-----------------------------------------------
Observations                   1,090           
R2                            0.0001           
Adjusted R2                   -0.001           
Residual Std. Error      0.724 (df = 1088)     
F Statistic            0.160 (df = 1; 1088)    
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

Results after clustering are not significant. Sessions logged between MOY and EOY did not seem to change math z scores significantly.

Regression model - District

# Full model
q5_model_pooled_by_district = lm(
  delta_math_z ~
    total_sessions * district, 
  data = pooled_vt
)
summary(q5_model_pooled_by_district) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions * district, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-5.0639 -0.3508  0.0080  0.3732  3.6653 

Coefficients:
                              Estimate Std. Error t value Pr(>|t|)  
(Intercept)                  7.495e-02  3.408e-02   2.199   0.0281 *
total_sessions              -6.109e-05  9.766e-04  -0.063   0.9501  
districtKent                 2.743e-01  1.605e-01   1.709   0.0877 .
districtPenn                 4.919e-02  5.870e-02   0.838   0.4022  
total_sessions:districtKent -2.628e-03  4.090e-03  -0.642   0.5207  
total_sessions:districtPenn -1.903e-03  2.020e-03  -0.942   0.3463  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.723 on 1084 degrees of freedom
  (104 observations deleted due to missingness)
Multiple R-squared:  0.005932,  Adjusted R-squared:  0.001347 
F-statistic: 1.294 on 5 and 1084 DF,  p-value: 0.2641
###################################
##Cluster-robust standard errors by school
q5_pooled_by_district_clustered = vcovCL(
  q5_model_pooled_by_district,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_district,
  vcov = q5_pooled_by_district_clustered #after clustering
)

t test of coefficients:

                               Estimate  Std. Error  t value  Pr(>|t|)    
(Intercept)                  7.4949e-02  1.8189e-02   4.1206 4.067e-05 ***
total_sessions              -6.1091e-05  5.7001e-05  -1.0717   0.28407    
districtKent                 2.7434e-01  1.8189e-02  15.0827 < 2.2e-16 ***
districtPenn                 4.9191e-02  6.0071e-02   0.8189   0.41303    
total_sessions:districtKent -2.6275e-03  5.7001e-05 -46.0965 < 2.2e-16 ***
total_sessions:districtPenn -1.9034e-03  7.3646e-04  -2.5846   0.00988 ** 
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Reference group is Cumberland. Results after clustering: Cumberland did not see significant changes in math score.

For Kent students, for each additional session, their change in math z scores will yield 0.003 points lower compared to Cumberland student logging sessions at the same rate, p<0.001 (2.7434e-01 represents the delta math score for Kent students with no sessions logged, p<0.001)

For Penn students, each additional session, their change in math z scores will yield 0.002 points lower compared to Cumberland students logging sessions at the same rate, p<0.01 (4.9191e-02 represent the delta math score for Penn students with no sessions logged, p>0.4).

—- SUBGROUPS —-

Regression model - Male

# Full model
q5_model_pooled_by_male = lm(
  delta_math_z ~
    total_sessions * male, 
  data = pooled_vt
)
summary(q5_model_pooled_by_male) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions * male, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-5.1163 -0.3384  0.0119  0.3650  3.6621 

Coefficients:
                     Estimate Std. Error t value Pr(>|t|)  
(Intercept)          0.082046   0.037285   2.201    0.028 *
total_sessions      -0.001026   0.001111  -0.924    0.356  
male                 0.035753   0.054516   0.656    0.512  
total_sessions:male  0.001523   0.001656   0.920    0.358  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.7234 on 1086 degrees of freedom
  (104 observations deleted due to missingness)
Multiple R-squared:  0.002974,  Adjusted R-squared:  0.0002201 
F-statistic:  1.08 on 3 and 1086 DF,  p-value: 0.3566
###################################
##Cluster-robust standard errors by school
q5_pooled_by_male_clustered = vcovCL(
  q5_model_pooled_by_male,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_male,
  vcov = q5_pooled_by_male_clustered #after clustering
)

t test of coefficients:

                       Estimate  Std. Error t value  Pr(>|t|)    
(Intercept)          0.08204650  0.02078827  3.9468 8.431e-05 ***
total_sessions      -0.00102622  0.00045808 -2.2403   0.02527 *  
male                 0.03575348  0.03644306  0.9811   0.32677    
total_sessions:male  0.00152331  0.00074228  2.0522   0.04039 *  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_male, type = "text", out = "q5_math_model_male.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                           delta_math_z        
-----------------------------------------------
total_sessions                -0.001           
                              (0.001)          
                                               
male                           0.036           
                              (0.055)          
                                               
total_sessions:male            0.002           
                              (0.002)          
                                               
Constant                      0.082**          
                              (0.037)          
                                               
-----------------------------------------------
Observations                   1,090           
R2                             0.003           
Adjusted R2                   0.0002           
Residual Std. Error      0.723 (df = 1086)     
F Statistic            1.080 (df = 3; 1086)    
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

Reference group is female. Results after clustering:

Females with zero sessions logged observed 0.082 point change in math z score, p<0.001. For female students, each additional session yielded -0.001 point change in math z scores, p<0.05. Males with zero sessions logged observed 0.04 points more in delta math z score when compared to females that no sessions logged (not sig).For male students, each additional session yielded an additional 0.001 point change in math z score, p<0.05.

Regression model - Race

# Full model
q5_model_pooled_by_race = lm(
  delta_math_z ~
    total_sessions * race_pooled, 
  data = pooled_vt
)
summary(q5_model_pooled_by_race) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions * race_pooled, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-5.0716 -0.3323  0.0087  0.3659  3.6458 

Coefficients:
                                        Estimate Std. Error t value Pr(>|t|)  
(Intercept)                            0.1871186  0.1127552   1.660   0.0973 .
total_sessions                        -0.0016315  0.0039479  -0.413   0.6795  
race_pooledBlack                      -0.0890280  0.1171837  -0.760   0.4476  
race_pooledHispanic                   -0.0767686  0.1379828  -0.556   0.5781  
race_pooledOther Races                -0.1386245  0.1446181  -0.959   0.3380  
total_sessions:race_pooledBlack        0.0006625  0.0040736   0.163   0.8708  
total_sessions:race_pooledHispanic     0.0016806  0.0045170   0.372   0.7099  
total_sessions:race_pooledOther Races  0.0042348  0.0045768   0.925   0.3550  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.7247 on 1082 degrees of freedom
  (104 observations deleted due to missingness)
Multiple R-squared:  0.003076,  Adjusted R-squared:  -0.003374 
F-statistic: 0.4769 on 7 and 1082 DF,  p-value: 0.8518
###################################
##Cluster-robust standard errors by school
q5_pooled_by_race_clustered = vcovCL(
  q5_model_pooled_by_race,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_race,
  vcov = q5_pooled_by_race_clustered #after clustering
)

t test of coefficients:

                                         Estimate  Std. Error t value Pr(>|t|)
(Intercept)                            0.18711859  0.07587595  2.4661  0.01381
total_sessions                        -0.00163145  0.00199286 -0.8186  0.41317
race_pooledBlack                      -0.08902804  0.07143075 -1.2464  0.21290
race_pooledHispanic                   -0.07676862  0.06705091 -1.1449  0.25249
race_pooledOther Races                -0.13862447  0.19634588 -0.7060  0.48033
total_sessions:race_pooledBlack        0.00066253  0.00192201  0.3447  0.73038
total_sessions:race_pooledHispanic     0.00168058  0.00155228  1.0827  0.27920
total_sessions:race_pooledOther Races  0.00423484  0.00427086  0.9916  0.32163
                                       
(Intercept)                           *
total_sessions                         
race_pooledBlack                       
race_pooledHispanic                    
race_pooledOther Races                 
total_sessions:race_pooledBlack        
total_sessions:race_pooledHispanic     
total_sessions:race_pooledOther Races  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_race, type = "text", out = "q5_math_model_race.txt")

=================================================================
                                          Dependent variable:    
                                      ---------------------------
                                             delta_math_z        
-----------------------------------------------------------------
total_sessions                                  -0.002           
                                                (0.004)          
                                                                 
race_pooledBlack                                -0.089           
                                                (0.117)          
                                                                 
race_pooledHispanic                             -0.077           
                                                (0.138)          
                                                                 
race_pooledOther Races                          -0.139           
                                                (0.145)          
                                                                 
total_sessions:race_pooledBlack                  0.001           
                                                (0.004)          
                                                                 
total_sessions:race_pooledHispanic               0.002           
                                                (0.005)          
                                                                 
total_sessions:race_pooledOther Races            0.004           
                                                (0.005)          
                                                                 
Constant                                        0.187*           
                                                (0.113)          
                                                                 
-----------------------------------------------------------------
Observations                                     1,090           
R2                                               0.003           
Adjusted R2                                     -0.003           
Residual Std. Error                        0.725 (df = 1082)     
F Statistic                              0.477 (df = 7; 1082)    
=================================================================
Note:                                 *p<0.1; **p<0.05; ***p<0.01

White is the reference group. Results after clustering:

Main race groups did not observe significant changes to math z scores from MOY to EOY.

Regarding Pacific Islanders (PIs): The average white students with no sessions logged saw 0.19 point increase in math z score, p<0.05. For white students, each additional session yielded -0.002 point decrease in math z score (not sig).

For PI studnets with no sessions logged, change in math z scores were 0.2 points less than whites with no sessions logged. decrease in math z scores (not sig). For PI students, each additional session yielded an additional 0.004 point change in math z score when compared to white students logging the same number of sessions, p<0.05.

Regression model - IEP

# Full model
q5_model_pooled_by_iep = lm(
  delta_math_z ~
    total_sessions * iep, 
  data = pooled_vt
)
summary(q5_model_pooled_by_iep) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions * iep, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-5.0872 -0.3434  0.0064  0.3615  3.6370 

Coefficients:
                     Estimate Std. Error t value Pr(>|t|)    
(Intercept)         0.1048386  0.0294227   3.563 0.000382 ***
total_sessions     -0.0004479  0.0009024  -0.496 0.619730    
iep                -0.0302648  0.0783942  -0.386 0.699528    
total_sessions:iep  0.0006983  0.0022312   0.313 0.754369    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.7244 on 1086 degrees of freedom
  (104 observations deleted due to missingness)
Multiple R-squared:  0.0002995, Adjusted R-squared:  -0.002462 
F-statistic: 0.1085 on 3 and 1086 DF,  p-value: 0.9552
###################################
##Cluster-robust standard errors by school
q5_pooled_by_iep_clustered = vcovCL(
  q5_model_pooled_by_iep,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_iep,
  vcov = q5_pooled_by_iep_clustered #after clustering
)

t test of coefficients:

                      Estimate  Std. Error t value Pr(>|t|)    
(Intercept)         0.10483863  0.02629755  3.9866 7.15e-05 ***
total_sessions     -0.00044793  0.00034447 -1.3004   0.1938    
iep                -0.03026483  0.07185257 -0.4212   0.6737    
total_sessions:iep  0.00069829  0.00129949  0.5374   0.5911    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_iep, type = "text", out = "q5_math_model_iep.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                           delta_math_z        
-----------------------------------------------
total_sessions                -0.0004          
                              (0.001)          
                                               
iep                           -0.030           
                              (0.078)          
                                               
total_sessions:iep             0.001           
                              (0.002)          
                                               
Constant                     0.105***          
                              (0.029)          
                                               
-----------------------------------------------
Observations                   1,090           
R2                            0.0003           
Adjusted R2                   -0.002           
Residual Std. Error      0.724 (df = 1086)     
F Statistic            0.108 (df = 3; 1086)    
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

Reference group is no-IEP. Results after clustering:

No observed significant changes to math z scores from MOY to EOY for IEP or nonIEP students using Live+AI.

Regression model - ELL

# Full model
q5_model_pooled_by_ell = lm(
  delta_math_z ~
    total_sessions * ell, 
  data = pooled_vt
)
summary(q5_model_pooled_by_ell) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions * ell, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-5.0864 -0.3421  0.0111  0.3664  3.6357 

Coefficients:
                     Estimate Std. Error t value Pr(>|t|)    
(Intercept)         0.1067335  0.0278783   3.829 0.000136 ***
total_sessions     -0.0006051  0.0008745  -0.692 0.489104    
ell                -0.1231560  0.1298468  -0.948 0.343100    
total_sessions:ell  0.0031921  0.0028485   1.121 0.262697    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.724 on 1086 degrees of freedom
  (104 observations deleted due to missingness)
Multiple R-squared:  0.001348,  Adjusted R-squared:  -0.001411 
F-statistic: 0.4885 on 3 and 1086 DF,  p-value: 0.6903
###################################
##Cluster-robust standard errors by school
q5_pooled_by_ell_clustered = vcovCL(
  q5_model_pooled_by_ell,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_ell,
  vcov = q5_pooled_by_ell_clustered #after clustering
)

t test of coefficients:

                      Estimate  Std. Error t value  Pr(>|t|)    
(Intercept)         0.10673348  0.02538405  4.2047 2.829e-05 ***
total_sessions     -0.00060515  0.00041772 -1.4487    0.1477    
ell                -0.12315598  0.21497960 -0.5729    0.5668    
total_sessions:ell  0.00319210  0.00274888  1.1612    0.2458    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_ell, type = "text", out = "q5_math_model_ell.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                           delta_math_z        
-----------------------------------------------
total_sessions                -0.001           
                              (0.001)          
                                               
ell                           -0.123           
                              (0.130)          
                                               
total_sessions:ell             0.003           
                              (0.003)          
                                               
Constant                     0.107***          
                              (0.028)          
                                               
-----------------------------------------------
Observations                   1,090           
R2                             0.001           
Adjusted R2                   -0.001           
Residual Std. Error      0.724 (df = 1086)     
F Statistic            0.488 (df = 3; 1086)    
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

Reference group is no-ELL. Results after clustering:

No observed significant changes to math z scores from MOY to EOY for ELL or nonELL students using Live+AI.

Regression model - Low-Income

# Full model
q5_model_pooled_by_low_inc = lm(
  delta_math_z ~
    total_sessions * low_inc, 
  data = pooled_vt
)
summary(q5_model_pooled_by_low_inc) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions * low_inc, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-1.4652 -0.3397 -0.0321  0.2896  3.1671 

Coefficients:
                         Estimate Std. Error t value Pr(>|t|)    
(Intercept)             0.1963326  0.0580308   3.383 0.000782 ***
total_sessions         -0.0013695  0.0014904  -0.919 0.358666    
low_inc                -0.0728116  0.0710396  -1.025 0.305970    
total_sessions:low_inc  0.0005118  0.0022121   0.231 0.817142    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.515 on 427 degrees of freedom
  (763 observations deleted due to missingness)
Multiple R-squared:  0.004977,  Adjusted R-squared:  -0.002014 
F-statistic: 0.7119 on 3 and 427 DF,  p-value: 0.5453
###################################
##Cluster-robust standard errors by school
q5_pooled_by_low_inc_clustered = vcovCL(
  q5_model_pooled_by_low_inc,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_low_inc,
  vcov = q5_pooled_by_low_inc_clustered #after clustering
)

t test of coefficients:

                          Estimate  Std. Error t value  Pr(>|t|)    
(Intercept)             0.19633255  0.06084849  3.2266  0.001349 ** 
total_sessions         -0.00136950  0.00069307 -1.9760  0.048799 *  
low_inc                -0.07281162  0.01510821 -4.8193 2.005e-06 ***
total_sessions:low_inc  0.00051180  0.00171584  0.2983  0.765635    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_low_inc, type = "text", out = "q5_math_model_frpl.txt")

==================================================
                           Dependent variable:    
                       ---------------------------
                              delta_math_z        
--------------------------------------------------
total_sessions                   -0.001           
                                 (0.001)          
                                                  
low_inc                          -0.073           
                                 (0.071)          
                                                  
total_sessions:low_inc            0.001           
                                 (0.002)          
                                                  
Constant                        0.196***          
                                 (0.058)          
                                                  
--------------------------------------------------
Observations                       431            
R2                                0.005           
Adjusted R2                      -0.002           
Residual Std. Error         0.515 (df = 427)      
F Statistic                0.712 (df = 3; 427)    
==================================================
Note:                  *p<0.1; **p<0.05; ***p<0.01

Reference group are students who are not low-income.

For non LC students with zero sessions, they observed a 0.2 point change in math z score, p<0.01. For non LC students, each additional session logged yielded a -0.001 point decrease in math z scores , p<0.05. For LC students with zero sessions, their change in math z scores was 0.07 less than non LC students with zero sessions, p<0.001. No significant changes in math z scores for low-income students using Live+AI.

Regression model - At-Risk

# Full model
q5_model_pooled_by_at_risk = lm(
  delta_math_z ~
    total_sessions * at_risk, 
  data = pooled_vt
)
summary(q5_model_pooled_by_at_risk) #before clustering by school

Call:
lm(formula = delta_math_z ~ total_sessions * at_risk, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-5.0249 -0.3360  0.0261  0.3792  3.6058 

Coefficients:
                         Estimate Std. Error t value Pr(>|t|)   
(Intercept)             0.1357022  0.0462215   2.936  0.00343 **
total_sessions         -0.0003874  0.0012116  -0.320  0.74928   
at_risk                -0.1276652  0.0801370  -1.593  0.11157   
total_sessions:at_risk  0.0019666  0.0024655   0.798  0.42534   
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.8148 on 738 degrees of freedom
  (452 observations deleted due to missingness)
Multiple R-squared:  0.003554,  Adjusted R-squared:  -0.0004962 
F-statistic: 0.8775 on 3 and 738 DF,  p-value: 0.4523
###################################
##Cluster-robust standard errors by school
q5_pooled_by_at_risk_clustered = vcovCL(
  q5_model_pooled_by_at_risk,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_at_risk,
  vcov = q5_pooled_by_at_risk_clustered #after clustering
)

t test of coefficients:

                          Estimate  Std. Error  t value  Pr(>|t|)    
(Intercept)             1.3570e-01  1.8517e-02   7.3285 6.122e-13 ***
total_sessions         -3.8736e-04  9.5267e-06 -40.6607 < 2.2e-16 ***
at_risk                -1.2767e-01  7.8747e-02  -1.6212    0.1054    
total_sessions:at_risk  1.9666e-03  1.9838e-03   0.9913    0.3219    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_at_risk, type = "text", out = "q5_math_model_at-risk.txt")

==================================================
                           Dependent variable:    
                       ---------------------------
                              delta_math_z        
--------------------------------------------------
total_sessions                   -0.0004          
                                 (0.001)          
                                                  
at_risk                          -0.128           
                                 (0.080)          
                                                  
total_sessions:at_risk            0.002           
                                 (0.002)          
                                                  
Constant                        0.136***          
                                 (0.046)          
                                                  
--------------------------------------------------
Observations                       742            
R2                                0.004           
Adjusted R2                      -0.0005          
Residual Std. Error         0.815 (df = 738)      
F Statistic                0.877 (df = 3; 738)    
==================================================
Note:                  *p<0.1; **p<0.05; ***p<0.01

Reference group is not at-risk. Results after clustering:

No observed significant changes to math z scores from MOY to EOY for at-risk students using Live+AI.

For non at-risk students, each additional session yielded -0.0003 point change in delta math z score, p<0.001.

POOLED - Regress delta ELA ~ Sessions per week

Change in ELA achievement

Calculate the change in ELA scores between MOY and EOY.

pooled_vt$delta_ela_z = pooled_vt$eoy_ela_z - pooled_vt$moy_ela_z

Descriptive statistics

Number of sessions (total). Among students who were offered tutoring, recoded NA to 0.

summary(pooled_vt$total_sessions)
   Min. 1st Qu.  Median    Mean 3rd Qu.    Max. 
   0.00    4.00   10.50   19.09   24.00  256.00 

MOY ELA Z score.

summary(pooled_vt$moy_ela_z)
   Min. 1st Qu.  Median    Mean 3rd Qu.    Max.    NA's 
-2.8100 -0.6400 -0.1000 -0.1518  0.4700  2.2900     105 

EOY ELA Z Score.

summary(pooled_vt$eoy_ela_z)
   Min. 1st Qu.  Median    Mean 3rd Qu.    Max.    NA's 
-3.6700 -0.7525 -0.1100 -0.1661  0.4800  2.2700      82 

Delta ELA Z Score.

summary(pooled_vt$delta_ela_z)
   Min. 1st Qu.  Median    Mean 3rd Qu.    Max.    NA's 
-4.3900 -0.3500 -0.0150 -0.0189  0.3000  2.9500     164 

—- MAIN —-

Regression model - Overall

# Full model
q5_model_pooled_overall = lm(
  delta_ela_z ~
    total_sessions, 
  data = pooled_vt
)
summary(q5_model_pooled_overall) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-4.3292 -0.3307  0.0034  0.3290  2.9063 

Coefficients:
                 Estimate Std. Error t value Pr(>|t|)   
(Intercept)    -0.0660193  0.0256010  -2.579   0.0101 * 
total_sessions  0.0026135  0.0008817   2.964   0.0031 **
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.6441 on 1028 degrees of freedom
  (164 observations deleted due to missingness)
Multiple R-squared:  0.008475,  Adjusted R-squared:  0.007511 
F-statistic: 8.787 on 1 and 1028 DF,  p-value: 0.003104
###################################
##Cluster-robust standard errors by school
q5_pooled_overall_clustered = vcovCL(
  q5_model_pooled_overall,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_overall,
  vcov = q5_pooled_overall_clustered #after clustering
)

t test of coefficients:

                 Estimate Std. Error t value Pr(>|t|)  
(Intercept)    -0.0660193  0.0286532 -2.3041  0.02142 *
total_sessions  0.0026135  0.0014400  1.8149  0.06983 .
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_overall, type = "text", out = "q5_ela_model_overall.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                            delta_ela_z        
-----------------------------------------------
total_sessions               0.003***          
                              (0.001)          
                                               
Constant                     -0.066**          
                              (0.026)          
                                               
-----------------------------------------------
Observations                   1,030           
R2                             0.008           
Adjusted R2                    0.008           
Residual Std. Error      0.644 (df = 1028)     
F Statistic           8.787*** (df = 1; 1028)  
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

For students with no sessions logged, they saw a -0.06 point change in delta ELA z scores, p<0.05.For each additional session logged, students on average observed a 0.002 point change increase in delta ELA z scores, p<0.1.

Regression model - District

# Full model
q5_model_pooled_by_district = lm(
  delta_ela_z ~
    total_sessions * district, 
  data = pooled_vt
)
summary(q5_model_pooled_by_district) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions * district, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-4.3431 -0.3252  0.0065  0.3337  2.6313 

Coefficients:
                              Estimate Std. Error t value Pr(>|t|)  
(Intercept)                 -4.960e-02  3.240e-02  -1.531   0.1260  
total_sessions               1.348e-03  1.138e-03   1.185   0.2362  
districtKent                 3.118e-01  1.367e-01   2.281   0.0227 *
districtPenn                -8.037e-02  5.301e-02  -1.516   0.1298  
total_sessions:districtKent  2.679e-03  3.605e-03   0.743   0.4576  
total_sessions:districtPenn -7.821e-05  1.923e-03  -0.041   0.9676  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.6334 on 1024 degrees of freedom
  (164 observations deleted due to missingness)
Multiple R-squared:  0.04489,   Adjusted R-squared:  0.04023 
F-statistic: 9.626 on 5 and 1024 DF,  p-value: 5.411e-09
###################################
##Cluster-robust standard errors by school
q5_pooled_by_district_clustered = vcovCL(
  q5_model_pooled_by_district,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_district,
  vcov = q5_pooled_by_district_clustered #after clustering
)

t test of coefficients:

                               Estimate  Std. Error t value  Pr(>|t|)    
(Intercept)                 -4.9604e-02  2.1692e-02 -2.2867   0.02241 *  
total_sessions               1.3484e-03  3.3203e-04  4.0611 5.256e-05 ***
districtKent                 3.1182e-01  2.1692e-02 14.3751 < 2.2e-16 ***
districtPenn                -8.0371e-02  3.9587e-02 -2.0302   0.04259 *  
total_sessions:districtKent  2.6787e-03  3.3203e-04  8.0676 1.994e-15 ***
total_sessions:districtPenn -7.8214e-05  9.4977e-04 -0.0824   0.93438    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Reference group is Cumberland. Results after clustering:

For Cumberland students, for each additional session, delta ELA z scores would increase by an additional 0.001 points, p<0.001.

For Kent students, for each additional session, delta ELA z scores increased by an additional 0.002 points more when compared to Cumberland students logging sessions at the same rate, p<0.001 (3.1182e-01 represents the delta ELA score for Kent students with no sessions logged, p<0.001).

For Penn students, each additional session, delta ELA z scores decreased by an additional 0.00007 points less when compared to Cumberland students, p>0.9 (-8.0371e-02 represent the delta ELA score for Penn students with no sessions logged, p<0.05).

—- SUBGROUPS —-

Regression model - Male

# Full model
q5_model_pooled_by_male = lm(
  delta_ela_z ~
    total_sessions * male, 
  data = pooled_vt
)
summary(q5_model_pooled_by_male) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions * male, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-4.3108 -0.3341  0.0018  0.3232  2.9199 

Coefficients:
                      Estimate Std. Error t value Pr(>|t|)  
(Intercept)         -0.0494346  0.0348837  -1.417   0.1567  
total_sessions       0.0025215  0.0011777   2.141   0.0325 *
male                -0.0352742  0.0514335  -0.686   0.4930  
total_sessions:male  0.0002123  0.0017779   0.119   0.9050  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.6445 on 1026 degrees of freedom
  (164 observations deleted due to missingness)
Multiple R-squared:  0.009079,  Adjusted R-squared:  0.006182 
F-statistic: 3.134 on 3 and 1026 DF,  p-value: 0.02484
###################################
##Cluster-robust standard errors by school
q5_pooled_by_male_clustered = vcovCL(
  q5_model_pooled_by_male,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_male,
  vcov = q5_pooled_by_male_clustered #after clustering
)

t test of coefficients:

                       Estimate  Std. Error t value Pr(>|t|)  
(Intercept)         -0.04943458  0.02304449 -2.1452  0.03217 *
total_sessions       0.00252148  0.00110103  2.2901  0.02222 *
male                -0.03527425  0.03344898 -1.0546  0.29187  
total_sessions:male  0.00021226  0.00232704  0.0912  0.92734  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_male, type = "text", out = "q5_ela_model_male.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                            delta_ela_z        
-----------------------------------------------
total_sessions                0.003**          
                              (0.001)          
                                               
male                          -0.035           
                              (0.051)          
                                               
total_sessions:male           0.0002           
                              (0.002)          
                                               
Constant                      -0.049           
                              (0.035)          
                                               
-----------------------------------------------
Observations                   1,030           
R2                             0.009           
Adjusted R2                    0.006           
Residual Std. Error      0.645 (df = 1026)     
F Statistic           3.134** (df = 3; 1026)   
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

Reference group is female. Results after clustering:

For female students, each additional session yielded -0.0025 point change in math z scores, p<0.05. For treatment female students with zero sessions, delta ELA z score equaled -0.04, p<0.05.

For male students, no significant changes to ELA z scores were observed with or without sessions logged, when compared to females.

Regression model - Race

# Full model
q5_model_pooled_by_race = lm(
  delta_ela_z ~
    total_sessions * race_pooled, 
  data = pooled_vt
)
summary(q5_model_pooled_by_race) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions * race_pooled, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-4.3010 -0.3379  0.0022  0.3312  2.9358 

Coefficients:
                                        Estimate Std. Error t value Pr(>|t|)
(Intercept)                           -6.516e-02  9.770e-02  -0.667    0.505
total_sessions                         2.654e-03  2.984e-03   0.889    0.374
race_pooledBlack                      -2.899e-02  1.023e-01  -0.283    0.777
race_pooledHispanic                    1.788e-01  1.232e-01   1.451    0.147
race_pooledOther Races                 9.529e-03  1.274e-01   0.075    0.940
total_sessions:race_pooledBlack       -7.509e-05  3.198e-03  -0.023    0.981
total_sessions:race_pooledHispanic    -2.413e-03  3.724e-03  -0.648    0.517
total_sessions:race_pooledOther Races  7.593e-04  3.753e-03   0.202    0.840

Residual standard error: 0.6434 on 1022 degrees of freedom
  (164 observations deleted due to missingness)
Multiple R-squared:  0.01625,   Adjusted R-squared:  0.009516 
F-statistic: 2.412 on 7 and 1022 DF,  p-value: 0.01881
###################################
##Cluster-robust standard errors by school
q5_pooled_by_race_clustered = vcovCL(
  q5_model_pooled_by_race,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_race,
  vcov = q5_pooled_by_race_clustered #after clustering
)

t test of coefficients:

                                         Estimate  Std. Error t value Pr(>|t|)
(Intercept)                           -6.5164e-02  7.7041e-02 -0.8458 0.397841
total_sessions                         2.6541e-03  2.9344e-03  0.9045 0.365940
race_pooledBlack                      -2.8986e-02  6.4294e-02 -0.4508 0.652207
race_pooledHispanic                    1.7875e-01  6.2828e-02  2.8451 0.004528
race_pooledOther Races                 9.5292e-03  7.7499e-02  0.1230 0.902164
total_sessions:race_pooledBlack       -7.5089e-05  3.2019e-03 -0.0235 0.981295
total_sessions:race_pooledHispanic    -2.4133e-03  2.1970e-03 -1.0984 0.272278
total_sessions:race_pooledOther Races  7.5931e-04  1.3743e-03  0.5525 0.580730
                                        
(Intercept)                             
total_sessions                          
race_pooledBlack                        
race_pooledHispanic                   **
race_pooledOther Races                  
total_sessions:race_pooledBlack         
total_sessions:race_pooledHispanic      
total_sessions:race_pooledOther Races   
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_race, type = "text", out = "q5_ela_model_race.txt")

=================================================================
                                          Dependent variable:    
                                      ---------------------------
                                              delta_ela_z        
-----------------------------------------------------------------
total_sessions                                   0.003           
                                                (0.003)          
                                                                 
race_pooledBlack                                -0.029           
                                                (0.102)          
                                                                 
race_pooledHispanic                              0.179           
                                                (0.123)          
                                                                 
race_pooledOther Races                           0.010           
                                                (0.127)          
                                                                 
total_sessions:race_pooledBlack                 -0.0001          
                                                (0.003)          
                                                                 
total_sessions:race_pooledHispanic              -0.002           
                                                (0.004)          
                                                                 
total_sessions:race_pooledOther Races            0.001           
                                                (0.004)          
                                                                 
Constant                                        -0.065           
                                                (0.098)          
                                                                 
-----------------------------------------------------------------
Observations                                     1,030           
R2                                               0.016           
Adjusted R2                                      0.010           
Residual Std. Error                        0.643 (df = 1022)     
F Statistic                             2.412** (df = 7; 1022)   
=================================================================
Note:                                 *p<0.1; **p<0.05; ***p<0.01

White is the reference group. Significant results after clustering:

Black students with no sessions logged observed a 0.03 point change less than White students with no sessions logged, p<0.1. Hispanic students with no sessions logged observed a 0.2 point change more than White student with no sessions logged, p<0.01. For Multi-racial students, for each additional session logged, ELA z scores saw -0.005 point change, p<0.01. For PI students, for each additional session logged, ELA z scores saw 0.01 point increase more than White students using at the same rate, p<0.05.

Regression model - IEP

# Full model
q5_model_pooled_by_iep = lm(
  delta_ela_z ~
    total_sessions * iep, 
  data = pooled_vt
)
summary(q5_model_pooled_by_iep) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions * iep, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-4.3282 -0.3306  0.0044  0.3295  2.9052 

Coefficients:
                     Estimate Std. Error t value Pr(>|t|)   
(Intercept)        -0.0670923  0.0275730  -2.433  0.01513 * 
total_sessions      0.0026630  0.0009461   2.815  0.00498 **
iep                 0.0076885  0.0747180   0.103  0.91806   
total_sessions:iep -0.0003789  0.0026302  -0.144  0.88548   
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.6447 on 1026 degrees of freedom
  (164 observations deleted due to missingness)
Multiple R-squared:  0.008496,  Adjusted R-squared:  0.005596 
F-statistic:  2.93 on 3 and 1026 DF,  p-value: 0.03269
###################################
##Cluster-robust standard errors by school
q5_pooled_by_iep_clustered = vcovCL(
  q5_model_pooled_by_iep,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_iep,
  vcov = q5_pooled_by_iep_clustered #after clustering
)

t test of coefficients:

                      Estimate  Std. Error t value Pr(>|t|)  
(Intercept)        -0.06709227  0.02976551 -2.2540   0.0244 *
total_sessions      0.00266301  0.00162511  1.6387   0.1016  
iep                 0.00768854  0.03291664  0.2336   0.8154  
total_sessions:iep -0.00037892  0.00171630 -0.2208   0.8253  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_iep, type = "text", out = "q5_ela_model_iep.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                            delta_ela_z        
-----------------------------------------------
total_sessions               0.003***          
                              (0.001)          
                                               
iep                            0.008           
                              (0.075)          
                                               
total_sessions:iep            -0.0004          
                              (0.003)          
                                               
Constant                     -0.067**          
                              (0.028)          
                                               
-----------------------------------------------
Observations                   1,030           
R2                             0.008           
Adjusted R2                    0.006           
Residual Std. Error      0.645 (df = 1026)     
F Statistic           2.930** (df = 3; 1026)   
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

Reference group is no-IEP. Results after clustering:

No observed significant changes to ELA z scores from MOY to EOY for IEP or nonIEP students using Live+AI.

Regression model - ELL

# Full model
q5_model_pooled_by_ell = lm(
  delta_ela_z ~
    total_sessions * ell, 
  data = pooled_vt
)
summary(q5_model_pooled_by_ell) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions * ell, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-4.3228 -0.3275  0.0039  0.3338  2.9063 

Coefficients:
                     Estimate Std. Error t value Pr(>|t|)   
(Intercept)        -0.0727908  0.0263345  -2.764  0.00581 **
total_sessions      0.0027743  0.0009447   2.937  0.00339 **
ell                 0.1354500  0.1184224   1.144  0.25298   
total_sessions:ell -0.0023535  0.0028404  -0.829  0.40753   
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.6443 on 1026 degrees of freedom
  (164 observations deleted due to missingness)
Multiple R-squared:  0.009739,  Adjusted R-squared:  0.006844 
F-statistic: 3.364 on 3 and 1026 DF,  p-value: 0.01818
###################################
##Cluster-robust standard errors by school
q5_pooled_by_ell_clustered = vcovCL(
  q5_model_pooled_by_ell,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_ell,
  vcov = q5_pooled_by_ell_clustered #after clustering
)

t test of coefficients:

                      Estimate  Std. Error t value Pr(>|t|)   
(Intercept)        -0.07279076  0.02519819 -2.8887 0.003949 **
total_sessions      0.00277429  0.00127184  2.1813 0.029386 * 
ell                 0.13544998  0.11680400  1.1596 0.246467   
total_sessions:ell -0.00235354  0.00077728 -3.0279 0.002524 **
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_ell, type = "text", out = "q5_ela_model_ell.txt")

===============================================
                        Dependent variable:    
                    ---------------------------
                            delta_ela_z        
-----------------------------------------------
total_sessions               0.003***          
                              (0.001)          
                                               
ell                            0.135           
                              (0.118)          
                                               
total_sessions:ell            -0.002           
                              (0.003)          
                                               
Constant                     -0.073***         
                              (0.026)          
                                               
-----------------------------------------------
Observations                   1,030           
R2                             0.010           
Adjusted R2                    0.007           
Residual Std. Error      0.644 (df = 1026)     
F Statistic           3.364** (df = 3; 1026)   
===============================================
Note:               *p<0.1; **p<0.05; ***p<0.01

Reference group are non ELLs.

For non ELLs with no sessions logged, they observed a -0.07 point change in ELA z score, p<0.01. For non ELLs, for each additional session, they observed 0.003 point change in ELA z scores, p<0.05. For ELLs with no sessions logged, ELA z scores did not change significantly compared to no ELLs with no sessions logged, p>0.2. For ELLs, for each additional session, they observed a 0.002 point change less compared to non ELLs using at the same rate, p<0.01.

Regression model - Low-Income

# Full model
q5_model_pooled_by_low_inc = lm(
  delta_ela_z ~
    total_sessions * low_inc, 
  data = pooled_vt
)
summary(q5_model_pooled_by_low_inc) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions * low_inc, data = pooled_vt)

Residuals:
     Min       1Q   Median       3Q      Max 
-1.67264 -0.28749  0.00877  0.29235  2.79314 

Coefficients:
                        Estimate Std. Error t value Pr(>|t|)  
(Intercept)            -0.020920   0.061328  -0.341   0.7332  
total_sessions          0.001840   0.001576   1.168   0.2436  
low_inc                -0.126439   0.074977  -1.686   0.0924 .
total_sessions:low_inc  0.005403   0.002336   2.313   0.0212 *
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.5435 on 430 degrees of freedom
  (760 observations deleted due to missingness)
Multiple R-squared:  0.04367,   Adjusted R-squared:  0.037 
F-statistic: 6.545 on 3 and 430 DF,  p-value: 0.0002459
###################################
##Cluster-robust standard errors by school
q5_pooled_by_low_inc_clustered = vcovCL(
  q5_model_pooled_by_low_inc,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_low_inc,
  vcov = q5_pooled_by_low_inc_clustered #after clustering
)

t test of coefficients:

                         Estimate Std. Error t value Pr(>|t|)   
(Intercept)            -0.0209202  0.1047096 -0.1998 0.841737   
total_sessions          0.0018404  0.0021022  0.8754 0.381829   
low_inc                -0.1264393  0.0980183 -1.2900 0.197759   
total_sessions:low_inc  0.0054029  0.0020389  2.6499 0.008348 **
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_low_inc, type = "text", out = "q5_ela_model_frpl.txt")

==================================================
                           Dependent variable:    
                       ---------------------------
                               delta_ela_z        
--------------------------------------------------
total_sessions                    0.002           
                                 (0.002)          
                                                  
low_inc                          -0.126*          
                                 (0.075)          
                                                  
total_sessions:low_inc           0.005**          
                                 (0.002)          
                                                  
Constant                         -0.021           
                                 (0.061)          
                                                  
--------------------------------------------------
Observations                       434            
R2                                0.044           
Adjusted R2                       0.037           
Residual Std. Error         0.543 (df = 430)      
F Statistic              6.545*** (df = 3; 430)   
==================================================
Note:                  *p<0.1; **p<0.05; ***p<0.01

Reference group are students who are not low-income.

Non LC students’ ELA z scores did not change significantly, regardless of sessions logged. For LC students, each additional session saw a 0.002 point change less compared non LC students using at the same rate, p<0.01.

Regression model - At-Risk

# Full model
q5_model_pooled_by_at_risk = lm(
  delta_ela_z ~
    total_sessions * at_risk, 
  data = pooled_vt
)
summary(q5_model_pooled_by_at_risk) #before clustering by school

Call:
lm(formula = delta_ela_z ~ total_sessions * at_risk, data = pooled_vt)

Residuals:
    Min      1Q  Median      3Q     Max 
-4.3154 -0.3362  0.0134  0.3538  2.7239 

Coefficients:
                        Estimate Std. Error t value Pr(>|t|)  
(Intercept)            -0.079278   0.043465  -1.824   0.0686 .
total_sessions          0.002335   0.001382   1.689   0.0916 .
at_risk                 0.116320   0.073798   1.576   0.1154  
total_sessions:at_risk  0.002167   0.002567   0.844   0.3989  
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1

Residual standard error: 0.7029 on 679 degrees of freedom
  (511 observations deleted due to missingness)
Multiple R-squared:  0.02216,   Adjusted R-squared:  0.01784 
F-statistic:  5.13 on 3 and 679 DF,  p-value: 0.00163
###################################
##Cluster-robust standard errors by school
q5_pooled_by_at_risk_clustered = vcovCL(
  q5_model_pooled_by_at_risk,
  cluster = ~ school_num
)

coeftest(
  q5_model_pooled_by_at_risk,
  vcov = q5_pooled_by_at_risk_clustered #after clustering
)

t test of coefficients:

                          Estimate  Std. Error t value  Pr(>|t|)    
(Intercept)            -0.07927836  0.01962216 -4.0402 5.950e-05 ***
total_sessions          0.00233470  0.00050267  4.6446 4.093e-06 ***
at_risk                 0.11631951  0.05171012  2.2495    0.0248 *  
total_sessions:at_risk  0.00216666  0.00403549  0.5369    0.5915    
---
Signif. codes:  0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
stargazer(q5_model_pooled_by_at_risk, type = "text", out = "q5_ela_model_at-risk.txt")

==================================================
                           Dependent variable:    
                       ---------------------------
                               delta_ela_z        
--------------------------------------------------
total_sessions                   0.002*           
                                 (0.001)          
                                                  
at_risk                           0.116           
                                 (0.074)          
                                                  
total_sessions:at_risk            0.002           
                                 (0.003)          
                                                  
Constant                         -0.079*          
                                 (0.043)          
                                                  
--------------------------------------------------
Observations                       683            
R2                                0.022           
Adjusted R2                       0.018           
Residual Std. Error         0.703 (df = 679)      
F Statistic              5.130*** (df = 3; 679)   
==================================================
Note:                  *p<0.1; **p<0.05; ***p<0.01

Reference group is not at-risk. Results after clustering:

For non at-risk students with no sessions, ELA z score changed by -0.08 points, p<0.001. For non at-risk students, each additional session yielded -0.002 point change in math z score, p<0.001. For at-risk student with no sessions, ELA z score changed by 0.11 points more than non at-risk students with no sessions logged, p<0.05. At-risk students did not see significant changes to ELA z scores when using Live+AI.

Request from QA

How many students have MOY or EOY z-score with absolute value greater than 3?

Performed calculations among students who were offered Live+AI (pooled_vt, where VT ID was present for a student; N=1194).

vars_z_outliers = c("moy_math_z", "moy_ela_z", "eoy_math_z", "eoy_ela_z")

summary_z_outliers = data.frame(
  variable = vars_z_outliers,
  count = sapply(vars_z_outliers, function(v) sum(pooled_vt[[v]] > 3, na.rm = TRUE))
)

summary_z_outliers
             variable count
moy_math_z moy_math_z     9
moy_ela_z   moy_ela_z     0
eoy_math_z eoy_math_z     9
eoy_ela_z   eoy_ela_z     0

Performed calculations among all students, regardless of Live+AI offering (N=15552).

summary_z_outliers_all = data.frame(
  variable = vars_z_outliers,
  count = sapply(vars_z_outliers, function(v) sum(pooled [[v]] > 3, na.rm = TRUE))
)

summary_z_outliers_all
             variable count
moy_math_z moy_math_z    76
moy_ela_z   moy_ela_z     1
eoy_math_z eoy_math_z    83
eoy_ela_z   eoy_ela_z     0