Introduction

This document presents the analysis and data processing steps for the Election Campaign Simulation. The focus is on precinct-level election data and voter files to identify GOTV (Get Out The Vote) and persuasion targets.

Libraries and Data Loading

Load Required Libraries

library(tidyverse)
library(knitr)

Load and Assign Column Names to Precinct Results Data

data_2016 <- read_delim("C:/Users/alesa/OneDrive/Desktop/grad school/POS 6933/Election Simulation/Data/Precinct Results/DAD_PctResults20161108.txt", col_names = FALSE, col_select = c(6, 12, 16, 19))
colnames(data_2016) <- c("precinct", "contest_name", "candidate_party", "vote_total")

data_2018 <- read_delim("C:/Users/alesa/OneDrive/Desktop/grad school/POS 6933/Election Simulation/Data/Precinct Results/DAD_PctResults20181106.txt", col_names = FALSE, col_select = c(6, 12, 16, 19))
colnames(data_2018) <- c("precinct", "contest_name", "candidate_party", "vote_total")

data_2020 <- read_delim("C:/Users/alesa/OneDrive/Desktop/grad school/POS 6933/Election Simulation/Data/Precinct Results/DAD_PctResults20201103.txt", col_names = FALSE, col_select = c(6, 12, 16, 19))
colnames(data_2020) <- c("precinct", "contest_name", "candidate_party", "vote_total")

data_2022 <- read_delim("C:/Users/alesa/OneDrive/Desktop/grad school/POS 6933/Election Simulation/Data/Precinct Results/DAD_PctResults20221108.txt", col_names = FALSE, col_select = c(6, 12, 16, 19))
colnames(data_2022) <- c("precinct", "contest_name", "candidate_party", "vote_total")

Data Cleaning

Standardize Precinct Numbers

data_2018 <- data_2018 %>% mutate(precinct = str_pad(precinct, 3, pad = "0"))
data_2020 <- data_2020 %>% mutate(precinct = str_sub(precinct, 1, 3))
data_2022 <- data_2022 %>% mutate(precinct = str_sub(precinct, 1, 3))

Standardize Contest Names

data_2022 <- data_2022 %>% mutate(contest_name = ifelse(contest_name == "Governor and Lieutenant Governor", "Governor", contest_name))

Process and Filter Election Data

data_2016_processed <- data_2016 %>%
  filter(contest_name %in% c("President of the United States", "Representative in Congress", "United States Senator", "Governor", "State Representative", "State Senator")) %>%
  group_by(precinct, contest_name) %>%
  summarize(
    dem_votes = sum(vote_total[candidate_party == "DEM"]),
    rep_votes = sum(vote_total[candidate_party == "REP"]),
    total_votes = dem_votes + rep_votes,
    vote_share_dem = dem_votes / total_votes * 100,
    vote_share_rep = rep_votes / total_votes * 100
  ) %>%
  ungroup() %>%
  mutate(year = 2016) %>%
  select(precinct, contest_name, year, vote_share_dem, vote_share_rep, total_votes) %>%  
  filter(total_votes >= 100)
  
data_2018_processed <- data_2018 %>%
  filter(contest_name %in% c("President of the United States", "Representative in Congress", "United States Senator", "Governor", "State Representative", "State Senator")) %>%
  group_by(precinct, contest_name) %>%
  summarize(
    dem_votes = sum(vote_total[candidate_party == "DEM"]),
    rep_votes = sum(vote_total[candidate_party == "REP"]),
    total_votes = dem_votes + rep_votes,
    vote_share_dem = dem_votes / total_votes * 100,
    vote_share_rep = rep_votes / total_votes * 100
  ) %>%
  ungroup() %>%
  mutate(year = 2018) %>%
  select(precinct, contest_name, year, vote_share_dem, vote_share_rep, total_votes) %>%  
  filter(total_votes >= 100)

data_2020_processed <- data_2020 %>%
  filter(contest_name %in% c("President of the United States", "Representative in Congress", "United States Senator", "Governor", "State Representative", "State Senator")) %>%
  group_by(precinct, contest_name) %>%
  summarize(
    dem_votes = sum(vote_total[candidate_party == "DEM"]),
    rep_votes = sum(vote_total[candidate_party == "REP"]),
    total_votes = dem_votes + rep_votes,
    vote_share_dem = dem_votes / total_votes * 100,
    vote_share_rep = rep_votes / total_votes * 100
  ) %>%
  ungroup() %>%
  mutate(year = 2020) %>%
  select(precinct, contest_name, year, vote_share_dem, vote_share_rep, total_votes) %>%  
  filter(total_votes >= 100)

data_2022_processed <- data_2022 %>%
  filter(contest_name %in% c("President of the United States", "Representative in Congress", "United States Senator", "Governor", "State Representative", "State Senator")) %>%
  group_by(precinct, contest_name) %>%
  summarize(
    dem_votes = sum(vote_total[candidate_party == "DEM"]),
    rep_votes = sum(vote_total[candidate_party == "REP"]),
    total_votes = dem_votes + rep_votes,
    vote_share_dem = dem_votes / total_votes * 100,
    vote_share_rep = rep_votes / total_votes * 100
  ) %>%
  ungroup() %>%
  mutate(year = 2022) %>%
  select(precinct, contest_name, year, vote_share_dem, vote_share_rep, total_votes) %>%  
  filter(total_votes >= 100)

Combining and Analyzing Data

Combine All Years

combined_data <- bind_rows(data_2016_processed, data_2018_processed, data_2020_processed, data_2022_processed)

Calculate Averages and Standard Deviations

average_and_sd <- combined_data %>%
  group_by(precinct) %>%
  summarize(
    avg_vote_share_dem = mean(vote_share_dem, na.rm = TRUE),
    sd_vote_share_dem = sd(vote_share_dem, na.rm = TRUE)       
  )

Target Precincts

GOTV Target Precincts

gotv_target_precincts <- average_and_sd %>%
  arrange(desc(avg_vote_share_dem)) %>%
  slice_head(n = 25)
Top 25 GOTV Target Precincts by Average Democratic Vote Share
Precinct- Dem Vote Share- Vote Variation
222 95.57956 1.350201
172 95.37798 2.949766
508 95.16253 3.238699
216 93.99078 2.530406
251 93.75685 3.565219
221 93.72252 2.520116
173 93.63684 4.108892
257 93.54195 3.282334
218 93.53312 3.430555
520 93.51469 3.085034
214 93.40552 2.250026
511 93.36604 3.622358
205 93.34540 3.238355
505 93.28404 4.008555
145 93.26859 3.324025
269 93.13085 2.718910
506 93.10736 4.299125
217 92.98806 2.657165
254 92.93799 4.164430
521 92.88133 4.514028
253 92.87898 3.727270
507 92.87455 4.384368
519 92.65620 4.167402
213 92.59933 2.512293
206 92.44235 2.838327

Persuasion Target Precincts

persuasion_target_precincts <- average_and_sd %>%
  arrange(desc(sd_vote_share_dem)) %>%
  slice_head(n = 25)
Top 25 Persuasion Target Precincts by Standard Deviation in Democratic Vote Share
Precinct- Dem Vote Share- Vote Variation
116 44.14024 18.68507
003 39.55866 18.67399
067 41.69257 18.40957
001 37.91634 18.21702
002 45.30230 17.70280
004 51.89023 17.57602
006 41.87254 17.17492
120 50.99085 16.48928
181 63.43092 16.47919
005 47.95894 16.34286
892 46.95068 16.16370
119 54.70992 15.89984
981 60.69190 15.79222
124 54.34991 15.62646
833 49.69262 15.43914
980 63.25100 15.01594
527 60.21388 14.90112
025 46.95226 14.70331
114 51.86225 14.47704
099 50.73125 14.38135
157 68.08453 14.34629
109 58.32753 14.25152
009 53.88160 14.24967
028 48.78921 14.21243
201 52.97533 14.06964

Voter Files

Load Voter Registration Data

voter_file <- read_delim("C:/Users/alesa/OneDrive/Desktop/grad school/POS 6933/Election Simulation/Data/VF/DAD_20240924.txt", col_names = FALSE, col_select = c(2, 3, 4, 5, 6, 8, 9, 10, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, 29, 35, 36, 37, 38))
colnames(voter_file) <- c("voter_id", "name_last", "name_suffix", "name_first", "name_middle", "res_address_1", "res_address_2", "city", "zipcode", "mail_address_1", "mail_address_2", "mail_address_3", "mail_city", "mail_state", "mail_zip", "mail_country", "gender", "race", "birth_date", "reg_date", "party", "precinct", "status", "area_code", "phone_number", "phone_ext", "email")

Filter GOTV and Persuasion Voters

gotv_voters <- voter_file %>%
  filter(precinct %in% gotv_target_precincts$precinct, party == "DEM")

persuasion_voters <- voter_file %>%
  filter(precinct %in% persuasion_target_precincts$precinct, party == "NPA")

Specialized Lists

Ballots Not Accepted

voter_history <- read_delim("C:/Users/alesa/OneDrive/Desktop/grad school/POS 6933/Election Simulation/Data/VF/DAD_H_20210112.txt", col_names = FALSE, col_select = c(2, 4, 5))
colnames(voter_history) <- c("voter_id", "election_type", "history_code")

ballots_not_accepted <- voter_history %>%
  filter(history_code %in% c("B", "L")) %>% 
  filter(election_type %in% c("GEN"))

gotv_targets_not_accepted <- gotv_voters %>%
  inner_join(ballots_not_accepted, by = "voter_id")

Save Output Files

write_csv(gotv_voters, "gotv_voter_contacts.csv")

write_csv(persuasion_voters, "persuasion_voter_contacts.csv")

write_csv(gotv_targets_not_accepted, "vote_by_mail_reminders.csv")

Conclusion

This analysis successfully identified target precincts and voter groups for GOTV and persuasion efforts. Additionally, specialized lists were created for micro-targeting efforts, such as reminding voters whose ballots were not accepted to participate in upcoming elections.