Hate Crimes

Author

Michael Kuzin

library(tidyverse)
── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
✔ dplyr     1.2.1     ✔ readr     2.2.0
✔ forcats   1.0.1     ✔ stringr   1.6.0
✔ ggplot2   4.0.3     ✔ tibble    3.3.1
✔ lubridate 1.9.5     ✔ tidyr     1.3.2
✔ purrr     1.2.2     
── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
✖ dplyr::filter() masks stats::filter()
✖ dplyr::lag()    masks stats::lag()
ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(knitr)

hatecrimes <- read_csv("NYPD_Hate_crimes_19-26.csv.csv")
Rows: 4029 Columns: 14
── Column specification ────────────────────────────────────────────────────────
Delimiter: ","
chr (9): Record Create Date, Patrol Borough Name, County, Law Code Category ...
dbl (4): Full Complaint ID, Complaint Year Number, Month Number, Complaint P...
lgl (1): Arrest Date

ℹ Use `spec()` to retrieve the full column specification for this data.
ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
names(hatecrimes) <- tolower(names(hatecrimes))
names(hatecrimes) <- gsub(" ","",names(hatecrimes))

head(hatecrimes)
# A tibble: 6 × 14
  fullcomplaintid complaintyearnumber monthnumber recordcreatedate
            <dbl>               <dbl>       <dbl> <chr>           
1         2.02e14                2019           1 1/23/2019       
2         2.02e14                2019           2 2/25/2019       
3         2.02e14                2019           2 2/27/2019       
4         2.02e14                2019           4 4/16/2019       
5         2.02e14                2019           6 6/20/2019       
6         2.02e14                2019           7 7/31/2019       
# ℹ 10 more variables: complaintprecinctcode <dbl>, patrolboroughname <chr>,
#   county <chr>, lawcodecategorydescription <chr>, offensedescription <chr>,
#   pdcodedescription <chr>, biasmotivedescription <chr>,
#   offensecategory <chr>, arrestdate <lgl>, arrestid <chr>
bias_count <- hatecrimes |>
  select(biasmotivedescription) |>
  group_by(biasmotivedescription) |>
  count() |>
  arrange(desc(n))

head(bias_count)
# A tibble: 6 × 2
# Groups:   biasmotivedescription [6]
  biasmotivedescription          n
  <chr>                      <int>
1 ANTI-JEWISH                 1906
2 ANTI-MALE HOMOSEXUAL (GAY)   489
3 ANTI-ASIAN                   401
4 ANTI-BLACK                   315
5 ANTI-OTHER ETHNICITY         168
6 ANTI-MUSLIM                  156
ggplot(hatecrimes, aes(x = biasmotivedescription)) +
  geom_bar()

bias_count |>
  head(10) |>
  ggplot(aes(x = biasmotivedescription, y = n)) +
  geom_col()

bias_count |>
  head(10) |>
  ggplot(aes(x = reorder(biasmotivedescription, n), y = n)) +
  geom_col() +
  coord_flip()

bias_count |>
  head(10) |>
  ggplot(aes(x = reorder(biasmotivedescription, n), y = n)) +
  geom_col() +
  coord_flip() +
  labs(
    x = "",
    y = "Counts of hatecrime types based on motive",
    title = "Bar Graph of Hate Crimes from 2019-2026",
    subtitle = "Counts based on the hatecrime motive",
    caption = "Source: NYPD Hate Crimes (NYC Open Data)"
  )

bias_count |>
  head(10) |>
  ggplot(aes(x = reorder(biasmotivedescription, n), y = n)) +
  geom_col(fill = "salmon") +
  coord_flip() +
  labs(
    x = "",
    y = "Counts of hatecrime types based on motive",
    title = "Bar Graph of Hate Crimes from 2019-2026",
    subtitle = "Counts based on the hatecrime motive",
    caption = "Source: NYPD Hate Crimes (NYC Open Data)"
  ) +
  theme_minimal()

bias_count |>
  head(10) |>
  ggplot(aes(x = reorder(biasmotivedescription, n), y = n)) +
  geom_col(fill = "salmon") +
  coord_flip() +
  labs(
    x = "",
    y = "Counts of hatecrime types based on motive",
    title = "Bar Graph of Hate Crimes from 2019-2026",
    subtitle = "Counts based on the hatecrime motive",
    caption = "Source: NYPD Hate Crimes (NYC Open Data)"
  ) +
  theme_minimal() +
  geom_text(aes(label = n), hjust = -.05, size = 3) +
  theme(axis.text.x = element_blank())

hate_year <- hatecrimes |>
  filter(biasmotivedescription %in% c(
    "ANTI-JEWISH",
    "ANTI-MALE HOMOSEXUAL (GAY)",
    "ANTI-ASIAN",
    "ANTI-BLACK"
  )) |>
  group_by(complaintyearnumber) |>
  count(biasmotivedescription) |>
  arrange(desc(n))

hate_year
# A tibble: 28 × 3
# Groups:   complaintyearnumber [7]
   complaintyearnumber biasmotivedescription          n
                 <dbl> <chr>                      <int>
 1                2024 ANTI-JEWISH                  371
 2                2023 ANTI-JEWISH                  343
 3                2025 ANTI-JEWISH                  320
 4                2022 ANTI-JEWISH                  279
 5                2019 ANTI-JEWISH                  252
 6                2021 ANTI-JEWISH                  215
 7                2021 ANTI-ASIAN                   150
 8                2020 ANTI-JEWISH                  126
 9                2023 ANTI-MALE HOMOSEXUAL (GAY)   116
10                2022 ANTI-ASIAN                    91
# ℹ 18 more rows
hate_county <- hatecrimes |>
  filter(biasmotivedescription %in% c(
    "ANTI-JEWISH",
    "ANTI-MALE HOMOSEXUAL (GAY)",
    "ANTI-ASIAN",
    "ANTI-BLACK"
  )) |>
  group_by(county) |>
  count(biasmotivedescription) |>
  arrange(desc(n))

hate_county
# A tibble: 20 × 3
# Groups:   county [5]
   county   biasmotivedescription          n
   <chr>    <chr>                      <int>
 1 KINGS    ANTI-JEWISH                  798
 2 NEW YORK ANTI-JEWISH                  651
 3 QUEENS   ANTI-JEWISH                  289
 4 NEW YORK ANTI-MALE HOMOSEXUAL (GAY)   237
 5 NEW YORK ANTI-ASIAN                   228
 6 KINGS    ANTI-MALE HOMOSEXUAL (GAY)   120
 7 KINGS    ANTI-BLACK                    99
 8 BRONX    ANTI-JEWISH                   92
 9 QUEENS   ANTI-MALE HOMOSEXUAL (GAY)    91
10 KINGS    ANTI-ASIAN                    80
11 NEW YORK ANTI-BLACK                    79
12 QUEENS   ANTI-ASIAN                    78
13 RICHMOND ANTI-JEWISH                   76
14 QUEENS   ANTI-BLACK                    75
15 BRONX    ANTI-MALE HOMOSEXUAL (GAY)    35
16 RICHMOND ANTI-BLACK                    35
17 BRONX    ANTI-BLACK                    27
18 BRONX    ANTI-ASIAN                    10
19 RICHMOND ANTI-MALE HOMOSEXUAL (GAY)     6
20 RICHMOND ANTI-ASIAN                     5
# MERGE: join census population data to the county hate crime counts
county_pop <- tibble(
  county = c("BRONX", "KINGS", "NEW YORK", "QUEENS", "RICHMOND"),
  population = c(1472654, 2736074, 1694251, 2405464, 495747)
)

hate_county_pop <- hate_county |>
  ungroup() |>
  left_join(county_pop, by = "county") |>
  mutate(rate_per_100k = round(n / population * 100000, 1)) |>
  arrange(desc(rate_per_100k))

hate_county_pop
# A tibble: 20 × 5
   county   biasmotivedescription          n population rate_per_100k
   <chr>    <chr>                      <int>      <dbl>         <dbl>
 1 NEW YORK ANTI-JEWISH                  651    1694251          38.4
 2 KINGS    ANTI-JEWISH                  798    2736074          29.2
 3 RICHMOND ANTI-JEWISH                   76     495747          15.3
 4 NEW YORK ANTI-MALE HOMOSEXUAL (GAY)   237    1694251          14  
 5 NEW YORK ANTI-ASIAN                   228    1694251          13.5
 6 QUEENS   ANTI-JEWISH                  289    2405464          12  
 7 RICHMOND ANTI-BLACK                    35     495747           7.1
 8 BRONX    ANTI-JEWISH                   92    1472654           6.2
 9 NEW YORK ANTI-BLACK                    79    1694251           4.7
10 KINGS    ANTI-MALE HOMOSEXUAL (GAY)   120    2736074           4.4
11 QUEENS   ANTI-MALE HOMOSEXUAL (GAY)    91    2405464           3.8
12 KINGS    ANTI-BLACK                    99    2736074           3.6
13 QUEENS   ANTI-ASIAN                    78    2405464           3.2
14 QUEENS   ANTI-BLACK                    75    2405464           3.1
15 KINGS    ANTI-ASIAN                    80    2736074           2.9
16 BRONX    ANTI-MALE HOMOSEXUAL (GAY)    35    1472654           2.4
17 BRONX    ANTI-BLACK                    27    1472654           1.8
18 RICHMOND ANTI-MALE HOMOSEXUAL (GAY)     6     495747           1.2
19 RICHMOND ANTI-ASIAN                     5     495747           1  
20 BRONX    ANTI-ASIAN                    10    1472654           0.7
hate2 <- hatecrimes |>
  filter(biasmotivedescription %in% c(
    "ANTI-JEWISH",
    "ANTI-MALE HOMOSEXUAL (GAY)",
    "ANTI-ASIAN",
    "ANTI-BLACK"
  )) |>
  group_by(complaintyearnumber, county) |>
  count(biasmotivedescription) |>
  arrange(desc(n))

hate2
# A tibble: 127 × 4
# Groups:   complaintyearnumber, county [35]
   complaintyearnumber county   biasmotivedescription     n
                 <dbl> <chr>    <chr>                 <int>
 1                2024 KINGS    ANTI-JEWISH             152
 2                2024 NEW YORK ANTI-JEWISH             136
 3                2025 KINGS    ANTI-JEWISH             136
 4                2019 KINGS    ANTI-JEWISH             128
 5                2023 KINGS    ANTI-JEWISH             126
 6                2022 KINGS    ANTI-JEWISH             125
 7                2023 NEW YORK ANTI-JEWISH             124
 8                2025 NEW YORK ANTI-JEWISH             110
 9                2022 NEW YORK ANTI-JEWISH             104
10                2021 NEW YORK ANTI-ASIAN               84
# ℹ 117 more rows
ggplot(data = hate2) +
  geom_bar(
    aes(
      x = complaintyearnumber,
      y = n,
      fill = biasmotivedescription
    ),
    position = "dodge",
    stat = "identity"
  ) +
  labs(
    fill = "Hate Crime Type",
    y = "Number of Hate Crime Incidents",
    title = "Hate Crime Type in NY Counties Between 2019-2026",
    caption = "Source: NYPD Hate Crimes (NYC Open Data)"
  )

One thing that I noticed about datasets like this is that it gives you the ability to look from year to year and compare how the city’s hate crime rate is doing. You can see if the city is improving or not and what types of hate crimes are becoming more or less common. Another positive thing is that you can look at different counties and compare them to each other. One negative thing is that the dataset might not show every hate crime that actually happened. Some people might not report what happened, so the data might not show the full situation.

One thing I would want to study is how hate crimes changed from 2019 to 2026 and see which years had the biggest changes. Another thing I would want to study is the different counties. I would compare them and see which counties have more hate crimes and what types of hate crimes are most common in each one.