library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.2.0
## ✔ forcats 1.0.1 ✔ stringr 1.6.0
## ✔ ggplot2 4.0.3 ✔ tibble 3.3.1
## ✔ lubridate 1.9.5 ✔ tidyr 1.3.2
## ✔ purrr 1.2.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(readxl)
library(pastecs)
##
## Attaching package: 'pastecs'
##
## The following objects are masked from 'package:dplyr':
##
## first, last
##
## The following object is masked from 'package:tidyr':
##
## extract
school_shootings <- read_csv("schoolshootings.csv")
## Rows: 398 Columns: 58
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (46): Full Date, Day of Week, Time of Day, During Class?, School Name, ...
## dbl (10): ID #, Month, Day, Year, Latitude, Longitude, Number of
## Victims Ki...
## num (1): Enrollment
## time (1): Time Homicide
## Occurred
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
For my assignment, I will examine the number of students who were fatally injured.
stat.desc(school_shootings$"Number of\nVictims Killed")
## nbr.val nbr.null nbr.na min max range
## 398.00000000 0.00000000 0.00000000 1.00000000 26.00000000 25.00000000
## sum median mean SE.mean CI.mean.0.95 var
## 529.00000000 1.00000000 1.32914573 0.09723471 0.19115931 3.76292672
## std.dev coef.var
## 1.93982647 1.45945356
# I remember some of the datasets had no deaths, so I ran this test to see all of the values in this column. This chart only includes events in which someone died.
#I am also able to see from this test and the chart above that I am not missing any values
school_shootings$"Number of\nVictims Killed"
## [1] 1 1 2 1 1 1 2 1 1 1 1 1 2 1 1 1 1 1 1 1 1 1 1 1 1
## [26] 1 1 1 1 1 1 1 1 2 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1
## [51] 1 1 1 1 1 1 9 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1
## [76] 1 1 1 5 1 1 1 1 1 1 1 1 3 1 1 1 1 1 1 1 1 1 1 1 1
## [101] 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 3 1 1 1 1 1 1
## [126] 26 1 1 1 2 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 2 1 4 1
## [151] 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 2 1 2 1 1 1 1 1 1 2
## [176] 2 1 17 1 1 10 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1 1
## [201] 1 1 1 2 1 1 1 1 1 1 1 1 2 1 2 1 1 1 1 1 1 1 1 3 1
## [226] 1 1 1 1 1 1 1 2 1 1 1 1 2 1 1 1 1 1 1 1 1 1 1 1 1
## [251] 1 1 1 1 1 1 1 1 1 1 1 1 1 1 4 2 1 1 2 1 1 1 1 1 1
## [276] 1 1 1 1 1 1 1 1 1 1 1 21 1 1 1 1 1 1 1 1 1 1 1 2 1
## [301] 1 1 1 2 1 1 1 1 1 1 1 6 1 1 1 1 1 1 1 1 1 1 1 1 1
## [326] 1 1 1 1 1 1 1 1 1 1 1 1 1 2 1 1 1 1 2 2 1 1 1 1 2
## [351] 1 1 1 1 1 1 1 1 1 1 4 1 1 1 1 1 1 1 1 1 1 1 2 1 1
## [376] 1 1 1 1 1 1 1 1 1 1 1 2 1 1 1 1 2 2 1 3 1 1 1
hist(school_shootings$"Number of\nVictims Killed", breaks = 26)
The histogram is extremely right-skewed because the mean average number
of students killed is 1, however there are extremes like 26. These
values create an instagram with a long tail on the right-hand side.
To process outliers, I am going to use log transformation.
# mt_cars_log<-mtcars %>% mutate(LOG_HP=log(hp)) %>% select(hp,LOG_HP)
# I'm running into issues with the way the column is written. I am going to rename it
# I got help and found this turns newlines and spaces into clean dots
library(dplyr)
colnames(school_shootings) <- make.names(colnames(school_shootings))
colnames(school_shootings)
## [1] "ID.."
## [2] "Full.Date"
## [3] "Month"
## [4] "Day"
## [5] "Year"
## [6] "Day.of.Week"
## [7] "Time.of.Day"
## [8] "Time.Homicide.Occurred"
## [9] "During.Class."
## [10] "School.Name"
## [11] "Address"
## [12] "City"
## [13] "State"
## [14] "Zip.code"
## [15] "Density"
## [16] "Latitude"
## [17] "Longitude"
## [18] "Industry...Type.of.Business."
## [19] "School.Level"
## [20] "Number.of.Employees.at.Site"
## [21] "Enrollment"
## [22] "Accessibility.to.Public"
## [23] "Gun.Free.Zone."
## [24] "Security.Systems"
## [25] "Lockdown."
## [26] "How.Perpetrator.Accessed"
## [27] "Armed.Person.on.Scene."
## [28] "Location.of..Homicide"
## [29] "Location.of.Homicide.Specify"
## [30] "Number.of.Victims.Killed"
## [31] "Number.of.Victims.Injured"
## [32] "Number.of.Shooting.Injuries"
## [33] "Number.of..Other.Injuries"
## [34] "Students.Killed."
## [35] "Staff.Killed."
## [36] "Students.Injured."
## [37] "Staff.Injured."
## [38] "Hostages."
## [39] "Weapon.Used"
## [40] "How.was.Weapon.Obtained."
## [41] "When.was.Weapon.Obtained."
## [42] "Other.Weapons"
## [43] "Preplanned."
## [44] "Degree.of.Planning"
## [45] "Inspired.by.or..Studied.Other..Perpetrators"
## [46] "Inspired.by.or.Studied..Other.Perpetrators..Specify"
## [47] "Connected.to..Another.Homicide."
## [48] "Connected.to.Another.Homicide.Specify"
## [49] "Active.Shooter."
## [50] "Gang.Group.Related"
## [51] "Drive.by."
## [52] "Domestic."
## [53] "Situation"
## [54] "On.Scene..Outcome"
## [55] "Conviction.Outcome"
## [56] "Short.Summary"
## [57] "Full.Description"
## [58] "Articles"
names(school_shootings)[30] <- "Number.of.Victims.Killed"
colnames(school_shootings)
## [1] "ID.."
## [2] "Full.Date"
## [3] "Month"
## [4] "Day"
## [5] "Year"
## [6] "Day.of.Week"
## [7] "Time.of.Day"
## [8] "Time.Homicide.Occurred"
## [9] "During.Class."
## [10] "School.Name"
## [11] "Address"
## [12] "City"
## [13] "State"
## [14] "Zip.code"
## [15] "Density"
## [16] "Latitude"
## [17] "Longitude"
## [18] "Industry...Type.of.Business."
## [19] "School.Level"
## [20] "Number.of.Employees.at.Site"
## [21] "Enrollment"
## [22] "Accessibility.to.Public"
## [23] "Gun.Free.Zone."
## [24] "Security.Systems"
## [25] "Lockdown."
## [26] "How.Perpetrator.Accessed"
## [27] "Armed.Person.on.Scene."
## [28] "Location.of..Homicide"
## [29] "Location.of.Homicide.Specify"
## [30] "Number.of.Victims.Killed"
## [31] "Number.of.Victims.Injured"
## [32] "Number.of.Shooting.Injuries"
## [33] "Number.of..Other.Injuries"
## [34] "Students.Killed."
## [35] "Staff.Killed."
## [36] "Students.Injured."
## [37] "Staff.Injured."
## [38] "Hostages."
## [39] "Weapon.Used"
## [40] "How.was.Weapon.Obtained."
## [41] "When.was.Weapon.Obtained."
## [42] "Other.Weapons"
## [43] "Preplanned."
## [44] "Degree.of.Planning"
## [45] "Inspired.by.or..Studied.Other..Perpetrators"
## [46] "Inspired.by.or.Studied..Other.Perpetrators..Specify"
## [47] "Connected.to..Another.Homicide."
## [48] "Connected.to.Another.Homicide.Specify"
## [49] "Active.Shooter."
## [50] "Gang.Group.Related"
## [51] "Drive.by."
## [52] "Domestic."
## [53] "Situation"
## [54] "On.Scene..Outcome"
## [55] "Conviction.Outcome"
## [56] "Short.Summary"
## [57] "Full.Description"
## [58] "Articles"
library(dplyr)
# Do the log-transformation
school_shootings_log <- school_shootings %>%
mutate(LOG_Number.of.Victims.Killed=log(Number.of.Victims.Killed))
head(school_shootings_log)
## # A tibble: 6 × 59
## ID.. Full.Date Month Day Year Day.of.Week Time.of.Day
## <dbl> <chr> <dbl> <dbl> <dbl> <chr> <chr>
## 1 1 01/19/2000 1 19 2000 Wednesday Afternoon
## 2 2 02/29/2000 2 29 2000 Tuesday Morning
## 3 3 03/10/2000 3 10 2000 Friday Night
## 4 4 05/05/2000 5 5 2000 Friday Unknown
## 5 5 05/10/2000 5 10 2000 Wednesday Afternoon
## 6 6 05/26/2000 5 26 2000 Friday Afternoon
## # ℹ 52 more variables: Time.Homicide.Occurred <time>, During.Class. <chr>,
## # School.Name <chr>, Address <chr>, City <chr>, State <chr>, Zip.code <chr>,
## # Density <chr>, Latitude <dbl>, Longitude <dbl>,
## # Industry...Type.of.Business. <chr>, School.Level <chr>,
## # Number.of.Employees.at.Site <chr>, Enrollment <dbl>,
## # Accessibility.to.Public <chr>, Gun.Free.Zone. <chr>,
## # Security.Systems <chr>, Lockdown. <chr>, How.Perpetrator.Accessed <chr>, …
# this took about 10 tries. I had to go back and rename columns. I am not sure exactly what was wrong but I think it may have been the column names
# Make a histogram of the converted numbers
hist(school_shootings_log$LOG_Number.of.Victims.Killed, breaks=10, probability = T)