In this analysis, from given weather data we are going to find
1. which weather events are more harmful for population health.
2. which weather events have greatest economic consequences.
Weather data link
is provided by National
Climatic Data Center.
Since data not in proper format in ‘Data Processing’ section we describe
cleaning and transforming of data and putting the relevant column data
in proper format.
Once data is in proper format in result section we summarize the data
and draw the conclusion from it in ‘Result’ section.
library(knitr)
library(dplyr)
setwd("/home/lotus/CourseraCourses/Data_Science_Spl/C5/Assignment2")
#reading the the data after downloading the file repdata_data_StormData.csv.bz2
filepath = "./repdata_data_StormData.csv.bz2"
if (!file.exists(filepath))
{
fileUrl <- "https://d396qusza40orc.cloudfront.net/repdata%2Fdata%2FStormData.csv.bz2"
download.file(fileUrl, filepath)
}
origData <- read.csv("./repdata_data_StormData.csv.bz2")
keeping only those columns which are required for the analysis
usefullColumns <- c("EVTYPE", "FATALITIES", "INJURIES", "PROPDMG", "PROPDMGEXP",
"CROPDMG", "CROPDMGEXP")
origData <- origData[, usefullColumns]
making EVTYPE lowercase, removing space, ‘/’, ‘-’ from it
origData<-mutate(origData, EVTYPE = tolower(EVTYPE))
origData<-mutate(origData, EVTYPE = gsub(" ", "",EVTYPE))
origData<-mutate(origData, EVTYPE = gsub("\\/", "",EVTYPE))
origData<-mutate(origData, EVTYPE = gsub("-", "",EVTYPE))
library(dplyr)
groupByEvtData<- group_by(origData, EVTYPE)
fatalitySumbyEvt <- summarize(groupByEvtData, sum(FATALITIES))
fatalitySumbyEvt <- as.data.frame(fatalitySumbyEvt)
fatalitySumbyEvt <- fatalitySumbyEvt[order(fatalitySumbyEvt[, 2], decreasing = TRUE), ]
fatalitySumbyEvt<-fatalitySumbyEvt[1:10, ]
barplot(fatalitySumbyEvt[, 2], names.arg = fatalitySumbyEvt$EVTYPE, las =2, ylab = "count",
col = "red", legend.text = "Fatalities vs EventType graph")
** From above graph Tornado event has most fatalities **
injurySumbyEvt <- summarize(groupByEvtData, sum(INJURIES))
injurySumbyEvt <- as.data.frame(injurySumbyEvt)
injurySumbyEvt <- injurySumbyEvt[order(injurySumbyEvt[, 2], decreasing = TRUE), ]
injurySumbyEvt<-injurySumbyEvt[1:10, ]
barplot(injurySumbyEvt[, 2], names.arg = injurySumbyEvt$EVTYPE, las =2, ylab = "count",
col = "red", legend.text = "Injuries vs EventType graph")
** From above graph Tornado event has most injuries **
Converting PROPDMG and CROPDMG into numeric value obtained after multiplying CROPDMGEXP and CROPDMG, PROPDMGEXP and PROPDMG.
Here K = Thousands (10^3), M = Millions (10^6), B = Billions (10^9) H = Hundreds (10^2) for other letters we are using multiplier as 1 only
getMultiplier<- function (st)
{
mult = 1
st = toupper(st)
if (st == "K")
mult <- 10^3
else if (st == "M")
mult <- 10^6
else if (st == "B")
mult <- 10^9
else if (st == "H")
mult <- 10^2
mult
}
library(dplyr)
groupByEvtData<- group_by(origData, EVTYPE)
cnames <- colnames(groupByEvtData)
# calculate crop damage by multiplying it with multiplier
X <- sapply(groupByEvtData$CROPDMGEXP, getMultiplier)*groupByEvtData$CROPDMG
# calculate property damage by multiplying it with multiplier
Y <- sapply(groupByEvtData$PROPDMGEXP, getMultiplier)*groupByEvtData$PROPDMG
# calculating economic damage by sum of crop and property damage
Z <- X + Y
# adding columns for real crop and property damage after multipiying it with exponent
groupByEvtData<- cbind(groupByEvtData, X, Y, Z)
cnames <- c(cnames, "cropDamage", "propDamage", "economicDamage")
colnames(groupByEvtData) <- cnames
library(dplyr)
cropDmgSumbyEvt <- summarize(groupByEvtData, sum(cropDamage)/10^9)
cropDmgSumbyEvt <- as.data.frame(cropDmgSumbyEvt)
cropDmgSumbyEvt <- cropDmgSumbyEvt[order(cropDmgSumbyEvt[, 2], decreasing = TRUE), ]
cropDmgSumbyEvt<-cropDmgSumbyEvt[1:10, ]
colnames(cropDmgSumbyEvt) <- c("Event", "crop damage in Billions")
printing top 10 events which has most crop damage
library(knitr)
kable(cropDmgSumbyEvt, row.names = FALSE)
| Event | crop damage in Billions |
|---|---|
| drought | 13.972566 |
| flood | 5.661968 |
| riverflood | 5.029459 |
| icestorm | 5.022113 |
| hail | 3.025955 |
| hurricane | 2.741910 |
| hurricanetyphoon | 2.607873 |
| flashflood | 1.421317 |
| extremecold | 1.312973 |
| frostfreeze | 1.094186 |
** From above table drought event has most crop damage **
library(dplyr)
propDmgSumbyEvt <- summarize(groupByEvtData, sum(propDamage)/10^9)
propDmgSumbyEvt <- as.data.frame(propDmgSumbyEvt)
propDmgSumbyEvt <- propDmgSumbyEvt[order(propDmgSumbyEvt[, 2], decreasing = TRUE), ]
propDmgSumbyEvt<-propDmgSumbyEvt[1:10, ]
colnames(propDmgSumbyEvt) <- c("Event", "property damage in Billions")
printing top 10 events which has most property damage
library(knitr)
kable(propDmgSumbyEvt, row.names = FALSE)
| Event | property damage in Billions |
|---|---|
| flood | 144.657710 |
| hurricanetyphoon | 69.305840 |
| tornado | 56.937161 |
| stormsurge | 43.323536 |
| flashflood | 16.141362 |
| hail | 15.732268 |
| hurricane | 11.868319 |
| tropicalstorm | 7.703890 |
| winterstorm | 6.688497 |
| highwind | 5.270046 |
** From above table flood event has most property damage **
library(dplyr)
ecoDmgSumbyEvt <- summarize(groupByEvtData, sum(economicDamage)/10^9)
ecoDmgSumbyEvt <- as.data.frame(ecoDmgSumbyEvt)
ecoDmgSumbyEvt <- ecoDmgSumbyEvt[order(ecoDmgSumbyEvt[, 2], decreasing = TRUE), ]
ecoDmgSumbyEvt<-ecoDmgSumbyEvt[1:10, ]
barplot(ecoDmgSumbyEvt[, 2], names.arg = ecoDmgSumbyEvt$EVTYPE, las =2, ylab = "In Billion dollars",
col = "red", legend.text = "Total Damage vs EventType graph")
** From above graph flood event has most damage (property + crop) **