Code
library(tidyverse)
library(lubridate)
library(scales)
library(knitr)In this assignment, I selected a dataset from kaggle, regarding an analytical investigation of worldwide natural disaster occurrences documented between 2023 and 2025. This dataset contains important characteristics such the type of disaster, location, degree of severity, number of victims, financial losses, and emergency response timings are all captured here and also gives an opportunity to working with enormous numerical data.
My goal is to work with this data and get insights information through data driven approach and analyze incidents in different locations worldwide and get more information & visualize data to improve emergency response plans, risk mitigation techniques and disaster preparedness. So data driven insights are essential for identifying susceptible areas and allocating resources as emergency needed.
How we can improve more awareness to people before any natural disaster happens and how we can prevent loss or degree of severity, number of victims through this analysis?
Including all related library
library(tidyverse)
library(lubridate)
library(scales)
library(knitr)Import Dataset
Loading dataset into dataframe to clean and preparing a dataset by modifying its columns into correct datatypes.
data_file <- "synthetic_disaster_events_2025.csv"
data <- read_csv(data_file, show_col_types = FALSE) |>
mutate(
date = as.Date(date),
disaster_type = factor(disaster_type),
location = factor(location),
aid_provided = factor(aid_provided, levels = c("No", "Yes")),
is_major_disaster = as.logical(is_major_disaster)
)
cat("Rows:", nrow(data), " Columns:", ncol(data))Rows: 20000 Columns: 13
Check Dataset Structure
str(data)tibble [20,000 × 13] (S3: tbl_df/tbl/data.frame)
$ event_id : num [1:20000] 1 2 3 4 5 6 7 8 9 10 ...
$ disaster_type : Factor w/ 7 levels "Drought","Earthquake",..: 7 4 6 1 6 7 1 6 4 1 ...
$ location : Factor w/ 8 levels "Chile","India",..: 1 2 4 1 7 3 2 5 7 8 ...
$ latitude : num [1:20000] -34.7 22.1 42.3 -33.4 39.4 ...
$ longitude : num [1:20000] -71.8 78 11 -70 37 ...
$ date : Date[1:20000], format: "2025-08-27" "2023-05-29" ...
$ severity_level : num [1:20000] 8 5 7 8 8 9 7 7 4 4 ...
$ affected_population : num [1:20000] 31104 29340 34804 31191 46284 ...
$ estimated_economic_loss_usd: num [1:20000] 2768213 5996227 9222541 1827703 13435921 ...
$ response_time_hours : num [1:20000] 5.12 44.43 49.3 65.56 60.96 ...
$ aid_provided : Factor w/ 2 levels "No","Yes": 2 1 1 2 1 2 2 2 1 2 ...
$ infrastructure_damage_index: num [1:20000] 0.59 0.26 0.94 0.94 0.92 0.95 0.4 1 0.53 0.64 ...
$ is_major_disaster : logi [1:20000] TRUE FALSE TRUE TRUE TRUE TRUE ...
head(data)# A tibble: 6 × 13
event_id disaster_type location latitude longitude date severity_level
<dbl> <fct> <fct> <dbl> <dbl> <date> <dbl>
1 1 Wildfire Chile -34.7 -71.8 2025-08-27 8
2 2 Hurricane India 22.1 78.0 2023-05-29 5
3 3 Volcanic Erupt… Italy 42.3 11.0 2023-01-15 7
4 4 Drought Chile -33.4 -70.0 2024-02-08 8
5 5 Volcanic Erupt… Turkey 39.4 37.0 2023-12-23 8
6 6 Wildfire Indones… -2.48 113. 2023-11-19 9
# ℹ 6 more variables: affected_population <dbl>,
# estimated_economic_loss_usd <dbl>, response_time_hours <dbl>,
# aid_provided <fct>, infrastructure_damage_index <dbl>,
# is_major_disaster <lgl>
Dataset Overview
overview <- tibble(
Measure = c(
"Number of events",
"Number of variables",
"Earliest event date",
"Latest event date",
"Total affected population",
"Total estimated economic loss",
"Average severity",
"Average response time (hours)"
),
Value = c(
nrow(data),
ncol(data),
format(min(data$date, na.rm = TRUE), "%Y-%m-%d"),
format(max(data$date, na.rm = TRUE), "%Y-%m-%d"),
comma(sum(data$affected_population, na.rm = TRUE)),
dollar(sum(data$estimated_economic_loss_usd, na.rm = TRUE)),
round(mean(data$severity_level, na.rm = TRUE), 2),
round(mean(data$response_time_hours, na.rm = TRUE), 2)
)
)
kable(overview, caption = "Dataset overview")| Measure | Value |
|---|---|
| Number of events | 20000 |
| Number of variables | 13 |
| Earliest event date | 2022-12-08 |
| Latest event date | 2025-12-07 |
| Total affected population | 552,824,979 |
| Total estimated economic loss | $96,621,452,916 |
| Average severity | 5.49 |
| Average response time (hours) | 36.37 |
Event Frequency
events_by_type <- data %>% count(disaster_type, sort = TRUE)
ggplot(events_by_type, aes(y = reorder(disaster_type, n), x = n)) +
geom_col() + coord_flip() +
labs(y = NULL, x = "Number of events", title = "Disaster Events by Type")knitr::kable(events_by_type, col.names = c("Disaster type", "Events"))| Disaster type | Events |
|---|---|
| Earthquake | 2910 |
| Landslide | 2891 |
| Wildfire | 2870 |
| Hurricane | 2866 |
| Drought | 2863 |
| Volcanic Eruption | 2827 |
| Flood | 2773 |
Economic loss
loss_by_type <- data %>%
group_by(disaster_type) %>%
summarise(
events = n(),
total_loss_usd = sum(estimated_economic_loss_usd, na.rm = TRUE),
mean_loss_usd = mean(estimated_economic_loss_usd, na.rm = TRUE),
median_loss_usd = median(estimated_economic_loss_usd, na.rm = TRUE),
.groups = "drop"
) %>% arrange(desc(total_loss_usd))
ggplot(loss_by_type, aes(x = reorder(disaster_type, total_loss_usd), y = total_loss_usd)) +
geom_col() + coord_flip() + scale_y_continuous(labels = dollar_format()) +
labs(x = NULL, y = "Total estimated loss (USD)", title = "Estimated economic loss by disaster type")knitr::kable(loss_by_type, digits = 0)| disaster_type | events | total_loss_usd | mean_loss_usd | median_loss_usd |
|---|---|---|---|---|
| Earthquake | 2910 | 14339311149 | 4927598 | 4104313 |
| Landslide | 2891 | 13981221492 | 4836120 | 4024796 |
| Hurricane | 2866 | 13916903531 | 4855863 | 4028622 |
| Drought | 2863 | 13831854096 | 4831245 | 4015134 |
| Volcanic Eruption | 2827 | 13820939149 | 4888907 | 4110298 |
| Wildfire | 2870 | 13602514413 | 4739552 | 3970607 |
| Flood | 2773 | 13128709086 | 4734479 | 3984243 |
Major-disaster summary
major_summary <- data %>%
group_by(is_major_disaster) %>%
summarise(
events = n(),
avg_severity = mean(severity_level, na.rm = TRUE),
avg_affected = mean(affected_population, na.rm = TRUE),
avg_loss_usd = mean(estimated_economic_loss_usd, na.rm = TRUE),
avg_response_hours = mean(response_time_hours, na.rm = TRUE),
.groups = "drop"
) %>%
mutate(category = if_else(is_major_disaster == 1, "Major disaster", "Other disaster")) %>%
select(category, everything(), -is_major_disaster)
knitr::kable(major_summary, digits = 2)| category | events | avg_severity | avg_affected | avg_loss_usd | avg_response_hours |
|---|---|---|---|---|---|
| Other disaster | 11999 | 3.49 | 17806.37 | 3107868 | 36.42 |
| Major disaster | 8001 | 8.48 | 42390.50 | 7415341 | 36.29 |
Summary by location
location_summary <- data %>%
group_by(location) %>%
summarise(
events = n(),
total_affected = sum(affected_population, na.rm = TRUE),
total_loss_usd = sum(estimated_economic_loss_usd, na.rm = TRUE),
avg_severity = mean(severity_level, na.rm = TRUE),
.groups = "drop"
) %>% arrange(desc(total_loss_usd))
knitr::kable(location_summary, digits = 2)| location | events | total_affected | total_loss_usd | avg_severity |
|---|---|---|---|---|
| India | 2583 | 72183422 | 12652716823 | 5.54 |
| Indonesia | 2530 | 70193413 | 12309428718 | 5.51 |
| Chile | 2515 | 70115117 | 12236377078 | 5.56 |
| Italy | 2485 | 69355495 | 12129163356 | 5.48 |
| USA | 2442 | 67448482 | 11940119463 | 5.48 |
| Philippines | 2453 | 67279471 | 11867675791 | 5.46 |
| Turkey | 2515 | 68387863 | 11804815025 | 5.46 |
| Japan | 2477 | 67861716 | 11681156662 | 5.43 |
Find relationships between among impact measures -
Severity level vs. Economic loss and correlation coefficient value between these two factors.
Response time vs. Affected population and correlation coefficient value between these two factors.