library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.2.0
## ✔ forcats 1.0.1 ✔ stringr 1.6.0
## ✔ ggplot2 4.0.3 ✔ tibble 3.3.1
## ✔ lubridate 1.9.5 ✔ tidyr 1.3.2
## ✔ purrr 1.2.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(janitor)
##
## Attaching package: 'janitor'
##
## The following objects are masked from 'package:stats':
##
## chisq.test, fisher.test
library(readxl)
garbage_raw <- read_excel("C:\\Users\\Grace Gallagher\\Documents\\PEP809A\\Garbage Weight.xlsx")
glimpse(garbage_raw)
## Rows: 62
## Columns: 10
## $ `HH SIZE` <dbl> 2, 3, 3, 6, 4, 2, 1, 5, 6, 4, 4, 7, 3, 5, 6, 2, 4, 4, 3, 3, …
## $ METAL <dbl> 1.09, 1.04, 2.57, 3.02, 1.50, 2.10, 1.93, 3.57, 2.32, 1.89, …
## $ PAPER <dbl> 2.41, 7.57, 9.55, 8.82, 8.72, 6.96, 6.83, 11.42, 16.08, 6.38…
## $ PLASTIC <dbl> 0.27, 1.41, 2.19, 2.83, 2.19, 1.81, 0.85, 3.05, 3.42, 2.10, …
## $ GLASS <dbl> 0.86, 3.46, 4.52, 4.92, 6.31, 2.49, 0.51, 5.81, 1.96, 17.67,…
## $ FOOD <dbl> 1.04, 3.68, 4.43, 2.98, 6.30, 1.46, 8.82, 9.62, 4.41, 2.73, …
## $ YARD <dbl> 0.38, 0.00, 0.24, 0.63, 0.15, 4.58, 0.07, 4.76, 0.13, 3.86, …
## $ TEXTILE <dbl> 0.05, 0.46, 0.50, 2.26, 0.55, 0.36, 0.60, 0.21, 0.81, 0.66, …
## $ OTHER <dbl> 4.66, 2.34, 3.60, 12.65, 2.18, 2.14, 2.22, 10.83, 4.14, 0.25…
## $ TOTAL <dbl> 10.76, 19.96, 27.60, 38.11, 27.90, 21.90, 21.83, 49.27, 33.2…
garbage <- garbage_raw %>%
clean_names()
glimpse(garbage)
## Rows: 62
## Columns: 10
## $ hh_size <dbl> 2, 3, 3, 6, 4, 2, 1, 5, 6, 4, 4, 7, 3, 5, 6, 2, 4, 4, 3, 3, 2,…
## $ metal <dbl> 1.09, 1.04, 2.57, 3.02, 1.50, 2.10, 1.93, 3.57, 2.32, 1.89, 3.…
## $ paper <dbl> 2.41, 7.57, 9.55, 8.82, 8.72, 6.96, 6.83, 11.42, 16.08, 6.38, …
## $ plastic <dbl> 0.27, 1.41, 2.19, 2.83, 2.19, 1.81, 0.85, 3.05, 3.42, 2.10, 2.…
## $ glass <dbl> 0.86, 3.46, 4.52, 4.92, 6.31, 2.49, 0.51, 5.81, 1.96, 17.67, 3…
## $ food <dbl> 1.04, 3.68, 4.43, 2.98, 6.30, 1.46, 8.82, 9.62, 4.41, 2.73, 9.…
## $ yard <dbl> 0.38, 0.00, 0.24, 0.63, 0.15, 4.58, 0.07, 4.76, 0.13, 3.86, 0.…
## $ textile <dbl> 0.05, 0.46, 0.50, 2.26, 0.55, 0.36, 0.60, 0.21, 0.81, 0.66, 0.…
## $ other <dbl> 4.66, 2.34, 3.60, 12.65, 2.18, 2.14, 2.22, 10.83, 4.14, 0.25, …
## $ total <dbl> 10.76, 19.96, 27.60, 38.11, 27.90, 21.90, 21.83, 49.27, 33.27,…
ggplot(data = garbage, aes(x = hh_size, y = total)) +
geom_point(alpha = 0.6) +
scale_x_continuous(name = "Household Size", breaks = seq(1, 10)) +
geom_smooth(aes(group = 1),
method = "lm", se = FALSE) +
labs(title = "Household Size vs Total Amount of Waste", y = "Total Amount of Waste")
## `geom_smooth()` using formula = 'y ~ x'
This scatter plot is showing the Household Size of community members on
the X-Axis and the Total Amount of Waste produced by the household on
the Y-Axis. The tread line shows that as the Household Size increases,
the amount of waste produced also increases.
garbage_levels <- setdiff(names(garbage), c("hh_size", "total"))
garbage_long <- garbage %>%
select(-total) %>%
pivot_longer(cols = -hh_size,
names_to = "garbage_type",
values_to = "garbage_weight") %>%
mutate(garbage_type = factor(garbage_type, levels = garbage_levels))
This is cleaning up the data so the Garbage Type is able to be used as a graphable variable.
ggplot(data = garbage_long, aes(x = hh_size, fill = garbage_type)) +
geom_histogram(binwidth = 1, boundary = 0.5, color = "white") +
scale_x_continuous(name = "Household Size", breaks = 1:11) +
scale_fill_brewer(palette = "Set2") +
labs(title = "Total Waste by Household Size and Waste Type", y = "Total Weight", fill = "Waste Type") +
theme_minimal(base_size = 13)
This Histogram is showing the amount of each category of waste produced by each family size. The household size is on the x-axis, and the y-axis is the total weight of each type. The legend on the right identifies the types of waste. I find it interesting that the 2 person household size produces more Waste in the “Other” category than other household sizes.
ggplot(data = garbage_long, aes(x = garbage_type, y = garbage_weight, fill = garbage_type)) +
geom_boxplot(show.legend = FALSE) +
labs(title = "Weight Distribution by Waste Type", x = "Waste Type", y = "Weight") +
theme_minimal(base_size = 13)
This boxplot shows the Weight Distribution by Waste Type. The x-axis displays the type of waste produced and the y-axis displays the weight of the waste. The waste type that weighs the most is Paper and Other has the most outliers.