Nama: Firnanda Dini
NIM: G5401231024
library(tidyverse)
## Warning: package 'tidyverse' was built under R version 4.4.3
## Warning: package 'ggplot2' was built under R version 4.4.3
## Warning: package 'tidyr' was built under R version 4.4.3
## Warning: package 'readr' was built under R version 4.4.3
## Warning: package 'dplyr' was built under R version 4.4.3
## Warning: package 'stringr' was built under R version 4.4.3
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.1.5
## ✔ forcats 1.0.0 ✔ stringr 1.6.0
## ✔ ggplot2 4.0.3 ✔ tibble 3.2.1
## ✔ lubridate 1.9.4 ✔ tidyr 1.3.2
## ✔ purrr 1.0.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(ggridges)
## Warning: package 'ggridges' was built under R version 4.4.3
library(plotly)
##
## Attaching package: 'plotly'
##
## The following object is masked from 'package:ggplot2':
##
## last_plot
##
## The following object is masked from 'package:stats':
##
## filter
##
## The following object is masked from 'package:graphics':
##
## layout
library(GGally)
## Warning: package 'GGally' was built under R version 4.4.3
glimpse(mpg)
## Rows: 234
## Columns: 11
## $ manufacturer <chr> "audi", "audi", "audi", "audi", "audi", "audi", "audi", "…
## $ model <chr> "a4", "a4", "a4", "a4", "a4", "a4", "a4", "a4 quattro", "…
## $ displ <dbl> 1.8, 1.8, 2.0, 2.0, 2.8, 2.8, 3.1, 1.8, 1.8, 2.0, 2.0, 2.…
## $ year <int> 1999, 1999, 2008, 2008, 1999, 1999, 2008, 1999, 1999, 200…
## $ cyl <int> 4, 4, 4, 4, 6, 6, 6, 4, 4, 4, 4, 6, 6, 6, 6, 6, 6, 8, 8, …
## $ trans <chr> "auto(l5)", "manual(m5)", "manual(m6)", "auto(av)", "auto…
## $ drv <chr> "f", "f", "f", "f", "f", "f", "f", "4", "4", "4", "4", "4…
## $ cty <int> 18, 21, 20, 21, 16, 18, 18, 18, 16, 20, 19, 15, 17, 17, 1…
## $ hwy <int> 29, 29, 31, 30, 26, 26, 27, 26, 25, 28, 27, 25, 25, 25, 2…
## $ fl <chr> "p", "p", "p", "p", "p", "p", "p", "p", "p", "p", "p", "p…
## $ class <chr> "compact", "compact", "compact", "compact", "compact", "c…
glimpse(diamonds)
## Rows: 53,940
## Columns: 10
## $ carat <dbl> 0.23, 0.21, 0.23, 0.29, 0.31, 0.24, 0.24, 0.26, 0.22, 0.23, 0.…
## $ cut <ord> Ideal, Premium, Good, Premium, Good, Very Good, Very Good, Ver…
## $ color <ord> E, E, E, I, J, J, I, H, E, H, J, J, F, J, E, E, I, J, J, J, I,…
## $ clarity <ord> SI2, SI1, VS1, VS2, SI2, VVS2, VVS1, SI1, VS2, VS1, SI1, VS1, …
## $ depth <dbl> 61.5, 59.8, 56.9, 62.4, 63.3, 62.8, 62.3, 61.9, 65.1, 59.4, 64…
## $ table <dbl> 55, 61, 65, 58, 58, 57, 57, 55, 61, 61, 55, 56, 61, 54, 62, 58…
## $ price <int> 326, 326, 327, 334, 335, 336, 336, 337, 337, 338, 339, 340, 34…
## $ x <dbl> 3.95, 3.89, 4.05, 4.20, 4.34, 3.94, 3.95, 4.07, 3.87, 4.00, 4.…
## $ y <dbl> 3.98, 3.84, 4.07, 4.23, 4.35, 3.96, 3.98, 4.11, 3.78, 4.05, 4.…
## $ z <dbl> 2.43, 2.31, 2.31, 2.63, 2.75, 2.48, 2.47, 2.53, 2.49, 2.39, 2.…
set.seed(123)
diamonds_sample <- diamonds |>
slice_sample(n = 3000)
glimpse(economics)
## Rows: 574
## Columns: 6
## $ date <date> 1967-07-01, 1967-08-01, 1967-09-01, 1967-10-01, 1967-11-01, …
## $ pce <dbl> 506.7, 509.8, 515.6, 512.2, 517.4, 525.1, 530.9, 533.6, 544.3…
## $ pop <dbl> 198712, 198911, 199113, 199311, 199498, 199657, 199808, 19992…
## $ psavert <dbl> 12.6, 12.6, 11.9, 12.9, 12.8, 11.8, 11.7, 12.3, 11.7, 12.3, 1…
## $ uempmed <dbl> 4.5, 4.7, 4.6, 4.9, 4.7, 4.8, 5.1, 4.5, 4.1, 4.6, 4.4, 4.4, 4…
## $ unemploy <dbl> 2944, 2945, 2958, 3143, 3066, 3018, 2878, 3001, 2877, 2709, 2…
us_states <- map_data("state")
glimpse(us_states)
## Rows: 15,537
## Columns: 6
## $ long <dbl> -87.46201, -87.48493, -87.52503, -87.53076, -87.57087, -87.5…
## $ lat <dbl> 30.38968, 30.37249, 30.37249, 30.33239, 30.32665, 30.32665, …
## $ group <dbl> 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, …
## $ order <int> 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 1…
## $ region <chr> "alabama", "alabama", "alabama", "alabama", "alabama", "alab…
## $ subregion <chr> NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, …
# Calculate average cty and select top 10
mpg_top10 <- mpg %>%
group_by(manufacturer) %>%
summarise(mean_cty = mean(cty), .groups = "drop") %>%
arrange(desc(mean_cty)) %>%
head(10)
# Create a bar chart for comparison
ggplot(data = mpg_top10, aes(x = reorder(manufacturer, mean_cty), y = mean_cty)) +
geom_col(fill = "steelblue") +
coord_flip() +
labs(
title = "Top 10 Average City Fuel Efficiency (cty) by Manufacturer",
x = "Manufacturer",
y = "Average cty (mpg)"
) +
theme_minimal()
# Distribution - diamonds
# Using a Violin Plot combined with a Boxplot for price (numeric) and cut (categorical) variables
ggplot(data = diamonds, aes(x = cut, y = price, fill = cut)) +
geom_violin(alpha = 0.7, show.legend = FALSE) +
geom_boxplot(width = 0.1, outlier.shape = NA, alpha = 0.8) +
labs(
title = "Distribution of Diamond Prices by Cut Quality",
x = "Cut Quality",
y = "Price (USD)"
) +
theme_minimal()
# Relationship - diamonds_sample
# Create a data sample to reduce overplotting
set.seed(123)
diamonds_sample <- diamonds %>% slice_sample(n = 3000)
# Scatter plot of the relationship between carat and price with added clarity aesthetic
ggplot(data = diamonds_sample, aes(x = carat, y = price, color = clarity)) +
geom_point(alpha = 0.5) +
geom_smooth(se = FALSE, linewidth = 1, color = "black") +
labs(
title = "Relationship Between Diamond Weight and Price",
x = "Weight (Carat)",
y = "Price (USD)",
color = "Clarity"
) +
theme_minimal()
## `geom_smooth()` using method = 'gam' and formula = 'y ~ s(x, bs = "cs")'
# 4. Time Series - economics
# Line chart for psavert time series
ggplot(data = economics, aes(x = date, y = psavert)) +
geom_line(linewidth = 0.7, color = "darkblue") +
annotate("rect", xmin = as.Date("2005-01-01"), xmax = as.Date("2010-01-01"),
ymin = -Inf, ymax = Inf, alpha = 0.2, fill = "red") +
annotate("text", x = as.Date("2007-06-01"), y = 15, label = "Post-2005 Spike", color = "red", fontface = "bold") +
labs(
title = "Personal Saving Rate (psavert) Over Time",
x = "Time (Year)",
y = "Personal Saving Rate"
) +
theme_minimal()
# 5. Improve a Visualization
# PROBLEMATIC VISUALIZATION (Should not be executed in final report, for illustration only)
# ggplot(data = mpg, aes(x = displ, y = hwy)) +
# geom_point(size = 6, color = "yellow") +
# coord_cartesian(ylim = c(20, 25)) +
# theme_dark()
# IMPROVED VISUALIZATION
ggplot(data = mpg, aes(x = displ, y = hwy, color = class)) +
geom_point(alpha = 0.7, size = 3) +
labs(
title = "Relationship Between Engine Displacement and Highway Fuel Efficiency by Class",
x = "Engine Displacement (liters)",
y = "Highway Efficiency (mpg)",
color = "Vehicle Class"
) +
theme_minimal()