Nama: Firnanda Dini

NIM: G5401231024

library(tidyverse)
## Warning: package 'tidyverse' was built under R version 4.4.3
## Warning: package 'ggplot2' was built under R version 4.4.3
## Warning: package 'tidyr' was built under R version 4.4.3
## Warning: package 'readr' was built under R version 4.4.3
## Warning: package 'dplyr' was built under R version 4.4.3
## Warning: package 'stringr' was built under R version 4.4.3
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr     1.2.1     ✔ readr     2.1.5
## ✔ forcats   1.0.0     ✔ stringr   1.6.0
## ✔ ggplot2   4.0.3     ✔ tibble    3.2.1
## ✔ lubridate 1.9.4     ✔ tidyr     1.3.2
## ✔ purrr     1.0.2     
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag()    masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(ggridges) 
## Warning: package 'ggridges' was built under R version 4.4.3
library(plotly)
## 
## Attaching package: 'plotly'
## 
## The following object is masked from 'package:ggplot2':
## 
##     last_plot
## 
## The following object is masked from 'package:stats':
## 
##     filter
## 
## The following object is masked from 'package:graphics':
## 
##     layout
library(GGally)
## Warning: package 'GGally' was built under R version 4.4.3
glimpse(mpg)
## Rows: 234
## Columns: 11
## $ manufacturer <chr> "audi", "audi", "audi", "audi", "audi", "audi", "audi", "…
## $ model        <chr> "a4", "a4", "a4", "a4", "a4", "a4", "a4", "a4 quattro", "…
## $ displ        <dbl> 1.8, 1.8, 2.0, 2.0, 2.8, 2.8, 3.1, 1.8, 1.8, 2.0, 2.0, 2.…
## $ year         <int> 1999, 1999, 2008, 2008, 1999, 1999, 2008, 1999, 1999, 200…
## $ cyl          <int> 4, 4, 4, 4, 6, 6, 6, 4, 4, 4, 4, 6, 6, 6, 6, 6, 6, 8, 8, …
## $ trans        <chr> "auto(l5)", "manual(m5)", "manual(m6)", "auto(av)", "auto…
## $ drv          <chr> "f", "f", "f", "f", "f", "f", "f", "4", "4", "4", "4", "4…
## $ cty          <int> 18, 21, 20, 21, 16, 18, 18, 18, 16, 20, 19, 15, 17, 17, 1…
## $ hwy          <int> 29, 29, 31, 30, 26, 26, 27, 26, 25, 28, 27, 25, 25, 25, 2…
## $ fl           <chr> "p", "p", "p", "p", "p", "p", "p", "p", "p", "p", "p", "p…
## $ class        <chr> "compact", "compact", "compact", "compact", "compact", "c…
glimpse(diamonds)
## Rows: 53,940
## Columns: 10
## $ carat   <dbl> 0.23, 0.21, 0.23, 0.29, 0.31, 0.24, 0.24, 0.26, 0.22, 0.23, 0.…
## $ cut     <ord> Ideal, Premium, Good, Premium, Good, Very Good, Very Good, Ver…
## $ color   <ord> E, E, E, I, J, J, I, H, E, H, J, J, F, J, E, E, I, J, J, J, I,…
## $ clarity <ord> SI2, SI1, VS1, VS2, SI2, VVS2, VVS1, SI1, VS2, VS1, SI1, VS1, …
## $ depth   <dbl> 61.5, 59.8, 56.9, 62.4, 63.3, 62.8, 62.3, 61.9, 65.1, 59.4, 64…
## $ table   <dbl> 55, 61, 65, 58, 58, 57, 57, 55, 61, 61, 55, 56, 61, 54, 62, 58…
## $ price   <int> 326, 326, 327, 334, 335, 336, 336, 337, 337, 338, 339, 340, 34…
## $ x       <dbl> 3.95, 3.89, 4.05, 4.20, 4.34, 3.94, 3.95, 4.07, 3.87, 4.00, 4.…
## $ y       <dbl> 3.98, 3.84, 4.07, 4.23, 4.35, 3.96, 3.98, 4.11, 3.78, 4.05, 4.…
## $ z       <dbl> 2.43, 2.31, 2.31, 2.63, 2.75, 2.48, 2.47, 2.53, 2.49, 2.39, 2.…
set.seed(123) 
diamonds_sample <- diamonds |> 
  slice_sample(n = 3000)
glimpse(economics)
## Rows: 574
## Columns: 6
## $ date     <date> 1967-07-01, 1967-08-01, 1967-09-01, 1967-10-01, 1967-11-01, …
## $ pce      <dbl> 506.7, 509.8, 515.6, 512.2, 517.4, 525.1, 530.9, 533.6, 544.3…
## $ pop      <dbl> 198712, 198911, 199113, 199311, 199498, 199657, 199808, 19992…
## $ psavert  <dbl> 12.6, 12.6, 11.9, 12.9, 12.8, 11.8, 11.7, 12.3, 11.7, 12.3, 1…
## $ uempmed  <dbl> 4.5, 4.7, 4.6, 4.9, 4.7, 4.8, 5.1, 4.5, 4.1, 4.6, 4.4, 4.4, 4…
## $ unemploy <dbl> 2944, 2945, 2958, 3143, 3066, 3018, 2878, 3001, 2877, 2709, 2…
us_states <- map_data("state") 
glimpse(us_states)
## Rows: 15,537
## Columns: 6
## $ long      <dbl> -87.46201, -87.48493, -87.52503, -87.53076, -87.57087, -87.5…
## $ lat       <dbl> 30.38968, 30.37249, 30.37249, 30.33239, 30.32665, 30.32665, …
## $ group     <dbl> 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, …
## $ order     <int> 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 1…
## $ region    <chr> "alabama", "alabama", "alabama", "alabama", "alabama", "alab…
## $ subregion <chr> NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, NA, …

1. COMPARISON - MPG

# Calculate average cty and select top 10
mpg_top10 <- mpg %>%
  group_by(manufacturer) %>%
  summarise(mean_cty = mean(cty), .groups = "drop") %>%
  arrange(desc(mean_cty)) %>%
  head(10)

# Create a bar chart for comparison
ggplot(data = mpg_top10, aes(x = reorder(manufacturer, mean_cty), y = mean_cty)) +
  geom_col(fill = "steelblue") +
  coord_flip() +
  labs(
    title = "Top 10 Average City Fuel Efficiency (cty) by Manufacturer",
    x = "Manufacturer",
    y = "Average cty (mpg)"
  ) +
  theme_minimal()

2. DISTRIBUTION- DIAMONDS

# Distribution - diamonds

# Using a Violin Plot combined with a Boxplot for price (numeric) and cut (categorical) variables
ggplot(data = diamonds, aes(x = cut, y = price, fill = cut)) +
  geom_violin(alpha = 0.7, show.legend = FALSE) +
  geom_boxplot(width = 0.1, outlier.shape = NA, alpha = 0.8) +
  labs(
    title = "Distribution of Diamond Prices by Cut Quality",
    x = "Cut Quality",
    y = "Price (USD)"
  ) +
  theme_minimal()

3. RELATIONSHIP - DIAMONDS_SAMPLE

# Relationship - diamonds_sample

# Create a data sample to reduce overplotting
set.seed(123)
diamonds_sample <- diamonds %>% slice_sample(n = 3000)

# Scatter plot of the relationship between carat and price with added clarity aesthetic
ggplot(data = diamonds_sample, aes(x = carat, y = price, color = clarity)) +
  geom_point(alpha = 0.5) +
  geom_smooth(se = FALSE, linewidth = 1, color = "black") +
  labs(
    title = "Relationship Between Diamond Weight and Price",
    x = "Weight (Carat)",
    y = "Price (USD)",
    color = "Clarity"
  ) +
  theme_minimal()
## `geom_smooth()` using method = 'gam' and formula = 'y ~ s(x, bs = "cs")'

4. TIME SERIES - ECONOMICS

# 4. Time Series - economics

# Line chart for psavert time series
ggplot(data = economics, aes(x = date, y = psavert)) +
  geom_line(linewidth = 0.7, color = "darkblue") +
  annotate("rect", xmin = as.Date("2005-01-01"), xmax = as.Date("2010-01-01"),
           ymin = -Inf, ymax = Inf, alpha = 0.2, fill = "red") +
  annotate("text", x = as.Date("2007-06-01"), y = 15, label = "Post-2005 Spike", color = "red", fontface = "bold") +
  labs(
    title = "Personal Saving Rate (psavert) Over Time",
    x = "Time (Year)",
    y = "Personal Saving Rate"
  ) +
  theme_minimal()

5. IMPROVE A VISUALIZATION

# 5. Improve a Visualization

# PROBLEMATIC VISUALIZATION (Should not be executed in final report, for illustration only)
# ggplot(data = mpg, aes(x = displ, y = hwy)) +
#   geom_point(size = 6, color = "yellow") +
#   coord_cartesian(ylim = c(20, 25)) +
#   theme_dark()

# IMPROVED VISUALIZATION
ggplot(data = mpg, aes(x = displ, y = hwy, color = class)) +
  geom_point(alpha = 0.7, size = 3) +
  labs(
    title = "Relationship Between Engine Displacement and Highway Fuel Efficiency by Class",
    x = "Engine Displacement (liters)",
    y = "Highway Efficiency (mpg)",
    color = "Vehicle Class"
  ) +
  theme_minimal()