library(ggplot2)
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(dslabs)

Part A: Constructing plots with the grammar of graphics

Exercise 1: Data, mappings, and geometry

1

str(murders)
## 'data.frame':    51 obs. of  5 variables:
##  $ state     : chr  "Alabama" "Alaska" "Arizona" "Arkansas" ...
##  $ abb       : chr  "AL" "AK" "AZ" "AR" ...
##  $ region    : Factor w/ 4 levels "Northeast","South",..: 2 4 4 2 4 4 1 2 2 2 ...
##  $ population: num  4779736 710231 6392017 2915918 37253956 ...
##  $ total     : num  135 19 232 93 1257 ...
names(murders)
## [1] "state"      "abb"        "region"     "population" "total"

2

p <- ggplot(murders)
print(p)

The plot is black because “p” contains the murders data but there is no geometry telling ggplot2 what to draw.

3

p <- p + geom_point(aes(x = population, y = total))
print(p)

4

The aesthetic mapping places population on the on the x-axis and total murders on the y-axis. geom_point() draws one point for each observation.

Exercise 2: Mapped versus fixed color

library(dslabs)
library(ggplot2)

p <- ggplot(murders, aes(x = population, y = total))

p +
  geom_point(color = "navy") +
  labs(
    x = "Population",
    y = "Total murders",
    title = "Total murder by population"
  )

colorblind_palette <- c(
  "Northeast" = "#E69F00",
  "South" = "#56B4E9",
  "North Central" = "#009E73",
  "West" = "#CC79A7"
)

p +
  geom_point(aes(color = region)) +
  scale_color_manual(values = colorblind_palette) +
  labs(
    x = "Population",
    y = "Total murders",
    color = "Region",
    title = "Total murders by population and region"
  )

The navy plot has no legend because color = “navy” sets one fixed color outside aes(). The second plot has a legend because aes(color = region) maps colors to the region values; scale_color_manual() supplied the colors and labs(color = “Region”) names the legend.

Exercise 3: Labels, scales, and annotations

library(dslabs)
library(ggplot2)

ggplot(murders, aes(x = population, y = total)) +
  geom_point(color = "navy") +
  geom_text(aes(label = abb), vjust = -0.6, size = 3) +
  scale_x_log10() +
  scale_y_log10() +
  labs(
    title = "Total murders and population by state",
    x = "Population (log10 scale)",
    y = "Total murders (log10 scale)",
    caption = "Data source: dslabs"
  )

The log10 scales spread out states with smaller populations and murder totals, making their labels and the relationship between population and total murders easier to see. The axes must be labeled because equal distances on a log10 scale represent equal multiplicative changes, not equal numerical increases. Withoutthose labels, a reader could misinterpret the distances between states.

Exercise 4: Histograms and analytical choices

library(dslabs)
library(ggplot2)

x_limits <- range(heights$height, na.rm = TRUE)

plot_1 <- ggplot(heights, aes(x = height)) +
  geom_histogram(binwidth = 1, fill = "navy", color = "white") +
  coord_cartesian(xlim = x_limits) +
  labs(
    title = "Heights with 1-inch bins",
    x = "Height (inches)",
    y = "Number of people"
  )

plot_3 <- ggplot(heights, aes(x = height)) +
  geom_histogram(binwidth = 3, fill = "navy", color = "white") +
  coord_cartesian(xlim = x_limits) +
  labs(
    title = "Heights with 3-inch bins",
    x = "Height (inches)",
    y = "Number of people"
  )

print(plot_1)

print(plot_3)

The 1-inch bins show more of the individual peaks and drips in reported heights. With 3-inch binds, neighboring heights are more combined, so the distribution looks smoother and some smaller peaks are less noticeable. Bin width is an analytical choise because it changes which patterns a reader can see and how prominent those patterns appear, even though the underlying data is the same.

Exercise 5: Comparing distributions

library(dslabs)
library(ggplot2)

sex_palette <- c(
  "Female" = "#D55E00",
  "Male" = "#0072B2"
)

# Density plots
ggplot(heights, aes(x = height, fill = sex, color = sex)) +
  geom_density(alpha = 0.35) +
  scale_fill_manual(values = sex_palette) +
  scale_color_manual(values = sex_palette) +
  labs(
    title = "Distribution of self-reported heights by sex",
    x = "Height (inches)",
    y = "Density",
    fill = "Sex",
    color = "Sex"
  )

# Boxplots with individual observations
ggplot(heights, aes(x = sex, y = height, fill = sex)) +
  geom_boxplot(alpha = 0.45, outlier.shape = NA) +
  geom_jitter(aes(color = sex),
              width = 0.15, height = 0,
              alpha = 0.25, size = 0.8,
              show.legend = FALSE) +
  scale_fill_manual(values = sex_palette) +
  scale_color_manual(values = sex_palette) +
  labs(
    title = "Self-reported heights by sex",
    x = "Sex",
    y = "Height (inches)",
    fill = "Sex"
  )

The density plot makes it easier to compare where heights are concentrated and how much the two distributions overlap. The boxplots make the medians, middle ranges, and spread easier to compare, while the jittered points show individual observations that the summaries hide. These plots describe differences in the recorded heights; they do not show that sex caused any particular height.

Part B: Applying visualization principles

Exercise 6: Meaningful category order

library(ggplot2)
library(dslabs)
library(dplyr)

measles_1967 <- us_contagious_diseases |>
  filter(
    year == 1967,
    disease == "Measles",
    !is.na(population),
    !is.na(count),
    weeks_reporting > 0
  ) |>
  mutate(
    rate = count / population * 10000 * 52 / weeks_reporting
  )

# Default state order
plot_default <- ggplot(measles_1967, aes(x = rate, y = state)) +
  geom_col(fill = "steelblue") +
  scale_x_continuous(
    limits = c(0, NA),
    expand = expansion(mult = c(0, 0.04))
  ) +
  labs(
    title = "1967 measles rates by state (annualized cases per 10,000 people)",
    x = "Annualized cases per 10,000 people",
    y = "State"
  ) +
  theme_minimal() +
  theme(axis.text.y = element_text(size = 8))

# States ordered by rate
plot_ordered <- ggplot(
  measles_1967,
  aes(x = rate, y = reorder(state, rate))
) +
  geom_col(fill = "steelblue") +
  scale_x_continuous(
    limits = c(0, NA),
    expand = expansion(mult = c(0, 0.04))
  ) +
  labs(
    title = "1967 measles rates by state, ordered by rate",
    subtitle = "Annualized cases per 10,000 people",
    x = "Annualized cases per 10,000 people",
    y = "State"
  ) +
  theme_minimal() +
  theme(axis.text.y = element_text(size = 8))

print(plot_default)

print(plot_ordered)

The ordered chart makes the highest- and lowest-rate states easier to identify because the bars are ranked by rate. In the default-order chart, a reader has to scan across all the state names and bar lengths to find the extremes. Both charts start the rate axis at zero, so bar lengths represent the rates without a truncated baseline.

Exercise 7: Show the data, not only an average

library(dslabs)
library(dplyr)
library(ggplot2)

# 1. Murders per 100,000 residents for each state
murders_rates <- murders |>
  mutate(murder_rate = total / population * 100000)

# 2. One mean rate per region
region_means <- murders_rates |>
  group_by(region) |>
  summarise(mean_rate = mean(murder_rate), .groups = "drop")

ggplot(region_means, aes(x = region, y = mean_rate)) +
  geom_col(fill = "navy") +
  labs(
    title = "Mean state murder rate by region",
    x = "Region",
    y = "Mean murders per 100,000 residents"
  )

# 3. Regional distributions, ordered by median
ggplot(
  murders_rates,
  aes(x = reorder(region, murder_rate, FUN = median),
      y = murder_rate)
) +
  geom_boxplot(fill = "lightblue", outlier.shape = NA) +
  geom_jitter(width = 0.15, height = 0,
              color = "navy", alpha = 0.7) +
  labs(
    title = "State murder rates by region",
    x = "Region (ordered by median rate)",
    y = "Murders per 100,000 residents"
  )

A regional mean reduces all the state rates in a region to one number. The boxplot and state points show how much rates vary within each region and whether rates from different regions overlap. Because a region with a higher mean can still contain states with lower rates than states in another region, the means alone provide weak support for choosing one region as uniformly safer or riskier.

Exercise 8: Visualization audit and redesign

ggplot(
  murders_rates,
  aes(
    x = reorder(region, murder_rate, FUN = median),
    y = murder_rate
  )
) +
  geom_boxplot(fill = "lightgray", outlier.shape = NA) +
  geom_jitter(
    width = 0.12, height = 0,
    color = "navy", alpha = 0.7, size = 2
  ) +
  scale_y_continuous(
    limits = c(0, NA),
    expand = expansion(mult = c(0, 0.05))
  ) +
  labs(
    title = "State murder rates vary within U.S. regions",
    subtitle = "Each point represents a state or Washington, DC; regions are ordered by median",
    x = "Region",
    y = "Murders per 100,000 residents",
    caption = "Source: dslabs murders; rate = total murders / population × 100,000"
  )

I replaced the mean-only bars with boxplots and individual state points because the analytical question concerns regional differences in state rates. The revision shows within-region variability and overlap that the four means conceal. It also orders the regions by median and identifies the unit and source, so readers can interpret the comparison without relying on color.

Exercise 9: Sensitivity of distribution displays

library(dslabs)
library(ggplot2)

adjustments <- c(0.5, 1, 2)

# Use identical horizontal and vertical limits on all three plots
x_limits <- range(heights$height, na.rm = TRUE)
y_max <- max(sapply(adjustments, function(a) {
  max(density(heights$height, adjust = a, na.rm = TRUE)$y)
})) * 1.05

make_plot <- function(a) {
  ggplot(heights, aes(x = height)) +
    geom_density(adjust = a, fill = "navy", alpha = 0.35) +
    coord_cartesian(xlim = x_limits, ylim = c(0, y_max)) +
    labs(
      title = paste("Height distribution: bandwidth adjustment", a),
      x = "Height (inches)",
      y = "Density"
    )
}

print(make_plot(0.5))

print(make_plot(1))

print(make_plot(2))

Across all three displays, heights remain concentrated in the middle of the observed range, with fewer observations toward the extremes. Small bumps are more prominent with adjust = 0.5 and may disappear with adjust = 2, so I would not treat every small peak as a distinct group. For exploratory analysis, I would use adjust = 0.5 to notice patterns worth checking further. For a general audience, I would use adjust = 1 because it shows the broad shape while reducing the risk that small fluctuations will be overinterpreted.

Exercise 10: Redesign under competing goals

library(dslabs)
library(ggplot2)

median_height <- median(heights$height, na.rm = TRUE)
x_limits <- c(
  floor(min(heights$height, na.rm = TRUE) / 3) * 3 - 3,
  ceiling(max(heights$height, na.rm = TRUE) / 3) * 3 + 3
)

# Exploratory: more detail for a data analyst
ggplot(heights, aes(x = height)) +
  geom_density(adjust = 0.5, fill = "navy", alpha = 0.25) +
  geom_rug(alpha = 0.15) +
  geom_vline(xintercept = median_height, linetype = "dashed") +
  coord_cartesian(xlim = x_limits) +
  labs(
    title = "Exploring the distribution of self-reported heights",
    subtitle = "Narrower smoothing; rug marks individual observations",
    x = "Height (inches)",
    y = "Density",
    caption = "Data source: dslabs heights; dashed line shows the median"
  )

# Communication: simpler display for a general audience
ggplot(heights, aes(x = height)) +
  geom_histogram(binwidth = 3, boundary = 0,
                 fill = "navy", color = "white") +
  geom_vline(xintercept = median_height, linetype = "dashed") +
  scale_y_continuous(expand = expansion(mult = c(0, 0.05))) +
  coord_cartesian(xlim = x_limits) +
  labs(
    title = "Most reported heights fall near the middle of the range",
    subtitle = paste("Three-inch groups; median:",
                     round(median_height, 1), "inches"),
    x = "Height (inches)",
    y = "Number of people",
    caption = "Data source: dslabs heights; dashed line shows the median"
  )

Both figures use the same observations, height units, and median. The exploratory figure preserves more detail through lighter smoothing and marks for individual observations, helping an analyst spot small peaks worth investigating. The communication figure groups heights into labeled three-inch bins and uses counts, which are easier for a general audience to read. It simplifies small fluctuations but keeps a zero baseline for bar lengths and states the bin width, so readers can see how the data were grouped. The small peaks in the exploratory display should not be interpreted as separate groups without further analysis.