# Load the dplyr package
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
# Load the dataset
df <- read.csv('Netflix.csv')
# Split the cast column into multiple rows using unnest
df <- df %>%
mutate(actor = strsplit(as.character(cast), ',')) %>%
tidyr::unnest(actor)
# Drop the original 'cast' column
df <- df %>%
select(-cast)
# Filter for TV shows and count actor appearances
tv_show_actors <- df %>%
filter(type == "TV Show") %>%
count(actor, sort = TRUE) %>%
head(6)
# Display the top six actors with the most appearances in TV shows
print(tv_show_actors)
## # A tibble: 6 × 2
## actor n
## <chr> <int>
## 1 " Takahiro Sakurai" 18
## 2 " Yuki Kaji" 14
## 3 "David Attenborough" 14
## 4 " Tomokazu Sugita" 12
## 5 " Ai Kayano" 11
## 6 " Daisuke Ono" 11