influencers <- read.csv("social media influencers - Tiktok sep 2022.csv")

head(influencers)
##   S.no  Tiktoker.name  Tiktok.name Subscribers Views.avg. Likes.avg.
## 1    1  jypestraykids   Stray Kids       13.8M       6.4M       2.3M
## 2    2     khaby.lame Khabane lame      149.2M      17.3M       2.3M
## 3    3 scarlettsspam2     scarlett        2.1M      17.9M     845.8K
## 4    4      addisonre  Addison Rae       88.7M        22M     906.6K
## 5    5     belindatok      Belinda        4.8M      14.2M       1.5M
## 6    6    onwardwanna      Wanna🥊        7.5M        12M         2M
##   Comments.avg. Shares.avg.
## 1         50.2K       34.2K
## 2         15.2K        8.7K
## 3         53.9K        6.3K
## 4          7.6K       26.2K
## 5         14.5K       15.3K
## 6         20.4K        4.2K
dim(influencers)
## [1] 1000    8
names(influencers)
## [1] "S.no"          "Tiktoker.name" "Tiktok.name"   "Subscribers"  
## [5] "Views.avg."    "Likes.avg."    "Comments.avg." "Shares.avg."
set.seed(123)

influencers_sample <- influencers[
  sample(1:nrow(influencers), 500),
]

dim(influencers_sample)
## [1] 500   8
convert_number <- function(x) {
  x <- gsub(",", "", x)
  multiplier <- ifelse(grepl("M$", x), 1000000,
                 ifelse(grepl("K$", x), 1000,
                 ifelse(grepl("B$", x), 1000000000, 1)))
  number <- as.numeric(gsub("[MK B]", "", x))
  number * multiplier
}

influencers_sample$Subscribers_num <-
  convert_number(influencers_sample$Subscribers)

influencers_sample$Views_num <-
  convert_number(influencers_sample$Views.avg.)

influencers_sample$Likes_num <-
  convert_number(influencers_sample$Likes.avg.)

Exploratory Data Analysis 1

summary(influencers_sample)
##       S.no          Tiktoker.name    Tiktok.name     Subscribers 
##  Min.   :   5.0   Length   :500   Length   :500   Length   :500  
##  1st Qu.: 239.8   N.unique :498   N.unique :497   N.unique :262  
##  Median : 486.5   N.blank  :  0   N.blank  :  0   N.blank  :  0  
##  Mean   : 496.2   Min.nchar:  3   Min.nchar:  2   Min.nchar:  2  
##  3rd Qu.: 751.2   Max.nchar: 24   Max.nchar: 29   Max.nchar:  6  
##  Max.   :1000.0                                                  
##      Views.avg.      Likes.avg.    Comments.avg.    Shares.avg. 
##  Length   :500   Length   :500   Length   :500   Length   :500  
##  N.unique :103   N.unique :469   N.unique :191   N.unique :235  
##  N.blank  :  0   N.blank  :  0   N.blank  :  0   N.blank  :  0  
##  Min.nchar:  2   Min.nchar:  2   Min.nchar:  2   Min.nchar:  2  
##  Max.nchar:  6   Max.nchar:  6   Max.nchar:  5   Max.nchar:  5  
##                                                                 
##  Subscribers_num       Views_num          Likes_num      
##  Min.   :     6300   Min.   :  503800   Min.   :  17600  
##  1st Qu.:  1275000   1st Qu.: 1600000   1st Qu.: 188175  
##  Median :  3200000   Median : 2200000   Median : 277300  
##  Mean   :  6938826   Mean   : 2856155   Mean   : 351867  
##  3rd Qu.:  7800000   3rd Qu.: 3500000   3rd Qu.: 399725  
##  Max.   :146200000   Max.   :16200000   Max.   :2700000

Exploratory Data Analysis 2

library(ggplot2)

ggplot(influencers_sample,
       aes(x = Subscribers_num, y = Views_num)) +
  geom_point(alpha = 0.5) +
  labs(
    title = "Relationship Between Subscribers and Views",
    x = "Subscribers",
    y = "Average Views"
  ) +
  theme_minimal()

What I Learned

  1. I was able to figure out the range and average values of TikTok subscribers, views, and likes in my sample by looking at the summary statistics.

  2. I was able to analyze the connection between the average number of views and the number of subscribers using the scatter plot, as well as whether the points had a unique pattern or major variance.