R Markdown

This is an R Markdown document. Markdown is a simple formatting syntax for authoring HTML, PDF, and MS Word documents. For more details on using R Markdown see http://rmarkdown.rstudio.com.

When you click the Knit button a document will be generated that includes both content as well as the output of any embedded R code chunks within the document. You can embed an R code chunk like this:

library('tidyr')
library('readr')
library('dplyr')
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library('ggplot2')
library('forcats')
library('tidyquant')
## Loading required package: lubridate
## 
## Attaching package: 'lubridate'
## The following objects are masked from 'package:base':
## 
##     date, intersect, setdiff, union
## Loading required package: PerformanceAnalytics
## Loading required package: xts
## Loading required package: zoo
## 
## Attaching package: 'zoo'
## The following objects are masked from 'package:base':
## 
##     as.Date, as.Date.numeric
## 
## ################################### WARNING ###################################
## # We noticed you have dplyr installed. The dplyr lag() function breaks how    #
## # base R's lag() function is supposed to work, which breaks lag(my_xts).      #
## #                                                                             #
## # Calls to lag(my_xts) that you enter or source() into this session won't     #
## # work correctly.                                                             #
## #                                                                             #
## # All package code is unaffected because it is protected by the R namespace   #
## # mechanism.                                                                  #
## #                                                                             #
## # Set `options(xts.warn_dplyr_breaks_lag = FALSE)` to suppress this warning.  #
## #                                                                             #
## # You can use stats::lag() to make sure you're not using dplyr::lag(), or you #
## # can add conflictRules('dplyr', exclude = 'lag') to your .Rprofile to stop   #
## # dplyr from breaking base R's lag() function.                                #
## ################################### WARNING ###################################
## 
## Attaching package: 'xts'
## The following objects are masked from 'package:dplyr':
## 
##     first, last
## 
## Attaching package: 'PerformanceAnalytics'
## The following object is masked from 'package:graphics':
## 
##     legend
## Loading required package: quantmod
## Loading required package: TTR
## Registered S3 method overwritten by 'quantmod':
##   method            from
##   as.zoo.data.frame zoo
setwd("C:/Users/jerem/Documents/Financial Database")
survey_data <- read.csv("multipleChoiceResponses.csv", header = TRUE, stringsAsFactors = FALSE)

selected_columns <- survey_data %>%
  select(starts_with("Learning"), starts_with("Working"), Age, EmployerIndustry, CurrentJobTitleSelect, starts_with("MLMethod"), starts_with("Formal"))
# Display the result using glimpse()
glimpse(selected_columns)
## Rows: 16,716
## Columns: 32
## $ LearningDataScience                     <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformSelect                  <chr> "College/University,Conference…
## $ LearningPlatformUsefulnessArxiv         <chr> "", "", "Very useful", "", "Ve…
## $ LearningPlatformUsefulnessBlogs         <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessCollege       <chr> "", "", "Somewhat useful", "Ve…
## $ LearningPlatformUsefulnessCompany       <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessConferences   <chr> "Very useful", "", "", "Very u…
## $ LearningPlatformUsefulnessFriends       <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessKaggle        <chr> "", "Somewhat useful", "Somewh…
## $ LearningPlatformUsefulnessNewsletters   <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessCommunities   <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessDocumentation <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessCourses       <chr> "", "", "Very useful", "Very u…
## $ LearningPlatformUsefulnessProjects      <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessPodcasts      <chr> "Very useful", "", "", "", "",…
## $ LearningPlatformUsefulnessSO            <chr> "", "", "", "", "", "Very usef…
## $ LearningPlatformUsefulnessTextbook      <chr> "", "", "", "", "Somewhat usef…
## $ LearningPlatformUsefulnessTradeBook     <chr> "Somewhat useful", "", "", "",…
## $ LearningPlatformUsefulnessTutoring      <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessYouTube       <chr> "", "", "Very useful", "", "",…
## $ LearningDataScienceTime                 <chr> "", "1-2 years", "1-2 years", …
## $ LearningCategorySelftTaught             <dbl> 0, 10, 20, 30, 60, 45, 40, 0, …
## $ LearningCategoryOnlineCourses           <dbl> 0, 30, 50, 0, 5, 25, 0, 40, 0,…
## $ LearningCategoryWork                    <dbl> 100, 0, 0, 40, 5, 20, 0, 0, 30…
## $ LearningCategoryUniversity              <dbl> 0, 30, 30, 30, 30, 0, 50, 50, …
## $ LearningCategoryKaggle                  <dbl> 0, 30, 0, 0, 0, 10, 10, 10, 0,…
## $ LearningCategoryOther                   <dbl> 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, …
## $ Age                                     <int> NA, 30, 28, 56, 38, 46, 35, 22…
## $ EmployerIndustry                        <chr> "Internet-based", "", "", "Mix…
## $ CurrentJobTitleSelect                   <chr> "DBA/Database Engineer", "", "…
## $ MLMethodNextYearSelect                  <chr> "Random Forests", "Random Fore…
## $ FormalEducation                         <chr> "Bachelor's degree", "Master's…
# 1.2 Change character columns to factors and find number of levels
factor_columns <- selected_columns %>%
  mutate(across(where(is.character), as.factor)) %>%
  summarise_all(nlevels)

# Display the result
print(factor_columns)
##   LearningDataScience LearningPlatformSelect LearningPlatformUsefulnessArxiv
## 1                   4                   5363                               4
##   LearningPlatformUsefulnessBlogs LearningPlatformUsefulnessCollege
## 1                               4                                 4
##   LearningPlatformUsefulnessCompany LearningPlatformUsefulnessConferences
## 1                                 4                                     4
##   LearningPlatformUsefulnessFriends LearningPlatformUsefulnessKaggle
## 1                                 4                                4
##   LearningPlatformUsefulnessNewsletters LearningPlatformUsefulnessCommunities
## 1                                     4                                     4
##   LearningPlatformUsefulnessDocumentation LearningPlatformUsefulnessCourses
## 1                                       4                                 4
##   LearningPlatformUsefulnessProjects LearningPlatformUsefulnessPodcasts
## 1                                  4                                  4
##   LearningPlatformUsefulnessSO LearningPlatformUsefulnessTextbook
## 1                            4                                  4
##   LearningPlatformUsefulnessTradeBook LearningPlatformUsefulnessTutoring
## 1                                   4                                  4
##   LearningPlatformUsefulnessYouTube LearningDataScienceTime
## 1                                 4                       7
##   LearningCategorySelftTaught LearningCategoryOnlineCourses
## 1                           0                             0
##   LearningCategoryWork LearningCategoryUniversity LearningCategoryKaggle
## 1                    0                          0                      0
##   LearningCategoryOther Age EmployerIndustry CurrentJobTitleSelect
## 1                     0   0               17                    17
##   MLMethodNextYearSelect FormalEducation
## 1                     26               8
# 1.3 Select 5 rows with the highest number of levels
top_levels <- factor_columns %>%
  gather(variable, num_levels) %>%
  arrange(desc(num_levels)) %>%
  slice_head(n = 5)

# Display the result
print(top_levels)
##                 variable num_levels
## 1 LearningPlatformSelect       5363
## 2 MLMethodNextYearSelect         26
## 3       EmployerIndustry         17
## 4  CurrentJobTitleSelect         17
## 5        FormalEducation          8
# 1.4 Filter for where the column called variable equals CurrentJobTitleSelect. Show its levels and number of levels.
current_job_levels <- survey_data %>%
  filter(!is.na(CurrentJobTitleSelect)) %>%
  summarise(levels = levels(factor(CurrentJobTitleSelect)), num_levels = nlevels(factor(CurrentJobTitleSelect)))
## Warning: Returning more (or less) than 1 row per `summarise()` group was deprecated in
## dplyr 1.1.0.
## ℹ Please use `reframe()` instead.
## ℹ When switching from `summarise()` to `reframe()`, remember that `reframe()`
##   always returns an ungrouped data frame and adjust accordingly.
# Display the result
print(current_job_levels)
##                                  levels num_levels
## 1                                               17
## 2                      Business Analyst         17
## 3                    Computer Scientist         17
## 4                          Data Analyst         17
## 5                            Data Miner         17
## 6                        Data Scientist         17
## 7                 DBA/Database Engineer         17
## 8                              Engineer         17
## 9             Machine Learning Engineer         17
## 10     Operations Research Practitioner         17
## 11                                Other         17
## 12                   Predictive Modeler         17
## 13                           Programmer         17
## 14                           Researcher         17
## 15                 Scientist/Researcher         17
## 16 Software Developer/Software Engineer         17
## 17                         Statistician         17
# 2.1 Flip the coordinates
ggplot(survey_data, aes(x = EmployerIndustry)) +
  geom_bar() +
  coord_flip()

# 2.2 Filter rows where Age and EmployerIndustry are not NA and replot
ggplot(survey_data %>% filter(!is.na(Age) & !is.na(EmployerIndustry)), aes(x = EmployerIndustry)) +
  geom_bar() +
  coord_flip()

# 2.3 Use geom_segment to plot and arrange data in descending order
age_means <- survey_data %>%
  group_by(EmployerIndustry) %>%
  summarise(mean_age = mean(Age, na.rm = TRUE)) %>%
  arrange(desc(mean_age))

ggplot(age_means, aes(x = reorder(EmployerIndustry, -mean_age), y = mean_age)) +
  geom_bar(stat = "identity") +
  coord_flip()

# 3. Get the levels of WorkInternalVsExternalTools and generate the plot
work_levels <- levels(survey_data$WorkInternalVsExternalTools)

# Manually change the order
custom_order <- c(
  "Entirely internal",
  "More internal than external",
  "Approximately half internal and half external",
  "More external than internal",
  "Entirely external",
  "Do not know"
)

# Change the order of levels
survey_data$WorkInternalVsExternalTools <- factor(
  survey_data$WorkInternalVsExternalTools,
  levels = custom_order
)

# Generate the plot
ggplot(survey_data, aes(x = WorkInternalVsExternalTools)) +
  geom_bar(stat = "count") +
  labs(title = "WorkInternalVsExternalTools Levels")

Including Plots

You can also embed plots, for example:

Note that the echo = FALSE parameter was added to the code chunk to prevent printing of the R code that generated the plot.