This is an R Markdown document. Markdown is a simple formatting syntax for authoring HTML, PDF, and MS Word documents. For more details on using R Markdown see http://rmarkdown.rstudio.com.
When you click the Knit button a document will be generated that includes both content as well as the output of any embedded R code chunks within the document. You can embed an R code chunk like this:
library('tidyr')
library('readr')
library('dplyr')
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library('ggplot2')
library('forcats')
library('tidyquant')
## Loading required package: lubridate
##
## Attaching package: 'lubridate'
## The following objects are masked from 'package:base':
##
## date, intersect, setdiff, union
## Loading required package: PerformanceAnalytics
## Loading required package: xts
## Loading required package: zoo
##
## Attaching package: 'zoo'
## The following objects are masked from 'package:base':
##
## as.Date, as.Date.numeric
##
## ################################### WARNING ###################################
## # We noticed you have dplyr installed. The dplyr lag() function breaks how #
## # base R's lag() function is supposed to work, which breaks lag(my_xts). #
## # #
## # Calls to lag(my_xts) that you enter or source() into this session won't #
## # work correctly. #
## # #
## # All package code is unaffected because it is protected by the R namespace #
## # mechanism. #
## # #
## # Set `options(xts.warn_dplyr_breaks_lag = FALSE)` to suppress this warning. #
## # #
## # You can use stats::lag() to make sure you're not using dplyr::lag(), or you #
## # can add conflictRules('dplyr', exclude = 'lag') to your .Rprofile to stop #
## # dplyr from breaking base R's lag() function. #
## ################################### WARNING ###################################
##
## Attaching package: 'xts'
## The following objects are masked from 'package:dplyr':
##
## first, last
##
## Attaching package: 'PerformanceAnalytics'
## The following object is masked from 'package:graphics':
##
## legend
## Loading required package: quantmod
## Loading required package: TTR
## Registered S3 method overwritten by 'quantmod':
## method from
## as.zoo.data.frame zoo
setwd("C:/Users/jerem/Documents/Financial Database")
survey_data <- read.csv("multipleChoiceResponses.csv", header = TRUE, stringsAsFactors = FALSE)
selected_columns <- survey_data %>%
select(starts_with("Learning"), starts_with("Working"), Age, EmployerIndustry, CurrentJobTitleSelect, starts_with("MLMethod"), starts_with("Formal"))
# Display the result using glimpse()
glimpse(selected_columns)
## Rows: 16,716
## Columns: 32
## $ LearningDataScience <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformSelect <chr> "College/University,Conference…
## $ LearningPlatformUsefulnessArxiv <chr> "", "", "Very useful", "", "Ve…
## $ LearningPlatformUsefulnessBlogs <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessCollege <chr> "", "", "Somewhat useful", "Ve…
## $ LearningPlatformUsefulnessCompany <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessConferences <chr> "Very useful", "", "", "Very u…
## $ LearningPlatformUsefulnessFriends <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessKaggle <chr> "", "Somewhat useful", "Somewh…
## $ LearningPlatformUsefulnessNewsletters <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessCommunities <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessDocumentation <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessCourses <chr> "", "", "Very useful", "Very u…
## $ LearningPlatformUsefulnessProjects <chr> "", "", "", "Very useful", "",…
## $ LearningPlatformUsefulnessPodcasts <chr> "Very useful", "", "", "", "",…
## $ LearningPlatformUsefulnessSO <chr> "", "", "", "", "", "Very usef…
## $ LearningPlatformUsefulnessTextbook <chr> "", "", "", "", "Somewhat usef…
## $ LearningPlatformUsefulnessTradeBook <chr> "Somewhat useful", "", "", "",…
## $ LearningPlatformUsefulnessTutoring <chr> "", "", "", "", "", "", "", ""…
## $ LearningPlatformUsefulnessYouTube <chr> "", "", "Very useful", "", "",…
## $ LearningDataScienceTime <chr> "", "1-2 years", "1-2 years", …
## $ LearningCategorySelftTaught <dbl> 0, 10, 20, 30, 60, 45, 40, 0, …
## $ LearningCategoryOnlineCourses <dbl> 0, 30, 50, 0, 5, 25, 0, 40, 0,…
## $ LearningCategoryWork <dbl> 100, 0, 0, 40, 5, 20, 0, 0, 30…
## $ LearningCategoryUniversity <dbl> 0, 30, 30, 30, 30, 0, 50, 50, …
## $ LearningCategoryKaggle <dbl> 0, 30, 0, 0, 0, 10, 10, 10, 0,…
## $ LearningCategoryOther <dbl> 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, …
## $ Age <int> NA, 30, 28, 56, 38, 46, 35, 22…
## $ EmployerIndustry <chr> "Internet-based", "", "", "Mix…
## $ CurrentJobTitleSelect <chr> "DBA/Database Engineer", "", "…
## $ MLMethodNextYearSelect <chr> "Random Forests", "Random Fore…
## $ FormalEducation <chr> "Bachelor's degree", "Master's…
# 1.2 Change character columns to factors and find number of levels
factor_columns <- selected_columns %>%
mutate(across(where(is.character), as.factor)) %>%
summarise_all(nlevels)
# Display the result
print(factor_columns)
## LearningDataScience LearningPlatformSelect LearningPlatformUsefulnessArxiv
## 1 4 5363 4
## LearningPlatformUsefulnessBlogs LearningPlatformUsefulnessCollege
## 1 4 4
## LearningPlatformUsefulnessCompany LearningPlatformUsefulnessConferences
## 1 4 4
## LearningPlatformUsefulnessFriends LearningPlatformUsefulnessKaggle
## 1 4 4
## LearningPlatformUsefulnessNewsletters LearningPlatformUsefulnessCommunities
## 1 4 4
## LearningPlatformUsefulnessDocumentation LearningPlatformUsefulnessCourses
## 1 4 4
## LearningPlatformUsefulnessProjects LearningPlatformUsefulnessPodcasts
## 1 4 4
## LearningPlatformUsefulnessSO LearningPlatformUsefulnessTextbook
## 1 4 4
## LearningPlatformUsefulnessTradeBook LearningPlatformUsefulnessTutoring
## 1 4 4
## LearningPlatformUsefulnessYouTube LearningDataScienceTime
## 1 4 7
## LearningCategorySelftTaught LearningCategoryOnlineCourses
## 1 0 0
## LearningCategoryWork LearningCategoryUniversity LearningCategoryKaggle
## 1 0 0 0
## LearningCategoryOther Age EmployerIndustry CurrentJobTitleSelect
## 1 0 0 17 17
## MLMethodNextYearSelect FormalEducation
## 1 26 8
# 1.3 Select 5 rows with the highest number of levels
top_levels <- factor_columns %>%
gather(variable, num_levels) %>%
arrange(desc(num_levels)) %>%
slice_head(n = 5)
# Display the result
print(top_levels)
## variable num_levels
## 1 LearningPlatformSelect 5363
## 2 MLMethodNextYearSelect 26
## 3 EmployerIndustry 17
## 4 CurrentJobTitleSelect 17
## 5 FormalEducation 8
# 1.4 Filter for where the column called variable equals CurrentJobTitleSelect. Show its levels and number of levels.
current_job_levels <- survey_data %>%
filter(!is.na(CurrentJobTitleSelect)) %>%
summarise(levels = levels(factor(CurrentJobTitleSelect)), num_levels = nlevels(factor(CurrentJobTitleSelect)))
## Warning: Returning more (or less) than 1 row per `summarise()` group was deprecated in
## dplyr 1.1.0.
## ℹ Please use `reframe()` instead.
## ℹ When switching from `summarise()` to `reframe()`, remember that `reframe()`
## always returns an ungrouped data frame and adjust accordingly.
# Display the result
print(current_job_levels)
## levels num_levels
## 1 17
## 2 Business Analyst 17
## 3 Computer Scientist 17
## 4 Data Analyst 17
## 5 Data Miner 17
## 6 Data Scientist 17
## 7 DBA/Database Engineer 17
## 8 Engineer 17
## 9 Machine Learning Engineer 17
## 10 Operations Research Practitioner 17
## 11 Other 17
## 12 Predictive Modeler 17
## 13 Programmer 17
## 14 Researcher 17
## 15 Scientist/Researcher 17
## 16 Software Developer/Software Engineer 17
## 17 Statistician 17
# 2.1 Flip the coordinates
ggplot(survey_data, aes(x = EmployerIndustry)) +
geom_bar() +
coord_flip()
# 2.2 Filter rows where Age and EmployerIndustry are not NA and replot
ggplot(survey_data %>% filter(!is.na(Age) & !is.na(EmployerIndustry)), aes(x = EmployerIndustry)) +
geom_bar() +
coord_flip()
# 2.3 Use geom_segment to plot and arrange data in descending order
age_means <- survey_data %>%
group_by(EmployerIndustry) %>%
summarise(mean_age = mean(Age, na.rm = TRUE)) %>%
arrange(desc(mean_age))
ggplot(age_means, aes(x = reorder(EmployerIndustry, -mean_age), y = mean_age)) +
geom_bar(stat = "identity") +
coord_flip()
# 3. Get the levels of WorkInternalVsExternalTools and generate the plot
work_levels <- levels(survey_data$WorkInternalVsExternalTools)
# Manually change the order
custom_order <- c(
"Entirely internal",
"More internal than external",
"Approximately half internal and half external",
"More external than internal",
"Entirely external",
"Do not know"
)
# Change the order of levels
survey_data$WorkInternalVsExternalTools <- factor(
survey_data$WorkInternalVsExternalTools,
levels = custom_order
)
# Generate the plot
ggplot(survey_data, aes(x = WorkInternalVsExternalTools)) +
geom_bar(stat = "count") +
labs(title = "WorkInternalVsExternalTools Levels")
You can also embed plots, for example:
Note that the echo = FALSE parameter was added to the
code chunk to prevent printing of the R code that generated the
plot.