library(tidyverse)
library(ggplot2)
library(readr)
#load the csv into R
skill_needs <- read_csv("industry_need_skill.csv")
This data set does not have any NAs.
#count the total of NAs for every column.
count_nas <- skill_needs %>%
summarize(
across(everything(), ~ sum(is.na(.x)))
)
count_nas
## # A tibble: 1 × 7
## year isic_section_index isic_section_name industry_name skill_group_category
## <int> <int> <int> <int> <int>
## 1 0 0 0 0 0
## # ℹ 2 more variables: skill_group_name <int>, skill_group_rank <int>
Dimension of the data frame is 3500 observations and 7 variables.
dim(skill_needs)
## [1] 3500 7
The names for each variable/column.
names(skill_needs)
## [1] "year" "isic_section_index" "isic_section_name"
## [4] "industry_name" "skill_group_category" "skill_group_name"
## [7] "skill_group_rank"
The structure and data types in the data frame.
str(skill_needs)
## spc_tbl_ [3,500 × 7] (S3: spec_tbl_df/tbl_df/tbl/data.frame)
## $ year : num [1:3500] 2015 2015 2015 2015 2015 ...
## $ isic_section_index : chr [1:3500] "B" "B" "B" "B" ...
## $ isic_section_name : chr [1:3500] "Mining and quarrying" "Mining and quarrying" "Mining and quarrying" "Mining and quarrying" ...
## $ industry_name : chr [1:3500] "Mining & Metals" "Mining & Metals" "Mining & Metals" "Mining & Metals" ...
## $ skill_group_category: chr [1:3500] "Specialized Industry Skills" "Soft Skills" "Business Skills" "Business Skills" ...
## $ skill_group_name : chr [1:3500] "Mining" "Negotiation" "Project Management" "Business Management" ...
## $ skill_group_rank : num [1:3500] 1 2 3 4 5 6 7 8 9 10 ...
## - attr(*, "spec")=
## .. cols(
## .. year = col_double(),
## .. isic_section_index = col_character(),
## .. isic_section_name = col_character(),
## .. industry_name = col_character(),
## .. skill_group_category = col_character(),
## .. skill_group_name = col_character(),
## .. skill_group_rank = col_double()
## .. )
## - attr(*, "problems")=<pointer: 0x126892c40>
top_five_category <- skill_needs %>%
filter(skill_group_rank <= 5) %>%
group_by(year, skill_group_category) %>%
summarise(
skill_count = n(),
.groups = "drop"
) %>%
group_by(year) %>%
mutate(
percentage = round(
skill_count / sum(skill_count) * 100, 1
)
) %>%
ungroup()
top_five_category
## # A tibble: 25 × 4
## year skill_group_category skill_count percentage
## <dbl> <chr> <int> <dbl>
## 1 2015 Business Skills 90 25.7
## 2 2015 Disruptive Tech Skills 13 3.7
## 3 2015 Soft Skills 35 10
## 4 2015 Specialized Industry Skills 149 42.6
## 5 2015 Tech Skills 63 18
## 6 2016 Business Skills 81 23.1
## 7 2016 Disruptive Tech Skills 14 4
## 8 2016 Soft Skills 52 14.9
## 9 2016 Specialized Industry Skills 143 40.9
## 10 2016 Tech Skills 60 17.1
## # ℹ 15 more rows
ggplot(
top_five_category,
aes(
x = factor(year),
y = percentage,
fill = skill_group_category
)
) +
geom_col() +
labs(
title = "Composition of Top-Five Skills by Year",
x = "Year",
y = "Percentage of Top-Five Positions",
fill = "Skill Category"
)
category_skills <- skill_needs %>%
group_by(skill_group_category) %>%
summarise(
distinct_skills = n_distinct(skill_group_name),
.groups = "drop"
) %>%
arrange(desc(distinct_skills))
category_skills
## # A tibble: 5 × 2
## skill_group_category distinct_skills
## <chr> <int>
## 1 Specialized Industry Skills 99
## 2 Business Skills 32
## 3 Tech Skills 18
## 4 Disruptive Tech Skills 9
## 5 Soft Skills 6
The frequency of skills plotted on a bar chart.
bar_skills <- ggplot(
category_skills,
aes(
x = reorder(skill_group_category, -distinct_skills),
y = distinct_skills
)
) +
geom_col(fill = "steelblue") +
labs(
title = "Distinct Skills by Category, 2015–2019",
x = "Skill Categories",
y = "Number of Distinct Skills"
)
bar_skills
### Top Five Skills Across Industry and year.
The top skills were digital literacy, teamwork, business management, leadership, and project management. This means that more industries identified digital literacy as a top skills followed by business management. When you think about technology in the work setting, I would want to understand if that finding still stands today.
top_common_skills <- skill_needs %>%
group_by(skill_group_name) %>%
summarise(
industry_count = n_distinct(industry_name),
.groups = "drop"
) %>%
filter(industry_count >= 2) %>%
slice_max(
order_by = industry_count,
n = 5,
with_ties = FALSE
) %>%
arrange(desc(industry_count))
top_common_skills
## # A tibble: 5 × 2
## skill_group_name industry_count
## <chr> <int>
## 1 Digital Literacy 68
## 2 Teamwork 53
## 3 Business Management 50
## 4 Leadership 50
## 5 Project Management 38