Research Question: Which skills are consistently prioritize across industries between 2015 and 2019?

library(tidyverse)
library(ggplot2)
library(readr)

#load the csv into R
skill_needs <- read_csv("industry_need_skill.csv")

Clean & Inspect Data

This data set does not have any NAs.

#count the total of NAs for every column. 
count_nas <- skill_needs %>%
        summarize(
                across(everything(), ~ sum(is.na(.x)))
        )
count_nas
## # A tibble: 1 × 7
##    year isic_section_index isic_section_name industry_name skill_group_category
##   <int>              <int>             <int>         <int>                <int>
## 1     0                  0                 0             0                    0
## # ℹ 2 more variables: skill_group_name <int>, skill_group_rank <int>

Dimension of the data frame is 3500 observations and 7 variables.

dim(skill_needs)
## [1] 3500    7

The names for each variable/column.

names(skill_needs)
## [1] "year"                 "isic_section_index"   "isic_section_name"   
## [4] "industry_name"        "skill_group_category" "skill_group_name"    
## [7] "skill_group_rank"

The structure and data types in the data frame.

str(skill_needs)
## spc_tbl_ [3,500 × 7] (S3: spec_tbl_df/tbl_df/tbl/data.frame)
##  $ year                : num [1:3500] 2015 2015 2015 2015 2015 ...
##  $ isic_section_index  : chr [1:3500] "B" "B" "B" "B" ...
##  $ isic_section_name   : chr [1:3500] "Mining and quarrying" "Mining and quarrying" "Mining and quarrying" "Mining and quarrying" ...
##  $ industry_name       : chr [1:3500] "Mining & Metals" "Mining & Metals" "Mining & Metals" "Mining & Metals" ...
##  $ skill_group_category: chr [1:3500] "Specialized Industry Skills" "Soft Skills" "Business Skills" "Business Skills" ...
##  $ skill_group_name    : chr [1:3500] "Mining" "Negotiation" "Project Management" "Business Management" ...
##  $ skill_group_rank    : num [1:3500] 1 2 3 4 5 6 7 8 9 10 ...
##  - attr(*, "spec")=
##   .. cols(
##   ..   year = col_double(),
##   ..   isic_section_index = col_character(),
##   ..   isic_section_name = col_character(),
##   ..   industry_name = col_character(),
##   ..   skill_group_category = col_character(),
##   ..   skill_group_name = col_character(),
##   ..   skill_group_rank = col_double()
##   .. )
##  - attr(*, "problems")=<pointer: 0x126892c40>

Top positions ranked by category and year

top_five_category <- skill_needs %>%
    filter(skill_group_rank <= 5) %>%
    group_by(year, skill_group_category) %>%
    summarise(
        skill_count = n(),
        .groups = "drop"
    ) %>%
    group_by(year) %>%
    mutate(
        percentage = round(
            skill_count / sum(skill_count) * 100, 1
        )
    ) %>%
    ungroup()

top_five_category
## # A tibble: 25 × 4
##     year skill_group_category        skill_count percentage
##    <dbl> <chr>                             <int>      <dbl>
##  1  2015 Business Skills                      90       25.7
##  2  2015 Disruptive Tech Skills               13        3.7
##  3  2015 Soft Skills                          35       10  
##  4  2015 Specialized Industry Skills         149       42.6
##  5  2015 Tech Skills                          63       18  
##  6  2016 Business Skills                      81       23.1
##  7  2016 Disruptive Tech Skills               14        4  
##  8  2016 Soft Skills                          52       14.9
##  9  2016 Specialized Industry Skills         143       40.9
## 10  2016 Tech Skills                          60       17.1
## # ℹ 15 more rows
ggplot(
    top_five_category,
    aes(
        x = factor(year),
        y = percentage,
        fill = skill_group_category
    )
) +
    geom_col() +
    labs(
        title = "Composition of Top-Five Skills by Year",
        x = "Year",
        y = "Percentage of Top-Five Positions",
        fill = "Skill Category"
    ) 

What is the top skill group

category_skills <- skill_needs %>%
    group_by(skill_group_category) %>%
    summarise(
        distinct_skills = n_distinct(skill_group_name),
        .groups = "drop"
    ) %>%
    arrange(desc(distinct_skills))

category_skills
## # A tibble: 5 × 2
##   skill_group_category        distinct_skills
##   <chr>                                 <int>
## 1 Specialized Industry Skills              99
## 2 Business Skills                          32
## 3 Tech Skills                              18
## 4 Disruptive Tech Skills                    9
## 5 Soft Skills                               6

The frequency of skills plotted on a bar chart.

bar_skills <- ggplot(
    category_skills,
    aes(
        x = reorder(skill_group_category, -distinct_skills),
        y = distinct_skills
    )
) +
    geom_col(fill = "steelblue") +
    labs(
        title = "Distinct Skills by Category, 2015–2019",
        x = "Skill Categories",
        y = "Number of Distinct Skills"
    ) 

bar_skills

### Top Five Skills Across Industry and year.

The top skills were digital literacy, teamwork, business management, leadership, and project management. This means that more industries identified digital literacy as a top skills followed by business management. When you think about technology in the work setting, I would want to understand if that finding still stands today.

top_common_skills <- skill_needs %>%
    group_by(skill_group_name) %>%
    summarise(
        industry_count = n_distinct(industry_name),
        .groups = "drop"
    ) %>%
    filter(industry_count >= 2) %>%
    slice_max(
        order_by = industry_count,
        n = 5,
        with_ties = FALSE
    ) %>%
    arrange(desc(industry_count))

top_common_skills
## # A tibble: 5 × 2
##   skill_group_name    industry_count
##   <chr>                        <int>
## 1 Digital Literacy                68
## 2 Teamwork                        53
## 3 Business Management             50
## 4 Leadership                      50
## 5 Project Management              38