##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
##
## Attaching package: 'janitor'
## The following objects are masked from 'package:stats':
##
## chisq.test, fisher.test
##
## Attaching package: 'scales'
## The following object is masked from 'package:readr':
##
## col_factor
##
## Attaching package: 'data.table'
## The following objects are masked from 'package:dplyr':
##
## between, first, last
## The following object is masked from 'package:base':
##
## %notin%
BELAJAR DATA SCIENCE MENYENANGKAN
Data Wrangling di R
path_file <- if (file.exists("D:/lab/datasets/raw/dataset_pendidikan.csv")) {
"D:/lab/datasets/raw/dataset_pendidikan.csv"
} else if (file.exists("D:/lab/datasets/raw/dataset_pendidikan_2.csv")) {
"D:/lab/datasets/raw/dataset_pendidikan_2.csv"
} else if (file.exists("dataset_pendidikan.csv")) {
"dataset_pendidikan.csv"
} else {
"dataset_pendidikan_2.csv"
}
df_pendidikan <- read.csv(path_file, stringsAsFactors = FALSE)
kolom_jam <- grep("jam|belajar", names(df_pendidikan), ignore.case = TRUE, value = TRUE)[1]
kolom_hadir <- grep("hadir|kehadiran", names(df_pendidikan), ignore.case = TRUE, value = TRUE)[1]
if (!is.na(kolom_jam)) df_pendidikan$Jam_Belajar <- df_pendidikan[[kolom_jam]]
if (!is.na(kolom_hadir)) df_pendidikan$Total_Kehadiran <- df_pendidikan[[kolom_hadir]]
head(df_pendidikan)## ID_Mahasiswa IPK Semester Total_Kehadiran Total_Jam_Belajar Rata2_Jam_Tidur
## 1 MHS001 3.45 2 58 238 7.5
## 2 MHS002 3.13 5 54 246 7.1
## 3 MHS003 2.75 4 61 91 7.0
## 4 MHS004 3.52 4 67 162 6.6
## 5 MHS005 3.57 4 56 258 7.4
## 6 MHS006 3.52 7 56 222 7.5
## Jam_Belajar
## 1 238
## 2 246
## 3 91
## 4 162
## 5 258
## 6 222
## [1] 49
df_pendidikan %>%
arrange(desc(IPK)) %>%
select(any_of(c("ID_Mahasiswa", "NIM", "Nama")), IPK) %>%
slice_head(n = 3)## ID_Mahasiswa IPK
## 1 MHS052 4
## 2 MHS111 4
## 3 MHS123 4
df_pendidikan <- df_pendidikan %>%
mutate(Kategori_IPK = case_when(
IPK < 2.50 ~ "Kurang",
IPK >= 2.50 & IPK <= 2.99 ~ "Cukup",
IPK >= 3.00 & IPK <= 3.49 ~ "Baik",
IPK >= 3.50 ~ "Sangat Baik"
))
# Sebaran kategori
table(df_pendidikan$Kategori_IPK)##
## Baik Cukup Kurang Sangat Baik
## 71 63 17 49
## [1] 71
df_pendidikan %>%
group_by(Semester) %>%
summarise(Rata_Rata_IPK = mean(IPK, na.rm = TRUE)) %>%
arrange(desc(Rata_Rata_IPK))## # A tibble: 8 × 2
## Semester Rata_Rata_IPK
## <int> <dbl>
## 1 4 3.20
## 2 6 3.19
## 3 3 3.17
## 4 5 3.12
## 5 7 3.06
## 6 1 3.06
## 7 8 3.04
## 8 2 3.04
df_pendidikan <- df_pendidikan %>%
mutate(Persen_Kehadiran = (Total_Kehadiran / 70) * 100)
df_pendidikan %>%
filter(Persen_Kehadiran < 75) %>%
nrow()## [1] 40
# Menghitung nilai tengah dan rata-rata
med_jam <- median(df_pendidikan$Jam_Belajar, na.rm = TRUE)
mean_ipk <- mean(df_pendidikan$IPK, na.rm = TRUE)
# Filter mahasiswa jam belajar > median dan IPK < rata-rata
df_pendidikan %>%
filter(Jam_Belajar > med_jam & IPK < mean_ipk) %>%
nrow()## [1] 17