library(readxl)
data_mahasiswa <- read_excel("D:\\Skripsi\\Alokasi Dosen Pembimbing Ketua Tugas Akhir Mahasiswa dengan Pendekatan Algoritma Genetika\\Olah Data\\Data Mahasiswa.xlsx")
data_dosen <- read_excel("D:\\Skripsi\\Alokasi Dosen Pembimbing Ketua Tugas Akhir Mahasiswa dengan Pendekatan Algoritma Genetika\\Olah Data\\Data Dosen.xlsx")
# sampel
data_mahasiswa$`Topik yang diminati`[46]
## [1] "Deteksi anomali pada residual model GARCH menggunakan pendekatan machine learning dan deep learning untuk meningkatkan akurasi pemodelan volatilitas saham."
substr(data_dosen$`Judul Penelitian`[12], 1, 500)
## [1] "Evaluation of Tree-Based Models for Predicting Social Assistance Recipient Status Based on National Socio-Economic Survey (SUSENAS) 2024, Application of support vector regression with walk-forward validation and trend accuracy-based for forecasting the Rupiah exchange rate, Explainable Machine Learning Models SHAP-based for Feature Importance Affecting Stunting Prevalence, TREE-BASED MIXED EFFECTS MODELING OF TEACHER CERTIFICATION OUTCOMES IN MADRASAH ALIYAH: A COMPARATIVE STUDY OF GLMM TREES AN"
data_dosen$`Bidang Penelitian` <- paste(
data_dosen$`Bidang Minat`,
data_dosen$`Judul Penelitian`,
sep = " ; "
)
# Mengeluarkan dosen yang tidak bisa menjadi pembimbing pertama dari pilihan mahasiswa
for (i in 1:nrow(data_mahasiswa)) {
pilihan <- c(
data_mahasiswa$`Pilihan 1`[i],
data_mahasiswa$`Pilihan 2`[i],
data_mahasiswa$`Pilihan 3`[i]
)
# hapus AHW dan KAN
pilihan <- pilihan[!pilihan %in% c("AHW", "KAN")]
# tambahkan NA kalau jumlah pilihan jadi kurang dari 3
pilihan <- c(pilihan, rep(NA, 3 - length(pilihan)))
# masukkan lagi ke kolom
data_mahasiswa$`Pilihan 1`[i] <- pilihan[1]
data_mahasiswa$`Pilihan 2`[i] <- pilihan[2]
data_mahasiswa$`Pilihan 3`[i] <- pilihan[3]
}
data_mahasiswa
## # A tibble: 53 × 5
## `Nama Mahasiswa` `Pilihan 1` `Pilihan 2` `Pilihan 3` `Topik yang diminati`
## <chr> <chr> <chr> <chr> <chr>
## 1 Nur Jannah Tuasikal ADJ IND <NA> "Spasial, Classifica…
## 2 Alma Alifia Halima… IND ERF <NA> "Clustering, Klasifi…
## 3 Ade Ariyo Yudanto CSI BSO AMS "Komparasi Efektivit…
## 4 Natalinda Erlina A… BSO CSI <NA> "Machine Learning"
## 5 Joice Junansi Tand… AMS BSO <NA> "Big Data Analytics …
## 6 Rizky Mardhatillah IND BSO SRO "forecasting"
## 7 Nur Aulia Maknunah BSO SRO SDO "Machine Learning"
## 8 DWI ERZALIANTI BSO SRO SDO "Mechine Learning (K…
## 9 Haniyathul Husna ADJ UDS IND "Spasial statistic"
## 10 Sitti Rahmawati Hu… BSO <NA> <NA> "Time Series And For…
## # ℹ 43 more rows
# Imputasi topik
idx_topik_kosong <- which(
is.na(data_mahasiswa$`Topik yang diminati`) |
data_mahasiswa$`Topik yang diminati` == ""
)
data_mahasiswa$`Topik yang diminati`[idx_topik_kosong] <- "Statistika dan Sains Data"
# Imputasi Dosen Pilihan
# Hanya jika pilihan1,2,3 semuanya kosong
set.seed(7)
# 1. Mahasiswa yg semua pilihan kosong
idx_kosong <- which(
(is.na(data_mahasiswa$`Pilihan 1`) | data_mahasiswa$`Pilihan 1` == "") &
(is.na(data_mahasiswa$`Pilihan 2`) | data_mahasiswa$`Pilihan 2` == "") &
(is.na(data_mahasiswa$`Pilihan 3`) | data_mahasiswa$`Pilihan 3` == "")
)
n_kosong <- length(idx_kosong)
# 2. Daftar semua dosen
daftar_dosen <- unique(data_dosen$`Dosen Pembimbing 1`)
# 3. Hitung frekuensi Pilihan 1
freq_p1 <- table(data_mahasiswa$`Pilihan 1`)
# jadi dataframe
df_freq <- data.frame(
dosen = daftar_dosen,
freq = 0
)
# isi frekuensi sesuai tabel
match_idx <- match(df_freq$dosen, names(freq_p1))
df_freq$freq[!is.na(match_idx)] <- as.numeric(freq_p1[match_idx[!is.na(match_idx)]])
# 4. Urutkan freq kecil -> besar
df_freq <- df_freq[order(df_freq$freq), ]
# 5. Ambil dosen sampai >= jumlah mahasiswa kosong
kandidat <- c()
i <- 1
while(length(kandidat) < n_kosong){
freq_i <- unique(df_freq$freq)[i]
dosen_i <- df_freq$dosen[df_freq$freq == freq_i]
kandidat <- c(kandidat, dosen_i)
i <- i + 1
}
# 6. Acak kandidat
hasil <- sample(
kandidat,
size = n_kosong,
replace = FALSE
)
# 7. Isi Pilihan 1
data_mahasiswa$`Pilihan 1`[idx_kosong] <- hasil
head(data_mahasiswa[idx_kosong,])
## # A tibble: 1 × 5
## `Nama Mahasiswa` `Pilihan 1` `Pilihan 2` `Pilihan 3` `Topik yang diminati`
## <chr> <chr> <chr> <chr> <chr>
## 1 Arnedi Rizki Adidha… ERF <NA> <NA> Statistika dan Sains…
library(polyglotr)
data_mahasiswa$topik_penelitian_eng <- NA
for (i in 1:nrow(data_mahasiswa)) {
teks <- data_mahasiswa$`Topik yang diminati`[i]
data_mahasiswa$topik_penelitian_eng[i] <- google_translate(
teks,
target_language = "en",
source_language = "id"
)
}
# hapus link karena dapat mengganggu proses translasi
data_dosen$`Bidang Penelitian` <- gsub("http[s]?://[^ ]+", "", data_dosen$`Bidang Penelitian`)
data_dosen$`Bidang Penelitian` <- gsub("www\\.[^ ]+", "", data_dosen$`Bidang Penelitian`)
data_dosen$bidang_penelitian_eng <- NA
for (i in 1:nrow(data_dosen)) {
data_dosen$bidang_penelitian_eng[i] <- google_translate_long_text(
data_dosen$`Bidang Penelitian`[i],
target_language = "en",
source_language = "id"
)
}
# sampel
data_mahasiswa$topik_penelitian_eng[46]
## [1] "Anomaly detection in GARCH model residuals uses machine learning and deep learning approaches to increase the accuracy of stock volatility modeling."
substr(data_dosen$bidang_penelitian_eng[12], 1, 500)
## [1] "1. Data Mining, 2. Experiment Design and Analysis, 3. Interpretable Machine Learning, 4. Classification Modeling, 5. Ensemble Learning; Evaluation of Tree-Based Models for Predicting Social Assistance Recipient Status Based on National Socio-Economic Survey (SUSENAS) 2024, Application of support vector regression with walk-forward validation and trend accuracy-based for forecasting the Rupiah exchange rate, Explainable Machine Learning Models SHAP-based for Feature Importance Affecting Stunting "
data_dosen$bidang_penelitian_bersih <- tolower(data_dosen$bidang_penelitian_eng)
data_mahasiswa$topik_penelitian_bersih <- tolower(data_mahasiswa$topik_penelitian_eng)
# sampel
data_mahasiswa$topik_penelitian_bersih[46]
## [1] "anomaly detection in garch model residuals uses machine learning and deep learning approaches to increase the accuracy of stock volatility modeling."
substr(data_dosen$bidang_penelitian_bersih[12], 1, 500)
## [1] "1. data mining, 2. experiment design and analysis, 3. interpretable machine learning, 4. classification modeling, 5. ensemble learning; evaluation of tree-based models for predicting social assistance recipient status based on national socio-economic survey (susenas) 2024, application of support vector regression with walk-forward validation and trend accuracy-based for forecasting the rupiah exchange rate, explainable machine learning models shap-based for feature importance affecting stunting "
# hapus tanda baca
data_dosen$bidang_penelitian_bersih <- gsub("[[:punct:]]", " ", data_dosen$bidang_penelitian_bersih)
data_mahasiswa$topik_penelitian_bersih <- gsub("[[:punct:]]", " ", data_mahasiswa$topik_penelitian_bersih)
# hapus angka
data_dosen$bidang_penelitian_bersih <- gsub("[[:digit:]]", " ", data_dosen$bidang_penelitian_bersih)
data_mahasiswa$topik_penelitian_bersih <- gsub("[[:digit:]]", " ", data_mahasiswa$topik_penelitian_bersih)
# hapus simbol
data_dosen$bidang_penelitian_bersih <- gsub("[^\x20-\x7E]", " ", data_dosen$bidang_penelitian_bersih)
data_mahasiswa$topik_penelitian_bersih <- gsub("[^\x20-\x7E]", " ", data_mahasiswa$topik_penelitian_bersih)
# hapus baris baru
data_dosen$bidang_penelitian_bersih <- gsub("[\r\n\t]", " ", data_dosen$bidang_penelitian_bersih)
data_mahasiswa$topik_penelitian_bersih <- gsub("[\r\n\t]", " ", data_mahasiswa$topik_penelitian_bersih)
# rapikan spasi
data_dosen$bidang_penelitian_bersih <- gsub(" +", " ", data_dosen$bidang_penelitian_bersih)
data_dosen$bidang_penelitian_bersih <- trimws(data_dosen$bidang_penelitian_bersih)
data_mahasiswa$topik_penelitian_bersih <- gsub(" +", " ", data_mahasiswa$topik_penelitian_bersih)
data_mahasiswa$topik_penelitian_bersih <- trimws(data_mahasiswa$topik_penelitian_bersih)
# sampel
data_mahasiswa$topik_penelitian_bersih[46]
## [1] "anomaly detection in garch model residuals uses machine learning and deep learning approaches to increase the accuracy of stock volatility modeling"
substr(data_dosen$bidang_penelitian_bersih[12], 1, 500)
## [1] "data mining experiment design and analysis interpretable machine learning classification modeling ensemble learning evaluation of tree based models for predicting social assistance recipient status based on national socio economic survey susenas application of support vector regression with walk forward validation and trend accuracy based for forecasting the rupiah exchange rate explainable machine learning models shap based for feature importance affecting stunting prevalence tree based mixed e"
https://github.com/stopwords-iso/stopwords-en/blob/master/stopwords-en.txt
library(tm)
## Loading required package: NLP
stopword_custom <- readLines("D:\\Skripsi\\Alokasi Dosen Pembimbing Ketua Tugas Akhir Mahasiswa dengan Pendekatan Algoritma Genetika\\Olah Data\\stopwords-en.txt")
## Warning in readLines("D:\\Skripsi\\Alokasi Dosen Pembimbing Ketua Tugas Akhir
## Mahasiswa dengan Pendekatan Algoritma Genetika\\Olah Data\\stopwords-en.txt"):
## incomplete final line found on 'D:\Skripsi\Alokasi Dosen Pembimbing Ketua Tugas
## Akhir Mahasiswa dengan Pendekatan Algoritma Genetika\Olah
## Data\stopwords-en.txt'
stopword_custom <- tolower(trimws(stopword_custom))
stopword_custom <- stopword_custom[stopword_custom != ""]
stopword_all <- c(
stopwords("en"),
stopword_custom
)
data_mahasiswa$topik_penelitian_bersih <- removeWords(
data_mahasiswa$topik_penelitian_bersih,
stopword_all
)
data_dosen$bidang_penelitian_bersih <- removeWords(
data_dosen$bidang_penelitian_bersih,
stopword_all
)
# rapikan spasi kembali
data_mahasiswa$topik_penelitian_bersih <- gsub(" +", " ", data_mahasiswa$topik_penelitian_bersih)
data_mahasiswa$topik_penelitian_bersih <- trimws(data_mahasiswa$topik_penelitian_bersih)
data_dosen$bidang_penelitian_bersih <- gsub(" +", " ", data_dosen$bidang_penelitian_bersih )
data_dosen$bidang_penelitian_bersih <- trimws(data_dosen$bidang_penelitian_bersih)
# sampel
data_mahasiswa$topik_penelitian_bersih[46]
## [1] "anomaly detection garch model residuals machine learning deep learning approaches increase accuracy stock volatility modeling"
substr(data_dosen$bidang_penelitian_bersih[12], 1, 500)
## [1] "data mining experiment design analysis interpretable machine learning classification modeling ensemble learning evaluation tree based models predicting social assistance recipient status based national socio economic survey susenas application support vector regression walk validation trend accuracy based forecasting rupiah exchange rate explainable machine learning models shap based feature stunting prevalence tree based mixed effects modeling teacher certification outcomes madrasah aliyah comp"
library(quanteda)
## Warning: package 'quanteda' was built under R version 4.5.3
## Package version: 4.4
## Unicode version: 15.1
## ICU version: 74.1
## Parallel computing: 4 of 4 threads used.
## See https://quanteda.io for tutorials and examples.
##
## Attaching package: 'quanteda'
## The following object is masked from 'package:tm':
##
## stopwords
## The following objects are masked from 'package:NLP':
##
## meta, meta<-
corp <- corpus(c(
data_mahasiswa$topik_penelitian_bersih,
data_dosen$bidang_penelitian_bersih
))
# unigram
tok_unigram <- tokens(corp)
dfm_unigram <- dfm(tok_unigram)
dim(dfm_unigram)
## [1] 72 2497
# bigram
tok_bigram <- tokens_ngrams(tokens(corp), n = 2)
dfm_bigram <- dfm(tok_bigram)
dim(dfm_bigram)
## [1] 72 7913
# unibigram
tok_unibigram <- tokens_ngrams(tokens(corp), n = 1:2)
dfm_unibigram <- dfm(tok_unibigram)
dim(dfm_unibigram)
## [1] 72 10410
# sampel
tok_unigram[53,]
## Tokens consisting of 1 document.
## text53 :
## [1] "classification" "beverage" "products" "unsupervised"
## [5] "supervised" "learning" "approaches" "title"
## [9] "development" "clustering" "based" "classification"
## [ ... and 10 more ]
tok_unibigram[53,]
## Tokens consisting of 1 document.
## text53 :
## [1] "classification" "beverage" "products" "unsupervised"
## [5] "supervised" "learning" "approaches" "title"
## [9] "development" "clustering" "based" "classification"
## [ ... and 31 more ]
tok_bigram[53,]
## Tokens consisting of 1 document.
## text53 :
## [1] "classification_beverage" "beverage_products"
## [3] "products_unsupervised" "unsupervised_supervised"
## [5] "supervised_learning" "learning_approaches"
## [7] "approaches_title" "title_development"
## [9] "development_clustering" "clustering_based"
## [11] "based_classification" "classification_method"
## [ ... and 9 more ]
tfidf_unigram <- dfm_tfidf(dfm_unigram, scheme_tf = "prop")
dim(tfidf_unigram)
## [1] 72 2497
tfidf_bigram <- dfm_tfidf(dfm_bigram, scheme_tf = "prop")
dim(tfidf_bigram)
## [1] 72 7913
tfidf_unibigram <- dfm_tfidf(dfm_unibigram, scheme_tf = "prop")
dim(tfidf_unibigram)
## [1] 72 10410
tfidf_unigram[1,]
## Document-feature matrix of: 1 document, 2,497 features (99.88% sparse) and 0 docvars.
## features
## docs spatial classification clustering data mining comparison
## text1 0.1783711 0.1316448 0.1652016 0 0 0
## features
## docs effectiveness fine tuned indobert
## text1 0 0 0 0
## [ reached max_nfeat ... 2,487 more features ]
tfidf_unigram[2,]
## Document-feature matrix of: 1 document, 2,497 features (99.84% sparse) and 0 docvars.
## features
## docs spatial classification clustering data mining comparison
## text2 0 0.09873362 0.1239012 0.08146339 0.2895906 0
## features
## docs effectiveness fine tuned indobert
## text2 0 0 0 0
## [ reached max_nfeat ... 2,487 more features ]
tfidf_unigram[3,]
## Document-feature matrix of: 1 document, 2,497 features (99.28% sparse) and 0 docvars.
## features
## docs spatial classification clustering data mining comparison
## text3 0 0 0 0.01810298 0 0.03214327
## features
## docs effectiveness fine tuned indobert
## text3 0.05017167 0.08646125 0.1031851 0.06973736
## [ reached max_nfeat ... 2,487 more features ]
tfidf_unigram[59,]
## Document-feature matrix of: 1 document, 2,497 features (88.87% sparse) and 0 docvars.
## features
## docs spatial classification clustering data mining comparison
## text59 0.01978714 0.0006349429 0.0007967921 0.005762684 0 0
## features
## docs effectiveness fine tuned indobert
## text59 0 0 0 0
## [ reached max_nfeat ... 2,487 more features ]
tfidf_unigram[60,]
## Document-feature matrix of: 1 document, 2,497 features (80.70% sparse) and 0 docvars.
## features
## docs spatial classification clustering data mining
## text60 0.0005471505 0.002422911 0.003040519 0.004664571 0.00118442
## features
## docs comparison effectiveness fine tuned indobert
## text60 0.005324346 0 0 0 0
## [ reached max_nfeat ... 2,487 more features ]
tfidf_unigram[61,]
## Document-feature matrix of: 1 document, 2,497 features (89.87% sparse) and 0 docvars.
## features
## docs spatial classification clustering data mining comparison
## text61 0 0.005052467 0.006340358 0.00555827 0 0.00986915
## features
## docs effectiveness fine tuned indobert
## text61 0.001925565 0 0 0
## [ reached max_nfeat ... 2,487 more features ]
tfidf_bigram[1,]
## Document-feature matrix of: 1 document, 7,913 features (99.97% sparse) and 0 docvars.
## features
## docs spatial_classification classification_clustering
## text1 0.9286662 0.6901056
## features
## docs clustering_classification classification_data data_mining
## text1 0 0 0
## features
## docs comparison_effectiveness effectiveness_fine fine_tuned tuned_indobert
## text1 0 0 0 0
## features
## docs indobert_support
## text1 0
## [ reached max_nfeat ... 7,903 more features ]
tfidf_bigram[2,]
## Document-feature matrix of: 1 document, 7,913 features (99.96% sparse) and 0 docvars.
## features
## docs spatial_classification classification_clustering
## text2 0 0
## features
## docs clustering_classification classification_data data_mining
## text2 0.6191108 0.6191108 0.4184242
## features
## docs comparison_effectiveness effectiveness_fine fine_tuned tuned_indobert
## text2 0 0 0 0
## features
## docs indobert_support
## text2 0
## [ reached max_nfeat ... 7,903 more features ]
tfidf_bigram[3,]
## Document-feature matrix of: 1 document, 7,913 features (99.79% sparse) and 0 docvars.
## features
## docs spatial_classification classification_clustering
## text3 0 0
## features
## docs clustering_classification classification_data data_mining
## text3 0 0 0
## features
## docs comparison_effectiveness effectiveness_fine fine_tuned tuned_indobert
## text3 0.1092549 0.1092549 0.1092549 0.1092549
## features
## docs indobert_support
## text3 0.1092549
## [ reached max_nfeat ... 7,903 more features ]
tfidf_bigram[59,]
## Document-feature matrix of: 1 document, 7,913 features (93.53% sparse) and 0 docvars.
## features
## docs spatial_classification classification_clustering
## text59 0 0
## features
## docs clustering_classification classification_data data_mining
## text59 0 0 0
## features
## docs comparison_effectiveness effectiveness_fine fine_tuned tuned_indobert
## text59 0 0 0 0
## features
## docs indobert_support
## text59 0
## [ reached max_nfeat ... 7,903 more features ]
tfidf_bigram[60,]
## Document-feature matrix of: 1 document, 7,913 features (89.32% sparse) and 0 docvars.
## features
## docs spatial_classification classification_clustering
## text60 0 0
## features
## docs clustering_classification classification_data data_mining
## text60 0 0 0
## features
## docs comparison_effectiveness effectiveness_fine fine_tuned tuned_indobert
## text60 0 0 0 0
## features
## docs indobert_support
## text60 0
## [ reached max_nfeat ... 7,903 more features ]
tfidf_bigram[61,]
## Document-feature matrix of: 1 document, 7,913 features (95.01% sparse) and 0 docvars.
## features
## docs spatial_classification classification_clustering
## text61 0 0
## features
## docs clustering_classification classification_data data_mining
## text61 0 0 0
## features
## docs comparison_effectiveness effectiveness_fine fine_tuned tuned_indobert
## text61 0 0 0 0
## features
## docs indobert_support
## text61 0
## [ reached max_nfeat ... 7,903 more features ]
library(proxy)
## Warning: package 'proxy' was built under R version 4.5.3
##
## Attaching package: 'proxy'
## The following objects are masked from 'package:stats':
##
## as.dist, dist
## The following object is masked from 'package:base':
##
## as.matrix
m <- nrow(data_mahasiswa)
n <- nrow(data_dosen)
mat_unigram <- as.matrix(tfidf_unigram)
unigram_mhs <- mat_unigram[1:m, ]
unigram_dosen <- mat_unigram[(m+1):(m+n), ]
cosine_unigram <- as.matrix(simil(unigram_mhs, unigram_dosen, method = "cosine"))
mat_unibigram <- as.matrix(tfidf_unibigram)
unibigram_mhs <- mat_unibigram[1:m, ]
unibigram_dosen <- mat_unibigram[(m+1):(m+n), ]
cosine_unibigram <- as.matrix(simil(unibigram_mhs, unibigram_dosen, method = "cosine"))
mat_bigram <- as.matrix(tfidf_bigram)
bigram_mhs <- mat_bigram[1:m, ]
bigram_dosen <- mat_bigram[(m+1):(m+n), ]
cosine_bigram <- as.matrix(simil(bigram_mhs, bigram_dosen, method = "cosine"))
max(cosine_unigram)
## [1] 0.2989592
max(cosine_unibigram)
## [1] 0.2117885
max(cosine_bigram)
## [1] 0.1822855
cosine_unigram[7,]
## text54 text55 text56 text57 text58 text59 text60
## 0.03064284 0.00000000 0.05191199 0.04946095 0.03184395 0.00000000 0.02246512
## text61 text62 text63 text64 text65 text66 text67
## 0.05994523 0.01333854 0.08899753 0.13279872 0.12206997 0.01551355 0.04900097
## text68 text69 text70 text71 text72
## 0.05294716 0.03133996 0.02037089 0.02486764 0.05339200
cosine_unibigram[7,]
## text54 text55 text56 text57 text58 text59 text60
## 0.02060720 0.00000000 0.03699369 0.03684595 0.02717309 0.00000000 0.02153603
## text61 text62 text63 text64 text65 text66 text67
## 0.05165348 0.01155613 0.06830393 0.09757788 0.10619090 0.01178585 0.03869660
## text68 text69 text70 text71 text72
## 0.03751104 0.01682334 0.01555433 0.01940150 0.03486209
cosine_bigram[7,]
## text54 text55 text56 text57 text58 text59
## 0.011261151 0.000000000 0.023405608 0.022967852 0.021368080 0.000000000
## text60 text61 text62 text63 text64 text65
## 0.020344879 0.043920347 0.009869181 0.046377121 0.060799380 0.086314813
## text66 text67 text68 text69 text70 text71
## 0.007152133 0.029530961 0.018822475 0.000000000 0.010610117 0.015424216
## text72
## 0.018687994
library(GA)
## Loading required package: foreach
## Loading required package: iterators
## Package 'GA' version 3.2.5
## Type 'citation("GA")' for citing this R package in publications.
##
## Attaching package: 'GA'
## The following object is masked from 'package:utils':
##
## de
# jumlah mahasiswa
m
## [1] 53
# jumlah dosen
n
## [1] 19
# buat id dosen dan id mahasiswa
data_dosen$id_dosen <- 1:nrow(data_dosen)
data_mahasiswa$id_mahasiswa <- 1:nrow(data_mahasiswa)
# ambil kapasitas dosen
kapasitas <- data_dosen$`Kapasitas Maksimum Bimbingan`
# ubah pilihan dosen mahasiswa menjadi id dosen
data_mahasiswa$pil1_id <- match(data_mahasiswa$`Pilihan 1`, data_dosen$`Dosen Pembimbing 1`)
data_mahasiswa$pil2_id <- match(data_mahasiswa$`Pilihan 2`, data_dosen$`Dosen Pembimbing 1`)
data_mahasiswa$pil3_id <- match(data_mahasiswa$`Pilihan 3`, data_dosen$`Dosen Pembimbing 1`)
hitung_penalti_pilihan <- function(g, p1, p2, p3,
w_p2, w_p3, w_pl) {
if (g == p1) {
return(0)
} else if (!is.na(p2) && g == p2) {
return(w_p2)
} else if (!is.na(p3) && g == p3) {
return(w_p3)
} else {
return(w_pl)
}
}
hitung_penalti_topik <- function(sim_ij, w_t) {
return(w_t * (1 - sim_ij))
}
# Penalti kapasitas konstan: kelebihan 1 atau 4 orang = penalti sama
hitung_penalti_kapasitas_konstan <- function(solusi, kapasitas, n, w_k) {
beban <- tabulate(solusi, nbins = n)
penalti <- sum(ifelse(beban > kapasitas, w_k, 0))
list(beban = beban, penalti = penalti)
}
# Penalti kapasitas proporsional: penalti = w_k x jumlah kelebihan
hitung_penalti_kapasitas_prop <- function(solusi, kapasitas, n, w_k) {
beban <- tabulate(solusi, nbins = n)
penalti <- sum(w_k * pmax(0, beban - kapasitas))
list(beban = beban, penalti = penalti)
}
fitness_dosbing <- function(x, sim_mat, data_mahasiswa, kapasitas, m, n,
w_k, w_p2, w_p3, w_pl, w_t,
fn_penalti_kapasitas) {
# ubah hasil GA menjadi nomor dosen
solusi <- floor(x)
# penalti kapasitas
penalti_kapasitas <- fn_penalti_kapasitas(
solusi = solusi,
kapasitas = kapasitas,
n = n,
w_k = w_k
)$penalti
penalti_pilihan <- 0
penalti_topik <- 0
for (i in 1:m) {
g <- solusi[i]
p1 <- data_mahasiswa$pil1_id[i]
p2 <- data_mahasiswa$pil2_id[i]
p3 <- data_mahasiswa$pil3_id[i]
sim_ij <- sim_mat[i, g]
# penalti pilihan
penalti_pilihan <- penalti_pilihan + hitung_penalti_pilihan(
g = g,
p1 = p1,
p2 = p2,
p3 = p3,
w_p2 = w_p2,
w_p3 = w_p3,
w_pl = w_pl
)
# penalti topik
penalti_topik <- penalti_topik + hitung_penalti_topik(
sim_ij = sim_ij,
w_t = w_t
)
}
total_cost <- penalti_kapasitas + penalti_pilihan + penalti_topik
fitness <- 1 / (1 + total_cost)
return(fitness)
}
# Bobot kecil
bobot_kecil <- list(
w_k = 50, # penalti kapasitas
w_p2 = 1, # dapat pilihan 2
w_p3 = 2, # dapat pilihan 3
w_pl = 10, # di luar semua pilihan
w_t = 5 # koefisien penalti similarity
)
# Bobot besar
bobot_besar <- list(
w_k = 1000, # penalti kapasitas
w_p2 = 1,
# dapat pilihan 2 (sama)
w_p3 = 5, # dapat pilihan 3
w_pl = 100, # di luar semua pilihan
w_t = 10 # koefisien penalti similarity
)
# Definisi 8 skenario (2 generasi x 2 penalti kapasitas x 2 bobot)
skenario <- list(
S1 = list(generasi = 100, fn_kapasitas = hitung_penalti_kapasitas_konstan, bobot = bobot_kecil),
S2 = list(generasi = 100, fn_kapasitas = hitung_penalti_kapasitas_konstan, bobot = bobot_besar),
S3 = list(generasi = 100, fn_kapasitas = hitung_penalti_kapasitas_prop, bobot = bobot_kecil),
S4 = list(generasi = 100, fn_kapasitas = hitung_penalti_kapasitas_prop, bobot = bobot_besar),
S5 = list(generasi = 1000, fn_kapasitas = hitung_penalti_kapasitas_konstan, bobot = bobot_kecil),
S6 = list(generasi = 1000, fn_kapasitas = hitung_penalti_kapasitas_konstan, bobot = bobot_besar),
S7 = list(generasi = 1000, fn_kapasitas = hitung_penalti_kapasitas_prop, bobot = bobot_kecil),
S8 = list(generasi = 1000, fn_kapasitas = hitung_penalti_kapasitas_prop, bobot = bobot_besar)
)
library(tictoc)
## Warning: package 'tictoc' was built under R version 4.5.3
# Daftar matriks similarity per versi n-gram
versi_ngram <- list(
unigram = cosine_unigram,
bigram = cosine_bigram,
unibigram = cosine_unibigram
)
hasil_semua <- list()
for (nama_ngram in names(versi_ngram)) {
cat("Versi N-gram:", toupper(nama_ngram), "\n")
sim_mat <- versi_ngram[[nama_ngram]]
hasil_skenario <- list()
for (nama in names(skenario)) {
cat("Menjalankan", nama, "\n")
s <- skenario[[nama]]
# Mulai hitung waktu
waktu_mulai <- proc.time()
ga_result <- ga(
type = "real-valued",
fitness = function(x) fitness_dosbing(
x = x,
sim_mat = sim_mat,
data_mahasiswa = data_mahasiswa,
kapasitas = kapasitas,
m = nrow(data_mahasiswa),
n = nrow(data_dosen),
w_k = s$bobot$w_k,
w_p2 = s$bobot$w_p2,
w_p3 = s$bobot$w_p3,
w_pl = s$bobot$w_pl,
w_t = s$bobot$w_t,
fn_penalti_kapasitas = s$fn_kapasitas
),
lower = rep(1, m),
upper = rep(n + 0.999, m),
popSize = 50,
maxiter = s$generasi,
keepBest = TRUE,
selection = gareal_rwSelection,
crossover = gareal_spCrossover,
mutation = gareal_raMutation
)
# Hitung waktu
waktu_selesai <- proc.time()
waktu_detik <- as.numeric((waktu_selesai - waktu_mulai)["elapsed"])
solusi_terbaik <- floor(ga_result@solution[1, ])
hasil_skenario[[nama]] <- list(
ga = ga_result,
solusi = solusi_terbaik,
fitness = ga_result@fitnessValue,
waktu = waktu_detik
)
cat(nama, "selesai - Fitness:", round(ga_result@fitnessValue, 6),
"| Waktu:", round(waktu_detik, 2), "detik\n")
}
hasil_semua[[nama_ngram]] <- hasil_skenario
}
## Versi N-gram: UNIGRAM
## Menjalankan S1
## S1 selesai - Fitness: 0.001776 | Waktu: 1.97 detik
## Menjalankan S2
## S2 selesai - Fitness: 0.000276 | Waktu: 1.96 detik
## Menjalankan S3
## S3 selesai - Fitness: 0.001648 | Waktu: 1.77 detik
## Menjalankan S4
## S4 selesai - Fitness: 0.000308 | Waktu: 1.87 detik
## Menjalankan S5
## S5 selesai - Fitness: 0.003286 | Waktu: 19.54 detik
## Menjalankan S6
## S6 selesai - Fitness: 0.001081 | Waktu: 18.53 detik
## Menjalankan S7
## S7 selesai - Fitness: 0.003144 | Waktu: 18.36 detik
## Menjalankan S8
## S8 selesai - Fitness: 0.001105 | Waktu: 20.14 detik
## Versi N-gram: BIGRAM
## Menjalankan S1
## S1 selesai - Fitness: 0.001689 | Waktu: 2.98 detik
## Menjalankan S2
## S2 selesai - Fitness: 0.000283 | Waktu: 1.99 detik
## Menjalankan S3
## S3 selesai - Fitness: 0.001751 | Waktu: 2.17 detik
## Menjalankan S4
## S4 selesai - Fitness: 0.000317 | Waktu: 1.98 detik
## Menjalankan S5
## S5 selesai - Fitness: 0.003101 | Waktu: 19.44 detik
## Menjalankan S6
## S6 selesai - Fitness: 0.000743 | Waktu: 19.66 detik
## Menjalankan S7
## S7 selesai - Fitness: 0.003033 | Waktu: 20.25 detik
## Menjalankan S8
## S8 selesai - Fitness: 0.000787 | Waktu: 19.76 detik
## Versi N-gram: UNIBIGRAM
## Menjalankan S1
## S1 selesai - Fitness: 0.001677 | Waktu: 2.25 detik
## Menjalankan S2
## S2 selesai - Fitness: 0.000268 | Waktu: 2.13 detik
## Menjalankan S3
## S3 selesai - Fitness: 0.001737 | Waktu: 2.12 detik
## Menjalankan S4
## S4 selesai - Fitness: 0.000275 | Waktu: 1.97 detik
## Menjalankan S5
## S5 selesai - Fitness: 0.00287 | Waktu: 19.47 detik
## Menjalankan S6
## S6 selesai - Fitness: 0.001776 | Waktu: 21.12 detik
## Menjalankan S7
## S7 selesai - Fitness: 0.003319 | Waktu: 19.07 detik
## Menjalankan S8
## S8 selesai - Fitness: 0.00087 | Waktu: 20.22 detik
ga_obj <- hasil_semua$unibigram$S8$ga
# Parameter utama
ga_obj@popSize # ukuran populasi
## [1] 50
ga_obj@maxiter # maksimum generasi
## [1] 1000
ga_obj@elitism # jumlah individu elitis
## [1] 2
ga_obj@pcrossover # probabilitas crossover
## [1] 0.8
ga_obj@pmutation # probabilitas mutasi
## [1] 0.1
hasil_semua$unigram$S1$ga@solution
## x1 x2 x3 x4 x5 x6 x7 x8
## [1,] 6.203849 8.365082 17.63632 14.90801 12.7332 9.116101 12.95703 17.36814
## x9 x10 x11 x12 x13 x14 x15 x16
## [1,] 6.76889 4.143011 11.16375 3.711134 15.05706 6.79034 16.09354 14.52816
## x17 x18 x19 x20 x21 x22 x23 x24
## [1,] 7.804916 16.94026 15.89214 18.68687 15.09372 15.57758 7.123582 14.80236
## x25 x26 x27 x28 x29 x30 x31 x32
## [1,] 16.95081 5.495818 3.367648 17.21342 18.73866 10.0383 18.96052 2.969131
## x33 x34 x35 x36 x37 x38 x39 x40
## [1,] 13.89409 6.432518 18.1481 3.725167 4.527768 8.633461 13.35532 12.22057
## x41 x42 x43 x44 x45 x46 x47 x48
## [1,] 19.22115 7.191742 19.32168 5.263879 12.26923 1.575757 1.559921 14.73812
## x49 x50 x51 x52 x53
## [1,] 8.259711 4.117639 7.114564 19.95071 4.967987
as.vector(hasil_semua$unibigram$S4$solusi)
## [1] 8 8 19 19 6 12 18 12 9 1 13 18 11 3 6 15 14 5 14 11 13 13 15 6 4
## [26] 10 14 17 9 1 9 17 11 9 6 16 19 3 1 10 10 12 11 16 5 16 4 7 4 2
## [51] 15 12 19
keterangan <- list(
S1 = c("100", "Konstan", "Kecil"),
S2 = c("100", "Konstan", "Besar"),
S3 = c("100", "Proporsional", "Kecil"),
S4 = c("100", "Proporsional", "Besar"),
S5 = c("1000", "Konstan", "Kecil"),
S6 = c("1000", "Konstan", "Besar"),
S7 = c("1000", "Proporsional", "Kecil"),
S8 = c("1000", "Proporsional", "Besar")
)
for (nama_ngram in names(hasil_semua)) {
cat("\nBEBAN DOSEN:", toupper(nama_ngram), "\n")
cat(sprintf("%-6s %-6s %-14s %-6s %-6s %-6s\n",
"Sken", "Gen", "Pen.Kapasitas", "Bobot", "Max", "Min"))
cat(strrep("-", 50), "\n")
for (nama in names(hasil_semua[[nama_ngram]])) {
solusi <- as.vector(hasil_semua[[nama_ngram]][[nama]]$solusi)
beban <- tabulate(solusi, nbins = nrow(data_dosen))
ket <- keterangan[[nama]]
cat(sprintf("%-6s %-6s %-14s %-6s %-6d %-6d\n",
nama, ket[1], ket[2], ket[3],
max(beban), min(beban)))
}
}
##
## BEBAN DOSEN: UNIGRAM
## Sken Gen Pen.Kapasitas Bobot Max Min
## --------------------------------------------------
## S1 100 Konstan Kecil 4 1
## S2 100 Konstan Besar 4 1
## S3 100 Proporsional Kecil 4 1
## S4 100 Proporsional Besar 4 1
## S5 1000 Konstan Kecil 4 0
## S6 1000 Konstan Besar 4 0
## S7 1000 Proporsional Kecil 4 0
## S8 1000 Proporsional Besar 4 0
##
## BEBAN DOSEN: BIGRAM
## Sken Gen Pen.Kapasitas Bobot Max Min
## --------------------------------------------------
## S1 100 Konstan Kecil 4 1
## S2 100 Konstan Besar 4 2
## S3 100 Proporsional Kecil 4 1
## S4 100 Proporsional Besar 4 1
## S5 1000 Konstan Kecil 4 0
## S6 1000 Konstan Besar 4 0
## S7 1000 Proporsional Kecil 4 0
## S8 1000 Proporsional Besar 4 0
##
## BEBAN DOSEN: UNIBIGRAM
## Sken Gen Pen.Kapasitas Bobot Max Min
## --------------------------------------------------
## S1 100 Konstan Kecil 4 0
## S2 100 Konstan Besar 4 0
## S3 100 Proporsional Kecil 4 1
## S4 100 Proporsional Besar 4 1
## S5 1000 Konstan Kecil 4 0
## S6 1000 Konstan Besar 4 0
## S7 1000 Proporsional Kecil 4 0
## S8 1000 Proporsional Besar 4 0
for (nama_ngram in names(hasil_semua)) {
cat("\nRINGKASAN:", toupper(nama_ngram), "\n")
cat(sprintf("%-8s %-8s %-16s %-8s %-12s %-10s\n",
"Skenario", "Generasi", "Penalti Kap", "Bobot",
"Fitness", "Waktu(dtk)"))
cat(strrep("-", 65), "\n")
for (nama in names(hasil_semua[[nama_ngram]])) {
ket <- keterangan[[nama]]
fit <- round(hasil_semua[[nama_ngram]][[nama]]$fitness, 6)
waktu <- hasil_semua[[nama_ngram]][[nama]]$waktu
cat(sprintf("%-8s %-8s %-16s %-8s %-12.6f %-10.2f\n",
nama,
ket[1],
ket[2],
ket[3],
fit,
waktu))
}
}
##
## RINGKASAN: UNIGRAM
## Skenario Generasi Penalti Kap Bobot Fitness Waktu(dtk)
## -----------------------------------------------------------------
## S1 100 Konstan Kecil 0.001776 1.97
## S2 100 Konstan Besar 0.000276 1.96
## S3 100 Proporsional Kecil 0.001648 1.77
## S4 100 Proporsional Besar 0.000308 1.87
## S5 1000 Konstan Kecil 0.003286 19.54
## S6 1000 Konstan Besar 0.001081 18.53
## S7 1000 Proporsional Kecil 0.003144 18.36
## S8 1000 Proporsional Besar 0.001105 20.14
##
## RINGKASAN: BIGRAM
## Skenario Generasi Penalti Kap Bobot Fitness Waktu(dtk)
## -----------------------------------------------------------------
## S1 100 Konstan Kecil 0.001689 2.98
## S2 100 Konstan Besar 0.000283 1.99
## S3 100 Proporsional Kecil 0.001751 2.17
## S4 100 Proporsional Besar 0.000317 1.98
## S5 1000 Konstan Kecil 0.003101 19.44
## S6 1000 Konstan Besar 0.000743 19.66
## S7 1000 Proporsional Kecil 0.003033 20.25
## S8 1000 Proporsional Besar 0.000787 19.76
##
## RINGKASAN: UNIBIGRAM
## Skenario Generasi Penalti Kap Bobot Fitness Waktu(dtk)
## -----------------------------------------------------------------
## S1 100 Konstan Kecil 0.001677 2.25
## S2 100 Konstan Besar 0.000268 2.13
## S3 100 Proporsional Kecil 0.001737 2.12
## S4 100 Proporsional Besar 0.000275 1.97
## S5 1000 Konstan Kecil 0.002870 19.47
## S6 1000 Konstan Besar 0.001776 21.12
## S7 1000 Proporsional Kecil 0.003319 19.07
## S8 1000 Proporsional Besar 0.000870 20.22
# Evaluasi Soft Constraint per Skenario dan N-gram
for (nama_ngram in names(hasil_semua)) {
sim_mat <- versi_ngram[[nama_ngram]]
cat("\nEVALUASI SOFT CONSTRAINT:", toupper(nama_ngram), "\n")
cat(sprintf("%-6s %-8s %-8s %-8s %-10s %-10s\n",
"Sken", "Pil1", "Pil2", "Pil3", "NonPilihan", "Avg.Sim"))
cat(strrep("-", 75), "\n")
for (nama in names(hasil_semua[[nama_ngram]])) {
solusi <- as.vector(hasil_semua[[nama_ngram]][[nama]]$solusi)
# Soft Constraint 1: Pelanggaran Dosen Pilihan
n_pil1 <- 0
n_pil2 <- 0
n_pil3 <- 0
n_non <- 0
for (i in 1:nrow(data_mahasiswa)) {
g <- solusi[i]
p1 <- data_mahasiswa$pil1_id[i]
p2 <- data_mahasiswa$pil2_id[i]
p3 <- data_mahasiswa$pil3_id[i]
if (!is.na(p1) && g == p1) {
n_pil1 <- n_pil1 + 1
} else if (!is.na(p2) && g == p2) {
n_pil2 <- n_pil2 + 1
} else if (!is.na(p3) && g == p3) {
n_pil3 <- n_pil3 + 1
} else {
n_non <- n_non + 1
}
}
# Soft Constraint 2: Kesesuaian Topik
sim_values <- sapply(1:nrow(data_mahasiswa), function(i) sim_mat[i, solusi[i]])
avg_sim <- mean(sim_values)
cat(sprintf("%-6s %-8d %-8d %-8d %-10d %-10.4f\n",
nama, n_pil1, n_pil2, n_pil3, n_non,
avg_sim))
}
}
##
## EVALUASI SOFT CONSTRAINT: UNIGRAM
## Sken Pil1 Pil2 Pil3 NonPilihan Avg.Sim
## ---------------------------------------------------------------------------
## S1 9 11 3 30 0.0756
## S2 9 9 4 31 0.0717
## S3 7 6 6 34 0.0650
## S4 7 11 8 27 0.0698
## S5 28 14 8 3 0.0818
## S6 29 16 4 4 0.0793
## S7 29 15 4 5 0.0790
## S8 32 16 1 4 0.0884
##
## EVALUASI SOFT CONSTRAINT: BIGRAM
## Sken Pil1 Pil2 Pil3 NonPilihan Avg.Sim
## ---------------------------------------------------------------------------
## S1 8 10 4 31 0.0076
## S2 14 8 1 30 0.0095
## S3 11 8 5 29 0.0105
## S4 13 10 4 26 0.0078
## S5 31 16 2 4 0.0132
## S6 27 17 1 8 0.0144
## S7 27 17 5 4 0.0125
## S8 27 12 7 7 0.0125
##
## EVALUASI SOFT CONSTRAINT: UNIBIGRAM
## Sken Pil1 Pil2 Pil3 NonPilihan Avg.Sim
## ---------------------------------------------------------------------------
## S1 9 6 6 32 0.0290
## S2 11 8 2 32 0.0311
## S3 10 8 5 30 0.0310
## S4 11 7 4 31 0.0324
## S5 30 10 6 7 0.0361
## S6 31 15 7 0 0.0341
## S7 31 16 4 2 0.0327
## S8 32 9 6 6 0.0377
tabel_alokasi_nama <- data.frame(
mahasiswa = data_mahasiswa$`Nama Mahasiswa`
)
for (nama_ngram in names(hasil_semua)) {
for (nama in names(hasil_semua[[nama_ngram]])) {
solusi <- as.vector(floor(hasil_semua[[nama_ngram]][[nama]]$solusi))
nama_col <- paste0(nama_ngram, "_", nama)
tabel_alokasi_nama[[nama_col]] <- data_dosen$`Dosen Pembimbing 1`[solusi]
}
}
print(tabel_alokasi_nama)
## mahasiswa unigram_S1 unigram_S2 unigram_S3 unigram_S4
## 1 Nur Jannah Tuasikal ADJ SDO AF ADJ
## 2 Alma Alifia Halimatunnisa ERF AMS IND IND
## 3 Ade Ariyo Yudanto SDO AS BSO BSO
## 4 Natalinda Erlina Amheka FMA KS ADJ IMS
## 5 Joice Junansi Tandirerung BSO AS YA AK
## 6 Rizky Mardhatillah IND AS IMS FMA
## 7 Nur Aulia Maknunah BSO BSO SDO SDO
## 8 DWI ERZALIANTI SDO IMS AK SRO
## 9 Haniyathul Husna ADJ IND UDS UDS
## 10 Sitti Rahmawati Hulukiti IMS FMA AMS AS
## 11 Sigap Abror Falah AMS BSO SRO CSI
## 12 Rahmi Nurul Ainun Fitrah HW FMA YA AF
## 13 Lalu Muhammad Yahya AF AMS SRO BSO
## 14 Rosita Ria Rusesta ADJ IMS ERF YA
## 15 Leni Pamularsih YA ERF KS AF
## 16 WILIA SONDRIVA FMA ADJ HW AS
## 17 Ika Lailia Nur Rohmatun Nazila BST CSI CSI MNA
## 18 Nurqalbu Abd.Mutalip YA BST AMS AF
## 19 Eliani AF BST FMA AMS
## 20 Avin Rahmadian SRO AMS AS AK
## 21 Chindy Putri Army AF UDS UDS BST
## 22 Inria Purwaningsih AF IND SRO YA
## 23 Syifa Ul Qalbi YMD BST AK AK UDS
## 24 ARIFATUN NISAK NUR AMANATI FMA HW BSO KS
## 25 Putri Aqila YA SRO AF UDS
## 26 Lilik Avitadia Prichanti AK SRO HW AS
## 27 Ni Made Ray Diantari HW HW FMA BSO
## 28 Nurita Suci Lestari SDO SRO UDS IMS
## 29 Markazul Adabiyah SRO ADJ AK ADJ
## 30 Daumi Rahmatika KS BSO UDS YA
## 31 Kinanti Rizky Pangestutik SRO IND AF CSI
## 32 Charisma Yasintasya Kafilla MNA AF AMS SDO
## 33 Arnedi Rizki Adidharma UDS ERF SRO KS
## 34 Afifah Humayrah ADJ BST KS ERF
## 35 Fitri Hayati SRO AF IMS HW
## 36 Siti Zakiah HW AK ADJ ADJ
## 37 Francisca Juventini Mandas IMS MNA AK ADJ
## 38 Dinda Ardhia Ramadhani Kusuma ERF ADJ BSO SRO
## 39 Viren Marcellya Clarenda Siboro UDS IMS IND BSO
## 40 Izzul Haq BSO AK FMA AMS
## 41 Aridha Pebriani Kusmiran CSI YA HW UDS
## 42 Maulana BST BST ERF FMA
## 43 Ain Fitri Basri CSI SDO ERF CSI
## 44 Freditasari Purwa Hidayat AK UDS ADJ SDO
## 45 Fatiya Hanifah BSO ADJ MNA SRO
## 46 FITRI DIANA MUSA AS YA AF YA
## 47 Muhammad Hanif Nafiis AS AMS SDO AMS
## 48 Muhammad Fadel Handiyono FMA SDO CSI AMS
## 49 Suci Rahmadani ERF UDS IND IND
## 50 Estyaningsi Purnama IMS IMS SDO IMS
## 51 Husnul Amira BST BSO BST SRO
## 52 Naila Nabiha Qonita CSI CSI CSI AK
## 53 Baiq Wita Rachmatia IMS SRO ERF CSI
## unigram_S5 unigram_S6 unigram_S7 unigram_S8 bigram_S1 bigram_S2 bigram_S3
## 1 ADJ IND IND IND IMS BST MNA
## 2 IND ERF ERF IND ERF FMA HW
## 3 BSO CSI CSI CSI YA FMA SDO
## 4 BSO KS AMS BSO AK MNA AMS
## 5 AMS KS AF KS AS SDO FMA
## 6 BSO BSO IND IND IMS BST KS
## 7 SDO SDO SRO SRO BSO AMS ERF
## 8 SDO SRO BSO SRO AMS SRO SRO
## 9 ADJ ADJ UDS UDS HW AF CSI
## 10 FMA BSO BSO FMA AF AF CSI
## 11 SRO CSI CSI CSI SDO CSI BSO
## 12 YA ADJ ADJ IMS SDO ADJ IMS
## 13 AMS AF AF AF AF AF AF
## 14 FMA FMA FMA FMA IMS IND FMA
## 15 ADJ ADJ ADJ ADJ AS CSI SDO
## 16 IMS HW HW HW IMS FMA FMA
## 17 AF CSI CSI BSO UDS AMS CSI
## 18 SRO YA SRO YA SRO YA BSO
## 19 YA IMS YA YA KS IND MNA
## 20 ADJ SRO ADJ ADJ CSI AMS AF
## 21 SDO SDO SDO SDO IND BSO SRO
## 22 UDS AMS UDS UDS CSI IMS UDS
## 23 AK YA UDS UDS UDS UDS SDO
## 24 HW HW HW HW AF AK IND
## 25 AF BSO AF BSO ADJ AF BSO
## 26 AK AK AK AK AK AS YA
## 27 KS KS KS KS KS KS AS
## 28 UDS IND IND IND IND UDS KS
## 29 IMS IND IMS IMS IND IMS IMS
## 30 UDS YA YA FMA SDO SRO IND
## 31 SRO SRO AMS SRO AMS SRO SRO
## 32 YA SDO YA YA ERF AS AK
## 33 ERF ERF ERF ERF BST ERF ADJ
## 34 IND IND IND ERF SDO IMS IND
## 35 SRO ADJ AMS ADJ HW ADJ ADJ
## 36 UDS UDS SRO UDS UDS ADJ MNA
## 37 IMS IMS IMS IMS CSI MNA AK
## 38 CSI SRO SRO CSI AMS HW AMS
## 39 IND UDS BSO BSO ERF AK UDS
## 40 AK AK AK AK BSO BSO CSI
## 41 AMS AMS YA AMS ADJ YA BSO
## 42 BST BST BST BST FMA HW FMA
## 43 CSI AMS CSI KS AK AMS KS
## 44 BSO AK AK AK BSO BSO AS
## 45 MNA UDS ADJ ADJ MNA SDO HW
## 46 YA YA IMS YA AS YA YA
## 47 CSI CSI AF AMS AF CSI ADJ
## 48 SDO FMA SDO SDO YA SDO BST
## 49 IMS UDS IMS IMS UDS KS UDS
## 50 MNA MNA MNA MNA YA IMS HW
## 51 BST BST AMS AMS BST ERF IMS
## 52 CSI BSO BSO CSI HW UDS IND
## 53 AMS AMS UDS AMS AMS BST AMS
## bigram_S4 bigram_S5 bigram_S6 bigram_S7 bigram_S8 unibigram_S1 unibigram_S2
## 1 ADJ IND IND ADJ ADJ AF ADJ
## 2 MNA IND ERF ERF ERF IMS IND
## 3 CSI CSI CSI AMS YA BSO BSO
## 4 AK BSO AMS KS CSI BST ERF
## 5 AS BSO BSO AMS AMS BSO HW
## 6 BSO IND IND IND SRO KS AK
## 7 SRO SRO SRO SRO SRO SRO CSI
## 8 SRO SRO BSO SRO BSO IMS BSO
## 9 IND ADJ UDS ADJ IND ADJ IND
## 10 AS KS BSO IMS BSO BSO ADJ
## 11 AF CSI BSO CSI CSI SRO IMS
## 12 BSO YA ADJ YA IMS HW AS
## 13 IMS AF AF AF AF IND AS
## 14 FMA FMA FMA FMA FMA CSI FMA
## 15 AS ADJ ADJ ADJ AMS IMS FMA
## 16 HW HW HW HW HW AS IMS
## 17 BSO AF CSI BSO CSI BSO BSO
## 18 YA BSO YA YA IMS SRO AS
## 19 HW KS AK YA YA UDS ADJ
## 20 IMS SRO SRO MNA ADJ AMS AK
## 21 UDS UDS UDS ERF UDS AMS KS
## 22 FMA UDS SRO CSI CSI AMS AF
## 23 SDO SDO SDO SDO SDO ADJ KS
## 24 AMS HW HW HW HW UDS AMS
## 25 IND AF AF BSO AF AF AMS
## 26 SRO AK AK AK YA YA YA
## 27 KS KS KS KS BSO AF BSO
## 28 UDS IND IND IND IND SDO UDS
## 29 CSI IMS IMS IND IMS HW IMS
## 30 YA YA IMS UDS FMA IMS YA
## 31 AMS SRO YA SRO SRO AF SDO
## 32 YA YA CSI SDO SDO SDO IND
## 33 BST ERF ERF ERF ERF FMA ERF
## 34 MNA ERF IND ERF IND AS UDS
## 35 SRO ADJ ADJ SRO ADJ ERF AS
## 36 KS ADJ ADJ ADJ UDS IND UDS
## 37 IMS AK IMS IMS AK YA SDO
## 38 CSI CSI AF CSI SRO CSI CSI
## 39 FMA BSO UDS UDS IND UDS SRO
## 40 ERF AK AK AK ERF AK AF
## 41 ERF AMS FMA AMS AMS IND AMS
## 42 BST BST BST BST BST SRO FMA
## 43 BSO AMS ERF CSI AMS HW ERF
## 44 AK AK AK BSO AK SDO AK
## 45 UDS MNA UDS UDS ADJ BST MNA
## 46 HW IMS YA IMS IMS BST AMS
## 47 CSI AMS AMS AMS YA CSI KS
## 48 SDO YA YA YA SDO ADJ KS
## 49 KS IMS IMS IMS UDS UDS SRO
## 50 UDS IMS MNA MNA MNA SDO MNA
## 51 YA AMS AMS BST BST KS UDS
## 52 SDO CSI CSI BSO BSO AK ERF
## 53 ERF UDS AMS UDS UDS AMS FMA
## unibigram_S3 unibigram_S4 unibigram_S5 unibigram_S6 unibigram_S7
## 1 IND ERF IND ADJ ADJ
## 2 AK ERF IMS ERF IND
## 3 AK CSI KS CSI CSI
## 4 BST CSI ERF CSI CSI
## 5 AMS ADJ AMS AMS AMS
## 6 SRO BSO IND SRO IND
## 7 SRO SRO SRO SRO SRO
## 8 AF BSO BSO SRO SRO
## 9 UDS IND IND IND ADJ
## 10 BSO AS IMS BSO FMA
## 11 KS UDS SRO BSO CSI
## 12 CSI SRO FMA IMS YA
## 13 ERF AMS AF AMS AF
## 14 IMS HW FMA FMA FMA
## 15 SDO ADJ ADJ SDO ADJ
## 16 ERF AF HW HW HW
## 17 CSI FMA CSI BSO BSO
## 18 SRO AK YA YA YA
## 19 YA FMA YA YA AMS
## 20 BSO AMS ADJ ADJ SRO
## 21 UDS UDS UDS SDO SDO
## 22 BST UDS UDS UDS UDS
## 23 YA AF YA UDS UDS
## 24 ADJ ADJ HW HW HW
## 25 HW IMS AF AF AF
## 26 KS KS CSI AK AK
## 27 KS FMA KS KS KS
## 28 AS SDO IND IND IND
## 29 IMS IND SRO IMS IMS
## 30 UDS AS BSO YA UDS
## 31 BSO IND SRO SRO CSI
## 32 IND SDO SDO SDO SDO
## 33 ERF AMS ERF ERF ERF
## 34 CSI IND ERF IND IND
## 35 SDO ADJ ADJ ADJ SRO
## 36 ADJ YA ADJ ADJ ADJ
## 37 AF CSI FMA IMS IMS
## 38 CSI HW BSO BSO BSO
## 39 AK AS UDS IND BSO
## 40 FMA KS AK AK AK
## 41 AF KS AMS AMS YA
## 42 ERF BSO BST BST BST
## 43 BST AMS CSI CSI BSO
## 44 FMA YA AK AK AK
## 45 ADJ AK UDS UDS UDS
## 46 IMS YA IMS IMS YA
## 47 FMA IMS AF AF AMS
## 48 FMA BST YA YA KS
## 49 ADJ IMS IMS UDS IMS
## 50 HW MNA MNA MNA MNA
## 51 AK AF AMS BST BST
## 52 MNA BSO BSO CSI SDO
## 53 SDO CSI AMS AMS AMS
## unibigram_S8
## 1 ADJ
## 2 ERF
## 3 CSI
## 4 BSO
## 5 KS
## 6 IND
## 7 SRO
## 8 HW
## 9 IND
## 10 FMA
## 11 CSI
## 12 FMA
## 13 AF
## 14 FMA
## 15 ADJ
## 16 IMS
## 17 AF
## 18 YA
## 19 AK
## 20 SRO
## 21 UDS
## 22 CSI
## 23 UDS
## 24 HW
## 25 AF
## 26 YA
## 27 KS
## 28 IND
## 29 IMS
## 30 YA
## 31 BSO
## 32 CSI
## 33 ERF
## 34 IND
## 35 ADJ
## 36 ADJ
## 37 IMS
## 38 SRO
## 39 BSO
## 40 AK
## 41 AMS
## 42 BST
## 43 AMS
## 44 BSO
## 45 MNA
## 46 SRO
## 47 AF
## 48 YA
## 49 UDS
## 50 IMS
## 51 AMS
## 52 SDO
## 53 AMS