install.packages(“caret”) install.packages(“rpart”) install.packages(“rpart.plot”) library(caret) library(rpart) library(rpart.plot) # 2. Memuat Dataset data_churn <- read.csv(“Iranian Churn Dataset.csv”)
head(data_churn, 10)
str(data_churn)
dim(data_churn)
data_churn\(Churn <- as.factor(data_churn\)Churn)
table(data_churn$Churn)
colSums(is.na(data_churn))
set.seed(42) # Agar hasil pembagian data konsisten index_latih <- createDataPartition( data_churn$Churn, p = 0.7, list = FALSE ) data_latih <- data_churn[index_latih, ] data_uji <- data_churn[-index_latih, ]
cat(“Jumlah data training:”, nrow(data_latih), “”) cat(“Jumlah data testing :”, nrow(data_uji), “”)
cat(“— MODEL 1: LOGISTIC REGRESSION —”) # 5. Membangun Model Logistic Regression model_logistik <- glm( Churn ~ ., data = data_latih, family = binomial )
summary(model_logistik)
prob_pred <- predict( model_logistik, newdata = data_uji, type = “response” ) # Mengubah probabilitas menjadi kelas # Jika probabilitas > 0.5 maka diprediksi sebagai 1 class_pred_log <- ifelse( prob_pred > 0.5, “1”, “0” ) class_pred_log <- as.factor(class_pred_log) # Menyamakan level factor dengan data aktual class_pred_log <- factor( class_pred_log, levels = levels(data_uji\(Churn) ) # 7. Confusion Matrix Logistic Regression cm_logistik <- confusionMatrix( class_pred_log, data_uji\)Churn, positive = “1” ) print(cm_logistik)
akurasi_logistik <- cm_logistik\(overall["Accuracy"] precision_logistik <- cm_logistik\)byClass[“Precision”] recall_logistik <- cm_logistik\(byClass["Recall"] f1_logistik <- cm_logistik\)byClass[“F1”]
cat(“Logistic Regression:”) cat(“Accuracy :”, round(akurasi_logistik * 100, 2), “%”) cat(“Precision:”, round(precision_logistik * 100, 2), “%”) cat(“Recall :”, round(recall_logistik * 100, 2), “%”) cat(“F1-Score :”, round(f1_logistik * 100, 2), “%”)
if (akurasi_logistik >= 0.85) { print(“Status Model 1: BERHASIL mencapai akurasi >= 85%!”) } else { print(“Status Model 1: Belum mencapai 85%.”) } # ============================================================ # MODEL 2: DECISION TREE # ============================================================ cat(“— MODEL 2: DECISION TREE —”) # 8. Membangun Model Decision Tree model_dt <- rpart( Churn ~ ., data = data_latih, method = “class” ) # Melihat ringkasan model print(model_dt) # 9. Visualisasi Decision Tree rpart.plot( model_dt, main = “Decision Tree - Iranian Churn” ) # 10. Melakukan Prediksi Decision Tree
class_pred_dt <- predict( model_dt, newdata = data_uji, type = “class” ) # 11. Confusion Matrix Decision Tree
cm_dt <- confusionMatrix( class_pred_dt, data_uji\(Churn, positive = "1" ) print(cm_dt) # Mengambil metrik evaluasi akurasi_dt <- cm_dt\)overall[“Accuracy”] precision_dt <- cm_dt\(byClass["Precision"] recall_dt <- cm_dt\)byClass[“Recall”] f1_dt <- cm_dt$byClass[“F1”]
cat(“Decision Tree:”) cat(“Accuracy :”, round(akurasi_dt * 100, 2), “%”) cat(“Precision:”, round(precision_dt * 100, 2), “%”) cat(“Recall :”, round(recall_dt * 100, 2), “%”) cat(“F1-Score :”, round(f1_dt * 100, 2), “%”)
if (akurasi_dt >= 0.85) { print(“Status Model 2: BERHASIL mencapai akurasi >= 85%!”) } else { print(“Status Model 2: Belum mencapai 85%.”) } # ============================================================ # 12. PERBANDINGAN HASIL KEDUA MODEL # ============================================================
hasil_model <- data.frame( Model = c( “Logistic Regression”, “Decision Tree” ), Accuracy = c( akurasi_logistik, akurasi_dt ), Precision = c( precision_logistik, precision_dt ), Recall = c( recall_logistik, recall_dt ), F1_Score = c( f1_logistik, f1_dt ) )
hasil_model\(Accuracy <- round( hasil_model\)Accuracy * 100, 2 )
hasil_model\(Precision <- round( hasil_model\)Precision * 100, 2 )
hasil_model\(Recall <- round( hasil_model\)Recall * 100, 2 )
hasil_model\(F1_Score <- round( hasil_model\)F1_Score * 100, 2 )
print(hasil_model)