1. Data Ingestion
bc_data <- read.csv("breast_cancer.csv", stringsAsFactors = FALSE)
bc_data$id <- NULL
empty_cols <- names(bc_data)[names(bc_data) == "" | grepl("^X$|^X\\.", names(bc_data))]
bc_data[empty_cols] <- NULL
bc_data <- bc_data[, colSums(is.na(bc_data)) < nrow(bc_data)]
numeric_cols <- sapply(bc_data, is.numeric)
for (col in names(bc_data)[numeric_cols]) {
if (any(is.na(bc_data[[col]]))) {
bc_data[[col]][is.na(bc_data[[col]])] <- median(bc_data[[col]], na.rm = TRUE)
}
}
bc_data$diagnosis <- factor(bc_data$diagnosis, levels = c("B", "M"),
labels = c("Benign", "Malignant"))
bc_data <- bc_data[!is.na(bc_data$diagnosis), ]
kable(data.frame(Metric = c("Rows", "Columns"),
Value = c(nrow(bc_data), ncol(bc_data))),
caption = "Dataset Dimensions")
Dataset Dimensions
| Rows |
569 |
| Columns |
31 |
kable(as.data.frame(table(bc_data$diagnosis)),
col.names = c("Diagnosis", "Count"),
caption = "Class Distribution")
Class Distribution
| Benign |
357 |
| Malignant |
212 |
2. Data Partitioning
set.seed(42)
train_index <- createDataPartition(bc_data$diagnosis, p = 0.80, list = FALSE)
train_data <- bc_data[train_index, ]
test_data <- bc_data[-train_index, ]
kable(data.frame(Set = c("Training", "Test"),
Rows = c(nrow(train_data), nrow(test_data))),
caption = "Train / Test Split (80/20)")
Train / Test Split (80/20)
| Training |
456 |
| Test |
113 |
3. Model Fitting
set.seed(42)
tuned_mtry <- tuneRF(
x = train_data[, -which(names(train_data) == "diagnosis")],
y = train_data$diagnosis,
ntreeTry = 500,
stepFactor = 1.5,
improve = 0.01,
trace = FALSE,
plot = FALSE
)
best_mtry <- tuned_mtry[which.min(tuned_mtry[, "OOBError"]), "mtry"]
Selected mtry: 3
rf_model <- randomForest(
diagnosis ~ .,
data = train_data,
ntree = 1000,
mtry = best_mtry,
importance = TRUE
)
saveRDS(rf_model, "rf_model.rds")
print(rf_model)
##
## Call:
## randomForest(formula = diagnosis ~ ., data = train_data, ntree = 1000, mtry = best_mtry, importance = TRUE)
## Type of random forest: classification
## Number of trees: 1000
## No. of variables tried at each split: 3
##
## OOB estimate of error rate: 4.39%
## Confusion matrix:
## Benign Malignant class.error
## Benign 279 7 0.02447552
## Malignant 13 157 0.07647059
The trained model is saved to rf_model.rds in the
working directory, and can be reloaded later without retraining:
rf_model <- readRDS("rf_model.rds")
predict(rf_model, newdata = new_patient_data)
4. Model Evaluation
rf_predictions <- predict(rf_model, newdata = test_data)
conf_matrix <- confusionMatrix(
rf_predictions,
test_data$diagnosis,
positive = "Malignant"
)
saveRDS(conf_matrix, "confusion_matrix.rds")
conf_matrix
## Confusion Matrix and Statistics
##
## Reference
## Prediction Benign Malignant
## Benign 69 0
## Malignant 2 42
##
## Accuracy : 0.9823
## 95% CI : (0.9375, 0.9978)
## No Information Rate : 0.6283
## P-Value [Acc > NIR] : <2e-16
##
## Kappa : 0.9625
##
## Mcnemar's Test P-Value : 0.4795
##
## Sensitivity : 1.0000
## Specificity : 0.9718
## Pos Pred Value : 0.9545
## Neg Pred Value : 1.0000
## Prevalence : 0.3717
## Detection Rate : 0.3717
## Detection Prevalence : 0.3894
## Balanced Accuracy : 0.9859
##
## 'Positive' Class : Malignant
##
5. Feature Importance
importance_df <- as.data.frame(round(importance(rf_model), 3))
kable(importance_df, caption = "Variable Importance Scores")
Variable Importance Scores
| radius_mean |
12.200 |
9.580 |
14.810 |
9.426 |
| texture_mean |
9.540 |
11.339 |
13.903 |
3.591 |
| perimeter_mean |
11.886 |
9.362 |
14.389 |
10.630 |
| area_mean |
14.463 |
9.339 |
15.967 |
11.822 |
| smoothness_mean |
3.759 |
9.487 |
10.108 |
1.690 |
| compactness_mean |
7.622 |
6.804 |
10.185 |
3.818 |
| concavity_mean |
10.555 |
11.049 |
15.351 |
10.667 |
| concave.points_mean |
14.498 |
15.465 |
19.711 |
19.383 |
| symmetry_mean |
2.073 |
6.083 |
6.588 |
1.006 |
| fractal_dimension_mean |
6.412 |
2.553 |
6.740 |
1.412 |
| radius_se |
9.762 |
7.989 |
12.616 |
4.994 |
| texture_se |
4.495 |
4.218 |
6.182 |
1.309 |
| perimeter_se |
10.427 |
9.676 |
14.146 |
5.248 |
| area_se |
15.152 |
11.325 |
18.681 |
11.276 |
| smoothness_se |
2.807 |
0.065 |
2.344 |
1.257 |
| compactness_se |
7.110 |
2.090 |
7.350 |
1.823 |
| concavity_se |
6.739 |
5.378 |
8.638 |
2.270 |
| concave.points_se |
6.976 |
3.215 |
7.853 |
2.106 |
| symmetry_se |
3.506 |
2.038 |
4.160 |
1.089 |
| fractal_dimension_se |
3.061 |
-0.787 |
2.010 |
1.417 |
| radius_worst |
17.810 |
15.278 |
21.863 |
18.631 |
| texture_worst |
11.425 |
13.369 |
16.371 |
3.767 |
| perimeter_worst |
17.038 |
15.039 |
21.162 |
19.923 |
| area_worst |
17.822 |
16.286 |
22.461 |
19.553 |
| smoothness_worst |
10.016 |
11.760 |
14.205 |
3.221 |
| compactness_worst |
8.632 |
8.718 |
12.425 |
6.053 |
| concavity_worst |
11.687 |
13.906 |
17.776 |
9.603 |
| concave.points_worst |
16.993 |
17.340 |
22.737 |
20.964 |
| symmetry_worst |
7.198 |
9.309 |
11.445 |
2.537 |
| fractal_dimension_worst |
5.848 |
4.532 |
7.340 |
2.103 |
varImpPlot(
rf_model,
main = "Random Forest - Feature Importance",
n.var = min(15, nrow(importance_df)),
pch = 19,
color = "steelblue"
)

6. Session Info
sessionInfo()
## R version 4.5.2 (2025-10-31)
## Platform: aarch64-apple-darwin20
## Running under: macOS Ventura 13.3
##
## Matrix products: default
## BLAS: /System/Library/Frameworks/Accelerate.framework/Versions/A/Frameworks/vecLib.framework/Versions/A/libBLAS.dylib
## LAPACK: /Library/Frameworks/R.framework/Versions/4.5-arm64/Resources/lib/libRlapack.dylib; LAPACK version 3.12.1
##
## locale:
## [1] en_US.UTF-8/en_US.UTF-8/en_US.UTF-8/C/en_US.UTF-8/en_US.UTF-8
##
## time zone: America/Chicago
## tzcode source: internal
##
## attached base packages:
## [1] stats graphics grDevices utils datasets methods base
##
## other attached packages:
## [1] knitr_1.51 caret_7.0-1 lattice_0.22-7
## [4] ggplot2_4.0.3 randomForest_4.7-1.2
##
## loaded via a namespace (and not attached):
## [1] gtable_0.3.6 xfun_0.60 bslib_0.12.0
## [4] recipes_1.3.3 vctrs_0.7.3 tools_4.5.2
## [7] generics_0.1.4 stats4_4.5.2 parallel_4.5.2
## [10] proxy_0.4-29 tibble_3.3.1 pkgconfig_2.0.3
## [13] ModelMetrics_1.2.2.2 Matrix_1.7-4 data.table_1.18.4
## [16] RColorBrewer_1.1-3 S7_0.2.2 lifecycle_1.0.5
## [19] compiler_4.5.2 farver_2.1.2 stringr_1.6.0
## [22] codetools_0.2-20 htmltools_0.5.9 class_7.3-23
## [25] sass_0.4.10 yaml_2.3.12 prodlim_2026.03.11
## [28] pillar_1.11.1 jquerylib_0.1.4 MASS_7.3-65
## [31] cachem_1.1.0 gower_1.0.2 iterators_1.0.14
## [34] rpart_4.1.24 foreach_1.5.2 nlme_3.1-168
## [37] parallelly_1.48.0 lava_1.9.2 tidyselect_1.2.1
## [40] digest_0.6.39 stringi_1.8.7 future_1.75.0
## [43] dplyr_1.2.1 reshape2_1.4.5 purrr_1.2.2
## [46] listenv_1.0.0 splines_4.5.2 fastmap_1.2.0
## [49] grid_4.5.2 cli_3.6.6 magrittr_2.0.5
## [52] survival_3.8-3 e1071_1.7-17 future.apply_1.20.2
## [55] withr_3.0.3 scales_1.4.0 lubridate_1.9.5
## [58] timechange_0.4.0 rmarkdown_2.31 globals_0.19.1
## [61] nnet_7.3-20 timeDate_4052.112 evaluate_1.0.5
## [64] hardhat_1.4.3 rlang_1.3.0 Rcpp_1.1.2
## [67] glue_1.8.1 pROC_1.19.0.1 ipred_0.9-15
## [70] jsonlite_2.0.0 R6_2.6.1 plyr_1.8.9