1. Data Ingestion

bc_data <- read.csv("breast_cancer.csv", stringsAsFactors = FALSE)

bc_data$id <- NULL
empty_cols <- names(bc_data)[names(bc_data) == "" | grepl("^X$|^X\\.", names(bc_data))]
bc_data[empty_cols] <- NULL
bc_data <- bc_data[, colSums(is.na(bc_data)) < nrow(bc_data)]

numeric_cols <- sapply(bc_data, is.numeric)
for (col in names(bc_data)[numeric_cols]) {
  if (any(is.na(bc_data[[col]]))) {
    bc_data[[col]][is.na(bc_data[[col]])] <- median(bc_data[[col]], na.rm = TRUE)
  }
}

bc_data$diagnosis <- factor(bc_data$diagnosis, levels = c("B", "M"),
                             labels = c("Benign", "Malignant"))
bc_data <- bc_data[!is.na(bc_data$diagnosis), ]
kable(data.frame(Metric = c("Rows", "Columns"),
                  Value = c(nrow(bc_data), ncol(bc_data))),
      caption = "Dataset Dimensions")
Dataset Dimensions
Metric Value
Rows 569
Columns 31
kable(as.data.frame(table(bc_data$diagnosis)),
      col.names = c("Diagnosis", "Count"),
      caption = "Class Distribution")
Class Distribution
Diagnosis Count
Benign 357
Malignant 212

2. Data Partitioning

set.seed(42)
train_index <- createDataPartition(bc_data$diagnosis, p = 0.80, list = FALSE)
train_data <- bc_data[train_index, ]
test_data  <- bc_data[-train_index, ]

kable(data.frame(Set = c("Training", "Test"),
                  Rows = c(nrow(train_data), nrow(test_data))),
      caption = "Train / Test Split (80/20)")
Train / Test Split (80/20)
Set Rows
Training 456
Test 113

3. Model Fitting

set.seed(42)
tuned_mtry <- tuneRF(
  x = train_data[, -which(names(train_data) == "diagnosis")],
  y = train_data$diagnosis,
  ntreeTry = 500,
  stepFactor = 1.5,
  improve = 0.01,
  trace = FALSE,
  plot = FALSE
)
best_mtry <- tuned_mtry[which.min(tuned_mtry[, "OOBError"]), "mtry"]

Selected mtry: 3

rf_model <- randomForest(
  diagnosis ~ .,
  data = train_data,
  ntree = 1000,
  mtry = best_mtry,
  importance = TRUE
)

saveRDS(rf_model, "rf_model.rds")

print(rf_model)
## 
## Call:
##  randomForest(formula = diagnosis ~ ., data = train_data, ntree = 1000,      mtry = best_mtry, importance = TRUE) 
##                Type of random forest: classification
##                      Number of trees: 1000
## No. of variables tried at each split: 3
## 
##         OOB estimate of  error rate: 4.39%
## Confusion matrix:
##           Benign Malignant class.error
## Benign       279         7  0.02447552
## Malignant     13       157  0.07647059

The trained model is saved to rf_model.rds in the working directory, and can be reloaded later without retraining:

rf_model <- readRDS("rf_model.rds")
predict(rf_model, newdata = new_patient_data)

4. Model Evaluation

rf_predictions <- predict(rf_model, newdata = test_data)

conf_matrix <- confusionMatrix(
  rf_predictions,
  test_data$diagnosis,
  positive = "Malignant"
)

saveRDS(conf_matrix, "confusion_matrix.rds")

conf_matrix
## Confusion Matrix and Statistics
## 
##            Reference
## Prediction  Benign Malignant
##   Benign        69         0
##   Malignant      2        42
##                                           
##                Accuracy : 0.9823          
##                  95% CI : (0.9375, 0.9978)
##     No Information Rate : 0.6283          
##     P-Value [Acc > NIR] : <2e-16          
##                                           
##                   Kappa : 0.9625          
##                                           
##  Mcnemar's Test P-Value : 0.4795          
##                                           
##             Sensitivity : 1.0000          
##             Specificity : 0.9718          
##          Pos Pred Value : 0.9545          
##          Neg Pred Value : 1.0000          
##              Prevalence : 0.3717          
##          Detection Rate : 0.3717          
##    Detection Prevalence : 0.3894          
##       Balanced Accuracy : 0.9859          
##                                           
##        'Positive' Class : Malignant       
## 

Key Performance Metrics

metrics_df <- data.frame(
  Metric = c("Accuracy", "Sensitivity (Malignant Recall)", "Specificity"),
  Value = round(c(conf_matrix$overall["Accuracy"],
                   conf_matrix$byClass["Sensitivity"],
                   conf_matrix$byClass["Specificity"]), 4)
)
kable(metrics_df, caption = "Test Set Performance")
Test Set Performance
Metric Value
Accuracy Accuracy 0.9823
Sensitivity Sensitivity (Malignant Recall) 1.0000
Specificity Specificity 0.9718
  • Accuracy: overall proportion of correct predictions across both classes.
  • Sensitivity: of all actual malignant cases, the proportion correctly flagged — the most clinically important number, since a missed malignant case is the costliest error.
  • Specificity: of all actual benign cases, the proportion correctly cleared.

5. Feature Importance

importance_df <- as.data.frame(round(importance(rf_model), 3))
kable(importance_df, caption = "Variable Importance Scores")
Variable Importance Scores
Benign Malignant MeanDecreaseAccuracy MeanDecreaseGini
radius_mean 12.200 9.580 14.810 9.426
texture_mean 9.540 11.339 13.903 3.591
perimeter_mean 11.886 9.362 14.389 10.630
area_mean 14.463 9.339 15.967 11.822
smoothness_mean 3.759 9.487 10.108 1.690
compactness_mean 7.622 6.804 10.185 3.818
concavity_mean 10.555 11.049 15.351 10.667
concave.points_mean 14.498 15.465 19.711 19.383
symmetry_mean 2.073 6.083 6.588 1.006
fractal_dimension_mean 6.412 2.553 6.740 1.412
radius_se 9.762 7.989 12.616 4.994
texture_se 4.495 4.218 6.182 1.309
perimeter_se 10.427 9.676 14.146 5.248
area_se 15.152 11.325 18.681 11.276
smoothness_se 2.807 0.065 2.344 1.257
compactness_se 7.110 2.090 7.350 1.823
concavity_se 6.739 5.378 8.638 2.270
concave.points_se 6.976 3.215 7.853 2.106
symmetry_se 3.506 2.038 4.160 1.089
fractal_dimension_se 3.061 -0.787 2.010 1.417
radius_worst 17.810 15.278 21.863 18.631
texture_worst 11.425 13.369 16.371 3.767
perimeter_worst 17.038 15.039 21.162 19.923
area_worst 17.822 16.286 22.461 19.553
smoothness_worst 10.016 11.760 14.205 3.221
compactness_worst 8.632 8.718 12.425 6.053
concavity_worst 11.687 13.906 17.776 9.603
concave.points_worst 16.993 17.340 22.737 20.964
symmetry_worst 7.198 9.309 11.445 2.537
fractal_dimension_worst 5.848 4.532 7.340 2.103
varImpPlot(
  rf_model,
  main = "Random Forest - Feature Importance",
  n.var = min(15, nrow(importance_df)),
  pch = 19,
  color = "steelblue"
)

6. Session Info

sessionInfo()
## R version 4.5.2 (2025-10-31)
## Platform: aarch64-apple-darwin20
## Running under: macOS Ventura 13.3
## 
## Matrix products: default
## BLAS:   /System/Library/Frameworks/Accelerate.framework/Versions/A/Frameworks/vecLib.framework/Versions/A/libBLAS.dylib 
## LAPACK: /Library/Frameworks/R.framework/Versions/4.5-arm64/Resources/lib/libRlapack.dylib;  LAPACK version 3.12.1
## 
## locale:
## [1] en_US.UTF-8/en_US.UTF-8/en_US.UTF-8/C/en_US.UTF-8/en_US.UTF-8
## 
## time zone: America/Chicago
## tzcode source: internal
## 
## attached base packages:
## [1] stats     graphics  grDevices utils     datasets  methods   base     
## 
## other attached packages:
## [1] knitr_1.51           caret_7.0-1          lattice_0.22-7      
## [4] ggplot2_4.0.3        randomForest_4.7-1.2
## 
## loaded via a namespace (and not attached):
##  [1] gtable_0.3.6         xfun_0.60            bslib_0.12.0        
##  [4] recipes_1.3.3        vctrs_0.7.3          tools_4.5.2         
##  [7] generics_0.1.4       stats4_4.5.2         parallel_4.5.2      
## [10] proxy_0.4-29         tibble_3.3.1         pkgconfig_2.0.3     
## [13] ModelMetrics_1.2.2.2 Matrix_1.7-4         data.table_1.18.4   
## [16] RColorBrewer_1.1-3   S7_0.2.2             lifecycle_1.0.5     
## [19] compiler_4.5.2       farver_2.1.2         stringr_1.6.0       
## [22] codetools_0.2-20     htmltools_0.5.9      class_7.3-23        
## [25] sass_0.4.10          yaml_2.3.12          prodlim_2026.03.11  
## [28] pillar_1.11.1        jquerylib_0.1.4      MASS_7.3-65         
## [31] cachem_1.1.0         gower_1.0.2          iterators_1.0.14    
## [34] rpart_4.1.24         foreach_1.5.2        nlme_3.1-168        
## [37] parallelly_1.48.0    lava_1.9.2           tidyselect_1.2.1    
## [40] digest_0.6.39        stringi_1.8.7        future_1.75.0       
## [43] dplyr_1.2.1          reshape2_1.4.5       purrr_1.2.2         
## [46] listenv_1.0.0        splines_4.5.2        fastmap_1.2.0       
## [49] grid_4.5.2           cli_3.6.6            magrittr_2.0.5      
## [52] survival_3.8-3       e1071_1.7-17         future.apply_1.20.2 
## [55] withr_3.0.3          scales_1.4.0         lubridate_1.9.5     
## [58] timechange_0.4.0     rmarkdown_2.31       globals_0.19.1      
## [61] nnet_7.3-20          timeDate_4052.112    evaluate_1.0.5      
## [64] hardhat_1.4.3        rlang_1.3.0          Rcpp_1.1.2          
## [67] glue_1.8.1           pROC_1.19.0.1        ipred_0.9-15        
## [70] jsonlite_2.0.0       R6_2.6.1             plyr_1.8.9