Assignment 2

Set the Working Directory

setwd("D:/Uni/Year 4 Sem 1/FIT3152/Assignment 2")

Access the library

#install.packages("tidyverse")
library(tidyverse)
## Warning: package 'tidyverse' was built under R version 4.2.3
## Warning: package 'ggplot2' was built under R version 4.2.3
## Warning: package 'tibble' was built under R version 4.2.3
## Warning: package 'tidyr' was built under R version 4.2.3
## Warning: package 'readr' was built under R version 4.2.3
## Warning: package 'purrr' was built under R version 4.2.3
## Warning: package 'dplyr' was built under R version 4.2.3
## Warning: package 'stringr' was built under R version 4.2.3
## Warning: package 'forcats' was built under R version 4.2.3
## Warning: package 'lubridate' was built under R version 4.2.3
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr     1.1.4     ✔ readr     2.1.5
## ✔ forcats   1.0.0     ✔ stringr   1.5.1
## ✔ ggplot2   3.5.0     ✔ tibble    3.2.1
## ✔ lubridate 1.9.3     ✔ tidyr     1.3.1
## ✔ purrr     1.0.2     
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag()    masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(ggplot2)

Read the data from the file and Create Dataset for The Analysis

rm(list = ls())
# read the data
Phish <- read.csv("PhishingData.csv", stringsAsFactors = TRUE)
set.seed(31225381) # Your Student ID is the random seed
L <- as.data.frame(c(1:50))
#L
L <- L[sample(nrow(L), 10, replace = FALSE),]
#L
Phish <- Phish[(Phish$A01 %in% L),]
PD <- Phish[sample(nrow(Phish), 2000, replace = FALSE),] # sample of 2000 rows

Question 1: Explore the dataset of Phishing

The summary only calculates the general descriptive statistic ## Descriptive Statistics

PD
# get the description of each attribute
summary(PD)
##       A01            A02               A03                A04       
##  Min.   : 4.0   Min.   : 0.0000   Min.   :0.000000   Min.   :2.000  
##  1st Qu.:17.0   1st Qu.: 0.0000   1st Qu.:0.000000   1st Qu.:2.000  
##  Median :35.0   Median : 0.0000   Median :0.000000   Median :3.000  
##  Mean   :30.8   Mean   : 0.1482   Mean   :0.000506   Mean   :2.767  
##  3rd Qu.:44.0   3rd Qu.: 0.0000   3rd Qu.:0.000000   3rd Qu.:3.000  
##  Max.   :47.0   Max.   :20.0000   Max.   :1.000000   Max.   :7.000  
##                 NA's   :16        NA's   :23         NA's   :20     
##       A05                A06             A07                A08        
##  Min.   : 0.00000   Min.   :0.000   Min.   :0.000000   Min.   :0.1667  
##  1st Qu.: 0.00000   1st Qu.:0.000   1st Qu.:0.000000   1st Qu.:0.6916  
##  Median : 0.00000   Median :0.000   Median :0.000000   Median :1.0000  
##  Mean   : 0.07284   Mean   :0.122   Mean   :0.002532   Mean   :0.8457  
##  3rd Qu.: 0.00000   3rd Qu.:0.000   3rd Qu.:0.000000   3rd Qu.:1.0000  
##  Max.   :97.00000   Max.   :1.000   Max.   :1.000000   Max.   :1.0000  
##  NA's   :23         NA's   :25      NA's   :25         NA's   :20      
##       A09               A10               A11                A12       
##  Min.   :0.00000   Min.   :0.00000   Min.   :  0.0000   Min.   :117.0  
##  1st Qu.:0.00000   1st Qu.:0.00000   1st Qu.:  0.0000   1st Qu.:232.0  
##  Median :0.00000   Median :0.00000   Median :  0.0000   Median :232.0  
##  Mean   :0.02833   Mean   :0.04529   Mean   :  0.1297   Mean   :317.5  
##  3rd Qu.:0.00000   3rd Qu.:0.00000   3rd Qu.:  0.0000   3rd Qu.:418.0  
##  Max.   :1.00000   Max.   :1.00000   Max.   :176.0000   Max.   :692.0  
##  NA's   :23        NA's   :13        NA's   :18         NA's   :18     
##       A13                A14              A15              A16        
##  Min.   :  0.0000   Min.   :0.0000   Min.   :0.0000   Min.   :0.0000  
##  1st Qu.:  0.0000   1st Qu.:0.0000   1st Qu.:0.0000   1st Qu.:0.0000  
##  Median :  0.0000   Median :0.0000   Median :0.0000   Median :0.0000  
##  Mean   :  0.1694   Mean   :0.1594   Mean   :0.1367   Mean   :0.0398  
##  3rd Qu.:  0.0000   3rd Qu.:0.0000   3rd Qu.:0.0000   3rd Qu.:0.0000  
##  Max.   :291.0000   Max.   :1.0000   Max.   :1.0000   Max.   :1.0000  
##  NA's   :16         NA's   :18       NA's   :18       NA's   :15      
##       A17             A18               A19              A20        
##  Min.   :0.000   Min.   :   5.00   Min.   :0.0000   Min.   :0.0000  
##  1st Qu.:1.000   1st Qu.:  12.00   1st Qu.:0.0000   1st Qu.:0.0000  
##  Median :1.000   Median :  31.00   Median :0.0000   Median :0.0000  
##  Mean   :1.153   Mean   :  57.22   Mean   :0.1065   Mean   :0.2352  
##  3rd Qu.:1.000   3rd Qu.:  89.00   3rd Qu.:0.0000   3rd Qu.:0.0000  
##  Max.   :6.000   Max.   :1017.00   Max.   :1.0000   Max.   :1.0000  
##  NA's   :16      NA's   :19        NA's   :18       NA's   :10      
##       A21               A22               A23               A24          
##  Min.   :0.00000   Min.   :0.01264   Min.   :   0.00   Min.   :0.000000  
##  1st Qu.:0.00000   1st Qu.:0.05103   1st Qu.:   2.00   1st Qu.:0.007505  
##  Median :0.00000   Median :0.05804   Median :  39.00   Median :0.079963  
##  Mean   :0.02375   Mean   :0.05596   Mean   :  60.05   Mean   :0.266114  
##  3rd Qu.:0.00000   3rd Qu.:0.06296   3rd Qu.: 100.00   3rd Qu.:0.522907  
##  Max.   :2.00000   Max.   :0.08039   Max.   :1868.00   Max.   :0.522907  
##  NA's   :21        NA's   :19        NA's   :27        NA's   :14        
##       A25               Class      
##  Min.   :0.000000   Min.   :0.000  
##  1st Qu.:0.000000   1st Qu.:0.000  
##  Median :0.000000   Median :0.000  
##  Mean   :0.000171   Mean   :0.428  
##  3rd Qu.:0.000000   3rd Qu.:1.000  
##  Max.   :0.143000   Max.   :1.000  
##  NA's   :15
dim(PD)
## [1] 2000   26

With the use of the “pastecs” library, it calculates the in-depth statistics.

#install.packages("pastecs")
library(pastecs)
## Warning: package 'pastecs' was built under R version 4.2.3
## 
## Attaching package: 'pastecs'
## The following objects are masked from 'package:dplyr':
## 
##     first, last
## The following object is masked from 'package:tidyr':
## 
##     extract
round(stat.desc(PD), 4)

Structure of The Dataset

str(PD)
## 'data.frame':    2000 obs. of  26 variables:
##  $ A01  : int  44 47 44 8 17 45 45 47 47 34 ...
##  $ A02  : int  0 0 5 0 0 3 0 0 0 0 ...
##  $ A03  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A04  : int  3 3 3 3 3 3 2 3 3 3 ...
##  $ A05  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A06  : int  0 0 0 0 0 0 0 0 0 NA ...
##  $ A07  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A08  : num  1 1 0.593 0.818 1 ...
##  $ A09  : int  0 0 NA 0 1 0 0 0 0 0 ...
##  $ A10  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A11  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A12  : int  232 504 232 171 232 232 379 232 232 482 ...
##  $ A13  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A14  : int  0 0 0 0 0 0 0 1 0 0 ...
##  $ A15  : int  0 0 0 1 0 0 0 0 0 0 ...
##  $ A16  : int  0 0 0 0 0 NA 0 0 0 1 ...
##  $ A17  : int  1 1 1 1 1 1 1 1 0 1 ...
##  $ A18  : int  78 24 16 11 97 212 9 98 12 24 ...
##  $ A19  : int  0 0 1 0 0 0 0 0 0 0 ...
##  $ A20  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A21  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A22  : num  0.0623 0.073 0.0489 0.0463 0.0664 ...
##  $ A23  : int  71 107 1 100 113 29 100 32 0 169 ...
##  $ A24  : num  0.5229 0.08 0.5229 0.0018 0.5229 ...
##  $ A25  : num  0 0 0 0 0 0 0 0 0 0 ...
##  $ Class: int  1 1 0 0 0 1 0 1 0 0 ...

The probability of phishing sites and probability of legitimate sites among all observations

counter_phishing = 0
counter_legitimate = 0
site_observation = count(PD)
for (site in PD$Class){
  if(site == 0)
  {
    counter_legitimate = counter_legitimate + 1
  }
  else
  {
    counter_phishing = counter_phishing + 1
  }
}
proportion.legitimateSites = as.numeric(round(counter_legitimate / site_observation, 2))
proportion.phishingSites = as.numeric(round(counter_phishing / site_observation, 2))
cat("The number of phishing sites is \n")
## The number of phishing sites is
print(counter_phishing)
## [1] 856
cat("The number of legitimate sites is \n")
## The number of legitimate sites is
print(counter_legitimate)
## [1] 1144
cat("Proportion of phishing sites is \n")
## Proportion of phishing sites is
print(proportion.phishingSites)
## [1] 0.43
cat("Proportion of legitimate sites is \n")
## Proportion of legitimate sites is
print(proportion.legitimateSites)
## [1] 0.57

According to the calculation, it can be seen that most of the observations tend to be identified as legitimate rather than phishing. It can also be proved by the use of the visualization.

ggplot(PD, aes(x = as.factor(Class))) +
  geom_bar(fill = c("blue", "red")) +
  labs(title = "Website Categories",
       x = "Category",
       y = "Frequency") +
  theme_minimal()

Check the number of missing values

# calculate the number of missing values in each column
colSums(is.na(PD))
##   A01   A02   A03   A04   A05   A06   A07   A08   A09   A10   A11   A12   A13 
##     0    16    23    20    23    25    25    20    23    13    18    18    16 
##   A14   A15   A16   A17   A18   A19   A20   A21   A22   A23   A24   A25 Class 
##    18    18    15    16    19    18    10    21    19    27    14    15     0

Omit any attributes based on the variance

The approach of omitting the number of attributes is by taking low variance based on the calculation I have done previously.

manipulated_PD <- PD %>% 
  select(-A03, -A07, -A22, -A25)
manipulated_PD

The attributes I am going to use for the next analysis is A01, A02, A04, A05, A06, A08, A09, A10, A11, A12, A13, A14, A15, A16, A17, A18, A19, A20, A21, A23, A24, and Class. What resonates with the use of these attributes is the low variances from A03, A07, A22, and A25. Because they are not varied and widespread, it is not suitable for the case of the classification. Therefore, I decide to make use of these attributes.

Question 2: Pre-Processing

Remove any NA values in the table

#manipulated_PD <- na.omit(manipulated_PD)
manipulated_PD = manipulated_PD[complete.cases(manipulated_PD),]
manipulated_PD

scale these numeric variables

There is no need to scale the data as the classification model and ANN do not concern how big or small the number is that causes the superiority. This feature can only be applied to the case that has a high-level concern over the superiority like “clustering”

#manipulated_PD[, 1:21] <- scale(manipulated_PD[, 1:21])
#manipulated_PD

Convert data type for Class to factor

#install.packages("car")
manipulated_PD$Class = as.factor(manipulated_PD$Class)
str(manipulated_PD)
## 'data.frame':    1659 obs. of  22 variables:
##  $ A01  : int  44 47 8 17 45 47 47 26 47 47 ...
##  $ A02  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A04  : int  3 3 3 3 2 3 3 3 3 3 ...
##  $ A05  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A06  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A08  : num  1 1 0.818 1 0.571 ...
##  $ A09  : int  0 0 0 1 0 0 0 0 0 0 ...
##  $ A10  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A11  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A12  : int  232 504 171 232 379 232 232 232 504 504 ...
##  $ A13  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A14  : int  0 0 0 0 0 1 0 0 1 0 ...
##  $ A15  : int  0 0 1 0 0 0 0 0 0 0 ...
##  $ A16  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A17  : int  1 1 1 1 1 1 0 1 1 1 ...
##  $ A18  : int  78 24 11 97 9 98 12 94 98 20 ...
##  $ A19  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A20  : int  0 0 0 0 0 0 0 0 0 1 ...
##  $ A21  : int  0 0 0 0 0 0 0 0 0 0 ...
##  $ A23  : int  71 107 100 113 100 32 0 110 110 40 ...
##  $ A24  : num  0.52291 0.07996 0.0018 0.52291 0.00122 ...
##  $ Class: Factor w/ 2 levels "0","1": 2 2 1 1 1 2 1 1 2 2 ...
manipulated_PD

Question 3: Divide Data Into 70% Training and 30% Testing

Train and Test Dataset by 70% and 30% Respectively

# set the seed with student id for reproducing the results randomly
set.seed(31225381)
train.row = sample(1:nrow(manipulated_PD), 0.7*nrow(manipulated_PD))
PD_train = manipulated_PD[train.row,]
PD_test = manipulated_PD[-train.row,]

View the content of train.row, the training dataset and the testing dataset

train.row
##    [1]  510  196 1196  602  684  611   97  132  222  528 1159  241 1658  174
##   [15] 1148 1244 1355 1208  453 1205  843  102  907  870 1043  458  604  677
##   [29]  697 1331 1036 1009  791 1544 1592 1269  727  168  804 1256  723 1408
##   [43]  629 1259 1231 1361  849  933  512  967 1130 1020  144 1597 1178  541
##   [57] 1216  786 1299 1252  800  182  572  586  889  487  269  420 1129  670
##   [71] 1446   48  834  501 1560  473  882 1233  124  919  107  781  582   54
##   [85] 1090 1497  221 1442 1541 1012  659 1423  937 1502  266  813  970  728
##   [99]  378 1104  811 1063  493  188  248  140  765  257 1444  608 1435 1441
##  [113] 1191  916 1484 1122  975 1149  306 1060 1290   71  855 1348 1625  210
##  [127]  923  776 1094  253  336  798 1558 1175 1265  382  486 1170  553  612
##  [141] 1163 1017 1445 1543 1450 1515 1153  900  702 1593  496 1508 1123  327
##  [155]  284  678  707 1443 1119  927  443 1067  979 1353 1649 1228  289  857
##  [169] 1027  324 1133 1453 1283  218  614 1106  640   88  789  628  125 1236
##  [183] 1195 1517  551  468  429  852  152 1401  773  930  605 1070 1330 1551
##  [197]  349  213  267  790  106 1141 1333  892  663 1197 1075 1209  595 1482
##  [211] 1118 1229 1193 1528 1303  672 1114   79  799 1552  454  370  520  185
##  [225]  720  936 1051 1388  906   99  295  817  850  488  929  459  178  100
##  [239]  385  696 1268 1336 1277 1217 1481 1350  198  679 1146  235  744 1594
##  [253] 1010 1616 1566 1214  298 1073  268 1610 1139 1320 1576 1111  539 1525
##  [267] 1248 1514 1367 1194 1556 1619 1375  175 1521    6  917 1121  425 1325
##  [281]  137  986 1506  435  835  448 1440  579 1332   34  607 1155  918  293
##  [295]  599 1169 1578  518  913 1232 1304  689  136  748 1377 1127   76 1251
##  [309]  194  163 1313 1239   38 1168  842  101 1429  272 1606  875  803  756
##  [323] 1189 1222 1025  361   63  555 1559 1569 1615 1587 1607  315  110  981
##  [337] 1645  203 1296 1298  301 1425 1369 1037   78   28  818 1109 1288 1034
##  [351]  525 1187  815  281  973 1653 1107 1489  560   77 1186 1092 1368  397
##  [365]  291  120 1014  795 1258  742  232  564  335 1091 1223 1328 1316  165
##  [379]  905 1432  760  264  575 1623  273  287  495  644  538 1650 1373 1086
##  [393]  725  166  965  345   43 1654  768   61  861  686  381  134  223  466
##  [407]  498   73 1021  669 1204 1307 1165  609 1385  226  150  802  422   85
##  [421] 1648 1393 1016  649  427  470  825  635  548 1142  746  949  187 1613
##  [435]  828  540   91 1397  212  617 1160 1074  355 1080 1211  252  714 1419
##  [449]  662  983  138  561  316 1053  729 1424  990  393 1631  565   33 1062
##  [463] 1533  709  606 1247   40  375 1003  329 1210  170  363  764 1421 1302
##  [477] 1415  695  181 1024 1310  147 1266  594  827  108  462 1378  903   30
##  [491]  647 1498  598  325  290  574 1351  532 1319  772 1584 1068  184  259
##  [505] 1470  711  968 1147  406  280  947  621 1207 1448  526  770 1055 1624
##  [519] 1087 1292  940  309  377  475   60 1534  954  514 1535  920  130 1061
##  [533] 1362 1028 1134   68 1305 1463   51  262  895 1243 1019  395 1356  961
##  [547] 1128 1026  793 1400 1618 1488  407  673  650  926  562 1499  492 1465
##  [561] 1156 1364 1475 1601 1571 1609  250  365 1505 1308  735  430  712  736
##  [575] 1387  285 1449  402 1190 1452  119  193  683   45  763 1407 1264  693
##  [589] 1124  701  874  636  366   17  339   12  491 1643 1273 1511 1044  154
##  [603] 1520 1054 1254  769 1093  552  616 1582  657  777  398 1237 1628  476
##  [617]  228  115 1235 1293  989    7 1512  648  996  403  334 1311 1374  423
##  [631] 1040  186  883  354  146  935  340   31  292  438   81 1483  372  206
##  [645]  481 1192  741  569   14  757  589  837 1172  271  499  877  880  739
##  [659]  312 1572  367  283  784 1436  219 1529 1509 1314 1263  554  263  246
##  [673]  866 1049  227  317 1279 1567  807  341 1038  758  161  785 1565 1652
##  [687] 1412 1640  216  960  680 1545  583  660   46   98  483  389  469   92
##  [701]  356  896  904 1485  353  944  597  921  368  121  978  419  417  591
##  [715]  705  439  816 1395  809   18  864  533  893  715   56  632 1039 1564
##  [729] 1492 1029 1460 1469  156 1224  664  418 1219  821   32  667  887  167
##  [743]  231  884  897  254 1103  991  987  601 1477 1651  687  627 1215 1225
##  [757]  651   80  956  637 1438  337  953  909 1253  824 1359  171  472  195
##  [771]  566 1570 1326  962 1622  244  860 1516  350 1548  180  392  706 1115
##  [785] 1513 1143  386  952 1532 1461 1085   89 1390 1487  265  237   21  311
##  [799]  829 1132  894    2 1457  787  531  820 1474 1583 1076 1504 1476 1341
##  [813]  779 1030  836 1405  549 1370   90 1230  457  792 1112  201  208 1001
##  [827]  114   20  925  942  775  529 1456  563 1006 1226 1180  357 1059    3
##  [841] 1335  322  524  256  690  808  788 1409 1414 1057 1599 1238  618  260
##  [855]  885 1426  762 1496 1282  465 1245  666 1346 1213  822  113  780  945
##  [869]   47 1338  369 1246  726 1183 1384  484  810  396 1077  623 1608  503
##  [883] 1334  854   25 1455 1554 1154  189  639 1272  275  225  391 1181  676
##  [897] 1152 1417  914 1052 1611 1164 1199 1278  747  869  330  969 1047 1203
##  [911]  358 1113  346 1162 1218  698 1431  872 1549  523  590 1614  643 1360
##  [925] 1150 1241 1048  333  881  550  111  557 1007 1500   24  296 1295 1234
##  [939] 1227  197  542 1557   87  401 1110  409  191  238 1285 1167 1416 1079
##  [953]  509 1379 1177 1337  980  494  738 1108 1309  774   70 1363 1546 1315
##  [967] 1553  143   41  665 1011 1503 1427   58 1644 1538  721 1527  584  688
##  [981]  988   86  505  303  305  199  482  570  164  504  383  699 1466  704
##  [995]  117  766 1531 1659 1638  848  915  126  948 1376 1433  928 1270  534
## [1009] 1603  819 1630 1641    9 1418  694  421  964 1324  943  288  958  360
## [1023]   66 1099  105  416 1354 1526  145 1045  674  814  536 1345  867 1089
## [1037] 1035  938  888  805 1352   57  220  323 1013 1398  844  838  343  249
## [1051]  658 1537  131 1005  517  571 1212 1301  400  997   15 1380  408 1633
## [1065] 1420   62 1468  708 1125 1404 1589  455  399 1620  671  413  862 1598
## [1079]  593 1145  630  116 1447 1344 1179 1069  911  277  682  891 1201  411
## [1093]  759 1642  202 1339   16 1323  331  373 1563  839  497 1574 1491  719
## [1107]  750  641  500 1406 1596  740  362 1501   94 1327 1126 1329  347   96
## [1121] 1347 1396 1255  901  899 1317  890  230 1015  215    8  135   27  567
## [1135]  782  302  767   49  431  868 1386  685  318  192  863   93  234  310
## [1149]  464  634  352  460  537 1173  516  432  845  507 1626 1144 1151
PD_train # dataset trained by 70%
PD_test # dataset test by 30%

Question 4, 5, 6, and 7: Calculate All Classification Methods

Library Package for Classification Model

#install.packages("tree")
library(tree)
## Warning: package 'tree' was built under R version 4.2.3
#install.packages("e1071") 
library(e1071) 
## Warning: package 'e1071' was built under R version 4.2.3
#install.packages(("ROCR")) 
library(ROCR) 
## Warning: package 'ROCR' was built under R version 4.2.3
#install.packages("randomForest") 
library(randomForest) 
## Warning: package 'randomForest' was built under R version 4.2.3
## randomForest 4.7-1.1
## Type rfNews() to see new features/changes/bug fixes.
## 
## Attaching package: 'randomForest'
## The following object is masked from 'package:dplyr':
## 
##     combine
## The following object is masked from 'package:ggplot2':
## 
##     margin
#install.packages("adabag") 
library(adabag)
## Warning: package 'adabag' was built under R version 4.2.3
## Loading required package: rpart
## Warning: package 'rpart' was built under R version 4.2.3
## Loading required package: caret
## Warning: package 'caret' was built under R version 4.2.3
## Loading required package: lattice
## 
## Attaching package: 'caret'
## The following object is masked from 'package:purrr':
## 
##     lift
## Loading required package: foreach
## Warning: package 'foreach' was built under R version 4.2.3
## 
## Attaching package: 'foreach'
## The following objects are masked from 'package:purrr':
## 
##     accumulate, when
## Loading required package: doParallel
## Warning: package 'doParallel' was built under R version 4.2.3
## Loading required package: iterators
## Warning: package 'iterators' was built under R version 4.2.3
## Loading required package: parallel
#install.packages("rpart") 
library(rpart)
#install.packages("rpart.plot")
library(rpart.plot)
## Warning: package 'rpart.plot' was built under R version 4.2.3

Decision Tree

Create decision Tree

PD.tree = tree(Class~., data = PD_train)
# to see all datas in each node
PD.tree
## node), split, n, deviance, yval, (yprob)
##       * denotes terminal node
## 
##  1) root 1161 1592.0 0 ( 0.5616 0.4384 )  
##    2) A18 < 14.5 341  291.2 0 ( 0.8475 0.1525 ) *
##    3) A18 > 14.5 820 1126.0 1 ( 0.4427 0.5573 )  
##      6) A01 < 30 309  371.0 0 ( 0.7120 0.2880 )  
##       12) A01 < 12.5 142  122.5 0 ( 0.8451 0.1549 ) *
##       13) A01 > 12.5 167  224.9 0 ( 0.5988 0.4012 ) *
##      7) A01 > 30 511  605.8 1 ( 0.2798 0.7202 )  
##       14) A23 < 2.5 90  113.1 0 ( 0.6778 0.3222 ) *
##       15) A23 > 2.5 421  415.2 1 ( 0.1948 0.8052 )  
##         30) A01 < 39 135  169.0 1 ( 0.3185 0.6815 ) *
##         31) A01 > 39 286  227.8 1 ( 0.1364 0.8636 ) *
print(summary(PD.tree))
## 
## Classification tree:
## tree(formula = Class ~ ., data = PD_train)
## Variables actually used in tree construction:
## [1] "A18" "A01" "A23"
## Number of terminal nodes:  6 
## Residual mean deviance:  0.9944 = 1149 / 1155 
## Misclassification error rate: 0.2171 = 252 / 1161
# draw the decision tree
plot(PD.tree)
# set the text size
text(PD.tree, pretty = 0)

In term of how to assign the label, the decision starts from the main root where it checks the record whether A18 is lower than 14.5. The root node is splitted into two branches on two sides (2nd and 3rd nodes). If the check is “YES”, then it is traversed to the left (assigning the 0 to the record as the conclusion). As it does not meet the condition, it goes to the node 3 (A18 > 14.5) in transition to the branch where it has 2 nodes (node 6 and node 7). Deciding on which node the record needs to point to, it relies on the condition of whether A01 < 30. If it meets, it is traversed to the left (node 6). Node 6 is split into two nodes (node 12 and node 13) with conditions (A1 < 12.5 and A1 > 12.5; going to the same label). As the condition fails to satisfies the condition, it visits the node 7 and then this node is divided into two nodes (node 14 and node 15) following the condition of whether A023 is lower than 2.5. As it is lower, then it stops. When the input is bigger than 2.5, it refers to the node 15 which in turn is separated into two nodes (node 30 and node 31). These nodes obey the condition rule (“A01 is either lower or greater than 39”). Choosing lower or higher than the expected number, it corresponds to the same outcome (i.e. 1).

Confusion Matrix and Accuracy (Prediction)

PD.predtree = predict(PD.tree, PD_test, type = "class")
tableDecisionTreeUsingTree = table(predicted = PD.predtree, actual = PD_test$Class)
cat("Decision Tree Confusion\n")
## Decision Tree Confusion
print(tableDecisionTreeUsingTree)
##          actual
## predicted   0   1
##         0 265  64
##         1  36 133
confusionMatrix(tableDecisionTreeUsingTree)
## Confusion Matrix and Statistics
## 
##          actual
## predicted   0   1
##         0 265  64
##         1  36 133
##                                           
##                Accuracy : 0.7992          
##                  95% CI : (0.7613, 0.8335)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : < 2.2e-16       
##                                           
##                   Kappa : 0.5695          
##                                           
##  Mcnemar's Test P-Value : 0.006934        
##                                           
##             Sensitivity : 0.8804          
##             Specificity : 0.6751          
##          Pos Pred Value : 0.8055          
##          Neg Pred Value : 0.7870          
##              Prevalence : 0.6044          
##          Detection Rate : 0.5321          
##    Detection Prevalence : 0.6606          
##       Balanced Accuracy : 0.7778          
##                                           
##        'Positive' Class : 0               
## 

Calculating Accuracy of Decision Tree Using Tree Library

cat("\nDecision Tree Accuracy\n")
## 
## Decision Tree Accuracy
tableDecisionTreeUsingTree.accuracy = (265 + 133) / (265 + 64 + 36 + 133)
tableDecisionTreeUsingTree.accuracy
## [1] 0.7991968

Measuring the accuracy can be completed by taking the diagonal and summing the value divided by the total of matrix. In the case of the “decision tree” accuracy, this implies that 0.7991 => 79.91% Its accuracy is 0.7991 meaning that 79.91% correct predictions among all observations.

Probabilities and Draw ROC (Prediction)

# do predictions as probabilities 
PD.pred.tree = predict(PD.tree, PD_test, type = "vector")
# computing a simple ROC curve (x-axis: fpr, y-axis = tpr)
# labels are actual values, predictors are probability of class
PDpred <- prediction(PD.pred.tree[, 2], PD_test$Class)
PDperf <- performance(PDpred, "tpr", "fpr")
plot(PDperf)
abline(0,1, lty = 2)

Calculate AUC of Decision Tree Using Tree

PDcauc.DecisionTreeUsingTree = performance(PDpred, "auc")
print(as.numeric(PDcauc.DecisionTreeUsingTree@y.values))
## [1] 0.8190887

The AUC of Decision Tree is 0.819 meaning that there is a 81.9% chance that the model will be able to distinguish between positive and negative class.

Decision Tree using rpart

PD.decisiontree = rpart(Class~., data = PD_train, method = "class")
print(summary(PD.decisiontree))
## Call:
## rpart(formula = Class ~ ., data = PD_train, method = "class")
##   n= 1161 
## 
##           CP nsplit rel error    xerror       xstd
## 1 0.22102161      0 1.0000000 1.0000000 0.03321611
## 2 0.06286837      2 0.5579568 0.5579568 0.02877565
## 3 0.01080550      3 0.4950884 0.5147348 0.02798315
## 4 0.01000000      5 0.4734774 0.5166994 0.02802090
## 
## Variable importance
## A23 A01 A18 A17 A08 A24 A04 A02 A11 
##  33  29  28   2   2   1   1   1   1 
## 
## Node number 1: 1161 observations,    complexity param=0.2210216
##   predicted class=0  expected loss=0.4384152  P(node) =1
##     class counts:   652   509
##    probabilities: 0.562 0.438 
##   left son=2 (341 obs) right son=3 (820 obs)
##   Primary splits:
##       A18 < 14.5       to the left,  improve=78.94041, (0 missing)
##       A01 < 21.5       to the left,  improve=60.95977, (0 missing)
##       A23 < 0.5        to the left,  improve=49.47025, (0 missing)
##       A14 < 0.5        to the left,  improve=48.43386, (0 missing)
##       A20 < 0.5        to the left,  improve=33.56203, (0 missing)
##   Surrogate splits:
##       A23 < 0.5        to the left,  agree=0.793, adj=0.296, (0 split)
##       A24 < 0.0003907  to the left,  agree=0.720, adj=0.047, (0 split)
##       A17 < 0.5        to the left,  agree=0.717, adj=0.038, (0 split)
##       A04 < 3.5        to the right, agree=0.714, adj=0.026, (0 split)
##       A08 < 0.3452083  to the left,  agree=0.707, adj=0.003, (0 split)
## 
## Node number 2: 341 observations
##   predicted class=0  expected loss=0.1524927  P(node) =0.2937123
##     class counts:   289    52
##    probabilities: 0.848 0.152 
## 
## Node number 3: 820 observations,    complexity param=0.2210216
##   predicted class=1  expected loss=0.4426829  P(node) =0.7062877
##     class counts:   363   457
##    probabilities: 0.443 0.557 
##   left son=6 (309 obs) right son=7 (511 obs)
##   Primary splits:
##       A01 < 30         to the left,  improve=71.91603, (0 missing)
##       A14 < 0.5        to the left,  improve=22.85848, (0 missing)
##       A23 < 0.5        to the left,  improve=22.00819, (0 missing)
##       A24 < 0.00178155 to the left,  improve=21.70071, (0 missing)
##       A08 < 0.568323   to the left,  improve=15.04439, (0 missing)
##   Surrogate splits:
##       A23 < 99.5       to the right, agree=0.771, adj=0.392, (0 split)
##       A02 < 2.5        to the right, agree=0.630, adj=0.019, (0 split)
##       A09 < 0.5        to the right, agree=0.629, adj=0.016, (0 split)
##       A17 < 3.5        to the right, agree=0.626, adj=0.006, (0 split)
##       A05 < 9          to the right, agree=0.624, adj=0.003, (0 split)
## 
## Node number 6: 309 observations,    complexity param=0.0108055
##   predicted class=0  expected loss=0.2880259  P(node) =0.2661499
##     class counts:   220    89
##    probabilities: 0.712 0.288 
##   left son=12 (142 obs) right son=13 (167 obs)
##   Primary splits:
##       A01 < 12.5       to the left,  improve=9.308772, (0 missing)
##       A24 < 0.00169035 to the left,  improve=3.798886, (0 missing)
##       A23 < 132.5      to the left,  improve=3.207926, (0 missing)
##       A18 < 112.5      to the left,  improve=2.464323, (0 missing)
##       A14 < 0.5        to the left,  improve=1.844179, (0 missing)
##   Surrogate splits:
##       A23 < 97         to the right, agree=0.625, adj=0.183, (0 split)
##       A04 < 2.5        to the left,  agree=0.563, adj=0.049, (0 split)
##       A02 < 0.5        to the right, agree=0.557, adj=0.035, (0 split)
##       A24 < 0.0125524  to the left,  agree=0.557, adj=0.035, (0 split)
##       A08 < 0.344065   to the left,  agree=0.550, adj=0.021, (0 split)
## 
## Node number 7: 511 observations,    complexity param=0.06286837
##   predicted class=1  expected loss=0.2798434  P(node) =0.4401378
##     class counts:   143   368
##    probabilities: 0.280 0.720 
##   left son=14 (90 obs) right son=15 (421 obs)
##   Primary splits:
##       A23 < 2.5        to the left,  improve=34.596660, (0 missing)
##       A08 < 0.568323   to the left,  improve=21.577230, (0 missing)
##       A24 < 0.00619405 to the left,  improve=16.310730, (0 missing)
##       A12 < 137.5      to the left,  improve= 8.362356, (0 missing)
##       A17 < 0.5        to the left,  improve= 6.383684, (0 missing)
##   Surrogate splits:
##       A08 < 0.568323   to the left,  agree=0.855, adj=0.178, (0 split)
##       A17 < 0.5        to the left,  agree=0.841, adj=0.100, (0 split)
##       A11 < 1.5        to the right, agree=0.832, adj=0.044, (0 split)
##       A04 < 3.5        to the right, agree=0.830, adj=0.033, (0 split)
##       A12 < 137.5      to the left,  agree=0.830, adj=0.033, (0 split)
## 
## Node number 12: 142 observations
##   predicted class=0  expected loss=0.1549296  P(node) =0.1223084
##     class counts:   120    22
##    probabilities: 0.845 0.155 
## 
## Node number 13: 167 observations,    complexity param=0.0108055
##   predicted class=0  expected loss=0.4011976  P(node) =0.1438415
##     class counts:   100    67
##    probabilities: 0.599 0.401 
##   left son=26 (134 obs) right son=27 (33 obs)
##   Primary splits:
##       A23 < 118.5      to the left,  improve=5.796735, (0 missing)
##       A24 < 0.00169035 to the left,  improve=5.663763, (0 missing)
##       A18 < 17.5       to the right, improve=2.504090, (0 missing)
##       A20 < 0.5        to the left,  improve=2.413963, (0 missing)
##       A02 < 0.5        to the left,  improve=1.899394, (0 missing)
##   Surrogate splits:
##       A02 < 5.5        to the left,  agree=0.808, adj=0.03, (0 split)
##       A16 < 0.5        to the left,  agree=0.808, adj=0.03, (0 split)
## 
## Node number 14: 90 observations
##   predicted class=0  expected loss=0.3222222  P(node) =0.07751938
##     class counts:    61    29
##    probabilities: 0.678 0.322 
## 
## Node number 15: 421 observations
##   predicted class=1  expected loss=0.1947743  P(node) =0.3626184
##     class counts:    82   339
##    probabilities: 0.195 0.805 
## 
## Node number 26: 134 observations
##   predicted class=0  expected loss=0.3358209  P(node) =0.1154177
##     class counts:    89    45
##    probabilities: 0.664 0.336 
## 
## Node number 27: 33 observations
##   predicted class=1  expected loss=0.3333333  P(node) =0.02842377
##     class counts:    11    22
##    probabilities: 0.333 0.667 
## 
## n= 1161 
## 
## node), split, n, loss, yval, (yprob)
##       * denotes terminal node
## 
##  1) root 1161 509 0 (0.5615848 0.4384152)  
##    2) A18< 14.5 341  52 0 (0.8475073 0.1524927) *
##    3) A18>=14.5 820 363 1 (0.4426829 0.5573171)  
##      6) A01< 30 309  89 0 (0.7119741 0.2880259)  
##       12) A01< 12.5 142  22 0 (0.8450704 0.1549296) *
##       13) A01>=12.5 167  67 0 (0.5988024 0.4011976)  
##         26) A23< 118.5 134  45 0 (0.6641791 0.3358209) *
##         27) A23>=118.5 33  11 1 (0.3333333 0.6666667) *
##      7) A01>=30 511 143 1 (0.2798434 0.7201566)  
##       14) A23< 2.5 90  29 0 (0.6777778 0.3222222) *
##       15) A23>=2.5 421  82 1 (0.1947743 0.8052257) *
rpart.plot(PD.decisiontree, main = "Decision Tree for Phishing Data")

In terms of the concept of this decision tree, the start point is always similar to the previous case (i.e. begins with the root split into two nodes (node 2 and node 3)). So The root verifies whether A18 is lower than 14.5. As the fulfillment is true, then the observation is labelled “0”. If not, then it is directed to node 3, divided into 2 nodes (node 6 and node 7 following the condition of whether A01 < 30): a. If the record of A01 is lower than 30, it visits node 6 that is split into node 12 (A01 < 12.5) and node 13 (A01 >= 12.5). As it is lower than 12.5, then the label is 0. If not, then it goes to the node 13, split into two nodes (node 26 and node 27). If the record of A23 is lower than 118.5 / 119, then it is marked as 0. if not, it is traversed to the right and labelled as 1. b. If the record of A01 is greater than or equal to 30, then it visits node 7, divided into node 14 and no 15 following the condition of whether A23 < 2.5. As the record of A23 < 2.5, it is traversed to the left and marked as 0. When A23’s record is greater than or equal to 2.5, then the label for the observation is 1.

Confusion Matrix of Decision Tree

PD.preddecisiontree = predict(PD.decisiontree, PD_test, type = "class")
tableDecisionTreeUsingRpart = table(predicted = PD.preddecisiontree, actual = PD_test$Class)
cat("Decision Tree Confusion Using Rpart Library\n")
## Decision Tree Confusion Using Rpart Library
print(tableDecisionTreeUsingRpart)
##          actual
## predicted   0   1
##         0 262  60
##         1  39 137
confusionMatrix(tableDecisionTreeUsingRpart)
## Confusion Matrix and Statistics
## 
##          actual
## predicted   0   1
##         0 262  60
##         1  39 137
##                                           
##                Accuracy : 0.8012          
##                  95% CI : (0.7634, 0.8354)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : < 2e-16         
##                                           
##                   Kappa : 0.5765          
##                                           
##  Mcnemar's Test P-Value : 0.04442         
##                                           
##             Sensitivity : 0.8704          
##             Specificity : 0.6954          
##          Pos Pred Value : 0.8137          
##          Neg Pred Value : 0.7784          
##              Prevalence : 0.6044          
##          Detection Rate : 0.5261          
##    Detection Prevalence : 0.6466          
##       Balanced Accuracy : 0.7829          
##                                           
##        'Positive' Class : 0               
## 

Calculating accuracy of Decision Tree Using Rpart

tableDecisionTreeUsingRpart.accuracy = (262 + 137) / (262 + 60 + 39 + 137)
cat("Decision Tree Accuracy Using Rpart \n")
## Decision Tree Accuracy Using Rpart
print(tableDecisionTreeUsingRpart.accuracy)
## [1] 0.8012048

The accuracy of Decision Tree using Rpart is 0.8012 meaning that 80.12% correct predictions of all observations

Probability and Draw ROC

# do predictions as probabilities 
PD.pred.rpart = predict(PD.decisiontree, PD_test, type = "prob")
# computing a simple ROC curve (x-axis: fpr, y-axis = tpr)
# labels are actual values, predictors are probability of class
PDRpartpred <- prediction(PD.pred.rpart[, 2], PD_test$Class)
PDRpartperf <- performance(PDRpartpred, "tpr", "fpr")
plot(PDRpartperf, main = "ROC for Decision Tree Using Rpart")
abline(0,1, lty = 2)

### Calculate AUC of Decision Tree Using Rpart

cat("AUC of 'Rpart' Decision Tree \n")
## AUC of 'Rpart' Decision Tree
PDcauc.DecisionTreeUsingRpart = performance(PDRpartpred, "auc")
decisiontree.auc = as.numeric(PDcauc.DecisionTreeUsingRpart@y.values)
print(decisiontree.auc)
## [1] 0.8111203

The AUC of Decision Tree is 0.811 meaning that there is a 81.11% chance that the model will be able to distinguish between positive and negative class.

Decision Tree Performance Conclusion

For the performance comparison, the tree-library-based decision tree shows better performance than the rpart-library-based one with respect to the area under curve, but talking about the prediction accuracy gives the opposite (rpart-library > tree-library). This depends on which purpose. If I want to maximize the accuracy of prediction, I will go for the rpart one. If the focus is on the distinguishment between classes, then choose the tree.

Naïve Bayes

PD.bayes = naiveBayes(Class~., data = PD_train)
PD.predbayes = predict(PD.bayes, PD_test)
tableNaiveBayes = table(predicted = PD.predbayes, actual = PD_test$Class)
cat("Naive Bayes Confusion\n")
## Naive Bayes Confusion
print(tableNaiveBayes)
##          actual
## predicted   0   1
##         0   9   2
##         1 292 195
confusionMatrix(tableNaiveBayes)
## Confusion Matrix and Statistics
## 
##          actual
## predicted   0   1
##         0   9   2
##         1 292 195
##                                           
##                Accuracy : 0.4096          
##                  95% CI : (0.3661, 0.4543)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : 1               
##                                           
##                   Kappa : 0.0157          
##                                           
##  Mcnemar's Test P-Value : <2e-16          
##                                           
##             Sensitivity : 0.02990         
##             Specificity : 0.98985         
##          Pos Pred Value : 0.81818         
##          Neg Pred Value : 0.40041         
##              Prevalence : 0.60442         
##          Detection Rate : 0.01807         
##    Detection Prevalence : 0.02209         
##       Balanced Accuracy : 0.50987         
##                                           
##        'Positive' Class : 0               
## 

Calculate the accuracy

cat("\nNaive Bayes Accuracy\n")
## 
## Naive Bayes Accuracy
tableNaiveBayes.accuracy = (9 + 195) / (9 + 2 + 292 + 195)
tableNaiveBayes.accuracy
## [1] 0.4096386

The accuracy of Naive Bayes is 0.4096 meaning that 40.96% correct predictions among the total number of observations. This classifier model brings poor performance at predicting the assignment of label options.

Output as confidence level of Naive Bayes and Draw ROC

PDpred.bayes = predict(PD.bayes, PD_test, type = 'raw')
PDBpred <- prediction(PDpred.bayes[, 2], PD_test$Class)
PDBperf <- performance(PDBpred, "tpr", "fpr")
plot(PDBperf, col = "blueviolet", main = "ROC for Naive Bayes")

### Calculate AUC of Naive Bayes

PDcauc.NaiveBayes <- performance(PDBpred, "auc")
naivebayes.auc = as.numeric(PDcauc.NaiveBayes@y.values)
print(naivebayes.auc)
## [1] 0.7298008

The AUC of Naive Bayes is 0.7298 meaning that there is a 72.98% chance that the model will be able to distinguish between positive and negative class.

Bagging

PD.Bag = bagging(Class~., data = PD_train, mfinal = 5)
# output as the confidence level of bagging 
PDpred.Bag = predict.bagging(PD.Bag, PD_test)
PDBagpred <- prediction(PDpred.Bag$prob[, 2], PD_test$Class)
PDBagperf <- performance(PDBagpred, "tpr", "fpr")
# calculate the accuracy
cat("Bagging Confusion \n")
## Bagging Confusion
print(PDpred.Bag$confusion)
##                Observed Class
## Predicted Class   0   1
##               0 265  61
##               1  36 136
confusionMatrix(PDpred.Bag$confusion)
## Confusion Matrix and Statistics
## 
##                Observed Class
## Predicted Class   0   1
##               0 265  61
##               1  36 136
##                                           
##                Accuracy : 0.8052          
##                  95% CI : (0.7677, 0.8391)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : < 2e-16         
##                                           
##                   Kappa : 0.5835          
##                                           
##  Mcnemar's Test P-Value : 0.01482         
##                                           
##             Sensitivity : 0.8804          
##             Specificity : 0.6904          
##          Pos Pred Value : 0.8129          
##          Neg Pred Value : 0.7907          
##              Prevalence : 0.6044          
##          Detection Rate : 0.5321          
##    Detection Prevalence : 0.6546          
##       Balanced Accuracy : 0.7854          
##                                           
##        'Positive' Class : 0               
## 

Calculate the accuracy

cat("\n Bagging accuracy \n")
## 
##  Bagging accuracy
tableBagging.accuracy <- (265 + 136) / (265 + 61 + 36 + 136)
print(tableBagging.accuracy)
## [1] 0.8052209

The accuracy of Bagging is 0.8052 meaning that 80.52% correct predictions among the total observations

Plot the ROC

# plot the ROC
plot(PDBagperf, col = "blue", main = "ROC for Bagging")

Calculate the AUC of Bagging

PDcauc.Bagging = performance(PDBagpred, "auc")
bagging.auc = as.numeric(PDcauc.Bagging@y.values)
print(bagging.auc)
## [1] 0.7931852

The AUC of Bagging is 0.7931 meaning that there is a 79.31% chance that the model will be able to distinguish between positive and negative class.

Boosting

PD.Boost = boosting(Class~., data = PD_train)
# output as the confidence level of Boosting
PDpred.Boost = predict.boosting(PD.Boost, PD_test)
PDBoostpred <- prediction(PDpred.Boost$prob[, 2], PD_test$Class)
PDBoostperf <- performance(PDBoostpred, "tpr", "fpr")
# Calculate the accuracy of performance
cat("Boosting Confusion \n")
## Boosting Confusion
print(PDpred.Boost$confusion)
##                Observed Class
## Predicted Class   0   1
##               0 243  56
##               1  58 141
confusionMatrix(PDpred.Boost$confusion)
## Confusion Matrix and Statistics
## 
##                Observed Class
## Predicted Class   0   1
##               0 243  56
##               1  58 141
##                                           
##                Accuracy : 0.7711          
##                  95% CI : (0.7316, 0.8073)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : 2.266e-15       
##                                           
##                   Kappa : 0.5221          
##                                           
##  Mcnemar's Test P-Value : 0.9254          
##                                           
##             Sensitivity : 0.8073          
##             Specificity : 0.7157          
##          Pos Pred Value : 0.8127          
##          Neg Pred Value : 0.7085          
##              Prevalence : 0.6044          
##          Detection Rate : 0.4880          
##    Detection Prevalence : 0.6004          
##       Balanced Accuracy : 0.7615          
##                                           
##        'Positive' Class : 0               
## 

Calculate the accuracy

cat("\n Boosting accuracy \n")
## 
##  Boosting accuracy
tableBoosting.accuracy <- (243 + 141) / (243 + 56 + 58 + 141)
print(tableBoosting.accuracy) # 0.771
## [1] 0.7710843

The accuracy of Boosting is 0.7710 meaning that 77.10% correct predictions of the total observations

Plot the ROC

# plot the ROC
plot(PDBoostperf, col ="red", main = "ROC for Boosting")

Calculate the AUC of Boosting

PDcauc.Boosting = performance(PDBoostpred, "auc")
boosting.auc = as.numeric(PDcauc.Boosting@y.values)
print(boosting.auc)
## [1] 0.8089532

The AUC of Boosting is 0.8089 meaning that there is a 80.89% chance that the model will be able to distinguish between the positive and negative class.

Random Forest

PD.RandomForest <- randomForest(Class~., data = PD_train, na.action = na.exclude)
# output as the confidence level of Random Forest
PDpredrf <- predict(PD.RandomForest, PD_test)
tableRandomForest <- table(Predicted_Class = PDpredrf, Actual_Class = PD_test$Class)
cat("Random Forest Confusion \n")
## Random Forest Confusion
print(tableRandomForest)
##                Actual_Class
## Predicted_Class   0   1
##               0 255  46
##               1  46 151
confusionMatrix(tableRandomForest)
## Confusion Matrix and Statistics
## 
##                Actual_Class
## Predicted_Class   0   1
##               0 255  46
##               1  46 151
##                                           
##                Accuracy : 0.8153          
##                  95% CI : (0.7783, 0.8484)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : <2e-16          
##                                           
##                   Kappa : 0.6137          
##                                           
##  Mcnemar's Test P-Value : 1               
##                                           
##             Sensitivity : 0.8472          
##             Specificity : 0.7665          
##          Pos Pred Value : 0.8472          
##          Neg Pred Value : 0.7665          
##              Prevalence : 0.6044          
##          Detection Rate : 0.5120          
##    Detection Prevalence : 0.6044          
##       Balanced Accuracy : 0.8068          
##                                           
##        'Positive' Class : 0               
## 

Calculate the accuracy

tableRandomForest.accuracy <- (255 + 151) / (255 + 46 + 46 + 151)
print(tableRandomForest.accuracy) 
## [1] 0.815261

The accuracy of Random Forest is 0.8152 meaning that 81.52% of the total observations

Plot the ROC

PDpred.RandomForest <- predict(PD.RandomForest, PD_test, type = "prob")
PDFpred <- prediction(PDpred.RandomForest[, 2], PD_test$Class)
PDFperf <- performance(PDFpred, "tpr", "fpr")
plot(PDFperf, col = "darkgreen", main = "ROC for Random Forest")

Calculate the AUC of Random Forest

PDcauc.RandomForest = performance(PDFpred, "auc")
randomtree.auc = as.numeric(PDcauc.RandomForest@y.values)
print(randomtree.auc)
## [1] 0.8387018

The AUC of Random Forest is 0.8387 meaning that there is a 83.87% chance that the model will be able to distinguish between the positive and negative class.

Question 6: Plot the ROC in the same axis

plot(PDRpartperf)
abline(0,1, lty = 2)
plot(PDBperf, add = TRUE,  col = "blueviolet")
plot(PDBagperf, add = TRUE,  col = "blue")
plot(PDBoostperf, add = TRUE, col = "red")
plot(PDFperf, add = TRUE,  col = "darkgreen")

# Question 7: Comparing Results From The Table

table_comparison <- data.frame(
  Classifier_Name = c("Decision Tree", "Naive Bayes", "Bagging", "Boosting", "Random Forest"),
  Accuracy = c(tableDecisionTreeUsingRpart.accuracy, tableNaiveBayes.accuracy, tableBagging.accuracy, tableBoosting.accuracy, tableRandomForest.accuracy),
  AUC = c(decisiontree.auc, naivebayes.auc, bagging.auc, boosting.auc, randomtree.auc))
table_comparison

Looking at the table, it can be seen that Random Forest shows better performance at assigning the label to the specified record because it focuses on reducing the overfitting (i.e. bias), and shows simplicity and flexibility of decision tree giving beautiful rating in accuracy and Area Under Curve.

Question 8: Attribute Importance

cat("Decision Tree Attribute Importance Using Tree Library \n")
## Decision Tree Attribute Importance Using Tree Library
print(summary(PD.tree))
## 
## Classification tree:
## tree(formula = Class ~ ., data = PD_train)
## Variables actually used in tree construction:
## [1] "A18" "A01" "A23"
## Number of terminal nodes:  6 
## Residual mean deviance:  0.9944 = 1149 / 1155 
## Misclassification error rate: 0.2171 = 252 / 1161
cat("\n Decision Tree Attribute Importance Using Rpart Library \n")
## 
##  Decision Tree Attribute Importance Using Rpart Library
print(PD.decisiontree$variable.importance)
##        A23        A01        A18        A17        A08        A24        A04 
## 93.6402846 81.2248006 78.9404102  6.9346000  6.5786778  4.0317225  3.6955763 
##        A02        A11        A09        A12        A05        A16 
##  1.8998600  1.5376292  1.1636898  1.1532219  0.2327380  0.1756586
cat("\n Bagging Attribute Importance \n")
## 
##  Bagging Attribute Importance
print(PD.Bag$importance)
##        A01        A02        A04        A05        A06        A08        A09 
## 36.2338722  0.0000000  0.0000000  0.0000000  0.0000000  3.6683625  0.0000000 
##        A10        A11        A12        A13        A14        A15        A16 
##  0.0000000  0.0000000  1.2051756  0.0000000  0.0000000  0.0000000  0.0000000 
##        A17        A18        A19        A20        A21        A23        A24 
##  0.7001734 40.0102393  0.0000000  0.5633931  0.0000000 16.9680590  0.6507249
cat("\n Boosting Attribute Importance \n")
## 
##  Boosting Attribute Importance
print(PD.Boost$importance)
##        A01        A02        A04        A05        A06        A08        A09 
## 13.5574373  1.0073009  1.6821161  0.1627032  1.2285223 11.7178864  0.7319051 
##        A10        A11        A12        A13        A14        A15        A16 
##  0.3716459  0.4630548  8.3607644  0.0000000  0.8905958  1.3007496  0.5062883 
##        A17        A18        A19        A20        A21        A23        A24 
##  2.7564020 25.7610662  1.0255450  1.0580330  0.6089782 19.3078819  7.5011235
cat("\n Random Forest Attribute Importance \n")
## 
##  Random Forest Attribute Importance
print(PD.RandomForest$importance)
##     MeanDecreaseGini
## A01       90.2781051
## A02        5.6118256
## A04        9.8210828
## A05        0.3669806
## A06        6.2783751
## A08       38.1170818
## A09        2.6974434
## A10        2.4205072
## A11        2.6167829
## A12       28.3095912
## A13        0.3180465
## A14       20.0887095
## A15        6.6058981
## A16        3.9362271
## A17       15.3126607
## A18       96.2740580
## A19        6.3004708
## A20       14.1659984
## A21        2.9399071
## A23       91.0375384
## A24       33.9490405

Question 9

Looking at the importance of all classifiers, I plan to utilize A01, A18, and A23 as it shows better rate of predicting the phishing data.

Create Decision Tree

PD.originaldecisiontree = tree(Class~A01 + A18 + A23, data = PD_train, method = "class")
plot(PD.originaldecisiontree, main = "New Decision Tree for Phishing Data")
text(PD.tree, pretty = 0)

### Creating confusion matrix and calculating accuracy

PD.originalpreddecisiontree = predict(PD.originaldecisiontree, PD_test, type = "class")
tableOriginalDecisionTreeUsingTree = table(predicted = PD.originalpreddecisiontree, actual = PD_test$Class)
#cat("Original Decision Tree Confusion Using Rpart Library\n")
print(tableOriginalDecisionTreeUsingTree)
##          actual
## predicted   0   1
##         0 265  64
##         1  36 133
confusionMatrix(tableOriginalDecisionTreeUsingTree)
## Confusion Matrix and Statistics
## 
##          actual
## predicted   0   1
##         0 265  64
##         1  36 133
##                                           
##                Accuracy : 0.7992          
##                  95% CI : (0.7613, 0.8335)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : < 2.2e-16       
##                                           
##                   Kappa : 0.5695          
##                                           
##  Mcnemar's Test P-Value : 0.006934        
##                                           
##             Sensitivity : 0.8804          
##             Specificity : 0.6751          
##          Pos Pred Value : 0.8055          
##          Neg Pred Value : 0.7870          
##              Prevalence : 0.6044          
##          Detection Rate : 0.5321          
##    Detection Prevalence : 0.6606          
##       Balanced Accuracy : 0.7778          
##                                           
##        'Positive' Class : 0               
## 
tableOriginalDecisionTreeUsingTree.accuracy = (265 + 133) / (265 + 64 + 36 + 133)
print(tableOriginalDecisionTreeUsingTree.accuracy)
## [1] 0.7991968

Looking at the previous and current accuracy based on the confusion matrix, there is no difference after adjusting the parameter. From the output, I can infer that it does not affect the performance of decision tree powerfully as their accuracies of predicting are equal.

calculate the number of leaves in the original data

originaldecisiontree.numberOfLeaves = sum(PD.originaldecisiontree$frame$var == "<leaf>")
originaldecisiontree.numberOfLeaves
## [1] 6

Analyze terminal node with cross validation

The cross validation in this case is used to estimate how well our model will perform on unseen data, providing a more accurate measure of its real-world effectiveness and enables to fine-tune the model for optimal performance. This involves calculating the standard deviation of each node size

cvtest = cv.tree(PD.originaldecisiontree, FUN = prune.misclass)
cvtest
## $size
## [1] 6 4 3 1
## 
## $dev
## [1] 261 261 286 509
## 
## $k
## [1]  -Inf   0.0  32.0 112.5
## 
## $method
## [1] "misclass"
## 
## attr(,"class")
## [1] "prune"         "tree.sequence"

Looking at this picture, it is obvious that the rate of CP decreases as the number of splits increases. This occurs in that the decreased CP penalizes the tree less which in turn leads to more production of branches / splits. Looking at the cross validation, it can be seen that the lowest misclassification error is between the sizes of 6 and 4. Then, I decide to choose 4

prune decision tree

The importance of pruning is to avoid the overfitting data by removing any rules that have little effect on the error rate. This is divided into two: a. pre-pruning -> to terminate the trees to grow before the ‘overfitting’ event happens b. post-pruning -> allow the tree to overfit the data after the tree is pruned (common practice)

pruned.decisiontree = prune.misclass(PD.originaldecisiontree, best = 4)
summary(pruned.decisiontree)
## 
## Classification tree:
## snip.tree(tree = PD.originaldecisiontree, nodes = c(6L, 15L))
## Number of terminal nodes:  4 
## Residual mean deviance:  1.029 = 1191 / 1157 
## Misclassification error rate: 0.2171 = 252 / 1161
plot(pruned.decisiontree)
text(pruned.decisiontree, pretty = 0)

Calculate the accuracy of the pruned model

# check the accuracy using the pruned model
PD.prunedmodel = predict(PD.originaldecisiontree, PD_test, type = "class")
tablePrunedDecisionTreeUsingTree = table(predicted = PD.prunedmodel, actual = PD_test$Class)
cat("Decision Tree Confusion\n")
## Decision Tree Confusion
print(tablePrunedDecisionTreeUsingTree)
##          actual
## predicted   0   1
##         0 265  64
##         1  36 133
confusionMatrix(tablePrunedDecisionTreeUsingTree)
## Confusion Matrix and Statistics
## 
##          actual
## predicted   0   1
##         0 265  64
##         1  36 133
##                                           
##                Accuracy : 0.7992          
##                  95% CI : (0.7613, 0.8335)
##     No Information Rate : 0.6044          
##     P-Value [Acc > NIR] : < 2.2e-16       
##                                           
##                   Kappa : 0.5695          
##                                           
##  Mcnemar's Test P-Value : 0.006934        
##                                           
##             Sensitivity : 0.8804          
##             Specificity : 0.6751          
##          Pos Pred Value : 0.8055          
##          Neg Pred Value : 0.7870          
##              Prevalence : 0.6044          
##          Detection Rate : 0.5321          
##    Detection Prevalence : 0.6606          
##       Balanced Accuracy : 0.7778          
##                                           
##        'Positive' Class : 0               
## 

Therefore, there is no change.

#Question 10

library(caret)
library(randomForest)

# Define the range for hyperparameters
mtry_values <- c(1:10)  # You can adjust this range as needed

# Set the seed for reproducibility
set.seed(123) # example

# Define the training control
train_control <- trainControl(method = "cv", number = 10, search = "random")

# Perform the random search
rf_random <- train(Class ~., data = PD_train,
                   method = "rf",
                   trControl = train_control,
                   tuneLength = 10,  # Number of random combinations to try
                   tuneGrid = expand.grid(mtry = mtry_values),
                   ntree = 100,  # You can set the number of trees here
                   nodesize = 5)  # You can set the node size here

# Get the best model
best_model <- rf_random$bestTune

# Print the best hyperparameters
print(best_model)
##   mtry
## 3    3
# Make predictions on the test data
predictions <- predict(rf_random, newdata = PD_test)

# Calculate the confusion matrix
cm <- table(predictions, PD_test$Class)

# Calculate the accuracy
accuracy <- sum(diag(cm)) / sum(cm)

# Print the accuracy
print(accuracy)
## [1] 0.813253

Hypertuning parameter consists of node size, number of trees and number of variables randomly sampled in each split. This also involves the cross validation and random search. The setting of the number utilizes the approach of brute-force. Looking at the result, it can be seen that the best accuracy after hypertuning is 0.813253 meaning that there is a decrease of the performance by 0.002% (from 0.8152 to 0.8132).

Question 11: Artifical Neuron Network

Plot Neuron Network

#install.packages("neuralnet")
#install.packages("car")
library(neuralnet)
## Warning: package 'neuralnet' was built under R version 4.2.3
## 
## Attaching package: 'neuralnet'
## The following object is masked from 'package:ROCR':
## 
##     prediction
## The following object is masked from 'package:dplyr':
## 
##     compute
library(car)
## Warning: package 'car' was built under R version 4.2.3
## Loading required package: carData
## Warning: package 'carData' was built under R version 4.2.3
## 
## Attaching package: 'car'
## The following object is masked from 'package:dplyr':
## 
##     recode
## The following object is masked from 'package:purrr':
## 
##     some
# Scale the data
PD_train_scaled <- as.data.frame(scale(PD_train[, c("A01", "A18", "A23")]))
PD_train_scaled$Class <- PD_train$Class

# Increase stepmax and adjust learning rate
PD_nn <- neuralnet(Class == 1 ~ A01 + A18 + A23, data = PD_train_scaled, hidden = 3, linear.output = FALSE, stepmax = 1e6, learningrate = 0.01)

plot(PD_nn)

Confusion matrix and Calculate the accuracy

PD_pred_neuron = compute(PD_nn, PD_test[c(1, 16, 20)])
prob <- PD_pred_neuron$net.result
pred <- ifelse(prob > 0.5, 1, 0)
# confusion matrix
tableNeuronNetwork = table(observed = PD_test$Class, predicted = pred)
confusionMatrix(tableNeuronNetwork)
## Confusion Matrix and Statistics
## 
##         predicted
## observed   0   1
##        0   2 299
##        1   0 197
##                                           
##                Accuracy : 0.3996          
##                  95% CI : (0.3563, 0.4441)
##     No Information Rate : 0.996           
##     P-Value [Acc > NIR] : 1               
##                                           
##                   Kappa : 0.0053          
##                                           
##  Mcnemar's Test P-Value : <2e-16          
##                                           
##             Sensitivity : 1.000000        
##             Specificity : 0.397177        
##          Pos Pred Value : 0.006645        
##          Neg Pred Value : 1.000000        
##              Prevalence : 0.004016        
##          Detection Rate : 0.004016        
##    Detection Prevalence : 0.604418        
##       Balanced Accuracy : 0.698589        
##                                           
##        'Positive' Class : 0               
## 

The accuracy of my own Artificial Neuron Network is 0.4217 showing poor performance as opposed to the other classifiers. This is because the error is too high during the prediction of classification for each observation. This requires parameter tuning or the changes of the network architecture.

Question 12: New Classifier Model

The new model I am choosing for predicting the phishing is support vector machine

Create a new sample

new_manipulated_PD = manipulated_PD
new_manipulated_PD[, 1:21] <- scale(new_manipulated_PD[, 1:21])
set.seed(31225381)
new_train_row = sample(1:nrow(new_manipulated_PD), 0.7*nrow(new_manipulated_PD))
new_PD_train = new_manipulated_PD[new_train_row,]
new_PD_test = new_manipulated_PD[-new_train_row,]

Fitting Support Vector Machine to Training Dataset

# Fitting SVM to the Training set 
#install.packages('e1071') 
library(e1071) 
new_classifier = svm(formula = Class~ ., 
                 data = new_PD_train, 
                 type = 'C-classification', 
                 kernel = 'linear') 

Predict the test result

PD_pred_svm = predict(new_classifier, newdata = new_PD_test[-22])

Create confusion matrix

PD_pred_svm_confusion_matrix = table(new_PD_test$Class, PD_pred_svm)
confusionMatrix(PD_pred_svm_confusion_matrix)
## Confusion Matrix and Statistics
## 
##    PD_pred_svm
##       0   1
##   0 242  59
##   1  78 119
##                                           
##                Accuracy : 0.7249          
##                  95% CI : (0.6834, 0.7637)
##     No Information Rate : 0.6426          
##     P-Value [Acc > NIR] : 5.704e-05       
##                                           
##                   Kappa : 0.415           
##                                           
##  Mcnemar's Test P-Value : 0.1241          
##                                           
##             Sensitivity : 0.7562          
##             Specificity : 0.6685          
##          Pos Pred Value : 0.8040          
##          Neg Pred Value : 0.6041          
##              Prevalence : 0.6426          
##          Detection Rate : 0.4859          
##    Detection Prevalence : 0.6044          
##       Balanced Accuracy : 0.7124          
##                                           
##        'Positive' Class : 0               
## 

The accuracy of Support Vector Machine is 72.49% meaning that 72.49% correct predictions of all observations.