setwd("D:/Uni/Year 4 Sem 1/FIT3152/Assignment 2")
#install.packages("tidyverse")
library(tidyverse)
## Warning: package 'tidyverse' was built under R version 4.2.3
## Warning: package 'ggplot2' was built under R version 4.2.3
## Warning: package 'tibble' was built under R version 4.2.3
## Warning: package 'tidyr' was built under R version 4.2.3
## Warning: package 'readr' was built under R version 4.2.3
## Warning: package 'purrr' was built under R version 4.2.3
## Warning: package 'dplyr' was built under R version 4.2.3
## Warning: package 'stringr' was built under R version 4.2.3
## Warning: package 'forcats' was built under R version 4.2.3
## Warning: package 'lubridate' was built under R version 4.2.3
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.1.4 ✔ readr 2.1.5
## ✔ forcats 1.0.0 ✔ stringr 1.5.1
## ✔ ggplot2 3.5.0 ✔ tibble 3.2.1
## ✔ lubridate 1.9.3 ✔ tidyr 1.3.1
## ✔ purrr 1.0.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(ggplot2)
rm(list = ls())
# read the data
Phish <- read.csv("PhishingData.csv", stringsAsFactors = TRUE)
set.seed(31225381) # Your Student ID is the random seed
L <- as.data.frame(c(1:50))
#L
L <- L[sample(nrow(L), 10, replace = FALSE),]
#L
Phish <- Phish[(Phish$A01 %in% L),]
PD <- Phish[sample(nrow(Phish), 2000, replace = FALSE),] # sample of 2000 rows
The summary only calculates the general descriptive statistic ## Descriptive Statistics
PD
# get the description of each attribute
summary(PD)
## A01 A02 A03 A04
## Min. : 4.0 Min. : 0.0000 Min. :0.000000 Min. :2.000
## 1st Qu.:17.0 1st Qu.: 0.0000 1st Qu.:0.000000 1st Qu.:2.000
## Median :35.0 Median : 0.0000 Median :0.000000 Median :3.000
## Mean :30.8 Mean : 0.1482 Mean :0.000506 Mean :2.767
## 3rd Qu.:44.0 3rd Qu.: 0.0000 3rd Qu.:0.000000 3rd Qu.:3.000
## Max. :47.0 Max. :20.0000 Max. :1.000000 Max. :7.000
## NA's :16 NA's :23 NA's :20
## A05 A06 A07 A08
## Min. : 0.00000 Min. :0.000 Min. :0.000000 Min. :0.1667
## 1st Qu.: 0.00000 1st Qu.:0.000 1st Qu.:0.000000 1st Qu.:0.6916
## Median : 0.00000 Median :0.000 Median :0.000000 Median :1.0000
## Mean : 0.07284 Mean :0.122 Mean :0.002532 Mean :0.8457
## 3rd Qu.: 0.00000 3rd Qu.:0.000 3rd Qu.:0.000000 3rd Qu.:1.0000
## Max. :97.00000 Max. :1.000 Max. :1.000000 Max. :1.0000
## NA's :23 NA's :25 NA's :25 NA's :20
## A09 A10 A11 A12
## Min. :0.00000 Min. :0.00000 Min. : 0.0000 Min. :117.0
## 1st Qu.:0.00000 1st Qu.:0.00000 1st Qu.: 0.0000 1st Qu.:232.0
## Median :0.00000 Median :0.00000 Median : 0.0000 Median :232.0
## Mean :0.02833 Mean :0.04529 Mean : 0.1297 Mean :317.5
## 3rd Qu.:0.00000 3rd Qu.:0.00000 3rd Qu.: 0.0000 3rd Qu.:418.0
## Max. :1.00000 Max. :1.00000 Max. :176.0000 Max. :692.0
## NA's :23 NA's :13 NA's :18 NA's :18
## A13 A14 A15 A16
## Min. : 0.0000 Min. :0.0000 Min. :0.0000 Min. :0.0000
## 1st Qu.: 0.0000 1st Qu.:0.0000 1st Qu.:0.0000 1st Qu.:0.0000
## Median : 0.0000 Median :0.0000 Median :0.0000 Median :0.0000
## Mean : 0.1694 Mean :0.1594 Mean :0.1367 Mean :0.0398
## 3rd Qu.: 0.0000 3rd Qu.:0.0000 3rd Qu.:0.0000 3rd Qu.:0.0000
## Max. :291.0000 Max. :1.0000 Max. :1.0000 Max. :1.0000
## NA's :16 NA's :18 NA's :18 NA's :15
## A17 A18 A19 A20
## Min. :0.000 Min. : 5.00 Min. :0.0000 Min. :0.0000
## 1st Qu.:1.000 1st Qu.: 12.00 1st Qu.:0.0000 1st Qu.:0.0000
## Median :1.000 Median : 31.00 Median :0.0000 Median :0.0000
## Mean :1.153 Mean : 57.22 Mean :0.1065 Mean :0.2352
## 3rd Qu.:1.000 3rd Qu.: 89.00 3rd Qu.:0.0000 3rd Qu.:0.0000
## Max. :6.000 Max. :1017.00 Max. :1.0000 Max. :1.0000
## NA's :16 NA's :19 NA's :18 NA's :10
## A21 A22 A23 A24
## Min. :0.00000 Min. :0.01264 Min. : 0.00 Min. :0.000000
## 1st Qu.:0.00000 1st Qu.:0.05103 1st Qu.: 2.00 1st Qu.:0.007505
## Median :0.00000 Median :0.05804 Median : 39.00 Median :0.079963
## Mean :0.02375 Mean :0.05596 Mean : 60.05 Mean :0.266114
## 3rd Qu.:0.00000 3rd Qu.:0.06296 3rd Qu.: 100.00 3rd Qu.:0.522907
## Max. :2.00000 Max. :0.08039 Max. :1868.00 Max. :0.522907
## NA's :21 NA's :19 NA's :27 NA's :14
## A25 Class
## Min. :0.000000 Min. :0.000
## 1st Qu.:0.000000 1st Qu.:0.000
## Median :0.000000 Median :0.000
## Mean :0.000171 Mean :0.428
## 3rd Qu.:0.000000 3rd Qu.:1.000
## Max. :0.143000 Max. :1.000
## NA's :15
dim(PD)
## [1] 2000 26
With the use of the “pastecs” library, it calculates the in-depth statistics.
#install.packages("pastecs")
library(pastecs)
## Warning: package 'pastecs' was built under R version 4.2.3
##
## Attaching package: 'pastecs'
## The following objects are masked from 'package:dplyr':
##
## first, last
## The following object is masked from 'package:tidyr':
##
## extract
round(stat.desc(PD), 4)
str(PD)
## 'data.frame': 2000 obs. of 26 variables:
## $ A01 : int 44 47 44 8 17 45 45 47 47 34 ...
## $ A02 : int 0 0 5 0 0 3 0 0 0 0 ...
## $ A03 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A04 : int 3 3 3 3 3 3 2 3 3 3 ...
## $ A05 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A06 : int 0 0 0 0 0 0 0 0 0 NA ...
## $ A07 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A08 : num 1 1 0.593 0.818 1 ...
## $ A09 : int 0 0 NA 0 1 0 0 0 0 0 ...
## $ A10 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A11 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A12 : int 232 504 232 171 232 232 379 232 232 482 ...
## $ A13 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A14 : int 0 0 0 0 0 0 0 1 0 0 ...
## $ A15 : int 0 0 0 1 0 0 0 0 0 0 ...
## $ A16 : int 0 0 0 0 0 NA 0 0 0 1 ...
## $ A17 : int 1 1 1 1 1 1 1 1 0 1 ...
## $ A18 : int 78 24 16 11 97 212 9 98 12 24 ...
## $ A19 : int 0 0 1 0 0 0 0 0 0 0 ...
## $ A20 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A21 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A22 : num 0.0623 0.073 0.0489 0.0463 0.0664 ...
## $ A23 : int 71 107 1 100 113 29 100 32 0 169 ...
## $ A24 : num 0.5229 0.08 0.5229 0.0018 0.5229 ...
## $ A25 : num 0 0 0 0 0 0 0 0 0 0 ...
## $ Class: int 1 1 0 0 0 1 0 1 0 0 ...
counter_phishing = 0
counter_legitimate = 0
site_observation = count(PD)
for (site in PD$Class){
if(site == 0)
{
counter_legitimate = counter_legitimate + 1
}
else
{
counter_phishing = counter_phishing + 1
}
}
proportion.legitimateSites = as.numeric(round(counter_legitimate / site_observation, 2))
proportion.phishingSites = as.numeric(round(counter_phishing / site_observation, 2))
cat("The number of phishing sites is \n")
## The number of phishing sites is
print(counter_phishing)
## [1] 856
cat("The number of legitimate sites is \n")
## The number of legitimate sites is
print(counter_legitimate)
## [1] 1144
cat("Proportion of phishing sites is \n")
## Proportion of phishing sites is
print(proportion.phishingSites)
## [1] 0.43
cat("Proportion of legitimate sites is \n")
## Proportion of legitimate sites is
print(proportion.legitimateSites)
## [1] 0.57
According to the calculation, it can be seen that most of the observations tend to be identified as legitimate rather than phishing. It can also be proved by the use of the visualization.
ggplot(PD, aes(x = as.factor(Class))) +
geom_bar(fill = c("blue", "red")) +
labs(title = "Website Categories",
x = "Category",
y = "Frequency") +
theme_minimal()
# calculate the number of missing values in each column
colSums(is.na(PD))
## A01 A02 A03 A04 A05 A06 A07 A08 A09 A10 A11 A12 A13
## 0 16 23 20 23 25 25 20 23 13 18 18 16
## A14 A15 A16 A17 A18 A19 A20 A21 A22 A23 A24 A25 Class
## 18 18 15 16 19 18 10 21 19 27 14 15 0
The approach of omitting the number of attributes is by taking low variance based on the calculation I have done previously.
manipulated_PD <- PD %>%
select(-A03, -A07, -A22, -A25)
manipulated_PD
The attributes I am going to use for the next analysis is A01, A02, A04, A05, A06, A08, A09, A10, A11, A12, A13, A14, A15, A16, A17, A18, A19, A20, A21, A23, A24, and Class. What resonates with the use of these attributes is the low variances from A03, A07, A22, and A25. Because they are not varied and widespread, it is not suitable for the case of the classification. Therefore, I decide to make use of these attributes.
#manipulated_PD <- na.omit(manipulated_PD)
manipulated_PD = manipulated_PD[complete.cases(manipulated_PD),]
manipulated_PD
There is no need to scale the data as the classification model and ANN do not concern how big or small the number is that causes the superiority. This feature can only be applied to the case that has a high-level concern over the superiority like “clustering”
#manipulated_PD[, 1:21] <- scale(manipulated_PD[, 1:21])
#manipulated_PD
#install.packages("car")
manipulated_PD$Class = as.factor(manipulated_PD$Class)
str(manipulated_PD)
## 'data.frame': 1659 obs. of 22 variables:
## $ A01 : int 44 47 8 17 45 47 47 26 47 47 ...
## $ A02 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A04 : int 3 3 3 3 2 3 3 3 3 3 ...
## $ A05 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A06 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A08 : num 1 1 0.818 1 0.571 ...
## $ A09 : int 0 0 0 1 0 0 0 0 0 0 ...
## $ A10 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A11 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A12 : int 232 504 171 232 379 232 232 232 504 504 ...
## $ A13 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A14 : int 0 0 0 0 0 1 0 0 1 0 ...
## $ A15 : int 0 0 1 0 0 0 0 0 0 0 ...
## $ A16 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A17 : int 1 1 1 1 1 1 0 1 1 1 ...
## $ A18 : int 78 24 11 97 9 98 12 94 98 20 ...
## $ A19 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A20 : int 0 0 0 0 0 0 0 0 0 1 ...
## $ A21 : int 0 0 0 0 0 0 0 0 0 0 ...
## $ A23 : int 71 107 100 113 100 32 0 110 110 40 ...
## $ A24 : num 0.52291 0.07996 0.0018 0.52291 0.00122 ...
## $ Class: Factor w/ 2 levels "0","1": 2 2 1 1 1 2 1 1 2 2 ...
manipulated_PD
# set the seed with student id for reproducing the results randomly
set.seed(31225381)
train.row = sample(1:nrow(manipulated_PD), 0.7*nrow(manipulated_PD))
PD_train = manipulated_PD[train.row,]
PD_test = manipulated_PD[-train.row,]
train.row
## [1] 510 196 1196 602 684 611 97 132 222 528 1159 241 1658 174
## [15] 1148 1244 1355 1208 453 1205 843 102 907 870 1043 458 604 677
## [29] 697 1331 1036 1009 791 1544 1592 1269 727 168 804 1256 723 1408
## [43] 629 1259 1231 1361 849 933 512 967 1130 1020 144 1597 1178 541
## [57] 1216 786 1299 1252 800 182 572 586 889 487 269 420 1129 670
## [71] 1446 48 834 501 1560 473 882 1233 124 919 107 781 582 54
## [85] 1090 1497 221 1442 1541 1012 659 1423 937 1502 266 813 970 728
## [99] 378 1104 811 1063 493 188 248 140 765 257 1444 608 1435 1441
## [113] 1191 916 1484 1122 975 1149 306 1060 1290 71 855 1348 1625 210
## [127] 923 776 1094 253 336 798 1558 1175 1265 382 486 1170 553 612
## [141] 1163 1017 1445 1543 1450 1515 1153 900 702 1593 496 1508 1123 327
## [155] 284 678 707 1443 1119 927 443 1067 979 1353 1649 1228 289 857
## [169] 1027 324 1133 1453 1283 218 614 1106 640 88 789 628 125 1236
## [183] 1195 1517 551 468 429 852 152 1401 773 930 605 1070 1330 1551
## [197] 349 213 267 790 106 1141 1333 892 663 1197 1075 1209 595 1482
## [211] 1118 1229 1193 1528 1303 672 1114 79 799 1552 454 370 520 185
## [225] 720 936 1051 1388 906 99 295 817 850 488 929 459 178 100
## [239] 385 696 1268 1336 1277 1217 1481 1350 198 679 1146 235 744 1594
## [253] 1010 1616 1566 1214 298 1073 268 1610 1139 1320 1576 1111 539 1525
## [267] 1248 1514 1367 1194 1556 1619 1375 175 1521 6 917 1121 425 1325
## [281] 137 986 1506 435 835 448 1440 579 1332 34 607 1155 918 293
## [295] 599 1169 1578 518 913 1232 1304 689 136 748 1377 1127 76 1251
## [309] 194 163 1313 1239 38 1168 842 101 1429 272 1606 875 803 756
## [323] 1189 1222 1025 361 63 555 1559 1569 1615 1587 1607 315 110 981
## [337] 1645 203 1296 1298 301 1425 1369 1037 78 28 818 1109 1288 1034
## [351] 525 1187 815 281 973 1653 1107 1489 560 77 1186 1092 1368 397
## [365] 291 120 1014 795 1258 742 232 564 335 1091 1223 1328 1316 165
## [379] 905 1432 760 264 575 1623 273 287 495 644 538 1650 1373 1086
## [393] 725 166 965 345 43 1654 768 61 861 686 381 134 223 466
## [407] 498 73 1021 669 1204 1307 1165 609 1385 226 150 802 422 85
## [421] 1648 1393 1016 649 427 470 825 635 548 1142 746 949 187 1613
## [435] 828 540 91 1397 212 617 1160 1074 355 1080 1211 252 714 1419
## [449] 662 983 138 561 316 1053 729 1424 990 393 1631 565 33 1062
## [463] 1533 709 606 1247 40 375 1003 329 1210 170 363 764 1421 1302
## [477] 1415 695 181 1024 1310 147 1266 594 827 108 462 1378 903 30
## [491] 647 1498 598 325 290 574 1351 532 1319 772 1584 1068 184 259
## [505] 1470 711 968 1147 406 280 947 621 1207 1448 526 770 1055 1624
## [519] 1087 1292 940 309 377 475 60 1534 954 514 1535 920 130 1061
## [533] 1362 1028 1134 68 1305 1463 51 262 895 1243 1019 395 1356 961
## [547] 1128 1026 793 1400 1618 1488 407 673 650 926 562 1499 492 1465
## [561] 1156 1364 1475 1601 1571 1609 250 365 1505 1308 735 430 712 736
## [575] 1387 285 1449 402 1190 1452 119 193 683 45 763 1407 1264 693
## [589] 1124 701 874 636 366 17 339 12 491 1643 1273 1511 1044 154
## [603] 1520 1054 1254 769 1093 552 616 1582 657 777 398 1237 1628 476
## [617] 228 115 1235 1293 989 7 1512 648 996 403 334 1311 1374 423
## [631] 1040 186 883 354 146 935 340 31 292 438 81 1483 372 206
## [645] 481 1192 741 569 14 757 589 837 1172 271 499 877 880 739
## [659] 312 1572 367 283 784 1436 219 1529 1509 1314 1263 554 263 246
## [673] 866 1049 227 317 1279 1567 807 341 1038 758 161 785 1565 1652
## [687] 1412 1640 216 960 680 1545 583 660 46 98 483 389 469 92
## [701] 356 896 904 1485 353 944 597 921 368 121 978 419 417 591
## [715] 705 439 816 1395 809 18 864 533 893 715 56 632 1039 1564
## [729] 1492 1029 1460 1469 156 1224 664 418 1219 821 32 667 887 167
## [743] 231 884 897 254 1103 991 987 601 1477 1651 687 627 1215 1225
## [757] 651 80 956 637 1438 337 953 909 1253 824 1359 171 472 195
## [771] 566 1570 1326 962 1622 244 860 1516 350 1548 180 392 706 1115
## [785] 1513 1143 386 952 1532 1461 1085 89 1390 1487 265 237 21 311
## [799] 829 1132 894 2 1457 787 531 820 1474 1583 1076 1504 1476 1341
## [813] 779 1030 836 1405 549 1370 90 1230 457 792 1112 201 208 1001
## [827] 114 20 925 942 775 529 1456 563 1006 1226 1180 357 1059 3
## [841] 1335 322 524 256 690 808 788 1409 1414 1057 1599 1238 618 260
## [855] 885 1426 762 1496 1282 465 1245 666 1346 1213 822 113 780 945
## [869] 47 1338 369 1246 726 1183 1384 484 810 396 1077 623 1608 503
## [883] 1334 854 25 1455 1554 1154 189 639 1272 275 225 391 1181 676
## [897] 1152 1417 914 1052 1611 1164 1199 1278 747 869 330 969 1047 1203
## [911] 358 1113 346 1162 1218 698 1431 872 1549 523 590 1614 643 1360
## [925] 1150 1241 1048 333 881 550 111 557 1007 1500 24 296 1295 1234
## [939] 1227 197 542 1557 87 401 1110 409 191 238 1285 1167 1416 1079
## [953] 509 1379 1177 1337 980 494 738 1108 1309 774 70 1363 1546 1315
## [967] 1553 143 41 665 1011 1503 1427 58 1644 1538 721 1527 584 688
## [981] 988 86 505 303 305 199 482 570 164 504 383 699 1466 704
## [995] 117 766 1531 1659 1638 848 915 126 948 1376 1433 928 1270 534
## [1009] 1603 819 1630 1641 9 1418 694 421 964 1324 943 288 958 360
## [1023] 66 1099 105 416 1354 1526 145 1045 674 814 536 1345 867 1089
## [1037] 1035 938 888 805 1352 57 220 323 1013 1398 844 838 343 249
## [1051] 658 1537 131 1005 517 571 1212 1301 400 997 15 1380 408 1633
## [1065] 1420 62 1468 708 1125 1404 1589 455 399 1620 671 413 862 1598
## [1079] 593 1145 630 116 1447 1344 1179 1069 911 277 682 891 1201 411
## [1093] 759 1642 202 1339 16 1323 331 373 1563 839 497 1574 1491 719
## [1107] 750 641 500 1406 1596 740 362 1501 94 1327 1126 1329 347 96
## [1121] 1347 1396 1255 901 899 1317 890 230 1015 215 8 135 27 567
## [1135] 782 302 767 49 431 868 1386 685 318 192 863 93 234 310
## [1149] 464 634 352 460 537 1173 516 432 845 507 1626 1144 1151
PD_train # dataset trained by 70%
PD_test # dataset test by 30%
#install.packages("tree")
library(tree)
## Warning: package 'tree' was built under R version 4.2.3
#install.packages("e1071")
library(e1071)
## Warning: package 'e1071' was built under R version 4.2.3
#install.packages(("ROCR"))
library(ROCR)
## Warning: package 'ROCR' was built under R version 4.2.3
#install.packages("randomForest")
library(randomForest)
## Warning: package 'randomForest' was built under R version 4.2.3
## randomForest 4.7-1.1
## Type rfNews() to see new features/changes/bug fixes.
##
## Attaching package: 'randomForest'
## The following object is masked from 'package:dplyr':
##
## combine
## The following object is masked from 'package:ggplot2':
##
## margin
#install.packages("adabag")
library(adabag)
## Warning: package 'adabag' was built under R version 4.2.3
## Loading required package: rpart
## Warning: package 'rpart' was built under R version 4.2.3
## Loading required package: caret
## Warning: package 'caret' was built under R version 4.2.3
## Loading required package: lattice
##
## Attaching package: 'caret'
## The following object is masked from 'package:purrr':
##
## lift
## Loading required package: foreach
## Warning: package 'foreach' was built under R version 4.2.3
##
## Attaching package: 'foreach'
## The following objects are masked from 'package:purrr':
##
## accumulate, when
## Loading required package: doParallel
## Warning: package 'doParallel' was built under R version 4.2.3
## Loading required package: iterators
## Warning: package 'iterators' was built under R version 4.2.3
## Loading required package: parallel
#install.packages("rpart")
library(rpart)
#install.packages("rpart.plot")
library(rpart.plot)
## Warning: package 'rpart.plot' was built under R version 4.2.3
PD.tree = tree(Class~., data = PD_train)
# to see all datas in each node
PD.tree
## node), split, n, deviance, yval, (yprob)
## * denotes terminal node
##
## 1) root 1161 1592.0 0 ( 0.5616 0.4384 )
## 2) A18 < 14.5 341 291.2 0 ( 0.8475 0.1525 ) *
## 3) A18 > 14.5 820 1126.0 1 ( 0.4427 0.5573 )
## 6) A01 < 30 309 371.0 0 ( 0.7120 0.2880 )
## 12) A01 < 12.5 142 122.5 0 ( 0.8451 0.1549 ) *
## 13) A01 > 12.5 167 224.9 0 ( 0.5988 0.4012 ) *
## 7) A01 > 30 511 605.8 1 ( 0.2798 0.7202 )
## 14) A23 < 2.5 90 113.1 0 ( 0.6778 0.3222 ) *
## 15) A23 > 2.5 421 415.2 1 ( 0.1948 0.8052 )
## 30) A01 < 39 135 169.0 1 ( 0.3185 0.6815 ) *
## 31) A01 > 39 286 227.8 1 ( 0.1364 0.8636 ) *
print(summary(PD.tree))
##
## Classification tree:
## tree(formula = Class ~ ., data = PD_train)
## Variables actually used in tree construction:
## [1] "A18" "A01" "A23"
## Number of terminal nodes: 6
## Residual mean deviance: 0.9944 = 1149 / 1155
## Misclassification error rate: 0.2171 = 252 / 1161
# draw the decision tree
plot(PD.tree)
# set the text size
text(PD.tree, pretty = 0)
In term of how to assign the label, the decision starts from the main
root where it checks the record whether A18 is lower than 14.5. The root
node is splitted into two branches on two sides (2nd and 3rd nodes). If
the check is “YES”, then it is traversed to the left (assigning the 0 to
the record as the conclusion). As it does not meet the condition, it
goes to the node 3 (A18 > 14.5) in transition to the branch where it
has 2 nodes (node 6 and node 7). Deciding on which node the record needs
to point to, it relies on the condition of whether A01 < 30. If it
meets, it is traversed to the left (node 6). Node 6 is split into two
nodes (node 12 and node 13) with conditions (A1 < 12.5 and A1 >
12.5; going to the same label). As the condition fails to satisfies the
condition, it visits the node 7 and then this node is divided into two
nodes (node 14 and node 15) following the condition of whether A023 is
lower than 2.5. As it is lower, then it stops. When the input is bigger
than 2.5, it refers to the node 15 which in turn is separated into two
nodes (node 30 and node 31). These nodes obey the condition rule (“A01
is either lower or greater than 39”). Choosing lower or higher than the
expected number, it corresponds to the same outcome (i.e. 1).
PD.predtree = predict(PD.tree, PD_test, type = "class")
tableDecisionTreeUsingTree = table(predicted = PD.predtree, actual = PD_test$Class)
cat("Decision Tree Confusion\n")
## Decision Tree Confusion
print(tableDecisionTreeUsingTree)
## actual
## predicted 0 1
## 0 265 64
## 1 36 133
confusionMatrix(tableDecisionTreeUsingTree)
## Confusion Matrix and Statistics
##
## actual
## predicted 0 1
## 0 265 64
## 1 36 133
##
## Accuracy : 0.7992
## 95% CI : (0.7613, 0.8335)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : < 2.2e-16
##
## Kappa : 0.5695
##
## Mcnemar's Test P-Value : 0.006934
##
## Sensitivity : 0.8804
## Specificity : 0.6751
## Pos Pred Value : 0.8055
## Neg Pred Value : 0.7870
## Prevalence : 0.6044
## Detection Rate : 0.5321
## Detection Prevalence : 0.6606
## Balanced Accuracy : 0.7778
##
## 'Positive' Class : 0
##
cat("\nDecision Tree Accuracy\n")
##
## Decision Tree Accuracy
tableDecisionTreeUsingTree.accuracy = (265 + 133) / (265 + 64 + 36 + 133)
tableDecisionTreeUsingTree.accuracy
## [1] 0.7991968
Measuring the accuracy can be completed by taking the diagonal and summing the value divided by the total of matrix. In the case of the “decision tree” accuracy, this implies that 0.7991 => 79.91% Its accuracy is 0.7991 meaning that 79.91% correct predictions among all observations.
# do predictions as probabilities
PD.pred.tree = predict(PD.tree, PD_test, type = "vector")
# computing a simple ROC curve (x-axis: fpr, y-axis = tpr)
# labels are actual values, predictors are probability of class
PDpred <- prediction(PD.pred.tree[, 2], PD_test$Class)
PDperf <- performance(PDpred, "tpr", "fpr")
plot(PDperf)
abline(0,1, lty = 2)
PDcauc.DecisionTreeUsingTree = performance(PDpred, "auc")
print(as.numeric(PDcauc.DecisionTreeUsingTree@y.values))
## [1] 0.8190887
The AUC of Decision Tree is 0.819 meaning that there is a 81.9% chance that the model will be able to distinguish between positive and negative class.
PD.decisiontree = rpart(Class~., data = PD_train, method = "class")
print(summary(PD.decisiontree))
## Call:
## rpart(formula = Class ~ ., data = PD_train, method = "class")
## n= 1161
##
## CP nsplit rel error xerror xstd
## 1 0.22102161 0 1.0000000 1.0000000 0.03321611
## 2 0.06286837 2 0.5579568 0.5579568 0.02877565
## 3 0.01080550 3 0.4950884 0.5147348 0.02798315
## 4 0.01000000 5 0.4734774 0.5166994 0.02802090
##
## Variable importance
## A23 A01 A18 A17 A08 A24 A04 A02 A11
## 33 29 28 2 2 1 1 1 1
##
## Node number 1: 1161 observations, complexity param=0.2210216
## predicted class=0 expected loss=0.4384152 P(node) =1
## class counts: 652 509
## probabilities: 0.562 0.438
## left son=2 (341 obs) right son=3 (820 obs)
## Primary splits:
## A18 < 14.5 to the left, improve=78.94041, (0 missing)
## A01 < 21.5 to the left, improve=60.95977, (0 missing)
## A23 < 0.5 to the left, improve=49.47025, (0 missing)
## A14 < 0.5 to the left, improve=48.43386, (0 missing)
## A20 < 0.5 to the left, improve=33.56203, (0 missing)
## Surrogate splits:
## A23 < 0.5 to the left, agree=0.793, adj=0.296, (0 split)
## A24 < 0.0003907 to the left, agree=0.720, adj=0.047, (0 split)
## A17 < 0.5 to the left, agree=0.717, adj=0.038, (0 split)
## A04 < 3.5 to the right, agree=0.714, adj=0.026, (0 split)
## A08 < 0.3452083 to the left, agree=0.707, adj=0.003, (0 split)
##
## Node number 2: 341 observations
## predicted class=0 expected loss=0.1524927 P(node) =0.2937123
## class counts: 289 52
## probabilities: 0.848 0.152
##
## Node number 3: 820 observations, complexity param=0.2210216
## predicted class=1 expected loss=0.4426829 P(node) =0.7062877
## class counts: 363 457
## probabilities: 0.443 0.557
## left son=6 (309 obs) right son=7 (511 obs)
## Primary splits:
## A01 < 30 to the left, improve=71.91603, (0 missing)
## A14 < 0.5 to the left, improve=22.85848, (0 missing)
## A23 < 0.5 to the left, improve=22.00819, (0 missing)
## A24 < 0.00178155 to the left, improve=21.70071, (0 missing)
## A08 < 0.568323 to the left, improve=15.04439, (0 missing)
## Surrogate splits:
## A23 < 99.5 to the right, agree=0.771, adj=0.392, (0 split)
## A02 < 2.5 to the right, agree=0.630, adj=0.019, (0 split)
## A09 < 0.5 to the right, agree=0.629, adj=0.016, (0 split)
## A17 < 3.5 to the right, agree=0.626, adj=0.006, (0 split)
## A05 < 9 to the right, agree=0.624, adj=0.003, (0 split)
##
## Node number 6: 309 observations, complexity param=0.0108055
## predicted class=0 expected loss=0.2880259 P(node) =0.2661499
## class counts: 220 89
## probabilities: 0.712 0.288
## left son=12 (142 obs) right son=13 (167 obs)
## Primary splits:
## A01 < 12.5 to the left, improve=9.308772, (0 missing)
## A24 < 0.00169035 to the left, improve=3.798886, (0 missing)
## A23 < 132.5 to the left, improve=3.207926, (0 missing)
## A18 < 112.5 to the left, improve=2.464323, (0 missing)
## A14 < 0.5 to the left, improve=1.844179, (0 missing)
## Surrogate splits:
## A23 < 97 to the right, agree=0.625, adj=0.183, (0 split)
## A04 < 2.5 to the left, agree=0.563, adj=0.049, (0 split)
## A02 < 0.5 to the right, agree=0.557, adj=0.035, (0 split)
## A24 < 0.0125524 to the left, agree=0.557, adj=0.035, (0 split)
## A08 < 0.344065 to the left, agree=0.550, adj=0.021, (0 split)
##
## Node number 7: 511 observations, complexity param=0.06286837
## predicted class=1 expected loss=0.2798434 P(node) =0.4401378
## class counts: 143 368
## probabilities: 0.280 0.720
## left son=14 (90 obs) right son=15 (421 obs)
## Primary splits:
## A23 < 2.5 to the left, improve=34.596660, (0 missing)
## A08 < 0.568323 to the left, improve=21.577230, (0 missing)
## A24 < 0.00619405 to the left, improve=16.310730, (0 missing)
## A12 < 137.5 to the left, improve= 8.362356, (0 missing)
## A17 < 0.5 to the left, improve= 6.383684, (0 missing)
## Surrogate splits:
## A08 < 0.568323 to the left, agree=0.855, adj=0.178, (0 split)
## A17 < 0.5 to the left, agree=0.841, adj=0.100, (0 split)
## A11 < 1.5 to the right, agree=0.832, adj=0.044, (0 split)
## A04 < 3.5 to the right, agree=0.830, adj=0.033, (0 split)
## A12 < 137.5 to the left, agree=0.830, adj=0.033, (0 split)
##
## Node number 12: 142 observations
## predicted class=0 expected loss=0.1549296 P(node) =0.1223084
## class counts: 120 22
## probabilities: 0.845 0.155
##
## Node number 13: 167 observations, complexity param=0.0108055
## predicted class=0 expected loss=0.4011976 P(node) =0.1438415
## class counts: 100 67
## probabilities: 0.599 0.401
## left son=26 (134 obs) right son=27 (33 obs)
## Primary splits:
## A23 < 118.5 to the left, improve=5.796735, (0 missing)
## A24 < 0.00169035 to the left, improve=5.663763, (0 missing)
## A18 < 17.5 to the right, improve=2.504090, (0 missing)
## A20 < 0.5 to the left, improve=2.413963, (0 missing)
## A02 < 0.5 to the left, improve=1.899394, (0 missing)
## Surrogate splits:
## A02 < 5.5 to the left, agree=0.808, adj=0.03, (0 split)
## A16 < 0.5 to the left, agree=0.808, adj=0.03, (0 split)
##
## Node number 14: 90 observations
## predicted class=0 expected loss=0.3222222 P(node) =0.07751938
## class counts: 61 29
## probabilities: 0.678 0.322
##
## Node number 15: 421 observations
## predicted class=1 expected loss=0.1947743 P(node) =0.3626184
## class counts: 82 339
## probabilities: 0.195 0.805
##
## Node number 26: 134 observations
## predicted class=0 expected loss=0.3358209 P(node) =0.1154177
## class counts: 89 45
## probabilities: 0.664 0.336
##
## Node number 27: 33 observations
## predicted class=1 expected loss=0.3333333 P(node) =0.02842377
## class counts: 11 22
## probabilities: 0.333 0.667
##
## n= 1161
##
## node), split, n, loss, yval, (yprob)
## * denotes terminal node
##
## 1) root 1161 509 0 (0.5615848 0.4384152)
## 2) A18< 14.5 341 52 0 (0.8475073 0.1524927) *
## 3) A18>=14.5 820 363 1 (0.4426829 0.5573171)
## 6) A01< 30 309 89 0 (0.7119741 0.2880259)
## 12) A01< 12.5 142 22 0 (0.8450704 0.1549296) *
## 13) A01>=12.5 167 67 0 (0.5988024 0.4011976)
## 26) A23< 118.5 134 45 0 (0.6641791 0.3358209) *
## 27) A23>=118.5 33 11 1 (0.3333333 0.6666667) *
## 7) A01>=30 511 143 1 (0.2798434 0.7201566)
## 14) A23< 2.5 90 29 0 (0.6777778 0.3222222) *
## 15) A23>=2.5 421 82 1 (0.1947743 0.8052257) *
rpart.plot(PD.decisiontree, main = "Decision Tree for Phishing Data")
In terms of the concept of this decision tree, the start point is always
similar to the previous case (i.e. begins with the root split into two
nodes (node 2 and node 3)). So The root verifies whether A18 is lower
than 14.5. As the fulfillment is true, then the observation is labelled
“0”. If not, then it is directed to node 3, divided into 2 nodes (node 6
and node 7 following the condition of whether A01 < 30): a. If the
record of A01 is lower than 30, it visits node 6 that is split into node
12 (A01 < 12.5) and node 13 (A01 >= 12.5). As it is lower than
12.5, then the label is 0. If not, then it goes to the node 13, split
into two nodes (node 26 and node 27). If the record of A23 is lower than
118.5 / 119, then it is marked as 0. if not, it is traversed to the
right and labelled as 1. b. If the record of A01 is greater than or
equal to 30, then it visits node 7, divided into node 14 and no 15
following the condition of whether A23 < 2.5. As the record of A23
< 2.5, it is traversed to the left and marked as 0. When A23’s record
is greater than or equal to 2.5, then the label for the observation is
1.
PD.preddecisiontree = predict(PD.decisiontree, PD_test, type = "class")
tableDecisionTreeUsingRpart = table(predicted = PD.preddecisiontree, actual = PD_test$Class)
cat("Decision Tree Confusion Using Rpart Library\n")
## Decision Tree Confusion Using Rpart Library
print(tableDecisionTreeUsingRpart)
## actual
## predicted 0 1
## 0 262 60
## 1 39 137
confusionMatrix(tableDecisionTreeUsingRpart)
## Confusion Matrix and Statistics
##
## actual
## predicted 0 1
## 0 262 60
## 1 39 137
##
## Accuracy : 0.8012
## 95% CI : (0.7634, 0.8354)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : < 2e-16
##
## Kappa : 0.5765
##
## Mcnemar's Test P-Value : 0.04442
##
## Sensitivity : 0.8704
## Specificity : 0.6954
## Pos Pred Value : 0.8137
## Neg Pred Value : 0.7784
## Prevalence : 0.6044
## Detection Rate : 0.5261
## Detection Prevalence : 0.6466
## Balanced Accuracy : 0.7829
##
## 'Positive' Class : 0
##
tableDecisionTreeUsingRpart.accuracy = (262 + 137) / (262 + 60 + 39 + 137)
cat("Decision Tree Accuracy Using Rpart \n")
## Decision Tree Accuracy Using Rpart
print(tableDecisionTreeUsingRpart.accuracy)
## [1] 0.8012048
The accuracy of Decision Tree using Rpart is 0.8012 meaning that 80.12% correct predictions of all observations
# do predictions as probabilities
PD.pred.rpart = predict(PD.decisiontree, PD_test, type = "prob")
# computing a simple ROC curve (x-axis: fpr, y-axis = tpr)
# labels are actual values, predictors are probability of class
PDRpartpred <- prediction(PD.pred.rpart[, 2], PD_test$Class)
PDRpartperf <- performance(PDRpartpred, "tpr", "fpr")
plot(PDRpartperf, main = "ROC for Decision Tree Using Rpart")
abline(0,1, lty = 2)
### Calculate AUC of Decision Tree Using Rpart
cat("AUC of 'Rpart' Decision Tree \n")
## AUC of 'Rpart' Decision Tree
PDcauc.DecisionTreeUsingRpart = performance(PDRpartpred, "auc")
decisiontree.auc = as.numeric(PDcauc.DecisionTreeUsingRpart@y.values)
print(decisiontree.auc)
## [1] 0.8111203
The AUC of Decision Tree is 0.811 meaning that there is a 81.11% chance that the model will be able to distinguish between positive and negative class.
For the performance comparison, the tree-library-based decision tree shows better performance than the rpart-library-based one with respect to the area under curve, but talking about the prediction accuracy gives the opposite (rpart-library > tree-library). This depends on which purpose. If I want to maximize the accuracy of prediction, I will go for the rpart one. If the focus is on the distinguishment between classes, then choose the tree.
PD.bayes = naiveBayes(Class~., data = PD_train)
PD.predbayes = predict(PD.bayes, PD_test)
tableNaiveBayes = table(predicted = PD.predbayes, actual = PD_test$Class)
cat("Naive Bayes Confusion\n")
## Naive Bayes Confusion
print(tableNaiveBayes)
## actual
## predicted 0 1
## 0 9 2
## 1 292 195
confusionMatrix(tableNaiveBayes)
## Confusion Matrix and Statistics
##
## actual
## predicted 0 1
## 0 9 2
## 1 292 195
##
## Accuracy : 0.4096
## 95% CI : (0.3661, 0.4543)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : 1
##
## Kappa : 0.0157
##
## Mcnemar's Test P-Value : <2e-16
##
## Sensitivity : 0.02990
## Specificity : 0.98985
## Pos Pred Value : 0.81818
## Neg Pred Value : 0.40041
## Prevalence : 0.60442
## Detection Rate : 0.01807
## Detection Prevalence : 0.02209
## Balanced Accuracy : 0.50987
##
## 'Positive' Class : 0
##
cat("\nNaive Bayes Accuracy\n")
##
## Naive Bayes Accuracy
tableNaiveBayes.accuracy = (9 + 195) / (9 + 2 + 292 + 195)
tableNaiveBayes.accuracy
## [1] 0.4096386
The accuracy of Naive Bayes is 0.4096 meaning that 40.96% correct predictions among the total number of observations. This classifier model brings poor performance at predicting the assignment of label options.
PDpred.bayes = predict(PD.bayes, PD_test, type = 'raw')
PDBpred <- prediction(PDpred.bayes[, 2], PD_test$Class)
PDBperf <- performance(PDBpred, "tpr", "fpr")
plot(PDBperf, col = "blueviolet", main = "ROC for Naive Bayes")
### Calculate AUC of Naive Bayes
PDcauc.NaiveBayes <- performance(PDBpred, "auc")
naivebayes.auc = as.numeric(PDcauc.NaiveBayes@y.values)
print(naivebayes.auc)
## [1] 0.7298008
The AUC of Naive Bayes is 0.7298 meaning that there is a 72.98% chance that the model will be able to distinguish between positive and negative class.
PD.Bag = bagging(Class~., data = PD_train, mfinal = 5)
# output as the confidence level of bagging
PDpred.Bag = predict.bagging(PD.Bag, PD_test)
PDBagpred <- prediction(PDpred.Bag$prob[, 2], PD_test$Class)
PDBagperf <- performance(PDBagpred, "tpr", "fpr")
# calculate the accuracy
cat("Bagging Confusion \n")
## Bagging Confusion
print(PDpred.Bag$confusion)
## Observed Class
## Predicted Class 0 1
## 0 265 61
## 1 36 136
confusionMatrix(PDpred.Bag$confusion)
## Confusion Matrix and Statistics
##
## Observed Class
## Predicted Class 0 1
## 0 265 61
## 1 36 136
##
## Accuracy : 0.8052
## 95% CI : (0.7677, 0.8391)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : < 2e-16
##
## Kappa : 0.5835
##
## Mcnemar's Test P-Value : 0.01482
##
## Sensitivity : 0.8804
## Specificity : 0.6904
## Pos Pred Value : 0.8129
## Neg Pred Value : 0.7907
## Prevalence : 0.6044
## Detection Rate : 0.5321
## Detection Prevalence : 0.6546
## Balanced Accuracy : 0.7854
##
## 'Positive' Class : 0
##
cat("\n Bagging accuracy \n")
##
## Bagging accuracy
tableBagging.accuracy <- (265 + 136) / (265 + 61 + 36 + 136)
print(tableBagging.accuracy)
## [1] 0.8052209
The accuracy of Bagging is 0.8052 meaning that 80.52% correct predictions among the total observations
# plot the ROC
plot(PDBagperf, col = "blue", main = "ROC for Bagging")
PDcauc.Bagging = performance(PDBagpred, "auc")
bagging.auc = as.numeric(PDcauc.Bagging@y.values)
print(bagging.auc)
## [1] 0.7931852
The AUC of Bagging is 0.7931 meaning that there is a 79.31% chance that the model will be able to distinguish between positive and negative class.
PD.Boost = boosting(Class~., data = PD_train)
# output as the confidence level of Boosting
PDpred.Boost = predict.boosting(PD.Boost, PD_test)
PDBoostpred <- prediction(PDpred.Boost$prob[, 2], PD_test$Class)
PDBoostperf <- performance(PDBoostpred, "tpr", "fpr")
# Calculate the accuracy of performance
cat("Boosting Confusion \n")
## Boosting Confusion
print(PDpred.Boost$confusion)
## Observed Class
## Predicted Class 0 1
## 0 243 56
## 1 58 141
confusionMatrix(PDpred.Boost$confusion)
## Confusion Matrix and Statistics
##
## Observed Class
## Predicted Class 0 1
## 0 243 56
## 1 58 141
##
## Accuracy : 0.7711
## 95% CI : (0.7316, 0.8073)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : 2.266e-15
##
## Kappa : 0.5221
##
## Mcnemar's Test P-Value : 0.9254
##
## Sensitivity : 0.8073
## Specificity : 0.7157
## Pos Pred Value : 0.8127
## Neg Pred Value : 0.7085
## Prevalence : 0.6044
## Detection Rate : 0.4880
## Detection Prevalence : 0.6004
## Balanced Accuracy : 0.7615
##
## 'Positive' Class : 0
##
cat("\n Boosting accuracy \n")
##
## Boosting accuracy
tableBoosting.accuracy <- (243 + 141) / (243 + 56 + 58 + 141)
print(tableBoosting.accuracy) # 0.771
## [1] 0.7710843
The accuracy of Boosting is 0.7710 meaning that 77.10% correct predictions of the total observations
# plot the ROC
plot(PDBoostperf, col ="red", main = "ROC for Boosting")
PDcauc.Boosting = performance(PDBoostpred, "auc")
boosting.auc = as.numeric(PDcauc.Boosting@y.values)
print(boosting.auc)
## [1] 0.8089532
The AUC of Boosting is 0.8089 meaning that there is a 80.89% chance that the model will be able to distinguish between the positive and negative class.
PD.RandomForest <- randomForest(Class~., data = PD_train, na.action = na.exclude)
# output as the confidence level of Random Forest
PDpredrf <- predict(PD.RandomForest, PD_test)
tableRandomForest <- table(Predicted_Class = PDpredrf, Actual_Class = PD_test$Class)
cat("Random Forest Confusion \n")
## Random Forest Confusion
print(tableRandomForest)
## Actual_Class
## Predicted_Class 0 1
## 0 255 46
## 1 46 151
confusionMatrix(tableRandomForest)
## Confusion Matrix and Statistics
##
## Actual_Class
## Predicted_Class 0 1
## 0 255 46
## 1 46 151
##
## Accuracy : 0.8153
## 95% CI : (0.7783, 0.8484)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : <2e-16
##
## Kappa : 0.6137
##
## Mcnemar's Test P-Value : 1
##
## Sensitivity : 0.8472
## Specificity : 0.7665
## Pos Pred Value : 0.8472
## Neg Pred Value : 0.7665
## Prevalence : 0.6044
## Detection Rate : 0.5120
## Detection Prevalence : 0.6044
## Balanced Accuracy : 0.8068
##
## 'Positive' Class : 0
##
tableRandomForest.accuracy <- (255 + 151) / (255 + 46 + 46 + 151)
print(tableRandomForest.accuracy)
## [1] 0.815261
The accuracy of Random Forest is 0.8152 meaning that 81.52% of the total observations
PDpred.RandomForest <- predict(PD.RandomForest, PD_test, type = "prob")
PDFpred <- prediction(PDpred.RandomForest[, 2], PD_test$Class)
PDFperf <- performance(PDFpred, "tpr", "fpr")
plot(PDFperf, col = "darkgreen", main = "ROC for Random Forest")
PDcauc.RandomForest = performance(PDFpred, "auc")
randomtree.auc = as.numeric(PDcauc.RandomForest@y.values)
print(randomtree.auc)
## [1] 0.8387018
The AUC of Random Forest is 0.8387 meaning that there is a 83.87% chance that the model will be able to distinguish between the positive and negative class.
plot(PDRpartperf)
abline(0,1, lty = 2)
plot(PDBperf, add = TRUE, col = "blueviolet")
plot(PDBagperf, add = TRUE, col = "blue")
plot(PDBoostperf, add = TRUE, col = "red")
plot(PDFperf, add = TRUE, col = "darkgreen")
# Question 7: Comparing Results From The Table
table_comparison <- data.frame(
Classifier_Name = c("Decision Tree", "Naive Bayes", "Bagging", "Boosting", "Random Forest"),
Accuracy = c(tableDecisionTreeUsingRpart.accuracy, tableNaiveBayes.accuracy, tableBagging.accuracy, tableBoosting.accuracy, tableRandomForest.accuracy),
AUC = c(decisiontree.auc, naivebayes.auc, bagging.auc, boosting.auc, randomtree.auc))
table_comparison
Looking at the table, it can be seen that Random Forest shows better performance at assigning the label to the specified record because it focuses on reducing the overfitting (i.e. bias), and shows simplicity and flexibility of decision tree giving beautiful rating in accuracy and Area Under Curve.
cat("Decision Tree Attribute Importance Using Tree Library \n")
## Decision Tree Attribute Importance Using Tree Library
print(summary(PD.tree))
##
## Classification tree:
## tree(formula = Class ~ ., data = PD_train)
## Variables actually used in tree construction:
## [1] "A18" "A01" "A23"
## Number of terminal nodes: 6
## Residual mean deviance: 0.9944 = 1149 / 1155
## Misclassification error rate: 0.2171 = 252 / 1161
cat("\n Decision Tree Attribute Importance Using Rpart Library \n")
##
## Decision Tree Attribute Importance Using Rpart Library
print(PD.decisiontree$variable.importance)
## A23 A01 A18 A17 A08 A24 A04
## 93.6402846 81.2248006 78.9404102 6.9346000 6.5786778 4.0317225 3.6955763
## A02 A11 A09 A12 A05 A16
## 1.8998600 1.5376292 1.1636898 1.1532219 0.2327380 0.1756586
cat("\n Bagging Attribute Importance \n")
##
## Bagging Attribute Importance
print(PD.Bag$importance)
## A01 A02 A04 A05 A06 A08 A09
## 36.2338722 0.0000000 0.0000000 0.0000000 0.0000000 3.6683625 0.0000000
## A10 A11 A12 A13 A14 A15 A16
## 0.0000000 0.0000000 1.2051756 0.0000000 0.0000000 0.0000000 0.0000000
## A17 A18 A19 A20 A21 A23 A24
## 0.7001734 40.0102393 0.0000000 0.5633931 0.0000000 16.9680590 0.6507249
cat("\n Boosting Attribute Importance \n")
##
## Boosting Attribute Importance
print(PD.Boost$importance)
## A01 A02 A04 A05 A06 A08 A09
## 13.5574373 1.0073009 1.6821161 0.1627032 1.2285223 11.7178864 0.7319051
## A10 A11 A12 A13 A14 A15 A16
## 0.3716459 0.4630548 8.3607644 0.0000000 0.8905958 1.3007496 0.5062883
## A17 A18 A19 A20 A21 A23 A24
## 2.7564020 25.7610662 1.0255450 1.0580330 0.6089782 19.3078819 7.5011235
cat("\n Random Forest Attribute Importance \n")
##
## Random Forest Attribute Importance
print(PD.RandomForest$importance)
## MeanDecreaseGini
## A01 90.2781051
## A02 5.6118256
## A04 9.8210828
## A05 0.3669806
## A06 6.2783751
## A08 38.1170818
## A09 2.6974434
## A10 2.4205072
## A11 2.6167829
## A12 28.3095912
## A13 0.3180465
## A14 20.0887095
## A15 6.6058981
## A16 3.9362271
## A17 15.3126607
## A18 96.2740580
## A19 6.3004708
## A20 14.1659984
## A21 2.9399071
## A23 91.0375384
## A24 33.9490405
Looking at the importance of all classifiers, I plan to utilize A01, A18, and A23 as it shows better rate of predicting the phishing data.
PD.originaldecisiontree = tree(Class~A01 + A18 + A23, data = PD_train, method = "class")
plot(PD.originaldecisiontree, main = "New Decision Tree for Phishing Data")
text(PD.tree, pretty = 0)
### Creating confusion matrix and calculating accuracy
PD.originalpreddecisiontree = predict(PD.originaldecisiontree, PD_test, type = "class")
tableOriginalDecisionTreeUsingTree = table(predicted = PD.originalpreddecisiontree, actual = PD_test$Class)
#cat("Original Decision Tree Confusion Using Rpart Library\n")
print(tableOriginalDecisionTreeUsingTree)
## actual
## predicted 0 1
## 0 265 64
## 1 36 133
confusionMatrix(tableOriginalDecisionTreeUsingTree)
## Confusion Matrix and Statistics
##
## actual
## predicted 0 1
## 0 265 64
## 1 36 133
##
## Accuracy : 0.7992
## 95% CI : (0.7613, 0.8335)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : < 2.2e-16
##
## Kappa : 0.5695
##
## Mcnemar's Test P-Value : 0.006934
##
## Sensitivity : 0.8804
## Specificity : 0.6751
## Pos Pred Value : 0.8055
## Neg Pred Value : 0.7870
## Prevalence : 0.6044
## Detection Rate : 0.5321
## Detection Prevalence : 0.6606
## Balanced Accuracy : 0.7778
##
## 'Positive' Class : 0
##
tableOriginalDecisionTreeUsingTree.accuracy = (265 + 133) / (265 + 64 + 36 + 133)
print(tableOriginalDecisionTreeUsingTree.accuracy)
## [1] 0.7991968
Looking at the previous and current accuracy based on the confusion matrix, there is no difference after adjusting the parameter. From the output, I can infer that it does not affect the performance of decision tree powerfully as their accuracies of predicting are equal.
originaldecisiontree.numberOfLeaves = sum(PD.originaldecisiontree$frame$var == "<leaf>")
originaldecisiontree.numberOfLeaves
## [1] 6
The cross validation in this case is used to estimate how well our model will perform on unseen data, providing a more accurate measure of its real-world effectiveness and enables to fine-tune the model for optimal performance. This involves calculating the standard deviation of each node size
cvtest = cv.tree(PD.originaldecisiontree, FUN = prune.misclass)
cvtest
## $size
## [1] 6 4 3 1
##
## $dev
## [1] 261 261 286 509
##
## $k
## [1] -Inf 0.0 32.0 112.5
##
## $method
## [1] "misclass"
##
## attr(,"class")
## [1] "prune" "tree.sequence"
Looking at this picture, it is obvious that the rate of CP decreases as the number of splits increases. This occurs in that the decreased CP penalizes the tree less which in turn leads to more production of branches / splits. Looking at the cross validation, it can be seen that the lowest misclassification error is between the sizes of 6 and 4. Then, I decide to choose 4
The importance of pruning is to avoid the overfitting data by removing any rules that have little effect on the error rate. This is divided into two: a. pre-pruning -> to terminate the trees to grow before the ‘overfitting’ event happens b. post-pruning -> allow the tree to overfit the data after the tree is pruned (common practice)
pruned.decisiontree = prune.misclass(PD.originaldecisiontree, best = 4)
summary(pruned.decisiontree)
##
## Classification tree:
## snip.tree(tree = PD.originaldecisiontree, nodes = c(6L, 15L))
## Number of terminal nodes: 4
## Residual mean deviance: 1.029 = 1191 / 1157
## Misclassification error rate: 0.2171 = 252 / 1161
plot(pruned.decisiontree)
text(pruned.decisiontree, pretty = 0)
# check the accuracy using the pruned model
PD.prunedmodel = predict(PD.originaldecisiontree, PD_test, type = "class")
tablePrunedDecisionTreeUsingTree = table(predicted = PD.prunedmodel, actual = PD_test$Class)
cat("Decision Tree Confusion\n")
## Decision Tree Confusion
print(tablePrunedDecisionTreeUsingTree)
## actual
## predicted 0 1
## 0 265 64
## 1 36 133
confusionMatrix(tablePrunedDecisionTreeUsingTree)
## Confusion Matrix and Statistics
##
## actual
## predicted 0 1
## 0 265 64
## 1 36 133
##
## Accuracy : 0.7992
## 95% CI : (0.7613, 0.8335)
## No Information Rate : 0.6044
## P-Value [Acc > NIR] : < 2.2e-16
##
## Kappa : 0.5695
##
## Mcnemar's Test P-Value : 0.006934
##
## Sensitivity : 0.8804
## Specificity : 0.6751
## Pos Pred Value : 0.8055
## Neg Pred Value : 0.7870
## Prevalence : 0.6044
## Detection Rate : 0.5321
## Detection Prevalence : 0.6606
## Balanced Accuracy : 0.7778
##
## 'Positive' Class : 0
##
Therefore, there is no change.
#Question 10
library(caret)
library(randomForest)
# Define the range for hyperparameters
mtry_values <- c(1:10) # You can adjust this range as needed
# Set the seed for reproducibility
set.seed(123) # example
# Define the training control
train_control <- trainControl(method = "cv", number = 10, search = "random")
# Perform the random search
rf_random <- train(Class ~., data = PD_train,
method = "rf",
trControl = train_control,
tuneLength = 10, # Number of random combinations to try
tuneGrid = expand.grid(mtry = mtry_values),
ntree = 100, # You can set the number of trees here
nodesize = 5) # You can set the node size here
# Get the best model
best_model <- rf_random$bestTune
# Print the best hyperparameters
print(best_model)
## mtry
## 3 3
# Make predictions on the test data
predictions <- predict(rf_random, newdata = PD_test)
# Calculate the confusion matrix
cm <- table(predictions, PD_test$Class)
# Calculate the accuracy
accuracy <- sum(diag(cm)) / sum(cm)
# Print the accuracy
print(accuracy)
## [1] 0.813253
Hypertuning parameter consists of node size, number of trees and number of variables randomly sampled in each split. This also involves the cross validation and random search. The setting of the number utilizes the approach of brute-force. Looking at the result, it can be seen that the best accuracy after hypertuning is 0.813253 meaning that there is a decrease of the performance by 0.002% (from 0.8152 to 0.8132).
#install.packages("neuralnet")
#install.packages("car")
library(neuralnet)
## Warning: package 'neuralnet' was built under R version 4.2.3
##
## Attaching package: 'neuralnet'
## The following object is masked from 'package:ROCR':
##
## prediction
## The following object is masked from 'package:dplyr':
##
## compute
library(car)
## Warning: package 'car' was built under R version 4.2.3
## Loading required package: carData
## Warning: package 'carData' was built under R version 4.2.3
##
## Attaching package: 'car'
## The following object is masked from 'package:dplyr':
##
## recode
## The following object is masked from 'package:purrr':
##
## some
# Scale the data
PD_train_scaled <- as.data.frame(scale(PD_train[, c("A01", "A18", "A23")]))
PD_train_scaled$Class <- PD_train$Class
# Increase stepmax and adjust learning rate
PD_nn <- neuralnet(Class == 1 ~ A01 + A18 + A23, data = PD_train_scaled, hidden = 3, linear.output = FALSE, stepmax = 1e6, learningrate = 0.01)
plot(PD_nn)
PD_pred_neuron = compute(PD_nn, PD_test[c(1, 16, 20)])
prob <- PD_pred_neuron$net.result
pred <- ifelse(prob > 0.5, 1, 0)
# confusion matrix
tableNeuronNetwork = table(observed = PD_test$Class, predicted = pred)
confusionMatrix(tableNeuronNetwork)
## Confusion Matrix and Statistics
##
## predicted
## observed 0 1
## 0 2 299
## 1 0 197
##
## Accuracy : 0.3996
## 95% CI : (0.3563, 0.4441)
## No Information Rate : 0.996
## P-Value [Acc > NIR] : 1
##
## Kappa : 0.0053
##
## Mcnemar's Test P-Value : <2e-16
##
## Sensitivity : 1.000000
## Specificity : 0.397177
## Pos Pred Value : 0.006645
## Neg Pred Value : 1.000000
## Prevalence : 0.004016
## Detection Rate : 0.004016
## Detection Prevalence : 0.604418
## Balanced Accuracy : 0.698589
##
## 'Positive' Class : 0
##
The accuracy of my own Artificial Neuron Network is 0.4217 showing poor performance as opposed to the other classifiers. This is because the error is too high during the prediction of classification for each observation. This requires parameter tuning or the changes of the network architecture.
The new model I am choosing for predicting the phishing is support vector machine
new_manipulated_PD = manipulated_PD
new_manipulated_PD[, 1:21] <- scale(new_manipulated_PD[, 1:21])
set.seed(31225381)
new_train_row = sample(1:nrow(new_manipulated_PD), 0.7*nrow(new_manipulated_PD))
new_PD_train = new_manipulated_PD[new_train_row,]
new_PD_test = new_manipulated_PD[-new_train_row,]
# Fitting SVM to the Training set
#install.packages('e1071')
library(e1071)
new_classifier = svm(formula = Class~ .,
data = new_PD_train,
type = 'C-classification',
kernel = 'linear')
PD_pred_svm = predict(new_classifier, newdata = new_PD_test[-22])
PD_pred_svm_confusion_matrix = table(new_PD_test$Class, PD_pred_svm)
confusionMatrix(PD_pred_svm_confusion_matrix)
## Confusion Matrix and Statistics
##
## PD_pred_svm
## 0 1
## 0 242 59
## 1 78 119
##
## Accuracy : 0.7249
## 95% CI : (0.6834, 0.7637)
## No Information Rate : 0.6426
## P-Value [Acc > NIR] : 5.704e-05
##
## Kappa : 0.415
##
## Mcnemar's Test P-Value : 0.1241
##
## Sensitivity : 0.7562
## Specificity : 0.6685
## Pos Pred Value : 0.8040
## Neg Pred Value : 0.6041
## Prevalence : 0.6426
## Detection Rate : 0.4859
## Detection Prevalence : 0.6044
## Balanced Accuracy : 0.7124
##
## 'Positive' Class : 0
##
The accuracy of Support Vector Machine is 72.49% meaning that 72.49% correct predictions of all observations.