dados<-read.csv("C:/Users/JOANA/Downloads/credit_customers.csv")
library(dplyr)
dados1<-dados%>%rename(c(verificando_status=checking_status,duracao=duration,
historico_credito=credit_history,proposito=purpose,
qualidade_credito=credit_amount,salvando_status=savings_status,
emprego=employment,compromisso_parcelamento=installment_commitment,
status_pessoal=personal_status,outras_partes=other_parties,
residencia_desde=residence_since,magnitude_propriedade=property_magnitude, idade=age,outros_planos_pagamento=other_payment_plans,
habitacao=housing,credito_existente=existing_credits,trabalho=job,
numero_dependente=num_dependents,telefone_proprio=own_telephone,
trabalhador_estrangeiro=foreign_worker,classe=class))
Ànalise Exploratória dos Dados
library(skimr)
skim(dados1)
Data summary
| Name |
dados1 |
| Number of rows |
1000 |
| Number of columns |
21 |
| _______________________ |
|
| Column type frequency: |
|
| character |
14 |
| numeric |
7 |
| ________________________ |
|
| Group variables |
None |
Variable type: character
| verificando_status |
0 |
1 |
2 |
11 |
0 |
4 |
0 |
| historico_credito |
0 |
1 |
8 |
30 |
0 |
5 |
0 |
| proposito |
0 |
1 |
5 |
19 |
0 |
10 |
0 |
| salvando_status |
0 |
1 |
4 |
16 |
0 |
5 |
0 |
| emprego |
0 |
1 |
2 |
10 |
0 |
5 |
0 |
| status_pessoal |
0 |
1 |
11 |
18 |
0 |
4 |
0 |
| outras_partes |
0 |
1 |
4 |
12 |
0 |
3 |
0 |
| magnitude_propriedade |
0 |
1 |
3 |
17 |
0 |
4 |
0 |
| outros_planos_pagamento |
0 |
1 |
4 |
6 |
0 |
3 |
0 |
| habitacao |
0 |
1 |
3 |
8 |
0 |
3 |
0 |
| trabalho |
0 |
1 |
7 |
25 |
0 |
4 |
0 |
| telefone_proprio |
0 |
1 |
3 |
4 |
0 |
2 |
0 |
| trabalhador_estrangeiro |
0 |
1 |
2 |
3 |
0 |
2 |
0 |
| classe |
0 |
1 |
3 |
4 |
0 |
2 |
0 |
Variable type: numeric
| duracao |
0 |
1 |
20.90 |
12.06 |
4 |
12.0 |
18.0 |
24.00 |
72 |
▇▇▂▁▁ |
| qualidade_credito |
0 |
1 |
3271.26 |
2822.74 |
250 |
1365.5 |
2319.5 |
3972.25 |
18424 |
▇▂▁▁▁ |
| compromisso_parcelamento |
0 |
1 |
2.97 |
1.12 |
1 |
2.0 |
3.0 |
4.00 |
4 |
▂▃▁▂▇ |
| residencia_desde |
0 |
1 |
2.85 |
1.10 |
1 |
2.0 |
3.0 |
4.00 |
4 |
▂▆▁▃▇ |
| idade |
0 |
1 |
35.55 |
11.38 |
19 |
27.0 |
33.0 |
42.00 |
75 |
▇▆▃▁▁ |
| credito_existente |
0 |
1 |
1.41 |
0.58 |
1 |
1.0 |
1.0 |
2.00 |
4 |
▇▅▁▁▁ |
| numero_dependente |
0 |
1 |
1.16 |
0.36 |
1 |
1.0 |
1.0 |
1.00 |
2 |
▇▁▁▁▂ |
attach(dados1)
Transformando as Variáveis Caracter em Fator
dados1$verificando_status<- as.factor(as.character(verificando_status))
dados1$historico_credito<- as.factor(as.character(historico_credito))
dados1$proposito<- as.factor(as.character(proposito))
dados1$salvando_status<- as.factor(as.character(salvando_status))
dados1$emprego<- as.factor(as.character(emprego))
dados1$status_pessoal<- as.factor(as.character(status_pessoal))
dados1$outras_partes<- as.factor(as.character(outras_partes))
dados1$magnitude_propriedade<- as.factor(as.character(magnitude_propriedade))
dados1$outros_planos_pagamento<- as.factor(as.character(outros_planos_pagamento))
dados1$habitacao<- as.factor(as.character(habitacao))
dados1$trabalho<- as.factor(as.character(trabalho))
dados1$telefone_proprio<- ifelse(dados1$telefone_proprio=="yes",1,0)
str(dados1)
## 'data.frame': 1000 obs. of 21 variables:
## $ verificando_status : Factor w/ 4 levels "<0",">=200","0<=X<200",..: 1 3 4 1 1 4 4 3 4 3 ...
## $ duracao : num 6 48 12 42 24 36 24 36 12 30 ...
## $ historico_credito : Factor w/ 5 levels "all paid","critical/other existing credit",..: 2 4 2 4 3 4 4 4 4 2 ...
## $ proposito : Factor w/ 10 levels "business","domestic appliance",..: 7 7 3 4 5 3 4 10 7 5 ...
## $ qualidade_credito : num 1169 5951 2096 7882 4870 ...
## $ salvando_status : Factor w/ 5 levels "<100",">=1000",..: 5 1 1 1 1 5 4 1 2 1 ...
## $ emprego : Factor w/ 5 levels "<1",">=7","1<=X<4",..: 2 3 4 4 3 3 2 3 4 5 ...
## $ compromisso_parcelamento: num 4 2 2 2 3 2 3 2 2 4 ...
## $ status_pessoal : Factor w/ 4 levels "female div/dep/mar",..: 4 1 4 4 4 4 4 4 2 3 ...
## $ outras_partes : Factor w/ 3 levels "co applicant",..: 3 3 3 2 3 3 3 3 3 3 ...
## $ residencia_desde : num 4 2 3 4 4 4 4 2 4 2 ...
## $ magnitude_propriedade : Factor w/ 4 levels "car","life insurance",..: 4 4 4 2 3 3 2 1 4 1 ...
## $ idade : num 67 22 49 45 53 35 53 35 61 28 ...
## $ outros_planos_pagamento : Factor w/ 3 levels "bank","none",..: 2 2 2 2 2 2 2 2 2 2 ...
## $ habitacao : Factor w/ 3 levels "for free","own",..: 2 2 2 1 1 1 2 3 2 2 ...
## $ credito_existente : num 2 1 1 1 2 1 1 1 1 2 ...
## $ trabalho : Factor w/ 4 levels "high qualif/self emp/mgmt",..: 2 2 4 2 2 4 2 1 4 1 ...
## $ numero_dependente : num 1 1 2 2 2 2 1 1 1 1 ...
## $ telefone_proprio : num 1 0 0 0 0 1 0 1 0 0 ...
## $ trabalhador_estrangeiro : chr "yes" "yes" "yes" "yes" ...
## $ classe : chr "good" "bad" "good" "good" ...
dados1$trabalhador_estrangeiro<- ifelse(verificando_status=="yes",1,0)
dados1$classe<- as.factor(as.character(classe))
str(dados1)
## 'data.frame': 1000 obs. of 21 variables:
## $ verificando_status : Factor w/ 4 levels "<0",">=200","0<=X<200",..: 1 3 4 1 1 4 4 3 4 3 ...
## $ duracao : num 6 48 12 42 24 36 24 36 12 30 ...
## $ historico_credito : Factor w/ 5 levels "all paid","critical/other existing credit",..: 2 4 2 4 3 4 4 4 4 2 ...
## $ proposito : Factor w/ 10 levels "business","domestic appliance",..: 7 7 3 4 5 3 4 10 7 5 ...
## $ qualidade_credito : num 1169 5951 2096 7882 4870 ...
## $ salvando_status : Factor w/ 5 levels "<100",">=1000",..: 5 1 1 1 1 5 4 1 2 1 ...
## $ emprego : Factor w/ 5 levels "<1",">=7","1<=X<4",..: 2 3 4 4 3 3 2 3 4 5 ...
## $ compromisso_parcelamento: num 4 2 2 2 3 2 3 2 2 4 ...
## $ status_pessoal : Factor w/ 4 levels "female div/dep/mar",..: 4 1 4 4 4 4 4 4 2 3 ...
## $ outras_partes : Factor w/ 3 levels "co applicant",..: 3 3 3 2 3 3 3 3 3 3 ...
## $ residencia_desde : num 4 2 3 4 4 4 4 2 4 2 ...
## $ magnitude_propriedade : Factor w/ 4 levels "car","life insurance",..: 4 4 4 2 3 3 2 1 4 1 ...
## $ idade : num 67 22 49 45 53 35 53 35 61 28 ...
## $ outros_planos_pagamento : Factor w/ 3 levels "bank","none",..: 2 2 2 2 2 2 2 2 2 2 ...
## $ habitacao : Factor w/ 3 levels "for free","own",..: 2 2 2 1 1 1 2 3 2 2 ...
## $ credito_existente : num 2 1 1 1 2 1 1 1 1 2 ...
## $ trabalho : Factor w/ 4 levels "high qualif/self emp/mgmt",..: 2 2 4 2 2 4 2 1 4 1 ...
## $ numero_dependente : num 1 1 2 2 2 2 1 1 1 1 ...
## $ telefone_proprio : num 1 0 0 0 0 1 0 1 0 0 ...
## $ trabalhador_estrangeiro : num 0 0 0 0 0 0 0 0 0 0 ...
## $ classe : Factor w/ 2 levels "bad","good": 2 1 2 2 1 2 2 2 2 1 ...
skim(dados1)
Data summary
| Name |
dados1 |
| Number of rows |
1000 |
| Number of columns |
21 |
| _______________________ |
|
| Column type frequency: |
|
| factor |
12 |
| numeric |
9 |
| ________________________ |
|
| Group variables |
None |
Variable type: factor
| verificando_status |
0 |
1 |
FALSE |
4 |
no : 394, <0: 274, 0<=: 269, >=2: 63 |
| historico_credito |
0 |
1 |
FALSE |
5 |
exi: 530, cri: 293, del: 88, all: 49 |
| proposito |
0 |
1 |
FALSE |
10 |
rad: 280, new: 234, fur: 181, use: 103 |
| salvando_status |
0 |
1 |
FALSE |
5 |
<10: 603, no : 183, 100: 103, 500: 63 |
| emprego |
0 |
1 |
FALSE |
5 |
1<=: 339, >=7: 253, 4<=: 174, <1: 172 |
| status_pessoal |
0 |
1 |
FALSE |
4 |
mal: 548, fem: 310, mal: 92, mal: 50 |
| outras_partes |
0 |
1 |
FALSE |
3 |
non: 907, gua: 52, co : 41 |
| magnitude_propriedade |
0 |
1 |
FALSE |
4 |
car: 332, rea: 282, lif: 232, no : 154 |
| outros_planos_pagamento |
0 |
1 |
FALSE |
3 |
non: 814, ban: 139, sto: 47 |
| habitacao |
0 |
1 |
FALSE |
3 |
own: 713, ren: 179, for: 108 |
| trabalho |
0 |
1 |
FALSE |
4 |
ski: 630, uns: 200, hig: 148, une: 22 |
| classe |
0 |
1 |
FALSE |
2 |
goo: 700, bad: 300 |
Variable type: numeric
| duracao |
0 |
1 |
20.90 |
12.06 |
4 |
12.0 |
18.0 |
24.00 |
72 |
▇▇▂▁▁ |
| qualidade_credito |
0 |
1 |
3271.26 |
2822.74 |
250 |
1365.5 |
2319.5 |
3972.25 |
18424 |
▇▂▁▁▁ |
| compromisso_parcelamento |
0 |
1 |
2.97 |
1.12 |
1 |
2.0 |
3.0 |
4.00 |
4 |
▂▃▁▂▇ |
| residencia_desde |
0 |
1 |
2.85 |
1.10 |
1 |
2.0 |
3.0 |
4.00 |
4 |
▂▆▁▃▇ |
| idade |
0 |
1 |
35.55 |
11.38 |
19 |
27.0 |
33.0 |
42.00 |
75 |
▇▆▃▁▁ |
| credito_existente |
0 |
1 |
1.41 |
0.58 |
1 |
1.0 |
1.0 |
2.00 |
4 |
▇▅▁▁▁ |
| numero_dependente |
0 |
1 |
1.16 |
0.36 |
1 |
1.0 |
1.0 |
1.00 |
2 |
▇▁▁▁▂ |
| telefone_proprio |
0 |
1 |
0.40 |
0.49 |
0 |
0.0 |
0.0 |
1.00 |
1 |
▇▁▁▁▆ |
| trabalhador_estrangeiro |
0 |
1 |
0.00 |
0.00 |
0 |
0.0 |
0.0 |
0.00 |
0 |
▁▁▇▁▁ |
Utilizando K-NN Classificação
Modelo Básico
library(caret)
fit= knn3(classe ~ ., data = dados1)
fit
## 5-nearest neighbor model
## Training set outcome distribution:
##
## bad good
## 300 700
Pré-processamento
set.seed(13)
fitp1 <- train(
classe ~ .,
data = dados1,
method = 'knn',
preProcess = c("center", "scale")
)
fitp1
## k-Nearest Neighbors
##
## 1000 samples
## 20 predictor
## 2 classes: 'bad', 'good'
##
## Pre-processing: centered (48), scaled (48)
## Resampling: Bootstrapped (25 reps)
## Summary of sample sizes: 1000, 1000, 1000, 1000, 1000, 1000, ...
## Resampling results across tuning parameters:
##
## k Accuracy Kappa
## 5 0.6829793 0.1963699
## 7 0.6854597 0.1862340
## 9 0.6947664 0.1955296
##
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was k = 9.
plot(fitp1)

Removendo a variável resposta do conjunto de teste.
tst.features = subset(tst, select = -c(classe))
tst.target = subset(tst, select = classe)[,1]
predictions = predict(fit10, newdata = tst.features)
Calculamos a matriz de confusão e medidas de avaliação
confusionMatrix(predictions, tst.target)
## Confusion Matrix and Statistics
##
## Reference
## Prediction bad good
## bad 21 7
## good 39 133
##
## Accuracy : 0.77
## 95% CI : (0.7054, 0.8264)
## No Information Rate : 0.7
## P-Value [Acc > NIR] : 0.01687
##
## Kappa : 0.3539
##
## Mcnemar's Test P-Value : 4.861e-06
##
## Sensitivity : 0.3500
## Specificity : 0.9500
## Pos Pred Value : 0.7500
## Neg Pred Value : 0.7733
## Prevalence : 0.3000
## Detection Rate : 0.1050
## Detection Prevalence : 0.1400
## Balanced Accuracy : 0.6500
##
## 'Positive' Class : bad
##
Tunning do número de vizinhos
set.seed(13)
tuneGrid <- expand.grid(
k = seq(7, 13, by = 1)
)
set.seed(13)
fit12 <- train(
classe ~ .,
data = trn,
method = 'knn',
preProcess = c("center", "scale"),
trControl = ctrl,
tuneGrid = tuneGrid
)
fit12
## k-Nearest Neighbors
##
## 800 samples
## 20 predictor
## 2 classes: 'bad', 'good'
##
## Pre-processing: centered (48), scaled (48)
## Resampling: Cross-Validated (10 fold)
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ...
## Resampling results across tuning parameters:
##
## k Accuracy Kappa
## 7 0.71500 0.2359817
## 8 0.72000 0.2411513
## 9 0.71750 0.2201487
## 10 0.71500 0.2071757
## 11 0.72125 0.2185028
## 12 0.71625 0.2026276
## 13 0.71750 0.1959997
##
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was k = 11.
predictions = predict(fit12, newdata = tst.features)
confusionMatrix(predictions, tst.target)
## Confusion Matrix and Statistics
##
## Reference
## Prediction bad good
## bad 18 6
## good 42 134
##
## Accuracy : 0.76
## 95% CI : (0.6947, 0.8174)
## No Information Rate : 0.7
## P-Value [Acc > NIR] : 0.03595
##
## Kappa : 0.3103
##
## Mcnemar's Test P-Value : 4.376e-07
##
## Sensitivity : 0.3000
## Specificity : 0.9571
## Pos Pred Value : 0.7500
## Neg Pred Value : 0.7614
## Prevalence : 0.3000
## Detection Rate : 0.0900
## Detection Prevalence : 0.1200
## Balanced Accuracy : 0.6286
##
## 'Positive' Class : bad
##
Utilizando o Método de Árvore Classificação
Método Básico
library(rpart)
fit_arvore = rpart(classe ~ ., data = dados1)
fit_arvore
## n= 1000
##
## node), split, n, loss, yval, (yprob)
## * denotes terminal node
##
## 1) root 1000 300 good (0.3000000 0.7000000)
## 2) verificando_status=<0,0<=X<200 543 240 good (0.4419890 0.5580110)
## 4) duracao>=22.5 237 103 bad (0.5654008 0.4345992)
## 8) salvando_status=<100,100<=X<500,500<=X<1000 196 74 bad (0.6224490 0.3775510)
## 16) duracao>=47.5 36 5 bad (0.8611111 0.1388889) *
## 17) duracao< 47.5 160 69 bad (0.5687500 0.4312500)
## 34) proposito=business,education,furniture/equipment,new car,other,radio/tv,repairs 137 52 bad (0.6204380 0.3795620) *
## 35) proposito=used car 23 6 good (0.2608696 0.7391304) *
## 9) salvando_status=>=1000,no known savings 41 12 good (0.2926829 0.7073171) *
## 5) duracao< 22.5 306 106 good (0.3464052 0.6535948)
## 10) historico_credito=all paid,no credits/all paid 28 7 bad (0.7500000 0.2500000) *
## 11) historico_credito=critical/other existing credit,delayed previously,existing paid 278 85 good (0.3057554 0.6942446)
## 22) qualidade_credito>=7491.5 7 1 bad (0.8571429 0.1428571) *
## 23) qualidade_credito< 7491.5 271 79 good (0.2915129 0.7084871)
## 46) proposito=domestic appliance,education 15 5 bad (0.6666667 0.3333333) *
## 47) proposito=business,furniture/equipment,new car,other,radio/tv,repairs,retraining,used car 256 69 good (0.2695312 0.7304688)
## 94) duracao>=11.5 183 60 good (0.3278689 0.6721311)
## 188) qualidade_credito< 1387.5 65 31 good (0.4769231 0.5230769)
## 376) magnitude_propriedade=car,no known property 20 3 bad (0.8500000 0.1500000) *
## 377) magnitude_propriedade=life insurance,real estate 45 14 good (0.3111111 0.6888889) *
## 189) qualidade_credito>=1387.5 118 29 good (0.2457627 0.7542373) *
## 95) duracao< 11.5 73 9 good (0.1232877 0.8767123) *
## 3) verificando_status=>=200,no checking 457 60 good (0.1312910 0.8687090) *
plot(fit_arvore)
text(fit_arvore, cex = 0.50)

Pré-processamento
set.seed(13)
fit21 <- train(
classe ~ .,
data = dados1,
method = 'rpart',
preProcess = c("center", "scale")
)
fit21
## CART
##
## 1000 samples
## 20 predictor
## 2 classes: 'bad', 'good'
##
## Pre-processing: centered (48), scaled (48)
## Resampling: Bootstrapped (25 reps)
## Summary of sample sizes: 1000, 1000, 1000, 1000, 1000, 1000, ...
## Resampling results across tuning parameters:
##
## cp Accuracy Kappa
## 0.01666667 0.7188142 0.2609517
## 0.02166667 0.7159060 0.2436917
## 0.03166667 0.7132175 0.2291673
##
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was cp = 0.01666667.
plot(fit21)

avaliando o modelo no conjunto de teste
tst.features_arv = subset(tst, select = -c(classe))
tst.target_arv = subset(tst, select = classe)[,1]
predictions_arv = predict(fit10, newdata = tst.features)
confusionMatrix(predictions_arv, tst.target_arv)
## Confusion Matrix and Statistics
##
## Reference
## Prediction bad good
## bad 21 7
## good 39 133
##
## Accuracy : 0.77
## 95% CI : (0.7054, 0.8264)
## No Information Rate : 0.7
## P-Value [Acc > NIR] : 0.01687
##
## Kappa : 0.3539
##
## Mcnemar's Test P-Value : 4.861e-06
##
## Sensitivity : 0.3500
## Specificity : 0.9500
## Pos Pred Value : 0.7500
## Neg Pred Value : 0.7733
## Prevalence : 0.3000
## Detection Rate : 0.1050
## Detection Prevalence : 0.1400
## Balanced Accuracy : 0.6500
##
## 'Positive' Class : bad
##
Tunning do parâmetro de complexidade
set.seed(13)
tuneGrid_arv <- expand.grid(
cp = seq(0, 1, by = .01)
)
set.seed(13)
fit24 <- train(
classe ~ .,
data = trn_arv,
method = 'rpart',
preProcess = c("center", "scale"),
trControl = ctrl_arv,
tuneGrid = tuneGrid_arv
)
fit24
## CART
##
## 800 samples
## 20 predictor
## 2 classes: 'bad', 'good'
##
## Pre-processing: centered (48), scaled (48)
## Resampling: Cross-Validated (10 fold)
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ...
## Resampling results across tuning parameters:
##
## cp Accuracy Kappa
## 0.00 0.68500 0.21858775
## 0.01 0.69875 0.22832196
## 0.02 0.68250 0.15504360
## 0.03 0.68625 0.08051324
## 0.04 0.69625 0.05752156
## 0.05 0.69625 0.03179106
## 0.06 0.70000 0.00000000
## 0.07 0.70000 0.00000000
## 0.08 0.70000 0.00000000
## 0.09 0.70000 0.00000000
## 0.10 0.70000 0.00000000
## 0.11 0.70000 0.00000000
## 0.12 0.70000 0.00000000
## 0.13 0.70000 0.00000000
## 0.14 0.70000 0.00000000
## 0.15 0.70000 0.00000000
## 0.16 0.70000 0.00000000
## 0.17 0.70000 0.00000000
## 0.18 0.70000 0.00000000
## 0.19 0.70000 0.00000000
## 0.20 0.70000 0.00000000
## 0.21 0.70000 0.00000000
## 0.22 0.70000 0.00000000
## 0.23 0.70000 0.00000000
## 0.24 0.70000 0.00000000
## 0.25 0.70000 0.00000000
## 0.26 0.70000 0.00000000
## 0.27 0.70000 0.00000000
## 0.28 0.70000 0.00000000
## 0.29 0.70000 0.00000000
## 0.30 0.70000 0.00000000
## 0.31 0.70000 0.00000000
## 0.32 0.70000 0.00000000
## 0.33 0.70000 0.00000000
## 0.34 0.70000 0.00000000
## 0.35 0.70000 0.00000000
## 0.36 0.70000 0.00000000
## 0.37 0.70000 0.00000000
## 0.38 0.70000 0.00000000
## 0.39 0.70000 0.00000000
## 0.40 0.70000 0.00000000
## 0.41 0.70000 0.00000000
## 0.42 0.70000 0.00000000
## 0.43 0.70000 0.00000000
## 0.44 0.70000 0.00000000
## 0.45 0.70000 0.00000000
## 0.46 0.70000 0.00000000
## 0.47 0.70000 0.00000000
## 0.48 0.70000 0.00000000
## 0.49 0.70000 0.00000000
## 0.50 0.70000 0.00000000
## 0.51 0.70000 0.00000000
## 0.52 0.70000 0.00000000
## 0.53 0.70000 0.00000000
## 0.54 0.70000 0.00000000
## 0.55 0.70000 0.00000000
## 0.56 0.70000 0.00000000
## 0.57 0.70000 0.00000000
## 0.58 0.70000 0.00000000
## 0.59 0.70000 0.00000000
## 0.60 0.70000 0.00000000
## 0.61 0.70000 0.00000000
## 0.62 0.70000 0.00000000
## 0.63 0.70000 0.00000000
## 0.64 0.70000 0.00000000
## 0.65 0.70000 0.00000000
## 0.66 0.70000 0.00000000
## 0.67 0.70000 0.00000000
## 0.68 0.70000 0.00000000
## 0.69 0.70000 0.00000000
## 0.70 0.70000 0.00000000
## 0.71 0.70000 0.00000000
## 0.72 0.70000 0.00000000
## 0.73 0.70000 0.00000000
## 0.74 0.70000 0.00000000
## 0.75 0.70000 0.00000000
## 0.76 0.70000 0.00000000
## 0.77 0.70000 0.00000000
## 0.78 0.70000 0.00000000
## 0.79 0.70000 0.00000000
## 0.80 0.70000 0.00000000
## 0.81 0.70000 0.00000000
## 0.82 0.70000 0.00000000
## 0.83 0.70000 0.00000000
## 0.84 0.70000 0.00000000
## 0.85 0.70000 0.00000000
## 0.86 0.70000 0.00000000
## 0.87 0.70000 0.00000000
## 0.88 0.70000 0.00000000
## 0.89 0.70000 0.00000000
## 0.90 0.70000 0.00000000
## 0.91 0.70000 0.00000000
## 0.92 0.70000 0.00000000
## 0.93 0.70000 0.00000000
## 0.94 0.70000 0.00000000
## 0.95 0.70000 0.00000000
## 0.96 0.70000 0.00000000
## 0.97 0.70000 0.00000000
## 0.98 0.70000 0.00000000
## 0.99 0.70000 0.00000000
## 1.00 0.70000 0.00000000
##
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was cp = 1.
plot(fit24)

predictions_arv3 = predict(fit24, newdata = tst.features_arv)
confusionMatrix(predictions_arv3, tst.target_arv)
## Confusion Matrix and Statistics
##
## Reference
## Prediction bad good
## bad 0 0
## good 60 140
##
## Accuracy : 0.7
## 95% CI : (0.6314, 0.7626)
## No Information Rate : 0.7
## P-Value [Acc > NIR] : 0.5348
##
## Kappa : 0
##
## Mcnemar's Test P-Value : 2.599e-14
##
## Sensitivity : 0.0
## Specificity : 1.0
## Pos Pred Value : NaN
## Neg Pred Value : 0.7
## Prevalence : 0.3
## Detection Rate : 0.0
## Detection Prevalence : 0.0
## Balanced Accuracy : 0.5
##
## 'Positive' Class : bad
##
Método Floresta Aleatória
ctrl_floresta <- trainControl(method = "repeatedcv",
number = 10,
repeats = 3,
search="random")
set.seed(13)
fit25<- train(
classe ~ .,
data = trn_arv,
method = 'rf',
preProcess = c("center", "scale"),
trControl = ctrl_floresta
)
fit25
## Random Forest
##
## 800 samples
## 20 predictor
## 2 classes: 'bad', 'good'
##
## Pre-processing: centered (48), scaled (48)
## Resampling: Cross-Validated (10 fold, repeated 3 times)
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ...
## Resampling results across tuning parameters:
##
## mtry Accuracy Kappa
## 10 0.7433333 0.2982631
## 13 0.7470833 0.3135297
## 37 0.7387500 0.3153433
##
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was mtry = 13.
plot(fit25)

predictions_floresta = predict(fit25, newdata = tst.features_arv)
confusionMatrix(predictions_floresta, tst.target_arv)
## Confusion Matrix and Statistics
##
## Reference
## Prediction bad good
## bad 25 13
## good 35 127
##
## Accuracy : 0.76
## 95% CI : (0.6947, 0.8174)
## No Information Rate : 0.7
## P-Value [Acc > NIR] : 0.035948
##
## Kappa : 0.3617
##
## Mcnemar's Test P-Value : 0.002437
##
## Sensitivity : 0.4167
## Specificity : 0.9071
## Pos Pred Value : 0.6579
## Neg Pred Value : 0.7840
## Prevalence : 0.3000
## Detection Rate : 0.1250
## Detection Prevalence : 0.1900
## Balanced Accuracy : 0.6619
##
## 'Positive' Class : bad
##
Etapa de tunning
set.seed(13)
tuneGrid_floresta <- expand.grid(
mtry = 1:4
)
set.seed(13)
fit26 <- train(
classe ~ .,
data = trn_arv,
method = 'rf',
preProcess = c("center", "scale"),
trControl = ctrl_floresta,
tuneGrid = tuneGrid_floresta
)
fit26
## Random Forest
##
## 800 samples
## 20 predictor
## 2 classes: 'bad', 'good'
##
## Pre-processing: centered (48), scaled (48)
## Resampling: Cross-Validated (10 fold, repeated 3 times)
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ...
## Resampling results across tuning parameters:
##
## mtry Accuracy Kappa
## 1 0.7000000 0.0000000
## 2 0.7187500 0.1001517
## 3 0.7395833 0.2268547
## 4 0.7362500 0.2483342
##
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was mtry = 3.
plot(fit26)

predictions_floresta1 = predict(fit26, newdata = tst.features_arv)
confusionMatrix(predictions_floresta1, tst.target_arv)
## Confusion Matrix and Statistics
##
## Reference
## Prediction bad good
## bad 17 2
## good 43 138
##
## Accuracy : 0.775
## 95% CI : (0.7108, 0.8309)
## No Information Rate : 0.7
## P-Value [Acc > NIR] : 0.01113
##
## Kappa : 0.3343
##
## Mcnemar's Test P-Value : 2.479e-09
##
## Sensitivity : 0.2833
## Specificity : 0.9857
## Pos Pred Value : 0.8947
## Neg Pred Value : 0.7624
## Prevalence : 0.3000
## Detection Rate : 0.0850
## Detection Prevalence : 0.0950
## Balanced Accuracy : 0.6345
##
## 'Positive' Class : bad
##