dados<-read.csv("C:/Users/JOANA/Downloads/credit_customers.csv")
library(dplyr)
dados1<-dados%>%rename(c(verificando_status=checking_status,duracao=duration,
                        historico_credito=credit_history,proposito=purpose,
                        qualidade_credito=credit_amount,salvando_status=savings_status,
                        emprego=employment,compromisso_parcelamento=installment_commitment,
                        status_pessoal=personal_status,outras_partes=other_parties,
                        residencia_desde=residence_since,magnitude_propriedade=property_magnitude,                      idade=age,outros_planos_pagamento=other_payment_plans,
                        habitacao=housing,credito_existente=existing_credits,trabalho=job,
                        numero_dependente=num_dependents,telefone_proprio=own_telephone,
                        trabalhador_estrangeiro=foreign_worker,classe=class))

Ànalise Exploratória dos Dados

library(skimr)
skim(dados1)
Data summary
Name dados1
Number of rows 1000
Number of columns 21
_______________________
Column type frequency:
character 14
numeric 7
________________________
Group variables None

Variable type: character

skim_variable n_missing complete_rate min max empty n_unique whitespace
verificando_status 0 1 2 11 0 4 0
historico_credito 0 1 8 30 0 5 0
proposito 0 1 5 19 0 10 0
salvando_status 0 1 4 16 0 5 0
emprego 0 1 2 10 0 5 0
status_pessoal 0 1 11 18 0 4 0
outras_partes 0 1 4 12 0 3 0
magnitude_propriedade 0 1 3 17 0 4 0
outros_planos_pagamento 0 1 4 6 0 3 0
habitacao 0 1 3 8 0 3 0
trabalho 0 1 7 25 0 4 0
telefone_proprio 0 1 3 4 0 2 0
trabalhador_estrangeiro 0 1 2 3 0 2 0
classe 0 1 3 4 0 2 0

Variable type: numeric

skim_variable n_missing complete_rate mean sd p0 p25 p50 p75 p100 hist
duracao 0 1 20.90 12.06 4 12.0 18.0 24.00 72 ▇▇▂▁▁
qualidade_credito 0 1 3271.26 2822.74 250 1365.5 2319.5 3972.25 18424 ▇▂▁▁▁
compromisso_parcelamento 0 1 2.97 1.12 1 2.0 3.0 4.00 4 ▂▃▁▂▇
residencia_desde 0 1 2.85 1.10 1 2.0 3.0 4.00 4 ▂▆▁▃▇
idade 0 1 35.55 11.38 19 27.0 33.0 42.00 75 ▇▆▃▁▁
credito_existente 0 1 1.41 0.58 1 1.0 1.0 2.00 4 ▇▅▁▁▁
numero_dependente 0 1 1.16 0.36 1 1.0 1.0 1.00 2 ▇▁▁▁▂
attach(dados1)

Transformando as Variáveis Caracter em Fator

dados1$verificando_status<- as.factor(as.character(verificando_status))
dados1$historico_credito<- as.factor(as.character(historico_credito))
dados1$proposito<- as.factor(as.character(proposito))
dados1$salvando_status<- as.factor(as.character(salvando_status))
dados1$emprego<- as.factor(as.character(emprego))
dados1$status_pessoal<- as.factor(as.character(status_pessoal))
dados1$outras_partes<- as.factor(as.character(outras_partes))
dados1$magnitude_propriedade<- as.factor(as.character(magnitude_propriedade))
dados1$outros_planos_pagamento<- as.factor(as.character(outros_planos_pagamento))
dados1$habitacao<- as.factor(as.character(habitacao))
dados1$trabalho<- as.factor(as.character(trabalho))
dados1$telefone_proprio<- ifelse(dados1$telefone_proprio=="yes",1,0)
str(dados1)
## 'data.frame':    1000 obs. of  21 variables:
##  $ verificando_status      : Factor w/ 4 levels "<0",">=200","0<=X<200",..: 1 3 4 1 1 4 4 3 4 3 ...
##  $ duracao                 : num  6 48 12 42 24 36 24 36 12 30 ...
##  $ historico_credito       : Factor w/ 5 levels "all paid","critical/other existing credit",..: 2 4 2 4 3 4 4 4 4 2 ...
##  $ proposito               : Factor w/ 10 levels "business","domestic appliance",..: 7 7 3 4 5 3 4 10 7 5 ...
##  $ qualidade_credito       : num  1169 5951 2096 7882 4870 ...
##  $ salvando_status         : Factor w/ 5 levels "<100",">=1000",..: 5 1 1 1 1 5 4 1 2 1 ...
##  $ emprego                 : Factor w/ 5 levels "<1",">=7","1<=X<4",..: 2 3 4 4 3 3 2 3 4 5 ...
##  $ compromisso_parcelamento: num  4 2 2 2 3 2 3 2 2 4 ...
##  $ status_pessoal          : Factor w/ 4 levels "female div/dep/mar",..: 4 1 4 4 4 4 4 4 2 3 ...
##  $ outras_partes           : Factor w/ 3 levels "co applicant",..: 3 3 3 2 3 3 3 3 3 3 ...
##  $ residencia_desde        : num  4 2 3 4 4 4 4 2 4 2 ...
##  $ magnitude_propriedade   : Factor w/ 4 levels "car","life insurance",..: 4 4 4 2 3 3 2 1 4 1 ...
##  $ idade                   : num  67 22 49 45 53 35 53 35 61 28 ...
##  $ outros_planos_pagamento : Factor w/ 3 levels "bank","none",..: 2 2 2 2 2 2 2 2 2 2 ...
##  $ habitacao               : Factor w/ 3 levels "for free","own",..: 2 2 2 1 1 1 2 3 2 2 ...
##  $ credito_existente       : num  2 1 1 1 2 1 1 1 1 2 ...
##  $ trabalho                : Factor w/ 4 levels "high qualif/self emp/mgmt",..: 2 2 4 2 2 4 2 1 4 1 ...
##  $ numero_dependente       : num  1 1 2 2 2 2 1 1 1 1 ...
##  $ telefone_proprio        : num  1 0 0 0 0 1 0 1 0 0 ...
##  $ trabalhador_estrangeiro : chr  "yes" "yes" "yes" "yes" ...
##  $ classe                  : chr  "good" "bad" "good" "good" ...
dados1$trabalhador_estrangeiro<- ifelse(verificando_status=="yes",1,0)
dados1$classe<- as.factor(as.character(classe))
str(dados1)
## 'data.frame':    1000 obs. of  21 variables:
##  $ verificando_status      : Factor w/ 4 levels "<0",">=200","0<=X<200",..: 1 3 4 1 1 4 4 3 4 3 ...
##  $ duracao                 : num  6 48 12 42 24 36 24 36 12 30 ...
##  $ historico_credito       : Factor w/ 5 levels "all paid","critical/other existing credit",..: 2 4 2 4 3 4 4 4 4 2 ...
##  $ proposito               : Factor w/ 10 levels "business","domestic appliance",..: 7 7 3 4 5 3 4 10 7 5 ...
##  $ qualidade_credito       : num  1169 5951 2096 7882 4870 ...
##  $ salvando_status         : Factor w/ 5 levels "<100",">=1000",..: 5 1 1 1 1 5 4 1 2 1 ...
##  $ emprego                 : Factor w/ 5 levels "<1",">=7","1<=X<4",..: 2 3 4 4 3 3 2 3 4 5 ...
##  $ compromisso_parcelamento: num  4 2 2 2 3 2 3 2 2 4 ...
##  $ status_pessoal          : Factor w/ 4 levels "female div/dep/mar",..: 4 1 4 4 4 4 4 4 2 3 ...
##  $ outras_partes           : Factor w/ 3 levels "co applicant",..: 3 3 3 2 3 3 3 3 3 3 ...
##  $ residencia_desde        : num  4 2 3 4 4 4 4 2 4 2 ...
##  $ magnitude_propriedade   : Factor w/ 4 levels "car","life insurance",..: 4 4 4 2 3 3 2 1 4 1 ...
##  $ idade                   : num  67 22 49 45 53 35 53 35 61 28 ...
##  $ outros_planos_pagamento : Factor w/ 3 levels "bank","none",..: 2 2 2 2 2 2 2 2 2 2 ...
##  $ habitacao               : Factor w/ 3 levels "for free","own",..: 2 2 2 1 1 1 2 3 2 2 ...
##  $ credito_existente       : num  2 1 1 1 2 1 1 1 1 2 ...
##  $ trabalho                : Factor w/ 4 levels "high qualif/self emp/mgmt",..: 2 2 4 2 2 4 2 1 4 1 ...
##  $ numero_dependente       : num  1 1 2 2 2 2 1 1 1 1 ...
##  $ telefone_proprio        : num  1 0 0 0 0 1 0 1 0 0 ...
##  $ trabalhador_estrangeiro : num  0 0 0 0 0 0 0 0 0 0 ...
##  $ classe                  : Factor w/ 2 levels "bad","good": 2 1 2 2 1 2 2 2 2 1 ...
skim(dados1)
Data summary
Name dados1
Number of rows 1000
Number of columns 21
_______________________
Column type frequency:
factor 12
numeric 9
________________________
Group variables None

Variable type: factor

skim_variable n_missing complete_rate ordered n_unique top_counts
verificando_status 0 1 FALSE 4 no : 394, <0: 274, 0<=: 269, >=2: 63
historico_credito 0 1 FALSE 5 exi: 530, cri: 293, del: 88, all: 49
proposito 0 1 FALSE 10 rad: 280, new: 234, fur: 181, use: 103
salvando_status 0 1 FALSE 5 <10: 603, no : 183, 100: 103, 500: 63
emprego 0 1 FALSE 5 1<=: 339, >=7: 253, 4<=: 174, <1: 172
status_pessoal 0 1 FALSE 4 mal: 548, fem: 310, mal: 92, mal: 50
outras_partes 0 1 FALSE 3 non: 907, gua: 52, co : 41
magnitude_propriedade 0 1 FALSE 4 car: 332, rea: 282, lif: 232, no : 154
outros_planos_pagamento 0 1 FALSE 3 non: 814, ban: 139, sto: 47
habitacao 0 1 FALSE 3 own: 713, ren: 179, for: 108
trabalho 0 1 FALSE 4 ski: 630, uns: 200, hig: 148, une: 22
classe 0 1 FALSE 2 goo: 700, bad: 300

Variable type: numeric

skim_variable n_missing complete_rate mean sd p0 p25 p50 p75 p100 hist
duracao 0 1 20.90 12.06 4 12.0 18.0 24.00 72 ▇▇▂▁▁
qualidade_credito 0 1 3271.26 2822.74 250 1365.5 2319.5 3972.25 18424 ▇▂▁▁▁
compromisso_parcelamento 0 1 2.97 1.12 1 2.0 3.0 4.00 4 ▂▃▁▂▇
residencia_desde 0 1 2.85 1.10 1 2.0 3.0 4.00 4 ▂▆▁▃▇
idade 0 1 35.55 11.38 19 27.0 33.0 42.00 75 ▇▆▃▁▁
credito_existente 0 1 1.41 0.58 1 1.0 1.0 2.00 4 ▇▅▁▁▁
numero_dependente 0 1 1.16 0.36 1 1.0 1.0 1.00 2 ▇▁▁▁▂
telefone_proprio 0 1 0.40 0.49 0 0.0 0.0 1.00 1 ▇▁▁▁▆
trabalhador_estrangeiro 0 1 0.00 0.00 0 0.0 0.0 0.00 0 ▁▁▇▁▁

Utilizando K-NN Classificação

Modelo Básico

library(caret)
fit= knn3(classe ~ ., data = dados1)
fit
## 5-nearest neighbor model
## Training set outcome distribution:
## 
##  bad good 
##  300  700

Treinando o modelo com a biblioteca caret

set.seed(13)
fitt1 <- train(
  classe ~ .,
  data = dados1,
  method = 'knn'
)
fitt1
## k-Nearest Neighbors 
## 
## 1000 samples
##   20 predictor
##    2 classes: 'bad', 'good' 
## 
## No pre-processing
## Resampling: Bootstrapped (25 reps) 
## Summary of sample sizes: 1000, 1000, 1000, 1000, 1000, 1000, ... 
## Resampling results across tuning parameters:
## 
##   k  Accuracy   Kappa     
##   5  0.6207552  0.05491485
##   7  0.6314183  0.06004447
##   9  0.6401425  0.05170687
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was k = 9.
plot(fitt1)

Pré-processamento

set.seed(13)
fitp1 <- train(
  classe ~ .,
  data = dados1,
  method = 'knn',
  preProcess = c("center", "scale")
)
fitp1
## k-Nearest Neighbors 
## 
## 1000 samples
##   20 predictor
##    2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Bootstrapped (25 reps) 
## Summary of sample sizes: 1000, 1000, 1000, 1000, 1000, 1000, ... 
## Resampling results across tuning parameters:
## 
##   k  Accuracy   Kappa    
##   5  0.6829793  0.1963699
##   7  0.6854597  0.1862340
##   9  0.6947664  0.1955296
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was k = 9.
plot(fitp1)

Avaliando o modelo com holdout

set.seed(13)
in.trn <- createDataPartition(dados1$classe, p = .80, list = FALSE)
trn <- dados1[in.trn,]
tst <- dados1[-in.trn,]
set.seed(13)
fit10 <- train(
  classe ~ .,
  data = trn,
  method = 'knn',
  preProcess = c("center", "scale")
)
fit10
## k-Nearest Neighbors 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Bootstrapped (25 reps) 
## Summary of sample sizes: 800, 800, 800, 800, 800, 800, ... 
## Resampling results across tuning parameters:
## 
##   k  Accuracy   Kappa    
##   5  0.6700123  0.1709260
##   7  0.6811970  0.1765205
##   9  0.6881976  0.1785956
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was k = 9.
plot(fit10)

Removendo a variável resposta do conjunto de teste.

tst.features = subset(tst, select = -c(classe))
tst.target = subset(tst, select = classe)[,1]

predictions = predict(fit10, newdata = tst.features)

Calculamos a matriz de confusão e medidas de avaliação

confusionMatrix(predictions, tst.target)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad   21    7
##       good  39  133
##                                           
##                Accuracy : 0.77            
##                  95% CI : (0.7054, 0.8264)
##     No Information Rate : 0.7             
##     P-Value [Acc > NIR] : 0.01687         
##                                           
##                   Kappa : 0.3539          
##                                           
##  Mcnemar's Test P-Value : 4.861e-06       
##                                           
##             Sensitivity : 0.3500          
##             Specificity : 0.9500          
##          Pos Pred Value : 0.7500          
##          Neg Pred Value : 0.7733          
##              Prevalence : 0.3000          
##          Detection Rate : 0.1050          
##    Detection Prevalence : 0.1400          
##       Balanced Accuracy : 0.6500          
##                                           
##        'Positive' Class : bad             
## 

Avaliando o modelo com validação cruzada

set.seed(13)
ctrl <- trainControl(
  method = "cv",
  number = 10,
)
set.seed(13)
fit11 <- train(
  classe ~ .,
  data = trn,
  method = 'knn',
  preProcess = c("center", "scale"),
  trControl = ctrl
)
fit11
## k-Nearest Neighbors 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Cross-Validated (10 fold) 
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ... 
## Resampling results across tuning parameters:
## 
##   k  Accuracy  Kappa    
##   5  0.69125   0.1860688
##   7  0.71500   0.2359817
##   9  0.71750   0.2201487
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was k = 9.
plot(fit11)

Avaliando o Modelo no Conjunto de Teste

predictions = predict(fit11, newdata = tst.features)

Calculamos a matriz de confusão e medidas de avaliação

confusionMatrix(predictions, tst.target)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad   20    7
##       good  40  133
##                                        
##                Accuracy : 0.765        
##                  95% CI : (0.7, 0.8219)
##     No Information Rate : 0.7          
##     P-Value [Acc > NIR] : 0.02493      
##                                        
##                   Kappa : 0.3362       
##                                        
##  Mcnemar's Test P-Value : 3.046e-06    
##                                        
##             Sensitivity : 0.3333       
##             Specificity : 0.9500       
##          Pos Pred Value : 0.7407       
##          Neg Pred Value : 0.7688       
##              Prevalence : 0.3000       
##          Detection Rate : 0.1000       
##    Detection Prevalence : 0.1350       
##       Balanced Accuracy : 0.6417       
##                                        
##        'Positive' Class : bad          
## 

Tunning do número de vizinhos

set.seed(13)
tuneGrid <- expand.grid(
  k = seq(7, 13, by = 1)
)
set.seed(13)
fit12 <- train(
  classe ~ .,
  data = trn,
  method = 'knn',
  preProcess = c("center", "scale"),
  trControl = ctrl,
  tuneGrid = tuneGrid
)
fit12
## k-Nearest Neighbors 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Cross-Validated (10 fold) 
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ... 
## Resampling results across tuning parameters:
## 
##   k   Accuracy  Kappa    
##    7  0.71500   0.2359817
##    8  0.72000   0.2411513
##    9  0.71750   0.2201487
##   10  0.71500   0.2071757
##   11  0.72125   0.2185028
##   12  0.71625   0.2026276
##   13  0.71750   0.1959997
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was k = 11.
predictions = predict(fit12, newdata = tst.features)
confusionMatrix(predictions, tst.target)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad   18    6
##       good  42  134
##                                           
##                Accuracy : 0.76            
##                  95% CI : (0.6947, 0.8174)
##     No Information Rate : 0.7             
##     P-Value [Acc > NIR] : 0.03595         
##                                           
##                   Kappa : 0.3103          
##                                           
##  Mcnemar's Test P-Value : 4.376e-07       
##                                           
##             Sensitivity : 0.3000          
##             Specificity : 0.9571          
##          Pos Pred Value : 0.7500          
##          Neg Pred Value : 0.7614          
##              Prevalence : 0.3000          
##          Detection Rate : 0.0900          
##    Detection Prevalence : 0.1200          
##       Balanced Accuracy : 0.6286          
##                                           
##        'Positive' Class : bad             
## 

Utilizando o Método de Árvore Classificação

Método Básico

library(rpart)
fit_arvore = rpart(classe ~ ., data = dados1)
fit_arvore
## n= 1000 
## 
## node), split, n, loss, yval, (yprob)
##       * denotes terminal node
## 
##   1) root 1000 300 good (0.3000000 0.7000000)  
##     2) verificando_status=<0,0<=X<200 543 240 good (0.4419890 0.5580110)  
##       4) duracao>=22.5 237 103 bad (0.5654008 0.4345992)  
##         8) salvando_status=<100,100<=X<500,500<=X<1000 196  74 bad (0.6224490 0.3775510)  
##          16) duracao>=47.5 36   5 bad (0.8611111 0.1388889) *
##          17) duracao< 47.5 160  69 bad (0.5687500 0.4312500)  
##            34) proposito=business,education,furniture/equipment,new car,other,radio/tv,repairs 137  52 bad (0.6204380 0.3795620) *
##            35) proposito=used car 23   6 good (0.2608696 0.7391304) *
##         9) salvando_status=>=1000,no known savings 41  12 good (0.2926829 0.7073171) *
##       5) duracao< 22.5 306 106 good (0.3464052 0.6535948)  
##        10) historico_credito=all paid,no credits/all paid 28   7 bad (0.7500000 0.2500000) *
##        11) historico_credito=critical/other existing credit,delayed previously,existing paid 278  85 good (0.3057554 0.6942446)  
##          22) qualidade_credito>=7491.5 7   1 bad (0.8571429 0.1428571) *
##          23) qualidade_credito< 7491.5 271  79 good (0.2915129 0.7084871)  
##            46) proposito=domestic appliance,education 15   5 bad (0.6666667 0.3333333) *
##            47) proposito=business,furniture/equipment,new car,other,radio/tv,repairs,retraining,used car 256  69 good (0.2695312 0.7304688)  
##              94) duracao>=11.5 183  60 good (0.3278689 0.6721311)  
##               188) qualidade_credito< 1387.5 65  31 good (0.4769231 0.5230769)  
##                 376) magnitude_propriedade=car,no known property 20   3 bad (0.8500000 0.1500000) *
##                 377) magnitude_propriedade=life insurance,real estate 45  14 good (0.3111111 0.6888889) *
##               189) qualidade_credito>=1387.5 118  29 good (0.2457627 0.7542373) *
##              95) duracao< 11.5 73   9 good (0.1232877 0.8767123) *
##     3) verificando_status=>=200,no checking 457  60 good (0.1312910 0.8687090) *
plot(fit_arvore)
text(fit_arvore, cex = 0.50)

Treinando o modelo com a biblioteca caret

library(caret)
set.seed(13)
fit20 <- train(
  classe ~ .,
  data = dados1,
  method = 'rpart'
)
fit20
## CART 
## 
## 1000 samples
##   20 predictor
##    2 classes: 'bad', 'good' 
## 
## No pre-processing
## Resampling: Bootstrapped (25 reps) 
## Summary of sample sizes: 1000, 1000, 1000, 1000, 1000, 1000, ... 
## Resampling results across tuning parameters:
## 
##   cp          Accuracy   Kappa    
##   0.01666667  0.7189184  0.2613195
##   0.02166667  0.7160102  0.2440485
##   0.03166667  0.7133217  0.2295218
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was cp = 0.01666667.
plot(fit20)

Pré-processamento

set.seed(13)
fit21 <- train(
  classe ~ .,
  data = dados1,
  method = 'rpart',
  preProcess = c("center", "scale")
)
fit21
## CART 
## 
## 1000 samples
##   20 predictor
##    2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Bootstrapped (25 reps) 
## Summary of sample sizes: 1000, 1000, 1000, 1000, 1000, 1000, ... 
## Resampling results across tuning parameters:
## 
##   cp          Accuracy   Kappa    
##   0.01666667  0.7188142  0.2609517
##   0.02166667  0.7159060  0.2436917
##   0.03166667  0.7132175  0.2291673
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was cp = 0.01666667.
plot(fit21)

Avaliando o modelo com holdout

set.seed(13)
in.trn_arv <- createDataPartition(dados1$classe, p = .80, list = FALSE)
trn_arv<- dados1[in.trn,]
tst_arv <- dados1[-in.trn,]
set.seed(13)
fit22 <- train(
  classe ~ .,
  data = trn,
  method = 'rpart',
  preProcess = c("center", "scale")
)
fit22
## CART 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Bootstrapped (25 reps) 
## Summary of sample sizes: 800, 800, 800, 800, 800, 800, ... 
## Resampling results across tuning parameters:
## 
##   cp          Accuracy   Kappa    
##   0.02916667  0.7099487  0.1891962
##   0.03750000  0.7051130  0.1379330
##   0.04583333  0.7076683  0.1064633
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was cp = 0.02916667.
plot(fit22)

avaliando o modelo no conjunto de teste

tst.features_arv = subset(tst, select = -c(classe))
tst.target_arv = subset(tst, select = classe)[,1]

predictions_arv = predict(fit10, newdata = tst.features)
confusionMatrix(predictions_arv, tst.target_arv)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad   21    7
##       good  39  133
##                                           
##                Accuracy : 0.77            
##                  95% CI : (0.7054, 0.8264)
##     No Information Rate : 0.7             
##     P-Value [Acc > NIR] : 0.01687         
##                                           
##                   Kappa : 0.3539          
##                                           
##  Mcnemar's Test P-Value : 4.861e-06       
##                                           
##             Sensitivity : 0.3500          
##             Specificity : 0.9500          
##          Pos Pred Value : 0.7500          
##          Neg Pred Value : 0.7733          
##              Prevalence : 0.3000          
##          Detection Rate : 0.1050          
##    Detection Prevalence : 0.1400          
##       Balanced Accuracy : 0.6500          
##                                           
##        'Positive' Class : bad             
## 

Avaliando o modelo com validação cruzada

library(caret)
set.seed(13)
ctrl_arv<- trainControl(
  method = "cv",
  number= 10,
)
set.seed(13)
fit23 <- train(
  classe ~ .,
  data = trn_arv
  ,
  method = 'rpart',
  preProcess = c("center", "scale"),
  trControl = ctrl_arv
)
fit23
## CART 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Cross-Validated (10 fold) 
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ... 
## Resampling results across tuning parameters:
## 
##   cp          Accuracy  Kappa     
##   0.02916667  0.68625   0.08051324
##   0.03750000  0.69625   0.05752156
##   0.04583333  0.69625   0.05752156
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was cp = 0.04583333.
plot(fit23)

predictions_arv2= predict(fit23, newdata = tst.features_arv)
confusionMatrix(predictions_arv2, tst.target_arv)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad    0    0
##       good  60  140
##                                           
##                Accuracy : 0.7             
##                  95% CI : (0.6314, 0.7626)
##     No Information Rate : 0.7             
##     P-Value [Acc > NIR] : 0.5348          
##                                           
##                   Kappa : 0               
##                                           
##  Mcnemar's Test P-Value : 2.599e-14       
##                                           
##             Sensitivity : 0.0             
##             Specificity : 1.0             
##          Pos Pred Value : NaN             
##          Neg Pred Value : 0.7             
##              Prevalence : 0.3             
##          Detection Rate : 0.0             
##    Detection Prevalence : 0.0             
##       Balanced Accuracy : 0.5             
##                                           
##        'Positive' Class : bad             
## 

Tunning do parâmetro de complexidade

set.seed(13)
tuneGrid_arv <- expand.grid(
  cp = seq(0, 1, by = .01)
)
set.seed(13)
fit24 <- train(
  classe ~ .,
  data = trn_arv,
  method = 'rpart',
  preProcess = c("center", "scale"),
  trControl = ctrl_arv,
  tuneGrid = tuneGrid_arv
)
fit24
## CART 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Cross-Validated (10 fold) 
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ... 
## Resampling results across tuning parameters:
## 
##   cp    Accuracy  Kappa     
##   0.00  0.68500   0.21858775
##   0.01  0.69875   0.22832196
##   0.02  0.68250   0.15504360
##   0.03  0.68625   0.08051324
##   0.04  0.69625   0.05752156
##   0.05  0.69625   0.03179106
##   0.06  0.70000   0.00000000
##   0.07  0.70000   0.00000000
##   0.08  0.70000   0.00000000
##   0.09  0.70000   0.00000000
##   0.10  0.70000   0.00000000
##   0.11  0.70000   0.00000000
##   0.12  0.70000   0.00000000
##   0.13  0.70000   0.00000000
##   0.14  0.70000   0.00000000
##   0.15  0.70000   0.00000000
##   0.16  0.70000   0.00000000
##   0.17  0.70000   0.00000000
##   0.18  0.70000   0.00000000
##   0.19  0.70000   0.00000000
##   0.20  0.70000   0.00000000
##   0.21  0.70000   0.00000000
##   0.22  0.70000   0.00000000
##   0.23  0.70000   0.00000000
##   0.24  0.70000   0.00000000
##   0.25  0.70000   0.00000000
##   0.26  0.70000   0.00000000
##   0.27  0.70000   0.00000000
##   0.28  0.70000   0.00000000
##   0.29  0.70000   0.00000000
##   0.30  0.70000   0.00000000
##   0.31  0.70000   0.00000000
##   0.32  0.70000   0.00000000
##   0.33  0.70000   0.00000000
##   0.34  0.70000   0.00000000
##   0.35  0.70000   0.00000000
##   0.36  0.70000   0.00000000
##   0.37  0.70000   0.00000000
##   0.38  0.70000   0.00000000
##   0.39  0.70000   0.00000000
##   0.40  0.70000   0.00000000
##   0.41  0.70000   0.00000000
##   0.42  0.70000   0.00000000
##   0.43  0.70000   0.00000000
##   0.44  0.70000   0.00000000
##   0.45  0.70000   0.00000000
##   0.46  0.70000   0.00000000
##   0.47  0.70000   0.00000000
##   0.48  0.70000   0.00000000
##   0.49  0.70000   0.00000000
##   0.50  0.70000   0.00000000
##   0.51  0.70000   0.00000000
##   0.52  0.70000   0.00000000
##   0.53  0.70000   0.00000000
##   0.54  0.70000   0.00000000
##   0.55  0.70000   0.00000000
##   0.56  0.70000   0.00000000
##   0.57  0.70000   0.00000000
##   0.58  0.70000   0.00000000
##   0.59  0.70000   0.00000000
##   0.60  0.70000   0.00000000
##   0.61  0.70000   0.00000000
##   0.62  0.70000   0.00000000
##   0.63  0.70000   0.00000000
##   0.64  0.70000   0.00000000
##   0.65  0.70000   0.00000000
##   0.66  0.70000   0.00000000
##   0.67  0.70000   0.00000000
##   0.68  0.70000   0.00000000
##   0.69  0.70000   0.00000000
##   0.70  0.70000   0.00000000
##   0.71  0.70000   0.00000000
##   0.72  0.70000   0.00000000
##   0.73  0.70000   0.00000000
##   0.74  0.70000   0.00000000
##   0.75  0.70000   0.00000000
##   0.76  0.70000   0.00000000
##   0.77  0.70000   0.00000000
##   0.78  0.70000   0.00000000
##   0.79  0.70000   0.00000000
##   0.80  0.70000   0.00000000
##   0.81  0.70000   0.00000000
##   0.82  0.70000   0.00000000
##   0.83  0.70000   0.00000000
##   0.84  0.70000   0.00000000
##   0.85  0.70000   0.00000000
##   0.86  0.70000   0.00000000
##   0.87  0.70000   0.00000000
##   0.88  0.70000   0.00000000
##   0.89  0.70000   0.00000000
##   0.90  0.70000   0.00000000
##   0.91  0.70000   0.00000000
##   0.92  0.70000   0.00000000
##   0.93  0.70000   0.00000000
##   0.94  0.70000   0.00000000
##   0.95  0.70000   0.00000000
##   0.96  0.70000   0.00000000
##   0.97  0.70000   0.00000000
##   0.98  0.70000   0.00000000
##   0.99  0.70000   0.00000000
##   1.00  0.70000   0.00000000
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was cp = 1.
plot(fit24)

predictions_arv3 = predict(fit24, newdata = tst.features_arv)
confusionMatrix(predictions_arv3, tst.target_arv)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad    0    0
##       good  60  140
##                                           
##                Accuracy : 0.7             
##                  95% CI : (0.6314, 0.7626)
##     No Information Rate : 0.7             
##     P-Value [Acc > NIR] : 0.5348          
##                                           
##                   Kappa : 0               
##                                           
##  Mcnemar's Test P-Value : 2.599e-14       
##                                           
##             Sensitivity : 0.0             
##             Specificity : 1.0             
##          Pos Pred Value : NaN             
##          Neg Pred Value : 0.7             
##              Prevalence : 0.3             
##          Detection Rate : 0.0             
##    Detection Prevalence : 0.0             
##       Balanced Accuracy : 0.5             
##                                           
##        'Positive' Class : bad             
## 

Método Floresta Aleatória

ctrl_floresta <- trainControl(method = "repeatedcv",
                     number = 10,
                     repeats = 3,
                     search="random")
set.seed(13)
fit25<- train(
  classe ~ .,
  data = trn_arv,
  method = 'rf',
  preProcess = c("center", "scale"),
  trControl = ctrl_floresta
)
fit25
## Random Forest 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Cross-Validated (10 fold, repeated 3 times) 
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ... 
## Resampling results across tuning parameters:
## 
##   mtry  Accuracy   Kappa    
##   10    0.7433333  0.2982631
##   13    0.7470833  0.3135297
##   37    0.7387500  0.3153433
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was mtry = 13.
plot(fit25)

predictions_floresta = predict(fit25, newdata = tst.features_arv)
confusionMatrix(predictions_floresta, tst.target_arv)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad   25   13
##       good  35  127
##                                           
##                Accuracy : 0.76            
##                  95% CI : (0.6947, 0.8174)
##     No Information Rate : 0.7             
##     P-Value [Acc > NIR] : 0.035948        
##                                           
##                   Kappa : 0.3617          
##                                           
##  Mcnemar's Test P-Value : 0.002437        
##                                           
##             Sensitivity : 0.4167          
##             Specificity : 0.9071          
##          Pos Pred Value : 0.6579          
##          Neg Pred Value : 0.7840          
##              Prevalence : 0.3000          
##          Detection Rate : 0.1250          
##    Detection Prevalence : 0.1900          
##       Balanced Accuracy : 0.6619          
##                                           
##        'Positive' Class : bad             
## 

Etapa de tunning

set.seed(13)
tuneGrid_floresta <- expand.grid(
  mtry = 1:4
)
set.seed(13)
fit26 <- train(
  classe ~ .,
  data = trn_arv,
  method = 'rf',
  preProcess = c("center", "scale"),
  trControl = ctrl_floresta,
  tuneGrid = tuneGrid_floresta
)
fit26
## Random Forest 
## 
## 800 samples
##  20 predictor
##   2 classes: 'bad', 'good' 
## 
## Pre-processing: centered (48), scaled (48) 
## Resampling: Cross-Validated (10 fold, repeated 3 times) 
## Summary of sample sizes: 720, 720, 720, 720, 720, 720, ... 
## Resampling results across tuning parameters:
## 
##   mtry  Accuracy   Kappa    
##   1     0.7000000  0.0000000
##   2     0.7187500  0.1001517
##   3     0.7395833  0.2268547
##   4     0.7362500  0.2483342
## 
## Accuracy was used to select the optimal model using the largest value.
## The final value used for the model was mtry = 3.
plot(fit26)

predictions_floresta1 = predict(fit26, newdata = tst.features_arv)
confusionMatrix(predictions_floresta1, tst.target_arv)
## Confusion Matrix and Statistics
## 
##           Reference
## Prediction bad good
##       bad   17    2
##       good  43  138
##                                           
##                Accuracy : 0.775           
##                  95% CI : (0.7108, 0.8309)
##     No Information Rate : 0.7             
##     P-Value [Acc > NIR] : 0.01113         
##                                           
##                   Kappa : 0.3343          
##                                           
##  Mcnemar's Test P-Value : 2.479e-09       
##                                           
##             Sensitivity : 0.2833          
##             Specificity : 0.9857          
##          Pos Pred Value : 0.8947          
##          Neg Pred Value : 0.7624          
##              Prevalence : 0.3000          
##          Detection Rate : 0.0850          
##    Detection Prevalence : 0.0950          
##       Balanced Accuracy : 0.6345          
##                                           
##        'Positive' Class : bad             
##