La regresión logística es un método de aprendizaje automático que sirve para predecir la probabilidad de que ocurra un evento categórico, con dos resultados posibles: Sí(1) o No(2).
Una empresa de servicios por suscripción ha observado un incremento en la pérdida de clientes, fenómeno conocido como Customer Churn. Se busca un modelo para estimar la probabilidad de que un cliente abandone el servicio.
#install.packages("caret") # Modelos de aprendizaje automatico
library(caret)
## Cargando paquete requerido: ggplot2
## Cargando paquete requerido: lattice
#install.packages("tidyverse") # Manipulación de datos
library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.2.0
## ✔ forcats 1.0.1 ✔ stringr 1.6.0
## ✔ lubridate 1.9.5 ✔ tibble 3.3.1
## ✔ purrr 1.2.2 ✔ tidyr 1.3.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ✖ purrr::lift() masks caret::lift()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
#install.packages("pROC") # Cálculo del área bajo la curva
library(pROC)
## Type 'citation("pROC")' for a citation.
##
## Adjuntando el paquete: 'pROC'
##
## The following objects are masked from 'package:stats':
##
## cov, smooth, var
df <- read.csv("C:/Users/dulce/OneDrive/Escritorio/IA Empresarial/customer_churn.csv")
df <- na.omit(df)
df$CustomerID <- NULL
df$Gender <- as.factor(df$Gender)
df$Subscription.Type <- as.factor(df$Subscription.Type)
df$Contract.Length <- as.factor(df$Contract.Length)
summary(df)
## Age Gender Tenure Usage.Frequency
## Min. :18.00 Female:190580 Min. : 1.00 Min. : 1.00
## 1st Qu.:29.00 Male :250252 1st Qu.:16.00 1st Qu.: 9.00
## Median :39.00 Median :32.00 Median :16.00
## Mean :39.37 Mean :31.26 Mean :15.81
## 3rd Qu.:48.00 3rd Qu.:46.00 3rd Qu.:23.00
## Max. :65.00 Max. :60.00 Max. :30.00
## Support.Calls Payment.Delay Subscription.Type Contract.Length
## Min. : 0.000 Min. : 0.00 Basic :143026 Annual :177198
## 1st Qu.: 1.000 1st Qu.: 6.00 Premium :148678 Monthly : 87104
## Median : 3.000 Median :12.00 Standard:149128 Quarterly:176530
## Mean : 3.604 Mean :12.97
## 3rd Qu.: 6.000 3rd Qu.:19.00
## Max. :10.000 Max. :30.00
## Total.Spend Last.Interaction Churn
## Min. : 100.0 Min. : 1.00 Min. :0.0000
## 1st Qu.: 480.0 1st Qu.: 7.00 1st Qu.:0.0000
## Median : 661.0 Median :14.00 Median :1.0000
## Mean : 631.6 Mean :14.48 Mean :0.5671
## 3rd Qu.: 830.0 3rd Qu.:22.00 3rd Qu.:1.0000
## Max. :1000.0 Max. :30.00 Max. :1.0000
str(df)
## 'data.frame': 440832 obs. of 11 variables:
## $ Age : int 30 65 55 58 23 51 58 55 39 64 ...
## $ Gender : Factor w/ 2 levels "Female","Male": 1 1 1 2 2 2 1 1 2 1 ...
## $ Tenure : int 39 49 14 38 32 33 49 37 12 3 ...
## $ Usage.Frequency : int 14 1 4 21 20 25 12 8 5 25 ...
## $ Support.Calls : int 5 10 6 7 5 9 3 4 7 2 ...
## $ Payment.Delay : int 18 8 18 7 8 26 16 15 4 11 ...
## $ Subscription.Type: Factor w/ 3 levels "Basic","Premium",..: 3 1 1 3 1 2 3 2 3 3 ...
## $ Contract.Length : Factor w/ 3 levels "Annual","Monthly",..: 1 2 3 2 2 1 3 1 3 3 ...
## $ Total.Spend : num 932 557 185 396 617 129 821 445 969 415 ...
## $ Last.Interaction : int 17 6 3 29 20 8 24 30 13 29 ...
## $ Churn : int 1 1 1 1 1 1 1 1 1 1 ...
## - attr(*, "na.action")= 'omit' Named int 199296
## ..- attr(*, "names")= chr "199296"
set.seed(123)
renglones_entrenamiento <- createDataPartition(df$Churn, p=0.7, list=FALSE)
entrenamiento <- df[renglones_entrenamiento, ]
prueba <- df[-renglones_entrenamiento, ]
modelo <- glm(Churn ~ ., data = entrenamiento, family = binomial)
## Warning: glm.fit: fitted probabilities numerically 0 or 1 occurred
summary(modelo)
##
## Call:
## glm(formula = Churn ~ ., family = binomial, data = entrenamiento)
##
## Coefficients:
## Estimate Std. Error z value Pr(>|z|)
## (Intercept) -7.565e-01 4.137e-02 -18.285 < 2e-16 ***
## Age 3.536e-02 5.914e-04 59.789 < 2e-16 ***
## GenderMale -1.149e+00 1.414e-02 -81.267 < 2e-16 ***
## Tenure -7.987e-03 3.814e-04 -20.938 < 2e-16 ***
## Usage.Frequency -1.513e-02 7.672e-04 -19.724 < 2e-16 ***
## Support.Calls 7.476e-01 3.671e-03 203.665 < 2e-16 ***
## Payment.Delay 1.124e-01 9.336e-04 120.411 < 2e-16 ***
## Subscription.TypePremium -1.316e-01 1.610e-02 -8.176 2.93e-16 ***
## Subscription.TypeStandard -1.180e-01 1.612e-02 -7.320 2.49e-13 ***
## Contract.LengthMonthly 2.017e+01 3.174e+01 0.635 0.525
## Contract.LengthQuarterly 9.906e-04 1.309e-02 0.076 0.940
## Total.Spend -6.053e-03 3.663e-05 -165.246 < 2e-16 ***
## Last.Interaction 6.068e-02 8.192e-04 74.076 < 2e-16 ***
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## (Dispersion parameter for binomial family taken to be 1)
##
## Null deviance: 422253 on 308582 degrees of freedom
## Residual deviance: 150361 on 308570 degrees of freedom
## AIC: 150387
##
## Number of Fisher Scoring iterations: 18
exp(coef(modelo))
## (Intercept) Age GenderMale
## 4.692980e-01 1.035992e+00 3.169417e-01
## Tenure Usage.Frequency Support.Calls
## 9.920451e-01 9.849826e-01 2.112012e+00
## Payment.Delay Subscription.TypePremium Subscription.TypeStandard
## 1.118971e+00 8.766661e-01 8.886939e-01
## Contract.LengthMonthly Contract.LengthQuarterly Total.Spend
## 5.730510e+08 1.000991e+00 9.939650e-01
## Last.Interaction
## 1.062561e+00
# Interpretación: Por cada año de edad, la probabilidad de abandono crece 3.9%.
resultado_entrenamiento <- predict(modelo,entrenamiento)
resultado_prueba <- predict(modelo, prueba)
nuevo_cliente <- data.frame (
Age=58,
Gender="Male",
Tenure=38,
Usage.Frequency=21,
Support.Calls=7,
Payment.Delay=7,
Subscription.Type="Standard",
Contract.Length="Monthly",
Total.Spend=396,
Last.Interaction=29
)
predict(modelo, newdata=nuevo_cliente, type="response")
## 1
## 1
# Probabilidad de churn
probabilidades <- predict(modelo, entrenamiento, type = "response")
# Puntaje de riesgo: probabilidad + penalizaciones por variables críticas
entrenamiento$puntaje_riesgo <- probabilidades +
0.15*(entrenamiento$Support.Calls > 5) +
0.10*(entrenamiento$Payment.Delay > 10) +
0.10*(entrenamiento$Tenure < 12) +
0.05*(entrenamiento$Usage.Frequency < 3)
# Clasificación en categorías
entrenamiento$nivel_riesgo <- cut(entrenamiento$puntaje_riesgo,
breaks = c(-Inf, 0.3, 0.6, Inf),
labels = c("Bajo", "Medio", "Alto"))
View(entrenamiento)