Inicio de los paquetes

library(lsm)      # Para descargar una base de datos
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(moments)  # Para hallar las medidas de forma
library(e1071)
## 
## Attaching package: 'e1071'
## The following objects are masked from 'package:moments':
## 
##     kurtosis, moment, skewness
library(ggplot2)
## 
## Attaching package: 'ggplot2'
## The following object is masked from 'package:e1071':
## 
##     element

Busqueda de Base de datos

datosCompleto <- lsm::survey

Buscar datos de forma sencilla

str(datosCompleto)
## tibble [800 Ɨ 66] (S3: tbl_df/tbl/data.frame)
##  $ Observation : num [1:800] 1 2 3 4 5 6 7 8 9 10 ...
##  $ ID          : chr [1:800] "SB11201910010435" "SB11201910004475" "SB11201910011427" "SB11201910041975" ...
##  $ Gender      : chr [1:800] "Female" "Male" "Male" "Male" ...
##  $ Like        : chr [1:800] "TV" "Network" "Network" "TV" ...
##  $ Age         : num [1:800] 21.4 21.1 20.9 18.4 16.6 ...
##  $ Smoke       : chr [1:800] "No" "Yes" "Yes" "Yes" ...
##  $ Height      : num [1:800] 1.58 1.6 1.5 1.53 1.78 1.65 1.73 1.53 1.64 1.52 ...
##  $ Weight      : num [1:800] 75 80 64 49 82 80 90 55 50 78 ...
##  $ BMI         : num [1:800] 30 31.2 28.4 20.9 25.9 ...
##  $ School      : chr [1:800] "Private" "Public" "Private" "Public" ...
##  $ SES         : chr [1:800] "Medium" "High" "High" "Low" ...
##  $ Enrollment  : chr [1:800] "Credit" "Scholarship" "Scholarship" "Credit" ...
##  $ Score       : num [1:800] 81 78 77 70 68 65 54 50 36 35 ...
##  $ MotherHeight: chr [1:800] "Short_M" "Normal_M" "Normal_M" "Tall_M" ...
##  $ MotherAge   : num [1:800] 41 45 45 45 46 46 47 48 48 48 ...
##  $ MotherCHD   : num [1:800] 0 0 0 0 1 0 0 0 0 1 ...
##  $ FatherHeight: chr [1:800] "Normal_F" "Short_F" "Tall_F" "Short_F" ...
##  $ FatherAge   : num [1:800] 40 43 44 45 45 46 46 48 48 49 ...
##  $ FatherCHD   : num [1:800] 1 1 1 2 1 1 1 1 1 1 ...
##  $ Status      : chr [1:800] "Distinguished" "Distinguished" "Distinguished" "Regular" ...
##  $ SemAcum     : num [1:800] 4.25 2.8 4.15 3.2 3.45 2.75 2.7 4.35 4.3 2.8 ...
##  $ Exam1       : num [1:800] 1.5 2.3 3.4 2.5 3.1 3.8 5 4 2.5 2.4 ...
##  $ Exam2       : num [1:800] 5 4.9 3.6 4.2 3.5 4.4 3 2.3 3.3 2.6 ...
##  $ Exam3       : num [1:800] 5 3.7 2 5 5 4.2 3.5 4.6 3.8 4.3 ...
##  $ Exam4       : num [1:800] 4.5 3.3 1.9 2.5 3 5 3.6 4.3 1.9 5 ...
##  $ ExamAcum    : num [1:800] 16 14.2 10.9 14.2 14.6 17.4 15.1 15.2 11.5 14.3 ...
##  $ Definitive  : num [1:800] 4 3.55 2.73 3.55 3.65 ...
##  $ Expense     : num [1:800] 48.9 72.1 85.2 56.6 64.6 63 40.8 65.4 37.3 63 ...
##  $ Income      : num [1:800] 1.61 2.07 2.84 1.55 2.32 2.1 1.69 2.18 1.71 2.1 ...
##  $ Gas         : num [1:800] 27.4 24.2 22.3 23.1 27.3 ...
##  $ Course      : chr [1:800] "Face-to-Face" "Virtual" "Face-to-Face" "Virtual" ...
##  $ Law         : chr [1:800] "Agree" "Agree" "Agree" "Agree" ...
##  $ Economic    : chr [1:800] "Regular" "Good" "Regular" "Bad" ...
##  $ Race        : chr [1:800] "Ethnic" "Ethnic" "Ethnic" "Ethnic" ...
##  $ Region      : chr [1:800] "North" "Center" "North" "Center" ...
##  $ EMO1        : num [1:800] 1 4 3 4 2 3 2 3 4 2 ...
##  $ EMO2        : num [1:800] 2 4 1 2 1 1 4 1 2 2 ...
##  $ EMO3        : num [1:800] 2 1 3 3 2 4 2 4 3 3 ...
##  $ EMO4        : num [1:800] 1 2 3 1 4 2 3 2 1 1 ...
##  $ EMO5        : num [1:800] 4 1 2 2 2 2 1 1 2 2 ...
##  $ GOAL1       : chr [1:800] "Strongly agree" "Undecided" "Agree" "Agree" ...
##  $ GOAL2       : chr [1:800] "Agree" "Disagree" "Disagree" "Undecided" ...
##  $ GOAL3       : chr [1:800] "Strongly agree" "Disagree" "Agree" "Strongly agree" ...
##  $ Pre_STAT1   : num [1:800] 2 1 5 4 1 4 4 2 2 2 ...
##  $ Pre_STAT2   : num [1:800] 4 1 1 3 4 1 2 3 3 5 ...
##  $ Pre_STAT3   : num [1:800] 2 1 3 1 1 5 4 3 3 2 ...
##  $ Pre_STAT4   : num [1:800] 5 1 1 2 2 3 2 3 2 4 ...
##  $ Post_STAT1  : num [1:800] 4 5 5 3 5 2 3 3 2 5 ...
##  $ Post_STAT2  : num [1:800] 5 1 2 2 3 3 2 3 2 3 ...
##  $ Post_STAT3  : num [1:800] 2 3 3 4 3 5 5 4 5 4 ...
##  $ Post_STAT4  : num [1:800] 2 3 3 5 4 4 3 5 5 1 ...
##  $ Pre_IDARE1  : chr [1:800] "Quite a bit" "Quite a bit" "Quite a bit" "Little" ...
##  $ Pre_IDARE2  : chr [1:800] "Little" "Little" "Little" "Nothing" ...
##  $ Pre_IDARE3  : chr [1:800] "Quite a bit" "A lot" "Quite a bit" "Quite a bit" ...
##  $ Pre_IDARE4  : chr [1:800] "Quite a bit" "Nothing" "Quite a bit" "Quite a bit" ...
##  $ Pre_IDARE5  : chr [1:800] "Little" "Quite a bit" "Little" "Nothing" ...
##  $ Post_IDARE1 : chr [1:800] "A lot" "A little" "Nothing" "Quite a bit" ...
##  $ Post_IDARE2 : chr [1:800] "A lot" "Nothing" "Quite a bit" "A little" ...
##  $ Post_IDARE3 : chr [1:800] "A little" "Quite a bit" "Nothing" "A lot" ...
##  $ Post_IDARE4 : chr [1:800] "Quite a bit" "A lot" "Nothing" "Quite a bit" ...
##  $ Post_IDARE5 : chr [1:800] "A lot" "Quite a bit" "Nothing" "A lot" ...
##  $ PSICO1      : chr [1:800] "Frequently" "Frequently" "Sometimes" "Almost always" ...
##  $ PSICO2      : chr [1:800] "Almost always" "Sometimes" "Sometimes" "Frequently" ...
##  $ PSICO3      : chr [1:800] "Frequently" "Sometimes" "Sometimes" "Frequently" ...
##  $ PSICO4      : chr [1:800] "Almost always" "Frequently" "Frequently" "Almost never" ...
##  $ PSICO5      : chr [1:800] "Almost always" "Frequently" "Sometimes" "Sometimes" ...
head(datosCompleto, 6)
## # A tibble: 6 Ɨ 66
##   Observation ID       Gender Like    Age Smoke Height Weight   BMI School SES  
##         <dbl> <chr>    <chr>  <chr> <dbl> <chr>  <dbl>  <dbl> <dbl> <chr>  <chr>
## 1           1 SB11201… Female TV     21.4 No      1.58     75  30.0 Priva… Medi…
## 2           2 SB11201… Male   Netw…  21.1 Yes     1.6      80  31.2 Public High 
## 3           3 SB11201… Male   Netw…  20.9 Yes     1.5      64  28.4 Priva… High 
## 4           4 SB11201… Male   TV     18.4 Yes     1.53     49  20.9 Public Low  
## 5           5 SB11201… Female TV     16.6 Yes     1.78     82  25.9 Priva… High 
## 6           6 SB11201… Female Netw…  16.0 No      1.65     80  29.4 Public Low  
## # ℹ 55 more variables: Enrollment <chr>, Score <dbl>, MotherHeight <chr>,
## #   MotherAge <dbl>, MotherCHD <dbl>, FatherHeight <chr>, FatherAge <dbl>,
## #   FatherCHD <dbl>, Status <chr>, SemAcum <dbl>, Exam1 <dbl>, Exam2 <dbl>,
## #   Exam3 <dbl>, Exam4 <dbl>, ExamAcum <dbl>, Definitive <dbl>, Expense <dbl>,
## #   Income <dbl>, Gas <dbl>, Course <chr>, Law <chr>, Economic <chr>,
## #   Race <chr>, Region <chr>, EMO1 <dbl>, EMO2 <dbl>, EMO3 <dbl>, EMO4 <dbl>,
## #   EMO5 <dbl>, GOAL1 <chr>, GOAL2 <chr>, GOAL3 <chr>, Pre_STAT1 <dbl>, …

Funcion Corchete / Tablas

Muestra <- datosCompleto[1:120,c(3,5,6,7,8,10)]       # A) Un nuevo data frame 

#A) Definiendo y convirtiendo en factor
Sexo <- as.factor(Muestra$Gender)

#B) Calcular tabla de frecuencias
Tabla1 <- table(Sexo)
Tabla1            
## Sexo
## Female   Male 
##     57     63
Muestra <- datosCompleto[1:120,c(3,5,6,7,8,10)]       # A) Un nuevo data frame

Colegio <- as.factor(Muestra$School)

Tabla2 <- table(Colegio)
Tabla2
## Colegio
## Private  Public 
##      60      60
Muestra <- datosCompleto[1:120,c(3,5,6,7,8,10)] 

Fuma <- as.factor(Muestra$Smoke)

Tabla3 <- table(Fuma)
Tabla3
## Fuma
##  No Yes 
##  53  67

Frecuencia relativa

63/120
## [1] 0.525
57/120
## [1] 0.475
53/120
## [1] 0.4416667
67/120
## [1] 0.5583333
60/120
## [1] 0.5
60/120
## [1] 0.5

Tablas cruzadas/de contingencias

Tabla4 <- table(Sexo, Colegio)
Tabla4       
##         Colegio
## Sexo     Private Public
##   Female      31     26
##   Male        29     34
Tabla5 <- table(Sexo, Fuma)
Tabla5
##         Fuma
## Sexo     No Yes
##   Female 24  33
##   Male   29  34

Diagramas

ggplot(Muestra, aes(x = Sexo, fill = Sexo)) +
  # GeometrĆ­a principal
  geom_bar(width = 0.5, colour = "black") +
  
  # Capa de texto (etiquetas de conteo)
  geom_text(
    aes(label = after_stat(count)),
    stat = "count",
    vjust = -0.5,
    size = 5
  ) +
  
  # Escalas (colores y ejes)
  scale_fill_manual(values = c("pink", "blue")) +
  scale_y_continuous(expand = expansion(mult = c(0, 0.1))) +
  
  # Etiquetas y tĆ­tulos
  labs(
    x = "GƩnero",
    y = "Cantidad",
    title = "Diagrama de gƩnero"
  ) +
  
  # Estilo visual / Tema
  theme_bw(base_size = 12)

ggplot(Muestra, aes(x = Fuma, fill = Fuma)) +
  # GeometrĆ­a principal
  geom_bar(width = 0.5, colour = "black") +
  
  # Capa de texto (etiquetas de conteo)
  geom_text(
    aes(label = after_stat(count)),
    stat = "count",
    vjust = -0.5,
    size = 5
  ) +
  
  # Escalas (colores y ejes)
  scale_fill_manual(values = c("green", "red")) +
  scale_y_continuous(expand = expansion(mult = c(0, 0.1))) +
  
  # Etiquetas y tĆ­tulos
  labs(
    x = "Fuma",
    y = "Cantidad",
    title = "Consumidores de tabaco"
  ) +
  
  # Estilo visual / Tema
  theme_bw(base_size = 12)

ggplot(Muestra, aes(x = Colegio, fill = Colegio)) +
  geom_bar(width = 0.5, colour = "black") +
  scale_fill_manual(
    values = c("Private" = "yellow", "Public" = "grey"),
    labels = c("Private" = "Privada", "Public" = "PĆŗblica")
  ) +
  scale_x_discrete(
    labels = c("Private" = "Privada", "Public" = "PĆŗblica")
  ) +
  labs(
    x = "Tipo de escolaridad",
    y = "Cantidad",
    fill = "Colegio"
  ) +
  ggtitle("Tipo de Colegio") +
  theme_bw(base_size = 12) +
  geom_text(
    aes(label = after_stat(count)),
    stat = "count",
    vjust = -0.5,
    size = 5
  ) +
  scale_y_continuous(
    expand = expansion(mult = c(0, 0.1))
  )

Diagrama combinado

ggplot(Muestra, aes(x = Sexo, fill = Colegio)) +
  geom_bar(
    width = 0.6,
    colour = "black",
    position = position_dodge()
  ) +
  scale_fill_manual(
    values = c(
      "Private" = "yellow",
      "Public" = "grey"
    ),
    labels = c(
      "Private" = "Privada",
      "Public" = "PĆŗblica"
    )
  ) +
  scale_x_discrete(
    labels = c(
      "Female" = "Femenina",
      "Male" = "Masculino"
    )
  ) +
  labs(
    x = "Sexo",
    y = "Cantidad",
    fill = "Tipo de colegio"
  ) +
  ggtitle("Comparación entre Sexo y Tipo de Colegio") +
  theme_bw(base_size = 12) +
  geom_text(
    aes(label = after_stat(count)),
    stat = "count",
    position = position_dodge(width = 0.6),
    vjust = -0.5,
    size = 4
  ) +
  scale_y_continuous(
    expand = expansion(mult = c(0, 0.1))
  )