BASE DE DATOS

library(readr)
insurance <- read_csv("insurance.csv")
## Rows: 1338 Columns: 7
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (3): sex, smoker, region
## dbl (4): age, bmi, children, charges
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
View(insurance)

Estadisticas Descriptivas

summary(insurance)
##       age               sex            bmi           children    
##  Min.   :18.00   Length   :1338   Min.   :15.96   Min.   :0.000  
##  1st Qu.:27.00   N.unique :   2   1st Qu.:26.30   1st Qu.:0.000  
##  Median :39.00   N.blank  :   0   Median :30.40   Median :1.000  
##  Mean   :39.21   Min.nchar:   4   Mean   :30.66   Mean   :1.095  
##  3rd Qu.:51.00   Max.nchar:   6   3rd Qu.:34.69   3rd Qu.:2.000  
##  Max.   :64.00                    Max.   :53.13   Max.   :5.000  
##        smoker           region        charges     
##  Length   :1338   Length   :1338   Min.   : 1122  
##  N.unique :   2   N.unique :   4   1st Qu.: 4740  
##  N.blank  :   0   N.blank  :   0   Median : 9382  
##  Min.nchar:   2   Min.nchar:   9   Mean   :13270  
##  Max.nchar:   3   Max.nchar:   9   3rd Qu.:16640  
##                                    Max.   :63770
str(insurance)
## spc_tbl_ [1,338 × 7] (S3: spec_tbl_df/tbl_df/tbl/data.frame)
##  $ age     : num [1:1338] 19 18 28 33 32 31 46 37 37 60 ...
##  $ sex     : chr [1:1338] "female" "male" "male" "male" ...
##  $ bmi     : num [1:1338] 27.9 33.8 33 22.7 28.9 ...
##  $ children: num [1:1338] 0 1 3 0 0 0 1 3 2 0 ...
##  $ smoker  : chr [1:1338] "yes" "no" "no" "no" ...
##  $ region  : chr [1:1338] "southwest" "southeast" "southeast" "northwest" ...
##  $ charges : num [1:1338] 16885 1726 4449 21984 3867 ...
##  - attr(*, "spec")=
##   .. cols(
##   ..   age = col_double(),
##   ..   sex = col_character(),
##   ..   bmi = col_double(),
##   ..   children = col_double(),
##   ..   smoker = col_character(),
##   ..   region = col_character(),
##   ..   charges = col_double()
##   .. )
##  - attr(*, "problems")=<pointer: 0x000001d55defcdc0>
skimr::skim(insurance)
Data summary
Name insurance
Number of rows 1338
Number of columns 7
_______________________
Column type frequency:
character 3
numeric 4
________________________
Group variables None

Variable type: character

skim_variable n_missing complete_rate min max empty n_unique whitespace
sex 0 1 4 6 0 2 0
smoker 0 1 2 3 0 2 0
region 0 1 9 9 0 4 0

Variable type: numeric

skim_variable n_missing complete_rate mean sd p0 p25 p50 p75 p100 hist
age 0 1 39.21 14.05 18.00 27.00 39.00 51.00 64.00 ▇▅▅▆▆
bmi 0 1 30.66 6.10 15.96 26.30 30.40 34.69 53.13 ▂▇▇▂▁
children 0 1 1.09 1.21 0.00 0.00 1.00 2.00 5.00 ▇▂▂▁▁
charges 0 1 13270.42 12110.01 1121.87 4740.29 9382.03 16639.91 63770.43 ▇▂▁▁▁

Graficos relacionados entre variables

plot(insurance$age, insurance$children)

plot(insurance$children, insurance$bmi)

plot(insurance$bmi, insurance$age)