Ejercicios LAB1

Ejercicio 1

library() #Comprobamos los paquetes instalados.

install.packages(c("MASS","survival")) #Instalamos los paquetes MASS y survival.
## Installing packages into '/usr/local/lib/R/site-library'
## (as 'lib' is unspecified)
??Rcmdr #Buscamos informacion sobre el paquete Rcmdr.

Ejercicio 2

 iris <- read.table("iris.txt", header = TRUE) #Apartado a), cargamos el archivo iris.txt.
 head(iris) 
##   Sepal.Length Sepal.Width Petal.Length Petal.Width Species
## 1          5.1         3.5          1.4         0.2  setosa
## 2          4.9         3.0          1.4         0.2  setosa
## 3          4.7         3.2          1.3         0.2  setosa
## 4          4.6         3.1          1.5         0.2  setosa
## 5          5.0         3.6          1.4         0.2  setosa
## 6          5.4         3.9          1.7         0.4  setosa
 summary(iris) #Obtenemos un resumen estadístico de las variables de iris.txt.
##   Sepal.Length    Sepal.Width     Petal.Length    Petal.Width   
##  Min.   :4.300   Min.   :2.000   Min.   :1.000   Min.   :0.100  
##  1st Qu.:5.100   1st Qu.:2.800   1st Qu.:1.600   1st Qu.:0.300  
##  Median :5.800   Median :3.000   Median :4.350   Median :1.300  
##  Mean   :5.843   Mean   :3.057   Mean   :3.758   Mean   :1.199  
##  3rd Qu.:6.400   3rd Qu.:3.300   3rd Qu.:5.100   3rd Qu.:1.800  
##  Max.   :7.900   Max.   :4.400   Max.   :6.900   Max.   :2.500  
##    Species         
##  Length:150        
##  Class :character  
##  Mode  :character  
##                    
##                    
## 
comets <- read.csv("prism_comet_het.csv", header = TRUE) #Apartado b), cargamos el archivo prism_comet_het.csv.
head(comets)
##   Replicate Exp1_Control Exp1_Treated Exp2_Control Exp2_Treated Exp3_Control
## 1         1    0.2318413    0.1996681    0.1238674    0.1419469    0.9692253
## 2         2    0.3565622    0.3806375    2.5868069    0.1498389    1.2773395
## 3         3    1.0939091    4.1272821    0.6051736    0.6922106    0.1678947
## 4         4    0.2774192    0.3815947    4.9243384    0.2420912    2.3401709
## 5         5    0.3151179    1.3713237    0.1709710    0.1487540   16.5688375
## 6         6    0.9288562    0.3430704    1.1968283    0.1156836   43.6138124
##   Exp3_Treated
## 1    0.1186164
## 2    1.9622727
## 3    0.6616416
## 4    2.7261976
## 5    3.6788026
## 6    6.3568654
fivenum(comets$Exp1_Control)
## [1]  0.000165759  0.532112967  0.915101114  1.928995670 56.154615340
fivenum(comets$Exp1_Treated)
## [1]   0.05410966   0.34541564   0.50702750   1.56835755 133.28313110

Ejercicio 3

library("MASS") 
head(anorexia) 
##   Treat Prewt Postwt
## 1  Cont  80.7   80.2
## 2  Cont  89.4   80.1
## 3  Cont  91.8   86.4
## 4  Cont  74.0   86.3
## 5  Cont  78.1   76.1
## 6  Cont  88.3   78.1
table(is.na(anorexia)) #Comprobamos el numero de valores NA.
## 
## FALSE 
##   216
table(is.null(anorexia)) #Comprobamos si el objeto anorexia es NULL.
## 
## FALSE 
##     1
levels(anorexia$Treat) #Mostramos los niveles actuales de la variable Treat.
## [1] "CBT"  "Cont" "FT"
anorexia_F <- factor(anorexia$Treat, levels = c("CBT", "Cont", "FT"), labels = c("Cogn Beh Tr","Contr","Fam Tr")) #Convertimos los valores de la columna Treat en variables categoricas y asignamos etiquetas.
anorexia_F
##  [1] Contr       Contr       Contr       Contr       Contr       Contr      
##  [7] Contr       Contr       Contr       Contr       Contr       Contr      
## [13] Contr       Contr       Contr       Contr       Contr       Contr      
## [19] Contr       Contr       Contr       Contr       Contr       Contr      
## [25] Contr       Contr       Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [31] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [37] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [43] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [49] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [55] Cogn Beh Tr Fam Tr      Fam Tr      Fam Tr      Fam Tr      Fam Tr     
## [61] Fam Tr      Fam Tr      Fam Tr      Fam Tr      Fam Tr      Fam Tr     
## [67] Fam Tr      Fam Tr      Fam Tr      Fam Tr      Fam Tr      Fam Tr     
## Levels: Cogn Beh Tr Contr Fam Tr

Ejercicio 4

library("MASS") #Apartado a)
data("biopsy")
head("biopsy")
## [1] "biopsy"
setwd="/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/" #Directorio de trabajo.
write.csv(biopsy, file=(paste0(setwd, "biopsy.csv"))) 

data("Melanoma") #Apartado b)
head("Melanoma") #Cargamos y visualizamos el data frame Melanoma.
## [1] "Melanoma"
setwd="/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/Melanoma/" #Directorio de trabajo.
write.csv(Melanoma, file=(paste0(setwd, "Melanoma.csv"))) #Lo guardamos en formato csv.
write.table(Melanoma, file=(paste0(setwd, "Melanoma.txt", quote = FALSE, sep = ","))) #Lo guardamos en formato txt (tabla).
library(openxlsx)
write.xlsx(Melanoma, file=(paste0(setwd, "Melanoma.xlsx"))) #O lo guardamos como archivo de Excel (xlsx).
Captura de pantalla con los distintos formatos de “Melanoma”.
Captura de pantalla con los distintos formatos de “Melanoma”.
summary_text <- capture.output(summary(Melanoma$age)) #Apartado c), generamos un summary de la variable age en Melanoma y lo pasamos a texto con capture.output().
writeLines(summary_text, "/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/summary_age.docx") #Salvamos el texto como un archivo doc.

#Apartado d), descargamos el dataset diabetes de https://hbiostat.org/data/ en formato CSV.
setwd="/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/"
diabetes <- read.csv(paste0(setwd, "diabetes.csv"))
head(diabetes)
##     id chol stab.glu hdl ratio glyhb   location age gender height weight  frame
## 1 1000  203       82  56   3.6  4.31 Buckingham  46 female     62    121 medium
## 2 1001  165       97  24   6.9  4.44 Buckingham  29 female     64    218  large
## 3 1002  228       92  37   6.2  4.64 Buckingham  58 female     61    256  large
## 4 1003   78       93  12   6.5  4.63 Buckingham  67   male     67    119  large
## 5 1005  249       90  28   8.9  7.72 Buckingham  64   male     68    183 medium
## 6 1008  248       94  69   3.6  4.81 Buckingham  34   male     71    190  large
##   bp.1s bp.1d bp.2s bp.2d waist hip time.ppn
## 1   118    59    NA    NA    29  38      720
## 2   112    68    NA    NA    46  48      360
## 3   190    92   185    92    49  57      180
## 4   110    50    NA    NA    33  38      480
## 5   138    80    NA    NA    44  41      300
## 6   132    86    NA    NA    36  42      195

Ejercicio 5

library("MASS")
head(birthwt)
##    low age lwt race smoke ptl ht ui ftv  bwt
## 85   0  19 182    2     0   0  0  1   0 2523
## 86   0  33 155    3     0   0  0  0   3 2551
## 87   0  20 105    1     1   0  0  0   1 2557
## 88   0  21 108    1     1   0  0  1   2 2594
## 89   0  18 107    1     1   0  0  1   0 2600
## 91   0  21 124    3     0   0  0  0   0 2622
max(birthwt$age) #Apartado a), mostramos la edad maxima de la variable edad (age).
## [1] 45
min(birthwt$age) #Apartado b), mostramos la edad minima de la variable edad (age).
## [1] 14
range(birthwt$age) #Apartado c), comprobamos el rango de edades mostrando el máximo y mínimo de la variable edad (age).
## [1] 14 45
diferencia <- max(birthwt$age)-min(birthwt$age)
diferencia #Mostramos el rango de edades como la diferencia entre el valor máximo y mínimo.
## [1] 31
library(dplyr) #Apartado d)
bwt_min <- birthwt %>%
  filter(bwt == min(bwt)) #Usando el paquete dplyr filtramos las filas que cumplen bwt = bwt minimo.
bwt_min$smoke #Imprimimos el valor del campo smoke.
## [1] 1
age_max <- birthwt %>% #Apartado e)
  filter(age == max(age)) #Usando el paquete dplyr filtramos las filas que cumplen age = age maximo.
age_max$bwt #Imprimimos el valor del campo bwt.
## [1] 4990
birthwt$smoke[birthwt$bwt==min(birthwt$bwt)] #De igual manera pero en una sola linea.
## [1] 1
birthwt$bwt[birthwt$ftv < 2] #Apartado f), mostramos el campo bwt (peso al nacer) de las entradas cuyo ftv (visitas al médico) sean menores de 2.
##   [1] 2523 2557 2600 2622 2637 2637 2663 2665 2722 2733 2751 2769 2769 2778 2807
##  [16] 2821 2836 2863 2877 2906 2920 2920 2920 2948 2948 2977 2977 2922 3033 3062
##  [31] 3062 3062 3062 3090 3090 3100 3104 3132 3175 3175 3203 3203 3203 3225 3225
##  [46] 3232 3234 3260 3274 3317 3317 3331 3374 3374 3402 3416 3444 3459 3460 3473
##  [61] 3544 3487 3544 3572 3572 3586 3600 3614 3614 3629 3637 3643 3651 3651 3651
##  [76] 3651 3699 3728 3756 3770 3770 3770 3790 3799 3827 3884 3912 3940 3941 3941
##  [91] 3969 3997 3997 4054 4054 4111 4174 4238 4593 4990  709 1135 1330 1474 1588
## [106] 1588 1701 1729 1790 1818 1885 1893 1899 1928 1936 1970 2055 2055 2084 2084
## [121] 2100 2125 2187 2187 2211 2225 2240 2240 2282 2296 2296 2325 2353 2353 2367
## [136] 2381 2381 2381 2410 2410 2410 2424 2442 2466 2466 2495 2495

Ejercicio 6

library("MASS")
data(anorexia)
matrix_anorexia <- matrix(c(anorexia$Prewt, anorexia$Postwt), ncol=2)
head(matrix_anorexia) 
##      [,1] [,2]
## [1,] 80.7 80.2
## [2,] 89.4 80.1
## [3,] 91.8 86.4
## [4,] 74.0 86.3
## [5,] 78.1 76.1
## [6,] 88.3 78.1

Ejercicio 7

#Introducimos el codigo del enunciado.
Identificador <- c("I1","I2","I3","I4","I5","I6","I7","I8","I9","I10","I11","I12","I13","I14", "I15","I16","I17","I18","I19","I20","I21","I22","I23","I24","I25")
Edad <- c(23,24,21,22,23,25,26,24,21,22,23,25,26,24,22,21,25,26,24,21,25,27,26,22,29)
Sexo <- c(1,2,1,1,1,2,2,2,1,2,1,2,2,2,1,1,1,2,2,2,1,2,1,1,2) #1 para mujeres y 2 para hombres
Peso <- c(76.5,81.2,79.3,59.5,67.3,78.6,67.9,100.2,97.8,56.4,65.4,67.5,87.4,99.7,87.6,93.4,65.4,73.7,85.1,61.2,54.8,103.4,65.8,71.7,85.0)
Alt <- c(165,154,178,165,164,175,182,165,178,165,158,183,184,164,189,167,182,179,165,158,183,184,189,166,175) #altura en cm
Fuma <-c("SÍ","NO","SÍ","SÍ","NO","NO","NO","SÍ","SÍ","SÍ","NO","NO","SÍ","SÍ","SÍ", "SÍ","NO","NO","SÍ","SÍ","SÍ","NO","SÍ","NO","SÍ")
Trat_Pulmon <- data.frame(Identificador,Edad,Sexo,Peso,Alt,Fuma)
Trat_Pulmon
##    Identificador Edad Sexo  Peso Alt Fuma
## 1             I1   23    1  76.5 165   SÍ
## 2             I2   24    2  81.2 154   NO
## 3             I3   21    1  79.3 178   SÍ
## 4             I4   22    1  59.5 165   SÍ
## 5             I5   23    1  67.3 164   NO
## 6             I6   25    2  78.6 175   NO
## 7             I7   26    2  67.9 182   NO
## 8             I8   24    2 100.2 165   SÍ
## 9             I9   21    1  97.8 178   SÍ
## 10           I10   22    2  56.4 165   SÍ
## 11           I11   23    1  65.4 158   NO
## 12           I12   25    2  67.5 183   NO
## 13           I13   26    2  87.4 184   SÍ
## 14           I14   24    2  99.7 164   SÍ
## 15           I15   22    1  87.6 189   SÍ
## 16           I16   21    1  93.4 167   SÍ
## 17           I17   25    1  65.4 182   NO
## 18           I18   26    2  73.7 179   NO
## 19           I19   24    2  85.1 165   SÍ
## 20           I20   21    2  61.2 158   SÍ
## 21           I21   25    1  54.8 183   SÍ
## 22           I22   27    2 103.4 184   NO
## 23           I23   26    1  65.8 189   SÍ
## 24           I24   22    1  71.7 166   NO
## 25           I25   29    2  85.0 175   SÍ
set1 <- subset(Trat_Pulmon, Trat_Pulmon$Edad > 22)  #Apartado a)
set1
##    Identificador Edad Sexo  Peso Alt Fuma
## 1             I1   23    1  76.5 165   SÍ
## 2             I2   24    2  81.2 154   NO
## 5             I5   23    1  67.3 164   NO
## 6             I6   25    2  78.6 175   NO
## 7             I7   26    2  67.9 182   NO
## 8             I8   24    2 100.2 165   SÍ
## 11           I11   23    1  65.4 158   NO
## 12           I12   25    2  67.5 183   NO
## 13           I13   26    2  87.4 184   SÍ
## 14           I14   24    2  99.7 164   SÍ
## 17           I17   25    1  65.4 182   NO
## 18           I18   26    2  73.7 179   NO
## 19           I19   24    2  85.1 165   SÍ
## 21           I21   25    1  54.8 183   SÍ
## 22           I22   27    2 103.4 184   NO
## 23           I23   26    1  65.8 189   SÍ
## 25           I25   29    2  85.0 175   SÍ
set2 <- Trat_Pulmon[3, 4] #Apartado b)
set2
## [1] 79.3
set3 <- subset(Trat_Pulmon, Edad < 27, select = -c(Alt)) #Apartado c)
set3
##    Identificador Edad Sexo  Peso Fuma
## 1             I1   23    1  76.5   SÍ
## 2             I2   24    2  81.2   NO
## 3             I3   21    1  79.3   SÍ
## 4             I4   22    1  59.5   SÍ
## 5             I5   23    1  67.3   NO
## 6             I6   25    2  78.6   NO
## 7             I7   26    2  67.9   NO
## 8             I8   24    2 100.2   SÍ
## 9             I9   21    1  97.8   SÍ
## 10           I10   22    2  56.4   SÍ
## 11           I11   23    1  65.4   NO
## 12           I12   25    2  67.5   NO
## 13           I13   26    2  87.4   SÍ
## 14           I14   24    2  99.7   SÍ
## 15           I15   22    1  87.6   SÍ
## 16           I16   21    1  93.4   SÍ
## 17           I17   25    1  65.4   NO
## 18           I18   26    2  73.7   NO
## 19           I19   24    2  85.1   SÍ
## 20           I20   21    2  61.2   SÍ
## 21           I21   25    1  54.8   SÍ
## 23           I23   26    1  65.8   SÍ
## 24           I24   22    1  71.7   NO

Ejercicio 8

library(datasets) #Apartado a)
df <-data.frame(ChickWeight)
plot(df$weight, col=blues9, main="Gráfico 1")

boxplot(df$Time, col="red", main="Gráfico 2") #Apartado b)

Ejercicio 9

library("MASS")
data("anorexia")
dif_wt <- c(anorexia$Postwt-anorexia$Prewt) #Calculamos la diferencia de peso.
anorexia_treat_df <- data.frame(anorexia, dif_wt) #Creamos el nuevo data frame con la columna extra.

anorexia_treat_C_df <- subset(anorexia_treat_df, anorexia_treat_df$Treat == "Cont" & anorexia_treat_df$difwt > 0) #Creamos el nuevo dataframe con las dos condiciones.