Ejercicios LAB1
Ejercicio 1
library() #Comprobamos los paquetes instalados.
install.packages(c("MASS","survival")) #Instalamos los paquetes MASS y survival.
## Installing packages into '/usr/local/lib/R/site-library'
## (as 'lib' is unspecified)
??Rcmdr #Buscamos informacion sobre el paquete Rcmdr.
Ejercicio 2
iris <- read.table("iris.txt", header = TRUE) #Apartado a), cargamos el archivo iris.txt.
head(iris)
## Sepal.Length Sepal.Width Petal.Length Petal.Width Species
## 1 5.1 3.5 1.4 0.2 setosa
## 2 4.9 3.0 1.4 0.2 setosa
## 3 4.7 3.2 1.3 0.2 setosa
## 4 4.6 3.1 1.5 0.2 setosa
## 5 5.0 3.6 1.4 0.2 setosa
## 6 5.4 3.9 1.7 0.4 setosa
summary(iris) #Obtenemos un resumen estadístico de las variables de iris.txt.
## Sepal.Length Sepal.Width Petal.Length Petal.Width
## Min. :4.300 Min. :2.000 Min. :1.000 Min. :0.100
## 1st Qu.:5.100 1st Qu.:2.800 1st Qu.:1.600 1st Qu.:0.300
## Median :5.800 Median :3.000 Median :4.350 Median :1.300
## Mean :5.843 Mean :3.057 Mean :3.758 Mean :1.199
## 3rd Qu.:6.400 3rd Qu.:3.300 3rd Qu.:5.100 3rd Qu.:1.800
## Max. :7.900 Max. :4.400 Max. :6.900 Max. :2.500
## Species
## Length:150
## Class :character
## Mode :character
##
##
##
comets <- read.csv("prism_comet_het.csv", header = TRUE) #Apartado b), cargamos el archivo prism_comet_het.csv.
head(comets)
## Replicate Exp1_Control Exp1_Treated Exp2_Control Exp2_Treated Exp3_Control
## 1 1 0.2318413 0.1996681 0.1238674 0.1419469 0.9692253
## 2 2 0.3565622 0.3806375 2.5868069 0.1498389 1.2773395
## 3 3 1.0939091 4.1272821 0.6051736 0.6922106 0.1678947
## 4 4 0.2774192 0.3815947 4.9243384 0.2420912 2.3401709
## 5 5 0.3151179 1.3713237 0.1709710 0.1487540 16.5688375
## 6 6 0.9288562 0.3430704 1.1968283 0.1156836 43.6138124
## Exp3_Treated
## 1 0.1186164
## 2 1.9622727
## 3 0.6616416
## 4 2.7261976
## 5 3.6788026
## 6 6.3568654
fivenum(comets$Exp1_Control)
## [1] 0.000165759 0.532112967 0.915101114 1.928995670 56.154615340
fivenum(comets$Exp1_Treated)
## [1] 0.05410966 0.34541564 0.50702750 1.56835755 133.28313110
Ejercicio 3
library("MASS")
head(anorexia)
## Treat Prewt Postwt
## 1 Cont 80.7 80.2
## 2 Cont 89.4 80.1
## 3 Cont 91.8 86.4
## 4 Cont 74.0 86.3
## 5 Cont 78.1 76.1
## 6 Cont 88.3 78.1
table(is.na(anorexia)) #Comprobamos el numero de valores NA.
##
## FALSE
## 216
table(is.null(anorexia)) #Comprobamos si el objeto anorexia es NULL.
##
## FALSE
## 1
levels(anorexia$Treat) #Mostramos los niveles actuales de la variable Treat.
## [1] "CBT" "Cont" "FT"
anorexia_F <- factor(anorexia$Treat, levels = c("CBT", "Cont", "FT"), labels = c("Cogn Beh Tr","Contr","Fam Tr")) #Convertimos los valores de la columna Treat en variables categoricas y asignamos etiquetas.
anorexia_F
## [1] Contr Contr Contr Contr Contr Contr
## [7] Contr Contr Contr Contr Contr Contr
## [13] Contr Contr Contr Contr Contr Contr
## [19] Contr Contr Contr Contr Contr Contr
## [25] Contr Contr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [31] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [37] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [43] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [49] Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr Cogn Beh Tr
## [55] Cogn Beh Tr Fam Tr Fam Tr Fam Tr Fam Tr Fam Tr
## [61] Fam Tr Fam Tr Fam Tr Fam Tr Fam Tr Fam Tr
## [67] Fam Tr Fam Tr Fam Tr Fam Tr Fam Tr Fam Tr
## Levels: Cogn Beh Tr Contr Fam Tr
Ejercicio 4
library("MASS") #Apartado a)
data("biopsy")
head("biopsy")
## [1] "biopsy"
setwd="/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/" #Directorio de trabajo.
write.csv(biopsy, file=(paste0(setwd, "biopsy.csv")))
data("Melanoma") #Apartado b)
head("Melanoma") #Cargamos y visualizamos el data frame Melanoma.
## [1] "Melanoma"
setwd="/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/Melanoma/" #Directorio de trabajo.
write.csv(Melanoma, file=(paste0(setwd, "Melanoma.csv"))) #Lo guardamos en formato csv.
write.table(Melanoma, file=(paste0(setwd, "Melanoma.txt", quote = FALSE, sep = ","))) #Lo guardamos en formato txt (tabla).
library(openxlsx)
write.xlsx(Melanoma, file=(paste0(setwd, "Melanoma.xlsx"))) #O lo guardamos como archivo de Excel (xlsx).
Captura de pantalla con los distintos formatos
de “Melanoma”.
summary_text <- capture.output(summary(Melanoma$age)) #Apartado c), generamos un summary de la variable age en Melanoma y lo pasamos a texto con capture.output().
writeLines(summary_text, "/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/summary_age.docx") #Salvamos el texto como un archivo doc.
#Apartado d), descargamos el dataset diabetes de https://hbiostat.org/data/ en formato CSV.
setwd="/home/pablo/Documentos/UOC/Software para el análisis de datos/LAB1/"
diabetes <- read.csv(paste0(setwd, "diabetes.csv"))
head(diabetes)
## id chol stab.glu hdl ratio glyhb location age gender height weight frame
## 1 1000 203 82 56 3.6 4.31 Buckingham 46 female 62 121 medium
## 2 1001 165 97 24 6.9 4.44 Buckingham 29 female 64 218 large
## 3 1002 228 92 37 6.2 4.64 Buckingham 58 female 61 256 large
## 4 1003 78 93 12 6.5 4.63 Buckingham 67 male 67 119 large
## 5 1005 249 90 28 8.9 7.72 Buckingham 64 male 68 183 medium
## 6 1008 248 94 69 3.6 4.81 Buckingham 34 male 71 190 large
## bp.1s bp.1d bp.2s bp.2d waist hip time.ppn
## 1 118 59 NA NA 29 38 720
## 2 112 68 NA NA 46 48 360
## 3 190 92 185 92 49 57 180
## 4 110 50 NA NA 33 38 480
## 5 138 80 NA NA 44 41 300
## 6 132 86 NA NA 36 42 195
Ejercicio 5
library("MASS")
head(birthwt)
## low age lwt race smoke ptl ht ui ftv bwt
## 85 0 19 182 2 0 0 0 1 0 2523
## 86 0 33 155 3 0 0 0 0 3 2551
## 87 0 20 105 1 1 0 0 0 1 2557
## 88 0 21 108 1 1 0 0 1 2 2594
## 89 0 18 107 1 1 0 0 1 0 2600
## 91 0 21 124 3 0 0 0 0 0 2622
max(birthwt$age) #Apartado a), mostramos la edad maxima de la variable edad (age).
## [1] 45
min(birthwt$age) #Apartado b), mostramos la edad minima de la variable edad (age).
## [1] 14
range(birthwt$age) #Apartado c), comprobamos el rango de edades mostrando el máximo y mínimo de la variable edad (age).
## [1] 14 45
diferencia <- max(birthwt$age)-min(birthwt$age)
diferencia #Mostramos el rango de edades como la diferencia entre el valor máximo y mínimo.
## [1] 31
library(dplyr) #Apartado d)
bwt_min <- birthwt %>%
filter(bwt == min(bwt)) #Usando el paquete dplyr filtramos las filas que cumplen bwt = bwt minimo.
bwt_min$smoke #Imprimimos el valor del campo smoke.
## [1] 1
age_max <- birthwt %>% #Apartado e)
filter(age == max(age)) #Usando el paquete dplyr filtramos las filas que cumplen age = age maximo.
age_max$bwt #Imprimimos el valor del campo bwt.
## [1] 4990
birthwt$smoke[birthwt$bwt==min(birthwt$bwt)] #De igual manera pero en una sola linea.
## [1] 1
birthwt$bwt[birthwt$ftv < 2] #Apartado f), mostramos el campo bwt (peso al nacer) de las entradas cuyo ftv (visitas al médico) sean menores de 2.
## [1] 2523 2557 2600 2622 2637 2637 2663 2665 2722 2733 2751 2769 2769 2778 2807
## [16] 2821 2836 2863 2877 2906 2920 2920 2920 2948 2948 2977 2977 2922 3033 3062
## [31] 3062 3062 3062 3090 3090 3100 3104 3132 3175 3175 3203 3203 3203 3225 3225
## [46] 3232 3234 3260 3274 3317 3317 3331 3374 3374 3402 3416 3444 3459 3460 3473
## [61] 3544 3487 3544 3572 3572 3586 3600 3614 3614 3629 3637 3643 3651 3651 3651
## [76] 3651 3699 3728 3756 3770 3770 3770 3790 3799 3827 3884 3912 3940 3941 3941
## [91] 3969 3997 3997 4054 4054 4111 4174 4238 4593 4990 709 1135 1330 1474 1588
## [106] 1588 1701 1729 1790 1818 1885 1893 1899 1928 1936 1970 2055 2055 2084 2084
## [121] 2100 2125 2187 2187 2211 2225 2240 2240 2282 2296 2296 2325 2353 2353 2367
## [136] 2381 2381 2381 2410 2410 2410 2424 2442 2466 2466 2495 2495
Ejercicio 6
library("MASS")
data(anorexia)
matrix_anorexia <- matrix(c(anorexia$Prewt, anorexia$Postwt), ncol=2)
head(matrix_anorexia)
## [,1] [,2]
## [1,] 80.7 80.2
## [2,] 89.4 80.1
## [3,] 91.8 86.4
## [4,] 74.0 86.3
## [5,] 78.1 76.1
## [6,] 88.3 78.1
Ejercicio 7
#Introducimos el codigo del enunciado.
Identificador <- c("I1","I2","I3","I4","I5","I6","I7","I8","I9","I10","I11","I12","I13","I14", "I15","I16","I17","I18","I19","I20","I21","I22","I23","I24","I25")
Edad <- c(23,24,21,22,23,25,26,24,21,22,23,25,26,24,22,21,25,26,24,21,25,27,26,22,29)
Sexo <- c(1,2,1,1,1,2,2,2,1,2,1,2,2,2,1,1,1,2,2,2,1,2,1,1,2) #1 para mujeres y 2 para hombres
Peso <- c(76.5,81.2,79.3,59.5,67.3,78.6,67.9,100.2,97.8,56.4,65.4,67.5,87.4,99.7,87.6,93.4,65.4,73.7,85.1,61.2,54.8,103.4,65.8,71.7,85.0)
Alt <- c(165,154,178,165,164,175,182,165,178,165,158,183,184,164,189,167,182,179,165,158,183,184,189,166,175) #altura en cm
Fuma <-c("SÍ","NO","SÍ","SÍ","NO","NO","NO","SÍ","SÍ","SÍ","NO","NO","SÍ","SÍ","SÍ", "SÍ","NO","NO","SÍ","SÍ","SÍ","NO","SÍ","NO","SÍ")
Trat_Pulmon <- data.frame(Identificador,Edad,Sexo,Peso,Alt,Fuma)
Trat_Pulmon
## Identificador Edad Sexo Peso Alt Fuma
## 1 I1 23 1 76.5 165 SÍ
## 2 I2 24 2 81.2 154 NO
## 3 I3 21 1 79.3 178 SÍ
## 4 I4 22 1 59.5 165 SÍ
## 5 I5 23 1 67.3 164 NO
## 6 I6 25 2 78.6 175 NO
## 7 I7 26 2 67.9 182 NO
## 8 I8 24 2 100.2 165 SÍ
## 9 I9 21 1 97.8 178 SÍ
## 10 I10 22 2 56.4 165 SÍ
## 11 I11 23 1 65.4 158 NO
## 12 I12 25 2 67.5 183 NO
## 13 I13 26 2 87.4 184 SÍ
## 14 I14 24 2 99.7 164 SÍ
## 15 I15 22 1 87.6 189 SÍ
## 16 I16 21 1 93.4 167 SÍ
## 17 I17 25 1 65.4 182 NO
## 18 I18 26 2 73.7 179 NO
## 19 I19 24 2 85.1 165 SÍ
## 20 I20 21 2 61.2 158 SÍ
## 21 I21 25 1 54.8 183 SÍ
## 22 I22 27 2 103.4 184 NO
## 23 I23 26 1 65.8 189 SÍ
## 24 I24 22 1 71.7 166 NO
## 25 I25 29 2 85.0 175 SÍ
set1 <- subset(Trat_Pulmon, Trat_Pulmon$Edad > 22) #Apartado a)
set1
## Identificador Edad Sexo Peso Alt Fuma
## 1 I1 23 1 76.5 165 SÍ
## 2 I2 24 2 81.2 154 NO
## 5 I5 23 1 67.3 164 NO
## 6 I6 25 2 78.6 175 NO
## 7 I7 26 2 67.9 182 NO
## 8 I8 24 2 100.2 165 SÍ
## 11 I11 23 1 65.4 158 NO
## 12 I12 25 2 67.5 183 NO
## 13 I13 26 2 87.4 184 SÍ
## 14 I14 24 2 99.7 164 SÍ
## 17 I17 25 1 65.4 182 NO
## 18 I18 26 2 73.7 179 NO
## 19 I19 24 2 85.1 165 SÍ
## 21 I21 25 1 54.8 183 SÍ
## 22 I22 27 2 103.4 184 NO
## 23 I23 26 1 65.8 189 SÍ
## 25 I25 29 2 85.0 175 SÍ
set2 <- Trat_Pulmon[3, 4] #Apartado b)
set2
## [1] 79.3
set3 <- subset(Trat_Pulmon, Edad < 27, select = -c(Alt)) #Apartado c)
set3
## Identificador Edad Sexo Peso Fuma
## 1 I1 23 1 76.5 SÍ
## 2 I2 24 2 81.2 NO
## 3 I3 21 1 79.3 SÍ
## 4 I4 22 1 59.5 SÍ
## 5 I5 23 1 67.3 NO
## 6 I6 25 2 78.6 NO
## 7 I7 26 2 67.9 NO
## 8 I8 24 2 100.2 SÍ
## 9 I9 21 1 97.8 SÍ
## 10 I10 22 2 56.4 SÍ
## 11 I11 23 1 65.4 NO
## 12 I12 25 2 67.5 NO
## 13 I13 26 2 87.4 SÍ
## 14 I14 24 2 99.7 SÍ
## 15 I15 22 1 87.6 SÍ
## 16 I16 21 1 93.4 SÍ
## 17 I17 25 1 65.4 NO
## 18 I18 26 2 73.7 NO
## 19 I19 24 2 85.1 SÍ
## 20 I20 21 2 61.2 SÍ
## 21 I21 25 1 54.8 SÍ
## 23 I23 26 1 65.8 SÍ
## 24 I24 22 1 71.7 NO
Ejercicio 8
library(datasets) #Apartado a)
df <-data.frame(ChickWeight)
plot(df$weight, col=blues9, main="Gráfico 1")

boxplot(df$Time, col="red", main="Gráfico 2") #Apartado b)

Ejercicio 9
library("MASS")
data("anorexia")
dif_wt <- c(anorexia$Postwt-anorexia$Prewt) #Calculamos la diferencia de peso.
anorexia_treat_df <- data.frame(anorexia, dif_wt) #Creamos el nuevo data frame con la columna extra.
anorexia_treat_C_df <- subset(anorexia_treat_df, anorexia_treat_df$Treat == "Cont" & anorexia_treat_df$difwt > 0) #Creamos el nuevo dataframe con las dos condiciones.