Load in the libraries that I need for this problem
library(datasets)
library(dplyr)
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(tidyr)
The problem is working with a state data set that is found in R. The first part of the problem is working with standardizing the data for each column. I first created a function that standardized just one column and then I created another function that then standardize the entire date set by calling each column.
## Loads in the data and stores it to a variable
data(state)
us_table = state.x77
us_table = as.data.frame(us_table)
## function to standardize the data in a column
stand_data_col = function(us_data){
i = 1
col_mean = mean(us_data)
col_sd = sd(us_data)
while (i <= length(us_data)){
us_data[i] = ((us_data[i] - col_mean) / col_sd)
i = i + 1
}
return(us_data)
}
## function to standardized the entire data set
stand_data = function(data_set){
num_col = ncol(data_set)
i = 1
while (i <= num_col){
data_set[,i] = stand_data_col(data_set[,i])
i = i + 1
}
return(data_set)
}
us_table_stand =stand_data(us_table)
us_table_stand
## View a summary of the data
us_table_stand %>% summary()
## Population Income Illiteracy Life Exp
## Min. :-0.8694 Min. :-2.1772 Min. :-1.0992 Min. :-2.1742
## 1st Qu.:-0.7094 1st Qu.:-0.7210 1st Qu.:-0.8941 1st Qu.:-0.5670
## Median :-0.3154 Median : 0.1354 Median :-0.3609 Median :-0.1517
## Mean : 0.0000 Mean : 0.0000 Mean : 0.0000 Mean : 0.0000
## 3rd Qu.: 0.1617 3rd Qu.: 0.6147 3rd Qu.: 0.6644 3rd Qu.: 0.7553
## Max. : 3.7970 Max. : 3.0582 Max. : 2.6742 Max. : 2.0273
## Murder HS Grad Frost Area
## Min. :-1.6194 Min. :-1.89526 Min. :-2.0096 Min. :-0.8167
## 1st Qu.:-0.8203 1st Qu.:-0.62622 1st Qu.:-0.7351 1st Qu.:-0.3955
## Median :-0.1430 Median : 0.01758 Median : 0.1931 Median :-0.1929
## Mean : 0.0000 Mean : 0.00000 Mean : 0.0000 Mean : 0.0000
## 3rd Qu.: 0.8931 3rd Qu.: 0.74805 3rd Qu.: 0.6789 3rd Qu.: 0.1222
## Max. : 2.0918 Max. : 1.75709 Max. : 1.6071 Max. : 5.8094
The second part of the problem is asking to get ten subsets with each having 20 randomly selected rows. It took me a bit to figure out how to incorporate both apply and lapply since apply works with matrices, while lapply works with lists or vectors. I ended up being able to create a function where I first get a subset, perform the mean on those columns and then store it in a list. Once all of the data is stored in the list, I then use lapply to get the mean of the ten subsets.
## Showing the output with just apply. Just wanted to show what its doing
## before lapply
get_mean_both = function(data_set, num_sample) {
i = 1
j = 1
data_list = list()
while (i <= num_sample) {
m_sample = sample_n(data_set, num_sample, replace = TRUE)
m_sample = apply(m_sample, MARGIN = 2, mean)
m_len_num = length(m_sample)
if (i == 1) {
while (j <= m_len_num) {
data_list[[j]] = m_sample[j]
j = j + 1
}
} else{
while (j <= m_len_num) {
data_list[[j]][i] = m_sample[j]
j = j + 1
}
}
j = 1
i = i + 1
}
return(data_list)
}
us_list = get_mean_both(us_table, 10)
us_list
## [[1]]
## Population
## 4180.3 6209.0 5909.9 9531.6 4044.2 2447.1 3991.8
##
## 2624.5 3714.4 4336.9
##
## [[2]]
## Income
## 4473.0 4373.4 4310.7 4712.4 4206.4 4604.6 4353.2 4034.8 4320.3 4252.1
##
## [[3]]
## Illiteracy
## 1.07 1.26 1.15 1.24 1.41 1.00 1.51
##
## 1.48 1.42 1.25
##
## [[4]]
## Life Exp
## 71.011 70.566 71.253 71.217 70.966 71.287 71.099 69.912
##
## 70.646 71.136
##
## [[5]]
## Murder
## 7.72 9.41 6.90 8.84 7.67 6.20 7.22 10.27 7.27 5.96
##
## [[6]]
## HS Grad
## 54.98 53.80 52.42 56.24 51.19 55.57 50.60 50.65 51.22 50.99
##
## [[7]]
## Frost
## 83.8 98.1 96.2 73.5 88.8 114.0 84.5 82.3 104.5 110.3
##
## [[8]]
## Area
## 69732.1 75339.2 77415.2 96145.6 61309.1 54777.0 51769.2 55893.3 55750.8 63214.7
Here is with the lapply in the function which would then get the mean of those columns.
get_mean_both = function(data_set, num_sample) {
i = 1
j = 1
data_list = list()
while (i <= num_sample) {
m_sample = sample_n(data_set, num_sample, replace = TRUE)
m_sample = apply(m_sample, MARGIN = 2, mean)
m_len_num = length(m_sample)
if (i == 1) {
while (j <= m_len_num) {
data_list[[j]] = m_sample[j]
j = j + 1
}
} else{
while (j <= m_len_num) {
data_list[[j]][i] = m_sample[j]
j = j + 1
}
}
j = 1
i = i + 1
}
names(data_list) = c("Population", "Income", "Illiteracy", "Life Exp",
"Murder", "HS Grad", "Frost", "Area")
data_list = lapply(data_list, mean)
return(data_list)
}
us_list = get_mean_both(us_table, 10)
us_list
## $Population
## [1] 4074.95
##
## $Income
## [1] 4504.15
##
## $Illiteracy
## [1] 1.117
##
## $`Life Exp`
## [1] 70.9344
##
## $Murder
## [1] 6.824
##
## $`HS Grad`
## [1] 54.059
##
## $Frost
## [1] 116.33
##
## $Area
## [1] 72392.25
Now with the lapply replace with sapply:
get_mean_both = function(data_set, num_sample) {
i = 1
j = 1
data_list = list()
while (i <= num_sample) {
m_sample = sample_n(data_set, num_sample, replace = TRUE)
m_sample = apply(m_sample, MARGIN = 2, mean)
m_len_num = length(m_sample)
if (i == 1) {
while (j <= m_len_num) {
data_list[[j]] = m_sample[j]
j = j + 1
}
} else{
while (j <= m_len_num) {
data_list[[j]][i] = m_sample[j]
j = j + 1
}
}
j = 1
i = i + 1
}
names(data_list) = c("Population", "Income", "Illiteracy", "Life Exp",
"Murder", "HS Grad", "Frost", "Area")
data_list = sapply(data_list, mean)
return(data_list)
}
us_list = get_mean_both(us_table, 10)
us_list
## Population Income Illiteracy Life Exp Murder HS Grad Frost
## 4425.220 4396.290 1.287 70.735 8.015 52.769 102.450
## Area
## 77582.640
The difference is that sapply is able to return a vector instead of a list.
The second problem is looking for you to use the reduce function and want you to find all the elements that appear in a least one entry. I read that as using the union, since the intersect will give you the elements that appear in every entry, instead of just one entry. Meaning, an element just have to appear once in the entry (one of the entries of the list.)
library(purrr)
my.list <- map(1:4, ~ sample(1:10, 15, replace = T))
my.list
## [[1]]
## [1] 6 7 9 10 4 6 5 5 2 9 4 2 4 9 3
##
## [[2]]
## [1] 10 7 3 4 4 8 9 2 8 4 10 5 2 1 1
##
## [[3]]
## [1] 9 4 3 10 9 8 1 10 3 5 8 3 9 8 2
##
## [[4]]
## [1] 7 3 8 10 3 6 7 5 1 9 2 10 3 9 10
x = reduce(my.list, union)
cat("The elemnts that appear at least once is: ", x)
## The elemnts that appear at least once is: 6 7 9 10 4 5 2 3 8 1