Problem 1

Load in the libraries that I need for this problem

library(datasets)
library(dplyr)
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(tidyr)

The problem is working with a state data set that is found in R. The first part of the problem is working with standardizing the data for each column. I first created a function that standardized just one column and then I created another function that then standardize the entire date set by calling each column.

## Loads in the data and stores it to a variable
data(state)
us_table = state.x77
us_table = as.data.frame(us_table)
## function to standardize the data in a column
stand_data_col = function(us_data){
  i = 1
  col_mean = mean(us_data)
  col_sd = sd(us_data)
  while (i <= length(us_data)){
    us_data[i] = ((us_data[i] - col_mean) / col_sd)
    i = i + 1
  }
  return(us_data)
}

## function to standardized the entire data set
stand_data = function(data_set){
  num_col = ncol(data_set)
  i = 1
  while (i <= num_col){
    data_set[,i] = stand_data_col(data_set[,i])
    i = i + 1
  }
  return(data_set)
}

us_table_stand =stand_data(us_table)
us_table_stand
## View a summary of the data
us_table_stand %>% summary()
##    Population          Income          Illiteracy         Life Exp      
##  Min.   :-0.8694   Min.   :-2.1772   Min.   :-1.0992   Min.   :-2.1742  
##  1st Qu.:-0.7094   1st Qu.:-0.7210   1st Qu.:-0.8941   1st Qu.:-0.5670  
##  Median :-0.3154   Median : 0.1354   Median :-0.3609   Median :-0.1517  
##  Mean   : 0.0000   Mean   : 0.0000   Mean   : 0.0000   Mean   : 0.0000  
##  3rd Qu.: 0.1617   3rd Qu.: 0.6147   3rd Qu.: 0.6644   3rd Qu.: 0.7553  
##  Max.   : 3.7970   Max.   : 3.0582   Max.   : 2.6742   Max.   : 2.0273  
##      Murder           HS Grad             Frost              Area        
##  Min.   :-1.6194   Min.   :-1.89526   Min.   :-2.0096   Min.   :-0.8167  
##  1st Qu.:-0.8203   1st Qu.:-0.62622   1st Qu.:-0.7351   1st Qu.:-0.3955  
##  Median :-0.1430   Median : 0.01758   Median : 0.1931   Median :-0.1929  
##  Mean   : 0.0000   Mean   : 0.00000   Mean   : 0.0000   Mean   : 0.0000  
##  3rd Qu.: 0.8931   3rd Qu.: 0.74805   3rd Qu.: 0.6789   3rd Qu.: 0.1222  
##  Max.   : 2.0918   Max.   : 1.75709   Max.   : 1.6071   Max.   : 5.8094

The second part of the problem is asking to get ten subsets with each having 20 randomly selected rows. It took me a bit to figure out how to incorporate both apply and lapply since apply works with matrices, while lapply works with lists or vectors. I ended up being able to create a function where I first get a subset, perform the mean on those columns and then store it in a list. Once all of the data is stored in the list, I then use lapply to get the mean of the ten subsets.

## Showing the output with just apply. Just wanted to show what its doing 
## before lapply
get_mean_both = function(data_set, num_sample) {
  i = 1
  j = 1
  data_list = list()
  while (i <= num_sample) {
    m_sample = sample_n(data_set, num_sample, replace = TRUE)
    m_sample = apply(m_sample, MARGIN = 2, mean)
    m_len_num = length(m_sample)
    if (i == 1) {
      while (j <= m_len_num) {
        data_list[[j]] = m_sample[j]
        j = j + 1
      }
    } else{
      while (j <= m_len_num) {
        data_list[[j]][i] = m_sample[j]
        j = j + 1
      }
    }
    j = 1
    i = i + 1
  }
  return(data_list)
}
us_list = get_mean_both(us_table, 10)
us_list
## [[1]]
## Population                                                                   
##     4180.3     6209.0     5909.9     9531.6     4044.2     2447.1     3991.8 
##                                  
##     2624.5     3714.4     4336.9 
## 
## [[2]]
## Income                                                                
## 4473.0 4373.4 4310.7 4712.4 4206.4 4604.6 4353.2 4034.8 4320.3 4252.1 
## 
## [[3]]
## Illiteracy                                                                   
##       1.07       1.26       1.15       1.24       1.41       1.00       1.51 
##                                  
##       1.48       1.42       1.25 
## 
## [[4]]
## Life Exp                                                                
##   71.011   70.566   71.253   71.217   70.966   71.287   71.099   69.912 
##                   
##   70.646   71.136 
## 
## [[5]]
## Murder                                                                
##   7.72   9.41   6.90   8.84   7.67   6.20   7.22  10.27   7.27   5.96 
## 
## [[6]]
## HS Grad                                                                         
##   54.98   53.80   52.42   56.24   51.19   55.57   50.60   50.65   51.22   50.99 
## 
## [[7]]
## Frost                                                       
##  83.8  98.1  96.2  73.5  88.8 114.0  84.5  82.3 104.5 110.3 
## 
## [[8]]
##    Area                                                                         
## 69732.1 75339.2 77415.2 96145.6 61309.1 54777.0 51769.2 55893.3 55750.8 63214.7

Here is with the lapply in the function which would then get the mean of those columns.

get_mean_both = function(data_set, num_sample) {
  i = 1
  j = 1
  data_list = list()
  while (i <= num_sample) {
    m_sample = sample_n(data_set, num_sample, replace = TRUE)
    m_sample = apply(m_sample, MARGIN = 2, mean)
    m_len_num = length(m_sample)
    if (i == 1) {
      while (j <= m_len_num) {
        data_list[[j]] = m_sample[j]
        j = j + 1
      }
    } else{
      while (j <= m_len_num) {
        data_list[[j]][i] = m_sample[j]
        j = j + 1
      }
    }
    j = 1
    i = i + 1
  }
  names(data_list) = c("Population", "Income", "Illiteracy", "Life Exp", 
                       "Murder", "HS Grad", "Frost", "Area")
  data_list = lapply(data_list, mean)
  return(data_list)
}
us_list = get_mean_both(us_table, 10)
us_list
## $Population
## [1] 4074.95
## 
## $Income
## [1] 4504.15
## 
## $Illiteracy
## [1] 1.117
## 
## $`Life Exp`
## [1] 70.9344
## 
## $Murder
## [1] 6.824
## 
## $`HS Grad`
## [1] 54.059
## 
## $Frost
## [1] 116.33
## 
## $Area
## [1] 72392.25

Now with the lapply replace with sapply:

get_mean_both = function(data_set, num_sample) {
  i = 1
  j = 1
  data_list = list()
  while (i <= num_sample) {
    m_sample = sample_n(data_set, num_sample, replace = TRUE)
    m_sample = apply(m_sample, MARGIN = 2, mean)
    m_len_num = length(m_sample)
    if (i == 1) {
      while (j <= m_len_num) {
        data_list[[j]] = m_sample[j]
        j = j + 1
      }
    } else{
      while (j <= m_len_num) {
        data_list[[j]][i] = m_sample[j]
        j = j + 1
      }
    }
    j = 1
    i = i + 1
  }
  names(data_list) = c("Population", "Income", "Illiteracy", "Life Exp", 
                       "Murder", "HS Grad", "Frost", "Area")
  data_list = sapply(data_list, mean)
  return(data_list)
}
us_list = get_mean_both(us_table, 10)
us_list
## Population     Income Illiteracy   Life Exp     Murder    HS Grad      Frost 
##   4425.220   4396.290      1.287     70.735      8.015     52.769    102.450 
##       Area 
##  77582.640

The difference is that sapply is able to return a vector instead of a list.


Problem 2

The second problem is looking for you to use the reduce function and want you to find all the elements that appear in a least one entry. I read that as using the union, since the intersect will give you the elements that appear in every entry, instead of just one entry. Meaning, an element just have to appear once in the entry (one of the entries of the list.)

library(purrr)
my.list <- map(1:4, ~ sample(1:10, 15, replace = T))
my.list
## [[1]]
##  [1]  6  7  9 10  4  6  5  5  2  9  4  2  4  9  3
## 
## [[2]]
##  [1] 10  7  3  4  4  8  9  2  8  4 10  5  2  1  1
## 
## [[3]]
##  [1]  9  4  3 10  9  8  1 10  3  5  8  3  9  8  2
## 
## [[4]]
##  [1]  7  3  8 10  3  6  7  5  1  9  2 10  3  9 10
x = reduce(my.list, union)
cat("The elemnts that appear at least once is: ", x)
## The elemnts that appear at least once is:  6 7 9 10 4 5 2 3 8 1