#####################################
## Chapter 4 - Dimension Reduction ##
#####################################
#NOTE: Prepared with R version 4.3.2
#q1: set the working directory to appropriate folder on your machine, so as to access the data
#files.
setwd("C:/Users/Kamcc/OneDrive/Desktop/LIS5802/Data/DMBA-R-datasets")
#q2: load the packages you plan to use here
read.csv("Cereals.csv")
## name mfr type calories protein fat sodium
## 1 100%_Bran N C 70 4 1 130
## 2 100%_Natural_Bran Q C 120 3 5 15
## 3 All-Bran K C 70 4 1 260
## 4 All-Bran_with_Extra_Fiber K C 50 4 0 140
## 5 Almond_Delight R C 110 2 2 200
## 6 Apple_Cinnamon_Cheerios G C 110 2 2 180
## 7 Apple_Jacks K C 110 2 0 125
## 8 Basic_4 G C 130 3 2 210
## 9 Bran_Chex R C 90 2 1 200
## 10 Bran_Flakes P C 90 3 0 210
## 11 Cap'n'Crunch Q C 120 1 2 220
## 12 Cheerios G C 110 6 2 290
## 13 Cinnamon_Toast_Crunch G C 120 1 3 210
## 14 Clusters G C 110 3 2 140
## 15 Cocoa_Puffs G C 110 1 1 180
## 16 Corn_Chex R C 110 2 0 280
## 17 Corn_Flakes K C 100 2 0 290
## 18 Corn_Pops K C 110 1 0 90
## 19 Count_Chocula G C 110 1 1 180
## 20 Cracklin'_Oat_Bran K C 110 3 3 140
## 21 Cream_of_Wheat_(Quick) N H 100 3 0 80
## 22 Crispix K C 110 2 0 220
## 23 Crispy_Wheat_&_Raisins G C 100 2 1 140
## 24 Double_Chex R C 100 2 0 190
## 25 Froot_Loops K C 110 2 1 125
## 26 Frosted_Flakes K C 110 1 0 200
## 27 Frosted_Mini-Wheats K C 100 3 0 0
## 28 Fruit_&_Fibre_Dates,_Walnuts,_and_Oats P C 120 3 2 160
## 29 Fruitful_Bran K C 120 3 0 240
## 30 Fruity_Pebbles P C 110 1 1 135
## 31 Golden_Crisp P C 100 2 0 45
## 32 Golden_Grahams G C 110 1 1 280
## 33 Grape_Nuts_Flakes P C 100 3 1 140
## 34 Grape-Nuts P C 110 3 0 170
## 35 Great_Grains_Pecan P C 120 3 3 75
## 36 Honey_Graham_Ohs Q C 120 1 2 220
## 37 Honey_Nut_Cheerios G C 110 3 1 250
## 38 Honey-comb P C 110 1 0 180
## 39 Just_Right_Crunchy__Nuggets K C 110 2 1 170
## 40 Just_Right_Fruit_&_Nut K C 140 3 1 170
## 41 Kix G C 110 2 1 260
## 42 Life Q C 100 4 2 150
## 43 Lucky_Charms G C 110 2 1 180
## 44 Maypo A H 100 4 1 0
## 45 Muesli_Raisins,_Dates,_&_Almonds R C 150 4 3 95
## 46 Muesli_Raisins,_Peaches,_&_Pecans R C 150 4 3 150
## 47 Mueslix_Crispy_Blend K C 160 3 2 150
## 48 Multi-Grain_Cheerios G C 100 2 1 220
## 49 Nut&Honey_Crunch K C 120 2 1 190
## 50 Nutri-Grain_Almond-Raisin K C 140 3 2 220
## 51 Nutri-grain_Wheat K C 90 3 0 170
## 52 Oatmeal_Raisin_Crisp G C 130 3 2 170
## 53 Post_Nat._Raisin_Bran P C 120 3 1 200
## 54 Product_19 K C 100 3 0 320
## 55 Puffed_Rice Q C 50 1 0 0
## 56 Puffed_Wheat Q C 50 2 0 0
## 57 Quaker_Oat_Squares Q C 100 4 1 135
## 58 Quaker_Oatmeal Q H 100 5 2 0
## 59 Raisin_Bran K C 120 3 1 210
## 60 Raisin_Nut_Bran G C 100 3 2 140
## 61 Raisin_Squares K C 90 2 0 0
## 62 Rice_Chex R C 110 1 0 240
## 63 Rice_Krispies K C 110 2 0 290
## 64 Shredded_Wheat N C 80 2 0 0
## 65 Shredded_Wheat_'n'Bran N C 90 3 0 0
## 66 Shredded_Wheat_spoon_size N C 90 3 0 0
## 67 Smacks K C 110 2 1 70
## 68 Special_K K C 110 6 0 230
## 69 Strawberry_Fruit_Wheats N C 90 2 0 15
## 70 Total_Corn_Flakes G C 110 2 1 200
## 71 Total_Raisin_Bran G C 140 3 1 190
## 72 Total_Whole_Grain G C 100 3 1 200
## 73 Triples G C 110 2 1 250
## 74 Trix G C 110 1 1 140
## 75 Wheat_Chex R C 100 3 1 230
## 76 Wheaties G C 100 3 1 200
## 77 Wheaties_Honey_Gold G C 110 2 1 200
## fiber carbo sugars potass vitamins shelf weight cups rating
## 1 10.0 5.0 6 280 25 3 1.00 0.33 68.40297
## 2 2.0 8.0 8 135 0 3 1.00 1.00 33.98368
## 3 9.0 7.0 5 320 25 3 1.00 0.33 59.42551
## 4 14.0 8.0 0 330 25 3 1.00 0.50 93.70491
## 5 1.0 14.0 8 NA 25 3 1.00 0.75 34.38484
## 6 1.5 10.5 10 70 25 1 1.00 0.75 29.50954
## 7 1.0 11.0 14 30 25 2 1.00 1.00 33.17409
## 8 2.0 18.0 8 100 25 3 1.33 0.75 37.03856
## 9 4.0 15.0 6 125 25 1 1.00 0.67 49.12025
## 10 5.0 13.0 5 190 25 3 1.00 0.67 53.31381
## 11 0.0 12.0 12 35 25 2 1.00 0.75 18.04285
## 12 2.0 17.0 1 105 25 1 1.00 1.25 50.76500
## 13 0.0 13.0 9 45 25 2 1.00 0.75 19.82357
## 14 2.0 13.0 7 105 25 3 1.00 0.50 40.40021
## 15 0.0 12.0 13 55 25 2 1.00 1.00 22.73645
## 16 0.0 22.0 3 25 25 1 1.00 1.00 41.44502
## 17 1.0 21.0 2 35 25 1 1.00 1.00 45.86332
## 18 1.0 13.0 12 20 25 2 1.00 1.00 35.78279
## 19 0.0 12.0 13 65 25 2 1.00 1.00 22.39651
## 20 4.0 10.0 7 160 25 3 1.00 0.50 40.44877
## 21 1.0 21.0 0 NA 0 2 1.00 1.00 64.53382
## 22 1.0 21.0 3 30 25 3 1.00 1.00 46.89564
## 23 2.0 11.0 10 120 25 3 1.00 0.75 36.17620
## 24 1.0 18.0 5 80 25 3 1.00 0.75 44.33086
## 25 1.0 11.0 13 30 25 2 1.00 1.00 32.20758
## 26 1.0 14.0 11 25 25 1 1.00 0.75 31.43597
## 27 3.0 14.0 7 100 25 2 1.00 0.80 58.34514
## 28 5.0 12.0 10 200 25 3 1.25 0.67 40.91705
## 29 5.0 14.0 12 190 25 3 1.33 0.67 41.01549
## 30 0.0 13.0 12 25 25 2 1.00 0.75 28.02576
## 31 0.0 11.0 15 40 25 1 1.00 0.88 35.25244
## 32 0.0 15.0 9 45 25 2 1.00 0.75 23.80404
## 33 3.0 15.0 5 85 25 3 1.00 0.88 52.07690
## 34 3.0 17.0 3 90 25 3 1.00 0.25 53.37101
## 35 3.0 13.0 4 100 25 3 1.00 0.33 45.81172
## 36 1.0 12.0 11 45 25 2 1.00 1.00 21.87129
## 37 1.5 11.5 10 90 25 1 1.00 0.75 31.07222
## 38 0.0 14.0 11 35 25 1 1.00 1.33 28.74241
## 39 1.0 17.0 6 60 100 3 1.00 1.00 36.52368
## 40 2.0 20.0 9 95 100 3 1.30 0.75 36.47151
## 41 0.0 21.0 3 40 25 2 1.00 1.50 39.24111
## 42 2.0 12.0 6 95 25 2 1.00 0.67 45.32807
## 43 0.0 12.0 12 55 25 2 1.00 1.00 26.73451
## 44 0.0 16.0 3 95 25 2 1.00 1.00 54.85092
## 45 3.0 16.0 11 170 25 3 1.00 1.00 37.13686
## 46 3.0 16.0 11 170 25 3 1.00 1.00 34.13976
## 47 3.0 17.0 13 160 25 3 1.50 0.67 30.31335
## 48 2.0 15.0 6 90 25 1 1.00 1.00 40.10596
## 49 0.0 15.0 9 40 25 2 1.00 0.67 29.92429
## 50 3.0 21.0 7 130 25 3 1.33 0.67 40.69232
## 51 3.0 18.0 2 90 25 3 1.00 1.00 59.64284
## 52 1.5 13.5 10 120 25 3 1.25 0.50 30.45084
## 53 6.0 11.0 14 260 25 3 1.33 0.67 37.84059
## 54 1.0 20.0 3 45 100 3 1.00 1.00 41.50354
## 55 0.0 13.0 0 15 0 3 0.50 1.00 60.75611
## 56 1.0 10.0 0 50 0 3 0.50 1.00 63.00565
## 57 2.0 14.0 6 110 25 3 1.00 0.50 49.51187
## 58 2.7 NA NA 110 0 1 1.00 0.67 50.82839
## 59 5.0 14.0 12 240 25 2 1.33 0.75 39.25920
## 60 2.5 10.5 8 140 25 3 1.00 0.50 39.70340
## 61 2.0 15.0 6 110 25 3 1.00 0.50 55.33314
## 62 0.0 23.0 2 30 25 1 1.00 1.13 41.99893
## 63 0.0 22.0 3 35 25 1 1.00 1.00 40.56016
## 64 3.0 16.0 0 95 0 1 0.83 1.00 68.23588
## 65 4.0 19.0 0 140 0 1 1.00 0.67 74.47295
## 66 3.0 20.0 0 120 0 1 1.00 0.67 72.80179
## 67 1.0 9.0 15 40 25 2 1.00 0.75 31.23005
## 68 1.0 16.0 3 55 25 1 1.00 1.00 53.13132
## 69 3.0 15.0 5 90 25 2 1.00 1.00 59.36399
## 70 0.0 21.0 3 35 100 3 1.00 1.00 38.83975
## 71 4.0 15.0 14 230 100 3 1.50 1.00 28.59278
## 72 3.0 16.0 3 110 100 3 1.00 1.00 46.65884
## 73 0.0 21.0 3 60 25 3 1.00 0.75 39.10617
## 74 0.0 13.0 12 25 25 2 1.00 1.00 27.75330
## 75 3.0 17.0 3 115 25 1 1.00 0.67 49.78744
## 76 3.0 17.0 3 110 25 1 1.00 1.00 51.59219
## 77 1.0 16.0 8 60 25 1 1.00 0.75 36.18756
library(ggplot2)
library(reshape)
## Problem 4.1 Breakfast Cereals.
##Use the data for the breakfast cereals example in Section 4.8 to explore and
##summarize the data as follows:
#load the data
cereals.df <- read.csv("Cereals.csv", stringsAsFactors = FALSE)
#in the Global Environment window, investigate variable types included in cereals.df
##1 Which variables are integer/numeric/character? ###
#Integer Variables are: calories, protein, fat, sodium, sugars, potass, vitamins, shelf
#numeric variables are: fiber, carbo, weight, cups, rating
#Character Variables are: mfr, type , name
##2 Compute the mean, median, min, max, and standard deviation for each of####
##the quantitative (numeric/integer) variables. This can be done through R's sapply() function
##(e.g., sapply(data, mean, na.rm = TRUE)).
##please make sure to select all columns except for the character variables.
##If you are unsure, revisit selecting elements from data frame subsection of datacamp certificates for R
#q3 through q6 I'm giving you the first line of codes to print out mean value. Do the same work
##to print out median, min, max, and standard deviation for each of the quantitative variables.
sapply(cereals.df[,-c(1:3)], mean, na.rm=TRUE)
## calories protein fat sodium fiber carbo sugars
## 106.883117 2.545455 1.012987 159.675325 2.151948 14.802632 7.026316
## potass vitamins shelf weight cups rating
## 98.666667 28.246753 2.207792 1.029610 0.821039 42.665705
sapply(cereals.df[,-c(1:3)], median, na.rm=TRUE)
## calories protein fat sodium fiber carbo sugars potass
## 110.00000 3.00000 1.00000 180.00000 2.00000 14.50000 7.00000 90.00000
## vitamins shelf weight cups rating
## 25.00000 2.00000 1.00000 0.75000 40.40021
sapply(cereals.df[,-c(1:3)], min, na.rm=TRUE)
## calories protein fat sodium fiber carbo sugars potass
## 50.00000 1.00000 0.00000 0.00000 0.00000 5.00000 0.00000 15.00000
## vitamins shelf weight cups rating
## 0.00000 1.00000 0.50000 0.25000 18.04285
sapply(cereals.df[,-c(1:3)], max, na.rm=TRUE)
## calories protein fat sodium fiber carbo sugars potass
## 160.00000 6.00000 5.00000 320.00000 14.00000 23.00000 15.00000 330.00000
## vitamins shelf weight cups rating
## 100.00000 3.00000 1.50000 1.50000 93.70491
sapply(cereals.df[,-c(1:3)], sd, na.rm=TRUE)
## calories protein fat sodium fiber carbo sugars
## 19.4841191 1.0947897 1.0064726 83.8322952 2.3833640 3.9073256 4.3786564
## potass vitamins shelf weight cups rating
## 70.4106360 22.3425225 0.8325241 0.1504768 0.2327161 14.0472887
## Use R to plot a histogram for each of the quantitative variables.
# The par() function in R min, max and standard deviations used to set or query graphical parameters. When using par(mfcol=c(3,5)),
# you are specifying a 3x5 grid for the placement of subsequent plots.
# This means that any plots created after this command will be arranged in a grid with 3 rows and 5 columns.
# Each plot will be placed in the next available cell of the grid, filling the grid by column
#adjust plot margins
par(mar = c(2, 2, 1, 1))
par(mfcol=c(3,5)) ## you may need to adjust the value in c() so that the size can fit to your own environment like c(2,2) for example.
for (i in c(4:16)) {
hist(cereals.df[,i], main=names(cereals.df)[i])
}
#q7: specify 2x7 grid for the placement of subsequent plots.
par(mfcol=c(2,7))

for (i in c(4:16)){
hist(cereals.df[,i], main=names(cereals.df)[i])
}
##3. Use R to plot a box plot for each of the quantitative variables.###########
##Based on the histograms and summary statistics, answer the following questions:
## i. Which variables have the largest variability? HINT: investigate standard deviation values
#Largest Variablity: Calories, Sodium, potassium, sugars
## ii. Which variables seem skewed?
#Fiber, Fat, potass, seem positively skewed while calories and sodium seem negatively scewed.
## iii. Are there any values that seem extreme?
#Calories, sodium, potassium, fiber, and sugars seem to have extreme outliers.
par(mfcol=c(4,4))

for (i in c(4:16)){
boxplot(cereals.df[,i], main=names(cereals.df)[i])
}
##4. Use R to plot a side-by-side boxplot comparing the calories in hot vs. ####
##cold cereals. What does this plot show us?
#It shows us that there is a wider distribution of calories in Cold cereals.
#side by side boxplot of calories in hot vs. cold cereals
par(mfcol=c(1,1))

boxplot(cereals.df$calories ~ cereals.df$type, main = "Distribution of Calories",
xlab = "Type", ylab = "Calories")

##5. Use R to plot a side-by-side boxplot of consumer rating as a function#####
##of the shelf height. If we were to predict consumer rating from shelf height,
##does it appear that we need to keep all three categories of shelf height? Just think about it.
##You do not need to write anything for this question.
#side-by-side boxplot of consumer rating as a function of the shelf height
boxplot(cereals.df$rating ~ cereals.df$shelf, main = "Distribution of Rating",
xlab = "Shelf Height", ylab = "Rating")

##6. Compute the correlation table for the quantitative variable (function ####
##cor()). In addition, generate a matrix plot for these variables (necessary
##codes are provided)).
## i. Which pair of variables is most strongly correlated?
#Sugars to carbos are highlt correlated, so are calories to sugars, and fiber to potass.
## ii. How can we reduce the number of variables based on these correlations?
#We can remove redundant variables (sugars/carbos), merge correlated variables (ie fiber/potass), or use PCA to retain variance but make a smaller set of components.
## iii. How would the correlations change if we normalized the dat a first?
#Normalizing the data wouldn't change the correlation values but it would make variables comparable, improve PCA performance, and keep correlation structure similar but standardize variables.
options(digits = 1) #to print less decimals in correlation matrix
cor_matrix = cor(na.omit(cereals.df[,-c(1:3)]))
cor_matrix
## calories protein fat sodium fiber carbo sugars potass vitamins
## calories 1.00 0.03 5e-01 3e-01 -0.30 0.27 0.569 -0.071 0.260
## protein 0.03 1.00 2e-01 1e-02 0.51 -0.04 -0.287 0.579 0.055
## fat 0.51 0.20 1e+00 8e-04 0.01 -0.28 0.287 0.200 -0.031
## sodium 0.30 0.01 8e-04 1e+00 -0.07 0.33 0.037 -0.039 0.332
## fiber -0.30 0.51 1e-02 -7e-02 1.00 -0.38 -0.151 0.912 -0.039
## carbo 0.27 -0.04 -3e-01 3e-01 -0.38 1.00 -0.452 -0.365 0.254
## sugars 0.57 -0.29 3e-01 4e-02 -0.15 -0.45 1.000 0.001 0.073
## potass -0.07 0.58 2e-01 -4e-02 0.91 -0.37 0.001 1.000 -0.003
## vitamins 0.26 0.05 -3e-02 3e-01 -0.04 0.25 0.073 -0.003 1.000
## shelf 0.09 0.20 3e-01 -1e-01 0.31 -0.19 0.061 0.395 0.284
## weight 0.70 0.23 2e-01 3e-01 0.25 0.14 0.461 0.421 0.320
## cups 0.09 -0.24 -2e-01 1e-01 -0.51 0.36 -0.032 -0.502 0.134
## rating -0.69 0.47 -4e-01 -4e-01 0.60 0.06 -0.756 0.416 -0.214
## shelf weight cups rating
## calories 0.09 0.7 0.09 -0.69
## protein 0.20 0.2 -0.24 0.47
## fat 0.28 0.2 -0.16 -0.41
## sodium -0.12 0.3 0.12 -0.38
## fiber 0.31 0.2 -0.51 0.60
## carbo -0.19 0.1 0.36 0.06
## sugars 0.06 0.5 -0.03 -0.76
## potass 0.39 0.4 -0.50 0.42
## vitamins 0.28 0.3 0.13 -0.21
## shelf 1.00 0.2 -0.35 0.05
## weight 0.19 1.0 -0.20 -0.30
## cups -0.35 -0.2 1.00 -0.22
## rating 0.05 -0.3 -0.22 1.00
# Melt the correlation matrix to long format for ggplot2
melted_cor_matrix <- melt(cor_matrix)
## Warning in type.convert.default(X[[i]], ...): 'as.is' should be specified by
## the caller; using TRUE
## Warning in type.convert.default(X[[i]], ...): 'as.is' should be specified by
## the caller; using TRUE
# Create a heatmap
ggplot(data = melted_cor_matrix, aes(x=X1, y=X2, fill=value)) +
geom_tile() +
scale_fill_gradient2(low = "blue", high = "red", mid = "white",
midpoint = 0, limit = c(-1,1), space = "Lab",
name="Pearson\nCorrelation") +
theme_minimal() +
theme(axis.text.x = element_text(angle = 45, vjust = 1,
size = 12, hjust = 1),
axis.text.y = element_text(size = 12)) +
coord_fixed()

#7. PCA ############################
#install and/or load ggbiplot and pacman packages with pacman
pacman::p_load(
ggbiplot,
pacman,
tidyr,
dplyr
)
prep <- na.omit(cereals.df[,-c(1:3)])
pc <- prcomp(prep, scale. = TRUE)
pc
## Standard deviations (1, .., p=13):
## [1] 2e+00 2e+00 1e+00 1e+00 1e+00 8e-01 8e-01 6e-01 6e-01 3e-01 3e-01 1e-01
## [13] 1e-08
##
## Rotation (n x k) = (13 x 13):
## PC1 PC2 PC3 PC4 PC5 PC6 PC7 PC8 PC9 PC10 PC11
## calories 0.30 0.4 -0.115 0.20 0.20 0.26 -0.03 -0.002 -0.03 -0.500 0.214
## protein -0.31 0.2 -0.277 0.30 0.32 -0.12 0.28 -0.427 -0.53 0.022 -0.032
## fat 0.04 0.3 0.205 0.19 0.59 -0.35 -0.05 0.063 0.46 0.145 0.067
## sodium 0.18 0.1 -0.389 0.12 -0.34 -0.66 -0.28 0.177 -0.22 0.001 0.087
## fiber -0.45 0.2 -0.070 0.04 -0.26 -0.06 0.11 0.216 0.24 -0.295 0.531
## carbo 0.19 -0.1 -0.562 0.09 0.18 0.33 -0.26 0.167 0.12 -0.241 -0.179
## sugars 0.23 0.4 0.355 -0.02 -0.31 0.15 0.23 -0.063 -0.23 -0.252 0.003
## potass -0.40 0.3 -0.068 0.09 -0.15 -0.03 0.15 0.262 0.17 -0.177 -0.729
## vitamins 0.12 0.2 -0.388 -0.60 -0.05 -0.13 0.29 -0.457 0.35 -0.052 -0.019
## shelf -0.17 0.3 0.002 -0.64 0.33 0.05 -0.17 0.414 -0.42 0.046 0.059
## weight 0.05 0.5 -0.247 0.15 -0.22 0.40 0.01 0.075 0.07 0.692 0.113
## cups 0.29 -0.2 -0.140 0.05 0.12 -0.10 0.75 0.499 -0.05 0.077 0.055
## rating -0.44 -0.3 -0.182 0.04 0.06 0.19 0.06 0.015 0.06 -0.012 0.276
## PC12 PC13
## calories -5e-01 -2e-01
## protein 1e-01 2e-01
## fat 3e-01 -9e-02
## sodium 5e-02 -2e-01
## fiber 6e-02 4e-01
## carbo 5e-01 2e-01
## sugars 6e-01 -2e-01
## potass -1e-01 -1e-01
## vitamins -4e-05 -6e-02
## shelf 2e-02 -1e-09
## weight -3e-02 -4e-09
## cups -3e-02 1e-09
## rating 2e-01 -7e-01
summary(pc)
## Importance of components:
## PC1 PC2 PC3 PC4 PC5 PC6 PC7 PC8
## Standard deviation 1.91 1.774 1.382 1.0097 0.9947 0.8497 0.8195 0.645
## Proportion of Variance 0.28 0.242 0.147 0.0784 0.0761 0.0555 0.0517 0.032
## Cumulative Proportion 0.28 0.522 0.669 0.7470 0.8231 0.8786 0.9303 0.962
## PC9 PC10 PC11 PC12 PC13
## Standard deviation 0.5619 0.30301 0.25194 0.13897 1.5e-08
## Proportion of Variance 0.0243 0.00706 0.00488 0.00149 0.0e+00
## Cumulative Proportion 0.9866 0.99363 0.99851 1.00000 1.0e+00
# The purpose of plotting eigenvalues in R is to visualize the importance
# of each principal component and to determine the number of principal
# components to retain in techniques such as Principal Component Analysis
# (PCA). The plot, often referred to as a scree plot, shows the eigenvalues
# in a downward curve, from highest to lowest, allowing the user to identify
# the most significant components. This visualization helps in making
# informed decisions about the number of components to keep in the analysis,
# based on the proportion of variance explained by each component
pc %>%
ggbiplot(
color = prep$rating, #color by rating
groups = factor(prep$rating),
labels = rownames(prep)
) +
theme(legend.position = "none")

# Extract the eigenvalues (variances) from the PCA results
eigenvalues <- pc$sdev^2
# Calculate the explained variance ratio for each principal component
explained_variance_ratio <- eigenvalues / sum(eigenvalues)
# View the explained variance ratio. Explained variance ration tells us how much informationis compressed
#into the first few components
#turn off scientific notations
options(scipen = 999)
round(explained_variance_ratio, digits=2)
## [1] 0.28 0.24 0.15 0.08 0.08 0.06 0.05 0.03 0.02 0.01 0.00 0.00 0.00
#first three principle components explain 67% of the entire variance.
sum(explained_variance_ratio)
## [1] 1
#1 --cumulative variance is 1
pc %>% plot(main="Eiganvalues")
