This report is Project 1 for MKTG3P98
getwd()
## [1] "/Users/charlesmba"
setwd("/Users/charlesmba")
library("ggplot2")
library(readr) #import library to read csv
Car1 <- read.csv("/Users/charlesmba/Desktop/Car_Survey_1 csv.csv")
str(Car1) #summary of Car_Survey_1
## 'data.frame': 1180 obs. of 23 variables:
## $ Resp : chr "Res1" "Res2" "Res3" "Res4" ...
## $ Att_1 : int 6 7 7 4 6 6 1 6 3 6 ...
## $ Att_2 : int 6 5 7 1 6 6 1 5 2 6 ...
## $ Enj_1 : int 6 5 7 1 6 6 1 5 3 4 ...
## $ Enj_2 : int 6 2 5 1 5 5 1 3 2 4 ...
## $ Perform_1 : int 5 2 5 1 5 5 2 5 2 4 ...
## $ Perform_2 : int 6 6 5 1 2 5 2 5 3 4 ...
## $ Perform_3 : int 3 7 3 1 1 7 2 2 1 1 ...
## $ WOM_1 : int 3 5 6 7 7 5 2 4 6 5 ...
## $ WOM_2 : int 3 5 6 7 7 5 3 6 6 6 ...
## $ Futu_Pur_1 : int 3 6 7 3 7 7 5 4 7 6 ...
## $ Futu_Pur_2 : int 3 6 7 3 6 7 2 4 7 6 ...
## $ Valu_Percp_1: int 5 6 5 6 6 7 2 4 6 6 ...
## $ Valu_Percp_2: int 2 7 7 5 5 7 2 4 6 6 ...
## $ Pur_Proces_1: int 6 7 7 5 6 7 2 4 6 6 ...
## $ Pur_Proces_2: int 4 6 7 4 7 7 6 4 6 6 ...
## $ Residence : int 2 2 1 2 1 2 2 1 2 1 ...
## $ Pay_Meth : int 2 2 2 2 2 2 2 2 2 2 ...
## $ Insur_Type : chr "Collision" "Collision" "Collision" "Collision" ...
## $ Gender : chr "Male" "Male" "Male" "Male" ...
## $ Age : int 18 18 19 19 19 19 19 21 21 21 ...
## $ Education : int 2 2 2 2 2 2 2 2 2 2 ...
## $ X : logi NA NA NA NA NA NA ...
head(Car1, n = 10)
## Resp Att_1 Att_2 Enj_1 Enj_2 Perform_1 Perform_2 Perform_3 WOM_1 WOM_2
## 1 Res1 6 6 6 6 5 6 3 3 3
## 2 Res2 7 5 5 2 2 6 7 5 5
## 3 Res3 7 7 7 5 5 5 3 6 6
## 4 Res4 4 1 1 1 1 1 1 7 7
## 5 Res5 6 6 6 5 5 2 1 7 7
## 6 Res6 6 6 6 5 5 5 7 5 5
## 7 Res7 1 1 1 1 2 2 2 2 3
## 8 Res8 6 5 5 3 5 5 2 4 6
## 9 Res9 3 2 3 2 2 3 1 6 6
## 10 Res10 6 6 4 4 4 4 1 5 6
## Futu_Pur_1 Futu_Pur_2 Valu_Percp_1 Valu_Percp_2 Pur_Proces_1 Pur_Proces_2
## 1 3 3 5 2 6 4
## 2 6 6 6 7 7 6
## 3 7 7 5 7 7 7
## 4 3 3 6 5 5 4
## 5 7 6 6 5 6 7
## 6 7 7 7 7 7 7
## 7 5 2 2 2 2 6
## 8 4 4 4 4 4 4
## 9 7 7 6 6 6 6
## 10 6 6 6 6 6 6
## Residence Pay_Meth Insur_Type Gender Age Education X
## 1 2 2 Collision Male 18 2 NA
## 2 2 2 Collision Male 18 2 NA
## 3 1 2 Collision Male 19 2 NA
## 4 2 2 Collision Male 19 2 NA
## 5 1 2 Collision Female 19 2 NA
## 6 2 2 Collision Female 19 2 NA
## 7 2 2 Collision Male 19 2 NA
## 8 1 2 Collision Male 21 2 NA
## 9 2 2 Collision Male 21 2 NA
## 10 1 2 Collision Male 21 2 NA
Car2 <- read.csv("/Users/charlesmba/Downloads/Car_Survey_2 csv.csv")
str(Car2) #summary of Car_Survey_2
## 'data.frame': 1049 obs. of 9 variables:
## $ Respondents: chr "Res1" "Res2" "Res3" "Res4" ...
## $ Region : chr "European" "European" "European" "European" ...
## $ Model : chr "Ford Expedition" "Ford Expedition" "Ford Expedition" "Ford Expedition" ...
## $ MPG : int 15 15 15 15 15 15 15 15 15 15 ...
## $ Cyl : int 8 8 8 8 8 8 8 8 8 8 ...
## $ acc1 : num 5.5 5.5 5.5 5.5 5.5 5.5 5.5 5.5 5.5 5.5 ...
## $ C_cost. : num 16 16 16 16 16 16 16 16 16 16 ...
## $ H_Cost : num 14 14 14 14 14 14 14 14 14 14 ...
## $ Post.Satis : int 4 3 5 5 5 3 3 6 3 5 ...
head(Car2, n = 10)
## Respondents Region Model MPG Cyl acc1 C_cost. H_Cost Post.Satis
## 1 Res1 European Ford Expedition 15 8 5.5 16 14 4
## 2 Res2 European Ford Expedition 15 8 5.5 16 14 3
## 3 Res3 European Ford Expedition 15 8 5.5 16 14 5
## 4 Res4 European Ford Expedition 15 8 5.5 16 14 5
## 5 Res5 European Ford Expedition 15 8 5.5 16 14 5
## 6 Res6 European Ford Expedition 15 8 5.5 16 14 3
## 7 Res7 European Ford Expedition 15 8 5.5 16 14 3
## 8 Res8 European Ford Expedition 15 8 5.5 16 14 6
## 9 Res9 European Ford Expedition 15 8 5.5 16 14 3
## 10 Res10 European Ford Expedition 15 8 5.5 16 14 5
names(Car2)[1] <- c("Resp")
head(Car2, n=1)
## Resp Region Model MPG Cyl acc1 C_cost. H_Cost Post.Satis
## 1 Res1 European Ford Expedition 15 8 5.5 16 14 4
Car_Total <- merge(Car1, Car2, by= "Resp")
str(Car_Total)
## 'data.frame': 1049 obs. of 31 variables:
## $ Resp : chr "Res1" "Res10" "Res100" "Res1000" ...
## $ Att_1 : int 6 6 6 6 6 3 2 7 2 6 ...
## $ Att_2 : int 6 6 7 6 6 1 2 7 1 6 ...
## $ Enj_1 : int 6 4 7 7 7 4 1 7 2 6 ...
## $ Enj_2 : int 6 4 3 6 6 3 2 6 1 5 ...
## $ Perform_1 : int 5 4 5 6 6 5 2 5 2 5 ...
## $ Perform_2 : int 6 4 6 6 6 6 2 6 2 5 ...
## $ Perform_3 : int 3 1 6 6 6 6 1 5 2 5 ...
## $ WOM_1 : int 3 5 3 6 4 2 6 6 7 3 ...
## $ WOM_2 : int 3 6 5 6 4 6 7 6 7 3 ...
## $ Futu_Pur_1 : int 3 6 6 6 4 6 6 6 7 6 ...
## $ Futu_Pur_2 : int 3 6 6 6 6 6 5 7 7 6 ...
## $ Valu_Percp_1: int 5 6 7 4 5 5 4 6 4 5 ...
## $ Valu_Percp_2: int 2 6 6 6 6 4 4 5 6 6 ...
## $ Pur_Proces_1: int 6 6 5 6 6 5 4 5 6 6 ...
## $ Pur_Proces_2: int 4 6 5 3 7 5 5 5 7 5 ...
## $ Residence : int 2 1 2 2 1 1 1 2 1 2 ...
## $ Pay_Meth : int 2 2 1 3 3 3 3 3 3 3 ...
## $ Insur_Type : chr "Collision" "Collision" "Collision" "Liability" ...
## $ Gender : chr "Male" "Male" "Female" "Female" ...
## $ Age : int 18 21 32 24 24 25 26 26 27 27 ...
## $ Education : int 2 2 1 2 2 2 2 2 2 2 ...
## $ X : logi NA NA NA NA NA NA ...
## $ Region : chr "European" "European" "American" "Asian" ...
## $ Model : chr "Ford Expedition" "Ford Expedition" "Toyota Rav4" "Toyota Corolla" ...
## $ MPG : int 15 15 24 26 26 26 26 26 26 26 ...
## $ Cyl : int 8 8 4 4 4 4 4 4 4 4 ...
## $ acc1 : num 5.5 5.5 8.2 8 8 8 8 8 8 8 ...
## $ C_cost. : num 16 16 10 7 7 7 7 7 7 7 ...
## $ H_Cost : num 14 14 8 6 6 6 6 6 6 6 ...
## $ Post.Satis : int 4 5 4 6 5 6 5 6 7 6 ...
na_rows <- Car_Total[is.na(Car_Total$Att_1),]
print(na_rows)
## Resp Att_1 Att_2 Enj_1 Enj_2 Perform_1 Perform_2 Perform_3 WOM_1 WOM_2
## 109 Res151 NA 2 NA 2 3 NA 2 2 NA
## 110 Res152 NA 5 5 4 5 6 4 5 6
## 112 Res154 NA 6 6 6 5 4 5 6 6
## 127 Res168 NA 3 3 3 6 5 3 5 6
## Futu_Pur_1 Futu_Pur_2 Valu_Percp_1 Valu_Percp_2 Pur_Proces_1 Pur_Proces_2
## 109 2 2 NA 2 2 3
## 110 6 6 6 2 4 5
## 112 5 6 5 6 5 5
## 127 6 7 5 3 5 4
## Residence Pay_Meth Insur_Type Gender Age Education X Region
## 109 1 2 Comprehensive Female 21 2 NA American
## 110 2 1 Comprehensive Female 21 2 NA American
## 112 2 2 Comprehensive Female 23 2 NA American
## 127 1 2 Collision Male 29 2 NA Asian
## Model MPG Cyl acc1 C_cost. H_Cost Post.Satis
## 109 Chrysler Jeep 18 6 3.6 12 10.0 4
## 110 Chrysler Jeep 18 6 3.6 12 10.0 6
## 112 Chrysler Jeep 18 6 3.6 12 10.0 6
## 127 Toyota Highlander 20 6 7.2 10 8.5 6
meanATT_1 <- mean(Car_Total$Att_1,na.rm = TRUE)
print(meanATT_1)
## [1] 4.882297
numeric_cols <- sapply(Car_Total, is.numeric)
Car_Total[, numeric_cols] <- lapply(Car_Total[, numeric_cols], function(x) {
mean_val <- mean (x, na.rm = TRUE)
x[is.na(x)] <- mean_val
return(x)
})
summary(Car_Total)
## Resp Att_1 Att_2 Enj_1
## Length:1049 Min. :1.000 Min. :1.000 Min. :1.000
## Class :character 1st Qu.:4.000 1st Qu.:4.000 1st Qu.:5.000
## Mode :character Median :5.000 Median :6.000 Median :6.000
## Mean :4.882 Mean :5.287 Mean :5.378
## 3rd Qu.:6.000 3rd Qu.:6.000 3rd Qu.:7.000
## Max. :7.000 Max. :7.000 Max. :7.000
## Enj_2 Perform_1 Perform_2 Perform_3
## Min. :1.000 Min. :1.000 Min. :1.000 Min. :1.000
## 1st Qu.:3.000 1st Qu.:4.000 1st Qu.:4.000 1st Qu.:3.000
## Median :5.000 Median :5.000 Median :5.000 Median :5.000
## Mean :4.575 Mean :4.947 Mean :4.831 Mean :4.217
## 3rd Qu.:6.000 3rd Qu.:6.000 3rd Qu.:6.000 3rd Qu.:6.000
## Max. :7.000 Max. :7.000 Max. :7.000 Max. :7.000
## WOM_1 WOM_2 Futu_Pur_1 Futu_Pur_2 Valu_Percp_1
## Min. :1.000 Min. :1.00 Min. :1.000 Min. :1.000 Min. :1.000
## 1st Qu.:4.000 1st Qu.:4.00 1st Qu.:5.000 1st Qu.:5.000 1st Qu.:5.000
## Median :6.000 Median :6.00 Median :6.000 Median :6.000 Median :6.000
## Mean :5.286 Mean :5.35 Mean :5.321 Mean :5.371 Mean :5.411
## 3rd Qu.:7.000 3rd Qu.:6.00 3rd Qu.:6.000 3rd Qu.:6.000 3rd Qu.:6.000
## Max. :7.000 Max. :7.00 Max. :9.000 Max. :7.000 Max. :7.000
## Valu_Percp_2 Pur_Proces_1 Pur_Proces_2 Residence
## Min. :1.000 Min. :1.000 Min. :1.000 Min. :1.000
## 1st Qu.:4.000 1st Qu.:5.000 1st Qu.:4.000 1st Qu.:1.000
## Median :5.000 Median :6.000 Median :5.000 Median :1.000
## Mean :5.114 Mean :5.256 Mean :4.923 Mean :1.474
## 3rd Qu.:6.000 3rd Qu.:6.000 3rd Qu.:6.000 3rd Qu.:2.000
## Max. :7.000 Max. :7.000 Max. :7.000 Max. :5.000
## Pay_Meth Insur_Type Gender Age
## Min. :1.000 Length:1049 Length:1049 Min. :18.00
## 1st Qu.:1.000 Class :character Class :character 1st Qu.:23.00
## Median :2.000 Mode :character Mode :character Median :34.00
## Mean :2.153 Mean :35.22
## 3rd Qu.:3.000 3rd Qu.:48.00
## Max. :3.000 Max. :60.00
## Education X Region Model
## Min. :1.000 Mode:logical Length:1049 Length:1049
## 1st Qu.:2.000 NA's:1049 Class :character Class :character
## Median :2.000 Mode :character Mode :character
## Mean :1.989
## 3rd Qu.:2.000
## Max. :3.000
## MPG Cyl acc1 C_cost. H_Cost
## Min. :14.00 Min. :4.0 Min. :3.600 Min. : 7.00 Min. : 6.000
## 1st Qu.:17.00 1st Qu.:4.0 1st Qu.:5.100 1st Qu.:10.00 1st Qu.: 8.000
## Median :19.00 Median :6.0 Median :6.500 Median :12.00 Median :10.000
## Mean :19.58 Mean :5.8 Mean :6.202 Mean :11.35 Mean : 9.634
## 3rd Qu.:22.00 3rd Qu.:6.0 3rd Qu.:7.500 3rd Qu.:13.00 3rd Qu.:11.000
## Max. :26.00 Max. :8.0 Max. :8.500 Max. :16.00 Max. :14.000
## Post.Satis
## Min. :2.00
## 1st Qu.:5.00
## Median :6.00
## Mean :5.28
## 3rd Qu.:6.00
## Max. :7.00
Car_Total$ATT_mean <- (Car_Total$Att_1 + Car_Total$Att_2) / 2
write.csv(Car_Total, "New_Car_Total.csv", row.names = FALSE)
list.files()
## [1] "Applications" "Assignment 1 .Rmd"
## [3] "Assignment 1.Rmd" "assignment code Untitled.R"
## [5] "Assignment-1-.html" "Assignment-1.html"
## [7] "Assignment-1.Rmd" "Desktop"
## [9] "Documents" "Downloads"
## [11] "Library" "Movies"
## [13] "Music" "New_Car_Total.csv"
## [15] "OneDrive - Brock University" "Pictures"
## [17] "Public" "Untitled.html"
## [19] "Untitled.Rmd"
New_Car_Total <- read.csv("New_Car_Total.csv")
ggplot(New_Car_Total, aes(x = Model, fill = Model)) +
geom_bar() +
labs(title = "Number of People Owning Different Car Brands",
x = "Car Brand",
y = "Count") +
theme_minimal() +
theme(axis.text.x = element_text(angle = 45, hjust = 1))
toyota_models <- c( "Toyota Corolla", "Toyota Rav4", "Toyota Highlander")
toyota_cars <- subset(New_Car_Total, Model %in% toyota_models)
meanT_Mpg <- mean(toyota_cars$MPG,na.rm = TRUE)
print(meanT_Mpg)
## [1] 22.73973
Ford_models <- c( "Ford Expedition", "Ford Explorer")
Ford_cars <- subset(New_Car_Total, Model %in% Ford_models)
meanF_Mpg <- mean(Ford_cars$MPG,na.rm = TRUE)
print(meanF_Mpg)
## [1] 16.9604
Honda_models <- c( "Honda CRV", "Honda Pilot")
Honda_cars <- subset(New_Car_Total, Model %in% Honda_models)
meanH_Mpg <- mean(Honda_cars$MPG,na.rm = TRUE)
print(meanH_Mpg)
## [1] 22.79245
Chrysler_models <- c( "Chrysler Jeep")
Chrysler_cars <- subset(New_Car_Total, Model %in% Chrysler_models)
meanC_Mpg <- mean(Chrysler_cars$MPG,na.rm = TRUE)
print(meanC_Mpg)
## [1] 18
mean_mpg_data <- data.frame(
Brand = c("Ford", "Toyota", "Honda", "Chrysler"),
Mean_MPG = c(meanF_Mpg, meanT_Mpg, meanH_Mpg, meanC_Mpg)
)
print(mean_mpg_data)
## Brand Mean_MPG
## 1 Ford 16.96040
## 2 Toyota 22.73973
## 3 Honda 22.79245
## 4 Chrysler 18.00000
ggplot(mean_mpg_data, aes(x = Brand, y = Mean_MPG, fill = Brand)) +
geom_bar(stat = "identity") +
labs(title = "Mean MPG for Top 4 owned brands",
x = "Car Brand",
y = "Mean MPG") +
theme_minimal()
New_Car_Total$WOM_mean <- (Car_Total$WOM_1 + Car_Total$WOM_2) / 2
toyota_models <- c( "Toyota Corolla", "Toyota Rav4", "Toyota Highlander")
Ford_models <- c( "Ford Expedition", "Ford Explorer")