#Loading data
q1 <- read_csv("~/data/Divvy_Trips_2018_Q1/Divvy_Trips_2018_Q1.csv")
## Rows: 387145 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (6): 01 - Rental Details Local Start Time, 01 - Rental Details Local End...
## dbl (5): 01 - Rental Details Rental ID, 01 - Rental Details Bike ID, 03 - Re...
## num (1): 01 - Rental Details Duration In Seconds Uncapped
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
q2 <- read_csv("~/data/Divvy_Trips_2018_Q2/Divvy_Trips_2018_Q2.csv")
## Rows: 1059681 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (4): from_station_name, to_station_name, usertype, gender
## dbl (5): trip_id, bikeid, from_station_id, to_station_id, birthyear
## num (1): tripduration
## dttm (2): start_time, end_time
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
q3 <- read_csv("~/data/Divvy_Trips_2018_Q3/Divvy_Trips_2018_Q3.csv")
## Rows: 1513570 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (4): from_station_name, to_station_name, usertype, gender
## dbl (5): trip_id, bikeid, from_station_id, to_station_id, birthyear
## num (1): tripduration
## dttm (2): start_time, end_time
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
q4 <- read_csv("~/data/Divvy_Trips_2018_Q4/Divvy_Trips_2018_Q4.csv")
## Rows: 642686 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (4): from_station_name, to_station_name, usertype, gender
## dbl (5): trip_id, bikeid, from_station_id, to_station_id, birthyear
## num (1): tripduration
## dttm (2): start_time, end_time
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
#change headings for q1 to enable binding
q1 <- rename(q1, "trip_id" = `01 - Rental Details Rental ID`, "start_time" = `01 - Rental Details Local Start Time`, "end_time" = `01 - Rental Details Local End Time`, "bikeid" = `01 - Rental Details Bike ID`, "tripduration" = `01 - Rental Details Duration In Seconds Uncapped`, "from_station_id" = `03 - Rental Start Station ID`,"from_station_name" = `03 - Rental Start Station Name`, "to_station_id" = `02 - Rental End Station ID`, "to_station_name" = `02 - Rental End Station Name`, "usertype" = `User Type`,"gender" = `Member Gender`, "birthyear" = `05 - Member Details Member Birthday Year`)
#view head in q1
head(q1, n=5)
## # A tibble: 5 × 12
## trip_id start_time end_time bikeid tripduration from_station_id
## <dbl> <chr> <chr> <dbl> <dbl> <dbl>
## 1 17536702 01/01/2018 00:12 01/01/2018 00:17 3304 323 69
## 2 17536703 01/01/2018 00:41 01/01/2018 00:47 5367 377 253
## 3 17536704 01/01/2018 00:44 01/01/2018 01:33 4599 2904 98
## 4 17536705 01/01/2018 00:53 01/01/2018 01:05 2302 747 125
## 5 17536706 01/01/2018 00:53 01/01/2018 00:56 3696 183 129
## # ℹ 6 more variables: from_station_name <chr>, to_station_id <dbl>,
## # to_station_name <chr>, usertype <chr>, gender <chr>, birthyear <dbl>
#combine the entire data set as yearly_report
all_quaters <- rbind(q1,q2,q3,q4)
#view the new dataset ‘all_quaters’ it contains a lot of NAs
view(all_quaters)
#checking for the NAs from the entire dataset
sum(is.na(all_quaters))
## [1] 1117714
#remove all NAs
all_quaters <- na.omit(all_quaters)
#verifying NAs removed
sum(is.na(all_quaters))
## [1] 0
#maximum and minimum ride time and other info. about the rider
all_quaters[which.max(all_quaters$tripduration),]
## # A tibble: 1 × 12
## trip_id start_time end_time bikeid tripduration from_station_id
## <dbl> <chr> <chr> <dbl> <dbl> <dbl>
## 1 17634739 25/01/2018 19:56 01/07/2018 18:56 5956 13557600 585
## # ℹ 6 more variables: from_station_name <chr>, to_station_id <dbl>,
## # to_station_name <chr>, usertype <chr>, gender <chr>, birthyear <dbl>
all_quaters[which.min(all_quaters$tripduration),]
## # A tibble: 1 × 12
## trip_id start_time end_time bikeid tripduration from_station_id
## <dbl> <chr> <chr> <dbl> <dbl> <dbl>
## 1 17543134 04/01/2018 06:11 04/01/2018 06:12 6032 61 91
## # ℹ 6 more variables: from_station_name <chr>, to_station_id <dbl>,
## # to_station_name <chr>, usertype <chr>, gender <chr>, birthyear <dbl>
#mean tripduration
mean(all_quaters$tripduration)
## [1] 950.9229
#Visualizing the dataset
ggplot(all_quaters, aes(gender))+geom_bar()
ggplot(all_quaters,aes(usertype))+geom_bar()