#Loading data

q1 <- read_csv("~/data/Divvy_Trips_2018_Q1/Divvy_Trips_2018_Q1.csv")
## Rows: 387145 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (6): 01 - Rental Details Local Start Time, 01 - Rental Details Local End...
## dbl (5): 01 - Rental Details Rental ID, 01 - Rental Details Bike ID, 03 - Re...
## num (1): 01 - Rental Details Duration In Seconds Uncapped
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
q2 <- read_csv("~/data/Divvy_Trips_2018_Q2/Divvy_Trips_2018_Q2.csv")
## Rows: 1059681 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr  (4): from_station_name, to_station_name, usertype, gender
## dbl  (5): trip_id, bikeid, from_station_id, to_station_id, birthyear
## num  (1): tripduration
## dttm (2): start_time, end_time
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
q3 <- read_csv("~/data/Divvy_Trips_2018_Q3/Divvy_Trips_2018_Q3.csv")
## Rows: 1513570 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr  (4): from_station_name, to_station_name, usertype, gender
## dbl  (5): trip_id, bikeid, from_station_id, to_station_id, birthyear
## num  (1): tripduration
## dttm (2): start_time, end_time
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
q4 <- read_csv("~/data/Divvy_Trips_2018_Q4/Divvy_Trips_2018_Q4.csv")
## Rows: 642686 Columns: 12
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr  (4): from_station_name, to_station_name, usertype, gender
## dbl  (5): trip_id, bikeid, from_station_id, to_station_id, birthyear
## num  (1): tripduration
## dttm (2): start_time, end_time
## 
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.

#change headings for q1 to enable binding

q1 <- rename(q1, "trip_id" = `01 - Rental Details Rental ID`, "start_time" = `01 - Rental Details Local Start Time`, "end_time" = `01 - Rental Details Local End Time`, "bikeid" = `01 - Rental Details Bike ID`, "tripduration" = `01 - Rental Details Duration In Seconds Uncapped`, "from_station_id" = `03 - Rental Start Station ID`,"from_station_name" = `03 - Rental Start Station Name`, "to_station_id" = `02 - Rental End Station ID`, "to_station_name" = `02 - Rental End Station Name`, "usertype" = `User Type`,"gender" = `Member Gender`, "birthyear" = `05 - Member Details Member Birthday Year`)

#view head in q1

head(q1, n=5)
## # A tibble: 5 × 12
##    trip_id start_time       end_time         bikeid tripduration from_station_id
##      <dbl> <chr>            <chr>             <dbl>        <dbl>           <dbl>
## 1 17536702 01/01/2018 00:12 01/01/2018 00:17   3304          323              69
## 2 17536703 01/01/2018 00:41 01/01/2018 00:47   5367          377             253
## 3 17536704 01/01/2018 00:44 01/01/2018 01:33   4599         2904              98
## 4 17536705 01/01/2018 00:53 01/01/2018 01:05   2302          747             125
## 5 17536706 01/01/2018 00:53 01/01/2018 00:56   3696          183             129
## # ℹ 6 more variables: from_station_name <chr>, to_station_id <dbl>,
## #   to_station_name <chr>, usertype <chr>, gender <chr>, birthyear <dbl>

#combine the entire data set as yearly_report

all_quaters <- rbind(q1,q2,q3,q4)

#view the new dataset ‘all_quaters’ it contains a lot of NAs

view(all_quaters)

#checking for the NAs from the entire dataset

sum(is.na(all_quaters))
## [1] 1117714

#remove all NAs

all_quaters <- na.omit(all_quaters)

#verifying NAs removed

sum(is.na(all_quaters))
## [1] 0

#maximum and minimum ride time and other info. about the rider

all_quaters[which.max(all_quaters$tripduration),]
## # A tibble: 1 × 12
##    trip_id start_time       end_time         bikeid tripduration from_station_id
##      <dbl> <chr>            <chr>             <dbl>        <dbl>           <dbl>
## 1 17634739 25/01/2018 19:56 01/07/2018 18:56   5956     13557600             585
## # ℹ 6 more variables: from_station_name <chr>, to_station_id <dbl>,
## #   to_station_name <chr>, usertype <chr>, gender <chr>, birthyear <dbl>
all_quaters[which.min(all_quaters$tripduration),]
## # A tibble: 1 × 12
##    trip_id start_time       end_time         bikeid tripduration from_station_id
##      <dbl> <chr>            <chr>             <dbl>        <dbl>           <dbl>
## 1 17543134 04/01/2018 06:11 04/01/2018 06:12   6032           61              91
## # ℹ 6 more variables: from_station_name <chr>, to_station_id <dbl>,
## #   to_station_name <chr>, usertype <chr>, gender <chr>, birthyear <dbl>

#mean tripduration

mean(all_quaters$tripduration)
## [1] 950.9229

#Visualizing the dataset

ggplot(all_quaters, aes(gender))+geom_bar()

ggplot(all_quaters,aes(usertype))+geom_bar()