Read in csv file.
bike <- read.csv("NYC-2016-Summary.csv")
head(bike)
## duration month hour day_of_week user_type
## 1 13.98333 1 0 Friday Customer
## 2 11.43333 1 0 Friday Subscriber
## 3 5.25000 1 0 Friday Subscriber
## 4 12.31667 1 0 Friday Subscriber
## 5 20.88333 1 0 Friday Customer
## 6 8.75000 1 0 Friday Subscriber
names(bike)
## [1] "duration" "month" "hour" "day_of_week" "user_type"
Created data files to later organize the original csv file.
months <- c("Jan", "Feb", "Mar", "Apr", "May", "Jun","Jul", "Aug", "Sep", "Oct", "Nov", "Dec")
days <- c("Monday", "Tuesday", "Wednesday","Thursday", "Friday", "Saturday", "Sunday")
hours <- sprintf("%02d:00", 0:23)
user <- c("Subscriber", "Customer")
Convert the columns into factors.
bike$month <- factor(months[bike$month],
levels = months,
ordered = TRUE)
bike$hour <- factor(sprintf("%02d:00", bike$hour),
levels = hours,
ordered = TRUE)
bike$day_of_week <- factor(bike$day_of_week,
levels = days,
ordered = TRUE)
bike$user_type <- factor(bike$user_type,
levels = user)
sapply(bike[, (ncol(bike)-3):ncol(bike)], class)
## $month
## [1] "ordered" "factor"
##
## $hour
## [1] "ordered" "factor"
##
## $day_of_week
## [1] "ordered" "factor"
##
## $user_type
## [1] "factor"