library(rmarkdown)
library(explore)
library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.1.4 ✔ readr 2.1.5
## ✔ forcats 1.0.0 ✔ stringr 1.5.1
## ✔ ggplot2 3.5.1 ✔ tibble 3.2.1
## ✔ lubridate 1.9.3 ✔ tidyr 1.3.1
## ✔ purrr 1.0.2
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
library(readr)
train_c <- read_csv("train_c.csv")
## Rows: 8693 Columns: 16
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (5): PassengerId, HomePlanet, Destination, deck, side
## dbl (8): Age, RoomService, FoodCourt, ShoppingMall, Spa, VRDeck, withgroup, ...
## lgl (3): CryoSleep, VIP, Transported
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
library(readr)
test_c <- read_csv("test_c.csv")
## Rows: 8693 Columns: 17
## ── Column specification ────────────────────────────────────────────────────────
## Delimiter: ","
## chr (6): PassengerId, HomePlanet, Destination, deck, side, destination
## dbl (8): Age, RoomService, FoodCourt, ShoppingMall, Spa, VRDeck, withgroup, ...
## lgl (3): CryoSleep, VIP, Transported
##
## ℹ Use `spec()` to retrieve the full column specification for this data.
## ℹ Specify the column types or set `show_col_types = FALSE` to quiet this message.
describe_all(train_c)
## # A tibble: 16 × 8
## variable type na na_pct unique min mean max
## <chr> <chr> <int> <dbl> <int> <dbl> <dbl> <dbl>
## 1 PassengerId chr 0 0 8693 NA NA NA
## 2 HomePlanet chr 0 0 3 NA NA NA
## 3 CryoSleep lgl 0 0 2 0 0.36 1
## 4 Destination chr 0 0 3 NA NA NA
## 5 Age dbl 0 0 88 0 28.8 79
## 6 VIP lgl 0 0 2 0 0.02 1
## 7 RoomService dbl 0 0 1273 0 220. 14327
## 8 FoodCourt dbl 0 0 1507 0 448. 29813
## 9 ShoppingMall dbl 0 0 1115 0 170. 23492
## 10 Spa dbl 0 0 1327 0 305. 22408
## 11 VRDeck dbl 0 0 1306 0 298. 24133
## 12 Transported lgl 0 0 2 0 0.5 1
## 13 withgroup dbl 0 0 2 0 0.45 1
## 14 deck chr 0 0 8 NA NA NA
## 15 side chr 0 0 2 NA NA NA
## 16 expense dbl 0 0 2336 0 1441. 35987
describe_all(train_c)
## # A tibble: 16 × 8
## variable type na na_pct unique min mean max
## <chr> <chr> <int> <dbl> <int> <dbl> <dbl> <dbl>
## 1 PassengerId chr 0 0 8693 NA NA NA
## 2 HomePlanet chr 0 0 3 NA NA NA
## 3 CryoSleep lgl 0 0 2 0 0.36 1
## 4 Destination chr 0 0 3 NA NA NA
## 5 Age dbl 0 0 88 0 28.8 79
## 6 VIP lgl 0 0 2 0 0.02 1
## 7 RoomService dbl 0 0 1273 0 220. 14327
## 8 FoodCourt dbl 0 0 1507 0 448. 29813
## 9 ShoppingMall dbl 0 0 1115 0 170. 23492
## 10 Spa dbl 0 0 1327 0 305. 22408
## 11 VRDeck dbl 0 0 1306 0 298. 24133
## 12 Transported lgl 0 0 2 0 0.5 1
## 13 withgroup dbl 0 0 2 0 0.45 1
## 14 deck chr 0 0 8 NA NA NA
## 15 side chr 0 0 2 NA NA NA
## 16 expense dbl 0 0 2336 0 1441. 35987
describe_all(test_c)
## # A tibble: 17 × 8
## variable type na na_pct unique min mean max
## <chr> <chr> <int> <dbl> <int> <dbl> <dbl> <dbl>
## 1 PassengerId chr 0 0 8693 NA NA NA
## 2 HomePlanet chr 0 0 3 NA NA NA
## 3 CryoSleep lgl 0 0 2 0 0.36 1
## 4 Destination chr 0 0 3 NA NA NA
## 5 Age dbl 0 0 88 0 28.8 79
## 6 VIP lgl 0 0 2 0 0.02 1
## 7 RoomService dbl 0 0 1273 0 220. 14327
## 8 FoodCourt dbl 0 0 1507 0 448. 29813
## 9 ShoppingMall dbl 0 0 1115 0 170. 23492
## 10 Spa dbl 0 0 1327 0 305. 22408
## 11 VRDeck dbl 0 0 1306 0 298. 24133
## 12 Transported lgl 0 0 2 0 0.5 1
## 13 withgroup dbl 0 0 2 0 0.28 1
## 14 deck chr 0 0 8 NA NA NA
## 15 side chr 199 2.3 3 NA NA NA
## 16 destination chr 0 0 3 NA NA NA
## 17 expense dbl 0 0 2336 0 1441. 35987
train_c$HomePlanet <- as.factor(train_c$HomePlanet)
train_c$Destination <- as.factor(train_c$Destination)
train_c$deck <- as.factor(train_c$deck)
train_c$side <- as.factor(train_c$side)
test_c$HomePlanet <- as.factor(test_c$HomePlanet)
test_c$Destination <- as.factor(train_c$Destination)
test_c$deck <- as.factor(test_c$deck)
test_c$side <- as.factor(test_c$side)
library(DataExplorer)
create_report(train_c)
##
##
## processing file: report.rmd
## | | | 0% | |. | 2% | |.. | 5% [global_options] | |... | 7% | |.... | 10% [introduce] | |.... | 12% | |..... | 14% [plot_intro] | |...... | 17% | |....... | 19% [data_structure] | |........ | 21% | |......... | 24% [missing_profile] | |.......... | 26% | |........... | 29% [univariate_distribution_header] | |........... | 31% | |............ | 33% [plot_histogram] | |............. | 36% | |.............. | 38% [plot_density] | |............... | 40% | |................ | 43% [plot_frequency_bar] | |................. | 45% | |.................. | 48% [plot_response_bar] | |.................. | 50% | |................... | 52% [plot_with_bar] | |.................... | 55% | |..................... | 57% [plot_normal_qq] | |...................... | 60% | |....................... | 62% [plot_response_qq] | |........................ | 64% | |......................... | 67% [plot_by_qq] | |.......................... | 69% | |.......................... | 71% [correlation_analysis] | |........................... | 74% | |............................ | 76% [principal_component_analysis] | |............................. | 79% | |.............................. | 81% [bivariate_distribution_header] | |............................... | 83% | |................................ | 86% [plot_response_boxplot] | |................................. | 88% | |................................. | 90% [plot_by_boxplot] | |.................................. | 93% | |................................... | 95% [plot_response_scatterplot] | |.................................... | 98% | |.....................................| 100% [plot_by_scatterplot]
## output file: C:/Users/Lenovo/OneDrive/Documents/SON PROJE/report.knit.md
## "C:/Program Files/RStudio/resources/app/bin/quarto/bin/tools/pandoc" +RTS -K512m -RTS "C:\Users\Lenovo\OneDrive\DOCUME~1\SONPRO~1\REPORT~1.MD" --to html4 --from markdown+autolink_bare_uris+tex_math_single_backslash --output pandoc54a4670f1bc0.html --lua-filter "C:\Users\Lenovo\AppData\Local\R\cache\R\renv\cache\v5\R-4.3\x86_64-w64-mingw32\rmarkdown\2.27\27f9502e1cdbfa195f94e03b0f517484\rmarkdown\rmarkdown\lua\pagebreak.lua" --lua-filter "C:\Users\Lenovo\AppData\Local\R\cache\R\renv\cache\v5\R-4.3\x86_64-w64-mingw32\rmarkdown\2.27\27f9502e1cdbfa195f94e03b0f517484\rmarkdown\rmarkdown\lua\latex-div.lua" --embed-resources --standalone --variable bs3=TRUE --section-divs --table-of-contents --toc-depth 6 --template "C:\Users\Lenovo\AppData\Local\R\cache\R\renv\cache\v5\R-4.3\x86_64-w64-mingw32\rmarkdown\2.27\27f9502e1cdbfa195f94e03b0f517484\rmarkdown\rmd\h\default.html" --no-highlight --variable highlightjs=1 --variable theme=yeti --mathjax --variable "mathjax-url=https://mathjax.rstudio.com/latest/MathJax.js?config=TeX-AMS-MML_HTMLorMML" --include-in-header "C:\Users\Lenovo\AppData\Local\Temp\Rtmp2ZLE7G\rmarkdown-str54a4326718ac.html"
##
## Output created: report.html
model <- lm(Transported ~ . , data = train_c[, 2:16])
summary(model)
##
## Call:
## lm(formula = Transported ~ ., data = train_c[, 2:16])
##
## Residuals:
## Min 1Q Median 3Q Max
## -1.48097 -0.30495 -0.02786 0.28279 1.75905
##
## Coefficients: (1 not defined because of singularities)
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 3.046e-01 4.003e-02 7.609 3.05e-14 ***
## HomePlanetEuropa 2.110e-01 2.864e-02 7.367 1.91e-13 ***
## HomePlanetMars 8.968e-02 1.479e-02 6.062 1.40e-09 ***
## CryoSleepTRUE 3.901e-01 1.158e-02 33.680 < 2e-16 ***
## DestinationPSO J318.5-22 -4.323e-02 1.802e-02 -2.400 0.016438 *
## DestinationTRAPPIST-1e -4.570e-02 1.128e-02 -4.051 5.14e-05 ***
## Age -2.217e-03 3.192e-04 -6.947 4.00e-12 ***
## VIPTRUE -3.367e-02 2.967e-02 -1.135 0.256466
## RoomService -1.147e-04 7.050e-06 -16.269 < 2e-16 ***
## FoodCourt 4.403e-05 3.053e-06 14.422 < 2e-16 ***
## ShoppingMall 8.240e-05 7.448e-06 11.063 < 2e-16 ***
## Spa -8.531e-05 4.106e-06 -20.777 < 2e-16 ***
## VRDeck -8.127e-05 4.109e-06 -19.776 < 2e-16 ***
## withgroup 1.999e-02 9.260e-03 2.159 0.030867 *
## deckB 1.076e-01 2.866e-02 3.755 0.000175 ***
## deckC 1.442e-01 2.893e-02 4.986 6.28e-07 ***
## deckD 5.455e-02 3.462e-02 1.575 0.115193
## deckE 1.033e-02 3.602e-02 0.287 0.774351
## deckF 1.074e-01 3.693e-02 2.907 0.003655 **
## deckG 5.576e-02 3.855e-02 1.446 0.148098
## deckT 6.774e-02 1.810e-01 0.374 0.708306
## sideS 8.559e-02 8.606e-03 9.946 < 2e-16 ***
## expense NA NA NA NA
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 0.4001 on 8671 degrees of freedom
## Multiple R-squared: 0.3612, Adjusted R-squared: 0.3596
## F-statistic: 233.4 on 21 and 8671 DF, p-value: < 2.2e-16
library(caTools)
set.seed(123)
split = sample.split(train_c$Transported, SplitRatio = 0.75)
train_train = subset(train_c, split == TRUE)
train_test = subset(train_c, split == FALSE)
regresyon <- lm(Transported ~ . ,data = train_train[, -c(1)])
summary(regresyon)
##
## Call:
## lm(formula = Transported ~ ., data = train_train[, -c(1)])
##
## Residuals:
## Min 1Q Median 3Q Max
## -1.55277 -0.30795 -0.02772 0.28588 1.74204
##
## Coefficients: (1 not defined because of singularities)
## Estimate Std. Error t value Pr(>|t|)
## (Intercept) 3.373e-01 4.627e-02 7.290 3.46e-13 ***
## HomePlanetEuropa 1.979e-01 3.338e-02 5.927 3.25e-09 ***
## HomePlanetMars 8.044e-02 1.706e-02 4.715 2.46e-06 ***
## CryoSleepTRUE 3.849e-01 1.335e-02 28.834 < 2e-16 ***
## DestinationPSO J318.5-22 -6.053e-02 2.076e-02 -2.915 0.003563 **
## DestinationTRAPPIST-1e -5.053e-02 1.304e-02 -3.876 0.000107 ***
## Age -2.382e-03 3.714e-04 -6.414 1.52e-10 ***
## VIPTRUE -1.268e-02 3.376e-02 -0.375 0.707329
## RoomService -1.147e-04 7.788e-06 -14.729 < 2e-16 ***
## FoodCourt 4.005e-05 3.466e-06 11.555 < 2e-16 ***
## ShoppingMall 8.546e-05 8.407e-06 10.166 < 2e-16 ***
## Spa -8.430e-05 4.704e-06 -17.921 < 2e-16 ***
## VRDeck -8.077e-05 4.771e-06 -16.930 < 2e-16 ***
## withgroup 1.620e-02 1.073e-02 1.510 0.130979
## deckB 1.003e-01 3.283e-02 3.057 0.002248 **
## deckC 1.388e-01 3.311e-02 4.193 2.78e-05 ***
## deckD 3.548e-02 3.952e-02 0.898 0.369332
## deckE -1.945e-02 4.186e-02 -0.465 0.642212
## deckF 8.940e-02 4.267e-02 2.095 0.036188 *
## deckG 4.025e-02 4.456e-02 0.903 0.366367
## deckT 5.872e-02 1.818e-01 0.323 0.746676
## sideS 8.994e-02 9.958e-03 9.033 < 2e-16 ***
## expense NA NA NA NA
## ---
## Signif. codes: 0 '***' 0.001 '**' 0.01 '*' 0.05 '.' 0.1 ' ' 1
##
## Residual standard error: 0.4003 on 6498 degrees of freedom
## Multiple R-squared: 0.3611, Adjusted R-squared: 0.3591
## F-statistic: 174.9 on 21 and 6498 DF, p-value: < 2.2e-16
reg_tahmin = predict(regresyon, newdata = train_test[, -c(1,12)])
reg_transported_tahmin <- ifelse(reg_tahmin > 0.5, 1, 0)