#Menginput data
library(ggplot2)
## Warning: package 'ggplot2' was built under R version 4.5.3
library(dplyr)
## Warning: package 'dplyr' was built under R version 4.5.3
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
library(gridExtra)
## Warning: package 'gridExtra' was built under R version 4.5.3
## 
## Attaching package: 'gridExtra'
## The following object is masked from 'package:dplyr':
## 
##     combine
data_ckd <- read.delim(file.choose(), header = TRUE, sep = "\t", stringsAsFactors = FALSE)
data
## function (..., list = character(), package = NULL, lib.loc = NULL, 
##     verbose = getOption("verbose"), envir = .GlobalEnv, overwrite = TRUE) 
## {
##     fileExt <- function(x) {
##         db <- grepl("\\.[^.]+\\.(gz|bz2|xz)$", x)
##         ans <- sub(".*\\.", "", x)
##         ans[db] <- sub(".*\\.([^.]+\\.)(gz|bz2|xz)$", "\\1\\2", 
##             x[db])
##         ans
##     }
##     my_read_table <- function(...) {
##         lcc <- Sys.getlocale("LC_COLLATE")
##         on.exit(Sys.setlocale("LC_COLLATE", lcc))
##         Sys.setlocale("LC_COLLATE", "C")
##         read.table(...)
##     }
##     stopifnot(is.character(list))
##     names <- c(as.character(substitute(list(...))[-1L]), list)
##     if (!is.null(package)) {
##         if (!is.character(package)) 
##             stop("'package' must be a character vector or NULL")
##     }
##     paths <- find.package(package, lib.loc, verbose = verbose)
##     if (is.null(lib.loc)) 
##         paths <- c(path.package(package, TRUE), if (!length(package)) getwd(), 
##             paths)
##     paths <- unique(normalizePath(paths[file.exists(paths)]))
##     paths <- paths[dir.exists(file.path(paths, "data"))]
##     dataExts <- tools:::.make_file_exts("data")
##     if (length(names) == 0L) {
##         db <- matrix(character(), nrow = 0L, ncol = 4L)
##         for (path in paths) {
##             entries <- NULL
##             packageName <- if (file_test("-f", file.path(path, 
##                 "DESCRIPTION"))) 
##                 basename(path)
##             else "."
##             if (file_test("-f", INDEX <- file.path(path, "Meta", 
##                 "data.rds"))) {
##                 entries <- readRDS(INDEX)
##             }
##             else {
##                 dataDir <- file.path(path, "data")
##                 entries <- tools::list_files_with_type(dataDir, 
##                   "data")
##                 if (length(entries)) {
##                   entries <- unique(tools::file_path_sans_ext(basename(entries)))
##                   entries <- cbind(entries, "")
##                 }
##             }
##             if (NROW(entries)) {
##                 if (is.matrix(entries) && ncol(entries) == 2L) 
##                   db <- rbind(db, cbind(packageName, dirname(path), 
##                     entries))
##                 else warning(gettextf("data index for package %s is invalid and will be ignored", 
##                   sQuote(packageName)), domain = NA, call. = FALSE)
##             }
##         }
##         colnames(db) <- c("Package", "LibPath", "Item", "Title")
##         footer <- if (missing(package)) 
##             paste0("Use ", sQuote(paste("data(package =", ".packages(all.available = TRUE))")), 
##                 "\n", "to list the data sets in all *available* packages.")
##         else NULL
##         y <- list(title = "Data sets", header = NULL, results = db, 
##             footer = footer)
##         class(y) <- "packageIQR"
##         return(y)
##     }
##     paths <- file.path(paths, "data")
##     for (name in names) {
##         found <- FALSE
##         for (p in paths) {
##             tmp_env <- if (overwrite) 
##                 envir
##             else new.env()
##             if (file_test("-f", file.path(p, "Rdata.rds"))) {
##                 rds <- readRDS(file.path(p, "Rdata.rds"))
##                 if (name %in% names(rds)) {
##                   found <- TRUE
##                   if (verbose) 
##                     message(sprintf("name=%s:\t found in Rdata.rds", 
##                       name), domain = NA)
##                   objs <- rds[[name]]
##                   lazyLoad(file.path(p, "Rdata"), envir = tmp_env, 
##                     filter = function(x) x %in% objs)
##                   break
##                 }
##                 else if (verbose) 
##                   message(sprintf("name=%s:\t NOT found in names() of Rdata.rds, i.e.,\n\t%s\n", 
##                     name, paste(names(rds), collapse = ",")), 
##                     domain = NA)
##             }
##             files <- list.files(p, full.names = TRUE)
##             files <- files[grep(name, files, fixed = TRUE)]
##             if (length(files) > 1L) {
##                 o <- match(fileExt(files), dataExts, nomatch = 100L)
##                 paths0 <- dirname(files)
##                 paths0 <- factor(paths0, levels = unique(paths0))
##                 files <- files[order(paths0, o)]
##             }
##             if (length(files)) {
##                 for (file in files) {
##                   if (verbose) 
##                     message("name=", name, ":\t file= ...", .Platform$file.sep, 
##                       basename(file), "::\t", appendLF = FALSE, 
##                       domain = NA)
##                   ext <- fileExt(file)
##                   if (basename(file) != paste0(name, ".", ext)) 
##                     found <- FALSE
##                   else {
##                     found <- TRUE
##                     switch(ext, R = , r = {
##                       library("utils")
##                       sys.source(file, chdir = TRUE, envir = tmp_env)
##                     }, RData = , rdata = , rda = load(file, envir = tmp_env), 
##                       TXT = , txt = , tab = , tab.gz = , tab.bz2 = , 
##                       tab.xz = , txt.gz = , txt.bz2 = , txt.xz = assign(name, 
##                         my_read_table(file, header = TRUE, as.is = FALSE), 
##                         envir = tmp_env), CSV = , csv = , csv.gz = , 
##                       csv.bz2 = , csv.xz = assign(name, my_read_table(file, 
##                         header = TRUE, sep = ";", as.is = FALSE), 
##                         envir = tmp_env), found <- FALSE)
##                   }
##                   if (found) 
##                     break
##                 }
##                 if (verbose) 
##                   message(if (!found) 
##                     "*NOT* ", "found", domain = NA)
##             }
##             if (found) 
##                 break
##         }
##         if (!found) {
##             warning(gettextf("data set %s not found", sQuote(name)), 
##                 domain = NA)
##         }
##         else if (!overwrite) {
##             for (o in ls(envir = tmp_env, all.names = TRUE)) {
##                 if (exists(o, envir = envir, inherits = FALSE)) 
##                   warning(gettextf("an object named %s already exists and will not be overwritten", 
##                     sQuote(o)))
##                 else assign(o, get(o, envir = tmp_env, inherits = FALSE), 
##                   envir = envir)
##             }
##             rm(tmp_env)
##         }
##     }
##     invisible(names)
## }
## <bytecode: 0x0000023b6b70fe38>
## <environment: namespace:utils>
head(data_ckd)
##   Age.of.the.patient Blood.pressure..mm.Hg. Red.blood.cells.in.urine
## 1                 54                    167                   normal
## 2                 42                    127                   normal
## 3                 38                    148                 abnormal
## 4                 67                    174                   normal
## 5                 67                    100                   normal
## 6                 42                    138                 abnormal
##   Bacteria.in.urine Blood.urea..mg.dl. Serum.creatinine..mg.dl.
## 1       not present           169.1014                     7.55
## 2           present           183.2235                    13.37
## 3       not present           193.1417                     9.49
## 4       not present           197.1886                     3.01
## 5           present            27.3082                     8.73
## 6           present           169.6566                     6.92
##   Hemoglobin.level..gms. Hypertension..yes.no. Diabetes.mellitus..yes.no.
## 1                   11.8                   yes                        yes
## 2                    8.2                    no                        yes
## 3                   10.1                    no                         no
## 4                   16.1                    no                         no
## 5                    7.7                    no                         no
## 6                    8.8                   yes                         no
##   Coronary.artery.disease..yes.no. Anemia..yes.no.
## 1                               no              no
## 2                               no             yes
## 3                              yes              no
## 4                               no             yes
## 5                              yes             yes
## 6                               no             yes
##   Urine.protein.to.creatinine.ratio Urine.output..ml.day. Cholesterol.level
## 1                              2.51                  1397               152
## 2                              4.27                  1632               242
## 3                              1.56                   889               103
## 4                              1.23                   893               149
## 5                              2.61                  1212               182
## 6                              2.42                  2720               260
##   Family.history.of.chronic.kidney.disease Smoking.status Body.Mass.Index..BMI.
## 1                                       no            yes                  25.3
## 2                                      yes             no                  20.6
## 3                                       no             no                  38.4
## 4                                       no            yes                  17.6
## 5                                      yes            yes                  37.2
## 6                                       no            yes                  23.5
##   Physical.activity.level CKD.Status
## 1                     low        Yes
## 2                moderate        Yes
## 3                    high         No
## 4                    high        Yes
## 5                moderate         No
## 6                moderate         No
str(data_ckd)
## 'data.frame':    50 obs. of  19 variables:
##  $ Age.of.the.patient                      : int  54 42 38 67 67 42 84 49 65 25 ...
##  $ Blood.pressure..mm.Hg.                  : int  167 127 148 174 100 138 104 113 126 174 ...
##  $ Red.blood.cells.in.urine                : chr  "normal" "normal" "abnormal" "normal" ...
##  $ Bacteria.in.urine                       : chr  "not present" "present" "not present" "not present" ...
##  $ Blood.urea..mg.dl.                      : num  169.1 183.2 193.1 197.2 27.3 ...
##  $ Serum.creatinine..mg.dl.                : num  7.55 13.37 9.49 3.01 8.73 ...
##  $ Hemoglobin.level..gms.                  : num  11.8 8.2 10.1 16.1 7.7 8.8 7.2 12.6 7.5 15.6 ...
##  $ Hypertension..yes.no.                   : chr  "yes" "no" "no" "no" ...
##  $ Diabetes.mellitus..yes.no.              : chr  "yes" "yes" "no" "no" ...
##  $ Coronary.artery.disease..yes.no.        : chr  "no" "no" "yes" "no" ...
##  $ Anemia..yes.no.                         : chr  "no" "yes" "no" "yes" ...
##  $ Urine.protein.to.creatinine.ratio       : num  2.51 4.27 1.56 1.23 2.61 2.42 1.67 0.41 2.8 0.46 ...
##  $ Urine.output..ml.day.                   : int  1397 1632 889 893 1212 2720 2859 2076 862 663 ...
##  $ Cholesterol.level                       : int  152 242 103 149 182 260 134 229 276 206 ...
##  $ Family.history.of.chronic.kidney.disease: chr  "no" "yes" "no" "no" ...
##  $ Smoking.status                          : chr  "yes" "no" "no" "yes" ...
##  $ Body.Mass.Index..BMI.                   : num  25.3 20.6 38.4 17.6 37.2 23.5 23 30.6 16.8 28.5 ...
##  $ Physical.activity.level                 : chr  "low" "moderate" "high" "high" ...
##  $ CKD.Status                              : chr  "Yes" "Yes" "No" "Yes" ...
dim(data_ckd)
## [1] 50 19
colnames(data_ckd)
##  [1] "Age.of.the.patient"                      
##  [2] "Blood.pressure..mm.Hg."                  
##  [3] "Red.blood.cells.in.urine"                
##  [4] "Bacteria.in.urine"                       
##  [5] "Blood.urea..mg.dl."                      
##  [6] "Serum.creatinine..mg.dl."                
##  [7] "Hemoglobin.level..gms."                  
##  [8] "Hypertension..yes.no."                   
##  [9] "Diabetes.mellitus..yes.no."              
## [10] "Coronary.artery.disease..yes.no."        
## [11] "Anemia..yes.no."                         
## [12] "Urine.protein.to.creatinine.ratio"       
## [13] "Urine.output..ml.day."                   
## [14] "Cholesterol.level"                       
## [15] "Family.history.of.chronic.kidney.disease"
## [16] "Smoking.status"                          
## [17] "Body.Mass.Index..BMI."                   
## [18] "Physical.activity.level"                 
## [19] "CKD.Status"
#3. Visualisasi Data
library(ggplot2)
library(gridExtra)

# 1. Visualisasi Distribusi Variabel Numerik (Histogram & Density)
p1 <- ggplot(data_ckd, aes(x = Age.of.the.patient)) + 
  geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") + 
  geom_density(color = "red", size = 1) + 
  labs(title = "Age of Patient", x = "Age", y = "Density") + theme_minimal()
## Warning: Using `size` aesthetic for lines was deprecated in ggplot2 3.4.0.
## ℹ Please use `linewidth` instead.
## This warning is displayed once per session.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.
p2 <- ggplot(data_ckd, aes(x = Blood.pressure..mm.Hg.)) + 
  geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") + 
  geom_density(color = "red", size = 1) + 
  labs(title = "Blood Pressure", x = "BP (mm/Hg)", y = "Density") + theme_minimal()

p3 <- ggplot(data_ckd, aes(x = Serum.creatinine..mg.dl.)) + 
  geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") + 
  geom_density(color = "red", size = 1) + 
  labs(title = "Serum Creatinine", x = "Creatinine (mg/dl)", y = "Density") + theme_minimal()

p4 <- ggplot(data_ckd, aes(x = Body.Mass.Index..BMI.)) + 
  geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") + 
  geom_density(color = "red", size = 1) + 
  labs(title = "BMI", x = "BMI", y = "Density") + theme_minimal()

grid.arrange(p1, p2, p3, p4, ncol = 2, nrow = 2)
## Warning: The dot-dot notation (`..density..`) was deprecated in ggplot2 3.4.0.
## ℹ Please use `after_stat(density)` instead.
## This warning is displayed once per session.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.

# 2. Visualisasi Hubungan CKD Status dan Hipertensi (Bar Plot)
ggplot(data_ckd, aes(x = CKD.Status, fill = Hypertension..yes.no.)) +
  geom_bar(position = "dodge") +
  scale_fill_manual(values = c("no" = "#F8766D", "yes" = "#00BFC4")) +
  labs(title = "CKD Status berdasarkan Hipertensi",
       x = "CKD Status",
       y = "Jumlah Pasien",
       fill = "Hypertension") +
  theme_minimal()

#4. Analisis Statistika Deskriptif
library(dplyr)
numeric_vars <- data_ckd %>% select(where(is.numeric))

deskriptif_summary <- data.frame(
  Mean = apply(numeric_vars, 2, mean, na.rm = TRUE),
  SD = apply(numeric_vars, 2, sd, na.rm = TRUE),
  Median = apply(numeric_vars, 2, median, na.rm = TRUE),
  Min = apply(numeric_vars, 2, min, na.rm = TRUE),
  Max = apply(numeric_vars, 2, max, na.rm = TRUE)
)

print("Statistik Variabel Numerik")
## [1] "Statistik Variabel Numerik"
print(round(deskriptif_summary, 2))
##                                      Mean     SD  Median    Min     Max
## Age.of.the.patient                  56.90  20.08   54.50  25.00   90.00
## Blood.pressure..mm.Hg.             131.90  29.45  127.50  80.00  179.00
## Blood.urea..mg.dl.                 101.27  58.01   97.62   8.69  198.73
## Serum.creatinine..mg.dl.             7.61   3.88    8.17   0.76   14.58
## Hemoglobin.level..gms.              12.09   3.37   11.85   6.00   17.50
## Urine.protein.to.creatinine.ratio    2.30   1.23    2.38   0.24    4.49
## Urine.output..ml.day.             1739.52 802.09 1740.50 308.00 2933.00
## Cholesterol.level                  199.52  66.92  214.00 100.00  298.00
## Body.Mass.Index..BMI.               26.33   7.64   25.10  15.20   39.00
# B. Frekuensi Variabel Kategorik Utama
print("FREKUENSI CKD STATUS")
## [1] "FREKUENSI CKD STATUS"
print(table(data_ckd$CKD.Status))
## 
##  No Yes 
##  31  19
print("FREKUENSI ANEMIA")
## [1] "FREKUENSI ANEMIA"
print(table(data_ckd$Anemia..yes.no.))
## 
##  no yes 
##  17  33
print("FREKUENSI HIPERTENSI")
## [1] "FREKUENSI HIPERTENSI"
print(table(data_ckd$Hypertension..yes.no.))
## 
##  no yes 
##  32  18
#5. UJI DISTRIBUSI & BENTUK DISTRIBUSI
shapiro_table <- data.frame(
  W_Statistic = numeric(),
  p_value = numeric(),
  Keputusan = character(),
  stringsAsFactors = FALSE
)

for (var in colnames(numeric_vars)) {
  test <- shapiro.test(numeric_vars[[var]])
  p_val <- test$p.value
  keputusan <- ifelse(p_val >= 0.05, "Berdistribusi Normal", "Tidak Normal")
  shapiro_table[var, ] <- c(
    round(test$statistic, 4),
    round(p_val, 4),
    keputusan
  )
}
print("HASIL UJI NORMALITAS SHAPIRO-WILK")
## [1] "HASIL UJI NORMALITAS SHAPIRO-WILK"
print(shapiro_table)
##                                   W_Statistic p_value            Keputusan
## Age.of.the.patient                     0.9476  0.0271         Tidak Normal
## Blood.pressure..mm.Hg.                 0.9464  0.0243         Tidak Normal
## Blood.urea..mg.dl.                     0.9354  0.0089         Tidak Normal
## Serum.creatinine..mg.dl.               0.9572  0.0677 Berdistribusi Normal
## Hemoglobin.level..gms.                 0.9484  0.0293         Tidak Normal
## Urine.protein.to.creatinine.ratio      0.9586  0.0777 Berdistribusi Normal
## Urine.output..ml.day.                  0.9294  0.0053         Tidak Normal
## Cholesterol.level                       0.901   5e-04         Tidak Normal
## Body.Mass.Index..BMI.                  0.9186  0.0021         Tidak Normal
#6.  HEATMAP & TABEL KONTINGENSI
library(ggplot2)
library(reshape2)
## Warning: package 'reshape2' was built under R version 4.5.3
# A. MEMBUAT TABEL KONTINGENSI (CKD Status vs Hipertensi)
tabel_kontingensi <- table(data_ckd$CKD.Status, data_ckd$Hypertension..yes.no.)
tabel_kontingensi_total <- addmargins(tabel_kontingensi)

print("TABEL KONTINGENSI (CKD Status vs Hipertensi)")
## [1] "TABEL KONTINGENSI (CKD Status vs Hipertensi)"
print(tabel_kontingensi_total)
##      
##       no yes Sum
##   No  21  10  31
##   Yes 11   8  19
##   Sum 32  18  50
# B. MEMBUAT HEATMAP KORELASI VARIABEL NUMERIK
numeric_cols <- data_ckd[, sapply(data_ckd, is.numeric)]
matriks_korelasi <- cor(numeric_cols, use = "complete.obs")
melted_cormat <- melt(matriks_korelasi)
ggplot(data = melted_cormat, aes(x = Var1, y = Var2, fill = value)) +
  geom_tile(color = "white") +
  scale_fill_gradient2(low = "#4575b4", mid = "white", high = "#d73027", 
                       midpoint = 0, limit = c(-1,1), name = "Korelasi") +
  geom_text(aes(label = round(value, 2)), color = "black", size = 3) +
  theme_minimal() + 
  theme(axis.text.x = element_text(angle = 45, vjust = 1, hjust = 1)) +
  labs(title = "Heatmap Korelasi Variabel Numerik", x = "", y = "")