#Menginput data
library(ggplot2)
## Warning: package 'ggplot2' was built under R version 4.5.3
library(dplyr)
## Warning: package 'dplyr' was built under R version 4.5.3
##
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
##
## filter, lag
## The following objects are masked from 'package:base':
##
## intersect, setdiff, setequal, union
library(gridExtra)
## Warning: package 'gridExtra' was built under R version 4.5.3
##
## Attaching package: 'gridExtra'
## The following object is masked from 'package:dplyr':
##
## combine
data_ckd <- read.delim(file.choose(), header = TRUE, sep = "\t", stringsAsFactors = FALSE)
data
## function (..., list = character(), package = NULL, lib.loc = NULL,
## verbose = getOption("verbose"), envir = .GlobalEnv, overwrite = TRUE)
## {
## fileExt <- function(x) {
## db <- grepl("\\.[^.]+\\.(gz|bz2|xz)$", x)
## ans <- sub(".*\\.", "", x)
## ans[db] <- sub(".*\\.([^.]+\\.)(gz|bz2|xz)$", "\\1\\2",
## x[db])
## ans
## }
## my_read_table <- function(...) {
## lcc <- Sys.getlocale("LC_COLLATE")
## on.exit(Sys.setlocale("LC_COLLATE", lcc))
## Sys.setlocale("LC_COLLATE", "C")
## read.table(...)
## }
## stopifnot(is.character(list))
## names <- c(as.character(substitute(list(...))[-1L]), list)
## if (!is.null(package)) {
## if (!is.character(package))
## stop("'package' must be a character vector or NULL")
## }
## paths <- find.package(package, lib.loc, verbose = verbose)
## if (is.null(lib.loc))
## paths <- c(path.package(package, TRUE), if (!length(package)) getwd(),
## paths)
## paths <- unique(normalizePath(paths[file.exists(paths)]))
## paths <- paths[dir.exists(file.path(paths, "data"))]
## dataExts <- tools:::.make_file_exts("data")
## if (length(names) == 0L) {
## db <- matrix(character(), nrow = 0L, ncol = 4L)
## for (path in paths) {
## entries <- NULL
## packageName <- if (file_test("-f", file.path(path,
## "DESCRIPTION")))
## basename(path)
## else "."
## if (file_test("-f", INDEX <- file.path(path, "Meta",
## "data.rds"))) {
## entries <- readRDS(INDEX)
## }
## else {
## dataDir <- file.path(path, "data")
## entries <- tools::list_files_with_type(dataDir,
## "data")
## if (length(entries)) {
## entries <- unique(tools::file_path_sans_ext(basename(entries)))
## entries <- cbind(entries, "")
## }
## }
## if (NROW(entries)) {
## if (is.matrix(entries) && ncol(entries) == 2L)
## db <- rbind(db, cbind(packageName, dirname(path),
## entries))
## else warning(gettextf("data index for package %s is invalid and will be ignored",
## sQuote(packageName)), domain = NA, call. = FALSE)
## }
## }
## colnames(db) <- c("Package", "LibPath", "Item", "Title")
## footer <- if (missing(package))
## paste0("Use ", sQuote(paste("data(package =", ".packages(all.available = TRUE))")),
## "\n", "to list the data sets in all *available* packages.")
## else NULL
## y <- list(title = "Data sets", header = NULL, results = db,
## footer = footer)
## class(y) <- "packageIQR"
## return(y)
## }
## paths <- file.path(paths, "data")
## for (name in names) {
## found <- FALSE
## for (p in paths) {
## tmp_env <- if (overwrite)
## envir
## else new.env()
## if (file_test("-f", file.path(p, "Rdata.rds"))) {
## rds <- readRDS(file.path(p, "Rdata.rds"))
## if (name %in% names(rds)) {
## found <- TRUE
## if (verbose)
## message(sprintf("name=%s:\t found in Rdata.rds",
## name), domain = NA)
## objs <- rds[[name]]
## lazyLoad(file.path(p, "Rdata"), envir = tmp_env,
## filter = function(x) x %in% objs)
## break
## }
## else if (verbose)
## message(sprintf("name=%s:\t NOT found in names() of Rdata.rds, i.e.,\n\t%s\n",
## name, paste(names(rds), collapse = ",")),
## domain = NA)
## }
## files <- list.files(p, full.names = TRUE)
## files <- files[grep(name, files, fixed = TRUE)]
## if (length(files) > 1L) {
## o <- match(fileExt(files), dataExts, nomatch = 100L)
## paths0 <- dirname(files)
## paths0 <- factor(paths0, levels = unique(paths0))
## files <- files[order(paths0, o)]
## }
## if (length(files)) {
## for (file in files) {
## if (verbose)
## message("name=", name, ":\t file= ...", .Platform$file.sep,
## basename(file), "::\t", appendLF = FALSE,
## domain = NA)
## ext <- fileExt(file)
## if (basename(file) != paste0(name, ".", ext))
## found <- FALSE
## else {
## found <- TRUE
## switch(ext, R = , r = {
## library("utils")
## sys.source(file, chdir = TRUE, envir = tmp_env)
## }, RData = , rdata = , rda = load(file, envir = tmp_env),
## TXT = , txt = , tab = , tab.gz = , tab.bz2 = ,
## tab.xz = , txt.gz = , txt.bz2 = , txt.xz = assign(name,
## my_read_table(file, header = TRUE, as.is = FALSE),
## envir = tmp_env), CSV = , csv = , csv.gz = ,
## csv.bz2 = , csv.xz = assign(name, my_read_table(file,
## header = TRUE, sep = ";", as.is = FALSE),
## envir = tmp_env), found <- FALSE)
## }
## if (found)
## break
## }
## if (verbose)
## message(if (!found)
## "*NOT* ", "found", domain = NA)
## }
## if (found)
## break
## }
## if (!found) {
## warning(gettextf("data set %s not found", sQuote(name)),
## domain = NA)
## }
## else if (!overwrite) {
## for (o in ls(envir = tmp_env, all.names = TRUE)) {
## if (exists(o, envir = envir, inherits = FALSE))
## warning(gettextf("an object named %s already exists and will not be overwritten",
## sQuote(o)))
## else assign(o, get(o, envir = tmp_env, inherits = FALSE),
## envir = envir)
## }
## rm(tmp_env)
## }
## }
## invisible(names)
## }
## <bytecode: 0x0000023b6b70fe38>
## <environment: namespace:utils>
head(data_ckd)
## Age.of.the.patient Blood.pressure..mm.Hg. Red.blood.cells.in.urine
## 1 54 167 normal
## 2 42 127 normal
## 3 38 148 abnormal
## 4 67 174 normal
## 5 67 100 normal
## 6 42 138 abnormal
## Bacteria.in.urine Blood.urea..mg.dl. Serum.creatinine..mg.dl.
## 1 not present 169.1014 7.55
## 2 present 183.2235 13.37
## 3 not present 193.1417 9.49
## 4 not present 197.1886 3.01
## 5 present 27.3082 8.73
## 6 present 169.6566 6.92
## Hemoglobin.level..gms. Hypertension..yes.no. Diabetes.mellitus..yes.no.
## 1 11.8 yes yes
## 2 8.2 no yes
## 3 10.1 no no
## 4 16.1 no no
## 5 7.7 no no
## 6 8.8 yes no
## Coronary.artery.disease..yes.no. Anemia..yes.no.
## 1 no no
## 2 no yes
## 3 yes no
## 4 no yes
## 5 yes yes
## 6 no yes
## Urine.protein.to.creatinine.ratio Urine.output..ml.day. Cholesterol.level
## 1 2.51 1397 152
## 2 4.27 1632 242
## 3 1.56 889 103
## 4 1.23 893 149
## 5 2.61 1212 182
## 6 2.42 2720 260
## Family.history.of.chronic.kidney.disease Smoking.status Body.Mass.Index..BMI.
## 1 no yes 25.3
## 2 yes no 20.6
## 3 no no 38.4
## 4 no yes 17.6
## 5 yes yes 37.2
## 6 no yes 23.5
## Physical.activity.level CKD.Status
## 1 low Yes
## 2 moderate Yes
## 3 high No
## 4 high Yes
## 5 moderate No
## 6 moderate No
str(data_ckd)
## 'data.frame': 50 obs. of 19 variables:
## $ Age.of.the.patient : int 54 42 38 67 67 42 84 49 65 25 ...
## $ Blood.pressure..mm.Hg. : int 167 127 148 174 100 138 104 113 126 174 ...
## $ Red.blood.cells.in.urine : chr "normal" "normal" "abnormal" "normal" ...
## $ Bacteria.in.urine : chr "not present" "present" "not present" "not present" ...
## $ Blood.urea..mg.dl. : num 169.1 183.2 193.1 197.2 27.3 ...
## $ Serum.creatinine..mg.dl. : num 7.55 13.37 9.49 3.01 8.73 ...
## $ Hemoglobin.level..gms. : num 11.8 8.2 10.1 16.1 7.7 8.8 7.2 12.6 7.5 15.6 ...
## $ Hypertension..yes.no. : chr "yes" "no" "no" "no" ...
## $ Diabetes.mellitus..yes.no. : chr "yes" "yes" "no" "no" ...
## $ Coronary.artery.disease..yes.no. : chr "no" "no" "yes" "no" ...
## $ Anemia..yes.no. : chr "no" "yes" "no" "yes" ...
## $ Urine.protein.to.creatinine.ratio : num 2.51 4.27 1.56 1.23 2.61 2.42 1.67 0.41 2.8 0.46 ...
## $ Urine.output..ml.day. : int 1397 1632 889 893 1212 2720 2859 2076 862 663 ...
## $ Cholesterol.level : int 152 242 103 149 182 260 134 229 276 206 ...
## $ Family.history.of.chronic.kidney.disease: chr "no" "yes" "no" "no" ...
## $ Smoking.status : chr "yes" "no" "no" "yes" ...
## $ Body.Mass.Index..BMI. : num 25.3 20.6 38.4 17.6 37.2 23.5 23 30.6 16.8 28.5 ...
## $ Physical.activity.level : chr "low" "moderate" "high" "high" ...
## $ CKD.Status : chr "Yes" "Yes" "No" "Yes" ...
dim(data_ckd)
## [1] 50 19
colnames(data_ckd)
## [1] "Age.of.the.patient"
## [2] "Blood.pressure..mm.Hg."
## [3] "Red.blood.cells.in.urine"
## [4] "Bacteria.in.urine"
## [5] "Blood.urea..mg.dl."
## [6] "Serum.creatinine..mg.dl."
## [7] "Hemoglobin.level..gms."
## [8] "Hypertension..yes.no."
## [9] "Diabetes.mellitus..yes.no."
## [10] "Coronary.artery.disease..yes.no."
## [11] "Anemia..yes.no."
## [12] "Urine.protein.to.creatinine.ratio"
## [13] "Urine.output..ml.day."
## [14] "Cholesterol.level"
## [15] "Family.history.of.chronic.kidney.disease"
## [16] "Smoking.status"
## [17] "Body.Mass.Index..BMI."
## [18] "Physical.activity.level"
## [19] "CKD.Status"
#3. Visualisasi Data
library(ggplot2)
library(gridExtra)
# 1. Visualisasi Distribusi Variabel Numerik (Histogram & Density)
p1 <- ggplot(data_ckd, aes(x = Age.of.the.patient)) +
geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") +
geom_density(color = "red", size = 1) +
labs(title = "Age of Patient", x = "Age", y = "Density") + theme_minimal()
## Warning: Using `size` aesthetic for lines was deprecated in ggplot2 3.4.0.
## ℹ Please use `linewidth` instead.
## This warning is displayed once per session.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.
p2 <- ggplot(data_ckd, aes(x = Blood.pressure..mm.Hg.)) +
geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") +
geom_density(color = "red", size = 1) +
labs(title = "Blood Pressure", x = "BP (mm/Hg)", y = "Density") + theme_minimal()
p3 <- ggplot(data_ckd, aes(x = Serum.creatinine..mg.dl.)) +
geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") +
geom_density(color = "red", size = 1) +
labs(title = "Serum Creatinine", x = "Creatinine (mg/dl)", y = "Density") + theme_minimal()
p4 <- ggplot(data_ckd, aes(x = Body.Mass.Index..BMI.)) +
geom_histogram(aes(y = ..density..), bins = 8, fill = "#87CEEB", color = "black") +
geom_density(color = "red", size = 1) +
labs(title = "BMI", x = "BMI", y = "Density") + theme_minimal()
grid.arrange(p1, p2, p3, p4, ncol = 2, nrow = 2)
## Warning: The dot-dot notation (`..density..`) was deprecated in ggplot2 3.4.0.
## ℹ Please use `after_stat(density)` instead.
## This warning is displayed once per session.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.

# 2. Visualisasi Hubungan CKD Status dan Hipertensi (Bar Plot)
ggplot(data_ckd, aes(x = CKD.Status, fill = Hypertension..yes.no.)) +
geom_bar(position = "dodge") +
scale_fill_manual(values = c("no" = "#F8766D", "yes" = "#00BFC4")) +
labs(title = "CKD Status berdasarkan Hipertensi",
x = "CKD Status",
y = "Jumlah Pasien",
fill = "Hypertension") +
theme_minimal()

#4. Analisis Statistika Deskriptif
library(dplyr)
numeric_vars <- data_ckd %>% select(where(is.numeric))
deskriptif_summary <- data.frame(
Mean = apply(numeric_vars, 2, mean, na.rm = TRUE),
SD = apply(numeric_vars, 2, sd, na.rm = TRUE),
Median = apply(numeric_vars, 2, median, na.rm = TRUE),
Min = apply(numeric_vars, 2, min, na.rm = TRUE),
Max = apply(numeric_vars, 2, max, na.rm = TRUE)
)
print("Statistik Variabel Numerik")
## [1] "Statistik Variabel Numerik"
print(round(deskriptif_summary, 2))
## Mean SD Median Min Max
## Age.of.the.patient 56.90 20.08 54.50 25.00 90.00
## Blood.pressure..mm.Hg. 131.90 29.45 127.50 80.00 179.00
## Blood.urea..mg.dl. 101.27 58.01 97.62 8.69 198.73
## Serum.creatinine..mg.dl. 7.61 3.88 8.17 0.76 14.58
## Hemoglobin.level..gms. 12.09 3.37 11.85 6.00 17.50
## Urine.protein.to.creatinine.ratio 2.30 1.23 2.38 0.24 4.49
## Urine.output..ml.day. 1739.52 802.09 1740.50 308.00 2933.00
## Cholesterol.level 199.52 66.92 214.00 100.00 298.00
## Body.Mass.Index..BMI. 26.33 7.64 25.10 15.20 39.00
# B. Frekuensi Variabel Kategorik Utama
print("FREKUENSI CKD STATUS")
## [1] "FREKUENSI CKD STATUS"
print(table(data_ckd$CKD.Status))
##
## No Yes
## 31 19
print("FREKUENSI ANEMIA")
## [1] "FREKUENSI ANEMIA"
print(table(data_ckd$Anemia..yes.no.))
##
## no yes
## 17 33
print("FREKUENSI HIPERTENSI")
## [1] "FREKUENSI HIPERTENSI"
print(table(data_ckd$Hypertension..yes.no.))
##
## no yes
## 32 18
#5. UJI DISTRIBUSI & BENTUK DISTRIBUSI
shapiro_table <- data.frame(
W_Statistic = numeric(),
p_value = numeric(),
Keputusan = character(),
stringsAsFactors = FALSE
)
for (var in colnames(numeric_vars)) {
test <- shapiro.test(numeric_vars[[var]])
p_val <- test$p.value
keputusan <- ifelse(p_val >= 0.05, "Berdistribusi Normal", "Tidak Normal")
shapiro_table[var, ] <- c(
round(test$statistic, 4),
round(p_val, 4),
keputusan
)
}
print("HASIL UJI NORMALITAS SHAPIRO-WILK")
## [1] "HASIL UJI NORMALITAS SHAPIRO-WILK"
print(shapiro_table)
## W_Statistic p_value Keputusan
## Age.of.the.patient 0.9476 0.0271 Tidak Normal
## Blood.pressure..mm.Hg. 0.9464 0.0243 Tidak Normal
## Blood.urea..mg.dl. 0.9354 0.0089 Tidak Normal
## Serum.creatinine..mg.dl. 0.9572 0.0677 Berdistribusi Normal
## Hemoglobin.level..gms. 0.9484 0.0293 Tidak Normal
## Urine.protein.to.creatinine.ratio 0.9586 0.0777 Berdistribusi Normal
## Urine.output..ml.day. 0.9294 0.0053 Tidak Normal
## Cholesterol.level 0.901 5e-04 Tidak Normal
## Body.Mass.Index..BMI. 0.9186 0.0021 Tidak Normal
#6. HEATMAP & TABEL KONTINGENSI
library(ggplot2)
library(reshape2)
## Warning: package 'reshape2' was built under R version 4.5.3
# A. MEMBUAT TABEL KONTINGENSI (CKD Status vs Hipertensi)
tabel_kontingensi <- table(data_ckd$CKD.Status, data_ckd$Hypertension..yes.no.)
tabel_kontingensi_total <- addmargins(tabel_kontingensi)
print("TABEL KONTINGENSI (CKD Status vs Hipertensi)")
## [1] "TABEL KONTINGENSI (CKD Status vs Hipertensi)"
print(tabel_kontingensi_total)
##
## no yes Sum
## No 21 10 31
## Yes 11 8 19
## Sum 32 18 50
# B. MEMBUAT HEATMAP KORELASI VARIABEL NUMERIK
numeric_cols <- data_ckd[, sapply(data_ckd, is.numeric)]
matriks_korelasi <- cor(numeric_cols, use = "complete.obs")
melted_cormat <- melt(matriks_korelasi)
ggplot(data = melted_cormat, aes(x = Var1, y = Var2, fill = value)) +
geom_tile(color = "white") +
scale_fill_gradient2(low = "#4575b4", mid = "white", high = "#d73027",
midpoint = 0, limit = c(-1,1), name = "Korelasi") +
geom_text(aes(label = round(value, 2)), color = "black", size = 3) +
theme_minimal() +
theme(axis.text.x = element_text(angle = 45, vjust = 1, hjust = 1)) +
labs(title = "Heatmap Korelasi Variabel Numerik", x = "", y = "")
