##BÀI TẬP VỀ NHÀ
Giải thích dataset Dataset: diamonds được chọn trong package ggplot2 BỘ dữ liệu chứa giá và các thuộc tính của khoảng 54.000 viên kim cương cụ thể là 53.940 viên kim cương cắt tròn. Làm sao mà chúng ta biết được? Mỗi hàng dữ liệu đại diện cho một viên kim cương khác nhau và có 53.940 hàng dữ liệu
Có 10 biến đo lường các thông tin khác nhau về kim cương 1.price: giá đô la Mỹ($326-$18,823) 2.carat: trọng lượng của viên kim cương(0.2–5.01) 3.cut chất lượng của vết cắt (Khá, Tốt, Rất tốt, Đặc biệt, Lý tưởng) 4.color màu kim cương J (kém nhất) đến D (tốt nhất) 5.clarity đo độ trong của viên kim cương I1 (kém nhất), SI2, SI1, VS2, VS1, VVS2, VVS1, IF (tốt nhất) 6.x chiều dài tính bằng mm 0-10,74 7.y chiều rộng tính bằng mm 0-58,9 8.z độ sâu tính bằng mm 0-31.8 depth tổng tỷ lệ phần trăm độ sâu 43-79 table chiều rộng của đỉnh kim cương so với điểm rộng nhất 43-95
library(ggplot2)
## Warning: package 'ggplot2' was built under R version 4.2.3
str(diamonds)
## tibble [53,940 × 10] (S3: tbl_df/tbl/data.frame)
## $ carat : num [1:53940] 0.23 0.21 0.23 0.29 0.31 0.24 0.24 0.26 0.22 0.23 ...
## $ cut : Ord.factor w/ 5 levels "Fair"<"Good"<..: 5 4 2 4 2 3 3 3 1 3 ...
## $ color : Ord.factor w/ 7 levels "D"<"E"<"F"<"G"<..: 2 2 2 6 7 7 6 5 2 5 ...
## $ clarity: Ord.factor w/ 8 levels "I1"<"SI2"<"SI1"<..: 2 3 5 4 2 6 7 3 4 5 ...
## $ depth : num [1:53940] 61.5 59.8 56.9 62.4 63.3 62.8 62.3 61.9 65.1 59.4 ...
## $ table : num [1:53940] 55 61 65 58 58 57 57 55 61 61 ...
## $ price : int [1:53940] 326 326 327 334 335 336 336 337 337 338 ...
## $ x : num [1:53940] 3.95 3.89 4.05 4.2 4.34 3.94 3.95 4.07 3.87 4 ...
## $ y : num [1:53940] 3.98 3.84 4.07 4.23 4.35 3.96 3.98 4.11 3.78 4.05 ...
## $ z : num [1:53940] 2.43 2.31 2.31 2.63 2.75 2.48 2.47 2.53 2.49 2.39 ...
hist(diamonds$price)
mean(diamonds$price)
## [1] 3932.8
summary(diamonds$price)
## Min. 1st Qu. Median Mean 3rd Qu. Max.
## 326 950 2401 3933 5324 18823
ggplot(data = diamonds) + geom_bar(mapping = aes(x = cut))
head(diamonds, 12)# prints the first dozen rows of data
## # A tibble: 12 × 10
## carat cut color clarity depth table price x y z
## <dbl> <ord> <ord> <ord> <dbl> <dbl> <int> <dbl> <dbl> <dbl>
## 1 0.23 Ideal E SI2 61.5 55 326 3.95 3.98 2.43
## 2 0.21 Premium E SI1 59.8 61 326 3.89 3.84 2.31
## 3 0.23 Good E VS1 56.9 65 327 4.05 4.07 2.31
## 4 0.29 Premium I VS2 62.4 58 334 4.2 4.23 2.63
## 5 0.31 Good J SI2 63.3 58 335 4.34 4.35 2.75
## 6 0.24 Very Good J VVS2 62.8 57 336 3.94 3.96 2.48
## 7 0.24 Very Good I VVS1 62.3 57 336 3.95 3.98 2.47
## 8 0.26 Very Good H SI1 61.9 55 337 4.07 4.11 2.53
## 9 0.22 Fair E VS2 65.1 61 337 3.87 3.78 2.49
## 10 0.23 Very Good H VS1 59.4 61 338 4 4.05 2.39
## 11 0.3 Good J SI1 64 55 339 4.25 4.28 2.73
## 12 0.23 Ideal J VS1 62.8 56 340 3.93 3.9 2.46
qplot(cut, data=diamonds) # cut is an ordered factor with 5 levels
## Warning: `qplot()` was deprecated in ggplot2 3.4.0.
## This warning is displayed once every 8 hours.
## Call `lifecycle::last_lifecycle_warnings()` to see where this warning was
## generated.
qplot(x=carat, y=price, color=clarity, facets=color~cut, data=diamonds)
# Tạo dữ liệu mẫu
set.seed(123)
filtered <- data.frame(x = rnorm(1000), y = rnorm(1000), z = rnorm(1000))
# Chọn các quan sát thỏa mãn điều kiện x \> -1 & y \< 1 & z \> -0.5
ggplot(data = diamonds) +
geom_bar(mapping = aes(x = cut))
ggplot(data = diamonds) +
geom_histogram(mapping = aes(x = carat), binwidth = .5)
## \`stat_bin()\` using \`bins = 30\`. Pick better value with \`binwidth\`.
smaller <- diamonds
ggplot(data = smaller, mapping = aes(x = carat)) +
geom_histogram(binwidth = .01)
ggplot(data = faithful, mapping = aes(x = eruptions)) +
geom_histogram()
## `stat_bin()` using `bins = 30`. Pick better value with `binwidth`.
ggplot(data = diamonds, mapping = aes(x = y)) +
geom_histogram(binwidth = .5)
ggplot(data = diamonds, mapping = aes(x = y)) +
geom_histogram() +
coord_cartesian(ylim = c(0, 50))
## `stat_bin()` using `bins = 30`. Pick better value with `binwidth`.