library(caret)
## Загрузка требуемого пакета: ggplot2
## Загрузка требуемого пакета: lattice
available_models <- names(getModelInfo())
head(available_models, 10)
## [1] "ada" "AdaBag" "AdaBoost.M1" "adaboost" "amdai"
## [6] "ANFIS" "avNNet" "awnb" "awtan" "bag"
set.seed(321)
x <- matrix(rnorm(50 * 5), ncol = 5)
y <- factor(rep(c("A", "B"), 25))
# 1. pairs
jpeg("caret_pairs_plot.jpg", width = 800, height = 800)
print(featurePlot(x = x, y = y, plot = "pairs"))
dev.off()
## png
## 2
# 2. density
jpeg("caret_density_plot.jpg", width = 800, height = 600)
print(featurePlot(x = x, y = y, plot = "density"))
dev.off()
## png
## 2
# 3. box
jpeg("caret_box_plot.jpg", width = 800, height = 600)
print(featurePlot(x = x, y = y, plot = "box"))
dev.off()
## png
## 2
# 4. ellipse
jpeg("caret_ellipse_plot.jpg", width = 800, height = 600)
print(featurePlot(x = x, y = y, plot = "ellipse"))
dev.off()
## png
## 2
Вывод по заданию 1: данные сгенерированы случайно из нормального распределения, классы «A» и «B» не разделяются — исходные признаки не обладают дискриминирующей способностью.
library(FSelectorRcpp)
data(iris)
# Информационный выигрыш (аналог information.gain)
weights <- information_gain(Species ~ ., data = iris)
print(weights)
## attributes importance
## 1 Sepal.Length 0.4521286
## 2 Sepal.Width 0.2672750
## 3 Petal.Length 0.9402853
## 4 Petal.Width 0.9554360
# Отбор двух лучших признаков (аналог cutoff.k)
library(dplyr)
##
## Присоединяю пакет: 'dplyr'
## Следующие объекты скрыты от 'package:stats':
##
## filter, lag
## Следующие объекты скрыты от 'package:base':
##
## intersect, setdiff, setequal, union
subset <- weights %>% top_n(2, importance) %>% pull(attributes)
cat("Выбранные признаки:", subset, "\n")
## Выбранные признаки: Petal.Length Petal.Width
Вывод по заданию 2: наибольший информационный выигрыш у признаков Petal.Length и Petal.Width.
library(arules)
## Загрузка требуемого пакета: Matrix
##
## Присоединяю пакет: 'arules'
## Следующий объект скрыт от 'package:dplyr':
##
## recode
## Следующий объект скрыт от 'package:FSelectorRcpp':
##
## discretize
## Следующие объекты скрыты от 'package:base':
##
## abbreviate, write
data(iris)
x <- iris$Sepal.Length
disc_interval <- discretize(x, method = "interval", breaks = 3)
disc_frequency <- discretize(x, method = "frequency", breaks = 3)
disc_cluster <- discretize(x, method = "cluster", breaks = 3)
disc_fixed <- discretize(x, method = "fixed", breaks = c(4.0, 5.5, 6.5, 8.0))
table(disc_interval)
## disc_interval
## [4.3,5.5) [5.5,6.7) [6.7,7.9]
## 52 70 28
table(disc_frequency)
## disc_frequency
## [4.3,5.4) [5.4,6.3) [6.3,7.9]
## 46 53 51
table(disc_cluster)
## disc_cluster
## [4.3,5.63) [5.63,6.71) [6.71,7.9]
## 65 65 20
table(disc_fixed)
## disc_fixed
## [4,5.5) [5.5,6.5) [6.5,8]
## 52 63 35
par(mfrow = c(2, 2))
hist(x, breaks = 20, main = "Interval (равная ширина)", col = "lightblue")
abline(v = discretize(x, method = "interval", breaks = 3, onlycuts = TRUE), col = "red", lwd = 2)
hist(x, breaks = 20, main = "Frequency (равная частота)", col = "lightgreen")
abline(v = discretize(x, method = "frequency", breaks = 3, onlycuts = TRUE), col = "red", lwd = 2)
hist(x, breaks = 20, main = "Cluster (k-means)", col = "lightyellow")
abline(v = discretize(x, method = "cluster", breaks = 3, onlycuts = TRUE), col = "red", lwd = 2)
hist(x, breaks = 20, main = "Fixed (ручные границы)", col = "lightpink")
abline(v = c(4.0, 5.5, 6.5, 8.0), col = "red", lwd = 2)
par(mfrow = c(1, 1))
Вывод по заданию 3: метод frequency даёт наиболее сбалансированные группы, cluster — наиболее «естественные» границы.
library(Boruta)
library(mlbench)
library(randomForest)
## randomForest 4.7-1.2
## Type rfNews() to see new features/changes/bug fixes.
##
## Присоединяю пакет: 'randomForest'
## Следующий объект скрыт от 'package:dplyr':
##
## combine
## Следующий объект скрыт от 'package:ggplot2':
##
## margin
data("Ozone")
ozone_clean <- na.omit(Ozone)
set.seed(123)
boruta_output <- Boruta(V4 ~ ., data = ozone_clean, doTrace = 0)
print(boruta_output)
## Boruta performed 18 iterations in 0.314796 secs.
## 9 attributes confirmed important: V1, V10, V11, V12, V13 and 4 more;
## 3 attributes confirmed unimportant: V2, V3, V6;
jpeg("boruta_boxplot.jpg", width = 1000, height = 600)
plot(boruta_output, cex.axis = 0.7, las = 2, xlab = "", main = "Boruta: важность признаков")
dev.off()
## png
## 2
Вывод по заданию 4: Boruta выделяет статистически значимые признаки, влияющие на концентрацию озона.