Задание 1. Пакет CARET и featurePlot

library(caret)
## Загрузка требуемого пакета: ggplot2
## Загрузка требуемого пакета: lattice
available_models <- names(getModelInfo())
head(available_models, 10)
##  [1] "ada"         "AdaBag"      "AdaBoost.M1" "adaboost"    "amdai"      
##  [6] "ANFIS"       "avNNet"      "awnb"        "awtan"       "bag"
set.seed(321)
x <- matrix(rnorm(50 * 5), ncol = 5)
y <- factor(rep(c("A", "B"), 25))

# 1. pairs
jpeg("caret_pairs_plot.jpg", width = 800, height = 800)
print(featurePlot(x = x, y = y, plot = "pairs"))
dev.off()
## png 
##   2
# 2. density
jpeg("caret_density_plot.jpg", width = 800, height = 600)
print(featurePlot(x = x, y = y, plot = "density"))
dev.off()
## png 
##   2
# 3. box
jpeg("caret_box_plot.jpg", width = 800, height = 600)
print(featurePlot(x = x, y = y, plot = "box"))
dev.off()
## png 
##   2
# 4. ellipse
jpeg("caret_ellipse_plot.jpg", width = 800, height = 600)
print(featurePlot(x = x, y = y, plot = "ellipse"))
dev.off()
## png 
##   2

Вывод по заданию 1: данные сгенерированы случайно из нормального распределения, классы «A» и «B» не разделяются — исходные признаки не обладают дискриминирующей способностью.

Задание 2. Важность признаков (пакет FSelector)

library(FSelectorRcpp)
data(iris)

# Информационный выигрыш (аналог information.gain)
weights <- information_gain(Species ~ ., data = iris)
print(weights)
##     attributes importance
## 1 Sepal.Length  0.4521286
## 2  Sepal.Width  0.2672750
## 3 Petal.Length  0.9402853
## 4  Petal.Width  0.9554360
# Отбор двух лучших признаков (аналог cutoff.k)
library(dplyr)
## 
## Присоединяю пакет: 'dplyr'
## Следующие объекты скрыты от 'package:stats':
## 
##     filter, lag
## Следующие объекты скрыты от 'package:base':
## 
##     intersect, setdiff, setequal, union
subset <- weights %>% top_n(2, importance) %>% pull(attributes)
cat("Выбранные признаки:", subset, "\n")
## Выбранные признаки: Petal.Length Petal.Width

Вывод по заданию 2: наибольший информационный выигрыш у признаков Petal.Length и Petal.Width.

Задание 3. Дискретизация (пакет arules)

library(arules)
## Загрузка требуемого пакета: Matrix
## 
## Присоединяю пакет: 'arules'
## Следующий объект скрыт от 'package:dplyr':
## 
##     recode
## Следующий объект скрыт от 'package:FSelectorRcpp':
## 
##     discretize
## Следующие объекты скрыты от 'package:base':
## 
##     abbreviate, write
data(iris)
x <- iris$Sepal.Length

disc_interval  <- discretize(x, method = "interval",  breaks = 3)
disc_frequency <- discretize(x, method = "frequency", breaks = 3)
disc_cluster   <- discretize(x, method = "cluster",   breaks = 3)
disc_fixed     <- discretize(x, method = "fixed",     breaks = c(4.0, 5.5, 6.5, 8.0))

table(disc_interval)
## disc_interval
## [4.3,5.5) [5.5,6.7) [6.7,7.9] 
##        52        70        28
table(disc_frequency)
## disc_frequency
## [4.3,5.4) [5.4,6.3) [6.3,7.9] 
##        46        53        51
table(disc_cluster)
## disc_cluster
##  [4.3,5.63) [5.63,6.71)  [6.71,7.9] 
##          65          65          20
table(disc_fixed)
## disc_fixed
##   [4,5.5) [5.5,6.5)   [6.5,8] 
##        52        63        35
par(mfrow = c(2, 2))
hist(x, breaks = 20, main = "Interval (равная ширина)", col = "lightblue")
abline(v = discretize(x, method = "interval", breaks = 3, onlycuts = TRUE), col = "red", lwd = 2)

hist(x, breaks = 20, main = "Frequency (равная частота)", col = "lightgreen")
abline(v = discretize(x, method = "frequency", breaks = 3, onlycuts = TRUE), col = "red", lwd = 2)

hist(x, breaks = 20, main = "Cluster (k-means)", col = "lightyellow")
abline(v = discretize(x, method = "cluster", breaks = 3, onlycuts = TRUE), col = "red", lwd = 2)

hist(x, breaks = 20, main = "Fixed (ручные границы)", col = "lightpink")
abline(v = c(4.0, 5.5, 6.5, 8.0), col = "red", lwd = 2)

par(mfrow = c(1, 1))

Вывод по заданию 3: метод frequency даёт наиболее сбалансированные группы, cluster — наиболее «естественные» границы.

Задание 4. Выбор признаков (пакет Boruta)

library(Boruta)
library(mlbench)
library(randomForest)
## randomForest 4.7-1.2
## Type rfNews() to see new features/changes/bug fixes.
## 
## Присоединяю пакет: 'randomForest'
## Следующий объект скрыт от 'package:dplyr':
## 
##     combine
## Следующий объект скрыт от 'package:ggplot2':
## 
##     margin
data("Ozone")
ozone_clean <- na.omit(Ozone)

set.seed(123)
boruta_output <- Boruta(V4 ~ ., data = ozone_clean, doTrace = 0)
print(boruta_output)
## Boruta performed 18 iterations in 0.314796 secs.
##  9 attributes confirmed important: V1, V10, V11, V12, V13 and 4 more;
##  3 attributes confirmed unimportant: V2, V3, V6;
jpeg("boruta_boxplot.jpg", width = 1000, height = 600)
plot(boruta_output, cex.axis = 0.7, las = 2, xlab = "", main = "Boruta: важность признаков")
dev.off()
## png 
##   2

Вывод по заданию 4: Boruta выделяет статистически значимые признаки, влияющие на концентрацию озона.