Задание 1. Пакет CARET: разведочный анализ

Код

library(caret)
head(names(getModelInfo()), 30)

set.seed(1)
x <- matrix(rnorm(50 * 5), ncol = 5)
colnames(x) <- paste0("X", 1:5)
x <- as.data.frame(x)
y <- factor(rep(c("A", "B"), 25))

jpeg("featurePlot_strip.jpg", width = 1000, height = 700, quality = 95)
featurePlot(x, y, plot = "strip", jitter = TRUE,
            auto.key = list(columns = 2))
dev.off()

jpeg("featurePlot_box.jpg", width = 1000, height = 700, quality = 95)
featurePlot(x, y, plot = "box",
            scales = list(y = list(relation = "free"),
                          x = list(rot = 90)),
            layout = c(5, 1),
            auto.key = list(columns = 2))
dev.off()

jpeg("featurePlot_pairs.jpg", width = 1000, height = 700, quality = 95)
featurePlot(x, y, plot = "pairs",
            auto.key = list(columns = 2))
dev.off()

jpeg("featurePlot_density.jpg", width = 1000, height = 700, quality = 95)
featurePlot(x = x, y = y, plot = "density",
            scales = list(x = list(relation = "free"),
                          y = list(relation = "free")),
            auto.key = list(columns = 2))
dev.off()

jpeg("featurePlot_density2.jpg", width = 1000, height = 700, quality = 95)
featurePlot(x, y, plot = "density",
            scales = list(x = list(relation = "free"),
                          y = list(relation = "free")),
            adjust = 1.5, pch = "|", layout = c(5, 1),
            auto.key = list(columns = 2))
dev.off()

Strip plot Box plot Pairs plot Densityplot Densityplot

Вывод

Данные сгенерированы случайно: признаки X1–X5 взяты из стандартного нормального распределения независимо от класса, а метки A и B чередуются. Ни один признак не разделяет классы, то есть на этих данных нет информативных признаков. Это ожидаемо для случайного набора, и он полезен как «нулевой» пример для сравнения с реальными данными, где графики выглядят иначе.


Задание 2. Пакет FSelectorRcpp: важность признаков

Код

library(FSelectorRcpp)
data(iris)
str(iris)

Результат str(iris)

'data.frame':   150 obs. of  5 variables:
 $ Sepal.Length: num  5.1 4.9 4.7 4.6 5 5.4 4.6 5 4.4 4.9 ...
 $ Sepal.Width : num  3.5 3 3.2 3.1 3.6 3.9 3.4 3.4 2.9 3.1 ...
 $ Petal.Length: num  1.4 1.4 1.3 1.5 1.4 1.7 1.4 1.5 1.4 1.5 ...
 $ Petal.Width : num  0.2 0.2 0.2 0.2 0.2 0.4 0.3 0.2 0.2 0.1 ...
 $ Species     : Factor w/ 3 levels "setosa","versicolor",..: 1 1 1 1 1 1 1 1 1 1 ...
inf <- information_gain(Species ~ ., data = iris)
inf

Результат

  attributes importance
1 Sepal.Length  0.4521286
2  Sepal.Width  0.2672750
3 Petal.Length  0.9402853
4  Petal.Width  0.9554360

Вывод

Все методы единодушно ранжируют признаки одинаково: Petal.Length и Petal.Width — наиболее информативные для классификации видов ириса, тогда как Sepal.Width вносит наименьший вклад. Это согласуется с известной структурой набора iris, где длина и ширина лепестка лучше всего разделяют три вида.


Задание 3. Пакет arules: дискретизация

Код

install.packages("arules")
library(arules)

# Метод interval (равная ширина)
iris$Sepal.Length <- discretize(iris$Sepal.Length, method = "interval")

# Метод frequency (равная частота)
iris$Sepal.Width <- discretize(iris$Sepal.Width, method = "frequency")

head(iris)

Результат после interval и frequency

  Sepal.Length Sepal.Width Petal.Length Petal.Width Species
1    [4.3,5.5)   [3.2,4.4]          1.4         0.2  setosa
2    [4.3,5.5)   [2.9,3.2)          1.4         0.2  setosa
3    [4.3,5.5)   [3.2,4.4]          1.3         0.2  setosa
4    [4.3,5.5)   [2.9,3.2)          1.5         0.2  setosa
5    [4.3,5.5)   [3.2,4.4]          1.4         0.2  setosa
6    [4.3,5.5)   [3.2,4.4]          1.7         0.4  setosa
# Метод cluster (кластеризация)
iris$Petal.Length <- discretize(iris$Petal.Length, method = "cluster")

# Метод fixed (заданные границы)
iris$Petal.Width <- discretize(iris$Petal.Width, method = "fixed",
                               breaks = c(0, 0.5, 1, 2))

head(iris)

Результат после cluster и fixed

  Sepal.Length Sepal.Width Petal.Length Petal.Width Species
1    [4.3,5.5)   [3.2,4.4]     [1,2.88)     [0,0.5)  setosa
2    [4.3,5.5)   [2.9,3.2)     [1,2.88)     [0,0.5)  setosa
3    [4.3,5.5)   [3.2,4.4]     [1,2.88)     [0,0.5)  setosa
4    [4.3,5.5)   [2.9,3.2)     [1,2.88)     [0,0.5)  setosa
5    [4.3,5.5)   [3.2,4.4]     [1,2.88)     [0,0.5)  setosa
6    [4.3,5.5)   [3.2,4.4]     [1,2.88)     [0,0.5)  setosa

Вывод

Выбор метода зависит от задачи: для ассоциативных правил часто предпочтительнее frequency (сбалансированные корзины), а для интерпретируемых моделей — fixed или cluster.


Задание 4. Пакет Boruta: выбор признаков (Ozone)

Код

library(mlbench)
library(Boruta)

data("Ozone", package = "mlbench")
dim(Ozone)

ozone_clean <- na.omit(Ozone)

set.seed(123)
b <- Boruta(V4 ~ ., data = ozone_clean, doTrace = 0)

plot(b, xlab = "", xaxt = "n",
     main = "Boruta: важность признаков (Ozone)")
lz <- lapply(1:ncol(b$ImpHistory), function(i)
  b$ImpHistory[is.finite(b$ImpHistory[, i]), i])
names(lz) <- colnames(b$ImpHistory)
Labels <- sort(sapply(lz, median))
axis(side = 1, las = 2, labels = names(Labels),
     at = 1:ncol(b$ImpHistory), cex.axis = 0.7)

plotImpHistory(b)

Densityplot Densityplot

Вывод

Алгоритм Boruta сравнивает реальные признаки с «теневыми» копиями и надёжно отделяет значимые признаки от шума; на наборе Ozone он выделил группу признаков, связанных с температурой и сезоном.