The main goal of this Milestone Report is to perform exploratory data analysis on the Coursera SwiftKey dataset (Blogs, News, and Twitter) and outline plans for creating a next-word prediction algorithm and Shiny web application.
We loaded three English text files and calculated their basic properties including file sizes, line counts, and word counts.
library(stringi)
library(knitr)
blogs_path <- "final/en_US/en_US.blogs.txt"
news_path <- "final/en_US/en_US.news.txt"
twitter_path <- "final/en_US/en_US.twitter.txt"
blogs <- readLines(blogs_path, encoding = "UTF-8", skipNul = TRUE)
news <- readLines(news_path, encoding = "UTF-8", skipNul = TRUE)
twitter <- readLines(twitter_path, encoding = "UTF-8", skipNul = TRUE)
summary_table <- data.frame(
File_Name = c("en_US.blogs.txt", "en_US.news.txt", "en_US.twitter.txt"),
File_Size_MB = c(file.info(blogs_path)$size / (1024^2),
file.info(news_path)$size / (1024^2),
file.info(twitter_path)$size / (1024^2)),
Line_Count = c(length(blogs), length(news), length(twitter)),
Word_Count = c(sum(stri_count_words(blogs)),
sum(stri_count_words(news)),
sum(stri_count_words(twitter)))
)
kable(summary_table, digits = 2, caption = "Summary Statistics of Raw Data")
| File_Name | File_Size_MB | Line_Count | Word_Count |
|---|---|---|---|
| en_US.blogs.txt | 200.42 | 899288 | 37546806 |
| en_US.news.txt | 196.28 | 1010206 | 34761151 |
| en_US.twitter.txt | 159.36 | 2360148 | 30096690 |
We sampled 1% of the dataset to analyze word frequencies efficiently and removed common English stopwords to identify meaningful patterns.
library(quanteda)
library(ggplot2)
set.seed(1234)
sample_size <- 0.01
combined_sample <- c(
sample(blogs, length(blogs) * sample_size),
sample(news, length(news) * sample_size),
sample(twitter, length(twitter) * sample_size)
)
tokens_data <- tokens(combined_sample,
remove_punct = TRUE,
remove_symbols = TRUE,
remove_numbers = TRUE)
tokens_clean <- tokens_select(tokens_data, pattern = stopwords("en"), selection = "remove")
dfm_clean <- dfm(tokens_clean)
top_words_clean <- topfeatures(dfm_clean, 10)
df_freq_clean <- data.frame(word = names(top_words_clean), freq = as.numeric(top_words_clean))
ggplot(df_freq_clean, aes(x = reorder(word, freq), y = freq)) +
geom_bar(stat = "identity", fill = "darkgreen") +
coord_flip() +
labs(title = "Top 10 Meaningful Words (Without Stopwords)",
x = "Words",
y = "Frequency") +
theme_minimal()