The objective of the project is to clean and organize a ches tournament data from a text file. The completed dataset will have each player’s name, state, total poiint, pre-rating, and the average pre-tournament rating of their opponents. The cleaned data will also be exported as a CSV file.
library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr 1.2.1 ✔ readr 2.2.0
## ✔ forcats 1.0.1 ✔ stringr 1.5.1
## ✔ ggplot2 3.5.2 ✔ tibble 3.3.0
## ✔ lubridate 1.9.5 ✔ tidyr 1.3.2
## ✔ purrr 1.2.1
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag() masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors
First, I read the tournament text file into R. Each line is stored separately so I can extract the player information.
chess_data <- readLines("chess_tournament.txt")
## Warning in readLines("chess_tournament.txt"): incomplete final line found on
## 'chess_tournament.txt'
head(chess_data)
## [1] "-----------------------------------------------------------------------------------------"
## [2] " Pair | Player Name |Total|Round|Round|Round|Round|Round|Round|Round| "
## [3] " Num | USCF ID / Rtg (Pre->Post) | Pts | 1 | 2 | 3 | 4 | 5 | 6 | 7 | "
## [4] "-----------------------------------------------------------------------------------------"
## [5] " 1 | GARY HUA |6.0 |W 39|W 21|W 18|W 14|W 7|D 12|D 4|"
## [6] " ON | 15445895 / R: 1794 ->1817 |N:2 |W |B |W |B |W |B |W |"
##Removing header and seperator lines
player_data <- chess_data[5:length(chess_data)]
player_data <- player_data[!grepl("^-+", trimws(player_data))]
head(player_data)
## [1] " 1 | GARY HUA |6.0 |W 39|W 21|W 18|W 14|W 7|D 12|D 4|"
## [2] " ON | 15445895 / R: 1794 ->1817 |N:2 |W |B |W |B |W |B |W |"
## [3] " 2 | DAKSHESH DARURI |6.0 |W 63|W 58|L 4|W 17|W 16|W 20|W 7|"
## [4] " MI | 14598900 / R: 1553 ->1663 |N:2 |B |W |B |W |B |W |B |"
## [5] " 3 | ADITYA BAJAJ |6.0 |L 8|W 61|W 25|W 21|W 11|W 13|W 12|"
## [6] " MI | 14959604 / R: 1384 ->1640 |N:2 |W |B |W |B |W |B |W |"
length(player_data)
## [1] 128
##Seperating two lines for players
player_lines <- player_data[seq(1, length(player_data), by = 2)]
detail_lines <- player_data[seq(2, length(player_data), by = 2)]
length(player_lines)
## [1] 64
length(detail_lines)
## [1] 64
head(player_lines, 3)
## [1] " 1 | GARY HUA |6.0 |W 39|W 21|W 18|W 14|W 7|D 12|D 4|"
## [2] " 2 | DAKSHESH DARURI |6.0 |W 63|W 58|L 4|W 17|W 16|W 20|W 7|"
## [3] " 3 | ADITYA BAJAJ |6.0 |L 8|W 61|W 25|W 21|W 11|W 13|W 12|"
head(detail_lines, 3)
## [1] " ON | 15445895 / R: 1794 ->1817 |N:2 |W |B |W |B |W |B |W |"
## [2] " MI | 14598900 / R: 1553 ->1663 |N:2 |B |W |B |W |B |W |B |"
## [3] " MI | 14959604 / R: 1384 ->1640 |N:2 |W |B |W |B |W |B |W |"
# Split every player line on the |
player_parts <- strsplit(player_lines, "\\|")
# Take player number, name, and total points
player_number <- as.numeric(trimws(sapply(player_parts, `[`, 1)))
player_name <- trimws(sapply(player_parts, `[`, 2))
total_points <- as.numeric(trimws(sapply(player_parts, `[`, 3)))
# verify the first several values
head(player_number)
## [1] 1 2 3 4 5 6
head(player_name)
## [1] "GARY HUA" "DAKSHESH DARURI" "ADITYA BAJAJ"
## [4] "PATRICK H SCHILLING" "HANSHI ZUO" "HANSEN SONG"
head(total_points)
## [1] 6.0 6.0 6.0 5.5 5.5 5.0
##Take state an pre-rating
# Split the detail lines at the |
detail_parts <- strsplit(detail_lines, "\\|")
# Take every player's state
player_state <- trimws(sapply(detail_parts, `[`, 1))
# Take the pre-tournament rating
rating_text <- sapply(detail_parts, `[`, 2)
pre_rating <- as.numeric(sub(".*R:\\s*([0-9]+).*", "\\1", rating_text))
# Verify the results
head(player_state)
## [1] "ON" "MI" "MI" "MI" "MI" "OH"
head(pre_rating)
## [1] 1794 1553 1384 1716 1655 1686
opponents <- lapply(player_parts, function(x) {
rounds <- x[4:10]
opponent_numbers <- as.numeric(gsub("[^0-9]", "", rounds))
opponent_numbers <- opponent_numbers[!is.na(opponent_numbers)]
return(opponent_numbers)
})
opponents[[1]]
## [1] 39 21 18 14 7 12 4
avg_opponent_rating <- sapply(opponents, function(x) {
mean(pre_rating[x])
})
head(avg_opponent_rating)
## [1] 1605.286 1469.286 1563.571 1573.571 1500.857 1518.714
final_data <- data.frame(
Player_Name = player_name,
State = player_state,
Total_Points = total_points,
Pre_Rating = pre_rating,
Avg_Opponent_Pre_Rating = round(avg_opponent_rating)
)
head(final_data)
## Player_Name State Total_Points Pre_Rating Avg_Opponent_Pre_Rating
## 1 GARY HUA ON 6.0 1794 1605
## 2 DAKSHESH DARURI MI 6.0 1553 1469
## 3 ADITYA BAJAJ MI 6.0 1384 1564
## 4 PATRICK H SCHILLING MI 5.5 1716 1574
## 5 HANSHI ZUO MI 5.5 1655 1501
## 6 HANSEN SONG OH 5.0 1686 1519
write.csv(final_data, "chess_results.csv", row.names = FALSE)
dim(final_data)
## [1] 64 5
head(final_data)
## Player_Name State Total_Points Pre_Rating Avg_Opponent_Pre_Rating
## 1 GARY HUA ON 6.0 1794 1605
## 2 DAKSHESH DARURI MI 6.0 1553 1469
## 3 ADITYA BAJAJ MI 6.0 1384 1564
## 4 PATRICK H SCHILLING MI 5.5 1716 1574
## 5 HANSHI ZUO MI 5.5 1655 1501
## 6 HANSEN SONG OH 5.0 1686 1519
final_data[1, ]
## Player_Name State Total_Points Pre_Rating Avg_Opponent_Pre_Rating
## 1 GARY HUA ON 6 1794 1605