Load the Tournament Data
# Read the chess tournament text file
tournament <- readLines("tournamentinfo.txt", warn = FALSE)
# Display the first 20 lines
head(tournament, 20)
## [1] "-----------------------------------------------------------------------------------------"
## [2] " Pair | Player Name |Total|Round|Round|Round|Round|Round|Round|Round| "
## [3] " Num | USCF ID / Rtg (Pre->Post) | Pts | 1 | 2 | 3 | 4 | 5 | 6 | 7 | "
## [4] "-----------------------------------------------------------------------------------------"
## [5] " 1 | GARY HUA |6.0 |W 39|W 21|W 18|W 14|W 7|D 12|D 4|"
## [6] " ON | 15445895 / R: 1794 ->1817 |N:2 |W |B |W |B |W |B |W |"
## [7] "-----------------------------------------------------------------------------------------"
## [8] " 2 | DAKSHESH DARURI |6.0 |W 63|W 58|L 4|W 17|W 16|W 20|W 7|"
## [9] " MI | 14598900 / R: 1553 ->1663 |N:2 |B |W |B |W |B |W |B |"
## [10] "-----------------------------------------------------------------------------------------"
## [11] " 3 | ADITYA BAJAJ |6.0 |L 8|W 61|W 25|W 21|W 11|W 13|W 12|"
## [12] " MI | 14959604 / R: 1384 ->1640 |N:2 |W |B |W |B |W |B |W |"
## [13] "-----------------------------------------------------------------------------------------"
## [14] " 4 | PATRICK H SCHILLING |5.5 |W 23|D 28|W 2|W 26|D 5|W 19|D 1|"
## [15] " MI | 12616049 / R: 1716 ->1744 |N:2 |W |B |W |B |W |B |B |"
## [16] "-----------------------------------------------------------------------------------------"
## [17] " 5 | HANSHI ZUO |5.5 |W 45|W 37|D 12|D 13|D 4|W 14|W 17|"
## [18] " MI | 14601533 / R: 1655 ->1690 |N:2 |B |W |B |W |B |W |B |"
## [19] "-----------------------------------------------------------------------------------------"
## [20] " 6 | HANSEN SONG |5.0 |W 34|D 29|L 11|W 35|D 10|W 27|W 21|"
Separate Player Number, Name, and Points
# Split each player record using the | symbol
player_split <- strsplit(player_lines, "\\|")
# Extract player number
player_number <- as.numeric(trimws(sapply(player_split, `[`, 1)))
# Extract player name
player_name <- trimws(sapply(player_split, `[`, 2))
# Extract total points
total_points <- as.numeric(trimws(sapply(player_split, `[`, 3)))
# Preview the results
head(data.frame(
Player_Number = player_number,
Player_Name = player_name,
Total_Points = total_points
))
## Player_Number Player_Name Total_Points
## 1 1 GARY HUA 6.0
## 2 2 DAKSHESH DARURI 6.0
## 3 3 ADITYA BAJAJ 6.0
## 4 4 PATRICK H SCHILLING 5.5
## 5 5 HANSHI ZUO 5.5
## 6 6 HANSEN SONG 5.0
Extract Player State and Pre-Rating
# Get the second line for each player record
detail_lines <- tournament[grepl("^\\s*[A-Z]{2}\\s*\\|", tournament)]
# Display the first few detail lines
head(detail_lines)
## [1] " ON | 15445895 / R: 1794 ->1817 |N:2 |W |B |W |B |W |B |W |"
## [2] " MI | 14598900 / R: 1553 ->1663 |N:2 |B |W |B |W |B |W |B |"
## [3] " MI | 14959604 / R: 1384 ->1640 |N:2 |W |B |W |B |W |B |W |"
## [4] " MI | 12616049 / R: 1716 ->1744 |N:2 |W |B |W |B |W |B |B |"
## [5] " MI | 14601533 / R: 1655 ->1690 |N:2 |B |W |B |W |B |W |B |"
## [6] " OH | 15055204 / R: 1686 ->1687 |N:3 |W |B |W |B |B |W |B |"
Separate State and Pre-Rating
# Split each detail line using the | symbol
detail_split <- strsplit(detail_lines, "\\|")
# Extract state
state <- trimws(sapply(detail_split, `[`, 1))
# Extract the rating section
rating_text <- trimws(sapply(detail_split, `[`, 2))
# Extract the pre-tournament rating after "R:"
pre_rating <- as.numeric(
sub(".*R:\\s*([0-9]+).*", "\\1", rating_text)
)
# Preview the results
head(data.frame(
State = state,
Pre_Rating = pre_rating
))
## State Pre_Rating
## 1 ON 1794
## 2 MI 1553
## 3 MI 1384
## 4 MI 1716
## 5 MI 1655
## 6 OH 1686
Create Player Data Table
# Combine the extracted player information into one data frame
players <- data.frame(
Player_Number = player_number,
Player_Name = player_name,
State = state,
Total_Points = total_points,
Pre_Rating = pre_rating
)
# Preview the first few rows
head(players)
## Player_Number Player_Name State Total_Points Pre_Rating
## 1 1 GARY HUA ON 6.0 1794
## 2 2 DAKSHESH DARURI MI 6.0 1553
## 3 3 ADITYA BAJAJ MI 6.0 1384
## 4 4 PATRICK H SCHILLING MI 5.5 1716
## 5 5 HANSHI ZUO MI 5.5 1655
## 6 6 HANSEN SONG OH 5.0 1686
Calculate Average Opponent Pre-Rating
# Calculate the average pre-rating of each player's opponents
average_opponent_rating <- sapply(opponents, function(opponent_ids) {
# Match opponent numbers to the player table
opponent_ratings <- players$Pre_Rating[
match(opponent_ids, players$Player_Number)
]
# Calculate the average opponent pre-rating
mean(opponent_ratings, na.rm = TRUE)
})
# Add the average opponent rating to the player table
players$Average_Opponent_Pre_Rating <- round(average_opponent_rating)
# Preview the first few rows
head(players)
## Player_Number Player_Name State Total_Points Pre_Rating
## 1 1 GARY HUA ON 6.0 1794
## 2 2 DAKSHESH DARURI MI 6.0 1553
## 3 3 ADITYA BAJAJ MI 6.0 1384
## 4 4 PATRICK H SCHILLING MI 5.5 1716
## 5 5 HANSHI ZUO MI 5.5 1655
## 6 6 HANSEN SONG OH 5.0 1686
## Average_Opponent_Pre_Rating
## 1 1605
## 2 1469
## 3 1564
## 4 1574
## 5 1501
## 6 1519
Create Final Dataset
# Keep only the columns required for the final project output
final_data <- players[, c(
"Player_Name",
"State",
"Total_Points",
"Pre_Rating",
"Average_Opponent_Pre_Rating"
)]
# Preview the final dataset
head(final_data)
## Player_Name State Total_Points Pre_Rating Average_Opponent_Pre_Rating
## 1 GARY HUA ON 6.0 1794 1605
## 2 DAKSHESH DARURI MI 6.0 1553 1469
## 3 ADITYA BAJAJ MI 6.0 1384 1564
## 4 PATRICK H SCHILLING MI 5.5 1716 1574
## 5 HANSHI ZUO MI 5.5 1655 1501
## 6 HANSEN SONG OH 5.0 1686 1519
Export Final Dataset
# Export the cleaned tournament data to a CSV file
write.csv(
final_data,
"Project1_Chess_Tournament_Results.csv",
row.names = FALSE
)
file.exists("Project1_Chess_Tournament_Results.csv")
## [1] TRUE