Introduction

The objective of the project is to clean and organize a ches tournament data from a text file. The completed dataset will have each player’s name, state, total poiint, pre-rating, and the average pre-tournament rating of their opponents. The cleaned data will also be exported as a CSV file.

library(tidyverse)
## ── Attaching core tidyverse packages ──────────────────────── tidyverse 2.0.0 ──
## ✔ dplyr     1.2.1     ✔ readr     2.2.0
## ✔ forcats   1.0.1     ✔ stringr   1.5.1
## ✔ ggplot2   3.5.2     ✔ tibble    3.3.0
## ✔ lubridate 1.9.5     ✔ tidyr     1.3.2
## ✔ purrr     1.2.1     
## ── Conflicts ────────────────────────────────────────── tidyverse_conflicts() ──
## ✖ dplyr::filter() masks stats::filter()
## ✖ dplyr::lag()    masks stats::lag()
## ℹ Use the conflicted package (<http://conflicted.r-lib.org/>) to force all conflicts to become errors

Importing Data

First, I read the tournament text file into R. Each line is stored separately so I can extract the player information.

chess_data <- readLines("chess_tournament.txt")
## Warning in readLines("chess_tournament.txt"): incomplete final line found on
## 'chess_tournament.txt'
head(chess_data)
## [1] "-----------------------------------------------------------------------------------------" 
## [2] " Pair | Player Name                     |Total|Round|Round|Round|Round|Round|Round|Round| "
## [3] " Num  | USCF ID / Rtg (Pre->Post)       | Pts |  1  |  2  |  3  |  4  |  5  |  6  |  7  | "
## [4] "-----------------------------------------------------------------------------------------" 
## [5] "    1 | GARY HUA                        |6.0  |W  39|W  21|W  18|W  14|W   7|D  12|D   4|" 
## [6] "   ON | 15445895 / R: 1794   ->1817     |N:2  |W    |B    |W    |B    |W    |B    |W    |"

##Removing header and seperator lines

player_data <- chess_data[5:length(chess_data)]
player_data <- player_data[!grepl("^-+", trimws(player_data))]

head(player_data)
## [1] "    1 | GARY HUA                        |6.0  |W  39|W  21|W  18|W  14|W   7|D  12|D   4|"
## [2] "   ON | 15445895 / R: 1794   ->1817     |N:2  |W    |B    |W    |B    |W    |B    |W    |"
## [3] "    2 | DAKSHESH DARURI                 |6.0  |W  63|W  58|L   4|W  17|W  16|W  20|W   7|"
## [4] "   MI | 14598900 / R: 1553   ->1663     |N:2  |B    |W    |B    |W    |B    |W    |B    |"
## [5] "    3 | ADITYA BAJAJ                    |6.0  |L   8|W  61|W  25|W  21|W  11|W  13|W  12|"
## [6] "   MI | 14959604 / R: 1384   ->1640     |N:2  |W    |B    |W    |B    |W    |B    |W    |"
length(player_data)
## [1] 128

##Seperating two lines for players

player_lines <- player_data[seq(1, length(player_data), by = 2)]
detail_lines <- player_data[seq(2, length(player_data), by = 2)]

length(player_lines)
## [1] 64
length(detail_lines)
## [1] 64
head(player_lines, 3)
## [1] "    1 | GARY HUA                        |6.0  |W  39|W  21|W  18|W  14|W   7|D  12|D   4|"
## [2] "    2 | DAKSHESH DARURI                 |6.0  |W  63|W  58|L   4|W  17|W  16|W  20|W   7|"
## [3] "    3 | ADITYA BAJAJ                    |6.0  |L   8|W  61|W  25|W  21|W  11|W  13|W  12|"
head(detail_lines, 3)
## [1] "   ON | 15445895 / R: 1794   ->1817     |N:2  |W    |B    |W    |B    |W    |B    |W    |"
## [2] "   MI | 14598900 / R: 1553   ->1663     |N:2  |B    |W    |B    |W    |B    |W    |B    |"
## [3] "   MI | 14959604 / R: 1384   ->1640     |N:2  |W    |B    |W    |B    |W    |B    |W    |"

Extrapolate player data

# Split every player line on the | 
player_parts <- strsplit(player_lines, "\\|")

# Take player number, name, and total points
player_number <- as.numeric(trimws(sapply(player_parts, `[`, 1)))
player_name <- trimws(sapply(player_parts, `[`, 2))
total_points <- as.numeric(trimws(sapply(player_parts, `[`, 3)))

# verify the first several values
head(player_number)
## [1] 1 2 3 4 5 6
head(player_name)
## [1] "GARY HUA"            "DAKSHESH DARURI"     "ADITYA BAJAJ"       
## [4] "PATRICK H SCHILLING" "HANSHI ZUO"          "HANSEN SONG"
head(total_points)
## [1] 6.0 6.0 6.0 5.5 5.5 5.0

##Take state an pre-rating

# Split the detail lines at the |
detail_parts <- strsplit(detail_lines, "\\|")

# Take every player's state
player_state <- trimws(sapply(detail_parts, `[`, 1))

# Take the pre-tournament rating
rating_text <- sapply(detail_parts, `[`, 2)
pre_rating <- as.numeric(sub(".*R:\\s*([0-9]+).*", "\\1", rating_text))

# Verify the results
head(player_state)
## [1] "ON" "MI" "MI" "MI" "MI" "OH"
head(pre_rating)
## [1] 1794 1553 1384 1716 1655 1686

Extrapolate player’s opponents

opponents <- lapply(player_parts, function(x) {

  rounds <- x[4:10]

  opponent_numbers <- as.numeric(gsub("[^0-9]", "", rounds))

  opponent_numbers <- opponent_numbers[!is.na(opponent_numbers)]

  return(opponent_numbers)
})

opponents[[1]]
## [1] 39 21 18 14  7 12  4

Avg apponent pre-rating

avg_opponent_rating <- sapply(opponents, function(x) {
  mean(pre_rating[x])
})

head(avg_opponent_rating)
## [1] 1605.286 1469.286 1563.571 1573.571 1500.857 1518.714

Resutls loaded into Data Frame

final_data <- data.frame(
  Player_Name = player_name,
  State = player_state,
  Total_Points = total_points,
  Pre_Rating = pre_rating,
  Avg_Opponent_Pre_Rating = round(avg_opponent_rating)
)

head(final_data)
##           Player_Name State Total_Points Pre_Rating Avg_Opponent_Pre_Rating
## 1            GARY HUA    ON          6.0       1794                    1605
## 2     DAKSHESH DARURI    MI          6.0       1553                    1469
## 3        ADITYA BAJAJ    MI          6.0       1384                    1564
## 4 PATRICK H SCHILLING    MI          5.5       1716                    1574
## 5          HANSHI ZUO    MI          5.5       1655                    1501
## 6         HANSEN SONG    OH          5.0       1686                    1519
write.csv(final_data, "chess_results.csv", row.names = FALSE)

Validation Check

dim(final_data)
## [1] 64  5
head(final_data)
##           Player_Name State Total_Points Pre_Rating Avg_Opponent_Pre_Rating
## 1            GARY HUA    ON          6.0       1794                    1605
## 2     DAKSHESH DARURI    MI          6.0       1553                    1469
## 3        ADITYA BAJAJ    MI          6.0       1384                    1564
## 4 PATRICK H SCHILLING    MI          5.5       1716                    1574
## 5          HANSHI ZUO    MI          5.5       1655                    1501
## 6         HANSEN SONG    OH          5.0       1686                    1519
final_data[1, ]
##   Player_Name State Total_Points Pre_Rating Avg_Opponent_Pre_Rating
## 1    GARY HUA    ON            6       1794                    1605