library(ncaahoopR)
## Loading required package: dplyr
## 
## Attaching package: 'dplyr'
## The following objects are masked from 'package:stats':
## 
##     filter, lag
## The following objects are masked from 'package:base':
## 
##     intersect, setdiff, setequal, union
## Loading required package: ggplot2
library(ggplot2)
library(ggimage)
library(dplyr)
library(networkD3)
dive_2000 <- read.csv("/Users/monkey_hou/Library/Mobile Documents/com~apple~CloudDocs/Temple/2023 Spring/analytics/week11/Diving2000_mod.csv")
mean(dive_2000$Difficulty)
## [1] 2.763141
median(dive_2000$Difficulty)
## [1] 3
quantile(dive_2000$Difficulty)
##   0%  25%  50%  75% 100% 
##  1.5  2.6  3.0  3.1  3.8
mean(dive_2000$AvgScore)
## [1] 6.832576
median(dive_2000$AvgScore)
## [1] 7.071429
quantile(dive_2000$AvgScore)
##        0%       25%       50%       75%      100% 
## 0.8571429 6.1428571 7.0714286 7.7857143 9.9285714
difficulty <- dive_2000 %>%
  count(Difficulty<= 2.0, 2.0<Difficulty & Difficulty<=2.5, 2.5<Difficulty & Difficulty<=3.0, 3.0<Difficulty & Difficulty<=3.5, 3.5<Difficulty)%>%
  select(n)%>%
  mutate("difficulty" = c("3.5⬆️","3.0-3.5","2.6-3.0","2.1-2.5","2.0⬇️"))
difficulty
##      n difficulty
## 1   42       3.5⬆️
## 2 3122    3.0-3.5
## 3 4935    2.6-3.0
## 4 1029    2.1-2.5
## 5 1659       2.0⬇️
score <- dive_2000 %>%
  count(AvgScore<= 5.0, 5.0<AvgScore & AvgScore<=6.5, 6.5<AvgScore & AvgScore<=7.5, 7.5<AvgScore & AvgScore<=8.5, 8.5<AvgScore)%>%
  select(n)%>%
  mutate("difficulty" = c("8.5⬆️","7.5-8.5","6.5-7.5","5.0-6.5","5.0⬇️"))

score
##      n difficulty
## 1  672       8.5⬆️
## 2 3024    7.5-8.5
## 3 3556    6.5-7.5
## 4 2282    5.0-6.5
## 5 1253       5.0⬇️
dive_2000<-dive_2000 %>%
  mutate("Difficulty Range" = case_when(Difficulty<=2.0 ~"2.0⬇️",2.0<Difficulty & Difficulty<=2.5 ~ "2.1-2.5", 2.5<Difficulty & Difficulty<=3.0~"2.6-3.0", 3.0<Difficulty & Difficulty<=3.5~"3.0-3.5", 3.5<Difficulty~"3.5⬆️"))%>%
  mutate("Avg Score Range"= case_when(AvgScore<= 5.0~"5.0⬇️", 5.0<AvgScore & AvgScore<=6.5~"5.0-6.5", 6.5<AvgScore & AvgScore<=7.5~"6.5-7.5", 7.5<AvgScore & AvgScore<=8.5~"7.5-8.5", 8.5<AvgScore~"8.5⬆️"))%>%
  mutate("all"=c("all diver"))
dive_2000%>%
  group_by(`Difficulty Range`,`Avg Score Range`)
## # A tibble: 10,787 × 14
## # Groups:   Difficulty Range, Avg Score Range [24]
##    Event Round Diver    Country  Rank DiveNo Difficulty JScore Judge    JCountry
##    <chr> <chr> <chr>    <chr>   <int>  <int>      <dbl>  <dbl> <chr>    <chr>   
##  1 M3mSB Final XIONG Ni CHN         1      1        3.1    8   RUIZ-PE… CUB     
##  2 M3mSB Final XIONG Ni CHN         1      1        3.1    9   GEAR De… NZL     
##  3 M3mSB Final XIONG Ni CHN         1      1        3.1    8.5 BOYS Be… CAN     
##  4 M3mSB Final XIONG Ni CHN         1      1        3.1    8.5 JOHNSON… NOR     
##  5 M3mSB Final XIONG Ni CHN         1      1        3.1    8.5 BOUSSAR… FRA     
##  6 M3mSB Final XIONG Ni CHN         1      1        3.1    8.5 CALDERO… PUR     
##  7 M3mSB Final XIONG Ni CHN         1      1        3.1    8.5 CRUZ Ju… ESP     
##  8 M3mSB Final XIONG Ni CHN         1      2        3      8.5 RUIZ-PE… CUB     
##  9 M3mSB Final XIONG Ni CHN         1      2        3      8   GEAR De… NZL     
## 10 M3mSB Final XIONG Ni CHN         1      2        3      8   BOYS Be… CAN     
## # … with 10,777 more rows, and 4 more variables: AvgScore <dbl>,
## #   `Difficulty Range` <chr>, `Avg Score Range` <chr>, all <chr>
networkdata1<-dive_2000 %>% select(all,`Difficulty Range`,`Avg Score Range`)%>%group_by(all,`Difficulty Range`)%>%count
networkdata2<-dive_2000 %>% select(all,`Difficulty Range`,`Avg Score Range`)%>%group_by(`Difficulty Range`,`Avg Score Range`)%>%count
netdf1 <-networkdata1 %>% ungroup()%>%select("source"= all,"target"=`Difficulty Range`,"value"= n)
netdf2 <-networkdata2 %>% ungroup()%>%select("source"= `Difficulty Range`,"target"= `Avg Score Range`,"value"=n)

network<-rbind(netdf1,netdf2)



nodes <- data.frame(name=c(as.character(dive_2000$`Difficulty Range`), as.character(dive_2000$`Avg Score Range`),as.character(dive_2000$all)))%>% unique()
                  
network$IDsource <- match(network$source, nodes$name)-1 
network$IDtarget <- match(network$target, nodes$name)-1
                    
p <- sankeyNetwork(Links = network, Nodes = nodes, Source = "IDsource", Target = "IDtarget",Value = "value", NodeID = "name", fontSize = 14, nodeWidth = 10)
## Links is a tbl_df. Converting to a plain data frame.
p <- htmlwidgets::prependContent(p, htmltools::tags$h3("2000 Olympics diving Difficulty/Average Score network graph"))
p <- htmlwidgets::prependContent(p, htmltools::tags$p("@PeterHou, Data: Diving2000_mod"))

htmlwidgets::onRender(p, '
  function(el) { 
    var cols_x = this.sankey.nodes().map(d => d.x).filter((v, i, a) => a.indexOf(v) === i).sort(function(a, b){return a - b});
    var labels = ["All Divers", "Difficulty", "Avg Scores"];
    cols_x.forEach((d, i) => {
      d3.select(el).select("svg")
        .append("text")
        .attr("x", d)
        .attr("y", 12)
        .attr("font-size", 12)
        .text(labels[i]);
    })
  }
')

2000 Olympics diving Difficulty/Average Score network graph

@PeterHou, Data: Diving2000_mod